From 6815d90f0b08eb692d994e4e552484b5d3d789cf Mon Sep 17 00:00:00 2001 From: Chuck <33324927+ChuckBuilds@users.noreply.github.com> Date: Fri, 2 Oct 2026 13:39:28 -0400 Subject: [PATCH 1/2] feat(fetch): one ESPN scoreboard cache key and a max-age response cache (fetch service stage 2) - espn_scoreboard_cache_key(sport, league, dates) names a scoreboard the same way for every consumer; get_espn_scoreboard / read_ / store_ are the shared cache-through read. A read checks the record's own timestamp, so nothing older than the reader's max_age comes back whatever ttl the writer stored. Old keys are read as legacy_keys for one release. - APIHelper.fetch_espn_scoreboard and SportsFetchMixin (_schedule_cache_key, _cached_schedule, _fetch_season_directly(cache_key=None)) use the key. - FetchService keeps 200s with Cache-Control max-age (minus Age) and answers identical GETs from them, never older than the caller's cache_max_age (default 30 s). Odds pass their interval, APIHelper its cache_ttl. - Counters: memo_hits, cache_hits, legacy_cache_hits; response_cache size. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 42 +++ docs/PLUGIN_API_REFERENCE.md | 44 ++- docs/REST_API_REFERENCE.md | 17 +- src/base_odds_manager.py | 5 +- src/common/api_helper.py | 59 +++- src/common/espn_dates.py | 284 ++++++++++++++++- src/common/fetch_service.py | 250 ++++++++++++++- src/common/sports_fetch.py | 92 +++++- test/test_api_helper.py | 47 ++- test/test_espn_scoreboard_cache.py | 477 +++++++++++++++++++++++++++++ test/test_fetch_service.py | 264 ++++++++++++++++ 11 files changed, 1527 insertions(+), 54 deletions(-) create mode 100644 test/test_espn_scoreboard_cache.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 5913a565d..3c2d361ae 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -96,6 +96,48 @@ policies are unchanged. `GET /api/v3/plugins/fetch-stats`. - `fetch_service` is a core config section (`src/core_config_keys.py`). +### Shared fetch service (stage 2: one scoreboard cache key, a max-age response cache) + +- **One cache key per ESPN scoreboard.** `espn_scoreboard_cache_key(sport, + league, dates)` in `src/common/espn_dates.py` names a scoreboard by ESPN's + own path (`football`/`nfl`) and its `dates=` value, so every consumer of + the same scoreboard shares one cached copy. Before, odds-ticker cached it as + `scoreboard_data_{sport}_{league}_{date}`, `APIHelper` as + `espn_{sport}_{league}_{date}` and the scoreboards as + `{sport_key}_schedule_{window}`. +- **Cache-through helpers.** `get_espn_scoreboard()` returns a cached copy at + most `max_age` seconds old and otherwise fetches with + `fetch_espn_scoreboard` and caches the result; `read_espn_scoreboard_cache()` + and `store_espn_scoreboard_cache()` are the two halves. A read checks the + record's own timestamp, so a writer's stored ttl can no longer make a + reader take data older than its own TTL; the shared entry stores no ttl. + Old keys are passed as `legacy_keys` and read after the canonical one for + one release, so an upgrade does not refetch every league at once. +- **Core callers use the key.** `APIHelper.fetch_espn_scoreboard` caches + under it by default (an explicit `cache_key` still works as before; the old + default key is read as a fallback). `SportsFetchMixin` gains + `_schedule_cache_key()` and `_cached_schedule()` for the scoreboards' + schedule windows (the same read as before, with the old key as fallback, + and a miss deletes the previous day's copy of a sliding window), and + `_fetch_season_directly(cache_key=None)` uses the canonical key. The + scoreboards and odds-ticker move to it in a plugins release that requires + this core. +- **Response cache.** A `200` with `Cache-Control: max-age=N` (minus `Age`) + answers an identical GET for N seconds without a request; ESPN sends no + validators, only max-age (1 to ~500 s, measured 2026-10-02). A caller says + how old a response it accepts with `fetch_get(..., cache_max_age=)` / + `fetch_espn_scoreboard(..., cache_max_age=)`; one that does not say gets at + most 30 s (`fetch_service.response_cache.default_max_age`). + `BaseOddsManager.get_odds` passes its update interval, `APIHelper.get` its + `cache_ttl`, and `get_espn_scoreboard` its `max_age`. `no-store`, + `no-cache`, `private`, `Vary: *` and `Set-Cookie` responses are never kept. + Bounded: 64 entries, 6 MB, 2 MB each; never longer than 10 minutes. +- **Counters.** `memo_hits` (answered by the response cache), `cache_hits` + (scoreboard fetches answered by a shared cache entry) and + `legacy_cache_hits` (reads from a pre-stage-2 key), per plugin and per + host, and the response cache's size, in `GET /api/v3/plugins/fetch-stats`. +- No new module; every change is additive to existing signatures. + ### Control socket (stage 2: wake-ups, brightness, plugin reload) - **Socket commands land at once.** Stage 1's socket was no faster than the diff --git a/docs/PLUGIN_API_REFERENCE.md b/docs/PLUGIN_API_REFERENCE.md index b3331a7ac..200dba875 100644 --- a/docs/PLUGIN_API_REFERENCE.md +++ b/docs/PLUGIN_API_REFERENCE.md @@ -1054,8 +1054,16 @@ values, exceptions and retries are what they were. next identical request revalidates, and a `304 Not Modified` comes back to your code as the original `200` with its body. ESPN currently sends neither, so this does nothing there. +- **Response cache.** A response whose server says `Cache-Control: + max-age=N` answers an identical GET for those N seconds without a + request (ESPN sends 1 to ~500 s). It never hands you a response older + than you accept: pass `cache_max_age=` to `fetch_get()` or + `fetch_espn_scoreboard()` (0 always asks the network); without it a + response is reused for at most 30 seconds. - **Counters.** Requests, merged requests, bytes, 304s, errors and time spent - waiting are counted per plugin and per host, and published for the web UI + waiting, and requests answered without the network (`memo_hits` from the + response cache, `cache_hits` from a shared scoreboard cache entry), are + counted per plugin and per host, and published for the web UI at `GET /api/v3/plugins/fetch-stats` (see [REST_API_REFERENCE.md](REST_API_REFERENCE.md#get-fetch-statistics)). A request is counted against your plugin when it runs inside your @@ -1066,6 +1074,36 @@ What is not covered yet: requests a plugin makes with its own `requests.get()` or `Session.get()` calls. They work as before but are invisible to the budgets and counters. +### One cache key per ESPN scoreboard + +Cache an ESPN scoreboard under `espn_scoreboard_cache_key(sport, league, +dates)` (`src.common.espn_dates`), not a key of your own, so every plugin +showing that league shares one fetch and one cached copy. `sport` and +`league` are ESPN's path segments (`football`, `college-football`), and +`dates` is what you send as `dates=` (`"20261004"`, `"202610"`, +`"20260925-20261016"`, a `date`, or `None` for the undated scoreboard). + +```python +from src.common.espn_dates import get_espn_scoreboard + +data = get_espn_scoreboard( + self.session, "football", "nfl", "20261004", + cache_manager=self.cache_manager, + max_age=300, # your TTL: nothing older comes back + legacy_keys=["my_old_key_20261004"], # read once while upgrading +) +``` + +`get_espn_scoreboard` returns a cached copy at most `max_age` seconds old, +whoever wrote it, and otherwise fetches with `fetch_espn_scoreboard` +(`limit=500`, ranges split the way ESPN requires) and caches the result +without a ttl, so each reader applies its own age limit. `max_age=0` always +fetches but still leaves the copy for others. For a two-step read, use +`read_espn_scoreboard_cache()` and `store_espn_scoreboard_cache()` around +your own fetch. Scoreboards built on `SportsFetchMixin` get +`_schedule_cache_key(datestring)` and `_cached_schedule(key, legacy_keys)` +for their schedule windows. All of this is in the core release after 3.8.0. + The settings live in `config.json` under `fetch_service`, read when the display starts and on a config reload: @@ -1084,7 +1122,9 @@ display starts and on a config reload: the bare domain); `"per_second": 0` removes a budget. `"enabled": false` turns the whole service into a plain `session.get()`. Two further switches, `"single_flight": false` and `"conditional_get": false`, turn off merging and -revalidation. +revalidation. `"response_cache": {"enabled": false}` turns off the response +cache; its `default_max_age` (30) is the limit for callers that pass no +`cache_max_age`. --- diff --git a/docs/REST_API_REFERENCE.md b/docs/REST_API_REFERENCE.md index 1e08576d0..9d9917273 100644 --- a/docs/REST_API_REFERENCE.md +++ b/docs/REST_API_REFERENCE.md @@ -1042,7 +1042,8 @@ or `unknown` (nothing published; `data.data` is `null`). "totals": {"requests": 412, "merged": 3, "not_modified": 0, "errors": 1, "http_errors": 2, "retries": 0, "throttled": 0, "overruns": 0, "bytes": 18234011, - "wait_seconds": 0.0}, + "wait_seconds": 0.0, "memo_hits": 21, "cache_hits": 40, + "legacy_cache_hits": 2}, "plugins": { "football-scoreboard": {"requests": 240, "merged": 2, "bytes": 9120330, "hosts": {"site.api.espn.com": 180, @@ -1053,9 +1054,11 @@ or `unknown` (nothing published; `data.data` is `null`). "site.api.espn.com": {"requests": 301, "...": "as in totals"} }, "validators": {"entries": 0, "bytes": 0}, + "response_cache": {"entries": 3, "bytes": 412004}, "config": {"enabled": true, "single_flight": true, "conditional_get": true, "max_wait_seconds": 2.0, - "rate_limits": {"*.espn.com": {"per_second": 20.0, "burst": 200.0}}} + "rate_limits": {"*.espn.com": {"per_second": 20.0, "burst": 200.0}}, + "response_cache": true, "default_max_age": 30.0} } } } @@ -1067,6 +1070,16 @@ flight, `not_modified` 304s served from the stored body, `errors` transport failures and `http_errors` responses with status 400 or above. `bytes` is the decoded body size. `core` is everything no plugin made. +Three counters are requests that never reached the network: `memo_hits` +were answered from the short response cache (a response still inside the +`Cache-Control: max-age` its server gave it), and `cache_hits` were +scoreboard fetches answered from a shared ESPN scoreboard cache entry +(`espn_scoreboard_cache_key`). `legacy_cache_hits` counts reads served from a +key that predates the shared one; it should fall to zero within a day of an +upgrade. A plugin's `hosts` counts are requests plus merged requests, +`memo_hits` and `cache_hits`: everything it asked for. +`response_cache` is the size of the response cache now. + ### Get/Set Plugin Limits **GET** `/api/v3/plugins/limits/` diff --git a/src/base_odds_manager.py b/src/base_odds_manager.py index eaccb4cc0..63d376fcd 100644 --- a/src/base_odds_manager.py +++ b/src/base_odds_manager.py @@ -175,7 +175,10 @@ def get_odds(self, sport: str | None, league: str | None, event_id: str, url = f"{self.base_url}/{sport}/leagues/{espn_league}/events/{event_id}/competitions/{event_id}/odds" self.logger.debug(f"Requesting odds from URL: {url}") - response = fetch_get(self.session, url, timeout=self.request_timeout) + # The response cache may answer only inside this caller's own + # interval, the age at which its cached odds expire anyway. + response = fetch_get(self.session, url, timeout=self.request_timeout, + cache_max_age=interval) response.raise_for_status() raw_data = response.json() diff --git a/src/common/api_helper.py b/src/common/api_helper.py index d9915f775..45b8b071e 100644 --- a/src/common/api_helper.py +++ b/src/common/api_helper.py @@ -10,7 +10,12 @@ import time from datetime import datetime from types import MappingProxyType -from src.common.espn_dates import ESPN_MAX_LIMIT +from src.common.espn_dates import ( + ESPN_MAX_LIMIT, + espn_scoreboard_cache_key, + read_espn_scoreboard_cache, + store_espn_scoreboard_cache, +) from src.common.fetch_service import fetch_get, fetch_post, share_connection_pool from typing import TYPE_CHECKING, Any, Dict, Mapping, Optional, cast @@ -117,6 +122,14 @@ def get(self, url: str, params: Optional[Dict] = None, Returns: Response data as dictionary or None if request fails """ + return self._get(url, params, headers, timeout, cache_key, cache_ttl, + cache_ttl if cache_key else None) + + def _get(self, url: str, params: Optional[Dict], headers: Optional[Dict], + timeout: Optional[int], cache_key: Optional[str], cache_ttl: int, + cache_max_age: Optional[float]) -> Optional[Dict]: + """:meth:`get`, saying how old a response the fetch service's short + response cache may hand back (``cache_max_age``, the caller's TTL).""" if cache_key and self.cache_manager: cached = self._get_from_cache(cache_key, cache_ttl) if cached is not None: @@ -138,7 +151,8 @@ def get(self, url: str, params: Optional[Dict] = None, url, params=params, headers=request_headers, - timeout=timeout or self.default_timeout + timeout=timeout or self.default_timeout, + cache_max_age=cache_max_age, ) response.raise_for_status() @@ -167,22 +181,23 @@ def fetch_espn_scoreboard(self, sport: str, league: str, sport: Sport name (e.g., 'basketball', 'football') league: League name (e.g., 'nba', 'nfl') date: Date in YYYYMMDD format (defaults to today) - cache_key: Cache key for response - cache_ttl: Cache time-to-live in seconds - + cache_key: Cache key for response. By default the canonical + ``espn_scoreboard_cache_key(sport, league, date)``, shared + with every other consumer of this scoreboard, with the key + this used before (``espn_{sport}_{league}_{date}``) read as a + fallback for one release. An explicit key works as before. + cache_ttl: Cache time-to-live in seconds. A shared entry is + returned only while it is at most this old. + Returns: ESPN API response data or None if request fails """ if date is None: date = datetime.now().strftime('%Y%m%d') - + # Build URL url = f"https://site.api.espn.com/apis/site/v2/sports/{sport}/{league}/scoreboard" - - # Build cache key if not provided - if cache_key is None: - cache_key = f"espn_{sport}_{league}_{date}" - + # Set parameters # limit above 500 makes ESPN truncate instead of erroring: college # football came back with 25 of 68 games. See src/common/espn_dates.py. @@ -190,8 +205,26 @@ def fetch_espn_scoreboard(self, sport: str, league: str, 'dates': date, 'limit': ESPN_MAX_LIMIT } - - return self.get(url, params=params, cache_key=cache_key, cache_ttl=cache_ttl) + + if cache_key is not None: + return self.get(url, params=params, cache_key=cache_key, cache_ttl=cache_ttl) + + legacy_key = f"espn_{sport}_{league}_{date}" + try: + shared_key = espn_scoreboard_cache_key(sport, league, date) + except ValueError: + # Not a path or date the canonical key covers: the old key. + return self.get(url, params=params, cache_key=legacy_key, cache_ttl=cache_ttl) + if self.cache_manager: + cached = read_espn_scoreboard_cache( + self.cache_manager, shared_key, cache_ttl, legacy_keys=(legacy_key,)) + if cached is not None: + self.logger.debug(f"Using cached response for {shared_key}") + return cast(Dict[Any, Any], cached) + data = self._get(url, params, None, None, None, cache_ttl, cache_ttl) + if data is not None and self.cache_manager: + store_espn_scoreboard_cache(self.cache_manager, shared_key, data) + return data def fetch_espn_standings(self, sport: str, league: str, cache_key: Optional[str] = None, diff --git a/src/common/espn_dates.py b/src/common/espn_dates.py index 66f859c80..8347c2700 100644 --- a/src/common/espn_dates.py +++ b/src/common/espn_dates.py @@ -30,15 +30,34 @@ ``RANGE_RETRY_SECONDS`` instead of spending a doomed request first -- live scoreboards ask every 30 seconds. After that the range is tried again, so the workaround retires itself if ESPN reverts. + +ONE CACHE KEY PER SCOREBOARD +---------------------------- +The same ESPN scoreboard used to be cached under a different key by every +consumer: odds-ticker as ``scoreboard_data_{sport}_{league}_{date}``, +``APIHelper`` as ``espn_{sport}_{league}_{date}``, the scoreboards as +``{sport_key}_schedule_{window}`` -- so two plugins showing the same league +fetched and stored it twice. :func:`espn_scoreboard_cache_key` is the one +name for "this sport/league scoreboard for these dates", and +:func:`get_espn_scoreboard` (or :func:`read_espn_scoreboard_cache` and +:func:`store_espn_scoreboard_cache` around :func:`fetch_espn_scoreboard`) +is the cache-through read every consumer can share. A read never returns an +entry older than the reader's own ``max_age``, whoever wrote it and whatever +ttl they stored with it. Old keys are passed as ``legacy_keys`` and read +after the canonical one, so an upgrade does not refetch everything at once; +they can go one release after the one that added this. """ import contextvars +import logging +import math +import re import threading import time from concurrent.futures import ThreadPoolExecutor -from datetime import date, timedelta +from datetime import date, datetime, timedelta from functools import partial -from typing import Any, Dict, List, Optional, Tuple, cast +from typing import Any, Dict, Iterable, List, Optional, Tuple, cast try: from src.common.json_body import response_json @@ -51,18 +70,23 @@ def response_json(response: Any) -> Any: try: # The core fetch service: counts, per-host budget, merging of identical # requests. Same call, same result and errors as ``session.get``. - from src.common.fetch_service import fetch_get, pinned_caller + from src.common.fetch_service import fetch_get, get_fetch_service, pinned_caller + _COUNTS_FETCHES = True except ImportError: # Bundled copies on cores without it call the session directly. import contextlib def fetch_get(session: Any, url: str, *, share_in_flight: bool = True, - **kwargs: Any) -> Any: + cache_max_age: Optional[float] = None, **kwargs: Any) -> Any: return session.get(url, **kwargs) def pinned_caller() -> Any: return contextlib.nullcontext() + _COUNTS_FETCHES = False + +_logger = logging.getLogger(__name__) + # Above this, ESPN returns a truncated list instead of an error. See module # docstring: 500 is the largest value measured to return complete data. ESPN_MAX_LIMIT = 500 @@ -89,8 +113,23 @@ def pinned_caller() -> Any: "merge_scoreboard_payloads", "fetch_espn_date_chunks", "fetch_espn_scoreboard", + "ESPN_SCOREBOARD_URL", + "espn_scoreboard_url", + "espn_scoreboard_cache_key", + "espn_scoreboard_cache_key_for_url", + "read_espn_scoreboard_cache", + "store_espn_scoreboard_cache", + "get_espn_scoreboard", ] +#: The site-API scoreboard every sport and league shares. +ESPN_SCOREBOARD_URL = "https://site.api.espn.com/apis/site/v2/sports/{sport}/{league}/scoreboard" +_ESPN_HOST_URL = "https://site.api.espn.com/" + +_PATH_PART = re.compile(r"^[a-z0-9][a-z0-9.\-]*$") +_DATES = re.compile(r"^\d{4}(?:\d{2}(?:\d{2})?)?$|^\d{8}-\d{8}$") +_SCOREBOARD_PATH = re.compile(r"/sports/([^/?#]+)/([^/?#]+)/scoreboard/?$") + def clamp_espn_limit(params: Optional[Dict[str, Any]]) -> Dict[str, Any]: """Return a copy of ``params`` with any ``limit`` over 500 pulled back to 500.""" @@ -129,6 +168,12 @@ def parse_espn_date_range(dates: Any) -> Optional[Tuple[date, date]]: return start, end +def _memo_kwargs(cache_max_age: Optional[float]) -> Dict[str, Any]: + """``cache_max_age`` for fetch_get, only when the caller gave one, so a + call that did not say is the call it always was.""" + return {} if cache_max_age is None else {"cache_max_age": cache_max_age} + + def _ranges_known_rejected() -> bool: with _range_lock: return time.monotonic() < _ranges_rejected_until @@ -204,6 +249,7 @@ def merge_scoreboard_payloads(payloads: List[Any]) -> Dict[str, Any]: def _fetch_one_chunk( session, url: str, params: Dict[str, Any], headers, timeout, logger, chunk: str, + cache_max_age: Optional[float] = None, ) -> Optional[Dict[str, Any]]: """GET a single ``dates=`` chunk, or None when it failed. @@ -217,6 +263,7 @@ def _fetch_one_chunk( params=dict(params, dates=chunk, limit=ESPN_MAX_LIMIT), headers=headers, timeout=timeout, + **_memo_kwargs(cache_max_age), ) response.raise_for_status() return cast(Optional[Dict[str, Any]], response_json(response)) @@ -228,7 +275,7 @@ def _fetch_one_chunk( def _fetch_chunks( session, url: str, params: Dict[str, Any], headers, timeout, logger, - chunks: List[str], + chunks: List[str], cache_max_age: Optional[float] = None, ) -> List[Optional[Dict[str, Any]]]: """Fetch every chunk, returning payloads positionally aligned with ``chunks``. @@ -246,6 +293,7 @@ def _fetch_chunks( return [] fetch = partial( _fetch_one_chunk, session, url, params, headers, timeout, logger, + cache_max_age=cache_max_age, ) if len(chunks) == 1: return [fetch(chunks[0])] @@ -268,6 +316,7 @@ def fetch_espn_date_chunks( headers: Optional[Dict[str, str]] = None, timeout: int = 15, logger=None, + cache_max_age: Optional[float] = None, ) -> Optional[Dict[str, Any]]: """Fetch a ``YYYYMMDD-YYYYMMDD`` window as month and day chunks. @@ -299,7 +348,7 @@ def fetch_espn_date_chunks( ) results = _fetch_chunks( - session, url, params, headers, timeout, logger, chunks, + session, url, params, headers, timeout, logger, chunks, cache_max_age, ) attempted = len(chunks) @@ -331,7 +380,7 @@ def fetch_espn_date_chunks( days = [day for index in sorted(capped) for day in capped[index]] attempted += len(days) by_day = dict(zip(days, _fetch_chunks( - session, url, params, headers, timeout, logger, days, + session, url, params, headers, timeout, logger, days, cache_max_age, ))) for index, month_days in capped.items(): slots[index] = [by_day.get(day) for day in month_days] @@ -364,6 +413,7 @@ def fetch_espn_scoreboard( headers: Optional[Dict[str, str]] = None, timeout: int = 15, logger=None, + cache_max_age: Optional[float] = None, ) -> Dict[str, Any]: """GET an ESPN scoreboard, re-asking in month/day chunks if a range 400s. @@ -373,6 +423,10 @@ def fetch_espn_scoreboard( and later ranges go straight to chunks for ``RANGE_RETRY_SECONDS``. A 400 on a non-range request, any other error, and a range whose every chunk fails all raise as before. + + ``cache_max_age`` is the oldest response, in seconds, the caller takes + from the fetch service's short response cache (its own TTL; 0 always + asks ESPN). None leaves it to the service default. """ params = clamp_espn_limit(params) is_range = parse_espn_date_range(params.get("dates")) is not None @@ -381,7 +435,7 @@ def fetch_espn_scoreboard( if is_range and _ranges_known_rejected(): data = fetch_espn_date_chunks( session, url, params=params, headers=headers, - timeout=timeout, logger=logger, + timeout=timeout, logger=logger, cache_max_age=cache_max_age, ) if data is not None: return data @@ -389,7 +443,8 @@ def fetch_espn_scoreboard( # real error to log, without spending the chunks a second time. chunks_tried = True - response = fetch_get(session, url, params=params, headers=headers, timeout=timeout) + response = fetch_get(session, url, params=params, headers=headers, timeout=timeout, + **_memo_kwargs(cache_max_age)) if is_range and response.status_code == 400 and not chunks_tried: _note_range_rejected() if logger: @@ -400,9 +455,218 @@ def fetch_espn_scoreboard( ) data = fetch_espn_date_chunks( session, url, params=params, headers=headers, - timeout=timeout, logger=logger, + timeout=timeout, logger=logger, cache_max_age=cache_max_age, ) if data is not None: return data response.raise_for_status() return cast(Dict[str, Any], response_json(response)) + + +# --- one cache key per scoreboard -------------------------------------------------- + +def espn_scoreboard_url(sport: str, league: str) -> str: + """The site-API scoreboard URL for an ESPN ``sport`` / ``league`` path.""" + return ESPN_SCOREBOARD_URL.format(sport=_path_part(sport, "sport"), + league=_path_part(league, "league")) + + +def _path_part(value: Any, what: str) -> str: + text = str(value or "").strip().lower() + if not _PATH_PART.match(text): + raise ValueError(f"not an ESPN {what} path segment: {value!r}") + return text + + +def _day(value: Any) -> str: + if isinstance(value, (date, datetime)): + return value.strftime("%Y%m%d") + text = str(value).strip() + if len(text) != 8 or not text.isdigit(): + raise ValueError(f"not an ESPN day (YYYYMMDD): {value!r}") + return text + + +def _dates_part(dates: Any) -> str: + """``dates`` as ESPN spells it, or ``current`` for no ``dates`` at all.""" + if dates is None or dates == "": + return "current" + if isinstance(dates, (date, datetime)): + return _day(dates) + if isinstance(dates, (tuple, list)): + if len(dates) != 2: + raise ValueError(f"a date range is (start, end): {dates!r}") + start, end = _day(dates[0]), _day(dates[1]) + return start if start == end else f"{start}-{end}" + text = str(dates).strip() + if isinstance(dates, bool) or not _DATES.match(text): + raise ValueError( + f"not an ESPN dates value (YYYY, YYYYMM, YYYYMMDD or " + f"YYYYMMDD-YYYYMMDD): {dates!r}") + return text + + +def espn_scoreboard_cache_key(sport: str, league: str, dates: Any = None) -> str: + """The one cache key for an ESPN scoreboard, whoever caches it. + + ``sport`` and ``league`` are ESPN's own path segments -- ``football`` / + ``college-football``, ``soccer`` / ``eng.1`` -- not a plugin's + ``sport_key``, so every plugin showing a league names it the same way. + ``dates`` is what the request sends as ``dates=``: ``"YYYYMMDD"``, + ``"YYYYMM"``, ``"YYYY"``, ``"YYYYMMDD-YYYYMMDD"``, a ``date``, or a + ``(start, end)`` pair of either; None is the undated "current" + scoreboard. Anything else raises ValueError rather than invent a key. + + The key says nothing about ``limit``: a cached copy is meant to be a + whole one (the helpers here always ask for ``ESPN_MAX_LIMIT``). + """ + return (f"espn_scoreboard_{_path_part(sport, 'sport')}_" + f"{_path_part(league, 'league')}_{_dates_part(dates)}") + + +def espn_scoreboard_cache_key_for_url(url: str, dates: Any = None) -> Optional[str]: + """:func:`espn_scoreboard_cache_key` for a scoreboard URL, or None when + ``url`` is not ``.../sports/{sport}/{league}/scoreboard``.""" + match = _SCOREBOARD_PATH.search(str(url or "").split("?", 1)[0]) + if match is None: + return None + try: + return espn_scoreboard_cache_key(match.group(1), match.group(2), dates) + except ValueError: + return None + + +def _note_cache_hit(legacy: bool, avoided_request: bool = True) -> None: + if not _COUNTS_FETCHES: + return + try: + get_fetch_service().note_cache_hit( + _ESPN_HOST_URL, legacy=legacy, avoided_request=avoided_request) + except Exception: # noqa: BLE001 - counting never breaks a read + _logger.debug("could not count a scoreboard cache hit", exc_info=True) + + +def _fresh_cached(cache_manager: Any, key: str, max_age: Optional[float], + now: float) -> Optional[Any]: + """The data cached under ``key`` if it is at most ``max_age`` seconds old. + + The age is the stored record's own timestamp, checked here: CacheManager + lets a ttl stored by the writer override the reader's max_age, and its + memory tier times an entry from when it was loaded, not written. A key + shared by readers with different TTLs can rely on neither. + """ + reader = getattr(cache_manager, "get_cached_data", None) + limit = None if max_age is None else max(1, int(math.ceil(max_age))) + if not callable(reader): + # A cache without records (a test double, a plugin's own store). + value = cache_manager.get(key, max_age=limit) + return value if isinstance(value, dict) else None + record = reader(key, max_age=limit, memory_ttl=limit) + if not isinstance(record, dict): + return None + if "data" not in record: + return record # unwrapped; the cache already judged it by mtime + if max_age is not None: + stamp = record.get("timestamp") + if isinstance(stamp, bool) or not isinstance(stamp, (int, float)): + return None + if now - float(stamp) > max_age: + return None + data = record["data"] + return data if isinstance(data, dict) else None + + +def read_espn_scoreboard_cache( + cache_manager: Any, + key: str, + max_age: Optional[float], + legacy_keys: Iterable[str] = (), + now: Optional[float] = None, +) -> Optional[Any]: + """The cached scoreboard under ``key``, or under the first of + ``legacy_keys`` that has one, if it is at most ``max_age`` seconds old. + + None on a miss, a stale entry, ``max_age`` of 0 or less, no cache + manager, or any cache error -- a read never raises. ``max_age=None`` + takes an entry of any age. A hit is counted in the fetch statistics + (``cache_hits``; ``legacy_cache_hits`` too for an old key). + """ + if cache_manager is None: + return None + if max_age is not None and max_age <= 0: + return None + clock = time.time() if now is None else now + for index, candidate in enumerate([key, *legacy_keys]): + if not candidate: + continue + try: + data = _fresh_cached(cache_manager, candidate, max_age, clock) + except Exception: # noqa: BLE001 - a broken cache is a miss + _logger.debug("scoreboard cache read failed for %s", candidate, exc_info=True) + continue + if data is not None: + _note_cache_hit(legacy=index > 0) + return data + return None + + +def store_espn_scoreboard_cache(cache_manager: Any, key: str, data: Any) -> None: + """Cache a fetched scoreboard under ``key``. Never raises. + + No ttl is stored: each reader applies its own ``max_age`` (a live + reader 30 s, a schedule reader an hour), and a stored ttl would + override theirs in CacheManager. + """ + if cache_manager is None or data is None: + return + try: + cache_manager.set(key, data) + except Exception: # noqa: BLE001 - the caller still has its data + _logger.warning("Could not cache scoreboard %s", key, exc_info=True) + + +def get_espn_scoreboard( + session: Any, + sport: str, + league: str, + dates: Any = None, + *, + cache_manager: Any = None, + max_age: Optional[float] = 300, + legacy_keys: Iterable[str] = (), + headers: Optional[Dict[str, str]] = None, + timeout: int = 15, + logger: Any = None, +) -> Dict[str, Any]: + """An ESPN scoreboard through the shared cache, fetched on a miss. + + Reads :func:`espn_scoreboard_cache_key` (then ``legacy_keys``) and + returns an entry at most ``max_age`` seconds old. Otherwise it fetches + with :func:`fetch_espn_scoreboard` -- ``limit=ESPN_MAX_LIMIT``, ranges + split as ESPN needs -- caches the result under the canonical key and + returns it. ``max_age=0`` always fetches (and still caches, for other + readers). Errors raise exactly as :func:`fetch_espn_scoreboard` does, + and nothing is cached then. ``session=None`` uses the fetch service's + pooled session for the ESPN host. + """ + key = espn_scoreboard_cache_key(sport, league, dates) + cached = read_espn_scoreboard_cache(cache_manager, key, max_age, legacy_keys) + if cached is not None: + return cast(Dict[str, Any], cached) + params: Dict[str, Any] = {"limit": ESPN_MAX_LIMIT} + spelled = _dates_part(dates) + if spelled != "current": + params["dates"] = spelled + data = fetch_espn_scoreboard( + session, + espn_scoreboard_url(sport, league), + params=params, + headers=headers, + timeout=timeout, + logger=logger, + # The response cache must not hand back anything older than the + # cache read above would have accepted. + cache_max_age=None if max_age is None else max(0.0, float(max_age)), + ) + store_espn_scoreboard_cache(cache_manager, key, data) + return data diff --git a/src/common/fetch_service.py b/src/common/fetch_service.py index 0764b0294..c116776bf 100644 --- a/src/common/fetch_service.py +++ b/src/common/fetch_service.py @@ -46,9 +46,25 @@ this was written -- see the PR that added this module -- so on ESPN the store stays empty and costs nothing.) +**Response cache (stage 2).** A 200 that says ``Cache-Control: max-age=N`` +is kept in memory for those N seconds (less its ``Age``), and an identical +GET inside that window is answered from it without a request. ESPN sends +max-age (1-496 s measured on 2026-10-02, most of it under 10 s) and no +validators, so this is the only revalidation-free reuse ESPN allows. It +never hands a caller a response older than the caller accepts: a caller +says how old with ``cache_max_age`` (``fetch_get(..., cache_max_age=ttl)``; +0 skips the cache), and one that does not say gets at most +``response_cache.default_max_age`` (30 s). ``no-store``, ``no-cache``, +``private``, ``Vary: *`` and ``Set-Cookie`` responses are never kept. +Identical means what the validator store keys on: URL, query, effective +headers and, for a session with cookies or auth, the session. + **Counters.** Requests, merged requests, bytes, 304s, errors, HTTP errors, -adapter retries, throttled requests and seconds waited, per plugin and per -host. :class:`FetchStatsPublisher` publishes them for the web interface +adapter retries, throttled requests and seconds waited, plus requests +answered without the network: ``memo_hits`` (the response cache) and +``cache_hits`` / ``legacy_cache_hits`` (a shared ESPN scoreboard cache entry, +counted by ``src/common/espn_dates.py``). Per plugin and per host. +:class:`FetchStatsPublisher` publishes them for the web interface (``GET /api/v3/plugins/fetch-stats``). CALLER IDENTITY @@ -68,7 +84,8 @@ 3. Otherwise the request is the core's own (``"core"``). Nothing here raises on account of bookkeeping: a failure in counting, -keying or the validator store falls back to a plain ``session.get``. +keying, the validator store or the response cache falls back to a plain +``session.get``. Core-internal for now (stage 1). Plugins reach it through ``APIHelper`` and ``espn_dates``; a plugin-facing API comes with stage 3. @@ -143,8 +160,22 @@ "max_bytes": 4 * 1024 * 1024, "max_entry_bytes": 1024 * 1024, }, + # Stage 2: responses ESPN calls fresh (Cache-Control: max-age), reused + # for identical GETs. A college-football Saturday is ~1 MB decoded, so + # one entry may be 2 MB; months (5-7 MB) are never kept. + "response_cache": { + "enabled": True, + "default_max_age": 30, + "max_entries": 64, + "max_bytes": 6 * 1024 * 1024, + "max_entry_bytes": 2 * 1024 * 1024, + }, } +#: However long a server says a response stays fresh, it is not kept longer +#: than this: the cache is for requests that coincide, not for storage. +_RESPONSE_CACHE_CEILING = 600.0 + #: Connection pools kept per shared adapter (one per host) and connections #: kept per pool. Larger than requests' 10 because one adapter now serves #: every core Session with its retry policy: three background workers each @@ -171,6 +202,9 @@ "overruns", # requests that went after max_wait_seconds anyway "bytes", # decoded response body bytes received "wait_seconds", # time spent waiting for host budgets + "memo_hits", # answered from the response cache (max-age); nothing sent + "cache_hits", # scoreboard fetches answered from a shared ESPN cache entry + "legacy_cache_hits", # cache reads answered from a pre-stage-2 key (any helper) ) @@ -393,6 +427,117 @@ def stats(self) -> Dict[str, int]: return {"entries": len(self._entries), "bytes": self._bytes} +# --- response cache (Cache-Control: max-age) -------------------------------------- + +@dataclass +class _Fresh: + response: requests.Response + stored_at: float + #: Seconds after stored_at the server said the response stays fresh. + lifetime: float + size: int + + +class _ResponseCache: + """LRU of finished 200 responses, each kept for its server max-age. + + :meth:`get` answers only while the entry is younger than both its own + lifetime and the caller's limit, so nobody is handed a response older + than they asked for. Expired entries are dropped as they are met and on + every insert, so the cache holds only what is still fresh. + """ + + def __init__(self, max_entries: int, max_bytes: int, max_entry_bytes: int, + clock: Callable[[], float]) -> None: + self.max_entries = max_entries + self.max_bytes = max_bytes + self.max_entry_bytes = max_entry_bytes + self._clock = clock + self._entries: "OrderedDict[Any, _Fresh]" = OrderedDict() + self._bytes = 0 + self._lock = threading.Lock() + + def get(self, key: Any, max_age: float) -> Optional[requests.Response]: + with self._lock: + entry = self._entries.get(key) + if entry is None: + return None + age = self._clock() - entry.stored_at + if age < 0 or age >= entry.lifetime: + self._drop_locked(key) + return None + if age > max_age: + return None # fresh for someone less strict; kept + self._entries.move_to_end(key) + clone: requests.Response = _clone_response(entry.response) + return clone + + def put(self, key: Any, response: requests.Response, lifetime: float, + size: int) -> None: + with self._lock: + self._drop_locked(key) + if size > self.max_entry_bytes or self.max_entries <= 0 or lifetime <= 0: + return + now = self._clock() + for old_key in [k for k, e in self._entries.items() + if now - e.stored_at >= e.lifetime]: + self._drop_locked(old_key) + self._entries[key] = _Fresh(_clone_response(response), now, lifetime, size) + self._bytes += size + while self._entries and (len(self._entries) > self.max_entries + or self._bytes > self.max_bytes): + _, old = self._entries.popitem(last=False) + self._bytes -= old.size + + def _drop_locked(self, key: Any) -> None: + old = self._entries.pop(key, None) + if old is not None: + self._bytes -= old.size + + def clear(self) -> None: + with self._lock: + self._entries.clear() + self._bytes = 0 + + def stats(self) -> Dict[str, int]: + with self._lock: + return {"entries": len(self._entries), "bytes": self._bytes} + + +def _cache_directives(value: Optional[str]) -> Dict[str, Optional[str]]: + directives: Dict[str, Optional[str]] = {} + for part in (value or "").split(","): + name, _, arg = part.strip().partition("=") + if name: + directives[name.strip().lower()] = arg.strip().strip('"') if arg else None + return directives + + +def _fresh_for(response: Any) -> Optional[float]: + """Seconds a finished response stays fresh by its own headers, or None + when it must not be reused: not a 200 with its body read, ``no-store``, + ``no-cache``, ``private``, ``Vary: *``, ``Set-Cookie``, or no max-age.""" + if _status_of(response) != 200 or _body_of(response) is None: + return None + directives = _cache_directives(_str_header(response, "Cache-Control")) + if {"no-store", "no-cache", "private"} & set(directives): + return None + if (_str_header(response, "Vary") or "").strip() == "*": + return None + if _str_header(response, "Set-Cookie"): + return None + try: + max_age = int(directives.get("max-age") or "") + except ValueError: + return None + try: + age = int(_str_header(response, "Age") or 0) + except ValueError: + age = 0 + fresh = float(min(max_age - max(age, 0), _RESPONSE_CACHE_CEILING)) + return fresh if fresh > 0 else None + + # --- helpers ------------------------------------------------------------------------ def _host_of(url: Any) -> str: @@ -580,6 +725,9 @@ def __init__(self, config: Any = None, *, self.max_wait_seconds = 2.0 self._rate_limits: Dict[str, Tuple[float, float]] = {} self._validators = _ValidatorStore(0, 0, 0) + self.response_cache = True + self.default_max_age = 30.0 + self._fresh = _ResponseCache(0, 0, 0, clock) self._applied: Optional[str] = None self.configure(config) @@ -635,6 +783,14 @@ def configure(self, config: Any = None) -> None: store = merged.get("validator_store") store = store if isinstance(store, Mapping) else {} default_store = DEFAULT_CONFIG["validator_store"] + fresh = merged.get("response_cache") + if fresh is not None and not isinstance(fresh, Mapping): + logger.warning("fetch_service.response_cache is not an object; using the defaults") + fresh = fresh if isinstance(fresh, Mapping) else {} + default_fresh = DEFAULT_CONFIG["response_cache"] + self.response_cache = fresh.get("enabled") is not False + self.default_max_age = _as_float(fresh.get("default_max_age"), + float(default_fresh["default_max_age"])) with self._lock: self._rate_limits = limits self._buckets.clear() @@ -643,6 +799,12 @@ def configure(self, config: Any = None) -> None: _as_int(store.get("max_bytes"), default_store["max_bytes"]), _as_int(store.get("max_entry_bytes"), default_store["max_entry_bytes"]), ) + self._fresh = _ResponseCache( + _as_int(fresh.get("max_entries"), default_fresh["max_entries"]), + _as_int(fresh.get("max_bytes"), default_fresh["max_bytes"]), + _as_int(fresh.get("max_entry_bytes"), default_fresh["max_entry_bytes"]), + self._clock, + ) self.change_count += 1 def describe_config(self) -> Dict[str, Any]: @@ -655,6 +817,8 @@ def describe_config(self) -> Dict[str, Any]: "conditional_get": self.conditional_get, "max_wait_seconds": self.max_wait_seconds, "rate_limits": limits, + "response_cache": self.response_cache, + "default_max_age": self.default_max_age, } def _limit_for(self, host: str) -> Optional[Tuple[float, float]]: @@ -723,7 +887,7 @@ def session_for(self, url: str) -> requests.Session: # -- requests -- def get(self, session: Any, url: str, *, share_in_flight: bool = True, - **kwargs: Any) -> Any: + cache_max_age: Optional[float] = None, **kwargs: Any) -> Any: """``session.get(url, **kwargs)`` through the service. Same return value, same exceptions, and ``session.get`` is called @@ -734,6 +898,11 @@ def get(self, session: Any, url: str, *, share_in_flight: bool = True, than joining an identical one in flight -- for a caller that may retry *because* an earlier request hung (BackgroundDataService cancels and replaces a fetch) and must not be handed that one. + + ``cache_max_age`` is the oldest response, in seconds, the caller + will take from the response cache (its own TTL); 0 always asks the + network. None means ``response_cache.default_max_age``. A response + is never reused past the max-age its server gave it either. """ transport = session if session is not None else self.session_for(url) if not self.enabled: @@ -744,8 +913,24 @@ def get(self, session: Any, url: str, *, share_in_flight: bool = True, logger.debug("fetch_service could not key a request to %s", url, exc_info=True) request = _Request(plugin=self._caller(), host=_host_of(url)) + reusable = (self.response_cache and request.representation is not None + and not _has_conditional_headers(kwargs.get("headers"))) + if reusable: + try: + limit = self._accepted_age(cache_max_age) + cached = self._fresh.get(request.representation, limit) if limit > 0 else None + except Exception: + logger.debug("fetch_service response cache lookup failed", exc_info=True) + cached = None + if cached is not None: + self._count(request.plugin, request.host, memo_hits=1) + return cached + if request.flight is None or not self.single_flight or not share_in_flight: - return self._send_get(transport, url, kwargs, request) + response = self._send_get(transport, url, kwargs, request) + if reusable: + self._remember(request, response) + return response with self._lock: flight = self._inflight.get(request.flight) @@ -764,6 +949,8 @@ def get(self, session: Any, url: str, *, share_in_flight: bool = True, try: response = self._send_get(transport, url, kwargs, request) flight.response = response + if reusable: + self._remember(request, response) return response except BaseException as error: flight.error = error @@ -790,6 +977,40 @@ def note_merged(self, url: Any = None, plugin_id: Optional[str] = None) -> None: except Exception: logger.debug("fetch_service could not count a merged request", exc_info=True) + def note_cache_hit(self, url: Any = None, *, legacy: bool = False, + avoided_request: bool = True, + plugin_id: Optional[str] = None) -> None: + """Count a read answered from a shared cache entry instead of the + network (``espn_dates``). ``legacy`` marks a read from a key that + predates the canonical one; ``avoided_request=False`` counts only + that, for a read that was never going to fetch on a miss.""" + try: + self._count(plugin_id or self._caller(), _host_of(url), + cache_hits=int(avoided_request), legacy_cache_hits=int(legacy)) + except Exception: + logger.debug("fetch_service could not count a cache hit", exc_info=True) + + def _accepted_age(self, cache_max_age: Any) -> float: + """The oldest cached response this call accepts, in seconds.""" + if cache_max_age is None: + return self.default_max_age + if isinstance(cache_max_age, bool) or not isinstance(cache_max_age, (int, float)): + return self.default_max_age + value = float(cache_max_age) + return value if math.isfinite(value) and value > 0 else 0.0 + + def _remember(self, request: _Request, response: Any) -> None: + """Keep a finished response for its server max-age. Never raises.""" + try: + lifetime = _fresh_for(response) + if lifetime is None: + return + body = _body_of(response) + self._fresh.put(request.representation, response, lifetime, + len(body) if body is not None else 0) + except Exception: + logger.debug("fetch_service could not keep a response", exc_info=True) + def _caller(self) -> str: return current_plugin_id() or CORE @@ -927,10 +1148,11 @@ def _apply_counts(self, plugin: str, host: str, deltas: Dict[str, float]) -> Non per_plugin[name] += value per_host[name] += value self._totals[name] += value - if changes.get("requests") or changes.get("merged"): + asked = int(changes.get("requests", 0) + changes.get("merged", 0) + + changes.get("memo_hits", 0) + changes.get("cache_hits", 0)) + if asked: hosts = self._plugin_hosts.setdefault(plugin, {}) - hosts[host] = hosts.get(host, 0) + int(changes.get("requests", 0) - + changes.get("merged", 0)) + hosts[host] = hosts.get(host, 0) + asked self.change_count += 1 def reset_counters(self) -> None: @@ -942,12 +1164,14 @@ def reset_counters(self) -> None: self.change_count += 1 def reset(self) -> None: - """Counters, validators, budgets and in-flight table (tests).""" + """Counters, validators, response cache, budgets and in-flight + table (tests).""" self.reset_counters() with self._lock: self._buckets.clear() self._inflight.clear() self._validators.clear() + self._fresh.clear() def snapshot(self) -> Dict[str, Any]: """Counters since the service started, JSON-ready.""" @@ -971,6 +1195,7 @@ def rounded(counters: Mapping[str, float]) -> Dict[str, Any]: "plugins": plugins, "hosts": hosts, "validators": self._validators.stats(), + "response_cache": self._fresh.stats(), "config": self.describe_config(), } @@ -1005,10 +1230,11 @@ def configure_fetch_service(config: Any) -> FetchService: def fetch_get(session: Any, url: str, *, share_in_flight: bool = True, - **kwargs: Any) -> Any: - """``session.get(url, **kwargs)`` through the process's FetchService.""" + cache_max_age: Optional[float] = None, **kwargs: Any) -> Any: + """``session.get(url, **kwargs)`` through the process's FetchService. + ``cache_max_age``: see :meth:`FetchService.get`.""" return get_fetch_service().get(session, url, share_in_flight=share_in_flight, - **kwargs) + cache_max_age=cache_max_age, **kwargs) def fetch_post(session: Any, url: str, **kwargs: Any) -> Any: diff --git a/src/common/sports_fetch.py b/src/common/sports_fetch.py index fa73d7206..d5f144b76 100644 --- a/src/common/sports_fetch.py +++ b/src/common/sports_fetch.py @@ -40,6 +40,20 @@ - ``live_games``, read with ``getattr`` -- ``_needs_previous_day``. - ``background_service``, read with ``getattr`` -- ``_background_fetches_espn_ranges``. +- ``sport`` and ``league`` (ESPN's path segments, e.g. ``football`` / + ``nfl``) -- ``_schedule_cache_key``, and ``_fetch_season_directly`` when + it is given no key and cannot read one from its URL. + +THE SCHEDULE CACHE KEY (fetch service stage 2) +---------------------------------------------- +``_schedule_cache_key`` names a schedule window with the canonical +``espn_scoreboard_cache_key`` instead of a plugin-built +``{sport_key}_schedule_{window}``, and ``_cached_schedule`` reads it with the +old key as a fallback for one release, so an upgrade serves the copy already +on disk instead of refetching every league at once. The canonical key +carries the window's dates, so it moves on a day as the window slides; a +miss on it also deletes the copy for the day before, so a league keeps one +window file instead of a week of them. Add it as a base of the plugin's ``SportsCore``, e.g. ``class SportsCore(SportsFetchMixin, SportsCoreSharedMixin, @@ -50,9 +64,31 @@ import logging import threading from datetime import datetime, timedelta -from typing import Any, ClassVar, Dict, Optional +from typing import Any, ClassVar, Dict, Iterable, Optional + +from src.common.espn_dates import ( + ESPN_MAX_LIMIT, + espn_scoreboard_cache_key, + espn_scoreboard_cache_key_for_url, + fetch_espn_scoreboard, + parse_espn_date_range, +) +from src.common.fetch_service import get_fetch_service + +_ESPN_SITE = "https://site.api.espn.com/" -from src.common.espn_dates import ESPN_MAX_LIMIT, fetch_espn_scoreboard + +def _previous_window_key(cache_key: str) -> Optional[str]: + """The canonical key of the same window one day earlier, or None when + ``cache_key`` is not a canonical day-range key.""" + head, sep, dates = cache_key.rpartition("_") + if not sep or not head.startswith("espn_scoreboard_"): + return None + span = parse_espn_date_range(dates) + if span is None: + return None + start, end = (day - timedelta(days=1) for day in span) + return f"{head}_{start.strftime('%Y%m%d')}-{end.strftime('%Y%m%d')}" class SportsFetchMixin: @@ -65,6 +101,8 @@ class SportsFetchMixin: cache_manager: Any logger: logging.Logger _games_lock: threading.RLock + sport: str + league: str #: How many games past the one on screen keep their odds warm. One is #: enough for the line to be ready when the rotation advances; more just @@ -154,18 +192,66 @@ def _background_fetches_espn_ranges(self) -> bool: service = getattr(self, "background_service", None) return bool(getattr(service, "handles_espn_date_ranges", False)) + def _schedule_cache_key(self, datestring: str) -> str: + """The canonical cache key for this league's schedule over + ``datestring`` (``espn_scoreboard_cache_key``).""" + return espn_scoreboard_cache_key(self.sport, self.league, datestring) + + def _cached_schedule(self, cache_key: str, legacy_keys: Iterable[str] = ()) -> Any: + """What ``self.cache_manager.get(cache_key)`` returns, falling back + to each of ``legacy_keys`` (the plugin's pre-canonical keys) in turn. + + The same read the managers made before -- same default max age, a + stored ttl still wins -- so moving to the canonical key changes + where a schedule is cached, not for how long. A read from an old key + is counted (``legacy_cache_hits``) so it is visible when the + fallback can go. A miss on the canonical key also deletes the same + window's copy from the day before (see the module docstring). + """ + cached = self.cache_manager.get(cache_key) + if cached: + return cached + self._retire_previous_window(cache_key) + for legacy in legacy_keys: + if not legacy or legacy == cache_key: + continue + cached = self.cache_manager.get(legacy) + if cached: + try: + get_fetch_service().note_cache_hit( + _ESPN_SITE, legacy=True, avoided_request=False) + except Exception: # noqa: BLE001 - counting never breaks a read + pass + return cached + return None + + def _retire_previous_window(self, cache_key: str) -> None: + previous = _previous_window_key(cache_key) + delete = getattr(self.cache_manager, "delete", None) + if previous is None or not callable(delete): + return + try: + delete(previous) + except Exception as e: # noqa: BLE001 - housekeeping only + self.logger.debug(f"Could not delete old schedule copy {previous}: {e}") + def _fetch_season_directly( self, url: str, datestring: str, - cache_key: str, + cache_key: Optional[str], label: str, ttl: Optional[int] = None, ) -> Optional[Dict]: """Fetch a season schedule on this thread, in chunks ESPN accepts, and cache it. ``label`` names the schedule in log lines, e.g. ``"2026 season"``. + ``cache_key=None`` caches it under the canonical key + (``espn_scoreboard_cache_key`` for ``url``'s sport and league). """ + if cache_key is None: + cache_key = (espn_scoreboard_cache_key_for_url(url, datestring) + or self._schedule_cache_key(datestring)) try: data = fetch_espn_scoreboard( self.session, diff --git a/test/test_api_helper.py b/test/test_api_helper.py index f31a5a9b5..b55b3fc21 100644 --- a/test/test_api_helper.py +++ b/test/test_api_helper.py @@ -205,27 +205,52 @@ def test_per_call_headers_merge_over_session_headers(self, helper): class TestEspnHelpers: @freeze_time('2026-08-07') - def test_fetch_espn_scoreboard_url_params_and_cache_key(self, helper): - helper.get = Mock(return_value={'ok': 1}) + def test_fetch_espn_scoreboard_url_params_and_cache_key(self, helper, cache): + # The canonical key (fetch service stage 2), shared with every other + # consumer of the scoreboard; the old key is only read as a fallback. + cache.get_cached_data.return_value = None + helper._get = Mock(return_value={'ok': 1}) result = helper.fetch_espn_scoreboard('football', 'nfl') assert result == {'ok': 1} - helper.get.assert_called_once_with( + helper._get.assert_called_once_with( 'https://site.api.espn.com/apis/site/v2/sports/football/nfl/scoreboard', - params={'dates': '20260807', 'limit': ESPN_MAX_LIMIT}, - cache_key='espn_football_nfl_20260807', - cache_ttl=300, + {'dates': '20260807', 'limit': ESPN_MAX_LIMIT}, + None, None, None, 300, 300, ) + read = [c.args[0] for c in cache.get_cached_data.call_args_list] + assert read == ['espn_scoreboard_football_nfl_20260807', 'espn_football_nfl_20260807'] + cache.set.assert_called_once_with('espn_scoreboard_football_nfl_20260807', {'ok': 1}) - def test_fetch_espn_scoreboard_explicit_date(self, helper): - helper.get = Mock(return_value=None) + def test_fetch_espn_scoreboard_explicit_date(self, helper, cache): + cache.get_cached_data.return_value = None + helper._get = Mock(return_value=None) helper.fetch_espn_scoreboard('basketball', 'nba', date='20250115') - kwargs = helper.get.call_args.kwargs - assert kwargs['params'] == {'dates': '20250115', 'limit': ESPN_MAX_LIMIT} - assert kwargs['cache_key'] == 'espn_basketball_nba_20250115' + args = helper._get.call_args.args + assert args[1] == {'dates': '20250115', 'limit': ESPN_MAX_LIMIT} + cache.set.assert_not_called() # a failed fetch caches nothing + + def test_fetch_espn_scoreboard_an_explicit_key_works_as_before(self, helper): + helper.get = Mock(return_value={'ok': 1}) + + helper.fetch_espn_scoreboard('basketball', 'nba', date='20250115', cache_key='mine') + + assert helper.get.call_args.kwargs['cache_key'] == 'mine' + + def test_fetch_espn_scoreboard_reads_the_old_key_after_an_upgrade(self, helper, cache): + import time as _time + + records = {'espn_basketball_nba_20250115': { + 'timestamp': _time.time() - 10, 'ttl': 300, 'data': {'events': ['old']}}} + cache.get_cached_data.side_effect = lambda key, **kw: records.get(key) + helper._get = Mock() + + assert helper.fetch_espn_scoreboard('basketball', 'nba', date='20250115') == { + 'events': ['old']} + helper._get.assert_not_called() def test_fetch_espn_standings_url_and_cache_key(self, helper): helper.get = Mock(return_value={'ok': 1}) diff --git a/test/test_espn_scoreboard_cache.py b/test/test_espn_scoreboard_cache.py new file mode 100644 index 000000000..8f509a4ba --- /dev/null +++ b/test/test_espn_scoreboard_cache.py @@ -0,0 +1,477 @@ +"""One cache key per ESPN scoreboard (fetch service stage 2). + +- every core caller names a scoreboard with the same key; +- the keys each consumer used before are read as a fallback; +- a shared entry is never returned older than the reader's own max_age, + whoever wrote it and whatever ttl they stored (checked against a real + CacheManager, whose stored ttl otherwise wins over the reader's); +- two consumers of one scoreboard cost one request; +- the counters see it. + +No network: sessions are fakes, and the fetch service is a fresh one per test. +""" + +import json +import logging +import threading +import time +from datetime import date, datetime + +import pytest +import requests +from requests.structures import CaseInsensitiveDict + +from src.common import espn_dates +from src.common import fetch_service as fs +from src.common.espn_dates import ( + ESPN_MAX_LIMIT, + espn_scoreboard_cache_key, + espn_scoreboard_cache_key_for_url, + espn_scoreboard_url, + get_espn_scoreboard, + read_espn_scoreboard_cache, + store_espn_scoreboard_cache, +) +from src.common.fetch_service import FetchService, plugin_scope +from src.common.sports_fetch import SportsFetchMixin + +NFL_URL = "https://site.api.espn.com/apis/site/v2/sports/football/nfl/scoreboard" + + +# --- fakes --------------------------------------------------------------------------- + +class Clock: + def __init__(self): + self.t = 1000.0 + + def now(self): + return self.t + + +def _response(body, url, max_age=None): + response = requests.Response() + response.status_code = 200 + response._content = json.dumps(body).encode() + response.headers = CaseInsensitiveDict( + {"Cache-Control": f"max-age={max_age}"} if max_age else {}) + response.url = url + response.encoding = "utf-8" + return response + + +class Espn(requests.Session): + """A Session answering every scoreboard GET with ``{"events": []}``.""" + + def __init__(self, max_age=None, fail=None): + super().__init__() + self.calls = [] + self.max_age = max_age + self.fail = fail + self._lock = threading.Lock() + + def get(self, url, **kwargs): + with self._lock: + self.calls.append((url, dict(kwargs.get("params") or {}))) + if self.fail: + raise self.fail + dates = (kwargs.get("params") or {}).get("dates", "current") + return _response({"events": [{"id": dates}]}, url, self.max_age) + + +class RecordCache: + """The CacheManager surface these helpers use, records and all.""" + + def __init__(self): + self.records = {} + self.deleted = [] + + def set(self, key, data, ttl=None): + record = {"timestamp": time.time()} + if ttl is not None: + record["ttl"] = ttl + record["data"] = data + self.records[key] = record + + def put(self, key, data, age, ttl=None): + self.set(key, data, ttl) + self.records[key]["timestamp"] -= age + + def get_cached_data(self, key, max_age=300, memory_ttl=None): + return self.records.get(key) + + def get(self, key, max_age=300, memory_ttl=None): + record = self.records.get(key) + return record["data"] if record else None + + def delete(self, key): + self.deleted.append(key) + self.records.pop(key, None) + + +@pytest.fixture +def service(monkeypatch): + svc = FetchService({"rate_limits": {}}) + monkeypatch.setattr(fs, "_service", svc) + return svc + + +def _totals(svc): + return svc.snapshot()["totals"] + + +# --- the key ------------------------------------------------------------------------- + +class TestCanonicalKey: + + def test_one_spelling_for_one_scoreboard(self): + key = "espn_scoreboard_football_nfl_20261004" + assert espn_scoreboard_cache_key("football", "nfl", "20261004") == key + assert espn_scoreboard_cache_key(" Football ", "NFL", "20261004") == key + assert espn_scoreboard_cache_key("football", "nfl", date(2026, 10, 4)) == key + assert espn_scoreboard_cache_key("football", "nfl", datetime(2026, 10, 4, 23, 59)) == key + assert espn_scoreboard_cache_key("football", "nfl", ("20261004", date(2026, 10, 4))) == key + assert espn_scoreboard_cache_key_for_url(NFL_URL, "20261004") == key + assert espn_scoreboard_cache_key_for_url(NFL_URL + "?limit=500", "20261004") == key + + @pytest.mark.parametrize("dates,suffix", [ + (None, "current"), + ("", "current"), + ("2026", "2026"), + ("202610", "202610"), + ("20260925-20261016", "20260925-20261016"), + ((date(2026, 9, 25), date(2026, 10, 16)), "20260925-20261016"), + ]) + def test_every_dates_form_espn_takes(self, dates, suffix): + assert espn_scoreboard_cache_key("soccer", "eng.1", dates) == f"espn_scoreboard_soccer_eng.1_{suffix}" + + @pytest.mark.parametrize("sport,league,dates", [ + ("football", "nfl", "2026-10-04"), + ("football", "nfl", "tomorrow"), + ("football", "nfl", ("20261004",)), + ("football", "nfl", True), + ("football", "", "20261004"), + ("foot ball", "nfl", "20261004"), + ("football", "nfl/../x", "20261004"), + ]) + def test_anything_else_is_refused_not_guessed(self, sport, league, dates): + with pytest.raises(ValueError): + espn_scoreboard_cache_key(sport, league, dates) + + def test_a_url_that_is_not_a_scoreboard_has_no_key(self): + assert espn_scoreboard_cache_key_for_url( + "https://site.api.espn.com/apis/site/v2/sports/football/nfl/teams") is None + + def test_the_url_it_names(self): + assert espn_scoreboard_url("Football", "college-football") == ( + "https://site.api.espn.com/apis/site/v2/sports/football/college-football/scoreboard") + + +class Host(SportsFetchMixin): + def __init__(self, session, cache): + self.session = session + self.headers = {"User-Agent": "test"} + self.cache_manager = cache + self.logger = logging.getLogger("test.espn_cache") + self._games_lock = threading.RLock() + self.sport = "football" + self.league = "nfl" + + +class TestEveryCallerUsesTheKey: + """The core helpers that cache a scoreboard all write the same key.""" + + DAY = "20261004" + KEY = "espn_scoreboard_football_nfl_20261004" + + def test_get_espn_scoreboard(self, service): + cache = RecordCache() + get_espn_scoreboard(Espn(), "football", "nfl", self.DAY, cache_manager=cache) + assert list(cache.records) == [self.KEY] + + def test_api_helper(self, service): + from src.common.api_helper import APIHelper + + cache = RecordCache() + helper = APIHelper(cache_manager=cache) + helper.set_rate_limit(0) + helper.session = Espn() + helper.fetch_espn_scoreboard("football", "nfl", date=self.DAY) + assert list(cache.records) == [self.KEY] + + def test_the_scoreboard_mixin(self, service): + cache = RecordCache() + host = Host(Espn(), cache) + assert host._schedule_cache_key(self.DAY) == self.KEY + host._fetch_season_directly(NFL_URL, self.DAY, None, "today") + assert list(cache.records) == [self.KEY] + + def test_the_mixin_without_a_scoreboard_url(self, service): + cache = RecordCache() + Host(Espn(), cache)._fetch_season_directly( + "https://example.test/feed", self.DAY, None, "today") + assert list(cache.records) == [self.KEY] + + def test_an_explicit_key_is_still_the_callers(self, service): + cache = RecordCache() + Host(Espn(), cache)._fetch_season_directly(NFL_URL, self.DAY, "mine", "today") + assert list(cache.records) == ["mine"] + + +# --- reading --------------------------------------------------------------------------- + +class TestRead: + + KEY = "espn_scoreboard_football_nfl_20261004" + OLD = "scoreboard_data_football_nfl_20261004" + + def test_a_fresh_canonical_entry(self, service): + cache = RecordCache() + cache.put(self.KEY, {"events": ["new"]}, age=10) + assert read_espn_scoreboard_cache(cache, self.KEY, 30) == {"events": ["new"]} + assert (_totals(service)["cache_hits"], _totals(service)["legacy_cache_hits"]) == (1, 0) + + def test_the_old_key_is_read_after_an_upgrade(self, service): + cache = RecordCache() + cache.put(self.OLD, {"events": ["old"]}, age=10) + assert read_espn_scoreboard_cache(cache, self.KEY, 30, [self.OLD]) == {"events": ["old"]} + assert (_totals(service)["cache_hits"], _totals(service)["legacy_cache_hits"]) == (1, 1) + + def test_the_canonical_key_wins_over_an_old_one(self, service): + cache = RecordCache() + cache.put(self.OLD, {"events": ["old"]}, age=1) + cache.put(self.KEY, {"events": ["new"]}, age=20) + assert read_espn_scoreboard_cache(cache, self.KEY, 30, [self.OLD]) == {"events": ["new"]} + + def test_a_stale_old_key_is_a_miss_too(self, service): + cache = RecordCache() + cache.put(self.OLD, {"events": ["old"]}, age=31) + assert read_espn_scoreboard_cache(cache, self.KEY, 30, [self.OLD]) is None + assert _totals(service)["cache_hits"] == 0 + + def test_age_is_judged_against_an_injected_clock(self, service): + cache = RecordCache() + cache.records[self.KEY] = {"timestamp": 5000.0, "data": {"events": []}} + assert read_espn_scoreboard_cache(cache, self.KEY, 30, now=5030.0) == {"events": []} + assert read_espn_scoreboard_cache(cache, self.KEY, 30, now=5030.5) is None + + def test_never_older_than_the_readers_ttl_whatever_the_writer_stored(self, service): + # A writer that stored ttl=3600 must not make a 30 s reader take an + # entry a minute old (CacheManager itself would let the ttl win). + cache = RecordCache() + cache.put(self.KEY, {"events": []}, age=60, ttl=3600) + assert read_espn_scoreboard_cache(cache, self.KEY, 30) is None + assert read_espn_scoreboard_cache(cache, self.KEY, 90) == {"events": []} + + @pytest.mark.parametrize("max_age", [0, -5]) + def test_no_max_age_no_read(self, service, max_age): + cache = RecordCache() + cache.put(self.KEY, {"events": []}, age=0) + assert read_espn_scoreboard_cache(cache, self.KEY, max_age) is None + + def test_a_broken_cache_is_a_miss(self, service): + class Broken: + def get_cached_data(self, *a, **k): + raise OSError("disk gone") + assert read_espn_scoreboard_cache(Broken(), self.KEY, 30) is None + assert read_espn_scoreboard_cache(None, self.KEY, 30) is None + + def test_a_cache_without_records(self, service): + class Plain: + def get(self, key, max_age=None): + return {"events": [key, max_age]} + assert read_espn_scoreboard_cache(Plain(), self.KEY, 29.5) == {"events": [self.KEY, 30]} + + +class TestWithARealCacheManager: + """The same promise against CacheManager, memory and disk tiers.""" + + KEY = "espn_scoreboard_football_nfl_20261004" + + @pytest.fixture + def cm(self, tmp_path): + from src.cache.disk_cache import DiskCache + from src.cache.memory_cache import MemoryCache + from src.cache_manager import CacheManager + + cm = CacheManager() + cm._disk_cache_component = DiskCache(cache_dir=str(tmp_path)) + cm._memory_cache_component = MemoryCache() + return cm + + def _age_on_disk(self, cm, seconds): + path = cm._disk_cache_component.get_cache_path(self.KEY) + with open(path, encoding="utf-8") as fh: + record = json.load(fh) + record["timestamp"] = time.time() - seconds + with open(path, "w", encoding="utf-8") as fh: + json.dump(record, fh) + cm._memory_cache_component.clear() + + def test_a_writers_long_ttl_does_not_outlast_the_readers(self, cm, service): + cm.set(self.KEY, {"events": [1]}, ttl=3600) + self._age_on_disk(cm, 60) + assert cm.get(self.KEY, max_age=30) == {"events": [1]} # what a plain read does + assert read_espn_scoreboard_cache(cm, self.KEY, 30) is None + assert read_espn_scoreboard_cache(cm, self.KEY, 120) == {"events": [1]} + + def test_a_copy_loaded_into_memory_keeps_its_real_age(self, cm, service): + store_espn_scoreboard_cache(cm, self.KEY, {"events": [1]}) + self._age_on_disk(cm, 60) + assert read_espn_scoreboard_cache(cm, self.KEY, 120) == {"events": [1]} # now in memory + assert read_espn_scoreboard_cache(cm, self.KEY, 30) is None + + def test_the_canonical_entry_stores_no_ttl(self, cm, service): + store_espn_scoreboard_cache(cm, self.KEY, {"events": []}) + path = cm._disk_cache_component.get_cache_path(self.KEY) + with open(path, encoding="utf-8") as fh: + assert "ttl" not in json.load(fh) + + +# --- fetching through the cache ---------------------------------------------------------------- + +class TestGetEspnScoreboard: + + def test_a_miss_fetches_a_whole_page_and_caches_it(self, service): + cache, espn = RecordCache(), Espn() + data = get_espn_scoreboard(espn, "football", "nfl", "20261004", cache_manager=cache) + assert data == {"events": [{"id": "20261004"}]} + assert espn.calls == [(NFL_URL, {"dates": "20261004", "limit": ESPN_MAX_LIMIT})] + assert "ttl" not in cache.records["espn_scoreboard_football_nfl_20261004"] + + def test_a_hit_sends_nothing(self, service): + cache, espn = RecordCache(), Espn() + get_espn_scoreboard(espn, "football", "nfl", "20261004", cache_manager=cache) + get_espn_scoreboard(espn, "football", "nfl", "20261004", cache_manager=cache, max_age=60) + assert len(espn.calls) == 1 + assert _totals(service)["cache_hits"] == 1 + + def test_max_age_zero_always_fetches_and_still_caches(self, service): + cache, espn = RecordCache(), Espn() + get_espn_scoreboard(espn, "football", "nfl", "20261004", cache_manager=cache, max_age=0) + get_espn_scoreboard(espn, "football", "nfl", "20261004", cache_manager=cache, max_age=0) + assert len(espn.calls) == 2 + assert "espn_scoreboard_football_nfl_20261004" in cache.records + + def test_the_current_scoreboard_sends_no_dates(self, service): + espn = Espn() + get_espn_scoreboard(espn, "football", "nfl", cache_manager=RecordCache()) + assert espn.calls == [(NFL_URL, {"limit": ESPN_MAX_LIMIT})] + + def test_a_range_is_fetched_in_chunks_once_espn_rejects_it(self, service, monkeypatch): + monkeypatch.setattr(espn_dates, "_ranges_rejected_until", time.monotonic() + 60) + cache, espn = RecordCache(), Espn() + data = get_espn_scoreboard(espn, "football", "nfl", "20261003-20261004", cache_manager=cache) + assert [e["id"] for e in data["events"]] == ["20261003", "20261004"] + assert list(cache.records) == ["espn_scoreboard_football_nfl_20261003-20261004"] + + def test_a_failure_raises_and_caches_nothing(self, service): + cache = RecordCache() + with pytest.raises(requests.ConnectionError): + get_espn_scoreboard(Espn(fail=requests.ConnectionError("down")), "football", "nfl", + "20261004", cache_manager=cache) + assert cache.records == {} + + def test_no_cache_manager_just_fetches(self, service): + espn = Espn() + get_espn_scoreboard(espn, "football", "nfl", "20261004") + get_espn_scoreboard(espn, "football", "nfl", "20261004") + assert len(espn.calls) == 2 + + def test_the_response_cache_is_bounded_by_the_callers_ttl(self, monkeypatch): + clock = Clock() + svc = FetchService({"rate_limits": {}}, clock=clock.now) + monkeypatch.setattr(fs, "_service", svc) + espn = Espn(max_age=450) + get_espn_scoreboard(espn, "football", "nfl", "20261004", max_age=0) + clock.t += 20 + get_espn_scoreboard(espn, "football", "nfl", "20261004", max_age=10) + assert len(espn.calls) == 2 # 20 s old > the caller's 10 + clock.t += 5 + get_espn_scoreboard(espn, "football", "nfl", "20261004", max_age=10) + assert len(espn.calls) == 2 # 5 s old, and ESPN said 450 + assert svc.snapshot()["totals"]["memo_hits"] == 1 + + +class TestTwoConsumersOneFetch: + """A scoreboard's live poll and odds-ticker asking for the same day.""" + + def test_the_second_consumer_is_served_from_the_first_ones_copy(self, service): + cache, espn = RecordCache(), Espn() + with plugin_scope("football-scoreboard"): + live = get_espn_scoreboard(espn, "football", "nfl", "20261004", + cache_manager=cache, max_age=0) + with plugin_scope("odds-ticker"): + ticker = get_espn_scoreboard( + espn, "football", "nfl", "20261004", cache_manager=cache, max_age=300, + legacy_keys=["scoreboard_data_football_nfl_20261004"]) + assert ticker == live + assert len(espn.calls) == 1 + snap = service.snapshot() + assert snap["plugins"]["odds-ticker"]["cache_hits"] == 1 + assert snap["plugins"]["odds-ticker"]["requests"] == 0 + assert snap["plugins"]["football-scoreboard"]["requests"] == 1 + + def test_concurrent_misses_go_to_espn_once(self, service): + gate = threading.Event() + + class Slow(Espn): + def get(self, url, **kwargs): + gate.wait(5) + return super().get(url, **kwargs) + + espn, results = Slow(), [] + threads = [threading.Thread(target=lambda: results.append(get_espn_scoreboard( + espn, "football", "nfl", "20261004", cache_manager=RecordCache(), max_age=60))) + for _ in range(2)] + for thread in threads: + thread.start() + deadline = time.monotonic() + 5 + while time.monotonic() < deadline: + with service._lock: + if any(f.waiters for f in service._inflight.values()): + break + time.sleep(0.005) + gate.set() + for thread in threads: + thread.join(5) + assert len(espn.calls) == 1 and results[0] == results[1] + + +# --- the scoreboards' schedule window ---------------------------------------------------------- + +class TestCachedSchedule: + + KEY = "espn_scoreboard_football_nfl_20260925-20261016" + YESTERDAY = "espn_scoreboard_football_nfl_20260924-20261015" + OLD = "nfl_schedule_window_7_14" + + def test_the_canonical_copy(self, service): + cache = RecordCache() + cache.put(self.KEY, {"events": ["new"]}, age=0) + assert Host(Espn(), cache)._cached_schedule(self.KEY, [self.OLD]) == {"events": ["new"]} + assert cache.deleted == [] + + def test_the_old_key_after_an_upgrade_counted_as_legacy(self, service): + cache = RecordCache() + cache.put(self.OLD, {"events": ["old"]}, age=0) + assert Host(Espn(), cache)._cached_schedule(self.KEY, [self.OLD]) == {"events": ["old"]} + totals = _totals(service) + assert (totals["legacy_cache_hits"], totals["cache_hits"]) == (1, 0) + + def test_a_miss_retires_yesterdays_window(self, service): + cache = RecordCache() + cache.put(self.YESTERDAY, {"events": []}, age=0) + assert Host(Espn(), cache)._cached_schedule(self.KEY) is None + assert cache.deleted == [self.YESTERDAY] + assert self.YESTERDAY not in cache.records + + def test_a_key_that_is_not_a_window_retires_nothing(self, service): + cache = RecordCache() + Host(Espn(), cache)._cached_schedule("espn_scoreboard_football_nfl_20261004") + Host(Espn(), cache)._cached_schedule(self.OLD) + assert cache.deleted == [] + + def test_a_cache_without_delete(self, service): + class NoDelete(RecordCache): + delete = None + assert Host(Espn(), NoDelete())._cached_schedule(self.KEY) is None diff --git a/test/test_fetch_service.py b/test/test_fetch_service.py index 397d9af93..e25ec56a7 100644 --- a/test/test_fetch_service.py +++ b/test/test_fetch_service.py @@ -876,3 +876,267 @@ def test_the_web_route_returns_the_published_counters(clock): assert body["status"] == "success" assert body["data"]["status"] == "live" assert body["data"]["data"]["plugins"]["weather"]["requests"] == 1 + + +# --- response cache (stage 2: Cache-Control max-age) ------------------------------------------- + +def fresh_for(seconds, body=b'{"ok": 1}', **headers): + """A handler answering 200 with ``Cache-Control: max-age=``.""" + def handler(url, kwargs): + return make_response(200, body, url=url, headers={ + "Cache-Control": f"max-age={seconds}", **headers}) + return handler + + +class NeverHits: + """A cache that always misses, so APIHelper always reaches the network.""" + + def get(self, key, max_age=None): + return None + + def set(self, key, value, ttl=None): + pass + + +class TestResponseCache: + + def test_an_identical_get_inside_max_age_is_not_sent(self, service): + session = FakeSession(fresh_for(60)) + first = service.get(session, "https://api.test/x", params={"d": 1}, timeout=5) + second = service.get(session, "https://api.test/x", params={"d": 1}, timeout=5) + assert len(session.calls) == 1 + assert second.json() == first.json() == {"ok": 1} + assert second is not first + totals = _counters(service) + assert (totals["requests"], totals["memo_hits"]) == (1, 1) + + def test_a_hit_is_a_copy_the_caller_may_change(self, service): + session = FakeSession(fresh_for(60)) + service.get(session, "https://api.test/x").headers["X-Mine"] = "1" + assert "X-Mine" not in service.get(session, "https://api.test/x").headers + + def test_past_max_age_the_network_is_asked_again(self, service, clock): + session = FakeSession(fresh_for(10)) + service.get(session, "https://api.test/x") + clock.advance(9.9) + service.get(session, "https://api.test/x") + assert len(session.calls) == 1 + clock.advance(0.2) + service.get(session, "https://api.test/x") + assert len(session.calls) == 2 + + def test_the_age_header_shortens_the_lifetime(self, service, clock): + session = FakeSession(fresh_for(10, Age="8")) + service.get(session, "https://api.test/x") + clock.advance(2.5) + service.get(session, "https://api.test/x") + assert len(session.calls) == 2 + + def test_never_older_than_the_callers_own_ttl(self, service, clock): + session = FakeSession(fresh_for(400)) + service.get(session, "https://api.test/x", cache_max_age=0) + clock.advance(20) + service.get(session, "https://api.test/x", cache_max_age=10) # 20 s > 10 + assert len(session.calls) == 2 + clock.advance(20) + service.get(session, "https://api.test/x", cache_max_age=30) # 20 s <= 30 + assert len(session.calls) == 2 + service.get(session, "https://api.test/x", cache_max_age=19.9) # 20 s > 19.9 + assert len(session.calls) == 3 + + def test_a_caller_that_does_not_say_gets_the_default_limit(self, service, clock): + session = FakeSession(fresh_for(450)) + service.get(session, "https://api.test/x") + clock.advance(service.default_max_age - 0.5) + service.get(session, "https://api.test/x") + assert len(session.calls) == 1 + clock.advance(1) + service.get(session, "https://api.test/x") + assert len(session.calls) == 2 + # ...while a caller with a longer TTL still takes the server at its word. + clock.advance(100) + service.get(session, "https://api.test/x", cache_max_age=300) + assert len(session.calls) == 2 + + def test_zero_always_asks_but_still_fills_the_cache(self, service): + session = FakeSession(fresh_for(60)) + service.get(session, "https://api.test/x", cache_max_age=0) + service.get(session, "https://api.test/x", cache_max_age=0) + assert len(session.calls) == 2 + service.get(session, "https://api.test/x") + assert len(session.calls) == 2 + + def test_cache_max_age_never_reaches_the_session(self, service): + session = FakeSession(fresh_for(60)) + service.get(session, "https://api.test/x", timeout=5, cache_max_age=30) + assert session.calls[0][1] == {"timeout": 5} + + @pytest.mark.parametrize("status,headers", [ + (200, {"Cache-Control": "no-store, max-age=60"}), + (200, {"Cache-Control": "no-cache, max-age=60"}), + (200, {"Cache-Control": "private, max-age=60"}), + (200, {"Cache-Control": "max-age=60", "Vary": "*"}), + (200, {"Cache-Control": "max-age=60", "Set-Cookie": "sid=1"}), + (200, {"Cache-Control": "max-age=0"}), + (200, {"Cache-Control": "max-age=soon"}), + (200, {"Cache-Control": "max-age=60", "Age": "60"}), + (200, {}), + (404, {"Cache-Control": "max-age=60"}), + (500, {"Cache-Control": "max-age=60"}), + ]) + def test_responses_that_must_not_be_reused_are_not(self, service, status, headers): + session = FakeSession(lambda url, kw: make_response(status, b"{}", headers=headers, url=url)) + service.get(session, "https://api.test/x") + service.get(session, "https://api.test/x") + assert len(session.calls) == 2 + assert service.snapshot()["response_cache"]["entries"] == 0 + + def test_a_stream_is_never_kept(self, service): + session = FakeSession(fresh_for(60)) + service.get(session, "https://api.test/x", stream=True) + service.get(session, "https://api.test/x", stream=True) + assert len(session.calls) == 2 + + @pytest.mark.parametrize("second", [ + {"params": {"d": 2}}, + {"headers": {"Accept": "text/html"}}, + {"headers": {"If-None-Match": '"x"'}}, # the caller's own revalidation + ]) + def test_requests_that_could_answer_differently_do_not_share(self, service, second): + session = FakeSession(fresh_for(60)) + service.get(session, "https://api.test/x", params={"d": 1}) + service.get(session, "https://api.test/x", **{"params": {"d": 1}, **second}) + assert len(session.calls) == 2 + + def test_another_timeout_or_retry_policy_still_shares(self, service): + # A finished 200 is the same answer however long the caller would have + # waited or however often retried; only in-flight merging needs those. + plain, retrying = FakeSession(fresh_for(60)), FakeSession(fresh_for(60)) + retrying.mount("https://", requests.adapters.HTTPAdapter(max_retries=Retry(total=5))) + service.get(plain, "https://api.test/x", timeout=5) + service.get(retrying, "https://api.test/x", timeout=30) + assert len(plain.calls) == 1 and retrying.calls == [] + + def test_a_session_with_cookies_only_reuses_its_own(self, service): + cookied, plain = FakeSession(fresh_for(60)), FakeSession(fresh_for(60)) + cookied.cookies.set("sid", "secret") + service.get(cookied, "https://api.test/x") + service.get(plain, "https://api.test/x") + service.get(cookied, "https://api.test/x") + assert len(cookied.calls) == len(plain.calls) == 1 + + def test_size_bounds(self, clock): + svc = FetchService({"rate_limits": {}, "response_cache": { + "max_entry_bytes": 10, "max_bytes": 25, "max_entries": 10}}, + clock=clock.now, sleep=clock.sleep) + big = FakeSession(fresh_for(60, body=b"x" * 11)) + svc.get(big, "https://api.test/big") + assert svc.snapshot()["response_cache"]["entries"] == 0 + small = FakeSession(fresh_for(60, body=b"y" * 10)) + for name in "abc": + svc.get(small, f"https://api.test/{name}") + assert svc.snapshot()["response_cache"] == {"entries": 2, "bytes": 20} + svc.get(small, "https://api.test/a") # evicted, least recently used + assert len(small.calls) == 4 + + def test_expired_entries_are_dropped_on_insert(self, service, clock): + session = FakeSession(fresh_for(5)) + service.get(session, "https://api.test/a") + clock.advance(6) + service.get(session, "https://api.test/b") + assert service.snapshot()["response_cache"]["entries"] == 1 + + def test_off_switch(self, clock): + svc = FetchService({"rate_limits": {}, "response_cache": {"enabled": False}}, + clock=clock.now, sleep=clock.sleep) + session = FakeSession(fresh_for(60)) + svc.get(session, "https://api.test/x") + svc.get(session, "https://api.test/x") + assert len(session.calls) == 2 + assert svc.describe_config()["response_cache"] is False + + def test_merged_callers_and_the_cache_together(self, service): + gate = threading.Event() + session = FakeSession(fresh_for(60), gate=gate) + a = threading.Thread(target=lambda: service.get(session, "https://api.test/x")) + a.start() + assert session.started.wait(5) + b = threading.Thread(target=lambda: service.get(session, "https://api.test/x")) + b.start() + _wait_for_waiters(service, 1) + gate.set() + a.join(5) + b.join(5) + service.get(session, "https://api.test/x") + assert len(session.calls) == 1 + totals = _counters(service) + assert (totals["requests"], totals["merged"], totals["memo_hits"]) == (1, 1, 1) + + def test_hits_are_counted_per_plugin_and_host(self, service): + session = FakeSession(fresh_for(60)) + with plugin_scope("odds-ticker"): + service.get(session, "https://site.api.espn.com/x") + with plugin_scope("football-scoreboard"): + service.get(session, "https://site.api.espn.com/x") + snap = service.snapshot() + assert snap["plugins"]["football-scoreboard"]["memo_hits"] == 1 + assert snap["plugins"]["football-scoreboard"]["requests"] == 0 + assert snap["plugins"]["football-scoreboard"]["hosts"] == {"site.api.espn.com": 1} + assert snap["hosts"]["site.api.espn.com"]["memo_hits"] == 1 + + def test_the_odds_manager_never_takes_odds_older_than_its_interval(self, global_service, clock): + from unittest.mock import MagicMock + from src.base_odds_manager import BaseOddsManager + + cache = MagicMock() + cache.get_with_auto_strategy.return_value = None + manager = BaseOddsManager(cache) + manager.session = FakeSession(fresh_for(450, body=b'{"count": 0, "items": []}')) + manager.get_odds("football", "nfl", "401", update_interval_seconds=60) + clock.advance(59) + manager.get_odds("football", "nfl", "401", update_interval_seconds=60) + assert len(manager.session.calls) == 1 + clock.advance(2) + manager.get_odds("football", "nfl", "401", update_interval_seconds=60) + assert len(manager.session.calls) == 2 + + def test_the_api_helper_bounds_a_cached_get_by_its_ttl(self, global_service, clock): + from src.common.api_helper import APIHelper + + helper = APIHelper() + helper.set_rate_limit(0) + helper.session = FakeSession(fresh_for(450, body=b'{"a": 1}')) + helper.get("https://api.test/x") + clock.advance(31) + helper.get("https://api.test/x") # no TTL: the 30 s default + assert len(helper.session.calls) == 2 + clock.advance(31) + helper.cache_manager = NeverHits() + helper.get("https://api.test/x", cache_key="k", cache_ttl=60) # 31 s <= 60 + assert len(helper.session.calls) == 2 + + +class TestCacheHitCounters: + + def test_a_shared_cache_hit(self, service): + with plugin_scope("odds-ticker"): + service.note_cache_hit("https://site.api.espn.com/") + counters = _counters(service, plugin="odds-ticker") + assert (counters["cache_hits"], counters["legacy_cache_hits"]) == (1, 0) + assert counters["hosts"] == {"site.api.espn.com": 1} + + def test_a_legacy_hit(self, service): + service.note_cache_hit("https://site.api.espn.com/", legacy=True) + totals = _counters(service) + assert (totals["cache_hits"], totals["legacy_cache_hits"]) == (1, 1) + + def test_a_read_that_avoided_no_request(self, service): + service.note_cache_hit("https://site.api.espn.com/", legacy=True, avoided_request=False) + totals = _counters(service) + assert (totals["cache_hits"], totals["legacy_cache_hits"]) == (0, 1) + + def test_every_counter_is_in_every_snapshot(self, service): + snap = service.snapshot() + for name in ("memo_hits", "cache_hits", "legacy_cache_hits"): + assert snap["totals"][name] == 0 + assert snap["response_cache"] == {"entries": 0, "bytes": 0} From 76267aa214294dda14a1e13aea0c89aaf9c77d14 Mon Sep 17 00:00:00 2001 From: Chuck <33324927+ChuckBuilds@users.noreply.github.com> Date: Fri, 2 Oct 2026 13:41:43 -0400 Subject: [PATCH 2/2] feat(espn): read_espn_scoreboard_cache takes an accept(data, age) predicate A consumer whose limit depends on the payload (odds-ticker holds a scoreboard with a live game to its live interval) can turn an entry down in one read, and a turned-down entry is not counted as a hit. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 4 ++- src/common/espn_dates.py | 41 +++++++++++++++++------------- test/test_espn_scoreboard_cache.py | 16 ++++++++++++ 3 files changed, 43 insertions(+), 18 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 3c2d361ae..dc449a3f6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -108,7 +108,9 @@ policies are unchanged. - **Cache-through helpers.** `get_espn_scoreboard()` returns a cached copy at most `max_age` seconds old and otherwise fetches with `fetch_espn_scoreboard` and caches the result; `read_espn_scoreboard_cache()` - and `store_espn_scoreboard_cache()` are the two halves. A read checks the + and `store_espn_scoreboard_cache()` are the two halves (the read takes an + `accept(data, age)` predicate, e.g. a shorter limit for a payload holding a + live game). A read checks the record's own timestamp, so a writer's stored ttl can no longer make a reader take data older than its own TTL; the shared entry stores no ttl. Old keys are passed as `legacy_keys` and read after the canonical one for diff --git a/src/common/espn_dates.py b/src/common/espn_dates.py index 8347c2700..3b34c9eaf 100644 --- a/src/common/espn_dates.py +++ b/src/common/espn_dates.py @@ -57,7 +57,7 @@ from concurrent.futures import ThreadPoolExecutor from datetime import date, datetime, timedelta from functools import partial -from typing import Any, Dict, Iterable, List, Optional, Tuple, cast +from typing import Any, Callable, Dict, Iterable, List, Optional, Tuple, cast try: from src.common.json_body import response_json @@ -547,8 +547,9 @@ def _note_cache_hit(legacy: bool, avoided_request: bool = True) -> None: def _fresh_cached(cache_manager: Any, key: str, max_age: Optional[float], - now: float) -> Optional[Any]: - """The data cached under ``key`` if it is at most ``max_age`` seconds old. + now: float) -> Tuple[Optional[Dict[str, Any]], Optional[float]]: + """The data cached under ``key`` if it is at most ``max_age`` seconds + old, and its age (None when the cache does not say). The age is the stored record's own timestamp, checked here: CacheManager lets a ttl stored by the writer override the reader's max_age, and its @@ -560,20 +561,20 @@ def _fresh_cached(cache_manager: Any, key: str, max_age: Optional[float], if not callable(reader): # A cache without records (a test double, a plugin's own store). value = cache_manager.get(key, max_age=limit) - return value if isinstance(value, dict) else None + return (value if isinstance(value, dict) else None), None record = reader(key, max_age=limit, memory_ttl=limit) if not isinstance(record, dict): - return None + return None, None if "data" not in record: - return record # unwrapped; the cache already judged it by mtime - if max_age is not None: - stamp = record.get("timestamp") - if isinstance(stamp, bool) or not isinstance(stamp, (int, float)): - return None - if now - float(stamp) > max_age: - return None + return record, None # unwrapped; the cache already judged it by mtime + stamp = record.get("timestamp") + age: Optional[float] = None + if not isinstance(stamp, bool) and isinstance(stamp, (int, float)): + age = max(0.0, now - float(stamp)) + if max_age is not None and (age is None or age > max_age): + return None, None data = record["data"] - return data if isinstance(data, dict) else None + return (data if isinstance(data, dict) else None), age def read_espn_scoreboard_cache( @@ -582,14 +583,18 @@ def read_espn_scoreboard_cache( max_age: Optional[float], legacy_keys: Iterable[str] = (), now: Optional[float] = None, -) -> Optional[Any]: + accept: Optional[Callable[[Dict[str, Any], Optional[float]], bool]] = None, +) -> Optional[Dict[str, Any]]: """The cached scoreboard under ``key``, or under the first of ``legacy_keys`` that has one, if it is at most ``max_age`` seconds old. None on a miss, a stale entry, ``max_age`` of 0 or less, no cache manager, or any cache error -- a read never raises. ``max_age=None`` - takes an entry of any age. A hit is counted in the fetch statistics - (``cache_hits``; ``legacy_cache_hits`` too for an old key). + takes an entry of any age. ``accept(data, age_seconds)`` can turn down + an entry the age alone would allow (a payload holding a live game wants + a shorter limit); ``age_seconds`` is None when the cache cannot say. A + hit is counted in the fetch statistics (``cache_hits``; + ``legacy_cache_hits`` too for an old key). """ if cache_manager is None: return None @@ -600,7 +605,9 @@ def read_espn_scoreboard_cache( if not candidate: continue try: - data = _fresh_cached(cache_manager, candidate, max_age, clock) + data, age = _fresh_cached(cache_manager, candidate, max_age, clock) + if data is not None and accept is not None and not accept(data, age): + data = None except Exception: # noqa: BLE001 - a broken cache is a miss _logger.debug("scoreboard cache read failed for %s", candidate, exc_info=True) continue diff --git a/test/test_espn_scoreboard_cache.py b/test/test_espn_scoreboard_cache.py index 8f509a4ba..fe7332aa6 100644 --- a/test/test_espn_scoreboard_cache.py +++ b/test/test_espn_scoreboard_cache.py @@ -262,6 +262,22 @@ def test_never_older_than_the_readers_ttl_whatever_the_writer_stored(self, servi assert read_espn_scoreboard_cache(cache, self.KEY, 30) is None assert read_espn_scoreboard_cache(cache, self.KEY, 90) == {"events": []} + def test_accept_can_turn_down_an_entry_and_sees_its_age(self, service): + cache = RecordCache() + cache.records[self.KEY] = {"timestamp": 5000.0, "data": {"events": ["live"]}} + seen = [] + + def live_needs_30s(data, age): + seen.append(age) + return age <= 30 + + assert read_espn_scoreboard_cache(cache, self.KEY, 300, now=5040.0, + accept=live_needs_30s) is None + assert read_espn_scoreboard_cache(cache, self.KEY, 300, now=5020.0, + accept=live_needs_30s) == {"events": ["live"]} + assert seen == [40.0, 20.0] + assert _totals(service)["cache_hits"] == 1 # the turned-down read is no hit + @pytest.mark.parametrize("max_age", [0, -5]) def test_no_max_age_no_read(self, service, max_age): cache = RecordCache()