diff --git a/CHANGELOG.md b/CHANGELOG.md index 35ac9182..a350177d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -19,6 +19,29 @@ accepts both, but the store flags the old spelling as deprecated ## Unreleased +### ESPN date-range fetches: fewer requests, fewer at once + +A soccer board (8 leagues, ESPN rejecting `dates=` ranges) logged ~90 +`NameResolutionError` lines and an `update() timed out` at every start on a +Pi: each league's fortnight-either-side window was 29 day requests, fetched +by several managers at once, ~40 in flight. Measured against live ESPN with +soccer-scoreboard 2.39.2, alternating runs: **~450 requests per start, peak +~45 in flight, ~75 DNS lookups -> 46 requests, peak 13, ~30 lookups**. + +- `fetch_espn_date_chunks()` asks for a window's partial edge month whole + when the window covers `ESPN_MONTH_COVER_MIN_DAYS` (7) or more of its days, + and trims the answer to the window's days by each event's US Eastern start + date -- the day ESPN's `dates=YYYYMMDD` means (417 of 417 live soccer + events matched). A 29-day window spanning two months is 2 requests instead + of 29. Short windows (a live poll's 1-2 days) stay day by day. A trimmed + month that comes back at the 500-event cap re-asks only the window's days. + An event with no readable date is kept. New: `espn_request_chunks()`. +- Chunk requests share one process-wide cap of `ESPN_CHUNK_WORKERS` (6) in + flight, across every window being fetched, instead of six per window. +- A new process starts as if a range had just been rejected, so it no longer + spends one doomed 400 per window at every start (eleven at once from a + soccer board); the range is still retried `RANGE_RETRY_SECONDS` in. + ### Cheap per-frame and per-fetch savings - `BaseOddsManager.get_odds()` no longer pretty-prints every odds response diff --git a/src/common/README.md b/src/common/README.md index 9181db49..a0f46287 100644 --- a/src/common/README.md +++ b/src/common/README.md @@ -109,8 +109,11 @@ and the plugin test harness all use it. Most plugins get BDF text through [`espn_dates.py`](espn_dates.py). ESPN's site API rejects `dates=` ranges and truncates results when `limit` is above 500. `fetch_espn_scoreboard()` splits a range into month and day requests ESPN accepts and merges the -results; `espn_date_chunks()`, `fetch_espn_date_chunks()`, -`clamp_espn_limit()` and `merge_scoreboard_payloads()` are the pieces. +results; `espn_date_chunks()`, `espn_request_chunks()`, +`fetch_espn_date_chunks()`, `clamp_espn_limit()` and +`merge_scoreboard_payloads()` are the pieces. A window's partial edge months +are asked whole and trimmed to its days (US Eastern), and chunk requests share +one process-wide cap of `ESPN_CHUNK_WORKERS` in flight. Every request goes through [`fetch_service`](#fetch_service), the chunks counted against the plugin that asked. Scoreboard plugins also bundle a copy for older cores. diff --git a/src/common/espn_dates.py b/src/common/espn_dates.py index 3b34c9ea..b8b84f70 100644 --- a/src/common/espn_dates.py +++ b/src/common/espn_dates.py @@ -26,10 +26,30 @@ A month can hold more than 500 events (college baseball's March does), and ESPN answers that with exactly ``limit`` events and no hint that more exist. A month chunk that comes back full is therefore re-asked day by day. +A window's *partial* edge months are asked for whole, too, once the window +covers ``ESPN_MONTH_COVER_MIN_DAYS`` or more of their days, and the answer is +trimmed back to the window's days. A scoreboard's default fortnight either side +of today (29 days, two partial months) was 29 day requests per league; it is +now 2. Trimming needs ESPN's "game day", which is the event's start in US +Eastern time -- checked against the live API on 2026-10-03: 417 of 417 soccer +events across five leagues and three months (one of them spanning the end of +daylight saving) came back from exactly the day query their Eastern date +names. A short window (a live poll's one or two days) stays day by day, so it +never downloads a whole month to read a day of it. + +Chunk requests share one process-wide budget of ``ESPN_CHUNK_WORKERS`` in +flight, however many windows are being fetched at once. Each window used to get +its own six, so a scoreboard starting eight leagues -- each with a recent and +an upcoming manager -- had ~40 requests in flight, every one beyond a session's +pool a new connection and a new DNS lookup. On a Pi whose resolver could not +keep up, that was ~90 ``NameResolutionError`` lines within a minute of every +start. + Once a range has been rejected, later ranges skip straight to chunks for ``RANGE_RETRY_SECONDS`` instead of spending a doomed request first -- live scoreboards ask every 30 seconds. After that the range is tried again, so the -workaround retires itself if ESPN reverts. +workaround retires itself if ESPN reverts. A process starts inside that +period, as if a range had just been rejected. ONE CACHE KEY PER SCOREBOARD ---------------------------- @@ -55,7 +75,7 @@ import re import threading import time from concurrent.futures import ThreadPoolExecutor -from datetime import date, datetime, timedelta +from datetime import date, datetime, timedelta, tzinfo from functools import partial from typing import Any, Callable, Dict, Iterable, List, Optional, Tuple, cast @@ -100,16 +120,54 @@ RANGE_RETRY_SECONDS = 6 * 60 * 60 # pool_maxsize of 10 so the shared Session never has to discard connections. ESPN_CHUNK_WORKERS = 6 +#: An edge month the window covers at least this many days of is asked for +#: whole and trimmed, instead of one request per day (see module docstring). +#: Below it the days are cheaper than the month: a whole month is two to +#: three times the bytes of the half of it a fortnight window holds. +ESPN_MONTH_COVER_MIN_DAYS = 7 + +# Every chunk request in the process holds one of these while it is in flight +# -- the cap is per process, not per window (see module docstring). +_chunk_slots = threading.BoundedSemaphore(ESPN_CHUNK_WORKERS) + + +def _eastern_zone() -> Optional[tzinfo]: + """US Eastern, the zone ESPN's ``dates=YYYYMMDD`` means, or None when + this Python has no time zone data (no edge month is trimmed then).""" + try: + from zoneinfo import ZoneInfo + return ZoneInfo("America/New_York") + except Exception: # noqa: BLE001 - no zoneinfo module or no tz database + pass + try: + import pytz + return cast(tzinfo, pytz.timezone("America/New_York")) + except Exception: # noqa: BLE001 + return None + + +_EASTERN = _eastern_zone() + +# What _fetch_one_chunk returns for a month that came back at the cap. +_CAPPED: Any = object() + _range_lock = threading.Lock() -_ranges_rejected_until = 0.0 +# A process starts out assuming ranges are still rejected, as they have been +# since 2026-09-15, and tries one again RANGE_RETRY_SECONDS in. Starting +# from "unknown" cost one doomed range request per window at every start -- +# eleven 400s at once from a soccer board, each fetching before any had +# answered -- to learn what every start learns. +_ranges_rejected_until = time.monotonic() + RANGE_RETRY_SECONDS __all__ = [ "ESPN_MAX_LIMIT", "ESPN_CHUNK_WORKERS", + "ESPN_MONTH_COVER_MIN_DAYS", "RANGE_RETRY_SECONDS", "clamp_espn_limit", "parse_espn_date_range", "espn_date_chunks", + "espn_request_chunks", "merge_scoreboard_payloads", "fetch_espn_date_chunks", "fetch_espn_scoreboard", @@ -220,6 +278,79 @@ def espn_date_chunks(start: date, end: date) -> List[str]: return chunks +def espn_request_chunks( + start: date, + end: date, + month_cover_min_days: Optional[int] = None, +) -> List[Tuple[str, Optional[Tuple[date, date]]]]: + """The requests that fetch ``[start, end]``, as ``(dates, trim)`` pairs. + + :func:`espn_date_chunks`, except that a partial edge month with + ``month_cover_min_days`` (default ``ESPN_MONTH_COVER_MIN_DAYS``) or more + of its days in the window becomes one ``YYYYMM`` request whose ``trim`` + is the first and last of those days: its events that start outside them + (US Eastern) are dropped. ``trim`` is None for every other request. + Without time zone data nothing can be trimmed, so the edge days stay day + requests. + """ + if month_cover_min_days is None: + month_cover_min_days = ESPN_MONTH_COVER_MIN_DAYS + planned: List[Tuple[str, Optional[Tuple[date, date]]]] = [] + run: List[str] = [] + + def flush() -> None: + if (_EASTERN is not None and month_cover_min_days > 0 + and len(run) >= month_cover_min_days): + planned.append((run[0][:6], (_parse_day(run[0]), _parse_day(run[-1])))) + else: + planned.extend((day, None) for day in run) + run.clear() + + for chunk in espn_date_chunks(start, end): + if run and (len(chunk) != 8 or chunk[:6] != run[0][:6]): + flush() + if len(chunk) == 8: + run.append(chunk) + else: + planned.append((chunk, None)) + flush() + return planned + + +def _parse_day(text: str) -> date: + return date(int(text[:4]), int(text[4:6]), int(text[6:8])) + + +def _eastern_day(stamp: Any) -> Optional[date]: + """The US Eastern date of an ESPN event ``date`` ("2026-10-10T11:30Z"), + or None when it cannot be read.""" + if not isinstance(stamp, str) or _EASTERN is None: + return None + try: + moment = datetime.fromisoformat(stamp.strip().replace("Z", "+00:00")) + except ValueError: + return None + if moment.tzinfo is None: + return None + return moment.astimezone(_EASTERN).date() + + +def _trim_to_days(payload: Any, first: date, last: date) -> Any: + """Drop the events of a month payload that start outside ``[first, last]`` + (US Eastern). An event whose date cannot be read is kept: its day query + might well have returned it, and a game is never dropped on a guess. + """ + if not isinstance(payload, dict) or not isinstance(payload.get("events"), list): + return payload + kept = [] + for event in payload["events"]: + day = _eastern_day(event.get("date")) if isinstance(event, dict) else None + if day is None or first <= day <= last: + kept.append(event) + payload["events"] = kept + return payload + + def merge_scoreboard_payloads(payloads: List[Any]) -> Dict[str, Any]: """Fold chunk responses into one scoreboard payload. @@ -250,37 +381,54 @@ def merge_scoreboard_payloads(payloads: List[Any]) -> Dict[str, Any]: def _fetch_one_chunk( session, url: str, params: Dict[str, Any], headers, timeout, logger, chunk: str, cache_max_age: Optional[float] = None, -) -> Optional[Dict[str, Any]]: + trims: Optional[Dict[str, Tuple[date, date]]] = None, +) -> Any: """GET a single ``dates=`` chunk, or None when it failed. One bad chunk must not sink the rest of the season, so every error is logged and swallowed here rather than raised to the gather below. + + A month that comes back at the cap is truncated: it returns ``_CAPPED``, + its payload dropped here before it is ever held beside the others. A + month in ``trims`` loses its events outside the days given there. + + The request holds one of the process-wide ``_chunk_slots`` while it runs. """ try: - response = fetch_get( - session, - url, - params=dict(params, dates=chunk, limit=ESPN_MAX_LIMIT), - headers=headers, - timeout=timeout, - **_memo_kwargs(cache_max_age), - ) - response.raise_for_status() - return cast(Optional[Dict[str, Any]], response_json(response)) + with _chunk_slots: + response = fetch_get( + session, + url, + params=dict(params, dates=chunk, limit=ESPN_MAX_LIMIT), + headers=headers, + timeout=timeout, + **_memo_kwargs(cache_max_age), + ) + response.raise_for_status() + payload = response_json(response) except Exception as exc: # noqa: BLE001 - see docstring if logger: logger.warning("ESPN chunk %s failed, skipping it: %s", chunk, exc) return None + if len(chunk) == 6 and isinstance(payload, dict): + if len(payload.get("events") or []) >= ESPN_MAX_LIMIT: + return _CAPPED + trim = (trims or {}).get(chunk) + if trim is not None: + payload = _trim_to_days(payload, *trim) + return payload def _fetch_chunks( session, url: str, params: Dict[str, Any], headers, timeout, logger, chunks: List[str], cache_max_age: Optional[float] = None, -) -> List[Optional[Dict[str, Any]]]: + trims: Optional[Dict[str, Tuple[date, date]]] = None, +) -> List[Any]: """Fetch every chunk, returning payloads positionally aligned with ``chunks``. Requests go out ``ESPN_CHUNK_WORKERS`` at a time because a cold season is - over a hundred of them. The order they come back in is not significant -- + over a hundred of them -- and no more than that across every window the + process is fetching, which ``_fetch_one_chunk``'s slot enforces. The order they come back in is not significant -- callers keep ``chunks`` order from the returned list -- but it does mean the session is shared across threads, which is why this only ever issues GETs and never touches session state. @@ -293,7 +441,7 @@ def _fetch_chunks( return [] fetch = partial( _fetch_one_chunk, session, url, params, headers, timeout, logger, - cache_max_age=cache_max_age, + cache_max_age=cache_max_age, trims=trims, ) if len(chunks) == 1: return [fetch(chunks[0])] @@ -340,7 +488,9 @@ def fetch_espn_date_chunks( if span is None: return None - chunks = espn_date_chunks(*span) + planned = espn_request_chunks(*span) + chunks = [chunk for chunk, _ in planned] + trims = {chunk: trim for chunk, trim in planned if trim is not None} if logger: logger.debug( "Fetching ESPN date range %s as %d month/day chunks", @@ -349,32 +499,31 @@ def fetch_espn_date_chunks( results = _fetch_chunks( session, url, params, headers, timeout, logger, chunks, cache_max_age, + trims, ) attempted = len(chunks) - # A month that came back at the cap is truncated; its days replace it in - # place, so merged events stay in chunk order however the requests raced. + # A month that came back at the cap is truncated; its days (only the + # window's, for a trimmed edge month) replace it in place, so merged + # events stay in chunk order however the requests raced. Its payload was + # already dropped in the worker: a capped college-baseball month is ~2MB + # of parsed JSON, and holding four of them through ~120 day requests added + # ~25MB to the peak -- more than the concurrency itself. Low-memory boards + # (docs/LOW_MEMORY_BOARDS.md) have under 200MB of headroom. slots: List[Any] = results capped: Dict[int, List[str]] = {} for index, chunk in enumerate(chunks): - payload = slots[index] - if payload is None or len(chunk) != 6: + if slots[index] is not _CAPPED: continue - events = payload.get("events") if isinstance(payload, dict) else None - if len(events or []) >= ESPN_MAX_LIMIT: - if logger: - logger.info( - "ESPN month %s hit the %d-event cap; re-asking it day by day", - chunk, ESPN_MAX_LIMIT, - ) - capped[index] = _days_of_month(chunk) - # Drop the truncated month now rather than after its days arrive: - # a capped college-baseball month is ~2MB of parsed JSON, and - # holding four of them through ~120 day requests added ~25MB to - # the peak -- more than the concurrency itself. Low-memory boards - # (docs/LOW_MEMORY_BOARDS.md) have under 200MB of headroom. - slots[index] = None - payload = events = None + if logger: + logger.info( + "ESPN month %s hit the %d-event cap; re-asking it day by day", + chunk, ESPN_MAX_LIMIT, + ) + trim = trims.get(chunk) + capped[index] = (_days_of_month(chunk) if trim is None + else espn_date_chunks(*trim)) + slots[index] = None if capped: days = [day for index in sorted(capped) for day in capped[index]] diff --git a/test/test_espn_dates.py b/test/test_espn_dates.py index 33fa0c5e..253bf95e 100644 --- a/test/test_espn_dates.py +++ b/test/test_espn_dates.py @@ -489,3 +489,171 @@ class TestConcurrency: assert live["peak"] <= espn_dates.ESPN_CHUNK_WORKERS assert live["peak"] > 1, "chunks should actually overlap" + + +class TestEdgeMonths: + """A window's partial edge months are asked whole and trimmed. + + The default scoreboard window -- a fortnight either side of today -- spans + two partial months, so it used to cost 29 day requests per league. ESPN's + ``dates=YYYYMMDD`` means a US Eastern day (verified against the live API + on 2026-10-03, 417 of 417 soccer events), so a month answer trimmed to + the window's Eastern days is what the day requests returned. + """ + + def test_a_fortnight_either_side_is_two_requests(self): + planned = espn_dates.espn_request_chunks(date(2026, 9, 20), date(2026, 10, 18)) + assert planned == [ + ("202609", (date(2026, 9, 20), date(2026, 9, 30))), + ("202610", (date(2026, 10, 1), date(2026, 10, 18))), + ] + + def test_a_live_polls_two_days_stay_two_days(self): + planned = espn_dates.espn_request_chunks(date(2026, 10, 2), date(2026, 10, 3)) + assert planned == [("20261002", None), ("20261003", None)] + + def test_the_threshold_is_inclusive(self): + n = espn_dates.ESPN_MONTH_COVER_MIN_DAYS + short = espn_dates.espn_request_chunks(date(2026, 10, 1), date(2026, 10, n - 1)) + assert [chunk for chunk, _ in short] == [ + "202610%02d" % day for day in range(1, n)] + enough = espn_dates.espn_request_chunks(date(2026, 10, 1), date(2026, 10, n)) + assert enough == [("202610", (date(2026, 10, 1), date(2026, 10, n)))] + + def test_whole_months_and_short_edges_are_unchanged(self): + planned = espn_dates.espn_request_chunks(date(2026, 8, 30), date(2026, 10, 2)) + assert planned == [ + ("20260830", None), ("20260831", None), ("202609", None), + ("20261001", None), ("20261002", None), + ] + + def test_without_time_zone_data_edges_stay_days(self, monkeypatch): + monkeypatch.setattr(espn_dates, "_EASTERN", None) + planned = espn_dates.espn_request_chunks(date(2026, 9, 20), date(2026, 10, 18)) + assert len(planned) == 29 + assert all(trim is None for _, trim in planned) + + @pytest.mark.parametrize("start,end", [ + (date(2026, 9, 20), date(2026, 10, 18)), + (date(2026, 1, 25), date(2026, 3, 3)), + (date(2026, 12, 20), date(2027, 1, 9)), + (date(2026, 10, 5), date(2026, 10, 9)), + ]) + def test_the_planned_requests_still_cover_every_day_exactly_once(self, start, end): + covered = [] + for chunk, trim in espn_dates.espn_request_chunks(start, end): + if trim is None: + covered.extend(days_covered_by([chunk])) + else: + assert chunk == trim[0].strftime("%Y%m") == trim[1].strftime("%Y%m") + covered.extend(trim[0] + timedelta(days=offset) + for offset in range((trim[1] - trim[0]).days + 1)) + expected = [start + timedelta(days=offset) for offset in range((end - start).days + 1)] + assert covered == expected + + def test_a_trimmed_month_keeps_only_the_windows_eastern_days(self): + september = [ + # 03:30Z on the 20th is still the 19th in New York: outside. + {"id": "before", "date": "2026-09-20T03:30Z"}, + {"id": "first", "date": "2026-09-20T14:00Z"}, + {"id": "late", "date": "2026-09-30T23:30Z"}, + ] + october = [ + {"id": "oct1", "date": "2026-10-01T19:00Z"}, + # 03:30Z on the 19th is the evening of the 18th in New York: inside. + {"id": "last", "date": "2026-10-19T03:30Z"}, + {"id": "after", "date": "2026-10-19T14:00Z"}, + {"id": "undated"}, + ] + session = FakeSession({"202609": september, "202610": october}) + + data = fetch_espn_date_chunks(session, URL, params={"dates": "20260920-20261018"}) + + assert sorted(call["dates"] for call in session.calls) == ["202609", "202610"] + # An event with no readable date is kept, never dropped on a guess. + assert [e["id"] for e in data["events"]] == ["first", "late", "oct1", "last", "undated"] + + def test_eastern_standard_time_is_honoured_after_the_clocks_change(self): + # 2026-11-01 ends daylight saving: Eastern is UTC-5 from then on. + november = [ + {"id": "out", "date": "2026-11-15T04:30Z"}, # Nov 14, 23:30 EST + {"id": "in", "date": "2026-11-15T05:30Z"}, # Nov 15, 00:30 EST + ] + session = FakeSession({"202611": november}) + data = fetch_espn_date_chunks(session, URL, params={"dates": "20261115-20261121"}) + assert [e["id"] for e in data["events"]] == ["in"] + + def test_a_capped_edge_month_re_asks_only_the_windows_days(self): + full = [{"id": "cap%d" % i, "date": "2026-10-05T18:00Z"} for i in range(ESPN_MAX_LIMIT)] + by_chunk = {"202610": full} + by_chunk.update({"202610%02d" % day: [{"id": "o%02d" % day}] for day in range(1, 32)}) + session = FakeSession(by_chunk) + + data = fetch_espn_date_chunks(session, URL, params={"dates": "20261001-20261010"}) + + sent = [call["dates"] for call in session.calls] + assert sent[0] == "202610" + assert sorted(sent[1:]) == ["202610%02d" % day for day in range(1, 11)] + assert [e["id"] for e in data["events"]] == ["o%02d" % day for day in range(1, 11)] + + +class TestProcessWideChunkCap: + """The chunk cap holds across windows, not per window. + + A soccer board starting eight leagues fetches sixteen windows at once. + With a pool of ``ESPN_CHUNK_WORKERS`` each, ~40 requests were in flight + and every one past a session's pool opened a connection -- and a DNS + lookup. On ledpi that was ~90 NameResolutionErrors per start. + """ + + def test_concurrent_windows_share_one_budget(self): + live = {"now": 0, "peak": 0} + guard = threading.Lock() + + class CountingSession(FakeSession): + def get(self, url, params=None, headers=None, timeout=None): + with guard: + live["now"] += 1 + live["peak"] = max(live["peak"], live["now"]) + try: + time.sleep(0.01) + return super().get(url, params=params, headers=headers, timeout=timeout) + finally: + with guard: + live["now"] -= 1 + + sessions = [CountingSession() for _ in range(6)] + # Six leagues, so the fetch service cannot merge them into one, on a + # host with no token bucket: earlier tests may have spent ESPN's + # burst, and a bucket paced at 20/s would serialise these by itself. + threads = [ + threading.Thread(target=fetch_espn_date_chunks, + args=(session, "https://scores.example.test/league%d" % index), + kwargs={"params": {"dates": "20260101-20261231"}}) + for index, session in enumerate(sessions) + ] + for thread in threads: + thread.start() + for thread in threads: + thread.join(timeout=30) + + assert all(len(session.calls) == 12 for session in sessions) + assert live["peak"] <= espn_dates.ESPN_CHUNK_WORKERS + assert live["peak"] > 1, "chunks should still overlap" + + +def test_a_fresh_process_skips_the_doomed_range_request(): + """Every start used to spend one 400 per window learning that ranges are + still rejected -- eleven at once from a soccer board. A new process now + starts inside the retry period instead.""" + import subprocess + import sys + from pathlib import Path + + out = subprocess.run( + [sys.executable, "-c", + "import src.common.espn_dates as e; print(e._ranges_known_rejected())"], + cwd=str(Path(__file__).resolve().parents[1]), + capture_output=True, text=True, timeout=60, + ) + assert out.stdout.strip() == "True", out.stderr diff --git a/test/test_fetch_service.py b/test/test_fetch_service.py index e25ec56a..d2c17f4c 100644 --- a/test/test_fetch_service.py +++ b/test/test_fetch_service.py @@ -670,14 +670,14 @@ class TestCallerIdentity: assert _counters(global_service, plugin="football-scoreboard")["requests"] == 1 def test_espn_chunks_on_worker_threads_count_against_the_caller(self, global_service): - from src.common.espn_dates import espn_date_chunks, fetch_espn_date_chunks, parse_espn_date_range + from src.common.espn_dates import espn_request_chunks, fetch_espn_date_chunks, parse_espn_date_range session = FakeSession(lambda url, kw: make_response(body=b'{"events": []}', url=url)) dates = "20260801-20261015" with plugin_scope("baseball-scoreboard"): fetch_espn_date_chunks(session, "https://site.api.espn.com/s/scoreboard", params={"dates": dates}) - chunks = len(espn_date_chunks(*parse_espn_date_range(dates))) + chunks = len(espn_request_chunks(*parse_espn_date_range(dates))) assert chunks > 1 assert len(session.calls) == chunks assert _counters(global_service, plugin="baseball-scoreboard")["requests"] == chunks