Compare commits

..
Author SHA1 Message Date
ChuckBuildsandClaude Opus 5.5 00456ebe99 perf(fetch): cache ESPN scoreboard windows without the parts nothing reads
The sports scoreboards cache their Recent/Upcoming window as the raw ESPN
response, and it stays parsed in the memory tier while fresh. Measured on
hdpi, most of it is never drawn: per-team stat leaders, athlete cards
(featuredAthletes, probables), team and event links, headlines, video
highlights and geo broadcasts. None of those keys is read by core or by any
plugin in ledmatrix-plugins.

BackgroundDataService now drops them from an ESPN /scoreboard response
before caching and delivering it (src/common/espn_payload.py), keeping
everything else. On the five hdpi windows that is 10.6MB -> 3.0MB of JSON
and ~40MB -> ~12MB of parsed objects, and parsing an expired window gets
3-4x cheaper. submit_fetch_request(slim_payload=False) caches a response
whole.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BkfgXMqqwn2w4NN7LRzhxy
2026-10-03 22:25:09 -04:00
6 changed files with 281 additions and 264 deletions
+20 -1
View File
@@ -34,6 +34,7 @@ from src.common.fetch_service import (
plugin_scope,
share_connection_pool,
)
from src.common.espn_payload import is_espn_scoreboard_url, slim_scoreboard_payload
from src.common.espn_dates import (
RANGE_RETRY_SECONDS,
_note_range_rejected,
@@ -83,6 +84,10 @@ class FetchRequest:
# the cache with the callbacks suppressed -- joiners waiting forever for a
# fetch that did, in fact, succeed.
commit_claimed: bool = False
# Trim an ESPN scoreboard response before it is cached and delivered
# (src/common/espn_payload.py). Set by whoever created the request; a
# submitter that joins the fetch gets the same payload.
slim_payload: bool = True
result: Optional[Any] = None
error: Optional[str] = None
# The plugin that submitted the request, so the fetch service counts the
@@ -249,7 +254,8 @@ class BackgroundDataService:
timeout: Optional[int] = None,
max_retries: int = 3,
priority: int = 1,
callback: Optional[Callable] = None) -> str:
callback: Optional[Callable] = None,
slim_payload: bool = True) -> str:
"""
Submit a background fetch request.
@@ -265,6 +271,11 @@ class BackgroundDataService:
priority: Accepted for compatibility and ignored; requests run in
submission order.
callback: Optional callback function when request completes
slim_payload: Drop the parts of an ESPN scoreboard response no
scoreboard reads (stat leaders, athlete cards, links,
headlines, highlights) before caching it; see
src/common/espn_payload.py. Only ESPN /scoreboard URLs are
touched. Pass False to cache the response whole.
Returns:
Request ID for tracking the fetch operation
@@ -336,6 +347,7 @@ class BackgroundDataService:
priority=priority,
callback=callback,
owner=owner,
slim_payload=slim_payload,
)
with self._lock:
@@ -497,6 +509,13 @@ class BackgroundDataService:
)
return result
# Most of an ESPN scoreboard response is never drawn, and the
# cached copy stays parsed in the memory tier while it is fresh.
# Trimmed before the write so the cache, request.result and the
# callbacks all see the same payload. See src/common/espn_payload.py.
if request.slim_payload and is_espn_scoreboard_url(request.url):
slim_scoreboard_payload(data)
# Cache the data
self.cache_manager.set(request.cache_key, data)
+2 -145
View File
@@ -46,39 +46,19 @@ entry older than the reader's own ``max_age``, whoever wrote it and whatever
ttl they stored with it. Old keys are passed as ``legacy_keys`` and read
after the canonical one, so an upgrade does not refetch everything at once;
they can go one release after the one that added this.
Chunks whose days are long over are kept in memory between fetches. The
scoreboards re-fetch their whole Recent/Upcoming window (14 days back, 7
ahead) every hour, and since ranges went away that is 22 day requests per
league. Measured on hdpi on 2026-10-02 (NFL, college football, MLB, college
baseball, NHL): the hourly window refresh was ~270 of 321 ESPN requests and
~21 of 24.6MB in the hour, and the 12 days that ended three or more days ago
were 68% of those bytes (6.9 of 10.2MB per copy of the five windows). A
settled chunk is answered from memory for ``SETTLED_CHUNK_TTL_SECONDS``,
stored as zlib-compressed JSON (~13x smaller than the body, and far smaller
than the parsed objects), so the hourly refresh only goes to ESPN for the
days that can still change.
"""
import contextvars
import json
import logging
import math
import re
import threading
import time
import zlib
from collections import OrderedDict
from concurrent.futures import ThreadPoolExecutor
from datetime import date, datetime, timedelta, timezone
from datetime import date, datetime, timedelta
from functools import partial
from typing import Any, Callable, Dict, Iterable, List, Optional, Tuple, cast
try:
import orjson
except ImportError: # optional; the stdlib parser gives the same objects
orjson = None
try:
from src.common.json_body import response_json
except ImportError:
@@ -123,37 +103,11 @@ ESPN_CHUNK_WORKERS = 6
_range_lock = threading.Lock()
_ranges_rejected_until = 0.0
# A chunk is "settled" once its last day is this many UTC days back. ESPN
# files games under the US Eastern date, and a late West-coast game ends after
# midnight UTC; three days leaves a full day of margin past both, so nothing
# still being played, finalised or rescheduled is ever served from memory.
SETTLED_AFTER_DAYS = 3
# How long a settled chunk is trusted. A day's finals do not change, but a
# rare correction (or an empty answer during an ESPN outage) should not live
# forever: once a day is plenty, and still skips 23 of every 24 hourly asks.
SETTLED_CHUNK_TTL_SECONDS = 24 * 60 * 60
# Bounds on the settled-chunk memory. A settled day measured 90KB (NHL) to
# 990KB (a college-football Saturday) of JSON and 9-74KB compressed; the five
# windows on hdpi need 60 entries and ~0.55MB. The caps only matter for a
# board fetching whole past seasons.
SETTLED_CACHE_MAX_ENTRIES = 512
SETTLED_CACHE_MAX_BYTES = 8 * 1024 * 1024
_settled_lock = threading.Lock()
# key -> (stored_at monotonic, compressed JSON)
_settled_chunks: "OrderedDict[Any, Tuple[float, bytes]]" = OrderedDict()
_settled_bytes = 0
__all__ = [
"ESPN_MAX_LIMIT",
"ESPN_CHUNK_WORKERS",
"RANGE_RETRY_SECONDS",
"SETTLED_AFTER_DAYS",
"SETTLED_CHUNK_TTL_SECONDS",
"clamp_espn_limit",
"clear_settled_chunk_cache",
"parse_espn_date_range",
"espn_date_chunks",
"merge_scoreboard_payloads",
@@ -245,88 +199,6 @@ def _days_of_month(chunk: str) -> List[str]:
return days
def _utc_today() -> date:
return datetime.now(timezone.utc).date()
def _chunk_last_day(chunk: str) -> Optional[date]:
try:
if len(chunk) == 8:
return date(int(chunk[:4]), int(chunk[4:6]), int(chunk[6:]))
if len(chunk) == 6:
first = date(int(chunk[:4]), int(chunk[4:6]), 1)
return _first_of_next_month(first) - timedelta(days=1)
except ValueError:
pass
return None
def _settled_key(url: str, params: Dict[str, Any], chunk: str) -> Optional[Any]:
"""Memory key for a chunk that can no longer change, else None."""
last_day = _chunk_last_day(chunk)
if last_day is None:
return None
if last_day > _utc_today() - timedelta(days=SETTLED_AFTER_DAYS):
return None
# dates is the chunk itself and limit is always ESPN_MAX_LIMIT here;
# anything else (groups=80 for FBS, a team filter) changes the answer.
rest = tuple(sorted(
(str(k), str(v)) for k, v in params.items() if k not in ("dates", "limit")
))
return (url, rest, chunk)
def _settled_get(key: Any) -> Optional[Dict[str, Any]]:
with _settled_lock:
entry = _settled_chunks.get(key)
if entry is None:
return None
if time.monotonic() - entry[0] > SETTLED_CHUNK_TTL_SECONDS:
_settled_drop(key)
return None
_settled_chunks.move_to_end(key)
blob = entry[1]
# Decompress and parse outside the lock: every hit gets its own objects,
# so a caller mutating its payload cannot reach another caller's.
body = zlib.decompress(blob)
return cast(Dict[str, Any], orjson.loads(body) if orjson else json.loads(body))
def _settled_drop(key: Any) -> None:
"""Remove one entry. Caller holds _settled_lock."""
global _settled_bytes
entry = _settled_chunks.pop(key, None)
if entry is not None:
_settled_bytes -= len(entry[1])
def _settled_put(key: Any, response: Any, payload: Dict[str, Any]) -> None:
global _settled_bytes
body = getattr(response, "content", None)
if not isinstance(body, (bytes, bytearray)):
body = json.dumps(payload).encode("utf-8")
blob = zlib.compress(bytes(body), 6)
if len(blob) > SETTLED_CACHE_MAX_BYTES:
return
with _settled_lock:
_settled_drop(key)
_settled_chunks[key] = (time.monotonic(), blob)
_settled_bytes += len(blob)
while _settled_chunks and (
len(_settled_chunks) > SETTLED_CACHE_MAX_ENTRIES
or _settled_bytes > SETTLED_CACHE_MAX_BYTES
):
_settled_drop(next(iter(_settled_chunks)))
def clear_settled_chunk_cache() -> None:
"""Forget every remembered settled chunk (tests, or a manual refresh)."""
global _settled_bytes
with _settled_lock:
_settled_chunks.clear()
_settled_bytes = 0
def espn_date_chunks(start: date, end: date) -> List[str]:
"""Cover ``[start, end]`` inclusive with ``dates=`` values ESPN accepts.
@@ -383,16 +255,8 @@ def _fetch_one_chunk(
One bad chunk must not sink the rest of the season, so every error is
logged and swallowed here rather than raised to the gather below.
A chunk whose days are settled (see ``SETTLED_AFTER_DAYS``) is answered
from memory when it was fetched in the last day.
"""
try:
settled = _settled_key(url, params, chunk)
if settled is not None:
cached = _settled_get(settled)
if cached is not None:
return cached
response = fetch_get(
session,
url,
@@ -402,14 +266,7 @@ def _fetch_one_chunk(
**_memo_kwargs(cache_max_age),
)
response.raise_for_status()
payload = response_json(response)
if settled is not None and isinstance(payload, dict):
events = payload.get("events")
# A capped month is truncated and gets re-asked day by day;
# remembering it would only cost memory.
if isinstance(events, list) and len(events) < ESPN_MAX_LIMIT:
_settled_put(settled, response, payload)
return cast(Optional[Dict[str, Any]], payload)
return cast(Optional[Dict[str, Any]], response_json(response))
except Exception as exc: # noqa: BLE001 - see docstring
if logger:
logger.warning("ESPN chunk %s failed, skipping it: %s", chunk, exc)
+97
View File
@@ -0,0 +1,97 @@
"""Drop the parts of an ESPN scoreboard payload no scoreboard reads.
The sports scoreboards cache their Recent/Upcoming window (14 days back, 7
ahead) as the raw ESPN response, and that record stays parsed in the memory
cache for as long as it is fresh. Most of it is never drawn. Measured on hdpi
(2026-10-02) the MLB window was 3.35MB of JSON and 13.5MB of Python objects,
and the five windows together ~40MB, mostly in:
* ``competitors[].leaders`` / ``competitions[].leaders`` -- per-team and
per-game stat leaders (28% of the MLB window)
* ``competitors[].team.links`` / ``event.links`` -- web and app URLs
* ``status.featuredAthletes`` and ``competitors[].probables`` -- athlete
cards with headshots and season stats
* ``competitions[].headlines`` / ``highlights`` -- article and video blurbs
(28% of the college-football window)
* ``competitions[].geoBroadcasts``
None of those keys is read by core or by any plugin in ledmatrix-plugins
(checked 2026-10-02 across every scoreboard, the odds ticker and the
leaderboard), while everything that is read -- odds, records, linescores,
situation, statistics, notes, broadcasts, venue -- is kept. Dropping them
takes the five windows from ~40MB to ~12MB of parsed objects and the files from
10.6MB to 3.0MB, so the reads that parse an expired window on the render
thread get 3-4x cheaper too.
:func:`slim_scoreboard_payload` changes the payload in place, and only ever
removes the keys listed here: anything it does not know about is left alone.
"""
from typing import Any, Dict
from urllib.parse import urlsplit
# Per level of the payload, the keys removed. Kept deliberately explicit:
# adding a key here means checking that nothing reads it first.
_EVENT_DROP = ("links",)
_COMPETITION_DROP = ("leaders", "headlines", "highlights", "geoBroadcasts")
_STATUS_DROP = ("featuredAthletes",)
_COMPETITOR_DROP = ("leaders", "probables")
_TEAM_DROP = ("links",)
def is_espn_scoreboard_url(url: Any) -> bool:
"""Whether ``url`` is an ESPN site-API scoreboard endpoint."""
if not isinstance(url, str):
return False
try:
parts = urlsplit(url)
except ValueError:
return False
host = (parts.hostname or "").lower()
if host != "espn.com" and not host.endswith(".espn.com"):
return False
return parts.path.rstrip("/").endswith("/scoreboard")
def _drop(obj: Any, keys) -> None:
if isinstance(obj, dict):
for key in keys:
obj.pop(key, None)
def slim_scoreboard_payload(payload: Any) -> Any:
"""Remove the unread parts of an ESPN scoreboard payload, in place.
Returns ``payload`` for convenience. Anything that is not shaped like a
scoreboard (not a dict, no ``events`` list, odd entries) is passed over
untouched rather than raising.
"""
if not isinstance(payload, dict):
return payload
events = payload.get("events")
if not isinstance(events, list):
return payload
for event in events:
if not isinstance(event, dict):
continue
_drop(event, _EVENT_DROP)
competitions = event.get("competitions")
if not isinstance(competitions, list):
continue
for competition in competitions:
if not isinstance(competition, dict):
continue
_drop(competition, _COMPETITION_DROP)
_drop(competition.get("status"), _STATUS_DROP)
competitors = competition.get("competitors")
if not isinstance(competitors, list):
continue
for competitor in competitors:
if not isinstance(competitor, dict):
continue
_drop(competitor, _COMPETITOR_DROP)
_drop(competitor.get("team"), _TEAM_DROP)
return payload
__all__ = ["is_espn_scoreboard_url", "slim_scoreboard_payload"]
-18
View File
@@ -343,24 +343,6 @@ def _hermetic_unit_refresh(monkeypatch, tmp_path_factory):
monkeypatch.setattr(unit_refresh, 'SYSTEMD_DIR', str(tmp_path_factory.getbasetemp() / 'no-systemd'))
@pytest.fixture(autouse=True)
def _forget_settled_espn_chunks(monkeypatch):
"""src.common.espn_dates remembers past days process-wide; tests fake
different answers for the same dates, so none may inherit another's.
"Today" is also pinned to 2000-01-01, so no date a test uses counts as
settled unless the test says so (by pinning _utc_today itself). Without
that, a test asking for last month twice passes while that month is
recent and fails once it is three days old: the second ask is answered
from memory."""
from datetime import date
from src.common import espn_dates
monkeypatch.setattr(espn_dates, "_utc_today", lambda: date(2000, 1, 1))
espn_dates.clear_settled_chunk_cache()
yield
espn_dates.clear_settled_chunk_cache()
@pytest.fixture(autouse=True)
def reset_logging():
"""Reset logging configuration before each test."""
-100
View File
@@ -39,14 +39,6 @@ def forget_rejected_ranges(monkeypatch):
monkeypatch.setattr(espn_dates, "_ranges_rejected_until", 0.0)
@pytest.fixture(autouse=True)
def nothing_is_settled_yet(monkeypatch):
"""Pin "today" before every date these tests use, so the settled-chunk
memory stays out of tests that are not about it whatever the real date.
TestSettledChunkCache moves it forward."""
monkeypatch.setattr(espn_dates, "_utc_today", lambda: date(2000, 1, 1))
class FakeResponse:
def __init__(self, status_code=200, payload=None):
self.status_code = status_code
@@ -497,95 +489,3 @@ class TestConcurrency:
assert live["peak"] <= espn_dates.ESPN_CHUNK_WORKERS
assert live["peak"] > 1, "chunks should actually overlap"
class TestSettledChunkCache:
"""Days that ended three or more days ago are fetched once a day, not hourly.
The scoreboards re-fetch a 22-day window every hour; on hdpi (2026-10-02)
the 12 settled days were 68% of that window's bytes.
"""
TODAY = date(2026, 10, 2)
# The scoreboards' default window on that day: 14 back, 7 ahead.
WINDOW = "20260918-20261009"
@pytest.fixture(autouse=True)
def frozen_today(self, monkeypatch):
monkeypatch.setattr(espn_dates, "_utc_today", lambda: self.TODAY)
def _events(self):
days = [(9, d) for d in range(18, 31)] + [(10, d) for d in range(1, 10)]
return {"2026%02d%02d" % (m, d): [{"id": f"{m}-{d}"}] for m, d in days}
def test_the_second_refresh_only_asks_for_unsettled_days(self):
session = FakeSession(self._events())
first = fetch_espn_scoreboard(session, URL, params={"dates": self.WINDOW})
session.calls.clear()
second = fetch_espn_scoreboard(session, URL, params={"dates": self.WINDOW})
asked = sorted(call["dates"] for call in session.calls)
# Sep 29 is the last settled day (today minus three).
assert asked == ["20260930"] + ["202610%02d" % d for d in range(1, 10)]
assert second["events"] == first["events"] # same events, same order
assert len(second["events"]) == 22
def test_a_hit_is_a_fresh_copy(self):
session = FakeSession(self._events())
fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
hit = fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
hit["events"][0]["id"] = "mutated"
again = fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
assert again["events"][0]["id"] == "9-18"
def test_other_params_are_part_of_the_key(self):
session = FakeSession(self._events())
fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW, "groups": 80})
session.calls.clear()
fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
assert len(session.calls) == 22 # a different question, nothing reused
def test_entries_expire_after_a_day(self, monkeypatch):
clock = [1000.0]
monkeypatch.setattr(espn_dates.time, "monotonic", lambda: clock[0])
session = FakeSession(self._events())
fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
clock[0] += espn_dates.SETTLED_CHUNK_TTL_SECONDS + 1
session.calls.clear()
fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
assert len(session.calls) == 22
def test_failed_chunks_are_not_remembered(self):
session = FakeSession(self._events(), fail_chunks={"20260920"})
fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
session.fail_chunks.clear()
session.calls.clear()
data = fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
assert "20260920" in [call["dates"] for call in session.calls]
assert len(data["events"]) == 22
def test_a_capped_month_is_not_remembered_but_its_days_are(self):
full = [{"id": f"x{i}"} for i in range(ESPN_MAX_LIMIT)]
session = FakeSession({"202608": full, "20260801": [{"id": "d1"}]})
fetch_espn_date_chunks(session, URL, params={"dates": "20260801-20260831"})
session.calls.clear()
data = fetch_espn_date_chunks(session, URL, params={"dates": "20260801-20260831"})
assert [call["dates"] for call in session.calls] == ["202608"]
assert [event["id"] for event in data["events"]] == ["d1"]
def test_memory_is_bounded(self, monkeypatch):
monkeypatch.setattr(espn_dates, "SETTLED_CACHE_MAX_ENTRIES", 5)
session = FakeSession(self._events())
fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
assert len(espn_dates._settled_chunks) == 5
assert espn_dates._settled_bytes == sum(
len(blob) for _, blob in espn_dates._settled_chunks.values())
def test_single_day_requests_are_untouched(self):
# The live path asks for today (or one day) as a plain request; that
# never goes through chunks or the memory.
session = FakeSession(self._events())
for _ in range(2):
fetch_espn_scoreboard(session, URL, params={"dates": "20260918"})
assert len(session.calls) == 2
+162
View File
@@ -0,0 +1,162 @@
"""Tests for src/common/espn_payload.py and its use by BackgroundDataService."""
import copy
import time
from unittest.mock import MagicMock, Mock, patch
import pytest
from src.background_data_service import BackgroundDataService, shutdown_background_service
from src.common.espn_payload import is_espn_scoreboard_url, slim_scoreboard_payload
SCOREBOARD = "https://site.api.espn.com/apis/site/v2/sports/baseball/mlb/scoreboard"
def _event():
"""One event carrying every key the slimming drops and a sample of the
keys scoreboards read, at the depth ESPN puts them."""
competitor = {
"id": "10",
"homeAway": "home",
"score": "5",
"team": {"abbreviation": "NYY", "logo": "https://a/l.png",
"links": [{"href": "https://espn.com/team"}]},
"records": [{"summary": "90-60"}],
"linescores": [{"value": 1}],
"statistics": [{"name": "hits", "displayValue": "9"}],
"leaders": [{"name": "avg", "leaders": [{"athlete": {"id": "1"}}]}],
"probables": [{"athlete": {"id": "2"}, "statistics": []}],
}
return {
"id": "401",
"date": "2026-10-01T23:05Z",
"links": [{"href": "https://espn.com/game"}],
"status": {"type": {"state": "post"}},
"competitions": [{
"status": {"type": {"state": "post", "shortDetail": "Final"},
"featuredAthletes": [{"athlete": {"id": "3"}}]},
"competitors": [competitor, dict(copy.deepcopy(competitor), homeAway="away")],
"odds": [{"details": "NYY -150", "overUnder": 8.5}],
"situation": {"outs": 2},
"notes": [{"headline": "Game 1"}],
"broadcasts": [{"names": ["FOX"]}],
"venue": {"fullName": "Yankee Stadium"},
"leaders": [{"name": "hits"}],
"headlines": [{"description": "recap"}],
"highlights": [{"links": {"source": {}}}],
"geoBroadcasts": [{"media": {"shortName": "FOX"}}],
}],
}
class TestSlimScoreboardPayload:
def test_drops_exactly_the_listed_keys(self):
payload = {"leagues": [{"id": "10"}], "events": [_event()]}
slim_scoreboard_payload(payload)
event = payload["events"][0]
competition = event["competitions"][0]
assert "links" not in event
for key in ("leaders", "headlines", "highlights", "geoBroadcasts"):
assert key not in competition
assert "featuredAthletes" not in competition["status"]
for competitor in competition["competitors"]:
assert "leaders" not in competitor
assert "probables" not in competitor
assert "links" not in competitor["team"]
def test_keeps_everything_else_unchanged(self):
"""Removing the dropped keys from the original by hand gives exactly
the slimmed payload: nothing else moved, changed or went missing."""
original = {"leagues": [{"id": "10"}], "events": [_event(), _event()]}
expected = copy.deepcopy(original)
for event in expected["events"]:
del event["links"]
competition = event["competitions"][0]
for key in ("leaders", "headlines", "highlights", "geoBroadcasts"):
del competition[key]
del competition["status"]["featuredAthletes"]
for competitor in competition["competitors"]:
del competitor["leaders"], competitor["probables"]
del competitor["team"]["links"]
assert slim_scoreboard_payload(original) == expected
def test_in_place_and_returns_payload(self):
payload = {"events": [_event()]}
assert slim_scoreboard_payload(payload) is payload
@pytest.mark.parametrize("payload", [
None, [], "x", {}, {"events": None}, {"events": "x"},
{"events": [None, 1, "x", {"competitions": None}]},
{"events": [{"competitions": [None, {"status": None, "competitors": None}]}]},
{"events": [{"competitions": [{"competitors": [None, {"team": None}]}]}]},
])
def test_odd_shapes_pass_through(self, payload):
before = copy.deepcopy(payload)
assert slim_scoreboard_payload(payload) == before
class TestIsEspnScoreboardUrl:
@pytest.mark.parametrize("url", [
SCOREBOARD,
SCOREBOARD + "/",
"http://site.api.espn.com/apis/site/v2/sports/football/college-football/scoreboard",
])
def test_scoreboards(self, url):
assert is_espn_scoreboard_url(url)
@pytest.mark.parametrize("url", [
None, "", 12,
"https://site.api.espn.com/apis/site/v2/sports/baseball/mlb/teams",
"https://site.api.espn.com/apis/site/v2/sports/football/nfl/summary",
"https://example.com/scoreboard",
"https://espn.com.evil.example/apis/x/scoreboard",
"https://notespn.com/apis/x/scoreboard",
])
def test_not_scoreboards(self, url):
assert not is_espn_scoreboard_url(url)
@pytest.fixture
def service():
shutdown_background_service()
cache = MagicMock()
cache.get.return_value = None
svc = BackgroundDataService(cache, max_workers=1, request_timeout=5)
yield svc
svc.shutdown(wait=False)
shutdown_background_service()
def _run(service, url, **kwargs):
response = Mock(status_code=200)
response.json.return_value = {"events": [_event()]}
response.raise_for_status.return_value = None
delivered = []
with patch.object(service.session, "get", return_value=response):
req_id = service.submit_fetch_request(
sport="mlb", year=2026, url=url, cache_key="mlb_schedule_window_14_7",
callback=lambda result: delivered.append(result.data), **kwargs)
deadline = time.time() + 5
while not service.is_request_complete(req_id) and time.time() < deadline:
time.sleep(0.02)
cached = service.cache_manager.set.call_args[0][1]
return cached, delivered
class TestBackgroundServiceSlims:
def test_espn_scoreboard_is_cached_and_delivered_slimmed(self, service):
cached, delivered = _run(service, SCOREBOARD)
competition = cached["events"][0]["competitions"][0]
assert "leaders" not in competition
assert "probables" not in competition["competitors"][0]
assert competition["odds"] and competition["situation"]
# The callback sees the very payload that was cached.
assert delivered and delivered[0] is cached
def test_opt_out_caches_whole_response(self, service):
cached, _ = _run(service, SCOREBOARD, slim_payload=False)
assert cached == {"events": [_event()]}
def test_other_urls_untouched(self, service):
cached, _ = _run(service, "https://example.com/feed")
assert cached == {"events": [_event()]}