mirror of
https://github.com/ChuckBuilds/LEDMatrix.git
synced 2026-10-04 14:25:08 +00:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
00456ebe99 |
@@ -34,6 +34,7 @@ from src.common.fetch_service import (
|
||||
plugin_scope,
|
||||
share_connection_pool,
|
||||
)
|
||||
from src.common.espn_payload import is_espn_scoreboard_url, slim_scoreboard_payload
|
||||
from src.common.espn_dates import (
|
||||
RANGE_RETRY_SECONDS,
|
||||
_note_range_rejected,
|
||||
@@ -83,6 +84,10 @@ class FetchRequest:
|
||||
# the cache with the callbacks suppressed -- joiners waiting forever for a
|
||||
# fetch that did, in fact, succeed.
|
||||
commit_claimed: bool = False
|
||||
# Trim an ESPN scoreboard response before it is cached and delivered
|
||||
# (src/common/espn_payload.py). Set by whoever created the request; a
|
||||
# submitter that joins the fetch gets the same payload.
|
||||
slim_payload: bool = True
|
||||
result: Optional[Any] = None
|
||||
error: Optional[str] = None
|
||||
# The plugin that submitted the request, so the fetch service counts the
|
||||
@@ -249,7 +254,8 @@ class BackgroundDataService:
|
||||
timeout: Optional[int] = None,
|
||||
max_retries: int = 3,
|
||||
priority: int = 1,
|
||||
callback: Optional[Callable] = None) -> str:
|
||||
callback: Optional[Callable] = None,
|
||||
slim_payload: bool = True) -> str:
|
||||
"""
|
||||
Submit a background fetch request.
|
||||
|
||||
@@ -265,6 +271,11 @@ class BackgroundDataService:
|
||||
priority: Accepted for compatibility and ignored; requests run in
|
||||
submission order.
|
||||
callback: Optional callback function when request completes
|
||||
slim_payload: Drop the parts of an ESPN scoreboard response no
|
||||
scoreboard reads (stat leaders, athlete cards, links,
|
||||
headlines, highlights) before caching it; see
|
||||
src/common/espn_payload.py. Only ESPN /scoreboard URLs are
|
||||
touched. Pass False to cache the response whole.
|
||||
|
||||
Returns:
|
||||
Request ID for tracking the fetch operation
|
||||
@@ -336,6 +347,7 @@ class BackgroundDataService:
|
||||
priority=priority,
|
||||
callback=callback,
|
||||
owner=owner,
|
||||
slim_payload=slim_payload,
|
||||
)
|
||||
|
||||
with self._lock:
|
||||
@@ -497,6 +509,13 @@ class BackgroundDataService:
|
||||
)
|
||||
return result
|
||||
|
||||
# Most of an ESPN scoreboard response is never drawn, and the
|
||||
# cached copy stays parsed in the memory tier while it is fresh.
|
||||
# Trimmed before the write so the cache, request.result and the
|
||||
# callbacks all see the same payload. See src/common/espn_payload.py.
|
||||
if request.slim_payload and is_espn_scoreboard_url(request.url):
|
||||
slim_scoreboard_payload(data)
|
||||
|
||||
# Cache the data
|
||||
self.cache_manager.set(request.cache_key, data)
|
||||
|
||||
|
||||
+2
-145
@@ -46,39 +46,19 @@ entry older than the reader's own ``max_age``, whoever wrote it and whatever
|
||||
ttl they stored with it. Old keys are passed as ``legacy_keys`` and read
|
||||
after the canonical one, so an upgrade does not refetch everything at once;
|
||||
they can go one release after the one that added this.
|
||||
|
||||
Chunks whose days are long over are kept in memory between fetches. The
|
||||
scoreboards re-fetch their whole Recent/Upcoming window (14 days back, 7
|
||||
ahead) every hour, and since ranges went away that is 22 day requests per
|
||||
league. Measured on hdpi on 2026-10-02 (NFL, college football, MLB, college
|
||||
baseball, NHL): the hourly window refresh was ~270 of 321 ESPN requests and
|
||||
~21 of 24.6MB in the hour, and the 12 days that ended three or more days ago
|
||||
were 68% of those bytes (6.9 of 10.2MB per copy of the five windows). A
|
||||
settled chunk is answered from memory for ``SETTLED_CHUNK_TTL_SECONDS``,
|
||||
stored as zlib-compressed JSON (~13x smaller than the body, and far smaller
|
||||
than the parsed objects), so the hourly refresh only goes to ESPN for the
|
||||
days that can still change.
|
||||
"""
|
||||
|
||||
import contextvars
|
||||
import json
|
||||
import logging
|
||||
import math
|
||||
import re
|
||||
import threading
|
||||
import time
|
||||
import zlib
|
||||
from collections import OrderedDict
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from datetime import date, datetime, timedelta, timezone
|
||||
from datetime import date, datetime, timedelta
|
||||
from functools import partial
|
||||
from typing import Any, Callable, Dict, Iterable, List, Optional, Tuple, cast
|
||||
|
||||
try:
|
||||
import orjson
|
||||
except ImportError: # optional; the stdlib parser gives the same objects
|
||||
orjson = None
|
||||
|
||||
try:
|
||||
from src.common.json_body import response_json
|
||||
except ImportError:
|
||||
@@ -123,37 +103,11 @@ ESPN_CHUNK_WORKERS = 6
|
||||
_range_lock = threading.Lock()
|
||||
_ranges_rejected_until = 0.0
|
||||
|
||||
# A chunk is "settled" once its last day is this many UTC days back. ESPN
|
||||
# files games under the US Eastern date, and a late West-coast game ends after
|
||||
# midnight UTC; three days leaves a full day of margin past both, so nothing
|
||||
# still being played, finalised or rescheduled is ever served from memory.
|
||||
SETTLED_AFTER_DAYS = 3
|
||||
|
||||
# How long a settled chunk is trusted. A day's finals do not change, but a
|
||||
# rare correction (or an empty answer during an ESPN outage) should not live
|
||||
# forever: once a day is plenty, and still skips 23 of every 24 hourly asks.
|
||||
SETTLED_CHUNK_TTL_SECONDS = 24 * 60 * 60
|
||||
|
||||
# Bounds on the settled-chunk memory. A settled day measured 90KB (NHL) to
|
||||
# 990KB (a college-football Saturday) of JSON and 9-74KB compressed; the five
|
||||
# windows on hdpi need 60 entries and ~0.55MB. The caps only matter for a
|
||||
# board fetching whole past seasons.
|
||||
SETTLED_CACHE_MAX_ENTRIES = 512
|
||||
SETTLED_CACHE_MAX_BYTES = 8 * 1024 * 1024
|
||||
|
||||
_settled_lock = threading.Lock()
|
||||
# key -> (stored_at monotonic, compressed JSON)
|
||||
_settled_chunks: "OrderedDict[Any, Tuple[float, bytes]]" = OrderedDict()
|
||||
_settled_bytes = 0
|
||||
|
||||
__all__ = [
|
||||
"ESPN_MAX_LIMIT",
|
||||
"ESPN_CHUNK_WORKERS",
|
||||
"RANGE_RETRY_SECONDS",
|
||||
"SETTLED_AFTER_DAYS",
|
||||
"SETTLED_CHUNK_TTL_SECONDS",
|
||||
"clamp_espn_limit",
|
||||
"clear_settled_chunk_cache",
|
||||
"parse_espn_date_range",
|
||||
"espn_date_chunks",
|
||||
"merge_scoreboard_payloads",
|
||||
@@ -245,88 +199,6 @@ def _days_of_month(chunk: str) -> List[str]:
|
||||
return days
|
||||
|
||||
|
||||
def _utc_today() -> date:
|
||||
return datetime.now(timezone.utc).date()
|
||||
|
||||
|
||||
def _chunk_last_day(chunk: str) -> Optional[date]:
|
||||
try:
|
||||
if len(chunk) == 8:
|
||||
return date(int(chunk[:4]), int(chunk[4:6]), int(chunk[6:]))
|
||||
if len(chunk) == 6:
|
||||
first = date(int(chunk[:4]), int(chunk[4:6]), 1)
|
||||
return _first_of_next_month(first) - timedelta(days=1)
|
||||
except ValueError:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def _settled_key(url: str, params: Dict[str, Any], chunk: str) -> Optional[Any]:
|
||||
"""Memory key for a chunk that can no longer change, else None."""
|
||||
last_day = _chunk_last_day(chunk)
|
||||
if last_day is None:
|
||||
return None
|
||||
if last_day > _utc_today() - timedelta(days=SETTLED_AFTER_DAYS):
|
||||
return None
|
||||
# dates is the chunk itself and limit is always ESPN_MAX_LIMIT here;
|
||||
# anything else (groups=80 for FBS, a team filter) changes the answer.
|
||||
rest = tuple(sorted(
|
||||
(str(k), str(v)) for k, v in params.items() if k not in ("dates", "limit")
|
||||
))
|
||||
return (url, rest, chunk)
|
||||
|
||||
|
||||
def _settled_get(key: Any) -> Optional[Dict[str, Any]]:
|
||||
with _settled_lock:
|
||||
entry = _settled_chunks.get(key)
|
||||
if entry is None:
|
||||
return None
|
||||
if time.monotonic() - entry[0] > SETTLED_CHUNK_TTL_SECONDS:
|
||||
_settled_drop(key)
|
||||
return None
|
||||
_settled_chunks.move_to_end(key)
|
||||
blob = entry[1]
|
||||
# Decompress and parse outside the lock: every hit gets its own objects,
|
||||
# so a caller mutating its payload cannot reach another caller's.
|
||||
body = zlib.decompress(blob)
|
||||
return cast(Dict[str, Any], orjson.loads(body) if orjson else json.loads(body))
|
||||
|
||||
|
||||
def _settled_drop(key: Any) -> None:
|
||||
"""Remove one entry. Caller holds _settled_lock."""
|
||||
global _settled_bytes
|
||||
entry = _settled_chunks.pop(key, None)
|
||||
if entry is not None:
|
||||
_settled_bytes -= len(entry[1])
|
||||
|
||||
|
||||
def _settled_put(key: Any, response: Any, payload: Dict[str, Any]) -> None:
|
||||
global _settled_bytes
|
||||
body = getattr(response, "content", None)
|
||||
if not isinstance(body, (bytes, bytearray)):
|
||||
body = json.dumps(payload).encode("utf-8")
|
||||
blob = zlib.compress(bytes(body), 6)
|
||||
if len(blob) > SETTLED_CACHE_MAX_BYTES:
|
||||
return
|
||||
with _settled_lock:
|
||||
_settled_drop(key)
|
||||
_settled_chunks[key] = (time.monotonic(), blob)
|
||||
_settled_bytes += len(blob)
|
||||
while _settled_chunks and (
|
||||
len(_settled_chunks) > SETTLED_CACHE_MAX_ENTRIES
|
||||
or _settled_bytes > SETTLED_CACHE_MAX_BYTES
|
||||
):
|
||||
_settled_drop(next(iter(_settled_chunks)))
|
||||
|
||||
|
||||
def clear_settled_chunk_cache() -> None:
|
||||
"""Forget every remembered settled chunk (tests, or a manual refresh)."""
|
||||
global _settled_bytes
|
||||
with _settled_lock:
|
||||
_settled_chunks.clear()
|
||||
_settled_bytes = 0
|
||||
|
||||
|
||||
def espn_date_chunks(start: date, end: date) -> List[str]:
|
||||
"""Cover ``[start, end]`` inclusive with ``dates=`` values ESPN accepts.
|
||||
|
||||
@@ -383,16 +255,8 @@ def _fetch_one_chunk(
|
||||
|
||||
One bad chunk must not sink the rest of the season, so every error is
|
||||
logged and swallowed here rather than raised to the gather below.
|
||||
|
||||
A chunk whose days are settled (see ``SETTLED_AFTER_DAYS``) is answered
|
||||
from memory when it was fetched in the last day.
|
||||
"""
|
||||
try:
|
||||
settled = _settled_key(url, params, chunk)
|
||||
if settled is not None:
|
||||
cached = _settled_get(settled)
|
||||
if cached is not None:
|
||||
return cached
|
||||
response = fetch_get(
|
||||
session,
|
||||
url,
|
||||
@@ -402,14 +266,7 @@ def _fetch_one_chunk(
|
||||
**_memo_kwargs(cache_max_age),
|
||||
)
|
||||
response.raise_for_status()
|
||||
payload = response_json(response)
|
||||
if settled is not None and isinstance(payload, dict):
|
||||
events = payload.get("events")
|
||||
# A capped month is truncated and gets re-asked day by day;
|
||||
# remembering it would only cost memory.
|
||||
if isinstance(events, list) and len(events) < ESPN_MAX_LIMIT:
|
||||
_settled_put(settled, response, payload)
|
||||
return cast(Optional[Dict[str, Any]], payload)
|
||||
return cast(Optional[Dict[str, Any]], response_json(response))
|
||||
except Exception as exc: # noqa: BLE001 - see docstring
|
||||
if logger:
|
||||
logger.warning("ESPN chunk %s failed, skipping it: %s", chunk, exc)
|
||||
|
||||
@@ -0,0 +1,97 @@
|
||||
"""Drop the parts of an ESPN scoreboard payload no scoreboard reads.
|
||||
|
||||
The sports scoreboards cache their Recent/Upcoming window (14 days back, 7
|
||||
ahead) as the raw ESPN response, and that record stays parsed in the memory
|
||||
cache for as long as it is fresh. Most of it is never drawn. Measured on hdpi
|
||||
(2026-10-02) the MLB window was 3.35MB of JSON and 13.5MB of Python objects,
|
||||
and the five windows together ~40MB, mostly in:
|
||||
|
||||
* ``competitors[].leaders`` / ``competitions[].leaders`` -- per-team and
|
||||
per-game stat leaders (28% of the MLB window)
|
||||
* ``competitors[].team.links`` / ``event.links`` -- web and app URLs
|
||||
* ``status.featuredAthletes`` and ``competitors[].probables`` -- athlete
|
||||
cards with headshots and season stats
|
||||
* ``competitions[].headlines`` / ``highlights`` -- article and video blurbs
|
||||
(28% of the college-football window)
|
||||
* ``competitions[].geoBroadcasts``
|
||||
|
||||
None of those keys is read by core or by any plugin in ledmatrix-plugins
|
||||
(checked 2026-10-02 across every scoreboard, the odds ticker and the
|
||||
leaderboard), while everything that is read -- odds, records, linescores,
|
||||
situation, statistics, notes, broadcasts, venue -- is kept. Dropping them
|
||||
takes the five windows from ~40MB to ~12MB of parsed objects and the files from
|
||||
10.6MB to 3.0MB, so the reads that parse an expired window on the render
|
||||
thread get 3-4x cheaper too.
|
||||
|
||||
:func:`slim_scoreboard_payload` changes the payload in place, and only ever
|
||||
removes the keys listed here: anything it does not know about is left alone.
|
||||
"""
|
||||
|
||||
from typing import Any, Dict
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
# Per level of the payload, the keys removed. Kept deliberately explicit:
|
||||
# adding a key here means checking that nothing reads it first.
|
||||
_EVENT_DROP = ("links",)
|
||||
_COMPETITION_DROP = ("leaders", "headlines", "highlights", "geoBroadcasts")
|
||||
_STATUS_DROP = ("featuredAthletes",)
|
||||
_COMPETITOR_DROP = ("leaders", "probables")
|
||||
_TEAM_DROP = ("links",)
|
||||
|
||||
|
||||
def is_espn_scoreboard_url(url: Any) -> bool:
|
||||
"""Whether ``url`` is an ESPN site-API scoreboard endpoint."""
|
||||
if not isinstance(url, str):
|
||||
return False
|
||||
try:
|
||||
parts = urlsplit(url)
|
||||
except ValueError:
|
||||
return False
|
||||
host = (parts.hostname or "").lower()
|
||||
if host != "espn.com" and not host.endswith(".espn.com"):
|
||||
return False
|
||||
return parts.path.rstrip("/").endswith("/scoreboard")
|
||||
|
||||
|
||||
def _drop(obj: Any, keys) -> None:
|
||||
if isinstance(obj, dict):
|
||||
for key in keys:
|
||||
obj.pop(key, None)
|
||||
|
||||
|
||||
def slim_scoreboard_payload(payload: Any) -> Any:
|
||||
"""Remove the unread parts of an ESPN scoreboard payload, in place.
|
||||
|
||||
Returns ``payload`` for convenience. Anything that is not shaped like a
|
||||
scoreboard (not a dict, no ``events`` list, odd entries) is passed over
|
||||
untouched rather than raising.
|
||||
"""
|
||||
if not isinstance(payload, dict):
|
||||
return payload
|
||||
events = payload.get("events")
|
||||
if not isinstance(events, list):
|
||||
return payload
|
||||
for event in events:
|
||||
if not isinstance(event, dict):
|
||||
continue
|
||||
_drop(event, _EVENT_DROP)
|
||||
competitions = event.get("competitions")
|
||||
if not isinstance(competitions, list):
|
||||
continue
|
||||
for competition in competitions:
|
||||
if not isinstance(competition, dict):
|
||||
continue
|
||||
_drop(competition, _COMPETITION_DROP)
|
||||
_drop(competition.get("status"), _STATUS_DROP)
|
||||
competitors = competition.get("competitors")
|
||||
if not isinstance(competitors, list):
|
||||
continue
|
||||
for competitor in competitors:
|
||||
if not isinstance(competitor, dict):
|
||||
continue
|
||||
_drop(competitor, _COMPETITOR_DROP)
|
||||
_drop(competitor.get("team"), _TEAM_DROP)
|
||||
return payload
|
||||
|
||||
|
||||
__all__ = ["is_espn_scoreboard_url", "slim_scoreboard_payload"]
|
||||
@@ -343,24 +343,6 @@ def _hermetic_unit_refresh(monkeypatch, tmp_path_factory):
|
||||
monkeypatch.setattr(unit_refresh, 'SYSTEMD_DIR', str(tmp_path_factory.getbasetemp() / 'no-systemd'))
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _forget_settled_espn_chunks(monkeypatch):
|
||||
"""src.common.espn_dates remembers past days process-wide; tests fake
|
||||
different answers for the same dates, so none may inherit another's.
|
||||
|
||||
"Today" is also pinned to 2000-01-01, so no date a test uses counts as
|
||||
settled unless the test says so (by pinning _utc_today itself). Without
|
||||
that, a test asking for last month twice passes while that month is
|
||||
recent and fails once it is three days old: the second ask is answered
|
||||
from memory."""
|
||||
from datetime import date
|
||||
from src.common import espn_dates
|
||||
monkeypatch.setattr(espn_dates, "_utc_today", lambda: date(2000, 1, 1))
|
||||
espn_dates.clear_settled_chunk_cache()
|
||||
yield
|
||||
espn_dates.clear_settled_chunk_cache()
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def reset_logging():
|
||||
"""Reset logging configuration before each test."""
|
||||
|
||||
@@ -39,14 +39,6 @@ def forget_rejected_ranges(monkeypatch):
|
||||
monkeypatch.setattr(espn_dates, "_ranges_rejected_until", 0.0)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def nothing_is_settled_yet(monkeypatch):
|
||||
"""Pin "today" before every date these tests use, so the settled-chunk
|
||||
memory stays out of tests that are not about it whatever the real date.
|
||||
TestSettledChunkCache moves it forward."""
|
||||
monkeypatch.setattr(espn_dates, "_utc_today", lambda: date(2000, 1, 1))
|
||||
|
||||
|
||||
class FakeResponse:
|
||||
def __init__(self, status_code=200, payload=None):
|
||||
self.status_code = status_code
|
||||
@@ -497,95 +489,3 @@ class TestConcurrency:
|
||||
|
||||
assert live["peak"] <= espn_dates.ESPN_CHUNK_WORKERS
|
||||
assert live["peak"] > 1, "chunks should actually overlap"
|
||||
|
||||
|
||||
class TestSettledChunkCache:
|
||||
"""Days that ended three or more days ago are fetched once a day, not hourly.
|
||||
|
||||
The scoreboards re-fetch a 22-day window every hour; on hdpi (2026-10-02)
|
||||
the 12 settled days were 68% of that window's bytes.
|
||||
"""
|
||||
|
||||
TODAY = date(2026, 10, 2)
|
||||
# The scoreboards' default window on that day: 14 back, 7 ahead.
|
||||
WINDOW = "20260918-20261009"
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def frozen_today(self, monkeypatch):
|
||||
monkeypatch.setattr(espn_dates, "_utc_today", lambda: self.TODAY)
|
||||
|
||||
def _events(self):
|
||||
days = [(9, d) for d in range(18, 31)] + [(10, d) for d in range(1, 10)]
|
||||
return {"2026%02d%02d" % (m, d): [{"id": f"{m}-{d}"}] for m, d in days}
|
||||
|
||||
def test_the_second_refresh_only_asks_for_unsettled_days(self):
|
||||
session = FakeSession(self._events())
|
||||
first = fetch_espn_scoreboard(session, URL, params={"dates": self.WINDOW})
|
||||
session.calls.clear()
|
||||
|
||||
second = fetch_espn_scoreboard(session, URL, params={"dates": self.WINDOW})
|
||||
|
||||
asked = sorted(call["dates"] for call in session.calls)
|
||||
# Sep 29 is the last settled day (today minus three).
|
||||
assert asked == ["20260930"] + ["202610%02d" % d for d in range(1, 10)]
|
||||
assert second["events"] == first["events"] # same events, same order
|
||||
assert len(second["events"]) == 22
|
||||
|
||||
def test_a_hit_is_a_fresh_copy(self):
|
||||
session = FakeSession(self._events())
|
||||
fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
|
||||
hit = fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
|
||||
hit["events"][0]["id"] = "mutated"
|
||||
again = fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
|
||||
assert again["events"][0]["id"] == "9-18"
|
||||
|
||||
def test_other_params_are_part_of_the_key(self):
|
||||
session = FakeSession(self._events())
|
||||
fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW, "groups": 80})
|
||||
session.calls.clear()
|
||||
fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
|
||||
assert len(session.calls) == 22 # a different question, nothing reused
|
||||
|
||||
def test_entries_expire_after_a_day(self, monkeypatch):
|
||||
clock = [1000.0]
|
||||
monkeypatch.setattr(espn_dates.time, "monotonic", lambda: clock[0])
|
||||
session = FakeSession(self._events())
|
||||
fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
|
||||
clock[0] += espn_dates.SETTLED_CHUNK_TTL_SECONDS + 1
|
||||
session.calls.clear()
|
||||
fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
|
||||
assert len(session.calls) == 22
|
||||
|
||||
def test_failed_chunks_are_not_remembered(self):
|
||||
session = FakeSession(self._events(), fail_chunks={"20260920"})
|
||||
fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
|
||||
session.fail_chunks.clear()
|
||||
session.calls.clear()
|
||||
data = fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
|
||||
assert "20260920" in [call["dates"] for call in session.calls]
|
||||
assert len(data["events"]) == 22
|
||||
|
||||
def test_a_capped_month_is_not_remembered_but_its_days_are(self):
|
||||
full = [{"id": f"x{i}"} for i in range(ESPN_MAX_LIMIT)]
|
||||
session = FakeSession({"202608": full, "20260801": [{"id": "d1"}]})
|
||||
fetch_espn_date_chunks(session, URL, params={"dates": "20260801-20260831"})
|
||||
session.calls.clear()
|
||||
data = fetch_espn_date_chunks(session, URL, params={"dates": "20260801-20260831"})
|
||||
assert [call["dates"] for call in session.calls] == ["202608"]
|
||||
assert [event["id"] for event in data["events"]] == ["d1"]
|
||||
|
||||
def test_memory_is_bounded(self, monkeypatch):
|
||||
monkeypatch.setattr(espn_dates, "SETTLED_CACHE_MAX_ENTRIES", 5)
|
||||
session = FakeSession(self._events())
|
||||
fetch_espn_date_chunks(session, URL, params={"dates": self.WINDOW})
|
||||
assert len(espn_dates._settled_chunks) == 5
|
||||
assert espn_dates._settled_bytes == sum(
|
||||
len(blob) for _, blob in espn_dates._settled_chunks.values())
|
||||
|
||||
def test_single_day_requests_are_untouched(self):
|
||||
# The live path asks for today (or one day) as a plain request; that
|
||||
# never goes through chunks or the memory.
|
||||
session = FakeSession(self._events())
|
||||
for _ in range(2):
|
||||
fetch_espn_scoreboard(session, URL, params={"dates": "20260918"})
|
||||
assert len(session.calls) == 2
|
||||
|
||||
@@ -0,0 +1,162 @@
|
||||
"""Tests for src/common/espn_payload.py and its use by BackgroundDataService."""
|
||||
|
||||
import copy
|
||||
import time
|
||||
from unittest.mock import MagicMock, Mock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from src.background_data_service import BackgroundDataService, shutdown_background_service
|
||||
from src.common.espn_payload import is_espn_scoreboard_url, slim_scoreboard_payload
|
||||
|
||||
SCOREBOARD = "https://site.api.espn.com/apis/site/v2/sports/baseball/mlb/scoreboard"
|
||||
|
||||
|
||||
def _event():
|
||||
"""One event carrying every key the slimming drops and a sample of the
|
||||
keys scoreboards read, at the depth ESPN puts them."""
|
||||
competitor = {
|
||||
"id": "10",
|
||||
"homeAway": "home",
|
||||
"score": "5",
|
||||
"team": {"abbreviation": "NYY", "logo": "https://a/l.png",
|
||||
"links": [{"href": "https://espn.com/team"}]},
|
||||
"records": [{"summary": "90-60"}],
|
||||
"linescores": [{"value": 1}],
|
||||
"statistics": [{"name": "hits", "displayValue": "9"}],
|
||||
"leaders": [{"name": "avg", "leaders": [{"athlete": {"id": "1"}}]}],
|
||||
"probables": [{"athlete": {"id": "2"}, "statistics": []}],
|
||||
}
|
||||
return {
|
||||
"id": "401",
|
||||
"date": "2026-10-01T23:05Z",
|
||||
"links": [{"href": "https://espn.com/game"}],
|
||||
"status": {"type": {"state": "post"}},
|
||||
"competitions": [{
|
||||
"status": {"type": {"state": "post", "shortDetail": "Final"},
|
||||
"featuredAthletes": [{"athlete": {"id": "3"}}]},
|
||||
"competitors": [competitor, dict(copy.deepcopy(competitor), homeAway="away")],
|
||||
"odds": [{"details": "NYY -150", "overUnder": 8.5}],
|
||||
"situation": {"outs": 2},
|
||||
"notes": [{"headline": "Game 1"}],
|
||||
"broadcasts": [{"names": ["FOX"]}],
|
||||
"venue": {"fullName": "Yankee Stadium"},
|
||||
"leaders": [{"name": "hits"}],
|
||||
"headlines": [{"description": "recap"}],
|
||||
"highlights": [{"links": {"source": {}}}],
|
||||
"geoBroadcasts": [{"media": {"shortName": "FOX"}}],
|
||||
}],
|
||||
}
|
||||
|
||||
|
||||
class TestSlimScoreboardPayload:
|
||||
def test_drops_exactly_the_listed_keys(self):
|
||||
payload = {"leagues": [{"id": "10"}], "events": [_event()]}
|
||||
slim_scoreboard_payload(payload)
|
||||
event = payload["events"][0]
|
||||
competition = event["competitions"][0]
|
||||
assert "links" not in event
|
||||
for key in ("leaders", "headlines", "highlights", "geoBroadcasts"):
|
||||
assert key not in competition
|
||||
assert "featuredAthletes" not in competition["status"]
|
||||
for competitor in competition["competitors"]:
|
||||
assert "leaders" not in competitor
|
||||
assert "probables" not in competitor
|
||||
assert "links" not in competitor["team"]
|
||||
|
||||
def test_keeps_everything_else_unchanged(self):
|
||||
"""Removing the dropped keys from the original by hand gives exactly
|
||||
the slimmed payload: nothing else moved, changed or went missing."""
|
||||
original = {"leagues": [{"id": "10"}], "events": [_event(), _event()]}
|
||||
expected = copy.deepcopy(original)
|
||||
for event in expected["events"]:
|
||||
del event["links"]
|
||||
competition = event["competitions"][0]
|
||||
for key in ("leaders", "headlines", "highlights", "geoBroadcasts"):
|
||||
del competition[key]
|
||||
del competition["status"]["featuredAthletes"]
|
||||
for competitor in competition["competitors"]:
|
||||
del competitor["leaders"], competitor["probables"]
|
||||
del competitor["team"]["links"]
|
||||
assert slim_scoreboard_payload(original) == expected
|
||||
|
||||
def test_in_place_and_returns_payload(self):
|
||||
payload = {"events": [_event()]}
|
||||
assert slim_scoreboard_payload(payload) is payload
|
||||
|
||||
@pytest.mark.parametrize("payload", [
|
||||
None, [], "x", {}, {"events": None}, {"events": "x"},
|
||||
{"events": [None, 1, "x", {"competitions": None}]},
|
||||
{"events": [{"competitions": [None, {"status": None, "competitors": None}]}]},
|
||||
{"events": [{"competitions": [{"competitors": [None, {"team": None}]}]}]},
|
||||
])
|
||||
def test_odd_shapes_pass_through(self, payload):
|
||||
before = copy.deepcopy(payload)
|
||||
assert slim_scoreboard_payload(payload) == before
|
||||
|
||||
|
||||
class TestIsEspnScoreboardUrl:
|
||||
@pytest.mark.parametrize("url", [
|
||||
SCOREBOARD,
|
||||
SCOREBOARD + "/",
|
||||
"http://site.api.espn.com/apis/site/v2/sports/football/college-football/scoreboard",
|
||||
])
|
||||
def test_scoreboards(self, url):
|
||||
assert is_espn_scoreboard_url(url)
|
||||
|
||||
@pytest.mark.parametrize("url", [
|
||||
None, "", 12,
|
||||
"https://site.api.espn.com/apis/site/v2/sports/baseball/mlb/teams",
|
||||
"https://site.api.espn.com/apis/site/v2/sports/football/nfl/summary",
|
||||
"https://example.com/scoreboard",
|
||||
"https://espn.com.evil.example/apis/x/scoreboard",
|
||||
"https://notespn.com/apis/x/scoreboard",
|
||||
])
|
||||
def test_not_scoreboards(self, url):
|
||||
assert not is_espn_scoreboard_url(url)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def service():
|
||||
shutdown_background_service()
|
||||
cache = MagicMock()
|
||||
cache.get.return_value = None
|
||||
svc = BackgroundDataService(cache, max_workers=1, request_timeout=5)
|
||||
yield svc
|
||||
svc.shutdown(wait=False)
|
||||
shutdown_background_service()
|
||||
|
||||
|
||||
def _run(service, url, **kwargs):
|
||||
response = Mock(status_code=200)
|
||||
response.json.return_value = {"events": [_event()]}
|
||||
response.raise_for_status.return_value = None
|
||||
delivered = []
|
||||
with patch.object(service.session, "get", return_value=response):
|
||||
req_id = service.submit_fetch_request(
|
||||
sport="mlb", year=2026, url=url, cache_key="mlb_schedule_window_14_7",
|
||||
callback=lambda result: delivered.append(result.data), **kwargs)
|
||||
deadline = time.time() + 5
|
||||
while not service.is_request_complete(req_id) and time.time() < deadline:
|
||||
time.sleep(0.02)
|
||||
cached = service.cache_manager.set.call_args[0][1]
|
||||
return cached, delivered
|
||||
|
||||
|
||||
class TestBackgroundServiceSlims:
|
||||
def test_espn_scoreboard_is_cached_and_delivered_slimmed(self, service):
|
||||
cached, delivered = _run(service, SCOREBOARD)
|
||||
competition = cached["events"][0]["competitions"][0]
|
||||
assert "leaders" not in competition
|
||||
assert "probables" not in competition["competitors"][0]
|
||||
assert competition["odds"] and competition["situation"]
|
||||
# The callback sees the very payload that was cached.
|
||||
assert delivered and delivered[0] is cached
|
||||
|
||||
def test_opt_out_caches_whole_response(self, service):
|
||||
cached, _ = _run(service, SCOREBOARD, slim_payload=False)
|
||||
assert cached == {"events": [_event()]}
|
||||
|
||||
def test_other_urls_untouched(self, service):
|
||||
cached, _ = _run(service, "https://example.com/feed")
|
||||
assert cached == {"events": [_event()]}
|
||||
Reference in New Issue
Block a user