fix(sports): recover from ESPN rejecting scoreboard date ranges (#591)

* fix(sports): recover from ESPN rejecting scoreboard date ranges

Since 2026-09-15 ESPN's site API answers `dates=YYYYMMDD-YYYYMMDD` with
400 "Failed to get events endpoint." for every sport. Single days, months
(`YYYYMM`) and season years still work. Every season and weeks-window fetch
in core failed, including the background service the scoreboards submit
their season schedules to.

src/common/espn_dates.py re-asks a rejected range as whole-month chunks
plus the leftover edge days, which tile the window exactly (a season is
8 requests, not 213). A month that comes back with exactly 500 events is
truncated (college baseball's March) and is re-asked day by day.

It also clamps `limit` to 500: above that ESPN truncates silently, e.g.
college football returns 25 of 68 games for one Saturday at limit=1000.

BackgroundDataService recovers rejected ranges on the worker thread and
advertises `handles_espn_date_ranges` so plugins can tell whether to hand
it a range. SportsCore, sports_shared, ESPNDataSource and APIHelper route
through the helper or the clamped limit.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>

* docs(changelog): ESPN date-range fallback and limit clamp

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>

* fix(sports): stop re-sending ESPN date ranges once one is rejected

Live scoreboards refresh every 30 seconds, and each refresh sent the range
first, got the 400, then fetched the chunks: three requests where one used
to do. After a rejection, ranges now go straight to chunks for six hours,
then the range is tried again so the workaround retires itself if ESPN
reverts. A single-day 400 does not set the memo, and when every chunk fails
the range request supplies the error without the chunks being fetched a
second time. Per-fetch chunk logging drops to debug; the rejection itself
stays a warning.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>

* fix(sports): clamp limit only on ESPN scoreboard submissions

The background service is generic, and limit above 500 only truncates
scoreboards. /teams needs limit=1000 (college football has 762 teams and
limit=500 returns 500), so a teams submission must keep its limit.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
Chuck
2026-09-16 17:12:28 -04:00
committed by GitHub
co-authored by Claude Opus 5
parent 2082665252
commit 7ae614aa35
10 changed files with 921 additions and 26 deletions
+42 -6
View File
@@ -25,6 +25,7 @@ from enum import Enum
import queue
from concurrent.futures import ThreadPoolExecutor
from src.cache_manager import CacheManager
from src.common.espn_dates import clamp_espn_limit, fetch_espn_date_chunks
# Configure logging
logger = logging.getLogger(__name__)
@@ -98,7 +99,12 @@ class BackgroundDataService:
This service manages a pool of background threads to fetch data asynchronously,
with intelligent caching, retry logic, and progress tracking.
"""
# Plugins feature-detect this. A core without it sends season ranges to
# ESPN as-is and gets 400s since 2026-09-15, so plugins fetch those
# ranges themselves instead of submitting them here.
handles_espn_date_ranges = True
def __init__(self, cache_manager: CacheManager, max_workers: int = 3, request_timeout: int = 30):
"""
Initialize the background data service.
@@ -247,6 +253,12 @@ class BackgroundDataService:
logger.debug(f"Cache hit for {sport} {year} data")
return request_id
# limit above 500 makes an ESPN *scoreboard* return a truncated list
# (src/common/espn_dates.py). Other endpoints need more: /teams has 762
# college-football teams, so only scoreboards are clamped.
if url.split('?', 1)[0].rstrip('/').endswith('/scoreboard'):
params = clamp_espn_limit(params)
# Create fetch request
request = FetchRequest(
id=request_id,
@@ -254,7 +266,7 @@ class BackgroundDataService:
year=year,
cache_key=cache_key,
url=url,
params=params or {},
params=dict(params or {}),
headers={**self.default_headers, **(headers or {})},
timeout=timeout or self.request_timeout,
max_retries=max_retries,
@@ -340,10 +352,17 @@ class BackgroundDataService:
# Perform HTTP request with retry logic
response = self._make_request_with_retry(request)
response.raise_for_status()
# Parse response
data = response.json()
# ESPN stopped accepting dates=YYYYMMDD-YYYYMMDD on 2026-09-15 and
# answers 400 for every sport. Re-ask in months and days rather
# than let a whole season fail. See src/common/espn_dates.py.
if response.status_code == 400:
data = self._fetch_in_date_chunks(request)
if data is None:
response.raise_for_status()
else:
response.raise_for_status()
data = response.json()
# Validate data structure
if not isinstance(data, dict):
@@ -519,6 +538,23 @@ class BackgroundDataService:
"""
result.data = None
def _fetch_in_date_chunks(self, request: FetchRequest) -> Optional[Dict[str, Any]]:
"""Re-fetch a rejected ``YYYYMMDD-YYYYMMDD`` range as month/day chunks.
None means the request was not a day range, or every chunk failed; the
caller then re-raises the original 400 instead of caching an empty
season. See src/common/espn_dates.py.
"""
logger.info("Recovering %s %s from a rejected date range", request.sport, request.year)
return fetch_espn_date_chunks(
self.session,
request.url,
params=request.params,
headers=request.headers,
timeout=request.timeout,
logger=logger,
)
def _make_request_with_retry(self, request: FetchRequest) -> requests.Response:
"""
Make HTTP request with retry logic and exponential backoff.
+9 -8
View File
@@ -10,6 +10,7 @@ from typing import Dict, List
import requests
import logging
from datetime import datetime
from src.common.espn_dates import fetch_espn_scoreboard
class DataSource(ABC):
"""Abstract base class for data sources."""
@@ -71,10 +72,10 @@ class ESPNDataSource(DataSource):
now = datetime.now()
formatted_date = now.strftime("%Y%m%d")
url = f"{self.base_url}/{sport}/{league}/scoreboard"
response = self.session.get(url, params={"dates": formatted_date, "limit": 1000}, headers=self.get_headers(), timeout=15)
response.raise_for_status()
data = response.json()
data = fetch_espn_scoreboard(
self.session, url, params={"dates": formatted_date, "limit": 1000},
headers=self.get_headers(), timeout=15, logger=self.logger,
)
events = data.get('events', [])
# Filter for live games
@@ -99,10 +100,10 @@ class ESPNDataSource(DataSource):
"limit": 1000
}
response = self.session.get(url, headers=self.get_headers(), params=params, timeout=15)
response.raise_for_status()
data = response.json()
data = fetch_espn_scoreboard(
self.session, url, params=params,
headers=self.get_headers(), timeout=15, logger=self.logger,
)
events = data.get('events', [])
self.logger.debug(f"Fetched {len(events)} scheduled games for {sport}/{league}")
+10 -6
View File
@@ -14,6 +14,7 @@ from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
import pytz
from src.common.espn_dates import fetch_espn_scoreboard
import requests
from PIL import Image, ImageDraw, ImageFont
from src.common.font_layout import load_truetype
@@ -844,9 +845,11 @@ class SportsCore(ABC):
formatted_date_yesterday = yesterday.strftime("%Y%m%d")
# Fetch todays games only
url = f"https://site.api.espn.com/apis/site/v2/sports/{self.sport}/{self.league}/scoreboard"
response = self.session.get(url, params={"dates": f"{formatted_date_yesterday}-{formatted_date}", "limit": 1000}, headers=self.headers, timeout=10)
response.raise_for_status()
data = response.json()
data = fetch_espn_scoreboard(
self.session, url,
params={"dates": f"{formatted_date_yesterday}-{formatted_date}", "limit": 1000},
headers=self.headers, timeout=10, logger=self.logger,
)
events = data.get('events', [])
self.logger.info(f"Fetched {len(events)} todays games for {self.sport} - {self.league}")
@@ -869,9 +872,10 @@ class SportsCore(ABC):
end_date = now + timedelta(weeks=1)
date_str = f"{start_date.strftime('%Y%m%d')}-{end_date.strftime('%Y%m%d')}"
url = f"https://site.api.espn.com/apis/site/v2/sports/{self.sport}/{self.league}/scoreboard"
response = self.session.get(url, params={"dates": date_str, "limit": 1000},headers=self.headers, timeout=10)
response.raise_for_status()
data = response.json()
data = fetch_espn_scoreboard(
self.session, url, params={"dates": date_str, "limit": 1000},
headers=self.headers, timeout=10, logger=self.logger,
)
immediate_events = data.get('events', [])
if immediate_events:
+4 -1
View File
@@ -8,6 +8,7 @@ Extracted from LEDMatrix core to provide reusable functionality for plugins.
import logging
import time
from datetime import datetime
from src.common.espn_dates import ESPN_MAX_LIMIT
from typing import Any, Dict, Optional
import requests
@@ -153,9 +154,11 @@ class APIHelper:
cache_key = f"espn_{sport}_{league}_{date}"
# Set parameters
# limit above 500 makes ESPN truncate instead of erroring: college
# football came back with 25 of 68 games. See src/common/espn_dates.py.
params = {
'dates': date,
'limit': 1000
'limit': ESPN_MAX_LIMIT
}
return self.get(url, params=params, cache_key=cache_key, cache_ttl=cache_ttl)
+294
View File
@@ -0,0 +1,294 @@
"""Working around two ESPN site-API behaviours that silently break scoreboards.
Both were found on 2026-09-15, when every scoreboard on the Pi started logging
``400 Client Error: Bad Request`` against URLs that had worked the day before.
1. **Date ranges are rejected.** ``?dates=YYYYMMDD-YYYYMMDD`` answers
``400 {"code":400,"message":"Failed to get events endpoint."}`` for *every*
sport -- football, baseball, hockey, basketball, soccer alike. Single days
(``?dates=YYYYMMDD``), whole months (``?dates=YYYYMM``) and season years
(``?dates=YYYY``) still answer 200.
2. **``limit`` above 500 corrupts the response.** ``limit=1000`` -- what every
caller in this codebase used to send -- makes college-football return 25
events where the truthful answer is 68 for a single Saturday and 323 for a
month. No error, just a short list. The cutoff sits between 500 and 600.
NFL-sized days never noticed, which is why this hid for so long.
The fix for (1) is to re-ask in units ESPN still honours. Whole calendar months
covered by the range become one ``YYYYMM`` request each and the leftover days at
either end become one ``YYYYMMDD`` request each, so the chunks cover the
requested window *exactly* -- no client-side date filtering, and therefore no
guessing at which timezone ESPN means by "a game day". A full NFL season
(20260801-20270301) costs 8 requests rather than 213 per-day ones.
A month can hold more than 500 events (college baseball's March does), and
ESPN answers that with exactly ``limit`` events and no hint that more exist. A
month chunk that comes back full is therefore re-asked day by day.
Once a range has been rejected, later ranges skip straight to chunks for
``RANGE_RETRY_SECONDS`` instead of spending a doomed request first -- live
scoreboards ask every 30 seconds. After that the range is tried again, so the
workaround retires itself if ESPN reverts.
"""
import threading
import time
from datetime import date, timedelta
from typing import Any, Dict, List, Optional, Tuple
# Above this, ESPN returns a truncated list instead of an error. See module
# docstring: 500 is the largest value measured to return complete data.
ESPN_MAX_LIMIT = 500
# How long a rejected range keeps later ranges from being tried as ranges.
RANGE_RETRY_SECONDS = 6 * 60 * 60
_range_lock = threading.Lock()
_ranges_rejected_until = 0.0
__all__ = [
"ESPN_MAX_LIMIT",
"RANGE_RETRY_SECONDS",
"clamp_espn_limit",
"parse_espn_date_range",
"espn_date_chunks",
"merge_scoreboard_payloads",
"fetch_espn_date_chunks",
"fetch_espn_scoreboard",
]
def clamp_espn_limit(params: Optional[Dict[str, Any]]) -> Dict[str, Any]:
"""Return a copy of ``params`` with any ``limit`` over 500 pulled back to 500."""
out = dict(params or {})
raw = out.get("limit")
if raw is None:
return out
try:
value = int(raw)
except (TypeError, ValueError):
return out
if value > ESPN_MAX_LIMIT:
out["limit"] = ESPN_MAX_LIMIT
return out
def parse_espn_date_range(dates: Any) -> Optional[Tuple[date, date]]:
"""Parse ``"YYYYMMDD-YYYYMMDD"`` into dates, or return None.
None means "not a day range" -- a single day, a month, a season year, or
anything unparseable. Those forms still work upstream and must be passed
through untouched rather than rewritten.
"""
if not isinstance(dates, str):
return None
halves = dates.split("-")
if len(halves) != 2 or len(halves[0]) != 8 or len(halves[1]) != 8:
return None
try:
start = date(int(halves[0][:4]), int(halves[0][4:6]), int(halves[0][6:]))
end = date(int(halves[1][:4]), int(halves[1][4:6]), int(halves[1][6:]))
except ValueError:
return None
if end < start:
return None
return start, end
def _ranges_known_rejected() -> bool:
with _range_lock:
return time.monotonic() < _ranges_rejected_until
def _note_range_rejected() -> None:
global _ranges_rejected_until
with _range_lock:
_ranges_rejected_until = time.monotonic() + RANGE_RETRY_SECONDS
def _first_of_next_month(day: date) -> date:
return date(day.year + (day.month == 12), day.month % 12 + 1, 1)
def _days_of_month(chunk: str) -> List[str]:
day = date(int(chunk[:4]), int(chunk[4:6]), 1)
stop = _first_of_next_month(day)
days = []
while day < stop:
days.append(day.strftime("%Y%m%d"))
day += timedelta(days=1)
return days
def espn_date_chunks(start: date, end: date) -> List[str]:
"""Cover ``[start, end]`` inclusive with ``dates=`` values ESPN accepts.
Whole calendar months inside the window collapse to one ``YYYYMM`` chunk;
partial months at the edges are spelled out day by day. The chunks tile the
window exactly -- they never reach outside it -- so merging their events
needs no date filtering afterwards.
"""
chunks: List[str] = []
cursor = start
while cursor <= end:
month_end = _first_of_next_month(cursor) - timedelta(days=1)
if cursor.day == 1 and month_end <= end:
chunks.append(cursor.strftime("%Y%m"))
cursor = month_end + timedelta(days=1)
else:
chunks.append(cursor.strftime("%Y%m%d"))
cursor += timedelta(days=1)
return chunks
def merge_scoreboard_payloads(payloads: List[Dict[str, Any]]) -> Dict[str, Any]:
"""Fold chunk responses into one scoreboard payload.
Events are de-duplicated by id and keep first-seen order. Non-event keys
(``leagues``, ``season``, ``week``) come from the first payload that has
them, matching what a single un-chunked response would have looked like.
"""
merged: Dict[str, Any] = {}
events: List[Dict[str, Any]] = []
seen = set()
for payload in payloads:
if not isinstance(payload, dict):
continue
for key, value in payload.items():
if key != "events" and key not in merged:
merged[key] = value
for event in payload.get("events") or []:
event_id = event.get("id") if isinstance(event, dict) else None
if event_id is not None:
if event_id in seen:
continue
seen.add(event_id)
events.append(event)
merged["events"] = events
return merged
def fetch_espn_date_chunks(
session,
url: str,
params: Optional[Dict[str, Any]] = None,
headers: Optional[Dict[str, str]] = None,
timeout: int = 15,
logger=None,
) -> Optional[Dict[str, Any]]:
"""Fetch a ``YYYYMMDD-YYYYMMDD`` window as month and day chunks.
Returns None when ``params["dates"]`` is not a day range, or when every
chunk failed. Callers treat None as "re-raise the original error": caching
an empty payload would read as "no games this season".
Chunks are always asked with ``limit=500``. ESPN's default page is smaller
than a busy month (100 for NFL, 300 for college football), and 500 is the
largest value that does not corrupt the answer. A month that comes back
with 500 events is assumed truncated and re-asked day by day. A failed
chunk is logged and skipped so one bad day cannot cost a whole season.
"""
params = dict(params or {})
span = parse_espn_date_range(params.get("dates"))
if span is None:
return None
chunks = espn_date_chunks(*span)
if logger:
logger.debug(
"Fetching ESPN date range %s as %d month/day chunks",
params.get("dates"), len(chunks),
)
payloads: List[Dict[str, Any]] = []
attempted = 0
pending = list(chunks)
while pending:
chunk = pending.pop(0)
attempted += 1
try:
response = session.get(
url,
params=dict(params, dates=chunk, limit=ESPN_MAX_LIMIT),
headers=headers,
timeout=timeout,
)
response.raise_for_status()
payload = response.json()
except Exception as exc: # noqa: BLE001 - one bad chunk must not sink the rest
if logger:
logger.warning("ESPN chunk %s failed, skipping it: %s", chunk, exc)
continue
events = payload.get("events") if isinstance(payload, dict) else None
if len(chunk) == 6 and len(events or []) >= ESPN_MAX_LIMIT:
if logger:
logger.info(
"ESPN month %s hit the %d-event cap; re-asking it day by day",
chunk, ESPN_MAX_LIMIT,
)
pending[:0] = _days_of_month(chunk)
continue
payloads.append(payload)
if not payloads:
return None
merged = merge_scoreboard_payloads(payloads)
if logger:
logger.debug(
"Recovered %d events for %s from %d/%d chunk requests",
len(merged["events"]), params.get("dates"), len(payloads), attempted,
)
return merged
def fetch_espn_scoreboard(
session,
url: str,
params: Optional[Dict[str, Any]] = None,
headers: Optional[Dict[str, str]] = None,
timeout: int = 15,
logger=None,
) -> Dict[str, Any]:
"""GET an ESPN scoreboard, re-asking in month/day chunks if a range 400s.
Anything that is not a ``YYYYMMDD-YYYYMMDD`` range is one request with the
caller's own parameters (``limit`` clamped), so single-day and season-year
callers see no change. A range that ESPN rejects is re-fetched in chunks,
and later ranges go straight to chunks for ``RANGE_RETRY_SECONDS``. A 400 on
a non-range request, any other error, and a range whose every chunk fails
all raise as before.
"""
params = clamp_espn_limit(params)
is_range = parse_espn_date_range(params.get("dates")) is not None
chunks_tried = False
if is_range and _ranges_known_rejected():
data = fetch_espn_date_chunks(
session, url, params=params, headers=headers,
timeout=timeout, logger=logger,
)
if data is not None:
return data
# Every chunk failed: ask for the range itself so the caller gets a
# real error to log, without spending the chunks a second time.
chunks_tried = True
response = session.get(url, params=params, headers=headers, timeout=timeout)
if is_range and response.status_code == 400 and not chunks_tried:
_note_range_rejected()
if logger:
logger.warning(
"ESPN rejected the date range %s (400); fetching it as month/day "
"chunks, and fetching ranges that way for the next %d hours",
params.get("dates"), RANGE_RETRY_SECONDS // 3600,
)
data = fetch_espn_date_chunks(
session, url, params=params, headers=headers,
timeout=timeout, logger=logger,
)
if data is not None:
return data
response.raise_for_status()
return response.json()
+4 -3
View File
@@ -84,6 +84,7 @@ from datetime import datetime, timedelta, timezone
from typing import Any, ClassVar, Dict, List, Optional, Tuple
import pytz
from src.common.espn_dates import fetch_espn_scoreboard
import requests
from PIL import Image, ImageDraw, ImageFont
from src.common.font_layout import load_truetype
@@ -947,14 +948,14 @@ class SportsCoreSharedMixin:
end_date = now + timedelta(days=self.schedule_lookahead_days)
date_str = f"{start_date.strftime('%Y%m%d')}-{end_date.strftime('%Y%m%d')}"
url = f"https://site.api.espn.com/apis/site/v2/sports/{self.sport}/{self.league}/scoreboard"
response = self.session.get(
data = fetch_espn_scoreboard(
self.session,
url,
params={"dates": date_str, "limit": 1000},
headers=self.headers,
timeout=10,
logger=self.logger,
)
response.raise_for_status()
data = response.json()
immediate_events = data.get("events", [])
if immediate_events: