fix(cache): cache keys too long to be a filename; memory hits judged by the record's own age (#738)

* fix(cache): store keys too long to be a filename

The calendar plugin's cache key joins every calendar id the user picked.
On hdpi it passed 300 bytes; ext4 caps a filename at 255, so every write
(the temp file, the direct-write fallback and the home-directory fallback)
failed with ENAMETOOLONG, once an hour, and the final warning said
"(permission denied)" whatever the error was.

DiskCache.get_cache_path keeps a key of up to 200 UTF-8 bytes as its
filename, exactly as before, and turns a longer one into its first 183
bytes (cut on a character boundary) plus a 16-hex-digit hash of the whole
key. The temp file adds 15 bytes, so the longest name is 215. The
shortened stem is itself short, so the web UI's cache list, which names a
key by its filename, deletes the same file. The give-up warning now names
the real error.

Validated on ledpi's ext4: the old module drops the hdpi-shaped key, the
new one writes a 205-byte filename and reads it back.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

* fix(cache): judge a memory hit by the record's own timestamp

A record loaded from disk went into the memory tier timed from the load,
so get(key, max_age=300) could return data close to 600 s old: after a
restart, after the memory sweep, or in a second process. A stored ttl was
stretched the same way. #728's _fresh_cached works around it for the
scoreboard; every other caller was exposed.

get_cached_data and load_cache now also check a memory hit against the
record's embedded timestamp, with DiskCache.get's rule that a stored ttl
wins over max_age. A stale copy is dropped and the read falls through to
disk, which returns the other process's newer write if there is one.
Records without a timestamp keep the memory tier's own clock.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Chuck
2026-10-03 22:18:00 -04:00
committed by GitHub
co-authored by Claude Opus 5.5
parent 41192b9588
commit a6e9e3ef1c
5 changed files with 285 additions and 7 deletions
+40 -5
View File
@@ -4,6 +4,7 @@ Disk Cache
Handles persistent disk-based caching with atomic writes and error recovery.
"""
import hashlib
import json
import math
import os
@@ -31,6 +32,35 @@ except ImportError: # pragma: no cover - exercised on hosts without the wheel
# useful, and a half-written file was never useful.
_ORPHAN_TEMP_MAX_AGE_SECONDS = 3600
# Longest key, in UTF-8 bytes, used verbatim as a filename stem. ext4 caps a
# name at 255 bytes and set()'s temp file is ".<stem>.json.<8 random>", 15
# bytes longer than the stem, so anything near the cap could never be written:
# the calendar plugin's key joins every calendar id and passed 300 bytes on a
# real install, failing every write with ENAMETOOLONG. Longer keys keep this
# many bytes as a readable prefix and end in a hash of the whole key.
_MAX_KEY_FILENAME_BYTES = 200
_KEY_HASH_CHARS = 16
def _filename_stem(key: str) -> str:
"""The filename stem for a key that is already a safe path component.
Short keys are used as they are, so every file already on disk keeps its
name. A long one becomes its first bytes plus a hash of the full key: the
prefix keeps the stem recognisable (and keeps the data-type words that
cleanup's retention lookup reads from it), the hash keeps two keys that
share a long prefix apart. The result is itself short, so a stem read back
from a filename -- which is how the web UI names a key it deletes -- maps to
the same file.
"""
encoded = key.encode('utf-8')
if len(encoded) <= _MAX_KEY_FILENAME_BYTES:
return key
digest = hashlib.sha256(encoded).hexdigest()[:_KEY_HASH_CHARS]
keep = _MAX_KEY_FILENAME_BYTES - _KEY_HASH_CHARS - 1
prefix = encoded[:keep].decode('utf-8', errors='ignore')
return f"{prefix}-{digest}"
class CacheStrategyProtocol(Protocol):
@@ -343,6 +373,8 @@ class DiskCache:
derives them), so rejecting anything with a path component turns
away only inputs that could never have been written here.
A key too long to be a filename is shortened by _filename_stem.
Args:
key: Cache key
@@ -356,7 +388,7 @@ class DiskCache:
if safe_key is None:
self.logger.warning("Rejected unsafe cache key %r", key)
return None
return os.path.join(self.cache_dir, f"{safe_key}.json")
return os.path.join(self.cache_dir, f"{_filename_stem(safe_key)}.json")
def get(self, key: str, max_age: Optional[int] = 300) -> Optional[Dict[str, Any]]:
"""
@@ -561,7 +593,7 @@ class DiskCache:
# If direct write also fails, try fallback location
self.logger.warning("Direct write failed for key '%s' to %s: %s", key, cache_path, write_error)
raise # Re-raise to trigger fallback logic
except (IOError, OSError, PermissionError):
except (IOError, OSError, PermissionError) as primary_error:
# Attempt one-time fallback write to user's home cache directory
try:
# Try user's home cache directory as fallback
@@ -587,11 +619,14 @@ class DiskCache:
self.logger.debug("Fallback cache write also failed for key '%s': %s", key, e2)
# If all write attempts failed, log warning but don't raise exception
# Cache is a performance optimization, not critical for operation
# Cache is a performance optimization, not critical for operation.
# Name the real error: this used to say "permission denied"
# whatever happened, which sent a too-long filename off to
# be debugged as a directory-ownership problem.
self.logger.warning(
"Could not write cache for key '%s' to %s (permission denied). "
"Could not write cache for key '%s' to %s (%s). "
"Cache will be unavailable for this key, but application will continue.",
key, cache_path
key, cache_path, primary_error.strerror or primary_error
)
return # Exit gracefully without raising exception
+34 -2
View File
@@ -46,6 +46,32 @@ from src.cache.disk_cache import DateTimeEncoder # noqa: F401 - deliberate re-e
# CacheManager.config_manager not built yet (None means "not available").
_UNSET: Any = object()
def _outlived(record: Any, max_age: Optional[float], now: float) -> bool:
"""Whether a record's own timestamp puts it past max_age.
The memory tier times an entry from when it was put there, and a record
loaded from disk is put there when it is read, not when it was written: a
record 290 s old, read after a restart, could be served for another
max_age from memory. This is the age check DiskCache.get makes, with the
same rule that a stored ttl wins over the caller's max_age. A record that
carries no timestamp is left to the memory tier's own clock.
"""
if not isinstance(record, dict):
return False
stored_ttl = record.get('ttl')
if isinstance(stored_ttl, (int, float)) and not isinstance(stored_ttl, bool) \
and stored_ttl >= 0:
max_age = stored_ttl
stamp = record.get('timestamp')
if max_age is None or stamp is None or isinstance(stamp, bool):
return False
try:
return now - float(stamp) > max_age
except (TypeError, ValueError):
return False
class CacheManager:
"""Manages caching of API responses to reduce API calls."""
@@ -284,7 +310,11 @@ class CacheManager:
# 1) Memory cache
cached = self._memory_cache_component.get(key, max_age=in_memory_ttl)
if cached is not None:
return cached
if not _outlived(cached, max_age, time.time()):
return cached
# Too old for this reader. Disk may hold a newer write (from the
# other process), and if it does not, the miss is the right answer.
self._memory_cache_component.clear(key)
# 2) Disk cache
record = self._disk_cache_component.get(key, max_age=max_age)
@@ -318,7 +348,9 @@ class CacheManager:
# Check memory cache first (1 minute TTL)
cached = self._memory_cache_component.get(key, max_age=60)
if cached is not None:
return cached
if not _outlived(cached, 3600, time.time()):
return cached
self._memory_cache_component.clear(key)
# Check disk cache
data = self._disk_cache_component.get(key, max_age=3600) # 1 hour for load_cache