mirror of
https://github.com/ChuckBuilds/LEDMatrix.git
synced 2026-10-04 14:25:08 +00:00
perf(cache): skip rewriting unchanged CacheManager.set() records (#730)
The disk cache's unchanged-payload skip now ignores a CacheManager.set() record's timestamp, so unchanged re-saves are skipped; a skip moves the file's mtime to the new timestamp instead, and readers take a record's age from the newer of the two (never more than an hour past the embedded timestamp). Per-plugin plugin_metrics:<id> records become one plugin_metrics_snapshot written at most once a minute, and CacheManager builds its ConfigManager on first use. On hdpi, cache file writes went from ~37 to 8.6 a minute. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Vendored
+181
-37
@@ -14,7 +14,7 @@ import tempfile
|
||||
import logging
|
||||
import threading
|
||||
import zlib
|
||||
from typing import Dict, Any, Optional, Protocol
|
||||
from typing import Dict, Any, Optional, Protocol, Tuple
|
||||
from datetime import datetime
|
||||
|
||||
from src.common.path_safety import safe_path_component
|
||||
@@ -111,18 +111,91 @@ _HEAD_RE = re.compile(
|
||||
)
|
||||
|
||||
|
||||
def _stale_from_head(head: bytes, max_age: Optional[int], now: float) -> bool:
|
||||
# UNCHANGED RE-SAVES: THE FILE'S MTIME CARRIES THE NEWER TIMESTAMP
|
||||
# ----------------------------------------------------------------
|
||||
# Plugins re-save unchanged API data every update cycle, and every one of
|
||||
# those saves was a full rewrite on the SD card. DiskCache.set skips the write
|
||||
# when the payload matches the last one it wrote for the key -- but
|
||||
# CacheManager.set stamps each record with time.time(), so for set() the
|
||||
# payload never matched and the skip never fired.
|
||||
#
|
||||
# The digest now leaves out a header-first record's timestamp, so an unchanged
|
||||
# set() is skipped. What the skip must not do is make the record look older
|
||||
# than it is: the timestamp inside the file is from the last real write, and
|
||||
# a reader in another process (the web interface, with memory_ttl=0) or after
|
||||
# a restart would call fresh data stale. So the newer timestamp goes where it
|
||||
# costs no data write -- the file's mtime -- and readers take a record's age
|
||||
# from the newer of the two. The invariant that makes that safe:
|
||||
#
|
||||
# a file's mtime is the timestamp of the newest record saved for its key
|
||||
#
|
||||
# real write mtime is set to the record's own timestamp, so a record saved
|
||||
# with an old timestamp (data as of some earlier time) cannot
|
||||
# borrow freshness from the moment it hit the disk
|
||||
# skip mtime is set to the skipped record's timestamp -- exactly what
|
||||
# a rewrite would have stored, without the rewrite
|
||||
#
|
||||
# Readers of the on-disk timestamp, all of which go through _effective_timestamp:
|
||||
# DiskCache.get (the header check and the full parse; it also returns the
|
||||
# record with 'timestamp' set to the effective value, so CacheManager.get's
|
||||
# max_age path, the memory tier hydrated from disk, and any plugin reading
|
||||
# record['timestamp'] all see it). Readers that use mtime alone already see the
|
||||
# newer value: the retention sweep below, CacheManager.list_cache_files (the
|
||||
# web UI's cache list). Nothing else opens cache files: web_interface and
|
||||
# scripts reach them only through CacheManager.
|
||||
#
|
||||
# Something other than this class can also move an mtime forward -- a copy
|
||||
# without -p, an rsync without -t, a `touch`. (backup_manager.py does not
|
||||
# back up or restore the cache directory, so the in-tree restore cannot.) That
|
||||
# must not make old data fresh, so the lift is bounded: a reader never takes
|
||||
# the mtime as more than _MAX_TIMESTAMP_LIFT past the embedded timestamp, and
|
||||
# set() rewrites the file for real once a skip would need more than that, so
|
||||
# an honest lift never reaches the bound. A file copied a day after it was
|
||||
# written therefore reads at most an hour fresher than its contents say, and a
|
||||
# 30-second live-score record from yesterday stays stale. CacheManager.set
|
||||
# records written before this change have mtime == write time == embedded
|
||||
# timestamp, give or take the write itself, and read exactly as before; a
|
||||
# file an older version wrote or touched later than its embedded timestamp
|
||||
# says reads at most the same hour fresher, once, until it is next saved.
|
||||
|
||||
#: Longest a skipped write may stand in for a real one, and so the furthest a
|
||||
#: file's mtime is ever trusted past the record's own timestamp. Unchanged data
|
||||
#: is rewritten at least this often, at most once an hour per key instead of
|
||||
#: once per update cycle.
|
||||
_MAX_TIMESTAMP_LIFT = 3600.0
|
||||
|
||||
|
||||
def _record_timestamp(value: Any) -> Optional[float]:
|
||||
"""A record's timestamp as a finite float, or None if it has no usable one."""
|
||||
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
||||
return None
|
||||
value = float(value)
|
||||
return value if math.isfinite(value) else None
|
||||
|
||||
|
||||
def _effective_timestamp(embedded: float, mtime: Optional[float]) -> float:
|
||||
"""When a record was last saved: its timestamp, or the file's mtime if a
|
||||
later unchanged save moved that forward -- never by more than
|
||||
_MAX_TIMESTAMP_LIFT. See "UNCHANGED RE-SAVES" above."""
|
||||
if mtime is None:
|
||||
return embedded
|
||||
return max(embedded, min(mtime, embedded + _MAX_TIMESTAMP_LIFT))
|
||||
|
||||
|
||||
def _stale_from_head(head: bytes, max_age: Optional[int], now: float,
|
||||
mtime: Optional[float] = None) -> bool:
|
||||
"""True when a record's header alone shows it has expired.
|
||||
|
||||
Mirrors the expiry rule in DiskCache.get: a per-entry ttl wins over the
|
||||
caller's max_age, and no limit at all means never stale. False whenever the
|
||||
header cannot be read, so the full parse decides as it always did.
|
||||
header cannot be read, so the full parse decides as it always did. ``mtime``
|
||||
is the file's, which may carry a newer save than the header does.
|
||||
"""
|
||||
match = _HEAD_RE.match(head)
|
||||
if not match:
|
||||
return False
|
||||
try:
|
||||
timestamp = float(match.group(1))
|
||||
timestamp = _effective_timestamp(float(match.group(1)), mtime)
|
||||
limit = max_age
|
||||
if match.group(2) is not None:
|
||||
ttl = float(match.group(2))
|
||||
@@ -179,7 +252,7 @@ else:
|
||||
# --------------------------------------------
|
||||
# The display service runs as root and the web interface as the installing
|
||||
# user, and the web interface reads records only the display writes
|
||||
# (display_current_state, display_on_demand_state, plugin_metrics:*). Files are
|
||||
# (display_current_state, display_on_demand_state, plugin_metrics_snapshot). Files are
|
||||
# written 0660, so the web interface can read one only through its group.
|
||||
#
|
||||
# The installers rely on the directory's setgid bit to set that group. That is
|
||||
@@ -248,11 +321,14 @@ class DiskCache:
|
||||
self.cache_dir = cache_dir
|
||||
self.logger = logger or logging.getLogger(__name__)
|
||||
self._lock = threading.Lock()
|
||||
# key -> adler32 of the last payload successfully written to the
|
||||
# primary cache path; lets set() skip rewriting identical data
|
||||
# (per-process only — worst case another process rewrites, never
|
||||
# a missed write). Guarded by _lock.
|
||||
self._write_digests: Dict[str, int] = {}
|
||||
# key -> what set() last put at the primary cache path: the adler32 of
|
||||
# the payload (less a header-first timestamp), the timestamp the file
|
||||
# holds (None for records without one), and the file's inode and size.
|
||||
# Lets set() skip rewriting identical data. Per-process only, and the
|
||||
# inode/size check means another process's write is never mistaken
|
||||
# for ours -- worst case a redundant write, never a missed one.
|
||||
# Guarded by _lock.
|
||||
self._write_digests: Dict[str, Tuple[int, Optional[float], int, int]] = {}
|
||||
|
||||
def get_cache_path(self, key: str) -> Optional[str]:
|
||||
"""
|
||||
@@ -301,32 +377,41 @@ class DiskCache:
|
||||
try:
|
||||
with self._lock:
|
||||
with open(cache_path, 'rb') as f:
|
||||
# The open file's mtime, not the path's: the file a skipped
|
||||
# write touched is the one being read.
|
||||
mtime = os.fstat(f.fileno()).st_mtime
|
||||
# Decide staleness from the header before paying for the
|
||||
# parse. A stale read is the common case for the biggest
|
||||
# records (a season schedule is re-fetched when its cache
|
||||
# expires), and parsing 53MB to throw it away held the GIL
|
||||
# for ~1.8s -- a visible freeze on the panel.
|
||||
if _stale_from_head(f.read(_HEAD_BYTES), max_age, time.time()):
|
||||
if _stale_from_head(f.read(_HEAD_BYTES), max_age, time.time(), mtime):
|
||||
return None
|
||||
f.seek(0)
|
||||
record = _loads(f.read())
|
||||
|
||||
# Determine record timestamp (prefer embedded, else file mtime)
|
||||
|
||||
# Determine record timestamp: the embedded one, moved forward by a
|
||||
# later unchanged save if there was one (see "UNCHANGED RE-SAVES"),
|
||||
# else the file mtime.
|
||||
record_ts = None
|
||||
if isinstance(record, dict):
|
||||
record_ts = record.get('timestamp')
|
||||
if record_ts is None:
|
||||
try:
|
||||
record_ts = os.path.getmtime(cache_path)
|
||||
except OSError:
|
||||
record_ts = None
|
||||
|
||||
if record_ts is not None:
|
||||
try:
|
||||
record_ts = float(record_ts)
|
||||
except (TypeError, ValueError):
|
||||
record_ts = None
|
||||
|
||||
record_ts = mtime
|
||||
else:
|
||||
embedded_ts = _record_timestamp(record_ts)
|
||||
if embedded_ts is None:
|
||||
try:
|
||||
record_ts = float(record_ts)
|
||||
except (TypeError, ValueError):
|
||||
record_ts = None
|
||||
else:
|
||||
record_ts = _effective_timestamp(embedded_ts, mtime)
|
||||
if record_ts != embedded_ts:
|
||||
# Hand the record back as a rewrite would have left it,
|
||||
# so callers that age it themselves agree with us.
|
||||
record['timestamp'] = record_ts
|
||||
|
||||
now = time.time()
|
||||
|
||||
# An explicit per-entry ttl wins over the caller's max_age. The
|
||||
@@ -403,7 +488,12 @@ class DiskCache:
|
||||
self.logger.warning("Cache data for key '%s' not serializable: %s", key, e)
|
||||
return
|
||||
|
||||
digest = zlib.adler32(payload)
|
||||
timestamp = _record_timestamp(data.get('timestamp')) if isinstance(data, dict) else None
|
||||
# A header-first record (CacheManager.set's layout) is compared without
|
||||
# its timestamp, which differs on every save; see "UNCHANGED RE-SAVES".
|
||||
# Any other layout is compared whole, as before.
|
||||
head = _HEAD_RE.match(payload) if timestamp is not None else None
|
||||
digest = zlib.adler32(memoryview(payload)[head.end(1):] if head else payload)
|
||||
|
||||
try:
|
||||
# Atomic write to avoid partial/corrupt files
|
||||
@@ -411,16 +501,10 @@ class DiskCache:
|
||||
# Skip the disk entirely when this exact payload was already
|
||||
# written for this key (plugins re-save unchanged API data
|
||||
# every update cycle — each write is real SD-card wear).
|
||||
# Refresh the file mtime so records that rely on it for TTL
|
||||
# (no embedded 'timestamp') don't expire early; a metadata
|
||||
# touch is journal-cheap compared to rewriting the data.
|
||||
if self._write_digests.get(key) == digest:
|
||||
try:
|
||||
os.utime(cache_path, None)
|
||||
return
|
||||
except OSError:
|
||||
# File vanished or perms changed — fall through and write
|
||||
self._write_digests.pop(key, None)
|
||||
# A metadata touch is journal-cheap compared to rewriting
|
||||
# the data.
|
||||
if self._skip_unchanged(key, cache_path, digest, timestamp):
|
||||
return
|
||||
|
||||
tmp_dir = os.path.dirname(cache_path)
|
||||
# Try to create temp file in cache directory first
|
||||
@@ -458,7 +542,7 @@ class DiskCache:
|
||||
# opened it in between was refused.
|
||||
_share_open_file(tmp_file.fileno(), _shared_group(tmp_dir))
|
||||
os.replace(tmp_path, cache_path)
|
||||
self._write_digests[key] = digest
|
||||
self._remember_write(key, cache_path, digest, timestamp)
|
||||
finally:
|
||||
if os.path.exists(tmp_path):
|
||||
try:
|
||||
@@ -471,7 +555,7 @@ class DiskCache:
|
||||
with open(cache_path, 'wb') as cache_file:
|
||||
cache_file.write(payload)
|
||||
_share_open_file(cache_file.fileno(), _shared_group(tmp_dir))
|
||||
self._write_digests[key] = digest
|
||||
self._remember_write(key, cache_path, digest, timestamp)
|
||||
self.logger.debug("Wrote cache for %s directly (non-atomic)", key)
|
||||
except (IOError, OSError, PermissionError) as write_error:
|
||||
# If direct write also fails, try fallback location
|
||||
@@ -520,6 +604,66 @@ class DiskCache:
|
||||
)
|
||||
return # Exit gracefully without raising exception
|
||||
|
||||
def _skip_unchanged(self, key: str, cache_path: str, digest: int,
|
||||
timestamp: Optional[float]) -> bool:
|
||||
"""Stand in for a write of an unchanged record by touching the file.
|
||||
|
||||
True when the file already holds this record bar its timestamp and the
|
||||
touch landed; False means write it. The touch sets mtime to the
|
||||
record's timestamp -- what a rewrite would have stored -- or to now
|
||||
for a record without one, whose age readers already take from mtime.
|
||||
Caller holds _lock.
|
||||
"""
|
||||
last = self._write_digests.get(key)
|
||||
if last is None or last[0] != digest:
|
||||
return False
|
||||
_, written_ts, ino, size = last
|
||||
if (timestamp is None) != (written_ts is None):
|
||||
return False
|
||||
if timestamp is not None and written_ts is not None:
|
||||
# Never backwards (a rewrite would make the record older), and
|
||||
# never further than readers will trust the mtime: past that the
|
||||
# record is rewritten, so its own timestamp catches up.
|
||||
if not written_ts <= timestamp <= written_ts + _MAX_TIMESTAMP_LIFT:
|
||||
return False
|
||||
try:
|
||||
st = os.stat(cache_path)
|
||||
if (st.st_ino, st.st_size) != (ino, size):
|
||||
# Replaced since our write (another process, a restore):
|
||||
# its contents are not the ones the digest describes.
|
||||
self._write_digests.pop(key, None)
|
||||
return False
|
||||
# Setting an explicit time needs the file's owner; a file someone
|
||||
# else wrote fails here and is rewritten (as our own file) instead.
|
||||
os.utime(cache_path, None if timestamp is None else (timestamp, timestamp))
|
||||
return True
|
||||
except OSError:
|
||||
# File vanished or perms changed — fall through and write
|
||||
self._write_digests.pop(key, None)
|
||||
return False
|
||||
|
||||
def _remember_write(self, key: str, cache_path: str, digest: int,
|
||||
timestamp: Optional[float]) -> None:
|
||||
"""After a real write: pin mtime to the record's timestamp and note
|
||||
what was written, so the next unchanged save can be skipped.
|
||||
|
||||
Pinning keeps a record saved with an older timestamp from looking as
|
||||
fresh as the moment it was written (see "UNCHANGED RE-SAVES"); for
|
||||
CacheManager.set's records the two differ only by the write itself.
|
||||
A timestamp in the future is left alone, mtime already being older.
|
||||
Never raises: the data is on disk, and anything failing here only
|
||||
costs the next save its skip. Caller holds _lock.
|
||||
"""
|
||||
self._write_digests.pop(key, None)
|
||||
try:
|
||||
if timestamp is not None and timestamp <= time.time():
|
||||
os.utime(cache_path, (timestamp, timestamp))
|
||||
st = os.stat(cache_path)
|
||||
except OSError as e:
|
||||
self.logger.debug("Could not pin mtime of %s: %s", cache_path, e)
|
||||
return
|
||||
self._write_digests[key] = (digest, timestamp, st.st_ino, st.st_size)
|
||||
|
||||
def clear(self, key: Optional[str] = None) -> None:
|
||||
"""
|
||||
Clear cache entry or all entries.
|
||||
|
||||
+49
-10
@@ -43,6 +43,9 @@ from src.logging_config import get_logger
|
||||
# it from either path.
|
||||
from src.cache.disk_cache import DateTimeEncoder # noqa: F401 - deliberate re-export
|
||||
|
||||
# CacheManager.config_manager not built yet (None means "not available").
|
||||
_UNSET: Any = object()
|
||||
|
||||
class CacheManager:
|
||||
"""Manages caching of API responses to reduce API calls."""
|
||||
|
||||
@@ -73,21 +76,19 @@ class CacheManager:
|
||||
self.logger.error("Could not find or create a writable cache directory. Caching will be disabled.")
|
||||
self.cache_dir = None
|
||||
|
||||
# Initialize config manager for sport-specific intervals
|
||||
try:
|
||||
from src.config_manager import ConfigManager
|
||||
self.config_manager: Optional[Any] = ConfigManager()
|
||||
self.config_manager.load_config()
|
||||
except ImportError:
|
||||
self.config_manager: Optional[Any] = None
|
||||
self.logger.warning("ConfigManager not available, using default cache intervals")
|
||||
|
||||
# The config manager is built on first use of self.config_manager; see
|
||||
# the property. Nothing in the cache reads it any more.
|
||||
self._config_manager: Any = _UNSET
|
||||
self._config_manager_lock = threading.Lock()
|
||||
|
||||
# Initialize cache components using composition
|
||||
self._memory_cache_component = MemoryCache(
|
||||
max_size=default_max_size(), cleanup_interval=300.0
|
||||
)
|
||||
self._disk_cache_component = DiskCache(cache_dir=self.cache_dir, logger=self.logger)
|
||||
self._strategy_component = CacheStrategy(config_manager=self.config_manager, logger=self.logger)
|
||||
# No config manager: CacheStrategy keeps the parameter for callers but
|
||||
# reads nothing from it, and passing ours would build it eagerly.
|
||||
self._strategy_component = CacheStrategy(logger=self.logger)
|
||||
self._metrics_component = CacheMetrics(logger=self.logger)
|
||||
|
||||
# Disk cleanup configuration
|
||||
@@ -115,6 +116,44 @@ class CacheManager:
|
||||
if self.cache_dir:
|
||||
self.start_cleanup_thread()
|
||||
|
||||
@property
|
||||
def config_manager(self) -> Optional[Any]:
|
||||
"""A loaded ConfigManager, built the first time it is asked for.
|
||||
|
||||
Every CacheManager used to build one and load the whole config in
|
||||
__init__, for a cache strategy that stopped reading it -- startup paid
|
||||
a config load (and the web interface another) per manager for nothing.
|
||||
It is still public: the sports plugins resolve the global timezone and
|
||||
display settings through ``cache_manager.config_manager``, and they get
|
||||
the same object they always did, on first access instead of at
|
||||
construction. None when ConfigManager cannot be imported, as before.
|
||||
Assigning replaces it, as assigning the attribute always did.
|
||||
"""
|
||||
# getattr: a manager made with __new__ (some tests) has no slot yet.
|
||||
value = getattr(self, '_config_manager', _UNSET)
|
||||
if value is not _UNSET:
|
||||
return value
|
||||
lock = getattr(self, '_config_manager_lock', None) or threading.Lock()
|
||||
with lock:
|
||||
value = getattr(self, '_config_manager', _UNSET)
|
||||
if value is _UNSET:
|
||||
try:
|
||||
from src.config_manager import ConfigManager
|
||||
except ImportError:
|
||||
self.logger.warning("ConfigManager not available, using default cache intervals")
|
||||
value = None
|
||||
else:
|
||||
value = ConfigManager()
|
||||
# Raises as it did from __init__; nothing is kept, so the
|
||||
# next access tries again.
|
||||
value.load_config()
|
||||
self._config_manager = value
|
||||
return value
|
||||
|
||||
@config_manager.setter
|
||||
def config_manager(self, value: Optional[Any]) -> None:
|
||||
self._config_manager = value
|
||||
|
||||
def _get_writable_cache_dir(self) -> Optional[str]:
|
||||
"""Tries to find or create a writable cache directory, preferring a system path when available."""
|
||||
# Attempt 1: System-wide persistent cache directory (preferred for services)
|
||||
|
||||
@@ -485,7 +485,7 @@ def record_error(
|
||||
# and only the display service's ever records anything (plugin_executor runs
|
||||
# the plugins there). The web interface therefore reads a snapshot the display
|
||||
# service publishes to the shared cache directory -- the same channel, and the
|
||||
# same file permissions, as display_current_state and plugin_metrics:*: files
|
||||
# same file permissions, as display_current_state and plugin_metrics_snapshot: files
|
||||
# are 0660 and carry the cache directory's group, so root writes and the web
|
||||
# user reads, and the other way round for the clear request.
|
||||
#
|
||||
|
||||
@@ -8,7 +8,7 @@ Provides resource limits and performance monitoring.
|
||||
import math
|
||||
import time
|
||||
import threading
|
||||
from typing import Dict, Optional, Any, Callable, cast
|
||||
from typing import Dict, Optional, Any, Callable, Set, cast
|
||||
from dataclasses import dataclass, field, fields
|
||||
|
||||
from src.logging_config import get_logger
|
||||
@@ -99,18 +99,33 @@ class ResourceMetrics:
|
||||
last_update_time: float = field(default_factory=time.time)
|
||||
|
||||
|
||||
#: How often a plugin's metrics are written to the cache, in seconds.
|
||||
#: How often the metrics snapshot is written to the cache, in seconds.
|
||||
#:
|
||||
#: Persisting on every call meant a small file rewritten roughly nine times a
|
||||
#: minute per plugin. On a rig with fourteen active plugins that was ~126
|
||||
#: writes a minute for metrics alone, and since each ~350-byte file costs a
|
||||
#: 4KB block plus an ext4 journal entry, it dominated the device's write
|
||||
#: volume -- on an SD card, which wears out.
|
||||
#: volume -- on an SD card, which wears out. Throttling each plugin's own
|
||||
#: record to once per 30 s still left two writes a minute per plugin, so all
|
||||
#: plugins now share one record (METRICS_SNAPSHOT_KEY), written at most once
|
||||
#: a minute: one write a minute however many plugins there are.
|
||||
#:
|
||||
#: The in-memory copy stays authoritative and exact; only the cross-process
|
||||
#: snapshot the web UI reads is delayed, and telemetry up to half a minute old
|
||||
#: is still a fair description of a long-running plugin.
|
||||
_METRICS_PERSIST_INTERVAL = 30.0
|
||||
#: snapshot the web UI reads is delayed, and telemetry up to a minute old is
|
||||
#: still a fair description of a long-running plugin.
|
||||
_METRICS_PERSIST_INTERVAL = 60.0
|
||||
|
||||
#: The one cache record holding every plugin's metrics:
|
||||
#: ``{"schema": 1, "plugins": {plugin_id: <metrics record>}}``, each metrics
|
||||
#: record shaped as the per-plugin ``plugin_metrics:<id>`` records were. Those
|
||||
#: older records are still read for a plugin the snapshot does not have yet
|
||||
#: (an upgrade, or a plugin that has not run since), never written.
|
||||
METRICS_SNAPSHOT_KEY = "plugin_metrics_snapshot"
|
||||
_METRICS_SNAPSHOT_SCHEMA = 1
|
||||
|
||||
#: A plugin with no call for this long is dropped from the snapshot -- what
|
||||
#: the cache's 30-day default retention did to its own record before.
|
||||
_METRICS_SNAPSHOT_ENTRY_MAX_AGE = 30 * 86400
|
||||
|
||||
|
||||
class PluginResourceMonitor:
|
||||
@@ -140,10 +155,15 @@ class PluginResourceMonitor:
|
||||
self._metrics: Dict[str, ResourceMetrics] = {}
|
||||
self._limits: Dict[str, ResourceLimits] = {}
|
||||
self._bad_limits_warned: set = set()
|
||||
# When each plugin's metrics last reached the cache. Metrics change on
|
||||
# every call, so they cannot be de-duplicated the way health state can;
|
||||
# they are rate-limited instead. See _METRICS_PERSIST_INTERVAL.
|
||||
self._metrics_persisted_at: Dict[str, float] = {}
|
||||
# When the metrics snapshot last reached the cache (monotonic), None
|
||||
# until it has. Metrics change on every call, so they cannot be
|
||||
# de-duplicated the way health state can; they are rate-limited
|
||||
# instead. See _METRICS_PERSIST_INTERVAL.
|
||||
self._snapshot_persisted_at: Optional[float] = None
|
||||
# Plugins whose metrics this process recorded since the last snapshot
|
||||
# write: only their entries are overwritten, the rest are kept as
|
||||
# found on disk.
|
||||
self._metrics_dirty: Set[str] = set()
|
||||
|
||||
# Lock for thread-safe access
|
||||
self._lock = threading.Lock()
|
||||
@@ -247,10 +267,14 @@ class PluginResourceMonitor:
|
||||
with self._lock:
|
||||
if force_reload or plugin_id not in self._metrics:
|
||||
# Try to load from cache
|
||||
cache_key = self._get_metrics_key(plugin_id)
|
||||
cached = self.cache_manager.get(
|
||||
cache_key, max_age=None, memory_ttl=0 if force_reload else None
|
||||
)
|
||||
memory_ttl = 0 if force_reload else None
|
||||
cached = self._read_snapshot(memory_ttl).get(plugin_id)
|
||||
if cached is None:
|
||||
# Not in the snapshot: the per-plugin record an older
|
||||
# version wrote, if there is one.
|
||||
cached = self.cache_manager.get(
|
||||
self._get_metrics_key(plugin_id), max_age=None,
|
||||
memory_ttl=memory_ttl)
|
||||
if cached:
|
||||
metrics = self._metrics_from_cache(plugin_id, cached)
|
||||
else:
|
||||
@@ -498,12 +522,70 @@ class PluginResourceMonitor:
|
||||
summaries[plugin_id] = self.get_metrics_summary(plugin_id)
|
||||
return summaries
|
||||
|
||||
def _persist_metrics(self, plugin_id: str, metrics: ResourceMetrics,
|
||||
force: bool = False) -> None:
|
||||
"""Write a plugin's metrics to the cache, at most once per interval.
|
||||
def _read_snapshot(self, memory_ttl: Optional[int] = None) -> Dict[str, Any]:
|
||||
"""The snapshot's per-plugin records, or {} if there is none usable.
|
||||
|
||||
Caller must hold ``self._lock``.
|
||||
"""
|
||||
cached = self.cache_manager.get(
|
||||
METRICS_SNAPSHOT_KEY, max_age=None, memory_ttl=memory_ttl)
|
||||
if not isinstance(cached, dict) or cached.get('schema') != _METRICS_SNAPSHOT_SCHEMA:
|
||||
return {}
|
||||
plugins = cached.get('plugins')
|
||||
return plugins if isinstance(plugins, dict) else {}
|
||||
|
||||
@staticmethod
|
||||
def _metrics_record(metrics: ResourceMetrics) -> Dict[str, Any]:
|
||||
"""One plugin's entry in the snapshot."""
|
||||
return {
|
||||
'memory_mb': metrics.memory_mb,
|
||||
'cpu_percent': metrics.cpu_percent,
|
||||
'execution_time': metrics.execution_time,
|
||||
'call_count': metrics.call_count,
|
||||
'total_execution_time': metrics.total_execution_time,
|
||||
'max_execution_time': metrics.max_execution_time,
|
||||
'min_execution_time': (metrics.min_execution_time
|
||||
if metrics.min_execution_time != float('inf')
|
||||
else 0.0),
|
||||
'last_update_time': metrics.last_update_time,
|
||||
}
|
||||
|
||||
def _write_snapshot(self, drop: Optional[str] = None) -> None:
|
||||
"""Write the snapshot: what is on disk, with this process's recorded
|
||||
plugins updated and ``drop`` removed.
|
||||
|
||||
Starting from the disk copy rather than from memory keeps the entries
|
||||
of plugins this process has not run -- disabled ones, which the web UI
|
||||
still shows -- and a reset made from the other process.
|
||||
|
||||
Caller must hold ``self._lock``.
|
||||
"""
|
||||
plugins = dict(self._read_snapshot(memory_ttl=0))
|
||||
if drop is not None:
|
||||
plugins.pop(drop, None)
|
||||
for plugin_id in self._metrics_dirty:
|
||||
if plugin_id in self._metrics:
|
||||
plugins[plugin_id] = self._metrics_record(self._metrics[plugin_id])
|
||||
cutoff = time.time() - _METRICS_SNAPSHOT_ENTRY_MAX_AGE
|
||||
for plugin_id, record in list(plugins.items()):
|
||||
last = record.get('last_update_time') if isinstance(record, dict) else None
|
||||
if isinstance(last, (int, float)) and last < cutoff:
|
||||
del plugins[plugin_id]
|
||||
self.cache_manager.set(METRICS_SNAPSHOT_KEY, {
|
||||
'schema': _METRICS_SNAPSHOT_SCHEMA,
|
||||
'plugins': plugins,
|
||||
})
|
||||
# Only once the write has landed, so a failed one is retried in full.
|
||||
self._metrics_dirty.clear()
|
||||
|
||||
def _persist_metrics(self, plugin_id: str, metrics: ResourceMetrics,
|
||||
force: bool = False) -> None:
|
||||
"""Record that a plugin's metrics changed, and write the snapshot if
|
||||
the last write is at least an interval old.
|
||||
|
||||
Caller must hold ``self._lock``.
|
||||
"""
|
||||
self._metrics_dirty.add(plugin_id)
|
||||
# Monotonic, not wall clock: these devices have no RTC, so the clock
|
||||
# jumps by however far off boot-time was the moment NTP first syncs.
|
||||
# A forward jump would allow an early write, a backward one would
|
||||
@@ -515,35 +597,26 @@ class PluginResourceMonitor:
|
||||
# single run -- the throttle swallowed the very first snapshot, which
|
||||
# is the one that matters most after a restart.
|
||||
now = time.monotonic()
|
||||
last_written = self._metrics_persisted_at.get(plugin_id)
|
||||
last_written = self._snapshot_persisted_at
|
||||
if (not force and last_written is not None
|
||||
and now - last_written < _METRICS_PERSIST_INTERVAL):
|
||||
return
|
||||
cache_key = self._get_metrics_key(plugin_id)
|
||||
self.cache_manager.set(cache_key, {
|
||||
'memory_mb': metrics.memory_mb,
|
||||
'cpu_percent': metrics.cpu_percent,
|
||||
'execution_time': metrics.execution_time,
|
||||
'call_count': metrics.call_count,
|
||||
'total_execution_time': metrics.total_execution_time,
|
||||
'max_execution_time': metrics.max_execution_time,
|
||||
'min_execution_time': (metrics.min_execution_time
|
||||
if metrics.min_execution_time != float('inf')
|
||||
else 0.0),
|
||||
'last_update_time': metrics.last_update_time,
|
||||
})
|
||||
self._write_snapshot()
|
||||
# Only after the write lands. Marking it first would mean a failed
|
||||
# set() bought the next interval's silence without leaving a snapshot.
|
||||
self._metrics_persisted_at[plugin_id] = now
|
||||
self._snapshot_persisted_at = now
|
||||
|
||||
def reset_metrics(self, plugin_id: str) -> None:
|
||||
"""Reset metrics for a plugin."""
|
||||
with self._lock:
|
||||
if plugin_id in self._metrics:
|
||||
self._metrics[plugin_id] = ResourceMetrics()
|
||||
cache_key = self._get_metrics_key(plugin_id)
|
||||
self.cache_manager.delete(cache_key)
|
||||
self._metrics_dirty.discard(plugin_id)
|
||||
self._write_snapshot(drop=plugin_id)
|
||||
# The record an older version wrote, so the reader's fallback
|
||||
# cannot bring the old numbers back.
|
||||
self.cache_manager.delete(self._get_metrics_key(plugin_id))
|
||||
# Let the next call persist immediately rather than leaving the
|
||||
# deleted key absent for the rest of the interval.
|
||||
self._metrics_persisted_at.pop(plugin_id, None)
|
||||
# plugin absent from the snapshot for the rest of the interval.
|
||||
self._snapshot_persisted_at = None
|
||||
|
||||
|
||||
Reference in New Issue
Block a user