mirror of
https://github.com/ChuckBuilds/LEDMatrix.git
synced 2026-10-06 23:35:08 +00:00
fix(plugins): one hung plugin no longer stops every plugin from updating
The single update worker took each plugin's lock with a blocking acquire(), and the render thread holds that lock while it runs the plugin's display(). A display() that never returned -- or a first frame still running on the executor's lingering thread after its 30s timeout -- parked the worker for good, and no plugin updated again. - The worker waits at most PLUGIN_LOCK_TIMEOUT (5s, the bound unload_plugin already uses), skips the busy plugin and records it through the normal update-failure path as a hang (PluginBusyError), so repeats open its circuit breaker. Log lines about it are rate-limited per plugin. - display() is timed on every frame (two monotonic reads). Calls of 2s or more are logged once a minute and counted in plugin health (slow_call_count, last_slow_call); calls past the executor timeout, and a first frame still running at it, are recorded as hangs (hang_count, last_hang) and no longer as successes. An update() still running after its timeout is recorded as a hang too. - on_config_change() runs under the plugin's lock via PluginManager.apply_config_change(); if the lock stays busy the latest change is deferred to the update worker, applied as soon as the lock frees and before the plugin's next update() at the latest. The plugin API is unchanged. Which thread runs each hook is documented in PluginManager.__init__. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -19,8 +19,23 @@ class PluginTimeoutError(Exception):
|
||||
"""Raised when a plugin operation times out."""
|
||||
|
||||
|
||||
class PluginBusyError(PluginTimeoutError):
|
||||
"""A plugin's lock stayed held past its bound.
|
||||
|
||||
Not raised; recorded. The lock is held by the plugin's own display(),
|
||||
update() or on_config_change() -- one that is hung or far slower than it
|
||||
should be -- so the caller skipped the plugin rather than wait on it.
|
||||
"""
|
||||
|
||||
|
||||
class PluginExecutor:
|
||||
"""Handles plugin execution with timeout and error isolation."""
|
||||
|
||||
#: A display() call at least this long is logged and counted as slow.
|
||||
#: A frame is milliseconds; two seconds is a plugin doing I/O in display().
|
||||
SLOW_DISPLAY_SECONDS = 2.0
|
||||
#: An update() call at least this long is logged as slow.
|
||||
SLOW_UPDATE_SECONDS = 5.0
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
@@ -117,15 +132,15 @@ class PluginExecutor:
|
||||
True if update succeeded, False otherwise
|
||||
"""
|
||||
try:
|
||||
start_time = time.time()
|
||||
start_time = time.monotonic()
|
||||
self.execute_with_timeout(
|
||||
lambda: plugin.update(),
|
||||
timeout=timeout,
|
||||
plugin_id=plugin_id
|
||||
)
|
||||
duration = time.time() - start_time
|
||||
duration = time.monotonic() - start_time
|
||||
|
||||
if duration > 5.0: # Warn if update takes more than 5 seconds
|
||||
if duration > self.SLOW_UPDATE_SECONDS:
|
||||
self.logger.warning(
|
||||
"Plugin %s update() took %.2fs (consider optimizing)",
|
||||
plugin_id,
|
||||
@@ -175,7 +190,7 @@ class PluginExecutor:
|
||||
True if display succeeded, False otherwise
|
||||
"""
|
||||
try:
|
||||
start_time = time.time()
|
||||
start_time = time.monotonic()
|
||||
|
||||
# Does display() take a display_mode keyword? The caller usually
|
||||
# knows and caches the answer, so prefer what it passed.
|
||||
@@ -206,9 +221,9 @@ class PluginExecutor:
|
||||
plugin_id=plugin_id
|
||||
)
|
||||
|
||||
duration = time.time() - start_time
|
||||
duration = time.monotonic() - start_time
|
||||
|
||||
if duration > 2.0: # Warn if display takes more than 2 seconds
|
||||
if duration > self.SLOW_DISPLAY_SECONDS:
|
||||
self.logger.warning(
|
||||
"Plugin %s display() took %.2fs (consider optimizing)",
|
||||
plugin_id,
|
||||
|
||||
@@ -254,6 +254,66 @@ class PluginHealthTracker:
|
||||
|
||||
self._save_health_state(plugin_id, state)
|
||||
|
||||
def record_hang(self, plugin_id: str, operation: str, seconds: float,
|
||||
error: Optional[Exception] = None) -> None:
|
||||
"""Record a call that ran past its limit, or held the plugin's lock past it.
|
||||
|
||||
Counts as a failure, so the ordinary circuit breaker handles a plugin
|
||||
that keeps hanging: after ``failure_threshold`` in a row it is skipped
|
||||
by both the update scheduler and the display rotation until the
|
||||
cooldown ends. The hang itself is kept alongside (``hang_count``,
|
||||
``last_hang``) so the health API can tell "hung" from "raised".
|
||||
|
||||
Args:
|
||||
plugin_id: Plugin identifier
|
||||
operation: What hung: ``"display"``, ``"update"``, or the lock
|
||||
wait that found one of them still running.
|
||||
seconds: How long it had been running, or how long the lock
|
||||
was waited on, when this was recorded.
|
||||
error: The error to store as ``last_error``; one is built from
|
||||
the other arguments when omitted.
|
||||
"""
|
||||
state = self.get_health_state(plugin_id)
|
||||
count = state.get('hang_count')
|
||||
state['hang_count'] = (count if isinstance(count, int) and not isinstance(count, bool)
|
||||
else 0) + 1
|
||||
state['last_hang'] = {
|
||||
'operation': operation,
|
||||
'seconds': round(float(seconds), 3),
|
||||
'time': time.time(),
|
||||
}
|
||||
if error is None:
|
||||
error = TimeoutError(f"{operation} still running after {seconds:.1f}s")
|
||||
# record_failure saves the record, hang fields included.
|
||||
self.record_failure(plugin_id, error)
|
||||
|
||||
#: Minimum seconds between persisting a plugin's slow-call counters. The
|
||||
#: in-memory record is updated on every slow call; a plugin that is slow
|
||||
#: on every frame must not become an SD-card write per frame.
|
||||
SLOW_CALL_PERSIST_INTERVAL = 60.0
|
||||
|
||||
def record_slow_call(self, plugin_id: str, operation: str, seconds: float) -> None:
|
||||
"""Note a call that finished, but slowly. Reporting only.
|
||||
|
||||
Unlike :meth:`record_hang` this never touches the circuit breaker: a
|
||||
slow display() still drew its frame.
|
||||
"""
|
||||
state = self.get_health_state(plugin_id)
|
||||
count = state.get('slow_call_count')
|
||||
state['slow_call_count'] = (count if isinstance(count, int) and not isinstance(count, bool)
|
||||
else 0) + 1
|
||||
now = time.time()
|
||||
state['last_slow_call'] = {
|
||||
'operation': operation,
|
||||
'seconds': round(float(seconds), 3),
|
||||
'time': now,
|
||||
}
|
||||
saved_at = self.__dict__.setdefault('_slow_call_saved_at', {})
|
||||
last = saved_at.get(plugin_id)
|
||||
if last is None or now - last >= self.SLOW_CALL_PERSIST_INTERVAL:
|
||||
saved_at[plugin_id] = now
|
||||
self._save_health_state(plugin_id, state)
|
||||
|
||||
def set_degraded(self, plugin_id: str, reason: Optional[str]) -> None:
|
||||
"""Flag (or clear) a plugin as degraded without touching the circuit breaker.
|
||||
|
||||
@@ -345,7 +405,11 @@ class PluginHealthTracker:
|
||||
'degraded': state.get('degraded', False),
|
||||
'degraded_reason': state.get('degraded_reason'),
|
||||
'circuit_opened_time': state.get('circuit_opened_time'),
|
||||
'half_open_start_time': state.get('half_open_start_time')
|
||||
'half_open_start_time': state.get('half_open_start_time'),
|
||||
'hang_count': state.get('hang_count', 0),
|
||||
'last_hang': state.get('last_hang'),
|
||||
'slow_call_count': state.get('slow_call_count', 0),
|
||||
'last_slow_call': state.get('last_slow_call'),
|
||||
}
|
||||
|
||||
def get_all_health_summaries(self) -> Dict[str, Dict[str, Any]]:
|
||||
|
||||
@@ -16,12 +16,14 @@ import time
|
||||
import threading
|
||||
import types
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Optional, Any, Tuple
|
||||
from typing import Dict, List, NamedTuple, Optional, Any, Tuple, Union
|
||||
import logging
|
||||
from src.exceptions import PluginError, ConfigError
|
||||
from src.logging_config import get_logger
|
||||
from src.plugin_system.plugin_loader import PluginLoader
|
||||
from src.plugin_system.plugin_executor import PluginExecutor
|
||||
from src.plugin_system.plugin_executor import (
|
||||
PluginBusyError, PluginExecutor, PluginTimeoutError,
|
||||
)
|
||||
from src.plugin_system.plugin_state import PluginStateManager, PluginState
|
||||
from src.plugin_system.schema_manager import (
|
||||
CORE_VEGAS_TUNING_KEYS, SchemaManager, normalize_legacy_booleans,
|
||||
@@ -36,6 +38,16 @@ from src.common.permission_utils import (
|
||||
)
|
||||
|
||||
|
||||
class _DeferredConfigChange(NamedTuple):
|
||||
"""Update-queue item: apply the config change parked for ``plugin_id``.
|
||||
|
||||
Queued by apply_config_change() when the plugin's lock was busy; the
|
||||
change itself waits in ``PluginManager._deferred_config_changes`` so only
|
||||
the latest one is ever applied.
|
||||
"""
|
||||
plugin_id: str
|
||||
|
||||
|
||||
class PluginManager:
|
||||
"""
|
||||
Manages plugin discovery, loading, and lifecycle.
|
||||
@@ -56,6 +68,19 @@ class PluginManager:
|
||||
# How long unload_plugin() waits for an in-flight update() to finish
|
||||
# before tearing the instance down anyway.
|
||||
UNLOAD_LOCK_TIMEOUT = 5.0
|
||||
|
||||
# How long the update worker and apply_config_change() wait for a
|
||||
# plugin's lock -- the same bound unload already uses for the same lock.
|
||||
# A display() frame holds it for milliseconds, so this only runs out when
|
||||
# the holder is hung or pathologically slow. The worker then skips that
|
||||
# plugin (recorded as a hang, so repeats open its circuit breaker)
|
||||
# instead of stalling every other plugin's update behind it.
|
||||
PLUGIN_LOCK_TIMEOUT = UNLOAD_LOCK_TIMEOUT
|
||||
|
||||
# Minimum seconds between repeats of the same hang/slow-call warning for
|
||||
# one plugin. A hung plugin is re-detected every interval; a slow
|
||||
# display() can be re-detected every frame.
|
||||
HANG_LOG_INTERVAL = 60.0
|
||||
|
||||
def __init__(self, plugins_dir: str = "plugins",
|
||||
config_manager: Optional[Any] = None,
|
||||
@@ -122,7 +147,33 @@ class PluginManager:
|
||||
# post-timeout window.
|
||||
# Kill switch: plugin_system.synchronous_updates: true restores the
|
||||
# inline path.
|
||||
self._update_queue: "queue.Queue[Optional[Tuple[str, float]]]" = queue.Queue()
|
||||
#
|
||||
# Which thread runs each plugin hook, and what it holds:
|
||||
# __init__, on_enable the loading thread (main thread at startup,
|
||||
# the render thread on a live enable).
|
||||
# update() plugin-update-worker, under the plugin lock,
|
||||
# via PluginExecutor (whose daemon thread runs
|
||||
# the call; if it outlives the executor's
|
||||
# timeout it keeps the lock until it returns).
|
||||
# Exceptions: the startup pass
|
||||
# (DisplayController._run_initial_updates, main
|
||||
# thread, before the display loop starts) and
|
||||
# the synchronous_updates kill switch (render
|
||||
# thread) run it without the lock.
|
||||
# display() the render thread, under a try-lock: a busy
|
||||
# lock skips the frame. The first frame of a
|
||||
# screen goes through PluginExecutor. Vegas
|
||||
# mode's adapter and coordinator take the lock
|
||||
# with a bounded wait.
|
||||
# on_config_change() ConfigService's watcher thread, under the
|
||||
# plugin lock via apply_config_change(); if the
|
||||
# lock stays busy it is deferred to the update
|
||||
# worker, which applies it under the lock.
|
||||
# cleanup(), on_disable() whoever calls unload_plugin(), under the
|
||||
# lock with UNLOAD_LOCK_TIMEOUT.
|
||||
# No wait on a plugin lock is unbounded, so one hung plugin can only
|
||||
# cost the worker PLUGIN_LOCK_TIMEOUT per attempt.
|
||||
self._update_queue: "queue.Queue[Union[None, Tuple[str, float], _DeferredConfigChange]]" = queue.Queue()
|
||||
self._pending_updates: set = set()
|
||||
self._pending_lock = threading.Lock()
|
||||
# Serializes the "is this plugin eligible?" -> "claim it (RUNNING)"
|
||||
@@ -145,6 +196,12 @@ class PluginManager:
|
||||
# run_scheduled_updates_with_changes().
|
||||
self._completed_updates: set = set()
|
||||
self._completed_updates_lock = threading.Lock()
|
||||
# Config changes that found the plugin's lock busy, latest per plugin,
|
||||
# with the instance they were meant for. See apply_config_change().
|
||||
self._deferred_config_changes: Dict[str, Tuple[Any, Dict[str, Any]]] = {}
|
||||
self._deferred_config_lock = threading.Lock()
|
||||
# key -> (monotonic time last logged, repeats suppressed since)
|
||||
self._rate_limited_warnings: Dict[str, Tuple[float, int]] = {}
|
||||
self._synchronous_updates = False
|
||||
if self.config_manager is not None:
|
||||
try:
|
||||
@@ -676,6 +733,8 @@ class PluginManager:
|
||||
|
||||
# Remove from active plugins
|
||||
del self.plugins[plugin_id]
|
||||
with self._deferred_config_lock:
|
||||
self._deferred_config_changes.pop(plugin_id, None)
|
||||
with self._plugin_last_update_lock:
|
||||
self.plugin_last_update.pop(plugin_id, None)
|
||||
self._update_interval_cache.pop(plugin_id, None)
|
||||
@@ -1012,6 +1071,8 @@ class PluginManager:
|
||||
self,
|
||||
plugin_id: str,
|
||||
exc: Optional[Exception] = None,
|
||||
hang: Optional[Tuple[str, float]] = None,
|
||||
log: bool = True,
|
||||
) -> None:
|
||||
"""Apply the standard failure-recovery path for a plugin update.
|
||||
|
||||
@@ -1025,6 +1086,10 @@ class PluginManager:
|
||||
exc: The exception that caused the failure, if any. When None a
|
||||
synthetic ExecutionFailure exception is constructed from the
|
||||
timeout/executor-error path.
|
||||
hang: ``(operation, seconds)`` when the failure is a hang rather
|
||||
than an error, so health records it as one.
|
||||
log: Log the generic failure line. Callers that already logged
|
||||
something more specific (rate-limited) pass False.
|
||||
"""
|
||||
failure_time = time.time()
|
||||
if exc is not None:
|
||||
@@ -1040,13 +1105,98 @@ class PluginManager:
|
||||
'timestamp': failure_time,
|
||||
'recoverable': True,
|
||||
}
|
||||
self.logger.warning("Plugin %s update() failed; will retry after interval", plugin_id)
|
||||
if log:
|
||||
self.logger.warning("Plugin %s update() failed; will retry after interval", plugin_id)
|
||||
with self._plugin_last_update_lock:
|
||||
self.plugin_last_update[plugin_id] = failure_time
|
||||
self.state_manager.set_state_with_error(plugin_id, PluginState.ENABLED, error_info)
|
||||
if self.health_tracker:
|
||||
if hang is not None:
|
||||
self._record_hang(plugin_id, hang[0], hang[1], err)
|
||||
elif self.health_tracker:
|
||||
self.health_tracker.record_failure(plugin_id, err)
|
||||
|
||||
def _warn_rate_limited(self, key: str, message: str, *args: Any) -> None:
|
||||
"""Log a warning at most once per HANG_LOG_INTERVAL for ``key``.
|
||||
|
||||
Repeats in between are counted and the count is appended to the next
|
||||
one that is logged, so the journal shows the problem continuing
|
||||
without a line per frame or per scheduler tick.
|
||||
"""
|
||||
# setdefault: tests build bare managers with PluginManager.__new__.
|
||||
seen = self.__dict__.setdefault('_rate_limited_warnings', {})
|
||||
now = time.monotonic()
|
||||
last, suppressed = seen.get(key, (None, 0))
|
||||
if last is not None and now - last < self.HANG_LOG_INTERVAL:
|
||||
seen[key] = (last, suppressed + 1)
|
||||
return
|
||||
seen[key] = (now, 0)
|
||||
if suppressed:
|
||||
message += " (%d more since the last warning)"
|
||||
args = args + (suppressed,)
|
||||
self.logger.warning(message, *args)
|
||||
|
||||
def _record_hang(self, plugin_id: str, operation: str, seconds: float,
|
||||
err: Exception) -> None:
|
||||
"""Record a hang in plugin health: a failure to the circuit breaker.
|
||||
|
||||
Goes through PluginHealthTracker.record_hang when the tracker has it
|
||||
(it also counts the hang separately), else plain record_failure. Never
|
||||
raises: this runs on the update worker and the render thread.
|
||||
"""
|
||||
tracker = self.health_tracker
|
||||
if tracker is None:
|
||||
return
|
||||
try:
|
||||
record_hang = getattr(tracker, 'record_hang', None)
|
||||
if callable(record_hang):
|
||||
record_hang(plugin_id, operation, seconds, err)
|
||||
else:
|
||||
tracker.record_failure(plugin_id, err)
|
||||
except Exception as e: # pylint: disable=broad-except
|
||||
self.logger.debug("Could not record hang for %s: %s", plugin_id, e)
|
||||
|
||||
def note_display_duration(self, plugin_id: str, seconds: float) -> None:
|
||||
"""Account for one display() call that took ``seconds``.
|
||||
|
||||
Called by the render loop for every frame, so the common case is one
|
||||
comparison. At or above PluginExecutor.SLOW_DISPLAY_SECONDS the call
|
||||
is logged (rate-limited) and counted as slow in plugin health; at or
|
||||
above the executor's timeout -- the limit the first frame of a screen
|
||||
is already held to -- it is recorded as a hang, which the circuit
|
||||
breaker counts as a failure.
|
||||
"""
|
||||
if seconds < PluginExecutor.SLOW_DISPLAY_SECONDS:
|
||||
return
|
||||
if seconds >= self.plugin_executor.default_timeout:
|
||||
self.record_display_hang(plugin_id, seconds)
|
||||
return
|
||||
self._warn_rate_limited(
|
||||
"slow-display:" + plugin_id,
|
||||
"Plugin %s display() took %.2fs; a frame should take milliseconds "
|
||||
"(is it fetching or loading files in display()?)", plugin_id, seconds)
|
||||
tracker = self.health_tracker
|
||||
record_slow = getattr(tracker, 'record_slow_call', None) if tracker is not None else None
|
||||
if callable(record_slow):
|
||||
try:
|
||||
record_slow(plugin_id, 'display', seconds)
|
||||
except Exception as e: # pylint: disable=broad-except
|
||||
self.logger.debug("Could not record slow display for %s: %s", plugin_id, e)
|
||||
|
||||
def record_display_hang(self, plugin_id: str, seconds: float) -> None:
|
||||
"""Record a display() call that ran ``seconds``, past its limit.
|
||||
|
||||
Either it has since returned (note_display_duration) or it is still
|
||||
running on the executor's lingering thread, holding the plugin's lock
|
||||
(the render loop's first-frame dispatch).
|
||||
"""
|
||||
self._warn_rate_limited(
|
||||
"hung-display:" + plugin_id,
|
||||
"Plugin %s display() ran for at least %.1fs (limit %.0fs); recorded "
|
||||
"as a hang -- repeated hangs open its circuit breaker",
|
||||
plugin_id, seconds, self.plugin_executor.default_timeout)
|
||||
self._record_hang(plugin_id, 'display', seconds, PluginTimeoutError(
|
||||
f"Plugin {plugin_id} display() ran for at least {seconds:.1f}s"))
|
||||
|
||||
def run_scheduled_updates(self, current_time: Optional[float] = None) -> None:
|
||||
"""
|
||||
Trigger plugin updates based on their defined update intervals.
|
||||
@@ -1132,11 +1282,13 @@ class PluginManager:
|
||||
self.state_manager.set_state(plugin_id, PluginState.ENABLED)
|
||||
|
||||
def get_plugin_lock(self, plugin_id: str) -> threading.Lock:
|
||||
"""Per-plugin lock keeping update() and display() mutually exclusive.
|
||||
"""Per-plugin lock keeping update(), display() and on_config_change()
|
||||
mutually exclusive.
|
||||
|
||||
The update worker holds it for the duration of a plugin's update();
|
||||
the display side acquires it non-blocking and skips that frame's
|
||||
display() call when the plugin is mid-update.
|
||||
display() call when the plugin is mid-update. Every other waiter uses
|
||||
a bounded acquire (see the thread notes in __init__).
|
||||
"""
|
||||
with self._plugin_locks_guard:
|
||||
lock = self._plugin_locks.get(plugin_id)
|
||||
@@ -1202,14 +1354,26 @@ class PluginManager:
|
||||
real update() call genuinely finishes (see _execute_update_now),
|
||||
which can be after this dispatch returns if PluginExecutor's own
|
||||
timeout elapses first.
|
||||
|
||||
The lock wait is bounded by PLUGIN_LOCK_TIMEOUT. Whatever holds it
|
||||
past that -- a hung display() on the render thread or a lingering
|
||||
executor thread -- costs this worker that long once per attempt, and
|
||||
the plugin is skipped and recorded as hung (_skip_busy_update); the
|
||||
other plugins' queued updates carry on.
|
||||
"""
|
||||
while True:
|
||||
item = self._update_queue.get()
|
||||
if item is None: # shutdown sentinel
|
||||
return
|
||||
if isinstance(item, _DeferredConfigChange):
|
||||
self._apply_deferred_config_change(item.plugin_id)
|
||||
continue
|
||||
plugin_id, scheduled_time = item
|
||||
lock = self.get_plugin_lock(plugin_id)
|
||||
lock.acquire()
|
||||
wait_start = time.monotonic()
|
||||
if not lock.acquire(timeout=self.PLUGIN_LOCK_TIMEOUT):
|
||||
self._skip_busy_update(plugin_id, time.monotonic() - wait_start)
|
||||
continue
|
||||
plugin_instance = self.plugins.get(plugin_id)
|
||||
if plugin_instance is None: # unloaded while queued; its
|
||||
# lifecycle state was already cleared by unload_plugin —
|
||||
@@ -1218,6 +1382,9 @@ class PluginManager:
|
||||
with self._pending_lock:
|
||||
self._pending_updates.discard(plugin_id)
|
||||
continue
|
||||
# A config change that found the lock busy goes in first, so
|
||||
# this update() runs against the settings the user saved.
|
||||
self._apply_deferred_config_locked(plugin_id, plugin_instance)
|
||||
try:
|
||||
self._execute_update_now(plugin_id, plugin_instance,
|
||||
scheduled_time, lock=lock)
|
||||
@@ -1228,6 +1395,129 @@ class PluginManager:
|
||||
self.logger.exception("update worker: unexpected error for %s",
|
||||
plugin_id)
|
||||
|
||||
def _skip_busy_update(self, plugin_id: str, waited: float) -> None:
|
||||
"""Give up on a queued update whose plugin lock stayed held.
|
||||
|
||||
Same bookkeeping as a failed update() -- pending slot dropped before
|
||||
the state returns to ENABLED, last-update stamped so the retry waits
|
||||
a full interval -- recorded as a hang, so a plugin stuck this way
|
||||
opens its circuit breaker after the usual number of attempts and
|
||||
stops being scheduled or displayed until the cooldown.
|
||||
"""
|
||||
with self._pending_lock:
|
||||
self._pending_updates.discard(plugin_id)
|
||||
if plugin_id not in self.plugins:
|
||||
# Unloaded while we waited: its lifecycle state is already
|
||||
# cleared; recording a failure would resurrect it as ENABLED.
|
||||
return
|
||||
self._warn_rate_limited(
|
||||
"busy-update:" + plugin_id,
|
||||
"Plugin %s update skipped: its lock was still held after %.1fs "
|
||||
"(a display() or update() of it is hung or very slow); other "
|
||||
"plugins keep updating", plugin_id, waited)
|
||||
self._record_update_failure(
|
||||
plugin_id,
|
||||
exc=PluginBusyError(
|
||||
f"Plugin {plugin_id} busy: its lock was held for over {waited:.1f}s "
|
||||
"by a hung or slow display()/update(); update skipped"),
|
||||
hang=('update lock wait', waited),
|
||||
log=False)
|
||||
|
||||
def apply_config_change(self, plugin_id: str, new_config: Dict[str, Any],
|
||||
plugin_instance: Optional[Any] = None) -> bool:
|
||||
"""Call ``on_config_change(new_config)`` without racing update()/display().
|
||||
|
||||
Runs on the calling thread -- ConfigService's watcher, for the display
|
||||
service -- holding the plugin's lock, waited on for at most
|
||||
PLUGIN_LOCK_TIMEOUT. If the lock is still busy (an update() mid-fetch
|
||||
can outlast that) the change is parked and handed to the update
|
||||
worker, which applies it under the same lock once it is free, and at
|
||||
the latest just before the plugin's next update(). A later change for
|
||||
the same plugin replaces a parked one.
|
||||
|
||||
Exceptions from on_config_change propagate on the immediate path,
|
||||
as they did when the caller invoked it directly.
|
||||
|
||||
Args:
|
||||
plugin_id: Plugin identifier.
|
||||
new_config: The prepared config to hand the plugin.
|
||||
plugin_instance: The instance to notify; defaults to the loaded one.
|
||||
|
||||
Returns:
|
||||
True if on_config_change ran now, False if it was deferred or there
|
||||
is no loaded plugin to notify.
|
||||
"""
|
||||
if plugin_instance is None:
|
||||
plugin_instance = self.plugins.get(plugin_id)
|
||||
if plugin_instance is None or not hasattr(plugin_instance, 'on_config_change'):
|
||||
return False
|
||||
lock = self.get_plugin_lock(plugin_id)
|
||||
if lock.acquire(timeout=self.PLUGIN_LOCK_TIMEOUT):
|
||||
try:
|
||||
with self._deferred_config_lock:
|
||||
# This change supersedes any older one still parked.
|
||||
self._deferred_config_changes.pop(plugin_id, None)
|
||||
plugin_instance.on_config_change(new_config)
|
||||
finally:
|
||||
lock.release()
|
||||
return True
|
||||
|
||||
with self._deferred_config_lock:
|
||||
self._deferred_config_changes[plugin_id] = (plugin_instance, new_config)
|
||||
self._warn_rate_limited(
|
||||
"busy-config:" + plugin_id,
|
||||
"Plugin %s is busy (lock held for over %.1fs); its config change "
|
||||
"will be applied by the update worker once it is free",
|
||||
plugin_id, self.PLUGIN_LOCK_TIMEOUT)
|
||||
try:
|
||||
self._ensure_update_worker()
|
||||
self._update_queue.put(_DeferredConfigChange(plugin_id))
|
||||
except Exception as exc: # pylint: disable=broad-except
|
||||
# No worker (thread start refused): still parked, so the next
|
||||
# update() of this plugin applies it.
|
||||
self.logger.error(
|
||||
"Could not queue the config change for plugin %s (%s: %s); it "
|
||||
"will be applied before its next update()",
|
||||
plugin_id, type(exc).__name__, exc)
|
||||
return False
|
||||
|
||||
def _apply_deferred_config_change(self, plugin_id: str) -> None:
|
||||
"""Worker side of a parked config change: take the lock, apply it."""
|
||||
with self._deferred_config_lock:
|
||||
if plugin_id not in self._deferred_config_changes:
|
||||
return # applied or superseded meanwhile
|
||||
lock = self.get_plugin_lock(plugin_id)
|
||||
wait_start = time.monotonic()
|
||||
if not lock.acquire(timeout=self.PLUGIN_LOCK_TIMEOUT):
|
||||
self._warn_rate_limited(
|
||||
"busy-config:" + plugin_id,
|
||||
"Plugin %s still busy after %.1fs; its config change stays "
|
||||
"parked until its next update()",
|
||||
plugin_id, time.monotonic() - wait_start)
|
||||
return
|
||||
try:
|
||||
self._apply_deferred_config_locked(plugin_id, self.plugins.get(plugin_id))
|
||||
finally:
|
||||
lock.release()
|
||||
|
||||
def _apply_deferred_config_locked(self, plugin_id: str,
|
||||
current_instance: Optional[Any]) -> None:
|
||||
"""Apply the parked config change for plugin_id; caller holds its lock."""
|
||||
with self._deferred_config_lock:
|
||||
entry = self._deferred_config_changes.pop(plugin_id, None)
|
||||
if entry is None:
|
||||
return
|
||||
instance, new_config = entry
|
||||
if current_instance is None or instance is not current_instance:
|
||||
# Unloaded, or reloaded as a new instance built from the current
|
||||
# config: nothing left to tell.
|
||||
return
|
||||
try:
|
||||
instance.on_config_change(new_config)
|
||||
self.logger.info("Applied deferred config change for plugin %s", plugin_id)
|
||||
except Exception: # pylint: disable=broad-except
|
||||
self.logger.exception("Error in plugin %s config change handler", plugin_id)
|
||||
|
||||
def stop_update_worker(self, timeout: float = 5.0) -> None:
|
||||
"""Signal the worker to exit (used by cleanup; thread is a daemon)."""
|
||||
if self._update_worker is not None and self._update_worker.is_alive():
|
||||
@@ -1338,14 +1628,29 @@ class PluginManager:
|
||||
else:
|
||||
_finish(True)
|
||||
|
||||
started = time.monotonic()
|
||||
try:
|
||||
self.plugin_executor.execute_update(
|
||||
success = self.plugin_executor.execute_update(
|
||||
types.SimpleNamespace(update=_target_update), plugin_id)
|
||||
except Exception as exc: # pragma: no cover - defensive; execute_update
|
||||
# catches everything internally, but guarantee _finish still
|
||||
# runs (releasing the lock) if something unexpected slips through.
|
||||
self.logger.exception("Unexpected error dispatching update for %s: %s", plugin_id, exc)
|
||||
_finish(False, exc=exc)
|
||||
return
|
||||
if not success and not finished['done']:
|
||||
# The executor stopped waiting but update() is still running: it
|
||||
# keeps the lock and the RUNNING state until it returns (then
|
||||
# _finish records the outcome). Say so now, rather than leave the
|
||||
# plugin silently stuck; record_success on a late return clears it.
|
||||
elapsed = time.monotonic() - started
|
||||
self._warn_rate_limited(
|
||||
"hung-update:" + plugin_id,
|
||||
"Plugin %s update() still running after %.1fs; it keeps its "
|
||||
"lock until it returns, and is not rescheduled until then",
|
||||
plugin_id, elapsed)
|
||||
self._record_hang(plugin_id, 'update', elapsed, PluginTimeoutError(
|
||||
f"Plugin {plugin_id} update() still running after {elapsed:.1f}s"))
|
||||
|
||||
def run_scheduled_updates_with_changes(self, current_time: Optional[float] = None) -> List[str]:
|
||||
"""
|
||||
|
||||
Reference in New Issue
Block a user