fix(plugins): one hung plugin no longer stops every plugin from updating (#677)

The update worker no longer blocks forever on a plugin whose display()
never returns. It waits at most PLUGIN_LOCK_TIMEOUT (5s) for a plugin's
lock, then skips that plugin's update (a report-only "busy skip" in
health) and keeps updating every other plugin. display() frames are timed
(slow calls logged and counted; calls past the executor timeout recorded as
hangs), a hung update() is recorded, and on_config_change() now runs under
the plugin lock or is deferred to the worker. The plugin-facing API is
unchanged.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Chuck
2026-09-30 09:10:41 -04:00
committed by GitHub
co-authored by Claude Opus 5.5
parent e3c85cece6
commit c0d97e4867
10 changed files with 1077 additions and 28 deletions
+23 -6
View File
@@ -19,8 +19,25 @@ class PluginTimeoutError(Exception):
"""Raised when a plugin operation times out."""
class PluginBusyError(PluginTimeoutError):
"""A plugin's lock stayed held past its bound.
Not raised; recorded. The lock is held by the plugin's own display(),
update(), on_config_change() or a Vegas content render -- slow, or hung
-- so the caller skipped the plugin rather than wait on it. Report-only:
it is kept as the plugin's state error info and counted as a busy skip in
health, never as a failure, so it cannot open the circuit breaker.
"""
class PluginExecutor:
"""Handles plugin execution with timeout and error isolation."""
#: A display() call at least this long is logged and counted as slow.
#: A frame is milliseconds; two seconds is a plugin doing I/O in display().
SLOW_DISPLAY_SECONDS = 2.0
#: An update() call at least this long is logged as slow.
SLOW_UPDATE_SECONDS = 5.0
def __init__(
self,
@@ -117,15 +134,15 @@ class PluginExecutor:
True if update succeeded, False otherwise
"""
try:
start_time = time.time()
start_time = time.monotonic()
self.execute_with_timeout(
lambda: plugin.update(),
timeout=timeout,
plugin_id=plugin_id
)
duration = time.time() - start_time
duration = time.monotonic() - start_time
if duration > 5.0: # Warn if update takes more than 5 seconds
if duration > self.SLOW_UPDATE_SECONDS:
self.logger.warning(
"Plugin %s update() took %.2fs (consider optimizing)",
plugin_id,
@@ -175,7 +192,7 @@ class PluginExecutor:
True if display succeeded, False otherwise
"""
try:
start_time = time.time()
start_time = time.monotonic()
# Does display() take a display_mode keyword? The caller usually
# knows and caches the answer, so prefer what it passed.
@@ -206,9 +223,9 @@ class PluginExecutor:
plugin_id=plugin_id
)
duration = time.time() - start_time
duration = time.monotonic() - start_time
if duration > 2.0: # Warn if display takes more than 2 seconds
if duration > self.SLOW_DISPLAY_SECONDS:
self.logger.warning(
"Plugin %s display() took %.2fs (consider optimizing)",
plugin_id,
+105 -1
View File
@@ -254,6 +254,104 @@ class PluginHealthTracker:
self._save_health_state(plugin_id, state)
def record_hang(self, plugin_id: str, operation: str, seconds: float,
error: Optional[Exception] = None) -> None:
"""Record a display() or update() call that ran past its limit.
Counts as a failure, so the ordinary circuit breaker handles a plugin
that keeps hanging: after ``failure_threshold`` in a row it is skipped
by both the update scheduler and the display rotation until the
cooldown ends. The hang itself is kept alongside (``hang_count``,
``last_hang``) so the health API can tell "hung" from "raised".
Not for an update skipped because the plugin's lock stayed held: the
holder may be a healthy but long render (Vegas prefetch). That is
:meth:`record_busy_skip`, which never touches the breaker.
Args:
plugin_id: Plugin identifier
operation: What hung: ``"display"`` or ``"update"``.
seconds: How long it had been running when this was recorded.
error: The error to store as ``last_error``; one is built from
the other arguments when omitted.
"""
state = self.get_health_state(plugin_id)
count = state.get('hang_count')
state['hang_count'] = (count if isinstance(count, int) and not isinstance(count, bool)
else 0) + 1
state['last_hang'] = {
'operation': operation,
'seconds': round(float(seconds), 3),
'time': time.time(),
}
if error is None:
error = TimeoutError(f"{operation} still running after {seconds:.1f}s")
# record_failure saves the record, hang fields included.
self.record_failure(plugin_id, error)
#: Minimum seconds between persisting a plugin's slow-call or busy-skip
#: counters. The in-memory record is updated every time; a plugin that is
#: slow on every frame must not become an SD-card write per frame.
SLOW_CALL_PERSIST_INTERVAL = 60.0
def record_slow_call(self, plugin_id: str, operation: str, seconds: float) -> None:
"""Note a call that finished, but slowly. Reporting only.
Unlike :meth:`record_hang` this never touches the circuit breaker: a
slow display() still drew its frame.
"""
state = self.get_health_state(plugin_id)
count = state.get('slow_call_count')
state['slow_call_count'] = (count if isinstance(count, int) and not isinstance(count, bool)
else 0) + 1
now = time.time()
state['last_slow_call'] = {
'operation': operation,
'seconds': round(float(seconds), 3),
'time': now,
}
self._save_reporting_throttled('slow', plugin_id, state, now)
def record_busy_skip(self, plugin_id: str, operation: str, seconds: float) -> None:
"""Note a call skipped because the plugin's lock stayed held. Reporting only.
The update worker gives up on a plugin's lock after
``PluginManager.PLUGIN_LOCK_TIMEOUT``. Whatever held it may be healthy
-- Vegas prefetch holds the lock for a plugin's whole content render,
which on a slow Pi can take longer than that -- so like
:meth:`record_slow_call` this never touches the circuit breaker, the
failure streak or ``last_error``. A real hang is recorded by
:meth:`record_hang` where it is measured.
Args:
plugin_id: Plugin identifier
operation: What was skipped, e.g. ``"update lock wait"``.
seconds: How long the lock was waited on.
"""
state = self.get_health_state(plugin_id)
count = state.get('busy_skip_count')
state['busy_skip_count'] = (count if isinstance(count, int) and not isinstance(count, bool)
else 0) + 1
now = time.time()
state['last_busy_skip'] = {
'operation': operation,
'seconds': round(float(seconds), 3),
'time': now,
}
self._save_reporting_throttled('busy', plugin_id, state, now)
def _save_reporting_throttled(self, kind: str, plugin_id: str,
state: Dict[str, Any], now: float) -> None:
"""Persist a reporting-only change at most once per
SLOW_CALL_PERSIST_INTERVAL per plugin and ``kind``. The first one is
saved at once, so the web process (which reads the persisted record)
sees it; repeats in between stay in memory until the next save."""
saved_at = self.__dict__.setdefault('_reporting_saved_at', {})
last = saved_at.get((kind, plugin_id))
if last is None or now - last >= self.SLOW_CALL_PERSIST_INTERVAL:
saved_at[(kind, plugin_id)] = now
self._save_health_state(plugin_id, state)
def set_degraded(self, plugin_id: str, reason: Optional[str]) -> None:
"""Flag (or clear) a plugin as degraded without touching the circuit breaker.
@@ -345,7 +443,13 @@ class PluginHealthTracker:
'degraded': state.get('degraded', False),
'degraded_reason': state.get('degraded_reason'),
'circuit_opened_time': state.get('circuit_opened_time'),
'half_open_start_time': state.get('half_open_start_time')
'half_open_start_time': state.get('half_open_start_time'),
'hang_count': state.get('hang_count', 0),
'last_hang': state.get('last_hang'),
'slow_call_count': state.get('slow_call_count', 0),
'last_slow_call': state.get('last_slow_call'),
'busy_skip_count': state.get('busy_skip_count', 0),
'last_busy_skip': state.get('last_busy_skip'),
}
def get_all_health_summaries(self) -> Dict[str, Dict[str, Any]]:
+323 -9
View File
@@ -16,12 +16,14 @@ import time
import threading
import types
from pathlib import Path
from typing import Dict, List, Optional, Any, Tuple
from typing import Dict, List, NamedTuple, Optional, Any, Tuple, Union
import logging
from src.exceptions import PluginError, ConfigError
from src.logging_config import get_logger
from src.plugin_system.plugin_loader import PluginLoader
from src.plugin_system.plugin_executor import PluginExecutor
from src.plugin_system.plugin_executor import (
PluginBusyError, PluginExecutor, PluginTimeoutError,
)
from src.plugin_system.plugin_state import PluginStateManager, PluginState
from src.plugin_system.schema_manager import (
CORE_VEGAS_TUNING_KEYS, SchemaManager, normalize_legacy_booleans,
@@ -36,6 +38,16 @@ from src.common.permission_utils import (
)
class _DeferredConfigChange(NamedTuple):
"""Update-queue item: apply the config change parked for ``plugin_id``.
Queued by apply_config_change() when the plugin's lock was busy; the
change itself waits in ``PluginManager._deferred_config_changes`` so only
the latest one is ever applied.
"""
plugin_id: str
class PluginManager:
"""
Manages plugin discovery, loading, and lifecycle.
@@ -56,6 +68,19 @@ class PluginManager:
# How long unload_plugin() waits for an in-flight update() to finish
# before tearing the instance down anyway.
UNLOAD_LOCK_TIMEOUT = 5.0
# How long the update worker and apply_config_change() wait for a
# plugin's lock -- the same bound unload already uses for the same lock.
# A display() frame holds it for milliseconds, so this only runs out when
# the holder is hung or pathologically slow. The worker then skips that
# plugin (recorded as a hang, so repeats open its circuit breaker)
# instead of stalling every other plugin's update behind it.
PLUGIN_LOCK_TIMEOUT = UNLOAD_LOCK_TIMEOUT
# Minimum seconds between repeats of the same hang/slow-call warning for
# one plugin. A hung plugin is re-detected every interval; a slow
# display() can be re-detected every frame.
HANG_LOG_INTERVAL = 60.0
def __init__(self, plugins_dir: str = "plugins",
config_manager: Optional[Any] = None,
@@ -122,7 +147,33 @@ class PluginManager:
# post-timeout window.
# Kill switch: plugin_system.synchronous_updates: true restores the
# inline path.
self._update_queue: "queue.Queue[Optional[Tuple[str, float]]]" = queue.Queue()
#
# Which thread runs each plugin hook, and what it holds:
# __init__, on_enable the loading thread (main thread at startup,
# the render thread on a live enable).
# update() plugin-update-worker, under the plugin lock,
# via PluginExecutor (whose daemon thread runs
# the call; if it outlives the executor's
# timeout it keeps the lock until it returns).
# Exceptions: the startup pass
# (DisplayController._run_initial_updates, main
# thread, before the display loop starts) and
# the synchronous_updates kill switch (render
# thread) run it without the lock.
# display() the render thread, under a try-lock: a busy
# lock skips the frame. The first frame of a
# screen goes through PluginExecutor. Vegas
# mode's adapter and coordinator take the lock
# with a bounded wait.
# on_config_change() ConfigService's watcher thread, under the
# plugin lock via apply_config_change(); if the
# lock stays busy it is deferred to the update
# worker, which applies it under the lock.
# cleanup(), on_disable() whoever calls unload_plugin(), under the
# lock with UNLOAD_LOCK_TIMEOUT.
# No wait on a plugin lock is unbounded, so one hung plugin can only
# cost the worker PLUGIN_LOCK_TIMEOUT per attempt.
self._update_queue: "queue.Queue[Union[None, Tuple[str, float], _DeferredConfigChange]]" = queue.Queue()
self._pending_updates: set = set()
self._pending_lock = threading.Lock()
# Serializes the "is this plugin eligible?" -> "claim it (RUNNING)"
@@ -145,6 +196,12 @@ class PluginManager:
# run_scheduled_updates_with_changes().
self._completed_updates: set = set()
self._completed_updates_lock = threading.Lock()
# Config changes that found the plugin's lock busy, latest per plugin,
# with the instance they were meant for. See apply_config_change().
self._deferred_config_changes: Dict[str, Tuple[Any, Dict[str, Any]]] = {}
self._deferred_config_lock = threading.Lock()
# key -> (monotonic time last logged, repeats suppressed since)
self._rate_limited_warnings: Dict[str, Tuple[float, int]] = {}
self._synchronous_updates = False
if self.config_manager is not None:
try:
@@ -686,6 +743,8 @@ class PluginManager:
# Remove from active plugins
del self.plugins[plugin_id]
with self._deferred_config_lock:
self._deferred_config_changes.pop(plugin_id, None)
with self._plugin_last_update_lock:
self.plugin_last_update.pop(plugin_id, None)
self._update_interval_cache.pop(plugin_id, None)
@@ -1022,6 +1081,8 @@ class PluginManager:
self,
plugin_id: str,
exc: Optional[Exception] = None,
log: bool = True,
count_failure: bool = True,
) -> None:
"""Apply the standard failure-recovery path for a plugin update.
@@ -1035,6 +1096,11 @@ class PluginManager:
exc: The exception that caused the failure, if any. When None a
synthetic ExecutionFailure exception is constructed from the
timeout/executor-error path.
log: Log the generic failure line. Callers that already logged
something more specific (rate-limited) pass False.
count_failure: Record the failure in plugin health, where it
counts toward the circuit breaker. A busy skip passes False:
it records itself as a busy skip, reporting only.
"""
failure_time = time.time()
if exc is not None:
@@ -1050,13 +1116,91 @@ class PluginManager:
'timestamp': failure_time,
'recoverable': True,
}
self.logger.warning("Plugin %s update() failed; will retry after interval", plugin_id)
if log:
self.logger.warning("Plugin %s update() failed; will retry after interval", plugin_id)
with self._plugin_last_update_lock:
self.plugin_last_update[plugin_id] = failure_time
self.state_manager.set_state_with_error(plugin_id, PluginState.ENABLED, error_info)
if self.health_tracker:
if count_failure and self.health_tracker:
self.health_tracker.record_failure(plugin_id, err)
def _warn_rate_limited(self, key: str, message: str, *args: Any) -> None:
"""Log a warning at most once per HANG_LOG_INTERVAL for ``key``.
Repeats in between are counted and the count is appended to the next
one that is logged, so the journal shows the problem continuing
without a line per frame or per scheduler tick.
"""
# setdefault: tests build bare managers with PluginManager.__new__.
seen = self.__dict__.setdefault('_rate_limited_warnings', {})
now = time.monotonic()
last, suppressed = seen.get(key, (None, 0))
if last is not None and now - last < self.HANG_LOG_INTERVAL:
seen[key] = (last, suppressed + 1)
return
seen[key] = (now, 0)
if suppressed:
message += " (%d more since the last warning)"
args = args + (suppressed,)
self.logger.warning(message, *args)
def _record_hang(self, plugin_id: str, operation: str, seconds: float,
err: Exception) -> None:
"""Record a hang in plugin health: a failure to the circuit breaker.
PluginHealthTracker.record_hang also counts the hang separately. Never
raises: this runs on the update worker and the render thread.
"""
tracker = self.health_tracker
if tracker is None:
return
try:
tracker.record_hang(plugin_id, operation, seconds, err)
except Exception as e: # pylint: disable=broad-except
self.logger.debug("Could not record hang for %s: %s", plugin_id, e)
def note_display_duration(self, plugin_id: str, seconds: float) -> None:
"""Account for one display() call that took ``seconds``.
Called by the render loop for every frame, so the common case is one
comparison. At or above PluginExecutor.SLOW_DISPLAY_SECONDS the call
is logged (rate-limited) and counted as slow in plugin health; at or
above the executor's timeout -- the limit the first frame of a screen
is already held to -- it is recorded as a hang, which the circuit
breaker counts as a failure.
"""
if seconds < PluginExecutor.SLOW_DISPLAY_SECONDS:
return
if seconds >= self.plugin_executor.default_timeout:
self.record_display_hang(plugin_id, seconds)
return
self._warn_rate_limited(
"slow-display:" + plugin_id,
"Plugin %s display() took %.2fs; a frame should take milliseconds "
"(is it fetching or loading files in display()?)", plugin_id, seconds)
tracker = self.health_tracker
record_slow = getattr(tracker, 'record_slow_call', None) if tracker is not None else None
if callable(record_slow):
try:
record_slow(plugin_id, 'display', seconds)
except Exception as e: # pylint: disable=broad-except
self.logger.debug("Could not record slow display for %s: %s", plugin_id, e)
def record_display_hang(self, plugin_id: str, seconds: float) -> None:
"""Record a display() call that ran ``seconds``, past its limit.
Either it has since returned (note_display_duration) or it is still
running on the executor's lingering thread, holding the plugin's lock
(the render loop's first-frame dispatch).
"""
self._warn_rate_limited(
"hung-display:" + plugin_id,
"Plugin %s display() ran for at least %.1fs (limit %.0fs); recorded "
"as a hang -- repeated hangs open its circuit breaker",
plugin_id, seconds, self.plugin_executor.default_timeout)
self._record_hang(plugin_id, 'display', seconds, PluginTimeoutError(
f"Plugin {plugin_id} display() ran for at least {seconds:.1f}s"))
def run_scheduled_updates(self, current_time: Optional[float] = None) -> None:
"""
Trigger plugin updates based on their defined update intervals.
@@ -1142,11 +1286,13 @@ class PluginManager:
self.state_manager.set_state(plugin_id, PluginState.ENABLED)
def get_plugin_lock(self, plugin_id: str) -> threading.Lock:
"""Per-plugin lock keeping update() and display() mutually exclusive.
"""Per-plugin lock keeping update(), display() and on_config_change()
mutually exclusive.
The update worker holds it for the duration of a plugin's update();
the display side acquires it non-blocking and skips that frame's
display() call when the plugin is mid-update.
display() call when the plugin is mid-update. Every other waiter uses
a bounded acquire (see the thread notes in __init__).
"""
with self._plugin_locks_guard:
lock = self._plugin_locks.get(plugin_id)
@@ -1212,14 +1358,28 @@ class PluginManager:
real update() call genuinely finishes (see _execute_update_now),
which can be after this dispatch returns if PluginExecutor's own
timeout elapses first.
The lock wait is bounded by PLUGIN_LOCK_TIMEOUT. Whatever holds it
past that -- a hung display() on the render thread, a lingering
executor thread, or a long but healthy Vegas content render -- costs
this worker that long once per attempt, and the plugin's update is
skipped and reported as a busy skip (_skip_busy_update), which never
counts toward the circuit breaker; the other plugins' queued updates
carry on.
"""
while True:
item = self._update_queue.get()
if item is None: # shutdown sentinel
return
if isinstance(item, _DeferredConfigChange):
self._apply_deferred_config_change(item.plugin_id)
continue
plugin_id, scheduled_time = item
lock = self.get_plugin_lock(plugin_id)
lock.acquire()
wait_start = time.monotonic()
if not lock.acquire(timeout=self.PLUGIN_LOCK_TIMEOUT):
self._skip_busy_update(plugin_id, time.monotonic() - wait_start)
continue
plugin_instance = self.plugins.get(plugin_id)
if plugin_instance is None: # unloaded while queued; its
# lifecycle state was already cleared by unload_plugin —
@@ -1228,6 +1388,9 @@ class PluginManager:
with self._pending_lock:
self._pending_updates.discard(plugin_id)
continue
# A config change that found the lock busy goes in first, so
# this update() runs against the settings the user saved.
self._apply_deferred_config_locked(plugin_id, plugin_instance)
try:
self._execute_update_now(plugin_id, plugin_instance,
scheduled_time, lock=lock)
@@ -1238,6 +1401,142 @@ class PluginManager:
self.logger.exception("update worker: unexpected error for %s",
plugin_id)
def _skip_busy_update(self, plugin_id: str, waited: float) -> None:
"""Give up on a queued update whose plugin lock stayed held.
Same bookkeeping as a failed update() -- pending slot dropped before
the state returns to ENABLED with PluginBusyError error info,
last-update stamped so the retry waits a full interval -- but
report-only in health: counted as a busy skip (``busy_skip_count`` /
``last_busy_skip``), never as a failure or a hang. The lock holder
may be perfectly healthy: Vegas prefetch holds a plugin's lock for its
whole content render, which on a slow Pi can outlast
PLUGIN_LOCK_TIMEOUT, and counting that would pull a healthy plugin
from rotation. Real hangs -- display() or update() past the executor
timeout -- are recorded where they are measured and still open the
breaker.
"""
with self._pending_lock:
self._pending_updates.discard(plugin_id)
if plugin_id not in self.plugins:
# Unloaded while we waited: its lifecycle state is already
# cleared; recording anything would resurrect it as ENABLED.
return
self._warn_rate_limited(
"busy-update:" + plugin_id,
"Plugin %s update skipped: its lock was still held after %.1fs "
"(a display(), Vegas render or update() of it is still running); "
"retrying next interval, not counted as a failure", plugin_id, waited)
self._record_update_failure(
plugin_id,
exc=PluginBusyError(
f"Plugin {plugin_id} busy: its lock was held for over {waited:.1f}s "
"by a slow or hung display()/update(); update skipped"),
log=False,
count_failure=False)
tracker = self.health_tracker
record_busy = getattr(tracker, 'record_busy_skip', None) if tracker is not None else None
if callable(record_busy):
try:
record_busy(plugin_id, 'update lock wait', waited)
except Exception as e: # pylint: disable=broad-except
self.logger.debug("Could not record busy skip for %s: %s", plugin_id, e)
def apply_config_change(self, plugin_id: str, new_config: Dict[str, Any],
plugin_instance: Optional[Any] = None) -> bool:
"""Call ``on_config_change(new_config)`` without racing update()/display().
Runs on the calling thread -- ConfigService's watcher, for the display
service -- holding the plugin's lock, waited on for at most
PLUGIN_LOCK_TIMEOUT. If the lock is still busy (an update() mid-fetch
can outlast that) the change is parked and handed to the update
worker, which applies it under the same lock once it is free, and at
the latest just before the plugin's next update(). A later change for
the same plugin replaces a parked one.
Exceptions from on_config_change propagate on the immediate path,
as they did when the caller invoked it directly.
Args:
plugin_id: Plugin identifier.
new_config: The prepared config to hand the plugin.
plugin_instance: The instance to notify; defaults to the loaded one.
Returns:
True if on_config_change ran now, False if it was deferred or there
is no loaded plugin to notify.
"""
if plugin_instance is None:
plugin_instance = self.plugins.get(plugin_id)
if plugin_instance is None or not hasattr(plugin_instance, 'on_config_change'):
return False
lock = self.get_plugin_lock(plugin_id)
if lock.acquire(timeout=self.PLUGIN_LOCK_TIMEOUT):
try:
with self._deferred_config_lock:
# This change supersedes any older one still parked.
self._deferred_config_changes.pop(plugin_id, None)
plugin_instance.on_config_change(new_config)
finally:
lock.release()
return True
with self._deferred_config_lock:
self._deferred_config_changes[plugin_id] = (plugin_instance, new_config)
self._warn_rate_limited(
"busy-config:" + plugin_id,
"Plugin %s is busy (lock held for over %.1fs); its config change "
"will be applied by the update worker once it is free",
plugin_id, self.PLUGIN_LOCK_TIMEOUT)
try:
self._ensure_update_worker()
self._update_queue.put(_DeferredConfigChange(plugin_id))
except Exception as exc: # pylint: disable=broad-except
# No worker (thread start refused): still parked, so the next
# update() of this plugin applies it.
self.logger.error(
"Could not queue the config change for plugin %s (%s: %s); it "
"will be applied before its next update()",
plugin_id, type(exc).__name__, exc)
return False
def _apply_deferred_config_change(self, plugin_id: str) -> None:
"""Worker side of a parked config change: take the lock, apply it."""
with self._deferred_config_lock:
if plugin_id not in self._deferred_config_changes:
return # applied or superseded meanwhile
lock = self.get_plugin_lock(plugin_id)
wait_start = time.monotonic()
if not lock.acquire(timeout=self.PLUGIN_LOCK_TIMEOUT):
self._warn_rate_limited(
"busy-config:" + plugin_id,
"Plugin %s still busy after %.1fs; its config change stays "
"parked until its next update()",
plugin_id, time.monotonic() - wait_start)
return
try:
self._apply_deferred_config_locked(plugin_id, self.plugins.get(plugin_id))
finally:
lock.release()
def _apply_deferred_config_locked(self, plugin_id: str,
current_instance: Optional[Any]) -> None:
"""Apply the parked config change for plugin_id; caller holds its lock."""
with self._deferred_config_lock:
entry = self._deferred_config_changes.pop(plugin_id, None)
if entry is None:
return
instance, new_config = entry
if current_instance is None or instance is not current_instance:
# Unloaded, or reloaded as a new instance built from the current
# config: nothing left to tell.
return
try:
instance.on_config_change(new_config)
self.logger.info("Applied deferred config change for plugin %s", plugin_id)
except Exception: # pylint: disable=broad-except
self.logger.exception("Error in plugin %s config change handler", plugin_id)
def stop_update_worker(self, timeout: float = 5.0) -> None:
"""Signal the worker to exit (used by cleanup; thread is a daemon)."""
if self._update_worker is not None and self._update_worker.is_alive():
@@ -1348,14 +1647,29 @@ class PluginManager:
else:
_finish(True)
started = time.monotonic()
try:
self.plugin_executor.execute_update(
success = self.plugin_executor.execute_update(
types.SimpleNamespace(update=_target_update), plugin_id)
except Exception as exc: # pragma: no cover - defensive; execute_update
# catches everything internally, but guarantee _finish still
# runs (releasing the lock) if something unexpected slips through.
self.logger.exception("Unexpected error dispatching update for %s: %s", plugin_id, exc)
_finish(False, exc=exc)
return
if not success and not finished['done']:
# The executor stopped waiting but update() is still running: it
# keeps the lock and the RUNNING state until it returns (then
# _finish records the outcome). Say so now, rather than leave the
# plugin silently stuck; record_success on a late return clears it.
elapsed = time.monotonic() - started
self._warn_rate_limited(
"hung-update:" + plugin_id,
"Plugin %s update() still running after %.1fs; it keeps its "
"lock until it returns, and is not rescheduled until then",
plugin_id, elapsed)
self._record_hang(plugin_id, 'update', elapsed, PluginTimeoutError(
f"Plugin {plugin_id} update() still running after {elapsed:.1f}s"))
def run_scheduled_updates_with_changes(self, current_time: Optional[float] = None) -> List[str]:
"""