mirror of
https://github.com/ChuckBuilds/LEDMatrix.git
synced 2026-10-04 06:15:09 +00:00
feat(display): systemd watchdog and heartbeat for a frozen render loop (#687)
If the render loop gets stuck inside a plugin's display(), ledmatrix.service stays active and the panel stays frozen. This adds a way to detect that. - src/display_watchdog.py (standard library only) sends sd_notify over $NOTIFY_SOCKET and writes /run/ledmatrix/display-heartbeat.json. Only the render thread counts: beats from other threads are ignored. - ledmatrix.service: WatchdogSec=120, NotifyAccess=main, RuntimeDirectory=ledmatrix (0755), RestartSteps=4 and RestartMaxDelaySec=2min. It stays Type=simple. run.py widens the watchdog to 15 min for start-up, and load_plugin() does the same on the render thread. The loop arms after its first frame. - /api/v3/health adds checks.display_loop: running, stalled (no heartbeat for over 60s, which makes the status degraded) or not_reported. With web login on, a caller who is not logged in still gets only healthy/degraded, and a stall degrades that answer. - The update verifier requires a fresh heartbeat from the restarted display when the display it replaced was writing one. A frozen panel is rolled back. - Existing installs get the systemd watchdog only after install_service.sh is re-run. The heartbeat works right away. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -14,6 +14,7 @@ from web_interface.blueprints.api_v3 import (
|
||||
subprocess, success_response, tempfile,
|
||||
)
|
||||
from src.common.path_safety import safe_path_component
|
||||
from src import display_watchdog
|
||||
from src.common import sync_manager as _sync
|
||||
from src import error_aggregator as _errors
|
||||
from web_interface import display_preview
|
||||
@@ -93,6 +94,34 @@ def get_health():
|
||||
'error': 'see logs for details'
|
||||
}
|
||||
|
||||
# Is the render loop still going round? The display rewrites this
|
||||
# heartbeat every few seconds from the render thread itself, so a
|
||||
# thread stuck inside a plugin lets it go stale even though the
|
||||
# service is "active". No heartbeat at all (the dev server, an older
|
||||
# display) is not a failure: the preview-frame check below is then
|
||||
# the only signal, as it always was.
|
||||
try:
|
||||
heartbeat = display_watchdog.read_heartbeat(display_watchdog.HEARTBEAT_PATH)
|
||||
age = display_watchdog.heartbeat_age(heartbeat) if heartbeat else None
|
||||
if age is None:
|
||||
health_status['checks']['display_loop'] = {
|
||||
'status': 'not_reported',
|
||||
'note': 'The display is not writing a heartbeat (not started yet, '
|
||||
'or a version or setup without one)',
|
||||
}
|
||||
else:
|
||||
fresh = age < display_watchdog.HEARTBEAT_STALE_SECONDS
|
||||
health_status['checks']['display_loop'] = {
|
||||
'status': 'running' if fresh else 'stalled',
|
||||
'heartbeat_age_seconds': round(age, 1),
|
||||
}
|
||||
except Exception:
|
||||
logger.warning("Health check could not read the display heartbeat", exc_info=True)
|
||||
health_status['checks']['display_loop'] = {
|
||||
'status': 'unknown',
|
||||
'error': 'see logs for details'
|
||||
}
|
||||
|
||||
# Check hardware connectivity (if display manager available)
|
||||
try:
|
||||
snapshot_path = display_preview.SNAPSHOT_PATH
|
||||
@@ -117,8 +146,10 @@ def get_health():
|
||||
}
|
||||
|
||||
# Determine overall health
|
||||
# 'not_reported' is the absence of a signal, not a bad one.
|
||||
all_healthy = all(
|
||||
check.get('status') in ['accessible', 'operational', 'connected', 'running', 'active']
|
||||
check.get('status') in ['accessible', 'operational', 'connected', 'running', 'active',
|
||||
'not_reported']
|
||||
for check in health_status['checks'].values()
|
||||
)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user