fix(display): Vegas resumes after live priority, and six smaller runtime fixes (#644)

- Vegas: a live-priority pause was only lifted from inside run_frame(),
  which returns before that check while paused, so the ticker never came
  back until a restart. run_iteration() now resumes it (the controller
  only calls it when nothing preempts Vegas); start()/stop() clear the
  pause state. Iteration length is timed with the monotonic clock.
- Dim schedule: a per-day disabled day now updates the minute-gate cache,
  so brightness no longer flips back to dim within each minute.
- On-demand: a second request no longer overwrites the rotation resume
  index with the first request's mode.
- Render pipeline: reset() drops the prepared group and deferred queue,
  and a prefetch in flight across a reset discards its result.
- Sync: stop() removes the status file (and the controller's cleanup now
  calls it), standalone removes a stale one at startup, and writes use a
  unique mkstemp temp file.
- render_gate.swap_releases_gil() delegates to frame_timing.
- Stale docstrings/comments corrected.

Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Chuck
2026-09-28 08:25:26 -04:00
committed by GitHub
co-authored by Claude Opus 5.5
parent bcef1957a9
commit 6bc13a8934
14 changed files with 401 additions and 32 deletions
+6 -12
View File
@@ -48,6 +48,8 @@ import time
from collections import deque
from typing import Any, Callable, Deque, List, Optional
from src.common.frame_timing import binding_releases_gil
#: Park background threads this long before the refresh a swap will return on,
#: so a short C call already under way has finished by then.
MARGIN_SECONDS = 0.002
@@ -90,19 +92,11 @@ def _unsafe(frame: Any, base: Any) -> bool:
def swap_releases_gil() -> Optional[bool]:
"""Whether the loaded rgbmatrix binding releases the GIL, or None if none is loaded.
The rebuilt binding links PyEval_SaveThread and the stock one never does.
The same test as src.common.frame_timing.binding_releases_gil (#629); one
of the two goes once both have landed.
A thin delegate to src.common.frame_timing.binding_releases_gil (#629),
which this used to duplicate line for line. The name stays because the
coordinator calls it here and tests replace it here.
"""
module = sys.modules.get("rgbmatrix.core")
path = getattr(module, "__file__", None)
if not path:
return None
try:
with open(path, "rb") as handle:
return b"PyEval_SaveThread" in handle.read()
except OSError:
return None
return binding_releases_gil()
def _held(lock: Any) -> bool:
+56 -5
View File
@@ -53,6 +53,17 @@ HEARTBEAT_INTERVAL = 2.0 # follower sends heartbeat every 2 s
PEER_TIMEOUT = 6.0 # leader: no heartbeat → follower gone
LEADER_TIMEOUT = 6.0 # follower: no frame → leader gone
STATUS_FILE = os.path.join(tempfile.gettempdir(), "led_matrix_sync_status.json")
# Serialises writes to STATUS_FILE (several threads report status) against
# its removal in stop(), so a write already under way cannot put the file back
# after the display process has shut down.
_STATUS_LOCK = threading.Lock()
def _remove_status_file() -> None:
try:
os.remove(STATUS_FILE)
except FileNotFoundError:
pass
class SyncRole(Enum):
@@ -83,6 +94,10 @@ class DisplaySyncManager:
back to its own plugins when the leader stops sending.
"""
# Set by stop(); status writes after that are dropped. Class-level so
# instances built without __init__ (tests) have it too.
_status_closed = False
def __init__(
self,
role_str: str,
@@ -138,6 +153,14 @@ class DisplaySyncManager:
self._send_sock: Optional[socket.socket] = None
if self.role == SyncRole.STANDALONE:
# Standalone never writes a status file, so one still here is
# from an earlier run as leader or follower. The web UI would
# keep reporting that run's peer as if it were live.
try:
with _STATUS_LOCK:
_remove_status_file()
except OSError as exc:
logger.debug("Sync: could not remove stale status file: %s", exc)
return
if self.role == SyncRole.LEADER:
@@ -692,18 +715,38 @@ class DisplaySyncManager:
def write_status_file(self) -> None:
"""Write current sync status to STATUS_FILE for the web UI to read."""
tmp = None
try:
status = self.get_status()
status["ts"] = time.time()
tmp = STATUS_FILE + ".tmp"
with open(tmp, "w") as f:
json.dump(status, f)
os.replace(tmp, STATUS_FILE)
with _STATUS_LOCK:
if self._status_closed:
return
# A unique temp name per write, like frame_timing's stats
# file: the receive loop, watchdog and hello handler all
# write, and with one fixed ".tmp" name one thread's
# os.replace() could move the other's half-written file.
fd, tmp = tempfile.mkstemp(
dir=os.path.dirname(STATUS_FILE) or ".",
prefix=".led_matrix_sync_status.", suffix=".tmp")
with os.fdopen(fd, "w") as f:
json.dump(status, f)
# mkstemp makes it owner-only; the web UI may run as a
# different user from the display service.
os.chmod(tmp, 0o644)
os.replace(tmp, STATUS_FILE)
tmp = None
except Exception as exc:
self.logger.debug("Sync: status file write error: %s", exc)
finally:
if tmp is not None:
try:
os.unlink(tmp)
except OSError:
pass
def stop(self) -> None:
"""Shut down threads and close sockets."""
"""Shut down threads, close sockets and withdraw the status file."""
self._running = False
for sock in (self._recv_sock, self._send_sock, self._img_server_sock):
if sock:
@@ -711,3 +754,11 @@ class DisplaySyncManager:
sock.close()
except Exception as exc:
self.logger.debug("Sync: error closing socket: %s", exc)
# The web UI reads this file as live status. Left behind, it went on
# reporting a connected peer after the display service had stopped.
try:
with _STATUS_LOCK:
self._status_closed = True
_remove_status_file()
except OSError as exc:
self.logger.debug("Sync: could not remove status file: %s", exc)
+29 -5
View File
@@ -284,7 +284,10 @@ class DisplayController:
if os.path.isabs(plugins_dir_name):
plugins_dir = plugins_dir_name
else:
# If relative, resolve relative to the project root (LEDMatrix directory)
# If relative, resolve against the current working directory.
# That is the project root only because ledmatrix.service
# sets WorkingDirectory to it; run from anywhere else, a
# relative path resolves against wherever that is.
project_root = os.getcwd()
plugins_dir = os.path.join(project_root, plugins_dir_name)
@@ -801,7 +804,15 @@ class DisplayController:
if use_per_day:
day_config = days_config[current_day]
if not day_config.get('enabled', True):
# Past the minute gate, so the cache must say the same thing:
# returning here without it left the previous minute's dim
# value to be served for the rest of this one, and the
# brightness flipped between dim and normal every minute.
if self._was_dimmed:
logger.info(f"Dim schedule deactivated: brightness restored to {normal_brightness}%")
self.is_dimmed = False
self._was_dimmed = False
self._cached_target_brightness = normal_brightness # persist for minute-gate
return normal_brightness
start_time_str = day_config.get('start_time', '20:00')
end_time_str = day_config.get('end_time', '07:00')
@@ -1710,10 +1721,16 @@ class DisplayController:
pinned = bool(request.get('pinned', False))
now = time.time()
if self.available_modes:
self.rotation_resume_index = self.current_mode_index
else:
self.rotation_resume_index = None
# Only a request that starts a session records where rotation was.
# A request made while on-demand is already showing would otherwise
# save the previous request's mode (current_mode_index points at it
# by now), and clearing would resume there instead of where the
# normal rotation was interrupted.
if not self.on_demand_active:
if self.available_modes:
self.rotation_resume_index = self.current_mode_index
else:
self.rotation_resume_index = None
if resolved_mode in self.available_modes:
self.current_mode_index = self.available_modes.index(resolved_mode)
@@ -3310,6 +3327,13 @@ class DisplayController:
self.vegas_coordinator.cleanup()
except Exception as e:
logger.warning("Error cleaning up Vegas mode: %s", e)
# After Vegas, which sends through it. Stopping also withdraws the
# sync status file, which the web UI otherwise kept showing as live.
if getattr(self, 'sync_manager', None) is not None:
try:
self.sync_manager.stop()
except Exception as e:
logger.warning("Error stopping display sync: %s", e)
# Shutdown config service if it exists
if hasattr(self, 'config_service'):
try:
+4 -1
View File
@@ -20,7 +20,10 @@ Key responsibilities
Singleton: only one ``DisplayManager`` instance exists per process. The
first call to ``DisplayManager(config)`` creates it; subsequent calls return
the same object.
the same object, but ``__init__`` runs again on it each time, so it is
re-initialised (matrix included) with the new arguments rather than handed
back as it was. Construct it once and pass that instance around;
:meth:`DisplayManager.cleanup` clears the singleton.
"""
import json
+29 -7
View File
@@ -273,6 +273,10 @@ class VegasModeCoordinator:
self._is_active = True
self._should_stop = False
# A pause belongs to the run it happened in; carrying it into a
# new run would have run_frame() refuse every frame.
self._is_paused = False
self._live_priority_active = False
self._start_time = time.time()
# A fresh run starts with a clean health slate: no stale
# "was degraded" from the previous run, and a heartbeat that is
@@ -298,6 +302,8 @@ class VegasModeCoordinator:
self._should_stop = True
self._is_active = False
self._is_paused = False
self._live_priority_active = False
if self._start_time:
self.stats['total_runtime_seconds'] += time.time() - self._start_time
@@ -473,6 +479,20 @@ class VegasModeCoordinator:
if not self.start():
return False
# A live-priority pause is only ever lifted by _check_live_priority(),
# and run_frame() returns before reaching it while paused -- so once
# paused, every later iteration returned False at its first frame and
# the ticker never came back until a restart. The display controller
# only calls run_iteration() when nothing preempts Vegas (no live mode,
# or live content is kept in the ticker), so being called at all means
# the live content that paused us has ended.
with self._state_lock:
paused_for_live = self._is_paused and self._live_priority_active
if paused_for_live:
self._live_priority_active = False
self.resume()
logger.info("Live priority ended - resuming Vegas")
if self.vegas_config.continuous_scroll:
# The strip is continuously extended and trimmed, so its width says
# nothing about how long to run. This is only how often control
@@ -481,7 +501,11 @@ class VegasModeCoordinator:
duration = float(self.vegas_config.max_cycle_duration)
else:
duration = self.render_pipeline.get_dynamic_duration()
start_time = time.time()
# Monotonic for the same reason as the per-frame clock below: this
# bounds how long the iteration runs, and an NTP step on an RTC-less
# Pi would otherwise end it at once (forward) or stretch it by the
# size of the correction (backward).
start_time = time.monotonic()
frame_count = 0
fps_log_interval = 5.0 # Sample FPS every 5 seconds
# Health state lives on the coordinator, not here: run_iteration() is
@@ -490,10 +514,8 @@ class VegasModeCoordinator:
# of every iteration rather than once per interval, and a recovery
# that crossed an iteration boundary was never reported at all --
# was_degraded had already gone back to False.
# Monotonic, and deliberately not start_time: start_time is wall
# clock and is used below to report the iteration's duration. Mixing
# the two here would make every delta hugely negative and silence the
# frame-rate reporting altogether.
# Monotonic. Never mix it with a wall-clock value: every delta would
# be hugely negative and silence the frame-rate reporting altogether.
last_fps_log_time = time.monotonic()
fps_frame_count = 0
# A mean hides stutter completely. At 120fps a five-second window is
@@ -633,7 +655,7 @@ class VegasModeCoordinator:
).start()
# Check elapsed time
elapsed = time.time() - start_time
elapsed = time.monotonic() - start_time
if elapsed >= duration:
break
@@ -652,7 +674,7 @@ class VegasModeCoordinator:
# cycle content multiple times within one iteration — acceptable for
# a continuous ticker.
logger.info("Vegas iteration completed after %.1fs", time.time() - start_time)
logger.info("Vegas iteration completed after %.1fs", time.monotonic() - start_time)
return True
def _check_live_priority(self) -> bool:
+20 -1
View File
@@ -52,6 +52,12 @@ class RenderPipeline:
# A panel measured within this fraction of its cap is keeping up with it.
REFRESH_TOLERANCE = 0.03
# Bumped by reset(). A prefetch thread records the value it started under
# and drops its group if a reset happened meanwhile, so a fetch still in
# flight when Vegas stops cannot land in the next run. Class-level so
# pipelines built without __init__ (tests) still have it.
_prefetch_generation = 0
def __init__(
self,
config: VegasModeConfig,
@@ -373,6 +379,7 @@ class RenderPipeline:
return
if self._prepared_group is not None:
return # already have one waiting
generation = self._prefetch_generation
def _work():
# Deprioritise against the render loop. Linux applies nice
@@ -394,6 +401,8 @@ class RenderPipeline:
logger.exception("Background prefetch failed")
group = []
with self._prefetch_lock:
if generation != self._prefetch_generation:
return # Vegas was reset while this was fetching
self._prepared_group = group
self._prefetch_thread = threading.Thread(
@@ -602,7 +611,8 @@ class RenderPipeline:
"""
Render a single frame to the display.
Should be called at ~125 FPS (8ms intervals).
Called once per frame by the coordinator, which paces the calls by
frame_interval (see that property) rather than a fixed rate.
Returns:
True if frame was rendered, False if no content
@@ -917,6 +927,15 @@ class RenderPipeline:
self._segments_in_scroll = []
self._frame_times = deque(maxlen=100)
# Content lined up for the old run belongs to it. Left in place, the
# first extension after Vegas is switched back on appended that stale
# group -- including plugins disabled in the meantime -- and the
# deferred queue went on fetching the old run's plugins.
with self._prefetch_lock:
self._prefetch_generation += 1
self._prepared_group = None
self._deferred_queue = []
self.display_manager.set_scrolling_state(False)
logger.info("RenderPipeline reset")