fix(display): thread-safety for deferred updates, BDF faces and follower image; one refresh default (#652)

- DisplayManager.defer_update()/process_deferred_updates(): one lock around
  every queue mutation (appends from the update thread were lost to the
  render thread's filter/slice reassignments); callables run outside it.
- FontManager and element_style no longer cache BDF freetype.Face objects
  process-wide (load_bdf_face caches them per thread); element_style's LRU
  is locked against get/move_to_end vs eviction races.
- limit_refresh_rate_hz default is one constant, DEFAULT_REFRESH_LIMIT_HZ =
  100 (the template's), for the library options, refresh_hz, the matrix
  guard, Vegas and scroll_config. Previously a missing key capped the panel
  at 90 while pacing assumed 100.
- Sync follower: the TCP thread queues the leader's scroll image; the render
  thread swaps image/array/width in between frames.
- update_display() error log rate-limited (traceback first, then once a
  minute with a count); swallowed DisplayController exceptions log at DEBUG.
- Root display_controller.py runs run.py via runpy.
- stream_manager: correct the RLock release comments; merge duplicate if.

Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Chuck
2026-09-28 10:39:40 -04:00
committed by GitHub
co-authored by Claude Opus 5.5
parent 1e62677257
commit 6f45ff5e63
17 changed files with 590 additions and 79 deletions
+5 -2
View File
@@ -51,6 +51,8 @@ import logging
from dataclasses import dataclass, replace
from typing import Any, Dict, Optional
from src.matrix_support import DEFAULT_REFRESH_LIMIT_HZ
logger = logging.getLogger(__name__)
#: Speed used when a plugin supplies nothing usable. One pixel per refresh on a
@@ -63,8 +65,9 @@ MIN_PIXELS_PER_SECOND = 1.0
MAX_PIXELS_PER_SECOND = 500.0
#: Assumed refresh when the caller does not say. Matches the usual
#: ``display.hardware.limit_refresh_rate_hz``.
DEFAULT_REFRESH_HZ = 100.0
#: ``display.hardware.limit_refresh_rate_hz``, and is the cap DisplayManager
#: applies when that key is missing.
DEFAULT_REFRESH_HZ = float(DEFAULT_REFRESH_LIMIT_HZ)
#: How far px/s may sit from a whole number of pixels per refresh before it is
#: worth warning about. 0.05px per frame is invisible; a third of a pixel is not.
+46 -14
View File
@@ -27,6 +27,7 @@ import signal
import json
import threading
import types
from collections import deque
from contextlib import contextmanager
from typing import Dict, Any, List, Optional, Callable, Tuple
from datetime import datetime
@@ -196,6 +197,12 @@ class DisplayController:
# scroll image arrives.
self._follower_pending_new_image = False
self._follower_last_frame = None
# (image, array) from the leader, handed from the sync TCP thread to
# the render thread, which adopts it at the start of a follower frame
# (_adopt_follower_scroll_image). One append / one popleft, each
# atomic, so the render thread never draws from a half-swapped
# cached_image / cached_array / total_scroll_width.
self._follower_incoming_image: deque = deque(maxlen=1)
self._follower_deadline: Optional[float] = None
# Leader: time.time() of the last follower frame sent.
self._last_follower_send = 0.0
@@ -559,21 +566,15 @@ class DisplayController:
# temporarily replace the leader's correct one.
# When the leader sends its scroll image (TCP), update our
# cached_array so both Pis have pixel-identical images.
# cached_array so both Pis have pixel-identical images. This runs
# on the sync TCP thread, so it only converts and queues the
# image; the render thread swaps it in between frames.
import numpy as _np
def _on_leader_scroll_image(image):
vc = self.vegas_coordinator
if vc and vc.render_pipeline:
rp = vc.render_pipeline
arr = _np.asarray(image.convert("RGB"), dtype=_np.uint8)
rp.scroll_helper.cached_image = image
rp.scroll_helper.cached_array = arr
rp.scroll_helper.total_scroll_width = image.width
self._follower_pending_new_image = False
logger.info(
"Sync: follower adopted leader scroll image %dx%d",
image.width, image.height,
)
self._follower_incoming_image.append((image, arr))
self.sync_manager.set_on_scroll_image(_on_leader_scroll_image)
if self.sync_manager.role == SyncRole.LEADER:
@@ -596,6 +597,29 @@ class DisplayController:
logger.error("Failed to initialize Vegas mode: %s", e, exc_info=True)
self.vegas_coordinator = None
def _adopt_follower_scroll_image(self, rp) -> None:
"""Swap in the leader's latest scroll image, on the render thread.
The sync TCP thread used to set cached_image, cached_array and
total_scroll_width one after another while this thread read them, so
a frame could slice the new array with the old width. It now queues
the image and this applies it between frames.
"""
try:
image, arr = self._follower_incoming_image.popleft()
except IndexError:
return
if rp is None:
return
rp.scroll_helper.cached_image = image
rp.scroll_helper.cached_array = arr
rp.scroll_helper.total_scroll_width = image.width
self._follower_pending_new_image = False
logger.info(
"Sync: follower adopted leader scroll image %dx%d",
image.width, image.height,
)
def _is_vegas_mode_active(self) -> bool:
"""Check if Vegas mode should be running."""
self._apply_pending_vegas_init()
@@ -1667,7 +1691,10 @@ class DisplayController:
if plugin_instance.has_live_content():
live_with_content.append(live_mode)
except Exception:
pass
# Treated as no live content; logged so a plugin whose
# check always raises is findable.
logger.debug("has_live_content() failed for %s", live_mode,
exc_info=True)
# Build mode list: live modes with content first, then other modes, then live modes without content
if live_with_content:
@@ -1904,7 +1931,9 @@ class DisplayController:
if bg_service and hasattr(bg_service, 'log_memory_stats'):
bg_service.log_memory_stats()
except Exception:
pass # Background service may not be initialized
# Background service may not be initialized
logger.debug("Background service memory stats unavailable",
exc_info=True)
# Log deferred updates stats
if hasattr(self.display_manager, '_scrolling_state'):
@@ -2106,6 +2135,7 @@ class DisplayController:
vc = self.vegas_coordinator
rp = vc.render_pipeline if (vc and vc.render_pipeline) else None
width = self.display_manager.width
self._adopt_follower_scroll_image(rp)
local_x = self._follower_local_x
if local_x is None:
@@ -2876,7 +2906,8 @@ class DisplayController:
try:
self.wifi_status_file.unlink()
except Exception:
pass
logger.debug("Could not remove WiFi status file %s",
self.wifi_status_file, exc_info=True)
return None
# Validate required fields
@@ -2909,7 +2940,8 @@ class DisplayController:
try:
self.wifi_status_file.unlink()
except Exception:
pass
logger.debug("Could not remove WiFi status file %s",
self.wifi_status_file, exc_info=True)
return None
# Message is valid and not expired — cache for the throttle window
+79 -35
View File
@@ -44,7 +44,9 @@ from src.display_geometry import (
DEFAULT_CHAIN_LENGTH, DEFAULT_COLS, DEFAULT_PARALLEL, DEFAULT_ROWS,
compose_pixel_mapper_config, physical_size, resolve_double_sided,
)
from src.matrix_support import MatrixSettingsRefused, library_refusals, refusal_message
from src.matrix_support import (
DEFAULT_REFRESH_LIMIT_HZ, MatrixSettingsRefused, library_refusals, refusal_message,
)
from src.pi5_matrix_support import is_raspberry_pi_5
import threading
import time
@@ -72,6 +74,10 @@ logger = get_logger(__name__)
#: and therefore get_font_height() -- report anything but 0.
_CALENDAR_FONT_PX = 7
#: Seconds between repeats of update_display()'s error log. It runs every
#: frame, so a fault that persists would otherwise log ~100 lines a second.
_UPDATE_ERROR_LOG_INTERVAL = 60.0
def _bdf_native_size(face) -> int:
"""The pixel height a BDF Face declares, or 0 if it does not say.
@@ -236,6 +242,11 @@ class DisplayManager:
_instance = None
# update_display()'s error-log throttle. Class defaults so instances built
# without __init__ (tests, doubles) have them too.
_update_error_logged_at: Optional[float] = None
_update_errors_suppressed = 0
def __new__(cls, *args, **kwargs):
if cls._instance is None:
cls._instance = super(DisplayManager, cls).__new__(cls)
@@ -345,6 +356,13 @@ class DisplayManager:
'max_deferred_updates': 50, # Limit queue size to prevent memory issues
'deferred_update_ttl': 300.0 # 5 minutes TTL for deferred updates
}
# Guards _scrolling_state['deferred_updates']. defer_update() is called
# from plugin update() on the update worker thread while
# process_deferred_updates() runs on the render thread, and both
# rebuild the list (TTL filter, [n:] slice) and assign it back -- an
# append landing between one side's read and its assignment was lost.
# Never held while a queued callable runs: those may defer again.
self._deferred_lock = threading.Lock()
self._setup_matrix()
logger.info("Matrix setup completed in %.3f seconds", time.time() - start_time)
@@ -976,7 +994,21 @@ class DisplayManager:
# Write a snapshot for the web preview (throttled)
self._write_snapshot_if_due(frame_checksum)
except Exception as e:
logger.error(f"Error updating display: {e}")
# Once with the traceback, then at most every
# _UPDATE_ERROR_LOG_INTERVAL with a count of what was skipped.
now = time.monotonic()
last = self._update_error_logged_at
if last is None:
self._update_error_logged_at = now
logger.error("Error updating display: %s", e, exc_info=True)
elif now - last >= _UPDATE_ERROR_LOG_INTERVAL:
skipped = self._update_errors_suppressed
self._update_error_logged_at = now
self._update_errors_suppressed = 0
logger.error("Error updating display: %s (%d more since the "
"last report)", e, skipped)
else:
self._update_errors_suppressed += 1
def _setup_scan_order_compensation(self) -> None:
"""Work out which rows to show a refresh behind while scrolling.
@@ -1547,7 +1579,7 @@ class DisplayManager:
options.panel_type = hardware_config.get('panel_type', '')
options.disable_hardware_pulsing = hardware_config.get('disable_hardware_pulsing', False)
options.show_refresh_rate = hardware_config.get('show_refresh_rate', False)
options.limit_refresh_rate_hz = hardware_config.get('limit_refresh_rate_hz', 90)
options.limit_refresh_rate_hz = hardware_config.get('limit_refresh_rate_hz', DEFAULT_REFRESH_LIMIT_HZ)
options.gpio_slowdown = runtime_config.get('gpio_slowdown', 3)
# Disable internal privilege dropping - we manage this via systemd or remain root
@@ -1593,7 +1625,7 @@ class DisplayManager:
value = float(hardware.get('limit_refresh_rate_hz') or 0)
except (TypeError, ValueError):
value = 0.0
return value if value > 0 else 100.0
return value if value > 0 else float(DEFAULT_REFRESH_LIMIT_HZ)
def _scrolling_now(self) -> bool:
"""Whether a scroll is running, without is_currently_scrolling()'s
@@ -1704,26 +1736,28 @@ class DisplayManager:
"""
current_time = time.time()
# Clean up expired updates before adding new ones
self._cleanup_expired_deferred_updates(current_time)
# Limit queue size to prevent memory issues
if len(self._scrolling_state['deferred_updates']) >= self._scrolling_state['max_deferred_updates']:
# Remove oldest update to make room
self._scrolling_state['deferred_updates'].pop(0)
logger.debug("Removed oldest deferred update due to queue size limit")
self._scrolling_state['deferred_updates'].append({
'func': update_func,
'priority': priority,
'timestamp': current_time
})
# Only sort if we have a reasonable number of updates to avoid excessive sorting
if len(self._scrolling_state['deferred_updates']) <= 20:
self._scrolling_state['deferred_updates'].sort(key=lambda x: x['priority'])
logger.debug(f"Deferred update added. Total deferred: {len(self._scrolling_state['deferred_updates'])}")
with self._deferred_lock:
# Clean up expired updates before adding new ones
self._cleanup_expired_deferred_updates(current_time)
# Limit queue size to prevent memory issues
if len(self._scrolling_state['deferred_updates']) >= self._scrolling_state['max_deferred_updates']:
# Remove oldest update to make room
self._scrolling_state['deferred_updates'].pop(0)
logger.debug("Removed oldest deferred update due to queue size limit")
self._scrolling_state['deferred_updates'].append({
'func': update_func,
'priority': priority,
'timestamp': current_time
})
# Only sort if we have a reasonable number of updates to avoid excessive sorting
if len(self._scrolling_state['deferred_updates']) <= 20:
self._scrolling_state['deferred_updates'].sort(key=lambda x: x['priority'])
queued = len(self._scrolling_state['deferred_updates'])
logger.debug(f"Deferred update added. Total deferred: {queued}")
def process_deferred_updates(self):
"""Process any deferred updates if not currently scrolling."""
@@ -1731,21 +1765,26 @@ class DisplayManager:
# Always clean up expired updates, even if scrolling
# This prevents memory leaks from accumulated expired updates
self._cleanup_expired_deferred_updates(current_time)
with self._deferred_lock:
self._cleanup_expired_deferred_updates(current_time)
if self.is_currently_scrolling():
return
if not self._scrolling_state['deferred_updates']:
return
# Process only a limited number of updates per call to avoid blocking
max_updates_per_call = min(5, len(self._scrolling_state['deferred_updates']))
updates_to_process = self._scrolling_state['deferred_updates'][:max_updates_per_call]
self._scrolling_state['deferred_updates'] = self._scrolling_state['deferred_updates'][max_updates_per_call:]
with self._deferred_lock:
if not self._scrolling_state['deferred_updates']:
return
# Process only a limited number of updates per call to avoid blocking
max_updates_per_call = min(5, len(self._scrolling_state['deferred_updates']))
updates_to_process = self._scrolling_state['deferred_updates'][:max_updates_per_call]
self._scrolling_state['deferred_updates'] = self._scrolling_state['deferred_updates'][max_updates_per_call:]
queued = len(self._scrolling_state['deferred_updates'])
logger.debug(f"Processing {len(updates_to_process)} deferred updates (queue size: {len(self._scrolling_state['deferred_updates'])})")
logger.debug(f"Processing {len(updates_to_process)} deferred updates (queue size: {queued})")
# The callables run outside the lock: they are plugin code of any
# length, and one that defers again would deadlock on it.
failed_updates = []
for update_info in updates_to_process:
try:
@@ -1764,10 +1803,15 @@ class DisplayManager:
# Re-add failed updates to the end of the queue (not the beginning)
if failed_updates:
self._scrolling_state['deferred_updates'].extend(failed_updates)
with self._deferred_lock:
self._scrolling_state['deferred_updates'].extend(failed_updates)
def _cleanup_expired_deferred_updates(self, current_time: float):
"""Remove expired deferred updates to prevent memory leaks."""
"""Remove expired deferred updates to prevent memory leaks.
Callers hold ``_deferred_lock``: this reads the list and assigns a
filtered copy back.
"""
ttl = self._scrolling_state['deferred_update_ttl']
initial_count = len(self._scrolling_state['deferred_updates'])
+33 -14
View File
@@ -48,6 +48,7 @@ import json
import logging
import math
import os
import threading
from collections import OrderedDict
from dataclasses import dataclass
from typing import Any, Dict, Optional, Tuple, Union
@@ -68,8 +69,10 @@ _FONTS_SUBDIR = os.path.join('assets', 'fonts')
_FALLBACK_FONT_NAME = 'PressStart2P-Regular.ttf'
# (resolved absolute path, requested size) -> (font face, realised size).
# BDF faces are stateful in principle, but the core's own FontManager shares
# faces the same way.
# TTF only: a BDF ``freetype.Face`` must never be shared between threads
# (FreeType does not allow it, and ``load_char`` rewrites the face's glyph
# slot), and this cache is process-wide. BDF faces come from
# ``load_bdf_face`` every time, which already caches them per thread.
#
# Bounded LRU rather than the unbounded dict this started as: the display
# process runs for weeks, and every config save can introduce a new
@@ -78,14 +81,28 @@ _FALLBACK_FONT_NAME = 'PressStart2P-Regular.ttf'
# every other hot cache (display_manager, font_manager, adaptive_layout).
_FONT_CACHE_MAX = 256
_font_cache: 'OrderedDict[Tuple[str, int], Tuple[Any, int]]' = OrderedDict()
# load_font is called from the display thread and from plugin update threads.
# A get() then move_to_end() pair on an unguarded OrderedDict raises KeyError
# when another thread evicts the key in between.
_font_cache_lock = threading.Lock()
def _cache_get(key: Tuple[str, int]) -> Optional[Tuple[Any, int]]:
"""The cached entry for ``key`` (marked most recently used), or None."""
with _font_cache_lock:
cached = _font_cache.get(key)
if cached is not None:
_font_cache.move_to_end(key)
return cached
def _cache_put(key: Tuple[str, int], value: Tuple[Any, int]) -> None:
"""Insert, evicting the least recently used entry past the bound."""
_font_cache[key] = value
_font_cache.move_to_end(key)
while len(_font_cache) > _FONT_CACHE_MAX:
_font_cache.popitem(last=False)
with _font_cache_lock:
_font_cache[key] = value
_font_cache.move_to_end(key)
while len(_font_cache) > _FONT_CACHE_MAX:
_font_cache.popitem(last=False)
# Config keys a style element block carries, in schema/UI order.
_STYLE_KEYS = ('font', 'font_size', 'text_color', 'visible', 'align')
@@ -222,14 +239,15 @@ def _load_font_sized(font_name: str, size: int) -> Tuple[Any, int]:
logger.warning("Font file not found: %s, using fallback", font_name)
return _load_fallback_font(size)
is_bdf = path.lower().endswith('.bdf')
cache_key = (path, size)
cached = _font_cache.get(cache_key)
if cached is not None:
_font_cache.move_to_end(cache_key)
return cached
if not is_bdf:
cached = _cache_get(cache_key)
if cached is not None:
return cached
try:
if path.lower().endswith('.bdf'):
if is_bdf:
font, effective = _load_bdf(path, size)
else:
font, effective = load_truetype(path, size), size
@@ -238,7 +256,9 @@ def _load_font_sized(font_name: str, size: int) -> Tuple[Any, int]:
path, size, e)
return _load_fallback_font(size)
_cache_put(cache_key, (font, effective))
# Not BDF: load_bdf_face caches those per thread (see _font_cache).
if not is_bdf:
_cache_put(cache_key, (font, effective))
return font, effective
@@ -247,9 +267,8 @@ def _load_fallback_font(size: int) -> Tuple[Any, int]:
path = resolve_font_path(_FALLBACK_FONT_NAME)
if path is not None:
cache_key = (path, size)
cached = _font_cache.get(cache_key)
cached = _cache_get(cache_key)
if cached is not None:
_font_cache.move_to_end(cache_key)
return cached
try:
entry = (load_truetype(path, size), size)
+7 -1
View File
@@ -498,6 +498,7 @@ class FontManager:
self.performance_stats["cache_misses"] += 1
# Load font
shareable = True
font_path = self.font_catalog.get(family)
if not font_path:
logger.warning(f"Font family '{family}' not found")
@@ -507,6 +508,7 @@ class FontManager:
try:
if font_path.endswith('.bdf'):
font = self._load_bdf_font(font_path, size_px)
shareable = False
else:
font = load_truetype(font_path, size_px)
except Exception as e:
@@ -516,7 +518,11 @@ class FontManager:
self.performance_stats["failed_loads"] += 1
font = ImageFont.load_default()
self.font_cache[cache_key] = font
# A BDF face is not cached here: font_cache is shared by every
# thread, and a freetype.Face must never be (see load_bdf_face, which
# already caches BDF faces per thread).
if shareable:
self.font_cache[cache_key] = font
return font
def _load_bdf_font(self, font_path: str, size_px: int) -> freetype.Face:
+8 -1
View File
@@ -74,6 +74,13 @@ MAPPING_OUTPUTS: Dict[str, int] = {
'classic-pi1': 1,
}
#: The refresh cap (``display.hardware.limit_refresh_rate_hz``) when config
#: omits it -- config/config.template.json's value. DisplayManager passes it to
#: the library and reports it as ``refresh_hz`` for scroll pacing, so the two
#: must be the same number: they were 90 and 100, and pacing solved against a
#: rate the panel was capped below.
DEFAULT_REFRESH_LIMIT_HZ = 100
#: What DisplayManager passes when a key is missing from display.hardware /
#: display.runtime. Config migration normally fills these from
#: config/config.template.json first, so they rarely apply.
@@ -81,7 +88,7 @@ DISPLAY_MANAGER_DEFAULTS: Dict[str, Any] = {
'rows': 32, 'cols': 64, 'chain_length': 2, 'parallel': 1,
'hardware_mapping': 'adafruit-hat-pwm', 'brightness': 90, 'pwm_bits': 10,
'pwm_lsb_nanoseconds': 150, 'led_rgb_sequence': 'RGB',
'row_address_type': 0, 'multiplexing': 0, 'limit_refresh_rate_hz': 90,
'row_address_type': 0, 'multiplexing': 0, 'limit_refresh_rate_hz': DEFAULT_REFRESH_LIMIT_HZ,
'gpio_slowdown': 3,
}
+2 -1
View File
@@ -16,6 +16,7 @@ from PIL import Image
from src.common.scroll_config import solve_crisp
from src.common.scroll_helper import ScrollHelper
from src.matrix_support import DEFAULT_REFRESH_LIMIT_HZ
from src.vegas_mode.config import VegasModeConfig
from src.vegas_mode.geometry import separation_gap
from src.vegas_mode.stream_manager import StreamManager
@@ -199,7 +200,7 @@ class RenderPipeline:
hz = float(getattr(self.display_manager, 'refresh_hz', 0) or 0)
except (TypeError, ValueError):
hz = 0.0
return hz if hz > 0 else 100.0
return hz if hz > 0 else float(DEFAULT_REFRESH_LIMIT_HZ)
def _refresh_hz(self) -> float:
"""The refresh to solve the crisp speed against: measured, else the cap."""
+9 -4
View File
@@ -74,8 +74,11 @@ class StreamManager:
# Segments composed into the current cycle (swap mode only).
self._active_buffer: Deque[ContentSegment] = deque()
# Reentrant: _prefetch_content releases and re-acquires it around the
# slow fetch while a caller may already hold it.
# Reentrant: get_next_segment holds it while calling
# _prefetch_content, which acquires it again. _prefetch_content's
# release() around the slow fetch only frees the lock when its caller
# did not already hold it (initialize); from get_next_segment the
# count only drops to 1, so the fetch runs with the lock held.
self._buffer_lock = threading.RLock()
# Plugin rotation, and the position of the next plugin to fetch in it.
@@ -536,7 +539,10 @@ class StreamManager:
plugin_id = self._ordered_plugins[self._prefetch_index]
# Release lock for potentially slow content fetch
# Release for the potentially slow content fetch. This frees
# the lock only when the caller did not hold it already
# (initialize); under get_next_segment's hold the RLock count
# just drops to 1 and other threads still wait.
self._buffer_lock.release()
try:
segment = self._fetch_plugin_content(plugin_id)
@@ -765,7 +771,6 @@ class StreamManager:
continue
if images:
self.stats['segments_fetched'] += 1
if images:
group.append((plugin_id, images))
else:
group.append((plugin_id, None if defer_empty else []))