fix(display): order snapshot writes, retry a failed one, stop the writer on cleanup; measure refresh from the low end

From review (CodeRabbit):

- A static frame's inline save now waits for a queued write in flight and
  drops a stale queued frame, under one write lock taken by both paths, so
  the last scrolling frame can no longer land on top of the first static one.
- A background write that fails clears the recorded digest, so an unchanged
  frame is written again instead of mtime-touching a stale file into looking
  healthy.
- cleanup() stops and joins the writer thread, dropping a frame it hasn't
  started, instead of leaving it (and its DisplayManager) alive.
- Vegas measures the panel's refresh from the 10th percentile of swap gaps,
  not the median: a late swap only lengthens a gap, so with most frames late
  in the window the median tracked the loop and locked in a rate too low
  (then the scroll ran fast until restart). Samples are dropped whenever
  the pacing is re-solved, so gaps from an old frame hold aren't divided by
  a new one.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Chuck
2026-09-24 18:28:42 -04:00
co-authored by Claude Opus 5.5
parent c43c10bd78
commit 00e14ea55f
4 changed files with 151 additions and 14 deletions
+43 -7
View File
@@ -199,6 +199,10 @@ class DisplayManager:
self._snapshot_cond = threading.Condition()
self._snapshot_pending: Optional[Image.Image] = None
self._snapshot_thread: Optional[threading.Thread] = None
self._snapshot_stop = False
# Held for the whole of each PNG write, by the writer thread and by
# the inline static path, so the two land on disk in order.
self._snapshot_write_lock = threading.Lock()
self._viewer_check_ts = 0.0
self._viewer_fresh = False
self._viewer_was_fresh = False
@@ -1274,6 +1278,8 @@ class DisplayManager:
def cleanup(self):
"""Clean up resources."""
if hasattr(self, '_snapshot_cond'):
self._stop_snapshot_writer()
if hasattr(self, 'matrix') and self.matrix is not None:
try:
self.matrix.Clear()
@@ -1611,7 +1617,12 @@ class DisplayManager:
if self.is_currently_scrolling():
self._queue_snapshot(self.image.copy())
else:
self._save_snapshot(self.image)
# A scroll that just ended can leave its last frame queued or
# mid-write; it must not land on top of this newer one.
with self._snapshot_write_lock:
with self._snapshot_cond:
self._snapshot_pending = None
self._save_snapshot(self.image)
self._last_snapshot_ts = now
self._last_snapshot_touch_ts = now
self._last_snapshot_digest = digest
@@ -1696,10 +1707,35 @@ class DisplayManager:
def _snapshot_writer(self) -> None:
while True:
with self._snapshot_cond:
while self._snapshot_pending is None:
while self._snapshot_pending is None and not self._snapshot_stop:
self._snapshot_cond.wait()
image, self._snapshot_pending = self._snapshot_pending, None
try:
self._save_snapshot(image)
except Exception as e:
self._log_snapshot_failure(e)
if self._snapshot_stop:
return # shutting down: a pending frame is dropped
# The write lock before the frame: whichever of this and an inline
# static save gets it first also writes first, and a static save
# clears the slot, so an older frame never lands on a newer one.
with self._snapshot_write_lock:
with self._snapshot_cond:
image, self._snapshot_pending = self._snapshot_pending, None
if image is None:
continue
try:
self._save_snapshot(image)
except Exception as e:
# The frame was recorded as written when it was queued.
# Forget that, so an unchanged frame is written again
# rather than only mtime-touching a stale file into
# looking healthy.
self._last_snapshot_digest = None
self._log_snapshot_failure(e)
def _stop_snapshot_writer(self, timeout: float = 1.0) -> None:
"""Stop the writer thread, dropping any frame it has not started."""
with self._snapshot_cond:
self._snapshot_stop = True
self._snapshot_pending = None
self._snapshot_cond.notify_all()
thread = self._snapshot_thread
if thread is not None and thread is not threading.current_thread():
thread.join(timeout)
self._snapshot_thread = None
+14 -7
View File
@@ -137,6 +137,8 @@ class RenderPipeline:
# a 95Hz panel, p99 21-28ms -- a visible hitch every few frames.
self._crisp = None
self._frame_hold = 1
# Gaps timed under the old hold would be divided by the new one.
self._swap_times.clear()
if self.config.smooth_scroll and not self.config.sub_pixel_blend:
self._crisp = solve_crisp(self.config.scroll_speed, self._refresh_hz())
self._frame_hold = self._crisp.frame_hold
@@ -197,10 +199,15 @@ class RenderPipeline:
refresh the panel never delivers -- 90px/s at "120Hz" is 3px every 4
refreshes, visibly jumpy, where the real 95Hz allows 1px every refresh.
SwapOnVSync blocks for frame_hold refreshes, so the median gap between
swaps is frame_hold refresh periods. The median ignores the odd frame
that missed its vsync or waited on a recompose. Measured once: the
refresh only changes with the hardware config, which restarts us.
SwapOnVSync blocks for frame_hold refreshes, and a swap can only come
back late -- a missed vsync lengthens its gap by whole refreshes, never
shortens one -- so the low end of the gaps is frame_hold refresh
periods: the 10th percentile, as src/common/frame_timing.py uses. The
median would track the render loop instead once most frames in the
window were late (startup, a prefetch, a recompose), lock in a rate
too low, and scroll faster than configured until restart. Measured
once: the refresh only changes with the hardware config, which
restarts us.
"""
if self._crisp is None or self._measured_hz is not None:
return
@@ -214,11 +221,11 @@ class RenderPipeline:
times = list(self._swap_times)
gaps = sorted(b - a for a, b in zip(times, times[1:]))
median = gaps[len(gaps) // 2]
period = gaps[len(gaps) // 10]
self._swap_times.clear()
if median <= 0:
if period <= 0:
return
measured = self._frame_hold / median
measured = self._frame_hold / period
cap = self._cap_hz()
if measured >= cap * (1.0 - self.REFRESH_TOLERANCE):
self._measured_hz = cap