feat(perf): frame timing for every presented frame, a soak tool, a render bench and a stall watchdog (#629)

src/common/frame_timing.py times every frame the display presents, whoever drew it, and writes cumulative counters to /dev/shm. scripts/frame_soak.py grades a running service (late frames, freezes, where the time goes) and scripts/render_bench.py the hardware and render path alone. A stall watchdog logs the stacks behind any scroll held up for 250 ms or more (LEDMATRIX_STALL_WATCHDOG_MS lowers that). See docs/SCROLL_PERFORMANCE.md, "Soaking a rig".

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Chuck
2026-09-24 19:38:49 -04:00
committed by GitHub
co-authored by Claude Opus 5.5
parent 7f9c73e9aa
commit 8ad9d191a7
9 changed files with 2217 additions and 13 deletions
+389
View File
@@ -0,0 +1,389 @@
#!/usr/bin/env python3
"""Soak a running display and report how often moving frames reached the panel late.
Runs NEXT TO the display service, as any user: it only reads the stats file the
service writes (src/common/frame_timing.py) at the start and end of the run and
reports the difference. Nothing is stopped, restarted or drawn.
# 10 minutes, as the display is now
python3 scripts/frame_soak.py
# the same with the web preview open (the preview's PNG encodes are one of
# the things that used to make the render loop miss refreshes)
python3 scripts/frame_soak.py --preview
# quick look at the totals since the service started
python3 scripts/frame_soak.py --show
# keep the report for a before/after comparison
python3 scripts/frame_soak.py --duration 600 --json soak-before.json
Exit status: 0 when the late-frame rate is within ``--max-late-pct``, 1 when it
is not, 2 when there was nothing to measure (no stats file, the service
restarted mid-run, or nothing scrolled).
What the numbers mean
---------------------
late frames frames that reached the panel one or more refreshes after they
were due -- the panel showed the previous frame again, which on
a moving strip is a visible hitch. This is the pass/fail number.
freezes gaps of 250ms+ inside a scroll: recomposes, plugin handovers,
blocking calls on the render thread. Reported, not failed on,
since some are handovers between plugins rather than faults.
blit copying the frame into the matrix canvas (rgbmatrix SetImage).
Grows with width x height x pwm_bits.
wait blocked in SwapOnVSync, i.e. slack before the refresh.
work everything else between two frames: drawing, scrolling, and
waiting for the GIL.
"""
from __future__ import annotations
import argparse
import json
import os
import sys
import time
from pathlib import Path
from typing import Any, Dict, Optional
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from src.common.frame_timing import ( # noqa: E402
BUCKET_COUNT,
SCHEMA_VERSION,
default_stats_path,
)
#: Touched by the web UI while someone has the preview open; a fresh marker
#: puts the display service's snapshot writer at full rate. Same path as
#: DisplayManager._viewer_marker_path.
VIEWER_MARKER = "/tmp/led_matrix_preview_viewer" # nosec B108 - fixed path shared with the service
#: A stats file not rewritten for this long means nothing is being presented.
STALE_SECONDS = 30.0
def load(path: str) -> Optional[Dict[str, Any]]:
try:
with open(path, encoding="utf-8") as handle:
stats = json.load(handle)
except (OSError, ValueError):
return None
if not isinstance(stats, dict) or stats.get("version") != SCHEMA_VERSION:
return None
return stats
def _histogram(stats: Dict[str, Any], name: str) -> Dict[int, int]:
raw = (stats.get("histograms") or {}).get(name) or {}
return {int(k): int(v) for k, v in raw.items()}
def diff(before: Dict[str, Any], after: Dict[str, Any]) -> Dict[str, Any]:
"""What happened between two snapshots of the same process."""
tb, ta = before["totals"], after["totals"]
totals = {}
for key, value in ta.items():
if isinstance(value, dict):
totals[key] = {k: v - tb.get(key, {}).get(k, 0)
for k, v in value.items()}
elif key == "worst_interval_ms":
# A running maximum can't be differenced; it is reported as the
# worst since the service started.
totals[key] = value
else:
totals[key] = value - tb.get(key, 0)
histograms = {}
for name in (after.get("histograms") or {}):
hb, ha = _histogram(before, name), _histogram(after, name)
histograms[name] = {k: v - hb.get(k, 0) for k, v in ha.items()
if v - hb.get(k, 0) > 0}
return {"totals": totals, "histograms": histograms,
"seconds": after["updated"] - before["updated"]}
def percentiles(histogram: Dict[int, int], bucket_ms: float) -> Dict[str, Any]:
"""p50/p95/p99/max from a sparse histogram, as each bucket's upper edge."""
count = sum(histogram.values())
if not count:
return {}
out = {}
targets = {"p50": 0.50, "p95": 0.95, "p99": 0.99}
running = 0
for index in sorted(histogram):
running += histogram[index]
for name, fraction in list(targets.items()):
if running >= fraction * count:
out[name] = _edge(index, bucket_ms)
del targets[name]
out["max"] = _edge(max(histogram), bucket_ms)
return out
def _edge(index: int, bucket_ms: float):
if index >= BUCKET_COUNT - 1:
return f">={index * bucket_ms:g}"
return round((index + 1) * bucket_ms, 2)
def build_report(before, after, preview: bool) -> Dict[str, Any]:
delta = diff(before, after)
totals = delta["totals"]
frames = totals["scroll_frames"]
# The rates are over frames judged against a known refresh period. Stats
# from a recorder that predates the count fall back to every frame.
timed = totals.get("timed_frames", frames) if "timed_frames" in totals else frames
hours = delta["seconds"] / 3600.0 if delta["seconds"] > 0 else 0.0
bucket_ms = after.get("bucket_ms", 0.25)
report = {
"seconds": round(delta["seconds"], 1),
"preview": preview,
"info": after.get("info"),
"binding_releases_gil": after.get("binding_releases_gil"),
"measured_refresh_hz": after.get("measured_refresh_hz"),
"scroll_frames": frames,
"static_frames": totals["static_frames"],
"late_frames": totals["late_frames"],
"timed_frames": timed,
"late_pct": round(100.0 * totals["late_frames"] / timed, 3) if timed else None,
"missed_refreshes": totals["missed_refreshes"],
"late_by": totals["late_by"],
"early_frames": totals.get("early_frames", 0),
"early_pct": (round(100.0 * totals.get("early_frames", 0) / timed, 3)
if timed else None),
"freeze_by": totals.get("freeze_by", {}),
"freezes": totals["freezes"],
"freezes_per_hour": round(totals["freezes"] / hours, 1) if hours else None,
"freeze_seconds": round(totals["freeze_seconds"], 2),
"worst_interval_ms": (round(totals["worst_interval_ms"], 1)
if totals["worst_interval_ms"] else None),
"timing_ms": {name: percentiles(h, bucket_ms)
for name, h in delta["histograms"].items()},
}
# The rate the panel held while rendering: the typical frame's interval
# per refresh held. A few percent under the idle rate is normal (the Pi is
# bit-banging the panel and pushing frames at once); a widening gap between
# the two is a render-cost regression even when nothing is late.
typical = (report["timing_ms"].get("interval_per_hold") or {}).get("p50")
# percentiles() reports a bucket's upper edge; the midpoint is the better
# estimate, and half a 0.25ms bucket is already ~1% at 100Hz -- the size
# of the idle-vs-held gap this number exists to show.
if isinstance(typical, (int, float)) and typical > bucket_ms / 2:
report["held_refresh_hz"] = round(1000.0 / (typical - bucket_ms / 2), 1)
else:
report["held_refresh_hz"] = None
return report
def print_report(report: Dict[str, Any], limit: float) -> None:
info = report.get("info") or {}
size = "{}x{}".format(
(info.get("cols") or 0) * (info.get("chain_length") or 1),
(info.get("rows") or 0) * (info.get("parallel") or 1))
gil = {True: "releases the GIL", False: "STOCK (holds the GIL in SwapOnVSync)",
None: "unknown"}[report.get("binding_releases_gil")]
print(f"Rig {info.get('pi_model') or 'unknown'}")
print(f"Panel {size} chain {info.get('chain_length')} x parallel "
f"{info.get('parallel')} pwm_bits {info.get('pwm_bits')} "
f"slowdown {info.get('gpio_slowdown')} mapping {info.get('hardware_mapping')}")
print(f"Refresh {report.get('measured_refresh_hz') or '?'} Hz measured, "
f"cap {info.get('limit_refresh_rate_hz')}")
print(f"Binding {gil}")
print(f"Run {report['seconds']:.0f}s, preview "
f"{'open (simulated)' if report['preview'] else 'as-is'}")
print()
frames = report["scroll_frames"]
print(f"Scrolling frames {frames}")
if frames:
late_by = report["late_by"]
print(f"Late frames {report['late_frames']} ({report['late_pct']}%)"
f" missed refreshes {report['missed_refreshes']}"
f" [by 1: {late_by['1']}, 2: {late_by['2']}, "
f"3-5: {late_by['3-5']}, 6+: {late_by['6+']}]")
if report["early_frames"]:
print(f"Early frames {report['early_frames']} "
f"({report['early_pct']}%) swaps returned a refresh early")
print(f"Freezes >=250ms {report['freezes']}"
f" ({report['freezes_per_hour']}/h, {report['freeze_seconds']}s total)"
f" worst gap since start {report['worst_interval_ms'] or '-'} ms")
if report["freezes"]:
print(" by length: " + ", ".join(
f"{k}: {v}" for k, v in report["freeze_by"].items()))
print()
print(f"{'ms':<18}{'p50':>8}{'p95':>8}{'p99':>8}{'max':>8}")
for name in ("blit", "wait", "work", "interval_per_hold"):
row = report["timing_ms"].get(name) or {}
print(f"{name:<18}" + "".join(f"{str(row.get(k, '-')):>8}"
for k in ("p50", "p95", "p99", "max")))
print()
if report["late_pct"] is None:
print("RESULT nothing scrolled - no verdict")
elif not locked(report, limit):
ceiling = refresh_ceiling(report)
if (report.get("early_pct") or 0.0) > limit:
why = (f"{report['early_pct']}% of frames came a refresh early, so the "
"swaps were not waiting for the panel")
else:
why = (f"frames arrived at {report['measured_refresh_hz']}Hz, faster than "
f"the panel can refresh ({ceiling:g}Hz)")
print(f"RESULT FAIL NOT LOCKED: {why}, and the late count means nothing")
elif report["late_pct"] <= limit:
print(f"RESULT PASS {report['late_pct']}% late <= {limit}%")
else:
print(f"RESULT FAIL {report['late_pct']}% late > {limit}%")
#: How far over the panel's rate frames may arrive before the loop cannot have
#: been waiting for it. The margin covers the refresh wandering a little.
CEILING_MARGIN = 1.05
def refresh_ceiling(report: Dict[str, Any]) -> Optional[float]:
"""The fastest the panel can refresh, as far as this run knows.
The benchmark measures it (``idle_refresh_hz``); the service only knows its
cap. With neither, there is no ceiling to check against.
"""
idle = report.get("idle_refresh_hz")
if idle:
return float(idle)
cap = (report.get("info") or {}).get("limit_refresh_rate_hz")
try:
cap = float(cap)
except (TypeError, ValueError):
return None
return cap if cap > 0 else None
def locked(report: Dict[str, Any], limit: float) -> bool:
"""Whether the loop was paced by the panel at all.
Two ways it is not. Frames a whole refresh early mean some swaps did not
wait. And a loop that never waited at all -- the dirty-tracking skip firing
mid-scroll let one free-run at 827fps -- looks self-consistent to a refresh
estimate taken from its own frames, so nothing registers as early; what
gives it away is a "refresh" faster than the panel can physically do.
"""
if (report.get("early_pct") or 0.0) > limit:
return False
ceiling = refresh_ceiling(report)
measured = report.get("measured_refresh_hz")
return not (ceiling and measured and measured > ceiling * CEILING_MARGIN)
def passed(report: Dict[str, Any], limit: float) -> bool:
return (report["late_pct"] is not None and locked(report, limit)
and report["late_pct"] <= limit)
def touch_marker() -> bool:
try:
with open(VIEWER_MARKER, "a"):
pass
os.utime(VIEWER_MARKER, None)
return True
except OSError:
return False
def wait_for_fresh(path: str, timeout: float) -> Optional[Dict[str, Any]]:
"""The first snapshot written after now, so both ends of the run are exact."""
first = load(path)
deadline = time.time() + timeout
while time.time() < deadline:
current = load(path)
if current and (first is None or current["updated"] != first["updated"]):
return current
time.sleep(0.5)
return None
def main(argv=None) -> int:
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
parser.add_argument("--duration", type=float, default=600.0,
help="seconds to soak (default 600)")
parser.add_argument("--preview", action="store_true",
help="keep the web-preview viewer marker fresh, as an "
"open preview tab does")
parser.add_argument("--max-late-pct", type=float, default=0.1,
help="fail above this percentage of late frames (default 0.1)")
parser.add_argument("--stats", default=default_stats_path(),
help="stats file written by the display service")
parser.add_argument("--json", metavar="PATH",
help="also write the report as JSON")
parser.add_argument("--show", action="store_true",
help="print totals since the service started and exit")
args = parser.parse_args(argv)
current = load(args.stats)
if current is None:
print(f"No frame stats at {args.stats}. Is the display service running a "
"build with frame timing, and has anything scrolled for ~10s?",
file=sys.stderr)
return 2
if time.time() - current["updated"] > STALE_SECONDS:
print(f"Frame stats are {time.time() - current['updated']:.0f}s old: nothing "
"has been presented recently (static screen, or the service stopped).",
file=sys.stderr)
if not args.show:
return 2
if args.show:
empty = json.loads(json.dumps(current))
for key, value in empty["totals"].items():
empty["totals"][key] = ({k: 0 for k in value} if isinstance(value, dict)
else 0)
empty["histograms"] = {}
empty["updated"] = current["started"]
report = build_report(empty, current, preview=False)
print_report(report, args.max_late_pct)
return 0
if args.preview and not touch_marker():
print(f"Cannot touch {VIEWER_MARKER}; run as the web service's user to "
"simulate an open preview.", file=sys.stderr)
return 2
print(f"Waiting for a fresh baseline from {args.stats} ...", flush=True)
before = wait_for_fresh(args.stats, timeout=60.0)
if before is None:
print("The stats file stopped updating.", file=sys.stderr)
return 2
end = time.time() + args.duration
next_progress = time.time() + 60.0
while time.time() < end:
if args.preview:
touch_marker()
time.sleep(1.0)
if time.time() >= next_progress:
now = load(args.stats)
if now and now.get("pid") == before["pid"]:
done = now["totals"]["scroll_frames"] - before["totals"]["scroll_frames"]
late = now["totals"]["late_frames"] - before["totals"]["late_frames"]
print(f" {int(end - time.time())}s left: {done} scrolling frames, "
f"{late} late", flush=True)
next_progress += 60.0
after = wait_for_fresh(args.stats, timeout=60.0)
if after is None:
print("The stats file stopped updating during the run.", file=sys.stderr)
return 2
if after.get("pid") != before.get("pid"):
print("The display service restarted during the run; results discarded.",
file=sys.stderr)
return 2
report = build_report(before, after, preview=args.preview)
print()
print_report(report, args.max_late_pct)
if args.json:
with open(args.json, "w", encoding="utf-8") as handle:
json.dump(report, handle, indent=2)
if report["late_pct"] is None:
return 2
return 0 if passed(report, args.max_late_pct) else 1
if __name__ == "__main__":
sys.exit(main())
+404
View File
@@ -0,0 +1,404 @@
#!/usr/bin/env python3
"""Benchmark the render loop against the panel's real refresh rate.
The question this answers is the one that decides whether a rig ships: *does
every frame present on the refresh it was meant to?* It drives the production
path -- a real ``DisplayManager`` and ``ScrollHelper``, the same crisp speed
resolver every ticker uses -- scrolls a synthetic strip for a while, and grades
it with the same frame-timing recorder the display service uses
(``src.common.frame_timing``), printing the same report as
``scripts/frame_soak.py``. A run passes when the loop was genuinely locked to
the panel and no more than ``--max-late-pct`` percent of frames were late.
Where frame_soak.py measures the service as it runs -- live content, plugin
updates, the web preview -- this measures the hardware and the render path
with nothing else in the way, on content that is identical every run. That is
what makes it the tool for comparing rigs (a Pi 3 against a Pi 4, one HAT
against another) and for A/B testing a change to the render path.
# stop the service first; it owns the GPIO
sudo systemctl stop ledmatrix
sudo python3 scripts/render_bench.py # 60s, default speed
sudo python3 scripts/render_bench.py --seconds 600 # the 10-minute gate
sudo python3 scripts/render_bench.py --speed 50 # a slower, held speed
sudo python3 scripts/render_bench.py --busy 2 # with background load
sudo python3 scripts/render_bench.py --json /tmp/pi4.json
sudo systemctl start ledmatrix
Like scripts/scroll_speeds.py, this never starts or stops the service itself,
so a crash here can never leave the panel dark.
Exit status is 0 when the run clears the gate, 1 when it does not, and 2 when
the run could not be set up (no hardware, no root, unusable config) -- so a rig
that cannot be measured is never mistaken for a rig that passed.
"""
from __future__ import annotations
import argparse
import json
import logging
import os
import sys
import threading
import time
import zlib
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from src.common import frame_timing, scroll_config # noqa: E402
sys.path.insert(0, str(Path(__file__).resolve().parent))
import frame_soak # noqa: E402 (same report, same verdict as the soak)
REPO = Path(__file__).resolve().parent.parent
CONFIG = REPO / "config" / "config.json"
#: Long enough to average out a scheduler hiccup, short enough that nobody
#: skips running it. The shipping gate is --seconds 600.
DEFAULT_SECONDS = 60.0
#: Seconds spent timing bare swaps before the scroll starts. The measurement
#: has to settle, but every second here is a second not scrolling.
MEASURE_SECONDS = 4.0
#: Scrolling discarded before the graded run starts: the first frames carry
#: first-touch costs and the scrolling state settling.
WARMUP_SECONDS = 2.0
def load_config() -> dict:
"""The config the display service would run with."""
try:
from src.config_manager import ConfigManager
config = ConfigManager().config
if isinstance(config, dict) and config:
return config
except Exception as exc: # noqa: BLE001 - any failure means use the plain read
print(f"ConfigManager unavailable ({exc}); reading {CONFIG} directly",
file=sys.stderr)
# ConfigManager pulls in a lot; a plain read is enough to drive the panel
# and keeps the benchmark usable on a half-installed machine.
try:
with open(CONFIG, encoding="utf-8") as handle:
config = json.load(handle)
except (OSError, ValueError) as exc:
sys.exit(f"could not read {CONFIG}: {exc}")
if not isinstance(config, dict):
sys.exit(f"{CONFIG} is not a config object")
return config
def build_strip(width: int, height: int, label: str):
"""A marquee strip a few screens wide, with text and colour.
Deliberately not plain white text on black: how long ``SetImage`` takes
depends on how many subpixels are lit, so a strip that is mostly dark
flatters the panel and hides exactly the regression this benchmark exists
to catch.
"""
from PIL import Image, ImageDraw, ImageFont
from src.common.font_layout import load_truetype
font = None
for path, size in (
(str(REPO / "assets/fonts/PressStart2P-Regular.ttf"), max(8, height // 4)),
("/usr/share/fonts/truetype/dejavu/DejaVuSansMono-Bold.ttf", max(10, height // 2)),
):
try:
font = load_truetype(path, size)
break
except OSError:
continue
if font is None:
font = ImageFont.load_default()
text = f" {label} *** THE QUICK BROWN FOX JUMPS OVER THE LAZY DOG *** "
probe = ImageDraw.Draw(Image.new("RGB", (8, 8)))
box = probe.textbbox((0, 0), text, font=font)
text_width = max(1, box[2] - box[0])
text_height = box[3] - box[1]
reps = max(2, (width * 4) // text_width + 1)
strip = Image.new("RGB", (text_width * reps, height), (0, 0, 0))
draw = ImageDraw.Draw(strip)
draw.fontmode = "1" # the panel has no partial brightness; see DisplayManager
palette = [(255, 210, 60), (80, 200, 255), (255, 90, 90), (140, 255, 140)]
for i in range(reps):
left = i * text_width
# A filled block per repeat, so a meaningful share of the strip is lit.
draw.rectangle(
[left + 4, height - 4, left + text_width - 4, height - 2],
fill=palette[i % len(palette)],
)
draw.text((left, (height - text_height) // 2 - box[1]), text,
font=font, fill=palette[(i + 1) % len(palette)])
return strip
class BackgroundLoad:
"""Threads that imitate plugins updating while the panel scrolls.
Not a simulation of any particular plugin -- it is the shape of the work
that competes with the render loop for the GIL: decoding JSON, resizing an
image, compressing bytes. A render loop that only holds its pacing on an
idle machine is not shippable, and this is how that shows up.
"""
def __init__(self, workers: int) -> None:
self.workers = max(0, workers)
self._stop = threading.Event()
self._threads: list = []
def __enter__(self) -> "BackgroundLoad":
for index in range(self.workers):
thread = threading.Thread(
target=self._run, args=(index,), name=f"bench-load-{index}", daemon=True)
thread.start()
self._threads.append(thread)
return self
def __exit__(self, *exc_info) -> None:
self._stop.set()
for thread in self._threads:
thread.join(timeout=2.0)
def _run(self, index: int) -> None:
from PIL import Image
payload = json.dumps({"games": [{"id": n, "score": [n, n + 1],
"name": f"team {n}"} for n in range(200)]})
image = Image.new("RGB", (256, 64), (12, 34, 56))
while not self._stop.wait(0.25 + 0.05 * index):
json.loads(payload)
image.resize((128, 32), Image.LANCZOS)
zlib.compress(image.tobytes(), 1)
def main(argv=None) -> int:
parser = argparse.ArgumentParser(
description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("--seconds", type=float, default=DEFAULT_SECONDS,
help=f"how long to scroll for (default {DEFAULT_SECONDS:.0f}; "
"the shipping gate is 600)")
parser.add_argument("--speed", type=float, default=None,
help="requested px/s; snapped to the nearest speed the "
"panel can show in whole pixels (default: one pixel "
"per refresh)")
parser.add_argument("--hz", type=float, default=None,
help="skip the idle measurement and take this as the "
"panel's rate (for reproducing a rig's numbers)")
parser.add_argument("--busy", type=int, default=0, metavar="N",
help="run N background workers imitating plugin updates")
parser.add_argument("--max-late-pct", "--max-missed", dest="max_late_pct",
type=float, default=0.1, metavar="PCT",
help="fail above this percentage of late frames (default 0.1)")
parser.add_argument("--json", dest="json_path", default=None, metavar="PATH",
help="also write the report as JSON, for comparing rigs")
parser.add_argument("--label", default=None,
help="name for this run in the JSON report (default: hostname)")
args = parser.parse_args(argv)
# Everything the display service logs would otherwise land in the middle of
# the report; the benchmark's own output is the point. The stall watchdog
# is the exception: a stack dump naming what held a frame up belongs here.
logging.basicConfig(level=logging.ERROR, stream=sys.stderr)
logging.getLogger("src.common.frame_timing").setLevel(logging.WARNING)
if hasattr(os, "geteuid") and os.geteuid() != 0:
print("this needs root for GPIO access - rerun with sudo", file=sys.stderr)
return 2
config = load_config()
from src.common.scroll_helper import ScrollHelper
from src.display_manager import DisplayManager
try:
display = DisplayManager(config, suppress_test_pattern=True)
except Exception as exc:
print(f"could not open the display ({exc}).\n"
"If the display service is running it owns the GPIO - stop it "
"first:\n sudo systemctl stop ledmatrix", file=sys.stderr)
return 2
if getattr(display, "matrix", None) is None:
print("the display came up in fallback mode - there is no panel here to "
"measure, and a software loop's frame times say nothing about "
"vsync. Run this on a rig.", file=sys.stderr)
return 2
width, height = display.width, display.height
if args.hz is not None:
idle_hz = float(args.hz)
print(f"taking the panel's rate as {idle_hz:.1f}Hz (given, not measured)")
else:
print(f"measuring the panel for {MEASURE_SECONDS:.0f}s...", flush=True)
idle_hz = frame_timing.measure_refresh_hz(display.matrix, MEASURE_SECONDS)
if idle_hz <= 0:
print("the panel did not answer a swap; cannot measure it",
file=sys.stderr)
return 2
cap = scroll_config.refresh_hz_from_config(config)
note = (f" (cap is {cap:.0f}Hz)" if idle_hz < cap * 0.98
else " (at its configured cap)")
print(f"panel refreshes at {idle_hz:.1f}Hz{note}")
requested = args.speed if args.speed else idle_hz
# Configured through the shared resolver rather than by setting the helper
# up by hand, so the benchmark measures the engine every ticker runs on. A
# speed the bench reached some other way would be measuring something no
# plugin does.
helper = ScrollHelper(width, height)
settings = scroll_config.configure(
helper,
plugin_config={"scroll_pixels_per_second": requested},
global_config=config,
refresh_hz=idle_hz,
display_manager=display,
)
choice = settings.crisp
if choice is None:
print("the resolver did not snap to a whole-pixel speed; nothing to "
"grade against", file=sys.stderr)
return 2
print(f"asked for {requested:.1f} px/s -> {choice.describe()}")
helper.set_sub_pixel_scrolling(False)
helper.set_scrolling_image(
build_strip(width, height, f"{choice.pixels_per_second:.0f} px/s"))
# The display service's own recorder, owned outright here: never flushed to
# the service's stats file, drained exactly at the start and end of the
# graded run, and seeded with the idle rate so a loop that never locked
# (free-running, or stuck at a fraction of the refresh) shows as early or
# late frames instead of looking self-consistent.
recorder = frame_timing.FrameTimingRecorder(
flush_interval=float("inf"),
info=display._frame_timing_info(), # pylint: disable=protected-access
refresh_hz=idle_hz,
)
recorder.scrolling_now = display._scrolling_now # pylint: disable=protected-access
display.frame_timing = recorder
print(f"scrolling {width}x{height} for {args.seconds:.0f}s"
+ (f" with {args.busy} background worker(s)" if args.busy else "")
+ " ...", flush=True)
frames = 0
duplicates = 0
blanks = 0
restarts = 0
last_column = None
before = None
started = time.perf_counter()
run_started = None
try:
with BackgroundLoad(args.busy):
while True:
now = time.perf_counter()
if run_started is None and now - started >= WARMUP_SECONDS:
recorder.drain()
before = recorder.snapshot()
run_started = now
frames = duplicates = blanks = restarts = 0
if run_started is not None and now - run_started >= args.seconds:
break
helper.update_scroll_position()
if helper.is_scroll_complete():
# The helper parks at the end of the strip and stops
# advancing, exactly as it does under a plugin -- which
# then hands over to the next one. Here there is nothing
# to hand over to, so start the strip again. Without this
# the benchmark measures a still image for the rest of the
# run and reports a smoothness it never demonstrated.
helper.reset_scroll()
restarts += 1
visible = helper.get_visible_portion()
column = int(helper.scroll_position)
if column == last_column:
duplicates += 1
last_column = column
if visible is None:
blanks += 1
else:
display.image.paste(visible, (0, 0))
# Every frame, not once before the loop. The scrolling state
# expires on its own inactivity threshold and takes the frame
# hold with it, so a scroll that announces itself once is
# presented at the wrong rate for all but its first moments --
# and its unchanged frames start taking the dirty-tracking
# skip, which returns without waiting for the panel at all.
# Every ticker re-announces per frame; so does this.
display.set_scrolling_state(True, frame_hold=choice.frame_hold)
display.update_display()
frames += 1
except KeyboardInterrupt:
print("\ninterrupted - reporting what was measured so far")
finally:
display.set_scrolling_state(False)
try:
display.clear()
except Exception as exc: # noqa: BLE001 - a lit panel is harmless; say so and go on
print(f"could not blank the panel: {exc}", file=sys.stderr)
if before is None:
print("interrupted during warm-up; nothing was graded", file=sys.stderr)
return 2
recorder.drain()
report = frame_soak.build_report(before, recorder.snapshot(), preview=False)
report["idle_refresh_hz"] = round(idle_hz, 2)
print()
frame_soak.print_report(report, args.max_late_pct)
held = report.get("held_refresh_hz")
if held:
drop = 100.0 * (idle_hz - held) / idle_hz
print(f"\npanel held ~{held:.1f}Hz while rendering, {drop:.1f}% below its "
f"{idle_hz:.1f}Hz idle rate (a widening gap is a render-cost "
"regression even with nothing late)")
if duplicates:
# A frame that shows the same columns as the one before it is work the
# panel did not need. It is not a miss -- the frame arrived on time --
# but it means the loop is presenting faster than the strip is moving.
print(f"duplicate {duplicates} frames advanced no pixels "
f"({100.0 * duplicates / max(1, frames):.2f}%)")
if blanks:
print(f"blank {blanks} frames had no visible slice to draw")
if restarts:
print(f"restarts {restarts} (the strip was scrolled through "
f"{restarts} time{'s' if restarts != 1 else ''})")
if args.json_path:
report.update({
"label": args.label or os.uname().nodename,
"bench": True,
"requested_pixels_per_second": requested,
"pixels_per_second": choice.pixels_per_second,
"pixels_per_frame": choice.pixels_per_frame,
"frame_hold": choice.frame_hold,
"busy_workers": args.busy,
"duplicate_frames": duplicates,
"blank_frames": blanks,
"strip_restarts": restarts,
"max_late_pct": args.max_late_pct,
"passed": frame_soak.passed(report, args.max_late_pct),
})
Path(args.json_path).write_text(json.dumps(report, indent=2) + "\n",
encoding="utf-8")
print(f"\nwrote {args.json_path}")
if report["late_pct"] is None:
return 2
return 0 if frame_soak.passed(report, args.max_late_pct) else 1
if __name__ == "__main__":
sys.exit(main())
+6 -13
View File
@@ -42,7 +42,7 @@ from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from src.common import scroll_config # noqa: E402
from src.common import frame_timing, scroll_config # noqa: E402
CONFIG = Path(__file__).resolve().parent.parent / "config" / "config.json"
@@ -99,20 +99,13 @@ def open_matrix(config, refresh_override=None):
def measure_refresh(config, seconds=6.0):
"""Actual refresh rate, by running uncapped and timing the swaps.
SwapOnVSync blocks until the panel's next refresh, so an unthrottled loop
runs at exactly the panel's rate. This is what an older Pi or a longer
chain will really give you, as opposed to whatever limit_refresh_rate_hz
optimistically asks for.
What an older Pi or a longer chain will really give you, as opposed to
whatever limit_refresh_rate_hz optimistically asks for. The timing loop
itself lives in src.common.frame_timing so the benchmark grades against
the same measurement this ladder is built from.
"""
matrix = open_matrix(config, refresh_override=0)
canvas = matrix.CreateFrameCanvas()
canvas = matrix.SwapOnVSync(canvas) # discard the first, it includes setup
frames = 0
started = time.perf_counter()
while time.perf_counter() - started < seconds:
canvas = matrix.SwapOnVSync(canvas)
frames += 1
measured = frames / (time.perf_counter() - started)
measured = frame_timing.measure_refresh_hz(matrix, seconds)
matrix.Clear()
return measured