mirror of
https://github.com/ChuckBuilds/LEDMatrix.git
synced 2026-10-05 14:55:08 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e745ae8060 | ||
|
|
a669d781f5 | ||
|
|
c20c0beac2 | ||
|
|
6fb2dc3595 | ||
|
|
b638b91169 | ||
|
|
bb475a79ea | ||
|
|
3f920f2899 | ||
|
|
0577c807eb | ||
|
|
5a7893b11a | ||
|
|
3866aa4519 | ||
|
|
26cae3e5d6 | ||
|
|
294bd522bd | ||
|
|
80048fe78e | ||
|
|
378478124c | ||
|
|
d447cdd965 | ||
|
|
ee7c3389f9 | ||
|
|
caec9f5bf5 | ||
|
|
a74b5a2f0f | ||
|
|
57d7df6705 | ||
|
|
e09e251553 | ||
|
|
7026eeb156 | ||
|
|
2236ff3081 | ||
|
|
5ad5e9aa59 | ||
|
|
8a0cce1aaf | ||
|
|
0b039c875f | ||
|
|
07abd87d5e | ||
|
|
e32d177cbd | ||
|
|
d18e4d3c9d | ||
|
|
04f0d8d134 | ||
|
|
064b9c9912 | ||
|
|
a6e9e3ef1c | ||
|
|
41192b9588 | ||
|
|
ef69201770 | ||
|
|
ed0753c1ca | ||
|
|
7bb85c0356 | ||
|
|
0d179fdf12 | ||
|
|
ee789775b4 | ||
|
|
05deb1ee7d | ||
|
|
f841fa36b6 | ||
|
|
515248b34e | ||
|
|
6bf7c3fa51 | ||
|
|
76ad71d1c5 | ||
|
|
a5ec645d25 | ||
|
|
084697346b | ||
|
|
a7b3f33952 | ||
|
|
c14002edc3 | ||
|
|
8136a2d525 | ||
|
|
5aa7a63127 | ||
|
|
85be4bf25d | ||
|
|
0e78e06eb9 | ||
|
|
746dcfcadb | ||
|
|
f72d69c2b0 | ||
|
|
1ecf3aba03 | ||
|
|
f5793e2134 | ||
|
|
34be83d595 | ||
|
|
41db488c73 | ||
|
|
77e0ea91ae | ||
|
|
6c6394a1a7 | ||
|
|
4be53d048b | ||
|
|
9edeb6da14 | ||
|
|
a21650e746 | ||
|
|
eb8128a981 | ||
|
|
16b566e14f |
@@ -10,3 +10,6 @@
|
||||
# Generated by scripts/build_css.py; collapsed in diffs, not hand-edited.
|
||||
web_interface/static/v3/tailwind.css linguist-generated=true
|
||||
web_interface/static/v3/plugin-frame.css linguist-generated=true
|
||||
|
||||
# Installed as an executable (its shebang runs it) by install_service.sh.
|
||||
scripts/install/ledmatrix_refresh_units.py text eol=lf
|
||||
|
||||
@@ -31,7 +31,7 @@ jobs:
|
||||
|
||||
- uses: actions/setup-python@0b93645e9fea7318ecaed2b359559ac225c90a2b # v5.3.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
python-version: "3.13"
|
||||
|
||||
# No dependencies: the script reads src/__init__.py and CHANGELOG.md only.
|
||||
- name: Assert the tag, CHANGELOG and src.__version__ agree
|
||||
|
||||
@@ -14,8 +14,14 @@ permissions:
|
||||
|
||||
jobs:
|
||||
plugin-safety:
|
||||
name: Plugin safety harness + unit tests
|
||||
name: Plugin safety harness + unit tests (Python ${{ matrix.python-version }})
|
||||
runs-on: ubuntu-latest
|
||||
# The two Pythons the installer supports: Raspberry Pi OS Bookworm ships
|
||||
# 3.11 and Trixie 3.13.
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
python-version: ["3.11", "3.13"]
|
||||
env:
|
||||
# The bundled fixture plugin gives the harness at least one real plugin
|
||||
# to render, and REQUIRE_PLUGINS turns "discovered zero plugins" into a
|
||||
@@ -29,7 +35,7 @@ jobs:
|
||||
|
||||
- uses: actions/setup-python@0b93645e9fea7318ecaed2b359559ac225c90a2b # v5.3.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
python-version: ${{ matrix.python-version }}
|
||||
cache: pip
|
||||
|
||||
- name: Install dependencies
|
||||
@@ -43,8 +49,13 @@ jobs:
|
||||
pytest --no-cov test/plugins/
|
||||
|
||||
unit-tests:
|
||||
name: Core unit tests
|
||||
name: Core unit tests (Python ${{ matrix.python-version }})
|
||||
runs-on: ubuntu-latest
|
||||
# Bookworm's Python (3.11) and Trixie's (3.13); see plugin-safety.
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
python-version: ["3.11", "3.13"]
|
||||
steps:
|
||||
- uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2
|
||||
with:
|
||||
@@ -52,7 +63,7 @@ jobs:
|
||||
|
||||
- uses: actions/setup-python@0b93645e9fea7318ecaed2b359559ac225c90a2b # v5.3.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
python-version: ${{ matrix.python-version }}
|
||||
cache: pip
|
||||
|
||||
- name: Install dependencies
|
||||
@@ -84,7 +95,7 @@ jobs:
|
||||
|
||||
- uses: actions/setup-python@0b93645e9fea7318ecaed2b359559ac225c90a2b # v5.3.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
python-version: "3.13"
|
||||
cache: pip
|
||||
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0
|
||||
@@ -123,7 +134,7 @@ jobs:
|
||||
|
||||
- uses: actions/setup-python@0b93645e9fea7318ecaed2b359559ac225c90a2b # v5.3.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
python-version: "3.13"
|
||||
|
||||
# Downloads the pinned standalone Tailwind CLI (SHA-256 checked; no
|
||||
# Node), rebuilds static/v3/tailwind.css and plugin-frame.css from the
|
||||
@@ -142,7 +153,7 @@ jobs:
|
||||
|
||||
- uses: actions/setup-python@0b93645e9fea7318ecaed2b359559ac225c90a2b # v5.3.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
python-version: "3.13"
|
||||
cache: pip
|
||||
|
||||
# The runtime requirements are installed so mypy sees the real types of
|
||||
@@ -181,7 +192,7 @@ jobs:
|
||||
|
||||
- uses: actions/setup-python@0b93645e9fea7318ecaed2b359559ac225c90a2b # v5.3.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
python-version: "3.13"
|
||||
|
||||
# Stdlib only; exits 0 whatever it finds.
|
||||
- name: Report method-family drift across the nine scoreboards
|
||||
|
||||
+1331
File diff suppressed because it is too large
Load Diff
@@ -151,6 +151,11 @@ The system supports live, recent, and upcoming game information for multiple spo
|
||||
- **1GB models (Pi 3B / 3B+), the 512MB Pi Zero 2 W and other low-memory boards**: supported, but the `rpi-rgb-led-matrix` C++ build needs more memory than the Pi has. The installer detects this automatically, compiles with fewer parallel jobs, and adds a temporary swapfile for the build which it removes afterwards. Expect that step to take 15-25 minutes instead of 2-5, and leave at least **3GB free** on the SD card. If you manage swap yourself, opt out with `--skip-swap`. To pin the compiler down further, use `--build-jobs 1`. Once running, keep an eye on memory: see [docs/LOW_MEMORY_BOARDS.md](docs/LOW_MEMORY_BOARDS.md).
|
||||
|
||||
|
||||
### Operating system
|
||||
- **Raspberry Pi OS Lite, Trixie (Debian 13) or Bookworm (Debian 12)**, 64-bit recommended. Trixie is the current release and the one to pick for a new SD card; an existing Bookworm install works as it is, no upgrade needed. The installer checks this first and stops with directions on anything else (Bullseye and older, the desktop edition, other distributions).
|
||||
- **Python**: whatever the OS ships, 3.13 on Trixie and 3.11 on Bookworm. Don't install a different Python; the installer and the services use the system `python3`.
|
||||
- **Networking**: NetworkManager, the default on both. Choosing a WiFi network from the web page and the `LEDMatrix-Setup` hotspot need it; if you switched to dhcpcd in `raspi-config`, switch back (Advanced Options → Network Config → NetworkManager).
|
||||
|
||||
### RGB Matrix Bonnet / HAT
|
||||
- [Adafruit RGB Matrix Bonnet/HAT](https://www.adafruit.com/product/3211) – supports one “chain” of horizontally connected displays
|
||||
- [Adafruit Triple LED Matrix Bonnet](https://www.adafruit.com/product/6358) – supports up to 3 vertical “chains” of horizontally connected displays *(use `regular` as hardware mapping)*
|
||||
@@ -249,7 +254,7 @@ These are not required and you can probably rig up something basic with stuff yo
|
||||
|
||||
<img width="512" height="361" alt="Step 2 Other " src="https://github.com/user-attachments/assets/166a22e8-8067-48df-9f80-50c91f573356" />
|
||||
|
||||
5. Then choose Raspbian OS (64-bit) Lite (Trixie)
|
||||
5. Then choose Raspbian OS (64-bit) Lite (Trixie). Bookworm Lite (listed as Legacy) also works; see [Operating system](#operating-system) below
|
||||
|
||||
<img width="512" height="361" alt="Step 4 Trixie Lite 64" src="https://github.com/user-attachments/assets/3b8590ce-b810-4dfe-9253-26e0d4f8ed1e" />
|
||||
|
||||
|
||||
@@ -174,6 +174,16 @@
|
||||
"plugin_system": {
|
||||
"plugins_directory": "plugin-repos"
|
||||
},
|
||||
"fetch_service": {
|
||||
"enabled": true,
|
||||
"max_wait_seconds": 2,
|
||||
"rate_limits": {
|
||||
"*.espn.com": {
|
||||
"per_second": 20,
|
||||
"burst": 200
|
||||
}
|
||||
}
|
||||
},
|
||||
"web-ui-info": {
|
||||
"enabled": true,
|
||||
"display_duration": 10
|
||||
|
||||
@@ -649,7 +649,9 @@ When nothing is running on demand, `data.state` is
|
||||
> on-demand machinery is internal — drive it through the REST endpoints
|
||||
> above (or the web UI buttons). The API handlers
|
||||
> (`start_on_demand_display()` / `stop_on_demand_display()` in
|
||||
> `web_interface/blueprints/api_v3/display.py`) write a request into the cache
|
||||
> `web_interface/blueprints/api_v3/display.py`) send the request over the
|
||||
> display's control socket ([IPC_CONTROL_SOCKET.md](IPC_CONTROL_SOCKET.md)).
|
||||
> Only when the socket cannot carry it do they write it into the cache
|
||||
> manager under the `display_on_demand_request` key, which
|
||||
> `DisplayController._poll_on_demand_requests()`
|
||||
> (`src/display_controller.py`) picks up. A separate
|
||||
@@ -747,8 +749,14 @@ keys helps troubleshoot stuck states.
|
||||
"timestamp": 1234567890.123
|
||||
}
|
||||
```
|
||||
**Purpose:** Communication from web interface to display controller
|
||||
**When Set:** API endpoint receives request
|
||||
**Purpose:** Communication from web interface to display controller, as the
|
||||
fallback when the control socket cannot carry the request (deprecated; it
|
||||
will be removed in a later release)
|
||||
**When Set:** API endpoint receives a request and the display's control
|
||||
socket is unavailable (display stopped, or older than the socket or the
|
||||
command); some plugins also write it directly
|
||||
**Read:** once a second while the display serves the control socket (0.25 s
|
||||
without it), and only when the file changed since the last look
|
||||
**Auto-Cleared:** After processing or 1 hour TTL
|
||||
|
||||
**2. display_on_demand_config** (No TTL)
|
||||
|
||||
+28
-15
@@ -42,26 +42,32 @@ each other. They share three things:
|
||||
| State | Where | Written by | Read by |
|
||||
|---|---|---|---|
|
||||
| On-demand command | control socket `/run/ledmatrix/control.sock` ([IPC_CONTROL_SOCKET.md](IPC_CONTROL_SOCKET.md)) | web: `start_on_demand_display()` / `stop_on_demand_display()` in [`api_v3/display.py`](../web_interface/blueprints/api_v3/display.py), via [`src/ipc/client.py`](../src/ipc/client.py) | display: [`src/ipc/server.py`](../src/ipc/server.py) acks; the render thread applies it in `_poll_on_demand_requests()` |
|
||||
| On-demand request (fallback) | cache `display_on_demand_request` | web, when the socket fails; four plugins write it directly | display: `_poll_on_demand_requests()` |
|
||||
| On-demand request (fallback) | cache `display_on_demand_request` | web, only when the socket could not carry the request (`should_fall_back`); four plugins write it directly | display: `_poll_on_demand_requests()`, a `stat()` every 1 s while the socket is up (0.25 s without), read only when the file changed |
|
||||
| On-demand state | cache `display_on_demand_state` | display: `_publish_on_demand_state()` | web: `/api/v3/display/on-demand/status` |
|
||||
| Current screen | cache `display_current_state` | display | web: `/api/v3/display/current-status` |
|
||||
| Plugin errors | cache `plugin_error_snapshot` | display: `ErrorSnapshotPublisher` ([`src/error_aggregator.py`](../src/error_aggregator.py)) | web: `read_error_report()` for `/api/v3/errors/*` |
|
||||
| Error clear | cache `plugin_error_clear_request` | web | display |
|
||||
| Error clear | control socket `errors.clear`; cache `plugin_error_clear_request` as the fallback | web: `POST /api/v3/errors/clear` | display: applied before the socket answers; the mailbox on the error publisher's 5 s tick, read only when the file changed |
|
||||
| Font usage | cache `font_usage_snapshot` | display: `FontUsagePublisher` ([`src/font_usage.py`](../src/font_usage.py)) | web: Fonts tab |
|
||||
| Fetch statistics (requests per plugin and host) | cache `fetch_stats_snapshot` | display: `FetchStatsPublisher` ([`src/common/fetch_service.py`](../src/common/fetch_service.py)), at most once a minute on change | web: `read_fetch_stats()` for `/api/v3/plugins/fetch-stats` |
|
||||
| Plugin health | cache `plugin_health:<id>` | display (web writes on reset) | web: `/api/v3/plugins/health` |
|
||||
| Plugin runtime (loaded, state, last error, version) | cache `plugin_runtime_snapshot` | display: `PluginRuntimePublisher` ([`src/plugin_system/plugin_runtime.py`](../src/plugin_system/plugin_runtime.py)) | web: `read_plugin_runtime()` for `/api/v3/plugins/installed`, `/plugins/state`, reconciliation |
|
||||
| Preview frame | `/tmp/led_matrix_preview.png` | display: `DisplayManager`, gated by [`snapshot_policy`](../src/common/snapshot_policy.py) | web: display SSE stream, `/api/v3/health` (file age) |
|
||||
| Preview viewer marker | `/tmp/led_matrix_preview_viewer` | web, while a preview is open | display: writes full-rate snapshots only while it is fresh |
|
||||
| Preview frame | `/tmp/led_matrix_preview.png` | display: `DisplayManager`, gated by [`snapshot_policy`](../src/common/snapshot_policy.py): a changed frame at most once a second with a viewer, every 30 s without | web: display SSE stream (checks the mtime every 0.25 s), `/api/v3/health` (file age) |
|
||||
| Preview viewer marker | `/tmp/led_matrix_preview_viewer` | web, about once a second while a preview is open | display: writes viewer-rate snapshots only while it is fresh (5 s) |
|
||||
| Hardware init status | `/tmp/led_matrix_hw_status.json` | display | web: `/api/v3/hardware/status` |
|
||||
| Render-loop heartbeat | `/run/ledmatrix/display-heartbeat.json` (tmpfs) | display: the render thread, via [`display_watchdog`](../src/display_watchdog.py) | web: `/api/v3/health` (`checks.display_loop`); the update health check |
|
||||
|
||||
The on-demand start route starts `ledmatrix.service` when it is not running
|
||||
(`start_service`, on by default) but never restarts a running one. The routes
|
||||
send the command over the display's control socket and get an ack; when that
|
||||
fails (a stopped display, one older than the socket) they write the mailbox
|
||||
instead, which the display reads every `ON_DEMAND_POLL_INTERVAL` (0.25s), from
|
||||
its dwell sleep, its render loops and Vegas's interrupt check as well as the
|
||||
main loop. Both ways end in the same handler, `_handle_on_demand_request()`.
|
||||
send the command over the display's control socket and get an ack; only when
|
||||
the socket could not carry it (a stopped display, one older than the socket
|
||||
or the command) do they write the mailbox instead. A display that had the
|
||||
request and refused it is answered with the error, not posted a mailbox
|
||||
copy. The display looks at the mailbox every
|
||||
`MAILBOX_POLL_INTERVAL_WITH_SOCKET` (1 s) while it serves the socket, and
|
||||
every `ON_DEMAND_POLL_INTERVAL` (0.25 s) without one, from its dwell sleep,
|
||||
its render loops and Vegas's interrupt check as well as the main loop; a
|
||||
look is one `stat()` unless the file changed. Both ways end in the same
|
||||
handler, `_handle_on_demand_request()`.
|
||||
The socket's handlers only queue; see [IPC_CONTROL_SOCKET.md](IPC_CONTROL_SOCKET.md)
|
||||
for the protocol, the permission model and the plan to retire the mailboxes.
|
||||
|
||||
@@ -85,7 +91,8 @@ How a web-side change reaches the running plugins:
|
||||
| Plugin enabled or disabled | `ConfigService` → `_controller_config_change` flags a reconcile; `_reconcile_enabled_plugins` loads it (fresh from disk) or unloads it on the render thread |
|
||||
| Plugin uninstalled (config removed) | the removed section flips its `enabled` flag, and the reconcile unloads it |
|
||||
| Plugin installed, not enabled | nothing to do until it is enabled, which loads it |
|
||||
| Plugin installed while already enabled, updated while enabled, or uninstalled with its config kept | **not picked up**: the display keeps running what it loaded. The route answers `restart_required: true` and the UI shows its restart banner |
|
||||
| Plugin updated while enabled | the update route asks the display over the control socket (`plugin.reload`) to reload it on the render thread, and answers `restart_required: false` once the new code runs. Without the socket, as the next row |
|
||||
| Plugin installed while already enabled, updated while enabled and not reloaded, or uninstalled with its config kept | **not picked up**: the display keeps running what it loaded. The route answers `restart_required: true` and the UI shows its restart banner |
|
||||
|
||||
`display_restart_required()` in `plugin_catalog.py` holds that last rule;
|
||||
routes return it as `restart_required` (with the banner's wording in
|
||||
@@ -109,9 +116,11 @@ other web-UI action runs its script as a subprocess. A later, explicit
|
||||
**plugin web-entry contract** -- a declared entry point for plugin web code
|
||||
-- replaces that function.
|
||||
|
||||
Next stages: a **control socket** from the web process to the display
|
||||
(reload one plugin, ask for its state) in place of `restart_required` and
|
||||
the cache-key mailboxes, and the plugin web-entry contract above.
|
||||
The **control socket** from the web process to the display
|
||||
([IPC_CONTROL_SOCKET.md](IPC_CONTROL_SOCKET.md)) carries on-demand
|
||||
commands and reloads an updated plugin; its next stages stream the
|
||||
display's state and retire the cache-key mailboxes. The plugin web-entry
|
||||
contract above is still to come.
|
||||
|
||||
### Plugin state: desired, observed, and who owns it
|
||||
|
||||
@@ -187,7 +196,8 @@ the scheduler), and sets up Vegas mode.
|
||||
enable/disable, poll on-demand requests, run scheduled plugin updates, check
|
||||
the on/off schedule and brightness, then show one screen. Priority is
|
||||
on-demand, then WiFi status messages, then live priority, then Vegas mode,
|
||||
then normal rotation.
|
||||
then normal rotation. [RUN_LOOP_REDESIGN.md](RUN_LOOP_REDESIGN.md) is the
|
||||
plan for restructuring this loop and lists its golden trace tests.
|
||||
|
||||
- **Rotation.** `available_modes` is the ordered list of display modes;
|
||||
`current_mode_index` advances after each screen.
|
||||
@@ -209,7 +219,10 @@ then normal rotation.
|
||||
to it, rotating between several live games.
|
||||
- **Schedule and dim schedule.** `_check_schedule()` reads `schedule`;
|
||||
`_check_dim_schedule()` reads `dim_schedule` and
|
||||
`display.hardware.brightness`. Both are re-evaluated once a minute.
|
||||
`display.hardware.brightness`. Both are re-evaluated once a minute, and
|
||||
both windows are half-open: on (or dimmed) from the start time, off at
|
||||
the end time. When an on-demand session ends, the on/off schedule is
|
||||
re-checked at once rather than at the next minute.
|
||||
- **Long screens.** While a screen is showing (a dwell, a scroll, a Vegas
|
||||
iteration), `_service_pending_changes()` repeats the on-demand, schedule
|
||||
and brightness checks every 0.25 s, so a change does not wait for the
|
||||
|
||||
@@ -31,6 +31,13 @@ tooling against it.
|
||||
| `start_time` / `end_time` | `"HH:MM"`, `07:00`–`23:00` | Global-mode on/off times |
|
||||
| `days.<weekday>.{enabled,start_time,end_time}` | per-day objects | Per-day-mode overrides |
|
||||
|
||||
The display is on from `start_time` up to, but not including, `end_time`:
|
||||
with `07:00`–`23:00` it turns on at 07:00 and off at 23:00. An end earlier
|
||||
than the start crosses midnight (`22:00`–`07:00` is on overnight). In
|
||||
per-day mode, the entry for the current day decides. An on-demand session
|
||||
keeps the display on during off hours; once it ends or is stopped, the
|
||||
display blanks within about a second.
|
||||
|
||||
Read by `DisplayController._check_schedule()` (`src/display_controller.py`).
|
||||
Managed in the web UI under Schedule.
|
||||
|
||||
@@ -44,7 +51,9 @@ Same shape as `schedule` (the template sets its `mode` to `"global"`), plus:
|
||||
|
||||
Read by `DisplayController._check_dim_schedule()` (`src/display_controller.py`;
|
||||
saved via `POST /api/v3/config/dim-schedule`). The display returns to
|
||||
`display.hardware.brightness` outside the window.
|
||||
`display.hardware.brightness` outside the window. The window has the same
|
||||
boundaries as `schedule`: dimmed from `start_time` up to, but not including,
|
||||
`end_time`.
|
||||
|
||||
## `display.hardware` — matrix panel hardware
|
||||
|
||||
|
||||
@@ -17,13 +17,13 @@ The LEDMatrix emulator allows you to run and test LEDMatrix displays on your com
|
||||
## Prerequisites
|
||||
|
||||
### System Requirements
|
||||
- Python 3.10 or higher
|
||||
- Python 3.11 or higher (3.11 and 3.13 are tested)
|
||||
- Windows, macOS, or Linux
|
||||
- At least 2GB RAM (4GB recommended)
|
||||
- Internet connection for plugin downloads
|
||||
|
||||
### Required Software
|
||||
- Python 3.10+
|
||||
- Python 3.11+
|
||||
- pip (Python package manager)
|
||||
- Git (for plugin management)
|
||||
|
||||
|
||||
@@ -203,6 +203,7 @@ Current methods:
|
||||
| `measure_text(text, font)` | `(width, height, baseline)` |
|
||||
| `get_font_height(font)` | Line height |
|
||||
| `register_plugin_fonts(plugin_id, font_manifest)` | Register a plugin's fonts (core calls it at load) |
|
||||
| `forget_plugin_fonts(plugin_id)` | Drop a plugin's manifest fonts and their cached objects (core calls it when a plugin unloads) |
|
||||
| `clear_cache()` | Drop cached fonts and metrics |
|
||||
| `font_catalog` (attribute) | Family name → file path |
|
||||
|
||||
@@ -218,5 +219,6 @@ Removed in 3.8.0, after logging a deprecation warning on first call since
|
||||
| `get_performance_stats()` | — |
|
||||
| `set_override()`, `remove_override()`, `get_overrides()` | a font field in your plugin's config schema |
|
||||
| `get_manager_fonts()`, `get_detected_fonts()` | — |
|
||||
| `get_plugin_fonts()`, `unregister_plugin_fonts()` | — |
|
||||
| `get_plugin_fonts()` | — |
|
||||
| `unregister_plugin_fonts()` | `forget_plugin_fonts()` (core calls it on unload) |
|
||||
| `add_font()`, `remove_font()`, `validate_font()` | the web UI's Fonts tab |
|
||||
|
||||
+19
-1
@@ -15,6 +15,12 @@ This guide will help you set up your LEDMatrix display for the first time and ge
|
||||
- Power supply (5V, 4A minimum recommended)
|
||||
- MicroSD card (16GB minimum)
|
||||
|
||||
**Software:**
|
||||
- Raspberry Pi OS Lite, Trixie (Debian 13) or Bookworm (Debian 12). Trixie
|
||||
is the current release; Bookworm is listed as Legacy in Raspberry Pi
|
||||
Imager. No other system is supported, and the installer says so up front.
|
||||
- The OS's own Python: 3.13 on Trixie, 3.11 on Bookworm
|
||||
|
||||
**Network:**
|
||||
- WiFi network (or Ethernet cable)
|
||||
- Computer with web browser on same network
|
||||
@@ -28,7 +34,8 @@ This guide will help you set up your LEDMatrix display for the first time and ge
|
||||
There is no prebuilt SD card image — you install LEDMatrix onto stock
|
||||
Raspberry Pi OS Lite yourself:
|
||||
|
||||
1. Flash Raspberry Pi OS Lite to the MicroSD card (Raspberry Pi Imager)
|
||||
1. Flash Raspberry Pi OS Lite (Trixie, or Bookworm) to the MicroSD card
|
||||
(Raspberry Pi Imager)
|
||||
2. Connect the LED matrix to your Raspberry Pi, insert the card, and
|
||||
power on
|
||||
3. SSH into the Pi and run the one-shot installer:
|
||||
@@ -39,6 +46,17 @@ Raspberry Pi OS Lite yourself:
|
||||
[README Installation Steps / Quick Install](../README.md#installation-steps)
|
||||
for full details
|
||||
|
||||
The one-shot installer installs the newest release (the **stable** update
|
||||
channel). To run the newest, unreleased code from `main` instead (the
|
||||
**beta** channel), put `LEDMATRIX_CHANNEL=beta` in front of `bash`:
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/ChuckBuilds/LEDMatrix/main/scripts/install/one-shot-install.sh | LEDMATRIX_CHANNEL=beta bash
|
||||
```
|
||||
A manual clone starts on `main`; add `--beta` to `first_time_install.sh`
|
||||
to stay on it, or leave it off and the first update after the next
|
||||
release moves the device onto releases. You can switch channels later on
|
||||
the General tab.
|
||||
|
||||
**Expected Behavior after install:**
|
||||
- LED matrix will light up
|
||||
- A fresh install ships only the bundled `starlark-apps` and
|
||||
|
||||
+547
-55
@@ -2,14 +2,23 @@
|
||||
|
||||
The display process serves a Unix socket that the web interface uses to send
|
||||
it commands and get an answer back. It replaces the cache-file "mailboxes" on
|
||||
the SD card one command at a time. Stage 1, described here, carries on-demand
|
||||
start/stop/status. The file mailbox stays as a fallback for one release.
|
||||
the SD card one command at a time. Stage 1 carries on-demand start, stop and
|
||||
status. Stage 2 makes those commands land within a frame on every kind of
|
||||
screen, and adds `brightness.set` and `plugin.reload`. Stage 3 adds a state
|
||||
stream (`state.get`, `state.subscribe`), so the web interface reads what the
|
||||
display is doing from the socket instead of from cache files the display
|
||||
wrote to the SD card. Stage 4 makes the socket the only way a command goes
|
||||
while it works: the web interface writes a mailbox only when the socket
|
||||
cannot carry the request, `errors.clear` replaces the last command that
|
||||
always went through a mailbox, and the display looks at the mailboxes once
|
||||
a second, with a `stat()`. The file mailboxes and the cache keys stay as a
|
||||
fallback for one release.
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Socket | `/run/ledmatrix/control.sock` (tmpfs) |
|
||||
| Served by | the display process ([`src/ipc/server.py`](../src/ipc/server.py)), started by `DisplayController.run()` |
|
||||
| Used by | the web interface ([`src/ipc/client.py`](../src/ipc/client.py)): `POST /api/v3/display/on-demand/start` and `/stop` |
|
||||
| Used by | the web interface ([`src/ipc/client.py`](../src/ipc/client.py)): `POST /api/v3/display/on-demand/start` and `/stop`, `POST /api/v3/plugins/update` (reload), `POST /api/v3/config/main` (brightness), `POST /api/v3/errors/clear`; and through [`web_interface/display_state.py`](../web_interface/display_state.py) (the state stream), `GET /api/v3/display/current-status`, `/display/on-demand/status`, `/plugins/installed` (`runtime`), `/plugins/state` and the reconciliations, `/health` (`display_loop`) |
|
||||
| Contract | [`src/ipc/contract.py`](../src/ipc/contract.py): messages, versions, framing and the socket path; both sides import it |
|
||||
| Override | `LEDMATRIX_CONTROL_SOCKET=/some/path.sock` for both processes, or `=off` to disable it |
|
||||
|
||||
@@ -34,6 +43,11 @@ running, the socket does not exist, and the web interface knows right away.
|
||||
|
||||
## Protocol (version 1)
|
||||
|
||||
Stage 3 is still version 1: `state.get` and `state.subscribe` are new
|
||||
commands, and a stage-2 display answers them `unknown_command`, which the
|
||||
web interface treats as "no socket" and falls back from.
|
||||
|
||||
|
||||
**Framing.** One JSON object per line (newline-delimited JSON), UTF-8, at
|
||||
most 64 KiB per line (`MAX_MESSAGE_BYTES`). Senders encode with
|
||||
`ensure_ascii`, so a newline never appears inside a message. A connection
|
||||
@@ -71,6 +85,21 @@ one. Clients branch on `error.code`, never on the message text.
|
||||
| `on_demand.start` | `{plugin_id?, mode?, duration?, pinned?}` (at least one of `plugin_id` and `mode`) | ack | queued |
|
||||
| `on_demand.stop` | — | ack | queued |
|
||||
| `on_demand.status` | — | `{on_demand: {...}, current_mode, display_active}` | answered directly |
|
||||
| `brightness.set` | `{brightness: int 0-100}` | `{brightness, panel_brightness, dimmed, display_active}` | queued, awaited (2 s) |
|
||||
| `plugin.reload` | `{plugin_id}` | `{plugin_id, reloaded: true, version, modes}` | queued, awaited (10 s) |
|
||||
| `state.get` | `{since?, epoch?}` | a state snapshot (see "The state stream") | answered directly |
|
||||
| `state.subscribe` | — | a state snapshot, then pushed `state` / `tick` events | answered directly, then a stream |
|
||||
| `errors.clear` | `{cutoff: number}` (epoch seconds, finite, ≥ 0) | `{request_id, cutoff, cleared}` | answered directly, once applied |
|
||||
|
||||
`errors.clear` (stage 4) forgets the plugin errors the display recorded at
|
||||
or before `cutoff` and rewrites its error snapshot (`plugin_error_snapshot`)
|
||||
before it answers, so the web interface's next read already has it. The
|
||||
request `id` is the clear's id, which the snapshot reports as
|
||||
`applied_clear_id`. It is answered on the connection thread by a handler the
|
||||
display registers (`ControlServer(handlers=...)`, the contract's
|
||||
`DIRECT_COMMANDS`): the error aggregator and its publisher have their own
|
||||
locks, and nothing the render thread owns is touched. A display that has no
|
||||
handler answers `unknown_command`, as an older display does.
|
||||
|
||||
`duration` is a number of seconds, or a numeric string. `0`, `null` or `""`
|
||||
mean "until stopped". `pinned` must be a real boolean: the REST route has
|
||||
@@ -78,28 +107,73 @@ already converted strings like `"false"` before it sends the command. The
|
||||
`on_demand` object in `on_demand.status` is the same dict the display
|
||||
publishes to `display_on_demand_state`.
|
||||
|
||||
**Acknowledgements.** A queued command is *accepted*, not *done*.
|
||||
`brightness.set` sets the panel's normal brightness. It is transient: it
|
||||
writes nothing to `config.json`, and the next config the display's watcher
|
||||
loads (or a restart) puts the configured value back. The web interface
|
||||
sends it after it has saved the setting, so the two agree. The dim schedule
|
||||
still applies on top, so `panel_brightness` is the dim level while the
|
||||
schedule dims. While the schedule has the display off, the new level is
|
||||
kept for when it comes back on.
|
||||
|
||||
`plugin.reload` loads a plugin the display is running again from disk,
|
||||
manifest included: the steps of disabling it live and enabling it again,
|
||||
with its modes kept in their place in the rotation. Only a running plugin
|
||||
can be reloaded (`not_loaded` otherwise), so the id never makes the display
|
||||
import anything new. A plugin loaded only for an on-demand session gets
|
||||
`busy`. A new version that fails to load gets `failed` and stays out of the
|
||||
rotation, as it would after a restart. The load runs off the render thread,
|
||||
so the panel keeps scrolling while it happens (see below).
|
||||
|
||||
**Acknowledgements.** A queued on-demand command is *accepted*, not *done*.
|
||||
`{"accepted": true, "request_id": …}` means the command is waiting in the
|
||||
render thread's queue, and the render thread will apply it at its next
|
||||
on-demand check. That is within one frame on a scrolling screen, 0.25 s
|
||||
during a dwell, and up to 1 s on a static screen, whose frame loop sleeps a
|
||||
second between frames. Except on a scrolling screen, where the mailbox waits
|
||||
up to 0.25 s, these are the mailbox's delays too: stage 1 adds
|
||||
acknowledgements, not speed. Any outcome is published as before
|
||||
render thread's queue. Since stage 2 the render thread waits on that queue
|
||||
instead of sleeping, so it applies the command within one frame on every
|
||||
kind of screen (see below). Any outcome is published as before
|
||||
(`display_on_demand_state`, and `status`/`error` for a bad plugin or mode),
|
||||
and it can be read with `on_demand.status`.
|
||||
|
||||
**Awaited commands.** `brightness.set` and `plugin.reload` are answered only
|
||||
once the render thread has applied them, with their result or their error.
|
||||
The connection thread waits for that (2 s and 10 s, `AWAIT_SECONDS` in the
|
||||
contract); the render thread never waits for a client. When the render thread
|
||||
has not got to the command in time, the answer is `pending`: the command
|
||||
stays queued and is still applied, so a client treats `pending` as "not known
|
||||
to be done", not as a refusal. The client's own timeout is one second longer
|
||||
than the display's wait, so `pending` arrives before the client gives up.
|
||||
|
||||
**Versions.** Every request carries `v`. For any command except `hello`, a
|
||||
`v` the display does not speak gets `unsupported_version`. `hello` is checked
|
||||
by its `versions` list instead, and its result names the highest version both
|
||||
sides share, so a client can find out what a display supports before it
|
||||
relies on anything newer. Stage 1's client sends `v: 1` and falls back to the
|
||||
relies on anything newer. The client sends `v: 1` and falls back to the
|
||||
mailbox when the display refuses it. It does not send `hello` first, which
|
||||
saves a round trip.
|
||||
|
||||
New commands are added within a version, so stage 2 is still version 1. A
|
||||
display that does not know a command answers `unknown_command`, which the
|
||||
web interface treats like any other socket failure and falls back from, and
|
||||
`hello` lists the commands a display knows. The version changes only when the
|
||||
envelope or the meaning of an existing command changes.
|
||||
|
||||
**Events.** `state.subscribe` is the one command with more than one message
|
||||
in reply. After its response, the display pushes events on the same
|
||||
connection until either side hangs up:
|
||||
|
||||
```json
|
||||
{"v": 1, "id": "<the subscribe id>", "event": "state", "result": {...a state snapshot...}}
|
||||
{"v": 1, "id": "<the subscribe id>", "event": "tick", "result": {"version": 7, "epoch": "…", "pid": 812, "served_at": 1790000000.1, "changed": false, "loop": {...}, "volatile": {"display": {"last_updated": 1790000000.0}, "...": "..."}}}
|
||||
```
|
||||
|
||||
An event has `event` where a response has `ok`, which is how a reader tells
|
||||
them apart. The client sends nothing after the subscribe; anything it does
|
||||
send is ignored.
|
||||
|
||||
**Error codes:** `bad_json`, `bad_request`, `message_too_large`,
|
||||
`unsupported_version`, `unknown_command`, `invalid_args`, `busy` (queue full,
|
||||
or too many connections), `forbidden` (peer credentials refused), `internal`.
|
||||
Stage 2 adds `pending` (accepted, not applied in time, still queued),
|
||||
`not_loaded` (`plugin.reload` of a plugin the display is not running) and
|
||||
`failed` (the render thread tried, and it did not work).
|
||||
|
||||
Try it on a device:
|
||||
|
||||
@@ -107,34 +181,346 @@ Try it on a device:
|
||||
python3 - <<'EOF'
|
||||
from src.ipc import client # run from the project directory
|
||||
print(client.on_demand_status())
|
||||
print(client.brightness_set(60))
|
||||
EOF
|
||||
```
|
||||
|
||||
## The state stream (stage 3)
|
||||
|
||||
Before stage 3 the web interface learned what the display was doing by
|
||||
reading files the display kept writing:
|
||||
|
||||
| What | Written by the display | How often | Medium |
|
||||
|---|---|---|---|
|
||||
| current mode, plugin, `is_display_active`, `on_demand_active` | `display_current_state` | every mode change, every flag change, and every 30 s | cache (SD card) |
|
||||
| on-demand session | `display_on_demand_state` | on each on-demand event | cache (SD card) |
|
||||
| plugin runtime snapshot (#690) | `plugin_runtime_snapshot` | on a change (at most every 10 s), else every 60 s | cache (SD card) |
|
||||
| render-loop liveness (#687) | `display-heartbeat.json` | every 5 s | tmpfs |
|
||||
|
||||
Now the display also keeps the same state in memory and serves it on the
|
||||
socket.
|
||||
|
||||
**The snapshot.** `state.get` and `state.subscribe` answer with one object:
|
||||
|
||||
```json
|
||||
{"schema": 1, "version": 42, "epoch": "3f9c0d1e2a4b5c6d", "pid": 812,
|
||||
"served_at": 1790000000.1, "changed": true,
|
||||
"loop": {"heartbeat_age_seconds": 1.8, "armed": true, "stale_after": 60.0},
|
||||
"state": {
|
||||
"display": {"mode": "nfl_live", "plugin_id": "football-scoreboard", "mode_index": 3,
|
||||
"total_modes": 9, "on_demand_active": false, "is_display_active": true,
|
||||
"last_updated": 1790000000.0},
|
||||
"on_demand": {"active": false, "status": "idle", "...": "as display_on_demand_state"},
|
||||
"brightness": {"brightness": 80, "panel_brightness": 40, "dimmed": true},
|
||||
"plugins": {"schema": 1, "running": true, "published_at": 1789999998.5, "...": "as plugin_runtime_snapshot"},
|
||||
"loop": {"heartbeat_age_seconds": 1.8, "armed": true, "stale_after": 60.0}
|
||||
}}
|
||||
```
|
||||
|
||||
- `display` and `on_demand` are the dicts the cache keys hold, `plugins` is
|
||||
the runtime snapshot (`build_runtime_snapshot`), and `brightness` is the
|
||||
configured level, what the panel shows now, and whether the dim schedule
|
||||
has it dimmed. A section not published yet is `null`.
|
||||
- `loop` is not published: the display measures it when it answers, from
|
||||
the render thread's last beat in memory (`RenderWatchdog.liveness()`),
|
||||
the same beat that writes the heartbeat file. So it keeps ageing while the
|
||||
render thread is stuck, and the socket's connection threads still answer.
|
||||
`heartbeat_age_seconds` is `null` until the loop has drawn its first frame.
|
||||
- `version` goes up whenever a section changes, ignoring the timestamps that
|
||||
move on every publish (`last_updated`, `remaining`, `published_at`). It
|
||||
counts within an `epoch`, one run of the display process, so a reader that
|
||||
sees a new `epoch` has a restarted display.
|
||||
- `state.get` with `since` and `epoch` from an earlier answer gets just
|
||||
`{changed: false, version, epoch, pid, served_at, loop, volatile}` while
|
||||
nothing has changed. `volatile` is `{section: {key: value}}`: the current
|
||||
values of those ignored timestamps, which the reader merges into the copy
|
||||
it has. They don't make a new version, but they are still news:
|
||||
`display.last_updated` is how a reader knows the render thread is still
|
||||
publishing, and `plugins.published_at` the runtime publisher. Without
|
||||
them a reader's copy kept the timestamps of the last real change, so a
|
||||
mode on screen for over 120 s read as unknown.
|
||||
- A snapshot that would not fit in a message (hundreds of plugins) is sent
|
||||
without `plugins`, and `truncated: ["plugins"]` says so. Readers then use
|
||||
the cache for that section only.
|
||||
|
||||
**The stream.** `state.subscribe` answers with the snapshot, then:
|
||||
|
||||
- a `state` event (a full snapshot) whenever the version changes, and
|
||||
- a `tick` at least every 5 s (`SUBSCRIBE_KEEPALIVE_SECONDS`) when nothing
|
||||
changed. It is the short `changed: false` answer, so it carries `loop`
|
||||
(a stalled render loop shows up within one tick) and `volatile` (the
|
||||
timestamps stay as fresh as the writers keep them), and it tells the
|
||||
reader the connection is alive.
|
||||
|
||||
A slow reader is never sent a backlog: each event is the latest version, so
|
||||
one that falls behind skips the versions in between. A reader that has heard
|
||||
nothing for 15 s (three keepalives) stops trusting its copy.
|
||||
|
||||
**Who publishes, and when.** All of it is in memory, with no disk writes:
|
||||
|
||||
- the render thread, at the places it already published the cache keys:
|
||||
`display` and `brightness` on every pass of
|
||||
`_publish_current_mode_state_if_changed()` (every loop pass, and every
|
||||
`_service_pending_changes()` in a dwell, a scrolling screen or Vegas), and
|
||||
`on_demand` in `_publish_on_demand_state()`. Every pass refreshes
|
||||
`display.last_updated`, so a reader can tell when the render thread has
|
||||
stopped publishing, just as the cache key's 120 s `max_age` does.
|
||||
- the plugin runtime publisher's thread, on every 5 s tick: the snapshot is
|
||||
rebuilt when the state machine changed, otherwise only its `published_at`
|
||||
moves. A change reaches subscribers within a tick, without the cache's
|
||||
10 s throttle.
|
||||
|
||||
Publishing is a hand-off, as the command queue is in the other direction.
|
||||
The hub (`StateHub` in [`src/ipc/server.py`](../src/ipc/server.py)) holds a
|
||||
lock only to swap a dict reference, compare it with the last one and bump the
|
||||
version. Every socket write happens on the subscriber's own connection
|
||||
thread. The render thread never waits for a reader.
|
||||
|
||||
### Readers in the web interface
|
||||
|
||||
[`web_interface/display_state.py`](../web_interface/display_state.py) holds
|
||||
one `state.subscribe` connection per web process
|
||||
(`src.ipc.client.StateSubscription`, a daemon thread, started on the first
|
||||
read and reconnecting with a backoff of 1 s up to 30 s). A route answers
|
||||
from the latest pushed snapshot in memory. Before the subscription has one,
|
||||
the route asks once with `state.get` (0.5 s timeout). When neither works, it
|
||||
reads the cache keys and the heartbeat file as before:
|
||||
|
||||
| Route | From the socket | Fallback |
|
||||
|---|---|---|
|
||||
| `GET /api/v3/display/current-status` | `state.display` | `display_current_state` |
|
||||
| `GET /api/v3/display/on-demand/status` | `state.on_demand`, with `remaining` worked out from `expires_at` now | `display_on_demand_state` |
|
||||
| `GET /api/v3/plugins/installed` (`runtime`), `/plugins/state`, `POST /plugins/state/reconcile` and the startup reconciliation | `state.plugins` + `state.loop` | `plugin_runtime_snapshot` + `display-heartbeat.json` |
|
||||
| `GET /api/v3/health` (`checks.display_loop`) | `state.loop` | `display-heartbeat.json` |
|
||||
|
||||
Each answer says where it came from: `source: "socket" | "cache"` (or
|
||||
`"heartbeat_file"` for the health check).
|
||||
|
||||
The SSE display stream (`/api/v3/stream/display`) reads the preview frame
|
||||
file, not a cache key, so it does not change.
|
||||
|
||||
**The same verdicts either way.** The socket's answers are judged by the
|
||||
rules the cache readers apply (#726):
|
||||
|
||||
- the runtime view is `stalled` when the render loop's heartbeat age is at
|
||||
least `HEARTBEAT_STALE_SECONDS` (60 s, the health check's threshold), and
|
||||
then reports no per-plugin facts;
|
||||
- it is `stale` when the snapshot is older than its `stale_after` (the
|
||||
publisher thread stopped);
|
||||
- with no beat yet, the snapshot is judged on its own;
|
||||
- there is no pid check, because the display that answered is alive;
|
||||
- a `display` section the render thread has not refreshed for 120 s reads
|
||||
as unknown, as the cache key does once it ages out.
|
||||
|
||||
The age a reader uses is the age the display measured, plus the time since
|
||||
the snapshot arrived.
|
||||
|
||||
### Fewer SD writes
|
||||
|
||||
The cache keys are still written, for one release, as the fallback. While
|
||||
the socket serves the readers, the display writes two of them less often.
|
||||
"Serves the readers" means a subscriber is connected, or a `state.get` came
|
||||
within the last 60 s (`StateHub.readers_active()`):
|
||||
|
||||
- `display_current_state` is no longer written on every mode change: once
|
||||
every 60 s (`CURRENT_STATE_RELAXED_REFRESH_SECONDS`, inside the readers'
|
||||
120 s `max_age`), and at once when `is_display_active` or
|
||||
`on_demand_active` changes.
|
||||
- `plugin_runtime_snapshot`'s refresh goes from 60 s to 120 s
|
||||
(`RELAXED_REFRESH_INTERVAL`), and the snapshot says so in its own
|
||||
`refresh_interval` and `stale_after` (360 s). Changes are still written at
|
||||
once, at most every 10 s.
|
||||
|
||||
`display_on_demand_state` is written only on events, so it is unchanged.
|
||||
The heartbeat file is on tmpfs, so it costs no SD writes, and it stays: the
|
||||
automatic update's health check reads it.
|
||||
|
||||
This is safe because the relaxed rate only applies while readers are using
|
||||
the socket. If they stop (the web interface loses the socket, or is stopped),
|
||||
the next publish after the reader window writes a changed mode at once, and
|
||||
the runtime refresh goes back to 60 s. A fallback reader in that window sees
|
||||
a mode up to 60 s old, never one older than its `max_age`.
|
||||
|
||||
Measured with fake clocks (`test_cache_writes_per_minute_with_and_without_socket_readers`
|
||||
in `test/test_state_stream_readers.py`), for a rotation of 15 s screens:
|
||||
|
||||
| Key | Writes/min, no socket readers | Writes/min, socket readers |
|
||||
|---|---|---|
|
||||
| `display_current_state` | 4.0 | 1.0 |
|
||||
| `plugin_runtime_snapshot` | 1.0 | 0.5 |
|
||||
| Total | 5.0 | 1.5 |
|
||||
|
||||
That is 70% fewer writes for these keys: about 2,200 a day instead of 7,200.
|
||||
Shorter screens save more, because the old rate followed the mode changes.
|
||||
A display that rarely changes mode (one plugin, a long live game) saves less. Plugin
|
||||
data caches, the error snapshot and font usage are written by other code
|
||||
and are not affected.
|
||||
|
||||
## How the display applies a command
|
||||
|
||||
The server's threads never touch rendering. A connection thread parses the
|
||||
request, validates it against the contract, and then does one of two things:
|
||||
|
||||
- For a command that changes the panel, it puts a `QueuedCommand` on a
|
||||
bounded queue (16 entries) and answers with the ack.
|
||||
bounded queue (16 entries) and answers with the ack, or, for an awaited
|
||||
command, with the outcome the render thread reports back through the
|
||||
command's `CommandOutcome`.
|
||||
- For a query, it answers from a status snapshot the display provides
|
||||
(`DisplayController._control_status`). The snapshot only reads attributes.
|
||||
|
||||
The render thread drains the queue in `_poll_on_demand_requests()`, the same
|
||||
place it reads the mailbox, and hands each command to
|
||||
`_handle_on_demand_request()`, which is the mailbox's own handler. The two
|
||||
paths share all of their code: activation, the processed-id guard, error
|
||||
publishing, and resuming the rotation afterwards. The 0.25 s floor on the
|
||||
mailbox read does not apply to the queue, because draining it costs no disk
|
||||
read. A queued command also lets `_service_pending_changes()` skip its own
|
||||
floor, so a long scrolling screen or a Vegas iteration takes the command at
|
||||
its next frame.
|
||||
place it reads the mailbox:
|
||||
|
||||
**Exactly once.** A command and a mailbox write for the same request share
|
||||
one `request_id`. If the client times out after the display queued the
|
||||
command and then also writes the mailbox, the display processes the request
|
||||
once. The existing `on_demand_request_id` and processed-id checks drop the
|
||||
second copy.
|
||||
- An on-demand command goes to `_handle_on_demand_request()`, which is the
|
||||
mailbox's own handler. The two paths share all of their code: activation,
|
||||
the processed-id guard, error publishing, and resuming the rotation
|
||||
afterwards.
|
||||
- `brightness.set` is applied there and then (`_apply_control_brightness`),
|
||||
and the current frame is pushed again so the panel shows it.
|
||||
- `plugin.reload` starts at the top of the next loop pass, the place where
|
||||
plugins are enabled and disabled live, because there no `display()` and no
|
||||
Vegas iteration is on the stack (`_apply_pending_plugin_reloads`). Until
|
||||
then the current screen ends early, as it does for a WiFi notice: the
|
||||
frame loops, the dwell and Vegas's interrupt check all treat a pending
|
||||
reload as a reason to stop (the frame loops through the Arbiter's
|
||||
mid-screen check, `Source.RELOAD`; the dwell through
|
||||
`_plugin_reload_pending`).
|
||||
- Only the quick half of the reload runs on the render thread
|
||||
(`_start_plugin_reload`): the plugin's modes leave the rotation, its
|
||||
config subscription is dropped, and `PluginManager.detach_plugin` takes
|
||||
the instance out of `plugins`. After that nothing new calls the old
|
||||
instance: no `update()`, and no Vegas fetch. The rotation then advances
|
||||
(Vegas resumes its strip), and frames keep coming.
|
||||
- The slow half runs on a `plugin-reload-<id>` thread (`_PluginReloadJob`).
|
||||
It waits for the plugin's lock, then tears the old instance down
|
||||
(`unload_detached_plugin`) and loads the new one (`reload_plugin`). The
|
||||
lock can be held for seconds by a Vegas render of the old instance. On
|
||||
ledpi the render thread used to wait for it here, and a football reload
|
||||
froze the panel for 3.0 s.
|
||||
- The new instance joins the rotation between two frames
|
||||
(`_finish_plugin_reloads`, from `_service_pending_changes` or the top of
|
||||
the loop). Its modes go back to their old places, Vegas is told to fetch
|
||||
it again, and the command is answered.
|
||||
- While the plugin reloads, it is out of the rotation. Vegas scrolls what
|
||||
its strip already holds of it. An on-demand request for it gets
|
||||
`plugin-reloading`. A config reconcile neither loads it a second time nor
|
||||
unloads it mid-load; a disable saved meanwhile is applied once the
|
||||
reload is done. A second reload of the same plugin runs after the first.
|
||||
|
||||
The floor on the mailbox read (0.25 s, 1 s since stage 4) does not apply to the queue, because
|
||||
draining it costs no disk read. A queued command also lets
|
||||
`_service_pending_changes()` skip its own floor.
|
||||
|
||||
### Waking the render thread (stage 2)
|
||||
|
||||
Stage 1 made the socket answer, but not land sooner: a queued command waited
|
||||
for the same polls the mailbox does. Measured on ledpi (Pi 4, 24 fps Vegas),
|
||||
a start took 1.02 s on a static screen and about 0.4 s in Vegas either way.
|
||||
Now the queue wakes the render thread:
|
||||
|
||||
- **The waits.** The server sets a `threading.Event` whenever it queues a
|
||||
command. The render thread waits on it (`ControlServer.wait_for_command`)
|
||||
where it used to sleep: the static screen's 1 s frame sleep
|
||||
(`_wait_frame_interval`) and the dwell's 0.25 s ticks
|
||||
(`_sleep_with_plugin_updates`, which also covers scheduled-off and the
|
||||
empty-rotation pause). On a wake it applies the command at once. A command
|
||||
that does not end the screen, such as a brightness, does not cut the frame
|
||||
short: the wait carries on to the end of the interval, so the plugin is
|
||||
still drawn once a second.
|
||||
- **Vegas.** The coordinator still runs its interrupt check every 10 frames,
|
||||
and now also at any frame where `urgent()` is true. The display passes
|
||||
"a control socket command is queued", which is one `Event.is_set()` per
|
||||
frame.
|
||||
- **Scrolling screens** already service pending changes every frame.
|
||||
|
||||
So a command lands within a millisecond or so on a static screen and in a
|
||||
dwell, and within one frame in Vegas and on a scrolling screen. The mailbox
|
||||
is slower on purpose (see "The mailboxes now"). Commands still run only on the render thread: the
|
||||
connection threads only queue them and set the event. The one exception is
|
||||
the slow half of `plugin.reload` (tearing down and loading the plugin),
|
||||
which runs on its own thread. Every change to the display's state still
|
||||
happens on the render thread.
|
||||
|
||||
The waits are timed `Event.wait()` calls: no polling, and no more wake-ups
|
||||
than the sleeps they replace when nothing arrives. Measured under WSL
|
||||
(Python 3.12, 20 s runs in the order before, after, after, before, with the
|
||||
socket's accept thread up), the idle process used 0.015–0.018% of a core
|
||||
before and 0.019–0.021% after on a static screen, and 0.035–0.037% before and
|
||||
0.047% after in a dwell: about 25 µs more per wait, from `Event.wait`'s own
|
||||
bookkeeping. A client's send to the render thread waking took 0.72 ms median
|
||||
(1.04 ms max), and a whole `brightness.set` round trip 0.64 ms median.
|
||||
|
||||
Without a socket (Windows, `LEDMATRIX_CONTROL_SOCKET=off`) the waits are the
|
||||
plain sleeps they were.
|
||||
|
||||
**Exactly once.** Since stage 4 the web interface writes the mailbox only
|
||||
when the display never had the request (see "When the web interface falls
|
||||
back"), so a request goes one way or the other, never both. A command and a
|
||||
mailbox write for the same request still share one `request_id`, and the
|
||||
`on_demand_request_id` and processed-id checks still drop a second copy: an
|
||||
older web interface (before stage 4) wrote the mailbox after a reply timed
|
||||
out, too. The display takes such a copy out of the mailbox when it drops it.
|
||||
|
||||
## When the web interface falls back (stage 4)
|
||||
|
||||
The client tells a request the display never had from one it had and then
|
||||
failed. `ControlError.sent` is True once the whole request was written to a
|
||||
connected display; a refusal the display sends before reading anything
|
||||
(`forbidden`, too many connections) carries no request id, and leaves it
|
||||
False. `src.ipc.client.should_fall_back()` is the one rule every route uses:
|
||||
|
||||
| What happened | Example reasons | Mailbox? | The route answers |
|
||||
|---|---|---|---|
|
||||
| The display never had it | `no_socket`, `refused`, `disabled`, `unsupported`, a connect or send that timed out, `forbidden` / `busy` at the door, `invalid_request` (refused by the client itself) | yes | success, `transport: "mailbox"`, `socket_error` |
|
||||
| A display too old to know it (the upgrade case) | `unknown_command`, `unsupported_version` | yes | as above |
|
||||
| The display had it and failed | `busy` (queue full), `invalid_args`, `internal`, a timeout or hang-up after the send, `bad_response` | no | `503` (`400` for `invalid_args`), `socket_error` |
|
||||
|
||||
A display that had the request may have applied it (a reply that timed out),
|
||||
or would refuse the mailbox copy as well (bad arguments), or is stuck and
|
||||
would not read the mailbox either (a full queue). Writing the copy anyway
|
||||
only turned that into a "success". An on-demand stop with `stop_service`
|
||||
still stops the service, which ends on-demand whatever happened.
|
||||
|
||||
Brightness and plugin reload never had a mailbox: without the socket, the
|
||||
config watcher applies the saved brightness and a reload becomes the
|
||||
restart banner, as before.
|
||||
|
||||
### The mailboxes now
|
||||
|
||||
| Mailbox | Written by | Read by the display | While the socket is up |
|
||||
|---|---|---|---|
|
||||
| `display_on_demand_request` | the web interface, only on fallback; plugins that predate `BasePlugin.request_on_demand()`, or run on a core without it | the render thread, `_poll_on_demand_requests()` | looked at every 1 s (`MAILBOX_POLL_INTERVAL_WITH_SOCKET`), 0.25 s without a socket |
|
||||
| `plugin_error_clear_request` | the web interface, only on fallback | the error publisher's thread, every 5 s tick | unchanged rate |
|
||||
|
||||
A look is one `stat()` of the mailbox file (`CacheManager.file_signature`):
|
||||
`(inode, mtime, size)`, and every write renames a new file into place, so a
|
||||
new write always looks different. `MailboxWatch` reads the file only when
|
||||
that changed since the last look, so a mailbox that holds nothing new, or
|
||||
nothing at all, costs no open and no parse. A socket command never reads or
|
||||
deletes the on-demand mailbox. A start already processed is taken out of
|
||||
the mailbox instead of being re-read until it expires.
|
||||
|
||||
A request that comes through the on-demand mailbox while the socket is up
|
||||
is logged once per writer (`came through the file mailbox although the
|
||||
control socket is up`), which names the plugins that still write it.
|
||||
|
||||
### Plugins in the display process
|
||||
|
||||
A plugin asks for the screen with `BasePlugin.request_on_demand()` and gives
|
||||
it back with `end_on_demand()` (see "On-demand display" in
|
||||
[PLUGIN_API_REFERENCE.md](PLUGIN_API_REFERENCE.md)). Neither goes through
|
||||
the socket or a file: `PluginManager` hands the mailbox-shaped request,
|
||||
marked `source: 'plugin'`, to `DisplayController.submit_plugin_on_demand`,
|
||||
which queues it in memory (at most `PLUGIN_ON_DEMAND_QUEUE_SIZE`, 32) from
|
||||
whatever thread the plugin called on, and wakes the render thread through
|
||||
the socket's queue flag (`ControlServer.wake()`). The render thread applies
|
||||
it in `_drain_control_commands`, after the socket's commands, through the
|
||||
same `_handle_on_demand_request`, so it lands within a frame like a socket
|
||||
command. Without a socket it lands on the next pending-changes pass (typically
|
||||
within 0.25 s). A plugin's stop ends only a session that plugin owns. The four
|
||||
plugins that wrote the mailbox (birdnet-go, mqtt-notifications, on-air,
|
||||
pomodoro-timer) use it where the core has it and write the mailbox
|
||||
otherwise.
|
||||
|
||||
## Robustness
|
||||
|
||||
@@ -151,9 +537,15 @@ block the render loop or crash it:
|
||||
and the connection is closed, because the next message boundary cannot be
|
||||
found. A client that disconnects mid-message is dropped silently. No
|
||||
exception from a handler leaves the connection thread.
|
||||
- **Full queue.** When the queue is full, the client gets `busy` and falls
|
||||
back to the mailbox. A full queue means the render thread is stuck, and the
|
||||
systemd watchdog deals with that.
|
||||
- **Full queue.** When the queue is full, the client gets `busy`, and the web
|
||||
interface answers `503` rather than write the mailbox, which the stuck
|
||||
render thread would not read either. A full queue means the render thread
|
||||
is stuck, and the systemd watchdog deals with that.
|
||||
- **Awaited commands.** The wait for an awaited command's outcome happens on
|
||||
its connection thread and is bounded (`AWAIT_SECONDS`), so a stuck render
|
||||
thread costs that client `pending` and one connection slot for at most
|
||||
10 s. The render thread settles an outcome without blocking; one nobody is
|
||||
waiting for any more is simply dropped.
|
||||
- **Startup.** The server binds under a temporary name, sets the mode and the
|
||||
group, then renames the socket into place, so it never appears with the
|
||||
umask's permissions. It removes a stale socket (a file that nothing is
|
||||
@@ -162,7 +554,16 @@ block the render loop or crash it:
|
||||
process created.
|
||||
- **Never fatal.** If the server cannot start (Windows, no `AF_UNIX`, a bind
|
||||
failure, `LEDMATRIX_CONTROL_SOCKET=off`), it logs that and the display runs
|
||||
as before. The web interface then uses the mailbox.
|
||||
as before. The web interface then uses the mailbox, and reads the cache
|
||||
keys and the heartbeat file.
|
||||
- **Subscribers (stage 3).** A `state.subscribe` connection gives its request
|
||||
slot back and takes one of 4 subscriber slots (`MAX_SUBSCRIBERS`). A fifth
|
||||
gets `busy`. So a few browsers' web processes holding streams can never
|
||||
use up the 8 slots that commands need. Each subscriber has its own thread.
|
||||
A send that cannot finish within the 2 s IO timeout (a reader that stopped
|
||||
reading) drops that subscriber. Nothing else waits for it, and the render
|
||||
thread only publishes to the hub. `close()` wakes every subscriber, so
|
||||
they end at once.
|
||||
|
||||
## Security model
|
||||
|
||||
@@ -191,9 +592,15 @@ anything else in the group they share:
|
||||
and is disconnected. This covers a socket mode that someone loosened by
|
||||
hand.
|
||||
|
||||
The commands are deliberately narrow. Stage 1 can start or stop on-demand
|
||||
display and read its state, which anyone who can reach the web UI can already
|
||||
do. Nothing on the socket runs a shell, writes a file, or names a path.
|
||||
The commands are deliberately narrow. They start or stop on-demand display,
|
||||
read its state, set the brightness, and reload a plugin the display is
|
||||
already running, all of which anyone who can reach the web UI can already do
|
||||
(the last by restarting the display). Nothing on the socket runs a shell,
|
||||
writes a file, or names a path, and `plugin.reload` cannot make the display
|
||||
import a plugin it was not running. Stages 2 and 3 changed none of the
|
||||
access rules above. The state stream carries what the cache keys already
|
||||
held, and those are readable by the same group. A subscriber goes through
|
||||
the same connect-time and peer-credential checks as any other connection.
|
||||
|
||||
**Development.** A display that is not root and cannot write to
|
||||
`/run/ledmatrix`, such as `python3 run.py -e` from a checkout, serves the
|
||||
@@ -205,34 +612,77 @@ device never touches the live display.
|
||||
|
||||
## Stage plan
|
||||
|
||||
1. **On-demand, with acks (this stage).** Contract, server, client.
|
||||
1. **On-demand, with acks (done, #706).** Contract, server, client.
|
||||
`on_demand.start`/`stop`/`status`, `hello`, `ping`. The REST routes try the
|
||||
socket first and report `transport: "socket" | "mailbox"` (plus
|
||||
`socket_error` on fallback). The mailbox is unchanged, and the plugins that
|
||||
write it directly (birdnet-go, mqtt-notifications, on-air, pomodoro-timer)
|
||||
keep working.
|
||||
2. **Commands that are restarts or polls today.**
|
||||
- `brightness.set`, transient and with no `config.json` write.
|
||||
2. **Commands that were restarts or polls (done).**
|
||||
- The render thread waits on the queue instead of sleeping, and Vegas
|
||||
checks it every frame, so a command lands within a frame on every kind
|
||||
of screen (see "Waking the render thread").
|
||||
- `brightness.set`, transient and with no `config.json` write. `POST
|
||||
/api/v3/config/main` sends it after saving a brightness and reports
|
||||
`brightness_transport`; without the socket the config watcher applies
|
||||
the saved value, as before.
|
||||
- `plugin.reload`, which replaces the `restart_required` answer from #688
|
||||
with a live reload of the updated plugin on the render thread.
|
||||
- `config.reload`, which applies a saved config without waiting for the 2 s
|
||||
mtime poll and acks which sections changed.
|
||||
- The dwell sleep and the static screen's 1 s frame sleep wait on the
|
||||
queue instead of sleeping, so a command lands within milliseconds on
|
||||
every kind of screen. Under WSL, with a static plugin on screen, a stop
|
||||
takes 1.0 s by either path today.
|
||||
3. **A state stream.** A `subscribe` command that keeps the connection open
|
||||
and pushes events: mode changes, on-demand state, plugin runtime state and
|
||||
the heartbeat. It replaces the polled `display_current_state`,
|
||||
`plugin_runtime_snapshot` (#690) and `display-heartbeat.json` (#687) for
|
||||
readers that hold a connection. The web interface relays it to its
|
||||
existing SSE stream. The files remain for one release for older readers.
|
||||
4. **Retire the mailboxes.** After a release in which every device has had the
|
||||
socket, the web interface stops writing `display_on_demand_request`, and
|
||||
the display stops polling it, logging the plugins that still write it so
|
||||
they can move to an in-process `request_display()`. The other cache keys
|
||||
used as messages (`plugin_error_clear_request` and the remaining
|
||||
`display_*` keys) move to the socket or to tmpfs.
|
||||
for a store update of an enabled plugin. `POST /api/v3/plugins/update`
|
||||
answers `restart_required: false, reloaded: true` once the new code
|
||||
runs, and falls back to the restart banner (with `reload_error`)
|
||||
otherwise.
|
||||
- `config.reload` was left out. Its only gain over the config watcher
|
||||
would be skipping the watcher's 2 s mtime poll, and the one setting
|
||||
where those seconds show, brightness, now has its own command. Plugin
|
||||
settings already reach the running plugin through the watcher, and the
|
||||
"which sections changed" ack had no reader: the web interface knows
|
||||
what it saved. A reload from the socket thread would also run every
|
||||
config subscriber on a second thread beside the watcher's.
|
||||
3. **A state stream (done).** `state.get` (a versioned snapshot) and
|
||||
`state.subscribe` (the snapshot, then pushed changes and keepalive ticks)
|
||||
carry the current mode, the on-demand state (including the outcome of an
|
||||
acked on-demand command), the brightness, the plugin runtime snapshot and
|
||||
the render loop's liveness, all served from memory (see "The state
|
||||
stream"). The web interface's readers use it and fall back to the cache
|
||||
keys and the heartbeat file. `display_current_state` and
|
||||
`plugin_runtime_snapshot` are written less often while it serves them.
|
||||
The keys remain for one release.
|
||||
- Left for later: the outcome of a `plugin.reload` that answered
|
||||
`pending` is visible only as the plugin's new `loaded_version` in
|
||||
`state.plugins`, not as an event of its own.
|
||||
- Left for later: the SSE display stream reads the preview frame, not
|
||||
state, so nothing relays the stream to the browser yet. A browser still
|
||||
polls the REST routes, which now answer from memory.
|
||||
- Left for later: the store's install of an already-enabled plugin, and
|
||||
an uninstall that keeps its config, still answer `restart_required`.
|
||||
They can now use a load/unload command and report the result the same
|
||||
way the update route does.
|
||||
4. **The mailboxes become a fallback (done).**
|
||||
- The web interface writes a mailbox only when the socket could not carry
|
||||
the request (`should_fall_back`); a display that had it and failed is
|
||||
answered as that (see "When the web interface falls back").
|
||||
- `errors.clear` replaces `plugin_error_clear_request` as the way a clear
|
||||
reaches the display.
|
||||
- The display looks at the on-demand mailbox once a second while the
|
||||
socket is up, reads either mailbox only when its file changed, and logs
|
||||
who still writes the on-demand one (see "The mailboxes now").
|
||||
- Not changed, deliberately: config saves (the schedule, the dim
|
||||
schedule, plugin settings) still reach the display through
|
||||
`config.json` and its watcher, which is the setting itself rather than
|
||||
a message; see `config.reload` under stage 2. The preview viewer marker
|
||||
(`/tmp/led_matrix_preview_viewer`) is a presence signal the display
|
||||
already stats at most once a second. Plugin health and metrics resets
|
||||
write the persisted record the display publishes and do not reach the
|
||||
running display (their routes say so); they are not mailboxes.
|
||||
5. **Remove the mailboxes (next release).** Once every device has run a
|
||||
display with stage 4, the web interface stops writing both mailboxes and
|
||||
the display stops reading them. The four plugins that wrote
|
||||
`display_on_demand_request` now have an in-process way to ask for the
|
||||
screen (`BasePlugin.request_on_demand()` / `end_on_demand()`, see
|
||||
"Plugins in the display process"); they keep the mailbox write only as
|
||||
their fallback on older cores. The display also stops writing `display_current_state`,
|
||||
`display_on_demand_state` and `plugin_runtime_snapshot` once the web
|
||||
interface no longer falls back to them.
|
||||
|
||||
## Checking it on a device
|
||||
|
||||
@@ -247,4 +697,46 @@ curl -s -X POST localhost:5000/api/v3/display/on-demand/start \
|
||||
If the response says `"transport": "mailbox"`, `socket_error` gives the
|
||||
reason. `no_socket` means the display is stopped or predates the socket.
|
||||
`refused` usually means the web user is not in the socket's group, which
|
||||
takes effect when the web service restarts after the user is added.
|
||||
takes effect when the web service restarts after the user is added. A `503`
|
||||
with `"transport": "socket"` means the display had the request and did not
|
||||
take it (`busy`, `timeout`, ...): nothing was written to the mailbox.
|
||||
|
||||
An error clear:
|
||||
|
||||
```bash
|
||||
curl -s -X POST localhost:5000/api/v3/errors/clear \
|
||||
-H 'Content-Type: application/json' -d '{"all":true}'
|
||||
# ... "applied": true, "transport": "socket"
|
||||
sudo journalctl -u ledmatrix | grep -E "Cleared .* plugin error|file mailbox"
|
||||
```
|
||||
|
||||
Brightness and a plugin reload:
|
||||
|
||||
```bash
|
||||
curl -s -X POST localhost:5000/api/v3/config/main \
|
||||
-H 'Content-Type: application/json' -d '{"brightness":40}'
|
||||
# ... "brightness_transport": "socket"
|
||||
curl -s -X POST localhost:5000/api/v3/plugins/update \
|
||||
-H 'Content-Type: application/json' -d '{"plugin_id":"clock-simple"}'
|
||||
# after a real update of an enabled plugin: "restart_required": false, "reloaded": true
|
||||
sudo journalctl -u ledmatrix | grep -E "Brightness set|Reload(ing|ed) plugin"
|
||||
```
|
||||
|
||||
`unknown_command` in `brightness_socket_error` or `reload_error` means the
|
||||
display runs a stage-1 build: restart it once to pick up this one.
|
||||
|
||||
The state stream:
|
||||
|
||||
```bash
|
||||
curl -s localhost:5000/api/v3/display/current-status # ... "source": "socket"
|
||||
curl -s localhost:5000/api/v3/health | python3 -m json.tool | grep -A3 display_loop
|
||||
python3 - <<'EOF'
|
||||
from src.ipc import client # run from the project directory
|
||||
snap = client.state_get()
|
||||
print(snap['version'], snap['epoch'], snap['loop'], snap['state']['display'])
|
||||
EOF
|
||||
```
|
||||
|
||||
`"source": "cache"` means the web interface could not use the socket: the
|
||||
display is stopped, predates stage 3, or the web user is not in the
|
||||
socket's group.
|
||||
|
||||
+20
-1
@@ -33,6 +33,9 @@ in again (services pick them up on restart).
|
||||
| `/run/ledmatrix/control.sock` | `root` : cache directory's group (`ledmatrix`) | `660` | The display's control socket; only root and that group can connect. See [IPC_CONTROL_SOCKET.md](IPC_CONTROL_SOCKET.md#security-model) |
|
||||
| `scripts/fix_perms/safe_plugin_rm.sh`, `safe_pip_install.sh` | `root:root` | `755` | Run as root through sudo, so the web user must not be able to edit them |
|
||||
| `/etc/sudoers.d/ledmatrix_web`, `ledmatrix_wifi` | `root` | `440` | |
|
||||
| `/usr/local/sbin/ledmatrix-refresh-units` | `root:root` | `755` | Copy of `scripts/install/ledmatrix_refresh_units.py`, installed by `install_service.sh`. Outside the project so the web user cannot edit what sudo runs |
|
||||
| `/var/lib/ledmatrix/unit-backup/` | `root` | `700` | The units the last refresh replaced, for the automatic update's rollback |
|
||||
| `/etc/systemd/system/ledmatrix*.service`, `.path` | `root:root` | `644` | Readable so the web interface can compare them with the templates after an update |
|
||||
|
||||
What keeps it that way at runtime:
|
||||
|
||||
@@ -86,6 +89,20 @@ password:
|
||||
- `journalctl -u ledmatrix.service *`, `-u ledmatrix *`, `-t ledmatrix *`,
|
||||
tagged `NOEXEC`: journalctl opens a pager on a terminal, and a shell
|
||||
escape from that pager would be a root shell
|
||||
- `/usr/local/sbin/ledmatrix-refresh-units ""` and
|
||||
`/usr/local/sbin/ledmatrix-refresh-units --restore` — exactly these two
|
||||
command lines (`""` means "no arguments"). After an update the first
|
||||
installs the systemd units whose templates changed and runs
|
||||
`systemctl daemon-reload`; the automatic update's rollback runs the second
|
||||
to put the previous units back. The helper takes nothing from the caller:
|
||||
the project folder and the web user come from the installed, root-owned
|
||||
`ledmatrix.service` and `ledmatrix-web.service`. It only replaces the four
|
||||
units `install_service.sh` installs, only if they are already installed,
|
||||
and refuses a template that would change a unit's `User=` (root for the
|
||||
display, the web user for the rest) or `WorkingDirectory=`, or that is a
|
||||
symlink, not a regular file, or over 64 KB. It grants nothing new: the
|
||||
templates are files the web user can edit, but so is `run.py`, which the
|
||||
display service already runs as root.
|
||||
|
||||
### `/etc/sudoers.d/ledmatrix_wifi`
|
||||
|
||||
@@ -136,7 +153,9 @@ directory.
|
||||
| `safe_plugin_rm.sh`, `safe_pip_install.sh` | — | Called by the web interface through sudo | Not for manual use |
|
||||
|
||||
To reinstall the sudoers rules, run
|
||||
`./scripts/install/configure_web_sudo.sh` (web rules) or
|
||||
`./scripts/install/configure_web_sudo.sh` (web rules; the
|
||||
`ledmatrix-refresh-units` rules also need the helper itself, which
|
||||
`sudo ./scripts/install/install_service.sh` installs) or
|
||||
`./scripts/install/configure_wifi_permissions.sh` (WiFi rules and polkit) as
|
||||
the web user, not with `sudo`.
|
||||
|
||||
|
||||
@@ -14,6 +14,7 @@ Complete API reference for plugin developers. This document describes all method
|
||||
- [Display Manager](#display-manager)
|
||||
- [Cache Manager](#cache-manager)
|
||||
- [Plugin Manager](#plugin-manager)
|
||||
- [Fetching data](#fetching-data)
|
||||
- [Deprecated APIs](#deprecated-apis)
|
||||
|
||||
---
|
||||
@@ -487,6 +488,78 @@ working for the plugin itself. `get_vegas_segment_width()` read the
|
||||
`vegas_panel_count` config value, which has never affected Vegas — a card's
|
||||
width comes from `get_vegas_content()` and `vegas_width_pct`.
|
||||
|
||||
### On-demand display
|
||||
|
||||
A plugin that reacts to something outside the rotation (an MQTT message, a
|
||||
timer, a detection) can take the screen for it, and give it back. Both
|
||||
methods are safe from any thread, including an MQTT callback: they only
|
||||
queue the request, and the display applies it on its render thread within a
|
||||
frame or so, exactly like an on-demand start or stop from the web interface.
|
||||
|
||||
#### `request_on_demand(mode=None, duration=None, pinned=False) -> Optional[str]`
|
||||
|
||||
Show this plugin now.
|
||||
|
||||
- `mode`: one of the plugin's display modes; `None` for its first.
|
||||
- `duration`: seconds before the rotation resumes; `None` (or `0`) for no
|
||||
limit, until `end_on_demand()` or the user stops it.
|
||||
- `pinned`: stay on `mode` instead of cycling through the plugin's other
|
||||
modes.
|
||||
|
||||
Returns the request id once the display has queued it, or `None` when
|
||||
there is no display in this process to ask (the web interface's plugin
|
||||
manager, `scripts/check_plugin.py`) or its queue is full. A bad argument
|
||||
(a `mode` that is not a string, a `duration` that is not a number) raises
|
||||
`ValueError`.
|
||||
|
||||
#### `end_on_demand() -> Optional[str]`
|
||||
|
||||
Give the screen back. Ends only a session this plugin owns: a session the
|
||||
user started for another plugin, or one that already ended, is left alone.
|
||||
Returns the request id once queued, or `None` as above.
|
||||
|
||||
#### Older cores: feature detection
|
||||
|
||||
These methods are new after core 3.8.0 (see `CHANGELOG.md`). Before them,
|
||||
plugins wrote the `display_on_demand_request` cache key (the "mailbox")
|
||||
themselves. The display reads it only once a second while the control
|
||||
socket is up, and it will be removed in a future release (see
|
||||
[IPC_CONTROL_SOCKET.md](IPC_CONTROL_SOCKET.md), stage 5). A plugin that
|
||||
must keep working on older cores checks for the method, and writes the
|
||||
mailbox only when the method is missing or answers `None`:
|
||||
|
||||
```python
|
||||
import time, uuid
|
||||
|
||||
def _show_alert(self):
|
||||
if hasattr(self, "request_on_demand") and self.request_on_demand(
|
||||
mode="my_alert", duration=15):
|
||||
return
|
||||
# Older core, or no display in this process: the mailbox, as before.
|
||||
self.cache_manager.set("display_on_demand_request", {
|
||||
"request_id": str(uuid.uuid4()), "action": "start",
|
||||
"plugin_id": self.plugin_id, "mode": "my_alert",
|
||||
"duration": 15, "pinned": False, "timestamp": time.time(),
|
||||
})
|
||||
|
||||
def _release(self):
|
||||
if hasattr(self, "end_on_demand") and self.end_on_demand():
|
||||
return
|
||||
self.cache_manager.set("display_on_demand_request", {
|
||||
"request_id": str(uuid.uuid4()), "action": "stop",
|
||||
"plugin_id": self.plugin_id, "timestamp": time.time(),
|
||||
})
|
||||
```
|
||||
|
||||
Keep `ledmatrix_min_version` where it is: the fallback is what keeps the
|
||||
plugin working on older cores. A mailbox stop ends any on-demand session,
|
||||
whoever started it; `end_on_demand()` ends only the plugin's own.
|
||||
|
||||
Both methods answer a request id only when the plugin manager returned a
|
||||
string, so a test that gives the plugin a `MagicMock()` plugin manager gets
|
||||
`None` and exercises the mailbox path. To test the new path, set
|
||||
`plugin_manager.request_on_demand.return_value = "some-id"`.
|
||||
|
||||
> The full source for `BasePlugin` lives in
|
||||
> `src/plugin_system/base_plugin.py`. If a method here disagrees with the
|
||||
> source, the source wins — please open an issue or PR to fix the doc.
|
||||
@@ -965,6 +1038,14 @@ if info:
|
||||
self.logger.info(f"Plugin: {info['name']}, Version: {info.get('version')}")
|
||||
```
|
||||
|
||||
#### `request_on_demand(plugin_id, mode=None, duration=None, pinned=False)` / `end_on_demand(plugin_id)`
|
||||
|
||||
What `BasePlugin.request_on_demand()` and `end_on_demand()` call, with the
|
||||
plugin's own id. Call those instead; see
|
||||
[On-demand display](#on-demand-display). The display controller routes them
|
||||
to itself with `set_on_demand_handler()`; a plugin manager without a
|
||||
display behind it answers `None`.
|
||||
|
||||
#### `get_all_plugin_info() -> List[Dict[str, Any]]`
|
||||
|
||||
Get information for all plugins.
|
||||
@@ -1030,6 +1111,103 @@ if weather is not None and weather.enabled:
|
||||
|
||||
---
|
||||
|
||||
## Fetching data
|
||||
|
||||
Use the core helpers for HTTP rather than a `requests.Session` of your own:
|
||||
`APIHelper` (`from src.common import APIHelper`) for JSON APIs, and
|
||||
`fetch_espn_scoreboard()` (`src.common.espn_dates`) or
|
||||
`BackgroundDataService` for ESPN scoreboards. Since the release after 3.7.0
|
||||
these go through the core **fetch service** (`src/common/fetch_service.py`),
|
||||
so a plugin that uses them gets the following with no code change. Return
|
||||
values, exceptions and retries are what they were.
|
||||
|
||||
- **Shared connections.** Core sessions with the same retry policy share one
|
||||
connection pool per host, instead of one pool per helper.
|
||||
- **Merged requests.** Identical GETs in flight at the same time (same URL
|
||||
and query, headers, timeout and retry policy) go to the network once, and
|
||||
every caller gets its own copy of the response, or the same exception.
|
||||
- **Host budgets.** A host can have a token-bucket budget. A request past it
|
||||
waits for a token, but never longer than `max_wait_seconds` (2 s by
|
||||
default). Only ESPN hosts have one by default (20 requests a second, burst
|
||||
200), which normal use never reaches.
|
||||
- **Conditional GET.** When a server sends `ETag` or `Last-Modified`, the
|
||||
next identical request revalidates, and a `304 Not Modified` comes back to
|
||||
your code as the original `200` with its body. ESPN currently sends
|
||||
neither, so this does nothing there.
|
||||
- **Response cache.** A response whose server says `Cache-Control:
|
||||
max-age=N` answers an identical GET for those N seconds without a
|
||||
request (ESPN sends 1 to ~500 s). It never hands you a response older
|
||||
than you accept: pass `cache_max_age=<your TTL>` to `fetch_get()` or
|
||||
`fetch_espn_scoreboard()` (0 always asks the network); without it a
|
||||
response is reused for at most 30 seconds.
|
||||
- **Counters.** Requests, merged requests, bytes, 304s, errors and time spent
|
||||
waiting, and requests answered without the network (`memo_hits` from the
|
||||
response cache, `cache_hits` from a shared scoreboard cache entry), are
|
||||
counted per plugin and per host, and published for the web UI
|
||||
at `GET /api/v3/plugins/fetch-stats` (see
|
||||
[REST_API_REFERENCE.md](REST_API_REFERENCE.md#get-fetch-statistics)). A
|
||||
request is counted against your plugin when it runs inside your
|
||||
`update()`/`display()`, your constructor or `on_enable()`, or anywhere in
|
||||
code under your plugin's directory, including threads you start.
|
||||
|
||||
What is not covered yet: requests a plugin makes with its own `requests.get()`
|
||||
or `Session.get()` calls. They work as before but are invisible to the
|
||||
budgets and counters.
|
||||
|
||||
### One cache key per ESPN scoreboard
|
||||
|
||||
Cache an ESPN scoreboard under `espn_scoreboard_cache_key(sport, league,
|
||||
dates)` (`src.common.espn_dates`), not a key of your own, so every plugin
|
||||
showing that league shares one fetch and one cached copy. `sport` and
|
||||
`league` are ESPN's path segments (`football`, `college-football`), and
|
||||
`dates` is what you send as `dates=` (`"20261004"`, `"202610"`,
|
||||
`"20260925-20261016"`, a `date`, or `None` for the undated scoreboard).
|
||||
|
||||
```python
|
||||
from src.common.espn_dates import get_espn_scoreboard
|
||||
|
||||
data = get_espn_scoreboard(
|
||||
self.session, "football", "nfl", "20261004",
|
||||
cache_manager=self.cache_manager,
|
||||
max_age=300, # your TTL: nothing older comes back
|
||||
legacy_keys=["my_old_key_20261004"], # read once while upgrading
|
||||
)
|
||||
```
|
||||
|
||||
`get_espn_scoreboard` returns a cached copy at most `max_age` seconds old,
|
||||
whoever wrote it, and otherwise fetches with `fetch_espn_scoreboard`
|
||||
(`limit=500`, ranges split the way ESPN requires) and caches the result
|
||||
without a ttl, so each reader applies its own age limit. `max_age=0` always
|
||||
fetches but still leaves the copy for others. For a two-step read, use
|
||||
`read_espn_scoreboard_cache()` and `store_espn_scoreboard_cache()` around
|
||||
your own fetch. Scoreboards built on `SportsFetchMixin` get
|
||||
`_schedule_cache_key(datestring)` and `_cached_schedule(key, legacy_keys)`
|
||||
for their schedule windows. All of this is in the core release after 3.8.0.
|
||||
|
||||
The settings live in `config.json` under `fetch_service`, read when the
|
||||
display starts and on a config reload:
|
||||
|
||||
```json
|
||||
"fetch_service": {
|
||||
"enabled": true,
|
||||
"max_wait_seconds": 2,
|
||||
"rate_limits": {
|
||||
"*.espn.com": {"per_second": 20, "burst": 200},
|
||||
"api.example.com": {"per_second": 1, "burst": 5}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
`rate_limits` keys are a host or a `*.domain` pattern (which also matches
|
||||
the bare domain); `"per_second": 0` removes a budget. `"enabled": false`
|
||||
turns the whole service into a plain `session.get()`. Two further switches,
|
||||
`"single_flight": false` and `"conditional_get": false`, turn off merging and
|
||||
revalidation. `"response_cache": {"enabled": false}` turns off the response
|
||||
cache; its `default_max_age` (30) is the limit for callers that pass no
|
||||
`cache_max_age`.
|
||||
|
||||
---
|
||||
|
||||
## Best Practices
|
||||
|
||||
### Caching
|
||||
|
||||
+196
-37
@@ -159,11 +159,21 @@ there an unchecked checkbox — which the browser omits — is saved as
|
||||
}
|
||||
```
|
||||
|
||||
`restart_required` is always true here: display hardware, rotation,
|
||||
durations and general settings take effect when the display restarts, and
|
||||
the web UI shows its restart banner on the flag. (Plugin sections saved
|
||||
through this route reach the running plugin live, like
|
||||
`POST /plugins/config`.)
|
||||
`restart_required` is true when the save changed a setting that takes
|
||||
effect when the display restarts: display hardware, rotation order,
|
||||
timezone, general settings and the rest. The web UI shows its restart banner
|
||||
on the flag. It is false when the save changed only what the running display
|
||||
applies by itself, or nothing: `brightness`, the per-mode durations
|
||||
(`duration__<mode>`, `display.display_durations`) and plugin sections, which
|
||||
reach the running plugin live, like `POST /plugins/config`.
|
||||
|
||||
A saved `brightness` reaches the panel without a restart. The route also
|
||||
sends it to the running display over the control
|
||||
socket (`brightness.set`), which puts it on the panel at once, and the
|
||||
response adds `"brightness_transport": "socket"`. Otherwise it is
|
||||
`"config"`, with `brightness_socket_error` giving the reason, and the
|
||||
display's config watcher applies the saved value within a few seconds, as
|
||||
before.
|
||||
|
||||
Invalid values (e.g. an out-of-range `target_fps`, a hardware option the
|
||||
Raspberry Pi 5 driver cannot use) are rejected with `400` and nothing is
|
||||
@@ -238,7 +248,10 @@ Replace the schedule configuration.
|
||||
```
|
||||
|
||||
A day whose `<day>_enabled` key is absent counts as enabled, with default
|
||||
times `07:00`-`23:00`. At least one day must be enabled.
|
||||
times `07:00`-`23:00`. An enabled schedule needs at least one day enabled; a
|
||||
disabled one (`"enabled": false`) may have every day off, as
|
||||
`config.template.json` ships it. A day that is off keeps the times sent for
|
||||
it, when they are valid `HH:MM`.
|
||||
|
||||
**Response**:
|
||||
```json
|
||||
@@ -323,12 +336,23 @@ by the display process (stale after 120 seconds).
|
||||
"data": {
|
||||
"mode": "nfl_live",
|
||||
"plugin_id": "football-scoreboard",
|
||||
"last_updated": 1234567890.123
|
||||
"last_updated": 1234567890.123,
|
||||
"source": "socket"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
When nothing has been published, every field is `null`.
|
||||
When nothing has been published, every field is `null`. `source` is
|
||||
`socket` when the answer came from the display's state stream over the
|
||||
control socket ([IPC_CONTROL_SOCKET.md](IPC_CONTROL_SOCKET.md)), and `cache`
|
||||
when it came from the `display_current_state` cache key (no socket: the
|
||||
display is stopped or older, or this is Windows). A display whose render
|
||||
loop has not refreshed its state for 120 seconds is reported with every
|
||||
field `null`, either way. So is a stopped display: when the socket does not
|
||||
answer and the render loop's heartbeat
|
||||
(`/run/ledmatrix/display-heartbeat.json`) is absent, stale or from a process
|
||||
that is gone, the cache's last entry is not used. A display still beating
|
||||
without a socket, Windows, or a socket switched off reads the cache.
|
||||
|
||||
### List Display Modes
|
||||
|
||||
@@ -405,11 +429,16 @@ Get the current on-demand display state.
|
||||
"returncode": 0,
|
||||
"stdout": "active",
|
||||
"stderr": ""
|
||||
}
|
||||
},
|
||||
"source": "socket"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
`source` is `socket` (the display's state stream, with `remaining` worked
|
||||
out at the time of the request) or `cache` (the `display_on_demand_state`
|
||||
cache key).
|
||||
|
||||
With no on-demand request, `state` is
|
||||
`{"active": false, "status": "idle", "last_updated": null}`.
|
||||
|
||||
@@ -435,7 +464,7 @@ Request a specific plugin to display on-demand.
|
||||
- `mode` (string, optional): Display mode name (plugin_id inferred if not provided)
|
||||
- `duration` (number, optional): Duration in seconds (0 = until stopped)
|
||||
- `pinned` (boolean, optional): Pin display (pause rotation)
|
||||
- `start_service` (boolean, optional): Start the display service if it is not running (default: true). A running service is never restarted: it picks the request up within about a quarter of a second. When false and the service is stopped, the route returns 400.
|
||||
- `start_service` (boolean, optional): Start the display service if it is not running (default: true). A running service is never restarted: it picks the request up within a frame over its control socket (within about a second through the mailbox fallback). When false and the service is stopped, the route returns 400.
|
||||
|
||||
**Response**:
|
||||
```json
|
||||
@@ -456,13 +485,23 @@ Request a specific plugin to display on-demand.
|
||||
`service` is `null` when `start_service` is false.
|
||||
|
||||
`transport` says how the request reached the display: `"socket"` means the
|
||||
display's control socket acknowledged it (it is queued for the render thread;
|
||||
see [IPC_CONTROL_SOCKET.md](IPC_CONTROL_SOCKET.md)), `"mailbox"` means it was
|
||||
display's control socket acknowledged it (it is queued for the render thread,
|
||||
which wakes for it and applies it within a frame; see
|
||||
[IPC_CONTROL_SOCKET.md](IPC_CONTROL_SOCKET.md)), `"mailbox"` means it was
|
||||
written to the cache mailbox the display polls, as before the socket existed.
|
||||
With `"mailbox"`, `socket_error` gives the reason the socket was not used
|
||||
(`no_socket` when the display is stopped or predates the socket, `timeout`,
|
||||
`refused`, `busy`, ...). Either way the request is applied the same way;
|
||||
`request_id` is the same id in both.
|
||||
The mailbox is used only when the socket could not carry the request. With
|
||||
`"mailbox"`, `socket_error` gives the reason (`no_socket` when the display is
|
||||
stopped or predates the socket, `refused`, a connect `timeout`,
|
||||
`unknown_command` from a display too old for the command, ...). Either way
|
||||
the request is applied the same way; `request_id` is the same id in both.
|
||||
|
||||
When the display had the request and did not take it -- a full queue
|
||||
(`busy`), bad arguments (`invalid_args`), no answer after the request was
|
||||
sent (`timeout`, `closed`) -- the route answers `503` (`400` for
|
||||
`invalid_args`) with `status: "error"` and `data: {request_id, transport:
|
||||
"socket", socket_error}`, and writes nothing to the mailbox. The stop route
|
||||
does the same, except that with `stop_service: true` it still stops the
|
||||
service and answers success.
|
||||
|
||||
### Stop On-Demand Display
|
||||
|
||||
@@ -542,7 +581,9 @@ List all installed plugins with their status and metadata.
|
||||
"status": "live",
|
||||
"published_at": 1790000030.0,
|
||||
"age_seconds": 12.4,
|
||||
"stale_after": 180.0
|
||||
"stale_after": 180.0,
|
||||
"heartbeat_age_seconds": 2.1,
|
||||
"source": "socket"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -563,10 +604,18 @@ until the display restarts. A plugin a live snapshot does not list is
|
||||
`loaded: false`, `state: "unloaded"`.
|
||||
|
||||
`runtime.status` says whether to believe them: `live` (fresh snapshot from
|
||||
a running display), `stale` (not refreshed within `stale_after` seconds: the
|
||||
display is hung or died), `stopped` (the display shut down) or `unknown`
|
||||
a running display), `stalled` (fresh snapshot, but the same process's
|
||||
render-loop heartbeat is 60 s or older -- the render loop is hung, as
|
||||
[`/health`](#health-check)'s `display_loop: stalled` says), `stale` (not refreshed
|
||||
within `stale_after` seconds, or the process that wrote it no longer exists:
|
||||
the display is hung or died), `stopped` (the display shut down) or `unknown`
|
||||
(nothing published yet). Unless it is `live`, every one of those fields is
|
||||
`null`. Health and metrics are at [`/plugins/health`](#get-plugin-health)
|
||||
`null`. `heartbeat_age_seconds` is the heartbeat's age when it was taken into
|
||||
account, `null` otherwise (no heartbeat, as on the dev server, or one from
|
||||
another process). `runtime.source` is `socket` when the snapshot and the
|
||||
heartbeat age came from the display's state stream over the control socket,
|
||||
and `cache` when they came from the `plugin_runtime_snapshot` cache key and
|
||||
the heartbeat file; the rules above are the same for both. Health and metrics are at [`/plugins/health`](#get-plugin-health)
|
||||
and `/plugins/metrics`.
|
||||
|
||||
`vegas_participation` is what Vegas mode does with the plugin: `"scroll"`,
|
||||
@@ -812,8 +861,29 @@ Update a plugin to the latest version. Runs synchronously.
|
||||
```
|
||||
|
||||
`update_status` is `updated`, `up_to_date` or `local_only`.
|
||||
`restart_required` is true when the plugin changed and is enabled: the
|
||||
running display keeps the code it loaded until it restarts.
|
||||
|
||||
When the plugin changed and is enabled, the route asks the running display
|
||||
to reload it over the control socket (`plugin.reload`, see
|
||||
[IPC_CONTROL_SOCKET.md](IPC_CONTROL_SOCKET.md)). Once the new code is
|
||||
running, the answer is:
|
||||
|
||||
```json
|
||||
{
|
||||
"status": "success",
|
||||
"message": "Plugin football-scoreboard updated to version 2.1.0; the display is running the new version",
|
||||
"restart_required": false,
|
||||
"reloaded": true,
|
||||
"reloaded_version": "2.1.0"
|
||||
}
|
||||
```
|
||||
|
||||
If the display could not reload it, `restart_required` is true (the running
|
||||
display keeps the code it loaded until it restarts) and `reload_error` says
|
||||
why: `no_socket` (the display is stopped or predates the socket),
|
||||
`unknown_command` (a display older than this command), `not_loaded`,
|
||||
`failed` (the new version did not load; it is out of the rotation),
|
||||
`pending` (not done within 10 s; it will still be reloaded), or another
|
||||
transport reason.
|
||||
|
||||
An update this core cannot run answers `409` with `Plugin update refused:`
|
||||
and the reason; the installed version is left as it was.
|
||||
@@ -980,6 +1050,76 @@ Metrics for one plugin; `data` has the same fields as one entry above.
|
||||
|
||||
Reset metrics for a plugin.
|
||||
|
||||
### Get Fetch Statistics
|
||||
|
||||
**GET** `/api/v3/plugins/fetch-stats`
|
||||
|
||||
Network requests made through the core fetch service
|
||||
(`src/common/fetch_service.py`), per plugin and per host, cumulative since
|
||||
the display started. Read-only. The display publishes the counters at most
|
||||
once a minute when they change (every 10 minutes otherwise), so they can be
|
||||
up to a minute old. Requests a plugin makes with its own `requests` calls,
|
||||
outside `APIHelper`, `espn_dates`, `BackgroundDataService` and
|
||||
`BaseOddsManager`, are not counted yet.
|
||||
|
||||
`data.status` is `live`, `stale` (no publish for longer than
|
||||
`stale_after`), `stopped` (the display exited; the last counters are kept)
|
||||
or `unknown` (nothing published; `data.data` is `null`).
|
||||
|
||||
**Response**:
|
||||
```json
|
||||
{
|
||||
"status": "success",
|
||||
"data": {
|
||||
"status": "live",
|
||||
"age_seconds": 12.4,
|
||||
"data": {
|
||||
"schema": 1,
|
||||
"running": true,
|
||||
"published_at": 1790000000.0,
|
||||
"stale_after": 720.0,
|
||||
"since": 1789990000.0,
|
||||
"totals": {"requests": 412, "merged": 3, "not_modified": 0,
|
||||
"errors": 1, "http_errors": 2, "retries": 0,
|
||||
"throttled": 0, "overruns": 0, "bytes": 18234011,
|
||||
"wait_seconds": 0.0, "memo_hits": 21, "cache_hits": 40,
|
||||
"legacy_cache_hits": 2},
|
||||
"plugins": {
|
||||
"football-scoreboard": {"requests": 240, "merged": 2, "bytes": 9120330,
|
||||
"hosts": {"site.api.espn.com": 180,
|
||||
"sports.core.api.espn.com": 62},
|
||||
"...": "the other counters, as in totals"}
|
||||
},
|
||||
"hosts": {
|
||||
"site.api.espn.com": {"requests": 301, "...": "as in totals"}
|
||||
},
|
||||
"validators": {"entries": 0, "bytes": 0},
|
||||
"response_cache": {"entries": 3, "bytes": 412004},
|
||||
"config": {"enabled": true, "single_flight": true,
|
||||
"conditional_get": true, "max_wait_seconds": 2.0,
|
||||
"rate_limits": {"*.espn.com": {"per_second": 20.0, "burst": 200.0}},
|
||||
"response_cache": true, "default_max_age": 30.0}
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
`requests` counts round trips sent (retries inside the HTTP adapter are in
|
||||
`retries`), `merged` requests answered by an identical one already in
|
||||
flight, `not_modified` 304s served from the stored body, `errors` transport
|
||||
failures and `http_errors` responses with status 400 or above. `bytes` is the
|
||||
decoded body size. `core` is everything no plugin made.
|
||||
|
||||
Three counters are requests that never reached the network: `memo_hits`
|
||||
were answered from the short response cache (a response still inside the
|
||||
`Cache-Control: max-age` its server gave it), and `cache_hits` were
|
||||
scoreboard fetches answered from a shared ESPN scoreboard cache entry
|
||||
(`espn_scoreboard_cache_key`). `legacy_cache_hits` counts reads served from a
|
||||
key that predates the shared one; it should fall to zero within a day of an
|
||||
upgrade. A plugin's `hosts` counts are requests plus merged requests,
|
||||
`memo_hits` and `cache_hits`: everything it asked for.
|
||||
`response_cache` is the size of the response cache now.
|
||||
|
||||
### Get/Set Plugin Limits
|
||||
|
||||
**GET** `/api/v3/plugins/limits/<plugin_id>`
|
||||
@@ -1051,7 +1191,7 @@ it is neither installed nor configured).
|
||||
"last_updated": "2025-01-15T10:30:00"
|
||||
}
|
||||
},
|
||||
"runtime": {"status": "live", "published_at": 1790000030.0, "age_seconds": 12.4, "stale_after": 180.0}
|
||||
"runtime": {"status": "live", "published_at": 1790000030.0, "age_seconds": 12.4, "stale_after": 180.0, "heartbeat_age_seconds": 2.1}
|
||||
}
|
||||
```
|
||||
|
||||
@@ -2117,13 +2257,11 @@ error with `"all": true` (`max_age_hours` is then ignored).
|
||||
}
|
||||
```
|
||||
|
||||
The clear is asynchronous. The web interface records a request
|
||||
(`plugin_error_clear_request` in the shared cache), and the display service
|
||||
applies it within about 5 seconds, rebuilding its counts from the errors it
|
||||
keeps and republishing. Reads hide the cleared errors from the moment the
|
||||
request is recorded. Until the display service applies an age-based clear,
|
||||
`recent_errors` and `active_patterns` are already filtered but the counts
|
||||
are the old ones, and `clear_pending` is `true`.
|
||||
The clear goes to the display service over its control socket
|
||||
(`errors.clear`, see [IPC_CONTROL_SOCKET.md](IPC_CONTROL_SOCKET.md)), which
|
||||
applies it, rebuilding its counts from the errors it keeps, and republishes
|
||||
before it answers: `applied` is `true`, `transport` is `"socket"`, and
|
||||
`cleared_count` is the display's own count.
|
||||
|
||||
```json
|
||||
{
|
||||
@@ -2131,17 +2269,30 @@ are the old ones, and `clear_pending` is `true`.
|
||||
"data": {
|
||||
"cleared_count": 13,
|
||||
"clear_requested": true,
|
||||
"applied": true,
|
||||
"transport": "socket",
|
||||
"request_id": "5f0c1e...",
|
||||
"cutoff": "2026-09-23T09:59:02.310000"
|
||||
},
|
||||
"message": "Clear of all errors requested; the display service applies it within about 5 seconds"
|
||||
"message": "Cleared all errors"
|
||||
}
|
||||
```
|
||||
|
||||
`cleared_count` is how many of the reported errors the clear hides. It is
|
||||
When the socket cannot carry it (the display is stopped, or older than
|
||||
`errors.clear`) the clear is asynchronous, as before: the web interface
|
||||
records a request (`plugin_error_clear_request` in the shared cache),
|
||||
`applied` is `false` and `transport` is `"mailbox"`, and the display service
|
||||
applies it within about 5 seconds. Reads hide the cleared errors from the
|
||||
moment the request is recorded. Until the display service applies an
|
||||
age-based clear, `recent_errors` and `active_patterns` are already filtered
|
||||
but the counts are the old ones, and `clear_pending` is `true`. Then
|
||||
`cleared_count` is how many of the reported errors the clear hides, and
|
||||
`null` when that cannot be known before the display service applies it (an
|
||||
age-based clear over more errors than the report lists). A request that
|
||||
could not be written to the shared cache answers `500`.
|
||||
age-based clear over more errors than the report lists).
|
||||
|
||||
A request that could not be written to the shared cache answers `500`. A
|
||||
display that had the request and failed it (`internal`, a timeout after the
|
||||
request was sent) answers `503`, with `context.socket_error`.
|
||||
|
||||
---
|
||||
|
||||
@@ -2159,7 +2310,14 @@ display snapshot. `data.status` is `healthy` or `degraded`, with
|
||||
(with `heartbeat_age_seconds`), `stalled` (no heartbeat for 60s: the panel is
|
||||
frozen even if the service is active; the status turns `degraded`), or
|
||||
`not_reported` when the display writes none (not started yet, the dev server,
|
||||
Windows), which does not affect the status.
|
||||
Windows), which does not affect the status, or `stopped` (with `source:
|
||||
"service"`) when the display service is not active, the control socket does
|
||||
not answer and there is no live heartbeat; the status then turns
|
||||
`degraded`. A platform with no control socket (Windows) or a socket switched
|
||||
off never reports `stopped`. Its `source` is `socket` when the
|
||||
age came from the display's state stream over the control socket (measured
|
||||
in memory by the display) and `heartbeat_file` when it came from
|
||||
`/run/ledmatrix/display-heartbeat.json`.
|
||||
|
||||
Open even when the web login is on, for uptime monitors; a caller that is not
|
||||
logged in (and has no token) then gets only `{"status": "success", "data":
|
||||
@@ -2219,8 +2377,9 @@ Replace the dim schedule. `dim_brightness` is 0-100 (default 30). In
|
||||
`per-day` mode the days can be sent either as the `days` object that GET
|
||||
returns, or as the web form's flat fields (`monday_enabled`,
|
||||
`monday_start`, `monday_end`, ...). A day that is not sent counts as
|
||||
enabled with default times `20:00`-`07:00`; at least one day must be
|
||||
enabled.
|
||||
enabled with default times `20:00`-`07:00`. As for the schedule above, an
|
||||
enabled dim schedule needs at least one day enabled and a disabled one may
|
||||
have every day off.
|
||||
|
||||
---
|
||||
|
||||
|
||||
@@ -0,0 +1,471 @@
|
||||
# Restructuring `DisplayController.run()`
|
||||
|
||||
`run()` in [`src/display_controller.py`](../src/display_controller.py) decides
|
||||
what the panel shows and runs it. This document is the plan for turning it
|
||||
from one long loop into three parts with clear jobs: an **Arbiter** that
|
||||
decides, a **ScreenRunner** that runs one screen, and **Sources** that each
|
||||
know about one kind of content. It covers the target design, the stages that
|
||||
get there, and how each stage is checked.
|
||||
|
||||
The goal is to change how the control flow is organised, not to move code
|
||||
into more files. Each stage ships as its own PR, and none of them changes
|
||||
what the panel shows unless that PR says so and updates the golden traces
|
||||
on purpose.
|
||||
|
||||
## Why
|
||||
|
||||
- **The priority order is written in branch order, twice.** It is
|
||||
Follower, on-demand, WiFi notice, live priority, Vegas, rotation. In
|
||||
`run()` that order exists only as the order of `if` blocks. Vegas
|
||||
repeats part of it in its interrupt callback (`_check_vegas_interrupt`).
|
||||
- **Preemption is found by re-checking.** A screen ends early when
|
||||
something else changed `current_display_mode` or `is_display_active`
|
||||
underneath it. `run()` notices with five separate
|
||||
`current_display_mode != active_mode` checks: after an empty pass, in each
|
||||
of the two frame loops, after the frame loops, and before rotating.
|
||||
- **Most recent fixes were ordering bugs** between these branches (#618,
|
||||
#644, #649, #652): a lost mode switch, rotating past an on-demand request,
|
||||
spinning when every mode is empty.
|
||||
- **It could not be tested** without threads, real sleeps and stopping the
|
||||
loop by raising from a patched method.
|
||||
|
||||
## What `run()` does today
|
||||
|
||||
Each pass, in order:
|
||||
|
||||
1. `loop_pass()` (watchdog). Apply a pending plugin enable/disable, then
|
||||
any plugin reloads the control socket asked for
|
||||
(`_apply_pending_plugin_reloads`; a pending reload ends the screen
|
||||
before it, like a WiFi notice, as `Source.RELOAD` at the runner's
|
||||
service points). The static screen's frame sleep and the dwell wait on
|
||||
the socket's queue instead of sleeping (`_wait_frame_interval`,
|
||||
`_sleep_with_plugin_updates`); without a socket, as in the golden
|
||||
traces, they are the plain sleeps.
|
||||
2. With no modes: dwell 1 s, next pass.
|
||||
3. Poll on-demand requests and expiry, release plugins loaded only for
|
||||
on-demand, tick plugin updates, drop an expired WiFi notice, evaluate
|
||||
the schedule (an on-demand session overrides scheduled-off), apply the
|
||||
brightness target. Then gather the Arbiter's inputs
|
||||
(`_arbiter_inputs`) and call `Arbiter.decide()`.
|
||||
4. **Scheduled off:** blank, dwell up to 60 s. `_blank_while_scheduled_off`
|
||||
5. **Follower:** render one frame from the leader. `_run_follower_frame`
|
||||
6. **WiFi notice** (unless on-demand): draw it, dwell 0.5 s. `_show_wifi_notice`.
|
||||
It also ends a running screen within about a second (the runner's
|
||||
service points), and a screen cut short resumes after it.
|
||||
7. **The Sources below the notice:** read whether Vegas is on and make the
|
||||
live-priority scan (`_arbiter_inputs_below_wifi`, where run() always
|
||||
read them), and call `decide()` again. It answers OnDemand (the
|
||||
session's current mode), Live (the next live mode, round-robin; a game
|
||||
that goes live during a screen takes over at the next service point, at
|
||||
most once a second), Vegas (`LEGACY`) or Rotation. `_take_plan` applies
|
||||
the answer: a live claim or the resume when live priority ends, the
|
||||
on-demand index.
|
||||
8. **Vegas** (`_run_vegas_iteration`): one iteration of up to
|
||||
`max_cycle_duration`. A completed iteration ends the pass, and so does
|
||||
one that yielded for the schedule, a reload or a WiFi notice. Any other
|
||||
interrupted one asks `decide()` once more (`vegas_yielded`): a game that
|
||||
stopped the ticker, or an on-demand session that started, shows next.
|
||||
9. **One screen:** pick the plugin (`_plugin_for_mode`) and hand the plan
|
||||
to the `ScreenRunner` (`src/screen_runner.py`). It draws the first frame
|
||||
through the executor (`_dispatch_first_frame`), has the controller fill
|
||||
in the plugin's durations, dynamic flag and frame policy
|
||||
(`_complete_plan`), runs the 125 Hz or 1 Hz frame loop with a service
|
||||
point after each frame, makes up the minimum duration, and returns an
|
||||
`Outcome`. On `PREEMPTED` the pass ends without advancing. On no
|
||||
content, rotate at once (`_note_empty_pass`, `_skip_failed_plugin_modes`).
|
||||
Otherwise `ArbiterState.after()` picks the next mode
|
||||
(`_advance_after_screen`).
|
||||
|
||||
## Design
|
||||
|
||||
```python
|
||||
def run(self):
|
||||
while True:
|
||||
inputs = self._arbiter_inputs() # schedule, follower, notice
|
||||
plan = Arbiter.decide(self._arbiter_state(), inputs, now)
|
||||
... # off / follower / notice
|
||||
plan = self._take_plan(Arbiter.decide(state, self._arbiter_inputs_below_wifi(inputs), now))
|
||||
if plan.source is Source.LEGACY: # Vegas, until stage 4
|
||||
plan = self._run_vegas_iteration(...)
|
||||
outcome = runner.run(plan, plugin) # ExitReason + elapsed
|
||||
if outcome.exit_reason is not ExitReason.PREEMPTED:
|
||||
self._advance_after_screen(plan, outcome) # ArbiterState.after
|
||||
```
|
||||
|
||||
The controller's attributes (`current_display_mode`, `current_mode_index`,
|
||||
`on_demand_*`, `_live_resume_index`) stay the record that the web UI, the
|
||||
control socket and the on-demand cache read. `_arbiter_state()` snapshots
|
||||
them into a frozen `ArbiterState`; the transitions are pure methods on it,
|
||||
and the controller writes their result back (`_adopt_state`).
|
||||
|
||||
### Sources
|
||||
|
||||
Each kind of content is a Source. A Source looks at the state and the
|
||||
inputs and either offers a screen or passes. The Arbiter asks them in this
|
||||
order:
|
||||
|
||||
| Order | Source | Offers a screen when | Code |
|
||||
|---|---|---|---|
|
||||
| gate | ScheduledOff | the schedule is off and no on-demand session overrides it | `decide` |
|
||||
| 1 | Follower | a sync leader is driving this panel | `decide` |
|
||||
| 2 | OnDemand | a session is active (its mode list, index, expiry and pin) | `_on_demand_plan` |
|
||||
| 3 | Wifi | a status message is pending and on-demand is not active | `decide` |
|
||||
| 4 | Live | a live-priority plugin has live content (round-robin across several) | `live_pick` |
|
||||
| 5 | Vegas | Vegas is enabled and nothing above wants the panel | `LEGACY`, run by `_run_vegas_iteration` |
|
||||
| 6 | Rotation | always: the rotation's current mode | `rotation_plan` |
|
||||
|
||||
ScheduledOff is a gate in front of the Sources because that is how it works
|
||||
today: a scheduled-off panel stays blank even for a follower, and only an
|
||||
on-demand session overrides it.
|
||||
|
||||
The Rotation answers `state.current_mode`, not
|
||||
`available_modes[current_mode_index]`: the two agree except where something
|
||||
moved the panel off the list and the rotation carries on from there (a live
|
||||
mode no rotation entry names, or None after a session ended with no enabled
|
||||
mode to resume to), and `run()` always showed `current_display_mode`.
|
||||
|
||||
### Arbiter
|
||||
|
||||
```python
|
||||
Arbiter.decide(state, inputs, now, running=None) -> ScreenPlan
|
||||
```
|
||||
|
||||
`decide` is a pure function: it does no I/O, takes no locks and does not
|
||||
sleep. It can be tested with plain tables of (state, inputs, now) mapped to
|
||||
an expected plan.
|
||||
|
||||
- `ArbiterState`: the current mode; the rotation and its index; the
|
||||
on-demand session's modes, index, expiry and pin; the live resume point;
|
||||
whether a mid-screen takeover has not shown yet. Transitions:
|
||||
`next_on_demand`, `showing`, `claim_live`, `release_live`, `after`.
|
||||
- `ArbiterInputs`: whether the schedule has the panel on, an on-demand
|
||||
session, a follower, the WiFi notice, the live modes (None where no scan
|
||||
was made), whether Vegas is on and keeps live content in its ticker,
|
||||
whether this pass's Vegas iteration has yielded, and (mid-screen) whether
|
||||
a plugin reload is waiting.
|
||||
- `ScreenPlan`:
|
||||
|
||||
| Field | Meaning |
|
||||
|---|---|
|
||||
| `source` | which Source won |
|
||||
| `mode`, `plugin` | what to draw (None for a blank or follower plan); the plugin id once resolved |
|
||||
| `min_duration`, `max_duration` | from `_resolve_durations` and the on-demand bound (`on_demand_bound`), filled in after the first frame; an on-demand plan's `max_duration` is what is left of the session at `now` |
|
||||
| `dynamic` | run until the plugin's cycle completes, between min and max |
|
||||
| `frame_policy` | `HIGH_FPS` or `STATIC`, today `_needs_high_fps`; see stage 5 |
|
||||
| `preemptible_by` | the Sources allowed to interrupt this plan mid-screen |
|
||||
| `notice`, `deadline`, `ends_live` | the WiFi notice; the on-demand expiry for the bound; "live priority just ended, resume the rotation first" |
|
||||
|
||||
`decide` cannot ask a plugin anything, so the fields a plugin answers are
|
||||
filled in by the controller after the first frame, where they were always
|
||||
read (`_complete_plan`).
|
||||
|
||||
With `running`, `decide` answers the mid-screen question instead: `running`
|
||||
itself while the screen holds, else the plan that ends it
|
||||
(`_hold_or_preempt`), in the order the frame loops always checked:
|
||||
|
||||
1. Live: a game went live while a non-live screen runs. It is the one
|
||||
preemption that changes the state (the rotation moves to the live mode
|
||||
and remembers where it was), and it is claimed even when a WiFi notice
|
||||
is also pending; the next pass shows the notice, then the game.
|
||||
2. The panel's mode moved under the screen (on-demand started, ended or
|
||||
changed mode; the rotation was rebuilt).
|
||||
3. The schedule turned the panel off.
|
||||
4. A WiFi notice (unless on-demand outranks it), compared with its expiry.
|
||||
5. A plugin reload is waiting (between frames only).
|
||||
|
||||
Every screen is preemptible by the gate, OnDemand, Wifi, Live, Rotation and
|
||||
a reload (`SCREEN_PREEMPTERS`), except that a live screen leaves Live out
|
||||
(`LIVE_PREEMPTERS`): live games take turns between screens. A follower and
|
||||
Vegas are looked at only between screens.
|
||||
|
||||
### ScreenRunner
|
||||
|
||||
```python
|
||||
ScreenRunner(clock: FrameClock, host: ScreenHost).run(plan, plugin) -> Outcome
|
||||
```
|
||||
|
||||
The ScreenRunner draws the first frame (`_dispatch_first_frame`), runs the
|
||||
frame loop that the plan's frame policy selects, services pending changes
|
||||
between frames, and returns one `ExitReason`:
|
||||
|
||||
| ExitReason | Golden-trace exit |
|
||||
|---|---|
|
||||
| `DURATION` | target duration reached (`duration`) |
|
||||
| `CYCLE_COMPLETE` | dynamic plugin finished after its minimum (`cycle-complete`) |
|
||||
| `EMPTY` | first frame returned False, or no plugin (`empty`; `raised` when display() raised inside the executor; `no-plugin`, `breaker`) |
|
||||
| `ERROR` | the dispatch itself raised (`error`) |
|
||||
| `DISPLAY_FALSE` | a later frame returned False (`display-false`) |
|
||||
| `PREEMPTED` | another Source took the panel (`on-demand-*`, `schedule-off`, `live`, `wifi`, ...) |
|
||||
| `RELOAD` | a plugin reload is waiting: the screen ends early but counts as shown, and the rotation advances |
|
||||
|
||||
`PREEMPTED` replaces the five `current_display_mode != active_mode` checks.
|
||||
The runner asks its host at named service points (`Checkpoint`): `FRAME`
|
||||
after each frame (and when a socket command wakes the 1 Hz wait),
|
||||
`AFTER_LOOP` / `AFTER_COMPLETED_LOOP` when the frame loop ends,
|
||||
`after_dwell` after the make-up dwell, and `FINAL` before the rotation
|
||||
advances. Each is one `decide(..., running=plan)` call
|
||||
(`DisplayController._screen_check`). The checkpoint says whether a pending
|
||||
reload counts there and when the WiFi notice file is read (`NoticeRead`):
|
||||
the read is throttled to once a second and deletes an expired file, so it
|
||||
happens exactly where the loop always read it.
|
||||
|
||||
In the 125 Hz loop the live-priority scan is made before the frame's sleep
|
||||
(`_screen_service`), at the moments it always was, and weighed by the
|
||||
service point after the sleep, where the loop always decided to end the
|
||||
screen.
|
||||
|
||||
`FrameClock` provides `time()`, `perf_counter()` and `sleep()`, the shape of
|
||||
the `time` module. In production it is `_ModuleClock`, which looks up
|
||||
`src.display_controller.time` on each call, so the golden traces' fake clock
|
||||
drives the runner as it drove the inline loops. The runner's log lines use
|
||||
the controller's logger, so they keep their source in the journal.
|
||||
|
||||
## Stages
|
||||
|
||||
| Stage | Change | Behaviour change | Verified by |
|
||||
|---|---|---|---|
|
||||
| 1 | Golden traces; extract helpers from `run()` | none | traces generated on main pass unchanged; mutation check |
|
||||
| 2 | Arbiter with Follower and Wifi Sources | none | traces unchanged; Arbiter unit tables; ledpi smoke |
|
||||
| 3 | ScreenRunner, FrameClock, ExitReason, `PREEMPTED`; OnDemand, Live, Rotation Sources | none | traces unchanged; ledpi frame soak A/B |
|
||||
| 4 | Vegas as a Source driven by `run_frame()` | none intended | traces against the real coordinator; ledpi Vegas soak A/B |
|
||||
| 5 | Plugins declare `frame_policy` | DEBUG instead of INFO for the FPS line | traces; soak on a static-heavy rotation |
|
||||
|
||||
### Stage 1 (#704)
|
||||
|
||||
- `test/_run_loop_harness.py` builds a real `DisplayController` through
|
||||
`__init__` on in-memory fakes (plugins, cache, config service, plugin
|
||||
manager, sync manager, display manager). It swaps the module's `time` and
|
||||
`datetime` for one fake clock and runs the real `run()` until a horizon.
|
||||
The first frame of each screen still goes through the real
|
||||
`PluginExecutor` and the per-plugin locks.
|
||||
- `test/test_run_loop_golden.py` has 15 scenarios, each compared with
|
||||
`test/fixtures/run_loop_golden/<scenario>.json`:
|
||||
- plain rotation (display_durations override, a high-FPS scroller, a
|
||||
plugin whose `display()` takes no `display_mode`)
|
||||
- empty modes and a mode with no plugin; an all-empty rotation (the 1 s
|
||||
pause)
|
||||
- plugin errors and the circuit breaker
|
||||
- dynamic duration (cycle complete, plugin cap, global cap)
|
||||
- live priority taking over and handing back; live round-robin
|
||||
- on-demand start/stop/expiry; pinned on-demand; a session resumed after
|
||||
a restart, and one that cannot resume (its plugin did not load); a
|
||||
request naming a live mode the plugin's live check would drop
|
||||
- schedule off and dim, with an on-demand override during downtime
|
||||
- WiFi notice; sync follower
|
||||
- Vegas, with and without `live_in_ticker`
|
||||
- Each trace row is `[start, mode, duration, exit_reason, frames,
|
||||
force_clear]`. The exit reason is the event that decided what came next.
|
||||
- All 18 tests run in under a second. The goldens were generated from
|
||||
main's `run()` before any code moved.
|
||||
- Vegas uses `FakeVegas`, which implements only the contract the controller
|
||||
depends on: `run_iteration()` returns True after its duration and False
|
||||
when the interrupt or live check asks it to yield, checking at the real
|
||||
coordinator's cadence. Running the real coordinator on the fake clock
|
||||
belongs to stage 4.
|
||||
- Twelve helpers were extracted from `run()` (listed under "What `run()`
|
||||
does today"). Breaking any one of them fails at least one golden trace.
|
||||
|
||||
### Stage 2: Arbiter, starting with Follower and Wifi (done)
|
||||
|
||||
1. Add `ScreenPlan` and an `Arbiter` with the ScheduledOff gate, Follower
|
||||
and Wifi. Every other case returns a `LEGACY` plan, which means "carry on
|
||||
with the existing code" (steps 7-9).
|
||||
2. `run()` calls `decide()` after the bookkeeping in step 3 and dispatches
|
||||
on `plan.source`: blank, `_run_follower_frame()`, the WiFi notice, or the
|
||||
existing path. Inputs that Sources read (follower active, the pending
|
||||
WiFi message, schedule state) are collected first, so `decide()` stays
|
||||
pure.
|
||||
3. Unit-test `decide()` with tables. The golden traces must not change.
|
||||
The Wifi Source must keep the mid-screen preemption described in step 6
|
||||
of "What `run()` does today".
|
||||
|
||||
Follower and Wifi go first because each is one self-contained branch that
|
||||
ends the pass. They prove the plumbing without touching the frame loops.
|
||||
|
||||
What shipped:
|
||||
|
||||
- `src/display_arbiter.py` (on the mypy ratchet) holds `Source`
|
||||
(`SCHEDULED_OFF`, `FOLLOWER`, `WIFI`, `LEGACY`), `ArbiterInputs`,
|
||||
`ArbiterState`, `WifiNotice`, `ScreenPlan` and `Arbiter.decide`.
|
||||
`ScreenPlan` has only the fields stage 2 uses: `source`, `max_duration`
|
||||
(60 s for the blank, 0.5 s for the notice, the constants `run()` used to
|
||||
hard-code) and `notice`. `mode`, `plugin`, the other durations,
|
||||
`frame_policy` and `preemptible_by` arrive with the Sources that need them.
|
||||
- `ArbiterState` is empty: no stage-2 Source remembers anything between
|
||||
passes. `now` is passed but not read, because the top-of-pass WiFi check
|
||||
never compared the expiry and must not start (the table pins this).
|
||||
- `ArbiterInputs` holds `schedule_on`, `on_demand_active`,
|
||||
`follower_active` and `wifi_notice`. `_arbiter_inputs` derives
|
||||
`schedule_on` as `is_display_active and not on_demand_schedule_override`,
|
||||
so the gate (blank when the schedule is off and no on-demand session
|
||||
overrides it) blanks exactly when `is_display_active` is False, as before,
|
||||
including #714's on-demand ending in off hours. It reads the WiFi notice
|
||||
only when the notice could win, because `_check_wifi_status_message` has
|
||||
side effects (its 1 Hz throttle, deleting an expired file) that those
|
||||
passes never had.
|
||||
- The mid-screen rule is `wifi_notice_preempts(notice, on_demand, now)`,
|
||||
which `_wifi_notice_pending` calls; it does compare the expiry.
|
||||
- `run()` still calls `_publish_current_mode_state_if_changed`,
|
||||
`_apply_pending_vegas_init` and `process_deferred_updates` at the same
|
||||
points relative to the branches, so the order of side effects in a pass
|
||||
is unchanged.
|
||||
- `test/test_display_arbiter.py`: the 16-row table (every combination of
|
||||
the four inputs, written out), the mid-screen table, purity checks (no
|
||||
clock reads, nothing mutated, no I/O imports), and the controller's
|
||||
snapshot through an on-demand session that overrides the schedule and
|
||||
ends. A mutation run broke 23 pieces once each (the gate, the order, each
|
||||
Source, the dwells, the expiry comparison, the snapshot's reads, each
|
||||
dispatch in `run()`); every one failed a test.
|
||||
|
||||
### Stage 3: ScreenRunner and `PREEMPTED` (done; awaiting the ledpi soak)
|
||||
|
||||
The plan, from where stage 2 left off:
|
||||
|
||||
1. `ArbiterState` gains the rotation index, the on-demand mode list, index,
|
||||
expiry and pin, and the live resume point. `ArbiterInputs` gains the
|
||||
live modes and whether Vegas is enabled and keeps live content in the
|
||||
ticker.
|
||||
2. OnDemand returns its current mode with the session's bound, reading
|
||||
`now` for the expiry. Live returns the next live mode (round-robin).
|
||||
Rotation returns the rotation's mode. `ScreenPlan` gains `mode`,
|
||||
`plugin`, `min_duration`, `max_duration`, `dynamic`, `frame_policy` and
|
||||
`preemptible_by`.
|
||||
3. `ScreenRunner.run(plan)` returns an `ExitReason`; `state.after(outcome)`
|
||||
replaces `_advance_after_screen`'s step and the live-resume bookkeeping.
|
||||
Each mid-screen check asks `decide()` whether a Source in
|
||||
`plan.preemptible_by` now wins.
|
||||
4. The control socket (`_drain_control_commands`, `_wait_for_control`) and
|
||||
state publishing stay where they are; the runner calls them at its
|
||||
service points.
|
||||
|
||||
What shipped, one commit each: the runner; then the OnDemand, Live and
|
||||
Rotation Sources; then one `decide()` call at the service points.
|
||||
|
||||
- `src/screen_runner.py` (on the mypy ratchet): `ScreenRunner`,
|
||||
`FrameClock`, `ExitReason`, `Outcome`, `Checkpoint`, `NoticeRead`,
|
||||
`Screen` and the `ScreenHost` protocol, which `DisplayController`
|
||||
implements through `_ScreenHost` (one-line forwards to its own methods).
|
||||
The two frame loops, the make-up dwell and the dynamic-duration exit
|
||||
moved in unchanged, pacing included.
|
||||
- `src/display_arbiter.py`: `Source` gains `ON_DEMAND`, `LIVE`, `ROTATION`
|
||||
and `RELOAD`; `LEGACY` means only Vegas. `FramePolicy`. `ArbiterState`
|
||||
and `ArbiterInputs` as listed under "Arbiter". The pure helpers
|
||||
`on_demand_bound` (`_clamp_to_on_demand`), `live_pick`
|
||||
(`_check_live_priority`'s pick), `live_takeover` (the mid-screen claim)
|
||||
and `rotation_plan`.
|
||||
- A pass asks `decide()` twice: once with the inputs every pass reads, and
|
||||
once, only when nothing above the notice took the panel, with the Vegas
|
||||
check and the live scan, read where run() always read them (the scan
|
||||
asks every live-priority plugin, and the Vegas check applies queued
|
||||
Vegas config, so reading them earlier, on a follower or notice pass,
|
||||
would be a change). A Vegas iteration that yields asks a third time.
|
||||
- `_resolve_active_mode`, `_clamp_to_on_demand` and `_screen_preempted` are
|
||||
gone. `_apply_live_priority`, `_check_live_priority`,
|
||||
`_check_live_takeover` and `_wifi_notice_pending` remain (Vegas, the
|
||||
dwell sleep and the tests call them), built on the same pure rules.
|
||||
- `_sleep_with_plugin_updates` keeps its own break rules. It also serves
|
||||
the blank, the notice and the idle wait, which are not screens, and its
|
||||
rules are edge-triggered (an on-demand session starting on the mode
|
||||
already showing ends a dwell but not a frame loop); folding them into
|
||||
`decide()` would change behaviour.
|
||||
|
||||
Behaviour, checked three ways:
|
||||
|
||||
- Golden traces: unchanged, no regeneration.
|
||||
- Every harness run in the suite (67: the goldens plus the live-takeover,
|
||||
WiFi+live, socket-wake, plugin-reload, schedule and tick tests) was
|
||||
captured with every sleep, `display()` call, WiFi read, live scan,
|
||||
publish, dwell and scroll-state call logged, and diffed against
|
||||
`origin/main`. Identical, except:
|
||||
- a Vegas pass used to scan the live plugins twice at the same instant
|
||||
(step 7, then step 8's "is anything live?"); it scans once;
|
||||
- `_apply_live_priority(None)` calls that changed nothing are not made;
|
||||
- throttled WiFi reads that returned the cached answer (no side effect)
|
||||
after a notice had already ended the screen are not made;
|
||||
- in the 125 Hz loop the live scan still runs before the frame's sleep,
|
||||
but the claim is made by the service point after it, so the "live"
|
||||
state change happens 8 ms later. The screen ends at the same frame as
|
||||
before.
|
||||
- Tables: `test/test_display_arbiter.py` (OnDemand, Live, Vegas/Rotation,
|
||||
`after`, the 24-row mid-screen table, `live_takeover`) and
|
||||
`test/test_screen_runner.py` (the runner on a scripted host and fake
|
||||
clock; the controller's service point and the reads it makes). A
|
||||
mutation run broke each moved or new piece once; see the PR.
|
||||
|
||||
This stage touches frame pacing (the 8 ms deadline sleep, the 1 ms yield),
|
||||
so it needs a frame soak on ledpi, A/B against main, before it merges.
|
||||
Coordinate with whoever owns scroll performance (`docs/SCROLL_PERFORMANCE.md`).
|
||||
|
||||
### Stage 4: Vegas as a Source
|
||||
|
||||
The controller calls `coordinator.run_frame()` once per frame from the
|
||||
ScreenRunner instead of handing over to `run_iteration()` for up to
|
||||
`max_cycle_duration`. The interrupt callback and the second copy of the
|
||||
priority order go away, because preemption becomes `PREEMPTED`. The
|
||||
`vegas-plugin-tick` thread that is spawned every 4 s becomes the
|
||||
controller's normal update tick. Extend the harness to drive the real
|
||||
coordinator on the fake clock, which means patching its `time` and running
|
||||
its prefetch inline. Verify with a Vegas soak on ledpi, A/B.
|
||||
|
||||
### Stage 5: `frame_policy`
|
||||
|
||||
Plugins declare `frame_policy` (STATIC, PERIODIC(hz), ANIMATED(fps),
|
||||
SCROLL). `_needs_high_fps` becomes the mapping for legacy plugins
|
||||
(`needs_high_fps`, the `static-image` special case, `enable_scrolling`),
|
||||
and its per-screen INFO line drops to DEBUG. It is already read twice per
|
||||
screen: once quietly before the first frame, so `_dispatch_first_frame` can
|
||||
end the previous scroll for a screen that runs the 1 Hz loop
|
||||
(`_start_screen_handover`), and once after it to pick the loop. A declared
|
||||
policy answers both.
|
||||
|
||||
## How each stage is verified
|
||||
|
||||
- **Golden traces.** Run `python -m pytest test/test_run_loop_golden.py`;
|
||||
it takes about a second. A refactoring stage must leave every trace
|
||||
unchanged. A deliberate behaviour change regenerates them with
|
||||
`LEDMATRIX_REGEN_GOLDEN=1` in its own commit, and the commit message
|
||||
explains each changed row. A new scenario's golden is generated against
|
||||
main's `run()` first, then checked against the branch.
|
||||
- **Mutation check.** Break each moved or new piece once, for example take
|
||||
`max` of the caps instead of `min`, or skip the live hold. At least one
|
||||
trace must fail each time. Stage 1 did this for all twelve helpers.
|
||||
- **Full suite.** Diff the FAILED/ERROR ids against a baseline run of main
|
||||
in a separate worktree. The Windows host has a stable set of
|
||||
pre-existing failures, so never compare against zero.
|
||||
- **ledpi soak** (stages 2-5). With the service running the branch:
|
||||
`python3 scripts/frame_soak.py --preview` for 10 minutes on a scrolling
|
||||
rotation, and on Vegas for stages 3-4. Alternate which build goes first.
|
||||
Compare late-frame rate and freezes with main. Also check by hand that
|
||||
on-demand start, stop and expiry, a live game taking over and handing
|
||||
back, and the schedule turning the panel off and on all behave as before.
|
||||
|
||||
## Behaviour the traces pin down that may be wrong
|
||||
|
||||
Stage 1 recorded six behaviours as they were, each to be fixed in its own
|
||||
PR that updates the affected trace and explains why. All six are fixed:
|
||||
|
||||
- A WiFi notice was only checked between screens, and Vegas yielded to one
|
||||
and then showed a rotation screen instead. Notices now preempt within
|
||||
about a second, and Vegas yields straight to them (#712; `wifi_notice`,
|
||||
`vegas`).
|
||||
- A live game only took over between screens, and Vegas yielded to one and
|
||||
then showed a rotation screen first. Games now take over within about a
|
||||
second, and Vegas yields straight to them (#713; `live_priority`,
|
||||
`vegas`).
|
||||
- An on-demand session that ended during scheduled-off kept the panel on
|
||||
until the next minute, and a schedule window's end minute counted as on
|
||||
only sometimes. Windows are now half-open `[start, end)`, and the panel
|
||||
blanks as soon as on-demand ends in off hours (#714; `schedule`).
|
||||
|
||||
A new one found later goes the same way: record it here with the trace that
|
||||
shows it, then fix it in its own PR, not inside a restructure stage.
|
||||
|
||||
Open:
|
||||
|
||||
- Vegas stops for a sync follower (its interrupt check includes
|
||||
`is_follower_active`), but the yield path never looks at a follower, so a
|
||||
full rotation screen (20 s in the test) runs before the next pass hands
|
||||
the panel to the leader. Found by stage 3's mutation run;
|
||||
`test_screen_runner.py::TestThroughRun::test_vegas_yielding_to_a_follower_shows_a_rotation_screen_first`
|
||||
pins it. Stage 4, which drops the interrupt callback, is the natural
|
||||
place to fix it.
|
||||
+62
-16
@@ -73,6 +73,10 @@ Sample ladder for a 100 Hz panel:
|
||||
100.0 px/s (1px every 1 refresh = 100.0 fps, smooth)
|
||||
```
|
||||
|
||||
The Vegas **Scroll Speed** slider in the web UI shows the same thing live: a
|
||||
line under it says what your speed will run as on this panel, and links to the
|
||||
nearest smooth speeds.
|
||||
|
||||
### How a slow speed stays crisp
|
||||
|
||||
`SwapOnVSync(canvas, framerate_fraction)` holds each frame for N panel
|
||||
@@ -242,8 +246,16 @@ mean exactly 10 ms, so a ticker stalling on half its frames still averages to a
|
||||
healthy 100 fps. The stats line reports the tail for that reason — read the
|
||||
percentiles, not the fps.
|
||||
|
||||
Every scroller emits one line every 5 seconds covering *every* frame in that
|
||||
window, tagged with the plugin it came from:
|
||||
Every scroller summarises each 5-second window, covering *every* frame in it,
|
||||
in one line tagged with the plugin it came from. At the default log level the
|
||||
line reaches the journal only when it is worth reading: a **degraded** window
|
||||
(frame rate below 90% of the rate the window was locked to, i.e. 1 / its own
|
||||
median -- the same 0.9 Vegas's `Vegas FPS` line uses -- or more than 1% of its
|
||||
frames stalled), the first window after one (the recovery), and otherwise once
|
||||
every 5 minutes per scroller as a heartbeat, so silence means stopped rather
|
||||
than fine. Every window is logged at DEBUG: to see them all, run the display
|
||||
with `-d` or `LEDMATRIX_DEBUG=true` (see
|
||||
[CONFIG_DEBUGGING.md](CONFIG_DEBUGGING.md#enable-debug-logging)).
|
||||
|
||||
```bash
|
||||
journalctl -u ledmatrix --since "-10min" --no-pager | grep "Scroll frame stats"
|
||||
@@ -282,6 +294,10 @@ journalctl -u ledmatrix --since "-3h" --no-pager | grep "Scroll frame stats" \
|
||||
| sort -k7 -rn
|
||||
```
|
||||
|
||||
At the default log level that ranks the windows the journal kept -- the
|
||||
degraded ones, recoveries and heartbeats -- so it over-weights bad windows;
|
||||
rank a debug run for an unbiased average, or soak the rig (below).
|
||||
|
||||
The `$2 < 1000` guard drops windows whose median is a whole second or more.
|
||||
Those are not frames. Until the idle-gap fix in `log_frame_rate()`, the first
|
||||
frame of every scroll was timed against the end of the *previous* scroll, so
|
||||
@@ -331,18 +347,24 @@ python3 scripts/frame_soak.py --json a.json # keep the report to compare later
|
||||
It runs as any user next to the display service and stops nothing. It needs
|
||||
something to *scroll* during the run: a live game holding a static scoreboard
|
||||
on screen gives no verdict. `--preview` keeps the web preview's viewer marker
|
||||
fresh, which puts the preview's PNG encoding at full rate -- run it as the web
|
||||
service's user.
|
||||
fresh, which puts the preview's PNG encoding at the viewer rate, as an open
|
||||
preview does -- run it as the web service's user. That rate is at most one
|
||||
frame a second. Through 3.8.0 it was up to five, so a `--preview` soak taken
|
||||
before that change is not comparable with one taken after it (the hdpi
|
||||
results below are from before it): take both sides of an A/B pair on
|
||||
the same side of it.
|
||||
|
||||
| line | what it tells you |
|
||||
|---|---|
|
||||
| **Late frames** | Frames presented one or more refreshes after they were due: the panel showed the previous frame again, a visible hitch. **The pass/fail number**, 0.1% by default (`--max-late-pct`). Only intervals between two scrolling frames count, and a frame held for `frame_hold` refreshes is due `frame_hold` refreshes after the last. |
|
||||
| **Freezes** | Gaps of 250 ms or more inside a scroll: recomposes, plugin handovers, blocking calls on the render thread. Reported but not failed on, because some are handovers between plugins rather than faults. A gap still counts when the display's scroll state went missing for one frame across it, as long as scrolling resumes within 1 s: both of that frame's intervals count. Two static frames in a row end the scroll. (The state expires after 2 s without scroll activity, and plugins can clear it from their own `display()`.) The late and early rates are over frames judged against a known refresh period, which the recorder adopts once two windows in a row agree on it. |
|
||||
| **Freezes** | Gaps of 250 ms or more inside a scroll: recomposes, plugin handovers the display controller does not tag (see *Handover gaps*), blocking calls on the render thread. Reported but not failed on, because some are handovers between plugins rather than faults. A gap still counts when the display's scroll state went missing for one frame across it, as long as scrolling resumes within 1 s: both of that frame's intervals count. Two static frames in a row end the scroll. (The state expires after 2 s without scroll activity, and plugins can clear it from their own `display()`.) The late and early rates are over frames judged against a known refresh period, which the recorder adopts once two windows in a row agree on it. Handovers to a static screen no longer show up here: the display controller ends the scroll state after a static screen's first frame, where it used to linger for 2 s and turn the 1 Hz loop's second frame into a ~1 s "freeze" (17 of 31 `Render stall over` lines on ledpi, 2026-09-15 to 10-01). **Freeze counts from before and after that change are not comparable.** |
|
||||
| **Handover gaps** | Gaps of 250 ms or more from a scroll's last frame to the next screen's first: the next plugin drawing, not a scroll stalling. The display controller tags that first frame `handover`, and these gaps are counted here instead of under Freezes (`handover_freezes` in the stats; the `handover` row under *after work* counts the same ones in its freezes column). Every turn's first frame is tagged, also when the rotation comes back to the same mode (a one-mode rotation, a pinned on-demand mode, live priority holding a screen), so a scroller rebuilding its content at the start of a turn is counted here; measure work on that rebuild with this line, not Freezes. Missing from stats written by an older service, whose freezes include them. |
|
||||
| **blit** | Copying the frame into the matrix canvas (`SetImage`). It grows with width × height × `pwm_bits`: ~5.5 ms at 512×64 with 8 bits on a Pi 4. It is the biggest fixed cost, and it sets the refresh rates a rig can hold one pixel per refresh at. |
|
||||
| **wait** | Time blocked in `SwapOnVSync`, i.e. the slack left in each refresh. A p50 near zero means the rig has no headroom and anything extra lands a frame late. |
|
||||
| **work** | Everything else between two frames: drawing, scrolling, and waiting for the GIL. A wide gap between its p50 and p99 is another thread getting in the way. |
|
||||
| **Garbage collection** | Python's cyclic collector stops every thread while it runs. Collections per generation in the run and the time they took, how many took 20 ms or more, and the longest since the service started. A long one tags the next frame `gc` (see *after work*), and a `Render stall` dump says when one ran inside the stall. Diagnostic only: nothing tunes the collector. Missing from stats written by an older service. |
|
||||
| **Binding** | `STOCK` means the rgbmatrix binding holds the GIL through the vsync wait, which starves every other thread. See *Rebuilding the binding*. |
|
||||
| **after work** | Frames presented straight after tagged render-thread work, with their own late rate: `extend` and `compose` (Vegas building its strip), `patch` (live elements, once they land). A kind whose late rate sits well above the overall one is the work making frames late. Shown only when something tagged its work. |
|
||||
| **after work** | Frames presented straight after tagged render-thread work, with their own late rate: `extend` and `compose` (Vegas building its strip), `patch` (live elements, once they land), `handover` (a new screen's first frame), `gc` (a garbage collection of 20 ms or more ran since the frame before). A kind whose late rate sits well above the overall one is the work making frames late. Shown only when something tagged its work. |
|
||||
|
||||
The refresh rate is estimated from the frames themselves (swaps that block on
|
||||
vsync can only land on refresh boundaries). Cross-check it with
|
||||
@@ -356,10 +378,13 @@ A/B two of them. A live-API workload drifts over time.
|
||||
The soak says how often; the service's log says why. A scroll that presents no
|
||||
frame for 250 ms logs `Render stall:` with the stack of the render thread and
|
||||
the top of every other thread's, and whether the whole interpreter was blocked
|
||||
(C code holding the GIL) rather than one thread. To see what is behind the
|
||||
shorter hitches, run the service with `LEDMATRIX_STALL_WATCHDOG_MS=30`, which
|
||||
dumps at three refreshes late instead: its extra polling costs a little GIL
|
||||
time of its own, so do that on a diagnostic run, not a soak you are grading.
|
||||
(C code holding the GIL) rather than one thread. A stall while the next
|
||||
screen's first `display()` is still drawing says `in a handover gap` instead of
|
||||
`mid-scroll`; that call runs on a thread named `display-<plugin id>`. To see
|
||||
what is behind the shorter hitches, run the service with
|
||||
`LEDMATRIX_STALL_WATCHDOG_MS=30`, which dumps at three refreshes late instead:
|
||||
its extra polling costs a little GIL time of its own, so do that on a
|
||||
diagnostic run, not a soak you are grading.
|
||||
`LEDMATRIX_STALL_WATCHDOG=0` turns it off.
|
||||
|
||||
### Results: hdpi, 2026-09-24
|
||||
@@ -517,19 +542,43 @@ On the 2×128×64 chain above, which refreshes at about 130 Hz flat out
|
||||
|
||||
### What the display does about it
|
||||
|
||||
At one pixel per refresh, the fastest crisp speed, the step is exactly one
|
||||
refresh's worth of motion, so it can be cancelled: show one half of the panel
|
||||
The step is the motion of one refresh, so it can be cancelled: show one half of the panel
|
||||
a refresh behind the other -- the half whose row at the seam lights at the
|
||||
start of each refresh. The two rows either side of the seam then show the same
|
||||
moment again. What is left is a
|
||||
lean of one pixel per half from top to bottom, continuous across the panel,
|
||||
which reads as nothing where the step read as a tear. `DisplayManager` does
|
||||
this while something scrolls at one frame per refresh
|
||||
this while something scrolls
|
||||
(`display.scan_order_compensation`, `"auto"` by default, `"off"` to disable;
|
||||
the geometry is in `src/scan_order.py`). The lagging rows come from the
|
||||
previous frame the display presented, so it works for Vegas and every plugin
|
||||
ticker without knowing how they scroll.
|
||||
|
||||
A frame held for several refreshes (any crisp speed below the panel's full
|
||||
refresh rate, e.g. 60 px/s at 120 Hz) is presented as two swaps instead of one:
|
||||
the lagging half shows the previous frame for the first refresh and the new one
|
||||
for the rest, so it steps one refresh after the rest rather than one frame.
|
||||
That costs a second blit inside the refresh after the first swap, so it is
|
||||
skipped when a blit takes more than half a refresh.
|
||||
|
||||
A plugin screen that runs the 1 Hz loop after a scroll is not composed: the
|
||||
display controller calls `DisplayManager.end_scroll_for_static_screen()` before
|
||||
its first `display()`, so the frames that call presents go out as drawn, in one
|
||||
swap each, instead of with the lagging half taken from the scroller's last
|
||||
frame.
|
||||
|
||||
The controller's own screens -- the blank shown when the schedule turns the
|
||||
panel off, and the WiFi status message -- end the scroll state before they are
|
||||
drawn, so they go out as drawn and are timed as static frames, not as freezes
|
||||
of the old scroll. A scroller that resumes after a WiFi notice sets the state
|
||||
again on its next frame.
|
||||
|
||||
One screen that follows a scroll is still composed while the scroll state lasts
|
||||
(it expires 2 s after the scroller's last frame): a screen that runs the
|
||||
high-FPS loop without scrolling (an older `static-image`, which is forced into
|
||||
it). Its first frame takes its lagging half from the scroller's last frame, for
|
||||
one refresh after a held scroll and otherwise until its next frame.
|
||||
|
||||
Checked on hdpi (4×128×64 on one chain, rotated 180, 2026-09-24) before it was
|
||||
written: `scan_mode: 1` (interlaced) made the step vanish but turned moving
|
||||
edges grainy, and halving the speed halved it, so it is the scan and not a torn
|
||||
@@ -537,9 +586,6 @@ frame. With the compensation the step is gone at 90 px/s.
|
||||
|
||||
It is left off where the row order is unknown or the maths does not hold:
|
||||
|
||||
- **Slower speeds**, where each frame is held for two or more refreshes. The
|
||||
offset there is half a pixel or less, and cancelling it would need a lag of
|
||||
a fraction of a frame.
|
||||
- **Other layouts:** pixel mappers other than a 0 or 180 degree rotation
|
||||
(U-mapper, 90/270), non-zero `multiplexing`, interlaced `scan_mode`, and a
|
||||
canvas remapped to another height (double-sided mode).
|
||||
|
||||
+36
-19
@@ -91,6 +91,7 @@ more. Shared sports code lives in `src/common`:
|
||||
| `sports_live_scroll.py` | next release | `SportsLiveScrollMixin` — rebuild a live scroll strip mid-cycle, keeping the marquee's place |
|
||||
| `sports_display_rules.py` | next release | `SportsCardOptionsMixin`, `SportsGameRulesMixin` — scorebug date options, the no-favourites filter, non-favourite live dwell |
|
||||
| `sports_font_path.py` | next release | `resolve_font_path` — what the plugins' `_resolve_font_path` copies return |
|
||||
| `sports_game_over.py` | 3.8.1 | `SportsGameOverMixin` — `_is_game_really_over`, with the `FINAL_PERIOD` seam (family 5) |
|
||||
|
||||
Each is described in [src/common/README.md](../src/common/README.md).
|
||||
|
||||
@@ -134,8 +135,7 @@ constants rather than behavior:
|
||||
|
||||
| Attribute | Meaning | Default |
|
||||
|---|---|---|
|
||||
| `FINAL_PERIOD` | Period at/after which a zero clock can mean "over" | `4` (hockey overrides to `3`) |
|
||||
| `CLOCK_COUNTS_DOWN` | Whether `0:00` means "expired" | `True` (soccer/afl/nrl override to `False` — their clocks count up, so `0:00` is kickoff) |
|
||||
| `FINAL_PERIOD` | Period from which a 0:00 clock ends a game (`sports_game_over`) | `None`: the clock never ends a game (afl, nrl, soccer, baseball, ufc). Hockey sets `3`; basketball, football and lacrosse `4` |
|
||||
| `COALESCE_SCORING_SEQUENCE` | Fold score increments arriving during an active celebration into that one celebration | `False` (football overrides to `True` — a touchdown lands as +6, then +1 for the extra point) |
|
||||
|
||||
### Why these are seams and not branches
|
||||
@@ -146,11 +146,14 @@ so NRL matches favorites on team ID. Flattening every plugin to abbreviations
|
||||
would silently select the wrong club for NRL users. The base declares the seam,
|
||||
NRL fills it, and core never learns the string `"nrl"`.
|
||||
|
||||
`CLOCK_COUNTS_DOWN` exists for the same reason in the opposite direction: a
|
||||
`FINAL_PERIOD` exists for the same reason in the opposite direction: a
|
||||
soccer clock reading `0:00` means the match has not kicked off, so running the
|
||||
clock-expiry branch there would evict live games.
|
||||
clock-expiry rule there would evict live games. Those sports declare `None`,
|
||||
and so do baseball (innings, not a clock) and ufc (a bout ends only on ESPN's
|
||||
final status). One attribute covers both questions, whether the clock can end
|
||||
a game and from which period, so no separate count-down flag was added.
|
||||
|
||||
`COALESCE_SCORING_SEQUENCE` is the third of the same kind. In football one
|
||||
`COALESCE_SCORING_SEQUENCE` is another of the same kind. In football one
|
||||
scoring play arrives as two score updates, so the follow-up must be folded into
|
||||
the first celebration; in soccer two increments a few seconds apart are two real
|
||||
goals, and folding them would swallow one. Neither default is "right" — which is
|
||||
@@ -298,6 +301,20 @@ Left in the plugins, though identical:
|
||||
renderers) is already core's, in `SportsHelpersMixin`; a renderer that
|
||||
wants it can inherit that.
|
||||
|
||||
### Family 5: the game-over check (core done; adoption waits for a release)
|
||||
|
||||
The pilot of the method below. ledmatrix-plugins `scripts/test_game_over_check.py`
|
||||
(#621) pinned 3,115 answers across the nine plugins first; the reconcile
|
||||
(ledmatrix-plugins #625) made the five bodies one and
|
||||
changed only the cells the owner's decisions under
|
||||
[Product decisions](#product-decisions-each-family-needs) explain: ufc's
|
||||
clock rule (65 cells), baseball's dormant one (53, every one a game with a
|
||||
`period` baseball's games never carry), and a level score at 0:00 (five
|
||||
cells in hockey, basketball, football and lacrosse). The harness renders
|
||||
were pixel-identical. `src/common/sports_game_over.py` holds the body;
|
||||
`test/test_sports_game_over_parity.py` compares it, and each plugin's
|
||||
`FINAL_PERIOD`, with the plugin copies.
|
||||
|
||||
### Why the method changes
|
||||
|
||||
Byte-identical promotion has nearly run dry. Measured on ledmatrix-plugins
|
||||
@@ -333,8 +350,8 @@ game-over check); the report measures each method in it. The procedure:
|
||||
line in each plugin.
|
||||
- *A per-sport fact* (hockey ends in period 3; a soccer clock counts up).
|
||||
Make it a declared class constant or override point with a default, as
|
||||
`FINAL_PERIOD`, `CLOCK_COUNTS_DOWN`, `COALESCE_SCORING_SEQUENCE` and
|
||||
`_favorite_key` are, and add it to the tables above. Never a sport-name
|
||||
`FINAL_PERIOD`, `COALESCE_SCORING_SEQUENCE` and `_favorite_key` are,
|
||||
and add it to the tables above. Never a sport-name
|
||||
branch: core must not learn sport names.
|
||||
- *A product difference*: anything a user can see (which games show, a
|
||||
colour, a date, a badge, how long a screen stays). The owner picks the
|
||||
@@ -382,7 +399,7 @@ release.
|
||||
| # | Family | Methods (variants) | Why here |
|
||||
|---|---|---|---|
|
||||
| 4 | Identical sweep | `manager.py`: `_dispatch_switch_refresh`, `_favorite_team_is_live`, `get_vegas_priority_weight`, `_game_involves`, `_favorite_scan_targets`, `_favorite_scan_games`, `_get_total_games_for_manager` (all nine, 1); the live-scroll helpers `_preserving_scroll_position`, `_refresh_live_scroll_managers`, `_live_scroll_managers`, `_note_live_scroll_built`, `_live_scroll_needs_rebuild`, `_live_scroll_fields` (eight, 1). `sports.py`: `_card_option`, `_filtered_or_all`, `_effective_live_duration`, `_recent_date_text` (eight, 1). 58 identical families in all | Nothing to decide; brings `manager.py` into core as a `SportsPluginHostMixin`. `_resolve_font_path` (identical in nine `sports.py` and eight renderers) becomes `sports_font_path.resolve_font_path`, not `font_layout.resolve_asset_path`, which skips the cwd. Core side done; see [Stage 4](#stage-4-the-identical-sweep-core-done-adoption-waits-for-a-release) |
|
||||
| 5 | Game-over check | `SportsLive._is_game_really_over` (5) | Pure logic, no pixels; its seams (`FINAL_PERIOD`, `CLOCK_COUNTS_DOWN`) were designed in B1. The pilot for the procedure |
|
||||
| 5 | Game-over check | `SportsLive._is_game_really_over` (5) | Pure logic, no pixels; one seam, `FINAL_PERIOD`. The pilot for the procedure. Reconciled to one body and promoted as `sports_game_over`; adoption waits for the release that ships it. See [Family 5](#family-5-the-game-over-check-core-done-adoption-waits-for-a-release) |
|
||||
| 6 | Favourite matching | `_is_favorite_game` (7 across three classes), `_select_games_for_display` (2: nrl), `_select_recent_games_for_display` (3) | Everything that asks "is this a favourite" goes through the 3.5.0 `_favorite_key` seam |
|
||||
| 7 | Other-games rotation | `_by_importance`, `_other_games_window`, `_advance_other_games_if_due` (2 each: football), `_rotate_other_games_on_display` (2: ufc) | One outlier each; football carries two fixes the other eight lack |
|
||||
| 8 | Rankings | `_fetch_team_rankings` (3), `_choose_poll` (3), `_load_division_team_ids`, `_passes_other_filters`, `_best_rank`, `_is_ranked_game` (2 each: football) | Needs 7; the rank badge and the "ranked only" filter read it |
|
||||
@@ -416,17 +433,17 @@ family 9 prepares.
|
||||
Owner calls to make before (or while) reconciling. Items marked *verify* are
|
||||
suspected behaviour that needs a payload or a rig to confirm first.
|
||||
|
||||
- **5, game-over check.** Which rule each sport gets: the clock never ends a
|
||||
game in afl, nrl and soccer (`CLOCK_COUNTS_DOWN = False`); hockey ends at
|
||||
0:00 from period 3, basketball, football and lacrosse from period 4.
|
||||
baseball and ufc share a copy that reads a missing clock as "0:00": dormant
|
||||
in baseball (its games carry no `period`), and not triggered by ufc's round
|
||||
breaks either. ESPN sends a break as `STATUS_END_OF_ROUND` with displayClock
|
||||
`-`, not `0:00` (verified against recorded payloads; ledmatrix-plugins#580
|
||||
pins it). Whatever rule ufc gets must not read `-` as `0:00`. Decide ufc's
|
||||
rule: no clock rule (ESPN's `STATUS_FINAL` is the only end signal it needs;
|
||||
this also closes a ~1 s window at the horn when the ticking clock reads
|
||||
`0:00`), or its own final period.
|
||||
- **5, game-over check. Decided 2026-10-05, done:** one seam,
|
||||
`FINAL_PERIOD`: hockey 3; basketball, football and lacrosse 4; `None` (the
|
||||
clock never ends a game) for afl, nrl and soccer (clocks that count up),
|
||||
baseball (its games carry no `period`, so the old rule was dormant) and
|
||||
ufc (a bout ends only on ESPN's final status, which also closes the ~1 s
|
||||
window at the horn when the ticking clock reads `0:00`; ESPN's round-break
|
||||
displayClock `-` was never a zero clock, ledmatrix-plugins#580). Only a
|
||||
non-empty clock string counts (the baseball/ufc copy read a missing clock
|
||||
as `0:00`). A score level at 0:00 is not over: the game stays live through
|
||||
the break before overtime, and one that really ends tied ends on its final
|
||||
status. Baseball keeps its postponed/suspended override in `BaseballLive`.
|
||||
- **6, favourite matching.** NRL keeps matching favourites by team id
|
||||
(abbreviations collide: NEW, CAN), through `_favorite_key` rather than its
|
||||
own copies of the selection methods. Six plugins log the recent-games
|
||||
|
||||
+91
-3
@@ -84,6 +84,43 @@ python3 web_interface/start.py
|
||||
|
||||
### Installation & Build Issues
|
||||
|
||||
#### "This version of Raspberry Pi OS is not supported"
|
||||
|
||||
LEDMatrix installs on Raspberry Pi OS Lite **Trixie** (Debian 13, Python
|
||||
3.13) or **Bookworm** (Debian 12, Python 3.11). The installer checks
|
||||
`/etc/os-release` before it changes anything and stops on anything else.
|
||||
|
||||
**Check what you have:**
|
||||
```bash
|
||||
grep -E '^(PRETTY_NAME|VERSION_ID)=' /etc/os-release
|
||||
python3 --version
|
||||
```
|
||||
|
||||
**Solutions:**
|
||||
- `VERSION_ID="11"` (Bullseye) or older: flash a new card with Raspberry Pi
|
||||
Imager, choosing Raspberry Pi OS Lite (64-bit). Trixie is recommended;
|
||||
Bookworm (Legacy) also works. An in-place upgrade from Bullseye is not
|
||||
supported by Raspberry Pi and is not worth the risk.
|
||||
- "Desktop environment detected": use the Lite image, not the desktop one.
|
||||
- "python3 is Python 3.x; LEDMatrix needs Python 3.11 or newer": something
|
||||
has replaced the system `python3`. Point it back at the OS's own Python
|
||||
(`/usr/bin/python3` should be 3.11 on Bookworm, 3.13 on Trixie).
|
||||
- `sudo bash scripts/check_system_compatibility.sh` runs the same checks
|
||||
without installing anything.
|
||||
|
||||
#### "This Pi manages its network with dhcpcd, not NetworkManager"
|
||||
|
||||
A warning, not an error: the install carries on and the display works. But
|
||||
choosing a WiFi network from the web page and the `LEDMatrix-Setup` hotspot
|
||||
both need NetworkManager, the default on Bookworm and Trixie. It appears
|
||||
when dhcpcd was selected in `raspi-config`. Switch back with a keyboard and
|
||||
screen attached (or over Ethernet), since the WiFi connection drops briefly:
|
||||
|
||||
```bash
|
||||
sudo raspi-config # Advanced Options -> Network Config -> NetworkManager
|
||||
sudo reboot
|
||||
```
|
||||
|
||||
#### Step 6 fails: "Failed building wheel for rgbmatrix"
|
||||
|
||||
**Symptoms:**
|
||||
@@ -328,6 +365,49 @@ commit, then switches to releases on its own.
|
||||
3. **Local changes after a channel switch:** edits that no longer fit the new
|
||||
version are kept in the git stash rather than lost; `git stash list`
|
||||
shows them as "LEDMatrix autostash before update".
|
||||
4. **A new install is on a release, not `main`.** The one-shot installer
|
||||
checks out the newest release. For the newest code instead, install with
|
||||
`LEDMATRIX_CHANNEL=beta`:
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/ChuckBuilds/LEDMatrix/main/scripts/install/one-shot-install.sh | LEDMATRIX_CHANNEL=beta bash
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
#### Issue: "service settings ... are not applied yet" after an update
|
||||
|
||||
**Symptoms:**
|
||||
- Update Code's message, or the web interface log, says an update changes
|
||||
service settings that are not applied yet, and to run the installer
|
||||
- The display logs `ledmatrix.service differs from systemd/ledmatrix.service`
|
||||
at startup
|
||||
|
||||
**Explanation:** updates install the systemd units a new version changes
|
||||
through the root helper `/usr/local/sbin/ledmatrix-refresh-units`, which the
|
||||
installer sets up and grants to the web user in
|
||||
`/etc/sudoers.d/ledmatrix_web`. A device installed before that has neither,
|
||||
so the new unit settings (for example the display's watchdog) wait for a
|
||||
reinstall. The update itself is fine.
|
||||
|
||||
**Solution:** re-run the installer once, as root:
|
||||
```bash
|
||||
cd ~/LEDMatrix
|
||||
sudo ./first_time_install.sh
|
||||
# or, lighter: install the units and helper, then the sudo rules
|
||||
sudo ./scripts/install/install_service.sh
|
||||
./scripts/install/configure_web_sudo.sh
|
||||
```
|
||||
Check it worked:
|
||||
```bash
|
||||
ls -l /usr/local/sbin/ledmatrix-refresh-units # root root, rwxr-xr-x
|
||||
sudo -l | grep ledmatrix-refresh-units # the two rules
|
||||
```
|
||||
A message that the helper **refused** a unit (`refusing to install it`)
|
||||
means a template in `systemd/` was edited so that it would run as another
|
||||
account or from another folder. The message names the template. Look at
|
||||
what changed with `git diff -- systemd/`, save any edit you want to keep,
|
||||
then restore only that file, for example
|
||||
`git checkout -- systemd/ledmatrix-web.service`.
|
||||
|
||||
---
|
||||
|
||||
@@ -364,9 +444,16 @@ commit, then switches to releases on its own.
|
||||
|
||||
5. **Check required services:**
|
||||
```bash
|
||||
systemctl is-active NetworkManager # must say "active"
|
||||
sudo systemctl status hostapd
|
||||
sudo systemctl status dnsmasq
|
||||
```
|
||||
On a fresh install `hostapd` shows as **masked**. That is expected, on
|
||||
Bookworm and Trixie alike: Debian's hostapd package masks the service
|
||||
when it is installed without a configuration, so the hotspot is brought
|
||||
up through NetworkManager instead (look for `nmcli hotspot fallback` in
|
||||
`journalctl -u ledmatrix-wifi-monitor`). If NetworkManager is not
|
||||
active, see "This Pi manages its network with dhcpcd" above.
|
||||
|
||||
6. **Manually enable AP mode:**
|
||||
```bash
|
||||
@@ -590,9 +677,10 @@ stack into the log, so it says which plugin was stuck.
|
||||
apart, so a plugin that hangs on every start does not restart the display
|
||||
hundreds of times an hour.
|
||||
|
||||
4. **Is the watchdog installed?** Installs from before it keep their old unit
|
||||
until the installer is re-run (a startup warning says the unit differs
|
||||
from its template):
|
||||
4. **Is the watchdog installed?** Updates install new unit settings once the
|
||||
installer has set up `ledmatrix-refresh-units`; installs from before that
|
||||
keep their old unit until the installer is re-run (a startup warning says
|
||||
the unit differs from its template):
|
||||
```bash
|
||||
systemctl show -p WatchdogUSec ledmatrix # 2min once running; 0 = not installed
|
||||
sudo ./scripts/install/install_service.sh
|
||||
|
||||
@@ -42,10 +42,12 @@ static/v3/js/
|
||||
registry.js page lifecycle: init/destroy on htmx swaps
|
||||
api.js fetch wrapper for /api/v3 (JSON envelope, login redirect)
|
||||
facade.js window.LEDMatrix and deprecated aliases
|
||||
(later) escape.js, notify.js, dialog.js, streams.js, visibility.js,
|
||||
visibility.js ctx.visibility: a page's timers run only while it is on screen
|
||||
(later) escape.js, notify.js, dialog.js, streams.js,
|
||||
store.js (the one installed-plugin store), form/renderer.js
|
||||
pages/ one module per tab partial
|
||||
cache.js export init(root, ctx), destroy(root, ctx)
|
||||
durations.js, operation-history.js, raw-json.js, backup-restore.js
|
||||
...
|
||||
```
|
||||
|
||||
@@ -57,9 +59,43 @@ A converted partial has no `<script>`. Its root element names its page:
|
||||
<div class="..." data-page="cache"> ... </div>
|
||||
```
|
||||
|
||||
`core/boot.js` registers each page with a loader:
|
||||
`registry.register('cache', () => import('../pages/cache.js'))`. A page's
|
||||
module is fetched only when its partial first appears.
|
||||
`core/boot.js` lists each page with a loader,
|
||||
`'cache': page(function() { return import('../pages/cache.js'); })`, and
|
||||
registers them all. A page's module is fetched only when its partial first
|
||||
appears. `page()` remembers the module once loaded, so the alias of an old
|
||||
synchronous global (`validateJSON` returns a boolean) still answers
|
||||
synchronously while its page is on screen.
|
||||
|
||||
The conventions the converted pages share:
|
||||
|
||||
- **Buttons name an action.** A partial's buttons carry `data-action` (and
|
||||
any argument as another `data-*` attribute) instead of an `onclick` that
|
||||
names a global. One delegated listener on the page root handles them all,
|
||||
including rows drawn later.
|
||||
- **Server data is drawn with `textContent`**, never a markup string.
|
||||
- **Reads are cancelled, writes are not.** Loads pass `ctx.signal`, so a swap
|
||||
cancels them. Saves, deletes, exports and restores do not: the server
|
||||
finishes them anyway, so the page still reports the result in a
|
||||
notification but draws nothing into a page that has gone.
|
||||
- **Old globals become aliases.** Each `window.*` name a page used to define
|
||||
is made in `boot.js` with `alias(page, name, replacement)`, which forwards
|
||||
to the module's export of the same name and warns once.
|
||||
- **Timers are cleared in `destroy()`**, the one thing `ctx.signal` cannot
|
||||
undo by itself.
|
||||
- **Polling goes through `ctx.visibility`.** A refresh that repeats
|
||||
(`ctx.visibility.every(ms, fn)`) or work that should run only while the
|
||||
page is on screen (`ctx.visibility.whileVisible(start, stop)`) is
|
||||
registered there, never with a bare `setInterval`. It runs only while the
|
||||
page's tab is the active tab and the browser tab is visible, and it ends
|
||||
when the page is destroyed, with no code in `destroy()`.
|
||||
- **A page reports its own htmx saves.** A form whose result a page module
|
||||
shows (an `htmx:afterRequest` listener on the page root, in place of an
|
||||
`hx-on` attribute naming a global) carries `data-reports-result`. `app.js`
|
||||
then leaves the server's message to the page, as it does for a form with
|
||||
an `hx-on` after-request handler, so a save shows one notification.
|
||||
- **Server data for the module goes in `data-*` attributes**, as JSON where
|
||||
it is structured (`data-schedule-config='{{ schedule_config | tojson }}'`),
|
||||
not templated into a script.
|
||||
|
||||
`core/registry.js` handles the rest:
|
||||
|
||||
@@ -82,6 +118,7 @@ Each mount gets a `ctx` object:
|
||||
| `ctx.state` | A per-mount object for the page's own state |
|
||||
| `ctx.api` | Shared service from `boot.js` |
|
||||
| `ctx.notify` | Shared service from `boot.js` |
|
||||
| `ctx.visibility` | This page's handle on `core/visibility.js` (below), made per mount by `boot.js` through the registry's `mountContext` option |
|
||||
|
||||
A page that passes `{ signal: ctx.signal }` to `addEventListener` and
|
||||
`fetch` needs no teardown code. Its listeners and in-flight requests go
|
||||
@@ -90,6 +127,30 @@ example: its delete buttons use one delegated listener, rows are built with
|
||||
`textContent` rather than markup strings, and a newer load supersedes an
|
||||
older one.
|
||||
|
||||
### Page visibility
|
||||
|
||||
`core/visibility.js` gives each mounted page `ctx.visibility`:
|
||||
|
||||
| Member | What it does |
|
||||
|---|---|
|
||||
| `whileVisible(start, stop)` | Runs `start()` when the page comes on screen (at once, if it mounts on screen) and `stop()` when it leaves. Returns a function that ends the registration, running `stop()` first if needed |
|
||||
| `every(ms, fn)` | `fn()` at once, then every `ms` while on screen. The interval is cleared while hidden and restarted, with an immediate `fn()`, when the page is back. Returns the same kind of end function |
|
||||
| `isVisible()` | True while the page is on screen |
|
||||
| `tab` | The tab the page belongs to: its name, or `forPage(ctx, { tab })` |
|
||||
|
||||
"On screen" means the page's tab is the active tab and the browser tab is
|
||||
visible. Everything a page registered ends when its `ctx.signal` aborts,
|
||||
after `destroy()`, so a swapped-out partial leaves no interval behind.
|
||||
|
||||
The answer comes from `window.LEDVisibility` (`app-shell.js`), read at call
|
||||
time, so the page modules and the classic partials that still call it
|
||||
(Overview, Logs, Tools) agree on the active tab, and the SSE streams keep
|
||||
pausing with them. Each registration takes its own `LEDVisibility` key, so
|
||||
registrations never replace each other or a classic partial's. Without
|
||||
`LEDVisibility` (a page outside `base.html`), the browser tab's visibility
|
||||
alone decides. Moving the tracker itself into the module (the shell table
|
||||
below) changes only `core/visibility.js`.
|
||||
|
||||
### One facade
|
||||
|
||||
`window.LEDMatrix` is the only global the module code adds:
|
||||
@@ -214,13 +275,13 @@ are the inline script in each partial today.
|
||||
| # | Page | Inline JS | Why it is here |
|
||||
|---|---|---|---|
|
||||
| 1 | Cache (`cache.html`) | 163 lines, now 0 | **Done in stage 1.** One endpoint pair, no globals other pages use. The reference conversion |
|
||||
| 2 | Rotation (`durations.html`) | 29 | Tiny. One htmx form |
|
||||
| 3 | Operation History | 293 | No globals, read-only list |
|
||||
| 4 | Config Editor (`raw_json.html`) | 212 | No globals. CodeMirror is set up and torn down in init/destroy |
|
||||
| 5 | Backup & Restore | 232 | 5 globals used only by its own `onclick`s; these become delegated listeners plus deprecated aliases |
|
||||
| 6 | Schedule | 193 | 2 globals used as `hx-on` response handlers. Moves `hx-on` handlers into page listeners |
|
||||
| 7 | General | 147 | `webLogin` global and the security section. The first page that touches login |
|
||||
| 8 | Display | 231 | First page with `LEDVisibility` timers: those move to a `ctx.visibility` service that stops on destroy |
|
||||
| 2 | Rotation (`durations.html`) | 29 lines, now 0 | **Done in stage 2.** The form stays plain htmx; the page starts the shared rotation-order widget, whose plugin-list request now takes `ctx.signal`. Its `hx-on` and `onsubmit` attributes call shared globals (`showSaveResult`, `fixInvalidNumberInputs`) and move with step 6 |
|
||||
| 3 | Operation History | 293 lines, now 0 | **Done in stage 2.** Read-only list; rows drawn with `textContent`, the search debounce cleared on destroy. The "Showing x to y" counters now also reset when nothing matches |
|
||||
| 4 | Config Editor (`raw_json.html`) | 212 lines, now 0 | **Done in stage 2.** Plain textareas (no CodeMirror on this page). It defined 5 globals after all (`formatJson`, `manualValidateJson`, `validateJSON`, `saveMainConfig`, `saveSecretsConfig`); nothing else used them, and they are deprecated aliases now. The live "Invalid JSON" line no longer puts the parser's message into `innerHTML` |
|
||||
| 5 | Backup & Restore | 232 lines, now 0 | **Done in stage 2.** Its 5 globals (`exportBackup`, `loadBackupList`, `validateRestoreFile`, `clearRestore`, `runRestore`) are deprecated aliases; the buttons are delegated `data-action`s. Uploads go through `ctx.api.request(..., { body: formData })` (`api.js` gained a raw `body` option) |
|
||||
| 6 | Schedule | 193 lines, now 0 | **Done in stage 3.** Its 2 `hx-on` response handlers (`handleScheduleResponse`, `handleDimScheduleResponse`) are one `htmx:afterRequest` listener on the page root, and deprecated aliases. The forms are marked `data-reports-result` so `app.js` does not repeat the server's message. The saved schedules reach the module as JSON in `data-schedule-config` / `data-dim-schedule-config` instead of being templated into the script |
|
||||
| 7 | General | 153 lines, now 0 | **Done in stage 3.** The Security section's three forms and two buttons are delegated `data-action`s (one submit and one click listener); `window.webLogin` is a deprecated alias of an object with its five methods. Login requests go through `ctx.api`, so the login redirect is quiet. The settings form keeps its `hx-on` call to the shared `showSaveResult`, as Rotation's does |
|
||||
| 8 | Display | 292 lines (2 scripts), now 0 | **Done in stage 4.** The first page with a timer: the 5 s multi-display sync poll is `ctx.visibility.every(5000, ...)` (above), so it runs only while the tab is on screen and stops when the partial is swapped out. Its one global, `updateSyncUI` (the Role menu's `onchange`), is a deprecated alias; the Advanced section's `onclick` is a delegated `data-action="toggle-section"` that calls the shared `toggleSection`. The status poll and the scroll-speed hint go through `ctx.api` with `ctx.signal`, as does the Vegas order widget's plugin-list request. The settings form keeps its `hx-on` call to `showSaveResult` and its `onsubmit` call to `fixInvalidNumberInputs`, as Rotation's does |
|
||||
| 9 | Overview | 410 (4 scripts) | First-run surface: Getting Started, update banner, live preview. Five globals |
|
||||
| 10 | WiFi | 364 | `x-data="wifiSetup()"` is defined by its own script. Moves to `Alpine.data()` registered from the module. AP-mode first screen, so it needs the AP-mode test on a real device |
|
||||
| 11 | Fonts | 681 | Large, but self-contained (6 globals) |
|
||||
@@ -237,7 +298,7 @@ the order:
|
||||
| `showNotification` | 4 versions | `core/notify.js` |
|
||||
| The modal helper | `utils/dialog.js` | `core/dialog.js` |
|
||||
| SSE streams | `app-shell.js` | `core/streams.js` |
|
||||
| `LEDVisibility` | `app-shell.js` | `core/visibility.js` |
|
||||
| `LEDVisibility` | `app-shell.js` | `core/visibility.js` (the page-facing `ctx.visibility` is there since step 8; it reads the tracker from `app-shell.js`) |
|
||||
|
||||
Each move leaves the old global as an alias. When the last inline script is
|
||||
gone, the script re-execution in `htmx-config.js` and the "HTMX never
|
||||
@@ -257,10 +318,18 @@ Unit suites need only node. They import the shipped modules directly:
|
||||
|
||||
| Suite | Kind | What it covers |
|
||||
|---|---|---|
|
||||
| `unit/test_page_registry.js` | Unit, minimal DOM shim | The lifecycle: one init per root, destroy on swap, a veto keeps the page, swaps elsewhere leave it alone, the sweep, lazy loading, a destroy while loading, error containment |
|
||||
| `unit/test_page_registry.js` | Unit, minimal DOM shim | The lifecycle: one init per root, destroy on swap, a veto keeps the page, swaps elsewhere leave it alone, the sweep, lazy loading, a destroy while loading, error containment, `mountContext` fields per mount |
|
||||
| `dom/test_visibility_service.js` | DOM: real `LEDVisibility` from `app-shell.js`, real registry, no server | `whileVisible` and `every` start and stop with the active tab and the browser tab's visibility; no interval runs while hidden or after a swap-out; one interval after five swaps; registrations never replace each other or a classic partial's; a destroyed page registers nothing; a throwing `start()` is contained; the no-`LEDVisibility` fallback |
|
||||
| `unit/test_core_modules.js` | Unit | `api.js` (envelope, errors, abort, login redirect, path check) and `facade.js` (facade, aliases) |
|
||||
| `dom/test_cache_page.js` | DOM: real partial, real API shape | No inline script; one request per swap and per Refresh after five swaps; a cancelled request draws nothing; hostile keys stay text; delete, empty, error, network and login states |
|
||||
| `test/web_interface/test_es_modules.py` | pytest | MIME type; `no-cache` without `?v` and immutable with it; `boot.js` loads last; every import resolves inside `core/` and `pages/`; every registered page has its module and exactly one partial root; a converted partial has no `<script>` |
|
||||
| `dom/test_durations_page.js` | DOM: real partial, real widget, real API shape | One plugin-list request per swap; Move down moves one place after five swaps; the swap cancels a request in flight; a late-loading widget is waited for, and a page swapped away while waiting starts nothing; hostile names stay text |
|
||||
| `dom/test_operation_history_page.js` | DOM: real partial, real API shape | One history request per swap and per Refresh; the plugin filter filled once (from `PluginAPI`'s cache when loaded); paging, filters, debounced search, Clear (one DELETE), error/network/login states, cancel on swap; hostile ids, users and errors stay text |
|
||||
| `dom/test_raw_json_page.js` | DOM: real partial, real config | One POST per Save after five swaps, to the right file; Format and Validate act once; invalid JSON never sent and its message stays text; a save survives a swap and is still reported; the old globals' entry points |
|
||||
| `dom/test_schedule_page.js` | DOM: real partial, real widget | Both pickers drawn once per swap from the saved config; after five swaps each form's answer is one notification (message, fallback, refused, non-JSON, `null`), a request from outside the forms none; the brightness label; a late widget waited for, a page swapped away while waiting draws nothing; the old globals' entry points |
|
||||
| `dom/test_display_page.js` | DOM: real partial, real widget, real `LEDVisibility`, real API shape | After five swaps one page, one sync interval, the Vegas order drawn once and each control acting once (brightness, resolution, the two show/hide toggles, the Advanced toggle, one debounced hint request); the sync poll only while on screen and never after a swap-out; sync states and hostile peer names as text, failure and login answers; a late widget waited for; `updateSyncUI`'s entry point |
|
||||
| `dom/test_general_page.js` | DOM: real partial, real widget, real API shape | The timezone picker drawn once per swap with the saved zone; the settings form left to htmx; after five swaps each Security action makes one request (create, copy, revoke and its cancel, password and its mismatch); hostile token names stay text; refused, network and login answers; a create made before a swap is still reported and draws nothing; `webLogin`'s entry points |
|
||||
| `dom/test_backup_restore_page.js` | DOM: real partial, real API shape | One request per Refresh, Delete, Export (busy button ignores a second click), Inspect and Restore after five swaps; the upload's fields and the six restore options; reads cancelled by a swap, writes not; hostile file and host names stay text; the old globals' entry points |
|
||||
| `test/web_interface/test_es_modules.py` | pytest | MIME type; `no-cache` without `?v` and immutable with it; `boot.js` loads last; every import resolves inside `core/` and `pages/`; the converted pages are exactly the registered ones, each with its module, `init`, and one root in the rendered partial; a converted partial has no `<script>` and no `onclick`; every moved global is aliased in `boot.js` and exported by its module, and no template defines it any more |
|
||||
| `test/test_field_model_parity.py` | pytest | The model against the macro for every available schema |
|
||||
|
||||
What each future step adds:
|
||||
|
||||
+160
-47
@@ -47,38 +47,51 @@ if echo "${DEVICE_MODEL:-}" | grep -qi "Raspberry Pi 5"; then
|
||||
echo "Raspberry Pi 5 detected — will verify RP1 library support."
|
||||
fi
|
||||
|
||||
# Check OS version - must be Raspberry Pi OS Lite (Trixie)
|
||||
# Check OS version - must be Raspberry Pi OS Lite, Bookworm or Trixie.
|
||||
# The rules live in scripts/install/lib_os.sh, shared with
|
||||
# scripts/check_system_compatibility.sh.
|
||||
echo ""
|
||||
echo "Checking operating system requirements..."
|
||||
echo "----------------------------------------"
|
||||
OS_CHECK_FAILED=0
|
||||
OS_RELEASE=""
|
||||
|
||||
if [ -f /etc/os-release ]; then
|
||||
. /etc/os-release
|
||||
echo "Detected OS: $PRETTY_NAME"
|
||||
echo "Version ID: ${VERSION_ID:-unknown}"
|
||||
|
||||
# Check if it's Raspberry Pi OS or Debian
|
||||
if [[ "$ID" != "raspbian" ]] && [[ "$ID" != "debian" ]]; then
|
||||
echo "✗ ERROR: This script requires Raspberry Pi OS (raspbian/debian)"
|
||||
echo " Detected OS ID: $ID"
|
||||
OS_CHECK_FAILED=1
|
||||
fi
|
||||
|
||||
# Check if it's Debian 13 (Trixie)
|
||||
if [ "${VERSION_ID:-0}" != "13" ]; then
|
||||
echo "✗ ERROR: This script requires Raspberry Pi OS Lite (Trixie) - Debian 13"
|
||||
echo " Detected version: ${VERSION_ID:-unknown}"
|
||||
echo " Please upgrade to Raspberry Pi OS Lite (Trixie) before continuing"
|
||||
OS_CHECK_FAILED=1
|
||||
OS_LIB="$(cd "$(dirname "$0")" && pwd)/scripts/install/lib_os.sh"
|
||||
if [ ! -f "$OS_LIB" ]; then
|
||||
echo "✗ ERROR: $OS_LIB is missing, so the operating system cannot be checked."
|
||||
echo " Your LEDMatrix download is incomplete. Download it again and re-run this script:"
|
||||
echo " git clone https://github.com/ChuckBuilds/LEDMatrix.git"
|
||||
exit 1
|
||||
fi
|
||||
# shellcheck source=scripts/install/lib_os.sh
|
||||
. "$OS_LIB"
|
||||
|
||||
if [ -r "$LM_OS_RELEASE_FILE" ]; then
|
||||
echo "Detected OS: $(lm_os_field PRETTY_NAME)"
|
||||
OS_VERSION_ID=$(lm_os_field VERSION_ID)
|
||||
echo "Version ID: ${OS_VERSION_ID:-unknown}"
|
||||
|
||||
if OS_RELEASE=$(lm_os_release); then
|
||||
echo "✓ $(lm_release_label "$OS_RELEASE") detected"
|
||||
else
|
||||
echo "✓ Debian 13 (Trixie) detected"
|
||||
OS_ID=$(lm_os_field ID)
|
||||
if [[ "$OS_ID" != "raspbian" ]] && [[ "$OS_ID" != "debian" ]]; then
|
||||
echo "✗ ERROR: This script requires Raspberry Pi OS (raspbian/debian)"
|
||||
echo " Detected OS ID: ${OS_ID:-unknown}"
|
||||
else
|
||||
echo "✗ ERROR: This version of Raspberry Pi OS is not supported"
|
||||
echo " Detected version: ${OS_VERSION_ID:-unknown}"
|
||||
echo " Supported: Trixie (Debian 13) and Bookworm (Debian 12)"
|
||||
fi
|
||||
OS_CHECK_FAILED=1
|
||||
fi
|
||||
|
||||
|
||||
# Check if it's the Lite version (no desktop environment)
|
||||
# Check for desktop packages or desktop services
|
||||
DESKTOP_DETECTED=0
|
||||
if dpkg -l | grep -qE "^ii.*raspberrypi-ui-mods|^ii.*lxde|^ii.*xfce|^ii.*gnome|^ii.*kde"; then
|
||||
# grep without -q: -q exits at the first match, dpkg then dies of SIGPIPE,
|
||||
# and pipefail turns a found desktop into "not found".
|
||||
if dpkg -l | grep -E "^ii.*raspberrypi-ui-mods|^ii.*lxde|^ii.*xfce|^ii.*gnome|^ii.*kde" >/dev/null; then
|
||||
DESKTOP_DETECTED=1
|
||||
fi
|
||||
if systemctl list-units --type=service --state=running 2>/dev/null | grep -qE "lightdm|gdm3|sddm|lxdm"; then
|
||||
@@ -96,23 +109,52 @@ if [ -f /etc/os-release ]; then
|
||||
echo "✓ Lite version confirmed (no desktop environment)"
|
||||
fi
|
||||
else
|
||||
echo "✗ ERROR: Could not detect OS version (/etc/os-release not found)"
|
||||
echo "✗ ERROR: Could not detect OS version ($LM_OS_RELEASE_FILE not found)"
|
||||
OS_CHECK_FAILED=1
|
||||
fi
|
||||
|
||||
# Python: whatever python3 the release ships (3.11 on Bookworm, 3.13 on
|
||||
# Trixie). Checked only when python3 is already there -- Step 1 installs it
|
||||
# otherwise, and on a supported release that brings the release's own version.
|
||||
if [ "$OS_CHECK_FAILED" -eq 0 ]; then
|
||||
if PYTHON3_VERSION=$(lm_python_version); then
|
||||
case "$(lm_python_check "$PYTHON3_VERSION")" in
|
||||
ok)
|
||||
echo "✓ Python $PYTHON3_VERSION detected"
|
||||
;;
|
||||
too-old)
|
||||
echo "✗ ERROR: python3 is Python $PYTHON3_VERSION; LEDMatrix needs Python 3.$LM_PYTHON_MIN_MINOR or newer"
|
||||
echo " $(lm_release_label "$OS_RELEASE") ships Python $(lm_release_python "$OS_RELEASE"). Something on this"
|
||||
echo " system has changed which Python 'python3' runs; point it back at the system Python."
|
||||
OS_CHECK_FAILED=1
|
||||
;;
|
||||
*)
|
||||
echo "⚠ python3 is Python $PYTHON3_VERSION, which LEDMatrix has not been tested with"
|
||||
echo " (tested: 3.$LM_PYTHON_MIN_MINOR to 3.$LM_PYTHON_MAX_MINOR). Continuing anyway."
|
||||
;;
|
||||
esac
|
||||
else
|
||||
echo "python3 not found yet; Step 1 installs it."
|
||||
fi
|
||||
fi
|
||||
|
||||
if [ "$OS_CHECK_FAILED" -eq 1 ]; then
|
||||
echo ""
|
||||
echo "Installation cannot continue. Please install Raspberry Pi OS Lite (Trixie) and try again."
|
||||
echo ""
|
||||
echo "To install Raspberry Pi OS Lite (Trixie):"
|
||||
echo " 1. Download from: https://www.raspberrypi.com/software/operating-systems/"
|
||||
echo " 2. Select 'Raspberry Pi OS Lite (64-bit)' with Debian 13 (Trixie)"
|
||||
echo " 3. Flash to SD card using Raspberry Pi Imager"
|
||||
echo " 4. Boot and run this script again"
|
||||
echo "Installation cannot continue."
|
||||
lm_print_supported_os_help
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "✓ OS requirements met"
|
||||
|
||||
# WiFi setup (the web page's WiFi tab and the LEDMatrix-Setup hotspot) needs
|
||||
# NetworkManager. Both releases use it by default; say so plainly if this Pi
|
||||
# does not, but carry on -- the display itself does not depend on it.
|
||||
case "$(lm_network_stack)" in
|
||||
networkmanager) echo "✓ NetworkManager is managing the network" ;;
|
||||
dhcpcd) lm_print_dhcpcd_advice ;;
|
||||
*) echo "⚠ Could not tell which service manages the network; WiFi setup from the web page needs NetworkManager" ;;
|
||||
esac
|
||||
echo ""
|
||||
|
||||
# The user who ran the installer: SUDO_USER once we are running under sudo
|
||||
@@ -190,6 +232,51 @@ _sync_rgb_submodule() {
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# LEDMatrix's own changes to the library live in patches/rpi-rgb-led-matrix/ and
|
||||
# are applied only for the build: _apply_rgb_patches before it, _revert_rgb_patches
|
||||
# after it, success or not. The checkout is left exactly as it was, so `git pull`
|
||||
# and _sync_rgb_submodule never meet local modifications in the submodule.
|
||||
# A patch that no longer applies (a submodule bump, a hand-edited checkout) is
|
||||
# reported and skipped -- the unpatched library still builds and works, so it is
|
||||
# never fatal. One that is already applied is left alone and not reverted.
|
||||
_RGB_APPLIED_PATCHES=()
|
||||
|
||||
_apply_rgb_patches() {
|
||||
local sub="$PROJECT_ROOT_DIR/rpi-rgb-led-matrix-master"
|
||||
local dir="$PROJECT_ROOT_DIR/patches/rpi-rgb-led-matrix" patch name
|
||||
_RGB_APPLIED_PATCHES=()
|
||||
[ -d "$dir" ] || return 0
|
||||
for patch in "$dir"/*.patch; do
|
||||
[ -f "$patch" ] || continue
|
||||
name=$(basename "$patch")
|
||||
if _git_as_repo_owner -C "$sub" apply --check "$patch" >/dev/null 2>&1; then
|
||||
if _git_as_repo_owner -C "$sub" apply "$patch"; then
|
||||
_RGB_APPLIED_PATCHES+=("$patch")
|
||||
echo "Applied library patch $name"
|
||||
else
|
||||
echo "⚠ Could not apply library patch $name; building without it"
|
||||
fi
|
||||
elif _git_as_repo_owner -C "$sub" apply --reverse --check "$patch" >/dev/null 2>&1; then
|
||||
echo "Library patch $name is already applied"
|
||||
else
|
||||
echo "⚠ Library patch $name does not apply to this checkout; building without it"
|
||||
fi
|
||||
done
|
||||
return 0
|
||||
}
|
||||
|
||||
_revert_rgb_patches() {
|
||||
local sub="$PROJECT_ROOT_DIR/rpi-rgb-led-matrix-master" i
|
||||
# Last applied first, in case two patches touch the same file.
|
||||
for ((i = ${#_RGB_APPLIED_PATCHES[@]} - 1; i >= 0; i--)); do
|
||||
if ! _git_as_repo_owner -C "$sub" apply --reverse "${_RGB_APPLIED_PATCHES[i]}"; then
|
||||
echo "⚠ Could not revert $(basename "${_RGB_APPLIED_PATCHES[i]}"); restore the checkout with: git -C $sub checkout -- ."
|
||||
fi
|
||||
done
|
||||
_RGB_APPLIED_PATCHES=()
|
||||
return 0
|
||||
}
|
||||
# --- end rpi-rgb-led-matrix checkout helpers ---------------------------------
|
||||
|
||||
# Determine the Project Root Directory (where this script is located)
|
||||
@@ -223,6 +310,8 @@ SKIP_SWAP=${LEDMATRIX_SKIP_SWAP:-0}
|
||||
BUILD_JOBS_OVERRIDE=${LEDMATRIX_BUILD_JOBS:-}
|
||||
# Weekly automatic updates: 1 on, 0 off, empty = ask (interactive) or leave as is.
|
||||
AUTO_UPDATE=${LEDMATRIX_AUTO_UPDATE:-}
|
||||
# Update channel written to config.json: stable, beta, or empty = leave as is.
|
||||
UPDATE_CHANNEL=$(printf '%s' "${LEDMATRIX_CHANNEL:-}" | tr '[:upper:]' '[:lower:]')
|
||||
|
||||
usage() {
|
||||
cat <<USAGE
|
||||
@@ -240,12 +329,18 @@ Options:
|
||||
--enable-auto-update Turn on weekly automatic updates (with health
|
||||
check and automatic rollback)
|
||||
--no-auto-update Leave weekly automatic updates off
|
||||
--beta Follow main, the newest code (the beta update
|
||||
channel). Without it, updates follow releases
|
||||
(stable). It sets the channel; it does not move
|
||||
this checkout -- the one-shot installer picks the
|
||||
version, and so does the next update.
|
||||
-h, --help Show this help message and exit
|
||||
|
||||
Environment variables (same effect as flags):
|
||||
LEDMATRIX_ASSUME_YES=1, RPI_RGB_FORCE_REBUILD=1, LEDMATRIX_SKIP_SOUND=1,
|
||||
LEDMATRIX_SKIP_PERF=1, LEDMATRIX_SKIP_REBOOT_PROMPT=1,
|
||||
LEDMATRIX_SKIP_SWAP=1, LEDMATRIX_BUILD_JOBS=N, LEDMATRIX_AUTO_UPDATE=1|0
|
||||
LEDMATRIX_SKIP_SWAP=1, LEDMATRIX_BUILD_JOBS=N, LEDMATRIX_AUTO_UPDATE=1|0,
|
||||
LEDMATRIX_CHANNEL=stable|beta
|
||||
|
||||
Low-memory devices:
|
||||
On a Pi with under 2GB of RAM the C++ build is limited to fewer parallel
|
||||
@@ -265,6 +360,7 @@ while [ $# -gt 0 ]; do
|
||||
--skip-swap) SKIP_SWAP=1 ;;
|
||||
--enable-auto-update) AUTO_UPDATE=1 ;;
|
||||
--no-auto-update) AUTO_UPDATE=0 ;;
|
||||
--beta) UPDATE_CHANNEL=beta ;;
|
||||
--build-jobs)
|
||||
shift
|
||||
if [ $# -eq 0 ]; then echo "--build-jobs requires a number"; usage; exit 1; fi
|
||||
@@ -291,10 +387,12 @@ else
|
||||
lm_remove_build_swap() { return 0; }
|
||||
fi
|
||||
|
||||
# Remove the temporary build swapfile no matter how the script ends. Step 6
|
||||
# tears it down itself; this is the backstop for the error path, since
|
||||
# on_error ends in `exit` and EXIT traps still run.
|
||||
trap 'lm_remove_build_swap' EXIT
|
||||
# Remove the temporary build swapfile, and take any library patches back out
|
||||
# of the submodule, no matter how the script ends. Step 6 does both itself;
|
||||
# this is the backstop for the error path (on_error ends in `exit` and EXIT
|
||||
# traps still run) and for an interrupted build. _revert_rgb_patches only
|
||||
# touches patches it applied, so running it twice is harmless.
|
||||
trap 'lm_remove_build_swap; _revert_rgb_patches' EXIT
|
||||
|
||||
# Helpers
|
||||
retry() {
|
||||
@@ -873,15 +971,25 @@ if [ -z "$AUTO_UPDATE" ] && [ "$ASSUME_YES" != "1" ] && [ -t 0 ]; then
|
||||
echo
|
||||
if [[ $REPLY =~ ^[Yy]$ ]]; then AUTO_UPDATE=1; else AUTO_UPDATE=0; fi
|
||||
fi
|
||||
if [ "$AUTO_UPDATE" = "1" ] || [ "$AUTO_UPDATE" = "0" ]; then
|
||||
if python3 - "$PROJECT_ROOT_DIR/config/config.json" "$AUTO_UPDATE" <<'PY'
|
||||
case "$UPDATE_CHANNEL" in
|
||||
stable|beta|"") ;;
|
||||
*) echo "⚠ LEDMATRIX_CHANNEL=$UPDATE_CHANNEL is not stable or beta; leaving the update channel as it is"
|
||||
UPDATE_CHANNEL="" ;;
|
||||
esac
|
||||
# The update channel, likewise only when asked for (--beta / LEDMATRIX_CHANNEL).
|
||||
if [ "$AUTO_UPDATE" = "1" ] || [ "$AUTO_UPDATE" = "0" ] || [ -n "$UPDATE_CHANNEL" ]; then
|
||||
if python3 - "$PROJECT_ROOT_DIR/config/config.json" "$AUTO_UPDATE" "$UPDATE_CHANNEL" <<'PY'
|
||||
import json, os, sys, tempfile
|
||||
path, enabled = sys.argv[1], sys.argv[2] == "1"
|
||||
path, enabled = sys.argv[1], sys.argv[2]
|
||||
channel = sys.argv[3] if len(sys.argv) > 3 else ""
|
||||
with open(path, encoding="utf-8") as f:
|
||||
config = json.load(f)
|
||||
if not isinstance(config.get("auto_update"), dict):
|
||||
config["auto_update"] = {}
|
||||
config["auto_update"]["enabled"] = enabled
|
||||
if enabled in ("0", "1"):
|
||||
config["auto_update"]["enabled"] = enabled == "1"
|
||||
if channel:
|
||||
config["auto_update"]["channel"] = channel
|
||||
# Written beside the original and swapped in whole: the display service's
|
||||
# config watcher may be running and must never read a half-written file.
|
||||
original = os.stat(path)
|
||||
@@ -902,9 +1010,11 @@ except BaseException:
|
||||
raise
|
||||
PY
|
||||
then
|
||||
if [ "$AUTO_UPDATE" = "1" ]; then echo "✓ Weekly automatic updates enabled"; else echo "✓ Weekly automatic updates off"; fi
|
||||
if [ "$AUTO_UPDATE" = "1" ]; then echo "✓ Weekly automatic updates enabled"
|
||||
elif [ "$AUTO_UPDATE" = "0" ]; then echo "✓ Weekly automatic updates off"; fi
|
||||
if [ -n "$UPDATE_CHANNEL" ]; then echo "✓ Update channel: $UPDATE_CHANNEL"; fi
|
||||
else
|
||||
echo "⚠ Could not set auto_update in config/config.json; turn it on from the General tab instead"
|
||||
echo "⚠ Could not set auto_update in config/config.json; set it from the General tab instead"
|
||||
fi
|
||||
fi
|
||||
|
||||
@@ -1226,9 +1336,11 @@ else
|
||||
fi
|
||||
BUILD_OUTPUT=$(mktemp)
|
||||
BUILD_SUCCESS=false
|
||||
_apply_rgb_patches
|
||||
if run_rgbmatrix_build "$BUILD_JOBS" "$BUILD_OUTPUT"; then
|
||||
BUILD_SUCCESS=true
|
||||
fi
|
||||
_revert_rgb_patches
|
||||
cat "$BUILD_OUTPUT" >> "$LOG_FILE"
|
||||
if [ "$BUILD_SUCCESS" != true ]; then
|
||||
print_rgbmatrix_build_failure "$BUILD_OUTPUT"
|
||||
@@ -1349,15 +1461,16 @@ if ! command -v setcap >/dev/null 2>&1; then
|
||||
echo "⚠ setcap not found, skipping capability configuration"
|
||||
echo " Install libcap2-bin if you need hardware timing capabilities"
|
||||
else
|
||||
# Find the Python binary and resolve symlinks to get the real binary
|
||||
# The binary the services run (ExecStart=/usr/bin/python3), symlinks
|
||||
# resolved: python3.11 on Bookworm, python3.13 on Trixie. This used to
|
||||
# prefer /usr/bin/python3.13 whenever it existed, which would set the
|
||||
# capability on an interpreter the services never run if python3 pointed
|
||||
# elsewhere.
|
||||
PYTHON_BIN=""
|
||||
PYTHON_VER=""
|
||||
if [ -f "/usr/bin/python3.13" ]; then
|
||||
PYTHON_BIN=$(readlink -f /usr/bin/python3.13)
|
||||
PYTHON_VER="3.13"
|
||||
elif [ -f "/usr/bin/python3" ]; then
|
||||
if [ -f "/usr/bin/python3" ]; then
|
||||
PYTHON_BIN=$(readlink -f /usr/bin/python3)
|
||||
PYTHON_VER=$(python3 --version 2>&1 | grep -oP '(?<=Python )\d+\.\d+' || echo "unknown")
|
||||
PYTHON_VER=$(lm_python_version /usr/bin/python3) || PYTHON_VER="unknown"
|
||||
fi
|
||||
|
||||
if [ -n "$PYTHON_BIN" ] && [ -f "$PYTHON_BIN" ]; then
|
||||
|
||||
@@ -22,6 +22,7 @@ src/common/api_helper.py
|
||||
src/common/bdf_font.py
|
||||
src/common/espn_dates.py
|
||||
src/common/favorite_team_check.py
|
||||
src/common/fetch_service.py
|
||||
src/common/font_layout.py
|
||||
src/common/frame_timing.py
|
||||
src/common/json_body.py
|
||||
@@ -37,6 +38,7 @@ src/common/sports_celebration.py
|
||||
src/common/sports_display_rules.py
|
||||
src/common/sports_fetch.py
|
||||
src/common/sports_font_path.py
|
||||
src/common/sports_game_over.py
|
||||
src/common/sports_live_scroll.py
|
||||
src/common/sports_plugin_host.py
|
||||
src/common/sports_scroll.py
|
||||
@@ -46,6 +48,7 @@ src/config_service.py
|
||||
src/core_config_keys.py
|
||||
src/deprecation.py
|
||||
src/device_location.py
|
||||
src/display_arbiter.py
|
||||
src/display_geometry.py
|
||||
src/dynamic_team_resolver.py
|
||||
src/exceptions.py
|
||||
@@ -56,6 +59,7 @@ src/ipc/contract.py
|
||||
src/ipc/server.py
|
||||
src/logging_config.py
|
||||
src/logo_downloader.py
|
||||
src/malloc_tuning.py
|
||||
src/matrix_support.py
|
||||
src/pi5_matrix_support.py
|
||||
src/plugin_system/__init__.py
|
||||
@@ -86,6 +90,7 @@ src/plugin_system/testing/vegas.py
|
||||
src/plugin_system/vegas_elements.py
|
||||
src/redaction.py
|
||||
src/scan_order.py
|
||||
src/screen_runner.py
|
||||
src/startup_validator.py
|
||||
src/vegas_mode/__init__.py
|
||||
src/vegas_mode/config.py
|
||||
|
||||
@@ -6,8 +6,9 @@
|
||||
files = src
|
||||
exclude = (^|/)(test|__pycache__)/
|
||||
|
||||
# Python version
|
||||
python_version = 3.10
|
||||
# Python version: the oldest the installer supports (Raspberry Pi OS
|
||||
# Bookworm ships 3.11; Trixie ships 3.13).
|
||||
python_version = 3.11
|
||||
|
||||
# Platform (Linux/Raspberry Pi)
|
||||
platform = linux
|
||||
@@ -103,8 +104,8 @@ ignore_missing_imports = True
|
||||
|
||||
|
||||
# numpy's own stubs (numpy>=2.3) use Python 3.12 `type` statements, which mypy
|
||||
# refuses to parse under python_version = 3.10 -- and 3.10 is the floor this
|
||||
# code has to run on, so it stays. Treat numpy as Any instead: skip it, and
|
||||
# refuses to parse under python_version = 3.11 -- and 3.11 (Bookworm) is the
|
||||
# floor this code has to run on, so it stays. Treat numpy as Any instead: skip it, and
|
||||
# follow_imports_for_stubs makes the skip apply to its .pyi files too.
|
||||
[mypy-numpy.*]
|
||||
follow_imports = skip
|
||||
|
||||
@@ -0,0 +1,174 @@
|
||||
Faster SetImage for rpi-rgb-led-matrix (applied by first_time_install.sh at build
|
||||
time; the submodule itself stays at its pinned commit).
|
||||
|
||||
Copying a frame into the panel buffer was the biggest CPU cost LEDMatrix owns on
|
||||
large panels: the binding's SetPixelsPillow walked the image column by column
|
||||
and called SetPixel per pixel, and each SetPixel read-modify-writes one word per
|
||||
PWM bit plane, 2KB apart, so consecutive pixels were a whole double-row apart and
|
||||
almost every write missed the cache. This patch:
|
||||
|
||||
* FrameCanvas gets its own SetPixelsPillow: row by row, one bulk SetPixels call
|
||||
per row;
|
||||
* Framebuffer::SetPixels clips once, looks colours up once per pixel, walks each
|
||||
row's designators in order and writes the bit planes branch-free;
|
||||
* the base Canvas.SetPixelsPillow loop (RGBMatrix.SetImage) is row-major.
|
||||
|
||||
The bit-plane buffer is byte-identical to the old code's (882 memcmp checks over
|
||||
noise/gradient/solid/low-value/sparse images, clipped offsets, pwm 7/8/11,
|
||||
brightness 1/50/90/100, inverse colours, luminance correction off and a pixel
|
||||
mapper). Measured on a Pi 4 at 512x64: 6.0-6.3 ms -> 1.8 ms per frame through
|
||||
the Python binding; on hdpi (Pi 4, 4x128x64) frame copy 6.57 -> 2.21 ms and the
|
||||
display process 139% -> 103% of a core.
|
||||
|
||||
LEDMatrix always draws into the canvas that is not on screen and swaps it in
|
||||
(DisplayManager.update_display), so the write order cannot show as tearing.
|
||||
|
||||
Against hzeller/rpi-rgb-led-matrix 1ee4f76.
|
||||
|
||||
diff --git a/bindings/python/rgbmatrix/core.pyx b/bindings/python/rgbmatrix/core.pyx
|
||||
index 230d87f..babc3bb 100644
|
||||
--- a/bindings/python/rgbmatrix/core.pyx
|
||||
+++ b/bindings/python/rgbmatrix/core.pyx
|
||||
@@ -2,6 +2,7 @@
|
||||
|
||||
from libcpp cimport bool
|
||||
from libc.stdint cimport uint8_t, uint32_t, uintptr_t
|
||||
+from libc.stdlib cimport malloc, free
|
||||
import cython
|
||||
|
||||
cdef extern from "Python.h":
|
||||
@@ -59,8 +60,9 @@ cdef class Canvas:
|
||||
|
||||
buffer = get_pillow_buffer(image_capsule)
|
||||
|
||||
- for col in range(max(0, -xstart), min(width, frame_width - xstart)):
|
||||
- for row in range(max(0, -ystart), min(height, frame_height - ystart)):
|
||||
+ # Row-major: walks both the image and the bitplane buffer sequentially.
|
||||
+ for row in range(max(0, -ystart), min(height, frame_height - ystart)):
|
||||
+ for col in range(max(0, -xstart), min(width, frame_width - xstart)):
|
||||
pixel = buffer[row][col]
|
||||
r = (pixel ) & 0xFF
|
||||
g = (pixel >> 8) & 0xFF
|
||||
@@ -86,6 +88,41 @@ cdef class FrameCanvas(Canvas):
|
||||
def SetPixel(self, int x, int y, uint8_t red, uint8_t green, uint8_t blue):
|
||||
(<cppinc.FrameCanvas*>self._getCanvas()).SetPixel(x, y, red, green, blue)
|
||||
|
||||
+ @cython.boundscheck(False)
|
||||
+ @cython.wraparound(False)
|
||||
+ def SetPixelsPillow(self, int xstart, int ystart, int width, int height, object image_capsule):
|
||||
+ # Same result as Canvas.SetPixelsPillow(), but hands each image row
|
||||
+ # to the C++ bulk FrameCanvas::SetPixels() instead of calling the
|
||||
+ # virtual SetPixel() once per pixel.
|
||||
+ cdef cppinc.FrameCanvas* my_canvas = <cppinc.FrameCanvas*>self._getCanvas()
|
||||
+ cdef int col_start = max(0, -xstart)
|
||||
+ cdef int col_end = min(width, my_canvas.width() - xstart)
|
||||
+ cdef int row_start = max(0, -ystart)
|
||||
+ cdef int row_end = min(height, my_canvas.height() - ystart)
|
||||
+ cdef int row, col, pixel
|
||||
+ cdef int *src
|
||||
+ cdef cppinc.Color *line
|
||||
+ cdef int **buffer
|
||||
+
|
||||
+ if col_end <= col_start or row_end <= row_start:
|
||||
+ return
|
||||
+ buffer = get_pillow_buffer(image_capsule)
|
||||
+ line = <cppinc.Color*>malloc((col_end - col_start) * sizeof(cppinc.Color))
|
||||
+ if line == NULL:
|
||||
+ raise MemoryError()
|
||||
+ try:
|
||||
+ for row in range(row_start, row_end):
|
||||
+ src = buffer[row]
|
||||
+ for col in range(col_start, col_end):
|
||||
+ pixel = src[col]
|
||||
+ line[col - col_start].r = pixel & 0xFF
|
||||
+ line[col - col_start].g = (pixel >> 8) & 0xFF
|
||||
+ line[col - col_start].b = (pixel >> 16) & 0xFF
|
||||
+ my_canvas.SetPixels(xstart + col_start, ystart + row,
|
||||
+ col_end - col_start, 1, line)
|
||||
+ finally:
|
||||
+ free(line)
|
||||
+
|
||||
|
||||
property width:
|
||||
def __get__(self): return (<cppinc.FrameCanvas*>self._getCanvas()).width()
|
||||
diff --git a/bindings/python/rgbmatrix/cppinc.pxd b/bindings/python/rgbmatrix/cppinc.pxd
|
||||
index 8bec241..314332d 100644
|
||||
--- a/bindings/python/rgbmatrix/cppinc.pxd
|
||||
+++ b/bindings/python/rgbmatrix/cppinc.pxd
|
||||
@@ -25,6 +25,7 @@ cdef extern from "led-matrix.h" namespace "rgb_matrix":
|
||||
FrameCanvas *SwapOnVSync(FrameCanvas*, uint8_t)
|
||||
|
||||
cdef cppclass FrameCanvas(Canvas):
|
||||
+ void SetPixels(int, int, int, int, Color*) nogil
|
||||
bool SetPWMBits(uint8_t)
|
||||
uint8_t pwmbits()
|
||||
void SetBrightness(uint8_t)
|
||||
diff --git a/lib/framebuffer.cc b/lib/framebuffer.cc
|
||||
index 36d138b..aee62ca 100644
|
||||
--- a/lib/framebuffer.cc
|
||||
+++ b/lib/framebuffer.cc
|
||||
@@ -807,11 +807,60 @@ void Framebuffer::SetPixel(int x, int y, uint8_t r, uint8_t g, uint8_t b) {
|
||||
}
|
||||
}
|
||||
|
||||
+// Bulk version of SetPixel(); produces exactly the same bitplane content.
|
||||
+// Faster because it hoists the per-pixel work out of the loop: the color
|
||||
+// mapping becomes one 256-entry table built per call (each channel maps
|
||||
+// independently through the same function), the pixel designators of a row
|
||||
+// are contiguous in the PixelDesignatorMap, and the bit-plane loop is
|
||||
+// branchless (the color bits are effectively random, so the branches in
|
||||
+// SetPixel() mispredict a lot).
|
||||
void Framebuffer::SetPixels(int x, int y, int width, int height, Color *colors) {
|
||||
- for (int iy = 0; iy < height; ++iy) {
|
||||
- for (int ix = 0; ix < width; ++ix) {
|
||||
- SetPixel(x + ix, y + iy, colors->r, colors->g, colors->b);
|
||||
- ++colors;
|
||||
+ PixelDesignatorMap *const mapper = *shared_mapper_;
|
||||
+ const int ix_start = std::max(0, -x);
|
||||
+ const int ix_end = std::min(width, mapper->width() - x);
|
||||
+ const int iy_start = std::max(0, -y);
|
||||
+ const int iy_end = std::min(height, mapper->height() - y);
|
||||
+ if (ix_start >= ix_end || iy_start >= iy_end) return;
|
||||
+
|
||||
+ // Common case (luminance correction, no inversion): use the precomputed
|
||||
+ // table directly; otherwise build one. Cheap enough to do per call, which
|
||||
+ // matters for callers that send one row at a time.
|
||||
+ uint16_t local_map[256];
|
||||
+ const uint16_t *color_map;
|
||||
+ if (do_luminance_correct_ && !inverse_color_) {
|
||||
+ color_map = ColorLookupTable::GetLookup(brightness_).color;
|
||||
+ } else {
|
||||
+ for (int c = 0; c < 256; ++c) {
|
||||
+ uint16_t unused1, unused2;
|
||||
+ MapColors(c, 0, 0, &local_map[c], &unused1, &unused2);
|
||||
+ }
|
||||
+ color_map = local_map;
|
||||
+ }
|
||||
+
|
||||
+ const int min_bit_plane = kBitPlanes - pwm_bits_;
|
||||
+ gpio_bits_t *const plane_start = bitplane_buffer_ + columns_ * min_bit_plane;
|
||||
+ for (int iy = iy_start; iy < iy_end; ++iy) {
|
||||
+ const Color *c = colors + iy * width + ix_start;
|
||||
+ const PixelDesignator *designator = mapper->get(x + ix_start, y + iy);
|
||||
+ for (int ix = ix_start; ix < ix_end; ++ix, ++c, ++designator) {
|
||||
+ const long pos = designator->gpio_word;
|
||||
+ if (pos < 0) continue; // non-used pixel marker.
|
||||
+ const uint16_t red = color_map[c->r];
|
||||
+ const uint16_t green = color_map[c->g];
|
||||
+ const uint16_t blue = color_map[c->b];
|
||||
+ const gpio_bits_t r_bits = designator->r_bit;
|
||||
+ const gpio_bits_t g_bits = designator->g_bit;
|
||||
+ const gpio_bits_t b_bits = designator->b_bit;
|
||||
+ const gpio_bits_t designator_mask = designator->mask;
|
||||
+ gpio_bits_t *bits = plane_start + pos;
|
||||
+ for (int plane = min_bit_plane; plane < kBitPlanes; ++plane) {
|
||||
+ const gpio_bits_t color_bits =
|
||||
+ (r_bits & -(gpio_bits_t)((red >> plane) & 1))
|
||||
+ | (g_bits & -(gpio_bits_t)((green >> plane) & 1))
|
||||
+ | (b_bits & -(gpio_bits_t)((blue >> plane) & 1));
|
||||
+ *bits = (*bits & designator_mask) | color_bits;
|
||||
+ bits += columns_;
|
||||
+ }
|
||||
}
|
||||
}
|
||||
}
|
||||
+1
-1
@@ -1,5 +1,5 @@
|
||||
# LEDMatrix Core Dependencies
|
||||
# Compatible with Python 3.10, 3.11, 3.12, and 3.13
|
||||
# Compatible with Python 3.11, 3.12 and 3.13; CI tests 3.11 and 3.13
|
||||
# Tested on Raspbian OS 12 (Bookworm) and 13 (Trixie)
|
||||
|
||||
# Image processing
|
||||
|
||||
@@ -14,6 +14,12 @@ project_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
if project_dir not in sys.path:
|
||||
sys.path.insert(0, project_dir)
|
||||
|
||||
# Cap glibc's malloc arenas before any thread exists (arenas already made
|
||||
# stay): the in-process twin of the unit's MALLOC_ARENA_MAX=2, for units
|
||||
# installed before that line. A no-op off glibc. See src/malloc_tuning.py.
|
||||
from src import malloc_tuning
|
||||
malloc_tuning.cap_arenas()
|
||||
|
||||
# Under systemd the watchdog clock is already running, and start-up (plugin
|
||||
# loads, initial updates) takes far longer than the render loop's limit. Widen
|
||||
# it before anything slow is imported; the render loop narrows it again once
|
||||
|
||||
@@ -53,26 +53,35 @@ else
|
||||
fi
|
||||
echo ""
|
||||
|
||||
# Check OS version
|
||||
# Check OS version. The supported releases come from the same library the
|
||||
# installer uses, so the two cannot disagree.
|
||||
echo "2. Checking Operating System Version..."
|
||||
echo "---------------------------------------"
|
||||
if [ -f /etc/os-release ]; then
|
||||
. /etc/os-release
|
||||
echo "OS: $PRETTY_NAME"
|
||||
echo "Version ID: ${VERSION_ID:-unknown}"
|
||||
|
||||
# first_time_install.sh refuses anything but Raspberry Pi OS / Debian 13
|
||||
# (Trixie), so anything else is an error here too, not a warning.
|
||||
if [[ "$ID" == "raspbian" ]] || [[ "$ID" == "debian" ]]; then
|
||||
if [ "${VERSION_ID:-0}" = "13" ]; then
|
||||
print_success "Detected Debian 13 Trixie - supported"
|
||||
elif [ "${VERSION_ID:-0}" = "12" ]; then
|
||||
print_error "Debian 12 Bookworm is not supported - the installer requires Raspberry Pi OS Lite (Trixie), Debian 13"
|
||||
else
|
||||
print_error "Debian/Raspbian ${VERSION_ID:-unknown} is not supported - the installer requires Raspberry Pi OS Lite (Trixie), Debian 13"
|
||||
fi
|
||||
OS_LIB="$(cd "$(dirname "$0")" && pwd)/install/lib_os.sh"
|
||||
OS_LIB_LOADED=0
|
||||
OS_RELEASE=""
|
||||
if [ -f "$OS_LIB" ]; then
|
||||
# shellcheck source=scripts/install/lib_os.sh
|
||||
. "$OS_LIB"
|
||||
OS_LIB_LOADED=1
|
||||
fi
|
||||
|
||||
if [ "$OS_LIB_LOADED" = "0" ]; then
|
||||
print_error "$OS_LIB is missing - download LEDMatrix again"
|
||||
elif [ -r "$LM_OS_RELEASE_FILE" ]; then
|
||||
OS_ID=$(lm_os_field ID)
|
||||
OS_VERSION_ID=$(lm_os_field VERSION_ID)
|
||||
echo "OS: $(lm_os_field PRETTY_NAME)"
|
||||
echo "Version ID: ${OS_VERSION_ID:-unknown}"
|
||||
|
||||
# first_time_install.sh refuses anything else, so this is an error here
|
||||
# too, not a warning.
|
||||
if OS_RELEASE=$(lm_os_release); then
|
||||
print_success "Detected $(lm_release_label "$OS_RELEASE") - supported"
|
||||
elif [[ "$OS_ID" == "raspbian" ]] || [[ "$OS_ID" == "debian" ]]; then
|
||||
print_error "Debian/Raspbian ${OS_VERSION_ID:-unknown} is not supported - the installer requires Raspberry Pi OS Lite, Trixie (Debian 13) or Bookworm (Debian 12)"
|
||||
else
|
||||
print_error "${ID:-unknown} is not supported - the installer requires Raspberry Pi OS Lite (Trixie), Debian 13"
|
||||
print_error "${OS_ID:-unknown} is not supported - the installer requires Raspberry Pi OS Lite, Trixie (Debian 13) or Bookworm (Debian 12)"
|
||||
fi
|
||||
else
|
||||
print_error "Could not detect OS version"
|
||||
@@ -92,7 +101,7 @@ if [ "$KERNEL_MAJOR" -ge "6" ]; then
|
||||
print_success "Kernel version is compatible (6.x or newer)"
|
||||
|
||||
if [ "$KERNEL_MAJOR" -eq "6" ] && [ "$KERNEL_MINOR" -ge "12" ]; then
|
||||
print_success "Running latest Trixie kernel (6.12 LTS)"
|
||||
print_success "Running a 6.12 LTS or newer kernel"
|
||||
fi
|
||||
elif [ "$KERNEL_MAJOR" -eq "5" ] && [ "$KERNEL_MINOR" -ge "10" ]; then
|
||||
print_success "Kernel version is compatible (5.10+)"
|
||||
@@ -104,25 +113,34 @@ echo ""
|
||||
# Check Python version
|
||||
echo "4. Checking Python Version..."
|
||||
echo "-----------------------------"
|
||||
if command -v python3 >/dev/null 2>&1; then
|
||||
PYTHON_VERSION=$(python3 -c 'import sys; print(f"{sys.version_info.major}.{sys.version_info.minor}.{sys.version_info.micro}")')
|
||||
PYTHON_MAJOR=$(python3 -c 'import sys; print(sys.version_info.major)')
|
||||
PYTHON_MINOR=$(python3 -c 'import sys; print(sys.version_info.minor)')
|
||||
|
||||
if [ "$OS_LIB_LOADED" = "1" ] && command -v python3 >/dev/null 2>&1; then
|
||||
PYTHON_VERSION=$(python3 -c 'import sys; print("%d.%d.%d" % sys.version_info[:3])')
|
||||
PYTHON_MINOR_VERSION=$(lm_python_version) || PYTHON_MINOR_VERSION=""
|
||||
PYTHON_RANGE="3.${LM_PYTHON_MIN_MINOR}-3.${LM_PYTHON_MAX_MINOR}"
|
||||
|
||||
echo "Python: $PYTHON_VERSION"
|
||||
|
||||
if [ "$PYTHON_MAJOR" -eq "3" ]; then
|
||||
if [ "$PYTHON_MINOR" -ge "10" ] && [ "$PYTHON_MINOR" -le "13" ]; then
|
||||
print_success "Python version is supported (3.10-3.13)"
|
||||
elif [ "$PYTHON_MINOR" -ge "14" ]; then
|
||||
print_warning "Python 3.${PYTHON_MINOR} is very new - some packages may not be compatible yet"
|
||||
else
|
||||
# Pillow 12 and the pinned test tools need 3.10+, so this won't install.
|
||||
print_error "Python 3.${PYTHON_MINOR} is too old - Python 3.10+ is required"
|
||||
fi
|
||||
else
|
||||
print_error "Python 2.x detected - Python 3.10+ is required"
|
||||
|
||||
case "$(lm_python_check "$PYTHON_MINOR_VERSION")" in
|
||||
ok)
|
||||
print_success "Python version is supported ($PYTHON_RANGE)"
|
||||
;;
|
||||
too-old)
|
||||
# The rgbmatrix bindings declare requires-python >=3.11, so the
|
||||
# display cannot be built on anything older.
|
||||
print_error "Python $PYTHON_MINOR_VERSION is too old - Python 3.${LM_PYTHON_MIN_MINOR}+ is required"
|
||||
;;
|
||||
too-new)
|
||||
print_warning "Python $PYTHON_MINOR_VERSION is newer than LEDMatrix has been tested with ($PYTHON_RANGE)"
|
||||
;;
|
||||
*)
|
||||
print_warning "Could not read the Python version"
|
||||
;;
|
||||
esac
|
||||
if [ -n "$OS_RELEASE" ] && [ "$PYTHON_MINOR_VERSION" != "$(lm_release_python "$OS_RELEASE")" ]; then
|
||||
print_warning "$(lm_release_label "$OS_RELEASE") ships Python $(lm_release_python "$OS_RELEASE"), but python3 runs $PYTHON_MINOR_VERSION"
|
||||
fi
|
||||
elif command -v python3 >/dev/null 2>&1; then
|
||||
print_warning "Cannot check the Python version without $OS_LIB"
|
||||
else
|
||||
print_error "Python 3 not found - installation required"
|
||||
fi
|
||||
@@ -268,6 +286,22 @@ if command -v ping >/dev/null 2>&1; then
|
||||
else
|
||||
print_warning "Ping command not available - cannot verify network"
|
||||
fi
|
||||
|
||||
# WiFi setup from the web page and the LEDMatrix-Setup hotspot drive
|
||||
# NetworkManager, the default on both Bookworm and Trixie.
|
||||
if [ "$OS_LIB_LOADED" = "1" ]; then
|
||||
case "$(lm_network_stack)" in
|
||||
networkmanager)
|
||||
print_success "NetworkManager manages the network (needed for WiFi setup)"
|
||||
;;
|
||||
dhcpcd)
|
||||
print_warning "dhcpcd manages the network - WiFi setup from the web page and the setup hotspot need NetworkManager (sudo raspi-config -> Advanced Options -> Network Config)"
|
||||
;;
|
||||
*)
|
||||
print_warning "Could not tell which service manages the network - WiFi setup from the web page needs NetworkManager"
|
||||
;;
|
||||
esac
|
||||
fi
|
||||
echo ""
|
||||
|
||||
# Print summary
|
||||
|
||||
+51
-3
@@ -9,7 +9,10 @@ reports the difference. Nothing is stopped, restarted or drawn.
|
||||
python3 scripts/frame_soak.py
|
||||
|
||||
# the same with the web preview open (the preview's PNG encodes are one of
|
||||
# the things that used to make the render loop miss refreshes)
|
||||
# the things that used to make the render loop miss refreshes). An open
|
||||
# preview is encoded at most once a second; through 3.8.0 it was up to
|
||||
# five times, so a --preview run from before that change is not comparable
|
||||
# with one from after it
|
||||
python3 scripts/frame_soak.py --preview
|
||||
|
||||
# quick look at the totals since the service started
|
||||
@@ -27,9 +30,16 @@ What the numbers mean
|
||||
late frames frames that reached the panel one or more refreshes after they
|
||||
were due -- the panel showed the previous frame again, which on
|
||||
a moving strip is a visible hitch. This is the pass/fail number.
|
||||
freezes gaps of 250ms+ inside a scroll: recomposes, plugin handovers,
|
||||
freezes gaps of 250ms+ inside a scroll: recomposes, plugin handovers
|
||||
the display controller does not tag (see handover gaps),
|
||||
blocking calls on the render thread. Reported, not failed on,
|
||||
since some are handovers between plugins rather than faults.
|
||||
handover gaps the same length of gap where the display controller had just
|
||||
started a screen's turn (also the same mode's again): its
|
||||
first display() drawing. Counted here instead of under
|
||||
freezes. Stats from a service older than this count have no
|
||||
such line, and their freezes include these, so do not
|
||||
compare freeze counts across that change.
|
||||
blit copying the frame into the matrix canvas (rgbmatrix SetImage).
|
||||
Grows with width x height x pwm_bits.
|
||||
wait blocked in SwapOnVSync, i.e. slack before the refresh.
|
||||
@@ -58,7 +68,8 @@ from src.common.frame_timing import ( # noqa: E402
|
||||
)
|
||||
|
||||
#: Touched by the web UI while someone has the preview open; a fresh marker
|
||||
#: puts the display service's snapshot writer at full rate. Same path as
|
||||
#: puts the display service's snapshot writer at the viewer rate
|
||||
#: (snapshot_policy.VIEWER_INTERVAL). Same path as
|
||||
#: DisplayManager._viewer_marker_path.
|
||||
VIEWER_MARKER = "/tmp/led_matrix_preview_viewer" # nosec B108 - fixed path shared with the service
|
||||
|
||||
@@ -156,6 +167,29 @@ def op_rows(totals: Dict[str, Any]) -> Dict[str, Dict[str, Any]]:
|
||||
return rows
|
||||
|
||||
|
||||
def gc_window(before: Dict[str, Any], after: Dict[str, Any]) -> Optional[Dict[str, Any]]:
|
||||
"""Garbage collection over the run, or None from a service without the
|
||||
monitor. The counters are cumulative since the service started, so they
|
||||
are differenced like the totals; the longest is since the start."""
|
||||
ga = after.get("gc")
|
||||
if not ga:
|
||||
return None
|
||||
gb = before.get("gc") or {}
|
||||
def minus(key):
|
||||
return [a - b for a, b in zip(ga.get(key, []),
|
||||
gb.get(key) or [0] * len(ga.get(key, [])))]
|
||||
seconds = minus("seconds")
|
||||
return {
|
||||
"collections": minus("collections"),
|
||||
"ms": [round(x * 1000.0, 1) for x in seconds],
|
||||
"long_pauses": ga.get("long_pauses", 0) - gb.get("long_pauses", 0),
|
||||
"long_ms": round((ga.get("long_seconds", 0.0)
|
||||
- gb.get("long_seconds", 0.0)) * 1000.0, 1),
|
||||
"threshold_ms": ga.get("threshold_ms"),
|
||||
"max_ms_since_start": ga.get("max_ms"),
|
||||
}
|
||||
|
||||
|
||||
def build_report(before, after, preview: bool) -> Dict[str, Any]:
|
||||
delta = diff(before, after)
|
||||
totals = delta["totals"]
|
||||
@@ -185,11 +219,15 @@ def build_report(before, after, preview: bool) -> Dict[str, Any]:
|
||||
"freezes": totals["freezes"],
|
||||
"freezes_per_hour": round(totals["freezes"] / hours, 1) if hours else None,
|
||||
"freeze_seconds": round(totals["freeze_seconds"], 2),
|
||||
# None from a service that predates the count: its handovers are
|
||||
# among the freezes above.
|
||||
"handover_freezes": totals.get("handover_freezes"),
|
||||
"worst_interval_ms": (round(totals["worst_interval_ms"], 1)
|
||||
if totals["worst_interval_ms"] else None),
|
||||
"timing_ms": {name: percentiles(h, bucket_ms)
|
||||
for name, h in delta["histograms"].items()},
|
||||
"ops": op_rows(totals),
|
||||
"gc": gc_window(before, after),
|
||||
}
|
||||
# The rate the panel held while rendering: the typical frame's interval
|
||||
# per refresh held. A few percent under the idle rate is normal (the Pi is
|
||||
@@ -240,6 +278,16 @@ def print_report(report: Dict[str, Any], limit: float) -> None:
|
||||
if report["freezes"]:
|
||||
print(" by length: " + ", ".join(
|
||||
f"{k}: {v}" for k, v in report["freeze_by"].items()))
|
||||
if report.get("handover_freezes") is not None:
|
||||
print(f"Handover gaps {report['handover_freezes']}"
|
||||
" >=250ms before a new screen's first frame; not in the freezes")
|
||||
gc_stats = report.get("gc")
|
||||
if gc_stats:
|
||||
counts, ms = gc_stats["collections"], gc_stats["ms"]
|
||||
print(f"Garbage collection gen0/1/2 {counts[0]}/{counts[1]}/{counts[2]}"
|
||||
f" ({ms[0]}/{ms[1]}/{ms[2]} ms) >={gc_stats['threshold_ms']:g}ms: "
|
||||
f"{gc_stats['long_pauses']} ({gc_stats['long_ms']} ms)"
|
||||
f" longest since start {gc_stats['max_ms_since_start']} ms")
|
||||
print()
|
||||
print(f"{'ms':<18}{'p50':>8}{'p95':>8}{'p99':>8}{'max':>8}")
|
||||
for name in ("blit", "wait", "work", "interval_per_hold"):
|
||||
|
||||
@@ -5,11 +5,18 @@ This directory contains scripts for installing and configuring the LEDMatrix sys
|
||||
## Scripts
|
||||
|
||||
- **`one-shot-install.sh`** - Single-command installer; clones the
|
||||
repo, checks prerequisites, then runs `first_time_install.sh`.
|
||||
Invoked via `curl ... | bash` from the project root README.
|
||||
repo, checks out the newest release (or `main` with
|
||||
`LEDMATRIX_CHANNEL=beta`), checks prerequisites, then runs
|
||||
`first_time_install.sh`. Invoked via `curl ... | bash` from the project
|
||||
root README. Re-running it never moves a checkout to an older version.
|
||||
- **`install_service.sh`** - Installs, enables and starts the display
|
||||
service (`ledmatrix.service`), the web interface service
|
||||
(`ledmatrix-web.service`) and the update-verify units (systemd)
|
||||
(`ledmatrix-web.service`) and the update-verify units (systemd), and
|
||||
installs `/usr/local/sbin/ledmatrix-refresh-units`
|
||||
- **`ledmatrix_refresh_units.py`** - Not run from here: `install_service.sh`
|
||||
installs a root-owned copy as `/usr/local/sbin/ledmatrix-refresh-units`,
|
||||
which updates run through sudo to install changed units (and the
|
||||
automatic update's rollback, with `--restore`, to put them back)
|
||||
- **`install_web_service.sh`** - Installs only the web interface service
|
||||
and the update-verify units (systemd)
|
||||
- **`install_wifi_monitor.sh`** - Installs the WiFi monitor daemon service
|
||||
@@ -34,6 +41,9 @@ Libraries (sourced, not run):
|
||||
script that renders a unit from `systemd/*.service`
|
||||
- **`lib_lowmem.sh`** - Build-job sizing and temporary swap for the C++
|
||||
build on low-memory Pis (`first_time_install.sh` Step 6)
|
||||
- **`lib_os.sh`** - Which releases (Bookworm, Trixie) and Python versions
|
||||
(3.11-3.13) the installer accepts, and which service runs the network;
|
||||
shared by `first_time_install.sh` and `scripts/check_system_compatibility.sh`
|
||||
|
||||
## Usage
|
||||
|
||||
|
||||
@@ -138,6 +138,8 @@ echo "- View system logs via journalctl"
|
||||
echo "- Reboot and shutdown the system"
|
||||
echo "- Remove plugin directories (for update/uninstall when root-owned files block deletion)"
|
||||
echo "- Install plugin/base requirements.txt as root (so ledmatrix.service can see them)"
|
||||
echo "- Install the LEDMatrix systemd units an update changed, and restore them on rollback"
|
||||
echo " (/usr/local/sbin/ledmatrix-refresh-units, installed by install_service.sh)"
|
||||
echo ""
|
||||
|
||||
# Ask for confirmation
|
||||
|
||||
@@ -143,6 +143,30 @@ for VERIFY_UNIT in ledmatrix-update-verify.service ledmatrix-update-verify.path;
|
||||
fi
|
||||
done
|
||||
|
||||
# The helper updates run (through sudo, see lib_sudoers.sh) to install these
|
||||
# same units when a new version changes their templates, and to put the old
|
||||
# ones back if the automatic update rolls back. Root-owned and outside the
|
||||
# checkout, so the web user who owns the checkout cannot change what sudo runs.
|
||||
# Not fatal: without it, updates leave the units for the next reinstall.
|
||||
REFRESH_UNITS_SRC="$PROJECT_ROOT_DIR/scripts/install/ledmatrix_refresh_units.py"
|
||||
REFRESH_UNITS_DEST=/usr/local/sbin/ledmatrix-refresh-units
|
||||
if [ -f "$REFRESH_UNITS_SRC" ]; then
|
||||
if sudo install -D -o root -g root -m 0755 "$REFRESH_UNITS_SRC" "$REFRESH_UNITS_DEST"; then
|
||||
echo "Installed $REFRESH_UNITS_DEST (lets updates refresh these units)"
|
||||
else
|
||||
echo "WARNING: could not install $REFRESH_UNITS_DEST; updates will not refresh the systemd units" >&2
|
||||
fi
|
||||
fi
|
||||
# The units above are copied from mktemp files, which are 0600. 0644 is what
|
||||
# first_time_install.sh (Step 8.1) sets, and lets the web interface compare
|
||||
# them with the templates after an update without root.
|
||||
for INSTALLED_UNIT in ledmatrix.service ledmatrix-web.service \
|
||||
ledmatrix-update-verify.service ledmatrix-update-verify.path; do
|
||||
if [ -f "/etc/systemd/system/$INSTALLED_UNIT" ]; then
|
||||
sudo chmod 644 "/etc/systemd/system/$INSTALLED_UNIT" || true
|
||||
fi
|
||||
done
|
||||
|
||||
echo "Reloading systemd daemon for web service..."
|
||||
sudo systemctl daemon-reload
|
||||
|
||||
|
||||
Executable
+451
@@ -0,0 +1,451 @@
|
||||
#!/usr/bin/python3 -I
|
||||
"""Refresh the installed LEDMatrix systemd units from the checkout's templates.
|
||||
|
||||
Installed by scripts/install/install_service.sh as a root-owned copy,
|
||||
/usr/local/sbin/ledmatrix-refresh-units, and granted to the web interface's
|
||||
user by /etc/sudoers.d/ledmatrix_web (scripts/install/lib_sudoers.sh) with
|
||||
exactly two command lines:
|
||||
|
||||
ledmatrix-refresh-units (no arguments)
|
||||
ledmatrix-refresh-units --restore
|
||||
|
||||
An update (Update Code, or the weekly automatic update) pulls new unit
|
||||
templates into systemd/, but the units systemd runs are the copies in
|
||||
/etc/systemd/system, which only the installer used to write. So a setting
|
||||
added to a template -- the render-loop watchdog, a memory limit -- never
|
||||
reached a device that was already installed. After an update the web
|
||||
interface runs this, and the next restart picks the new units up.
|
||||
|
||||
* **No arguments:** render each installed unit from systemd/<unit> exactly as
|
||||
install_service.sh does (__PROJECT_ROOT_DIR__ and __USER__ replaced
|
||||
literally), and install the ones whose content differs (comments and blank
|
||||
lines aside, as src/startup_validator.py compares them), then
|
||||
``systemctl daemon-reload``. The units replaced are saved first, so the
|
||||
automatic update's rollback can put them back.
|
||||
* ``--restore``: put back the units the last refresh replaced, and
|
||||
daemon-reload. Nothing saved means nothing to do.
|
||||
* ``--check``: print the units that would change, one per line. Needs no
|
||||
root and changes nothing.
|
||||
|
||||
What it trusts, and why. It takes no other input: the project directory and
|
||||
the web interface's user come from the installed, root-owned
|
||||
ledmatrix.service and ledmatrix-web.service, not from the caller, and sudo
|
||||
strips the caller's environment (``-I`` ignores the PYTHON* variables too).
|
||||
It only replaces units that are already installed, only the four
|
||||
install_service.sh installs, and only with a rendering that keeps each unit's
|
||||
User= (root for the display, the web user for the others) and
|
||||
WorkingDirectory=. The templates are files the web user can edit -- but so is
|
||||
run.py, which ledmatrix.service already runs as root, so a template grants
|
||||
nothing that user did not have; the checks keep a damaged or hostile template
|
||||
from changing who a unit runs as, and keep this from reading anything but a
|
||||
regular file under the checkout's systemd/ folder.
|
||||
|
||||
Standard library only, and no imports from the checkout: the installed copy
|
||||
must not run code the web user can change.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import stat
|
||||
import subprocess # nosec B404 - fixed argv, no shell # nosemgrep
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
SYSTEMD_DIR = '/etc/systemd/system'
|
||||
#: Root-only: the units the last refresh replaced, for --restore.
|
||||
BACKUP_DIR = '/var/lib/ledmatrix/unit-backup'
|
||||
MANIFEST = 'manifest.json'
|
||||
INSTALLED_PATH = '/usr/local/sbin/ledmatrix-refresh-units'
|
||||
|
||||
DISPLAY_UNIT = 'ledmatrix.service'
|
||||
WEB_UNIT = 'ledmatrix-web.service'
|
||||
VERIFY_SERVICE = 'ledmatrix-update-verify.service'
|
||||
VERIFY_PATH = 'ledmatrix-update-verify.path'
|
||||
#: What install_service.sh installs, in its order. Nothing else is touched.
|
||||
UNITS = (DISPLAY_UNIT, WEB_UNIT, VERIFY_SERVICE, VERIFY_PATH)
|
||||
|
||||
MAX_TEMPLATE_BYTES = 64 * 1024
|
||||
_USER_RE = re.compile(r'^[a-z_][a-z0-9_-]{0,31}$')
|
||||
#: systemd expands % specifiers, and a quote, backslash or line break would
|
||||
#: be reinterpreted in a unit file (src/auto_update_setup.py refuses the same).
|
||||
#: (On Windows, where the tests also run, a backslash is the path separator.)
|
||||
_UNSAFE_PATH_CHARS = set('%"') | ({'\\'} if os.sep == '/' else set())
|
||||
|
||||
EXIT_OK = 0
|
||||
EXIT_FAILED = 1
|
||||
EXIT_USAGE = 2
|
||||
|
||||
|
||||
class RefreshError(Exception):
|
||||
"""Why the units were left alone, in words for the web interface's log."""
|
||||
|
||||
|
||||
class UnitsUnreadable(RefreshError):
|
||||
"""An installed unit is not readable by this (unprivileged) user.
|
||||
|
||||
install_service.sh used to leave units mode 0600 (first_time_install.sh's
|
||||
Step 8.1 makes them 0644), so ``--check`` as the web user cannot always
|
||||
tell; the root helper itself can.
|
||||
"""
|
||||
|
||||
|
||||
def directive_values(text, key):
|
||||
"""Every value of ``key=`` in a unit's text, in order (systemd allows spaces around ``=``)."""
|
||||
return [m.group(1).strip() for m in re.finditer(rf'^[ \t]*{key}[ \t]*=(.*)$', text or '', re.M)]
|
||||
|
||||
|
||||
def layout_problem(text, section, keys):
|
||||
"""What would make ``directive_values`` misread the unit as systemd reads it, or None.
|
||||
|
||||
A ``User=`` inside a backslash-continued line is part of the line before,
|
||||
and one under [Unit] is ignored, so either could pass a check that systemd
|
||||
then does not apply. Neither appears in the shipped templates.
|
||||
"""
|
||||
current = None
|
||||
for raw in (text or '').splitlines():
|
||||
line = raw.strip()
|
||||
if not line or line.startswith(('#', ';')):
|
||||
continue
|
||||
if line.endswith('\\'):
|
||||
return 'continues a line with a backslash'
|
||||
if line.startswith('[') and line.endswith(']'):
|
||||
current = line[1:-1]
|
||||
continue
|
||||
key = line.split('=', 1)[0].strip()
|
||||
if key in keys and current != section:
|
||||
return f'sets {key}= outside [{section}]'
|
||||
return None
|
||||
|
||||
|
||||
def unit_body(text):
|
||||
"""A unit's meaningful lines in order: no comments, no blank lines.
|
||||
|
||||
The same comparison src/startup_validator.py uses for its drift warning,
|
||||
so what this refreshes is exactly what that warns about.
|
||||
"""
|
||||
lines = []
|
||||
for line in (text or '').splitlines():
|
||||
line = line.strip()
|
||||
if line and not line.startswith('#'):
|
||||
lines.append(line)
|
||||
return '\n'.join(lines)
|
||||
|
||||
|
||||
def render(template, project_root, user):
|
||||
"""install_service.sh's ``sed "s|__PROJECT_ROOT_DIR__|...|g; s|__USER__|...|g"``."""
|
||||
return template.replace('__PROJECT_ROOT_DIR__', project_root).replace('__USER__', user)
|
||||
|
||||
|
||||
def _read_regular(path, limit=MAX_TEMPLATE_BYTES, dir_fd=None):
|
||||
"""A regular file's text, never through a symlink, a FIFO or a device."""
|
||||
flags = os.O_RDONLY | getattr(os, 'O_NOFOLLOW', 0) | getattr(os, 'O_NONBLOCK', 0)
|
||||
kwargs = {'dir_fd': dir_fd} if dir_fd is not None else {}
|
||||
fd = os.open(path, flags, **kwargs)
|
||||
try:
|
||||
info = os.fstat(fd)
|
||||
if not stat.S_ISREG(info.st_mode):
|
||||
raise RefreshError(f'{path} is not a regular file')
|
||||
if info.st_size > limit:
|
||||
raise RefreshError(f'{path} is larger than {limit} bytes')
|
||||
data = b''
|
||||
while True:
|
||||
chunk = os.read(fd, limit + 1 - len(data))
|
||||
if not chunk:
|
||||
break
|
||||
data += chunk
|
||||
if len(data) > limit:
|
||||
raise RefreshError(f'{path} is larger than {limit} bytes')
|
||||
finally:
|
||||
os.close(fd)
|
||||
if b'\0' in data:
|
||||
raise RefreshError(f'{path} is not a text file')
|
||||
try:
|
||||
return data.decode('utf-8')
|
||||
except UnicodeDecodeError as e:
|
||||
raise RefreshError(f'{path} is not UTF-8') from e
|
||||
|
||||
|
||||
def _read_installed(systemd_dir, name):
|
||||
path = os.path.join(systemd_dir, name)
|
||||
try:
|
||||
with open(path, 'r', encoding='utf-8') as f:
|
||||
return f.read()
|
||||
except FileNotFoundError:
|
||||
return None
|
||||
except PermissionError as e:
|
||||
raise UnitsUnreadable(f'cannot read the installed {name}: {e}') from e
|
||||
except (OSError, UnicodeDecodeError) as e:
|
||||
raise RefreshError(f'cannot read the installed {name}: {e}') from e
|
||||
|
||||
|
||||
def _lookup_user(user):
|
||||
try:
|
||||
import pwd
|
||||
except ImportError: # not a POSIX host (the tests on Windows)
|
||||
return True
|
||||
try:
|
||||
pwd.getpwnam(user)
|
||||
return True
|
||||
except KeyError:
|
||||
return False
|
||||
|
||||
|
||||
class Refresher:
|
||||
def __init__(self, systemd_dir=SYSTEMD_DIR, backup_dir=BACKUP_DIR, run=subprocess.run,
|
||||
is_root=None, user_exists=_lookup_user, log=None):
|
||||
self.systemd_dir = systemd_dir
|
||||
self.backup_dir = backup_dir
|
||||
self.run = run
|
||||
self.is_root = is_root or (lambda: hasattr(os, 'geteuid') and os.geteuid() == 0)
|
||||
self.user_exists = user_exists
|
||||
self.log = log or (lambda msg: print(msg, flush=True))
|
||||
|
||||
# -- what the installed units say -------------------------------------
|
||||
|
||||
def context(self, installed):
|
||||
"""(project root, web user) from the installed, root-owned units."""
|
||||
display = installed.get(DISPLAY_UNIT)
|
||||
if display is None:
|
||||
raise RefreshError(f'{DISPLAY_UNIT} is not installed; run scripts/install/install_service.sh')
|
||||
roots = directive_values(display, 'WorkingDirectory')
|
||||
if len(roots) != 1:
|
||||
raise RefreshError(f'the installed {DISPLAY_UNIT} does not name one WorkingDirectory')
|
||||
root = roots[0]
|
||||
if (not os.path.isabs(root) or any(ch in _UNSAFE_PATH_CHARS or ord(ch) < 32 for ch in root)
|
||||
or os.path.normpath(root) != root):
|
||||
raise RefreshError(f'the installed {DISPLAY_UNIT} runs from {root!r}, which cannot be used')
|
||||
if not os.path.isdir(root):
|
||||
raise RefreshError(f'{root} (the installed {DISPLAY_UNIT} WorkingDirectory) does not exist')
|
||||
|
||||
user = None
|
||||
web = installed.get(WEB_UNIT)
|
||||
if web is not None:
|
||||
users = directive_values(web, 'User')
|
||||
user = users[0] if len(users) == 1 else ('root' if not users else None)
|
||||
if user is None or not _USER_RE.match(user) or not self.user_exists(user):
|
||||
raise RefreshError(f'the installed {WEB_UNIT} runs as an account that cannot be used')
|
||||
if directive_values(web, 'WorkingDirectory') != [root]:
|
||||
raise RefreshError(f'the installed {WEB_UNIT} and {DISPLAY_UNIT} run from different folders')
|
||||
return root, user
|
||||
|
||||
@staticmethod
|
||||
def expected_user(name, web_user):
|
||||
return 'root' if name == DISPLAY_UNIT else web_user
|
||||
|
||||
def _template(self, root, name):
|
||||
"""systemd/<name> under the checkout, as a regular file, never via a symlink."""
|
||||
dir_flags = os.O_RDONLY | getattr(os, 'O_DIRECTORY', 0) | getattr(os, 'O_NOFOLLOW', 0)
|
||||
if os.open in getattr(os, 'supports_dir_fd', set()):
|
||||
try:
|
||||
dfd = os.open(os.path.join(root, 'systemd'), dir_flags)
|
||||
except OSError as e:
|
||||
raise RefreshError(f'cannot open {root}/systemd: {e}') from e
|
||||
try:
|
||||
return _read_regular(name, dir_fd=dfd)
|
||||
except FileNotFoundError:
|
||||
return None
|
||||
except OSError as e:
|
||||
raise RefreshError(f'cannot read systemd/{name}: {e}') from e
|
||||
finally:
|
||||
os.close(dfd)
|
||||
path = os.path.join(root, 'systemd', name)
|
||||
if os.path.islink(os.path.join(root, 'systemd')):
|
||||
raise RefreshError(f'{root}/systemd is a symlink')
|
||||
try:
|
||||
return _read_regular(path)
|
||||
except FileNotFoundError:
|
||||
return None
|
||||
except OSError as e:
|
||||
raise RefreshError(f'cannot read systemd/{name}: {e}') from e
|
||||
|
||||
def _validate(self, name, rendered, root, user):
|
||||
problem = layout_problem(rendered, 'Service', ('User', 'WorkingDirectory'))
|
||||
if problem:
|
||||
raise RefreshError(f'systemd/{name} {problem}; refusing to install it')
|
||||
if directive_values(rendered, 'User') != [user]:
|
||||
raise RefreshError(f'systemd/{name} would not run as {user}; refusing to install it')
|
||||
if directive_values(rendered, 'WorkingDirectory') != [root]:
|
||||
raise RefreshError(f'systemd/{name} would not run from {root}; refusing to install it')
|
||||
|
||||
def plan(self):
|
||||
"""{unit: (installed text, new text)} for every installed unit that would change.
|
||||
|
||||
Raises RefreshError, and so changes nothing, if any unit cannot be
|
||||
rendered safely: four units refreshed as a set or not at all.
|
||||
"""
|
||||
installed = {name: _read_installed(self.systemd_dir, name) for name in UNITS}
|
||||
root, web_user = self.context(installed)
|
||||
changes = {}
|
||||
for name in UNITS:
|
||||
current = installed[name]
|
||||
if current is None:
|
||||
continue # never installed here: installing is the installer's job
|
||||
user = self.expected_user(name, web_user)
|
||||
if user is None:
|
||||
continue # the web unit is not installed, so neither is its user
|
||||
template = self._template(root, name)
|
||||
if template is None:
|
||||
continue # a version without this unit leaves the installed one alone
|
||||
rendered = render(template, root, user)
|
||||
# A path unit runs nothing itself; what matters is what it starts.
|
||||
if name.endswith('.service'):
|
||||
self._validate(name, rendered, root, user)
|
||||
else:
|
||||
self._validate_path(name, rendered)
|
||||
if unit_body(rendered) != unit_body(current):
|
||||
changes[name] = (current, rendered)
|
||||
return changes
|
||||
|
||||
def _validate_path(self, name, rendered):
|
||||
problem = layout_problem(rendered, 'Path', ('Unit',))
|
||||
if problem:
|
||||
raise RefreshError(f'systemd/{name} {problem}; refusing to install it')
|
||||
if directive_values(rendered, 'Unit') != [VERIFY_SERVICE]:
|
||||
raise RefreshError(f'systemd/{name} does not start {VERIFY_SERVICE}; refusing to install it')
|
||||
if directive_values(rendered, 'User'):
|
||||
raise RefreshError(f'systemd/{name} sets User=; refusing to install it')
|
||||
|
||||
# -- writing ------------------------------------------------------------
|
||||
|
||||
def _write_unit(self, name, text):
|
||||
fd, tmp = tempfile.mkstemp(dir=self.systemd_dir, prefix=f'.{name}.')
|
||||
try:
|
||||
with os.fdopen(fd, 'w', encoding='utf-8', newline='\n') as f:
|
||||
f.write(text)
|
||||
os.chmod(tmp, 0o644)
|
||||
os.replace(tmp, os.path.join(self.systemd_dir, name))
|
||||
except BaseException:
|
||||
try:
|
||||
os.unlink(tmp)
|
||||
except OSError:
|
||||
pass
|
||||
raise
|
||||
|
||||
def _backup_dir(self):
|
||||
"""The backup folder, created root-only; refused if it is not a plain folder."""
|
||||
os.makedirs(os.path.dirname(self.backup_dir), mode=0o755, exist_ok=True)
|
||||
try:
|
||||
os.mkdir(self.backup_dir, 0o700)
|
||||
except FileExistsError:
|
||||
pass
|
||||
info = os.lstat(self.backup_dir)
|
||||
if not stat.S_ISDIR(info.st_mode):
|
||||
raise RefreshError(f'{self.backup_dir} is not a folder')
|
||||
if hasattr(os, 'geteuid') and info.st_uid != os.geteuid():
|
||||
raise RefreshError(f'{self.backup_dir} is not owned by root')
|
||||
return self.backup_dir
|
||||
|
||||
def _clear_backup(self, folder):
|
||||
for entry in os.listdir(folder):
|
||||
path = os.path.join(folder, entry)
|
||||
if os.path.isfile(path) or os.path.islink(path):
|
||||
os.unlink(path)
|
||||
|
||||
def _systemctl(self, *args):
|
||||
result = self.run(['systemctl', *args], capture_output=True, text=True, timeout=60)
|
||||
if result.returncode != 0:
|
||||
raise RefreshError(f'"systemctl {" ".join(args)}" failed: '
|
||||
f'{(result.stderr or result.stdout or "").strip()}')
|
||||
|
||||
def _restart_path_unit_if_active(self, names):
|
||||
"""A rewritten path unit watches the old path until it is restarted."""
|
||||
if VERIFY_PATH not in names:
|
||||
return
|
||||
state = self.run(['systemctl', 'is-active', VERIFY_PATH], capture_output=True, text=True, timeout=30)
|
||||
if (state.stdout or '').strip() == 'active':
|
||||
self._systemctl('restart', VERIFY_PATH)
|
||||
|
||||
def refresh(self):
|
||||
if not self.is_root():
|
||||
raise RefreshError('must run as root (sudo)')
|
||||
changes = self.plan()
|
||||
folder = self._backup_dir()
|
||||
# Always reset: the backup belongs to this refresh, so a --restore
|
||||
# after an update that changed nothing restores nothing.
|
||||
self._clear_backup(folder)
|
||||
if not changes:
|
||||
self.log('units: up to date')
|
||||
return []
|
||||
for name, (current, _) in changes.items():
|
||||
with open(os.path.join(folder, name), 'w', encoding='utf-8', newline='\n') as f:
|
||||
f.write(current)
|
||||
with open(os.path.join(folder, MANIFEST), 'w', encoding='utf-8') as f:
|
||||
json.dump({'units': sorted(changes)}, f)
|
||||
try:
|
||||
for name, (_, rendered) in changes.items():
|
||||
self._write_unit(name, rendered)
|
||||
self._systemctl('daemon-reload')
|
||||
except BaseException:
|
||||
# A failed refresh is reported as a failure, so the update records
|
||||
# no units_refreshed and a rollback would not --restore. Put the
|
||||
# replaced units back now, rather than leave a half-written set
|
||||
# under the old code.
|
||||
self._undo(changes, folder)
|
||||
raise
|
||||
self._restart_path_unit_if_active(changes)
|
||||
self.log('units refreshed: ' + ' '.join(sorted(changes)))
|
||||
return sorted(changes)
|
||||
|
||||
def _undo(self, changes, folder):
|
||||
"""Best effort: reinstall the units a failed refresh replaced."""
|
||||
undone = True
|
||||
for name, (current, _) in changes.items():
|
||||
try:
|
||||
self._write_unit(name, current)
|
||||
except OSError as e:
|
||||
undone = False
|
||||
self.log(f'units: could not put back {name}: {e}')
|
||||
try:
|
||||
self.run(['systemctl', 'daemon-reload'], capture_output=True, text=True, timeout=60)
|
||||
except (OSError, subprocess.SubprocessError) as e:
|
||||
self.log(f'units: daemon-reload after putting units back failed: {e}')
|
||||
if undone:
|
||||
# Nothing is left to restore; keep the backup only if a unit could
|
||||
# not be put back, so a manual --restore still can.
|
||||
self._clear_backup(folder)
|
||||
|
||||
def restore(self):
|
||||
if not self.is_root():
|
||||
raise RefreshError('must run as root (sudo)')
|
||||
folder = self._backup_dir()
|
||||
try:
|
||||
manifest = json.loads(_read_regular(os.path.join(folder, MANIFEST)))
|
||||
except FileNotFoundError:
|
||||
self.log('units: nothing to restore')
|
||||
return []
|
||||
names = [n for n in (manifest or {}).get('units', []) if n in UNITS]
|
||||
for name in names:
|
||||
self._write_unit(name, _read_regular(os.path.join(folder, name)))
|
||||
self._systemctl('daemon-reload')
|
||||
self._restart_path_unit_if_active(names)
|
||||
self._clear_backup(folder)
|
||||
self.log('units restored: ' + ' '.join(names))
|
||||
return names
|
||||
|
||||
|
||||
def main(argv, refresher=None):
|
||||
args = argv[1:]
|
||||
if args not in ([], ['--restore'], ['--check']):
|
||||
print('usage: ledmatrix-refresh-units [--restore | --check]', file=sys.stderr)
|
||||
return EXIT_USAGE
|
||||
refresher = refresher or Refresher()
|
||||
try:
|
||||
if args == ['--check']:
|
||||
for name in sorted(refresher.plan()):
|
||||
print(name)
|
||||
elif args == ['--restore']:
|
||||
refresher.restore()
|
||||
else:
|
||||
refresher.refresh()
|
||||
except (RefreshError, OSError, subprocess.SubprocessError, ValueError) as e:
|
||||
print(f'ledmatrix-refresh-units: {e}', file=sys.stderr)
|
||||
return EXIT_FAILED
|
||||
return EXIT_OK
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
# Only as the installed program: sudo already sets a secure PATH, and
|
||||
# this pins the one systemctl comes from. (Not in main(), which the
|
||||
# tests call in-process.)
|
||||
os.environ['PATH'] = '/usr/sbin:/usr/bin:/sbin:/bin'
|
||||
sys.exit(main(sys.argv))
|
||||
@@ -0,0 +1,138 @@
|
||||
#!/bin/bash
|
||||
# Which operating systems and Python versions LEDMatrix installs on.
|
||||
#
|
||||
# Sourced by first_time_install.sh and scripts/check_system_compatibility.sh,
|
||||
# so the installer and the compatibility checker cannot disagree about what
|
||||
# is supported. Pure functions: nothing here installs, changes or exits --
|
||||
# the callers decide what to do with the answers.
|
||||
#
|
||||
# Supported (Lite, no desktop):
|
||||
# Raspberry Pi OS / Debian 12 "Bookworm" -- Python 3.11
|
||||
# Raspberry Pi OS / Debian 13 "Trixie" -- Python 3.13
|
||||
#
|
||||
# Everything the installer asks apt for (python3-pip, python3-venv,
|
||||
# python-dev-is-python3, python3-pil, python3-pil.imagetk, build-essential,
|
||||
# python3-setuptools, python3-wheel, cmake, ninja-build, git, curl, wget,
|
||||
# unzip, and hostapd, dnsmasq, network-manager for WiFi setup) has the same
|
||||
# name on both releases. Both ship a pip (23.0.1 and 25.1.1) that is PEP 668
|
||||
# "externally managed" and accepts --break-system-packages, and a cmake (3.25
|
||||
# and 3.31) new enough for the rgbmatrix build (3.22). So no step needs a
|
||||
# per-release branch today; if one ever does, the release name comes from
|
||||
# lm_os_release below.
|
||||
|
||||
# Test hook: the os-release file to read.
|
||||
LM_OS_RELEASE_FILE="${LM_OS_RELEASE_FILE:-/etc/os-release}"
|
||||
|
||||
# Oldest and newest python3 minor versions the installer accepts. 3.11 is
|
||||
# Bookworm's, and also the floor of the rgbmatrix bindings (requires-python
|
||||
# >=3.11 in rpi-rgb-led-matrix-master/pyproject.toml); 3.13 is Trixie's.
|
||||
LM_PYTHON_MIN_MINOR=11
|
||||
LM_PYTHON_MAX_MINOR=13
|
||||
|
||||
# lm_os_field KEY -- one value from os-release with its quotes removed; empty
|
||||
# when the key or the file is missing. Parsed rather than sourced so that
|
||||
# os-release's ID, VERSION and friends do not land in the caller's variables.
|
||||
lm_os_field() {
|
||||
[ -r "$LM_OS_RELEASE_FILE" ] || return 0
|
||||
sed -n "/^$1=/{s/^$1=//;s/^[\"']//;s/[\"']\$//;p;q;}" "$LM_OS_RELEASE_FILE"
|
||||
}
|
||||
|
||||
# lm_os_release -- print "bookworm" or "trixie" and succeed on a supported
|
||||
# release; print nothing and fail on anything else. VERSION_ID decides; the
|
||||
# codename is used only when VERSION_ID is missing.
|
||||
lm_os_release() {
|
||||
local id version
|
||||
id=$(lm_os_field ID)
|
||||
version=$(lm_os_field VERSION_ID)
|
||||
[ -n "$version" ] || version=$(lm_os_field VERSION_CODENAME)
|
||||
case "$id" in
|
||||
raspbian|debian) ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
case "$version" in
|
||||
12|bookworm) echo bookworm ;;
|
||||
13|trixie) echo trixie ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# lm_release_label RELEASE -- how to name a release to a person.
|
||||
lm_release_label() {
|
||||
case "$1" in
|
||||
bookworm) echo "Debian 12 (Bookworm)" ;;
|
||||
trixie) echo "Debian 13 (Trixie)" ;;
|
||||
*) echo "$1" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# lm_release_python RELEASE -- the python3 version a release ships, e.g. 3.11.
|
||||
lm_release_python() {
|
||||
case "$1" in
|
||||
bookworm) echo 3.11 ;;
|
||||
trixie) echo 3.13 ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# lm_python_version [PYTHON] -- "3.11" and so on for python3 (or PYTHON);
|
||||
# prints nothing and fails when it cannot be run.
|
||||
lm_python_version() {
|
||||
"${1:-python3}" -c 'import sys; print("%d.%d" % sys.version_info[:2])' 2>/dev/null
|
||||
}
|
||||
|
||||
# lm_python_check VERSION -- print "ok", "too-old", "too-new" or "unknown"
|
||||
# for a version such as 3.11. Always succeeds, so it is safe under set -e.
|
||||
lm_python_check() {
|
||||
local major minor
|
||||
major=${1%%.*}
|
||||
minor=${1#*.}
|
||||
minor=${minor%%.*}
|
||||
case "$major:$minor" in
|
||||
*[!0-9:]*|:*|*:) echo unknown; return 0 ;;
|
||||
esac
|
||||
if [ "$major" -lt 3 ] || { [ "$major" -eq 3 ] && [ "$minor" -lt "$LM_PYTHON_MIN_MINOR" ]; }; then
|
||||
echo too-old
|
||||
elif [ "$major" -gt 3 ] || [ "$minor" -gt "$LM_PYTHON_MAX_MINOR" ]; then
|
||||
echo too-new
|
||||
else
|
||||
echo ok
|
||||
fi
|
||||
}
|
||||
|
||||
# lm_network_stack -- which service runs the network: "networkmanager",
|
||||
# "dhcpcd" or "unknown". Raspberry Pi OS uses NetworkManager on both Bookworm
|
||||
# and Trixie; dhcpcd appears when someone switched back to it in raspi-config.
|
||||
lm_network_stack() {
|
||||
if systemctl is-active --quiet NetworkManager 2>/dev/null; then
|
||||
echo networkmanager
|
||||
elif systemctl is-active --quiet dhcpcd 2>/dev/null; then
|
||||
echo dhcpcd
|
||||
else
|
||||
echo unknown
|
||||
fi
|
||||
}
|
||||
|
||||
# lm_print_dhcpcd_advice -- the explanation for a Pi running dhcpcd. WiFi
|
||||
# setup from the web page and the LEDMatrix-Setup hotspot both drive
|
||||
# NetworkManager (nmcli). The installer does not switch the network stack
|
||||
# itself: doing that over SSH can cut the connection it is running on.
|
||||
lm_print_dhcpcd_advice() {
|
||||
echo "⚠ This Pi manages its network with dhcpcd, not NetworkManager."
|
||||
echo " LEDMatrix installs and the display works, but choosing a WiFi network"
|
||||
echo " from the web page and the LEDMatrix-Setup hotspot both need NetworkManager."
|
||||
echo " To switch (with a keyboard and screen attached, or over Ethernet):"
|
||||
echo " sudo raspi-config -> Advanced Options -> Network Config -> NetworkManager"
|
||||
echo " then reboot."
|
||||
}
|
||||
|
||||
# lm_print_supported_os_help -- what to do on an unsupported system.
|
||||
lm_print_supported_os_help() {
|
||||
echo "LEDMatrix needs Raspberry Pi OS Lite: Trixie (Debian 13) or Bookworm (Debian 12)."
|
||||
echo ""
|
||||
echo "To install Raspberry Pi OS Lite:"
|
||||
echo " 1. Download Raspberry Pi Imager from: https://www.raspberrypi.com/software/"
|
||||
echo " 2. Choose 'Raspberry Pi OS Lite (64-bit)'. Trixie is the current version and"
|
||||
echo " is recommended; Bookworm (listed as Legacy) also works"
|
||||
echo " 3. Flash it to the SD card"
|
||||
echo " 4. Boot the Pi and run this script again"
|
||||
}
|
||||
@@ -10,6 +10,11 @@
|
||||
#
|
||||
# Add or remove a grant here and nowhere else.
|
||||
|
||||
# Root-owned copy of scripts/install/ledmatrix_refresh_units.py, installed by
|
||||
# install_service.sh. Outside the checkout on purpose: the web user owns the
|
||||
# checkout, so a granted file inside it could be rewritten and run as root.
|
||||
LEDMATRIX_REFRESH_UNITS_PATH=/usr/local/sbin/ledmatrix-refresh-units
|
||||
|
||||
# web_sudoers_rules WEB_USER PROJECT_ROOT SYSTEMCTL_PATH BASH_PATH REBOOT_PATH POWEROFF_PATH JOURNALCTL_PATH
|
||||
#
|
||||
# Print the ledmatrix_web sudoers rules to stdout.
|
||||
@@ -58,6 +63,10 @@ $WEB_USER ALL=(ALL) NOPASSWD: $BASH_PATH $PROJECT_ROOT/scripts/fix_perms/safe_pl
|
||||
# Install a requirements.txt as root via vetted helper, so packages are visible
|
||||
# to root-run ledmatrix.service (not just the web interface's own user).
|
||||
$WEB_USER ALL=(ALL) NOPASSWD: $BASH_PATH $PROJECT_ROOT/scripts/fix_perms/safe_pip_install.sh *
|
||||
# After an update, install the new systemd units (no arguments: "" allows none)
|
||||
# and, on the automatic update's rollback, put the previous ones back.
|
||||
$WEB_USER ALL=(ALL) NOPASSWD: $LEDMATRIX_REFRESH_UNITS_PATH ""
|
||||
$WEB_USER ALL=(ALL) NOPASSWD: $LEDMATRIX_REFRESH_UNITS_PATH --restore
|
||||
EOF
|
||||
if [ -n "$JOURNALCTL_PATH" ]; then
|
||||
cat << EOF
|
||||
|
||||
@@ -3,6 +3,10 @@
|
||||
# LED Matrix One-Shot Installation Script
|
||||
# This script provides a single-command installation experience
|
||||
# Usage: curl -fsSL https://raw.githubusercontent.com/ChuckBuilds/LEDMatrix/main/scripts/install/one-shot-install.sh | bash
|
||||
#
|
||||
# A new install runs the newest release (the stable update channel). For the
|
||||
# newest code from main instead (the beta channel), set LEDMATRIX_CHANNEL=beta:
|
||||
# curl -fsSL https://raw.githubusercontent.com/ChuckBuilds/LEDMatrix/main/scripts/install/one-shot-install.sh | LEDMATRIX_CHANNEL=beta bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
@@ -205,6 +209,114 @@ check_sudo() {
|
||||
print_success "Sudo access confirmed"
|
||||
}
|
||||
|
||||
# --- release checkout helpers ------------------------------------------------
|
||||
# Which version an install runs. The rules are web_interface/update_channel.py's,
|
||||
# so the installer and the web interface's updates agree:
|
||||
# stable (default) the newest vX.Y.Z tag by semantic version; pre-releases
|
||||
# (v3.8.0-rc1), leading zeros and other tags are ignored
|
||||
# beta main, the newest code
|
||||
# Never backwards: an existing checkout moves to a release only when that
|
||||
# release contains its current commit (git merge-base --is-ancestor).
|
||||
# Never fatal: whatever goes wrong, the install carries on with the checkout
|
||||
# as it is.
|
||||
|
||||
# Print "stable" or "beta": LEDMATRIX_CHANNEL when it is set, else the
|
||||
# existing install's auto_update.channel (CONFIG_FILE), else stable.
|
||||
_lm_channel() {
|
||||
local config_file="${1:-}" value
|
||||
value=$(printf '%s' "${LEDMATRIX_CHANNEL:-}" | tr '[:upper:]' '[:lower:]' | tr -d '[:space:]')
|
||||
case "$value" in
|
||||
stable|beta) printf '%s\n' "$value"; return 0 ;;
|
||||
"") ;;
|
||||
*) print_warning "LEDMATRIX_CHANNEL=${LEDMATRIX_CHANNEL} is not stable or beta; using stable" >&2
|
||||
printf 'stable\n'; return 0 ;;
|
||||
esac
|
||||
if [ -n "$config_file" ] && [ -f "$config_file" ] && command -v python3 >/dev/null 2>&1; then
|
||||
value=$(python3 - "$config_file" 2>/dev/null <<'PY' || true
|
||||
import json, sys
|
||||
try:
|
||||
with open(sys.argv[1], encoding="utf-8") as f:
|
||||
section = json.load(f).get("auto_update")
|
||||
value = section.get("channel") if isinstance(section, dict) else None
|
||||
print(value.strip().lower() if isinstance(value, str) else "")
|
||||
except Exception:
|
||||
print("")
|
||||
PY
|
||||
)
|
||||
if [ "$value" = "beta" ]; then
|
||||
printf 'beta\n'
|
||||
return 0
|
||||
fi
|
||||
fi
|
||||
printf 'stable\n'
|
||||
}
|
||||
|
||||
# Print the newest release tag of the repository in the current directory,
|
||||
# or nothing when it has none.
|
||||
_lm_newest_release_tag() {
|
||||
git tag --list 'v*' 2>/dev/null \
|
||||
| grep -E '^v(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)$' \
|
||||
| sort -t. -k1.2,1n -k2,2n -k3,3n \
|
||||
| tail -n 1 || true
|
||||
}
|
||||
|
||||
# A fresh clone (on main): move to the newest release unless beta was asked for.
|
||||
_lm_checkout_release_after_clone() {
|
||||
local channel tag
|
||||
channel=$(_lm_channel "")
|
||||
if [ "$channel" = "beta" ]; then
|
||||
print_success "Beta channel: installing the newest code from main"
|
||||
return 0
|
||||
fi
|
||||
tag=$(_lm_newest_release_tag)
|
||||
if [ -z "$tag" ]; then
|
||||
print_warning "No release found; installing the newest code from main"
|
||||
return 0
|
||||
fi
|
||||
if git -c advice.detachedHead=false checkout --quiet --detach "${tag}^{commit}"; then
|
||||
print_success "Installing release $tag (stable channel)"
|
||||
else
|
||||
print_warning "Could not check out release $tag; installing the newest code from main"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# An existing checkout: move it forward along its channel, never backwards.
|
||||
# Returns 1 when it should be updated the way it always was (a fast-forward
|
||||
# pull of its branch): beta, or stable on a branch newer than every release.
|
||||
_lm_update_existing_checkout() {
|
||||
local channel tag head tag_sha
|
||||
channel=$(_lm_channel "config/config.json")
|
||||
if [ "$channel" = "beta" ]; then
|
||||
return 1
|
||||
fi
|
||||
if ! git fetch --quiet --tags --force origin >/dev/null 2>&1; then
|
||||
print_warning "Could not fetch release tags; keeping the current version"
|
||||
return 0
|
||||
fi
|
||||
tag=$(_lm_newest_release_tag)
|
||||
head=$(git rev-parse --verify --quiet HEAD 2>/dev/null || true)
|
||||
if [ -n "$tag" ] && [ -n "$head" ] && git merge-base --is-ancestor "$head" "$tag" 2>/dev/null; then
|
||||
tag_sha=$(git rev-parse --verify --quiet "${tag}^{commit}" 2>/dev/null || true)
|
||||
if [ "$head" = "$tag_sha" ]; then
|
||||
print_success "Already on the newest release, $tag"
|
||||
elif git -c advice.detachedHead=false checkout --quiet --detach "${tag}^{commit}"; then
|
||||
print_success "Updated to release $tag (stable channel)"
|
||||
else
|
||||
print_warning "Could not move to release $tag (local changes?); keeping the current version"
|
||||
fi
|
||||
return 0
|
||||
fi
|
||||
if git symbolic-ref --quiet HEAD >/dev/null 2>&1; then
|
||||
# Newer than the newest release (or no release yet): follow the branch
|
||||
# until a release includes this version, as updates do.
|
||||
return 1
|
||||
fi
|
||||
print_success "This checkout is newer than the newest release${tag:+ ($tag)}; leaving it as it is"
|
||||
return 0
|
||||
}
|
||||
# --- end release checkout helpers --------------------------------------------
|
||||
|
||||
# Main installation function
|
||||
main() {
|
||||
print_step "LED Matrix One-Shot Installation"
|
||||
@@ -292,7 +404,10 @@ main() {
|
||||
|
||||
# Try to safely update current branch first (fast-forward only to avoid unintended merges)
|
||||
PULL_SUCCESS=false
|
||||
if git pull --ff-only origin "$CURRENT_BRANCH" >/dev/null 2>&1; then
|
||||
# Stable: the newest release, if it contains this version.
|
||||
if _lm_update_existing_checkout; then
|
||||
PULL_SUCCESS=true
|
||||
elif git pull --ff-only origin "$CURRENT_BRANCH" >/dev/null 2>&1; then
|
||||
print_success "Repository updated successfully (branch: $CURRENT_BRANCH)"
|
||||
PULL_SUCCESS=true
|
||||
else
|
||||
@@ -323,10 +438,12 @@ main() {
|
||||
rm -rf "$REPO_DIR"
|
||||
print_success "Cloning repository..."
|
||||
retry git clone "$REPO_URL" "$REPO_DIR"
|
||||
(cd "$REPO_DIR" && _lm_checkout_release_after_clone) || print_warning "Could not choose a release; installing the newest code from main"
|
||||
fi
|
||||
else
|
||||
print_success "Cloning repository to $REPO_DIR..."
|
||||
retry git clone "$REPO_URL" "$REPO_DIR"
|
||||
(cd "$REPO_DIR" && _lm_checkout_release_after_clone) || print_warning "Could not choose a release; installing the newest code from main"
|
||||
fi
|
||||
|
||||
# Verify repository is accessible
|
||||
@@ -397,6 +514,7 @@ main() {
|
||||
sudo -E env TMPDIR=/tmp LEDMATRIX_ASSUME_YES=1 \
|
||||
LEDMATRIX_APT_UPDATED="${LEDMATRIX_APT_UPDATED:-0}" \
|
||||
LEDMATRIX_AUTO_UPDATE="${LEDMATRIX_AUTO_UPDATE:-}" \
|
||||
LEDMATRIX_CHANNEL="${LEDMATRIX_CHANNEL:-}" \
|
||||
bash ./first_time_install.sh -y </dev/null
|
||||
fi
|
||||
INSTALL_EXIT_CODE=$?
|
||||
|
||||
@@ -4,13 +4,14 @@ Alternative dependency installer that tries apt packages first,
|
||||
then falls back to pip with --break-system-packages
|
||||
"""
|
||||
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import warnings
|
||||
from collections import deque
|
||||
from pathlib import Path
|
||||
from typing import List, Tuple
|
||||
from typing import Dict, List, Tuple
|
||||
|
||||
# How many trailing lines of a failed command's output to keep for the
|
||||
# end-of-run failure summary. Keeps the root cause near the end of the log,
|
||||
@@ -81,6 +82,8 @@ def install_via_pip(package_name: str) -> Tuple[bool, str]:
|
||||
|
||||
Returns (success, output).
|
||||
"""
|
||||
# pip knows PIL as Pillow; the others are asked for by their own name.
|
||||
package_name = _dist_name(package_name)
|
||||
print(f"Installing {package_name} via pip...")
|
||||
success, output = _run([
|
||||
sys.executable, '-m', 'pip', 'install',
|
||||
@@ -99,26 +102,66 @@ IMPORT_NAME_MAP = {
|
||||
'freetype-py': 'freetype',
|
||||
}
|
||||
|
||||
# Minimum versions that must be met for an already-installed package to count
|
||||
# as satisfied. Debian Bookworm's python3-freetype is 2.3.0, below the
|
||||
# freetype-py>=2.5.1 pin in requirements.txt, so an import-only check would
|
||||
# wrongly skip the pip upgrade.
|
||||
MIN_VERSIONS = {
|
||||
'freetype-py': (2, 5, 1),
|
||||
# The packages above are keyed by what main() lists; these are the ones whose
|
||||
# pip distribution name differs from that key.
|
||||
DIST_NAME_MAP = {
|
||||
'PIL': 'Pillow',
|
||||
}
|
||||
|
||||
REQUIREMENTS_FILE = Path(__file__).resolve().parent.parent / 'web_interface' / 'requirements.txt'
|
||||
|
||||
|
||||
def _version_tuple(text: str) -> tuple:
|
||||
parts = []
|
||||
for part in text.split('.'):
|
||||
digits = ''.join(ch for ch in part if ch.isdigit())
|
||||
if not digits:
|
||||
break
|
||||
parts.append(int(digits))
|
||||
return tuple(parts)
|
||||
|
||||
|
||||
def _requirement_floors(path: Path = REQUIREMENTS_FILE) -> Dict[str, tuple]:
|
||||
"""``>=`` floors from a requirements file, keyed by lower-cased name.
|
||||
|
||||
The apt copies of these packages are older than the pins on both
|
||||
supported releases -- Bookworm ships Flask and Werkzeug 2.2.2, Pillow 9.4,
|
||||
requests 2.28, psutil 5.9, pytz 2022.7 and freetype-py 2.3; Trixie ships
|
||||
Flask 3.1.1, Werkzeug 3.1.3, Pillow 11.1 and requests 2.32 --
|
||||
so a package that merely imports is not enough. Read from the file rather
|
||||
than copied here so the two cannot drift.
|
||||
"""
|
||||
floors: Dict[str, tuple] = {}
|
||||
try:
|
||||
lines = path.read_text(encoding='utf-8').splitlines()
|
||||
except OSError:
|
||||
return floors
|
||||
for line in lines:
|
||||
match = re.match(r'\s*([A-Za-z0-9][A-Za-z0-9._-]*)[^#]*?>=\s*([0-9][0-9.]*)', line)
|
||||
if match:
|
||||
floors[match.group(1).lower()] = _version_tuple(match.group(2))
|
||||
return floors
|
||||
|
||||
|
||||
def _dist_name(package_name: str) -> str:
|
||||
return DIST_NAME_MAP.get(package_name, package_name)
|
||||
|
||||
|
||||
def _minimum_version(package_name: str) -> tuple:
|
||||
"""The required floor for ``package_name``, or () when there is none."""
|
||||
return MIN_VERSIONS.get(_dist_name(package_name).lower(), ())
|
||||
|
||||
|
||||
# Minimum versions that must be met for an already-installed package to count
|
||||
# as satisfied.
|
||||
MIN_VERSIONS = _requirement_floors()
|
||||
|
||||
|
||||
def _installed_version_tuple(dist_name: str) -> tuple:
|
||||
"""Return the installed distribution version as an int tuple, or () if unknown."""
|
||||
try:
|
||||
from importlib.metadata import version
|
||||
parts = []
|
||||
for part in version(dist_name).split('.'):
|
||||
digits = ''.join(ch for ch in part if ch.isdigit())
|
||||
if not digits:
|
||||
break
|
||||
parts.append(int(digits))
|
||||
return tuple(parts)
|
||||
return _version_tuple(version(dist_name))
|
||||
except Exception:
|
||||
return ()
|
||||
|
||||
@@ -134,9 +177,9 @@ def check_package_installed(package_name: str) -> bool:
|
||||
__import__(import_name)
|
||||
except ImportError:
|
||||
return False
|
||||
minimum = MIN_VERSIONS.get(package_name)
|
||||
minimum = _minimum_version(package_name)
|
||||
if minimum:
|
||||
installed = _installed_version_tuple(package_name)
|
||||
installed = _installed_version_tuple(_dist_name(package_name))
|
||||
if not installed or installed < minimum:
|
||||
print(f"{package_name} is installed but below the required "
|
||||
f"{'.'.join(map(str, minimum))}; will upgrade via pip")
|
||||
@@ -188,10 +231,11 @@ def main():
|
||||
continue
|
||||
|
||||
# Try apt first, then pip. An apt install only counts if it also
|
||||
# satisfies any minimum version (Debian's python3-freetype can be
|
||||
# older than the freetype-py pin), otherwise fall through to pip.
|
||||
# satisfies the requirements floor (the apt copies of most of these
|
||||
# are older than the pins on both Bookworm and Trixie), otherwise
|
||||
# fall through to pip.
|
||||
ok, apt_output = install_via_apt(package)
|
||||
if ok and package in MIN_VERSIONS and not check_package_installed(package):
|
||||
if ok and _minimum_version(package) and not check_package_installed(package):
|
||||
ok = False
|
||||
apt_output = f"apt version of {package} is below the required minimum"
|
||||
if not ok:
|
||||
|
||||
@@ -438,6 +438,7 @@ def main(argv=None) -> int:
|
||||
flush_interval=float("inf"),
|
||||
info=display._frame_timing_info(), # pylint: disable=protected-access
|
||||
refresh_hz=idle_hz,
|
||||
gc_monitor=frame_timing.install_gc_monitor(),
|
||||
)
|
||||
recorder.scrolling_now = display._scrolling_now # pylint: disable=protected-access
|
||||
display.frame_timing = recorder
|
||||
|
||||
@@ -15,7 +15,8 @@ The updater leaves data/auto_update_pending.json:
|
||||
|
||||
{"status": "pending", "old_head": ..., "new_head": ...,
|
||||
"old_ref": "main" | "" (detached) | absent (older updaters),
|
||||
"display_was_active": bool, "dependency_failures": [...], "created_at": ...}
|
||||
"display_was_active": bool, "dependency_failures": [...],
|
||||
"units_refreshed": bool (absent from older updaters), "created_at": ...}
|
||||
|
||||
This moves its status to "verifying" and then to one of "success",
|
||||
"rolled_back" or "rollback_failed", with "reason" and "detail" saying why.
|
||||
@@ -74,6 +75,10 @@ BASH_CANDIDATES = ('/usr/bin/bash', '/bin/bash')
|
||||
#: ...and, like it, moves to the next one only when sudo refused the command
|
||||
#: line (permission_utils.SUDO_REFUSAL_PHRASES), never after pip itself ran.
|
||||
SUDO_REFUSAL_PHRASES = ('a password is required', 'is not allowed to run', 'no tty present')
|
||||
#: The root-owned helper that installed the update's systemd units
|
||||
#: (web_interface/unit_refresh.py); ``--restore`` puts the previous ones back.
|
||||
REFRESH_UNITS_PATH = '/usr/local/sbin/ledmatrix-refresh-units'
|
||||
UNIT_RESTORE_TIMEOUT_SECONDS = 90
|
||||
|
||||
#: The longest one health check can take: restart and wait, roll back
|
||||
#: (diff, reset, reinstalls), restart and wait again. A wait's last poll can
|
||||
@@ -81,7 +86,8 @@ SUDO_REFUSAL_PHRASES = ('a password is required', 'is not allowed to run', 'no t
|
||||
_WAIT_WORST_SECONDS = (HEALTH_TIMEOUT_SECONDS + STABLE_SECONDS + WEB_CHECK_TIMEOUT_SECONDS
|
||||
+ 2 * SYSTEMCTL_QUERY_TIMEOUT_SECONDS + POLL_SECONDS)
|
||||
WORST_CASE_SECONDS = (2 * (2 * RESTART_TIMEOUT_SECONDS + _WAIT_WORST_SECONDS)
|
||||
+ GIT_TIMEOUT_SECONDS + GIT_RESET_TIMEOUT_SECONDS + PIP_BUDGET_SECONDS)
|
||||
+ GIT_TIMEOUT_SECONDS + GIT_RESET_TIMEOUT_SECONDS + UNIT_RESTORE_TIMEOUT_SECONDS
|
||||
+ PIP_BUDGET_SECONDS)
|
||||
|
||||
#: What a command that could not run at all reports: its callers only read
|
||||
#: these three fields, the same ones a completed subprocess has.
|
||||
@@ -300,12 +306,26 @@ class Verifier:
|
||||
if result.returncode != 0:
|
||||
return False, (f'"git reset --hard {old}" failed: '
|
||||
f'{(result.stderr or result.stdout or "").strip()}')
|
||||
notes = []
|
||||
# The update also installed its own systemd units: put the previous
|
||||
# ones back before anything restarts onto the rolled-back code.
|
||||
if pending.get('units_refreshed') and not self.restore_units():
|
||||
notes.append('restoring the previous service settings failed; run '
|
||||
'"sudo ./scripts/install/install_service.sh" in the LEDMatrix folder')
|
||||
deadline = self.clock() + PIP_BUDGET_SECONDS
|
||||
failed = [rel for rel in requirements if not self.install_requirements(rel, deadline)]
|
||||
if failed:
|
||||
return True, ('reinstalling the previous dependencies from ' + ', '.join(failed)
|
||||
+ ' failed; run Install Base Requirements from the Tools tab')
|
||||
return True, ''
|
||||
notes.append('reinstalling the previous dependencies from ' + ', '.join(failed)
|
||||
+ ' failed; run Install Base Requirements from the Tools tab')
|
||||
return True, '; '.join(notes)
|
||||
|
||||
def restore_units(self):
|
||||
"""Reinstall the systemd units the update replaced. True on success."""
|
||||
result = self._run(['sudo', '-n', REFRESH_UNITS_PATH, '--restore'],
|
||||
timeout=UNIT_RESTORE_TIMEOUT_SECONDS)
|
||||
if result.returncode != 0:
|
||||
self.log(f'restoring the previous systemd units failed: {(result.stderr or "").strip()}')
|
||||
return result.returncode == 0
|
||||
|
||||
# -- the check itself -------------------------------------------------
|
||||
|
||||
|
||||
+1
-1
@@ -4,5 +4,5 @@ LEDMatrix Display System
|
||||
Core source package for the LED Matrix Display project.
|
||||
"""
|
||||
|
||||
__version__ = "3.8.0"
|
||||
__version__ = "3.8.1"
|
||||
|
||||
|
||||
+34
-12
@@ -28,6 +28,7 @@ freetype.Face, so it drops straight into DisplayManager.draw_text().
|
||||
"""
|
||||
|
||||
import logging
|
||||
import weakref
|
||||
from collections import OrderedDict
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Dict, List, Optional, Sequence, Tuple, Union
|
||||
@@ -332,9 +333,10 @@ class LayoutContext:
|
||||
# a plugin fitting changing text (a live game clock, a ticker) on a
|
||||
# 24/7 service would otherwise grow this without bound.
|
||||
self._fit_cache: "OrderedDict[Any, FitResult]" = OrderedDict()
|
||||
# LRU-bounded (images are big). Entries hold a strong reference to
|
||||
# the source image when keyed by id() so the id can't be recycled
|
||||
# out from under the cache.
|
||||
# LRU-bounded (images are big). An id()-keyed entry watches its
|
||||
# source image through a weak reference and is dropped when the
|
||||
# source is freed (see fit_image), so the id can't be recycled out
|
||||
# from under the cache and the cache never keeps the source alive.
|
||||
self._image_cache: "OrderedDict[Any, Tuple[Any, Any]]" = OrderedDict()
|
||||
|
||||
_IMAGE_CACHE_MAX = 64
|
||||
@@ -536,8 +538,15 @@ class LayoutContext:
|
||||
cached per (image, box size, options) for this panel size.
|
||||
|
||||
Prefer a stable ``cache_key`` (e.g. "logo:KC") for images that get
|
||||
reloaded — the default id()-based key is safe (the entry pins the
|
||||
source image) but misses across reloads of the same content.
|
||||
reloaded — the default id()-based key misses across reloads of the
|
||||
same content.
|
||||
|
||||
An id()-keyed entry lives only as long as its source image: it holds
|
||||
a weak reference and is dropped when the source is freed. It used to
|
||||
pin the source instead, so a plugin passing a freshly loaded image
|
||||
each frame (``draw_image(Image.open(path), box)``, the documented
|
||||
one-liner) never hit and kept the last 64 sources alive — ~64MB for
|
||||
500x500 RGBA team logos, the median size under assets/sports.
|
||||
"""
|
||||
from src.adaptive_images import fit_image as _fit_image
|
||||
|
||||
@@ -547,18 +556,31 @@ class LayoutContext:
|
||||
key = ("image", identity, img.size, box_w, box_h, mode,
|
||||
crop_to_ink, anchor, resample_name, upscale)
|
||||
|
||||
cached = self._image_cache.get(key)
|
||||
if cached is not None:
|
||||
self._image_cache.move_to_end(key)
|
||||
cache = self._image_cache
|
||||
cached = cache.get(key)
|
||||
# An id()-keyed hit must still be this very image; the callback below
|
||||
# normally removes a dead source's entry before its id can recur.
|
||||
if cached is not None and (cache_key is not None or cached[1]() is img):
|
||||
cache.move_to_end(key)
|
||||
return cached[0]
|
||||
|
||||
result = _fit_image(img, (box_w, box_h), mode=mode,
|
||||
crop_to_ink=crop_to_ink, anchor=anchor,
|
||||
resample=resample, upscale=upscale)
|
||||
# Pin the source only for id()-keyed entries (see docstring).
|
||||
self._image_cache[key] = (result, img if cache_key is None else None)
|
||||
while len(self._image_cache) > self._IMAGE_CACHE_MAX:
|
||||
self._image_cache.popitem(last=False)
|
||||
source = None
|
||||
if cache_key is None:
|
||||
def _forget(ref: Any, key: Any = key) -> None:
|
||||
entry = cache.get(key)
|
||||
if entry is not None and entry[1] is ref:
|
||||
cache.pop(key, None)
|
||||
try:
|
||||
source = weakref.ref(img, _forget)
|
||||
except TypeError:
|
||||
# Not weak-referenceable: pin it, as before.
|
||||
source = lambda img=img: img # noqa: E731
|
||||
cache[key] = (result, source)
|
||||
while len(cache) > self._IMAGE_CACHE_MAX:
|
||||
cache.popitem(last=False)
|
||||
return result
|
||||
|
||||
# ---- text utilities ------------------------------------------------
|
||||
|
||||
@@ -27,6 +27,13 @@ from concurrent.futures import ThreadPoolExecutor
|
||||
import pytz
|
||||
from src.cache_manager import CacheManager
|
||||
from src.common.json_body import response_json
|
||||
from src.common.fetch_service import (
|
||||
current_plugin_id,
|
||||
fetch_get,
|
||||
get_fetch_service,
|
||||
plugin_scope,
|
||||
share_connection_pool,
|
||||
)
|
||||
from src.common.espn_dates import (
|
||||
RANGE_RETRY_SECONDS,
|
||||
_note_range_rejected,
|
||||
@@ -78,6 +85,9 @@ class FetchRequest:
|
||||
commit_claimed: bool = False
|
||||
result: Optional[Any] = None
|
||||
error: Optional[str] = None
|
||||
# The plugin that submitted the request, so the fetch service counts the
|
||||
# worker's requests against it (fetch_service, caller identity).
|
||||
owner: Optional[str] = None
|
||||
|
||||
@dataclass
|
||||
class FetchResult:
|
||||
@@ -119,6 +129,12 @@ class _ConnectionRetryingSession:
|
||||
def __init__(self, session):
|
||||
self._session = session
|
||||
|
||||
@property
|
||||
def fetch_identity_session(self):
|
||||
"""The wrapped Session, whose headers and adapter the fetch service
|
||||
reads to key this request (src/common/fetch_service.py)."""
|
||||
return self._session
|
||||
|
||||
def get(self, *args, **kwargs):
|
||||
for attempt in range(self.ATTEMPTS):
|
||||
try:
|
||||
@@ -196,9 +212,12 @@ class BackgroundDataService:
|
||||
# connection errors three times, a dead network cost up to 16
|
||||
# connection attempts per request and held one of the few worker
|
||||
# threads for all of them.
|
||||
#
|
||||
# The adapter is the fetch service's shared no-retry one: the same
|
||||
# max_retries=0, with the connection pool shared with the other core
|
||||
# sessions that do not retry (the odds managers).
|
||||
self.session = requests.Session()
|
||||
self.session.mount('http://', requests.adapters.HTTPAdapter(max_retries=0))
|
||||
self.session.mount('https://', requests.adapters.HTTPAdapter(max_retries=0))
|
||||
share_connection_pool(self.session, max_retries=0)
|
||||
|
||||
# Default headers: core's shared set (real User-Agent, no hand-set
|
||||
# Accept-Encoding) -- see src/common/api_helper.py.
|
||||
@@ -299,6 +318,10 @@ class BackgroundDataService:
|
||||
if url.split('?', 1)[0].rstrip('/').endswith('/scoreboard'):
|
||||
params = clamp_espn_limit(params)
|
||||
|
||||
# Who asked, resolved on the submitting thread: the worker thread
|
||||
# runs no plugin code, so it could not tell (fetch_service).
|
||||
owner = current_plugin_id()
|
||||
|
||||
# Create fetch request
|
||||
request = FetchRequest(
|
||||
id=request_id,
|
||||
@@ -311,7 +334,8 @@ class BackgroundDataService:
|
||||
timeout=timeout or self.request_timeout,
|
||||
max_retries=max_retries,
|
||||
priority=priority,
|
||||
callback=callback
|
||||
callback=callback,
|
||||
owner=owner,
|
||||
)
|
||||
|
||||
with self._lock:
|
||||
@@ -330,6 +354,7 @@ class BackgroundDataService:
|
||||
self.stats['deduplicated_requests'] = (
|
||||
self.stats.get('deduplicated_requests', 0) + 1
|
||||
)
|
||||
get_fetch_service().note_merged(url, owner)
|
||||
logger.info(
|
||||
"Joined in-flight fetch %s for %s (cache_key=%s) instead of "
|
||||
"starting a duplicate", existing_id, sport, cache_key
|
||||
@@ -357,6 +382,11 @@ class BackgroundDataService:
|
||||
Returns:
|
||||
Fetch result with data or error information
|
||||
"""
|
||||
with plugin_scope(request.owner):
|
||||
return self._fetch_data_worker_scoped(request)
|
||||
|
||||
def _fetch_data_worker_scoped(self, request: FetchRequest) -> FetchResult:
|
||||
"""_fetch_data_worker's body, run with the submitter as the caller."""
|
||||
start_time = time.time()
|
||||
result = FetchResult(request_id=request.id, success=False, retry_count=request.retry_count)
|
||||
|
||||
@@ -621,8 +651,14 @@ class BackgroundDataService:
|
||||
|
||||
for attempt in range(request.max_retries + 1):
|
||||
try:
|
||||
response = self.session.get(
|
||||
# Not shared with an identical request in flight: this
|
||||
# service cancels and replaces fetches, and a replacement
|
||||
# must not join the one it replaced. Its own cache_key
|
||||
# dedup already merges what should be merged.
|
||||
response = fetch_get(
|
||||
self.session,
|
||||
request.url,
|
||||
share_in_flight=False,
|
||||
params=request.params,
|
||||
headers=request.headers,
|
||||
timeout=request.timeout
|
||||
|
||||
@@ -213,8 +213,9 @@ def list_installed_plugins(project_root: Path) -> List[Dict[str, Any]]:
|
||||
The plugins are the ``manifest.json`` files in the configured plugin
|
||||
directory (see :func:`_plugins_directory`), with the manifest's version;
|
||||
``enabled`` is config.json's flag by the display's rule (a missing flag
|
||||
is disabled). A restore reinstalls every listed plugin and takes enabled
|
||||
state from the restored config.json, so ``enabled`` is informational.
|
||||
is disabled). A restore installs each listed plugin that is missing and
|
||||
takes enabled state from the restored config.json, so ``enabled`` is
|
||||
informational.
|
||||
|
||||
``data/plugin_state.json`` is not read: it only ever repeated config's
|
||||
enabled flags and the manifests' versions, and is retired (nothing
|
||||
|
||||
+33
-14
@@ -19,6 +19,8 @@ import json
|
||||
from typing import Dict, Any, Optional, List, cast
|
||||
|
||||
from src.common.api_helper import DEFAULT_HTTP_HEADERS
|
||||
from src.common.fetch_service import fetch_get, share_connection_pool
|
||||
from src.common.json_body import response_json
|
||||
|
||||
|
||||
|
||||
@@ -59,7 +61,13 @@ class BaseOddsManager:
|
||||
# Deliberately no retry adapter, unlike api_helper: retries multiply
|
||||
# request_timeout, which is set to 5s precisely to stay inside that
|
||||
# budget. One try, then the cooldown below.
|
||||
#
|
||||
# Every scoreboard league manager builds one of these, so the session
|
||||
# mounts the fetch service's shared no-retry adapter: the same single
|
||||
# try, over one connection pool per host for all of them instead of
|
||||
# one pool per instance.
|
||||
self.session = requests.Session()
|
||||
share_connection_pool(self.session, max_retries=0)
|
||||
self.session.headers.update(DEFAULT_HTTP_HEADERS)
|
||||
|
||||
# Configuration with defaults
|
||||
@@ -139,7 +147,7 @@ class BaseOddsManager:
|
||||
if _is_no_odds_marker(cached_data):
|
||||
self.logger.debug("Cached no-odds marker for %s", cache_key)
|
||||
return None
|
||||
self.logger.debug(f"Using cached odds from ESPN for {cache_key}")
|
||||
self.logger.debug("Using cached odds from ESPN for %s", cache_key)
|
||||
return cached_data
|
||||
|
||||
if time.monotonic() < self._skip_network_until:
|
||||
@@ -152,7 +160,7 @@ class BaseOddsManager:
|
||||
self._skip_network_until - time.monotonic())
|
||||
return None
|
||||
|
||||
self.logger.debug(f"Cache miss - fetching fresh odds from ESPN for {cache_key}")
|
||||
self.logger.debug("Cache miss - fetching fresh odds from ESPN for %s", cache_key)
|
||||
|
||||
try:
|
||||
# Map league names to ESPN API format
|
||||
@@ -166,23 +174,30 @@ class BaseOddsManager:
|
||||
|
||||
espn_league = league_mapping.get(league, league)
|
||||
url = f"{self.base_url}/{sport}/leagues/{espn_league}/events/{event_id}/competitions/{event_id}/odds"
|
||||
self.logger.debug(f"Requesting odds from URL: {url}")
|
||||
self.logger.debug("Requesting odds from URL: %s", url)
|
||||
|
||||
response = self.session.get(url, timeout=self.request_timeout)
|
||||
# The response cache may answer only inside this caller's own
|
||||
# interval, the age at which its cached odds expire anyway.
|
||||
response = fetch_get(self.session, url, timeout=self.request_timeout,
|
||||
cache_max_age=interval)
|
||||
response.raise_for_status()
|
||||
raw_data = response.json()
|
||||
raw_data = response_json(response)
|
||||
|
||||
self._skip_network_until = 0.0 # reachable again
|
||||
|
||||
self.logger.debug(f"Received raw odds data from ESPN: {json.dumps(raw_data, indent=2)}")
|
||||
# Guarded, not just %-style: the json.dumps argument would still be
|
||||
# built for every response with DEBUG off.
|
||||
if self.logger.isEnabledFor(logging.DEBUG):
|
||||
self.logger.debug("Received raw odds data from ESPN: %s",
|
||||
json.dumps(raw_data, indent=2))
|
||||
|
||||
odds_data = self._extract_espn_data(raw_data)
|
||||
if odds_data:
|
||||
self.logger.debug(f"Successfully extracted odds data: {odds_data}")
|
||||
self.logger.debug("Successfully extracted odds data: %s", odds_data)
|
||||
self.cache_manager.set(cache_key, odds_data, ttl=interval)
|
||||
self.logger.debug(f"Saved odds data to cache for {cache_key} with TTL {interval}s")
|
||||
self.logger.debug("Saved odds data to cache for %s with TTL %ss", cache_key, interval)
|
||||
else:
|
||||
self.logger.debug(f"No odds data available for {cache_key}")
|
||||
self.logger.debug("No odds data available for %s", cache_key)
|
||||
# Cache the absence too, so the game is not re-requested
|
||||
# on every update until the interval passes.
|
||||
self.cache_manager.set(cache_key, {"no_odds": True}, ttl=interval)
|
||||
@@ -216,12 +231,12 @@ class BaseOddsManager:
|
||||
Returns:
|
||||
Formatted odds data dictionary or None
|
||||
"""
|
||||
self.logger.debug(f"Extracting ESPN odds data. Data keys: {list(data.keys())}")
|
||||
self.logger.debug("Extracting ESPN odds data. Data keys: %s", list(data.keys()))
|
||||
|
||||
if "items" in data and data["items"]:
|
||||
self.logger.debug(f"Found {len(data['items'])} items in odds data")
|
||||
self.logger.debug("Found %d items in odds data", len(data['items']))
|
||||
item = data["items"][0]
|
||||
self.logger.debug(f"First item keys: {list(item.keys())}")
|
||||
self.logger.debug("First item keys: %s", list(item.keys()))
|
||||
|
||||
# The ESPN API returns odds data directly in the item, not in a
|
||||
# providers array. ESPN sends explicit JSON nulls for absent
|
||||
@@ -244,13 +259,17 @@ class BaseOddsManager:
|
||||
.get("pointSpread") or {}).get("value")
|
||||
}
|
||||
}
|
||||
self.logger.debug(f"Returning extracted odds data: {json.dumps(extracted_data, indent=2)}")
|
||||
if self.logger.isEnabledFor(logging.DEBUG):
|
||||
self.logger.debug("Returning extracted odds data: %s",
|
||||
json.dumps(extracted_data, indent=2))
|
||||
return extracted_data
|
||||
|
||||
# Check if this is a valid empty response or an unexpected structure
|
||||
if "count" in data and data["count"] == 0 and "items" in data and data["items"] == []:
|
||||
# This is a valid empty response - no odds available for this game
|
||||
self.logger.debug(f"No odds available for this game. Response: {json.dumps(data, indent=2)}")
|
||||
if self.logger.isEnabledFor(logging.DEBUG):
|
||||
self.logger.debug("No odds available for this game. Response: %s",
|
||||
json.dumps(data, indent=2))
|
||||
return None
|
||||
else:
|
||||
# This is an unexpected response structure
|
||||
|
||||
Vendored
+221
-42
@@ -4,6 +4,7 @@ Disk Cache
|
||||
Handles persistent disk-based caching with atomic writes and error recovery.
|
||||
"""
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
@@ -14,7 +15,7 @@ import tempfile
|
||||
import logging
|
||||
import threading
|
||||
import zlib
|
||||
from typing import Dict, Any, Optional, Protocol
|
||||
from typing import Dict, Any, Optional, Protocol, Tuple
|
||||
from datetime import datetime
|
||||
|
||||
from src.common.path_safety import safe_path_component
|
||||
@@ -31,6 +32,35 @@ except ImportError: # pragma: no cover - exercised on hosts without the wheel
|
||||
# useful, and a half-written file was never useful.
|
||||
_ORPHAN_TEMP_MAX_AGE_SECONDS = 3600
|
||||
|
||||
# Longest key, in UTF-8 bytes, used verbatim as a filename stem. ext4 caps a
|
||||
# name at 255 bytes and set()'s temp file is ".<stem>.json.<8 random>", 15
|
||||
# bytes longer than the stem, so anything near the cap could never be written:
|
||||
# the calendar plugin's key joins every calendar id and passed 300 bytes on a
|
||||
# real install, failing every write with ENAMETOOLONG. Longer keys keep this
|
||||
# many bytes as a readable prefix and end in a hash of the whole key.
|
||||
_MAX_KEY_FILENAME_BYTES = 200
|
||||
_KEY_HASH_CHARS = 16
|
||||
|
||||
|
||||
def _filename_stem(key: str) -> str:
|
||||
"""The filename stem for a key that is already a safe path component.
|
||||
|
||||
Short keys are used as they are, so every file already on disk keeps its
|
||||
name. A long one becomes its first bytes plus a hash of the full key: the
|
||||
prefix keeps the stem recognisable (and keeps the data-type words that
|
||||
cleanup's retention lookup reads from it), the hash keeps two keys that
|
||||
share a long prefix apart. The result is itself short, so a stem read back
|
||||
from a filename -- which is how the web UI names a key it deletes -- maps to
|
||||
the same file.
|
||||
"""
|
||||
encoded = key.encode('utf-8')
|
||||
if len(encoded) <= _MAX_KEY_FILENAME_BYTES:
|
||||
return key
|
||||
digest = hashlib.sha256(encoded).hexdigest()[:_KEY_HASH_CHARS]
|
||||
keep = _MAX_KEY_FILENAME_BYTES - _KEY_HASH_CHARS - 1
|
||||
prefix = encoded[:keep].decode('utf-8', errors='ignore')
|
||||
return f"{prefix}-{digest}"
|
||||
|
||||
|
||||
|
||||
class CacheStrategyProtocol(Protocol):
|
||||
@@ -111,18 +141,91 @@ _HEAD_RE = re.compile(
|
||||
)
|
||||
|
||||
|
||||
def _stale_from_head(head: bytes, max_age: Optional[int], now: float) -> bool:
|
||||
# UNCHANGED RE-SAVES: THE FILE'S MTIME CARRIES THE NEWER TIMESTAMP
|
||||
# ----------------------------------------------------------------
|
||||
# Plugins re-save unchanged API data every update cycle, and every one of
|
||||
# those saves was a full rewrite on the SD card. DiskCache.set skips the write
|
||||
# when the payload matches the last one it wrote for the key -- but
|
||||
# CacheManager.set stamps each record with time.time(), so for set() the
|
||||
# payload never matched and the skip never fired.
|
||||
#
|
||||
# The digest now leaves out a header-first record's timestamp, so an unchanged
|
||||
# set() is skipped. What the skip must not do is make the record look older
|
||||
# than it is: the timestamp inside the file is from the last real write, and
|
||||
# a reader in another process (the web interface, with memory_ttl=0) or after
|
||||
# a restart would call fresh data stale. So the newer timestamp goes where it
|
||||
# costs no data write -- the file's mtime -- and readers take a record's age
|
||||
# from the newer of the two. The invariant that makes that safe:
|
||||
#
|
||||
# a file's mtime is the timestamp of the newest record saved for its key
|
||||
#
|
||||
# real write mtime is set to the record's own timestamp, so a record saved
|
||||
# with an old timestamp (data as of some earlier time) cannot
|
||||
# borrow freshness from the moment it hit the disk
|
||||
# skip mtime is set to the skipped record's timestamp -- exactly what
|
||||
# a rewrite would have stored, without the rewrite
|
||||
#
|
||||
# Readers of the on-disk timestamp, all of which go through _effective_timestamp:
|
||||
# DiskCache.get (the header check and the full parse; it also returns the
|
||||
# record with 'timestamp' set to the effective value, so CacheManager.get's
|
||||
# max_age path, the memory tier hydrated from disk, and any plugin reading
|
||||
# record['timestamp'] all see it). Readers that use mtime alone already see the
|
||||
# newer value: the retention sweep below, CacheManager.list_cache_files (the
|
||||
# web UI's cache list). Nothing else opens cache files: web_interface and
|
||||
# scripts reach them only through CacheManager.
|
||||
#
|
||||
# Something other than this class can also move an mtime forward -- a copy
|
||||
# without -p, an rsync without -t, a `touch`. (backup_manager.py does not
|
||||
# back up or restore the cache directory, so the in-tree restore cannot.) That
|
||||
# must not make old data fresh, so the lift is bounded: a reader never takes
|
||||
# the mtime as more than _MAX_TIMESTAMP_LIFT past the embedded timestamp, and
|
||||
# set() rewrites the file for real once a skip would need more than that, so
|
||||
# an honest lift never reaches the bound. A file copied a day after it was
|
||||
# written therefore reads at most an hour fresher than its contents say, and a
|
||||
# 30-second live-score record from yesterday stays stale. CacheManager.set
|
||||
# records written before this change have mtime == write time == embedded
|
||||
# timestamp, give or take the write itself, and read exactly as before; a
|
||||
# file an older version wrote or touched later than its embedded timestamp
|
||||
# says reads at most the same hour fresher, once, until it is next saved.
|
||||
|
||||
#: Longest a skipped write may stand in for a real one, and so the furthest a
|
||||
#: file's mtime is ever trusted past the record's own timestamp. Unchanged data
|
||||
#: is rewritten at least this often, at most once an hour per key instead of
|
||||
#: once per update cycle.
|
||||
_MAX_TIMESTAMP_LIFT = 3600.0
|
||||
|
||||
|
||||
def _record_timestamp(value: Any) -> Optional[float]:
|
||||
"""A record's timestamp as a finite float, or None if it has no usable one."""
|
||||
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
||||
return None
|
||||
value = float(value)
|
||||
return value if math.isfinite(value) else None
|
||||
|
||||
|
||||
def _effective_timestamp(embedded: float, mtime: Optional[float]) -> float:
|
||||
"""When a record was last saved: its timestamp, or the file's mtime if a
|
||||
later unchanged save moved that forward -- never by more than
|
||||
_MAX_TIMESTAMP_LIFT. See "UNCHANGED RE-SAVES" above."""
|
||||
if mtime is None:
|
||||
return embedded
|
||||
return max(embedded, min(mtime, embedded + _MAX_TIMESTAMP_LIFT))
|
||||
|
||||
|
||||
def _stale_from_head(head: bytes, max_age: Optional[int], now: float,
|
||||
mtime: Optional[float] = None) -> bool:
|
||||
"""True when a record's header alone shows it has expired.
|
||||
|
||||
Mirrors the expiry rule in DiskCache.get: a per-entry ttl wins over the
|
||||
caller's max_age, and no limit at all means never stale. False whenever the
|
||||
header cannot be read, so the full parse decides as it always did.
|
||||
header cannot be read, so the full parse decides as it always did. ``mtime``
|
||||
is the file's, which may carry a newer save than the header does.
|
||||
"""
|
||||
match = _HEAD_RE.match(head)
|
||||
if not match:
|
||||
return False
|
||||
try:
|
||||
timestamp = float(match.group(1))
|
||||
timestamp = _effective_timestamp(float(match.group(1)), mtime)
|
||||
limit = max_age
|
||||
if match.group(2) is not None:
|
||||
ttl = float(match.group(2))
|
||||
@@ -179,7 +282,7 @@ else:
|
||||
# --------------------------------------------
|
||||
# The display service runs as root and the web interface as the installing
|
||||
# user, and the web interface reads records only the display writes
|
||||
# (display_current_state, display_on_demand_state, plugin_metrics:*). Files are
|
||||
# (display_current_state, display_on_demand_state, plugin_metrics_snapshot). Files are
|
||||
# written 0660, so the web interface can read one only through its group.
|
||||
#
|
||||
# The installers rely on the directory's setgid bit to set that group. That is
|
||||
@@ -248,11 +351,14 @@ class DiskCache:
|
||||
self.cache_dir = cache_dir
|
||||
self.logger = logger or logging.getLogger(__name__)
|
||||
self._lock = threading.Lock()
|
||||
# key -> adler32 of the last payload successfully written to the
|
||||
# primary cache path; lets set() skip rewriting identical data
|
||||
# (per-process only — worst case another process rewrites, never
|
||||
# a missed write). Guarded by _lock.
|
||||
self._write_digests: Dict[str, int] = {}
|
||||
# key -> what set() last put at the primary cache path: the adler32 of
|
||||
# the payload (less a header-first timestamp), the timestamp the file
|
||||
# holds (None for records without one), and the file's inode and size.
|
||||
# Lets set() skip rewriting identical data. Per-process only, and the
|
||||
# inode/size check means another process's write is never mistaken
|
||||
# for ours -- worst case a redundant write, never a missed one.
|
||||
# Guarded by _lock.
|
||||
self._write_digests: Dict[str, Tuple[int, Optional[float], int, int]] = {}
|
||||
|
||||
def get_cache_path(self, key: str) -> Optional[str]:
|
||||
"""
|
||||
@@ -267,6 +373,8 @@ class DiskCache:
|
||||
derives them), so rejecting anything with a path component turns
|
||||
away only inputs that could never have been written here.
|
||||
|
||||
A key too long to be a filename is shortened by _filename_stem.
|
||||
|
||||
Args:
|
||||
key: Cache key
|
||||
|
||||
@@ -280,7 +388,7 @@ class DiskCache:
|
||||
if safe_key is None:
|
||||
self.logger.warning("Rejected unsafe cache key %r", key)
|
||||
return None
|
||||
return os.path.join(self.cache_dir, f"{safe_key}.json")
|
||||
return os.path.join(self.cache_dir, f"{_filename_stem(safe_key)}.json")
|
||||
|
||||
def get(self, key: str, max_age: Optional[int] = 300) -> Optional[Dict[str, Any]]:
|
||||
"""
|
||||
@@ -301,32 +409,41 @@ class DiskCache:
|
||||
try:
|
||||
with self._lock:
|
||||
with open(cache_path, 'rb') as f:
|
||||
# The open file's mtime, not the path's: the file a skipped
|
||||
# write touched is the one being read.
|
||||
mtime = os.fstat(f.fileno()).st_mtime
|
||||
# Decide staleness from the header before paying for the
|
||||
# parse. A stale read is the common case for the biggest
|
||||
# records (a season schedule is re-fetched when its cache
|
||||
# expires), and parsing 53MB to throw it away held the GIL
|
||||
# for ~1.8s -- a visible freeze on the panel.
|
||||
if _stale_from_head(f.read(_HEAD_BYTES), max_age, time.time()):
|
||||
if _stale_from_head(f.read(_HEAD_BYTES), max_age, time.time(), mtime):
|
||||
return None
|
||||
f.seek(0)
|
||||
record = _loads(f.read())
|
||||
|
||||
# Determine record timestamp (prefer embedded, else file mtime)
|
||||
|
||||
# Determine record timestamp: the embedded one, moved forward by a
|
||||
# later unchanged save if there was one (see "UNCHANGED RE-SAVES"),
|
||||
# else the file mtime.
|
||||
record_ts = None
|
||||
if isinstance(record, dict):
|
||||
record_ts = record.get('timestamp')
|
||||
if record_ts is None:
|
||||
try:
|
||||
record_ts = os.path.getmtime(cache_path)
|
||||
except OSError:
|
||||
record_ts = None
|
||||
|
||||
if record_ts is not None:
|
||||
try:
|
||||
record_ts = float(record_ts)
|
||||
except (TypeError, ValueError):
|
||||
record_ts = None
|
||||
|
||||
record_ts = mtime
|
||||
else:
|
||||
embedded_ts = _record_timestamp(record_ts)
|
||||
if embedded_ts is None:
|
||||
try:
|
||||
record_ts = float(record_ts)
|
||||
except (TypeError, ValueError):
|
||||
record_ts = None
|
||||
else:
|
||||
record_ts = _effective_timestamp(embedded_ts, mtime)
|
||||
if record_ts != embedded_ts:
|
||||
# Hand the record back as a rewrite would have left it,
|
||||
# so callers that age it themselves agree with us.
|
||||
record['timestamp'] = record_ts
|
||||
|
||||
now = time.time()
|
||||
|
||||
# An explicit per-entry ttl wins over the caller's max_age. The
|
||||
@@ -403,7 +520,12 @@ class DiskCache:
|
||||
self.logger.warning("Cache data for key '%s' not serializable: %s", key, e)
|
||||
return
|
||||
|
||||
digest = zlib.adler32(payload)
|
||||
timestamp = _record_timestamp(data.get('timestamp')) if isinstance(data, dict) else None
|
||||
# A header-first record (CacheManager.set's layout) is compared without
|
||||
# its timestamp, which differs on every save; see "UNCHANGED RE-SAVES".
|
||||
# Any other layout is compared whole, as before.
|
||||
head = _HEAD_RE.match(payload) if timestamp is not None else None
|
||||
digest = zlib.adler32(memoryview(payload)[head.end(1):] if head else payload)
|
||||
|
||||
try:
|
||||
# Atomic write to avoid partial/corrupt files
|
||||
@@ -411,16 +533,10 @@ class DiskCache:
|
||||
# Skip the disk entirely when this exact payload was already
|
||||
# written for this key (plugins re-save unchanged API data
|
||||
# every update cycle — each write is real SD-card wear).
|
||||
# Refresh the file mtime so records that rely on it for TTL
|
||||
# (no embedded 'timestamp') don't expire early; a metadata
|
||||
# touch is journal-cheap compared to rewriting the data.
|
||||
if self._write_digests.get(key) == digest:
|
||||
try:
|
||||
os.utime(cache_path, None)
|
||||
return
|
||||
except OSError:
|
||||
# File vanished or perms changed — fall through and write
|
||||
self._write_digests.pop(key, None)
|
||||
# A metadata touch is journal-cheap compared to rewriting
|
||||
# the data.
|
||||
if self._skip_unchanged(key, cache_path, digest, timestamp):
|
||||
return
|
||||
|
||||
tmp_dir = os.path.dirname(cache_path)
|
||||
# Try to create temp file in cache directory first
|
||||
@@ -458,7 +574,7 @@ class DiskCache:
|
||||
# opened it in between was refused.
|
||||
_share_open_file(tmp_file.fileno(), _shared_group(tmp_dir))
|
||||
os.replace(tmp_path, cache_path)
|
||||
self._write_digests[key] = digest
|
||||
self._remember_write(key, cache_path, digest, timestamp)
|
||||
finally:
|
||||
if os.path.exists(tmp_path):
|
||||
try:
|
||||
@@ -471,13 +587,13 @@ class DiskCache:
|
||||
with open(cache_path, 'wb') as cache_file:
|
||||
cache_file.write(payload)
|
||||
_share_open_file(cache_file.fileno(), _shared_group(tmp_dir))
|
||||
self._write_digests[key] = digest
|
||||
self._remember_write(key, cache_path, digest, timestamp)
|
||||
self.logger.debug("Wrote cache for %s directly (non-atomic)", key)
|
||||
except (IOError, OSError, PermissionError) as write_error:
|
||||
# If direct write also fails, try fallback location
|
||||
self.logger.warning("Direct write failed for key '%s' to %s: %s", key, cache_path, write_error)
|
||||
raise # Re-raise to trigger fallback logic
|
||||
except (IOError, OSError, PermissionError):
|
||||
except (IOError, OSError, PermissionError) as primary_error:
|
||||
# Attempt one-time fallback write to user's home cache directory
|
||||
try:
|
||||
# Try user's home cache directory as fallback
|
||||
@@ -503,11 +619,14 @@ class DiskCache:
|
||||
self.logger.debug("Fallback cache write also failed for key '%s': %s", key, e2)
|
||||
|
||||
# If all write attempts failed, log warning but don't raise exception
|
||||
# Cache is a performance optimization, not critical for operation
|
||||
# Cache is a performance optimization, not critical for operation.
|
||||
# Name the real error: this used to say "permission denied"
|
||||
# whatever happened, which sent a too-long filename off to
|
||||
# be debugged as a directory-ownership problem.
|
||||
self.logger.warning(
|
||||
"Could not write cache for key '%s' to %s (permission denied). "
|
||||
"Could not write cache for key '%s' to %s (%s). "
|
||||
"Cache will be unavailable for this key, but application will continue.",
|
||||
key, cache_path
|
||||
key, cache_path, primary_error.strerror or primary_error
|
||||
)
|
||||
return # Exit gracefully without raising exception
|
||||
|
||||
@@ -520,6 +639,66 @@ class DiskCache:
|
||||
)
|
||||
return # Exit gracefully without raising exception
|
||||
|
||||
def _skip_unchanged(self, key: str, cache_path: str, digest: int,
|
||||
timestamp: Optional[float]) -> bool:
|
||||
"""Stand in for a write of an unchanged record by touching the file.
|
||||
|
||||
True when the file already holds this record bar its timestamp and the
|
||||
touch landed; False means write it. The touch sets mtime to the
|
||||
record's timestamp -- what a rewrite would have stored -- or to now
|
||||
for a record without one, whose age readers already take from mtime.
|
||||
Caller holds _lock.
|
||||
"""
|
||||
last = self._write_digests.get(key)
|
||||
if last is None or last[0] != digest:
|
||||
return False
|
||||
_, written_ts, ino, size = last
|
||||
if (timestamp is None) != (written_ts is None):
|
||||
return False
|
||||
if timestamp is not None and written_ts is not None:
|
||||
# Never backwards (a rewrite would make the record older), and
|
||||
# never further than readers will trust the mtime: past that the
|
||||
# record is rewritten, so its own timestamp catches up.
|
||||
if not written_ts <= timestamp <= written_ts + _MAX_TIMESTAMP_LIFT:
|
||||
return False
|
||||
try:
|
||||
st = os.stat(cache_path)
|
||||
if (st.st_ino, st.st_size) != (ino, size):
|
||||
# Replaced since our write (another process, a restore):
|
||||
# its contents are not the ones the digest describes.
|
||||
self._write_digests.pop(key, None)
|
||||
return False
|
||||
# Setting an explicit time needs the file's owner; a file someone
|
||||
# else wrote fails here and is rewritten (as our own file) instead.
|
||||
os.utime(cache_path, None if timestamp is None else (timestamp, timestamp))
|
||||
return True
|
||||
except OSError:
|
||||
# File vanished or perms changed — fall through and write
|
||||
self._write_digests.pop(key, None)
|
||||
return False
|
||||
|
||||
def _remember_write(self, key: str, cache_path: str, digest: int,
|
||||
timestamp: Optional[float]) -> None:
|
||||
"""After a real write: pin mtime to the record's timestamp and note
|
||||
what was written, so the next unchanged save can be skipped.
|
||||
|
||||
Pinning keeps a record saved with an older timestamp from looking as
|
||||
fresh as the moment it was written (see "UNCHANGED RE-SAVES"); for
|
||||
CacheManager.set's records the two differ only by the write itself.
|
||||
A timestamp in the future is left alone, mtime already being older.
|
||||
Never raises: the data is on disk, and anything failing here only
|
||||
costs the next save its skip. Caller holds _lock.
|
||||
"""
|
||||
self._write_digests.pop(key, None)
|
||||
try:
|
||||
if timestamp is not None and timestamp <= time.time():
|
||||
os.utime(cache_path, (timestamp, timestamp))
|
||||
st = os.stat(cache_path)
|
||||
except OSError as e:
|
||||
self.logger.debug("Could not pin mtime of %s: %s", cache_path, e)
|
||||
return
|
||||
self._write_digests[key] = (digest, timestamp, st.st_ino, st.st_size)
|
||||
|
||||
def clear(self, key: Optional[str] = None) -> None:
|
||||
"""
|
||||
Clear cache entry or all entries.
|
||||
|
||||
+140
-14
@@ -28,7 +28,7 @@ import os
|
||||
import time
|
||||
from datetime import datetime
|
||||
import pytz
|
||||
from typing import Any, Dict, List, Optional
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
import logging
|
||||
import threading
|
||||
import tempfile
|
||||
@@ -43,6 +43,72 @@ from src.logging_config import get_logger
|
||||
# it from either path.
|
||||
from src.cache.disk_cache import DateTimeEncoder # noqa: F401 - deliberate re-export
|
||||
|
||||
# CacheManager.config_manager not built yet (None means "not available").
|
||||
_UNSET: Any = object()
|
||||
|
||||
|
||||
def _outlived(record: Any, max_age: Optional[float], now: float) -> bool:
|
||||
"""Whether a record's own timestamp puts it past max_age.
|
||||
|
||||
The memory tier times an entry from when it was put there, and a record
|
||||
loaded from disk is put there when it is read, not when it was written: a
|
||||
record 290 s old, read after a restart, could be served for another
|
||||
max_age from memory. This is the age check DiskCache.get makes, with the
|
||||
same rule that a stored ttl wins over the caller's max_age. A record that
|
||||
carries no timestamp is left to the memory tier's own clock.
|
||||
"""
|
||||
if not isinstance(record, dict):
|
||||
return False
|
||||
stored_ttl = record.get('ttl')
|
||||
if isinstance(stored_ttl, (int, float)) and not isinstance(stored_ttl, bool) \
|
||||
and stored_ttl >= 0:
|
||||
max_age = stored_ttl
|
||||
stamp = record.get('timestamp')
|
||||
if max_age is None or stamp is None or isinstance(stamp, bool):
|
||||
return False
|
||||
try:
|
||||
return now - float(stamp) > max_age
|
||||
except (TypeError, ValueError):
|
||||
return False
|
||||
|
||||
|
||||
_NOT_SEEN: Any = object()
|
||||
|
||||
|
||||
class MailboxWatch:
|
||||
"""Tells the poller of a mailbox key whether its file changed since the
|
||||
last look, from one stat() (:meth:`CacheManager.file_signature`).
|
||||
|
||||
The display polls the mailboxes the web interface falls back to. Reading
|
||||
one is an open and a JSON parse; with this a poll that finds the same file
|
||||
(or none) costs a stat, and the file is read only after a new write. A
|
||||
cache without ``file_signature`` (a test double) is read every time.
|
||||
"""
|
||||
|
||||
def __init__(self, key: str):
|
||||
self.key = key
|
||||
self._seen: Any = _NOT_SEEN
|
||||
|
||||
def changed(self, cache_manager: Any) -> bool:
|
||||
"""True when the poller should read the key now."""
|
||||
signature = getattr(cache_manager, 'file_signature', None)
|
||||
sig = signature(self.key) if callable(signature) else _NOT_SEEN
|
||||
if sig is not None and not isinstance(sig, tuple):
|
||||
return True # cannot tell: read it
|
||||
if sig is None:
|
||||
self._seen = None
|
||||
return False # no file, nothing to read
|
||||
if sig == self._seen:
|
||||
return False
|
||||
self._seen = sig
|
||||
return True
|
||||
|
||||
def forget(self) -> None:
|
||||
"""Read the key on the next poll even if its file has not changed
|
||||
(the last read failed)."""
|
||||
self._seen = _NOT_SEEN
|
||||
|
||||
|
||||
class CacheManager:
|
||||
"""Manages caching of API responses to reduce API calls."""
|
||||
|
||||
@@ -73,21 +139,19 @@ class CacheManager:
|
||||
self.logger.error("Could not find or create a writable cache directory. Caching will be disabled.")
|
||||
self.cache_dir = None
|
||||
|
||||
# Initialize config manager for sport-specific intervals
|
||||
try:
|
||||
from src.config_manager import ConfigManager
|
||||
self.config_manager: Optional[Any] = ConfigManager()
|
||||
self.config_manager.load_config()
|
||||
except ImportError:
|
||||
self.config_manager: Optional[Any] = None
|
||||
self.logger.warning("ConfigManager not available, using default cache intervals")
|
||||
|
||||
# The config manager is built on first use of self.config_manager; see
|
||||
# the property. Nothing in the cache reads it any more.
|
||||
self._config_manager: Any = _UNSET
|
||||
self._config_manager_lock = threading.Lock()
|
||||
|
||||
# Initialize cache components using composition
|
||||
self._memory_cache_component = MemoryCache(
|
||||
max_size=default_max_size(), cleanup_interval=300.0
|
||||
)
|
||||
self._disk_cache_component = DiskCache(cache_dir=self.cache_dir, logger=self.logger)
|
||||
self._strategy_component = CacheStrategy(config_manager=self.config_manager, logger=self.logger)
|
||||
# No config manager: CacheStrategy keeps the parameter for callers but
|
||||
# reads nothing from it, and passing ours would build it eagerly.
|
||||
self._strategy_component = CacheStrategy(logger=self.logger)
|
||||
self._metrics_component = CacheMetrics(logger=self.logger)
|
||||
|
||||
# Disk cleanup configuration
|
||||
@@ -115,6 +179,44 @@ class CacheManager:
|
||||
if self.cache_dir:
|
||||
self.start_cleanup_thread()
|
||||
|
||||
@property
|
||||
def config_manager(self) -> Optional[Any]:
|
||||
"""A loaded ConfigManager, built the first time it is asked for.
|
||||
|
||||
Every CacheManager used to build one and load the whole config in
|
||||
__init__, for a cache strategy that stopped reading it -- startup paid
|
||||
a config load (and the web interface another) per manager for nothing.
|
||||
It is still public: the sports plugins resolve the global timezone and
|
||||
display settings through ``cache_manager.config_manager``, and they get
|
||||
the same object they always did, on first access instead of at
|
||||
construction. None when ConfigManager cannot be imported, as before.
|
||||
Assigning replaces it, as assigning the attribute always did.
|
||||
"""
|
||||
# getattr: a manager made with __new__ (some tests) has no slot yet.
|
||||
value = getattr(self, '_config_manager', _UNSET)
|
||||
if value is not _UNSET:
|
||||
return value
|
||||
lock = getattr(self, '_config_manager_lock', None) or threading.Lock()
|
||||
with lock:
|
||||
value = getattr(self, '_config_manager', _UNSET)
|
||||
if value is _UNSET:
|
||||
try:
|
||||
from src.config_manager import ConfigManager
|
||||
except ImportError:
|
||||
self.logger.warning("ConfigManager not available, using default cache intervals")
|
||||
value = None
|
||||
else:
|
||||
value = ConfigManager()
|
||||
# Raises as it did from __init__; nothing is kept, so the
|
||||
# next access tries again.
|
||||
value.load_config()
|
||||
self._config_manager = value
|
||||
return value
|
||||
|
||||
@config_manager.setter
|
||||
def config_manager(self, value: Optional[Any]) -> None:
|
||||
self._config_manager = value
|
||||
|
||||
def _get_writable_cache_dir(self) -> Optional[str]:
|
||||
"""Tries to find or create a writable cache directory, preferring a system path when available."""
|
||||
# Attempt 1: System-wide persistent cache directory (preferred for services)
|
||||
@@ -230,7 +332,25 @@ class CacheManager:
|
||||
def _get_cache_path(self, key: str) -> Optional[str]:
|
||||
"""Get the path for a cache file."""
|
||||
return self._disk_cache_component.get_cache_path(key)
|
||||
|
||||
|
||||
def file_signature(self, key: str) -> Optional[Tuple[int, int, int]]:
|
||||
"""``(st_ino, st_mtime_ns, st_size)`` of ``key``'s file, or None when
|
||||
there is no file (the key is absent, or this cache has no disk tier).
|
||||
|
||||
One stat(), no read: a poller of a mailbox another process writes
|
||||
compares it with the last one it saw and reads the file only when it
|
||||
changed. Every write replaces the file (a temp file renamed into
|
||||
place), so a new write always has a new inode, however fast it came.
|
||||
"""
|
||||
path = self._get_cache_path(key)
|
||||
if not path:
|
||||
return None
|
||||
try:
|
||||
st = os.stat(path)
|
||||
except OSError:
|
||||
return None
|
||||
return (st.st_ino, st.st_mtime_ns, st.st_size)
|
||||
|
||||
def get_cached_data(self, key: str, max_age: int = 300, memory_ttl: Optional[int] = None) -> Optional[Dict[str, Any]]:
|
||||
"""Get data from cache (memory first, then disk) honoring TTLs.
|
||||
|
||||
@@ -245,7 +365,11 @@ class CacheManager:
|
||||
# 1) Memory cache
|
||||
cached = self._memory_cache_component.get(key, max_age=in_memory_ttl)
|
||||
if cached is not None:
|
||||
return cached
|
||||
if not _outlived(cached, max_age, time.time()):
|
||||
return cached
|
||||
# Too old for this reader. Disk may hold a newer write (from the
|
||||
# other process), and if it does not, the miss is the right answer.
|
||||
self._memory_cache_component.clear(key)
|
||||
|
||||
# 2) Disk cache
|
||||
record = self._disk_cache_component.get(key, max_age=max_age)
|
||||
@@ -279,7 +403,9 @@ class CacheManager:
|
||||
# Check memory cache first (1 minute TTL)
|
||||
cached = self._memory_cache_component.get(key, max_age=60)
|
||||
if cached is not None:
|
||||
return cached
|
||||
if not _outlived(cached, 3600, time.time()):
|
||||
return cached
|
||||
self._memory_cache_component.clear(key)
|
||||
|
||||
# Check disk cache
|
||||
data = self._disk_cache_component.get(key, max_age=3600) # 1 hour for load_cache
|
||||
|
||||
+59
-5
@@ -17,7 +17,9 @@ Rules for the package:
|
||||
- `from src.common import ...` re-exports `APIHelper`, `ScrollHelper`,
|
||||
`LogoHelper`, `TextHelper`, `scroll_config` (plus `ScrollSettings`,
|
||||
`configure_scroll`, `resolve_scroll_settings`, `refresh_hz_from_config`) and
|
||||
the adaptive layout names below ([`__init__.py`](__init__.py)).
|
||||
the adaptive layout names below ([`__init__.py`](__init__.py)). Each is
|
||||
imported on first use, so `import src.common` or a submodule import stays
|
||||
cheap; add a new re-export to `_LAZY` there as well as `__all__`.
|
||||
|
||||
## Summary
|
||||
|
||||
@@ -27,6 +29,7 @@ Rules for the package:
|
||||
| [`bdf_font`](#bdf_font) | Load and draw BDF bitmap fonts | Yes, if drawing BDF text directly | 3.5.0 |
|
||||
| [`espn_dates`](#espn_dates) | Fetch ESPN scoreboards across a date range | Yes (scoreboards) | 3.5.0 |
|
||||
| [`favorite_team_check`](#favorite_team_check) | Log why a favourite team code shows nothing | Yes (scoreboards) | 3.6.0 |
|
||||
| [`fetch_service`](#fetch_service) | Pooled, merged, budgeted and counted HTTP for core fetch paths | No, core-internal (reached through `api_helper` and `espn_dates`) | n/a |
|
||||
| [`font_layout`](#font_layout) | Reproducible TrueType loading, crisp sizes | Yes | 3.4.0 |
|
||||
| [`frame_timing`](#frame_timing) | Timing of every presented frame, stall watchdog | No, core-internal | n/a |
|
||||
| [`json_body`](#json_body) | Parse a response body as JSON, with orjson if installed | Optional (large payloads) | 3.5.0 |
|
||||
@@ -43,6 +46,7 @@ Rules for the package:
|
||||
| [`sports_display_rules`](#sports_display_rules) | Which games a scoreboard shows, for how long, and its scorebug date line | Yes (scoreboards) | 3.8.0 |
|
||||
| [`sports_fetch`](#sports_fetch) | Scoreboard season fetch, lookback and live-odds decisions | Yes (scoreboards) | 3.7.0 |
|
||||
| [`sports_font_path`](#sports_font_path) | Find a scoreboard's bundled font whatever the cwd | Yes (scoreboards) | 3.8.0 |
|
||||
| [`sports_game_over`](#sports_game_over) | Whether a game ESPN still lists as live has ended | Yes (scoreboards) | 3.8.1 |
|
||||
| [`sports_game_renderer`](#sports_game_renderer) | Scoreboard scroll/Vegas card geometry | Yes (scoreboards) | 3.3.0 |
|
||||
| [`sports_helpers`](#sports_helpers) | Small helpers every scoreboard `sports.py` copies | Yes (scoreboards) | 3.5.0 |
|
||||
| [`sports_live_scroll`](#sports_live_scroll) | Rebuild a live scroll strip mid-cycle without moving it | Yes (scoreboards) | 3.8.0 |
|
||||
@@ -106,9 +110,14 @@ and the plugin test harness all use it. Most plugins get BDF text through
|
||||
[`espn_dates.py`](espn_dates.py). ESPN's site API rejects `dates=` ranges
|
||||
and truncates results when `limit` is above 500. `fetch_espn_scoreboard()`
|
||||
splits a range into month and day requests ESPN accepts and merges the
|
||||
results; `espn_date_chunks()`, `fetch_espn_date_chunks()`,
|
||||
`clamp_espn_limit()` and `merge_scoreboard_payloads()` are the pieces.
|
||||
Scoreboard plugins also bundle a copy for older cores.
|
||||
results; `espn_date_chunks()`, `espn_request_chunks()`,
|
||||
`fetch_espn_date_chunks()`, `clamp_espn_limit()` and
|
||||
`merge_scoreboard_payloads()` are the pieces. A window's partial edge months
|
||||
are asked whole and trimmed to its days (US Eastern), and chunk requests share
|
||||
one process-wide cap of `ESPN_CHUNK_WORKERS` in flight.
|
||||
Every request goes through [`fetch_service`](#fetch_service), the chunks
|
||||
counted against the plugin that asked. Scoreboard plugins also bundle a copy
|
||||
for older cores.
|
||||
|
||||
### favorite_team_check
|
||||
|
||||
@@ -121,6 +130,23 @@ says the league has nothing on yet; `reset()` re-arms it after a config edit.
|
||||
Diagnostics only: every failure is swallowed. Scoreboard plugins also bundle
|
||||
a copy for older cores.
|
||||
|
||||
### fetch_service
|
||||
|
||||
[`fetch_service.py`](fetch_service.py). Core-internal for now. Every core
|
||||
fetch path -- `APIHelper.get`/`post`, `espn_dates` (so every scoreboard's
|
||||
ESPN scoreboard fetch and `SportsFetchMixin`), `BackgroundDataService` and
|
||||
`BaseOddsManager` -- calls `fetch_get(session, url, ...)` instead of
|
||||
`session.get(url, ...)`. Same arguments, return value and exceptions; on top
|
||||
it shares one connection pool per host per retry policy
|
||||
(`share_connection_pool`), merges identical GETs in flight, applies per-host
|
||||
token buckets (`fetch_service.rate_limits` in config.json; ESPN gets 20/s,
|
||||
burst 200), revalidates with server-sent `ETag`/`Last-Modified` and counts
|
||||
requests per plugin and per host. The display publishes the counters
|
||||
(`FetchStatsPublisher`) for `GET /api/v3/plugins/fetch-stats`. Which plugin
|
||||
made a request comes from `plugin_scope()`, set by the plugin executor, or
|
||||
else from the plugin directory on the stack. See
|
||||
[docs/PLUGIN_API_REFERENCE.md](../../docs/PLUGIN_API_REFERENCE.md#fetching-data).
|
||||
|
||||
### font_layout
|
||||
|
||||
[`font_layout.py`](font_layout.py). `load_truetype(path, size)` is
|
||||
@@ -211,7 +237,8 @@ rather than the `set_*` methods. Vegas mode reads a plugin's
|
||||
[`snapshot_policy.py`](snapshot_policy.py). Core-internal. `decide()`
|
||||
tells `DisplayManager` whether to write `/tmp/led_matrix_preview.png`, only
|
||||
touch its mtime, or skip, based on whether a browser is watching the preview.
|
||||
The web health check reads the file's age.
|
||||
The web health check reads the file's age, and the web preview stream checks
|
||||
its mtime every `VIEWER_POLL_INTERVAL`.
|
||||
|
||||
### sports_card
|
||||
|
||||
@@ -268,6 +295,16 @@ path as given when it exists (relative to the cwd), else
|
||||
`font_layout.resolve_asset_path(path)`. What the scoreboards'
|
||||
`_resolve_font_path` copies return on a core that ships it.
|
||||
|
||||
### sports_game_over
|
||||
|
||||
[`sports_game_over.py`](sports_game_over.py). `SportsGameOverMixin`:
|
||||
`_is_game_really_over(game)`, the `SportsLive` check that drops a game ESPN
|
||||
still lists as live (`SportsLiveSharedMixin._detect_stale_games` calls it).
|
||||
Over on a final period text, or on a 0:00 clock from period `FINAL_PERIOD`
|
||||
on unless the score is level. `FINAL_PERIOD` is a class attribute the host
|
||||
sets per sport; the default `None` means the clock never ends a game. List
|
||||
it before `SportsLiveSharedMixin`.
|
||||
|
||||
### sports_game_renderer
|
||||
|
||||
[`sports_game_renderer.py`](sports_game_renderer.py).
|
||||
@@ -311,6 +348,10 @@ dynamic-duration helpers. List it before `BasePlugin`.
|
||||
scoreboards share (Vegas items, dynamic duration, frame loop), paced through
|
||||
`scroll_config`. Subclasses supply `prepare_scroll_content()` and set
|
||||
`SCROLL_LEAGUE_KEYS`; see the module docstring for an example.
|
||||
`prepare_and_display()` rewinds a recent or upcoming strip whose games,
|
||||
rankings, config, panel size and date are unchanged instead of calling
|
||||
`prepare_scroll_content()` again, with one display per slate (game type and
|
||||
leagues).
|
||||
|
||||
### sports_shared
|
||||
|
||||
@@ -360,6 +401,19 @@ Created by `DisplayController`; works with any plugin.
|
||||
`get_text_dimensions()`, `center_text()`, `wrap_text()`,
|
||||
`draw_multiline_text()`, `create_text_image()`.
|
||||
|
||||
`draw_text_outlined(draw, xy, text, font, fill, outline_color=(0, 0, 0),
|
||||
offsets=OUTLINE_SQUARE)` (3.8.1) draws the text in `outline_color` at
|
||||
each offset, then in `fill` on top: the same pixels as one `draw.text` per
|
||||
offset, but the string is rasterized once. `OUTLINE_SQUARE` is the
|
||||
eight-sided one-pixel outline the scoreboards draw, `OUTLINE_CROSS` the
|
||||
four-sided one. Fractional coordinates (a whole-pixel float such as `52.0`
|
||||
is fine), multiline text, fonts other than a plain `FreeTypeFont`, image modes
|
||||
other than RGB, RGBA and L, and a subclassed or replaced `draw.text` take
|
||||
the `draw.text` loop unchanged. `TextHelper.draw_text_with_outline()` and
|
||||
the scoreboards' `SportsCoreSharedMixin._draw_text_with_outline()` use it.
|
||||
A plugin that also runs on older cores should guard the import and keep its
|
||||
own loop as the fallback.
|
||||
|
||||
## Logging
|
||||
|
||||
Modules here create their logger with `logging.getLogger(__name__)`, which is
|
||||
|
||||
+103
-36
@@ -6,45 +6,90 @@ This package provides reusable functionality for plugins and core modules:
|
||||
- Logo helpers
|
||||
- Text/scroll helpers
|
||||
- Adaptive layout and image helpers
|
||||
|
||||
The names below are imported on first use (PEP 562), not when the package is
|
||||
imported. ``from src.common import ScrollHelper`` and
|
||||
``src.common.ScrollHelper`` work as before and return the same objects, but
|
||||
``import src.common`` -- or importing any submodule, such as
|
||||
``src.common.path_safety`` -- no longer loads numpy, requests and freetype
|
||||
along with every helper. The web interface imports src.common only for a few
|
||||
small modules and never needs those.
|
||||
"""
|
||||
|
||||
# Export commonly used utilities
|
||||
from src.common.api_helper import APIHelper
|
||||
from src.common.scroll_helper import ScrollHelper
|
||||
from src.common import scroll_config
|
||||
from src.common.scroll_config import (
|
||||
ScrollSettings,
|
||||
configure as configure_scroll,
|
||||
resolve as resolve_scroll_settings,
|
||||
refresh_hz_from_config,
|
||||
)
|
||||
from src.common.logo_helper import LogoHelper
|
||||
from src.common.text_helper import TextHelper
|
||||
import importlib
|
||||
from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple
|
||||
|
||||
# Adaptive layout & images (canonical homes: src.adaptive_layout /
|
||||
# src.adaptive_images — re-exported here so plugin authors find them in the
|
||||
# blessed-helpers package). See docs/ADAPTIVE_LAYOUT.md.
|
||||
from src.adaptive_layout import (
|
||||
Region,
|
||||
LayoutContext,
|
||||
FontStep,
|
||||
FontLadder,
|
||||
LADDER_GRID,
|
||||
LADDER_ARCADE,
|
||||
FitResult,
|
||||
draw_fitted_text,
|
||||
ScoreboardRegions,
|
||||
scoreboard_regions,
|
||||
MediaRow,
|
||||
media_row,
|
||||
)
|
||||
from src.adaptive_images import (
|
||||
ImageFitResult,
|
||||
fit_image,
|
||||
draw_fitted_image,
|
||||
RESAMPLE_LANCZOS,
|
||||
RESAMPLE_NEAREST,
|
||||
)
|
||||
if TYPE_CHECKING:
|
||||
# What mypy and editors see: the real names and their types.
|
||||
from src.common.api_helper import APIHelper
|
||||
from src.common.scroll_helper import ScrollHelper
|
||||
from src.common import scroll_config
|
||||
from src.common.scroll_config import (
|
||||
ScrollSettings,
|
||||
configure as configure_scroll,
|
||||
resolve as resolve_scroll_settings,
|
||||
refresh_hz_from_config,
|
||||
)
|
||||
from src.common.logo_helper import LogoHelper
|
||||
from src.common.text_helper import TextHelper
|
||||
|
||||
# Adaptive layout & images (canonical homes: src.adaptive_layout /
|
||||
# src.adaptive_images — re-exported here so plugin authors find them in the
|
||||
# blessed-helpers package). See docs/ADAPTIVE_LAYOUT.md.
|
||||
from src.adaptive_layout import (
|
||||
Region,
|
||||
LayoutContext,
|
||||
FontStep,
|
||||
FontLadder,
|
||||
LADDER_GRID,
|
||||
LADDER_ARCADE,
|
||||
FitResult,
|
||||
draw_fitted_text,
|
||||
ScoreboardRegions,
|
||||
scoreboard_regions,
|
||||
MediaRow,
|
||||
media_row,
|
||||
)
|
||||
from src.adaptive_images import (
|
||||
ImageFitResult,
|
||||
fit_image,
|
||||
draw_fitted_image,
|
||||
RESAMPLE_LANCZOS,
|
||||
RESAMPLE_NEAREST,
|
||||
)
|
||||
|
||||
#: Exported name -> (module it lives in, attribute name there). An attribute
|
||||
#: of None means the name is the module itself. Keep in step with the
|
||||
#: TYPE_CHECKING imports above and with __all__.
|
||||
_LAZY: Dict[str, Tuple[str, Optional[str]]] = {
|
||||
'APIHelper': ('src.common.api_helper', 'APIHelper'),
|
||||
'ScrollHelper': ('src.common.scroll_helper', 'ScrollHelper'),
|
||||
'scroll_config': ('src.common.scroll_config', None),
|
||||
'ScrollSettings': ('src.common.scroll_config', 'ScrollSettings'),
|
||||
'configure_scroll': ('src.common.scroll_config', 'configure'),
|
||||
'resolve_scroll_settings': ('src.common.scroll_config', 'resolve'),
|
||||
'refresh_hz_from_config': ('src.common.scroll_config', 'refresh_hz_from_config'),
|
||||
'LogoHelper': ('src.common.logo_helper', 'LogoHelper'),
|
||||
'TextHelper': ('src.common.text_helper', 'TextHelper'),
|
||||
# adaptive layout & images
|
||||
'Region': ('src.adaptive_layout', 'Region'),
|
||||
'LayoutContext': ('src.adaptive_layout', 'LayoutContext'),
|
||||
'FontStep': ('src.adaptive_layout', 'FontStep'),
|
||||
'FontLadder': ('src.adaptive_layout', 'FontLadder'),
|
||||
'LADDER_GRID': ('src.adaptive_layout', 'LADDER_GRID'),
|
||||
'LADDER_ARCADE': ('src.adaptive_layout', 'LADDER_ARCADE'),
|
||||
'FitResult': ('src.adaptive_layout', 'FitResult'),
|
||||
'draw_fitted_text': ('src.adaptive_layout', 'draw_fitted_text'),
|
||||
'ScoreboardRegions': ('src.adaptive_layout', 'ScoreboardRegions'),
|
||||
'scoreboard_regions': ('src.adaptive_layout', 'scoreboard_regions'),
|
||||
'MediaRow': ('src.adaptive_layout', 'MediaRow'),
|
||||
'media_row': ('src.adaptive_layout', 'media_row'),
|
||||
'ImageFitResult': ('src.adaptive_images', 'ImageFitResult'),
|
||||
'fit_image': ('src.adaptive_images', 'fit_image'),
|
||||
'draw_fitted_image': ('src.adaptive_images', 'draw_fitted_image'),
|
||||
'RESAMPLE_LANCZOS': ('src.adaptive_images', 'RESAMPLE_LANCZOS'),
|
||||
'RESAMPLE_NEAREST': ('src.adaptive_images', 'RESAMPLE_NEAREST'),
|
||||
}
|
||||
|
||||
__all__ = [
|
||||
'APIHelper',
|
||||
@@ -75,3 +120,25 @@ __all__ = [
|
||||
'RESAMPLE_LANCZOS',
|
||||
'RESAMPLE_NEAREST',
|
||||
]
|
||||
|
||||
|
||||
def __getattr__(name: str) -> Any:
|
||||
"""Import an exported name on first access (PEP 562).
|
||||
|
||||
Only called for names not already in the module namespace, so after the
|
||||
first access the cached value below is returned directly. Unknown names
|
||||
raise AttributeError, which ``from src.common import <submodule>`` relies
|
||||
on to fall through to importing the submodule.
|
||||
"""
|
||||
try:
|
||||
module_name, attr = _LAZY[name]
|
||||
except KeyError:
|
||||
raise AttributeError(f"module {__name__!r} has no attribute {name!r}") from None
|
||||
module = importlib.import_module(module_name) # nosemgrep: python.lang.security.audit.non-literal-import.non-literal-import -- module_name comes from the fixed _LAZY table
|
||||
value = module if attr is None else getattr(module, attr)
|
||||
globals()[name] = value
|
||||
return value
|
||||
|
||||
|
||||
def __dir__() -> List[str]:
|
||||
return sorted(set(globals()) | set(__all__))
|
||||
|
||||
+63
-22
@@ -10,11 +10,17 @@ import logging
|
||||
import time
|
||||
from datetime import datetime
|
||||
from types import MappingProxyType
|
||||
from src.common.espn_dates import ESPN_MAX_LIMIT
|
||||
from src.common.espn_dates import (
|
||||
ESPN_MAX_LIMIT,
|
||||
espn_scoreboard_cache_key,
|
||||
read_espn_scoreboard_cache,
|
||||
store_espn_scoreboard_cache,
|
||||
)
|
||||
from src.common.fetch_service import fetch_get, fetch_post, share_connection_pool
|
||||
from src.common.json_body import response_json
|
||||
from typing import TYPE_CHECKING, Any, Dict, Mapping, Optional, cast
|
||||
|
||||
import requests
|
||||
from requests.adapters import HTTPAdapter
|
||||
from urllib3.util.retry import Retry
|
||||
|
||||
if TYPE_CHECKING:
|
||||
@@ -45,7 +51,11 @@ class APIHelper:
|
||||
|
||||
- Requests go through one ``requests.Session`` that retries GET, HEAD
|
||||
and OPTIONS on 429 and 5xx with exponential backoff, and sends
|
||||
:data:`DEFAULT_HTTP_HEADERS`.
|
||||
:data:`DEFAULT_HTTP_HEADERS`. Its connection pool is shared with every
|
||||
other helper using the same retry policy, and requests go through the
|
||||
core fetch service (``src/common/fetch_service.py``): identical GETs in
|
||||
flight are merged, hosts with a budget are paced, and requests are
|
||||
counted per plugin. Return values and errors are unchanged.
|
||||
- Consecutive requests from one helper are spaced at least
|
||||
``set_rate_limit()`` seconds apart (1 second by default). A cache hit
|
||||
does not count.
|
||||
@@ -81,9 +91,10 @@ class APIHelper:
|
||||
status_forcelist=[429, 500, 502, 503, 504],
|
||||
allowed_methods=["GET", "HEAD", "OPTIONS"]
|
||||
)
|
||||
adapter = HTTPAdapter(max_retries=retry_strategy)
|
||||
self.session.mount("https://", adapter)
|
||||
self.session.mount("http://", adapter)
|
||||
# The shared adapter for this retry policy: the same retries as a
|
||||
# private HTTPAdapter(max_retries=retry_strategy), with the connection
|
||||
# pool shared by every helper (fetch_service).
|
||||
share_connection_pool(self.session, retry_strategy)
|
||||
|
||||
self.session.headers.update({**DEFAULT_HTTP_HEADERS, 'Connection': 'keep-alive'})
|
||||
|
||||
@@ -112,6 +123,14 @@ class APIHelper:
|
||||
Returns:
|
||||
Response data as dictionary or None if request fails
|
||||
"""
|
||||
return self._get(url, params, headers, timeout, cache_key, cache_ttl,
|
||||
cache_ttl if cache_key else None)
|
||||
|
||||
def _get(self, url: str, params: Optional[Dict], headers: Optional[Dict],
|
||||
timeout: Optional[int], cache_key: Optional[str], cache_ttl: int,
|
||||
cache_max_age: Optional[float]) -> Optional[Dict]:
|
||||
""":meth:`get`, saying how old a response the fetch service's short
|
||||
response cache may hand back (``cache_max_age``, the caller's TTL)."""
|
||||
if cache_key and self.cache_manager:
|
||||
cached = self._get_from_cache(cache_key, cache_ttl)
|
||||
if cached is not None:
|
||||
@@ -128,16 +147,18 @@ class APIHelper:
|
||||
request_headers.update(headers)
|
||||
|
||||
# Make request
|
||||
response = self.session.get(
|
||||
response = fetch_get(
|
||||
self.session,
|
||||
url,
|
||||
params=params,
|
||||
headers=request_headers,
|
||||
timeout=timeout or self.default_timeout
|
||||
timeout=timeout or self.default_timeout,
|
||||
cache_max_age=cache_max_age,
|
||||
)
|
||||
response.raise_for_status()
|
||||
|
||||
# Parse JSON response
|
||||
data: Dict[Any, Any] = response.json()
|
||||
data: Dict[Any, Any] = response_json(response)
|
||||
|
||||
# Cache response if cache key provided
|
||||
if cache_key and self.cache_manager:
|
||||
@@ -161,22 +182,23 @@ class APIHelper:
|
||||
sport: Sport name (e.g., 'basketball', 'football')
|
||||
league: League name (e.g., 'nba', 'nfl')
|
||||
date: Date in YYYYMMDD format (defaults to today)
|
||||
cache_key: Cache key for response
|
||||
cache_ttl: Cache time-to-live in seconds
|
||||
|
||||
cache_key: Cache key for response. By default the canonical
|
||||
``espn_scoreboard_cache_key(sport, league, date)``, shared
|
||||
with every other consumer of this scoreboard, with the key
|
||||
this used before (``espn_{sport}_{league}_{date}``) read as a
|
||||
fallback for one release. An explicit key works as before.
|
||||
cache_ttl: Cache time-to-live in seconds. A shared entry is
|
||||
returned only while it is at most this old.
|
||||
|
||||
Returns:
|
||||
ESPN API response data or None if request fails
|
||||
"""
|
||||
if date is None:
|
||||
date = datetime.now().strftime('%Y%m%d')
|
||||
|
||||
|
||||
# Build URL
|
||||
url = f"https://site.api.espn.com/apis/site/v2/sports/{sport}/{league}/scoreboard"
|
||||
|
||||
# Build cache key if not provided
|
||||
if cache_key is None:
|
||||
cache_key = f"espn_{sport}_{league}_{date}"
|
||||
|
||||
|
||||
# Set parameters
|
||||
# limit above 500 makes ESPN truncate instead of erroring: college
|
||||
# football came back with 25 of 68 games. See src/common/espn_dates.py.
|
||||
@@ -184,8 +206,26 @@ class APIHelper:
|
||||
'dates': date,
|
||||
'limit': ESPN_MAX_LIMIT
|
||||
}
|
||||
|
||||
return self.get(url, params=params, cache_key=cache_key, cache_ttl=cache_ttl)
|
||||
|
||||
if cache_key is not None:
|
||||
return self.get(url, params=params, cache_key=cache_key, cache_ttl=cache_ttl)
|
||||
|
||||
legacy_key = f"espn_{sport}_{league}_{date}"
|
||||
try:
|
||||
shared_key = espn_scoreboard_cache_key(sport, league, date)
|
||||
except ValueError:
|
||||
# Not a path or date the canonical key covers: the old key.
|
||||
return self.get(url, params=params, cache_key=legacy_key, cache_ttl=cache_ttl)
|
||||
if self.cache_manager:
|
||||
cached = read_espn_scoreboard_cache(
|
||||
self.cache_manager, shared_key, cache_ttl, legacy_keys=(legacy_key,))
|
||||
if cached is not None:
|
||||
self.logger.debug(f"Using cached response for {shared_key}")
|
||||
return cast(Dict[Any, Any], cached)
|
||||
data = self._get(url, params, None, None, None, cache_ttl, cache_ttl)
|
||||
if data is not None and self.cache_manager:
|
||||
store_espn_scoreboard_cache(self.cache_manager, shared_key, data)
|
||||
return data
|
||||
|
||||
def fetch_espn_standings(self, sport: str, league: str,
|
||||
cache_key: Optional[str] = None,
|
||||
@@ -255,7 +295,8 @@ class APIHelper:
|
||||
if headers:
|
||||
request_headers.update(headers)
|
||||
|
||||
response = self.session.post(
|
||||
response = fetch_post(
|
||||
self.session,
|
||||
url,
|
||||
data=data,
|
||||
json=json_data,
|
||||
@@ -264,7 +305,7 @@ class APIHelper:
|
||||
)
|
||||
response.raise_for_status()
|
||||
|
||||
return cast(Optional[Dict[Any, Any]], response.json())
|
||||
return cast(Optional[Dict[Any, Any]], response_json(response))
|
||||
|
||||
except requests.exceptions.RequestException as e:
|
||||
self.logger.error(f"POST request failed for {url}: {e}")
|
||||
|
||||
+491
-42
@@ -26,18 +26,58 @@ A month can hold more than 500 events (college baseball's March does), and
|
||||
ESPN answers that with exactly ``limit`` events and no hint that more exist. A
|
||||
month chunk that comes back full is therefore re-asked day by day.
|
||||
|
||||
A window's *partial* edge months are asked for whole, too, once the window
|
||||
covers ``ESPN_MONTH_COVER_MIN_DAYS`` or more of their days, and the answer is
|
||||
trimmed back to the window's days. A scoreboard's default fortnight either side
|
||||
of today (29 days, two partial months) was 29 day requests per league; it is
|
||||
now 2. Trimming needs ESPN's "game day", which is the event's start in US
|
||||
Eastern time -- checked against the live API on 2026-10-03: 417 of 417 soccer
|
||||
events across five leagues and three months (one of them spanning the end of
|
||||
daylight saving) came back from exactly the day query their Eastern date
|
||||
names. A short window (a live poll's one or two days) stays day by day, so it
|
||||
never downloads a whole month to read a day of it.
|
||||
|
||||
Chunk requests share one process-wide budget of ``ESPN_CHUNK_WORKERS`` in
|
||||
flight, however many windows are being fetched at once. Each window used to get
|
||||
its own six, so a scoreboard starting eight leagues -- each with a recent and
|
||||
an upcoming manager -- had ~40 requests in flight, every one beyond a session's
|
||||
pool a new connection and a new DNS lookup. On a Pi whose resolver could not
|
||||
keep up, that was ~90 ``NameResolutionError`` lines within a minute of every
|
||||
start.
|
||||
|
||||
Once a range has been rejected, later ranges skip straight to chunks for
|
||||
``RANGE_RETRY_SECONDS`` instead of spending a doomed request first -- live
|
||||
scoreboards ask every 30 seconds. After that the range is tried again, so the
|
||||
workaround retires itself if ESPN reverts.
|
||||
workaround retires itself if ESPN reverts. A process starts inside that
|
||||
period, as if a range had just been rejected.
|
||||
|
||||
ONE CACHE KEY PER SCOREBOARD
|
||||
----------------------------
|
||||
The same ESPN scoreboard used to be cached under a different key by every
|
||||
consumer: odds-ticker as ``scoreboard_data_{sport}_{league}_{date}``,
|
||||
``APIHelper`` as ``espn_{sport}_{league}_{date}``, the scoreboards as
|
||||
``{sport_key}_schedule_{window}`` -- so two plugins showing the same league
|
||||
fetched and stored it twice. :func:`espn_scoreboard_cache_key` is the one
|
||||
name for "this sport/league scoreboard for these dates", and
|
||||
:func:`get_espn_scoreboard` (or :func:`read_espn_scoreboard_cache` and
|
||||
:func:`store_espn_scoreboard_cache` around :func:`fetch_espn_scoreboard`)
|
||||
is the cache-through read every consumer can share. A read never returns an
|
||||
entry older than the reader's own ``max_age``, whoever wrote it and whatever
|
||||
ttl they stored with it. Old keys are passed as ``legacy_keys`` and read
|
||||
after the canonical one, so an upgrade does not refetch everything at once;
|
||||
they can go one release after the one that added this.
|
||||
"""
|
||||
|
||||
import contextvars
|
||||
import logging
|
||||
import math
|
||||
import re
|
||||
import threading
|
||||
import time
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from datetime import date, timedelta
|
||||
from datetime import date, datetime, timedelta, tzinfo
|
||||
from functools import partial
|
||||
from typing import Any, Dict, List, Optional, Tuple, cast
|
||||
from typing import Any, Callable, Dict, Iterable, List, Optional, Tuple, cast
|
||||
|
||||
try:
|
||||
from src.common.json_body import response_json
|
||||
@@ -47,6 +87,26 @@ except ImportError:
|
||||
def response_json(response: Any) -> Any:
|
||||
return response.json()
|
||||
|
||||
try:
|
||||
# The core fetch service: counts, per-host budget, merging of identical
|
||||
# requests. Same call, same result and errors as ``session.get``.
|
||||
from src.common.fetch_service import fetch_get, get_fetch_service, pinned_caller
|
||||
_COUNTS_FETCHES = True
|
||||
except ImportError:
|
||||
# Bundled copies on cores without it call the session directly.
|
||||
import contextlib
|
||||
|
||||
def fetch_get(session: Any, url: str, *, share_in_flight: bool = True,
|
||||
cache_max_age: Optional[float] = None, **kwargs: Any) -> Any:
|
||||
return session.get(url, **kwargs)
|
||||
|
||||
def pinned_caller() -> Any:
|
||||
return contextlib.nullcontext()
|
||||
|
||||
_COUNTS_FETCHES = False
|
||||
|
||||
_logger = logging.getLogger(__name__)
|
||||
|
||||
# Above this, ESPN returns a truncated list instead of an error. See module
|
||||
# docstring: 500 is the largest value measured to return complete data.
|
||||
ESPN_MAX_LIMIT = 500
|
||||
@@ -60,21 +120,77 @@ RANGE_RETRY_SECONDS = 6 * 60 * 60
|
||||
# pool_maxsize of 10 so the shared Session never has to discard connections.
|
||||
ESPN_CHUNK_WORKERS = 6
|
||||
|
||||
#: An edge month the window covers at least this many days of is asked for
|
||||
#: whole and trimmed, instead of one request per day (see module docstring).
|
||||
#: Below it the days are cheaper than the month: a whole month is two to
|
||||
#: three times the bytes of the half of it a fortnight window holds.
|
||||
ESPN_MONTH_COVER_MIN_DAYS = 7
|
||||
|
||||
# Every chunk request in the process holds one of these while it is in flight
|
||||
# -- the cap is per process, not per window (see module docstring).
|
||||
_chunk_slots = threading.BoundedSemaphore(ESPN_CHUNK_WORKERS)
|
||||
|
||||
|
||||
def _eastern_zone() -> Optional[tzinfo]:
|
||||
"""US Eastern, the zone ESPN's ``dates=YYYYMMDD`` means, or None when
|
||||
this Python has no time zone data (no edge month is trimmed then)."""
|
||||
zone: Optional[tzinfo] = None
|
||||
try:
|
||||
from zoneinfo import ZoneInfo
|
||||
zone = ZoneInfo("America/New_York")
|
||||
except Exception: # noqa: BLE001 - no zoneinfo module or no tz database
|
||||
zone = None
|
||||
if zone is not None:
|
||||
return zone
|
||||
try:
|
||||
import pytz
|
||||
return cast(tzinfo, pytz.timezone("America/New_York"))
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
|
||||
|
||||
_EASTERN = _eastern_zone()
|
||||
|
||||
# What _fetch_one_chunk returns for a month that came back at the cap.
|
||||
_CAPPED: Any = object()
|
||||
|
||||
_range_lock = threading.Lock()
|
||||
_ranges_rejected_until = 0.0
|
||||
# A process starts out assuming ranges are still rejected, as they have been
|
||||
# since 2026-09-15, and tries one again RANGE_RETRY_SECONDS in. Starting
|
||||
# from "unknown" cost one doomed range request per window at every start --
|
||||
# eleven 400s at once from a soccer board, each fetching before any had
|
||||
# answered -- to learn what every start learns.
|
||||
_ranges_rejected_until = time.monotonic() + RANGE_RETRY_SECONDS
|
||||
|
||||
__all__ = [
|
||||
"ESPN_MAX_LIMIT",
|
||||
"ESPN_CHUNK_WORKERS",
|
||||
"ESPN_MONTH_COVER_MIN_DAYS",
|
||||
"RANGE_RETRY_SECONDS",
|
||||
"clamp_espn_limit",
|
||||
"parse_espn_date_range",
|
||||
"espn_date_chunks",
|
||||
"espn_request_chunks",
|
||||
"merge_scoreboard_payloads",
|
||||
"fetch_espn_date_chunks",
|
||||
"fetch_espn_scoreboard",
|
||||
"ESPN_SCOREBOARD_URL",
|
||||
"espn_scoreboard_url",
|
||||
"espn_scoreboard_cache_key",
|
||||
"espn_scoreboard_cache_key_for_url",
|
||||
"read_espn_scoreboard_cache",
|
||||
"store_espn_scoreboard_cache",
|
||||
"get_espn_scoreboard",
|
||||
]
|
||||
|
||||
#: The site-API scoreboard every sport and league shares.
|
||||
ESPN_SCOREBOARD_URL = "https://site.api.espn.com/apis/site/v2/sports/{sport}/{league}/scoreboard"
|
||||
_ESPN_HOST_URL = "https://site.api.espn.com/"
|
||||
|
||||
_PATH_PART = re.compile(r"^[a-z0-9][a-z0-9.\-]*$")
|
||||
_DATES = re.compile(r"^\d{4}(?:\d{2}(?:\d{2})?)?$|^\d{8}-\d{8}$")
|
||||
_SCOREBOARD_PATH = re.compile(r"/sports/([^/?#]+)/([^/?#]+)/scoreboard/?$")
|
||||
|
||||
|
||||
def clamp_espn_limit(params: Optional[Dict[str, Any]]) -> Dict[str, Any]:
|
||||
"""Return a copy of ``params`` with any ``limit`` over 500 pulled back to 500."""
|
||||
@@ -113,6 +229,12 @@ def parse_espn_date_range(dates: Any) -> Optional[Tuple[date, date]]:
|
||||
return start, end
|
||||
|
||||
|
||||
def _memo_kwargs(cache_max_age: Optional[float]) -> Dict[str, Any]:
|
||||
"""``cache_max_age`` for fetch_get, only when the caller gave one, so a
|
||||
call that did not say is the call it always was."""
|
||||
return {} if cache_max_age is None else {"cache_max_age": cache_max_age}
|
||||
|
||||
|
||||
def _ranges_known_rejected() -> bool:
|
||||
with _range_lock:
|
||||
return time.monotonic() < _ranges_rejected_until
|
||||
@@ -159,6 +281,79 @@ def espn_date_chunks(start: date, end: date) -> List[str]:
|
||||
return chunks
|
||||
|
||||
|
||||
def espn_request_chunks(
|
||||
start: date,
|
||||
end: date,
|
||||
month_cover_min_days: Optional[int] = None,
|
||||
) -> List[Tuple[str, Optional[Tuple[date, date]]]]:
|
||||
"""The requests that fetch ``[start, end]``, as ``(dates, trim)`` pairs.
|
||||
|
||||
:func:`espn_date_chunks`, except that a partial edge month with
|
||||
``month_cover_min_days`` (default ``ESPN_MONTH_COVER_MIN_DAYS``) or more
|
||||
of its days in the window becomes one ``YYYYMM`` request whose ``trim``
|
||||
is the first and last of those days: its events that start outside them
|
||||
(US Eastern) are dropped. ``trim`` is None for every other request.
|
||||
Without time zone data nothing can be trimmed, so the edge days stay day
|
||||
requests.
|
||||
"""
|
||||
if month_cover_min_days is None:
|
||||
month_cover_min_days = ESPN_MONTH_COVER_MIN_DAYS
|
||||
planned: List[Tuple[str, Optional[Tuple[date, date]]]] = []
|
||||
run: List[str] = []
|
||||
|
||||
def flush() -> None:
|
||||
if (_EASTERN is not None and month_cover_min_days > 0
|
||||
and len(run) >= month_cover_min_days):
|
||||
planned.append((run[0][:6], (_parse_day(run[0]), _parse_day(run[-1]))))
|
||||
else:
|
||||
planned.extend((day, None) for day in run)
|
||||
run.clear()
|
||||
|
||||
for chunk in espn_date_chunks(start, end):
|
||||
if run and (len(chunk) != 8 or chunk[:6] != run[0][:6]):
|
||||
flush()
|
||||
if len(chunk) == 8:
|
||||
run.append(chunk)
|
||||
else:
|
||||
planned.append((chunk, None))
|
||||
flush()
|
||||
return planned
|
||||
|
||||
|
||||
def _parse_day(text: str) -> date:
|
||||
return date(int(text[:4]), int(text[4:6]), int(text[6:8]))
|
||||
|
||||
|
||||
def _eastern_day(stamp: Any) -> Optional[date]:
|
||||
"""The US Eastern date of an ESPN event ``date`` ("2026-10-10T11:30Z"),
|
||||
or None when it cannot be read."""
|
||||
if not isinstance(stamp, str) or _EASTERN is None:
|
||||
return None
|
||||
try:
|
||||
moment = datetime.fromisoformat(stamp.strip().replace("Z", "+00:00"))
|
||||
except ValueError:
|
||||
return None
|
||||
if moment.tzinfo is None:
|
||||
return None
|
||||
return moment.astimezone(_EASTERN).date()
|
||||
|
||||
|
||||
def _trim_to_days(payload: Any, first: date, last: date) -> Any:
|
||||
"""Drop the events of a month payload that start outside ``[first, last]``
|
||||
(US Eastern). An event whose date cannot be read is kept: its day query
|
||||
might well have returned it, and a game is never dropped on a guess.
|
||||
"""
|
||||
if not isinstance(payload, dict) or not isinstance(payload.get("events"), list):
|
||||
return payload
|
||||
kept = []
|
||||
for event in payload["events"]:
|
||||
day = _eastern_day(event.get("date")) if isinstance(event, dict) else None
|
||||
if day is None or first <= day <= last:
|
||||
kept.append(event)
|
||||
payload["events"] = kept
|
||||
return payload
|
||||
|
||||
|
||||
def merge_scoreboard_payloads(payloads: List[Any]) -> Dict[str, Any]:
|
||||
"""Fold chunk responses into one scoreboard payload.
|
||||
|
||||
@@ -188,51 +383,81 @@ def merge_scoreboard_payloads(payloads: List[Any]) -> Dict[str, Any]:
|
||||
|
||||
def _fetch_one_chunk(
|
||||
session, url: str, params: Dict[str, Any], headers, timeout, logger, chunk: str,
|
||||
) -> Optional[Dict[str, Any]]:
|
||||
cache_max_age: Optional[float] = None,
|
||||
trims: Optional[Dict[str, Tuple[date, date]]] = None,
|
||||
) -> Any:
|
||||
"""GET a single ``dates=`` chunk, or None when it failed.
|
||||
|
||||
One bad chunk must not sink the rest of the season, so every error is
|
||||
logged and swallowed here rather than raised to the gather below.
|
||||
|
||||
A month that comes back at the cap is truncated: it returns ``_CAPPED``,
|
||||
its payload dropped here before it is ever held beside the others. A
|
||||
month in ``trims`` loses its events outside the days given there.
|
||||
|
||||
The request holds one of the process-wide ``_chunk_slots`` while it runs.
|
||||
"""
|
||||
try:
|
||||
response = session.get(
|
||||
url,
|
||||
params=dict(params, dates=chunk, limit=ESPN_MAX_LIMIT),
|
||||
headers=headers,
|
||||
timeout=timeout,
|
||||
)
|
||||
response.raise_for_status()
|
||||
return cast(Optional[Dict[str, Any]], response_json(response))
|
||||
with _chunk_slots:
|
||||
response = fetch_get(
|
||||
session,
|
||||
url,
|
||||
params=dict(params, dates=chunk, limit=ESPN_MAX_LIMIT),
|
||||
headers=headers,
|
||||
timeout=timeout,
|
||||
**_memo_kwargs(cache_max_age),
|
||||
)
|
||||
response.raise_for_status()
|
||||
payload = response_json(response)
|
||||
except Exception as exc: # noqa: BLE001 - see docstring
|
||||
if logger:
|
||||
logger.warning("ESPN chunk %s failed, skipping it: %s", chunk, exc)
|
||||
return None
|
||||
if len(chunk) == 6 and isinstance(payload, dict):
|
||||
if len(payload.get("events") or []) >= ESPN_MAX_LIMIT:
|
||||
return _CAPPED
|
||||
trim = (trims or {}).get(chunk)
|
||||
if trim is not None:
|
||||
payload = _trim_to_days(payload, *trim)
|
||||
return payload
|
||||
|
||||
|
||||
def _fetch_chunks(
|
||||
session, url: str, params: Dict[str, Any], headers, timeout, logger,
|
||||
chunks: List[str],
|
||||
) -> List[Optional[Dict[str, Any]]]:
|
||||
chunks: List[str], cache_max_age: Optional[float] = None,
|
||||
trims: Optional[Dict[str, Tuple[date, date]]] = None,
|
||||
) -> List[Any]:
|
||||
"""Fetch every chunk, returning payloads positionally aligned with ``chunks``.
|
||||
|
||||
Requests go out ``ESPN_CHUNK_WORKERS`` at a time because a cold season is
|
||||
over a hundred of them. The order they come back in is not significant --
|
||||
over a hundred of them -- and no more than that across every window the
|
||||
process is fetching, which ``_fetch_one_chunk``'s slot enforces. The order they come back in is not significant --
|
||||
callers keep ``chunks`` order from the returned list -- but it does mean
|
||||
the session is shared across threads, which is why this only ever issues
|
||||
GETs and never touches session state.
|
||||
|
||||
Each chunk runs in a copy of the caller's context, with the caller pinned
|
||||
into it, so the fetch service counts the chunks against the plugin that
|
||||
asked for the range rather than against the core.
|
||||
"""
|
||||
if not chunks:
|
||||
return []
|
||||
fetch = partial(
|
||||
_fetch_one_chunk, session, url, params, headers, timeout, logger,
|
||||
cache_max_age=cache_max_age, trims=trims,
|
||||
)
|
||||
if len(chunks) == 1:
|
||||
return [fetch(chunks[0])]
|
||||
workers = min(ESPN_CHUNK_WORKERS, len(chunks))
|
||||
with pinned_caller():
|
||||
# One copy per chunk: a Context cannot be entered by two threads.
|
||||
contexts = [contextvars.copy_context() for _ in chunks]
|
||||
with ThreadPoolExecutor(
|
||||
max_workers=workers, thread_name_prefix="espn-chunk",
|
||||
) as pool:
|
||||
return list(pool.map(fetch, chunks))
|
||||
futures = [pool.submit(context.run, fetch, chunk)
|
||||
for context, chunk in zip(contexts, chunks)]
|
||||
return [future.result() for future in futures]
|
||||
|
||||
|
||||
def fetch_espn_date_chunks(
|
||||
@@ -242,6 +467,7 @@ def fetch_espn_date_chunks(
|
||||
headers: Optional[Dict[str, str]] = None,
|
||||
timeout: int = 15,
|
||||
logger=None,
|
||||
cache_max_age: Optional[float] = None,
|
||||
) -> Optional[Dict[str, Any]]:
|
||||
"""Fetch a ``YYYYMMDD-YYYYMMDD`` window as month and day chunks.
|
||||
|
||||
@@ -265,7 +491,9 @@ def fetch_espn_date_chunks(
|
||||
if span is None:
|
||||
return None
|
||||
|
||||
chunks = espn_date_chunks(*span)
|
||||
planned = espn_request_chunks(*span)
|
||||
chunks = [chunk for chunk, _ in planned]
|
||||
trims = {chunk: trim for chunk, trim in planned if trim is not None}
|
||||
if logger:
|
||||
logger.debug(
|
||||
"Fetching ESPN date range %s as %d month/day chunks",
|
||||
@@ -273,39 +501,38 @@ def fetch_espn_date_chunks(
|
||||
)
|
||||
|
||||
results = _fetch_chunks(
|
||||
session, url, params, headers, timeout, logger, chunks,
|
||||
session, url, params, headers, timeout, logger, chunks, cache_max_age,
|
||||
trims,
|
||||
)
|
||||
attempted = len(chunks)
|
||||
|
||||
# A month that came back at the cap is truncated; its days replace it in
|
||||
# place, so merged events stay in chunk order however the requests raced.
|
||||
# A month that came back at the cap is truncated; its days (only the
|
||||
# window's, for a trimmed edge month) replace it in place, so merged
|
||||
# events stay in chunk order however the requests raced. Its payload was
|
||||
# already dropped in the worker: a capped college-baseball month is ~2MB
|
||||
# of parsed JSON, and holding four of them through ~120 day requests added
|
||||
# ~25MB to the peak -- more than the concurrency itself. Low-memory boards
|
||||
# (docs/LOW_MEMORY_BOARDS.md) have under 200MB of headroom.
|
||||
slots: List[Any] = results
|
||||
capped: Dict[int, List[str]] = {}
|
||||
for index, chunk in enumerate(chunks):
|
||||
payload = slots[index]
|
||||
if payload is None or len(chunk) != 6:
|
||||
if slots[index] is not _CAPPED:
|
||||
continue
|
||||
events = payload.get("events") if isinstance(payload, dict) else None
|
||||
if len(events or []) >= ESPN_MAX_LIMIT:
|
||||
if logger:
|
||||
logger.info(
|
||||
"ESPN month %s hit the %d-event cap; re-asking it day by day",
|
||||
chunk, ESPN_MAX_LIMIT,
|
||||
)
|
||||
capped[index] = _days_of_month(chunk)
|
||||
# Drop the truncated month now rather than after its days arrive:
|
||||
# a capped college-baseball month is ~2MB of parsed JSON, and
|
||||
# holding four of them through ~120 day requests added ~25MB to
|
||||
# the peak -- more than the concurrency itself. Low-memory boards
|
||||
# (docs/LOW_MEMORY_BOARDS.md) have under 200MB of headroom.
|
||||
slots[index] = None
|
||||
payload = events = None
|
||||
if logger:
|
||||
logger.info(
|
||||
"ESPN month %s hit the %d-event cap; re-asking it day by day",
|
||||
chunk, ESPN_MAX_LIMIT,
|
||||
)
|
||||
trim = trims.get(chunk)
|
||||
capped[index] = (_days_of_month(chunk) if trim is None
|
||||
else espn_date_chunks(*trim))
|
||||
slots[index] = None
|
||||
|
||||
if capped:
|
||||
days = [day for index in sorted(capped) for day in capped[index]]
|
||||
attempted += len(days)
|
||||
by_day = dict(zip(days, _fetch_chunks(
|
||||
session, url, params, headers, timeout, logger, days,
|
||||
session, url, params, headers, timeout, logger, days, cache_max_age,
|
||||
)))
|
||||
for index, month_days in capped.items():
|
||||
slots[index] = [by_day.get(day) for day in month_days]
|
||||
@@ -338,6 +565,7 @@ def fetch_espn_scoreboard(
|
||||
headers: Optional[Dict[str, str]] = None,
|
||||
timeout: int = 15,
|
||||
logger=None,
|
||||
cache_max_age: Optional[float] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""GET an ESPN scoreboard, re-asking in month/day chunks if a range 400s.
|
||||
|
||||
@@ -347,6 +575,10 @@ def fetch_espn_scoreboard(
|
||||
and later ranges go straight to chunks for ``RANGE_RETRY_SECONDS``. A 400 on
|
||||
a non-range request, any other error, and a range whose every chunk fails
|
||||
all raise as before.
|
||||
|
||||
``cache_max_age`` is the oldest response, in seconds, the caller takes
|
||||
from the fetch service's short response cache (its own TTL; 0 always
|
||||
asks ESPN). None leaves it to the service default.
|
||||
"""
|
||||
params = clamp_espn_limit(params)
|
||||
is_range = parse_espn_date_range(params.get("dates")) is not None
|
||||
@@ -355,7 +587,7 @@ def fetch_espn_scoreboard(
|
||||
if is_range and _ranges_known_rejected():
|
||||
data = fetch_espn_date_chunks(
|
||||
session, url, params=params, headers=headers,
|
||||
timeout=timeout, logger=logger,
|
||||
timeout=timeout, logger=logger, cache_max_age=cache_max_age,
|
||||
)
|
||||
if data is not None:
|
||||
return data
|
||||
@@ -363,7 +595,8 @@ def fetch_espn_scoreboard(
|
||||
# real error to log, without spending the chunks a second time.
|
||||
chunks_tried = True
|
||||
|
||||
response = session.get(url, params=params, headers=headers, timeout=timeout)
|
||||
response = fetch_get(session, url, params=params, headers=headers, timeout=timeout,
|
||||
**_memo_kwargs(cache_max_age))
|
||||
if is_range and response.status_code == 400 and not chunks_tried:
|
||||
_note_range_rejected()
|
||||
if logger:
|
||||
@@ -374,9 +607,225 @@ def fetch_espn_scoreboard(
|
||||
)
|
||||
data = fetch_espn_date_chunks(
|
||||
session, url, params=params, headers=headers,
|
||||
timeout=timeout, logger=logger,
|
||||
timeout=timeout, logger=logger, cache_max_age=cache_max_age,
|
||||
)
|
||||
if data is not None:
|
||||
return data
|
||||
response.raise_for_status()
|
||||
return cast(Dict[str, Any], response_json(response))
|
||||
|
||||
|
||||
# --- one cache key per scoreboard --------------------------------------------------
|
||||
|
||||
def espn_scoreboard_url(sport: str, league: str) -> str:
|
||||
"""The site-API scoreboard URL for an ESPN ``sport`` / ``league`` path."""
|
||||
return ESPN_SCOREBOARD_URL.format(sport=_path_part(sport, "sport"),
|
||||
league=_path_part(league, "league"))
|
||||
|
||||
|
||||
def _path_part(value: Any, what: str) -> str:
|
||||
text = str(value or "").strip().lower()
|
||||
if not _PATH_PART.match(text):
|
||||
raise ValueError(f"not an ESPN {what} path segment: {value!r}")
|
||||
return text
|
||||
|
||||
|
||||
def _day(value: Any) -> str:
|
||||
if isinstance(value, (date, datetime)):
|
||||
return value.strftime("%Y%m%d")
|
||||
text = str(value).strip()
|
||||
if len(text) != 8 or not text.isdigit():
|
||||
raise ValueError(f"not an ESPN day (YYYYMMDD): {value!r}")
|
||||
return text
|
||||
|
||||
|
||||
def _dates_part(dates: Any) -> str:
|
||||
"""``dates`` as ESPN spells it, or ``current`` for no ``dates`` at all."""
|
||||
if dates is None or dates == "":
|
||||
return "current"
|
||||
if isinstance(dates, (date, datetime)):
|
||||
return _day(dates)
|
||||
if isinstance(dates, (tuple, list)):
|
||||
if len(dates) != 2:
|
||||
raise ValueError(f"a date range is (start, end): {dates!r}")
|
||||
start, end = _day(dates[0]), _day(dates[1])
|
||||
return start if start == end else f"{start}-{end}"
|
||||
text = str(dates).strip()
|
||||
if isinstance(dates, bool) or not _DATES.match(text):
|
||||
raise ValueError(
|
||||
f"not an ESPN dates value (YYYY, YYYYMM, YYYYMMDD or "
|
||||
f"YYYYMMDD-YYYYMMDD): {dates!r}")
|
||||
return text
|
||||
|
||||
|
||||
def espn_scoreboard_cache_key(sport: str, league: str, dates: Any = None) -> str:
|
||||
"""The one cache key for an ESPN scoreboard, whoever caches it.
|
||||
|
||||
``sport`` and ``league`` are ESPN's own path segments -- ``football`` /
|
||||
``college-football``, ``soccer`` / ``eng.1`` -- not a plugin's
|
||||
``sport_key``, so every plugin showing a league names it the same way.
|
||||
``dates`` is what the request sends as ``dates=``: ``"YYYYMMDD"``,
|
||||
``"YYYYMM"``, ``"YYYY"``, ``"YYYYMMDD-YYYYMMDD"``, a ``date``, or a
|
||||
``(start, end)`` pair of either; None is the undated "current"
|
||||
scoreboard. Anything else raises ValueError rather than invent a key.
|
||||
|
||||
The key says nothing about ``limit``: a cached copy is meant to be a
|
||||
whole one (the helpers here always ask for ``ESPN_MAX_LIMIT``).
|
||||
"""
|
||||
return (f"espn_scoreboard_{_path_part(sport, 'sport')}_"
|
||||
f"{_path_part(league, 'league')}_{_dates_part(dates)}")
|
||||
|
||||
|
||||
def espn_scoreboard_cache_key_for_url(url: str, dates: Any = None) -> Optional[str]:
|
||||
""":func:`espn_scoreboard_cache_key` for a scoreboard URL, or None when
|
||||
``url`` is not ``.../sports/{sport}/{league}/scoreboard``."""
|
||||
match = _SCOREBOARD_PATH.search(str(url or "").split("?", 1)[0])
|
||||
if match is None:
|
||||
return None
|
||||
try:
|
||||
return espn_scoreboard_cache_key(match.group(1), match.group(2), dates)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
def _note_cache_hit(legacy: bool, avoided_request: bool = True) -> None:
|
||||
if not _COUNTS_FETCHES:
|
||||
return
|
||||
try:
|
||||
get_fetch_service().note_cache_hit(
|
||||
_ESPN_HOST_URL, legacy=legacy, avoided_request=avoided_request)
|
||||
except Exception: # noqa: BLE001 - counting never breaks a read
|
||||
_logger.debug("could not count a scoreboard cache hit", exc_info=True)
|
||||
|
||||
|
||||
def _fresh_cached(cache_manager: Any, key: str, max_age: Optional[float],
|
||||
now: float) -> Tuple[Optional[Dict[str, Any]], Optional[float]]:
|
||||
"""The data cached under ``key`` if it is at most ``max_age`` seconds
|
||||
old, and its age (None when the cache does not say).
|
||||
|
||||
The age is the stored record's own timestamp, checked here: CacheManager
|
||||
lets a ttl stored by the writer override the reader's max_age, and its
|
||||
memory tier times an entry from when it was loaded, not written. A key
|
||||
shared by readers with different TTLs can rely on neither.
|
||||
"""
|
||||
reader = getattr(cache_manager, "get_cached_data", None)
|
||||
limit = None if max_age is None else max(1, int(math.ceil(max_age)))
|
||||
if not callable(reader):
|
||||
# A cache without records (a test double, a plugin's own store).
|
||||
value = cache_manager.get(key, max_age=limit)
|
||||
return (value if isinstance(value, dict) else None), None
|
||||
record = reader(key, max_age=limit, memory_ttl=limit)
|
||||
if not isinstance(record, dict):
|
||||
return None, None
|
||||
if "data" not in record:
|
||||
return record, None # unwrapped; the cache already judged it by mtime
|
||||
stamp = record.get("timestamp")
|
||||
age: Optional[float] = None
|
||||
if not isinstance(stamp, bool) and isinstance(stamp, (int, float)):
|
||||
age = max(0.0, now - float(stamp))
|
||||
if max_age is not None and (age is None or age > max_age):
|
||||
return None, None
|
||||
data = record["data"]
|
||||
return (data if isinstance(data, dict) else None), age
|
||||
|
||||
|
||||
def read_espn_scoreboard_cache(
|
||||
cache_manager: Any,
|
||||
key: str,
|
||||
max_age: Optional[float],
|
||||
legacy_keys: Iterable[str] = (),
|
||||
now: Optional[float] = None,
|
||||
accept: Optional[Callable[[Dict[str, Any], Optional[float]], bool]] = None,
|
||||
) -> Optional[Dict[str, Any]]:
|
||||
"""The cached scoreboard under ``key``, or under the first of
|
||||
``legacy_keys`` that has one, if it is at most ``max_age`` seconds old.
|
||||
|
||||
None on a miss, a stale entry, ``max_age`` of 0 or less, no cache
|
||||
manager, or any cache error -- a read never raises. ``max_age=None``
|
||||
takes an entry of any age. ``accept(data, age_seconds)`` can turn down
|
||||
an entry the age alone would allow (a payload holding a live game wants
|
||||
a shorter limit); ``age_seconds`` is None when the cache cannot say. A
|
||||
hit is counted in the fetch statistics (``cache_hits``;
|
||||
``legacy_cache_hits`` too for an old key).
|
||||
"""
|
||||
if cache_manager is None:
|
||||
return None
|
||||
if max_age is not None and max_age <= 0:
|
||||
return None
|
||||
clock = time.time() if now is None else now
|
||||
for index, candidate in enumerate([key, *legacy_keys]):
|
||||
if not candidate:
|
||||
continue
|
||||
try:
|
||||
data, age = _fresh_cached(cache_manager, candidate, max_age, clock)
|
||||
if data is not None and accept is not None and not accept(data, age):
|
||||
data = None
|
||||
except Exception: # noqa: BLE001 - a broken cache is a miss
|
||||
_logger.debug("scoreboard cache read failed for %s", candidate, exc_info=True)
|
||||
continue
|
||||
if data is not None:
|
||||
_note_cache_hit(legacy=index > 0)
|
||||
return data
|
||||
return None
|
||||
|
||||
|
||||
def store_espn_scoreboard_cache(cache_manager: Any, key: str, data: Any) -> None:
|
||||
"""Cache a fetched scoreboard under ``key``. Never raises.
|
||||
|
||||
No ttl is stored: each reader applies its own ``max_age`` (a live
|
||||
reader 30 s, a schedule reader an hour), and a stored ttl would
|
||||
override theirs in CacheManager.
|
||||
"""
|
||||
if cache_manager is None or data is None:
|
||||
return
|
||||
try:
|
||||
cache_manager.set(key, data)
|
||||
except Exception: # noqa: BLE001 - the caller still has its data
|
||||
_logger.warning("Could not cache scoreboard %s", key, exc_info=True)
|
||||
|
||||
|
||||
def get_espn_scoreboard(
|
||||
session: Any,
|
||||
sport: str,
|
||||
league: str,
|
||||
dates: Any = None,
|
||||
*,
|
||||
cache_manager: Any = None,
|
||||
max_age: Optional[float] = 300,
|
||||
legacy_keys: Iterable[str] = (),
|
||||
headers: Optional[Dict[str, str]] = None,
|
||||
timeout: int = 15,
|
||||
logger: Any = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""An ESPN scoreboard through the shared cache, fetched on a miss.
|
||||
|
||||
Reads :func:`espn_scoreboard_cache_key` (then ``legacy_keys``) and
|
||||
returns an entry at most ``max_age`` seconds old. Otherwise it fetches
|
||||
with :func:`fetch_espn_scoreboard` -- ``limit=ESPN_MAX_LIMIT``, ranges
|
||||
split as ESPN needs -- caches the result under the canonical key and
|
||||
returns it. ``max_age=0`` always fetches (and still caches, for other
|
||||
readers). Errors raise exactly as :func:`fetch_espn_scoreboard` does,
|
||||
and nothing is cached then. ``session=None`` uses the fetch service's
|
||||
pooled session for the ESPN host.
|
||||
"""
|
||||
key = espn_scoreboard_cache_key(sport, league, dates)
|
||||
cached = read_espn_scoreboard_cache(cache_manager, key, max_age, legacy_keys)
|
||||
if cached is not None:
|
||||
return cast(Dict[str, Any], cached)
|
||||
params: Dict[str, Any] = {"limit": ESPN_MAX_LIMIT}
|
||||
spelled = _dates_part(dates)
|
||||
if spelled != "current":
|
||||
params["dates"] = spelled
|
||||
data = fetch_espn_scoreboard(
|
||||
session,
|
||||
espn_scoreboard_url(sport, league),
|
||||
params=params,
|
||||
headers=headers,
|
||||
timeout=timeout,
|
||||
logger=logger,
|
||||
# The response cache must not hand back anything older than the
|
||||
# cache read above would have accepted.
|
||||
cache_max_age=None if max_age is None else max(0.0, float(max_age)),
|
||||
)
|
||||
store_espn_scoreboard_cache(cache_manager, key, data)
|
||||
return data
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
+222
-15
@@ -19,8 +19,9 @@ Only intervals between two consecutive *scrolling* frames count: a static
|
||||
screen that changes once a second has no timing to get wrong, and the first
|
||||
frame of a scroll has no predecessor worth measuring against.
|
||||
|
||||
"Scrolling" is DisplayManager's scroll state when the frame is presented, and
|
||||
that state can go missing in the middle of a scroll. It expires after 2s
|
||||
"Scrolling" is the scroll state ``DisplayManager.update_display`` acted on for
|
||||
the frame, sampled once before the blit and swap, and that state can go missing
|
||||
in the middle of a scroll. It expires after 2s
|
||||
without scroll activity, which a long enough stall outlasts, and any thread can
|
||||
clear it: plugins call ``set_scrolling_state(False)`` from their own
|
||||
``display()``, and Vegas captures some of those on the render thread between
|
||||
@@ -38,12 +39,20 @@ after the one before it. One that arrives a whole refresh or more after that is
|
||||
a visible hitch. ``missed_refreshes`` sums how many refreshes late.
|
||||
|
||||
An interval of ``FREEZE_SECONDS`` or more is a **freeze** instead -- a
|
||||
recompose, a plugin handover, a blocking call on the render thread. Those are
|
||||
counted separately, both because they are a different fault and because
|
||||
folding a single 400ms handover into the late count as "40 missed refreshes"
|
||||
would drown the jitter the late count exists to measure. ``freeze_by`` splits
|
||||
them by length. Intervals of ``GAP_SECONDS`` or more are ignored as not being
|
||||
frames of one scroll at all.
|
||||
recompose, a plugin handover nobody tagged (see below), a blocking call on the
|
||||
render thread. Those are counted separately, both because they are a different
|
||||
fault and because folding a single 400ms handover into the late count as "40
|
||||
missed refreshes" would drown the jitter the late count exists to measure.
|
||||
``freeze_by`` splits them by length. Intervals of ``GAP_SECONDS`` or more are
|
||||
ignored as not being frames of one scroll at all.
|
||||
|
||||
One kind of freeze is not a scroll stalling at all: the gap from one screen's
|
||||
last frame to the next screen's first, while the next screen draws. The
|
||||
display controller tags that frame ``handover`` (see "Operations") at the
|
||||
start of every turn, the same mode's again included, and a tagged freeze is
|
||||
counted in ``handover_freezes`` instead of ``freezes`` and ``freeze_by``.
|
||||
Stats written before that field existed have handovers among their freezes,
|
||||
so freeze counts from before and after it are not comparable.
|
||||
|
||||
A frame that arrives a whole refresh or more *early* means the swap did not
|
||||
wait for the panel: the emulator, the fallback display, or a hold that was not
|
||||
@@ -77,6 +86,24 @@ the work landed in. ``op_frames`` counts timed frames per kind,
|
||||
freeze instead, and ``op_bytes`` what the work moved. A kind whose late rate
|
||||
sits well above the overall one is the work to look at.
|
||||
|
||||
``handover`` (:data:`HANDOVER_OP`) is noted off the render thread: the display
|
||||
controller notes it just before it starts a screen's first ``display()``,
|
||||
which presents from a thread of its own, and drops the note again with
|
||||
:meth:`FrameTimingRecorder.drop_op` once that call returns, so a first
|
||||
``display()`` that drew nothing cannot leave the tag for an unrelated frame.
|
||||
|
||||
Garbage collection
|
||||
------------------
|
||||
|
||||
Python's cyclic collector stops every thread while it runs. :class:`GcMonitor`
|
||||
times each collection from ``gc.callbacks``; the display manager installs one
|
||||
per process. A collection of ``GC_PAUSE_SECONDS`` or more tags the next
|
||||
presented frame ``gc`` (:data:`GC_OP`), so it shows in ``op_frames``,
|
||||
``late_op_frames`` and ``op_freezes`` like noted work, and the snapshot carries
|
||||
a ``gc`` block of cumulative counters: collections and seconds per generation,
|
||||
the longest, and the long ones. A stall dump says when a long collection ran
|
||||
inside the stall. Diagnostic only: nothing tunes or freezes the collector.
|
||||
|
||||
Stall watchdog
|
||||
--------------
|
||||
Counting a freeze says that it happened, not why. ``StallWatchdog`` watches the
|
||||
@@ -94,7 +121,9 @@ three times per threshold, so keep it to diagnostic runs, not soaks.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import atexit
|
||||
import copy
|
||||
import gc
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
@@ -133,6 +162,16 @@ RESUME_SECONDS = 1.0
|
||||
FREEZE_BUCKETS = ((0.5, "<0.5s"), (1.0, "0.5-1s"), (2.0, "1-2s"),
|
||||
(float("inf"), "2s+"))
|
||||
|
||||
#: The op the display controller notes before a screen's first frame. A
|
||||
#: freeze it ends is a handover, counted apart from the freezes; see
|
||||
#: "What is counted".
|
||||
HANDOVER_OP = "handover"
|
||||
|
||||
#: The op a garbage collection of ``GC_PAUSE_SECONDS`` or more tags the next
|
||||
#: frame with; see "Garbage collection".
|
||||
GC_OP = "gc"
|
||||
GC_PAUSE_SECONDS = 0.020
|
||||
|
||||
#: A window may lower the refresh-period estimate by at most this fraction.
|
||||
MAX_REFRESH_DROP = 0.2
|
||||
|
||||
@@ -164,6 +203,114 @@ def default_stats_path() -> str:
|
||||
return os.path.join(base, STATS_FILENAME)
|
||||
|
||||
|
||||
class GcMonitor:
|
||||
"""Times every garbage collection, from ``gc.callbacks``.
|
||||
|
||||
Python's cyclic collector stops every thread for as long as a collection
|
||||
takes, and a full one over a large heap (a season of game dicts) can take
|
||||
longer than a frame. Nothing measured that, so a stall it caused looked
|
||||
like any other. The callback runs inside the collection, with the GIL
|
||||
held, and collections never overlap, so these plain counters need no
|
||||
lock: the render thread and the stats writer only read them.
|
||||
|
||||
Install it once per process with :func:`install_gc_monitor`.
|
||||
|
||||
Collections still run while the interpreter shuts down, after module
|
||||
globals such as ``time`` may already be torn down to ``None``. The clock
|
||||
and ``sys.is_finalizing`` are bound here so the callback never looks a
|
||||
global up, it does nothing once finalization has begun, and
|
||||
:func:`install_gc_monitor` unregisters it at exit anyway.
|
||||
"""
|
||||
|
||||
def __init__(self, threshold: float = GC_PAUSE_SECONDS,
|
||||
clock: Callable[[], float] = time.perf_counter):
|
||||
self.threshold = threshold
|
||||
self._clock = clock
|
||||
self._is_finalizing = sys.is_finalizing
|
||||
self._started: Optional[float] = None
|
||||
#: Per generation (0, 1, 2), since the monitor was installed.
|
||||
self.collections = [0, 0, 0]
|
||||
self.seconds = [0.0, 0.0, 0.0]
|
||||
self.max_seconds = 0.0
|
||||
#: Collections of ``threshold`` or more, and their total length. The
|
||||
#: recorder compares ``long_pauses`` with the count it last saw to tag
|
||||
#: the next frame.
|
||||
self.long_pauses = 0
|
||||
self.long_seconds = 0.0
|
||||
#: ``time.perf_counter()`` at the end of the last long collection,
|
||||
#: and its length, for the stall watchdog.
|
||||
self.last_long: Optional[Tuple[float, float]] = None
|
||||
|
||||
def __call__(self, phase: str, info: Dict[str, Any]) -> None:
|
||||
if self._is_finalizing():
|
||||
return
|
||||
now = self._clock()
|
||||
if phase == "start":
|
||||
self._started = now
|
||||
return
|
||||
started, self._started = self._started, None
|
||||
if started is None:
|
||||
return
|
||||
took = now - started
|
||||
generation = min(max(int(info.get("generation", 0)), 0), 2)
|
||||
self.collections[generation] += 1
|
||||
self.seconds[generation] += took
|
||||
if took > self.max_seconds:
|
||||
self.max_seconds = took
|
||||
if took >= self.threshold:
|
||||
self.long_seconds += took
|
||||
self.last_long = (now, took)
|
||||
self.long_pauses += 1
|
||||
|
||||
def snapshot(self) -> Dict[str, Any]:
|
||||
"""Cumulative counters for the stats file (all since installation)."""
|
||||
return {
|
||||
"threshold_ms": round(self.threshold * 1000.0, 3),
|
||||
"collections": list(self.collections),
|
||||
"seconds": [round(x, 6) for x in self.seconds],
|
||||
"max_ms": round(self.max_seconds * 1000.0, 3),
|
||||
"long_pauses": self.long_pauses,
|
||||
"long_seconds": round(self.long_seconds, 6),
|
||||
}
|
||||
|
||||
|
||||
_gc_monitor: Optional[GcMonitor] = None
|
||||
_gc_monitor_lock = threading.Lock()
|
||||
|
||||
|
||||
def install_gc_monitor() -> GcMonitor:
|
||||
"""The process's GcMonitor, installed in ``gc.callbacks`` on first call.
|
||||
|
||||
It is unregistered at exit (:func:`uninstall_gc_monitor`), before the
|
||||
interpreter tears module globals down.
|
||||
"""
|
||||
global _gc_monitor
|
||||
with _gc_monitor_lock:
|
||||
if _gc_monitor is None:
|
||||
_gc_monitor = GcMonitor()
|
||||
gc.callbacks.append(_gc_monitor)
|
||||
atexit.register(uninstall_gc_monitor)
|
||||
return _gc_monitor
|
||||
|
||||
|
||||
def uninstall_gc_monitor() -> None:
|
||||
"""Take the process's GcMonitor out of ``gc.callbacks``; safe to repeat.
|
||||
|
||||
A recorder that still holds the monitor keeps its counters; they just
|
||||
stop moving. The next :func:`install_gc_monitor` installs a fresh one.
|
||||
"""
|
||||
global _gc_monitor
|
||||
with _gc_monitor_lock:
|
||||
monitor, _gc_monitor = _gc_monitor, None
|
||||
if monitor is None:
|
||||
return
|
||||
atexit.unregister(uninstall_gc_monitor)
|
||||
try:
|
||||
gc.callbacks.remove(monitor)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
|
||||
#: One presented frame's interval: (interval, blit, wait, hold, ops), where
|
||||
#: ops is the work noted before it (kind -> bytes) or None.
|
||||
_Frame = Tuple[float, float, float, int, Optional[Dict[str, int]]]
|
||||
@@ -259,11 +406,15 @@ class FrameTimingRecorder:
|
||||
flush_interval: float = FLUSH_INTERVAL,
|
||||
info: Optional[Dict[str, Any]] = None,
|
||||
refresh_hz: Optional[float] = None,
|
||||
gc_monitor: Optional[GcMonitor] = None,
|
||||
):
|
||||
"""
|
||||
:param refresh_hz: the panel's rate, measured independently (see the
|
||||
module docstring). Omit it to estimate from the frames alone, as
|
||||
the display service does.
|
||||
:param gc_monitor: tags frames after a long garbage collection and
|
||||
adds its counters to the stats (see "Garbage collection"). The
|
||||
display manager passes the process's :func:`install_gc_monitor`.
|
||||
"""
|
||||
self.path = path or default_stats_path()
|
||||
self.flush_interval = flush_interval
|
||||
@@ -278,6 +429,8 @@ class FrameTimingRecorder:
|
||||
self._unsure: Optional[_Frame] = None
|
||||
# Work noted since the last frame (kind -> bytes), for the next one.
|
||||
self._ops: Optional[Dict[str, int]] = None
|
||||
self.gc_monitor = gc_monitor
|
||||
self._gc_seen = gc_monitor.long_pauses if gc_monitor is not None else 0
|
||||
self._last_flush: Optional[float] = None
|
||||
self._queue: "queue.SimpleQueue" = queue.SimpleQueue()
|
||||
self._worker: Optional[threading.Thread] = None
|
||||
@@ -302,6 +455,10 @@ class FrameTimingRecorder:
|
||||
"freezes": 0,
|
||||
"freeze_seconds": 0.0,
|
||||
"freeze_by": {label: 0 for _, label in FREEZE_BUCKETS},
|
||||
# Freezes that ended a screen handover rather than stalled a
|
||||
# scroll: in neither of the two above. Additive; see "What is
|
||||
# counted".
|
||||
"handover_freezes": 0,
|
||||
"worst_interval_ms": 0.0,
|
||||
# Per kind of noted render-thread work; see "Operations".
|
||||
"op_frames": {},
|
||||
@@ -337,7 +494,8 @@ class FrameTimingRecorder:
|
||||
|
||||
Render thread only, like :meth:`record`, which consumes the tag: the
|
||||
interval the next frame ends is the one this work landed in. Several
|
||||
notes before one frame accumulate, per kind. See "Operations".
|
||||
notes before one frame accumulate, per kind. See "Operations" (and
|
||||
:data:`HANDOVER_OP`, the one note made from another thread).
|
||||
|
||||
:param kind: a short name for the work, e.g. ``"extend"``, ``"patch"``.
|
||||
:param nbytes: how much the work moved, summed into ``op_bytes``.
|
||||
@@ -347,6 +505,21 @@ class FrameTimingRecorder:
|
||||
ops = self._ops = {}
|
||||
ops[kind] = ops.get(kind, 0) + int(nbytes)
|
||||
|
||||
def drop_op(self, kind: str) -> None:
|
||||
"""Forget a note of ``kind`` that no frame has carried yet.
|
||||
|
||||
For work that may present nothing: the display controller notes a
|
||||
handover before a screen's first ``display()`` and drops it once that
|
||||
returns. When the call drew a frame, the frame already took the tag
|
||||
and this does nothing; when it drew nothing (no content), the tag
|
||||
would otherwise land on whatever frame came next -- seconds or minutes
|
||||
later, and nothing to do with the handover. Other kinds noted for the
|
||||
same frame are kept.
|
||||
"""
|
||||
ops = self._ops
|
||||
if ops is not None:
|
||||
ops.pop(kind, None)
|
||||
|
||||
def record(self, blit: float, wait: float, hold: int, scrolling: bool,
|
||||
presented_at: float) -> None:
|
||||
"""One frame reached the panel.
|
||||
@@ -354,13 +527,24 @@ class FrameTimingRecorder:
|
||||
:param blit: seconds spent copying the frame into the canvas.
|
||||
:param wait: seconds SwapOnVSync blocked.
|
||||
:param hold: the refreshes this frame was held for.
|
||||
:param scrolling: whether a scroll was running when it was presented.
|
||||
:param scrolling: whether a scroll was running for this frame: the
|
||||
scroll state ``update_display`` acted on, sampled once before the
|
||||
blit and swap.
|
||||
:param presented_at: ``time.perf_counter()`` when the swap returned.
|
||||
"""
|
||||
previous = self._previous
|
||||
self._previous = (presented_at, scrolling, hold)
|
||||
self.last_frame = (presented_at, scrolling, threading.get_ident())
|
||||
ops, self._ops = self._ops, None
|
||||
monitor = self.gc_monitor
|
||||
if monitor is not None and monitor.long_pauses != self._gc_seen:
|
||||
# A long collection ran since the last frame: the interval this
|
||||
# frame ends is the one it landed in. Read here rather than
|
||||
# noted, since note_op is the render thread's and a collection
|
||||
# runs on whichever thread triggered it.
|
||||
self._gc_seen = monitor.long_pauses
|
||||
ops = dict(ops) if ops else {}
|
||||
ops[GC_OP] = ops.get(GC_OP, 0)
|
||||
if not scrolling:
|
||||
self._static_frames += 1
|
||||
# The scroll ended, or its state went missing for this frame: the
|
||||
@@ -467,13 +651,19 @@ class FrameTimingRecorder:
|
||||
for kind, nbytes in ops.items():
|
||||
_bump(totals["op_bytes"], kind, nbytes)
|
||||
if interval >= FREEZE_SECONDS:
|
||||
for kind in ops or ():
|
||||
_bump(totals["op_freezes"], kind)
|
||||
if ops and HANDOVER_OP in ops:
|
||||
# The next screen drawing its first frame, not a scroll
|
||||
# that stalled: counted apart, so the freezes keep
|
||||
# meaning the second. See "What is counted".
|
||||
totals["handover_freezes"] += 1
|
||||
continue
|
||||
totals["freezes"] += 1
|
||||
totals["freeze_seconds"] += interval
|
||||
label = next(name for limit, name in FREEZE_BUCKETS
|
||||
if interval < limit)
|
||||
totals["freeze_by"][label] += 1
|
||||
for kind in ops or ():
|
||||
_bump(totals["op_freezes"], kind)
|
||||
continue
|
||||
totals["scroll_frames"] += 1
|
||||
for name, value in (("blit", blit), ("wait", wait),
|
||||
@@ -517,6 +707,9 @@ class FrameTimingRecorder:
|
||||
"binding_releases_gil": self._binding_gil,
|
||||
"info": info,
|
||||
"totals": copy.deepcopy(self.totals),
|
||||
# Additive: absent from older files and when no monitor is set.
|
||||
**({"gc": self.gc_monitor.snapshot()}
|
||||
if self.gc_monitor is not None else {}),
|
||||
# JSON keys are strings; readers convert back.
|
||||
"histograms": {name: {str(k): v for k, v in sorted(h.items())}
|
||||
for name, h in self.histograms.items()},
|
||||
@@ -647,14 +840,28 @@ class StallWatchdog:
|
||||
return stall_from, dumped
|
||||
|
||||
def describe(self, ident: int, age: float, late: float) -> str:
|
||||
"""The stack dump: the stalled thread in full, the rest in brief."""
|
||||
"""The stack dump: the stalled thread in full, the rest in brief.
|
||||
|
||||
A stall while a ``handover`` note is still waiting for its frame is
|
||||
the next screen's first ``display()`` taking its time, not a scroll
|
||||
that stopped, and is labelled a handover gap.
|
||||
"""
|
||||
names = {t.ident: t.name for t in threading.enumerate()}
|
||||
frames = sys._current_frames()
|
||||
pending = getattr(self.recorder, "_ops", None)
|
||||
where = ("in a handover gap" if pending and HANDOVER_OP in pending
|
||||
else "mid-scroll")
|
||||
monitor = getattr(self.recorder, "gc_monitor", None)
|
||||
last_long = getattr(monitor, "last_long", None)
|
||||
gc_note = ""
|
||||
if last_long is not None and time.perf_counter() - last_long[0] <= age:
|
||||
gc_note = (f"; a {last_long[1] * 1000.0:.0f}ms garbage collection "
|
||||
"ran inside it")
|
||||
lines = [
|
||||
f"Render stall: no frame for {age * 1000.0:.0f}ms mid-scroll "
|
||||
f"Render stall: no frame for {age * 1000.0:.0f}ms {where} "
|
||||
f"(watchdog woke {late * 1000.0:.0f}ms late"
|
||||
+ ("; the interpreter itself was blocked" if late >= age / 2 else "")
|
||||
+ ")",
|
||||
+ gc_note + ")",
|
||||
f"-- {names.get(ident, ident)} (presents frames):",
|
||||
]
|
||||
stalled = frames.get(ident)
|
||||
|
||||
@@ -157,7 +157,13 @@ def crisp_ladder(
|
||||
#: when 30 was asked for -- being 11% slow is worth far less than looking bad.
|
||||
_STEP_PENALTY = 0.05
|
||||
_SLOW_FPS_PENALTY = 0.25 # below 20fps
|
||||
_LOWISH_FPS_PENALTY = 0.10 # below 25fps
|
||||
_LOWISH_FPS_PENALTY = 0.16 # below 30fps, i.e. "slightly stepped"
|
||||
# Up to 30fps, matching CrispSpeed.steppiness: a measured 125.7Hz panel makes
|
||||
# 50.3px/s (2px every 5 refreshes) 25.1fps, which a 25fps cutoff let through.
|
||||
# 0.16, not less: asked for 50px/s on a 120Hz panel, 48px/s (2px every 5
|
||||
# refreshes, 24fps) costs 0.04 + 0.05 + this, and has to lose to both 60px/s
|
||||
# and 40px/s (1px, smooth, 20% off = 0.20). At 0.10 it won and shipped a
|
||||
# visibly stepped scroll to anyone asking for the default.
|
||||
|
||||
|
||||
def _quality_cost(candidate: "CrispSpeed", target: float) -> float:
|
||||
@@ -173,7 +179,7 @@ def _quality_cost(candidate: "CrispSpeed", target: float) -> float:
|
||||
fps = candidate.frames_per_second
|
||||
if fps < 20:
|
||||
cost += _SLOW_FPS_PENALTY
|
||||
elif fps < 25:
|
||||
elif fps < 30:
|
||||
cost += _LOWISH_FPS_PENALTY
|
||||
return cost
|
||||
|
||||
@@ -474,3 +480,61 @@ def refresh_hz_from_config(global_config: Optional[Dict[str, Any]]) -> float:
|
||||
if not isinstance(hardware, dict):
|
||||
return DEFAULT_REFRESH_HZ
|
||||
return _coerce(hardware.get("limit_refresh_rate_hz")) or DEFAULT_REFRESH_HZ
|
||||
|
||||
|
||||
#: Smooth options offered next to a speed that is not one itself.
|
||||
_ADVICE_ALTERNATIVES = 2
|
||||
|
||||
|
||||
def speed_advice(
|
||||
requested_pixels_per_second: float,
|
||||
refresh_hz: float,
|
||||
min_pixels_per_second: float = MIN_PIXELS_PER_SECOND,
|
||||
max_pixels_per_second: float = MAX_PIXELS_PER_SECOND,
|
||||
) -> Dict[str, Any]:
|
||||
"""What the panel will do with a requested speed, for showing in a UI.
|
||||
|
||||
``applied`` is what :func:`solve_crisp` picks, i.e. what really runs.
|
||||
``smooth`` is true when that is single-pixel-ish, 30fps-or-better motion.
|
||||
``alternatives`` are the smooth ladder entries nearest the request inside
|
||||
the given range, for a click-to-apply suggestion; empty when the request
|
||||
already is one.
|
||||
"""
|
||||
hz = _coerce(refresh_hz) or DEFAULT_REFRESH_HZ
|
||||
requested = max(MIN_PIXELS_PER_SECOND,
|
||||
min(MAX_PIXELS_PER_SECOND, _coerce(requested_pixels_per_second) or 0.0))
|
||||
applied = solve_crisp(requested, hz)
|
||||
|
||||
def as_dict(c: CrispSpeed) -> Dict[str, Any]:
|
||||
return {
|
||||
"pixels_per_second": round(c.pixels_per_second, 1),
|
||||
"pixels_per_frame": c.pixels_per_frame,
|
||||
"frame_hold": c.frame_hold,
|
||||
"frames_per_second": round(c.frames_per_second, 1),
|
||||
"steppiness": c.steppiness,
|
||||
}
|
||||
|
||||
smooth_ladder = [
|
||||
c for c in crisp_ladder(hz)
|
||||
if c.steppiness == "smooth"
|
||||
and min_pixels_per_second <= c.pixels_per_second <= max_pixels_per_second
|
||||
]
|
||||
# 2%: a UI hands over whole numbers, and 63 asked of a 62.9 px/s panel is
|
||||
# as good as exact.
|
||||
exact = abs(applied.pixels_per_second - requested) <= max(0.05, 0.02 * requested)
|
||||
smooth = applied.steppiness == "smooth"
|
||||
alternatives: List[CrispSpeed] = []
|
||||
if not (exact and smooth):
|
||||
alternatives = sorted(
|
||||
smooth_ladder,
|
||||
key=lambda c: abs(c.pixels_per_second - requested),
|
||||
)[:_ADVICE_ALTERNATIVES]
|
||||
alternatives.sort(key=lambda c: c.pixels_per_second)
|
||||
return {
|
||||
"requested": round(requested, 1),
|
||||
"refresh_hz": round(hz, 1),
|
||||
"applied": as_dict(applied),
|
||||
"exact": exact,
|
||||
"smooth": smooth,
|
||||
"alternatives": [as_dict(c) for c in alternatives],
|
||||
}
|
||||
|
||||
+76
-31
@@ -28,6 +28,30 @@ import numpy as np
|
||||
# long over one frame, so a sample this large is an idle gap between scrolls.
|
||||
FPS_LOG_INTERVAL = 5.0
|
||||
|
||||
# The stats line goes to INFO only when a window is worth an operator's
|
||||
# attention, as Vegas's FPS line does (src/vegas_mode/coordinator.py): every
|
||||
# 5s from every scroller was most of the journal on a healthy rig. A window is
|
||||
# degraded when its frame rate falls below this fraction of the rate it was
|
||||
# locked to (1 / its own median frame time; same 0.9 as Vegas) ...
|
||||
STATS_HEALTHY_FRACTION = 0.9
|
||||
# ... or when more than this share of its frames stalled (past 1.5x the
|
||||
# median). A 1% stall rate barely moves the mean, so the fps test alone would
|
||||
# miss the judder this line exists to show.
|
||||
STATS_DEGRADED_STALL_RATE = 0.01
|
||||
# A healthy scroller still logs at INFO this often, so silence in the journal
|
||||
# means stopped rather than fine. Every window is still logged at DEBUG.
|
||||
STATS_HEARTBEAT_INTERVAL = 300.0
|
||||
|
||||
|
||||
def frame_stats_degraded(stats: Dict[str, Any]) -> bool:
|
||||
"""Whether one frame_stats() window is worth logging at INFO."""
|
||||
n = stats["frames"]
|
||||
if n == 0 or stats["median"] <= 0:
|
||||
return False
|
||||
locked_fps = 1.0 / stats["median"]
|
||||
return (stats["fps"] < locked_fps * STATS_HEALTHY_FRACTION
|
||||
or stats["stalls"] > n * STATS_DEGRADED_STALL_RATE)
|
||||
|
||||
|
||||
def _rgb_pixels(item) -> np.ndarray:
|
||||
"""An appended item's pixels as an RGB array, as pasting it would draw them."""
|
||||
@@ -189,6 +213,11 @@ class ScrollHelper:
|
||||
# Every frame time since the last stats line, so the 5s summary can
|
||||
# report the tail rather than one arbitrary sample. Cleared on log.
|
||||
self._window: list = []
|
||||
# INFO-level stats bookkeeping (see STATS_HEARTBEAT_INTERVAL). Kept
|
||||
# across reset_scroll(): a heartbeat per scroll start would bring the
|
||||
# chatter back. 0.0 so the first window after start-up is at INFO.
|
||||
self._stats_last_info_log = 0.0
|
||||
self._stats_was_degraded = False
|
||||
|
||||
# Scrolling state management
|
||||
self.is_scrolling = False
|
||||
@@ -532,7 +561,7 @@ class ScrollHelper:
|
||||
width = self.display_width
|
||||
strip_width = self.cached_array.shape[1]
|
||||
|
||||
if start_x + width + 1 <= strip_width:
|
||||
if 0 <= start_x and start_x + width + 1 <= strip_width:
|
||||
# Slice the backing array directly. Going via
|
||||
# _get_visible_portion_integer would build two PIL images only for
|
||||
# them to be converted straight back to arrays, which measured 15x
|
||||
@@ -540,9 +569,10 @@ class ScrollHelper:
|
||||
near = self.cached_array[:, start_x:start_x + width]
|
||||
far = self.cached_array[:, start_x + 1:start_x + 1 + width]
|
||||
else:
|
||||
# Close enough to the end that one of the slices wraps; let the
|
||||
# integer path handle that and pay the conversion. Continuous mode
|
||||
# extends the strip before reaching here, so this is the rare case.
|
||||
# One of the slices wraps (close to the end, or a strip narrower
|
||||
# than the panel); let the integer path handle that and pay the
|
||||
# conversion. Continuous mode extends the strip before reaching
|
||||
# here, so this is the rare case.
|
||||
near = np.asarray(
|
||||
self._get_visible_portion_integer(start_x, start_x + width))
|
||||
far = np.asarray(
|
||||
@@ -572,31 +602,33 @@ class ScrollHelper:
|
||||
_size = (self.display_width, self.display_height)
|
||||
img_w = self.cached_array.shape[1]
|
||||
|
||||
if end_x <= img_w:
|
||||
# Normal case: single contiguous slice (fastest path)
|
||||
frame_array = np.ascontiguousarray(self.cached_array[:, start_x:end_x])
|
||||
return Image.frombytes('RGB', _size, frame_array.tobytes())
|
||||
if 0 <= start_x and end_x <= img_w:
|
||||
# Normal case: single contiguous slice (fastest path). tobytes()
|
||||
# on the column-slice view already returns C-order bytes, so
|
||||
# ascontiguousarray() first only added a second full-frame copy.
|
||||
return Image.frombytes(
|
||||
'RGB', _size,
|
||||
self.cached_array[:, start_x:end_x].tobytes())
|
||||
|
||||
# Ensure frame buffer is allocated for all non-simple paths
|
||||
if self._frame_buffer is None or self._frame_buffer.shape != (self.display_height, self.display_width, 3):
|
||||
self._frame_buffer = np.zeros((self.display_height, self.display_width, 3), dtype=np.uint8)
|
||||
|
||||
if img_w == 0:
|
||||
self._frame_buffer[:] = 0
|
||||
else:
|
||||
# Ensure frame buffer is allocated for all non-simple paths
|
||||
if self._frame_buffer is None or self._frame_buffer.shape != (self.display_height, self.display_width, 3):
|
||||
self._frame_buffer = np.zeros((self.display_height, self.display_width, 3), dtype=np.uint8)
|
||||
# The frame runs off the strip, so it carries on from the head:
|
||||
# frame column j is strip column (start_x + j) modulo the strip's
|
||||
# width -- the tail and then the head, and a strip narrower than
|
||||
# the panel repeated across it. Copying the tail and then the rest
|
||||
# of the frame from the head assumed the head was that wide, and
|
||||
# raised at every position for a strip narrower than the panel
|
||||
# (Vegas composes one, with no lead-in, when its content is
|
||||
# narrower than the chain).
|
||||
np.take(self.cached_array, np.arange(start_x, end_x), axis=1,
|
||||
mode='wrap', out=self._frame_buffer)
|
||||
|
||||
width1 = img_w - start_x
|
||||
if width1 > 0:
|
||||
# Wrap-around: tail of image + head of image
|
||||
self._frame_buffer[:, :width1] = self.cached_array[:, start_x:]
|
||||
remaining_width = self.display_width - width1
|
||||
self._frame_buffer[:, width1:] = self.cached_array[:, :remaining_width]
|
||||
else:
|
||||
# Edge case: start_x at or past image end — show from beginning,
|
||||
# clamped to available width (scroll_position should wrap before
|
||||
# reaching this state in normal operation).
|
||||
available = min(self.display_width, img_w)
|
||||
self._frame_buffer[:, :available] = self.cached_array[:, :available]
|
||||
if available < self.display_width:
|
||||
self._frame_buffer[:, available:] = 0
|
||||
|
||||
return Image.frombytes('RGB', _size, self._frame_buffer.tobytes())
|
||||
return Image.frombytes('RGB', _size, self._frame_buffer.tobytes())
|
||||
|
||||
def calculate_dynamic_duration(self) -> int:
|
||||
"""
|
||||
@@ -1207,10 +1239,23 @@ class ScrollHelper:
|
||||
# as an idle gap. There is nothing to report, and reporting the
|
||||
# gap itself is the bug above.
|
||||
if self._window:
|
||||
self.logger.info(
|
||||
"Scroll frame stats - %s",
|
||||
format_frame_stats(self._window),
|
||||
)
|
||||
# INFO when degraded, on the window that recovers from it, and
|
||||
# as a slow heartbeat; DEBUG otherwise.
|
||||
degraded = frame_stats_degraded(frame_stats(self._window))
|
||||
if (degraded or self._stats_was_degraded
|
||||
or current_time - self._stats_last_info_log
|
||||
>= STATS_HEARTBEAT_INTERVAL):
|
||||
self.logger.info(
|
||||
"Scroll frame stats - %s",
|
||||
format_frame_stats(self._window),
|
||||
)
|
||||
self._stats_last_info_log = current_time
|
||||
elif self.logger.isEnabledFor(logging.DEBUG):
|
||||
self.logger.debug(
|
||||
"Scroll frame stats - %s",
|
||||
format_frame_stats(self._window),
|
||||
)
|
||||
self._stats_was_degraded = degraded
|
||||
self.last_fps_log_time = current_time
|
||||
self.frame_count = 0
|
||||
self._window = []
|
||||
|
||||
@@ -26,14 +26,33 @@ Policy:
|
||||
- Unchanged frames are never re-encoded; the mtime is touched every
|
||||
TOUCH_INTERVAL so the health check (60s threshold) never degrades.
|
||||
|
||||
The writer and the SSE reader (web_interface/app.py) have two periods, not
|
||||
one shared value. The reader sends each write it sees, so the preview shows
|
||||
at most one frame per VIEWER_INTERVAL. (That was 0.2 s while the reader
|
||||
slept 1 s between reads, so four encodes in five were overwritten unread.)
|
||||
The reader checks the file's mtime every VIEWER_POLL_INTERVAL, which is only
|
||||
a stat, and so sends each write within that long of it landing. Equal
|
||||
periods would alias: two unsynchronised 1 s clocks leave the preview up to a
|
||||
second stale, and now and then 2 s between frames.
|
||||
|
||||
decide() is monotone in frame_changed: SKIP for a changed frame means SKIP
|
||||
for an unchanged one. DisplayManager relies on that to skip hashing the
|
||||
frame when even a changed one would be skipped; test_snapshot_policy.py
|
||||
checks it.
|
||||
|
||||
If any constant here changes, re-check the health threshold in
|
||||
api_v3/misc.py (get_hardware_status) — TOUCH_INTERVAL must stay well under it.
|
||||
"""
|
||||
|
||||
from enum import Enum
|
||||
|
||||
# Snapshot cadence with a browser preview open (seconds).
|
||||
VIEWER_INTERVAL = 0.2
|
||||
# Snapshot cadence with a browser preview open (seconds): the shortest gap
|
||||
# between two preview frames.
|
||||
VIEWER_INTERVAL = 1.0
|
||||
# How often the web SSE reader checks the snapshot's mtime (seconds). Must
|
||||
# stay well under VIEWER_INTERVAL -- half of it at most -- or the two clocks
|
||||
# alias (see above).
|
||||
VIEWER_POLL_INTERVAL = 0.25
|
||||
# Snapshot cadence with no viewers — cheap freshness for page-open (seconds).
|
||||
IDLE_INTERVAL = 30.0
|
||||
# Max age of the last write/touch before bumping mtime for the health
|
||||
|
||||
@@ -18,7 +18,7 @@ the extra guard only stops a None size raising TypeError.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from datetime import datetime, timezone
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
from zoneinfo import ZoneInfo
|
||||
|
||||
@@ -338,10 +338,46 @@ def format_game_date(config: Optional[Dict[str, Any]], logger, date_text: str,
|
||||
if not raw:
|
||||
return ""
|
||||
fmt = str(scroll_card_option(config, "date_format", "abbrev") or "abbrev")
|
||||
return _format_date_as(fmt, raw, lambda: weekday_for(config, logger, game))
|
||||
return _format_date_as(fmt, raw, lambda: weekday_for(config, logger, game),
|
||||
game=game)
|
||||
|
||||
|
||||
def _format_date_as(fmt: str, raw: str, weekday, months=MONTH_ABBR) -> str:
|
||||
def _printed_weekday(game: Optional[Dict], month: int, day: int) -> str:
|
||||
"""The weekday of the date a card prints as month/day, or '' if unknown.
|
||||
|
||||
The extractor prints "M/D" in the plugin's resolved zone (its own setting,
|
||||
else the global one, else the system zone). The card cannot see that zone:
|
||||
it is handed the plugin's config, whose ``timezone`` ships as "", so
|
||||
card_tzinfo answers UTC and an evening kickoff in the Americas got the
|
||||
next day's weekday ("Sat Oct 2" for a Friday game). Every zone is within
|
||||
a day of UTC, so the printed date is the start's UTC date or a neighbour
|
||||
of it; the one with that month and day is the date on the card.
|
||||
"""
|
||||
if not isinstance(game, dict):
|
||||
return ""
|
||||
raw = game.get("start_time_utc") or game.get("start_time")
|
||||
if not raw:
|
||||
return ""
|
||||
try:
|
||||
start = raw if isinstance(raw, datetime) else datetime.fromisoformat(
|
||||
str(raw).replace("Z", "+00:00"))
|
||||
if start.utcoffset() is None:
|
||||
return "" # naive: no instant to place the date against
|
||||
utc_day = start.astimezone(timezone.utc).date()
|
||||
except (ValueError, TypeError, OverflowError):
|
||||
return ""
|
||||
for offset in (0, -1, 1):
|
||||
try:
|
||||
candidate = utc_day + timedelta(days=offset)
|
||||
except OverflowError:
|
||||
continue
|
||||
if (candidate.month, candidate.day) == (month, day):
|
||||
return WEEKDAY_ABBR[candidate.weekday()]
|
||||
return ""
|
||||
|
||||
|
||||
def _format_date_as(fmt: str, raw: str, weekday, months=MONTH_ABBR,
|
||||
game: Optional[Dict] = None) -> str:
|
||||
"""Render a stripped, non-empty "M/D" *raw* in style *fmt*.
|
||||
|
||||
The body both date formatters share. They differ in which setting names the
|
||||
@@ -349,6 +385,9 @@ def _format_date_as(fmt: str, raw: str, weekday, months=MONTH_ABBR) -> str:
|
||||
``SportsCoreSharedMixin._format_game_date``), so those arrive as arguments:
|
||||
*weekday* is a zero-argument callable, only called for the "weekday" style.
|
||||
*months* lets the mixin keep reading its (overridable) ``_MONTH_ABBR``.
|
||||
With *game*, the "weekday" style names the printed date's own weekday
|
||||
(:func:`_printed_weekday`), and *weekday* is only the fallback for a
|
||||
date its start time cannot place.
|
||||
"""
|
||||
if fmt == "numeric":
|
||||
return raw
|
||||
@@ -364,7 +403,7 @@ def _format_date_as(fmt: str, raw: str, weekday, months=MONTH_ABBR) -> str:
|
||||
if fmt == "day_first":
|
||||
return f"{day} {name}"
|
||||
if fmt == "weekday":
|
||||
day_name = weekday()
|
||||
day_name = _printed_weekday(game, month, day) or weekday()
|
||||
return f"{day_name} {name} {day}" if day_name else f"{name} {day}"
|
||||
return f"{name} {day}"
|
||||
|
||||
|
||||
@@ -40,6 +40,20 @@ listed here.
|
||||
- ``live_games``, read with ``getattr`` -- ``_needs_previous_day``.
|
||||
- ``background_service``, read with ``getattr`` --
|
||||
``_background_fetches_espn_ranges``.
|
||||
- ``sport`` and ``league`` (ESPN's path segments, e.g. ``football`` /
|
||||
``nfl``) -- ``_schedule_cache_key``, and ``_fetch_season_directly`` when
|
||||
it is given no key and cannot read one from its URL.
|
||||
|
||||
THE SCHEDULE CACHE KEY (fetch service stage 2)
|
||||
----------------------------------------------
|
||||
``_schedule_cache_key`` names a schedule window with the canonical
|
||||
``espn_scoreboard_cache_key`` instead of a plugin-built
|
||||
``{sport_key}_schedule_{window}``, and ``_cached_schedule`` reads it with the
|
||||
old key as a fallback for one release, so an upgrade serves the copy already
|
||||
on disk instead of refetching every league at once. The canonical key
|
||||
carries the window's dates, so it moves on a day as the window slides; a
|
||||
miss on it also deletes the copy for the day before, so a league keeps one
|
||||
window file instead of a week of them.
|
||||
|
||||
Add it as a base of the plugin's ``SportsCore``, e.g.
|
||||
``class SportsCore(SportsFetchMixin, SportsCoreSharedMixin,
|
||||
@@ -50,9 +64,31 @@ constant on the plugin's own class still wins over the mixin's.
|
||||
import logging
|
||||
import threading
|
||||
from datetime import datetime, timedelta
|
||||
from typing import Any, ClassVar, Dict, Optional
|
||||
from typing import Any, ClassVar, Dict, Iterable, Optional
|
||||
|
||||
from src.common.espn_dates import ESPN_MAX_LIMIT, fetch_espn_scoreboard
|
||||
from src.common.espn_dates import (
|
||||
ESPN_MAX_LIMIT,
|
||||
espn_scoreboard_cache_key,
|
||||
espn_scoreboard_cache_key_for_url,
|
||||
fetch_espn_scoreboard,
|
||||
parse_espn_date_range,
|
||||
)
|
||||
from src.common.fetch_service import get_fetch_service
|
||||
|
||||
_ESPN_SITE = "https://site.api.espn.com/"
|
||||
|
||||
|
||||
def _previous_window_key(cache_key: str) -> Optional[str]:
|
||||
"""The canonical key of the same window one day earlier, or None when
|
||||
``cache_key`` is not a canonical day-range key."""
|
||||
head, sep, dates = cache_key.rpartition("_")
|
||||
if not sep or not head.startswith("espn_scoreboard_"):
|
||||
return None
|
||||
span = parse_espn_date_range(dates)
|
||||
if span is None:
|
||||
return None
|
||||
start, end = (day - timedelta(days=1) for day in span)
|
||||
return f"{head}_{start.strftime('%Y%m%d')}-{end.strftime('%Y%m%d')}"
|
||||
|
||||
|
||||
class SportsFetchMixin:
|
||||
@@ -65,6 +101,8 @@ class SportsFetchMixin:
|
||||
cache_manager: Any
|
||||
logger: logging.Logger
|
||||
_games_lock: threading.RLock
|
||||
sport: str
|
||||
league: str
|
||||
|
||||
#: How many games past the one on screen keep their odds warm. One is
|
||||
#: enough for the line to be ready when the rotation advances; more just
|
||||
@@ -154,18 +192,66 @@ class SportsFetchMixin:
|
||||
service = getattr(self, "background_service", None)
|
||||
return bool(getattr(service, "handles_espn_date_ranges", False))
|
||||
|
||||
def _schedule_cache_key(self, datestring: str) -> str:
|
||||
"""The canonical cache key for this league's schedule over
|
||||
``datestring`` (``espn_scoreboard_cache_key``)."""
|
||||
return espn_scoreboard_cache_key(self.sport, self.league, datestring)
|
||||
|
||||
def _cached_schedule(self, cache_key: str, legacy_keys: Iterable[str] = ()) -> Any:
|
||||
"""What ``self.cache_manager.get(cache_key)`` returns, falling back
|
||||
to each of ``legacy_keys`` (the plugin's pre-canonical keys) in turn.
|
||||
|
||||
The same read the managers made before -- same default max age, a
|
||||
stored ttl still wins -- so moving to the canonical key changes
|
||||
where a schedule is cached, not for how long. A read from an old key
|
||||
is counted (``legacy_cache_hits``) so it is visible when the
|
||||
fallback can go. A miss on the canonical key also deletes the same
|
||||
window's copy from the day before (see the module docstring).
|
||||
"""
|
||||
cached = self.cache_manager.get(cache_key)
|
||||
if cached:
|
||||
return cached
|
||||
self._retire_previous_window(cache_key)
|
||||
for legacy in legacy_keys:
|
||||
if not legacy or legacy == cache_key:
|
||||
continue
|
||||
cached = self.cache_manager.get(legacy)
|
||||
if cached:
|
||||
try:
|
||||
get_fetch_service().note_cache_hit(
|
||||
_ESPN_SITE, legacy=True, avoided_request=False)
|
||||
except Exception: # noqa: BLE001 - counting never breaks a read
|
||||
pass
|
||||
return cached
|
||||
return None
|
||||
|
||||
def _retire_previous_window(self, cache_key: str) -> None:
|
||||
previous = _previous_window_key(cache_key)
|
||||
delete = getattr(self.cache_manager, "delete", None)
|
||||
if previous is None or not callable(delete):
|
||||
return
|
||||
try:
|
||||
delete(previous)
|
||||
except Exception as e: # noqa: BLE001 - housekeeping only
|
||||
self.logger.debug(f"Could not delete old schedule copy {previous}: {e}")
|
||||
|
||||
def _fetch_season_directly(
|
||||
self,
|
||||
url: str,
|
||||
datestring: str,
|
||||
cache_key: str,
|
||||
cache_key: Optional[str],
|
||||
label: str,
|
||||
ttl: Optional[int] = None,
|
||||
) -> Optional[Dict]:
|
||||
"""Fetch a season schedule on this thread, in chunks ESPN accepts, and cache it.
|
||||
|
||||
``label`` names the schedule in log lines, e.g. ``"2026 season"``.
|
||||
``cache_key=None`` caches it under the canonical key
|
||||
(``espn_scoreboard_cache_key`` for ``url``'s sport and league).
|
||||
"""
|
||||
if cache_key is None:
|
||||
cache_key = (espn_scoreboard_cache_key_for_url(url, datestring)
|
||||
or self._schedule_cache_key(datestring))
|
||||
try:
|
||||
data = fetch_espn_scoreboard(
|
||||
self.session,
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
"""Whether a game ESPN still lists as live has in fact ended (sports family 5).
|
||||
|
||||
``SportsGameOverMixin._is_game_really_over`` is the scoreboards'
|
||||
``SportsLive._is_game_really_over``, reconciled in ledmatrix-plugins
|
||||
#625 from five bodies into one and copied here under
|
||||
its existing name. ``SportsLiveSharedMixin._detect_stale_games``
|
||||
(``src.common.sports_shared``) calls it on every live game, and the plugins'
|
||||
live-priority filters call it too, to drop a game ESPN still reports as
|
||||
in progress.
|
||||
|
||||
A game is over when its period text says final. From period ``FINAL_PERIOD``
|
||||
on, a clock reading 0:00 ends it too, unless the score is level: a tie at the
|
||||
end of regulation goes to overtime (or a shootout), and a game that does end
|
||||
tied says final. Only a clock *string* is read ("0:00" and ":00" are zero;
|
||||
":40", "0.0" and ESPN's "-" between MMA rounds are not), and a missing or
|
||||
unreadable score leaves the decision to the clock.
|
||||
|
||||
``FINAL_PERIOD`` is the one per-sport fact, a class attribute rather than a
|
||||
sport-name branch. The scoreboards declare it on their ``SportsLive``:
|
||||
|
||||
- 3: hockey;
|
||||
- 4: basketball, football, lacrosse;
|
||||
- ``None`` (this default; the clock never ends a game): afl, nrl and soccer,
|
||||
whose clocks count up; baseball, which has innings; ufc, whose bouts end
|
||||
only on ESPN's final status.
|
||||
|
||||
A sport can still override the method and defer to it, as baseball's
|
||||
``BaseballLive`` does to end postponed and suspended games first.
|
||||
|
||||
A new module rather than another method on ``sports_shared``, for the reason
|
||||
``sports_helpers`` gives: a missing module fails at load, where the version
|
||||
checks see it; a missing method fails mid-update.
|
||||
|
||||
WHAT A HOST MUST PROVIDE
|
||||
------------------------
|
||||
Derived by walking every ``self.<attr>`` the mixin reads; the host-contract
|
||||
test in ``test/test_sports_game_over.py`` fails if a read is added without
|
||||
being listed here.
|
||||
|
||||
- ``logger`` -- a ``logging.Logger``; the method logs its verdict at DEBUG.
|
||||
- ``FINAL_PERIOD`` -- defaulted here to ``None``; set it on the host class.
|
||||
|
||||
The method reads the game dict's ``away_abbr``, ``home_abbr``,
|
||||
``period_text``, ``period``, ``clock``, ``away_score`` and ``home_score``
|
||||
(``_extract_game_details_common``'s keys); any of them may be missing or
|
||||
null.
|
||||
|
||||
BASE ORDER
|
||||
----------
|
||||
List the mixin before ``SportsLiveSharedMixin`` --
|
||||
``class SportsLive(SportsGameOverMixin, SportsLiveSharedMixin, SportsCore)`` --
|
||||
so the shared mixin's ``_detect_stale_games`` finds this method through the
|
||||
MRO. Neither shared mixin defines it, so the order does not change which body
|
||||
runs today; it keeps the method next to its caller should one ever be added
|
||||
there. A method on the plugin's own class still wins, and its ``super()``
|
||||
reaches this one. The mixin has no ``__init__`` and no state.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from typing import Dict, Optional
|
||||
|
||||
|
||||
class SportsGameOverMixin:
|
||||
"""The live manager's "is this game really over?" check. See module docstring."""
|
||||
|
||||
# The host contract, declared for type checking only.
|
||||
logger: logging.Logger
|
||||
|
||||
#: Period from which a 0:00 clock ends a game; None: the clock never does.
|
||||
FINAL_PERIOD: Optional[int] = None
|
||||
|
||||
def _is_game_really_over(self, game: Dict) -> bool:
|
||||
"""Whether a game ESPN still lists as live has in fact ended.
|
||||
|
||||
It has when its period text says final. From period ``FINAL_PERIOD``
|
||||
on, a clock at 0:00 ends it too, unless the score is level: a tie at
|
||||
the end of regulation goes to overtime, and a game that does end tied
|
||||
says final. With ``FINAL_PERIOD = None`` the clock never ends a game.
|
||||
"""
|
||||
game_str = f"{game.get('away_abbr')}@{game.get('home_abbr')}"
|
||||
|
||||
# ESPN can send the key as null, and .get()'s default only covers a
|
||||
# missing key, so a None here crashed the whole live update.
|
||||
raw_period_text = game.get("period_text")
|
||||
period_text = raw_period_text.lower() if isinstance(raw_period_text, str) else ""
|
||||
if "final" in period_text:
|
||||
self.logger.debug(
|
||||
f"_is_game_really_over({game_str}): "
|
||||
f"returning True - 'final' in period_text='{period_text}'"
|
||||
)
|
||||
return True
|
||||
|
||||
# Same for a null or non-numeric period: treat it as period 0.
|
||||
try:
|
||||
period = int(game.get("period") or 0)
|
||||
except (TypeError, ValueError, OverflowError):
|
||||
period = 0
|
||||
# Only a clock string is read: "0:00" and ":00" are zero; ":40" is not.
|
||||
clock = game.get("clock")
|
||||
clock_at_zero = isinstance(clock, str) and clock.replace(":", "").strip() in ("000", "00")
|
||||
|
||||
if self.FINAL_PERIOD is not None and period >= self.FINAL_PERIOD and clock_at_zero:
|
||||
try:
|
||||
tied = int(game["away_score"]) == int(game["home_score"])
|
||||
except (KeyError, TypeError, ValueError, OverflowError):
|
||||
tied = False # a missing or unreadable score leaves it to the clock
|
||||
if not tied:
|
||||
self.logger.debug(
|
||||
f"_is_game_really_over({game_str}): "
|
||||
f"returning True - clock at 0:00 (clock='{clock}', period={period})"
|
||||
)
|
||||
return True
|
||||
self.logger.debug(
|
||||
f"_is_game_really_over({game_str}): "
|
||||
f"returning False - tied at 0:00 (period={period}), overtime next"
|
||||
)
|
||||
return False
|
||||
|
||||
self.logger.debug(
|
||||
f"_is_game_really_over({game_str}): returning False"
|
||||
)
|
||||
return False
|
||||
|
||||
|
||||
__all__ = ["SportsGameOverMixin"]
|
||||
+372
-13
@@ -44,7 +44,10 @@ from __future__ import annotations
|
||||
|
||||
import functools
|
||||
import logging
|
||||
import threading
|
||||
import time
|
||||
import weakref
|
||||
from collections import OrderedDict
|
||||
from typing import Any, Callable, Dict, List, Optional, Tuple
|
||||
|
||||
from PIL import Image
|
||||
@@ -320,7 +323,7 @@ class SportsScrollDisplay:
|
||||
:returns: True if a frame was drawn; False when there is no content or
|
||||
the frame could not be rendered.
|
||||
"""
|
||||
if not self.scroll_helper.cached_image:
|
||||
if not self._has_strip():
|
||||
return False
|
||||
|
||||
try:
|
||||
@@ -413,7 +416,21 @@ class SportsScrollDisplay:
|
||||
|
||||
def has_cached_content(self) -> bool:
|
||||
"""Whether content is prepared and ready to scroll."""
|
||||
return bool(self.scroll_helper.cached_image)
|
||||
return self._has_strip()
|
||||
|
||||
def _has_strip(self) -> bool:
|
||||
"""Whether the helper holds a strip, without building its PIL image.
|
||||
|
||||
Reading ``cached_image`` after the strip was extended or trimmed builds
|
||||
the image from the array and keeps it, so the strip is held twice;
|
||||
display_scroll_frame asks this every frame. ``has_strip()`` answers
|
||||
from the helper's bookkeeping. A helper without it (a plugin's own, a
|
||||
test double) is asked the old way.
|
||||
"""
|
||||
helper = self.scroll_helper
|
||||
if callable(getattr(type(helper), "has_strip", None)):
|
||||
return bool(helper.has_strip())
|
||||
return bool(helper.cached_image)
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Live Vegas cards
|
||||
@@ -589,16 +606,80 @@ class SportsScrollDisplay:
|
||||
return info
|
||||
|
||||
|
||||
class _StripSlot:
|
||||
"""One slate's display in a game type's pool, and what its strip shows.
|
||||
|
||||
``key`` is None when the strip must not be reused: never built, built
|
||||
from something that could not be fingerprinted, a build that failed, or
|
||||
released to stay inside the memory budget.
|
||||
"""
|
||||
|
||||
__slots__ = ("display", "key", "built_at", "strip", "last_used", "epoch")
|
||||
|
||||
def __init__(self, display: SportsScrollDisplay) -> None:
|
||||
self.display = display
|
||||
self.key: Optional[Tuple[Any, ...]] = None
|
||||
self.built_at = 0.0
|
||||
#: The helper's strip array when it was built, held weakly: a strip
|
||||
#: replaced or cleared by anything since (a Vegas build on the same
|
||||
#: display, a plugin calling clear()) no longer matches it.
|
||||
self.strip: Optional[Callable[[], Any]] = None
|
||||
self.last_used = 0.0
|
||||
#: Bumped by every forget(). A build records its key only if this is
|
||||
#: what it was when the build started: anything that forgot the slot
|
||||
#: meanwhile (another build on this display, which may finish first
|
||||
#: and leave its strip in the helper) means the strip in the helper
|
||||
#: is not known to be this build's.
|
||||
self.epoch = 0
|
||||
|
||||
def forget(self) -> None:
|
||||
self.key = None
|
||||
self.strip = None
|
||||
self.epoch += 1
|
||||
|
||||
|
||||
class SportsScrollDisplayManager:
|
||||
"""One :class:`SportsScrollDisplay` per game type ('live'/'recent'/'upcoming').
|
||||
|
||||
Subclasses set :attr:`display_class`; everything else was near-identical
|
||||
across the eight plugin copies.
|
||||
|
||||
A recent or upcoming strip that has not changed is reused rather than
|
||||
redrawn when its turn comes round again -- see :meth:`prepare_and_display`.
|
||||
"""
|
||||
|
||||
#: The SportsScrollDisplay subclass to instantiate per game type.
|
||||
display_class = SportsScrollDisplay
|
||||
|
||||
#: Game types whose strip is reused while nothing it is drawn from has
|
||||
#: changed. Live is left out: its games change every poll, so the check
|
||||
#: would never pay, and sports_live_scroll rebuilds a live strip in place
|
||||
#: mid-cycle around get_scroll_display('live'), which has to stay the one
|
||||
#: display it always was. Vegas's 'mixed' never comes through
|
||||
#: prepare_and_display.
|
||||
STRIP_MEMO_GAME_TYPES = frozenset({"recent", "upcoming"})
|
||||
|
||||
#: Oldest a reused strip may be. Not everything a card draws is in the
|
||||
#: game dicts -- a team logo that was missing at the first build appears
|
||||
#: only when the card is drawn again -- so an unchanged slate is still
|
||||
#: redrawn this often.
|
||||
STRIP_MEMO_MAX_AGE_S = 600.0
|
||||
|
||||
#: Displays kept per game type, one per slate (its leagues): the one on
|
||||
#: screen plus the most recently shown others. When all are taken, the
|
||||
#: least recently shown one draws the new slate, as the one shared display
|
||||
#: always did, so a rotation with more slates than this costs no more
|
||||
#: than before.
|
||||
STRIP_MEMO_SLATES_PER_TYPE = 4
|
||||
|
||||
#: Ceiling, per plugin, on the strips kept for displays not on screen, in
|
||||
#: bytes (the strip's array and image, and its Vegas items). Seven
|
||||
#: football games at 192x48 come to ~0.65MB, thirty at 512x64 to ~4MB.
|
||||
#: It bounds strip pixels only: each extra display also keeps its own
|
||||
#: logo and separator-icon caches and frame buffer, and each slot a frozen
|
||||
#: copy of the config in its key (~50KB), none of which is counted here.
|
||||
STRIP_MEMO_MAX_PARKED_BYTES = 6 * 1024 * 1024
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
display_manager,
|
||||
@@ -616,16 +697,34 @@ class SportsScrollDisplayManager:
|
||||
# either way, but two spellings of "nothing active" across two classes
|
||||
# is a trap for anyone comparing state between them.
|
||||
self._current_game_type: str = ""
|
||||
# Per game type, its displays by slate, least recently shown first.
|
||||
# _scroll_displays[game_type] is always the one on screen, so every
|
||||
# reader of it (display_frame, is_complete, the plugins' own
|
||||
# get_dynamic_duration and has_cached_content) sees what it did when
|
||||
# there was only one display per game type.
|
||||
self._strip_pools: Dict[str, "OrderedDict[Tuple[str, Tuple[Any, ...]], _StripSlot]"] = {}
|
||||
# Only the bookkeeping is under it (and creating a slate's display,
|
||||
# the first time that slate is drawn), never a build: a display()
|
||||
# call that outlived its timeout can still be building when the next
|
||||
# one starts.
|
||||
self._strip_lock = threading.RLock()
|
||||
|
||||
def _new_scroll_display(self) -> SportsScrollDisplay:
|
||||
return self.display_class(
|
||||
self.display_manager,
|
||||
self.config,
|
||||
self.logger,
|
||||
global_config=self.global_config,
|
||||
)
|
||||
|
||||
def get_scroll_display(self, game_type: str) -> SportsScrollDisplay:
|
||||
"""The display for ``game_type``, created on first use."""
|
||||
"""The display for ``game_type``, created on first use.
|
||||
|
||||
For a recent or upcoming game type, the display of the slate prepared
|
||||
last -- the strip display_frame() draws.
|
||||
"""
|
||||
if game_type not in self._scroll_displays:
|
||||
self._scroll_displays[game_type] = self.display_class(
|
||||
self.display_manager,
|
||||
self.config,
|
||||
self.logger,
|
||||
global_config=self.global_config,
|
||||
)
|
||||
self._scroll_displays[game_type] = self._new_scroll_display()
|
||||
return self._scroll_displays[game_type]
|
||||
|
||||
def prepare_and_display(
|
||||
@@ -635,8 +734,63 @@ class SportsScrollDisplayManager:
|
||||
leagues: List[str],
|
||||
rankings_cache: Optional[Dict[str, int]] = None,
|
||||
) -> bool:
|
||||
"""Build content for ``game_type`` and make it the active strip."""
|
||||
scroll_display = self.get_scroll_display(game_type)
|
||||
"""Build content for ``game_type`` and make it the active strip.
|
||||
|
||||
Building a strip draws every card while the render thread waits for
|
||||
it, with the panel frozen on its last frame: ~1.4s for seven football
|
||||
cards at 192x48 on a Pi 4, at the start of every turn and again each
|
||||
time the cycle completes. A recent or upcoming turn usually draws
|
||||
exactly the strip its slate drew last time. So when nothing the strip
|
||||
is drawn from has changed -- the games, the rankings, the config, the
|
||||
panel size and the date -- that strip is rewound and shown again
|
||||
instead, which leaves the plugin's prepare_scroll_content() uncalled.
|
||||
|
||||
Two leagues usually take turns on one game type (nfl_recent, then
|
||||
ncaa_fb_recent), so each slate keeps its own display rather than
|
||||
sharing one; see STRIP_MEMO_SLATES_PER_TYPE. Anything in doubt is
|
||||
drawn again: a live strip, a turn with no games, inputs that cannot
|
||||
be fingerprinted, a strip older than STRIP_MEMO_MAX_AGE_S, or one
|
||||
changed since it was built.
|
||||
"""
|
||||
keyed = self._strip_memo_key(games, game_type, leagues, rankings_cache)
|
||||
restore: Optional[SportsScrollDisplay] = None
|
||||
epoch = 0
|
||||
with self._strip_lock:
|
||||
if keyed is None:
|
||||
self._forget_shown_strip(game_type)
|
||||
scroll_display = self.get_scroll_display(game_type)
|
||||
else:
|
||||
reused = self._reuse_strip(game_type, *keyed)
|
||||
if reused is not None:
|
||||
# What a fresh build leaves: the strip at its start, a new
|
||||
# cycle not yet complete, this game type active.
|
||||
reused.reset_scroll()
|
||||
self._current_game_type = game_type
|
||||
self.logger.debug(
|
||||
"Reusing the unchanged %s strip for %s",
|
||||
game_type, ", ".join(map(str, keyed[0][1])))
|
||||
return True
|
||||
scroll_display, restore, epoch = self._display_to_build(
|
||||
game_type, keyed[0])
|
||||
# Before the build, so data that changes during it reads as changed.
|
||||
started = time.monotonic()
|
||||
success = self._prepare_on(
|
||||
scroll_display, games, game_type, leagues, rankings_cache)
|
||||
if keyed is not None:
|
||||
with self._strip_lock:
|
||||
self._note_strip_built(
|
||||
game_type, keyed, scroll_display, restore, success, started,
|
||||
epoch)
|
||||
return success
|
||||
|
||||
def _prepare_on(
|
||||
self,
|
||||
scroll_display: SportsScrollDisplay,
|
||||
games: List[Dict],
|
||||
game_type: str,
|
||||
leagues: List[str],
|
||||
rankings_cache: Optional[Dict[str, int]],
|
||||
) -> bool:
|
||||
try:
|
||||
success = scroll_display.prepare_scroll_content(
|
||||
games, game_type, leagues, rankings_cache
|
||||
@@ -654,6 +808,200 @@ class SportsScrollDisplayManager:
|
||||
self._current_game_type = game_type
|
||||
return success
|
||||
|
||||
# ------------------------------------------------------------------
|
||||
# Reusing an unchanged strip
|
||||
# ------------------------------------------------------------------
|
||||
|
||||
def _strip_memo_key(
|
||||
self,
|
||||
games: Any,
|
||||
game_type: str,
|
||||
leagues: Any,
|
||||
rankings_cache: Any,
|
||||
) -> Optional[Tuple[Tuple[str, Tuple[Any, ...]], Tuple[Any, ...]]]:
|
||||
"""``(slate, key)``: which display, and everything its strip is drawn
|
||||
from. None when this strip must be drawn regardless.
|
||||
|
||||
The games go through sports_vegas.game_fingerprint, the same "has
|
||||
this card changed" the live Vegas cards are redrawn on: the whole
|
||||
game dict, so no field a card draws can be missed. The config is
|
||||
fingerprinted by value, not by identity, so a config edited in place
|
||||
counts as changed.
|
||||
"""
|
||||
if game_type not in self.STRIP_MEMO_GAME_TYPES or not games:
|
||||
return None
|
||||
# Iterated twice (here and by the build), so a one-shot iterable
|
||||
# would reach the build empty.
|
||||
if not isinstance(games, (list, tuple)) or not isinstance(leagues, (list, tuple)):
|
||||
return None
|
||||
try:
|
||||
slate = (game_type, tuple(leagues))
|
||||
hash(slate)
|
||||
key = (
|
||||
tuple(sports_vegas.game_fingerprint(game) for game in games),
|
||||
sports_vegas._freeze(rankings_cache),
|
||||
sports_vegas._freeze(self.config),
|
||||
(getattr(self.display_manager, "width", None),
|
||||
getattr(self.display_manager, "height", None)),
|
||||
# A backstop: no card reads the clock today (game dates come
|
||||
# in the game dicts), but a strip must not outlive its day.
|
||||
time.localtime()[:3],
|
||||
)
|
||||
except Exception:
|
||||
# A game dict changing size under a background update, say. The
|
||||
# build reads it anyway; only the reuse is given up.
|
||||
self.logger.debug("Strip for %s not reusable this turn", game_type,
|
||||
exc_info=True)
|
||||
return None
|
||||
return slate, key
|
||||
|
||||
def _strip_reusable(self, slot: _StripSlot, key: Tuple[Any, ...]) -> bool:
|
||||
if slot.key is None or slot.strip is None:
|
||||
return False
|
||||
if time.monotonic() - slot.built_at >= self.STRIP_MEMO_MAX_AGE_S:
|
||||
return False
|
||||
helper = getattr(slot.display, "scroll_helper", None)
|
||||
array = getattr(helper, "cached_array", None)
|
||||
if array is None or slot.strip() is not array:
|
||||
return False
|
||||
has_strip = getattr(helper, "has_strip", None)
|
||||
if callable(has_strip) and not has_strip():
|
||||
return False
|
||||
return bool(slot.key == key)
|
||||
|
||||
def _reuse_strip(
|
||||
self, game_type: str, slate: Tuple[str, Tuple[Any, ...]], key: Tuple[Any, ...],
|
||||
) -> Optional[SportsScrollDisplay]:
|
||||
"""The display already showing this exact strip, made the active one."""
|
||||
pool = self._strip_pools.get(game_type)
|
||||
slot = pool.get(slate) if pool else None
|
||||
if pool is None or slot is None or not self._strip_reusable(slot, key):
|
||||
return None
|
||||
pool.move_to_end(slate)
|
||||
slot.last_used = time.monotonic()
|
||||
# The display this replaces keeps its slot and strip for its own next
|
||||
# turn, within the parked-strip budget.
|
||||
self._scroll_displays[game_type] = slot.display
|
||||
self._trim_parked_strips()
|
||||
return slot.display
|
||||
|
||||
def _display_to_build(
|
||||
self, game_type: str, slate: Tuple[str, Tuple[Any, ...]],
|
||||
) -> Tuple[SportsScrollDisplay, Optional[SportsScrollDisplay], int]:
|
||||
"""The display to draw ``slate`` on, made the active one.
|
||||
|
||||
Returns it; when it replaced another as the active display, that
|
||||
one, to put back if the build fails -- a failed build left the
|
||||
previous strip showing when the game type had one display; and the
|
||||
slot's epoch the build must still find to record its key.
|
||||
"""
|
||||
pool = self._strip_pools.setdefault(game_type, OrderedDict())
|
||||
active = self._scroll_displays.get(game_type)
|
||||
slot = pool.get(slate)
|
||||
if slot is None:
|
||||
if active is not None and not any(s.display is active for s in pool.values()):
|
||||
# The game type's display, holding nothing reusable: this
|
||||
# slate is drawn on it, exactly as before slates had their own.
|
||||
slot = _StripSlot(active)
|
||||
elif len(pool) < max(1, self.STRIP_MEMO_SLATES_PER_TYPE):
|
||||
slot = _StripSlot(self._new_scroll_display())
|
||||
else:
|
||||
# The least recently shown slate's display draws this one.
|
||||
_, slot = pool.popitem(last=False)
|
||||
pool[slate] = slot
|
||||
# Whatever was reusable on it is about to be drawn over.
|
||||
slot.forget()
|
||||
pool.move_to_end(slate)
|
||||
slot.last_used = time.monotonic()
|
||||
self._scroll_displays[game_type] = slot.display
|
||||
replaced = active if active is not None and active is not slot.display else None
|
||||
return slot.display, replaced, slot.epoch
|
||||
|
||||
def _note_strip_built(
|
||||
self,
|
||||
game_type: str,
|
||||
keyed: Tuple[Tuple[str, Tuple[Any, ...]], Tuple[Any, ...]],
|
||||
scroll_display: SportsScrollDisplay,
|
||||
restore: Optional[SportsScrollDisplay],
|
||||
success: bool,
|
||||
started: float,
|
||||
epoch: int,
|
||||
) -> None:
|
||||
slate, key = keyed
|
||||
pool = self._strip_pools.get(game_type)
|
||||
slot = pool.get(slate) if pool else None
|
||||
if (slot is None or slot.display is not scroll_display or slot.epoch != epoch
|
||||
or self._scroll_displays.get(game_type) is not scroll_display):
|
||||
# Another prepare moved on while this one built, or drew on this
|
||||
# display too. Record nothing; the strip is drawn again next time.
|
||||
return
|
||||
array = getattr(scroll_display.scroll_helper, "cached_array", None)
|
||||
if success and array is not None:
|
||||
slot.key = key
|
||||
slot.built_at = started
|
||||
slot.strip = weakref.ref(array)
|
||||
elif not success and restore is not None:
|
||||
self._scroll_displays[game_type] = restore
|
||||
self._trim_parked_strips()
|
||||
|
||||
def _forget_shown_strip(self, game_type: str) -> None:
|
||||
"""A build this memo cannot key is about to draw on the active display."""
|
||||
active = self._scroll_displays.get(game_type)
|
||||
for slot in (self._strip_pools.get(game_type) or {}).values():
|
||||
if slot.display is active:
|
||||
slot.forget()
|
||||
|
||||
@staticmethod
|
||||
def _strip_bytes(scroll_display: SportsScrollDisplay) -> int:
|
||||
"""What keeping this display's strip costs: the strip's array, the
|
||||
image beside it, and its Vegas items."""
|
||||
array = getattr(scroll_display.scroll_helper, "cached_array", None)
|
||||
total = int(array.nbytes) * 2 if array is not None else 0
|
||||
for item in getattr(scroll_display, "_vegas_content_items", None) or ():
|
||||
total += item.width * item.height * len(item.getbands())
|
||||
return total
|
||||
|
||||
def _trim_parked_strips(self) -> None:
|
||||
"""Release the strips of displays not on screen until they fit
|
||||
STRIP_MEMO_MAX_PARKED_BYTES: those that can never be reused first (a
|
||||
failed build left an older slate's strip behind), then the least
|
||||
recently shown. The display itself is kept, so its slate's next turn
|
||||
draws on it again."""
|
||||
# list() first: get_scroll_display() can add a game type from another
|
||||
# thread (Vegas's 'mixed'), and a dict must not grow mid-iteration.
|
||||
on_screen = {id(display) for display in list(self._scroll_displays.values())}
|
||||
parked = []
|
||||
total = 0
|
||||
for pool in self._strip_pools.values():
|
||||
for slot in pool.values():
|
||||
if id(slot.display) in on_screen:
|
||||
continue
|
||||
try:
|
||||
size = self._strip_bytes(slot.display)
|
||||
except Exception:
|
||||
# Unmeasurable: assume it does not fit.
|
||||
size = self.STRIP_MEMO_MAX_PARKED_BYTES + 1
|
||||
if size:
|
||||
parked.append((slot.last_used, size, slot))
|
||||
total += size
|
||||
parked.sort(key=lambda entry: (entry[2].key is not None, entry[0]))
|
||||
for _, size, slot in parked:
|
||||
if total <= self.STRIP_MEMO_MAX_PARKED_BYTES:
|
||||
break
|
||||
slot.forget()
|
||||
self._release_strip(slot.display)
|
||||
total -= size
|
||||
|
||||
def _release_strip(self, scroll_display: SportsScrollDisplay) -> None:
|
||||
"""Drop a display's strip without SportsScrollDisplay.clear(), which
|
||||
also tells the display manager nothing is scrolling -- not this
|
||||
display's to say while another one is on screen."""
|
||||
try:
|
||||
scroll_display.scroll_helper.clear_cache()
|
||||
scroll_display._vegas_content_items = []
|
||||
except Exception:
|
||||
self.logger.debug("Could not release a parked strip", exc_info=True)
|
||||
|
||||
def display_frame(self, game_type: Optional[str] = None) -> bool:
|
||||
"""Advance the active strip (or a named one) by one frame."""
|
||||
game_type = game_type or self._current_game_type
|
||||
@@ -676,8 +1024,19 @@ class SportsScrollDisplayManager:
|
||||
return scroll_display.is_scroll_complete()
|
||||
|
||||
def clear_all(self) -> None:
|
||||
"""Clear every display and forget which one was active."""
|
||||
for scroll_display in self._scroll_displays.values():
|
||||
"""Clear every display and forget which one was active.
|
||||
|
||||
The displays of slates not on screen too, and nothing cleared is
|
||||
reused: the next prepare draws its strip again.
|
||||
"""
|
||||
displays = list(self._scroll_displays.values())
|
||||
with self._strip_lock:
|
||||
for pool in self._strip_pools.values():
|
||||
for slot in pool.values():
|
||||
slot.forget()
|
||||
if not any(slot.display is shown for shown in displays):
|
||||
displays.append(slot.display)
|
||||
for scroll_display in displays:
|
||||
scroll_display.clear()
|
||||
self._current_game_type = ""
|
||||
|
||||
|
||||
+66
-22
@@ -57,7 +57,9 @@ Methods that stay per-plugin, because they are not identical across the eight
|
||||
``_get_layout_offset``, ``_by_importance``, ``_other_games_window``,
|
||||
``_upcoming_date_and_time_text``, ``_extract_game_details_common``,
|
||||
``_load_division_team_ids``, ``_get_timezone``, ``_is_favorite_game``,
|
||||
``_is_game_really_over``, ``_is_ranked_game``, ``_passes_other_filters``.
|
||||
``_is_ranked_game``, ``_passes_other_filters``. (``_is_game_really_over``,
|
||||
which ``_detect_stale_games`` below calls, was here too until the plugins
|
||||
reconciled it; it is now ``src.common.sports_game_over``.)
|
||||
|
||||
Of the fourteen shared class constants, thirteen are identical everywhere and
|
||||
live here. Only ``_SCORE_PROBE_TEXT`` varies -- afl and basketball reach three digits
|
||||
@@ -102,6 +104,7 @@ import requests
|
||||
from PIL import Image, ImageDraw
|
||||
from src.common import sports_card as _card
|
||||
from src.common.font_layout import load_truetype, resolve_asset_path
|
||||
from src.common.text_helper import OUTLINE_SQUARE, draw_text_outlined
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -119,6 +122,32 @@ _DEFAULT_LIVE_IDLE_MAX_SECONDS = 900
|
||||
_KICKOFF_GRACE_SECONDS = 900
|
||||
#: Fallback cadence around a kickoff when the manager has no update_interval.
|
||||
_KICKOFF_POLL_FLOOR = 30
|
||||
#: How many kickoffs after the current one a live manager remembers. Only the
|
||||
#: earliest few can matter before the next look refreshes the list, so this
|
||||
#: bounds the memory without dropping a kickoff the board would wait for.
|
||||
_KICKOFF_QUEUE_MAX = 8
|
||||
|
||||
|
||||
def _current_scheduled_start(host: Any, now: float) -> Optional[float]:
|
||||
"""The kickoff a live manager is honouring now, promoting the next queued one.
|
||||
|
||||
``_next_scheduled_start_ts`` is the kickoff being honoured: the earliest
|
||||
one ahead of us, or one that has just passed and is inside its grace.
|
||||
Kickoffs behind it wait in ``_later_scheduled_starts``. When the current
|
||||
one's grace runs out, the earliest queued kickoff that is not itself past
|
||||
its grace takes over -- including one that has already passed, so a
|
||||
second kickoff inside the first one's grace still gets a grace of its own.
|
||||
"""
|
||||
current: Optional[float] = getattr(host, "_next_scheduled_start_ts", None)
|
||||
if current and current > now - _KICKOFF_GRACE_SECONDS:
|
||||
return current
|
||||
queued: Optional[List[float]] = getattr(host, "_later_scheduled_starts", None)
|
||||
if queued:
|
||||
alive = sorted(s for s in queued if s > now - _KICKOFF_GRACE_SECONDS)
|
||||
current = alive.pop(0) if alive else None
|
||||
host._later_scheduled_starts = alive
|
||||
host._next_scheduled_start_ts = current
|
||||
return current if current and current > now - _KICKOFF_GRACE_SECONDS else None
|
||||
|
||||
|
||||
def _resolve_font_path(path: str) -> str:
|
||||
@@ -359,14 +388,16 @@ class SportsCoreSharedMixin:
|
||||
The formatting is sports_card's. What differs from the card's
|
||||
``format_game_date`` is passed in: the setting (``switch_date_format``,
|
||||
see :meth:`_switch_date_format`) and the weekday, which comes from
|
||||
:meth:`_weekday_for` and so from this plugin's resolved timezone.
|
||||
:meth:`_weekday_for` and so from this plugin's resolved timezone
|
||||
when the game's start cannot place the printed date. The game goes
|
||||
in too, so both formatters name the printed date's own weekday.
|
||||
"""
|
||||
raw = str(date_text or "").strip()
|
||||
if not raw:
|
||||
return raw
|
||||
return _card._format_date_as(self._switch_date_format(), raw,
|
||||
lambda: self._weekday_for(game),
|
||||
self._MONTH_ABBR)
|
||||
self._MONTH_ABBR, game=game)
|
||||
|
||||
def _weekday_for(self, game: Optional[Dict]) -> str:
|
||||
"""Weekday abbreviation from the game's start time, or ''."""
|
||||
@@ -851,19 +882,12 @@ class SportsCoreSharedMixin:
|
||||
elif fill is None:
|
||||
fill = self._font_color(font)
|
||||
draw.fontmode = "1"
|
||||
x, y = position
|
||||
for dx, dy in [
|
||||
(-1, -1),
|
||||
(-1, 0),
|
||||
(-1, 1),
|
||||
(0, -1),
|
||||
(0, 1),
|
||||
(1, -1),
|
||||
(1, 0),
|
||||
(1, 1),
|
||||
]:
|
||||
draw.text((x + dx, y + dy), text, font=font, fill=outline_color)
|
||||
draw.text((x, y), text, font=font, fill=fill)
|
||||
# The eight-neighbour outline, then the text on top. Rasterized once
|
||||
# and stamped nine times rather than drawn nine times; the pixels are
|
||||
# the same (draw_text_outlined falls back to the nine draws wherever
|
||||
# that is not proven).
|
||||
draw_text_outlined(draw, position, text, font, fill, outline_color,
|
||||
OUTLINE_SQUARE)
|
||||
|
||||
def _should_log(self, warning_type: str, cooldown: int = 60) -> bool:
|
||||
"""True at most once per ``cooldown`` seconds, for rate-limiting a
|
||||
@@ -1298,11 +1322,11 @@ class SportsLiveSharedMixin:
|
||||
otherwise look like another empty check and escalate the back-off
|
||||
again, right when the game is actually starting.
|
||||
"""
|
||||
start = getattr(self, "_next_scheduled_start_ts", None)
|
||||
now = time.time()
|
||||
start = _current_scheduled_start(self, now)
|
||||
if not start:
|
||||
return interval
|
||||
live = getattr(self, "update_interval", None) or _KICKOFF_POLL_FLOOR
|
||||
now = time.time()
|
||||
if now < start:
|
||||
return max(live, min(interval, int(start - now)))
|
||||
if now - start <= _KICKOFF_GRACE_SECONDS:
|
||||
@@ -1317,6 +1341,16 @@ class SportsLiveSharedMixin:
|
||||
already has. Self-correcting: a stored start that has passed is
|
||||
replaced by the next one offered, so a postponed game cannot pin the
|
||||
cadence to a kickoff that never happens.
|
||||
|
||||
Every pending kickoff is honoured, not just the first. A kickoff that
|
||||
arrives while an earlier one is inside its grace is queued in
|
||||
``_later_scheduled_starts`` (the earliest _KICKOFF_QUEUE_MAX of them)
|
||||
and takes over when that grace ends, with a grace of its own. Keeping
|
||||
only the one kickoff dropped the second of two favourites starting
|
||||
within the grace of each other: it was refused while the first held
|
||||
the slot, and refused again once it had passed, so if ESPN had not
|
||||
flipped it live by the end of the first grace the back-off went
|
||||
straight back to its ceiling and the game was noticed up to that late.
|
||||
"""
|
||||
if not isinstance(details, dict):
|
||||
return
|
||||
@@ -1333,7 +1367,7 @@ class SportsLiveSharedMixin:
|
||||
now = time.time()
|
||||
if candidate <= now:
|
||||
return
|
||||
current = getattr(self, "_next_scheduled_start_ts", None)
|
||||
current = _current_scheduled_start(self, now)
|
||||
# A kickoff that has only just passed is *kept*, not replaced by the
|
||||
# next one on the card. Replacing it immediately is what made the grace
|
||||
# window in _clamp_to_scheduled_start dead code: the moment 13:00 came
|
||||
@@ -1343,10 +1377,20 @@ class SportsLiveSharedMixin:
|
||||
# polled at 13:00:45, found nothing live because ESPN had not flipped
|
||||
# the status yet, and then went quiet for the next quarter of an hour,
|
||||
# which is the behaviour this whole clamp exists to prevent.
|
||||
if (current is None
|
||||
or current <= now - _KICKOFF_GRACE_SECONDS
|
||||
or candidate < current):
|
||||
#
|
||||
# Nor is it forgotten: whichever kickoff loses is queued behind the
|
||||
# one honoured now, so it gets its own grace when that one's ends.
|
||||
if current is None:
|
||||
self._next_scheduled_start_ts = candidate
|
||||
return
|
||||
if candidate == current:
|
||||
return
|
||||
if candidate < current:
|
||||
self._next_scheduled_start_ts, candidate = candidate, current
|
||||
queued = getattr(self, "_later_scheduled_starts", None) or []
|
||||
if candidate not in queued:
|
||||
self._later_scheduled_starts = sorted(
|
||||
[*queued, candidate])[:_KICKOFF_QUEUE_MAX]
|
||||
|
||||
#: How long a game that finished live is still reported by
|
||||
#: finished_games_snapshot(): long enough for the recent-games list, which
|
||||
|
||||
@@ -29,7 +29,6 @@ import time
|
||||
import logging
|
||||
from enum import Enum
|
||||
from typing import Callable, Optional
|
||||
import numpy as np
|
||||
from PIL import Image
|
||||
|
||||
from src.config_manager_atomic import _replace
|
||||
@@ -434,6 +433,12 @@ class DisplaySyncManager:
|
||||
return
|
||||
if self._leader_state != LeaderState.CONNECTED or not self._peer_ip:
|
||||
return
|
||||
# numpy is imported here, not at module level: the web interface
|
||||
# imports this module for its constants (STATUS_FILE, SYNC_PORT) and
|
||||
# would otherwise load numpy for nothing. Only a connected leader
|
||||
# gets this far, and after the first frame the import is a
|
||||
# sys.modules lookup.
|
||||
import numpy as np
|
||||
try:
|
||||
arr = np.asarray(image.convert("RGB"), dtype=np.uint8)
|
||||
header = _RAW_MAGIC + _RAW_HEADER.pack(image.width, image.height)
|
||||
|
||||
+178
-10
@@ -7,7 +7,7 @@ Extracted from LEDMatrix core to provide reusable functionality for plugins.
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Optional, Tuple, Union
|
||||
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple, Union
|
||||
|
||||
from PIL import Image, ImageDraw, ImageFont
|
||||
from src.common.font_layout import load_truetype, resolve_asset_path
|
||||
@@ -15,6 +15,174 @@ from src.common.font_layout import load_truetype, resolve_asset_path
|
||||
# Shared throwaway draw surface for measuring text without a target canvas.
|
||||
_measure_draw = ImageDraw.Draw(Image.new("RGB", (1, 1)))
|
||||
|
||||
#: A one-pixel outline on all eight sides, in the order the scoreboards have
|
||||
#: always drawn it (dx outer, dy inner). The order can change pixels only
|
||||
#: where anti-aliased (fontmode "L") edges overlap; it is kept anyway.
|
||||
OUTLINE_SQUARE: Tuple[Tuple[int, int], ...] = (
|
||||
(-1, -1), (-1, 0), (-1, 1), (0, -1), (0, 1), (1, -1), (1, 0), (1, 1))
|
||||
|
||||
#: A one-pixel outline on the four edge sides only, leaving the diagonal
|
||||
#: corners open: the thinner outline ufc's fight card draws.
|
||||
OUTLINE_CROSS: Tuple[Tuple[int, int], ...] = ((-1, 0), (1, 0), (0, -1), (0, 1))
|
||||
|
||||
# What the stamping path in draw_text_outlined is proven pixel-identical for
|
||||
# (test/test_text_helper.py compares it with the draw.text loop across every
|
||||
# combination). Anything else takes the loop. Compared with ``in`` on tuples
|
||||
# rather than sets so an unhashable fontmode falls back instead of raising.
|
||||
_STAMP_DRAW_MODES = ("RGB", "RGBA", "L")
|
||||
_STAMP_FONT_MODES = ("1", "L")
|
||||
|
||||
# ImageDraw.text as Pillow defines it, which the stamping path stands in for.
|
||||
# A draw whose text has been replaced since -- on the class or the instance,
|
||||
# as a test recording the strings drawn does -- takes the loop, so the
|
||||
# replacement still sees every call.
|
||||
_PILLOW_DRAW_TEXT = ImageDraw.ImageDraw.text
|
||||
|
||||
|
||||
def draw_text_outlined(draw: ImageDraw.ImageDraw, xy: Sequence[Any], text: Any,
|
||||
font: Any, fill: Any,
|
||||
outline_color: Any = (0, 0, 0),
|
||||
offsets: Iterable[Sequence[Any]] = OUTLINE_SQUARE) -> None:
|
||||
"""Draw ``text`` in ``outline_color`` at each of ``offsets``, then in ``fill`` on top.
|
||||
|
||||
The result is pixel-identical to the loop every outlined draw used to be::
|
||||
|
||||
x, y = xy
|
||||
for dx, dy in offsets:
|
||||
draw.text((x + dx, y + dy), text, font=font, fill=outline_color)
|
||||
draw.text((x, y), text, font=font, fill=fill)
|
||||
|
||||
but each ``draw.text`` rasterizes the whole string through FreeType again,
|
||||
so the default nine draws did the same glyph work nine times, and on a
|
||||
scoreboard card that text work is much of the render. Here the string is
|
||||
rasterized once and the one mask is stamped at every offset, which is
|
||||
what ``draw.text`` itself does with the mask, so the pixels are the same.
|
||||
|
||||
That holds only where it has been checked: a plain ``ImageDraw`` whose
|
||||
``text`` is Pillow's, a ``FreeTypeFont``, one line of ``str``, whole-pixel
|
||||
``xy`` (an int, or a float with nothing after the point, which is what
|
||||
centring on a measured ``textlength`` with ``// 2`` gives) and int
|
||||
offsets, and the image and font modes in ``_STAMP_DRAW_MODES`` /
|
||||
``_STAMP_FONT_MODES``. Fractional coordinates change the raster itself
|
||||
(Pillow rasterizes at the sub-pixel start), and multiline text is laid
|
||||
out line by line. Every other case, and anything
|
||||
the stamping path cannot prepare, runs the loop above unchanged, so it
|
||||
behaves exactly as before, errors included.
|
||||
|
||||
Args:
|
||||
draw: The ``ImageDraw`` to draw on.
|
||||
xy: Top-left (x, y) of the text, as for ``draw.text``.
|
||||
text: The text.
|
||||
font: The font, as for ``draw.text``.
|
||||
fill: Colour of the text itself, drawn last.
|
||||
outline_color: Colour of the outline.
|
||||
offsets: (dx, dy) of each outline draw, in drawing order.
|
||||
:data:`OUTLINE_SQUARE` (the default) or :data:`OUTLINE_CROSS`.
|
||||
"""
|
||||
x, y = xy
|
||||
# Read once: the loop below may have to start over after the stamping
|
||||
# path looked at them.
|
||||
offsets = tuple(offsets)
|
||||
if _can_stamp(draw, x, y, text, font, offsets):
|
||||
if _stamp_outlined(draw, int(x), int(y), text, font, fill,
|
||||
outline_color, offsets):
|
||||
return
|
||||
for dx, dy in offsets:
|
||||
draw.text((x + dx, y + dy), text, font=font, fill=outline_color)
|
||||
draw.text((x, y), text, font=font, fill=fill)
|
||||
|
||||
|
||||
def _can_stamp(draw: Any, x: Any, y: Any, text: Any, font: Any,
|
||||
offsets: Tuple[Any, ...]) -> bool:
|
||||
"""Whether draw_text_outlined may stamp one mask instead of drawing N times.
|
||||
|
||||
Exact types for the draw and the font, and Pillow's own ``draw.text``: a
|
||||
subclass may override ``text`` or ``getmask2``, or a test may replace
|
||||
``draw.text`` to record what is drawn, and stamping would skip either.
|
||||
"""
|
||||
return (
|
||||
type(draw) is ImageDraw.ImageDraw
|
||||
and ImageDraw.ImageDraw.text is _PILLOW_DRAW_TEXT
|
||||
and "text" not in vars(draw)
|
||||
and type(font) is ImageFont.FreeTypeFont
|
||||
and isinstance(text, str)
|
||||
and "\n" not in text
|
||||
and "\r" not in text
|
||||
and _whole_pixel(x)
|
||||
and _whole_pixel(y)
|
||||
and all(isinstance(o, (tuple, list)) and len(o) == 2
|
||||
and isinstance(o[0], int) and isinstance(o[1], int)
|
||||
for o in offsets)
|
||||
and draw.mode in _STAMP_DRAW_MODES
|
||||
and draw.fontmode in _STAMP_FONT_MODES
|
||||
)
|
||||
|
||||
|
||||
def _whole_pixel(v: Any) -> bool:
|
||||
"""An int, or a float on a whole pixel, as a draw.text coordinate.
|
||||
|
||||
For those, draw.text's ``int(x + dx)`` is ``int(x) + dx`` and its
|
||||
sub-pixel start is 0 (or -0.0, which renders the same), so one mask fits
|
||||
every offset. Floats are held well inside the range where ``x + dx`` is
|
||||
exact; nothing that far out is on any canvas, so the loop the rest take
|
||||
costs nothing that matters.
|
||||
"""
|
||||
if isinstance(v, int):
|
||||
return True
|
||||
return isinstance(v, float) and v.is_integer() and -2**31 < v < 2**31
|
||||
|
||||
|
||||
def _text_ink(draw: ImageDraw.ImageDraw, color: Any) -> Any:
|
||||
"""The ink ``ImageDraw.text`` resolves ``color`` to (its inner getink)."""
|
||||
ink, fill_ink = draw._getink(color)
|
||||
return fill_ink if ink is None else ink
|
||||
|
||||
|
||||
def _stamp_outlined(draw: ImageDraw.ImageDraw, x: int, y: int, text: str,
|
||||
font: ImageFont.FreeTypeFont, fill: Any, outline_color: Any,
|
||||
offsets: Tuple[Sequence[Any], ...]) -> bool:
|
||||
"""Rasterize once and stamp; False, with nothing drawn, to take the loop.
|
||||
|
||||
Replays what ``ImageDraw.text`` does for one line at an integer position
|
||||
(Pillow 11 and 12): ``font.getmask2`` with these arguments, then
|
||||
``draw.draw.draw_bitmap`` at the position plus the mask's offset.
|
||||
``draw.draw`` and ``draw._getink`` are Pillow internals, so everything up
|
||||
to the first pixel is guarded: if anything fails before then, nothing has
|
||||
been drawn and the loop runs instead, which then fails (or not) exactly
|
||||
as it always did -- a bad fill colour still raises after the outline is
|
||||
drawn, as it did from the last ``draw.text``.
|
||||
"""
|
||||
try:
|
||||
outline_ink = _text_ink(draw, outline_color)
|
||||
text_ink = _text_ink(draw, fill)
|
||||
# What draw.text passes for a single line with no anchor at a whole
|
||||
# pixel position, by keyword so a getmask2 with another parameter
|
||||
# order cannot shift them. ink only matters to an RGBA (colour-glyph)
|
||||
# mask, which the font modes allowed here never produce.
|
||||
mask, (ox, oy) = font.getmask2(
|
||||
text, draw.fontmode, direction=None, features=None,
|
||||
language=None, stroke_width=0, anchor="la", ink=text_ink,
|
||||
start=(0.0, 0.0), stroke_filled=True)
|
||||
draw_bitmap = draw.draw.draw_bitmap
|
||||
except Exception:
|
||||
return False
|
||||
stamped = False
|
||||
try:
|
||||
# draw.text returns without drawing when its ink resolves to None.
|
||||
if outline_ink is not None:
|
||||
for dx, dy in offsets:
|
||||
draw_bitmap((x + dx + ox, y + dy + oy), mask, outline_ink)
|
||||
stamped = True
|
||||
if text_ink is not None:
|
||||
draw_bitmap((x + ox, y + oy), mask, text_ink)
|
||||
except Exception:
|
||||
# A rejected call draws nothing, but one that got through has: never
|
||||
# draw the outline twice (anti-aliased edges would be blended twice).
|
||||
if stamped:
|
||||
raise
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
class TextHelper:
|
||||
"""
|
||||
@@ -103,15 +271,15 @@ class TextHelper:
|
||||
outline_width: Width of outline in pixels
|
||||
"""
|
||||
x, y = position
|
||||
|
||||
# Draw outline by drawing text in outline color at offset positions
|
||||
for dx in range(-outline_width, outline_width + 1):
|
||||
for dy in range(-outline_width, outline_width + 1):
|
||||
if dx != 0 or dy != 0: # Skip center position
|
||||
draw.text((x + dx, y + dy), text, font=font, fill=outline_color)
|
||||
|
||||
# Draw main text
|
||||
draw.text((x, y), text, font=font, fill=fill)
|
||||
|
||||
# Outline: every offset up to outline_width away on each axis, centre
|
||||
# skipped, in the order this has always drawn them (OUTLINE_SQUARE at
|
||||
# width 1). The main text is drawn last, on top.
|
||||
offsets = [(dx, dy)
|
||||
for dx in range(-outline_width, outline_width + 1)
|
||||
for dy in range(-outline_width, outline_width + 1)
|
||||
if dx != 0 or dy != 0]
|
||||
draw_text_outlined(draw, (x, y), text, font, fill, outline_color, offsets)
|
||||
|
||||
def get_text_width(self, text: str, font: ImageFont.ImageFont) -> int:
|
||||
"""
|
||||
|
||||
+43
-10
@@ -46,6 +46,35 @@ from src.common.permission_utils import (
|
||||
get_config_dir_mode
|
||||
)
|
||||
|
||||
|
||||
def _private_copy(config: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""A deep copy of ``config`` that shares nothing with it.
|
||||
|
||||
load_config() hands one out per call, and the saves keep one, so the
|
||||
cached config is never an object a caller holds. A web handler edits what
|
||||
it loaded, validates, and may refuse the save; when the cache was that
|
||||
same object, the refused edit stayed in it, and the next save of any
|
||||
other setting wrote it to config.json -- a nested secret included, in
|
||||
plain text, since it had never reached config_secrets.json to be
|
||||
stripped.
|
||||
|
||||
The config is JSON data, so only its dicts and lists need copying; every
|
||||
other value in it is immutable. On a Pi 4 with a real 60 KiB config this
|
||||
takes 2.1 ms against copy.deepcopy's 6.8 ms, on a path ~30 handlers call
|
||||
(a pickle round trip is no faster, 1.9 ms, and brings pickle into the
|
||||
config path for nothing).
|
||||
"""
|
||||
return _copy_containers(config)
|
||||
|
||||
|
||||
def _copy_containers(value: Any) -> Any:
|
||||
if isinstance(value, dict):
|
||||
return {key: _copy_containers(item) for key, item in value.items()}
|
||||
if isinstance(value, list):
|
||||
return [_copy_containers(item) for item in value]
|
||||
return value
|
||||
|
||||
|
||||
class ConfigManager:
|
||||
"""
|
||||
Reads and writes the main application configuration files.
|
||||
@@ -126,9 +155,10 @@ class ConfigManager:
|
||||
validate_after_write=validate_after_write
|
||||
)
|
||||
|
||||
# Update in-memory config if save was successful
|
||||
# Update in-memory config if save was successful. A copy: the caller
|
||||
# still holds new_config_data (see _private_copy).
|
||||
if result.status == SaveResultStatus.SUCCESS:
|
||||
self.config = new_config_data
|
||||
self.config = _private_copy(new_config_data)
|
||||
# In-memory config now matches what was just written, so the
|
||||
# load_config fast path may return it. It still carries the
|
||||
# merged secrets that were stripped on disk; that matches a full
|
||||
@@ -208,14 +238,16 @@ class ConfigManager:
|
||||
|
||||
Fast path: when config.json, config_secrets.json and the template
|
||||
are all unchanged since the last successful load (mtime_ns + size),
|
||||
the already-parsed self.config is returned without touching the
|
||||
files — same aliasing semantics as the full path, which also
|
||||
returns self.config.
|
||||
a copy of the already-parsed self.config is returned without
|
||||
touching the files.
|
||||
|
||||
Either way the caller gets its own copy (see _private_copy): editing
|
||||
it changes nothing here until it is saved.
|
||||
"""
|
||||
try:
|
||||
current_sig = self._files_signature()
|
||||
if self.config and self._loaded_sig == current_sig:
|
||||
return self.config
|
||||
return _private_copy(self.config)
|
||||
|
||||
# Check if config file exists, if not create from template
|
||||
if not os.path.exists(self.config_path):
|
||||
@@ -249,8 +281,8 @@ class ConfigManager:
|
||||
# Signature taken AFTER load + migration (migration may write the
|
||||
# config back), so it reflects exactly what was read/written.
|
||||
self._loaded_sig = self._files_signature()
|
||||
return self.config
|
||||
|
||||
return _private_copy(self.config)
|
||||
|
||||
except FileNotFoundError as e:
|
||||
# Only config.json can get here: a missing or unreadable secrets
|
||||
# file is handled where it is read.
|
||||
@@ -355,8 +387,9 @@ class ConfigManager:
|
||||
try:
|
||||
atomic_write_json(self.config_path, config_to_write)
|
||||
|
||||
# Update the in-memory config to the new state (which includes secrets for runtime)
|
||||
self.config = new_config_data
|
||||
# Update the in-memory config to the new state (which includes
|
||||
# secrets for runtime), as a copy -- see _private_copy
|
||||
self.config = _private_copy(new_config_data)
|
||||
self._loaded_sig = self._files_signature()
|
||||
self.logger.info(f"Configuration successfully saved to {os.path.abspath(self.config_path)}")
|
||||
if secrets_content:
|
||||
|
||||
+92
-42
@@ -14,7 +14,7 @@ import json
|
||||
import time
|
||||
import threading
|
||||
from pathlib import Path
|
||||
from typing import Dict, Any, Optional, List, Callable
|
||||
from typing import Dict, Any, Optional, List, Callable, Tuple
|
||||
from collections import defaultdict
|
||||
import logging
|
||||
import hashlib
|
||||
@@ -52,7 +52,18 @@ class ConfigService:
|
||||
|
||||
# Thread safety
|
||||
self._lock: threading.RLock = threading.RLock()
|
||||
|
||||
# Held across a whole reload -- read, swap, notify -- so one reload's
|
||||
# notifications finish before the next one's start. Subscribers run
|
||||
# under this lock and never under _lock: the display's per-plugin
|
||||
# subscriber can wait seconds for a busy plugin, and get_config(),
|
||||
# subscribe() and unsubscribe() -- called from the render thread --
|
||||
# must not wait behind it.
|
||||
self._notify_lock: threading.RLock = threading.RLock()
|
||||
# (key, callback, thread id) of the callback a notification is running,
|
||||
# so unsubscribe() can wait for that one call; signalled on its return.
|
||||
self._running_callback: Optional[Tuple[str, Callable[..., None], int]] = None
|
||||
self._callback_done = threading.Condition(self._lock)
|
||||
|
||||
# Current configuration
|
||||
self._current_config: Dict[str, Any] = {}
|
||||
self._current_checksum: Optional[str] = None
|
||||
@@ -87,32 +98,33 @@ class ConfigService:
|
||||
True if config changed, False otherwise
|
||||
"""
|
||||
try:
|
||||
new_config = self.config_manager.load_config()
|
||||
new_checksum = self._calculate_checksum(new_config)
|
||||
|
||||
with self._lock:
|
||||
# Check if config actually changed
|
||||
if new_checksum == self._current_checksum:
|
||||
self.logger.debug("Configuration unchanged, skipping reload")
|
||||
return False
|
||||
|
||||
# Store old config for change detection
|
||||
old_config = self._current_config.copy()
|
||||
|
||||
# Update current config
|
||||
self._current_config = new_config
|
||||
self._current_checksum = new_checksum
|
||||
|
||||
# Notify subscribers
|
||||
with self._notify_lock:
|
||||
new_config = self.config_manager.load_config()
|
||||
new_checksum = self._calculate_checksum(new_config)
|
||||
|
||||
with self._lock:
|
||||
# Check if config actually changed
|
||||
if new_checksum == self._current_checksum:
|
||||
self.logger.debug("Configuration unchanged, skipping reload")
|
||||
return False
|
||||
|
||||
# Store old config for change detection
|
||||
old_config = self._current_config.copy()
|
||||
|
||||
# Update current config
|
||||
self._current_config = new_config
|
||||
self._current_checksum = new_checksum
|
||||
|
||||
# Notify subscribers, outside _lock (see _notify_lock)
|
||||
self._notify_subscribers(old_config, new_config)
|
||||
|
||||
|
||||
self.logger.info(
|
||||
"Configuration reloaded (checksum: %s)",
|
||||
new_checksum[:8]
|
||||
)
|
||||
|
||||
|
||||
return True
|
||||
|
||||
|
||||
except ConfigError as e:
|
||||
self.logger.error("Error loading configuration: %s", e, exc_info=True)
|
||||
return False
|
||||
@@ -127,35 +139,64 @@ class ConfigService:
|
||||
Args:
|
||||
old_config: Previous configuration
|
||||
new_config: New configuration
|
||||
|
||||
Called without _lock held. The subscriber lists are copied under it,
|
||||
and each callback is checked against them again just before it runs.
|
||||
"""
|
||||
with self._lock:
|
||||
subscribers = {key: list(callbacks) for key, callbacks in self._subscribers.items()}
|
||||
|
||||
# Notify global subscribers (key: '*')
|
||||
for callback in self._subscribers.get('*', []):
|
||||
try:
|
||||
callback(old_config, new_config)
|
||||
except Exception as e:
|
||||
self.logger.error("Error in global config change callback: %s", e, exc_info=True)
|
||||
|
||||
for callback in subscribers.get('*', []):
|
||||
self._call_subscriber('*', callback, old_config, new_config)
|
||||
|
||||
# Notify plugin-specific subscribers
|
||||
for plugin_id in self._subscribers.keys():
|
||||
for plugin_id, callbacks in subscribers.items():
|
||||
if plugin_id == '*':
|
||||
continue
|
||||
|
||||
|
||||
old_plugin_config = old_config.get(plugin_id, {})
|
||||
new_plugin_config = new_config.get(plugin_id, {})
|
||||
|
||||
|
||||
# Only notify if plugin config actually changed
|
||||
if old_plugin_config != new_plugin_config:
|
||||
for callback in self._subscribers[plugin_id]:
|
||||
try:
|
||||
callback(old_plugin_config, new_plugin_config)
|
||||
except Exception as e:
|
||||
self.logger.error(
|
||||
"Error in config change callback for %s: %s",
|
||||
plugin_id,
|
||||
e,
|
||||
exc_info=True
|
||||
)
|
||||
|
||||
for callback in callbacks:
|
||||
self._call_subscriber(plugin_id, callback,
|
||||
old_plugin_config, new_plugin_config)
|
||||
|
||||
def _call_subscriber(
|
||||
self,
|
||||
key: str,
|
||||
callback: Callable[[Dict[str, Any], Dict[str, Any]], None],
|
||||
old_config: Dict[str, Any],
|
||||
new_config: Dict[str, Any],
|
||||
) -> None:
|
||||
"""Run one callback, unless it was unsubscribed since the snapshot.
|
||||
|
||||
unsubscribe() promises that once it returns the callback is neither
|
||||
running nor will run: the display unloads the plugin straight after.
|
||||
"""
|
||||
with self._lock:
|
||||
if callback not in self._subscribers.get(key, ()):
|
||||
return
|
||||
self._running_callback = (key, callback, threading.get_ident())
|
||||
try:
|
||||
callback(old_config, new_config)
|
||||
except Exception as e:
|
||||
if key == '*':
|
||||
self.logger.error("Error in global config change callback: %s", e, exc_info=True)
|
||||
else:
|
||||
self.logger.error(
|
||||
"Error in config change callback for %s: %s",
|
||||
key,
|
||||
e,
|
||||
exc_info=True
|
||||
)
|
||||
finally:
|
||||
with self._lock:
|
||||
self._running_callback = None
|
||||
self._callback_done.notify_all()
|
||||
|
||||
def _check_file_changes(self) -> bool:
|
||||
"""
|
||||
Check if configuration files have been modified.
|
||||
@@ -276,6 +317,11 @@ class ConfigService:
|
||||
"""
|
||||
Unsubscribe from configuration changes.
|
||||
|
||||
Once this returns the callback is not running and will not be called
|
||||
again. A notification that is running this very callback is waited
|
||||
for (unless the callback is the caller); one running any other
|
||||
callback is not.
|
||||
|
||||
Args:
|
||||
callback: Callback function to remove
|
||||
plugin_id: Optional plugin ID (must match subscription)
|
||||
@@ -285,6 +331,10 @@ class ConfigService:
|
||||
if callback in self._subscribers[key]:
|
||||
self._subscribers[key].remove(callback)
|
||||
self.logger.debug("Unsubscribed from config changes for %s", key)
|
||||
while (self._running_callback is not None
|
||||
and self._running_callback[:2] == (key, callback)
|
||||
and self._running_callback[2] != threading.get_ident()):
|
||||
self._callback_done.wait()
|
||||
|
||||
def shutdown(self) -> None:
|
||||
"""Shutdown the configuration service."""
|
||||
|
||||
@@ -29,6 +29,7 @@ CORE_CONFIG_KEYS = frozenset({
|
||||
'display',
|
||||
'sync',
|
||||
'plugin_system',
|
||||
'fetch_service',
|
||||
# Older or optional core sections still found in existing config files.
|
||||
'logging',
|
||||
'network',
|
||||
|
||||
@@ -0,0 +1,550 @@
|
||||
"""What the panel shows next: the Arbiter of docs/RUN_LOOP_REDESIGN.md.
|
||||
|
||||
``Arbiter.decide(state, inputs, now)`` takes a snapshot that
|
||||
``DisplayController.run()`` gathers and returns a :class:`ScreenPlan` naming
|
||||
the Source that gets the panel. It is a pure function: no I/O, no clock
|
||||
reads (``now`` is passed in), no locks, and it changes nothing it is given.
|
||||
That is what lets a plain table of cases test the priority order, which
|
||||
used to exist only as the order of ``if`` blocks in ``run()``.
|
||||
|
||||
The order is
|
||||
|
||||
ScheduledOff (a gate), Follower, OnDemand, Wifi, Live, Vegas, Rotation
|
||||
|
||||
Every Source but Vegas is decided here (stage 3). Vegas is the ``LEGACY``
|
||||
plan: the Arbiter picks it, but its iteration is still run()'s own code
|
||||
until stage 4.
|
||||
|
||||
A pass asks twice: once with the inputs every pass reads (the gate,
|
||||
Follower, OnDemand, Wifi), and once more, only when nothing above the
|
||||
notice took the panel, with the inputs the Sources below it need (whether
|
||||
Vegas is on, the live-priority scan), read where run() always read them.
|
||||
|
||||
``decide(..., running=plan)`` is the other question, asked by the
|
||||
ScreenRunner (src/screen_runner.py) at its service points: does a Source
|
||||
in ``plan.preemptible_by`` now take the panel from the screen that is
|
||||
running? The mid-screen rules are :func:`_hold_or_preempt`.
|
||||
|
||||
The state transitions (the next on-demand mode, a live claim and its
|
||||
release, the rotation's step after a screen) are pure methods of
|
||||
:class:`ArbiterState`; the controller applies what they return.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass, replace
|
||||
from enum import Enum
|
||||
from typing import FrozenSet, Optional, Protocol, Tuple
|
||||
|
||||
__all__ = [
|
||||
"Arbiter",
|
||||
"ArbiterInputs",
|
||||
"ArbiterState",
|
||||
"FramePolicy",
|
||||
"LIVE_PREEMPTERS",
|
||||
"SCHEDULED_OFF_DWELL",
|
||||
"SCREEN_PREEMPTERS",
|
||||
"ScreenEnd",
|
||||
"ScreenPlan",
|
||||
"Source",
|
||||
"WIFI_NOTICE_DWELL",
|
||||
"WifiNotice",
|
||||
"live_pick",
|
||||
"live_takeover",
|
||||
"on_demand_bound",
|
||||
"rotation_plan",
|
||||
"wifi_notice_preempts",
|
||||
]
|
||||
|
||||
# How long one scheduled-off pass blanks the panel. The dwell ends early when
|
||||
# on-demand starts or the schedule turns the panel back on.
|
||||
SCHEDULED_OFF_DWELL = 60.0
|
||||
|
||||
# How long one WiFi-notice pass holds the notice before the next pass looks
|
||||
# again; the notice stays up, pass after pass, until it expires.
|
||||
WIFI_NOTICE_DWELL = 0.5
|
||||
|
||||
|
||||
class Source(Enum):
|
||||
"""Who gets the panel this pass."""
|
||||
|
||||
SCHEDULED_OFF = "scheduled-off"
|
||||
FOLLOWER = "follower"
|
||||
ON_DEMAND = "on-demand"
|
||||
WIFI = "wifi"
|
||||
LIVE = "live"
|
||||
# Vegas: the Arbiter picks it, but its iteration is still run()'s own
|
||||
# code (and its interrupt callback a second copy of this order) until
|
||||
# stage 4 makes it a Source driven frame by frame.
|
||||
LEGACY = "legacy"
|
||||
ROTATION = "rotation"
|
||||
# Not a screen: a plugin reload waits at the top of the loop. It ends a
|
||||
# screen between frames (the screen counts as shown and the rotation
|
||||
# moves on), and the next pass reloads before it draws.
|
||||
RELOAD = "reload"
|
||||
|
||||
|
||||
class FramePolicy(Enum):
|
||||
"""How often a screen draws: today's two frame loops (see
|
||||
DisplayController._needs_high_fps). Stage 5 lets plugins declare it."""
|
||||
|
||||
#: The 125 Hz loop, paced to an 8 ms deadline: scrolling plugins.
|
||||
HIGH_FPS = "high-fps"
|
||||
#: The 1 Hz loop.
|
||||
STATIC = "static"
|
||||
|
||||
|
||||
class ScreenEnd(Protocol):
|
||||
"""What ArbiterState.after needs to know about how a screen ended
|
||||
(screen_runner.Outcome, filled in by the controller)."""
|
||||
|
||||
@property
|
||||
def on_demand_active(self) -> bool:
|
||||
"""An on-demand session was running when the screen ended."""
|
||||
|
||||
@property
|
||||
def still_live(self) -> bool:
|
||||
"""The mode's plugin still had live content: hold the rotation."""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class WifiNotice:
|
||||
"""A WiFi status message waiting to be drawn.
|
||||
|
||||
``expires_at`` is wall-clock time (``time.time()``), as written by the
|
||||
WiFi manager.
|
||||
"""
|
||||
|
||||
message: str
|
||||
expires_at: float
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ArbiterState:
|
||||
"""What the Arbiter remembers between passes.
|
||||
|
||||
A snapshot of the controller's own fields, taken when decide() is
|
||||
called (DisplayController._arbiter_state); the transitions below return
|
||||
the next state, and the controller writes it back.
|
||||
|
||||
Attributes:
|
||||
current_mode: The mode on the panel or about to be
|
||||
(``current_display_mode``).
|
||||
on_demand_modes: The on-demand session's modes, in the order it
|
||||
shows them (a pinned mode already moved to the front when the
|
||||
session started, by _apply_on_demand_pin).
|
||||
on_demand_index: Which of them is showing.
|
||||
on_demand_expires_at: When the session ends (wall clock), or None
|
||||
for a session with no duration.
|
||||
on_demand_pinned: The session was started pinned. Carried for the
|
||||
snapshot; the pin itself is already in ``on_demand_modes``.
|
||||
rotation: The rotation's modes (``available_modes``).
|
||||
rotation_index: Where the rotation is (``current_mode_index``).
|
||||
live_resume_index: Where the rotation was when live priority took
|
||||
the panel, so it resumes there once nothing is live; None while
|
||||
live priority holds nothing.
|
||||
live_takeover_unshown: A mid-screen takeover chose current_mode and
|
||||
it has not been shown yet, so the next pass must not advance the
|
||||
live round-robin past it.
|
||||
"""
|
||||
|
||||
current_mode: Optional[str] = None
|
||||
on_demand_modes: Tuple[str, ...] = ()
|
||||
on_demand_index: int = 0
|
||||
on_demand_expires_at: Optional[float] = None
|
||||
on_demand_pinned: bool = False
|
||||
rotation: Tuple[str, ...] = ()
|
||||
rotation_index: int = 0
|
||||
live_resume_index: Optional[int] = None
|
||||
live_takeover_unshown: bool = False
|
||||
|
||||
def next_on_demand(self) -> "ArbiterState":
|
||||
"""The session's next mode, wrapping round. Needs a mode list."""
|
||||
index = (self.on_demand_index + 1) % len(self.on_demand_modes)
|
||||
return replace(self, on_demand_index=index,
|
||||
current_mode=self.on_demand_modes[index])
|
||||
|
||||
def claim_live(self, mode: str) -> "ArbiterState":
|
||||
"""Live priority takes the panel for ``mode``.
|
||||
|
||||
The rotation's position is saved only on the first claim, not on
|
||||
each re-check while the hold continues, so it resumes where live
|
||||
priority interrupted it instead of after the live mode (which would
|
||||
skip every mode between the two).
|
||||
"""
|
||||
if self.current_mode == mode:
|
||||
return self
|
||||
resume = self.rotation_index if self.live_resume_index is None else self.live_resume_index
|
||||
index = self.rotation.index(mode) if mode in self.rotation else self.rotation_index
|
||||
return replace(self, current_mode=mode, rotation_index=index,
|
||||
live_resume_index=resume)
|
||||
|
||||
def after(self, outcome: "ScreenEnd") -> "ArbiterState":
|
||||
"""The state once a screen has run its course: the next mode.
|
||||
|
||||
An on-demand session moves to its next mode. Otherwise the rotation
|
||||
advances -- unless the mode just shown is a live-priority mode that
|
||||
is still live, which holds the panel. A session with no modes left
|
||||
is ended by the controller before it asks (that is not pure: it
|
||||
resumes the rotation and clears the cache).
|
||||
"""
|
||||
if outcome.on_demand_active:
|
||||
return self.next_on_demand() if self.on_demand_modes else self
|
||||
if outcome.still_live or not self.rotation:
|
||||
return self
|
||||
index = (self.rotation_index + 1) % len(self.rotation)
|
||||
return replace(self, rotation_index=index, current_mode=self.rotation[index])
|
||||
|
||||
def release_live(self) -> "ArbiterState":
|
||||
"""Nothing is live any more: the rotation resumes where it was."""
|
||||
if self.live_resume_index is None or not self.rotation:
|
||||
return self
|
||||
index = self.live_resume_index % len(self.rotation)
|
||||
return replace(self, current_mode=self.rotation[index], rotation_index=index,
|
||||
live_resume_index=None)
|
||||
|
||||
def showing(self, plan: "ScreenPlan") -> "ArbiterState":
|
||||
"""The state once ``plan`` is on the panel.
|
||||
|
||||
An on-demand plan puts the session's index on the mode it shows (an
|
||||
index past the end of a shortened list starts it again at 0).
|
||||
"""
|
||||
state = replace(self, current_mode=plan.mode)
|
||||
if plan.source is Source.ON_DEMAND and self.on_demand_modes:
|
||||
state = replace(state, on_demand_index=_on_demand_index(self))
|
||||
return state
|
||||
|
||||
|
||||
def _on_demand_index(state: ArbiterState) -> int:
|
||||
"""The session's index, or 0 once it is past the end of its list."""
|
||||
index = state.on_demand_index
|
||||
return index if index < len(state.on_demand_modes) else 0
|
||||
|
||||
|
||||
def on_demand_bound(min_duration: float, max_duration: float,
|
||||
deadline: Optional[float],
|
||||
now: float) -> Optional[Tuple[float, float]]:
|
||||
"""Shorten a screen's (min, max) seconds to what is left of a timed
|
||||
on-demand session ending at ``deadline``. None when nothing is left.
|
||||
|
||||
The OnDemand Source's bound, applied after the screen's first frame,
|
||||
where it always was (``now`` is read then).
|
||||
"""
|
||||
if deadline is None:
|
||||
return min_duration, max_duration
|
||||
remaining = max(0.0, deadline - now)
|
||||
min_duration = min(min_duration, remaining)
|
||||
max_duration = min(max_duration, remaining)
|
||||
if max_duration <= 0:
|
||||
return None
|
||||
return min_duration, max_duration
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ArbiterInputs:
|
||||
"""One pass's snapshot, gathered by run() before it calls decide().
|
||||
|
||||
Attributes:
|
||||
schedule_on: The display schedule has the panel on, not counting an
|
||||
on-demand override of a scheduled-off window.
|
||||
on_demand_active: An on-demand session is running.
|
||||
follower_active: A sync leader is driving this panel.
|
||||
wifi_notice: The pending WiFi notice, or None. run() reads it only
|
||||
when it could win (the panel is on, and neither a follower nor
|
||||
on-demand outranks it), because reading it has side effects: a
|
||||
1 Hz throttle and deleting an expired file.
|
||||
live_modes: The modes with live content, from a live-priority scan,
|
||||
in registration order; None when no scan was made (on-demand,
|
||||
Vegas keeping live content in its ticker, a throttled
|
||||
mid-screen check). A scan asks every live-priority plugin, so it
|
||||
is made only where run() always made it.
|
||||
vegas_enabled: Vegas mode is on (and no on-demand session holds it
|
||||
off).
|
||||
vegas_live_in_ticker: Vegas keeps live content in its ticker
|
||||
instead of yielding the panel to it.
|
||||
vegas_yielded: This pass's Vegas iteration has run and yielded, so
|
||||
the Vegas Source passes and the screen it fell through to is
|
||||
decided.
|
||||
reload_pending: Mid-screen only: a plugin reload is waiting for the
|
||||
top of the loop, at a service point where that ends the screen.
|
||||
"""
|
||||
|
||||
schedule_on: bool
|
||||
on_demand_active: bool
|
||||
follower_active: bool
|
||||
wifi_notice: Optional[WifiNotice] = None
|
||||
live_modes: Optional[Tuple[str, ...]] = None
|
||||
vegas_enabled: bool = False
|
||||
vegas_live_in_ticker: bool = False
|
||||
vegas_yielded: bool = False
|
||||
reload_pending: bool = False
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ScreenPlan:
|
||||
"""The Arbiter's answer for one pass.
|
||||
|
||||
decide() is pure, so it cannot ask a plugin anything: the fields a
|
||||
plugin answers (its durations, whether it runs a dynamic cycle, how
|
||||
often it draws) are filled in by the controller after the screen's first
|
||||
frame, when they have always been read (DisplayController.complete_plan).
|
||||
|
||||
Attributes:
|
||||
source: The Source that gets the panel.
|
||||
mode: The display mode to draw (None for a blank, follower or notice).
|
||||
plugin: The id of the plugin drawing ``mode``, once resolved.
|
||||
min_duration: Seconds the screen runs at least (dynamic duration).
|
||||
max_duration: How long the plan holds the panel, in seconds, at most
|
||||
(its dwell ends early when what the panel should show changes).
|
||||
None when the Source paces itself (a follower frame, Vegas), and
|
||||
for a rotation or live plan until its first frame.
|
||||
dynamic: Run until the plugin's cycle completes, between min and max.
|
||||
frame_policy: Which frame loop the screen runs.
|
||||
preemptible_by: The Sources that may end the screen mid-way.
|
||||
notice: The WiFi notice to draw, for a WIFI plan.
|
||||
deadline: For an on-demand plan, when the session ends (wall
|
||||
clock): after the first frame the screen's durations are cut to
|
||||
what is left (:func:`on_demand_bound`).
|
||||
ends_live: Nothing is live any more and live priority had
|
||||
interrupted the rotation: taking this plan resumes the rotation
|
||||
where it was (ArbiterState.release_live) before it shows.
|
||||
"""
|
||||
|
||||
source: Source
|
||||
mode: Optional[str] = None
|
||||
plugin: Optional[str] = None
|
||||
min_duration: Optional[float] = None
|
||||
max_duration: Optional[float] = None
|
||||
dynamic: bool = False
|
||||
frame_policy: Optional[FramePolicy] = None
|
||||
preemptible_by: FrozenSet[Source] = frozenset()
|
||||
notice: Optional[WifiNotice] = None
|
||||
deadline: Optional[float] = None
|
||||
ends_live: bool = False
|
||||
|
||||
|
||||
SCHEDULED_OFF_PLAN = ScreenPlan(Source.SCHEDULED_OFF, max_duration=SCHEDULED_OFF_DWELL)
|
||||
FOLLOWER_PLAN = ScreenPlan(Source.FOLLOWER)
|
||||
RELOAD_PLAN = ScreenPlan(Source.RELOAD)
|
||||
|
||||
#: What may end a screen mid-way: the schedule, an on-demand session
|
||||
#: starting or ending, a WiFi notice, a live game, the rotation moving
|
||||
#: under the screen, and a plugin reload. Not a follower or Vegas: those
|
||||
#: are only looked at between screens.
|
||||
SCREEN_PREEMPTERS: FrozenSet[Source] = frozenset(
|
||||
{Source.SCHEDULED_OFF, Source.ON_DEMAND, Source.WIFI, Source.LIVE, Source.ROTATION,
|
||||
Source.RELOAD})
|
||||
|
||||
#: A live screen is not preempted by Live: live games take turns between
|
||||
#: screens, never mid-screen.
|
||||
LIVE_PREEMPTERS: FrozenSet[Source] = SCREEN_PREEMPTERS - {Source.LIVE}
|
||||
|
||||
|
||||
class Arbiter:
|
||||
"""Decides which Source gets the panel. Stateless; see the module docstring."""
|
||||
|
||||
@staticmethod
|
||||
def decide(state: ArbiterState, inputs: ArbiterInputs, now: float,
|
||||
running: Optional[ScreenPlan] = None) -> ScreenPlan:
|
||||
"""The plan for this pass, from the Sources in priority order.
|
||||
|
||||
With ``running``, the question is the ScreenRunner's at one of its
|
||||
service points instead: does a Source in ``running.preemptible_by``
|
||||
now take the panel from that screen? The answer is ``running``
|
||||
itself (the same object) while it holds, else the plan that ends it.
|
||||
See :func:`_hold_or_preempt` for the rules.
|
||||
|
||||
Args:
|
||||
state: What the Arbiter remembers between passes.
|
||||
inputs: This pass's snapshot.
|
||||
now: Wall-clock time of the snapshot. The OnDemand Source reads
|
||||
it for what is left of a timed session, and the mid-screen
|
||||
WiFi rule (:func:`wifi_notice_preempts`) to compare with the
|
||||
notice's expiry. The top-of-pass WiFi check does not: it
|
||||
takes the notice as read.
|
||||
running: The screen on the panel, for a mid-screen check.
|
||||
|
||||
Returns:
|
||||
The winning Source's plan (LEGACY for Vegas), or ``running``.
|
||||
"""
|
||||
if running is not None:
|
||||
return _hold_or_preempt(state, inputs, now, running)
|
||||
|
||||
# ScheduledOff is a gate, not a Source: a scheduled-off panel stays
|
||||
# blank even for a follower, and only an on-demand session overrides
|
||||
# it (#714 -- one ending in off hours blanks at the next pass).
|
||||
if not inputs.schedule_on and not inputs.on_demand_active:
|
||||
return SCHEDULED_OFF_PLAN
|
||||
|
||||
# 1. Follower: a sync leader drives this panel, ahead of on-demand.
|
||||
if inputs.follower_active:
|
||||
return FOLLOWER_PLAN
|
||||
|
||||
# 2. OnDemand: the session's current mode. It outranks the notice.
|
||||
if inputs.on_demand_active:
|
||||
return _on_demand_plan(state, now)
|
||||
|
||||
# 3. Wifi: a pending notice, held for one short dwell per pass.
|
||||
if inputs.wifi_notice is not None:
|
||||
return ScreenPlan(Source.WIFI, max_duration=WIFI_NOTICE_DWELL,
|
||||
notice=inputs.wifi_notice)
|
||||
|
||||
# 4. Live: the next live game, round-robin across several. With
|
||||
# nothing live, a rotation that live priority interrupted resumes.
|
||||
ends_live = False
|
||||
if _live_applies(inputs):
|
||||
pick = live_pick(inputs.live_modes, state.current_mode,
|
||||
advance=not state.live_takeover_unshown)
|
||||
if pick is not None:
|
||||
return ScreenPlan(Source.LIVE, mode=pick, preemptible_by=LIVE_PREEMPTERS)
|
||||
ends_live = state.live_resume_index is not None and bool(state.rotation)
|
||||
|
||||
# 5. Vegas: one iteration of the ticker, run by run()'s own code
|
||||
# until stage 4. Passes once this pass's iteration has yielded.
|
||||
if inputs.vegas_enabled and not inputs.vegas_yielded:
|
||||
return ScreenPlan(Source.LEGACY, ends_live=ends_live)
|
||||
|
||||
# 6. Rotation: the rotation's current mode (after the resume, when
|
||||
# live priority just ended).
|
||||
return rotation_plan(state.release_live() if ends_live else state,
|
||||
ends_live=ends_live)
|
||||
|
||||
|
||||
def _hold_or_preempt(state: ArbiterState, inputs: ArbiterInputs, now: float,
|
||||
running: ScreenPlan) -> ScreenPlan:
|
||||
"""The mid-screen rules: ``running``, or the plan that ends it.
|
||||
|
||||
What the frame loops used to check one by one (_check_live_takeover,
|
||||
then _screen_preempted with _wifi_notice_pending in it, before stage 3),
|
||||
in their order:
|
||||
|
||||
1. Live: a game went live while a non-live screen runs (the inputs
|
||||
carry a scan only when one was due, at most once a second). Checked
|
||||
first because it is the one preemption that changes the state -- the
|
||||
rotation moves to the live mode and remembers where it was -- and it
|
||||
still happens when a WiFi notice is also pending: the next pass then
|
||||
shows the notice, and the game after it.
|
||||
2. The panel's mode moved under the screen: an on-demand session
|
||||
started, ended or changed mode, or the rotation was rebuilt (a
|
||||
plugin enabled, disabled or reloaded).
|
||||
3. The schedule turned the panel off.
|
||||
4. A WiFi notice arrived (unless on-demand outranks it), compared with
|
||||
its expiry because the read throttle can hand back a stale one.
|
||||
5. A plugin reload is waiting at the top of the loop.
|
||||
|
||||
A follower and Vegas are never mid-screen preemptions; they are looked
|
||||
at between screens.
|
||||
"""
|
||||
by = running.preemptible_by
|
||||
if Source.LIVE in by:
|
||||
takeover = live_takeover(state, inputs)
|
||||
if takeover is not None:
|
||||
return ScreenPlan(Source.LIVE, mode=takeover, preemptible_by=LIVE_PREEMPTERS)
|
||||
if state.current_mode != running.mode:
|
||||
source = Source.ON_DEMAND if inputs.on_demand_active else Source.ROTATION
|
||||
if source in by:
|
||||
return ScreenPlan(source, mode=state.current_mode, preemptible_by=SCREEN_PREEMPTERS)
|
||||
if (Source.SCHEDULED_OFF in by and not inputs.schedule_on
|
||||
and not inputs.on_demand_active):
|
||||
return SCHEDULED_OFF_PLAN
|
||||
notice = inputs.wifi_notice
|
||||
if (Source.WIFI in by and notice is not None
|
||||
and wifi_notice_preempts(notice, inputs.on_demand_active, now)):
|
||||
return ScreenPlan(Source.WIFI, max_duration=WIFI_NOTICE_DWELL, notice=notice)
|
||||
if Source.RELOAD in by and inputs.reload_pending:
|
||||
return RELOAD_PLAN
|
||||
return running
|
||||
|
||||
|
||||
def live_takeover(state: ArbiterState, inputs: ArbiterInputs) -> Optional[str]:
|
||||
"""The live mode that takes the panel mid-screen, or None.
|
||||
|
||||
The first live mode, when a scan found one and the panel is not on a
|
||||
live mode already. Never while on-demand holds the panel, while it is
|
||||
scheduled off, or while Vegas keeps live content in its ticker.
|
||||
"""
|
||||
if not _live_applies(inputs) or inputs.on_demand_active or not inputs.schedule_on:
|
||||
return None
|
||||
live = inputs.live_modes
|
||||
if not live or state.current_mode in live:
|
||||
return None
|
||||
return live[0]
|
||||
|
||||
|
||||
def _on_demand_plan(state: ArbiterState, now: float) -> ScreenPlan:
|
||||
"""The OnDemand Source: the session's current mode.
|
||||
|
||||
``max_duration`` is what is left of a timed session at ``now`` (None
|
||||
without a duration); ``deadline`` carries the expiry so the bound can be
|
||||
applied again after the first frame. A session with no modes left (its
|
||||
plugin was unloaded under it) gets a plan with no mode: the controller
|
||||
ends the session and shows the rotation's mode instead.
|
||||
"""
|
||||
modes = state.on_demand_modes
|
||||
if not modes:
|
||||
return ScreenPlan(Source.ON_DEMAND)
|
||||
expires_at = state.on_demand_expires_at
|
||||
remaining = None if expires_at is None else max(0.0, expires_at - now)
|
||||
return ScreenPlan(Source.ON_DEMAND, mode=modes[_on_demand_index(state)],
|
||||
max_duration=remaining, deadline=expires_at,
|
||||
preemptible_by=SCREEN_PREEMPTERS)
|
||||
|
||||
|
||||
def rotation_plan(state: ArbiterState, ends_live: bool = False) -> ScreenPlan:
|
||||
"""The Rotation Source: the mode the rotation is on.
|
||||
|
||||
That is ``state.current_mode``, which is ``rotation[rotation_index]``
|
||||
except where something moved the panel off the list and the rotation
|
||||
carries on from there: a live mode no rotation entry names, or None
|
||||
when a session ended with no enabled mode to resume to.
|
||||
"""
|
||||
return ScreenPlan(Source.ROTATION, mode=state.current_mode, ends_live=ends_live,
|
||||
preemptible_by=SCREEN_PREEMPTERS)
|
||||
|
||||
|
||||
def _live_applies(inputs: ArbiterInputs) -> bool:
|
||||
"""Whether the Live Source has a say: a scan was made, and Vegas is not
|
||||
keeping live content in its ticker (where the live plugin takes extra
|
||||
turns in the marquee instead of the panel)."""
|
||||
if inputs.live_modes is None:
|
||||
return False
|
||||
return not (inputs.vegas_enabled and inputs.vegas_live_in_ticker)
|
||||
|
||||
|
||||
def live_pick(live_modes: Optional[Tuple[str, ...]], current_mode: Optional[str],
|
||||
advance: bool) -> Optional[str]:
|
||||
"""The live mode to show, or None when nothing is live.
|
||||
|
||||
When several plugins are live at once this round-robins between them, so
|
||||
the panel alternates each dwell instead of pinning to the first one
|
||||
registered. The mode on the panel is the cursor, so this stays right as
|
||||
games start and end.
|
||||
|
||||
Args:
|
||||
live_modes: The live modes, in registration order.
|
||||
current_mode: The mode on the panel.
|
||||
advance: True for the rotation's pick (the live mode after the one
|
||||
showing). False for a peek (the one showing if it is still live,
|
||||
else the first), which Vegas uses to ask whether anything is.
|
||||
"""
|
||||
if not live_modes:
|
||||
return None
|
||||
if current_mode in live_modes:
|
||||
if advance:
|
||||
index = live_modes.index(current_mode)
|
||||
return live_modes[(index + 1) % len(live_modes)]
|
||||
return current_mode
|
||||
return live_modes[0]
|
||||
|
||||
|
||||
def wifi_notice_preempts(notice: Optional[WifiNotice], on_demand_active: bool,
|
||||
now: float) -> bool:
|
||||
"""Whether a WiFi notice should end the current screen early.
|
||||
|
||||
The Wifi Source's mid-screen rule, polled between frames, during dwells
|
||||
and when a Vegas iteration yields. On-demand outranks the notice, as in
|
||||
:meth:`Arbiter.decide`. Unlike the top-of-pass check it also compares
|
||||
``now`` with the expiry, because the 1 Hz read throttle can hand back a
|
||||
notice that has expired since it was read.
|
||||
"""
|
||||
if on_demand_active or notice is None:
|
||||
return False
|
||||
return now < notice.expires_at
|
||||
+2108
-834
File diff suppressed because it is too large
Load Diff
+232
-44
@@ -51,13 +51,14 @@ from src.pi5_matrix_support import is_raspberry_pi_5
|
||||
import threading
|
||||
import time
|
||||
from collections import OrderedDict, deque
|
||||
from typing import Dict, Any, Optional, Tuple, TYPE_CHECKING
|
||||
from typing import Dict, Any, List, Optional, Tuple, TYPE_CHECKING
|
||||
import zlib
|
||||
import freetype
|
||||
|
||||
from src.common import snapshot_policy
|
||||
from src import display_watchdog
|
||||
from src.common.frame_timing import FrameTimingRecorder
|
||||
from src.common.frame_timing import (
|
||||
FrameTimingRecorder, install_gc_monitor, uninstall_gc_monitor)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from src.common.render_gate import RenderGate
|
||||
@@ -80,6 +81,15 @@ _CALENDAR_FONT_PX = 7
|
||||
#: frame, so a fault that persists would otherwise log ~100 lines a second.
|
||||
_UPDATE_ERROR_LOG_INTERVAL = 60.0
|
||||
|
||||
#: zlib level for the preview snapshot PNG. The fastest level: each file is
|
||||
#: read by the web UI and soon replaced by the next, so encode time (paid on
|
||||
#: the render thread for a static screen) matters more than its size.
|
||||
#: Lossless at any level. Against Pillow's default (6), on a desktop with
|
||||
#: Pillow 12.3, a text-dense 512x64 frame encoded in about half the time,
|
||||
#: into 12 KB instead of 7 KB; sparser frames saved less time (10-20%) and
|
||||
#: stayed under 2 KB.
|
||||
_SNAPSHOT_PNG_COMPRESS_LEVEL = 1
|
||||
|
||||
|
||||
def _bdf_native_size(face) -> int:
|
||||
"""The pixel height a BDF Face declares, or 0 if it does not say.
|
||||
@@ -223,6 +233,11 @@ def _per_thread_canvas_attr(name: str) -> property:
|
||||
|
||||
|
||||
|
||||
#: A held frame is only split when its blit takes less than this share of a
|
||||
#: refresh: the second blit has to land before the next vsync.
|
||||
_SPLIT_BLIT_FRACTION = 0.5
|
||||
|
||||
|
||||
class DisplayManager:
|
||||
"""
|
||||
Singleton hardware abstraction layer for the RGB LED matrix.
|
||||
@@ -291,8 +306,8 @@ class DisplayManager:
|
||||
self._TEXT_WIDTH_CACHE_MAX = 1024
|
||||
# Snapshot mirror for web preview + health check (service writes, web
|
||||
# reads). Cadence/skip decisions live in src/common/snapshot_policy.py:
|
||||
# full rate only while the web SSE broadcaster keeps the viewer marker
|
||||
# fresh; unchanged frames are never re-encoded, only mtime-touched.
|
||||
# the viewer rate only while the web SSE broadcaster keeps the viewer
|
||||
# marker fresh; unchanged frames are never re-encoded, only mtime-touched.
|
||||
self._snapshot_path = "/tmp/led_matrix_preview.png" # nosec B108 - fixed path intentional; web UI reads same path
|
||||
self._viewer_marker_path = "/tmp/led_matrix_preview_viewer" # nosec B108 - touched by web SSE broadcaster
|
||||
self._last_snapshot_ts = 0.0
|
||||
@@ -302,6 +317,11 @@ class DisplayManager:
|
||||
# is handed to the writer; this only once it has been saved, so an
|
||||
# mtime touch never vouches for a frame still waiting to be written.
|
||||
self._saved_snapshot_digest: Optional[int] = None
|
||||
# A changed frame reached _write_snapshot_if_due() inside the write
|
||||
# interval and was skipped. Nothing writes it unless update_display()
|
||||
# runs again, and a screen that draws once and holds never calls it
|
||||
# again -- see write_owed_snapshot().
|
||||
self._snapshot_owed = False
|
||||
self._snapshot_dir_prepared = False
|
||||
# Background writer used mid-scroll; see _write_snapshot_if_due.
|
||||
self._snapshot_cond = threading.Condition()
|
||||
@@ -339,6 +359,10 @@ class DisplayManager:
|
||||
# advances a whole pixel every Nth refresh instead of every one.
|
||||
# See src/common/scroll_config.py and scripts/scroll_speeds.py.
|
||||
self._frame_hold = 1
|
||||
# True while a static screen draws its first frame after a scroll,
|
||||
# whose state is left set until then: those frames go out without
|
||||
# scan-order compensation. See end_scroll_for_static_screen().
|
||||
self._static_handover = False
|
||||
|
||||
# A src.common.render_gate.RenderGate while Vegas runs with
|
||||
# vegas_scroll.prefetch_gate on: opened around each swap so the
|
||||
@@ -347,7 +371,8 @@ class DisplayManager:
|
||||
|
||||
# Timing of every presented frame, whoever drew it, for
|
||||
# scripts/frame_soak.py. See src/common/frame_timing.py.
|
||||
self.frame_timing = FrameTimingRecorder(info=self._frame_timing_info())
|
||||
self.frame_timing = FrameTimingRecorder(
|
||||
info=self._frame_timing_info(), gc_monitor=install_gc_monitor())
|
||||
self.frame_timing.scrolling_now = self._scrolling_now
|
||||
|
||||
self._scrolling_state = {
|
||||
@@ -937,16 +962,32 @@ class DisplayManager:
|
||||
self._write_snapshot_if_due()
|
||||
return
|
||||
|
||||
# Asked once per frame and the answer reused below: the call
|
||||
# has side effects (it expires a stale scroll and drops its
|
||||
# frame hold), so asking again further down could disagree
|
||||
# with what this frame was already treated as. Asked first,
|
||||
# so the dirty check, the scan-order segments, the pacing gate,
|
||||
# the swaps and frame timing all see one answer and the frame
|
||||
# hold it leaves.
|
||||
scrolling = self.is_currently_scrolling()
|
||||
digest = None
|
||||
frame_checksum = None
|
||||
if self._dirty_tracking_enabled:
|
||||
# No digest mid-scroll. The skip it feeds is never taken while
|
||||
# scrolling (see below), so all it bought there was the
|
||||
# snapshot's changed-frame check -- a tobytes() plus adler32
|
||||
# over the whole framebuffer every frame (~0.17ms at 256x64
|
||||
# on a Pi 4, twice that at 512x64) for a decision acted on at
|
||||
# most once a second. _write_snapshot_if_due hashes for
|
||||
# itself when a write or touch is actually due. The cost: the
|
||||
# first static frame after a scroll is always pushed, once.
|
||||
if self._dirty_tracking_enabled and not scrolling:
|
||||
try:
|
||||
brightness = getattr(self.matrix, 'brightness', None)
|
||||
except AttributeError:
|
||||
brightness = None
|
||||
frame_checksum = zlib.adler32(self.image.tobytes())
|
||||
digest = (frame_checksum, brightness)
|
||||
if digest == self._last_pushed_digest and not self.is_currently_scrolling():
|
||||
if digest == self._last_pushed_digest:
|
||||
# Nothing changed since the last push — the panel is
|
||||
# already showing exactly this frame.
|
||||
#
|
||||
@@ -971,27 +1012,35 @@ class DisplayManager:
|
||||
# mode the logical screen is first tiled across the full chain.
|
||||
blit_started = time.perf_counter()
|
||||
if self._double_sided is not None:
|
||||
self.offscreen_canvas.SetImage(self._composite_double_sided())
|
||||
segments = [(self._composite_double_sided(), self._frame_hold)]
|
||||
else:
|
||||
self.offscreen_canvas.SetImage(self._scan_compensated(self.image))
|
||||
blit_done = time.perf_counter()
|
||||
|
||||
# Swap buffers immediately. framerate_fraction holds the frame
|
||||
# for N refreshes; SwapOnVSync blocks for all of them, which is
|
||||
# what paces the render loop to the chosen frame rate.
|
||||
segments = self._scan_segments(self.image, scrolling)
|
||||
gate = self.render_gate
|
||||
blit_time = swap_time = 0.0
|
||||
# Usually one segment: the frame, held for _frame_hold
|
||||
# refreshes. SwapOnVSync blocks for all of them, which is what
|
||||
# paces the render loop to the chosen frame rate. Scan-order
|
||||
# compensation on a held frame splits it, so the lagging rows
|
||||
# change one refresh after the rest.
|
||||
if gate is not None:
|
||||
gate.before_swap(self._frame_hold)
|
||||
self.matrix.SwapOnVSync(self.offscreen_canvas, self._frame_hold)
|
||||
for index, (shown, hold) in enumerate(segments):
|
||||
if index:
|
||||
blit_started = time.perf_counter()
|
||||
self.offscreen_canvas.SetImage(shown)
|
||||
blit_done = time.perf_counter()
|
||||
blit_time += blit_done - blit_started
|
||||
self.matrix.SwapOnVSync(self.offscreen_canvas, hold)
|
||||
swap_time += time.perf_counter() - blit_done
|
||||
# Swap our canvas references
|
||||
self.offscreen_canvas, self.current_canvas = self.current_canvas, self.offscreen_canvas
|
||||
if gate is not None:
|
||||
gate.after_swap(self._frame_hold)
|
||||
presented_at = time.perf_counter()
|
||||
self._last_blit_seconds = blit_time / len(segments)
|
||||
self.frame_timing.record(
|
||||
blit_done - blit_started, presented_at - blit_done,
|
||||
self._frame_hold, self.is_currently_scrolling(), presented_at)
|
||||
|
||||
# Swap our canvas references
|
||||
self.offscreen_canvas, self.current_canvas = self.current_canvas, self.offscreen_canvas
|
||||
blit_time, swap_time,
|
||||
self._frame_hold, scrolling, presented_at)
|
||||
|
||||
self._last_pushed_digest = digest
|
||||
|
||||
@@ -1038,23 +1087,48 @@ class DisplayManager:
|
||||
", ".join(f"rows {top}-{bottom - 1} show {lag} refresh(es) behind"
|
||||
for top, bottom, lag in bands))
|
||||
|
||||
def _scan_compensated(self, image: Image.Image) -> Image.Image:
|
||||
"""The frame to present, with lagging rows taken from earlier frames.
|
||||
def _scan_segments(self, image: Image.Image,
|
||||
scrolling: Optional[bool] = None
|
||||
) -> List[Tuple[Image.Image, int]]:
|
||||
"""What to present for this frame: ``[(image, refreshes), ...]``.
|
||||
|
||||
Only mid-scroll at one frame per refresh: that is when consecutive
|
||||
frames are consecutive refreshes. At a longer hold, or on a static
|
||||
screen, the history is dropped and the frame goes out as it is.
|
||||
Mid-scroll with compensation on, lagging rows are taken from earlier
|
||||
refreshes (see src/scan_order.py). At one refresh per frame that is one
|
||||
image. A frame held longer is split at the refresh where the lagging
|
||||
rows catch up, so those rows step a refresh after the rest. The split
|
||||
needs a second blit inside the refresh that follows the first swap, so
|
||||
it is skipped when a blit is too slow to fit. A static screen goes out
|
||||
as it is, and drops the history. So does a static screen's first frame
|
||||
after a scroll, while the scroll state is still set (see
|
||||
end_scroll_for_static_screen): one segment, held for the scroll's
|
||||
hold, with no rows from the scroller's frames.
|
||||
|
||||
``scrolling`` is the caller's is_currently_scrolling() answer for this
|
||||
frame. update_display() asks once, before calling this, so the frame
|
||||
hold read here is the one that answer left (an expired scroll's hold
|
||||
is already dropped). None asks here.
|
||||
"""
|
||||
hold = self._frame_hold
|
||||
bands = getattr(self, '_scan_lag_bands', None)
|
||||
if not bands:
|
||||
return image
|
||||
if self._frame_hold != 1 or not self.is_currently_scrolling():
|
||||
self._scan_history.clear()
|
||||
return image
|
||||
presented = scan_order.compose(image, self._scan_history, bands)
|
||||
if (not bands
|
||||
or not (scrolling if scrolling is not None
|
||||
else self.is_currently_scrolling())
|
||||
or self._static_handover):
|
||||
if bands:
|
||||
self._scan_history.clear()
|
||||
return [(image, hold)]
|
||||
if hold > 1:
|
||||
blit = getattr(self, '_last_blit_seconds', 0.0)
|
||||
if blit > _SPLIT_BLIT_FRACTION / max(1.0, self.refresh_hz):
|
||||
self._scan_history.clear()
|
||||
return [(image, hold)]
|
||||
segments = [
|
||||
(scan_order.compose(image, self._scan_history, bands, backs), count)
|
||||
for backs, count in scan_order.refresh_plan(bands, hold)
|
||||
]
|
||||
# A copy: plugins draw into the same image object frame after frame.
|
||||
self._scan_history.appendleft(image.copy())
|
||||
return presented
|
||||
return segments
|
||||
|
||||
def clear(self):
|
||||
"""Clear the display completely."""
|
||||
@@ -1339,6 +1413,8 @@ class DisplayManager:
|
||||
# The stall watchdog would otherwise outlive this manager.
|
||||
if getattr(self, 'frame_timing', None) is not None:
|
||||
self.frame_timing.close()
|
||||
# Installed with the recorder; stop timing collections with it.
|
||||
uninstall_gc_monitor()
|
||||
# Reset the singleton state when cleaning up
|
||||
DisplayManager._instance = None
|
||||
|
||||
@@ -1499,6 +1575,9 @@ class DisplayManager:
|
||||
# A plugin captured for Vegas calls this from its own display();
|
||||
# it must not change the live scroll's state or frame hold.
|
||||
return
|
||||
# A scroll starting or ending also ends a static screen's handover;
|
||||
# see end_scroll_for_static_screen.
|
||||
self._static_handover = False
|
||||
current_time = time.time()
|
||||
# Scrolling callers set this every frame; log transitions only.
|
||||
changed = self._scrolling_state['is_scrolling'] != is_scrolling
|
||||
@@ -1511,6 +1590,47 @@ class DisplayManager:
|
||||
if changed:
|
||||
logger.debug("Scrolling state set to: %s", is_scrolling)
|
||||
|
||||
def end_scroll_for_static_screen(self) -> None:
|
||||
"""Ready the panel for a static screen's first frame after a scroll.
|
||||
|
||||
The display controller calls this just before it dispatches the first
|
||||
frame of a screen that runs its 1 Hz loop, and
|
||||
``set_scrolling_state(False)`` once that dispatch returns. Nothing
|
||||
else ends a scroll at a handover: the state belongs to the screen
|
||||
before, and would only expire 2 s after its last frame.
|
||||
|
||||
Until then, the frames that dispatch presents go out as drawn, not
|
||||
scan-order composed: each as one segment, held for the scroll's hold.
|
||||
With the state still "scrolling", ``_scan_segments`` would take their
|
||||
lagging rows from the frame before: for the first, the scroller's last
|
||||
frame -- the bottom half of the old ticker under the new screen on a
|
||||
96x48 panel. At hold 1 that frame stays up for a whole second; at a
|
||||
longer hold its first refresh flashes the old rows. For a second frame
|
||||
in the same call, the rows would come from the first, shown for as long
|
||||
as the first's would be. Dirty tracking does not keep such a frame up
|
||||
past the screen's next redraw: a frame pushed while the scroll state
|
||||
is set leaves it no digest to match, so that redraw is pushed.
|
||||
|
||||
The rest of that scroll is left on purpose, until the controller ends
|
||||
it:
|
||||
|
||||
* the scroll state, so the gap from the scroller's last frame to this
|
||||
screen's first is still timed by the frame-timing recorder and
|
||||
watched by the stall watchdog, which is where a slow first
|
||||
``display()`` shows up;
|
||||
* its frame hold. On a frame that stays up for a second it only moves
|
||||
the swap to the scroll's next hold boundary, and it is the pacing
|
||||
that gap is due at: judged at hold 1, a handover that kept the
|
||||
scroller's own schedule would count as frames late.
|
||||
|
||||
The next ``set_scrolling_state()`` call, whoever makes it, ends this.
|
||||
One attribute store, so no lock: ``update_display`` reads it once per
|
||||
frame, under its own, and the history is dropped there.
|
||||
"""
|
||||
if self._writes_suppressed():
|
||||
return # a thread drawing off-screen cannot end the live scroll
|
||||
self._static_handover = True
|
||||
|
||||
def is_currently_scrolling(self) -> bool:
|
||||
"""Check if the display is currently in a scrolling state."""
|
||||
current_time = time.time()
|
||||
@@ -1653,11 +1773,14 @@ class DisplayManager:
|
||||
|
||||
Args:
|
||||
frame_checksum: adler32 of the current frame, when the caller has
|
||||
already computed one. Dirty tracking checksums every frame a
|
||||
few lines above the call site, and re-deriving it here meant a
|
||||
second tobytes() plus a second pass over the whole framebuffer
|
||||
on every single frame — ~0.17ms per frame of the two combined
|
||||
at 256x64, paid 100 times a second to reach the same number.
|
||||
already computed one. Dirty tracking checksums every static
|
||||
frame a few lines above the call site, and re-deriving it here
|
||||
meant a second tobytes() plus a second pass over the whole
|
||||
framebuffer on every single frame — ~0.17ms per frame of the
|
||||
two combined at 256x64, paid 100 times a second to reach the
|
||||
same number. None (mid-scroll, dirty tracking off, no
|
||||
hardware): the frame is hashed here, and only when the policy
|
||||
could act on it.
|
||||
"""
|
||||
try:
|
||||
now = time.time()
|
||||
@@ -1668,20 +1791,53 @@ class DisplayManager:
|
||||
self._last_snapshot_ts = 0.0
|
||||
self._viewer_was_fresh = viewer_fresh
|
||||
|
||||
digest = (frame_checksum if frame_checksum is not None
|
||||
else zlib.adler32(self.image.tobytes()))
|
||||
action = snapshot_policy.decide(
|
||||
now, self._last_snapshot_ts, self._last_snapshot_touch_ts,
|
||||
viewer_fresh, digest != self._last_snapshot_digest)
|
||||
if frame_checksum is not None:
|
||||
digest = frame_checksum
|
||||
frame_changed = digest != self._last_snapshot_digest
|
||||
action = snapshot_policy.decide(
|
||||
now, self._last_snapshot_ts, self._last_snapshot_touch_ts,
|
||||
viewer_fresh, frame_changed)
|
||||
else:
|
||||
# Ask as if the frame had changed before paying to find out.
|
||||
# decide() is monotone in frame_changed -- a SKIP for a
|
||||
# changed frame is a SKIP for an unchanged one too (its touch
|
||||
# branch ignores frame_changed) -- so returning here gives the
|
||||
# same answer the hash would have, and on most frames the hash
|
||||
# is never taken. test_snapshot_policy.py holds decide() to it.
|
||||
action = snapshot_policy.decide(
|
||||
now, self._last_snapshot_ts, self._last_snapshot_touch_ts,
|
||||
viewer_fresh, True)
|
||||
if action is snapshot_policy.SnapshotAction.SKIP:
|
||||
# Not hashed, so not known to be unchanged: owed until a
|
||||
# later look finds it written or unchanged.
|
||||
self._snapshot_owed = True
|
||||
return
|
||||
digest = zlib.adler32(self.image.tobytes())
|
||||
frame_changed = digest != self._last_snapshot_digest
|
||||
if not frame_changed:
|
||||
# Unchanged after all: the decision an unchanged frame gets.
|
||||
action = snapshot_policy.decide(
|
||||
now, self._last_snapshot_ts,
|
||||
self._last_snapshot_touch_ts, viewer_fresh, False)
|
||||
if action is snapshot_policy.SnapshotAction.SKIP:
|
||||
# A changed frame inside the write interval stays owed: the
|
||||
# next update_display() would write it, but a static screen
|
||||
# may not make one -- write_owed_snapshot() covers that.
|
||||
self._snapshot_owed = frame_changed
|
||||
return
|
||||
if (action is snapshot_policy.SnapshotAction.TOUCH
|
||||
and self._saved_snapshot_digest == digest):
|
||||
# mtime bump only: keeps the health check (snapshot age)
|
||||
# green without paying for a PNG encode of an unchanged frame
|
||||
# (this frame is already on disk, so nothing is owed).
|
||||
self._snapshot_owed = False
|
||||
os.utime(self._snapshot_path, None)
|
||||
self._last_snapshot_touch_ts = now
|
||||
return
|
||||
# Owed until the write below succeeds: if it raises, the frame
|
||||
# stays owed and write_owed_snapshot() retries it, rather than a
|
||||
# held screen leaving the preview stale after one failed write.
|
||||
self._snapshot_owed = True
|
||||
# (A TOUCH for a frame that isn't on disk yet -- still queued, or
|
||||
# its write failed -- is written instead: touching would make the
|
||||
# older file on disk look current.)
|
||||
@@ -1706,9 +1862,39 @@ class DisplayManager:
|
||||
self._last_snapshot_ts = now
|
||||
self._last_snapshot_touch_ts = now
|
||||
self._last_snapshot_digest = digest
|
||||
self._snapshot_owed = False
|
||||
except Exception as e:
|
||||
self._log_snapshot_failure(e)
|
||||
|
||||
def write_owed_snapshot(self) -> None:
|
||||
"""Write a frame the snapshot throttle skipped, once it is due.
|
||||
|
||||
The preview snapshot is only ever written from update_display(), and
|
||||
at most once per write interval (snapshot_policy). A frame pushed
|
||||
inside that interval is skipped, and is written by the next
|
||||
update_display() that comes after it -- but a screen that draws its
|
||||
card once and then holds it makes no further call. Its frame was on
|
||||
the panel and never in the preview: soccer's recent/upcoming cards
|
||||
skip redundant redraws, and the first one after an on-demand start
|
||||
(pushed a few milliseconds after the controller's clear) left
|
||||
/api/v3/display/current and the web preview black for the whole
|
||||
screen while the panel showed the card.
|
||||
|
||||
The render loop calls this after each frame. Cheap when nothing is
|
||||
owed (one attribute read); otherwise the usual policy decides, so
|
||||
the write still waits out the interval and an unchanged frame is
|
||||
never re-encoded.
|
||||
"""
|
||||
if not self._snapshot_owed:
|
||||
return
|
||||
try:
|
||||
if self._writes_suppressed():
|
||||
return
|
||||
with self._update_lock:
|
||||
self._write_snapshot_if_due()
|
||||
except Exception as e: # pylint: disable=broad-except
|
||||
self._log_snapshot_failure(e)
|
||||
|
||||
def _log_snapshot_failure(self, error: Exception) -> None:
|
||||
# Snapshot failures must never break display — but they must not
|
||||
# be silent either: the snapshot's mtime is the web UI's display
|
||||
@@ -1750,7 +1936,8 @@ class DisplayManager:
|
||||
prefix=f".{snapshot_path_obj.name}.", suffix=".tmp")
|
||||
try:
|
||||
with os.fdopen(_fd, "wb") as _f:
|
||||
image.save(_f, format='PNG')
|
||||
image.save(_f, format='PNG',
|
||||
compress_level=_SNAPSHOT_PNG_COMPRESS_LEVEL)
|
||||
os.chmod(tmp_path, 0o644)
|
||||
os.replace(tmp_path, self._snapshot_path)
|
||||
except Exception:
|
||||
@@ -1761,7 +1948,8 @@ class DisplayManager:
|
||||
except OSError:
|
||||
pass
|
||||
# Fallback to direct save if replace not supported
|
||||
image.save(self._snapshot_path, format='PNG')
|
||||
image.save(self._snapshot_path, format='PNG',
|
||||
compress_level=_SNAPSHOT_PNG_COMPRESS_LEVEL)
|
||||
# Set proper file permissions after saving
|
||||
try:
|
||||
ensure_file_permissions(snapshot_path_obj, get_assets_file_mode())
|
||||
|
||||
@@ -216,6 +216,23 @@ class RenderWatchdog:
|
||||
def armed(self) -> bool:
|
||||
return self._armed
|
||||
|
||||
def liveness(self) -> Dict[str, Any]:
|
||||
"""The heartbeat, in memory: what the control socket's state stream
|
||||
reports as ``loop``.
|
||||
|
||||
``heartbeat_age_seconds`` is the age of the render thread's last beat,
|
||||
the beat that writes the heartbeat file, so it ages at the same rate
|
||||
and is judged by the same ``HEARTBEAT_STALE_SECONDS``. None until the
|
||||
loop has drawn its first frame. Any thread may call this: it only
|
||||
reads two attributes.
|
||||
"""
|
||||
last = self._last_beat
|
||||
age = None
|
||||
if self._armed and last is not None:
|
||||
age = max(self._clock() - last, 0.0)
|
||||
return {'heartbeat_age_seconds': age, 'armed': self._armed,
|
||||
'stale_after': HEARTBEAT_STALE_SECONDS}
|
||||
|
||||
def _on_render_thread(self) -> bool:
|
||||
return self._render_thread is not None and threading.get_ident() == self._render_thread
|
||||
|
||||
|
||||
@@ -23,6 +23,7 @@ import requests
|
||||
from typing import Any, Dict, List
|
||||
|
||||
from src.common.api_helper import DEFAULT_HTTP_HEADERS
|
||||
from src.common.json_body import response_json
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -157,7 +158,7 @@ class DynamicTeamResolver:
|
||||
response = requests.get(rankings_url, headers=dict(DEFAULT_HTTP_HEADERS),
|
||||
timeout=self.request_timeout)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
data = response_json(response)
|
||||
|
||||
rankings = {}
|
||||
rankings_data = data.get('rankings', [])
|
||||
|
||||
+126
-32
@@ -485,15 +485,18 @@ def record_error(
|
||||
# and only the display service's ever records anything (plugin_executor runs
|
||||
# the plugins there). The web interface therefore reads a snapshot the display
|
||||
# service publishes to the shared cache directory -- the same channel, and the
|
||||
# same file permissions, as display_current_state and plugin_metrics:*: files
|
||||
# same file permissions, as display_current_state and plugin_metrics_snapshot: files
|
||||
# are 0660 and carry the cache directory's group, so root writes and the web
|
||||
# user reads, and the other way round for the clear request.
|
||||
#
|
||||
# ERROR_SNAPSHOT_KEY written by the display service only
|
||||
# ERROR_CLEAR_REQUEST_KEY written by the web interface only
|
||||
# ERROR_CLEAR_REQUEST_KEY written by the web interface only, as a fallback
|
||||
#
|
||||
# A clear is asynchronous: the web interface records a request, and the
|
||||
# display service applies it (clear_before) on its next tick and republishes.
|
||||
# A clear goes over the control socket (``errors.clear``): the display applies
|
||||
# it (clear_before) and republishes the snapshot before it answers. Only when
|
||||
# the socket cannot carry it (no socket, or a display older than the command)
|
||||
# does the web interface record a request in the mailbox, which the display
|
||||
# applies on its next tick; its tick reads that file only when it changed.
|
||||
# Until it has, the web interface hides whatever the snapshot shows from
|
||||
# before the cutoff, so a clear takes effect for readers immediately and a
|
||||
# snapshot published just before the request cannot bring old errors back.
|
||||
@@ -598,13 +601,43 @@ class ErrorSnapshotPublisher:
|
||||
self._published_version: Optional[int] = None
|
||||
self._last_attempt: Optional[float] = None
|
||||
self._applied_clear_id: Optional[str] = None
|
||||
# The widest cutoff applied in this process: a clear request at or
|
||||
# before it has nothing left to clear (see _pending_cutoff).
|
||||
self._applied_clear_cutoff: Optional[float] = None
|
||||
from src.cache_manager import MailboxWatch # the display's cache, loaded already
|
||||
self._mailbox = MailboxWatch(ERROR_CLEAR_REQUEST_KEY)
|
||||
self._tick_lock = threading.Lock()
|
||||
self._stop = threading.Event()
|
||||
self._thread: Optional[threading.Thread] = None
|
||||
|
||||
def _clear(self, request_id: str, cutoff: float) -> int:
|
||||
"""Apply one clear and remember it. Caller holds _tick_lock."""
|
||||
cleared = 0
|
||||
if math.isfinite(cutoff):
|
||||
cleared = self.aggregator.clear_before(datetime.fromtimestamp(cutoff))
|
||||
_snapshot_logger.info("Cleared %d plugin error record(s) as requested (%s)",
|
||||
cleared, request_id)
|
||||
if self._applied_clear_cutoff is None or cutoff > self._applied_clear_cutoff:
|
||||
self._applied_clear_cutoff = cutoff
|
||||
# A malformed request is acknowledged too, so it is not retried forever.
|
||||
self._applied_clear_id = request_id
|
||||
return cleared
|
||||
|
||||
def _apply_clear_request(self) -> bool:
|
||||
"""Honour a clear request we have not applied yet. True if one was."""
|
||||
request = self.cache_manager.get(ERROR_CLEAR_REQUEST_KEY, max_age=None, memory_ttl=0)
|
||||
"""Honour a mailbox clear request we have not applied yet. True if one was.
|
||||
|
||||
The mailbox is the fallback for a web interface that could not use
|
||||
the control socket (``errors.clear``, :meth:`clear_now`). It is read
|
||||
only when its file changed since the last tick; otherwise a tick
|
||||
costs one stat().
|
||||
"""
|
||||
if not self._mailbox.changed(self.cache_manager):
|
||||
return False
|
||||
try:
|
||||
request = self.cache_manager.get(ERROR_CLEAR_REQUEST_KEY, max_age=None, memory_ttl=0)
|
||||
except Exception:
|
||||
self._mailbox.forget()
|
||||
raise
|
||||
if not isinstance(request, dict):
|
||||
return False
|
||||
request_id = request.get("request_id")
|
||||
@@ -614,14 +647,30 @@ class ErrorSnapshotPublisher:
|
||||
cutoff = float(request.get("cutoff"))
|
||||
except (TypeError, ValueError):
|
||||
cutoff = float("nan")
|
||||
if math.isfinite(cutoff):
|
||||
cleared = self.aggregator.clear_before(datetime.fromtimestamp(cutoff))
|
||||
_snapshot_logger.info("Cleared %d plugin error record(s) as requested (%s)",
|
||||
cleared, request_id)
|
||||
# A malformed request is acknowledged too, so it is not retried forever.
|
||||
self._applied_clear_id = request_id
|
||||
self._clear(request_id, cutoff)
|
||||
return True
|
||||
|
||||
def clear_now(self, request_id: str, cutoff: float) -> int:
|
||||
"""``errors.clear`` over the control socket: apply a clear at once and
|
||||
republish the snapshot, so the web interface's next read has it.
|
||||
Returns how many records were cleared. Raises when the snapshot
|
||||
could not be written, so the caller is not told it worked."""
|
||||
with self._tick_lock:
|
||||
cleared = self._clear(request_id, float(cutoff))
|
||||
self._publish(self.aggregator.version, self._clock())
|
||||
return cleared
|
||||
|
||||
def _publish(self, version: int, now: float) -> None:
|
||||
"""Write the snapshot. Caller holds _tick_lock."""
|
||||
# Stamp the attempt before writing: a cache that keeps failing
|
||||
# is retried at the throttled rate, not on every tick.
|
||||
self._last_attempt = now
|
||||
snapshot = self.aggregator.build_snapshot()
|
||||
snapshot["applied_clear_id"] = self._applied_clear_id
|
||||
snapshot["applied_clear_cutoff"] = self._applied_clear_cutoff
|
||||
self.cache_manager.set(ERROR_SNAPSHOT_KEY, snapshot)
|
||||
self._published_version = version
|
||||
|
||||
def tick(self) -> bool:
|
||||
"""Apply a pending clear and publish if due. True if a snapshot was written."""
|
||||
with self._tick_lock:
|
||||
@@ -635,13 +684,7 @@ class ErrorSnapshotPublisher:
|
||||
if (self._last_attempt is not None
|
||||
and now - self._last_attempt < self.min_interval):
|
||||
return False
|
||||
# Stamp the attempt before writing: a cache that keeps failing
|
||||
# is retried at the throttled rate, not on every tick.
|
||||
self._last_attempt = now
|
||||
snapshot = self.aggregator.build_snapshot()
|
||||
snapshot["applied_clear_id"] = self._applied_clear_id
|
||||
self.cache_manager.set(ERROR_SNAPSHOT_KEY, snapshot)
|
||||
self._published_version = version
|
||||
self._publish(version, now)
|
||||
return True
|
||||
except Exception as err: # never let reporting break the display
|
||||
_snapshot_logger.debug("Could not publish the plugin error snapshot: %s",
|
||||
@@ -693,6 +736,22 @@ def start_error_snapshot_publisher(cache_manager: Any) -> Optional[ErrorSnapshot
|
||||
return None
|
||||
|
||||
|
||||
def apply_error_clear(request_id: str, args: Any) -> Dict[str, Any]:
|
||||
"""The display's handler for ``errors.clear`` on the control socket.
|
||||
|
||||
``args`` is the contract's ErrorsClearArgs (``cutoff``, epoch seconds).
|
||||
Runs on the socket's connection thread: the aggregator and the publisher
|
||||
have their own locks, and nothing here touches rendering. Returns
|
||||
ErrorsClearResult once the clear is applied and the snapshot rewritten.
|
||||
"""
|
||||
publisher = _snapshot_publisher
|
||||
if publisher is None:
|
||||
raise RuntimeError("the error snapshot publisher is not running")
|
||||
cutoff = float(args.cutoff)
|
||||
cleared = publisher.clear_now(request_id, cutoff)
|
||||
return {"request_id": request_id, "cutoff": cutoff, "cleared": cleared}
|
||||
|
||||
|
||||
# --- Reading side (web interface) -------------------------------------------
|
||||
|
||||
def read_error_report(cache_manager: Any) -> Tuple[Optional[Dict[str, Any]], Optional[Dict[str, Any]]]:
|
||||
@@ -731,7 +790,15 @@ def _pending_cutoff(snapshot: Optional[Dict[str, Any]],
|
||||
cutoff = float(clear_request.get("cutoff"))
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
return cutoff if math.isfinite(cutoff) else None
|
||||
if not math.isfinite(cutoff):
|
||||
return None
|
||||
# A wider clear has been applied since (over the control socket): this
|
||||
# older request has nothing left to hide.
|
||||
applied = snapshot.get("applied_clear_cutoff") if snapshot is not None else None
|
||||
if (isinstance(applied, (int, float)) and not isinstance(applied, bool)
|
||||
and applied >= cutoff):
|
||||
return None
|
||||
return cutoff
|
||||
|
||||
|
||||
def _is_after(item: Any, field_name: str, cutoff: float) -> bool:
|
||||
@@ -831,13 +898,31 @@ def _count_cleared(summary: Dict[str, Any], cutoff: float) -> Optional[int]:
|
||||
return None
|
||||
|
||||
|
||||
def request_error_clear(cache_manager: Any, cutoff: float) -> Dict[str, Any]:
|
||||
#: ``send(request_id, cutoff)`` hands a clear to the display over the control
|
||||
#: socket and returns its ErrorsClearResult, or None when the socket could
|
||||
#: not carry it and the mailbox should be written instead. Any exception it
|
||||
#: raises reaches the caller: the display had the request and failed it.
|
||||
ClearSender = Callable[[str, float], Optional[Dict[str, Any]]]
|
||||
|
||||
|
||||
def request_error_clear(cache_manager: Any, cutoff: float,
|
||||
send: Optional[ClearSender] = None) -> Dict[str, Any]:
|
||||
"""Ask the display service to forget errors recorded at or before ``cutoff``.
|
||||
|
||||
Returns ``request_id``, ``cutoff`` (ISO, local time), ``cleared_count``
|
||||
(see _count_cleared) and ``clear_requested``. Raises OSError when the
|
||||
request did not reach the shared cache, since a cache without a usable
|
||||
directory accepts set() and keeps nothing.
|
||||
Over the control socket when ``send`` is given and carries it: the
|
||||
display applies the clear and republishes its snapshot before it
|
||||
answers, so nothing is written here. Otherwise (no socket, or a display
|
||||
older than ``errors.clear``) a request is written to the
|
||||
``plugin_error_clear_request`` mailbox, which the display applies on
|
||||
its next tick, and readers hide the cleared errors until then.
|
||||
|
||||
Returns ``request_id``, ``cutoff`` (ISO, local time), ``cleared_count``,
|
||||
``clear_requested``, ``applied`` (the display has already cleared them)
|
||||
and ``transport`` (``socket`` or ``mailbox``). ``cleared_count`` is the
|
||||
display's own count over the socket, else an estimate from the snapshot
|
||||
(see _count_cleared). Raises OSError when a mailbox request did not reach
|
||||
the shared cache, since a cache without a usable directory accepts set()
|
||||
and keeps nothing.
|
||||
|
||||
A request the display has not applied yet is only ever widened: a later,
|
||||
narrower one ("older than 24 hours" after "everything") overwriting it
|
||||
@@ -848,8 +933,21 @@ def request_error_clear(cache_manager: Any, cutoff: float) -> Dict[str, Any]:
|
||||
if pending is not None:
|
||||
cutoff = max(cutoff, pending)
|
||||
before = error_summary_from_report(snapshot, clear_request)
|
||||
request_id = uuid.uuid4().hex
|
||||
answer = {
|
||||
"clear_requested": True,
|
||||
"request_id": request_id,
|
||||
"cutoff": datetime.fromtimestamp(cutoff).isoformat(),
|
||||
}
|
||||
if send is not None:
|
||||
result = send(request_id, cutoff)
|
||||
if result is not None:
|
||||
count = result.get("cleared")
|
||||
return dict(answer, applied=True, transport="socket",
|
||||
cleared_count=count if isinstance(count, int) and not isinstance(count, bool)
|
||||
else _count_cleared(before, cutoff))
|
||||
request = {
|
||||
"request_id": uuid.uuid4().hex,
|
||||
"request_id": request_id,
|
||||
"cutoff": cutoff,
|
||||
"requested_at": time.time(),
|
||||
}
|
||||
@@ -857,9 +955,5 @@ def request_error_clear(cache_manager: Any, cutoff: float) -> Dict[str, Any]:
|
||||
stored = cache_manager.get(ERROR_CLEAR_REQUEST_KEY, max_age=None, memory_ttl=0)
|
||||
if not isinstance(stored, dict) or stored.get("request_id") != request["request_id"]:
|
||||
raise OSError("the clear request was not stored in the shared cache")
|
||||
return {
|
||||
"cleared_count": _count_cleared(before, cutoff),
|
||||
"clear_requested": True,
|
||||
"request_id": request["request_id"],
|
||||
"cutoff": datetime.fromtimestamp(cutoff).isoformat(),
|
||||
}
|
||||
return dict(answer, applied=False, transport="mailbox",
|
||||
cleared_count=_count_cleared(before, cutoff))
|
||||
|
||||
@@ -222,6 +222,39 @@ class FontManager:
|
||||
logger.error(f"Error registering fonts for plugin {plugin_id}: {e}", exc_info=True)
|
||||
return False
|
||||
|
||||
def forget_plugin_fonts(self, plugin_id: str) -> bool:
|
||||
"""Drop the fonts ``plugin_id``'s manifest registered: its manifest
|
||||
and catalog, its ``plugin_id::family`` entries in font_catalog, and
|
||||
cached font objects for those families. Called by core when a plugin
|
||||
is unloaded, so a reload registers from its current manifest and a
|
||||
removed plugin's fonts stop resolving.
|
||||
|
||||
FontManager takes no locks; like forget_manager_fonts this relies on
|
||||
single dict operations being atomic and iterates snapshots, so a
|
||||
render thread calling get_font() meanwhile cannot break it. Returns
|
||||
True if the plugin had registered fonts.
|
||||
"""
|
||||
prefix = f"{plugin_id}::"
|
||||
manifest = self.plugin_fonts.pop(plugin_id, None)
|
||||
catalog = self.plugin_font_catalogs.pop(plugin_id, None)
|
||||
# Every namespaced entry, not just the families in the catalog: one
|
||||
# whose file failed to load never made it into the catalog, and a
|
||||
# caller may have added one directly.
|
||||
for family in list(self.font_catalog):
|
||||
if family.startswith(prefix):
|
||||
self.font_catalog.pop(family, None)
|
||||
# get_font() keys the cache f"{family}_{size_px}".
|
||||
dropped = [key for key in list(self.font_cache) if key.startswith(prefix)]
|
||||
for key in dropped:
|
||||
self.font_cache.pop(key, None)
|
||||
if dropped:
|
||||
# Font objects someone may hold were dropped; see cache_generation.
|
||||
self.cache_generation += 1
|
||||
if manifest is None and catalog is None:
|
||||
return False
|
||||
logger.info("Forgot fonts of plugin %s", plugin_id)
|
||||
return True
|
||||
|
||||
def _validate_font_manifest(self, font_manifest: Dict[str, Any]) -> bool:
|
||||
"""Validate the structure of a plugin's font manifest."""
|
||||
required_fields = ["fonts"]
|
||||
|
||||
+368
-11
@@ -3,26 +3,37 @@
|
||||
Every failure -- no socket (the display is stopped, or predates the socket),
|
||||
a refused or timed-out connection, a reply that breaks the contract, or an
|
||||
error the display returned -- raises :class:`ControlError` with a short
|
||||
``reason``, and the caller falls back to the file mailbox. Nothing here
|
||||
blocks for longer than ``timeout`` in total.
|
||||
``reason``. Nothing here blocks for longer than ``timeout`` in total.
|
||||
|
||||
Whether the caller may then write the file mailbox instead is
|
||||
:func:`should_fall_back`: only when the display never took the request (it
|
||||
could not be reached, or it is too old to know the command). A display that
|
||||
took the request and then failed, refused or went quiet is answered as
|
||||
that, not posted a second time through the mailbox.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import socket
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
from typing import Any, Dict, List, Mapping, Optional, Sequence
|
||||
from typing import Any, Callable, Dict, List, Mapping, Optional, Sequence
|
||||
|
||||
from src.ipc.contract import (
|
||||
AWAIT_SECONDS,
|
||||
MAX_MESSAGE_BYTES,
|
||||
PROTOCOL_VERSION,
|
||||
SUBSCRIBE_KEEPALIVE_SECONDS,
|
||||
SUPPORTED_VERSIONS,
|
||||
Command,
|
||||
ErrorCode,
|
||||
FrameReader,
|
||||
ProtocolError,
|
||||
Request,
|
||||
Response,
|
||||
StateEvent,
|
||||
StateEventKind,
|
||||
client_socket_paths,
|
||||
decode_message,
|
||||
encode_message,
|
||||
@@ -44,17 +55,49 @@ class ControlError(Exception):
|
||||
``refused``, ``timeout``, ``closed``, ``bad_response``, ``invalid_request``.
|
||||
When the display answered with an error, ``reason`` is that error's
|
||||
:class:`~src.ipc.contract.ErrorCode` (``busy``, ``unknown_command``, ...).
|
||||
|
||||
``sent`` is True once the whole request was written to a connected
|
||||
display, which may then have acted on it. A refusal the display sends
|
||||
before it reads anything (``forbidden``, too many connections) carries
|
||||
no request id and leaves ``sent`` False.
|
||||
"""
|
||||
|
||||
def __init__(self, reason: str, message: str = ''):
|
||||
def __init__(self, reason: str, message: str = '', *, sent: bool = False):
|
||||
super().__init__(reason, message)
|
||||
self.reason = reason
|
||||
self.message = message
|
||||
self.sent = sent
|
||||
|
||||
def __str__(self) -> str:
|
||||
return f'{self.reason}: {self.message}' if self.message else self.reason
|
||||
|
||||
|
||||
#: Answers from a display that read the request but does not speak it: one
|
||||
#: older than the command (an upgrade in progress) or the protocol version.
|
||||
#: It did nothing, so the mailbox is the way to reach it.
|
||||
UPGRADE_REASONS = frozenset({ErrorCode.UNKNOWN_COMMAND, ErrorCode.UNSUPPORTED_VERSION})
|
||||
|
||||
|
||||
def should_fall_back(error: BaseException) -> bool:
|
||||
"""May the caller write the file mailbox after ``error``?
|
||||
|
||||
Yes when the display never took the request: there is no socket (the
|
||||
display is stopped, predates the socket, or it is switched off), the
|
||||
connection was refused or timed out, the display turned the connection
|
||||
away before reading it, or it is too old to know the command
|
||||
(:data:`UPGRADE_REASONS`). Also for an error that is not a
|
||||
:class:`ControlError` (a bug in the client), as before.
|
||||
|
||||
No once the display had the request: a ``busy`` queue, ``invalid_args``,
|
||||
an ``internal`` error, or a timeout or hang-up after the request was
|
||||
sent. The display may have applied it, or would refuse it from the
|
||||
mailbox too, so a second copy there only hides the failure.
|
||||
"""
|
||||
if not isinstance(error, ControlError):
|
||||
return True
|
||||
return not error.sent or error.reason in UPGRADE_REASONS
|
||||
|
||||
|
||||
def request(cmd: str, args: Optional[Mapping[str, Any]] = None, *,
|
||||
request_id: Optional[str] = None,
|
||||
timeout: float = DEFAULT_TIMEOUT_SECONDS,
|
||||
@@ -88,11 +131,14 @@ def request(cmd: str, args: Optional[Mapping[str, Any]] = None, *,
|
||||
# A refusal before the request was read (forbidden, too many
|
||||
# connections) carries no id.
|
||||
if response.id != request_id and not (response.id is None and not response.ok):
|
||||
raise ControlError('bad_response', 'the reply is for a different request')
|
||||
raise ControlError('bad_response', 'the reply is for a different request', sent=True)
|
||||
if not response.ok:
|
||||
error = response.error
|
||||
# No id: refused at the door (forbidden, too many connections),
|
||||
# before the display read the request.
|
||||
raise ControlError(error.code if error else 'bad_response',
|
||||
error.message if error else '')
|
||||
error.message if error else '',
|
||||
sent=response.id is not None)
|
||||
return dict(response.result or {})
|
||||
|
||||
|
||||
@@ -137,26 +183,31 @@ def _connect(paths: Sequence[str], deadline: float) -> socket.socket:
|
||||
|
||||
|
||||
def _exchange(sock: socket.socket, payload: bytes, deadline: float) -> Response:
|
||||
"""Send ``payload`` and read the reply. A failure once the whole request
|
||||
is written raises with ``sent=True``: the display may have it."""
|
||||
sent = False
|
||||
try:
|
||||
sock.settimeout(_remaining(deadline))
|
||||
sock.sendall(payload)
|
||||
sent = True
|
||||
reader = FrameReader(MAX_MESSAGE_BYTES)
|
||||
while True:
|
||||
sock.settimeout(_remaining(deadline))
|
||||
data = sock.recv(4096)
|
||||
if not data:
|
||||
raise ControlError('closed', 'the display closed the connection')
|
||||
raise ControlError('closed', 'the display closed the connection', sent=sent)
|
||||
lines = reader.feed(data)
|
||||
if lines:
|
||||
return Response.from_dict(decode_message(lines[0]))
|
||||
except socket.timeout:
|
||||
raise ControlError('timeout', 'no reply in time') from None
|
||||
raise ControlError('timeout', 'no reply in time', sent=sent) from None
|
||||
except ProtocolError as e:
|
||||
raise ControlError('bad_response', e.message) from None
|
||||
except ControlError:
|
||||
raise ControlError('bad_response', e.message, sent=sent) from None
|
||||
except ControlError as e:
|
||||
e.sent = e.sent or sent
|
||||
raise
|
||||
except OSError as e:
|
||||
raise ControlError('closed', str(e)) from None
|
||||
raise ControlError('closed', str(e), sent=sent) from None
|
||||
|
||||
|
||||
# -- commands ---------------------------------------------------------------------------
|
||||
@@ -188,6 +239,55 @@ def on_demand_status(*, timeout: float = DEFAULT_TIMEOUT_SECONDS,
|
||||
return request(Command.ON_DEMAND_STATUS, {}, timeout=timeout, paths=paths)
|
||||
|
||||
|
||||
#: Headroom over the display's own wait for an awaited command, so its
|
||||
#: ``pending`` answer arrives before the client gives up.
|
||||
_AWAIT_MARGIN_SECONDS = 1.0
|
||||
|
||||
|
||||
def _awaited_timeout(cmd: str) -> float:
|
||||
return AWAIT_SECONDS[cmd] + _AWAIT_MARGIN_SECONDS
|
||||
|
||||
|
||||
def brightness_set(brightness: int, *, timeout: Optional[float] = None,
|
||||
paths: Optional[Sequence[str]] = None) -> Dict[str, Any]:
|
||||
"""Set the panel's normal brightness now (transient: config.json is not
|
||||
written). Returns the applied :class:`~src.ipc.contract.BrightnessResult`;
|
||||
raises :class:`ControlError`.
|
||||
"""
|
||||
return request(Command.BRIGHTNESS_SET, {'brightness': brightness},
|
||||
timeout=_awaited_timeout(Command.BRIGHTNESS_SET) if timeout is None
|
||||
else timeout, paths=paths)
|
||||
|
||||
|
||||
def plugin_reload(plugin_id: str, *, timeout: Optional[float] = None,
|
||||
paths: Optional[Sequence[str]] = None) -> Dict[str, Any]:
|
||||
"""Have the display reload a running plugin from disk.
|
||||
|
||||
Returns :class:`~src.ipc.contract.PluginReloadResult` once the new code is
|
||||
running. Raises :class:`ControlError`: ``not_loaded`` (not running it),
|
||||
``failed`` (the new version did not load), ``pending`` (not done in
|
||||
time; it will still happen), or a transport reason.
|
||||
"""
|
||||
return request(Command.PLUGIN_RELOAD, {'plugin_id': plugin_id},
|
||||
timeout=_awaited_timeout(Command.PLUGIN_RELOAD) if timeout is None
|
||||
else timeout, paths=paths)
|
||||
|
||||
|
||||
def errors_clear(request_id: str, cutoff: float, *,
|
||||
timeout: float = DEFAULT_TIMEOUT_SECONDS,
|
||||
paths: Optional[Sequence[str]] = None) -> Dict[str, Any]:
|
||||
"""Have the display forget the plugin errors recorded at or before
|
||||
``cutoff`` (epoch seconds) and publish its error snapshot again.
|
||||
|
||||
Returns :class:`~src.ipc.contract.ErrorsClearResult` once it is done.
|
||||
Raises :class:`ControlError`: ``unknown_command`` from a display older
|
||||
than the command, which still reads the ``plugin_error_clear_request``
|
||||
mailbox.
|
||||
"""
|
||||
return request(Command.ERRORS_CLEAR, {'cutoff': cutoff}, request_id=request_id,
|
||||
timeout=timeout, paths=paths)
|
||||
|
||||
|
||||
def ping(*, timeout: float = DEFAULT_TIMEOUT_SECONDS,
|
||||
paths: Optional[Sequence[str]] = None) -> Dict[str, Any]:
|
||||
return request(Command.PING, {}, timeout=timeout, paths=paths)
|
||||
@@ -198,3 +298,260 @@ def hello(client: str = 'web', *, timeout: float = DEFAULT_TIMEOUT_SECONDS,
|
||||
"""Version negotiation: the result's ``version`` is the one both sides speak."""
|
||||
return request(Command.HELLO, {'versions': list(SUPPORTED_VERSIONS), 'client': client},
|
||||
timeout=timeout, paths=paths)
|
||||
|
||||
|
||||
# -- the state stream (stage 3) ---------------------------------------------------------
|
||||
|
||||
def state_get(since: Optional[int] = None, epoch: Optional[str] = None, *,
|
||||
timeout: float = DEFAULT_TIMEOUT_SECONDS,
|
||||
paths: Optional[Sequence[str]] = None) -> Dict[str, Any]:
|
||||
"""The display's state now, as a :class:`~src.ipc.contract.StateSnapshot`.
|
||||
|
||||
With ``since``/``epoch`` from an earlier answer, an unchanged state comes
|
||||
back in the short ``changed: false`` form. Raises :class:`ControlError`
|
||||
(``unknown_command`` from a display older than stage 3).
|
||||
"""
|
||||
args: Dict[str, Any] = {}
|
||||
if since is not None:
|
||||
args['since'] = since
|
||||
if epoch is not None:
|
||||
args['epoch'] = epoch
|
||||
return request(Command.STATE_GET, args, timeout=timeout, paths=paths)
|
||||
|
||||
|
||||
def snapshot_age(snapshot: Mapping[str, Any], now_mono: Optional[float] = None) -> float:
|
||||
"""Seconds since ``snapshot`` arrived: ``received_mono`` (set by
|
||||
:meth:`StateSubscription.latest`) to now; 0 for a one-shot answer."""
|
||||
received = snapshot.get('received_mono')
|
||||
if isinstance(received, (int, float)) and not isinstance(received, bool):
|
||||
now_mono = time.monotonic() if now_mono is None else now_mono
|
||||
return max(now_mono - float(received), 0.0)
|
||||
return 0.0
|
||||
|
||||
|
||||
def snapshot_loop_age(snapshot: Mapping[str, Any],
|
||||
now_mono: Optional[float] = None) -> Optional[float]:
|
||||
"""The render loop's heartbeat age now, from a state snapshot: the age the
|
||||
display measured when it answered, plus the time since the answer
|
||||
arrived. None when the display has no beat to report yet."""
|
||||
loop = snapshot.get('loop')
|
||||
if not isinstance(loop, dict):
|
||||
state = snapshot.get('state')
|
||||
loop = state.get('loop') if isinstance(state, dict) else None
|
||||
age = loop.get('heartbeat_age_seconds') if isinstance(loop, dict) else None
|
||||
if not isinstance(age, (int, float)) or isinstance(age, bool):
|
||||
return None
|
||||
return max(float(age), 0.0) + snapshot_age(snapshot, now_mono)
|
||||
|
||||
|
||||
def _merge_volatile(state: Dict[str, Any], volatile: Any) -> None:
|
||||
"""Fold a tick's ``volatile`` values (``{section: {key: value}}``) into
|
||||
``state``, copying each section it touches.
|
||||
|
||||
These are the timestamps the hub leaves out of its version --
|
||||
``display.last_updated``, ``on_demand.last_updated``/``remaining``,
|
||||
``plugins.published_at`` -- and the readers judge freshness by them, so
|
||||
a copy that only full ``state`` events updated would go stale while the
|
||||
same mode stayed on screen. Only keys the section already has are taken:
|
||||
a tick never adds a section or a key the last snapshot did not carry
|
||||
(a section left out of a truncated snapshot stays out).
|
||||
"""
|
||||
if not isinstance(volatile, dict):
|
||||
return # a display from before ticks carried them
|
||||
for name, values in volatile.items():
|
||||
section = state.get(name)
|
||||
if not isinstance(section, dict) or not isinstance(values, dict):
|
||||
continue
|
||||
fresh = {k: v for k, v in values.items() if k in section}
|
||||
if fresh:
|
||||
state[name] = dict(section, **fresh)
|
||||
|
||||
|
||||
#: A subscription that has heard nothing for this long is not trusted: the
|
||||
#: display sends a tick at least every SUBSCRIBE_KEEPALIVE_SECONDS.
|
||||
SUBSCRIPTION_SILENCE_SECONDS = 3 * SUBSCRIBE_KEEPALIVE_SECONDS
|
||||
|
||||
#: Reconnect backoff: the first retry, and the cap. A display that does not
|
||||
#: know state.subscribe (stage 2 or older) is retried at the cap.
|
||||
_RECONNECT_MIN_SECONDS = 1.0
|
||||
_RECONNECT_MAX_SECONDS = 30.0
|
||||
|
||||
#: Failures that another try soon will not fix.
|
||||
_SLOW_RETRY_REASONS = frozenset({'unknown_command', 'unsupported_version', 'disabled',
|
||||
'unsupported'})
|
||||
|
||||
|
||||
class StateSubscription:
|
||||
"""One ``state.subscribe`` connection, held on a daemon thread.
|
||||
|
||||
Keeps the latest snapshot the display pushed, so a reader answers from
|
||||
memory (:meth:`latest`). Reconnects with a backoff when the display goes
|
||||
away. Never raises into the caller: :meth:`latest` is None whenever the
|
||||
copy cannot be vouched for (not connected, or silent for longer than
|
||||
``silence``), and the caller falls back.
|
||||
"""
|
||||
|
||||
def __init__(self, paths: Optional[Sequence[str]] = None, *,
|
||||
silence: float = SUBSCRIPTION_SILENCE_SECONDS,
|
||||
connect_timeout: float = DEFAULT_TIMEOUT_SECONDS,
|
||||
clock: Callable[[], float] = time.monotonic):
|
||||
self._paths = list(paths) if paths is not None else None
|
||||
self._silence = silence
|
||||
self._connect_timeout = connect_timeout
|
||||
self._clock = clock
|
||||
self._lock = threading.Lock()
|
||||
self._snapshot: Optional[Dict[str, Any]] = None
|
||||
self._received: Optional[float] = None
|
||||
self._connected = False
|
||||
self._stop = threading.Event()
|
||||
self._sock: Optional[socket.socket] = None
|
||||
self._thread: Optional[threading.Thread] = None
|
||||
#: The reason the last connection ended (a ControlError reason).
|
||||
self.last_error: Optional[str] = None
|
||||
#: Full snapshots received: the subscribe answer and each state event.
|
||||
self.snapshots = 0
|
||||
|
||||
# -- the reader's side ---------------------------------------------------
|
||||
|
||||
@property
|
||||
def connected(self) -> bool:
|
||||
return self._connected
|
||||
|
||||
def latest(self) -> Optional[Dict[str, Any]]:
|
||||
"""A copy of the latest snapshot, with ``received_mono`` (this
|
||||
process's monotonic clock when it arrived); None when not trusted."""
|
||||
with self._lock:
|
||||
if not self._connected or self._snapshot is None or self._received is None:
|
||||
return None
|
||||
if self._clock() - self._received > self._silence:
|
||||
return None
|
||||
snap = dict(self._snapshot)
|
||||
snap['received_mono'] = self._received
|
||||
return snap
|
||||
|
||||
# -- lifecycle -----------------------------------------------------------
|
||||
|
||||
def start(self) -> 'StateSubscription':
|
||||
if self._thread is None or not self._thread.is_alive():
|
||||
self._stop.clear()
|
||||
self._thread = threading.Thread(target=self._run, name='ledmatrix-state-feed',
|
||||
daemon=True)
|
||||
self._thread.start()
|
||||
return self
|
||||
|
||||
def stop(self, timeout: float = 2.0) -> None:
|
||||
self._stop.set()
|
||||
sock = self._sock
|
||||
if sock is not None:
|
||||
try:
|
||||
sock.shutdown(socket.SHUT_RDWR)
|
||||
except OSError:
|
||||
pass
|
||||
thread = self._thread
|
||||
if thread is not None and thread is not threading.current_thread():
|
||||
thread.join(timeout)
|
||||
self._thread = None
|
||||
|
||||
# -- the feed thread -----------------------------------------------------
|
||||
|
||||
def _run(self) -> None:
|
||||
backoff = _RECONNECT_MIN_SECONDS
|
||||
while not self._stop.is_set():
|
||||
snapshots = self.snapshots
|
||||
try:
|
||||
self._follow()
|
||||
except ControlError as e:
|
||||
self.last_error = e.reason
|
||||
if e.reason in _SLOW_RETRY_REASONS:
|
||||
backoff = _RECONNECT_MAX_SECONDS
|
||||
except Exception as e: # pylint: disable=broad-except
|
||||
self.last_error = type(e).__name__
|
||||
finally:
|
||||
with self._lock:
|
||||
self._connected = False
|
||||
sock, self._sock = self._sock, None
|
||||
if sock is not None:
|
||||
try:
|
||||
sock.close()
|
||||
except OSError:
|
||||
pass
|
||||
if self.snapshots != snapshots:
|
||||
# This connection got as far as the display's state: whatever
|
||||
# ended it (a restart, most often), it was working, so the
|
||||
# next try starts from the shortest wait again.
|
||||
backoff = _RECONNECT_MIN_SECONDS
|
||||
if self._stop.wait(backoff):
|
||||
return
|
||||
backoff = min(backoff * 2, _RECONNECT_MAX_SECONDS)
|
||||
|
||||
def _follow(self) -> None:
|
||||
"""Subscribe, then read events until the connection ends. Raises ControlError."""
|
||||
if not socket_supported():
|
||||
raise ControlError('unsupported', 'no Unix sockets on this platform')
|
||||
candidates = list(self._paths) if self._paths is not None else client_socket_paths()
|
||||
if not candidates:
|
||||
raise ControlError('disabled', 'the control socket is turned off')
|
||||
request_id = str(uuid.uuid4())
|
||||
payload = encode_message(Request(id=request_id, cmd=Command.STATE_SUBSCRIBE,
|
||||
args={}).to_dict())
|
||||
sock = _connect(candidates, time.monotonic() + self._connect_timeout)
|
||||
self._sock = sock
|
||||
try:
|
||||
sock.settimeout(self._connect_timeout)
|
||||
sock.sendall(payload)
|
||||
# A read waits for the next event; the display sends one at least
|
||||
# every keepalive, so this much silence means it is gone.
|
||||
sock.settimeout(self._silence)
|
||||
reader = FrameReader(MAX_MESSAGE_BYTES)
|
||||
first = True
|
||||
while not self._stop.is_set():
|
||||
data = sock.recv(65536)
|
||||
if not data:
|
||||
raise ControlError('closed', 'the display closed the connection')
|
||||
for line in reader.feed(data):
|
||||
obj = decode_message(line)
|
||||
if first:
|
||||
response = Response.from_dict(obj)
|
||||
if not response.ok:
|
||||
error = response.error
|
||||
raise ControlError(error.code if error else 'bad_response',
|
||||
error.message if error else '')
|
||||
self._store(dict(response.result or {}), full=True)
|
||||
first = False
|
||||
continue
|
||||
event = StateEvent.from_dict(obj)
|
||||
self._store(event.result, full=event.event == StateEventKind.STATE)
|
||||
except socket.timeout:
|
||||
raise ControlError('timeout', 'the display went quiet') from None
|
||||
except ProtocolError as e:
|
||||
raise ControlError('bad_response', e.message) from None
|
||||
except OSError as e:
|
||||
if self._stop.is_set():
|
||||
return
|
||||
raise ControlError('closed', str(e)) from None
|
||||
|
||||
def _store(self, result: Dict[str, Any], full: bool) -> None:
|
||||
now = self._clock()
|
||||
with self._lock:
|
||||
if full and isinstance(result.get('state'), dict):
|
||||
self._snapshot = result
|
||||
self.snapshots += 1
|
||||
elif (self._snapshot is not None
|
||||
and result.get('epoch') == self._snapshot.get('epoch')):
|
||||
# A tick: nothing changed but the render loop's liveness and
|
||||
# the volatile keys (timestamps) the writers keep refreshing.
|
||||
snap = dict(self._snapshot)
|
||||
state = dict(snap.get('state') or {})
|
||||
if result.get('version') == snap.get('version'):
|
||||
_merge_volatile(state, result.get('volatile'))
|
||||
loop = result.get('loop')
|
||||
if isinstance(loop, dict):
|
||||
state['loop'] = loop
|
||||
snap['loop'] = loop
|
||||
snap['state'] = state
|
||||
snap['served_at'] = result.get('served_at', snap.get('served_at'))
|
||||
self._snapshot = snap
|
||||
else:
|
||||
return # a tick before any state, or from another epoch
|
||||
self._received = now
|
||||
self._connected = True
|
||||
|
||||
+339
-4
@@ -26,7 +26,19 @@ order. Commands that change what the panel shows are *acknowledged*, not
|
||||
completed: ``{"accepted": true, "request_id": ...}`` means the render thread
|
||||
has the command queued and will apply it at its next on-demand check. Its
|
||||
outcome is published the way it always was (``display_on_demand_state``,
|
||||
later the state stream).
|
||||
later the state stream). A few commands (:data:`AWAITED_COMMANDS`) are
|
||||
answered only once the render thread has applied them, or with ``pending``
|
||||
when it has not within :data:`AWAIT_SECONDS`.
|
||||
|
||||
``state.subscribe`` is the one exception to "one response per request": its
|
||||
response is followed, on the same connection, by :class:`StateEvent` lines
|
||||
the display pushes until either side hangs up. Events carry ``event``
|
||||
instead of ``ok``.
|
||||
|
||||
New commands are added within a protocol version: a display that does not
|
||||
know one answers ``unknown_command``, the client falls back, and ``hello``
|
||||
lists the commands a display knows. The version changes only when the
|
||||
envelope or the meaning of an existing command changes.
|
||||
|
||||
See docs/IPC_CONTROL_SOCKET.md for the full description.
|
||||
"""
|
||||
@@ -133,19 +145,74 @@ class Command:
|
||||
ON_DEMAND_START = 'on_demand.start'
|
||||
ON_DEMAND_STOP = 'on_demand.stop'
|
||||
ON_DEMAND_STATUS = 'on_demand.status'
|
||||
BRIGHTNESS_SET = 'brightness.set'
|
||||
PLUGIN_RELOAD = 'plugin.reload'
|
||||
STATE_GET = 'state.get'
|
||||
STATE_SUBSCRIBE = 'state.subscribe'
|
||||
ERRORS_CLEAR = 'errors.clear'
|
||||
|
||||
|
||||
#: Every command version 1 defines, in the order ``hello`` reports them.
|
||||
#: ``brightness.set`` and ``plugin.reload`` came in stage 2, ``state.get``
|
||||
#: and ``state.subscribe`` in stage 3, and ``errors.clear`` in stage 4, all
|
||||
#: within version 1 (see the module docstring on adding commands).
|
||||
COMMANDS: Tuple[str, ...] = (
|
||||
Command.HELLO,
|
||||
Command.PING,
|
||||
Command.ON_DEMAND_START,
|
||||
Command.ON_DEMAND_STOP,
|
||||
Command.ON_DEMAND_STATUS,
|
||||
Command.BRIGHTNESS_SET,
|
||||
Command.PLUGIN_RELOAD,
|
||||
Command.STATE_GET,
|
||||
Command.STATE_SUBSCRIBE,
|
||||
Command.ERRORS_CLEAR,
|
||||
)
|
||||
|
||||
#: Commands that are queued for the render thread and answered with an ack.
|
||||
QUEUED_COMMANDS = frozenset({Command.ON_DEMAND_START, Command.ON_DEMAND_STOP})
|
||||
#: Commands the connection thread answers itself, through a handler the
|
||||
#: display registers (``ControlServer(handlers=...)``), because they touch
|
||||
#: nothing the render thread owns. A display that registered none answers
|
||||
#: ``unknown_command``, and the client falls back as from an older display.
|
||||
DIRECT_COMMANDS = frozenset({Command.ERRORS_CLEAR})
|
||||
|
||||
#: Commands that are queued for the render thread.
|
||||
QUEUED_COMMANDS = frozenset({Command.ON_DEMAND_START, Command.ON_DEMAND_STOP,
|
||||
Command.BRIGHTNESS_SET, Command.PLUGIN_RELOAD})
|
||||
|
||||
#: Queued commands whose answer waits for the render thread's outcome
|
||||
#: instead of being an ack. The value is how long the display waits before
|
||||
#: answering ``pending``; the command stays queued and is still applied.
|
||||
#: A plugin reload first lets the current screen end (within a frame on a
|
||||
#: scrolling screen, at once on a static one) and then imports the plugin,
|
||||
#: which can take a few seconds on a slow board.
|
||||
AWAIT_SECONDS: Dict[str, float] = {
|
||||
Command.BRIGHTNESS_SET: 2.0,
|
||||
Command.PLUGIN_RELOAD: 10.0,
|
||||
}
|
||||
AWAITED_COMMANDS = frozenset(AWAIT_SECONDS)
|
||||
|
||||
#: The state stream (stage 3). ``state.subscribe`` turns its connection into
|
||||
#: a one-way stream of :class:`StateEvent` lines. Subscribers have their own
|
||||
#: bound, separate from the short request connections, so they can never
|
||||
#: take the slots a command needs.
|
||||
MAX_SUBSCRIBERS = 4
|
||||
|
||||
#: A subscriber hears from the display at least this often: a ``state``
|
||||
#: event when something changed, else a ``tick`` carrying the render loop's
|
||||
#: liveness and the latest volatile timestamps. A client that has heard nothing for a few of these treats its
|
||||
#: copy as unknown.
|
||||
SUBSCRIBE_KEEPALIVE_SECONDS = 5.0
|
||||
|
||||
#: The shape of the ``state`` object in a state snapshot. Bumped only when a
|
||||
#: field changes meaning; new fields are added within a schema.
|
||||
STATE_SCHEMA = 1
|
||||
|
||||
#: The sections of a state snapshot, in the order they are documented.
|
||||
STATE_SECTIONS: Tuple[str, ...] = ('display', 'on_demand', 'brightness', 'plugins', 'loop')
|
||||
|
||||
#: Brightness, in percent, as the display's hardware setting takes it.
|
||||
MIN_BRIGHTNESS = 0
|
||||
MAX_BRIGHTNESS = 100
|
||||
|
||||
|
||||
class ErrorCode:
|
||||
@@ -159,6 +226,10 @@ class ErrorCode:
|
||||
BUSY = 'busy' # queue full / too many clients
|
||||
FORBIDDEN = 'forbidden' # peer credentials refused
|
||||
INTERNAL = 'internal' # a bug on the display side
|
||||
# From the awaited commands (stage 2):
|
||||
PENDING = 'pending' # accepted, not applied within AWAIT_SECONDS; still queued
|
||||
NOT_LOADED = 'not_loaded' # plugin.reload: the display is not running that plugin
|
||||
FAILED = 'failed' # the render thread tried, and it did not work
|
||||
|
||||
|
||||
class ProtocolError(Exception):
|
||||
@@ -398,7 +469,139 @@ class NoArgs:
|
||||
return cls()
|
||||
|
||||
|
||||
CommandArgs = Union[HelloArgs, OnDemandStartArgs, OnDemandStopArgs, NoArgs]
|
||||
@dataclass(frozen=True)
|
||||
class BrightnessSetArgs:
|
||||
"""``brightness.set``: the panel's normal brightness, in percent, now.
|
||||
|
||||
Transient: nothing is written to config.json, and the next config change
|
||||
the display picks up (or a restart) goes back to the configured value.
|
||||
The web interface sends it after saving the setting, so the two agree.
|
||||
The dim schedule still applies on top, as it does to the saved value.
|
||||
"""
|
||||
brightness: int
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {'brightness': self.brightness}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, args: Mapping[str, Any]) -> 'BrightnessSetArgs':
|
||||
value = args.get('brightness')
|
||||
if not _is_int(value) or not MIN_BRIGHTNESS <= value <= MAX_BRIGHTNESS:
|
||||
raise ProtocolError(ErrorCode.INVALID_ARGS,
|
||||
f'brightness must be an integer from {MIN_BRIGHTNESS} '
|
||||
f'to {MAX_BRIGHTNESS}')
|
||||
return cls(brightness=value)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PluginReloadArgs:
|
||||
"""``plugin.reload``: load a running plugin again from disk.
|
||||
|
||||
For a plugin the store has just updated. Only a plugin the display is
|
||||
running can be reloaded (``not_loaded`` otherwise), so the id never
|
||||
makes the display import anything it was not already running.
|
||||
"""
|
||||
plugin_id: str
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {'plugin_id': self.plugin_id}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, args: Mapping[str, Any]) -> 'PluginReloadArgs':
|
||||
plugin_id = _optional_name(args, 'plugin_id')
|
||||
if plugin_id is None:
|
||||
raise ProtocolError(ErrorCode.INVALID_ARGS, 'plugin_id is required')
|
||||
return cls(plugin_id=plugin_id)
|
||||
|
||||
|
||||
def _optional_version(args: Mapping[str, Any], key: str) -> Optional[int]:
|
||||
value = args.get(key)
|
||||
if value is None:
|
||||
return None
|
||||
if not _is_int(value) or value < 0:
|
||||
raise ProtocolError(ErrorCode.INVALID_ARGS, f'{key} must be a non-negative integer')
|
||||
return value
|
||||
|
||||
|
||||
def _optional_epoch(args: Mapping[str, Any]) -> Optional[str]:
|
||||
value = args.get('epoch')
|
||||
if value is None or value == '':
|
||||
return None
|
||||
if not _valid_id(value):
|
||||
raise ProtocolError(ErrorCode.INVALID_ARGS,
|
||||
f'epoch must be a printable string of 1-{MAX_ID_LENGTH} characters')
|
||||
return str(value)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class StateGetArgs:
|
||||
"""``state.get``: the display's state, as a versioned snapshot.
|
||||
|
||||
With ``since`` and the ``epoch`` it came from, the answer is only
|
||||
``{changed: false, version, epoch, served_at, loop, volatile}`` while the
|
||||
state is still at that version, so a poller that already has it is sent
|
||||
no state -- only the latest values of the keys that do not count as a
|
||||
change (``volatile``, see :class:`StateSnapshot`).
|
||||
"""
|
||||
since: Optional[int] = None
|
||||
epoch: Optional[str] = None
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {'since': self.since, 'epoch': self.epoch}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, args: Mapping[str, Any]) -> 'StateGetArgs':
|
||||
return cls(since=_optional_version(args, 'since'), epoch=_optional_epoch(args))
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class StateSubscribeArgs:
|
||||
"""``state.subscribe``: the snapshot now, then a push stream of changes.
|
||||
|
||||
The response is the snapshot ``state.get`` returns. After it the
|
||||
connection carries only :class:`StateEvent` lines from the display: a
|
||||
``state`` event whenever the state changes (always the latest version,
|
||||
so a reader that falls behind skips versions instead of queueing them),
|
||||
and a ``tick`` at least every :data:`SUBSCRIBE_KEEPALIVE_SECONDS`.
|
||||
"""
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, args: Mapping[str, Any]) -> 'StateSubscribeArgs':
|
||||
return cls()
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ErrorsClearArgs:
|
||||
"""``errors.clear``: forget the plugin errors recorded at or before
|
||||
``cutoff`` (seconds since the epoch), as ``POST /api/v3/errors/clear``
|
||||
asks. The request id is the clear's id, which the display's error
|
||||
snapshot then reports as ``applied_clear_id``.
|
||||
"""
|
||||
cutoff: float
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {'cutoff': self.cutoff}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, args: Mapping[str, Any]) -> 'ErrorsClearArgs':
|
||||
value = args.get('cutoff')
|
||||
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
||||
raise ProtocolError(ErrorCode.INVALID_ARGS, 'cutoff must be a number of seconds')
|
||||
if not math.isfinite(value) or value < 0:
|
||||
raise ProtocolError(ErrorCode.INVALID_ARGS,
|
||||
'cutoff must be a finite, non-negative number of seconds')
|
||||
return cls(cutoff=float(value))
|
||||
|
||||
|
||||
CommandArgs = Union[HelloArgs, OnDemandStartArgs, OnDemandStopArgs, NoArgs,
|
||||
BrightnessSetArgs, PluginReloadArgs, StateGetArgs, StateSubscribeArgs,
|
||||
ErrorsClearArgs]
|
||||
|
||||
#: The arguments of a command that goes on the render thread's queue.
|
||||
QueuedArgs = Union[OnDemandStartArgs, OnDemandStopArgs, BrightnessSetArgs, PluginReloadArgs]
|
||||
|
||||
_ARG_TYPES: Dict[str, Any] = {
|
||||
Command.HELLO: HelloArgs,
|
||||
@@ -406,6 +609,11 @@ _ARG_TYPES: Dict[str, Any] = {
|
||||
Command.ON_DEMAND_START: OnDemandStartArgs,
|
||||
Command.ON_DEMAND_STOP: OnDemandStopArgs,
|
||||
Command.ON_DEMAND_STATUS: NoArgs,
|
||||
Command.BRIGHTNESS_SET: BrightnessSetArgs,
|
||||
Command.PLUGIN_RELOAD: PluginReloadArgs,
|
||||
Command.STATE_GET: StateGetArgs,
|
||||
Command.STATE_SUBSCRIBE: StateSubscribeArgs,
|
||||
Command.ERRORS_CLEAR: ErrorsClearArgs,
|
||||
}
|
||||
|
||||
|
||||
@@ -457,6 +665,133 @@ class AckResult(TypedDict):
|
||||
queued: int
|
||||
|
||||
|
||||
class BrightnessResult(TypedDict):
|
||||
"""``brightness.set``, once applied.
|
||||
|
||||
``panel_brightness`` is what the panel shows now: the dim schedule's
|
||||
level while it dims, and unchanged while the schedule has the display
|
||||
off (the new level applies when it comes back on).
|
||||
"""
|
||||
brightness: int
|
||||
panel_brightness: int
|
||||
dimmed: bool
|
||||
display_active: bool
|
||||
|
||||
|
||||
class PluginReloadResult(TypedDict):
|
||||
"""``plugin.reload``, once the plugin is running again."""
|
||||
plugin_id: str
|
||||
reloaded: bool
|
||||
version: Optional[str]
|
||||
modes: List[str]
|
||||
|
||||
|
||||
class ErrorsClearResult(TypedDict):
|
||||
"""``errors.clear``, once applied and the error snapshot republished."""
|
||||
request_id: str
|
||||
cutoff: float
|
||||
cleared: int
|
||||
|
||||
|
||||
class LoopState(TypedDict):
|
||||
"""``loop``: is the render loop still going round?
|
||||
|
||||
``heartbeat_age_seconds`` is the age of the render thread's last beat,
|
||||
measured in memory by the display when it answered -- the same beat that
|
||||
writes ``display-heartbeat.json``. None until the loop has drawn its first
|
||||
frame. At ``stale_after`` or more the loop is stalled: the threshold
|
||||
``/api/v3/health`` uses.
|
||||
"""
|
||||
heartbeat_age_seconds: Optional[float]
|
||||
armed: bool
|
||||
stale_after: float
|
||||
|
||||
|
||||
class StateSnapshot(TypedDict, total=False):
|
||||
"""The answer to ``state.get`` and ``state.subscribe``, and the
|
||||
``result`` of a ``state`` event.
|
||||
|
||||
``version`` counts changes to the state within one ``epoch`` (one run of
|
||||
the display process): a reader that sees a new epoch starts over.
|
||||
``changed`` is False only for a ``state.get`` whose ``since`` is still
|
||||
current, and then ``state`` is absent and ``volatile`` is there instead:
|
||||
``{section: {key: value}}``, the current values of the keys the version
|
||||
ignores (``display.last_updated``, ``on_demand.last_updated`` and
|
||||
``remaining``, ``plugins.published_at``). A reader merges them into the
|
||||
copy it has; they are how it can tell the writers are still publishing.
|
||||
``served_at`` is the display's wall clock when it answered. ``loop`` is
|
||||
measured at that moment, so it is also inside ``state``.
|
||||
|
||||
``state`` holds the sections in :data:`STATE_SECTIONS`:
|
||||
|
||||
* ``display``: what ``display_current_state`` holds (mode, plugin_id,
|
||||
mode_index, total_modes, on_demand_active, is_display_active,
|
||||
last_updated);
|
||||
* ``on_demand``: what ``display_on_demand_state`` holds;
|
||||
* ``brightness``: ``{brightness, panel_brightness, dimmed}``;
|
||||
* ``plugins``: the plugin runtime snapshot (``plugin_runtime_snapshot``),
|
||||
or None when there is none (or it was too large to send);
|
||||
* ``loop``: :class:`LoopState`.
|
||||
|
||||
A section the display has not published yet is None.
|
||||
"""
|
||||
schema: int
|
||||
version: int
|
||||
epoch: str
|
||||
pid: int
|
||||
served_at: float
|
||||
changed: bool
|
||||
state: Dict[str, Any]
|
||||
volatile: Dict[str, Dict[str, Any]]
|
||||
loop: LoopState
|
||||
|
||||
|
||||
class StateEventKind:
|
||||
STATE = 'state' # result: a full StateSnapshot, the latest version
|
||||
TICK = 'tick' # result: {version, epoch, pid, served_at, loop, volatile}; nothing changed
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class StateEvent:
|
||||
"""One message the display pushes to a subscriber.
|
||||
|
||||
``{"v": 1, "id": "<the subscribe request's id>", "event": "state" | "tick",
|
||||
"result": {...}}``. It has no ``ok``, which is how a reader tells it from
|
||||
a response.
|
||||
"""
|
||||
id: str
|
||||
event: str
|
||||
result: Dict[str, Any]
|
||||
v: int = PROTOCOL_VERSION
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
return {'v': self.v, 'id': self.id, 'event': self.event, 'result': dict(self.result)}
|
||||
|
||||
@classmethod
|
||||
def from_dict(cls, obj: Any) -> 'StateEvent':
|
||||
"""Validate an event. Raises :class:`ProtocolError` (BAD_REQUEST)."""
|
||||
if not isinstance(obj, dict):
|
||||
raise ProtocolError(ErrorCode.BAD_REQUEST, 'an event must be a JSON object')
|
||||
version = obj.get('v')
|
||||
if not _is_int(version):
|
||||
raise ProtocolError(ErrorCode.BAD_REQUEST, 'v must be an integer')
|
||||
raw_id = obj.get('id')
|
||||
if not isinstance(raw_id, str):
|
||||
raise ProtocolError(ErrorCode.BAD_REQUEST, 'id must be a string')
|
||||
event = obj.get('event')
|
||||
if event not in (StateEventKind.STATE, StateEventKind.TICK):
|
||||
raise ProtocolError(ErrorCode.BAD_REQUEST, 'event must be "state" or "tick"')
|
||||
result = obj.get('result')
|
||||
if not isinstance(result, dict):
|
||||
raise ProtocolError(ErrorCode.BAD_REQUEST, 'result must be a JSON object')
|
||||
return cls(id=raw_id, event=event, result=result, v=version)
|
||||
|
||||
|
||||
def is_event(obj: Any) -> bool:
|
||||
"""Whether a decoded message is a pushed event rather than a response."""
|
||||
return isinstance(obj, dict) and 'event' in obj and 'ok' not in obj
|
||||
|
||||
|
||||
def negotiate_version(client_versions: Tuple[int, ...]) -> Optional[int]:
|
||||
"""The highest version both sides speak, or None."""
|
||||
common = set(client_versions) & set(SUPPORTED_VERSIONS)
|
||||
|
||||
+536
-26
@@ -6,7 +6,22 @@ rendering: a command that changes the panel is validated, put on a bounded
|
||||
queue and acknowledged, and the render thread drains that queue at the point
|
||||
where it reads the file mailbox (``DisplayController._poll_on_demand_requests``),
|
||||
handing each command to the same code. Queries (``on_demand.status``) are
|
||||
answered from a snapshot callable the display provides.
|
||||
answered from a snapshot callable the display provides, and the few commands
|
||||
that touch nothing the render thread owns (``errors.clear``) by a handler the
|
||||
display registers, on the connection thread.
|
||||
|
||||
The queue also wakes the render thread: :meth:`ControlServer.wait_for_command`
|
||||
is what it waits on in place of a sleep, so a command lands within a frame on
|
||||
every kind of screen. An awaited command (``brightness.set``,
|
||||
``plugin.reload``) carries a :class:`CommandOutcome` that the render thread
|
||||
fills in; its connection thread waits for that, bounded, before answering.
|
||||
|
||||
The state stream (stage 3): the display publishes what it is doing into a
|
||||
:class:`StateHub`, in memory, and ``state.get`` / ``state.subscribe`` read
|
||||
it. A subscriber's connection gives back its request slot, takes one of
|
||||
:data:`~src.ipc.contract.MAX_SUBSCRIBERS`, and is pushed the latest version
|
||||
on every change plus a keepalive tick, from its own thread: publishing never
|
||||
waits for a reader, and a reader that stops reading is dropped.
|
||||
|
||||
Robustness rules, because this runs inside the display process:
|
||||
|
||||
@@ -31,6 +46,7 @@ server checks them again: root, its own user, or a member of that group.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import queue
|
||||
@@ -39,18 +55,27 @@ import stat
|
||||
import struct
|
||||
import threading
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Callable, Dict, FrozenSet, List, Mapping, Optional, Union
|
||||
import uuid
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, Callable, Dict, FrozenSet, Iterable, List, Mapping, Optional, Tuple
|
||||
|
||||
from src.ipc.contract import (
|
||||
AWAIT_SECONDS,
|
||||
AWAITED_COMMANDS,
|
||||
COMMANDS,
|
||||
DEFAULT_SOCKET_DIR,
|
||||
DIRECT_COMMANDS,
|
||||
DEFAULT_SOCKET_PATH,
|
||||
MAX_MESSAGE_BYTES,
|
||||
MAX_SUBSCRIBERS,
|
||||
PROTOCOL_VERSION,
|
||||
QUEUED_COMMANDS,
|
||||
STATE_SCHEMA,
|
||||
STATE_SECTIONS,
|
||||
SUBSCRIBE_KEEPALIVE_SECONDS,
|
||||
SUPPORTED_VERSIONS,
|
||||
AckResult,
|
||||
BrightnessSetArgs,
|
||||
Command,
|
||||
ErrorCode,
|
||||
FrameReader,
|
||||
@@ -58,9 +83,14 @@ from src.ipc.contract import (
|
||||
HelloResult,
|
||||
OnDemandStartArgs,
|
||||
OnDemandStopArgs,
|
||||
PluginReloadArgs,
|
||||
ProtocolError,
|
||||
QueuedArgs,
|
||||
Request,
|
||||
Response,
|
||||
StateEvent,
|
||||
StateEventKind,
|
||||
StateGetArgs,
|
||||
configured_socket_path,
|
||||
decode_message,
|
||||
dev_socket_path,
|
||||
@@ -80,7 +110,8 @@ MAX_CLIENTS = 8
|
||||
|
||||
#: Commands waiting for the render thread. It drains them at least every
|
||||
#: 0.25 s, so a full queue means the render thread is stuck, and the client
|
||||
#: is told ``busy`` (and falls back to the mailbox) instead of piling up work.
|
||||
#: is told ``busy`` instead of piling up work. The mailbox would not be read
|
||||
#: either, so the web interface reports the failure rather than fall back.
|
||||
QUEUE_SIZE = 16
|
||||
|
||||
#: Timeout for one recv()/send() on a connection.
|
||||
@@ -100,19 +131,310 @@ _LISTEN_BACKLOG = 64
|
||||
|
||||
# -- queued work ---------------------------------------------------------------------
|
||||
|
||||
class CommandOutcome:
|
||||
"""How an awaited command turned out, handed from the render thread back
|
||||
to the connection thread that is waiting to answer.
|
||||
|
||||
The render thread calls :meth:`succeed` or :meth:`fail` once; the first
|
||||
call wins. The connection thread may have stopped waiting already (it
|
||||
answered ``pending``), and then nobody reads it.
|
||||
"""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._done = threading.Event()
|
||||
self._lock = threading.Lock()
|
||||
self.result: Optional[Dict[str, Any]] = None
|
||||
self.error_code: Optional[str] = None
|
||||
self.error_message = ''
|
||||
|
||||
@property
|
||||
def done(self) -> bool:
|
||||
return self._done.is_set()
|
||||
|
||||
def succeed(self, result: Mapping[str, Any]) -> None:
|
||||
with self._lock:
|
||||
if self._done.is_set():
|
||||
return
|
||||
self.result = dict(result)
|
||||
self._done.set()
|
||||
|
||||
def fail(self, code: str, message: str) -> None:
|
||||
with self._lock:
|
||||
if self._done.is_set():
|
||||
return
|
||||
self.error_code = code
|
||||
self.error_message = message
|
||||
self._done.set()
|
||||
|
||||
def wait(self, timeout: float) -> bool:
|
||||
return self._done.wait(timeout)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class QueuedCommand:
|
||||
"""A command waiting for the render thread."""
|
||||
"""A command waiting for the render thread.
|
||||
|
||||
``outcome`` is set for an awaited command (``AWAITED_COMMANDS``): the
|
||||
render thread reports through it, and the client's answer waits for it.
|
||||
"""
|
||||
request_id: str
|
||||
cmd: str
|
||||
args: Union[OnDemandStartArgs, OnDemandStopArgs]
|
||||
args: QueuedArgs
|
||||
received_at: float # time.time() when it was accepted
|
||||
peer_uid: Optional[int] = None
|
||||
outcome: Optional[CommandOutcome] = field(default=None, compare=False, repr=False)
|
||||
|
||||
def as_on_demand_request(self) -> Dict[str, Any]:
|
||||
"""The mailbox-shaped payload the display's on-demand handler takes."""
|
||||
if not isinstance(self.args, (OnDemandStartArgs, OnDemandStopArgs)):
|
||||
raise TypeError(f'{self.cmd} is not an on-demand command')
|
||||
return on_demand_request(self.request_id, self.args, self.received_at)
|
||||
|
||||
def succeed(self, result: Mapping[str, Any]) -> None:
|
||||
"""Report success to a waiting client (a no-op for an acked command)."""
|
||||
if self.outcome is not None:
|
||||
self.outcome.succeed(result)
|
||||
|
||||
def fail(self, code: str, message: str) -> None:
|
||||
"""Report failure to a waiting client (a no-op for an acked command)."""
|
||||
if self.outcome is not None:
|
||||
self.outcome.fail(code, message)
|
||||
|
||||
|
||||
# -- the state stream (stage 3) ----------------------------------------------------------
|
||||
|
||||
#: How long a ``state.get`` keeps the display counting its readers as served
|
||||
#: over the socket (:meth:`StateHub.readers_active`). A subscriber counts for
|
||||
#: as long as it is connected.
|
||||
READER_WINDOW_SECONDS = 60.0
|
||||
|
||||
#: Room kept for the envelope (``v``, ``id``, ``event``) around a snapshot,
|
||||
#: within MAX_MESSAGE_BYTES.
|
||||
_ENVELOPE_ROOM = 512
|
||||
|
||||
_MISSING = object()
|
||||
|
||||
LoopProbe = Callable[[], Mapping[str, Any]]
|
||||
|
||||
|
||||
def _fingerprint(value: Optional[Mapping[str, Any]], volatile: Iterable[str]) -> Any:
|
||||
"""What a section's version is judged on: the value minus its volatile keys
|
||||
(timestamps that move on every publish without anything changing)."""
|
||||
if value is None:
|
||||
return None
|
||||
skip = frozenset(volatile)
|
||||
return {k: v for k, v in value.items() if k not in skip} if skip else dict(value)
|
||||
|
||||
|
||||
def _volatile_values(sections: Mapping[str, Optional[Dict[str, Any]]],
|
||||
volatile: Mapping[str, FrozenSet[str]]) -> Dict[str, Dict[str, Any]]:
|
||||
"""``{section: {key: value}}``: the volatile keys each published section
|
||||
has now. The sections are never mutated after publish (a publish swaps
|
||||
in a new dict), so reading them outside the lock is safe."""
|
||||
values: Dict[str, Dict[str, Any]] = {}
|
||||
for name, keys in volatile.items():
|
||||
value = sections.get(name)
|
||||
if not keys or not isinstance(value, dict):
|
||||
continue
|
||||
present = {k: value[k] for k in keys if k in value}
|
||||
if present:
|
||||
values[name] = present
|
||||
return values
|
||||
|
||||
|
||||
def _unknown_loop() -> Dict[str, Any]:
|
||||
return {'heartbeat_age_seconds': None, 'armed': False, 'stale_after': None}
|
||||
|
||||
|
||||
def fit_snapshot(snapshot: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"""``snapshot``, or a copy without the plugin runtime section when the
|
||||
message would be over MAX_MESSAGE_BYTES (hundreds of plugins). The
|
||||
reader then falls back to the cache for that section only; ``truncated``
|
||||
says which was left out."""
|
||||
state = snapshot.get('state')
|
||||
if not isinstance(state, dict) or state.get('plugins') is None:
|
||||
return snapshot
|
||||
try:
|
||||
size = len(json.dumps(snapshot, separators=(',', ':'), ensure_ascii=True,
|
||||
allow_nan=False))
|
||||
except (TypeError, ValueError):
|
||||
size = MAX_MESSAGE_BYTES
|
||||
if size <= MAX_MESSAGE_BYTES - _ENVELOPE_ROOM:
|
||||
return snapshot
|
||||
logger.warning("State snapshot is %d bytes; sending it without the plugin runtime "
|
||||
"section", size)
|
||||
trimmed = dict(snapshot)
|
||||
trimmed['state'] = dict(state, plugins=None)
|
||||
trimmed['truncated'] = ['plugins']
|
||||
return trimmed
|
||||
|
||||
|
||||
class StateHub:
|
||||
"""The display's live state, in memory, for ``state.get`` and ``state.subscribe``.
|
||||
|
||||
Writers publish whole sections (:meth:`publish`): the render thread
|
||||
publishes ``display``, ``on_demand`` and ``brightness``, and the plugin
|
||||
runtime publisher's thread publishes ``plugins``. Each section has one
|
||||
writer. ``loop`` is not published: it is measured when a reader asks
|
||||
(``loop_probe``), so it keeps ageing while the render thread is stuck.
|
||||
|
||||
The version goes up when a section's value changes, ignoring the keys
|
||||
the publisher names as volatile (timestamps). Those keys still carry
|
||||
news -- ``display.last_updated`` is the render thread's proof of life --
|
||||
so the short ``changed: false`` answer, which is what a subscriber's
|
||||
tick carries, has their current values in ``volatile``; a reader merges
|
||||
them into its copy. Publishing never blocks on
|
||||
a reader: the lock is held only to swap a dict reference and compare it,
|
||||
and every socket write happens on the reader's own thread, outside it.
|
||||
A reader that is slow gets the latest version when it next asks, not
|
||||
every version in between.
|
||||
"""
|
||||
|
||||
def __init__(self, loop_probe: Optional[LoopProbe] = None, *,
|
||||
clock: Callable[[], float] = time.monotonic,
|
||||
wall_clock: Callable[[], float] = time.time,
|
||||
epoch: Optional[str] = None, pid: Optional[int] = None,
|
||||
reader_window: float = READER_WINDOW_SECONDS):
|
||||
self._cond = threading.Condition(threading.Lock())
|
||||
self._sections: Dict[str, Optional[Dict[str, Any]]] = {}
|
||||
self._fingerprints: Dict[str, Any] = {}
|
||||
self._volatile: Dict[str, FrozenSet[str]] = {}
|
||||
self._version = 0
|
||||
self.epoch = epoch or uuid.uuid4().hex[:16]
|
||||
self.pid = os.getpid() if pid is None else pid
|
||||
self._loop_probe = loop_probe
|
||||
self._clock = clock
|
||||
self._wall_clock = wall_clock
|
||||
self._reader_window = reader_window
|
||||
self._last_read: Optional[float] = None
|
||||
self._subscribers = 0
|
||||
|
||||
@property
|
||||
def version(self) -> int:
|
||||
return self._version
|
||||
|
||||
@property
|
||||
def subscribers(self) -> int:
|
||||
return self._subscribers
|
||||
|
||||
# -- writers -------------------------------------------------------------
|
||||
|
||||
def publish(self, section: str, value: Optional[Mapping[str, Any]],
|
||||
volatile: Iterable[str] = ()) -> bool:
|
||||
"""Store a section's latest value; True when that is a new version.
|
||||
|
||||
The value is copied (one level), so the caller may reuse its dict.
|
||||
"""
|
||||
stored = None if value is None else dict(value)
|
||||
skip = frozenset(volatile)
|
||||
fingerprint = _fingerprint(stored, skip)
|
||||
with self._cond:
|
||||
self._sections[section] = stored
|
||||
self._volatile[section] = skip
|
||||
if self._fingerprints.get(section, _MISSING) == fingerprint:
|
||||
return False
|
||||
self._fingerprints[section] = fingerprint
|
||||
self._version += 1
|
||||
self._cond.notify_all()
|
||||
return True
|
||||
|
||||
def wake(self) -> None:
|
||||
"""Wake every waiting reader (the server is closing)."""
|
||||
with self._cond:
|
||||
self._cond.notify_all()
|
||||
|
||||
# -- readers -------------------------------------------------------------
|
||||
|
||||
def loop(self) -> Dict[str, Any]:
|
||||
"""The render loop's liveness now. Never raises."""
|
||||
if self._loop_probe is None:
|
||||
return _unknown_loop()
|
||||
try:
|
||||
return dict(self._loop_probe())
|
||||
except Exception: # pylint: disable=broad-except
|
||||
logger.debug("Render loop liveness probe failed", exc_info=True)
|
||||
return _unknown_loop()
|
||||
|
||||
def snapshot(self, since: Optional[int] = None,
|
||||
epoch: Optional[str] = None) -> Dict[str, Any]:
|
||||
"""The :class:`~src.ipc.contract.StateSnapshot` now.
|
||||
|
||||
``since`` with this hub's ``epoch``, still the current version, gives
|
||||
the short ``changed: false`` form, with ``volatile``: each section's
|
||||
volatile keys at their latest values (the rest of the section is
|
||||
what the reader already has).
|
||||
"""
|
||||
with self._cond:
|
||||
version = self._version
|
||||
sections = dict(self._sections)
|
||||
volatile = dict(self._volatile)
|
||||
loop = self.loop()
|
||||
result: Dict[str, Any] = {
|
||||
'schema': STATE_SCHEMA,
|
||||
'version': version,
|
||||
'epoch': self.epoch,
|
||||
'pid': self.pid,
|
||||
'served_at': self._wall_clock(),
|
||||
'loop': loop,
|
||||
}
|
||||
if since is not None and epoch == self.epoch and since == version:
|
||||
result['changed'] = False
|
||||
result['volatile'] = _volatile_values(sections, volatile)
|
||||
return result
|
||||
state: Dict[str, Any] = {name: sections.get(name) for name in STATE_SECTIONS
|
||||
if name != 'loop'}
|
||||
state['loop'] = loop
|
||||
result['changed'] = True
|
||||
result['state'] = state
|
||||
return result
|
||||
|
||||
def wait_for_change(self, version: int, timeout: float,
|
||||
stop: Optional[threading.Event] = None) -> bool:
|
||||
"""Block up to ``timeout`` for a version other than ``version``."""
|
||||
with self._cond:
|
||||
self._cond.wait_for(
|
||||
lambda: self._version != version or (stop is not None and stop.is_set()),
|
||||
timeout)
|
||||
return self._version != version
|
||||
|
||||
# -- who is reading ------------------------------------------------------
|
||||
|
||||
def note_read(self) -> None:
|
||||
self._last_read = self._clock()
|
||||
|
||||
def subscriber_joined(self) -> None:
|
||||
with self._cond:
|
||||
self._subscribers += 1
|
||||
|
||||
def subscriber_left(self) -> None:
|
||||
with self._cond:
|
||||
self._subscribers = max(0, self._subscribers - 1)
|
||||
self._last_read = self._clock()
|
||||
|
||||
def readers_active(self) -> bool:
|
||||
"""Is the socket serving state readers? A subscriber is connected, or a
|
||||
``state.get`` came within the reader window. The display uses this to
|
||||
write the cache copies of the same state less often."""
|
||||
if self._subscribers > 0:
|
||||
return True
|
||||
last = self._last_read
|
||||
return last is not None and self._clock() - last < self._reader_window
|
||||
|
||||
|
||||
class _Slot:
|
||||
"""A connection slot, released once (a subscriber gives its back early)."""
|
||||
|
||||
def __init__(self, semaphore: threading.BoundedSemaphore):
|
||||
self._semaphore = semaphore
|
||||
self._held = True
|
||||
self._lock = threading.Lock()
|
||||
|
||||
def release(self) -> None:
|
||||
with self._lock:
|
||||
if self._held:
|
||||
self._held = False
|
||||
self._semaphore.release()
|
||||
|
||||
|
||||
# -- peer credentials ------------------------------------------------------------------
|
||||
|
||||
@@ -234,6 +556,12 @@ def server_socket_path(environ: Optional[Mapping[str, str]] = None) -> Optional[
|
||||
|
||||
StatusProvider = Callable[[], Dict[str, Any]]
|
||||
|
||||
#: A handler for one of DIRECT_COMMANDS, ``(request_id, args) -> result``. It
|
||||
#: runs on the connection thread, so it must not touch what the render thread
|
||||
#: owns. It may raise ProtocolError to answer with that error's code; any
|
||||
#: other exception is answered ``internal``.
|
||||
DirectHandler = Callable[[str, Any], Mapping[str, Any]]
|
||||
|
||||
|
||||
class ControlServer:
|
||||
"""Serves the control socket on background threads.
|
||||
@@ -247,8 +575,21 @@ class ControlServer:
|
||||
max_clients: int = MAX_CLIENTS, io_timeout: float = IO_TIMEOUT_SECONDS,
|
||||
message_timeout: float = MESSAGE_TIMEOUT_SECONDS,
|
||||
idle_timeout: float = IDLE_TIMEOUT_SECONDS,
|
||||
check_peer: bool = True):
|
||||
check_peer: bool = True,
|
||||
await_seconds: Optional[Mapping[str, float]] = None,
|
||||
state_hub: Optional[StateHub] = None,
|
||||
max_subscribers: int = MAX_SUBSCRIBERS,
|
||||
keepalive: float = SUBSCRIBE_KEEPALIVE_SECONDS,
|
||||
handlers: Optional[Mapping[str, DirectHandler]] = None):
|
||||
self.path = path
|
||||
self.state_hub = state_hub
|
||||
self._handlers: Dict[str, DirectHandler] = {
|
||||
cmd: fn for cmd, fn in (handlers or {}).items() if cmd in DIRECT_COMMANDS}
|
||||
self._subscriber_slots = threading.BoundedSemaphore(max_subscribers)
|
||||
self._keepalive = keepalive
|
||||
self._await_seconds: Dict[str, float] = dict(AWAIT_SECONDS)
|
||||
if await_seconds:
|
||||
self._await_seconds.update(await_seconds)
|
||||
self._status_provider = status_provider
|
||||
self._group = group
|
||||
self._queue: 'queue.Queue[QueuedCommand]' = queue.Queue(maxsize=queue_size)
|
||||
@@ -307,6 +648,8 @@ class ControlServer:
|
||||
"""Stop accepting and remove the socket file (only if it is still ours)."""
|
||||
self._stopping.set()
|
||||
self._close_socket()
|
||||
if self.state_hub is not None:
|
||||
self.state_hub.wake() # subscribers see _stopping and hang up
|
||||
thread = self._thread
|
||||
if thread is not None and thread is not threading.current_thread():
|
||||
thread.join(timeout=2.0)
|
||||
@@ -418,8 +761,33 @@ class ControlServer:
|
||||
"""Cheap check for queued commands, for the render thread's fast path."""
|
||||
return self._pending.is_set()
|
||||
|
||||
def wait_for_command(self, timeout: float) -> bool:
|
||||
"""Block up to ``timeout`` seconds for a queued command; True if one is.
|
||||
|
||||
The render thread waits here instead of sleeping, in the dwell and
|
||||
on a static screen, so a command wakes it at once. It is a timed
|
||||
wait on an Event: no polling, and nothing more than the sleep it
|
||||
replaces when no command comes. The flag stays set until drain(),
|
||||
so a caller that does not drain would return at once every time.
|
||||
"""
|
||||
return self._pending.wait(timeout)
|
||||
|
||||
def wake(self) -> None:
|
||||
"""Wake the render thread as a queued command would, with nothing queued.
|
||||
|
||||
For work that reaches the display another way in the same process (a
|
||||
plugin's on-demand request, ``DisplayController.submit_plugin_on_demand``):
|
||||
the render thread returns from :meth:`wait_for_command` and drains,
|
||||
and reads the caller's own queue there. Safe from any thread.
|
||||
"""
|
||||
self._pending.set()
|
||||
|
||||
def drain(self) -> List[QueuedCommand]:
|
||||
"""Every queued command, oldest first. Called from the render thread."""
|
||||
"""Every queued command, oldest first. Called from the render thread.
|
||||
|
||||
Clears the wake flag first, so anything queued (or woken for) while
|
||||
this runs wakes the next wait again.
|
||||
"""
|
||||
commands: List[QueuedCommand] = []
|
||||
self._pending.clear()
|
||||
while True:
|
||||
@@ -467,6 +835,7 @@ class ControlServer:
|
||||
|
||||
def _serve(self, conn: socket.socket) -> None:
|
||||
"""One connection: authenticate, then answer requests until it ends."""
|
||||
slot = _Slot(self._slots)
|
||||
try:
|
||||
conn.settimeout(self._io_timeout)
|
||||
peer = peer_credentials(conn)
|
||||
@@ -475,7 +844,7 @@ class ControlServer:
|
||||
"this user or group %s", peer.pid, peer.uid, peer.gid, self._group)
|
||||
self._send(conn, Response.failure(None, ErrorCode.FORBIDDEN, 'not permitted'))
|
||||
return
|
||||
self._read_requests(conn, peer)
|
||||
self._read_requests(conn, peer, slot)
|
||||
except Exception: # pylint: disable=broad-except
|
||||
logger.exception("Control socket connection failed")
|
||||
finally:
|
||||
@@ -483,7 +852,7 @@ class ControlServer:
|
||||
conn.close()
|
||||
except OSError:
|
||||
pass
|
||||
self._slots.release()
|
||||
slot.release()
|
||||
|
||||
def _peer_ok(self, peer: PeerCredentials) -> bool:
|
||||
groups = None
|
||||
@@ -491,7 +860,8 @@ class ControlServer:
|
||||
groups = process_groups(peer.pid)
|
||||
return peer_allowed(peer, self._own_uid, self._group, groups)
|
||||
|
||||
def _read_requests(self, conn: socket.socket, peer: Optional[PeerCredentials]) -> None:
|
||||
def _read_requests(self, conn: socket.socket, peer: Optional[PeerCredentials],
|
||||
slot: Optional[_Slot] = None) -> None:
|
||||
reader = FrameReader(MAX_MESSAGE_BYTES)
|
||||
idle_since = time.monotonic()
|
||||
message_started: Optional[float] = None
|
||||
@@ -516,7 +886,13 @@ class ControlServer:
|
||||
self._send(conn, Response.failure(None, e.code, e.message))
|
||||
return # can't find the next message boundary: hang up
|
||||
for line in lines:
|
||||
if not self._send(conn, self.handle_line(line, peer)):
|
||||
response, cmd = self._handle(line, peer)
|
||||
if cmd == Command.STATE_SUBSCRIBE and response.ok:
|
||||
# The connection becomes a one-way stream; anything the
|
||||
# client sent after the subscribe is ignored.
|
||||
self._subscribe(conn, response, slot)
|
||||
return
|
||||
if not self._send(conn, response):
|
||||
return
|
||||
if reader.pending:
|
||||
if message_started is None or lines:
|
||||
@@ -542,20 +918,91 @@ class ControlServer:
|
||||
# -- requests --------------------------------------------------------------------
|
||||
|
||||
def handle_line(self, line: bytes, peer: Optional[PeerCredentials] = None) -> Response:
|
||||
"""Answer one request line. Never raises."""
|
||||
"""Answer one request line. Never raises.
|
||||
|
||||
A ``state.subscribe`` answered here gets its snapshot only; the
|
||||
stream that follows needs a connection (``_read_requests``).
|
||||
"""
|
||||
return self._handle(line, peer)[0]
|
||||
|
||||
def _handle(self, line: bytes,
|
||||
peer: Optional[PeerCredentials]) -> Tuple[Response, Optional[str]]:
|
||||
"""The response to one line, and the command it answered (when known)."""
|
||||
request_id: Optional[str] = None
|
||||
cmd: Optional[str] = None
|
||||
try:
|
||||
obj = decode_message(line)
|
||||
raw_id = obj.get('id')
|
||||
request_id = raw_id if isinstance(raw_id, str) and len(raw_id) <= 128 else None
|
||||
request = Request.from_dict(obj)
|
||||
request_id = request.id
|
||||
return self._dispatch(request, peer)
|
||||
cmd = request.cmd
|
||||
return self._dispatch(request, peer), cmd
|
||||
except ProtocolError as e:
|
||||
return Response.failure(e.request_id or request_id, e.code, e.message)
|
||||
return Response.failure(e.request_id or request_id, e.code, e.message), cmd
|
||||
except Exception: # pylint: disable=broad-except
|
||||
logger.exception("Control socket handler failed")
|
||||
return Response.failure(request_id, ErrorCode.INTERNAL, 'internal error')
|
||||
return Response.failure(request_id, ErrorCode.INTERNAL, 'internal error'), cmd
|
||||
|
||||
# -- the state stream ----------------------------------------------------------
|
||||
|
||||
def _subscribe(self, conn: socket.socket, response: Response,
|
||||
slot: Optional[_Slot]) -> None:
|
||||
"""Answer a ``state.subscribe`` and push state events until it ends.
|
||||
|
||||
Subscribers have their own bound (MAX_SUBSCRIBERS) and give their
|
||||
request slot back, so a few browsers watching never use up the slots
|
||||
commands need. Everything here runs on this connection's thread: a
|
||||
reader that does not keep up only stalls its own sends, and one that
|
||||
stops reading for a whole IO timeout is dropped. The render thread
|
||||
only ever publishes into the hub.
|
||||
"""
|
||||
hub = self.state_hub
|
||||
if hub is None or not self._subscriber_slots.acquire(blocking=False):
|
||||
self._send(conn, Response.failure(response.id, ErrorCode.BUSY,
|
||||
'too many state subscribers', v=response.v))
|
||||
return
|
||||
if slot is not None:
|
||||
slot.release()
|
||||
hub.subscriber_joined()
|
||||
try:
|
||||
if not self._send(conn, response):
|
||||
return
|
||||
result = response.result or {}
|
||||
version = result.get('version', -1)
|
||||
sub_id = response.id or ''
|
||||
logger.debug("Control socket: state subscriber joined at version %s", version)
|
||||
while not self._stopping.is_set():
|
||||
hub.wait_for_change(version, self._keepalive, self._stopping)
|
||||
if self._stopping.is_set():
|
||||
return
|
||||
snap = hub.snapshot(since=version, epoch=hub.epoch)
|
||||
if snap.get('changed'):
|
||||
version = snap['version']
|
||||
event = StateEvent(sub_id, StateEventKind.STATE, fit_snapshot(snap),
|
||||
v=response.v)
|
||||
else:
|
||||
event = StateEvent(sub_id, StateEventKind.TICK, snap, v=response.v)
|
||||
if not self._send_event(conn, event):
|
||||
return
|
||||
finally:
|
||||
hub.subscriber_left()
|
||||
self._subscriber_slots.release()
|
||||
|
||||
def _send_event(self, conn: socket.socket, event: StateEvent) -> bool:
|
||||
try:
|
||||
data = encode_message(event.to_dict())
|
||||
except ProtocolError as e:
|
||||
logger.error("Control socket state event not sent: %s", e.message)
|
||||
return False
|
||||
try:
|
||||
conn.sendall(data)
|
||||
return True
|
||||
except socket.timeout:
|
||||
logger.info("Control socket: dropping a state subscriber that stopped reading")
|
||||
return False
|
||||
except OSError:
|
||||
return False
|
||||
|
||||
def _dispatch(self, request: Request, peer: Optional[PeerCredentials]) -> Response:
|
||||
if request.cmd == Command.HELLO:
|
||||
@@ -596,11 +1043,28 @@ class ControlServer:
|
||||
v=request.v)
|
||||
return Response.success(request.id, self._status_provider(), v=request.v)
|
||||
|
||||
if request.cmd in QUEUED_COMMANDS and isinstance(args, (OnDemandStartArgs,
|
||||
OnDemandStopArgs)):
|
||||
if request.cmd in (Command.STATE_GET, Command.STATE_SUBSCRIBE):
|
||||
hub = self.state_hub
|
||||
if hub is None:
|
||||
return Response.failure(request.id, ErrorCode.INTERNAL, 'no state available',
|
||||
v=request.v)
|
||||
if isinstance(args, StateGetArgs):
|
||||
hub.note_read()
|
||||
snap = hub.snapshot(since=args.since, epoch=args.epoch)
|
||||
else:
|
||||
snap = hub.snapshot()
|
||||
return Response.success(request.id, fit_snapshot(snap), v=request.v)
|
||||
|
||||
if request.cmd in DIRECT_COMMANDS:
|
||||
return self._direct(request, args)
|
||||
|
||||
if request.cmd in QUEUED_COMMANDS and isinstance(args, (
|
||||
OnDemandStartArgs, OnDemandStopArgs, BrightnessSetArgs, PluginReloadArgs)):
|
||||
awaited = request.cmd in AWAITED_COMMANDS
|
||||
command = QueuedCommand(request_id=request.id, cmd=request.cmd, args=args,
|
||||
received_at=time.time(),
|
||||
peer_uid=peer.uid if peer is not None else None)
|
||||
peer_uid=peer.uid if peer is not None else None,
|
||||
outcome=CommandOutcome() if awaited else None)
|
||||
try:
|
||||
self._queue.put_nowait(command)
|
||||
except queue.Full:
|
||||
@@ -610,34 +1074,80 @@ class ControlServer:
|
||||
'the display is not taking commands right now',
|
||||
v=request.v)
|
||||
self._pending.set()
|
||||
logger.info("Control socket accepted %s %s", request.cmd, request.id)
|
||||
if command.outcome is not None:
|
||||
return self._await_outcome(request, command.outcome)
|
||||
ack: AckResult = {'accepted': True, 'request_id': request.id,
|
||||
'queued': self._queue.qsize()}
|
||||
logger.info("Control socket accepted %s %s", request.cmd, request.id)
|
||||
return Response.success(request.id, dict(ack), v=request.v)
|
||||
|
||||
# A command in COMMANDS with no handler here is a bug in this module.
|
||||
return Response.failure(request.id, ErrorCode.INTERNAL,
|
||||
f'{request.cmd} is not implemented', v=request.v)
|
||||
|
||||
def _direct(self, request: Request, args: Any) -> Response:
|
||||
"""A command the display answers on this thread (DIRECT_COMMANDS)."""
|
||||
handler = self._handlers.get(request.cmd)
|
||||
if handler is None:
|
||||
# Answered as an older display would, so the client falls back.
|
||||
return Response.failure(request.id, ErrorCode.UNKNOWN_COMMAND,
|
||||
f'{request.cmd} is not served by this display',
|
||||
v=request.v)
|
||||
try:
|
||||
result = handler(request.id, args)
|
||||
except ProtocolError as e:
|
||||
return Response.failure(request.id, e.code, e.message, v=request.v)
|
||||
except Exception: # pylint: disable=broad-except
|
||||
logger.exception("Control socket: %s %s failed", request.cmd, request.id)
|
||||
return Response.failure(request.id, ErrorCode.INTERNAL,
|
||||
'the display failed to apply it', v=request.v)
|
||||
logger.info("Control socket applied %s %s", request.cmd, request.id)
|
||||
return Response.success(request.id, dict(result), v=request.v)
|
||||
|
||||
def _await_outcome(self, request: Request, outcome: CommandOutcome) -> Response:
|
||||
"""Answer an awaited command once the render thread has applied it.
|
||||
|
||||
Waits on this connection's thread, never the render thread's. If the
|
||||
render thread does not get to it in time, the answer is ``pending``:
|
||||
the command stays queued and is still applied, so a client treats
|
||||
that as "not known to be done" rather than as a refusal.
|
||||
"""
|
||||
timeout = self._await_seconds.get(request.cmd, 0.0)
|
||||
if not outcome.wait(timeout):
|
||||
logger.warning("Control socket: %s %s not applied within %.1fs; answering pending",
|
||||
request.cmd, request.id, timeout)
|
||||
return Response.failure(request.id, ErrorCode.PENDING,
|
||||
f'accepted, but not applied within {timeout:g}s; '
|
||||
'the display will still apply it', v=request.v)
|
||||
if outcome.error_code is not None:
|
||||
return Response.failure(request.id, outcome.error_code, outcome.error_message,
|
||||
v=request.v)
|
||||
return Response.success(request.id, outcome.result or {}, v=request.v)
|
||||
|
||||
|
||||
def start_control_server(status_provider: Optional[StatusProvider] = None,
|
||||
cache_dir: Optional[str] = None,
|
||||
environ: Optional[Mapping[str, str]] = None) -> Optional[ControlServer]:
|
||||
environ: Optional[Mapping[str, str]] = None,
|
||||
state_hub: Optional[StateHub] = None,
|
||||
handlers: Optional[Mapping[str, DirectHandler]] = None,
|
||||
) -> Optional[ControlServer]:
|
||||
"""Start the display's control socket, or return None when it can't run.
|
||||
|
||||
None covers Windows, ``LEDMATRIX_CONTROL_SOCKET=off`` and any failure to
|
||||
bind; in every case the web interface falls back to the file mailbox.
|
||||
bind; in every case the web interface falls back to the file mailbox
|
||||
and to the cache keys the display still writes.
|
||||
"""
|
||||
path = server_socket_path(environ)
|
||||
if path is None:
|
||||
logger.debug("Control socket disabled or unsupported here; using the file mailbox only")
|
||||
return None
|
||||
server = ControlServer(path, status_provider, resolve_socket_group(cache_dir))
|
||||
server = ControlServer(path, status_provider, resolve_socket_group(cache_dir),
|
||||
state_hub=state_hub, handlers=handlers)
|
||||
return server if server.start() else None
|
||||
|
||||
|
||||
__all__ = [
|
||||
'ControlServer', 'PeerCredentials', 'QueuedCommand', 'StatusProvider',
|
||||
'peer_allowed', 'peer_credentials', 'process_groups', 'resolve_socket_group',
|
||||
'server_socket_path', 'start_control_server', 'PROTOCOL_VERSION',
|
||||
'CommandOutcome', 'ControlServer', 'PeerCredentials', 'QueuedCommand', 'StateHub',
|
||||
'StatusProvider', 'fit_snapshot', 'peer_allowed', 'peer_credentials', 'process_groups',
|
||||
'resolve_socket_group', 'server_socket_path', 'start_control_server', 'PROTOCOL_VERSION',
|
||||
]
|
||||
|
||||
@@ -21,6 +21,7 @@ from PIL.PngImagePlugin import PngInfo
|
||||
from requests.adapters import HTTPAdapter
|
||||
from urllib3.util.retry import Retry
|
||||
from src.common.api_helper import DEFAULT_HTTP_HEADERS
|
||||
from src.common.json_body import response_json
|
||||
from src.common.logo_helper import MAX_LOGO_BYTES
|
||||
from src.common.permission_utils import (
|
||||
ensure_directory_permissions,
|
||||
@@ -481,7 +482,7 @@ class LogoDownloader:
|
||||
logger.info(f"Fetching team data for {league} from ESPN API...")
|
||||
response = self.session.get(api_url, params={'limit':1000},headers=self.headers, timeout=self.request_timeout)
|
||||
response.raise_for_status()
|
||||
data: Dict = response.json()
|
||||
data: Dict = response_json(response)
|
||||
|
||||
logger.info(f"Successfully fetched team data for {league}")
|
||||
return data
|
||||
@@ -505,7 +506,7 @@ class LogoDownloader:
|
||||
logger.info(f"Fetching team data for team {team_id} in {league} from ESPN API...")
|
||||
response = self.session.get(f"{api_url}/{team_id}", headers=self.headers, timeout=self.request_timeout)
|
||||
response.raise_for_status()
|
||||
data: Dict = response.json()
|
||||
data: Dict = response_json(response)
|
||||
|
||||
logger.info(f"Successfully fetched team data for {team_id} in {league}")
|
||||
return data
|
||||
|
||||
@@ -0,0 +1,123 @@
|
||||
"""Keep glibc's malloc from holding on to memory the display has freed.
|
||||
|
||||
The display process allocates and frees PIL images and numpy buffers all day
|
||||
from a dozen threads. glibc gives each allocating thread its own malloc arena
|
||||
(up to 8 x CPU count) and returns little of what is freed inside them to the
|
||||
OS, so resident memory climbs for hours while the live data stays flat. Two
|
||||
in-process remedies, both standard library only (ctypes) and both no-ops off
|
||||
Linux/glibc:
|
||||
|
||||
* :func:`cap_arenas` -- ``mallopt(M_ARENA_MAX, 2)``, the in-process twin of the
|
||||
unit's ``Environment=MALLOC_ARENA_MAX=2``. Units installed before that line
|
||||
existed never got it (systemd runs the copy in /etc/systemd/system), so the
|
||||
process applies it itself. Call it before any other thread starts: arenas
|
||||
already created stay. A ``MALLOC_ARENA_MAX`` set in the environment wins.
|
||||
* :class:`MallocTrimmer` -- ``malloc_trim(0)`` at most every few minutes,
|
||||
called from the render loop between screens, where no frame is being drawn.
|
||||
glibc 2.8+ releases free pages from the middle of every arena, not only the
|
||||
top of the main heap.
|
||||
|
||||
Without glibc (macOS, Windows, musl, the dev server on any of them) nothing is
|
||||
loaded and every call returns False.
|
||||
"""
|
||||
import ctypes
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from typing import Any, Callable, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
#: glibc's mallopt() parameter number for the arena cap (malloc.h).
|
||||
M_ARENA_MAX = -8
|
||||
|
||||
#: The arena cap applied when the environment does not set one; the same value
|
||||
#: as the unit's ``MALLOC_ARENA_MAX``.
|
||||
DEFAULT_ARENA_MAX = 2
|
||||
|
||||
#: Seconds between malloc_trim() calls. A trim takes about 1-20 ms on a Pi 4,
|
||||
#: so this keeps it far from frame timing while still returning memory long
|
||||
#: before it piles up.
|
||||
TRIM_INTERVAL_SECONDS = 300.0
|
||||
|
||||
_UNLOADED = object()
|
||||
_libc: Any = _UNLOADED
|
||||
|
||||
|
||||
def _load_libc() -> Optional[Any]:
|
||||
"""The process's C library if it is glibc with malloc_trim, else None."""
|
||||
global _libc
|
||||
if _libc is _UNLOADED:
|
||||
_libc = None
|
||||
if sys.platform.startswith('linux'):
|
||||
try:
|
||||
libc = ctypes.CDLL(None)
|
||||
# gnu_get_libc_version is glibc-only, so musl (which has
|
||||
# mallopt but no malloc_trim) is left alone as a whole.
|
||||
if all(hasattr(libc, name) for name in
|
||||
('gnu_get_libc_version', 'malloc_trim', 'mallopt')):
|
||||
libc.malloc_trim.argtypes = [ctypes.c_size_t]
|
||||
libc.malloc_trim.restype = ctypes.c_int
|
||||
libc.mallopt.argtypes = [ctypes.c_int, ctypes.c_int]
|
||||
libc.mallopt.restype = ctypes.c_int
|
||||
_libc = libc
|
||||
except (OSError, AttributeError, TypeError):
|
||||
logger.debug("glibc malloc controls unavailable", exc_info=True)
|
||||
return _libc
|
||||
|
||||
|
||||
def cap_arenas(max_arenas: int = DEFAULT_ARENA_MAX) -> bool:
|
||||
"""Cap glibc's malloc arenas at ``max_arenas``. True when the cap was set.
|
||||
|
||||
Skipped when ``MALLOC_ARENA_MAX`` is in the environment: glibc has read it
|
||||
already, and an operator who set it chose that value.
|
||||
"""
|
||||
if os.environ.get('MALLOC_ARENA_MAX'):
|
||||
return False
|
||||
libc = _load_libc()
|
||||
if libc is None:
|
||||
return False
|
||||
try:
|
||||
return bool(libc.mallopt(M_ARENA_MAX, int(max_arenas)))
|
||||
except Exception: # pylint: disable=broad-except
|
||||
logger.debug("mallopt(M_ARENA_MAX) failed", exc_info=True)
|
||||
return False
|
||||
|
||||
|
||||
class MallocTrimmer:
|
||||
"""Calls ``malloc_trim(0)`` at most once per ``interval`` seconds.
|
||||
|
||||
:meth:`maybe_trim` is meant for an idle point of the render loop; it costs
|
||||
one clock read when no trim is due. The first trim comes one interval
|
||||
after construction, so start-up's allocations have settled.
|
||||
"""
|
||||
|
||||
def __init__(self, interval: float = TRIM_INTERVAL_SECONDS,
|
||||
clock: Callable[[], float] = time.monotonic) -> None:
|
||||
self._interval = interval
|
||||
self._clock = clock
|
||||
self._libc = _load_libc()
|
||||
self._next = clock() + interval
|
||||
|
||||
@property
|
||||
def available(self) -> bool:
|
||||
return self._libc is not None
|
||||
|
||||
def maybe_trim(self) -> bool:
|
||||
"""Trim if one is due. True when malloc_trim ran and released memory."""
|
||||
if self._libc is None:
|
||||
return False
|
||||
now = self._clock()
|
||||
if now < self._next:
|
||||
return False
|
||||
self._next = now + self._interval
|
||||
try:
|
||||
released = bool(self._libc.malloc_trim(0))
|
||||
except Exception: # pylint: disable=broad-except
|
||||
logger.debug("malloc_trim failed; not trying again", exc_info=True)
|
||||
self._libc = None
|
||||
return False
|
||||
logger.debug("malloc_trim(0) took %.1f ms, released=%s",
|
||||
(self._clock() - now) * 1000.0, released)
|
||||
return released
|
||||
@@ -3,15 +3,46 @@ LEDMatrix Plugin System
|
||||
|
||||
This module provides the core plugin infrastructure for the LEDMatrix project.
|
||||
It enables dynamic loading, management, and discovery of display plugins.
|
||||
|
||||
BasePlugin and PluginManager are imported on first use (PEP 562), not when
|
||||
the package is imported: the web interface imports several submodules
|
||||
(store_manager, schema_manager, ...) and never needs PluginManager, which
|
||||
pulls in the loader, executor and the shared helpers behind them.
|
||||
``from src.plugin_system import BasePlugin`` works as before and returns the
|
||||
same class.
|
||||
"""
|
||||
|
||||
import importlib
|
||||
from typing import TYPE_CHECKING, Any, Dict, List, Tuple
|
||||
|
||||
__version__ = "1.0.0"
|
||||
|
||||
from .base_plugin import BasePlugin
|
||||
from .plugin_manager import PluginManager
|
||||
if TYPE_CHECKING:
|
||||
from .base_plugin import BasePlugin
|
||||
from .plugin_manager import PluginManager
|
||||
|
||||
#: Exported name -> (module it lives in, attribute name there).
|
||||
_LAZY: Dict[str, Tuple[str, str]] = {
|
||||
'BasePlugin': ('src.plugin_system.base_plugin', 'BasePlugin'),
|
||||
'PluginManager': ('src.plugin_system.plugin_manager', 'PluginManager'),
|
||||
}
|
||||
|
||||
__all__ = [
|
||||
'BasePlugin',
|
||||
'PluginManager',
|
||||
]
|
||||
|
||||
|
||||
def __getattr__(name: str) -> Any:
|
||||
"""Import an exported name on first access (PEP 562); see src.common."""
|
||||
try:
|
||||
module_name, attr = _LAZY[name]
|
||||
except KeyError:
|
||||
raise AttributeError(f"module {__name__!r} has no attribute {name!r}") from None
|
||||
value = getattr(importlib.import_module(module_name), attr) # nosemgrep: python.lang.security.audit.non-literal-import.non-literal-import -- module_name comes from the fixed _LAZY table
|
||||
globals()[name] = value
|
||||
return value
|
||||
|
||||
|
||||
def __dir__() -> List[str]:
|
||||
return sorted(set(globals()) | set(__all__))
|
||||
|
||||
@@ -1070,6 +1070,65 @@ class BasePlugin(ABC):
|
||||
if callable(notify):
|
||||
notify(self.plugin_id)
|
||||
|
||||
def request_on_demand(self, mode: Optional[str] = None,
|
||||
duration: Optional[float] = None,
|
||||
pinned: bool = False) -> Optional[str]:
|
||||
"""
|
||||
Take the screen now: show this plugin on demand. Safe from any thread.
|
||||
|
||||
For a plugin that reacts to something outside the rotation -- an MQTT
|
||||
message, a timer, a detection -- and wants the panel for it. The
|
||||
request goes straight to the display in this process and is applied
|
||||
on its render thread within a frame or so, exactly like an on-demand
|
||||
start from the web interface.
|
||||
|
||||
Args:
|
||||
mode: One of this plugin's display modes; None for its first.
|
||||
duration: Seconds to show it before the rotation resumes; None
|
||||
(or zero) for no limit, until end_on_demand() or the user
|
||||
stops it.
|
||||
pinned: Stay on ``mode`` instead of cycling through the
|
||||
plugin's other modes.
|
||||
|
||||
Returns:
|
||||
The request id once the display has queued it, or None when
|
||||
there is no display in this process to ask (the web interface,
|
||||
scripts/check_plugin.py) or its queue is full. A plugin that
|
||||
also runs on cores without this method writes the
|
||||
``display_on_demand_request`` mailbox on None, as before; see
|
||||
"On-demand display" in docs/PLUGIN_API_REFERENCE.md.
|
||||
|
||||
Example::
|
||||
|
||||
if not (hasattr(self, 'request_on_demand')
|
||||
and self.request_on_demand(mode='my_alert', duration=15)):
|
||||
self._write_on_demand_mailbox(...) # older cores
|
||||
"""
|
||||
request = getattr(getattr(self, 'plugin_manager', None), 'request_on_demand', None)
|
||||
if not callable(request):
|
||||
return None
|
||||
request_id = request(self.plugin_id, mode=mode, duration=duration, pinned=pinned)
|
||||
# Only a real id counts: a test's MagicMock manager answers a mock,
|
||||
# which must read as "not taken" so the plugin's fallback runs.
|
||||
return request_id if isinstance(request_id, str) else None
|
||||
|
||||
def end_on_demand(self) -> Optional[str]:
|
||||
"""
|
||||
Give the screen back: end this plugin's on-demand session. Any thread.
|
||||
|
||||
Ends only a session this plugin owns. One the user started for
|
||||
another plugin, or a session that already ended, is left alone. The
|
||||
rotation resumes where it left off.
|
||||
|
||||
Returns:
|
||||
The request id once queued, or None as request_on_demand() does.
|
||||
"""
|
||||
end = getattr(getattr(self, 'plugin_manager', None), 'end_on_demand', None)
|
||||
if not callable(end):
|
||||
return None
|
||||
request_id = end(self.plugin_id)
|
||||
return request_id if isinstance(request_id, str) else None
|
||||
|
||||
def get_vegas_participation(self) -> str:
|
||||
"""
|
||||
How this plugin takes part in Vegas mode: ``'scroll'``, ``'pause'`` or
|
||||
|
||||
@@ -48,12 +48,19 @@ class PluginOperation:
|
||||
completed_at: Optional[datetime] = None
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
"""Convert operation to dictionary for serialization."""
|
||||
"""Convert operation to dictionary for serialization.
|
||||
|
||||
Parameters whose name starts with ``_`` are internal and left out:
|
||||
PluginOperationQueue keeps the operation's callback there as
|
||||
``_callback`` until its worker runs it, and a pending operation's
|
||||
status answered 500 because that function cannot be serialized.
|
||||
"""
|
||||
return {
|
||||
'operation_id': self.operation_id,
|
||||
'operation_type': self.operation_type.value,
|
||||
'plugin_id': self.plugin_id,
|
||||
'parameters': self.parameters,
|
||||
'parameters': {key: value for key, value in self.parameters.items()
|
||||
if not str(key).startswith('_')},
|
||||
'status': self.status.value,
|
||||
'progress': self.progress,
|
||||
'message': self.message,
|
||||
|
||||
@@ -238,7 +238,9 @@ def display_restart_required(action: str, plugin_enabled: bool, *,
|
||||
config carried over) is not picked up until a restart.
|
||||
- ``update``: the display keeps running the code it loaded until it
|
||||
restarts, if it runs the plugin at all -- only when it is enabled.
|
||||
``changed=False`` (already up to date) needs nothing.
|
||||
``changed=False`` (already up to date) needs nothing. The update route
|
||||
first asks the display to reload it over the control socket
|
||||
(``_reload_after_store_update``); this answer stands when it cannot.
|
||||
- ``uninstall``: removing the plugin's config section flips its enabled
|
||||
flag, and the reconcile unloads it. With ``preserve_config`` the flag
|
||||
stays, and an enabled plugin keeps running until a restart.
|
||||
|
||||
@@ -10,6 +10,7 @@ from typing import Any, Dict, Optional, Callable
|
||||
from threading import Thread
|
||||
import logging
|
||||
|
||||
from src.common.fetch_service import plugin_scope
|
||||
from src.exceptions import PluginError
|
||||
from src.logging_config import get_logger
|
||||
from src.error_aggregator import record_error
|
||||
@@ -58,7 +59,8 @@ class PluginExecutor:
|
||||
self,
|
||||
operation: Callable[[], Any],
|
||||
timeout: Optional[float] = None,
|
||||
plugin_id: Optional[str] = None
|
||||
plugin_id: Optional[str] = None,
|
||||
thread_name: Optional[str] = None
|
||||
) -> Any:
|
||||
"""
|
||||
Execute a plugin operation with timeout.
|
||||
@@ -67,6 +69,8 @@ class PluginExecutor:
|
||||
operation: Function to execute
|
||||
timeout: Timeout in seconds (None = use default)
|
||||
plugin_id: Optional plugin ID for logging
|
||||
thread_name: Name for the thread the operation runs on (None
|
||||
keeps Python's default). Stack dumps list threads by name.
|
||||
|
||||
Returns:
|
||||
Result of operation
|
||||
@@ -83,13 +87,19 @@ class PluginExecutor:
|
||||
|
||||
def target():
|
||||
try:
|
||||
result_container['value'] = operation()
|
||||
# Fetches made by the operation (and by threads the core
|
||||
# starts from it) are counted against this plugin.
|
||||
with plugin_scope(plugin_id):
|
||||
result_container['value'] = operation()
|
||||
result_container['completed'] = True
|
||||
except Exception as e:
|
||||
except BaseException as e: # pylint: disable=broad-except
|
||||
# asyncio.CancelledError and SystemExit too: uncaught, one
|
||||
# ended this thread with 'completed' unset, and an operation
|
||||
# that failed at once was reported as timing out.
|
||||
result_container['exception'] = e
|
||||
result_container['completed'] = True
|
||||
|
||||
thread = Thread(target=target, daemon=True)
|
||||
thread = Thread(target=target, daemon=True, name=thread_name)
|
||||
thread.start()
|
||||
thread.join(timeout=timeout)
|
||||
|
||||
@@ -173,7 +183,8 @@ class PluginExecutor:
|
||||
force_clear: bool = False,
|
||||
display_mode: Optional[str] = None,
|
||||
timeout: Optional[float] = None,
|
||||
accepts_display_mode: Optional[bool] = None
|
||||
accepts_display_mode: Optional[bool] = None,
|
||||
raise_errors: bool = False
|
||||
) -> bool:
|
||||
"""
|
||||
Execute plugin display() method with error handling.
|
||||
@@ -187,9 +198,18 @@ class PluginExecutor:
|
||||
accepts_display_mode: Whether plugin.display() takes a
|
||||
display_mode keyword. Pass it when the caller already knows;
|
||||
None falls back to inspecting the callable.
|
||||
raise_errors: Re-raise the PluginError wrapping an exception
|
||||
display() raised, instead of returning False. False alone
|
||||
cannot tell "no content" from "raised", and a caller that
|
||||
feeds the circuit breaker needs that difference. The error
|
||||
is still logged and recorded first. A timeout still returns
|
||||
False either way.
|
||||
|
||||
Returns:
|
||||
True if display succeeded, False otherwise
|
||||
|
||||
Raises:
|
||||
PluginError: Only with ``raise_errors``, when display() raised.
|
||||
"""
|
||||
try:
|
||||
start_time = time.monotonic()
|
||||
@@ -209,18 +229,24 @@ class PluginExecutor:
|
||||
'display_mode' in inspect.signature(plugin.display).parameters)
|
||||
has_display_mode = accepts_display_mode
|
||||
|
||||
# Named for the plugin: this thread presents a screen's first
|
||||
# frame, so the frame-timing stall watchdog's stack dumps name it.
|
||||
thread_name = f"display-{plugin_id}"
|
||||
|
||||
# Capture the return value from the plugin's display() method
|
||||
if has_display_mode and display_mode:
|
||||
result = self.execute_with_timeout(
|
||||
lambda: plugin.display(display_mode=display_mode, force_clear=force_clear),
|
||||
timeout=timeout,
|
||||
plugin_id=plugin_id
|
||||
plugin_id=plugin_id,
|
||||
thread_name=thread_name
|
||||
)
|
||||
else:
|
||||
result = self.execute_with_timeout(
|
||||
lambda: plugin.display(force_clear=force_clear),
|
||||
timeout=timeout,
|
||||
plugin_id=plugin_id
|
||||
plugin_id=plugin_id,
|
||||
thread_name=thread_name
|
||||
)
|
||||
|
||||
duration = time.monotonic() - start_time
|
||||
@@ -245,6 +271,8 @@ class PluginExecutor:
|
||||
return False
|
||||
except PluginError:
|
||||
# Already logged and recorded in execute_with_timeout
|
||||
if raise_errors:
|
||||
raise
|
||||
return False
|
||||
except Exception as e:
|
||||
self.logger.error(
|
||||
|
||||
@@ -199,9 +199,22 @@ def contained_plugin_dir(plugin_dir: Path, plugins_dir: Path) -> Optional[str]:
|
||||
name that came out of ``os.scandir()`` on the trusted root carries no
|
||||
taint, which is a real containment guarantee (and one CodeQL's
|
||||
path-injection query can follow), not a string sanitiser.
|
||||
|
||||
The entry looked for is the one ``plugin_dir`` itself names when it sits
|
||||
directly in ``plugins_dir``: for a dev plugin symlinked in under its id,
|
||||
the link's name. Resolving the link first and looking for the target's
|
||||
folder name refused ``plugins/foo -> ~/.ledmatrix-dev-plugins/ledmatrix-foo``
|
||||
(what ``dev_plugin_setup.sh link-github foo <url>`` makes), so the plugin
|
||||
never loaded. Any other path is resolved and matched by its final name,
|
||||
as before.
|
||||
"""
|
||||
plugin_dir_real = os.path.realpath(str(plugin_dir))
|
||||
plugins_dir_real = os.path.realpath(str(plugins_dir))
|
||||
plugin_dir_abs = os.path.abspath(str(plugin_dir))
|
||||
if os.path.realpath(os.path.dirname(plugin_dir_abs)) == plugins_dir_real:
|
||||
matched_name = find_trusted_subdir(plugins_dir_real, os.path.basename(plugin_dir_abs))
|
||||
if matched_name is not None:
|
||||
return os.path.join(plugins_dir_real, matched_name)
|
||||
plugin_dir_real = os.path.realpath(str(plugin_dir))
|
||||
matched_name = find_trusted_subdir(plugins_dir_real, os.path.basename(plugin_dir_real))
|
||||
if matched_name is None:
|
||||
return None
|
||||
@@ -243,6 +256,10 @@ class PluginLoader:
|
||||
self.logger = logger or get_logger(__name__)
|
||||
self._loaded_modules: Dict[str, Any] = {}
|
||||
self._plugin_module_registry: Dict[str, set] = {} # Maps plugin_id to set of module names
|
||||
# plugin_id -> {dotted name: module} for the modules of the plugin's
|
||||
# own packages (``providers.feed``). They keep their names while the
|
||||
# plugin runs and are dropped with it; see _iter_plugin_submodules.
|
||||
self._plugin_submodules: Dict[str, Dict[str, Any]] = {}
|
||||
# Lock to serialize module loading when plugins share module names
|
||||
# (e.g., scroll_display.py, game_renderer.py across sport plugins).
|
||||
# During exec_module, bare-name sub-modules temporarily appear in
|
||||
@@ -449,6 +466,45 @@ class PluginLoader:
|
||||
continue
|
||||
return result
|
||||
|
||||
@staticmethod
|
||||
def _iter_plugin_submodules(
|
||||
plugin_dir: Path, before_keys: set
|
||||
) -> list:
|
||||
"""Return dotted-name modules from plugin_dir added after before_keys.
|
||||
|
||||
The modules of a package the plugin ships (``providers.feed`` from
|
||||
``providers/feed.py``). _iter_plugin_bare_modules skips them, so the
|
||||
bare ``providers`` was namespaced and dropped on unload while
|
||||
``providers.feed`` stayed in sys.modules: a reload after a store update
|
||||
imported a fresh ``providers`` and then got the old ``feed`` back from
|
||||
the cache, running the new manager.py against the old helpers until the
|
||||
display restarted.
|
||||
|
||||
A module counts when its ``__file__`` -- or, for a namespace package,
|
||||
which has none, every ``__path__`` entry -- is inside plugin_dir, so a
|
||||
library the plugin imports (``requests.adapters``) never does.
|
||||
|
||||
Returns a list of (mod_name, module) tuples.
|
||||
"""
|
||||
resolved_dir = plugin_dir.resolve()
|
||||
result = []
|
||||
for key in set(sys.modules.keys()) - before_keys:
|
||||
if "." not in key:
|
||||
continue
|
||||
mod = sys.modules.get(key)
|
||||
if mod is None:
|
||||
continue
|
||||
mod_file = getattr(mod, "__file__", None)
|
||||
locations = [mod_file] if mod_file else list(getattr(mod, "__path__", None) or [])
|
||||
if not locations:
|
||||
continue
|
||||
try:
|
||||
if all(Path(loc).resolve().is_relative_to(resolved_dir) for loc in locations):
|
||||
result.append((key, mod))
|
||||
except (ValueError, TypeError, OSError):
|
||||
continue
|
||||
return result
|
||||
|
||||
def _evict_stale_bare_modules(self, plugin_dir: Path) -> dict:
|
||||
"""Temporarily remove bare-name sys.modules entries from other plugins.
|
||||
|
||||
@@ -527,6 +583,13 @@ class PluginLoader:
|
||||
# Track for cleanup during unload
|
||||
self._plugin_module_registry[plugin_id] = namespaced_names
|
||||
|
||||
# The modules of the plugin's own packages keep their dotted names
|
||||
# while it runs -- as they always have, so the package and its
|
||||
# children stay a matching set in sys.modules -- and are dropped
|
||||
# with the plugin by unregister_plugin_modules().
|
||||
self._plugin_submodules[plugin_id] = dict(
|
||||
self._iter_plugin_submodules(plugin_dir, before_keys))
|
||||
|
||||
if namespaced_names:
|
||||
self.logger.info(
|
||||
"Namespace-isolated %d module(s) for plugin %s",
|
||||
@@ -537,10 +600,16 @@ class PluginLoader:
|
||||
"""Remove namespaced sub-modules and cached module for a plugin from sys.modules.
|
||||
|
||||
Called by PluginManager during unload to clean up all module entries
|
||||
that were created when the plugin was loaded.
|
||||
that were created when the plugin was loaded, including the dotted
|
||||
modules of its packages. A dotted name is dropped only while it still
|
||||
holds this plugin's module: the name is not namespaced, so another
|
||||
plugin may have put its own there since.
|
||||
"""
|
||||
for ns_name in self._plugin_module_registry.pop(plugin_id, set()):
|
||||
sys.modules.pop(ns_name, None)
|
||||
for name, mod in self._plugin_submodules.pop(plugin_id, {}).items():
|
||||
if sys.modules.get(name) is mod:
|
||||
sys.modules.pop(name, None)
|
||||
self._loaded_modules.pop(plugin_id, None)
|
||||
|
||||
def load_module(
|
||||
@@ -646,11 +715,13 @@ class PluginLoader:
|
||||
if evicted_name not in sys.modules:
|
||||
sys.modules[evicted_name] = evicted_mod
|
||||
# Clean up the partially-initialized main module and any
|
||||
# bare-name sub-modules that were added during exec_module
|
||||
# so they don't leak into subsequent plugin loads.
|
||||
# bare-name or package sub-modules that were added during
|
||||
# exec_module so they don't leak into subsequent plugin loads.
|
||||
sys.modules.pop(module_name, None)
|
||||
for key, _ in self._iter_plugin_bare_modules(plugin_dir, before_keys):
|
||||
sys.modules.pop(key, None)
|
||||
for key, _ in self._iter_plugin_submodules(plugin_dir, before_keys):
|
||||
sys.modules.pop(key, None)
|
||||
raise
|
||||
|
||||
self._loaded_modules[plugin_id] = module
|
||||
|
||||
@@ -0,0 +1,204 @@
|
||||
"""
|
||||
Files a plugin writes beside itself at runtime, which an update must keep.
|
||||
|
||||
A store update replaces a plugin's directory with a fresh download and then
|
||||
deletes the old copy. Anything the plugin created there -- OAuth tokens, a
|
||||
client-secrets file, a PKCE verifier, cached state -- is in no release, so the
|
||||
fresh download does not contain it and deleting the old copy destroys it. On
|
||||
2026-10-04 updating calendar 1.2.9 -> 1.2.12 that way deleted its
|
||||
``token.pickle`` and ``credentials.json``, and the calendar stopped until they
|
||||
were restored from a backup.
|
||||
|
||||
What counts as "the plugin's own local file" is the union of:
|
||||
|
||||
* :data:`KNOWN_STATE_PATTERNS` -- secret and state files plugins are known to
|
||||
write, kept even when a plugin forgot to gitignore them; and
|
||||
* whatever the plugin's own ``.gitignore`` (old copy or new) excludes. A file
|
||||
the author ignores is by definition not part of a release.
|
||||
|
||||
A file the new release ships is never overwritten: tracked content wins. Byte
|
||||
code (``__pycache__``, ``*.pyc``) and ``.git`` are never carried, since they
|
||||
belong to the old code rather than to the user.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import fnmatch
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
from typing import Iterable, List, Optional, Pattern, Tuple
|
||||
|
||||
__all__ = [
|
||||
'KNOWN_STATE_PATTERNS',
|
||||
'carry_over_local_files',
|
||||
'is_known_state_file',
|
||||
'local_files_to_keep',
|
||||
]
|
||||
|
||||
# Basename globs. Kept even when the plugin's .gitignore does not list them.
|
||||
KNOWN_STATE_PATTERNS: Tuple[str, ...] = (
|
||||
'token.pickle',
|
||||
'*.pickle',
|
||||
'token.json',
|
||||
'credentials.json',
|
||||
'config_secrets.json',
|
||||
'.pkce_code_verifier',
|
||||
)
|
||||
|
||||
_NEVER_CARRY_DIRS = frozenset({'.git', '__pycache__'})
|
||||
_NEVER_CARRY_SUFFIXES = ('.pyc', '.pyo')
|
||||
|
||||
|
||||
def is_known_state_file(rel_path: str) -> bool:
|
||||
"""True when ``rel_path``'s basename is a known secret/state file."""
|
||||
name = rel_path.replace('\\', '/').rsplit('/', 1)[-1]
|
||||
return any(fnmatch.fnmatchcase(name, p) for p in KNOWN_STATE_PATTERNS)
|
||||
|
||||
|
||||
class _GitIgnore:
|
||||
"""The subset of gitignore semantics plugin .gitignore files use.
|
||||
|
||||
Supports comments, ``!`` negation (last match wins), a trailing ``/`` for
|
||||
directory-only patterns, anchoring by a leading or embedded ``/``, ``*``,
|
||||
``?``, ``[...]`` and ``**``. As in git, a file under an ignored directory
|
||||
is ignored regardless of later negations.
|
||||
"""
|
||||
|
||||
def __init__(self, lines: Iterable[str]):
|
||||
self._rules: List[Tuple[Pattern[str], bool, bool]] = []
|
||||
for raw in lines:
|
||||
line = raw.rstrip('\n').rstrip()
|
||||
if not line or line.startswith('#'):
|
||||
continue
|
||||
negate = line.startswith('!')
|
||||
if negate:
|
||||
line = line[1:]
|
||||
elif line.startswith('\\'):
|
||||
line = line[1:]
|
||||
dir_only = line.endswith('/')
|
||||
line = line.rstrip('/')
|
||||
if not line:
|
||||
continue
|
||||
anchored = '/' in line
|
||||
line = line.lstrip('/')
|
||||
body = self._translate(line)
|
||||
regex = body if anchored else r'(?:.*/)?' + body
|
||||
self._rules.append((re.compile(r'\A' + regex + r'\Z'), negate, dir_only))
|
||||
|
||||
@staticmethod
|
||||
def _translate(pattern: str) -> str:
|
||||
out, i, n = [], 0, len(pattern)
|
||||
while i < n:
|
||||
if pattern.startswith('**/', i):
|
||||
out.append(r'(?:.*/)?')
|
||||
i += 3
|
||||
elif pattern.startswith('/**', i) and i + 3 == n:
|
||||
out.append(r'/.*')
|
||||
i += 3
|
||||
elif pattern.startswith('**', i):
|
||||
out.append(r'.*')
|
||||
i += 2
|
||||
elif pattern[i] == '*':
|
||||
out.append(r'[^/]*')
|
||||
i += 1
|
||||
elif pattern[i] == '?':
|
||||
out.append(r'[^/]')
|
||||
i += 1
|
||||
elif pattern[i] == '[':
|
||||
end = pattern.find(']', i + 1)
|
||||
if end == -1:
|
||||
out.append(re.escape('['))
|
||||
i += 1
|
||||
else:
|
||||
cls = pattern[i + 1:end]
|
||||
if cls.startswith('!'):
|
||||
cls = '^' + cls[1:]
|
||||
out.append('[' + cls.replace('\\', '\\\\') + ']')
|
||||
i = end + 1
|
||||
else:
|
||||
out.append(re.escape(pattern[i]))
|
||||
i += 1
|
||||
return ''.join(out)
|
||||
|
||||
def _decide(self, rel: str, is_dir: bool) -> Optional[bool]:
|
||||
verdict = None
|
||||
for regex, negate, dir_only in self._rules:
|
||||
if dir_only and not is_dir:
|
||||
continue
|
||||
if regex.match(rel):
|
||||
verdict = not negate
|
||||
return verdict
|
||||
|
||||
def ignores(self, rel_path: str) -> bool:
|
||||
if not self._rules:
|
||||
return False
|
||||
parts = rel_path.replace('\\', '/').split('/')
|
||||
for depth in range(1, len(parts)):
|
||||
if self._decide('/'.join(parts[:depth]), True):
|
||||
return True
|
||||
return bool(self._decide('/'.join(parts), False))
|
||||
|
||||
|
||||
def _read_gitignore(plugin_dir: Path) -> List[str]:
|
||||
try:
|
||||
return (plugin_dir / '.gitignore').read_text(
|
||||
encoding='utf-8', errors='replace').splitlines()
|
||||
except OSError:
|
||||
return []
|
||||
|
||||
|
||||
def local_files_to_keep(old_dir: Path, new_dir: Path) -> List[str]:
|
||||
"""Relative paths (``/``-separated) in ``old_dir`` to copy into ``new_dir``.
|
||||
|
||||
Regular files only; symlinks and anything the new release already ships
|
||||
are skipped.
|
||||
"""
|
||||
old_dir, new_dir = Path(old_dir), Path(new_dir)
|
||||
ignore = _GitIgnore(_read_gitignore(old_dir) + _read_gitignore(new_dir))
|
||||
keep: List[str] = []
|
||||
for root, dirs, files in os.walk(old_dir):
|
||||
dirs[:] = sorted(d for d in dirs if d not in _NEVER_CARRY_DIRS
|
||||
and not os.path.islink(os.path.join(root, d)))
|
||||
rel_root = os.path.relpath(root, old_dir)
|
||||
for name in sorted(files):
|
||||
if name.endswith(_NEVER_CARRY_SUFFIXES):
|
||||
continue
|
||||
full = os.path.join(root, name)
|
||||
if os.path.islink(full) or not os.path.isfile(full):
|
||||
continue
|
||||
rel = name if rel_root == '.' else f"{rel_root}/{name}".replace('\\', '/')
|
||||
if not (is_known_state_file(rel) or ignore.ignores(rel)):
|
||||
continue
|
||||
if os.path.lexists(new_dir / rel):
|
||||
continue
|
||||
keep.append(rel)
|
||||
return keep
|
||||
|
||||
|
||||
def carry_over_local_files(
|
||||
old_dir: Path, new_dir: Path
|
||||
) -> Tuple[List[str], List[Tuple[str, str]]]:
|
||||
"""Copy the plugin's local files from ``old_dir`` into ``new_dir``.
|
||||
|
||||
Copies rather than moves, so ``old_dir`` stays a complete copy until the
|
||||
caller deletes it. Returns ``(copied, failed)`` where ``failed`` pairs a
|
||||
relative path with the error; the caller should keep ``old_dir`` when
|
||||
anything failed.
|
||||
"""
|
||||
copied: List[str] = []
|
||||
failed: List[Tuple[str, str]] = []
|
||||
try:
|
||||
candidates = local_files_to_keep(old_dir, new_dir)
|
||||
except OSError as e:
|
||||
return copied, [('.', str(e))]
|
||||
for rel in candidates:
|
||||
dest = Path(new_dir) / rel
|
||||
try:
|
||||
dest.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copy2(Path(old_dir) / rel, dest)
|
||||
copied.append(rel)
|
||||
except OSError as e:
|
||||
failed.append((rel, str(e)))
|
||||
return copied, failed
|
||||
@@ -15,6 +15,7 @@ import sys
|
||||
import time
|
||||
import threading
|
||||
import types
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
from typing import Callable, Dict, List, NamedTuple, Optional, Any, Tuple, Union
|
||||
import logging
|
||||
@@ -32,6 +33,7 @@ from src.plugin_system.schema_manager import (
|
||||
from src.plugin_system.plugin_dirs import (
|
||||
ManifestStatus, PluginDirectoryIndex, resolve_plugin_dir,
|
||||
)
|
||||
from src.common.fetch_service import plugin_scope, register_plugin_directory
|
||||
from src.common.permission_utils import (
|
||||
ensure_directory_permissions,
|
||||
get_plugin_dir_mode
|
||||
@@ -69,6 +71,14 @@ class PluginManager:
|
||||
# before tearing the instance down anyway.
|
||||
UNLOAD_LOCK_TIMEOUT = 5.0
|
||||
|
||||
# How long unload_detached_plugin() (a live reload, off the render thread)
|
||||
# waits for the old instance's lock. Longer than UNLOAD_LOCK_TIMEOUT
|
||||
# because nothing is blocked by the wait, and a Vegas content build of the
|
||||
# old instance can hold the lock for several seconds (5.9 s seen on
|
||||
# ledpi). Past it the reload is refused rather than tearing down an
|
||||
# instance a Vegas call may still be running in.
|
||||
DETACHED_UNLOAD_LOCK_TIMEOUT = 30.0
|
||||
|
||||
# How long the update worker and apply_config_change() wait for a
|
||||
# plugin's lock -- the same bound unload already uses for the same lock.
|
||||
# A display() frame holds it for milliseconds, so this only runs out when
|
||||
@@ -150,7 +160,9 @@ class PluginManager:
|
||||
#
|
||||
# Which thread runs each plugin hook, and what it holds:
|
||||
# __init__, on_enable the loading thread (main thread at startup,
|
||||
# the render thread on a live enable).
|
||||
# the render thread on a live enable, a
|
||||
# plugin-reload thread for a control socket
|
||||
# reload).
|
||||
# update() plugin-update-worker, under the plugin lock,
|
||||
# via PluginExecutor (whose daemon thread runs
|
||||
# the call; if it outlives the executor's
|
||||
@@ -169,8 +181,10 @@ class PluginManager:
|
||||
# plugin lock via apply_config_change(); if the
|
||||
# lock stays busy it is deferred to the update
|
||||
# worker, which applies it under the lock.
|
||||
# cleanup(), on_disable() whoever calls unload_plugin(), under the
|
||||
# lock with UNLOAD_LOCK_TIMEOUT.
|
||||
# cleanup(), on_disable() whoever calls unload_plugin() (or, for a
|
||||
# reload, unload_detached_plugin() on its
|
||||
# plugin-reload thread), under the lock with
|
||||
# UNLOAD_LOCK_TIMEOUT.
|
||||
# No wait on a plugin lock is unbounded, so one hung plugin can only
|
||||
# cost the worker PLUGIN_LOCK_TIMEOUT per attempt.
|
||||
self._update_queue: "queue.Queue[Union[None, Tuple[str, float], _DeferredConfigChange]]" = queue.Queue()
|
||||
@@ -201,6 +215,9 @@ class PluginManager:
|
||||
# add_update_listener(). A tuple, replaced rather than mutated, so the
|
||||
# worker can iterate it without a lock.
|
||||
self._update_listeners: Tuple[Callable[[str], None], ...] = ()
|
||||
# Where plugins' on-demand requests go: the display controller's
|
||||
# submit_plugin_on_demand. See set_on_demand_handler().
|
||||
self._on_demand_handler: Optional[Callable[[Dict[str, Any]], bool]] = None
|
||||
# Config changes that found the plugin's lock busy, latest per plugin,
|
||||
# with the instance they were meant for. See apply_config_change().
|
||||
self._deferred_config_changes: Dict[str, Tuple[Any, Dict[str, Any]]] = {}
|
||||
@@ -423,6 +440,11 @@ class PluginManager:
|
||||
# Update mapping if found via search
|
||||
if plugin_id not in self.plugin_directories:
|
||||
self.plugin_directories[plugin_id] = plugin_dir
|
||||
|
||||
# Code under this directory is this plugin's: the fetch service
|
||||
# counts a request against it even from a thread the plugin
|
||||
# started itself (src/common/fetch_service.py, caller identity).
|
||||
register_plugin_directory(plugin_id, plugin_dir)
|
||||
|
||||
# Get plugin config
|
||||
if self.config_manager:
|
||||
@@ -462,18 +484,20 @@ class PluginManager:
|
||||
config = dict(config)
|
||||
config['enabled'] = True
|
||||
|
||||
# Use PluginLoader to load plugin
|
||||
plugin_instance, _module = self.plugin_loader.load_plugin(
|
||||
plugin_id=plugin_id,
|
||||
manifest=manifest,
|
||||
plugin_dir=plugin_dir,
|
||||
config=config,
|
||||
display_manager=self.display_manager,
|
||||
cache_manager=self.cache_manager,
|
||||
plugin_manager=self,
|
||||
install_deps=True,
|
||||
plugins_dir=self.plugins_dir,
|
||||
)
|
||||
# Use PluginLoader to load plugin. Fetches the constructor makes
|
||||
# count against the plugin.
|
||||
with plugin_scope(plugin_id):
|
||||
plugin_instance, _module = self.plugin_loader.load_plugin(
|
||||
plugin_id=plugin_id,
|
||||
manifest=manifest,
|
||||
plugin_dir=plugin_dir,
|
||||
config=config,
|
||||
display_manager=self.display_manager,
|
||||
cache_manager=self.cache_manager,
|
||||
plugin_manager=self,
|
||||
install_deps=True,
|
||||
plugins_dir=self.plugins_dir,
|
||||
)
|
||||
|
||||
# Register plugin-shipped fonts with the FontManager (if any).
|
||||
# Plugin manifests can declare a "fonts" block that ships custom
|
||||
@@ -527,7 +551,8 @@ class PluginManager:
|
||||
# Call on_enable if plugin is enabled
|
||||
if hasattr(plugin_instance, 'on_enable'):
|
||||
try:
|
||||
plugin_instance.on_enable()
|
||||
with plugin_scope(plugin_id):
|
||||
plugin_instance.on_enable()
|
||||
except Exception:
|
||||
# Undo the registration above before the outer
|
||||
# handler marks it ERROR: left in self.plugins, the
|
||||
@@ -576,11 +601,21 @@ class PluginManager:
|
||||
self.plugin_loader.unregister_plugin_modules(plugin_id)
|
||||
except Exception as e: # pragma: no cover - defensive
|
||||
self.logger.debug("Could not drop modules of %s: %s", plugin_id, e)
|
||||
try:
|
||||
if self.font_manager is not None and hasattr(self.font_manager, 'forget_manager_fonts'):
|
||||
self.font_manager.forget_manager_fonts(plugin_id)
|
||||
except Exception as e:
|
||||
self.logger.debug("Could not forget fonts of %s: %s", plugin_id, e)
|
||||
self._forget_plugin_fonts(plugin_id)
|
||||
|
||||
def _forget_plugin_fonts(self, plugin_id: str) -> None:
|
||||
"""Drop what the FontManager holds for a plugin: the fonts its
|
||||
instance reported using (the Fonts tab's "Used by") and the fonts its
|
||||
manifest registered. Never raises."""
|
||||
if self.font_manager is None:
|
||||
return
|
||||
for name in ('forget_manager_fonts', 'forget_plugin_fonts'):
|
||||
if not hasattr(self.font_manager, name):
|
||||
continue
|
||||
try:
|
||||
getattr(self.font_manager, name)(plugin_id)
|
||||
except Exception as e:
|
||||
self.logger.debug("Could not forget fonts of %s (%s): %s", plugin_id, name, e)
|
||||
|
||||
#: Config keys the **core** reads out of a plugin's own config block. The
|
||||
#: plugin never declares them, so a schema with
|
||||
@@ -743,14 +778,57 @@ class PluginManager:
|
||||
if lock_acquired:
|
||||
lock.release()
|
||||
|
||||
def _unload_plugin_locked(self, plugin_id: str) -> bool:
|
||||
"""Body of unload_plugin(); caller holds (or gave up on) the plugin lock."""
|
||||
if plugin_id not in self.plugins: # unloaded while we waited
|
||||
def detach_plugin(self, plugin_id: str) -> Optional[Any]:
|
||||
"""Take a loaded plugin out of ``plugins`` without tearing it down.
|
||||
|
||||
The first half of a reload that must not block its caller, the render
|
||||
thread (DisplayController._start_plugin_reload). Every new call into a
|
||||
plugin starts by looking it up in ``plugins``: the update scheduler,
|
||||
the update worker (which looks again under the plugin's lock) and
|
||||
Vegas's fetches. So once detached, nothing new reaches the instance.
|
||||
Work already running on it under its lock -- an update(), or a Vegas
|
||||
content render that can take seconds -- carries on;
|
||||
unload_detached_plugin() waits for it, on another thread.
|
||||
|
||||
Returns the instance, or None when the plugin was not loaded.
|
||||
"""
|
||||
return self.plugins.pop(plugin_id, None)
|
||||
|
||||
def unload_detached_plugin(self, plugin_id: str, plugin: Any) -> bool:
|
||||
"""Tear down an instance taken out by detach_plugin(): unload_plugin()
|
||||
for an instance that is no longer in ``plugins``.
|
||||
|
||||
Waits for the plugin's lock, bounded by DETACHED_UNLOAD_LOCK_TIMEOUT,
|
||||
so it belongs off the render thread. Call it before loading the plugin
|
||||
again: it drops the plugin's modules and lifecycle state along with the
|
||||
instance. Unlike unload_plugin() it never tears down without the lock:
|
||||
a call that took the lock before the detach (a Vegas content build)
|
||||
may still be running in this instance. Returns False then, and the
|
||||
caller must not load the plugin again over it.
|
||||
"""
|
||||
lock = self.get_plugin_lock(plugin_id)
|
||||
if not lock.acquire(timeout=self.DETACHED_UNLOAD_LOCK_TIMEOUT):
|
||||
self.logger.warning(
|
||||
"Plugin %s still busy after %.1fs; not unloading it while in use",
|
||||
plugin_id, self.DETACHED_UNLOAD_LOCK_TIMEOUT)
|
||||
return False
|
||||
try:
|
||||
return self._unload_plugin_locked(plugin_id, plugin)
|
||||
finally:
|
||||
lock.release()
|
||||
|
||||
def _unload_plugin_locked(self, plugin_id: str, detached: Optional[Any] = None) -> bool:
|
||||
"""Body of unload_plugin(); caller holds (or gave up on) the plugin lock.
|
||||
|
||||
``detached`` is an instance already taken out of ``plugins``
|
||||
(detach_plugin); without it, the loaded instance is unloaded.
|
||||
"""
|
||||
if detached is None and plugin_id not in self.plugins: # unloaded while we waited
|
||||
self.logger.warning("Plugin %s not loaded", plugin_id)
|
||||
return False
|
||||
|
||||
try:
|
||||
plugin = self.plugins[plugin_id]
|
||||
plugin = self.plugins[plugin_id] if detached is None else detached
|
||||
|
||||
# Call cleanup if available
|
||||
if hasattr(plugin, 'cleanup'):
|
||||
@@ -766,8 +844,9 @@ class PluginManager:
|
||||
except Exception as e:
|
||||
self.logger.warning("Error during plugin on_disable: %s", e)
|
||||
|
||||
# Remove from active plugins
|
||||
del self.plugins[plugin_id]
|
||||
# Remove from active plugins (a detached one already is)
|
||||
if detached is None:
|
||||
del self.plugins[plugin_id]
|
||||
with self._deferred_config_lock:
|
||||
self._deferred_config_changes.pop(plugin_id, None)
|
||||
with self._plugin_last_update_lock:
|
||||
@@ -781,12 +860,9 @@ class PluginManager:
|
||||
# Delegate sub-module and cached-module cleanup to the loader
|
||||
self.plugin_loader.unregister_plugin_modules(plugin_id)
|
||||
|
||||
# Its font registrations go with it (the Fonts tab's "Used by").
|
||||
try:
|
||||
if self.font_manager is not None and hasattr(self.font_manager, 'forget_manager_fonts'):
|
||||
self.font_manager.forget_manager_fonts(plugin_id)
|
||||
except Exception as e:
|
||||
self.logger.debug("Could not forget fonts of %s: %s", plugin_id, e)
|
||||
# Its font registrations go with it: the fonts it reported using
|
||||
# and the ones its manifest registered.
|
||||
self._forget_plugin_fonts(plugin_id)
|
||||
|
||||
# Update state
|
||||
self.state_manager.set_state(plugin_id, PluginState.UNLOADED)
|
||||
@@ -1098,7 +1174,7 @@ class PluginManager:
|
||||
def _record_update_failure(
|
||||
self,
|
||||
plugin_id: str,
|
||||
exc: Optional[Exception] = None,
|
||||
exc: Optional[BaseException] = None,
|
||||
log: bool = True,
|
||||
count_failure: bool = True,
|
||||
) -> None:
|
||||
@@ -1122,7 +1198,7 @@ class PluginManager:
|
||||
"""
|
||||
failure_time = time.time()
|
||||
if exc is not None:
|
||||
err: Exception = exc
|
||||
err: BaseException = exc
|
||||
error_type = type(exc).__name__
|
||||
else:
|
||||
err = Exception(f"Plugin {plugin_id} execution failed (timeout or executor error)")
|
||||
@@ -1588,7 +1664,7 @@ class PluginManager:
|
||||
finish_guard = threading.Lock()
|
||||
finished = {'done': False}
|
||||
|
||||
def _finish(success: bool, exc: Optional[Exception] = None) -> None:
|
||||
def _finish(success: bool, exc: Optional[BaseException] = None) -> None:
|
||||
with finish_guard:
|
||||
if finished['done']:
|
||||
return
|
||||
@@ -1662,7 +1738,13 @@ class PluginManager:
|
||||
self.resource_monitor.monitor_call(plugin_id, plugin_instance.update)
|
||||
else:
|
||||
plugin_instance.update()
|
||||
except Exception as exc:
|
||||
except BaseException as exc: # pylint: disable=broad-except
|
||||
# BaseException, not just Exception: asyncio.CancelledError
|
||||
# and SystemExit derive from it. Either one skipped _finish,
|
||||
# so the plugin kept its lock and stayed RUNNING for good --
|
||||
# never rescheduled, and every display() skipped as busy.
|
||||
# Re-raised for the executor, which reports it as this
|
||||
# update's failure.
|
||||
_finish(False, exc=exc)
|
||||
raise
|
||||
else:
|
||||
@@ -1773,3 +1855,73 @@ class PluginManager:
|
||||
done = sorted(self._completed_updates)
|
||||
self._completed_updates.clear()
|
||||
return done
|
||||
|
||||
# -- on-demand requests from plugins -------------------------------------
|
||||
|
||||
def set_on_demand_handler(
|
||||
self, handler: Optional[Callable[[Dict[str, Any]], bool]]) -> None:
|
||||
"""Route plugins' on-demand requests to ``handler`` (None: nowhere).
|
||||
|
||||
The display controller sets its ``submit_plugin_on_demand`` here
|
||||
before any plugin loads. The handler takes a mailbox-shaped request
|
||||
from any thread, queues it for the render thread and returns True,
|
||||
or False when it could not. A plugin manager with no handler (the
|
||||
web interface's, a test's, scripts/check_plugin.py's) has no screen
|
||||
to give, so request_on_demand() there answers None.
|
||||
"""
|
||||
self._on_demand_handler = handler
|
||||
|
||||
def request_on_demand(self, plugin_id: str, mode: Optional[str] = None,
|
||||
duration: Optional[float] = None,
|
||||
pinned: bool = False) -> Optional[str]:
|
||||
"""Ask the display to show ``plugin_id`` now. Safe from any thread.
|
||||
|
||||
BasePlugin.request_on_demand() lands here; see it for the arguments.
|
||||
Returns the request id once the display has queued the request (it
|
||||
is applied on the render thread within a frame or so), or None when
|
||||
this process has no display to ask or its queue is full.
|
||||
"""
|
||||
if not isinstance(plugin_id, str) or not plugin_id:
|
||||
raise ValueError('plugin_id is required')
|
||||
if mode is not None and (not isinstance(mode, str) or not mode):
|
||||
raise ValueError('mode must be a non-empty string or None')
|
||||
if duration is not None:
|
||||
if isinstance(duration, bool) or not isinstance(duration, (int, float)):
|
||||
raise ValueError('duration must be a number of seconds or None')
|
||||
if not math.isfinite(duration) or duration <= 0:
|
||||
duration = None # the display reads these as "no limit" too
|
||||
else:
|
||||
duration = float(duration)
|
||||
return self._submit_on_demand({
|
||||
'action': 'start', 'plugin_id': plugin_id, 'mode': mode,
|
||||
'duration': duration, 'pinned': bool(pinned)})
|
||||
|
||||
def end_on_demand(self, plugin_id: str) -> Optional[str]:
|
||||
"""Give the screen back, if ``plugin_id``'s on-demand session has it.
|
||||
|
||||
BasePlugin.end_on_demand() lands here. A session the plugin does not
|
||||
own (the user started another plugin from the web interface, say) is
|
||||
left alone. Returns the request id once queued, or None as
|
||||
request_on_demand() does.
|
||||
"""
|
||||
if not isinstance(plugin_id, str) or not plugin_id:
|
||||
raise ValueError('plugin_id is required')
|
||||
return self._submit_on_demand({'action': 'stop', 'plugin_id': plugin_id})
|
||||
|
||||
def _submit_on_demand(self, request: Dict[str, Any]) -> Optional[str]:
|
||||
# __dict__.get: tests build bare managers with PluginManager.__new__.
|
||||
handler = self.__dict__.get('_on_demand_handler')
|
||||
if handler is None:
|
||||
return None
|
||||
request_id = str(uuid.uuid4())
|
||||
request.update({'request_id': request_id, 'timestamp': time.time(),
|
||||
'source': 'plugin'})
|
||||
try:
|
||||
accepted = handler(request)
|
||||
except Exception as exc: # pylint: disable=broad-except
|
||||
self._warn_rate_limited(
|
||||
"on-demand-handler",
|
||||
"The on-demand request from plugin %s failed: %r",
|
||||
request.get('plugin_id'), exc)
|
||||
return None
|
||||
return request_id if accepted else None
|
||||
|
||||
@@ -25,15 +25,40 @@ snapshot behind. A display that stops cleanly publishes ``running: false``
|
||||
on the way out, so readers see "stopped" at once rather than after the
|
||||
stale window. Nothing on the reading side reports a runtime fact from a
|
||||
snapshot that is not live.
|
||||
|
||||
Render-loop liveness. The snapshot is written from its own thread, which
|
||||
keeps going when the render loop hangs inside a plugin. So the reader also
|
||||
checks the render loop's heartbeat (``src/display_watchdog.py``, the file
|
||||
``/api/v3/health`` reports as ``checks.display_loop``): a live snapshot from
|
||||
the process whose heartbeat has gone stale is ``stalled``, as the health
|
||||
check says, not ``live``. No extra writes: the heartbeat already exists, on
|
||||
tmpfs. A missing heartbeat (dev server, emulator, Windows, a display still
|
||||
starting up) or one from another process (a display restarted after a
|
||||
watchdog kill) says nothing, and the snapshot is judged on its own.
|
||||
|
||||
The control socket. Where the display serves its state stream (stage 3,
|
||||
docs/IPC_CONTROL_SOCKET.md), every tick also hands the snapshot to it, in
|
||||
memory, and the web interface reads it there first
|
||||
(``view_from_socket_state``, judged by the same rules). While the socket
|
||||
serves those readers, the cache copy is their fallback and an unchanged
|
||||
snapshot is rewritten every ``RELAXED_REFRESH_INTERVAL`` instead.
|
||||
|
||||
A dead publisher. systemd removes the heartbeat's directory when the
|
||||
service stops, so after a watchdog kill there is no heartbeat to go stale.
|
||||
The reader then asks whether the snapshot's ``pid`` still exists (POSIX
|
||||
``kill(pid, 0)``, which sends nothing): a running snapshot from a process
|
||||
that is gone is ``stale`` at once rather than ``live`` for the rest of its
|
||||
``stale_after`` window.
|
||||
"""
|
||||
|
||||
import math
|
||||
import os
|
||||
import threading
|
||||
import time
|
||||
from dataclasses import dataclass, field
|
||||
from dataclasses import dataclass, field, replace
|
||||
from typing import Any, Callable, Dict, Optional
|
||||
|
||||
from src import display_watchdog
|
||||
from src.logging_config import get_logger
|
||||
from src.redaction import redact_credentials
|
||||
|
||||
@@ -57,6 +82,15 @@ TICK_INTERVAL = 5.0
|
||||
#: A snapshot older than this is stale: three missed refreshes.
|
||||
STALE_AFTER = 3 * REFRESH_INTERVAL
|
||||
|
||||
#: The refresh while the control socket serves the web interface's readers
|
||||
#: (``StateHub.readers_active``). The cache copy is then only their fallback,
|
||||
#: so an unchanged snapshot is rewritten half as often; the snapshot says so
|
||||
#: in its own ``refresh_interval`` and ``stale_after``.
|
||||
RELAXED_REFRESH_INTERVAL = 2 * REFRESH_INTERVAL
|
||||
|
||||
#: The control socket's section for this snapshot (``state.plugins``).
|
||||
STATE_SECTION = "plugins"
|
||||
|
||||
#: Bounds on a published ``stale_after``, so a corrupt value can make a
|
||||
#: reader neither trust a dead display for hours nor distrust a live one.
|
||||
_STALE_AFTER_MIN = 30.0
|
||||
@@ -70,6 +104,9 @@ _VERSION_CHARS = 40
|
||||
#: Reader statuses. Only LIVE carries runtime facts.
|
||||
LIVE = "live"
|
||||
STALE = "stale"
|
||||
#: The snapshot is fresh but the render loop's heartbeat is not: the display
|
||||
#: is hung (or was just killed by the watchdog), as /api/v3/health reports.
|
||||
STALLED = "stalled"
|
||||
STOPPED = "stopped"
|
||||
UNKNOWN = "unknown"
|
||||
|
||||
@@ -119,7 +156,8 @@ def summarize_error(error_info: Optional[Dict[str, Any]]) -> Optional[Dict[str,
|
||||
|
||||
def build_runtime_snapshot(state_manager: Any, *, started_at: float,
|
||||
now: Optional[float] = None,
|
||||
running: bool = True) -> Dict[str, Any]:
|
||||
running: bool = True,
|
||||
refresh_interval: float = REFRESH_INTERVAL) -> Dict[str, Any]:
|
||||
"""The snapshot for ``state_manager`` (a plugin_state.PluginStateManager).
|
||||
|
||||
A stopped snapshot (``running=False``) lists no plugins: nothing is
|
||||
@@ -141,8 +179,8 @@ def build_runtime_snapshot(state_manager: Any, *, started_at: float,
|
||||
"running": running,
|
||||
"published_at": time.time() if now is None else now,
|
||||
"started_at": started_at,
|
||||
"refresh_interval": REFRESH_INTERVAL,
|
||||
"stale_after": STALE_AFTER,
|
||||
"refresh_interval": refresh_interval,
|
||||
"stale_after": 3 * refresh_interval,
|
||||
"pid": os.getpid(),
|
||||
"plugins": plugins,
|
||||
}
|
||||
@@ -176,30 +214,87 @@ class PluginRuntimePublisher:
|
||||
self._tick_lock = threading.Lock()
|
||||
self._stop = threading.Event()
|
||||
self._thread: Optional[threading.Thread] = None
|
||||
# The control socket's state stream (src/ipc/server.StateHub), when
|
||||
# the display serves one: every tick also hands it the snapshot, in
|
||||
# memory, and the cache refresh relaxes while it has readers.
|
||||
self._hub: Any = None
|
||||
self._hub_change: Optional[int] = None
|
||||
self._hub_snapshot: Optional[Dict[str, Any]] = None
|
||||
self.relaxed_refresh_interval = RELAXED_REFRESH_INTERVAL
|
||||
|
||||
def _write(self, running: bool) -> None:
|
||||
snapshot = build_runtime_snapshot(self.state_manager, started_at=self.started_at,
|
||||
now=self._wall_clock(), running=running)
|
||||
def attach_hub(self, hub: Any) -> None:
|
||||
"""Also publish to the control socket's state hub, starting now."""
|
||||
with self._tick_lock:
|
||||
self._hub = hub
|
||||
self._hub_change = None
|
||||
self._hub_snapshot = None
|
||||
try:
|
||||
self._push_to_hub(self.state_manager.change_count)
|
||||
except Exception as err: # never let reporting break the display
|
||||
logger.debug("Could not publish the plugin runtime state: %s", err,
|
||||
exc_info=True)
|
||||
|
||||
def _push_to_hub(self, change: int) -> None:
|
||||
"""The snapshot to the state hub: rebuilt when the state machine
|
||||
changed, otherwise the last one with a new ``published_at``, which
|
||||
the hub does not count as a new version. In memory, every tick, so
|
||||
the socket's copy is never more than a tick old."""
|
||||
hub = self._hub
|
||||
if hub is None:
|
||||
return
|
||||
now = self._wall_clock()
|
||||
if self._hub_snapshot is None or change != self._hub_change:
|
||||
snapshot = build_runtime_snapshot(self.state_manager, started_at=self.started_at,
|
||||
now=now)
|
||||
else:
|
||||
snapshot = dict(self._hub_snapshot, published_at=now)
|
||||
hub.publish(STATE_SECTION, snapshot, volatile=("published_at",))
|
||||
self._hub_snapshot = snapshot
|
||||
self._hub_change = change
|
||||
|
||||
def _cache_refresh_interval(self) -> float:
|
||||
"""The cache refresh: relaxed while the socket serves the readers."""
|
||||
hub = self._hub
|
||||
try:
|
||||
if hub is not None and hub.readers_active():
|
||||
return self.relaxed_refresh_interval
|
||||
except Exception: # pylint: disable=broad-except
|
||||
# The normal interval is the safe answer: it only writes more.
|
||||
logger.debug("State hub readers_active() failed; using the normal refresh", exc_info=True)
|
||||
return self.refresh_interval
|
||||
|
||||
def _write(self, running: bool, refresh_interval: Optional[float] = None) -> None:
|
||||
snapshot = build_runtime_snapshot(
|
||||
self.state_manager, started_at=self.started_at, now=self._wall_clock(),
|
||||
running=running,
|
||||
refresh_interval=self.refresh_interval if refresh_interval is None
|
||||
else refresh_interval)
|
||||
self.cache_manager.set(PLUGIN_RUNTIME_KEY, snapshot)
|
||||
|
||||
def tick(self) -> bool:
|
||||
"""Publish if something changed (throttled) or the refresh is due.
|
||||
True if a snapshot was written."""
|
||||
True if a snapshot was written to the cache."""
|
||||
with self._tick_lock:
|
||||
try:
|
||||
change = self.state_manager.change_count
|
||||
try:
|
||||
self._push_to_hub(change)
|
||||
except Exception as err: # the cache copy still goes out below
|
||||
logger.debug("Could not publish the plugin runtime state: %s", err,
|
||||
exc_info=True)
|
||||
now = self._clock()
|
||||
refresh = self._cache_refresh_interval()
|
||||
since = None if self._last_attempt is None else now - self._last_attempt
|
||||
if since is not None:
|
||||
if change == self._published_change:
|
||||
if since < self.refresh_interval:
|
||||
if since < refresh:
|
||||
return False
|
||||
elif since < self.min_interval:
|
||||
return False
|
||||
# Stamp the attempt before writing: a cache that keeps failing
|
||||
# is retried at the throttled rate, not on every tick.
|
||||
self._last_attempt = now
|
||||
self._write(running=True)
|
||||
self._write(running=True, refresh_interval=refresh)
|
||||
self._published_change = change
|
||||
return True
|
||||
except Exception as err: # never let reporting break the display
|
||||
@@ -281,7 +376,9 @@ class PluginRuntimeView:
|
||||
|
||||
``status``: ``live`` (a fresh snapshot from a running display),
|
||||
``stale`` (the last snapshot is older than its ``stale_after``: the
|
||||
display is hung or died without cleaning up), ``stopped`` (the display
|
||||
display is hung or died without cleaning up), ``stalled`` (the snapshot
|
||||
is fresh but the same process's render-loop heartbeat is stale: the
|
||||
render loop is hung), ``stopped`` (the display
|
||||
said so on its way out) or ``unknown`` (no readable snapshot). Only a
|
||||
live view reports per-plugin facts; every other status answers None for
|
||||
them, so a caller cannot pass stale truth on by accident.
|
||||
@@ -292,6 +389,11 @@ class PluginRuntimeView:
|
||||
age_seconds: Optional[float] = None
|
||||
stale_after: float = STALE_AFTER
|
||||
plugins: Dict[str, Dict[str, Any]] = field(default_factory=dict)
|
||||
#: Age of the render loop's heartbeat, when it was taken into account.
|
||||
heartbeat_age_seconds: Optional[float] = None
|
||||
#: Where the snapshot came from: ``cache`` (the shared cache file and the
|
||||
#: heartbeat file) or ``socket`` (the control socket's state stream).
|
||||
source: str = "cache"
|
||||
|
||||
@property
|
||||
def live(self) -> bool:
|
||||
@@ -321,6 +423,9 @@ class PluginRuntimeView:
|
||||
"published_at": self.published_at,
|
||||
"age_seconds": None if self.age_seconds is None else round(self.age_seconds, 1),
|
||||
"stale_after": self.stale_after,
|
||||
"heartbeat_age_seconds": (None if self.heartbeat_age_seconds is None
|
||||
else round(self.heartbeat_age_seconds, 1)),
|
||||
"source": self.source,
|
||||
}
|
||||
|
||||
|
||||
@@ -334,8 +439,56 @@ def _stale_after_of(snapshot: Dict[str, Any]) -> float:
|
||||
return min(max(number, _STALE_AFTER_MIN), _STALE_AFTER_MAX)
|
||||
|
||||
|
||||
def view_from_snapshot(snapshot: Any, now: Optional[float] = None) -> PluginRuntimeView:
|
||||
"""Judge a snapshot read from the cache; never raises."""
|
||||
def _heartbeat_age_for(snapshot: Dict[str, Any], heartbeat: Any,
|
||||
now_mono: Optional[float]) -> Optional[float]:
|
||||
"""Age of ``heartbeat`` if it comes from the process that published
|
||||
``snapshot``; None when there is none, it has no time, or it belongs to
|
||||
another process (a restarted display, or a heartbeat left by a killed one)."""
|
||||
if not isinstance(heartbeat, dict):
|
||||
return None
|
||||
beat_pid = heartbeat.get("pid")
|
||||
snap_pid = snapshot.get("pid")
|
||||
if (isinstance(beat_pid, bool) or not isinstance(beat_pid, int)
|
||||
or isinstance(snap_pid, bool) or not isinstance(snap_pid, int)
|
||||
or beat_pid != snap_pid):
|
||||
return None
|
||||
return display_watchdog.heartbeat_age(heartbeat, now_mono=now_mono)
|
||||
|
||||
|
||||
def process_exists(pid: int) -> Optional[bool]:
|
||||
"""Whether process ``pid`` exists: True, False, or None when this
|
||||
platform cannot tell. POSIX only -- on Windows ``os.kill`` terminates.
|
||||
Signal 0 sends nothing; EPERM (the display runs as root, the web
|
||||
interface does not) still means the process is there."""
|
||||
if os.name != "posix" or pid <= 0:
|
||||
return None
|
||||
try:
|
||||
os.kill(pid, 0)
|
||||
except ProcessLookupError:
|
||||
return False
|
||||
except PermissionError:
|
||||
return True
|
||||
except OSError:
|
||||
return None
|
||||
return True
|
||||
|
||||
|
||||
def view_from_snapshot(snapshot: Any, now: Optional[float] = None,
|
||||
heartbeat: Any = None,
|
||||
now_mono: Optional[float] = None,
|
||||
process_alive: Optional[Callable[[int], Optional[bool]]] = None,
|
||||
) -> PluginRuntimeView:
|
||||
"""Judge a snapshot read from the cache; never raises.
|
||||
|
||||
``heartbeat`` is the render loop's heartbeat
|
||||
(``display_watchdog.read_heartbeat()``), or None when there is none. A
|
||||
live snapshot whose process's heartbeat is at least
|
||||
``display_watchdog.HEARTBEAT_STALE_SECONDS`` old is ``stalled``: the
|
||||
threshold /api/v3/health uses for ``checks.display_loop``.
|
||||
``process_alive`` (``process_exists`` when reading the real cache) says
|
||||
whether the snapshot's publisher still exists; a running snapshot from
|
||||
one that is gone is ``stale``.
|
||||
"""
|
||||
if not isinstance(snapshot, dict) or snapshot.get("schema") != SNAPSHOT_SCHEMA:
|
||||
return PluginRuntimeView(status=UNKNOWN)
|
||||
published_at = _epoch(snapshot.get("published_at"))
|
||||
@@ -351,21 +504,64 @@ def view_from_snapshot(snapshot: Any, now: Optional[float] = None) -> PluginRunt
|
||||
if age > stale_after or age < -stale_after:
|
||||
return PluginRuntimeView(status=STALE, published_at=published_at,
|
||||
age_seconds=age, stale_after=stale_after)
|
||||
pid = snapshot.get("pid")
|
||||
if (process_alive is not None and isinstance(pid, int) and not isinstance(pid, bool)
|
||||
and process_alive(pid) is False):
|
||||
return PluginRuntimeView(status=STALE, published_at=published_at,
|
||||
age_seconds=max(age, 0.0), stale_after=stale_after)
|
||||
beat_age = _heartbeat_age_for(snapshot, heartbeat, now_mono)
|
||||
if beat_age is not None and beat_age >= display_watchdog.HEARTBEAT_STALE_SECONDS:
|
||||
return PluginRuntimeView(status=STALLED, published_at=published_at,
|
||||
age_seconds=max(age, 0.0), stale_after=stale_after,
|
||||
heartbeat_age_seconds=beat_age)
|
||||
plugins = snapshot.get("plugins")
|
||||
return PluginRuntimeView(
|
||||
status=LIVE, published_at=published_at, age_seconds=max(age, 0.0),
|
||||
stale_after=stale_after,
|
||||
stale_after=stale_after, heartbeat_age_seconds=beat_age,
|
||||
plugins={k: v for k, v in plugins.items() if isinstance(v, dict)}
|
||||
if isinstance(plugins, dict) else {},
|
||||
)
|
||||
|
||||
|
||||
def read_plugin_runtime(cache_manager: Any, now: Optional[float] = None) -> PluginRuntimeView:
|
||||
"""The display's latest snapshot, judged for staleness. Never raises; a
|
||||
missing cache manager or an unreadable snapshot is ``unknown``.
|
||||
def view_from_socket_state(snapshot: Any, now: Optional[float] = None,
|
||||
now_mono: Optional[float] = None) -> Optional[PluginRuntimeView]:
|
||||
"""Judge the ``plugins`` section of a control-socket state snapshot by
|
||||
the same rules as the cache copy; None when it has none (an older
|
||||
display, or a snapshot too large to carry it), so the caller reads the
|
||||
cache instead.
|
||||
|
||||
The display measured its render loop's heartbeat age when it answered
|
||||
(``state.loop``); that is the heartbeat here, aged by the time since the
|
||||
answer arrived. A live snapshot with a stalled loop is ``stalled``, and a
|
||||
snapshot older than its ``stale_after`` (the publisher thread stopped)
|
||||
is ``stale``, exactly as for the cache. The display answered, so its
|
||||
process is alive: there is no pid check.
|
||||
"""
|
||||
from src.ipc.client import snapshot_loop_age # stdlib-only module
|
||||
if not isinstance(snapshot, dict):
|
||||
return None
|
||||
state = snapshot.get("state")
|
||||
plugins = state.get(STATE_SECTION) if isinstance(state, dict) else None
|
||||
if not isinstance(plugins, dict):
|
||||
return None
|
||||
now_mono = time.monotonic() if now_mono is None else now_mono
|
||||
beat_age = snapshot_loop_age(snapshot, now_mono=now_mono)
|
||||
heartbeat = None
|
||||
if beat_age is not None:
|
||||
heartbeat = {"pid": plugins.get("pid"), "mono": now_mono - beat_age}
|
||||
view = view_from_snapshot(plugins, now=now, heartbeat=heartbeat, now_mono=now_mono)
|
||||
return replace(view, source="socket")
|
||||
|
||||
|
||||
def read_plugin_runtime(cache_manager: Any, now: Optional[float] = None,
|
||||
heartbeat_path: Optional[str] = None) -> PluginRuntimeView:
|
||||
"""The display's latest snapshot, judged for staleness and against the
|
||||
render loop's heartbeat. Never raises; a missing cache manager or an
|
||||
unreadable snapshot is ``unknown``.
|
||||
|
||||
memory_ttl=0: the key is written by the other process, so only the file
|
||||
is current.
|
||||
is current. ``heartbeat_path`` defaults to
|
||||
``display_watchdog.HEARTBEAT_PATH``.
|
||||
"""
|
||||
if cache_manager is None:
|
||||
return PluginRuntimeView(status=UNKNOWN)
|
||||
@@ -374,4 +570,11 @@ def read_plugin_runtime(cache_manager: Any, now: Optional[float] = None) -> Plug
|
||||
except Exception as err:
|
||||
logger.debug("Could not read the plugin runtime snapshot: %s", err, exc_info=True)
|
||||
return PluginRuntimeView(status=UNKNOWN)
|
||||
return view_from_snapshot(snapshot, now=now)
|
||||
try:
|
||||
heartbeat = display_watchdog.read_heartbeat(
|
||||
heartbeat_path or display_watchdog.HEARTBEAT_PATH)
|
||||
except Exception as err: # read_heartbeat does not raise; belt and braces
|
||||
logger.debug("Could not read the display heartbeat: %s", err, exc_info=True)
|
||||
heartbeat = None
|
||||
return view_from_snapshot(snapshot, now=now, heartbeat=heartbeat,
|
||||
process_alive=process_exists)
|
||||
|
||||
@@ -8,7 +8,7 @@ Provides resource limits and performance monitoring.
|
||||
import math
|
||||
import time
|
||||
import threading
|
||||
from typing import Dict, Optional, Any, Callable, cast
|
||||
from typing import Dict, Optional, Any, Callable, Set, cast
|
||||
from dataclasses import dataclass, field, fields
|
||||
|
||||
from src.logging_config import get_logger
|
||||
@@ -99,18 +99,33 @@ class ResourceMetrics:
|
||||
last_update_time: float = field(default_factory=time.time)
|
||||
|
||||
|
||||
#: How often a plugin's metrics are written to the cache, in seconds.
|
||||
#: How often the metrics snapshot is written to the cache, in seconds.
|
||||
#:
|
||||
#: Persisting on every call meant a small file rewritten roughly nine times a
|
||||
#: minute per plugin. On a rig with fourteen active plugins that was ~126
|
||||
#: writes a minute for metrics alone, and since each ~350-byte file costs a
|
||||
#: 4KB block plus an ext4 journal entry, it dominated the device's write
|
||||
#: volume -- on an SD card, which wears out.
|
||||
#: volume -- on an SD card, which wears out. Throttling each plugin's own
|
||||
#: record to once per 30 s still left two writes a minute per plugin, so all
|
||||
#: plugins now share one record (METRICS_SNAPSHOT_KEY), written at most once
|
||||
#: a minute: one write a minute however many plugins there are.
|
||||
#:
|
||||
#: The in-memory copy stays authoritative and exact; only the cross-process
|
||||
#: snapshot the web UI reads is delayed, and telemetry up to half a minute old
|
||||
#: is still a fair description of a long-running plugin.
|
||||
_METRICS_PERSIST_INTERVAL = 30.0
|
||||
#: snapshot the web UI reads is delayed, and telemetry up to a minute old is
|
||||
#: still a fair description of a long-running plugin.
|
||||
_METRICS_PERSIST_INTERVAL = 60.0
|
||||
|
||||
#: The one cache record holding every plugin's metrics:
|
||||
#: ``{"schema": 1, "plugins": {plugin_id: <metrics record>}}``, each metrics
|
||||
#: record shaped as the per-plugin ``plugin_metrics:<id>`` records were. Those
|
||||
#: older records are still read for a plugin the snapshot does not have yet
|
||||
#: (an upgrade, or a plugin that has not run since), never written.
|
||||
METRICS_SNAPSHOT_KEY = "plugin_metrics_snapshot"
|
||||
_METRICS_SNAPSHOT_SCHEMA = 1
|
||||
|
||||
#: A plugin with no call for this long is dropped from the snapshot -- what
|
||||
#: the cache's 30-day default retention did to its own record before.
|
||||
_METRICS_SNAPSHOT_ENTRY_MAX_AGE = 30 * 86400
|
||||
|
||||
|
||||
class PluginResourceMonitor:
|
||||
@@ -140,10 +155,15 @@ class PluginResourceMonitor:
|
||||
self._metrics: Dict[str, ResourceMetrics] = {}
|
||||
self._limits: Dict[str, ResourceLimits] = {}
|
||||
self._bad_limits_warned: set = set()
|
||||
# When each plugin's metrics last reached the cache. Metrics change on
|
||||
# every call, so they cannot be de-duplicated the way health state can;
|
||||
# they are rate-limited instead. See _METRICS_PERSIST_INTERVAL.
|
||||
self._metrics_persisted_at: Dict[str, float] = {}
|
||||
# When the metrics snapshot last reached the cache (monotonic), None
|
||||
# until it has. Metrics change on every call, so they cannot be
|
||||
# de-duplicated the way health state can; they are rate-limited
|
||||
# instead. See _METRICS_PERSIST_INTERVAL.
|
||||
self._snapshot_persisted_at: Optional[float] = None
|
||||
# Plugins whose metrics this process recorded since the last snapshot
|
||||
# write: only their entries are overwritten, the rest are kept as
|
||||
# found on disk.
|
||||
self._metrics_dirty: Set[str] = set()
|
||||
|
||||
# Lock for thread-safe access
|
||||
self._lock = threading.Lock()
|
||||
@@ -247,10 +267,14 @@ class PluginResourceMonitor:
|
||||
with self._lock:
|
||||
if force_reload or plugin_id not in self._metrics:
|
||||
# Try to load from cache
|
||||
cache_key = self._get_metrics_key(plugin_id)
|
||||
cached = self.cache_manager.get(
|
||||
cache_key, max_age=None, memory_ttl=0 if force_reload else None
|
||||
)
|
||||
memory_ttl = 0 if force_reload else None
|
||||
cached = self._read_snapshot(memory_ttl).get(plugin_id)
|
||||
if cached is None:
|
||||
# Not in the snapshot: the per-plugin record an older
|
||||
# version wrote, if there is one.
|
||||
cached = self.cache_manager.get(
|
||||
self._get_metrics_key(plugin_id), max_age=None,
|
||||
memory_ttl=memory_ttl)
|
||||
if cached:
|
||||
metrics = self._metrics_from_cache(plugin_id, cached)
|
||||
else:
|
||||
@@ -498,12 +522,70 @@ class PluginResourceMonitor:
|
||||
summaries[plugin_id] = self.get_metrics_summary(plugin_id)
|
||||
return summaries
|
||||
|
||||
def _persist_metrics(self, plugin_id: str, metrics: ResourceMetrics,
|
||||
force: bool = False) -> None:
|
||||
"""Write a plugin's metrics to the cache, at most once per interval.
|
||||
def _read_snapshot(self, memory_ttl: Optional[int] = None) -> Dict[str, Any]:
|
||||
"""The snapshot's per-plugin records, or {} if there is none usable.
|
||||
|
||||
Caller must hold ``self._lock``.
|
||||
"""
|
||||
cached = self.cache_manager.get(
|
||||
METRICS_SNAPSHOT_KEY, max_age=None, memory_ttl=memory_ttl)
|
||||
if not isinstance(cached, dict) or cached.get('schema') != _METRICS_SNAPSHOT_SCHEMA:
|
||||
return {}
|
||||
plugins = cached.get('plugins')
|
||||
return plugins if isinstance(plugins, dict) else {}
|
||||
|
||||
@staticmethod
|
||||
def _metrics_record(metrics: ResourceMetrics) -> Dict[str, Any]:
|
||||
"""One plugin's entry in the snapshot."""
|
||||
return {
|
||||
'memory_mb': metrics.memory_mb,
|
||||
'cpu_percent': metrics.cpu_percent,
|
||||
'execution_time': metrics.execution_time,
|
||||
'call_count': metrics.call_count,
|
||||
'total_execution_time': metrics.total_execution_time,
|
||||
'max_execution_time': metrics.max_execution_time,
|
||||
'min_execution_time': (metrics.min_execution_time
|
||||
if metrics.min_execution_time != float('inf')
|
||||
else 0.0),
|
||||
'last_update_time': metrics.last_update_time,
|
||||
}
|
||||
|
||||
def _write_snapshot(self, drop: Optional[str] = None) -> None:
|
||||
"""Write the snapshot: what is on disk, with this process's recorded
|
||||
plugins updated and ``drop`` removed.
|
||||
|
||||
Starting from the disk copy rather than from memory keeps the entries
|
||||
of plugins this process has not run -- disabled ones, which the web UI
|
||||
still shows -- and a reset made from the other process.
|
||||
|
||||
Caller must hold ``self._lock``.
|
||||
"""
|
||||
plugins = dict(self._read_snapshot(memory_ttl=0))
|
||||
if drop is not None:
|
||||
plugins.pop(drop, None)
|
||||
for plugin_id in self._metrics_dirty:
|
||||
if plugin_id in self._metrics:
|
||||
plugins[plugin_id] = self._metrics_record(self._metrics[plugin_id])
|
||||
cutoff = time.time() - _METRICS_SNAPSHOT_ENTRY_MAX_AGE
|
||||
for plugin_id, record in list(plugins.items()):
|
||||
last = record.get('last_update_time') if isinstance(record, dict) else None
|
||||
if isinstance(last, (int, float)) and last < cutoff:
|
||||
del plugins[plugin_id]
|
||||
self.cache_manager.set(METRICS_SNAPSHOT_KEY, {
|
||||
'schema': _METRICS_SNAPSHOT_SCHEMA,
|
||||
'plugins': plugins,
|
||||
})
|
||||
# Only once the write has landed, so a failed one is retried in full.
|
||||
self._metrics_dirty.clear()
|
||||
|
||||
def _persist_metrics(self, plugin_id: str, metrics: ResourceMetrics,
|
||||
force: bool = False) -> None:
|
||||
"""Record that a plugin's metrics changed, and write the snapshot if
|
||||
the last write is at least an interval old.
|
||||
|
||||
Caller must hold ``self._lock``.
|
||||
"""
|
||||
self._metrics_dirty.add(plugin_id)
|
||||
# Monotonic, not wall clock: these devices have no RTC, so the clock
|
||||
# jumps by however far off boot-time was the moment NTP first syncs.
|
||||
# A forward jump would allow an early write, a backward one would
|
||||
@@ -515,35 +597,26 @@ class PluginResourceMonitor:
|
||||
# single run -- the throttle swallowed the very first snapshot, which
|
||||
# is the one that matters most after a restart.
|
||||
now = time.monotonic()
|
||||
last_written = self._metrics_persisted_at.get(plugin_id)
|
||||
last_written = self._snapshot_persisted_at
|
||||
if (not force and last_written is not None
|
||||
and now - last_written < _METRICS_PERSIST_INTERVAL):
|
||||
return
|
||||
cache_key = self._get_metrics_key(plugin_id)
|
||||
self.cache_manager.set(cache_key, {
|
||||
'memory_mb': metrics.memory_mb,
|
||||
'cpu_percent': metrics.cpu_percent,
|
||||
'execution_time': metrics.execution_time,
|
||||
'call_count': metrics.call_count,
|
||||
'total_execution_time': metrics.total_execution_time,
|
||||
'max_execution_time': metrics.max_execution_time,
|
||||
'min_execution_time': (metrics.min_execution_time
|
||||
if metrics.min_execution_time != float('inf')
|
||||
else 0.0),
|
||||
'last_update_time': metrics.last_update_time,
|
||||
})
|
||||
self._write_snapshot()
|
||||
# Only after the write lands. Marking it first would mean a failed
|
||||
# set() bought the next interval's silence without leaving a snapshot.
|
||||
self._metrics_persisted_at[plugin_id] = now
|
||||
self._snapshot_persisted_at = now
|
||||
|
||||
def reset_metrics(self, plugin_id: str) -> None:
|
||||
"""Reset metrics for a plugin."""
|
||||
with self._lock:
|
||||
if plugin_id in self._metrics:
|
||||
self._metrics[plugin_id] = ResourceMetrics()
|
||||
cache_key = self._get_metrics_key(plugin_id)
|
||||
self.cache_manager.delete(cache_key)
|
||||
self._metrics_dirty.discard(plugin_id)
|
||||
self._write_snapshot(drop=plugin_id)
|
||||
# The record an older version wrote, so the reader's fallback
|
||||
# cannot bring the old numbers back.
|
||||
self.cache_manager.delete(self._get_metrics_key(plugin_id))
|
||||
# Let the next call persist immediately rather than leaving the
|
||||
# deleted key absent for the rest of the interval.
|
||||
self._metrics_persisted_at.pop(plugin_id, None)
|
||||
# plugin absent from the snapshot for the rest of the interval.
|
||||
self._snapshot_persisted_at = None
|
||||
|
||||
|
||||
@@ -22,6 +22,7 @@ from src.plugin_system.plugin_loader import (
|
||||
contained_plugin_dir, requirements_to_install,
|
||||
)
|
||||
from src.plugin_system.plugin_dirs import BACKUP_MARKER
|
||||
from src.plugin_system.plugin_local_files import carry_over_local_files
|
||||
from src.plugin_system.repo_urls import (
|
||||
USER_AGENT, github_api_headers, github_owner_repo, normalize_repo_url,
|
||||
)
|
||||
@@ -92,7 +93,9 @@ class _InstallMixin:
|
||||
raise
|
||||
|
||||
if installed:
|
||||
self._discard_backup(plugin_id, backup_path, "install")
|
||||
self._discard_backup(
|
||||
plugin_id, backup_path, "install",
|
||||
new_path=self._existing_install(plugin_id) or plugin_path)
|
||||
return True
|
||||
|
||||
self._restore_backup(plugin_id, plugin_path, backup_path, "Install")
|
||||
@@ -133,8 +136,33 @@ class _InstallMixin:
|
||||
return f"could not set aside {plugin_path}: {e}"
|
||||
return None
|
||||
|
||||
def _discard_backup(self, plugin_id: str, backup_path: Path, action: str) -> None:
|
||||
"""Remove the set-aside copy after a successful (re)install."""
|
||||
def _discard_backup(
|
||||
self, plugin_id: str, backup_path: Path, action: str,
|
||||
new_path: Optional[Path] = None,
|
||||
) -> None:
|
||||
"""Remove the set-aside copy after a successful (re)install.
|
||||
|
||||
With ``new_path`` (where the new copy landed), first carries the
|
||||
plugin's own runtime files -- OAuth tokens, client secrets, anything
|
||||
its .gitignore excludes -- from the old copy into the new one: no
|
||||
release contains them, so deleting the old copy would destroy them.
|
||||
See src/plugin_system/plugin_local_files.py. If any could not be
|
||||
copied the old copy is kept, so nothing is lost.
|
||||
"""
|
||||
if new_path is not None and new_path.is_dir():
|
||||
copied, failed = carry_over_local_files(backup_path, new_path)
|
||||
if copied:
|
||||
self.logger.info(
|
||||
"Kept %d local file(s) of %s across the %s: %s",
|
||||
len(copied), plugin_id, action, ", ".join(copied))
|
||||
if failed:
|
||||
self.logger.error(
|
||||
"Could not carry %s's local files into the new copy (%s); "
|
||||
"the previous copy is kept at %s -- copy them back by hand",
|
||||
plugin_id,
|
||||
"; ".join(f"{rel}: {err}" for rel, err in failed),
|
||||
backup_path)
|
||||
return
|
||||
if not self._safe_remove_directory(backup_path):
|
||||
self.logger.warning(
|
||||
"%s of %s succeeded but the previous copy at %s could not be "
|
||||
@@ -542,7 +570,8 @@ class _InstallMixin:
|
||||
raise
|
||||
temp_dir = None # Prevent cleanup since we moved it
|
||||
if backup_path is not None:
|
||||
self._discard_backup(plugin_id, backup_path, "install")
|
||||
self._discard_backup(
|
||||
plugin_id, backup_path, "install", new_path=final_path)
|
||||
|
||||
# Install dependencies
|
||||
self._install_dependencies(final_path)
|
||||
|
||||
@@ -138,6 +138,11 @@ class PluginStoreManager(_RegistryMixin, _InstallMixin, _UpdateMixin):
|
||||
# the registry cache expires. Only one thread fetches; others wait and
|
||||
# then get the result from the warm cache (double-checked locking).
|
||||
self._registry_fetch_lock = threading.Lock()
|
||||
# refresh_registry_in_background: the one refresh thread, and when
|
||||
# an offline one may be retried (see that method).
|
||||
self._registry_refresh_lock = threading.Lock()
|
||||
self._registry_refresh_thread: Optional[threading.Thread] = None
|
||||
self._registry_refresh_retry_after = 0.0
|
||||
|
||||
# Per-plugin locks for _reinstall_with_rollback: the web UI runs
|
||||
# Flask with threaded=True, so two overlapping requests for the
|
||||
@@ -344,12 +349,26 @@ class PluginStoreManager(_RegistryMixin, _InstallMixin, _UpdateMixin):
|
||||
2. Fix permissions via os.chmod() then retry (works for same-owner files)
|
||||
3. Use sudo rm -rf as last resort (works for root-owned __pycache__, etc.)
|
||||
|
||||
A symlink -- a dev plugin linked in by scripts/dev/dev_plugin_setup.sh
|
||||
-- is removed as a link, before any of that: rmtree refuses one, and
|
||||
stage 2 would walk through it and chmod the developer's checkout.
|
||||
|
||||
Args:
|
||||
path: Path to directory to remove
|
||||
|
||||
Returns:
|
||||
True if directory was removed successfully, False otherwise
|
||||
"""
|
||||
if path.is_symlink():
|
||||
# Checked before exists(), which follows the link: a dangling one
|
||||
# would read as already removed and be left behind.
|
||||
try:
|
||||
path.unlink()
|
||||
return True
|
||||
except OSError as e:
|
||||
self.logger.error(f"Could not remove the symlink {path}: {e}")
|
||||
return False
|
||||
|
||||
if not path.exists():
|
||||
return True # Already removed
|
||||
|
||||
|
||||
@@ -7,6 +7,7 @@ methods reach shared state and helpers through ``self``.
|
||||
|
||||
import json
|
||||
import requests
|
||||
import threading
|
||||
import time
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from datetime import datetime
|
||||
@@ -985,10 +986,14 @@ class _RegistryMixin:
|
||||
|
||||
def get_registry_info(self, plugin_id: str) -> Optional[Dict]:
|
||||
"""
|
||||
Get plugin information from the registry cache only (no GitHub API calls).
|
||||
Get plugin information from the registry (plugins.json).
|
||||
|
||||
Use this for lightweight lookups where only registry fields are needed
|
||||
(e.g., verified status, latest_version).
|
||||
Makes no GitHub API calls, but it does go through `fetch_registry`:
|
||||
when the in-memory copy is missing or older than
|
||||
``registry_cache_timeout`` it downloads plugins.json, and with no
|
||||
network that waits out the timeout and retries. A caller that must
|
||||
not block on the network (the installed-plugins list) uses
|
||||
`get_cached_registry_info` instead.
|
||||
|
||||
Args:
|
||||
plugin_id: Plugin identifier
|
||||
@@ -999,3 +1004,52 @@ class _RegistryMixin:
|
||||
registry = self.fetch_registry()
|
||||
plugins = registry.get('plugins', []) or []
|
||||
return self._match_registry_entry(plugins, plugin_id)
|
||||
|
||||
def get_cached_registry_info(self, plugin_id: str) -> Optional[Dict]:
|
||||
"""The registry entry for ``plugin_id`` from the copy already in
|
||||
memory, however old; never touches the network.
|
||||
|
||||
None when no registry has been loaded yet, or the plugin isn't in it.
|
||||
When the copy is missing or past ``registry_cache_timeout`` this
|
||||
starts `refresh_registry_in_background`, so a later call has it.
|
||||
"""
|
||||
cache = getattr(self, 'registry_cache', None)
|
||||
cache_time = getattr(self, 'registry_cache_time', None)
|
||||
if (not cache or not cache_time
|
||||
or (time.time() - cache_time) >= self.registry_cache_timeout):
|
||||
self.refresh_registry_in_background()
|
||||
plugins = cache.get('plugins') if isinstance(cache, dict) else None
|
||||
if not isinstance(plugins, list):
|
||||
return None
|
||||
return self._match_registry_entry(
|
||||
[p for p in plugins if isinstance(p, dict)], plugin_id)
|
||||
|
||||
def refresh_registry_in_background(self) -> bool:
|
||||
"""Fetch the registry on a daemon thread; True when one was started.
|
||||
|
||||
At most one runs at a time. After a fetch that left no registry in
|
||||
memory (offline), no new one starts for ``_failure_backoff_seconds``,
|
||||
so an offline Pi doesn't retry on every page load.
|
||||
"""
|
||||
with self._registry_refresh_lock:
|
||||
running = self._registry_refresh_thread
|
||||
if running is not None and running.is_alive():
|
||||
return False
|
||||
if time.time() < self._registry_refresh_retry_after:
|
||||
return False
|
||||
thread = threading.Thread(
|
||||
target=self._background_registry_refresh,
|
||||
name='registry-refresh', daemon=True)
|
||||
self._registry_refresh_thread = thread
|
||||
thread.start()
|
||||
return True
|
||||
|
||||
def _background_registry_refresh(self) -> None:
|
||||
try:
|
||||
self.fetch_registry()
|
||||
except Exception as e: # noqa: BLE001 - a background warm-up must not crash
|
||||
self.logger.warning("Background registry refresh failed: %s", e)
|
||||
if not getattr(self, 'registry_cache', None):
|
||||
with self._registry_refresh_lock:
|
||||
self._registry_refresh_retry_after = (
|
||||
time.time() + self._failure_backoff_seconds)
|
||||
|
||||
@@ -10,6 +10,9 @@ import subprocess # nosec B404 - list-form argv only, no shell # nosemgrep
|
||||
from pathlib import Path
|
||||
from typing import Dict, Optional, Tuple
|
||||
from src.plugin_system.plugin_dirs import BACKUP_MARKER
|
||||
from src.plugin_system.plugin_local_files import (
|
||||
KNOWN_STATE_PATTERNS, is_known_state_file,
|
||||
)
|
||||
from src.plugin_system.repo_urls import same_repo
|
||||
|
||||
|
||||
@@ -302,7 +305,11 @@ class _UpdateMixin:
|
||||
installed = False
|
||||
|
||||
if installed:
|
||||
self._discard_backup(plugin_id, backup_path, "update")
|
||||
# install_plugin may land the new copy under the manifest id
|
||||
# rather than the old directory name.
|
||||
self._discard_backup(
|
||||
plugin_id, backup_path, "update",
|
||||
new_path=self._existing_install(plugin_id) or plugin_path)
|
||||
return True
|
||||
|
||||
# Bad network, registry error...: the user keeps a working plugin.
|
||||
@@ -509,8 +516,12 @@ class _UpdateMixin:
|
||||
for line in untracked_result.stdout.strip().split('\n'):
|
||||
if line.startswith('??'):
|
||||
# Untracked file
|
||||
file_path = line[3:].strip()
|
||||
untracked_files.append(file_path)
|
||||
file_path = line[3:].strip().strip('"')
|
||||
# Tokens and secrets stay out of the
|
||||
# stash (see below), so they alone are
|
||||
# not a reason to stash.
|
||||
if not is_known_state_file(file_path):
|
||||
untracked_files.append(file_path)
|
||||
|
||||
# Check for tracked file changes
|
||||
status_result = subprocess.run(
|
||||
@@ -537,9 +548,17 @@ class _UpdateMixin:
|
||||
if has_changes:
|
||||
self.logger.info(f"Stashing local changes in {plugin_id} before update")
|
||||
try:
|
||||
# Use -u to include untracked files in stash
|
||||
# Use -u to include untracked files in stash --
|
||||
# except the plugin's tokens and secrets, which a
|
||||
# repo may have forgotten to gitignore. The stash
|
||||
# is never popped, so a stashed token.pickle would
|
||||
# vanish from the plugin and break it.
|
||||
stash_cmd = (
|
||||
['git', '-C', str(plugin_path), 'stash', 'push', '-u',
|
||||
'-m', f'LEDMatrix auto-stash before update {plugin_id}', '--', '.']
|
||||
+ [f':(exclude,glob)**/{p}' for p in KNOWN_STATE_PATTERNS])
|
||||
stash_result = subprocess.run(
|
||||
['git', '-C', str(plugin_path), 'stash', 'push', '-u', '-m', f'LEDMatrix auto-stash before update {plugin_id}'],
|
||||
stash_cmd,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=30,
|
||||
|
||||
+41
-9
@@ -23,6 +23,10 @@ above it near the end, the section below must show one more refresh of lag to
|
||||
stay continuous with it (and one less where the order jumps the other way).
|
||||
Stacked parallel chains are lit simultaneously, so each further half adds one.
|
||||
|
||||
A frame held for several refreshes (a slower, crisp scroll) is presented as a
|
||||
sequence of swaps instead of one long hold, so the lagging half can step one
|
||||
refresh after the rest: see :func:`refresh_plan`.
|
||||
|
||||
Only layouts whose physical row order is known are compensated: plain chains,
|
||||
parallel chains, and a 0 or 180 degree rotation. Other pixel mappers
|
||||
(U-mapper, 90/270 rotation, ...), special multiplexing and interlaced scan are
|
||||
@@ -109,19 +113,47 @@ def scan_lag_bands(hardware: Mapping[str, Any], height: int,
|
||||
return [band for band in bands if band[2] > 0] or None
|
||||
|
||||
|
||||
def compose(image: Image.Image, history: Sequence[Image.Image],
|
||||
bands: Sequence[Band]) -> Image.Image:
|
||||
"""``image`` with each band taken from the frame ``lag`` refreshes back.
|
||||
def refresh_plan(bands: Sequence[Band], hold: int) -> List[Tuple[Tuple[int, ...], int]]:
|
||||
"""How to present one frame that is held for ``hold`` refreshes.
|
||||
|
||||
``history[0]`` is the previous frame. A band whose frame is not available
|
||||
yet (the first frames of a scroll) is left current. Returns ``image`` itself
|
||||
when nothing changes, so the caller pays for a copy only when it must.
|
||||
A band lagging ``lag`` refreshes shows, on refresh ``r`` of the frame, what
|
||||
the panel showed ``lag`` refreshes earlier: the current frame once
|
||||
``r >= lag``, else a frame ``ceil((lag - r) / hold)`` back. At one refresh
|
||||
per frame that is just ``lag`` frames back. Held longer, the lagging band
|
||||
steps one refresh after the rest instead of one frame, which is the only
|
||||
way to cancel the offset: it is a fraction of a frame there.
|
||||
|
||||
Returns ``[(frames_back_per_band, refreshes), ...]`` in order, merging
|
||||
neighbouring refreshes that show the same thing so each costs one swap.
|
||||
"""
|
||||
plan: List[Tuple[Tuple[int, ...], int]] = []
|
||||
for r in range(max(1, hold)):
|
||||
backs = tuple(max(0, -((r - lag) // max(1, hold))) for _, _, lag in bands)
|
||||
if plan and plan[-1][0] == backs:
|
||||
plan[-1] = (backs, plan[-1][1] + 1)
|
||||
else:
|
||||
plan.append((backs, 1))
|
||||
return plan
|
||||
|
||||
|
||||
def compose(image: Image.Image, history: Sequence[Image.Image],
|
||||
bands: Sequence[Band],
|
||||
backs: Optional[Sequence[int]] = None) -> Image.Image:
|
||||
"""``image`` with each band taken from an earlier frame.
|
||||
|
||||
``history[0]`` is the previous frame. ``backs`` is how many frames back each
|
||||
band is taken from (0 = the current one); by default that is the band's lag,
|
||||
which is right when every frame is held for one refresh. A band whose frame
|
||||
is not available yet (the first frames of a scroll) is left current.
|
||||
Returns ``image`` itself when nothing changes, so the caller pays for a copy
|
||||
only when it must.
|
||||
"""
|
||||
out = image
|
||||
for top, bottom, lag in bands:
|
||||
if lag > len(history):
|
||||
for i, (top, bottom, lag) in enumerate(bands):
|
||||
back = lag if backs is None else backs[i]
|
||||
if back <= 0 or back > len(history):
|
||||
continue
|
||||
source = history[lag - 1]
|
||||
source = history[back - 1]
|
||||
if source.size != image.size:
|
||||
continue
|
||||
if out is image:
|
||||
|
||||
@@ -0,0 +1,484 @@
|
||||
"""Runs one screen: the ScreenRunner of docs/RUN_LOOP_REDESIGN.md.
|
||||
|
||||
``ScreenRunner.run(plan, plugin)`` draws a screen's first frame, runs the
|
||||
frame loop its plan's ``frame_policy`` picks (125 Hz or 1 Hz), makes up the
|
||||
minimum duration when the loop ended early, and returns one
|
||||
:class:`Outcome` saying why the screen ended. Everything that touches the
|
||||
plugin, the panel or the controller's state goes through a
|
||||
:class:`ScreenHost` (the DisplayController); everything that reads or waits
|
||||
on the clock goes through an injected :class:`FrameClock`. The runner itself
|
||||
holds no state between screens.
|
||||
|
||||
What can end a screen early is decided at the runner's service points: after
|
||||
each frame, after the frame loop, and after the make-up dwell. At each one
|
||||
the host gathers a snapshot and asks the Arbiter, once, whether a Source in
|
||||
``plan.preemptible_by`` now wants the panel (:meth:`ScreenHost.check`). A yes
|
||||
is ``ExitReason.PREEMPTED``: the next pass of the loop decides what shows,
|
||||
and the rotation does not advance past the screen that was cut short.
|
||||
|
||||
The frame pacing is the loop that used to be inline in
|
||||
``DisplayController.run()``, unchanged: the 125 Hz loop paces to an 8 ms
|
||||
deadline from the start of each frame (sleeping at least 1 ms, so a frame
|
||||
that overran still yields the GIL), and the 1 Hz loop sleeps a flat second
|
||||
between frames, woken early by a control socket command.
|
||||
"""
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
from typing import Any, NamedTuple, Optional, Protocol, Tuple
|
||||
|
||||
from src.display_arbiter import FramePolicy, ScreenPlan, Source
|
||||
|
||||
__all__ = [
|
||||
"AFTER_COMPLETED_LOOP",
|
||||
"AFTER_LOOP",
|
||||
"Checkpoint",
|
||||
"DYNAMIC_GRACE",
|
||||
"ExitReason",
|
||||
"FINAL",
|
||||
"FRAME",
|
||||
"FirstFrame",
|
||||
"FrameClock",
|
||||
"HIGH_FPS_INTERVAL",
|
||||
"NoticeRead",
|
||||
"Outcome",
|
||||
"STATIC_INTERVAL",
|
||||
"Screen",
|
||||
"ScreenHost",
|
||||
"ScreenRunner",
|
||||
"after_dwell",
|
||||
]
|
||||
|
||||
#: Seconds between frames in the high-FPS loop (125 Hz), for scrolling plugins.
|
||||
HIGH_FPS_INTERVAL = 0.008
|
||||
|
||||
#: Seconds between frames in the static loop (1 Hz).
|
||||
STATIC_INTERVAL = 1.0
|
||||
|
||||
#: A dynamic-duration screen ends on cycle completion only this long after its
|
||||
#: minimum, so timing jitter around the minimum can't end it early.
|
||||
DYNAMIC_GRACE = 0.5
|
||||
|
||||
|
||||
class ExitReason(Enum):
|
||||
"""Why a screen ended. The value is the golden traces' exit column where
|
||||
one exists (test/test_run_loop_golden.py)."""
|
||||
|
||||
#: The screen ran its target duration.
|
||||
DURATION = "duration"
|
||||
#: A dynamic-duration plugin finished its cycle after its minimum.
|
||||
CYCLE_COMPLETE = "cycle-complete"
|
||||
#: The first frame had nothing to show (display() returned False or
|
||||
#: raised inside the executor), or no plugin draws the mode.
|
||||
EMPTY = "empty"
|
||||
#: The first frame's dispatch itself raised.
|
||||
ERROR = "error"
|
||||
#: A later frame returned False (a dynamic-duration screen on the 1 Hz
|
||||
#: loop keeps going instead).
|
||||
DISPLAY_FALSE = "display-false"
|
||||
#: Another Source took the panel, or an on-demand session ran out before
|
||||
#: the screen began. The rotation does not advance.
|
||||
PREEMPTED = "preempted"
|
||||
#: A plugin reload is waiting for the top of the loop. The screen is cut
|
||||
#: short but counts as shown: the rotation advances, and the next pass
|
||||
#: reloads before it draws.
|
||||
RELOAD = "reload"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Outcome:
|
||||
"""How a screen ended.
|
||||
|
||||
Attributes:
|
||||
exit_reason: Why it ended.
|
||||
elapsed: Seconds from the end of the first frame to the end.
|
||||
preempted_by: For PREEMPTED, the plan that took the panel when the
|
||||
Arbiter named one (None when the session simply ran out).
|
||||
on_demand_active: Filled in by the controller when the screen is
|
||||
over: an on-demand session was running at that moment.
|
||||
still_live: Filled in by the controller: the mode's plugin still had
|
||||
live content at that moment, which holds the rotation on it.
|
||||
"""
|
||||
|
||||
exit_reason: ExitReason
|
||||
elapsed: float = 0.0
|
||||
preempted_by: Optional[ScreenPlan] = None
|
||||
on_demand_active: bool = False
|
||||
still_live: bool = False
|
||||
|
||||
|
||||
class FrameClock(Protocol):
|
||||
"""The clocks the runner reads and the sleep it paces with.
|
||||
|
||||
The shape of the ``time`` module, so production passes it (through an
|
||||
indirection that lets tests patch the module) and the golden traces pass
|
||||
their fake clock.
|
||||
"""
|
||||
|
||||
def time(self) -> float:
|
||||
"""Wall-clock seconds: what screen durations are measured in."""
|
||||
|
||||
def perf_counter(self) -> float:
|
||||
"""A monotonic high-resolution clock: what the 8 ms pacing reads."""
|
||||
|
||||
def sleep(self, seconds: float) -> None:
|
||||
"""Block for ``seconds``."""
|
||||
|
||||
|
||||
class NoticeRead(Enum):
|
||||
"""When a service point reads the WiFi notice file.
|
||||
|
||||
The read is throttled to once a second and deletes an expired file, so
|
||||
*when* it happens is behaviour: each service point reads it exactly
|
||||
when the loop always did.
|
||||
"""
|
||||
|
||||
#: Not at all.
|
||||
NEVER = "never"
|
||||
#: Only if nothing cheaper has already ended the screen: no on-demand
|
||||
#: session, the panel on, the mode unchanged and no live takeover.
|
||||
IF_UNDECIDED = "if-undecided"
|
||||
#: Whenever no on-demand session is running.
|
||||
ALWAYS = "always"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Checkpoint:
|
||||
"""What one kind of service point considers.
|
||||
|
||||
Attributes:
|
||||
name: For logs and tests.
|
||||
notice: When the WiFi notice is read (see NoticeRead).
|
||||
notice_counts: Whether a pending notice ends the screen here. After
|
||||
the make-up dwell it does only while the screen had time left.
|
||||
reload: Whether a pending plugin reload ends the screen here. Only
|
||||
between frames: once the frame loop is over the screen is too.
|
||||
"""
|
||||
|
||||
name: str
|
||||
notice: NoticeRead
|
||||
reload: bool
|
||||
notice_counts: bool = True
|
||||
|
||||
|
||||
#: Between frames: the frame loops' check, and the socket wake in the 1 Hz wait.
|
||||
FRAME = Checkpoint("frame", NoticeRead.IF_UNDECIDED, reload=True)
|
||||
#: After a frame loop that ended early (display() returned False, a reload).
|
||||
AFTER_LOOP = Checkpoint("after-loop", NoticeRead.IF_UNDECIDED, reload=False)
|
||||
#: After a frame loop that ran its course: only a mode change or the schedule.
|
||||
AFTER_COMPLETED_LOOP = Checkpoint("after-completed-loop", NoticeRead.NEVER, reload=False)
|
||||
#: The last look before the rotation advances.
|
||||
FINAL = Checkpoint("final", NoticeRead.NEVER, reload=False)
|
||||
|
||||
|
||||
def after_dwell(time_left: bool) -> Checkpoint:
|
||||
"""After the make-up dwell: a notice that cut it short ends the screen,
|
||||
so the mode resumes after the notice instead of rotating past it."""
|
||||
return Checkpoint("after-dwell", NoticeRead.ALWAYS, reload=False,
|
||||
notice_counts=time_left)
|
||||
|
||||
|
||||
class FirstFrame(NamedTuple):
|
||||
"""What the first frame's dispatch returned (see _dispatch_first_frame)."""
|
||||
|
||||
shown: bool
|
||||
raised: bool
|
||||
accepts_display_mode: bool
|
||||
|
||||
|
||||
@dataclass
|
||||
class Screen:
|
||||
"""One running screen: its completed plan, the plugin drawing it and
|
||||
when it started. Mutable only in that the runner owns it."""
|
||||
|
||||
plan: ScreenPlan
|
||||
plugin: Any
|
||||
accepts_display_mode: bool
|
||||
start: float
|
||||
|
||||
@property
|
||||
def mode(self) -> Optional[str]:
|
||||
return self.plan.mode
|
||||
|
||||
|
||||
class ScreenHost(Protocol):
|
||||
"""The controller's side of a screen. See DisplayController."""
|
||||
|
||||
def first_frame(self, plan: ScreenPlan, plugin: Any) -> FirstFrame:
|
||||
"""Draw the first frame through the plugin executor."""
|
||||
|
||||
def complete_plan(self, plan: ScreenPlan, plugin: Any) -> Optional[ScreenPlan]:
|
||||
"""The plan with the plugin's durations, dynamic flag and frame
|
||||
policy, read after the first frame. None when an on-demand session
|
||||
has no time left for it."""
|
||||
|
||||
def draw(self, screen: Screen) -> Any:
|
||||
"""One later frame: what display() returned."""
|
||||
|
||||
def after_frame(self, screen: Screen) -> None:
|
||||
"""After a frame that did not end the screen (the follower frame)."""
|
||||
|
||||
def tick(self) -> None:
|
||||
"""Plugin updates that have come due (throttled)."""
|
||||
|
||||
def service(self, screen: Screen) -> Optional[Tuple[str, ...]]:
|
||||
"""Apply pending changes (on-demand requests, schedule, brightness,
|
||||
finished reloads). Returns the live modes when a live-priority scan
|
||||
was due, else None."""
|
||||
|
||||
def wait_frame(self, interval: float, screen: Screen) -> Optional[ScreenPlan]:
|
||||
"""The 1 Hz loop's sleep between frames. The plan that takes the
|
||||
panel when a control socket command ended the screen, else None."""
|
||||
|
||||
def check(self, screen: Screen, checkpoint: Checkpoint,
|
||||
live_scan: Optional[Tuple[str, ...]] = None) -> Optional[ScreenPlan]:
|
||||
"""The service point: the plan that now takes the panel from this
|
||||
screen, or None while it holds. One Arbiter.decide() call."""
|
||||
|
||||
def dwell(self, seconds: float) -> None:
|
||||
"""Sleep up to ``seconds``, servicing changes; returns early on one."""
|
||||
|
||||
def cycle_complete(self, screen: Screen) -> bool:
|
||||
"""The plugin's dynamic-duration cycle is complete."""
|
||||
|
||||
|
||||
class ScreenRunner:
|
||||
"""Runs one screen at a time for a ScreenHost. See the module docstring."""
|
||||
|
||||
def __init__(self, clock: FrameClock, host: ScreenHost,
|
||||
log: Optional[logging.Logger] = None):
|
||||
self.clock = clock
|
||||
self.host = host
|
||||
# The controller passes its own logger, so these lines keep the
|
||||
# source they always had in the journal.
|
||||
self.log = log or logging.getLogger(__name__)
|
||||
|
||||
# -- the screen ------------------------------------------------------
|
||||
|
||||
def run(self, plan: ScreenPlan, plugin: Any) -> Outcome:
|
||||
"""Run ``plan``'s screen, drawn by ``plugin`` (None: nothing draws it)."""
|
||||
if plugin is None:
|
||||
return Outcome(ExitReason.EMPTY)
|
||||
first = self.host.first_frame(plan, plugin)
|
||||
if not first.shown:
|
||||
return Outcome(ExitReason.ERROR if first.raised else ExitReason.EMPTY)
|
||||
completed = self.host.complete_plan(plan, plugin)
|
||||
if completed is None:
|
||||
return Outcome(ExitReason.PREEMPTED)
|
||||
screen = Screen(completed, plugin, first.accepts_display_mode,
|
||||
start=self.clock.time())
|
||||
|
||||
if completed.frame_policy is FramePolicy.HIGH_FPS:
|
||||
reason, by = self._high_fps_loop(screen)
|
||||
else:
|
||||
reason, by = self._static_loop(screen)
|
||||
if reason is ExitReason.PREEMPTED:
|
||||
# The service point that ended the loop has decided; looking
|
||||
# again now, at the same instant, gives the same answer.
|
||||
return self._outcome(screen, reason, by)
|
||||
|
||||
loop_completed = reason in (ExitReason.DURATION, ExitReason.CYCLE_COMPLETE)
|
||||
# LOAD-BEARING: a change the frame loop did not end on (a dwell
|
||||
# inside it, a later frame returning False) must not fall into the
|
||||
# make-up dwell below. It can run for the rest of the screen's
|
||||
# duration, and a freshly requested on-demand mode would sit
|
||||
# invisible for that long -- or be clobbered by a queued stop.
|
||||
by = self.host.check(screen, AFTER_COMPLETED_LOOP if loop_completed else AFTER_LOOP)
|
||||
if by is not None:
|
||||
return self._outcome(screen, ExitReason.PREEMPTED, by)
|
||||
|
||||
# Honour the minimum duration when a static, non-dynamic screen's
|
||||
# loop ended early. A screen cut short for a plugin reload is over:
|
||||
# the dwell returns at once and the rotation advances.
|
||||
if (not completed.dynamic and not loop_completed
|
||||
and completed.frame_policy is not FramePolicy.HIGH_FPS):
|
||||
elapsed = self.clock.time() - screen.start
|
||||
remaining = max(0.0, self._max(screen) - elapsed)
|
||||
if remaining > 0:
|
||||
self.host.dwell(remaining)
|
||||
time_left = self.clock.time() - screen.start < self._max(screen)
|
||||
by = self.host.check(screen, after_dwell(time_left))
|
||||
if by is not None:
|
||||
return self._outcome(screen, ExitReason.PREEMPTED, by)
|
||||
|
||||
if completed.dynamic:
|
||||
self._log_dynamic_end(screen)
|
||||
|
||||
# The dwells above return early when a pending change (on-demand
|
||||
# started or stopped, the panel scheduled off) has already decided
|
||||
# what comes next; rotating now would skip it.
|
||||
by = self.host.check(screen, FINAL)
|
||||
if by is not None:
|
||||
return self._outcome(screen, ExitReason.PREEMPTED, by)
|
||||
return self._outcome(screen, reason, None)
|
||||
|
||||
def _outcome(self, screen: Screen, reason: ExitReason,
|
||||
by: Optional[ScreenPlan]) -> Outcome:
|
||||
return Outcome(reason, self.clock.time() - screen.start, preempted_by=by)
|
||||
|
||||
@staticmethod
|
||||
def _max(screen: Screen) -> float:
|
||||
return float(screen.plan.max_duration or 0.0)
|
||||
|
||||
@staticmethod
|
||||
def _min(screen: Screen) -> float:
|
||||
return float(screen.plan.min_duration or 0.0)
|
||||
|
||||
@staticmethod
|
||||
def _ended_by(by: ScreenPlan) -> ExitReason:
|
||||
return ExitReason.RELOAD if by.source is Source.RELOAD else ExitReason.PREEMPTED
|
||||
|
||||
# -- the frame loops ---------------------------------------------------
|
||||
|
||||
def _high_fps_loop(self, screen: Screen) -> Tuple[ExitReason, Optional[ScreenPlan]]:
|
||||
"""Ultra-smooth frames for scrolling plugins (8 ms = 125 FPS)."""
|
||||
clock, host, log = self.clock, self.host, self.log
|
||||
interval = HIGH_FPS_INTERVAL
|
||||
log.debug("Entering high-FPS loop for %s with display_interval=%.3fs (%.1f FPS)",
|
||||
screen.mode, interval, 1.0 / interval)
|
||||
target = self._max(screen)
|
||||
while True:
|
||||
frame_start = clock.perf_counter()
|
||||
try:
|
||||
result = host.draw(screen)
|
||||
if isinstance(result, bool) and not result:
|
||||
log.debug("Display returned False, breaking early")
|
||||
return ExitReason.DISPLAY_FALSE, None
|
||||
except Exception: # pylint: disable=broad-except
|
||||
log.exception("Error during display update")
|
||||
|
||||
# Multi-display sync: send follower frame after each render
|
||||
host.after_frame(screen)
|
||||
host.tick()
|
||||
# Throttled: one clock compare between passes. A live-priority
|
||||
# scan, when one is due, happens here, before the sleep, as it
|
||||
# always has; the Arbiter weighs it after the sleep.
|
||||
live_scan = host.service(screen)
|
||||
|
||||
# Pace to the frame deadline rather than sleeping a flat
|
||||
# interval on top of the work. display() has already blocked on
|
||||
# the panel's vsync by this point, so an unconditional sleep is
|
||||
# added to a wait that already happened. Measured on a 2x128x64
|
||||
# chain at limit_refresh_rate_hz=100: ~4ms of render plus a flat
|
||||
# 8ms put each iteration at ~12ms against a 10ms refresh grid, so
|
||||
# every swap missed a refresh and the loop settled at 50fps where
|
||||
# display_interval asks for 125 -- and with zero headroom, ~14% of
|
||||
# frames slipped a further refresh, which is what reads as scroll
|
||||
# stutter.
|
||||
remaining = interval - (clock.perf_counter() - frame_start)
|
||||
# Yield even when the frame overran its budget, so plugin update
|
||||
# threads and the web UI are not starved of the GIL.
|
||||
clock.sleep(remaining if remaining > 0 else 0.001)
|
||||
|
||||
by = host.check(screen, FRAME, live_scan)
|
||||
if by is not None:
|
||||
log.debug("Mode changed during high-FPS loop, breaking early")
|
||||
return self._ended_by(by), by
|
||||
|
||||
elapsed = clock.time() - screen.start
|
||||
if elapsed >= target:
|
||||
log.debug("Reached high-FPS target duration %.2fs for mode %s",
|
||||
target, screen.mode)
|
||||
return ExitReason.DURATION, None
|
||||
if self._should_exit_dynamic(screen, elapsed):
|
||||
log.debug("Dynamic duration cycle complete for %s after %.2fs",
|
||||
screen.mode, elapsed)
|
||||
return ExitReason.CYCLE_COMPLETE, None
|
||||
|
||||
def _static_loop(self, screen: Screen) -> Tuple[ExitReason, Optional[ScreenPlan]]:
|
||||
"""One frame a second for everything else."""
|
||||
clock, host, log = self.clock, self.host, self.log
|
||||
interval = STATIC_INTERVAL
|
||||
log.debug("Entering normal FPS loop for %s with display_interval=%.3fs",
|
||||
screen.mode, interval)
|
||||
target = self._max(screen)
|
||||
dynamic = screen.plan.dynamic
|
||||
while True:
|
||||
# Wakes for a control socket command and applies it at once,
|
||||
# instead of up to a second later.
|
||||
by = host.wait_frame(interval, screen)
|
||||
if by is not None:
|
||||
log.info("Mode changed during display loop from %s to %s (%s), "
|
||||
"breaking early", screen.mode, by.mode, by.source.value)
|
||||
return self._ended_by(by), by
|
||||
host.tick()
|
||||
|
||||
elapsed = clock.time() - screen.start
|
||||
if elapsed >= target:
|
||||
log.debug("Reached standard target duration %.2fs for mode %s",
|
||||
target, screen.mode)
|
||||
return ExitReason.DURATION, None
|
||||
|
||||
try:
|
||||
result = host.draw(screen)
|
||||
if isinstance(result, bool) and not result:
|
||||
# A dynamic-duration screen doesn't end on False: it
|
||||
# keeps looping until its cycle completes or its maximum.
|
||||
if not dynamic:
|
||||
log.info("Display returned False for %s (no dynamic duration), "
|
||||
"breaking early", screen.mode)
|
||||
return ExitReason.DISPLAY_FALSE, None
|
||||
log.debug("Display returned False for %s (dynamic duration enabled), "
|
||||
"continuing loop", screen.mode)
|
||||
except Exception: # pylint: disable=broad-except
|
||||
log.exception("Error during display update")
|
||||
|
||||
# Multi-display sync: send follower frame after each render
|
||||
host.after_frame(screen)
|
||||
|
||||
live_scan = host.service(screen)
|
||||
by = host.check(screen, FRAME, live_scan)
|
||||
if by is not None:
|
||||
log.info("Mode changed during display loop from %s to %s (%s), "
|
||||
"breaking early", screen.mode, by.mode, by.source.value)
|
||||
return self._ended_by(by), by
|
||||
|
||||
if self._should_exit_dynamic(screen, elapsed):
|
||||
log.info("Dynamic duration cycle complete for %s after %.2fs",
|
||||
screen.mode, elapsed)
|
||||
return ExitReason.CYCLE_COMPLETE, None
|
||||
|
||||
# -- dynamic duration --------------------------------------------------
|
||||
|
||||
def _should_exit_dynamic(self, screen: Screen, elapsed: float) -> bool:
|
||||
if not screen.plan.dynamic:
|
||||
return False
|
||||
minimum = self._min(screen)
|
||||
# A small grace period after min_duration prevents premature exits
|
||||
# due to timing issues.
|
||||
if elapsed < minimum + DYNAMIC_GRACE:
|
||||
self.log.debug(
|
||||
"_should_exit_dynamic: elapsed %.2fs < min_duration %.2fs + grace %.2fs, "
|
||||
"returning False", elapsed, minimum, DYNAMIC_GRACE)
|
||||
return False
|
||||
cycle_complete = self.host.cycle_complete(screen)
|
||||
self.log.debug(
|
||||
"_should_exit_dynamic: elapsed %.2fs >= min %.2fs, cycle_complete=%s, returning %s",
|
||||
elapsed, minimum + DYNAMIC_GRACE, cycle_complete, cycle_complete)
|
||||
if cycle_complete:
|
||||
self.log.debug("Cycle complete detected for %s after %.2fs (min: %.2fs, grace: %.2fs)",
|
||||
screen.mode, elapsed, minimum, DYNAMIC_GRACE)
|
||||
return cycle_complete
|
||||
|
||||
def _log_dynamic_end(self, screen: Screen) -> None:
|
||||
"""How a dynamic-duration screen ended, for the log. Asks the plugin
|
||||
once more whether its cycle is complete, as the loop always did."""
|
||||
elapsed_total = self.clock.time() - screen.start
|
||||
cycle_done = self.host.cycle_complete(screen)
|
||||
minimum, maximum = self._min(screen), self._max(screen)
|
||||
if cycle_done:
|
||||
self.log.info(
|
||||
"Dynamic duration cycle completed for %s after %.2fs "
|
||||
"(target: %.2fs, min: %.2fs, max: %.2fs)",
|
||||
screen.mode, elapsed_total, maximum, minimum, maximum)
|
||||
elif elapsed_total >= maximum:
|
||||
self.log.info(
|
||||
"Dynamic duration cap reached before cycle completion for %s "
|
||||
"(%.2fs/%ds, min: %.2fs)",
|
||||
screen.mode, elapsed_total, int(maximum), minimum)
|
||||
else:
|
||||
self.log.debug(
|
||||
"Dynamic duration cycle in progress for %s: %.2fs elapsed "
|
||||
"(target: %.2fs, min: %.2fs, max: %.2fs)",
|
||||
screen.mode, elapsed_total, maximum, minimum, maximum)
|
||||
@@ -95,11 +95,13 @@ class StartupValidator:
|
||||
def _validate_systemd_units(self) -> None:
|
||||
"""Warn when an installed unit has drifted from the repo's template.
|
||||
|
||||
Nothing re-applies these after the first install. `git pull` -- which is
|
||||
what the web UI's update button runs -- brings a new template into the
|
||||
checkout, but nothing copies it to /etc/systemd/system and nothing runs
|
||||
`systemctl daemon-reload`, so the unit that actually runs is whatever
|
||||
first_time_install.sh wrote on day one.
|
||||
Before updates refreshed units, nothing re-applied these after the
|
||||
first install: `git pull` brought a new template into the checkout,
|
||||
but nothing copied it to /etc/systemd/system, so the unit that
|
||||
actually ran was whatever first_time_install.sh wrote on day one.
|
||||
Updates now install changed units through the root helper
|
||||
ledmatrix-refresh-units (web_interface/unit_refresh.py) -- but only on
|
||||
a device whose installer granted it, so this still catches the rest.
|
||||
|
||||
That makes every hardening added to a unit inert on existing installs.
|
||||
Measured on one rig: the installed unit was thirteen days older than the
|
||||
@@ -141,10 +143,11 @@ class StartupValidator:
|
||||
|
||||
if self._unit_body(expected) != self._unit_body(actual):
|
||||
self.warnings.append(
|
||||
f"{installed.name} differs from {template_rel}; the "
|
||||
"installed unit is not refreshed by an update, so "
|
||||
f"{installed.name} differs from {template_rel}, so "
|
||||
"settings added to the template are not in effect. "
|
||||
"Re-run scripts/install/install_service.sh to apply them."
|
||||
"Updates apply them only once the installer has granted "
|
||||
"ledmatrix-refresh-units: re-run "
|
||||
"scripts/install/install_service.sh (or first_time_install.sh) to apply them."
|
||||
)
|
||||
except OSError as e:
|
||||
self.logger.debug("Could not compare systemd units: %s", e)
|
||||
|
||||
@@ -152,6 +152,8 @@ class VegasModeCoordinator:
|
||||
# Interrupt checker for yielding control back to display controller
|
||||
self._interrupt_check: Optional[Callable[[], bool]] = None
|
||||
self._interrupt_check_interval: int = 10 # Check every N frames
|
||||
# Checked every frame; True runs the interrupt check at once.
|
||||
self._interrupt_urgent: Optional[Callable[[], bool]] = None
|
||||
|
||||
# Plugin update callback — fired from a background thread inside the loop
|
||||
# so the main loop's _tick_plugin_updates() finds nothing due when Vegas
|
||||
@@ -226,7 +228,8 @@ class VegasModeCoordinator:
|
||||
def set_interrupt_checker(
|
||||
self,
|
||||
checker: Callable[[], bool],
|
||||
check_interval: int = 10
|
||||
check_interval: int = 10,
|
||||
urgent: Optional[Callable[[], bool]] = None,
|
||||
) -> None:
|
||||
"""
|
||||
Set the callback for checking if Vegas should yield control.
|
||||
@@ -237,9 +240,25 @@ class VegasModeCoordinator:
|
||||
Args:
|
||||
checker: Callable that returns True if Vegas should yield
|
||||
check_interval: Check every N frames (default 10)
|
||||
urgent: A cheap per-frame test; when it is True the checker
|
||||
runs at this frame instead of waiting for the interval (the
|
||||
display controller passes "a control socket command is
|
||||
queued", so a command waits one frame, not ten)
|
||||
"""
|
||||
self._interrupt_check = checker
|
||||
self._interrupt_check_interval = max(1, check_interval)
|
||||
self._interrupt_urgent = urgent
|
||||
|
||||
def _interrupt_is_urgent(self) -> bool:
|
||||
"""The per-frame test set with ``urgent``; never raises."""
|
||||
urgent = getattr(self, '_interrupt_urgent', None)
|
||||
if urgent is None:
|
||||
return False
|
||||
try:
|
||||
return bool(urgent()) # pylint: disable=not-callable
|
||||
except Exception: # pylint: disable=broad-except
|
||||
logger.debug("Urgent interrupt test failed", exc_info=True)
|
||||
return False
|
||||
|
||||
def set_update_callback(self, callback: Callable[[], None]) -> None:
|
||||
"""
|
||||
@@ -709,7 +728,8 @@ class VegasModeCoordinator:
|
||||
frame_times.clear()
|
||||
|
||||
if (self._interrupt_check and
|
||||
frame_count % self._interrupt_check_interval == 0):
|
||||
(frame_count % self._interrupt_check_interval == 0
|
||||
or self._interrupt_is_urgent())):
|
||||
try:
|
||||
if self._interrupt_check():
|
||||
logger.debug(
|
||||
|
||||
@@ -183,9 +183,27 @@ class PluginAdapter:
|
||||
"round", plugin_id, self.PLUGIN_LOCK_TIMEOUT
|
||||
)
|
||||
return None
|
||||
if not self._still_loaded(plugin, plugin_id):
|
||||
return None
|
||||
return self._fetch_content(plugin, plugin_id, restricted=False,
|
||||
keyed=keyed)
|
||||
|
||||
def _still_loaded(self, plugin: 'BasePlugin', plugin_id: str) -> bool:
|
||||
"""Whether ``plugin`` is still the loaded instance of ``plugin_id``.
|
||||
|
||||
Checked once the plugin's lock is held: a reload or a disable can
|
||||
take the instance out and tear it down while this fetch waited for
|
||||
the lock (PluginManager.detach_plugin), and a torn-down instance is
|
||||
not asked for content. True when the manager keeps no ``plugins``
|
||||
mapping to ask.
|
||||
"""
|
||||
plugins = getattr(self.plugin_manager, 'plugins', None)
|
||||
if not isinstance(plugins, dict) or plugins.get(plugin_id) is plugin:
|
||||
return True
|
||||
logger.debug("[%s] Unloaded or reloaded while waiting for its lock; "
|
||||
"skipping the old instance", plugin_id)
|
||||
return False
|
||||
|
||||
def is_live_capable(self, plugin: 'BasePlugin', plugin_id: str) -> bool:
|
||||
"""Whether to ask this plugin for live elements rather than pictures.
|
||||
|
||||
@@ -922,6 +940,8 @@ class PluginAdapter:
|
||||
return None
|
||||
epochs = self.live_epochs
|
||||
epoch = epochs.get(plugin_id) if epochs is not None else 0
|
||||
if not self._still_loaded(plugin, plugin_id):
|
||||
return epoch, {}
|
||||
render_width = self.resolve_render_width(plugin, plugin_id)
|
||||
plugin._vegas_render_width = render_width
|
||||
try:
|
||||
|
||||
@@ -333,8 +333,11 @@ class StreamManager:
|
||||
loaded = 0
|
||||
|
||||
if hasattr(self.plugin_manager, 'plugins'):
|
||||
loaded = len(self.plugin_manager.plugins)
|
||||
for plugin_id, plugin in self.plugin_manager.plugins.items():
|
||||
# A snapshot: a plugin reload adds and removes entries on its
|
||||
# own thread (DisplayController._start_plugin_reload).
|
||||
plugins = list(self.plugin_manager.plugins.items())
|
||||
loaded = len(plugins)
|
||||
for plugin_id, plugin in plugins:
|
||||
if not getattr(plugin, 'enabled', False):
|
||||
logger.debug("[%s] Vegas: skipped (not enabled)", plugin_id)
|
||||
continue
|
||||
|
||||
@@ -19,6 +19,11 @@ Type=simple
|
||||
User=__USER__
|
||||
WorkingDirectory=__PROJECT_ROOT_DIR__
|
||||
Environment=USE_THREADING=1
|
||||
# Cap glibc's malloc arenas, as ledmatrix.service does: each allocating thread
|
||||
# can get its own arena, up to 8 x CPU count (24 on a 3-core Pi), and a grown
|
||||
# arena is never handed back to the OS. This threaded Flask process would hold
|
||||
# that memory the same way. See ledmatrix.service for the measurement.
|
||||
Environment=MALLOC_ARENA_MAX=2
|
||||
ExecStart=/usr/bin/python3 __PROJECT_ROOT_DIR__/scripts/utils/start_web_conditionally.py
|
||||
Restart=on-failure
|
||||
RestartSec=10
|
||||
|
||||
@@ -0,0 +1,923 @@
|
||||
"""Drive the real DisplayController.run() on a fake clock and record a trace.
|
||||
|
||||
The golden trace tests (test_run_loop_golden.py) use this to pin down what
|
||||
run() does today -- which mode is on the panel, for how long, and why it
|
||||
left -- so that the loop can be restructured (docs/RUN_LOOP_REDESIGN.md)
|
||||
without changing any of it.
|
||||
|
||||
What is real and what is fake
|
||||
-----------------------------
|
||||
Real: DisplayController itself (constructed through __init__, then run()),
|
||||
PluginExecutor (each screen's first frame still goes through its thread),
|
||||
the per-plugin display locks, and every controller method run() calls.
|
||||
|
||||
Fake, so the run is deterministic and takes milliseconds:
|
||||
|
||||
* the clock -- ``src.display_controller.time`` and ``datetime`` are replaced
|
||||
by one FakeClock; sleeping only advances it. Scripted events (an on-demand
|
||||
request, a WiFi notice, live content starting) fire as it passes them.
|
||||
* plugins -- FakePlugin, whose content, liveness and dynamic-duration answers
|
||||
are functions of the fake clock.
|
||||
* the plugin manager, cache, config service, display manager and sync
|
||||
manager -- in-memory stand-ins with no threads.
|
||||
* the Vegas coordinator -- FakeVegas implements only the contract the
|
||||
controller relies on (run_iteration() returning True when it ran its
|
||||
duration and False when interrupted, the interrupt and live checks it
|
||||
calls back into). The real coordinator spawns threads and renders a strip;
|
||||
driving it on the fake clock is part of stage 4 (Vegas as a Source).
|
||||
|
||||
The run ends when the fake clock passes the scenario's horizon: the clock
|
||||
raises StopRun, a BaseException, which run()'s ``except Exception`` lets
|
||||
through after its ``finally`` has run cleanup().
|
||||
|
||||
How the trace is read
|
||||
---------------------
|
||||
Everything observable is appended to one ordered event log. reduce_trace()
|
||||
folds it into screens: a screen starts at the first display() call of a
|
||||
loop pass (a "pass" is one call of the watchdog's loop_pass(), at the top of
|
||||
run()'s loop), or at the first follower / Vegas / WiFi / blank frame. Its
|
||||
exit reason is the first reason-bearing event logged before the next screen
|
||||
starts, else ``duration``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import threading
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
from typing import Any, Callable, Dict, List, Optional, Tuple
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from src.common.sync_manager import SyncRole
|
||||
from src.plugin_system.plugin_executor import PluginExecutor
|
||||
|
||||
GOLDEN_DIR = Path(__file__).parent / "fixtures" / "run_loop_golden"
|
||||
|
||||
#: Monday 2026-01-05 22:59:30 UTC. The schedule scenario's windows are set
|
||||
#: around 23:00; every other scenario has no schedule, so the date is moot.
|
||||
T0 = datetime(2026, 1, 5, 22, 59, 30, tzinfo=timezone.utc).timestamp()
|
||||
|
||||
#: Loop passes allowed without the clock moving before the run is called a
|
||||
#: spin. run() must sleep somewhere on every few passes.
|
||||
SPIN_LIMIT = 500
|
||||
|
||||
|
||||
class StopRun(BaseException):
|
||||
"""Ends a harness run. A BaseException so run()'s handlers pass it on."""
|
||||
|
||||
|
||||
class SpinError(BaseException):
|
||||
"""run() went round SPIN_LIMIT times without the clock moving.
|
||||
|
||||
A BaseException for the same reason as StopRun: run() would log and
|
||||
swallow anything less, and the test would see a short trace."""
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Clock
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class FakeClock:
|
||||
"""time.time/monotonic/perf_counter all read ``now``; sleep() advances it.
|
||||
|
||||
Alarms are (time, callback) pairs fired, in time order, by the sleep that
|
||||
carries the clock past them. Reaching the horizon raises StopRun.
|
||||
"""
|
||||
|
||||
def __init__(self, start: float, horizon: float):
|
||||
self.start = start
|
||||
self.now = start
|
||||
self.horizon = start + horizon
|
||||
self._alarms: List[Tuple[float, int, Callable[[], None]]] = []
|
||||
self._seq = 0
|
||||
self.passes_since_advance = 0
|
||||
|
||||
def rel(self) -> float:
|
||||
return self.now - self.start
|
||||
|
||||
def at(self, t: float, callback: Callable[[], None]) -> None:
|
||||
self._alarms.append((self.start + t, self._seq, callback))
|
||||
self._seq += 1
|
||||
self._alarms.sort()
|
||||
|
||||
def time(self) -> float:
|
||||
return self.now
|
||||
|
||||
def sleep(self, seconds: float) -> None:
|
||||
target = self.now + max(0.0, seconds)
|
||||
while self._alarms and self._alarms[0][0] <= target:
|
||||
when, _, callback = self._alarms.pop(0)
|
||||
self.now = max(self.now, when)
|
||||
callback()
|
||||
self.now = target
|
||||
if seconds > 0:
|
||||
self.passes_since_advance = 0
|
||||
if self.now >= self.horizon:
|
||||
raise StopRun()
|
||||
|
||||
def time_module(self) -> SimpleNamespace:
|
||||
return SimpleNamespace(time=self.time, monotonic=self.time,
|
||||
perf_counter=self.time, sleep=self.sleep)
|
||||
|
||||
def datetime_class(self):
|
||||
clock = self
|
||||
|
||||
class FakeDateTime(datetime):
|
||||
@classmethod
|
||||
def now(cls, tz=None): # type: ignore[override]
|
||||
return datetime.fromtimestamp(clock.now, tz or timezone.utc)
|
||||
|
||||
return FakeDateTime
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Fakes
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
class FakeCache:
|
||||
"""The in-memory slice of CacheManager that run() and its helpers use."""
|
||||
|
||||
def __init__(self):
|
||||
self.data: Dict[str, Any] = {}
|
||||
self.cache_dir = "/nonexistent/run-loop-harness"
|
||||
self._writes = 0
|
||||
self._written: Dict[str, int] = {}
|
||||
|
||||
def get(self, key, max_age=None, memory_ttl=None):
|
||||
return self.data.get(key)
|
||||
|
||||
def set(self, key, data, ttl=None):
|
||||
self.data[key] = data
|
||||
# Every write is a new file, as DiskCache's rename makes it.
|
||||
self._writes += 1
|
||||
self._written[key] = self._writes
|
||||
|
||||
def file_signature(self, key):
|
||||
"""CacheManager.file_signature: None without a file, else a value
|
||||
that changes with every write."""
|
||||
if key not in self.data:
|
||||
return None
|
||||
return (self._written.get(key, 0), 0, 0)
|
||||
|
||||
def delete(self, key):
|
||||
self.data.pop(key, None)
|
||||
|
||||
def clear_cache(self, key=None):
|
||||
if key is None:
|
||||
self.data.clear()
|
||||
else:
|
||||
self.data.pop(key, None)
|
||||
|
||||
def __getattr__(self, name):
|
||||
# Anything else (stats, cleanup hooks) is a no-op.
|
||||
return lambda *a, **k: None
|
||||
|
||||
|
||||
class FakeConfigService:
|
||||
def __init__(self, config):
|
||||
self.config = config
|
||||
|
||||
def get_config(self):
|
||||
return self.config
|
||||
|
||||
def subscribe(self, *a, **k):
|
||||
pass
|
||||
|
||||
def unsubscribe(self, *a, **k):
|
||||
pass
|
||||
|
||||
def shutdown(self):
|
||||
pass
|
||||
|
||||
|
||||
class FakeSync:
|
||||
"""A standalone sync manager whose follower state follows the script."""
|
||||
|
||||
role = SyncRole.STANDALONE
|
||||
|
||||
def __init__(self, harness: "RunLoopHarness"):
|
||||
self._h = harness
|
||||
self.follower_windows: List[Tuple[float, float]] = []
|
||||
|
||||
def is_follower_active(self) -> bool:
|
||||
t = self._h.clock.rel()
|
||||
return any(a <= t < b for a, b in self.follower_windows)
|
||||
|
||||
def get_latest_scroll_x(self):
|
||||
return None
|
||||
|
||||
def get_latest_frame(self):
|
||||
return "leader-frame"
|
||||
|
||||
def stop(self):
|
||||
pass
|
||||
|
||||
def __getattr__(self, name):
|
||||
return lambda *a, **k: None
|
||||
|
||||
|
||||
class FakeHealthTracker:
|
||||
"""Circuit breaker stand-in: opens after two consecutive failures and
|
||||
stays open (no wall-clock cooldown, which would not be deterministic)."""
|
||||
|
||||
def __init__(self, harness: "RunLoopHarness"):
|
||||
self._h = harness
|
||||
self.failures: Dict[str, int] = {}
|
||||
|
||||
def should_skip_plugin(self, plugin_id):
|
||||
skip = self.failures.get(plugin_id, 0) >= 2
|
||||
if skip:
|
||||
self._h.log("breaker-open", plugin_id, quiet=True)
|
||||
return skip
|
||||
|
||||
def record_success(self, plugin_id):
|
||||
self.failures[plugin_id] = 0
|
||||
|
||||
def record_failure(self, plugin_id, exc=None):
|
||||
self.failures[plugin_id] = self.failures.get(plugin_id, 0) + 1
|
||||
self._h.log("health-failure", plugin_id)
|
||||
|
||||
|
||||
class FakePluginManager:
|
||||
def __init__(self):
|
||||
self.plugins: Dict[str, Any] = {}
|
||||
self.plugin_manifests: Dict[str, Any] = {}
|
||||
self.plugin_last_update: Dict[str, float] = {}
|
||||
self.health_tracker = None
|
||||
self.resource_monitor = None
|
||||
self.state_manager = None
|
||||
self.plugin_executor = PluginExecutor()
|
||||
self.no_lock: set = set()
|
||||
self._locks: Dict[str, threading.Lock] = {}
|
||||
self.hangs: List[str] = []
|
||||
|
||||
def discover_plugins(self):
|
||||
return []
|
||||
|
||||
def discovered_plugin_ids(self):
|
||||
return set(self.plugins)
|
||||
|
||||
def load_plugin(self, plugin_id, force_enabled=False):
|
||||
return False
|
||||
|
||||
def get_plugin(self, plugin_id):
|
||||
return self.plugins.get(plugin_id)
|
||||
|
||||
def unload_plugin(self, plugin_id):
|
||||
self.plugins.pop(plugin_id, None)
|
||||
return True
|
||||
|
||||
def detach_plugin(self, plugin_id):
|
||||
return self.plugins.pop(plugin_id, None)
|
||||
|
||||
def unload_detached_plugin(self, plugin_id, plugin):
|
||||
return True
|
||||
|
||||
def reload_plugin(self, plugin_id):
|
||||
return False
|
||||
|
||||
def get_plugin_lock(self, plugin_id):
|
||||
if plugin_id in self.no_lock:
|
||||
return None # as when loading failed part-way
|
||||
return self._locks.setdefault(plugin_id, threading.Lock())
|
||||
|
||||
def record_display_hang(self, plugin_id, seconds):
|
||||
self.hangs.append(plugin_id)
|
||||
|
||||
def note_display_duration(self, plugin_id, seconds):
|
||||
pass
|
||||
|
||||
def run_scheduled_updates(self):
|
||||
pass
|
||||
|
||||
def run_scheduled_updates_with_changes(self):
|
||||
return []
|
||||
|
||||
def stop_update_worker(self):
|
||||
pass
|
||||
|
||||
|
||||
class FakePlugin:
|
||||
"""A plugin whose answers are functions of the harness clock.
|
||||
|
||||
Args:
|
||||
plugin_id: The plugin id.
|
||||
modes: Its display modes, registered in this order.
|
||||
duration: get_display_duration().
|
||||
content: ``content(t, mode) -> bool``: what display() returns.
|
||||
Defaults to always True.
|
||||
live: ``(start, end)`` seconds during which has_live_content() is
|
||||
True; get_live_modes() then names its modes ending in ``_live``.
|
||||
live_priority: has_live_priority().
|
||||
dynamic: Enables dynamic duration. Keys: ``cap`` (the plugin's cap),
|
||||
``cycle`` (get_cycle_duration()), ``complete_after`` (seconds
|
||||
after reset_cycle_state() that is_cycle_complete() turns True;
|
||||
None means never).
|
||||
needs_high_fps / enable_scrolling: Set as attributes only when given,
|
||||
since run() tests for their presence.
|
||||
raises: display() raises RuntimeError.
|
||||
first_frame_only: display() returns True on a screen's first frame
|
||||
and False on every later one.
|
||||
"""
|
||||
|
||||
def __init__(self, plugin_id: str, modes: List[str], duration: float = 30,
|
||||
content: Optional[Callable[[float, str], bool]] = None,
|
||||
live: Optional[Tuple[float, float]] = None,
|
||||
live_priority: bool = False,
|
||||
dynamic: Optional[Dict[str, Any]] = None,
|
||||
needs_high_fps: Optional[bool] = None,
|
||||
enable_scrolling: Optional[bool] = None,
|
||||
raises: bool = False,
|
||||
first_frame_only: bool = False):
|
||||
self.plugin_id = plugin_id
|
||||
self.modes = list(modes)
|
||||
self.duration = duration
|
||||
self.content = content
|
||||
self.live = live
|
||||
self.live_priority = live_priority
|
||||
self.dynamic = dynamic
|
||||
self.raises = raises
|
||||
self.first_frame_only = first_frame_only
|
||||
if needs_high_fps is not None:
|
||||
self.needs_high_fps = needs_high_fps
|
||||
if enable_scrolling is not None:
|
||||
self.enable_scrolling = enable_scrolling
|
||||
self._h: Optional["RunLoopHarness"] = None
|
||||
self._reset_at: Optional[float] = None
|
||||
|
||||
# -- display -----------------------------------------------------------
|
||||
def display(self, display_mode=None, force_clear=False):
|
||||
assert self._h is not None
|
||||
return self._h.on_display(self, display_mode or self.modes[0], force_clear)
|
||||
|
||||
def get_display_duration(self):
|
||||
return self.duration
|
||||
|
||||
# -- live --------------------------------------------------------------
|
||||
def _is_live(self) -> bool:
|
||||
if not self.live or self._h is None:
|
||||
return False
|
||||
t = self._h.clock.rel()
|
||||
return self.live[0] <= t < self.live[1]
|
||||
|
||||
def has_live_priority(self):
|
||||
return self.live_priority
|
||||
|
||||
def has_live_content(self):
|
||||
return self._is_live()
|
||||
|
||||
def get_live_modes(self):
|
||||
return [m for m in self.modes if m.endswith("_live")]
|
||||
|
||||
# -- dynamic duration ----------------------------------------------------
|
||||
def supports_dynamic_duration(self):
|
||||
return bool(self.dynamic)
|
||||
|
||||
def get_dynamic_duration_cap(self):
|
||||
return (self.dynamic or {}).get("cap")
|
||||
|
||||
def get_cycle_duration(self, display_mode=None):
|
||||
return (self.dynamic or {}).get("cycle")
|
||||
|
||||
def reset_cycle_state(self):
|
||||
assert self._h is not None
|
||||
self._reset_at = self._h.clock.rel()
|
||||
self._h.log("cycle-reset", self.plugin_id)
|
||||
|
||||
def is_cycle_complete(self):
|
||||
if not self.dynamic:
|
||||
return True
|
||||
after = self.dynamic.get("complete_after")
|
||||
if after is None or self._reset_at is None or self._h is None:
|
||||
return False
|
||||
done = self._h.clock.rel() - self._reset_at >= after
|
||||
if done:
|
||||
self._h.log("cycle-complete", self.plugin_id, quiet=True)
|
||||
return done
|
||||
|
||||
|
||||
class LegacyFakePlugin(FakePlugin):
|
||||
"""display() without a display_mode parameter, as older plugins have."""
|
||||
|
||||
def display(self, force_clear=False): # type: ignore[override]
|
||||
assert self._h is not None
|
||||
return self._h.on_display(self, self.modes[0], force_clear)
|
||||
|
||||
|
||||
class FakeVegas:
|
||||
"""The coordinator contract DisplayController relies on, nothing more.
|
||||
|
||||
run_iteration() renders frames at 125 Hz on the fake clock for
|
||||
``cycle`` seconds and returns True, or returns False as soon as the
|
||||
interrupt checker (every 10 frames, or at the next frame when the
|
||||
``urgent`` test says so) or the live-priority checker (every 0.25 s)
|
||||
asks it to yield -- the same cadence the real coordinator uses.
|
||||
A live-priority pause is lifted by the next call, as in the real one.
|
||||
"""
|
||||
|
||||
FRAME = 1.0 / 125
|
||||
INTERRUPT_EVERY = 10
|
||||
LIVE_EVERY = 0.25
|
||||
|
||||
def __init__(self, harness: "RunLoopHarness", cycle: float = 30.0,
|
||||
live_in_ticker: bool = False):
|
||||
self._h = harness
|
||||
self.cycle = cycle
|
||||
self.is_enabled = True
|
||||
self.vegas_config = SimpleNamespace(live_in_ticker=live_in_ticker)
|
||||
self.render_pipeline = None
|
||||
self._interrupt: Optional[Callable[[], bool]] = None
|
||||
self._urgent: Optional[Callable[[], bool]] = None
|
||||
self._live: Optional[Callable[[], Any]] = None
|
||||
self._paused_for_live = False
|
||||
|
||||
def set_live_priority_checker(self, fn):
|
||||
self._live = fn
|
||||
|
||||
def set_interrupt_checker(self, fn, check_interval=10, urgent=None):
|
||||
self._interrupt = fn
|
||||
self._urgent = urgent
|
||||
|
||||
def apply_pending_config_if_idle(self):
|
||||
pass
|
||||
|
||||
def cleanup(self):
|
||||
pass
|
||||
|
||||
def run_iteration(self) -> bool:
|
||||
h = self._h
|
||||
clock = h.clock
|
||||
if self._paused_for_live:
|
||||
self._paused_for_live = False
|
||||
h.log("vegas-start", None, quiet=True)
|
||||
start = clock.now
|
||||
last_live = None
|
||||
frames = 0
|
||||
while True:
|
||||
now = clock.now
|
||||
if (self._live and not self.vegas_config.live_in_ticker
|
||||
and (last_live is None or now - last_live >= self.LIVE_EVERY)):
|
||||
last_live = now
|
||||
if self._live():
|
||||
self._paused_for_live = True
|
||||
h.log("vegas-live")
|
||||
return False
|
||||
h.log("vegas-frame", None, quiet=True)
|
||||
clock.sleep(self.FRAME)
|
||||
frames += 1
|
||||
due = frames % self.INTERRUPT_EVERY == 0 or (self._urgent and self._urgent())
|
||||
if self._interrupt and due and self._interrupt():
|
||||
h.log("vegas-interrupt")
|
||||
return False
|
||||
if clock.now - start >= self.cycle:
|
||||
return True
|
||||
|
||||
|
||||
class FakeControlServer:
|
||||
"""The ControlServer surface run() uses (src/ipc/server.py), on the fake clock.
|
||||
|
||||
``post()`` queues a real QueuedCommand when the clock reaches ``t``, as a
|
||||
connection thread would. ``wait_for_command`` advances the clock to the
|
||||
first of: a command being queued, or the timeout -- the timed Event wait
|
||||
the real server does, without threads.
|
||||
"""
|
||||
|
||||
def __init__(self, harness: "RunLoopHarness"):
|
||||
self._h = harness
|
||||
self.queue: List[Any] = []
|
||||
self.posted: List[Any] = []
|
||||
self.closed = False
|
||||
|
||||
@property
|
||||
def has_pending(self) -> bool:
|
||||
return bool(self.queue)
|
||||
|
||||
def drain(self) -> List[Any]:
|
||||
out, self.queue = self.queue, []
|
||||
return out
|
||||
|
||||
def wait_for_command(self, timeout: float) -> bool:
|
||||
clock = self._h.clock
|
||||
target = clock.now + max(0.0, timeout)
|
||||
while not self.queue:
|
||||
nxt = clock._alarms[0][0] if clock._alarms else None
|
||||
if nxt is None or nxt > target:
|
||||
clock.sleep(target - clock.now)
|
||||
return bool(self.queue)
|
||||
clock.sleep(max(0.0, nxt - clock.now))
|
||||
return True
|
||||
|
||||
def post(self, t: float, cmd: str, args: Dict[str, Any], request_id: str = "sock"):
|
||||
from src.ipc.contract import AWAITED_COMMANDS, parse_args
|
||||
from src.ipc.server import CommandOutcome, QueuedCommand
|
||||
|
||||
command = QueuedCommand(
|
||||
request_id=request_id, cmd=cmd, args=parse_args(cmd, args), # type: ignore[arg-type]
|
||||
received_at=0.0,
|
||||
outcome=CommandOutcome() if cmd in AWAITED_COMMANDS else None)
|
||||
self.posted.append(command)
|
||||
|
||||
def enqueue():
|
||||
self._h.log("socket", cmd)
|
||||
self.queue.append(command)
|
||||
self._h.clock.at(t, enqueue)
|
||||
return command
|
||||
|
||||
def close(self):
|
||||
self.closed = True
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Harness
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
#: Events that can end a screen, as they appear in the trace.
|
||||
REASON_EVENTS = {
|
||||
"schedule-off", "schedule-on", "live", "live-ended", "on-demand-start",
|
||||
"on-demand-requested-stop", "on-demand-expired",
|
||||
"on-demand-no-modes-available", "vegas-live", "vegas-interrupt",
|
||||
"cycle-complete", "display-false",
|
||||
}
|
||||
|
||||
_SEGMENT_FOR = {
|
||||
"follower-frame": "<follower>",
|
||||
"vegas-frame": "<vegas>",
|
||||
"wifi": "<wifi>",
|
||||
"blank": "<off>",
|
||||
}
|
||||
|
||||
|
||||
class RunLoopHarness:
|
||||
"""Build a DisplayController on fakes, run it, and return its trace."""
|
||||
|
||||
def __init__(self, tmp_path: Path, horizon: float):
|
||||
self.clock = FakeClock(T0, horizon)
|
||||
self.events: List[Tuple[float, str, Any, Dict[str, Any]]] = []
|
||||
self.tmp_path = tmp_path
|
||||
# The controller keeps this very dict as self.config, so a scenario
|
||||
# can edit what run() reads live (durations, schedules). Values
|
||||
# __init__ copies out (global_dynamic_config) are set on the
|
||||
# controller instead.
|
||||
self.config: Dict[str, Any] = {
|
||||
"timezone": "UTC",
|
||||
"display": {"hardware": {"brightness": 90}},
|
||||
}
|
||||
self.cache = FakeCache()
|
||||
self.pm = FakePluginManager()
|
||||
self.sync = FakeSync(self)
|
||||
self.dm = self._display_manager()
|
||||
self._displayed_this_pass = False
|
||||
#: How long a plugin reload's own thread takes, on the fake clock.
|
||||
self.reload_seconds = 0.0
|
||||
self.controller = self._build()
|
||||
self.controller._spawn_plugin_reload = self._spawn_plugin_reload
|
||||
|
||||
def _spawn_plugin_reload(self, job) -> None:
|
||||
"""The plugin-reload thread, on the fake clock: the job runs
|
||||
``reload_seconds`` after it was started, meanwhile the render thread
|
||||
carries on (at once when that is 0)."""
|
||||
if self.reload_seconds <= 0:
|
||||
job.run(self.pm)
|
||||
else:
|
||||
self.clock.at(self.clock.rel() + self.reload_seconds, lambda: job.run(self.pm))
|
||||
|
||||
# -- event log -----------------------------------------------------------
|
||||
def log(self, kind: str, subject: Any = None, quiet: bool = False, **data):
|
||||
data["quiet"] = quiet
|
||||
self.events.append((round(self.clock.rel(), 3), kind, subject, data))
|
||||
|
||||
def on_display(self, plugin: FakePlugin, mode: str, force_clear: bool):
|
||||
first = not self._displayed_this_pass
|
||||
self._displayed_this_pass = True
|
||||
if plugin.raises:
|
||||
self.log("first" if first else "frame", mode, quiet=True,
|
||||
clear=bool(force_clear), result="raised")
|
||||
raise RuntimeError(f"{plugin.plugin_id} display() failed")
|
||||
if plugin.first_frame_only:
|
||||
result = first
|
||||
elif plugin.content is None:
|
||||
result = True
|
||||
else:
|
||||
result = bool(plugin.content(self.clock.rel(), mode))
|
||||
self.log("first" if first else "frame", mode, quiet=True,
|
||||
clear=bool(force_clear), result=result)
|
||||
return result
|
||||
|
||||
# -- construction ----------------------------------------------------------
|
||||
def _display_manager(self):
|
||||
dm = MagicMock(name="DisplayManager")
|
||||
dm.width = 128
|
||||
dm.height = 32
|
||||
dm._sync_render_allowed = False
|
||||
dm.set_brightness = MagicMock(side_effect=self._on_set_brightness)
|
||||
dm.update_display = MagicMock(side_effect=self._on_update_display)
|
||||
dm.get_font_height = MagicMock(return_value=8)
|
||||
return dm
|
||||
|
||||
def _on_set_brightness(self, value):
|
||||
self.log("brightness", value)
|
||||
return True
|
||||
|
||||
def _on_update_display(self):
|
||||
if getattr(self.dm, "_sync_render_allowed", False):
|
||||
self.log("follower-frame", None, quiet=True)
|
||||
elif not self.controller.is_display_active:
|
||||
self.log("blank", None, quiet=True)
|
||||
|
||||
def _build(self):
|
||||
from src import display_controller as dc_mod
|
||||
|
||||
clock = self.clock
|
||||
env = {"LEDMATRIX_HOT_RELOAD": "false", "EMULATOR": "true"}
|
||||
with patch.dict(os.environ, env), \
|
||||
patch.object(dc_mod, "time", clock.time_module()), \
|
||||
patch.object(dc_mod, "datetime", clock.datetime_class()), \
|
||||
patch.object(dc_mod, "ConfigManager", MagicMock()), \
|
||||
patch.object(dc_mod, "ConfigService", lambda **kw: FakeConfigService(self.config)), \
|
||||
patch.object(dc_mod, "CacheManager", lambda: self.cache), \
|
||||
patch.object(dc_mod, "DisplayManager", lambda config: self.dm), \
|
||||
patch.object(dc_mod, "FontManager", MagicMock()), \
|
||||
patch.object(dc_mod, "DisplaySyncManager", lambda **kw: self.sync), \
|
||||
patch("src.plugin_system.PluginManager", lambda **kw: self.pm), \
|
||||
patch("src.error_aggregator.start_error_snapshot_publisher", lambda cm: None), \
|
||||
patch("src.font_usage.start_font_usage_publisher", lambda *a, **k: None), \
|
||||
patch("src.plugin_system.plugin_runtime.start_plugin_runtime_publisher",
|
||||
lambda *a, **k: None), \
|
||||
patch("src.auto_update_setup.ensure_update_helper", lambda config: None):
|
||||
controller = dc_mod.DisplayController()
|
||||
|
||||
# __init__ wires real health/resource monitors; swap in the fake
|
||||
# breaker so failures and skips are deterministic.
|
||||
self.pm.health_tracker = FakeHealthTracker(self)
|
||||
self.pm.resource_monitor = None
|
||||
controller.wifi_status_file = self.tmp_path / "wifi_status.json"
|
||||
self._instrument(controller)
|
||||
return controller
|
||||
|
||||
def _instrument(self, dc) -> None:
|
||||
"""Log the controller's decisions without changing any of them.
|
||||
|
||||
Each wrapper calls straight through to the real method; only methods
|
||||
that exist both before and after the stage-1 extraction are wrapped,
|
||||
so the same harness records the same trace from either.
|
||||
"""
|
||||
h = self
|
||||
|
||||
def wrap(name, before, after):
|
||||
real = getattr(dc, name)
|
||||
|
||||
def wrapper(*args, **kwargs):
|
||||
token = before(*args, **kwargs)
|
||||
result = real(*args, **kwargs)
|
||||
after(token, *args, **kwargs)
|
||||
return result
|
||||
setattr(dc, name, wrapper)
|
||||
|
||||
wrap("_evaluate_schedule",
|
||||
lambda: dc.is_display_active,
|
||||
lambda was: (h.log("schedule-off") if was and not dc.is_display_active
|
||||
else h.log("schedule-on") if not was and dc.is_display_active
|
||||
else None))
|
||||
wrap("_activate_on_demand",
|
||||
lambda request: None,
|
||||
lambda _, request: h.log("on-demand-start", request.get("plugin_id"))
|
||||
if dc.on_demand_active else h.log("on-demand-error", dc.on_demand_last_error))
|
||||
wrap("_clear_on_demand",
|
||||
lambda reason=None: dc.on_demand_active,
|
||||
lambda was, reason=None: h.log(f"on-demand-{reason}") if was else None)
|
||||
wrap("_apply_live_priority",
|
||||
lambda mode: dc.current_display_mode,
|
||||
lambda prev, mode: (None if dc.current_display_mode == prev
|
||||
else h.log("live" if mode else "live-ended",
|
||||
dc.current_display_mode)))
|
||||
real_note = dc._note_empty_pass
|
||||
|
||||
def note_empty_pass():
|
||||
h.log("empty", dc.current_display_mode, quiet=True)
|
||||
return real_note()
|
||||
dc._note_empty_pass = note_empty_pass
|
||||
|
||||
real_wifi = dc._display_wifi_status_message
|
||||
|
||||
def display_wifi(status):
|
||||
shown = real_wifi(status)
|
||||
if shown:
|
||||
h.log("wifi", status.get("message"), quiet=True)
|
||||
return shown
|
||||
dc._display_wifi_status_message = display_wifi
|
||||
|
||||
# -- scenario setup ------------------------------------------------------
|
||||
def add_plugin(self, plugin: FakePlugin, lock: bool = True) -> FakePlugin:
|
||||
"""Register a plugin the way _register_loaded_plugin leaves things."""
|
||||
dc = self.controller
|
||||
plugin._h = self
|
||||
self.pm.plugins[plugin.plugin_id] = plugin
|
||||
if not lock:
|
||||
self.pm.no_lock.add(plugin.plugin_id)
|
||||
dc.plugin_display_modes[plugin.plugin_id] = list(plugin.modes)
|
||||
for mode in plugin.modes:
|
||||
if mode not in dc.available_modes:
|
||||
dc.available_modes.append(mode)
|
||||
dc.plugin_modes[mode] = plugin
|
||||
dc.mode_to_plugin_id[mode] = plugin.plugin_id
|
||||
return plugin
|
||||
|
||||
def add_mode_without_plugin(self, mode: str) -> None:
|
||||
self.controller.available_modes.append(mode)
|
||||
|
||||
def on_demand_request(self, t: float, request_id: str, action: str = "start", **fields):
|
||||
def post():
|
||||
self.log("request", f"{action}:{request_id}")
|
||||
self.cache.set("display_on_demand_request",
|
||||
{"request_id": request_id, "action": action, **fields})
|
||||
self.clock.at(t, post)
|
||||
|
||||
def restore_on_demand(self, plugin_id: str, mode: Optional[str] = None,
|
||||
duration: Optional[float] = None, pinned: bool = False,
|
||||
named_mode: Optional[str] = None):
|
||||
"""Start with an on-demand session resumed from the cache, as after
|
||||
a restart: the state _select_startup_plugins restores, then
|
||||
_populate_on_demand_modes_from_plugin, as __init__ calls it. A
|
||||
session that cannot resume is logged as ``on-demand-error``."""
|
||||
dc = self.controller
|
||||
dc._on_demand_named_mode = named_mode
|
||||
dc.on_demand_active = True
|
||||
dc.on_demand_plugin_id = plugin_id
|
||||
dc.on_demand_mode = mode
|
||||
dc.on_demand_duration = duration
|
||||
dc.on_demand_pinned = pinned
|
||||
dc.on_demand_requested_at = self.clock.now
|
||||
dc.on_demand_expires_at = self.clock.now + duration if duration else None
|
||||
dc.on_demand_status = 'active'
|
||||
dc.on_demand_schedule_override = True
|
||||
dc._populate_on_demand_modes_from_plugin()
|
||||
if dc.on_demand_status == 'error':
|
||||
self.log("on-demand-error", dc.on_demand_last_error)
|
||||
|
||||
def wifi_message(self, t: float, message: str, duration: float = 5):
|
||||
def write():
|
||||
self.log("wifi-file", message)
|
||||
self.controller.wifi_status_file.write_text(json.dumps(
|
||||
{"message": message, "timestamp": self.clock.now, "duration": duration}),
|
||||
encoding="utf-8")
|
||||
self.clock.at(t, write)
|
||||
|
||||
def control_socket(self) -> FakeControlServer:
|
||||
"""Serve the control socket (a FakeControlServer) for this run."""
|
||||
server = FakeControlServer(self)
|
||||
self.controller._control_server = server
|
||||
return server
|
||||
|
||||
def enable_vegas(self, cycle: float = 30.0, live_in_ticker: bool = False) -> FakeVegas:
|
||||
"""Install FakeVegas, wired up as _initialize_vegas_mode wires the real one."""
|
||||
dc = self.controller
|
||||
vegas = FakeVegas(self, cycle=cycle, live_in_ticker=live_in_ticker)
|
||||
vegas.set_live_priority_checker(dc._check_live_priority)
|
||||
vegas.set_interrupt_checker(
|
||||
lambda: dc._check_vegas_interrupt() or dc.sync_manager.is_follower_active(),
|
||||
check_interval=10, urgent=dc._control_command_pending)
|
||||
dc.vegas_coordinator = vegas
|
||||
return vegas
|
||||
|
||||
# -- running -------------------------------------------------------------
|
||||
def run(self) -> Dict[str, Any]:
|
||||
from src import display_controller as dc_mod
|
||||
from src import display_watchdog
|
||||
|
||||
clock = self.clock
|
||||
watchdog = display_watchdog.watchdog
|
||||
real_loop_pass = watchdog.loop_pass
|
||||
|
||||
def loop_pass():
|
||||
self._displayed_this_pass = False
|
||||
self.log("pass", None, quiet=True)
|
||||
clock.passes_since_advance += 1
|
||||
if clock.passes_since_advance > SPIN_LIMIT:
|
||||
raise SpinError(f"run() spun {SPIN_LIMIT} passes at t={clock.rel():.3f}")
|
||||
return real_loop_pass()
|
||||
|
||||
with patch.object(dc_mod, "time", clock.time_module()), \
|
||||
patch.object(dc_mod, "datetime", clock.datetime_class()), \
|
||||
patch.object(watchdog, "loop_pass", loop_pass):
|
||||
try:
|
||||
self.controller.run()
|
||||
except StopRun:
|
||||
pass
|
||||
else:
|
||||
# run() only returns after catching something itself.
|
||||
raise AssertionError(
|
||||
f"run() returned at t={clock.rel():.3f} before the horizon")
|
||||
return reduce_trace(self.events, round(clock.horizon - clock.start, 3))
|
||||
|
||||
|
||||
def reduce_trace(events, horizon: float) -> Dict[str, Any]:
|
||||
"""Fold the event log into screens and the notable events."""
|
||||
screens: List[Dict[str, Any]] = []
|
||||
notable: List[List[Any]] = []
|
||||
cur: Optional[Dict[str, Any]] = None
|
||||
# What happened in the current loop pass, for attributing an empty pass.
|
||||
shown_this_pass = False
|
||||
failed_this_pass = False
|
||||
breaker_this_pass = False
|
||||
|
||||
def start(t, mode, clear=None):
|
||||
nonlocal cur
|
||||
cur = {"t": t, "mode": mode, "frames": 0, "clear": clear, "exit": None}
|
||||
screens.append(cur)
|
||||
|
||||
for t, kind, subject, data in events:
|
||||
if not data.get("quiet"):
|
||||
notable.append([t, kind] + ([subject] if subject is not None else []))
|
||||
if kind == "pass":
|
||||
shown_this_pass = failed_this_pass = breaker_this_pass = False
|
||||
elif kind in ("first", "frame"):
|
||||
if kind == "first" or cur is None:
|
||||
start(t, subject, data["clear"])
|
||||
shown_this_pass = True
|
||||
cur["result"] = data["result"]
|
||||
cur["frames"] += 1
|
||||
if kind == "frame" and data["result"] is False and cur["exit"] is None:
|
||||
cur["exit"] = "display-false"
|
||||
elif kind in _SEGMENT_FOR:
|
||||
segment = _SEGMENT_FOR[kind]
|
||||
if cur is None or cur["mode"] != segment or cur["exit"] is not None:
|
||||
start(t, segment)
|
||||
cur["frames"] += 1
|
||||
elif kind == "vegas-start":
|
||||
start(t, "<vegas>")
|
||||
elif kind == "health-failure":
|
||||
failed_this_pass = True
|
||||
elif kind == "breaker-open":
|
||||
breaker_this_pass = True
|
||||
elif kind == "empty":
|
||||
if shown_this_pass and cur is not None and cur["exit"] is None:
|
||||
# display() ran and had nothing (False) or raised.
|
||||
cur["exit"] = "raised" if cur.get("result") == "raised" else "empty"
|
||||
else:
|
||||
# Never reached display(): no plugin, the breaker is open, or
|
||||
# the dispatch itself raised.
|
||||
start(t, subject)
|
||||
cur["exit"] = ("error" if failed_this_pass
|
||||
else "breaker" if breaker_this_pass else "no-plugin")
|
||||
elif kind in REASON_EVENTS and cur is not None and cur["exit"] is None:
|
||||
cur["exit"] = kind
|
||||
|
||||
rows = []
|
||||
for i, screen in enumerate(screens):
|
||||
end = screens[i + 1]["t"] if i + 1 < len(screens) else horizon
|
||||
nxt = screens[i + 1] if i + 1 < len(screens) else None
|
||||
# A WiFi notice logs no event at the moment it takes the panel (the
|
||||
# file is written earlier), so a screen followed by one is labelled
|
||||
# "wifi". Its duration column shows whether it was cut short.
|
||||
exit_reason = screen["exit"] or (
|
||||
"horizon" if nxt is None
|
||||
else "wifi" if nxt["mode"] == "<wifi>" and screen["mode"] != "<wifi>"
|
||||
else "duration")
|
||||
rows.append([screen["t"], screen["mode"], round(end - screen["t"], 3),
|
||||
exit_reason, screen["frames"], screen["clear"]])
|
||||
return {"screens": rows, "events": notable}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Golden files
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def dump_golden(trace: Dict[str, Any]) -> str:
|
||||
"""One screen or event per line, so a diff points at the row that moved."""
|
||||
def block(name, rows, last=False):
|
||||
end = "" if last else ","
|
||||
if not rows:
|
||||
return [f' "{name}": []{end}']
|
||||
return [f' "{name}": [',
|
||||
",\n".join(" " + json.dumps(row) for row in rows),
|
||||
f" ]{end}"]
|
||||
|
||||
lines = (["{"] + block("screens", trace["screens"])
|
||||
+ block("events", trace["events"], last=True) + ["}"])
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def check_golden(name: str, trace: Dict[str, Any]) -> None:
|
||||
"""Compare against test/fixtures/run_loop_golden/<name>.json.
|
||||
|
||||
LEDMATRIX_REGEN_GOLDEN=1 rewrites the file instead. Only do that for a
|
||||
deliberate behaviour change, and say why in the commit.
|
||||
"""
|
||||
path = GOLDEN_DIR / f"{name}.json"
|
||||
text = dump_golden(trace)
|
||||
if os.environ.get("LEDMATRIX_REGEN_GOLDEN") == "1":
|
||||
GOLDEN_DIR.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(text, encoding="utf-8", newline="\n")
|
||||
return
|
||||
assert path.exists(), f"no golden trace {path}; run with LEDMATRIX_REGEN_GOLDEN=1"
|
||||
expected = json.loads(path.read_text(encoding="utf-8"))
|
||||
actual = json.loads(text)
|
||||
if actual != expected:
|
||||
import difflib
|
||||
diff = "\n".join(difflib.unified_diff(
|
||||
dump_golden(expected).splitlines(), text.splitlines(),
|
||||
"golden", "actual", lineterm="", n=2))
|
||||
raise AssertionError(f"run() trace for {name!r} changed:\n{diff}")
|
||||
@@ -328,6 +328,21 @@ def _hermetic_control_socket(monkeypatch):
|
||||
monkeypatch.setenv(SOCKET_PATH_ENV, 'off')
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _hermetic_unit_refresh(monkeypatch, tmp_path_factory):
|
||||
"""Keep updates' systemd unit refresh off the host.
|
||||
|
||||
perform_core_update runs web_interface/unit_refresh.py after any update
|
||||
that moves HEAD, and several tests run the real one against a test clone.
|
||||
On a device -- or a machine where install_service.sh was tried out -- it
|
||||
would compare the clone's templates with the real /etc/systemd/system and
|
||||
run the real sudo helper. Point it at a folder that does not exist: no units
|
||||
installed, nothing to do. The unit refresh tests pass their own.
|
||||
"""
|
||||
from web_interface import unit_refresh
|
||||
monkeypatch.setattr(unit_refresh, 'SYSTEMD_DIR', str(tmp_path_factory.getbasetemp() / 'no-systemd'))
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def reset_logging():
|
||||
"""Reset logging configuration before each test."""
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user