Merge branch 'main' into claude/phase2-one-sudoers-generator

This commit is contained in:
Chuck
2026-09-24 15:51:04 -04:00
committed by GitHub
13 changed files with 441 additions and 12 deletions
+118
View File
@@ -0,0 +1,118 @@
"""A stale cache record is recognised from its header, without parsing it.
The sports plugins cache whole season schedules -- 53MB for MLB, 18MB for NHL.
When one expired, DiskCache.get parsed all of it (~1.8s of orjson.loads on a
Pi 4, GIL held, the whole display frozen) only to find the timestamp too old
and throw the result away. CacheManager.set now writes timestamp and ttl ahead
of the data, and DiskCache.get reads them from the first bytes of the file.
"""
import json
import time
from types import SimpleNamespace
import pytest
from src.cache import disk_cache as disk_cache_module
from src.cache.disk_cache import DiskCache, _stale_from_head
from src.common import json_body
@pytest.fixture
def disk(tmp_path):
return DiskCache(cache_dir=str(tmp_path))
@pytest.fixture
def parses(monkeypatch):
"""Count full parses of cache files."""
calls = []
real = disk_cache_module._loads
def counting(raw):
calls.append(len(raw))
return real(raw)
monkeypatch.setattr(disk_cache_module, "_loads", counting)
return calls
def _header_first(age=0.0, ttl=None, events=100):
record = {"timestamp": time.time() - age}
if ttl is not None:
record["ttl"] = ttl
record["data"] = {"events": [{"id": n, "name": "x" * 50} for n in range(events)]}
return record
def test_cache_manager_writes_the_header_first(monkeypatch):
from src.cache_manager import CacheManager
written = {}
manager = CacheManager.__new__(CacheManager)
monkeypatch.setattr(manager, "save_cache",
lambda key, record: written.update({key: record}),
raising=False)
CacheManager.set(manager, "k", {"events": []}, ttl=60)
assert list(written["k"]) == ["timestamp", "ttl", "data"]
CacheManager.set(manager, "k", {"events": []})
assert list(written["k"]) == ["timestamp", "data"]
def test_a_stale_record_is_not_parsed(disk, parses):
disk.set("season", _header_first(age=600))
assert disk.get("season", max_age=300) is None
assert parses == []
def test_a_fresh_record_is_parsed_and_returned(disk, parses):
disk.set("season", _header_first(age=10))
record = disk.get("season", max_age=300)
assert record["data"]["events"][0]["id"] == 0
assert len(parses) == 1
def test_the_entry_ttl_wins_over_max_age(disk, parses):
disk.set("long", _header_first(age=600, ttl=3600))
assert disk.get("long", max_age=300) is not None # ttl says fresh
disk.set("short", _header_first(age=60, ttl=30))
parses.clear()
assert disk.get("short", max_age=300) is None # ttl says stale
assert parses == []
def test_no_limit_means_never_stale(disk):
disk.set("forever", _header_first(age=10 ** 7))
assert disk.get("forever", max_age=None) is not None
def test_older_files_with_data_first_still_work(disk, parses):
# Records written before the header moved: parsed in full, as before.
disk.set("legacy_fresh", {"data": {"v": 1}, "timestamp": time.time()})
disk.set("legacy_stale", {"data": {"v": 1}, "timestamp": time.time() - 600})
assert disk.get("legacy_fresh", max_age=300)["data"] == {"v": 1}
assert disk.get("legacy_stale", max_age=300) is None
assert len(parses) == 2
@pytest.mark.parametrize("head, stale", [
(b'{"timestamp":100.0,"data":{}}', True),
(b'{"timestamp": 100.0, "ttl": 1000, "data": {}}', False), # stdlib spacing
(b'{"timestamp":1e2,"ttl":5,"data":1}', True),
(b'{"timestamp":100.0}', True),
(b'{"data":{},"timestamp":100.0}', False), # unknown layout
(b'{"timestamp":"100.0","data":{}}', False), # string: parse it
(b'', False),
])
def test_reading_the_header(head, stale):
assert _stale_from_head(head, 300, now=1000.0) is stale
def test_response_json_prefers_orjson_and_falls_back():
payload = {"events": [1, 2, 3]}
response = SimpleNamespace(content=json.dumps(payload).encode(),
json=lambda: pytest.fail("used the slow path"))
if json_body.orjson is None:
pytest.skip("orjson not installed")
assert json_body.response_json(response) == payload
# A response object without bytes content (a test double) still works.
assert json_body.response_json(SimpleNamespace(json=lambda: payload)) == payload
+110
View File
@@ -0,0 +1,110 @@
"""redact_credentials must stay linear in the length of its input.
Regressions under test, both quadratic regexes in src/redaction.py:
- The URL-userinfo pattern (`scheme://user:password@`) could start a match at
every letter of a run of scheme characters, and each attempt read to the end
of the run looking for `://`: 1.6s for a 20k-character run.
- The Authorization-header pattern had two `\\s*` separated only by an
optional quote, so a header followed by whitespace and no credential tried
every split of that whitespace between them: 8s for 20k spaces.
The display service redacts every message, stack trace and context value it
publishes in the error snapshot, and re.sub holds the GIL throughout, so an
exception quoting a hex digest or a long ID stalled the render loop with it.
test_error_snapshot_cross_process.py's snapshot-size test spent 140s here.
The fixed patterns have to redact exactly what the old ones did.
"""
import time
import pytest
from src.redaction import redact_credentials
# Each timed input took seconds before the fix and takes about a millisecond
# after it; the bound leaves CI plenty of headroom while still failing on a
# quadratic pattern.
_TIME_LIMIT = 1.0
def _timed(text):
start = time.perf_counter()
result = redact_credentials(text)
return result, time.perf_counter() - start
class TestUrlUserinfo:
@pytest.mark.parametrize("text,expected", [
("401 for https://user:hunter2@example.com/api",
"401 for https://user:<redacted>@example.com/api"),
("HTTPS://USER:HUNTER2@EXAMPLE.COM",
"HTTPS://USER:<redacted>@EXAMPLE.COM"),
("git+ssh://deploy:hunter2@host/repo",
"git+ssh://deploy:<redacted>@host/repo"),
# The scheme starts after digits or +.- in the same run. Those
# characters must survive, and the password must still go.
("1http://user:hunter2@host", "1http://user:<redacted>@host"),
("+.-http://user:hunter2@host", "+.-http://user:<redacted>@host"),
("a1+http://user:hunter2@host", "a1+http://user:<redacted>@host"),
("see a://u:first@b and c://v:second@d",
"see a://u:<redacted>@b and c://v:<redacted>@d"),
])
def test_password_is_redacted_and_the_rest_kept(self, text, expected):
assert redact_credentials(text) == expected
def test_a_url_without_a_password_is_untouched(self):
text = "GET https://user@example.com/path failed"
assert redact_credentials(text) == text
class TestAuthorizationHeader:
@pytest.mark.parametrize("text,expected", [
("Authorization: Bearer eyJ.SECRET.sig", "Authorization: Bearer <redacted>"),
("Proxy-Authorization: Basic dXNlcg==", "Proxy-Authorization: Basic <redacted>"),
("authorization: barecredential", "authorization: <redacted>"),
# Whitespace and an opening quote around the value, in either order.
('authorization=" Bearer tok"', 'authorization=" Bearer <redacted>"'),
("authorization: ' tok'", "authorization: ' <redacted>'"),
("authorization:\n\tBearer tok", "authorization:\n\tBearer <redacted>"),
])
def test_credential_is_redacted_and_the_rest_kept(self, text, expected):
assert redact_credentials(text) == expected
@pytest.mark.parametrize("text", ["authorization: ", "authorization: , next"])
def test_a_header_without_a_credential_is_untouched(self, text):
assert redact_credentials(text) == text
class TestLinearTime:
@pytest.mark.parametrize("unit", ["x", "0123456789abcdef", "1a", "a+", "1"])
def test_long_scheme_character_runs(self, unit):
text = (unit * 50_000)[:50_000]
result, elapsed = _timed(text)
assert result == text
assert elapsed < _TIME_LIMIT, f"{elapsed:.2f}s to redact {len(text)} chars of {unit!r}"
def test_a_credential_after_a_long_run_is_still_found(self):
run = "ab12" * 10_000
result, elapsed = _timed(f"{run} https://user:hunter2@example.com")
assert result == f"{run} https://user:<redacted>@example.com"
assert elapsed < _TIME_LIMIT
@pytest.mark.parametrize("header,whitespace", [
("authorization:", " "),
("Proxy-Authorization:", "\t"),
("authorization=", "\n"),
])
def test_a_header_followed_by_long_whitespace(self, header, whitespace):
text = header + whitespace * 20_000 + ","
result, elapsed = _timed(text)
assert result == text
assert elapsed < _TIME_LIMIT, (
f"{elapsed:.2f}s to redact {header!r} and {len(text) - len(header)} more chars")
def test_a_credential_after_long_whitespace_is_still_found(self):
gap = " " * 20_000
result, elapsed = _timed(f"authorization:{gap}Bearer tok")
assert result == f"authorization:{gap}Bearer <redacted>"
assert elapsed < _TIME_LIMIT
@@ -0,0 +1,25 @@
"""Names two rarely-run api_v3 paths call must exist.
Both slipped through because nothing exercised them: the Pixlet editor's
stop route only restarts the display after a SIGKILL, and the Starlark
device-location resolver only builds a cache manager when the web app has
not set one. Either raised NameError when it finally ran.
"""
from unittest.mock import patch
from test._api_v3_test_helpers import api_v3_module # noqa: F401
def test_the_editor_stop_route_can_restart_the_display():
from web_interface.blueprints.api_v3 import starlark
assert callable(starlark._run_systemctl_command)
def test_the_device_location_resolver_builds_without_a_cache_manager(api_v3_module):
pkg = api_v3_module
with patch.object(pkg.api_v3, 'cache_manager', None, create=True), \
patch.object(pkg, '_starlark_device_location', None):
resolver = pkg._get_starlark_device_location()
assert resolver.cache_manager is None