mirror of
https://github.com/ChuckBuilds/LEDMatrix.git
synced 2026-10-10 17:16:36 +00:00
Compare commits
3
Commits
v3.6.0
...
9be61e7e0c
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9be61e7e0c | ||
|
|
9171d82562 | ||
|
|
eaaea93c77 |
@@ -120,6 +120,13 @@ floor on the release that ships them):
|
|||||||
when the count is only known to the display service.
|
when the count is only known to the display service.
|
||||||
- The Logs tab has a **Plugin errors** panel: per-plugin counts, repeating
|
- The Logs tab has a **Plugin errors** panel: per-plugin counts, repeating
|
||||||
errors and a Clear button.
|
errors and a Clear button.
|
||||||
|
- Credential redaction in exception text (`src/redaction.py`) takes time
|
||||||
|
proportional to the text, not its square. Two patterns were quadratic: URL
|
||||||
|
`user:password@`, on a long unbroken run of letters or digits (a hex digest,
|
||||||
|
an ID), and `Authorization:` followed by a long run of whitespace. Either
|
||||||
|
used to stall every thread of the display service for up to seconds each
|
||||||
|
time the snapshot was published: about 0.5s for 20k characters of hex, 8s
|
||||||
|
for 20k spaces. What gets redacted is unchanged.
|
||||||
|
|
||||||
### Removed
|
### Removed
|
||||||
|
|
||||||
|
|||||||
+16
-3
@@ -24,8 +24,13 @@ _REDACT_CREDENTIAL = re.compile(
|
|||||||
# silently leak the ones nobody thought of. Not covered by the generic pattern
|
# silently leak the ones nobody thought of. Not covered by the generic pattern
|
||||||
# above, whose value part stops at whitespace and so would keep the credential
|
# above, whose value part stops at whitespace and so would keep the credential
|
||||||
# once a space follows the scheme.
|
# once a space follows the scheme.
|
||||||
|
#
|
||||||
|
# The opening quote and the whitespace after it are one optional unit. Written
|
||||||
|
# `\s*["\']?\s*`, a whitespace run with no quote in it could be split between
|
||||||
|
# the two `\s*` in every possible way, and a header with no credential after
|
||||||
|
# it tried them all: quadratic, 8s for 20k spaces.
|
||||||
_REDACT_AUTH_HEADER = re.compile(
|
_REDACT_AUTH_HEADER = re.compile(
|
||||||
r'((?:proxy-)?authorization["\']?\s*[=:]\s*["\']?\s*'
|
r'((?:proxy-)?authorization["\']?\s*[=:]\s*(?:["\']\s*)?'
|
||||||
r'(?:[A-Za-z][\w.+-]*[ \t]+)?)' # optional scheme name, kept
|
r'(?:[A-Za-z][\w.+-]*[ \t]+)?)' # optional scheme name, kept
|
||||||
r'([^\s,"\'<>}]+)', # the credential, redacted
|
r'([^\s,"\'<>}]+)', # the credential, redacted
|
||||||
re.IGNORECASE,
|
re.IGNORECASE,
|
||||||
@@ -34,8 +39,16 @@ _REDACT_AUTH_HEADER = re.compile(
|
|||||||
# Credentials embedded in a URL: https://user:password@host. requests quotes
|
# Credentials embedded in a URL: https://user:password@host. requests quotes
|
||||||
# the full URL in its exceptions, so this is a realistic leak. The username is
|
# the full URL in its exceptions, so this is a realistic leak. The username is
|
||||||
# kept -- it identifies which account failed without being the secret.
|
# kept -- it identifies which account failed without being the secret.
|
||||||
_REDACT_URL_USERINFO = re.compile(r'([a-z][a-z0-9+.-]*://[^/\s:@]+:)([^/\s@]+)(@)',
|
#
|
||||||
re.IGNORECASE)
|
# A match may only start where a run of scheme characters starts. Unanchored,
|
||||||
|
# `[a-z][a-z0-9+.-]*://` was tried from every letter of a long run (a hex
|
||||||
|
# digest, an ID, a blob of response body), each attempt reading to the end of
|
||||||
|
# the run: quadratic, 1.6s for 20k characters, all of it holding the GIL.
|
||||||
|
# Leading digits and `+.-` sit inside group 1 so the substitution puts them
|
||||||
|
# back; the scheme proper still has to start with a letter.
|
||||||
|
_REDACT_URL_USERINFO = re.compile(
|
||||||
|
r'((?<![a-z0-9+.-])[0-9+.-]*[a-z][a-z0-9+.-]*://[^/\s:@]+:)([^/\s@]+)(@)',
|
||||||
|
re.IGNORECASE)
|
||||||
|
|
||||||
|
|
||||||
def redact_credentials(text: str) -> str:
|
def redact_credentials(text: str) -> str:
|
||||||
|
|||||||
@@ -0,0 +1,110 @@
|
|||||||
|
"""redact_credentials must stay linear in the length of its input.
|
||||||
|
|
||||||
|
Regressions under test, both quadratic regexes in src/redaction.py:
|
||||||
|
|
||||||
|
- The URL-userinfo pattern (`scheme://user:password@`) could start a match at
|
||||||
|
every letter of a run of scheme characters, and each attempt read to the end
|
||||||
|
of the run looking for `://`: 1.6s for a 20k-character run.
|
||||||
|
- The Authorization-header pattern had two `\\s*` separated only by an
|
||||||
|
optional quote, so a header followed by whitespace and no credential tried
|
||||||
|
every split of that whitespace between them: 8s for 20k spaces.
|
||||||
|
|
||||||
|
The display service redacts every message, stack trace and context value it
|
||||||
|
publishes in the error snapshot, and re.sub holds the GIL throughout, so an
|
||||||
|
exception quoting a hex digest or a long ID stalled the render loop with it.
|
||||||
|
test_error_snapshot_cross_process.py's snapshot-size test spent 140s here.
|
||||||
|
|
||||||
|
The fixed patterns have to redact exactly what the old ones did.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import time
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from src.redaction import redact_credentials
|
||||||
|
|
||||||
|
# Each timed input took seconds before the fix and takes about a millisecond
|
||||||
|
# after it; the bound leaves CI plenty of headroom while still failing on a
|
||||||
|
# quadratic pattern.
|
||||||
|
_TIME_LIMIT = 1.0
|
||||||
|
|
||||||
|
|
||||||
|
def _timed(text):
|
||||||
|
start = time.perf_counter()
|
||||||
|
result = redact_credentials(text)
|
||||||
|
return result, time.perf_counter() - start
|
||||||
|
|
||||||
|
|
||||||
|
class TestUrlUserinfo:
|
||||||
|
@pytest.mark.parametrize("text,expected", [
|
||||||
|
("401 for https://user:hunter2@example.com/api",
|
||||||
|
"401 for https://user:<redacted>@example.com/api"),
|
||||||
|
("HTTPS://USER:HUNTER2@EXAMPLE.COM",
|
||||||
|
"HTTPS://USER:<redacted>@EXAMPLE.COM"),
|
||||||
|
("git+ssh://deploy:hunter2@host/repo",
|
||||||
|
"git+ssh://deploy:<redacted>@host/repo"),
|
||||||
|
# The scheme starts after digits or +.- in the same run. Those
|
||||||
|
# characters must survive, and the password must still go.
|
||||||
|
("1http://user:hunter2@host", "1http://user:<redacted>@host"),
|
||||||
|
("+.-http://user:hunter2@host", "+.-http://user:<redacted>@host"),
|
||||||
|
("a1+http://user:hunter2@host", "a1+http://user:<redacted>@host"),
|
||||||
|
("see a://u:first@b and c://v:second@d",
|
||||||
|
"see a://u:<redacted>@b and c://v:<redacted>@d"),
|
||||||
|
])
|
||||||
|
def test_password_is_redacted_and_the_rest_kept(self, text, expected):
|
||||||
|
assert redact_credentials(text) == expected
|
||||||
|
|
||||||
|
def test_a_url_without_a_password_is_untouched(self):
|
||||||
|
text = "GET https://user@example.com/path failed"
|
||||||
|
assert redact_credentials(text) == text
|
||||||
|
|
||||||
|
|
||||||
|
class TestAuthorizationHeader:
|
||||||
|
@pytest.mark.parametrize("text,expected", [
|
||||||
|
("Authorization: Bearer eyJ.SECRET.sig", "Authorization: Bearer <redacted>"),
|
||||||
|
("Proxy-Authorization: Basic dXNlcg==", "Proxy-Authorization: Basic <redacted>"),
|
||||||
|
("authorization: barecredential", "authorization: <redacted>"),
|
||||||
|
# Whitespace and an opening quote around the value, in either order.
|
||||||
|
('authorization=" Bearer tok"', 'authorization=" Bearer <redacted>"'),
|
||||||
|
("authorization: ' tok'", "authorization: ' <redacted>'"),
|
||||||
|
("authorization:\n\tBearer tok", "authorization:\n\tBearer <redacted>"),
|
||||||
|
])
|
||||||
|
def test_credential_is_redacted_and_the_rest_kept(self, text, expected):
|
||||||
|
assert redact_credentials(text) == expected
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("text", ["authorization: ", "authorization: , next"])
|
||||||
|
def test_a_header_without_a_credential_is_untouched(self, text):
|
||||||
|
assert redact_credentials(text) == text
|
||||||
|
|
||||||
|
|
||||||
|
class TestLinearTime:
|
||||||
|
@pytest.mark.parametrize("unit", ["x", "0123456789abcdef", "1a", "a+", "1"])
|
||||||
|
def test_long_scheme_character_runs(self, unit):
|
||||||
|
text = (unit * 50_000)[:50_000]
|
||||||
|
result, elapsed = _timed(text)
|
||||||
|
assert result == text
|
||||||
|
assert elapsed < _TIME_LIMIT, f"{elapsed:.2f}s to redact {len(text)} chars of {unit!r}"
|
||||||
|
|
||||||
|
def test_a_credential_after_a_long_run_is_still_found(self):
|
||||||
|
run = "ab12" * 10_000
|
||||||
|
result, elapsed = _timed(f"{run} https://user:hunter2@example.com")
|
||||||
|
assert result == f"{run} https://user:<redacted>@example.com"
|
||||||
|
assert elapsed < _TIME_LIMIT
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("header,whitespace", [
|
||||||
|
("authorization:", " "),
|
||||||
|
("Proxy-Authorization:", "\t"),
|
||||||
|
("authorization=", "\n"),
|
||||||
|
])
|
||||||
|
def test_a_header_followed_by_long_whitespace(self, header, whitespace):
|
||||||
|
text = header + whitespace * 20_000 + ","
|
||||||
|
result, elapsed = _timed(text)
|
||||||
|
assert result == text
|
||||||
|
assert elapsed < _TIME_LIMIT, (
|
||||||
|
f"{elapsed:.2f}s to redact {header!r} and {len(text) - len(header)} more chars")
|
||||||
|
|
||||||
|
def test_a_credential_after_long_whitespace_is_still_found(self):
|
||||||
|
gap = " " * 20_000
|
||||||
|
result, elapsed = _timed(f"authorization:{gap}Bearer tok")
|
||||||
|
assert result == f"authorization:{gap}Bearer <redacted>"
|
||||||
|
assert elapsed < _TIME_LIMIT
|
||||||
Reference in New Issue
Block a user