mirror of
https://github.com/ChuckBuilds/LEDMatrix.git
synced 2026-10-04 14:25:08 +00:00
* fix(redaction): make URL-userinfo redaction linear, not quadratic _REDACT_URL_USERINFO could start a match at every letter of a run of scheme characters, and each attempt read to the end of the run looking for `://`. On a long unbroken run of letters or digits (a hex digest, an ID, part of a response body) that is quadratic: 1.6s for 20k characters. The display service redacts every message, stack trace and context value it publishes in the error snapshot, holding the aggregator lock, and re.sub holds the GIL for the whole call, so one such exception stalled every thread, render loop included (~0.5s measured for 20k chars of hex). It also made test_snapshot_stays_small the slowest test in the suite by far: 142s of a 383s run, 139s of it in this one regex. A match may now only start where a run of scheme characters starts (negative lookbehind). Leading digits and `+.-` are captured in group 1 so the substitution restores them, and the scheme still has to start with a letter, so what gets redacted is unchanged: old and new output were identical on 300k fuzzed inputs. 20k chars now take ~0.5ms, 200k ~6ms, and test_snapshot_stays_small takes 0.8s. test/test_redaction.py pins the exact output for schemes that begin after digits or `+.-`, and bounds 50k-character runs at 1s; against the old pattern those timing tests fail at 3-11s each. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01KMXdS2S4NXTJ8ET96GymhK * fix(redaction): make Authorization-header redaction linear too _REDACT_AUTH_HEADER matched the value's opening as `\s*["\']?\s*`: two `\s*` separated only by an optional quote. With no quote, a whitespace run could be split between them in every possible way, and when no credential followed (end of text, or `,` `"` `<` ...) the engine tried them all before giving up: quadratic, 8s for `authorization:` and 20k spaces, 17s with `Proxy-Authorization:` (tried again at the inner `authorization`). Same stall as the URL pattern: re.sub holds the GIL, and the display service redacts everything it publishes. The quote and the whitespace after it are now one optional unit, `\s*(?:["\']\s*)?`, which matches the same strings with only one way to split them. Output is identical to the old pattern on 300k fuzzed inputs; 20k spaces now take ~1.6ms. A scan of all three redaction patterns over prefix/run/suffix shapes finds none left that scales superlinearly. test/test_redaction.py pins exact output for quoted, tabbed, multi-line and credential-less headers, and bounds header + 20k whitespace at 1s; against the previous pattern those fail at 8-17s each. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01KMXdS2S4NXTJ8ET96GymhK --------- Co-authored-by: Claude <noreply@anthropic.com>
63 lines
3.1 KiB
Python
63 lines
3.1 KiB
Python
"""Credential redaction for text that leaves the process that produced it.
|
|
|
|
Kept free of Flask so the display service can redact what it publishes (see
|
|
src/error_aggregator.py) as well as the web interface what it returns.
|
|
"""
|
|
|
|
import re
|
|
|
|
# Credentials that turn up inside exception text. A requests error quotes the
|
|
# URL it failed on, and plugins that authenticate by query string put their key
|
|
# there, so echoing an exception verbatim can hand out an API key. Redact the
|
|
# value, keep the parameter name -- knowing *which* credential was involved is
|
|
# part of the diagnosis.
|
|
_REDACT_CREDENTIAL = re.compile(
|
|
r'((?:api[_-]?key|access[_-]?token|auth|apikey|key|passwd|password|pwd|'
|
|
r'secret|sig|signature|token)["\']?\s*[=:]\s*["\']?)([^\s&"\'<>,}]+)',
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
# `Authorization: <scheme> <credential>`. The scheme name is kept because it
|
|
# says which kind of credential failed; the credential goes. Any scheme
|
|
# matches, not a fixed list: ApiKey, Negotiate, NTLM, AWS4-HMAC-SHA256 and
|
|
# whatever a plugin's API invents next are all credentials, and a list would
|
|
# silently leak the ones nobody thought of. Not covered by the generic pattern
|
|
# above, whose value part stops at whitespace and so would keep the credential
|
|
# once a space follows the scheme.
|
|
#
|
|
# The opening quote and the whitespace after it are one optional unit. Written
|
|
# `\s*["\']?\s*`, a whitespace run with no quote in it could be split between
|
|
# the two `\s*` in every possible way, and a header with no credential after
|
|
# it tried them all: quadratic, 8s for 20k spaces.
|
|
_REDACT_AUTH_HEADER = re.compile(
|
|
r'((?:proxy-)?authorization["\']?\s*[=:]\s*(?:["\']\s*)?'
|
|
r'(?:[A-Za-z][\w.+-]*[ \t]+)?)' # optional scheme name, kept
|
|
r'([^\s,"\'<>}]+)', # the credential, redacted
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
# Credentials embedded in a URL: https://user:password@host. requests quotes
|
|
# the full URL in its exceptions, so this is a realistic leak. The username is
|
|
# kept -- it identifies which account failed without being the secret.
|
|
#
|
|
# A match may only start where a run of scheme characters starts. Unanchored,
|
|
# `[a-z][a-z0-9+.-]*://` was tried from every letter of a long run (a hex
|
|
# digest, an ID, a blob of response body), each attempt reading to the end of
|
|
# the run: quadratic, 1.6s for 20k characters, all of it holding the GIL.
|
|
# Leading digits and `+.-` sit inside group 1 so the substitution puts them
|
|
# back; the scheme proper still has to start with a letter.
|
|
_REDACT_URL_USERINFO = re.compile(
|
|
r'((?<![a-z0-9+.-])[0-9+.-]*[a-z][a-z0-9+.-]*://[^/\s:@]+:)([^/\s@]+)(@)',
|
|
re.IGNORECASE)
|
|
|
|
|
|
def redact_credentials(text: str) -> str:
|
|
"""Replace credentials in ``text`` with ``<redacted>``; keep everything
|
|
else, including line breaks, so a stack trace stays readable."""
|
|
text = text or ''
|
|
# Order matters: the URL and header forms are more specific than the
|
|
# generic key=value pattern, which would otherwise chew the scheme.
|
|
text = _REDACT_URL_USERINFO.sub(r'\1<redacted>\3', text)
|
|
text = _REDACT_AUTH_HEADER.sub(r'\1<redacted>', text)
|
|
return _REDACT_CREDENTIAL.sub(r'\1<redacted>', text)
|