Coverage for dataexcept/redaction.py: 100%
75 statements
« prev ^ index » next coverage.py v7.16.0, created at 2026-09-27 14:44 +0000
« prev ^ index » next coverage.py v7.16.0, created at 2026-09-27 14:44 +0000
1"""Redaction helpers for values that must not reach a log.
3Several exceptions here are raised with credentials in hand: an authentication
4token, a database URL carrying a password, a webhook URL whose *path* is the
5secret. Those values end up in the exception message, and
6:func:`dataexcept.logging_helpers.log_exception` logs ``str(exc)``, so without
7redaction a failed delivery writes the credential to the log.
9The aim is to keep an error debuggable while giving up the secret. A redacted
10value carries a short, non-reversible fingerprint, so repeated failures of the
11*same* credential stay recognisable in a log without the credential appearing
12in it.
14What this can and cannot do is stated in ``SECURITY.md``. In short: values the
15library is *given* as credentials are redacted, and URLs are redacted wherever
16they appear -- including inside a message you supplied and inside the text of a
17wrapped exception. A bare, non-URL secret pasted into free-form text cannot be
18recognised and is not redacted.
19"""
21from __future__ import annotations
23import hashlib
24import re
25from typing import Optional
26from urllib.parse import parse_qsl, urlencode, urlsplit, urlunsplit
28__all__ = [
29 "fingerprint",
30 "redact_if_url",
31 "redact_secret",
32 "redact_url",
33 "redact_urls_in_text",
34 "remove_secret",
35]
37PLACEHOLDER = "***"
39#: Below this length a "secret" is not removed from free text. Substring
40#: replacement of a short value corrupts ordinary words -- removing "tok" from
41#: "Invalid authentication token" mangles the message and tells a reader
42#: nothing. The structured field is redacted regardless of length.
43MIN_REMOVABLE_SECRET_LENGTH = 8
45#: Parameter-name tokens that mark a value as a secret. A name is split into
46#: tokens on separators and camelCase boundaries, and matched token by token.
47#:
48#: Substring matching was tried first and was wrong in both directions: it
49#: redacted "monkey", "design", "assign", "keyword" and "authors" -- mangling
50#: ordinary debugging information -- while still missing "passphrase". The
51#: point of keeping host, port and path is that the error stays actionable, and
52#: shredding a legitimate query parameter works against that.
53#:
54#: Deliberately absent: "code", "state", "nonce" and "client_id". An OAuth
55#: authorization code is a secret, but "code" is far more often a country
56#: code, an HTTP status or a discount code, and redacting those would destroy
57#: more debugging information than it protects. Checked against the parameter
58#: names used by AWS SigV4, Azure SAS, Google Cloud and OAuth 2.
59SENSITIVE_PARAM_TOKENS = frozenset(
60 {
61 "apikey",
62 "auth",
63 "authorization",
64 "bearer",
65 "credential",
66 "credentials",
67 "hmac",
68 "jwt",
69 "key",
70 "keys",
71 "passphrase",
72 "passwd",
73 "password",
74 "pwd",
75 "sas",
76 "secret",
77 "secrets",
78 "session",
79 "sig",
80 "signature",
81 "token",
82 "tokens",
83 }
84)
86#: Splits a parameter name into words: on separators, and between a lower-case
87#: or digit character and an upper-case one, so ``accessToken`` yields
88#: ``["access", "token"]``.
89_NAME_TOKENS = re.compile(r"[A-Za-z0-9]+")
90_CAMEL_BOUNDARY = re.compile(r"(?<=[a-z0-9])(?=[A-Z])")
93def _tokens(name: str) -> list[str]:
94 words: list[str] = []
95 for chunk in _NAME_TOKENS.findall(name):
96 words.extend(part.lower() for part in _CAMEL_BOUNDARY.split(chunk) if part)
97 return words
100def _is_sensitive(name: str) -> bool:
101 return any(token in SENSITIVE_PARAM_TOKENS for token in _tokens(name))
104#: Finds URLs inside free-form text, so a credential cannot slip through in a
105#: caller-supplied message or in the text of a wrapped exception.
106# No word-boundary anchor: a URL can directly follow a word character, as
107# in the step name "feature_https://..." that FeaturePreprocessingError
108# builds. The scheme character class excludes "_", so a match still starts
109# at the scheme rather than mid-word.
110#
111# Where the URL ends needs two character classes rather than one. A comma,
112# semicolon, closing bracket or quote never belongs to a URL in prose, so the
113# body simply excludes them. A full stop, colon, exclamation or question mark
114# does belong inside one -- and also ends the sentence it sits in -- so the
115# body runs through them and only the last character is required not to be
116# one. "see https://h/p?token=SECRET: retry" then keeps its separator instead
117# of losing the colon into the redacted URL.
118_URL_BODY = r"[^\s'\"<>,;)\]}]"
119_URL_END = r"[^\s'\"<>,;)\]}.:!?]"
120_URL_IN_TEXT = re.compile(rf"[a-zA-Z][a-zA-Z0-9+.\-]*://{_URL_BODY}*{_URL_END}")
123def fingerprint(value: str) -> str:
124 """Return a short, one-way fingerprint of *value*.
126 Enough to tell "the same bad token again" from "a different bad token",
127 and not enough to recover the token.
128 """
129 digest = hashlib.sha256(value.encode("utf-8", "replace")).hexdigest()
130 return digest[:8]
133def redact_secret(value: Optional[str]) -> Optional[str]:
134 """Replace a secret with a placeholder and its fingerprint."""
135 if value is None:
136 return None
137 if not value:
138 return PLACEHOLDER
139 return f"{PLACEHOLDER}({fingerprint(value)})"
142def remove_secret(text: str, secret: Optional[str]) -> str:
143 """Replace every occurrence of a known *secret* in *text*.
145 Used where the library was handed the secret explicitly, so it can be
146 removed even from a message the caller wrote themselves.
147 """
148 if not secret or not text or len(secret) < MIN_REMOVABLE_SECRET_LENGTH:
149 return text
150 return text.replace(secret, f"{PLACEHOLDER}({fingerprint(secret)})")
153def _redact_params(query: str) -> tuple[str, bool]:
154 if not query:
155 return query, False
156 pairs = parse_qsl(query, keep_blank_values=True)
157 if not any(_is_sensitive(key) for key, _ in pairs):
158 return query, False
159 return (
160 urlencode(
161 [
162 (key, PLACEHOLDER if _is_sensitive(key) else value)
163 for key, value in pairs
164 ],
165 # Keep the placeholder legible rather than percent-encoded.
166 safe="*",
167 ),
168 True,
169 )
172def redact_url(url: Optional[str], *, keep_path: bool = True) -> Optional[str]:
173 """Strip credentials from *url*.
175 Scheme, host and port are always kept: those are what make an error
176 actionable. Userinfo, sensitive query parameters and sensitive fragment
177 parameters are always removed.
179 Pass ``keep_path=False`` where the path itself is the credential. An
180 incoming webhook URL is the common case -- Slack, Discord and others put
181 the secret in the path, so preserving it would defeat the point.
182 """
183 if not url or not isinstance(url, str):
184 # Anything that is not a string is handed back untouched. urlsplit
185 # would raise AttributeError from inside urllib, masking whatever the
186 # caller's real mistake was with a message about `.decode`.
187 return url
189 try:
190 parts = urlsplit(url)
191 except ValueError: # pragma: no cover - urlsplit is extremely permissive
192 return PLACEHOLDER
194 if not parts.scheme or not parts.netloc:
195 # Not a URL with a host; a bare path or plain string is returned
196 # untouched rather than mangled.
197 return url
199 redacted = False
201 netloc = parts.netloc
202 if "@" in netloc:
203 _, _, host = netloc.rpartition("@")
204 netloc = f"{PLACEHOLDER}:{PLACEHOLDER}@{host}"
205 redacted = True
207 query, query_redacted = _redact_params(parts.query)
208 redacted = redacted or query_redacted
210 fragment = parts.fragment
211 if "=" in fragment:
212 # OAuth implicit flow returns the token in the fragment.
213 fragment, fragment_redacted = _redact_params(fragment)
214 redacted = redacted or fragment_redacted
216 path = parts.path
217 if not keep_path and path.strip("/"):
218 path = f"/{PLACEHOLDER}"
219 redacted = True
221 if not redacted:
222 # Hand back exactly what was passed in. Rebuilding would normalise it,
223 # and "sqlite://" loses its slashes on the way through.
224 return url
226 return urlunsplit((parts.scheme, netloc, path, query, fragment))
229def redact_if_url(value: Optional[str], *, keep_path: bool = True) -> Optional[str]:
230 """Redact *value* only if it is a URL, leaving file paths untouched.
232 Fields such as ``DataLoadingError.source`` document themselves as "file
233 path or URL", so they cannot be redacted unconditionally without mangling
234 ordinary paths.
235 """
236 if not isinstance(value, str) or "://" not in value:
237 return value
238 return redact_url(value, keep_path=keep_path)
241def redact_urls_in_text(text: str, *, keep_path: bool = True) -> str:
242 """Redact every URL found in free-form *text*.
244 This is the boundary that stops a secret being reintroduced after the
245 structured argument was redacted -- through a caller-supplied ``message``,
246 or through the text of a wrapped exception that quotes the original URL.
247 """
248 if not text or "://" not in text:
249 return text
250 return _URL_IN_TEXT.sub(
251 lambda match: redact_url(match.group(0), keep_path=keep_path) or "",
252 text,
253 )