scrubbr 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
scrubbr/__init__.py ADDED
@@ -0,0 +1,6 @@
1
+ from scrubbr.alias import AliasBook
2
+ from scrubbr.identity import LocalIdentity
3
+ from scrubbr.kinds import Finding, Kind
4
+ from scrubbr.scrub import ScrubResult, scrub
5
+
6
+ __all__ = ["AliasBook", "Finding", "Kind", "LocalIdentity", "ScrubResult", "scrub"]
scrubbr/alias.py ADDED
@@ -0,0 +1,38 @@
1
+ import random
2
+
3
+ from scrubbr.kinds import Kind
4
+ from scrubbr.shapes import mint, normalize, render
5
+
6
+
7
+ class AliasBook:
8
+ """The only thing that may mint an alias.
9
+
10
+ Callers ask for the alias of a value and cannot obtain an inconsistent answer, which
11
+ is what makes "one value, one replacement" a structural guarantee rather than a
12
+ convention every call site has to remember.
13
+ """
14
+
15
+ def __init__(self, rng: random.Random | None = None) -> None:
16
+ # SystemRandom is the CSPRNG secrets is built on; a seeded Random makes the
17
+ # output reproducible, which tests and correlated multi-document runs need.
18
+ self._rng = rng if rng is not None else random.SystemRandom()
19
+ self._canonical: dict[tuple[Kind, str], str] = {}
20
+ self._issued: dict[Kind, int] = {}
21
+
22
+ def alias_for(self, kind: Kind, text: str) -> str:
23
+ canonical = self.canonical_alias(kind, normalize(kind, text))
24
+ return render(kind, canonical, text)
25
+
26
+ def canonical_alias(self, kind: Kind, normalized: str) -> str:
27
+ key = (kind, normalized)
28
+ existing = self._canonical.get(key)
29
+ if existing is not None:
30
+ return existing
31
+ minted = mint(kind, normalized, self._rng, self._next(kind))
32
+ self._canonical[key] = minted
33
+ return minted
34
+
35
+ def _next(self, kind: Kind) -> int:
36
+ index = self._issued.get(kind, 0)
37
+ self._issued[kind] = index + 1
38
+ return index
scrubbr/cli.py ADDED
@@ -0,0 +1,168 @@
1
+ import argparse
2
+ import io
3
+ import os
4
+ import shutil
5
+ import sys
6
+ from collections import Counter
7
+ from collections.abc import Callable
8
+ from dataclasses import replace
9
+ from importlib.metadata import version
10
+ from pathlib import Path
11
+
12
+ import structlog
13
+
14
+ from scrubbr.identity import LocalIdentity
15
+ from scrubbr.report import clip, render_report
16
+ from scrubbr.review import NoTerminal, Terminal, confirm, open_terminal
17
+ from scrubbr.scrub import ScrubResult, scrub
18
+
19
+
20
+ def _configure_logging() -> None:
21
+ renderer = (
22
+ structlog.dev.ConsoleRenderer(colors=sys.stderr.isatty())
23
+ if sys.stderr.isatty()
24
+ else structlog.processors.JSONRenderer()
25
+ )
26
+ structlog.configure(
27
+ processors=[structlog.processors.add_log_level, renderer],
28
+ logger_factory=structlog.PrintLoggerFactory(file=sys.stderr),
29
+ )
30
+
31
+
32
+ def _parse(argv: list[str] | None) -> argparse.Namespace:
33
+ parser = argparse.ArgumentParser(
34
+ prog="scrubbr",
35
+ description="Sanitize Linux diagnostics before pasting them into an LLM.",
36
+ )
37
+ parser.add_argument(
38
+ "--version",
39
+ action="version",
40
+ version=f"%(prog)s {version('scrubbr')}",
41
+ )
42
+ parser.add_argument(
43
+ "infile",
44
+ nargs="?",
45
+ # Kernel and firmware strings in dmesg are not always valid UTF-8, and crashing on
46
+ # a decode error would lose whatever the caller piped in.
47
+ type=lambda path: argparse.FileType("r", encoding="utf-8", errors="replace")(path),
48
+ default=None,
49
+ help="file to read; defaults to stdin",
50
+ )
51
+ parser.add_argument(
52
+ "-o",
53
+ "--output",
54
+ metavar="FILE",
55
+ help="write the scrubbed text to FILE instead of stdout",
56
+ )
57
+ parser.add_argument(
58
+ "-y",
59
+ "--no-review",
60
+ action="store_true",
61
+ help="skip the interactive review",
62
+ )
63
+ parser.add_argument(
64
+ "-v",
65
+ "--verbose",
66
+ action="store_true",
67
+ help="also report each replaced value with its count and alias"
68
+ " (prints the original values to stderr)",
69
+ )
70
+ parser.add_argument(
71
+ "--strict",
72
+ action="store_true",
73
+ help="refuse to emit anything while unscrubbed suspicious strings remain",
74
+ )
75
+ parser.add_argument(
76
+ "--no-identity",
77
+ action="store_true",
78
+ help="do not seed the scanner with this machine's hostname, user and machine-id",
79
+ )
80
+ parser.add_argument(
81
+ "--also",
82
+ action="append",
83
+ default=[],
84
+ metavar="TEXT",
85
+ help="scrub this value too; repeatable. IPs are always replaced (even private or"
86
+ " loopback), long hex, UUIDs and emails keep their shape; anything else becomes"
87
+ " [REDACTED]",
88
+ )
89
+ return parser.parse_args(argv)
90
+
91
+
92
+ def _report(log: structlog.stdlib.BoundLogger, result: ScrubResult, verbose: bool) -> None:
93
+ if sys.stderr.isatty():
94
+ sys.stderr.write("\n".join(render_report(result, verbose, _stderr_width())) + "\n")
95
+ return
96
+ log.info(
97
+ "scrubbed",
98
+ **{kind.value: count for kind, count in sorted(result.counts.items())},
99
+ replacements=len(result.findings),
100
+ )
101
+ if verbose:
102
+ occurrences = Counter((f.kind, f.text, f.alias) for f in result.findings)
103
+ for (kind, text, alias), count in sorted(occurrences.items()):
104
+ log.info("replaced", kind=kind.value, text=clip(text), count=count, alias=clip(alias))
105
+ for residual in result.residuals:
106
+ log.warning("unscrubbed", line=residual.line, reason=residual.reason, text=residual.text)
107
+
108
+
109
+ def _stderr_width() -> int:
110
+ try:
111
+ return os.get_terminal_size(sys.stderr.fileno()).columns
112
+ except (OSError, ValueError):
113
+ # shutil probes stdout, which is routinely a pipe here (`scrubbr | wl-copy`), so
114
+ # stderr's own fd is tried first; shutil still honours $COLUMNS before its 80.
115
+ return shutil.get_terminal_size().columns
116
+
117
+
118
+ def main(argv: list[str] | None = None, open_tty: Callable[[], Terminal] = open_terminal) -> int:
119
+ args = _parse(argv)
120
+ _configure_logging()
121
+ log = structlog.get_logger()
122
+
123
+ if args.infile is None:
124
+ if isinstance(sys.stdin, io.TextIOWrapper):
125
+ sys.stdin.reconfigure(errors="replace")
126
+ args.infile = sys.stdin
127
+ text = args.infile.read()
128
+ base = LocalIdentity() if args.no_identity else LocalIdentity.local()
129
+ identity = replace(base, extra=base.extra + tuple(args.also))
130
+ result = scrub(text, identity)
131
+ _report(log, result, args.verbose)
132
+
133
+ if args.strict and result.residuals:
134
+ log.error("refusing to emit", reason="strict mode with unscrubbed strings")
135
+ return 2
136
+
137
+ if not args.no_review:
138
+ try:
139
+ tty = open_tty()
140
+ except NoTerminal:
141
+ # Failing open here would emit unreviewed text precisely when the safety gate
142
+ # could not run. Skipping review has to be a decision the caller makes.
143
+ log.error("refusing to emit", reason="no terminal for review; pass -y to skip it")
144
+ return 3
145
+ try:
146
+ confirmed = confirm(text, result.text, result.residuals, tty)
147
+ finally:
148
+ tty.close()
149
+ if not confirmed:
150
+ log.error("discarded", reason="not confirmed at review")
151
+ return 1
152
+
153
+ if args.output is not None:
154
+ try:
155
+ # Opened only after the review passes: opening at parse time would leave an
156
+ # empty or truncated file behind on every refused run.
157
+ Path(args.output).write_text(result.text, encoding="utf-8")
158
+ except OSError as error:
159
+ log.error("could not write", path=args.output, error=str(error))
160
+ return 4
161
+ log.info("written", path=args.output)
162
+ else:
163
+ sys.stdout.write(result.text)
164
+ return 0
165
+
166
+
167
+ if __name__ == "__main__":
168
+ raise SystemExit(main())
scrubbr/detect.py ADDED
@@ -0,0 +1,230 @@
1
+ import re
2
+ from dataclasses import dataclass
3
+ from functools import lru_cache
4
+
5
+ from scrubbr.identity import LocalIdentity
6
+ from scrubbr.kinds import Kind
7
+ from scrubbr.shapes import EMAIL_PATTERN, HEX_PATTERN, UUID_PATTERN, classify_literal
8
+
9
+ # An RSA-8192 key is around 12 KB of base64, so this fits any real key while keeping an
10
+ # unterminated BEGIN marker from backtracking across the whole file.
11
+ PEM_MAX_BODY = 20_000
12
+
13
+ NO_IDENTITY = LocalIdentity()
14
+
15
+ SECRET_KEYWORDS = (
16
+ "psk",
17
+ "password",
18
+ "passwd",
19
+ "secret",
20
+ "api_key",
21
+ "apikey",
22
+ "access_token",
23
+ "auth_token",
24
+ "client_secret",
25
+ "private_key",
26
+ )
27
+
28
+
29
+ @dataclass(frozen=True)
30
+ class Rule:
31
+ name: str
32
+ kind: Kind
33
+ pattern: str
34
+ value_group: str | None = None
35
+ forced: bool = False
36
+
37
+
38
+ @dataclass(frozen=True)
39
+ class Match:
40
+ kind: Kind
41
+ start: int
42
+ end: int
43
+ text: str
44
+ forced: bool = False
45
+
46
+
47
+ # Ordered most-specific first. Alternation resolves precedence: at any position the
48
+ # earliest alternative that matches wins, which is what makes the colon-hex family
49
+ # (fingerprint / MAC / IPv6) disambiguate correctly.
50
+ STRUCTURAL_RULES: tuple[Rule, ...] = (
51
+ Rule(
52
+ "pem",
53
+ Kind.PEM,
54
+ # The body stays a wildcard on purpose. Narrowing it to the base64 alphabet would
55
+ # make one stray character anywhere in the block fail the whole alternative and
56
+ # pass the key through verbatim; the length bound alone removes the backtracking.
57
+ r"-----BEGIN [A-Z0-9 ._-]{0,100}-----"
58
+ rf"(?P<pem_body>[\s\S]{{0,{PEM_MAX_BODY}}}?)"
59
+ r"-----END [A-Z0-9 ._-]{0,100}-----",
60
+ value_group="pem_body",
61
+ ),
62
+ # A log cut off mid-key has a BEGIN marker and no END. The complete rule above is
63
+ # tried first, so this only ever fires on a truncated block -- without it the key body
64
+ # matches nothing at all and survives verbatim, which truncated diagnostics make common.
65
+ Rule(
66
+ "pem_truncated",
67
+ Kind.PEM,
68
+ r"-----BEGIN [A-Z0-9 ._-]{0,100}-----"
69
+ rf"(?P<pem_open>(?:\r?\n[A-Za-z0-9+/=]{{16,80}}){{1,{PEM_MAX_BODY // 16}}})",
70
+ value_group="pem_open",
71
+ ),
72
+ Rule(
73
+ "crypt_hash",
74
+ Kind.CRYPT_HASH,
75
+ r"\$(?:1|5|6|2[aby]|apr1)\$(?:rounds=\d+\$)?[./A-Za-z0-9]{1,16}\$[./A-Za-z0-9]{20,90}",
76
+ ),
77
+ Rule("jwt", Kind.JWT, r"eyJ[A-Za-z0-9_-]{6,}\.[A-Za-z0-9_-]{6,}\.[A-Za-z0-9_-]*"),
78
+ Rule("disk_id", Kind.DISK_ID, r"/dev/disk/by-id/(?P<disk_id_val>[A-Za-z0-9._:+-]+)",
79
+ value_group="disk_id_val"),
80
+ # 8+ colon-hex groups: a digest fingerprint, never a MAC.
81
+ Rule(
82
+ "fingerprint",
83
+ Kind.FINGERPRINT,
84
+ r"(?<![0-9A-Fa-f:])(?:[0-9A-Fa-f]{2}:){7,}[0-9A-Fa-f]{2}(?![0-9A-Fa-f:])",
85
+ ),
86
+ Rule("uuid", Kind.UUID, rf"(?<![0-9A-Fa-f-]){UUID_PATTERN}(?![0-9A-Fa-f-])"),
87
+ # Exactly six 2-digit groups with a consistent separator, fenced by lookarounds so it
88
+ # cannot bite a slice out of a longer colon-hex chain.
89
+ # One rule per separator rather than one rule with a backreferenced separator: each
90
+ # fence then excludes only its OWN separator, so a colon-separated address is still
91
+ # found in "aa:bb:cc:dd:ee:ff-eth0" and "wlan0-aa:bb:cc:dd:ee:ff", while a
92
+ # hyphen-separated one is still fenced off from a longer hyphenated run.
93
+ Rule(
94
+ "mac_colon",
95
+ Kind.MAC,
96
+ r"(?<![0-9A-Fa-f:])(?:[0-9A-Fa-f]{2}:){5}[0-9A-Fa-f]{2}(?![0-9A-Fa-f:])",
97
+ ),
98
+ Rule(
99
+ "mac_hyphen",
100
+ Kind.MAC,
101
+ r"(?<![0-9A-Fa-f-])(?:[0-9A-Fa-f]{2}-){5}[0-9A-Fa-f]{2}(?![0-9A-Fa-f-])",
102
+ ),
103
+ # A trailing dot only disqualifies the dotted form when more hex follows it.
104
+ Rule(
105
+ "mac_cisco",
106
+ Kind.MAC,
107
+ r"(?<![0-9A-Fa-f])(?<![0-9A-Fa-f]\.)"
108
+ r"[0-9A-Fa-f]{4}\.[0-9A-Fa-f]{4}\.[0-9A-Fa-f]{4}"
109
+ r"(?![0-9A-Fa-f])(?!\.[0-9A-Fa-f])",
110
+ ),
111
+ # Bare 12-hex is only a MAC when something says so; on its own it is indistinguishable
112
+ # from a truncated hash, and scrubbing every 12-hex run would wreck ordinary logs.
113
+ Rule(
114
+ "mac_bare",
115
+ Kind.MAC,
116
+ r"(?i:mac|hwaddr|hw_addr|bssid|ether|lladdr|hw)(?:\s+address)?\s*[=:]?\s*"
117
+ r"(?P<mac_bare_val>[0-9A-Fa-f]{12})(?![0-9A-Za-z])",
118
+ value_group="mac_bare_val",
119
+ ),
120
+ # Ahead of the identity literals: a username is usually the local part of its owner's
121
+ # address, and letting the literal win would rewrite "dev@example.com" to
122
+ # "user-a@example.com", keeping the domain — normally the more identifying half.
123
+ Rule("email", Kind.EMAIL, EMAIL_PATTERN),
124
+ )
125
+
126
+ CONTEXTUAL_RULES: tuple[Rule, ...] = (
127
+ Rule(
128
+ "ssid",
129
+ Kind.SSID,
130
+ r"(?i:ssid)\s*[=:]\s*(?P<ssid_q>[\"'])(?P<ssid_val>[^\"']*)(?P=ssid_q)",
131
+ value_group="ssid_val",
132
+ ),
133
+ Rule(
134
+ "secret_kv",
135
+ Kind.SECRET_VALUE,
136
+ r"(?i:" + "|".join(SECRET_KEYWORDS) + r")\s*[=:]\s*"
137
+ r"(?P<secret_q>[\"']?)(?P<secret_val>[^\s\"',;]+)(?P=secret_q)",
138
+ value_group="secret_val",
139
+ ),
140
+ Rule(
141
+ "home_path",
142
+ Kind.USERNAME,
143
+ r"/home/(?P<home_user>[a-z_][a-z0-9_-]{0,31})",
144
+ value_group="home_user",
145
+ ),
146
+ Rule(
147
+ "ipv6",
148
+ Kind.IPV6,
149
+ r"(?<![0-9A-Fa-f:.])[0-9A-Fa-f]{0,4}(?::[0-9A-Fa-f]{0,4}){2,7}(?![0-9A-Fa-f:])",
150
+ ),
151
+ Rule("ipv4", Kind.IPV4, r"(?<![0-9.])(?:[0-9]{1,3}\.){3}[0-9]{1,3}(?![0-9.])"),
152
+ Rule(
153
+ "hex",
154
+ Kind.HEX,
155
+ rf"(?<![0-9A-Za-z])(?:0[xX])?(?P<hex_val>{HEX_PATTERN})(?![0-9A-Za-z])",
156
+ value_group="hex_val",
157
+ ),
158
+ )
159
+
160
+
161
+ def _literal_rules(
162
+ identity: LocalIdentity, promoted: tuple[tuple[str, Kind], ...]
163
+ ) -> tuple[Rule, ...]:
164
+ literals: list[tuple[str, Kind, bool]] = []
165
+ if identity.hostname:
166
+ literals.append((identity.hostname, Kind.HOSTNAME, False))
167
+ short = identity.hostname.split(".")[0]
168
+ if short != identity.hostname:
169
+ literals.append((short, Kind.HOSTNAME, False))
170
+ # An extra value is explicitly declared, so it is scrubbed unconditionally -- forced
171
+ # past the keep-allowlists that judge only incidental matches.
172
+ literals.extend((value, classify_literal(value), True) for value in identity.extra)
173
+ if identity.username:
174
+ literals.append((identity.username, Kind.USERNAME, False))
175
+ literals.extend((value, kind, False) for value, kind in promoted)
176
+ # Longest first: a username is frequently a substring of the hostname ("dev" inside
177
+ # "dev-thinkpad"), and the longer match has to win at that position.
178
+ literals.sort(key=lambda entry: len(entry[0]), reverse=True)
179
+ return tuple(
180
+ Rule(f"literal_{index}", kind, _literal_pattern(value, kind), forced=forced)
181
+ for index, (value, kind, forced) in enumerate(literals)
182
+ )
183
+
184
+
185
+ def _literal_pattern(value: str, kind: Kind) -> str:
186
+ escaped = re.escape(value)
187
+ if kind is Kind.IPV6:
188
+ # Forcing must survive the log spelling FE80::1 as fe80::1.
189
+ escaped = f"(?i:{escaped})"
190
+ return rf"(?<![A-Za-z0-9_]){escaped}(?![A-Za-z0-9_])"
191
+
192
+
193
+ @lru_cache(maxsize=64)
194
+ def _compiled(
195
+ identity: LocalIdentity, promoted: tuple[tuple[str, Kind], ...]
196
+ ) -> tuple[re.Pattern[str], dict[str, Rule]]:
197
+ rules = STRUCTURAL_RULES + _literal_rules(identity, promoted) + CONTEXTUAL_RULES
198
+ combined = "|".join(f"(?P<{rule.name}>{rule.pattern})" for rule in rules)
199
+ return re.compile(combined), {rule.name: rule for rule in rules}
200
+
201
+
202
+ def detect(
203
+ text: str,
204
+ identity: LocalIdentity = NO_IDENTITY,
205
+ promoted: tuple[tuple[str, Kind], ...] = (),
206
+ ) -> list[Match]:
207
+ """Locate everything worth replacing.
208
+
209
+ `promoted` carries values a first pass discovered behind a keyword — `psk=secret` —
210
+ so that a second pass also catches them where they appear bare and unlabelled.
211
+ """
212
+ pattern, by_name = _compiled(identity, promoted)
213
+ matches: list[Match] = []
214
+ for found in pattern.finditer(text):
215
+ rule = by_name[_which(found, by_name)]
216
+ group = rule.value_group or rule.name
217
+ start, end = found.span(group)
218
+ if start < 0 or start == end:
219
+ continue
220
+ matches.append(
221
+ Match(kind=rule.kind, start=start, end=end, text=text[start:end], forced=rule.forced)
222
+ )
223
+ return matches
224
+
225
+
226
+ def _which(found: re.Match[str], by_name: dict[str, Rule]) -> str:
227
+ for name in by_name:
228
+ if found.group(name) is not None:
229
+ return name
230
+ raise AssertionError("a match must belong to exactly one rule")
scrubbr/identity.py ADDED
@@ -0,0 +1,59 @@
1
+ import getpass
2
+ import socket
3
+ from dataclasses import dataclass, field
4
+ from pathlib import Path
5
+
6
+ SYSTEM_USERNAMES = frozenset(
7
+ {
8
+ "root",
9
+ "daemon",
10
+ "bin",
11
+ "sys",
12
+ "sync",
13
+ "man",
14
+ "nobody",
15
+ "messagebus",
16
+ "dbus",
17
+ "polkitd",
18
+ "sshd",
19
+ "www-data",
20
+ "systemd-network",
21
+ "systemd-resolve",
22
+ "systemd-timesync",
23
+ "systemd-journal",
24
+ "user",
25
+ }
26
+ )
27
+
28
+
29
+ @dataclass(frozen=True)
30
+ class LocalIdentity:
31
+ """Literal strings naming *this* machine and *this* person.
32
+
33
+ Seeding the scanner with these is the highest-precision rule available: the surest
34
+ way to know a hostname is yours is to ask the system for it, rather than to guess at
35
+ which token in a syslog line is a hostname.
36
+ """
37
+
38
+ hostname: str | None = None
39
+ username: str | None = None
40
+ extra: tuple[str, ...] = field(default_factory=tuple)
41
+
42
+ @classmethod
43
+ def local(cls) -> "LocalIdentity":
44
+ extra: list[str] = []
45
+ try:
46
+ machine_id = Path("/etc/machine-id").read_text(encoding="utf-8").strip()
47
+ except OSError:
48
+ machine_id = ""
49
+ if machine_id:
50
+ extra.append(machine_id)
51
+ try:
52
+ username: str | None = getpass.getuser()
53
+ except (OSError, KeyError):
54
+ username = None
55
+ try:
56
+ hostname: str | None = socket.gethostname()
57
+ except OSError:
58
+ hostname = None
59
+ return cls(hostname=hostname, username=username, extra=tuple(extra))
scrubbr/kinds.py ADDED
@@ -0,0 +1,41 @@
1
+ from enum import StrEnum
2
+
3
+ from pydantic import BaseModel, ConfigDict
4
+
5
+
6
+ class Kind(StrEnum):
7
+ PEM = "pem"
8
+ CRYPT_HASH = "crypt_hash"
9
+ JWT = "jwt"
10
+ FINGERPRINT = "fingerprint"
11
+ DISK_ID = "disk_id"
12
+ UUID = "uuid"
13
+ MAC = "mac"
14
+ IPV6 = "ipv6"
15
+ LINK_LOCAL_ID = "link_local_id"
16
+ IPV4 = "ipv4"
17
+ HEX = "hex"
18
+ EMAIL = "email"
19
+ SECRET_VALUE = "secret_value"
20
+ REDACTED = "redacted"
21
+ SSID = "ssid"
22
+ HOSTNAME = "hostname"
23
+ USERNAME = "username"
24
+
25
+
26
+ class Finding(BaseModel):
27
+ model_config = ConfigDict(frozen=True)
28
+
29
+ kind: Kind
30
+ start: int
31
+ end: int
32
+ text: str
33
+ alias: str
34
+
35
+
36
+ class Residual(BaseModel):
37
+ model_config = ConfigDict(frozen=True)
38
+
39
+ line: int
40
+ text: str
41
+ reason: str
scrubbr/report.py ADDED
@@ -0,0 +1,108 @@
1
+ import textwrap
2
+ from collections import Counter
3
+
4
+ from scrubbr.kinds import Finding, Kind, Residual
5
+ from scrubbr.scrub import ScrubResult
6
+
7
+ # The report names values, it does not reproduce them: a PEM body is kilobytes across
8
+ # many lines, and no terminal width justifies showing more of a secret.
9
+ CLIP = 60
10
+ # Below this the columns stop being readable anyway; render as if the terminal were this
11
+ # wide and let it wrap.
12
+ MIN_WIDTH = 40
13
+ _GUTTER = " "
14
+ # StrEnum sorts alphabetically; declaration order is the sensitivity order, which is what
15
+ # a security report should lead with.
16
+ _KIND_ORDER = {kind: index for index, kind in enumerate(Kind)}
17
+
18
+
19
+ def render_report(result: ScrubResult, verbose: bool, width: int) -> list[str]:
20
+ width = max(width, MIN_WIDTH)
21
+ lines = _summary(result, width)
22
+ if verbose and result.findings:
23
+ lines += ["", *_findings_table(result.findings, width)]
24
+ if result.residuals:
25
+ lines += ["", *_residuals_table(result.residuals, width)]
26
+ return lines
27
+
28
+
29
+ def _summary(result: ScrubResult, width: int) -> list[str]:
30
+ distinct = sum(result.counts.values())
31
+ lines = [f"scrubbed {distinct} distinct values, {len(result.findings)} replacements"]
32
+ if result.counts:
33
+ by_kind = ", ".join(
34
+ f"{kind.value} {result.counts[kind]}" for kind in Kind if kind in result.counts
35
+ )
36
+ lines += textwrap.wrap(
37
+ by_kind, width=width, initial_indent=_GUTTER, subsequent_indent=_GUTTER
38
+ )
39
+ return lines
40
+
41
+
42
+ def _findings_table(findings: list[Finding], width: int) -> list[str]:
43
+ occurrences = Counter((f.kind, f.text, f.alias) for f in findings)
44
+ rows = sorted(
45
+ (
46
+ (kind, _flatten(text), count, _flatten(alias))
47
+ for (kind, text, alias), count in occurrences.items()
48
+ ),
49
+ key=lambda row: (_KIND_ORDER[row[0]], -row[2], row[1]),
50
+ )
51
+ kind_w = max(len("kind"), *(len(kind.value) for kind, _, _, _ in rows))
52
+ count_w = max(len("count"), *(len(str(count)) for _, _, count, _ in rows))
53
+ text_w, alias_w = _apportion(
54
+ width - kind_w - count_w - 3 * len(_GUTTER),
55
+ min(CLIP, max(len("text"), *(len(text) for _, text, _, _ in rows))),
56
+ min(CLIP, max(len("alias"), *(len(alias) for _, _, _, alias in rows))),
57
+ )
58
+ template = f"{{:<{kind_w}}}{_GUTTER}{{:>{count_w}}}{_GUTTER}{{:<{text_w}}}{_GUTTER}{{}}"
59
+ header = template.format("kind", "count", _fit("text", text_w), _fit("alias", alias_w))
60
+ lines = [header.rstrip()]
61
+ lines += [
62
+ template.format(kind.value, count, _fit(text, text_w), _fit(alias, alias_w)).rstrip()
63
+ for kind, text, count, alias in rows
64
+ ]
65
+ return lines
66
+
67
+
68
+ def _residuals_table(residuals: list[Residual], width: int) -> list[str]:
69
+ line_w = max(len("line"), *(len(str(residual.line)) for residual in residuals))
70
+ reason_w = max(len("reason"), *(len(residual.reason) for residual in residuals))
71
+ text_w = min(CLIP, width - line_w - reason_w - 3 * len(_GUTTER))
72
+ template = f"{_GUTTER}{{:>{line_w}}}{_GUTTER}{{:<{reason_w}}}{_GUTTER}{{}}"
73
+ lines = [f"{len(residuals)} unscrubbed strings look sensitive:"]
74
+ lines.append(template.format("line", "reason", "text").rstrip())
75
+ lines += [
76
+ template.format(residual.line, residual.reason, _fit(_flatten(residual.text), text_w))
77
+ for residual in residuals
78
+ ]
79
+ return lines
80
+
81
+
82
+ def _apportion(avail: int, text_nat: int, alias_nat: int) -> tuple[int, int]:
83
+ if text_nat + alias_nat <= avail:
84
+ return text_nat, alias_nat
85
+ half = avail // 2
86
+ if text_nat <= half:
87
+ return text_nat, avail - text_nat
88
+ if alias_nat <= half:
89
+ return avail - alias_nat, alias_nat
90
+ return max(avail - half, 0), max(half, 0)
91
+
92
+
93
+ def _fit(value: str, width: int) -> str:
94
+ if len(value) <= width:
95
+ return value
96
+ if width <= 0:
97
+ return ""
98
+ keep = width - 1
99
+ tail = keep // 2
100
+ return value[: keep - tail] + "…" + value[len(value) - tail :]
101
+
102
+
103
+ def _flatten(value: str) -> str:
104
+ return " ".join(value.split())
105
+
106
+
107
+ def clip(value: str) -> str:
108
+ return _flatten(value)[:CLIP]