scrubbr 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scrubbr/__init__.py +6 -0
- scrubbr/alias.py +38 -0
- scrubbr/cli.py +168 -0
- scrubbr/detect.py +230 -0
- scrubbr/identity.py +59 -0
- scrubbr/kinds.py +41 -0
- scrubbr/report.py +108 -0
- scrubbr/residual.py +103 -0
- scrubbr/review.py +90 -0
- scrubbr/scrub.py +218 -0
- scrubbr/shapes.py +239 -0
- scrubbr-0.2.0.dist-info/METADATA +255 -0
- scrubbr-0.2.0.dist-info/RECORD +16 -0
- scrubbr-0.2.0.dist-info/WHEEL +4 -0
- scrubbr-0.2.0.dist-info/entry_points.txt +3 -0
- scrubbr-0.2.0.dist-info/licenses/LICENSE +21 -0
scrubbr/__init__.py
ADDED
scrubbr/alias.py
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
import random
|
|
2
|
+
|
|
3
|
+
from scrubbr.kinds import Kind
|
|
4
|
+
from scrubbr.shapes import mint, normalize, render
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class AliasBook:
|
|
8
|
+
"""The only thing that may mint an alias.
|
|
9
|
+
|
|
10
|
+
Callers ask for the alias of a value and cannot obtain an inconsistent answer, which
|
|
11
|
+
is what makes "one value, one replacement" a structural guarantee rather than a
|
|
12
|
+
convention every call site has to remember.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
def __init__(self, rng: random.Random | None = None) -> None:
|
|
16
|
+
# SystemRandom is the CSPRNG secrets is built on; a seeded Random makes the
|
|
17
|
+
# output reproducible, which tests and correlated multi-document runs need.
|
|
18
|
+
self._rng = rng if rng is not None else random.SystemRandom()
|
|
19
|
+
self._canonical: dict[tuple[Kind, str], str] = {}
|
|
20
|
+
self._issued: dict[Kind, int] = {}
|
|
21
|
+
|
|
22
|
+
def alias_for(self, kind: Kind, text: str) -> str:
|
|
23
|
+
canonical = self.canonical_alias(kind, normalize(kind, text))
|
|
24
|
+
return render(kind, canonical, text)
|
|
25
|
+
|
|
26
|
+
def canonical_alias(self, kind: Kind, normalized: str) -> str:
|
|
27
|
+
key = (kind, normalized)
|
|
28
|
+
existing = self._canonical.get(key)
|
|
29
|
+
if existing is not None:
|
|
30
|
+
return existing
|
|
31
|
+
minted = mint(kind, normalized, self._rng, self._next(kind))
|
|
32
|
+
self._canonical[key] = minted
|
|
33
|
+
return minted
|
|
34
|
+
|
|
35
|
+
def _next(self, kind: Kind) -> int:
|
|
36
|
+
index = self._issued.get(kind, 0)
|
|
37
|
+
self._issued[kind] = index + 1
|
|
38
|
+
return index
|
scrubbr/cli.py
ADDED
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import io
|
|
3
|
+
import os
|
|
4
|
+
import shutil
|
|
5
|
+
import sys
|
|
6
|
+
from collections import Counter
|
|
7
|
+
from collections.abc import Callable
|
|
8
|
+
from dataclasses import replace
|
|
9
|
+
from importlib.metadata import version
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
import structlog
|
|
13
|
+
|
|
14
|
+
from scrubbr.identity import LocalIdentity
|
|
15
|
+
from scrubbr.report import clip, render_report
|
|
16
|
+
from scrubbr.review import NoTerminal, Terminal, confirm, open_terminal
|
|
17
|
+
from scrubbr.scrub import ScrubResult, scrub
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _configure_logging() -> None:
|
|
21
|
+
renderer = (
|
|
22
|
+
structlog.dev.ConsoleRenderer(colors=sys.stderr.isatty())
|
|
23
|
+
if sys.stderr.isatty()
|
|
24
|
+
else structlog.processors.JSONRenderer()
|
|
25
|
+
)
|
|
26
|
+
structlog.configure(
|
|
27
|
+
processors=[structlog.processors.add_log_level, renderer],
|
|
28
|
+
logger_factory=structlog.PrintLoggerFactory(file=sys.stderr),
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _parse(argv: list[str] | None) -> argparse.Namespace:
|
|
33
|
+
parser = argparse.ArgumentParser(
|
|
34
|
+
prog="scrubbr",
|
|
35
|
+
description="Sanitize Linux diagnostics before pasting them into an LLM.",
|
|
36
|
+
)
|
|
37
|
+
parser.add_argument(
|
|
38
|
+
"--version",
|
|
39
|
+
action="version",
|
|
40
|
+
version=f"%(prog)s {version('scrubbr')}",
|
|
41
|
+
)
|
|
42
|
+
parser.add_argument(
|
|
43
|
+
"infile",
|
|
44
|
+
nargs="?",
|
|
45
|
+
# Kernel and firmware strings in dmesg are not always valid UTF-8, and crashing on
|
|
46
|
+
# a decode error would lose whatever the caller piped in.
|
|
47
|
+
type=lambda path: argparse.FileType("r", encoding="utf-8", errors="replace")(path),
|
|
48
|
+
default=None,
|
|
49
|
+
help="file to read; defaults to stdin",
|
|
50
|
+
)
|
|
51
|
+
parser.add_argument(
|
|
52
|
+
"-o",
|
|
53
|
+
"--output",
|
|
54
|
+
metavar="FILE",
|
|
55
|
+
help="write the scrubbed text to FILE instead of stdout",
|
|
56
|
+
)
|
|
57
|
+
parser.add_argument(
|
|
58
|
+
"-y",
|
|
59
|
+
"--no-review",
|
|
60
|
+
action="store_true",
|
|
61
|
+
help="skip the interactive review",
|
|
62
|
+
)
|
|
63
|
+
parser.add_argument(
|
|
64
|
+
"-v",
|
|
65
|
+
"--verbose",
|
|
66
|
+
action="store_true",
|
|
67
|
+
help="also report each replaced value with its count and alias"
|
|
68
|
+
" (prints the original values to stderr)",
|
|
69
|
+
)
|
|
70
|
+
parser.add_argument(
|
|
71
|
+
"--strict",
|
|
72
|
+
action="store_true",
|
|
73
|
+
help="refuse to emit anything while unscrubbed suspicious strings remain",
|
|
74
|
+
)
|
|
75
|
+
parser.add_argument(
|
|
76
|
+
"--no-identity",
|
|
77
|
+
action="store_true",
|
|
78
|
+
help="do not seed the scanner with this machine's hostname, user and machine-id",
|
|
79
|
+
)
|
|
80
|
+
parser.add_argument(
|
|
81
|
+
"--also",
|
|
82
|
+
action="append",
|
|
83
|
+
default=[],
|
|
84
|
+
metavar="TEXT",
|
|
85
|
+
help="scrub this value too; repeatable. IPs are always replaced (even private or"
|
|
86
|
+
" loopback), long hex, UUIDs and emails keep their shape; anything else becomes"
|
|
87
|
+
" [REDACTED]",
|
|
88
|
+
)
|
|
89
|
+
return parser.parse_args(argv)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _report(log: structlog.stdlib.BoundLogger, result: ScrubResult, verbose: bool) -> None:
|
|
93
|
+
if sys.stderr.isatty():
|
|
94
|
+
sys.stderr.write("\n".join(render_report(result, verbose, _stderr_width())) + "\n")
|
|
95
|
+
return
|
|
96
|
+
log.info(
|
|
97
|
+
"scrubbed",
|
|
98
|
+
**{kind.value: count for kind, count in sorted(result.counts.items())},
|
|
99
|
+
replacements=len(result.findings),
|
|
100
|
+
)
|
|
101
|
+
if verbose:
|
|
102
|
+
occurrences = Counter((f.kind, f.text, f.alias) for f in result.findings)
|
|
103
|
+
for (kind, text, alias), count in sorted(occurrences.items()):
|
|
104
|
+
log.info("replaced", kind=kind.value, text=clip(text), count=count, alias=clip(alias))
|
|
105
|
+
for residual in result.residuals:
|
|
106
|
+
log.warning("unscrubbed", line=residual.line, reason=residual.reason, text=residual.text)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _stderr_width() -> int:
|
|
110
|
+
try:
|
|
111
|
+
return os.get_terminal_size(sys.stderr.fileno()).columns
|
|
112
|
+
except (OSError, ValueError):
|
|
113
|
+
# shutil probes stdout, which is routinely a pipe here (`scrubbr | wl-copy`), so
|
|
114
|
+
# stderr's own fd is tried first; shutil still honours $COLUMNS before its 80.
|
|
115
|
+
return shutil.get_terminal_size().columns
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def main(argv: list[str] | None = None, open_tty: Callable[[], Terminal] = open_terminal) -> int:
|
|
119
|
+
args = _parse(argv)
|
|
120
|
+
_configure_logging()
|
|
121
|
+
log = structlog.get_logger()
|
|
122
|
+
|
|
123
|
+
if args.infile is None:
|
|
124
|
+
if isinstance(sys.stdin, io.TextIOWrapper):
|
|
125
|
+
sys.stdin.reconfigure(errors="replace")
|
|
126
|
+
args.infile = sys.stdin
|
|
127
|
+
text = args.infile.read()
|
|
128
|
+
base = LocalIdentity() if args.no_identity else LocalIdentity.local()
|
|
129
|
+
identity = replace(base, extra=base.extra + tuple(args.also))
|
|
130
|
+
result = scrub(text, identity)
|
|
131
|
+
_report(log, result, args.verbose)
|
|
132
|
+
|
|
133
|
+
if args.strict and result.residuals:
|
|
134
|
+
log.error("refusing to emit", reason="strict mode with unscrubbed strings")
|
|
135
|
+
return 2
|
|
136
|
+
|
|
137
|
+
if not args.no_review:
|
|
138
|
+
try:
|
|
139
|
+
tty = open_tty()
|
|
140
|
+
except NoTerminal:
|
|
141
|
+
# Failing open here would emit unreviewed text precisely when the safety gate
|
|
142
|
+
# could not run. Skipping review has to be a decision the caller makes.
|
|
143
|
+
log.error("refusing to emit", reason="no terminal for review; pass -y to skip it")
|
|
144
|
+
return 3
|
|
145
|
+
try:
|
|
146
|
+
confirmed = confirm(text, result.text, result.residuals, tty)
|
|
147
|
+
finally:
|
|
148
|
+
tty.close()
|
|
149
|
+
if not confirmed:
|
|
150
|
+
log.error("discarded", reason="not confirmed at review")
|
|
151
|
+
return 1
|
|
152
|
+
|
|
153
|
+
if args.output is not None:
|
|
154
|
+
try:
|
|
155
|
+
# Opened only after the review passes: opening at parse time would leave an
|
|
156
|
+
# empty or truncated file behind on every refused run.
|
|
157
|
+
Path(args.output).write_text(result.text, encoding="utf-8")
|
|
158
|
+
except OSError as error:
|
|
159
|
+
log.error("could not write", path=args.output, error=str(error))
|
|
160
|
+
return 4
|
|
161
|
+
log.info("written", path=args.output)
|
|
162
|
+
else:
|
|
163
|
+
sys.stdout.write(result.text)
|
|
164
|
+
return 0
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
if __name__ == "__main__":
|
|
168
|
+
raise SystemExit(main())
|
scrubbr/detect.py
ADDED
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
import re
|
|
2
|
+
from dataclasses import dataclass
|
|
3
|
+
from functools import lru_cache
|
|
4
|
+
|
|
5
|
+
from scrubbr.identity import LocalIdentity
|
|
6
|
+
from scrubbr.kinds import Kind
|
|
7
|
+
from scrubbr.shapes import EMAIL_PATTERN, HEX_PATTERN, UUID_PATTERN, classify_literal
|
|
8
|
+
|
|
9
|
+
# An RSA-8192 key is around 12 KB of base64, so this fits any real key while keeping an
|
|
10
|
+
# unterminated BEGIN marker from backtracking across the whole file.
|
|
11
|
+
PEM_MAX_BODY = 20_000
|
|
12
|
+
|
|
13
|
+
NO_IDENTITY = LocalIdentity()
|
|
14
|
+
|
|
15
|
+
SECRET_KEYWORDS = (
|
|
16
|
+
"psk",
|
|
17
|
+
"password",
|
|
18
|
+
"passwd",
|
|
19
|
+
"secret",
|
|
20
|
+
"api_key",
|
|
21
|
+
"apikey",
|
|
22
|
+
"access_token",
|
|
23
|
+
"auth_token",
|
|
24
|
+
"client_secret",
|
|
25
|
+
"private_key",
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass(frozen=True)
|
|
30
|
+
class Rule:
|
|
31
|
+
name: str
|
|
32
|
+
kind: Kind
|
|
33
|
+
pattern: str
|
|
34
|
+
value_group: str | None = None
|
|
35
|
+
forced: bool = False
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class Match:
|
|
40
|
+
kind: Kind
|
|
41
|
+
start: int
|
|
42
|
+
end: int
|
|
43
|
+
text: str
|
|
44
|
+
forced: bool = False
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
# Ordered most-specific first. Alternation resolves precedence: at any position the
|
|
48
|
+
# earliest alternative that matches wins, which is what makes the colon-hex family
|
|
49
|
+
# (fingerprint / MAC / IPv6) disambiguate correctly.
|
|
50
|
+
STRUCTURAL_RULES: tuple[Rule, ...] = (
|
|
51
|
+
Rule(
|
|
52
|
+
"pem",
|
|
53
|
+
Kind.PEM,
|
|
54
|
+
# The body stays a wildcard on purpose. Narrowing it to the base64 alphabet would
|
|
55
|
+
# make one stray character anywhere in the block fail the whole alternative and
|
|
56
|
+
# pass the key through verbatim; the length bound alone removes the backtracking.
|
|
57
|
+
r"-----BEGIN [A-Z0-9 ._-]{0,100}-----"
|
|
58
|
+
rf"(?P<pem_body>[\s\S]{{0,{PEM_MAX_BODY}}}?)"
|
|
59
|
+
r"-----END [A-Z0-9 ._-]{0,100}-----",
|
|
60
|
+
value_group="pem_body",
|
|
61
|
+
),
|
|
62
|
+
# A log cut off mid-key has a BEGIN marker and no END. The complete rule above is
|
|
63
|
+
# tried first, so this only ever fires on a truncated block -- without it the key body
|
|
64
|
+
# matches nothing at all and survives verbatim, which truncated diagnostics make common.
|
|
65
|
+
Rule(
|
|
66
|
+
"pem_truncated",
|
|
67
|
+
Kind.PEM,
|
|
68
|
+
r"-----BEGIN [A-Z0-9 ._-]{0,100}-----"
|
|
69
|
+
rf"(?P<pem_open>(?:\r?\n[A-Za-z0-9+/=]{{16,80}}){{1,{PEM_MAX_BODY // 16}}})",
|
|
70
|
+
value_group="pem_open",
|
|
71
|
+
),
|
|
72
|
+
Rule(
|
|
73
|
+
"crypt_hash",
|
|
74
|
+
Kind.CRYPT_HASH,
|
|
75
|
+
r"\$(?:1|5|6|2[aby]|apr1)\$(?:rounds=\d+\$)?[./A-Za-z0-9]{1,16}\$[./A-Za-z0-9]{20,90}",
|
|
76
|
+
),
|
|
77
|
+
Rule("jwt", Kind.JWT, r"eyJ[A-Za-z0-9_-]{6,}\.[A-Za-z0-9_-]{6,}\.[A-Za-z0-9_-]*"),
|
|
78
|
+
Rule("disk_id", Kind.DISK_ID, r"/dev/disk/by-id/(?P<disk_id_val>[A-Za-z0-9._:+-]+)",
|
|
79
|
+
value_group="disk_id_val"),
|
|
80
|
+
# 8+ colon-hex groups: a digest fingerprint, never a MAC.
|
|
81
|
+
Rule(
|
|
82
|
+
"fingerprint",
|
|
83
|
+
Kind.FINGERPRINT,
|
|
84
|
+
r"(?<![0-9A-Fa-f:])(?:[0-9A-Fa-f]{2}:){7,}[0-9A-Fa-f]{2}(?![0-9A-Fa-f:])",
|
|
85
|
+
),
|
|
86
|
+
Rule("uuid", Kind.UUID, rf"(?<![0-9A-Fa-f-]){UUID_PATTERN}(?![0-9A-Fa-f-])"),
|
|
87
|
+
# Exactly six 2-digit groups with a consistent separator, fenced by lookarounds so it
|
|
88
|
+
# cannot bite a slice out of a longer colon-hex chain.
|
|
89
|
+
# One rule per separator rather than one rule with a backreferenced separator: each
|
|
90
|
+
# fence then excludes only its OWN separator, so a colon-separated address is still
|
|
91
|
+
# found in "aa:bb:cc:dd:ee:ff-eth0" and "wlan0-aa:bb:cc:dd:ee:ff", while a
|
|
92
|
+
# hyphen-separated one is still fenced off from a longer hyphenated run.
|
|
93
|
+
Rule(
|
|
94
|
+
"mac_colon",
|
|
95
|
+
Kind.MAC,
|
|
96
|
+
r"(?<![0-9A-Fa-f:])(?:[0-9A-Fa-f]{2}:){5}[0-9A-Fa-f]{2}(?![0-9A-Fa-f:])",
|
|
97
|
+
),
|
|
98
|
+
Rule(
|
|
99
|
+
"mac_hyphen",
|
|
100
|
+
Kind.MAC,
|
|
101
|
+
r"(?<![0-9A-Fa-f-])(?:[0-9A-Fa-f]{2}-){5}[0-9A-Fa-f]{2}(?![0-9A-Fa-f-])",
|
|
102
|
+
),
|
|
103
|
+
# A trailing dot only disqualifies the dotted form when more hex follows it.
|
|
104
|
+
Rule(
|
|
105
|
+
"mac_cisco",
|
|
106
|
+
Kind.MAC,
|
|
107
|
+
r"(?<![0-9A-Fa-f])(?<![0-9A-Fa-f]\.)"
|
|
108
|
+
r"[0-9A-Fa-f]{4}\.[0-9A-Fa-f]{4}\.[0-9A-Fa-f]{4}"
|
|
109
|
+
r"(?![0-9A-Fa-f])(?!\.[0-9A-Fa-f])",
|
|
110
|
+
),
|
|
111
|
+
# Bare 12-hex is only a MAC when something says so; on its own it is indistinguishable
|
|
112
|
+
# from a truncated hash, and scrubbing every 12-hex run would wreck ordinary logs.
|
|
113
|
+
Rule(
|
|
114
|
+
"mac_bare",
|
|
115
|
+
Kind.MAC,
|
|
116
|
+
r"(?i:mac|hwaddr|hw_addr|bssid|ether|lladdr|hw)(?:\s+address)?\s*[=:]?\s*"
|
|
117
|
+
r"(?P<mac_bare_val>[0-9A-Fa-f]{12})(?![0-9A-Za-z])",
|
|
118
|
+
value_group="mac_bare_val",
|
|
119
|
+
),
|
|
120
|
+
# Ahead of the identity literals: a username is usually the local part of its owner's
|
|
121
|
+
# address, and letting the literal win would rewrite "dev@example.com" to
|
|
122
|
+
# "user-a@example.com", keeping the domain — normally the more identifying half.
|
|
123
|
+
Rule("email", Kind.EMAIL, EMAIL_PATTERN),
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
CONTEXTUAL_RULES: tuple[Rule, ...] = (
|
|
127
|
+
Rule(
|
|
128
|
+
"ssid",
|
|
129
|
+
Kind.SSID,
|
|
130
|
+
r"(?i:ssid)\s*[=:]\s*(?P<ssid_q>[\"'])(?P<ssid_val>[^\"']*)(?P=ssid_q)",
|
|
131
|
+
value_group="ssid_val",
|
|
132
|
+
),
|
|
133
|
+
Rule(
|
|
134
|
+
"secret_kv",
|
|
135
|
+
Kind.SECRET_VALUE,
|
|
136
|
+
r"(?i:" + "|".join(SECRET_KEYWORDS) + r")\s*[=:]\s*"
|
|
137
|
+
r"(?P<secret_q>[\"']?)(?P<secret_val>[^\s\"',;]+)(?P=secret_q)",
|
|
138
|
+
value_group="secret_val",
|
|
139
|
+
),
|
|
140
|
+
Rule(
|
|
141
|
+
"home_path",
|
|
142
|
+
Kind.USERNAME,
|
|
143
|
+
r"/home/(?P<home_user>[a-z_][a-z0-9_-]{0,31})",
|
|
144
|
+
value_group="home_user",
|
|
145
|
+
),
|
|
146
|
+
Rule(
|
|
147
|
+
"ipv6",
|
|
148
|
+
Kind.IPV6,
|
|
149
|
+
r"(?<![0-9A-Fa-f:.])[0-9A-Fa-f]{0,4}(?::[0-9A-Fa-f]{0,4}){2,7}(?![0-9A-Fa-f:])",
|
|
150
|
+
),
|
|
151
|
+
Rule("ipv4", Kind.IPV4, r"(?<![0-9.])(?:[0-9]{1,3}\.){3}[0-9]{1,3}(?![0-9.])"),
|
|
152
|
+
Rule(
|
|
153
|
+
"hex",
|
|
154
|
+
Kind.HEX,
|
|
155
|
+
rf"(?<![0-9A-Za-z])(?:0[xX])?(?P<hex_val>{HEX_PATTERN})(?![0-9A-Za-z])",
|
|
156
|
+
value_group="hex_val",
|
|
157
|
+
),
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _literal_rules(
|
|
162
|
+
identity: LocalIdentity, promoted: tuple[tuple[str, Kind], ...]
|
|
163
|
+
) -> tuple[Rule, ...]:
|
|
164
|
+
literals: list[tuple[str, Kind, bool]] = []
|
|
165
|
+
if identity.hostname:
|
|
166
|
+
literals.append((identity.hostname, Kind.HOSTNAME, False))
|
|
167
|
+
short = identity.hostname.split(".")[0]
|
|
168
|
+
if short != identity.hostname:
|
|
169
|
+
literals.append((short, Kind.HOSTNAME, False))
|
|
170
|
+
# An extra value is explicitly declared, so it is scrubbed unconditionally -- forced
|
|
171
|
+
# past the keep-allowlists that judge only incidental matches.
|
|
172
|
+
literals.extend((value, classify_literal(value), True) for value in identity.extra)
|
|
173
|
+
if identity.username:
|
|
174
|
+
literals.append((identity.username, Kind.USERNAME, False))
|
|
175
|
+
literals.extend((value, kind, False) for value, kind in promoted)
|
|
176
|
+
# Longest first: a username is frequently a substring of the hostname ("dev" inside
|
|
177
|
+
# "dev-thinkpad"), and the longer match has to win at that position.
|
|
178
|
+
literals.sort(key=lambda entry: len(entry[0]), reverse=True)
|
|
179
|
+
return tuple(
|
|
180
|
+
Rule(f"literal_{index}", kind, _literal_pattern(value, kind), forced=forced)
|
|
181
|
+
for index, (value, kind, forced) in enumerate(literals)
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _literal_pattern(value: str, kind: Kind) -> str:
|
|
186
|
+
escaped = re.escape(value)
|
|
187
|
+
if kind is Kind.IPV6:
|
|
188
|
+
# Forcing must survive the log spelling FE80::1 as fe80::1.
|
|
189
|
+
escaped = f"(?i:{escaped})"
|
|
190
|
+
return rf"(?<![A-Za-z0-9_]){escaped}(?![A-Za-z0-9_])"
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
@lru_cache(maxsize=64)
|
|
194
|
+
def _compiled(
|
|
195
|
+
identity: LocalIdentity, promoted: tuple[tuple[str, Kind], ...]
|
|
196
|
+
) -> tuple[re.Pattern[str], dict[str, Rule]]:
|
|
197
|
+
rules = STRUCTURAL_RULES + _literal_rules(identity, promoted) + CONTEXTUAL_RULES
|
|
198
|
+
combined = "|".join(f"(?P<{rule.name}>{rule.pattern})" for rule in rules)
|
|
199
|
+
return re.compile(combined), {rule.name: rule for rule in rules}
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def detect(
|
|
203
|
+
text: str,
|
|
204
|
+
identity: LocalIdentity = NO_IDENTITY,
|
|
205
|
+
promoted: tuple[tuple[str, Kind], ...] = (),
|
|
206
|
+
) -> list[Match]:
|
|
207
|
+
"""Locate everything worth replacing.
|
|
208
|
+
|
|
209
|
+
`promoted` carries values a first pass discovered behind a keyword — `psk=secret` —
|
|
210
|
+
so that a second pass also catches them where they appear bare and unlabelled.
|
|
211
|
+
"""
|
|
212
|
+
pattern, by_name = _compiled(identity, promoted)
|
|
213
|
+
matches: list[Match] = []
|
|
214
|
+
for found in pattern.finditer(text):
|
|
215
|
+
rule = by_name[_which(found, by_name)]
|
|
216
|
+
group = rule.value_group or rule.name
|
|
217
|
+
start, end = found.span(group)
|
|
218
|
+
if start < 0 or start == end:
|
|
219
|
+
continue
|
|
220
|
+
matches.append(
|
|
221
|
+
Match(kind=rule.kind, start=start, end=end, text=text[start:end], forced=rule.forced)
|
|
222
|
+
)
|
|
223
|
+
return matches
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _which(found: re.Match[str], by_name: dict[str, Rule]) -> str:
|
|
227
|
+
for name in by_name:
|
|
228
|
+
if found.group(name) is not None:
|
|
229
|
+
return name
|
|
230
|
+
raise AssertionError("a match must belong to exactly one rule")
|
scrubbr/identity.py
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
import getpass
|
|
2
|
+
import socket
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
SYSTEM_USERNAMES = frozenset(
|
|
7
|
+
{
|
|
8
|
+
"root",
|
|
9
|
+
"daemon",
|
|
10
|
+
"bin",
|
|
11
|
+
"sys",
|
|
12
|
+
"sync",
|
|
13
|
+
"man",
|
|
14
|
+
"nobody",
|
|
15
|
+
"messagebus",
|
|
16
|
+
"dbus",
|
|
17
|
+
"polkitd",
|
|
18
|
+
"sshd",
|
|
19
|
+
"www-data",
|
|
20
|
+
"systemd-network",
|
|
21
|
+
"systemd-resolve",
|
|
22
|
+
"systemd-timesync",
|
|
23
|
+
"systemd-journal",
|
|
24
|
+
"user",
|
|
25
|
+
}
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass(frozen=True)
|
|
30
|
+
class LocalIdentity:
|
|
31
|
+
"""Literal strings naming *this* machine and *this* person.
|
|
32
|
+
|
|
33
|
+
Seeding the scanner with these is the highest-precision rule available: the surest
|
|
34
|
+
way to know a hostname is yours is to ask the system for it, rather than to guess at
|
|
35
|
+
which token in a syslog line is a hostname.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
hostname: str | None = None
|
|
39
|
+
username: str | None = None
|
|
40
|
+
extra: tuple[str, ...] = field(default_factory=tuple)
|
|
41
|
+
|
|
42
|
+
@classmethod
|
|
43
|
+
def local(cls) -> "LocalIdentity":
|
|
44
|
+
extra: list[str] = []
|
|
45
|
+
try:
|
|
46
|
+
machine_id = Path("/etc/machine-id").read_text(encoding="utf-8").strip()
|
|
47
|
+
except OSError:
|
|
48
|
+
machine_id = ""
|
|
49
|
+
if machine_id:
|
|
50
|
+
extra.append(machine_id)
|
|
51
|
+
try:
|
|
52
|
+
username: str | None = getpass.getuser()
|
|
53
|
+
except (OSError, KeyError):
|
|
54
|
+
username = None
|
|
55
|
+
try:
|
|
56
|
+
hostname: str | None = socket.gethostname()
|
|
57
|
+
except OSError:
|
|
58
|
+
hostname = None
|
|
59
|
+
return cls(hostname=hostname, username=username, extra=tuple(extra))
|
scrubbr/kinds.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
from enum import StrEnum
|
|
2
|
+
|
|
3
|
+
from pydantic import BaseModel, ConfigDict
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class Kind(StrEnum):
|
|
7
|
+
PEM = "pem"
|
|
8
|
+
CRYPT_HASH = "crypt_hash"
|
|
9
|
+
JWT = "jwt"
|
|
10
|
+
FINGERPRINT = "fingerprint"
|
|
11
|
+
DISK_ID = "disk_id"
|
|
12
|
+
UUID = "uuid"
|
|
13
|
+
MAC = "mac"
|
|
14
|
+
IPV6 = "ipv6"
|
|
15
|
+
LINK_LOCAL_ID = "link_local_id"
|
|
16
|
+
IPV4 = "ipv4"
|
|
17
|
+
HEX = "hex"
|
|
18
|
+
EMAIL = "email"
|
|
19
|
+
SECRET_VALUE = "secret_value"
|
|
20
|
+
REDACTED = "redacted"
|
|
21
|
+
SSID = "ssid"
|
|
22
|
+
HOSTNAME = "hostname"
|
|
23
|
+
USERNAME = "username"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class Finding(BaseModel):
|
|
27
|
+
model_config = ConfigDict(frozen=True)
|
|
28
|
+
|
|
29
|
+
kind: Kind
|
|
30
|
+
start: int
|
|
31
|
+
end: int
|
|
32
|
+
text: str
|
|
33
|
+
alias: str
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class Residual(BaseModel):
|
|
37
|
+
model_config = ConfigDict(frozen=True)
|
|
38
|
+
|
|
39
|
+
line: int
|
|
40
|
+
text: str
|
|
41
|
+
reason: str
|
scrubbr/report.py
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
import textwrap
|
|
2
|
+
from collections import Counter
|
|
3
|
+
|
|
4
|
+
from scrubbr.kinds import Finding, Kind, Residual
|
|
5
|
+
from scrubbr.scrub import ScrubResult
|
|
6
|
+
|
|
7
|
+
# The report names values, it does not reproduce them: a PEM body is kilobytes across
|
|
8
|
+
# many lines, and no terminal width justifies showing more of a secret.
|
|
9
|
+
CLIP = 60
|
|
10
|
+
# Below this the columns stop being readable anyway; render as if the terminal were this
|
|
11
|
+
# wide and let it wrap.
|
|
12
|
+
MIN_WIDTH = 40
|
|
13
|
+
_GUTTER = " "
|
|
14
|
+
# StrEnum sorts alphabetically; declaration order is the sensitivity order, which is what
|
|
15
|
+
# a security report should lead with.
|
|
16
|
+
_KIND_ORDER = {kind: index for index, kind in enumerate(Kind)}
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def render_report(result: ScrubResult, verbose: bool, width: int) -> list[str]:
|
|
20
|
+
width = max(width, MIN_WIDTH)
|
|
21
|
+
lines = _summary(result, width)
|
|
22
|
+
if verbose and result.findings:
|
|
23
|
+
lines += ["", *_findings_table(result.findings, width)]
|
|
24
|
+
if result.residuals:
|
|
25
|
+
lines += ["", *_residuals_table(result.residuals, width)]
|
|
26
|
+
return lines
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _summary(result: ScrubResult, width: int) -> list[str]:
|
|
30
|
+
distinct = sum(result.counts.values())
|
|
31
|
+
lines = [f"scrubbed {distinct} distinct values, {len(result.findings)} replacements"]
|
|
32
|
+
if result.counts:
|
|
33
|
+
by_kind = ", ".join(
|
|
34
|
+
f"{kind.value} {result.counts[kind]}" for kind in Kind if kind in result.counts
|
|
35
|
+
)
|
|
36
|
+
lines += textwrap.wrap(
|
|
37
|
+
by_kind, width=width, initial_indent=_GUTTER, subsequent_indent=_GUTTER
|
|
38
|
+
)
|
|
39
|
+
return lines
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _findings_table(findings: list[Finding], width: int) -> list[str]:
|
|
43
|
+
occurrences = Counter((f.kind, f.text, f.alias) for f in findings)
|
|
44
|
+
rows = sorted(
|
|
45
|
+
(
|
|
46
|
+
(kind, _flatten(text), count, _flatten(alias))
|
|
47
|
+
for (kind, text, alias), count in occurrences.items()
|
|
48
|
+
),
|
|
49
|
+
key=lambda row: (_KIND_ORDER[row[0]], -row[2], row[1]),
|
|
50
|
+
)
|
|
51
|
+
kind_w = max(len("kind"), *(len(kind.value) for kind, _, _, _ in rows))
|
|
52
|
+
count_w = max(len("count"), *(len(str(count)) for _, _, count, _ in rows))
|
|
53
|
+
text_w, alias_w = _apportion(
|
|
54
|
+
width - kind_w - count_w - 3 * len(_GUTTER),
|
|
55
|
+
min(CLIP, max(len("text"), *(len(text) for _, text, _, _ in rows))),
|
|
56
|
+
min(CLIP, max(len("alias"), *(len(alias) for _, _, _, alias in rows))),
|
|
57
|
+
)
|
|
58
|
+
template = f"{{:<{kind_w}}}{_GUTTER}{{:>{count_w}}}{_GUTTER}{{:<{text_w}}}{_GUTTER}{{}}"
|
|
59
|
+
header = template.format("kind", "count", _fit("text", text_w), _fit("alias", alias_w))
|
|
60
|
+
lines = [header.rstrip()]
|
|
61
|
+
lines += [
|
|
62
|
+
template.format(kind.value, count, _fit(text, text_w), _fit(alias, alias_w)).rstrip()
|
|
63
|
+
for kind, text, count, alias in rows
|
|
64
|
+
]
|
|
65
|
+
return lines
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _residuals_table(residuals: list[Residual], width: int) -> list[str]:
|
|
69
|
+
line_w = max(len("line"), *(len(str(residual.line)) for residual in residuals))
|
|
70
|
+
reason_w = max(len("reason"), *(len(residual.reason) for residual in residuals))
|
|
71
|
+
text_w = min(CLIP, width - line_w - reason_w - 3 * len(_GUTTER))
|
|
72
|
+
template = f"{_GUTTER}{{:>{line_w}}}{_GUTTER}{{:<{reason_w}}}{_GUTTER}{{}}"
|
|
73
|
+
lines = [f"{len(residuals)} unscrubbed strings look sensitive:"]
|
|
74
|
+
lines.append(template.format("line", "reason", "text").rstrip())
|
|
75
|
+
lines += [
|
|
76
|
+
template.format(residual.line, residual.reason, _fit(_flatten(residual.text), text_w))
|
|
77
|
+
for residual in residuals
|
|
78
|
+
]
|
|
79
|
+
return lines
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _apportion(avail: int, text_nat: int, alias_nat: int) -> tuple[int, int]:
|
|
83
|
+
if text_nat + alias_nat <= avail:
|
|
84
|
+
return text_nat, alias_nat
|
|
85
|
+
half = avail // 2
|
|
86
|
+
if text_nat <= half:
|
|
87
|
+
return text_nat, avail - text_nat
|
|
88
|
+
if alias_nat <= half:
|
|
89
|
+
return avail - alias_nat, alias_nat
|
|
90
|
+
return max(avail - half, 0), max(half, 0)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _fit(value: str, width: int) -> str:
|
|
94
|
+
if len(value) <= width:
|
|
95
|
+
return value
|
|
96
|
+
if width <= 0:
|
|
97
|
+
return ""
|
|
98
|
+
keep = width - 1
|
|
99
|
+
tail = keep // 2
|
|
100
|
+
return value[: keep - tail] + "…" + value[len(value) - tail :]
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _flatten(value: str) -> str:
|
|
104
|
+
return " ".join(value.split())
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def clip(value: str) -> str:
|
|
108
|
+
return _flatten(value)[:CLIP]
|