ecoportal-api 0.10.15 → 0.10.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of ecoportal-api might be problematic. Click here for more details.
- checksums.yaml +4 -4
- data/.ai-assistance/.gitignore +2 -0
- data/.ai-assistance/bridge/.gitignore +10 -0
- data/.ai-assistance/bridge/CLAUDE.md +96 -0
- data/.ai-assistance/bridge/archive/.gitkeep +0 -0
- data/.ai-assistance/bridge/inbox/.gitkeep +0 -0
- data/.ai-assistance/bridge/outbox/.gitkeep +0 -0
- data/.ai-assistance/capabilities/assumptions-log.md +23 -0
- data/.ai-assistance/scripts/bridge-inbox-check.sh +119 -0
- data/.ai-assistance/scripts/bridge-init.sh +86 -0
- data/.ai-assistance/scripts/confine-to-subtree.sh +58 -0
- data/.ai-assistance/scripts/dirty-tree-guard.sh +96 -0
- data/.ai-assistance/scripts/distill_procedural.py +602 -0
- data/.ai-assistance/scripts/log-mcp-access.sh +24 -0
- data/.ai-assistance/scripts/log-skill-usage.sh +79 -0
- data/.ai-assistance/scripts/log_mcp_access.py +158 -0
- data/.ai-assistance/scripts/observe-session.sh +13 -0
- data/.ai-assistance/scripts/observe_session.py +287 -0
- data/.ai-assistance/scripts/protect-host-paths.sh +135 -0
- data/.ai-assistance/scripts/scrub.py +1149 -0
- data/.ai-assistance/scripts/scrub.py.sha256 +6 -0
- data/.ai-assistance/scripts/surface-procedural.sh +9 -0
- data/.ai-assistance/scripts/surface_procedural.py +101 -0
- data/.ai-assistance/skills/ep-ai-manager/SKILL.md +519 -0
- data/.ai-assistance/skills/project-self-docs/SKILL.md +259 -0
- data/.ai-assistance/skills/project-self-docs/scripts/self_docs_scan.py +378 -0
- data/.ai-assistance/standards-version.json +12 -0
- data/.ai-assistance/version.json +8 -0
- data/.claude/.gitignore +2 -0
- data/.claude/settings.json +128 -0
- data/CHANGELOG.md +14 -1
- data/CLAUDE.md +95 -0
- data/docs/self-docs/ARCHITECTURE.md +145 -0
- data/docs/self-docs/CHANGES.jsonl +7 -0
- data/docs/self-docs/COMPLIANCE.md +66 -0
- data/docs/self-docs/CONVENTIONS.md +74 -0
- data/docs/self-docs/INTEGRATIONS.md +62 -0
- data/docs/self-docs/OPERATIONS.md +64 -0
- data/docs/self-docs/OVERVIEW.md +61 -0
- data/docs/self-docs/STATUS.md +71 -0
- data/docs/self-docs/self-docs-index.json +51 -0
- data/docs/worklog.md +48 -0
- data/lib/ecoportal/api/common/client/with_retry.rb +6 -0
- data/lib/ecoportal/api/common/client.rb +12 -10
- data/lib/ecoportal/api/version.rb +1 -1
- metadata +41 -1
|
@@ -0,0 +1,1149 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
scrub.py -- conservative, deterministic PII / secret / local-path redactor.
|
|
4
|
+
|
|
5
|
+
FIRST PASS -- PENDING SECURITY REVIEW. This module is the ADR-011 /
|
|
6
|
+
external-llm-review prerequisite: before any learning or worklog content is
|
|
7
|
+
egressed to an external system (EPAI Confluence, Gemini), it must be stripped of
|
|
8
|
+
PII, secrets, and machine-local filesystem paths. Today seed-epai-learnings.py
|
|
9
|
+
--push is gated behind a manual EPAI_PUSH_CONFIRM=1 attestation precisely because
|
|
10
|
+
no scrub function existed. This file provides that function.
|
|
11
|
+
|
|
12
|
+
WIRING STATUS. Two consumers rely on this module:
|
|
13
|
+
- scripts/toggl-timesheet.py -- scrub() redaction of PII-adjacent free text (PII docking).
|
|
14
|
+
- scripts/seed-epai-learnings.py --push -- find_secrets() as a fail-closed BLOCK gate
|
|
15
|
+
before publishing auto-harvested content to the ROVO-indexed EPAI Confluence space.
|
|
16
|
+
Two egress models, deliberately different (see find_secrets vs scrub below):
|
|
17
|
+
- REDACT (scrub) for LLM egress: over-redaction is acceptable, so replace secrets with
|
|
18
|
+
placeholders and send the cleaned text.
|
|
19
|
+
- BLOCK (find_secrets) for PUBLISH / bridge egress: silently rewriting a human-readable
|
|
20
|
+
doc would corrupt it, so instead REFUSE egress when a high-confidence secret is present.
|
|
21
|
+
- .ai-assistance/skills/gemini-assist/gemini_ask.rb -- scrub_for_egress() REDACTS the full
|
|
22
|
+
prompt (files + task) before it is sent to the Gemini API. Fail-closed: it shells out to
|
|
23
|
+
THIS module and aborts the send if scrubbing cannot run.
|
|
24
|
+
The regexes are heuristics; they over-redact by design (false positives) and may still have
|
|
25
|
+
false negatives. Treat findings as a strong prompt, not an absolute guarantee. Egress paths not
|
|
26
|
+
separately gated are covered by compensating controls: the docs/structure Confluence seeders
|
|
27
|
+
publish COMMITTED content already guarded by the secret_scan CI check + human review; bridge
|
|
28
|
+
inbox/outbox are gitignored (no git leak) and the committed bridge archive is covered by
|
|
29
|
+
secret_scan. Wire an explicit gate there too if belt-and-suspenders is wanted.
|
|
30
|
+
|
|
31
|
+
Design notes:
|
|
32
|
+
- Stdlib only. ASCII only (see .ai-assistance/conventions/documentation-style.md).
|
|
33
|
+
- No network. No LLM. Pure-function, deterministic given the same input.
|
|
34
|
+
- Conservative: prefer false-positives (over-redaction) over leaks. When in doubt
|
|
35
|
+
for a high-entropy token, we redact and record a finding so a human sees it.
|
|
36
|
+
- scrub() NEVER raises: on any internal error it returns the original text plus an
|
|
37
|
+
{"type": "error", "count": 1} finding, so a caller can detect that scrubbing
|
|
38
|
+
did not run cleanly and refuse to egress.
|
|
39
|
+
- Redactions replace the match with a clear placeholder, e.g. [REDACTED:email].
|
|
40
|
+
|
|
41
|
+
Public API:
|
|
42
|
+
scrub(text) -> (scrubbed_text, findings)
|
|
43
|
+
findings is a list of {"type": str, "count": int}, one entry per pattern
|
|
44
|
+
that fired at least once.
|
|
45
|
+
scrub_file(path) -> (scrubbed_text, findings)
|
|
46
|
+
|
|
47
|
+
CLI:
|
|
48
|
+
python scripts/lib/scrub.py <file> # scrubbed text -> stdout, summary -> stderr
|
|
49
|
+
python scripts/lib/scrub.py --preview <file>
|
|
50
|
+
# NON-destructive redaction report (scrubbed
|
|
51
|
+
# text + findings table + MASKED samples);
|
|
52
|
+
# sends nothing, reveals no secret
|
|
53
|
+
python scripts/lib/scrub.py --selftest # run built-in pattern assertions, exit 0/1
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
import json
|
|
57
|
+
import os
|
|
58
|
+
import re
|
|
59
|
+
import sys
|
|
60
|
+
from pathlib import Path
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
# ---------------------------------------------------------------------------
|
|
64
|
+
# Code-fence masking
|
|
65
|
+
# ---------------------------------------------------------------------------
|
|
66
|
+
# High-entropy redaction is intentionally NOT applied inside fenced code blocks
|
|
67
|
+
# (``` ... ``` or ~~~ ... ~~~), because legitimate hashes, base64 fixtures, and
|
|
68
|
+
# sample data commonly live there and redacting them would mangle documentation.
|
|
69
|
+
# All the *targeted* patterns (emails, paths, known secret prefixes, AWS keys,
|
|
70
|
+
# JWTs, etc.) ARE still applied everywhere, including inside code fences -- a real
|
|
71
|
+
# secret pasted into a code block is still a leak.
|
|
72
|
+
|
|
73
|
+
_FENCE_RE = re.compile(r"(?ms)^[ \t]*(`{3,}|~{3,}).*?^[ \t]*\1[ \t]*$")
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _fence_spans(text):
|
|
77
|
+
"""Return a list of (start, end) char spans covered by fenced code blocks."""
|
|
78
|
+
spans = []
|
|
79
|
+
try:
|
|
80
|
+
for m in _FENCE_RE.finditer(text):
|
|
81
|
+
spans.append((m.start(), m.end()))
|
|
82
|
+
except Exception:
|
|
83
|
+
# Never let fence detection break scrubbing; just treat as no fences.
|
|
84
|
+
return []
|
|
85
|
+
return spans
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _in_spans(pos, spans):
|
|
89
|
+
for s, e in spans:
|
|
90
|
+
if s <= pos < e:
|
|
91
|
+
return True
|
|
92
|
+
return False
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
# ---------------------------------------------------------------------------
|
|
96
|
+
# Pattern table
|
|
97
|
+
# ---------------------------------------------------------------------------
|
|
98
|
+
# Each entry: (type_name, compiled_regex, respect_code_fences)
|
|
99
|
+
# respect_code_fences=True means matches inside fenced code blocks are skipped
|
|
100
|
+
# (used only for the broad high-entropy catch-alls).
|
|
101
|
+
#
|
|
102
|
+
# Order matters: more specific / higher-confidence patterns run first so that,
|
|
103
|
+
# e.g., an AWS key or JWT is labelled as such rather than swallowed by the
|
|
104
|
+
# generic high-entropy rule. Once a region is redacted it is replaced by an
|
|
105
|
+
# ASCII placeholder that later patterns will not re-match.
|
|
106
|
+
|
|
107
|
+
# Email -- deliberately broad local part.
|
|
108
|
+
_EMAIL = re.compile(
|
|
109
|
+
r"[A-Za-z0-9._%+\-]+@[A-Za-z0-9.\-]+\.[A-Za-z]{2,}"
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
# Windows absolute path: drive letter + (\ or /) + at least one segment.
|
|
113
|
+
# Matches C:\Users\rella\... and C:/claude/Projects/...
|
|
114
|
+
_WIN_PATH = re.compile(
|
|
115
|
+
r"[A-Za-z]:[\\/](?:[^\s\\/:*?\"<>|]+[\\/]?)+"
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
# POSIX user-home / mount absolute paths.
|
|
119
|
+
# /home/<user>/... /Users/<user>/... /mnt/... /root/... /var/.../<user>...
|
|
120
|
+
# Kept reasonably tight (requires a known root) to limit false positives on
|
|
121
|
+
# ordinary prose containing slashes. Still over-redacts URLs paths sometimes;
|
|
122
|
+
# acceptable per the conservative contract.
|
|
123
|
+
_POSIX_HOME = re.compile(
|
|
124
|
+
r"(?:/home/|/Users/|/mnt/|/media/|/root/|/srv/|/opt/)"
|
|
125
|
+
r"[^\s\"'<>|:]*"
|
|
126
|
+
)
|
|
127
|
+
# Bare ~ home expansion: ~/something or ~user/something
|
|
128
|
+
_TILDE_HOME = re.compile(r"~[A-Za-z0-9_\-]*\/[^\s\"'<>|:]+")
|
|
129
|
+
|
|
130
|
+
# Secret prefixes (high confidence).
|
|
131
|
+
_AWS_AKIA = re.compile(r"\b(?:AKIA|ASIA|AROA|AIDA|ANPA|ANVA|AIPA)[0-9A-Z]{16}\b")
|
|
132
|
+
_GITLAB_PAT = re.compile(r"\bglpat-[A-Za-z0-9_\-]{20,}\b")
|
|
133
|
+
_GITHUB_PAT = re.compile(r"\b(?:ghp|gho|ghu|ghs|ghr)_[A-Za-z0-9]{36,}\b")
|
|
134
|
+
_GITHUB_FINE_PAT = re.compile(r"\bgithub_pat_[A-Za-z0-9_]{30,}\b")
|
|
135
|
+
_GOOGLE_API = re.compile(r"\bAIza[0-9A-Za-z_\-]{35}\b")
|
|
136
|
+
# Google/Gemini newer API-key format: "AQ." + base64url body (e.g. AQ.Ab8R...).
|
|
137
|
+
# The AIza pattern above does not cover it; observed on live Gemini keys.
|
|
138
|
+
_GOOGLE_AQ = re.compile(r"\bAQ\.[A-Za-z0-9_\-]{30,}\b")
|
|
139
|
+
# Slack tokens (bonus, high confidence).
|
|
140
|
+
_SLACK = re.compile(r"\bxox[baprs]-[A-Za-z0-9\-]{10,}\b")
|
|
141
|
+
# JWT: three dot-separated base64url segments starting with the typical eyJ header.
|
|
142
|
+
_JWT = re.compile(r"\beyJ[A-Za-z0-9_\-]+\.[A-Za-z0-9_\-]+\.[A-Za-z0-9_\-]+\b")
|
|
143
|
+
|
|
144
|
+
# Ported from the AWS corpus-pipeline scrubber (do not lose these scoped detectors):
|
|
145
|
+
_ATLASSIAN = re.compile(r"\bATATT3x[A-Za-z0-9]{30,}\b") # Atlassian API token
|
|
146
|
+
_ANTHROPIC = re.compile(r"\bsk-ant-[A-Za-z0-9_\-]{40,}\b") # Anthropic API key
|
|
147
|
+
_OPENAI = re.compile(r"\bsk-[A-Za-z0-9]{48,}\b") # OpenAI API key
|
|
148
|
+
_AWS_ARN = re.compile(r"\barn:aws:[a-z0-9\-]+:[a-z0-9\-]*:\d{12}:[^\s`\]\n]+")
|
|
149
|
+
_BASIC_AUTH_URL = re.compile(r"//[^@\s:/?#]{3,}:[^@\s:/?#]{3,}@") # user:pass@ in a URL
|
|
150
|
+
# Phone numbers (E.164 international + NZ/AU domestic; NZ-scoped per eP).
|
|
151
|
+
_PHONE_E164 = re.compile(r"\+\d{1,3}[\s.\-]?\(?\d{1,4}\)?[\s.\-]?\d{3,4}[\s.\-]?\d{3,4}\b")
|
|
152
|
+
_PHONE_NZ_AU = re.compile(r"\b0[2-9]\d?[\s.\-]?\d{3,4}[\s.\-]?\d{3,4}\b")
|
|
153
|
+
# Private-range IP in CIDR notation (bare IPs still caught by _IPV4 below).
|
|
154
|
+
_CIDR = re.compile(r"\b(?:\d{1,3}\.){3}\d{1,3}/\d{1,2}\b")
|
|
155
|
+
|
|
156
|
+
# Bearer tokens: "Bearer <token>".
|
|
157
|
+
_BEARER = re.compile(r"(?i)\bbearer\s+[A-Za-z0-9._\-]{8,}")
|
|
158
|
+
|
|
159
|
+
# key=value style secret assignments. Captures common key names with an optional
|
|
160
|
+
# quote around the value. value is anything non-space / non-quote, >=6 chars.
|
|
161
|
+
# NOTE: no leading \b -- the keyword must match even when it is a SUFFIX of a
|
|
162
|
+
# prefixed identifier (GEMINI_API_KEY, CONFLUENCE_EPAI_API_KEY), where the char
|
|
163
|
+
# before "api" is "_" (a word char, so \b fails). The trailing suffix group also
|
|
164
|
+
# lets access-level-suffixed names match (FOO_TOKEN_RO=, BAR_TOKEN_RW=) -- exactly
|
|
165
|
+
# the shape of our own env-var-naming convention. Matching starts mid-identifier,
|
|
166
|
+
# so the redaction keeps the prefix (GEMINI_[REDACTED:secret_assignment]).
|
|
167
|
+
_ASSIGN = re.compile(
|
|
168
|
+
r"(?i)(?:api[_\-]?key|apikey|secret(?:[_\-]?key)?|token|access[_\-]?token|"
|
|
169
|
+
r"auth[_\-]?token|password|passwd|pwd|client[_\-]?secret|private[_\-]?key)"
|
|
170
|
+
r"(?:[_\-][A-Za-z0-9]+)*"
|
|
171
|
+
r"\s*[:=]\s*['\"]?([^\s'\"]{6,})['\"]?"
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
# IPv4 (low priority, optional). Bounded octets to reduce matching version-like
|
|
175
|
+
# strings; still may catch some non-address dotted quads -- acceptable.
|
|
176
|
+
_IPV4 = re.compile(
|
|
177
|
+
r"\b(?:(?:25[0-5]|2[0-4]\d|1?\d?\d)\.){3}(?:25[0-5]|2[0-4]\d|1?\d?\d)\b"
|
|
178
|
+
)
|
|
179
|
+
|
|
180
|
+
# High-entropy catch-alls (only OUTSIDE code fences). Long base64-ish or hex runs
|
|
181
|
+
# that are likely keys/hashes pasted into prose. >= 40 hex chars or >= 32 base64
|
|
182
|
+
# chars. These deliberately over-redact; every hit is recorded as a finding.
|
|
183
|
+
_HEX_LONG = re.compile(r"\b[0-9a-fA-F]{40,}\b")
|
|
184
|
+
_B64_LONG = re.compile(r"\b[A-Za-z0-9+/]{32,}={0,2}\b")
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
# (type, regex, respect_fences). Targeted patterns ignore fences (real secrets
|
|
188
|
+
# in code blocks still leak); only the generic entropy rules respect fences.
|
|
189
|
+
_PATTERNS = [
|
|
190
|
+
("basic_auth_url", _BASIC_AUTH_URL, False),
|
|
191
|
+
("email", _EMAIL, False),
|
|
192
|
+
("aws_access_key", _AWS_AKIA, False),
|
|
193
|
+
("gitlab_pat", _GITLAB_PAT, False),
|
|
194
|
+
("github_pat", _GITHUB_PAT, False),
|
|
195
|
+
("github_fine_pat", _GITHUB_FINE_PAT, False),
|
|
196
|
+
("google_api_key", _GOOGLE_API, False),
|
|
197
|
+
("google_aq_key", _GOOGLE_AQ, False),
|
|
198
|
+
("slack_token", _SLACK, False),
|
|
199
|
+
("jwt", _JWT, False),
|
|
200
|
+
("atlassian_token", _ATLASSIAN, False),
|
|
201
|
+
("anthropic_key", _ANTHROPIC, False),
|
|
202
|
+
("openai_key", _OPENAI, False),
|
|
203
|
+
("aws_arn", _AWS_ARN, False),
|
|
204
|
+
("bearer_token", _BEARER, False),
|
|
205
|
+
("secret_assignment", _ASSIGN, False),
|
|
206
|
+
("windows_path", _WIN_PATH, False),
|
|
207
|
+
("posix_home_path", _POSIX_HOME, False),
|
|
208
|
+
("tilde_home_path", _TILDE_HOME, False),
|
|
209
|
+
("network_cidr", _CIDR, False),
|
|
210
|
+
("phone", _PHONE_E164, False),
|
|
211
|
+
("phone", _PHONE_NZ_AU, False),
|
|
212
|
+
("ipv4", _IPV4, False),
|
|
213
|
+
("high_entropy_hex", _HEX_LONG, True),
|
|
214
|
+
("high_entropy_base64", _B64_LONG, True),
|
|
215
|
+
]
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _placeholder(type_name):
|
|
219
|
+
return "[REDACTED:%s]" % type_name
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _apply_pattern(text, type_name, regex, respect_fences, matches=None):
|
|
223
|
+
"""Replace all matches of one pattern. Returns (new_text, count).
|
|
224
|
+
|
|
225
|
+
Recomputes code-fence spans against the CURRENT text when respect_fences is
|
|
226
|
+
True, so a match's position is checked against fences in the same string.
|
|
227
|
+
|
|
228
|
+
If `matches` is a list, one dict per redaction is appended to it:
|
|
229
|
+
{"type", "start", "end", "replacement", "original"} where start/end are
|
|
230
|
+
offsets INTO THE OUTPUT (scrubbed) text and "original" is the raw matched
|
|
231
|
+
substring (used only for preview masking -- never persisted to egress).
|
|
232
|
+
"""
|
|
233
|
+
placeholder = _placeholder(type_name)
|
|
234
|
+
if respect_fences:
|
|
235
|
+
spans = _fence_spans(text)
|
|
236
|
+
else:
|
|
237
|
+
spans = None
|
|
238
|
+
|
|
239
|
+
count = 0
|
|
240
|
+
out = []
|
|
241
|
+
out_len = 0
|
|
242
|
+
last = 0
|
|
243
|
+
for m in regex.finditer(text):
|
|
244
|
+
start = m.start()
|
|
245
|
+
if spans is not None and _in_spans(start, spans):
|
|
246
|
+
continue
|
|
247
|
+
gap = text[last:start]
|
|
248
|
+
out.append(gap)
|
|
249
|
+
out_len += len(gap)
|
|
250
|
+
if matches is not None:
|
|
251
|
+
matches.append({
|
|
252
|
+
"type": type_name,
|
|
253
|
+
"start": out_len,
|
|
254
|
+
"end": out_len + len(placeholder),
|
|
255
|
+
"replacement": placeholder,
|
|
256
|
+
"original": m.group(0),
|
|
257
|
+
})
|
|
258
|
+
out.append(placeholder)
|
|
259
|
+
out_len += len(placeholder)
|
|
260
|
+
last = m.end()
|
|
261
|
+
count += 1
|
|
262
|
+
out.append(text[last:])
|
|
263
|
+
return "".join(out), count
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
# ---------------------------------------------------------------------------
|
|
267
|
+
# Layer-1 dictionary pass (known private entities -> generic placeholder)
|
|
268
|
+
# ---------------------------------------------------------------------------
|
|
269
|
+
# A deterministic, dictionary-backed redaction of KNOWN private entities
|
|
270
|
+
# (customer organizations, people, internal project names) that recur across
|
|
271
|
+
# ecoPortal work and would otherwise leak on egress. The dictionary is built by
|
|
272
|
+
# scripts/build-private-data-dictionary.py and lives, gitignored, at
|
|
273
|
+
# .ai-assistance/local/private-data/ (override with $PRIVATE_DATA_DICT_DIR).
|
|
274
|
+
#
|
|
275
|
+
# Contract (shared with the builder):
|
|
276
|
+
# Sharded files organizations.json / people.json / projects.json, each:
|
|
277
|
+
# {"version":1, "generated":..., "category":..., "count":N,
|
|
278
|
+
# "entities":[{"key","display","placeholder","aliases":[],"sources":[]}]}
|
|
279
|
+
# Each entity's display + aliases all map, FORWARD-ONLY, to the SAME generic
|
|
280
|
+
# placeholder ([ORG]/[PERSON]/[PROJECT]). There is NO reverse map -- once
|
|
281
|
+
# redacted, the original name is unrecoverable from the output.
|
|
282
|
+
#
|
|
283
|
+
# Semantics: case-insensitive, word-boundary, LONGEST-MATCH-FIRST (so
|
|
284
|
+
# "Briscoe Group" wins over "Briscoe"). A single precompiled regex alternation
|
|
285
|
+
# (sorted longest-first, re.escape'd) does the matching in one pass.
|
|
286
|
+
#
|
|
287
|
+
# Graceful skip: if the directory is absent or holds no usable entities, the pass
|
|
288
|
+
# is a no-op and regex-only behaviour is unchanged.
|
|
289
|
+
|
|
290
|
+
_DICT_SHARD_FILES = ("organizations.json", "people.json", "projects.json")
|
|
291
|
+
# Generic placeholders by category (must agree with the builder's PLACEHOLDERS).
|
|
292
|
+
_DICT_PLACEHOLDERS = {
|
|
293
|
+
"organization": "[ORG]",
|
|
294
|
+
"person": "[PERSON]",
|
|
295
|
+
"project": "[PROJECT]",
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def _dict_dir():
|
|
300
|
+
"""Resolve the private-data dictionary directory.
|
|
301
|
+
|
|
302
|
+
Order: $PRIVATE_DATA_DICT_DIR, else the repo-default
|
|
303
|
+
.ai-assistance/local/private-data/ relative to this file (scripts/lib/).
|
|
304
|
+
"""
|
|
305
|
+
env = os.environ.get("PRIVATE_DATA_DICT_DIR")
|
|
306
|
+
if env:
|
|
307
|
+
return Path(env)
|
|
308
|
+
# scripts/lib/scrub.py -> repo root is two parents up.
|
|
309
|
+
root = Path(__file__).resolve().parent.parent.parent
|
|
310
|
+
return root / ".ai-assistance" / "local" / "private-data"
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def load_dictionary(directory=None):
|
|
314
|
+
"""Load the sharded dictionary into a list of (phrase, placeholder) pairs.
|
|
315
|
+
|
|
316
|
+
Reads organizations.json / people.json / projects.json from `directory`
|
|
317
|
+
(default: _dict_dir()). Each entity contributes its display and every alias
|
|
318
|
+
as a phrase, all mapped to the entity's generic placeholder. Returns [] if
|
|
319
|
+
the directory is absent/empty or nothing usable is found. NEVER raises.
|
|
320
|
+
"""
|
|
321
|
+
try:
|
|
322
|
+
directory = Path(directory) if directory is not None else _dict_dir()
|
|
323
|
+
if not directory.is_dir():
|
|
324
|
+
return []
|
|
325
|
+
pairs = []
|
|
326
|
+
seen = set()
|
|
327
|
+
for fname in _DICT_SHARD_FILES:
|
|
328
|
+
fpath = directory / fname
|
|
329
|
+
if not fpath.is_file():
|
|
330
|
+
continue
|
|
331
|
+
try:
|
|
332
|
+
data = json.loads(fpath.read_text(encoding="utf-8"))
|
|
333
|
+
except Exception:
|
|
334
|
+
# A malformed shard is skipped, not fatal -- regex passes still run.
|
|
335
|
+
continue
|
|
336
|
+
entities = data.get("entities") if isinstance(data, dict) else None
|
|
337
|
+
if not isinstance(entities, list):
|
|
338
|
+
continue
|
|
339
|
+
for ent in entities:
|
|
340
|
+
if not isinstance(ent, dict):
|
|
341
|
+
continue
|
|
342
|
+
placeholder = ent.get("placeholder")
|
|
343
|
+
if not placeholder:
|
|
344
|
+
# Fall back to category default if placeholder missing.
|
|
345
|
+
placeholder = _DICT_PLACEHOLDERS.get(ent.get("category"))
|
|
346
|
+
if not placeholder:
|
|
347
|
+
continue
|
|
348
|
+
phrases = []
|
|
349
|
+
disp = ent.get("display")
|
|
350
|
+
if isinstance(disp, str):
|
|
351
|
+
phrases.append(disp)
|
|
352
|
+
aliases = ent.get("aliases")
|
|
353
|
+
if isinstance(aliases, list):
|
|
354
|
+
phrases.extend(a for a in aliases if isinstance(a, str))
|
|
355
|
+
for phrase in phrases:
|
|
356
|
+
phrase = phrase.strip()
|
|
357
|
+
if not phrase:
|
|
358
|
+
continue
|
|
359
|
+
dedup_key = (phrase.casefold(), placeholder)
|
|
360
|
+
if dedup_key in seen:
|
|
361
|
+
continue
|
|
362
|
+
seen.add(dedup_key)
|
|
363
|
+
pairs.append((phrase, placeholder))
|
|
364
|
+
return pairs
|
|
365
|
+
except Exception:
|
|
366
|
+
return []
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def _compile_dictionary(pairs):
|
|
370
|
+
"""Compile (phrase, placeholder) pairs into (regex, {group->placeholder}).
|
|
371
|
+
|
|
372
|
+
Longest-match-first: phrases are sorted by descending length before being
|
|
373
|
+
joined into a single alternation, so the regex engine prefers the longest
|
|
374
|
+
phrase at a given position. Case-insensitive, word-boundary anchored.
|
|
375
|
+
Returns (None, {}) when there is nothing to compile.
|
|
376
|
+
"""
|
|
377
|
+
if not pairs:
|
|
378
|
+
return None, {}
|
|
379
|
+
# Sort longest phrase first (tie-break casefold for determinism).
|
|
380
|
+
ordered = sorted(pairs, key=lambda p: (-len(p[0]), p[0].casefold()))
|
|
381
|
+
alts = []
|
|
382
|
+
group_placeholder = {}
|
|
383
|
+
for i, (phrase, placeholder) in enumerate(ordered):
|
|
384
|
+
gname = "d%d" % i
|
|
385
|
+
alts.append("(?P<%s>%s)" % (gname, re.escape(phrase)))
|
|
386
|
+
group_placeholder[gname] = placeholder
|
|
387
|
+
# Alphanumeric lookarounds (not \b) so entity names containing punctuation
|
|
388
|
+
# -- "AT&T", "Acme Corp." -- still match at their real boundaries (ported from
|
|
389
|
+
# the AWS corpus-pipeline registry cleaner). re.IGNORECASE for case.
|
|
390
|
+
pattern = r"(?<![A-Za-z0-9])(?:%s)(?![A-Za-z0-9])" % "|".join(alts)
|
|
391
|
+
try:
|
|
392
|
+
regex = re.compile(pattern, re.IGNORECASE)
|
|
393
|
+
except Exception:
|
|
394
|
+
return None, {}
|
|
395
|
+
return regex, group_placeholder
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def _apply_dictionary(text, compiled, matches=None):
|
|
399
|
+
"""Replace known-entity matches with their generic placeholder.
|
|
400
|
+
|
|
401
|
+
Returns (new_text, count). `compiled` is (regex, group->placeholder) from
|
|
402
|
+
_compile_dictionary. No-op (returns text, 0) when regex is None.
|
|
403
|
+
|
|
404
|
+
If `matches` is a list, one dict per redaction is appended (same shape as
|
|
405
|
+
_apply_pattern), with type "known-entity" and offsets into the OUTPUT text.
|
|
406
|
+
"""
|
|
407
|
+
regex, group_placeholder = compiled
|
|
408
|
+
if regex is None:
|
|
409
|
+
return text, 0
|
|
410
|
+
count = 0
|
|
411
|
+
out = []
|
|
412
|
+
out_len = 0
|
|
413
|
+
last = 0
|
|
414
|
+
for m in regex.finditer(text):
|
|
415
|
+
# Identify which named group actually matched.
|
|
416
|
+
placeholder = None
|
|
417
|
+
for gname, ph in group_placeholder.items():
|
|
418
|
+
if m.group(gname) is not None:
|
|
419
|
+
placeholder = ph
|
|
420
|
+
break
|
|
421
|
+
if placeholder is None:
|
|
422
|
+
continue
|
|
423
|
+
gap = text[last:m.start()]
|
|
424
|
+
out.append(gap)
|
|
425
|
+
out_len += len(gap)
|
|
426
|
+
if matches is not None:
|
|
427
|
+
matches.append({
|
|
428
|
+
"type": "known-entity",
|
|
429
|
+
"start": out_len,
|
|
430
|
+
"end": out_len + len(placeholder),
|
|
431
|
+
"replacement": placeholder,
|
|
432
|
+
"original": m.group(0),
|
|
433
|
+
})
|
|
434
|
+
out.append(placeholder)
|
|
435
|
+
out_len += len(placeholder)
|
|
436
|
+
last = m.end()
|
|
437
|
+
count += 1
|
|
438
|
+
out.append(text[last:])
|
|
439
|
+
return "".join(out), count
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
# ---------------------------------------------------------------------------
|
|
443
|
+
# Aho-Corasick fast path for the dictionary pass
|
|
444
|
+
# ---------------------------------------------------------------------------
|
|
445
|
+
# The regex oracle above compiles ~4,435 phrases into ONE alternation and lets
|
|
446
|
+
# Python `re` backtrack over it (O(text x alternatives)); on an 84KB file that is
|
|
447
|
+
# ~100s. This path builds an Aho-Corasick automaton once (linear build) and scans
|
|
448
|
+
# the text once (linear O(text)), reproducing the oracle's selection rule EXACTLY:
|
|
449
|
+
#
|
|
450
|
+
# - Case-insensitive via a length-preserving `str.lower()` fold. `re.IGNORECASE`
|
|
451
|
+
# matches char-by-char and (as proven exhaustively over all of Unicode) agrees
|
|
452
|
+
# with `.lower()` on every character pair EXCEPT a fixed set of 68 non-ASCII
|
|
453
|
+
# "exotic" codepoints (long-s, micro-sign, Greek/Cyrillic glyph variants, the
|
|
454
|
+
# U+0130 dotted-I whose lower() is 2 chars, a couple of ligatures). If ANY such
|
|
455
|
+
# char appears in the text or a phrase we do NOT fast-path: build_automaton
|
|
456
|
+
# returns None and scrub() uses the regex oracle. For all ASCII + Maori-macron
|
|
457
|
+
# text (the entire real corpus) `.lower()` is byte-for-byte equivalent to
|
|
458
|
+
# re.IGNORECASE and length-preserving, so offsets never drift.
|
|
459
|
+
# - Boundaries are the SAME ASCII-alnum lookarounds (not \b): a match needs a
|
|
460
|
+
# non-[A-Za-z0-9] char (or string end) immediately before its start and after
|
|
461
|
+
# its end, checked against the ORIGINAL text positions.
|
|
462
|
+
# - Leftmost, longest-at-start, non-overlapping: AC collects every phrase
|
|
463
|
+
# occurrence; a single left-to-right sweep then picks, at the earliest start
|
|
464
|
+
# with a boundary-valid phrase, the LONGEST such phrase (ties broken by the
|
|
465
|
+
# oracle's sort order: (-len, casefold), stable), emits it, and resumes at the
|
|
466
|
+
# match end -- identical to re.finditer over the sorted alternation.
|
|
467
|
+
# - The winning phrase's placeholder replaces the matched ORIGINAL-case span.
|
|
468
|
+
|
|
469
|
+
# Codepoints where str.lower() is not a length-preserving, re.IGNORECASE-faithful
|
|
470
|
+
# fold. Derived by exhaustively comparing str.lower() equivalence against
|
|
471
|
+
# re.IGNORECASE over every Unicode codepoint (see the perf/scrub-aho-corasick
|
|
472
|
+
# design notes). Presence of ANY of these in text or a phrase forces the oracle.
|
|
473
|
+
_AC_EXOTIC_CODEPOINTS = frozenset({
|
|
474
|
+
0xB5, 0x130, 0x17F, 0x345, 0x390, 0x392, 0x395, 0x398, 0x399, 0x39A,
|
|
475
|
+
0x39C, 0x3A0, 0x3A1, 0x3A3, 0x3A6, 0x3B0, 0x3B2, 0x3B5, 0x3B8, 0x3B9,
|
|
476
|
+
0x3BA, 0x3BC, 0x3C0, 0x3C1, 0x3C2, 0x3C3, 0x3C6, 0x3D0, 0x3D1, 0x3D5,
|
|
477
|
+
0x3D6, 0x3F0, 0x3F1, 0x3F4, 0x3F5, 0x412, 0x414, 0x41E, 0x421, 0x422,
|
|
478
|
+
0x42A, 0x432, 0x434, 0x43E, 0x441, 0x442, 0x44A, 0x462, 0x463, 0x1C80,
|
|
479
|
+
0x1C81, 0x1C82, 0x1C83, 0x1C84, 0x1C85, 0x1C86, 0x1C87, 0x1C88, 0x1E60,
|
|
480
|
+
0x1E61, 0x1E9B, 0x1FBE, 0x1FD3, 0x1FE3, 0xA64A, 0xA64B, 0xFB05, 0xFB06,
|
|
481
|
+
})
|
|
482
|
+
|
|
483
|
+
|
|
484
|
+
def _ac_has_exotic(s):
|
|
485
|
+
"""True if s contains any codepoint where str.lower() is not a faithful,
|
|
486
|
+
length-preserving stand-in for re.IGNORECASE (forces the regex oracle)."""
|
|
487
|
+
for ch in s:
|
|
488
|
+
if ord(ch) in _AC_EXOTIC_CODEPOINTS:
|
|
489
|
+
return True
|
|
490
|
+
return False
|
|
491
|
+
|
|
492
|
+
|
|
493
|
+
def build_automaton(pairs):
|
|
494
|
+
"""Build an Aho-Corasick automaton from (phrase, placeholder) pairs.
|
|
495
|
+
|
|
496
|
+
Returns an opaque dict consumed by _apply_dictionary_ac, or None when the
|
|
497
|
+
fast path is not usable (no pairs, or a phrase contains an exotic char --
|
|
498
|
+
see _AC_EXOTIC_CODEPOINTS -- in which case the caller falls back to the
|
|
499
|
+
regex oracle for guaranteed equivalence).
|
|
500
|
+
|
|
501
|
+
Phrases are keyed on their str.lower() fold. The same ordering the oracle
|
|
502
|
+
uses -- sorted by (-len(phrase), phrase.casefold()), stable -- is preserved
|
|
503
|
+
as each phrase's `rank`, so ties at a start position resolve identically.
|
|
504
|
+
"""
|
|
505
|
+
if not pairs:
|
|
506
|
+
return None
|
|
507
|
+
# Same order the regex oracle imposes on its alternation.
|
|
508
|
+
ordered = sorted(pairs, key=lambda p: (-len(p[0]), p[0].casefold()))
|
|
509
|
+
# Node 0 is root. goto: list of dicts (char -> node). out: list of terminal
|
|
510
|
+
# payloads per node, each (phrase_len, rank, placeholder). fail: list of ints.
|
|
511
|
+
goto = [{}]
|
|
512
|
+
out = [[]]
|
|
513
|
+
for rank, (phrase, placeholder) in enumerate(ordered):
|
|
514
|
+
if _ac_has_exotic(phrase):
|
|
515
|
+
# A phrase with an exotic char cannot be folded safely -> no fast path.
|
|
516
|
+
return None
|
|
517
|
+
folded = phrase.lower()
|
|
518
|
+
if not folded:
|
|
519
|
+
continue
|
|
520
|
+
node = 0
|
|
521
|
+
for ch in folded:
|
|
522
|
+
nxt = goto[node].get(ch)
|
|
523
|
+
if nxt is None:
|
|
524
|
+
nxt = len(goto)
|
|
525
|
+
goto.append({})
|
|
526
|
+
out.append([])
|
|
527
|
+
goto[node][ch] = nxt
|
|
528
|
+
node = nxt
|
|
529
|
+
out[node].append((len(folded), rank, placeholder))
|
|
530
|
+
# Build failure links (BFS). fail[root]=0; fail of depth-1 nodes = root.
|
|
531
|
+
fail = [0] * len(goto)
|
|
532
|
+
queue = []
|
|
533
|
+
for ch, nxt in goto[0].items():
|
|
534
|
+
fail[nxt] = 0
|
|
535
|
+
queue.append(nxt)
|
|
536
|
+
head = 0
|
|
537
|
+
while head < len(queue):
|
|
538
|
+
node = queue[head]
|
|
539
|
+
head += 1
|
|
540
|
+
for ch, nxt in goto[node].items():
|
|
541
|
+
queue.append(nxt)
|
|
542
|
+
f = fail[node]
|
|
543
|
+
while f and ch not in goto[f]:
|
|
544
|
+
f = fail[f]
|
|
545
|
+
fail[nxt] = goto[f].get(ch, 0) if f or ch in goto[0] else 0
|
|
546
|
+
# Merge terminal outputs along the failure chain (report ALL phrases
|
|
547
|
+
# that end here, including shorter suffix phrases).
|
|
548
|
+
if out[fail[nxt]]:
|
|
549
|
+
out[nxt] = out[nxt] + out[fail[nxt]]
|
|
550
|
+
return {"goto": goto, "out": out, "fail": fail}
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
def _ac_alnum(ch):
|
|
554
|
+
"""ASCII alnum test matching the oracle's [A-Za-z0-9] character class."""
|
|
555
|
+
return ("0" <= ch <= "9") or ("A" <= ch <= "Z") or ("a" <= ch <= "z")
|
|
556
|
+
|
|
557
|
+
|
|
558
|
+
def _apply_dictionary_ac(text, automaton, matches=None):
|
|
559
|
+
"""Aho-Corasick equivalent of _apply_dictionary. Returns (new_text, count).
|
|
560
|
+
|
|
561
|
+
Reproduces the regex oracle byte-for-byte for text free of exotic chars.
|
|
562
|
+
Raises on any unexpected condition so scrub() can fall back to the oracle;
|
|
563
|
+
callers MUST wrap this and fall back on exception (fail-safe, never open).
|
|
564
|
+
|
|
565
|
+
`matches`, if a list, receives one dict per redaction (same shape/semantics
|
|
566
|
+
as _apply_dictionary): type "known-entity", offsets into the OUTPUT text,
|
|
567
|
+
"original" the matched original-case span, "replacement" the placeholder.
|
|
568
|
+
"""
|
|
569
|
+
if automaton is None:
|
|
570
|
+
return text, 0
|
|
571
|
+
if _ac_has_exotic(text):
|
|
572
|
+
# Exotic char in the text: not safe to fold with .lower() -> force oracle.
|
|
573
|
+
raise ValueError("exotic codepoint in text; use regex oracle")
|
|
574
|
+
|
|
575
|
+
goto = automaton["goto"]
|
|
576
|
+
out = automaton["out"]
|
|
577
|
+
fail = automaton["fail"]
|
|
578
|
+
folded = text.lower()
|
|
579
|
+
# .lower() is length-preserving for all non-exotic input (guarded above);
|
|
580
|
+
# assert alignment so any surprise drift trips the fallback instead of leaking.
|
|
581
|
+
if len(folded) != len(text):
|
|
582
|
+
raise ValueError("fold changed length; use regex oracle")
|
|
583
|
+
|
|
584
|
+
n = len(text)
|
|
585
|
+
# Pass 1: AC sweep. For every end position, record the BEST phrase ending
|
|
586
|
+
# there whose start boundary is valid, keyed for start-position selection.
|
|
587
|
+
# best_at_start[start] = (phrase_len, rank, placeholder, end) for the phrase
|
|
588
|
+
# the oracle would choose if a match begins at `start` (longest, then lowest
|
|
589
|
+
# rank). We only keep boundary-valid candidates.
|
|
590
|
+
best_at_start = {}
|
|
591
|
+
node = 0
|
|
592
|
+
for i in range(n):
|
|
593
|
+
ch = folded[i]
|
|
594
|
+
while node and ch not in goto[node]:
|
|
595
|
+
node = fail[node]
|
|
596
|
+
node = goto[node].get(ch, 0)
|
|
597
|
+
if not out[node]:
|
|
598
|
+
continue
|
|
599
|
+
end = i + 1 # exclusive
|
|
600
|
+
# Trailing boundary: char after end must not be ASCII alnum (or be EOS).
|
|
601
|
+
if end < n and _ac_alnum(text[end]):
|
|
602
|
+
trailing_ok = False
|
|
603
|
+
else:
|
|
604
|
+
trailing_ok = True
|
|
605
|
+
if not trailing_ok:
|
|
606
|
+
continue
|
|
607
|
+
for (plen, rank, placeholder) in out[node]:
|
|
608
|
+
start = end - plen
|
|
609
|
+
# Leading boundary: char before start must not be ASCII alnum (or BOS).
|
|
610
|
+
if start > 0 and _ac_alnum(text[start - 1]):
|
|
611
|
+
continue
|
|
612
|
+
cur = best_at_start.get(start)
|
|
613
|
+
# Oracle preference at a start: longer wins, then lower rank.
|
|
614
|
+
if cur is None or (plen, -rank) > (cur[0], -cur[1]):
|
|
615
|
+
best_at_start[start] = (plen, rank, placeholder, end)
|
|
616
|
+
|
|
617
|
+
if not best_at_start:
|
|
618
|
+
return text, 0
|
|
619
|
+
|
|
620
|
+
# Pass 2: leftmost, non-overlapping selection -- resume at match end.
|
|
621
|
+
count = 0
|
|
622
|
+
out_parts = []
|
|
623
|
+
out_len = 0
|
|
624
|
+
last = 0
|
|
625
|
+
pos = 0
|
|
626
|
+
starts = sorted(best_at_start)
|
|
627
|
+
si = 0
|
|
628
|
+
ns = len(starts)
|
|
629
|
+
while si < ns:
|
|
630
|
+
start = starts[si]
|
|
631
|
+
if start < pos:
|
|
632
|
+
si += 1
|
|
633
|
+
continue
|
|
634
|
+
plen, rank, placeholder, end = best_at_start[start]
|
|
635
|
+
gap = text[last:start]
|
|
636
|
+
out_parts.append(gap)
|
|
637
|
+
out_len += len(gap)
|
|
638
|
+
if matches is not None:
|
|
639
|
+
matches.append({
|
|
640
|
+
"type": "known-entity",
|
|
641
|
+
"start": out_len,
|
|
642
|
+
"end": out_len + len(placeholder),
|
|
643
|
+
"replacement": placeholder,
|
|
644
|
+
"original": text[start:end],
|
|
645
|
+
})
|
|
646
|
+
out_parts.append(placeholder)
|
|
647
|
+
out_len += len(placeholder)
|
|
648
|
+
last = end
|
|
649
|
+
pos = end
|
|
650
|
+
count += 1
|
|
651
|
+
si += 1
|
|
652
|
+
out_parts.append(text[last:])
|
|
653
|
+
return "".join(out_parts), count
|
|
654
|
+
|
|
655
|
+
|
|
656
|
+
def _run_dictionary_pass(text, matches=None):
|
|
657
|
+
"""Run the layer-1 dictionary pass, preferring the Aho-Corasick fast path
|
|
658
|
+
and FALLING BACK to the regex oracle on any AC exception (fail-safe: the
|
|
659
|
+
oracle is the known-correct path, so a fallback can only be correct, never
|
|
660
|
+
a leak). Returns (new_text, count). Never raises.
|
|
661
|
+
|
|
662
|
+
Loads + builds the automaton (and the oracle) on each call, mirroring the
|
|
663
|
+
prior behaviour where _compile_dictionary(load_dictionary()) was rebuilt per
|
|
664
|
+
scrub. When AC is unavailable (no dict, or an exotic phrase) build_automaton
|
|
665
|
+
returns None and we use the oracle directly.
|
|
666
|
+
"""
|
|
667
|
+
pairs = load_dictionary()
|
|
668
|
+
if not pairs:
|
|
669
|
+
return text, 0
|
|
670
|
+
automaton = None
|
|
671
|
+
try:
|
|
672
|
+
automaton = build_automaton(pairs)
|
|
673
|
+
except Exception:
|
|
674
|
+
automaton = None
|
|
675
|
+
if automaton is not None:
|
|
676
|
+
# AC fast path; any surprise -> oracle. If `matches` was partially
|
|
677
|
+
# populated before an exception, drop those entries so the oracle
|
|
678
|
+
# reruns cleanly (the oracle appends from a clean slate).
|
|
679
|
+
marker = len(matches) if matches is not None else 0
|
|
680
|
+
try:
|
|
681
|
+
return _apply_dictionary_ac(text, automaton, matches)
|
|
682
|
+
except Exception:
|
|
683
|
+
if matches is not None:
|
|
684
|
+
del matches[marker:]
|
|
685
|
+
# Regex oracle (fallback, or the only path when AC is unavailable).
|
|
686
|
+
compiled = _compile_dictionary(pairs)
|
|
687
|
+
return _apply_dictionary(text, compiled, matches)
|
|
688
|
+
|
|
689
|
+
|
|
690
|
+
def scrub(text):
|
|
691
|
+
"""Redact PII / secrets / local paths from text.
|
|
692
|
+
|
|
693
|
+
Returns (scrubbed_text, findings) where findings is a list of
|
|
694
|
+
{"type": str, "count": int}, one per pattern that matched at least once.
|
|
695
|
+
|
|
696
|
+
NEVER raises. On any internal error returns (original_text, [{"type":
|
|
697
|
+
"error", "count": 1}]) so a caller can detect failure and refuse to egress.
|
|
698
|
+
"""
|
|
699
|
+
try:
|
|
700
|
+
if text is None:
|
|
701
|
+
return "", []
|
|
702
|
+
if not isinstance(text, str):
|
|
703
|
+
text = str(text)
|
|
704
|
+
|
|
705
|
+
findings = []
|
|
706
|
+
scrubbed = text
|
|
707
|
+
|
|
708
|
+
# Layer-1 dictionary pass FIRST: fold known private entities to their
|
|
709
|
+
# generic placeholder before the regex heuristics run. Graceful no-op if
|
|
710
|
+
# no dictionary is present. Never raises (loader/compile swallow errors).
|
|
711
|
+
try:
|
|
712
|
+
scrubbed, dict_count = _run_dictionary_pass(scrubbed)
|
|
713
|
+
if dict_count:
|
|
714
|
+
findings.append({"type": "known-entity", "count": dict_count})
|
|
715
|
+
except Exception:
|
|
716
|
+
# Dictionary pass must never break scrubbing; regex passes still run.
|
|
717
|
+
pass
|
|
718
|
+
|
|
719
|
+
for type_name, regex, respect_fences in _PATTERNS:
|
|
720
|
+
scrubbed, count = _apply_pattern(
|
|
721
|
+
scrubbed, type_name, regex, respect_fences
|
|
722
|
+
)
|
|
723
|
+
if count:
|
|
724
|
+
findings.append({"type": type_name, "count": count})
|
|
725
|
+
return scrubbed, findings
|
|
726
|
+
except Exception as exc: # never raise out of scrub()
|
|
727
|
+
try:
|
|
728
|
+
sys.stderr.write("scrub: internal error: %r\n" % (exc,))
|
|
729
|
+
except Exception:
|
|
730
|
+
pass
|
|
731
|
+
return text if isinstance(text, str) else "", [
|
|
732
|
+
{"type": "error", "count": 1}
|
|
733
|
+
]
|
|
734
|
+
|
|
735
|
+
|
|
736
|
+
def scrub_detailed(text):
|
|
737
|
+
"""Like scrub(), but also return per-match detail for PREVIEW/reporting.
|
|
738
|
+
|
|
739
|
+
Returns (scrubbed_text, findings, matches) where:
|
|
740
|
+
- scrubbed_text / findings are IDENTICAL to what scrub(text) returns
|
|
741
|
+
(same values, same shapes) -- this function does not change scrub()'s
|
|
742
|
+
public contract, it augments it.
|
|
743
|
+
- matches is a list of dicts, one per individual redaction, in the order
|
|
744
|
+
they were applied:
|
|
745
|
+
{"type": str, "start": int, "end": int,
|
|
746
|
+
"replacement": str, "original": str}
|
|
747
|
+
"original" is the RAW matched substring; it is provided ONLY so a caller
|
|
748
|
+
can render a MASKED sample and MUST NOT be egressed or logged verbatim.
|
|
749
|
+
"start"/"end" are offsets into the scrubbed text AT THE TIME that pattern
|
|
750
|
+
pass ran (passes run sequentially and later passes rewrite the string, so
|
|
751
|
+
treat them as an ordering/locality hint, not a guaranteed final offset).
|
|
752
|
+
|
|
753
|
+
NEVER raises: on internal error returns (original_text, [error-finding], []).
|
|
754
|
+
"""
|
|
755
|
+
try:
|
|
756
|
+
if text is None:
|
|
757
|
+
return "", [], []
|
|
758
|
+
if not isinstance(text, str):
|
|
759
|
+
text = str(text)
|
|
760
|
+
|
|
761
|
+
findings = []
|
|
762
|
+
matches = []
|
|
763
|
+
scrubbed = text
|
|
764
|
+
|
|
765
|
+
try:
|
|
766
|
+
scrubbed, dict_count = _run_dictionary_pass(scrubbed, matches)
|
|
767
|
+
if dict_count:
|
|
768
|
+
findings.append({"type": "known-entity", "count": dict_count})
|
|
769
|
+
except Exception:
|
|
770
|
+
pass
|
|
771
|
+
|
|
772
|
+
for type_name, regex, respect_fences in _PATTERNS:
|
|
773
|
+
scrubbed, count = _apply_pattern(
|
|
774
|
+
scrubbed, type_name, regex, respect_fences, matches
|
|
775
|
+
)
|
|
776
|
+
if count:
|
|
777
|
+
findings.append({"type": type_name, "count": count})
|
|
778
|
+
return scrubbed, findings, matches
|
|
779
|
+
except Exception as exc: # never raise
|
|
780
|
+
try:
|
|
781
|
+
sys.stderr.write("scrub_detailed: internal error: %r\n" % (exc,))
|
|
782
|
+
except Exception:
|
|
783
|
+
pass
|
|
784
|
+
return (text if isinstance(text, str) else ""), [
|
|
785
|
+
{"type": "error", "count": 1}
|
|
786
|
+
], []
|
|
787
|
+
|
|
788
|
+
|
|
789
|
+
# ---------------------------------------------------------------------------
|
|
790
|
+
# Egress block-gate (B / ADR-011)
|
|
791
|
+
# ---------------------------------------------------------------------------
|
|
792
|
+
# High-confidence secret types: a match here is almost certainly a REAL credential,
|
|
793
|
+
# not a heuristic false-positive. find_secrets() filters to these so it can be used as
|
|
794
|
+
# a BLOCK gate before egress (publish to Confluence, write to a bridge file), where
|
|
795
|
+
# silently redacting would corrupt legitimate content (governance docs contain example
|
|
796
|
+
# tokens, IPs, paths). The noisy heuristics -- email, *_path, ipv4, network_cidr, phone,
|
|
797
|
+
# high_entropy_*, known-entity -- are deliberately EXCLUDED here; they belong to
|
|
798
|
+
# redaction (scrub()), used for LLM egress where over-redaction is acceptable.
|
|
799
|
+
HIGH_CONFIDENCE_TYPES = frozenset({
|
|
800
|
+
"basic_auth_url", "aws_access_key", "gitlab_pat", "github_pat", "github_fine_pat",
|
|
801
|
+
"google_api_key", "google_aq_key", "slack_token", "jwt", "atlassian_token",
|
|
802
|
+
"anthropic_key", "openai_key", "bearer_token", "secret_assignment",
|
|
803
|
+
})
|
|
804
|
+
|
|
805
|
+
|
|
806
|
+
def find_secrets(text):
|
|
807
|
+
"""Return high-confidence secret findings in `text` (list of {"type","count"}).
|
|
808
|
+
|
|
809
|
+
Non-empty => `text` contains something that must NOT be egressed. Uses scrub()'s
|
|
810
|
+
detector but filters to HIGH_CONFIDENCE_TYPES so noisy heuristics do not false-block.
|
|
811
|
+
Fail-closed: if scrub() itself errored (could not verify), that error finding is
|
|
812
|
+
included, so callers treat an un-verifiable input as unsafe. Never raises.
|
|
813
|
+
"""
|
|
814
|
+
try:
|
|
815
|
+
_, findings = scrub(text)
|
|
816
|
+
except Exception:
|
|
817
|
+
return [{"type": "error", "count": 1}]
|
|
818
|
+
return [f for f in findings
|
|
819
|
+
if f["type"] in HIGH_CONFIDENCE_TYPES or f["type"] == "error"]
|
|
820
|
+
|
|
821
|
+
|
|
822
|
+
def scrub_file(path):
|
|
823
|
+
"""Read a file (UTF-8, errors replaced) and scrub its contents.
|
|
824
|
+
|
|
825
|
+
Returns (scrubbed_text, findings). Like scrub(), never raises: a read error
|
|
826
|
+
is reported as an {"type": "error", "count": 1} finding.
|
|
827
|
+
"""
|
|
828
|
+
try:
|
|
829
|
+
with open(path, "r", encoding="utf-8", errors="replace") as fh:
|
|
830
|
+
raw = fh.read()
|
|
831
|
+
except Exception as exc:
|
|
832
|
+
try:
|
|
833
|
+
sys.stderr.write("scrub_file: cannot read %s: %r\n" % (path, exc))
|
|
834
|
+
except Exception:
|
|
835
|
+
pass
|
|
836
|
+
return "", [{"type": "error", "count": 1}]
|
|
837
|
+
return scrub(raw)
|
|
838
|
+
|
|
839
|
+
|
|
840
|
+
# ---------------------------------------------------------------------------
|
|
841
|
+
# Self-test
|
|
842
|
+
# ---------------------------------------------------------------------------
|
|
843
|
+
# A built-in sample containing one of each pattern. Each entry asserts that the
|
|
844
|
+
# raw fragment no longer appears verbatim in the scrubbed output and that the
|
|
845
|
+
# expected finding type fired. Demonstrates correctness without any fixtures.
|
|
846
|
+
|
|
847
|
+
_SELFTEST_CASES = [
|
|
848
|
+
("email", "contact me at oscar@ecoportal.co.nz today"),
|
|
849
|
+
("windows_path", r"see C:\Users\rella\.claude\paths.json for config"),
|
|
850
|
+
("windows_path_fwd", "open C:/claude/Projects/ep-ai-standards/TODO.md"),
|
|
851
|
+
("posix_home_path", "log lives at /home/oscar/.config/app/secret.cfg"),
|
|
852
|
+
("posix_users_path", "build dir /Users/oscar/work/build/out"),
|
|
853
|
+
("mnt_path", "data on /mnt/data/private/dump.sql"),
|
|
854
|
+
("tilde_home", "config at ~/secrets/keys.txt loaded"),
|
|
855
|
+
("aws_access_key", "key AKIAIOSFODNN7EXAMPLE used"),
|
|
856
|
+
("gitlab_pat", "token glpat-ABCDEFGHIJKLMNOPQRSTUV set"),
|
|
857
|
+
("github_pat", "ghp_ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789 here"),
|
|
858
|
+
("github_fine_pat",
|
|
859
|
+
"github_pat_11ABCDEFG0abcdefghijklmnopqrstuvwxyz12345 here"),
|
|
860
|
+
("google_api_key",
|
|
861
|
+
"AIzaSyA1234567890abcdefghijklmnopqrstuv in url"),
|
|
862
|
+
("google_aq_key",
|
|
863
|
+
"gemini key AQ.Ab8RN6JjFAKEfakefakekey1234567890xyzABCdefg used"),
|
|
864
|
+
("prefixed_api_key_assignment", "GEMINI_API_KEY=AbCdEfFake123456value"),
|
|
865
|
+
("suffixed_token_assignment", "TOGGL_TOKEN_RW=zzzzfakevalue1234"),
|
|
866
|
+
("slack_token", "xoxb-1234567890-abcdefghijklmnop posted"),
|
|
867
|
+
("jwt",
|
|
868
|
+
"eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxMjM0NTY3ODkwIn0.dozjgNryP4J3jVmNHl0w "
|
|
869
|
+
"is the token"),
|
|
870
|
+
("bearer_token", "Authorization: Bearer abcdef1234567890XYZ"),
|
|
871
|
+
("secret_assignment", 'api_key="s3cr3tValue12345"'),
|
|
872
|
+
("password_assignment", "password = hunter2hunter2"),
|
|
873
|
+
("ipv4", "host at 192.168.10.254 reachable"),
|
|
874
|
+
("network_cidr", "subnet 10.0.0.0/8 allocated"),
|
|
875
|
+
("phone_e164", "call +64 9 123 4567 tomorrow"),
|
|
876
|
+
("phone_nz_au", "mobile 021 123 4567 for the site"),
|
|
877
|
+
("atlassian_token", "atlassian ATATT3xFfGF0abcdefghijklmnopqrstuvwxyz012345 set"),
|
|
878
|
+
("anthropic_key", "claude sk-ant-api03-abcdefghijklmnopqrstuvwxyz0123456789ABCD used"),
|
|
879
|
+
("openai_key", "openai sk-abcdefghijklmnopqrstuvwxyz0123456789ABCDEFGHIJKL here"),
|
|
880
|
+
("aws_arn", "role arn:aws:iam::123456789012:role/MyExampleRole attached"),
|
|
881
|
+
("basic_auth_url", "clone https://guser:s3cretpw@git.example.com/repo.git now"),
|
|
882
|
+
("high_entropy_hex",
|
|
883
|
+
"digest 0123456789abcdef0123456789abcdef01234567 outside fence"),
|
|
884
|
+
("high_entropy_base64",
|
|
885
|
+
"blob YWJjZGVmZ2hpamtsbW5vcHFyc3R1dnd4eXoxMjM0NTY3ODkw end"),
|
|
886
|
+
]
|
|
887
|
+
|
|
888
|
+
# Raw secret fragments that MUST disappear from the scrubbed output.
|
|
889
|
+
_SELFTEST_SECRETS = [
|
|
890
|
+
"oscar@ecoportal.co.nz",
|
|
891
|
+
r"C:\Users\rella",
|
|
892
|
+
"C:/claude/Projects",
|
|
893
|
+
"/home/oscar/",
|
|
894
|
+
"/Users/oscar/",
|
|
895
|
+
"/mnt/data/private",
|
|
896
|
+
"~/secrets/keys.txt",
|
|
897
|
+
"AKIAIOSFODNN7EXAMPLE",
|
|
898
|
+
"glpat-ABCDEFGHIJKLMNOPQRSTUV",
|
|
899
|
+
"ghp_ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789",
|
|
900
|
+
"github_pat_11ABCDEFG0abcdefghijklmnopqrstuvwxyz12345",
|
|
901
|
+
"AIzaSyA1234567890abcdefghijklmnopqrstuv",
|
|
902
|
+
"AQ.Ab8RN6JjFAKEfakefakekey1234567890xyzABCdefg",
|
|
903
|
+
"AbCdEfFake123456value",
|
|
904
|
+
"zzzzfakevalue1234",
|
|
905
|
+
"xoxb-1234567890-abcdefghijklmnop",
|
|
906
|
+
"eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxMjM0NTY3ODkwIn0.dozjgNryP4J3jVmNHl0w",
|
|
907
|
+
"hunter2hunter2",
|
|
908
|
+
"s3cr3tValue12345",
|
|
909
|
+
"192.168.10.254",
|
|
910
|
+
"10.0.0.0/8",
|
|
911
|
+
"+64 9 123 4567",
|
|
912
|
+
"021 123 4567",
|
|
913
|
+
"ATATT3xFfGF0abcdefghijklmnopqrstuvwxyz012345",
|
|
914
|
+
"sk-ant-api03-abcdefghijklmnopqrstuvwxyz0123456789ABCD",
|
|
915
|
+
"sk-abcdefghijklmnopqrstuvwxyz0123456789ABCDEFGHIJKL",
|
|
916
|
+
"arn:aws:iam::123456789012:role/MyExampleRole",
|
|
917
|
+
"s3cretpw",
|
|
918
|
+
"0123456789abcdef0123456789abcdef01234567",
|
|
919
|
+
"YWJjZGVmZ2hpamtsbW5vcHFyc3R1dnd4eXoxMjM0NTY3ODkw",
|
|
920
|
+
]
|
|
921
|
+
|
|
922
|
+
|
|
923
|
+
def _selftest():
|
|
924
|
+
"""Run built-in assertions. Returns 0 on success, 1 on any failure.
|
|
925
|
+
|
|
926
|
+
Hermetic: the machine-local entity dictionary is disabled so this exercises
|
|
927
|
+
the regex layer in isolation (the dictionary pass has its own tests).
|
|
928
|
+
"""
|
|
929
|
+
os.environ["PRIVATE_DATA_DICT_DIR"] = os.path.join(
|
|
930
|
+
os.path.dirname(os.path.abspath(__file__)), "__no_such_dict__")
|
|
931
|
+
sample = "\n".join(desc for _, desc in _SELFTEST_CASES)
|
|
932
|
+
scrubbed, findings = scrub(sample)
|
|
933
|
+
|
|
934
|
+
ok = True
|
|
935
|
+
|
|
936
|
+
# 1. No raw secret fragment survives.
|
|
937
|
+
for secret in _SELFTEST_SECRETS:
|
|
938
|
+
if secret in scrubbed:
|
|
939
|
+
sys.stderr.write("FAIL: secret survived scrub: %r\n" % secret)
|
|
940
|
+
ok = False
|
|
941
|
+
|
|
942
|
+
# 2. The expected finding types all fired.
|
|
943
|
+
fired = {f["type"] for f in findings}
|
|
944
|
+
expected_types = {
|
|
945
|
+
"email", "windows_path", "posix_home_path", "tilde_home_path",
|
|
946
|
+
"aws_access_key", "gitlab_pat", "github_pat", "github_fine_pat",
|
|
947
|
+
"google_api_key", "google_aq_key", "slack_token", "jwt", "bearer_token",
|
|
948
|
+
"secret_assignment", "ipv4", "high_entropy_hex", "high_entropy_base64",
|
|
949
|
+
"atlassian_token", "anthropic_key", "openai_key", "aws_arn",
|
|
950
|
+
"basic_auth_url", "phone", "network_cidr",
|
|
951
|
+
}
|
|
952
|
+
missing = expected_types - fired
|
|
953
|
+
if missing:
|
|
954
|
+
sys.stderr.write("FAIL: finding types never fired: %s\n"
|
|
955
|
+
% ", ".join(sorted(missing)))
|
|
956
|
+
ok = False
|
|
957
|
+
|
|
958
|
+
# 3. Code-fence sample: a hash INSIDE a fence is preserved; a real secret
|
|
959
|
+
# (email) inside a fence is still redacted.
|
|
960
|
+
fenced = (
|
|
961
|
+
"Here is a legitimate hash in a code block:\n"
|
|
962
|
+
"```\n"
|
|
963
|
+
"deadbeefdeadbeefdeadbeefdeadbeefdeadbeef\n"
|
|
964
|
+
"user@example.com\n"
|
|
965
|
+
"```\n"
|
|
966
|
+
"And outside: deadbeefdeadbeefdeadbeefdeadbeefdeadbeef\n"
|
|
967
|
+
)
|
|
968
|
+
fscrub, ffind = scrub(fenced)
|
|
969
|
+
# The in-fence hash should remain (not high-entropy-redacted).
|
|
970
|
+
if "deadbeefdeadbeefdeadbeefdeadbeefdeadbeef" not in fscrub:
|
|
971
|
+
sys.stderr.write("FAIL: in-fence hash was redacted (should be kept)\n")
|
|
972
|
+
ok = False
|
|
973
|
+
# The out-of-fence hash should be redacted.
|
|
974
|
+
if fscrub.count("deadbeefdeadbeefdeadbeefdeadbeefdeadbeef") != 1:
|
|
975
|
+
sys.stderr.write(
|
|
976
|
+
"FAIL: out-of-fence hash not redacted "
|
|
977
|
+
"(in-fence copy should be the only survivor)\n"
|
|
978
|
+
)
|
|
979
|
+
ok = False
|
|
980
|
+
# The in-fence email should still be redacted (targeted patterns ignore fences).
|
|
981
|
+
if "user@example.com" in fscrub:
|
|
982
|
+
sys.stderr.write("FAIL: in-fence email survived (should be redacted)\n")
|
|
983
|
+
ok = False
|
|
984
|
+
|
|
985
|
+
# 4. scrub() never raises and handles odd input.
|
|
986
|
+
for bad in (None, 12345, b"bytes-not-str"):
|
|
987
|
+
try:
|
|
988
|
+
out, _ = scrub(bad)
|
|
989
|
+
if not isinstance(out, str):
|
|
990
|
+
sys.stderr.write("FAIL: scrub(%r) did not return str\n" % (bad,))
|
|
991
|
+
ok = False
|
|
992
|
+
except Exception as exc:
|
|
993
|
+
sys.stderr.write("FAIL: scrub(%r) raised %r\n" % (bad, exc))
|
|
994
|
+
ok = False
|
|
995
|
+
|
|
996
|
+
if ok:
|
|
997
|
+
sys.stderr.write("selftest: OK (%d findings on sample)\n" % len(findings))
|
|
998
|
+
return 0
|
|
999
|
+
sys.stderr.write("selftest: FAILED\n")
|
|
1000
|
+
return 1
|
|
1001
|
+
|
|
1002
|
+
|
|
1003
|
+
def _summarize(findings, stream):
|
|
1004
|
+
if not findings:
|
|
1005
|
+
stream.write("scrub: no findings\n")
|
|
1006
|
+
return
|
|
1007
|
+
stream.write("scrub: findings:\n")
|
|
1008
|
+
total = 0
|
|
1009
|
+
for f in findings:
|
|
1010
|
+
stream.write(" %-22s %d\n" % (f["type"], f["count"]))
|
|
1011
|
+
total += f["count"]
|
|
1012
|
+
stream.write("scrub: %d redaction(s) across %d pattern type(s)\n"
|
|
1013
|
+
% (total, len(findings)))
|
|
1014
|
+
|
|
1015
|
+
|
|
1016
|
+
def _mask_sample(original):
|
|
1017
|
+
"""Return a masked, safe-to-print rendering of a raw matched secret.
|
|
1018
|
+
|
|
1019
|
+
Reveals at most the first 2 and last 2 characters and masks the middle with
|
|
1020
|
+
a run of '*', so the preview shows the SHAPE of what was redacted without
|
|
1021
|
+
disclosing the value. Short values (<= 4 chars) are fully masked. Newlines
|
|
1022
|
+
and tabs are collapsed so the sample stays on one line.
|
|
1023
|
+
"""
|
|
1024
|
+
s = "".join(" " if c in "\r\n\t" else c for c in (original or ""))
|
|
1025
|
+
n = len(s)
|
|
1026
|
+
if n == 0:
|
|
1027
|
+
return ""
|
|
1028
|
+
if n <= 4:
|
|
1029
|
+
return "*" * n
|
|
1030
|
+
head = s[:2]
|
|
1031
|
+
tail = s[-2:]
|
|
1032
|
+
return "%s%s%s" % (head, "*" * (n - 4), tail)
|
|
1033
|
+
|
|
1034
|
+
|
|
1035
|
+
def _render_preview(scrubbed, findings, matches, stream):
|
|
1036
|
+
"""Write a human-readable, NON-leaking redaction report to `stream`.
|
|
1037
|
+
|
|
1038
|
+
Sections:
|
|
1039
|
+
1. The scrubbed text (already safe -- secrets replaced by placeholders).
|
|
1040
|
+
2. A findings table: one row per finding TYPE -> count.
|
|
1041
|
+
3. A per-redaction list with a MASKED sample for each (never the raw value).
|
|
1042
|
+
"""
|
|
1043
|
+
line = "=" * 60
|
|
1044
|
+
stream.write(line + "\n")
|
|
1045
|
+
stream.write("SCRUB PREVIEW -- REDACTION REPORT (nothing was sent)\n")
|
|
1046
|
+
stream.write(line + "\n\n")
|
|
1047
|
+
|
|
1048
|
+
stream.write("--- Scrubbed text (safe to read) ---\n")
|
|
1049
|
+
stream.write(scrubbed)
|
|
1050
|
+
if scrubbed and not scrubbed.endswith("\n"):
|
|
1051
|
+
stream.write("\n")
|
|
1052
|
+
stream.write("\n")
|
|
1053
|
+
|
|
1054
|
+
stream.write("--- Findings (type -> count) ---\n")
|
|
1055
|
+
if not findings:
|
|
1056
|
+
stream.write(" (none -- nothing matched)\n")
|
|
1057
|
+
else:
|
|
1058
|
+
total = 0
|
|
1059
|
+
for f in findings:
|
|
1060
|
+
stream.write(" %-22s %d\n" % (f["type"], f["count"]))
|
|
1061
|
+
total += f["count"]
|
|
1062
|
+
stream.write(" %-22s %d redaction(s) across %d type(s)\n"
|
|
1063
|
+
% ("TOTAL", total, len(findings)))
|
|
1064
|
+
stream.write("\n")
|
|
1065
|
+
|
|
1066
|
+
stream.write("--- Redactions (masked samples -- secrets NOT shown) ---\n")
|
|
1067
|
+
if not matches:
|
|
1068
|
+
stream.write(" (none)\n")
|
|
1069
|
+
else:
|
|
1070
|
+
for i, m in enumerate(matches, 1):
|
|
1071
|
+
stream.write(" %3d. %-22s %s (masked: %s)\n"
|
|
1072
|
+
% (i, m["type"], m["replacement"],
|
|
1073
|
+
_mask_sample(m.get("original", ""))))
|
|
1074
|
+
stream.write("\n")
|
|
1075
|
+
stream.write(line + "\n")
|
|
1076
|
+
|
|
1077
|
+
|
|
1078
|
+
def _preview(path):
|
|
1079
|
+
"""Render a non-destructive redaction report for `path` to stdout. Exit code
|
|
1080
|
+
matches the scrubbed-text CLI: 2 if scrubbing errored, else 0."""
|
|
1081
|
+
scrubbed, findings, matches = scrub_detailed_file(path)
|
|
1082
|
+
_render_preview(scrubbed, findings, matches, sys.stdout)
|
|
1083
|
+
if any(f["type"] == "error" for f in findings):
|
|
1084
|
+
return 2
|
|
1085
|
+
return 0
|
|
1086
|
+
|
|
1087
|
+
|
|
1088
|
+
def scrub_detailed_file(path):
|
|
1089
|
+
"""Read a file (UTF-8, errors replaced) and scrub_detailed() its contents.
|
|
1090
|
+
|
|
1091
|
+
Returns (scrubbed_text, findings, matches). Never raises: a read error is
|
|
1092
|
+
reported as an {"type": "error", "count": 1} finding with no matches.
|
|
1093
|
+
"""
|
|
1094
|
+
try:
|
|
1095
|
+
with open(path, "r", encoding="utf-8", errors="replace") as fh:
|
|
1096
|
+
raw = fh.read()
|
|
1097
|
+
except Exception as exc:
|
|
1098
|
+
try:
|
|
1099
|
+
sys.stderr.write("scrub_detailed_file: cannot read %s: %r\n"
|
|
1100
|
+
% (path, exc))
|
|
1101
|
+
except Exception:
|
|
1102
|
+
pass
|
|
1103
|
+
return "", [{"type": "error", "count": 1}], []
|
|
1104
|
+
return scrub_detailed(raw)
|
|
1105
|
+
|
|
1106
|
+
|
|
1107
|
+
def _main(argv):
|
|
1108
|
+
# Windows stdout/stderr default to a locale codepage (cp1252) that cannot encode
|
|
1109
|
+
# non-ASCII content (e.g. a stray glyph in a doc), which crashes the scrubber. Because
|
|
1110
|
+
# the Ruby egress client (gemini_ask.rb) shells out to this and fail-closes on ANY
|
|
1111
|
+
# scrubber error, that crash would silently BLOCK all Gemini egress of non-ASCII
|
|
1112
|
+
# content. Force UTF-8 so the scrubbed text round-trips regardless of platform locale.
|
|
1113
|
+
for _stream in (sys.stdout, sys.stderr):
|
|
1114
|
+
try:
|
|
1115
|
+
_stream.reconfigure(encoding="utf-8")
|
|
1116
|
+
except Exception:
|
|
1117
|
+
pass
|
|
1118
|
+
args = argv[1:]
|
|
1119
|
+
if not args or args[0] in ("-h", "--help"):
|
|
1120
|
+
sys.stderr.write(
|
|
1121
|
+
"usage: python scripts/lib/scrub.py <file>\n"
|
|
1122
|
+
" python scripts/lib/scrub.py --preview <file>\n"
|
|
1123
|
+
" python scripts/lib/scrub.py --selftest\n"
|
|
1124
|
+
)
|
|
1125
|
+
return 0 if args else 1
|
|
1126
|
+
if args[0] == "--selftest":
|
|
1127
|
+
return _selftest()
|
|
1128
|
+
if args[0] == "--preview":
|
|
1129
|
+
if len(args) < 2:
|
|
1130
|
+
sys.stderr.write(
|
|
1131
|
+
"usage: python scripts/lib/scrub.py --preview <file>\n"
|
|
1132
|
+
)
|
|
1133
|
+
return 1
|
|
1134
|
+
return _preview(args[1])
|
|
1135
|
+
|
|
1136
|
+
path = args[0]
|
|
1137
|
+
scrubbed, findings = scrub_file(path)
|
|
1138
|
+
sys.stdout.write(scrubbed)
|
|
1139
|
+
if scrubbed and not scrubbed.endswith("\n"):
|
|
1140
|
+
sys.stdout.write("\n")
|
|
1141
|
+
_summarize(findings, sys.stderr)
|
|
1142
|
+
# Non-zero exit if scrubbing errored, so callers/CI can detect it.
|
|
1143
|
+
if any(f["type"] == "error" for f in findings):
|
|
1144
|
+
return 2
|
|
1145
|
+
return 0
|
|
1146
|
+
|
|
1147
|
+
|
|
1148
|
+
if __name__ == "__main__":
|
|
1149
|
+
sys.exit(_main(sys.argv))
|