ecoportal-api 0.10.15 → 0.10.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of ecoportal-api might be problematic. Click here for more details.

Files changed (46) hide show
  1. checksums.yaml +4 -4
  2. data/.ai-assistance/.gitignore +2 -0
  3. data/.ai-assistance/bridge/.gitignore +10 -0
  4. data/.ai-assistance/bridge/CLAUDE.md +96 -0
  5. data/.ai-assistance/bridge/archive/.gitkeep +0 -0
  6. data/.ai-assistance/bridge/inbox/.gitkeep +0 -0
  7. data/.ai-assistance/bridge/outbox/.gitkeep +0 -0
  8. data/.ai-assistance/capabilities/assumptions-log.md +23 -0
  9. data/.ai-assistance/scripts/bridge-inbox-check.sh +119 -0
  10. data/.ai-assistance/scripts/bridge-init.sh +86 -0
  11. data/.ai-assistance/scripts/confine-to-subtree.sh +58 -0
  12. data/.ai-assistance/scripts/dirty-tree-guard.sh +96 -0
  13. data/.ai-assistance/scripts/distill_procedural.py +602 -0
  14. data/.ai-assistance/scripts/log-mcp-access.sh +24 -0
  15. data/.ai-assistance/scripts/log-skill-usage.sh +79 -0
  16. data/.ai-assistance/scripts/log_mcp_access.py +158 -0
  17. data/.ai-assistance/scripts/observe-session.sh +13 -0
  18. data/.ai-assistance/scripts/observe_session.py +287 -0
  19. data/.ai-assistance/scripts/protect-host-paths.sh +135 -0
  20. data/.ai-assistance/scripts/scrub.py +1149 -0
  21. data/.ai-assistance/scripts/scrub.py.sha256 +6 -0
  22. data/.ai-assistance/scripts/surface-procedural.sh +9 -0
  23. data/.ai-assistance/scripts/surface_procedural.py +101 -0
  24. data/.ai-assistance/skills/ep-ai-manager/SKILL.md +519 -0
  25. data/.ai-assistance/skills/project-self-docs/SKILL.md +259 -0
  26. data/.ai-assistance/skills/project-self-docs/scripts/self_docs_scan.py +378 -0
  27. data/.ai-assistance/standards-version.json +12 -0
  28. data/.ai-assistance/version.json +8 -0
  29. data/.claude/.gitignore +2 -0
  30. data/.claude/settings.json +128 -0
  31. data/CHANGELOG.md +14 -1
  32. data/CLAUDE.md +95 -0
  33. data/docs/self-docs/ARCHITECTURE.md +145 -0
  34. data/docs/self-docs/CHANGES.jsonl +7 -0
  35. data/docs/self-docs/COMPLIANCE.md +66 -0
  36. data/docs/self-docs/CONVENTIONS.md +74 -0
  37. data/docs/self-docs/INTEGRATIONS.md +62 -0
  38. data/docs/self-docs/OPERATIONS.md +64 -0
  39. data/docs/self-docs/OVERVIEW.md +61 -0
  40. data/docs/self-docs/STATUS.md +71 -0
  41. data/docs/self-docs/self-docs-index.json +51 -0
  42. data/docs/worklog.md +48 -0
  43. data/lib/ecoportal/api/common/client/with_retry.rb +6 -0
  44. data/lib/ecoportal/api/common/client.rb +12 -10
  45. data/lib/ecoportal/api/version.rb +1 -1
  46. metadata +41 -1
@@ -0,0 +1,1149 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ scrub.py -- conservative, deterministic PII / secret / local-path redactor.
4
+
5
+ FIRST PASS -- PENDING SECURITY REVIEW. This module is the ADR-011 /
6
+ external-llm-review prerequisite: before any learning or worklog content is
7
+ egressed to an external system (EPAI Confluence, Gemini), it must be stripped of
8
+ PII, secrets, and machine-local filesystem paths. Today seed-epai-learnings.py
9
+ --push is gated behind a manual EPAI_PUSH_CONFIRM=1 attestation precisely because
10
+ no scrub function existed. This file provides that function.
11
+
12
+ WIRING STATUS. Two consumers rely on this module:
13
+ - scripts/toggl-timesheet.py -- scrub() redaction of PII-adjacent free text (PII docking).
14
+ - scripts/seed-epai-learnings.py --push -- find_secrets() as a fail-closed BLOCK gate
15
+ before publishing auto-harvested content to the ROVO-indexed EPAI Confluence space.
16
+ Two egress models, deliberately different (see find_secrets vs scrub below):
17
+ - REDACT (scrub) for LLM egress: over-redaction is acceptable, so replace secrets with
18
+ placeholders and send the cleaned text.
19
+ - BLOCK (find_secrets) for PUBLISH / bridge egress: silently rewriting a human-readable
20
+ doc would corrupt it, so instead REFUSE egress when a high-confidence secret is present.
21
+ - .ai-assistance/skills/gemini-assist/gemini_ask.rb -- scrub_for_egress() REDACTS the full
22
+ prompt (files + task) before it is sent to the Gemini API. Fail-closed: it shells out to
23
+ THIS module and aborts the send if scrubbing cannot run.
24
+ The regexes are heuristics; they over-redact by design (false positives) and may still have
25
+ false negatives. Treat findings as a strong prompt, not an absolute guarantee. Egress paths not
26
+ separately gated are covered by compensating controls: the docs/structure Confluence seeders
27
+ publish COMMITTED content already guarded by the secret_scan CI check + human review; bridge
28
+ inbox/outbox are gitignored (no git leak) and the committed bridge archive is covered by
29
+ secret_scan. Wire an explicit gate there too if belt-and-suspenders is wanted.
30
+
31
+ Design notes:
32
+ - Stdlib only. ASCII only (see .ai-assistance/conventions/documentation-style.md).
33
+ - No network. No LLM. Pure-function, deterministic given the same input.
34
+ - Conservative: prefer false-positives (over-redaction) over leaks. When in doubt
35
+ for a high-entropy token, we redact and record a finding so a human sees it.
36
+ - scrub() NEVER raises: on any internal error it returns the original text plus an
37
+ {"type": "error", "count": 1} finding, so a caller can detect that scrubbing
38
+ did not run cleanly and refuse to egress.
39
+ - Redactions replace the match with a clear placeholder, e.g. [REDACTED:email].
40
+
41
+ Public API:
42
+ scrub(text) -> (scrubbed_text, findings)
43
+ findings is a list of {"type": str, "count": int}, one entry per pattern
44
+ that fired at least once.
45
+ scrub_file(path) -> (scrubbed_text, findings)
46
+
47
+ CLI:
48
+ python scripts/lib/scrub.py <file> # scrubbed text -> stdout, summary -> stderr
49
+ python scripts/lib/scrub.py --preview <file>
50
+ # NON-destructive redaction report (scrubbed
51
+ # text + findings table + MASKED samples);
52
+ # sends nothing, reveals no secret
53
+ python scripts/lib/scrub.py --selftest # run built-in pattern assertions, exit 0/1
54
+ """
55
+
56
+ import json
57
+ import os
58
+ import re
59
+ import sys
60
+ from pathlib import Path
61
+
62
+
63
+ # ---------------------------------------------------------------------------
64
+ # Code-fence masking
65
+ # ---------------------------------------------------------------------------
66
+ # High-entropy redaction is intentionally NOT applied inside fenced code blocks
67
+ # (``` ... ``` or ~~~ ... ~~~), because legitimate hashes, base64 fixtures, and
68
+ # sample data commonly live there and redacting them would mangle documentation.
69
+ # All the *targeted* patterns (emails, paths, known secret prefixes, AWS keys,
70
+ # JWTs, etc.) ARE still applied everywhere, including inside code fences -- a real
71
+ # secret pasted into a code block is still a leak.
72
+
73
+ _FENCE_RE = re.compile(r"(?ms)^[ \t]*(`{3,}|~{3,}).*?^[ \t]*\1[ \t]*$")
74
+
75
+
76
+ def _fence_spans(text):
77
+ """Return a list of (start, end) char spans covered by fenced code blocks."""
78
+ spans = []
79
+ try:
80
+ for m in _FENCE_RE.finditer(text):
81
+ spans.append((m.start(), m.end()))
82
+ except Exception:
83
+ # Never let fence detection break scrubbing; just treat as no fences.
84
+ return []
85
+ return spans
86
+
87
+
88
+ def _in_spans(pos, spans):
89
+ for s, e in spans:
90
+ if s <= pos < e:
91
+ return True
92
+ return False
93
+
94
+
95
+ # ---------------------------------------------------------------------------
96
+ # Pattern table
97
+ # ---------------------------------------------------------------------------
98
+ # Each entry: (type_name, compiled_regex, respect_code_fences)
99
+ # respect_code_fences=True means matches inside fenced code blocks are skipped
100
+ # (used only for the broad high-entropy catch-alls).
101
+ #
102
+ # Order matters: more specific / higher-confidence patterns run first so that,
103
+ # e.g., an AWS key or JWT is labelled as such rather than swallowed by the
104
+ # generic high-entropy rule. Once a region is redacted it is replaced by an
105
+ # ASCII placeholder that later patterns will not re-match.
106
+
107
+ # Email -- deliberately broad local part.
108
+ _EMAIL = re.compile(
109
+ r"[A-Za-z0-9._%+\-]+@[A-Za-z0-9.\-]+\.[A-Za-z]{2,}"
110
+ )
111
+
112
+ # Windows absolute path: drive letter + (\ or /) + at least one segment.
113
+ # Matches C:\Users\rella\... and C:/claude/Projects/...
114
+ _WIN_PATH = re.compile(
115
+ r"[A-Za-z]:[\\/](?:[^\s\\/:*?\"<>|]+[\\/]?)+"
116
+ )
117
+
118
+ # POSIX user-home / mount absolute paths.
119
+ # /home/<user>/... /Users/<user>/... /mnt/... /root/... /var/.../<user>...
120
+ # Kept reasonably tight (requires a known root) to limit false positives on
121
+ # ordinary prose containing slashes. Still over-redacts URLs paths sometimes;
122
+ # acceptable per the conservative contract.
123
+ _POSIX_HOME = re.compile(
124
+ r"(?:/home/|/Users/|/mnt/|/media/|/root/|/srv/|/opt/)"
125
+ r"[^\s\"'<>|:]*"
126
+ )
127
+ # Bare ~ home expansion: ~/something or ~user/something
128
+ _TILDE_HOME = re.compile(r"~[A-Za-z0-9_\-]*\/[^\s\"'<>|:]+")
129
+
130
+ # Secret prefixes (high confidence).
131
+ _AWS_AKIA = re.compile(r"\b(?:AKIA|ASIA|AROA|AIDA|ANPA|ANVA|AIPA)[0-9A-Z]{16}\b")
132
+ _GITLAB_PAT = re.compile(r"\bglpat-[A-Za-z0-9_\-]{20,}\b")
133
+ _GITHUB_PAT = re.compile(r"\b(?:ghp|gho|ghu|ghs|ghr)_[A-Za-z0-9]{36,}\b")
134
+ _GITHUB_FINE_PAT = re.compile(r"\bgithub_pat_[A-Za-z0-9_]{30,}\b")
135
+ _GOOGLE_API = re.compile(r"\bAIza[0-9A-Za-z_\-]{35}\b")
136
+ # Google/Gemini newer API-key format: "AQ." + base64url body (e.g. AQ.Ab8R...).
137
+ # The AIza pattern above does not cover it; observed on live Gemini keys.
138
+ _GOOGLE_AQ = re.compile(r"\bAQ\.[A-Za-z0-9_\-]{30,}\b")
139
+ # Slack tokens (bonus, high confidence).
140
+ _SLACK = re.compile(r"\bxox[baprs]-[A-Za-z0-9\-]{10,}\b")
141
+ # JWT: three dot-separated base64url segments starting with the typical eyJ header.
142
+ _JWT = re.compile(r"\beyJ[A-Za-z0-9_\-]+\.[A-Za-z0-9_\-]+\.[A-Za-z0-9_\-]+\b")
143
+
144
+ # Ported from the AWS corpus-pipeline scrubber (do not lose these scoped detectors):
145
+ _ATLASSIAN = re.compile(r"\bATATT3x[A-Za-z0-9]{30,}\b") # Atlassian API token
146
+ _ANTHROPIC = re.compile(r"\bsk-ant-[A-Za-z0-9_\-]{40,}\b") # Anthropic API key
147
+ _OPENAI = re.compile(r"\bsk-[A-Za-z0-9]{48,}\b") # OpenAI API key
148
+ _AWS_ARN = re.compile(r"\barn:aws:[a-z0-9\-]+:[a-z0-9\-]*:\d{12}:[^\s`\]\n]+")
149
+ _BASIC_AUTH_URL = re.compile(r"//[^@\s:/?#]{3,}:[^@\s:/?#]{3,}@") # user:pass@ in a URL
150
+ # Phone numbers (E.164 international + NZ/AU domestic; NZ-scoped per eP).
151
+ _PHONE_E164 = re.compile(r"\+\d{1,3}[\s.\-]?\(?\d{1,4}\)?[\s.\-]?\d{3,4}[\s.\-]?\d{3,4}\b")
152
+ _PHONE_NZ_AU = re.compile(r"\b0[2-9]\d?[\s.\-]?\d{3,4}[\s.\-]?\d{3,4}\b")
153
+ # Private-range IP in CIDR notation (bare IPs still caught by _IPV4 below).
154
+ _CIDR = re.compile(r"\b(?:\d{1,3}\.){3}\d{1,3}/\d{1,2}\b")
155
+
156
+ # Bearer tokens: "Bearer <token>".
157
+ _BEARER = re.compile(r"(?i)\bbearer\s+[A-Za-z0-9._\-]{8,}")
158
+
159
+ # key=value style secret assignments. Captures common key names with an optional
160
+ # quote around the value. value is anything non-space / non-quote, >=6 chars.
161
+ # NOTE: no leading \b -- the keyword must match even when it is a SUFFIX of a
162
+ # prefixed identifier (GEMINI_API_KEY, CONFLUENCE_EPAI_API_KEY), where the char
163
+ # before "api" is "_" (a word char, so \b fails). The trailing suffix group also
164
+ # lets access-level-suffixed names match (FOO_TOKEN_RO=, BAR_TOKEN_RW=) -- exactly
165
+ # the shape of our own env-var-naming convention. Matching starts mid-identifier,
166
+ # so the redaction keeps the prefix (GEMINI_[REDACTED:secret_assignment]).
167
+ _ASSIGN = re.compile(
168
+ r"(?i)(?:api[_\-]?key|apikey|secret(?:[_\-]?key)?|token|access[_\-]?token|"
169
+ r"auth[_\-]?token|password|passwd|pwd|client[_\-]?secret|private[_\-]?key)"
170
+ r"(?:[_\-][A-Za-z0-9]+)*"
171
+ r"\s*[:=]\s*['\"]?([^\s'\"]{6,})['\"]?"
172
+ )
173
+
174
+ # IPv4 (low priority, optional). Bounded octets to reduce matching version-like
175
+ # strings; still may catch some non-address dotted quads -- acceptable.
176
+ _IPV4 = re.compile(
177
+ r"\b(?:(?:25[0-5]|2[0-4]\d|1?\d?\d)\.){3}(?:25[0-5]|2[0-4]\d|1?\d?\d)\b"
178
+ )
179
+
180
+ # High-entropy catch-alls (only OUTSIDE code fences). Long base64-ish or hex runs
181
+ # that are likely keys/hashes pasted into prose. >= 40 hex chars or >= 32 base64
182
+ # chars. These deliberately over-redact; every hit is recorded as a finding.
183
+ _HEX_LONG = re.compile(r"\b[0-9a-fA-F]{40,}\b")
184
+ _B64_LONG = re.compile(r"\b[A-Za-z0-9+/]{32,}={0,2}\b")
185
+
186
+
187
+ # (type, regex, respect_fences). Targeted patterns ignore fences (real secrets
188
+ # in code blocks still leak); only the generic entropy rules respect fences.
189
+ _PATTERNS = [
190
+ ("basic_auth_url", _BASIC_AUTH_URL, False),
191
+ ("email", _EMAIL, False),
192
+ ("aws_access_key", _AWS_AKIA, False),
193
+ ("gitlab_pat", _GITLAB_PAT, False),
194
+ ("github_pat", _GITHUB_PAT, False),
195
+ ("github_fine_pat", _GITHUB_FINE_PAT, False),
196
+ ("google_api_key", _GOOGLE_API, False),
197
+ ("google_aq_key", _GOOGLE_AQ, False),
198
+ ("slack_token", _SLACK, False),
199
+ ("jwt", _JWT, False),
200
+ ("atlassian_token", _ATLASSIAN, False),
201
+ ("anthropic_key", _ANTHROPIC, False),
202
+ ("openai_key", _OPENAI, False),
203
+ ("aws_arn", _AWS_ARN, False),
204
+ ("bearer_token", _BEARER, False),
205
+ ("secret_assignment", _ASSIGN, False),
206
+ ("windows_path", _WIN_PATH, False),
207
+ ("posix_home_path", _POSIX_HOME, False),
208
+ ("tilde_home_path", _TILDE_HOME, False),
209
+ ("network_cidr", _CIDR, False),
210
+ ("phone", _PHONE_E164, False),
211
+ ("phone", _PHONE_NZ_AU, False),
212
+ ("ipv4", _IPV4, False),
213
+ ("high_entropy_hex", _HEX_LONG, True),
214
+ ("high_entropy_base64", _B64_LONG, True),
215
+ ]
216
+
217
+
218
+ def _placeholder(type_name):
219
+ return "[REDACTED:%s]" % type_name
220
+
221
+
222
+ def _apply_pattern(text, type_name, regex, respect_fences, matches=None):
223
+ """Replace all matches of one pattern. Returns (new_text, count).
224
+
225
+ Recomputes code-fence spans against the CURRENT text when respect_fences is
226
+ True, so a match's position is checked against fences in the same string.
227
+
228
+ If `matches` is a list, one dict per redaction is appended to it:
229
+ {"type", "start", "end", "replacement", "original"} where start/end are
230
+ offsets INTO THE OUTPUT (scrubbed) text and "original" is the raw matched
231
+ substring (used only for preview masking -- never persisted to egress).
232
+ """
233
+ placeholder = _placeholder(type_name)
234
+ if respect_fences:
235
+ spans = _fence_spans(text)
236
+ else:
237
+ spans = None
238
+
239
+ count = 0
240
+ out = []
241
+ out_len = 0
242
+ last = 0
243
+ for m in regex.finditer(text):
244
+ start = m.start()
245
+ if spans is not None and _in_spans(start, spans):
246
+ continue
247
+ gap = text[last:start]
248
+ out.append(gap)
249
+ out_len += len(gap)
250
+ if matches is not None:
251
+ matches.append({
252
+ "type": type_name,
253
+ "start": out_len,
254
+ "end": out_len + len(placeholder),
255
+ "replacement": placeholder,
256
+ "original": m.group(0),
257
+ })
258
+ out.append(placeholder)
259
+ out_len += len(placeholder)
260
+ last = m.end()
261
+ count += 1
262
+ out.append(text[last:])
263
+ return "".join(out), count
264
+
265
+
266
+ # ---------------------------------------------------------------------------
267
+ # Layer-1 dictionary pass (known private entities -> generic placeholder)
268
+ # ---------------------------------------------------------------------------
269
+ # A deterministic, dictionary-backed redaction of KNOWN private entities
270
+ # (customer organizations, people, internal project names) that recur across
271
+ # ecoPortal work and would otherwise leak on egress. The dictionary is built by
272
+ # scripts/build-private-data-dictionary.py and lives, gitignored, at
273
+ # .ai-assistance/local/private-data/ (override with $PRIVATE_DATA_DICT_DIR).
274
+ #
275
+ # Contract (shared with the builder):
276
+ # Sharded files organizations.json / people.json / projects.json, each:
277
+ # {"version":1, "generated":..., "category":..., "count":N,
278
+ # "entities":[{"key","display","placeholder","aliases":[],"sources":[]}]}
279
+ # Each entity's display + aliases all map, FORWARD-ONLY, to the SAME generic
280
+ # placeholder ([ORG]/[PERSON]/[PROJECT]). There is NO reverse map -- once
281
+ # redacted, the original name is unrecoverable from the output.
282
+ #
283
+ # Semantics: case-insensitive, word-boundary, LONGEST-MATCH-FIRST (so
284
+ # "Briscoe Group" wins over "Briscoe"). A single precompiled regex alternation
285
+ # (sorted longest-first, re.escape'd) does the matching in one pass.
286
+ #
287
+ # Graceful skip: if the directory is absent or holds no usable entities, the pass
288
+ # is a no-op and regex-only behaviour is unchanged.
289
+
290
+ _DICT_SHARD_FILES = ("organizations.json", "people.json", "projects.json")
291
+ # Generic placeholders by category (must agree with the builder's PLACEHOLDERS).
292
+ _DICT_PLACEHOLDERS = {
293
+ "organization": "[ORG]",
294
+ "person": "[PERSON]",
295
+ "project": "[PROJECT]",
296
+ }
297
+
298
+
299
+ def _dict_dir():
300
+ """Resolve the private-data dictionary directory.
301
+
302
+ Order: $PRIVATE_DATA_DICT_DIR, else the repo-default
303
+ .ai-assistance/local/private-data/ relative to this file (scripts/lib/).
304
+ """
305
+ env = os.environ.get("PRIVATE_DATA_DICT_DIR")
306
+ if env:
307
+ return Path(env)
308
+ # scripts/lib/scrub.py -> repo root is two parents up.
309
+ root = Path(__file__).resolve().parent.parent.parent
310
+ return root / ".ai-assistance" / "local" / "private-data"
311
+
312
+
313
+ def load_dictionary(directory=None):
314
+ """Load the sharded dictionary into a list of (phrase, placeholder) pairs.
315
+
316
+ Reads organizations.json / people.json / projects.json from `directory`
317
+ (default: _dict_dir()). Each entity contributes its display and every alias
318
+ as a phrase, all mapped to the entity's generic placeholder. Returns [] if
319
+ the directory is absent/empty or nothing usable is found. NEVER raises.
320
+ """
321
+ try:
322
+ directory = Path(directory) if directory is not None else _dict_dir()
323
+ if not directory.is_dir():
324
+ return []
325
+ pairs = []
326
+ seen = set()
327
+ for fname in _DICT_SHARD_FILES:
328
+ fpath = directory / fname
329
+ if not fpath.is_file():
330
+ continue
331
+ try:
332
+ data = json.loads(fpath.read_text(encoding="utf-8"))
333
+ except Exception:
334
+ # A malformed shard is skipped, not fatal -- regex passes still run.
335
+ continue
336
+ entities = data.get("entities") if isinstance(data, dict) else None
337
+ if not isinstance(entities, list):
338
+ continue
339
+ for ent in entities:
340
+ if not isinstance(ent, dict):
341
+ continue
342
+ placeholder = ent.get("placeholder")
343
+ if not placeholder:
344
+ # Fall back to category default if placeholder missing.
345
+ placeholder = _DICT_PLACEHOLDERS.get(ent.get("category"))
346
+ if not placeholder:
347
+ continue
348
+ phrases = []
349
+ disp = ent.get("display")
350
+ if isinstance(disp, str):
351
+ phrases.append(disp)
352
+ aliases = ent.get("aliases")
353
+ if isinstance(aliases, list):
354
+ phrases.extend(a for a in aliases if isinstance(a, str))
355
+ for phrase in phrases:
356
+ phrase = phrase.strip()
357
+ if not phrase:
358
+ continue
359
+ dedup_key = (phrase.casefold(), placeholder)
360
+ if dedup_key in seen:
361
+ continue
362
+ seen.add(dedup_key)
363
+ pairs.append((phrase, placeholder))
364
+ return pairs
365
+ except Exception:
366
+ return []
367
+
368
+
369
+ def _compile_dictionary(pairs):
370
+ """Compile (phrase, placeholder) pairs into (regex, {group->placeholder}).
371
+
372
+ Longest-match-first: phrases are sorted by descending length before being
373
+ joined into a single alternation, so the regex engine prefers the longest
374
+ phrase at a given position. Case-insensitive, word-boundary anchored.
375
+ Returns (None, {}) when there is nothing to compile.
376
+ """
377
+ if not pairs:
378
+ return None, {}
379
+ # Sort longest phrase first (tie-break casefold for determinism).
380
+ ordered = sorted(pairs, key=lambda p: (-len(p[0]), p[0].casefold()))
381
+ alts = []
382
+ group_placeholder = {}
383
+ for i, (phrase, placeholder) in enumerate(ordered):
384
+ gname = "d%d" % i
385
+ alts.append("(?P<%s>%s)" % (gname, re.escape(phrase)))
386
+ group_placeholder[gname] = placeholder
387
+ # Alphanumeric lookarounds (not \b) so entity names containing punctuation
388
+ # -- "AT&T", "Acme Corp." -- still match at their real boundaries (ported from
389
+ # the AWS corpus-pipeline registry cleaner). re.IGNORECASE for case.
390
+ pattern = r"(?<![A-Za-z0-9])(?:%s)(?![A-Za-z0-9])" % "|".join(alts)
391
+ try:
392
+ regex = re.compile(pattern, re.IGNORECASE)
393
+ except Exception:
394
+ return None, {}
395
+ return regex, group_placeholder
396
+
397
+
398
+ def _apply_dictionary(text, compiled, matches=None):
399
+ """Replace known-entity matches with their generic placeholder.
400
+
401
+ Returns (new_text, count). `compiled` is (regex, group->placeholder) from
402
+ _compile_dictionary. No-op (returns text, 0) when regex is None.
403
+
404
+ If `matches` is a list, one dict per redaction is appended (same shape as
405
+ _apply_pattern), with type "known-entity" and offsets into the OUTPUT text.
406
+ """
407
+ regex, group_placeholder = compiled
408
+ if regex is None:
409
+ return text, 0
410
+ count = 0
411
+ out = []
412
+ out_len = 0
413
+ last = 0
414
+ for m in regex.finditer(text):
415
+ # Identify which named group actually matched.
416
+ placeholder = None
417
+ for gname, ph in group_placeholder.items():
418
+ if m.group(gname) is not None:
419
+ placeholder = ph
420
+ break
421
+ if placeholder is None:
422
+ continue
423
+ gap = text[last:m.start()]
424
+ out.append(gap)
425
+ out_len += len(gap)
426
+ if matches is not None:
427
+ matches.append({
428
+ "type": "known-entity",
429
+ "start": out_len,
430
+ "end": out_len + len(placeholder),
431
+ "replacement": placeholder,
432
+ "original": m.group(0),
433
+ })
434
+ out.append(placeholder)
435
+ out_len += len(placeholder)
436
+ last = m.end()
437
+ count += 1
438
+ out.append(text[last:])
439
+ return "".join(out), count
440
+
441
+
442
+ # ---------------------------------------------------------------------------
443
+ # Aho-Corasick fast path for the dictionary pass
444
+ # ---------------------------------------------------------------------------
445
+ # The regex oracle above compiles ~4,435 phrases into ONE alternation and lets
446
+ # Python `re` backtrack over it (O(text x alternatives)); on an 84KB file that is
447
+ # ~100s. This path builds an Aho-Corasick automaton once (linear build) and scans
448
+ # the text once (linear O(text)), reproducing the oracle's selection rule EXACTLY:
449
+ #
450
+ # - Case-insensitive via a length-preserving `str.lower()` fold. `re.IGNORECASE`
451
+ # matches char-by-char and (as proven exhaustively over all of Unicode) agrees
452
+ # with `.lower()` on every character pair EXCEPT a fixed set of 68 non-ASCII
453
+ # "exotic" codepoints (long-s, micro-sign, Greek/Cyrillic glyph variants, the
454
+ # U+0130 dotted-I whose lower() is 2 chars, a couple of ligatures). If ANY such
455
+ # char appears in the text or a phrase we do NOT fast-path: build_automaton
456
+ # returns None and scrub() uses the regex oracle. For all ASCII + Maori-macron
457
+ # text (the entire real corpus) `.lower()` is byte-for-byte equivalent to
458
+ # re.IGNORECASE and length-preserving, so offsets never drift.
459
+ # - Boundaries are the SAME ASCII-alnum lookarounds (not \b): a match needs a
460
+ # non-[A-Za-z0-9] char (or string end) immediately before its start and after
461
+ # its end, checked against the ORIGINAL text positions.
462
+ # - Leftmost, longest-at-start, non-overlapping: AC collects every phrase
463
+ # occurrence; a single left-to-right sweep then picks, at the earliest start
464
+ # with a boundary-valid phrase, the LONGEST such phrase (ties broken by the
465
+ # oracle's sort order: (-len, casefold), stable), emits it, and resumes at the
466
+ # match end -- identical to re.finditer over the sorted alternation.
467
+ # - The winning phrase's placeholder replaces the matched ORIGINAL-case span.
468
+
469
+ # Codepoints where str.lower() is not a length-preserving, re.IGNORECASE-faithful
470
+ # fold. Derived by exhaustively comparing str.lower() equivalence against
471
+ # re.IGNORECASE over every Unicode codepoint (see the perf/scrub-aho-corasick
472
+ # design notes). Presence of ANY of these in text or a phrase forces the oracle.
473
+ _AC_EXOTIC_CODEPOINTS = frozenset({
474
+ 0xB5, 0x130, 0x17F, 0x345, 0x390, 0x392, 0x395, 0x398, 0x399, 0x39A,
475
+ 0x39C, 0x3A0, 0x3A1, 0x3A3, 0x3A6, 0x3B0, 0x3B2, 0x3B5, 0x3B8, 0x3B9,
476
+ 0x3BA, 0x3BC, 0x3C0, 0x3C1, 0x3C2, 0x3C3, 0x3C6, 0x3D0, 0x3D1, 0x3D5,
477
+ 0x3D6, 0x3F0, 0x3F1, 0x3F4, 0x3F5, 0x412, 0x414, 0x41E, 0x421, 0x422,
478
+ 0x42A, 0x432, 0x434, 0x43E, 0x441, 0x442, 0x44A, 0x462, 0x463, 0x1C80,
479
+ 0x1C81, 0x1C82, 0x1C83, 0x1C84, 0x1C85, 0x1C86, 0x1C87, 0x1C88, 0x1E60,
480
+ 0x1E61, 0x1E9B, 0x1FBE, 0x1FD3, 0x1FE3, 0xA64A, 0xA64B, 0xFB05, 0xFB06,
481
+ })
482
+
483
+
484
+ def _ac_has_exotic(s):
485
+ """True if s contains any codepoint where str.lower() is not a faithful,
486
+ length-preserving stand-in for re.IGNORECASE (forces the regex oracle)."""
487
+ for ch in s:
488
+ if ord(ch) in _AC_EXOTIC_CODEPOINTS:
489
+ return True
490
+ return False
491
+
492
+
493
+ def build_automaton(pairs):
494
+ """Build an Aho-Corasick automaton from (phrase, placeholder) pairs.
495
+
496
+ Returns an opaque dict consumed by _apply_dictionary_ac, or None when the
497
+ fast path is not usable (no pairs, or a phrase contains an exotic char --
498
+ see _AC_EXOTIC_CODEPOINTS -- in which case the caller falls back to the
499
+ regex oracle for guaranteed equivalence).
500
+
501
+ Phrases are keyed on their str.lower() fold. The same ordering the oracle
502
+ uses -- sorted by (-len(phrase), phrase.casefold()), stable -- is preserved
503
+ as each phrase's `rank`, so ties at a start position resolve identically.
504
+ """
505
+ if not pairs:
506
+ return None
507
+ # Same order the regex oracle imposes on its alternation.
508
+ ordered = sorted(pairs, key=lambda p: (-len(p[0]), p[0].casefold()))
509
+ # Node 0 is root. goto: list of dicts (char -> node). out: list of terminal
510
+ # payloads per node, each (phrase_len, rank, placeholder). fail: list of ints.
511
+ goto = [{}]
512
+ out = [[]]
513
+ for rank, (phrase, placeholder) in enumerate(ordered):
514
+ if _ac_has_exotic(phrase):
515
+ # A phrase with an exotic char cannot be folded safely -> no fast path.
516
+ return None
517
+ folded = phrase.lower()
518
+ if not folded:
519
+ continue
520
+ node = 0
521
+ for ch in folded:
522
+ nxt = goto[node].get(ch)
523
+ if nxt is None:
524
+ nxt = len(goto)
525
+ goto.append({})
526
+ out.append([])
527
+ goto[node][ch] = nxt
528
+ node = nxt
529
+ out[node].append((len(folded), rank, placeholder))
530
+ # Build failure links (BFS). fail[root]=0; fail of depth-1 nodes = root.
531
+ fail = [0] * len(goto)
532
+ queue = []
533
+ for ch, nxt in goto[0].items():
534
+ fail[nxt] = 0
535
+ queue.append(nxt)
536
+ head = 0
537
+ while head < len(queue):
538
+ node = queue[head]
539
+ head += 1
540
+ for ch, nxt in goto[node].items():
541
+ queue.append(nxt)
542
+ f = fail[node]
543
+ while f and ch not in goto[f]:
544
+ f = fail[f]
545
+ fail[nxt] = goto[f].get(ch, 0) if f or ch in goto[0] else 0
546
+ # Merge terminal outputs along the failure chain (report ALL phrases
547
+ # that end here, including shorter suffix phrases).
548
+ if out[fail[nxt]]:
549
+ out[nxt] = out[nxt] + out[fail[nxt]]
550
+ return {"goto": goto, "out": out, "fail": fail}
551
+
552
+
553
+ def _ac_alnum(ch):
554
+ """ASCII alnum test matching the oracle's [A-Za-z0-9] character class."""
555
+ return ("0" <= ch <= "9") or ("A" <= ch <= "Z") or ("a" <= ch <= "z")
556
+
557
+
558
+ def _apply_dictionary_ac(text, automaton, matches=None):
559
+ """Aho-Corasick equivalent of _apply_dictionary. Returns (new_text, count).
560
+
561
+ Reproduces the regex oracle byte-for-byte for text free of exotic chars.
562
+ Raises on any unexpected condition so scrub() can fall back to the oracle;
563
+ callers MUST wrap this and fall back on exception (fail-safe, never open).
564
+
565
+ `matches`, if a list, receives one dict per redaction (same shape/semantics
566
+ as _apply_dictionary): type "known-entity", offsets into the OUTPUT text,
567
+ "original" the matched original-case span, "replacement" the placeholder.
568
+ """
569
+ if automaton is None:
570
+ return text, 0
571
+ if _ac_has_exotic(text):
572
+ # Exotic char in the text: not safe to fold with .lower() -> force oracle.
573
+ raise ValueError("exotic codepoint in text; use regex oracle")
574
+
575
+ goto = automaton["goto"]
576
+ out = automaton["out"]
577
+ fail = automaton["fail"]
578
+ folded = text.lower()
579
+ # .lower() is length-preserving for all non-exotic input (guarded above);
580
+ # assert alignment so any surprise drift trips the fallback instead of leaking.
581
+ if len(folded) != len(text):
582
+ raise ValueError("fold changed length; use regex oracle")
583
+
584
+ n = len(text)
585
+ # Pass 1: AC sweep. For every end position, record the BEST phrase ending
586
+ # there whose start boundary is valid, keyed for start-position selection.
587
+ # best_at_start[start] = (phrase_len, rank, placeholder, end) for the phrase
588
+ # the oracle would choose if a match begins at `start` (longest, then lowest
589
+ # rank). We only keep boundary-valid candidates.
590
+ best_at_start = {}
591
+ node = 0
592
+ for i in range(n):
593
+ ch = folded[i]
594
+ while node and ch not in goto[node]:
595
+ node = fail[node]
596
+ node = goto[node].get(ch, 0)
597
+ if not out[node]:
598
+ continue
599
+ end = i + 1 # exclusive
600
+ # Trailing boundary: char after end must not be ASCII alnum (or be EOS).
601
+ if end < n and _ac_alnum(text[end]):
602
+ trailing_ok = False
603
+ else:
604
+ trailing_ok = True
605
+ if not trailing_ok:
606
+ continue
607
+ for (plen, rank, placeholder) in out[node]:
608
+ start = end - plen
609
+ # Leading boundary: char before start must not be ASCII alnum (or BOS).
610
+ if start > 0 and _ac_alnum(text[start - 1]):
611
+ continue
612
+ cur = best_at_start.get(start)
613
+ # Oracle preference at a start: longer wins, then lower rank.
614
+ if cur is None or (plen, -rank) > (cur[0], -cur[1]):
615
+ best_at_start[start] = (plen, rank, placeholder, end)
616
+
617
+ if not best_at_start:
618
+ return text, 0
619
+
620
+ # Pass 2: leftmost, non-overlapping selection -- resume at match end.
621
+ count = 0
622
+ out_parts = []
623
+ out_len = 0
624
+ last = 0
625
+ pos = 0
626
+ starts = sorted(best_at_start)
627
+ si = 0
628
+ ns = len(starts)
629
+ while si < ns:
630
+ start = starts[si]
631
+ if start < pos:
632
+ si += 1
633
+ continue
634
+ plen, rank, placeholder, end = best_at_start[start]
635
+ gap = text[last:start]
636
+ out_parts.append(gap)
637
+ out_len += len(gap)
638
+ if matches is not None:
639
+ matches.append({
640
+ "type": "known-entity",
641
+ "start": out_len,
642
+ "end": out_len + len(placeholder),
643
+ "replacement": placeholder,
644
+ "original": text[start:end],
645
+ })
646
+ out_parts.append(placeholder)
647
+ out_len += len(placeholder)
648
+ last = end
649
+ pos = end
650
+ count += 1
651
+ si += 1
652
+ out_parts.append(text[last:])
653
+ return "".join(out_parts), count
654
+
655
+
656
+ def _run_dictionary_pass(text, matches=None):
657
+ """Run the layer-1 dictionary pass, preferring the Aho-Corasick fast path
658
+ and FALLING BACK to the regex oracle on any AC exception (fail-safe: the
659
+ oracle is the known-correct path, so a fallback can only be correct, never
660
+ a leak). Returns (new_text, count). Never raises.
661
+
662
+ Loads + builds the automaton (and the oracle) on each call, mirroring the
663
+ prior behaviour where _compile_dictionary(load_dictionary()) was rebuilt per
664
+ scrub. When AC is unavailable (no dict, or an exotic phrase) build_automaton
665
+ returns None and we use the oracle directly.
666
+ """
667
+ pairs = load_dictionary()
668
+ if not pairs:
669
+ return text, 0
670
+ automaton = None
671
+ try:
672
+ automaton = build_automaton(pairs)
673
+ except Exception:
674
+ automaton = None
675
+ if automaton is not None:
676
+ # AC fast path; any surprise -> oracle. If `matches` was partially
677
+ # populated before an exception, drop those entries so the oracle
678
+ # reruns cleanly (the oracle appends from a clean slate).
679
+ marker = len(matches) if matches is not None else 0
680
+ try:
681
+ return _apply_dictionary_ac(text, automaton, matches)
682
+ except Exception:
683
+ if matches is not None:
684
+ del matches[marker:]
685
+ # Regex oracle (fallback, or the only path when AC is unavailable).
686
+ compiled = _compile_dictionary(pairs)
687
+ return _apply_dictionary(text, compiled, matches)
688
+
689
+
690
+ def scrub(text):
691
+ """Redact PII / secrets / local paths from text.
692
+
693
+ Returns (scrubbed_text, findings) where findings is a list of
694
+ {"type": str, "count": int}, one per pattern that matched at least once.
695
+
696
+ NEVER raises. On any internal error returns (original_text, [{"type":
697
+ "error", "count": 1}]) so a caller can detect failure and refuse to egress.
698
+ """
699
+ try:
700
+ if text is None:
701
+ return "", []
702
+ if not isinstance(text, str):
703
+ text = str(text)
704
+
705
+ findings = []
706
+ scrubbed = text
707
+
708
+ # Layer-1 dictionary pass FIRST: fold known private entities to their
709
+ # generic placeholder before the regex heuristics run. Graceful no-op if
710
+ # no dictionary is present. Never raises (loader/compile swallow errors).
711
+ try:
712
+ scrubbed, dict_count = _run_dictionary_pass(scrubbed)
713
+ if dict_count:
714
+ findings.append({"type": "known-entity", "count": dict_count})
715
+ except Exception:
716
+ # Dictionary pass must never break scrubbing; regex passes still run.
717
+ pass
718
+
719
+ for type_name, regex, respect_fences in _PATTERNS:
720
+ scrubbed, count = _apply_pattern(
721
+ scrubbed, type_name, regex, respect_fences
722
+ )
723
+ if count:
724
+ findings.append({"type": type_name, "count": count})
725
+ return scrubbed, findings
726
+ except Exception as exc: # never raise out of scrub()
727
+ try:
728
+ sys.stderr.write("scrub: internal error: %r\n" % (exc,))
729
+ except Exception:
730
+ pass
731
+ return text if isinstance(text, str) else "", [
732
+ {"type": "error", "count": 1}
733
+ ]
734
+
735
+
736
+ def scrub_detailed(text):
737
+ """Like scrub(), but also return per-match detail for PREVIEW/reporting.
738
+
739
+ Returns (scrubbed_text, findings, matches) where:
740
+ - scrubbed_text / findings are IDENTICAL to what scrub(text) returns
741
+ (same values, same shapes) -- this function does not change scrub()'s
742
+ public contract, it augments it.
743
+ - matches is a list of dicts, one per individual redaction, in the order
744
+ they were applied:
745
+ {"type": str, "start": int, "end": int,
746
+ "replacement": str, "original": str}
747
+ "original" is the RAW matched substring; it is provided ONLY so a caller
748
+ can render a MASKED sample and MUST NOT be egressed or logged verbatim.
749
+ "start"/"end" are offsets into the scrubbed text AT THE TIME that pattern
750
+ pass ran (passes run sequentially and later passes rewrite the string, so
751
+ treat them as an ordering/locality hint, not a guaranteed final offset).
752
+
753
+ NEVER raises: on internal error returns (original_text, [error-finding], []).
754
+ """
755
+ try:
756
+ if text is None:
757
+ return "", [], []
758
+ if not isinstance(text, str):
759
+ text = str(text)
760
+
761
+ findings = []
762
+ matches = []
763
+ scrubbed = text
764
+
765
+ try:
766
+ scrubbed, dict_count = _run_dictionary_pass(scrubbed, matches)
767
+ if dict_count:
768
+ findings.append({"type": "known-entity", "count": dict_count})
769
+ except Exception:
770
+ pass
771
+
772
+ for type_name, regex, respect_fences in _PATTERNS:
773
+ scrubbed, count = _apply_pattern(
774
+ scrubbed, type_name, regex, respect_fences, matches
775
+ )
776
+ if count:
777
+ findings.append({"type": type_name, "count": count})
778
+ return scrubbed, findings, matches
779
+ except Exception as exc: # never raise
780
+ try:
781
+ sys.stderr.write("scrub_detailed: internal error: %r\n" % (exc,))
782
+ except Exception:
783
+ pass
784
+ return (text if isinstance(text, str) else ""), [
785
+ {"type": "error", "count": 1}
786
+ ], []
787
+
788
+
789
+ # ---------------------------------------------------------------------------
790
+ # Egress block-gate (B / ADR-011)
791
+ # ---------------------------------------------------------------------------
792
+ # High-confidence secret types: a match here is almost certainly a REAL credential,
793
+ # not a heuristic false-positive. find_secrets() filters to these so it can be used as
794
+ # a BLOCK gate before egress (publish to Confluence, write to a bridge file), where
795
+ # silently redacting would corrupt legitimate content (governance docs contain example
796
+ # tokens, IPs, paths). The noisy heuristics -- email, *_path, ipv4, network_cidr, phone,
797
+ # high_entropy_*, known-entity -- are deliberately EXCLUDED here; they belong to
798
+ # redaction (scrub()), used for LLM egress where over-redaction is acceptable.
799
+ HIGH_CONFIDENCE_TYPES = frozenset({
800
+ "basic_auth_url", "aws_access_key", "gitlab_pat", "github_pat", "github_fine_pat",
801
+ "google_api_key", "google_aq_key", "slack_token", "jwt", "atlassian_token",
802
+ "anthropic_key", "openai_key", "bearer_token", "secret_assignment",
803
+ })
804
+
805
+
806
+ def find_secrets(text):
807
+ """Return high-confidence secret findings in `text` (list of {"type","count"}).
808
+
809
+ Non-empty => `text` contains something that must NOT be egressed. Uses scrub()'s
810
+ detector but filters to HIGH_CONFIDENCE_TYPES so noisy heuristics do not false-block.
811
+ Fail-closed: if scrub() itself errored (could not verify), that error finding is
812
+ included, so callers treat an un-verifiable input as unsafe. Never raises.
813
+ """
814
+ try:
815
+ _, findings = scrub(text)
816
+ except Exception:
817
+ return [{"type": "error", "count": 1}]
818
+ return [f for f in findings
819
+ if f["type"] in HIGH_CONFIDENCE_TYPES or f["type"] == "error"]
820
+
821
+
822
+ def scrub_file(path):
823
+ """Read a file (UTF-8, errors replaced) and scrub its contents.
824
+
825
+ Returns (scrubbed_text, findings). Like scrub(), never raises: a read error
826
+ is reported as an {"type": "error", "count": 1} finding.
827
+ """
828
+ try:
829
+ with open(path, "r", encoding="utf-8", errors="replace") as fh:
830
+ raw = fh.read()
831
+ except Exception as exc:
832
+ try:
833
+ sys.stderr.write("scrub_file: cannot read %s: %r\n" % (path, exc))
834
+ except Exception:
835
+ pass
836
+ return "", [{"type": "error", "count": 1}]
837
+ return scrub(raw)
838
+
839
+
840
+ # ---------------------------------------------------------------------------
841
+ # Self-test
842
+ # ---------------------------------------------------------------------------
843
+ # A built-in sample containing one of each pattern. Each entry asserts that the
844
+ # raw fragment no longer appears verbatim in the scrubbed output and that the
845
+ # expected finding type fired. Demonstrates correctness without any fixtures.
846
+
847
+ _SELFTEST_CASES = [
848
+ ("email", "contact me at oscar@ecoportal.co.nz today"),
849
+ ("windows_path", r"see C:\Users\rella\.claude\paths.json for config"),
850
+ ("windows_path_fwd", "open C:/claude/Projects/ep-ai-standards/TODO.md"),
851
+ ("posix_home_path", "log lives at /home/oscar/.config/app/secret.cfg"),
852
+ ("posix_users_path", "build dir /Users/oscar/work/build/out"),
853
+ ("mnt_path", "data on /mnt/data/private/dump.sql"),
854
+ ("tilde_home", "config at ~/secrets/keys.txt loaded"),
855
+ ("aws_access_key", "key AKIAIOSFODNN7EXAMPLE used"),
856
+ ("gitlab_pat", "token glpat-ABCDEFGHIJKLMNOPQRSTUV set"),
857
+ ("github_pat", "ghp_ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789 here"),
858
+ ("github_fine_pat",
859
+ "github_pat_11ABCDEFG0abcdefghijklmnopqrstuvwxyz12345 here"),
860
+ ("google_api_key",
861
+ "AIzaSyA1234567890abcdefghijklmnopqrstuv in url"),
862
+ ("google_aq_key",
863
+ "gemini key AQ.Ab8RN6JjFAKEfakefakekey1234567890xyzABCdefg used"),
864
+ ("prefixed_api_key_assignment", "GEMINI_API_KEY=AbCdEfFake123456value"),
865
+ ("suffixed_token_assignment", "TOGGL_TOKEN_RW=zzzzfakevalue1234"),
866
+ ("slack_token", "xoxb-1234567890-abcdefghijklmnop posted"),
867
+ ("jwt",
868
+ "eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxMjM0NTY3ODkwIn0.dozjgNryP4J3jVmNHl0w "
869
+ "is the token"),
870
+ ("bearer_token", "Authorization: Bearer abcdef1234567890XYZ"),
871
+ ("secret_assignment", 'api_key="s3cr3tValue12345"'),
872
+ ("password_assignment", "password = hunter2hunter2"),
873
+ ("ipv4", "host at 192.168.10.254 reachable"),
874
+ ("network_cidr", "subnet 10.0.0.0/8 allocated"),
875
+ ("phone_e164", "call +64 9 123 4567 tomorrow"),
876
+ ("phone_nz_au", "mobile 021 123 4567 for the site"),
877
+ ("atlassian_token", "atlassian ATATT3xFfGF0abcdefghijklmnopqrstuvwxyz012345 set"),
878
+ ("anthropic_key", "claude sk-ant-api03-abcdefghijklmnopqrstuvwxyz0123456789ABCD used"),
879
+ ("openai_key", "openai sk-abcdefghijklmnopqrstuvwxyz0123456789ABCDEFGHIJKL here"),
880
+ ("aws_arn", "role arn:aws:iam::123456789012:role/MyExampleRole attached"),
881
+ ("basic_auth_url", "clone https://guser:s3cretpw@git.example.com/repo.git now"),
882
+ ("high_entropy_hex",
883
+ "digest 0123456789abcdef0123456789abcdef01234567 outside fence"),
884
+ ("high_entropy_base64",
885
+ "blob YWJjZGVmZ2hpamtsbW5vcHFyc3R1dnd4eXoxMjM0NTY3ODkw end"),
886
+ ]
887
+
888
+ # Raw secret fragments that MUST disappear from the scrubbed output.
889
+ _SELFTEST_SECRETS = [
890
+ "oscar@ecoportal.co.nz",
891
+ r"C:\Users\rella",
892
+ "C:/claude/Projects",
893
+ "/home/oscar/",
894
+ "/Users/oscar/",
895
+ "/mnt/data/private",
896
+ "~/secrets/keys.txt",
897
+ "AKIAIOSFODNN7EXAMPLE",
898
+ "glpat-ABCDEFGHIJKLMNOPQRSTUV",
899
+ "ghp_ABCDEFGHIJKLMNOPQRSTUVWXYZ0123456789",
900
+ "github_pat_11ABCDEFG0abcdefghijklmnopqrstuvwxyz12345",
901
+ "AIzaSyA1234567890abcdefghijklmnopqrstuv",
902
+ "AQ.Ab8RN6JjFAKEfakefakekey1234567890xyzABCdefg",
903
+ "AbCdEfFake123456value",
904
+ "zzzzfakevalue1234",
905
+ "xoxb-1234567890-abcdefghijklmnop",
906
+ "eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxMjM0NTY3ODkwIn0.dozjgNryP4J3jVmNHl0w",
907
+ "hunter2hunter2",
908
+ "s3cr3tValue12345",
909
+ "192.168.10.254",
910
+ "10.0.0.0/8",
911
+ "+64 9 123 4567",
912
+ "021 123 4567",
913
+ "ATATT3xFfGF0abcdefghijklmnopqrstuvwxyz012345",
914
+ "sk-ant-api03-abcdefghijklmnopqrstuvwxyz0123456789ABCD",
915
+ "sk-abcdefghijklmnopqrstuvwxyz0123456789ABCDEFGHIJKL",
916
+ "arn:aws:iam::123456789012:role/MyExampleRole",
917
+ "s3cretpw",
918
+ "0123456789abcdef0123456789abcdef01234567",
919
+ "YWJjZGVmZ2hpamtsbW5vcHFyc3R1dnd4eXoxMjM0NTY3ODkw",
920
+ ]
921
+
922
+
923
+ def _selftest():
924
+ """Run built-in assertions. Returns 0 on success, 1 on any failure.
925
+
926
+ Hermetic: the machine-local entity dictionary is disabled so this exercises
927
+ the regex layer in isolation (the dictionary pass has its own tests).
928
+ """
929
+ os.environ["PRIVATE_DATA_DICT_DIR"] = os.path.join(
930
+ os.path.dirname(os.path.abspath(__file__)), "__no_such_dict__")
931
+ sample = "\n".join(desc for _, desc in _SELFTEST_CASES)
932
+ scrubbed, findings = scrub(sample)
933
+
934
+ ok = True
935
+
936
+ # 1. No raw secret fragment survives.
937
+ for secret in _SELFTEST_SECRETS:
938
+ if secret in scrubbed:
939
+ sys.stderr.write("FAIL: secret survived scrub: %r\n" % secret)
940
+ ok = False
941
+
942
+ # 2. The expected finding types all fired.
943
+ fired = {f["type"] for f in findings}
944
+ expected_types = {
945
+ "email", "windows_path", "posix_home_path", "tilde_home_path",
946
+ "aws_access_key", "gitlab_pat", "github_pat", "github_fine_pat",
947
+ "google_api_key", "google_aq_key", "slack_token", "jwt", "bearer_token",
948
+ "secret_assignment", "ipv4", "high_entropy_hex", "high_entropy_base64",
949
+ "atlassian_token", "anthropic_key", "openai_key", "aws_arn",
950
+ "basic_auth_url", "phone", "network_cidr",
951
+ }
952
+ missing = expected_types - fired
953
+ if missing:
954
+ sys.stderr.write("FAIL: finding types never fired: %s\n"
955
+ % ", ".join(sorted(missing)))
956
+ ok = False
957
+
958
+ # 3. Code-fence sample: a hash INSIDE a fence is preserved; a real secret
959
+ # (email) inside a fence is still redacted.
960
+ fenced = (
961
+ "Here is a legitimate hash in a code block:\n"
962
+ "```\n"
963
+ "deadbeefdeadbeefdeadbeefdeadbeefdeadbeef\n"
964
+ "user@example.com\n"
965
+ "```\n"
966
+ "And outside: deadbeefdeadbeefdeadbeefdeadbeefdeadbeef\n"
967
+ )
968
+ fscrub, ffind = scrub(fenced)
969
+ # The in-fence hash should remain (not high-entropy-redacted).
970
+ if "deadbeefdeadbeefdeadbeefdeadbeefdeadbeef" not in fscrub:
971
+ sys.stderr.write("FAIL: in-fence hash was redacted (should be kept)\n")
972
+ ok = False
973
+ # The out-of-fence hash should be redacted.
974
+ if fscrub.count("deadbeefdeadbeefdeadbeefdeadbeefdeadbeef") != 1:
975
+ sys.stderr.write(
976
+ "FAIL: out-of-fence hash not redacted "
977
+ "(in-fence copy should be the only survivor)\n"
978
+ )
979
+ ok = False
980
+ # The in-fence email should still be redacted (targeted patterns ignore fences).
981
+ if "user@example.com" in fscrub:
982
+ sys.stderr.write("FAIL: in-fence email survived (should be redacted)\n")
983
+ ok = False
984
+
985
+ # 4. scrub() never raises and handles odd input.
986
+ for bad in (None, 12345, b"bytes-not-str"):
987
+ try:
988
+ out, _ = scrub(bad)
989
+ if not isinstance(out, str):
990
+ sys.stderr.write("FAIL: scrub(%r) did not return str\n" % (bad,))
991
+ ok = False
992
+ except Exception as exc:
993
+ sys.stderr.write("FAIL: scrub(%r) raised %r\n" % (bad, exc))
994
+ ok = False
995
+
996
+ if ok:
997
+ sys.stderr.write("selftest: OK (%d findings on sample)\n" % len(findings))
998
+ return 0
999
+ sys.stderr.write("selftest: FAILED\n")
1000
+ return 1
1001
+
1002
+
1003
+ def _summarize(findings, stream):
1004
+ if not findings:
1005
+ stream.write("scrub: no findings\n")
1006
+ return
1007
+ stream.write("scrub: findings:\n")
1008
+ total = 0
1009
+ for f in findings:
1010
+ stream.write(" %-22s %d\n" % (f["type"], f["count"]))
1011
+ total += f["count"]
1012
+ stream.write("scrub: %d redaction(s) across %d pattern type(s)\n"
1013
+ % (total, len(findings)))
1014
+
1015
+
1016
+ def _mask_sample(original):
1017
+ """Return a masked, safe-to-print rendering of a raw matched secret.
1018
+
1019
+ Reveals at most the first 2 and last 2 characters and masks the middle with
1020
+ a run of '*', so the preview shows the SHAPE of what was redacted without
1021
+ disclosing the value. Short values (<= 4 chars) are fully masked. Newlines
1022
+ and tabs are collapsed so the sample stays on one line.
1023
+ """
1024
+ s = "".join(" " if c in "\r\n\t" else c for c in (original or ""))
1025
+ n = len(s)
1026
+ if n == 0:
1027
+ return ""
1028
+ if n <= 4:
1029
+ return "*" * n
1030
+ head = s[:2]
1031
+ tail = s[-2:]
1032
+ return "%s%s%s" % (head, "*" * (n - 4), tail)
1033
+
1034
+
1035
+ def _render_preview(scrubbed, findings, matches, stream):
1036
+ """Write a human-readable, NON-leaking redaction report to `stream`.
1037
+
1038
+ Sections:
1039
+ 1. The scrubbed text (already safe -- secrets replaced by placeholders).
1040
+ 2. A findings table: one row per finding TYPE -> count.
1041
+ 3. A per-redaction list with a MASKED sample for each (never the raw value).
1042
+ """
1043
+ line = "=" * 60
1044
+ stream.write(line + "\n")
1045
+ stream.write("SCRUB PREVIEW -- REDACTION REPORT (nothing was sent)\n")
1046
+ stream.write(line + "\n\n")
1047
+
1048
+ stream.write("--- Scrubbed text (safe to read) ---\n")
1049
+ stream.write(scrubbed)
1050
+ if scrubbed and not scrubbed.endswith("\n"):
1051
+ stream.write("\n")
1052
+ stream.write("\n")
1053
+
1054
+ stream.write("--- Findings (type -> count) ---\n")
1055
+ if not findings:
1056
+ stream.write(" (none -- nothing matched)\n")
1057
+ else:
1058
+ total = 0
1059
+ for f in findings:
1060
+ stream.write(" %-22s %d\n" % (f["type"], f["count"]))
1061
+ total += f["count"]
1062
+ stream.write(" %-22s %d redaction(s) across %d type(s)\n"
1063
+ % ("TOTAL", total, len(findings)))
1064
+ stream.write("\n")
1065
+
1066
+ stream.write("--- Redactions (masked samples -- secrets NOT shown) ---\n")
1067
+ if not matches:
1068
+ stream.write(" (none)\n")
1069
+ else:
1070
+ for i, m in enumerate(matches, 1):
1071
+ stream.write(" %3d. %-22s %s (masked: %s)\n"
1072
+ % (i, m["type"], m["replacement"],
1073
+ _mask_sample(m.get("original", ""))))
1074
+ stream.write("\n")
1075
+ stream.write(line + "\n")
1076
+
1077
+
1078
+ def _preview(path):
1079
+ """Render a non-destructive redaction report for `path` to stdout. Exit code
1080
+ matches the scrubbed-text CLI: 2 if scrubbing errored, else 0."""
1081
+ scrubbed, findings, matches = scrub_detailed_file(path)
1082
+ _render_preview(scrubbed, findings, matches, sys.stdout)
1083
+ if any(f["type"] == "error" for f in findings):
1084
+ return 2
1085
+ return 0
1086
+
1087
+
1088
+ def scrub_detailed_file(path):
1089
+ """Read a file (UTF-8, errors replaced) and scrub_detailed() its contents.
1090
+
1091
+ Returns (scrubbed_text, findings, matches). Never raises: a read error is
1092
+ reported as an {"type": "error", "count": 1} finding with no matches.
1093
+ """
1094
+ try:
1095
+ with open(path, "r", encoding="utf-8", errors="replace") as fh:
1096
+ raw = fh.read()
1097
+ except Exception as exc:
1098
+ try:
1099
+ sys.stderr.write("scrub_detailed_file: cannot read %s: %r\n"
1100
+ % (path, exc))
1101
+ except Exception:
1102
+ pass
1103
+ return "", [{"type": "error", "count": 1}], []
1104
+ return scrub_detailed(raw)
1105
+
1106
+
1107
+ def _main(argv):
1108
+ # Windows stdout/stderr default to a locale codepage (cp1252) that cannot encode
1109
+ # non-ASCII content (e.g. a stray glyph in a doc), which crashes the scrubber. Because
1110
+ # the Ruby egress client (gemini_ask.rb) shells out to this and fail-closes on ANY
1111
+ # scrubber error, that crash would silently BLOCK all Gemini egress of non-ASCII
1112
+ # content. Force UTF-8 so the scrubbed text round-trips regardless of platform locale.
1113
+ for _stream in (sys.stdout, sys.stderr):
1114
+ try:
1115
+ _stream.reconfigure(encoding="utf-8")
1116
+ except Exception:
1117
+ pass
1118
+ args = argv[1:]
1119
+ if not args or args[0] in ("-h", "--help"):
1120
+ sys.stderr.write(
1121
+ "usage: python scripts/lib/scrub.py <file>\n"
1122
+ " python scripts/lib/scrub.py --preview <file>\n"
1123
+ " python scripts/lib/scrub.py --selftest\n"
1124
+ )
1125
+ return 0 if args else 1
1126
+ if args[0] == "--selftest":
1127
+ return _selftest()
1128
+ if args[0] == "--preview":
1129
+ if len(args) < 2:
1130
+ sys.stderr.write(
1131
+ "usage: python scripts/lib/scrub.py --preview <file>\n"
1132
+ )
1133
+ return 1
1134
+ return _preview(args[1])
1135
+
1136
+ path = args[0]
1137
+ scrubbed, findings = scrub_file(path)
1138
+ sys.stdout.write(scrubbed)
1139
+ if scrubbed and not scrubbed.endswith("\n"):
1140
+ sys.stdout.write("\n")
1141
+ _summarize(findings, sys.stderr)
1142
+ # Non-zero exit if scrubbing errored, so callers/CI can detect it.
1143
+ if any(f["type"] == "error" for f in findings):
1144
+ return 2
1145
+ return 0
1146
+
1147
+
1148
+ if __name__ == "__main__":
1149
+ sys.exit(_main(sys.argv))