froid-loop 0.11.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- froid_loop/__init__.py +11 -0
- froid_loop/__main__.py +12 -0
- froid_loop/adapters/__init__.py +3 -0
- froid_loop/adapters/base.py +254 -0
- froid_loop/adapters/entrypoints.py +63 -0
- froid_loop/adapters/env_fault.py +290 -0
- froid_loop/adapters/generic.py +2013 -0
- froid_loop/adapters/mock.py +49 -0
- froid_loop/adapters/multiplexer.py +914 -0
- froid_loop/adapters/opencode_http.py +1687 -0
- froid_loop/adapters/profile.py +650 -0
- froid_loop/adapters/psmux_backend.py +1428 -0
- froid_loop/adapters/registry.py +322 -0
- froid_loop/adapters/tmux_backend.py +35 -0
- froid_loop/adapters/tmux_base.py +630 -0
- froid_loop/checks.py +187 -0
- froid_loop/cli.py +5041 -0
- froid_loop/data/__init__.py +0 -0
- froid_loop/data/froid_loop_hook.py +228 -0
- froid_loop/data/froid_loop_probe_hook.py +88 -0
- froid_loop/data/plugins/example/plugin.toml +21 -0
- froid_loop/data/plugins/tea/plugin.toml +184 -0
- froid_loop/data/plugins/tea/tea_plugin.py +258 -0
- froid_loop/data/plugins/unity/plugin.toml +140 -0
- froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef +16 -0
- froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef.meta +7 -0
- froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs +221 -0
- froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs.meta +11 -0
- froid_loop/data/plugins/unity/unity_assets/_folders/Editor.meta +8 -0
- froid_loop/data/plugins/unity/unity_assets/_folders/FroidLoop.meta +8 -0
- froid_loop/data/plugins/unity/unity_cleanup.py +125 -0
- froid_loop/data/plugins/unity/unity_dialog_probe.py +239 -0
- froid_loop/data/plugins/unity/unity_facts.md +17 -0
- froid_loop/data/plugins/unity/unity_plugin.py +415 -0
- froid_loop/data/plugins/unity/unity_quiesce.py +234 -0
- froid_loop/data/plugins/unity/unity_ready.py +230 -0
- froid_loop/data/plugins/unity/unity_seed_assets.py +298 -0
- froid_loop/data/plugins/unity/unity_setup.py +551 -0
- froid_loop/data/plugins/unity/unity_teardown.py +362 -0
- froid_loop/data/profiles/antigravity.toml +52 -0
- froid_loop/data/profiles/claude.toml +85 -0
- froid_loop/data/profiles/codex.toml +22 -0
- froid_loop/data/profiles/copilot.toml +52 -0
- froid_loop/data/profiles/gemini.toml +26 -0
- froid_loop/data/profiles/opencode.toml +54 -0
- froid_loop/data/settings/core.toml +458 -0
- froid_loop/data/skills/README.md +93 -0
- froid_loop/data/skills/froid-loop-resolve/SKILL.md +288 -0
- froid_loop/data/skills/froid-loop-setup/SKILL.md +161 -0
- froid_loop/data/skills/froid-loop-setup/assets/module-help.csv +3 -0
- froid_loop/data/skills/froid-loop-setup/assets/module.yaml +19 -0
- froid_loop/data/skills/froid-loop-sweep/SKILL.md +100 -0
- froid_loop/data/skills/froid-loop-sweep/automation-mode.md +127 -0
- froid_loop/data/skills/froid-loop-sweep/deferred-work-format.md +302 -0
- froid_loop/data/skills/froid-loop-sweep/migration-mode.md +86 -0
- froid_loop/decisions.py +202 -0
- froid_loop/deferredwork.py +2282 -0
- froid_loop/devcontract.py +892 -0
- froid_loop/diagnostics.py +1104 -0
- froid_loop/documents.py +532 -0
- froid_loop/engine.py +7732 -0
- froid_loop/envvars.py +111 -0
- froid_loop/escalation.py +225 -0
- froid_loop/events.py +266 -0
- froid_loop/fences.py +103 -0
- froid_loop/froidconfig.py +226 -0
- froid_loop/frontmatter.py +526 -0
- froid_loop/gates.py +133 -0
- froid_loop/install.py +2936 -0
- froid_loop/journal.py +178 -0
- froid_loop/machine.py +148 -0
- froid_loop/model.py +898 -0
- froid_loop/operatoractions.py +474 -0
- froid_loop/platform_util.py +1490 -0
- froid_loop/plugins/__init__.py +64 -0
- froid_loop/plugins/bus.py +259 -0
- froid_loop/plugins/context.py +319 -0
- froid_loop/plugins/loader.py +145 -0
- froid_loop/plugins/manifest.py +279 -0
- froid_loop/plugins/model.py +296 -0
- froid_loop/plugins/registry.py +245 -0
- froid_loop/plugins/trust.py +75 -0
- froid_loop/policy.py +1569 -0
- froid_loop/probe.py +1044 -0
- froid_loop/process_host.py +408 -0
- froid_loop/recovery_flow.py +1561 -0
- froid_loop/resolve.py +283 -0
- froid_loop/runs.py +4715 -0
- froid_loop/runsetup.py +1293 -0
- froid_loop/sanitize.py +593 -0
- froid_loop/settings_schema.py +276 -0
- froid_loop/signals.py +160 -0
- froid_loop/sprintstatus.py +609 -0
- froid_loop/statemachine.py +57 -0
- froid_loop/stories.py +615 -0
- froid_loop/stories_engine.py +796 -0
- froid_loop/sweep.py +1892 -0
- froid_loop/tokens.py +196 -0
- froid_loop/tui/__init__.py +11 -0
- froid_loop/tui/app.py +1584 -0
- froid_loop/tui/data.py +840 -0
- froid_loop/tui/launch.py +1003 -0
- froid_loop/tui/screens/__init__.py +1 -0
- froid_loop/tui/screens/dashboard.py +1071 -0
- froid_loop/tui/screens/modals.py +943 -0
- froid_loop/tui/screens/settings_screen.py +477 -0
- froid_loop/tui/settings.py +135 -0
- froid_loop/tui/widgets.py +981 -0
- froid_loop/verify.py +4545 -0
- froid_loop/workspace.py +320 -0
- froid_loop/worktree_flow.py +2301 -0
- froid_loop-0.11.1.dist-info/METADATA +728 -0
- froid_loop-0.11.1.dist-info/RECORD +116 -0
- froid_loop-0.11.1.dist-info/WHEEL +4 -0
- froid_loop-0.11.1.dist-info/entry_points.txt +2 -0
- froid_loop-0.11.1.dist-info/licenses/LICENSE +30 -0
froid_loop/sanitize.py
ADDED
|
@@ -0,0 +1,593 @@
|
|
|
1
|
+
"""PII-scrubbing chokepoint for `froid-loop probe-adapter` and `froid-loop diagnose`.
|
|
2
|
+
|
|
3
|
+
Pure stdlib, no froid_loop imports — the single audited place that decides what
|
|
4
|
+
data from a foreign CLI (or a user's own run dir) is safe to show a maintainer.
|
|
5
|
+
Both the probe and the diagnostic-dump commands route every captured payload,
|
|
6
|
+
help/version blob, discovered path, and run-state value through here before
|
|
7
|
+
rendering; nothing is displayed raw.
|
|
8
|
+
|
|
9
|
+
Guarantees:
|
|
10
|
+
- token *counts* are non-PII, so numbers/bools/null pass through verbatim;
|
|
11
|
+
- dict **keys** get the same scrub as leaf strings (a home-path or
|
|
12
|
+
credential-shaped key can't leak where the equivalent value would be caught);
|
|
13
|
+
every leaf **string** is `$HOME`-redacted and then kept ONLY if it matches a
|
|
14
|
+
conservative identifier shape (a short slug with no spaces / `@` / `/`, e.g.
|
|
15
|
+
``claude-opus-4-8`` or ``session-abc_123``); anything else (prose, code,
|
|
16
|
+
paths, emails) becomes ``<redacted:str>``;
|
|
17
|
+
- an identifier-shaped string that looks like a **secret** (a known credential
|
|
18
|
+
prefix such as ``ghp_``/``sk-``/``AKIA``, a JWT, or a long high-entropy blob)
|
|
19
|
+
becomes ``<redacted:secret>`` even though it would otherwise pass — the one
|
|
20
|
+
hole the identifier shape would leave open;
|
|
21
|
+
- list lengths are preserved (the count is structural, the contents aren't);
|
|
22
|
+
- recursion is depth-guarded so a pathological payload can't blow the stack.
|
|
23
|
+
|
|
24
|
+
It also owns the shared egress backstop both commands run over their own
|
|
25
|
+
rendered bytes before emitting: :class:`Pseudonymizer` (stable, irreversible
|
|
26
|
+
per-report aliases for proprietary identifiers — story keys, branches, SHAs,
|
|
27
|
+
project names — that *are* identifier-shaped and so would otherwise survive
|
|
28
|
+
verbatim), :func:`assert_no_leak` (the raw re-scan; accepts labeled extras
|
|
29
|
+
``(value, label)`` so a hit is reported by a printable label instead of an
|
|
30
|
+
opaque index), and :func:`guard` / :func:`assert_clean` (the fail-closed
|
|
31
|
+
policy around it: hard-rule hits — email/secret/home-path/url-creds/username —
|
|
32
|
+
refuse outright by raising :class:`LeakDetected`, while a stray pseudonymizer
|
|
33
|
+
original is repaired via :func:`replace_standalone` under identical
|
|
34
|
+
word-boundary semantics, re-verified, and disclosed to the caller). A routing
|
|
35
|
+
bug or a future field can therefore never silently ship a secret/PII/path.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
# Strict-checked under #245 Stage 2, with the two "expression fully known" rules
|
|
39
|
+
# below relaxed for this file only: `_scrub` walks arbitrary, foreign JSON whose
|
|
40
|
+
# leaves are `Any` by design (it exists to redact unknown/future fields), so
|
|
41
|
+
# isinstance-narrowing that `Any` yields dict/list `[Unknown, ...]` and every
|
|
42
|
+
# key/value it iterates is Unknown. Typing it away would defeat the point of a
|
|
43
|
+
# catch-all scrubber. Every other strict rule stays on.
|
|
44
|
+
# pyright: reportUnknownArgumentType=false, reportUnknownVariableType=false
|
|
45
|
+
|
|
46
|
+
from __future__ import annotations
|
|
47
|
+
|
|
48
|
+
import getpass
|
|
49
|
+
import hashlib
|
|
50
|
+
import math
|
|
51
|
+
import os
|
|
52
|
+
import re
|
|
53
|
+
import secrets
|
|
54
|
+
from collections import Counter
|
|
55
|
+
from typing import Any, Iterable, Iterator
|
|
56
|
+
|
|
57
|
+
# A conservative "this is a machine identifier, not prose or PII" shape: starts
|
|
58
|
+
# alphanumeric, then only word-ish chars (letters, digits, ``.`` ``_`` ``-``),
|
|
59
|
+
# bounded length. No spaces, no ``@``, no ``/`` — so emails, paths, and sentences
|
|
60
|
+
# can never satisfy it. Model ids and session/conversation ids do.
|
|
61
|
+
_IDENTIFIER_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]*$")
|
|
62
|
+
_IDENTIFIER_MAX = 80
|
|
63
|
+
|
|
64
|
+
# Per-line character bound for :func:`scrub_text`. Real ``--version``/``--help``
|
|
65
|
+
# lines from the coding CLIs this probes run well under ~120 characters, so 200
|
|
66
|
+
# is roughly 2x headroom over legitimate output while still bounding an
|
|
67
|
+
# adversarial or corrupted one. Public because `probe.py` passes it explicitly
|
|
68
|
+
# rather than relying on a hidden default.
|
|
69
|
+
SCRUB_TEXT_MAX_CHARS = 200
|
|
70
|
+
|
|
71
|
+
_EMAIL_RE = re.compile(r"[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}")
|
|
72
|
+
|
|
73
|
+
# Known credential token shapes — provider prefixes plus the JWT header. These
|
|
74
|
+
# are exactly the strings that are identifier-shaped (so would pass the slug
|
|
75
|
+
# gate) yet must never be surfaced. Anchored: a value *starting* with one of
|
|
76
|
+
# these is treated as a secret.
|
|
77
|
+
_SECRET_PREFIX_RE = re.compile(
|
|
78
|
+
r"^(?:"
|
|
79
|
+
r"sk-|ghp_|gho_|ghu_|ghs_|ghr_|github_pat_|glpat-|gss_|"
|
|
80
|
+
r"xox[baprs]-|AKIA|ASIA|AIza|ya29\.|AGAPP|hf_|npm_|dop_v1_|sk-ant-"
|
|
81
|
+
r")"
|
|
82
|
+
)
|
|
83
|
+
_JWT_RE = re.compile(r"^eyJ[A-Za-z0-9_-]+\.eyJ[A-Za-z0-9_-]+")
|
|
84
|
+
# Contiguous alphanumeric runs — a UUID/slug breaks into short runs at its
|
|
85
|
+
# hyphens, but a raw API token / hex secret is one long dense run.
|
|
86
|
+
_ALNUM_RUN_RE = re.compile(r"[A-Za-z0-9]+")
|
|
87
|
+
_SECRET_RUN_MIN = 32 # length of contiguous alnum run that triggers the entropy gate
|
|
88
|
+
_SECRET_ENTROPY_MIN = 3.5 # bits/char; pure hex ~4.0, base64 ~6.0, prose/slug well below
|
|
89
|
+
|
|
90
|
+
# Token shape used by assert_no_leak to re-scan rendered output for secrets.
|
|
91
|
+
_LEAK_TOKEN_RE = re.compile(r"[A-Za-z0-9._/+-]{6,}")
|
|
92
|
+
_URL_CRED_RE = re.compile(r"https?://[^/\s]*:[^/@\s]+@")
|
|
93
|
+
# The guard constructs whose match can straddle a truncation point AND whose
|
|
94
|
+
# leading fragment would still be sensitive once the rest is clipped away.
|
|
95
|
+
# `scrub_text`'s cut retracts out of these -- see `_truncate_line`. This sits
|
|
96
|
+
# beside assert_no_leak's rule list deliberately: a rule added there needs an
|
|
97
|
+
# entry here only if a PREFIX of its match stays sensitive. `_EMAIL_RE` is absent
|
|
98
|
+
# because scrub_text redacts emails BEFORE it truncates, so none survives to be
|
|
99
|
+
# split; `_ABS_HOME_RE` is absent because a fragment too short to match `/home/`
|
|
100
|
+
# no longer carries the home tree the rule names.
|
|
101
|
+
_TRUNCATION_HAZARD_RES = (_LEAK_TOKEN_RE, _URL_CRED_RE)
|
|
102
|
+
# The same bytes reach assert_no_leak either as raw text (the markdown report)
|
|
103
|
+
# or as JSON text (the --json document), and json.dumps DOUBLES a backslash —
|
|
104
|
+
# `C:\Users\alice` is serialized as `C:\\Users\\alice`. Matching only the raw
|
|
105
|
+
# form let a Windows home path through the JSON render untouched, so the
|
|
106
|
+
# separator alternates one-or-two backslashes. POSIX prefixes need no such
|
|
107
|
+
# treatment: `/` is not escaped by JSON.
|
|
108
|
+
#
|
|
109
|
+
# The drive-letter arm is for BACKSLASHES only, deliberately: the forward-slash
|
|
110
|
+
# Windows form (`C:/Users/alice`, from Path.as_posix or MSYS-ish tooling) already
|
|
111
|
+
# matches the `/Users/` arm as a substring, as does git-bash's `/c/Users/alice`.
|
|
112
|
+
# Widening the drive-letter arm to `[\\/]` buys only the mixed-separator oddity
|
|
113
|
+
# `C:\Users/alice` — and any string carrying a separator at all is rejected by
|
|
114
|
+
# looks_like_identifier upstream and redacted before it can reach here.
|
|
115
|
+
#
|
|
116
|
+
# The backslash `home`/`root` arm is for the Windows→WSL UNC bridge (#512):
|
|
117
|
+
# `\\wsl.localhost\<distro>\home\<user>`, the legacy `\\wsl$\...`, and the
|
|
118
|
+
# extended-length `\\?\UNC\wsl.localhost\...` folding all reach a POSIX home
|
|
119
|
+
# directory whose separators are BACKSLASHES, so none of the forward-slash arms
|
|
120
|
+
# can see them. That shape needs a path rule for a second reason: the identifier
|
|
121
|
+
# at risk is the *Linux* username, while the username rule in assert_no_leak
|
|
122
|
+
# compares `getpass.getuser()` — the *Windows* account — so on a native-Windows
|
|
123
|
+
# interpreter reaching a distro path, no other rule can fire on the name that
|
|
124
|
+
# matters. The arm anchors on the home segment rather than the `wsl` host
|
|
125
|
+
# because a host anchor misses `\\?\UNC\wsl.localhost\...` entirely: the
|
|
126
|
+
# `?\UNC\` segment sits between the leading backslashes and `wsl`. `Users` is
|
|
127
|
+
# deliberately NOT in this arm — it stays behind the drive-letter arm above,
|
|
128
|
+
# which is the bound the preceding paragraph justifies.
|
|
129
|
+
# tests/test_sanitize.py::test_assert_no_leak_fires pins each arm;
|
|
130
|
+
# ::test_assert_no_leak_home_rule_is_not_any_absolute_path pins the other
|
|
131
|
+
# direction, since a hit makes diagnose refuse to emit at all.
|
|
132
|
+
_ABS_HOME_RE = re.compile(
|
|
133
|
+
r"/home/|/Users/|/root/|[A-Za-z]:\\{1,2}Users\\{1,2}|\\{1,2}(?:home|root)\\{1,2}", re.I
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
_REDACTED_STR = "<redacted:str>"
|
|
137
|
+
_REDACTED_SECRET = "<redacted:secret>" # nosec B105 - redaction marker, not a credential
|
|
138
|
+
_REDACTED_EMAIL = "<redacted:email>"
|
|
139
|
+
_REDACTED_DEPTH = "<redacted:depth>"
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _home() -> str:
|
|
143
|
+
home = os.path.expanduser("~")
|
|
144
|
+
return home if home and home != "~" else ""
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def redact_home(s: str) -> str:
|
|
148
|
+
"""Replace the current user's home directory prefix with ``~``.
|
|
149
|
+
|
|
150
|
+
Catches the literal expanded home (``/home/alice`` -> ``~``); the munged,
|
|
151
|
+
slash-stripped forms some CLIs use for directory names (``-home-alice-...``)
|
|
152
|
+
do not match a path and are handled by the identifier filter instead.
|
|
153
|
+
"""
|
|
154
|
+
home = _home()
|
|
155
|
+
if home and home != "/" and home in s:
|
|
156
|
+
s = s.replace(home, "~")
|
|
157
|
+
return s
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def looks_like_identifier(s: str) -> bool:
|
|
161
|
+
"""True for a short machine slug safe to surface verbatim (no PII)."""
|
|
162
|
+
return 0 < len(s) <= _IDENTIFIER_MAX and bool(_IDENTIFIER_RE.match(s))
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _shannon_entropy(s: str) -> float:
|
|
166
|
+
"""Bits per character — high for random tokens, low for words/slugs."""
|
|
167
|
+
if not s:
|
|
168
|
+
return 0.0
|
|
169
|
+
n = len(s)
|
|
170
|
+
return -sum((c / n) * math.log2(c / n) for c in Counter(s).values())
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def looks_like_secret(s: str) -> bool:
|
|
174
|
+
"""True for a credential-shaped string that must never be surfaced.
|
|
175
|
+
|
|
176
|
+
Catches values that *are* identifier-shaped (so :func:`looks_like_identifier`
|
|
177
|
+
would pass them) but are secrets: a known provider prefix (``ghp_``, ``sk-``,
|
|
178
|
+
``AKIA``, ``xoxb-`` …), a JWT, or a long high-entropy contiguous run (a raw
|
|
179
|
+
API token / hex key). A UUID or hyphenated slug breaks into short runs at its
|
|
180
|
+
separators, so ``claude-opus-4-8`` and ``01234567-89ab-cdef-…`` stay safe."""
|
|
181
|
+
if _SECRET_PREFIX_RE.match(s) or _JWT_RE.match(s):
|
|
182
|
+
return True
|
|
183
|
+
runs = _ALNUM_RUN_RE.findall(s)
|
|
184
|
+
longest = max(runs, key=len) if runs else ""
|
|
185
|
+
return len(longest) >= _SECRET_RUN_MIN and _shannon_entropy(longest) >= _SECRET_ENTROPY_MIN
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _truncate_line(line: str, max_chars: int) -> str:
|
|
189
|
+
"""One line clipped to ``max_chars``, never in a way that blinds the guard.
|
|
190
|
+
|
|
191
|
+
The naive cut is at ``max_chars``. That is unsafe when it lands INSIDE
|
|
192
|
+
something :func:`assert_no_leak` would have flagged, because the fragment it
|
|
193
|
+
leaves behind can keep the sensitive part while no longer tripping the rule
|
|
194
|
+
— turning a fail-closed refusal into an emission. Two shapes reach here:
|
|
195
|
+
|
|
196
|
+
* a credential-shaped token, where the entropy arm of
|
|
197
|
+
:func:`looks_like_secret` needs a contiguous alnum run of
|
|
198
|
+
``_SECRET_RUN_MIN``, so clipping a 36-character token to 30 leaves a
|
|
199
|
+
credential prefix the guard no longer recognizes; and
|
|
200
|
+
* a URL credential, whose match ends at the ``@`` — clip that off and
|
|
201
|
+
``_URL_CRED_RE`` stops matching while the password prefix still ships.
|
|
202
|
+
|
|
203
|
+
Rather than special-case either, the cut retracts out of any
|
|
204
|
+
``_TRUNCATION_HAZARD_RES`` match it splits whose whole trips the guard while
|
|
205
|
+
its surviving prefix does not, using :func:`assert_no_leak` itself as the
|
|
206
|
+
oracle. Deferring to the guard is what keeps this general: it is the same
|
|
207
|
+
verdict the rendered bytes are judged by, so a rule added there is covered
|
|
208
|
+
here without a second implementation of what "sensitive" means.
|
|
209
|
+
|
|
210
|
+
That verdict test is also what keeps the cap from over-firing. ``"x" * 5000``
|
|
211
|
+
is a single 5000-character token, but it trips no rule whole OR clipped, so
|
|
212
|
+
it still truncates at exactly ``max_chars``; a ``ghp_``-prefixed token is
|
|
213
|
+
matched at its start and still trips the guard once clipped, so it is not
|
|
214
|
+
retracted either and the refusal is preserved with no content dropped (#481).
|
|
215
|
+
|
|
216
|
+
What this does NOT cover, deliberately: the guard's rules keyed on values
|
|
217
|
+
only the CALLER knows. ``assert_no_leak`` is consulted here without its
|
|
218
|
+
``extra`` argument, so a caller-supplied sensitive value is not considered,
|
|
219
|
+
and the hazard patterns need six characters, so a standalone username at the
|
|
220
|
+
guard's five-character floor is matched by neither. Splitting one of those
|
|
221
|
+
still costs the guard its verdict. Closing that needs the caller's values
|
|
222
|
+
threaded into this function, which is tracked separately (#654) rather than
|
|
223
|
+
widened into #481 — this bounds the cap to the guard's STATIC rules.
|
|
224
|
+
"""
|
|
225
|
+
if len(line) <= max_chars:
|
|
226
|
+
return line
|
|
227
|
+
cut = max_chars
|
|
228
|
+
# Retracting out of one hazard can land the cut inside an earlier one, so
|
|
229
|
+
# iterate to a fixed point. `cut` only ever decreases, so this terminates.
|
|
230
|
+
changed = True
|
|
231
|
+
while changed:
|
|
232
|
+
changed = False
|
|
233
|
+
for pattern in _TRUNCATION_HAZARD_RES:
|
|
234
|
+
for match in pattern.finditer(line):
|
|
235
|
+
if (
|
|
236
|
+
match.start() < cut < match.end()
|
|
237
|
+
and assert_no_leak(match.group())
|
|
238
|
+
and not assert_no_leak(line[match.start() : cut])
|
|
239
|
+
):
|
|
240
|
+
cut = match.start()
|
|
241
|
+
changed = True
|
|
242
|
+
return line[:cut] + f"… ({len(line) - cut} more chars redacted)"
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def scrub_text(s: str, *, max_lines: int | None = None, max_chars: int | None = None) -> str:
|
|
246
|
+
"""Sanitize free text (a CLI's ``--help`` / ``--version`` / a log tail).
|
|
247
|
+
|
|
248
|
+
Less aggressive than :func:`scrub_json` — help text is the CLI's own and
|
|
249
|
+
flag lines must survive — so we only redact the home dir and any emails,
|
|
250
|
+
then optionally cap the line count and each line's length.
|
|
251
|
+
|
|
252
|
+
``max_chars`` bounds each line's **content** at ``max_chars``; a truncated
|
|
253
|
+
line is emitted as at most ``max_chars`` characters plus the
|
|
254
|
+
``… (N more chars redacted)`` marker, so its total length is at most
|
|
255
|
+
``max_chars + len(marker)`` — "at most" because a cut that would split a
|
|
256
|
+
credential-shaped token retracts to that token's start rather than emit a
|
|
257
|
+
guard-invisible prefix of it (see :func:`_truncate_line`). That overshoot is
|
|
258
|
+
the same off-by-one ``max_lines`` already has — ``max_lines=5`` emits six lines, five kept plus
|
|
259
|
+
the marker line — and it is deliberate (#481): a marker made to fit inside
|
|
260
|
+
the budget would have to be a bare ellipsis, which does not say that
|
|
261
|
+
redaction happened. The line-count marker is never itself char-truncated.
|
|
262
|
+
Composing both caps bounds the whole result at
|
|
263
|
+
``max_lines * (max_chars + len(char marker)) + len(line marker)``.
|
|
264
|
+
|
|
265
|
+
Passing either cap normalizes line endings and drops a trailing newline
|
|
266
|
+
(``splitlines``/``join``); passing neither is a byte-identical passthrough
|
|
267
|
+
of the redaction step.
|
|
268
|
+
"""
|
|
269
|
+
s = redact_home(s)
|
|
270
|
+
s = _EMAIL_RE.sub(_REDACTED_EMAIL, s)
|
|
271
|
+
if max_lines is None and max_chars is None:
|
|
272
|
+
return s
|
|
273
|
+
lines = s.splitlines()
|
|
274
|
+
# Held aside, not appended yet: the line-count marker is a report of what the
|
|
275
|
+
# line cap removed, so char-truncating it would redact the redaction notice.
|
|
276
|
+
line_marker: list[str] = []
|
|
277
|
+
if max_lines is not None and len(lines) > max_lines:
|
|
278
|
+
dropped = len(lines) - max_lines
|
|
279
|
+
line_marker = [f"… ({dropped} more lines redacted)"]
|
|
280
|
+
lines = lines[:max_lines]
|
|
281
|
+
if max_chars is not None:
|
|
282
|
+
lines = [_truncate_line(line, max_chars) for line in lines]
|
|
283
|
+
return "\n".join(lines + line_marker)
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def _is_word_boundary(ch: str) -> bool:
|
|
287
|
+
# a string edge ("") or any non-word char counts as a boundary; "word" = [A-Za-z0-9_]
|
|
288
|
+
return ch == "" or not (ch.isalnum() or ch == "_")
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _iter_standalone(text: str, needle: str) -> Iterator[int]:
|
|
292
|
+
"""Yield start indices of standalone occurrences of an opaque needle: each is
|
|
293
|
+
flanked by a string edge or a non-word char on both sides. Unlike re's ``\\b``
|
|
294
|
+
on the needle's own edges, this still fires when the needle begins or ends
|
|
295
|
+
with punctuation (e.g. ``.acme``). str.find, never regex — needles routinely
|
|
296
|
+
carry regex metachars. Non-overlapping: a hit resumes scanning past the
|
|
297
|
+
needle. Both detection and repair walk this one loop so their boundary
|
|
298
|
+
semantics can never drift apart."""
|
|
299
|
+
start = 0
|
|
300
|
+
while (idx := text.find(needle, start)) >= 0:
|
|
301
|
+
before = text[idx - 1] if idx else ""
|
|
302
|
+
after = text[idx + len(needle)] if idx + len(needle) < len(text) else ""
|
|
303
|
+
if _is_word_boundary(before) and _is_word_boundary(after):
|
|
304
|
+
yield idx
|
|
305
|
+
start = idx + len(needle)
|
|
306
|
+
else:
|
|
307
|
+
start = idx + 1
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def _contains_standalone(text: str, needle: str) -> bool:
|
|
311
|
+
"""True when ``needle`` occurs standalone (word-boundary-flanked) in ``text``."""
|
|
312
|
+
return next(_iter_standalone(text, needle), None) is not None
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def replace_standalone(text: str, needle: str, replacement: str) -> tuple[str, int]:
|
|
316
|
+
"""Replace every standalone occurrence of ``needle``; return ``(text, count)``.
|
|
317
|
+
|
|
318
|
+
Occurrences embedded in a longer word stay untouched — that containment is
|
|
319
|
+
the detection side's deliberate false-positive exclusion (``proj`` inside
|
|
320
|
+
``project``), and this must mirror :func:`_contains_standalone` exactly.
|
|
321
|
+
Scanning consumes past each replaced span, so the replacement is never
|
|
322
|
+
itself rescanned and a replacement containing the needle still terminates."""
|
|
323
|
+
parts: list[str] = []
|
|
324
|
+
pos = 0
|
|
325
|
+
count = 0
|
|
326
|
+
for idx in _iter_standalone(text, needle):
|
|
327
|
+
parts.append(text[pos:idx])
|
|
328
|
+
parts.append(replacement)
|
|
329
|
+
pos = idx + len(needle)
|
|
330
|
+
count += 1
|
|
331
|
+
if not count:
|
|
332
|
+
return text, 0
|
|
333
|
+
parts.append(text[pos:])
|
|
334
|
+
return "".join(parts), count
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def _scrub_str(s: str) -> str:
|
|
338
|
+
"""Sanitize a single string: redact a home path, drop free-form prose, and
|
|
339
|
+
redact a credential-shaped token; an identifier-shaped slug passes verbatim."""
|
|
340
|
+
red = redact_home(s)
|
|
341
|
+
if not looks_like_identifier(red):
|
|
342
|
+
return _REDACTED_STR
|
|
343
|
+
return _REDACTED_SECRET if looks_like_secret(red) else red
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def _scrub(obj: Any, depth: int, max_depth: int) -> Any:
|
|
347
|
+
if depth > max_depth:
|
|
348
|
+
return _REDACTED_DEPTH
|
|
349
|
+
# bool is an int subclass — handled by the numeric branch; both pass through.
|
|
350
|
+
if obj is None or isinstance(obj, (bool, int, float)):
|
|
351
|
+
return obj
|
|
352
|
+
if isinstance(obj, str):
|
|
353
|
+
return _scrub_str(obj)
|
|
354
|
+
if isinstance(obj, dict):
|
|
355
|
+
# Keys get the same scrub as string values, so a home-path or
|
|
356
|
+
# credential-shaped key can't leak where the equivalent value would be
|
|
357
|
+
# caught (diagnostics routes unknown/future fields through here). Two
|
|
358
|
+
# distinct non-identifier keys can collapse to the same <redacted:str>
|
|
359
|
+
# and merge — acceptable under safe-by-default (redaction over fidelity).
|
|
360
|
+
return {_scrub_str(str(k)): _scrub(v, depth + 1, max_depth) for k, v in obj.items()}
|
|
361
|
+
if isinstance(obj, (list, tuple)):
|
|
362
|
+
return [_scrub(v, depth + 1, max_depth) for v in obj]
|
|
363
|
+
# any other type (shouldn't appear in JSON) is treated as an opaque string
|
|
364
|
+
return _REDACTED_STR
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def scrub_json(obj: Any, *, max_depth: int = 40) -> Any:
|
|
368
|
+
"""Recursively sanitize a JSON-shaped value (see module docstring)."""
|
|
369
|
+
return _scrub(obj, 0, max_depth)
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
def scrub_event_payload(payload: Any) -> Any:
|
|
373
|
+
"""Sanitize one captured hook payload — the probe's per-event chokepoint."""
|
|
374
|
+
return scrub_json(payload)
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
class Pseudonymizer:
|
|
378
|
+
"""Stable, irreversible aliases for proprietary identifiers in a dump.
|
|
379
|
+
|
|
380
|
+
Story keys, branch names, spec filenames and SHAs are identifier-shaped, so
|
|
381
|
+
:func:`scrub_json` would pass them verbatim and leak the customer's feature
|
|
382
|
+
names. The dump routes each through :meth:`alias` instead: a per-invocation
|
|
383
|
+
random salt makes the alias unguessable across dumps, while caching makes it
|
|
384
|
+
*stable within* one dump — so a maintainer can see that the story which
|
|
385
|
+
escalated is the same one that later deferred, without ever learning its
|
|
386
|
+
name. The alias is a salted BLAKE2s digest; the salt is never persisted, so
|
|
387
|
+
no map survives that could reverse it.
|
|
388
|
+
|
|
389
|
+
The :meth:`legend` (alias -> original) exists only as a LOCAL convenience for
|
|
390
|
+
the user who generated the dump; it is never written into the shipped report,
|
|
391
|
+
and :func:`assert_no_leak` is fed its values to prove none slipped through.
|
|
392
|
+
"""
|
|
393
|
+
|
|
394
|
+
def __init__(self, salt: bytes | None = None):
|
|
395
|
+
self._salt = salt if salt is not None else secrets.token_bytes(16)
|
|
396
|
+
self._map: dict[tuple[str, str], str] = {}
|
|
397
|
+
self._aliases: dict[str, str] = {} # alias -> original, for collision rejection
|
|
398
|
+
|
|
399
|
+
def alias(self, value: Any, *, ns: str = "id", epic: int | None = None) -> Any:
|
|
400
|
+
"""Map ``value`` to its stable alias. ``None``/empty pass through; a story
|
|
401
|
+
alias prefixes the epic for legibility (``s1-3f2a9c``)."""
|
|
402
|
+
if value is None or value == "":
|
|
403
|
+
return value
|
|
404
|
+
value = str(value)
|
|
405
|
+
key = (ns, value)
|
|
406
|
+
cached = self._map.get(key)
|
|
407
|
+
if cached is not None:
|
|
408
|
+
return cached
|
|
409
|
+
prefix = f"s{epic}" if (ns == "story" and epic is not None) else ns
|
|
410
|
+
# 48-bit digest makes collisions vanishingly unlikely even for large
|
|
411
|
+
# dumps; but a clash is not cosmetic — it would merge two stories'
|
|
412
|
+
# per_alias_event_counts and overwrite a legend() entry — so on the rare
|
|
413
|
+
# collision re-hash with a counter until the alias is free.
|
|
414
|
+
counter = 0
|
|
415
|
+
while True:
|
|
416
|
+
material = self._salt + value.encode("utf-8")
|
|
417
|
+
if counter:
|
|
418
|
+
material += counter.to_bytes(4, "big")
|
|
419
|
+
alias = f"{prefix}-{hashlib.blake2s(material, digest_size=6).hexdigest()}"
|
|
420
|
+
owner = self._aliases.get(alias)
|
|
421
|
+
if owner is None or owner == value:
|
|
422
|
+
break
|
|
423
|
+
counter += 1
|
|
424
|
+
self._aliases[alias] = value
|
|
425
|
+
self._map[key] = alias
|
|
426
|
+
return alias
|
|
427
|
+
|
|
428
|
+
def legend(self) -> dict[str, str]:
|
|
429
|
+
"""alias -> original, for LOCAL use only. Never write this into a dump."""
|
|
430
|
+
return {alias: value for (_, value), alias in self._map.items()}
|
|
431
|
+
|
|
432
|
+
def entries(self) -> list[tuple[str, str, str]]:
|
|
433
|
+
"""``(ns, original, alias)`` triples in insertion order — feeds the
|
|
434
|
+
labeled leak check and alias-substitution repair. Like :meth:`legend`,
|
|
435
|
+
LOCAL ONLY: the originals must never be written into a dump."""
|
|
436
|
+
return [(ns, value, alias) for (ns, value), alias in self._map.items()]
|
|
437
|
+
|
|
438
|
+
|
|
439
|
+
def embeds_current_username(s: str) -> bool:
|
|
440
|
+
"""True when the current username (≥5 chars — the :func:`assert_no_leak`
|
|
441
|
+
hard rule's threshold) appears anywhere in ``s``. Collection-time callers
|
|
442
|
+
redact such values pre-emptively — e.g. a kept-verbatim path component like
|
|
443
|
+
``pytest-of-alice`` — so the guard's username rule (standalone semantics,
|
|
444
|
+
strictly narrower than this substring check) can never fire on them."""
|
|
445
|
+
try:
|
|
446
|
+
user = getpass.getuser()
|
|
447
|
+
except Exception:
|
|
448
|
+
# No passwd entry / no USER env (minimal containers): same degradation
|
|
449
|
+
# as the assert_no_leak username rule.
|
|
450
|
+
return False
|
|
451
|
+
return len(user) >= 5 and user in s
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def assert_no_leak(text: str, *, extra: Iterable[str | tuple[str, str]] = ()) -> list[str]:
|
|
455
|
+
"""Re-scan already-rendered output for anything that must not ship.
|
|
456
|
+
|
|
457
|
+
The defense-in-depth backstop to the per-field routing: even if a handler is
|
|
458
|
+
wrong or a new field is added, this catches an email, a credential-shaped
|
|
459
|
+
token (same logic as :func:`looks_like_secret`), URL-embedded creds, an
|
|
460
|
+
absolute home path, the current username, or any caller-supplied sensitive
|
|
461
|
+
string (e.g. a project basename, or every :meth:`Pseudonymizer.legend` value)
|
|
462
|
+
in the final bytes. Returns the list of rule names that fired — empty means
|
|
463
|
+
clean. Callers fail closed (refuse to write) on a non-empty result.
|
|
464
|
+
|
|
465
|
+
An ``extra`` item is either a bare sensitive value (fires as
|
|
466
|
+
``sensitive[<index>]``) or a ``(value, label)`` pair (fires as
|
|
467
|
+
``sensitive[<label>]``). The label is echoed verbatim into rule names and
|
|
468
|
+
thence CLI output, so callers must NEVER put the sensitive value (or any
|
|
469
|
+
part of it) in the label — diagnostics builds labels from ``ns:alias``
|
|
470
|
+
only, which are safe to print by construction.
|
|
471
|
+
"""
|
|
472
|
+
fired: list[str] = []
|
|
473
|
+
if _EMAIL_RE.search(text):
|
|
474
|
+
fired.append("email")
|
|
475
|
+
if _URL_CRED_RE.search(text):
|
|
476
|
+
fired.append("url-credentials")
|
|
477
|
+
if _ABS_HOME_RE.search(text):
|
|
478
|
+
fired.append("absolute-home-path")
|
|
479
|
+
if any(looks_like_secret(tok) for tok in _LEAK_TOKEN_RE.findall(text)):
|
|
480
|
+
fired.append("secret")
|
|
481
|
+
try:
|
|
482
|
+
user = getpass.getuser()
|
|
483
|
+
except Exception:
|
|
484
|
+
# No passwd entry / no USER env (minimal containers): this one
|
|
485
|
+
# defense-in-depth rule simply can't run; the rest still cover the output.
|
|
486
|
+
user = ""
|
|
487
|
+
if len(user) >= 5 and _contains_standalone(text, user):
|
|
488
|
+
fired.append("username")
|
|
489
|
+
for i, item in enumerate(extra):
|
|
490
|
+
if isinstance(item, tuple):
|
|
491
|
+
value, label = str(item[0]), item[1]
|
|
492
|
+
else:
|
|
493
|
+
# Bare value: report the position only — never echo the value, since
|
|
494
|
+
# this rule name is surfaced in the CLI failure message and would
|
|
495
|
+
# otherwise leak it.
|
|
496
|
+
value, label = str(item), str(i)
|
|
497
|
+
# delimiter check so a short basename ("proj") can't false-positive on a
|
|
498
|
+
# common word that contains it ("project"), yet a value whose own edge is
|
|
499
|
+
# punctuation (".acme") is still caught — a blind spot of a \b regex.
|
|
500
|
+
if len(value) >= 4 and _contains_standalone(text, value):
|
|
501
|
+
fired.append(f"sensitive[{label}]")
|
|
502
|
+
return fired
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
class LeakDetected(Exception):
|
|
506
|
+
"""The rendered report tripped :func:`assert_no_leak` — emission is refused.
|
|
507
|
+
|
|
508
|
+
Raised only for hard rules (email/secret/home-path/url-creds/username) or a
|
|
509
|
+
``sensitive[*]`` repair that did not converge; a plain stray-original hit is
|
|
510
|
+
repaired by alias substitution instead. ``rules`` carries the fired rule
|
|
511
|
+
names — ``sensitive[<ns>:<alias>]`` for pseudonymizer originals, printable
|
|
512
|
+
because the label never contains the original value."""
|
|
513
|
+
|
|
514
|
+
def __init__(self, rules: list[str]):
|
|
515
|
+
self.rules = rules
|
|
516
|
+
super().__init__("rendered report tripped leak self-check: " + ", ".join(rules))
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
_MAX_REPAIR_PASSES = 3
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
def _repair_candidates(pseudo: Pseudonymizer | None) -> list[tuple[str, str, str]]:
|
|
523
|
+
"""``(original, alias, label)`` triples for the leak check and repair.
|
|
524
|
+
|
|
525
|
+
Filtered to assert_no_leak's ≥4-char detection threshold (repair must never
|
|
526
|
+
rewrite an occurrence detection would not fire on) and deduped by original
|
|
527
|
+
(a value aliased under two namespaces gets one deterministic label — the
|
|
528
|
+
first insertion's). Labels are ``ns:alias`` — safe to print by construction,
|
|
529
|
+
never the original."""
|
|
530
|
+
if pseudo is None:
|
|
531
|
+
return []
|
|
532
|
+
candidates: list[tuple[str, str, str]] = []
|
|
533
|
+
seen: set[str] = set()
|
|
534
|
+
for ns, original, alias in pseudo.entries():
|
|
535
|
+
if len(original) >= 4 and original not in seen:
|
|
536
|
+
seen.add(original)
|
|
537
|
+
candidates.append((original, alias, f"{ns}:{alias}"))
|
|
538
|
+
return candidates
|
|
539
|
+
|
|
540
|
+
|
|
541
|
+
def assert_clean(rendered: str, pseudo: Pseudonymizer | None = None) -> None:
|
|
542
|
+
"""One plain, no-repair re-check — run after a repair note is appended so
|
|
543
|
+
the note itself sits inside the verified bytes."""
|
|
544
|
+
extras = [(orig, label) for orig, _alias, label in _repair_candidates(pseudo)]
|
|
545
|
+
fired = assert_no_leak(rendered, extra=extras)
|
|
546
|
+
if fired:
|
|
547
|
+
raise LeakDetected(fired)
|
|
548
|
+
|
|
549
|
+
|
|
550
|
+
def guard(
|
|
551
|
+
rendered: str,
|
|
552
|
+
pseudo: Pseudonymizer | None = None,
|
|
553
|
+
*,
|
|
554
|
+
max_passes: int = _MAX_REPAIR_PASSES,
|
|
555
|
+
) -> tuple[str, list[tuple[str, int]]]:
|
|
556
|
+
"""Verify the rendered bytes; repair stray pseudonymized originals; fail closed.
|
|
557
|
+
|
|
558
|
+
When the pseudonymizer is supplied, its legend's original values (the real
|
|
559
|
+
story keys/branches/project names) are fed into the self-check too, so any
|
|
560
|
+
that slipped through the per-field routing are caught here in the final
|
|
561
|
+
bytes. Unlike a hard-rule hit, such a miss is repairable: the backstop knows
|
|
562
|
+
the original's safe alias, so it substitutes it and re-verifies instead of
|
|
563
|
+
refusing outright. Returns ``(text, [(label, count), ...])`` of applied
|
|
564
|
+
repairs. Raises :class:`LeakDetected` on any hard rule (email / secret /
|
|
565
|
+
home-path / url-creds / username — genuine PII never auto-repairs) or if
|
|
566
|
+
repair does not converge within the pass bound."""
|
|
567
|
+
candidates = _repair_candidates(pseudo)
|
|
568
|
+
extras = [(orig, label) for orig, _alias, label in candidates]
|
|
569
|
+
# Longest-first: a branch embedding a story slug at a "-" boundary must be
|
|
570
|
+
# replaced whole, not spliced into a half-alias mongrel by the inner slug.
|
|
571
|
+
by_length = sorted(candidates, key=lambda c: len(c[0]), reverse=True)
|
|
572
|
+
|
|
573
|
+
tally: dict[str, int] = {}
|
|
574
|
+
for _ in range(max_passes):
|
|
575
|
+
fired = assert_no_leak(rendered, extra=extras)
|
|
576
|
+
if not fired:
|
|
577
|
+
return rendered, sorted(tally.items())
|
|
578
|
+
if any(not rule.startswith("sensitive[") for rule in fired):
|
|
579
|
+
raise LeakDetected(fired)
|
|
580
|
+
# A repair pass replaces every standalone occurrence of every candidate,
|
|
581
|
+
# so pass 2 is reachable only if a substitution manufactured a NEW
|
|
582
|
+
# standalone occurrence of a different original — a hash-output
|
|
583
|
+
# coincidence (the alias alphabet is [A-Za-z0-9-] and "-" is itself a
|
|
584
|
+
# boundary char). The bound turns a pathological substitution cycle
|
|
585
|
+
# into a fail-closed refusal instead of a loop.
|
|
586
|
+
for original, alias, label in by_length:
|
|
587
|
+
rendered, n = replace_standalone(rendered, original, alias)
|
|
588
|
+
if n:
|
|
589
|
+
tally[label] = tally.get(label, 0) + n
|
|
590
|
+
fired = assert_no_leak(rendered, extra=extras)
|
|
591
|
+
if fired:
|
|
592
|
+
raise LeakDetected(fired)
|
|
593
|
+
return rendered, sorted(tally.items())
|