coherence-check 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
coherence/__init__.py ADDED
@@ -0,0 +1,26 @@
1
+ """Coherence — one law, many costumes.
2
+
3
+ Nothing is done unless there is evidence.
4
+ Nothing is finished unless there is a next.
5
+ Nothing is remembered unless it was done.
6
+ """
7
+
8
+ from coherence.core.fact import LAW, Fact, FactError, FactKind
9
+ from coherence.core.spine import Coherence
10
+ from coherence.core.types import Artifact, Bundle, Record, Truth
11
+ from coherence.evolve import DominoChain, EvolutionMemory
12
+
13
+ __all__ = [
14
+ "Coherence",
15
+ "Fact",
16
+ "FactError",
17
+ "FactKind",
18
+ "LAW",
19
+ "Bundle",
20
+ "Record",
21
+ "Artifact",
22
+ "Truth",
23
+ "DominoChain",
24
+ "EvolutionMemory",
25
+ ]
26
+ __version__ = "0.5.1"
coherence/__main__.py ADDED
@@ -0,0 +1,437 @@
1
+ """python -m coherence demo|evolve|law|prove-cmd|check|report"""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import sys
8
+ import tempfile
9
+ from pathlib import Path
10
+
11
+ from coherence import Coherence
12
+ from coherence.ci.session import (
13
+ DEFAULT_SESSION,
14
+ SessionStore,
15
+ build_report,
16
+ check_exit_code,
17
+ report_markdown,
18
+ )
19
+
20
+
21
+ def law() -> int:
22
+ from coherence.core.fact import LAW
23
+
24
+ print()
25
+ print("COHERENCE · the law")
26
+ print("=" * 50)
27
+ print(LAW)
28
+ print()
29
+ print("Atom: Fact(claim, evidence, next)")
30
+ print(" done ⇔ evidence non-empty")
31
+ print(" finished ⇔ next non-empty (always)")
32
+ print(" remember ⇔ done only")
33
+ print()
34
+ print("Everything else is a view over the same Fact.")
35
+ print("Docs: README.md")
36
+ return 0
37
+
38
+
39
+ def demo() -> int:
40
+ print()
41
+ print("COHERENCE — one law, five kinds of Fact")
42
+ print("=" * 50)
43
+ from coherence.core.fact import LAW
44
+
45
+ print(LAW)
46
+ print()
47
+
48
+ c = Coherence(title="demo-pr-agent-fix")
49
+ c.skills.audit("web-search-skill", ["network", "read"], source="marketplace:example")
50
+ c.skills.audit("shell-runner", ["shell", "filesystem_write"], source="random-gist")
51
+ c.decisions.lock(
52
+ "auth-boundary",
53
+ "MUST NOT rewrite auth without human; tests must stay green",
54
+ by="staff-eng",
55
+ )
56
+ c.claimproof.illusion("unit tests", "agent said tests passed in chat")
57
+ c.claimproof.claim("refactored utils", "looks cleaner", next_action="add proof")
58
+ c.claimproof.cmd("unit tests", "pytest -q", exit_code=0)
59
+ c.claimproof.cmd("typecheck", "mypy src", exit_code=1)
60
+ c.decisions.check_violation("auth-boundary", "I did not rewrite auth; only utils")
61
+ c.replay.check()
62
+ triage = c.review.triage()
63
+ guard = c.require_coherent()
64
+
65
+ print(c.plain_english())
66
+ print("summary:", c.summary())
67
+ print(f"triage priority: {triage.meta.get('priority')}")
68
+ print(f"coherence guard: {guard.summary if guard else 'ok'}")
69
+ print("PASS — rungs 1–5 share one Bundle")
70
+ return 0
71
+
72
+
73
+ def evolve_demo() -> int:
74
+ print()
75
+ print("COHERENCE EVOLVE — dominos + next steps + memory")
76
+ print("=" * 56)
77
+
78
+ mem = Path(tempfile.gettempdir()) / "coherence_evolve_demo.json"
79
+ if mem.exists():
80
+ mem.unlink()
81
+
82
+ c1 = Coherence(title="session-1", memory_path=mem, seed_cascade=True)
83
+ print(c1.dominos.plain_english())
84
+ print()
85
+
86
+ head = c1.dominos.head()
87
+ assert head is not None
88
+ print(f"ACTIVE: {head.title}")
89
+ print(f"GILBERT NEXT: {head.next_action}")
90
+ print()
91
+
92
+ proof_rec = c1.claimproof.cmd("unit tests", "pytest -q", exit_code=0)
93
+ c1.solve_domino(
94
+ head.id,
95
+ proof=f"cmd proven {proof_rec.proven}",
96
+ lesson="Never trust chat 'tests passed' — require cmd_exit artifact",
97
+ proof_record_id=proof_rec.id,
98
+ tags=["tests", "illusion"],
99
+ )
100
+ print("After knock #1:")
101
+ print(c1.dominos.plain_english())
102
+ print()
103
+ print(c1.evolve.plain_english())
104
+ print()
105
+
106
+ head2 = c1.dominos.head()
107
+ assert head2 is not None
108
+ sk = c1.skills.audit("shell-runner", ["shell"], source="gist")
109
+ c1.solve_domino(
110
+ head2.id,
111
+ proof=f"skill bill {sk.proven}",
112
+ lesson="High-risk skills must surface on the same report as test proof",
113
+ proof_record_id=sk.id,
114
+ tags=["skills"],
115
+ )
116
+
117
+ print("Session 1 summary:", c1.summary())
118
+ print()
119
+
120
+ c2 = Coherence(title="session-2", memory_path=mem)
121
+ print("SESSION 2 — evolution memory loaded")
122
+ print(c2.evolve.plain_english())
123
+ hints = c2.evolve.apply_hints("agent said tests passed in chat")
124
+ print(f"hints for 'tests passed in chat': {len(hints)}")
125
+ if hints:
126
+ print(f" lesson: {hints[0].lesson}")
127
+ print(f" GILBERT NEXT from memory: {hints[0].next_domino}")
128
+ print()
129
+ print("stats:", c2.evolve.stats())
130
+ print()
131
+ print("PASS — use → solve → learn → next session more coherent")
132
+ print("Docs: docs/EVOLUTION-AND-DOMINOS.md")
133
+ return 0
134
+
135
+
136
+ def cmd_prove(argv: list[str]) -> int:
137
+ p = argparse.ArgumentParser(prog="coherence prove-cmd")
138
+ p.add_argument("command", help="shell command to run and prove")
139
+ p.add_argument("--session", default=str(DEFAULT_SESSION))
140
+ p.add_argument("--claim", default="")
141
+ p.add_argument(
142
+ "--next",
143
+ default="close remaining open facts or chain complete",
144
+ dest="next_action",
145
+ )
146
+ args = p.parse_args(argv)
147
+ store = SessionStore(args.session)
148
+ c, fact, code = store.prove_command(
149
+ args.command,
150
+ claim=args.claim or None,
151
+ next_action=args.next_action,
152
+ )
153
+ print(fact.plain_english())
154
+ print(f"session: {store.path} facts_done={len(c.done_facts())} open={len(c.open_facts())}")
155
+ # CI: non-zero if command failed
156
+ return 0 if code == 0 else 1
157
+
158
+
159
+ def cmd_said(argv: list[str]) -> int:
160
+ p = argparse.ArgumentParser(prog="coherence said")
161
+ p.add_argument("claim")
162
+ p.add_argument("--next", required=True, dest="next_action")
163
+ p.add_argument("--session", default=str(DEFAULT_SESSION))
164
+ args = p.parse_args(argv)
165
+ store = SessionStore(args.session)
166
+ c = store.load()
167
+ f = c.said(args.claim, args.next_action)
168
+ store.save(c)
169
+ print(f.plain_english())
170
+ return 0
171
+
172
+
173
+ def cmd_check(argv: list[str]) -> int:
174
+ p = argparse.ArgumentParser(prog="coherence check")
175
+ p.add_argument("--session", default=str(DEFAULT_SESSION))
176
+ p.add_argument(
177
+ "--strict",
178
+ action="store_true",
179
+ default=True,
180
+ help="fail if zero proven facts (default)",
181
+ )
182
+ p.add_argument("--no-strict", action="store_true", help="allow empty session")
183
+ args = p.parse_args(argv)
184
+ strict = not args.no_strict
185
+ store = SessionStore(args.session)
186
+
187
+ # Integrity FIRST. If the session file was edited after the facts were
188
+ # recorded, nothing else it says can be trusted — so a tampered chain is a
189
+ # hard failure (exit 3) before any fact is read. This is the guard that
190
+ # makes "agents can't fake green" true for the case that actually matters:
191
+ # the policed agent editing its own session on disk.
192
+ integ = store.verify()
193
+ if integ["status"] == "tampered":
194
+ print(json.dumps({"ok": False, "integrity": integ}, indent=2))
195
+ print(f"CHECK FAIL: session tampered at entry {integ.get('position')} "
196
+ f"— {integ.get('detail')}", file=sys.stderr)
197
+ return 3
198
+
199
+ c = store.load()
200
+ report = build_report(c)
201
+ report["integrity"] = integ
202
+ print(json.dumps(report, indent=2))
203
+ code = check_exit_code(c, strict=strict)
204
+ if code == 0:
205
+ note = "" if integ["status"] == "ok" else f" ({integ['status']})"
206
+ print(f"CHECK PASS{note}", file=sys.stderr)
207
+ elif code == 2:
208
+ print("CHECK FAIL: no proven facts (strict)", file=sys.stderr)
209
+ else:
210
+ print("CHECK FAIL: open or blocked facts remain", file=sys.stderr)
211
+ return code
212
+
213
+
214
+ def cmd_scope(argv: list[str]) -> int:
215
+ """Blast radius: what the agent touched, and where this report's edge is."""
216
+ p = argparse.ArgumentParser(prog="coherence scope")
217
+ p.add_argument("transcript")
218
+ p.add_argument("--json", action="store_true", dest="as_json")
219
+ p.add_argument("--full", action="store_true", help="list every item, not a sample")
220
+ args = p.parse_args(argv)
221
+ from coherence.audit.scope import scope_transcript
222
+ sc = scope_transcript(args.transcript)
223
+ if args.as_json:
224
+ print(json.dumps({
225
+ "commands": sc.commands, "bounded": sc.bounded(),
226
+ "files": sorted(sc.files), "pushes": sorted(sc.pushes),
227
+ "hosts": sorted(sc.hosts), "installs": sorted(sc.installs),
228
+ "opaque": [{"line": a, "command": b, "why": c} for a, b, c in sc.opaque],
229
+ }, indent=2))
230
+ return sc.exit_code()
231
+
232
+ def show(label, items):
233
+ items = sorted(items)
234
+ print(f"\n {label} ({len(items)})")
235
+ for i in (items if args.full else items[:8]):
236
+ print(f" {i}")
237
+ if not args.full and len(items) > 8:
238
+ print(f" … {len(items) - 8} more (--full)")
239
+
240
+ print(f"blast radius of {sc.commands} commands\n")
241
+ show("files touched", sc.files)
242
+ show("repos pushed", sc.pushes)
243
+ show("network hosts contacted", sc.hosts)
244
+ show("packages installed", sc.installs)
245
+
246
+ print(f"\n OPAQUE — effects this report CANNOT see ({len(sc.opaque)})")
247
+ for seq, cmd, why in (sc.opaque if args.full else sc.opaque[:8]):
248
+ print(f" line {seq}: {cmd}")
249
+ print(f" why: {why}")
250
+ if not args.full and len(sc.opaque) > 8:
251
+ print(f" … {len(sc.opaque) - 8} more (--full)")
252
+
253
+ if sc.bounded():
254
+ print(f"\nVERDICT: BOUNDED — {len(sc.opaque)} command(s) could do anything this")
255
+ print("report cannot see. Everything above is what IS visible, not everything")
256
+ print("that happened. Unknown is reported as unknown, never as 'nothing'.")
257
+ else:
258
+ print("\nVERDICT: fully readable — every command's effects were determinable.")
259
+ return sc.exit_code()
260
+
261
+
262
+ def cmd_audit(argv: list[str]) -> int:
263
+ """Audit an agent transcript: every checkable claim vs. what actually ran."""
264
+ p = argparse.ArgumentParser(prog="coherence audit")
265
+ p.add_argument("transcript", help="agent session .jsonl (Claude Code format)")
266
+ p.add_argument("--json", action="store_true", dest="as_json")
267
+ args = p.parse_args(argv)
268
+ from coherence.audit.transcript import (
269
+ audit_transcript, SUPPORTED, WEAK, UNSUPPORTED, CONTRADICTED)
270
+ a = audit_transcript(args.transcript)
271
+ c = a.counts()
272
+ if args.as_json:
273
+ print(json.dumps({
274
+ "commands": a.commands, "claims": len(a.claims), "counts": c,
275
+ "findings": [vars(x) for x in a.claims
276
+ if x.verdict in (UNSUPPORTED, CONTRADICTED, WEAK)],
277
+ }, indent=2))
278
+ return a.exit_code()
279
+ print(f"audited: {a.commands} commands, {len(a.claims)} checkable claims\n")
280
+ print(f" supported {c[SUPPORTED]}")
281
+ print(f" weak evidence {c[WEAK]} (piped exit codes — pytest | tail class)")
282
+ print(f" unsupported {c[UNSUPPORTED]} (claims resting on nothing)")
283
+ print(f" CONTRADICTED {c[CONTRADICTED]} (claimed success; its own transcript says failure)")
284
+ for x in a.claims:
285
+ if x.verdict == CONTRADICTED:
286
+ print(f"\n LIE at line {x.seq} [{x.kind}]: \"{x.text}\"")
287
+ print(f" evidence against: {x.evidence}")
288
+ for x in a.claims:
289
+ if x.verdict == UNSUPPORTED:
290
+ print(f"\n unsupported line {x.seq} [{x.kind}]: \"{x.text[:100]}\"")
291
+ print(f"\nexit {a.exit_code()} (0 all supported · 1 unsupported · 2 contradicted)")
292
+ return a.exit_code()
293
+
294
+
295
+ def cmd_tamper_demo(argv: list[str]) -> int:
296
+ """Show, in one command, the thing that is hard to believe from prose:
297
+ a forged 'proven' in the session file is caught.
298
+
299
+ Runs entirely in a throwaway temp directory — touches nothing of yours.
300
+ """
301
+ import tempfile
302
+ from coherence.ci.session import SessionStore
303
+
304
+ with tempfile.TemporaryDirectory() as td:
305
+ sess = Path(td) / "session.json"
306
+ store = SessionStore(sess)
307
+
308
+ print("1. An agent runs a check. It FAILS — so no proof is recorded.")
309
+ store.prove_command("false", claim="test suite")
310
+ print(f" coherence check -> exit {cmd_exit(sess)} (open fact, no evidence)\n")
311
+
312
+ print("2. The agent edits its own session file to claim it passed.")
313
+ data = json.loads(sess.read_text())
314
+ data["facts"][0]["evidence"] = "exit_code=0 output_digest=deadbeef"
315
+ data["facts"][0]["next"] = "chain complete"
316
+ sess.write_text(json.dumps(data, indent=2))
317
+ print(' forged: evidence = "exit_code=0 ..."\n')
318
+
319
+ print("3. The check runs again. The hash chain does not match.")
320
+ code = cmd_exit(sess)
321
+ v = store.verify()
322
+ print(f" coherence check -> exit {code} {v['status'].upper()} at entry {v.get('position')}")
323
+ print(f" {v.get('detail')}\n")
324
+
325
+ if code != 3:
326
+ print("UNEXPECTED: tampering was not caught", file=sys.stderr)
327
+ return 1
328
+ print("A forged green is caught. That is the whole idea.")
329
+ print("Exit codes: 0 pass · 1 open facts · 2 empty · 3 tampered")
330
+ return 0
331
+
332
+
333
+ def cmd_exit(session: Path) -> int:
334
+ """Run the real check logic quietly and return only its exit code."""
335
+ from coherence.ci.session import SessionStore, check_exit_code
336
+ store = SessionStore(session)
337
+ integ = store.verify()
338
+ if integ["status"] == "tampered":
339
+ return 3
340
+ return check_exit_code(store.load(), strict=True)
341
+
342
+
343
+ def cmd_report(argv: list[str]) -> int:
344
+ p = argparse.ArgumentParser(prog="coherence report")
345
+ p.add_argument("--session", default=str(DEFAULT_SESSION))
346
+ p.add_argument("--json", action="store_true")
347
+ p.add_argument("--out", default="", help="write markdown or json to file")
348
+ args = p.parse_args(argv)
349
+ store = SessionStore(args.session)
350
+ c = store.load()
351
+ if args.json:
352
+ body = json.dumps(build_report(c), indent=2)
353
+ else:
354
+ body = report_markdown(c)
355
+ if args.out:
356
+ Path(args.out).write_text(body + "\n", encoding="utf-8")
357
+ print(f"wrote {args.out}")
358
+ else:
359
+ print(body)
360
+ return 0
361
+
362
+
363
+ def main(argv: list[str] | None = None) -> None:
364
+ argv = list(sys.argv[1:] if argv is None else argv)
365
+ if not argv:
366
+ raise SystemExit(demo())
367
+ cmd = argv[0].lower()
368
+ rest = argv[1:]
369
+ if cmd in ("demo", "run"):
370
+ raise SystemExit(demo())
371
+ if cmd in ("evolve", "domino", "flywheel"):
372
+ raise SystemExit(evolve_demo())
373
+ if cmd in ("law", "320", "iq"):
374
+ raise SystemExit(law())
375
+ if cmd in ("health", "doctor", "status"):
376
+ from coherence.health import plain_english, run_health, write_report
377
+
378
+ p = argparse.ArgumentParser(prog="coherence health")
379
+ p.add_argument("--out", default="", help="write JSON report path")
380
+ p.add_argument("--memory", default="", help="optional evolution memory to verify")
381
+ p.add_argument("--no-storm", action="store_true")
382
+ args = p.parse_args(rest)
383
+ report = run_health(
384
+ include_storm=not args.no_storm,
385
+ memory_path=Path(args.memory) if args.memory else None,
386
+ )
387
+ print(plain_english(report))
388
+ if args.out:
389
+ write_report(report, Path(args.out))
390
+ print(f"wrote {args.out}")
391
+ raise SystemExit(0 if report.ok else 1)
392
+ if cmd in ("storm", "proof", "storm-proof"):
393
+ # Hostile proof harness (EffectFence/Seal style)
394
+ from pathlib import Path as _P
395
+ import runpy
396
+
397
+ storm = _P(__file__).resolve().parents[2] / "storm.py"
398
+ if not storm.exists():
399
+ # installed wheel: look beside package or cwd
400
+ storm = _P.cwd() / "storm.py"
401
+ if storm.exists():
402
+ raise SystemExit(runpy.run_path(str(storm), run_name="__main__") or 0)
403
+ print("storm.py not found — run from repo: python storm.py", file=sys.stderr)
404
+ raise SystemExit(2)
405
+ if cmd in ("prove-cmd", "prove_cmd", "prove"):
406
+ raise SystemExit(cmd_prove(rest))
407
+ if cmd == "said":
408
+ raise SystemExit(cmd_said(rest))
409
+ if cmd == "check":
410
+ raise SystemExit(cmd_check(rest))
411
+ if cmd == "scope":
412
+ raise SystemExit(cmd_scope(rest))
413
+ if cmd == "audit":
414
+ raise SystemExit(cmd_audit(rest))
415
+ if cmd in ("tamper-demo", "tamper_demo", "tamper"):
416
+ raise SystemExit(cmd_tamper_demo(rest))
417
+ if cmd == "report":
418
+ raise SystemExit(cmd_report(rest))
419
+ if cmd in ("-h", "--help", "help"):
420
+ print(
421
+ "Usage: python -m coherence <command>\n"
422
+ " law | demo | evolve | storm | health\n"
423
+ " said CLAIM --next NEXT\n"
424
+ " prove-cmd 'pytest -q'\n"
425
+ " tamper-demo (10s: forge a green, watch it get caught)\n"
426
+ " audit FILE.jsonl (agent transcript: claims vs. what actually ran)\n"
427
+ " scope FILE.jsonl (blast radius: what it touched + what we cannot see)\n"
428
+ " check [--no-strict]\n"
429
+ " report [--json] [--out file.md]\n"
430
+ )
431
+ raise SystemExit(0)
432
+ print(f"Unknown command: {cmd}. Try: law | demo | prove-cmd | check | report", file=sys.stderr)
433
+ raise SystemExit(2)
434
+
435
+
436
+ if __name__ == "__main__":
437
+ main()
File without changes
@@ -0,0 +1,156 @@
1
+ """Blast radius: what did the agent touch — and what can't we tell?
2
+
3
+ `audit` answers "did it do what it said?". This answers the harder twin:
4
+ "did it do anything it did NOT say?" — the 3am question for anyone who lets an
5
+ agent run unattended, and the one every security review asks as "show me
6
+ everything the agent touched."
7
+
8
+ Proving a negative from a log is a logic trap: absence of evidence is not
9
+ evidence of absence. A single `bash deploy.sh` can hide any effect in the
10
+ world. The usual responses are both wrong:
11
+
12
+ * claim completeness anyway (a lie that gets someone breached), or
13
+ * demand syscall-level sandboxing, ship nothing, and leave the question
14
+ unanswered forever — which is where the industry actually sits.
15
+
16
+ This module takes the third road, and it is the same rule Seal's witness is
17
+ built on: **UNKNOWN never collapses into ABSENT.** Commands whose effects
18
+ cannot be determined statically are not skipped and not guessed at — they are
19
+ reported as OPAQUE, counted, and printed, and the verdict states plainly that
20
+ the report is bounded by them. A bounded answer that says where its edge is
21
+ beats both a false complete one and no answer at all.
22
+
23
+ What is extracted (best effort, from the transcript already on disk):
24
+ files written · repos pushed · network hosts contacted · packages installed
25
+
26
+ What makes a command OPAQUE:
27
+ running a script file, `eval`, command substitution, piping a download into
28
+ a shell, `source`, or a base64 decode into execution — anything whose real
29
+ effects live somewhere this file cannot see.
30
+ """
31
+ from __future__ import annotations
32
+
33
+ import json
34
+ import re
35
+ from dataclasses import dataclass, field
36
+ from pathlib import Path
37
+
38
+ from .transcript import _events # same parser, one source of truth
39
+
40
+ # ── effects we CAN read directly off the command line ────────────────────
41
+ _WRITE_TOOLS = {"Write", "Edit", "NotebookEdit", "MultiEdit"}
42
+
43
+ _NET = re.compile(
44
+ r"\b(?:curl|wget|http|https)\b[^\n|;]*?"
45
+ r"(?:https?://)?([a-z0-9.-]+\.[a-z]{2,})(?:[/:\s]|$)", re.I)
46
+ _GH_API = re.compile(r"\bgh\s+(?:api|repo|pr|issue|release|run|search)\b", re.I)
47
+ _PUSH = re.compile(r"\bgit\s+push\b[^\n;&|]*", re.I)
48
+ _INSTALL = re.compile(
49
+ r"\b(?:pip|pip3|python3?\s+-m\s+pip)\s+install\s+([^\n;&|]+)"
50
+ r"|\bnpm\s+(?:i|install|publish)\s*([^\n;&|]*)"
51
+ r"|\bcargo\s+(?:install|publish)\s*([^\n;&|]*)"
52
+ r"|\bbrew\s+install\s+([^\n;&|]+)", re.I)
53
+ _REDIRECT = re.compile(r"(?<![>\d])>{1,2}\s*([^\s&|;]+)")
54
+ _FILE_MUT = re.compile(
55
+ r"\b(?:rm|mv|cp|chmod|chown|touch|mkdir|ln)\s+(-[^\s]*\s+)*([^\n;&|]+)", re.I)
56
+
57
+ # ── what we deliberately CANNOT read: the honest edge ────────────────────
58
+ _OPAQUE_RULES = [
59
+ (re.compile(r"\beval\b"), "eval — constructs code at runtime"),
60
+ (re.compile(r"\bsource\s+|\.\s+/"), "source — runs another file's contents"),
61
+ (re.compile(r"(?:curl|wget)[^\n]*\|\s*(?:ba|z|)sh\b"),
62
+ "download piped into a shell"),
63
+ (re.compile(r"\bbase64\s+(?:-d|--decode)[^\n]*\|"),
64
+ "base64 decoded into execution"),
65
+ (re.compile(r"\b(?:ba|z|)sh\s+[^\s-][^\n;&|]*\.(?:sh|bash)\b"),
66
+ "runs a script file"),
67
+ (re.compile(r"\b(?:python3?|node|ruby|perl)\s+[^\s-][^\n;&|]*\.(?:py|js|rb|pl)\b"),
68
+ "runs a program file"),
69
+ (re.compile(r"\$\((?!\s*(?:pwd|date|basename|dirname|echo)\b)"),
70
+ "command substitution"),
71
+ (re.compile(r"\bmake\b|\bnpm\s+run\b|\byarn\s+\w+"),
72
+ "delegates to a script target"),
73
+ (re.compile(r"\bdocker\s+run\b|\bssh\b"), "executes in another environment"),
74
+ ]
75
+
76
+ # hosts that are noise, not reach: the agent's own machine
77
+ _LOCAL = re.compile(r"^(?:localhost|127\.0\.0\.1|0\.0\.0\.0|\[::1\])")
78
+
79
+
80
+ @dataclass
81
+ class Scope:
82
+ files: set = field(default_factory=set)
83
+ pushes: set = field(default_factory=set)
84
+ hosts: set = field(default_factory=set)
85
+ installs: set = field(default_factory=set)
86
+ opaque: list = field(default_factory=list) # (seq, command, why)
87
+ commands: int = 0
88
+
89
+ def bounded(self) -> bool:
90
+ """True when opaque commands exist — the report has a known edge."""
91
+ return bool(self.opaque)
92
+
93
+ def exit_code(self) -> int:
94
+ """0 = fully readable · 1 = readable but BOUNDED by opaque commands.
95
+
96
+ Never 0 while anything is opaque: a report that cannot see everything
97
+ must not exit like one that can.
98
+ """
99
+ return 1 if self.opaque else 0
100
+
101
+
102
+ def _clean_path(p: str) -> str | None:
103
+ p = p.strip().strip("\"'")
104
+ if not p or p.startswith("-") or p in ("/dev/null", "/dev/stdout", "/dev/stderr"):
105
+ return None
106
+ return p if len(p) < 200 else None
107
+
108
+
109
+ def scope_transcript(path: Path | str) -> Scope:
110
+ s = Scope()
111
+ pending_writes: dict = {}
112
+
113
+ for ev in _events(Path(path)):
114
+ if ev[0] == "text":
115
+ continue
116
+ if ev[0] == "cmd_use":
117
+ _, seq, _tid, cmd = ev
118
+ s.commands += 1
119
+
120
+ for rx, why in _OPAQUE_RULES:
121
+ if rx.search(cmd):
122
+ s.opaque.append((seq, cmd.strip()[:120], why))
123
+ break # one reason is enough to mark it unreadable
124
+
125
+ for m in _NET.finditer(cmd):
126
+ h = m.group(1).lower()
127
+ if not _LOCAL.match(h):
128
+ s.hosts.add(h)
129
+ if _GH_API.search(cmd):
130
+ s.hosts.add("api.github.com")
131
+ for m in _PUSH.finditer(cmd):
132
+ s.pushes.add(m.group(0).strip()[:80])
133
+ for m in _INSTALL.finditer(cmd):
134
+ pkg = next((g for g in m.groups() if g), "").strip()[:60]
135
+ s.installs.add(pkg or "(unnamed)")
136
+ for rx in (_REDIRECT, _FILE_MUT):
137
+ for m in rx.finditer(cmd):
138
+ p = _clean_path(m.group(m.lastindex or 1))
139
+ if p:
140
+ s.files.add(p)
141
+
142
+ # file-writing TOOL calls (Write/Edit) are recorded structurally, not by regex
143
+ with open(path, encoding="utf-8", errors="replace") as f:
144
+ for raw in f:
145
+ try:
146
+ d = json.loads(raw)
147
+ except Exception:
148
+ continue
149
+ msg = d.get("message") or {}
150
+ for c in (msg.get("content") or []) if isinstance(msg.get("content"), list) else []:
151
+ if isinstance(c, dict) and c.get("type") == "tool_use" \
152
+ and c.get("name") in _WRITE_TOOLS:
153
+ fp = (c.get("input") or {}).get("file_path")
154
+ if fp:
155
+ s.files.add(str(fp))
156
+ return s