coherence-check 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- coherence/__init__.py +26 -0
- coherence/__main__.py +437 -0
- coherence/audit/__init__.py +0 -0
- coherence/audit/scope.py +156 -0
- coherence/audit/transcript.py +213 -0
- coherence/ci/__init__.py +15 -0
- coherence/ci/session.py +304 -0
- coherence/claimproof/__init__.py +3 -0
- coherence/claimproof/steps.py +101 -0
- coherence/core/__init__.py +13 -0
- coherence/core/fact.py +211 -0
- coherence/core/spine.py +209 -0
- coherence/core/types.py +115 -0
- coherence/decisions/__init__.py +3 -0
- coherence/decisions/log.py +105 -0
- coherence/evolve/__init__.py +11 -0
- coherence/evolve/dominos.py +256 -0
- coherence/evolve/memory.py +231 -0
- coherence/health.py +193 -0
- coherence/py.typed +1 -0
- coherence/replay/__init__.py +3 -0
- coherence/replay/engine.py +85 -0
- coherence/review/__init__.py +3 -0
- coherence/review/triage.py +76 -0
- coherence/skills/__init__.py +3 -0
- coherence/skills/audit.py +73 -0
- coherence_check-0.6.0.dist-info/METADATA +342 -0
- coherence_check-0.6.0.dist-info/RECORD +33 -0
- coherence_check-0.6.0.dist-info/WHEEL +5 -0
- coherence_check-0.6.0.dist-info/entry_points.txt +2 -0
- coherence_check-0.6.0.dist-info/licenses/LICENSE +202 -0
- coherence_check-0.6.0.dist-info/licenses/NOTICE +4 -0
- coherence_check-0.6.0.dist-info/top_level.txt +1 -0
coherence/__init__.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Coherence — one law, many costumes.
|
|
2
|
+
|
|
3
|
+
Nothing is done unless there is evidence.
|
|
4
|
+
Nothing is finished unless there is a next.
|
|
5
|
+
Nothing is remembered unless it was done.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from coherence.core.fact import LAW, Fact, FactError, FactKind
|
|
9
|
+
from coherence.core.spine import Coherence
|
|
10
|
+
from coherence.core.types import Artifact, Bundle, Record, Truth
|
|
11
|
+
from coherence.evolve import DominoChain, EvolutionMemory
|
|
12
|
+
|
|
13
|
+
__all__ = [
|
|
14
|
+
"Coherence",
|
|
15
|
+
"Fact",
|
|
16
|
+
"FactError",
|
|
17
|
+
"FactKind",
|
|
18
|
+
"LAW",
|
|
19
|
+
"Bundle",
|
|
20
|
+
"Record",
|
|
21
|
+
"Artifact",
|
|
22
|
+
"Truth",
|
|
23
|
+
"DominoChain",
|
|
24
|
+
"EvolutionMemory",
|
|
25
|
+
]
|
|
26
|
+
__version__ = "0.5.1"
|
coherence/__main__.py
ADDED
|
@@ -0,0 +1,437 @@
|
|
|
1
|
+
"""python -m coherence demo|evolve|law|prove-cmd|check|report"""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import sys
|
|
8
|
+
import tempfile
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
from coherence import Coherence
|
|
12
|
+
from coherence.ci.session import (
|
|
13
|
+
DEFAULT_SESSION,
|
|
14
|
+
SessionStore,
|
|
15
|
+
build_report,
|
|
16
|
+
check_exit_code,
|
|
17
|
+
report_markdown,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def law() -> int:
|
|
22
|
+
from coherence.core.fact import LAW
|
|
23
|
+
|
|
24
|
+
print()
|
|
25
|
+
print("COHERENCE · the law")
|
|
26
|
+
print("=" * 50)
|
|
27
|
+
print(LAW)
|
|
28
|
+
print()
|
|
29
|
+
print("Atom: Fact(claim, evidence, next)")
|
|
30
|
+
print(" done ⇔ evidence non-empty")
|
|
31
|
+
print(" finished ⇔ next non-empty (always)")
|
|
32
|
+
print(" remember ⇔ done only")
|
|
33
|
+
print()
|
|
34
|
+
print("Everything else is a view over the same Fact.")
|
|
35
|
+
print("Docs: README.md")
|
|
36
|
+
return 0
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def demo() -> int:
|
|
40
|
+
print()
|
|
41
|
+
print("COHERENCE — one law, five kinds of Fact")
|
|
42
|
+
print("=" * 50)
|
|
43
|
+
from coherence.core.fact import LAW
|
|
44
|
+
|
|
45
|
+
print(LAW)
|
|
46
|
+
print()
|
|
47
|
+
|
|
48
|
+
c = Coherence(title="demo-pr-agent-fix")
|
|
49
|
+
c.skills.audit("web-search-skill", ["network", "read"], source="marketplace:example")
|
|
50
|
+
c.skills.audit("shell-runner", ["shell", "filesystem_write"], source="random-gist")
|
|
51
|
+
c.decisions.lock(
|
|
52
|
+
"auth-boundary",
|
|
53
|
+
"MUST NOT rewrite auth without human; tests must stay green",
|
|
54
|
+
by="staff-eng",
|
|
55
|
+
)
|
|
56
|
+
c.claimproof.illusion("unit tests", "agent said tests passed in chat")
|
|
57
|
+
c.claimproof.claim("refactored utils", "looks cleaner", next_action="add proof")
|
|
58
|
+
c.claimproof.cmd("unit tests", "pytest -q", exit_code=0)
|
|
59
|
+
c.claimproof.cmd("typecheck", "mypy src", exit_code=1)
|
|
60
|
+
c.decisions.check_violation("auth-boundary", "I did not rewrite auth; only utils")
|
|
61
|
+
c.replay.check()
|
|
62
|
+
triage = c.review.triage()
|
|
63
|
+
guard = c.require_coherent()
|
|
64
|
+
|
|
65
|
+
print(c.plain_english())
|
|
66
|
+
print("summary:", c.summary())
|
|
67
|
+
print(f"triage priority: {triage.meta.get('priority')}")
|
|
68
|
+
print(f"coherence guard: {guard.summary if guard else 'ok'}")
|
|
69
|
+
print("PASS — rungs 1–5 share one Bundle")
|
|
70
|
+
return 0
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def evolve_demo() -> int:
|
|
74
|
+
print()
|
|
75
|
+
print("COHERENCE EVOLVE — dominos + next steps + memory")
|
|
76
|
+
print("=" * 56)
|
|
77
|
+
|
|
78
|
+
mem = Path(tempfile.gettempdir()) / "coherence_evolve_demo.json"
|
|
79
|
+
if mem.exists():
|
|
80
|
+
mem.unlink()
|
|
81
|
+
|
|
82
|
+
c1 = Coherence(title="session-1", memory_path=mem, seed_cascade=True)
|
|
83
|
+
print(c1.dominos.plain_english())
|
|
84
|
+
print()
|
|
85
|
+
|
|
86
|
+
head = c1.dominos.head()
|
|
87
|
+
assert head is not None
|
|
88
|
+
print(f"ACTIVE: {head.title}")
|
|
89
|
+
print(f"GILBERT NEXT: {head.next_action}")
|
|
90
|
+
print()
|
|
91
|
+
|
|
92
|
+
proof_rec = c1.claimproof.cmd("unit tests", "pytest -q", exit_code=0)
|
|
93
|
+
c1.solve_domino(
|
|
94
|
+
head.id,
|
|
95
|
+
proof=f"cmd proven {proof_rec.proven}",
|
|
96
|
+
lesson="Never trust chat 'tests passed' — require cmd_exit artifact",
|
|
97
|
+
proof_record_id=proof_rec.id,
|
|
98
|
+
tags=["tests", "illusion"],
|
|
99
|
+
)
|
|
100
|
+
print("After knock #1:")
|
|
101
|
+
print(c1.dominos.plain_english())
|
|
102
|
+
print()
|
|
103
|
+
print(c1.evolve.plain_english())
|
|
104
|
+
print()
|
|
105
|
+
|
|
106
|
+
head2 = c1.dominos.head()
|
|
107
|
+
assert head2 is not None
|
|
108
|
+
sk = c1.skills.audit("shell-runner", ["shell"], source="gist")
|
|
109
|
+
c1.solve_domino(
|
|
110
|
+
head2.id,
|
|
111
|
+
proof=f"skill bill {sk.proven}",
|
|
112
|
+
lesson="High-risk skills must surface on the same report as test proof",
|
|
113
|
+
proof_record_id=sk.id,
|
|
114
|
+
tags=["skills"],
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
print("Session 1 summary:", c1.summary())
|
|
118
|
+
print()
|
|
119
|
+
|
|
120
|
+
c2 = Coherence(title="session-2", memory_path=mem)
|
|
121
|
+
print("SESSION 2 — evolution memory loaded")
|
|
122
|
+
print(c2.evolve.plain_english())
|
|
123
|
+
hints = c2.evolve.apply_hints("agent said tests passed in chat")
|
|
124
|
+
print(f"hints for 'tests passed in chat': {len(hints)}")
|
|
125
|
+
if hints:
|
|
126
|
+
print(f" lesson: {hints[0].lesson}")
|
|
127
|
+
print(f" GILBERT NEXT from memory: {hints[0].next_domino}")
|
|
128
|
+
print()
|
|
129
|
+
print("stats:", c2.evolve.stats())
|
|
130
|
+
print()
|
|
131
|
+
print("PASS — use → solve → learn → next session more coherent")
|
|
132
|
+
print("Docs: docs/EVOLUTION-AND-DOMINOS.md")
|
|
133
|
+
return 0
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def cmd_prove(argv: list[str]) -> int:
|
|
137
|
+
p = argparse.ArgumentParser(prog="coherence prove-cmd")
|
|
138
|
+
p.add_argument("command", help="shell command to run and prove")
|
|
139
|
+
p.add_argument("--session", default=str(DEFAULT_SESSION))
|
|
140
|
+
p.add_argument("--claim", default="")
|
|
141
|
+
p.add_argument(
|
|
142
|
+
"--next",
|
|
143
|
+
default="close remaining open facts or chain complete",
|
|
144
|
+
dest="next_action",
|
|
145
|
+
)
|
|
146
|
+
args = p.parse_args(argv)
|
|
147
|
+
store = SessionStore(args.session)
|
|
148
|
+
c, fact, code = store.prove_command(
|
|
149
|
+
args.command,
|
|
150
|
+
claim=args.claim or None,
|
|
151
|
+
next_action=args.next_action,
|
|
152
|
+
)
|
|
153
|
+
print(fact.plain_english())
|
|
154
|
+
print(f"session: {store.path} facts_done={len(c.done_facts())} open={len(c.open_facts())}")
|
|
155
|
+
# CI: non-zero if command failed
|
|
156
|
+
return 0 if code == 0 else 1
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def cmd_said(argv: list[str]) -> int:
|
|
160
|
+
p = argparse.ArgumentParser(prog="coherence said")
|
|
161
|
+
p.add_argument("claim")
|
|
162
|
+
p.add_argument("--next", required=True, dest="next_action")
|
|
163
|
+
p.add_argument("--session", default=str(DEFAULT_SESSION))
|
|
164
|
+
args = p.parse_args(argv)
|
|
165
|
+
store = SessionStore(args.session)
|
|
166
|
+
c = store.load()
|
|
167
|
+
f = c.said(args.claim, args.next_action)
|
|
168
|
+
store.save(c)
|
|
169
|
+
print(f.plain_english())
|
|
170
|
+
return 0
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def cmd_check(argv: list[str]) -> int:
|
|
174
|
+
p = argparse.ArgumentParser(prog="coherence check")
|
|
175
|
+
p.add_argument("--session", default=str(DEFAULT_SESSION))
|
|
176
|
+
p.add_argument(
|
|
177
|
+
"--strict",
|
|
178
|
+
action="store_true",
|
|
179
|
+
default=True,
|
|
180
|
+
help="fail if zero proven facts (default)",
|
|
181
|
+
)
|
|
182
|
+
p.add_argument("--no-strict", action="store_true", help="allow empty session")
|
|
183
|
+
args = p.parse_args(argv)
|
|
184
|
+
strict = not args.no_strict
|
|
185
|
+
store = SessionStore(args.session)
|
|
186
|
+
|
|
187
|
+
# Integrity FIRST. If the session file was edited after the facts were
|
|
188
|
+
# recorded, nothing else it says can be trusted — so a tampered chain is a
|
|
189
|
+
# hard failure (exit 3) before any fact is read. This is the guard that
|
|
190
|
+
# makes "agents can't fake green" true for the case that actually matters:
|
|
191
|
+
# the policed agent editing its own session on disk.
|
|
192
|
+
integ = store.verify()
|
|
193
|
+
if integ["status"] == "tampered":
|
|
194
|
+
print(json.dumps({"ok": False, "integrity": integ}, indent=2))
|
|
195
|
+
print(f"CHECK FAIL: session tampered at entry {integ.get('position')} "
|
|
196
|
+
f"— {integ.get('detail')}", file=sys.stderr)
|
|
197
|
+
return 3
|
|
198
|
+
|
|
199
|
+
c = store.load()
|
|
200
|
+
report = build_report(c)
|
|
201
|
+
report["integrity"] = integ
|
|
202
|
+
print(json.dumps(report, indent=2))
|
|
203
|
+
code = check_exit_code(c, strict=strict)
|
|
204
|
+
if code == 0:
|
|
205
|
+
note = "" if integ["status"] == "ok" else f" ({integ['status']})"
|
|
206
|
+
print(f"CHECK PASS{note}", file=sys.stderr)
|
|
207
|
+
elif code == 2:
|
|
208
|
+
print("CHECK FAIL: no proven facts (strict)", file=sys.stderr)
|
|
209
|
+
else:
|
|
210
|
+
print("CHECK FAIL: open or blocked facts remain", file=sys.stderr)
|
|
211
|
+
return code
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def cmd_scope(argv: list[str]) -> int:
|
|
215
|
+
"""Blast radius: what the agent touched, and where this report's edge is."""
|
|
216
|
+
p = argparse.ArgumentParser(prog="coherence scope")
|
|
217
|
+
p.add_argument("transcript")
|
|
218
|
+
p.add_argument("--json", action="store_true", dest="as_json")
|
|
219
|
+
p.add_argument("--full", action="store_true", help="list every item, not a sample")
|
|
220
|
+
args = p.parse_args(argv)
|
|
221
|
+
from coherence.audit.scope import scope_transcript
|
|
222
|
+
sc = scope_transcript(args.transcript)
|
|
223
|
+
if args.as_json:
|
|
224
|
+
print(json.dumps({
|
|
225
|
+
"commands": sc.commands, "bounded": sc.bounded(),
|
|
226
|
+
"files": sorted(sc.files), "pushes": sorted(sc.pushes),
|
|
227
|
+
"hosts": sorted(sc.hosts), "installs": sorted(sc.installs),
|
|
228
|
+
"opaque": [{"line": a, "command": b, "why": c} for a, b, c in sc.opaque],
|
|
229
|
+
}, indent=2))
|
|
230
|
+
return sc.exit_code()
|
|
231
|
+
|
|
232
|
+
def show(label, items):
|
|
233
|
+
items = sorted(items)
|
|
234
|
+
print(f"\n {label} ({len(items)})")
|
|
235
|
+
for i in (items if args.full else items[:8]):
|
|
236
|
+
print(f" {i}")
|
|
237
|
+
if not args.full and len(items) > 8:
|
|
238
|
+
print(f" … {len(items) - 8} more (--full)")
|
|
239
|
+
|
|
240
|
+
print(f"blast radius of {sc.commands} commands\n")
|
|
241
|
+
show("files touched", sc.files)
|
|
242
|
+
show("repos pushed", sc.pushes)
|
|
243
|
+
show("network hosts contacted", sc.hosts)
|
|
244
|
+
show("packages installed", sc.installs)
|
|
245
|
+
|
|
246
|
+
print(f"\n OPAQUE — effects this report CANNOT see ({len(sc.opaque)})")
|
|
247
|
+
for seq, cmd, why in (sc.opaque if args.full else sc.opaque[:8]):
|
|
248
|
+
print(f" line {seq}: {cmd}")
|
|
249
|
+
print(f" why: {why}")
|
|
250
|
+
if not args.full and len(sc.opaque) > 8:
|
|
251
|
+
print(f" … {len(sc.opaque) - 8} more (--full)")
|
|
252
|
+
|
|
253
|
+
if sc.bounded():
|
|
254
|
+
print(f"\nVERDICT: BOUNDED — {len(sc.opaque)} command(s) could do anything this")
|
|
255
|
+
print("report cannot see. Everything above is what IS visible, not everything")
|
|
256
|
+
print("that happened. Unknown is reported as unknown, never as 'nothing'.")
|
|
257
|
+
else:
|
|
258
|
+
print("\nVERDICT: fully readable — every command's effects were determinable.")
|
|
259
|
+
return sc.exit_code()
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def cmd_audit(argv: list[str]) -> int:
|
|
263
|
+
"""Audit an agent transcript: every checkable claim vs. what actually ran."""
|
|
264
|
+
p = argparse.ArgumentParser(prog="coherence audit")
|
|
265
|
+
p.add_argument("transcript", help="agent session .jsonl (Claude Code format)")
|
|
266
|
+
p.add_argument("--json", action="store_true", dest="as_json")
|
|
267
|
+
args = p.parse_args(argv)
|
|
268
|
+
from coherence.audit.transcript import (
|
|
269
|
+
audit_transcript, SUPPORTED, WEAK, UNSUPPORTED, CONTRADICTED)
|
|
270
|
+
a = audit_transcript(args.transcript)
|
|
271
|
+
c = a.counts()
|
|
272
|
+
if args.as_json:
|
|
273
|
+
print(json.dumps({
|
|
274
|
+
"commands": a.commands, "claims": len(a.claims), "counts": c,
|
|
275
|
+
"findings": [vars(x) for x in a.claims
|
|
276
|
+
if x.verdict in (UNSUPPORTED, CONTRADICTED, WEAK)],
|
|
277
|
+
}, indent=2))
|
|
278
|
+
return a.exit_code()
|
|
279
|
+
print(f"audited: {a.commands} commands, {len(a.claims)} checkable claims\n")
|
|
280
|
+
print(f" supported {c[SUPPORTED]}")
|
|
281
|
+
print(f" weak evidence {c[WEAK]} (piped exit codes — pytest | tail class)")
|
|
282
|
+
print(f" unsupported {c[UNSUPPORTED]} (claims resting on nothing)")
|
|
283
|
+
print(f" CONTRADICTED {c[CONTRADICTED]} (claimed success; its own transcript says failure)")
|
|
284
|
+
for x in a.claims:
|
|
285
|
+
if x.verdict == CONTRADICTED:
|
|
286
|
+
print(f"\n LIE at line {x.seq} [{x.kind}]: \"{x.text}\"")
|
|
287
|
+
print(f" evidence against: {x.evidence}")
|
|
288
|
+
for x in a.claims:
|
|
289
|
+
if x.verdict == UNSUPPORTED:
|
|
290
|
+
print(f"\n unsupported line {x.seq} [{x.kind}]: \"{x.text[:100]}\"")
|
|
291
|
+
print(f"\nexit {a.exit_code()} (0 all supported · 1 unsupported · 2 contradicted)")
|
|
292
|
+
return a.exit_code()
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def cmd_tamper_demo(argv: list[str]) -> int:
|
|
296
|
+
"""Show, in one command, the thing that is hard to believe from prose:
|
|
297
|
+
a forged 'proven' in the session file is caught.
|
|
298
|
+
|
|
299
|
+
Runs entirely in a throwaway temp directory — touches nothing of yours.
|
|
300
|
+
"""
|
|
301
|
+
import tempfile
|
|
302
|
+
from coherence.ci.session import SessionStore
|
|
303
|
+
|
|
304
|
+
with tempfile.TemporaryDirectory() as td:
|
|
305
|
+
sess = Path(td) / "session.json"
|
|
306
|
+
store = SessionStore(sess)
|
|
307
|
+
|
|
308
|
+
print("1. An agent runs a check. It FAILS — so no proof is recorded.")
|
|
309
|
+
store.prove_command("false", claim="test suite")
|
|
310
|
+
print(f" coherence check -> exit {cmd_exit(sess)} (open fact, no evidence)\n")
|
|
311
|
+
|
|
312
|
+
print("2. The agent edits its own session file to claim it passed.")
|
|
313
|
+
data = json.loads(sess.read_text())
|
|
314
|
+
data["facts"][0]["evidence"] = "exit_code=0 output_digest=deadbeef"
|
|
315
|
+
data["facts"][0]["next"] = "chain complete"
|
|
316
|
+
sess.write_text(json.dumps(data, indent=2))
|
|
317
|
+
print(' forged: evidence = "exit_code=0 ..."\n')
|
|
318
|
+
|
|
319
|
+
print("3. The check runs again. The hash chain does not match.")
|
|
320
|
+
code = cmd_exit(sess)
|
|
321
|
+
v = store.verify()
|
|
322
|
+
print(f" coherence check -> exit {code} {v['status'].upper()} at entry {v.get('position')}")
|
|
323
|
+
print(f" {v.get('detail')}\n")
|
|
324
|
+
|
|
325
|
+
if code != 3:
|
|
326
|
+
print("UNEXPECTED: tampering was not caught", file=sys.stderr)
|
|
327
|
+
return 1
|
|
328
|
+
print("A forged green is caught. That is the whole idea.")
|
|
329
|
+
print("Exit codes: 0 pass · 1 open facts · 2 empty · 3 tampered")
|
|
330
|
+
return 0
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def cmd_exit(session: Path) -> int:
|
|
334
|
+
"""Run the real check logic quietly and return only its exit code."""
|
|
335
|
+
from coherence.ci.session import SessionStore, check_exit_code
|
|
336
|
+
store = SessionStore(session)
|
|
337
|
+
integ = store.verify()
|
|
338
|
+
if integ["status"] == "tampered":
|
|
339
|
+
return 3
|
|
340
|
+
return check_exit_code(store.load(), strict=True)
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def cmd_report(argv: list[str]) -> int:
|
|
344
|
+
p = argparse.ArgumentParser(prog="coherence report")
|
|
345
|
+
p.add_argument("--session", default=str(DEFAULT_SESSION))
|
|
346
|
+
p.add_argument("--json", action="store_true")
|
|
347
|
+
p.add_argument("--out", default="", help="write markdown or json to file")
|
|
348
|
+
args = p.parse_args(argv)
|
|
349
|
+
store = SessionStore(args.session)
|
|
350
|
+
c = store.load()
|
|
351
|
+
if args.json:
|
|
352
|
+
body = json.dumps(build_report(c), indent=2)
|
|
353
|
+
else:
|
|
354
|
+
body = report_markdown(c)
|
|
355
|
+
if args.out:
|
|
356
|
+
Path(args.out).write_text(body + "\n", encoding="utf-8")
|
|
357
|
+
print(f"wrote {args.out}")
|
|
358
|
+
else:
|
|
359
|
+
print(body)
|
|
360
|
+
return 0
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
def main(argv: list[str] | None = None) -> None:
|
|
364
|
+
argv = list(sys.argv[1:] if argv is None else argv)
|
|
365
|
+
if not argv:
|
|
366
|
+
raise SystemExit(demo())
|
|
367
|
+
cmd = argv[0].lower()
|
|
368
|
+
rest = argv[1:]
|
|
369
|
+
if cmd in ("demo", "run"):
|
|
370
|
+
raise SystemExit(demo())
|
|
371
|
+
if cmd in ("evolve", "domino", "flywheel"):
|
|
372
|
+
raise SystemExit(evolve_demo())
|
|
373
|
+
if cmd in ("law", "320", "iq"):
|
|
374
|
+
raise SystemExit(law())
|
|
375
|
+
if cmd in ("health", "doctor", "status"):
|
|
376
|
+
from coherence.health import plain_english, run_health, write_report
|
|
377
|
+
|
|
378
|
+
p = argparse.ArgumentParser(prog="coherence health")
|
|
379
|
+
p.add_argument("--out", default="", help="write JSON report path")
|
|
380
|
+
p.add_argument("--memory", default="", help="optional evolution memory to verify")
|
|
381
|
+
p.add_argument("--no-storm", action="store_true")
|
|
382
|
+
args = p.parse_args(rest)
|
|
383
|
+
report = run_health(
|
|
384
|
+
include_storm=not args.no_storm,
|
|
385
|
+
memory_path=Path(args.memory) if args.memory else None,
|
|
386
|
+
)
|
|
387
|
+
print(plain_english(report))
|
|
388
|
+
if args.out:
|
|
389
|
+
write_report(report, Path(args.out))
|
|
390
|
+
print(f"wrote {args.out}")
|
|
391
|
+
raise SystemExit(0 if report.ok else 1)
|
|
392
|
+
if cmd in ("storm", "proof", "storm-proof"):
|
|
393
|
+
# Hostile proof harness (EffectFence/Seal style)
|
|
394
|
+
from pathlib import Path as _P
|
|
395
|
+
import runpy
|
|
396
|
+
|
|
397
|
+
storm = _P(__file__).resolve().parents[2] / "storm.py"
|
|
398
|
+
if not storm.exists():
|
|
399
|
+
# installed wheel: look beside package or cwd
|
|
400
|
+
storm = _P.cwd() / "storm.py"
|
|
401
|
+
if storm.exists():
|
|
402
|
+
raise SystemExit(runpy.run_path(str(storm), run_name="__main__") or 0)
|
|
403
|
+
print("storm.py not found — run from repo: python storm.py", file=sys.stderr)
|
|
404
|
+
raise SystemExit(2)
|
|
405
|
+
if cmd in ("prove-cmd", "prove_cmd", "prove"):
|
|
406
|
+
raise SystemExit(cmd_prove(rest))
|
|
407
|
+
if cmd == "said":
|
|
408
|
+
raise SystemExit(cmd_said(rest))
|
|
409
|
+
if cmd == "check":
|
|
410
|
+
raise SystemExit(cmd_check(rest))
|
|
411
|
+
if cmd == "scope":
|
|
412
|
+
raise SystemExit(cmd_scope(rest))
|
|
413
|
+
if cmd == "audit":
|
|
414
|
+
raise SystemExit(cmd_audit(rest))
|
|
415
|
+
if cmd in ("tamper-demo", "tamper_demo", "tamper"):
|
|
416
|
+
raise SystemExit(cmd_tamper_demo(rest))
|
|
417
|
+
if cmd == "report":
|
|
418
|
+
raise SystemExit(cmd_report(rest))
|
|
419
|
+
if cmd in ("-h", "--help", "help"):
|
|
420
|
+
print(
|
|
421
|
+
"Usage: python -m coherence <command>\n"
|
|
422
|
+
" law | demo | evolve | storm | health\n"
|
|
423
|
+
" said CLAIM --next NEXT\n"
|
|
424
|
+
" prove-cmd 'pytest -q'\n"
|
|
425
|
+
" tamper-demo (10s: forge a green, watch it get caught)\n"
|
|
426
|
+
" audit FILE.jsonl (agent transcript: claims vs. what actually ran)\n"
|
|
427
|
+
" scope FILE.jsonl (blast radius: what it touched + what we cannot see)\n"
|
|
428
|
+
" check [--no-strict]\n"
|
|
429
|
+
" report [--json] [--out file.md]\n"
|
|
430
|
+
)
|
|
431
|
+
raise SystemExit(0)
|
|
432
|
+
print(f"Unknown command: {cmd}. Try: law | demo | prove-cmd | check | report", file=sys.stderr)
|
|
433
|
+
raise SystemExit(2)
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
if __name__ == "__main__":
|
|
437
|
+
main()
|
|
File without changes
|
coherence/audit/scope.py
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"""Blast radius: what did the agent touch — and what can't we tell?
|
|
2
|
+
|
|
3
|
+
`audit` answers "did it do what it said?". This answers the harder twin:
|
|
4
|
+
"did it do anything it did NOT say?" — the 3am question for anyone who lets an
|
|
5
|
+
agent run unattended, and the one every security review asks as "show me
|
|
6
|
+
everything the agent touched."
|
|
7
|
+
|
|
8
|
+
Proving a negative from a log is a logic trap: absence of evidence is not
|
|
9
|
+
evidence of absence. A single `bash deploy.sh` can hide any effect in the
|
|
10
|
+
world. The usual responses are both wrong:
|
|
11
|
+
|
|
12
|
+
* claim completeness anyway (a lie that gets someone breached), or
|
|
13
|
+
* demand syscall-level sandboxing, ship nothing, and leave the question
|
|
14
|
+
unanswered forever — which is where the industry actually sits.
|
|
15
|
+
|
|
16
|
+
This module takes the third road, and it is the same rule Seal's witness is
|
|
17
|
+
built on: **UNKNOWN never collapses into ABSENT.** Commands whose effects
|
|
18
|
+
cannot be determined statically are not skipped and not guessed at — they are
|
|
19
|
+
reported as OPAQUE, counted, and printed, and the verdict states plainly that
|
|
20
|
+
the report is bounded by them. A bounded answer that says where its edge is
|
|
21
|
+
beats both a false complete one and no answer at all.
|
|
22
|
+
|
|
23
|
+
What is extracted (best effort, from the transcript already on disk):
|
|
24
|
+
files written · repos pushed · network hosts contacted · packages installed
|
|
25
|
+
|
|
26
|
+
What makes a command OPAQUE:
|
|
27
|
+
running a script file, `eval`, command substitution, piping a download into
|
|
28
|
+
a shell, `source`, or a base64 decode into execution — anything whose real
|
|
29
|
+
effects live somewhere this file cannot see.
|
|
30
|
+
"""
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import json
|
|
34
|
+
import re
|
|
35
|
+
from dataclasses import dataclass, field
|
|
36
|
+
from pathlib import Path
|
|
37
|
+
|
|
38
|
+
from .transcript import _events # same parser, one source of truth
|
|
39
|
+
|
|
40
|
+
# ── effects we CAN read directly off the command line ────────────────────
|
|
41
|
+
_WRITE_TOOLS = {"Write", "Edit", "NotebookEdit", "MultiEdit"}
|
|
42
|
+
|
|
43
|
+
_NET = re.compile(
|
|
44
|
+
r"\b(?:curl|wget|http|https)\b[^\n|;]*?"
|
|
45
|
+
r"(?:https?://)?([a-z0-9.-]+\.[a-z]{2,})(?:[/:\s]|$)", re.I)
|
|
46
|
+
_GH_API = re.compile(r"\bgh\s+(?:api|repo|pr|issue|release|run|search)\b", re.I)
|
|
47
|
+
_PUSH = re.compile(r"\bgit\s+push\b[^\n;&|]*", re.I)
|
|
48
|
+
_INSTALL = re.compile(
|
|
49
|
+
r"\b(?:pip|pip3|python3?\s+-m\s+pip)\s+install\s+([^\n;&|]+)"
|
|
50
|
+
r"|\bnpm\s+(?:i|install|publish)\s*([^\n;&|]*)"
|
|
51
|
+
r"|\bcargo\s+(?:install|publish)\s*([^\n;&|]*)"
|
|
52
|
+
r"|\bbrew\s+install\s+([^\n;&|]+)", re.I)
|
|
53
|
+
_REDIRECT = re.compile(r"(?<![>\d])>{1,2}\s*([^\s&|;]+)")
|
|
54
|
+
_FILE_MUT = re.compile(
|
|
55
|
+
r"\b(?:rm|mv|cp|chmod|chown|touch|mkdir|ln)\s+(-[^\s]*\s+)*([^\n;&|]+)", re.I)
|
|
56
|
+
|
|
57
|
+
# ── what we deliberately CANNOT read: the honest edge ────────────────────
|
|
58
|
+
_OPAQUE_RULES = [
|
|
59
|
+
(re.compile(r"\beval\b"), "eval — constructs code at runtime"),
|
|
60
|
+
(re.compile(r"\bsource\s+|\.\s+/"), "source — runs another file's contents"),
|
|
61
|
+
(re.compile(r"(?:curl|wget)[^\n]*\|\s*(?:ba|z|)sh\b"),
|
|
62
|
+
"download piped into a shell"),
|
|
63
|
+
(re.compile(r"\bbase64\s+(?:-d|--decode)[^\n]*\|"),
|
|
64
|
+
"base64 decoded into execution"),
|
|
65
|
+
(re.compile(r"\b(?:ba|z|)sh\s+[^\s-][^\n;&|]*\.(?:sh|bash)\b"),
|
|
66
|
+
"runs a script file"),
|
|
67
|
+
(re.compile(r"\b(?:python3?|node|ruby|perl)\s+[^\s-][^\n;&|]*\.(?:py|js|rb|pl)\b"),
|
|
68
|
+
"runs a program file"),
|
|
69
|
+
(re.compile(r"\$\((?!\s*(?:pwd|date|basename|dirname|echo)\b)"),
|
|
70
|
+
"command substitution"),
|
|
71
|
+
(re.compile(r"\bmake\b|\bnpm\s+run\b|\byarn\s+\w+"),
|
|
72
|
+
"delegates to a script target"),
|
|
73
|
+
(re.compile(r"\bdocker\s+run\b|\bssh\b"), "executes in another environment"),
|
|
74
|
+
]
|
|
75
|
+
|
|
76
|
+
# hosts that are noise, not reach: the agent's own machine
|
|
77
|
+
_LOCAL = re.compile(r"^(?:localhost|127\.0\.0\.1|0\.0\.0\.0|\[::1\])")
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass
|
|
81
|
+
class Scope:
|
|
82
|
+
files: set = field(default_factory=set)
|
|
83
|
+
pushes: set = field(default_factory=set)
|
|
84
|
+
hosts: set = field(default_factory=set)
|
|
85
|
+
installs: set = field(default_factory=set)
|
|
86
|
+
opaque: list = field(default_factory=list) # (seq, command, why)
|
|
87
|
+
commands: int = 0
|
|
88
|
+
|
|
89
|
+
def bounded(self) -> bool:
|
|
90
|
+
"""True when opaque commands exist — the report has a known edge."""
|
|
91
|
+
return bool(self.opaque)
|
|
92
|
+
|
|
93
|
+
def exit_code(self) -> int:
|
|
94
|
+
"""0 = fully readable · 1 = readable but BOUNDED by opaque commands.
|
|
95
|
+
|
|
96
|
+
Never 0 while anything is opaque: a report that cannot see everything
|
|
97
|
+
must not exit like one that can.
|
|
98
|
+
"""
|
|
99
|
+
return 1 if self.opaque else 0
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _clean_path(p: str) -> str | None:
|
|
103
|
+
p = p.strip().strip("\"'")
|
|
104
|
+
if not p or p.startswith("-") or p in ("/dev/null", "/dev/stdout", "/dev/stderr"):
|
|
105
|
+
return None
|
|
106
|
+
return p if len(p) < 200 else None
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def scope_transcript(path: Path | str) -> Scope:
|
|
110
|
+
s = Scope()
|
|
111
|
+
pending_writes: dict = {}
|
|
112
|
+
|
|
113
|
+
for ev in _events(Path(path)):
|
|
114
|
+
if ev[0] == "text":
|
|
115
|
+
continue
|
|
116
|
+
if ev[0] == "cmd_use":
|
|
117
|
+
_, seq, _tid, cmd = ev
|
|
118
|
+
s.commands += 1
|
|
119
|
+
|
|
120
|
+
for rx, why in _OPAQUE_RULES:
|
|
121
|
+
if rx.search(cmd):
|
|
122
|
+
s.opaque.append((seq, cmd.strip()[:120], why))
|
|
123
|
+
break # one reason is enough to mark it unreadable
|
|
124
|
+
|
|
125
|
+
for m in _NET.finditer(cmd):
|
|
126
|
+
h = m.group(1).lower()
|
|
127
|
+
if not _LOCAL.match(h):
|
|
128
|
+
s.hosts.add(h)
|
|
129
|
+
if _GH_API.search(cmd):
|
|
130
|
+
s.hosts.add("api.github.com")
|
|
131
|
+
for m in _PUSH.finditer(cmd):
|
|
132
|
+
s.pushes.add(m.group(0).strip()[:80])
|
|
133
|
+
for m in _INSTALL.finditer(cmd):
|
|
134
|
+
pkg = next((g for g in m.groups() if g), "").strip()[:60]
|
|
135
|
+
s.installs.add(pkg or "(unnamed)")
|
|
136
|
+
for rx in (_REDIRECT, _FILE_MUT):
|
|
137
|
+
for m in rx.finditer(cmd):
|
|
138
|
+
p = _clean_path(m.group(m.lastindex or 1))
|
|
139
|
+
if p:
|
|
140
|
+
s.files.add(p)
|
|
141
|
+
|
|
142
|
+
# file-writing TOOL calls (Write/Edit) are recorded structurally, not by regex
|
|
143
|
+
with open(path, encoding="utf-8", errors="replace") as f:
|
|
144
|
+
for raw in f:
|
|
145
|
+
try:
|
|
146
|
+
d = json.loads(raw)
|
|
147
|
+
except Exception:
|
|
148
|
+
continue
|
|
149
|
+
msg = d.get("message") or {}
|
|
150
|
+
for c in (msg.get("content") or []) if isinstance(msg.get("content"), list) else []:
|
|
151
|
+
if isinstance(c, dict) and c.get("type") == "tool_use" \
|
|
152
|
+
and c.get("name") in _WRITE_TOOLS:
|
|
153
|
+
fp = (c.get("input") or {}).get("file_path")
|
|
154
|
+
if fp:
|
|
155
|
+
s.files.add(str(fp))
|
|
156
|
+
return s
|