loki-mode 9.12.6 → 9.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/README.md +81 -101
  2. package/SKILL.md +2 -2
  3. package/VERSION +1 -1
  4. package/autonomy/intent.sh +414 -0
  5. package/autonomy/issue-providers.sh +24 -0
  6. package/autonomy/lib/agent_readiness.py +280 -0
  7. package/autonomy/lib/claim_grounding.py +171 -0
  8. package/autonomy/lib/config-map.sh +10 -6
  9. package/autonomy/lib/decision_record.py +198 -0
  10. package/autonomy/lib/failure_memory.py +199 -0
  11. package/autonomy/lib/gate_policy.py +166 -0
  12. package/autonomy/lib/outcome_ledger.py +620 -0
  13. package/autonomy/lib/preedit_snapshot.py +216 -0
  14. package/autonomy/lib/proof-generator.py +71 -4
  15. package/autonomy/lib/verdict.py +204 -0
  16. package/autonomy/loki +430 -15
  17. package/autonomy/notify.sh +70 -1
  18. package/autonomy/provider-offer.sh +25 -1
  19. package/autonomy/queue-consumer.sh +290 -18
  20. package/autonomy/run.sh +527 -12
  21. package/autonomy/telemetry.sh +8 -1
  22. package/completions/_loki +5 -0
  23. package/completions/loki.bash +2 -1
  24. package/dashboard/__init__.py +1 -1
  25. package/dashboard/run.py +13 -2
  26. package/dashboard/scim.py +221 -0
  27. package/dashboard/server.py +190 -1
  28. package/dashboard/static/index.html +248 -55
  29. package/docs/GATE-FAILURE-TRIAGE.md +254 -0
  30. package/docs/LOOP-CANDIDATE-PROPOSAL-v1.md +167 -0
  31. package/docs/LOOP-HARNESS-AUDIT.md +53 -0
  32. package/docs/QUEUE-OPERATIONS.md +107 -0
  33. package/docs/VERIFICATION-COST.md +273 -0
  34. package/docs/WANG-PRINCIPLES-PLAN.md +1 -1
  35. package/loki-ts/dist/loki.js +414 -416
  36. package/mcp/__init__.py +1 -1
  37. package/mcp/_sdk_loader.py +25 -0
  38. package/package.json +1 -1
  39. package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
@@ -0,0 +1,620 @@
1
+ #!/usr/bin/env python3
2
+ """Outcome Ledger: did the change turn out to be RIGHT, not merely that it happened.
3
+
4
+ WHY THIS EXISTS. Every competing agent reports VOLUME. Measured against their own
5
+ published docs: Factory AI's analytics expose files_created, files_edited,
6
+ lines_modified, git_commits, git_prs_created, tokens and DAU -- and no defect
7
+ rate, no rework rate, no revert rate, no change-failure rate. Their telemetry doc
8
+ states outright that correlating usage with delivery outcomes is left to the
9
+ customer's own stack. Devin's security page concedes the agent "can still
10
+ experience hallucinations, introduce bugs into code" and points you at your own
11
+ code review and branch protection. So an engineering leader using either can
12
+ prove the agent was BUSY. Neither can show it was RIGHT.
13
+
14
+ This module answers the other question, from local git history alone: after the
15
+ receipt was written, did the change survive?
16
+
17
+ WHAT MAKES A SIGNAL DEFENSIBLE HERE. The temptation is to count "the file was
18
+ touched again" as rework. That is a misleading signal -- an unrelated feature
19
+ landing in the same file would inflate it, and a leader acting on that number
20
+ would be acting on noise. So each signal below is either a deterministic fact or
21
+ it reports UNKNOWN:
22
+
23
+ reverted FACT. `git revert` writes "This reverts commit <sha>" into the
24
+ message. That is a machine-parseable link, not an inference.
25
+ line_survival FACT. `git blame --porcelain` attributes every surviving line to
26
+ the commit that introduced it. Counting lines still attributed to
27
+ head_sha is a measurement, not an estimate.
28
+ reworked MEASURED, line-scoped. Only lines this change INTRODUCED and that
29
+ a LATER commit replaced count as rework. A file touched elsewhere
30
+ does not.
31
+ UNKNOWN Any case we cannot compute: absent head_sha, unreachable commit
32
+ (never merged, shallow clone, pruned), or a file since deleted.
33
+ Never reported as zero, never as a pass.
34
+
35
+ The last rule is the whole point. A change-failure rate that silently scores
36
+ unmeasurable cases as successes is the exact false-green this product exists to
37
+ refuse. `loki proof` already sets the house convention -- its help says
38
+ "version_is_ahead reads UNKNOWN when it cannot be computed" -- and this inherits it.
39
+
40
+ READ-ONLY. Never writes to the repo under analysis. Every number it prints comes
41
+ from a git command the reader can rerun by hand; --json emits those commands.
42
+ """
43
+
44
+ from __future__ import annotations
45
+
46
+ import json
47
+ import os
48
+ import subprocess
49
+ import sys
50
+ from datetime import datetime, timezone
51
+
52
+ SCHEMA_VERSION = "1.0"
53
+
54
+ # A status that is not a number. Kept as a module constant so a caller can never
55
+ # accidentally coerce it to 0 in a tally.
56
+ UNKNOWN = "UNKNOWN"
57
+
58
+ # git's empty-tree object. proof-generator falls back to it for a greenfield run,
59
+ # so a receipt carrying it has no real baseline to diff against.
60
+ EMPTY_TREE_SHA = "4b825dc642cb6eb9a060e54bf8d69288fbee4904"
61
+
62
+ # THE ANCHOR GATE, and the reason this module is not a metrics generator.
63
+ #
64
+ # `facts.git.head_sha` is `git rev-parse HEAD` AT RECEIPT-GENERATION TIME. When
65
+ # the agent's work is still uncommitted -- the normal case -- that is the run's
66
+ # STARTING commit, not the commit the change became. Measured on this repo's own
67
+ # 9 receipts:
68
+ #
69
+ # 8 of 9 carry base_sha="" and head_sha=1385e71c. That commit touches TWO files
70
+ # (providers/codex.sh, tests/test-provider-degraded-mode.sh) while the receipts
71
+ # attest to EIGHT including autonomy/telemetry.sh. They are not the same change.
72
+ # The 9th carries base_sha=4b825dc6 (empty tree) and 4017 files.
73
+ #
74
+ # Following head_sha regardless would have attributed one commit's fate to eight
75
+ # unrelated runs and reported it as a change-failure rate. That is precisely the
76
+ # fabricated metric this product exists to refuse, and it would have been
77
+ # invisible in the output.
78
+ #
79
+ # File-overlap was tested as a fallback discriminator and REJECTED: the bad
80
+ # receipts overlap that commit by 2 files, so any overlap>0 rule marks all eight
81
+ # as anchored. Overlap is not evidence of identity.
82
+ #
83
+ # So: a receipt yields numbers only when sha algebra proves the range is the
84
+ # change. Everything else is UNKNOWN with a named reason.
85
+ ANCHOR_REASONS = {
86
+ "no_git": "not a git repository",
87
+ "head_sha_empty": "receipt records no head_sha",
88
+ "base_sha_empty": "run baseline was not recorded at generation",
89
+ "greenfield_no_baseline": "base_sha is the empty tree, so there is no baseline",
90
+ "change_not_committed": "base_sha == head_sha, so the work was never committed",
91
+ "sha_not_in_history": "a sha is not resolvable in this clone",
92
+ "not_reachable_from_head": "head_sha is not an ancestor of HEAD (never merged)",
93
+ "diff_range_mismatch": "the base..head diff does not match the receipt's file set",
94
+ }
95
+
96
+ # HISTORICAL vs LIVE: the distinction that stops an old receipt from reading as
97
+ # a regression.
98
+ #
99
+ # `ANCHORED 0 of 9` is alarming until you know WHY. Two very different things
100
+ # produce it, and a flat count of reasons cannot tell them apart:
101
+ #
102
+ # FROZEN The receipt file itself lacks the anchor. base_sha was written as
103
+ # "" and that JSON is on disk forever. No future fix can anchor it,
104
+ # and rewriting a receipt to make a metric look better is the exact
105
+ # dishonesty this module exists to refuse. These are HISTORY.
106
+ # LIVE The receipt records real shas; only the ENVIRONMENT cannot resolve
107
+ # them right now -- a shallow clone, an unmerged branch, or a file
108
+ # set that still has uncommitted edits. These can anchor later, with
109
+ # no change to the receipt.
110
+ #
111
+ # Measured on this repo: all 9 receipts are frozen (8 base_sha_empty, 1
112
+ # greenfield_no_baseline) and were generated 2026-07-27 and 2026-07-31, BEFORE
113
+ # proof-generator learned to read .loki/state/start-sha (commit 99ce689d,
114
+ # 2026-08-07). So the zero is fully explained by history.
115
+ # Frozen because the GENERATOR failed to record something it should have. These
116
+ # are the only two where recency is meaningful: before the fix they are history,
117
+ # after it they are a live bug in proof-generator.
118
+ FROZEN_GENERATOR_REASONS = frozenset({
119
+ "head_sha_empty",
120
+ "base_sha_empty",
121
+ })
122
+
123
+ # Frozen BY DESIGN, and correct at any date. A genuinely greenfield repo has no
124
+ # earlier commit to diff against, and a receipt written over uncommitted work
125
+ # honestly has base == head. Neither can ever anchor, and neither is a defect --
126
+ # so recency must NOT be applied to them. Calling a correct greenfield receipt a
127
+ # regression would be a false alarm on the signal added to prevent false alarms.
128
+ FROZEN_BY_DESIGN_REASONS = frozenset({
129
+ "greenfield_no_baseline",
130
+ "change_not_committed",
131
+ })
132
+
133
+ FROZEN_REASONS = FROZEN_GENERATOR_REASONS | FROZEN_BY_DESIGN_REASONS
134
+
135
+ # The two sets must stay disjoint. If a reason appeared in both, classification
136
+ # would depend on which branch is checked first -- and a by-design case that
137
+ # drifted into the generator set would start alarming as a regression the moment
138
+ # someone reordered the checks. Assert the invariant rather than trusting order.
139
+ assert not (FROZEN_GENERATOR_REASONS & FROZEN_BY_DESIGN_REASONS), \
140
+ "a reason cannot be both a generator failure and correct by design"
141
+
142
+ # The commit that taught proof-generator to resolve base_sha from
143
+ # .loki/state/start-sha when the env var is absent. A receipt generated at or
144
+ # after this instant should carry a real baseline, so a FROZEN reason on one is
145
+ # NOT history -- it is a live regression in the generator.
146
+ #
147
+ # Recency is the discriminator because it is the only one the receipts actually
148
+ # support: they carry generated_at (verified on all 9), and loki_version tracks
149
+ # releases rather than this fix. A frozen-vs-live split ALONE would file a newly
150
+ # broken receipt under history, which is the green-wash this guards against.
151
+ BASE_SHA_FIX_UTC = "2026-08-07T13:39:32Z" # 99ce689d, committed 09:39:32 -04:00
152
+
153
+
154
+ def classify_unanchored(reason, generated_at):
155
+ """Bucket an unanchored receipt: historical, regression, or live.
156
+
157
+ Returns one of:
158
+ "by_design" -- unanchorable and CORRECT: a greenfield run with no earlier
159
+ commit, or a receipt over uncommitted work. Never a defect,
160
+ at any date, so recency is not applied.
161
+ "historical" -- the generator failed to record an anchor, in a receipt
162
+ written BEFORE the fix. History, and never anchorable now.
163
+ "regression" -- the same generator failure AFTER the fix. The anchor is
164
+ being dropped again. This is the only bucket that alarms.
165
+ "live" -- the receipt is fine; the environment cannot resolve it yet.
166
+ """
167
+ if reason in FROZEN_BY_DESIGN_REASONS:
168
+ return "by_design"
169
+ if reason not in FROZEN_GENERATOR_REASONS:
170
+ return "live"
171
+ # No timestamp means we cannot place it relative to the fix. Refuse to call
172
+ # it history, because that is the direction that hides a regression.
173
+ #
174
+ # ponytail: lexicographic compare, correct only for the "...Z" ISO form the
175
+ # generator writes (verified on all 9 receipts). An offset-form timestamp
176
+ # would sort below the constant and read as historical -- the hiding
177
+ # direction. Parse properly only if a generator ever emits offsets.
178
+ if not generated_at:
179
+ return "regression"
180
+ return "historical" if generated_at < BASE_SHA_FIX_UTC else "regression"
181
+
182
+
183
+ def _git(args, cwd, timeout=30):
184
+ """Run a git command read-only. Returns (rc, stdout). Never raises.
185
+
186
+ Failure is data here, not an exception: an unreachable sha and a shallow
187
+ clone both surface as a non-zero rc, and each caller turns that into an
188
+ explicit UNKNOWN with a reason rather than a silent zero.
189
+ """
190
+ try:
191
+ p = subprocess.run(
192
+ ["git"] + list(args),
193
+ cwd=cwd, capture_output=True, text=True, timeout=timeout,
194
+ )
195
+ return p.returncode, p.stdout
196
+ except (subprocess.TimeoutExpired, OSError) as exc:
197
+ return 1, f"__error__ {type(exc).__name__}: {exc}"
198
+
199
+
200
+ def _commit_exists(sha, cwd):
201
+ """True only if the object is present AND is a commit."""
202
+ if not sha:
203
+ return False
204
+ rc, _ = _git(["cat-file", "-e", f"{sha}^{{commit}}"], cwd)
205
+ return rc == 0
206
+
207
+
208
+ def resolve_anchor(base_sha, head_sha, files, cwd):
209
+ """Decide whether this receipt can be followed at all. Returns (state, reason).
210
+
211
+ Cheap sha algebra, run BEFORE any metric. Only "anchored" produces numbers;
212
+ every other outcome is UNKNOWN with a named reason from ANCHOR_REASONS. See
213
+ the ANCHOR_REASONS comment for the measured evidence that makes this gate
214
+ mandatory rather than defensive.
215
+ """
216
+ rc, _ = _git(["rev-parse", "--git-dir"], cwd)
217
+ if rc != 0:
218
+ return "unanchored", "no_git"
219
+ if not head_sha:
220
+ return "unanchored", "head_sha_empty"
221
+ if not base_sha:
222
+ return "unanchored", "base_sha_empty"
223
+ if base_sha == EMPTY_TREE_SHA or base_sha.startswith(EMPTY_TREE_SHA[:12]):
224
+ return "unanchored", "greenfield_no_baseline"
225
+ if base_sha == head_sha:
226
+ return "unanchored", "change_not_committed"
227
+ if not _commit_exists(base_sha, cwd) or not _commit_exists(head_sha, cwd):
228
+ return "unanchored", "sha_not_in_history"
229
+
230
+ # Unreachable is NOT the same as reverted, and conflating them would invent
231
+ # failures out of unmerged branches. merge-base separates the two.
232
+ rc, _ = _git(["merge-base", "--is-ancestor", head_sha, "HEAD"], cwd)
233
+ if rc != 0:
234
+ return "unanchored", "not_reachable_from_head"
235
+
236
+ # Final identity check: the range must actually produce the receipt's files.
237
+ # This is what overlap-matching cannot do -- it demands the SET match, so a
238
+ # coincidental 2-file overlap can never masquerade as the same change.
239
+ rc, out = _git(["diff", "--name-only", f"{base_sha}..{head_sha}"], cwd)
240
+ if rc != 0:
241
+ return "unanchored", "sha_not_in_history"
242
+ range_files = {l.strip() for l in out.splitlines() if l.strip()}
243
+ receipt_files = {f.get("path") for f in files
244
+ if isinstance(f, dict) and f.get("path")}
245
+ if receipt_files and range_files != receipt_files:
246
+ return "unanchored", "diff_range_mismatch"
247
+
248
+ return "anchored", None
249
+
250
+
251
+ def detect_revert(head_sha, cwd):
252
+ """Was head_sha reverted? Returns (status, evidence).
253
+
254
+ FACT, not inference. `git revert` writes the canonical trailer
255
+ "This reverts commit <full-sha>." into the message body. We search for that
256
+ exact trailer, so a commit that merely mentions the sha in prose -- a
257
+ follow-up, a doc reference, a changelog entry -- does not count.
258
+ """
259
+ if not _commit_exists(head_sha, cwd):
260
+ return UNKNOWN, "head_sha is not a reachable commit in this clone"
261
+
262
+ # --fixed-strings so a sha can never be read as a regex; --all so a revert
263
+ # on any branch counts, not only the current one.
264
+ rc, out = _git(
265
+ ["log", "--all", "--fixed-strings",
266
+ f"--grep=This reverts commit {head_sha}",
267
+ "--format=%H %s"],
268
+ cwd,
269
+ )
270
+ if rc != 0:
271
+ return UNKNOWN, "git log failed while searching for a revert trailer"
272
+ hits = [l for l in out.splitlines() if l.strip()]
273
+ if hits:
274
+ return True, hits
275
+ return False, []
276
+
277
+
278
+ def line_survival(head_sha, files, cwd):
279
+ """How many lines introduced by head_sha still survive at HEAD.
280
+
281
+ FACT via `git blame --porcelain`: every line in the current file carries the
282
+ sha of the commit that last touched it. Lines still attributed to head_sha
283
+ are lines this change introduced that nothing has since replaced.
284
+
285
+ Returns a dict. Files that cannot be blamed (deleted, renamed away, binary)
286
+ are counted in `unknown_files` rather than being scored as zero survival --
287
+ a deleted file is not evidence that the work was wrong.
288
+ """
289
+ if not _commit_exists(head_sha, cwd):
290
+ return {"status": UNKNOWN,
291
+ "reason": "head_sha is not a reachable commit in this clone"}
292
+
293
+ surviving = 0
294
+ unknown_files = []
295
+ checked = 0
296
+
297
+ for f in files:
298
+ path = f.get("path") if isinstance(f, dict) else f
299
+ if not path:
300
+ continue
301
+ # A file removed after the fact cannot be blamed. That is UNKNOWN, not 0.
302
+ if not os.path.exists(os.path.join(cwd, path)):
303
+ unknown_files.append({"path": path, "reason": "not present at HEAD"})
304
+ continue
305
+ rc, out = _git(["blame", "--porcelain", "--", path], cwd, timeout=60)
306
+ if rc != 0:
307
+ unknown_files.append({"path": path, "reason": "blame failed"})
308
+ continue
309
+ checked += 1
310
+ # In porcelain output a line beginning with a 40-hex sha starts a block;
311
+ # the sha is the commit that introduced that line.
312
+ for line in out.splitlines():
313
+ parts = line.split(" ", 1)
314
+ token = parts[0]
315
+ if len(token) == 40 and all(c in "0123456789abcdef" for c in token):
316
+ if token == head_sha:
317
+ surviving += 1
318
+
319
+ return {
320
+ "status": "measured" if checked else UNKNOWN,
321
+ "reason": None if checked else "no file from this change could be blamed",
322
+ "surviving_lines": surviving if checked else UNKNOWN,
323
+ "files_checked": checked,
324
+ "files_unknown": unknown_files,
325
+ }
326
+
327
+
328
+ def detect_rework(head_sha, files, cwd, since_days=None):
329
+ """Lines this change introduced that a LATER commit replaced.
330
+
331
+ This is the signal most easily faked. "The file was touched again" is NOT
332
+ rework -- an unrelated feature in the same file would inflate it, and a
333
+ leader acting on that number would be acting on noise. So this is scoped to
334
+ LINES: a later commit counts only where it replaced a line that head_sha
335
+ introduced, which `git blame` establishes by attribution.
336
+
337
+ Derived, deliberately, from the same blame data as line_survival: lines
338
+ introduced minus lines still attributed. That keeps the two numbers
339
+ arithmetically consistent instead of two estimates that can disagree.
340
+ """
341
+ if not _commit_exists(head_sha, cwd):
342
+ return {"status": UNKNOWN,
343
+ "reason": "head_sha is not a reachable commit in this clone"}
344
+
345
+ introduced = 0
346
+ for f in files:
347
+ if isinstance(f, dict):
348
+ ins = f.get("insertions")
349
+ if isinstance(ins, int):
350
+ introduced += ins
351
+
352
+ if introduced == 0:
353
+ return {"status": UNKNOWN,
354
+ "reason": "receipt records no insertion counts to compare against"}
355
+
356
+ surv = line_survival(head_sha, files, cwd)
357
+ if surv.get("status") != "measured":
358
+ return {"status": UNKNOWN, "reason": surv.get("reason") or "survival unmeasurable"}
359
+
360
+ survived = surv["surviving_lines"]
361
+ # Clamp: blame can attribute MORE lines than the receipt counted when a file
362
+ # was reformatted, so a negative would be an artifact, not a measurement.
363
+ replaced = max(0, introduced - survived)
364
+ return {
365
+ "status": "measured",
366
+ "lines_introduced": introduced,
367
+ "lines_surviving": survived,
368
+ "lines_replaced": replaced,
369
+ "rework_ratio": round(replaced / introduced, 4) if introduced else UNKNOWN,
370
+ }
371
+
372
+
373
+ def outcome_for_receipt(proof_path, cwd):
374
+ """Compute the full outcome record for one receipt."""
375
+ try:
376
+ with open(proof_path, "r", encoding="utf-8") as fh:
377
+ proof = json.load(fh)
378
+ except (OSError, ValueError) as exc:
379
+ return {"status": UNKNOWN, "reason": f"receipt unreadable: {exc}",
380
+ "proof_path": proof_path}
381
+
382
+ run_id = proof.get("run_id") or os.path.basename(os.path.dirname(proof_path))
383
+ git_facts = (proof.get("facts") or {}).get("git") or {}
384
+ head_sha = git_facts.get("head_sha")
385
+ base_sha = git_facts.get("base_sha")
386
+ files = ((git_facts.get("diff") or {}).get("files")) or []
387
+
388
+ rec = {
389
+ "run_id": run_id,
390
+ "generated_at": proof.get("generated_at"),
391
+ "headline": (proof.get("verification") or {}).get("headline")
392
+ or proof.get("headline"),
393
+ "base_sha": base_sha,
394
+ "head_sha": head_sha,
395
+ "files_changed": len(files),
396
+ }
397
+
398
+ # THE GATE. No metric runs unless sha algebra proves base..head IS this
399
+ # change. Without it, 8 of this repo's 9 receipts would have been scored
400
+ # against an unrelated 2-file commit and reported as a change-failure rate.
401
+ state, reason = resolve_anchor(base_sha, head_sha, files, cwd)
402
+ rec["anchor"] = {"state": state, "reason": reason}
403
+ if state != "anchored":
404
+ # An old receipt is not a regression. Classify so a reader can tell a
405
+ # frozen pre-fix receipt from a generator that started dropping anchors.
406
+ rec["anchor"]["klass"] = classify_unanchored(reason, rec["generated_at"])
407
+ rec["outcome"] = UNKNOWN
408
+ rec["reason"] = ANCHOR_REASONS.get(reason, reason or "not anchored")
409
+ rec["commands"] = [
410
+ f"git merge-base --is-ancestor {head_sha or '<head_sha>'} HEAD",
411
+ f"git diff --name-only {base_sha or '<base_sha>'}..{head_sha or '<head_sha>'}",
412
+ ]
413
+ return rec
414
+
415
+ reverted, revert_evidence = detect_revert(head_sha, cwd)
416
+ rec["reverted"] = reverted
417
+ if revert_evidence:
418
+ rec["revert_evidence"] = revert_evidence
419
+ rec["survival"] = line_survival(head_sha, files, cwd)
420
+ rec["rework"] = detect_rework(head_sha, files, cwd)
421
+
422
+ # The verdict is deliberately coarse and refuses to guess.
423
+ if reverted is True:
424
+ rec["outcome"] = "REVERTED"
425
+ elif reverted is UNKNOWN:
426
+ rec["outcome"] = UNKNOWN
427
+ rec["reason"] = "could not determine whether the change was reverted"
428
+ elif rec["rework"].get("status") == "measured":
429
+ rec["outcome"] = "SURVIVED"
430
+ else:
431
+ rec["outcome"] = UNKNOWN
432
+ rec["reason"] = rec["rework"].get("reason") or "outcome unmeasurable"
433
+ return rec
434
+
435
+
436
+ def collect(loki_dir, cwd, run_id=None):
437
+ """Compute outcomes for every receipt under <loki_dir>/proofs."""
438
+ proofs_dir = os.path.join(loki_dir, "proofs")
439
+ records = []
440
+ if not os.path.isdir(proofs_dir):
441
+ return records, "no .loki/proofs directory in this project"
442
+
443
+ for entry in sorted(os.listdir(proofs_dir)):
444
+ if run_id and entry != run_id:
445
+ continue
446
+ p = os.path.join(proofs_dir, entry, "proof.json")
447
+ if os.path.isfile(p):
448
+ records.append(outcome_for_receipt(p, cwd))
449
+ if not records:
450
+ return records, "no receipts found (run a build first, then re-run this)"
451
+ return records, None
452
+
453
+
454
+ def summarize(records):
455
+ """Aggregate. UNKNOWN is carried, never folded into a pass.
456
+
457
+ change_failure_rate is computed over MEASURED receipts only, and the count of
458
+ unmeasured ones is reported alongside it. A rate that quietly treated
459
+ unmeasurable changes as successes would be precisely the false-green this
460
+ tool exists to refuse -- so if nothing is measurable the rate is UNKNOWN, not
461
+ 0.0, no matter how good that would look.
462
+ """
463
+ total = len(records)
464
+ measured = [r for r in records if r.get("outcome") in ("SURVIVED", "REVERTED")]
465
+ reverted = [r for r in measured if r.get("outcome") == "REVERTED"]
466
+ unknown = [r for r in records if r.get("outcome") == UNKNOWN]
467
+
468
+ # Why receipts could not be measured is the most useful thing this tool
469
+ # prints when nothing is anchored. Without the distribution the report reads
470
+ # as "no data" instead of "your receipts are not recording a landed sha",
471
+ # which is an actionable defect.
472
+ reasons = {}
473
+ klasses = {"by_design": 0, "historical": 0, "regression": 0, "live": 0}
474
+ for r in records:
475
+ a = r.get("anchor") or {}
476
+ if a.get("state") and a["state"] != "anchored":
477
+ key = a.get("reason") or "unknown"
478
+ reasons[key] = reasons.get(key, 0) + 1
479
+ k = a.get("klass")
480
+ if k in klasses:
481
+ klasses[k] += 1
482
+
483
+ summary = {
484
+ "receipts_total": total,
485
+ "receipts_measured": len(measured),
486
+ "receipts_unknown": len(unknown),
487
+ "reverted": len(reverted),
488
+ "unanchored_reasons": reasons,
489
+ # The count that turns an alarming zero into an explained one. A
490
+ # regression here is the only bucket that warrants action.
491
+ "unanchored_by_design": klasses["by_design"],
492
+ "unanchored_historical": klasses["historical"],
493
+ "unanchored_regression": klasses["regression"],
494
+ "unanchored_live": klasses["live"],
495
+ }
496
+ if measured:
497
+ summary["change_failure_rate"] = round(len(reverted) / len(measured), 4)
498
+ else:
499
+ summary["change_failure_rate"] = UNKNOWN
500
+ summary["change_failure_rate_reason"] = (
501
+ "no receipt could be measured, so a rate would be fabricated")
502
+
503
+ ratios = [r["rework"]["rework_ratio"] for r in records
504
+ if isinstance(r.get("rework"), dict)
505
+ and r["rework"].get("status") == "measured"
506
+ and isinstance(r["rework"].get("rework_ratio"), (int, float))]
507
+ if ratios:
508
+ summary["rework_ratio_avg"] = round(sum(ratios) / len(ratios), 4)
509
+ else:
510
+ summary["rework_ratio_avg"] = UNKNOWN
511
+ return summary
512
+
513
+
514
+ def render_text(records, summary, note=None):
515
+ out = []
516
+ out.append("Outcome Ledger -- what happened to the work AFTER the receipt")
517
+ out.append("")
518
+ if note:
519
+ out.append(f" {note}")
520
+ out.append("")
521
+ return "\n".join(out)
522
+
523
+ for r in records:
524
+ line = f" {r.get('run_id', '?')[:34]:36}"
525
+ oc = r.get("outcome", UNKNOWN)
526
+ line += f"{oc:10}"
527
+ rw = r.get("rework")
528
+ if isinstance(rw, dict) and rw.get("status") == "measured":
529
+ line += (f" lines {rw['lines_surviving']}/{rw['lines_introduced']} survive"
530
+ f" rework {rw['rework_ratio']}")
531
+ elif oc == UNKNOWN and r.get("reason"):
532
+ line += f" ({r['reason']})"
533
+ out.append(line)
534
+
535
+ out.append("")
536
+ out.append(f" ANCHORED {summary['receipts_measured']} of "
537
+ f"{summary['receipts_total']} receipts.")
538
+ reasons = summary.get("unanchored_reasons") or {}
539
+ if reasons:
540
+ out.append("")
541
+ out.append(" Why not anchored (a receipt must prove base..head IS the change):")
542
+ for k, n in sorted(reasons.items(), key=lambda kv: -kv[1]):
543
+ out.append(f" {k:24} {n:3} {ANCHOR_REASONS.get(k, '')}")
544
+
545
+ # Without this, ANCHORED 0 of N reads as a regression when it is history.
546
+ hist = summary.get("unanchored_historical", 0)
547
+ regr = summary.get("unanchored_regression", 0)
548
+ live = summary.get("unanchored_live", 0)
549
+ design = summary.get("unanchored_by_design", 0)
550
+ if hist or regr or live or design:
551
+ out.append("")
552
+ if design:
553
+ out.append(f" {design} by design: a greenfield run or uncommitted"
554
+ f" work has no baseline to diff against. Correct, not a"
555
+ f" defect, and never anchorable.")
556
+ if hist:
557
+ out.append(f" {hist} historical: written before the base_sha fix"
558
+ f" ({BASE_SHA_FIX_UTC[:10]}); the receipt itself has no"
559
+ f" baseline, so these can never anchor. Not a defect.")
560
+ if live:
561
+ out.append(f" {live} live: the receipt records real shas; this"
562
+ f" clone cannot resolve them yet. May anchor later.")
563
+ if regr:
564
+ out.append(f" {regr} REGRESSION: written AFTER the fix and still"
565
+ f" missing a baseline. The generator is dropping the"
566
+ f" anchor -- this one is a defect.")
567
+ out.append("")
568
+ cfr = summary.get("change_failure_rate")
569
+ if cfr == UNKNOWN:
570
+ out.append(f" change-failure rate: UNKNOWN"
571
+ f" ({summary.get('change_failure_rate_reason', '')})")
572
+ else:
573
+ out.append(f" change-failure rate: {cfr}"
574
+ f" (over {summary['receipts_measured']} measured receipts)")
575
+ out.append(f" average rework ratio: {summary.get('rework_ratio_avg')}")
576
+ out.append("")
577
+ out.append(" Every number above is derivable from git. Check any of them:")
578
+ out.append(" git log --all --fixed-strings --grep='This reverts commit <head_sha>'")
579
+ out.append(" git blame --porcelain -- <path>")
580
+ return "\n".join(out)
581
+
582
+
583
+ def main(argv):
584
+ as_json = "--json" in argv
585
+ run_id = None
586
+ if "--run-id" in argv:
587
+ i = argv.index("--run-id")
588
+ if i + 1 < len(argv):
589
+ run_id = argv[i + 1]
590
+
591
+ cwd = os.environ.get("LOKI_OUTCOMES_CWD") or os.getcwd()
592
+ loki_dir = os.environ.get("LOKI_DIR") or os.path.join(cwd, ".loki")
593
+
594
+ records, note = collect(loki_dir, cwd, run_id=run_id)
595
+ summary = summarize(records)
596
+
597
+ if as_json:
598
+ print(json.dumps({
599
+ "schema_version": SCHEMA_VERSION,
600
+ "generated_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
601
+ "note": note,
602
+ "summary": summary,
603
+ "receipts": records,
604
+ "how_to_verify": [
605
+ "git log --all --fixed-strings --grep='This reverts commit <head_sha>'",
606
+ "git blame --porcelain -- <path>",
607
+ ],
608
+ }, indent=2))
609
+ else:
610
+ print(render_text(records, summary, note=note))
611
+
612
+ # Exit 3 when there is nothing to measure, mirroring `loki proof releases`:
613
+ # an empty result is a real answer, distinct from a failure.
614
+ if note:
615
+ return 3
616
+ return 0
617
+
618
+
619
+ if __name__ == "__main__":
620
+ sys.exit(main(sys.argv[1:]))