loki-mode 9.12.6 → 9.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,498 @@
1
+ #!/usr/bin/env python3
2
+ """Outcome Ledger: did the change turn out to be RIGHT, not merely that it happened.
3
+
4
+ WHY THIS EXISTS. Every competing agent reports VOLUME. Measured against their own
5
+ published docs: Factory AI's analytics expose files_created, files_edited,
6
+ lines_modified, git_commits, git_prs_created, tokens and DAU -- and no defect
7
+ rate, no rework rate, no revert rate, no change-failure rate. Their telemetry doc
8
+ states outright that correlating usage with delivery outcomes is left to the
9
+ customer's own stack. Devin's security page concedes the agent "can still
10
+ experience hallucinations, introduce bugs into code" and points you at your own
11
+ code review and branch protection. So an engineering leader using either can
12
+ prove the agent was BUSY. Neither can show it was RIGHT.
13
+
14
+ This module answers the other question, from local git history alone: after the
15
+ receipt was written, did the change survive?
16
+
17
+ WHAT MAKES A SIGNAL DEFENSIBLE HERE. The temptation is to count "the file was
18
+ touched again" as rework. That is a misleading signal -- an unrelated feature
19
+ landing in the same file would inflate it, and a leader acting on that number
20
+ would be acting on noise. So each signal below is either a deterministic fact or
21
+ it reports UNKNOWN:
22
+
23
+ reverted FACT. `git revert` writes "This reverts commit <sha>" into the
24
+ message. That is a machine-parseable link, not an inference.
25
+ line_survival FACT. `git blame --porcelain` attributes every surviving line to
26
+ the commit that introduced it. Counting lines still attributed to
27
+ head_sha is a measurement, not an estimate.
28
+ reworked MEASURED, line-scoped. Only lines this change INTRODUCED and that
29
+ a LATER commit replaced count as rework. A file touched elsewhere
30
+ does not.
31
+ UNKNOWN Any case we cannot compute: absent head_sha, unreachable commit
32
+ (never merged, shallow clone, pruned), or a file since deleted.
33
+ Never reported as zero, never as a pass.
34
+
35
+ The last rule is the whole point. A change-failure rate that silently scores
36
+ unmeasurable cases as successes is the exact false-green this product exists to
37
+ refuse. `loki proof` already sets the house convention -- its help says
38
+ "version_is_ahead reads UNKNOWN when it cannot be computed" -- and this inherits it.
39
+
40
+ READ-ONLY. Never writes to the repo under analysis. Every number it prints comes
41
+ from a git command the reader can rerun by hand; --json emits those commands.
42
+ """
43
+
44
+ from __future__ import annotations
45
+
46
+ import json
47
+ import os
48
+ import subprocess
49
+ import sys
50
+ from datetime import datetime, timezone
51
+
52
+ SCHEMA_VERSION = "1.0"
53
+
54
+ # A status that is not a number. Kept as a module constant so a caller can never
55
+ # accidentally coerce it to 0 in a tally.
56
+ UNKNOWN = "UNKNOWN"
57
+
58
+ # git's empty-tree object. proof-generator falls back to it for a greenfield run,
59
+ # so a receipt carrying it has no real baseline to diff against.
60
+ EMPTY_TREE_SHA = "4b825dc642cb6eb9a060e54bf8d69288fbee4904"
61
+
62
+ # THE ANCHOR GATE, and the reason this module is not a metrics generator.
63
+ #
64
+ # `facts.git.head_sha` is `git rev-parse HEAD` AT RECEIPT-GENERATION TIME. When
65
+ # the agent's work is still uncommitted -- the normal case -- that is the run's
66
+ # STARTING commit, not the commit the change became. Measured on this repo's own
67
+ # 9 receipts:
68
+ #
69
+ # 8 of 9 carry base_sha="" and head_sha=1385e71c. That commit touches TWO files
70
+ # (providers/codex.sh, tests/test-provider-degraded-mode.sh) while the receipts
71
+ # attest to EIGHT including autonomy/telemetry.sh. They are not the same change.
72
+ # The 9th carries base_sha=4b825dc6 (empty tree) and 4017 files.
73
+ #
74
+ # Following head_sha regardless would have attributed one commit's fate to eight
75
+ # unrelated runs and reported it as a change-failure rate. That is precisely the
76
+ # fabricated metric this product exists to refuse, and it would have been
77
+ # invisible in the output.
78
+ #
79
+ # File-overlap was tested as a fallback discriminator and REJECTED: the bad
80
+ # receipts overlap that commit by 2 files, so any overlap>0 rule marks all eight
81
+ # as anchored. Overlap is not evidence of identity.
82
+ #
83
+ # So: a receipt yields numbers only when sha algebra proves the range is the
84
+ # change. Everything else is UNKNOWN with a named reason.
85
+ ANCHOR_REASONS = {
86
+ "no_git": "not a git repository",
87
+ "head_sha_empty": "receipt records no head_sha",
88
+ "base_sha_empty": "run baseline was not recorded at generation",
89
+ "greenfield_no_baseline": "base_sha is the empty tree, so there is no baseline",
90
+ "change_not_committed": "base_sha == head_sha, so the work was never committed",
91
+ "sha_not_in_history": "a sha is not resolvable in this clone",
92
+ "not_reachable_from_head": "head_sha is not an ancestor of HEAD (never merged)",
93
+ "diff_range_mismatch": "the base..head diff does not match the receipt's file set",
94
+ }
95
+
96
+
97
+ def _git(args, cwd, timeout=30):
98
+ """Run a git command read-only. Returns (rc, stdout). Never raises.
99
+
100
+ Failure is data here, not an exception: an unreachable sha and a shallow
101
+ clone both surface as a non-zero rc, and each caller turns that into an
102
+ explicit UNKNOWN with a reason rather than a silent zero.
103
+ """
104
+ try:
105
+ p = subprocess.run(
106
+ ["git"] + list(args),
107
+ cwd=cwd, capture_output=True, text=True, timeout=timeout,
108
+ )
109
+ return p.returncode, p.stdout
110
+ except (subprocess.TimeoutExpired, OSError) as exc:
111
+ return 1, f"__error__ {type(exc).__name__}: {exc}"
112
+
113
+
114
+ def _commit_exists(sha, cwd):
115
+ """True only if the object is present AND is a commit."""
116
+ if not sha:
117
+ return False
118
+ rc, _ = _git(["cat-file", "-e", f"{sha}^{{commit}}"], cwd)
119
+ return rc == 0
120
+
121
+
122
+ def resolve_anchor(base_sha, head_sha, files, cwd):
123
+ """Decide whether this receipt can be followed at all. Returns (state, reason).
124
+
125
+ Cheap sha algebra, run BEFORE any metric. Only "anchored" produces numbers;
126
+ every other outcome is UNKNOWN with a named reason from ANCHOR_REASONS. See
127
+ the ANCHOR_REASONS comment for the measured evidence that makes this gate
128
+ mandatory rather than defensive.
129
+ """
130
+ rc, _ = _git(["rev-parse", "--git-dir"], cwd)
131
+ if rc != 0:
132
+ return "unanchored", "no_git"
133
+ if not head_sha:
134
+ return "unanchored", "head_sha_empty"
135
+ if not base_sha:
136
+ return "unanchored", "base_sha_empty"
137
+ if base_sha == EMPTY_TREE_SHA or base_sha.startswith(EMPTY_TREE_SHA[:12]):
138
+ return "unanchored", "greenfield_no_baseline"
139
+ if base_sha == head_sha:
140
+ return "unanchored", "change_not_committed"
141
+ if not _commit_exists(base_sha, cwd) or not _commit_exists(head_sha, cwd):
142
+ return "unanchored", "sha_not_in_history"
143
+
144
+ # Unreachable is NOT the same as reverted, and conflating them would invent
145
+ # failures out of unmerged branches. merge-base separates the two.
146
+ rc, _ = _git(["merge-base", "--is-ancestor", head_sha, "HEAD"], cwd)
147
+ if rc != 0:
148
+ return "unanchored", "not_reachable_from_head"
149
+
150
+ # Final identity check: the range must actually produce the receipt's files.
151
+ # This is what overlap-matching cannot do -- it demands the SET match, so a
152
+ # coincidental 2-file overlap can never masquerade as the same change.
153
+ rc, out = _git(["diff", "--name-only", f"{base_sha}..{head_sha}"], cwd)
154
+ if rc != 0:
155
+ return "unanchored", "sha_not_in_history"
156
+ range_files = {l.strip() for l in out.splitlines() if l.strip()}
157
+ receipt_files = {f.get("path") for f in files
158
+ if isinstance(f, dict) and f.get("path")}
159
+ if receipt_files and range_files != receipt_files:
160
+ return "unanchored", "diff_range_mismatch"
161
+
162
+ return "anchored", None
163
+
164
+
165
+ def detect_revert(head_sha, cwd):
166
+ """Was head_sha reverted? Returns (status, evidence).
167
+
168
+ FACT, not inference. `git revert` writes the canonical trailer
169
+ "This reverts commit <full-sha>." into the message body. We search for that
170
+ exact trailer, so a commit that merely mentions the sha in prose -- a
171
+ follow-up, a doc reference, a changelog entry -- does not count.
172
+ """
173
+ if not _commit_exists(head_sha, cwd):
174
+ return UNKNOWN, "head_sha is not a reachable commit in this clone"
175
+
176
+ # --fixed-strings so a sha can never be read as a regex; --all so a revert
177
+ # on any branch counts, not only the current one.
178
+ rc, out = _git(
179
+ ["log", "--all", "--fixed-strings",
180
+ f"--grep=This reverts commit {head_sha}",
181
+ "--format=%H %s"],
182
+ cwd,
183
+ )
184
+ if rc != 0:
185
+ return UNKNOWN, "git log failed while searching for a revert trailer"
186
+ hits = [l for l in out.splitlines() if l.strip()]
187
+ if hits:
188
+ return True, hits
189
+ return False, []
190
+
191
+
192
+ def line_survival(head_sha, files, cwd):
193
+ """How many lines introduced by head_sha still survive at HEAD.
194
+
195
+ FACT via `git blame --porcelain`: every line in the current file carries the
196
+ sha of the commit that last touched it. Lines still attributed to head_sha
197
+ are lines this change introduced that nothing has since replaced.
198
+
199
+ Returns a dict. Files that cannot be blamed (deleted, renamed away, binary)
200
+ are counted in `unknown_files` rather than being scored as zero survival --
201
+ a deleted file is not evidence that the work was wrong.
202
+ """
203
+ if not _commit_exists(head_sha, cwd):
204
+ return {"status": UNKNOWN,
205
+ "reason": "head_sha is not a reachable commit in this clone"}
206
+
207
+ surviving = 0
208
+ unknown_files = []
209
+ checked = 0
210
+
211
+ for f in files:
212
+ path = f.get("path") if isinstance(f, dict) else f
213
+ if not path:
214
+ continue
215
+ # A file removed after the fact cannot be blamed. That is UNKNOWN, not 0.
216
+ if not os.path.exists(os.path.join(cwd, path)):
217
+ unknown_files.append({"path": path, "reason": "not present at HEAD"})
218
+ continue
219
+ rc, out = _git(["blame", "--porcelain", "--", path], cwd, timeout=60)
220
+ if rc != 0:
221
+ unknown_files.append({"path": path, "reason": "blame failed"})
222
+ continue
223
+ checked += 1
224
+ # In porcelain output a line beginning with a 40-hex sha starts a block;
225
+ # the sha is the commit that introduced that line.
226
+ for line in out.splitlines():
227
+ parts = line.split(" ", 1)
228
+ token = parts[0]
229
+ if len(token) == 40 and all(c in "0123456789abcdef" for c in token):
230
+ if token == head_sha:
231
+ surviving += 1
232
+
233
+ return {
234
+ "status": "measured" if checked else UNKNOWN,
235
+ "reason": None if checked else "no file from this change could be blamed",
236
+ "surviving_lines": surviving if checked else UNKNOWN,
237
+ "files_checked": checked,
238
+ "files_unknown": unknown_files,
239
+ }
240
+
241
+
242
+ def detect_rework(head_sha, files, cwd, since_days=None):
243
+ """Lines this change introduced that a LATER commit replaced.
244
+
245
+ This is the signal most easily faked. "The file was touched again" is NOT
246
+ rework -- an unrelated feature in the same file would inflate it, and a
247
+ leader acting on that number would be acting on noise. So this is scoped to
248
+ LINES: a later commit counts only where it replaced a line that head_sha
249
+ introduced, which `git blame` establishes by attribution.
250
+
251
+ Derived, deliberately, from the same blame data as line_survival: lines
252
+ introduced minus lines still attributed. That keeps the two numbers
253
+ arithmetically consistent instead of two estimates that can disagree.
254
+ """
255
+ if not _commit_exists(head_sha, cwd):
256
+ return {"status": UNKNOWN,
257
+ "reason": "head_sha is not a reachable commit in this clone"}
258
+
259
+ introduced = 0
260
+ for f in files:
261
+ if isinstance(f, dict):
262
+ ins = f.get("insertions")
263
+ if isinstance(ins, int):
264
+ introduced += ins
265
+
266
+ if introduced == 0:
267
+ return {"status": UNKNOWN,
268
+ "reason": "receipt records no insertion counts to compare against"}
269
+
270
+ surv = line_survival(head_sha, files, cwd)
271
+ if surv.get("status") != "measured":
272
+ return {"status": UNKNOWN, "reason": surv.get("reason") or "survival unmeasurable"}
273
+
274
+ survived = surv["surviving_lines"]
275
+ # Clamp: blame can attribute MORE lines than the receipt counted when a file
276
+ # was reformatted, so a negative would be an artifact, not a measurement.
277
+ replaced = max(0, introduced - survived)
278
+ return {
279
+ "status": "measured",
280
+ "lines_introduced": introduced,
281
+ "lines_surviving": survived,
282
+ "lines_replaced": replaced,
283
+ "rework_ratio": round(replaced / introduced, 4) if introduced else UNKNOWN,
284
+ }
285
+
286
+
287
+ def outcome_for_receipt(proof_path, cwd):
288
+ """Compute the full outcome record for one receipt."""
289
+ try:
290
+ with open(proof_path, "r", encoding="utf-8") as fh:
291
+ proof = json.load(fh)
292
+ except (OSError, ValueError) as exc:
293
+ return {"status": UNKNOWN, "reason": f"receipt unreadable: {exc}",
294
+ "proof_path": proof_path}
295
+
296
+ run_id = proof.get("run_id") or os.path.basename(os.path.dirname(proof_path))
297
+ git_facts = (proof.get("facts") or {}).get("git") or {}
298
+ head_sha = git_facts.get("head_sha")
299
+ base_sha = git_facts.get("base_sha")
300
+ files = ((git_facts.get("diff") or {}).get("files")) or []
301
+
302
+ rec = {
303
+ "run_id": run_id,
304
+ "generated_at": proof.get("generated_at"),
305
+ "headline": (proof.get("verification") or {}).get("headline")
306
+ or proof.get("headline"),
307
+ "base_sha": base_sha,
308
+ "head_sha": head_sha,
309
+ "files_changed": len(files),
310
+ }
311
+
312
+ # THE GATE. No metric runs unless sha algebra proves base..head IS this
313
+ # change. Without it, 8 of this repo's 9 receipts would have been scored
314
+ # against an unrelated 2-file commit and reported as a change-failure rate.
315
+ state, reason = resolve_anchor(base_sha, head_sha, files, cwd)
316
+ rec["anchor"] = {"state": state, "reason": reason}
317
+ if state != "anchored":
318
+ rec["outcome"] = UNKNOWN
319
+ rec["reason"] = ANCHOR_REASONS.get(reason, reason or "not anchored")
320
+ rec["commands"] = [
321
+ f"git merge-base --is-ancestor {head_sha or '<head_sha>'} HEAD",
322
+ f"git diff --name-only {base_sha or '<base_sha>'}..{head_sha or '<head_sha>'}",
323
+ ]
324
+ return rec
325
+
326
+ reverted, revert_evidence = detect_revert(head_sha, cwd)
327
+ rec["reverted"] = reverted
328
+ if revert_evidence:
329
+ rec["revert_evidence"] = revert_evidence
330
+ rec["survival"] = line_survival(head_sha, files, cwd)
331
+ rec["rework"] = detect_rework(head_sha, files, cwd)
332
+
333
+ # The verdict is deliberately coarse and refuses to guess.
334
+ if reverted is True:
335
+ rec["outcome"] = "REVERTED"
336
+ elif reverted is UNKNOWN:
337
+ rec["outcome"] = UNKNOWN
338
+ rec["reason"] = "could not determine whether the change was reverted"
339
+ elif rec["rework"].get("status") == "measured":
340
+ rec["outcome"] = "SURVIVED"
341
+ else:
342
+ rec["outcome"] = UNKNOWN
343
+ rec["reason"] = rec["rework"].get("reason") or "outcome unmeasurable"
344
+ return rec
345
+
346
+
347
+ def collect(loki_dir, cwd, run_id=None):
348
+ """Compute outcomes for every receipt under <loki_dir>/proofs."""
349
+ proofs_dir = os.path.join(loki_dir, "proofs")
350
+ records = []
351
+ if not os.path.isdir(proofs_dir):
352
+ return records, "no .loki/proofs directory in this project"
353
+
354
+ for entry in sorted(os.listdir(proofs_dir)):
355
+ if run_id and entry != run_id:
356
+ continue
357
+ p = os.path.join(proofs_dir, entry, "proof.json")
358
+ if os.path.isfile(p):
359
+ records.append(outcome_for_receipt(p, cwd))
360
+ if not records:
361
+ return records, "no receipts found (run a build first, then re-run this)"
362
+ return records, None
363
+
364
+
365
+ def summarize(records):
366
+ """Aggregate. UNKNOWN is carried, never folded into a pass.
367
+
368
+ change_failure_rate is computed over MEASURED receipts only, and the count of
369
+ unmeasured ones is reported alongside it. A rate that quietly treated
370
+ unmeasurable changes as successes would be precisely the false-green this
371
+ tool exists to refuse -- so if nothing is measurable the rate is UNKNOWN, not
372
+ 0.0, no matter how good that would look.
373
+ """
374
+ total = len(records)
375
+ measured = [r for r in records if r.get("outcome") in ("SURVIVED", "REVERTED")]
376
+ reverted = [r for r in measured if r.get("outcome") == "REVERTED"]
377
+ unknown = [r for r in records if r.get("outcome") == UNKNOWN]
378
+
379
+ # Why receipts could not be measured is the most useful thing this tool
380
+ # prints when nothing is anchored. Without the distribution the report reads
381
+ # as "no data" instead of "your receipts are not recording a landed sha",
382
+ # which is an actionable defect.
383
+ reasons = {}
384
+ for r in records:
385
+ a = r.get("anchor") or {}
386
+ if a.get("state") and a["state"] != "anchored":
387
+ key = a.get("reason") or "unknown"
388
+ reasons[key] = reasons.get(key, 0) + 1
389
+
390
+ summary = {
391
+ "receipts_total": total,
392
+ "receipts_measured": len(measured),
393
+ "receipts_unknown": len(unknown),
394
+ "reverted": len(reverted),
395
+ "unanchored_reasons": reasons,
396
+ }
397
+ if measured:
398
+ summary["change_failure_rate"] = round(len(reverted) / len(measured), 4)
399
+ else:
400
+ summary["change_failure_rate"] = UNKNOWN
401
+ summary["change_failure_rate_reason"] = (
402
+ "no receipt could be measured, so a rate would be fabricated")
403
+
404
+ ratios = [r["rework"]["rework_ratio"] for r in records
405
+ if isinstance(r.get("rework"), dict)
406
+ and r["rework"].get("status") == "measured"
407
+ and isinstance(r["rework"].get("rework_ratio"), (int, float))]
408
+ if ratios:
409
+ summary["rework_ratio_avg"] = round(sum(ratios) / len(ratios), 4)
410
+ else:
411
+ summary["rework_ratio_avg"] = UNKNOWN
412
+ return summary
413
+
414
+
415
+ def render_text(records, summary, note=None):
416
+ out = []
417
+ out.append("Outcome Ledger -- what happened to the work AFTER the receipt")
418
+ out.append("")
419
+ if note:
420
+ out.append(f" {note}")
421
+ out.append("")
422
+ return "\n".join(out)
423
+
424
+ for r in records:
425
+ line = f" {r.get('run_id', '?')[:34]:36}"
426
+ oc = r.get("outcome", UNKNOWN)
427
+ line += f"{oc:10}"
428
+ rw = r.get("rework")
429
+ if isinstance(rw, dict) and rw.get("status") == "measured":
430
+ line += (f" lines {rw['lines_surviving']}/{rw['lines_introduced']} survive"
431
+ f" rework {rw['rework_ratio']}")
432
+ elif oc == UNKNOWN and r.get("reason"):
433
+ line += f" ({r['reason']})"
434
+ out.append(line)
435
+
436
+ out.append("")
437
+ out.append(f" ANCHORED {summary['receipts_measured']} of "
438
+ f"{summary['receipts_total']} receipts.")
439
+ reasons = summary.get("unanchored_reasons") or {}
440
+ if reasons:
441
+ out.append("")
442
+ out.append(" Why not anchored (a receipt must prove base..head IS the change):")
443
+ for k, n in sorted(reasons.items(), key=lambda kv: -kv[1]):
444
+ out.append(f" {k:24} {n:3} {ANCHOR_REASONS.get(k, '')}")
445
+ out.append("")
446
+ cfr = summary.get("change_failure_rate")
447
+ if cfr == UNKNOWN:
448
+ out.append(f" change-failure rate: UNKNOWN"
449
+ f" ({summary.get('change_failure_rate_reason', '')})")
450
+ else:
451
+ out.append(f" change-failure rate: {cfr}"
452
+ f" (over {summary['receipts_measured']} measured receipts)")
453
+ out.append(f" average rework ratio: {summary.get('rework_ratio_avg')}")
454
+ out.append("")
455
+ out.append(" Every number above is derivable from git. Check any of them:")
456
+ out.append(" git log --all --fixed-strings --grep='This reverts commit <head_sha>'")
457
+ out.append(" git blame --porcelain -- <path>")
458
+ return "\n".join(out)
459
+
460
+
461
+ def main(argv):
462
+ as_json = "--json" in argv
463
+ run_id = None
464
+ if "--run-id" in argv:
465
+ i = argv.index("--run-id")
466
+ if i + 1 < len(argv):
467
+ run_id = argv[i + 1]
468
+
469
+ cwd = os.environ.get("LOKI_OUTCOMES_CWD") or os.getcwd()
470
+ loki_dir = os.environ.get("LOKI_DIR") or os.path.join(cwd, ".loki")
471
+
472
+ records, note = collect(loki_dir, cwd, run_id=run_id)
473
+ summary = summarize(records)
474
+
475
+ if as_json:
476
+ print(json.dumps({
477
+ "schema_version": SCHEMA_VERSION,
478
+ "generated_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
479
+ "note": note,
480
+ "summary": summary,
481
+ "receipts": records,
482
+ "how_to_verify": [
483
+ "git log --all --fixed-strings --grep='This reverts commit <head_sha>'",
484
+ "git blame --porcelain -- <path>",
485
+ ],
486
+ }, indent=2))
487
+ else:
488
+ print(render_text(records, summary, note=note))
489
+
490
+ # Exit 3 when there is nothing to measure, mirroring `loki proof releases`:
491
+ # an empty result is a real answer, distinct from a failure.
492
+ if note:
493
+ return 3
494
+ return 0
495
+
496
+
497
+ if __name__ == "__main__":
498
+ sys.exit(main(sys.argv[1:]))