loki-mode 9.12.6 → 9.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +81 -101
- package/SKILL.md +2 -2
- package/VERSION +1 -1
- package/autonomy/intent.sh +414 -0
- package/autonomy/issue-providers.sh +24 -0
- package/autonomy/lib/agent_readiness.py +280 -0
- package/autonomy/lib/claim_grounding.py +171 -0
- package/autonomy/lib/config-map.sh +10 -6
- package/autonomy/lib/decision_record.py +198 -0
- package/autonomy/lib/failure_memory.py +199 -0
- package/autonomy/lib/gate_policy.py +166 -0
- package/autonomy/lib/outcome_ledger.py +620 -0
- package/autonomy/lib/preedit_snapshot.py +216 -0
- package/autonomy/lib/proof-generator.py +71 -4
- package/autonomy/lib/verdict.py +204 -0
- package/autonomy/loki +430 -15
- package/autonomy/notify.sh +70 -1
- package/autonomy/provider-offer.sh +25 -1
- package/autonomy/queue-consumer.sh +290 -18
- package/autonomy/run.sh +527 -12
- package/autonomy/telemetry.sh +8 -1
- package/completions/_loki +5 -0
- package/completions/loki.bash +2 -1
- package/dashboard/__init__.py +1 -1
- package/dashboard/run.py +13 -2
- package/dashboard/scim.py +221 -0
- package/dashboard/server.py +190 -1
- package/dashboard/static/index.html +248 -55
- package/docs/GATE-FAILURE-TRIAGE.md +254 -0
- package/docs/LOOP-CANDIDATE-PROPOSAL-v1.md +167 -0
- package/docs/LOOP-HARNESS-AUDIT.md +53 -0
- package/docs/QUEUE-OPERATIONS.md +107 -0
- package/docs/VERIFICATION-COST.md +273 -0
- package/docs/WANG-PRINCIPLES-PLAN.md +1 -1
- package/loki-ts/dist/loki.js +414 -416
- package/mcp/__init__.py +1 -1
- package/mcp/_sdk_loader.py +25 -0
- package/package.json +1 -1
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
|
@@ -0,0 +1,620 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Outcome Ledger: did the change turn out to be RIGHT, not merely that it happened.
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. Every competing agent reports VOLUME. Measured against their own
|
|
5
|
+
published docs: Factory AI's analytics expose files_created, files_edited,
|
|
6
|
+
lines_modified, git_commits, git_prs_created, tokens and DAU -- and no defect
|
|
7
|
+
rate, no rework rate, no revert rate, no change-failure rate. Their telemetry doc
|
|
8
|
+
states outright that correlating usage with delivery outcomes is left to the
|
|
9
|
+
customer's own stack. Devin's security page concedes the agent "can still
|
|
10
|
+
experience hallucinations, introduce bugs into code" and points you at your own
|
|
11
|
+
code review and branch protection. So an engineering leader using either can
|
|
12
|
+
prove the agent was BUSY. Neither can show it was RIGHT.
|
|
13
|
+
|
|
14
|
+
This module answers the other question, from local git history alone: after the
|
|
15
|
+
receipt was written, did the change survive?
|
|
16
|
+
|
|
17
|
+
WHAT MAKES A SIGNAL DEFENSIBLE HERE. The temptation is to count "the file was
|
|
18
|
+
touched again" as rework. That is a misleading signal -- an unrelated feature
|
|
19
|
+
landing in the same file would inflate it, and a leader acting on that number
|
|
20
|
+
would be acting on noise. So each signal below is either a deterministic fact or
|
|
21
|
+
it reports UNKNOWN:
|
|
22
|
+
|
|
23
|
+
reverted FACT. `git revert` writes "This reverts commit <sha>" into the
|
|
24
|
+
message. That is a machine-parseable link, not an inference.
|
|
25
|
+
line_survival FACT. `git blame --porcelain` attributes every surviving line to
|
|
26
|
+
the commit that introduced it. Counting lines still attributed to
|
|
27
|
+
head_sha is a measurement, not an estimate.
|
|
28
|
+
reworked MEASURED, line-scoped. Only lines this change INTRODUCED and that
|
|
29
|
+
a LATER commit replaced count as rework. A file touched elsewhere
|
|
30
|
+
does not.
|
|
31
|
+
UNKNOWN Any case we cannot compute: absent head_sha, unreachable commit
|
|
32
|
+
(never merged, shallow clone, pruned), or a file since deleted.
|
|
33
|
+
Never reported as zero, never as a pass.
|
|
34
|
+
|
|
35
|
+
The last rule is the whole point. A change-failure rate that silently scores
|
|
36
|
+
unmeasurable cases as successes is the exact false-green this product exists to
|
|
37
|
+
refuse. `loki proof` already sets the house convention -- its help says
|
|
38
|
+
"version_is_ahead reads UNKNOWN when it cannot be computed" -- and this inherits it.
|
|
39
|
+
|
|
40
|
+
READ-ONLY. Never writes to the repo under analysis. Every number it prints comes
|
|
41
|
+
from a git command the reader can rerun by hand; --json emits those commands.
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
from __future__ import annotations
|
|
45
|
+
|
|
46
|
+
import json
|
|
47
|
+
import os
|
|
48
|
+
import subprocess
|
|
49
|
+
import sys
|
|
50
|
+
from datetime import datetime, timezone
|
|
51
|
+
|
|
52
|
+
SCHEMA_VERSION = "1.0"
|
|
53
|
+
|
|
54
|
+
# A status that is not a number. Kept as a module constant so a caller can never
|
|
55
|
+
# accidentally coerce it to 0 in a tally.
|
|
56
|
+
UNKNOWN = "UNKNOWN"
|
|
57
|
+
|
|
58
|
+
# git's empty-tree object. proof-generator falls back to it for a greenfield run,
|
|
59
|
+
# so a receipt carrying it has no real baseline to diff against.
|
|
60
|
+
EMPTY_TREE_SHA = "4b825dc642cb6eb9a060e54bf8d69288fbee4904"
|
|
61
|
+
|
|
62
|
+
# THE ANCHOR GATE, and the reason this module is not a metrics generator.
|
|
63
|
+
#
|
|
64
|
+
# `facts.git.head_sha` is `git rev-parse HEAD` AT RECEIPT-GENERATION TIME. When
|
|
65
|
+
# the agent's work is still uncommitted -- the normal case -- that is the run's
|
|
66
|
+
# STARTING commit, not the commit the change became. Measured on this repo's own
|
|
67
|
+
# 9 receipts:
|
|
68
|
+
#
|
|
69
|
+
# 8 of 9 carry base_sha="" and head_sha=1385e71c. That commit touches TWO files
|
|
70
|
+
# (providers/codex.sh, tests/test-provider-degraded-mode.sh) while the receipts
|
|
71
|
+
# attest to EIGHT including autonomy/telemetry.sh. They are not the same change.
|
|
72
|
+
# The 9th carries base_sha=4b825dc6 (empty tree) and 4017 files.
|
|
73
|
+
#
|
|
74
|
+
# Following head_sha regardless would have attributed one commit's fate to eight
|
|
75
|
+
# unrelated runs and reported it as a change-failure rate. That is precisely the
|
|
76
|
+
# fabricated metric this product exists to refuse, and it would have been
|
|
77
|
+
# invisible in the output.
|
|
78
|
+
#
|
|
79
|
+
# File-overlap was tested as a fallback discriminator and REJECTED: the bad
|
|
80
|
+
# receipts overlap that commit by 2 files, so any overlap>0 rule marks all eight
|
|
81
|
+
# as anchored. Overlap is not evidence of identity.
|
|
82
|
+
#
|
|
83
|
+
# So: a receipt yields numbers only when sha algebra proves the range is the
|
|
84
|
+
# change. Everything else is UNKNOWN with a named reason.
|
|
85
|
+
ANCHOR_REASONS = {
|
|
86
|
+
"no_git": "not a git repository",
|
|
87
|
+
"head_sha_empty": "receipt records no head_sha",
|
|
88
|
+
"base_sha_empty": "run baseline was not recorded at generation",
|
|
89
|
+
"greenfield_no_baseline": "base_sha is the empty tree, so there is no baseline",
|
|
90
|
+
"change_not_committed": "base_sha == head_sha, so the work was never committed",
|
|
91
|
+
"sha_not_in_history": "a sha is not resolvable in this clone",
|
|
92
|
+
"not_reachable_from_head": "head_sha is not an ancestor of HEAD (never merged)",
|
|
93
|
+
"diff_range_mismatch": "the base..head diff does not match the receipt's file set",
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
# HISTORICAL vs LIVE: the distinction that stops an old receipt from reading as
|
|
97
|
+
# a regression.
|
|
98
|
+
#
|
|
99
|
+
# `ANCHORED 0 of 9` is alarming until you know WHY. Two very different things
|
|
100
|
+
# produce it, and a flat count of reasons cannot tell them apart:
|
|
101
|
+
#
|
|
102
|
+
# FROZEN The receipt file itself lacks the anchor. base_sha was written as
|
|
103
|
+
# "" and that JSON is on disk forever. No future fix can anchor it,
|
|
104
|
+
# and rewriting a receipt to make a metric look better is the exact
|
|
105
|
+
# dishonesty this module exists to refuse. These are HISTORY.
|
|
106
|
+
# LIVE The receipt records real shas; only the ENVIRONMENT cannot resolve
|
|
107
|
+
# them right now -- a shallow clone, an unmerged branch, or a file
|
|
108
|
+
# set that still has uncommitted edits. These can anchor later, with
|
|
109
|
+
# no change to the receipt.
|
|
110
|
+
#
|
|
111
|
+
# Measured on this repo: all 9 receipts are frozen (8 base_sha_empty, 1
|
|
112
|
+
# greenfield_no_baseline) and were generated 2026-07-27 and 2026-07-31, BEFORE
|
|
113
|
+
# proof-generator learned to read .loki/state/start-sha (commit 99ce689d,
|
|
114
|
+
# 2026-08-07). So the zero is fully explained by history.
|
|
115
|
+
# Frozen because the GENERATOR failed to record something it should have. These
|
|
116
|
+
# are the only two where recency is meaningful: before the fix they are history,
|
|
117
|
+
# after it they are a live bug in proof-generator.
|
|
118
|
+
FROZEN_GENERATOR_REASONS = frozenset({
|
|
119
|
+
"head_sha_empty",
|
|
120
|
+
"base_sha_empty",
|
|
121
|
+
})
|
|
122
|
+
|
|
123
|
+
# Frozen BY DESIGN, and correct at any date. A genuinely greenfield repo has no
|
|
124
|
+
# earlier commit to diff against, and a receipt written over uncommitted work
|
|
125
|
+
# honestly has base == head. Neither can ever anchor, and neither is a defect --
|
|
126
|
+
# so recency must NOT be applied to them. Calling a correct greenfield receipt a
|
|
127
|
+
# regression would be a false alarm on the signal added to prevent false alarms.
|
|
128
|
+
FROZEN_BY_DESIGN_REASONS = frozenset({
|
|
129
|
+
"greenfield_no_baseline",
|
|
130
|
+
"change_not_committed",
|
|
131
|
+
})
|
|
132
|
+
|
|
133
|
+
FROZEN_REASONS = FROZEN_GENERATOR_REASONS | FROZEN_BY_DESIGN_REASONS
|
|
134
|
+
|
|
135
|
+
# The two sets must stay disjoint. If a reason appeared in both, classification
|
|
136
|
+
# would depend on which branch is checked first -- and a by-design case that
|
|
137
|
+
# drifted into the generator set would start alarming as a regression the moment
|
|
138
|
+
# someone reordered the checks. Assert the invariant rather than trusting order.
|
|
139
|
+
assert not (FROZEN_GENERATOR_REASONS & FROZEN_BY_DESIGN_REASONS), \
|
|
140
|
+
"a reason cannot be both a generator failure and correct by design"
|
|
141
|
+
|
|
142
|
+
# The commit that taught proof-generator to resolve base_sha from
|
|
143
|
+
# .loki/state/start-sha when the env var is absent. A receipt generated at or
|
|
144
|
+
# after this instant should carry a real baseline, so a FROZEN reason on one is
|
|
145
|
+
# NOT history -- it is a live regression in the generator.
|
|
146
|
+
#
|
|
147
|
+
# Recency is the discriminator because it is the only one the receipts actually
|
|
148
|
+
# support: they carry generated_at (verified on all 9), and loki_version tracks
|
|
149
|
+
# releases rather than this fix. A frozen-vs-live split ALONE would file a newly
|
|
150
|
+
# broken receipt under history, which is the green-wash this guards against.
|
|
151
|
+
BASE_SHA_FIX_UTC = "2026-08-07T13:39:32Z" # 99ce689d, committed 09:39:32 -04:00
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def classify_unanchored(reason, generated_at):
|
|
155
|
+
"""Bucket an unanchored receipt: historical, regression, or live.
|
|
156
|
+
|
|
157
|
+
Returns one of:
|
|
158
|
+
"by_design" -- unanchorable and CORRECT: a greenfield run with no earlier
|
|
159
|
+
commit, or a receipt over uncommitted work. Never a defect,
|
|
160
|
+
at any date, so recency is not applied.
|
|
161
|
+
"historical" -- the generator failed to record an anchor, in a receipt
|
|
162
|
+
written BEFORE the fix. History, and never anchorable now.
|
|
163
|
+
"regression" -- the same generator failure AFTER the fix. The anchor is
|
|
164
|
+
being dropped again. This is the only bucket that alarms.
|
|
165
|
+
"live" -- the receipt is fine; the environment cannot resolve it yet.
|
|
166
|
+
"""
|
|
167
|
+
if reason in FROZEN_BY_DESIGN_REASONS:
|
|
168
|
+
return "by_design"
|
|
169
|
+
if reason not in FROZEN_GENERATOR_REASONS:
|
|
170
|
+
return "live"
|
|
171
|
+
# No timestamp means we cannot place it relative to the fix. Refuse to call
|
|
172
|
+
# it history, because that is the direction that hides a regression.
|
|
173
|
+
#
|
|
174
|
+
# ponytail: lexicographic compare, correct only for the "...Z" ISO form the
|
|
175
|
+
# generator writes (verified on all 9 receipts). An offset-form timestamp
|
|
176
|
+
# would sort below the constant and read as historical -- the hiding
|
|
177
|
+
# direction. Parse properly only if a generator ever emits offsets.
|
|
178
|
+
if not generated_at:
|
|
179
|
+
return "regression"
|
|
180
|
+
return "historical" if generated_at < BASE_SHA_FIX_UTC else "regression"
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _git(args, cwd, timeout=30):
|
|
184
|
+
"""Run a git command read-only. Returns (rc, stdout). Never raises.
|
|
185
|
+
|
|
186
|
+
Failure is data here, not an exception: an unreachable sha and a shallow
|
|
187
|
+
clone both surface as a non-zero rc, and each caller turns that into an
|
|
188
|
+
explicit UNKNOWN with a reason rather than a silent zero.
|
|
189
|
+
"""
|
|
190
|
+
try:
|
|
191
|
+
p = subprocess.run(
|
|
192
|
+
["git"] + list(args),
|
|
193
|
+
cwd=cwd, capture_output=True, text=True, timeout=timeout,
|
|
194
|
+
)
|
|
195
|
+
return p.returncode, p.stdout
|
|
196
|
+
except (subprocess.TimeoutExpired, OSError) as exc:
|
|
197
|
+
return 1, f"__error__ {type(exc).__name__}: {exc}"
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _commit_exists(sha, cwd):
|
|
201
|
+
"""True only if the object is present AND is a commit."""
|
|
202
|
+
if not sha:
|
|
203
|
+
return False
|
|
204
|
+
rc, _ = _git(["cat-file", "-e", f"{sha}^{{commit}}"], cwd)
|
|
205
|
+
return rc == 0
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def resolve_anchor(base_sha, head_sha, files, cwd):
|
|
209
|
+
"""Decide whether this receipt can be followed at all. Returns (state, reason).
|
|
210
|
+
|
|
211
|
+
Cheap sha algebra, run BEFORE any metric. Only "anchored" produces numbers;
|
|
212
|
+
every other outcome is UNKNOWN with a named reason from ANCHOR_REASONS. See
|
|
213
|
+
the ANCHOR_REASONS comment for the measured evidence that makes this gate
|
|
214
|
+
mandatory rather than defensive.
|
|
215
|
+
"""
|
|
216
|
+
rc, _ = _git(["rev-parse", "--git-dir"], cwd)
|
|
217
|
+
if rc != 0:
|
|
218
|
+
return "unanchored", "no_git"
|
|
219
|
+
if not head_sha:
|
|
220
|
+
return "unanchored", "head_sha_empty"
|
|
221
|
+
if not base_sha:
|
|
222
|
+
return "unanchored", "base_sha_empty"
|
|
223
|
+
if base_sha == EMPTY_TREE_SHA or base_sha.startswith(EMPTY_TREE_SHA[:12]):
|
|
224
|
+
return "unanchored", "greenfield_no_baseline"
|
|
225
|
+
if base_sha == head_sha:
|
|
226
|
+
return "unanchored", "change_not_committed"
|
|
227
|
+
if not _commit_exists(base_sha, cwd) or not _commit_exists(head_sha, cwd):
|
|
228
|
+
return "unanchored", "sha_not_in_history"
|
|
229
|
+
|
|
230
|
+
# Unreachable is NOT the same as reverted, and conflating them would invent
|
|
231
|
+
# failures out of unmerged branches. merge-base separates the two.
|
|
232
|
+
rc, _ = _git(["merge-base", "--is-ancestor", head_sha, "HEAD"], cwd)
|
|
233
|
+
if rc != 0:
|
|
234
|
+
return "unanchored", "not_reachable_from_head"
|
|
235
|
+
|
|
236
|
+
# Final identity check: the range must actually produce the receipt's files.
|
|
237
|
+
# This is what overlap-matching cannot do -- it demands the SET match, so a
|
|
238
|
+
# coincidental 2-file overlap can never masquerade as the same change.
|
|
239
|
+
rc, out = _git(["diff", "--name-only", f"{base_sha}..{head_sha}"], cwd)
|
|
240
|
+
if rc != 0:
|
|
241
|
+
return "unanchored", "sha_not_in_history"
|
|
242
|
+
range_files = {l.strip() for l in out.splitlines() if l.strip()}
|
|
243
|
+
receipt_files = {f.get("path") for f in files
|
|
244
|
+
if isinstance(f, dict) and f.get("path")}
|
|
245
|
+
if receipt_files and range_files != receipt_files:
|
|
246
|
+
return "unanchored", "diff_range_mismatch"
|
|
247
|
+
|
|
248
|
+
return "anchored", None
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def detect_revert(head_sha, cwd):
|
|
252
|
+
"""Was head_sha reverted? Returns (status, evidence).
|
|
253
|
+
|
|
254
|
+
FACT, not inference. `git revert` writes the canonical trailer
|
|
255
|
+
"This reverts commit <full-sha>." into the message body. We search for that
|
|
256
|
+
exact trailer, so a commit that merely mentions the sha in prose -- a
|
|
257
|
+
follow-up, a doc reference, a changelog entry -- does not count.
|
|
258
|
+
"""
|
|
259
|
+
if not _commit_exists(head_sha, cwd):
|
|
260
|
+
return UNKNOWN, "head_sha is not a reachable commit in this clone"
|
|
261
|
+
|
|
262
|
+
# --fixed-strings so a sha can never be read as a regex; --all so a revert
|
|
263
|
+
# on any branch counts, not only the current one.
|
|
264
|
+
rc, out = _git(
|
|
265
|
+
["log", "--all", "--fixed-strings",
|
|
266
|
+
f"--grep=This reverts commit {head_sha}",
|
|
267
|
+
"--format=%H %s"],
|
|
268
|
+
cwd,
|
|
269
|
+
)
|
|
270
|
+
if rc != 0:
|
|
271
|
+
return UNKNOWN, "git log failed while searching for a revert trailer"
|
|
272
|
+
hits = [l for l in out.splitlines() if l.strip()]
|
|
273
|
+
if hits:
|
|
274
|
+
return True, hits
|
|
275
|
+
return False, []
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def line_survival(head_sha, files, cwd):
|
|
279
|
+
"""How many lines introduced by head_sha still survive at HEAD.
|
|
280
|
+
|
|
281
|
+
FACT via `git blame --porcelain`: every line in the current file carries the
|
|
282
|
+
sha of the commit that last touched it. Lines still attributed to head_sha
|
|
283
|
+
are lines this change introduced that nothing has since replaced.
|
|
284
|
+
|
|
285
|
+
Returns a dict. Files that cannot be blamed (deleted, renamed away, binary)
|
|
286
|
+
are counted in `unknown_files` rather than being scored as zero survival --
|
|
287
|
+
a deleted file is not evidence that the work was wrong.
|
|
288
|
+
"""
|
|
289
|
+
if not _commit_exists(head_sha, cwd):
|
|
290
|
+
return {"status": UNKNOWN,
|
|
291
|
+
"reason": "head_sha is not a reachable commit in this clone"}
|
|
292
|
+
|
|
293
|
+
surviving = 0
|
|
294
|
+
unknown_files = []
|
|
295
|
+
checked = 0
|
|
296
|
+
|
|
297
|
+
for f in files:
|
|
298
|
+
path = f.get("path") if isinstance(f, dict) else f
|
|
299
|
+
if not path:
|
|
300
|
+
continue
|
|
301
|
+
# A file removed after the fact cannot be blamed. That is UNKNOWN, not 0.
|
|
302
|
+
if not os.path.exists(os.path.join(cwd, path)):
|
|
303
|
+
unknown_files.append({"path": path, "reason": "not present at HEAD"})
|
|
304
|
+
continue
|
|
305
|
+
rc, out = _git(["blame", "--porcelain", "--", path], cwd, timeout=60)
|
|
306
|
+
if rc != 0:
|
|
307
|
+
unknown_files.append({"path": path, "reason": "blame failed"})
|
|
308
|
+
continue
|
|
309
|
+
checked += 1
|
|
310
|
+
# In porcelain output a line beginning with a 40-hex sha starts a block;
|
|
311
|
+
# the sha is the commit that introduced that line.
|
|
312
|
+
for line in out.splitlines():
|
|
313
|
+
parts = line.split(" ", 1)
|
|
314
|
+
token = parts[0]
|
|
315
|
+
if len(token) == 40 and all(c in "0123456789abcdef" for c in token):
|
|
316
|
+
if token == head_sha:
|
|
317
|
+
surviving += 1
|
|
318
|
+
|
|
319
|
+
return {
|
|
320
|
+
"status": "measured" if checked else UNKNOWN,
|
|
321
|
+
"reason": None if checked else "no file from this change could be blamed",
|
|
322
|
+
"surviving_lines": surviving if checked else UNKNOWN,
|
|
323
|
+
"files_checked": checked,
|
|
324
|
+
"files_unknown": unknown_files,
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
def detect_rework(head_sha, files, cwd, since_days=None):
|
|
329
|
+
"""Lines this change introduced that a LATER commit replaced.
|
|
330
|
+
|
|
331
|
+
This is the signal most easily faked. "The file was touched again" is NOT
|
|
332
|
+
rework -- an unrelated feature in the same file would inflate it, and a
|
|
333
|
+
leader acting on that number would be acting on noise. So this is scoped to
|
|
334
|
+
LINES: a later commit counts only where it replaced a line that head_sha
|
|
335
|
+
introduced, which `git blame` establishes by attribution.
|
|
336
|
+
|
|
337
|
+
Derived, deliberately, from the same blame data as line_survival: lines
|
|
338
|
+
introduced minus lines still attributed. That keeps the two numbers
|
|
339
|
+
arithmetically consistent instead of two estimates that can disagree.
|
|
340
|
+
"""
|
|
341
|
+
if not _commit_exists(head_sha, cwd):
|
|
342
|
+
return {"status": UNKNOWN,
|
|
343
|
+
"reason": "head_sha is not a reachable commit in this clone"}
|
|
344
|
+
|
|
345
|
+
introduced = 0
|
|
346
|
+
for f in files:
|
|
347
|
+
if isinstance(f, dict):
|
|
348
|
+
ins = f.get("insertions")
|
|
349
|
+
if isinstance(ins, int):
|
|
350
|
+
introduced += ins
|
|
351
|
+
|
|
352
|
+
if introduced == 0:
|
|
353
|
+
return {"status": UNKNOWN,
|
|
354
|
+
"reason": "receipt records no insertion counts to compare against"}
|
|
355
|
+
|
|
356
|
+
surv = line_survival(head_sha, files, cwd)
|
|
357
|
+
if surv.get("status") != "measured":
|
|
358
|
+
return {"status": UNKNOWN, "reason": surv.get("reason") or "survival unmeasurable"}
|
|
359
|
+
|
|
360
|
+
survived = surv["surviving_lines"]
|
|
361
|
+
# Clamp: blame can attribute MORE lines than the receipt counted when a file
|
|
362
|
+
# was reformatted, so a negative would be an artifact, not a measurement.
|
|
363
|
+
replaced = max(0, introduced - survived)
|
|
364
|
+
return {
|
|
365
|
+
"status": "measured",
|
|
366
|
+
"lines_introduced": introduced,
|
|
367
|
+
"lines_surviving": survived,
|
|
368
|
+
"lines_replaced": replaced,
|
|
369
|
+
"rework_ratio": round(replaced / introduced, 4) if introduced else UNKNOWN,
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def outcome_for_receipt(proof_path, cwd):
|
|
374
|
+
"""Compute the full outcome record for one receipt."""
|
|
375
|
+
try:
|
|
376
|
+
with open(proof_path, "r", encoding="utf-8") as fh:
|
|
377
|
+
proof = json.load(fh)
|
|
378
|
+
except (OSError, ValueError) as exc:
|
|
379
|
+
return {"status": UNKNOWN, "reason": f"receipt unreadable: {exc}",
|
|
380
|
+
"proof_path": proof_path}
|
|
381
|
+
|
|
382
|
+
run_id = proof.get("run_id") or os.path.basename(os.path.dirname(proof_path))
|
|
383
|
+
git_facts = (proof.get("facts") or {}).get("git") or {}
|
|
384
|
+
head_sha = git_facts.get("head_sha")
|
|
385
|
+
base_sha = git_facts.get("base_sha")
|
|
386
|
+
files = ((git_facts.get("diff") or {}).get("files")) or []
|
|
387
|
+
|
|
388
|
+
rec = {
|
|
389
|
+
"run_id": run_id,
|
|
390
|
+
"generated_at": proof.get("generated_at"),
|
|
391
|
+
"headline": (proof.get("verification") or {}).get("headline")
|
|
392
|
+
or proof.get("headline"),
|
|
393
|
+
"base_sha": base_sha,
|
|
394
|
+
"head_sha": head_sha,
|
|
395
|
+
"files_changed": len(files),
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
# THE GATE. No metric runs unless sha algebra proves base..head IS this
|
|
399
|
+
# change. Without it, 8 of this repo's 9 receipts would have been scored
|
|
400
|
+
# against an unrelated 2-file commit and reported as a change-failure rate.
|
|
401
|
+
state, reason = resolve_anchor(base_sha, head_sha, files, cwd)
|
|
402
|
+
rec["anchor"] = {"state": state, "reason": reason}
|
|
403
|
+
if state != "anchored":
|
|
404
|
+
# An old receipt is not a regression. Classify so a reader can tell a
|
|
405
|
+
# frozen pre-fix receipt from a generator that started dropping anchors.
|
|
406
|
+
rec["anchor"]["klass"] = classify_unanchored(reason, rec["generated_at"])
|
|
407
|
+
rec["outcome"] = UNKNOWN
|
|
408
|
+
rec["reason"] = ANCHOR_REASONS.get(reason, reason or "not anchored")
|
|
409
|
+
rec["commands"] = [
|
|
410
|
+
f"git merge-base --is-ancestor {head_sha or '<head_sha>'} HEAD",
|
|
411
|
+
f"git diff --name-only {base_sha or '<base_sha>'}..{head_sha or '<head_sha>'}",
|
|
412
|
+
]
|
|
413
|
+
return rec
|
|
414
|
+
|
|
415
|
+
reverted, revert_evidence = detect_revert(head_sha, cwd)
|
|
416
|
+
rec["reverted"] = reverted
|
|
417
|
+
if revert_evidence:
|
|
418
|
+
rec["revert_evidence"] = revert_evidence
|
|
419
|
+
rec["survival"] = line_survival(head_sha, files, cwd)
|
|
420
|
+
rec["rework"] = detect_rework(head_sha, files, cwd)
|
|
421
|
+
|
|
422
|
+
# The verdict is deliberately coarse and refuses to guess.
|
|
423
|
+
if reverted is True:
|
|
424
|
+
rec["outcome"] = "REVERTED"
|
|
425
|
+
elif reverted is UNKNOWN:
|
|
426
|
+
rec["outcome"] = UNKNOWN
|
|
427
|
+
rec["reason"] = "could not determine whether the change was reverted"
|
|
428
|
+
elif rec["rework"].get("status") == "measured":
|
|
429
|
+
rec["outcome"] = "SURVIVED"
|
|
430
|
+
else:
|
|
431
|
+
rec["outcome"] = UNKNOWN
|
|
432
|
+
rec["reason"] = rec["rework"].get("reason") or "outcome unmeasurable"
|
|
433
|
+
return rec
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
def collect(loki_dir, cwd, run_id=None):
|
|
437
|
+
"""Compute outcomes for every receipt under <loki_dir>/proofs."""
|
|
438
|
+
proofs_dir = os.path.join(loki_dir, "proofs")
|
|
439
|
+
records = []
|
|
440
|
+
if not os.path.isdir(proofs_dir):
|
|
441
|
+
return records, "no .loki/proofs directory in this project"
|
|
442
|
+
|
|
443
|
+
for entry in sorted(os.listdir(proofs_dir)):
|
|
444
|
+
if run_id and entry != run_id:
|
|
445
|
+
continue
|
|
446
|
+
p = os.path.join(proofs_dir, entry, "proof.json")
|
|
447
|
+
if os.path.isfile(p):
|
|
448
|
+
records.append(outcome_for_receipt(p, cwd))
|
|
449
|
+
if not records:
|
|
450
|
+
return records, "no receipts found (run a build first, then re-run this)"
|
|
451
|
+
return records, None
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def summarize(records):
|
|
455
|
+
"""Aggregate. UNKNOWN is carried, never folded into a pass.
|
|
456
|
+
|
|
457
|
+
change_failure_rate is computed over MEASURED receipts only, and the count of
|
|
458
|
+
unmeasured ones is reported alongside it. A rate that quietly treated
|
|
459
|
+
unmeasurable changes as successes would be precisely the false-green this
|
|
460
|
+
tool exists to refuse -- so if nothing is measurable the rate is UNKNOWN, not
|
|
461
|
+
0.0, no matter how good that would look.
|
|
462
|
+
"""
|
|
463
|
+
total = len(records)
|
|
464
|
+
measured = [r for r in records if r.get("outcome") in ("SURVIVED", "REVERTED")]
|
|
465
|
+
reverted = [r for r in measured if r.get("outcome") == "REVERTED"]
|
|
466
|
+
unknown = [r for r in records if r.get("outcome") == UNKNOWN]
|
|
467
|
+
|
|
468
|
+
# Why receipts could not be measured is the most useful thing this tool
|
|
469
|
+
# prints when nothing is anchored. Without the distribution the report reads
|
|
470
|
+
# as "no data" instead of "your receipts are not recording a landed sha",
|
|
471
|
+
# which is an actionable defect.
|
|
472
|
+
reasons = {}
|
|
473
|
+
klasses = {"by_design": 0, "historical": 0, "regression": 0, "live": 0}
|
|
474
|
+
for r in records:
|
|
475
|
+
a = r.get("anchor") or {}
|
|
476
|
+
if a.get("state") and a["state"] != "anchored":
|
|
477
|
+
key = a.get("reason") or "unknown"
|
|
478
|
+
reasons[key] = reasons.get(key, 0) + 1
|
|
479
|
+
k = a.get("klass")
|
|
480
|
+
if k in klasses:
|
|
481
|
+
klasses[k] += 1
|
|
482
|
+
|
|
483
|
+
summary = {
|
|
484
|
+
"receipts_total": total,
|
|
485
|
+
"receipts_measured": len(measured),
|
|
486
|
+
"receipts_unknown": len(unknown),
|
|
487
|
+
"reverted": len(reverted),
|
|
488
|
+
"unanchored_reasons": reasons,
|
|
489
|
+
# The count that turns an alarming zero into an explained one. A
|
|
490
|
+
# regression here is the only bucket that warrants action.
|
|
491
|
+
"unanchored_by_design": klasses["by_design"],
|
|
492
|
+
"unanchored_historical": klasses["historical"],
|
|
493
|
+
"unanchored_regression": klasses["regression"],
|
|
494
|
+
"unanchored_live": klasses["live"],
|
|
495
|
+
}
|
|
496
|
+
if measured:
|
|
497
|
+
summary["change_failure_rate"] = round(len(reverted) / len(measured), 4)
|
|
498
|
+
else:
|
|
499
|
+
summary["change_failure_rate"] = UNKNOWN
|
|
500
|
+
summary["change_failure_rate_reason"] = (
|
|
501
|
+
"no receipt could be measured, so a rate would be fabricated")
|
|
502
|
+
|
|
503
|
+
ratios = [r["rework"]["rework_ratio"] for r in records
|
|
504
|
+
if isinstance(r.get("rework"), dict)
|
|
505
|
+
and r["rework"].get("status") == "measured"
|
|
506
|
+
and isinstance(r["rework"].get("rework_ratio"), (int, float))]
|
|
507
|
+
if ratios:
|
|
508
|
+
summary["rework_ratio_avg"] = round(sum(ratios) / len(ratios), 4)
|
|
509
|
+
else:
|
|
510
|
+
summary["rework_ratio_avg"] = UNKNOWN
|
|
511
|
+
return summary
|
|
512
|
+
|
|
513
|
+
|
|
514
|
+
def render_text(records, summary, note=None):
|
|
515
|
+
out = []
|
|
516
|
+
out.append("Outcome Ledger -- what happened to the work AFTER the receipt")
|
|
517
|
+
out.append("")
|
|
518
|
+
if note:
|
|
519
|
+
out.append(f" {note}")
|
|
520
|
+
out.append("")
|
|
521
|
+
return "\n".join(out)
|
|
522
|
+
|
|
523
|
+
for r in records:
|
|
524
|
+
line = f" {r.get('run_id', '?')[:34]:36}"
|
|
525
|
+
oc = r.get("outcome", UNKNOWN)
|
|
526
|
+
line += f"{oc:10}"
|
|
527
|
+
rw = r.get("rework")
|
|
528
|
+
if isinstance(rw, dict) and rw.get("status") == "measured":
|
|
529
|
+
line += (f" lines {rw['lines_surviving']}/{rw['lines_introduced']} survive"
|
|
530
|
+
f" rework {rw['rework_ratio']}")
|
|
531
|
+
elif oc == UNKNOWN and r.get("reason"):
|
|
532
|
+
line += f" ({r['reason']})"
|
|
533
|
+
out.append(line)
|
|
534
|
+
|
|
535
|
+
out.append("")
|
|
536
|
+
out.append(f" ANCHORED {summary['receipts_measured']} of "
|
|
537
|
+
f"{summary['receipts_total']} receipts.")
|
|
538
|
+
reasons = summary.get("unanchored_reasons") or {}
|
|
539
|
+
if reasons:
|
|
540
|
+
out.append("")
|
|
541
|
+
out.append(" Why not anchored (a receipt must prove base..head IS the change):")
|
|
542
|
+
for k, n in sorted(reasons.items(), key=lambda kv: -kv[1]):
|
|
543
|
+
out.append(f" {k:24} {n:3} {ANCHOR_REASONS.get(k, '')}")
|
|
544
|
+
|
|
545
|
+
# Without this, ANCHORED 0 of N reads as a regression when it is history.
|
|
546
|
+
hist = summary.get("unanchored_historical", 0)
|
|
547
|
+
regr = summary.get("unanchored_regression", 0)
|
|
548
|
+
live = summary.get("unanchored_live", 0)
|
|
549
|
+
design = summary.get("unanchored_by_design", 0)
|
|
550
|
+
if hist or regr or live or design:
|
|
551
|
+
out.append("")
|
|
552
|
+
if design:
|
|
553
|
+
out.append(f" {design} by design: a greenfield run or uncommitted"
|
|
554
|
+
f" work has no baseline to diff against. Correct, not a"
|
|
555
|
+
f" defect, and never anchorable.")
|
|
556
|
+
if hist:
|
|
557
|
+
out.append(f" {hist} historical: written before the base_sha fix"
|
|
558
|
+
f" ({BASE_SHA_FIX_UTC[:10]}); the receipt itself has no"
|
|
559
|
+
f" baseline, so these can never anchor. Not a defect.")
|
|
560
|
+
if live:
|
|
561
|
+
out.append(f" {live} live: the receipt records real shas; this"
|
|
562
|
+
f" clone cannot resolve them yet. May anchor later.")
|
|
563
|
+
if regr:
|
|
564
|
+
out.append(f" {regr} REGRESSION: written AFTER the fix and still"
|
|
565
|
+
f" missing a baseline. The generator is dropping the"
|
|
566
|
+
f" anchor -- this one is a defect.")
|
|
567
|
+
out.append("")
|
|
568
|
+
cfr = summary.get("change_failure_rate")
|
|
569
|
+
if cfr == UNKNOWN:
|
|
570
|
+
out.append(f" change-failure rate: UNKNOWN"
|
|
571
|
+
f" ({summary.get('change_failure_rate_reason', '')})")
|
|
572
|
+
else:
|
|
573
|
+
out.append(f" change-failure rate: {cfr}"
|
|
574
|
+
f" (over {summary['receipts_measured']} measured receipts)")
|
|
575
|
+
out.append(f" average rework ratio: {summary.get('rework_ratio_avg')}")
|
|
576
|
+
out.append("")
|
|
577
|
+
out.append(" Every number above is derivable from git. Check any of them:")
|
|
578
|
+
out.append(" git log --all --fixed-strings --grep='This reverts commit <head_sha>'")
|
|
579
|
+
out.append(" git blame --porcelain -- <path>")
|
|
580
|
+
return "\n".join(out)
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
def main(argv):
|
|
584
|
+
as_json = "--json" in argv
|
|
585
|
+
run_id = None
|
|
586
|
+
if "--run-id" in argv:
|
|
587
|
+
i = argv.index("--run-id")
|
|
588
|
+
if i + 1 < len(argv):
|
|
589
|
+
run_id = argv[i + 1]
|
|
590
|
+
|
|
591
|
+
cwd = os.environ.get("LOKI_OUTCOMES_CWD") or os.getcwd()
|
|
592
|
+
loki_dir = os.environ.get("LOKI_DIR") or os.path.join(cwd, ".loki")
|
|
593
|
+
|
|
594
|
+
records, note = collect(loki_dir, cwd, run_id=run_id)
|
|
595
|
+
summary = summarize(records)
|
|
596
|
+
|
|
597
|
+
if as_json:
|
|
598
|
+
print(json.dumps({
|
|
599
|
+
"schema_version": SCHEMA_VERSION,
|
|
600
|
+
"generated_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
|
|
601
|
+
"note": note,
|
|
602
|
+
"summary": summary,
|
|
603
|
+
"receipts": records,
|
|
604
|
+
"how_to_verify": [
|
|
605
|
+
"git log --all --fixed-strings --grep='This reverts commit <head_sha>'",
|
|
606
|
+
"git blame --porcelain -- <path>",
|
|
607
|
+
],
|
|
608
|
+
}, indent=2))
|
|
609
|
+
else:
|
|
610
|
+
print(render_text(records, summary, note=note))
|
|
611
|
+
|
|
612
|
+
# Exit 3 when there is nothing to measure, mirroring `loki proof releases`:
|
|
613
|
+
# an empty result is a real answer, distinct from a failure.
|
|
614
|
+
if note:
|
|
615
|
+
return 3
|
|
616
|
+
return 0
|
|
617
|
+
|
|
618
|
+
|
|
619
|
+
if __name__ == "__main__":
|
|
620
|
+
sys.exit(main(sys.argv[1:]))
|