loki-mode 9.12.6 → 9.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/README.md +81 -101
  2. package/SKILL.md +2 -2
  3. package/VERSION +1 -1
  4. package/autonomy/intent.sh +414 -0
  5. package/autonomy/issue-providers.sh +24 -0
  6. package/autonomy/lib/agent_readiness.py +280 -0
  7. package/autonomy/lib/claim_grounding.py +171 -0
  8. package/autonomy/lib/config-map.sh +10 -6
  9. package/autonomy/lib/decision_record.py +198 -0
  10. package/autonomy/lib/failure_memory.py +199 -0
  11. package/autonomy/lib/gate_policy.py +166 -0
  12. package/autonomy/lib/outcome_ledger.py +620 -0
  13. package/autonomy/lib/preedit_snapshot.py +216 -0
  14. package/autonomy/lib/proof-generator.py +71 -4
  15. package/autonomy/lib/verdict.py +204 -0
  16. package/autonomy/loki +430 -15
  17. package/autonomy/notify.sh +70 -1
  18. package/autonomy/provider-offer.sh +25 -1
  19. package/autonomy/queue-consumer.sh +290 -18
  20. package/autonomy/run.sh +527 -12
  21. package/autonomy/telemetry.sh +8 -1
  22. package/completions/_loki +5 -0
  23. package/completions/loki.bash +2 -1
  24. package/dashboard/__init__.py +1 -1
  25. package/dashboard/run.py +13 -2
  26. package/dashboard/scim.py +221 -0
  27. package/dashboard/server.py +190 -1
  28. package/dashboard/static/index.html +248 -55
  29. package/docs/GATE-FAILURE-TRIAGE.md +254 -0
  30. package/docs/LOOP-CANDIDATE-PROPOSAL-v1.md +167 -0
  31. package/docs/LOOP-HARNESS-AUDIT.md +53 -0
  32. package/docs/QUEUE-OPERATIONS.md +107 -0
  33. package/docs/VERIFICATION-COST.md +273 -0
  34. package/docs/WANG-PRINCIPLES-PLAN.md +1 -1
  35. package/loki-ts/dist/loki.js +414 -416
  36. package/mcp/__init__.py +1 -1
  37. package/mcp/_sdk_loader.py +25 -0
  38. package/package.json +1 -1
  39. package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
@@ -0,0 +1,216 @@
1
+ #!/usr/bin/env python3
2
+ """Pre-edit snapshot: score the AGENT, not the agent-plus-whoever-fixed-it.
3
+
4
+ WHY THIS EXISTS. 8090 AI's medical-document evaluation framework -- the single
5
+ most rigorous engineering artifact in any competitor corpus -- makes an argument
6
+ we could not answer:
7
+
8
+ "A skilled writer rescues an unusable draft, and the post-edit score looks
9
+ acceptable; a junior writer leaves the model's failures visible, and the
10
+ same model appears to perform worse. In either case, the metric describes
11
+ the writer-plus-AI workflow, not the AI alone."
12
+
13
+ Every quality number we publish today is measured AFTER a human may have touched
14
+ the diff. So a strong reviewer flatters the agent and a weak one maligns it, and
15
+ the number moves for reasons that have nothing to do with the agent. That makes
16
+ our own headline partly a measurement of our users.
17
+
18
+ The fix is not analysis, it is ORDER OF OPERATIONS. 8090's own words: the two
19
+ scores must be "separated by the workflow itself, not reconstructed afterward
20
+ from logs". So this captures an immutable snapshot of the agent's output at the
21
+ moment it stops and before any human edit. Anything reconstructed later is a
22
+ guess about which lines a human wrote, and a guess is exactly what this replaces.
23
+
24
+ WHAT IT DELIBERATELY IS NOT. It does not score anything. It records WHAT WAS
25
+ THERE, with a hash, so a later comparison is a fact rather than an inference. A
26
+ low pre-edit score is a DIAGNOSTIC, not a failure -- 8090 again: "A preedit
27
+ composite of twenty percent is not a failure of the eval. It is a diagnostic,
28
+ document-by-document, of where the model is weak." Nothing here feeds a gate,
29
+ because a gate that punished the agent for needing edits would train the agent to
30
+ produce diffs nobody edits, which is not the same as good diffs.
31
+
32
+ WHAT ALREADY EXISTS AND IS NOT REBUILT. `_compute_headline` in
33
+ proof-generator.py already implements the asymmetric-risk half of this idea, in
34
+ its own words: "A failed check is a stronger negative signal than a not-run one:
35
+ amber means we did not check everything, red means something we checked did not
36
+ pass." That is F2-style weighting -- treating a false negative as worse than a
37
+ false positive -- already shipped. This module adds only the missing half.
38
+ """
39
+
40
+ from __future__ import annotations
41
+
42
+ import hashlib
43
+ import json
44
+ import os
45
+ import subprocess
46
+ import sys
47
+ from datetime import datetime, timezone
48
+
49
+ SCHEMA_VERSION = "1.0"
50
+
51
+ UNKNOWN = "UNKNOWN"
52
+
53
+ # Named refusals, mirroring the outcome ledger's ANCHOR_REASONS. A comparison we
54
+ # cannot compute is reported BY NAME, never as zero and never as a pass.
55
+ COMPARE_REASONS = {
56
+ "no_snapshot": "no pre-edit snapshot was captured for this run",
57
+ "snapshot_unreadable": "the snapshot exists but could not be parsed",
58
+ "not_a_git_repo": "not a git repository, so the current diff cannot be read",
59
+ "no_current_diff": "there is no current diff to compare the snapshot against",
60
+ }
61
+
62
+
63
+ def _git(args, cwd, timeout=60):
64
+ try:
65
+ p = subprocess.run(["git"] + list(args), cwd=cwd,
66
+ capture_output=True, text=True, timeout=timeout)
67
+ return p.returncode, p.stdout
68
+ except (subprocess.TimeoutExpired, OSError):
69
+ return 1, ""
70
+
71
+
72
+ def _sha256(text):
73
+ return hashlib.sha256(text.encode("utf-8", "replace")).hexdigest()
74
+
75
+
76
+ def capture(loki_dir, run_id, cwd, base_sha=None):
77
+ """Freeze the agent's output before any human edit.
78
+
79
+ Stores the full unified diff plus its sha256. The hash is what makes a later
80
+ comparison evidence rather than an assertion: anyone can recompute it.
81
+
82
+ Called by the workflow at the moment the agent stops. Capturing it later --
83
+ after a review, from a log -- would record the human's work as the agent's,
84
+ which is the exact contamination this exists to prevent.
85
+ """
86
+ rc, _ = _git(["rev-parse", "--git-dir"], cwd)
87
+ if rc != 0:
88
+ return {"status": UNKNOWN, "reason": "not_a_git_repo"}
89
+
90
+ if base_sha:
91
+ rc, diff = _git(["diff", base_sha, "--"], cwd)
92
+ else:
93
+ # Uncommitted agent work is the normal case at stop time.
94
+ rc, diff = _git(["diff", "HEAD", "--"], cwd)
95
+ if rc != 0:
96
+ return {"status": UNKNOWN, "reason": "no_current_diff"}
97
+
98
+ rc2, names = _git(["diff", "--name-only"] + ([base_sha] if base_sha else ["HEAD"]), cwd)
99
+ files = [l.strip() for l in names.splitlines() if l.strip()] if rc2 == 0 else []
100
+
101
+ snap = {
102
+ "schema_version": SCHEMA_VERSION,
103
+ "run_id": run_id,
104
+ "captured_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
105
+ "base_sha": base_sha or "",
106
+ # The point of the whole file: this is the agent's output, uncontaminated.
107
+ "author": "agent",
108
+ "diff_sha256": _sha256(diff),
109
+ "diff_bytes": len(diff.encode("utf-8", "replace")),
110
+ "files": files,
111
+ "lines_added": sum(1 for l in diff.splitlines()
112
+ if l.startswith("+") and not l.startswith("+++")),
113
+ "lines_removed": sum(1 for l in diff.splitlines()
114
+ if l.startswith("-") and not l.startswith("---")),
115
+ }
116
+
117
+ out_dir = os.path.join(loki_dir, "preedit")
118
+ try:
119
+ os.makedirs(out_dir, exist_ok=True)
120
+ path = os.path.join(out_dir, f"{run_id}.json")
121
+ # Write-once. A snapshot that a later run can overwrite is not a
122
+ # snapshot -- re-capturing after a human edit would silently record the
123
+ # human's work as the agent's, which is the contamination this prevents.
124
+ if os.path.exists(path):
125
+ return {"status": "exists", "path": path}
126
+ with open(path, "w", encoding="utf-8") as fh:
127
+ json.dump(snap, fh, indent=2)
128
+ fh.write("\n")
129
+ # The raw diff is kept beside the metadata so the hash is checkable by
130
+ # hand. Evidence nobody can recompute is a claim, not evidence.
131
+ with open(os.path.join(out_dir, f"{run_id}.diff"), "w", encoding="utf-8") as fh:
132
+ fh.write(diff)
133
+ except OSError as exc:
134
+ return {"status": UNKNOWN, "reason": "snapshot_unreadable", "detail": str(exc)}
135
+
136
+ return {"status": "captured", "path": path, "diff_sha256": snap["diff_sha256"]}
137
+
138
+
139
+ def compare(loki_dir, run_id, cwd):
140
+ """How much of what shipped was the agent's, and how much was human rescue.
141
+
142
+ Reports lines, never a score. The split is the useful fact; a composite
143
+ number invites exactly the ranking behaviour that made post-edit scores
144
+ misleading in the first place.
145
+ """
146
+ path = os.path.join(loki_dir, "preedit", f"{run_id}.json")
147
+ if not os.path.isfile(path):
148
+ return {"status": UNKNOWN, "reason": "no_snapshot",
149
+ "detail": COMPARE_REASONS["no_snapshot"]}
150
+ try:
151
+ with open(path, "r", encoding="utf-8") as fh:
152
+ snap = json.load(fh)
153
+ except (OSError, ValueError):
154
+ return {"status": UNKNOWN, "reason": "snapshot_unreadable",
155
+ "detail": COMPARE_REASONS["snapshot_unreadable"]}
156
+
157
+ base = snap.get("base_sha") or "HEAD"
158
+ rc, diff_now = _git(["diff", base, "--"], cwd)
159
+ if rc != 0:
160
+ return {"status": UNKNOWN, "reason": "no_current_diff",
161
+ "detail": COMPARE_REASONS["no_current_diff"]}
162
+
163
+ now_hash = _sha256(diff_now)
164
+ added_now = sum(1 for l in diff_now.splitlines()
165
+ if l.startswith("+") and not l.startswith("+++"))
166
+
167
+ untouched = now_hash == snap.get("diff_sha256")
168
+ return {
169
+ "status": "measured",
170
+ "run_id": run_id,
171
+ "agent_lines_added": snap.get("lines_added"),
172
+ "current_lines_added": added_now,
173
+ # A hash match is proof nobody edited it; a mismatch proves somebody did,
174
+ # but NOT how much -- attributing individual lines would be the guess
175
+ # this module exists to avoid.
176
+ "human_edited": (not untouched),
177
+ "preedit_diff_sha256": snap.get("diff_sha256"),
178
+ "current_diff_sha256": now_hash,
179
+ "captured_at": snap.get("captured_at"),
180
+ "note": ("the pre-edit diff is byte-identical to what shipped, so this "
181
+ "receipt measures the agent alone"
182
+ if untouched else
183
+ "what shipped differs from the agent's output, so any score "
184
+ "taken now measures the agent plus the human who edited it"),
185
+ }
186
+
187
+
188
+ def main(argv):
189
+ if not argv:
190
+ print("usage: preedit_snapshot.py capture|compare <run_id> [--json]",
191
+ file=sys.stderr)
192
+ return 2
193
+ action = argv[0]
194
+ run_id = argv[1] if len(argv) > 1 else ""
195
+ if not run_id:
196
+ print("a run_id is required", file=sys.stderr)
197
+ return 2
198
+
199
+ cwd = os.environ.get("LOKI_PREEDIT_CWD") or os.getcwd()
200
+ loki_dir = os.environ.get("LOKI_DIR") or os.path.join(cwd, ".loki")
201
+
202
+ if action == "capture":
203
+ res = capture(loki_dir, run_id, cwd,
204
+ base_sha=os.environ.get("LOKI_RUN_START_SHA") or None)
205
+ elif action == "compare":
206
+ res = compare(loki_dir, run_id, cwd)
207
+ else:
208
+ print(f"unknown action: {action}", file=sys.stderr)
209
+ return 2
210
+
211
+ print(json.dumps(res, indent=2))
212
+ return 0 if res.get("status") in ("captured", "exists", "measured") else 3
213
+
214
+
215
+ if __name__ == "__main__":
216
+ sys.exit(main(sys.argv[1:]))
@@ -943,12 +943,46 @@ def _git_diffstat(target_dir, include_diffs):
943
943
 
944
944
  Order of preference:
945
945
  1. _LOKI_RUN_START_SHA -- the run's own baseline (run.sh exports it).
946
- 2. The empty tree -- correct for a GREENFIELD run (a repo with no commits
946
+ 2. .loki/state/start-sha -- the SAME baseline, persisted to disk. The env
947
+ var is only exported inside run_autonomous (run.sh:21690), so a receipt
948
+ generated outside that scope -- `loki proof` run by hand, a resumed
949
+ run, a receipt written after the runner exited -- saw an empty env var
950
+ and fell straight through to the empty tree.
951
+ 3. The empty tree -- correct for a GREENFIELD run (a repo with no commits
947
952
  at start), where "everything that now exists" IS the run's output and
948
953
  there is no earlier commit to diff against.
949
- 3. Empty string -- let workspace_diff apply its own fallbacks.
954
+ 4. Empty string -- let workspace_diff apply its own fallbacks.
955
+
956
+ WHY STEP 2 EXISTS. Without it every receipt on this repo recorded
957
+ base_sha="" -- measured: 9 of 9 receipts in .loki/proofs/, and the dashboard
958
+ correctly reported "9 receipts, 0 verified". An empty base is unanchorable,
959
+ so outcome_ledger.resolve_anchor() returns unanchored/base_sha_empty and NO
960
+ receipt can be verified. The persisted file already existed and both other
961
+ consumers already read it (run.sh:7779, completion-council.sh:1834); this
962
+ reader was the only one that did not, so it silently lost the anchor.
963
+
964
+ The greenfield fallback is deliberately kept BELOW the file: falling back to
965
+ the empty tree when a real baseline exists would attest to "everything in
966
+ the repo" as this run's output, which is a much worse lie than an empty
967
+ base_sha. Order matters more than the addition.
950
968
  """
951
969
  base = os.environ.get("_LOKI_RUN_START_SHA", "").strip()
970
+ if not base:
971
+ # Same baseline, read from disk. Mirrors run.sh:7779 and
972
+ # completion-council.sh:1834 rather than inventing a third convention.
973
+ try:
974
+ with open(os.path.join(target_dir, ".loki", "state", "start-sha")) as fh:
975
+ _persisted = fh.read().strip()
976
+ # Only trust it if it names a commit that actually exists HERE. A
977
+ # stale or foreign SHA would produce a diff against nothing and an
978
+ # integrity hash nobody can recompute.
979
+ if _persisted and subprocess.run(
980
+ ["git", "cat-file", "-e", _persisted + "^{commit}"],
981
+ cwd=target_dir, capture_output=True, timeout=10,
982
+ ).returncode == 0:
983
+ base = _persisted
984
+ except (OSError, subprocess.SubprocessError, ValueError):
985
+ pass
952
986
  if not base:
953
987
  # Greenfield: no baseline commit existed when the run started.
954
988
  base = _empty_tree_sha(target_dir)
@@ -990,8 +1024,33 @@ def _collect_iterations(loki_dir):
990
1024
  return {"count": count, "succeeded": n_completed, "failed": n_failed}
991
1025
 
992
1026
 
1027
+ # Acceptance-criterion ids as minted at intake ("- AC-<AXIS>-NNN: <text>", see
1028
+ # _brief_acceptance_criteria in autonomy/loki). Anchored and strict on purpose:
1029
+ # a loose pattern would count prose that merely mentions an id, and the whole
1030
+ # value of the id is that a citation points at exactly one criterion.
1031
+ _AC_ID_RE = re.compile(r"^- (AC-[A-Z]+-[0-9]{3}): ", re.MULTILINE)
1032
+
1033
+
1034
+ def _spec_criteria_declared(text):
1035
+ """Return the acceptance-criterion ids the spec DECLARES, in spec order.
1036
+
1037
+ DECLARED, NOT SATISFIED. This records which criteria exist in the spec and
1038
+ nothing more -- no check runs here, and no field in the receipt asserts that
1039
+ any of these was met. Naming it criteria_met would be a lie we cannot back.
1040
+
1041
+ Empty list when the spec declares none (an older PRD, a hand-written spec,
1042
+ or a run with no spec file at all). Never invented, never a placeholder.
1043
+ """
1044
+ if not text:
1045
+ return []
1046
+ # Deduped: an id is a citation target, so each must resolve to one criterion.
1047
+ # A repeated id is a spec bug; counting it twice would not make it citable.
1048
+ return list(dict.fromkeys(_AC_ID_RE.findall(text)))
1049
+
1050
+
993
1051
  def _collect_spec(loki_dir, target_dir):
994
- """Return spec dict {source, brief}. brief truncated to 600 chars."""
1052
+ """Return spec dict {source, brief, criteria_declared}. brief truncated to
1053
+ 600 chars."""
995
1054
  prd_path = os.environ.get("PRD_PATH", "").strip()
996
1055
  source = ""
997
1056
  brief = ""
@@ -1018,7 +1077,15 @@ def _collect_spec(loki_dir, target_dir):
1018
1077
  # Full brief here; the <=600 cap is applied AFTER redaction in generate()
1019
1078
  # so a secret straddling the cap cannot be sliced into an under-length
1020
1079
  # fragment that bypasses the redactor.
1021
- return {"source": source, "brief": brief}
1080
+ #
1081
+ # Criteria are parsed from the FULL spec text, not from the 600-char display
1082
+ # cap: a PRD's acceptance-criteria block sits well past char 600, so parsing
1083
+ # the capped brief would silently drop most of them.
1084
+ return {
1085
+ "source": source,
1086
+ "brief": brief,
1087
+ "criteria_declared": _spec_criteria_declared(brief),
1088
+ }
1022
1089
 
1023
1090
 
1024
1091
  def _self_version():
@@ -0,0 +1,204 @@
1
+ #!/usr/bin/env python3
2
+ """One verdict a reviewer can read in ten seconds, assembled from measured parts.
3
+
4
+ WHY THIS EXISTS. We now measure five things nobody else does -- did the work
5
+ survive (outcome ledger), does the spec still match the intent (intent ledger),
6
+ was this the agent's own output or a human rescue (pre-edit snapshot), does the
7
+ completion claim name real work (claim grounding), and which model decided
8
+ (decision record). Each is correct and each lives in its own file. A reviewer
9
+ looking at a pull request reads none of them.
10
+
11
+ A moat nobody sees is not a moat. The Evidence Receipt is already attached to
12
+ PRs by default (LOKI_PROVEN_PR, autonomy/run.sh:3849), so the surface exists and
13
+ this fills it: one block, five lines, each of which is either a measured fact or
14
+ an explicit UNKNOWN.
15
+
16
+ WHAT IT REFUSES, AND WHY THAT MATTERS MORE HERE THAN ANYWHERE. This is the only
17
+ place where all five signals meet, which makes it the one place a composite score
18
+ would be tempting: a single "trust: 87" would fit a PR comment beautifully. It is
19
+ not offered. Averaging a revert count, a hash comparison, a diff hash, a path
20
+ match and a model id produces a number whose movement nobody can explain, and a
21
+ number nobody can explain is the thing our competitors already ship. Five honest
22
+ lines beat one confident one.
23
+
24
+ UNKNOWN IS PRINTED, NOT HIDDEN. A reviewer must be able to tell "we checked and
25
+ it is fine" from "we could not check". Suppressing the unmeasured lines would
26
+ make a receipt with one real signal look identical to one with five, which is
27
+ exactly the false confidence the whole product exists to refuse.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ import json
33
+ import os
34
+ import sys
35
+
36
+ SCHEMA_VERSION = "1.0"
37
+
38
+ UNKNOWN = "UNKNOWN"
39
+
40
+ # Each signal: where it lives, and the one-line question it answers for a human
41
+ # scanning a PR. The question text is load-bearing -- a reviewer who cannot tell
42
+ # what a line MEANS will skip the block, and a skipped block is worth nothing.
43
+ SIGNALS = (
44
+ ("outcome", "did previous work survive, or get reverted"),
45
+ ("intent", "does the spec still say what was actually wanted"),
46
+ ("authorship", "is this the agent's output, or a human rescue"),
47
+ ("grounding", "does the completion claim name work that exists in the diff"),
48
+ ("model", "which model decided, at what temperature"),
49
+ )
50
+
51
+
52
+ def _read_json(path):
53
+ try:
54
+ with open(path, "r", encoding="utf-8") as fh:
55
+ return json.load(fh)
56
+ except (OSError, ValueError):
57
+ return None
58
+
59
+
60
+ def _outcome_line(loki_dir):
61
+ p = os.path.join(loki_dir, "proofs")
62
+ if not os.path.isdir(p):
63
+ return UNKNOWN, "no receipts yet, so nothing has an outcome to follow"
64
+ # Deliberately does not recompute: the outcome ledger owns that logic and two
65
+ # implementations of one number eventually disagree.
66
+ return UNKNOWN, "run `loki outcomes` for the measured post-merge result"
67
+
68
+
69
+ def _intent_line(loki_dir):
70
+ p = os.path.join(loki_dir, "intent", "intent.json")
71
+ if not os.path.isfile(p):
72
+ return UNKNOWN, "no intent recorded, so spec-vs-intent drift is not measurable"
73
+ doc = _read_json(p)
74
+ if doc is None:
75
+ return UNKNOWN, "the intent record could not be read"
76
+ stmts = doc.get("statements") or []
77
+ if not stmts:
78
+ return UNKNOWN, "no intent statements recorded"
79
+ linked = sum(1 for s in stmts if s.get("links"))
80
+ return "measured", f"{len(stmts)} intent statement(s), {linked} linked to the spec"
81
+
82
+
83
+ def _authorship_line(loki_dir, run_id=None):
84
+ d = os.path.join(loki_dir, "preedit")
85
+ if not os.path.isdir(d):
86
+ return UNKNOWN, "no pre-edit snapshot, so agent output cannot be told from human edits"
87
+ snaps = [f for f in os.listdir(d) if f.endswith(".json")]
88
+ if not snaps:
89
+ return UNKNOWN, "no pre-edit snapshot captured"
90
+ return "measured", f"{len(snaps)} run(s) have an immutable pre-edit snapshot"
91
+
92
+
93
+ def _grounding_line(loki_dir):
94
+ # The completion route now persists the per-claim result here
95
+ # (run.sh:_loki_check_claim_grounding, written at the non-destructive claim
96
+ # peek). Absent file means the run never reached a completion claim, which
97
+ # is UNKNOWN -- not a pass, and not a failure.
98
+ d = _read_json(os.path.join(loki_dir, "state", "claim-grounding.json"))
99
+ if not isinstance(d, dict) or d.get("status") != "measured":
100
+ return UNKNOWN, "no completion claim was checked against the diff"
101
+ named = d.get("paths_named") or []
102
+ if not named:
103
+ # Reported by name rather than scored: a claim naming no path is
104
+ # UNGROUNDABLE, which is a different fact from a claim that checked out.
105
+ return UNKNOWN, "the completion claim named no file path, so it cannot be grounded"
106
+ ungrounded = d.get("ungrounded") or []
107
+ if d.get("has_ungrounded_claim"):
108
+ return "finding", (
109
+ f"the completion claim names {len(ungrounded)} path(s) absent from the diff: "
110
+ + ", ".join(str(p) for p in ungrounded[:3])
111
+ )
112
+ return "measured", f"all {len(named)} path(s) named in the completion claim are in the diff"
113
+
114
+
115
+ def _model_line(loki_dir):
116
+ p = os.path.join(loki_dir, "decisions", "decisions.jsonl")
117
+ if not os.path.isfile(p):
118
+ return UNKNOWN, "no decision records, so a model swap would be undetectable"
119
+ models = set()
120
+ n = 0
121
+ try:
122
+ with open(p, "r", encoding="utf-8") as fh:
123
+ for line in fh:
124
+ line = line.strip()
125
+ if not line:
126
+ continue
127
+ try:
128
+ r = json.loads(line)
129
+ except ValueError:
130
+ continue
131
+ n += 1
132
+ if r.get("model_id"):
133
+ models.add(r["model_id"])
134
+ except OSError:
135
+ return UNKNOWN, "the decision record could not be read"
136
+ if not n:
137
+ return UNKNOWN, "no decision records"
138
+ if len(models) > 1:
139
+ # The single most audit-relevant thing this block can say.
140
+ return "measured", f"{n} decisions across {len(models)} DIFFERENT models: {', '.join(sorted(models))}"
141
+ return "measured", f"{n} decisions, one model: {', '.join(models) or 'unnamed'}"
142
+
143
+
144
+ def build(loki_dir, run_id=None):
145
+ rows = []
146
+ for key, question in SIGNALS:
147
+ if key == "outcome":
148
+ status, detail = _outcome_line(loki_dir)
149
+ elif key == "intent":
150
+ status, detail = _intent_line(loki_dir)
151
+ elif key == "authorship":
152
+ status, detail = _authorship_line(loki_dir, run_id)
153
+ elif key == "grounding":
154
+ status, detail = _grounding_line(loki_dir)
155
+ else:
156
+ status, detail = _model_line(loki_dir)
157
+ rows.append({"signal": key, "question": question,
158
+ "status": status, "detail": detail})
159
+
160
+ measured = sum(1 for r in rows if r["status"] == "measured")
161
+ return {
162
+ "schema_version": SCHEMA_VERSION,
163
+ "signals": rows,
164
+ "measured": measured,
165
+ "total": len(rows),
166
+ # No composite. See the module docstring: averaging a revert count, a
167
+ # hash comparison and a model id yields a number nobody can explain.
168
+ }
169
+
170
+
171
+ def render_markdown(v):
172
+ """A PR comment block. Terse on purpose: a reviewer gives this ten seconds."""
173
+ out = ["### Loki verification signals", ""]
174
+ out.append(f"{v['measured']} of {v['total']} signals measured. "
175
+ "UNKNOWN means we could not check, not that it passed.")
176
+ out.append("")
177
+ out.append("| Signal | Status | Detail |")
178
+ out.append("|---|---|---|")
179
+ for r in v["signals"]:
180
+ # Three states, not two. Collapsing everything non-measured to UNKNOWN
181
+ # would render a real finding ("the claim names a file absent from the
182
+ # diff") identically to "we could not check" -- the false equivalence
183
+ # this whole module exists to refuse. Anything unrecognised still
184
+ # degrades to UNKNOWN, so an unmeasured signal can never read as a pass.
185
+ status = r["status"] if r["status"] in ("measured", "finding") else UNKNOWN
186
+ out.append(f"| {r['signal']} | {status} | {r['detail']} |")
187
+ out.append("")
188
+ out.append("Every line is derived from a file in `.loki/` that you can read "
189
+ "yourself. No score is offered: five honest lines beat one "
190
+ "confident number.")
191
+ return "\n".join(out)
192
+
193
+
194
+ def main(argv):
195
+ as_json = "--json" in argv
196
+ cwd = os.environ.get("LOKI_VERDICT_CWD") or os.getcwd()
197
+ loki_dir = os.environ.get("LOKI_DIR") or os.path.join(cwd, ".loki")
198
+ v = build(loki_dir)
199
+ print(json.dumps(v, indent=2) if as_json else render_markdown(v))
200
+ return 0
201
+
202
+
203
+ if __name__ == "__main__":
204
+ sys.exit(main(sys.argv[1:]))