loki-mode 9.12.6 → 9.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,216 @@
1
+ #!/usr/bin/env python3
2
+ """Pre-edit snapshot: score the AGENT, not the agent-plus-whoever-fixed-it.
3
+
4
+ WHY THIS EXISTS. 8090 AI's medical-document evaluation framework -- the single
5
+ most rigorous engineering artifact in any competitor corpus -- makes an argument
6
+ we could not answer:
7
+
8
+ "A skilled writer rescues an unusable draft, and the post-edit score looks
9
+ acceptable; a junior writer leaves the model's failures visible, and the
10
+ same model appears to perform worse. In either case, the metric describes
11
+ the writer-plus-AI workflow, not the AI alone."
12
+
13
+ Every quality number we publish today is measured AFTER a human may have touched
14
+ the diff. So a strong reviewer flatters the agent and a weak one maligns it, and
15
+ the number moves for reasons that have nothing to do with the agent. That makes
16
+ our own headline partly a measurement of our users.
17
+
18
+ The fix is not analysis, it is ORDER OF OPERATIONS. 8090's own words: the two
19
+ scores must be "separated by the workflow itself, not reconstructed afterward
20
+ from logs". So this captures an immutable snapshot of the agent's output at the
21
+ moment it stops and before any human edit. Anything reconstructed later is a
22
+ guess about which lines a human wrote, and a guess is exactly what this replaces.
23
+
24
+ WHAT IT DELIBERATELY IS NOT. It does not score anything. It records WHAT WAS
25
+ THERE, with a hash, so a later comparison is a fact rather than an inference. A
26
+ low pre-edit score is a DIAGNOSTIC, not a failure -- 8090 again: "A preedit
27
+ composite of twenty percent is not a failure of the eval. It is a diagnostic,
28
+ document-by-document, of where the model is weak." Nothing here feeds a gate,
29
+ because a gate that punished the agent for needing edits would train the agent to
30
+ produce diffs nobody edits, which is not the same as good diffs.
31
+
32
+ WHAT ALREADY EXISTS AND IS NOT REBUILT. `_compute_headline` in
33
+ proof-generator.py already implements the asymmetric-risk half of this idea, in
34
+ its own words: "A failed check is a stronger negative signal than a not-run one:
35
+ amber means we did not check everything, red means something we checked did not
36
+ pass." That is F2-style weighting -- treating a false negative as worse than a
37
+ false positive -- already shipped. This module adds only the missing half.
38
+ """
39
+
40
+ from __future__ import annotations
41
+
42
+ import hashlib
43
+ import json
44
+ import os
45
+ import subprocess
46
+ import sys
47
+ from datetime import datetime, timezone
48
+
49
+ SCHEMA_VERSION = "1.0"
50
+
51
+ UNKNOWN = "UNKNOWN"
52
+
53
+ # Named refusals, mirroring the outcome ledger's ANCHOR_REASONS. A comparison we
54
+ # cannot compute is reported BY NAME, never as zero and never as a pass.
55
+ COMPARE_REASONS = {
56
+ "no_snapshot": "no pre-edit snapshot was captured for this run",
57
+ "snapshot_unreadable": "the snapshot exists but could not be parsed",
58
+ "not_a_git_repo": "not a git repository, so the current diff cannot be read",
59
+ "no_current_diff": "there is no current diff to compare the snapshot against",
60
+ }
61
+
62
+
63
+ def _git(args, cwd, timeout=60):
64
+ try:
65
+ p = subprocess.run(["git"] + list(args), cwd=cwd,
66
+ capture_output=True, text=True, timeout=timeout)
67
+ return p.returncode, p.stdout
68
+ except (subprocess.TimeoutExpired, OSError):
69
+ return 1, ""
70
+
71
+
72
+ def _sha256(text):
73
+ return hashlib.sha256(text.encode("utf-8", "replace")).hexdigest()
74
+
75
+
76
+ def capture(loki_dir, run_id, cwd, base_sha=None):
77
+ """Freeze the agent's output before any human edit.
78
+
79
+ Stores the full unified diff plus its sha256. The hash is what makes a later
80
+ comparison evidence rather than an assertion: anyone can recompute it.
81
+
82
+ Called by the workflow at the moment the agent stops. Capturing it later --
83
+ after a review, from a log -- would record the human's work as the agent's,
84
+ which is the exact contamination this exists to prevent.
85
+ """
86
+ rc, _ = _git(["rev-parse", "--git-dir"], cwd)
87
+ if rc != 0:
88
+ return {"status": UNKNOWN, "reason": "not_a_git_repo"}
89
+
90
+ if base_sha:
91
+ rc, diff = _git(["diff", base_sha, "--"], cwd)
92
+ else:
93
+ # Uncommitted agent work is the normal case at stop time.
94
+ rc, diff = _git(["diff", "HEAD", "--"], cwd)
95
+ if rc != 0:
96
+ return {"status": UNKNOWN, "reason": "no_current_diff"}
97
+
98
+ rc2, names = _git(["diff", "--name-only"] + ([base_sha] if base_sha else ["HEAD"]), cwd)
99
+ files = [l.strip() for l in names.splitlines() if l.strip()] if rc2 == 0 else []
100
+
101
+ snap = {
102
+ "schema_version": SCHEMA_VERSION,
103
+ "run_id": run_id,
104
+ "captured_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
105
+ "base_sha": base_sha or "",
106
+ # The point of the whole file: this is the agent's output, uncontaminated.
107
+ "author": "agent",
108
+ "diff_sha256": _sha256(diff),
109
+ "diff_bytes": len(diff.encode("utf-8", "replace")),
110
+ "files": files,
111
+ "lines_added": sum(1 for l in diff.splitlines()
112
+ if l.startswith("+") and not l.startswith("+++")),
113
+ "lines_removed": sum(1 for l in diff.splitlines()
114
+ if l.startswith("-") and not l.startswith("---")),
115
+ }
116
+
117
+ out_dir = os.path.join(loki_dir, "preedit")
118
+ try:
119
+ os.makedirs(out_dir, exist_ok=True)
120
+ path = os.path.join(out_dir, f"{run_id}.json")
121
+ # Write-once. A snapshot that a later run can overwrite is not a
122
+ # snapshot -- re-capturing after a human edit would silently record the
123
+ # human's work as the agent's, which is the contamination this prevents.
124
+ if os.path.exists(path):
125
+ return {"status": "exists", "path": path}
126
+ with open(path, "w", encoding="utf-8") as fh:
127
+ json.dump(snap, fh, indent=2)
128
+ fh.write("\n")
129
+ # The raw diff is kept beside the metadata so the hash is checkable by
130
+ # hand. Evidence nobody can recompute is a claim, not evidence.
131
+ with open(os.path.join(out_dir, f"{run_id}.diff"), "w", encoding="utf-8") as fh:
132
+ fh.write(diff)
133
+ except OSError as exc:
134
+ return {"status": UNKNOWN, "reason": "snapshot_unreadable", "detail": str(exc)}
135
+
136
+ return {"status": "captured", "path": path, "diff_sha256": snap["diff_sha256"]}
137
+
138
+
139
+ def compare(loki_dir, run_id, cwd):
140
+ """How much of what shipped was the agent's, and how much was human rescue.
141
+
142
+ Reports lines, never a score. The split is the useful fact; a composite
143
+ number invites exactly the ranking behaviour that made post-edit scores
144
+ misleading in the first place.
145
+ """
146
+ path = os.path.join(loki_dir, "preedit", f"{run_id}.json")
147
+ if not os.path.isfile(path):
148
+ return {"status": UNKNOWN, "reason": "no_snapshot",
149
+ "detail": COMPARE_REASONS["no_snapshot"]}
150
+ try:
151
+ with open(path, "r", encoding="utf-8") as fh:
152
+ snap = json.load(fh)
153
+ except (OSError, ValueError):
154
+ return {"status": UNKNOWN, "reason": "snapshot_unreadable",
155
+ "detail": COMPARE_REASONS["snapshot_unreadable"]}
156
+
157
+ base = snap.get("base_sha") or "HEAD"
158
+ rc, diff_now = _git(["diff", base, "--"], cwd)
159
+ if rc != 0:
160
+ return {"status": UNKNOWN, "reason": "no_current_diff",
161
+ "detail": COMPARE_REASONS["no_current_diff"]}
162
+
163
+ now_hash = _sha256(diff_now)
164
+ added_now = sum(1 for l in diff_now.splitlines()
165
+ if l.startswith("+") and not l.startswith("+++"))
166
+
167
+ untouched = now_hash == snap.get("diff_sha256")
168
+ return {
169
+ "status": "measured",
170
+ "run_id": run_id,
171
+ "agent_lines_added": snap.get("lines_added"),
172
+ "current_lines_added": added_now,
173
+ # A hash match is proof nobody edited it; a mismatch proves somebody did,
174
+ # but NOT how much -- attributing individual lines would be the guess
175
+ # this module exists to avoid.
176
+ "human_edited": (not untouched),
177
+ "preedit_diff_sha256": snap.get("diff_sha256"),
178
+ "current_diff_sha256": now_hash,
179
+ "captured_at": snap.get("captured_at"),
180
+ "note": ("the pre-edit diff is byte-identical to what shipped, so this "
181
+ "receipt measures the agent alone"
182
+ if untouched else
183
+ "what shipped differs from the agent's output, so any score "
184
+ "taken now measures the agent plus the human who edited it"),
185
+ }
186
+
187
+
188
+ def main(argv):
189
+ if not argv:
190
+ print("usage: preedit_snapshot.py capture|compare <run_id> [--json]",
191
+ file=sys.stderr)
192
+ return 2
193
+ action = argv[0]
194
+ run_id = argv[1] if len(argv) > 1 else ""
195
+ if not run_id:
196
+ print("a run_id is required", file=sys.stderr)
197
+ return 2
198
+
199
+ cwd = os.environ.get("LOKI_PREEDIT_CWD") or os.getcwd()
200
+ loki_dir = os.environ.get("LOKI_DIR") or os.path.join(cwd, ".loki")
201
+
202
+ if action == "capture":
203
+ res = capture(loki_dir, run_id, cwd,
204
+ base_sha=os.environ.get("LOKI_RUN_START_SHA") or None)
205
+ elif action == "compare":
206
+ res = compare(loki_dir, run_id, cwd)
207
+ else:
208
+ print(f"unknown action: {action}", file=sys.stderr)
209
+ return 2
210
+
211
+ print(json.dumps(res, indent=2))
212
+ return 0 if res.get("status") in ("captured", "exists", "measured") else 3
213
+
214
+
215
+ if __name__ == "__main__":
216
+ sys.exit(main(sys.argv[1:]))
@@ -0,0 +1,204 @@
1
+ #!/usr/bin/env python3
2
+ """One verdict a reviewer can read in ten seconds, assembled from measured parts.
3
+
4
+ WHY THIS EXISTS. We now measure five things nobody else does -- did the work
5
+ survive (outcome ledger), does the spec still match the intent (intent ledger),
6
+ was this the agent's own output or a human rescue (pre-edit snapshot), does the
7
+ completion claim name real work (claim grounding), and which model decided
8
+ (decision record). Each is correct and each lives in its own file. A reviewer
9
+ looking at a pull request reads none of them.
10
+
11
+ A moat nobody sees is not a moat. The Evidence Receipt is already attached to
12
+ PRs by default (LOKI_PROVEN_PR, autonomy/run.sh:3849), so the surface exists and
13
+ this fills it: one block, five lines, each of which is either a measured fact or
14
+ an explicit UNKNOWN.
15
+
16
+ WHAT IT REFUSES, AND WHY THAT MATTERS MORE HERE THAN ANYWHERE. This is the only
17
+ place where all five signals meet, which makes it the one place a composite score
18
+ would be tempting: a single "trust: 87" would fit a PR comment beautifully. It is
19
+ not offered. Averaging a revert count, a hash comparison, a diff hash, a path
20
+ match and a model id produces a number whose movement nobody can explain, and a
21
+ number nobody can explain is the thing our competitors already ship. Five honest
22
+ lines beat one confident one.
23
+
24
+ UNKNOWN IS PRINTED, NOT HIDDEN. A reviewer must be able to tell "we checked and
25
+ it is fine" from "we could not check". Suppressing the unmeasured lines would
26
+ make a receipt with one real signal look identical to one with five, which is
27
+ exactly the false confidence the whole product exists to refuse.
28
+ """
29
+
30
+ from __future__ import annotations
31
+
32
+ import json
33
+ import os
34
+ import sys
35
+
36
+ SCHEMA_VERSION = "1.0"
37
+
38
+ UNKNOWN = "UNKNOWN"
39
+
40
+ # Each signal: where it lives, and the one-line question it answers for a human
41
+ # scanning a PR. The question text is load-bearing -- a reviewer who cannot tell
42
+ # what a line MEANS will skip the block, and a skipped block is worth nothing.
43
+ SIGNALS = (
44
+ ("outcome", "did previous work survive, or get reverted"),
45
+ ("intent", "does the spec still say what was actually wanted"),
46
+ ("authorship", "is this the agent's output, or a human rescue"),
47
+ ("grounding", "does the completion claim name work that exists in the diff"),
48
+ ("model", "which model decided, at what temperature"),
49
+ )
50
+
51
+
52
+ def _read_json(path):
53
+ try:
54
+ with open(path, "r", encoding="utf-8") as fh:
55
+ return json.load(fh)
56
+ except (OSError, ValueError):
57
+ return None
58
+
59
+
60
+ def _outcome_line(loki_dir):
61
+ p = os.path.join(loki_dir, "proofs")
62
+ if not os.path.isdir(p):
63
+ return UNKNOWN, "no receipts yet, so nothing has an outcome to follow"
64
+ # Deliberately does not recompute: the outcome ledger owns that logic and two
65
+ # implementations of one number eventually disagree.
66
+ return UNKNOWN, "run `loki outcomes` for the measured post-merge result"
67
+
68
+
69
+ def _intent_line(loki_dir):
70
+ p = os.path.join(loki_dir, "intent", "intent.json")
71
+ if not os.path.isfile(p):
72
+ return UNKNOWN, "no intent recorded, so spec-vs-intent drift is not measurable"
73
+ doc = _read_json(p)
74
+ if doc is None:
75
+ return UNKNOWN, "the intent record could not be read"
76
+ stmts = doc.get("statements") or []
77
+ if not stmts:
78
+ return UNKNOWN, "no intent statements recorded"
79
+ linked = sum(1 for s in stmts if s.get("links"))
80
+ return "measured", f"{len(stmts)} intent statement(s), {linked} linked to the spec"
81
+
82
+
83
+ def _authorship_line(loki_dir, run_id=None):
84
+ d = os.path.join(loki_dir, "preedit")
85
+ if not os.path.isdir(d):
86
+ return UNKNOWN, "no pre-edit snapshot, so agent output cannot be told from human edits"
87
+ snaps = [f for f in os.listdir(d) if f.endswith(".json")]
88
+ if not snaps:
89
+ return UNKNOWN, "no pre-edit snapshot captured"
90
+ return "measured", f"{len(snaps)} run(s) have an immutable pre-edit snapshot"
91
+
92
+
93
+ def _grounding_line(loki_dir):
94
+ # The completion route now persists the per-claim result here
95
+ # (run.sh:_loki_check_claim_grounding, written at the non-destructive claim
96
+ # peek). Absent file means the run never reached a completion claim, which
97
+ # is UNKNOWN -- not a pass, and not a failure.
98
+ d = _read_json(os.path.join(loki_dir, "state", "claim-grounding.json"))
99
+ if not isinstance(d, dict) or d.get("status") != "measured":
100
+ return UNKNOWN, "no completion claim was checked against the diff"
101
+ named = d.get("paths_named") or []
102
+ if not named:
103
+ # Reported by name rather than scored: a claim naming no path is
104
+ # UNGROUNDABLE, which is a different fact from a claim that checked out.
105
+ return UNKNOWN, "the completion claim named no file path, so it cannot be grounded"
106
+ ungrounded = d.get("ungrounded") or []
107
+ if d.get("has_ungrounded_claim"):
108
+ return "finding", (
109
+ f"the completion claim names {len(ungrounded)} path(s) absent from the diff: "
110
+ + ", ".join(str(p) for p in ungrounded[:3])
111
+ )
112
+ return "measured", f"all {len(named)} path(s) named in the completion claim are in the diff"
113
+
114
+
115
+ def _model_line(loki_dir):
116
+ p = os.path.join(loki_dir, "decisions", "decisions.jsonl")
117
+ if not os.path.isfile(p):
118
+ return UNKNOWN, "no decision records, so a model swap would be undetectable"
119
+ models = set()
120
+ n = 0
121
+ try:
122
+ with open(p, "r", encoding="utf-8") as fh:
123
+ for line in fh:
124
+ line = line.strip()
125
+ if not line:
126
+ continue
127
+ try:
128
+ r = json.loads(line)
129
+ except ValueError:
130
+ continue
131
+ n += 1
132
+ if r.get("model_id"):
133
+ models.add(r["model_id"])
134
+ except OSError:
135
+ return UNKNOWN, "the decision record could not be read"
136
+ if not n:
137
+ return UNKNOWN, "no decision records"
138
+ if len(models) > 1:
139
+ # The single most audit-relevant thing this block can say.
140
+ return "measured", f"{n} decisions across {len(models)} DIFFERENT models: {', '.join(sorted(models))}"
141
+ return "measured", f"{n} decisions, one model: {', '.join(models) or 'unnamed'}"
142
+
143
+
144
+ def build(loki_dir, run_id=None):
145
+ rows = []
146
+ for key, question in SIGNALS:
147
+ if key == "outcome":
148
+ status, detail = _outcome_line(loki_dir)
149
+ elif key == "intent":
150
+ status, detail = _intent_line(loki_dir)
151
+ elif key == "authorship":
152
+ status, detail = _authorship_line(loki_dir, run_id)
153
+ elif key == "grounding":
154
+ status, detail = _grounding_line(loki_dir)
155
+ else:
156
+ status, detail = _model_line(loki_dir)
157
+ rows.append({"signal": key, "question": question,
158
+ "status": status, "detail": detail})
159
+
160
+ measured = sum(1 for r in rows if r["status"] == "measured")
161
+ return {
162
+ "schema_version": SCHEMA_VERSION,
163
+ "signals": rows,
164
+ "measured": measured,
165
+ "total": len(rows),
166
+ # No composite. See the module docstring: averaging a revert count, a
167
+ # hash comparison and a model id yields a number nobody can explain.
168
+ }
169
+
170
+
171
+ def render_markdown(v):
172
+ """A PR comment block. Terse on purpose: a reviewer gives this ten seconds."""
173
+ out = ["### Loki verification signals", ""]
174
+ out.append(f"{v['measured']} of {v['total']} signals measured. "
175
+ "UNKNOWN means we could not check, not that it passed.")
176
+ out.append("")
177
+ out.append("| Signal | Status | Detail |")
178
+ out.append("|---|---|---|")
179
+ for r in v["signals"]:
180
+ # Three states, not two. Collapsing everything non-measured to UNKNOWN
181
+ # would render a real finding ("the claim names a file absent from the
182
+ # diff") identically to "we could not check" -- the false equivalence
183
+ # this whole module exists to refuse. Anything unrecognised still
184
+ # degrades to UNKNOWN, so an unmeasured signal can never read as a pass.
185
+ status = r["status"] if r["status"] in ("measured", "finding") else UNKNOWN
186
+ out.append(f"| {r['signal']} | {status} | {r['detail']} |")
187
+ out.append("")
188
+ out.append("Every line is derived from a file in `.loki/` that you can read "
189
+ "yourself. No score is offered: five honest lines beat one "
190
+ "confident number.")
191
+ return "\n".join(out)
192
+
193
+
194
+ def main(argv):
195
+ as_json = "--json" in argv
196
+ cwd = os.environ.get("LOKI_VERDICT_CWD") or os.getcwd()
197
+ loki_dir = os.environ.get("LOKI_DIR") or os.path.join(cwd, ".loki")
198
+ v = build(loki_dir)
199
+ print(json.dumps(v, indent=2) if as_json else render_markdown(v))
200
+ return 0
201
+
202
+
203
+ if __name__ == "__main__":
204
+ sys.exit(main(sys.argv[1:]))