loki-mode 9.8.0 → 9.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +19 -14
  2. package/SKILL.md +3 -2
  3. package/VERSION +1 -1
  4. package/autonomy/loki +122 -1
  5. package/autonomy/run.sh +49 -2
  6. package/dashboard/__init__.py +1 -1
  7. package/dashboard/api_evidence.py +411 -0
  8. package/dashboard/api_operator.py +283 -0
  9. package/dashboard/api_phases.py +262 -0
  10. package/dashboard/api_releases.py +242 -0
  11. package/dashboard/api_runs.py +477 -0
  12. package/dashboard/api_tests.py +444 -0
  13. package/dashboard/api_v2.py +47 -1
  14. package/dashboard/server.py +54 -0
  15. package/dashboard/static/index.html +246 -135
  16. package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
  17. package/docs/CAPABILITY-BACKLOG.md +53 -0
  18. package/docs/COMPARISON.md +2 -2
  19. package/docs/COMPETITIVE-ANALYSIS.md +1 -1
  20. package/docs/COMPETITIVE-SCORECARD.md +422 -0
  21. package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
  22. package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
  23. package/docs/DEMOS.md +21 -23
  24. package/docs/HANDOFF-2026-08-03.md +439 -0
  25. package/docs/INSTALLATION.md +17 -10
  26. package/docs/OUTCOME-FRONTIER.md +536 -0
  27. package/docs/PROMPT-ABLATION-RESULT.md +97 -0
  28. package/docs/TOOLS.md +800 -0
  29. package/docs/alternative-installations.md +2 -3
  30. package/docs/audit-logging.md +44 -35
  31. package/docs/authentication.md +13 -2
  32. package/docs/authorization.md +87 -81
  33. package/docs/git-workflow.md +6 -3
  34. package/docs/metrics.md +15 -16
  35. package/docs/network-security.md +16 -13
  36. package/docs/openclaw-integration.md +36 -556
  37. package/docs/show-hn-post.md +2 -2
  38. package/docs/siem-integration.md +39 -36
  39. package/loki-ts/dist/loki.js +18 -18
  40. package/mcp/__init__.py +1 -1
  41. package/package.json +2 -2
  42. package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
  43. package/references/confidence-routing.md +18 -1
  44. package/references/invariant-checks.md +13 -8
  45. package/references/magic-rarv-integration.md +0 -1
  46. package/references/multi-provider.md +27 -5
  47. package/skills/healing.md +4 -2
  48. package/tools/audit-docs.py +488 -0
  49. package/tools/baseline-pin.py +19 -1
  50. package/tools/calibration-audit.py +523 -0
  51. package/tools/ci-gate.py +19 -1
  52. package/tools/cost-forecast.py +344 -0
  53. package/tools/cost-guard.py +19 -1
  54. package/tools/cost-history.py +19 -1
  55. package/tools/cost-per-outcome.py +394 -0
  56. package/tools/estimate-run.py +19 -1
  57. package/tools/evidence-freshness.py +307 -0
  58. package/tools/gate-init.py +19 -1
  59. package/tools/gate-report.py +19 -1
  60. package/tools/gate-simulate.py +570 -0
  61. package/tools/gate-trend.py +354 -0
  62. package/tools/model-advisor.py +52 -1
  63. package/tools/policy-load.py +19 -1
  64. package/tools/prompt-cost.py +363 -0
  65. package/tools/prompt-diff.py +448 -0
  66. package/tools/prompt-lint.py +448 -0
  67. package/tools/receipt-bundle.py +72 -2
  68. package/tools/receipt-diff.py +19 -1
  69. package/tools/receipt-find.py +19 -1
  70. package/tools/receipt-stats.py +380 -0
  71. package/tools/receipt-timeline.py +478 -0
  72. package/tools/receipt-verify-batch.py +291 -0
  73. package/tools/run-replay.py +19 -1
  74. package/tools/signing-status.py +19 -1
  75. package/tools/token-guard.py +19 -1
  76. package/tools/token-tax.py +375 -0
  77. package/tools/tool-index.py +19 -1
  78. package/tools/verification-tax.py +277 -0
  79. package/tools/verify-chain.py +361 -0
@@ -0,0 +1,375 @@
1
+ #!/usr/bin/env python3
2
+ """What does verification cost in TOKENS, versus the build itself?
3
+
4
+ WHY THIS EXISTS. tools/verification-tax.py measures the gate in WALL TIME and
5
+ in how often it changed an outcome, and says in its own header that it does not
6
+ measure token cost. docs/CAPABILITY-BACKLOG.md carries that as the open half of
7
+ a SHIPPED-partial row ("Remaining: token cost"). This is that half.
8
+
9
+ WHAT IT REPORTS, and the one thing it REFUSES.
10
+
11
+ Tokens are recorded per ITERATION, in .loki/metrics/efficiency/iteration-N.json.
12
+ Stages are recorded per STAGE, as stage_complete events carrying a name, a
13
+ status and a duration_s. Those are the only two artifacts a run writes, and
14
+ NEITHER attributes a token to a stage. So:
15
+
16
+ THE VERIFICATION SHARE OF TOKENS IS NOT REPORTED, BECAUSE NOTHING
17
+ RECORDS IT.
18
+
19
+ The tempting fix is to weight each iteration's tokens by each stage's share of
20
+ that iteration's wall clock. tools/cost-attribute.py already refused exactly
21
+ that for dollars and named it correctly: the same invention wearing a
22
+ defensible-looking coat, and worse than an even split precisely because a
23
+ reader will believe it. A gate that shells out to eslint for thirty seconds
24
+ spends no tokens; a review stage that streams three completions in ten seconds
25
+ spends most of the iteration. Wall clock is not a token proxy.
26
+
27
+ What IS measured, and is reported:
28
+
29
+ tokens per iteration measured, unmeasured iterations excluded
30
+ verification stages run counted from stage_complete, by name
31
+ build stages run same, for the one stage that IS the agent
32
+ tokens per verification-stage-execution the denominator is measured and
33
+ the numerator is measured, but the DIVISION is
34
+ only an upper bound and is labelled as one
35
+
36
+ THE HONESTY RULE, which is the reason for the whole file. An iteration with no
37
+ recorded token counts reads UNKNOWN and is EXCLUDED from every average. It is
38
+ never averaged in as zero. Averaging absent measurements toward zero slides the
39
+ mean toward "verification is free" -- which is the conclusion someone building
40
+ a case for removing gates would want, reached by arithmetic rather than by
41
+ evidence. Measured-ness is record_is_measured() in
42
+ autonomy/lib/efficiency_cost.py, imported and never restated, fed the FOUR
43
+ TOKEN FIELDS ONLY: a record carrying a real cost_usd and zero tokens (the shape
44
+ codex wrote before v8.51.0) is a run whose TOKENS were never measured, and
45
+ feeding cost_usd in would report it as "0 tokens", which is the exact lie this
46
+ file exists to prevent.
47
+
48
+ WHAT THIS IS NOT. Not a gate. A workspace whose every gate BLOCKED still exits
49
+ 0 here, because the question asked was "what did it cost", and that question
50
+ was answered. tools/token-guard.py is the gate.
51
+
52
+ Usage:
53
+ tools/token-tax.py [workspace] [--json]
54
+
55
+ Exit: 0 reported, 2 could not check, 3 nothing to report, 64 usage error,
56
+ 66 input path missing.
57
+ """
58
+
59
+ import argparse
60
+ import json
61
+ import os
62
+ import sys
63
+
64
+ sys.dont_write_bytecode = True
65
+
66
+ _HERE = os.path.dirname(os.path.abspath(__file__))
67
+ # Resolved from __file__, never from the workspace argument: the workspace being
68
+ # read is a different tree and has no autonomy/lib.
69
+ sys.path.insert(0, os.path.join(os.path.dirname(_HERE), "autonomy", "lib"))
70
+
71
+ from efficiency_cost import record_is_measured # noqa: E402
72
+
73
+ OK = 0
74
+ COULD_NOT_CHECK = 2
75
+ NOTHING_TO_REPORT = 3
76
+ USAGE_ERROR = 64
77
+ INPUT_MISSING = 66
78
+
79
+ # The four token fields. cost_usd is deliberately absent; see the docstring.
80
+ TOKEN_FIELDS = ("input_tokens", "output_tokens", "cache_read_tokens",
81
+ "cache_creation_tokens")
82
+
83
+ # The one stage that IS the build. Every other stage emitted by
84
+ # emit_stage_complete in autonomy/run.sh is verification or gate work.
85
+ BUILD_STAGES = frozenset(["agent"])
86
+
87
+ WHY_NO_SPLIT = (
88
+ "REFUSED. Tokens are recorded per ITERATION and stages are recorded per "
89
+ "STAGE; no artifact this repo writes attributes a token to a stage. "
90
+ "Weighting each iteration's tokens by a stage's share of wall clock would "
91
+ "look like a measurement and would not be one -- a gate shelling out to a "
92
+ "linter burns seconds and no tokens, while one streaming a completion "
93
+ "burns tokens in no time. The split becomes reportable when a writer "
94
+ "records per-stage usage, not before."
95
+ )
96
+
97
+
98
+ class _Parser(argparse.ArgumentParser):
99
+ """Usage errors exit 64, not argparse's default 2.
100
+
101
+ In this repo's convention 2 means "could NOT be checked" -- a real answer
102
+ about the subject. A mistyped flag is not that: it is an error about the
103
+ INVOCATION, and nothing about the subject was examined. The two call for
104
+ opposite responses, since retrying cannot fix a typo.
105
+ """
106
+
107
+ def error(self, message):
108
+ self.print_usage(sys.stderr)
109
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
110
+ raise SystemExit(USAGE_ERROR)
111
+
112
+
113
+ def _num(v):
114
+ """A number as itself; None, "", or a bool as None."""
115
+ if isinstance(v, bool) or not isinstance(v, (int, float)):
116
+ return None
117
+ return v
118
+
119
+
120
+ def measured_tokens(rec):
121
+ """The four token fields summed, or None when this run never measured them.
122
+
123
+ None means UNKNOWN. The caller must not treat it as zero -- that is the
124
+ single property this whole file is built around.
125
+ """
126
+ if not isinstance(rec, dict):
127
+ return None
128
+ fields = {f: _num(rec.get(f)) for f in TOKEN_FIELDS}
129
+ if not record_is_measured(fields):
130
+ return None
131
+ # Measured, so a None in one field is a real gap inside a real measurement.
132
+ # Count it as 0 for summing rather than poisoning the whole record.
133
+ return {f: (0 if fields[f] is None else fields[f]) for f in TOKEN_FIELDS}
134
+
135
+
136
+ def read_iterations(loki_dir):
137
+ """[{iteration, tokens}] for every iteration-N.json, tokens None if unmeasured.
138
+
139
+ Unreadable and non-dict records are kept with tokens=None rather than
140
+ dropped: a file that exists and cannot be read is an iteration that
141
+ happened and was not measured, which is exactly what this reports.
142
+ """
143
+ eff_dir = os.path.join(loki_dir, "metrics", "efficiency")
144
+ try:
145
+ names = sorted(os.listdir(eff_dir))
146
+ except OSError:
147
+ return []
148
+ out = []
149
+ for name in names:
150
+ if not (name.startswith("iteration-") and name.endswith(".json")):
151
+ continue
152
+ try:
153
+ n = int(name[len("iteration-"):-len(".json")])
154
+ except ValueError:
155
+ continue
156
+ try:
157
+ with open(os.path.join(eff_dir, name), "r", encoding="utf-8") as fh:
158
+ rec = json.load(fh)
159
+ except (OSError, ValueError):
160
+ rec = None
161
+ out.append({"iteration": n, "tokens": measured_tokens(rec)})
162
+ out.sort(key=lambda r: r["iteration"])
163
+ return out
164
+
165
+
166
+ def read_stages(loki_dir):
167
+ """({stage: executions}, corrupt_line_count) from stage_complete events.
168
+
169
+ Corrupt lines are counted, never silently dropped: a tool that skips bad
170
+ lines reports a cleaner run than happened, and skips most often when
171
+ something upstream is broken.
172
+ """
173
+ path = os.path.join(loki_dir, "events.jsonl")
174
+ counts, corrupt = {}, 0
175
+ try:
176
+ fh = open(path, "r", encoding="utf-8", errors="replace")
177
+ except OSError:
178
+ return counts, corrupt
179
+ with fh:
180
+ for line in fh:
181
+ if not line.strip():
182
+ continue
183
+ try:
184
+ e = json.loads(line)
185
+ except ValueError:
186
+ corrupt += 1
187
+ continue
188
+ if not isinstance(e, dict):
189
+ corrupt += 1
190
+ continue
191
+ if (e.get("type") or e.get("event")) != "stage_complete":
192
+ continue
193
+ data = e.get("data") if isinstance(e.get("data"), dict) else {}
194
+ name = data.get("stage")
195
+ if name:
196
+ counts[str(name)] = counts.get(str(name), 0) + 1
197
+ return counts, corrupt
198
+
199
+
200
+ def summarize(iterations, stages, corrupt):
201
+ """The report. Averages cover MEASURED iterations only, and say so."""
202
+ # THE EXCLUSION. One filter site, not a repeated guard: an unmeasured
203
+ # iteration leaves the arithmetic entirely rather than entering it as zero.
204
+ measured = [r for r in iterations if r["tokens"] is not None]
205
+ unmeasured = len(iterations) - len(measured)
206
+
207
+ totals = {f: sum(r["tokens"][f] for r in measured) for f in TOKEN_FIELDS}
208
+ grand = sum(totals.values()) if measured else None
209
+
210
+ verification = {k: v for k, v in stages.items() if k not in BUILD_STAGES}
211
+ build = {k: v for k, v in stages.items() if k in BUILD_STAGES}
212
+ v_execs = sum(verification.values())
213
+
214
+ # An UPPER BOUND, and labelled as one everywhere it appears: it divides ALL
215
+ # of a run's tokens by the verification executions, as though the build
216
+ # spent none. It is the largest the true figure could be, which makes it
217
+ # useful for bounding a claim and useless as an estimate.
218
+ ceiling = None
219
+ if grand is not None and v_execs:
220
+ ceiling = round(grand / v_execs, 1)
221
+
222
+ return {
223
+ "iterations": len(iterations),
224
+ "measured_iterations": len(measured),
225
+ "unmeasured_iterations": unmeasured,
226
+ "corrupt_lines": corrupt,
227
+ "totals": totals if measured else {f: None for f in TOKEN_FIELDS},
228
+ "total_tokens": grand,
229
+ "mean_tokens_per_measured_iteration": (
230
+ round(grand / len(measured), 1) if measured else None),
231
+ "per_iteration": [
232
+ {"iteration": r["iteration"],
233
+ "tokens": (None if r["tokens"] is None
234
+ else sum(r["tokens"].values()))}
235
+ for r in iterations
236
+ ],
237
+ "verification_stage_executions": v_execs,
238
+ "verification_stages": dict(sorted(verification.items())),
239
+ "build_stage_executions": sum(build.values()),
240
+ "build_stages": dict(sorted(build.items())),
241
+ "verification_tokens": None,
242
+ "build_tokens": None,
243
+ "verification_token_share": None,
244
+ "verification_token_share_why": WHY_NO_SPLIT,
245
+ "tokens_per_verification_execution_upper_bound": ceiling,
246
+ }
247
+
248
+
249
+ def render(s):
250
+ out = ["TOKEN TAX -- read from artifacts only; nothing was started and "
251
+ "nothing was spent.", ""]
252
+ out.append(" iterations %d" % s["iterations"])
253
+ out.append(" measured %d of %d"
254
+ % (s["measured_iterations"], s["iterations"]))
255
+
256
+ if s["measured_iterations"]:
257
+ out.append(" total tokens %d" % s["total_tokens"])
258
+ for f in TOKEN_FIELDS:
259
+ out.append(" %-22s %d" % (f, s["totals"][f]))
260
+ out.append(" mean per measured %.1f"
261
+ % s["mean_tokens_per_measured_iteration"])
262
+ else:
263
+ out.append(" total tokens UNKNOWN (no iteration recorded any "
264
+ "token count)")
265
+ out.append(" mean per measured UNKNOWN")
266
+
267
+ if s["unmeasured_iterations"]:
268
+ out.append(" UNMEASURED %d iteration(s) carry no token count; "
269
+ "EXCLUDED from" % s["unmeasured_iterations"])
270
+ out.append(" the totals and the mean above, NOT "
271
+ "counted as 0 tokens.")
272
+ out.append(" The real figure is HIGHER than what "
273
+ "is printed here.")
274
+
275
+ out.append("")
276
+ out.append("PER ITERATION")
277
+ out.append("-" * 60)
278
+ for r in s["per_iteration"]:
279
+ if r["tokens"] is None:
280
+ out.append(" iteration %-8d UNKNOWN (not measured; excluded)"
281
+ % r["iteration"])
282
+ else:
283
+ out.append(" iteration %-8d %d tokens" % (r["iteration"],
284
+ r["tokens"]))
285
+
286
+ out.append("")
287
+ out.append("STAGES RUN (counted, not priced)")
288
+ out.append("-" * 60)
289
+ if s["verification_stages"]:
290
+ for name, n in s["verification_stages"].items():
291
+ out.append(" verification %-22s %d execution(s)" % (name, n))
292
+ else:
293
+ out.append(" verification no stage_complete record -- stages not "
294
+ "recorded (not 0 runs)")
295
+ for name, n in s["build_stages"].items():
296
+ out.append(" build %-22s %d execution(s)" % (name, n))
297
+
298
+ out.append("")
299
+ out.append("VERIFICATION SHARE OF TOKENS")
300
+ out.append("-" * 60)
301
+ for chunk in WHY_NO_SPLIT.split(" -- "):
302
+ out.append(" %s" % chunk)
303
+ if s["tokens_per_verification_execution_upper_bound"] is not None:
304
+ out.append("")
305
+ out.append(" UPPER BOUND ONLY: %.1f tokens per verification execution "
306
+ "IF the build"
307
+ % s["tokens_per_verification_execution_upper_bound"])
308
+ out.append(" had spent none, which it did not. This bounds a claim; "
309
+ "it does not estimate one.")
310
+
311
+ if s["corrupt_lines"]:
312
+ out.append("")
313
+ out.append(" CORRUPT %d unreadable event line(s), counted not "
314
+ "dropped" % s["corrupt_lines"])
315
+
316
+ out.append("")
317
+ if not s["measured_iterations"]:
318
+ out.append(" READ: token cost is UNMEASURED. No claim that "
319
+ "verification is cheap or")
320
+ out.append(" expensive can be made from this workspace. Absence of a "
321
+ "number is not a")
322
+ out.append(" small number.")
323
+ return "\n".join(out)
324
+
325
+
326
+ def main(argv=None):
327
+ ap = _Parser(
328
+ description="Report what verification cost in TOKENS versus the "
329
+ "build. Reports only; never blocks.")
330
+ ap.add_argument("workspace", nargs="?", default=".",
331
+ help="workspace root containing .loki/ (default .)")
332
+ ap.add_argument("--json", action="store_true", dest="as_json",
333
+ help="emit the report as JSON")
334
+ args = ap.parse_args(argv)
335
+
336
+ if not os.path.exists(args.workspace):
337
+ # 66, not COULD_NOT_CHECK. "The path you named does not exist" is a
338
+ # fact about the INPUT; "I could not evaluate" is a fact about the
339
+ # subject. A caller retrying on 2 would retry forever against a typo.
340
+ sys.stderr.write("token-tax: no such workspace: %s\n" % args.workspace)
341
+ return INPUT_MISSING
342
+
343
+ loki = args.workspace
344
+ if os.path.basename(os.path.normpath(loki)) != ".loki":
345
+ loki = os.path.join(loki, ".loki")
346
+ if not os.path.isdir(loki):
347
+ sys.stderr.write("token-tax: no .loki directory under %s; nothing to "
348
+ "report\n" % args.workspace)
349
+ return NOTHING_TO_REPORT
350
+
351
+ try:
352
+ iterations = read_iterations(loki)
353
+ stages, corrupt = read_stages(loki)
354
+ except OSError as exc:
355
+ sys.stderr.write("token-tax: could not read %s: %s\n" % (loki, exc))
356
+ return COULD_NOT_CHECK
357
+
358
+ if not iterations and not stages and not corrupt:
359
+ sys.stderr.write("token-tax: no efficiency records and no stage "
360
+ "events under %s; nothing to report\n" % loki)
361
+ return NOTHING_TO_REPORT
362
+
363
+ s = summarize(iterations, stages, corrupt)
364
+ # Records present but NONE measured is exit 0, deliberately: the question
365
+ # "what did this cost" was answered, and the answer is UNKNOWN. Collapsing
366
+ # it into NOTHING_TO_REPORT would hide the exact case this file exists for.
367
+ if args.as_json:
368
+ print(json.dumps(s, indent=2, sort_keys=True))
369
+ else:
370
+ print(render(s))
371
+ return OK
372
+
373
+
374
+ if __name__ == "__main__":
375
+ sys.exit(main())
@@ -43,6 +43,24 @@ _ROOT = os.path.dirname(_HERE)
43
43
  NO_DESC = "(no description)"
44
44
 
45
45
 
46
+ class _Parser(argparse.ArgumentParser):
47
+ """Usage errors exit 64, not argparse's default 2.
48
+
49
+ In this repo's convention 2 means "could NOT be checked" -- a real
50
+ answer about the subject. A mistyped flag is not that: it is an error
51
+ about the INVOCATION, and nothing about the subject was examined. The
52
+ two call for opposite responses, since retrying cannot fix a typo.
53
+
54
+ argparse exits 2 for every usage error unless this is overridden, so
55
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
56
+ """
57
+
58
+ def error(self, message):
59
+ self.print_usage(sys.stderr)
60
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
61
+ raise SystemExit(64)
62
+
63
+
46
64
  def _first_sentence(text):
47
65
  """First line of a docstring or header block, trimmed.
48
66
 
@@ -162,7 +180,7 @@ def render(rows):
162
180
 
163
181
 
164
182
  def main(argv=None):
165
- ap = argparse.ArgumentParser(description="List loki's bundled tools.")
183
+ ap = _Parser(description="List loki's bundled tools.")
166
184
  ap.add_argument("--json", action="store_true")
167
185
  ap.add_argument("--tools-dir", default=None)
168
186
  args = ap.parse_args(argv)
@@ -0,0 +1,277 @@
1
+ #!/usr/bin/env python3
2
+ """What does verification COST, and how often does it change the answer?
3
+
4
+ WHY THIS EXISTS. tools/gate-log.py made verdicts durable and tools/gate-trend.py
5
+ says whether the gate is improving or going blind. Both answer questions about
6
+ the gate's OUTPUT. Neither answers what the gate costs to run, and that is now
7
+ the load-bearing question: verification here is adaptive scaffolding whose
8
+ mandated trajectory is toward near-zero overhead, applied selectively where
9
+ calibrated confidence is low and bypassed where it is high.
10
+
11
+ Neither half of that mandate can be validated without a baseline. "Make it
12
+ cheaper" needs a current cost. "Bypass it where confidence is high" needs to
13
+ know how often running it changed an outcome at all -- because a gate that never
14
+ changes an outcome is pure tax, and a gate that frequently does is buying
15
+ something real. This file computes both from the log that already exists.
16
+
17
+ WHAT THIS IS NOT. It is not a gate. It never blocks, never returns a
18
+ merge-blocking exit code, and adds no new artifact format. It reads a log
19
+ written by other tools and reports two numbers. Verification machinery is
20
+ exactly what this must not become; it exists to make the existing machinery
21
+ measurable so it can be shrunk or skipped.
22
+
23
+ THE HONESTY RULE THIS INHERITS. A missing duration is not a duration of zero.
24
+ Entries that carry no timing are counted and reported as UNTIMED rather than
25
+ folded into the average as zeros, because averaging an absent measurement toward
26
+ zero manufactures exactly the "verification is already free" conclusion this
27
+ tool exists to test. An empty log reports NO DATA, never "0.0s, looks cheap".
28
+ """
29
+
30
+ import argparse
31
+ import json
32
+ import os
33
+ import sys
34
+
35
+ # The repo-wide exit convention, enforced by tests/test_tool_exit_contract.py.
36
+ # COULD_NOT_CHECK was 4 here, which is not in the convention at all: a CI
37
+ # caller branching on 2 would have read "could not check" as an unrecognised
38
+ # code and fallen through to its default arm.
39
+ OK = 0
40
+ COULD_NOT_CHECK = 2
41
+ NO_DATA = 3
42
+ USAGE_ERROR = 64
43
+ INPUT_MISSING = 66
44
+
45
+ # Keys a duration may plausibly appear under. The verify evidence doc carries
46
+ # run_started_at/run_completed_at; a future writer may record an explicit
47
+ # duration. Accept either rather than mandating one shape.
48
+ DURATION_KEYS = ("duration_s", "duration_seconds", "elapsed_s")
49
+ START_KEYS = ("run_started_at", "started_at")
50
+ END_KEYS = ("run_completed_at", "completed_at", "ended_at")
51
+
52
+ # A verdict that blocks. Anything here means the gate CHANGED the outcome:
53
+ # without it the change would have proceeded.
54
+ BLOCKING = {"BLOCKED", "FAIL", "FAILED", "BLOCK"}
55
+
56
+
57
+ def _parse_iso(value):
58
+ """Return epoch seconds, or None. None means unknown, never zero."""
59
+ if not isinstance(value, str) or not value:
60
+ return None
61
+ import datetime
62
+ text = value.strip().replace("Z", "+00:00")
63
+ try:
64
+ return datetime.datetime.fromisoformat(text).timestamp()
65
+ except (ValueError, TypeError):
66
+ return None
67
+
68
+
69
+ def _first(mapping, keys):
70
+ for key in keys:
71
+ if key in mapping and mapping[key] is not None:
72
+ return mapping[key]
73
+ return None
74
+
75
+
76
+ def duration_of(entry):
77
+ """Seconds this verification took, or None if the entry never recorded it.
78
+
79
+ Checked in two ways because two writers exist: an explicit duration field,
80
+ or a start/end pair as the verify evidence document carries. A negative or
81
+ non-numeric result is discarded as unknown rather than trusted.
82
+ """
83
+ verdict = entry.get("verdict")
84
+ sources = [entry]
85
+ if isinstance(verdict, dict):
86
+ sources.append(verdict)
87
+ produced = verdict.get("produced_by")
88
+ if isinstance(produced, dict):
89
+ sources.append(produced)
90
+
91
+ for src in sources:
92
+ if not isinstance(src, dict):
93
+ continue
94
+ explicit = _first(src, DURATION_KEYS)
95
+ if isinstance(explicit, (int, float)) and explicit >= 0:
96
+ return float(explicit)
97
+
98
+ for src in sources:
99
+ if not isinstance(src, dict):
100
+ continue
101
+ start = _parse_iso(_first(src, START_KEYS))
102
+ end = _parse_iso(_first(src, END_KEYS))
103
+ if start is not None and end is not None and end >= start:
104
+ return end - start
105
+ return None
106
+
107
+
108
+ def verdict_string(entry):
109
+ """The verdict label, wherever this writer happened to put it."""
110
+ verdict = entry.get("verdict")
111
+ if isinstance(verdict, str):
112
+ return verdict.upper()
113
+ if isinstance(verdict, dict):
114
+ for key in ("verdict", "status", "result"):
115
+ value = verdict.get(key)
116
+ if isinstance(value, str):
117
+ return value.upper()
118
+ category = entry.get("category")
119
+ return category.upper() if isinstance(category, str) else ""
120
+
121
+
122
+ def changed_outcome(entry):
123
+ """Did running the gate change what would otherwise have happened?
124
+
125
+ Only a blocking verdict changes an outcome. A pass means the change would
126
+ have shipped either way, so the time spent verifying bought information but
127
+ not a different result. This is the hit rate that justifies the tax.
128
+ """
129
+ return verdict_string(entry) in BLOCKING
130
+
131
+
132
+ def read_log(path):
133
+ """Every line. Corrupt lines counted, never silently dropped."""
134
+ entries, corrupt = [], 0
135
+ with open(path, "r", encoding="utf-8") as fh:
136
+ for line in fh:
137
+ line = line.strip()
138
+ if not line:
139
+ continue
140
+ try:
141
+ obj = json.loads(line)
142
+ except ValueError:
143
+ corrupt += 1
144
+ continue
145
+ if isinstance(obj, dict):
146
+ entries.append(obj)
147
+ else:
148
+ corrupt += 1
149
+ return entries, corrupt
150
+
151
+
152
+ def summarize(entries, corrupt):
153
+ """Two numbers and the honesty around them.
154
+
155
+ total_s and mean_s are computed over TIMED entries only, and untimed is
156
+ reported alongside so a reader can see how much of the log the average
157
+ actually covers. An average over 2 of 200 runs is not a baseline.
158
+ """
159
+ timed = [d for d in (duration_of(e) for e in entries) if d is not None]
160
+ changed = sum(1 for e in entries if changed_outcome(e))
161
+ total = len(entries)
162
+ return {
163
+ "runs": total,
164
+ "timed": len(timed),
165
+ "untimed": total - len(timed),
166
+ "corrupt": corrupt,
167
+ "total_s": round(sum(timed), 3) if timed else None,
168
+ "mean_s": round(sum(timed) / len(timed), 3) if timed else None,
169
+ "max_s": round(max(timed), 3) if timed else None,
170
+ "changed_outcome": changed,
171
+ "hit_rate": round(changed / total, 4) if total else None,
172
+ }
173
+
174
+
175
+ def render(summary):
176
+ lines = ["VERIFICATION TAX", ""]
177
+ runs = summary["runs"]
178
+ lines.append(" runs logged %d" % runs)
179
+
180
+ if summary["timed"]:
181
+ lines.append(" timed %d of %d" % (summary["timed"], runs))
182
+ lines.append(" total time %.1fs" % summary["total_s"])
183
+ lines.append(" mean per run %.2fs" % summary["mean_s"])
184
+ lines.append(" slowest run %.2fs" % summary["max_s"])
185
+ else:
186
+ lines.append(" timed 0 of %d" % runs)
187
+ lines.append(" total time UNKNOWN (no entry carried timing)")
188
+ lines.append(" mean per run UNKNOWN")
189
+
190
+ if summary["untimed"]:
191
+ lines.append(" UNTIMED %d run(s) carry no duration; excluded"
192
+ % summary["untimed"])
193
+ lines.append(" from the averages above, NOT counted as 0s")
194
+
195
+ lines.append("")
196
+ if summary["hit_rate"] is None:
197
+ lines.append(" outcome changed UNKNOWN")
198
+ else:
199
+ lines.append(" outcome changed %d of %d runs (%.1f%%)" % (
200
+ summary["changed_outcome"], runs, summary["hit_rate"] * 100))
201
+ lines.append(" a pass would have shipped anyway; only a")
202
+ lines.append(" block changed what happened")
203
+
204
+ if summary["corrupt"]:
205
+ lines.append("")
206
+ lines.append(" CORRUPT %d unreadable line(s), counted not dropped"
207
+ % summary["corrupt"])
208
+
209
+ lines.append("")
210
+ if summary["timed"] and summary["hit_rate"] == 0:
211
+ lines.append(" READ: every logged run passed. On this sample the tax bought")
212
+ lines.append(" information but changed no outcome -- the case for applying it")
213
+ lines.append(" selectively rather than universally.")
214
+ elif not summary["timed"]:
215
+ lines.append(" READ: cost is UNMEASURED. No claim that verification is cheap")
216
+ lines.append(" or expensive can be made from this log until writers record")
217
+ lines.append(" timing. Absence of a number is not a small number.")
218
+ return "\n".join(lines)
219
+
220
+
221
+ class _Parser(argparse.ArgumentParser):
222
+ """Usage errors exit 64, not argparse's default 2.
223
+
224
+ In this repo's convention 2 means "could NOT be checked" -- a real
225
+ answer about the subject. A mistyped flag is not that; it is an error
226
+ about the INVOCATION, and reporting it as 2 tells a CI caller that the
227
+ axis was evaluated and found unevaluable, when in fact nothing was
228
+ examined at all.
229
+
230
+ argparse defaults to 2 for every usage error, so a tool that does not
231
+ override this silently emits the wrong code. This one did: `--bogus`
232
+ exited 2 and a missing positional exited 2.
233
+ """
234
+
235
+ def error(self, message):
236
+ self.print_usage(sys.stderr)
237
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
238
+ raise SystemExit(USAGE_ERROR)
239
+
240
+
241
+ def main(argv=None):
242
+ parser = _Parser(
243
+ description="Measure what verification costs and how often it "
244
+ "changed an outcome. Reports only; never blocks.")
245
+ parser.add_argument("log", help="JSONL log (e.g. from tools/gate-log.py)")
246
+ parser.add_argument("--json", action="store_true",
247
+ help="emit the summary as JSON")
248
+ args = parser.parse_args(argv)
249
+
250
+ if not os.path.exists(args.log):
251
+ # 66, not COULD_NOT_CHECK. "The path you named does not exist" is a
252
+ # fact about the INPUT; "I could not evaluate" is a fact about the
253
+ # subject. A caller retrying on 2 would retry forever against a
254
+ # typo, while 66 says the argument itself is wrong.
255
+ sys.stderr.write("verification-tax: no such log: %s\n" % args.log)
256
+ return INPUT_MISSING
257
+ try:
258
+ entries, corrupt = read_log(args.log)
259
+ except OSError as exc:
260
+ sys.stderr.write("verification-tax: could not read %s: %s\n"
261
+ % (args.log, exc))
262
+ return COULD_NOT_CHECK
263
+
264
+ if not entries and not corrupt:
265
+ sys.stderr.write("verification-tax: log is empty; NO DATA\n")
266
+ return NO_DATA
267
+
268
+ summary = summarize(entries, corrupt)
269
+ if args.json:
270
+ sys.stdout.write(json.dumps(summary, sort_keys=True, indent=2) + "\n")
271
+ else:
272
+ sys.stdout.write(render(summary) + "\n")
273
+ return OK
274
+
275
+
276
+ if __name__ == "__main__":
277
+ sys.exit(main())