loki-mode 9.8.0 → 9.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +19 -14
  2. package/SKILL.md +3 -2
  3. package/VERSION +1 -1
  4. package/autonomy/loki +122 -1
  5. package/autonomy/run.sh +49 -2
  6. package/dashboard/__init__.py +1 -1
  7. package/dashboard/api_evidence.py +411 -0
  8. package/dashboard/api_operator.py +283 -0
  9. package/dashboard/api_phases.py +262 -0
  10. package/dashboard/api_releases.py +242 -0
  11. package/dashboard/api_runs.py +477 -0
  12. package/dashboard/api_tests.py +444 -0
  13. package/dashboard/api_v2.py +47 -1
  14. package/dashboard/server.py +54 -0
  15. package/dashboard/static/index.html +246 -135
  16. package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
  17. package/docs/CAPABILITY-BACKLOG.md +53 -0
  18. package/docs/COMPARISON.md +2 -2
  19. package/docs/COMPETITIVE-ANALYSIS.md +1 -1
  20. package/docs/COMPETITIVE-SCORECARD.md +422 -0
  21. package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
  22. package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
  23. package/docs/DEMOS.md +21 -23
  24. package/docs/HANDOFF-2026-08-03.md +439 -0
  25. package/docs/INSTALLATION.md +17 -10
  26. package/docs/OUTCOME-FRONTIER.md +536 -0
  27. package/docs/PROMPT-ABLATION-RESULT.md +97 -0
  28. package/docs/TOOLS.md +800 -0
  29. package/docs/alternative-installations.md +2 -3
  30. package/docs/audit-logging.md +44 -35
  31. package/docs/authentication.md +13 -2
  32. package/docs/authorization.md +87 -81
  33. package/docs/git-workflow.md +6 -3
  34. package/docs/metrics.md +15 -16
  35. package/docs/network-security.md +16 -13
  36. package/docs/openclaw-integration.md +36 -556
  37. package/docs/show-hn-post.md +2 -2
  38. package/docs/siem-integration.md +39 -36
  39. package/loki-ts/dist/loki.js +18 -18
  40. package/mcp/__init__.py +1 -1
  41. package/package.json +2 -2
  42. package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
  43. package/references/confidence-routing.md +18 -1
  44. package/references/invariant-checks.md +13 -8
  45. package/references/magic-rarv-integration.md +0 -1
  46. package/references/multi-provider.md +27 -5
  47. package/skills/healing.md +4 -2
  48. package/tools/audit-docs.py +488 -0
  49. package/tools/baseline-pin.py +19 -1
  50. package/tools/calibration-audit.py +523 -0
  51. package/tools/ci-gate.py +19 -1
  52. package/tools/cost-forecast.py +344 -0
  53. package/tools/cost-guard.py +19 -1
  54. package/tools/cost-history.py +19 -1
  55. package/tools/cost-per-outcome.py +394 -0
  56. package/tools/estimate-run.py +19 -1
  57. package/tools/evidence-freshness.py +307 -0
  58. package/tools/gate-init.py +19 -1
  59. package/tools/gate-report.py +19 -1
  60. package/tools/gate-simulate.py +570 -0
  61. package/tools/gate-trend.py +354 -0
  62. package/tools/model-advisor.py +52 -1
  63. package/tools/policy-load.py +19 -1
  64. package/tools/prompt-cost.py +363 -0
  65. package/tools/prompt-diff.py +448 -0
  66. package/tools/prompt-lint.py +448 -0
  67. package/tools/receipt-bundle.py +72 -2
  68. package/tools/receipt-diff.py +19 -1
  69. package/tools/receipt-find.py +19 -1
  70. package/tools/receipt-stats.py +380 -0
  71. package/tools/receipt-timeline.py +478 -0
  72. package/tools/receipt-verify-batch.py +291 -0
  73. package/tools/run-replay.py +19 -1
  74. package/tools/signing-status.py +19 -1
  75. package/tools/token-guard.py +19 -1
  76. package/tools/token-tax.py +375 -0
  77. package/tools/tool-index.py +19 -1
  78. package/tools/verification-tax.py +277 -0
  79. package/tools/verify-chain.py +361 -0
@@ -65,6 +65,24 @@ _LIB = os.path.join(os.path.dirname(_HERE), "autonomy", "lib")
65
65
  sys.path.insert(0, _LIB)
66
66
 
67
67
 
68
+ class _Parser(argparse.ArgumentParser):
69
+ """Usage errors exit 64, not argparse's default 2.
70
+
71
+ In this repo's convention 2 means "could NOT be checked" -- a real
72
+ answer about the subject. A mistyped flag is not that: it is an error
73
+ about the INVOCATION, and nothing about the subject was examined. The
74
+ two call for opposite responses, since retrying cannot fix a typo.
75
+
76
+ argparse exits 2 for every usage error unless this is overridden, so
77
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
78
+ """
79
+
80
+ def error(self, message):
81
+ self.print_usage(sys.stderr)
82
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
83
+ raise SystemExit(64)
84
+
85
+
68
86
  def _load(name, path):
69
87
  spec = importlib.util.spec_from_file_location(name, path)
70
88
  mod = importlib.util.module_from_spec(spec)
@@ -176,7 +194,7 @@ def check_pin(rec):
176
194
 
177
195
 
178
196
  def main(argv):
179
- ap = argparse.ArgumentParser(
197
+ ap = _Parser(
180
198
  description="Pin a run as the cost baseline, then resolve it later.")
181
199
  sub = ap.add_subparsers(dest="cmd")
182
200
 
@@ -0,0 +1,523 @@
1
+ #!/usr/bin/env python3
2
+ """Score council voters against the council's own outcome, over history already on disk.
3
+
4
+ WHY THIS EXISTS. Every iteration writes a council transcript recording who voted
5
+ which way and what the council decided. Nothing has ever gone back and asked the
6
+ obvious follow-up: when a voter says APPROVE, how often does that hold? A voter
7
+ who approves everything is indistinguishable from a careful one on any single
8
+ iteration, and the transcripts have been accumulating the evidence to tell them
9
+ apart the whole time.
10
+
11
+ This reads those files and nothing else. It starts no build, calls no model,
12
+ spends nothing, and changes no default behaviour. It is a REPORTER, not a gate:
13
+ it exits 0 on terrible calibration, because a reporter that fails CI would just
14
+ be turned off.
15
+
16
+ THE LABELLING CONVENTION, which is the one thing a reader should argue with.
17
+
18
+ prediction = 1.0 when a voter's verdict is APPROVE, 0.0 when REJECT
19
+ label = 1 when the transcript's `outcome` is APPROVED, else 0
20
+
21
+ That convention has a circularity in it, and pretending otherwise would be the
22
+ dishonest move. The council outcome is MECHANICALLY DERIVED from the votes
23
+ (`approve_count >= threshold`), so a voter's own prediction partially CAUSES the
24
+ label it is then scored against. What this measures is therefore AGREEMENT WITH
25
+ THE COUNCIL MAJORITY, not accuracy against ground truth. A voter scoring
26
+ perfectly here may simply be voting with the crowd. No artifact on disk records
27
+ whether the council was actually right, so ground-truth calibration is not
28
+ computable from this substrate at all -- and it is reported that way rather than
29
+ approximated. The convention is printed with every run so a reader can disagree
30
+ with it without reading this source.
31
+
32
+ PREDICTIONS ARE BINARY, which shapes everything downstream. Voters emit a
33
+ verdict, not a probability, so every prediction is exactly 0.0 or 1.0. The
34
+ reliability curve therefore has at most two occupied bins no matter how many
35
+ bins are requested; the rest are empty by construction, not by accident. They
36
+ print as UNKNOWN, never as 0.0, because "no voter ever predicted 0.35" and
37
+ "voters predicted 0.35 and were never right" are opposite findings.
38
+
39
+ ECE DEPENDS ON ITS BIN COUNT. The number changes when the bin count changes,
40
+ so the bin count is printed next to every ECE and stated as a knob. An ECE
41
+ quoted without its bin count is not a measurement.
42
+
43
+ SUPPORT IS PRINTED EVERYWHERE, and a headline is REFUSED below a floor
44
+ (default 30 predictions). A Brier score over four samples is noise wearing a
45
+ decimal point, and the most expensive thing this tool could do is hand someone
46
+ a confident number derived from a handful of rows.
47
+
48
+ PREDICTIONS ARE CLUSTERED, and the count is printed alongside the transcript
49
+ count for that reason. Every voter in one transcript is scored against the SAME
50
+ label, so N votes drawn from M transcripts carry nowhere near N independent
51
+ observations -- the effective sample size is nearer M. Printing the pooled vote
52
+ count alone would inflate apparent support by roughly the council size, which
53
+ in a tool built to refuse overstatement would be the same defect it exists to
54
+ prevent. The floor is applied to predictions because that is the stated knob,
55
+ and the transcript count is printed next to it so a reader can apply their own.
56
+
57
+ WHAT THE ARTIFACTS CANNOT SUPPORT. The brief asked for breakdowns by gate,
58
+ model and task. None of the three is recorded, and each is reported NOT
59
+ AVAILABLE with its reason rather than approximated by a nearby field:
60
+
61
+ by gate -- `outcome` can be BLOCKED_BY_GATE, but that is a gate OUTCOME,
62
+ not a gate IDENTITY. No gate id or name appears anywhere in the
63
+ transcript, so votes cannot be grouped by which gate blocked.
64
+ by model -- `voters[].name` is the ROLE (it is populated from `v.role` in
65
+ councilWriteTranscript). Which model backed a role is not
66
+ written. Role is not a model and is reported as its own axis.
67
+ by task -- `prd_path` and `task_or_prd` (the first 200 chars of the PRD)
68
+ identify the RUN, not a task, and are effectively constant
69
+ across every transcript in one .loki dir, so they separate
70
+ nothing.
71
+
72
+ Nearby numbers exist that would each make a plausible-looking proxy, and all
73
+ are deliberately refused. `last_confidence` is a real number, but it lives in
74
+ council STATE (loki-ts/src/runner/council.ts:129) as a single run-level scalar,
75
+ not per voter and not in the transcript; joining it onto voters would invent a
76
+ per-voter confidence that was never recorded. `.loki/state/uncertainty.json`
77
+ holds BOOLEAN uncertainty proxies, not a confidence, and cannot be read as one.
78
+
79
+ MISSINGNESS IS A RESULT, not a footnote. Skipped transcripts, absent fields and
80
+ CANNOT_VALIDATE votes each get their own counted line. CANNOT_VALIDATE is
81
+ EXCLUDED from the calibration sample -- a voter declining to assert is not a
82
+ wrong probabilistic assertion -- which means the sample size here will NOT match
83
+ the transcript's own `reject_count`, since that field lumps CANNOT_VALIDATE in
84
+ with REJECT. That discrepancy is stated in the output so it does not read as
85
+ dropped rows.
86
+
87
+ Exit codes follow the tools/ convention:
88
+ 0 reported
89
+ 2 could NOT check (transcript files found, none parseable)
90
+ 3 nothing to report (no transcript files)
91
+ 64 usage error
92
+ 66 input path missing
93
+
94
+ Usage:
95
+ tools/calibration-audit.py [workspace] [--bins N] [--min-support N] [--json]
96
+ """
97
+
98
+ import argparse
99
+ import json
100
+ import os
101
+ import sys
102
+
103
+ sys.dont_write_bytecode = True
104
+
105
+ UNKNOWN = "UNKNOWN"
106
+
107
+ # Below this many usable predictions, no headline calibration number is
108
+ # presented. Stated in the output, and overridable, because the right floor is
109
+ # a judgement call and a hidden judgement call is one nobody can challenge.
110
+ DEFAULT_MIN_SUPPORT = 30
111
+ DEFAULT_BINS = 10
112
+
113
+ # Dimensions the brief asked for that the artifacts do not record. Printed
114
+ # verbatim so the refusal is visible to a reader who never opens this file.
115
+ NOT_AVAILABLE = [
116
+ ("gate", "outcome can be BLOCKED_BY_GATE, but that is a gate OUTCOME, "
117
+ "not a gate IDENTITY; no gate id or name is written to the "
118
+ "transcript, so votes cannot be grouped by gate"),
119
+ ("model", "voters[].name is the ROLE (populated from v.role in "
120
+ "councilWriteTranscript); which model backed a role is never "
121
+ "recorded, and role is reported as its own axis instead"),
122
+ ("task", "prd_path and task_or_prd (first 200 chars of the PRD) identify "
123
+ "the RUN, not a task, and are effectively constant across every "
124
+ "transcript in one .loki dir, so they separate nothing"),
125
+ ]
126
+
127
+
128
+ class _Parser(argparse.ArgumentParser):
129
+ """argparse exits 2 on a usage error, and 2 already means something else.
130
+
131
+ In this convention 2 is "could NOT check" -- a real answer about the
132
+ history. A typo in a flag is not that; it is 64. Left alone, `--bnis`
133
+ would report as a failed scan and a caller could not tell the two apart.
134
+ """
135
+
136
+ def error(self, message):
137
+ self.print_usage(sys.stderr)
138
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
139
+ raise SystemExit(64)
140
+
141
+
142
+ def transcripts_dir(workspace):
143
+ return os.path.join(workspace, ".loki", "council", "transcripts")
144
+
145
+
146
+ def load_votes(directory):
147
+ """Read every transcript, returning (votes, missingness).
148
+
149
+ A vote is a flat record so the breakdowns are plain groupings. Anything
150
+ that could not be turned into a vote is counted rather than dropped
151
+ silently: a scan that quietly discards half its input reports on a
152
+ population nobody chose.
153
+ """
154
+ miss = {
155
+ "files_found": 0,
156
+ "files_unparseable": 0,
157
+ "files_missing_voters": 0,
158
+ "files_missing_outcome": 0,
159
+ "transcripts_used": 0,
160
+ "voters_seen": 0,
161
+ "votes_cannot_validate": 0,
162
+ "votes_unknown_verdict": 0,
163
+ "votes_missing_name": 0,
164
+ "votes_missing_role_index": 0,
165
+ "outcome_blocked_by_gate": 0,
166
+ }
167
+ votes = []
168
+
169
+ try:
170
+ names = sorted(n for n in os.listdir(directory) if n.endswith(".json"))
171
+ except OSError:
172
+ return votes, miss
173
+
174
+ for name in names:
175
+ miss["files_found"] += 1
176
+ path = os.path.join(directory, name)
177
+ try:
178
+ with open(path) as handle:
179
+ data = json.load(handle)
180
+ if not isinstance(data, dict):
181
+ raise ValueError("not an object")
182
+ except (OSError, ValueError):
183
+ miss["files_unparseable"] += 1
184
+ continue
185
+
186
+ outcome = data.get("outcome")
187
+ if not isinstance(outcome, str):
188
+ miss["files_missing_outcome"] += 1
189
+ continue
190
+ voters = data.get("voters")
191
+ if not isinstance(voters, list) or not voters:
192
+ miss["files_missing_voters"] += 1
193
+ continue
194
+
195
+ # The label. BLOCKED_BY_GATE is not APPROVED, so under the stated
196
+ # convention it labels 0 -- counted separately so a reader who reads
197
+ # it differently can re-derive without re-running.
198
+ if outcome == "BLOCKED_BY_GATE":
199
+ miss["outcome_blocked_by_gate"] += 1
200
+ label = 1 if outcome == "APPROVED" else 0
201
+
202
+ miss["transcripts_used"] += 1
203
+ for voter in voters:
204
+ if not isinstance(voter, dict):
205
+ miss["votes_unknown_verdict"] += 1
206
+ continue
207
+ miss["voters_seen"] += 1
208
+ verdict = voter.get("verdict")
209
+
210
+ # CANNOT_VALIDATE is a refusal to assert, not a wrong assertion.
211
+ # Excluding it is a choice, so it is counted where it is visible.
212
+ if verdict == "CANNOT_VALIDATE":
213
+ miss["votes_cannot_validate"] += 1
214
+ continue
215
+ if verdict == "APPROVE":
216
+ prediction = 1.0
217
+ elif verdict == "REJECT":
218
+ prediction = 0.0
219
+ else:
220
+ miss["votes_unknown_verdict"] += 1
221
+ continue
222
+
223
+ voter_name = voter.get("name")
224
+ if not isinstance(voter_name, str) or not voter_name:
225
+ miss["votes_missing_name"] += 1
226
+ voter_name = UNKNOWN
227
+ role_index = voter.get("role_index")
228
+ if not isinstance(role_index, int):
229
+ miss["votes_missing_role_index"] += 1
230
+ role_index = UNKNOWN
231
+
232
+ contrarian = voter.get("is_contrarian")
233
+ votes.append({
234
+ "prediction": prediction,
235
+ "label": label,
236
+ "name": voter_name,
237
+ "role_index": role_index,
238
+ "is_contrarian": (contrarian if isinstance(contrarian, bool)
239
+ else UNKNOWN),
240
+ })
241
+
242
+ return votes, miss
243
+
244
+
245
+ def brier(votes):
246
+ """Mean squared error of prediction against label. UNKNOWN over nothing."""
247
+ if not votes:
248
+ return UNKNOWN
249
+ total = sum((v["prediction"] - v["label"]) ** 2 for v in votes)
250
+ return total / len(votes)
251
+
252
+
253
+ def reliability(votes, bins):
254
+ """Per-bin predicted rate vs observed rate, WITH support.
255
+
256
+ EVERY bin is returned, including the empty ones, and an empty bin carries
257
+ UNKNOWN rather than 0.0. Predictions here are binary, so most bins are
258
+ empty by construction -- reporting those as a 0.0 observed rate would
259
+ manufacture a perfectly-wrong-looking region of the curve out of the
260
+ absence of data.
261
+ """
262
+ table = []
263
+ for index in range(bins):
264
+ low = index / bins
265
+ high = (index + 1) / bins
266
+ # Half-open bins, with the last one closed so prediction 1.0 lands.
267
+ if index == bins - 1:
268
+ members = [v for v in votes if low <= v["prediction"] <= high]
269
+ else:
270
+ members = [v for v in votes if low <= v["prediction"] < high]
271
+ if members:
272
+ predicted = sum(v["prediction"] for v in members) / len(members)
273
+ observed = sum(v["label"] for v in members) / len(members)
274
+ else:
275
+ predicted = UNKNOWN
276
+ observed = UNKNOWN
277
+ table.append({
278
+ "bin": index,
279
+ "range": [low, high],
280
+ "support": len(members),
281
+ "predicted_rate": predicted,
282
+ "observed_rate": observed,
283
+ })
284
+ return table
285
+
286
+
287
+ def ece(votes, bins):
288
+ """Support-weighted mean gap between predicted and observed rate.
289
+
290
+ Empty bins contribute nothing. That is the DEFINITION of the
291
+ support-weighted sum, not an imputation: a bin with no support has no
292
+ weight, so it cannot pull the number in either direction. It is a
293
+ different thing from treating its observed rate as 0.0, which would.
294
+ """
295
+ if not votes:
296
+ return UNKNOWN
297
+ table = reliability(votes, bins)
298
+ total = 0.0
299
+ for row in table:
300
+ if row["support"] == 0:
301
+ continue
302
+ gap = abs(row["predicted_rate"] - row["observed_rate"])
303
+ total += (row["support"] / len(votes)) * gap
304
+ return total
305
+
306
+
307
+ def group(votes, key, bins):
308
+ """Break the sample down by one recorded field, keeping support per group."""
309
+ buckets = {}
310
+ for vote in votes:
311
+ buckets.setdefault(str(vote[key]), []).append(vote)
312
+ return [
313
+ {
314
+ "value": value,
315
+ "support": len(members),
316
+ "brier": brier(members),
317
+ "ece": ece(members, bins),
318
+ "approve_rate": sum(v["prediction"] for v in members) / len(members),
319
+ "observed_rate": sum(v["label"] for v in members) / len(members),
320
+ }
321
+ for value, members in sorted(buckets.items())
322
+ ]
323
+
324
+
325
+ def audit(workspace, bins, min_support):
326
+ votes, miss = load_votes(transcripts_dir(workspace))
327
+ enough = len(votes) >= min_support
328
+ return {
329
+ "bins": bins,
330
+ "min_support": min_support,
331
+ "total_predictions": len(votes),
332
+ "sufficient_support": enough,
333
+ # The headline is WITHHELD below the floor rather than printed small.
334
+ # A number a reader can see is a number a reader will quote.
335
+ "brier": brier(votes) if enough else UNKNOWN,
336
+ "ece": ece(votes, bins) if enough else UNKNOWN,
337
+ "reliability": reliability(votes, bins),
338
+ "by_name": group(votes, "name", bins),
339
+ "by_role_index": group(votes, "role_index", bins),
340
+ "by_is_contrarian": group(votes, "is_contrarian", bins),
341
+ "missingness": miss,
342
+ "not_available": [{"dimension": d, "reason": r}
343
+ for d, r in NOT_AVAILABLE],
344
+ }
345
+
346
+
347
+ def _num(value):
348
+ return UNKNOWN if value == UNKNOWN else "%.4f" % value
349
+
350
+
351
+ def _exit_code(report):
352
+ miss = report["missingness"]
353
+ if miss["files_found"] == 0:
354
+ return 3 # nothing to report
355
+ if miss["transcripts_used"] == 0:
356
+ # Files were there and none survived parsing. That is a FAILED scan,
357
+ # not an empty one, and collapsing it into 3 would let a directory of
358
+ # corrupt transcripts read as "nothing to report".
359
+ return 2
360
+ return 0
361
+
362
+
363
+ def _render(report, code):
364
+ bins = report["bins"]
365
+ total = report["total_predictions"]
366
+ miss = report["missingness"]
367
+ lines = ["COUNCIL CALIBRATION AUDIT"]
368
+ lines.append(" reads .loki/council/transcripts only; starts nothing, "
369
+ "spends nothing")
370
+ lines.append("")
371
+ lines.append("LABELLING CONVENTION (disagree with this before the numbers)")
372
+ lines.append(" prediction = 1.0 for APPROVE, 0.0 for REJECT")
373
+ lines.append(" label = 1 when outcome == APPROVED, else 0")
374
+ lines.append(" CANNOT_VALIDATE is EXCLUDED: declining to assert is not a")
375
+ lines.append(" wrong assertion. So this sample size will NOT match the")
376
+ lines.append(" transcript reject_count, which lumps it in with REJECT.")
377
+ lines.append(" CIRCULARITY: outcome is derived from the votes")
378
+ lines.append(" (approve_count >= threshold), so a voter's prediction")
379
+ lines.append(" partly CAUSES its own label. This measures AGREEMENT")
380
+ lines.append(" WITH THE MAJORITY, not accuracy against ground truth.")
381
+ lines.append(" No artifact records whether the council was right.")
382
+ lines.append(" Predictions are BINARY (a verdict, not a probability), so")
383
+ lines.append(" at most two bins can ever be occupied.")
384
+
385
+ if code == 3:
386
+ lines.append("")
387
+ lines.append("NOTHING TO REPORT -- no transcript files found.")
388
+ lines.append(" Scanning nothing is not evidence of good calibration.")
389
+ return "\n".join(lines)
390
+ if code == 2:
391
+ lines.append("")
392
+ lines.append("CANNOT CHECK -- %d transcript file(s) found, none "
393
+ "parseable." % miss["files_found"])
394
+ return "\n".join(lines)
395
+
396
+ lines.append("")
397
+ lines.append("SUPPORT")
398
+ lines.append(" usable predictions: %d from %d transcript(s)"
399
+ % (total, miss["transcripts_used"]))
400
+ lines.append(" Votes within one transcript share a label, so predictions")
401
+ lines.append(" are CLUSTERED, not independent: the effective sample")
402
+ lines.append(" size is nearer the transcript count than the vote count.")
403
+ lines.append(" The floor below is applied to PREDICTIONS (the stated "
404
+ "knob): %d" % report["min_support"])
405
+ if not report["sufficient_support"]:
406
+ lines.append("")
407
+ lines.append(" INSUFFICIENT SUPPORT -- headline calibration numbers "
408
+ "are WITHHELD.")
409
+ lines.append(" Usable predictions (%d) is below the stated floor (%d)."
410
+ % (total, report["min_support"]))
411
+ lines.append(" The breakdowns below are printed with their support so "
412
+ "they can be")
413
+ lines.append(" read as counts, not as calibration.")
414
+ else:
415
+ lines.append("")
416
+ lines.append("HEADLINE")
417
+ lines.append(" Brier score: %s (0 is perfect, lower is better)"
418
+ % _num(report["brier"]))
419
+ lines.append(" ECE: %s at %d bins"
420
+ % (_num(report["ece"]), bins))
421
+ lines.append(" ECE depends on its bin count: changing --bins changes "
422
+ "this number.")
423
+
424
+ lines.append("")
425
+ lines.append("RELIABILITY TABLE (%d bins; empty bins are UNKNOWN, not 0)"
426
+ % bins)
427
+ lines.append(" %-14s %8s %10s %10s" % ("bin", "support", "predicted",
428
+ "observed"))
429
+ for row in report["reliability"]:
430
+ lines.append(" %-14s %8d %10s %10s"
431
+ % ("[%.2f,%.2f]" % (row["range"][0], row["range"][1]),
432
+ row["support"], _num(row["predicted_rate"]),
433
+ _num(row["observed_rate"])))
434
+
435
+ for title, key in (("BY VOTER NAME (role)", "by_name"),
436
+ ("BY ROLE INDEX", "by_role_index"),
437
+ ("BY IS_CONTRARIAN", "by_is_contrarian")):
438
+ lines.append("")
439
+ lines.append(title)
440
+ rows = report[key]
441
+ if not rows:
442
+ lines.append(" UNKNOWN -- no usable votes carried this field")
443
+ continue
444
+ lines.append(" %-28s %8s %9s %9s" % ("value", "support", "brier",
445
+ "ece"))
446
+ for row in rows:
447
+ flag = "" if row["support"] >= report["min_support"] else " (low)"
448
+ lines.append(" %-28s %8d %9s %9s%s"
449
+ % (row["value"][:28], row["support"],
450
+ _num(row["brier"]), _num(row["ece"]), flag))
451
+
452
+ lines.append("")
453
+ lines.append("NOT AVAILABLE -- asked for, not recorded, not approximated")
454
+ for item in report["not_available"]:
455
+ lines.append(" by %s: %s" % (item["dimension"], item["reason"]))
456
+
457
+ lines.append("")
458
+ lines.append("MISSINGNESS")
459
+ lines.append(" transcript files found: %d" % miss["files_found"])
460
+ lines.append(" transcripts used: %d" % miss["transcripts_used"])
461
+ lines.append(" files unparseable: %d" % miss["files_unparseable"])
462
+ lines.append(" files with no voters[]: %d"
463
+ % miss["files_missing_voters"])
464
+ lines.append(" files with no outcome: %d"
465
+ % miss["files_missing_outcome"])
466
+ lines.append(" voter entries seen: %d" % miss["voters_seen"])
467
+ lines.append(" CANNOT_VALIDATE (excluded): %d"
468
+ % miss["votes_cannot_validate"])
469
+ lines.append(" unusable verdict: %d"
470
+ % miss["votes_unknown_verdict"])
471
+ lines.append(" votes with no name: %d" % miss["votes_missing_name"])
472
+ lines.append(" votes with no role_index: %d"
473
+ % miss["votes_missing_role_index"])
474
+ lines.append(" outcome BLOCKED_BY_GATE: %d (labelled 0 here)"
475
+ % miss["outcome_blocked_by_gate"])
476
+
477
+ lines.append("")
478
+ lines.append("This is a REPORTER, not a gate. It exits 0 even when "
479
+ "calibration is bad.")
480
+ return "\n".join(lines)
481
+
482
+
483
+ def main(argv=None):
484
+ parser = _Parser(
485
+ description="Audit council voter calibration over transcripts on disk.")
486
+ parser.add_argument("workspace", nargs="?", default=".",
487
+ help="workspace root holding .loki (default: cwd)")
488
+ parser.add_argument("--bins", type=int, default=DEFAULT_BINS,
489
+ help="reliability bin count (default: %d); changing "
490
+ "it changes ECE" % DEFAULT_BINS)
491
+ parser.add_argument("--min-support", type=int, default=DEFAULT_MIN_SUPPORT,
492
+ help="predictions required before a headline "
493
+ "calibration number is presented (default: %d)"
494
+ % DEFAULT_MIN_SUPPORT)
495
+ parser.add_argument("--json", action="store_true",
496
+ help="emit machine-readable output")
497
+ args = parser.parse_args(argv)
498
+
499
+ if not os.path.exists(args.workspace):
500
+ payload = {"status": "input_missing", "exit_code": 66,
501
+ "error": "no such path: " + args.workspace}
502
+ print(json.dumps(payload, indent=2) if args.json
503
+ else "INPUT MISSING -- no such path: " + args.workspace)
504
+ return 66
505
+ if args.bins < 1:
506
+ parser.error("--bins must be at least 1")
507
+ if args.min_support < 0:
508
+ parser.error("--min-support cannot be negative")
509
+
510
+ report = audit(args.workspace, args.bins, args.min_support)
511
+ code = _exit_code(report)
512
+ if args.json:
513
+ report["status"] = {3: "nothing_to_report", 2: "cannot_check"}.get(
514
+ code, "reported")
515
+ report["exit_code"] = code
516
+ print(json.dumps(report, indent=2))
517
+ else:
518
+ print(_render(report, code))
519
+ return code
520
+
521
+
522
+ if __name__ == "__main__":
523
+ sys.exit(main())
package/tools/ci-gate.py CHANGED
@@ -81,6 +81,24 @@ PASS, FAIL, UNEVALUABLE = 0, 1, 2
81
81
  _STATE = {PASS: "PASS", FAIL: "FAIL", UNEVALUABLE: "UNEVALUABLE"}
82
82
 
83
83
 
84
+ class _Parser(argparse.ArgumentParser):
85
+ """Usage errors exit 64, not argparse's default 2.
86
+
87
+ In this repo's convention 2 means "could NOT be checked" -- a real
88
+ answer about the subject. A mistyped flag is not that: it is an error
89
+ about the INVOCATION, and nothing about the subject was examined. The
90
+ two call for opposite responses, since retrying cannot fix a typo.
91
+
92
+ argparse exits 2 for every usage error unless this is overridden, so
93
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
94
+ """
95
+
96
+ def error(self, message):
97
+ self.print_usage(sys.stderr)
98
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
99
+ raise SystemExit(64)
100
+
101
+
84
102
  def _row(policy, code, reason):
85
103
  return {"policy": policy, "state": _STATE[code], "exit_code": code,
86
104
  "reason": reason}
@@ -201,7 +219,7 @@ def render(d):
201
219
 
202
220
 
203
221
  def main(argv=None):
204
- ap = argparse.ArgumentParser(
222
+ ap = _Parser(
205
223
  description="One exit code over every configured merge policy.")
206
224
  ap.add_argument("workspace", nargs="?", default=".",
207
225
  help="workspace root (or its .loki dir); default .")