loki-mode 9.8.0 → 9.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -14
- package/SKILL.md +3 -2
- package/VERSION +1 -1
- package/autonomy/loki +122 -1
- package/autonomy/run.sh +49 -2
- package/dashboard/__init__.py +1 -1
- package/dashboard/api_evidence.py +411 -0
- package/dashboard/api_operator.py +283 -0
- package/dashboard/api_phases.py +262 -0
- package/dashboard/api_releases.py +242 -0
- package/dashboard/api_runs.py +477 -0
- package/dashboard/api_tests.py +444 -0
- package/dashboard/api_v2.py +47 -1
- package/dashboard/server.py +54 -0
- package/dashboard/static/index.html +246 -135
- package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
- package/docs/CAPABILITY-BACKLOG.md +53 -0
- package/docs/COMPARISON.md +2 -2
- package/docs/COMPETITIVE-ANALYSIS.md +1 -1
- package/docs/COMPETITIVE-SCORECARD.md +422 -0
- package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
- package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
- package/docs/DEMOS.md +21 -23
- package/docs/HANDOFF-2026-08-03.md +439 -0
- package/docs/INSTALLATION.md +17 -10
- package/docs/OUTCOME-FRONTIER.md +536 -0
- package/docs/PROMPT-ABLATION-RESULT.md +97 -0
- package/docs/TOOLS.md +800 -0
- package/docs/alternative-installations.md +2 -3
- package/docs/audit-logging.md +44 -35
- package/docs/authentication.md +13 -2
- package/docs/authorization.md +87 -81
- package/docs/git-workflow.md +6 -3
- package/docs/metrics.md +15 -16
- package/docs/network-security.md +16 -13
- package/docs/openclaw-integration.md +36 -556
- package/docs/show-hn-post.md +2 -2
- package/docs/siem-integration.md +39 -36
- package/loki-ts/dist/loki.js +18 -18
- package/mcp/__init__.py +1 -1
- package/package.json +2 -2
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
- package/references/confidence-routing.md +18 -1
- package/references/invariant-checks.md +13 -8
- package/references/magic-rarv-integration.md +0 -1
- package/references/multi-provider.md +27 -5
- package/skills/healing.md +4 -2
- package/tools/audit-docs.py +488 -0
- package/tools/baseline-pin.py +19 -1
- package/tools/calibration-audit.py +523 -0
- package/tools/ci-gate.py +19 -1
- package/tools/cost-forecast.py +344 -0
- package/tools/cost-guard.py +19 -1
- package/tools/cost-history.py +19 -1
- package/tools/cost-per-outcome.py +394 -0
- package/tools/estimate-run.py +19 -1
- package/tools/evidence-freshness.py +307 -0
- package/tools/gate-init.py +19 -1
- package/tools/gate-report.py +19 -1
- package/tools/gate-simulate.py +570 -0
- package/tools/gate-trend.py +354 -0
- package/tools/model-advisor.py +52 -1
- package/tools/policy-load.py +19 -1
- package/tools/prompt-cost.py +363 -0
- package/tools/prompt-diff.py +448 -0
- package/tools/prompt-lint.py +448 -0
- package/tools/receipt-bundle.py +72 -2
- package/tools/receipt-diff.py +19 -1
- package/tools/receipt-find.py +19 -1
- package/tools/receipt-stats.py +380 -0
- package/tools/receipt-timeline.py +478 -0
- package/tools/receipt-verify-batch.py +291 -0
- package/tools/run-replay.py +19 -1
- package/tools/signing-status.py +19 -1
- package/tools/token-guard.py +19 -1
- package/tools/token-tax.py +375 -0
- package/tools/tool-index.py +19 -1
- package/tools/verification-tax.py +277 -0
- package/tools/verify-chain.py +361 -0
package/tools/baseline-pin.py
CHANGED
|
@@ -65,6 +65,24 @@ _LIB = os.path.join(os.path.dirname(_HERE), "autonomy", "lib")
|
|
|
65
65
|
sys.path.insert(0, _LIB)
|
|
66
66
|
|
|
67
67
|
|
|
68
|
+
class _Parser(argparse.ArgumentParser):
|
|
69
|
+
"""Usage errors exit 64, not argparse's default 2.
|
|
70
|
+
|
|
71
|
+
In this repo's convention 2 means "could NOT be checked" -- a real
|
|
72
|
+
answer about the subject. A mistyped flag is not that: it is an error
|
|
73
|
+
about the INVOCATION, and nothing about the subject was examined. The
|
|
74
|
+
two call for opposite responses, since retrying cannot fix a typo.
|
|
75
|
+
|
|
76
|
+
argparse exits 2 for every usage error unless this is overridden, so
|
|
77
|
+
every tool needs it. tests/test_tool_exit_contract.py asserts it.
|
|
78
|
+
"""
|
|
79
|
+
|
|
80
|
+
def error(self, message):
|
|
81
|
+
self.print_usage(sys.stderr)
|
|
82
|
+
sys.stderr.write("%s: error: %s\n" % (self.prog, message))
|
|
83
|
+
raise SystemExit(64)
|
|
84
|
+
|
|
85
|
+
|
|
68
86
|
def _load(name, path):
|
|
69
87
|
spec = importlib.util.spec_from_file_location(name, path)
|
|
70
88
|
mod = importlib.util.module_from_spec(spec)
|
|
@@ -176,7 +194,7 @@ def check_pin(rec):
|
|
|
176
194
|
|
|
177
195
|
|
|
178
196
|
def main(argv):
|
|
179
|
-
ap =
|
|
197
|
+
ap = _Parser(
|
|
180
198
|
description="Pin a run as the cost baseline, then resolve it later.")
|
|
181
199
|
sub = ap.add_subparsers(dest="cmd")
|
|
182
200
|
|
|
@@ -0,0 +1,523 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Score council voters against the council's own outcome, over history already on disk.
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. Every iteration writes a council transcript recording who voted
|
|
5
|
+
which way and what the council decided. Nothing has ever gone back and asked the
|
|
6
|
+
obvious follow-up: when a voter says APPROVE, how often does that hold? A voter
|
|
7
|
+
who approves everything is indistinguishable from a careful one on any single
|
|
8
|
+
iteration, and the transcripts have been accumulating the evidence to tell them
|
|
9
|
+
apart the whole time.
|
|
10
|
+
|
|
11
|
+
This reads those files and nothing else. It starts no build, calls no model,
|
|
12
|
+
spends nothing, and changes no default behaviour. It is a REPORTER, not a gate:
|
|
13
|
+
it exits 0 on terrible calibration, because a reporter that fails CI would just
|
|
14
|
+
be turned off.
|
|
15
|
+
|
|
16
|
+
THE LABELLING CONVENTION, which is the one thing a reader should argue with.
|
|
17
|
+
|
|
18
|
+
prediction = 1.0 when a voter's verdict is APPROVE, 0.0 when REJECT
|
|
19
|
+
label = 1 when the transcript's `outcome` is APPROVED, else 0
|
|
20
|
+
|
|
21
|
+
That convention has a circularity in it, and pretending otherwise would be the
|
|
22
|
+
dishonest move. The council outcome is MECHANICALLY DERIVED from the votes
|
|
23
|
+
(`approve_count >= threshold`), so a voter's own prediction partially CAUSES the
|
|
24
|
+
label it is then scored against. What this measures is therefore AGREEMENT WITH
|
|
25
|
+
THE COUNCIL MAJORITY, not accuracy against ground truth. A voter scoring
|
|
26
|
+
perfectly here may simply be voting with the crowd. No artifact on disk records
|
|
27
|
+
whether the council was actually right, so ground-truth calibration is not
|
|
28
|
+
computable from this substrate at all -- and it is reported that way rather than
|
|
29
|
+
approximated. The convention is printed with every run so a reader can disagree
|
|
30
|
+
with it without reading this source.
|
|
31
|
+
|
|
32
|
+
PREDICTIONS ARE BINARY, which shapes everything downstream. Voters emit a
|
|
33
|
+
verdict, not a probability, so every prediction is exactly 0.0 or 1.0. The
|
|
34
|
+
reliability curve therefore has at most two occupied bins no matter how many
|
|
35
|
+
bins are requested; the rest are empty by construction, not by accident. They
|
|
36
|
+
print as UNKNOWN, never as 0.0, because "no voter ever predicted 0.35" and
|
|
37
|
+
"voters predicted 0.35 and were never right" are opposite findings.
|
|
38
|
+
|
|
39
|
+
ECE DEPENDS ON ITS BIN COUNT. The number changes when the bin count changes,
|
|
40
|
+
so the bin count is printed next to every ECE and stated as a knob. An ECE
|
|
41
|
+
quoted without its bin count is not a measurement.
|
|
42
|
+
|
|
43
|
+
SUPPORT IS PRINTED EVERYWHERE, and a headline is REFUSED below a floor
|
|
44
|
+
(default 30 predictions). A Brier score over four samples is noise wearing a
|
|
45
|
+
decimal point, and the most expensive thing this tool could do is hand someone
|
|
46
|
+
a confident number derived from a handful of rows.
|
|
47
|
+
|
|
48
|
+
PREDICTIONS ARE CLUSTERED, and the count is printed alongside the transcript
|
|
49
|
+
count for that reason. Every voter in one transcript is scored against the SAME
|
|
50
|
+
label, so N votes drawn from M transcripts carry nowhere near N independent
|
|
51
|
+
observations -- the effective sample size is nearer M. Printing the pooled vote
|
|
52
|
+
count alone would inflate apparent support by roughly the council size, which
|
|
53
|
+
in a tool built to refuse overstatement would be the same defect it exists to
|
|
54
|
+
prevent. The floor is applied to predictions because that is the stated knob,
|
|
55
|
+
and the transcript count is printed next to it so a reader can apply their own.
|
|
56
|
+
|
|
57
|
+
WHAT THE ARTIFACTS CANNOT SUPPORT. The brief asked for breakdowns by gate,
|
|
58
|
+
model and task. None of the three is recorded, and each is reported NOT
|
|
59
|
+
AVAILABLE with its reason rather than approximated by a nearby field:
|
|
60
|
+
|
|
61
|
+
by gate -- `outcome` can be BLOCKED_BY_GATE, but that is a gate OUTCOME,
|
|
62
|
+
not a gate IDENTITY. No gate id or name appears anywhere in the
|
|
63
|
+
transcript, so votes cannot be grouped by which gate blocked.
|
|
64
|
+
by model -- `voters[].name` is the ROLE (it is populated from `v.role` in
|
|
65
|
+
councilWriteTranscript). Which model backed a role is not
|
|
66
|
+
written. Role is not a model and is reported as its own axis.
|
|
67
|
+
by task -- `prd_path` and `task_or_prd` (the first 200 chars of the PRD)
|
|
68
|
+
identify the RUN, not a task, and are effectively constant
|
|
69
|
+
across every transcript in one .loki dir, so they separate
|
|
70
|
+
nothing.
|
|
71
|
+
|
|
72
|
+
Nearby numbers exist that would each make a plausible-looking proxy, and all
|
|
73
|
+
are deliberately refused. `last_confidence` is a real number, but it lives in
|
|
74
|
+
council STATE (loki-ts/src/runner/council.ts:129) as a single run-level scalar,
|
|
75
|
+
not per voter and not in the transcript; joining it onto voters would invent a
|
|
76
|
+
per-voter confidence that was never recorded. `.loki/state/uncertainty.json`
|
|
77
|
+
holds BOOLEAN uncertainty proxies, not a confidence, and cannot be read as one.
|
|
78
|
+
|
|
79
|
+
MISSINGNESS IS A RESULT, not a footnote. Skipped transcripts, absent fields and
|
|
80
|
+
CANNOT_VALIDATE votes each get their own counted line. CANNOT_VALIDATE is
|
|
81
|
+
EXCLUDED from the calibration sample -- a voter declining to assert is not a
|
|
82
|
+
wrong probabilistic assertion -- which means the sample size here will NOT match
|
|
83
|
+
the transcript's own `reject_count`, since that field lumps CANNOT_VALIDATE in
|
|
84
|
+
with REJECT. That discrepancy is stated in the output so it does not read as
|
|
85
|
+
dropped rows.
|
|
86
|
+
|
|
87
|
+
Exit codes follow the tools/ convention:
|
|
88
|
+
0 reported
|
|
89
|
+
2 could NOT check (transcript files found, none parseable)
|
|
90
|
+
3 nothing to report (no transcript files)
|
|
91
|
+
64 usage error
|
|
92
|
+
66 input path missing
|
|
93
|
+
|
|
94
|
+
Usage:
|
|
95
|
+
tools/calibration-audit.py [workspace] [--bins N] [--min-support N] [--json]
|
|
96
|
+
"""
|
|
97
|
+
|
|
98
|
+
import argparse
|
|
99
|
+
import json
|
|
100
|
+
import os
|
|
101
|
+
import sys
|
|
102
|
+
|
|
103
|
+
sys.dont_write_bytecode = True
|
|
104
|
+
|
|
105
|
+
UNKNOWN = "UNKNOWN"
|
|
106
|
+
|
|
107
|
+
# Below this many usable predictions, no headline calibration number is
|
|
108
|
+
# presented. Stated in the output, and overridable, because the right floor is
|
|
109
|
+
# a judgement call and a hidden judgement call is one nobody can challenge.
|
|
110
|
+
DEFAULT_MIN_SUPPORT = 30
|
|
111
|
+
DEFAULT_BINS = 10
|
|
112
|
+
|
|
113
|
+
# Dimensions the brief asked for that the artifacts do not record. Printed
|
|
114
|
+
# verbatim so the refusal is visible to a reader who never opens this file.
|
|
115
|
+
NOT_AVAILABLE = [
|
|
116
|
+
("gate", "outcome can be BLOCKED_BY_GATE, but that is a gate OUTCOME, "
|
|
117
|
+
"not a gate IDENTITY; no gate id or name is written to the "
|
|
118
|
+
"transcript, so votes cannot be grouped by gate"),
|
|
119
|
+
("model", "voters[].name is the ROLE (populated from v.role in "
|
|
120
|
+
"councilWriteTranscript); which model backed a role is never "
|
|
121
|
+
"recorded, and role is reported as its own axis instead"),
|
|
122
|
+
("task", "prd_path and task_or_prd (first 200 chars of the PRD) identify "
|
|
123
|
+
"the RUN, not a task, and are effectively constant across every "
|
|
124
|
+
"transcript in one .loki dir, so they separate nothing"),
|
|
125
|
+
]
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
class _Parser(argparse.ArgumentParser):
|
|
129
|
+
"""argparse exits 2 on a usage error, and 2 already means something else.
|
|
130
|
+
|
|
131
|
+
In this convention 2 is "could NOT check" -- a real answer about the
|
|
132
|
+
history. A typo in a flag is not that; it is 64. Left alone, `--bnis`
|
|
133
|
+
would report as a failed scan and a caller could not tell the two apart.
|
|
134
|
+
"""
|
|
135
|
+
|
|
136
|
+
def error(self, message):
|
|
137
|
+
self.print_usage(sys.stderr)
|
|
138
|
+
sys.stderr.write("%s: error: %s\n" % (self.prog, message))
|
|
139
|
+
raise SystemExit(64)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def transcripts_dir(workspace):
|
|
143
|
+
return os.path.join(workspace, ".loki", "council", "transcripts")
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def load_votes(directory):
|
|
147
|
+
"""Read every transcript, returning (votes, missingness).
|
|
148
|
+
|
|
149
|
+
A vote is a flat record so the breakdowns are plain groupings. Anything
|
|
150
|
+
that could not be turned into a vote is counted rather than dropped
|
|
151
|
+
silently: a scan that quietly discards half its input reports on a
|
|
152
|
+
population nobody chose.
|
|
153
|
+
"""
|
|
154
|
+
miss = {
|
|
155
|
+
"files_found": 0,
|
|
156
|
+
"files_unparseable": 0,
|
|
157
|
+
"files_missing_voters": 0,
|
|
158
|
+
"files_missing_outcome": 0,
|
|
159
|
+
"transcripts_used": 0,
|
|
160
|
+
"voters_seen": 0,
|
|
161
|
+
"votes_cannot_validate": 0,
|
|
162
|
+
"votes_unknown_verdict": 0,
|
|
163
|
+
"votes_missing_name": 0,
|
|
164
|
+
"votes_missing_role_index": 0,
|
|
165
|
+
"outcome_blocked_by_gate": 0,
|
|
166
|
+
}
|
|
167
|
+
votes = []
|
|
168
|
+
|
|
169
|
+
try:
|
|
170
|
+
names = sorted(n for n in os.listdir(directory) if n.endswith(".json"))
|
|
171
|
+
except OSError:
|
|
172
|
+
return votes, miss
|
|
173
|
+
|
|
174
|
+
for name in names:
|
|
175
|
+
miss["files_found"] += 1
|
|
176
|
+
path = os.path.join(directory, name)
|
|
177
|
+
try:
|
|
178
|
+
with open(path) as handle:
|
|
179
|
+
data = json.load(handle)
|
|
180
|
+
if not isinstance(data, dict):
|
|
181
|
+
raise ValueError("not an object")
|
|
182
|
+
except (OSError, ValueError):
|
|
183
|
+
miss["files_unparseable"] += 1
|
|
184
|
+
continue
|
|
185
|
+
|
|
186
|
+
outcome = data.get("outcome")
|
|
187
|
+
if not isinstance(outcome, str):
|
|
188
|
+
miss["files_missing_outcome"] += 1
|
|
189
|
+
continue
|
|
190
|
+
voters = data.get("voters")
|
|
191
|
+
if not isinstance(voters, list) or not voters:
|
|
192
|
+
miss["files_missing_voters"] += 1
|
|
193
|
+
continue
|
|
194
|
+
|
|
195
|
+
# The label. BLOCKED_BY_GATE is not APPROVED, so under the stated
|
|
196
|
+
# convention it labels 0 -- counted separately so a reader who reads
|
|
197
|
+
# it differently can re-derive without re-running.
|
|
198
|
+
if outcome == "BLOCKED_BY_GATE":
|
|
199
|
+
miss["outcome_blocked_by_gate"] += 1
|
|
200
|
+
label = 1 if outcome == "APPROVED" else 0
|
|
201
|
+
|
|
202
|
+
miss["transcripts_used"] += 1
|
|
203
|
+
for voter in voters:
|
|
204
|
+
if not isinstance(voter, dict):
|
|
205
|
+
miss["votes_unknown_verdict"] += 1
|
|
206
|
+
continue
|
|
207
|
+
miss["voters_seen"] += 1
|
|
208
|
+
verdict = voter.get("verdict")
|
|
209
|
+
|
|
210
|
+
# CANNOT_VALIDATE is a refusal to assert, not a wrong assertion.
|
|
211
|
+
# Excluding it is a choice, so it is counted where it is visible.
|
|
212
|
+
if verdict == "CANNOT_VALIDATE":
|
|
213
|
+
miss["votes_cannot_validate"] += 1
|
|
214
|
+
continue
|
|
215
|
+
if verdict == "APPROVE":
|
|
216
|
+
prediction = 1.0
|
|
217
|
+
elif verdict == "REJECT":
|
|
218
|
+
prediction = 0.0
|
|
219
|
+
else:
|
|
220
|
+
miss["votes_unknown_verdict"] += 1
|
|
221
|
+
continue
|
|
222
|
+
|
|
223
|
+
voter_name = voter.get("name")
|
|
224
|
+
if not isinstance(voter_name, str) or not voter_name:
|
|
225
|
+
miss["votes_missing_name"] += 1
|
|
226
|
+
voter_name = UNKNOWN
|
|
227
|
+
role_index = voter.get("role_index")
|
|
228
|
+
if not isinstance(role_index, int):
|
|
229
|
+
miss["votes_missing_role_index"] += 1
|
|
230
|
+
role_index = UNKNOWN
|
|
231
|
+
|
|
232
|
+
contrarian = voter.get("is_contrarian")
|
|
233
|
+
votes.append({
|
|
234
|
+
"prediction": prediction,
|
|
235
|
+
"label": label,
|
|
236
|
+
"name": voter_name,
|
|
237
|
+
"role_index": role_index,
|
|
238
|
+
"is_contrarian": (contrarian if isinstance(contrarian, bool)
|
|
239
|
+
else UNKNOWN),
|
|
240
|
+
})
|
|
241
|
+
|
|
242
|
+
return votes, miss
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def brier(votes):
|
|
246
|
+
"""Mean squared error of prediction against label. UNKNOWN over nothing."""
|
|
247
|
+
if not votes:
|
|
248
|
+
return UNKNOWN
|
|
249
|
+
total = sum((v["prediction"] - v["label"]) ** 2 for v in votes)
|
|
250
|
+
return total / len(votes)
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def reliability(votes, bins):
|
|
254
|
+
"""Per-bin predicted rate vs observed rate, WITH support.
|
|
255
|
+
|
|
256
|
+
EVERY bin is returned, including the empty ones, and an empty bin carries
|
|
257
|
+
UNKNOWN rather than 0.0. Predictions here are binary, so most bins are
|
|
258
|
+
empty by construction -- reporting those as a 0.0 observed rate would
|
|
259
|
+
manufacture a perfectly-wrong-looking region of the curve out of the
|
|
260
|
+
absence of data.
|
|
261
|
+
"""
|
|
262
|
+
table = []
|
|
263
|
+
for index in range(bins):
|
|
264
|
+
low = index / bins
|
|
265
|
+
high = (index + 1) / bins
|
|
266
|
+
# Half-open bins, with the last one closed so prediction 1.0 lands.
|
|
267
|
+
if index == bins - 1:
|
|
268
|
+
members = [v for v in votes if low <= v["prediction"] <= high]
|
|
269
|
+
else:
|
|
270
|
+
members = [v for v in votes if low <= v["prediction"] < high]
|
|
271
|
+
if members:
|
|
272
|
+
predicted = sum(v["prediction"] for v in members) / len(members)
|
|
273
|
+
observed = sum(v["label"] for v in members) / len(members)
|
|
274
|
+
else:
|
|
275
|
+
predicted = UNKNOWN
|
|
276
|
+
observed = UNKNOWN
|
|
277
|
+
table.append({
|
|
278
|
+
"bin": index,
|
|
279
|
+
"range": [low, high],
|
|
280
|
+
"support": len(members),
|
|
281
|
+
"predicted_rate": predicted,
|
|
282
|
+
"observed_rate": observed,
|
|
283
|
+
})
|
|
284
|
+
return table
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def ece(votes, bins):
|
|
288
|
+
"""Support-weighted mean gap between predicted and observed rate.
|
|
289
|
+
|
|
290
|
+
Empty bins contribute nothing. That is the DEFINITION of the
|
|
291
|
+
support-weighted sum, not an imputation: a bin with no support has no
|
|
292
|
+
weight, so it cannot pull the number in either direction. It is a
|
|
293
|
+
different thing from treating its observed rate as 0.0, which would.
|
|
294
|
+
"""
|
|
295
|
+
if not votes:
|
|
296
|
+
return UNKNOWN
|
|
297
|
+
table = reliability(votes, bins)
|
|
298
|
+
total = 0.0
|
|
299
|
+
for row in table:
|
|
300
|
+
if row["support"] == 0:
|
|
301
|
+
continue
|
|
302
|
+
gap = abs(row["predicted_rate"] - row["observed_rate"])
|
|
303
|
+
total += (row["support"] / len(votes)) * gap
|
|
304
|
+
return total
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def group(votes, key, bins):
|
|
308
|
+
"""Break the sample down by one recorded field, keeping support per group."""
|
|
309
|
+
buckets = {}
|
|
310
|
+
for vote in votes:
|
|
311
|
+
buckets.setdefault(str(vote[key]), []).append(vote)
|
|
312
|
+
return [
|
|
313
|
+
{
|
|
314
|
+
"value": value,
|
|
315
|
+
"support": len(members),
|
|
316
|
+
"brier": brier(members),
|
|
317
|
+
"ece": ece(members, bins),
|
|
318
|
+
"approve_rate": sum(v["prediction"] for v in members) / len(members),
|
|
319
|
+
"observed_rate": sum(v["label"] for v in members) / len(members),
|
|
320
|
+
}
|
|
321
|
+
for value, members in sorted(buckets.items())
|
|
322
|
+
]
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def audit(workspace, bins, min_support):
|
|
326
|
+
votes, miss = load_votes(transcripts_dir(workspace))
|
|
327
|
+
enough = len(votes) >= min_support
|
|
328
|
+
return {
|
|
329
|
+
"bins": bins,
|
|
330
|
+
"min_support": min_support,
|
|
331
|
+
"total_predictions": len(votes),
|
|
332
|
+
"sufficient_support": enough,
|
|
333
|
+
# The headline is WITHHELD below the floor rather than printed small.
|
|
334
|
+
# A number a reader can see is a number a reader will quote.
|
|
335
|
+
"brier": brier(votes) if enough else UNKNOWN,
|
|
336
|
+
"ece": ece(votes, bins) if enough else UNKNOWN,
|
|
337
|
+
"reliability": reliability(votes, bins),
|
|
338
|
+
"by_name": group(votes, "name", bins),
|
|
339
|
+
"by_role_index": group(votes, "role_index", bins),
|
|
340
|
+
"by_is_contrarian": group(votes, "is_contrarian", bins),
|
|
341
|
+
"missingness": miss,
|
|
342
|
+
"not_available": [{"dimension": d, "reason": r}
|
|
343
|
+
for d, r in NOT_AVAILABLE],
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def _num(value):
|
|
348
|
+
return UNKNOWN if value == UNKNOWN else "%.4f" % value
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def _exit_code(report):
|
|
352
|
+
miss = report["missingness"]
|
|
353
|
+
if miss["files_found"] == 0:
|
|
354
|
+
return 3 # nothing to report
|
|
355
|
+
if miss["transcripts_used"] == 0:
|
|
356
|
+
# Files were there and none survived parsing. That is a FAILED scan,
|
|
357
|
+
# not an empty one, and collapsing it into 3 would let a directory of
|
|
358
|
+
# corrupt transcripts read as "nothing to report".
|
|
359
|
+
return 2
|
|
360
|
+
return 0
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
def _render(report, code):
|
|
364
|
+
bins = report["bins"]
|
|
365
|
+
total = report["total_predictions"]
|
|
366
|
+
miss = report["missingness"]
|
|
367
|
+
lines = ["COUNCIL CALIBRATION AUDIT"]
|
|
368
|
+
lines.append(" reads .loki/council/transcripts only; starts nothing, "
|
|
369
|
+
"spends nothing")
|
|
370
|
+
lines.append("")
|
|
371
|
+
lines.append("LABELLING CONVENTION (disagree with this before the numbers)")
|
|
372
|
+
lines.append(" prediction = 1.0 for APPROVE, 0.0 for REJECT")
|
|
373
|
+
lines.append(" label = 1 when outcome == APPROVED, else 0")
|
|
374
|
+
lines.append(" CANNOT_VALIDATE is EXCLUDED: declining to assert is not a")
|
|
375
|
+
lines.append(" wrong assertion. So this sample size will NOT match the")
|
|
376
|
+
lines.append(" transcript reject_count, which lumps it in with REJECT.")
|
|
377
|
+
lines.append(" CIRCULARITY: outcome is derived from the votes")
|
|
378
|
+
lines.append(" (approve_count >= threshold), so a voter's prediction")
|
|
379
|
+
lines.append(" partly CAUSES its own label. This measures AGREEMENT")
|
|
380
|
+
lines.append(" WITH THE MAJORITY, not accuracy against ground truth.")
|
|
381
|
+
lines.append(" No artifact records whether the council was right.")
|
|
382
|
+
lines.append(" Predictions are BINARY (a verdict, not a probability), so")
|
|
383
|
+
lines.append(" at most two bins can ever be occupied.")
|
|
384
|
+
|
|
385
|
+
if code == 3:
|
|
386
|
+
lines.append("")
|
|
387
|
+
lines.append("NOTHING TO REPORT -- no transcript files found.")
|
|
388
|
+
lines.append(" Scanning nothing is not evidence of good calibration.")
|
|
389
|
+
return "\n".join(lines)
|
|
390
|
+
if code == 2:
|
|
391
|
+
lines.append("")
|
|
392
|
+
lines.append("CANNOT CHECK -- %d transcript file(s) found, none "
|
|
393
|
+
"parseable." % miss["files_found"])
|
|
394
|
+
return "\n".join(lines)
|
|
395
|
+
|
|
396
|
+
lines.append("")
|
|
397
|
+
lines.append("SUPPORT")
|
|
398
|
+
lines.append(" usable predictions: %d from %d transcript(s)"
|
|
399
|
+
% (total, miss["transcripts_used"]))
|
|
400
|
+
lines.append(" Votes within one transcript share a label, so predictions")
|
|
401
|
+
lines.append(" are CLUSTERED, not independent: the effective sample")
|
|
402
|
+
lines.append(" size is nearer the transcript count than the vote count.")
|
|
403
|
+
lines.append(" The floor below is applied to PREDICTIONS (the stated "
|
|
404
|
+
"knob): %d" % report["min_support"])
|
|
405
|
+
if not report["sufficient_support"]:
|
|
406
|
+
lines.append("")
|
|
407
|
+
lines.append(" INSUFFICIENT SUPPORT -- headline calibration numbers "
|
|
408
|
+
"are WITHHELD.")
|
|
409
|
+
lines.append(" Usable predictions (%d) is below the stated floor (%d)."
|
|
410
|
+
% (total, report["min_support"]))
|
|
411
|
+
lines.append(" The breakdowns below are printed with their support so "
|
|
412
|
+
"they can be")
|
|
413
|
+
lines.append(" read as counts, not as calibration.")
|
|
414
|
+
else:
|
|
415
|
+
lines.append("")
|
|
416
|
+
lines.append("HEADLINE")
|
|
417
|
+
lines.append(" Brier score: %s (0 is perfect, lower is better)"
|
|
418
|
+
% _num(report["brier"]))
|
|
419
|
+
lines.append(" ECE: %s at %d bins"
|
|
420
|
+
% (_num(report["ece"]), bins))
|
|
421
|
+
lines.append(" ECE depends on its bin count: changing --bins changes "
|
|
422
|
+
"this number.")
|
|
423
|
+
|
|
424
|
+
lines.append("")
|
|
425
|
+
lines.append("RELIABILITY TABLE (%d bins; empty bins are UNKNOWN, not 0)"
|
|
426
|
+
% bins)
|
|
427
|
+
lines.append(" %-14s %8s %10s %10s" % ("bin", "support", "predicted",
|
|
428
|
+
"observed"))
|
|
429
|
+
for row in report["reliability"]:
|
|
430
|
+
lines.append(" %-14s %8d %10s %10s"
|
|
431
|
+
% ("[%.2f,%.2f]" % (row["range"][0], row["range"][1]),
|
|
432
|
+
row["support"], _num(row["predicted_rate"]),
|
|
433
|
+
_num(row["observed_rate"])))
|
|
434
|
+
|
|
435
|
+
for title, key in (("BY VOTER NAME (role)", "by_name"),
|
|
436
|
+
("BY ROLE INDEX", "by_role_index"),
|
|
437
|
+
("BY IS_CONTRARIAN", "by_is_contrarian")):
|
|
438
|
+
lines.append("")
|
|
439
|
+
lines.append(title)
|
|
440
|
+
rows = report[key]
|
|
441
|
+
if not rows:
|
|
442
|
+
lines.append(" UNKNOWN -- no usable votes carried this field")
|
|
443
|
+
continue
|
|
444
|
+
lines.append(" %-28s %8s %9s %9s" % ("value", "support", "brier",
|
|
445
|
+
"ece"))
|
|
446
|
+
for row in rows:
|
|
447
|
+
flag = "" if row["support"] >= report["min_support"] else " (low)"
|
|
448
|
+
lines.append(" %-28s %8d %9s %9s%s"
|
|
449
|
+
% (row["value"][:28], row["support"],
|
|
450
|
+
_num(row["brier"]), _num(row["ece"]), flag))
|
|
451
|
+
|
|
452
|
+
lines.append("")
|
|
453
|
+
lines.append("NOT AVAILABLE -- asked for, not recorded, not approximated")
|
|
454
|
+
for item in report["not_available"]:
|
|
455
|
+
lines.append(" by %s: %s" % (item["dimension"], item["reason"]))
|
|
456
|
+
|
|
457
|
+
lines.append("")
|
|
458
|
+
lines.append("MISSINGNESS")
|
|
459
|
+
lines.append(" transcript files found: %d" % miss["files_found"])
|
|
460
|
+
lines.append(" transcripts used: %d" % miss["transcripts_used"])
|
|
461
|
+
lines.append(" files unparseable: %d" % miss["files_unparseable"])
|
|
462
|
+
lines.append(" files with no voters[]: %d"
|
|
463
|
+
% miss["files_missing_voters"])
|
|
464
|
+
lines.append(" files with no outcome: %d"
|
|
465
|
+
% miss["files_missing_outcome"])
|
|
466
|
+
lines.append(" voter entries seen: %d" % miss["voters_seen"])
|
|
467
|
+
lines.append(" CANNOT_VALIDATE (excluded): %d"
|
|
468
|
+
% miss["votes_cannot_validate"])
|
|
469
|
+
lines.append(" unusable verdict: %d"
|
|
470
|
+
% miss["votes_unknown_verdict"])
|
|
471
|
+
lines.append(" votes with no name: %d" % miss["votes_missing_name"])
|
|
472
|
+
lines.append(" votes with no role_index: %d"
|
|
473
|
+
% miss["votes_missing_role_index"])
|
|
474
|
+
lines.append(" outcome BLOCKED_BY_GATE: %d (labelled 0 here)"
|
|
475
|
+
% miss["outcome_blocked_by_gate"])
|
|
476
|
+
|
|
477
|
+
lines.append("")
|
|
478
|
+
lines.append("This is a REPORTER, not a gate. It exits 0 even when "
|
|
479
|
+
"calibration is bad.")
|
|
480
|
+
return "\n".join(lines)
|
|
481
|
+
|
|
482
|
+
|
|
483
|
+
def main(argv=None):
|
|
484
|
+
parser = _Parser(
|
|
485
|
+
description="Audit council voter calibration over transcripts on disk.")
|
|
486
|
+
parser.add_argument("workspace", nargs="?", default=".",
|
|
487
|
+
help="workspace root holding .loki (default: cwd)")
|
|
488
|
+
parser.add_argument("--bins", type=int, default=DEFAULT_BINS,
|
|
489
|
+
help="reliability bin count (default: %d); changing "
|
|
490
|
+
"it changes ECE" % DEFAULT_BINS)
|
|
491
|
+
parser.add_argument("--min-support", type=int, default=DEFAULT_MIN_SUPPORT,
|
|
492
|
+
help="predictions required before a headline "
|
|
493
|
+
"calibration number is presented (default: %d)"
|
|
494
|
+
% DEFAULT_MIN_SUPPORT)
|
|
495
|
+
parser.add_argument("--json", action="store_true",
|
|
496
|
+
help="emit machine-readable output")
|
|
497
|
+
args = parser.parse_args(argv)
|
|
498
|
+
|
|
499
|
+
if not os.path.exists(args.workspace):
|
|
500
|
+
payload = {"status": "input_missing", "exit_code": 66,
|
|
501
|
+
"error": "no such path: " + args.workspace}
|
|
502
|
+
print(json.dumps(payload, indent=2) if args.json
|
|
503
|
+
else "INPUT MISSING -- no such path: " + args.workspace)
|
|
504
|
+
return 66
|
|
505
|
+
if args.bins < 1:
|
|
506
|
+
parser.error("--bins must be at least 1")
|
|
507
|
+
if args.min_support < 0:
|
|
508
|
+
parser.error("--min-support cannot be negative")
|
|
509
|
+
|
|
510
|
+
report = audit(args.workspace, args.bins, args.min_support)
|
|
511
|
+
code = _exit_code(report)
|
|
512
|
+
if args.json:
|
|
513
|
+
report["status"] = {3: "nothing_to_report", 2: "cannot_check"}.get(
|
|
514
|
+
code, "reported")
|
|
515
|
+
report["exit_code"] = code
|
|
516
|
+
print(json.dumps(report, indent=2))
|
|
517
|
+
else:
|
|
518
|
+
print(_render(report, code))
|
|
519
|
+
return code
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
if __name__ == "__main__":
|
|
523
|
+
sys.exit(main())
|
package/tools/ci-gate.py
CHANGED
|
@@ -81,6 +81,24 @@ PASS, FAIL, UNEVALUABLE = 0, 1, 2
|
|
|
81
81
|
_STATE = {PASS: "PASS", FAIL: "FAIL", UNEVALUABLE: "UNEVALUABLE"}
|
|
82
82
|
|
|
83
83
|
|
|
84
|
+
class _Parser(argparse.ArgumentParser):
|
|
85
|
+
"""Usage errors exit 64, not argparse's default 2.
|
|
86
|
+
|
|
87
|
+
In this repo's convention 2 means "could NOT be checked" -- a real
|
|
88
|
+
answer about the subject. A mistyped flag is not that: it is an error
|
|
89
|
+
about the INVOCATION, and nothing about the subject was examined. The
|
|
90
|
+
two call for opposite responses, since retrying cannot fix a typo.
|
|
91
|
+
|
|
92
|
+
argparse exits 2 for every usage error unless this is overridden, so
|
|
93
|
+
every tool needs it. tests/test_tool_exit_contract.py asserts it.
|
|
94
|
+
"""
|
|
95
|
+
|
|
96
|
+
def error(self, message):
|
|
97
|
+
self.print_usage(sys.stderr)
|
|
98
|
+
sys.stderr.write("%s: error: %s\n" % (self.prog, message))
|
|
99
|
+
raise SystemExit(64)
|
|
100
|
+
|
|
101
|
+
|
|
84
102
|
def _row(policy, code, reason):
|
|
85
103
|
return {"policy": policy, "state": _STATE[code], "exit_code": code,
|
|
86
104
|
"reason": reason}
|
|
@@ -201,7 +219,7 @@ def render(d):
|
|
|
201
219
|
|
|
202
220
|
|
|
203
221
|
def main(argv=None):
|
|
204
|
-
ap =
|
|
222
|
+
ap = _Parser(
|
|
205
223
|
description="One exit code over every configured merge policy.")
|
|
206
224
|
ap.add_argument("workspace", nargs="?", default=".",
|
|
207
225
|
help="workspace root (or its .loki dir); default .")
|