loki-mode 9.8.0 → 9.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -14
- package/SKILL.md +3 -2
- package/VERSION +1 -1
- package/autonomy/loki +122 -1
- package/autonomy/run.sh +49 -2
- package/dashboard/__init__.py +1 -1
- package/dashboard/api_evidence.py +411 -0
- package/dashboard/api_operator.py +283 -0
- package/dashboard/api_phases.py +262 -0
- package/dashboard/api_releases.py +242 -0
- package/dashboard/api_runs.py +477 -0
- package/dashboard/api_tests.py +444 -0
- package/dashboard/api_v2.py +47 -1
- package/dashboard/server.py +54 -0
- package/dashboard/static/index.html +246 -135
- package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
- package/docs/CAPABILITY-BACKLOG.md +53 -0
- package/docs/COMPARISON.md +2 -2
- package/docs/COMPETITIVE-ANALYSIS.md +1 -1
- package/docs/COMPETITIVE-SCORECARD.md +422 -0
- package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
- package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
- package/docs/DEMOS.md +21 -23
- package/docs/HANDOFF-2026-08-03.md +439 -0
- package/docs/INSTALLATION.md +17 -10
- package/docs/OUTCOME-FRONTIER.md +536 -0
- package/docs/PROMPT-ABLATION-RESULT.md +97 -0
- package/docs/TOOLS.md +800 -0
- package/docs/alternative-installations.md +2 -3
- package/docs/audit-logging.md +44 -35
- package/docs/authentication.md +13 -2
- package/docs/authorization.md +87 -81
- package/docs/git-workflow.md +6 -3
- package/docs/metrics.md +15 -16
- package/docs/network-security.md +16 -13
- package/docs/openclaw-integration.md +36 -556
- package/docs/show-hn-post.md +2 -2
- package/docs/siem-integration.md +39 -36
- package/loki-ts/dist/loki.js +18 -18
- package/mcp/__init__.py +1 -1
- package/package.json +2 -2
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
- package/references/confidence-routing.md +18 -1
- package/references/invariant-checks.md +13 -8
- package/references/magic-rarv-integration.md +0 -1
- package/references/multi-provider.md +27 -5
- package/skills/healing.md +4 -2
- package/tools/audit-docs.py +488 -0
- package/tools/baseline-pin.py +19 -1
- package/tools/calibration-audit.py +523 -0
- package/tools/ci-gate.py +19 -1
- package/tools/cost-forecast.py +344 -0
- package/tools/cost-guard.py +19 -1
- package/tools/cost-history.py +19 -1
- package/tools/cost-per-outcome.py +394 -0
- package/tools/estimate-run.py +19 -1
- package/tools/evidence-freshness.py +307 -0
- package/tools/gate-init.py +19 -1
- package/tools/gate-report.py +19 -1
- package/tools/gate-simulate.py +570 -0
- package/tools/gate-trend.py +354 -0
- package/tools/model-advisor.py +52 -1
- package/tools/policy-load.py +19 -1
- package/tools/prompt-cost.py +363 -0
- package/tools/prompt-diff.py +448 -0
- package/tools/prompt-lint.py +448 -0
- package/tools/receipt-bundle.py +72 -2
- package/tools/receipt-diff.py +19 -1
- package/tools/receipt-find.py +19 -1
- package/tools/receipt-stats.py +380 -0
- package/tools/receipt-timeline.py +478 -0
- package/tools/receipt-verify-batch.py +291 -0
- package/tools/run-replay.py +19 -1
- package/tools/signing-status.py +19 -1
- package/tools/token-guard.py +19 -1
- package/tools/token-tax.py +375 -0
- package/tools/tool-index.py +19 -1
- package/tools/verification-tax.py +277 -0
- package/tools/verify-chain.py +361 -0
|
@@ -0,0 +1,375 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""What does verification cost in TOKENS, versus the build itself?
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. tools/verification-tax.py measures the gate in WALL TIME and
|
|
5
|
+
in how often it changed an outcome, and says in its own header that it does not
|
|
6
|
+
measure token cost. docs/CAPABILITY-BACKLOG.md carries that as the open half of
|
|
7
|
+
a SHIPPED-partial row ("Remaining: token cost"). This is that half.
|
|
8
|
+
|
|
9
|
+
WHAT IT REPORTS, and the one thing it REFUSES.
|
|
10
|
+
|
|
11
|
+
Tokens are recorded per ITERATION, in .loki/metrics/efficiency/iteration-N.json.
|
|
12
|
+
Stages are recorded per STAGE, as stage_complete events carrying a name, a
|
|
13
|
+
status and a duration_s. Those are the only two artifacts a run writes, and
|
|
14
|
+
NEITHER attributes a token to a stage. So:
|
|
15
|
+
|
|
16
|
+
THE VERIFICATION SHARE OF TOKENS IS NOT REPORTED, BECAUSE NOTHING
|
|
17
|
+
RECORDS IT.
|
|
18
|
+
|
|
19
|
+
The tempting fix is to weight each iteration's tokens by each stage's share of
|
|
20
|
+
that iteration's wall clock. tools/cost-attribute.py already refused exactly
|
|
21
|
+
that for dollars and named it correctly: the same invention wearing a
|
|
22
|
+
defensible-looking coat, and worse than an even split precisely because a
|
|
23
|
+
reader will believe it. A gate that shells out to eslint for thirty seconds
|
|
24
|
+
spends no tokens; a review stage that streams three completions in ten seconds
|
|
25
|
+
spends most of the iteration. Wall clock is not a token proxy.
|
|
26
|
+
|
|
27
|
+
What IS measured, and is reported:
|
|
28
|
+
|
|
29
|
+
tokens per iteration measured, unmeasured iterations excluded
|
|
30
|
+
verification stages run counted from stage_complete, by name
|
|
31
|
+
build stages run same, for the one stage that IS the agent
|
|
32
|
+
tokens per verification-stage-execution the denominator is measured and
|
|
33
|
+
the numerator is measured, but the DIVISION is
|
|
34
|
+
only an upper bound and is labelled as one
|
|
35
|
+
|
|
36
|
+
THE HONESTY RULE, which is the reason for the whole file. An iteration with no
|
|
37
|
+
recorded token counts reads UNKNOWN and is EXCLUDED from every average. It is
|
|
38
|
+
never averaged in as zero. Averaging absent measurements toward zero slides the
|
|
39
|
+
mean toward "verification is free" -- which is the conclusion someone building
|
|
40
|
+
a case for removing gates would want, reached by arithmetic rather than by
|
|
41
|
+
evidence. Measured-ness is record_is_measured() in
|
|
42
|
+
autonomy/lib/efficiency_cost.py, imported and never restated, fed the FOUR
|
|
43
|
+
TOKEN FIELDS ONLY: a record carrying a real cost_usd and zero tokens (the shape
|
|
44
|
+
codex wrote before v8.51.0) is a run whose TOKENS were never measured, and
|
|
45
|
+
feeding cost_usd in would report it as "0 tokens", which is the exact lie this
|
|
46
|
+
file exists to prevent.
|
|
47
|
+
|
|
48
|
+
WHAT THIS IS NOT. Not a gate. A workspace whose every gate BLOCKED still exits
|
|
49
|
+
0 here, because the question asked was "what did it cost", and that question
|
|
50
|
+
was answered. tools/token-guard.py is the gate.
|
|
51
|
+
|
|
52
|
+
Usage:
|
|
53
|
+
tools/token-tax.py [workspace] [--json]
|
|
54
|
+
|
|
55
|
+
Exit: 0 reported, 2 could not check, 3 nothing to report, 64 usage error,
|
|
56
|
+
66 input path missing.
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
import argparse
|
|
60
|
+
import json
|
|
61
|
+
import os
|
|
62
|
+
import sys
|
|
63
|
+
|
|
64
|
+
sys.dont_write_bytecode = True
|
|
65
|
+
|
|
66
|
+
_HERE = os.path.dirname(os.path.abspath(__file__))
|
|
67
|
+
# Resolved from __file__, never from the workspace argument: the workspace being
|
|
68
|
+
# read is a different tree and has no autonomy/lib.
|
|
69
|
+
sys.path.insert(0, os.path.join(os.path.dirname(_HERE), "autonomy", "lib"))
|
|
70
|
+
|
|
71
|
+
from efficiency_cost import record_is_measured # noqa: E402
|
|
72
|
+
|
|
73
|
+
OK = 0
|
|
74
|
+
COULD_NOT_CHECK = 2
|
|
75
|
+
NOTHING_TO_REPORT = 3
|
|
76
|
+
USAGE_ERROR = 64
|
|
77
|
+
INPUT_MISSING = 66
|
|
78
|
+
|
|
79
|
+
# The four token fields. cost_usd is deliberately absent; see the docstring.
|
|
80
|
+
TOKEN_FIELDS = ("input_tokens", "output_tokens", "cache_read_tokens",
|
|
81
|
+
"cache_creation_tokens")
|
|
82
|
+
|
|
83
|
+
# The one stage that IS the build. Every other stage emitted by
|
|
84
|
+
# emit_stage_complete in autonomy/run.sh is verification or gate work.
|
|
85
|
+
BUILD_STAGES = frozenset(["agent"])
|
|
86
|
+
|
|
87
|
+
WHY_NO_SPLIT = (
|
|
88
|
+
"REFUSED. Tokens are recorded per ITERATION and stages are recorded per "
|
|
89
|
+
"STAGE; no artifact this repo writes attributes a token to a stage. "
|
|
90
|
+
"Weighting each iteration's tokens by a stage's share of wall clock would "
|
|
91
|
+
"look like a measurement and would not be one -- a gate shelling out to a "
|
|
92
|
+
"linter burns seconds and no tokens, while one streaming a completion "
|
|
93
|
+
"burns tokens in no time. The split becomes reportable when a writer "
|
|
94
|
+
"records per-stage usage, not before."
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class _Parser(argparse.ArgumentParser):
|
|
99
|
+
"""Usage errors exit 64, not argparse's default 2.
|
|
100
|
+
|
|
101
|
+
In this repo's convention 2 means "could NOT be checked" -- a real answer
|
|
102
|
+
about the subject. A mistyped flag is not that: it is an error about the
|
|
103
|
+
INVOCATION, and nothing about the subject was examined. The two call for
|
|
104
|
+
opposite responses, since retrying cannot fix a typo.
|
|
105
|
+
"""
|
|
106
|
+
|
|
107
|
+
def error(self, message):
|
|
108
|
+
self.print_usage(sys.stderr)
|
|
109
|
+
sys.stderr.write("%s: error: %s\n" % (self.prog, message))
|
|
110
|
+
raise SystemExit(USAGE_ERROR)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _num(v):
|
|
114
|
+
"""A number as itself; None, "", or a bool as None."""
|
|
115
|
+
if isinstance(v, bool) or not isinstance(v, (int, float)):
|
|
116
|
+
return None
|
|
117
|
+
return v
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def measured_tokens(rec):
|
|
121
|
+
"""The four token fields summed, or None when this run never measured them.
|
|
122
|
+
|
|
123
|
+
None means UNKNOWN. The caller must not treat it as zero -- that is the
|
|
124
|
+
single property this whole file is built around.
|
|
125
|
+
"""
|
|
126
|
+
if not isinstance(rec, dict):
|
|
127
|
+
return None
|
|
128
|
+
fields = {f: _num(rec.get(f)) for f in TOKEN_FIELDS}
|
|
129
|
+
if not record_is_measured(fields):
|
|
130
|
+
return None
|
|
131
|
+
# Measured, so a None in one field is a real gap inside a real measurement.
|
|
132
|
+
# Count it as 0 for summing rather than poisoning the whole record.
|
|
133
|
+
return {f: (0 if fields[f] is None else fields[f]) for f in TOKEN_FIELDS}
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def read_iterations(loki_dir):
|
|
137
|
+
"""[{iteration, tokens}] for every iteration-N.json, tokens None if unmeasured.
|
|
138
|
+
|
|
139
|
+
Unreadable and non-dict records are kept with tokens=None rather than
|
|
140
|
+
dropped: a file that exists and cannot be read is an iteration that
|
|
141
|
+
happened and was not measured, which is exactly what this reports.
|
|
142
|
+
"""
|
|
143
|
+
eff_dir = os.path.join(loki_dir, "metrics", "efficiency")
|
|
144
|
+
try:
|
|
145
|
+
names = sorted(os.listdir(eff_dir))
|
|
146
|
+
except OSError:
|
|
147
|
+
return []
|
|
148
|
+
out = []
|
|
149
|
+
for name in names:
|
|
150
|
+
if not (name.startswith("iteration-") and name.endswith(".json")):
|
|
151
|
+
continue
|
|
152
|
+
try:
|
|
153
|
+
n = int(name[len("iteration-"):-len(".json")])
|
|
154
|
+
except ValueError:
|
|
155
|
+
continue
|
|
156
|
+
try:
|
|
157
|
+
with open(os.path.join(eff_dir, name), "r", encoding="utf-8") as fh:
|
|
158
|
+
rec = json.load(fh)
|
|
159
|
+
except (OSError, ValueError):
|
|
160
|
+
rec = None
|
|
161
|
+
out.append({"iteration": n, "tokens": measured_tokens(rec)})
|
|
162
|
+
out.sort(key=lambda r: r["iteration"])
|
|
163
|
+
return out
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def read_stages(loki_dir):
|
|
167
|
+
"""({stage: executions}, corrupt_line_count) from stage_complete events.
|
|
168
|
+
|
|
169
|
+
Corrupt lines are counted, never silently dropped: a tool that skips bad
|
|
170
|
+
lines reports a cleaner run than happened, and skips most often when
|
|
171
|
+
something upstream is broken.
|
|
172
|
+
"""
|
|
173
|
+
path = os.path.join(loki_dir, "events.jsonl")
|
|
174
|
+
counts, corrupt = {}, 0
|
|
175
|
+
try:
|
|
176
|
+
fh = open(path, "r", encoding="utf-8", errors="replace")
|
|
177
|
+
except OSError:
|
|
178
|
+
return counts, corrupt
|
|
179
|
+
with fh:
|
|
180
|
+
for line in fh:
|
|
181
|
+
if not line.strip():
|
|
182
|
+
continue
|
|
183
|
+
try:
|
|
184
|
+
e = json.loads(line)
|
|
185
|
+
except ValueError:
|
|
186
|
+
corrupt += 1
|
|
187
|
+
continue
|
|
188
|
+
if not isinstance(e, dict):
|
|
189
|
+
corrupt += 1
|
|
190
|
+
continue
|
|
191
|
+
if (e.get("type") or e.get("event")) != "stage_complete":
|
|
192
|
+
continue
|
|
193
|
+
data = e.get("data") if isinstance(e.get("data"), dict) else {}
|
|
194
|
+
name = data.get("stage")
|
|
195
|
+
if name:
|
|
196
|
+
counts[str(name)] = counts.get(str(name), 0) + 1
|
|
197
|
+
return counts, corrupt
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def summarize(iterations, stages, corrupt):
|
|
201
|
+
"""The report. Averages cover MEASURED iterations only, and say so."""
|
|
202
|
+
# THE EXCLUSION. One filter site, not a repeated guard: an unmeasured
|
|
203
|
+
# iteration leaves the arithmetic entirely rather than entering it as zero.
|
|
204
|
+
measured = [r for r in iterations if r["tokens"] is not None]
|
|
205
|
+
unmeasured = len(iterations) - len(measured)
|
|
206
|
+
|
|
207
|
+
totals = {f: sum(r["tokens"][f] for r in measured) for f in TOKEN_FIELDS}
|
|
208
|
+
grand = sum(totals.values()) if measured else None
|
|
209
|
+
|
|
210
|
+
verification = {k: v for k, v in stages.items() if k not in BUILD_STAGES}
|
|
211
|
+
build = {k: v for k, v in stages.items() if k in BUILD_STAGES}
|
|
212
|
+
v_execs = sum(verification.values())
|
|
213
|
+
|
|
214
|
+
# An UPPER BOUND, and labelled as one everywhere it appears: it divides ALL
|
|
215
|
+
# of a run's tokens by the verification executions, as though the build
|
|
216
|
+
# spent none. It is the largest the true figure could be, which makes it
|
|
217
|
+
# useful for bounding a claim and useless as an estimate.
|
|
218
|
+
ceiling = None
|
|
219
|
+
if grand is not None and v_execs:
|
|
220
|
+
ceiling = round(grand / v_execs, 1)
|
|
221
|
+
|
|
222
|
+
return {
|
|
223
|
+
"iterations": len(iterations),
|
|
224
|
+
"measured_iterations": len(measured),
|
|
225
|
+
"unmeasured_iterations": unmeasured,
|
|
226
|
+
"corrupt_lines": corrupt,
|
|
227
|
+
"totals": totals if measured else {f: None for f in TOKEN_FIELDS},
|
|
228
|
+
"total_tokens": grand,
|
|
229
|
+
"mean_tokens_per_measured_iteration": (
|
|
230
|
+
round(grand / len(measured), 1) if measured else None),
|
|
231
|
+
"per_iteration": [
|
|
232
|
+
{"iteration": r["iteration"],
|
|
233
|
+
"tokens": (None if r["tokens"] is None
|
|
234
|
+
else sum(r["tokens"].values()))}
|
|
235
|
+
for r in iterations
|
|
236
|
+
],
|
|
237
|
+
"verification_stage_executions": v_execs,
|
|
238
|
+
"verification_stages": dict(sorted(verification.items())),
|
|
239
|
+
"build_stage_executions": sum(build.values()),
|
|
240
|
+
"build_stages": dict(sorted(build.items())),
|
|
241
|
+
"verification_tokens": None,
|
|
242
|
+
"build_tokens": None,
|
|
243
|
+
"verification_token_share": None,
|
|
244
|
+
"verification_token_share_why": WHY_NO_SPLIT,
|
|
245
|
+
"tokens_per_verification_execution_upper_bound": ceiling,
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def render(s):
|
|
250
|
+
out = ["TOKEN TAX -- read from artifacts only; nothing was started and "
|
|
251
|
+
"nothing was spent.", ""]
|
|
252
|
+
out.append(" iterations %d" % s["iterations"])
|
|
253
|
+
out.append(" measured %d of %d"
|
|
254
|
+
% (s["measured_iterations"], s["iterations"]))
|
|
255
|
+
|
|
256
|
+
if s["measured_iterations"]:
|
|
257
|
+
out.append(" total tokens %d" % s["total_tokens"])
|
|
258
|
+
for f in TOKEN_FIELDS:
|
|
259
|
+
out.append(" %-22s %d" % (f, s["totals"][f]))
|
|
260
|
+
out.append(" mean per measured %.1f"
|
|
261
|
+
% s["mean_tokens_per_measured_iteration"])
|
|
262
|
+
else:
|
|
263
|
+
out.append(" total tokens UNKNOWN (no iteration recorded any "
|
|
264
|
+
"token count)")
|
|
265
|
+
out.append(" mean per measured UNKNOWN")
|
|
266
|
+
|
|
267
|
+
if s["unmeasured_iterations"]:
|
|
268
|
+
out.append(" UNMEASURED %d iteration(s) carry no token count; "
|
|
269
|
+
"EXCLUDED from" % s["unmeasured_iterations"])
|
|
270
|
+
out.append(" the totals and the mean above, NOT "
|
|
271
|
+
"counted as 0 tokens.")
|
|
272
|
+
out.append(" The real figure is HIGHER than what "
|
|
273
|
+
"is printed here.")
|
|
274
|
+
|
|
275
|
+
out.append("")
|
|
276
|
+
out.append("PER ITERATION")
|
|
277
|
+
out.append("-" * 60)
|
|
278
|
+
for r in s["per_iteration"]:
|
|
279
|
+
if r["tokens"] is None:
|
|
280
|
+
out.append(" iteration %-8d UNKNOWN (not measured; excluded)"
|
|
281
|
+
% r["iteration"])
|
|
282
|
+
else:
|
|
283
|
+
out.append(" iteration %-8d %d tokens" % (r["iteration"],
|
|
284
|
+
r["tokens"]))
|
|
285
|
+
|
|
286
|
+
out.append("")
|
|
287
|
+
out.append("STAGES RUN (counted, not priced)")
|
|
288
|
+
out.append("-" * 60)
|
|
289
|
+
if s["verification_stages"]:
|
|
290
|
+
for name, n in s["verification_stages"].items():
|
|
291
|
+
out.append(" verification %-22s %d execution(s)" % (name, n))
|
|
292
|
+
else:
|
|
293
|
+
out.append(" verification no stage_complete record -- stages not "
|
|
294
|
+
"recorded (not 0 runs)")
|
|
295
|
+
for name, n in s["build_stages"].items():
|
|
296
|
+
out.append(" build %-22s %d execution(s)" % (name, n))
|
|
297
|
+
|
|
298
|
+
out.append("")
|
|
299
|
+
out.append("VERIFICATION SHARE OF TOKENS")
|
|
300
|
+
out.append("-" * 60)
|
|
301
|
+
for chunk in WHY_NO_SPLIT.split(" -- "):
|
|
302
|
+
out.append(" %s" % chunk)
|
|
303
|
+
if s["tokens_per_verification_execution_upper_bound"] is not None:
|
|
304
|
+
out.append("")
|
|
305
|
+
out.append(" UPPER BOUND ONLY: %.1f tokens per verification execution "
|
|
306
|
+
"IF the build"
|
|
307
|
+
% s["tokens_per_verification_execution_upper_bound"])
|
|
308
|
+
out.append(" had spent none, which it did not. This bounds a claim; "
|
|
309
|
+
"it does not estimate one.")
|
|
310
|
+
|
|
311
|
+
if s["corrupt_lines"]:
|
|
312
|
+
out.append("")
|
|
313
|
+
out.append(" CORRUPT %d unreadable event line(s), counted not "
|
|
314
|
+
"dropped" % s["corrupt_lines"])
|
|
315
|
+
|
|
316
|
+
out.append("")
|
|
317
|
+
if not s["measured_iterations"]:
|
|
318
|
+
out.append(" READ: token cost is UNMEASURED. No claim that "
|
|
319
|
+
"verification is cheap or")
|
|
320
|
+
out.append(" expensive can be made from this workspace. Absence of a "
|
|
321
|
+
"number is not a")
|
|
322
|
+
out.append(" small number.")
|
|
323
|
+
return "\n".join(out)
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def main(argv=None):
|
|
327
|
+
ap = _Parser(
|
|
328
|
+
description="Report what verification cost in TOKENS versus the "
|
|
329
|
+
"build. Reports only; never blocks.")
|
|
330
|
+
ap.add_argument("workspace", nargs="?", default=".",
|
|
331
|
+
help="workspace root containing .loki/ (default .)")
|
|
332
|
+
ap.add_argument("--json", action="store_true", dest="as_json",
|
|
333
|
+
help="emit the report as JSON")
|
|
334
|
+
args = ap.parse_args(argv)
|
|
335
|
+
|
|
336
|
+
if not os.path.exists(args.workspace):
|
|
337
|
+
# 66, not COULD_NOT_CHECK. "The path you named does not exist" is a
|
|
338
|
+
# fact about the INPUT; "I could not evaluate" is a fact about the
|
|
339
|
+
# subject. A caller retrying on 2 would retry forever against a typo.
|
|
340
|
+
sys.stderr.write("token-tax: no such workspace: %s\n" % args.workspace)
|
|
341
|
+
return INPUT_MISSING
|
|
342
|
+
|
|
343
|
+
loki = args.workspace
|
|
344
|
+
if os.path.basename(os.path.normpath(loki)) != ".loki":
|
|
345
|
+
loki = os.path.join(loki, ".loki")
|
|
346
|
+
if not os.path.isdir(loki):
|
|
347
|
+
sys.stderr.write("token-tax: no .loki directory under %s; nothing to "
|
|
348
|
+
"report\n" % args.workspace)
|
|
349
|
+
return NOTHING_TO_REPORT
|
|
350
|
+
|
|
351
|
+
try:
|
|
352
|
+
iterations = read_iterations(loki)
|
|
353
|
+
stages, corrupt = read_stages(loki)
|
|
354
|
+
except OSError as exc:
|
|
355
|
+
sys.stderr.write("token-tax: could not read %s: %s\n" % (loki, exc))
|
|
356
|
+
return COULD_NOT_CHECK
|
|
357
|
+
|
|
358
|
+
if not iterations and not stages and not corrupt:
|
|
359
|
+
sys.stderr.write("token-tax: no efficiency records and no stage "
|
|
360
|
+
"events under %s; nothing to report\n" % loki)
|
|
361
|
+
return NOTHING_TO_REPORT
|
|
362
|
+
|
|
363
|
+
s = summarize(iterations, stages, corrupt)
|
|
364
|
+
# Records present but NONE measured is exit 0, deliberately: the question
|
|
365
|
+
# "what did this cost" was answered, and the answer is UNKNOWN. Collapsing
|
|
366
|
+
# it into NOTHING_TO_REPORT would hide the exact case this file exists for.
|
|
367
|
+
if args.as_json:
|
|
368
|
+
print(json.dumps(s, indent=2, sort_keys=True))
|
|
369
|
+
else:
|
|
370
|
+
print(render(s))
|
|
371
|
+
return OK
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
if __name__ == "__main__":
|
|
375
|
+
sys.exit(main())
|
package/tools/tool-index.py
CHANGED
|
@@ -43,6 +43,24 @@ _ROOT = os.path.dirname(_HERE)
|
|
|
43
43
|
NO_DESC = "(no description)"
|
|
44
44
|
|
|
45
45
|
|
|
46
|
+
class _Parser(argparse.ArgumentParser):
|
|
47
|
+
"""Usage errors exit 64, not argparse's default 2.
|
|
48
|
+
|
|
49
|
+
In this repo's convention 2 means "could NOT be checked" -- a real
|
|
50
|
+
answer about the subject. A mistyped flag is not that: it is an error
|
|
51
|
+
about the INVOCATION, and nothing about the subject was examined. The
|
|
52
|
+
two call for opposite responses, since retrying cannot fix a typo.
|
|
53
|
+
|
|
54
|
+
argparse exits 2 for every usage error unless this is overridden, so
|
|
55
|
+
every tool needs it. tests/test_tool_exit_contract.py asserts it.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
def error(self, message):
|
|
59
|
+
self.print_usage(sys.stderr)
|
|
60
|
+
sys.stderr.write("%s: error: %s\n" % (self.prog, message))
|
|
61
|
+
raise SystemExit(64)
|
|
62
|
+
|
|
63
|
+
|
|
46
64
|
def _first_sentence(text):
|
|
47
65
|
"""First line of a docstring or header block, trimmed.
|
|
48
66
|
|
|
@@ -162,7 +180,7 @@ def render(rows):
|
|
|
162
180
|
|
|
163
181
|
|
|
164
182
|
def main(argv=None):
|
|
165
|
-
ap =
|
|
183
|
+
ap = _Parser(description="List loki's bundled tools.")
|
|
166
184
|
ap.add_argument("--json", action="store_true")
|
|
167
185
|
ap.add_argument("--tools-dir", default=None)
|
|
168
186
|
args = ap.parse_args(argv)
|
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""What does verification COST, and how often does it change the answer?
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. tools/gate-log.py made verdicts durable and tools/gate-trend.py
|
|
5
|
+
says whether the gate is improving or going blind. Both answer questions about
|
|
6
|
+
the gate's OUTPUT. Neither answers what the gate costs to run, and that is now
|
|
7
|
+
the load-bearing question: verification here is adaptive scaffolding whose
|
|
8
|
+
mandated trajectory is toward near-zero overhead, applied selectively where
|
|
9
|
+
calibrated confidence is low and bypassed where it is high.
|
|
10
|
+
|
|
11
|
+
Neither half of that mandate can be validated without a baseline. "Make it
|
|
12
|
+
cheaper" needs a current cost. "Bypass it where confidence is high" needs to
|
|
13
|
+
know how often running it changed an outcome at all -- because a gate that never
|
|
14
|
+
changes an outcome is pure tax, and a gate that frequently does is buying
|
|
15
|
+
something real. This file computes both from the log that already exists.
|
|
16
|
+
|
|
17
|
+
WHAT THIS IS NOT. It is not a gate. It never blocks, never returns a
|
|
18
|
+
merge-blocking exit code, and adds no new artifact format. It reads a log
|
|
19
|
+
written by other tools and reports two numbers. Verification machinery is
|
|
20
|
+
exactly what this must not become; it exists to make the existing machinery
|
|
21
|
+
measurable so it can be shrunk or skipped.
|
|
22
|
+
|
|
23
|
+
THE HONESTY RULE THIS INHERITS. A missing duration is not a duration of zero.
|
|
24
|
+
Entries that carry no timing are counted and reported as UNTIMED rather than
|
|
25
|
+
folded into the average as zeros, because averaging an absent measurement toward
|
|
26
|
+
zero manufactures exactly the "verification is already free" conclusion this
|
|
27
|
+
tool exists to test. An empty log reports NO DATA, never "0.0s, looks cheap".
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
import argparse
|
|
31
|
+
import json
|
|
32
|
+
import os
|
|
33
|
+
import sys
|
|
34
|
+
|
|
35
|
+
# The repo-wide exit convention, enforced by tests/test_tool_exit_contract.py.
|
|
36
|
+
# COULD_NOT_CHECK was 4 here, which is not in the convention at all: a CI
|
|
37
|
+
# caller branching on 2 would have read "could not check" as an unrecognised
|
|
38
|
+
# code and fallen through to its default arm.
|
|
39
|
+
OK = 0
|
|
40
|
+
COULD_NOT_CHECK = 2
|
|
41
|
+
NO_DATA = 3
|
|
42
|
+
USAGE_ERROR = 64
|
|
43
|
+
INPUT_MISSING = 66
|
|
44
|
+
|
|
45
|
+
# Keys a duration may plausibly appear under. The verify evidence doc carries
|
|
46
|
+
# run_started_at/run_completed_at; a future writer may record an explicit
|
|
47
|
+
# duration. Accept either rather than mandating one shape.
|
|
48
|
+
DURATION_KEYS = ("duration_s", "duration_seconds", "elapsed_s")
|
|
49
|
+
START_KEYS = ("run_started_at", "started_at")
|
|
50
|
+
END_KEYS = ("run_completed_at", "completed_at", "ended_at")
|
|
51
|
+
|
|
52
|
+
# A verdict that blocks. Anything here means the gate CHANGED the outcome:
|
|
53
|
+
# without it the change would have proceeded.
|
|
54
|
+
BLOCKING = {"BLOCKED", "FAIL", "FAILED", "BLOCK"}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _parse_iso(value):
|
|
58
|
+
"""Return epoch seconds, or None. None means unknown, never zero."""
|
|
59
|
+
if not isinstance(value, str) or not value:
|
|
60
|
+
return None
|
|
61
|
+
import datetime
|
|
62
|
+
text = value.strip().replace("Z", "+00:00")
|
|
63
|
+
try:
|
|
64
|
+
return datetime.datetime.fromisoformat(text).timestamp()
|
|
65
|
+
except (ValueError, TypeError):
|
|
66
|
+
return None
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _first(mapping, keys):
|
|
70
|
+
for key in keys:
|
|
71
|
+
if key in mapping and mapping[key] is not None:
|
|
72
|
+
return mapping[key]
|
|
73
|
+
return None
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def duration_of(entry):
|
|
77
|
+
"""Seconds this verification took, or None if the entry never recorded it.
|
|
78
|
+
|
|
79
|
+
Checked in two ways because two writers exist: an explicit duration field,
|
|
80
|
+
or a start/end pair as the verify evidence document carries. A negative or
|
|
81
|
+
non-numeric result is discarded as unknown rather than trusted.
|
|
82
|
+
"""
|
|
83
|
+
verdict = entry.get("verdict")
|
|
84
|
+
sources = [entry]
|
|
85
|
+
if isinstance(verdict, dict):
|
|
86
|
+
sources.append(verdict)
|
|
87
|
+
produced = verdict.get("produced_by")
|
|
88
|
+
if isinstance(produced, dict):
|
|
89
|
+
sources.append(produced)
|
|
90
|
+
|
|
91
|
+
for src in sources:
|
|
92
|
+
if not isinstance(src, dict):
|
|
93
|
+
continue
|
|
94
|
+
explicit = _first(src, DURATION_KEYS)
|
|
95
|
+
if isinstance(explicit, (int, float)) and explicit >= 0:
|
|
96
|
+
return float(explicit)
|
|
97
|
+
|
|
98
|
+
for src in sources:
|
|
99
|
+
if not isinstance(src, dict):
|
|
100
|
+
continue
|
|
101
|
+
start = _parse_iso(_first(src, START_KEYS))
|
|
102
|
+
end = _parse_iso(_first(src, END_KEYS))
|
|
103
|
+
if start is not None and end is not None and end >= start:
|
|
104
|
+
return end - start
|
|
105
|
+
return None
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def verdict_string(entry):
|
|
109
|
+
"""The verdict label, wherever this writer happened to put it."""
|
|
110
|
+
verdict = entry.get("verdict")
|
|
111
|
+
if isinstance(verdict, str):
|
|
112
|
+
return verdict.upper()
|
|
113
|
+
if isinstance(verdict, dict):
|
|
114
|
+
for key in ("verdict", "status", "result"):
|
|
115
|
+
value = verdict.get(key)
|
|
116
|
+
if isinstance(value, str):
|
|
117
|
+
return value.upper()
|
|
118
|
+
category = entry.get("category")
|
|
119
|
+
return category.upper() if isinstance(category, str) else ""
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def changed_outcome(entry):
|
|
123
|
+
"""Did running the gate change what would otherwise have happened?
|
|
124
|
+
|
|
125
|
+
Only a blocking verdict changes an outcome. A pass means the change would
|
|
126
|
+
have shipped either way, so the time spent verifying bought information but
|
|
127
|
+
not a different result. This is the hit rate that justifies the tax.
|
|
128
|
+
"""
|
|
129
|
+
return verdict_string(entry) in BLOCKING
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def read_log(path):
|
|
133
|
+
"""Every line. Corrupt lines counted, never silently dropped."""
|
|
134
|
+
entries, corrupt = [], 0
|
|
135
|
+
with open(path, "r", encoding="utf-8") as fh:
|
|
136
|
+
for line in fh:
|
|
137
|
+
line = line.strip()
|
|
138
|
+
if not line:
|
|
139
|
+
continue
|
|
140
|
+
try:
|
|
141
|
+
obj = json.loads(line)
|
|
142
|
+
except ValueError:
|
|
143
|
+
corrupt += 1
|
|
144
|
+
continue
|
|
145
|
+
if isinstance(obj, dict):
|
|
146
|
+
entries.append(obj)
|
|
147
|
+
else:
|
|
148
|
+
corrupt += 1
|
|
149
|
+
return entries, corrupt
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def summarize(entries, corrupt):
|
|
153
|
+
"""Two numbers and the honesty around them.
|
|
154
|
+
|
|
155
|
+
total_s and mean_s are computed over TIMED entries only, and untimed is
|
|
156
|
+
reported alongside so a reader can see how much of the log the average
|
|
157
|
+
actually covers. An average over 2 of 200 runs is not a baseline.
|
|
158
|
+
"""
|
|
159
|
+
timed = [d for d in (duration_of(e) for e in entries) if d is not None]
|
|
160
|
+
changed = sum(1 for e in entries if changed_outcome(e))
|
|
161
|
+
total = len(entries)
|
|
162
|
+
return {
|
|
163
|
+
"runs": total,
|
|
164
|
+
"timed": len(timed),
|
|
165
|
+
"untimed": total - len(timed),
|
|
166
|
+
"corrupt": corrupt,
|
|
167
|
+
"total_s": round(sum(timed), 3) if timed else None,
|
|
168
|
+
"mean_s": round(sum(timed) / len(timed), 3) if timed else None,
|
|
169
|
+
"max_s": round(max(timed), 3) if timed else None,
|
|
170
|
+
"changed_outcome": changed,
|
|
171
|
+
"hit_rate": round(changed / total, 4) if total else None,
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def render(summary):
|
|
176
|
+
lines = ["VERIFICATION TAX", ""]
|
|
177
|
+
runs = summary["runs"]
|
|
178
|
+
lines.append(" runs logged %d" % runs)
|
|
179
|
+
|
|
180
|
+
if summary["timed"]:
|
|
181
|
+
lines.append(" timed %d of %d" % (summary["timed"], runs))
|
|
182
|
+
lines.append(" total time %.1fs" % summary["total_s"])
|
|
183
|
+
lines.append(" mean per run %.2fs" % summary["mean_s"])
|
|
184
|
+
lines.append(" slowest run %.2fs" % summary["max_s"])
|
|
185
|
+
else:
|
|
186
|
+
lines.append(" timed 0 of %d" % runs)
|
|
187
|
+
lines.append(" total time UNKNOWN (no entry carried timing)")
|
|
188
|
+
lines.append(" mean per run UNKNOWN")
|
|
189
|
+
|
|
190
|
+
if summary["untimed"]:
|
|
191
|
+
lines.append(" UNTIMED %d run(s) carry no duration; excluded"
|
|
192
|
+
% summary["untimed"])
|
|
193
|
+
lines.append(" from the averages above, NOT counted as 0s")
|
|
194
|
+
|
|
195
|
+
lines.append("")
|
|
196
|
+
if summary["hit_rate"] is None:
|
|
197
|
+
lines.append(" outcome changed UNKNOWN")
|
|
198
|
+
else:
|
|
199
|
+
lines.append(" outcome changed %d of %d runs (%.1f%%)" % (
|
|
200
|
+
summary["changed_outcome"], runs, summary["hit_rate"] * 100))
|
|
201
|
+
lines.append(" a pass would have shipped anyway; only a")
|
|
202
|
+
lines.append(" block changed what happened")
|
|
203
|
+
|
|
204
|
+
if summary["corrupt"]:
|
|
205
|
+
lines.append("")
|
|
206
|
+
lines.append(" CORRUPT %d unreadable line(s), counted not dropped"
|
|
207
|
+
% summary["corrupt"])
|
|
208
|
+
|
|
209
|
+
lines.append("")
|
|
210
|
+
if summary["timed"] and summary["hit_rate"] == 0:
|
|
211
|
+
lines.append(" READ: every logged run passed. On this sample the tax bought")
|
|
212
|
+
lines.append(" information but changed no outcome -- the case for applying it")
|
|
213
|
+
lines.append(" selectively rather than universally.")
|
|
214
|
+
elif not summary["timed"]:
|
|
215
|
+
lines.append(" READ: cost is UNMEASURED. No claim that verification is cheap")
|
|
216
|
+
lines.append(" or expensive can be made from this log until writers record")
|
|
217
|
+
lines.append(" timing. Absence of a number is not a small number.")
|
|
218
|
+
return "\n".join(lines)
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
class _Parser(argparse.ArgumentParser):
|
|
222
|
+
"""Usage errors exit 64, not argparse's default 2.
|
|
223
|
+
|
|
224
|
+
In this repo's convention 2 means "could NOT be checked" -- a real
|
|
225
|
+
answer about the subject. A mistyped flag is not that; it is an error
|
|
226
|
+
about the INVOCATION, and reporting it as 2 tells a CI caller that the
|
|
227
|
+
axis was evaluated and found unevaluable, when in fact nothing was
|
|
228
|
+
examined at all.
|
|
229
|
+
|
|
230
|
+
argparse defaults to 2 for every usage error, so a tool that does not
|
|
231
|
+
override this silently emits the wrong code. This one did: `--bogus`
|
|
232
|
+
exited 2 and a missing positional exited 2.
|
|
233
|
+
"""
|
|
234
|
+
|
|
235
|
+
def error(self, message):
|
|
236
|
+
self.print_usage(sys.stderr)
|
|
237
|
+
sys.stderr.write("%s: error: %s\n" % (self.prog, message))
|
|
238
|
+
raise SystemExit(USAGE_ERROR)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def main(argv=None):
|
|
242
|
+
parser = _Parser(
|
|
243
|
+
description="Measure what verification costs and how often it "
|
|
244
|
+
"changed an outcome. Reports only; never blocks.")
|
|
245
|
+
parser.add_argument("log", help="JSONL log (e.g. from tools/gate-log.py)")
|
|
246
|
+
parser.add_argument("--json", action="store_true",
|
|
247
|
+
help="emit the summary as JSON")
|
|
248
|
+
args = parser.parse_args(argv)
|
|
249
|
+
|
|
250
|
+
if not os.path.exists(args.log):
|
|
251
|
+
# 66, not COULD_NOT_CHECK. "The path you named does not exist" is a
|
|
252
|
+
# fact about the INPUT; "I could not evaluate" is a fact about the
|
|
253
|
+
# subject. A caller retrying on 2 would retry forever against a
|
|
254
|
+
# typo, while 66 says the argument itself is wrong.
|
|
255
|
+
sys.stderr.write("verification-tax: no such log: %s\n" % args.log)
|
|
256
|
+
return INPUT_MISSING
|
|
257
|
+
try:
|
|
258
|
+
entries, corrupt = read_log(args.log)
|
|
259
|
+
except OSError as exc:
|
|
260
|
+
sys.stderr.write("verification-tax: could not read %s: %s\n"
|
|
261
|
+
% (args.log, exc))
|
|
262
|
+
return COULD_NOT_CHECK
|
|
263
|
+
|
|
264
|
+
if not entries and not corrupt:
|
|
265
|
+
sys.stderr.write("verification-tax: log is empty; NO DATA\n")
|
|
266
|
+
return NO_DATA
|
|
267
|
+
|
|
268
|
+
summary = summarize(entries, corrupt)
|
|
269
|
+
if args.json:
|
|
270
|
+
sys.stdout.write(json.dumps(summary, sort_keys=True, indent=2) + "\n")
|
|
271
|
+
else:
|
|
272
|
+
sys.stdout.write(render(summary) + "\n")
|
|
273
|
+
return OK
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
if __name__ == "__main__":
|
|
277
|
+
sys.exit(main())
|