loki-mode 9.8.1 → 9.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -14
- package/SKILL.md +3 -2
- package/VERSION +1 -1
- package/autonomy/loki +122 -1
- package/autonomy/run.sh +49 -2
- package/dashboard/__init__.py +1 -1
- package/dashboard/api_evidence.py +411 -0
- package/dashboard/api_operator.py +283 -0
- package/dashboard/api_phases.py +262 -0
- package/dashboard/api_releases.py +242 -0
- package/dashboard/api_runs.py +477 -0
- package/dashboard/api_tests.py +444 -0
- package/dashboard/api_v2.py +47 -1
- package/dashboard/server.py +54 -0
- package/dashboard/static/index.html +246 -135
- package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
- package/docs/CAPABILITY-BACKLOG.md +53 -0
- package/docs/COMPARISON.md +2 -2
- package/docs/COMPETITIVE-ANALYSIS.md +1 -1
- package/docs/COMPETITIVE-SCORECARD.md +422 -0
- package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
- package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
- package/docs/DEMOS.md +21 -23
- package/docs/HANDOFF-2026-08-03.md +439 -0
- package/docs/INSTALLATION.md +17 -10
- package/docs/OUTCOME-FRONTIER.md +536 -0
- package/docs/PROMPT-ABLATION-RESULT.md +97 -0
- package/docs/TOOLS.md +800 -0
- package/docs/alternative-installations.md +2 -3
- package/docs/audit-logging.md +44 -35
- package/docs/authentication.md +13 -2
- package/docs/authorization.md +87 -81
- package/docs/git-workflow.md +6 -3
- package/docs/metrics.md +15 -16
- package/docs/network-security.md +16 -13
- package/docs/openclaw-integration.md +36 -556
- package/docs/show-hn-post.md +2 -2
- package/docs/siem-integration.md +39 -36
- package/loki-ts/dist/loki.js +18 -18
- package/mcp/__init__.py +1 -1
- package/package.json +1 -1
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
- package/references/confidence-routing.md +18 -1
- package/references/invariant-checks.md +13 -8
- package/references/magic-rarv-integration.md +0 -1
- package/references/multi-provider.md +27 -5
- package/skills/healing.md +4 -2
- package/tools/audit-docs.py +488 -0
- package/tools/baseline-pin.py +19 -1
- package/tools/calibration-audit.py +523 -0
- package/tools/ci-gate.py +19 -1
- package/tools/cost-forecast.py +344 -0
- package/tools/cost-guard.py +19 -1
- package/tools/cost-history.py +19 -1
- package/tools/cost-per-outcome.py +394 -0
- package/tools/estimate-run.py +19 -1
- package/tools/evidence-freshness.py +307 -0
- package/tools/gate-init.py +19 -1
- package/tools/gate-report.py +19 -1
- package/tools/gate-simulate.py +570 -0
- package/tools/gate-trend.py +354 -0
- package/tools/model-advisor.py +52 -1
- package/tools/policy-load.py +19 -1
- package/tools/prompt-cost.py +363 -0
- package/tools/prompt-diff.py +448 -0
- package/tools/prompt-lint.py +448 -0
- package/tools/receipt-bundle.py +72 -2
- package/tools/receipt-diff.py +19 -1
- package/tools/receipt-find.py +19 -1
- package/tools/receipt-stats.py +380 -0
- package/tools/receipt-timeline.py +478 -0
- package/tools/receipt-verify-batch.py +291 -0
- package/tools/run-replay.py +19 -1
- package/tools/signing-status.py +19 -1
- package/tools/token-guard.py +19 -1
- package/tools/token-tax.py +375 -0
- package/tools/tool-index.py +19 -1
- package/tools/verification-tax.py +277 -0
- package/tools/verify-chain.py +361 -0
|
@@ -0,0 +1,394 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""What did a VERIFIED outcome cost, versus a failed one?
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. receipt-stats.py answers "what IS this archive": how many
|
|
5
|
+
receipts, how many verified, what the whole thing cost. It reports ONE cost
|
|
6
|
+
figure across every receipt regardless of verdict. That answers the accounting
|
|
7
|
+
question and not the one a team actually asks when deciding whether the trust
|
|
8
|
+
layer pays for itself:
|
|
9
|
+
|
|
10
|
+
Is a run that ends VERIFIED cheaper or more expensive than one that ends
|
|
11
|
+
FAILED, and what are we paying for the runs we cannot verify at all?
|
|
12
|
+
|
|
13
|
+
Those three numbers are a different report from one blended average, and the
|
|
14
|
+
blended average actively hides the answer. An archive whose failures cost 5x
|
|
15
|
+
its successes and an archive where every run costs the same produce the SAME
|
|
16
|
+
total and the SAME overall median. Splitting by verdict class is the entire
|
|
17
|
+
product here; everything else is bookkeeping to keep the split honest.
|
|
18
|
+
|
|
19
|
+
THE HONESTY RULES. Each is a specific way a per-class average can claim more
|
|
20
|
+
than it measured, and each is sharper here than in a single-figure summary,
|
|
21
|
+
because a class is a SMALL sample. One fake zero folded into a total of forty
|
|
22
|
+
receipts moves it slightly. One fake zero folded into the three receipts that
|
|
23
|
+
happened to FAIL moves that class's average by a third.
|
|
24
|
+
|
|
25
|
+
1. AN UNMEASURED RECEIPT IS EXCLUDED FROM EVERY AVERAGE, AND THE EXCLUSION IS
|
|
26
|
+
COUNTED PER CLASS. A receipt that never recorded cost did not cost $0. It
|
|
27
|
+
is dropped from the mean and the total for its class, and the count of what
|
|
28
|
+
was dropped is reported ON THE SAME ROW -- an average over 2 of 11 FAILED
|
|
29
|
+
receipts is not wrong, but shown without that ratio it reads as the cost of
|
|
30
|
+
failing, which it is not.
|
|
31
|
+
|
|
32
|
+
The predicate is record_is_measured() in autonomy/lib/efficiency_cost.py,
|
|
33
|
+
reached through receipt-diff.measured_cost() (re-exported by
|
|
34
|
+
receipt-bundle.py), which already maps the receipt's `cost.usd` onto the
|
|
35
|
+
per-iteration `cost_usd` key that predicate reads. Neither half is restated
|
|
36
|
+
here; a second copy is exactly how this rule drifted across four surfaces
|
|
37
|
+
before (v8.51.0-v8.54.0).
|
|
38
|
+
|
|
39
|
+
Measured is necessary but NOT sufficient for a dollar figure:
|
|
40
|
+
record_is_measured is true when ANY of five fields is non-zero, so a receipt
|
|
41
|
+
carrying tokens and a null `usd` is honestly measured and still has no
|
|
42
|
+
amount. Both are required -- see _usd().
|
|
43
|
+
|
|
44
|
+
2. A CLASS WITH NO MEASURED RECEIPTS READS UNKNOWN, AND STILL GETS A ROW.
|
|
45
|
+
Two failures here, and they point opposite ways:
|
|
46
|
+
|
|
47
|
+
- Rendering that class as "$0.00" says failing runs are free. They are
|
|
48
|
+
the most expensive thing in the archive; we simply did not measure them.
|
|
49
|
+
- OMITTING the row says no receipt landed in that class at all. "Three
|
|
50
|
+
runs FAILED and none recorded a cost" and "nothing FAILED" are opposite
|
|
51
|
+
facts, and dropping the row turns the first into the second silently.
|
|
52
|
+
So every class in _CLASSES is rendered on every run, with its receipt
|
|
53
|
+
count, whether or not anything in it measured.
|
|
54
|
+
|
|
55
|
+
The MEASURED COUNT decides UNKNOWN, never the total. `if not total` would
|
|
56
|
+
erase a class whose receipts each genuinely measured $0.0000 -- rule 1
|
|
57
|
+
failing in the opposite direction, and just as untrue.
|
|
58
|
+
|
|
59
|
+
3. A MEASURED ZERO SURVIVES AS ZERO. A run that genuinely cost $0.00 is an
|
|
60
|
+
observation, not an absence. `is None` throughout, never truthiness. The
|
|
61
|
+
two ways to lie about cost are to invent a number and to discard one.
|
|
62
|
+
|
|
63
|
+
4. A MALFORMED RECEIPT IS COUNTED AND NAMED. Kept separate from UNVERIFIABLE,
|
|
64
|
+
for the same reason receipt-stats.py keeps it separate: "this file is not
|
|
65
|
+
JSON" and "this receipt cannot be re-derived here" have different fixes.
|
|
66
|
+
Folding the first into the second inflates the UNVERIFIABLE class with
|
|
67
|
+
files that were never receipts.
|
|
68
|
+
|
|
69
|
+
5. ZERO RECEIPTS IS NOT A CLEAN ARCHIVE. Exit 3, and say so in words. It is
|
|
70
|
+
most often the wrong directory.
|
|
71
|
+
|
|
72
|
+
WHAT THIS IS NOT. An ADVISOR, not a gate, and the distinction is pinned by
|
|
73
|
+
tests/test_tool_exit_contract.py. A FAILED receipt in the archive does not make
|
|
74
|
+
this exit 1 -- receipt-bundle.py is the gate that refuses to merge on that, and
|
|
75
|
+
re-deriving the weakest-link rule here would be a second copy of a verdict
|
|
76
|
+
predicate. An archive where NOTHING measured cost still exits 0: "every receipt
|
|
77
|
+
is readable, and none recorded a cost" is a complete honest answer, and the
|
|
78
|
+
output says UNKNOWN in words. Forcing it non-zero would train operators to
|
|
79
|
+
ignore a tool that is working correctly.
|
|
80
|
+
|
|
81
|
+
Usage:
|
|
82
|
+
tools/cost-per-outcome.py [workspace] [--repo-dir DIR] [--json]
|
|
83
|
+
|
|
84
|
+
Exit codes:
|
|
85
|
+
0 receipts were found and split by verdict class
|
|
86
|
+
3 the workspace exists but holds no receipts -- nothing to split
|
|
87
|
+
64 usage error (unknown flag, bad argument)
|
|
88
|
+
66 the workspace path does not exist
|
|
89
|
+
"""
|
|
90
|
+
|
|
91
|
+
import argparse
|
|
92
|
+
import importlib.util
|
|
93
|
+
import json
|
|
94
|
+
import os
|
|
95
|
+
import pathlib
|
|
96
|
+
import sys
|
|
97
|
+
|
|
98
|
+
# A stale .pyc for a hyphenated module loaded by path makes a mutation probe
|
|
99
|
+
# report a FALSE survival: the probe edits the source, the loader serves old
|
|
100
|
+
# bytecode, invalidation is mtime+size and the restore is byte-identical.
|
|
101
|
+
# Must be set before any loader below runs.
|
|
102
|
+
sys.dont_write_bytecode = True
|
|
103
|
+
|
|
104
|
+
_ROOT = pathlib.Path(__file__).resolve().parents[1]
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _load(name, path):
|
|
108
|
+
spec = importlib.util.spec_from_file_location(name, path)
|
|
109
|
+
mod = importlib.util.module_from_spec(spec)
|
|
110
|
+
spec.loader.exec_module(mod)
|
|
111
|
+
return mod
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
# receipt_state() wraps verify() from autonomy/lib/proof-verify.py and already
|
|
115
|
+
# keeps VERIFIED / FAILED / UNVERIFIABLE apart with the `is` comparisons that
|
|
116
|
+
# make that correct (gpg_ok is the truthy string "n/a"; diff_drift None means
|
|
117
|
+
# unverifiable, not clean). measured_cost() reuses record_is_measured() AND
|
|
118
|
+
# maps cost.usd -> cost_usd. Both imported, neither restated.
|
|
119
|
+
_rb = _load("receipt_bundle", _ROOT / "tools" / "receipt-bundle.py")
|
|
120
|
+
receipt_state = _rb.receipt_state
|
|
121
|
+
measured_cost = _rb.measured_cost
|
|
122
|
+
|
|
123
|
+
VERIFIED = _rb.VERIFIED
|
|
124
|
+
FAILED = _rb.FAILED
|
|
125
|
+
UNVERIFIABLE = _rb.UNVERIFIABLE
|
|
126
|
+
MALFORMED = "MALFORMED"
|
|
127
|
+
|
|
128
|
+
# Rendered in this order on EVERY run, present or not (rule 2). MALFORMED is
|
|
129
|
+
# last because it is a census bucket, not one of receipt_state's verdicts.
|
|
130
|
+
_CLASSES = (VERIFIED, FAILED, UNVERIFIABLE, MALFORMED)
|
|
131
|
+
|
|
132
|
+
OK, NOTHING_TO_CHECK, USAGE, NO_INPUT = 0, 3, 64, 66
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _usd(proof):
|
|
136
|
+
"""The receipt's cost in dollars, or None when there is no such number.
|
|
137
|
+
|
|
138
|
+
None means UNMEASURED. Every caller must EXCLUDE rather than substitute:
|
|
139
|
+
returning 0.0 here is rule 1, and it would drag its class's average toward
|
|
140
|
+
zero using a value nobody observed.
|
|
141
|
+
|
|
142
|
+
Two gates, both required. measured_cost() returning None means the receipt
|
|
143
|
+
recorded no measurement at all. A non-None record can still carry
|
|
144
|
+
cost_usd=None -- tokens moved, dollars were never priced -- and defaulting
|
|
145
|
+
that to 0.0 would invent a measurement out of a receipt proving only that
|
|
146
|
+
work happened.
|
|
147
|
+
"""
|
|
148
|
+
rec = measured_cost(proof)
|
|
149
|
+
if rec is None:
|
|
150
|
+
return None
|
|
151
|
+
return rec.get("cost_usd")
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def find_receipts(workspace):
|
|
155
|
+
"""Every proof.json under the workspace, sorted for a stable report.
|
|
156
|
+
|
|
157
|
+
ponytail: same rglob as receipt-bundle.find_receipts, so a receipt archived
|
|
158
|
+
outside .loki/proofs/ is still counted. Not imported, because that one
|
|
159
|
+
assumes the directory exists and this tool must tell a MISSING workspace
|
|
160
|
+
(exit 66) apart from an EMPTY one (exit 3).
|
|
161
|
+
"""
|
|
162
|
+
root = pathlib.Path(workspace)
|
|
163
|
+
if not root.is_dir():
|
|
164
|
+
return []
|
|
165
|
+
return sorted(p for p in root.rglob("proof.json") if p.is_file())
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _class_block(costs, receipt_n):
|
|
169
|
+
"""One verdict class's cost figures, carrying the basis they came from.
|
|
170
|
+
|
|
171
|
+
`len(costs)`, never `sum(costs)`, decides UNKNOWN. Rules 2 and 3 in one
|
|
172
|
+
line: a class of three receipts that each genuinely measured $0.0000 has a
|
|
173
|
+
real total of 0.0 and a real mean of 0.0, and `if not total` would report
|
|
174
|
+
both as UNKNOWN. Erasing a measurement is the same dishonesty as inventing
|
|
175
|
+
one, pointed the other way.
|
|
176
|
+
"""
|
|
177
|
+
measured_n = len(costs)
|
|
178
|
+
return {
|
|
179
|
+
"receipts": receipt_n,
|
|
180
|
+
"measured_receipts": measured_n,
|
|
181
|
+
"unmeasured_receipts": receipt_n - measured_n,
|
|
182
|
+
"total_usd": sum(costs) if measured_n else None,
|
|
183
|
+
"mean_usd": (sum(costs) / measured_n) if measured_n else None,
|
|
184
|
+
"min_usd": min(costs) if measured_n else None,
|
|
185
|
+
"max_usd": max(costs) if measured_n else None,
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def cost_per_outcome(workspace, repo_dir="."):
|
|
190
|
+
"""Join every receipt to its measured cost, split by verdict class.
|
|
191
|
+
|
|
192
|
+
Pure: no writes, no network, no subprocess beyond what verify() already
|
|
193
|
+
runs to re-derive a diff.
|
|
194
|
+
"""
|
|
195
|
+
receipts = []
|
|
196
|
+
malformed = []
|
|
197
|
+
by_class = {c: [] for c in _CLASSES} # class -> [measured usd, ...]
|
|
198
|
+
counts = {c: 0 for c in _CLASSES}
|
|
199
|
+
|
|
200
|
+
for path in find_receipts(workspace):
|
|
201
|
+
try:
|
|
202
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
203
|
+
proof = json.load(f)
|
|
204
|
+
if not isinstance(proof, dict):
|
|
205
|
+
raise ValueError("receipt is not a JSON object")
|
|
206
|
+
except Exception as exc:
|
|
207
|
+
# Rule 4: counted and NAMED in its own bucket, never dropped and
|
|
208
|
+
# never folded into UNVERIFIABLE. Same keys as every other row so
|
|
209
|
+
# a consumer reading receipts[i]["cost_usd"] does not KeyError on
|
|
210
|
+
# exactly the rows this rule exists to surface.
|
|
211
|
+
malformed.append({"path": str(path), "reason": str(exc)})
|
|
212
|
+
counts[MALFORMED] += 1
|
|
213
|
+
receipts.append({
|
|
214
|
+
"path": str(path), "class": MALFORMED, "reason": str(exc),
|
|
215
|
+
"cost_usd": None,
|
|
216
|
+
})
|
|
217
|
+
continue
|
|
218
|
+
|
|
219
|
+
state, reason = receipt_state(path, repo_dir)
|
|
220
|
+
usd = _usd(proof)
|
|
221
|
+
|
|
222
|
+
counts[state] += 1
|
|
223
|
+
if usd is not None:
|
|
224
|
+
# `is not None`, never truthiness: a measured 0.0 is an
|
|
225
|
+
# observation and must enter the average as a real data point.
|
|
226
|
+
by_class[state].append(usd)
|
|
227
|
+
|
|
228
|
+
receipts.append({
|
|
229
|
+
"path": str(path), "class": state, "reason": reason,
|
|
230
|
+
"cost_usd": usd,
|
|
231
|
+
})
|
|
232
|
+
|
|
233
|
+
classes = {c: _class_block(by_class[c], counts[c]) for c in _CLASSES}
|
|
234
|
+
|
|
235
|
+
return {
|
|
236
|
+
"report": "loki-cost-per-outcome/v1",
|
|
237
|
+
"workspace": os.path.abspath(str(workspace)),
|
|
238
|
+
"checked_from": os.path.abspath(repo_dir),
|
|
239
|
+
"receipts": receipts,
|
|
240
|
+
"receipt_count": len(receipts),
|
|
241
|
+
"classes": classes,
|
|
242
|
+
"malformed": malformed,
|
|
243
|
+
"malformed_count": len(malformed),
|
|
244
|
+
"comparison": _comparison(classes),
|
|
245
|
+
"summary": _summary(len(receipts), classes),
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _comparison(classes):
|
|
250
|
+
"""The headline this tool exists to produce, or why it cannot be produced.
|
|
251
|
+
|
|
252
|
+
None on either side means UNKNOWN. A ratio computed against an invented
|
|
253
|
+
zero is the exact lie this file is built to refuse, and a ratio computed
|
|
254
|
+
against a genuine measured 0.0 is a division by zero -- so both are
|
|
255
|
+
refused, in words that say which case it was.
|
|
256
|
+
"""
|
|
257
|
+
v = classes[VERIFIED]["mean_usd"]
|
|
258
|
+
f = classes[FAILED]["mean_usd"]
|
|
259
|
+
if v is None or f is None:
|
|
260
|
+
missing = [n for n, m in ((VERIFIED, v), (FAILED, f)) if m is None]
|
|
261
|
+
return {
|
|
262
|
+
"ratio_failed_over_verified": None,
|
|
263
|
+
"note": ("UNKNOWN -- no measured cost for %s, so there is no "
|
|
264
|
+
"basis to compare what a failure costs against what a "
|
|
265
|
+
"success costs. Not a ratio of 1.0, and not $0.00."
|
|
266
|
+
% " and ".join(missing)),
|
|
267
|
+
}
|
|
268
|
+
if v == 0:
|
|
269
|
+
return {
|
|
270
|
+
"ratio_failed_over_verified": None,
|
|
271
|
+
"note": ("UNKNOWN -- measured VERIFIED mean is exactly $0.0000, "
|
|
272
|
+
"so a ratio is undefined. That zero is real data, not a "
|
|
273
|
+
"missing measurement."),
|
|
274
|
+
}
|
|
275
|
+
return {
|
|
276
|
+
"ratio_failed_over_verified": f / v,
|
|
277
|
+
"note": ("a FAILED outcome costs %.2fx a VERIFIED one on the measured "
|
|
278
|
+
"receipts in each class" % (f / v)),
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def _amount(value):
|
|
283
|
+
"""A dollar figure, or the honest dash. Never "$0.0000" for an absence."""
|
|
284
|
+
return "-" if value is None else "$%.4f" % value
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def _class_line(name, block):
|
|
288
|
+
"""One rendered row. Emitted for every class whether or not it measured.
|
|
289
|
+
|
|
290
|
+
Three distinct states, and collapsing any two is a claim the data does not
|
|
291
|
+
support. "No run ended this way", "runs ended this way and none was
|
|
292
|
+
priced", and "here is what they cost" are three different findings, and the
|
|
293
|
+
first two are the pair a reader most often has to act on.
|
|
294
|
+
"""
|
|
295
|
+
if block["receipts"] == 0:
|
|
296
|
+
# Still UNKNOWN, with a DIFFERENT reason clause. "Nothing landed here"
|
|
297
|
+
# is a fact about the archive; "we could not price what landed here" is
|
|
298
|
+
# a fact about the instrumentation, and they have different fixes -- so
|
|
299
|
+
# the clause differs. But the leading word does not: _summary() renders
|
|
300
|
+
# this same class from `mean_usd is None`, and if the row said "none"
|
|
301
|
+
# while the summary said UNKNOWN, one report would carry two words for
|
|
302
|
+
# one class. That drift is the thing this file argues against, so the
|
|
303
|
+
# two surfaces are kept on the same token deliberately.
|
|
304
|
+
figures = ("UNKNOWN (0 of 0 measured) -- no receipt ended with this "
|
|
305
|
+
"outcome")
|
|
306
|
+
elif block["measured_receipts"] == 0 or block["mean_usd"] is None:
|
|
307
|
+
# Phrased off the data, never a hardcoded count. The second disjunct is
|
|
308
|
+
# reachable with measured_receipts > 0 if a figure went missing without
|
|
309
|
+
# the count going with it, and a tool arguing that a report must not
|
|
310
|
+
# claim more than it measured cannot ship a sentence able to state a
|
|
311
|
+
# false count.
|
|
312
|
+
figures = ("UNKNOWN (%d of %d measured) -- not $0.00: unmeasured is "
|
|
313
|
+
"not free" % (block["measured_receipts"], block["receipts"]))
|
|
314
|
+
else:
|
|
315
|
+
figures = ("mean %s total %s range %s-%s (%d of %d measured)"
|
|
316
|
+
% (_amount(block["mean_usd"]), _amount(block["total_usd"]),
|
|
317
|
+
_amount(block["min_usd"]), _amount(block["max_usd"]),
|
|
318
|
+
block["measured_receipts"], block["receipts"]))
|
|
319
|
+
if block["unmeasured_receipts"]:
|
|
320
|
+
figures += ("; %d EXCLUDED, cost never measured"
|
|
321
|
+
% block["unmeasured_receipts"])
|
|
322
|
+
return "%-13s %2d receipt(s) %s" % (name, block["receipts"], figures)
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def _summary(n, classes):
|
|
326
|
+
if n == 0:
|
|
327
|
+
# Rule 5, distinct in words from "we split an empty archive".
|
|
328
|
+
return ("NO RECEIPTS -- no proof.json found under this workspace, so "
|
|
329
|
+
"there was nothing to split by outcome. Zero receipts is not "
|
|
330
|
+
"a clean archive; it is most often the wrong directory.")
|
|
331
|
+
parts = []
|
|
332
|
+
for c in _CLASSES:
|
|
333
|
+
b = classes[c]
|
|
334
|
+
parts.append("%s %d receipt(s) %s"
|
|
335
|
+
% (c, b["receipts"],
|
|
336
|
+
"cost UNKNOWN" if b["mean_usd"] is None
|
|
337
|
+
else "mean %s" % _amount(b["mean_usd"])))
|
|
338
|
+
return "%d receipt(s) split by outcome: %s." % (n, "; ".join(parts))
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
class _Parser(argparse.ArgumentParser):
|
|
342
|
+
"""argparse exits 2 on a usage error; here 2 means "could not check".
|
|
343
|
+
|
|
344
|
+
Those are opposite facts. A mistyped flag would otherwise be
|
|
345
|
+
indistinguishable from blind instrumentation, and an operator would go
|
|
346
|
+
looking for a broken measurement that is really a typo. 64 is the usage
|
|
347
|
+
error.
|
|
348
|
+
|
|
349
|
+
error() only. --help routes through exit(), not error(), and overriding
|
|
350
|
+
exit() would break the exit-0 contract test_tool_exit_contract.py asserts
|
|
351
|
+
for every tool's --help.
|
|
352
|
+
"""
|
|
353
|
+
|
|
354
|
+
def error(self, message):
|
|
355
|
+
self.print_usage(sys.stderr)
|
|
356
|
+
sys.stderr.write("%s: error: %s\n" % (self.prog, message))
|
|
357
|
+
raise SystemExit(USAGE)
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
def main(argv=None):
|
|
361
|
+
ap = _Parser(
|
|
362
|
+
description="What a VERIFIED outcome cost, versus a failed one.")
|
|
363
|
+
ap.add_argument("workspace", nargs="?", default=".",
|
|
364
|
+
help="workspace holding the receipts (default: .)")
|
|
365
|
+
ap.add_argument("--repo-dir", default=".",
|
|
366
|
+
help="repository the receipts are verified against "
|
|
367
|
+
"(default: .)")
|
|
368
|
+
ap.add_argument("--json", action="store_true",
|
|
369
|
+
help="emit the full report as JSON")
|
|
370
|
+
args = ap.parse_args(argv)
|
|
371
|
+
|
|
372
|
+
if not os.path.isdir(args.workspace):
|
|
373
|
+
# 66, not 3. "You pointed me at nothing" and "this archive is empty"
|
|
374
|
+
# are different facts, and only one of them is about the archive.
|
|
375
|
+
sys.stderr.write(
|
|
376
|
+
"cost-per-outcome: workspace does not exist: %s\n" % args.workspace)
|
|
377
|
+
return NO_INPUT
|
|
378
|
+
|
|
379
|
+
report = cost_per_outcome(args.workspace, args.repo_dir)
|
|
380
|
+
|
|
381
|
+
if args.json:
|
|
382
|
+
print(json.dumps(report, indent=2))
|
|
383
|
+
else:
|
|
384
|
+
for c in _CLASSES:
|
|
385
|
+
print(_class_line(c, report["classes"][c]))
|
|
386
|
+
print("")
|
|
387
|
+
print("comparison: %s" % report["comparison"]["note"])
|
|
388
|
+
print(report["summary"])
|
|
389
|
+
|
|
390
|
+
return OK if report["receipt_count"] else NOTHING_TO_CHECK
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
if __name__ == "__main__":
|
|
394
|
+
sys.exit(main())
|
package/tools/estimate-run.py
CHANGED
|
@@ -86,6 +86,24 @@ PRICING_PATH = os.path.join(
|
|
|
86
86
|
_REPO_ROOT, "loki-ts", "data", "model-pricing.json")
|
|
87
87
|
|
|
88
88
|
|
|
89
|
+
class _Parser(argparse.ArgumentParser):
|
|
90
|
+
"""Usage errors exit 64, not argparse's default 2.
|
|
91
|
+
|
|
92
|
+
In this repo's convention 2 means "could NOT be checked" -- a real
|
|
93
|
+
answer about the subject. A mistyped flag is not that: it is an error
|
|
94
|
+
about the INVOCATION, and nothing about the subject was examined. The
|
|
95
|
+
two call for opposite responses, since retrying cannot fix a typo.
|
|
96
|
+
|
|
97
|
+
argparse exits 2 for every usage error unless this is overridden, so
|
|
98
|
+
every tool needs it. tests/test_tool_exit_contract.py asserts it.
|
|
99
|
+
"""
|
|
100
|
+
|
|
101
|
+
def error(self, message):
|
|
102
|
+
self.print_usage(sys.stderr)
|
|
103
|
+
sys.stderr.write("%s: error: %s\n" % (self.prog, message))
|
|
104
|
+
raise SystemExit(64)
|
|
105
|
+
|
|
106
|
+
|
|
89
107
|
def _num(v):
|
|
90
108
|
"""Non-bool int/float, else None. Never coerces junk to 0."""
|
|
91
109
|
if isinstance(v, bool) or not isinstance(v, (int, float)):
|
|
@@ -322,7 +340,7 @@ def render(est):
|
|
|
322
340
|
|
|
323
341
|
|
|
324
342
|
def main(argv=None):
|
|
325
|
-
ap =
|
|
343
|
+
ap = _Parser(
|
|
326
344
|
description="Estimate what a run is likely to cost, from measured history.")
|
|
327
345
|
ap.add_argument("workspace", nargs="?", default=".")
|
|
328
346
|
ap.add_argument("--iterations", type=int, default=None,
|