loki-mode 9.8.1 → 9.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -14
- package/SKILL.md +3 -2
- package/VERSION +1 -1
- package/autonomy/loki +122 -1
- package/autonomy/run.sh +49 -2
- package/dashboard/__init__.py +1 -1
- package/dashboard/api_evidence.py +411 -0
- package/dashboard/api_operator.py +283 -0
- package/dashboard/api_phases.py +262 -0
- package/dashboard/api_releases.py +242 -0
- package/dashboard/api_runs.py +477 -0
- package/dashboard/api_tests.py +444 -0
- package/dashboard/api_v2.py +47 -1
- package/dashboard/server.py +54 -0
- package/dashboard/static/index.html +246 -135
- package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
- package/docs/CAPABILITY-BACKLOG.md +53 -0
- package/docs/COMPARISON.md +2 -2
- package/docs/COMPETITIVE-ANALYSIS.md +1 -1
- package/docs/COMPETITIVE-SCORECARD.md +422 -0
- package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
- package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
- package/docs/DEMOS.md +21 -23
- package/docs/HANDOFF-2026-08-03.md +439 -0
- package/docs/INSTALLATION.md +17 -10
- package/docs/OUTCOME-FRONTIER.md +536 -0
- package/docs/PROMPT-ABLATION-RESULT.md +97 -0
- package/docs/TOOLS.md +800 -0
- package/docs/alternative-installations.md +2 -3
- package/docs/audit-logging.md +44 -35
- package/docs/authentication.md +13 -2
- package/docs/authorization.md +87 -81
- package/docs/git-workflow.md +6 -3
- package/docs/metrics.md +15 -16
- package/docs/network-security.md +16 -13
- package/docs/openclaw-integration.md +36 -556
- package/docs/show-hn-post.md +2 -2
- package/docs/siem-integration.md +39 -36
- package/loki-ts/dist/loki.js +18 -18
- package/mcp/__init__.py +1 -1
- package/package.json +1 -1
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
- package/references/confidence-routing.md +18 -1
- package/references/invariant-checks.md +13 -8
- package/references/magic-rarv-integration.md +0 -1
- package/references/multi-provider.md +27 -5
- package/skills/healing.md +4 -2
- package/tools/audit-docs.py +488 -0
- package/tools/baseline-pin.py +19 -1
- package/tools/calibration-audit.py +523 -0
- package/tools/ci-gate.py +19 -1
- package/tools/cost-forecast.py +344 -0
- package/tools/cost-guard.py +19 -1
- package/tools/cost-history.py +19 -1
- package/tools/cost-per-outcome.py +394 -0
- package/tools/estimate-run.py +19 -1
- package/tools/evidence-freshness.py +307 -0
- package/tools/gate-init.py +19 -1
- package/tools/gate-report.py +19 -1
- package/tools/gate-simulate.py +570 -0
- package/tools/gate-trend.py +354 -0
- package/tools/model-advisor.py +52 -1
- package/tools/policy-load.py +19 -1
- package/tools/prompt-cost.py +363 -0
- package/tools/prompt-diff.py +448 -0
- package/tools/prompt-lint.py +448 -0
- package/tools/receipt-bundle.py +72 -2
- package/tools/receipt-diff.py +19 -1
- package/tools/receipt-find.py +19 -1
- package/tools/receipt-stats.py +380 -0
- package/tools/receipt-timeline.py +478 -0
- package/tools/receipt-verify-batch.py +291 -0
- package/tools/run-replay.py +19 -1
- package/tools/signing-status.py +19 -1
- package/tools/token-guard.py +19 -1
- package/tools/token-tax.py +375 -0
- package/tools/tool-index.py +19 -1
- package/tools/verification-tax.py +277 -0
- package/tools/verify-chain.py +361 -0
|
@@ -0,0 +1,344 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Project agent spend forward from recorded history, with an honest basis.
|
|
3
|
+
|
|
4
|
+
cost-history.py answers "is our spend trending up?" over runs already made.
|
|
5
|
+
The question that follows immediately, and the one a budget owner actually
|
|
6
|
+
asks, is forward-looking: we are about to do N more runs, what will they cost?
|
|
7
|
+
That is a PROJECTION, and a projection is the easiest place in this repo to
|
|
8
|
+
launder an absent measurement into a confident number.
|
|
9
|
+
|
|
10
|
+
THE FAILURE MODE THIS TOOL IS SHAPED AROUND. A forecast is arithmetic over a
|
|
11
|
+
sample, and arithmetic is happy to run on an empty sample: sum([]) is 0, and
|
|
12
|
+
0 * N is 0. So the naive implementation answers "$0.00" for a project with no
|
|
13
|
+
recorded history at all, which is not a cautious estimate -- it is the single
|
|
14
|
+
most confident claim the tool can make ("these runs are free"), produced at the
|
|
15
|
+
exact moment it knows nothing. An earlier attempt at this tool did precisely
|
|
16
|
+
that, projecting $0.00 from zero measured runs.
|
|
17
|
+
|
|
18
|
+
So the rule, and it is the whole tool:
|
|
19
|
+
|
|
20
|
+
NO MEASURED HISTORY MEANS NO FORECAST. Not a zero, not a cautious low
|
|
21
|
+
estimate, not a wide range around zero. No number at all.
|
|
22
|
+
|
|
23
|
+
Absent is not zero. A forecast is a claim about the future built from a sample
|
|
24
|
+
of the past, and with an empty sample there is no claim to make. Exit 3
|
|
25
|
+
(nothing to check), print the absence in words, and emit no dollar figure
|
|
26
|
+
anywhere -- including in --json, where `projected_usd` is null.
|
|
27
|
+
|
|
28
|
+
FOUR MORE PROPERTIES, each the honest half of a shortcut that would read better:
|
|
29
|
+
|
|
30
|
+
1. A TREND FROM ONE POINT IS NOT A TREND, but it is still a measurement. With
|
|
31
|
+
exactly one measured run this DOES forecast -- refusing would throw away
|
|
32
|
+
real data, which is the over-strict mirror of the same dishonesty -- and it
|
|
33
|
+
labels the projection SINGLE OBSERVATION, stating in words that it assumes
|
|
34
|
+
a flat rate from one point. The reader gets the number AND the reason to
|
|
35
|
+
distrust it. Silently presenting it like a 40-run mean is the lie.
|
|
36
|
+
|
|
37
|
+
2. THE BASIS IS PRINTED ON EVERY PROJECTION, never on request. "M measured of
|
|
38
|
+
N recorded" is what makes the number auditable: $4.20 from 40 measured runs
|
|
39
|
+
and $4.20 from 1 measured run of 39 are different facts, and the number
|
|
40
|
+
alone cannot tell them apart. A basis available behind a --verbose flag is a
|
|
41
|
+
basis nobody reads.
|
|
42
|
+
|
|
43
|
+
3. AN UNMEASURED ROW IS EXCLUDED, NEVER READ AS 0. cost-history.py records an
|
|
44
|
+
unmeasured run as usd=null on purpose, so the measurement gap stays
|
|
45
|
+
countable. Averaging those nulls as 0 would drag the mean down and understate
|
|
46
|
+
the forecast -- and it would do so worst when instrumentation is broken, i.e.
|
|
47
|
+
when the estimate matters most. Excluded rows are COUNTED and reported.
|
|
48
|
+
|
|
49
|
+
4. A MEASURED ZERO IS NOT AN UNMEASURED ROW. A run that genuinely cost $0.0000
|
|
50
|
+
(cached, free tier) is data and stays in the mean as 0. This is why every
|
|
51
|
+
test here is `is None` and never a truthiness check: `if usd:` would drop a
|
|
52
|
+
real zero and silently bias the forecast upward. record_is_measured() is
|
|
53
|
+
imported for the legacy row that carries no `measured` flag, and is
|
|
54
|
+
deliberately NOT applied to the flagged rows -- it is a truthiness predicate
|
|
55
|
+
over token fields, so it answers False for a measured $0.00, which is
|
|
56
|
+
correct for its own question and wrong for this one.
|
|
57
|
+
|
|
58
|
+
5. A CORRUPT LINE IS COUNTED AND REPORTED, never silently skipped. A history
|
|
59
|
+
that quietly drops rows forecasts from a cleaner sample than reality.
|
|
60
|
+
|
|
61
|
+
WHY THE MEAN AND NOT THE MEDIAN. cost-history.py trends on the median, because
|
|
62
|
+
a trend must not be named by one outlier. A forecast of TOTAL spend over N runs
|
|
63
|
+
is a different question: the expensive runs are real money and the total has to
|
|
64
|
+
include them. mean * N is the unbiased estimator of that total; median * N
|
|
65
|
+
systematically under-forecasts any right-skewed cost distribution, which agent
|
|
66
|
+
spend always is. Both numbers are printed so the skew is visible.
|
|
67
|
+
|
|
68
|
+
NOT A GATE. This advises a budget owner; no CI job branches on its exit code,
|
|
69
|
+
and it never blocks a merge. Exit 3 on an absent basis says "nothing to check",
|
|
70
|
+
which is the honest answer to "what will N runs cost" when nothing was measured.
|
|
71
|
+
|
|
72
|
+
Usage:
|
|
73
|
+
tools/cost-forecast.py --runs N [--file .loki/cost-history.jsonl] [--json]
|
|
74
|
+
|
|
75
|
+
Exit: 0 forecast produced, 2 history unreadable, 3 nothing to forecast from,
|
|
76
|
+
64 usage error, 66 history file does not exist.
|
|
77
|
+
"""
|
|
78
|
+
|
|
79
|
+
import sys
|
|
80
|
+
|
|
81
|
+
sys.dont_write_bytecode = True
|
|
82
|
+
|
|
83
|
+
import argparse # noqa: E402
|
|
84
|
+
import json # noqa: E402
|
|
85
|
+
import os # noqa: E402
|
|
86
|
+
|
|
87
|
+
_HERE = os.path.dirname(os.path.abspath(__file__))
|
|
88
|
+
sys.path.insert(0, os.path.join(os.path.dirname(_HERE), "autonomy", "lib"))
|
|
89
|
+
|
|
90
|
+
from efficiency_cost import record_is_measured # noqa: E402
|
|
91
|
+
|
|
92
|
+
OK, CANNOT, NOTHING_TO_CHECK, USAGE, NO_INPUT = 0, 2, 3, 64, 66
|
|
93
|
+
|
|
94
|
+
DEFAULT_FILE = os.path.join(".loki", "cost-history.jsonl")
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class _Parser(argparse.ArgumentParser):
|
|
98
|
+
"""argparse exits 2 on a usage error, and 2 here means "could not check".
|
|
99
|
+
|
|
100
|
+
Those are opposite facts. "You typed the flag wrong" and "the instrument is
|
|
101
|
+
blind" must not share an exit code: a typo would masquerade as a blind
|
|
102
|
+
gate, and the operator would go hunting for missing instrumentation.
|
|
103
|
+
"""
|
|
104
|
+
|
|
105
|
+
def error(self, message):
|
|
106
|
+
self.print_usage(sys.stderr)
|
|
107
|
+
print("%s: error: %s" % (self.prog, message), file=sys.stderr)
|
|
108
|
+
raise SystemExit(USAGE)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _num(v):
|
|
112
|
+
"""A number as itself; None, "", or a bool as None. A real 0 survives as 0."""
|
|
113
|
+
if isinstance(v, bool) or not isinstance(v, (int, float)):
|
|
114
|
+
return None
|
|
115
|
+
return v
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def row_usd(row):
|
|
119
|
+
"""The measured USD for one history row, or None when it was not measured.
|
|
120
|
+
|
|
121
|
+
THE ONE PLACE the measured/unmeasured distinction is decided, so it cannot
|
|
122
|
+
drift between the loader and the reporter.
|
|
123
|
+
|
|
124
|
+
`measured` is cost-history.py's own flag, written at record time by
|
|
125
|
+
record_is_measured(), so trusting it REUSES that rule rather than restating
|
|
126
|
+
it. A legacy row without the flag falls back to record_is_measured() here.
|
|
127
|
+
|
|
128
|
+
The final test is `is not None`, never truthiness: a genuine $0.0000 run is
|
|
129
|
+
a measurement and must survive as 0.0. Dropping it would bias every
|
|
130
|
+
forecast upward, and would do it invisibly.
|
|
131
|
+
|
|
132
|
+
THE ORDER OF THE LEGACY BRANCH IS LOAD-BEARING. record_is_measured() is a
|
|
133
|
+
TRUTHINESS predicate over token fields, so it answers False for a measured
|
|
134
|
+
$0.0000 -- correct for its own question ("did this record observe
|
|
135
|
+
anything?"), wrong for this one ("is this dollar figure present?"). Asking
|
|
136
|
+
it first would drop a real zero from a flag-less row while the identical
|
|
137
|
+
flagged row was kept, so the same fact got two answers depending on which
|
|
138
|
+
version of cost-history.py wrote it. `usd is None` is therefore tested
|
|
139
|
+
FIRST and short-circuits; record_is_measured() only ever adjudicates rows
|
|
140
|
+
whose usd is already absent.
|
|
141
|
+
"""
|
|
142
|
+
if not isinstance(row, dict):
|
|
143
|
+
return None
|
|
144
|
+
usd = _num(row.get("usd"))
|
|
145
|
+
if "measured" in row:
|
|
146
|
+
if row.get("measured") is not True:
|
|
147
|
+
return None
|
|
148
|
+
elif usd is None and not record_is_measured(row):
|
|
149
|
+
return None
|
|
150
|
+
return usd
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def load(path):
|
|
154
|
+
"""Read the history. Returns (rows, corrupt_count) or (None, 0) if unreadable.
|
|
155
|
+
|
|
156
|
+
A line that is not JSON, or is JSON but not an object carrying `usd`, is
|
|
157
|
+
corrupt and COUNTED. It is never dropped on the floor: a history that
|
|
158
|
+
quietly loses rows forecasts from a tidier sample than the real one, and
|
|
159
|
+
loses them most often when something upstream just broke.
|
|
160
|
+
"""
|
|
161
|
+
rows, corrupt = [], 0
|
|
162
|
+
try:
|
|
163
|
+
with open(path, "r", encoding="utf-8") as handle:
|
|
164
|
+
raw = handle.read()
|
|
165
|
+
except OSError:
|
|
166
|
+
return None, 0
|
|
167
|
+
for line in raw.splitlines():
|
|
168
|
+
if not line.strip():
|
|
169
|
+
continue
|
|
170
|
+
try:
|
|
171
|
+
obj = json.loads(line)
|
|
172
|
+
except ValueError:
|
|
173
|
+
corrupt += 1
|
|
174
|
+
continue
|
|
175
|
+
if not isinstance(obj, dict) or "usd" not in obj:
|
|
176
|
+
corrupt += 1
|
|
177
|
+
continue
|
|
178
|
+
rows.append(obj)
|
|
179
|
+
return rows, corrupt
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def median(values):
|
|
183
|
+
"""Median of a non-empty list. Even length averages the middle pair."""
|
|
184
|
+
s = sorted(values)
|
|
185
|
+
n = len(s)
|
|
186
|
+
mid = n // 2
|
|
187
|
+
return s[mid] if n % 2 else (s[mid - 1] + s[mid]) / 2.0
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def forecast(path, runs):
|
|
191
|
+
"""Project spend over `runs` future runs. Returns a verdict dict.
|
|
192
|
+
|
|
193
|
+
Every absence path returns projected_usd=None. There is no branch in this
|
|
194
|
+
function that produces a number from an empty sample, which is the property
|
|
195
|
+
the whole file exists to hold.
|
|
196
|
+
"""
|
|
197
|
+
base = {
|
|
198
|
+
"runs": runs,
|
|
199
|
+
"recorded": 0,
|
|
200
|
+
"measured": 0,
|
|
201
|
+
"excluded_unmeasured": 0,
|
|
202
|
+
"corrupt_lines": 0,
|
|
203
|
+
"mean_usd": None,
|
|
204
|
+
"median_usd": None,
|
|
205
|
+
"projected_usd": None,
|
|
206
|
+
"confidence": "NONE",
|
|
207
|
+
"basis": None,
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
if not os.path.exists(path):
|
|
211
|
+
base["status"] = "no_history"
|
|
212
|
+
base["exit_code"] = NO_INPUT
|
|
213
|
+
base["why"] = ("no history file at %s -- record runs with "
|
|
214
|
+
"tools/cost-history.py record first. There is no "
|
|
215
|
+
"sample to project from, so there is no forecast."
|
|
216
|
+
% path)
|
|
217
|
+
return base
|
|
218
|
+
|
|
219
|
+
rows, corrupt = load(path)
|
|
220
|
+
if rows is None:
|
|
221
|
+
base["status"] = "unreadable"
|
|
222
|
+
base["exit_code"] = CANNOT
|
|
223
|
+
base["why"] = ("history at %s exists but could not be read; the "
|
|
224
|
+
"instrument is blind, which is not the same as a "
|
|
225
|
+
"cheap forecast." % path)
|
|
226
|
+
return base
|
|
227
|
+
|
|
228
|
+
base["recorded"] = len(rows)
|
|
229
|
+
base["corrupt_lines"] = corrupt
|
|
230
|
+
|
|
231
|
+
costs = []
|
|
232
|
+
unmeasured = 0
|
|
233
|
+
for row in rows:
|
|
234
|
+
usd = row_usd(row)
|
|
235
|
+
if usd is None:
|
|
236
|
+
unmeasured += 1
|
|
237
|
+
else:
|
|
238
|
+
costs.append(usd)
|
|
239
|
+
base["measured"] = len(costs)
|
|
240
|
+
base["excluded_unmeasured"] = unmeasured
|
|
241
|
+
|
|
242
|
+
if not costs:
|
|
243
|
+
# THE PATH THAT MUST NEVER PRODUCE A NUMBER. Zero measured runs, whether
|
|
244
|
+
# from an empty file, an all-null history, or nothing but corrupt lines.
|
|
245
|
+
# sum([])/0 does not even divide, and the tempting repair -- treat the
|
|
246
|
+
# nulls as 0 -- is the lie itself. projected_usd stays None.
|
|
247
|
+
base["status"] = "no_measured_history"
|
|
248
|
+
base["exit_code"] = NOTHING_TO_CHECK
|
|
249
|
+
base["why"] = (
|
|
250
|
+
# No dollar SIGN anywhere on this path, not even in prose. The
|
|
251
|
+
# tests assert the absence of "$" in the whole output, which is one
|
|
252
|
+
# assertion that catches every number at once -- worth more than
|
|
253
|
+
# the rhetorical flourish of quoting the figure being refused.
|
|
254
|
+
"%d run(s) recorded, 0 measured%s. A forecast is arithmetic over a "
|
|
255
|
+
"sample and there is no sample, so no figure is given: an absent "
|
|
256
|
+
"measurement reads UNKNOWN, never zero."
|
|
257
|
+
% (len(rows),
|
|
258
|
+
" (%d unmeasured, excluded not counted as 0)" % unmeasured
|
|
259
|
+
if unmeasured else ""))
|
|
260
|
+
return base
|
|
261
|
+
|
|
262
|
+
mean = sum(costs) / float(len(costs))
|
|
263
|
+
base["mean_usd"] = mean
|
|
264
|
+
base["median_usd"] = median(costs)
|
|
265
|
+
base["projected_usd"] = mean * runs
|
|
266
|
+
base["status"] = "ok"
|
|
267
|
+
base["exit_code"] = OK
|
|
268
|
+
|
|
269
|
+
if len(costs) < 2:
|
|
270
|
+
# One point is a measurement but not a trend. Forecast anyway -- the
|
|
271
|
+
# data is real -- and say plainly what the projection assumes, so the
|
|
272
|
+
# number is never mistaken for a 40-run mean.
|
|
273
|
+
base["confidence"] = "SINGLE OBSERVATION"
|
|
274
|
+
base["assumption"] = (
|
|
275
|
+
"flat rate assumed from a SINGLE observation; one point cannot "
|
|
276
|
+
"establish a trend, so this projects that one run's cost forward "
|
|
277
|
+
"unchanged and says nothing about whether spend is rising.")
|
|
278
|
+
else:
|
|
279
|
+
base["confidence"] = "MULTI OBSERVATION"
|
|
280
|
+
base["assumption"] = (
|
|
281
|
+
"flat rate assumed: the mean of %d measured run(s) projected "
|
|
282
|
+
"forward. This does not model a trend; use "
|
|
283
|
+
"tools/cost-history.py report for direction." % len(costs))
|
|
284
|
+
|
|
285
|
+
base["basis"] = ("%d measured of %d recorded run(s)"
|
|
286
|
+
% (len(costs), len(rows)))
|
|
287
|
+
return base
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def render(d):
|
|
291
|
+
"""Text rendering. Every absence branch returns BEFORE any %-format of a
|
|
292
|
+
dollar figure, so a None projection can never reach a float format."""
|
|
293
|
+
if d["status"] == "no_history":
|
|
294
|
+
return "NO FORECAST: %s" % d["why"]
|
|
295
|
+
if d["status"] == "unreadable":
|
|
296
|
+
return "CANNOT FORECAST: %s" % d["why"]
|
|
297
|
+
|
|
298
|
+
lines = []
|
|
299
|
+
if d["corrupt_lines"]:
|
|
300
|
+
lines.append("%d CORRUPT line(s) in the history -- counted, not "
|
|
301
|
+
"skipped; excluded from the sample below."
|
|
302
|
+
% d["corrupt_lines"])
|
|
303
|
+
if d["excluded_unmeasured"]:
|
|
304
|
+
lines.append("%d unmeasured run(s) EXCLUDED (recorded as null, never "
|
|
305
|
+
"averaged as 0)." % d["excluded_unmeasured"])
|
|
306
|
+
|
|
307
|
+
if d["status"] == "no_measured_history":
|
|
308
|
+
lines.append("NO FORECAST: %s" % d["why"])
|
|
309
|
+
return "\n".join(lines)
|
|
310
|
+
|
|
311
|
+
lines.append("basis: %s" % d["basis"])
|
|
312
|
+
lines.append("mean $%.4f/run, median $%.4f/run"
|
|
313
|
+
% (d["mean_usd"], d["median_usd"]))
|
|
314
|
+
lines.append("projected over %d run(s): $%.4f"
|
|
315
|
+
% (d["runs"], d["projected_usd"]))
|
|
316
|
+
lines.append("confidence: %s" % d["confidence"])
|
|
317
|
+
lines.append(d["assumption"])
|
|
318
|
+
return "\n".join(lines)
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def main(argv=None):
|
|
322
|
+
ap = _Parser(
|
|
323
|
+
description="Project agent spend forward from recorded cost history.")
|
|
324
|
+
ap.add_argument("--runs", type=int, required=True,
|
|
325
|
+
help="number of future runs to project over")
|
|
326
|
+
ap.add_argument("--file", default=DEFAULT_FILE,
|
|
327
|
+
help="history JSONL (default %s)" % DEFAULT_FILE)
|
|
328
|
+
ap.add_argument("--json", action="store_true", dest="as_json",
|
|
329
|
+
help="emit the forecast as JSON")
|
|
330
|
+
|
|
331
|
+
args = ap.parse_args(argv)
|
|
332
|
+
if args.runs < 1:
|
|
333
|
+
# Not a forecast question. Refusing beats answering $0.00 for 0 runs,
|
|
334
|
+
# which is a real number that reads like a measurement.
|
|
335
|
+
ap.error("--runs must be 1 or greater (got %d)" % args.runs)
|
|
336
|
+
|
|
337
|
+
d = forecast(args.file, args.runs)
|
|
338
|
+
print(json.dumps(d, indent=2, sort_keys=True) if args.as_json
|
|
339
|
+
else render(d))
|
|
340
|
+
return d["exit_code"]
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
if __name__ == "__main__":
|
|
344
|
+
sys.exit(main())
|
package/tools/cost-guard.py
CHANGED
|
@@ -69,6 +69,24 @@ from efficiency_cost import collect_efficiency, record_is_measured # noqa: E402
|
|
|
69
69
|
OK, OVER, CANNOT = 0, 1, 2
|
|
70
70
|
|
|
71
71
|
|
|
72
|
+
class _Parser(argparse.ArgumentParser):
|
|
73
|
+
"""Usage errors exit 64, not argparse's default 2.
|
|
74
|
+
|
|
75
|
+
In this repo's convention 2 means "could NOT be checked" -- a real
|
|
76
|
+
answer about the subject. A mistyped flag is not that: it is an error
|
|
77
|
+
about the INVOCATION, and nothing about the subject was examined. The
|
|
78
|
+
two call for opposite responses, since retrying cannot fix a typo.
|
|
79
|
+
|
|
80
|
+
argparse exits 2 for every usage error unless this is overridden, so
|
|
81
|
+
every tool needs it. tests/test_tool_exit_contract.py asserts it.
|
|
82
|
+
"""
|
|
83
|
+
|
|
84
|
+
def error(self, message):
|
|
85
|
+
self.print_usage(sys.stderr)
|
|
86
|
+
sys.stderr.write("%s: error: %s\n" % (self.prog, message))
|
|
87
|
+
raise SystemExit(64)
|
|
88
|
+
|
|
89
|
+
|
|
72
90
|
def _num(v):
|
|
73
91
|
"""A number as itself; None, "", or a bool as None."""
|
|
74
92
|
if isinstance(v, bool) or not isinstance(v, (int, float)):
|
|
@@ -214,7 +232,7 @@ def render(d):
|
|
|
214
232
|
|
|
215
233
|
|
|
216
234
|
def main(argv=None):
|
|
217
|
-
ap =
|
|
235
|
+
ap = _Parser(
|
|
218
236
|
description="Fail CI when a run's cost regressed past a budget policy.")
|
|
219
237
|
ap.add_argument("workspace", nargs="?", default=".",
|
|
220
238
|
help="workspace root (or its .loki dir); default .")
|
package/tools/cost-history.py
CHANGED
|
@@ -67,6 +67,24 @@ DEFAULT_FILE = os.path.join(".loki", "cost-history.jsonl")
|
|
|
67
67
|
FLAT_PCT = 5.0
|
|
68
68
|
|
|
69
69
|
|
|
70
|
+
class _Parser(argparse.ArgumentParser):
|
|
71
|
+
"""Usage errors exit 64, not argparse's default 2.
|
|
72
|
+
|
|
73
|
+
In this repo's convention 2 means "could NOT be checked" -- a real
|
|
74
|
+
answer about the subject. A mistyped flag is not that: it is an error
|
|
75
|
+
about the INVOCATION, and nothing about the subject was examined. The
|
|
76
|
+
two call for opposite responses, since retrying cannot fix a typo.
|
|
77
|
+
|
|
78
|
+
argparse exits 2 for every usage error unless this is overridden, so
|
|
79
|
+
every tool needs it. tests/test_tool_exit_contract.py asserts it.
|
|
80
|
+
"""
|
|
81
|
+
|
|
82
|
+
def error(self, message):
|
|
83
|
+
self.print_usage(sys.stderr)
|
|
84
|
+
sys.stderr.write("%s: error: %s\n" % (self.prog, message))
|
|
85
|
+
raise SystemExit(64)
|
|
86
|
+
|
|
87
|
+
|
|
70
88
|
def _num(v):
|
|
71
89
|
"""A number as itself; None, "", or a bool as None."""
|
|
72
90
|
if isinstance(v, bool) or not isinstance(v, (int, float)):
|
|
@@ -287,7 +305,7 @@ def render(d):
|
|
|
287
305
|
|
|
288
306
|
|
|
289
307
|
def main(argv=None):
|
|
290
|
-
ap =
|
|
308
|
+
ap = _Parser(
|
|
291
309
|
description="Track agent cost across many runs and report the trend.")
|
|
292
310
|
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
293
311
|
|