loki-mode 9.8.0 → 9.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -14
- package/SKILL.md +3 -2
- package/VERSION +1 -1
- package/autonomy/loki +122 -1
- package/autonomy/run.sh +49 -2
- package/dashboard/__init__.py +1 -1
- package/dashboard/api_evidence.py +411 -0
- package/dashboard/api_operator.py +283 -0
- package/dashboard/api_phases.py +262 -0
- package/dashboard/api_releases.py +242 -0
- package/dashboard/api_runs.py +477 -0
- package/dashboard/api_tests.py +444 -0
- package/dashboard/api_v2.py +47 -1
- package/dashboard/server.py +54 -0
- package/dashboard/static/index.html +246 -135
- package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
- package/docs/CAPABILITY-BACKLOG.md +53 -0
- package/docs/COMPARISON.md +2 -2
- package/docs/COMPETITIVE-ANALYSIS.md +1 -1
- package/docs/COMPETITIVE-SCORECARD.md +422 -0
- package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
- package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
- package/docs/DEMOS.md +21 -23
- package/docs/HANDOFF-2026-08-03.md +439 -0
- package/docs/INSTALLATION.md +17 -10
- package/docs/OUTCOME-FRONTIER.md +536 -0
- package/docs/PROMPT-ABLATION-RESULT.md +97 -0
- package/docs/TOOLS.md +800 -0
- package/docs/alternative-installations.md +2 -3
- package/docs/audit-logging.md +44 -35
- package/docs/authentication.md +13 -2
- package/docs/authorization.md +87 -81
- package/docs/git-workflow.md +6 -3
- package/docs/metrics.md +15 -16
- package/docs/network-security.md +16 -13
- package/docs/openclaw-integration.md +36 -556
- package/docs/show-hn-post.md +2 -2
- package/docs/siem-integration.md +39 -36
- package/loki-ts/dist/loki.js +18 -18
- package/mcp/__init__.py +1 -1
- package/package.json +2 -2
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
- package/references/confidence-routing.md +18 -1
- package/references/invariant-checks.md +13 -8
- package/references/magic-rarv-integration.md +0 -1
- package/references/multi-provider.md +27 -5
- package/skills/healing.md +4 -2
- package/tools/audit-docs.py +488 -0
- package/tools/baseline-pin.py +19 -1
- package/tools/calibration-audit.py +523 -0
- package/tools/ci-gate.py +19 -1
- package/tools/cost-forecast.py +344 -0
- package/tools/cost-guard.py +19 -1
- package/tools/cost-history.py +19 -1
- package/tools/cost-per-outcome.py +394 -0
- package/tools/estimate-run.py +19 -1
- package/tools/evidence-freshness.py +307 -0
- package/tools/gate-init.py +19 -1
- package/tools/gate-report.py +19 -1
- package/tools/gate-simulate.py +570 -0
- package/tools/gate-trend.py +354 -0
- package/tools/model-advisor.py +52 -1
- package/tools/policy-load.py +19 -1
- package/tools/prompt-cost.py +363 -0
- package/tools/prompt-diff.py +448 -0
- package/tools/prompt-lint.py +448 -0
- package/tools/receipt-bundle.py +72 -2
- package/tools/receipt-diff.py +19 -1
- package/tools/receipt-find.py +19 -1
- package/tools/receipt-stats.py +380 -0
- package/tools/receipt-timeline.py +478 -0
- package/tools/receipt-verify-batch.py +291 -0
- package/tools/run-replay.py +19 -1
- package/tools/signing-status.py +19 -1
- package/tools/token-guard.py +19 -1
- package/tools/token-tax.py +375 -0
- package/tools/tool-index.py +19 -1
- package/tools/verification-tax.py +277 -0
- package/tools/verify-chain.py +361 -0
|
@@ -0,0 +1,354 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Is the merge gate getting better, getting worse, or going BLIND?
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. tools/gate-log.py made the verdicts durable and reports the
|
|
5
|
+
totals. Totals answer "how have we done", which is the wrong question for a
|
|
6
|
+
gate. The question an engineer asks before trusting a merge is directional:
|
|
7
|
+
|
|
8
|
+
is this gate improving, or is it quietly degrading?
|
|
9
|
+
|
|
10
|
+
gate-log's own trend field compares two halves on one axis, the pass rate. That
|
|
11
|
+
is enough to say "healthier" or "worse", and it is deliberately not enough to
|
|
12
|
+
say WHY, because on one axis the two ways a gate degrades are indistinguishable.
|
|
13
|
+
This file separates them, because they call for opposite responses.
|
|
14
|
+
|
|
15
|
+
THE DISTINCTION THIS TOOL IS FOR:
|
|
16
|
+
|
|
17
|
+
a rising FAIL rate = the gate is doing its job, and the code is worse
|
|
18
|
+
a rising UNEVALUABLE = the gate is going BLIND, and knows nothing at all
|
|
19
|
+
|
|
20
|
+
The second is more urgent and reads as less urgent, which is how it survives. A
|
|
21
|
+
failing gate is loud: someone is blocked, someone investigates. A blind gate is
|
|
22
|
+
silent and its numbers look BETTER the longer it stays blind, because an axis
|
|
23
|
+
that cannot be evaluated can never fail. Three weeks of blindness on the receipt
|
|
24
|
+
axis renders as a falling failure rate, which is indistinguishable from three
|
|
25
|
+
weeks of improvement. So GOING BLIND outranks WORSENING in the verdict here: an
|
|
26
|
+
operator told only "worsening" fixes the failures, re-runs, sees the failure
|
|
27
|
+
rate drop, and is still blind.
|
|
28
|
+
|
|
29
|
+
THE ARITHMETIC THAT KILLS THIS, and the reason there is no subtraction below:
|
|
30
|
+
|
|
31
|
+
unevaluable = total - passes - failures
|
|
32
|
+
pass_rate = 1 - fail_rate
|
|
33
|
+
|
|
34
|
+
Both are natural, both fold the third category into one of the other two, and
|
|
35
|
+
both are the exact defect gate-log.py's header names. Every rate here is counted
|
|
36
|
+
from its OWN bucket over the shared denominator. Nothing is derived by
|
|
37
|
+
complement, so no category can absorb another's mass.
|
|
38
|
+
|
|
39
|
+
FEWER THAN 2 RECORDS IS NOT "FLAT". Flat is a claim about change over time and
|
|
40
|
+
needs two points to make. One record reports INSUFFICIENT DATA and exits
|
|
41
|
+
non-zero, because this is a gate: saying "I cannot establish a trend" while
|
|
42
|
+
exiting 0 tells CI the trend axis was checked and was fine.
|
|
43
|
+
|
|
44
|
+
A CORRUPT LINE IS COUNTED, NEVER SKIPPED. Skipping is how a log half-eaten by a
|
|
45
|
+
crashed writer reports a clean history of whatever survived. Corrupt lines get
|
|
46
|
+
their own count, are included in the window and the denominator, and force the
|
|
47
|
+
blind exit code, since a line we could not read is a verdict we do not have.
|
|
48
|
+
|
|
49
|
+
AN EMPTY LOG EXITS 3, NOT 0. "No verdicts recorded" and "no verdict ever
|
|
50
|
+
blocked" are opposite facts. Absent is not zero.
|
|
51
|
+
|
|
52
|
+
WHAT IS DELIBERATELY NOT HERE: any cost figure, and therefore no use of
|
|
53
|
+
record_is_measured() from autonomy/lib/efficiency_cost.py. gate-log rows carry
|
|
54
|
+
no measured cost by design (see its "WHAT IS DELIBERATELY NOT HERE"), and
|
|
55
|
+
recovering one by parsing a policy's human-readable reason string would restate
|
|
56
|
+
a predicate efficiency_cost.py owns. That drift is what this repo fixed across
|
|
57
|
+
four surfaces. No cost surface beats a re-derived one, so this tool reports
|
|
58
|
+
categories only.
|
|
59
|
+
|
|
60
|
+
Usage:
|
|
61
|
+
tools/gate-trend.py [--file .loki/gate-log.jsonl] [--window N] [--json]
|
|
62
|
+
|
|
63
|
+
Exit: 0 improving or steady with every record readable and passing-or-failing
|
|
64
|
+
cleanly, 1 a genuine regression in the FAIL rate, 2 could not evaluate (the
|
|
65
|
+
gate is blind: unevaluable or corrupt records present, or fewer than 2 records
|
|
66
|
+
to compare), 3 the log exists but holds no records, 64 usage error, 66 no log
|
|
67
|
+
file at that path.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
import argparse
|
|
71
|
+
import json
|
|
72
|
+
import os
|
|
73
|
+
import sys
|
|
74
|
+
|
|
75
|
+
sys.dont_write_bytecode = True
|
|
76
|
+
|
|
77
|
+
PASSED, FAILED, COULD_NOT_CHECK, NOTHING, USAGE, MISSING = 0, 1, 2, 3, 64, 66
|
|
78
|
+
|
|
79
|
+
DEFAULT_LOG = os.path.join(".loki", "gate-log.jsonl")
|
|
80
|
+
|
|
81
|
+
# THE ONE MAPPING from a recorded verdict word to a bucket. Three buckets,
|
|
82
|
+
# never two. Folding the blind bucket into the passing one is the whole defect
|
|
83
|
+
# this file exists to prevent, and it is a one-word edit -- which is exactly
|
|
84
|
+
# what tests/test_gate_trend.py mutates.
|
|
85
|
+
_CATEGORY = {"PASS": "pass", "FAIL": "fail", "UNEVALUABLE": "unevaluable"}
|
|
86
|
+
|
|
87
|
+
# Not a verdict any gate can emit; what a damaged line becomes. Kept out of
|
|
88
|
+
# _CATEGORY so no recorded state word can ever map into it.
|
|
89
|
+
_CORRUPT = "corrupt"
|
|
90
|
+
|
|
91
|
+
_ORDER = ["pass", "fail", "unevaluable", _CORRUPT]
|
|
92
|
+
|
|
93
|
+
# A category that means the history itself is unreadable on that record. Both
|
|
94
|
+
# force the blind exit: a verdict we could not read is a verdict we do not have.
|
|
95
|
+
_BLIND = ("unevaluable", _CORRUPT)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class _Parser(argparse.ArgumentParser):
|
|
99
|
+
"""argparse exits 2 on a usage error. Here 2 means "could not check".
|
|
100
|
+
|
|
101
|
+
A mistyped flag would otherwise be indistinguishable from this gate
|
|
102
|
+
reporting that it was blind, and a CI job branching on the code would treat
|
|
103
|
+
an operator's typo as a real finding about the merge. 64 is the convention.
|
|
104
|
+
It also means a stray positional is an ERROR rather than something read as
|
|
105
|
+
a log path: receipt-attest once read `--help` as a proof path and issued a
|
|
106
|
+
verdict about it.
|
|
107
|
+
"""
|
|
108
|
+
|
|
109
|
+
def error(self, message):
|
|
110
|
+
self.print_usage(sys.stderr)
|
|
111
|
+
sys.stderr.write("gate-trend: %s\n" % message)
|
|
112
|
+
raise SystemExit(USAGE)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def classify(verdict):
|
|
116
|
+
"""Bucket ONE recorded verdict. Anything unrecognised is unevaluable.
|
|
117
|
+
|
|
118
|
+
Never defaults to "pass". A default of pass is how an aggregator launders
|
|
119
|
+
every shape it did not anticipate into green, and the shapes it did not
|
|
120
|
+
anticipate are precisely the broken ones.
|
|
121
|
+
"""
|
|
122
|
+
if not isinstance(verdict, dict):
|
|
123
|
+
return _CATEGORY["UNEVALUABLE"]
|
|
124
|
+
state = verdict.get("state")
|
|
125
|
+
if isinstance(state, str) and state.upper() in _CATEGORY:
|
|
126
|
+
return _CATEGORY[state.upper()]
|
|
127
|
+
return _CATEGORY["UNEVALUABLE"]
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def category_of(entry):
|
|
131
|
+
"""A log entry's bucket, re-derived from its verdict when it lacks one.
|
|
132
|
+
|
|
133
|
+
Trusting a stored "category" blindly would let a hand-edited log assert
|
|
134
|
+
anything; the embedded verdict stays authoritative. An entry with neither is
|
|
135
|
+
unevaluable, not a pass. Deliberately checks _CATEGORY.values() and not
|
|
136
|
+
_ORDER, so a stored "corrupt" cannot inflate the count of lines this reader
|
|
137
|
+
actually failed to parse.
|
|
138
|
+
"""
|
|
139
|
+
if not isinstance(entry, dict):
|
|
140
|
+
return _CATEGORY["UNEVALUABLE"]
|
|
141
|
+
stored = entry.get("category")
|
|
142
|
+
if isinstance(stored, str) and stored in _CATEGORY.values():
|
|
143
|
+
return stored
|
|
144
|
+
return classify(entry.get("verdict"))
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def read_categories(path):
|
|
148
|
+
"""Every line in order, as a list of buckets. Corrupt lines INCLUDED.
|
|
149
|
+
|
|
150
|
+
A corrupt line that is merely skipped is a verdict deleted from the record
|
|
151
|
+
by the reader, and the resulting trend describes a history that never
|
|
152
|
+
happened. Returning it in sequence also keeps it inside --window, so a
|
|
153
|
+
window cannot launder corruption by sliding past it.
|
|
154
|
+
"""
|
|
155
|
+
out = []
|
|
156
|
+
with open(path, "r", encoding="utf-8") as fh:
|
|
157
|
+
for raw in fh:
|
|
158
|
+
if not raw.strip():
|
|
159
|
+
continue # a trailing newline is not a damaged record
|
|
160
|
+
try:
|
|
161
|
+
entry = json.loads(raw)
|
|
162
|
+
except ValueError:
|
|
163
|
+
out.append(_CORRUPT)
|
|
164
|
+
continue
|
|
165
|
+
if not isinstance(entry, dict):
|
|
166
|
+
out.append(_CORRUPT)
|
|
167
|
+
continue
|
|
168
|
+
out.append(category_of(entry))
|
|
169
|
+
return out
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _rates(seq):
|
|
173
|
+
"""One rate per bucket, each counted from its OWN members.
|
|
174
|
+
|
|
175
|
+
No complement, no subtraction from the total. `unevaluable = 1 - pass` is
|
|
176
|
+
the one line that would make a blind gate look healthy, so every rate is
|
|
177
|
+
an independent count over the shared denominator.
|
|
178
|
+
"""
|
|
179
|
+
n = float(len(seq))
|
|
180
|
+
return dict((name, sum(1 for c in seq if c == name) / n) for name in _ORDER)
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def analyse(seq):
|
|
184
|
+
"""Split the window in half and compare each category's rate independently.
|
|
185
|
+
|
|
186
|
+
Returns a summary whose "direction" is None -- UNKNOWN, never "flat" --
|
|
187
|
+
when there are fewer than 2 records. Flat is a claim about change, and one
|
|
188
|
+
point cannot support it.
|
|
189
|
+
"""
|
|
190
|
+
counts = dict((name, sum(1 for c in seq if c == name)) for name in _ORDER)
|
|
191
|
+
blind = sum(counts[name] for name in _BLIND)
|
|
192
|
+
summary = {
|
|
193
|
+
"records": len(seq),
|
|
194
|
+
"counts": counts,
|
|
195
|
+
"blind_records": blind,
|
|
196
|
+
"direction": None,
|
|
197
|
+
"older": None,
|
|
198
|
+
"newer": None,
|
|
199
|
+
"window": None,
|
|
200
|
+
}
|
|
201
|
+
if len(seq) < 2:
|
|
202
|
+
return summary
|
|
203
|
+
|
|
204
|
+
half = len(seq) // 2
|
|
205
|
+
older, newer = _rates(seq[:half]), _rates(seq[half:])
|
|
206
|
+
summary["older"] = older
|
|
207
|
+
summary["newer"] = newer
|
|
208
|
+
summary["window"] = [half, len(seq) - half]
|
|
209
|
+
|
|
210
|
+
# PRECEDENCE. Blindness first: a gate that cannot evaluate has not passed,
|
|
211
|
+
# and its failure rate falls precisely because it is blind. Reporting
|
|
212
|
+
# "improving" off a falling failure rate while the blind rate climbs is the
|
|
213
|
+
# inversion this tool exists to make impossible.
|
|
214
|
+
blind_older = older["unevaluable"] + older[_CORRUPT]
|
|
215
|
+
blind_newer = newer["unevaluable"] + newer[_CORRUPT]
|
|
216
|
+
if blind_newer > blind_older:
|
|
217
|
+
summary["direction"] = "going_blind"
|
|
218
|
+
elif blind_newer:
|
|
219
|
+
# Blind and getting less blind is STILL BLIND, never "improving". A
|
|
220
|
+
# gate 75% unable to evaluate, printing "IMPROVING -- fewer blocked or
|
|
221
|
+
# blind runs than before", is the same inversion as the rising case
|
|
222
|
+
# wearing a recovery story: the headline an operator reads says the
|
|
223
|
+
# thing is getting better while most of its axes report nothing. The
|
|
224
|
+
# exit code alone is not enough here, because the human reads the word.
|
|
225
|
+
summary["direction"] = "still_blind"
|
|
226
|
+
elif newer["fail"] > older["fail"]:
|
|
227
|
+
summary["direction"] = "worsening"
|
|
228
|
+
elif newer["fail"] < older["fail"] or blind_newer < blind_older:
|
|
229
|
+
summary["direction"] = "improving"
|
|
230
|
+
else:
|
|
231
|
+
summary["direction"] = "steady"
|
|
232
|
+
summary["blind_rate_older"] = blind_older
|
|
233
|
+
summary["blind_rate_newer"] = blind_newer
|
|
234
|
+
return summary
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def exit_code(summary):
|
|
238
|
+
"""Weakest link. Blind outranks failed, and INSUFFICIENT DATA is blind.
|
|
239
|
+
|
|
240
|
+
A gate exiting 0 while its own output says it could not establish a trend
|
|
241
|
+
tells CI the axis was checked and was fine. There is no reading of "fewer
|
|
242
|
+
than 2 records" that earns a green.
|
|
243
|
+
"""
|
|
244
|
+
if summary["blind_records"]:
|
|
245
|
+
return COULD_NOT_CHECK
|
|
246
|
+
if summary["direction"] is None:
|
|
247
|
+
return COULD_NOT_CHECK
|
|
248
|
+
if summary["direction"] == "worsening":
|
|
249
|
+
return FAILED
|
|
250
|
+
return PASSED
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
_HEADLINE = {
|
|
254
|
+
"going_blind": "GOING BLIND -- the unevaluable rate is RISING. This is "
|
|
255
|
+
"more urgent than a rising failure rate: a gate that "
|
|
256
|
+
"cannot evaluate can never fail, so blindness renders as "
|
|
257
|
+
"improvement.",
|
|
258
|
+
"still_blind": "STILL BLIND -- the unevaluable rate fell but is NOT zero. "
|
|
259
|
+
"Less blind is not sighted: the remaining blind runs "
|
|
260
|
+
"checked nothing, so this is not an improvement to report.",
|
|
261
|
+
"worsening": "WORSENING -- the failure rate is rising. The gate is "
|
|
262
|
+
"working; the code getting to it is worse.",
|
|
263
|
+
"improving": "IMPROVING -- fewer blocked or blind runs than before.",
|
|
264
|
+
"steady": "STEADY -- no measured change in either rate.",
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def render(summary):
|
|
269
|
+
counts = summary["counts"]
|
|
270
|
+
lines = ["gate-trend: %d record(s) in window" % summary["records"], ""]
|
|
271
|
+
for name in _ORDER:
|
|
272
|
+
label = {"unevaluable": "unevaluable (NOT a pass)",
|
|
273
|
+
_CORRUPT: "corrupt (counted, not skipped)"}.get(name, name)
|
|
274
|
+
lines.append(" %-30s %d" % (label, counts[name]))
|
|
275
|
+
lines.append("")
|
|
276
|
+
if summary["direction"] is None:
|
|
277
|
+
lines.append(
|
|
278
|
+
"trend: INSUFFICIENT DATA -- %d record(s), 2 needed to compare. "
|
|
279
|
+
"Not flat: flat is a claim about change." % summary["records"])
|
|
280
|
+
return "\n".join(lines)
|
|
281
|
+
lines.append("trend: %s" % _HEADLINE[summary["direction"]])
|
|
282
|
+
lines.append("")
|
|
283
|
+
older, newer = summary["older"], summary["newer"]
|
|
284
|
+
lines.append(" %-14s %8s -> %8s" % ("rate", "older", "newer"))
|
|
285
|
+
for name in _ORDER:
|
|
286
|
+
lines.append(" %-14s %7.0f%% -> %7.0f%%"
|
|
287
|
+
% (name, older[name] * 100, newer[name] * 100))
|
|
288
|
+
lines.append("")
|
|
289
|
+
lines.append(" window: %d older then %d newer record(s)"
|
|
290
|
+
% (summary["window"][0], summary["window"][1]))
|
|
291
|
+
if summary["blind_records"]:
|
|
292
|
+
lines.append(
|
|
293
|
+
" NOTE: %d record(s) could not be evaluated or read. Those are "
|
|
294
|
+
"counted here and are neither passes nor failures."
|
|
295
|
+
% summary["blind_records"])
|
|
296
|
+
return "\n".join(lines)
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def main(argv=None):
|
|
300
|
+
ap = _Parser(
|
|
301
|
+
description="Report whether the merge gate is improving, worsening, "
|
|
302
|
+
"or going blind over its recorded verdicts.")
|
|
303
|
+
# No positional argument, deliberately. A stray path then lands on
|
|
304
|
+
# _Parser.error() -> 64, so a mistyped invocation can never be read as a
|
|
305
|
+
# log to judge.
|
|
306
|
+
ap.add_argument("--file", default=DEFAULT_LOG,
|
|
307
|
+
help="JSONL log written by gate-log.py (default: %s)"
|
|
308
|
+
% DEFAULT_LOG)
|
|
309
|
+
ap.add_argument("--window", type=int, default=None,
|
|
310
|
+
help="compare only the most recent N records")
|
|
311
|
+
ap.add_argument("--json", action="store_true", dest="as_json",
|
|
312
|
+
help="emit the summary as JSON")
|
|
313
|
+
args = ap.parse_args(argv)
|
|
314
|
+
|
|
315
|
+
# `is None` and never falsy: --window 0 is a value an operator typed, and
|
|
316
|
+
# swallowing it as "not given" would silently analyse the whole log while
|
|
317
|
+
# the operator believes a window is in force.
|
|
318
|
+
if args.window is not None and args.window < 2:
|
|
319
|
+
ap.error("--window must be at least 2: fewer than 2 records cannot "
|
|
320
|
+
"establish a trend")
|
|
321
|
+
|
|
322
|
+
if not os.path.exists(args.file):
|
|
323
|
+
sys.stderr.write(
|
|
324
|
+
"gate-trend: no log at %s -- UNKNOWN, not a clean history. A gate "
|
|
325
|
+
"whose verdicts were never recorded is not a gate that never "
|
|
326
|
+
"blocked.\n" % args.file)
|
|
327
|
+
return MISSING
|
|
328
|
+
try:
|
|
329
|
+
seq = read_categories(args.file)
|
|
330
|
+
except OSError as exc:
|
|
331
|
+
# An unreachable log is "could not check", never "checked and failed".
|
|
332
|
+
sys.stderr.write("gate-trend: could not read %s: %s\n"
|
|
333
|
+
% (args.file, exc))
|
|
334
|
+
return COULD_NOT_CHECK
|
|
335
|
+
|
|
336
|
+
if not seq:
|
|
337
|
+
sys.stderr.write(
|
|
338
|
+
"gate-trend: %s holds no records -- nothing to trend. An empty log "
|
|
339
|
+
"must never read as 'never blocked'.\n" % args.file)
|
|
340
|
+
return NOTHING
|
|
341
|
+
|
|
342
|
+
if args.window is not None:
|
|
343
|
+
seq = seq[-args.window:]
|
|
344
|
+
|
|
345
|
+
summary = analyse(seq)
|
|
346
|
+
if args.as_json:
|
|
347
|
+
print(json.dumps(summary, indent=2, sort_keys=True))
|
|
348
|
+
else:
|
|
349
|
+
print(render(summary))
|
|
350
|
+
return exit_code(summary)
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
if __name__ == "__main__":
|
|
354
|
+
sys.exit(main())
|
package/tools/model-advisor.py
CHANGED
|
@@ -138,6 +138,24 @@ SWEBENCH_CITATION = {
|
|
|
138
138
|
}
|
|
139
139
|
|
|
140
140
|
|
|
141
|
+
class _Parser(argparse.ArgumentParser):
|
|
142
|
+
"""Usage errors exit 64, not argparse's default 2.
|
|
143
|
+
|
|
144
|
+
In this repo's convention 2 means "could NOT be checked" -- a real
|
|
145
|
+
answer about the subject. A mistyped flag is not that: it is an error
|
|
146
|
+
about the INVOCATION, and nothing about the subject was examined. The
|
|
147
|
+
two call for opposite responses, since retrying cannot fix a typo.
|
|
148
|
+
|
|
149
|
+
argparse exits 2 for every usage error unless this is overridden, so
|
|
150
|
+
every tool needs it. tests/test_tool_exit_contract.py asserts it.
|
|
151
|
+
"""
|
|
152
|
+
|
|
153
|
+
def error(self, message):
|
|
154
|
+
self.print_usage(sys.stderr)
|
|
155
|
+
sys.stderr.write("%s: error: %s\n" % (self.prog, message))
|
|
156
|
+
raise SystemExit(64)
|
|
157
|
+
|
|
158
|
+
|
|
141
159
|
def _num(v):
|
|
142
160
|
"""Non-bool int/float, else None. Never coerces junk to 0."""
|
|
143
161
|
if isinstance(v, bool) or not isinstance(v, (int, float)):
|
|
@@ -460,6 +478,39 @@ def render(adv):
|
|
|
460
478
|
lines.append(" Cheapest rate: none cheaper than the model "
|
|
461
479
|
"already in use")
|
|
462
480
|
|
|
481
|
+
# LOCAL CALIBRATION, shown as a CAVEAT and never as a ranking input.
|
|
482
|
+
#
|
|
483
|
+
# This module's own docstring says quality "is not something any cost
|
|
484
|
+
# record can answer", and that stands: nothing below re-ranks a candidate
|
|
485
|
+
# or weights a saving. advise() is untouched, so the recommendation is
|
|
486
|
+
# provably identical with and without this block.
|
|
487
|
+
#
|
|
488
|
+
# What it adds is one honest local fact. tools/calibration-audit.py scores
|
|
489
|
+
# council votes against council outcomes on THIS workload, so unlike the
|
|
490
|
+
# SWE-bench citation it is not borrowed from someone else's task. It is
|
|
491
|
+
# surfaced ABOVE that citation for exactly that reason: local evidence
|
|
492
|
+
# first, external evidence second.
|
|
493
|
+
#
|
|
494
|
+
# THE LIMIT, restated here because a reader arriving at a cost tool will
|
|
495
|
+
# not have read the audit's header: the audit measures AGREEMENT WITH THE
|
|
496
|
+
# MAJORITY, not correctness. The council outcome is derived from the votes,
|
|
497
|
+
# so a voter partly causes its own label. Treating that as a quality score
|
|
498
|
+
# would be the fabricated authority this tool exists to refuse -- so it is
|
|
499
|
+
# printed as a pointer, with the caveat attached, and never as a number
|
|
500
|
+
# that moves a recommendation.
|
|
501
|
+
lines.append("")
|
|
502
|
+
lines.append(" Local calibration -- a CAVEAT, not a ranking input:")
|
|
503
|
+
lines.append(" Cheaper is not better if the cheaper model agrees with "
|
|
504
|
+
"your council less often.")
|
|
505
|
+
lines.append(" This tool does NOT measure that and does not pretend to. "
|
|
506
|
+
"For the local signal:")
|
|
507
|
+
lines.append(" python3 tools/calibration-audit.py <workspace>")
|
|
508
|
+
lines.append(" Read its header first: it scores agreement with the "
|
|
509
|
+
"majority, NOT accuracy.")
|
|
510
|
+
lines.append(" No artifact records whether the council was right, so no "
|
|
511
|
+
"quality claim is")
|
|
512
|
+
lines.append(" available from any tool in this repo today.")
|
|
513
|
+
|
|
463
514
|
cite = adv["swebench_citation"]
|
|
464
515
|
lines.append("")
|
|
465
516
|
lines.append(" Cited external benchmark (%s) -- NOT a measurement of your "
|
|
@@ -477,7 +528,7 @@ def render(adv):
|
|
|
477
528
|
|
|
478
529
|
|
|
479
530
|
def main(argv=None):
|
|
480
|
-
ap =
|
|
531
|
+
ap = _Parser(
|
|
481
532
|
description="Recommend a cheaper model from this workspace's measured "
|
|
482
533
|
"cost history, and quantify the saving.")
|
|
483
534
|
ap.add_argument("workspace", nargs="?", default=".")
|
package/tools/policy-load.py
CHANGED
|
@@ -63,6 +63,24 @@ class PolicyError(Exception):
|
|
|
63
63
|
"""A policy that must not be handed to a gate."""
|
|
64
64
|
|
|
65
65
|
|
|
66
|
+
class _Parser(argparse.ArgumentParser):
|
|
67
|
+
"""Usage errors exit 64, not argparse's default 2.
|
|
68
|
+
|
|
69
|
+
In this repo's convention 2 means "could NOT be checked" -- a real
|
|
70
|
+
answer about the subject. A mistyped flag is not that: it is an error
|
|
71
|
+
about the INVOCATION, and nothing about the subject was examined. The
|
|
72
|
+
two call for opposite responses, since retrying cannot fix a typo.
|
|
73
|
+
|
|
74
|
+
argparse exits 2 for every usage error unless this is overridden, so
|
|
75
|
+
every tool needs it. tests/test_tool_exit_contract.py asserts it.
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
def error(self, message):
|
|
79
|
+
self.print_usage(sys.stderr)
|
|
80
|
+
sys.stderr.write("%s: error: %s\n" % (self.prog, message))
|
|
81
|
+
raise SystemExit(64)
|
|
82
|
+
|
|
83
|
+
|
|
66
84
|
def _check_max_usd(value):
|
|
67
85
|
# bool is a subclass of int: `true` would otherwise become a $1.00 ceiling.
|
|
68
86
|
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
@@ -157,7 +175,7 @@ def as_args(policy):
|
|
|
157
175
|
|
|
158
176
|
|
|
159
177
|
def main(argv=None):
|
|
160
|
-
ap =
|
|
178
|
+
ap = _Parser(
|
|
161
179
|
description="Load and validate a merge policy file for ci-gate.py.")
|
|
162
180
|
ap.add_argument("--file", default=DEFAULT_FILE,
|
|
163
181
|
help="policy file to load; default {}".format(DEFAULT_FILE))
|