loki-mode 9.8.1 → 9.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -14
- package/SKILL.md +3 -2
- package/VERSION +1 -1
- package/autonomy/loki +122 -1
- package/autonomy/run.sh +49 -2
- package/dashboard/__init__.py +1 -1
- package/dashboard/api_evidence.py +411 -0
- package/dashboard/api_operator.py +283 -0
- package/dashboard/api_phases.py +262 -0
- package/dashboard/api_releases.py +242 -0
- package/dashboard/api_runs.py +477 -0
- package/dashboard/api_tests.py +444 -0
- package/dashboard/api_v2.py +47 -1
- package/dashboard/server.py +54 -0
- package/dashboard/static/index.html +246 -135
- package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
- package/docs/CAPABILITY-BACKLOG.md +53 -0
- package/docs/COMPARISON.md +2 -2
- package/docs/COMPETITIVE-ANALYSIS.md +1 -1
- package/docs/COMPETITIVE-SCORECARD.md +422 -0
- package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
- package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
- package/docs/DEMOS.md +21 -23
- package/docs/HANDOFF-2026-08-03.md +439 -0
- package/docs/INSTALLATION.md +17 -10
- package/docs/OUTCOME-FRONTIER.md +536 -0
- package/docs/PROMPT-ABLATION-RESULT.md +97 -0
- package/docs/TOOLS.md +800 -0
- package/docs/alternative-installations.md +2 -3
- package/docs/audit-logging.md +44 -35
- package/docs/authentication.md +13 -2
- package/docs/authorization.md +87 -81
- package/docs/git-workflow.md +6 -3
- package/docs/metrics.md +15 -16
- package/docs/network-security.md +16 -13
- package/docs/openclaw-integration.md +36 -556
- package/docs/show-hn-post.md +2 -2
- package/docs/siem-integration.md +39 -36
- package/loki-ts/dist/loki.js +18 -18
- package/mcp/__init__.py +1 -1
- package/package.json +1 -1
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
- package/references/confidence-routing.md +18 -1
- package/references/invariant-checks.md +13 -8
- package/references/magic-rarv-integration.md +0 -1
- package/references/multi-provider.md +27 -5
- package/skills/healing.md +4 -2
- package/tools/audit-docs.py +488 -0
- package/tools/baseline-pin.py +19 -1
- package/tools/calibration-audit.py +523 -0
- package/tools/ci-gate.py +19 -1
- package/tools/cost-forecast.py +344 -0
- package/tools/cost-guard.py +19 -1
- package/tools/cost-history.py +19 -1
- package/tools/cost-per-outcome.py +394 -0
- package/tools/estimate-run.py +19 -1
- package/tools/evidence-freshness.py +307 -0
- package/tools/gate-init.py +19 -1
- package/tools/gate-report.py +19 -1
- package/tools/gate-simulate.py +570 -0
- package/tools/gate-trend.py +354 -0
- package/tools/model-advisor.py +52 -1
- package/tools/policy-load.py +19 -1
- package/tools/prompt-cost.py +363 -0
- package/tools/prompt-diff.py +448 -0
- package/tools/prompt-lint.py +448 -0
- package/tools/receipt-bundle.py +72 -2
- package/tools/receipt-diff.py +19 -1
- package/tools/receipt-find.py +19 -1
- package/tools/receipt-stats.py +380 -0
- package/tools/receipt-timeline.py +478 -0
- package/tools/receipt-verify-batch.py +291 -0
- package/tools/run-replay.py +19 -1
- package/tools/signing-status.py +19 -1
- package/tools/token-guard.py +19 -1
- package/tools/token-tax.py +375 -0
- package/tools/tool-index.py +19 -1
- package/tools/verification-tax.py +277 -0
- package/tools/verify-chain.py +361 -0
|
@@ -0,0 +1,570 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Would this policy have blocked my last N runs? Ask before adopting it.
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. Lowering a ceiling is a code change (tools/policy-load.py
|
|
5
|
+
exists to make it one), but a reviewed diff still tells you nothing about its
|
|
6
|
+
CONSEQUENCE. `max_usd: 2.00` looks reasonable in a pull request and is
|
|
7
|
+
indistinguishable, on the page, from a ceiling that would have blocked eleven
|
|
8
|
+
of the last twelve merges. Teams find out by merging it and watching CI go red,
|
|
9
|
+
which is the most expensive possible way to learn a number.
|
|
10
|
+
|
|
11
|
+
So: replay the recorded history against a CANDIDATE policy and report which
|
|
12
|
+
runs it would have blocked. Nothing is enforced, nothing is written, no
|
|
13
|
+
provider is contacted. Simulation is free.
|
|
14
|
+
|
|
15
|
+
tools/gate-simulate.py [workspace] --policy candidate.json
|
|
16
|
+
|
|
17
|
+
THE ONE PROPERTY THAT MAKES THE ANSWER WORTH ANYTHING:
|
|
18
|
+
|
|
19
|
+
A RUN THAT CANNOT BE SIMULATED IS UNEVALUABLE, AND IS EXCLUDED FROM THE
|
|
20
|
+
BLOCK COUNT. NEVER COUNTED AS PASSING.
|
|
21
|
+
|
|
22
|
+
A run whose cost was never measured carries ZERO information about whether it
|
|
23
|
+
would clear a cost ceiling. Sliding it into the "would not have blocked" pile
|
|
24
|
+
produces the single most dangerous output this tool could emit: "0 of 5 runs
|
|
25
|
+
would have been blocked" when three of the five were unmeasurable. That reads
|
|
26
|
+
as a green light to adopt the policy, it is derived from two data points, and
|
|
27
|
+
it is loudest exactly when the instrumentation is broken and it is least
|
|
28
|
+
entitled to speak. It is the same lie -- absence rendered as compliance -- that
|
|
29
|
+
this repo has now paid for across seventeen surfaces.
|
|
30
|
+
|
|
31
|
+
Hence every count in the output states its BASIS: "2 of 5 evaluable". A bare
|
|
32
|
+
count is a number without a denominator, and a denominator is the only thing
|
|
33
|
+
that distinguishes a clean history from a blind one.
|
|
34
|
+
|
|
35
|
+
WHY THIS IS A GATE and not an advisor, despite only ever reporting. The name
|
|
36
|
+
says gate-*, so tests/test_tool_exit_contract.py holds it to the strictest rule
|
|
37
|
+
in the file: it must never exit 0 while its own output says it could not
|
|
38
|
+
evaluate. That is the right rule for it. This tool's whole purpose is to be the
|
|
39
|
+
input to an adoption decision, and an adoption decision made from a partly
|
|
40
|
+
blind simulation is worse than one made from no simulation, because it comes
|
|
41
|
+
with a number attached. So any unevaluable run exits 2, and the per-run table
|
|
42
|
+
still shows every verdict it did manage to reach.
|
|
43
|
+
|
|
44
|
+
PRECEDENCE, inherited from ci-gate.py's own docstring: BLIND OUTRANKS BLOCKED.
|
|
45
|
+
When the history holds both an unevaluable run and one the policy would block,
|
|
46
|
+
this exits 2, not 1. With 1 the operator lowers the ceiling, re-runs, sees the
|
|
47
|
+
block clear, and is still blind on the runs that were never measured.
|
|
48
|
+
|
|
49
|
+
NOTHING HERE RE-IMPLEMENTS A VALIDATOR OR A PREDICATE:
|
|
50
|
+
|
|
51
|
+
the policy schema tools/policy-load.py, by SUBPROCESS. Its docstring is
|
|
52
|
+
explicit that a loader shrugging at bad input yields a
|
|
53
|
+
gate with nothing to enforce. A candidate policy this
|
|
54
|
+
tool blesses is one an operator is about to hand to a
|
|
55
|
+
real gate, so it must clear the SAME bar, decided by the
|
|
56
|
+
same code. Refusing here names the file.
|
|
57
|
+
measured vs not record_is_measured() in autonomy/lib/efficiency_cost.py,
|
|
58
|
+
imported directly. row_usd() below maps a HISTORY ROW
|
|
59
|
+
onto it -- see that docstring for why the mapping is a
|
|
60
|
+
key mapping and not a second copy of the rule.
|
|
61
|
+
reading the history load() in tools/cost-history.py, which owns the file
|
|
62
|
+
format and already counts corrupt lines rather than
|
|
63
|
+
dropping them.
|
|
64
|
+
receipt integrity verify() in autonomy/lib/proof-verify.py.
|
|
65
|
+
the newest receipt the same glob convention as ci-gate.py/baseline-pin.py.
|
|
66
|
+
|
|
67
|
+
A MEASURED $0.00 IS DATA AND IS SIMULATED AS 0. Zero is falsy, so every guard
|
|
68
|
+
here is an explicit `is None`. A truthiness guard would silently reclassify the
|
|
69
|
+
cheapest real runs as unmeasurable, which is this same lie pointed the other
|
|
70
|
+
way and would make a free-tier history look entirely blind.
|
|
71
|
+
|
|
72
|
+
BOTH POLICY AXES ARE SIMULATED, because simulating one and ignoring the other
|
|
73
|
+
is the vacuously-green shape one level up: a report headed "0 would be blocked"
|
|
74
|
+
for a policy whose require_receipt axis was never looked at. The receipt axis
|
|
75
|
+
resolves the row's recorded workspace to its newest receipt and asks verify().
|
|
76
|
+
A workspace that no longer exists, or holds no receipt, is UNEVALUABLE on that
|
|
77
|
+
axis -- the run may well have had one at the time; a deleted directory is an
|
|
78
|
+
absent measurement, not a failure. If policy-load ever grows a key this tool
|
|
79
|
+
cannot simulate, the run is refused NAMING THE KEY rather than quietly scored
|
|
80
|
+
on the axes it does understand.
|
|
81
|
+
|
|
82
|
+
Usage:
|
|
83
|
+
tools/gate-simulate.py [workspace] --policy <file> [--json]
|
|
84
|
+
tools/gate-simulate.py [workspace] --policy <file> --file <history.jsonl>
|
|
85
|
+
|
|
86
|
+
Exit: 0 every run evaluable and none blocked, 1 a run would be BLOCKED,
|
|
87
|
+
2 a run could not be evaluated (or the history is unreadable), 3 no history to
|
|
88
|
+
simulate against, 64 usage error, 66 the policy or history file is missing.
|
|
89
|
+
"""
|
|
90
|
+
|
|
91
|
+
import sys
|
|
92
|
+
|
|
93
|
+
sys.dont_write_bytecode = True
|
|
94
|
+
|
|
95
|
+
import argparse # noqa: E402
|
|
96
|
+
import glob # noqa: E402
|
|
97
|
+
import importlib.util # noqa: E402
|
|
98
|
+
import json # noqa: E402
|
|
99
|
+
import os # noqa: E402
|
|
100
|
+
import subprocess # noqa: E402
|
|
101
|
+
|
|
102
|
+
_HERE = os.path.dirname(os.path.abspath(__file__))
|
|
103
|
+
_ROOT = os.path.dirname(_HERE)
|
|
104
|
+
_LIB = os.path.join(_ROOT, "autonomy", "lib")
|
|
105
|
+
sys.path.insert(0, _LIB)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _load(name, path):
|
|
109
|
+
"""Import a hyphenated file as a module. Hyphens block a plain import."""
|
|
110
|
+
spec = importlib.util.spec_from_file_location(name, path)
|
|
111
|
+
mod = importlib.util.module_from_spec(spec)
|
|
112
|
+
spec.loader.exec_module(mod)
|
|
113
|
+
return mod
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
_pv = _load("proof_verify", os.path.join(_LIB, "proof-verify.py"))
|
|
117
|
+
_ch = _load("cost_history", os.path.join(_HERE, "cost-history.py"))
|
|
118
|
+
|
|
119
|
+
from efficiency_cost import record_is_measured # noqa: E402
|
|
120
|
+
|
|
121
|
+
_POLICY_LOAD = os.path.join(_HERE, "policy-load.py")
|
|
122
|
+
|
|
123
|
+
PASS, BLOCKED, CANNOT, NOTHING = 0, 1, 2, 3
|
|
124
|
+
USAGE, NO_INPUT = 64, 66
|
|
125
|
+
|
|
126
|
+
_STATE = {PASS: "WOULD PASS", BLOCKED: "WOULD BLOCK",
|
|
127
|
+
CANNOT: "UNEVALUABLE", NOTHING: "NOTHING TO CHECK"}
|
|
128
|
+
|
|
129
|
+
DEFAULT_HISTORY = os.path.join(".loki", "cost-history.jsonl")
|
|
130
|
+
|
|
131
|
+
# Every policy key this tool knows how to replay. Cross-checked against the
|
|
132
|
+
# policy actually loaded: a key policy-load accepts and this cannot simulate is
|
|
133
|
+
# a REFUSAL naming the key, never a silent omission from the score. Scoring a
|
|
134
|
+
# run on the axes we happen to understand, under a headline that names the
|
|
135
|
+
# whole policy, is how a partial simulation passes for a complete one.
|
|
136
|
+
SIMULATABLE_KEYS = ("max_usd", "require_receipt")
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class _Parser(argparse.ArgumentParser):
|
|
140
|
+
"""argparse exits 2 on a usage error; here 2 means "could not check".
|
|
141
|
+
|
|
142
|
+
An unknown flag and a blind instrument are opposite facts and must not
|
|
143
|
+
share an exit code. Left at the default, a typo'd flag reads to a CI job as
|
|
144
|
+
a gate that went blind, and the operator goes hunting for missing
|
|
145
|
+
instrumentation that was never missing.
|
|
146
|
+
"""
|
|
147
|
+
|
|
148
|
+
def error(self, message):
|
|
149
|
+
self.print_usage(sys.stderr)
|
|
150
|
+
print("%s: error: %s" % (self.prog, message), file=sys.stderr)
|
|
151
|
+
raise SystemExit(USAGE)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def load_policy(path):
|
|
155
|
+
"""The validated policy, via policy-load.py as a SUBPROCESS.
|
|
156
|
+
|
|
157
|
+
Returns (policy_dict, None) or (None, reason). The validation itself is
|
|
158
|
+
never restated: policy-load.py owns the schema, the value checks (negative,
|
|
159
|
+
NaN, bool-as-int) and the enforces-nothing rule, and a candidate blessed
|
|
160
|
+
here is one an operator is about to hand to a real gate. Two validators
|
|
161
|
+
drift, and the drift shows up as a policy this tool approved and the gate
|
|
162
|
+
then rejected.
|
|
163
|
+
|
|
164
|
+
policy-load exits 1 for BOTH "file missing" and "file invalid", so the
|
|
165
|
+
missing case is settled by stat() here BEFORE the subprocess runs. They
|
|
166
|
+
need different exit codes (66 vs the refusal) because they are different
|
|
167
|
+
operator actions: create the file, or fix the file.
|
|
168
|
+
"""
|
|
169
|
+
if not os.path.exists(path):
|
|
170
|
+
return None, ("no policy file at %s -- there is no candidate to "
|
|
171
|
+
"simulate." % path)
|
|
172
|
+
proc = subprocess.run(
|
|
173
|
+
[sys.executable, _POLICY_LOAD, "--file", path, "--json"],
|
|
174
|
+
capture_output=True, text=True, timeout=60)
|
|
175
|
+
if proc.returncode != 0:
|
|
176
|
+
detail = (proc.stderr or proc.stdout).strip() or "no detail reported"
|
|
177
|
+
return None, ("policy %s was REFUSED by tools/policy-load.py, so it "
|
|
178
|
+
"must not be simulated: a simulation of an invalid "
|
|
179
|
+
"policy answers a question about a policy no gate would "
|
|
180
|
+
"accept.\n %s" % (path, detail))
|
|
181
|
+
try:
|
|
182
|
+
policy = json.loads(proc.stdout)
|
|
183
|
+
except ValueError as exc:
|
|
184
|
+
return None, ("could not read the validated policy back from "
|
|
185
|
+
"tools/policy-load.py --json for %s: %s" % (path, exc))
|
|
186
|
+
if not isinstance(policy, dict):
|
|
187
|
+
return None, ("tools/policy-load.py returned a %s for %s, not a policy "
|
|
188
|
+
"object" % (type(policy).__name__, path))
|
|
189
|
+
return policy, None
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def _num(v):
|
|
193
|
+
"""A number as itself; None, "", or a bool as None. A real 0 survives as 0."""
|
|
194
|
+
if isinstance(v, bool) or not isinstance(v, (int, float)):
|
|
195
|
+
return None
|
|
196
|
+
return v
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def row_usd(row):
|
|
200
|
+
"""The measured USD for one HISTORY ROW, or None when it was not measured.
|
|
201
|
+
|
|
202
|
+
THE ONE PLACE the measured/unmeasured distinction is decided for a row, so
|
|
203
|
+
it cannot drift between the loader and the reporter.
|
|
204
|
+
|
|
205
|
+
WHY THIS IS NOT cost-history.measured_usd(), which is right there and which
|
|
206
|
+
a reviewer will reasonably ask about. That function reads a RECEIPT COST
|
|
207
|
+
BLOCK -- {usd, input_tokens, output_tokens, ...} -- and funnels every field
|
|
208
|
+
through record_is_measured(). A history ROW has no token fields at all, so
|
|
209
|
+
record_is_measured() answers False for every row, including one carrying a
|
|
210
|
+
perfectly good `usd`. Pointing it at a row does not reuse the rule, it
|
|
211
|
+
misapplies it, and it would report an entire measured history as blind.
|
|
212
|
+
|
|
213
|
+
What IS reused is the root predicate itself, imported from
|
|
214
|
+
autonomy/lib/efficiency_cost.py and never restated. cost-history.py writes
|
|
215
|
+
the `measured` flag at record time BY CALLING record_is_measured(), so
|
|
216
|
+
trusting that flag reuses the rule at one remove, which is the whole point
|
|
217
|
+
of it being recorded.
|
|
218
|
+
|
|
219
|
+
THE ORDER OF THE LEGACY BRANCH IS LOAD-BEARING. record_is_measured() is a
|
|
220
|
+
truthiness predicate, so it answers False for a measured $0.0000 --
|
|
221
|
+
correct for its own question ("did this record observe anything?"), wrong
|
|
222
|
+
for this one ("is this dollar figure present?"). So `usd is None` is tested
|
|
223
|
+
FIRST and short-circuits; record_is_measured() only ever adjudicates a
|
|
224
|
+
legacy row whose usd is already absent. Asking it first would drop a real
|
|
225
|
+
zero from a flag-less row while keeping the identical flagged row, so the
|
|
226
|
+
same fact would get two answers depending on which version of
|
|
227
|
+
cost-history.py wrote it.
|
|
228
|
+
|
|
229
|
+
The final test is `is not None`, never truthiness: a genuinely measured
|
|
230
|
+
$0.0000 run (cached, free tier) is a real observation and must be
|
|
231
|
+
simulated as 0.0 against the ceiling, not filed under "cannot evaluate".
|
|
232
|
+
"""
|
|
233
|
+
if not isinstance(row, dict):
|
|
234
|
+
return None
|
|
235
|
+
usd = _num(row.get("usd"))
|
|
236
|
+
if "measured" in row:
|
|
237
|
+
if row.get("measured") is not True:
|
|
238
|
+
return None
|
|
239
|
+
elif usd is None and not record_is_measured(row):
|
|
240
|
+
return None
|
|
241
|
+
return usd
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def newest_receipt(workspace):
|
|
245
|
+
"""The most recent proof.json under a workspace, or None.
|
|
246
|
+
|
|
247
|
+
Same glob convention as ci-gate.py and baseline-pin.py.
|
|
248
|
+
"""
|
|
249
|
+
root = os.path.normpath(workspace)
|
|
250
|
+
if os.path.basename(root) == ".loki":
|
|
251
|
+
root = os.path.dirname(root) or "."
|
|
252
|
+
found = glob.glob(os.path.join(root, ".loki", "proofs", "*", "proof.json"))
|
|
253
|
+
return max(found, key=os.path.getmtime) if found else None
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def simulate_cost(row, max_usd):
|
|
257
|
+
"""(verdict, detail) for one run against a cost ceiling.
|
|
258
|
+
|
|
259
|
+
THE FUNCTION THE WHOLE FILE PROTECTS. Its unmeasured branch is the one
|
|
260
|
+
place absence could become compliance, so it is written once, here, and has
|
|
261
|
+
nowhere to hide.
|
|
262
|
+
|
|
263
|
+
`is None`, never truthiness: a genuinely measured $0.0000 run (cached, free
|
|
264
|
+
tier) is a real observation and must be simulated as 0.0. A falsy guard
|
|
265
|
+
would file every cheap run under "cannot evaluate" and report a blind
|
|
266
|
+
history to a team whose instrumentation is perfect.
|
|
267
|
+
"""
|
|
268
|
+
usd = row_usd(row)
|
|
269
|
+
if usd is None:
|
|
270
|
+
return CANNOT, ("cost was never measured for this run, so whether it "
|
|
271
|
+
"clears a $%.4f ceiling is unknown. Unmeasured is not "
|
|
272
|
+
"$0.00 and is not a pass." % max_usd)
|
|
273
|
+
if usd > max_usd:
|
|
274
|
+
return BLOCKED, ("$%.4f exceeds the $%.4f ceiling" % (usd, max_usd))
|
|
275
|
+
return PASS, "$%.4f is within the $%.4f ceiling" % (usd, max_usd)
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def simulate_receipt(row):
|
|
279
|
+
"""(verdict, detail) for one run against require_receipt.
|
|
280
|
+
|
|
281
|
+
A row records the workspace it came from. The receipt axis therefore needs
|
|
282
|
+
that directory to still be on disk, which for older history it often is
|
|
283
|
+
not. THAT IS UNEVALUABLE, NOT A FAILURE: the run may well have produced a
|
|
284
|
+
perfectly good receipt, and a deleted directory is an absent measurement
|
|
285
|
+
rather than evidence of a missing receipt. Scoring it as a block would
|
|
286
|
+
manufacture an adoption objection out of nothing, which is the same
|
|
287
|
+
dishonesty as manufacturing a pass.
|
|
288
|
+
|
|
289
|
+
Verification itself is verify() in autonomy/lib/proof-verify.py, never
|
|
290
|
+
re-implemented here. But its VERDICT is not consumed whole, and the reason
|
|
291
|
+
is the same rule one level down.
|
|
292
|
+
|
|
293
|
+
verify() answers "is this receipt verified?", and for that question it is
|
|
294
|
+
correct that diff_drift=None (could not re-check against the repo) collapses
|
|
295
|
+
into ok=False -- its own docstring says so by design, because a receipt
|
|
296
|
+
whose central fact cannot be re-checked is not a verified receipt. This tool
|
|
297
|
+
asks a DIFFERENT question: "would the policy have blocked this run?" A
|
|
298
|
+
historical run whose workspace is no longer a resolvable git repo -- branch
|
|
299
|
+
deleted, refs gone, never a repo -- comes back hash_ok=True with
|
|
300
|
+
diff_drift=None. Reporting that as WOULD BLOCK manufactures an adoption
|
|
301
|
+
objection out of a missing measurement, which is exactly the dishonesty the
|
|
302
|
+
paragraph above disclaims, pointed the other way. So the two are split here:
|
|
303
|
+
a receipt that FAILED a check blocks, a receipt that could not BE checked is
|
|
304
|
+
unevaluable.
|
|
305
|
+
|
|
306
|
+
That deliberately makes this axis NARROWER than verify()'s verdict, and
|
|
307
|
+
narrower still than what ci-gate.py enforces (it routes the receipt axis
|
|
308
|
+
through receipt-attest.py, which uses verify_integrity() rather than
|
|
309
|
+
verify()). A simulator stricter than the gate it simulates answers a
|
|
310
|
+
question nobody asked; one that reports blindness as blindness does not.
|
|
311
|
+
"""
|
|
312
|
+
workspace = row.get("workspace")
|
|
313
|
+
if not isinstance(workspace, str) or not workspace:
|
|
314
|
+
return CANNOT, ("this history row records no workspace, so its "
|
|
315
|
+
"receipt cannot be located")
|
|
316
|
+
if not os.path.isdir(workspace):
|
|
317
|
+
return CANNOT, ("workspace %s no longer exists, so whether it carried "
|
|
318
|
+
"a receipt cannot be established now" % workspace)
|
|
319
|
+
receipt = newest_receipt(workspace)
|
|
320
|
+
if receipt is None:
|
|
321
|
+
return BLOCKED, ("no receipt under %s/.loki/proofs/*/proof.json -- "
|
|
322
|
+
"checked, and the answer is no" % workspace)
|
|
323
|
+
try:
|
|
324
|
+
result = _pv.verify(receipt)
|
|
325
|
+
except Exception as exc:
|
|
326
|
+
return CANNOT, ("receipt %s could not be verified: %s" % (receipt, exc))
|
|
327
|
+
if result.get("ok"):
|
|
328
|
+
return PASS, "receipt %s verifies" % receipt
|
|
329
|
+
# Intact receipt, unresolvable repo: checked the receipt, could not check it
|
|
330
|
+
# AGAINST anything. Not a failure, and not a pass.
|
|
331
|
+
if result.get("hash_ok") and result.get("diff_drift") is None:
|
|
332
|
+
return CANNOT, ("receipt %s is intact but could not be re-checked "
|
|
333
|
+
"against its repo (the recorded git state is no longer "
|
|
334
|
+
"resolvable), so whether it would have satisfied "
|
|
335
|
+
"require_receipt is unknown" % receipt)
|
|
336
|
+
return BLOCKED, ("receipt %s FAILS verification: %s"
|
|
337
|
+
% (receipt, result.get("reason") or "no reason recorded"))
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def simulate_run(row, policy):
|
|
341
|
+
"""The worst verdict across every configured axis, for ONE run.
|
|
342
|
+
|
|
343
|
+
WEAKEST LINK, with CANNOT outranking BLOCKED -- the same precedence
|
|
344
|
+
ci-gate.py applies to a live merge, for the same reason. A run that is
|
|
345
|
+
blocked on cost AND blind on receipts is reported blind: fixing the cost
|
|
346
|
+
would otherwise clear the verdict and leave the blindness in place.
|
|
347
|
+
"""
|
|
348
|
+
axes = []
|
|
349
|
+
if "max_usd" in policy:
|
|
350
|
+
verdict, detail = simulate_cost(row, policy["max_usd"])
|
|
351
|
+
axes.append({"axis": "max_usd", "verdict": verdict, "detail": detail})
|
|
352
|
+
if policy.get("require_receipt"):
|
|
353
|
+
verdict, detail = simulate_receipt(row)
|
|
354
|
+
axes.append({"axis": "require_receipt", "verdict": verdict,
|
|
355
|
+
"detail": detail})
|
|
356
|
+
|
|
357
|
+
if not axes:
|
|
358
|
+
# policy-load already refuses an enforces-nothing policy, so this is
|
|
359
|
+
# unreachable through the CLI. Kept because it is the correct answer if
|
|
360
|
+
# it ever becomes reachable: no axis checked is no pass earned.
|
|
361
|
+
return {"verdict": CANNOT, "axes": [],
|
|
362
|
+
"detail": "no axis of this policy applies to this run"}
|
|
363
|
+
|
|
364
|
+
codes = [a["verdict"] for a in axes]
|
|
365
|
+
worst = CANNOT if CANNOT in codes else (
|
|
366
|
+
BLOCKED if BLOCKED in codes else PASS)
|
|
367
|
+
detail = "; ".join(a["detail"] for a in axes if a["verdict"] == worst)
|
|
368
|
+
return {"verdict": worst, "axes": axes, "detail": detail}
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def simulate(policy_path, history_path):
|
|
372
|
+
"""Replay the history against a candidate policy. Returns a verdict dict.
|
|
373
|
+
|
|
374
|
+
Every count carries its denominator. There is no branch here that reports a
|
|
375
|
+
block count without also reporting how many runs were evaluable of how
|
|
376
|
+
many, because a bare "0 blocked" over a blind history is the exact output
|
|
377
|
+
this file exists to make impossible.
|
|
378
|
+
"""
|
|
379
|
+
base = {
|
|
380
|
+
"policy_file": policy_path,
|
|
381
|
+
"history_file": history_path,
|
|
382
|
+
"policy": None,
|
|
383
|
+
"runs": 0,
|
|
384
|
+
"evaluable": 0,
|
|
385
|
+
"unevaluable": 0,
|
|
386
|
+
"blocked": 0,
|
|
387
|
+
"would_pass": 0,
|
|
388
|
+
"corrupt_lines": 0,
|
|
389
|
+
"results": [],
|
|
390
|
+
"basis": None,
|
|
391
|
+
"why": None,
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
policy, refusal = load_policy(policy_path)
|
|
395
|
+
if policy is None:
|
|
396
|
+
base["status"] = "policy_refused"
|
|
397
|
+
# A missing file is a different operator action from an invalid one.
|
|
398
|
+
base["exit_code"] = (NO_INPUT if not os.path.exists(policy_path)
|
|
399
|
+
else CANNOT)
|
|
400
|
+
base["why"] = refusal
|
|
401
|
+
return base
|
|
402
|
+
base["policy"] = policy
|
|
403
|
+
|
|
404
|
+
unsupported = sorted(k for k in policy if k not in SIMULATABLE_KEYS)
|
|
405
|
+
if unsupported:
|
|
406
|
+
base["status"] = "policy_unsimulatable"
|
|
407
|
+
base["exit_code"] = CANNOT
|
|
408
|
+
base["why"] = (
|
|
409
|
+
"policy %s uses key(s) this tool cannot replay: %s. Reporting a "
|
|
410
|
+
"block count over the remaining axes would headline the whole "
|
|
411
|
+
"policy while silently ignoring part of it." % (
|
|
412
|
+
policy_path, ", ".join(unsupported)))
|
|
413
|
+
return base
|
|
414
|
+
|
|
415
|
+
if not os.path.exists(history_path):
|
|
416
|
+
base["status"] = "no_history"
|
|
417
|
+
base["exit_code"] = NO_INPUT
|
|
418
|
+
base["why"] = (
|
|
419
|
+
"no history file at %s -- record runs with "
|
|
420
|
+
"tools/cost-history.py record first. Simulating against nothing "
|
|
421
|
+
"is not a clean result." % history_path)
|
|
422
|
+
return base
|
|
423
|
+
|
|
424
|
+
rows, corrupt = _ch.load(history_path)
|
|
425
|
+
if rows is None:
|
|
426
|
+
base["status"] = "unreadable"
|
|
427
|
+
base["exit_code"] = CANNOT
|
|
428
|
+
base["why"] = ("history at %s exists but could not be read; the "
|
|
429
|
+
"instrument is blind, which is not the same as a "
|
|
430
|
+
"policy that blocks nothing." % history_path)
|
|
431
|
+
return base
|
|
432
|
+
|
|
433
|
+
base["corrupt_lines"] = corrupt
|
|
434
|
+
base["runs"] = len(rows)
|
|
435
|
+
|
|
436
|
+
if not rows:
|
|
437
|
+
base["status"] = "empty_history"
|
|
438
|
+
base["exit_code"] = NOTHING
|
|
439
|
+
base["why"] = (
|
|
440
|
+
"history %s holds no runs (%d corrupt line(s)); a policy simulated "
|
|
441
|
+
"against zero runs blocks nothing, and that is not evidence it is "
|
|
442
|
+
"safe to adopt." % (history_path, corrupt))
|
|
443
|
+
return base
|
|
444
|
+
|
|
445
|
+
for index, row in enumerate(rows):
|
|
446
|
+
outcome = simulate_run(row, policy)
|
|
447
|
+
base["results"].append({
|
|
448
|
+
"index": index,
|
|
449
|
+
"workspace": row.get("workspace"),
|
|
450
|
+
"usd": row_usd(row), # None stays None. Never 0 as a stand-in.
|
|
451
|
+
"verdict": outcome["verdict"],
|
|
452
|
+
"state": _STATE[outcome["verdict"]],
|
|
453
|
+
"detail": outcome["detail"],
|
|
454
|
+
"axes": outcome["axes"],
|
|
455
|
+
})
|
|
456
|
+
|
|
457
|
+
verdicts = [r["verdict"] for r in base["results"]]
|
|
458
|
+
base["unevaluable"] = verdicts.count(CANNOT)
|
|
459
|
+
base["blocked"] = verdicts.count(BLOCKED)
|
|
460
|
+
base["would_pass"] = verdicts.count(PASS)
|
|
461
|
+
base["evaluable"] = base["blocked"] + base["would_pass"]
|
|
462
|
+
|
|
463
|
+
# THE BASIS, on every count and never behind a flag. "1 of 5 would have
|
|
464
|
+
# been blocked" is a different fact depending on whether 5, 2 or 0 runs
|
|
465
|
+
# could be evaluated, and the reader cannot recover the difference.
|
|
466
|
+
base["basis"] = ("%d of %d run(s) were evaluable against this policy; "
|
|
467
|
+
"%d could not be simulated and are excluded from the "
|
|
468
|
+
"block count." % (base["evaluable"], base["runs"],
|
|
469
|
+
base["unevaluable"]))
|
|
470
|
+
|
|
471
|
+
if base["unevaluable"]:
|
|
472
|
+
# BLIND OUTRANKS BLOCKED. Fixing a ceiling clears a block and leaves
|
|
473
|
+
# the blindness exactly where it was.
|
|
474
|
+
base["status"] = "partly_unevaluable"
|
|
475
|
+
base["exit_code"] = CANNOT
|
|
476
|
+
base["why"] = (
|
|
477
|
+
"%d of %d run(s) could NOT be simulated against this policy, so "
|
|
478
|
+
"this simulation cannot tell you whether adopting it is safe. An "
|
|
479
|
+
"unevaluable run is not a run that would have passed."
|
|
480
|
+
% (base["unevaluable"], base["runs"]))
|
|
481
|
+
elif base["blocked"]:
|
|
482
|
+
base["status"] = "would_block"
|
|
483
|
+
base["exit_code"] = BLOCKED
|
|
484
|
+
base["why"] = ("%d of %d evaluable run(s) would have been BLOCKED by "
|
|
485
|
+
"this policy." % (base["blocked"], base["evaluable"]))
|
|
486
|
+
else:
|
|
487
|
+
base["status"] = "would_pass"
|
|
488
|
+
base["exit_code"] = PASS
|
|
489
|
+
base["why"] = None
|
|
490
|
+
|
|
491
|
+
return base
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
def render(d):
|
|
495
|
+
if d["status"] == "policy_refused":
|
|
496
|
+
return "CANNOT SIMULATE: %s" % d["why"]
|
|
497
|
+
if d["status"] == "policy_unsimulatable":
|
|
498
|
+
return "CANNOT SIMULATE: %s" % d["why"]
|
|
499
|
+
if d["status"] in ("no_history", "empty_history", "unreadable"):
|
|
500
|
+
return "NOTHING TO SIMULATE: %s" % d["why"]
|
|
501
|
+
|
|
502
|
+
policy = ", ".join("%s=%s" % (k, json.dumps(d["policy"][k]))
|
|
503
|
+
for k in sorted(d["policy"]))
|
|
504
|
+
lines = ["policy: %s" % policy,
|
|
505
|
+
"history: %s" % d["history_file"],
|
|
506
|
+
""]
|
|
507
|
+
# THE COST COLUMN IS ONLY PRINTED WHEN THE POLICY HAS A COST AXIS, and the
|
|
508
|
+
# reason is not cosmetic. An unmeasured cost renders as UNKNOWN, which is a
|
|
509
|
+
# cannot-evaluate token; under a receipt-only policy that column is not part
|
|
510
|
+
# of the verdict at all, so printing UNKNOWN there would put "I could not
|
|
511
|
+
# evaluate" in the output of a run that WAS fully evaluated on every axis
|
|
512
|
+
# the policy configures -- a gate saying it is blind while honestly exiting
|
|
513
|
+
# 0. The rule this tool is held to reads exit code against output text, so
|
|
514
|
+
# an irrelevant column must not speak in that vocabulary.
|
|
515
|
+
costed = "max_usd" in d["policy"]
|
|
516
|
+
header = ("%-5s %-12s %-12s %s" % ("RUN", "COST", "VERDICT", "DETAIL")
|
|
517
|
+
if costed else "%-5s %-12s %s" % ("RUN", "VERDICT", "DETAIL"))
|
|
518
|
+
lines.append(header)
|
|
519
|
+
for row in d["results"]:
|
|
520
|
+
if not costed:
|
|
521
|
+
lines.append("%-5d %-12s %s" % (
|
|
522
|
+
row["index"], row["state"], row["detail"]))
|
|
523
|
+
continue
|
|
524
|
+
# UNKNOWN, never $0.00. Absent is not zero, and a measured zero is not
|
|
525
|
+
# absent -- so this is `is None` and prints a real 0 as $0.0000.
|
|
526
|
+
cost = "UNKNOWN" if row["usd"] is None else "$%.4f" % row["usd"]
|
|
527
|
+
lines.append("%-5d %-12s %-12s %s" % (
|
|
528
|
+
row["index"], cost, row["state"], row["detail"]))
|
|
529
|
+
lines.append("")
|
|
530
|
+
if d["corrupt_lines"]:
|
|
531
|
+
lines.append("%d CORRUPT line(s) in the history -- counted, not "
|
|
532
|
+
"skipped." % d["corrupt_lines"])
|
|
533
|
+
lines.append("%d of %d evaluable run(s) WOULD HAVE BEEN BLOCKED"
|
|
534
|
+
% (d["blocked"], d["evaluable"]))
|
|
535
|
+
lines.append("basis: %s" % d["basis"])
|
|
536
|
+
if d["why"]:
|
|
537
|
+
lines.append(d["why"])
|
|
538
|
+
return "\n".join(lines)
|
|
539
|
+
|
|
540
|
+
|
|
541
|
+
def main(argv=None):
|
|
542
|
+
ap = _Parser(
|
|
543
|
+
description="Replay recorded cost history against a candidate policy.")
|
|
544
|
+
ap.add_argument("workspace", nargs="?", default=".",
|
|
545
|
+
help="workspace root (or its .loki dir); default .")
|
|
546
|
+
ap.add_argument("--policy", required=True,
|
|
547
|
+
help="candidate policy file to simulate")
|
|
548
|
+
ap.add_argument("--file", dest="history",
|
|
549
|
+
help="history JSONL; default <workspace>/%s"
|
|
550
|
+
% DEFAULT_HISTORY)
|
|
551
|
+
ap.add_argument("--json", action="store_true", dest="as_json",
|
|
552
|
+
help="emit the simulation as JSON")
|
|
553
|
+
args = ap.parse_args(argv)
|
|
554
|
+
|
|
555
|
+
root = os.path.normpath(args.workspace)
|
|
556
|
+
if os.path.basename(root) == ".loki":
|
|
557
|
+
root = os.path.dirname(root) or "."
|
|
558
|
+
history = args.history or os.path.join(root, DEFAULT_HISTORY)
|
|
559
|
+
|
|
560
|
+
d = simulate(args.policy, history)
|
|
561
|
+
# --json stays JSON on EVERY path, refusals included: a caller that parses
|
|
562
|
+
# stdout must not have to special-case the failure branch (fixed once
|
|
563
|
+
# already in 9b879986). Diagnostics go to stderr.
|
|
564
|
+
print(json.dumps(d, indent=2, sort_keys=True) if args.as_json
|
|
565
|
+
else render(d))
|
|
566
|
+
return d["exit_code"]
|
|
567
|
+
|
|
568
|
+
|
|
569
|
+
if __name__ == "__main__":
|
|
570
|
+
sys.exit(main())
|