loki-mode 9.8.1 → 9.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +19 -14
  2. package/SKILL.md +3 -2
  3. package/VERSION +1 -1
  4. package/autonomy/loki +122 -1
  5. package/autonomy/run.sh +49 -2
  6. package/dashboard/__init__.py +1 -1
  7. package/dashboard/api_evidence.py +411 -0
  8. package/dashboard/api_operator.py +283 -0
  9. package/dashboard/api_phases.py +262 -0
  10. package/dashboard/api_releases.py +242 -0
  11. package/dashboard/api_runs.py +477 -0
  12. package/dashboard/api_tests.py +444 -0
  13. package/dashboard/api_v2.py +47 -1
  14. package/dashboard/server.py +54 -0
  15. package/dashboard/static/index.html +246 -135
  16. package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
  17. package/docs/CAPABILITY-BACKLOG.md +53 -0
  18. package/docs/COMPARISON.md +2 -2
  19. package/docs/COMPETITIVE-ANALYSIS.md +1 -1
  20. package/docs/COMPETITIVE-SCORECARD.md +422 -0
  21. package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
  22. package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
  23. package/docs/DEMOS.md +21 -23
  24. package/docs/HANDOFF-2026-08-03.md +439 -0
  25. package/docs/INSTALLATION.md +17 -10
  26. package/docs/OUTCOME-FRONTIER.md +536 -0
  27. package/docs/PROMPT-ABLATION-RESULT.md +97 -0
  28. package/docs/TOOLS.md +800 -0
  29. package/docs/alternative-installations.md +2 -3
  30. package/docs/audit-logging.md +44 -35
  31. package/docs/authentication.md +13 -2
  32. package/docs/authorization.md +87 -81
  33. package/docs/git-workflow.md +6 -3
  34. package/docs/metrics.md +15 -16
  35. package/docs/network-security.md +16 -13
  36. package/docs/openclaw-integration.md +36 -556
  37. package/docs/show-hn-post.md +2 -2
  38. package/docs/siem-integration.md +39 -36
  39. package/loki-ts/dist/loki.js +18 -18
  40. package/mcp/__init__.py +1 -1
  41. package/package.json +1 -1
  42. package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
  43. package/references/confidence-routing.md +18 -1
  44. package/references/invariant-checks.md +13 -8
  45. package/references/magic-rarv-integration.md +0 -1
  46. package/references/multi-provider.md +27 -5
  47. package/skills/healing.md +4 -2
  48. package/tools/audit-docs.py +488 -0
  49. package/tools/baseline-pin.py +19 -1
  50. package/tools/calibration-audit.py +523 -0
  51. package/tools/ci-gate.py +19 -1
  52. package/tools/cost-forecast.py +344 -0
  53. package/tools/cost-guard.py +19 -1
  54. package/tools/cost-history.py +19 -1
  55. package/tools/cost-per-outcome.py +394 -0
  56. package/tools/estimate-run.py +19 -1
  57. package/tools/evidence-freshness.py +307 -0
  58. package/tools/gate-init.py +19 -1
  59. package/tools/gate-report.py +19 -1
  60. package/tools/gate-simulate.py +570 -0
  61. package/tools/gate-trend.py +354 -0
  62. package/tools/model-advisor.py +52 -1
  63. package/tools/policy-load.py +19 -1
  64. package/tools/prompt-cost.py +363 -0
  65. package/tools/prompt-diff.py +448 -0
  66. package/tools/prompt-lint.py +448 -0
  67. package/tools/receipt-bundle.py +72 -2
  68. package/tools/receipt-diff.py +19 -1
  69. package/tools/receipt-find.py +19 -1
  70. package/tools/receipt-stats.py +380 -0
  71. package/tools/receipt-timeline.py +478 -0
  72. package/tools/receipt-verify-batch.py +291 -0
  73. package/tools/run-replay.py +19 -1
  74. package/tools/signing-status.py +19 -1
  75. package/tools/token-guard.py +19 -1
  76. package/tools/token-tax.py +375 -0
  77. package/tools/tool-index.py +19 -1
  78. package/tools/verification-tax.py +277 -0
  79. package/tools/verify-chain.py +361 -0
@@ -0,0 +1,570 @@
1
+ #!/usr/bin/env python3
2
+ """Would this policy have blocked my last N runs? Ask before adopting it.
3
+
4
+ WHY THIS EXISTS. Lowering a ceiling is a code change (tools/policy-load.py
5
+ exists to make it one), but a reviewed diff still tells you nothing about its
6
+ CONSEQUENCE. `max_usd: 2.00` looks reasonable in a pull request and is
7
+ indistinguishable, on the page, from a ceiling that would have blocked eleven
8
+ of the last twelve merges. Teams find out by merging it and watching CI go red,
9
+ which is the most expensive possible way to learn a number.
10
+
11
+ So: replay the recorded history against a CANDIDATE policy and report which
12
+ runs it would have blocked. Nothing is enforced, nothing is written, no
13
+ provider is contacted. Simulation is free.
14
+
15
+ tools/gate-simulate.py [workspace] --policy candidate.json
16
+
17
+ THE ONE PROPERTY THAT MAKES THE ANSWER WORTH ANYTHING:
18
+
19
+ A RUN THAT CANNOT BE SIMULATED IS UNEVALUABLE, AND IS EXCLUDED FROM THE
20
+ BLOCK COUNT. NEVER COUNTED AS PASSING.
21
+
22
+ A run whose cost was never measured carries ZERO information about whether it
23
+ would clear a cost ceiling. Sliding it into the "would not have blocked" pile
24
+ produces the single most dangerous output this tool could emit: "0 of 5 runs
25
+ would have been blocked" when three of the five were unmeasurable. That reads
26
+ as a green light to adopt the policy, it is derived from two data points, and
27
+ it is loudest exactly when the instrumentation is broken and it is least
28
+ entitled to speak. It is the same lie -- absence rendered as compliance -- that
29
+ this repo has now paid for across seventeen surfaces.
30
+
31
+ Hence every count in the output states its BASIS: "2 of 5 evaluable". A bare
32
+ count is a number without a denominator, and a denominator is the only thing
33
+ that distinguishes a clean history from a blind one.
34
+
35
+ WHY THIS IS A GATE and not an advisor, despite only ever reporting. The name
36
+ says gate-*, so tests/test_tool_exit_contract.py holds it to the strictest rule
37
+ in the file: it must never exit 0 while its own output says it could not
38
+ evaluate. That is the right rule for it. This tool's whole purpose is to be the
39
+ input to an adoption decision, and an adoption decision made from a partly
40
+ blind simulation is worse than one made from no simulation, because it comes
41
+ with a number attached. So any unevaluable run exits 2, and the per-run table
42
+ still shows every verdict it did manage to reach.
43
+
44
+ PRECEDENCE, inherited from ci-gate.py's own docstring: BLIND OUTRANKS BLOCKED.
45
+ When the history holds both an unevaluable run and one the policy would block,
46
+ this exits 2, not 1. With 1 the operator lowers the ceiling, re-runs, sees the
47
+ block clear, and is still blind on the runs that were never measured.
48
+
49
+ NOTHING HERE RE-IMPLEMENTS A VALIDATOR OR A PREDICATE:
50
+
51
+ the policy schema tools/policy-load.py, by SUBPROCESS. Its docstring is
52
+ explicit that a loader shrugging at bad input yields a
53
+ gate with nothing to enforce. A candidate policy this
54
+ tool blesses is one an operator is about to hand to a
55
+ real gate, so it must clear the SAME bar, decided by the
56
+ same code. Refusing here names the file.
57
+ measured vs not record_is_measured() in autonomy/lib/efficiency_cost.py,
58
+ imported directly. row_usd() below maps a HISTORY ROW
59
+ onto it -- see that docstring for why the mapping is a
60
+ key mapping and not a second copy of the rule.
61
+ reading the history load() in tools/cost-history.py, which owns the file
62
+ format and already counts corrupt lines rather than
63
+ dropping them.
64
+ receipt integrity verify() in autonomy/lib/proof-verify.py.
65
+ the newest receipt the same glob convention as ci-gate.py/baseline-pin.py.
66
+
67
+ A MEASURED $0.00 IS DATA AND IS SIMULATED AS 0. Zero is falsy, so every guard
68
+ here is an explicit `is None`. A truthiness guard would silently reclassify the
69
+ cheapest real runs as unmeasurable, which is this same lie pointed the other
70
+ way and would make a free-tier history look entirely blind.
71
+
72
+ BOTH POLICY AXES ARE SIMULATED, because simulating one and ignoring the other
73
+ is the vacuously-green shape one level up: a report headed "0 would be blocked"
74
+ for a policy whose require_receipt axis was never looked at. The receipt axis
75
+ resolves the row's recorded workspace to its newest receipt and asks verify().
76
+ A workspace that no longer exists, or holds no receipt, is UNEVALUABLE on that
77
+ axis -- the run may well have had one at the time; a deleted directory is an
78
+ absent measurement, not a failure. If policy-load ever grows a key this tool
79
+ cannot simulate, the run is refused NAMING THE KEY rather than quietly scored
80
+ on the axes it does understand.
81
+
82
+ Usage:
83
+ tools/gate-simulate.py [workspace] --policy <file> [--json]
84
+ tools/gate-simulate.py [workspace] --policy <file> --file <history.jsonl>
85
+
86
+ Exit: 0 every run evaluable and none blocked, 1 a run would be BLOCKED,
87
+ 2 a run could not be evaluated (or the history is unreadable), 3 no history to
88
+ simulate against, 64 usage error, 66 the policy or history file is missing.
89
+ """
90
+
91
+ import sys
92
+
93
+ sys.dont_write_bytecode = True
94
+
95
+ import argparse # noqa: E402
96
+ import glob # noqa: E402
97
+ import importlib.util # noqa: E402
98
+ import json # noqa: E402
99
+ import os # noqa: E402
100
+ import subprocess # noqa: E402
101
+
102
+ _HERE = os.path.dirname(os.path.abspath(__file__))
103
+ _ROOT = os.path.dirname(_HERE)
104
+ _LIB = os.path.join(_ROOT, "autonomy", "lib")
105
+ sys.path.insert(0, _LIB)
106
+
107
+
108
+ def _load(name, path):
109
+ """Import a hyphenated file as a module. Hyphens block a plain import."""
110
+ spec = importlib.util.spec_from_file_location(name, path)
111
+ mod = importlib.util.module_from_spec(spec)
112
+ spec.loader.exec_module(mod)
113
+ return mod
114
+
115
+
116
+ _pv = _load("proof_verify", os.path.join(_LIB, "proof-verify.py"))
117
+ _ch = _load("cost_history", os.path.join(_HERE, "cost-history.py"))
118
+
119
+ from efficiency_cost import record_is_measured # noqa: E402
120
+
121
+ _POLICY_LOAD = os.path.join(_HERE, "policy-load.py")
122
+
123
+ PASS, BLOCKED, CANNOT, NOTHING = 0, 1, 2, 3
124
+ USAGE, NO_INPUT = 64, 66
125
+
126
+ _STATE = {PASS: "WOULD PASS", BLOCKED: "WOULD BLOCK",
127
+ CANNOT: "UNEVALUABLE", NOTHING: "NOTHING TO CHECK"}
128
+
129
+ DEFAULT_HISTORY = os.path.join(".loki", "cost-history.jsonl")
130
+
131
+ # Every policy key this tool knows how to replay. Cross-checked against the
132
+ # policy actually loaded: a key policy-load accepts and this cannot simulate is
133
+ # a REFUSAL naming the key, never a silent omission from the score. Scoring a
134
+ # run on the axes we happen to understand, under a headline that names the
135
+ # whole policy, is how a partial simulation passes for a complete one.
136
+ SIMULATABLE_KEYS = ("max_usd", "require_receipt")
137
+
138
+
139
+ class _Parser(argparse.ArgumentParser):
140
+ """argparse exits 2 on a usage error; here 2 means "could not check".
141
+
142
+ An unknown flag and a blind instrument are opposite facts and must not
143
+ share an exit code. Left at the default, a typo'd flag reads to a CI job as
144
+ a gate that went blind, and the operator goes hunting for missing
145
+ instrumentation that was never missing.
146
+ """
147
+
148
+ def error(self, message):
149
+ self.print_usage(sys.stderr)
150
+ print("%s: error: %s" % (self.prog, message), file=sys.stderr)
151
+ raise SystemExit(USAGE)
152
+
153
+
154
+ def load_policy(path):
155
+ """The validated policy, via policy-load.py as a SUBPROCESS.
156
+
157
+ Returns (policy_dict, None) or (None, reason). The validation itself is
158
+ never restated: policy-load.py owns the schema, the value checks (negative,
159
+ NaN, bool-as-int) and the enforces-nothing rule, and a candidate blessed
160
+ here is one an operator is about to hand to a real gate. Two validators
161
+ drift, and the drift shows up as a policy this tool approved and the gate
162
+ then rejected.
163
+
164
+ policy-load exits 1 for BOTH "file missing" and "file invalid", so the
165
+ missing case is settled by stat() here BEFORE the subprocess runs. They
166
+ need different exit codes (66 vs the refusal) because they are different
167
+ operator actions: create the file, or fix the file.
168
+ """
169
+ if not os.path.exists(path):
170
+ return None, ("no policy file at %s -- there is no candidate to "
171
+ "simulate." % path)
172
+ proc = subprocess.run(
173
+ [sys.executable, _POLICY_LOAD, "--file", path, "--json"],
174
+ capture_output=True, text=True, timeout=60)
175
+ if proc.returncode != 0:
176
+ detail = (proc.stderr or proc.stdout).strip() or "no detail reported"
177
+ return None, ("policy %s was REFUSED by tools/policy-load.py, so it "
178
+ "must not be simulated: a simulation of an invalid "
179
+ "policy answers a question about a policy no gate would "
180
+ "accept.\n %s" % (path, detail))
181
+ try:
182
+ policy = json.loads(proc.stdout)
183
+ except ValueError as exc:
184
+ return None, ("could not read the validated policy back from "
185
+ "tools/policy-load.py --json for %s: %s" % (path, exc))
186
+ if not isinstance(policy, dict):
187
+ return None, ("tools/policy-load.py returned a %s for %s, not a policy "
188
+ "object" % (type(policy).__name__, path))
189
+ return policy, None
190
+
191
+
192
+ def _num(v):
193
+ """A number as itself; None, "", or a bool as None. A real 0 survives as 0."""
194
+ if isinstance(v, bool) or not isinstance(v, (int, float)):
195
+ return None
196
+ return v
197
+
198
+
199
+ def row_usd(row):
200
+ """The measured USD for one HISTORY ROW, or None when it was not measured.
201
+
202
+ THE ONE PLACE the measured/unmeasured distinction is decided for a row, so
203
+ it cannot drift between the loader and the reporter.
204
+
205
+ WHY THIS IS NOT cost-history.measured_usd(), which is right there and which
206
+ a reviewer will reasonably ask about. That function reads a RECEIPT COST
207
+ BLOCK -- {usd, input_tokens, output_tokens, ...} -- and funnels every field
208
+ through record_is_measured(). A history ROW has no token fields at all, so
209
+ record_is_measured() answers False for every row, including one carrying a
210
+ perfectly good `usd`. Pointing it at a row does not reuse the rule, it
211
+ misapplies it, and it would report an entire measured history as blind.
212
+
213
+ What IS reused is the root predicate itself, imported from
214
+ autonomy/lib/efficiency_cost.py and never restated. cost-history.py writes
215
+ the `measured` flag at record time BY CALLING record_is_measured(), so
216
+ trusting that flag reuses the rule at one remove, which is the whole point
217
+ of it being recorded.
218
+
219
+ THE ORDER OF THE LEGACY BRANCH IS LOAD-BEARING. record_is_measured() is a
220
+ truthiness predicate, so it answers False for a measured $0.0000 --
221
+ correct for its own question ("did this record observe anything?"), wrong
222
+ for this one ("is this dollar figure present?"). So `usd is None` is tested
223
+ FIRST and short-circuits; record_is_measured() only ever adjudicates a
224
+ legacy row whose usd is already absent. Asking it first would drop a real
225
+ zero from a flag-less row while keeping the identical flagged row, so the
226
+ same fact would get two answers depending on which version of
227
+ cost-history.py wrote it.
228
+
229
+ The final test is `is not None`, never truthiness: a genuinely measured
230
+ $0.0000 run (cached, free tier) is a real observation and must be
231
+ simulated as 0.0 against the ceiling, not filed under "cannot evaluate".
232
+ """
233
+ if not isinstance(row, dict):
234
+ return None
235
+ usd = _num(row.get("usd"))
236
+ if "measured" in row:
237
+ if row.get("measured") is not True:
238
+ return None
239
+ elif usd is None and not record_is_measured(row):
240
+ return None
241
+ return usd
242
+
243
+
244
+ def newest_receipt(workspace):
245
+ """The most recent proof.json under a workspace, or None.
246
+
247
+ Same glob convention as ci-gate.py and baseline-pin.py.
248
+ """
249
+ root = os.path.normpath(workspace)
250
+ if os.path.basename(root) == ".loki":
251
+ root = os.path.dirname(root) or "."
252
+ found = glob.glob(os.path.join(root, ".loki", "proofs", "*", "proof.json"))
253
+ return max(found, key=os.path.getmtime) if found else None
254
+
255
+
256
+ def simulate_cost(row, max_usd):
257
+ """(verdict, detail) for one run against a cost ceiling.
258
+
259
+ THE FUNCTION THE WHOLE FILE PROTECTS. Its unmeasured branch is the one
260
+ place absence could become compliance, so it is written once, here, and has
261
+ nowhere to hide.
262
+
263
+ `is None`, never truthiness: a genuinely measured $0.0000 run (cached, free
264
+ tier) is a real observation and must be simulated as 0.0. A falsy guard
265
+ would file every cheap run under "cannot evaluate" and report a blind
266
+ history to a team whose instrumentation is perfect.
267
+ """
268
+ usd = row_usd(row)
269
+ if usd is None:
270
+ return CANNOT, ("cost was never measured for this run, so whether it "
271
+ "clears a $%.4f ceiling is unknown. Unmeasured is not "
272
+ "$0.00 and is not a pass." % max_usd)
273
+ if usd > max_usd:
274
+ return BLOCKED, ("$%.4f exceeds the $%.4f ceiling" % (usd, max_usd))
275
+ return PASS, "$%.4f is within the $%.4f ceiling" % (usd, max_usd)
276
+
277
+
278
+ def simulate_receipt(row):
279
+ """(verdict, detail) for one run against require_receipt.
280
+
281
+ A row records the workspace it came from. The receipt axis therefore needs
282
+ that directory to still be on disk, which for older history it often is
283
+ not. THAT IS UNEVALUABLE, NOT A FAILURE: the run may well have produced a
284
+ perfectly good receipt, and a deleted directory is an absent measurement
285
+ rather than evidence of a missing receipt. Scoring it as a block would
286
+ manufacture an adoption objection out of nothing, which is the same
287
+ dishonesty as manufacturing a pass.
288
+
289
+ Verification itself is verify() in autonomy/lib/proof-verify.py, never
290
+ re-implemented here. But its VERDICT is not consumed whole, and the reason
291
+ is the same rule one level down.
292
+
293
+ verify() answers "is this receipt verified?", and for that question it is
294
+ correct that diff_drift=None (could not re-check against the repo) collapses
295
+ into ok=False -- its own docstring says so by design, because a receipt
296
+ whose central fact cannot be re-checked is not a verified receipt. This tool
297
+ asks a DIFFERENT question: "would the policy have blocked this run?" A
298
+ historical run whose workspace is no longer a resolvable git repo -- branch
299
+ deleted, refs gone, never a repo -- comes back hash_ok=True with
300
+ diff_drift=None. Reporting that as WOULD BLOCK manufactures an adoption
301
+ objection out of a missing measurement, which is exactly the dishonesty the
302
+ paragraph above disclaims, pointed the other way. So the two are split here:
303
+ a receipt that FAILED a check blocks, a receipt that could not BE checked is
304
+ unevaluable.
305
+
306
+ That deliberately makes this axis NARROWER than verify()'s verdict, and
307
+ narrower still than what ci-gate.py enforces (it routes the receipt axis
308
+ through receipt-attest.py, which uses verify_integrity() rather than
309
+ verify()). A simulator stricter than the gate it simulates answers a
310
+ question nobody asked; one that reports blindness as blindness does not.
311
+ """
312
+ workspace = row.get("workspace")
313
+ if not isinstance(workspace, str) or not workspace:
314
+ return CANNOT, ("this history row records no workspace, so its "
315
+ "receipt cannot be located")
316
+ if not os.path.isdir(workspace):
317
+ return CANNOT, ("workspace %s no longer exists, so whether it carried "
318
+ "a receipt cannot be established now" % workspace)
319
+ receipt = newest_receipt(workspace)
320
+ if receipt is None:
321
+ return BLOCKED, ("no receipt under %s/.loki/proofs/*/proof.json -- "
322
+ "checked, and the answer is no" % workspace)
323
+ try:
324
+ result = _pv.verify(receipt)
325
+ except Exception as exc:
326
+ return CANNOT, ("receipt %s could not be verified: %s" % (receipt, exc))
327
+ if result.get("ok"):
328
+ return PASS, "receipt %s verifies" % receipt
329
+ # Intact receipt, unresolvable repo: checked the receipt, could not check it
330
+ # AGAINST anything. Not a failure, and not a pass.
331
+ if result.get("hash_ok") and result.get("diff_drift") is None:
332
+ return CANNOT, ("receipt %s is intact but could not be re-checked "
333
+ "against its repo (the recorded git state is no longer "
334
+ "resolvable), so whether it would have satisfied "
335
+ "require_receipt is unknown" % receipt)
336
+ return BLOCKED, ("receipt %s FAILS verification: %s"
337
+ % (receipt, result.get("reason") or "no reason recorded"))
338
+
339
+
340
+ def simulate_run(row, policy):
341
+ """The worst verdict across every configured axis, for ONE run.
342
+
343
+ WEAKEST LINK, with CANNOT outranking BLOCKED -- the same precedence
344
+ ci-gate.py applies to a live merge, for the same reason. A run that is
345
+ blocked on cost AND blind on receipts is reported blind: fixing the cost
346
+ would otherwise clear the verdict and leave the blindness in place.
347
+ """
348
+ axes = []
349
+ if "max_usd" in policy:
350
+ verdict, detail = simulate_cost(row, policy["max_usd"])
351
+ axes.append({"axis": "max_usd", "verdict": verdict, "detail": detail})
352
+ if policy.get("require_receipt"):
353
+ verdict, detail = simulate_receipt(row)
354
+ axes.append({"axis": "require_receipt", "verdict": verdict,
355
+ "detail": detail})
356
+
357
+ if not axes:
358
+ # policy-load already refuses an enforces-nothing policy, so this is
359
+ # unreachable through the CLI. Kept because it is the correct answer if
360
+ # it ever becomes reachable: no axis checked is no pass earned.
361
+ return {"verdict": CANNOT, "axes": [],
362
+ "detail": "no axis of this policy applies to this run"}
363
+
364
+ codes = [a["verdict"] for a in axes]
365
+ worst = CANNOT if CANNOT in codes else (
366
+ BLOCKED if BLOCKED in codes else PASS)
367
+ detail = "; ".join(a["detail"] for a in axes if a["verdict"] == worst)
368
+ return {"verdict": worst, "axes": axes, "detail": detail}
369
+
370
+
371
+ def simulate(policy_path, history_path):
372
+ """Replay the history against a candidate policy. Returns a verdict dict.
373
+
374
+ Every count carries its denominator. There is no branch here that reports a
375
+ block count without also reporting how many runs were evaluable of how
376
+ many, because a bare "0 blocked" over a blind history is the exact output
377
+ this file exists to make impossible.
378
+ """
379
+ base = {
380
+ "policy_file": policy_path,
381
+ "history_file": history_path,
382
+ "policy": None,
383
+ "runs": 0,
384
+ "evaluable": 0,
385
+ "unevaluable": 0,
386
+ "blocked": 0,
387
+ "would_pass": 0,
388
+ "corrupt_lines": 0,
389
+ "results": [],
390
+ "basis": None,
391
+ "why": None,
392
+ }
393
+
394
+ policy, refusal = load_policy(policy_path)
395
+ if policy is None:
396
+ base["status"] = "policy_refused"
397
+ # A missing file is a different operator action from an invalid one.
398
+ base["exit_code"] = (NO_INPUT if not os.path.exists(policy_path)
399
+ else CANNOT)
400
+ base["why"] = refusal
401
+ return base
402
+ base["policy"] = policy
403
+
404
+ unsupported = sorted(k for k in policy if k not in SIMULATABLE_KEYS)
405
+ if unsupported:
406
+ base["status"] = "policy_unsimulatable"
407
+ base["exit_code"] = CANNOT
408
+ base["why"] = (
409
+ "policy %s uses key(s) this tool cannot replay: %s. Reporting a "
410
+ "block count over the remaining axes would headline the whole "
411
+ "policy while silently ignoring part of it." % (
412
+ policy_path, ", ".join(unsupported)))
413
+ return base
414
+
415
+ if not os.path.exists(history_path):
416
+ base["status"] = "no_history"
417
+ base["exit_code"] = NO_INPUT
418
+ base["why"] = (
419
+ "no history file at %s -- record runs with "
420
+ "tools/cost-history.py record first. Simulating against nothing "
421
+ "is not a clean result." % history_path)
422
+ return base
423
+
424
+ rows, corrupt = _ch.load(history_path)
425
+ if rows is None:
426
+ base["status"] = "unreadable"
427
+ base["exit_code"] = CANNOT
428
+ base["why"] = ("history at %s exists but could not be read; the "
429
+ "instrument is blind, which is not the same as a "
430
+ "policy that blocks nothing." % history_path)
431
+ return base
432
+
433
+ base["corrupt_lines"] = corrupt
434
+ base["runs"] = len(rows)
435
+
436
+ if not rows:
437
+ base["status"] = "empty_history"
438
+ base["exit_code"] = NOTHING
439
+ base["why"] = (
440
+ "history %s holds no runs (%d corrupt line(s)); a policy simulated "
441
+ "against zero runs blocks nothing, and that is not evidence it is "
442
+ "safe to adopt." % (history_path, corrupt))
443
+ return base
444
+
445
+ for index, row in enumerate(rows):
446
+ outcome = simulate_run(row, policy)
447
+ base["results"].append({
448
+ "index": index,
449
+ "workspace": row.get("workspace"),
450
+ "usd": row_usd(row), # None stays None. Never 0 as a stand-in.
451
+ "verdict": outcome["verdict"],
452
+ "state": _STATE[outcome["verdict"]],
453
+ "detail": outcome["detail"],
454
+ "axes": outcome["axes"],
455
+ })
456
+
457
+ verdicts = [r["verdict"] for r in base["results"]]
458
+ base["unevaluable"] = verdicts.count(CANNOT)
459
+ base["blocked"] = verdicts.count(BLOCKED)
460
+ base["would_pass"] = verdicts.count(PASS)
461
+ base["evaluable"] = base["blocked"] + base["would_pass"]
462
+
463
+ # THE BASIS, on every count and never behind a flag. "1 of 5 would have
464
+ # been blocked" is a different fact depending on whether 5, 2 or 0 runs
465
+ # could be evaluated, and the reader cannot recover the difference.
466
+ base["basis"] = ("%d of %d run(s) were evaluable against this policy; "
467
+ "%d could not be simulated and are excluded from the "
468
+ "block count." % (base["evaluable"], base["runs"],
469
+ base["unevaluable"]))
470
+
471
+ if base["unevaluable"]:
472
+ # BLIND OUTRANKS BLOCKED. Fixing a ceiling clears a block and leaves
473
+ # the blindness exactly where it was.
474
+ base["status"] = "partly_unevaluable"
475
+ base["exit_code"] = CANNOT
476
+ base["why"] = (
477
+ "%d of %d run(s) could NOT be simulated against this policy, so "
478
+ "this simulation cannot tell you whether adopting it is safe. An "
479
+ "unevaluable run is not a run that would have passed."
480
+ % (base["unevaluable"], base["runs"]))
481
+ elif base["blocked"]:
482
+ base["status"] = "would_block"
483
+ base["exit_code"] = BLOCKED
484
+ base["why"] = ("%d of %d evaluable run(s) would have been BLOCKED by "
485
+ "this policy." % (base["blocked"], base["evaluable"]))
486
+ else:
487
+ base["status"] = "would_pass"
488
+ base["exit_code"] = PASS
489
+ base["why"] = None
490
+
491
+ return base
492
+
493
+
494
+ def render(d):
495
+ if d["status"] == "policy_refused":
496
+ return "CANNOT SIMULATE: %s" % d["why"]
497
+ if d["status"] == "policy_unsimulatable":
498
+ return "CANNOT SIMULATE: %s" % d["why"]
499
+ if d["status"] in ("no_history", "empty_history", "unreadable"):
500
+ return "NOTHING TO SIMULATE: %s" % d["why"]
501
+
502
+ policy = ", ".join("%s=%s" % (k, json.dumps(d["policy"][k]))
503
+ for k in sorted(d["policy"]))
504
+ lines = ["policy: %s" % policy,
505
+ "history: %s" % d["history_file"],
506
+ ""]
507
+ # THE COST COLUMN IS ONLY PRINTED WHEN THE POLICY HAS A COST AXIS, and the
508
+ # reason is not cosmetic. An unmeasured cost renders as UNKNOWN, which is a
509
+ # cannot-evaluate token; under a receipt-only policy that column is not part
510
+ # of the verdict at all, so printing UNKNOWN there would put "I could not
511
+ # evaluate" in the output of a run that WAS fully evaluated on every axis
512
+ # the policy configures -- a gate saying it is blind while honestly exiting
513
+ # 0. The rule this tool is held to reads exit code against output text, so
514
+ # an irrelevant column must not speak in that vocabulary.
515
+ costed = "max_usd" in d["policy"]
516
+ header = ("%-5s %-12s %-12s %s" % ("RUN", "COST", "VERDICT", "DETAIL")
517
+ if costed else "%-5s %-12s %s" % ("RUN", "VERDICT", "DETAIL"))
518
+ lines.append(header)
519
+ for row in d["results"]:
520
+ if not costed:
521
+ lines.append("%-5d %-12s %s" % (
522
+ row["index"], row["state"], row["detail"]))
523
+ continue
524
+ # UNKNOWN, never $0.00. Absent is not zero, and a measured zero is not
525
+ # absent -- so this is `is None` and prints a real 0 as $0.0000.
526
+ cost = "UNKNOWN" if row["usd"] is None else "$%.4f" % row["usd"]
527
+ lines.append("%-5d %-12s %-12s %s" % (
528
+ row["index"], cost, row["state"], row["detail"]))
529
+ lines.append("")
530
+ if d["corrupt_lines"]:
531
+ lines.append("%d CORRUPT line(s) in the history -- counted, not "
532
+ "skipped." % d["corrupt_lines"])
533
+ lines.append("%d of %d evaluable run(s) WOULD HAVE BEEN BLOCKED"
534
+ % (d["blocked"], d["evaluable"]))
535
+ lines.append("basis: %s" % d["basis"])
536
+ if d["why"]:
537
+ lines.append(d["why"])
538
+ return "\n".join(lines)
539
+
540
+
541
+ def main(argv=None):
542
+ ap = _Parser(
543
+ description="Replay recorded cost history against a candidate policy.")
544
+ ap.add_argument("workspace", nargs="?", default=".",
545
+ help="workspace root (or its .loki dir); default .")
546
+ ap.add_argument("--policy", required=True,
547
+ help="candidate policy file to simulate")
548
+ ap.add_argument("--file", dest="history",
549
+ help="history JSONL; default <workspace>/%s"
550
+ % DEFAULT_HISTORY)
551
+ ap.add_argument("--json", action="store_true", dest="as_json",
552
+ help="emit the simulation as JSON")
553
+ args = ap.parse_args(argv)
554
+
555
+ root = os.path.normpath(args.workspace)
556
+ if os.path.basename(root) == ".loki":
557
+ root = os.path.dirname(root) or "."
558
+ history = args.history or os.path.join(root, DEFAULT_HISTORY)
559
+
560
+ d = simulate(args.policy, history)
561
+ # --json stays JSON on EVERY path, refusals included: a caller that parses
562
+ # stdout must not have to special-case the failure branch (fixed once
563
+ # already in 9b879986). Diagnostics go to stderr.
564
+ print(json.dumps(d, indent=2, sort_keys=True) if args.as_json
565
+ else render(d))
566
+ return d["exit_code"]
567
+
568
+
569
+ if __name__ == "__main__":
570
+ sys.exit(main())