loki-mode 9.8.1 → 9.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +19 -14
  2. package/SKILL.md +3 -2
  3. package/VERSION +1 -1
  4. package/autonomy/loki +122 -1
  5. package/autonomy/run.sh +49 -2
  6. package/dashboard/__init__.py +1 -1
  7. package/dashboard/api_evidence.py +411 -0
  8. package/dashboard/api_operator.py +283 -0
  9. package/dashboard/api_phases.py +262 -0
  10. package/dashboard/api_releases.py +242 -0
  11. package/dashboard/api_runs.py +477 -0
  12. package/dashboard/api_tests.py +444 -0
  13. package/dashboard/api_v2.py +47 -1
  14. package/dashboard/server.py +54 -0
  15. package/dashboard/static/index.html +246 -135
  16. package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
  17. package/docs/CAPABILITY-BACKLOG.md +53 -0
  18. package/docs/COMPARISON.md +2 -2
  19. package/docs/COMPETITIVE-ANALYSIS.md +1 -1
  20. package/docs/COMPETITIVE-SCORECARD.md +422 -0
  21. package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
  22. package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
  23. package/docs/DEMOS.md +21 -23
  24. package/docs/HANDOFF-2026-08-03.md +439 -0
  25. package/docs/INSTALLATION.md +17 -10
  26. package/docs/OUTCOME-FRONTIER.md +536 -0
  27. package/docs/PROMPT-ABLATION-RESULT.md +97 -0
  28. package/docs/TOOLS.md +800 -0
  29. package/docs/alternative-installations.md +2 -3
  30. package/docs/audit-logging.md +44 -35
  31. package/docs/authentication.md +13 -2
  32. package/docs/authorization.md +87 -81
  33. package/docs/git-workflow.md +6 -3
  34. package/docs/metrics.md +15 -16
  35. package/docs/network-security.md +16 -13
  36. package/docs/openclaw-integration.md +36 -556
  37. package/docs/show-hn-post.md +2 -2
  38. package/docs/siem-integration.md +39 -36
  39. package/loki-ts/dist/loki.js +222 -222
  40. package/mcp/__init__.py +1 -1
  41. package/package.json +1 -1
  42. package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
  43. package/references/confidence-routing.md +18 -1
  44. package/references/invariant-checks.md +13 -8
  45. package/references/magic-rarv-integration.md +0 -1
  46. package/references/multi-provider.md +27 -5
  47. package/skills/healing.md +4 -2
  48. package/tools/audit-docs.py +488 -0
  49. package/tools/baseline-pin.py +19 -1
  50. package/tools/calibration-audit.py +523 -0
  51. package/tools/ci-gate.py +19 -1
  52. package/tools/cost-forecast.py +344 -0
  53. package/tools/cost-guard.py +19 -1
  54. package/tools/cost-history.py +19 -1
  55. package/tools/cost-per-outcome.py +394 -0
  56. package/tools/estimate-run.py +19 -1
  57. package/tools/evidence-freshness.py +307 -0
  58. package/tools/gate-init.py +19 -1
  59. package/tools/gate-report.py +19 -1
  60. package/tools/gate-simulate.py +570 -0
  61. package/tools/gate-trend.py +354 -0
  62. package/tools/model-advisor.py +52 -1
  63. package/tools/policy-load.py +19 -1
  64. package/tools/prompt-cost.py +363 -0
  65. package/tools/prompt-diff.py +448 -0
  66. package/tools/prompt-lint.py +448 -0
  67. package/tools/receipt-bundle.py +72 -2
  68. package/tools/receipt-diff.py +19 -1
  69. package/tools/receipt-find.py +19 -1
  70. package/tools/receipt-stats.py +380 -0
  71. package/tools/receipt-timeline.py +478 -0
  72. package/tools/receipt-verify-batch.py +291 -0
  73. package/tools/run-replay.py +19 -1
  74. package/tools/signing-status.py +19 -1
  75. package/tools/token-guard.py +19 -1
  76. package/tools/token-tax.py +375 -0
  77. package/tools/tool-index.py +19 -1
  78. package/tools/verification-tax.py +277 -0
  79. package/tools/verify-chain.py +361 -0
@@ -0,0 +1,291 @@
1
+ #!/usr/bin/env python3
2
+ """Verify an EXPLICIT LIST of receipts, one verdict per line.
3
+
4
+ WHY THIS EXISTS, GIVEN receipt-bundle.py ALREADY SHIPS. receipt-bundle walks a
5
+ workspace: it answers "is everything under this tree sound". That is the right
6
+ question for an archive and the WRONG question for a pull request, because its
7
+ input is discovered rather than declared. A PR touches a known, enumerable set
8
+ of receipts, and the difference is not cosmetic:
9
+
10
+ A receipt that should have been in the set, but is not on disk, is
11
+ INVISIBLE to a walker and NAMED by a list.
12
+
13
+ The walker cannot report an absence. `rglob("proof.json")` over a tree missing
14
+ the one receipt that mattered returns the others and reports VERIFIED on them,
15
+ truthfully, while the reviewer reads that verdict as covering the receipt they
16
+ asked about. The list route is handed the expected paths, so a path that does
17
+ not exist is a COUNTED, NAMED result (rule 4) rather than a silent shortfall.
18
+
19
+ That is the whole feature. Everything else here is deliberately NOT new:
20
+
21
+ receipt_state() three-state per-receipt classifier \\ imported from
22
+ rollup() weakest-link aggregation / receipt-bundle.py
23
+ verify() the actual verification -- proof-verify.py
24
+
25
+ Re-deriving any of them would be the drift this codebase has paid for
26
+ repeatedly: the honesty rule survives because there is ONE copy of it. If the
27
+ classifier gains a fourth failure axis upstream, this tool gains it for free.
28
+
29
+ THE FOUR STATES, NEVER COLLAPSED:
30
+
31
+ VERIFIED every axis was checked and passed
32
+ FAILED an axis was checked and said no
33
+ UNVERIFIABLE an axis could NOT be checked here, with its reason
34
+ MISSING the path does not exist -- named, counted, never skipped
35
+
36
+ MISSING is kept distinct from UNVERIFIABLE on purpose. "This receipt could not
37
+ be re-checked from this directory" and "this receipt is not there" have
38
+ different remedies: the first is re-run from the right repo, the second is a
39
+ receipt that was never generated or was deleted. Folding the second into the
40
+ first sends the operator looking for a directory problem that does not exist.
41
+ Both are non-passing, so no green leaks either way.
42
+
43
+ ROLLUP IS WEAKEST-LINK, NEVER AN AVERAGE. One FAILED sinks the batch no matter
44
+ how many VERIFIED surround it. Any scoring rule -- mean, majority, "90% or
45
+ better" -- makes a bad receipt cheaper to hide the more good ones are added,
46
+ which is exactly backwards for a gate.
47
+
48
+ AN EMPTY LIST IS NOT A PASS. Verifying nothing and finding nothing wrong are
49
+ different facts. `receipt-verify-batch.py` with no paths, or `--stdin` fed an
50
+ empty pipe, exits 3: nothing was checked, so nothing is certified. A CI job
51
+ whose glob silently matched zero files must not read as green.
52
+
53
+ Exit codes (the repo-wide convention, see tests/test_tool_exit_contract.py):
54
+
55
+ 0 every listed receipt VERIFIED
56
+ 1 at least one receipt FAILED
57
+ 2 nothing failed, but at least one receipt was UNVERIFIABLE
58
+ 3 the list was empty -- nothing to check
59
+ 64 usage error (unknown flag, bad invocation)
60
+ 66 at least one listed path does not exist
61
+
62
+ 66 outranks 1 and 2. An absent receipt means the SET this batch was asked about
63
+ is not the set it checked, so every other verdict here is a statement about a
64
+ different question than the one asked. Fix the input before reading the output.
65
+ """
66
+
67
+ import argparse
68
+ import importlib.util
69
+ import json
70
+ import os
71
+ import pathlib
72
+ import sys
73
+
74
+ # A stale .pyc for a hyphenated module loaded by path makes mutation probes
75
+ # report FALSE failures (the probe edits the source, the loader serves the old
76
+ # bytecode). Must be set before any loader below runs.
77
+ sys.dont_write_bytecode = True
78
+
79
+ _ROOT = pathlib.Path(__file__).resolve().parents[1]
80
+ _LIB = _ROOT / "autonomy" / "lib"
81
+ _TOOLS = _ROOT / "tools"
82
+
83
+
84
+ def _load(name, path):
85
+ spec = importlib.util.spec_from_file_location(name, path)
86
+ mod = importlib.util.module_from_spec(spec)
87
+ spec.loader.exec_module(mod)
88
+ return mod
89
+
90
+
91
+ # The three-state classifier and the weakest-link rule are imported, never
92
+ # restated. receipt-bundle.py is the single definition of both.
93
+ _rb = _load("receipt_bundle", _TOOLS / "receipt-bundle.py")
94
+ _pv = _load("proof_verify", _LIB / "proof-verify.py")
95
+
96
+ receipt_state = _rb.receipt_state
97
+ measured_cost = _rb.measured_cost
98
+
99
+ VERIFIED = _rb.VERIFIED
100
+ FAILED = _rb.FAILED
101
+ UNVERIFIABLE = _rb.UNVERIFIABLE
102
+ EMPTY = _rb.EMPTY
103
+ MISSING = "MISSING"
104
+
105
+ # Ordered worst-first. rollup() takes the min index, which IS the weakest-link
106
+ # rule. MISSING sits at the worst end: a set that is not the set we were asked
107
+ # about invalidates the batch, it does not average out against it.
108
+ _ORDER = (MISSING, FAILED, UNVERIFIABLE, VERIFIED)
109
+
110
+ EXIT = {VERIFIED: 0, FAILED: 1, UNVERIFIABLE: 2, EMPTY: 3, MISSING: 66}
111
+
112
+
113
+ def rollup(states):
114
+ """The batch verdict: the WEAKEST state present, never an average.
115
+
116
+ Empty is its own verdict (EMPTY), not a vacuous pass. A batch of zero
117
+ receipts has not been audited, and "we found nothing wrong" is a claim
118
+ nobody is entitled to make about a set they never opened.
119
+ """
120
+ if not states:
121
+ return EMPTY
122
+ return min(states, key=_ORDER.index)
123
+
124
+
125
+ def check_one(path, repo_dir="."):
126
+ """Classify ONE listed path into (state, reason).
127
+
128
+ A path that does not exist returns MISSING with its own name in the reason,
129
+ rather than being dropped from the list. This is the branch rule 4 is
130
+ about and the one the walker route structurally cannot have.
131
+ """
132
+ p = pathlib.Path(path)
133
+ if not p.exists():
134
+ return MISSING, ("no such receipt: %s -- listed for verification but "
135
+ "not present on disk" % path)
136
+ if p.is_dir():
137
+ # A directory is a plausible typo for the receipt inside it. Say which
138
+ # one we could not read rather than letting verify() raise something
139
+ # about bytes.
140
+ return MISSING, ("not a receipt file: %s is a directory" % path)
141
+ return receipt_state(str(p), repo_dir)
142
+
143
+
144
+ def batch(paths, repo_dir="."):
145
+ """Verify exactly the listed receipts. Pure: no writes, no network."""
146
+ receipts = []
147
+ total = 0.0
148
+ measured_n = 0
149
+
150
+ for path in paths:
151
+ state, reason = check_one(path, repo_dir)
152
+ entry = {"path": str(path), "state": state, "reason": reason,
153
+ "cost_usd": None}
154
+
155
+ # Cost is read regardless of verdict, but only a MEASURED cost
156
+ # contributes. measured_cost() returns None when the block is absent,
157
+ # malformed, or all-zero.
158
+ if state != MISSING:
159
+ try:
160
+ cost = measured_cost(_pv._load_proof(str(path)))
161
+ except Exception:
162
+ cost = None
163
+ # `is not None`, never truthiness: a genuinely measured $0.00 is a
164
+ # real observation and must survive as 0.0.
165
+ if cost is not None and cost.get("cost_usd") is not None:
166
+ entry["cost_usd"] = cost["cost_usd"]
167
+ total += cost["cost_usd"]
168
+ measured_n += 1
169
+
170
+ receipts.append(entry)
171
+
172
+ verdict = rollup([r["state"] for r in receipts])
173
+
174
+ # measured_n, NOT the total, decides UNKNOWN. Receipts that each genuinely
175
+ # measured $0.00 sum to 0.0, and `if not total` would erase that into
176
+ # "unmeasured" -- the exact defect four surfaces already shipped.
177
+ cost_block = {
178
+ "measured_receipts": measured_n,
179
+ "total_receipts": len(receipts),
180
+ "total_usd": total if measured_n else None,
181
+ }
182
+
183
+ not_verified = [r for r in receipts if r["state"] != VERIFIED]
184
+
185
+ return {
186
+ "batch": "loki-receipt-verify-batch/v1",
187
+ "checked_from": os.path.abspath(repo_dir),
188
+ "receipts": receipts,
189
+ "counts": {s: sum(1 for r in receipts if r["state"] == s)
190
+ for s in _ORDER},
191
+ "cost": cost_block,
192
+ "not_verified": [
193
+ {"path": r["path"], "state": r["state"], "reason": r["reason"]}
194
+ for r in not_verified
195
+ ],
196
+ "missing": [r["path"] for r in receipts if r["state"] == MISSING],
197
+ "verdict": verdict,
198
+ "summary": _summary(verdict, receipts, cost_block),
199
+ }
200
+
201
+
202
+ def _cost_line(cost):
203
+ """The total, always carrying its own ratio. UNKNOWN when nothing measured."""
204
+ if cost["total_usd"] is None:
205
+ return "total cost UNKNOWN (0 of %d receipts measured cost)" % (
206
+ cost["total_receipts"])
207
+ return "total cost $%.4f across %d of %d receipts measured" % (
208
+ cost["total_usd"], cost["measured_receipts"], cost["total_receipts"])
209
+
210
+
211
+ def _summary(verdict, receipts, cost):
212
+ if verdict == EMPTY:
213
+ return ("EMPTY -- no receipts were listed, so nothing was checked. "
214
+ "Verifying nothing is not a pass.")
215
+ n = len(receipts)
216
+ miss = sum(1 for r in receipts if r["state"] == MISSING)
217
+ bad = sum(1 for r in receipts if r["state"] == FAILED)
218
+ unv = sum(1 for r in receipts if r["state"] == UNVERIFIABLE)
219
+ if verdict == MISSING:
220
+ head = ("MISSING -- %d of %d listed receipts are not on disk; the set "
221
+ "checked is not the set asked about" % (miss, n))
222
+ elif verdict == FAILED:
223
+ head = ("FAILED -- %d of %d receipts FAILED verification; the batch is "
224
+ "only as good as its weakest receipt" % (bad, n))
225
+ elif verdict == UNVERIFIABLE:
226
+ head = ("UNVERIFIABLE -- %d of %d receipts could not be checked here; "
227
+ "nothing failed, but the batch is not proven" % (unv, n))
228
+ else:
229
+ head = "VERIFIED -- all %d listed receipts verified" % n
230
+ return head + ". " + _cost_line(cost)
231
+
232
+
233
+ def _render(report):
234
+ """One verdict per line, which is the surface a reviewer reads."""
235
+ lines = []
236
+ for r in report["receipts"]:
237
+ lines.append("%-13s %s" % (r["state"], r["path"]))
238
+ if r["reason"]:
239
+ lines.append(" %s" % r["reason"])
240
+ if report["receipts"]:
241
+ lines.append("")
242
+ if report["not_verified"]:
243
+ lines.append("Not verified (%d) -- counted, never dropped:"
244
+ % len(report["not_verified"]))
245
+ for r in report["not_verified"]:
246
+ lines.append(" %s [%s]" % (r["path"], r["state"]))
247
+ lines.append("")
248
+ lines.append(_cost_line(report["cost"]))
249
+ lines.append("")
250
+ lines.append(report["summary"])
251
+ return "\n".join(lines)
252
+
253
+
254
+ class _Parser(argparse.ArgumentParser):
255
+ """argparse exits 2 on a usage error, and 2 here means COULD NOT CHECK.
256
+
257
+ Left at the default, a mistyped flag would report the same code as a gate
258
+ that ran and found itself blind -- a typo masquerading as an honest
259
+ inability. 64 is the usage code, and it is distinct precisely so an
260
+ operator can tell "I invoked this wrong" from "this could not evaluate".
261
+ """
262
+
263
+ def error(self, message):
264
+ self.exit(64, "usage error: %s\n" % message)
265
+
266
+
267
+ def main(argv=None):
268
+ ap = _Parser(
269
+ prog="receipt-verify-batch.py",
270
+ description="Verify an explicit list of receipts, one verdict per line.")
271
+ ap.add_argument("paths", nargs="*",
272
+ help="receipt paths (proof.json) to verify")
273
+ ap.add_argument("--stdin", action="store_true",
274
+ help="read receipt paths from stdin, one per line")
275
+ ap.add_argument("--json", action="store_true", help="emit the raw record")
276
+ ap.add_argument("--repo-dir", default=".",
277
+ help="repository the receipts are re-checked against")
278
+ args = ap.parse_args(argv)
279
+
280
+ paths = list(args.paths)
281
+ if args.stdin:
282
+ paths += [ln.strip() for ln in sys.stdin.read().splitlines()
283
+ if ln.strip()]
284
+
285
+ report = batch(paths, args.repo_dir)
286
+ print(json.dumps(report, indent=2) if args.json else _render(report))
287
+ return EXIT[report["verdict"]]
288
+
289
+
290
+ if __name__ == "__main__":
291
+ sys.exit(main())
@@ -50,6 +50,24 @@ from efficiency_cost import record_is_measured # noqa: E402
50
50
  EXIT_NO_DATA = 66 # matches the missing-workspace convention in measure-run.sh
51
51
 
52
52
 
53
+ class _Parser(argparse.ArgumentParser):
54
+ """Usage errors exit 64, not argparse's default 2.
55
+
56
+ In this repo's convention 2 means "could NOT be checked" -- a real
57
+ answer about the subject. A mistyped flag is not that: it is an error
58
+ about the INVOCATION, and nothing about the subject was examined. The
59
+ two call for opposite responses, since retrying cannot fix a typo.
60
+
61
+ argparse exits 2 for every usage error unless this is overridden, so
62
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
63
+ """
64
+
65
+ def error(self, message):
66
+ self.print_usage(sys.stderr)
67
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
68
+ raise SystemExit(64)
69
+
70
+
53
71
  def _load_events(path):
54
72
  """Return (events, skipped_line_count).
55
73
 
@@ -355,7 +373,7 @@ def render(rep):
355
373
 
356
374
 
357
375
  def main(argv=None):
358
- ap = argparse.ArgumentParser(
376
+ ap = _Parser(
359
377
  description="Replay a completed run from its artifacts. Reads only; "
360
378
  "starts nothing, spends nothing.")
361
379
  ap.add_argument("workspace", nargs="?", default=".")
@@ -78,6 +78,24 @@ _REASONS = (
78
78
  )
79
79
 
80
80
 
81
+ class _Parser(argparse.ArgumentParser):
82
+ """Usage errors exit 64, not argparse's default 2.
83
+
84
+ In this repo's convention 2 means "could NOT be checked" -- a real
85
+ answer about the subject. A mistyped flag is not that: it is an error
86
+ about the INVOCATION, and nothing about the subject was examined. The
87
+ two call for opposite responses, since retrying cannot fix a typo.
88
+
89
+ argparse exits 2 for every usage error unless this is overridden, so
90
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
91
+ """
92
+
93
+ def error(self, message):
94
+ self.print_usage(sys.stderr)
95
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
96
+ raise SystemExit(64)
97
+
98
+
81
99
  def _classify(stderr, fallback):
82
100
  """Map gpg stderr onto a known reason. NEVER returns the stderr itself.
83
101
 
@@ -256,7 +274,7 @@ def render(result):
256
274
 
257
275
 
258
276
  def main(argv=None):
259
- parser = argparse.ArgumentParser(
277
+ parser = _Parser(
260
278
  description="Report whether this machine can produce SIGNED receipts.")
261
279
  parser.add_argument("--json", action="store_true",
262
280
  help="emit the result as JSON")
@@ -83,6 +83,24 @@ MEASURED_FIELDS = TOTAL_FIELDS
83
83
  TOTAL_DEFINITION = "total = " + " + ".join(TOTAL_FIELDS)
84
84
 
85
85
 
86
+ class _Parser(argparse.ArgumentParser):
87
+ """Usage errors exit 64, not argparse's default 2.
88
+
89
+ In this repo's convention 2 means "could NOT be checked" -- a real
90
+ answer about the subject. A mistyped flag is not that: it is an error
91
+ about the INVOCATION, and nothing about the subject was examined. The
92
+ two call for opposite responses, since retrying cannot fix a typo.
93
+
94
+ argparse exits 2 for every usage error unless this is overridden, so
95
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
96
+ """
97
+
98
+ def error(self, message):
99
+ self.print_usage(sys.stderr)
100
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
101
+ raise SystemExit(64)
102
+
103
+
86
104
  def _num(v):
87
105
  """A number as itself; None, "", or a bool as None.
88
106
 
@@ -195,7 +213,7 @@ def render(d):
195
213
 
196
214
 
197
215
  def main(argv=None):
198
- ap = argparse.ArgumentParser(
216
+ ap = _Parser(
199
217
  description="Fail CI when a run's token usage regressed past a "
200
218
  "budget policy.")
201
219
  ap.add_argument("workspace", nargs="?", default=".",