loki-mode 9.8.1 → 9.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +19 -14
  2. package/SKILL.md +3 -2
  3. package/VERSION +1 -1
  4. package/autonomy/loki +122 -1
  5. package/autonomy/run.sh +49 -2
  6. package/dashboard/__init__.py +1 -1
  7. package/dashboard/api_evidence.py +411 -0
  8. package/dashboard/api_operator.py +283 -0
  9. package/dashboard/api_phases.py +262 -0
  10. package/dashboard/api_releases.py +242 -0
  11. package/dashboard/api_runs.py +477 -0
  12. package/dashboard/api_tests.py +444 -0
  13. package/dashboard/api_v2.py +47 -1
  14. package/dashboard/server.py +54 -0
  15. package/dashboard/static/index.html +246 -135
  16. package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
  17. package/docs/CAPABILITY-BACKLOG.md +53 -0
  18. package/docs/COMPARISON.md +2 -2
  19. package/docs/COMPETITIVE-ANALYSIS.md +1 -1
  20. package/docs/COMPETITIVE-SCORECARD.md +422 -0
  21. package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
  22. package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
  23. package/docs/DEMOS.md +21 -23
  24. package/docs/HANDOFF-2026-08-03.md +439 -0
  25. package/docs/INSTALLATION.md +17 -10
  26. package/docs/OUTCOME-FRONTIER.md +536 -0
  27. package/docs/PROMPT-ABLATION-RESULT.md +97 -0
  28. package/docs/TOOLS.md +800 -0
  29. package/docs/alternative-installations.md +2 -3
  30. package/docs/audit-logging.md +44 -35
  31. package/docs/authentication.md +13 -2
  32. package/docs/authorization.md +87 -81
  33. package/docs/git-workflow.md +6 -3
  34. package/docs/metrics.md +15 -16
  35. package/docs/network-security.md +16 -13
  36. package/docs/openclaw-integration.md +36 -556
  37. package/docs/show-hn-post.md +2 -2
  38. package/docs/siem-integration.md +39 -36
  39. package/loki-ts/dist/loki.js +222 -222
  40. package/mcp/__init__.py +1 -1
  41. package/package.json +1 -1
  42. package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
  43. package/references/confidence-routing.md +18 -1
  44. package/references/invariant-checks.md +13 -8
  45. package/references/magic-rarv-integration.md +0 -1
  46. package/references/multi-provider.md +27 -5
  47. package/skills/healing.md +4 -2
  48. package/tools/audit-docs.py +488 -0
  49. package/tools/baseline-pin.py +19 -1
  50. package/tools/calibration-audit.py +523 -0
  51. package/tools/ci-gate.py +19 -1
  52. package/tools/cost-forecast.py +344 -0
  53. package/tools/cost-guard.py +19 -1
  54. package/tools/cost-history.py +19 -1
  55. package/tools/cost-per-outcome.py +394 -0
  56. package/tools/estimate-run.py +19 -1
  57. package/tools/evidence-freshness.py +307 -0
  58. package/tools/gate-init.py +19 -1
  59. package/tools/gate-report.py +19 -1
  60. package/tools/gate-simulate.py +570 -0
  61. package/tools/gate-trend.py +354 -0
  62. package/tools/model-advisor.py +52 -1
  63. package/tools/policy-load.py +19 -1
  64. package/tools/prompt-cost.py +363 -0
  65. package/tools/prompt-diff.py +448 -0
  66. package/tools/prompt-lint.py +448 -0
  67. package/tools/receipt-bundle.py +72 -2
  68. package/tools/receipt-diff.py +19 -1
  69. package/tools/receipt-find.py +19 -1
  70. package/tools/receipt-stats.py +380 -0
  71. package/tools/receipt-timeline.py +478 -0
  72. package/tools/receipt-verify-batch.py +291 -0
  73. package/tools/run-replay.py +19 -1
  74. package/tools/signing-status.py +19 -1
  75. package/tools/token-guard.py +19 -1
  76. package/tools/token-tax.py +375 -0
  77. package/tools/tool-index.py +19 -1
  78. package/tools/verification-tax.py +277 -0
  79. package/tools/verify-chain.py +361 -0
@@ -0,0 +1,361 @@
1
+ #!/usr/bin/env python3
2
+ """Run the whole verification chain end to end and report ONE verdict.
3
+
4
+ WHY THIS EXISTS. The chain already ships, one tool per link: receipt-bundle.py
5
+ rolls every receipt into a bundle verdict, receipt-stats.py censuses the
6
+ archive, cost-guard.py decides budget. Each is honest on its own axis. Nobody
7
+ runs one. An operator asking "is this workspace's evidence sound" runs all of
8
+ them and folds the answers by hand, and every hand-fold gets the aggregation
9
+ wrong in the same direction: toward green. `&&` stops at the first non-zero and
10
+ never reports the rest; `;` reports everything and returns only the last one.
11
+
12
+ So this composes them. It re-implements NOTHING. Every stage is a subprocess,
13
+ every state is that child's own exit code mapped, and every reason is that
14
+ child's own words. A second copy of a rule is how the rule drifts -- this repo
15
+ has watched that happen five separate times with provider lists -- so the only
16
+ logic here is the fold, and the fold is one function.
17
+
18
+ THE FIVE RULES, each a specific way a chain report can claim more than it ran.
19
+
20
+ 1. COMPOSE, NEVER REIMPLEMENT. This file contains no verification, no cost
21
+ predicate, no receipt parsing. record_is_measured() and verify() are applied
22
+ by the children, once, where they already live. Reaching in to re-derive a
23
+ cost figure here would create the second copy that drifts. The one honesty
24
+ rule that does bite at THIS layer is the render: a passed-through total of a
25
+ genuinely measured $0.0000 must survive as 0.0000, so every guard is
26
+ `is None` and never truthiness.
27
+
28
+ 2. A STAGE WHOSE TOOL IS MISSING READS UNAVAILABLE, AND THE ROLLUP IS NOT OK.
29
+ An absent tool means the link was never run, which is the absence of
30
+ evidence, not evidence of soundness. Silently skipping it and still
31
+ reporting health is the exact defect this whole tool line exists to
32
+ prevent -- and it is the one that hides best, because a skipped stage
33
+ leaves no row to notice. So it leaves a row, named, with a reason.
34
+
35
+ 3. THE ROLLUP IS THE WEAKEST LINK, NEVER A COUNT OR A PERCENTAGE. One failed
36
+ stage fails the chain. "2 of 3 stages passed" is a score, and a score cannot
37
+ stop anything: it gets EASIER to pass as stages are added, which is exactly
38
+ backwards. See rollup(), which is three lines and is the product.
39
+
40
+ 4. THREE STATES, NEVER A BOOLEAN. PASSED, FAILED, and UNAVAILABLE are three
41
+ different facts. "We checked and it is sound" and "we could not check" are
42
+ opposite claims about the world and only one earns a green. Collapsing them
43
+ makes this loudest at exactly the moment its instrumentation broke.
44
+
45
+ UNAVAILABLE outranks FAILED in the rollup, following ci-gate.py: with a
46
+ FAILED verdict the operator fixes the named failure, re-runs, sees green,
47
+ and is still blind on the dead link. The stage table shows both regardless.
48
+
49
+ 5. AN EMPTY WORKSPACE IS NOT A PASSING CHAIN, and it is not a blind one
50
+ either. EMPTY (exit 3) requires that EVERY stage reported nothing to check.
51
+ One blind stage among empty ones is UNAVAILABLE, never 3 -- otherwise a
52
+ broken link launders itself as "nothing to do here".
53
+
54
+ COST-GUARD IS CONDITIONAL, AND THAT IS NOT THE SAME AS UNAVAILABLE. Budget is
55
+ enforced only when a policy exists (.loki-policy.json, read through
56
+ policy-load.py so the filename and the schema are not restated). No policy
57
+ means the budget stage was never CONFIGURED, which is a different fact from a
58
+ budget tool that is missing from disk. A never-configured stage is reported as
59
+ a named note and does not sink the chain; a missing TOOL does. Merging those
60
+ two would make this tool exit non-zero on every workspace that has no policy,
61
+ which is most of them, and a gate that is always red is a gate nobody reads.
62
+
63
+ READ-ONLY. This starts nothing, spends nothing, and contacts no provider. It
64
+ runs three local subprocesses that only read files. The output says so, because
65
+ an operator deciding whether to run this against a live workspace should not
66
+ have to read the source to find out.
67
+
68
+ Usage:
69
+ tools/verify-chain.py [workspace] [--json]
70
+
71
+ Exit: 0 every stage checked and passed, 1 a stage FAILED, 2 a stage could not
72
+ be evaluated, 3 nothing to check anywhere, 64 usage error, 66 workspace absent.
73
+ """
74
+
75
+ import argparse
76
+ import importlib.util
77
+ import json
78
+ import os
79
+ import subprocess
80
+ import sys
81
+
82
+ # Set before any spec_from_file_location below. Loading policy-load.py through
83
+ # the loader otherwise drops a __pycache__/ into tools/ as a side effect of
84
+ # merely REPORTING on a workspace, and this tool's whole claim is that it is
85
+ # read-only. A tool that writes while saying it does not is the same category
86
+ # of untrue as a gate that greens while saying it could not check.
87
+ sys.dont_write_bytecode = True
88
+
89
+ _HERE = os.path.dirname(os.path.abspath(__file__))
90
+
91
+ # Module-level so a test can point one at a nonexistent path and assert that a
92
+ # missing stage tool reads UNAVAILABLE rather than being skipped into a green
93
+ # rollup. Same reason ci-gate.py holds its policy tools this way.
94
+ _RECEIPT_BUNDLE = os.path.join(_HERE, "receipt-bundle.py")
95
+ _RECEIPT_STATS = os.path.join(_HERE, "receipt-stats.py")
96
+ _COST_GUARD = os.path.join(_HERE, "cost-guard.py")
97
+ _POLICY_LOAD = os.path.join(_HERE, "policy-load.py")
98
+
99
+ PASSED = "PASSED"
100
+ FAILED = "FAILED"
101
+ UNAVAILABLE = "UNAVAILABLE"
102
+ NOTHING = "NOTHING"
103
+
104
+ # Ordered worst-first; rollup() takes the min index. UNAVAILABLE outranks
105
+ # FAILED deliberately: see rule 4. NOTHING is last because a chain is only
106
+ # EMPTY when there is nothing anywhere.
107
+ _ORDER = (UNAVAILABLE, FAILED, PASSED, NOTHING)
108
+
109
+ EXIT = {PASSED: 0, FAILED: 1, UNAVAILABLE: 2, NOTHING: 3}
110
+
111
+ # The children's exit codes, mapped. 0/1/2/3 is the repo-wide convention; ANY
112
+ # other code (64 usage, 66 missing input, a signal, an import blowup) is not a
113
+ # verdict this chain recognises and therefore is not a pass.
114
+ _FROM_CHILD = {0: PASSED, 1: FAILED, 2: UNAVAILABLE, 3: NOTHING}
115
+
116
+
117
+ def _policy_load():
118
+ """policy-load.py, loaded lazily so an import blowup cannot break --help.
119
+
120
+ A module-level load that raised would make this tool exit non-zero on
121
+ `--help`, which fails the repo-wide exit contract for every tool at once.
122
+ """
123
+ spec = importlib.util.spec_from_file_location("policy_load", _POLICY_LOAD)
124
+ module = importlib.util.module_from_spec(spec)
125
+ spec.loader.exec_module(module)
126
+ return module
127
+
128
+
129
+ def _workspace_root(workspace):
130
+ """Accept either a workspace root or its .loki dir, as ci-gate.py does."""
131
+ normalised = os.path.normpath(workspace)
132
+ if os.path.basename(normalised) == ".loki":
133
+ return os.path.dirname(normalised) or "."
134
+ return workspace
135
+
136
+
137
+ def _stage(name, state, reason, exit_code=None):
138
+ return {"stage": name, "state": state, "reason": reason,
139
+ "child_exit_code": exit_code}
140
+
141
+
142
+ def _run(name, argv):
143
+ """Invoke one stage tool and map its exit code. Never recompute its rule."""
144
+ tool = argv[0]
145
+ if not os.path.isfile(tool):
146
+ # Rule 2. The link was never run, so there is no result to report and
147
+ # certainly no pass. A skipped stage leaves a row precisely because a
148
+ # skipped stage is what nobody notices.
149
+ return _stage(name, UNAVAILABLE,
150
+ "stage tool is missing from disk: %s -- this link of "
151
+ "the chain never ran, so it is not a pass" % tool)
152
+ try:
153
+ proc = subprocess.run([sys.executable] + argv, capture_output=True,
154
+ text=True)
155
+ except OSError as exc:
156
+ return _stage(name, UNAVAILABLE, "could not run %s: %s" % (tool, exc))
157
+
158
+ rc = proc.returncode
159
+ if rc not in _FROM_CHILD:
160
+ return _stage(name, UNAVAILABLE,
161
+ "%s exited %d, which is not a verdict this chain "
162
+ "recognises: %s"
163
+ % (os.path.basename(tool), rc,
164
+ (proc.stderr or proc.stdout).strip() or "no output"),
165
+ rc)
166
+ return _stage(name, _FROM_CHILD[rc], _reason(proc), rc)
167
+
168
+
169
+ def _reason(proc):
170
+ """The child's own words, never this file's paraphrase of its rule.
171
+
172
+ `status`/`state` are last on purpose. They are single tokens like
173
+ "within_budget" -- true, but a cell that reports a machine enum where the
174
+ child had a whole sentence ("WITHIN BUDGET: measured cost $0.4200") throws
175
+ away the only figure an operator wanted to read.
176
+ """
177
+ try:
178
+ detail = json.loads(proc.stdout)
179
+ except (ValueError, TypeError):
180
+ detail = None
181
+ if isinstance(detail, dict):
182
+ for key in ("summary", "reason", "why", "verdict", "status", "state"):
183
+ value = detail.get(key)
184
+ if isinstance(value, str) and value.strip():
185
+ return value.strip()
186
+ text = (proc.stdout or proc.stderr or "").strip()
187
+ if text.startswith(("{", "[")):
188
+ text = (proc.stderr or "").strip()
189
+ for line in text.splitlines():
190
+ if line.strip():
191
+ return line.strip()
192
+ return "no detail reported"
193
+
194
+
195
+ def rollup(states):
196
+ """The chain verdict: the WEAKEST state present, never a count.
197
+
198
+ This is the whole product. One FAILED stage among any number of PASSED ones
199
+ makes the chain FAILED, and one UNAVAILABLE stage outranks even that,
200
+ because a blind link means the operator does not know what they shipped.
201
+
202
+ Any scoring rule -- "2 of 3", a mean, a majority -- makes a bad stage
203
+ cheaper to hide the more good stages surround it, and gets easier to pass
204
+ as the chain grows longer. Exactly backwards.
205
+
206
+ NOTHING sorts last, so a chain is EMPTY only when EVERY stage found nothing
207
+ to check. One blind stage beside empty ones is UNAVAILABLE, never 3.
208
+ """
209
+ if not states:
210
+ return UNAVAILABLE
211
+ return min(states, key=_ORDER.index)
212
+
213
+
214
+ def _budget_stage(root, workspace):
215
+ """Run cost-guard only when a policy configures a ceiling.
216
+
217
+ Three distinct outcomes, kept apart on purpose:
218
+
219
+ - no policy file, or a policy with no max_usd: the budget stage was never
220
+ CONFIGURED. Reported as a note, absent from the rollup. A gate that
221
+ went red on every workspace without a policy would be a gate nobody
222
+ reads, and "not configured" is a different fact from "not checked".
223
+ - a policy file that exists and is invalid: that IS a blind stage. The
224
+ operator asked for budget enforcement and did not get it.
225
+ - a valid ceiling: cost-guard decides, and its exit code is the state.
226
+ """
227
+ try:
228
+ policy_module = _policy_load()
229
+ except Exception as exc:
230
+ # Consistent with _run()'s missing-tool branch. Letting this raise
231
+ # would exit 1 through main(), reporting FAILED for what is actually a
232
+ # blind stage -- the wrong one of the three states.
233
+ return _stage("budget", UNAVAILABLE,
234
+ "could not load the policy reader %s, so no ceiling "
235
+ "could be enforced: %s" % (_POLICY_LOAD, exc)), None
236
+
237
+ path = os.path.join(root, policy_module.DEFAULT_FILE)
238
+ if not os.path.isfile(path):
239
+ return None, ("budget: not configured -- no %s in %s, so no cost "
240
+ "ceiling was enforced. Not a failure and not a blind "
241
+ "spot: the rule was never requested."
242
+ % (policy_module.DEFAULT_FILE, root))
243
+ try:
244
+ policy = policy_module.load(path)
245
+ except policy_module.PolicyError as exc:
246
+ return _stage("budget", UNAVAILABLE,
247
+ "policy file exists but could not be read, so the "
248
+ "ceiling it asks for was never enforced: %s" % exc), None
249
+
250
+ max_usd = policy.get("max_usd")
251
+ if max_usd is None:
252
+ return None, ("budget: not configured -- %s sets no max_usd, so no "
253
+ "cost ceiling was enforced." % path)
254
+
255
+ # No --json here, deliberately. cost-guard's JSON pass carries only
256
+ # status="within_budget", while its render() prints the measured figure the
257
+ # operator actually wants. _reason() reads whichever the child emits.
258
+ return _run("budget", [_COST_GUARD, workspace,
259
+ "--max-usd", str(max_usd)]), None
260
+
261
+
262
+ def chain(workspace=".", repo_dir="."):
263
+ """Run every stage and fold them into one verdict. Returns the record."""
264
+ root = _workspace_root(workspace)
265
+ stages = [
266
+ _run("bundle", [_RECEIPT_BUNDLE, workspace, "--repo-dir", repo_dir,
267
+ "--json"]),
268
+ _run("stats", [_RECEIPT_STATS, workspace, "--repo-dir", repo_dir,
269
+ "--json"]),
270
+ ]
271
+ notes = []
272
+ budget, note = _budget_stage(root, workspace)
273
+ if budget is not None:
274
+ stages.append(budget)
275
+ if note is not None:
276
+ notes.append(note)
277
+
278
+ verdict = rollup([s["state"] for s in stages])
279
+ return {
280
+ "workspace": workspace,
281
+ "read_only": True,
282
+ "stages": stages,
283
+ "notes": notes,
284
+ "verdict": verdict,
285
+ "exit_code": EXIT[verdict],
286
+ "summary": _summary(verdict, stages),
287
+ }
288
+
289
+
290
+ def _summary(verdict, stages):
291
+ total = len(stages)
292
+ blind = [s["stage"] for s in stages if s["state"] == UNAVAILABLE]
293
+ bad = [s["stage"] for s in stages if s["state"] == FAILED]
294
+ if verdict == NOTHING:
295
+ return ("NOTHING TO CHECK -- all %d stages found no evidence under "
296
+ "this workspace. Zero receipts is not a passing chain; it is "
297
+ "most often the wrong directory." % total)
298
+ if verdict == UNAVAILABLE:
299
+ return ("UNAVAILABLE -- %d of %d stages could not be evaluated (%s). "
300
+ "The chain is not proven: an unrun link is the absence of "
301
+ "evidence, not evidence of soundness."
302
+ % (len(blind), total, ", ".join(blind)))
303
+ if verdict == FAILED:
304
+ return ("FAILED -- %d of %d stages FAILED (%s). The chain is only as "
305
+ "good as its weakest link." % (len(bad), total, ", ".join(bad)))
306
+ return ("OK -- all %d stages were evaluated and passed. This is the "
307
+ "weakest link, not a score." % total)
308
+
309
+
310
+ def _render(report):
311
+ lines = ["Verification chain: %s" % report["workspace"],
312
+ "Read-only: starts nothing, spends nothing, contacts no provider.",
313
+ ""]
314
+ for s in report["stages"]:
315
+ lines.append(" %-12s %-12s %s" % (s["stage"], s["state"], s["reason"]))
316
+ for note in report["notes"]:
317
+ lines.append(" %s" % note)
318
+ lines.append("")
319
+ lines.append(report["summary"])
320
+ return "\n".join(lines)
321
+
322
+
323
+ class _Parser(argparse.ArgumentParser):
324
+ """Exit 64 on a usage error, never argparse's default 2.
325
+
326
+ In this repo 2 means "could not check", so a mistyped flag falling through
327
+ as 2 would be indistinguishable from an honest blind spot -- a typo
328
+ masquerading as a gate that ran and came back uncertain.
329
+ """
330
+
331
+ def error(self, message):
332
+ self.print_usage(sys.stderr)
333
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
334
+ raise SystemExit(64)
335
+
336
+
337
+ def main(argv=None):
338
+ ap = _Parser(
339
+ description="Run the whole verification chain and report one verdict.")
340
+ ap.add_argument("workspace", nargs="?", default=".",
341
+ help="workspace to verify (or its .loki dir); default .")
342
+ ap.add_argument("--repo-dir", default=".",
343
+ help="repository the receipts are re-checked against")
344
+ ap.add_argument("--json", action="store_true", dest="as_json",
345
+ help="emit the full record as JSON")
346
+ args = ap.parse_args(argv)
347
+
348
+ if not os.path.isdir(args.workspace):
349
+ # 66, not 3. "You pointed me at nothing" and "this workspace holds no
350
+ # evidence" are different facts, and only one is about the workspace.
351
+ sys.stderr.write(
352
+ "verify-chain: workspace does not exist: %s\n" % args.workspace)
353
+ return 66
354
+
355
+ report = chain(args.workspace, args.repo_dir)
356
+ print(json.dumps(report, indent=2) if args.as_json else _render(report))
357
+ return report["exit_code"]
358
+
359
+
360
+ if __name__ == "__main__":
361
+ sys.exit(main())