loki-mode 9.8.1 → 9.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +19 -14
  2. package/SKILL.md +3 -2
  3. package/VERSION +1 -1
  4. package/autonomy/loki +122 -1
  5. package/autonomy/run.sh +49 -2
  6. package/dashboard/__init__.py +1 -1
  7. package/dashboard/api_evidence.py +411 -0
  8. package/dashboard/api_operator.py +283 -0
  9. package/dashboard/api_phases.py +262 -0
  10. package/dashboard/api_releases.py +242 -0
  11. package/dashboard/api_runs.py +477 -0
  12. package/dashboard/api_tests.py +444 -0
  13. package/dashboard/api_v2.py +47 -1
  14. package/dashboard/server.py +54 -0
  15. package/dashboard/static/index.html +246 -135
  16. package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
  17. package/docs/CAPABILITY-BACKLOG.md +53 -0
  18. package/docs/COMPARISON.md +2 -2
  19. package/docs/COMPETITIVE-ANALYSIS.md +1 -1
  20. package/docs/COMPETITIVE-SCORECARD.md +422 -0
  21. package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
  22. package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
  23. package/docs/DEMOS.md +21 -23
  24. package/docs/HANDOFF-2026-08-03.md +439 -0
  25. package/docs/INSTALLATION.md +17 -10
  26. package/docs/OUTCOME-FRONTIER.md +536 -0
  27. package/docs/PROMPT-ABLATION-RESULT.md +97 -0
  28. package/docs/TOOLS.md +800 -0
  29. package/docs/alternative-installations.md +2 -3
  30. package/docs/audit-logging.md +44 -35
  31. package/docs/authentication.md +13 -2
  32. package/docs/authorization.md +87 -81
  33. package/docs/git-workflow.md +6 -3
  34. package/docs/metrics.md +15 -16
  35. package/docs/network-security.md +16 -13
  36. package/docs/openclaw-integration.md +36 -556
  37. package/docs/show-hn-post.md +2 -2
  38. package/docs/siem-integration.md +39 -36
  39. package/loki-ts/dist/loki.js +18 -18
  40. package/mcp/__init__.py +1 -1
  41. package/package.json +1 -1
  42. package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
  43. package/references/confidence-routing.md +18 -1
  44. package/references/invariant-checks.md +13 -8
  45. package/references/magic-rarv-integration.md +0 -1
  46. package/references/multi-provider.md +27 -5
  47. package/skills/healing.md +4 -2
  48. package/tools/audit-docs.py +488 -0
  49. package/tools/baseline-pin.py +19 -1
  50. package/tools/calibration-audit.py +523 -0
  51. package/tools/ci-gate.py +19 -1
  52. package/tools/cost-forecast.py +344 -0
  53. package/tools/cost-guard.py +19 -1
  54. package/tools/cost-history.py +19 -1
  55. package/tools/cost-per-outcome.py +394 -0
  56. package/tools/estimate-run.py +19 -1
  57. package/tools/evidence-freshness.py +307 -0
  58. package/tools/gate-init.py +19 -1
  59. package/tools/gate-report.py +19 -1
  60. package/tools/gate-simulate.py +570 -0
  61. package/tools/gate-trend.py +354 -0
  62. package/tools/model-advisor.py +52 -1
  63. package/tools/policy-load.py +19 -1
  64. package/tools/prompt-cost.py +363 -0
  65. package/tools/prompt-diff.py +448 -0
  66. package/tools/prompt-lint.py +448 -0
  67. package/tools/receipt-bundle.py +72 -2
  68. package/tools/receipt-diff.py +19 -1
  69. package/tools/receipt-find.py +19 -1
  70. package/tools/receipt-stats.py +380 -0
  71. package/tools/receipt-timeline.py +478 -0
  72. package/tools/receipt-verify-batch.py +291 -0
  73. package/tools/run-replay.py +19 -1
  74. package/tools/signing-status.py +19 -1
  75. package/tools/token-guard.py +19 -1
  76. package/tools/token-tax.py +375 -0
  77. package/tools/tool-index.py +19 -1
  78. package/tools/verification-tax.py +277 -0
  79. package/tools/verify-chain.py +361 -0
@@ -0,0 +1,307 @@
1
+ #!/usr/bin/env python3
2
+ """Is a DIRECTORY of evidence documents still true of the current tree?
3
+
4
+ WHY THIS EXISTS, GIVEN `loki verify --check-fresh` ALREADY SHIPS. That answers
5
+ the freshness question for ONE evidence doc, from inside the workspace that
6
+ produced it. The question an operator actually has before a release is plural:
7
+
8
+ I have a folder of receipts. Which of them still describe THIS tree?
9
+
10
+ Nothing answers that. `receipt-bundle.py` walks a directory but rolls every
11
+ axis into one integrity verdict, and `receipt-verify-batch.py` takes a declared
12
+ list. Neither separates "the tree moved underneath this receipt" from "this
13
+ receipt is forged" -- and for a staleness sweep those are the only two things
14
+ that matter, because they have opposite remedies. A stale receipt is REGENERATED.
15
+ A forged one is INVESTIGATED.
16
+
17
+ THE THREE STATES, and why the third one is the whole point:
18
+
19
+ FRESH the recorded diff (and tree digest, when recorded) still matches
20
+ the repository -- this document is true of the tree right now
21
+ STALE a freshness axis was CHECKED and says the tree has moved since
22
+ UNKNOWN a freshness axis could NOT be checked here, with its reason
23
+
24
+ UNKNOWN IS NEVER FOLDED INTO FRESH AND NEVER COUNTED AS STALE. An absent
25
+ measurement is not a verdict. Folding it into FRESH is the false green this
26
+ whole tool line exists to prevent; counting it as STALE would send an operator
27
+ regenerating receipts that were never shown to be out of date. It gets its own
28
+ count, its own reason, and its own exit code (2), which is the honest report:
29
+ "nothing was proven stale, and I could not check all of it".
30
+
31
+ WHAT THIS IS NOT. This is NOT the integrity gate. A receipt whose hash does not
32
+ match, whose gpg signature fails, or whose headline contradicts its own facts is
33
+ reported here as UNKNOWN -- because its freshness genuinely cannot be determined
34
+ (a tampered receipt's recorded base sha is not trustworthy input to a drift
35
+ comparison, and its tree may not have moved at all). The integrity failure is
36
+ NAMED in the reason string and the receipt is never dropped, but the verdict it
37
+ sinks the sweep to is UNKNOWN (exit 2), not FAILED (exit 1). That is weaker than
38
+ `receipt-bundle.py`, which is the correct tool for the integrity question. Run
39
+ both; this one answers only "still true of the tree?".
40
+
41
+ Nothing here re-implements hashing, receipt parsing, verification, or the walk.
42
+ `verify()` (autonomy/lib/proof-verify.py) is the single source of truth for the
43
+ drift signal, and `find_receipts()` (tools/receipt-bundle.py) is the single
44
+ definition of the walk.
45
+
46
+ Usage:
47
+ tools/evidence-freshness.py [directory] [--json] [--repo-dir DIR]
48
+
49
+ Exit codes (the repo-wide convention, tests/test_tool_exit_contract.py):
50
+
51
+ 0 every receipt found is FRESH
52
+ 1 at least one receipt is STALE (checked, and the tree has moved)
53
+ 2 nothing stale, but at least one receipt's freshness is UNKNOWN
54
+ 3 no receipts found -- nothing to check, which is not a pass
55
+ 64 usage error (unknown flag, bad invocation)
56
+ 66 the directory to scan does not exist
57
+
58
+ 66 outranks everything. A scan of a directory that is not there measured no
59
+ receipts at all, and reporting 3 ("nothing to check") for a mistyped path would
60
+ tell an operator their evidence folder is empty when it is merely elsewhere.
61
+ """
62
+
63
+ import argparse
64
+ import importlib.util
65
+ import json
66
+ import os
67
+ import pathlib
68
+ import sys
69
+
70
+ # A stale .pyc for a hyphenated module loaded by path makes mutation probes
71
+ # report FALSE results (the probe edits the source, the loader serves the old
72
+ # bytecode). Must be set before any loader below runs.
73
+ sys.dont_write_bytecode = True
74
+
75
+ _ROOT = pathlib.Path(__file__).resolve().parents[1]
76
+ _LIB = _ROOT / "autonomy" / "lib"
77
+ _TOOLS = _ROOT / "tools"
78
+
79
+
80
+ def _load(name, path):
81
+ spec = importlib.util.spec_from_file_location(name, path)
82
+ mod = importlib.util.module_from_spec(spec)
83
+ spec.loader.exec_module(mod)
84
+ return mod
85
+
86
+
87
+ _pv = _load("proof_verify", _LIB / "proof-verify.py")
88
+ _rb = _load("receipt_bundle", _TOOLS / "receipt-bundle.py")
89
+
90
+ verify = _pv.verify
91
+ # The walk is imported, never restated. If receipt-bundle learns about a new
92
+ # receipt location, this tool learns about it for free.
93
+ find_receipts = _rb.find_receipts
94
+
95
+ FRESH = "FRESH"
96
+ STALE = "STALE"
97
+ UNKNOWN = "UNKNOWN"
98
+ EMPTY = "EMPTY"
99
+
100
+ # Ordered worst-first. rollup() takes the min index, which IS the weakest-link
101
+ # rule: one STALE receipt sinks the sweep no matter how many FRESH surround it.
102
+ # STALE outranks UNKNOWN because a proven staleness is a stronger finding than
103
+ # an unproven one -- an operator with both should act on the one that is known.
104
+ _ORDER = (STALE, UNKNOWN, FRESH)
105
+
106
+ EXIT = {FRESH: 0, STALE: 1, UNKNOWN: 2, EMPTY: 3}
107
+
108
+ MISSING_DIR_EXIT = 66
109
+
110
+
111
+ def rollup(states):
112
+ """The sweep verdict: the WEAKEST state present, never an average.
113
+
114
+ Any scoring rule -- mean, majority, "90% fresh" -- makes a stale receipt
115
+ cheaper to hide the more fresh ones surround it, which is backwards for a
116
+ freshness report. Empty is its own verdict, not a vacuous pass.
117
+ """
118
+ if not states:
119
+ return EMPTY
120
+ return min(states, key=_ORDER.index)
121
+
122
+
123
+ def freshness_state(proof_path, repo_dir="."):
124
+ """Project ONE receipt onto (state, reason), keeping UNKNOWN distinct.
125
+
126
+ This deliberately does NOT reuse receipt-bundle.receipt_state(): that
127
+ classifier folds drift together with hash/gpg/headline failures into a
128
+ single FAILED, which destroys the FRESH/STALE distinction this tool is for.
129
+ It reads verify()'s drift axes directly instead.
130
+
131
+ The integrity axes are still read, but they route to UNKNOWN rather than to
132
+ a freshness verdict. A receipt whose bytes were edited after hashing has an
133
+ untrustworthy recorded base sha, so comparing it to the tree answers nothing
134
+ about freshness -- the honest report is that freshness could not be
135
+ determined, with the integrity problem named. See the module docstring: this
136
+ is not the integrity gate.
137
+
138
+ `is` comparisons are load-bearing: diff_drift is three-valued
139
+ (True/False/None) and `if not diff_drift` would read None as "no drift".
140
+ """
141
+ try:
142
+ result = verify(str(proof_path), repo_dir)
143
+ except Exception as exc: # unreadable, unparseable, wrong shape
144
+ return UNKNOWN, ("freshness could not be determined: the receipt could "
145
+ "not be loaded or verified: %s" % exc)
146
+
147
+ reasons = result.get("reasons") or []
148
+ detail = reasons[0] if reasons else (result.get("reason") or "")
149
+
150
+ # Integrity first. A receipt that fails these is not evidence about the tree
151
+ # at all, so no freshness claim can be made from it in either direction.
152
+ if not result.get("hash_ok"):
153
+ return UNKNOWN, ("freshness undeterminable: the recorded integrity hash "
154
+ "does not match the receipt bytes, so its recorded "
155
+ "base sha cannot be trusted as a comparison point "
156
+ "(run receipt-bundle.py -- this is an integrity "
157
+ "finding, not a staleness one). %s" % detail).strip()
158
+ if result.get("gpg_ok") is False:
159
+ return UNKNOWN, ("freshness undeterminable: the gpg signature does not "
160
+ "verify, so the recorded facts cannot be trusted as a "
161
+ "comparison point (run receipt-bundle.py). "
162
+ "%s" % detail).strip()
163
+ if result.get("headline_consistent") is False:
164
+ return UNKNOWN, ("freshness undeterminable: the headline disagrees with "
165
+ "the recorded facts, so the receipt is internally "
166
+ "inconsistent (run receipt-bundle.py). "
167
+ "%s" % detail).strip()
168
+
169
+ # Now the freshness axes proper.
170
+ drift = result.get("diff_drift")
171
+ tree_drift = result.get("tree_drift")
172
+ recorded_tree = (result.get("tree_recheck") or {}).get("recorded")
173
+
174
+ if drift is None:
175
+ return UNKNOWN, (detail or "the recorded diff could not be re-derived "
176
+ "here, so freshness is unknown -- re-run from the "
177
+ "repository this receipt was generated in")
178
+ if drift is True:
179
+ return STALE, (detail or "the recorded diff no longer matches the "
180
+ "repository: the tree has moved since this evidence was "
181
+ "written")
182
+ if recorded_tree and tree_drift is None:
183
+ # A tree digest was recorded but could not be recomputed. The diff axis
184
+ # passed, but a receipt that binds exact file content and cannot have
185
+ # that content re-checked is not proven fresh.
186
+ return UNKNOWN, (detail or "the recorded final workspace tree digest "
187
+ "could not be re-derived, so content freshness is "
188
+ "unknown even though the diff stat matched")
189
+ if tree_drift is True:
190
+ return STALE, (detail or "the recorded workspace tree digest no longer "
191
+ "matches: file content changed since this evidence was "
192
+ "written")
193
+ return FRESH, ""
194
+
195
+
196
+ def scan(directory, repo_dir="."):
197
+ """Classify every receipt under `directory`. Pure: no writes, no network."""
198
+ paths = find_receipts(directory)
199
+
200
+ receipts = []
201
+ for path in paths:
202
+ state, reason = freshness_state(path, repo_dir)
203
+ receipts.append({"path": str(path), "state": state, "reason": reason})
204
+
205
+ verdict = rollup([r["state"] for r in receipts])
206
+ counts = {s: sum(1 for r in receipts if r["state"] == s) for s in _ORDER}
207
+
208
+ return {
209
+ "report": "loki-evidence-freshness/v1",
210
+ "directory": os.path.abspath(str(directory)),
211
+ "checked_from": os.path.abspath(repo_dir),
212
+ "receipts": receipts,
213
+ "counts": counts,
214
+ "stale": [{"path": r["path"], "reason": r["reason"]}
215
+ for r in receipts if r["state"] == STALE],
216
+ # Carried as its OWN list, never merged into stale[]. An operator acts
217
+ # on these differently: stale means regenerate, unknown means find out.
218
+ "unknown": [{"path": r["path"], "reason": r["reason"]}
219
+ for r in receipts if r["state"] == UNKNOWN],
220
+ "verdict": verdict,
221
+ "summary": _summary(verdict, receipts, counts),
222
+ }
223
+
224
+
225
+ def _summary(verdict, receipts, counts):
226
+ if verdict == EMPTY:
227
+ return ("EMPTY -- no receipts found under this directory, so no "
228
+ "evidence was checked. Zero receipts is not a fresh sweep.")
229
+ n = len(receipts)
230
+ if verdict == FRESH:
231
+ return ("FRESH -- all %d receipts still describe the current tree" % n)
232
+ if verdict == STALE:
233
+ return ("STALE -- %d of %d receipts no longer describe the current "
234
+ "tree (%d fresh, %d unknown). Regenerate the stale ones; the "
235
+ "sweep is only as current as its oldest receipt."
236
+ % (counts[STALE], n, counts[FRESH], counts[UNKNOWN]))
237
+ return ("UNKNOWN -- %d of %d receipts could not be checked for freshness "
238
+ "(%d fresh, 0 proven stale). Nothing was shown to be out of date, "
239
+ "and the sweep is not complete: an absent measurement is not a "
240
+ "verdict." % (counts[UNKNOWN], n, counts[FRESH]))
241
+
242
+
243
+ def _render(report):
244
+ lines = ["Evidence freshness: %s" % report["directory"], ""]
245
+ for r in report["receipts"]:
246
+ lines.append(" %-8s %s" % (r["state"], r["path"]))
247
+ if r["reason"]:
248
+ lines.append(" %s" % r["reason"])
249
+ if report["receipts"]:
250
+ lines.append("")
251
+ c = report["counts"]
252
+ lines.append("%d fresh, %d stale, %d unknown"
253
+ % (c[FRESH], c[STALE], c[UNKNOWN]))
254
+ lines.append("")
255
+ lines.append(report["summary"])
256
+ return "\n".join(lines)
257
+
258
+
259
+ class _Parser(argparse.ArgumentParser):
260
+ """Usage errors exit 64, not argparse's default 2.
261
+
262
+ In this repo's convention 2 means "could NOT be checked" -- a real answer
263
+ about the subject. A mistyped flag is not that: it is an error about the
264
+ INVOCATION, and nothing about the subject was examined. The two call for
265
+ opposite responses, since retrying cannot fix a typo.
266
+
267
+ argparse exits 2 for every usage error unless this is overridden, so every
268
+ tool needs it. tests/test_tool_exit_contract.py asserts it.
269
+ """
270
+
271
+ def error(self, message):
272
+ self.print_usage(sys.stderr)
273
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
274
+ raise SystemExit(64)
275
+
276
+
277
+ def main(argv=None):
278
+ ap = _Parser(
279
+ prog="evidence-freshness.py",
280
+ description="Report which evidence receipts still describe the "
281
+ "current tree.")
282
+ ap.add_argument("directory", nargs="?", default=".",
283
+ help="directory of receipts to scan (default: .)")
284
+ ap.add_argument("--json", action="store_true", help="emit the raw record")
285
+ ap.add_argument("--repo-dir", default=".",
286
+ help="repository the receipts are re-checked against")
287
+ args = ap.parse_args(argv)
288
+
289
+ # Checked BEFORE the scan, and kept distinct from an empty result. A
290
+ # directory that is not there measured nothing; reporting 3 ("nothing to
291
+ # check") would tell an operator their evidence folder is empty when it is
292
+ # merely somewhere else.
293
+ if not os.path.isdir(args.directory):
294
+ msg = ("no such directory: %s -- nothing was scanned, so no freshness "
295
+ "claim is made about anything" % args.directory)
296
+ print(json.dumps({"verdict": "MISSING", "directory": args.directory,
297
+ "summary": msg}, indent=2)
298
+ if args.json else "MISSING -- %s" % msg)
299
+ return MISSING_DIR_EXIT
300
+
301
+ report = scan(args.directory, args.repo_dir)
302
+ print(json.dumps(report, indent=2) if args.json else _render(report))
303
+ return EXIT[report["verdict"]]
304
+
305
+
306
+ if __name__ == "__main__":
307
+ sys.exit(main())
@@ -87,6 +87,24 @@ HEADROOM = 2.0
87
87
  sys.path.insert(0, _HERE)
88
88
 
89
89
 
90
+ class _Parser(argparse.ArgumentParser):
91
+ """Usage errors exit 64, not argparse's default 2.
92
+
93
+ In this repo's convention 2 means "could NOT be checked" -- a real
94
+ answer about the subject. A mistyped flag is not that: it is an error
95
+ about the INVOCATION, and nothing about the subject was examined. The
96
+ two call for opposite responses, since retrying cannot fix a typo.
97
+
98
+ argparse exits 2 for every usage error unless this is overridden, so
99
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
100
+ """
101
+
102
+ def error(self, message):
103
+ self.print_usage(sys.stderr)
104
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
105
+ raise SystemExit(64)
106
+
107
+
90
108
  def _load_cost_history():
91
109
  """cost-history.py as a module, despite the hyphen in its name."""
92
110
  import importlib.util
@@ -231,7 +249,7 @@ def generate(workspace, out_path, force=False, history_file=None):
231
249
 
232
250
 
233
251
  def main(argv=None):
234
- ap = argparse.ArgumentParser(
252
+ ap = _Parser(
235
253
  description="Generate a ci-gate policy file and the CI snippet that "
236
254
  "enforces it.")
237
255
  ap.add_argument("workspace", nargs="?", default=".",
@@ -60,6 +60,24 @@ _UNKNOWN = ("UNEVALUABLE", "state not reported by the gate: the input carried "
60
60
  "no recognised verdict for this policy, so it has not passed")
61
61
 
62
62
 
63
+ class _Parser(argparse.ArgumentParser):
64
+ """Usage errors exit 64, not argparse's default 2.
65
+
66
+ In this repo's convention 2 means "could NOT be checked" -- a real
67
+ answer about the subject. A mistyped flag is not that: it is an error
68
+ about the INVOCATION, and nothing about the subject was examined. The
69
+ two call for opposite responses, since retrying cannot fix a typo.
70
+
71
+ argparse exits 2 for every usage error unless this is overridden, so
72
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
73
+ """
74
+
75
+ def error(self, message):
76
+ self.print_usage(sys.stderr)
77
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
78
+ raise SystemExit(64)
79
+
80
+
63
81
  def _state_of(row):
64
82
  """The row's own verdict, or UNEVALUABLE. Never a default of PASS."""
65
83
  if not isinstance(row, dict):
@@ -206,7 +224,7 @@ def exit_code(verdict):
206
224
 
207
225
 
208
226
  def main(argv=None):
209
- ap = argparse.ArgumentParser(
227
+ ap = _Parser(
210
228
  description="Render a ci-gate JSON verdict as CI-native output.")
211
229
  ap.add_argument("--format", choices=sorted(_FORMATS), default="markdown",
212
230
  help="markdown (step summary), github (annotations), text")