loki-mode 9.8.1 → 9.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +19 -14
  2. package/SKILL.md +3 -2
  3. package/VERSION +1 -1
  4. package/autonomy/loki +122 -1
  5. package/autonomy/run.sh +49 -2
  6. package/dashboard/__init__.py +1 -1
  7. package/dashboard/api_evidence.py +411 -0
  8. package/dashboard/api_operator.py +283 -0
  9. package/dashboard/api_phases.py +262 -0
  10. package/dashboard/api_releases.py +242 -0
  11. package/dashboard/api_runs.py +477 -0
  12. package/dashboard/api_tests.py +444 -0
  13. package/dashboard/api_v2.py +47 -1
  14. package/dashboard/server.py +54 -0
  15. package/dashboard/static/index.html +246 -135
  16. package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
  17. package/docs/CAPABILITY-BACKLOG.md +53 -0
  18. package/docs/COMPARISON.md +2 -2
  19. package/docs/COMPETITIVE-ANALYSIS.md +1 -1
  20. package/docs/COMPETITIVE-SCORECARD.md +422 -0
  21. package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
  22. package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
  23. package/docs/DEMOS.md +21 -23
  24. package/docs/HANDOFF-2026-08-03.md +439 -0
  25. package/docs/INSTALLATION.md +17 -10
  26. package/docs/OUTCOME-FRONTIER.md +536 -0
  27. package/docs/PROMPT-ABLATION-RESULT.md +97 -0
  28. package/docs/TOOLS.md +800 -0
  29. package/docs/alternative-installations.md +2 -3
  30. package/docs/audit-logging.md +44 -35
  31. package/docs/authentication.md +13 -2
  32. package/docs/authorization.md +87 -81
  33. package/docs/git-workflow.md +6 -3
  34. package/docs/metrics.md +15 -16
  35. package/docs/network-security.md +16 -13
  36. package/docs/openclaw-integration.md +36 -556
  37. package/docs/show-hn-post.md +2 -2
  38. package/docs/siem-integration.md +39 -36
  39. package/loki-ts/dist/loki.js +18 -18
  40. package/mcp/__init__.py +1 -1
  41. package/package.json +1 -1
  42. package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
  43. package/references/confidence-routing.md +18 -1
  44. package/references/invariant-checks.md +13 -8
  45. package/references/magic-rarv-integration.md +0 -1
  46. package/references/multi-provider.md +27 -5
  47. package/skills/healing.md +4 -2
  48. package/tools/audit-docs.py +488 -0
  49. package/tools/baseline-pin.py +19 -1
  50. package/tools/calibration-audit.py +523 -0
  51. package/tools/ci-gate.py +19 -1
  52. package/tools/cost-forecast.py +344 -0
  53. package/tools/cost-guard.py +19 -1
  54. package/tools/cost-history.py +19 -1
  55. package/tools/cost-per-outcome.py +394 -0
  56. package/tools/estimate-run.py +19 -1
  57. package/tools/evidence-freshness.py +307 -0
  58. package/tools/gate-init.py +19 -1
  59. package/tools/gate-report.py +19 -1
  60. package/tools/gate-simulate.py +570 -0
  61. package/tools/gate-trend.py +354 -0
  62. package/tools/model-advisor.py +52 -1
  63. package/tools/policy-load.py +19 -1
  64. package/tools/prompt-cost.py +363 -0
  65. package/tools/prompt-diff.py +448 -0
  66. package/tools/prompt-lint.py +448 -0
  67. package/tools/receipt-bundle.py +72 -2
  68. package/tools/receipt-diff.py +19 -1
  69. package/tools/receipt-find.py +19 -1
  70. package/tools/receipt-stats.py +380 -0
  71. package/tools/receipt-timeline.py +478 -0
  72. package/tools/receipt-verify-batch.py +291 -0
  73. package/tools/run-replay.py +19 -1
  74. package/tools/signing-status.py +19 -1
  75. package/tools/token-guard.py +19 -1
  76. package/tools/token-tax.py +375 -0
  77. package/tools/tool-index.py +19 -1
  78. package/tools/verification-tax.py +277 -0
  79. package/tools/verify-chain.py +361 -0
@@ -0,0 +1,363 @@
1
+ #!/usr/bin/env python3
2
+ """What the system prompt COSTS per iteration, split at the cache breakpoint.
3
+
4
+ WHY THIS EXISTS. Every iteration re-sends the whole prompt. build_prompt.ts
5
+ splits it at the literal [CACHE_BREAKPOINT] marker into a cache-stable
6
+ <loki_system> prefix (COACHING -- how to work) and a volatile
7
+ <dynamic_context> tail (STATE -- which gate failed, what self-heal found), and
8
+ sdk_invoker.ts puts cache_control on that split. CLAUDE.md warns that any new
9
+ always-on instruction must go in the prefix or it busts the cache every
10
+ iteration. Nothing reported what that prefix actually WEIGHS, so the warning
11
+ had no number attached to it and the LOKI_SIMPLE ablation had no denominator.
12
+
13
+ This reads the byte-exact fixture corpus under
14
+ loki-ts/tests/fixtures/build_prompt/fixture-*/expected.txt -- the same bytes
15
+ the parity job pins -- and reports bytes and estimated tokens for each half.
16
+
17
+ THE RULES. Each is a specific way this report could claim more than it measured.
18
+
19
+ 1. THE PREFIX IS AN UPPER BOUND ON THE LOKI_SIMPLE SAVING, NEVER THE SAVING.
20
+ This is the sharpest rule in the file and the easiest one to get wrong.
21
+
22
+ LOKI_SIMPLE=1 does NOT delete the prefix. At build_prompt.ts:1607 the
23
+ `if (!simple)` block wraps NINE pushes; `<loki_system>`, the PRD anchor, the
24
+ goal-score line and the closing tag survive it. So the prefix is a ceiling.
25
+
26
+ Worse, the two obvious numbers to reach for are both wrong for THIS corpus:
27
+
28
+ - The source comment at build_prompt.ts:1602 says "Measured on fixture-1:
29
+ 8090 -> 1776 bytes, -78%". fixture-1 in this corpus is 7909 bytes. That
30
+ comment does not describe these fixtures; whatever it measured, it was
31
+ not what this tool reads. Lifting -78% here would be reporting someone
32
+ else's measurement as ours.
33
+ - "the prefix is 94%, so LOKI_SIMPLE saves 94%" over-claims by everything
34
+ that survives the strip.
35
+
36
+ And NO fixture sets LOKI_SIMPLE (there is no post-ablation output in the
37
+ corpus), so the exact saving is UNMEASURED. It is reported as a bound plus
38
+ the word UNKNOWN, which is the honest pair. Re-deriving the nine-block strip
39
+ list in Python to compute it exactly would be a second copy of the ablation
40
+ predicate, and a second copy is how the rule it encodes drifts.
41
+
42
+ 2. A FIXTURE WITH NO MARKER IS UNKNOWN AND IS EXCLUDED, NEVER COUNTED AS 0.
43
+ fixture-30 and fixture-51 carry no [CACHE_BREAKPOINT]. Their prefix share is
44
+ an absent measurement, not a 0% one. Folding two zeros into a 60-fixture
45
+ mean drags the average toward the bottom without changing any real value --
46
+ the same defect receipt-stats.py rule 1 exists to prevent, transplanted from
47
+ cost to percentage. They are excluded from every aggregate, and the count of
48
+ exclusions is printed in words.
49
+
50
+ 3. AN UNSPLITTABLE CORPUS READS UNKNOWN, NOT 0%. If fixtures were found but
51
+ none carried the marker, there is no aggregate to report. The count of
52
+ SPLIT fixtures decides that, never the byte total -- `if not total` would
53
+ also erase a real, measured empty prompt.
54
+
55
+ 4. ZERO FIXTURES IS NOT A CHEAP PROMPT. Finding nothing to measure is an
56
+ invocation pointed at the wrong directory, and a substring scan over an
57
+ empty listing reports nothing missing. It gets exit 3 and says NOTHING TO
58
+ MEASURE in words rather than printing a tidy 0-byte report.
59
+
60
+ WHAT THIS IS NOT. An ADVISOR, not a gate -- `_is_gate()` in
61
+ tests/test_tool_exit_contract.py does not match this name, deliberately. There
62
+ is no prompt size that is a FAILURE, so exit 1 is unreachable here BY DESIGN;
63
+ a future reader should not "fix" that by inventing a byte threshold. A budget
64
+ belongs in token-guard.py, which is a gate and is named like one.
65
+
66
+ Usage:
67
+ tools/prompt-cost.py [fixture-dir] [--json] [--bytes-per-token N]
68
+
69
+ Exit codes:
70
+ 0 fixtures were found and measured
71
+ 3 the directory exists but holds no fixtures -- nothing to measure
72
+ 64 usage error (unknown flag, bad argument)
73
+ 66 the fixture directory does not exist
74
+ """
75
+
76
+ import argparse
77
+ import glob
78
+ import json
79
+ import os
80
+ import pathlib
81
+ import sys
82
+
83
+ # A stale .pyc can mask a mutation and turn a real probe into a false
84
+ # "MUTATION SURVIVED", since invalidation is mtime+size and a restore is
85
+ # byte-identical.
86
+ sys.dont_write_bytecode = True
87
+
88
+ _ROOT = pathlib.Path(__file__).resolve().parents[1]
89
+
90
+ # The literal marker build_prompt.ts emits. Matched as bytes-in-text exactly as
91
+ # written; this is not a regex and must not become one.
92
+ MARKER = "[CACHE_BREAKPOINT]"
93
+
94
+ DEFAULT_FIXTURES = _ROOT / "loki-ts" / "tests" / "fixtures" / "build_prompt"
95
+
96
+ # The standard rough estimate. Named, not inlined, because it is an ESTIMATE
97
+ # and every figure derived from it is labelled as one -- a real tokenizer would
98
+ # disagree, and a number that looks exact invites being quoted as exact.
99
+ BYTES_PER_TOKEN = 4
100
+
101
+
102
+ def find_fixtures(fixture_dir):
103
+ """Every expected.txt under fixture-*/, sorted numerically for a stable report.
104
+
105
+ Sorted by the fixture NUMBER, so fixture-2 precedes fixture-10. Plain
106
+ lexicographic sort puts fixture-10 second and makes two runs of the report
107
+ diff cleanly but read wrongly.
108
+ """
109
+ root = pathlib.Path(fixture_dir)
110
+ if not root.is_dir():
111
+ return []
112
+ paths = glob.glob(str(root / "fixture-*" / "expected.txt"))
113
+
114
+ def key(p):
115
+ name = pathlib.Path(p).parent.name
116
+ tail = name.rsplit("-", 1)[-1]
117
+ return (0, int(tail)) if tail.isdigit() else (1, 0, name)
118
+
119
+ return sorted((pathlib.Path(p) for p in paths if os.path.isfile(p)), key=key)
120
+
121
+
122
+ def split_prompt(text):
123
+ """Split at the first MARKER into (prefix, tail) as BYTE counts, or None.
124
+
125
+ Returns None when the marker is absent -- rule 2. The caller must exclude
126
+ rather than substitute; returning (0, 0) here would be indistinguishable
127
+ from a genuinely empty prompt.
128
+
129
+ Splits on the FIRST occurrence only. A second marker is content of the tail,
130
+ and str.split(MARKER, 1) keeps it there rather than dropping it.
131
+ """
132
+ if MARKER not in text:
133
+ return None
134
+ prefix, tail = text.split(MARKER, 1)
135
+ return len(prefix.encode("utf-8")), len(tail.encode("utf-8"))
136
+
137
+
138
+ def tokens(byte_count, bytes_per_token=BYTES_PER_TOKEN):
139
+ """Estimated tokens. None in, None out -- an unmeasured half has no estimate."""
140
+ if byte_count is None:
141
+ return None
142
+ return byte_count // bytes_per_token
143
+
144
+
145
+ def measure(fixture_dir, bytes_per_token=BYTES_PER_TOKEN):
146
+ """Measure every fixture. Pure: no writes, no network."""
147
+ paths = find_fixtures(fixture_dir)
148
+
149
+ rows = []
150
+ unsplit = []
151
+ shares = []
152
+
153
+ for path in paths:
154
+ name = path.parent.name
155
+ try:
156
+ text = path.read_text(encoding="utf-8")
157
+ except Exception as exc:
158
+ # Counted and NAMED, never silently dropped. A report that skips
159
+ # what it cannot read describes a tidier corpus than exists.
160
+ unsplit.append({"fixture": name, "reason": str(exc)})
161
+ rows.append({
162
+ "fixture": name, "path": str(path), "total_bytes": None,
163
+ "prefix_bytes": None, "tail_bytes": None,
164
+ "prefix_tokens": None, "tail_tokens": None,
165
+ "prefix_share_pct": None, "reason": str(exc),
166
+ })
167
+ continue
168
+
169
+ total = len(text.encode("utf-8"))
170
+ split = split_prompt(text)
171
+
172
+ if split is None:
173
+ unsplit.append({"fixture": name, "reason": "no %s marker" % MARKER})
174
+ rows.append({
175
+ "fixture": name, "path": str(path), "total_bytes": total,
176
+ "prefix_bytes": None, "tail_bytes": None,
177
+ "prefix_tokens": None, "tail_tokens": None,
178
+ "prefix_share_pct": None,
179
+ "reason": "no %s marker" % MARKER,
180
+ })
181
+ continue
182
+
183
+ prefix_b, tail_b = split
184
+ # The marker's own bytes are the third slice. prefix + marker + tail is
185
+ # the whole file exactly; asserting it here catches a slicing off-by-one
186
+ # that would otherwise hide in the ~0.2% rounding gap between the two
187
+ # reported percentages.
188
+ assert prefix_b + len(MARKER.encode("utf-8")) + tail_b == total, (
189
+ "%s: split does not reconstruct the file" % name)
190
+
191
+ share = 100.0 * prefix_b / total if total else None
192
+ if share is not None:
193
+ shares.append(share)
194
+
195
+ rows.append({
196
+ "fixture": name, "path": str(path), "total_bytes": total,
197
+ "prefix_bytes": prefix_b, "tail_bytes": tail_b,
198
+ "prefix_tokens": tokens(prefix_b, bytes_per_token),
199
+ "tail_tokens": tokens(tail_b, bytes_per_token),
200
+ "prefix_share_pct": share, "reason": None,
201
+ })
202
+
203
+ # len(shares), NOT the byte total, decides UNKNOWN -- rule 3.
204
+ split_n = len(shares)
205
+ prefix_total = sum(r["prefix_bytes"] for r in rows
206
+ if r["prefix_bytes"] is not None) if split_n else None
207
+ tail_total = sum(r["tail_bytes"] for r in rows
208
+ if r["tail_bytes"] is not None) if split_n else None
209
+
210
+ aggregate = {
211
+ "fixtures": len(rows),
212
+ "split_fixtures": split_n,
213
+ "unsplit_fixtures": len(rows) - split_n,
214
+ "prefix_bytes_total": prefix_total,
215
+ "tail_bytes_total": tail_total,
216
+ "prefix_tokens_total": tokens(prefix_total, bytes_per_token),
217
+ "tail_tokens_total": tokens(tail_total, bytes_per_token),
218
+ # Mean of the per-fixture shares, over SPLIT fixtures only (rule 2).
219
+ "mean_prefix_share_pct": (sum(shares) / split_n) if split_n else None,
220
+ }
221
+
222
+ return {
223
+ "report": "loki-prompt-cost/v1",
224
+ "fixture_dir": os.path.abspath(str(fixture_dir)),
225
+ "marker": MARKER,
226
+ "bytes_per_token": bytes_per_token,
227
+ "fixtures": rows,
228
+ "aggregate": aggregate,
229
+ "unsplit": unsplit,
230
+ "unsplit_count": len(unsplit),
231
+ "summary": _summary(aggregate, bytes_per_token),
232
+ }
233
+
234
+
235
+ def _simple_line(agg):
236
+ """What LOKI_SIMPLE=1 would save -- as a BOUND plus UNKNOWN. Rule 1.
237
+
238
+ Never prints a savings figure. The prefix is a ceiling (the strip keeps
239
+ <loki_system>, the PRD anchor, the goal-score line and the closing tag), and
240
+ no fixture in this corpus was generated with LOKI_SIMPLE=1, so the actual
241
+ post-ablation size was never observed here.
242
+ """
243
+ if agg["split_fixtures"] == 0 or agg["prefix_tokens_total"] is None:
244
+ return ("LOKI_SIMPLE=1 saving UNKNOWN -- no fixture could be split, so "
245
+ "there is no prefix to bound it with")
246
+ return ("LOKI_SIMPLE=1 saving UNKNOWN -- not measured by this corpus (no "
247
+ "fixture sets LOKI_SIMPLE). BOUND: it strips coaching from the "
248
+ "prefix, so it removes AT MOST the %d prefix bytes (~%d est "
249
+ "tokens) across %d fixture(s), and strictly less in practice "
250
+ "because <loki_system>, the PRD anchor and the closing tag survive "
251
+ "the strip. Not a saving figure: a ceiling."
252
+ % (agg["prefix_bytes_total"], agg["prefix_tokens_total"],
253
+ agg["split_fixtures"]))
254
+
255
+
256
+ def _summary(agg, bytes_per_token):
257
+ if agg["fixtures"] == 0:
258
+ # Rule 4. Distinct in words from "we measured a corpus and it was small".
259
+ return ("NOTHING TO MEASURE -- no fixture-*/expected.txt found under "
260
+ "this directory, so no prompt was read. Zero fixtures is not a "
261
+ "cheap prompt; it is most often the wrong directory.")
262
+
263
+ if agg["split_fixtures"] == 0:
264
+ # Rule 3: fixtures were read, none could be split. Real fact, no aggregate.
265
+ head = ("%d fixture(s) read, NONE splittable: not one carried the %s "
266
+ "marker, so the prefix/tail share is UNKNOWN -- not 0%%."
267
+ % (agg["fixtures"], MARKER))
268
+ return head + " " + _simple_line(agg)
269
+
270
+ head = ("%d fixture(s): prefix %d bytes (~%d est tokens), tail %d bytes "
271
+ "(~%d est tokens) totalled across %d splittable fixture(s). "
272
+ "Prefix is %.1f%% of the prompt on average (mean of per-fixture "
273
+ "shares); at ~%d bytes/token that prefix is re-sent every "
274
+ "iteration and is what the cache breakpoint exists to hold."
275
+ % (agg["fixtures"], agg["prefix_bytes_total"],
276
+ agg["prefix_tokens_total"], agg["tail_bytes_total"],
277
+ agg["tail_tokens_total"], agg["split_fixtures"],
278
+ agg["mean_prefix_share_pct"], bytes_per_token))
279
+
280
+ if agg["unsplit_fixtures"]:
281
+ # Rule 2, stated in words with the count. Phrased off the data so this
282
+ # sentence can never assert a false number.
283
+ head += (" %d fixture(s) EXCLUDED from every aggregate: no %s marker, "
284
+ "and counting an absent split as 0%% would drag the mean down "
285
+ "without changing any real measurement."
286
+ % (agg["unsplit_fixtures"], MARKER))
287
+
288
+ return head + " " + _simple_line(agg)
289
+
290
+
291
+ class _Parser(argparse.ArgumentParser):
292
+ """argparse exits 2 on a usage error; here 2 means "could not check".
293
+
294
+ A mistyped flag would otherwise be indistinguishable from a tool that ran
295
+ and could not measure anything. 64 is the usage error.
296
+
297
+ error() only. --help routes through exit(), not error(), and overriding
298
+ exit() would break the exit-0 contract test_tool_exit_contract.py asserts
299
+ for every tool's --help.
300
+ """
301
+
302
+ def error(self, message):
303
+ self.print_usage(sys.stderr)
304
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
305
+ raise SystemExit(64)
306
+
307
+
308
+ def main(argv=None):
309
+ ap = _Parser(
310
+ description="Report what the system prompt costs per iteration, split "
311
+ "at the cache breakpoint into coaching prefix and state tail.")
312
+ ap.add_argument("fixture_dir", nargs="?", default=str(DEFAULT_FIXTURES),
313
+ help="directory holding fixture-*/expected.txt "
314
+ "(default: the build_prompt parity corpus)")
315
+ ap.add_argument("--json", action="store_true",
316
+ help="emit the full report as JSON")
317
+ ap.add_argument("--bytes-per-token", type=int, default=BYTES_PER_TOKEN,
318
+ metavar="N",
319
+ help="bytes per token for the ESTIMATE (default: %d)"
320
+ % BYTES_PER_TOKEN)
321
+ args = ap.parse_args(argv)
322
+
323
+ if args.bytes_per_token < 1:
324
+ ap.error("--bytes-per-token must be >= 1")
325
+
326
+ if not os.path.isdir(args.fixture_dir):
327
+ # 66, not 3. "You pointed me at nothing" and "this corpus is empty" are
328
+ # different facts, and only one of them is about the corpus.
329
+ sys.stderr.write(
330
+ "prompt-cost: fixture directory does not exist: %s\n"
331
+ % args.fixture_dir)
332
+ return 66
333
+
334
+ report = measure(args.fixture_dir, args.bytes_per_token)
335
+
336
+ if args.json:
337
+ print(json.dumps(report, indent=2))
338
+ else:
339
+ # "UNKNOWN" for an unsplittable fixture, never "0". The table is the
340
+ # surface an operator eyeballs, and it must not be the one place a
341
+ # missing measurement reads as a cheap prompt.
342
+ print("%-12s %8s %8s %8s %7s" % (
343
+ "FIXTURE", "TOTAL", "PREFIX", "TAIL", "PREFIX%"))
344
+ for r in report["fixtures"]:
345
+ if r["prefix_bytes"] is None:
346
+ total = "UNKNOWN" if r["total_bytes"] is None else str(r["total_bytes"])
347
+ print("%-12s %8s %8s %8s %7s (%s)" % (
348
+ r["fixture"], total, "UNKNOWN", "UNKNOWN", "UNKNOWN",
349
+ r["reason"]))
350
+ else:
351
+ print("%-12s %8d %8d %8d %6.1f%%" % (
352
+ r["fixture"], r["total_bytes"], r["prefix_bytes"],
353
+ r["tail_bytes"], r["prefix_share_pct"]))
354
+ print("")
355
+ print(report["summary"])
356
+
357
+ # Exit 1 is unreachable BY DESIGN -- see the module docstring. There is no
358
+ # prompt size that constitutes a FAILURE; a budget belongs in token-guard.py.
359
+ return 0 if report["aggregate"]["fixtures"] else 3
360
+
361
+
362
+ if __name__ == "__main__":
363
+ sys.exit(main())