loki-mode 9.8.1 → 9.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +19 -14
  2. package/SKILL.md +3 -2
  3. package/VERSION +1 -1
  4. package/autonomy/loki +122 -1
  5. package/autonomy/run.sh +49 -2
  6. package/dashboard/__init__.py +1 -1
  7. package/dashboard/api_evidence.py +411 -0
  8. package/dashboard/api_operator.py +283 -0
  9. package/dashboard/api_phases.py +262 -0
  10. package/dashboard/api_releases.py +242 -0
  11. package/dashboard/api_runs.py +477 -0
  12. package/dashboard/api_tests.py +444 -0
  13. package/dashboard/api_v2.py +47 -1
  14. package/dashboard/server.py +54 -0
  15. package/dashboard/static/index.html +246 -135
  16. package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
  17. package/docs/CAPABILITY-BACKLOG.md +53 -0
  18. package/docs/COMPARISON.md +2 -2
  19. package/docs/COMPETITIVE-ANALYSIS.md +1 -1
  20. package/docs/COMPETITIVE-SCORECARD.md +422 -0
  21. package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
  22. package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
  23. package/docs/DEMOS.md +21 -23
  24. package/docs/HANDOFF-2026-08-03.md +439 -0
  25. package/docs/INSTALLATION.md +17 -10
  26. package/docs/OUTCOME-FRONTIER.md +536 -0
  27. package/docs/PROMPT-ABLATION-RESULT.md +97 -0
  28. package/docs/TOOLS.md +800 -0
  29. package/docs/alternative-installations.md +2 -3
  30. package/docs/audit-logging.md +44 -35
  31. package/docs/authentication.md +13 -2
  32. package/docs/authorization.md +87 -81
  33. package/docs/git-workflow.md +6 -3
  34. package/docs/metrics.md +15 -16
  35. package/docs/network-security.md +16 -13
  36. package/docs/openclaw-integration.md +36 -556
  37. package/docs/show-hn-post.md +2 -2
  38. package/docs/siem-integration.md +39 -36
  39. package/loki-ts/dist/loki.js +18 -18
  40. package/mcp/__init__.py +1 -1
  41. package/package.json +1 -1
  42. package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
  43. package/references/confidence-routing.md +18 -1
  44. package/references/invariant-checks.md +13 -8
  45. package/references/magic-rarv-integration.md +0 -1
  46. package/references/multi-provider.md +27 -5
  47. package/skills/healing.md +4 -2
  48. package/tools/audit-docs.py +488 -0
  49. package/tools/baseline-pin.py +19 -1
  50. package/tools/calibration-audit.py +523 -0
  51. package/tools/ci-gate.py +19 -1
  52. package/tools/cost-forecast.py +344 -0
  53. package/tools/cost-guard.py +19 -1
  54. package/tools/cost-history.py +19 -1
  55. package/tools/cost-per-outcome.py +394 -0
  56. package/tools/estimate-run.py +19 -1
  57. package/tools/evidence-freshness.py +307 -0
  58. package/tools/gate-init.py +19 -1
  59. package/tools/gate-report.py +19 -1
  60. package/tools/gate-simulate.py +570 -0
  61. package/tools/gate-trend.py +354 -0
  62. package/tools/model-advisor.py +52 -1
  63. package/tools/policy-load.py +19 -1
  64. package/tools/prompt-cost.py +363 -0
  65. package/tools/prompt-diff.py +448 -0
  66. package/tools/prompt-lint.py +448 -0
  67. package/tools/receipt-bundle.py +72 -2
  68. package/tools/receipt-diff.py +19 -1
  69. package/tools/receipt-find.py +19 -1
  70. package/tools/receipt-stats.py +380 -0
  71. package/tools/receipt-timeline.py +478 -0
  72. package/tools/receipt-verify-batch.py +291 -0
  73. package/tools/run-replay.py +19 -1
  74. package/tools/signing-status.py +19 -1
  75. package/tools/token-guard.py +19 -1
  76. package/tools/token-tax.py +375 -0
  77. package/tools/tool-index.py +19 -1
  78. package/tools/verification-tax.py +277 -0
  79. package/tools/verify-chain.py +361 -0
@@ -0,0 +1,448 @@
1
+ #!/usr/bin/env python3
2
+ """Show exactly what LOKI_SIMPLE=1 deletes from the prompt, before anyone
3
+ trusts an ablation result built on it.
4
+
5
+ The flag strips the coaching half of the system prompt (measured -78%, roughly
6
+ 1562 tokens per iteration). A percentage is not a reason to believe an arm is
7
+ sound. WHICH instructions vanished is, and nothing printed that until now: the
8
+ ablation tests assert that named anchors are absent, which proves the strip
9
+ happened, not that what it took was safe to take.
10
+
11
+ THE ONE ASSERTION THIS FILE EXISTS FOR, and the reason it can exit 1:
12
+
13
+ the dynamic tail must be IDENTICAL between the two arms.
14
+
15
+ Everything above [CACHE_BREAKPOINT] is the cache-stable prefix -- coaching,
16
+ which is how to work, and which a frontier model does natively. Everything
17
+ below is per-iteration STATE: which gate failed, what self-heal found, what
18
+ iteration this is. The model cannot derive state. An arm that drops coaching is
19
+ an experiment about prompt bloat; an arm that drops state is a run going blind
20
+ to its own history, and the two are indistinguishable from a byte count alone.
21
+
22
+ WHY THE STRIP IS SIMULATED DOCUMENT-WIDE. The removal predicate is applied to
23
+ EVERY line of the fixture, then both arms are split at the marker and the tails
24
+ compared. Applying it only above the marker and copying the tail verbatim would
25
+ compare the tail to itself, and the most important check in this file could
26
+ never fail. Simulating the whole document means a predicate that drifts into
27
+ matching tail content produces a real FAIL, which is the point.
28
+
29
+ The anchors are derived from the nine values pushed inside `if (!simple)` at
30
+ loki-ts/src/runner/build_prompt.ts:1607-1617. Two prefix lines pushed just
31
+ AFTER that block are deliberately absent: goal-sharpening and
32
+ CODEBASE_ANALYSIS_MODE survive the flag, and listing them here would
33
+ over-report the deletion. A stale anchor under-reports it, so an anchor that
34
+ matches nothing anywhere in the corpus is reported loudly rather than ignored.
35
+
36
+ Honesty rules, same as tools/receipt-diff.py: an unmeasured value reads
37
+ UNKNOWN, never 0; a fixture that cannot be split is reported as NOT CHECKED
38
+ rather than silently skipped; and tokens are labelled as the bytes/4 estimate
39
+ they are, because printing a bare integer would dress a derivation up as a
40
+ measurement.
41
+ """
42
+
43
+ import argparse
44
+ import json
45
+ import os
46
+ import sys
47
+
48
+ UNKNOWN = "UNKNOWN"
49
+ MARKER = "[CACHE_BREAKPOINT]"
50
+
51
+ _HERE = os.path.dirname(os.path.abspath(__file__))
52
+ DEFAULT_CORPUS = os.path.join(
53
+ os.path.dirname(_HERE), "loki-ts", "tests", "fixtures", "build_prompt")
54
+
55
+ # The nine coaching values gated by `if (!simple)` in build_prompt.ts. Each is a
56
+ # whole-line, start-anchored prefix: a bare substring would match this file's
57
+ # own prose and any tail line quoting an instruction back.
58
+ STRIP_ANCHORS = (
59
+ "RALPH WIGGUM MODE ACTIVE.", # rarvText
60
+ "SDLC_PHASES_ENABLED: [", # sdlcText
61
+ "CRITICAL AUTONOMY RULES: ", # autonomyText
62
+ "MEMORY SYSTEM: ", # MEMORY_INSTRUCTION
63
+ "USAGE_DOC_REQUIRED: ", # USAGE_DOC_INSTRUCTION
64
+ "DOC_SCOPE: ", # docScope
65
+ "RUN_CONTRACT: ", # COMPOSE_INSTRUCTION
66
+ "LSP_GROUNDING: ", # LSP_GROUNDING_INSTRUCTION
67
+ "Project conventions: read AGENTS.md", # AGENTS_MD_INSTRUCTION
68
+ )
69
+
70
+
71
+ class UsageError(Exception):
72
+ """Raised for a malformed invocation, so main() can exit 64."""
73
+
74
+
75
+ def _num(v):
76
+ """A number as itself, anything else (None, "", bool) as None.
77
+
78
+ Lifted verbatim from tools/receipt-diff.py rather than re-derived: an
79
+ absent value must never arrive downstream as a real 0.
80
+ """
81
+ if isinstance(v, bool) or not isinstance(v, (int, float)):
82
+ return None
83
+ return v
84
+
85
+
86
+ def is_coaching(line):
87
+ """True when LOKI_SIMPLE=1 would delete this line.
88
+
89
+ Start-anchored on purpose. Substring matching would let a tail line that
90
+ mentions an instruction be scored as coaching, which is exactly the
91
+ misclassification the tail check is supposed to catch.
92
+ """
93
+ return any(line.startswith(a) for a in STRIP_ANCHORS)
94
+
95
+
96
+ def simulate(text):
97
+ """The simple arm: apply the strip to the WHOLE document, not the prefix.
98
+
99
+ Returns (kept_lines, removed_lines).
100
+ """
101
+ kept, removed = [], []
102
+ for line in text.split("\n"):
103
+ (removed if is_coaching(line) else kept).append(line)
104
+ return kept, removed
105
+
106
+
107
+ def split_at_marker(text):
108
+ """(prefix, tail) around the literal marker, or None when absent."""
109
+ if MARKER not in text:
110
+ return None
111
+ prefix, tail = text.split(MARKER, 1)
112
+ return prefix, tail
113
+
114
+
115
+ def est_tokens(n_bytes):
116
+ """The repo's own bytes/4 estimator, never presented as a measurement."""
117
+ b = _num(n_bytes)
118
+ return None if b is None else int(round(b / 4.0))
119
+
120
+
121
+ def is_degraded(fixture_dir):
122
+ """True when this fixture takes the degraded-provider path.
123
+
124
+ LOAD-BEARING, and invisible to a text-only reading of expected.txt. On
125
+ PROVIDER_DEGRADED=true, buildPrompt returns buildStaticFirstDegraded at
126
+ build_prompt.ts:1574-1577 -- BEFORE `const simple` is even read at 1606 and
127
+ before the `if (!simple)` gate at 1607. The flag is inert there: both arms
128
+ emit identical bytes (verified against the real builder, 4202 == 4202,
129
+ against 7900 -> 1653 on the normal path).
130
+
131
+ Without this, the tool matches coaching-shaped anchors in a degraded
132
+ prompt and reports a saving the flag cannot produce -- a fabricated
133
+ number attached to the one thing a human reads this tool to learn.
134
+ """
135
+ env_path = os.path.join(fixture_dir, "env.txt") if fixture_dir else None
136
+ if not env_path or not os.path.isfile(env_path):
137
+ return False
138
+ with open(env_path, "r", encoding="utf-8") as fh:
139
+ return any(line.strip() == "PROVIDER_DEGRADED=true" for line in fh)
140
+
141
+
142
+ def analyze(name, text, fixture_dir=None):
143
+ """Diff one fixture's two arms. Never raises on shape; reports instead."""
144
+ full_bytes = len(text.encode("utf-8"))
145
+
146
+ if is_degraded(fixture_dir):
147
+ # A MEASURED zero, not an unmeasured one: the arms were compared and
148
+ # found equal. Reported, never folded into NOT CHECKED -- this fixture
149
+ # is perfectly checkable and the answer is "the flag does nothing".
150
+ split = split_at_marker(text)
151
+ tail_lines = ([ln for ln in split[1].split("\n") if ln.strip()]
152
+ if split else [])
153
+ return {
154
+ "fixture": name,
155
+ "checked": split is not None,
156
+ "degraded": True,
157
+ "why": "degraded provider path returns before the LOKI_SIMPLE "
158
+ "gate, so the flag has no effect here",
159
+ "full_bytes": full_bytes,
160
+ "simple_bytes": full_bytes,
161
+ "removed": [],
162
+ "survived": [ln for ln in (split[0] if split else text).split("\n")
163
+ if ln.strip()],
164
+ "tail_lines": tail_lines,
165
+ "tail_identical": True if split is not None else None,
166
+ "tail_diff": None,
167
+ }
168
+
169
+ split_full = split_at_marker(text)
170
+ if split_full is None:
171
+ return {
172
+ "fixture": name,
173
+ "checked": False,
174
+ "degraded": False,
175
+ "why": "no %s marker, so prefix and tail cannot be separated "
176
+ "(legacy flat prompt ordering)" % MARKER,
177
+ "full_bytes": full_bytes,
178
+ "simple_bytes": None,
179
+ "removed": [],
180
+ "survived": [],
181
+ "tail_identical": None,
182
+ }
183
+
184
+ kept, removed = simulate(text)
185
+ simple_text = "\n".join(kept)
186
+ split_simple = split_at_marker(simple_text)
187
+
188
+ # The strip eating the marker itself would be the most severe form of the
189
+ # failure this file guards, so it is a FAIL and not a "cannot check".
190
+ if split_simple is None:
191
+ return {
192
+ "fixture": name,
193
+ "checked": True,
194
+ "degraded": False,
195
+ "tail_identical": False,
196
+ "why": "the simulated strip removed the %s marker itself" % MARKER,
197
+ "full_bytes": full_bytes,
198
+ "simple_bytes": len(simple_text.encode("utf-8")),
199
+ "removed": removed,
200
+ "survived": [],
201
+ "tail_diff": None,
202
+ }
203
+
204
+ tail_full, tail_simple = split_full[1], split_simple[1]
205
+ identical = tail_full == tail_simple
206
+
207
+ prefix_simple = split_simple[0]
208
+ survived = [ln for ln in prefix_simple.split("\n") if ln.strip()]
209
+
210
+ out = {
211
+ "fixture": name,
212
+ "checked": True,
213
+ "degraded": False,
214
+ "tail_identical": identical,
215
+ "full_bytes": full_bytes,
216
+ "simple_bytes": len(simple_text.encode("utf-8")),
217
+ "removed": removed,
218
+ "survived": survived,
219
+ "tail_lines": [ln for ln in tail_full.split("\n") if ln.strip()],
220
+ "tail_diff": None,
221
+ }
222
+ if not identical:
223
+ out["tail_diff"] = _first_divergence(tail_full, tail_simple)
224
+ return out
225
+
226
+
227
+ def _first_divergence(a, b):
228
+ """The first differing tail line, so a FAIL names what went missing."""
229
+ la, lb = a.split("\n"), b.split("\n")
230
+ for i in range(max(len(la), len(lb))):
231
+ x = la[i] if i < len(la) else None
232
+ y = lb[i] if i < len(lb) else None
233
+ if x != y:
234
+ return {"line": i + 1, "full": x, "simple": y}
235
+ return None
236
+
237
+
238
+ def scan(corpus):
239
+ """Every fixture under the corpus root, in stable lexicographic order."""
240
+ if not os.path.isdir(corpus):
241
+ return None
242
+ results = []
243
+ for entry in sorted(os.listdir(corpus)):
244
+ fixture_dir = os.path.join(corpus, entry)
245
+ path = os.path.join(fixture_dir, "expected.txt")
246
+ if os.path.isfile(path):
247
+ with open(path, "r", encoding="utf-8") as fh:
248
+ results.append(analyze(entry, fh.read(), fixture_dir))
249
+ return results
250
+
251
+
252
+ def unused_anchors(results):
253
+ """Anchors matching nothing anywhere: a silent under-report of the strip."""
254
+ seen = set()
255
+ for r in results:
256
+ for line in r["removed"]:
257
+ for a in STRIP_ANCHORS:
258
+ if line.startswith(a):
259
+ seen.add(a)
260
+ return [a for a in STRIP_ANCHORS if a not in seen]
261
+
262
+
263
+ def _delta_line(full_b, simple_b):
264
+ fb, sb = _num(full_b), _num(simple_b)
265
+ if fb is None or sb is None:
266
+ return "bytes %s -> %s delta %s" % (
267
+ fb if fb is not None else UNKNOWN,
268
+ sb if sb is not None else UNKNOWN, UNKNOWN)
269
+ d = sb - fb
270
+ pct = (100.0 * d / fb) if fb else None
271
+ tok = est_tokens(-d)
272
+ return ("bytes %d -> %d delta %+d (%s), ~%s tokens saved "
273
+ "(est., bytes/4)" % (
274
+ fb, sb, d,
275
+ "%+.1f%%" % pct if pct is not None else UNKNOWN,
276
+ tok if tok is not None else UNKNOWN))
277
+
278
+
279
+ def _abbrev(line, width=96):
280
+ line = line.rstrip()
281
+ return line if len(line) <= width else line[:width - 3] + "..."
282
+
283
+
284
+ def render(results, corpus, verbose=False):
285
+ lines = ["LOKI_SIMPLE=1 prompt ablation diff",
286
+ " corpus: %s" % corpus,
287
+ " arms: full (default) vs simple (LOKI_SIMPLE=1)", ""]
288
+
289
+ checked = [r for r in results if r["checked"]]
290
+ failed = [r for r in checked if r["tail_identical"] is False]
291
+ skipped = [r for r in results if not r["checked"]]
292
+
293
+ for r in results:
294
+ if not r["checked"]:
295
+ continue
296
+ # A degraded fixture has nothing removed BECAUSE the flag is inert
297
+ # there, which is a finding. Hiding it would leave a reader believing
298
+ # the corpus is uniform.
299
+ if (not verbose and not r["removed"] and r["tail_identical"]
300
+ and not r.get("degraded")):
301
+ continue
302
+ lines.append(" %s" % r["fixture"])
303
+ lines.append(" %s" % _delta_line(r["full_bytes"], r["simple_bytes"]))
304
+ if r["removed"]:
305
+ lines.append(" REMOVED under the flag (coaching, %d lines):"
306
+ % len(r["removed"]))
307
+ for line in r["removed"]:
308
+ lines.append(" - %s" % _abbrev(line))
309
+ elif r.get("degraded"):
310
+ lines.append(" REMOVED under the flag: NOTHING -- %s"
311
+ % r.get("why"))
312
+ else:
313
+ lines.append(" REMOVED under the flag: nothing (this prompt "
314
+ "carries no coaching to strip)")
315
+ lines.append(" SURVIVES in the prefix (%d lines):"
316
+ % len(r["survived"]))
317
+ for line in r["survived"]:
318
+ lines.append(" + %s" % _abbrev(line))
319
+ tail_n = len(r.get("tail_lines") or [])
320
+ if r["tail_identical"]:
321
+ lines.append(" SURVIVES in the dynamic tail: all %d lines, "
322
+ "byte-identical between arms" % tail_n)
323
+ else:
324
+ lines.append(" FAIL the dynamic tail DIFFERS between arms: %s"
325
+ % (r.get("why") or "state was deleted, not coaching"))
326
+ d = r.get("tail_diff")
327
+ if d:
328
+ lines.append(" tail line %d" % d["line"])
329
+ lines.append(" full: %s" % _abbrev(str(d["full"])))
330
+ lines.append(" simple: %s" % _abbrev(str(d["simple"])))
331
+ lines.append("")
332
+
333
+ if skipped:
334
+ lines.append(" NOT CHECKED: %d fixture(s) (reported, not skipped):"
335
+ % len(skipped))
336
+ for r in skipped:
337
+ lines.append(" %s: %s" % (r["fixture"], r["why"]))
338
+ lines.append("")
339
+
340
+ stale = unused_anchors(checked)
341
+ if stale:
342
+ lines.append(" WARNING: %d strip anchor(s) matched nothing in the "
343
+ "whole corpus, so this tool may be UNDER-reporting what "
344
+ "the flag deletes:" % len(stale))
345
+ for a in stale:
346
+ lines.append(" %s" % a)
347
+ lines.append("")
348
+
349
+ degraded = [r for r in results if r.get("degraded")]
350
+ if degraded:
351
+ lines.append(" FLAG INERT on %d degraded-provider fixture(s) "
352
+ "(measured 0-byte delta, not an unmeasured one): %s"
353
+ % (len(degraded), ", ".join(r["fixture"]
354
+ for r in degraded)))
355
+ lines.append("")
356
+
357
+ lines.append(" %d fixture(s) scanned, %d checked, %d not checked"
358
+ % (len(results), len(checked), len(skipped)))
359
+ if failed:
360
+ for r in failed:
361
+ lines.append(" FAIL %s: dynamic tail is NOT identical between "
362
+ "arms" % r["fixture"])
363
+ lines.append("VERDICT: FAIL -- %d fixture(s) would lose STATE, not "
364
+ "coaching. That is not an ablation, it is the run going "
365
+ "blind to its own history." % len(failed))
366
+ else:
367
+ lines.append("VERDICT: PASS -- the dynamic tail is byte-identical "
368
+ "between arms in all %d checked fixture(s); only prefix "
369
+ "coaching is removed." % len(checked))
370
+ return "\n".join(lines)
371
+
372
+
373
+ def main(argv=None):
374
+ ap = argparse.ArgumentParser(
375
+ description="Show what LOKI_SIMPLE=1 removes from the prompt, and "
376
+ "assert the dynamic tail is identical between arms.")
377
+
378
+ def _usage_error(message):
379
+ raise UsageError(message)
380
+
381
+ # argparse exits 2 for a usage error, which in this tool line means
382
+ # "could NOT check" -- a materially different claim from "you typed it
383
+ # wrong". Reroute to 64. --help is unaffected: it goes through
384
+ # parser.exit(0), not error().
385
+ ap.error = _usage_error
386
+
387
+ ap.add_argument("corpus", nargs="?", default=DEFAULT_CORPUS,
388
+ help="build_prompt fixture corpus root "
389
+ "(default: %s)" % DEFAULT_CORPUS)
390
+ ap.add_argument("--json", action="store_true", dest="as_json",
391
+ help="emit the diff as JSON")
392
+ ap.add_argument("--verbose", action="store_true",
393
+ help="include fixtures with nothing removed")
394
+
395
+ try:
396
+ args = ap.parse_args(argv)
397
+ except UsageError as exc:
398
+ sys.stderr.write("usage error: %s\n" % exc)
399
+ return 64
400
+
401
+ if not os.path.isdir(args.corpus):
402
+ payload = {"checked": False,
403
+ "reason": "fixture corpus not found: %s" % args.corpus}
404
+ print(json.dumps(payload, indent=2) if args.as_json
405
+ else "INPUT MISSING: fixture corpus not found: %s" % args.corpus)
406
+ return 66
407
+
408
+ results = scan(args.corpus)
409
+
410
+ # An empty diff must never read as "no changes"; there was nothing to read.
411
+ if not results:
412
+ payload = {"checked": False, "fixtures": 0,
413
+ "reason": "no fixture-*/expected.txt under %s" % args.corpus}
414
+ print(json.dumps(payload, indent=2) if args.as_json
415
+ else "NOTHING TO COMPARE: no fixture-*/expected.txt under %s"
416
+ % args.corpus)
417
+ return 3
418
+
419
+ checked = [r for r in results if r["checked"]]
420
+ if not checked:
421
+ payload = {"checked": False, "fixtures": len(results),
422
+ "reason": "no fixture carries the %s marker, so no arm "
423
+ "could be split" % MARKER}
424
+ print(json.dumps(payload, indent=2) if args.as_json
425
+ else "CANNOT CHECK: no fixture carries the %s marker, so no "
426
+ "prefix/tail split was possible" % MARKER)
427
+ return 2
428
+
429
+ failed = [r for r in checked if r["tail_identical"] is False]
430
+
431
+ if args.as_json:
432
+ print(json.dumps({
433
+ "corpus": args.corpus,
434
+ "fixtures": len(results),
435
+ "checked": len(checked),
436
+ "not_checked": len(results) - len(checked),
437
+ "tail_failures": [r["fixture"] for r in failed],
438
+ "stale_anchors": unused_anchors(checked),
439
+ "results": results,
440
+ }, indent=2))
441
+ else:
442
+ print(render(results, args.corpus, verbose=args.verbose))
443
+
444
+ return 1 if failed else 0
445
+
446
+
447
+ if __name__ == "__main__":
448
+ sys.exit(main())