loki-mode 9.8.0 → 9.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +19 -14
  2. package/SKILL.md +3 -2
  3. package/VERSION +1 -1
  4. package/autonomy/loki +122 -1
  5. package/autonomy/run.sh +49 -2
  6. package/dashboard/__init__.py +1 -1
  7. package/dashboard/api_evidence.py +411 -0
  8. package/dashboard/api_operator.py +283 -0
  9. package/dashboard/api_phases.py +262 -0
  10. package/dashboard/api_releases.py +242 -0
  11. package/dashboard/api_runs.py +477 -0
  12. package/dashboard/api_tests.py +444 -0
  13. package/dashboard/api_v2.py +47 -1
  14. package/dashboard/server.py +54 -0
  15. package/dashboard/static/index.html +246 -135
  16. package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
  17. package/docs/CAPABILITY-BACKLOG.md +53 -0
  18. package/docs/COMPARISON.md +2 -2
  19. package/docs/COMPETITIVE-ANALYSIS.md +1 -1
  20. package/docs/COMPETITIVE-SCORECARD.md +422 -0
  21. package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
  22. package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
  23. package/docs/DEMOS.md +21 -23
  24. package/docs/HANDOFF-2026-08-03.md +439 -0
  25. package/docs/INSTALLATION.md +17 -10
  26. package/docs/OUTCOME-FRONTIER.md +536 -0
  27. package/docs/PROMPT-ABLATION-RESULT.md +97 -0
  28. package/docs/TOOLS.md +800 -0
  29. package/docs/alternative-installations.md +2 -3
  30. package/docs/audit-logging.md +44 -35
  31. package/docs/authentication.md +13 -2
  32. package/docs/authorization.md +87 -81
  33. package/docs/git-workflow.md +6 -3
  34. package/docs/metrics.md +15 -16
  35. package/docs/network-security.md +16 -13
  36. package/docs/openclaw-integration.md +36 -556
  37. package/docs/show-hn-post.md +2 -2
  38. package/docs/siem-integration.md +39 -36
  39. package/loki-ts/dist/loki.js +18 -18
  40. package/mcp/__init__.py +1 -1
  41. package/package.json +2 -2
  42. package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
  43. package/references/confidence-routing.md +18 -1
  44. package/references/invariant-checks.md +13 -8
  45. package/references/magic-rarv-integration.md +0 -1
  46. package/references/multi-provider.md +27 -5
  47. package/skills/healing.md +4 -2
  48. package/tools/audit-docs.py +488 -0
  49. package/tools/baseline-pin.py +19 -1
  50. package/tools/calibration-audit.py +523 -0
  51. package/tools/ci-gate.py +19 -1
  52. package/tools/cost-forecast.py +344 -0
  53. package/tools/cost-guard.py +19 -1
  54. package/tools/cost-history.py +19 -1
  55. package/tools/cost-per-outcome.py +394 -0
  56. package/tools/estimate-run.py +19 -1
  57. package/tools/evidence-freshness.py +307 -0
  58. package/tools/gate-init.py +19 -1
  59. package/tools/gate-report.py +19 -1
  60. package/tools/gate-simulate.py +570 -0
  61. package/tools/gate-trend.py +354 -0
  62. package/tools/model-advisor.py +52 -1
  63. package/tools/policy-load.py +19 -1
  64. package/tools/prompt-cost.py +363 -0
  65. package/tools/prompt-diff.py +448 -0
  66. package/tools/prompt-lint.py +448 -0
  67. package/tools/receipt-bundle.py +72 -2
  68. package/tools/receipt-diff.py +19 -1
  69. package/tools/receipt-find.py +19 -1
  70. package/tools/receipt-stats.py +380 -0
  71. package/tools/receipt-timeline.py +478 -0
  72. package/tools/receipt-verify-batch.py +291 -0
  73. package/tools/run-replay.py +19 -1
  74. package/tools/signing-status.py +19 -1
  75. package/tools/token-guard.py +19 -1
  76. package/tools/token-tax.py +375 -0
  77. package/tools/tool-index.py +19 -1
  78. package/tools/verification-tax.py +277 -0
  79. package/tools/verify-chain.py +361 -0
@@ -0,0 +1,448 @@
1
+ #!/usr/bin/env python3
2
+ """Rank system-prompt instruction blocks by how much they look like dead weight.
3
+
4
+ WHY THIS EXISTS. Anthropic deleted roughly 80% of Claude Code's system prompt
5
+ for Opus 5. The deleted text was not wrong; it was written to correct behaviours
6
+ of OLDER models that a newer one no longer needs. Coaching a capable model
7
+ through a procedure it already runs natively costs tokens on EVERY iteration and
8
+ buys nothing. Our prompt has the same shape and has never been pruned.
9
+
10
+ Pruning by intuition is how you delete the one line that was load-bearing. So
11
+ this does not prune. It RANKS, so a human can spend an ablation trial on the
12
+ highest-value candidate first instead of the first block they happened to read.
13
+
14
+ WHAT IS READ, and where the line is drawn. The corpus is the byte-exact
15
+ build_prompt fixtures. Each expected.txt splits at the literal
16
+ `[CACHE_BREAKPOINT]` marker:
17
+
18
+ ABOVE it the cache-stable prefix: standing instructions, sent every
19
+ iteration, identical every iteration. This is the coaching half and
20
+ the only half this tool scores.
21
+ BELOW it `<dynamic_context>`: iteration number, retry count, live state.
22
+ Per-iteration FACTS the model cannot infer. Never scored, never
23
+ flagged, not even read.
24
+
25
+ A fixture with no marker has no such split to read. It is EXCLUDED and counted,
26
+ never treated as all-prefix -- fixture-30's first line is "Resume iteration #3
27
+ (retry #1)", which is state, and scoring it would flag exactly the thing this
28
+ tool must never flag. Exclusions are printed with the run, because an exclusion
29
+ nobody can see is one nobody can challenge.
30
+
31
+ DEDUPE, which decides whether the output is usable. 58 fixtures carry a prefix
32
+ but they hold only 39 DISTINCT lines: the RALPH WIGGUM block alone recurs seven
33
+ times at near-identical size. Ranked per occurrence, the entire top of the
34
+ report is seven copies of one block and a reader cannot find the second
35
+ candidate. So blocks are deduped on exact text and carry an OCCURRENCES count.
36
+ That count is also the honest measure of reclaim: a block in 46 of 58 prompts is
37
+ worth more to delete than one appearing once.
38
+
39
+ Dedupe is on the text with RUNTIME-SUBSTITUTED NUMBERS masked, and that detail
40
+ was not a guess. Exact-string dedupe was tried first and produced SEVEN distinct
41
+ RALPH WIGGUM entries occupying ranks 3 through 6 of the report. The variants
42
+ differ at byte 533 by a single character: `MAX_PARALLEL_AGENTS=10` against `=12`
43
+ -- one interpolated value, not one authored instruction. A reader cannot run an
44
+ ablation on "the =10 variant". Masking digit runs collapses them to one entry
45
+ carrying the summed occurrence count, which is what a human actually deletes.
46
+
47
+ Masking is deliberately limited to digits, and the limit is load-bearing in
48
+ BOTH directions. Two RALPH WIGGUM rows survive the mask, and they should: they
49
+ carry OPPOSITE completion instructions -- "There is NEVER a 'finished' state"
50
+ against "claim done via loki_complete_task and STOP". Collapsing those would
51
+ hide a real fork in the prompt behind whichever variant happened to be seen
52
+ first. Blocks differing by an SDLC phase LIST (`[UNIT_TESTS]` against
53
+ `[UNIT_TESTS,API_TESTS]`) stay separate for the same reason. Anything beyond
54
+ digit masking would need a similarity threshold, which is a tuning knob and a
55
+ new way to be wrong.
56
+
57
+ SCORING is a transparent sum of named signals, and every matched signal name is
58
+ printed. A ranking whose reasons are invisible is a ranking nobody can argue
59
+ with.
60
+
61
+ UP size (tokens reclaimed), and coaching shape: "you must", "always",
62
+ "never", "make sure", numbered procedure steps, and naming a process a
63
+ capable model already runs (reason, reflect, verify, iterate).
64
+ DOWN a concrete repo path, a file to write, a specific command, an exit
65
+ criterion. Those carry information the model cannot infer, and deleting
66
+ them removes a fact rather than a redundant instruction.
67
+
68
+ THE TENSION IS REAL, and it is why this is advisory rather than a gate. The
69
+ top-ranked block scores high on both directions at once: RALPH WIGGUM is large,
70
+ numbered, and names REASON/REFLECT/VERIFY -- and it also names
71
+ `.loki/CONTINUITY.md` and `.loki/state/`. It is coaching and information braided
72
+ together. No static analysis can say which half is carrying the run. Only a
73
+ measured ablation can, which is precisely what a ranking is for.
74
+
75
+ ADVISORY, NOT A VERDICT. A high score is a hypothesis to test, never a licence
76
+ to delete.
77
+
78
+ HONESTY. An unmeasured value prints UNKNOWN, never 0. No fixtures, or fixtures
79
+ with no scoreable block, exits 3 -- an empty ranking would read as "nothing to
80
+ delete", which is the opposite of "nobody looked".
81
+
82
+ Exit codes follow the tools/ convention:
83
+ 0 analysed, ranking emitted
84
+ 2 the scan itself could not run
85
+ 3 nothing to analyse (no fixtures, or none with a cache prefix)
86
+ 64 usage error
87
+ 66 the given fixture root does not exist
88
+
89
+ This is an ADVISOR: no gate consumes its exit code, and 0 means the question was
90
+ answered honestly. Reads the filesystem only. Starts nothing, spends nothing.
91
+
92
+ Usage:
93
+ tools/prompt-lint.py [fixture-root] [--json] [--top N]
94
+ """
95
+
96
+ import argparse
97
+ import json
98
+ import os
99
+ import re
100
+ import sys
101
+
102
+ sys.dont_write_bytecode = True
103
+
104
+ _HERE = os.path.dirname(os.path.abspath(__file__))
105
+ _ROOT = os.path.dirname(_HERE)
106
+
107
+ # The byte-exact corpus. These fixtures gate build_prompt parity on both routes,
108
+ # so they are the closest thing to the real shipped prompt that can be read
109
+ # without running a build.
110
+ FIXTURE_GLOB = os.path.join("loki-ts", "tests", "fixtures", "build_prompt")
111
+
112
+ # The split literal. Everything above is cache-stable coaching; everything below
113
+ # is per-iteration state and is never read.
114
+ BREAKPOINT = "[CACHE_BREAKPOINT]"
115
+
116
+ # Structural scaffolding, not instructions. Scoring `<loki_system>` as a
117
+ # deletion candidate is noise: it is 13 bytes and it is a tag.
118
+ _SCAFFOLD = {"<loki_system>", "</loki_system>", "<dynamic_context>",
119
+ "</dynamic_context>"}
120
+
121
+ # A line under this many bytes cannot repay an ablation trial even if deleted
122
+ # outright. Keeps "Loki Mode" (9 bytes) out of a ranking of things worth testing.
123
+ MIN_BLOCK_BYTES = 40
124
+
125
+ # Roughly 4 bytes per token for English prose. Deliberately crude and labelled
126
+ # as an estimate everywhere it is printed -- a precise tokeniser would imply a
127
+ # precision the ranking does not have and does not need.
128
+ BYTES_PER_TOKEN = 4
129
+
130
+ # Signals, each with a weight and a name that is printed when it fires. The
131
+ # names are the evidence line: a reader disagreeing with a rank can see exactly
132
+ # which cue produced it and argue with that cue rather than with the number.
133
+ #
134
+ # UP-weighted signals are shapes that coach a model through something. DOWN
135
+ # weights are negative because the text carries a fact instead: a path, a
136
+ # filename, a command, a condition for stopping. Deleting information is a
137
+ # different and worse trade than deleting redundant instruction.
138
+ _SIGNALS = (
139
+ (re.compile(r"\byou must\b", re.I), 3, "imperative:you-must"),
140
+ (re.compile(r"\b(?:ALWAYS|NEVER)\b"), 3, "imperative:always-never"),
141
+ (re.compile(r"\bmake sure\b", re.I), 3, "imperative:make-sure"),
142
+ (re.compile(r"\bdo not\b|\bdon't\b", re.I), 2, "imperative:prohibition"),
143
+ (re.compile(r"\bCRITICAL\b|\bMUST\b|\bREQUIRED\b"), 2, "imperative:emphasis"),
144
+ (re.compile(r"\b\d\)\s"), 4, "procedure:numbered-steps"),
145
+ (re.compile(r"\b(?:REASON|REFLECT|VERIFY|ITERATE|ANALYZE|ANALYSE)\b", re.I),
146
+ 4, "native-capability:named-process"),
147
+ (re.compile(r"\bstep\s+\d\b|\bfirst,|\bthen,|\bfinally,", re.I),
148
+ 2, "procedure:sequencing"),
149
+ # DOWN. Information the model cannot infer from the task.
150
+ (re.compile(r"\.loki/[A-Za-z0-9_./-]+|[A-Za-z0-9_-]+\.(?:md|json|ya?ml|toml|txt)"),
151
+ -3, "information:concrete-path"),
152
+ (re.compile(r"\bwrite\s+[A-Za-z0-9_.-]+\.(?:md|json|ya?ml)|\bcreate\s+\.loki/",
153
+ re.I), -3, "information:file-to-write"),
154
+ (re.compile(r"`[^`]+`|\b(?:npm|pip|docker|git|curl|bun|pytest|python3?)\s+"
155
+ r"[a-z-]+"), -3, "information:specific-command"),
156
+ (re.compile(r"\bmcp__[a-z_-]+|\bloki_[a-z_]+\b"), -3, "information:tool-name"),
157
+ (re.compile(r"\bunder \d+ lines\b|\bkeep it under\b|\bmax(?:imum)? of \d+"
158
+ r"|\blimit to\b|\bMAX_[A-Z_]+=", re.I),
159
+ -3, "information:exit-criterion"),
160
+ )
161
+
162
+ # Runtime-substituted values. `MAX_PARALLEL_AGENTS=10` and `=12` are one
163
+ # authored instruction, and collapsing them is what keeps one block from taking
164
+ # four consecutive ranks. See the docstring: this was measured, not assumed.
165
+ _DIGITS = re.compile(r"\d+")
166
+
167
+
168
+ def dedupe_key(text):
169
+ return _DIGITS.sub("#", text)
170
+
171
+
172
+ # Size contributes on a log-ish curve: a 1400-byte block should outrank a
173
+ # 200-byte one, but not by seven times, or size alone would decide the whole
174
+ # ranking and the signal names would be decoration.
175
+ def _size_points(nbytes):
176
+ points = 0
177
+ for threshold in (100, 250, 500, 1000, 2000, 4000):
178
+ if nbytes >= threshold:
179
+ points += 2
180
+ return points
181
+
182
+
183
+ # A block sent in 42 of 58 prompts costs 42x what a block sent in one does, so
184
+ # the reclaim differs by that factor even at identical size. Kept as a small
185
+ # bounded bonus rather than a multiplier: multiplying would let a ubiquitous but
186
+ # information-dense block outrank a rarer block that is pure coaching, and
187
+ # frequency is a cost argument, not evidence the text is dead weight.
188
+ def _occurrence_points(count):
189
+ points = 0
190
+ for threshold in (2, 10, 30):
191
+ if count >= threshold:
192
+ points += 2
193
+ return points
194
+
195
+
196
+ class ScanError(Exception):
197
+ """The scan could not run. Exit 2, never a ranking."""
198
+
199
+
200
+ class _Parser(argparse.ArgumentParser):
201
+ """argparse exits 2 on a usage error, and 2 already means something else.
202
+
203
+ In this convention 2 is "could NOT analyse" -- a real answer about the
204
+ corpus. A typo in a flag is not that; it is 64. Left alone, `--tpo` would
205
+ report as a failed scan and a CI job could not tell the two apart.
206
+ """
207
+
208
+ def error(self, message):
209
+ self.print_usage(sys.stderr)
210
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
211
+ raise SystemExit(64)
212
+
213
+
214
+ def _read(path):
215
+ with open(path, "r", encoding="utf-8", errors="replace") as fh:
216
+ return fh.read()
217
+
218
+
219
+ def find_fixtures(root):
220
+ """Every fixture-*/expected.txt under root, sorted for stable output."""
221
+ out = []
222
+ if not os.path.isdir(root):
223
+ return out
224
+ for name in sorted(os.listdir(root)):
225
+ if not name.startswith("fixture-"):
226
+ continue
227
+ full = os.path.join(root, name, "expected.txt")
228
+ if os.path.isfile(full):
229
+ out.append((name, full))
230
+ return out
231
+
232
+
233
+ def static_prefix(text):
234
+ """Everything ABOVE the breakpoint, or None when there is no split.
235
+
236
+ None is not an empty prefix. It means this fixture cannot be read at all --
237
+ the caller must EXCLUDE it, never fall back to scoring the whole file, which
238
+ would score per-iteration state as coaching.
239
+ """
240
+ idx = text.find(BREAKPOINT)
241
+ if idx < 0:
242
+ return None
243
+ return text[:idx]
244
+
245
+
246
+ def split_blocks(prefix):
247
+ """One instruction block per non-empty, non-scaffold, non-trivial line."""
248
+ blocks = []
249
+ for line in prefix.splitlines():
250
+ line = line.strip()
251
+ if not line or line in _SCAFFOLD:
252
+ continue
253
+ if len(line.encode("utf-8")) < MIN_BLOCK_BYTES:
254
+ continue
255
+ blocks.append(line)
256
+ return blocks
257
+
258
+
259
+ def score_block(text, occurrences=1):
260
+ """Sum of size, frequency and named signal weights.
261
+
262
+ Returns (score, reasons). Every contributing term is named in reasons: a
263
+ rank whose reasons are invisible is one nobody can argue with.
264
+ """
265
+ nbytes = len(text.encode("utf-8"))
266
+ size = _size_points(nbytes)
267
+ freq = _occurrence_points(occurrences)
268
+ reasons = []
269
+ if size:
270
+ reasons.append("size:+%d" % size)
271
+ if freq:
272
+ reasons.append("frequency(x%d):+%d" % (occurrences, freq))
273
+ total = size + freq
274
+ for pattern, weight, label in _SIGNALS:
275
+ if pattern.search(text):
276
+ total += weight
277
+ reasons.append("%s:%+d" % (label, weight))
278
+ return total, reasons
279
+
280
+
281
+ def analyse(root):
282
+ """Dedupe blocks across fixtures, score each once, rank by score.
283
+
284
+ Deduped on EXACT text with an occurrence count. Ranking per occurrence puts
285
+ seven copies of the same block at the top and hides every other candidate.
286
+ """
287
+ fixtures = find_fixtures(root)
288
+ if not fixtures:
289
+ raise ScanError("no fixture-*/expected.txt found under " + root)
290
+
291
+ excluded = []
292
+ seen = {}
293
+ order = []
294
+ prefixed = 0
295
+ for name, full in fixtures:
296
+ try:
297
+ text = _read(full)
298
+ except OSError as exc:
299
+ excluded.append((name, "unreadable: %s" % exc))
300
+ continue
301
+ prefix = static_prefix(text)
302
+ if prefix is None:
303
+ excluded.append(
304
+ (name, "no %s; no static/volatile split to read" % BREAKPOINT))
305
+ continue
306
+ prefixed += 1
307
+ for block in split_blocks(prefix):
308
+ key = dedupe_key(block)
309
+ if key not in seen:
310
+ # First text seen for this key is the printed representative.
311
+ # The variants differ only in a substituted number, so any one
312
+ # of them describes the block a human would actually delete.
313
+ seen[key] = [0, block]
314
+ order.append(key)
315
+ seen[key][0] += 1
316
+
317
+ ranked = []
318
+ for key in order:
319
+ count, block = seen[key]
320
+ nbytes = len(block.encode("utf-8"))
321
+ score, reasons = score_block(block, count)
322
+ ranked.append({
323
+ "text": block,
324
+ "bytes": nbytes,
325
+ "est_tokens": nbytes // BYTES_PER_TOKEN,
326
+ "occurrences": count,
327
+ "score": score,
328
+ "signals": reasons,
329
+ })
330
+ # Score first, then size, then text -- so the order is total and two runs
331
+ # over the same corpus cannot disagree.
332
+ ranked.sort(key=lambda b: (-b["score"], -b["bytes"], b["text"]))
333
+ return {
334
+ "fixtures_found": len(fixtures),
335
+ "fixtures_with_prefix": prefixed,
336
+ "fixtures_excluded": [{"fixture": n, "reason": r} for n, r in excluded],
337
+ "blocks_analyzed": len(ranked),
338
+ "blocks": ranked,
339
+ }
340
+
341
+
342
+ ADVISORY = ("ADVISORY: this ranks deletion CANDIDATES, it does not decide. "
343
+ "A high score is a hypothesis to test with a measured ablation "
344
+ "trial, never a licence to delete.")
345
+
346
+
347
+ def _exit_code(report):
348
+ if not report["blocks_analyzed"]:
349
+ return 3
350
+ return 0
351
+
352
+
353
+ def _render(report, code, top):
354
+ lines = ["PROMPT LINT -- deletion candidates in the cache-stable prefix"]
355
+ lines.append(" " + ADVISORY)
356
+ lines.append("")
357
+ lines.append(" fixtures found: %d" % report["fixtures_found"])
358
+ lines.append(" with cache prefix: %d" % report["fixtures_with_prefix"])
359
+ lines.append(" distinct blocks: %d" % report["blocks_analyzed"])
360
+
361
+ # NOT "tokens per prompt" and NOT "tokens across the corpus". It is the sum
362
+ # over DISTINCT blocks: no single prompt carries all of them, and a block
363
+ # sent 48 times counts once. Labelled for exactly what it measures, because
364
+ # a reader sizing an ablation will quote this number.
365
+ distinct = sum(b["est_tokens"] for b in report["blocks"])
366
+ lines.append(" distinct coaching tokens: %s (deduped, NOT per-prompt; "
367
+ "bytes/%d, an estimate)"
368
+ % (distinct if report["blocks"] else "UNKNOWN",
369
+ BYTES_PER_TOKEN))
370
+ lines.append(" per-prompt reclaim: see the x<N> prompts column per block")
371
+ lines.append(" measured deletion value: UNKNOWN -- requires an ablation "
372
+ "trial; this tool ranks only")
373
+
374
+ for item in report["fixtures_excluded"]:
375
+ lines.append(" excluded: %-14s %s" % (item["fixture"], item["reason"]))
376
+
377
+ if code == 3:
378
+ lines.append("")
379
+ lines.append("NOTHING TO ANALYZE -- no block found in any cache "
380
+ "prefix. This is an absent measurement, not a finding "
381
+ "that the prompt is already lean.")
382
+ return "\n".join(lines)
383
+
384
+ shown = report["blocks"][:top]
385
+ lines.append("")
386
+ lines.append("RANKED DELETION CANDIDATES (top %d of %d)"
387
+ % (len(shown), report["blocks_analyzed"]))
388
+ for rank, b in enumerate(shown, 1):
389
+ head = b["text"][:88].replace("\n", " ")
390
+ lines.append("")
391
+ lines.append(" %2d. score %-4d %d bytes ~%d tokens x%d prompts"
392
+ % (rank, b["score"], b["bytes"], b["est_tokens"],
393
+ b["occurrences"]))
394
+ lines.append(" %s..." % head)
395
+ lines.append(" signals: %s" % (", ".join(b["signals"]) or "none"))
396
+ lines.append("")
397
+ lines.append(" " + ADVISORY)
398
+ return "\n".join(lines)
399
+
400
+
401
+ def main(argv=None):
402
+ parser = _Parser(
403
+ description="Rank system-prompt blocks by deletion-candidate value. "
404
+ "Advisory: ranks candidates, does not decide.")
405
+ parser.add_argument("fixture_root", nargs="?", default=None,
406
+ help="directory of fixture-*/expected.txt "
407
+ "(default: the repo's build_prompt corpus)")
408
+ parser.add_argument("--json", action="store_true",
409
+ help="emit machine-readable output")
410
+ parser.add_argument("--top", type=int, default=10,
411
+ help="how many ranked blocks to print (default 10)")
412
+ args = parser.parse_args(argv)
413
+
414
+ if args.fixture_root is not None and not os.path.exists(args.fixture_root):
415
+ # 66 input missing. Emitted as JSON under --json: a consumer that asked
416
+ # for machine output must not get a bare line it cannot parse.
417
+ payload = {"status": "input_missing", "exit_code": 66,
418
+ "error": "no such path: " + args.fixture_root}
419
+ print(json.dumps(payload, indent=2) if args.json
420
+ else "INPUT MISSING -- no such path: " + args.fixture_root)
421
+ return 66
422
+
423
+ root = args.fixture_root or os.path.join(_ROOT, FIXTURE_GLOB)
424
+
425
+ try:
426
+ report = analyse(root)
427
+ except ScanError as exc:
428
+ payload = {"status": "scan_failed", "exit_code": 2, "error": str(exc)}
429
+ print(json.dumps(payload, indent=2) if args.json
430
+ else "CANNOT SCAN -- " + str(exc))
431
+ return 2
432
+
433
+ code = _exit_code(report)
434
+ if args.json:
435
+ print(json.dumps({
436
+ "status": "nothing_to_analyze" if code == 3 else "ranked",
437
+ "exit_code": code,
438
+ "advisory": ADVISORY,
439
+ "measured_deletion_value": "UNKNOWN",
440
+ "report": report,
441
+ }, indent=2))
442
+ else:
443
+ print(_render(report, code, max(args.top, 0)))
444
+ return code
445
+
446
+
447
+ if __name__ == "__main__":
448
+ sys.exit(main())
@@ -71,6 +71,7 @@ import json
71
71
  import os
72
72
  import pathlib
73
73
  import sys
74
+ import time
74
75
 
75
76
  # A stale .pyc for a hyphenated module loaded by path makes mutation probes
76
77
  # report FALSE failures (the probe edits the source, the loader serves the old
@@ -81,6 +82,24 @@ _ROOT = pathlib.Path(__file__).resolve().parents[1]
81
82
  _LIB = _ROOT / "autonomy" / "lib"
82
83
 
83
84
 
85
+ class _Parser(argparse.ArgumentParser):
86
+ """Usage errors exit 64, not argparse's default 2.
87
+
88
+ In this repo's convention 2 means "could NOT be checked" -- a real
89
+ answer about the subject. A mistyped flag is not that: it is an error
90
+ about the INVOCATION, and nothing about the subject was examined. The
91
+ two call for opposite responses, since retrying cannot fix a typo.
92
+
93
+ argparse exits 2 for every usage error unless this is overridden, so
94
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
95
+ """
96
+
97
+ def error(self, message):
98
+ self.print_usage(sys.stderr)
99
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
100
+ raise SystemExit(64)
101
+
102
+
84
103
  def _load(name, path):
85
104
  spec = importlib.util.spec_from_file_location(name, path)
86
105
  mod = importlib.util.module_from_spec(spec)
@@ -178,9 +197,60 @@ def find_receipts(workspace):
178
197
  receipt archived elsewhere in the workspace still gets audited. Missing
179
198
  evidence is the failure mode this file exists to prevent; over-collecting
180
199
  is not.
200
+
201
+ UNBOUNDED ON PURPOSE, and only safe because every caller of THIS function
202
+ is a CLI auditing a workspace the operator chose. Truncating such an audit
203
+ silently is the missing-evidence failure above. A caller reached from an
204
+ HTTP request must use find_receipts_bounded() instead: there the walk is a
205
+ denial-of-service surface, not a chore.
206
+ """
207
+ paths, _ = find_receipts_bounded(workspace)
208
+ return paths
209
+
210
+
211
+ def find_receipts_bounded(workspace, max_entries=None, max_seconds=None):
212
+ """find_receipts with explicit limits, returning (paths, truncated_reason).
213
+
214
+ truncated_reason is None on a COMPLETE walk, otherwise a string naming the
215
+ limit that stopped it. It is a return value rather than a log line because
216
+ a caller must not be able to report a partial audit as a complete one: a
217
+ truncated walk that looks complete lets a FAILED receipt sitting past the
218
+ cutoff read as "no problems found", which is the laundering this whole
219
+ module exists to prevent.
220
+
221
+ Both limits default to None, so this is a superset of find_receipts and the
222
+ unbounded CLI path keeps its exact behaviour.
181
223
  """
182
224
  root = pathlib.Path(workspace)
183
- return sorted(p for p in root.rglob("proof.json") if p.is_file())
225
+ started = time.monotonic()
226
+ found = []
227
+ scanned = 0
228
+ reason = None
229
+ # rglob("proof.json") yields only MATCHES, so counting its results counts
230
+ # receipts, not work. The cost being bounded here is the TRAVERSAL: a tree
231
+ # of 400 empty directories yields zero matches while still walking every
232
+ # one of them. os.walk exposes the directories actually visited, which is
233
+ # the quantity that makes this a denial-of-service surface.
234
+ for dirpath, dirnames, filenames in os.walk(str(root)):
235
+ scanned += 1 + len(filenames)
236
+ if max_entries is not None and scanned > max_entries:
237
+ reason = ("stopped after scanning %d entries (limit); results are "
238
+ "PARTIAL" % max_entries)
239
+ break
240
+ if max_seconds is not None and (time.monotonic() - started) > max_seconds:
241
+ reason = ("stopped after %.1fs (limit); results are PARTIAL"
242
+ % max_seconds)
243
+ break
244
+ if "proof.json" in filenames:
245
+ p = pathlib.Path(dirpath) / "proof.json"
246
+ try:
247
+ if p.is_file():
248
+ found.append(p)
249
+ except OSError:
250
+ # Vanished mid-walk or unreadable: skipped rather than
251
+ # aborting the whole audit.
252
+ continue
253
+ return sorted(found), reason
184
254
 
185
255
 
186
256
  def bundle(workspace, repo_dir="."):
@@ -296,7 +366,7 @@ def _render(report):
296
366
 
297
367
 
298
368
  def main(argv=None):
299
- ap = argparse.ArgumentParser(
369
+ ap = _Parser(
300
370
  description="Verify every receipt under a workspace as one bundle.")
301
371
  ap.add_argument("workspace", nargs="?", default=".",
302
372
  help="workspace to scan for receipts (default: .)")
@@ -54,6 +54,24 @@ class NotComparable(Exception):
54
54
  """Raised when two receipts must not be diffed at all."""
55
55
 
56
56
 
57
+ class _Parser(argparse.ArgumentParser):
58
+ """Usage errors exit 64, not argparse's default 2.
59
+
60
+ In this repo's convention 2 means "could NOT be checked" -- a real
61
+ answer about the subject. A mistyped flag is not that: it is an error
62
+ about the INVOCATION, and nothing about the subject was examined. The
63
+ two call for opposite responses, since retrying cannot fix a typo.
64
+
65
+ argparse exits 2 for every usage error unless this is overridden, so
66
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
67
+ """
68
+
69
+ def error(self, message):
70
+ self.print_usage(sys.stderr)
71
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
72
+ raise SystemExit(64)
73
+
74
+
57
75
  def delta(a, b):
58
76
  """b - a, or UNKNOWN when either side was never measured.
59
77
 
@@ -274,7 +292,7 @@ def render(d):
274
292
 
275
293
 
276
294
  def main(argv=None):
277
- ap = argparse.ArgumentParser(
295
+ ap = _Parser(
278
296
  description="Compare two Evidence Receipts (proof.json).")
279
297
  ap.add_argument("a", help="baseline proof.json")
280
298
  ap.add_argument("b", help="proof.json to compare against the baseline")
@@ -77,6 +77,24 @@ sys.dont_write_bytecode = True
77
77
  _ROOT = pathlib.Path(__file__).resolve().parents[1]
78
78
 
79
79
 
80
+ class _Parser(argparse.ArgumentParser):
81
+ """Usage errors exit 64, not argparse's default 2.
82
+
83
+ In this repo's convention 2 means "could NOT be checked" -- a real
84
+ answer about the subject. A mistyped flag is not that: it is an error
85
+ about the INVOCATION, and nothing about the subject was examined. The
86
+ two call for opposite responses, since retrying cannot fix a typo.
87
+
88
+ argparse exits 2 for every usage error unless this is overridden, so
89
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
90
+ """
91
+
92
+ def error(self, message):
93
+ self.print_usage(sys.stderr)
94
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
95
+ raise SystemExit(64)
96
+
97
+
80
98
  def _load(name, path):
81
99
  spec = importlib.util.spec_from_file_location(name, path)
82
100
  mod = importlib.util.module_from_spec(spec)
@@ -293,7 +311,7 @@ def _since(value):
293
311
 
294
312
 
295
313
  def main(argv=None):
296
- ap = argparse.ArgumentParser(
314
+ ap = _Parser(
297
315
  description="Find receipts under a workspace by measurable criteria.")
298
316
  ap.add_argument("workspace", nargs="?", default=".",
299
317
  help="workspace to search for receipts (default: .)")