loki-mode 9.8.1 → 9.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +19 -14
  2. package/SKILL.md +3 -2
  3. package/VERSION +1 -1
  4. package/autonomy/loki +122 -1
  5. package/autonomy/run.sh +49 -2
  6. package/dashboard/__init__.py +1 -1
  7. package/dashboard/api_evidence.py +411 -0
  8. package/dashboard/api_operator.py +283 -0
  9. package/dashboard/api_phases.py +262 -0
  10. package/dashboard/api_releases.py +242 -0
  11. package/dashboard/api_runs.py +477 -0
  12. package/dashboard/api_tests.py +444 -0
  13. package/dashboard/api_v2.py +47 -1
  14. package/dashboard/server.py +54 -0
  15. package/dashboard/static/index.html +246 -135
  16. package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
  17. package/docs/CAPABILITY-BACKLOG.md +53 -0
  18. package/docs/COMPARISON.md +2 -2
  19. package/docs/COMPETITIVE-ANALYSIS.md +1 -1
  20. package/docs/COMPETITIVE-SCORECARD.md +422 -0
  21. package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
  22. package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
  23. package/docs/DEMOS.md +21 -23
  24. package/docs/HANDOFF-2026-08-03.md +439 -0
  25. package/docs/INSTALLATION.md +17 -10
  26. package/docs/OUTCOME-FRONTIER.md +536 -0
  27. package/docs/PROMPT-ABLATION-RESULT.md +97 -0
  28. package/docs/TOOLS.md +800 -0
  29. package/docs/alternative-installations.md +2 -3
  30. package/docs/audit-logging.md +44 -35
  31. package/docs/authentication.md +13 -2
  32. package/docs/authorization.md +87 -81
  33. package/docs/git-workflow.md +6 -3
  34. package/docs/metrics.md +15 -16
  35. package/docs/network-security.md +16 -13
  36. package/docs/openclaw-integration.md +36 -556
  37. package/docs/show-hn-post.md +2 -2
  38. package/docs/siem-integration.md +39 -36
  39. package/loki-ts/dist/loki.js +18 -18
  40. package/mcp/__init__.py +1 -1
  41. package/package.json +1 -1
  42. package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
  43. package/references/confidence-routing.md +18 -1
  44. package/references/invariant-checks.md +13 -8
  45. package/references/magic-rarv-integration.md +0 -1
  46. package/references/multi-provider.md +27 -5
  47. package/skills/healing.md +4 -2
  48. package/tools/audit-docs.py +488 -0
  49. package/tools/baseline-pin.py +19 -1
  50. package/tools/calibration-audit.py +523 -0
  51. package/tools/ci-gate.py +19 -1
  52. package/tools/cost-forecast.py +344 -0
  53. package/tools/cost-guard.py +19 -1
  54. package/tools/cost-history.py +19 -1
  55. package/tools/cost-per-outcome.py +394 -0
  56. package/tools/estimate-run.py +19 -1
  57. package/tools/evidence-freshness.py +307 -0
  58. package/tools/gate-init.py +19 -1
  59. package/tools/gate-report.py +19 -1
  60. package/tools/gate-simulate.py +570 -0
  61. package/tools/gate-trend.py +354 -0
  62. package/tools/model-advisor.py +52 -1
  63. package/tools/policy-load.py +19 -1
  64. package/tools/prompt-cost.py +363 -0
  65. package/tools/prompt-diff.py +448 -0
  66. package/tools/prompt-lint.py +448 -0
  67. package/tools/receipt-bundle.py +72 -2
  68. package/tools/receipt-diff.py +19 -1
  69. package/tools/receipt-find.py +19 -1
  70. package/tools/receipt-stats.py +380 -0
  71. package/tools/receipt-timeline.py +478 -0
  72. package/tools/receipt-verify-batch.py +291 -0
  73. package/tools/run-replay.py +19 -1
  74. package/tools/signing-status.py +19 -1
  75. package/tools/token-guard.py +19 -1
  76. package/tools/token-tax.py +375 -0
  77. package/tools/tool-index.py +19 -1
  78. package/tools/verification-tax.py +277 -0
  79. package/tools/verify-chain.py +361 -0
@@ -0,0 +1,344 @@
1
+ #!/usr/bin/env python3
2
+ """Project agent spend forward from recorded history, with an honest basis.
3
+
4
+ cost-history.py answers "is our spend trending up?" over runs already made.
5
+ The question that follows immediately, and the one a budget owner actually
6
+ asks, is forward-looking: we are about to do N more runs, what will they cost?
7
+ That is a PROJECTION, and a projection is the easiest place in this repo to
8
+ launder an absent measurement into a confident number.
9
+
10
+ THE FAILURE MODE THIS TOOL IS SHAPED AROUND. A forecast is arithmetic over a
11
+ sample, and arithmetic is happy to run on an empty sample: sum([]) is 0, and
12
+ 0 * N is 0. So the naive implementation answers "$0.00" for a project with no
13
+ recorded history at all, which is not a cautious estimate -- it is the single
14
+ most confident claim the tool can make ("these runs are free"), produced at the
15
+ exact moment it knows nothing. An earlier attempt at this tool did precisely
16
+ that, projecting $0.00 from zero measured runs.
17
+
18
+ So the rule, and it is the whole tool:
19
+
20
+ NO MEASURED HISTORY MEANS NO FORECAST. Not a zero, not a cautious low
21
+ estimate, not a wide range around zero. No number at all.
22
+
23
+ Absent is not zero. A forecast is a claim about the future built from a sample
24
+ of the past, and with an empty sample there is no claim to make. Exit 3
25
+ (nothing to check), print the absence in words, and emit no dollar figure
26
+ anywhere -- including in --json, where `projected_usd` is null.
27
+
28
+ FOUR MORE PROPERTIES, each the honest half of a shortcut that would read better:
29
+
30
+ 1. A TREND FROM ONE POINT IS NOT A TREND, but it is still a measurement. With
31
+ exactly one measured run this DOES forecast -- refusing would throw away
32
+ real data, which is the over-strict mirror of the same dishonesty -- and it
33
+ labels the projection SINGLE OBSERVATION, stating in words that it assumes
34
+ a flat rate from one point. The reader gets the number AND the reason to
35
+ distrust it. Silently presenting it like a 40-run mean is the lie.
36
+
37
+ 2. THE BASIS IS PRINTED ON EVERY PROJECTION, never on request. "M measured of
38
+ N recorded" is what makes the number auditable: $4.20 from 40 measured runs
39
+ and $4.20 from 1 measured run of 39 are different facts, and the number
40
+ alone cannot tell them apart. A basis available behind a --verbose flag is a
41
+ basis nobody reads.
42
+
43
+ 3. AN UNMEASURED ROW IS EXCLUDED, NEVER READ AS 0. cost-history.py records an
44
+ unmeasured run as usd=null on purpose, so the measurement gap stays
45
+ countable. Averaging those nulls as 0 would drag the mean down and understate
46
+ the forecast -- and it would do so worst when instrumentation is broken, i.e.
47
+ when the estimate matters most. Excluded rows are COUNTED and reported.
48
+
49
+ 4. A MEASURED ZERO IS NOT AN UNMEASURED ROW. A run that genuinely cost $0.0000
50
+ (cached, free tier) is data and stays in the mean as 0. This is why every
51
+ test here is `is None` and never a truthiness check: `if usd:` would drop a
52
+ real zero and silently bias the forecast upward. record_is_measured() is
53
+ imported for the legacy row that carries no `measured` flag, and is
54
+ deliberately NOT applied to the flagged rows -- it is a truthiness predicate
55
+ over token fields, so it answers False for a measured $0.00, which is
56
+ correct for its own question and wrong for this one.
57
+
58
+ 5. A CORRUPT LINE IS COUNTED AND REPORTED, never silently skipped. A history
59
+ that quietly drops rows forecasts from a cleaner sample than reality.
60
+
61
+ WHY THE MEAN AND NOT THE MEDIAN. cost-history.py trends on the median, because
62
+ a trend must not be named by one outlier. A forecast of TOTAL spend over N runs
63
+ is a different question: the expensive runs are real money and the total has to
64
+ include them. mean * N is the unbiased estimator of that total; median * N
65
+ systematically under-forecasts any right-skewed cost distribution, which agent
66
+ spend always is. Both numbers are printed so the skew is visible.
67
+
68
+ NOT A GATE. This advises a budget owner; no CI job branches on its exit code,
69
+ and it never blocks a merge. Exit 3 on an absent basis says "nothing to check",
70
+ which is the honest answer to "what will N runs cost" when nothing was measured.
71
+
72
+ Usage:
73
+ tools/cost-forecast.py --runs N [--file .loki/cost-history.jsonl] [--json]
74
+
75
+ Exit: 0 forecast produced, 2 history unreadable, 3 nothing to forecast from,
76
+ 64 usage error, 66 history file does not exist.
77
+ """
78
+
79
+ import sys
80
+
81
+ sys.dont_write_bytecode = True
82
+
83
+ import argparse # noqa: E402
84
+ import json # noqa: E402
85
+ import os # noqa: E402
86
+
87
+ _HERE = os.path.dirname(os.path.abspath(__file__))
88
+ sys.path.insert(0, os.path.join(os.path.dirname(_HERE), "autonomy", "lib"))
89
+
90
+ from efficiency_cost import record_is_measured # noqa: E402
91
+
92
+ OK, CANNOT, NOTHING_TO_CHECK, USAGE, NO_INPUT = 0, 2, 3, 64, 66
93
+
94
+ DEFAULT_FILE = os.path.join(".loki", "cost-history.jsonl")
95
+
96
+
97
+ class _Parser(argparse.ArgumentParser):
98
+ """argparse exits 2 on a usage error, and 2 here means "could not check".
99
+
100
+ Those are opposite facts. "You typed the flag wrong" and "the instrument is
101
+ blind" must not share an exit code: a typo would masquerade as a blind
102
+ gate, and the operator would go hunting for missing instrumentation.
103
+ """
104
+
105
+ def error(self, message):
106
+ self.print_usage(sys.stderr)
107
+ print("%s: error: %s" % (self.prog, message), file=sys.stderr)
108
+ raise SystemExit(USAGE)
109
+
110
+
111
+ def _num(v):
112
+ """A number as itself; None, "", or a bool as None. A real 0 survives as 0."""
113
+ if isinstance(v, bool) or not isinstance(v, (int, float)):
114
+ return None
115
+ return v
116
+
117
+
118
+ def row_usd(row):
119
+ """The measured USD for one history row, or None when it was not measured.
120
+
121
+ THE ONE PLACE the measured/unmeasured distinction is decided, so it cannot
122
+ drift between the loader and the reporter.
123
+
124
+ `measured` is cost-history.py's own flag, written at record time by
125
+ record_is_measured(), so trusting it REUSES that rule rather than restating
126
+ it. A legacy row without the flag falls back to record_is_measured() here.
127
+
128
+ The final test is `is not None`, never truthiness: a genuine $0.0000 run is
129
+ a measurement and must survive as 0.0. Dropping it would bias every
130
+ forecast upward, and would do it invisibly.
131
+
132
+ THE ORDER OF THE LEGACY BRANCH IS LOAD-BEARING. record_is_measured() is a
133
+ TRUTHINESS predicate over token fields, so it answers False for a measured
134
+ $0.0000 -- correct for its own question ("did this record observe
135
+ anything?"), wrong for this one ("is this dollar figure present?"). Asking
136
+ it first would drop a real zero from a flag-less row while the identical
137
+ flagged row was kept, so the same fact got two answers depending on which
138
+ version of cost-history.py wrote it. `usd is None` is therefore tested
139
+ FIRST and short-circuits; record_is_measured() only ever adjudicates rows
140
+ whose usd is already absent.
141
+ """
142
+ if not isinstance(row, dict):
143
+ return None
144
+ usd = _num(row.get("usd"))
145
+ if "measured" in row:
146
+ if row.get("measured") is not True:
147
+ return None
148
+ elif usd is None and not record_is_measured(row):
149
+ return None
150
+ return usd
151
+
152
+
153
+ def load(path):
154
+ """Read the history. Returns (rows, corrupt_count) or (None, 0) if unreadable.
155
+
156
+ A line that is not JSON, or is JSON but not an object carrying `usd`, is
157
+ corrupt and COUNTED. It is never dropped on the floor: a history that
158
+ quietly loses rows forecasts from a tidier sample than the real one, and
159
+ loses them most often when something upstream just broke.
160
+ """
161
+ rows, corrupt = [], 0
162
+ try:
163
+ with open(path, "r", encoding="utf-8") as handle:
164
+ raw = handle.read()
165
+ except OSError:
166
+ return None, 0
167
+ for line in raw.splitlines():
168
+ if not line.strip():
169
+ continue
170
+ try:
171
+ obj = json.loads(line)
172
+ except ValueError:
173
+ corrupt += 1
174
+ continue
175
+ if not isinstance(obj, dict) or "usd" not in obj:
176
+ corrupt += 1
177
+ continue
178
+ rows.append(obj)
179
+ return rows, corrupt
180
+
181
+
182
+ def median(values):
183
+ """Median of a non-empty list. Even length averages the middle pair."""
184
+ s = sorted(values)
185
+ n = len(s)
186
+ mid = n // 2
187
+ return s[mid] if n % 2 else (s[mid - 1] + s[mid]) / 2.0
188
+
189
+
190
+ def forecast(path, runs):
191
+ """Project spend over `runs` future runs. Returns a verdict dict.
192
+
193
+ Every absence path returns projected_usd=None. There is no branch in this
194
+ function that produces a number from an empty sample, which is the property
195
+ the whole file exists to hold.
196
+ """
197
+ base = {
198
+ "runs": runs,
199
+ "recorded": 0,
200
+ "measured": 0,
201
+ "excluded_unmeasured": 0,
202
+ "corrupt_lines": 0,
203
+ "mean_usd": None,
204
+ "median_usd": None,
205
+ "projected_usd": None,
206
+ "confidence": "NONE",
207
+ "basis": None,
208
+ }
209
+
210
+ if not os.path.exists(path):
211
+ base["status"] = "no_history"
212
+ base["exit_code"] = NO_INPUT
213
+ base["why"] = ("no history file at %s -- record runs with "
214
+ "tools/cost-history.py record first. There is no "
215
+ "sample to project from, so there is no forecast."
216
+ % path)
217
+ return base
218
+
219
+ rows, corrupt = load(path)
220
+ if rows is None:
221
+ base["status"] = "unreadable"
222
+ base["exit_code"] = CANNOT
223
+ base["why"] = ("history at %s exists but could not be read; the "
224
+ "instrument is blind, which is not the same as a "
225
+ "cheap forecast." % path)
226
+ return base
227
+
228
+ base["recorded"] = len(rows)
229
+ base["corrupt_lines"] = corrupt
230
+
231
+ costs = []
232
+ unmeasured = 0
233
+ for row in rows:
234
+ usd = row_usd(row)
235
+ if usd is None:
236
+ unmeasured += 1
237
+ else:
238
+ costs.append(usd)
239
+ base["measured"] = len(costs)
240
+ base["excluded_unmeasured"] = unmeasured
241
+
242
+ if not costs:
243
+ # THE PATH THAT MUST NEVER PRODUCE A NUMBER. Zero measured runs, whether
244
+ # from an empty file, an all-null history, or nothing but corrupt lines.
245
+ # sum([])/0 does not even divide, and the tempting repair -- treat the
246
+ # nulls as 0 -- is the lie itself. projected_usd stays None.
247
+ base["status"] = "no_measured_history"
248
+ base["exit_code"] = NOTHING_TO_CHECK
249
+ base["why"] = (
250
+ # No dollar SIGN anywhere on this path, not even in prose. The
251
+ # tests assert the absence of "$" in the whole output, which is one
252
+ # assertion that catches every number at once -- worth more than
253
+ # the rhetorical flourish of quoting the figure being refused.
254
+ "%d run(s) recorded, 0 measured%s. A forecast is arithmetic over a "
255
+ "sample and there is no sample, so no figure is given: an absent "
256
+ "measurement reads UNKNOWN, never zero."
257
+ % (len(rows),
258
+ " (%d unmeasured, excluded not counted as 0)" % unmeasured
259
+ if unmeasured else ""))
260
+ return base
261
+
262
+ mean = sum(costs) / float(len(costs))
263
+ base["mean_usd"] = mean
264
+ base["median_usd"] = median(costs)
265
+ base["projected_usd"] = mean * runs
266
+ base["status"] = "ok"
267
+ base["exit_code"] = OK
268
+
269
+ if len(costs) < 2:
270
+ # One point is a measurement but not a trend. Forecast anyway -- the
271
+ # data is real -- and say plainly what the projection assumes, so the
272
+ # number is never mistaken for a 40-run mean.
273
+ base["confidence"] = "SINGLE OBSERVATION"
274
+ base["assumption"] = (
275
+ "flat rate assumed from a SINGLE observation; one point cannot "
276
+ "establish a trend, so this projects that one run's cost forward "
277
+ "unchanged and says nothing about whether spend is rising.")
278
+ else:
279
+ base["confidence"] = "MULTI OBSERVATION"
280
+ base["assumption"] = (
281
+ "flat rate assumed: the mean of %d measured run(s) projected "
282
+ "forward. This does not model a trend; use "
283
+ "tools/cost-history.py report for direction." % len(costs))
284
+
285
+ base["basis"] = ("%d measured of %d recorded run(s)"
286
+ % (len(costs), len(rows)))
287
+ return base
288
+
289
+
290
+ def render(d):
291
+ """Text rendering. Every absence branch returns BEFORE any %-format of a
292
+ dollar figure, so a None projection can never reach a float format."""
293
+ if d["status"] == "no_history":
294
+ return "NO FORECAST: %s" % d["why"]
295
+ if d["status"] == "unreadable":
296
+ return "CANNOT FORECAST: %s" % d["why"]
297
+
298
+ lines = []
299
+ if d["corrupt_lines"]:
300
+ lines.append("%d CORRUPT line(s) in the history -- counted, not "
301
+ "skipped; excluded from the sample below."
302
+ % d["corrupt_lines"])
303
+ if d["excluded_unmeasured"]:
304
+ lines.append("%d unmeasured run(s) EXCLUDED (recorded as null, never "
305
+ "averaged as 0)." % d["excluded_unmeasured"])
306
+
307
+ if d["status"] == "no_measured_history":
308
+ lines.append("NO FORECAST: %s" % d["why"])
309
+ return "\n".join(lines)
310
+
311
+ lines.append("basis: %s" % d["basis"])
312
+ lines.append("mean $%.4f/run, median $%.4f/run"
313
+ % (d["mean_usd"], d["median_usd"]))
314
+ lines.append("projected over %d run(s): $%.4f"
315
+ % (d["runs"], d["projected_usd"]))
316
+ lines.append("confidence: %s" % d["confidence"])
317
+ lines.append(d["assumption"])
318
+ return "\n".join(lines)
319
+
320
+
321
+ def main(argv=None):
322
+ ap = _Parser(
323
+ description="Project agent spend forward from recorded cost history.")
324
+ ap.add_argument("--runs", type=int, required=True,
325
+ help="number of future runs to project over")
326
+ ap.add_argument("--file", default=DEFAULT_FILE,
327
+ help="history JSONL (default %s)" % DEFAULT_FILE)
328
+ ap.add_argument("--json", action="store_true", dest="as_json",
329
+ help="emit the forecast as JSON")
330
+
331
+ args = ap.parse_args(argv)
332
+ if args.runs < 1:
333
+ # Not a forecast question. Refusing beats answering $0.00 for 0 runs,
334
+ # which is a real number that reads like a measurement.
335
+ ap.error("--runs must be 1 or greater (got %d)" % args.runs)
336
+
337
+ d = forecast(args.file, args.runs)
338
+ print(json.dumps(d, indent=2, sort_keys=True) if args.as_json
339
+ else render(d))
340
+ return d["exit_code"]
341
+
342
+
343
+ if __name__ == "__main__":
344
+ sys.exit(main())
@@ -69,6 +69,24 @@ from efficiency_cost import collect_efficiency, record_is_measured # noqa: E402
69
69
  OK, OVER, CANNOT = 0, 1, 2
70
70
 
71
71
 
72
+ class _Parser(argparse.ArgumentParser):
73
+ """Usage errors exit 64, not argparse's default 2.
74
+
75
+ In this repo's convention 2 means "could NOT be checked" -- a real
76
+ answer about the subject. A mistyped flag is not that: it is an error
77
+ about the INVOCATION, and nothing about the subject was examined. The
78
+ two call for opposite responses, since retrying cannot fix a typo.
79
+
80
+ argparse exits 2 for every usage error unless this is overridden, so
81
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
82
+ """
83
+
84
+ def error(self, message):
85
+ self.print_usage(sys.stderr)
86
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
87
+ raise SystemExit(64)
88
+
89
+
72
90
  def _num(v):
73
91
  """A number as itself; None, "", or a bool as None."""
74
92
  if isinstance(v, bool) or not isinstance(v, (int, float)):
@@ -214,7 +232,7 @@ def render(d):
214
232
 
215
233
 
216
234
  def main(argv=None):
217
- ap = argparse.ArgumentParser(
235
+ ap = _Parser(
218
236
  description="Fail CI when a run's cost regressed past a budget policy.")
219
237
  ap.add_argument("workspace", nargs="?", default=".",
220
238
  help="workspace root (or its .loki dir); default .")
@@ -67,6 +67,24 @@ DEFAULT_FILE = os.path.join(".loki", "cost-history.jsonl")
67
67
  FLAT_PCT = 5.0
68
68
 
69
69
 
70
+ class _Parser(argparse.ArgumentParser):
71
+ """Usage errors exit 64, not argparse's default 2.
72
+
73
+ In this repo's convention 2 means "could NOT be checked" -- a real
74
+ answer about the subject. A mistyped flag is not that: it is an error
75
+ about the INVOCATION, and nothing about the subject was examined. The
76
+ two call for opposite responses, since retrying cannot fix a typo.
77
+
78
+ argparse exits 2 for every usage error unless this is overridden, so
79
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
80
+ """
81
+
82
+ def error(self, message):
83
+ self.print_usage(sys.stderr)
84
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
85
+ raise SystemExit(64)
86
+
87
+
70
88
  def _num(v):
71
89
  """A number as itself; None, "", or a bool as None."""
72
90
  if isinstance(v, bool) or not isinstance(v, (int, float)):
@@ -287,7 +305,7 @@ def render(d):
287
305
 
288
306
 
289
307
  def main(argv=None):
290
- ap = argparse.ArgumentParser(
308
+ ap = _Parser(
291
309
  description="Track agent cost across many runs and report the trend.")
292
310
  sub = ap.add_subparsers(dest="cmd", required=True)
293
311