loki-mode 9.8.0 → 9.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -14
- package/SKILL.md +3 -2
- package/VERSION +1 -1
- package/autonomy/loki +122 -1
- package/autonomy/run.sh +49 -2
- package/dashboard/__init__.py +1 -1
- package/dashboard/api_evidence.py +411 -0
- package/dashboard/api_operator.py +283 -0
- package/dashboard/api_phases.py +262 -0
- package/dashboard/api_releases.py +242 -0
- package/dashboard/api_runs.py +477 -0
- package/dashboard/api_tests.py +444 -0
- package/dashboard/api_v2.py +47 -1
- package/dashboard/server.py +54 -0
- package/dashboard/static/index.html +246 -135
- package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
- package/docs/CAPABILITY-BACKLOG.md +53 -0
- package/docs/COMPARISON.md +2 -2
- package/docs/COMPETITIVE-ANALYSIS.md +1 -1
- package/docs/COMPETITIVE-SCORECARD.md +422 -0
- package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
- package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
- package/docs/DEMOS.md +21 -23
- package/docs/HANDOFF-2026-08-03.md +439 -0
- package/docs/INSTALLATION.md +17 -10
- package/docs/OUTCOME-FRONTIER.md +536 -0
- package/docs/PROMPT-ABLATION-RESULT.md +97 -0
- package/docs/TOOLS.md +800 -0
- package/docs/alternative-installations.md +2 -3
- package/docs/audit-logging.md +44 -35
- package/docs/authentication.md +13 -2
- package/docs/authorization.md +87 -81
- package/docs/git-workflow.md +6 -3
- package/docs/metrics.md +15 -16
- package/docs/network-security.md +16 -13
- package/docs/openclaw-integration.md +36 -556
- package/docs/show-hn-post.md +2 -2
- package/docs/siem-integration.md +39 -36
- package/loki-ts/dist/loki.js +18 -18
- package/mcp/__init__.py +1 -1
- package/package.json +2 -2
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
- package/references/confidence-routing.md +18 -1
- package/references/invariant-checks.md +13 -8
- package/references/magic-rarv-integration.md +0 -1
- package/references/multi-provider.md +27 -5
- package/skills/healing.md +4 -2
- package/tools/audit-docs.py +488 -0
- package/tools/baseline-pin.py +19 -1
- package/tools/calibration-audit.py +523 -0
- package/tools/ci-gate.py +19 -1
- package/tools/cost-forecast.py +344 -0
- package/tools/cost-guard.py +19 -1
- package/tools/cost-history.py +19 -1
- package/tools/cost-per-outcome.py +394 -0
- package/tools/estimate-run.py +19 -1
- package/tools/evidence-freshness.py +307 -0
- package/tools/gate-init.py +19 -1
- package/tools/gate-report.py +19 -1
- package/tools/gate-simulate.py +570 -0
- package/tools/gate-trend.py +354 -0
- package/tools/model-advisor.py +52 -1
- package/tools/policy-load.py +19 -1
- package/tools/prompt-cost.py +363 -0
- package/tools/prompt-diff.py +448 -0
- package/tools/prompt-lint.py +448 -0
- package/tools/receipt-bundle.py +72 -2
- package/tools/receipt-diff.py +19 -1
- package/tools/receipt-find.py +19 -1
- package/tools/receipt-stats.py +380 -0
- package/tools/receipt-timeline.py +478 -0
- package/tools/receipt-verify-batch.py +291 -0
- package/tools/run-replay.py +19 -1
- package/tools/signing-status.py +19 -1
- package/tools/token-guard.py +19 -1
- package/tools/token-tax.py +375 -0
- package/tools/tool-index.py +19 -1
- package/tools/verification-tax.py +277 -0
- package/tools/verify-chain.py +361 -0
|
@@ -0,0 +1,478 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""What happened across an archive of receipts, in time order, and what CHANGED.
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. receipt-stats.py answers "what IS this archive" -- a census,
|
|
5
|
+
one number per axis, no order. receipt-diff.py answers "how do these TWO runs
|
|
6
|
+
compare". Neither answers the question an operator asks when something started
|
|
7
|
+
going wrong: what changed, and WHEN. That question is answered today by
|
|
8
|
+
sorting proof.json paths by eye and subtracting two cost figures in your head,
|
|
9
|
+
which is exactly the pipeline that gets the answer wrong in a predictable
|
|
10
|
+
direction -- toward a tidy, continuous, complete-looking story.
|
|
11
|
+
|
|
12
|
+
A timeline is more dangerous than a census, because a timeline asserts things a
|
|
13
|
+
census never does: that these events are ORDERED, that they are ADJACENT, and
|
|
14
|
+
that the difference between two neighbours is a real change in the world. Every
|
|
15
|
+
rule below exists because one of those three assertions can be made without
|
|
16
|
+
evidence.
|
|
17
|
+
|
|
18
|
+
THE RULES.
|
|
19
|
+
|
|
20
|
+
1. A RECEIPT WITH NO READABLE TIMESTAMP IS COUNTED AND REPORTED AS UNDATED.
|
|
21
|
+
Never dropped, and never given an invented position in the order. Dropping
|
|
22
|
+
it reports a shorter, cleaner history than exists -- and does it invisibly,
|
|
23
|
+
which is the whole defect class. Slotting it in "where it probably goes"
|
|
24
|
+
(path order, mtime, the end of the list) is worse: it manufactures an
|
|
25
|
+
ADJACENCY the data does not support, and every delta computed across that
|
|
26
|
+
adjacency is a fabricated change between two runs that may be months apart.
|
|
27
|
+
|
|
28
|
+
So undated receipts land in their own section, they are counted in the
|
|
29
|
+
census, and they carry NO delta at all -- not a zero one.
|
|
30
|
+
|
|
31
|
+
2. A COST DELTA AGAINST AN UNMEASURED RECEIPT IS UNKNOWN, NEVER A NUMBER.
|
|
32
|
+
This is v8.51.0-v8.54.0 in its most seductive form. `curr - prev` with an
|
|
33
|
+
absent measurement read as 0.0 does not merely print a wrong figure, it
|
|
34
|
+
prints a wrong STORY: the run whose instrumentation broke shows up as the
|
|
35
|
+
moment cost collapsed to nothing, or as the moment it exploded, depending
|
|
36
|
+
on which side of the pair went dark. An unmeasured neighbour means there is
|
|
37
|
+
no delta to state.
|
|
38
|
+
|
|
39
|
+
The predicate is record_is_measured() in autonomy/lib/efficiency_cost.py,
|
|
40
|
+
reached through receipt-diff.py's measured_cost(), which already maps the
|
|
41
|
+
receipt's `cost.usd` onto the per-iteration `cost_usd` key that predicate
|
|
42
|
+
reads. Neither half is restated here; a second copy is how the rule drifts.
|
|
43
|
+
|
|
44
|
+
Measured is necessary but not sufficient. record_is_measured is true when
|
|
45
|
+
ANY of five fields is non-zero, so a receipt carrying tokens and a null
|
|
46
|
+
`usd` is honestly measured and still has no dollar figure. Both are
|
|
47
|
+
required -- see _usd().
|
|
48
|
+
|
|
49
|
+
3. A MEASURED ZERO DELTA READS 0, NOT UNKNOWN. `is None` throughout, never
|
|
50
|
+
truthiness. "The cost did not change" is a finding; a run that genuinely
|
|
51
|
+
cost the same as the one before it is the single most useful cell in this
|
|
52
|
+
table, and `if delta:` erases exactly that cell. The two ways to be
|
|
53
|
+
dishonest about a measurement are to invent one and to discard one, and
|
|
54
|
+
rule 3 is rule 2 pointed the other way.
|
|
55
|
+
|
|
56
|
+
4. FEWER THAN TWO RECEIPTS CANNOT SHOW A CHANGE, AND SAYS SO. One receipt has
|
|
57
|
+
nothing to be compared against. Printing it with an empty change column
|
|
58
|
+
renders as "nothing changed" -- a claim about stability drawn from an
|
|
59
|
+
archive that could not have detected instability. That is stated in words,
|
|
60
|
+
and it is a DIFFERENT statement from "this receipt is the first in a longer
|
|
61
|
+
chain", which is why they are separate messages.
|
|
62
|
+
|
|
63
|
+
5. ZERO RECEIPTS IS NOT A CLEAN ARCHIVE. It gets exit 3 and says in words that
|
|
64
|
+
it is most often the wrong directory.
|
|
65
|
+
|
|
66
|
+
WHAT THIS IS NOT. This is an ADVISOR, not a gate, and the distinction is
|
|
67
|
+
pinned by tests/test_tool_exit_contract.py. A FAILED receipt in the history
|
|
68
|
+
does not make this exit 1 -- receipt-bundle.py is the gate that refuses to
|
|
69
|
+
merge on one, and re-deriving its weakest-link rule here would be a second
|
|
70
|
+
copy of a verdict predicate. An archive whose costs were never measured still
|
|
71
|
+
exits 0: "every receipt is readable, and none recorded a cost" is a complete,
|
|
72
|
+
honest answer, and the output says UNKNOWN in words. Forcing it non-zero would
|
|
73
|
+
make the honest answer look like a tool failure.
|
|
74
|
+
|
|
75
|
+
Verdict classification is receipt_state() from receipt-bundle.py, wrapping
|
|
76
|
+
verify() from autonomy/lib/proof-verify.py. Nothing here re-implements it.
|
|
77
|
+
|
|
78
|
+
Usage:
|
|
79
|
+
tools/receipt-timeline.py [workspace] [--repo-dir DIR] [--json]
|
|
80
|
+
|
|
81
|
+
Exit codes:
|
|
82
|
+
0 receipts were found and laid out in time order
|
|
83
|
+
3 the workspace exists but holds no receipts -- nothing to show
|
|
84
|
+
64 usage error (unknown flag, bad argument)
|
|
85
|
+
66 the workspace path does not exist
|
|
86
|
+
"""
|
|
87
|
+
|
|
88
|
+
import argparse
|
|
89
|
+
import importlib.util
|
|
90
|
+
import json
|
|
91
|
+
import os
|
|
92
|
+
import pathlib
|
|
93
|
+
import sys
|
|
94
|
+
|
|
95
|
+
# A stale .pyc can mask a mutation and turn a real probe into a false
|
|
96
|
+
# "MUTATION SURVIVED", since invalidation is mtime+size and a restore is
|
|
97
|
+
# byte-identical. Set before the loader below runs.
|
|
98
|
+
sys.dont_write_bytecode = True
|
|
99
|
+
|
|
100
|
+
_ROOT = pathlib.Path(__file__).resolve().parents[1]
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _load(name, path):
|
|
104
|
+
spec = importlib.util.spec_from_file_location(name, path)
|
|
105
|
+
mod = importlib.util.module_from_spec(spec)
|
|
106
|
+
spec.loader.exec_module(mod)
|
|
107
|
+
return mod
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
# receipt_state() wraps verify() and keeps VERIFIED / FAILED / UNVERIFIABLE
|
|
111
|
+
# apart with the `is`-comparisons that make that correct. measured_cost()
|
|
112
|
+
# reuses record_is_measured() AND maps cost.usd -> cost_usd. Both imported,
|
|
113
|
+
# never restated.
|
|
114
|
+
_rb = _load("receipt_bundle", _ROOT / "tools" / "receipt-bundle.py")
|
|
115
|
+
receipt_state = _rb.receipt_state
|
|
116
|
+
measured_cost = _rb.measured_cost
|
|
117
|
+
|
|
118
|
+
VERIFIED = _rb.VERIFIED
|
|
119
|
+
UNVERIFIABLE = _rb.UNVERIFIABLE
|
|
120
|
+
FAILED = _rb.FAILED
|
|
121
|
+
|
|
122
|
+
MALFORMED = "MALFORMED"
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _usd(proof):
|
|
126
|
+
"""The receipt's cost in dollars, or None when there is no such number.
|
|
127
|
+
|
|
128
|
+
None means UNMEASURED, and every caller must abstain rather than
|
|
129
|
+
substitute. Two independent ways to be absent, both mapped to None: the
|
|
130
|
+
whole record was never measured, and the record was measured on tokens but
|
|
131
|
+
carries no `usd` figure. Rule 2 covers both, because a delta needs the
|
|
132
|
+
dollar number specifically.
|
|
133
|
+
"""
|
|
134
|
+
rec = measured_cost(proof)
|
|
135
|
+
if rec is None:
|
|
136
|
+
return None
|
|
137
|
+
return rec.get("cost_usd")
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _when(proof):
|
|
141
|
+
"""The receipt's generated_at as a sortable ISO string, or None.
|
|
142
|
+
|
|
143
|
+
Compared as a string, not parsed. generated_at ends in "Z", which
|
|
144
|
+
datetime.fromisoformat rejects before Python 3.11, and this tool must
|
|
145
|
+
behave identically on every interpreter it runs under. ISO-8601 UTC
|
|
146
|
+
timestamps sort lexicographically in exactly chronological order, so the
|
|
147
|
+
string IS the key.
|
|
148
|
+
|
|
149
|
+
None is the answer for anything that is not a plausible timestamp, and
|
|
150
|
+
None routes the receipt to the undated section (rule 1) rather than to a
|
|
151
|
+
guessed position. The bar is deliberately low -- a full-precision parse
|
|
152
|
+
would reject valid variants and manufacture undated receipts out of dated
|
|
153
|
+
ones -- but it is not zero: a value must at least carry a date-shaped
|
|
154
|
+
prefix, or "not a timestamp" and "a timestamp we mis-sorted" become the
|
|
155
|
+
same outcome.
|
|
156
|
+
"""
|
|
157
|
+
v = proof.get("generated_at")
|
|
158
|
+
if not isinstance(v, str) or len(v) < 10:
|
|
159
|
+
return None
|
|
160
|
+
head = v[:10]
|
|
161
|
+
if head[4] != "-" or head[7] != "-":
|
|
162
|
+
return None
|
|
163
|
+
if not (head[:4].isdigit() and head[5:7].isdigit() and head[8:10].isdigit()):
|
|
164
|
+
return None
|
|
165
|
+
return v
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _cost_change(prev_usd, curr_usd):
|
|
169
|
+
"""(delta, why) for one step. delta is None exactly when UNKNOWN.
|
|
170
|
+
|
|
171
|
+
Rules 2 and 3 in one place. `is None` on BOTH sides: a measured $0.0000 is
|
|
172
|
+
a real observation on either end of the subtraction, and truthiness would
|
|
173
|
+
silently reclassify it as unmeasured -- the same erasure this file exists
|
|
174
|
+
to prevent, applied to the operand instead of the result.
|
|
175
|
+
"""
|
|
176
|
+
if prev_usd is None or curr_usd is None:
|
|
177
|
+
side = ("neither receipt" if prev_usd is None and curr_usd is None
|
|
178
|
+
else "the previous receipt" if prev_usd is None
|
|
179
|
+
else "this receipt")
|
|
180
|
+
return None, ("cost delta UNKNOWN -- %s recorded a measured cost, and "
|
|
181
|
+
"an absent measurement is not $0.00" % side)
|
|
182
|
+
return curr_usd - prev_usd, ""
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _gate_regressions(prev_gates, curr_gates):
|
|
186
|
+
"""Gates that PASSED in the previous receipt and FAIL in this one.
|
|
187
|
+
|
|
188
|
+
Only that direction, and only on gates present in BOTH receipts. A gate
|
|
189
|
+
absent from one side did not change state; it was not observed, and
|
|
190
|
+
reporting "gate X now fails" on the strength of its first-ever appearance
|
|
191
|
+
is an invented transition. Absent is not passing and absent is not failing.
|
|
192
|
+
"""
|
|
193
|
+
if not isinstance(prev_gates, dict) or not isinstance(curr_gates, dict):
|
|
194
|
+
return []
|
|
195
|
+
out = []
|
|
196
|
+
for name in sorted(set(prev_gates) & set(curr_gates)):
|
|
197
|
+
if _passed(prev_gates[name]) is True and _passed(curr_gates[name]) is False:
|
|
198
|
+
out.append(name)
|
|
199
|
+
return out
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _passed(value):
|
|
203
|
+
"""True / False / None for one gate entry. None means NOT OBSERVED.
|
|
204
|
+
|
|
205
|
+
Three states, because a gate whose result we cannot read has not passed
|
|
206
|
+
and has not failed. Folding the unreadable case into False would report a
|
|
207
|
+
fabricated regression the first time a receipt schema changes shape.
|
|
208
|
+
"""
|
|
209
|
+
if isinstance(value, bool):
|
|
210
|
+
return value
|
|
211
|
+
if isinstance(value, dict):
|
|
212
|
+
for key in ("passed", "ok", "pass"):
|
|
213
|
+
if isinstance(value.get(key), bool):
|
|
214
|
+
return value[key]
|
|
215
|
+
status = value.get("status")
|
|
216
|
+
if isinstance(status, str):
|
|
217
|
+
low = status.strip().lower()
|
|
218
|
+
if low in ("pass", "passed", "ok", "green"):
|
|
219
|
+
return True
|
|
220
|
+
if low in ("fail", "failed", "error", "red"):
|
|
221
|
+
return False
|
|
222
|
+
if isinstance(value, str):
|
|
223
|
+
low = value.strip().lower()
|
|
224
|
+
if low in ("pass", "passed", "ok", "green"):
|
|
225
|
+
return True
|
|
226
|
+
if low in ("fail", "failed", "error", "red"):
|
|
227
|
+
return False
|
|
228
|
+
return None
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _gates(proof):
|
|
232
|
+
"""The receipt's gate results, or {} when the receipt records none."""
|
|
233
|
+
for key in ("gates", "quality_gates"):
|
|
234
|
+
block = proof.get(key)
|
|
235
|
+
if isinstance(block, dict):
|
|
236
|
+
return block
|
|
237
|
+
if isinstance(block, list):
|
|
238
|
+
named = {}
|
|
239
|
+
for entry in block:
|
|
240
|
+
if isinstance(entry, dict) and isinstance(entry.get("name"), str):
|
|
241
|
+
named[entry["name"]] = entry
|
|
242
|
+
return named
|
|
243
|
+
return {}
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def timeline(rows):
|
|
247
|
+
"""Order `rows` oldest-first and compute what changed at each step.
|
|
248
|
+
|
|
249
|
+
Pure: no filesystem, no verification, no clock. Every rule above is
|
|
250
|
+
enforced here, so a defect in any of them is provable against a literal
|
|
251
|
+
list of dicts.
|
|
252
|
+
|
|
253
|
+
Each input row is {path, state, reason, when, cost_usd, gates}. `when` is
|
|
254
|
+
None for an undated receipt (rule 1).
|
|
255
|
+
|
|
256
|
+
Returns {dated, undated, comparable, note}. Undated rows keep their input
|
|
257
|
+
order, are never merged into `dated`, and carry no delta -- an adjacency
|
|
258
|
+
the data does not support is the same invention as a made-up sort position.
|
|
259
|
+
"""
|
|
260
|
+
dated = [r for r in rows if r.get("when") is not None]
|
|
261
|
+
undated = [r for r in rows if r.get("when") is None]
|
|
262
|
+
|
|
263
|
+
# Sort by timestamp, path second, so two receipts sharing a timestamp still
|
|
264
|
+
# land in a stable order rather than an arbitrary one.
|
|
265
|
+
dated = sorted(dated, key=lambda r: (r["when"], r["path"]))
|
|
266
|
+
|
|
267
|
+
steps = []
|
|
268
|
+
prev = None
|
|
269
|
+
for row in dated:
|
|
270
|
+
entry = dict(row)
|
|
271
|
+
if prev is None:
|
|
272
|
+
# Distinct from rule 4's message. "First in a longer chain" and
|
|
273
|
+
# "the only receipt there is" are different facts, and only the
|
|
274
|
+
# second one says the archive cannot show change at all.
|
|
275
|
+
entry["cost_delta_usd"] = None
|
|
276
|
+
entry["change"] = ("first receipt in the timeline -- there is no "
|
|
277
|
+
"earlier run to compare it against")
|
|
278
|
+
entry["gate_regressions"] = []
|
|
279
|
+
else:
|
|
280
|
+
delta, why = _cost_change(prev.get("cost_usd"), row.get("cost_usd"))
|
|
281
|
+
entry["cost_delta_usd"] = delta
|
|
282
|
+
regressions = _gate_regressions(prev.get("gates"), row.get("gates"))
|
|
283
|
+
entry["gate_regressions"] = regressions
|
|
284
|
+
entry["change"] = _change_line(delta, why, regressions)
|
|
285
|
+
steps.append(entry)
|
|
286
|
+
prev = row
|
|
287
|
+
|
|
288
|
+
undated_out = []
|
|
289
|
+
for row in undated:
|
|
290
|
+
entry = dict(row)
|
|
291
|
+
# Rule 1: counted, reported, and explicitly WITHOUT a delta. Not a
|
|
292
|
+
# zero delta, which would read as "nothing changed".
|
|
293
|
+
entry["cost_delta_usd"] = None
|
|
294
|
+
entry["gate_regressions"] = []
|
|
295
|
+
entry["change"] = ("UNDATED -- this receipt records no readable "
|
|
296
|
+
"generated_at, so it has no position in the order "
|
|
297
|
+
"and no previous receipt to compare against. It is "
|
|
298
|
+
"counted here, not dropped.")
|
|
299
|
+
undated_out.append(entry)
|
|
300
|
+
|
|
301
|
+
total = len(dated) + len(undated)
|
|
302
|
+
comparable = len(dated) >= 2
|
|
303
|
+
|
|
304
|
+
if total == 0:
|
|
305
|
+
note = ("NO RECEIPTS -- no proof.json found under this workspace, so "
|
|
306
|
+
"there was no history to lay out. Zero receipts is not a clean "
|
|
307
|
+
"archive; it is most often the wrong directory.")
|
|
308
|
+
elif not comparable:
|
|
309
|
+
# Rule 4, stated in words rather than left as an empty column.
|
|
310
|
+
note = ("%d receipt(s), %d of them dated: a timeline needs at least "
|
|
311
|
+
"TWO dated receipts to show a change, so NOTHING here is a "
|
|
312
|
+
"comparison. An empty change column would read as 'nothing "
|
|
313
|
+
"changed', which is a claim this archive cannot support."
|
|
314
|
+
% (total, len(dated)))
|
|
315
|
+
else:
|
|
316
|
+
note = "%d dated receipt(s) in order" % len(dated)
|
|
317
|
+
if undated:
|
|
318
|
+
note += (", plus %d UNDATED receipt(s) held out of the order "
|
|
319
|
+
"(counted, not dropped)" % len(undated))
|
|
320
|
+
|
|
321
|
+
return {
|
|
322
|
+
"dated": steps,
|
|
323
|
+
"undated": undated_out,
|
|
324
|
+
"comparable": comparable,
|
|
325
|
+
"note": note,
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
def _change_line(delta, why, regressions):
|
|
330
|
+
"""One sentence for the change column. Never blank on a comparable step."""
|
|
331
|
+
if delta is None:
|
|
332
|
+
parts = [why]
|
|
333
|
+
elif delta > 0:
|
|
334
|
+
parts = ["cost ROSE $%.4f" % delta]
|
|
335
|
+
elif delta < 0:
|
|
336
|
+
parts = ["cost FELL $%.4f" % -delta]
|
|
337
|
+
else:
|
|
338
|
+
# Rule 3. A measured zero is an observation, and it is the cell an
|
|
339
|
+
# operator most often wants; `if delta:` would blank exactly this one.
|
|
340
|
+
parts = ["cost UNCHANGED (measured $0.0000 delta)"]
|
|
341
|
+
if regressions:
|
|
342
|
+
parts.append("GATE REGRESSION: %s passed in the previous receipt and "
|
|
343
|
+
"now FAILS" % ", ".join(regressions))
|
|
344
|
+
return "; ".join(parts)
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def find_receipts(workspace):
|
|
348
|
+
"""Every proof.json under the workspace, sorted for a stable report.
|
|
349
|
+
|
|
350
|
+
ponytail: same rglob as receipt-bundle.find_receipts, so a receipt archived
|
|
351
|
+
outside .loki/proofs/ is still counted. Not imported, because that one
|
|
352
|
+
assumes the directory exists and this tool must tell a missing workspace
|
|
353
|
+
(exit 66) apart from an empty one (exit 3).
|
|
354
|
+
"""
|
|
355
|
+
root = pathlib.Path(workspace)
|
|
356
|
+
if not root.is_dir():
|
|
357
|
+
return []
|
|
358
|
+
return sorted(p for p in root.rglob("proof.json") if p.is_file())
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def scan(workspace, repo_dir="."):
|
|
362
|
+
"""Read every receipt under `workspace` and lay it out in time order."""
|
|
363
|
+
rows = []
|
|
364
|
+
malformed = []
|
|
365
|
+
|
|
366
|
+
for path in find_receipts(workspace):
|
|
367
|
+
try:
|
|
368
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
369
|
+
proof = json.load(f)
|
|
370
|
+
if not isinstance(proof, dict):
|
|
371
|
+
raise ValueError("receipt is not a JSON object")
|
|
372
|
+
except Exception as exc:
|
|
373
|
+
# Counted and NAMED, never dropped -- and undated by construction,
|
|
374
|
+
# so it lands in the undated section rather than at a guessed
|
|
375
|
+
# point in the order.
|
|
376
|
+
malformed.append({"path": str(path), "reason": str(exc)})
|
|
377
|
+
rows.append({
|
|
378
|
+
"path": str(path), "state": MALFORMED, "reason": str(exc),
|
|
379
|
+
"when": None, "cost_usd": None, "gates": {},
|
|
380
|
+
})
|
|
381
|
+
continue
|
|
382
|
+
|
|
383
|
+
state, reason = receipt_state(path, repo_dir)
|
|
384
|
+
rows.append({
|
|
385
|
+
"path": str(path), "state": state, "reason": reason,
|
|
386
|
+
"when": _when(proof), "cost_usd": _usd(proof),
|
|
387
|
+
"gates": _gates(proof),
|
|
388
|
+
})
|
|
389
|
+
|
|
390
|
+
tl = timeline(rows)
|
|
391
|
+
return {
|
|
392
|
+
"report": "loki-receipt-timeline/v1",
|
|
393
|
+
"workspace": os.path.abspath(str(workspace)),
|
|
394
|
+
"checked_from": os.path.abspath(repo_dir),
|
|
395
|
+
"dated": tl["dated"],
|
|
396
|
+
"undated": tl["undated"],
|
|
397
|
+
"receipt_count": len(rows),
|
|
398
|
+
"dated_count": len(tl["dated"]),
|
|
399
|
+
"undated_count": len(tl["undated"]),
|
|
400
|
+
"comparable": tl["comparable"],
|
|
401
|
+
"malformed": malformed,
|
|
402
|
+
"malformed_count": len(malformed),
|
|
403
|
+
"note": tl["note"],
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
def _row_line(row):
|
|
408
|
+
# `is None`, not truthiness, per rule 3 -- consistently with every other
|
|
409
|
+
# absence test in this file, so a reader never has to work out which
|
|
410
|
+
# falsy values this one line happens to be safe against.
|
|
411
|
+
when = "UNDATED" if row["when"] is None else row["when"]
|
|
412
|
+
# "-" for an unmeasured cost, never "$0.0000". The table is the surface an
|
|
413
|
+
# operator eyeballs, and it must not be the one place the archive looks
|
|
414
|
+
# free.
|
|
415
|
+
usd = "-" if row["cost_usd"] is None else "$%.4f" % row["cost_usd"]
|
|
416
|
+
return "%-26s %-13s %-11s %s" % (when, row["state"], usd, row["path"])
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
class _Parser(argparse.ArgumentParser):
|
|
420
|
+
"""argparse exits 2 on a usage error; here 2 means "could not check".
|
|
421
|
+
|
|
422
|
+
A mistyped flag would otherwise be indistinguishable from a blind gate --
|
|
423
|
+
the operator sees the code that means "your instrumentation is broken" and
|
|
424
|
+
goes looking for broken instrumentation. 64 is the usage error.
|
|
425
|
+
|
|
426
|
+
error() only. --help routes through exit(), not error(), and overriding
|
|
427
|
+
exit() would break the exit-0 contract that test_tool_exit_contract.py
|
|
428
|
+
asserts for every tool's --help.
|
|
429
|
+
"""
|
|
430
|
+
|
|
431
|
+
def error(self, message):
|
|
432
|
+
self.print_usage(sys.stderr)
|
|
433
|
+
sys.stderr.write("%s: error: %s\n" % (self.prog, message))
|
|
434
|
+
raise SystemExit(64)
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
def main(argv=None):
|
|
438
|
+
ap = _Parser(
|
|
439
|
+
description="Show what happened across an archive of receipts, in "
|
|
440
|
+
"time order, and what changed at each step.")
|
|
441
|
+
ap.add_argument("workspace", nargs="?", default=".",
|
|
442
|
+
help="workspace holding the receipts (default: .)")
|
|
443
|
+
ap.add_argument("--repo-dir", default=".",
|
|
444
|
+
help="repository the receipts are verified against "
|
|
445
|
+
"(default: .)")
|
|
446
|
+
ap.add_argument("--json", action="store_true",
|
|
447
|
+
help="emit the full report as JSON")
|
|
448
|
+
args = ap.parse_args(argv)
|
|
449
|
+
|
|
450
|
+
if not os.path.isdir(args.workspace):
|
|
451
|
+
# 66, not 3. "You pointed me at nothing" and "this archive is empty"
|
|
452
|
+
# are different facts, and only one of them is about the archive.
|
|
453
|
+
sys.stderr.write(
|
|
454
|
+
"receipt-timeline: workspace does not exist: %s\n" % args.workspace)
|
|
455
|
+
return 66
|
|
456
|
+
|
|
457
|
+
report = scan(args.workspace, args.repo_dir)
|
|
458
|
+
|
|
459
|
+
if args.json:
|
|
460
|
+
print(json.dumps(report, indent=2))
|
|
461
|
+
else:
|
|
462
|
+
for row in report["dated"]:
|
|
463
|
+
print(_row_line(row))
|
|
464
|
+
print(" %s" % row["change"])
|
|
465
|
+
if report["undated"]:
|
|
466
|
+
print("")
|
|
467
|
+
print("UNDATED -- held out of the order, counted, not dropped:")
|
|
468
|
+
for row in report["undated"]:
|
|
469
|
+
print(_row_line(row))
|
|
470
|
+
print(" %s" % row["change"])
|
|
471
|
+
print("")
|
|
472
|
+
print(report["note"])
|
|
473
|
+
|
|
474
|
+
return 0 if report["receipt_count"] else 3
|
|
475
|
+
|
|
476
|
+
|
|
477
|
+
if __name__ == "__main__":
|
|
478
|
+
sys.exit(main())
|