loki-mode 9.8.1 → 9.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -14
- package/SKILL.md +3 -2
- package/VERSION +1 -1
- package/autonomy/loki +122 -1
- package/autonomy/run.sh +49 -2
- package/dashboard/__init__.py +1 -1
- package/dashboard/api_evidence.py +411 -0
- package/dashboard/api_operator.py +283 -0
- package/dashboard/api_phases.py +262 -0
- package/dashboard/api_releases.py +242 -0
- package/dashboard/api_runs.py +477 -0
- package/dashboard/api_tests.py +444 -0
- package/dashboard/api_v2.py +47 -1
- package/dashboard/server.py +54 -0
- package/dashboard/static/index.html +246 -135
- package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
- package/docs/CAPABILITY-BACKLOG.md +53 -0
- package/docs/COMPARISON.md +2 -2
- package/docs/COMPETITIVE-ANALYSIS.md +1 -1
- package/docs/COMPETITIVE-SCORECARD.md +422 -0
- package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
- package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
- package/docs/DEMOS.md +21 -23
- package/docs/HANDOFF-2026-08-03.md +439 -0
- package/docs/INSTALLATION.md +17 -10
- package/docs/OUTCOME-FRONTIER.md +536 -0
- package/docs/PROMPT-ABLATION-RESULT.md +97 -0
- package/docs/TOOLS.md +800 -0
- package/docs/alternative-installations.md +2 -3
- package/docs/audit-logging.md +44 -35
- package/docs/authentication.md +13 -2
- package/docs/authorization.md +87 -81
- package/docs/git-workflow.md +6 -3
- package/docs/metrics.md +15 -16
- package/docs/network-security.md +16 -13
- package/docs/openclaw-integration.md +36 -556
- package/docs/show-hn-post.md +2 -2
- package/docs/siem-integration.md +39 -36
- package/loki-ts/dist/loki.js +18 -18
- package/mcp/__init__.py +1 -1
- package/package.json +1 -1
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
- package/references/confidence-routing.md +18 -1
- package/references/invariant-checks.md +13 -8
- package/references/magic-rarv-integration.md +0 -1
- package/references/multi-provider.md +27 -5
- package/skills/healing.md +4 -2
- package/tools/audit-docs.py +488 -0
- package/tools/baseline-pin.py +19 -1
- package/tools/calibration-audit.py +523 -0
- package/tools/ci-gate.py +19 -1
- package/tools/cost-forecast.py +344 -0
- package/tools/cost-guard.py +19 -1
- package/tools/cost-history.py +19 -1
- package/tools/cost-per-outcome.py +394 -0
- package/tools/estimate-run.py +19 -1
- package/tools/evidence-freshness.py +307 -0
- package/tools/gate-init.py +19 -1
- package/tools/gate-report.py +19 -1
- package/tools/gate-simulate.py +570 -0
- package/tools/gate-trend.py +354 -0
- package/tools/model-advisor.py +52 -1
- package/tools/policy-load.py +19 -1
- package/tools/prompt-cost.py +363 -0
- package/tools/prompt-diff.py +448 -0
- package/tools/prompt-lint.py +448 -0
- package/tools/receipt-bundle.py +72 -2
- package/tools/receipt-diff.py +19 -1
- package/tools/receipt-find.py +19 -1
- package/tools/receipt-stats.py +380 -0
- package/tools/receipt-timeline.py +478 -0
- package/tools/receipt-verify-batch.py +291 -0
- package/tools/run-replay.py +19 -1
- package/tools/signing-status.py +19 -1
- package/tools/token-guard.py +19 -1
- package/tools/token-tax.py +375 -0
- package/tools/tool-index.py +19 -1
- package/tools/verification-tax.py +277 -0
- package/tools/verify-chain.py +361 -0
|
@@ -0,0 +1,448 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Show exactly what LOKI_SIMPLE=1 deletes from the prompt, before anyone
|
|
3
|
+
trusts an ablation result built on it.
|
|
4
|
+
|
|
5
|
+
The flag strips the coaching half of the system prompt (measured -78%, roughly
|
|
6
|
+
1562 tokens per iteration). A percentage is not a reason to believe an arm is
|
|
7
|
+
sound. WHICH instructions vanished is, and nothing printed that until now: the
|
|
8
|
+
ablation tests assert that named anchors are absent, which proves the strip
|
|
9
|
+
happened, not that what it took was safe to take.
|
|
10
|
+
|
|
11
|
+
THE ONE ASSERTION THIS FILE EXISTS FOR, and the reason it can exit 1:
|
|
12
|
+
|
|
13
|
+
the dynamic tail must be IDENTICAL between the two arms.
|
|
14
|
+
|
|
15
|
+
Everything above [CACHE_BREAKPOINT] is the cache-stable prefix -- coaching,
|
|
16
|
+
which is how to work, and which a frontier model does natively. Everything
|
|
17
|
+
below is per-iteration STATE: which gate failed, what self-heal found, what
|
|
18
|
+
iteration this is. The model cannot derive state. An arm that drops coaching is
|
|
19
|
+
an experiment about prompt bloat; an arm that drops state is a run going blind
|
|
20
|
+
to its own history, and the two are indistinguishable from a byte count alone.
|
|
21
|
+
|
|
22
|
+
WHY THE STRIP IS SIMULATED DOCUMENT-WIDE. The removal predicate is applied to
|
|
23
|
+
EVERY line of the fixture, then both arms are split at the marker and the tails
|
|
24
|
+
compared. Applying it only above the marker and copying the tail verbatim would
|
|
25
|
+
compare the tail to itself, and the most important check in this file could
|
|
26
|
+
never fail. Simulating the whole document means a predicate that drifts into
|
|
27
|
+
matching tail content produces a real FAIL, which is the point.
|
|
28
|
+
|
|
29
|
+
The anchors are derived from the nine values pushed inside `if (!simple)` at
|
|
30
|
+
loki-ts/src/runner/build_prompt.ts:1607-1617. Two prefix lines pushed just
|
|
31
|
+
AFTER that block are deliberately absent: goal-sharpening and
|
|
32
|
+
CODEBASE_ANALYSIS_MODE survive the flag, and listing them here would
|
|
33
|
+
over-report the deletion. A stale anchor under-reports it, so an anchor that
|
|
34
|
+
matches nothing anywhere in the corpus is reported loudly rather than ignored.
|
|
35
|
+
|
|
36
|
+
Honesty rules, same as tools/receipt-diff.py: an unmeasured value reads
|
|
37
|
+
UNKNOWN, never 0; a fixture that cannot be split is reported as NOT CHECKED
|
|
38
|
+
rather than silently skipped; and tokens are labelled as the bytes/4 estimate
|
|
39
|
+
they are, because printing a bare integer would dress a derivation up as a
|
|
40
|
+
measurement.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
import argparse
|
|
44
|
+
import json
|
|
45
|
+
import os
|
|
46
|
+
import sys
|
|
47
|
+
|
|
48
|
+
UNKNOWN = "UNKNOWN"
|
|
49
|
+
MARKER = "[CACHE_BREAKPOINT]"
|
|
50
|
+
|
|
51
|
+
_HERE = os.path.dirname(os.path.abspath(__file__))
|
|
52
|
+
DEFAULT_CORPUS = os.path.join(
|
|
53
|
+
os.path.dirname(_HERE), "loki-ts", "tests", "fixtures", "build_prompt")
|
|
54
|
+
|
|
55
|
+
# The nine coaching values gated by `if (!simple)` in build_prompt.ts. Each is a
|
|
56
|
+
# whole-line, start-anchored prefix: a bare substring would match this file's
|
|
57
|
+
# own prose and any tail line quoting an instruction back.
|
|
58
|
+
STRIP_ANCHORS = (
|
|
59
|
+
"RALPH WIGGUM MODE ACTIVE.", # rarvText
|
|
60
|
+
"SDLC_PHASES_ENABLED: [", # sdlcText
|
|
61
|
+
"CRITICAL AUTONOMY RULES: ", # autonomyText
|
|
62
|
+
"MEMORY SYSTEM: ", # MEMORY_INSTRUCTION
|
|
63
|
+
"USAGE_DOC_REQUIRED: ", # USAGE_DOC_INSTRUCTION
|
|
64
|
+
"DOC_SCOPE: ", # docScope
|
|
65
|
+
"RUN_CONTRACT: ", # COMPOSE_INSTRUCTION
|
|
66
|
+
"LSP_GROUNDING: ", # LSP_GROUNDING_INSTRUCTION
|
|
67
|
+
"Project conventions: read AGENTS.md", # AGENTS_MD_INSTRUCTION
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class UsageError(Exception):
|
|
72
|
+
"""Raised for a malformed invocation, so main() can exit 64."""
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _num(v):
|
|
76
|
+
"""A number as itself, anything else (None, "", bool) as None.
|
|
77
|
+
|
|
78
|
+
Lifted verbatim from tools/receipt-diff.py rather than re-derived: an
|
|
79
|
+
absent value must never arrive downstream as a real 0.
|
|
80
|
+
"""
|
|
81
|
+
if isinstance(v, bool) or not isinstance(v, (int, float)):
|
|
82
|
+
return None
|
|
83
|
+
return v
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def is_coaching(line):
|
|
87
|
+
"""True when LOKI_SIMPLE=1 would delete this line.
|
|
88
|
+
|
|
89
|
+
Start-anchored on purpose. Substring matching would let a tail line that
|
|
90
|
+
mentions an instruction be scored as coaching, which is exactly the
|
|
91
|
+
misclassification the tail check is supposed to catch.
|
|
92
|
+
"""
|
|
93
|
+
return any(line.startswith(a) for a in STRIP_ANCHORS)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def simulate(text):
|
|
97
|
+
"""The simple arm: apply the strip to the WHOLE document, not the prefix.
|
|
98
|
+
|
|
99
|
+
Returns (kept_lines, removed_lines).
|
|
100
|
+
"""
|
|
101
|
+
kept, removed = [], []
|
|
102
|
+
for line in text.split("\n"):
|
|
103
|
+
(removed if is_coaching(line) else kept).append(line)
|
|
104
|
+
return kept, removed
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def split_at_marker(text):
|
|
108
|
+
"""(prefix, tail) around the literal marker, or None when absent."""
|
|
109
|
+
if MARKER not in text:
|
|
110
|
+
return None
|
|
111
|
+
prefix, tail = text.split(MARKER, 1)
|
|
112
|
+
return prefix, tail
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def est_tokens(n_bytes):
|
|
116
|
+
"""The repo's own bytes/4 estimator, never presented as a measurement."""
|
|
117
|
+
b = _num(n_bytes)
|
|
118
|
+
return None if b is None else int(round(b / 4.0))
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def is_degraded(fixture_dir):
|
|
122
|
+
"""True when this fixture takes the degraded-provider path.
|
|
123
|
+
|
|
124
|
+
LOAD-BEARING, and invisible to a text-only reading of expected.txt. On
|
|
125
|
+
PROVIDER_DEGRADED=true, buildPrompt returns buildStaticFirstDegraded at
|
|
126
|
+
build_prompt.ts:1574-1577 -- BEFORE `const simple` is even read at 1606 and
|
|
127
|
+
before the `if (!simple)` gate at 1607. The flag is inert there: both arms
|
|
128
|
+
emit identical bytes (verified against the real builder, 4202 == 4202,
|
|
129
|
+
against 7900 -> 1653 on the normal path).
|
|
130
|
+
|
|
131
|
+
Without this, the tool matches coaching-shaped anchors in a degraded
|
|
132
|
+
prompt and reports a saving the flag cannot produce -- a fabricated
|
|
133
|
+
number attached to the one thing a human reads this tool to learn.
|
|
134
|
+
"""
|
|
135
|
+
env_path = os.path.join(fixture_dir, "env.txt") if fixture_dir else None
|
|
136
|
+
if not env_path or not os.path.isfile(env_path):
|
|
137
|
+
return False
|
|
138
|
+
with open(env_path, "r", encoding="utf-8") as fh:
|
|
139
|
+
return any(line.strip() == "PROVIDER_DEGRADED=true" for line in fh)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def analyze(name, text, fixture_dir=None):
|
|
143
|
+
"""Diff one fixture's two arms. Never raises on shape; reports instead."""
|
|
144
|
+
full_bytes = len(text.encode("utf-8"))
|
|
145
|
+
|
|
146
|
+
if is_degraded(fixture_dir):
|
|
147
|
+
# A MEASURED zero, not an unmeasured one: the arms were compared and
|
|
148
|
+
# found equal. Reported, never folded into NOT CHECKED -- this fixture
|
|
149
|
+
# is perfectly checkable and the answer is "the flag does nothing".
|
|
150
|
+
split = split_at_marker(text)
|
|
151
|
+
tail_lines = ([ln for ln in split[1].split("\n") if ln.strip()]
|
|
152
|
+
if split else [])
|
|
153
|
+
return {
|
|
154
|
+
"fixture": name,
|
|
155
|
+
"checked": split is not None,
|
|
156
|
+
"degraded": True,
|
|
157
|
+
"why": "degraded provider path returns before the LOKI_SIMPLE "
|
|
158
|
+
"gate, so the flag has no effect here",
|
|
159
|
+
"full_bytes": full_bytes,
|
|
160
|
+
"simple_bytes": full_bytes,
|
|
161
|
+
"removed": [],
|
|
162
|
+
"survived": [ln for ln in (split[0] if split else text).split("\n")
|
|
163
|
+
if ln.strip()],
|
|
164
|
+
"tail_lines": tail_lines,
|
|
165
|
+
"tail_identical": True if split is not None else None,
|
|
166
|
+
"tail_diff": None,
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
split_full = split_at_marker(text)
|
|
170
|
+
if split_full is None:
|
|
171
|
+
return {
|
|
172
|
+
"fixture": name,
|
|
173
|
+
"checked": False,
|
|
174
|
+
"degraded": False,
|
|
175
|
+
"why": "no %s marker, so prefix and tail cannot be separated "
|
|
176
|
+
"(legacy flat prompt ordering)" % MARKER,
|
|
177
|
+
"full_bytes": full_bytes,
|
|
178
|
+
"simple_bytes": None,
|
|
179
|
+
"removed": [],
|
|
180
|
+
"survived": [],
|
|
181
|
+
"tail_identical": None,
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
kept, removed = simulate(text)
|
|
185
|
+
simple_text = "\n".join(kept)
|
|
186
|
+
split_simple = split_at_marker(simple_text)
|
|
187
|
+
|
|
188
|
+
# The strip eating the marker itself would be the most severe form of the
|
|
189
|
+
# failure this file guards, so it is a FAIL and not a "cannot check".
|
|
190
|
+
if split_simple is None:
|
|
191
|
+
return {
|
|
192
|
+
"fixture": name,
|
|
193
|
+
"checked": True,
|
|
194
|
+
"degraded": False,
|
|
195
|
+
"tail_identical": False,
|
|
196
|
+
"why": "the simulated strip removed the %s marker itself" % MARKER,
|
|
197
|
+
"full_bytes": full_bytes,
|
|
198
|
+
"simple_bytes": len(simple_text.encode("utf-8")),
|
|
199
|
+
"removed": removed,
|
|
200
|
+
"survived": [],
|
|
201
|
+
"tail_diff": None,
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
tail_full, tail_simple = split_full[1], split_simple[1]
|
|
205
|
+
identical = tail_full == tail_simple
|
|
206
|
+
|
|
207
|
+
prefix_simple = split_simple[0]
|
|
208
|
+
survived = [ln for ln in prefix_simple.split("\n") if ln.strip()]
|
|
209
|
+
|
|
210
|
+
out = {
|
|
211
|
+
"fixture": name,
|
|
212
|
+
"checked": True,
|
|
213
|
+
"degraded": False,
|
|
214
|
+
"tail_identical": identical,
|
|
215
|
+
"full_bytes": full_bytes,
|
|
216
|
+
"simple_bytes": len(simple_text.encode("utf-8")),
|
|
217
|
+
"removed": removed,
|
|
218
|
+
"survived": survived,
|
|
219
|
+
"tail_lines": [ln for ln in tail_full.split("\n") if ln.strip()],
|
|
220
|
+
"tail_diff": None,
|
|
221
|
+
}
|
|
222
|
+
if not identical:
|
|
223
|
+
out["tail_diff"] = _first_divergence(tail_full, tail_simple)
|
|
224
|
+
return out
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _first_divergence(a, b):
|
|
228
|
+
"""The first differing tail line, so a FAIL names what went missing."""
|
|
229
|
+
la, lb = a.split("\n"), b.split("\n")
|
|
230
|
+
for i in range(max(len(la), len(lb))):
|
|
231
|
+
x = la[i] if i < len(la) else None
|
|
232
|
+
y = lb[i] if i < len(lb) else None
|
|
233
|
+
if x != y:
|
|
234
|
+
return {"line": i + 1, "full": x, "simple": y}
|
|
235
|
+
return None
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def scan(corpus):
|
|
239
|
+
"""Every fixture under the corpus root, in stable lexicographic order."""
|
|
240
|
+
if not os.path.isdir(corpus):
|
|
241
|
+
return None
|
|
242
|
+
results = []
|
|
243
|
+
for entry in sorted(os.listdir(corpus)):
|
|
244
|
+
fixture_dir = os.path.join(corpus, entry)
|
|
245
|
+
path = os.path.join(fixture_dir, "expected.txt")
|
|
246
|
+
if os.path.isfile(path):
|
|
247
|
+
with open(path, "r", encoding="utf-8") as fh:
|
|
248
|
+
results.append(analyze(entry, fh.read(), fixture_dir))
|
|
249
|
+
return results
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def unused_anchors(results):
|
|
253
|
+
"""Anchors matching nothing anywhere: a silent under-report of the strip."""
|
|
254
|
+
seen = set()
|
|
255
|
+
for r in results:
|
|
256
|
+
for line in r["removed"]:
|
|
257
|
+
for a in STRIP_ANCHORS:
|
|
258
|
+
if line.startswith(a):
|
|
259
|
+
seen.add(a)
|
|
260
|
+
return [a for a in STRIP_ANCHORS if a not in seen]
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _delta_line(full_b, simple_b):
|
|
264
|
+
fb, sb = _num(full_b), _num(simple_b)
|
|
265
|
+
if fb is None or sb is None:
|
|
266
|
+
return "bytes %s -> %s delta %s" % (
|
|
267
|
+
fb if fb is not None else UNKNOWN,
|
|
268
|
+
sb if sb is not None else UNKNOWN, UNKNOWN)
|
|
269
|
+
d = sb - fb
|
|
270
|
+
pct = (100.0 * d / fb) if fb else None
|
|
271
|
+
tok = est_tokens(-d)
|
|
272
|
+
return ("bytes %d -> %d delta %+d (%s), ~%s tokens saved "
|
|
273
|
+
"(est., bytes/4)" % (
|
|
274
|
+
fb, sb, d,
|
|
275
|
+
"%+.1f%%" % pct if pct is not None else UNKNOWN,
|
|
276
|
+
tok if tok is not None else UNKNOWN))
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def _abbrev(line, width=96):
|
|
280
|
+
line = line.rstrip()
|
|
281
|
+
return line if len(line) <= width else line[:width - 3] + "..."
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def render(results, corpus, verbose=False):
|
|
285
|
+
lines = ["LOKI_SIMPLE=1 prompt ablation diff",
|
|
286
|
+
" corpus: %s" % corpus,
|
|
287
|
+
" arms: full (default) vs simple (LOKI_SIMPLE=1)", ""]
|
|
288
|
+
|
|
289
|
+
checked = [r for r in results if r["checked"]]
|
|
290
|
+
failed = [r for r in checked if r["tail_identical"] is False]
|
|
291
|
+
skipped = [r for r in results if not r["checked"]]
|
|
292
|
+
|
|
293
|
+
for r in results:
|
|
294
|
+
if not r["checked"]:
|
|
295
|
+
continue
|
|
296
|
+
# A degraded fixture has nothing removed BECAUSE the flag is inert
|
|
297
|
+
# there, which is a finding. Hiding it would leave a reader believing
|
|
298
|
+
# the corpus is uniform.
|
|
299
|
+
if (not verbose and not r["removed"] and r["tail_identical"]
|
|
300
|
+
and not r.get("degraded")):
|
|
301
|
+
continue
|
|
302
|
+
lines.append(" %s" % r["fixture"])
|
|
303
|
+
lines.append(" %s" % _delta_line(r["full_bytes"], r["simple_bytes"]))
|
|
304
|
+
if r["removed"]:
|
|
305
|
+
lines.append(" REMOVED under the flag (coaching, %d lines):"
|
|
306
|
+
% len(r["removed"]))
|
|
307
|
+
for line in r["removed"]:
|
|
308
|
+
lines.append(" - %s" % _abbrev(line))
|
|
309
|
+
elif r.get("degraded"):
|
|
310
|
+
lines.append(" REMOVED under the flag: NOTHING -- %s"
|
|
311
|
+
% r.get("why"))
|
|
312
|
+
else:
|
|
313
|
+
lines.append(" REMOVED under the flag: nothing (this prompt "
|
|
314
|
+
"carries no coaching to strip)")
|
|
315
|
+
lines.append(" SURVIVES in the prefix (%d lines):"
|
|
316
|
+
% len(r["survived"]))
|
|
317
|
+
for line in r["survived"]:
|
|
318
|
+
lines.append(" + %s" % _abbrev(line))
|
|
319
|
+
tail_n = len(r.get("tail_lines") or [])
|
|
320
|
+
if r["tail_identical"]:
|
|
321
|
+
lines.append(" SURVIVES in the dynamic tail: all %d lines, "
|
|
322
|
+
"byte-identical between arms" % tail_n)
|
|
323
|
+
else:
|
|
324
|
+
lines.append(" FAIL the dynamic tail DIFFERS between arms: %s"
|
|
325
|
+
% (r.get("why") or "state was deleted, not coaching"))
|
|
326
|
+
d = r.get("tail_diff")
|
|
327
|
+
if d:
|
|
328
|
+
lines.append(" tail line %d" % d["line"])
|
|
329
|
+
lines.append(" full: %s" % _abbrev(str(d["full"])))
|
|
330
|
+
lines.append(" simple: %s" % _abbrev(str(d["simple"])))
|
|
331
|
+
lines.append("")
|
|
332
|
+
|
|
333
|
+
if skipped:
|
|
334
|
+
lines.append(" NOT CHECKED: %d fixture(s) (reported, not skipped):"
|
|
335
|
+
% len(skipped))
|
|
336
|
+
for r in skipped:
|
|
337
|
+
lines.append(" %s: %s" % (r["fixture"], r["why"]))
|
|
338
|
+
lines.append("")
|
|
339
|
+
|
|
340
|
+
stale = unused_anchors(checked)
|
|
341
|
+
if stale:
|
|
342
|
+
lines.append(" WARNING: %d strip anchor(s) matched nothing in the "
|
|
343
|
+
"whole corpus, so this tool may be UNDER-reporting what "
|
|
344
|
+
"the flag deletes:" % len(stale))
|
|
345
|
+
for a in stale:
|
|
346
|
+
lines.append(" %s" % a)
|
|
347
|
+
lines.append("")
|
|
348
|
+
|
|
349
|
+
degraded = [r for r in results if r.get("degraded")]
|
|
350
|
+
if degraded:
|
|
351
|
+
lines.append(" FLAG INERT on %d degraded-provider fixture(s) "
|
|
352
|
+
"(measured 0-byte delta, not an unmeasured one): %s"
|
|
353
|
+
% (len(degraded), ", ".join(r["fixture"]
|
|
354
|
+
for r in degraded)))
|
|
355
|
+
lines.append("")
|
|
356
|
+
|
|
357
|
+
lines.append(" %d fixture(s) scanned, %d checked, %d not checked"
|
|
358
|
+
% (len(results), len(checked), len(skipped)))
|
|
359
|
+
if failed:
|
|
360
|
+
for r in failed:
|
|
361
|
+
lines.append(" FAIL %s: dynamic tail is NOT identical between "
|
|
362
|
+
"arms" % r["fixture"])
|
|
363
|
+
lines.append("VERDICT: FAIL -- %d fixture(s) would lose STATE, not "
|
|
364
|
+
"coaching. That is not an ablation, it is the run going "
|
|
365
|
+
"blind to its own history." % len(failed))
|
|
366
|
+
else:
|
|
367
|
+
lines.append("VERDICT: PASS -- the dynamic tail is byte-identical "
|
|
368
|
+
"between arms in all %d checked fixture(s); only prefix "
|
|
369
|
+
"coaching is removed." % len(checked))
|
|
370
|
+
return "\n".join(lines)
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def main(argv=None):
|
|
374
|
+
ap = argparse.ArgumentParser(
|
|
375
|
+
description="Show what LOKI_SIMPLE=1 removes from the prompt, and "
|
|
376
|
+
"assert the dynamic tail is identical between arms.")
|
|
377
|
+
|
|
378
|
+
def _usage_error(message):
|
|
379
|
+
raise UsageError(message)
|
|
380
|
+
|
|
381
|
+
# argparse exits 2 for a usage error, which in this tool line means
|
|
382
|
+
# "could NOT check" -- a materially different claim from "you typed it
|
|
383
|
+
# wrong". Reroute to 64. --help is unaffected: it goes through
|
|
384
|
+
# parser.exit(0), not error().
|
|
385
|
+
ap.error = _usage_error
|
|
386
|
+
|
|
387
|
+
ap.add_argument("corpus", nargs="?", default=DEFAULT_CORPUS,
|
|
388
|
+
help="build_prompt fixture corpus root "
|
|
389
|
+
"(default: %s)" % DEFAULT_CORPUS)
|
|
390
|
+
ap.add_argument("--json", action="store_true", dest="as_json",
|
|
391
|
+
help="emit the diff as JSON")
|
|
392
|
+
ap.add_argument("--verbose", action="store_true",
|
|
393
|
+
help="include fixtures with nothing removed")
|
|
394
|
+
|
|
395
|
+
try:
|
|
396
|
+
args = ap.parse_args(argv)
|
|
397
|
+
except UsageError as exc:
|
|
398
|
+
sys.stderr.write("usage error: %s\n" % exc)
|
|
399
|
+
return 64
|
|
400
|
+
|
|
401
|
+
if not os.path.isdir(args.corpus):
|
|
402
|
+
payload = {"checked": False,
|
|
403
|
+
"reason": "fixture corpus not found: %s" % args.corpus}
|
|
404
|
+
print(json.dumps(payload, indent=2) if args.as_json
|
|
405
|
+
else "INPUT MISSING: fixture corpus not found: %s" % args.corpus)
|
|
406
|
+
return 66
|
|
407
|
+
|
|
408
|
+
results = scan(args.corpus)
|
|
409
|
+
|
|
410
|
+
# An empty diff must never read as "no changes"; there was nothing to read.
|
|
411
|
+
if not results:
|
|
412
|
+
payload = {"checked": False, "fixtures": 0,
|
|
413
|
+
"reason": "no fixture-*/expected.txt under %s" % args.corpus}
|
|
414
|
+
print(json.dumps(payload, indent=2) if args.as_json
|
|
415
|
+
else "NOTHING TO COMPARE: no fixture-*/expected.txt under %s"
|
|
416
|
+
% args.corpus)
|
|
417
|
+
return 3
|
|
418
|
+
|
|
419
|
+
checked = [r for r in results if r["checked"]]
|
|
420
|
+
if not checked:
|
|
421
|
+
payload = {"checked": False, "fixtures": len(results),
|
|
422
|
+
"reason": "no fixture carries the %s marker, so no arm "
|
|
423
|
+
"could be split" % MARKER}
|
|
424
|
+
print(json.dumps(payload, indent=2) if args.as_json
|
|
425
|
+
else "CANNOT CHECK: no fixture carries the %s marker, so no "
|
|
426
|
+
"prefix/tail split was possible" % MARKER)
|
|
427
|
+
return 2
|
|
428
|
+
|
|
429
|
+
failed = [r for r in checked if r["tail_identical"] is False]
|
|
430
|
+
|
|
431
|
+
if args.as_json:
|
|
432
|
+
print(json.dumps({
|
|
433
|
+
"corpus": args.corpus,
|
|
434
|
+
"fixtures": len(results),
|
|
435
|
+
"checked": len(checked),
|
|
436
|
+
"not_checked": len(results) - len(checked),
|
|
437
|
+
"tail_failures": [r["fixture"] for r in failed],
|
|
438
|
+
"stale_anchors": unused_anchors(checked),
|
|
439
|
+
"results": results,
|
|
440
|
+
}, indent=2))
|
|
441
|
+
else:
|
|
442
|
+
print(render(results, args.corpus, verbose=args.verbose))
|
|
443
|
+
|
|
444
|
+
return 1 if failed else 0
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
if __name__ == "__main__":
|
|
448
|
+
sys.exit(main())
|