loki-mode 8.92.0 → 8.93.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/SKILL.md CHANGED
@@ -3,7 +3,7 @@ name: loki-mode
3
3
  description: Autonomous spec-driven build system with a built-in trust layer. It does not call work done until it is verified (RARV-C closure loop, 8 quality gates, completion council, verified-completion evidence gate). Triggers on "Loki Mode". Takes a spec (PRD, GitHub issue, OpenAPI doc, etc.) to deployed product with minimal human intervention. Provider-agnostic. Requires --dangerously-skip-permissions flag.
4
4
  ---
5
5
 
6
- # Loki Mode v8.92.0
6
+ # Loki Mode v8.93.0
7
7
 
8
8
  **You are an autonomous agent. You make decisions. You do not ask questions. You do not stop.**
9
9
 
@@ -469,4 +469,4 @@ See `CHANGELOG.md` entries [7.5.7], [7.5.8], [7.5.13] for the per-fix list and r
469
469
 
470
470
  ---
471
471
 
472
- **v8.92.0 | [Autonomi](https://www.autonomi.dev/) flagship product | ~410 lines core**
472
+ **v8.93.0 | [Autonomi](https://www.autonomi.dev/) flagship product | ~410 lines core**
package/VERSION CHANGED
@@ -1 +1 @@
1
- 8.92.0
1
+ 8.93.0
@@ -7,7 +7,7 @@ Modules:
7
7
  control: Session control API (start/stop/pause/resume)
8
8
  """
9
9
 
10
- __version__ = "8.92.0"
10
+ __version__ = "8.93.0"
11
11
 
12
12
  # Expose the control app for easy import
13
13
  try:
@@ -1,5 +1,5 @@
1
1
  // @bun
2
- var m_=Object.create;var{getPrototypeOf:u_,defineProperty:eK,getOwnPropertyNames:p_}=Object;var d_=Object.prototype.hasOwnProperty;function c_(Z){return this[Z]}var l_,i_,a_=(Z,X,Q)=>{var Y=Z!=null&&typeof Z==="object";if(Y){var J=X?l_??=new WeakMap:i_??=new WeakMap,z=J.get(Z);if(z)return z}Q=Z!=null?m_(u_(Z)):{};let K=X||!Z||!Z.__esModule?eK(Q,"default",{value:Z,enumerable:!0}):Q;for(let $ of p_(Z))if(!d_.call(K,$))eK(K,$,{get:c_.bind(Z,$),enumerable:!0});if(Y)J.set(Z,K);return K};var HQ=(Z,X)=>()=>(X||Z((X={exports:{}}).exports,X),X.exports);var s_=(Z)=>Z;function n_(Z,X){this[Z]=s_.bind(null,X)}var l0=(Z,X)=>{for(var Q in X)eK(Z,Q,{get:X[Q],enumerable:!0,configurable:!0,set:n_.bind(X,Q)})};var p=(Z,X)=>()=>(Z&&(X=Z(Z=0)),X);var e0=import.meta.require;var kO={};l0(kO,{lokiDir:()=>j0,homeLokiDir:()=>R4,findRepoRootForVersion:()=>X$,REPO_ROOT:()=>i0});import{resolve as n7,dirname as Z$}from"path";import{fileURLToPath as o_}from"url";import{existsSync as UQ}from"fs";import{homedir as r_}from"os";function t_(){let Z=RO;for(let X=0;X<6;X++){if(UQ(n7(Z,"VERSION"))&&UQ(n7(Z,"autonomy/run.sh")))return Z;let Q=Z$(Z);if(Q===Z)break;Z=Q}return n7(RO,"..","..","..")}function X$(Z){let X=Z;for(let Q=0;Q<6;Q++){if(UQ(n7(X,"VERSION"))&&UQ(n7(X,"autonomy/run.sh")))return X;let Y=Z$(X);if(Y===X)break;X=Y}return n7(Z,"..","..","..")}function j0(){return process.env.LOKI_DIR??n7(process.cwd(),".loki")}function R4(){return n7(r_(),".loki")}var RO,i0;var H8=p(()=>{RO=Z$(o_(import.meta.url));i0=t_()});import{readFileSync as e_}from"fs";import{resolve as Zf,dirname as Xf}from"path";import{fileURLToPath as Qf}from"url";function h3(){if(h5!==null)return h5;let Z="8.92.0";if(typeof Z==="string"&&Z.length>0)return h5=Z,h5;try{let X=Xf(Qf(import.meta.url)),Q=X$(X);h5=e_(Zf(Q,"VERSION"),"utf-8").trim()}catch{h5="unknown"}return h5}var h5=null;var BQ=p(()=>{H8()});var bO={};l0(bO,{runOrThrow:()=>jf,run:()=>E0,readStreamCapped:()=>NQ,commandVersion:()=>Tf,commandExists:()=>X9,ShellError:()=>Q$,MAX_STDOUT_BYTES:()=>yO});async function NQ(Z,X=yO){let Q=Z.getReader(),Y=new TextDecoder,J="",z=0;try{while(z<X){let{done:K,value:$}=await Q.read();if(K)break;if(!$)continue;if(z+=$.byteLength,z>X){let W=$.byteLength-(z-X);J+=Y.decode($.subarray(0,W),{stream:!0});break}J+=Y.decode($,{stream:!0})}J+=Y.decode()}finally{try{await Q.cancel()}catch{}Q.releaseLock()}return J}async function E0(Z,X={}){let Q=Bun.spawn({cmd:[...Z],stdout:"pipe",stderr:"pipe",env:X.env?{...process.env,...X.env}:process.env,cwd:X.cwd}),Y,J;if(X.timeoutMs&&X.timeoutMs>0)Y=setTimeout(()=>{try{Q.kill("SIGTERM")}catch{}J=setTimeout(()=>{try{Q.kill("SIGKILL")}catch{}},2000)},X.timeoutMs);try{let[z,K,$]=await Promise.all([NQ(Q.stdout),new Response(Q.stderr).text(),Q.exited]);return{stdout:z,stderr:K,exitCode:$}}finally{if(Y)clearTimeout(Y);if(J)clearTimeout(J)}}async function jf(Z,X={}){let Q=await E0(Z,X);if(Q.exitCode!==0)throw new Q$(`command failed (${Q.exitCode}): ${Z.join(" ")}`,Q.exitCode,Q.stdout,Q.stderr);return Q}async function X9(Z){let X=Mf(Z),Q=await E0(["sh","-c",`command -v ${X}`],{timeoutMs:5000});if(Q.exitCode===0)return Q.stdout.trim()||null;return null}function Mf(Z){if(!/^[A-Za-z0-9._/-]+$/.test(Z))throw Error(`refused to shell-escape suspect token: ${Z}`);return Z}async function Tf(Z,X="--version"){if(!await X9(Z))return null;let Y=await E0([Z,X],{timeoutMs:5000});if(Y.exitCode!==0)return null;return((Y.stdout||Y.stderr).split(/\r?\n/)[0]?.trim()??"")||null}var yO=16777216,Q$;var x9=p(()=>{Q$=class Q$ extends Error{message;exitCode;stdout;stderr;constructor(Z,X,Q,Y){super(Z);this.message=Z;this.exitCode=X;this.stdout=Q;this.stderr=Y;this.name="ShellError"}}});function o7(Z){return wf?"":Z}var wf,L0,F8,p0,zV0,a0,W8,Q9,v;var S6=p(()=>{wf=(process.env.NO_COLOR??"").length>0;L0=o7("\x1B[0;31m"),F8=o7("\x1B[0;32m"),p0=o7("\x1B[1;33m"),zV0=o7("\x1B[0;34m"),a0=o7("\x1B[0;36m"),W8=o7("\x1B[1m"),Q9=o7("\x1B[2m"),v=o7("\x1B[0m")});import{existsSync as bf}from"fs";async function E7(){if(x4!==void 0)return x4;let Z="/opt/homebrew/bin/python3.12";if(bf(Z))return x4=Z,Z;let X=await X9("python3.12");if(X)return x4=X,X;let Q=await X9("python3");return x4=Q,Q}async function Y7(Z,X={}){let Q=await E7();if(!Q)return{stdout:"",stderr:"python3 not found",exitCode:127};return E0([Q,"-c",Z],X)}var x4;var r7=p(()=>{x9()});var ZL={};l0(ZL,{runStatus:()=>Kh});import{existsSync as Y9,readFileSync as g3,readdirSync as iO,statSync as aO}from"fs";import{resolve as h8,basename as rf}from"path";import{homedir as tf}from"os";function sO(Z){let X=Math.trunc(Z);if(X>=1e6)return`${(Math.trunc(X/1e6*10)/10).toFixed(1)}M`;if(X>=1000)return`${(Math.trunc(X/1000*10)/10).toFixed(1)}K`;return String(X)}function nO(Z,X,Q){if(X===0)return null;let Y=Math.trunc(Z*100/X),J=Math.trunc(Z*LQ/X);if(J>LQ)J=LQ;let z=LQ-J,K=F8;if(Y>=80)K=L0;else if(Y>=50)K=p0;let $="=".repeat(Math.max(0,J))+" ".repeat(Math.max(0,z)),W=sO(Z),V=sO(X);return` ${W8}${Q}${v} ${K}[${$}]${v} ${Y}% (${W} / ${V})`}async function Zh(){if(await X9("jq"))return!0;return process.stdout.write(`${L0}Error: jq is required but not installed.${v}
2
+ var m_=Object.create;var{getPrototypeOf:u_,defineProperty:eK,getOwnPropertyNames:p_}=Object;var d_=Object.prototype.hasOwnProperty;function c_(Z){return this[Z]}var l_,i_,a_=(Z,X,Q)=>{var Y=Z!=null&&typeof Z==="object";if(Y){var J=X?l_??=new WeakMap:i_??=new WeakMap,z=J.get(Z);if(z)return z}Q=Z!=null?m_(u_(Z)):{};let K=X||!Z||!Z.__esModule?eK(Q,"default",{value:Z,enumerable:!0}):Q;for(let $ of p_(Z))if(!d_.call(K,$))eK(K,$,{get:c_.bind(Z,$),enumerable:!0});if(Y)J.set(Z,K);return K};var HQ=(Z,X)=>()=>(X||Z((X={exports:{}}).exports,X),X.exports);var s_=(Z)=>Z;function n_(Z,X){this[Z]=s_.bind(null,X)}var l0=(Z,X)=>{for(var Q in X)eK(Z,Q,{get:X[Q],enumerable:!0,configurable:!0,set:n_.bind(X,Q)})};var p=(Z,X)=>()=>(Z&&(X=Z(Z=0)),X);var e0=import.meta.require;var kO={};l0(kO,{lokiDir:()=>j0,homeLokiDir:()=>R4,findRepoRootForVersion:()=>X$,REPO_ROOT:()=>i0});import{resolve as n7,dirname as Z$}from"path";import{fileURLToPath as o_}from"url";import{existsSync as UQ}from"fs";import{homedir as r_}from"os";function t_(){let Z=RO;for(let X=0;X<6;X++){if(UQ(n7(Z,"VERSION"))&&UQ(n7(Z,"autonomy/run.sh")))return Z;let Q=Z$(Z);if(Q===Z)break;Z=Q}return n7(RO,"..","..","..")}function X$(Z){let X=Z;for(let Q=0;Q<6;Q++){if(UQ(n7(X,"VERSION"))&&UQ(n7(X,"autonomy/run.sh")))return X;let Y=Z$(X);if(Y===X)break;X=Y}return n7(Z,"..","..","..")}function j0(){return process.env.LOKI_DIR??n7(process.cwd(),".loki")}function R4(){return n7(r_(),".loki")}var RO,i0;var H8=p(()=>{RO=Z$(o_(import.meta.url));i0=t_()});import{readFileSync as e_}from"fs";import{resolve as Zf,dirname as Xf}from"path";import{fileURLToPath as Qf}from"url";function h3(){if(h5!==null)return h5;let Z="8.93.0";if(typeof Z==="string"&&Z.length>0)return h5=Z,h5;try{let X=Xf(Qf(import.meta.url)),Q=X$(X);h5=e_(Zf(Q,"VERSION"),"utf-8").trim()}catch{h5="unknown"}return h5}var h5=null;var BQ=p(()=>{H8()});var bO={};l0(bO,{runOrThrow:()=>jf,run:()=>E0,readStreamCapped:()=>NQ,commandVersion:()=>Tf,commandExists:()=>X9,ShellError:()=>Q$,MAX_STDOUT_BYTES:()=>yO});async function NQ(Z,X=yO){let Q=Z.getReader(),Y=new TextDecoder,J="",z=0;try{while(z<X){let{done:K,value:$}=await Q.read();if(K)break;if(!$)continue;if(z+=$.byteLength,z>X){let W=$.byteLength-(z-X);J+=Y.decode($.subarray(0,W),{stream:!0});break}J+=Y.decode($,{stream:!0})}J+=Y.decode()}finally{try{await Q.cancel()}catch{}Q.releaseLock()}return J}async function E0(Z,X={}){let Q=Bun.spawn({cmd:[...Z],stdout:"pipe",stderr:"pipe",env:X.env?{...process.env,...X.env}:process.env,cwd:X.cwd}),Y,J;if(X.timeoutMs&&X.timeoutMs>0)Y=setTimeout(()=>{try{Q.kill("SIGTERM")}catch{}J=setTimeout(()=>{try{Q.kill("SIGKILL")}catch{}},2000)},X.timeoutMs);try{let[z,K,$]=await Promise.all([NQ(Q.stdout),new Response(Q.stderr).text(),Q.exited]);return{stdout:z,stderr:K,exitCode:$}}finally{if(Y)clearTimeout(Y);if(J)clearTimeout(J)}}async function jf(Z,X={}){let Q=await E0(Z,X);if(Q.exitCode!==0)throw new Q$(`command failed (${Q.exitCode}): ${Z.join(" ")}`,Q.exitCode,Q.stdout,Q.stderr);return Q}async function X9(Z){let X=Mf(Z),Q=await E0(["sh","-c",`command -v ${X}`],{timeoutMs:5000});if(Q.exitCode===0)return Q.stdout.trim()||null;return null}function Mf(Z){if(!/^[A-Za-z0-9._/-]+$/.test(Z))throw Error(`refused to shell-escape suspect token: ${Z}`);return Z}async function Tf(Z,X="--version"){if(!await X9(Z))return null;let Y=await E0([Z,X],{timeoutMs:5000});if(Y.exitCode!==0)return null;return((Y.stdout||Y.stderr).split(/\r?\n/)[0]?.trim()??"")||null}var yO=16777216,Q$;var x9=p(()=>{Q$=class Q$ extends Error{message;exitCode;stdout;stderr;constructor(Z,X,Q,Y){super(Z);this.message=Z;this.exitCode=X;this.stdout=Q;this.stderr=Y;this.name="ShellError"}}});function o7(Z){return wf?"":Z}var wf,L0,F8,p0,zV0,a0,W8,Q9,v;var S6=p(()=>{wf=(process.env.NO_COLOR??"").length>0;L0=o7("\x1B[0;31m"),F8=o7("\x1B[0;32m"),p0=o7("\x1B[1;33m"),zV0=o7("\x1B[0;34m"),a0=o7("\x1B[0;36m"),W8=o7("\x1B[1m"),Q9=o7("\x1B[2m"),v=o7("\x1B[0m")});import{existsSync as bf}from"fs";async function E7(){if(x4!==void 0)return x4;let Z="/opt/homebrew/bin/python3.12";if(bf(Z))return x4=Z,Z;let X=await X9("python3.12");if(X)return x4=X,X;let Q=await X9("python3");return x4=Q,Q}async function Y7(Z,X={}){let Q=await E7();if(!Q)return{stdout:"",stderr:"python3 not found",exitCode:127};return E0([Q,"-c",Z],X)}var x4;var r7=p(()=>{x9()});var ZL={};l0(ZL,{runStatus:()=>Kh});import{existsSync as Y9,readFileSync as g3,readdirSync as iO,statSync as aO}from"fs";import{resolve as h8,basename as rf}from"path";import{homedir as tf}from"os";function sO(Z){let X=Math.trunc(Z);if(X>=1e6)return`${(Math.trunc(X/1e6*10)/10).toFixed(1)}M`;if(X>=1000)return`${(Math.trunc(X/1000*10)/10).toFixed(1)}K`;return String(X)}function nO(Z,X,Q){if(X===0)return null;let Y=Math.trunc(Z*100/X),J=Math.trunc(Z*LQ/X);if(J>LQ)J=LQ;let z=LQ-J,K=F8;if(Y>=80)K=L0;else if(Y>=50)K=p0;let $="=".repeat(Math.max(0,J))+" ".repeat(Math.max(0,z)),W=sO(Z),V=sO(X);return` ${W8}${Q}${v} ${K}[${$}]${v} ${Y}% (${W} / ${V})`}async function Zh(){if(await X9("jq"))return!0;return process.stdout.write(`${L0}Error: jq is required but not installed.${v}
3
3
  `),process.stdout.write(`Install with:
4
4
  `),process.stdout.write(` brew install jq (macOS)
5
5
  `),process.stdout.write(` apt install jq (Debian/Ubuntu)
@@ -1232,4 +1232,4 @@ Set LOKI_LEGACY_BASH=1 to force the bash CLI for every command.
1232
1232
  `),2}case"start":{let{runStart:Y}=await Promise.resolve().then(() => (h_(),f_));return Y(Q)}default:return process.stderr.write(`Unknown command: ${X}
1233
1233
  `),process.stderr.write(v_),2}}cO();process.on("SIGINT",()=>process.exit(130));process.on("SIGTERM",()=>process.exit(143));var uW0=await mW0(Bun.argv.slice(2));process.exit(uW0);
1234
1234
 
1235
- //# debugId=E465E6ADABBF4BDF64756E2164756E21
1235
+ //# debugId=232EF166A0E3908D64756E2164756E21
package/mcp/__init__.py CHANGED
@@ -75,4 +75,4 @@ try:
75
75
  except ImportError:
76
76
  __all__ = ['mcp']
77
77
 
78
- __version__ = '8.92.0'
78
+ __version__ = '8.93.0'
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "loki-mode",
3
3
  "mcpName": "io.github.asklokesh/loki-mode",
4
- "version": "8.92.0",
4
+ "version": "8.93.0",
5
5
  "description": "Loki Mode by Autonomi. Autonomous spec-to-product system: takes a PRD, GitHub issue, OpenAPI/JSON/YAML, or one-line brief to a deployed app via the RARV-C closure loop with 8 quality gates. Provider-agnostic (Claude Code, OpenAI Codex, Cline, Aider).",
6
6
  "keywords": [
7
7
  "agent",
@@ -2,7 +2,7 @@
2
2
  "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
3
3
  "name": "loki-mode",
4
4
  "displayName": "Loki Mode",
5
- "version": "8.92.0",
5
+ "version": "8.93.0",
6
6
  "description": "Autonomous spec-to-product build system with a built-in trust layer (RARV-C closure loop, 8 quality gates, completion council). Ships Loki's spec-hardening, drift-detection, and deterministic PR verification commands plus the Loki MCP server.",
7
7
  "author": {
8
8
  "name": "Autonomi",
@@ -0,0 +1,495 @@
1
+ #!/usr/bin/env python3
2
+ """Would a cheaper model have done this job, and what would it have saved.
3
+
4
+ WHY THIS EXISTS. This project has one order-of-magnitude finding from its own
5
+ benchmark work: on SWE-bench verified, MiniMax M2.5 (open weights) scored 75.8
6
+ against Claude Opus 4.6's 75.6, at $36.64 against $275.76 -- roughly 7.5x
7
+ cheaper for equal-or-better quality. Separately, the HARNESS was worth about
8
+ 3.4 points on an identical model. Nothing in the product surfaces any of that
9
+ to an operator deciding what to run. estimate-run.py answers "what will this
10
+ cost on the model I am already using". Nothing answers "should I be using a
11
+ different model at all".
12
+
13
+ WHAT SEPARATION MAKES THIS HONEST. Two things are kept strictly apart, and
14
+ conflating them is the failure mode this tool exists to avoid:
15
+
16
+ RATE, which we can compute. The pricing table gives USD per Mtok. Replaying
17
+ THIS workspace's observed token vector through another model's rates is
18
+ arithmetic, and the resulting ratio is a fact about prices.
19
+
20
+ QUALITY, which we cannot compute. Whether the cheaper model would have
21
+ produced the same result on YOUR workload is not something any cost record
22
+ can answer. The SWE-bench figures above are reported as a CITED EXTERNAL
23
+ BENCHMARK with the numbers attached, on a different workload, and they never
24
+ rank a candidate, never weight a recommendation, and are never phrased as a
25
+ promise. A ranking driven by a benchmark we did not run on your task would be
26
+ exactly the fabricated authority this repo has spent thirteen surfaces
27
+ removing.
28
+
29
+ THE HONESTY RULES, inherited from the cost lineage (v8.51.0 - v8.54.0, where a
30
+ single missing measurement became four different renderings of "$0.00"):
31
+
32
+ 1. NO MEASURED BASIS MEANS NO RECOMMENDATION. Not a smaller saving, not a
33
+ zero -- no number at all, and the output says why. A projected saving with
34
+ no history behind it is an invented fact, and it is worse than an invented
35
+ past cost because it invites a switch.
36
+ 2. Every projection is labelled ESTIMATE with the count of records it rests
37
+ on. A basis of one is stated as a basis of one.
38
+ 3. A CHEAPER RATE IS NOT A CHEAPER RUN. Token counts differ between models: a
39
+ weaker model can need more iterations, longer outputs, more retries, and
40
+ the saving assumes comparable token usage. We cannot verify that
41
+ assumption from cost records alone, and the output says so in the output
42
+ itself, not only in this docstring.
43
+ 4. A model absent from the pricing table reads "unpriced". Never $0, never
44
+ omitted silently, and never ranked -- an unpriced model with no ratio
45
+ cannot be claimed to be cheaper.
46
+
47
+ HOW THE RATIO IS COMPUTED, and why not the obvious way. The naive form is
48
+ candidate_input_rate / incumbent_input_rate. That is wrong here: output tokens
49
+ cost 5x input, cache reads a tenth, and the mix varies enormously between
50
+ workloads. So the observed token vector from the measured records is replayed
51
+ through BOTH models' full rate cards:
52
+
53
+ ratio = modelled_candidate / modelled_incumbent
54
+
55
+ then the saving scales the RECORDED cost, not the modelled one:
56
+
57
+ projected_candidate_usd = recorded_usd * ratio
58
+
59
+ The ratio is dimensionless, so it cancels drift between the table and what was
60
+ actually billed -- and sonnet is on intro pricing through 2026-08-31 per the
61
+ table's own _source note, so that drift is real today. The saving therefore
62
+ stays anchored to a measured number rather than to a price-card reconstruction.
63
+
64
+ If the modelled incumbent cost is zero (records carry cost but no tokens), the
65
+ ratio is undefined and we say so rather than falling back to an input-only
66
+ ratio, which would quietly answer a different question.
67
+
68
+ Usage:
69
+ python3 tools/model-advisor.py [WORKSPACE] [--json]
70
+ """
71
+
72
+ from __future__ import annotations
73
+
74
+ import argparse
75
+ import importlib.util
76
+ import json
77
+ import os
78
+ import sys
79
+
80
+ # ponytail: bytecode writing off before any file-based loader, so a probe that
81
+ # mutates and restores a file cannot be served a stale .pyc of the pre-mutation
82
+ # source.
83
+ sys.dont_write_bytecode = True
84
+
85
+ _HERE = os.path.dirname(os.path.abspath(__file__))
86
+ _REPO_ROOT = os.path.dirname(_HERE)
87
+ _LIB = os.path.join(_REPO_ROOT, "autonomy", "lib")
88
+ if _LIB not in sys.path:
89
+ sys.path.insert(0, _LIB)
90
+
91
+ # THE single definition of "measured". Restating it is how the honesty rule
92
+ # drifts; the four surfaces that once rendered an unmeasured run as "$0.00"
93
+ # each had their own idea of what counted.
94
+ from efficiency_cost import record_is_measured # noqa: E402
95
+
96
+ # One reader of .loki/metrics/efficiency/ -- it already skips malformed records
97
+ # rather than defaulting them to zero. cost-summary.py and estimate-run.py load
98
+ # it exactly this way; a third copy would drift the same way a second predicate
99
+ # would.
100
+ _ia_spec = importlib.util.spec_from_file_location(
101
+ "iteration_attribution", os.path.join(_LIB, "iteration_attribution.py"))
102
+ _ia = importlib.util.module_from_spec(_ia_spec)
103
+ _ia_spec.loader.exec_module(_ia)
104
+
105
+ # Alias-keyed: {"pricing": {"sonnet": {"input": 3.0, "output": 15.0,
106
+ # "cache_read": 0.3, "cache_write": 3.75}}}, USD per 1M tokens. NOT the schema
107
+ # of benchmarks/bench/prices.json (models.<x>.input_per_mtok), so this reads
108
+ # the file directly rather than routing through price_from_tokens.
109
+ PRICING_PATH = os.path.join(
110
+ _REPO_ROOT, "loki-ts", "data", "model-pricing.json")
111
+
112
+ # Record token field -> pricing rate field. cache_creation is billed at the
113
+ # write rate, which is 1.25x input and NOT the same as fresh input.
114
+ _TOKEN_TO_RATE = (
115
+ ("input_tokens", "input"),
116
+ ("output_tokens", "output"),
117
+ ("cache_read_tokens", "cache_read"),
118
+ ("cache_creation_tokens", "cache_write"),
119
+ )
120
+
121
+ # A cited EXTERNAL benchmark on an EXTERNAL workload. Reported verbatim with
122
+ # its numbers so a reader can weigh it; never an input to the ranking, which is
123
+ # ordered purely by computed rate. See module docstring.
124
+ SWEBENCH_CITATION = {
125
+ "benchmark": "SWE-bench verified",
126
+ "workload": "SWE-bench verified issues -- NOT this workspace's workload",
127
+ "results": [
128
+ {"model": "MiniMax M2.5 (open weights)", "score": 75.8,
129
+ "cost_usd": 36.64},
130
+ {"model": "Claude Opus 4.6", "score": 75.6, "cost_usd": 275.76},
131
+ ],
132
+ "harness_effect": (
133
+ "the harness itself was worth about 3.4 points on an identical model"),
134
+ "caveat": (
135
+ "equal-or-better score at roughly 7.5x lower cost ON THAT BENCHMARK. "
136
+ "It is not a measurement of your workload and does not predict that a "
137
+ "cheaper model would complete YOUR task"),
138
+ }
139
+
140
+
141
+ def _num(v):
142
+ """Non-bool int/float, else None. Never coerces junk to 0."""
143
+ if isinstance(v, bool) or not isinstance(v, (int, float)):
144
+ return None
145
+ return v
146
+
147
+
148
+ def load_pricing(path=None):
149
+ """Return the alias -> rates map, or {} when unreadable.
150
+
151
+ A missing table means we cannot quote any rate, which is an honest null:
152
+ every model then reads unpriced and no ratio is offered.
153
+ """
154
+ try:
155
+ with open(path or PRICING_PATH, encoding="utf-8") as handle:
156
+ data = json.load(handle)
157
+ except Exception:
158
+ return {}
159
+ pricing = data.get("pricing") if isinstance(data, dict) else None
160
+ return pricing if isinstance(pricing, dict) else {}
161
+
162
+
163
+ def modelled_cost(tokens, rates):
164
+ """Replay an observed token vector through one model's full rate card.
165
+
166
+ Returns USD, or None when the model has no usable rates -- unpriced, which
167
+ must never render as 0. Rates are per 1M tokens.
168
+ """
169
+ if not isinstance(rates, dict):
170
+ return None
171
+ total = 0.0
172
+ priced_any = False
173
+ for tok_key, rate_key in _TOKEN_TO_RATE:
174
+ rate = _num(rates.get(rate_key))
175
+ if rate is None:
176
+ continue
177
+ priced_any = True
178
+ total += (tokens.get(tok_key, 0) / 1_000_000.0) * rate
179
+ # A table entry naming the model but carrying no numeric rate prices
180
+ # nothing. That is unpriced, not free.
181
+ return total if priced_any else None
182
+
183
+
184
+ def advise(workspace=".", pricing_path=None):
185
+ """Build the advice dict. Pure derivation from measured records."""
186
+ loki_dir = os.path.join(workspace, ".loki")
187
+ recs = _ia._iteration_records(loki_dir)
188
+
189
+ found = len(recs)
190
+ measured = 0
191
+ priced = 0
192
+ recorded_usd = 0.0
193
+ tokens = {key: 0 for key, _ in _TOKEN_TO_RATE}
194
+ by_model_cost = {}
195
+
196
+ for rec in recs:
197
+ if not record_is_measured(rec):
198
+ # EXCLUDED, not summed as zero. An unmeasured iteration averaged in
199
+ # as 0 is indistinguishable from a real measurement of 0.
200
+ continue
201
+ measured += 1
202
+ usd = _num(rec.get("cost_usd"))
203
+ # Measured on tokens but carrying no cost is real for token accounting
204
+ # and useless as a COST basis. Real costs are stored raw (0.018719), so
205
+ # a sub-cent charge is 0.0001 and never exactly 0; an exact zero is a
206
+ # reliable unpriced signal. estimate-run.py draws the same line.
207
+ if usd is None or usd == 0:
208
+ continue
209
+ priced += 1
210
+ recorded_usd += float(usd)
211
+ for key, _ in _TOKEN_TO_RATE:
212
+ tokens[key] += _num(rec.get(key)) or 0
213
+ name = str(rec.get("model") or "").strip()
214
+ by_model_cost[name] = by_model_cost.get(name, 0.0) + float(usd)
215
+
216
+ pricing = load_pricing(pricing_path)
217
+
218
+ # THE INCUMBENT. Not collect_efficiency()'s last-non-empty-seen model:
219
+ # across mixed history that names whichever model happened to be recorded
220
+ # last, which is not the model the spend belongs to. Dominant by COST
221
+ # share, with the mix always reported.
222
+ incumbent = None
223
+ if by_model_cost:
224
+ named = {k: v for k, v in by_model_cost.items() if k}
225
+ if named:
226
+ incumbent = max(named, key=lambda k: named[k])
227
+ mixed = len([k for k in by_model_cost if k]) > 1
228
+
229
+ incumbent_rates = pricing.get(incumbent) if incumbent else None
230
+ incumbent_modelled = modelled_cost(tokens, incumbent_rates)
231
+
232
+ out = {
233
+ "workspace": os.path.abspath(workspace),
234
+ "label": "ESTIMATE",
235
+ "iterations_found": found,
236
+ "iterations_measured": measured,
237
+ "iterations_priced": priced,
238
+ "basis_count": priced,
239
+ "has_basis": priced > 0,
240
+ "single_point_basis": priced == 1,
241
+ "incumbent_model": incumbent,
242
+ "incumbent_priced": incumbent_rates is not None,
243
+ "incumbent_recorded_cost_usd": (
244
+ round(recorded_usd, 4) if priced else None),
245
+ "observed_tokens": dict(tokens) if priced else None,
246
+ "basis_models": sorted(k for k in by_model_cost if k),
247
+ "mixed_basis_models": mixed,
248
+ "candidates": [],
249
+ "best_candidate": None,
250
+ "swebench_citation": SWEBENCH_CITATION,
251
+ "assumption_verified": False,
252
+ "notes": [],
253
+ }
254
+
255
+ n = out["notes"]
256
+
257
+ # ---- THE HONESTY GUARD -------------------------------------------------
258
+ # No priced history means no basis, so there is no recommendation and no
259
+ # number. Softening this into a 0.0 saving, or into a rate-only ranking
260
+ # dressed as advice, is the "unmeasured becomes free" defect pointed at a
261
+ # purchasing decision.
262
+ if not out["has_basis"]:
263
+ out["incumbent_recorded_cost_usd"] = None
264
+ if found == 0:
265
+ n.append(
266
+ "NO BASIS: no iteration records in this workspace. There is no "
267
+ "measured history to compare models against, so no saving is "
268
+ "projected (not $0.00) and no model is recommended")
269
+ else:
270
+ n.append(
271
+ "NO BASIS: no measured, priced iteration among %d record(s). "
272
+ "There is no measured history to compare models against, so no "
273
+ "saving is projected (not $0.00) and no model is recommended"
274
+ % found)
275
+ if measured > priced:
276
+ n.append(
277
+ "%d of %d measured record(s) carried tokens but no cost "
278
+ "(unpriced): that spend is UNKNOWN rather than zero, so it "
279
+ "cannot form a basis" % (measured - priced, measured))
280
+ n.append(
281
+ "run an iteration with cost recording enabled, then re-run this "
282
+ "tool")
283
+ return out
284
+ # ------------------------------------------------------------------------
285
+
286
+ n.append(
287
+ "ESTIMATE based on %d measured, priced iteration(s) of %d found -- "
288
+ "not a guarantee" % (priced, found))
289
+ if priced == 1:
290
+ n.append(
291
+ "the basis is a SINGLE data point: one observation extrapolated, "
292
+ "not a distribution")
293
+ if priced < found:
294
+ n.append(
295
+ "PARTIAL: %d of %d record(s) did not inform this comparison "
296
+ "(unmeasured or unpriced), and were excluded rather than counted "
297
+ "as zero" % (found - priced, found))
298
+ if mixed:
299
+ n.append(
300
+ "the basis MIXES models (%s): the incumbent below is the one with "
301
+ "the largest share of recorded cost, and a single ratio across "
302
+ "different price points may not transfer to either"
303
+ % ", ".join(out["basis_models"]))
304
+
305
+ if incumbent is None:
306
+ n.append(
307
+ "the priced records name no model: the incumbent is unknown, so no "
308
+ "rate ratio can be computed and no saving is projected")
309
+ return out
310
+
311
+ if incumbent_rates is None:
312
+ n.append(
313
+ "no price listed for the incumbent %s in the pricing table: it "
314
+ "reads UNPRICED (not $0), and without its rates no ratio against "
315
+ "any candidate can be computed, so no saving is projected"
316
+ % incumbent)
317
+ return out
318
+
319
+ if not incumbent_modelled:
320
+ # Cost recorded but no tokens (or all rates zero): the ratio's
321
+ # denominator is zero. Falling back to an input-rate-only ratio here
322
+ # would silently answer a different question.
323
+ n.append(
324
+ "the measured records carry cost but no usable token counts, so "
325
+ "the incumbent's modelled cost is zero and every ratio would be "
326
+ "undefined: no saving is projected")
327
+ return out
328
+
329
+ # ---- CANDIDATES --------------------------------------------------------
330
+ # Ordered purely by computed rate ratio. The SWE-bench citation is attached
331
+ # to the report as external context and deliberately does NOT rank anything.
332
+ for name in sorted(pricing):
333
+ if name == incumbent:
334
+ continue
335
+ cand_modelled = modelled_cost(tokens, pricing.get(name))
336
+ if cand_modelled is None:
337
+ # Named in the table but with no usable rate: unpriced, so it is
338
+ # listed as unpriced and carries no ratio and no saving.
339
+ out["candidates"].append({
340
+ "model": name, "priced": False, "rate_ratio": None,
341
+ "projected_cost_usd": None, "projected_saving_usd": None,
342
+ "cheaper": None,
343
+ })
344
+ continue
345
+ ratio = cand_modelled / incumbent_modelled
346
+ projected = recorded_usd * ratio
347
+ out["candidates"].append({
348
+ "model": name,
349
+ "priced": True,
350
+ "rate_ratio": round(ratio, 4),
351
+ "projected_cost_usd": round(projected, 4),
352
+ "projected_saving_usd": round(recorded_usd - projected, 4),
353
+ "cheaper": ratio < 1.0,
354
+ })
355
+
356
+ cheaper = [c for c in out["candidates"] if c["cheaper"]]
357
+ cheaper.sort(key=lambda c: c["rate_ratio"])
358
+ out["candidates"].sort(
359
+ key=lambda c: (c["rate_ratio"] is None, c["rate_ratio"]))
360
+
361
+ if cheaper:
362
+ out["best_candidate"] = cheaper[0]["model"]
363
+ n.append(
364
+ "%d cheaper-RATE candidate(s) found; cheapest is %s at %.2fx the "
365
+ "incumbent's rate on this workspace's observed token mix"
366
+ % (len(cheaper), cheaper[0]["model"], cheaper[0]["rate_ratio"]))
367
+ else:
368
+ # Already on the cheapest priced model. Saying "no cheaper candidate"
369
+ # is a real answer; an empty ranked list rendered as a saving of
370
+ # nothing is not.
371
+ n.append(
372
+ "no cheaper-rate candidate: %s is already the cheapest priced "
373
+ "model in the table for this token mix" % incumbent)
374
+
375
+ unpriced = [c["model"] for c in out["candidates"] if not c["priced"]]
376
+ if unpriced:
377
+ n.append(
378
+ "unpriced (no rate in the pricing table, so UNKNOWN and not $0, "
379
+ "and not ranked): %s" % ", ".join(unpriced))
380
+
381
+ # RULE 3, stated in the output and not only in the source. Every number
382
+ # above is a rate comparison holding tokens fixed; nothing here has
383
+ # measured a second model on this workload.
384
+ n.append(
385
+ "A CHEAPER RATE IS NOT A CHEAPER RUN. Every projection above replays "
386
+ "THIS run's observed token counts through another model's rates. Token "
387
+ "counts differ between models -- a different model may need more "
388
+ "iterations, longer outputs or more retries -- so the saving assumes "
389
+ "COMPARABLE TOKEN USAGE")
390
+ n.append(
391
+ "that assumption is NOT VERIFIED here: this workspace has no measured "
392
+ "run on any candidate model, so its real token usage is unknown. The "
393
+ "only way to verify it is to run the candidate and compare")
394
+ n.append(
395
+ "quality is NOT projected. The SWE-bench figures in this report are a "
396
+ "cited external benchmark on a different workload, not a prediction "
397
+ "about your task")
398
+ return out
399
+
400
+
401
+ def _fmt_usd(v):
402
+ """UNKNOWN, never $0.00, when there is nothing to report."""
403
+ return "UNKNOWN" if v is None else "$%.4f" % v
404
+
405
+
406
+ def _fmt_saving(v):
407
+ """A negative saving is EXTRA COST and must not read as one.
408
+
409
+ "$-1.6558" in a SAVING column is a number a reader skims as a saving with a
410
+ stray character. A more expensive model costs more; say that.
411
+ """
412
+ if v is None:
413
+ return "UNKNOWN"
414
+ return "$%.4f" % v if v >= 0 else "+$%.4f more" % abs(v)
415
+
416
+
417
+ def render(adv):
418
+ """Human-readable report. The honesty lives here too, not only in the dict."""
419
+ lines = []
420
+ lines.append("Model cost advisor -- %s" % adv["workspace"])
421
+ lines.append("")
422
+
423
+ if not adv["has_basis"]:
424
+ lines.append(" NO BASIS: no measured, priced iteration in this workspace.")
425
+ lines.append(" Model used: UNKNOWN")
426
+ lines.append(" Measured cost: UNKNOWN")
427
+ lines.append(" Recommendation: NONE -- there is no measured basis")
428
+ lines.append(" Projected saving: UNKNOWN")
429
+ lines.append(" Records found: %d measured: %d priced: %d"
430
+ % (adv["iterations_found"], adv["iterations_measured"],
431
+ adv["iterations_priced"]))
432
+ else:
433
+ lines.append(" Model used: %s%s" % (
434
+ adv["incumbent_model"] or "not recorded",
435
+ "" if adv["incumbent_priced"] else " (UNPRICED)"))
436
+ lines.append(" Measured cost: %s (ESTIMATE basis: %d priced "
437
+ "iteration(s) of %d found)" % (
438
+ _fmt_usd(adv["incumbent_recorded_cost_usd"]),
439
+ adv["basis_count"], adv["iterations_found"]))
440
+ lines.append("")
441
+ if adv["candidates"]:
442
+ lines.append(" Candidates (ESTIMATE, basis %d iteration(s)):"
443
+ % adv["basis_count"])
444
+ lines.append(" %-16s %-10s %-12s %s"
445
+ % ("MODEL", "RATE", "PROJECTED", "SAVING"))
446
+ for cand in adv["candidates"]:
447
+ if not cand["priced"]:
448
+ lines.append(" %-16s %-10s %-12s %s" % (
449
+ cand["model"], "unpriced", "unpriced", "unpriced"))
450
+ continue
451
+ lines.append(" %-16s %-10s %-12s %s" % (
452
+ cand["model"],
453
+ "%.2fx" % cand["rate_ratio"],
454
+ _fmt_usd(cand["projected_cost_usd"]),
455
+ _fmt_saving(cand["projected_saving_usd"])))
456
+ lines.append("")
457
+ if adv["best_candidate"]:
458
+ lines.append(" Cheapest rate: %s" % adv["best_candidate"])
459
+ else:
460
+ lines.append(" Cheapest rate: none cheaper than the model "
461
+ "already in use")
462
+
463
+ cite = adv["swebench_citation"]
464
+ lines.append("")
465
+ lines.append(" Cited external benchmark (%s) -- NOT a measurement of your "
466
+ "workload:" % cite["benchmark"])
467
+ for row in cite["results"]:
468
+ lines.append(" %-30s score %.1f cost $%.2f"
469
+ % (row["model"], row["score"], row["cost_usd"]))
470
+ lines.append(" %s" % cite["harness_effect"])
471
+ lines.append(" %s" % cite["caveat"])
472
+
473
+ lines.append("")
474
+ for note in adv["notes"]:
475
+ lines.append(" - %s" % note)
476
+ return "\n".join(lines)
477
+
478
+
479
+ def main(argv=None):
480
+ ap = argparse.ArgumentParser(
481
+ description="Recommend a cheaper model from this workspace's measured "
482
+ "cost history, and quantify the saving.")
483
+ ap.add_argument("workspace", nargs="?", default=".")
484
+ ap.add_argument("--json", action="store_true")
485
+ args = ap.parse_args(argv)
486
+
487
+ adv = advise(args.workspace)
488
+ print(json.dumps(adv, indent=2) if args.json else render(adv))
489
+ # Exit 0 either way: "no basis" is a successful, honest answer, not a tool
490
+ # failure. Callers read has_basis.
491
+ return 0
492
+
493
+
494
+ if __name__ == "__main__":
495
+ sys.exit(main())
@@ -0,0 +1,314 @@
1
+ #!/usr/bin/env python3
2
+ """Collect every receipt under a workspace into ONE verifiable audit trail.
3
+
4
+ WHY THIS EXISTS. Single-receipt verification exists (autonomy/lib/
5
+ proof-verify.py). Cross-run comparison exists (receipt-diff.py). Portable
6
+ attestation exists (receipt-attest.py). All three answer questions about ONE
7
+ run, or two. Nobody has to hand a compliance reviewer one run. They have to
8
+ hand over a SEQUENCE -- every run in the workspace, and a single verdict over
9
+ the lot.
10
+
11
+ That rollup is where the laundering happens, so it is what this file is
12
+ mostly about.
13
+
14
+ THE FOUR RULES. Each one is a specific way a bundle report can claim more than
15
+ it earned, and each is the honest direction, not the flattering one.
16
+
17
+ 1. THE ROLLUP IS THE WEAKEST LINK, NEVER AN AVERAGE. Nine good receipts and
18
+ one forged one is a FAILED bundle. Not "90% verified", not "mostly clean",
19
+ not a score. A percentage lets a bad run hide inside a crowd of good ones,
20
+ and hiding a bad run inside good ones is the entire attack this product
21
+ line exists to prevent. Averaging also gets EASIER to pass as the bundle
22
+ grows, which is exactly backwards: more runs should be harder to certify,
23
+ not cheaper. See rollup() -- it is four lines and it is the product.
24
+
25
+ 2. AN UNVERIFIABLE RECEIPT IS NEVER SILENTLY DROPPED. Absent is not clean. A
26
+ bundle that quietly excludes what it could not check reports a stronger
27
+ claim than it earned, and it does so invisibly -- the reader sees only
28
+ receipts that passed and cannot tell whether that is because the rest were
29
+ good or because the rest were omitted. So an unverifiable receipt is
30
+ counted, NAMED, given a reason, and it holds the rollup down to
31
+ UNVERIFIABLE. This is why the per-receipt state is three-valued and is NOT
32
+ read off verify()'s `ok` field: `ok` is False for BOTH a tampered receipt
33
+ and an honest receipt read from the wrong directory (proof-verify.py:706
34
+ says so outright -- diff_drift None makes ok False by design). Collapsing
35
+ those two together makes this rule unsatisfiable, because "name why" has no
36
+ answer when FAILED and UNVERIFIABLE are the same bucket.
37
+
38
+ 3. TOTAL COST SUMS ONLY MEASURED RECEIPTS, AND SAYS HOW MANY OF HOW MANY. A
39
+ total over 10 receipts of which 3 measured cost is not a bundle total; it
40
+ is a third of one wearing the label of the whole. The ratio rides with the
41
+ number, always. And if NOTHING measured, the total reads UNKNOWN -- never
42
+ $0.00, because unmeasured and free are different claims and only one of
43
+ them is honest. The measured/unmeasured predicate is record_is_measured()
44
+ in autonomy/lib/efficiency_cost.py, reached through receipt-diff.py's
45
+ measured_cost() which already maps the receipt's `usd` onto the
46
+ per-iteration `cost_usd` key. Imported twice over, restated zero times: a
47
+ second copy of that predicate is precisely how the honesty rule drifts.
48
+
49
+ 4. AN EMPTY WORKSPACE IS NOT A PASSING AUDIT. Zero receipts verified is not
50
+ "everything verified". Vacuous truth is the cheapest false green there is,
51
+ and a tool that reports EMPTY as clean can be passed by deleting the
52
+ evidence. So EMPTY is its own verdict with its own non-zero exit code.
53
+
54
+ Nothing here re-implements verification. verify() in proof-verify.py is the
55
+ single source of truth; this projects its result onto three states and folds
56
+ them with min().
57
+
58
+ Usage:
59
+ tools/receipt-bundle.py [workspace] [--json]
60
+
61
+ Exit codes:
62
+ 0 every receipt in the bundle VERIFIED
63
+ 1 at least one receipt FAILED
64
+ 2 nothing failed, but at least one receipt was UNVERIFIABLE
65
+ 3 no receipts found at all -- nothing was audited
66
+ """
67
+
68
+ import argparse
69
+ import importlib.util
70
+ import json
71
+ import os
72
+ import pathlib
73
+ import sys
74
+
75
+ # A stale .pyc for a hyphenated module loaded by path makes mutation probes
76
+ # report FALSE failures (the probe edits the source, the loader serves the old
77
+ # bytecode). Must be set before any loader below runs.
78
+ sys.dont_write_bytecode = True
79
+
80
+ _ROOT = pathlib.Path(__file__).resolve().parents[1]
81
+ _LIB = _ROOT / "autonomy" / "lib"
82
+
83
+
84
+ def _load(name, path):
85
+ spec = importlib.util.spec_from_file_location(name, path)
86
+ mod = importlib.util.module_from_spec(spec)
87
+ spec.loader.exec_module(mod)
88
+ return mod
89
+
90
+
91
+ _pv = _load("proof_verify", _LIB / "proof-verify.py")
92
+ # measured_cost() already maps the receipt's cost block onto the per-iteration
93
+ # key names record_is_measured() expects. Reused rather than re-derived.
94
+ _rd = _load("receipt_diff", _ROOT / "tools" / "receipt-diff.py")
95
+
96
+ verify = _pv.verify
97
+ measured_cost = _rd.measured_cost
98
+
99
+ VERIFIED = "VERIFIED"
100
+ UNVERIFIABLE = "UNVERIFIABLE"
101
+ FAILED = "FAILED"
102
+ EMPTY = "EMPTY"
103
+
104
+ # Ordered worst-first. rollup() takes the min index, which IS the weakest-link
105
+ # rule: one FAILED sinks any number of VERIFIED.
106
+ _ORDER = (FAILED, UNVERIFIABLE, VERIFIED)
107
+
108
+ EXIT = {VERIFIED: 0, FAILED: 1, UNVERIFIABLE: 2, EMPTY: 3}
109
+
110
+
111
+ def rollup(states):
112
+ """The bundle verdict: the WEAKEST state present, never an average.
113
+
114
+ This is the whole product. One FAILED receipt among a thousand VERIFIED
115
+ ones makes the bundle FAILED, because a buyer handing this to a compliance
116
+ reviewer is claiming the SEQUENCE is sound, and a sequence containing a
117
+ forged run is not sound no matter what fraction it represents.
118
+
119
+ Any scoring rule -- mean, median, majority, "90% or better" -- makes a bad
120
+ run cheaper to hide the more good runs surround it. Empty is its own
121
+ verdict, not a vacuous pass (rule 4).
122
+ """
123
+ if not states:
124
+ return EMPTY
125
+ return min(states, key=_ORDER.index)
126
+
127
+
128
+ def receipt_state(proof_path, repo_dir="."):
129
+ """Project one receipt onto (state, reason), keeping three states apart.
130
+
131
+ verify()'s `ok` is deliberately NOT the source here. It is False both for
132
+ a receipt that failed a check and for one whose checks could not run at
133
+ all -- and telling those apart is the entire point of rule 2. So the
134
+ failure signals are read individually:
135
+
136
+ a check RAN and said no -> FAILED (hash_ok False,
137
+ diff_drift True,
138
+ gpg_ok False,
139
+ headline_consistent False)
140
+ a check COULD NOT RUN -> UNVERIFIABLE (diff_drift None, or the
141
+ receipt would not load)
142
+
143
+ `is` comparisons are load-bearing throughout: gpg_ok is the truthy string
144
+ "n/a" when there is nothing to check, and `if not gpg_ok` would read that
145
+ as a signature failure.
146
+ """
147
+ try:
148
+ result = verify(str(proof_path), repo_dir)
149
+ except Exception as exc: # unreadable, unparseable, wrong shape
150
+ # Named and counted, never dropped. This is the branch rule 2 is about.
151
+ return UNVERIFIABLE, "the receipt could not be loaded or verified: %s" % exc
152
+
153
+ reasons = result.get("reasons") or []
154
+ detail = reasons[0] if reasons else (result.get("reason") or "")
155
+
156
+ if not result.get("hash_ok"):
157
+ return FAILED, detail or "the recorded integrity hash does not match the receipt bytes"
158
+ if result.get("diff_drift") is True:
159
+ return FAILED, detail or "the recorded diff no longer matches the repository"
160
+ if result.get("gpg_ok") is False:
161
+ return FAILED, detail or "the gpg signature does not verify"
162
+ if result.get("headline_consistent") is False:
163
+ return FAILED, detail or "the headline disagrees with the recorded facts"
164
+ if result.get("diff_drift") is None:
165
+ return UNVERIFIABLE, detail or "the recorded diff could not be re-derived here"
166
+ if not result.get("ok"):
167
+ # verify() sank the verdict on an axis not enumerated above. Report it
168
+ # as FAILED with its own words rather than silently upgrading to
169
+ # VERIFIED -- a new check added upstream must not arrive here as a pass.
170
+ return FAILED, detail or "verification failed"
171
+ return VERIFIED, ""
172
+
173
+
174
+ def find_receipts(workspace):
175
+ """Every proof.json under the workspace, sorted for a stable report.
176
+
177
+ ponytail: rglob over the whole tree rather than only .loki/proofs/*, so a
178
+ receipt archived elsewhere in the workspace still gets audited. Missing
179
+ evidence is the failure mode this file exists to prevent; over-collecting
180
+ is not.
181
+ """
182
+ root = pathlib.Path(workspace)
183
+ return sorted(p for p in root.rglob("proof.json") if p.is_file())
184
+
185
+
186
+ def bundle(workspace, repo_dir="."):
187
+ """Audit every receipt under `workspace`. Pure: no writes, no network."""
188
+ paths = find_receipts(workspace)
189
+
190
+ receipts = []
191
+ total = 0.0
192
+ measured_n = 0
193
+
194
+ for path in paths:
195
+ state, reason = receipt_state(path, repo_dir)
196
+ entry = {
197
+ "path": str(path),
198
+ "state": state,
199
+ "reason": reason,
200
+ "cost_usd": None,
201
+ }
202
+
203
+ # Cost is read from the receipt regardless of verdict, but only a
204
+ # MEASURED cost contributes. measured_cost() returns None when the
205
+ # block is absent, malformed, or all-zero.
206
+ try:
207
+ cost = measured_cost(_pv._load_proof(str(path)))
208
+ except Exception:
209
+ cost = None
210
+ if cost is not None and cost.get("cost_usd") is not None:
211
+ entry["cost_usd"] = cost["cost_usd"]
212
+ total += cost["cost_usd"]
213
+ measured_n += 1
214
+
215
+ receipts.append(entry)
216
+
217
+ verdict = rollup([r["state"] for r in receipts])
218
+
219
+ # measured_n, NOT the total, decides UNKNOWN. Three receipts that each
220
+ # genuinely measured $0.00 is a real observation and must survive as a
221
+ # number; `if not total` would erase it and reintroduce the exact
222
+ # unmeasured-reads-as-free defect this codebase spent four releases on.
223
+ cost_block = {
224
+ "measured_receipts": measured_n,
225
+ "total_receipts": len(receipts),
226
+ "total_usd": total if measured_n else None,
227
+ }
228
+
229
+ unverifiable = [r for r in receipts if r["state"] != VERIFIED]
230
+
231
+ return {
232
+ "bundle": "loki-receipt-bundle/v1",
233
+ "workspace": os.path.abspath(str(workspace)),
234
+ "checked_from": os.path.abspath(repo_dir),
235
+ "receipts": receipts,
236
+ "counts": {
237
+ state: sum(1 for r in receipts if r["state"] == state)
238
+ for state in _ORDER
239
+ },
240
+ "cost": cost_block,
241
+ "not_verified": [
242
+ {"path": r["path"], "state": r["state"], "reason": r["reason"]}
243
+ for r in unverifiable
244
+ ],
245
+ "verdict": verdict,
246
+ "summary": _summary(verdict, receipts, cost_block),
247
+ }
248
+
249
+
250
+ def _cost_line(cost):
251
+ """The total, always carrying its own ratio. UNKNOWN when nothing measured."""
252
+ if cost["total_usd"] is None:
253
+ return "total cost UNKNOWN (0 of %d receipts measured cost)" % (
254
+ cost["total_receipts"])
255
+ return "total cost $%.4f across %d of %d receipts measured" % (
256
+ cost["total_usd"], cost["measured_receipts"], cost["total_receipts"])
257
+
258
+
259
+ def _summary(verdict, receipts, cost):
260
+ if verdict == EMPTY:
261
+ return ("EMPTY -- no receipts found under this workspace, so nothing "
262
+ "was audited. Zero receipts is not a passing audit.")
263
+ n = len(receipts)
264
+ bad = sum(1 for r in receipts if r["state"] == FAILED)
265
+ unv = sum(1 for r in receipts if r["state"] == UNVERIFIABLE)
266
+ if verdict == VERIFIED:
267
+ head = "VERIFIED -- all %d receipts verified" % n
268
+ elif verdict == FAILED:
269
+ head = ("FAILED -- %d of %d receipts FAILED verification; the bundle "
270
+ "is only as good as its weakest receipt" % (bad, n))
271
+ else:
272
+ head = ("UNVERIFIABLE -- %d of %d receipts could not be checked here; "
273
+ "nothing failed, but the bundle is not proven" % (unv, n))
274
+ return head + ". " + _cost_line(cost)
275
+
276
+
277
+ def _render(report):
278
+ lines = ["Receipt bundle: %s" % report["workspace"], ""]
279
+ for r in report["receipts"]:
280
+ lines.append(" %-13s %s" % (r["state"], r["path"]))
281
+ if r["reason"]:
282
+ lines.append(" %s" % r["reason"])
283
+ if report["receipts"]:
284
+ lines.append("")
285
+ if report["not_verified"]:
286
+ lines.append("Not verified (%d) -- counted, never dropped:"
287
+ % len(report["not_verified"]))
288
+ for r in report["not_verified"]:
289
+ lines.append(" %s [%s]" % (r["path"], r["state"]))
290
+ lines.append(" %s" % r["reason"])
291
+ lines.append("")
292
+ lines.append(_cost_line(report["cost"]))
293
+ lines.append("")
294
+ lines.append(report["summary"])
295
+ return "\n".join(lines)
296
+
297
+
298
+ def main(argv=None):
299
+ ap = argparse.ArgumentParser(
300
+ description="Verify every receipt under a workspace as one bundle.")
301
+ ap.add_argument("workspace", nargs="?", default=".",
302
+ help="workspace to scan for receipts (default: .)")
303
+ ap.add_argument("--json", action="store_true", help="emit the raw record")
304
+ ap.add_argument("--repo-dir", default=".",
305
+ help="repository the receipts are re-checked against")
306
+ args = ap.parse_args(argv)
307
+
308
+ report = bundle(args.workspace, args.repo_dir)
309
+ print(json.dumps(report, indent=2) if args.json else _render(report))
310
+ return EXIT[report["verdict"]]
311
+
312
+
313
+ if __name__ == "__main__":
314
+ sys.exit(main())