great-cto 3.26.2 → 3.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,266 @@
1
+ /**
2
+ * agent-posture — what does this agent's tool grant actually LET IT DO?
3
+ *
4
+ * ADR-009 ends with an instruction nobody had a way to follow: "Ask at design
5
+ * time, when the capability is added — not after the incident." The question it
6
+ * asks is about consequence — is this expensive to undo? — while every agent
7
+ * file answers a different question, in a different language: which tools are on
8
+ * the `tools:` line. Reviewing the second does not answer the first. `Bash` and
9
+ * `Bash(node:*)` sit two characters apart and read as careful and careless; in
10
+ * consequence they are the same grant.
11
+ *
12
+ * So this names the grant in the language of the decision. Vocabulary shape
13
+ * borrowed from OpenFirma's capability postures (`credential.read`,
14
+ * `communication.external.send`, `code.destructive`) — the idea only; that
15
+ * project is GPL-3.0 and this one is MIT, so nothing was copied.
16
+ *
17
+ * NOTHING here decides anything, exactly as in `gate-reversibility.mjs`, whose
18
+ * ADR-009 categories it reuses rather than inventing a second vocabulary for the
19
+ * same axis. It classifies, so a reviewer can see the grant they are approving.
20
+ */
21
+
22
+ import { CATEGORIES } from './gate-reversibility.mjs';
23
+
24
+ /**
25
+ * The postures. Each cites the ADR-009 category that makes it expensive, or
26
+ * `null` when the repair is simply to do the thing again.
27
+ */
28
+ export const POSTURES = Object.freeze({
29
+ 'code.read': {
30
+ category: null,
31
+ means: 'read files in the working tree',
32
+ },
33
+ 'code.write': {
34
+ category: null,
35
+ means: 'create or modify files — the repair is another edit',
36
+ },
37
+ 'code.destructive': {
38
+ category: 'destroys-evidence',
39
+ means: 'delete files, rewrite history, or overwrite work that is not in the index',
40
+ },
41
+ 'credential.read': {
42
+ category: 'unrevocable-disclosure',
43
+ means: 'reach secrets on disk or in the environment',
44
+ },
45
+ 'communication.external.send': {
46
+ category: 'escapes-the-machine',
47
+ means: 'send data off this machine — a push, a publish, a request body, a URL',
48
+ },
49
+ 'network.fetch': {
50
+ category: null,
51
+ means: 'pull from the network; nothing of the user\'s leaves except the request',
52
+ },
53
+ 'process.spawn': {
54
+ category: null,
55
+ means: 'start a process or another agent, which then holds its own grant',
56
+ },
57
+ payments: {
58
+ category: 'costs-money',
59
+ means: 'spend money — paid API capacity or provisioned infrastructure',
60
+ },
61
+ });
62
+
63
+ /**
64
+ * A shell is a shell. Every posture an unrestricted `Bash` confers, in one list,
65
+ * so the interpreters below cannot drift away from it.
66
+ */
67
+ const FULL_SHELL = Object.freeze([
68
+ 'code.read', 'code.write', 'code.destructive',
69
+ 'credential.read', 'communication.external.send', 'network.fetch', 'process.spawn',
70
+ ]);
71
+
72
+ /**
73
+ * Bash sub-grants, by the command they scope to.
74
+ *
75
+ * `full: true` marks the ones that scope to a name but not to a capability — an
76
+ * interpreter, or a command that runs other commands. `Bash(node:*)` is
77
+ * `node -e '<anything>'`; `Bash(find:*)` is `find . -exec <anything>`;
78
+ * `Bash(xargs:*)` and `Bash(awk:*)` likewise. These read as restrictions and are
79
+ * not, which is the reason this file exists.
80
+ */
81
+ const BASH_SCOPES = Object.freeze({
82
+ // Scoped in name only — a full shell wearing a command name.
83
+ node: { full: true, why: 'node -e runs arbitrary JavaScript, including child_process' },
84
+ python3: { full: true, why: 'python3 -c runs arbitrary Python, including os.system' },
85
+ python: { full: true, why: 'python -c runs arbitrary Python, including os.system' },
86
+ xargs: { full: true, why: 'xargs exists to run other commands' },
87
+ find: { full: true, why: 'find -exec and -delete run other commands and remove files' },
88
+ awk: { full: true, why: 'awk has system() and can redirect print into a file' },
89
+ sh: { full: true, why: 'a shell' },
90
+ bash: { full: true, why: 'a shell' },
91
+ zsh: { full: true, why: 'a shell' },
92
+ env: { full: true, why: 'env runs the command that follows it' },
93
+ eval: { full: true, why: 'eval runs the string that follows it' },
94
+
95
+ // Genuinely narrower.
96
+ git: { postures: ['code.read', 'code.write', 'code.destructive', 'communication.external.send'],
97
+ why: 'push sends the tree to a remote; checkout -- and reset --hard destroy uncommitted work' },
98
+ npm: { postures: ['code.read', 'code.write', 'network.fetch', 'communication.external.send', 'process.spawn'],
99
+ why: 'install runs lifecycle scripts; publish escapes the machine' },
100
+ bd: { postures: ['code.read', 'code.write', 'communication.external.send'],
101
+ why: 'bd sync pushes the task store to its remote' },
102
+ cat: { postures: ['code.read', 'credential.read'],
103
+ why: 'the file it reads may be ~/.great_cto/secrets.env' },
104
+ source: { postures: ['code.read', 'credential.read'],
105
+ why: 'sourcing an env file puts its secrets in the environment' },
106
+ sort: { postures: ['code.read', 'code.write'], why: 'sort -o writes' },
107
+ tee: { postures: ['code.read', 'code.write'], why: 'writes what it reads' },
108
+ rm: { postures: ['code.destructive'], why: 'removes files' },
109
+
110
+ ls: { postures: ['code.read'] },
111
+ grep: { postures: ['code.read'] },
112
+ wc: { postures: ['code.read'] },
113
+ head: { postures: ['code.read'] },
114
+ tail: { postures: ['code.read'] },
115
+ date: { postures: ['code.read'] },
116
+ echo: { postures: ['code.read'] },
117
+ printf: { postures: ['code.read'] },
118
+ export: { postures: ['code.read'] },
119
+ mkdir: { postures: ['code.write'] },
120
+ touch: { postures: ['code.write'] },
121
+ });
122
+
123
+ /** Non-Bash tools. */
124
+ const TOOL_POSTURES = Object.freeze({
125
+ Read: ['code.read'],
126
+ Glob: ['code.read'],
127
+ Grep: ['code.read'],
128
+ Write: ['code.write'],
129
+ Edit: ['code.write'],
130
+ NotebookEdit: ['code.write'],
131
+ // A URL is a channel. A fetch of `https://evil/?leak=<secret>` is a send, and
132
+ // an agent that can read a file and reach the network can move it.
133
+ WebFetch: ['network.fetch', 'communication.external.send'],
134
+ WebSearch: ['network.fetch', 'communication.external.send'],
135
+ Agent: ['process.spawn'],
136
+ Task: ['process.spawn'],
137
+ });
138
+
139
+ /** MCP and beta tools, matched by prefix — the list of these grows monthly. */
140
+ const PREFIX_POSTURES = Object.freeze([
141
+ [/^advisor_/, ['network.fetch', 'communication.external.send', 'payments']],
142
+ [/^memory_/, ['code.read', 'code.write']],
143
+ [/^mcp__great_cto_llm_router__/, ['network.fetch', 'communication.external.send', 'payments']],
144
+ [/^mcp__grafana__/, ['network.fetch']],
145
+ ]);
146
+
147
+ /**
148
+ * Split a `tools:` frontmatter value into tokens. `Bash(git:*)` contains a comma
149
+ * in no case we ship, but the split is on commas outside parentheses anyway, so
150
+ * a future `Bash(a:*, b:*)` does not silently become two broken tokens.
151
+ */
152
+ export function splitTools(line) {
153
+ const out = [];
154
+ let depth = 0; let cur = '';
155
+ for (const ch of String(line ?? '')) {
156
+ if (ch === '(') depth++;
157
+ if (ch === ')') depth--;
158
+ if (ch === ',' && depth === 0) { out.push(cur.trim()); cur = ''; continue; }
159
+ cur += ch;
160
+ }
161
+ if (cur.trim()) out.push(cur.trim());
162
+ return out.filter(Boolean);
163
+ }
164
+
165
+ /**
166
+ * @returns {{postures:string[], unknown:boolean, fullShell:boolean, why:string}}
167
+ *
168
+ * THREE states for the tool itself, and `unknown` is the one that earns its
169
+ * keep: a tool this table has never heard of grants `unknown`, never nothing.
170
+ * A grant nobody classified must not read as a grant that was classified and
171
+ * found harmless.
172
+ */
173
+ export function postureOfTool(tool) {
174
+ const t = String(tool ?? '').trim();
175
+ if (!t) return { postures: [], unknown: true, fullShell: false, why: 'no tool given' };
176
+
177
+ if (t === '*' || t === 'All tools') {
178
+ return { postures: [...FULL_SHELL, 'payments'], unknown: false, fullShell: true, why: 'every tool' };
179
+ }
180
+ if (t === 'Bash') {
181
+ return { postures: [...FULL_SHELL], unknown: false, fullShell: true, why: 'unrestricted shell' };
182
+ }
183
+
184
+ const scoped = /^Bash\(([^:)]+)/.exec(t);
185
+ if (scoped) {
186
+ const cmd = scoped[1].trim();
187
+ const hit = BASH_SCOPES[cmd];
188
+ if (!hit) {
189
+ return {
190
+ postures: [], unknown: true, fullShell: false,
191
+ why: `Bash scope '${cmd}' is not in the table — treat as unjudged, not as narrow. Add it to agent-posture.mjs.`,
192
+ };
193
+ }
194
+ if (hit.full) {
195
+ return { postures: [...FULL_SHELL], unknown: false, fullShell: true, why: hit.why };
196
+ }
197
+ return { postures: [...hit.postures], unknown: false, fullShell: false, why: hit.why ?? '' };
198
+ }
199
+
200
+ if (TOOL_POSTURES[t]) {
201
+ return { postures: [...TOOL_POSTURES[t]], unknown: false, fullShell: false, why: '' };
202
+ }
203
+ for (const [re, postures] of PREFIX_POSTURES) {
204
+ if (re.test(t)) return { postures: [...postures], unknown: false, fullShell: false, why: '' };
205
+ }
206
+ return {
207
+ postures: [], unknown: true, fullShell: false,
208
+ why: `'${t}' is not in the table — treat as unjudged, not as harmless. Add it to agent-posture.mjs.`,
209
+ };
210
+ }
211
+
212
+ /**
213
+ * The posture of a whole `tools:` line.
214
+ *
215
+ * @returns {{postures:string[], expensive:string[], unknownTools:string[],
216
+ * fullShellVia:string[], scopedInNameOnly:string[]}}
217
+ */
218
+ export function postureOf(toolsLine) {
219
+ const tools = splitTools(toolsLine);
220
+ const postures = new Set();
221
+ const unknownTools = [];
222
+ const fullShellVia = [];
223
+ const scopedInNameOnly = [];
224
+
225
+ for (const t of tools) {
226
+ const r = postureOfTool(t);
227
+ if (r.unknown) { unknownTools.push(t); continue; }
228
+ for (const p of r.postures) postures.add(p);
229
+ if (r.fullShell) {
230
+ fullShellVia.push(t);
231
+ // `Bash` is honest about being a shell. `Bash(node:*)` is not.
232
+ if (t !== 'Bash' && t !== '*' && t !== 'All tools') scopedInNameOnly.push(t);
233
+ }
234
+ }
235
+
236
+ const ordered = Object.keys(POSTURES).filter((p) => postures.has(p));
237
+ return {
238
+ postures: ordered,
239
+ expensive: ordered.filter((p) => POSTURES[p].category),
240
+ unknownTools,
241
+ fullShellVia,
242
+ scopedInNameOnly,
243
+ };
244
+ }
245
+
246
+ /** One line for a human reviewing a grant, in their words. */
247
+ export function describePosture(r) {
248
+ const parts = [];
249
+ if (r.expensive.length) {
250
+ parts.push(`expensive: ${r.expensive.map((p) => `${p} (${CATEGORIES[POSTURES[p].category]})`).join('; ')}`);
251
+ } else if (r.postures.length) {
252
+ parts.push(`routine: ${r.postures.join(', ')}`);
253
+ }
254
+ if (r.scopedInNameOnly.length) {
255
+ parts.push(`scoped in name only: ${r.scopedInNameOnly.join(', ')} — a full shell`);
256
+ }
257
+ if (r.unknownTools.length) {
258
+ parts.push(`NOT CLASSIFIED: ${r.unknownTools.join(', ')} — unjudged, not harmless`);
259
+ }
260
+ return parts.join(' · ') || 'no tools granted';
261
+ }
262
+
263
+ /** Every posture name, for a surface that wants a legend. */
264
+ export function knownPostures() {
265
+ return Object.keys(POSTURES);
266
+ }
@@ -0,0 +1,236 @@
1
+ // scripts/lib/cost-meter.mjs — turn real Anthropic `usage` into real USD.
2
+ //
3
+ // Why it exists (DEEPEN-PIPELINE Wave 1, cost loop):
4
+ // cost-guard.mjs guesses with a hardcoded ROUGH_COST_USD table and
5
+ // log-verdict.sh trusts a typed CLI arg — spend is never measured. This module
6
+ // is the single place that converts an API response's token usage into dollars,
7
+ // so the runner, log-verdict, and any LLM-calling script can record TRUE cost.
8
+ //
9
+ // Prices are USD per 1,000,000 tokens (list prices). They change — override
10
+ // without editing code via either:
11
+ // GREAT_CTO_MODEL_PRICES='{"claude-opus-4-8":{"input":15,"output":75}}' (env, JSON)
12
+ // ~/.great_cto/model-prices.json (file, JSON)
13
+ //
14
+ // Pure + offline-testable: priceForModel() and costForUsage() take an explicit
15
+ // `prices` arg so unit tests never touch env or disk.
16
+
17
+ import { readFileSync } from 'node:fs';
18
+ import { homedir } from 'node:os';
19
+ import { join } from 'node:path';
20
+
21
+ /**
22
+ * Default list prices, USD per 1M tokens.
23
+ *
24
+ * Anthropic rates below are first-party API list prices as of 2026-06-24. They
25
+ * also apply to Claude on Microsoft Foundry; Bedrock and Vertex are partner-
26
+ * operated with separate pricing — override there.
27
+ *
28
+ * Why the current models are listed explicitly rather than left to the family
29
+ * fallback: the fallback bills anything matching /opus/i at the Opus 4 rate, and
30
+ * Opus 5 is $5/$25, not $15/$75. Every Opus 5 turn on this machine — 8,763 of
31
+ * them in one session — was being costed at THREE TIMES its real price, and the
32
+ * total looked like a total. A guess that silently triples the number is worse
33
+ * than no number, because it is spendable.
34
+ *
35
+ * Keep this list ahead of the fallback. A model that reaches the fallback is
36
+ * reported as `assumed` by priceUsage(); one that reaches neither is reported as
37
+ * unpriced rather than free.
38
+ */
39
+ export const DEFAULT_PRICES = {
40
+ // Claude 5 family
41
+ 'claude-fable-5': { input: 10, output: 50 },
42
+ 'claude-mythos-5': { input: 10, output: 50 },
43
+ 'claude-opus-5': { input: 5, output: 25 },
44
+ 'claude-sonnet-5': { input: 2, output: 10 },
45
+ // Claude 4.6–4.8
46
+ 'claude-opus-4-8': { input: 5, output: 25 },
47
+ 'claude-opus-4-7': { input: 5, output: 25 },
48
+ 'claude-opus-4-6': { input: 5, output: 25 },
49
+ 'claude-sonnet-4-6': { input: 3, output: 15 },
50
+ 'claude-haiku-4-5': { input: 1, output: 5 },
51
+ // Claude 4.x family (bare ids; OpenRouter "anthropic/<id>" slugs resolve via prefix-strip)
52
+ 'claude-opus-4': { input: 15, output: 75 },
53
+ 'claude-sonnet-4': { input: 3, output: 15 },
54
+ 'claude-haiku-4': { input: 0.8, output: 4 },
55
+ // Claude 3.x (still referenced by some evals/agents)
56
+ 'claude-3-5-sonnet': { input: 3, output: 15 },
57
+ 'claude-3-5-haiku': { input: 0.8, output: 4 },
58
+ 'claude-3-opus': { input: 15, output: 75 },
59
+ // OpenRouter non-Anthropic slugs the project routes to (approx list prices —
60
+ // override via ~/.great_cto/model-prices.json or GREAT_CTO_MODEL_PRICES).
61
+ 'moonshotai/kimi-k2': { input: 0.55, output: 2.2 },
62
+ 'moonshotai/kimi-k3': { input: 3, output: 15 },
63
+ // Read from OpenRouter's /models on 2026-08-27, not guessed. Until now these
64
+ // reported `priced: false` — correctly, and that honesty is why an eval run on
65
+ // glm-5.3-flash showed $0.000: not free, unpriced. Now they are priced exactly.
66
+ 'z-ai/glm-5.3-flash': { input: 0.075, output: 0.25 },
67
+ 'z-ai/glm-5.3': { input: 1.4, output: 4.4 },
68
+ };
69
+
70
+ /**
71
+ * NOT modelled, and named here so it is a known gap rather than a silent one:
72
+ * Opus 5 fast mode bills at $10/$50 instead of $5/$25. Turns carry the rate they
73
+ * ran at in `usage.speed`, which this module does not read — a fast-mode turn is
74
+ * therefore under-costed by 2×. Wire it when a transcript in the wild shows
75
+ * `speed: "fast"`; until then the figure is right for standard turns and low for
76
+ * fast ones, which is the direction that does not create false confidence.
77
+ */
78
+ export const UNMODELLED_RATES = Object.freeze(['opus-5 fast mode ($10/$50)']);
79
+
80
+ /** Load price overrides from env (preferred) then ~/.great_cto/model-prices.json. */
81
+ export function loadPriceOverrides() {
82
+ try {
83
+ if (process.env.GREAT_CTO_MODEL_PRICES) return JSON.parse(process.env.GREAT_CTO_MODEL_PRICES);
84
+ } catch { /* malformed env JSON → ignore */ }
85
+ try {
86
+ return JSON.parse(readFileSync(join(homedir(), '.great_cto', 'model-prices.json'), 'utf8'));
87
+ } catch { /* no override file → ignore */ }
88
+ return {};
89
+ }
90
+
91
+ /** Effective price table = defaults merged with overrides. */
92
+ export function effectivePrices() {
93
+ return { ...DEFAULT_PRICES, ...loadPriceOverrides() };
94
+ }
95
+
96
+ /**
97
+ * Resolve a per-MTok price for a model id.
98
+ * 1. exact key match
99
+ * 2. longest prefix match (so "claude-opus-4-8-2026..." → "claude-opus-4")
100
+ * 3. family heuristic on /opus|sonnet|haiku/
101
+ * Returns { input, output } in USD/MTok, or null if unknown.
102
+ */
103
+ export function priceForModel(model, prices = effectivePrices()) {
104
+ if (!model) return null;
105
+ if (prices[model]) return prices[model]; // exact (incl. full OpenRouter slug)
106
+
107
+ // Strip a leading "provider/" segment so OpenRouter slugs like
108
+ // "anthropic/claude-sonnet-4" resolve to the bare "claude-sonnet-4" key.
109
+ const bare = model.includes('/') ? model.slice(model.indexOf('/') + 1) : model;
110
+ if (prices[bare]) return prices[bare];
111
+
112
+ let best = null, bestLen = 0;
113
+ for (const k of Object.keys(prices)) {
114
+ if (bare.startsWith(k) && k.length > bestLen) { best = prices[k]; bestLen = k.length; }
115
+ }
116
+ if (best) return best;
117
+
118
+ if (/opus/i.test(model)) return prices['claude-opus-4'] || { input: 15, output: 75 };
119
+ if (/sonnet/i.test(model)) return prices['claude-sonnet-4'] || { input: 3, output: 15 };
120
+ if (/haiku/i.test(model)) return prices['claude-haiku-4'] || { input: 0.8, output: 4 };
121
+ return null;
122
+ }
123
+
124
+ /**
125
+ * The same lookup, but it says HOW it found the price.
126
+ *
127
+ * `priceForModel` answers with a number or null, and both callers and readers
128
+ * then treat "priced exactly" and "priced by guessing the family" as the same
129
+ * thing. They are not. `claude-opus-5` is billed here at Opus 4's rate because
130
+ * its name contains "opus" — a guess that may be right and is not a fact, and a
131
+ * total built from it should be able to say so.
132
+ *
133
+ * @returns {{price: {input:number,output:number}|null, source: 'exact'|'bare'|'prefix'|'family'|'none'}}
134
+ */
135
+ export function resolvePrice(model, prices = effectivePrices()) {
136
+ if (!model) return { price: null, source: 'none' };
137
+ if (prices[model]) return { price: prices[model], source: 'exact' };
138
+
139
+ const bare = model.includes('/') ? model.slice(model.indexOf('/') + 1) : model;
140
+ if (prices[bare]) return { price: prices[bare], source: 'bare' };
141
+
142
+ let best = null, bestLen = 0;
143
+ for (const k of Object.keys(prices)) {
144
+ if (bare.startsWith(k) && k.length > bestLen) { best = prices[k]; bestLen = k.length; }
145
+ }
146
+ if (best) return { price: best, source: 'prefix' };
147
+
148
+ if (/opus/i.test(model)) return { price: prices['claude-opus-4'] || { input: 15, output: 75 }, source: 'family' };
149
+ if (/sonnet/i.test(model)) return { price: prices['claude-sonnet-4'] || { input: 3, output: 15 }, source: 'family' };
150
+ if (/haiku/i.test(model)) return { price: prices['claude-haiku-4'] || { input: 0.8, output: 4 }, source: 'family' };
151
+ return { price: null, source: 'none' };
152
+ }
153
+
154
+ /**
155
+ * Cost of one call, with the third state kept.
156
+ *
157
+ * `costForUsage` returns a number, so an unknown model has to come back as 0 —
158
+ * and a model nobody has priced then reads as a model that costs nothing.
159
+ * 172 turns of `claude-fable-5` were billed at $0.00 for exactly that reason,
160
+ * and the total looked like a total rather than a total plus a hole.
161
+ *
162
+ * @returns {{usd:number, priced:boolean, assumed:boolean, source:string, model:string}}
163
+ */
164
+ export function priceUsage({ model, usage, prices }) {
165
+ const { price, source } = resolvePrice(model, prices || effectivePrices());
166
+ if (!usage || !price) {
167
+ return { usd: 0, priced: false, assumed: false, source, model: model || '' };
168
+ }
169
+ const inTok = usage.input_tokens || 0;
170
+ const outTok = usage.output_tokens || 0;
171
+ const cacheWrite = usage.cache_creation_input_tokens || 0;
172
+ const cacheRead = usage.cache_read_input_tokens || 0;
173
+ const usd = (inTok * price.input + outTok * price.output
174
+ + cacheWrite * price.input * 1.25 + cacheRead * price.input * 0.1) / 1_000_000;
175
+ return { usd, priced: true, assumed: source === 'family', source, model: model || '' };
176
+ }
177
+
178
+ /**
179
+ * Dollar cost of a single API call.
180
+ * @param {object} opts
181
+ * @param {string} opts.model
182
+ * @param {{input_tokens?:number, output_tokens?:number}} opts.usage Anthropic response.usage
183
+ * @param {object} [opts.prices] override table (for tests)
184
+ * @returns {number} USD (0 if usage or price unknown)
185
+ */
186
+ export function costForUsage({ model, usage, prices }) {
187
+ if (!usage) return 0;
188
+ const p = priceForModel(model, prices);
189
+ if (!p) return 0;
190
+ const inTok = usage.input_tokens || 0;
191
+ const outTok = usage.output_tokens || 0;
192
+ // Prompt-caching tokens bill at Anthropic's standard multipliers off the base
193
+ // input price: cache WRITE = 1.25× input, cache READ = 0.1× input. Ignoring
194
+ // them under-counts real spend badly (a cached turn is often 50k+ cache tokens
195
+ // vs a few hundred fresh input tokens).
196
+ const cacheWrite = usage.cache_creation_input_tokens || 0;
197
+ const cacheRead = usage.cache_read_input_tokens || 0;
198
+ return (inTok * p.input + outTok * p.output
199
+ + cacheWrite * p.input * 1.25 + cacheRead * p.input * 0.1) / 1_000_000;
200
+ }
201
+
202
+ export function round4(n) { return Math.round(n * 10000) / 10000; }
203
+
204
+ // ── CLI: compute one cost from args/env (used by log-verdict.sh `auto` mode) ──
205
+ // node scripts/lib/cost-meter.mjs --model M --in 1234 --out 567
206
+ // prints the USD number (4 dp) to stdout.
207
+ function main(argv) {
208
+ let model = process.env.LLM_MODEL || '';
209
+ let inTok = parseInt(process.env.LLM_INPUT_TOKENS || '0', 10) || 0;
210
+ let outTok = parseInt(process.env.LLM_OUTPUT_TOKENS || '0', 10) || 0;
211
+ for (let i = 0; i < argv.length; i++) {
212
+ if (argv[i] === '--model' && argv[i + 1]) model = argv[++i];
213
+ else if (argv[i] === '--in' && argv[i + 1]) inTok = parseInt(argv[++i], 10) || 0;
214
+ else if (argv[i] === '--out' && argv[i + 1]) outTok = parseInt(argv[++i], 10) || 0;
215
+ }
216
+ // No tokens means the caller never had a usage block to hand us — a
217
+ // measurement that did not happen. Printing 0 here made every verdict in the
218
+ // fleet record a MEASURED zero: the portfolio reported $0.00 spend for twelve
219
+ // projects, and a per-agent budget would have read "spent $0.00 of $25,
220
+ // measured" forever. All 35 agents pass `auto`, so this was every verdict.
221
+ //
222
+ // Exit 2 with nothing on stdout. `log-verdict.sh` omits the field, and every
223
+ // reader downstream already distinguishes an absent cost from a zero one —
224
+ // portfolio.mjs calls it "spend nobody recorded", agent-budget.mjs calls it
225
+ // `unmeasured`. They were right and had nothing to be right about.
226
+ if (inTok <= 0 && outTok <= 0) {
227
+ process.stderr.write('cost-meter: no token usage supplied — cost not measured\n');
228
+ return process.exit(2);
229
+ }
230
+ const cost = costForUsage({ model, usage: { input_tokens: inTok, output_tokens: outTok } });
231
+ process.stdout.write(String(round4(cost)));
232
+ }
233
+
234
+ import { fileURLToPath } from 'node:url';
235
+ const isMain = process.argv[1] && fileURLToPath(import.meta.url) === process.argv[1];
236
+ if (isMain) main(process.argv.slice(2));