great-cto 3.26.2 → 3.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/board/.claude-plugin/plugin.json +1 -1
- package/board/packages/board/lib/fleet.mjs +43 -1
- package/board/packages/board/lib/routes.mjs +147 -9
- package/board/packages/board/lib/view-counter.mjs +122 -0
- package/board/packages/board/public/index.html +946 -545
- package/board/scripts/lib/agent-posture.mjs +266 -0
- package/board/scripts/lib/cost-meter.mjs +236 -0
- package/board/scripts/lib/cross-model-review.mjs +225 -0
- package/board/scripts/lib/provider-exhaustion.mjs +152 -0
- package/package.json +1 -1
|
@@ -0,0 +1,266 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* agent-posture — what does this agent's tool grant actually LET IT DO?
|
|
3
|
+
*
|
|
4
|
+
* ADR-009 ends with an instruction nobody had a way to follow: "Ask at design
|
|
5
|
+
* time, when the capability is added — not after the incident." The question it
|
|
6
|
+
* asks is about consequence — is this expensive to undo? — while every agent
|
|
7
|
+
* file answers a different question, in a different language: which tools are on
|
|
8
|
+
* the `tools:` line. Reviewing the second does not answer the first. `Bash` and
|
|
9
|
+
* `Bash(node:*)` sit two characters apart and read as careful and careless; in
|
|
10
|
+
* consequence they are the same grant.
|
|
11
|
+
*
|
|
12
|
+
* So this names the grant in the language of the decision. Vocabulary shape
|
|
13
|
+
* borrowed from OpenFirma's capability postures (`credential.read`,
|
|
14
|
+
* `communication.external.send`, `code.destructive`) — the idea only; that
|
|
15
|
+
* project is GPL-3.0 and this one is MIT, so nothing was copied.
|
|
16
|
+
*
|
|
17
|
+
* NOTHING here decides anything, exactly as in `gate-reversibility.mjs`, whose
|
|
18
|
+
* ADR-009 categories it reuses rather than inventing a second vocabulary for the
|
|
19
|
+
* same axis. It classifies, so a reviewer can see the grant they are approving.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import { CATEGORIES } from './gate-reversibility.mjs';
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* The postures. Each cites the ADR-009 category that makes it expensive, or
|
|
26
|
+
* `null` when the repair is simply to do the thing again.
|
|
27
|
+
*/
|
|
28
|
+
export const POSTURES = Object.freeze({
|
|
29
|
+
'code.read': {
|
|
30
|
+
category: null,
|
|
31
|
+
means: 'read files in the working tree',
|
|
32
|
+
},
|
|
33
|
+
'code.write': {
|
|
34
|
+
category: null,
|
|
35
|
+
means: 'create or modify files — the repair is another edit',
|
|
36
|
+
},
|
|
37
|
+
'code.destructive': {
|
|
38
|
+
category: 'destroys-evidence',
|
|
39
|
+
means: 'delete files, rewrite history, or overwrite work that is not in the index',
|
|
40
|
+
},
|
|
41
|
+
'credential.read': {
|
|
42
|
+
category: 'unrevocable-disclosure',
|
|
43
|
+
means: 'reach secrets on disk or in the environment',
|
|
44
|
+
},
|
|
45
|
+
'communication.external.send': {
|
|
46
|
+
category: 'escapes-the-machine',
|
|
47
|
+
means: 'send data off this machine — a push, a publish, a request body, a URL',
|
|
48
|
+
},
|
|
49
|
+
'network.fetch': {
|
|
50
|
+
category: null,
|
|
51
|
+
means: 'pull from the network; nothing of the user\'s leaves except the request',
|
|
52
|
+
},
|
|
53
|
+
'process.spawn': {
|
|
54
|
+
category: null,
|
|
55
|
+
means: 'start a process or another agent, which then holds its own grant',
|
|
56
|
+
},
|
|
57
|
+
payments: {
|
|
58
|
+
category: 'costs-money',
|
|
59
|
+
means: 'spend money — paid API capacity or provisioned infrastructure',
|
|
60
|
+
},
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* A shell is a shell. Every posture an unrestricted `Bash` confers, in one list,
|
|
65
|
+
* so the interpreters below cannot drift away from it.
|
|
66
|
+
*/
|
|
67
|
+
const FULL_SHELL = Object.freeze([
|
|
68
|
+
'code.read', 'code.write', 'code.destructive',
|
|
69
|
+
'credential.read', 'communication.external.send', 'network.fetch', 'process.spawn',
|
|
70
|
+
]);
|
|
71
|
+
|
|
72
|
+
/**
|
|
73
|
+
* Bash sub-grants, by the command they scope to.
|
|
74
|
+
*
|
|
75
|
+
* `full: true` marks the ones that scope to a name but not to a capability — an
|
|
76
|
+
* interpreter, or a command that runs other commands. `Bash(node:*)` is
|
|
77
|
+
* `node -e '<anything>'`; `Bash(find:*)` is `find . -exec <anything>`;
|
|
78
|
+
* `Bash(xargs:*)` and `Bash(awk:*)` likewise. These read as restrictions and are
|
|
79
|
+
* not, which is the reason this file exists.
|
|
80
|
+
*/
|
|
81
|
+
const BASH_SCOPES = Object.freeze({
|
|
82
|
+
// Scoped in name only — a full shell wearing a command name.
|
|
83
|
+
node: { full: true, why: 'node -e runs arbitrary JavaScript, including child_process' },
|
|
84
|
+
python3: { full: true, why: 'python3 -c runs arbitrary Python, including os.system' },
|
|
85
|
+
python: { full: true, why: 'python -c runs arbitrary Python, including os.system' },
|
|
86
|
+
xargs: { full: true, why: 'xargs exists to run other commands' },
|
|
87
|
+
find: { full: true, why: 'find -exec and -delete run other commands and remove files' },
|
|
88
|
+
awk: { full: true, why: 'awk has system() and can redirect print into a file' },
|
|
89
|
+
sh: { full: true, why: 'a shell' },
|
|
90
|
+
bash: { full: true, why: 'a shell' },
|
|
91
|
+
zsh: { full: true, why: 'a shell' },
|
|
92
|
+
env: { full: true, why: 'env runs the command that follows it' },
|
|
93
|
+
eval: { full: true, why: 'eval runs the string that follows it' },
|
|
94
|
+
|
|
95
|
+
// Genuinely narrower.
|
|
96
|
+
git: { postures: ['code.read', 'code.write', 'code.destructive', 'communication.external.send'],
|
|
97
|
+
why: 'push sends the tree to a remote; checkout -- and reset --hard destroy uncommitted work' },
|
|
98
|
+
npm: { postures: ['code.read', 'code.write', 'network.fetch', 'communication.external.send', 'process.spawn'],
|
|
99
|
+
why: 'install runs lifecycle scripts; publish escapes the machine' },
|
|
100
|
+
bd: { postures: ['code.read', 'code.write', 'communication.external.send'],
|
|
101
|
+
why: 'bd sync pushes the task store to its remote' },
|
|
102
|
+
cat: { postures: ['code.read', 'credential.read'],
|
|
103
|
+
why: 'the file it reads may be ~/.great_cto/secrets.env' },
|
|
104
|
+
source: { postures: ['code.read', 'credential.read'],
|
|
105
|
+
why: 'sourcing an env file puts its secrets in the environment' },
|
|
106
|
+
sort: { postures: ['code.read', 'code.write'], why: 'sort -o writes' },
|
|
107
|
+
tee: { postures: ['code.read', 'code.write'], why: 'writes what it reads' },
|
|
108
|
+
rm: { postures: ['code.destructive'], why: 'removes files' },
|
|
109
|
+
|
|
110
|
+
ls: { postures: ['code.read'] },
|
|
111
|
+
grep: { postures: ['code.read'] },
|
|
112
|
+
wc: { postures: ['code.read'] },
|
|
113
|
+
head: { postures: ['code.read'] },
|
|
114
|
+
tail: { postures: ['code.read'] },
|
|
115
|
+
date: { postures: ['code.read'] },
|
|
116
|
+
echo: { postures: ['code.read'] },
|
|
117
|
+
printf: { postures: ['code.read'] },
|
|
118
|
+
export: { postures: ['code.read'] },
|
|
119
|
+
mkdir: { postures: ['code.write'] },
|
|
120
|
+
touch: { postures: ['code.write'] },
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
/** Non-Bash tools. */
|
|
124
|
+
const TOOL_POSTURES = Object.freeze({
|
|
125
|
+
Read: ['code.read'],
|
|
126
|
+
Glob: ['code.read'],
|
|
127
|
+
Grep: ['code.read'],
|
|
128
|
+
Write: ['code.write'],
|
|
129
|
+
Edit: ['code.write'],
|
|
130
|
+
NotebookEdit: ['code.write'],
|
|
131
|
+
// A URL is a channel. A fetch of `https://evil/?leak=<secret>` is a send, and
|
|
132
|
+
// an agent that can read a file and reach the network can move it.
|
|
133
|
+
WebFetch: ['network.fetch', 'communication.external.send'],
|
|
134
|
+
WebSearch: ['network.fetch', 'communication.external.send'],
|
|
135
|
+
Agent: ['process.spawn'],
|
|
136
|
+
Task: ['process.spawn'],
|
|
137
|
+
});
|
|
138
|
+
|
|
139
|
+
/** MCP and beta tools, matched by prefix — the list of these grows monthly. */
|
|
140
|
+
const PREFIX_POSTURES = Object.freeze([
|
|
141
|
+
[/^advisor_/, ['network.fetch', 'communication.external.send', 'payments']],
|
|
142
|
+
[/^memory_/, ['code.read', 'code.write']],
|
|
143
|
+
[/^mcp__great_cto_llm_router__/, ['network.fetch', 'communication.external.send', 'payments']],
|
|
144
|
+
[/^mcp__grafana__/, ['network.fetch']],
|
|
145
|
+
]);
|
|
146
|
+
|
|
147
|
+
/**
|
|
148
|
+
* Split a `tools:` frontmatter value into tokens. `Bash(git:*)` contains a comma
|
|
149
|
+
* in no case we ship, but the split is on commas outside parentheses anyway, so
|
|
150
|
+
* a future `Bash(a:*, b:*)` does not silently become two broken tokens.
|
|
151
|
+
*/
|
|
152
|
+
export function splitTools(line) {
|
|
153
|
+
const out = [];
|
|
154
|
+
let depth = 0; let cur = '';
|
|
155
|
+
for (const ch of String(line ?? '')) {
|
|
156
|
+
if (ch === '(') depth++;
|
|
157
|
+
if (ch === ')') depth--;
|
|
158
|
+
if (ch === ',' && depth === 0) { out.push(cur.trim()); cur = ''; continue; }
|
|
159
|
+
cur += ch;
|
|
160
|
+
}
|
|
161
|
+
if (cur.trim()) out.push(cur.trim());
|
|
162
|
+
return out.filter(Boolean);
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/**
|
|
166
|
+
* @returns {{postures:string[], unknown:boolean, fullShell:boolean, why:string}}
|
|
167
|
+
*
|
|
168
|
+
* THREE states for the tool itself, and `unknown` is the one that earns its
|
|
169
|
+
* keep: a tool this table has never heard of grants `unknown`, never nothing.
|
|
170
|
+
* A grant nobody classified must not read as a grant that was classified and
|
|
171
|
+
* found harmless.
|
|
172
|
+
*/
|
|
173
|
+
export function postureOfTool(tool) {
|
|
174
|
+
const t = String(tool ?? '').trim();
|
|
175
|
+
if (!t) return { postures: [], unknown: true, fullShell: false, why: 'no tool given' };
|
|
176
|
+
|
|
177
|
+
if (t === '*' || t === 'All tools') {
|
|
178
|
+
return { postures: [...FULL_SHELL, 'payments'], unknown: false, fullShell: true, why: 'every tool' };
|
|
179
|
+
}
|
|
180
|
+
if (t === 'Bash') {
|
|
181
|
+
return { postures: [...FULL_SHELL], unknown: false, fullShell: true, why: 'unrestricted shell' };
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
const scoped = /^Bash\(([^:)]+)/.exec(t);
|
|
185
|
+
if (scoped) {
|
|
186
|
+
const cmd = scoped[1].trim();
|
|
187
|
+
const hit = BASH_SCOPES[cmd];
|
|
188
|
+
if (!hit) {
|
|
189
|
+
return {
|
|
190
|
+
postures: [], unknown: true, fullShell: false,
|
|
191
|
+
why: `Bash scope '${cmd}' is not in the table — treat as unjudged, not as narrow. Add it to agent-posture.mjs.`,
|
|
192
|
+
};
|
|
193
|
+
}
|
|
194
|
+
if (hit.full) {
|
|
195
|
+
return { postures: [...FULL_SHELL], unknown: false, fullShell: true, why: hit.why };
|
|
196
|
+
}
|
|
197
|
+
return { postures: [...hit.postures], unknown: false, fullShell: false, why: hit.why ?? '' };
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
if (TOOL_POSTURES[t]) {
|
|
201
|
+
return { postures: [...TOOL_POSTURES[t]], unknown: false, fullShell: false, why: '' };
|
|
202
|
+
}
|
|
203
|
+
for (const [re, postures] of PREFIX_POSTURES) {
|
|
204
|
+
if (re.test(t)) return { postures: [...postures], unknown: false, fullShell: false, why: '' };
|
|
205
|
+
}
|
|
206
|
+
return {
|
|
207
|
+
postures: [], unknown: true, fullShell: false,
|
|
208
|
+
why: `'${t}' is not in the table — treat as unjudged, not as harmless. Add it to agent-posture.mjs.`,
|
|
209
|
+
};
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
/**
|
|
213
|
+
* The posture of a whole `tools:` line.
|
|
214
|
+
*
|
|
215
|
+
* @returns {{postures:string[], expensive:string[], unknownTools:string[],
|
|
216
|
+
* fullShellVia:string[], scopedInNameOnly:string[]}}
|
|
217
|
+
*/
|
|
218
|
+
export function postureOf(toolsLine) {
|
|
219
|
+
const tools = splitTools(toolsLine);
|
|
220
|
+
const postures = new Set();
|
|
221
|
+
const unknownTools = [];
|
|
222
|
+
const fullShellVia = [];
|
|
223
|
+
const scopedInNameOnly = [];
|
|
224
|
+
|
|
225
|
+
for (const t of tools) {
|
|
226
|
+
const r = postureOfTool(t);
|
|
227
|
+
if (r.unknown) { unknownTools.push(t); continue; }
|
|
228
|
+
for (const p of r.postures) postures.add(p);
|
|
229
|
+
if (r.fullShell) {
|
|
230
|
+
fullShellVia.push(t);
|
|
231
|
+
// `Bash` is honest about being a shell. `Bash(node:*)` is not.
|
|
232
|
+
if (t !== 'Bash' && t !== '*' && t !== 'All tools') scopedInNameOnly.push(t);
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
const ordered = Object.keys(POSTURES).filter((p) => postures.has(p));
|
|
237
|
+
return {
|
|
238
|
+
postures: ordered,
|
|
239
|
+
expensive: ordered.filter((p) => POSTURES[p].category),
|
|
240
|
+
unknownTools,
|
|
241
|
+
fullShellVia,
|
|
242
|
+
scopedInNameOnly,
|
|
243
|
+
};
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
/** One line for a human reviewing a grant, in their words. */
|
|
247
|
+
export function describePosture(r) {
|
|
248
|
+
const parts = [];
|
|
249
|
+
if (r.expensive.length) {
|
|
250
|
+
parts.push(`expensive: ${r.expensive.map((p) => `${p} (${CATEGORIES[POSTURES[p].category]})`).join('; ')}`);
|
|
251
|
+
} else if (r.postures.length) {
|
|
252
|
+
parts.push(`routine: ${r.postures.join(', ')}`);
|
|
253
|
+
}
|
|
254
|
+
if (r.scopedInNameOnly.length) {
|
|
255
|
+
parts.push(`scoped in name only: ${r.scopedInNameOnly.join(', ')} — a full shell`);
|
|
256
|
+
}
|
|
257
|
+
if (r.unknownTools.length) {
|
|
258
|
+
parts.push(`NOT CLASSIFIED: ${r.unknownTools.join(', ')} — unjudged, not harmless`);
|
|
259
|
+
}
|
|
260
|
+
return parts.join(' · ') || 'no tools granted';
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
/** Every posture name, for a surface that wants a legend. */
|
|
264
|
+
export function knownPostures() {
|
|
265
|
+
return Object.keys(POSTURES);
|
|
266
|
+
}
|
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
// scripts/lib/cost-meter.mjs — turn real Anthropic `usage` into real USD.
|
|
2
|
+
//
|
|
3
|
+
// Why it exists (DEEPEN-PIPELINE Wave 1, cost loop):
|
|
4
|
+
// cost-guard.mjs guesses with a hardcoded ROUGH_COST_USD table and
|
|
5
|
+
// log-verdict.sh trusts a typed CLI arg — spend is never measured. This module
|
|
6
|
+
// is the single place that converts an API response's token usage into dollars,
|
|
7
|
+
// so the runner, log-verdict, and any LLM-calling script can record TRUE cost.
|
|
8
|
+
//
|
|
9
|
+
// Prices are USD per 1,000,000 tokens (list prices). They change — override
|
|
10
|
+
// without editing code via either:
|
|
11
|
+
// GREAT_CTO_MODEL_PRICES='{"claude-opus-4-8":{"input":15,"output":75}}' (env, JSON)
|
|
12
|
+
// ~/.great_cto/model-prices.json (file, JSON)
|
|
13
|
+
//
|
|
14
|
+
// Pure + offline-testable: priceForModel() and costForUsage() take an explicit
|
|
15
|
+
// `prices` arg so unit tests never touch env or disk.
|
|
16
|
+
|
|
17
|
+
import { readFileSync } from 'node:fs';
|
|
18
|
+
import { homedir } from 'node:os';
|
|
19
|
+
import { join } from 'node:path';
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Default list prices, USD per 1M tokens.
|
|
23
|
+
*
|
|
24
|
+
* Anthropic rates below are first-party API list prices as of 2026-06-24. They
|
|
25
|
+
* also apply to Claude on Microsoft Foundry; Bedrock and Vertex are partner-
|
|
26
|
+
* operated with separate pricing — override there.
|
|
27
|
+
*
|
|
28
|
+
* Why the current models are listed explicitly rather than left to the family
|
|
29
|
+
* fallback: the fallback bills anything matching /opus/i at the Opus 4 rate, and
|
|
30
|
+
* Opus 5 is $5/$25, not $15/$75. Every Opus 5 turn on this machine — 8,763 of
|
|
31
|
+
* them in one session — was being costed at THREE TIMES its real price, and the
|
|
32
|
+
* total looked like a total. A guess that silently triples the number is worse
|
|
33
|
+
* than no number, because it is spendable.
|
|
34
|
+
*
|
|
35
|
+
* Keep this list ahead of the fallback. A model that reaches the fallback is
|
|
36
|
+
* reported as `assumed` by priceUsage(); one that reaches neither is reported as
|
|
37
|
+
* unpriced rather than free.
|
|
38
|
+
*/
|
|
39
|
+
export const DEFAULT_PRICES = {
|
|
40
|
+
// Claude 5 family
|
|
41
|
+
'claude-fable-5': { input: 10, output: 50 },
|
|
42
|
+
'claude-mythos-5': { input: 10, output: 50 },
|
|
43
|
+
'claude-opus-5': { input: 5, output: 25 },
|
|
44
|
+
'claude-sonnet-5': { input: 2, output: 10 },
|
|
45
|
+
// Claude 4.6–4.8
|
|
46
|
+
'claude-opus-4-8': { input: 5, output: 25 },
|
|
47
|
+
'claude-opus-4-7': { input: 5, output: 25 },
|
|
48
|
+
'claude-opus-4-6': { input: 5, output: 25 },
|
|
49
|
+
'claude-sonnet-4-6': { input: 3, output: 15 },
|
|
50
|
+
'claude-haiku-4-5': { input: 1, output: 5 },
|
|
51
|
+
// Claude 4.x family (bare ids; OpenRouter "anthropic/<id>" slugs resolve via prefix-strip)
|
|
52
|
+
'claude-opus-4': { input: 15, output: 75 },
|
|
53
|
+
'claude-sonnet-4': { input: 3, output: 15 },
|
|
54
|
+
'claude-haiku-4': { input: 0.8, output: 4 },
|
|
55
|
+
// Claude 3.x (still referenced by some evals/agents)
|
|
56
|
+
'claude-3-5-sonnet': { input: 3, output: 15 },
|
|
57
|
+
'claude-3-5-haiku': { input: 0.8, output: 4 },
|
|
58
|
+
'claude-3-opus': { input: 15, output: 75 },
|
|
59
|
+
// OpenRouter non-Anthropic slugs the project routes to (approx list prices —
|
|
60
|
+
// override via ~/.great_cto/model-prices.json or GREAT_CTO_MODEL_PRICES).
|
|
61
|
+
'moonshotai/kimi-k2': { input: 0.55, output: 2.2 },
|
|
62
|
+
'moonshotai/kimi-k3': { input: 3, output: 15 },
|
|
63
|
+
// Read from OpenRouter's /models on 2026-08-27, not guessed. Until now these
|
|
64
|
+
// reported `priced: false` — correctly, and that honesty is why an eval run on
|
|
65
|
+
// glm-5.3-flash showed $0.000: not free, unpriced. Now they are priced exactly.
|
|
66
|
+
'z-ai/glm-5.3-flash': { input: 0.075, output: 0.25 },
|
|
67
|
+
'z-ai/glm-5.3': { input: 1.4, output: 4.4 },
|
|
68
|
+
};
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* NOT modelled, and named here so it is a known gap rather than a silent one:
|
|
72
|
+
* Opus 5 fast mode bills at $10/$50 instead of $5/$25. Turns carry the rate they
|
|
73
|
+
* ran at in `usage.speed`, which this module does not read — a fast-mode turn is
|
|
74
|
+
* therefore under-costed by 2×. Wire it when a transcript in the wild shows
|
|
75
|
+
* `speed: "fast"`; until then the figure is right for standard turns and low for
|
|
76
|
+
* fast ones, which is the direction that does not create false confidence.
|
|
77
|
+
*/
|
|
78
|
+
export const UNMODELLED_RATES = Object.freeze(['opus-5 fast mode ($10/$50)']);
|
|
79
|
+
|
|
80
|
+
/** Load price overrides from env (preferred) then ~/.great_cto/model-prices.json. */
|
|
81
|
+
export function loadPriceOverrides() {
|
|
82
|
+
try {
|
|
83
|
+
if (process.env.GREAT_CTO_MODEL_PRICES) return JSON.parse(process.env.GREAT_CTO_MODEL_PRICES);
|
|
84
|
+
} catch { /* malformed env JSON → ignore */ }
|
|
85
|
+
try {
|
|
86
|
+
return JSON.parse(readFileSync(join(homedir(), '.great_cto', 'model-prices.json'), 'utf8'));
|
|
87
|
+
} catch { /* no override file → ignore */ }
|
|
88
|
+
return {};
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/** Effective price table = defaults merged with overrides. */
|
|
92
|
+
export function effectivePrices() {
|
|
93
|
+
return { ...DEFAULT_PRICES, ...loadPriceOverrides() };
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* Resolve a per-MTok price for a model id.
|
|
98
|
+
* 1. exact key match
|
|
99
|
+
* 2. longest prefix match (so "claude-opus-4-8-2026..." → "claude-opus-4")
|
|
100
|
+
* 3. family heuristic on /opus|sonnet|haiku/
|
|
101
|
+
* Returns { input, output } in USD/MTok, or null if unknown.
|
|
102
|
+
*/
|
|
103
|
+
export function priceForModel(model, prices = effectivePrices()) {
|
|
104
|
+
if (!model) return null;
|
|
105
|
+
if (prices[model]) return prices[model]; // exact (incl. full OpenRouter slug)
|
|
106
|
+
|
|
107
|
+
// Strip a leading "provider/" segment so OpenRouter slugs like
|
|
108
|
+
// "anthropic/claude-sonnet-4" resolve to the bare "claude-sonnet-4" key.
|
|
109
|
+
const bare = model.includes('/') ? model.slice(model.indexOf('/') + 1) : model;
|
|
110
|
+
if (prices[bare]) return prices[bare];
|
|
111
|
+
|
|
112
|
+
let best = null, bestLen = 0;
|
|
113
|
+
for (const k of Object.keys(prices)) {
|
|
114
|
+
if (bare.startsWith(k) && k.length > bestLen) { best = prices[k]; bestLen = k.length; }
|
|
115
|
+
}
|
|
116
|
+
if (best) return best;
|
|
117
|
+
|
|
118
|
+
if (/opus/i.test(model)) return prices['claude-opus-4'] || { input: 15, output: 75 };
|
|
119
|
+
if (/sonnet/i.test(model)) return prices['claude-sonnet-4'] || { input: 3, output: 15 };
|
|
120
|
+
if (/haiku/i.test(model)) return prices['claude-haiku-4'] || { input: 0.8, output: 4 };
|
|
121
|
+
return null;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* The same lookup, but it says HOW it found the price.
|
|
126
|
+
*
|
|
127
|
+
* `priceForModel` answers with a number or null, and both callers and readers
|
|
128
|
+
* then treat "priced exactly" and "priced by guessing the family" as the same
|
|
129
|
+
* thing. They are not. `claude-opus-5` is billed here at Opus 4's rate because
|
|
130
|
+
* its name contains "opus" — a guess that may be right and is not a fact, and a
|
|
131
|
+
* total built from it should be able to say so.
|
|
132
|
+
*
|
|
133
|
+
* @returns {{price: {input:number,output:number}|null, source: 'exact'|'bare'|'prefix'|'family'|'none'}}
|
|
134
|
+
*/
|
|
135
|
+
export function resolvePrice(model, prices = effectivePrices()) {
|
|
136
|
+
if (!model) return { price: null, source: 'none' };
|
|
137
|
+
if (prices[model]) return { price: prices[model], source: 'exact' };
|
|
138
|
+
|
|
139
|
+
const bare = model.includes('/') ? model.slice(model.indexOf('/') + 1) : model;
|
|
140
|
+
if (prices[bare]) return { price: prices[bare], source: 'bare' };
|
|
141
|
+
|
|
142
|
+
let best = null, bestLen = 0;
|
|
143
|
+
for (const k of Object.keys(prices)) {
|
|
144
|
+
if (bare.startsWith(k) && k.length > bestLen) { best = prices[k]; bestLen = k.length; }
|
|
145
|
+
}
|
|
146
|
+
if (best) return { price: best, source: 'prefix' };
|
|
147
|
+
|
|
148
|
+
if (/opus/i.test(model)) return { price: prices['claude-opus-4'] || { input: 15, output: 75 }, source: 'family' };
|
|
149
|
+
if (/sonnet/i.test(model)) return { price: prices['claude-sonnet-4'] || { input: 3, output: 15 }, source: 'family' };
|
|
150
|
+
if (/haiku/i.test(model)) return { price: prices['claude-haiku-4'] || { input: 0.8, output: 4 }, source: 'family' };
|
|
151
|
+
return { price: null, source: 'none' };
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* Cost of one call, with the third state kept.
|
|
156
|
+
*
|
|
157
|
+
* `costForUsage` returns a number, so an unknown model has to come back as 0 —
|
|
158
|
+
* and a model nobody has priced then reads as a model that costs nothing.
|
|
159
|
+
* 172 turns of `claude-fable-5` were billed at $0.00 for exactly that reason,
|
|
160
|
+
* and the total looked like a total rather than a total plus a hole.
|
|
161
|
+
*
|
|
162
|
+
* @returns {{usd:number, priced:boolean, assumed:boolean, source:string, model:string}}
|
|
163
|
+
*/
|
|
164
|
+
export function priceUsage({ model, usage, prices }) {
|
|
165
|
+
const { price, source } = resolvePrice(model, prices || effectivePrices());
|
|
166
|
+
if (!usage || !price) {
|
|
167
|
+
return { usd: 0, priced: false, assumed: false, source, model: model || '' };
|
|
168
|
+
}
|
|
169
|
+
const inTok = usage.input_tokens || 0;
|
|
170
|
+
const outTok = usage.output_tokens || 0;
|
|
171
|
+
const cacheWrite = usage.cache_creation_input_tokens || 0;
|
|
172
|
+
const cacheRead = usage.cache_read_input_tokens || 0;
|
|
173
|
+
const usd = (inTok * price.input + outTok * price.output
|
|
174
|
+
+ cacheWrite * price.input * 1.25 + cacheRead * price.input * 0.1) / 1_000_000;
|
|
175
|
+
return { usd, priced: true, assumed: source === 'family', source, model: model || '' };
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* Dollar cost of a single API call.
|
|
180
|
+
* @param {object} opts
|
|
181
|
+
* @param {string} opts.model
|
|
182
|
+
* @param {{input_tokens?:number, output_tokens?:number}} opts.usage Anthropic response.usage
|
|
183
|
+
* @param {object} [opts.prices] override table (for tests)
|
|
184
|
+
* @returns {number} USD (0 if usage or price unknown)
|
|
185
|
+
*/
|
|
186
|
+
export function costForUsage({ model, usage, prices }) {
|
|
187
|
+
if (!usage) return 0;
|
|
188
|
+
const p = priceForModel(model, prices);
|
|
189
|
+
if (!p) return 0;
|
|
190
|
+
const inTok = usage.input_tokens || 0;
|
|
191
|
+
const outTok = usage.output_tokens || 0;
|
|
192
|
+
// Prompt-caching tokens bill at Anthropic's standard multipliers off the base
|
|
193
|
+
// input price: cache WRITE = 1.25× input, cache READ = 0.1× input. Ignoring
|
|
194
|
+
// them under-counts real spend badly (a cached turn is often 50k+ cache tokens
|
|
195
|
+
// vs a few hundred fresh input tokens).
|
|
196
|
+
const cacheWrite = usage.cache_creation_input_tokens || 0;
|
|
197
|
+
const cacheRead = usage.cache_read_input_tokens || 0;
|
|
198
|
+
return (inTok * p.input + outTok * p.output
|
|
199
|
+
+ cacheWrite * p.input * 1.25 + cacheRead * p.input * 0.1) / 1_000_000;
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
export function round4(n) { return Math.round(n * 10000) / 10000; }
|
|
203
|
+
|
|
204
|
+
// ── CLI: compute one cost from args/env (used by log-verdict.sh `auto` mode) ──
|
|
205
|
+
// node scripts/lib/cost-meter.mjs --model M --in 1234 --out 567
|
|
206
|
+
// prints the USD number (4 dp) to stdout.
|
|
207
|
+
function main(argv) {
|
|
208
|
+
let model = process.env.LLM_MODEL || '';
|
|
209
|
+
let inTok = parseInt(process.env.LLM_INPUT_TOKENS || '0', 10) || 0;
|
|
210
|
+
let outTok = parseInt(process.env.LLM_OUTPUT_TOKENS || '0', 10) || 0;
|
|
211
|
+
for (let i = 0; i < argv.length; i++) {
|
|
212
|
+
if (argv[i] === '--model' && argv[i + 1]) model = argv[++i];
|
|
213
|
+
else if (argv[i] === '--in' && argv[i + 1]) inTok = parseInt(argv[++i], 10) || 0;
|
|
214
|
+
else if (argv[i] === '--out' && argv[i + 1]) outTok = parseInt(argv[++i], 10) || 0;
|
|
215
|
+
}
|
|
216
|
+
// No tokens means the caller never had a usage block to hand us — a
|
|
217
|
+
// measurement that did not happen. Printing 0 here made every verdict in the
|
|
218
|
+
// fleet record a MEASURED zero: the portfolio reported $0.00 spend for twelve
|
|
219
|
+
// projects, and a per-agent budget would have read "spent $0.00 of $25,
|
|
220
|
+
// measured" forever. All 35 agents pass `auto`, so this was every verdict.
|
|
221
|
+
//
|
|
222
|
+
// Exit 2 with nothing on stdout. `log-verdict.sh` omits the field, and every
|
|
223
|
+
// reader downstream already distinguishes an absent cost from a zero one —
|
|
224
|
+
// portfolio.mjs calls it "spend nobody recorded", agent-budget.mjs calls it
|
|
225
|
+
// `unmeasured`. They were right and had nothing to be right about.
|
|
226
|
+
if (inTok <= 0 && outTok <= 0) {
|
|
227
|
+
process.stderr.write('cost-meter: no token usage supplied — cost not measured\n');
|
|
228
|
+
return process.exit(2);
|
|
229
|
+
}
|
|
230
|
+
const cost = costForUsage({ model, usage: { input_tokens: inTok, output_tokens: outTok } });
|
|
231
|
+
process.stdout.write(String(round4(cost)));
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
import { fileURLToPath } from 'node:url';
|
|
235
|
+
const isMain = process.argv[1] && fileURLToPath(import.meta.url) === process.argv[1];
|
|
236
|
+
if (isMain) main(process.argv.slice(2));
|