great-cto 3.26.4 → 3.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/board/.claude-plugin/plugin.json +1 -1
- package/board/packages/board/lib/routes.mjs +127 -2
- package/board/packages/board/lib/view-counter.mjs +122 -0
- package/board/packages/board/public/index.html +938 -560
- package/board/scripts/lib/cost-meter.mjs +236 -0
- package/board/scripts/lib/cross-model-review.mjs +225 -0
- package/board/scripts/lib/provider-exhaustion.mjs +152 -0
- package/package.json +1 -1
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
// scripts/lib/cost-meter.mjs — turn real Anthropic `usage` into real USD.
|
|
2
|
+
//
|
|
3
|
+
// Why it exists (DEEPEN-PIPELINE Wave 1, cost loop):
|
|
4
|
+
// cost-guard.mjs guesses with a hardcoded ROUGH_COST_USD table and
|
|
5
|
+
// log-verdict.sh trusts a typed CLI arg — spend is never measured. This module
|
|
6
|
+
// is the single place that converts an API response's token usage into dollars,
|
|
7
|
+
// so the runner, log-verdict, and any LLM-calling script can record TRUE cost.
|
|
8
|
+
//
|
|
9
|
+
// Prices are USD per 1,000,000 tokens (list prices). They change — override
|
|
10
|
+
// without editing code via either:
|
|
11
|
+
// GREAT_CTO_MODEL_PRICES='{"claude-opus-4-8":{"input":15,"output":75}}' (env, JSON)
|
|
12
|
+
// ~/.great_cto/model-prices.json (file, JSON)
|
|
13
|
+
//
|
|
14
|
+
// Pure + offline-testable: priceForModel() and costForUsage() take an explicit
|
|
15
|
+
// `prices` arg so unit tests never touch env or disk.
|
|
16
|
+
|
|
17
|
+
import { readFileSync } from 'node:fs';
|
|
18
|
+
import { homedir } from 'node:os';
|
|
19
|
+
import { join } from 'node:path';
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Default list prices, USD per 1M tokens.
|
|
23
|
+
*
|
|
24
|
+
* Anthropic rates below are first-party API list prices as of 2026-06-24. They
|
|
25
|
+
* also apply to Claude on Microsoft Foundry; Bedrock and Vertex are partner-
|
|
26
|
+
* operated with separate pricing — override there.
|
|
27
|
+
*
|
|
28
|
+
* Why the current models are listed explicitly rather than left to the family
|
|
29
|
+
* fallback: the fallback bills anything matching /opus/i at the Opus 4 rate, and
|
|
30
|
+
* Opus 5 is $5/$25, not $15/$75. Every Opus 5 turn on this machine — 8,763 of
|
|
31
|
+
* them in one session — was being costed at THREE TIMES its real price, and the
|
|
32
|
+
* total looked like a total. A guess that silently triples the number is worse
|
|
33
|
+
* than no number, because it is spendable.
|
|
34
|
+
*
|
|
35
|
+
* Keep this list ahead of the fallback. A model that reaches the fallback is
|
|
36
|
+
* reported as `assumed` by priceUsage(); one that reaches neither is reported as
|
|
37
|
+
* unpriced rather than free.
|
|
38
|
+
*/
|
|
39
|
+
export const DEFAULT_PRICES = {
|
|
40
|
+
// Claude 5 family
|
|
41
|
+
'claude-fable-5': { input: 10, output: 50 },
|
|
42
|
+
'claude-mythos-5': { input: 10, output: 50 },
|
|
43
|
+
'claude-opus-5': { input: 5, output: 25 },
|
|
44
|
+
'claude-sonnet-5': { input: 2, output: 10 },
|
|
45
|
+
// Claude 4.6–4.8
|
|
46
|
+
'claude-opus-4-8': { input: 5, output: 25 },
|
|
47
|
+
'claude-opus-4-7': { input: 5, output: 25 },
|
|
48
|
+
'claude-opus-4-6': { input: 5, output: 25 },
|
|
49
|
+
'claude-sonnet-4-6': { input: 3, output: 15 },
|
|
50
|
+
'claude-haiku-4-5': { input: 1, output: 5 },
|
|
51
|
+
// Claude 4.x family (bare ids; OpenRouter "anthropic/<id>" slugs resolve via prefix-strip)
|
|
52
|
+
'claude-opus-4': { input: 15, output: 75 },
|
|
53
|
+
'claude-sonnet-4': { input: 3, output: 15 },
|
|
54
|
+
'claude-haiku-4': { input: 0.8, output: 4 },
|
|
55
|
+
// Claude 3.x (still referenced by some evals/agents)
|
|
56
|
+
'claude-3-5-sonnet': { input: 3, output: 15 },
|
|
57
|
+
'claude-3-5-haiku': { input: 0.8, output: 4 },
|
|
58
|
+
'claude-3-opus': { input: 15, output: 75 },
|
|
59
|
+
// OpenRouter non-Anthropic slugs the project routes to (approx list prices —
|
|
60
|
+
// override via ~/.great_cto/model-prices.json or GREAT_CTO_MODEL_PRICES).
|
|
61
|
+
'moonshotai/kimi-k2': { input: 0.55, output: 2.2 },
|
|
62
|
+
'moonshotai/kimi-k3': { input: 3, output: 15 },
|
|
63
|
+
// Read from OpenRouter's /models on 2026-08-27, not guessed. Until now these
|
|
64
|
+
// reported `priced: false` — correctly, and that honesty is why an eval run on
|
|
65
|
+
// glm-5.3-flash showed $0.000: not free, unpriced. Now they are priced exactly.
|
|
66
|
+
'z-ai/glm-5.3-flash': { input: 0.075, output: 0.25 },
|
|
67
|
+
'z-ai/glm-5.3': { input: 1.4, output: 4.4 },
|
|
68
|
+
};
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* NOT modelled, and named here so it is a known gap rather than a silent one:
|
|
72
|
+
* Opus 5 fast mode bills at $10/$50 instead of $5/$25. Turns carry the rate they
|
|
73
|
+
* ran at in `usage.speed`, which this module does not read — a fast-mode turn is
|
|
74
|
+
* therefore under-costed by 2×. Wire it when a transcript in the wild shows
|
|
75
|
+
* `speed: "fast"`; until then the figure is right for standard turns and low for
|
|
76
|
+
* fast ones, which is the direction that does not create false confidence.
|
|
77
|
+
*/
|
|
78
|
+
export const UNMODELLED_RATES = Object.freeze(['opus-5 fast mode ($10/$50)']);
|
|
79
|
+
|
|
80
|
+
/** Load price overrides from env (preferred) then ~/.great_cto/model-prices.json. */
|
|
81
|
+
export function loadPriceOverrides() {
|
|
82
|
+
try {
|
|
83
|
+
if (process.env.GREAT_CTO_MODEL_PRICES) return JSON.parse(process.env.GREAT_CTO_MODEL_PRICES);
|
|
84
|
+
} catch { /* malformed env JSON → ignore */ }
|
|
85
|
+
try {
|
|
86
|
+
return JSON.parse(readFileSync(join(homedir(), '.great_cto', 'model-prices.json'), 'utf8'));
|
|
87
|
+
} catch { /* no override file → ignore */ }
|
|
88
|
+
return {};
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/** Effective price table = defaults merged with overrides. */
|
|
92
|
+
export function effectivePrices() {
|
|
93
|
+
return { ...DEFAULT_PRICES, ...loadPriceOverrides() };
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* Resolve a per-MTok price for a model id.
|
|
98
|
+
* 1. exact key match
|
|
99
|
+
* 2. longest prefix match (so "claude-opus-4-8-2026..." → "claude-opus-4")
|
|
100
|
+
* 3. family heuristic on /opus|sonnet|haiku/
|
|
101
|
+
* Returns { input, output } in USD/MTok, or null if unknown.
|
|
102
|
+
*/
|
|
103
|
+
export function priceForModel(model, prices = effectivePrices()) {
|
|
104
|
+
if (!model) return null;
|
|
105
|
+
if (prices[model]) return prices[model]; // exact (incl. full OpenRouter slug)
|
|
106
|
+
|
|
107
|
+
// Strip a leading "provider/" segment so OpenRouter slugs like
|
|
108
|
+
// "anthropic/claude-sonnet-4" resolve to the bare "claude-sonnet-4" key.
|
|
109
|
+
const bare = model.includes('/') ? model.slice(model.indexOf('/') + 1) : model;
|
|
110
|
+
if (prices[bare]) return prices[bare];
|
|
111
|
+
|
|
112
|
+
let best = null, bestLen = 0;
|
|
113
|
+
for (const k of Object.keys(prices)) {
|
|
114
|
+
if (bare.startsWith(k) && k.length > bestLen) { best = prices[k]; bestLen = k.length; }
|
|
115
|
+
}
|
|
116
|
+
if (best) return best;
|
|
117
|
+
|
|
118
|
+
if (/opus/i.test(model)) return prices['claude-opus-4'] || { input: 15, output: 75 };
|
|
119
|
+
if (/sonnet/i.test(model)) return prices['claude-sonnet-4'] || { input: 3, output: 15 };
|
|
120
|
+
if (/haiku/i.test(model)) return prices['claude-haiku-4'] || { input: 0.8, output: 4 };
|
|
121
|
+
return null;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/**
|
|
125
|
+
* The same lookup, but it says HOW it found the price.
|
|
126
|
+
*
|
|
127
|
+
* `priceForModel` answers with a number or null, and both callers and readers
|
|
128
|
+
* then treat "priced exactly" and "priced by guessing the family" as the same
|
|
129
|
+
* thing. They are not. `claude-opus-5` is billed here at Opus 4's rate because
|
|
130
|
+
* its name contains "opus" — a guess that may be right and is not a fact, and a
|
|
131
|
+
* total built from it should be able to say so.
|
|
132
|
+
*
|
|
133
|
+
* @returns {{price: {input:number,output:number}|null, source: 'exact'|'bare'|'prefix'|'family'|'none'}}
|
|
134
|
+
*/
|
|
135
|
+
export function resolvePrice(model, prices = effectivePrices()) {
|
|
136
|
+
if (!model) return { price: null, source: 'none' };
|
|
137
|
+
if (prices[model]) return { price: prices[model], source: 'exact' };
|
|
138
|
+
|
|
139
|
+
const bare = model.includes('/') ? model.slice(model.indexOf('/') + 1) : model;
|
|
140
|
+
if (prices[bare]) return { price: prices[bare], source: 'bare' };
|
|
141
|
+
|
|
142
|
+
let best = null, bestLen = 0;
|
|
143
|
+
for (const k of Object.keys(prices)) {
|
|
144
|
+
if (bare.startsWith(k) && k.length > bestLen) { best = prices[k]; bestLen = k.length; }
|
|
145
|
+
}
|
|
146
|
+
if (best) return { price: best, source: 'prefix' };
|
|
147
|
+
|
|
148
|
+
if (/opus/i.test(model)) return { price: prices['claude-opus-4'] || { input: 15, output: 75 }, source: 'family' };
|
|
149
|
+
if (/sonnet/i.test(model)) return { price: prices['claude-sonnet-4'] || { input: 3, output: 15 }, source: 'family' };
|
|
150
|
+
if (/haiku/i.test(model)) return { price: prices['claude-haiku-4'] || { input: 0.8, output: 4 }, source: 'family' };
|
|
151
|
+
return { price: null, source: 'none' };
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* Cost of one call, with the third state kept.
|
|
156
|
+
*
|
|
157
|
+
* `costForUsage` returns a number, so an unknown model has to come back as 0 —
|
|
158
|
+
* and a model nobody has priced then reads as a model that costs nothing.
|
|
159
|
+
* 172 turns of `claude-fable-5` were billed at $0.00 for exactly that reason,
|
|
160
|
+
* and the total looked like a total rather than a total plus a hole.
|
|
161
|
+
*
|
|
162
|
+
* @returns {{usd:number, priced:boolean, assumed:boolean, source:string, model:string}}
|
|
163
|
+
*/
|
|
164
|
+
export function priceUsage({ model, usage, prices }) {
|
|
165
|
+
const { price, source } = resolvePrice(model, prices || effectivePrices());
|
|
166
|
+
if (!usage || !price) {
|
|
167
|
+
return { usd: 0, priced: false, assumed: false, source, model: model || '' };
|
|
168
|
+
}
|
|
169
|
+
const inTok = usage.input_tokens || 0;
|
|
170
|
+
const outTok = usage.output_tokens || 0;
|
|
171
|
+
const cacheWrite = usage.cache_creation_input_tokens || 0;
|
|
172
|
+
const cacheRead = usage.cache_read_input_tokens || 0;
|
|
173
|
+
const usd = (inTok * price.input + outTok * price.output
|
|
174
|
+
+ cacheWrite * price.input * 1.25 + cacheRead * price.input * 0.1) / 1_000_000;
|
|
175
|
+
return { usd, priced: true, assumed: source === 'family', source, model: model || '' };
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* Dollar cost of a single API call.
|
|
180
|
+
* @param {object} opts
|
|
181
|
+
* @param {string} opts.model
|
|
182
|
+
* @param {{input_tokens?:number, output_tokens?:number}} opts.usage Anthropic response.usage
|
|
183
|
+
* @param {object} [opts.prices] override table (for tests)
|
|
184
|
+
* @returns {number} USD (0 if usage or price unknown)
|
|
185
|
+
*/
|
|
186
|
+
export function costForUsage({ model, usage, prices }) {
|
|
187
|
+
if (!usage) return 0;
|
|
188
|
+
const p = priceForModel(model, prices);
|
|
189
|
+
if (!p) return 0;
|
|
190
|
+
const inTok = usage.input_tokens || 0;
|
|
191
|
+
const outTok = usage.output_tokens || 0;
|
|
192
|
+
// Prompt-caching tokens bill at Anthropic's standard multipliers off the base
|
|
193
|
+
// input price: cache WRITE = 1.25× input, cache READ = 0.1× input. Ignoring
|
|
194
|
+
// them under-counts real spend badly (a cached turn is often 50k+ cache tokens
|
|
195
|
+
// vs a few hundred fresh input tokens).
|
|
196
|
+
const cacheWrite = usage.cache_creation_input_tokens || 0;
|
|
197
|
+
const cacheRead = usage.cache_read_input_tokens || 0;
|
|
198
|
+
return (inTok * p.input + outTok * p.output
|
|
199
|
+
+ cacheWrite * p.input * 1.25 + cacheRead * p.input * 0.1) / 1_000_000;
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
export function round4(n) { return Math.round(n * 10000) / 10000; }
|
|
203
|
+
|
|
204
|
+
// ── CLI: compute one cost from args/env (used by log-verdict.sh `auto` mode) ──
|
|
205
|
+
// node scripts/lib/cost-meter.mjs --model M --in 1234 --out 567
|
|
206
|
+
// prints the USD number (4 dp) to stdout.
|
|
207
|
+
function main(argv) {
|
|
208
|
+
let model = process.env.LLM_MODEL || '';
|
|
209
|
+
let inTok = parseInt(process.env.LLM_INPUT_TOKENS || '0', 10) || 0;
|
|
210
|
+
let outTok = parseInt(process.env.LLM_OUTPUT_TOKENS || '0', 10) || 0;
|
|
211
|
+
for (let i = 0; i < argv.length; i++) {
|
|
212
|
+
if (argv[i] === '--model' && argv[i + 1]) model = argv[++i];
|
|
213
|
+
else if (argv[i] === '--in' && argv[i + 1]) inTok = parseInt(argv[++i], 10) || 0;
|
|
214
|
+
else if (argv[i] === '--out' && argv[i + 1]) outTok = parseInt(argv[++i], 10) || 0;
|
|
215
|
+
}
|
|
216
|
+
// No tokens means the caller never had a usage block to hand us — a
|
|
217
|
+
// measurement that did not happen. Printing 0 here made every verdict in the
|
|
218
|
+
// fleet record a MEASURED zero: the portfolio reported $0.00 spend for twelve
|
|
219
|
+
// projects, and a per-agent budget would have read "spent $0.00 of $25,
|
|
220
|
+
// measured" forever. All 35 agents pass `auto`, so this was every verdict.
|
|
221
|
+
//
|
|
222
|
+
// Exit 2 with nothing on stdout. `log-verdict.sh` omits the field, and every
|
|
223
|
+
// reader downstream already distinguishes an absent cost from a zero one —
|
|
224
|
+
// portfolio.mjs calls it "spend nobody recorded", agent-budget.mjs calls it
|
|
225
|
+
// `unmeasured`. They were right and had nothing to be right about.
|
|
226
|
+
if (inTok <= 0 && outTok <= 0) {
|
|
227
|
+
process.stderr.write('cost-meter: no token usage supplied — cost not measured\n');
|
|
228
|
+
return process.exit(2);
|
|
229
|
+
}
|
|
230
|
+
const cost = costForUsage({ model, usage: { input_tokens: inTok, output_tokens: outTok } });
|
|
231
|
+
process.stdout.write(String(round4(cost)));
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
import { fileURLToPath } from 'node:url';
|
|
235
|
+
const isMain = process.argv[1] && fileURLToPath(import.meta.url) === process.argv[1];
|
|
236
|
+
if (isMain) main(process.argv.slice(2));
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
// scripts/lib/cross-model-review.mjs — cross-model adversarial review (architect-loop R3).
|
|
2
|
+
//
|
|
3
|
+
// great_cto's reviews are Claude-on-Claude → same-model blind spots. This red-teams
|
|
4
|
+
// the diff with a DIFFERENT model via OpenRouter (default openai/gpt-5), flagging
|
|
5
|
+
// ONLY correctness / requirement / invariant gaps with file:line — no style. The
|
|
6
|
+
// code-reviewer agent merges these with its own findings for high-stakes changes.
|
|
7
|
+
//
|
|
8
|
+
// Pure (buildReviewPrompt / parseFindings / pickReviewerModel) is unit-tested with
|
|
9
|
+
// no network; the CLI does the live OpenRouter call.
|
|
10
|
+
//
|
|
11
|
+
// Usage:
|
|
12
|
+
// git diff main...HEAD | node scripts/lib/cross-model-review.mjs --diff -
|
|
13
|
+
// node scripts/lib/cross-model-review.mjs --diff /tmp/d.diff --spec docs/architecture/ARCH-x.md
|
|
14
|
+
// GREAT_CTO_CROSS_REVIEW_MODEL=google/gemini-2.5-pro node ... --diff -
|
|
15
|
+
|
|
16
|
+
import { readFileSync } from 'node:fs';
|
|
17
|
+
import { fileURLToPath } from 'node:url';
|
|
18
|
+
import { execFileSync } from 'node:child_process';
|
|
19
|
+
import { costForUsage, round4, resolvePrice } from './cost-meter.mjs';
|
|
20
|
+
import { resolveSecondOpinion, codexReview } from './second-opinion.mjs';
|
|
21
|
+
import { principalError } from './provider-exhaustion.mjs';
|
|
22
|
+
import { existsSync, appendFileSync, mkdirSync } from 'node:fs';
|
|
23
|
+
import { join } from 'node:path';
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Exit codes. The first version had two: 0 for PASS and 1 for everything else —
|
|
27
|
+
* BLOCK, a missing API key, a dead network. So "the review blocked this" and
|
|
28
|
+
* "the review did not happen" were the same number to the agent reading it,
|
|
29
|
+
* and the instruction to "note the cross-model pass was skipped" rested on the
|
|
30
|
+
* agent noticing a stderr line. A skipped review now has its own code, and it
|
|
31
|
+
* is neither of the two that mean a verdict was reached.
|
|
32
|
+
*/
|
|
33
|
+
export const EXIT = Object.freeze({ PASS: 0, BLOCK: 1, USAGE: 2, SKIPPED: 3 });
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Which provider reviews, decided from three sources in a fixed order:
|
|
37
|
+
* an explicit `--provider`, then the project's `capabilities: second_opinion`,
|
|
38
|
+
* then — for compatibility with every script that set it — the OpenRouter env.
|
|
39
|
+
*
|
|
40
|
+
* Returns the resolver's four states plus `source`, so the log line can say
|
|
41
|
+
* why this provider and not another. Pure: `codex` and `env` are injected.
|
|
42
|
+
*/
|
|
43
|
+
export function decideProvider({ argv = [], projectMd = '', codex = null, env = process.env } = {}) {
|
|
44
|
+
const forced = readArg(argv, '--provider');
|
|
45
|
+
if (forced) {
|
|
46
|
+
const r = resolveSecondOpinion({ projectMd: `capabilities:\n second_opinion: ${forced}\n`, codex, env });
|
|
47
|
+
return { ...r, source: '--provider' };
|
|
48
|
+
}
|
|
49
|
+
const fromProject = resolveSecondOpinion({ projectMd, codex, env });
|
|
50
|
+
if (fromProject.state !== 'undeclared') return { ...fromProject, source: 'PROJECT.md' };
|
|
51
|
+
if (env.OPENROUTER_API_KEY) {
|
|
52
|
+
return { state: 'declared', provider: 'openrouter', why: '', source: 'OPENROUTER_API_KEY (second_opinion undeclared)' };
|
|
53
|
+
}
|
|
54
|
+
return { ...fromProject, source: 'PROJECT.md' };
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* One line per review, so the board can show what the second opinion DID.
|
|
59
|
+
*
|
|
60
|
+
* `sha` (git HEAD) and `dirty` (working tree had uncommitted changes) are the
|
|
61
|
+
* diff-identity fields BRD-R2's reader keys on, to tell "this line covers the
|
|
62
|
+
* code you're looking at" from "it covered something else". Additive only:
|
|
63
|
+
* both default to `null` — never `undefined`, never omitted — so a line
|
|
64
|
+
* written before this field existed, and any caller that doesn't supply them,
|
|
65
|
+
* still serializes to the same shape a reader already knows how to parse.
|
|
66
|
+
*/
|
|
67
|
+
export function reviewLogLine({ provider, model, state, verdict, findings, cost, source, error_kind, resets_at, sha, dirty }) {
|
|
68
|
+
return JSON.stringify({
|
|
69
|
+
ts: new Date().toISOString(), provider, model: model ?? null, state, verdict: verdict ?? null,
|
|
70
|
+
error_kind: error_kind ?? null, resets_at: resets_at ?? null,
|
|
71
|
+
findings: Array.isArray(findings) ? findings.length : null,
|
|
72
|
+
p0: Array.isArray(findings) ? findings.filter((f) => f.severity === 'P0').length : null,
|
|
73
|
+
cost: cost ?? null, source, sha: sha ?? null, dirty: dirty ?? null,
|
|
74
|
+
});
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
const OPENROUTER_API = 'https://openrouter.ai/api/v1/chat/completions';
|
|
78
|
+
|
|
79
|
+
/** A genuinely non-Claude reviewer model (cross-model). Override via env. */
|
|
80
|
+
export function pickReviewerModel(env = process.env) {
|
|
81
|
+
return env.GREAT_CTO_CROSS_REVIEW_MODEL || 'openai/gpt-5';
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/** Build the red-team prompt. Calibrated: correctness/invariant only, file:line, no style. */
|
|
85
|
+
export function buildReviewPrompt({ diff, spec }) {
|
|
86
|
+
const system =
|
|
87
|
+
'You are an adversarial code reviewer from a DIFFERENT model family than the author. ' +
|
|
88
|
+
'Your job is to catch what a same-model reviewer would miss. Review ONLY for: ' +
|
|
89
|
+
'correctness bugs, violated requirements, broken invariants, security holes, data loss. ' +
|
|
90
|
+
'Do NOT report style, naming, or preferences. Ground every finding in the diff with file:line. ' +
|
|
91
|
+
'Default to silence over a weak finding. ' +
|
|
92
|
+
'Output ONE finding per line in EXACTLY this format:\n' +
|
|
93
|
+
'<file>:<line> | <P0|P1|P2> | <one-sentence concrete issue>\n' +
|
|
94
|
+
'P0 = data loss / security / broken build or prod path. ' +
|
|
95
|
+
'After the findings, output a final line: VERDICT: BLOCK (if any P0) or VERDICT: PASS.';
|
|
96
|
+
const user =
|
|
97
|
+
(spec ? `Spec / intent:\n${spec}\n\n` : '') +
|
|
98
|
+
`Diff under review:\n${diff}\n\n` +
|
|
99
|
+
`Report findings (file:line | severity | issue), then VERDICT:`;
|
|
100
|
+
return { system, user };
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/** Parse the model's findings + verdict. */
|
|
104
|
+
export function parseFindings(text) {
|
|
105
|
+
const findings = [];
|
|
106
|
+
let verdict = null;
|
|
107
|
+
for (const raw of String(text).split('\n')) {
|
|
108
|
+
const line = raw.trim();
|
|
109
|
+
const v = line.match(/^VERDICT:\s*(BLOCK|PASS)/i);
|
|
110
|
+
if (v) { verdict = v[1].toUpperCase(); continue; }
|
|
111
|
+
// <file>:<line> | <SEV> | <issue>
|
|
112
|
+
const m = line.match(/^(.+?):(\d+)\s*\|\s*(P[012])\s*\|\s*(.+)$/i);
|
|
113
|
+
if (m) findings.push({ file: m[1].trim(), line: parseInt(m[2], 10), severity: m[3].toUpperCase(), issue: m[4].trim() });
|
|
114
|
+
}
|
|
115
|
+
// Derive verdict if the model omitted it: any P0 → BLOCK.
|
|
116
|
+
if (!verdict) verdict = findings.some(f => f.severity === 'P0') ? 'BLOCK' : 'PASS';
|
|
117
|
+
return { findings, verdict };
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
// ── CLI ───────────────────────────────────────────────────────────────────────
|
|
121
|
+
|
|
122
|
+
async function callOpenRouter({ apiKey, model, system, user }) {
|
|
123
|
+
const res = await fetch(OPENROUTER_API, {
|
|
124
|
+
method: 'POST',
|
|
125
|
+
headers: { Authorization: `Bearer ${apiKey}`, 'HTTP-Referer': 'https://greatcto.systems', 'X-Title': 'great_cto-xmodel-review', 'content-type': 'application/json' },
|
|
126
|
+
body: JSON.stringify({ model, max_tokens: 1200, temperature: 0, messages: [{ role: 'system', content: system }, { role: 'user', content: user }] }),
|
|
127
|
+
});
|
|
128
|
+
if (!res.ok) throw new Error(`OpenRouter ${res.status}: ${(await res.text()).slice(0, 200)}`);
|
|
129
|
+
const data = await res.json();
|
|
130
|
+
const u = data.usage || null;
|
|
131
|
+
return { text: data.choices?.[0]?.message?.content?.trim() || '', usage: u ? { input_tokens: u.prompt_tokens ?? 0, output_tokens: u.completion_tokens ?? 0 } : null, model };
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
function readArg(argv, name) { const i = argv.indexOf(name); return i > -1 ? argv[i + 1] : null; }
|
|
135
|
+
|
|
136
|
+
/**
|
|
137
|
+
* The tree's identity at review time — git HEAD and whether it was dirty.
|
|
138
|
+
* CLI-only: the pure `reviewLogLine` above never shells out; per this file's
|
|
139
|
+
* own pure/CLI split (see file header), the git call lives here and the
|
|
140
|
+
* result is injected. Returns nulls outside a git repo rather than throwing —
|
|
141
|
+
* "couldn't determine identity" is data for the log line, not a reason to
|
|
142
|
+
* fail the review.
|
|
143
|
+
*/
|
|
144
|
+
function gitIdentity(cwd) {
|
|
145
|
+
try {
|
|
146
|
+
const sha = execFileSync('git', ['rev-parse', 'HEAD'], { cwd, encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'] }).trim();
|
|
147
|
+
const status = execFileSync('git', ['status', '--porcelain'], { cwd, encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'] });
|
|
148
|
+
return { sha: sha || null, dirty: status.trim().length > 0 };
|
|
149
|
+
} catch {
|
|
150
|
+
return { sha: null, dirty: null };
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
async function main(argv) {
|
|
155
|
+
const diffPath = readArg(argv, '--diff');
|
|
156
|
+
if (!diffPath) { console.error('Usage: cross-model-review.mjs --diff <file|-> [--spec <file>] [--model <slug>] [--provider codex|openrouter]'); process.exit(EXIT.USAGE); }
|
|
157
|
+
|
|
158
|
+
const root = readArg(argv, '--root') || process.cwd();
|
|
159
|
+
const mdPath = join(root, '.great_cto', 'PROJECT.md');
|
|
160
|
+
const projectMd = existsSync(mdPath) ? readFileSync(mdPath, 'utf8') : '';
|
|
161
|
+
const decision = decideProvider({ argv, projectMd });
|
|
162
|
+
const identity = gitIdentity(root);
|
|
163
|
+
const logPath = join(root, '.great_cto', 'cross-review.log');
|
|
164
|
+
const log = (rec) => { try { mkdirSync(join(root, '.great_cto'), { recursive: true }); appendFileSync(logPath, reviewLogLine({ ...rec, source: decision.source, sha: identity.sha, dirty: identity.dirty }) + '\n'); } catch { /* the log is evidence, not a gate */ } };
|
|
165
|
+
|
|
166
|
+
// Anything that is not a reviewer reviewing exits SKIPPED — not PASS, and not
|
|
167
|
+
// the BLOCK code either. The line says why, and the log keeps it.
|
|
168
|
+
if (decision.state !== 'declared') {
|
|
169
|
+
console.log(`cross-model-review: SKIPPED (${decision.state}) — ${decision.why}`);
|
|
170
|
+
log({ provider: decision.provider, state: decision.state, verdict: null, findings: null, cost: null });
|
|
171
|
+
process.exit(EXIT.SKIPPED);
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
const diff = diffPath === '-' ? readFileSync(0, 'utf8') : readFileSync(diffPath, 'utf8');
|
|
175
|
+
if (!diff.trim()) { console.log('cross-model-review: empty diff, nothing to review.'); process.exit(EXIT.PASS); }
|
|
176
|
+
const specFile = readArg(argv, '--spec');
|
|
177
|
+
const spec = specFile ? readFileSync(specFile, 'utf8').slice(0, 4000) : null;
|
|
178
|
+
const prompt = buildReviewPrompt({ diff: diff.slice(0, 24000), spec });
|
|
179
|
+
|
|
180
|
+
let res;
|
|
181
|
+
if (decision.provider === 'codex') {
|
|
182
|
+
const model = readArg(argv, '--model') || null; // null = whatever ~/.codex/config.toml names
|
|
183
|
+
console.error(`cross-model-review: reviewer=codex${model ? ' -m ' + model : ' (' + (decision.codex?.model || 'default model') + ')'} (cross-model red-team, read-only sandbox)`);
|
|
184
|
+
const r = await codexReview({ ...prompt, cwd: root, model, bin: process.env.GREAT_CTO_CODEX_BIN || 'codex' });
|
|
185
|
+
if (r.state !== 'ok') {
|
|
186
|
+
// The reason a human is shown is RANKED, not the first thing Codex said.
|
|
187
|
+
// A quota-exhausted review used to display "Skill descriptions were
|
|
188
|
+
// shortened…" — advisory noise that arrived first — while the sentence
|
|
189
|
+
// naming the cause and its reset date was truncated away.
|
|
190
|
+
const principal = principalError(r.errors);
|
|
191
|
+
const reason = principal ? `${principal.kind}: ${principal.why}` : 'no answer';
|
|
192
|
+
console.log(`cross-model-review: SKIPPED (codex ${r.state}) — ${reason}`);
|
|
193
|
+
if (principal && r.errors.length > 1) {
|
|
194
|
+
console.log(` (${r.errors.length - 1} other message(s) from codex, not the cause)`);
|
|
195
|
+
}
|
|
196
|
+
log({
|
|
197
|
+
provider: 'codex', model: r.model ?? decision.codex?.model, state: r.state,
|
|
198
|
+
verdict: null, findings: null, cost: null,
|
|
199
|
+
error_kind: principal?.kind ?? null, resets_at: principal?.resetsAt ?? null,
|
|
200
|
+
});
|
|
201
|
+
process.exit(EXIT.SKIPPED);
|
|
202
|
+
}
|
|
203
|
+
res = { text: r.text, usage: r.usage, model: r.model ?? decision.codex?.model ?? 'codex' };
|
|
204
|
+
} else {
|
|
205
|
+
const model = readArg(argv, '--model') || pickReviewerModel();
|
|
206
|
+
console.error(`cross-model-review: reviewer=${model} (cross-model red-team via OpenRouter)`);
|
|
207
|
+
res = await callOpenRouter({ apiKey: process.env.OPENROUTER_API_KEY, model, ...prompt });
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
const { findings, verdict } = parseFindings(res.text);
|
|
211
|
+
// Unpriced is null, not zero. The first real Codex review logged `cost: 0`
|
|
212
|
+
// for gpt-5.6-terra — a model the price table does not carry — because usage
|
|
213
|
+
// was present and costForUsage prices an unknown model at nothing. A reviewer
|
|
214
|
+
// that reads as free is the defect this repository has removed twice already.
|
|
215
|
+
const priced = resolvePrice(res.model).price != null;
|
|
216
|
+
const cost = res.usage && priced ? round4(costForUsage({ model: res.model, usage: res.usage })) : null;
|
|
217
|
+
|
|
218
|
+
for (const f of findings) console.log(` ${f.severity} ${f.file}:${f.line} — ${f.issue}`);
|
|
219
|
+
console.log(`\ncross-model-review (${decision.provider}:${res.model}): ${findings.length} finding(s), VERDICT: ${verdict} (${cost == null ? 'cost unpriced' : '$' + cost})`);
|
|
220
|
+
log({ provider: decision.provider, model: res.model, state: 'ok', verdict, findings, cost });
|
|
221
|
+
process.exit(verdict === 'BLOCK' ? EXIT.BLOCK : EXIT.PASS);
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
const isMain = process.argv[1] && fileURLToPath(import.meta.url) === process.argv[1];
|
|
225
|
+
if (isMain) main(process.argv.slice(2)).catch(e => { console.error('FATAL:', e.message); process.exit(EXIT.SKIPPED); });
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
// Some provider failures mean "try again". Others mean "every remaining call
|
|
2
|
+
// will fail exactly like this one".
|
|
3
|
+
//
|
|
4
|
+
// What happened
|
|
5
|
+
// -------------
|
|
6
|
+
// A 75-file eval run spent $13.99, ran out of OpenRouter credits partway, and
|
|
7
|
+
// then made 147 more calls that could not possibly succeed — one per remaining
|
|
8
|
+
// case, each returning the same 402. The dropout gate did its job at the end and
|
|
9
|
+
// reported thirteen files as NOT MEASURED rather than as scores.
|
|
10
|
+
//
|
|
11
|
+
// But the run had already written those thirteen files into
|
|
12
|
+
// `results-history.jsonl` with `rate: 0`, and the drift detector reads `rate`.
|
|
13
|
+
// So the loop's next comparison would have read thirteen evals as having
|
|
14
|
+
// collapsed from ~0.85 to 0.00 overnight, and alarmed on a regression that is
|
|
15
|
+
// really an empty wallet.
|
|
16
|
+
//
|
|
17
|
+
// A run that did not happen recorded as a score of zero. Same defect this
|
|
18
|
+
// repository keeps finding, this time between two of its own components.
|
|
19
|
+
//
|
|
20
|
+
// So: recognise the terminal states, stop the run at the first one, and keep the
|
|
21
|
+
// unrunnable files out of the history entirely.
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* What kind of failure this is, from the error a provider call threw.
|
|
25
|
+
*
|
|
26
|
+
* The distinction that matters is not the status code but whether waiting or
|
|
27
|
+
* retrying could change the answer. 429 is the provider saying "slow down" —
|
|
28
|
+
* that resolves. 402 is the provider saying "you have no money" — that resolves
|
|
29
|
+
* only by someone topping up, which will not happen inside this run.
|
|
30
|
+
*
|
|
31
|
+
* @param {Error|string} err
|
|
32
|
+
* @returns {{terminal:boolean, kind:'credits'|'billing'|'quota'|'auth'|'rate-limit'|'transient',
|
|
33
|
+
* why:string, resetsAt?:string}}
|
|
34
|
+
*/
|
|
35
|
+
export function classifyProviderError(err) {
|
|
36
|
+
const msg = String(err?.message ?? err ?? '');
|
|
37
|
+
|
|
38
|
+
// Match the status as a distinct token so a `402` inside a response body — an
|
|
39
|
+
// id, a byte count — does not read as the status of the call itself.
|
|
40
|
+
const status = msg.match(/\bAPI\s+(\d{3})\b/)?.[1] ?? msg.match(/\b(4\d{2}|5\d{2})\b/)?.[1] ?? null;
|
|
41
|
+
const body = msg.toLowerCase();
|
|
42
|
+
|
|
43
|
+
if (status === '402' || /insufficient (credit|balance|fund)|no credits|out of credits|payment required/.test(body)) {
|
|
44
|
+
return { terminal: true, kind: 'credits', why: 'the provider account is out of credits — every remaining call fails identically until someone tops it up' };
|
|
45
|
+
}
|
|
46
|
+
// An account locked over billing is terminal, and it is NOT the same state as
|
|
47
|
+
// an empty balance. This repository's own GitHub Actions have been refused with
|
|
48
|
+
// this exact message since 2026-06-25 — one hundred consecutive runs, each
|
|
49
|
+
// failing identically, none able to succeed until a human settles a bill.
|
|
50
|
+
// Classified as transient it would earn a retry every time, which is the 402
|
|
51
|
+
// mistake this module exists to prevent, wearing different words.
|
|
52
|
+
//
|
|
53
|
+
// Kept apart from `credits` deliberately: topping up a balance and unlocking an
|
|
54
|
+
// account are different actions by possibly different people, and a message
|
|
55
|
+
// that merges them sends someone to the wrong screen.
|
|
56
|
+
if (/account is locked|billing (issue|problem|lock)|locked due to.*billing|billing.*(suspend|disabled)/.test(body)) {
|
|
57
|
+
return { terminal: true, kind: 'billing', why: 'the provider account is locked over billing — no retry clears it until a human settles the bill' };
|
|
58
|
+
}
|
|
59
|
+
// A PLAN QUOTA is its own kind, and merging it into rate-limit was costing a
|
|
60
|
+
// real answer. Codex on a ChatGPT plan answers "You've hit your usage limit.
|
|
61
|
+
// Upgrade to Plus to continue, or try again at Oct 5th, 2026 9:41 AM" — which
|
|
62
|
+
// is terminal for anything running today and NOT terminal in the way `credits`
|
|
63
|
+
// is: nobody has to do anything, it comes back by itself, on a date the
|
|
64
|
+
// message names. Read as `rate-limit` it earns a retry loop that cannot
|
|
65
|
+
// succeed for a month; read as `credits` it sends someone to a billing page
|
|
66
|
+
// they do not need.
|
|
67
|
+
//
|
|
68
|
+
// The date is the actionable half, so it is extracted rather than described.
|
|
69
|
+
if (/usage limit|quota (exceeded|exhausted)|monthly limit|plan limit/.test(body)) {
|
|
70
|
+
const at = msg.match(/try again at ([^.\n"]{4,40})/i)?.[1]?.trim() ?? null;
|
|
71
|
+
return {
|
|
72
|
+
terminal: true, kind: 'quota',
|
|
73
|
+
why: at
|
|
74
|
+
? `the provider plan's usage limit is spent — it returns on its own at ${at}, and no retry before then can succeed`
|
|
75
|
+
: "the provider plan's usage limit is spent — it returns on its own, and no retry before then can succeed",
|
|
76
|
+
...(at ? { resetsAt: at } : {}),
|
|
77
|
+
};
|
|
78
|
+
}
|
|
79
|
+
if (status === '401' || status === '403' || /invalid api key|unauthorized|forbidden/.test(body)) {
|
|
80
|
+
return { terminal: true, kind: 'auth', why: 'the provider rejected the key — no retry inside this run can fix that' };
|
|
81
|
+
}
|
|
82
|
+
if (status === '429' || /rate.?limit|too many requests/.test(body)) {
|
|
83
|
+
return { terminal: false, kind: 'rate-limit', why: 'rate limited — this resolves on its own' };
|
|
84
|
+
}
|
|
85
|
+
return { terminal: false, kind: 'transient', why: msg.slice(0, 160) || 'unclassified provider error' };
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* What to print when a run gives up.
|
|
90
|
+
*
|
|
91
|
+
* Names the money spent, because the next question anybody asks is "did I pay
|
|
92
|
+
* for that", and says plainly that the remaining files were not measured rather
|
|
93
|
+
* than letting a reader infer a result from a truncated table.
|
|
94
|
+
*/
|
|
95
|
+
export function exhaustionReport({ kind, why, completed, total, costUsd }) {
|
|
96
|
+
const spent = typeof costUsd === 'number' ? `$${costUsd.toFixed(2)}` : 'an unrecorded amount';
|
|
97
|
+
return [
|
|
98
|
+
`RUN STOPPED — ${kind}: ${why}`,
|
|
99
|
+
` ${completed} of ${total} eval file(s) completed; ${spent} spent.`,
|
|
100
|
+
` The rest were NOT MEASURED. They are not zeros, and they are not written to`,
|
|
101
|
+
` history — a run that did not happen must not become a data point.`,
|
|
102
|
+
` Re-run once the account is funded.`,
|
|
103
|
+
].join('\n');
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Should this result be allowed into the trend history?
|
|
108
|
+
*
|
|
109
|
+
* A file whose cases never reached the provider has a rate computed over the
|
|
110
|
+
* prefix that did run, which is not a draw from the case list. `eval-power`
|
|
111
|
+
* already refuses to compare it against a threshold; this refuses to let it
|
|
112
|
+
* become tomorrow's baseline.
|
|
113
|
+
*/
|
|
114
|
+
export function admissibleToHistory(result) {
|
|
115
|
+
if (!result) return { ok: false, why: 'no result' };
|
|
116
|
+
if (result.dropout?.severe) {
|
|
117
|
+
return { ok: false, why: `dropout: ${result.dropout.why ?? 'the run stopped partway through this file'}` };
|
|
118
|
+
}
|
|
119
|
+
if (!result.judged) return { ok: false, why: 'no case was judged' };
|
|
120
|
+
return { ok: true };
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Pick the error a human should be shown, out of everything a provider emitted.
|
|
125
|
+
*
|
|
126
|
+
* Codex reports advisory problems and fatal ones through the same channel, in
|
|
127
|
+
* arrival order. On 2026-09-06 a review that failed on an exhausted plan quota
|
|
128
|
+
* displayed "Skill descriptions were shortened to fit the skills context
|
|
129
|
+
* budget" — the first error in the array, and pure noise — while the sentence
|
|
130
|
+
* naming the cause and its reset date sat second and was cut off by a 300-char
|
|
131
|
+
* truncation. The reader is then sent to disable skills over a quota problem.
|
|
132
|
+
*
|
|
133
|
+
* Terminal beats non-terminal; among terminal, the earliest listed wins. An
|
|
134
|
+
* empty list is `null`, not an invented reason.
|
|
135
|
+
*
|
|
136
|
+
* @param {string[]} errors
|
|
137
|
+
* @returns {{message:string, kind:string, why:string, terminal:boolean, resetsAt?:string}|null}
|
|
138
|
+
*/
|
|
139
|
+
export function principalError(errors) {
|
|
140
|
+
const list = (Array.isArray(errors) ? errors : []).filter((e) => String(e ?? '').trim());
|
|
141
|
+
if (!list.length) return null;
|
|
142
|
+
const RANK = { credits: 0, billing: 0, auth: 0, quota: 0, 'rate-limit': 1, transient: 2 };
|
|
143
|
+
let best = null;
|
|
144
|
+
for (const [i, message] of list.entries()) {
|
|
145
|
+
const c = classifyProviderError(message);
|
|
146
|
+
const score = [RANK[c.kind] ?? 2, i];
|
|
147
|
+
if (!best || score[0] < best.score[0] || (score[0] === best.score[0] && score[1] < best.score[1])) {
|
|
148
|
+
best = { score, value: { message: String(message), ...c } };
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
return best.value;
|
|
152
|
+
}
|
package/package.json
CHANGED