great-cto 3.26.2 → 3.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/board/.claude-plugin/plugin.json +1 -1
- package/board/packages/board/lib/fleet.mjs +43 -1
- package/board/packages/board/lib/routes.mjs +147 -9
- package/board/packages/board/lib/view-counter.mjs +122 -0
- package/board/packages/board/public/index.html +946 -545
- package/board/scripts/lib/agent-posture.mjs +266 -0
- package/board/scripts/lib/cost-meter.mjs +236 -0
- package/board/scripts/lib/cross-model-review.mjs +225 -0
- package/board/scripts/lib/provider-exhaustion.mjs +152 -0
- package/package.json +1 -1
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
// scripts/lib/cross-model-review.mjs — cross-model adversarial review (architect-loop R3).
|
|
2
|
+
//
|
|
3
|
+
// great_cto's reviews are Claude-on-Claude → same-model blind spots. This red-teams
|
|
4
|
+
// the diff with a DIFFERENT model via OpenRouter (default openai/gpt-5), flagging
|
|
5
|
+
// ONLY correctness / requirement / invariant gaps with file:line — no style. The
|
|
6
|
+
// code-reviewer agent merges these with its own findings for high-stakes changes.
|
|
7
|
+
//
|
|
8
|
+
// Pure (buildReviewPrompt / parseFindings / pickReviewerModel) is unit-tested with
|
|
9
|
+
// no network; the CLI does the live OpenRouter call.
|
|
10
|
+
//
|
|
11
|
+
// Usage:
|
|
12
|
+
// git diff main...HEAD | node scripts/lib/cross-model-review.mjs --diff -
|
|
13
|
+
// node scripts/lib/cross-model-review.mjs --diff /tmp/d.diff --spec docs/architecture/ARCH-x.md
|
|
14
|
+
// GREAT_CTO_CROSS_REVIEW_MODEL=google/gemini-2.5-pro node ... --diff -
|
|
15
|
+
|
|
16
|
+
import { readFileSync } from 'node:fs';
|
|
17
|
+
import { fileURLToPath } from 'node:url';
|
|
18
|
+
import { execFileSync } from 'node:child_process';
|
|
19
|
+
import { costForUsage, round4, resolvePrice } from './cost-meter.mjs';
|
|
20
|
+
import { resolveSecondOpinion, codexReview } from './second-opinion.mjs';
|
|
21
|
+
import { principalError } from './provider-exhaustion.mjs';
|
|
22
|
+
import { existsSync, appendFileSync, mkdirSync } from 'node:fs';
|
|
23
|
+
import { join } from 'node:path';
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Exit codes. The first version had two: 0 for PASS and 1 for everything else —
|
|
27
|
+
* BLOCK, a missing API key, a dead network. So "the review blocked this" and
|
|
28
|
+
* "the review did not happen" were the same number to the agent reading it,
|
|
29
|
+
* and the instruction to "note the cross-model pass was skipped" rested on the
|
|
30
|
+
* agent noticing a stderr line. A skipped review now has its own code, and it
|
|
31
|
+
* is neither of the two that mean a verdict was reached.
|
|
32
|
+
*/
|
|
33
|
+
export const EXIT = Object.freeze({ PASS: 0, BLOCK: 1, USAGE: 2, SKIPPED: 3 });
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Which provider reviews, decided from three sources in a fixed order:
|
|
37
|
+
* an explicit `--provider`, then the project's `capabilities: second_opinion`,
|
|
38
|
+
* then — for compatibility with every script that set it — the OpenRouter env.
|
|
39
|
+
*
|
|
40
|
+
* Returns the resolver's four states plus `source`, so the log line can say
|
|
41
|
+
* why this provider and not another. Pure: `codex` and `env` are injected.
|
|
42
|
+
*/
|
|
43
|
+
export function decideProvider({ argv = [], projectMd = '', codex = null, env = process.env } = {}) {
|
|
44
|
+
const forced = readArg(argv, '--provider');
|
|
45
|
+
if (forced) {
|
|
46
|
+
const r = resolveSecondOpinion({ projectMd: `capabilities:\n second_opinion: ${forced}\n`, codex, env });
|
|
47
|
+
return { ...r, source: '--provider' };
|
|
48
|
+
}
|
|
49
|
+
const fromProject = resolveSecondOpinion({ projectMd, codex, env });
|
|
50
|
+
if (fromProject.state !== 'undeclared') return { ...fromProject, source: 'PROJECT.md' };
|
|
51
|
+
if (env.OPENROUTER_API_KEY) {
|
|
52
|
+
return { state: 'declared', provider: 'openrouter', why: '', source: 'OPENROUTER_API_KEY (second_opinion undeclared)' };
|
|
53
|
+
}
|
|
54
|
+
return { ...fromProject, source: 'PROJECT.md' };
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* One line per review, so the board can show what the second opinion DID.
|
|
59
|
+
*
|
|
60
|
+
* `sha` (git HEAD) and `dirty` (working tree had uncommitted changes) are the
|
|
61
|
+
* diff-identity fields BRD-R2's reader keys on, to tell "this line covers the
|
|
62
|
+
* code you're looking at" from "it covered something else". Additive only:
|
|
63
|
+
* both default to `null` — never `undefined`, never omitted — so a line
|
|
64
|
+
* written before this field existed, and any caller that doesn't supply them,
|
|
65
|
+
* still serializes to the same shape a reader already knows how to parse.
|
|
66
|
+
*/
|
|
67
|
+
export function reviewLogLine({ provider, model, state, verdict, findings, cost, source, error_kind, resets_at, sha, dirty }) {
|
|
68
|
+
return JSON.stringify({
|
|
69
|
+
ts: new Date().toISOString(), provider, model: model ?? null, state, verdict: verdict ?? null,
|
|
70
|
+
error_kind: error_kind ?? null, resets_at: resets_at ?? null,
|
|
71
|
+
findings: Array.isArray(findings) ? findings.length : null,
|
|
72
|
+
p0: Array.isArray(findings) ? findings.filter((f) => f.severity === 'P0').length : null,
|
|
73
|
+
cost: cost ?? null, source, sha: sha ?? null, dirty: dirty ?? null,
|
|
74
|
+
});
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
const OPENROUTER_API = 'https://openrouter.ai/api/v1/chat/completions';
|
|
78
|
+
|
|
79
|
+
/** A genuinely non-Claude reviewer model (cross-model). Override via env. */
|
|
80
|
+
export function pickReviewerModel(env = process.env) {
|
|
81
|
+
return env.GREAT_CTO_CROSS_REVIEW_MODEL || 'openai/gpt-5';
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
/** Build the red-team prompt. Calibrated: correctness/invariant only, file:line, no style. */
|
|
85
|
+
export function buildReviewPrompt({ diff, spec }) {
|
|
86
|
+
const system =
|
|
87
|
+
'You are an adversarial code reviewer from a DIFFERENT model family than the author. ' +
|
|
88
|
+
'Your job is to catch what a same-model reviewer would miss. Review ONLY for: ' +
|
|
89
|
+
'correctness bugs, violated requirements, broken invariants, security holes, data loss. ' +
|
|
90
|
+
'Do NOT report style, naming, or preferences. Ground every finding in the diff with file:line. ' +
|
|
91
|
+
'Default to silence over a weak finding. ' +
|
|
92
|
+
'Output ONE finding per line in EXACTLY this format:\n' +
|
|
93
|
+
'<file>:<line> | <P0|P1|P2> | <one-sentence concrete issue>\n' +
|
|
94
|
+
'P0 = data loss / security / broken build or prod path. ' +
|
|
95
|
+
'After the findings, output a final line: VERDICT: BLOCK (if any P0) or VERDICT: PASS.';
|
|
96
|
+
const user =
|
|
97
|
+
(spec ? `Spec / intent:\n${spec}\n\n` : '') +
|
|
98
|
+
`Diff under review:\n${diff}\n\n` +
|
|
99
|
+
`Report findings (file:line | severity | issue), then VERDICT:`;
|
|
100
|
+
return { system, user };
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/** Parse the model's findings + verdict. */
|
|
104
|
+
export function parseFindings(text) {
|
|
105
|
+
const findings = [];
|
|
106
|
+
let verdict = null;
|
|
107
|
+
for (const raw of String(text).split('\n')) {
|
|
108
|
+
const line = raw.trim();
|
|
109
|
+
const v = line.match(/^VERDICT:\s*(BLOCK|PASS)/i);
|
|
110
|
+
if (v) { verdict = v[1].toUpperCase(); continue; }
|
|
111
|
+
// <file>:<line> | <SEV> | <issue>
|
|
112
|
+
const m = line.match(/^(.+?):(\d+)\s*\|\s*(P[012])\s*\|\s*(.+)$/i);
|
|
113
|
+
if (m) findings.push({ file: m[1].trim(), line: parseInt(m[2], 10), severity: m[3].toUpperCase(), issue: m[4].trim() });
|
|
114
|
+
}
|
|
115
|
+
// Derive verdict if the model omitted it: any P0 → BLOCK.
|
|
116
|
+
if (!verdict) verdict = findings.some(f => f.severity === 'P0') ? 'BLOCK' : 'PASS';
|
|
117
|
+
return { findings, verdict };
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
// ── CLI ───────────────────────────────────────────────────────────────────────
|
|
121
|
+
|
|
122
|
+
async function callOpenRouter({ apiKey, model, system, user }) {
|
|
123
|
+
const res = await fetch(OPENROUTER_API, {
|
|
124
|
+
method: 'POST',
|
|
125
|
+
headers: { Authorization: `Bearer ${apiKey}`, 'HTTP-Referer': 'https://greatcto.systems', 'X-Title': 'great_cto-xmodel-review', 'content-type': 'application/json' },
|
|
126
|
+
body: JSON.stringify({ model, max_tokens: 1200, temperature: 0, messages: [{ role: 'system', content: system }, { role: 'user', content: user }] }),
|
|
127
|
+
});
|
|
128
|
+
if (!res.ok) throw new Error(`OpenRouter ${res.status}: ${(await res.text()).slice(0, 200)}`);
|
|
129
|
+
const data = await res.json();
|
|
130
|
+
const u = data.usage || null;
|
|
131
|
+
return { text: data.choices?.[0]?.message?.content?.trim() || '', usage: u ? { input_tokens: u.prompt_tokens ?? 0, output_tokens: u.completion_tokens ?? 0 } : null, model };
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
function readArg(argv, name) { const i = argv.indexOf(name); return i > -1 ? argv[i + 1] : null; }
|
|
135
|
+
|
|
136
|
+
/**
|
|
137
|
+
* The tree's identity at review time — git HEAD and whether it was dirty.
|
|
138
|
+
* CLI-only: the pure `reviewLogLine` above never shells out; per this file's
|
|
139
|
+
* own pure/CLI split (see file header), the git call lives here and the
|
|
140
|
+
* result is injected. Returns nulls outside a git repo rather than throwing —
|
|
141
|
+
* "couldn't determine identity" is data for the log line, not a reason to
|
|
142
|
+
* fail the review.
|
|
143
|
+
*/
|
|
144
|
+
function gitIdentity(cwd) {
|
|
145
|
+
try {
|
|
146
|
+
const sha = execFileSync('git', ['rev-parse', 'HEAD'], { cwd, encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'] }).trim();
|
|
147
|
+
const status = execFileSync('git', ['status', '--porcelain'], { cwd, encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'] });
|
|
148
|
+
return { sha: sha || null, dirty: status.trim().length > 0 };
|
|
149
|
+
} catch {
|
|
150
|
+
return { sha: null, dirty: null };
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
async function main(argv) {
|
|
155
|
+
const diffPath = readArg(argv, '--diff');
|
|
156
|
+
if (!diffPath) { console.error('Usage: cross-model-review.mjs --diff <file|-> [--spec <file>] [--model <slug>] [--provider codex|openrouter]'); process.exit(EXIT.USAGE); }
|
|
157
|
+
|
|
158
|
+
const root = readArg(argv, '--root') || process.cwd();
|
|
159
|
+
const mdPath = join(root, '.great_cto', 'PROJECT.md');
|
|
160
|
+
const projectMd = existsSync(mdPath) ? readFileSync(mdPath, 'utf8') : '';
|
|
161
|
+
const decision = decideProvider({ argv, projectMd });
|
|
162
|
+
const identity = gitIdentity(root);
|
|
163
|
+
const logPath = join(root, '.great_cto', 'cross-review.log');
|
|
164
|
+
const log = (rec) => { try { mkdirSync(join(root, '.great_cto'), { recursive: true }); appendFileSync(logPath, reviewLogLine({ ...rec, source: decision.source, sha: identity.sha, dirty: identity.dirty }) + '\n'); } catch { /* the log is evidence, not a gate */ } };
|
|
165
|
+
|
|
166
|
+
// Anything that is not a reviewer reviewing exits SKIPPED — not PASS, and not
|
|
167
|
+
// the BLOCK code either. The line says why, and the log keeps it.
|
|
168
|
+
if (decision.state !== 'declared') {
|
|
169
|
+
console.log(`cross-model-review: SKIPPED (${decision.state}) — ${decision.why}`);
|
|
170
|
+
log({ provider: decision.provider, state: decision.state, verdict: null, findings: null, cost: null });
|
|
171
|
+
process.exit(EXIT.SKIPPED);
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
const diff = diffPath === '-' ? readFileSync(0, 'utf8') : readFileSync(diffPath, 'utf8');
|
|
175
|
+
if (!diff.trim()) { console.log('cross-model-review: empty diff, nothing to review.'); process.exit(EXIT.PASS); }
|
|
176
|
+
const specFile = readArg(argv, '--spec');
|
|
177
|
+
const spec = specFile ? readFileSync(specFile, 'utf8').slice(0, 4000) : null;
|
|
178
|
+
const prompt = buildReviewPrompt({ diff: diff.slice(0, 24000), spec });
|
|
179
|
+
|
|
180
|
+
let res;
|
|
181
|
+
if (decision.provider === 'codex') {
|
|
182
|
+
const model = readArg(argv, '--model') || null; // null = whatever ~/.codex/config.toml names
|
|
183
|
+
console.error(`cross-model-review: reviewer=codex${model ? ' -m ' + model : ' (' + (decision.codex?.model || 'default model') + ')'} (cross-model red-team, read-only sandbox)`);
|
|
184
|
+
const r = await codexReview({ ...prompt, cwd: root, model, bin: process.env.GREAT_CTO_CODEX_BIN || 'codex' });
|
|
185
|
+
if (r.state !== 'ok') {
|
|
186
|
+
// The reason a human is shown is RANKED, not the first thing Codex said.
|
|
187
|
+
// A quota-exhausted review used to display "Skill descriptions were
|
|
188
|
+
// shortened…" — advisory noise that arrived first — while the sentence
|
|
189
|
+
// naming the cause and its reset date was truncated away.
|
|
190
|
+
const principal = principalError(r.errors);
|
|
191
|
+
const reason = principal ? `${principal.kind}: ${principal.why}` : 'no answer';
|
|
192
|
+
console.log(`cross-model-review: SKIPPED (codex ${r.state}) — ${reason}`);
|
|
193
|
+
if (principal && r.errors.length > 1) {
|
|
194
|
+
console.log(` (${r.errors.length - 1} other message(s) from codex, not the cause)`);
|
|
195
|
+
}
|
|
196
|
+
log({
|
|
197
|
+
provider: 'codex', model: r.model ?? decision.codex?.model, state: r.state,
|
|
198
|
+
verdict: null, findings: null, cost: null,
|
|
199
|
+
error_kind: principal?.kind ?? null, resets_at: principal?.resetsAt ?? null,
|
|
200
|
+
});
|
|
201
|
+
process.exit(EXIT.SKIPPED);
|
|
202
|
+
}
|
|
203
|
+
res = { text: r.text, usage: r.usage, model: r.model ?? decision.codex?.model ?? 'codex' };
|
|
204
|
+
} else {
|
|
205
|
+
const model = readArg(argv, '--model') || pickReviewerModel();
|
|
206
|
+
console.error(`cross-model-review: reviewer=${model} (cross-model red-team via OpenRouter)`);
|
|
207
|
+
res = await callOpenRouter({ apiKey: process.env.OPENROUTER_API_KEY, model, ...prompt });
|
|
208
|
+
}
|
|
209
|
+
|
|
210
|
+
const { findings, verdict } = parseFindings(res.text);
|
|
211
|
+
// Unpriced is null, not zero. The first real Codex review logged `cost: 0`
|
|
212
|
+
// for gpt-5.6-terra — a model the price table does not carry — because usage
|
|
213
|
+
// was present and costForUsage prices an unknown model at nothing. A reviewer
|
|
214
|
+
// that reads as free is the defect this repository has removed twice already.
|
|
215
|
+
const priced = resolvePrice(res.model).price != null;
|
|
216
|
+
const cost = res.usage && priced ? round4(costForUsage({ model: res.model, usage: res.usage })) : null;
|
|
217
|
+
|
|
218
|
+
for (const f of findings) console.log(` ${f.severity} ${f.file}:${f.line} — ${f.issue}`);
|
|
219
|
+
console.log(`\ncross-model-review (${decision.provider}:${res.model}): ${findings.length} finding(s), VERDICT: ${verdict} (${cost == null ? 'cost unpriced' : '$' + cost})`);
|
|
220
|
+
log({ provider: decision.provider, model: res.model, state: 'ok', verdict, findings, cost });
|
|
221
|
+
process.exit(verdict === 'BLOCK' ? EXIT.BLOCK : EXIT.PASS);
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
const isMain = process.argv[1] && fileURLToPath(import.meta.url) === process.argv[1];
|
|
225
|
+
if (isMain) main(process.argv.slice(2)).catch(e => { console.error('FATAL:', e.message); process.exit(EXIT.SKIPPED); });
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
// Some provider failures mean "try again". Others mean "every remaining call
|
|
2
|
+
// will fail exactly like this one".
|
|
3
|
+
//
|
|
4
|
+
// What happened
|
|
5
|
+
// -------------
|
|
6
|
+
// A 75-file eval run spent $13.99, ran out of OpenRouter credits partway, and
|
|
7
|
+
// then made 147 more calls that could not possibly succeed — one per remaining
|
|
8
|
+
// case, each returning the same 402. The dropout gate did its job at the end and
|
|
9
|
+
// reported thirteen files as NOT MEASURED rather than as scores.
|
|
10
|
+
//
|
|
11
|
+
// But the run had already written those thirteen files into
|
|
12
|
+
// `results-history.jsonl` with `rate: 0`, and the drift detector reads `rate`.
|
|
13
|
+
// So the loop's next comparison would have read thirteen evals as having
|
|
14
|
+
// collapsed from ~0.85 to 0.00 overnight, and alarmed on a regression that is
|
|
15
|
+
// really an empty wallet.
|
|
16
|
+
//
|
|
17
|
+
// A run that did not happen recorded as a score of zero. Same defect this
|
|
18
|
+
// repository keeps finding, this time between two of its own components.
|
|
19
|
+
//
|
|
20
|
+
// So: recognise the terminal states, stop the run at the first one, and keep the
|
|
21
|
+
// unrunnable files out of the history entirely.
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* What kind of failure this is, from the error a provider call threw.
|
|
25
|
+
*
|
|
26
|
+
* The distinction that matters is not the status code but whether waiting or
|
|
27
|
+
* retrying could change the answer. 429 is the provider saying "slow down" —
|
|
28
|
+
* that resolves. 402 is the provider saying "you have no money" — that resolves
|
|
29
|
+
* only by someone topping up, which will not happen inside this run.
|
|
30
|
+
*
|
|
31
|
+
* @param {Error|string} err
|
|
32
|
+
* @returns {{terminal:boolean, kind:'credits'|'billing'|'quota'|'auth'|'rate-limit'|'transient',
|
|
33
|
+
* why:string, resetsAt?:string}}
|
|
34
|
+
*/
|
|
35
|
+
export function classifyProviderError(err) {
|
|
36
|
+
const msg = String(err?.message ?? err ?? '');
|
|
37
|
+
|
|
38
|
+
// Match the status as a distinct token so a `402` inside a response body — an
|
|
39
|
+
// id, a byte count — does not read as the status of the call itself.
|
|
40
|
+
const status = msg.match(/\bAPI\s+(\d{3})\b/)?.[1] ?? msg.match(/\b(4\d{2}|5\d{2})\b/)?.[1] ?? null;
|
|
41
|
+
const body = msg.toLowerCase();
|
|
42
|
+
|
|
43
|
+
if (status === '402' || /insufficient (credit|balance|fund)|no credits|out of credits|payment required/.test(body)) {
|
|
44
|
+
return { terminal: true, kind: 'credits', why: 'the provider account is out of credits — every remaining call fails identically until someone tops it up' };
|
|
45
|
+
}
|
|
46
|
+
// An account locked over billing is terminal, and it is NOT the same state as
|
|
47
|
+
// an empty balance. This repository's own GitHub Actions have been refused with
|
|
48
|
+
// this exact message since 2026-06-25 — one hundred consecutive runs, each
|
|
49
|
+
// failing identically, none able to succeed until a human settles a bill.
|
|
50
|
+
// Classified as transient it would earn a retry every time, which is the 402
|
|
51
|
+
// mistake this module exists to prevent, wearing different words.
|
|
52
|
+
//
|
|
53
|
+
// Kept apart from `credits` deliberately: topping up a balance and unlocking an
|
|
54
|
+
// account are different actions by possibly different people, and a message
|
|
55
|
+
// that merges them sends someone to the wrong screen.
|
|
56
|
+
if (/account is locked|billing (issue|problem|lock)|locked due to.*billing|billing.*(suspend|disabled)/.test(body)) {
|
|
57
|
+
return { terminal: true, kind: 'billing', why: 'the provider account is locked over billing — no retry clears it until a human settles the bill' };
|
|
58
|
+
}
|
|
59
|
+
// A PLAN QUOTA is its own kind, and merging it into rate-limit was costing a
|
|
60
|
+
// real answer. Codex on a ChatGPT plan answers "You've hit your usage limit.
|
|
61
|
+
// Upgrade to Plus to continue, or try again at Oct 5th, 2026 9:41 AM" — which
|
|
62
|
+
// is terminal for anything running today and NOT terminal in the way `credits`
|
|
63
|
+
// is: nobody has to do anything, it comes back by itself, on a date the
|
|
64
|
+
// message names. Read as `rate-limit` it earns a retry loop that cannot
|
|
65
|
+
// succeed for a month; read as `credits` it sends someone to a billing page
|
|
66
|
+
// they do not need.
|
|
67
|
+
//
|
|
68
|
+
// The date is the actionable half, so it is extracted rather than described.
|
|
69
|
+
if (/usage limit|quota (exceeded|exhausted)|monthly limit|plan limit/.test(body)) {
|
|
70
|
+
const at = msg.match(/try again at ([^.\n"]{4,40})/i)?.[1]?.trim() ?? null;
|
|
71
|
+
return {
|
|
72
|
+
terminal: true, kind: 'quota',
|
|
73
|
+
why: at
|
|
74
|
+
? `the provider plan's usage limit is spent — it returns on its own at ${at}, and no retry before then can succeed`
|
|
75
|
+
: "the provider plan's usage limit is spent — it returns on its own, and no retry before then can succeed",
|
|
76
|
+
...(at ? { resetsAt: at } : {}),
|
|
77
|
+
};
|
|
78
|
+
}
|
|
79
|
+
if (status === '401' || status === '403' || /invalid api key|unauthorized|forbidden/.test(body)) {
|
|
80
|
+
return { terminal: true, kind: 'auth', why: 'the provider rejected the key — no retry inside this run can fix that' };
|
|
81
|
+
}
|
|
82
|
+
if (status === '429' || /rate.?limit|too many requests/.test(body)) {
|
|
83
|
+
return { terminal: false, kind: 'rate-limit', why: 'rate limited — this resolves on its own' };
|
|
84
|
+
}
|
|
85
|
+
return { terminal: false, kind: 'transient', why: msg.slice(0, 160) || 'unclassified provider error' };
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/**
|
|
89
|
+
* What to print when a run gives up.
|
|
90
|
+
*
|
|
91
|
+
* Names the money spent, because the next question anybody asks is "did I pay
|
|
92
|
+
* for that", and says plainly that the remaining files were not measured rather
|
|
93
|
+
* than letting a reader infer a result from a truncated table.
|
|
94
|
+
*/
|
|
95
|
+
export function exhaustionReport({ kind, why, completed, total, costUsd }) {
|
|
96
|
+
const spent = typeof costUsd === 'number' ? `$${costUsd.toFixed(2)}` : 'an unrecorded amount';
|
|
97
|
+
return [
|
|
98
|
+
`RUN STOPPED — ${kind}: ${why}`,
|
|
99
|
+
` ${completed} of ${total} eval file(s) completed; ${spent} spent.`,
|
|
100
|
+
` The rest were NOT MEASURED. They are not zeros, and they are not written to`,
|
|
101
|
+
` history — a run that did not happen must not become a data point.`,
|
|
102
|
+
` Re-run once the account is funded.`,
|
|
103
|
+
].join('\n');
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Should this result be allowed into the trend history?
|
|
108
|
+
*
|
|
109
|
+
* A file whose cases never reached the provider has a rate computed over the
|
|
110
|
+
* prefix that did run, which is not a draw from the case list. `eval-power`
|
|
111
|
+
* already refuses to compare it against a threshold; this refuses to let it
|
|
112
|
+
* become tomorrow's baseline.
|
|
113
|
+
*/
|
|
114
|
+
export function admissibleToHistory(result) {
|
|
115
|
+
if (!result) return { ok: false, why: 'no result' };
|
|
116
|
+
if (result.dropout?.severe) {
|
|
117
|
+
return { ok: false, why: `dropout: ${result.dropout.why ?? 'the run stopped partway through this file'}` };
|
|
118
|
+
}
|
|
119
|
+
if (!result.judged) return { ok: false, why: 'no case was judged' };
|
|
120
|
+
return { ok: true };
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Pick the error a human should be shown, out of everything a provider emitted.
|
|
125
|
+
*
|
|
126
|
+
* Codex reports advisory problems and fatal ones through the same channel, in
|
|
127
|
+
* arrival order. On 2026-09-06 a review that failed on an exhausted plan quota
|
|
128
|
+
* displayed "Skill descriptions were shortened to fit the skills context
|
|
129
|
+
* budget" — the first error in the array, and pure noise — while the sentence
|
|
130
|
+
* naming the cause and its reset date sat second and was cut off by a 300-char
|
|
131
|
+
* truncation. The reader is then sent to disable skills over a quota problem.
|
|
132
|
+
*
|
|
133
|
+
* Terminal beats non-terminal; among terminal, the earliest listed wins. An
|
|
134
|
+
* empty list is `null`, not an invented reason.
|
|
135
|
+
*
|
|
136
|
+
* @param {string[]} errors
|
|
137
|
+
* @returns {{message:string, kind:string, why:string, terminal:boolean, resetsAt?:string}|null}
|
|
138
|
+
*/
|
|
139
|
+
export function principalError(errors) {
|
|
140
|
+
const list = (Array.isArray(errors) ? errors : []).filter((e) => String(e ?? '').trim());
|
|
141
|
+
if (!list.length) return null;
|
|
142
|
+
const RANK = { credits: 0, billing: 0, auth: 0, quota: 0, 'rate-limit': 1, transient: 2 };
|
|
143
|
+
let best = null;
|
|
144
|
+
for (const [i, message] of list.entries()) {
|
|
145
|
+
const c = classifyProviderError(message);
|
|
146
|
+
const score = [RANK[c.kind] ?? 2, i];
|
|
147
|
+
if (!best || score[0] < best.score[0] || (score[0] === best.score[0] && score[1] < best.score[1])) {
|
|
148
|
+
best = { score, value: { message: String(message), ...c } };
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
return best.value;
|
|
152
|
+
}
|
package/package.json
CHANGED