great-cto 3.26.2 → 3.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,225 @@
1
+ // scripts/lib/cross-model-review.mjs — cross-model adversarial review (architect-loop R3).
2
+ //
3
+ // great_cto's reviews are Claude-on-Claude → same-model blind spots. This red-teams
4
+ // the diff with a DIFFERENT model via OpenRouter (default openai/gpt-5), flagging
5
+ // ONLY correctness / requirement / invariant gaps with file:line — no style. The
6
+ // code-reviewer agent merges these with its own findings for high-stakes changes.
7
+ //
8
+ // Pure (buildReviewPrompt / parseFindings / pickReviewerModel) is unit-tested with
9
+ // no network; the CLI does the live OpenRouter call.
10
+ //
11
+ // Usage:
12
+ // git diff main...HEAD | node scripts/lib/cross-model-review.mjs --diff -
13
+ // node scripts/lib/cross-model-review.mjs --diff /tmp/d.diff --spec docs/architecture/ARCH-x.md
14
+ // GREAT_CTO_CROSS_REVIEW_MODEL=google/gemini-2.5-pro node ... --diff -
15
+
16
+ import { readFileSync } from 'node:fs';
17
+ import { fileURLToPath } from 'node:url';
18
+ import { execFileSync } from 'node:child_process';
19
+ import { costForUsage, round4, resolvePrice } from './cost-meter.mjs';
20
+ import { resolveSecondOpinion, codexReview } from './second-opinion.mjs';
21
+ import { principalError } from './provider-exhaustion.mjs';
22
+ import { existsSync, appendFileSync, mkdirSync } from 'node:fs';
23
+ import { join } from 'node:path';
24
+
25
+ /**
26
+ * Exit codes. The first version had two: 0 for PASS and 1 for everything else —
27
+ * BLOCK, a missing API key, a dead network. So "the review blocked this" and
28
+ * "the review did not happen" were the same number to the agent reading it,
29
+ * and the instruction to "note the cross-model pass was skipped" rested on the
30
+ * agent noticing a stderr line. A skipped review now has its own code, and it
31
+ * is neither of the two that mean a verdict was reached.
32
+ */
33
+ export const EXIT = Object.freeze({ PASS: 0, BLOCK: 1, USAGE: 2, SKIPPED: 3 });
34
+
35
+ /**
36
+ * Which provider reviews, decided from three sources in a fixed order:
37
+ * an explicit `--provider`, then the project's `capabilities: second_opinion`,
38
+ * then — for compatibility with every script that set it — the OpenRouter env.
39
+ *
40
+ * Returns the resolver's four states plus `source`, so the log line can say
41
+ * why this provider and not another. Pure: `codex` and `env` are injected.
42
+ */
43
+ export function decideProvider({ argv = [], projectMd = '', codex = null, env = process.env } = {}) {
44
+ const forced = readArg(argv, '--provider');
45
+ if (forced) {
46
+ const r = resolveSecondOpinion({ projectMd: `capabilities:\n second_opinion: ${forced}\n`, codex, env });
47
+ return { ...r, source: '--provider' };
48
+ }
49
+ const fromProject = resolveSecondOpinion({ projectMd, codex, env });
50
+ if (fromProject.state !== 'undeclared') return { ...fromProject, source: 'PROJECT.md' };
51
+ if (env.OPENROUTER_API_KEY) {
52
+ return { state: 'declared', provider: 'openrouter', why: '', source: 'OPENROUTER_API_KEY (second_opinion undeclared)' };
53
+ }
54
+ return { ...fromProject, source: 'PROJECT.md' };
55
+ }
56
+
57
+ /**
58
+ * One line per review, so the board can show what the second opinion DID.
59
+ *
60
+ * `sha` (git HEAD) and `dirty` (working tree had uncommitted changes) are the
61
+ * diff-identity fields BRD-R2's reader keys on, to tell "this line covers the
62
+ * code you're looking at" from "it covered something else". Additive only:
63
+ * both default to `null` — never `undefined`, never omitted — so a line
64
+ * written before this field existed, and any caller that doesn't supply them,
65
+ * still serializes to the same shape a reader already knows how to parse.
66
+ */
67
+ export function reviewLogLine({ provider, model, state, verdict, findings, cost, source, error_kind, resets_at, sha, dirty }) {
68
+ return JSON.stringify({
69
+ ts: new Date().toISOString(), provider, model: model ?? null, state, verdict: verdict ?? null,
70
+ error_kind: error_kind ?? null, resets_at: resets_at ?? null,
71
+ findings: Array.isArray(findings) ? findings.length : null,
72
+ p0: Array.isArray(findings) ? findings.filter((f) => f.severity === 'P0').length : null,
73
+ cost: cost ?? null, source, sha: sha ?? null, dirty: dirty ?? null,
74
+ });
75
+ }
76
+
77
+ const OPENROUTER_API = 'https://openrouter.ai/api/v1/chat/completions';
78
+
79
+ /** A genuinely non-Claude reviewer model (cross-model). Override via env. */
80
+ export function pickReviewerModel(env = process.env) {
81
+ return env.GREAT_CTO_CROSS_REVIEW_MODEL || 'openai/gpt-5';
82
+ }
83
+
84
+ /** Build the red-team prompt. Calibrated: correctness/invariant only, file:line, no style. */
85
+ export function buildReviewPrompt({ diff, spec }) {
86
+ const system =
87
+ 'You are an adversarial code reviewer from a DIFFERENT model family than the author. ' +
88
+ 'Your job is to catch what a same-model reviewer would miss. Review ONLY for: ' +
89
+ 'correctness bugs, violated requirements, broken invariants, security holes, data loss. ' +
90
+ 'Do NOT report style, naming, or preferences. Ground every finding in the diff with file:line. ' +
91
+ 'Default to silence over a weak finding. ' +
92
+ 'Output ONE finding per line in EXACTLY this format:\n' +
93
+ '<file>:<line> | <P0|P1|P2> | <one-sentence concrete issue>\n' +
94
+ 'P0 = data loss / security / broken build or prod path. ' +
95
+ 'After the findings, output a final line: VERDICT: BLOCK (if any P0) or VERDICT: PASS.';
96
+ const user =
97
+ (spec ? `Spec / intent:\n${spec}\n\n` : '') +
98
+ `Diff under review:\n${diff}\n\n` +
99
+ `Report findings (file:line | severity | issue), then VERDICT:`;
100
+ return { system, user };
101
+ }
102
+
103
+ /** Parse the model's findings + verdict. */
104
+ export function parseFindings(text) {
105
+ const findings = [];
106
+ let verdict = null;
107
+ for (const raw of String(text).split('\n')) {
108
+ const line = raw.trim();
109
+ const v = line.match(/^VERDICT:\s*(BLOCK|PASS)/i);
110
+ if (v) { verdict = v[1].toUpperCase(); continue; }
111
+ // <file>:<line> | <SEV> | <issue>
112
+ const m = line.match(/^(.+?):(\d+)\s*\|\s*(P[012])\s*\|\s*(.+)$/i);
113
+ if (m) findings.push({ file: m[1].trim(), line: parseInt(m[2], 10), severity: m[3].toUpperCase(), issue: m[4].trim() });
114
+ }
115
+ // Derive verdict if the model omitted it: any P0 → BLOCK.
116
+ if (!verdict) verdict = findings.some(f => f.severity === 'P0') ? 'BLOCK' : 'PASS';
117
+ return { findings, verdict };
118
+ }
119
+
120
+ // ── CLI ───────────────────────────────────────────────────────────────────────
121
+
122
+ async function callOpenRouter({ apiKey, model, system, user }) {
123
+ const res = await fetch(OPENROUTER_API, {
124
+ method: 'POST',
125
+ headers: { Authorization: `Bearer ${apiKey}`, 'HTTP-Referer': 'https://greatcto.systems', 'X-Title': 'great_cto-xmodel-review', 'content-type': 'application/json' },
126
+ body: JSON.stringify({ model, max_tokens: 1200, temperature: 0, messages: [{ role: 'system', content: system }, { role: 'user', content: user }] }),
127
+ });
128
+ if (!res.ok) throw new Error(`OpenRouter ${res.status}: ${(await res.text()).slice(0, 200)}`);
129
+ const data = await res.json();
130
+ const u = data.usage || null;
131
+ return { text: data.choices?.[0]?.message?.content?.trim() || '', usage: u ? { input_tokens: u.prompt_tokens ?? 0, output_tokens: u.completion_tokens ?? 0 } : null, model };
132
+ }
133
+
134
+ function readArg(argv, name) { const i = argv.indexOf(name); return i > -1 ? argv[i + 1] : null; }
135
+
136
+ /**
137
+ * The tree's identity at review time — git HEAD and whether it was dirty.
138
+ * CLI-only: the pure `reviewLogLine` above never shells out; per this file's
139
+ * own pure/CLI split (see file header), the git call lives here and the
140
+ * result is injected. Returns nulls outside a git repo rather than throwing —
141
+ * "couldn't determine identity" is data for the log line, not a reason to
142
+ * fail the review.
143
+ */
144
+ function gitIdentity(cwd) {
145
+ try {
146
+ const sha = execFileSync('git', ['rev-parse', 'HEAD'], { cwd, encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'] }).trim();
147
+ const status = execFileSync('git', ['status', '--porcelain'], { cwd, encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'] });
148
+ return { sha: sha || null, dirty: status.trim().length > 0 };
149
+ } catch {
150
+ return { sha: null, dirty: null };
151
+ }
152
+ }
153
+
154
+ async function main(argv) {
155
+ const diffPath = readArg(argv, '--diff');
156
+ if (!diffPath) { console.error('Usage: cross-model-review.mjs --diff <file|-> [--spec <file>] [--model <slug>] [--provider codex|openrouter]'); process.exit(EXIT.USAGE); }
157
+
158
+ const root = readArg(argv, '--root') || process.cwd();
159
+ const mdPath = join(root, '.great_cto', 'PROJECT.md');
160
+ const projectMd = existsSync(mdPath) ? readFileSync(mdPath, 'utf8') : '';
161
+ const decision = decideProvider({ argv, projectMd });
162
+ const identity = gitIdentity(root);
163
+ const logPath = join(root, '.great_cto', 'cross-review.log');
164
+ const log = (rec) => { try { mkdirSync(join(root, '.great_cto'), { recursive: true }); appendFileSync(logPath, reviewLogLine({ ...rec, source: decision.source, sha: identity.sha, dirty: identity.dirty }) + '\n'); } catch { /* the log is evidence, not a gate */ } };
165
+
166
+ // Anything that is not a reviewer reviewing exits SKIPPED — not PASS, and not
167
+ // the BLOCK code either. The line says why, and the log keeps it.
168
+ if (decision.state !== 'declared') {
169
+ console.log(`cross-model-review: SKIPPED (${decision.state}) — ${decision.why}`);
170
+ log({ provider: decision.provider, state: decision.state, verdict: null, findings: null, cost: null });
171
+ process.exit(EXIT.SKIPPED);
172
+ }
173
+
174
+ const diff = diffPath === '-' ? readFileSync(0, 'utf8') : readFileSync(diffPath, 'utf8');
175
+ if (!diff.trim()) { console.log('cross-model-review: empty diff, nothing to review.'); process.exit(EXIT.PASS); }
176
+ const specFile = readArg(argv, '--spec');
177
+ const spec = specFile ? readFileSync(specFile, 'utf8').slice(0, 4000) : null;
178
+ const prompt = buildReviewPrompt({ diff: diff.slice(0, 24000), spec });
179
+
180
+ let res;
181
+ if (decision.provider === 'codex') {
182
+ const model = readArg(argv, '--model') || null; // null = whatever ~/.codex/config.toml names
183
+ console.error(`cross-model-review: reviewer=codex${model ? ' -m ' + model : ' (' + (decision.codex?.model || 'default model') + ')'} (cross-model red-team, read-only sandbox)`);
184
+ const r = await codexReview({ ...prompt, cwd: root, model, bin: process.env.GREAT_CTO_CODEX_BIN || 'codex' });
185
+ if (r.state !== 'ok') {
186
+ // The reason a human is shown is RANKED, not the first thing Codex said.
187
+ // A quota-exhausted review used to display "Skill descriptions were
188
+ // shortened…" — advisory noise that arrived first — while the sentence
189
+ // naming the cause and its reset date was truncated away.
190
+ const principal = principalError(r.errors);
191
+ const reason = principal ? `${principal.kind}: ${principal.why}` : 'no answer';
192
+ console.log(`cross-model-review: SKIPPED (codex ${r.state}) — ${reason}`);
193
+ if (principal && r.errors.length > 1) {
194
+ console.log(` (${r.errors.length - 1} other message(s) from codex, not the cause)`);
195
+ }
196
+ log({
197
+ provider: 'codex', model: r.model ?? decision.codex?.model, state: r.state,
198
+ verdict: null, findings: null, cost: null,
199
+ error_kind: principal?.kind ?? null, resets_at: principal?.resetsAt ?? null,
200
+ });
201
+ process.exit(EXIT.SKIPPED);
202
+ }
203
+ res = { text: r.text, usage: r.usage, model: r.model ?? decision.codex?.model ?? 'codex' };
204
+ } else {
205
+ const model = readArg(argv, '--model') || pickReviewerModel();
206
+ console.error(`cross-model-review: reviewer=${model} (cross-model red-team via OpenRouter)`);
207
+ res = await callOpenRouter({ apiKey: process.env.OPENROUTER_API_KEY, model, ...prompt });
208
+ }
209
+
210
+ const { findings, verdict } = parseFindings(res.text);
211
+ // Unpriced is null, not zero. The first real Codex review logged `cost: 0`
212
+ // for gpt-5.6-terra — a model the price table does not carry — because usage
213
+ // was present and costForUsage prices an unknown model at nothing. A reviewer
214
+ // that reads as free is the defect this repository has removed twice already.
215
+ const priced = resolvePrice(res.model).price != null;
216
+ const cost = res.usage && priced ? round4(costForUsage({ model: res.model, usage: res.usage })) : null;
217
+
218
+ for (const f of findings) console.log(` ${f.severity} ${f.file}:${f.line} — ${f.issue}`);
219
+ console.log(`\ncross-model-review (${decision.provider}:${res.model}): ${findings.length} finding(s), VERDICT: ${verdict} (${cost == null ? 'cost unpriced' : '$' + cost})`);
220
+ log({ provider: decision.provider, model: res.model, state: 'ok', verdict, findings, cost });
221
+ process.exit(verdict === 'BLOCK' ? EXIT.BLOCK : EXIT.PASS);
222
+ }
223
+
224
+ const isMain = process.argv[1] && fileURLToPath(import.meta.url) === process.argv[1];
225
+ if (isMain) main(process.argv.slice(2)).catch(e => { console.error('FATAL:', e.message); process.exit(EXIT.SKIPPED); });
@@ -0,0 +1,152 @@
1
+ // Some provider failures mean "try again". Others mean "every remaining call
2
+ // will fail exactly like this one".
3
+ //
4
+ // What happened
5
+ // -------------
6
+ // A 75-file eval run spent $13.99, ran out of OpenRouter credits partway, and
7
+ // then made 147 more calls that could not possibly succeed — one per remaining
8
+ // case, each returning the same 402. The dropout gate did its job at the end and
9
+ // reported thirteen files as NOT MEASURED rather than as scores.
10
+ //
11
+ // But the run had already written those thirteen files into
12
+ // `results-history.jsonl` with `rate: 0`, and the drift detector reads `rate`.
13
+ // So the loop's next comparison would have read thirteen evals as having
14
+ // collapsed from ~0.85 to 0.00 overnight, and alarmed on a regression that is
15
+ // really an empty wallet.
16
+ //
17
+ // A run that did not happen recorded as a score of zero. Same defect this
18
+ // repository keeps finding, this time between two of its own components.
19
+ //
20
+ // So: recognise the terminal states, stop the run at the first one, and keep the
21
+ // unrunnable files out of the history entirely.
22
+
23
+ /**
24
+ * What kind of failure this is, from the error a provider call threw.
25
+ *
26
+ * The distinction that matters is not the status code but whether waiting or
27
+ * retrying could change the answer. 429 is the provider saying "slow down" —
28
+ * that resolves. 402 is the provider saying "you have no money" — that resolves
29
+ * only by someone topping up, which will not happen inside this run.
30
+ *
31
+ * @param {Error|string} err
32
+ * @returns {{terminal:boolean, kind:'credits'|'billing'|'quota'|'auth'|'rate-limit'|'transient',
33
+ * why:string, resetsAt?:string}}
34
+ */
35
+ export function classifyProviderError(err) {
36
+ const msg = String(err?.message ?? err ?? '');
37
+
38
+ // Match the status as a distinct token so a `402` inside a response body — an
39
+ // id, a byte count — does not read as the status of the call itself.
40
+ const status = msg.match(/\bAPI\s+(\d{3})\b/)?.[1] ?? msg.match(/\b(4\d{2}|5\d{2})\b/)?.[1] ?? null;
41
+ const body = msg.toLowerCase();
42
+
43
+ if (status === '402' || /insufficient (credit|balance|fund)|no credits|out of credits|payment required/.test(body)) {
44
+ return { terminal: true, kind: 'credits', why: 'the provider account is out of credits — every remaining call fails identically until someone tops it up' };
45
+ }
46
+ // An account locked over billing is terminal, and it is NOT the same state as
47
+ // an empty balance. This repository's own GitHub Actions have been refused with
48
+ // this exact message since 2026-06-25 — one hundred consecutive runs, each
49
+ // failing identically, none able to succeed until a human settles a bill.
50
+ // Classified as transient it would earn a retry every time, which is the 402
51
+ // mistake this module exists to prevent, wearing different words.
52
+ //
53
+ // Kept apart from `credits` deliberately: topping up a balance and unlocking an
54
+ // account are different actions by possibly different people, and a message
55
+ // that merges them sends someone to the wrong screen.
56
+ if (/account is locked|billing (issue|problem|lock)|locked due to.*billing|billing.*(suspend|disabled)/.test(body)) {
57
+ return { terminal: true, kind: 'billing', why: 'the provider account is locked over billing — no retry clears it until a human settles the bill' };
58
+ }
59
+ // A PLAN QUOTA is its own kind, and merging it into rate-limit was costing a
60
+ // real answer. Codex on a ChatGPT plan answers "You've hit your usage limit.
61
+ // Upgrade to Plus to continue, or try again at Oct 5th, 2026 9:41 AM" — which
62
+ // is terminal for anything running today and NOT terminal in the way `credits`
63
+ // is: nobody has to do anything, it comes back by itself, on a date the
64
+ // message names. Read as `rate-limit` it earns a retry loop that cannot
65
+ // succeed for a month; read as `credits` it sends someone to a billing page
66
+ // they do not need.
67
+ //
68
+ // The date is the actionable half, so it is extracted rather than described.
69
+ if (/usage limit|quota (exceeded|exhausted)|monthly limit|plan limit/.test(body)) {
70
+ const at = msg.match(/try again at ([^.\n"]{4,40})/i)?.[1]?.trim() ?? null;
71
+ return {
72
+ terminal: true, kind: 'quota',
73
+ why: at
74
+ ? `the provider plan's usage limit is spent — it returns on its own at ${at}, and no retry before then can succeed`
75
+ : "the provider plan's usage limit is spent — it returns on its own, and no retry before then can succeed",
76
+ ...(at ? { resetsAt: at } : {}),
77
+ };
78
+ }
79
+ if (status === '401' || status === '403' || /invalid api key|unauthorized|forbidden/.test(body)) {
80
+ return { terminal: true, kind: 'auth', why: 'the provider rejected the key — no retry inside this run can fix that' };
81
+ }
82
+ if (status === '429' || /rate.?limit|too many requests/.test(body)) {
83
+ return { terminal: false, kind: 'rate-limit', why: 'rate limited — this resolves on its own' };
84
+ }
85
+ return { terminal: false, kind: 'transient', why: msg.slice(0, 160) || 'unclassified provider error' };
86
+ }
87
+
88
+ /**
89
+ * What to print when a run gives up.
90
+ *
91
+ * Names the money spent, because the next question anybody asks is "did I pay
92
+ * for that", and says plainly that the remaining files were not measured rather
93
+ * than letting a reader infer a result from a truncated table.
94
+ */
95
+ export function exhaustionReport({ kind, why, completed, total, costUsd }) {
96
+ const spent = typeof costUsd === 'number' ? `$${costUsd.toFixed(2)}` : 'an unrecorded amount';
97
+ return [
98
+ `RUN STOPPED — ${kind}: ${why}`,
99
+ ` ${completed} of ${total} eval file(s) completed; ${spent} spent.`,
100
+ ` The rest were NOT MEASURED. They are not zeros, and they are not written to`,
101
+ ` history — a run that did not happen must not become a data point.`,
102
+ ` Re-run once the account is funded.`,
103
+ ].join('\n');
104
+ }
105
+
106
+ /**
107
+ * Should this result be allowed into the trend history?
108
+ *
109
+ * A file whose cases never reached the provider has a rate computed over the
110
+ * prefix that did run, which is not a draw from the case list. `eval-power`
111
+ * already refuses to compare it against a threshold; this refuses to let it
112
+ * become tomorrow's baseline.
113
+ */
114
+ export function admissibleToHistory(result) {
115
+ if (!result) return { ok: false, why: 'no result' };
116
+ if (result.dropout?.severe) {
117
+ return { ok: false, why: `dropout: ${result.dropout.why ?? 'the run stopped partway through this file'}` };
118
+ }
119
+ if (!result.judged) return { ok: false, why: 'no case was judged' };
120
+ return { ok: true };
121
+ }
122
+
123
+ /**
124
+ * Pick the error a human should be shown, out of everything a provider emitted.
125
+ *
126
+ * Codex reports advisory problems and fatal ones through the same channel, in
127
+ * arrival order. On 2026-09-06 a review that failed on an exhausted plan quota
128
+ * displayed "Skill descriptions were shortened to fit the skills context
129
+ * budget" — the first error in the array, and pure noise — while the sentence
130
+ * naming the cause and its reset date sat second and was cut off by a 300-char
131
+ * truncation. The reader is then sent to disable skills over a quota problem.
132
+ *
133
+ * Terminal beats non-terminal; among terminal, the earliest listed wins. An
134
+ * empty list is `null`, not an invented reason.
135
+ *
136
+ * @param {string[]} errors
137
+ * @returns {{message:string, kind:string, why:string, terminal:boolean, resetsAt?:string}|null}
138
+ */
139
+ export function principalError(errors) {
140
+ const list = (Array.isArray(errors) ? errors : []).filter((e) => String(e ?? '').trim());
141
+ if (!list.length) return null;
142
+ const RANK = { credits: 0, billing: 0, auth: 0, quota: 0, 'rate-limit': 1, transient: 2 };
143
+ let best = null;
144
+ for (const [i, message] of list.entries()) {
145
+ const c = classifyProviderError(message);
146
+ const score = [RANK[c.kind] ?? 2, i];
147
+ if (!best || score[0] < best.score[0] || (score[0] === best.score[0] && score[1] < best.score[1])) {
148
+ best = { score, value: { message: String(message), ...c } };
149
+ }
150
+ }
151
+ return best.value;
152
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "great-cto",
3
- "version": "3.26.2",
3
+ "version": "3.27.0",
4
4
  "description": "One command install for the great_cto Claude Code plugin. Auto-detects your stack, picks the right archetype, bootstraps PROJECT.md.",
5
5
  "keywords": [
6
6
  "claude-code",