@rigour-labs/core 6.7.8 → 6.7.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -12,6 +12,11 @@ export interface ApiJudgeOptions {
|
|
|
12
12
|
roots: string[];
|
|
13
13
|
reasoning?: Reasoning;
|
|
14
14
|
fetchImpl?: typeof fetch;
|
|
15
|
+
/** The review's input files, given inline in the first message: a model that never calls a tool still has what it needs. */
|
|
16
|
+
inputs?: Array<{
|
|
17
|
+
path: string;
|
|
18
|
+
text: string;
|
|
19
|
+
}>;
|
|
15
20
|
}
|
|
16
21
|
export interface ApiJudgeRun {
|
|
17
22
|
exitCode: number;
|
|
@@ -11,6 +11,10 @@ import path from 'path';
|
|
|
11
11
|
import { execa } from 'execa';
|
|
12
12
|
const SYSTEM = 'You review code with read-only tools. Read the files the task names with read_file (the review inputs are named by absolute path), search the repository with search, read history with git. When you are done, reply with the final answer the task asks for and nothing else.';
|
|
13
13
|
const MAX_RESULT_CHARS = 60_000;
|
|
14
|
+
/** Output tokens a turn may use, reasoning included: a review verdict is long, and a reasoning model's thinking counts against the default budget. */
|
|
15
|
+
const MAX_OUTPUT_TOKENS = 32_000;
|
|
16
|
+
/** Characters of the inputs given inline in the first message; the rest stays in the files, for the tools. */
|
|
17
|
+
const MAX_INLINE_CHARS = 240_000;
|
|
14
18
|
const GIT_ALLOWED = new Set(['log', 'show', 'diff', 'blame', 'grep', 'ls-files', 'rev-parse', 'merge-base']);
|
|
15
19
|
/** Git options that write, or point git at another repository. */
|
|
16
20
|
const GIT_REFUSED = /^(--output|-o$|--git-dir|--work-tree|-C$|--exec-path|-c$|--config-env)/;
|
|
@@ -24,7 +28,7 @@ const n = (v) => (typeof v === 'number' && Number.isFinite(v) ? v : 0);
|
|
|
24
28
|
export async function runApiJudge(prompt, o) {
|
|
25
29
|
const fetchImpl = o.fetchImpl ?? fetch;
|
|
26
30
|
const deadline = Date.now() + o.timeoutMs;
|
|
27
|
-
const messages = [{ role: 'system', content: SYSTEM }, { role: 'user', content: prompt }];
|
|
31
|
+
const messages = [{ role: 'system', content: SYSTEM }, { role: 'user', content: withInputs(prompt, o.inputs ?? []) }];
|
|
28
32
|
const usage = { input: 0, cacheRead: 0, cacheWrite: 0, output: 0 };
|
|
29
33
|
const calls = [];
|
|
30
34
|
let cost;
|
|
@@ -40,7 +44,7 @@ export async function runApiJudge(prompt, o) {
|
|
|
40
44
|
response = await fetchImpl(`${o.url.replace(/\/$/, '')}/chat/completions`, {
|
|
41
45
|
method: 'POST',
|
|
42
46
|
headers: { 'content-type': 'application/json', authorization: `Bearer ${o.key}` },
|
|
43
|
-
body: JSON.stringify({ model: o.model, messages, tools: TOOLS, tool_choice: 'auto', ...(o.reasoning ? { reasoning_effort: o.reasoning } : {}) }),
|
|
47
|
+
body: JSON.stringify({ model: o.model, messages, tools: TOOLS, tool_choice: 'auto', max_tokens: MAX_OUTPUT_TOKENS, ...(o.reasoning ? { reasoning_effort: o.reasoning } : {}) }),
|
|
44
48
|
signal: controller.signal,
|
|
45
49
|
});
|
|
46
50
|
}
|
|
@@ -66,13 +70,22 @@ export async function runApiJudge(prompt, o) {
|
|
|
66
70
|
usage.output += n(u.completion_tokens);
|
|
67
71
|
if (typeof u.cost === 'number')
|
|
68
72
|
cost = (cost ?? 0) + u.cost;
|
|
69
|
-
const
|
|
73
|
+
const choice = body.choices?.[0];
|
|
74
|
+
// A gateway reports a provider's failure inside the choice, with the choice's finish_reason "error": say what it said.
|
|
75
|
+
if (choice?.error || body.error)
|
|
76
|
+
return fail(`the API reported an error: ${JSON.stringify(choice?.error ?? body.error).slice(0, 300)}`);
|
|
77
|
+
const message = choice?.message;
|
|
70
78
|
if (!message)
|
|
71
79
|
return fail('no choices in the answer');
|
|
72
|
-
|
|
80
|
+
// Echo back what the API needs to continue (reasoning_details carries a reasoning model's chain), not the reasoning prose.
|
|
81
|
+
messages.push({ role: 'assistant', content: message.content ?? null, ...(message.tool_calls ? { tool_calls: message.tool_calls } : {}), ...(message.reasoning_details ? { reasoning_details: message.reasoning_details } : {}) });
|
|
73
82
|
const toolCalls = Array.isArray(message.tool_calls) ? message.tool_calls : [];
|
|
74
83
|
if (toolCalls.length === 0) {
|
|
75
|
-
const
|
|
84
|
+
const text = String(message.content ?? '').trim();
|
|
85
|
+
// An empty final message is not an answer: say why (the output budget ran out, a refusal), never report it as one.
|
|
86
|
+
if (!text)
|
|
87
|
+
return fail(`an empty answer on turn ${turn} (finish_reason ${String(body.choices?.[0]?.finish_reason ?? 'unknown')}${message.refusal ? `, refusal: ${String(message.refusal).slice(0, 120)}` : ''})`);
|
|
88
|
+
const answer = { result: text, usage, ...(cost !== undefined ? { cost_usd: cost } : {}), trace: { turns: turn, usage, calls } };
|
|
76
89
|
return { exitCode: 0, stdout: JSON.stringify(answer), stderr: '' };
|
|
77
90
|
}
|
|
78
91
|
for (const call of toolCalls) {
|
|
@@ -83,6 +96,22 @@ export async function runApiJudge(prompt, o) {
|
|
|
83
96
|
}
|
|
84
97
|
return fail(`no answer within ${o.maxTurns} turns`);
|
|
85
98
|
}
|
|
99
|
+
/**
|
|
100
|
+
* The prompt with the inputs it names appended inline, smallest first so the diff takes what budget is left: a judge
|
|
101
|
+
* is fed, not left to decide whether to read. What does not fit is cut with a note naming the file to read.
|
|
102
|
+
*/
|
|
103
|
+
function withInputs(prompt, inputs) {
|
|
104
|
+
if (inputs.length === 0)
|
|
105
|
+
return prompt;
|
|
106
|
+
const ordered = [...inputs].sort((a, b) => a.text.length - b.text.length);
|
|
107
|
+
let left = MAX_INLINE_CHARS;
|
|
108
|
+
const parts = ordered.map(input => {
|
|
109
|
+
const text = input.text.length > left ? `${input.text.slice(0, Math.max(0, left))}\n…[cut here: read ${input.path} with read_file for the rest]` : input.text;
|
|
110
|
+
left = Math.max(0, left - input.text.length);
|
|
111
|
+
return `### ${input.path}\n${text}`;
|
|
112
|
+
});
|
|
113
|
+
return `${prompt}\n\nThe inputs named above, inline (the files are also there for your tools):\n\n${parts.join('\n\n')}`;
|
|
114
|
+
}
|
|
86
115
|
/** One tool call, inside the allowed roots only; the trace names tools as the CLI judges do (Read, Grep, Glob, Bash). */
|
|
87
116
|
async function runTool(call, o) {
|
|
88
117
|
const name = String(call?.function?.name ?? '');
|
|
@@ -78,6 +78,27 @@ describe('the API judge', () => {
|
|
|
78
78
|
const slow = (async (_u, init) => new Promise((_r, reject) => init.signal.addEventListener('abort', () => reject(new Error('aborted')))));
|
|
79
79
|
expect(await runApiJudge('p', options(slow, { timeoutMs: 50 }))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('request failed') });
|
|
80
80
|
});
|
|
81
|
+
it('fails closed on an empty final message, saying why, and echoes back only what the API needs to continue', async () => {
|
|
82
|
+
const empty = (async () => new Response(JSON.stringify({ choices: [{ message: { role: 'assistant', content: '', reasoning: 'thinking…' }, finish_reason: 'length' }], usage: {} }), { status: 200 }));
|
|
83
|
+
expect(await runApiJudge('p', options(empty))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('an empty answer on turn 1 (finish_reason length)') });
|
|
84
|
+
const providerError = (async () => new Response(JSON.stringify({ choices: [{ message: { role: 'assistant', content: '' }, finish_reason: 'error', error: { message: 'Provider returned error', code: 502 } }], usage: {} }), { status: 200 }));
|
|
85
|
+
expect(await runApiJudge('p', options(providerError))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('the API reported an error: {"message":"Provider returned error","code":502}') });
|
|
86
|
+
const { fetchImpl, seen } = model([{ tools: [{ name: 'list_dir', args: { path: '.' } }] }, { text: 'ok' }]);
|
|
87
|
+
await runApiJudge('p', options(fetchImpl));
|
|
88
|
+
const echoed = seen[1].messages.find((m) => m.role === 'assistant');
|
|
89
|
+
expect(Object.keys(echoed).sort()).toEqual(['content', 'role', 'tool_calls']); // no reasoning prose sent back
|
|
90
|
+
expect(seen[0].max_tokens).toBe(32000);
|
|
91
|
+
});
|
|
92
|
+
it('gives the judge its inputs inline, smallest first, the diff cut with a note when the budget runs out', async () => {
|
|
93
|
+
const { fetchImpl, seen } = model([{ text: 'ok' }]);
|
|
94
|
+
const inputs = [{ path: '/w/full.diff', text: 'x'.repeat(300_000) }, { path: '/w/pr-description.md', text: 'the description' }, { path: '/w/previous-reviews.md', text: 'the reviews' }];
|
|
95
|
+
await runApiJudge('review this', options(fetchImpl, { inputs }));
|
|
96
|
+
const first = seen[0].messages[1].content;
|
|
97
|
+
expect(first.startsWith('review this\n\nThe inputs named above, inline')).toBe(true);
|
|
98
|
+
expect(first.indexOf('### /w/previous-reviews.md')).toBeLessThan(first.indexOf('### /w/full.diff')); // smallest first
|
|
99
|
+
expect(first).toContain('…[cut here: read /w/full.diff with read_file for the rest]');
|
|
100
|
+
expect(first.length).toBeLessThan(241_000);
|
|
101
|
+
});
|
|
81
102
|
it('passes the reasoning effort when asked', async () => {
|
|
82
103
|
const { fetchImpl, seen } = model([{ text: 'ok' }]);
|
|
83
104
|
await runApiJudge('p', options(fetchImpl, { reasoning: 'low' }));
|
package/dist/review/reviewer.js
CHANGED
|
@@ -208,8 +208,9 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
208
208
|
return none(settings.required.panel || settings.required.mode ? 'unavailable' : 'skipped', over, { reviewers, scope, why, pr: pr?.number });
|
|
209
209
|
const work = fs.mkdtempSync(path.join(os.tmpdir(), 'rigour-reviewer-'));
|
|
210
210
|
// One judge run, by CLI or by API: the same prompt, the same cost accounting, the same trace.
|
|
211
|
+
let inlineInputs = [];
|
|
211
212
|
const runJudge = (name, prompt, model) => name === 'api'
|
|
212
|
-
? runApiJudge(prompt, { url: settings.api.url, model: settings.api.model, key: process.env[settings.api.key_env] ?? '', maxTurns: settings.api.max_turns, timeoutMs: settings.timeout_ms, cwd, roots: [cwd, work], ...(settings.reasoning[name] ? { reasoning: settings.reasoning[name] } : {}), ...(options.fetch ? { fetchImpl: options.fetch } : {}) })
|
|
213
|
+
? runApiJudge(prompt, { url: settings.api.url, model: settings.api.model, key: process.env[settings.api.key_env] ?? '', maxTurns: settings.api.max_turns, timeoutMs: settings.timeout_ms, cwd, roots: [cwd, work], inputs: inlineInputs, ...(settings.reasoning[name] ? { reasoning: settings.reasoning[name] } : {}), ...(options.fetch ? { fetchImpl: options.fetch } : {}) })
|
|
213
214
|
: exec(installed.get(name).binary, ADAPTERS[name].args(prompt, model, { reasoning: settings.reasoning[name] }), { cwd, timeoutMs: settings.timeout_ms, unset: judgeUnset(name, settings.judge_env) });
|
|
214
215
|
try {
|
|
215
216
|
const file = (name, text) => {
|
|
@@ -223,6 +224,7 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
223
224
|
const diffFile = file('full.diff', fullDiff);
|
|
224
225
|
const contextFile = file('team-knowledge.md', context.text);
|
|
225
226
|
const hintsFile = file('hints.txt', options.hints?.trim() || 'none\n');
|
|
227
|
+
inlineInputs = [[reviewsFile, reviews.markdown], [prBodyFile, body], [diffstatFile, await git(['diff', '--stat', `${baseSha}...HEAD`])], [diffFile, fullDiff], [contextFile, context.text], [hintsFile, options.hints?.trim() || 'none\n']].map(([p, text]) => ({ path: p, text }));
|
|
226
228
|
let delta = '';
|
|
227
229
|
// A reviewer must report on the human reviews, unless every point was settled by the previous verdict and is carried.
|
|
228
230
|
let needsPriorPoints = reviews.count > 0;
|
|
@@ -258,13 +260,29 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
258
260
|
return { run, answer, verdict: run.exitCode === 0 || answer.text.trim() ? parseVerdict(answer.text, needsPriorPoints, name, answer) : undefined };
|
|
259
261
|
};
|
|
260
262
|
let first = await ask();
|
|
261
|
-
//
|
|
262
|
-
if (first.verdict
|
|
263
|
-
progress(`Rigour reviewer: ${name} gave no valid verdict; asking once more`);
|
|
263
|
+
// No verdict, whether a malformed answer or a run that died, is a slip, not a decision: asked once more, inside the caps.
|
|
264
|
+
if ((!first.verdict || 'error' in first.verdict) && !overBudget(store.spend(), settings, 1)) {
|
|
265
|
+
progress(`Rigour reviewer: ${name} gave no ${first.verdict ? 'valid verdict' : 'answer'}; asking once more`);
|
|
264
266
|
first = await ask();
|
|
265
267
|
}
|
|
266
268
|
return first.verdict ?? { error: `${name}: no answer (exit ${first.run.exitCode}): ${first.run.stderr.trim().slice(-200)}` };
|
|
267
269
|
}));
|
|
270
|
+
// A judge that gives nothing is replaced by the next one installed, so the boundary stays up: a review ends unavailable only when every judge failed.
|
|
271
|
+
const spare = candidates.filter(c => installed.has(c) && !reviewers.includes(c));
|
|
272
|
+
for (let i = 0; i < answers.length; i++) {
|
|
273
|
+
let answer = answers[i];
|
|
274
|
+
while ('error' in answer && spare.length && !overBudget(store.spend(), settings, 1)) {
|
|
275
|
+
const next = spare.shift();
|
|
276
|
+
progress(`Rigour reviewer: ${reviewers[i]} gave no verdict (${answer.error}); ${next} judges instead`);
|
|
277
|
+
modeRecord = { ...modeRecord, degraded: `${modeRecord.degraded ? `${modeRecord.degraded}; ` : ''}${reviewers[i]} gave no verdict, ${next} judged instead` };
|
|
278
|
+
reviewers[i] = next;
|
|
279
|
+
const run = await runJudge(next, prompt, modelFor(next));
|
|
280
|
+
const got = ADAPTERS[next].answer(run.stdout);
|
|
281
|
+
store.addSpend(1, got.costUsd);
|
|
282
|
+
answer = run.exitCode === 0 || got.text.trim() ? parseVerdict(got.text, needsPriorPoints, next, got) : { error: `${next}: no answer (exit ${run.exitCode}): ${run.stderr.trim().slice(-200)}` };
|
|
283
|
+
}
|
|
284
|
+
answers[i] = answer;
|
|
285
|
+
}
|
|
268
286
|
const failed = answers.find(a => 'error' in a);
|
|
269
287
|
if (failed && 'error' in failed)
|
|
270
288
|
return none('unavailable', failed.error, { reviewers, scope, why, pr: pr?.number });
|
|
@@ -170,7 +170,7 @@ describe('the reviewer', () => {
|
|
|
170
170
|
expect(hidden.outcome).toBe('passed');
|
|
171
171
|
const shown = await runReviewer(repo, 'main', config, fakes(() => JSON.stringify({ ...EMPTY, prior_points: [] }), seen), () => undefined, { pr: 42, reviewsBefore: '2026-10-04', force: true });
|
|
172
172
|
expect(seen.files['previous-reviews.md']).toContain('Review by senior');
|
|
173
|
-
expect(shown).toMatchObject({ outcome: 'unavailable', reason: '
|
|
173
|
+
expect(shown).toMatchObject({ outcome: 'unavailable', reason: expect.stringContaining('did not report on the human reviews') });
|
|
174
174
|
});
|
|
175
175
|
it('for a backtest, gives the description as it read at the review, never a later edit', async () => {
|
|
176
176
|
const versions = { lastEditedAt: '2026-10-06T00:00:00Z', body: 'today: refunds are issued by the nightly job', userContentEdits: { totalCount: 3, nodes: [
|
|
@@ -200,7 +200,7 @@ describe('the reviewer', () => {
|
|
|
200
200
|
});
|
|
201
201
|
it('never passes without a verdict: a crash, a malformed answer, an unreadable pull request or no installed reviewer', async () => {
|
|
202
202
|
const crashed = await runReviewer(repo, 'main', config, fakes(() => ({ exitCode: 1, stdout: '', stderr: 'API error' }), seenNow()), () => undefined);
|
|
203
|
-
expect(crashed).toMatchObject({ outcome: 'unavailable', reason: expect.stringContaining('
|
|
203
|
+
expect(crashed).toMatchObject({ outcome: 'unavailable', reason: expect.stringContaining('cursor: no answer (exit 1)'), mode: { degraded: expect.stringContaining('claude gave no verdict, cursor judged instead') } }); // asked twice, then the spare judge, which failed too
|
|
204
204
|
const prose = await runReviewer(repo, 'main', config, fakes(() => 'Looks good to me!', seenNow()), () => undefined);
|
|
205
205
|
expect(prose).toMatchObject({ outcome: 'unavailable', reason: expect.stringContaining('no valid verdict') });
|
|
206
206
|
const broken = async (command, args, options) => command === 'gh' && args[0] === 'pr' ? { exitCode: 1, stdout: '', stderr: 'HTTP 500' } : fakes(() => '', seenNow())(command, args, options);
|
|
@@ -283,16 +283,34 @@ describe('the reviewer', () => {
|
|
|
283
283
|
delete process.env.TEST_JUDGE_KEY;
|
|
284
284
|
}
|
|
285
285
|
});
|
|
286
|
+
it('replaces a judge that gives nothing with the next one installed, and says so', async () => {
|
|
287
|
+
const seen = seenNow();
|
|
288
|
+
const silent = (async () => new Response(JSON.stringify({ choices: [{ message: { role: 'assistant', content: '' }, finish_reason: 'stop' }], usage: { prompt_tokens: 5, completion_tokens: 0 } }), { status: 200 }));
|
|
289
|
+
const twoJudges = ConfigSchema.parse({ version: 1, review: { reviewer: { enabled: true, reviewers: ['api', 'claude'], api: { url: 'https://example.test/v1', model: 'silent-model', key_env: 'TEST_JUDGE_KEY' } } } });
|
|
290
|
+
process.env.TEST_JUDGE_KEY = 'secret';
|
|
291
|
+
try {
|
|
292
|
+
const result = await runReviewer(repo, 'main', twoJudges, fakes(() => JSON.stringify(EMPTY), seen), () => undefined, { fetch: silent, force: true });
|
|
293
|
+
expect(result).toMatchObject({ outcome: 'passed', reviewers: ['claude'], mode: { degraded: expect.stringContaining('api gave no verdict, claude judged instead') } });
|
|
294
|
+
expect(seen.prompts).toHaveLength(1); // claude ran once, after the api judge's two empty answers
|
|
295
|
+
}
|
|
296
|
+
finally {
|
|
297
|
+
delete process.env.TEST_JUDGE_KEY;
|
|
298
|
+
}
|
|
299
|
+
});
|
|
286
300
|
it('asks a judge once more after an answer that is not a verdict, and is unavailable only when the second is not one either', async () => {
|
|
287
301
|
const seen = seenNow();
|
|
288
302
|
let calls = 0;
|
|
289
303
|
const slipOnce = await runReviewer(repo, 'main', config, fakes(() => (++calls === 1 ? '{"prior_points":[], "findings":[{"class"' : JSON.stringify(EMPTY)), seen), () => undefined, { force: true });
|
|
290
304
|
expect(slipOnce.outcome).toBe('passed');
|
|
291
305
|
expect(seen.prompts).toHaveLength(2);
|
|
306
|
+
let crashes = 0;
|
|
307
|
+
const crashOnce = seenNow();
|
|
308
|
+
const recovered = await runReviewer(repo, 'main', config, fakes(() => (++crashes === 1 ? { exitCode: 1, stdout: '', stderr: 'API error' } : JSON.stringify(EMPTY)), crashOnce), () => undefined, { force: true });
|
|
309
|
+
expect(recovered.outcome).toBe('passed'); // a run that died is asked once more too
|
|
292
310
|
const twice = seenNow();
|
|
293
311
|
const slipTwice = await runReviewer(repo, 'main', config, fakes(() => 'not json', twice), () => undefined, { force: true });
|
|
294
312
|
expect(slipTwice).toMatchObject({ outcome: 'unavailable', reason: expect.stringContaining('no valid verdict') });
|
|
295
|
-
expect(twice.prompts).toHaveLength(
|
|
313
|
+
expect(twice.prompts).toHaveLength(3); // once more, then the spare judge once: never a loop
|
|
296
314
|
});
|
|
297
315
|
it('says what was asked and that nothing ran when a review ends early, with why a judge is missing', async () => {
|
|
298
316
|
const seen = { ...seenNow(), installed: ['claude'] }; // cursor is listed but not installed
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@rigour-labs/core",
|
|
3
|
-
"version": "6.7.
|
|
3
|
+
"version": "6.7.9",
|
|
4
4
|
"description": "Rigour's review engine: deterministic gates on changed lines, rules and lessons learned from your team's fixes, and per-check precision from what you fix versus dismiss, across TypeScript, JavaScript, Python, Go, Ruby and C#.",
|
|
5
5
|
"engines": {
|
|
6
6
|
"node": ">=22.13"
|
|
@@ -72,11 +72,11 @@
|
|
|
72
72
|
"@anthropic-ai/sdk": "^0.30.1",
|
|
73
73
|
"pg": "^8.16.3",
|
|
74
74
|
"openai": "^5.23.2",
|
|
75
|
-
"@rigour-labs/brain-darwin-arm64": "6.7.
|
|
76
|
-
"@rigour-labs/brain-
|
|
77
|
-
"@rigour-labs/brain-linux-x64": "6.7.
|
|
78
|
-
"@rigour-labs/brain-
|
|
79
|
-
"@rigour-labs/brain-win-x64": "6.7.
|
|
75
|
+
"@rigour-labs/brain-darwin-arm64": "6.7.9",
|
|
76
|
+
"@rigour-labs/brain-linux-arm64": "6.7.9",
|
|
77
|
+
"@rigour-labs/brain-linux-x64": "6.7.9",
|
|
78
|
+
"@rigour-labs/brain-darwin-x64": "6.7.9",
|
|
79
|
+
"@rigour-labs/brain-win-x64": "6.7.9"
|
|
80
80
|
},
|
|
81
81
|
"devDependencies": {
|
|
82
82
|
"@types/fs-extra": "^11.0.4",
|