@rigour-labs/core 6.7.6 → 6.7.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,28 @@
1
+ import type { RunTrace } from './adapters.js';
2
+ export type Reasoning = 'low' | 'medium' | 'high';
3
+ export interface ApiJudgeOptions {
4
+ /** The API's base URL; `/chat/completions` is appended. */
5
+ url: string;
6
+ model: string;
7
+ key: string;
8
+ maxTurns: number;
9
+ timeoutMs: number;
10
+ cwd: string;
11
+ /** Where the tools may read: the checkout and the review's input folder. */
12
+ roots: string[];
13
+ reasoning?: Reasoning;
14
+ fetchImpl?: typeof fetch;
15
+ }
16
+ export interface ApiJudgeRun {
17
+ exitCode: number;
18
+ stdout: string;
19
+ stderr: string;
20
+ }
21
+ /** What a run's stdout carries on success, for the adapter to read. */
22
+ export interface ApiJudgeAnswer {
23
+ result: string;
24
+ usage: RunTrace['usage'];
25
+ cost_usd?: number;
26
+ trace: RunTrace;
27
+ }
28
+ export declare function runApiJudge(prompt: string, o: ApiJudgeOptions): Promise<ApiJudgeRun>;
@@ -0,0 +1,161 @@
1
+ /**
2
+ * A judge reached through a model API instead of an agent CLI, so any model a team can call works as
3
+ * a reviewer: OpenAI-compatible chat completions with tools (OpenAI, OpenRouter, a local server, other
4
+ * vendors through a gateway). Rigour runs the loop itself: the model asks for a read-only tool, Rigour
5
+ * runs it inside the checkout (and the review's own input folder) and hands the result back, until
6
+ * the model answers. The same prompt, the same evidence contract and the same trace as a CLI judge;
7
+ * the key comes from an environment variable named in rigour.yml, never from the file.
8
+ */
9
+ import fs from 'fs';
10
+ import path from 'path';
11
+ import { execa } from 'execa';
12
+ const SYSTEM = 'You review code with read-only tools. Read the files the task names with read_file (the review inputs are named by absolute path), search the repository with search, read history with git. When you are done, reply with the final answer the task asks for and nothing else.';
13
+ const MAX_RESULT_CHARS = 60_000;
14
+ const GIT_ALLOWED = new Set(['log', 'show', 'diff', 'blame', 'grep', 'ls-files', 'rev-parse', 'merge-base']);
15
+ /** Git options that write, or point git at another repository. */
16
+ const GIT_REFUSED = /^(--output|-o$|--git-dir|--work-tree|-C$|--exec-path|-c$|--config-env)/;
17
+ const TOOLS = [
18
+ { type: 'function', function: { name: 'read_file', description: 'Read a file in the repository or the review inputs, or a line range of it.', parameters: { type: 'object', properties: { path: { type: 'string' }, start_line: { type: 'integer' }, end_line: { type: 'integer' } }, required: ['path'] } } },
19
+ { type: 'function', function: { name: 'search', description: 'Search the tracked files for a regular expression (git grep -n), optionally under one path.', parameters: { type: 'object', properties: { pattern: { type: 'string' }, path: { type: 'string' } }, required: ['pattern'] } } },
20
+ { type: 'function', function: { name: 'list_dir', description: 'List a directory in the repository or the review inputs.', parameters: { type: 'object', properties: { path: { type: 'string' } }, required: ['path'] } } },
21
+ { type: 'function', function: { name: 'git', description: 'Run a read-only git command in the repository: log, show, diff, blame, grep, ls-files, rev-parse, merge-base.', parameters: { type: 'object', properties: { args: { type: 'array', items: { type: 'string' } } }, required: ['args'] } } },
22
+ ];
23
+ const n = (v) => (typeof v === 'number' && Number.isFinite(v) ? v : 0);
24
+ export async function runApiJudge(prompt, o) {
25
+ const fetchImpl = o.fetchImpl ?? fetch;
26
+ const deadline = Date.now() + o.timeoutMs;
27
+ const messages = [{ role: 'system', content: SYSTEM }, { role: 'user', content: prompt }];
28
+ const usage = { input: 0, cacheRead: 0, cacheWrite: 0, output: 0 };
29
+ const calls = [];
30
+ let cost;
31
+ const fail = (why) => ({ exitCode: 1, stdout: '', stderr: `api judge (${o.model}): ${why}` });
32
+ for (let turn = 1; turn <= o.maxTurns; turn++) {
33
+ const left = deadline - Date.now();
34
+ if (left <= 0)
35
+ return fail(`timed out after ${o.timeoutMs} ms`);
36
+ const controller = new AbortController();
37
+ const timer = setTimeout(() => controller.abort(), left);
38
+ let response;
39
+ try {
40
+ response = await fetchImpl(`${o.url.replace(/\/$/, '')}/chat/completions`, {
41
+ method: 'POST',
42
+ headers: { 'content-type': 'application/json', authorization: `Bearer ${o.key}` },
43
+ body: JSON.stringify({ model: o.model, messages, tools: TOOLS, tool_choice: 'auto', ...(o.reasoning ? { reasoning_effort: o.reasoning } : {}) }),
44
+ signal: controller.signal,
45
+ });
46
+ }
47
+ catch (error) {
48
+ return fail(`request failed: ${error instanceof Error ? error.message : String(error)}`);
49
+ }
50
+ finally {
51
+ clearTimeout(timer);
52
+ }
53
+ if (!response.ok)
54
+ return fail(`HTTP ${response.status}: ${(await response.text()).slice(0, 300)}`);
55
+ let body;
56
+ try {
57
+ body = await response.json();
58
+ }
59
+ catch {
60
+ return fail('the answer was not JSON');
61
+ }
62
+ const u = body.usage ?? {};
63
+ const cached = n(u.prompt_tokens_details?.cached_tokens);
64
+ usage.input += Math.max(0, n(u.prompt_tokens) - cached);
65
+ usage.cacheRead += cached;
66
+ usage.output += n(u.completion_tokens);
67
+ if (typeof u.cost === 'number')
68
+ cost = (cost ?? 0) + u.cost;
69
+ const message = body.choices?.[0]?.message;
70
+ if (!message)
71
+ return fail('no choices in the answer');
72
+ messages.push(message);
73
+ const toolCalls = Array.isArray(message.tool_calls) ? message.tool_calls : [];
74
+ if (toolCalls.length === 0) {
75
+ const answer = { result: String(message.content ?? ''), usage, ...(cost !== undefined ? { cost_usd: cost } : {}), trace: { turns: turn, usage, calls } };
76
+ return { exitCode: 0, stdout: JSON.stringify(answer), stderr: '' };
77
+ }
78
+ for (const call of toolCalls) {
79
+ const ran = await runTool(call, o);
80
+ calls.push({ turn, tool: ran.tool, target: ran.target, resultChars: ran.result.length });
81
+ messages.push({ role: 'tool', tool_call_id: call.id, content: ran.result });
82
+ }
83
+ }
84
+ return fail(`no answer within ${o.maxTurns} turns`);
85
+ }
86
+ /** One tool call, inside the allowed roots only; the trace names tools as the CLI judges do (Read, Grep, Glob, Bash). */
87
+ async function runTool(call, o) {
88
+ const name = String(call?.function?.name ?? '');
89
+ let args = {};
90
+ try {
91
+ args = JSON.parse(call?.function?.arguments || '{}');
92
+ }
93
+ catch {
94
+ return { tool: name, target: '', result: 'refused: the arguments were not JSON' };
95
+ }
96
+ const clip = (text) => (text.length > MAX_RESULT_CHARS ? `${text.slice(0, MAX_RESULT_CHARS)}\n…[truncated at ${MAX_RESULT_CHARS} characters]` : text);
97
+ if (name === 'read_file') {
98
+ const target = inside(String(args.path ?? ''), o);
99
+ if (!target)
100
+ return { tool: 'Read', target: String(args.path ?? ''), result: 'refused: outside the repository and the review inputs' };
101
+ try {
102
+ const lines = fs.readFileSync(target, 'utf8').split('\n');
103
+ const start = Math.max(1, Number(args.start_line) || 1);
104
+ const end = Math.min(lines.length, Number(args.end_line) || lines.length);
105
+ return { tool: 'Read', target, result: clip(lines.slice(start - 1, end).map((l, i) => `${start + i}: ${l}`).join('\n')) };
106
+ }
107
+ catch (error) {
108
+ return { tool: 'Read', target, result: `error: ${error instanceof Error ? error.message : String(error)}` };
109
+ }
110
+ }
111
+ if (name === 'list_dir') {
112
+ const target = inside(String(args.path ?? '.'), o);
113
+ if (!target)
114
+ return { tool: 'Glob', target: String(args.path ?? ''), result: 'refused: outside the repository and the review inputs' };
115
+ try {
116
+ return { tool: 'Glob', target, result: clip(fs.readdirSync(target, { withFileTypes: true }).map(e => (e.isDirectory() ? `${e.name}/` : e.name)).join('\n')) };
117
+ }
118
+ catch (error) {
119
+ return { tool: 'Glob', target, result: `error: ${error instanceof Error ? error.message : String(error)}` };
120
+ }
121
+ }
122
+ if (name === 'search') {
123
+ const pattern = String(args.pattern ?? '');
124
+ const under = args.path ? inside(String(args.path), o) : undefined;
125
+ if (args.path && !under)
126
+ return { tool: 'Grep', target: pattern, result: 'refused: outside the repository' };
127
+ const run = await execa('git', ['grep', '-n', '-I', '-e', pattern, ...(under ? ['--', path.relative(o.cwd, under) || '.'] : [])], { cwd: o.cwd, reject: false, maxBuffer: 16 * 1024 * 1024 });
128
+ return { tool: 'Grep', target: pattern, result: clip(run.stdout || (run.exitCode === 1 ? '(no match)' : run.stderr)) };
129
+ }
130
+ if (name === 'git') {
131
+ const gitArgs = Array.isArray(args.args) ? args.args.map(String) : [];
132
+ if (!GIT_ALLOWED.has(gitArgs[0] ?? '') || gitArgs.some(a => GIT_REFUSED.test(a)))
133
+ return { tool: 'Bash', target: `git ${gitArgs.join(' ')}`, result: 'refused: only read-only git commands, without options that write or point elsewhere' };
134
+ const run = await execa('git', ['--no-pager', ...gitArgs], { cwd: o.cwd, reject: false, maxBuffer: 16 * 1024 * 1024 });
135
+ return { tool: 'Bash', target: `git ${gitArgs.join(' ')}`, result: clip(run.stdout || run.stderr) };
136
+ }
137
+ return { tool: name, target: '', result: `refused: no tool named ${name}` };
138
+ }
139
+ /** The real path of `p` when it lies under one of the roots; undefined otherwise (a symlink out is outside). */
140
+ function inside(p, o) {
141
+ const resolved = path.resolve(o.cwd, p);
142
+ let real;
143
+ try {
144
+ real = fs.realpathSync(resolved);
145
+ }
146
+ catch {
147
+ return undefined;
148
+ }
149
+ for (const root of o.roots) {
150
+ let base;
151
+ try {
152
+ base = fs.realpathSync(root);
153
+ }
154
+ catch {
155
+ continue;
156
+ }
157
+ if (real === base || real.startsWith(base + path.sep))
158
+ return real;
159
+ }
160
+ return undefined;
161
+ }
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1,87 @@
1
+ import { execFileSync } from 'child_process';
2
+ import fs from 'fs';
3
+ import os from 'os';
4
+ import path from 'path';
5
+ import { afterEach, beforeEach, describe, expect, it } from 'vitest';
6
+ import { runApiJudge } from './api-judge.js';
7
+ let repo;
8
+ let work;
9
+ beforeEach(() => {
10
+ repo = fs.mkdtempSync(path.join(os.tmpdir(), 'api-judge-'));
11
+ work = fs.mkdtempSync(path.join(os.tmpdir(), 'api-judge-inputs-'));
12
+ execFileSync('git', ['-C', repo, 'init', '-q']);
13
+ fs.mkdirSync(path.join(repo, 'src'));
14
+ fs.writeFileSync(path.join(repo, 'src/job.ts'), 'export function job() {\n return 1;\n}\n');
15
+ execFileSync('git', ['-C', repo, 'add', '-A']);
16
+ execFileSync('git', ['-C', repo, '-c', 'user.email=t@x', '-c', 'user.name=t', 'commit', '-qm', 'init']);
17
+ fs.writeFileSync(path.join(work, 'full.diff'), '+++ b/src/job.ts\n');
18
+ });
19
+ afterEach(() => { for (const d of [repo, work])
20
+ fs.rmSync(d, { recursive: true, force: true }); });
21
+ /** A model that plays a scripted conversation: each entry is what it answers to the next request. */
22
+ function model(turns) {
23
+ const seen = [];
24
+ let i = 0;
25
+ const fetchImpl = (async (_url, init) => {
26
+ const body = JSON.parse(init.body);
27
+ seen.push(body);
28
+ const turn = turns[i++] ?? { text: 'done' };
29
+ if (turn.status)
30
+ return new Response('nope', { status: turn.status });
31
+ const message = turn.tools
32
+ ? { role: 'assistant', content: null, tool_calls: turn.tools.map((t, k) => ({ id: `c${i}-${k}`, type: 'function', function: { name: t.name, arguments: JSON.stringify(t.args) } })) }
33
+ : { role: 'assistant', content: turn.text };
34
+ return new Response(JSON.stringify({ choices: [{ message }], usage: turn.usage ?? { prompt_tokens: 100, completion_tokens: 10, prompt_tokens_details: { cached_tokens: 60 } } }), { status: 200, headers: { 'content-type': 'application/json' } });
35
+ });
36
+ return { fetchImpl, seen };
37
+ }
38
+ const options = (fetchImpl, extra = {}) => ({ url: 'https://example.test/v1', model: 'some-model', key: 'k', maxTurns: 10, timeoutMs: 30_000, cwd: repo, roots: [repo, work], fetchImpl, ...extra });
39
+ describe('the API judge', () => {
40
+ it('runs the tools the model asks for inside the checkout and the inputs, then returns its answer with the usage and the trace', async () => {
41
+ const { fetchImpl, seen } = model([
42
+ { tools: [{ name: 'read_file', args: { path: path.join(work, 'full.diff') } }, { name: 'read_file', args: { path: 'src/job.ts', start_line: 2, end_line: 2 } }] },
43
+ { tools: [{ name: 'search', args: { pattern: 'return' } }, { name: 'git', args: { args: ['log', '-1', '--format=%s'] } }, { name: 'list_dir', args: { path: 'src' } }] },
44
+ { text: '{"prior_points":[]}', usage: { prompt_tokens: 50, completion_tokens: 5, cost: 0.01 } },
45
+ ]);
46
+ const run = await runApiJudge('review this', options(fetchImpl));
47
+ expect(run.exitCode).toBe(0);
48
+ const answer = JSON.parse(run.stdout);
49
+ expect(answer.result).toBe('{"prior_points":[]}');
50
+ expect(answer.usage).toEqual({ input: 130, cacheRead: 120, cacheWrite: 0, output: 25 });
51
+ expect(answer.cost_usd).toBe(0.01);
52
+ expect(answer.trace.turns).toBe(3);
53
+ expect(answer.trace.calls.map(c => c.tool)).toEqual(['Read', 'Read', 'Grep', 'Bash', 'Glob']);
54
+ const toolResults = (request) => request.messages.filter((m) => m.role === 'tool').map((m) => m.content);
55
+ expect(toolResults(seen[1])).toEqual(['1: +++ b/src/job.ts\n2: ', '2: return 1;']); // turn 1's reads, as the model got them
56
+ expect(toolResults(seen[2]).slice(2)).toEqual(['src/job.ts:2: return 1;', 'init', 'job.ts']); // turn 2's search, git and listing
57
+ expect(seen[0].messages[0].role).toBe('system');
58
+ expect(seen[0].tools.map((t) => t.function.name)).toEqual(['read_file', 'search', 'list_dir', 'git']);
59
+ });
60
+ it('refuses to read outside the roots, a symlink out included, and refuses git that writes or points elsewhere', async () => {
61
+ fs.symlinkSync(os.homedir(), path.join(repo, 'out'));
62
+ const { fetchImpl, seen } = model([
63
+ { tools: [{ name: 'read_file', args: { path: '/etc/hosts' } }, { name: 'read_file', args: { path: 'out/.bashrc' } }, { name: 'git', args: { args: ['log', '--output=/tmp/x'] } }, { name: 'git', args: { args: ['push'] } }, { name: 'nope', args: {} }] },
64
+ { text: 'ok' },
65
+ ]);
66
+ await runApiJudge('p', options(fetchImpl));
67
+ const results = seen[1].messages.filter((m) => m.role === 'tool').map((m) => m.content);
68
+ expect(results).toEqual([
69
+ 'refused: outside the repository and the review inputs', 'refused: outside the repository and the review inputs',
70
+ 'refused: only read-only git commands, without options that write or point elsewhere', 'refused: only read-only git commands, without options that write or point elsewhere',
71
+ 'refused: no tool named nope',
72
+ ]);
73
+ });
74
+ it('fails closed on a turn cap, an HTTP error and a timeout, saying why', async () => {
75
+ const loop = model(Array.from({ length: 5 }, () => ({ tools: [{ name: 'list_dir', args: { path: '.' } }] })));
76
+ expect(await runApiJudge('p', options(loop.fetchImpl, { maxTurns: 3 }))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('no answer within 3 turns') });
77
+ expect(await runApiJudge('p', options(model([{ status: 429 }]).fetchImpl))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('HTTP 429') });
78
+ const slow = (async (_u, init) => new Promise((_r, reject) => init.signal.addEventListener('abort', () => reject(new Error('aborted')))));
79
+ expect(await runApiJudge('p', options(slow, { timeoutMs: 50 }))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('request failed') });
80
+ });
81
+ it('passes the reasoning effort when asked', async () => {
82
+ const { fetchImpl, seen } = model([{ text: 'ok' }]);
83
+ await runApiJudge('p', options(fetchImpl, { reasoning: 'low' }));
84
+ expect(seen[0].reasoning_effort).toBe('low');
85
+ expect(seen[0].model).toBe('some-model');
86
+ });
87
+ });
@@ -35,6 +35,16 @@ export interface ResolvedReviewer {
35
35
  judges: 2 | 3;
36
36
  escalate: 'always' | 'risk';
37
37
  cross_models: Record<string, string>;
38
+ /** Reasoning effort per reviewer name (codex, api). */
39
+ reasoning: Record<string, 'low' | 'medium' | 'high'>;
40
+ /** The API judge, when the team configured one (review.reviewer.api). */
41
+ api?: {
42
+ url: string;
43
+ model: string;
44
+ key_env: string;
45
+ vendor?: 'anthropic' | 'openai' | 'google' | 'other';
46
+ max_turns: number;
47
+ };
38
48
  /** Environment variables each judge's CLI must not see (the team's, never a person's). */
39
49
  judge_env: Record<string, {
40
50
  unset: string[];
@@ -13,7 +13,7 @@ const RANK = { single: 0, cross: 1, full: 2 };
13
13
  /** How near a layer is to this run: the nearer wins. */
14
14
  const NEAR = { flag: 0, env: 1, user: 2, team: 3 };
15
15
  export function resolveReviewer(config, choice = {}, user = loadSettings().reviewer, env = process.env) {
16
- const team = config.review?.reviewer ?? { enabled: false, on_push: 'background', reviewers: ['claude'], mode: 'single', models: {}, timeout_ms: 15 * 60_000, panel: 'off', mode_required: false, panel_max_items: 20, dismissals: false, judges: 2, escalate: 'always', cross_models: {}, judge_env: {} };
16
+ const team = config.review?.reviewer ?? { enabled: false, on_push: 'background', reviewers: ['claude'], mode: 'single', models: {}, timeout_ms: 15 * 60_000, panel: 'off', mode_required: false, panel_max_items: 20, dismissals: false, judges: 2, escalate: 'always', cross_models: {}, judge_env: {}, reasoning: {} };
17
17
  const refused = [];
18
18
  const envMode = parseMode(env.RIGOUR_REVIEWER_MODE);
19
19
  const envPanel = parseSwitch(env.RIGOUR_REVIEWER_PANEL);
@@ -79,6 +79,8 @@ export function resolveReviewer(config, choice = {}, user = loadSettings().revie
79
79
  escalate: requiredEscalation(team, user, refused),
80
80
  cross_models: team.cross_models,
81
81
  judge_env: team.judge_env ?? {},
82
+ reasoning: team.reasoning ?? {},
83
+ ...(team.api ? { api: team.api } : {}),
82
84
  source: { mode: modeSource, panel: panelSource },
83
85
  required: { mode: team.mode_required, panel: requirePanel },
84
86
  refused,
@@ -9,6 +9,8 @@ export { defaultExec, githubEnv, githubToken, parseJsonArrays, type Exec, type P
9
9
  export { itemLine, type OpenItem } from './reviewer/verdict.js';
10
10
  export type ReviewerOutcome = 'passed' | 'findings' | 'unavailable' | 'skipped';
11
11
  export interface ReviewerOptions {
12
+ /** For tests: what the API judge calls instead of fetch. */
13
+ fetch?: typeof fetch;
12
14
  /** The pull request to read when the checkout is detached (a backtest). */
13
15
  pr?: number;
14
16
  /** ISO time: a review or comment posted from then on is not shown to the reviewer (a backtest). */
@@ -21,7 +21,8 @@
21
21
  import fs from 'fs';
22
22
  import os from 'os';
23
23
  import path from 'path';
24
- import { ADAPTERS, isReviewerName, resolveAdapter, selectReviewers, vendorsOf } from './reviewer/adapters.js';
24
+ import { runApiJudge } from './reviewer/api-judge.js';
25
+ import { ADAPTERS, apiVendor, isReviewerName, resolveAdapter, selectReviewers, vendorsOf } from './reviewer/adapters.js';
25
26
  import { defaultExec, GH_TIMEOUT_MS, githubEnv } from './reviewer/exec.js';
26
27
  import { bodyAsOf, findPullRequest, ghFor, humanReviews, linesChanged, mergesBaseIn, rulesText, sha } from './reviewer/inputs.js';
27
28
  import { mergeImpact } from './reviewer/merge-impact.js';
@@ -76,13 +77,20 @@ async function review(cwd, base, config, exec, progress, options) {
76
77
  const candidates = settings.reviewers.filter(isReviewerName);
77
78
  const installed = new Map();
78
79
  for (const name of candidates) {
80
+ if (name === 'api') {
81
+ // The API judge is installed when the team configured it and the key it names is set.
82
+ if (settings.api && process.env[settings.api.key_env])
83
+ installed.set('api', { binary: 'api', version: settings.api.model });
84
+ continue;
85
+ }
79
86
  const found = await resolveAdapter(ADAPTERS[name], cwd, exec);
80
87
  if (found)
81
88
  installed.set(name, found);
82
89
  }
90
+ const vendorOf = (name) => (name === 'api' ? apiVendor(settings.api) : ADAPTERS[name].vendor);
83
91
  const authors = vendorsOf(await git(['log', '--format=%(trailers:key=Co-Authored-By,valueonly)%(trailers:key=Co-authored-by,valueonly)', `${baseSha}..HEAD`]));
84
92
  const mode = settings.mode;
85
- let reviewers = selectReviewers(candidates, mode, authors, new Set(installed.keys()), settings.judges);
93
+ let reviewers = selectReviewers(candidates, mode, authors, new Set(installed.keys()), settings.judges, vendorOf);
86
94
  if (reviewers.length === 0)
87
95
  return none('unavailable', `no reviewer installed: ${candidates.map(n => ADAPTERS[n].binary).join(', ') || 'review.reviewer.reviewers is empty'}`);
88
96
  modeRecord = modeRan(settings, reviewers, candidates, installed);
@@ -199,6 +207,10 @@ async function review(cwd, base, config, exec, progress, options) {
199
207
  if (over)
200
208
  return none(settings.required.panel || settings.required.mode ? 'unavailable' : 'skipped', over, { reviewers, scope, why, pr: pr?.number });
201
209
  const work = fs.mkdtempSync(path.join(os.tmpdir(), 'rigour-reviewer-'));
210
+ // One judge run, by CLI or by API: the same prompt, the same cost accounting, the same trace.
211
+ const runJudge = (name, prompt, model) => name === 'api'
212
+ ? runApiJudge(prompt, { url: settings.api.url, model: settings.api.model, key: process.env[settings.api.key_env] ?? '', maxTurns: settings.api.max_turns, timeoutMs: settings.timeout_ms, cwd, roots: [cwd, work], ...(settings.reasoning[name] ? { reasoning: settings.reasoning[name] } : {}), ...(options.fetch ? { fetchImpl: options.fetch } : {}) })
213
+ : exec(installed.get(name).binary, ADAPTERS[name].args(prompt, model, { reasoning: settings.reasoning[name] }), { cwd, timeoutMs: settings.timeout_ms, unset: judgeUnset(name, settings.judge_env) });
202
214
  try {
203
215
  const file = (name, text) => {
204
216
  const target = path.join(work, name);
@@ -239,7 +251,7 @@ async function review(cwd, base, config, exec, progress, options) {
239
251
  const answers = await Promise.all(reviewers.map(async (name) => {
240
252
  const adapter = ADAPTERS[name];
241
253
  const ask = async () => {
242
- const run = await exec(installed.get(name).binary, adapter.args(prompt, modelFor(name)), { cwd, timeoutMs: settings.timeout_ms, unset: judgeUnset(name, settings.judge_env) });
254
+ const run = await runJudge(name, prompt, modelFor(name));
243
255
  progress(`Rigour reviewer: ${name} finished in ${Math.round((Date.now() - started) / 1000)}s (exit ${run.exitCode})`);
244
256
  const answer = adapter.answer(run.stdout);
245
257
  store.addSpend(1, answer.costUsd); // every run counts against the caps, an answer or not
@@ -293,7 +305,7 @@ async function review(cwd, base, config, exec, progress, options) {
293
305
  },
294
306
  ask: async (judge, asked) => {
295
307
  const name = judge;
296
- const run = await exec(installed.get(name).binary, ADAPTERS[name].args(crossExamPrompt(repoRoot, head.slice(0, 9), diffFile, asked), settings.cross_models[name] ?? modelFor(name)), { cwd, timeoutMs: settings.timeout_ms, unset: judgeUnset(name, settings.judge_env) });
308
+ const run = await runJudge(name, crossExamPrompt(repoRoot, head.slice(0, 9), diffFile, asked), settings.cross_models[name] ?? modelFor(name));
297
309
  const answer = ADAPTERS[name].answer(run.stdout);
298
310
  store.addSpend(1, answer.costUsd);
299
311
  reserved--;
@@ -256,6 +256,33 @@ describe('the reviewer', () => {
256
256
  expect(again.cached).toBe(true);
257
257
  expect(again.record?.integrity).toBe(first.record?.integrity);
258
258
  });
259
+ it('reviews through the API judge when the team configured one and its key is set, with the same prompt and accounting', async () => {
260
+ const seen = seenNow();
261
+ const calls = [];
262
+ const fetchImpl = (async (_url, init) => {
263
+ const body = JSON.parse(init.body);
264
+ calls.push(body);
265
+ const last = body.messages.at(-1);
266
+ const message = last.role === 'user'
267
+ ? { role: 'assistant', content: null, tool_calls: [{ id: 't1', type: 'function', function: { name: 'read_file', arguments: JSON.stringify({ path: /(\S+full\.diff)/.exec(last.content)[1] }) } }] }
268
+ : { role: 'assistant', content: JSON.stringify({ ...EMPTY, findings: [{ class: 'correctness', file: 'src/job.ts', line: 2, issue: 'returns before the lock', input: 'two runs', consequence: 'two emails', quote: 'return 1;', severity: 'blocking' }] }) };
269
+ return new Response(JSON.stringify({ choices: [{ message }], usage: { prompt_tokens: 100, completion_tokens: 20, cost: 0.05 } }), { status: 200 });
270
+ });
271
+ const apiConfig = ConfigSchema.parse({ version: 1, review: { reviewer: { enabled: true, reviewers: ['api'], api: { url: 'https://example.test/v1', model: 'qwen3-coder', key_env: 'TEST_JUDGE_KEY' }, reasoning: { api: 'low' } } } });
272
+ const without = await runReviewer(repo, 'main', apiConfig, fakes(() => '', seen), () => undefined, { fetch: fetchImpl });
273
+ expect(without).toMatchObject({ outcome: 'unavailable', reason: expect.stringContaining('no reviewer installed') }); // the key is not set
274
+ process.env.TEST_JUDGE_KEY = 'secret';
275
+ try {
276
+ const result = await runReviewer(repo, 'main', apiConfig, fakes(() => '', seen), () => undefined, { fetch: fetchImpl, force: true });
277
+ expect(result).toMatchObject({ outcome: 'findings', reviewers: ['api'], costUsd: 0.1, items: [expect.objectContaining({ issue: 'returns before the lock', reviewer: 'api' })] });
278
+ expect(calls[0].reasoning_effort).toBe('low');
279
+ expect(calls[0].messages[1].content).toContain('full.diff'); // the same prompt a CLI judge gets
280
+ expect(result.record?.judges).toEqual([{ reviewer: 'api', version: 'qwen3-coder', cost_usd: 0.1, turns: 2 }]);
281
+ }
282
+ finally {
283
+ delete process.env.TEST_JUDGE_KEY;
284
+ }
285
+ });
259
286
  it('asks a judge once more after an answer that is not a verdict, and is unavailable only when the second is not one either', async () => {
260
287
  const seen = seenNow();
261
288
  let calls = 0;
@@ -293,6 +293,7 @@ export const UNIVERSAL_CONFIG = {
293
293
  escalate: 'always',
294
294
  cross_models: {},
295
295
  judge_env: {},
296
+ reasoning: {},
296
297
  },
297
298
  },
298
299
  ignore: [],
@@ -2457,6 +2457,32 @@ export declare const ConfigSchema: z.ZodObject<{
2457
2457
  escalate: z.ZodDefault<z.ZodOptional<z.ZodEnum<["always", "risk"]>>>;
2458
2458
  /** A model per reviewer name for cross-examination (a narrow verification task), e.g. { claude: "haiku" }. */
2459
2459
  cross_models: z.ZodDefault<z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodString>>>;
2460
+ /** Reasoning effort per reviewer name where the CLI or API takes one (codex, api): low, medium or high. */
2461
+ reasoning: z.ZodDefault<z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodEnum<["low", "medium", "high"]>>>>;
2462
+ /**
2463
+ * A judge reached through a model API (OpenAI-compatible chat completions with tools), named `api` in
2464
+ * `reviewers`: any model the team can call. The key is read from the environment variable `key_env`, never
2465
+ * from this file. `vendor` is the model's maker, for cross and full modes (inferred from the model name when unset).
2466
+ */
2467
+ api: z.ZodOptional<z.ZodObject<{
2468
+ url: z.ZodString;
2469
+ model: z.ZodString;
2470
+ key_env: z.ZodDefault<z.ZodOptional<z.ZodString>>;
2471
+ vendor: z.ZodOptional<z.ZodEnum<["anthropic", "openai", "google", "other"]>>;
2472
+ max_turns: z.ZodDefault<z.ZodOptional<z.ZodNumber>>;
2473
+ }, "strip", z.ZodTypeAny, {
2474
+ model: string;
2475
+ url: string;
2476
+ key_env: string;
2477
+ max_turns: number;
2478
+ vendor?: "anthropic" | "openai" | "google" | "other" | undefined;
2479
+ }, {
2480
+ model: string;
2481
+ url: string;
2482
+ key_env?: string | undefined;
2483
+ vendor?: "anthropic" | "openai" | "google" | "other" | undefined;
2484
+ max_turns?: number | undefined;
2485
+ }>>;
2460
2486
  /**
2461
2487
  * Environment variables a judge's CLI must not see, per reviewer name, e.g. { codex: { unset: [OPENAI_API_KEY] } }
2462
2488
  * when that variable holds a key meant for another service. RIGOUR_API_KEY is never passed to a judge.
@@ -2482,12 +2508,20 @@ export declare const ConfigSchema: z.ZodObject<{
2482
2508
  judges: 2 | 3;
2483
2509
  escalate: "always" | "risk";
2484
2510
  cross_models: Record<string, string>;
2511
+ reasoning: Record<string, "high" | "medium" | "low">;
2485
2512
  judge_env: Record<string, {
2486
2513
  unset: string[];
2487
2514
  }>;
2488
2515
  model?: string | undefined;
2489
2516
  max_runs_per_day?: number | undefined;
2490
2517
  max_usd_per_day?: number | undefined;
2518
+ api?: {
2519
+ model: string;
2520
+ url: string;
2521
+ key_env: string;
2522
+ max_turns: number;
2523
+ vendor?: "anthropic" | "openai" | "google" | "other" | undefined;
2524
+ } | undefined;
2491
2525
  }, {
2492
2526
  enabled?: boolean | undefined;
2493
2527
  mode?: "single" | "cross" | "full" | undefined;
@@ -2505,6 +2539,14 @@ export declare const ConfigSchema: z.ZodObject<{
2505
2539
  judges?: 2 | 3 | undefined;
2506
2540
  escalate?: "always" | "risk" | undefined;
2507
2541
  cross_models?: Record<string, string> | undefined;
2542
+ reasoning?: Record<string, "high" | "medium" | "low"> | undefined;
2543
+ api?: {
2544
+ model: string;
2545
+ url: string;
2546
+ key_env?: string | undefined;
2547
+ vendor?: "anthropic" | "openai" | "google" | "other" | undefined;
2548
+ max_turns?: number | undefined;
2549
+ } | undefined;
2508
2550
  judge_env?: Record<string, {
2509
2551
  unset?: string[] | undefined;
2510
2552
  }> | undefined;
@@ -2526,12 +2568,20 @@ export declare const ConfigSchema: z.ZodObject<{
2526
2568
  judges: 2 | 3;
2527
2569
  escalate: "always" | "risk";
2528
2570
  cross_models: Record<string, string>;
2571
+ reasoning: Record<string, "high" | "medium" | "low">;
2529
2572
  judge_env: Record<string, {
2530
2573
  unset: string[];
2531
2574
  }>;
2532
2575
  model?: string | undefined;
2533
2576
  max_runs_per_day?: number | undefined;
2534
2577
  max_usd_per_day?: number | undefined;
2578
+ api?: {
2579
+ model: string;
2580
+ url: string;
2581
+ key_env: string;
2582
+ max_turns: number;
2583
+ vendor?: "anthropic" | "openai" | "google" | "other" | undefined;
2584
+ } | undefined;
2535
2585
  };
2536
2586
  github_account?: string | undefined;
2537
2587
  }, {
@@ -2555,6 +2605,14 @@ export declare const ConfigSchema: z.ZodObject<{
2555
2605
  judges?: 2 | 3 | undefined;
2556
2606
  escalate?: "always" | "risk" | undefined;
2557
2607
  cross_models?: Record<string, string> | undefined;
2608
+ reasoning?: Record<string, "high" | "medium" | "low"> | undefined;
2609
+ api?: {
2610
+ model: string;
2611
+ url: string;
2612
+ key_env?: string | undefined;
2613
+ vendor?: "anthropic" | "openai" | "google" | "other" | undefined;
2614
+ max_turns?: number | undefined;
2615
+ } | undefined;
2558
2616
  judge_env?: Record<string, {
2559
2617
  unset?: string[] | undefined;
2560
2618
  }> | undefined;
@@ -2832,12 +2890,20 @@ export declare const ConfigSchema: z.ZodObject<{
2832
2890
  judges: 2 | 3;
2833
2891
  escalate: "always" | "risk";
2834
2892
  cross_models: Record<string, string>;
2893
+ reasoning: Record<string, "high" | "medium" | "low">;
2835
2894
  judge_env: Record<string, {
2836
2895
  unset: string[];
2837
2896
  }>;
2838
2897
  model?: string | undefined;
2839
2898
  max_runs_per_day?: number | undefined;
2840
2899
  max_usd_per_day?: number | undefined;
2900
+ api?: {
2901
+ model: string;
2902
+ url: string;
2903
+ key_env: string;
2904
+ max_turns: number;
2905
+ vendor?: "anthropic" | "openai" | "google" | "other" | undefined;
2906
+ } | undefined;
2841
2907
  };
2842
2908
  github_account?: string | undefined;
2843
2909
  };
@@ -3120,6 +3186,14 @@ export declare const ConfigSchema: z.ZodObject<{
3120
3186
  judges?: 2 | 3 | undefined;
3121
3187
  escalate?: "always" | "risk" | undefined;
3122
3188
  cross_models?: Record<string, string> | undefined;
3189
+ reasoning?: Record<string, "high" | "medium" | "low"> | undefined;
3190
+ api?: {
3191
+ model: string;
3192
+ url: string;
3193
+ key_env?: string | undefined;
3194
+ vendor?: "anthropic" | "openai" | "google" | "other" | undefined;
3195
+ max_turns?: number | undefined;
3196
+ } | undefined;
3123
3197
  judge_env?: Record<string, {
3124
3198
  unset?: string[] | undefined;
3125
3199
  }> | undefined;
@@ -423,6 +423,20 @@ export const ConfigSchema = z.object({
423
423
  escalate: z.enum(['always', 'risk']).optional().default('always'),
424
424
  /** A model per reviewer name for cross-examination (a narrow verification task), e.g. { claude: "haiku" }. */
425
425
  cross_models: z.record(ModelName).optional().default({}),
426
+ /** Reasoning effort per reviewer name where the CLI or API takes one (codex, api): low, medium or high. */
427
+ reasoning: z.record(z.enum(['low', 'medium', 'high'])).optional().default({}),
428
+ /**
429
+ * A judge reached through a model API (OpenAI-compatible chat completions with tools), named `api` in
430
+ * `reviewers`: any model the team can call. The key is read from the environment variable `key_env`, never
431
+ * from this file. `vendor` is the model's maker, for cross and full modes (inferred from the model name when unset).
432
+ */
433
+ api: z.object({
434
+ url: z.string().url(),
435
+ model: z.string().min(1),
436
+ key_env: z.string().regex(/^[A-Za-z_][A-Za-z0-9_]*$/, 'an environment variable name').optional().default('RIGOUR_JUDGE_API_KEY'),
437
+ vendor: z.enum(['anthropic', 'openai', 'google', 'other']).optional(),
438
+ max_turns: z.number().int().positive().optional().default(60),
439
+ }).optional(),
426
440
  /**
427
441
  * Environment variables a judge's CLI must not see, per reviewer name, e.g. { codex: { unset: [OPENAI_API_KEY] } }
428
442
  * when that variable holds a key meant for another service. RIGOUR_API_KEY is never passed to a judge.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@rigour-labs/core",
3
- "version": "6.7.6",
3
+ "version": "6.7.7",
4
4
  "description": "Rigour's review engine: deterministic gates on changed lines, rules and lessons learned from your team's fixes, and per-check precision from what you fix versus dismiss, across TypeScript, JavaScript, Python, Go, Ruby and C#.",
5
5
  "engines": {
6
6
  "node": ">=22.13"
@@ -72,11 +72,11 @@
72
72
  "@anthropic-ai/sdk": "^0.30.1",
73
73
  "pg": "^8.16.3",
74
74
  "openai": "^5.23.2",
75
- "@rigour-labs/brain-darwin-arm64": "6.7.6",
76
- "@rigour-labs/brain-darwin-x64": "6.7.6",
77
- "@rigour-labs/brain-linux-arm64": "6.7.6",
78
- "@rigour-labs/brain-linux-x64": "6.7.6",
79
- "@rigour-labs/brain-win-x64": "6.7.6"
75
+ "@rigour-labs/brain-darwin-arm64": "6.7.7",
76
+ "@rigour-labs/brain-darwin-x64": "6.7.7",
77
+ "@rigour-labs/brain-linux-arm64": "6.7.7",
78
+ "@rigour-labs/brain-win-x64": "6.7.7",
79
+ "@rigour-labs/brain-linux-x64": "6.7.7"
80
80
  },
81
81
  "devDependencies": {
82
82
  "@types/fs-extra": "^11.0.4",