@rigour-labs/core 6.7.6 → 6.7.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/dist/index.d.ts +1 -0
  2. package/dist/index.js +1 -0
  3. package/dist/review/backtest-init.d.ts +15 -0
  4. package/dist/review/backtest-init.js +30 -16
  5. package/dist/review/backtest-last.d.ts +40 -0
  6. package/dist/review/backtest-last.js +116 -0
  7. package/dist/review/backtest-last.test.d.ts +1 -0
  8. package/dist/review/backtest-last.test.js +109 -0
  9. package/dist/review/backtest.d.ts +26 -0
  10. package/dist/review/backtest.js +4 -4
  11. package/dist/review/reviewer/adapters.d.ts +11 -4
  12. package/dist/review/reviewer/adapters.js +44 -5
  13. package/dist/review/reviewer/adapters.test.js +16 -1
  14. package/dist/review/reviewer/api-judge.d.ts +28 -0
  15. package/dist/review/reviewer/api-judge.js +161 -0
  16. package/dist/review/reviewer/api-judge.test.d.ts +1 -0
  17. package/dist/review/reviewer/api-judge.test.js +87 -0
  18. package/dist/review/reviewer/context.d.ts +2 -0
  19. package/dist/review/reviewer/context.js +4 -1
  20. package/dist/review/reviewer/settings.d.ts +10 -0
  21. package/dist/review/reviewer/settings.js +3 -1
  22. package/dist/review/reviewer/verdict.js +8 -3
  23. package/dist/review/reviewer.d.ts +2 -0
  24. package/dist/review/reviewer.js +17 -5
  25. package/dist/review/reviewer.test.js +35 -3
  26. package/dist/review-learning/lessons.d.ts +2 -0
  27. package/dist/review-learning/lessons.js +17 -1
  28. package/dist/review-learning/repo-rules.test.js +3 -0
  29. package/dist/review-learning/review-learning.test.js +9 -0
  30. package/dist/review-learning/team-lessons.d.ts +2 -2
  31. package/dist/review-learning/team-lessons.js +3 -3
  32. package/dist/templates/universal-config.js +1 -0
  33. package/dist/types/index.d.ts +74 -0
  34. package/dist/types/index.js +14 -0
  35. package/package.json +6 -6
@@ -0,0 +1,28 @@
1
+ import type { RunTrace } from './adapters.js';
2
+ export type Reasoning = 'low' | 'medium' | 'high';
3
+ export interface ApiJudgeOptions {
4
+ /** The API's base URL; `/chat/completions` is appended. */
5
+ url: string;
6
+ model: string;
7
+ key: string;
8
+ maxTurns: number;
9
+ timeoutMs: number;
10
+ cwd: string;
11
+ /** Where the tools may read: the checkout and the review's input folder. */
12
+ roots: string[];
13
+ reasoning?: Reasoning;
14
+ fetchImpl?: typeof fetch;
15
+ }
16
+ export interface ApiJudgeRun {
17
+ exitCode: number;
18
+ stdout: string;
19
+ stderr: string;
20
+ }
21
+ /** What a run's stdout carries on success, for the adapter to read. */
22
+ export interface ApiJudgeAnswer {
23
+ result: string;
24
+ usage: RunTrace['usage'];
25
+ cost_usd?: number;
26
+ trace: RunTrace;
27
+ }
28
+ export declare function runApiJudge(prompt: string, o: ApiJudgeOptions): Promise<ApiJudgeRun>;
@@ -0,0 +1,161 @@
1
+ /**
2
+ * A judge reached through a model API instead of an agent CLI, so any model a team can call works as
3
+ * a reviewer: OpenAI-compatible chat completions with tools (OpenAI, OpenRouter, a local server, other
4
+ * vendors through a gateway). Rigour runs the loop itself: the model asks for a read-only tool, Rigour
5
+ * runs it inside the checkout (and the review's own input folder) and hands the result back, until
6
+ * the model answers. The same prompt, the same evidence contract and the same trace as a CLI judge;
7
+ * the key comes from an environment variable named in rigour.yml, never from the file.
8
+ */
9
+ import fs from 'fs';
10
+ import path from 'path';
11
+ import { execa } from 'execa';
12
+ const SYSTEM = 'You review code with read-only tools. Read the files the task names with read_file (the review inputs are named by absolute path), search the repository with search, read history with git. When you are done, reply with the final answer the task asks for and nothing else.';
13
+ const MAX_RESULT_CHARS = 60_000;
14
+ const GIT_ALLOWED = new Set(['log', 'show', 'diff', 'blame', 'grep', 'ls-files', 'rev-parse', 'merge-base']);
15
+ /** Git options that write, or point git at another repository. */
16
+ const GIT_REFUSED = /^(--output|-o$|--git-dir|--work-tree|-C$|--exec-path|-c$|--config-env)/;
17
+ const TOOLS = [
18
+ { type: 'function', function: { name: 'read_file', description: 'Read a file in the repository or the review inputs, or a line range of it.', parameters: { type: 'object', properties: { path: { type: 'string' }, start_line: { type: 'integer' }, end_line: { type: 'integer' } }, required: ['path'] } } },
19
+ { type: 'function', function: { name: 'search', description: 'Search the tracked files for a regular expression (git grep -n), optionally under one path.', parameters: { type: 'object', properties: { pattern: { type: 'string' }, path: { type: 'string' } }, required: ['pattern'] } } },
20
+ { type: 'function', function: { name: 'list_dir', description: 'List a directory in the repository or the review inputs.', parameters: { type: 'object', properties: { path: { type: 'string' } }, required: ['path'] } } },
21
+ { type: 'function', function: { name: 'git', description: 'Run a read-only git command in the repository: log, show, diff, blame, grep, ls-files, rev-parse, merge-base.', parameters: { type: 'object', properties: { args: { type: 'array', items: { type: 'string' } } }, required: ['args'] } } },
22
+ ];
23
+ const n = (v) => (typeof v === 'number' && Number.isFinite(v) ? v : 0);
24
+ export async function runApiJudge(prompt, o) {
25
+ const fetchImpl = o.fetchImpl ?? fetch;
26
+ const deadline = Date.now() + o.timeoutMs;
27
+ const messages = [{ role: 'system', content: SYSTEM }, { role: 'user', content: prompt }];
28
+ const usage = { input: 0, cacheRead: 0, cacheWrite: 0, output: 0 };
29
+ const calls = [];
30
+ let cost;
31
+ const fail = (why) => ({ exitCode: 1, stdout: '', stderr: `api judge (${o.model}): ${why}` });
32
+ for (let turn = 1; turn <= o.maxTurns; turn++) {
33
+ const left = deadline - Date.now();
34
+ if (left <= 0)
35
+ return fail(`timed out after ${o.timeoutMs} ms`);
36
+ const controller = new AbortController();
37
+ const timer = setTimeout(() => controller.abort(), left);
38
+ let response;
39
+ try {
40
+ response = await fetchImpl(`${o.url.replace(/\/$/, '')}/chat/completions`, {
41
+ method: 'POST',
42
+ headers: { 'content-type': 'application/json', authorization: `Bearer ${o.key}` },
43
+ body: JSON.stringify({ model: o.model, messages, tools: TOOLS, tool_choice: 'auto', ...(o.reasoning ? { reasoning_effort: o.reasoning } : {}) }),
44
+ signal: controller.signal,
45
+ });
46
+ }
47
+ catch (error) {
48
+ return fail(`request failed: ${error instanceof Error ? error.message : String(error)}`);
49
+ }
50
+ finally {
51
+ clearTimeout(timer);
52
+ }
53
+ if (!response.ok)
54
+ return fail(`HTTP ${response.status}: ${(await response.text()).slice(0, 300)}`);
55
+ let body;
56
+ try {
57
+ body = await response.json();
58
+ }
59
+ catch {
60
+ return fail('the answer was not JSON');
61
+ }
62
+ const u = body.usage ?? {};
63
+ const cached = n(u.prompt_tokens_details?.cached_tokens);
64
+ usage.input += Math.max(0, n(u.prompt_tokens) - cached);
65
+ usage.cacheRead += cached;
66
+ usage.output += n(u.completion_tokens);
67
+ if (typeof u.cost === 'number')
68
+ cost = (cost ?? 0) + u.cost;
69
+ const message = body.choices?.[0]?.message;
70
+ if (!message)
71
+ return fail('no choices in the answer');
72
+ messages.push(message);
73
+ const toolCalls = Array.isArray(message.tool_calls) ? message.tool_calls : [];
74
+ if (toolCalls.length === 0) {
75
+ const answer = { result: String(message.content ?? ''), usage, ...(cost !== undefined ? { cost_usd: cost } : {}), trace: { turns: turn, usage, calls } };
76
+ return { exitCode: 0, stdout: JSON.stringify(answer), stderr: '' };
77
+ }
78
+ for (const call of toolCalls) {
79
+ const ran = await runTool(call, o);
80
+ calls.push({ turn, tool: ran.tool, target: ran.target, resultChars: ran.result.length });
81
+ messages.push({ role: 'tool', tool_call_id: call.id, content: ran.result });
82
+ }
83
+ }
84
+ return fail(`no answer within ${o.maxTurns} turns`);
85
+ }
86
+ /** One tool call, inside the allowed roots only; the trace names tools as the CLI judges do (Read, Grep, Glob, Bash). */
87
+ async function runTool(call, o) {
88
+ const name = String(call?.function?.name ?? '');
89
+ let args = {};
90
+ try {
91
+ args = JSON.parse(call?.function?.arguments || '{}');
92
+ }
93
+ catch {
94
+ return { tool: name, target: '', result: 'refused: the arguments were not JSON' };
95
+ }
96
+ const clip = (text) => (text.length > MAX_RESULT_CHARS ? `${text.slice(0, MAX_RESULT_CHARS)}\n…[truncated at ${MAX_RESULT_CHARS} characters]` : text);
97
+ if (name === 'read_file') {
98
+ const target = inside(String(args.path ?? ''), o);
99
+ if (!target)
100
+ return { tool: 'Read', target: String(args.path ?? ''), result: 'refused: outside the repository and the review inputs' };
101
+ try {
102
+ const lines = fs.readFileSync(target, 'utf8').split('\n');
103
+ const start = Math.max(1, Number(args.start_line) || 1);
104
+ const end = Math.min(lines.length, Number(args.end_line) || lines.length);
105
+ return { tool: 'Read', target, result: clip(lines.slice(start - 1, end).map((l, i) => `${start + i}: ${l}`).join('\n')) };
106
+ }
107
+ catch (error) {
108
+ return { tool: 'Read', target, result: `error: ${error instanceof Error ? error.message : String(error)}` };
109
+ }
110
+ }
111
+ if (name === 'list_dir') {
112
+ const target = inside(String(args.path ?? '.'), o);
113
+ if (!target)
114
+ return { tool: 'Glob', target: String(args.path ?? ''), result: 'refused: outside the repository and the review inputs' };
115
+ try {
116
+ return { tool: 'Glob', target, result: clip(fs.readdirSync(target, { withFileTypes: true }).map(e => (e.isDirectory() ? `${e.name}/` : e.name)).join('\n')) };
117
+ }
118
+ catch (error) {
119
+ return { tool: 'Glob', target, result: `error: ${error instanceof Error ? error.message : String(error)}` };
120
+ }
121
+ }
122
+ if (name === 'search') {
123
+ const pattern = String(args.pattern ?? '');
124
+ const under = args.path ? inside(String(args.path), o) : undefined;
125
+ if (args.path && !under)
126
+ return { tool: 'Grep', target: pattern, result: 'refused: outside the repository' };
127
+ const run = await execa('git', ['grep', '-n', '-I', '-e', pattern, ...(under ? ['--', path.relative(o.cwd, under) || '.'] : [])], { cwd: o.cwd, reject: false, maxBuffer: 16 * 1024 * 1024 });
128
+ return { tool: 'Grep', target: pattern, result: clip(run.stdout || (run.exitCode === 1 ? '(no match)' : run.stderr)) };
129
+ }
130
+ if (name === 'git') {
131
+ const gitArgs = Array.isArray(args.args) ? args.args.map(String) : [];
132
+ if (!GIT_ALLOWED.has(gitArgs[0] ?? '') || gitArgs.some(a => GIT_REFUSED.test(a)))
133
+ return { tool: 'Bash', target: `git ${gitArgs.join(' ')}`, result: 'refused: only read-only git commands, without options that write or point elsewhere' };
134
+ const run = await execa('git', ['--no-pager', ...gitArgs], { cwd: o.cwd, reject: false, maxBuffer: 16 * 1024 * 1024 });
135
+ return { tool: 'Bash', target: `git ${gitArgs.join(' ')}`, result: clip(run.stdout || run.stderr) };
136
+ }
137
+ return { tool: name, target: '', result: `refused: no tool named ${name}` };
138
+ }
139
+ /** The real path of `p` when it lies under one of the roots; undefined otherwise (a symlink out is outside). */
140
+ function inside(p, o) {
141
+ const resolved = path.resolve(o.cwd, p);
142
+ let real;
143
+ try {
144
+ real = fs.realpathSync(resolved);
145
+ }
146
+ catch {
147
+ return undefined;
148
+ }
149
+ for (const root of o.roots) {
150
+ let base;
151
+ try {
152
+ base = fs.realpathSync(root);
153
+ }
154
+ catch {
155
+ continue;
156
+ }
157
+ if (real === base || real.startsWith(base + path.sep))
158
+ return real;
159
+ }
160
+ return undefined;
161
+ }
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1,87 @@
1
+ import { execFileSync } from 'child_process';
2
+ import fs from 'fs';
3
+ import os from 'os';
4
+ import path from 'path';
5
+ import { afterEach, beforeEach, describe, expect, it } from 'vitest';
6
+ import { runApiJudge } from './api-judge.js';
7
+ let repo;
8
+ let work;
9
+ beforeEach(() => {
10
+ repo = fs.mkdtempSync(path.join(os.tmpdir(), 'api-judge-'));
11
+ work = fs.mkdtempSync(path.join(os.tmpdir(), 'api-judge-inputs-'));
12
+ execFileSync('git', ['-C', repo, 'init', '-q']);
13
+ fs.mkdirSync(path.join(repo, 'src'));
14
+ fs.writeFileSync(path.join(repo, 'src/job.ts'), 'export function job() {\n return 1;\n}\n');
15
+ execFileSync('git', ['-C', repo, 'add', '-A']);
16
+ execFileSync('git', ['-C', repo, '-c', 'user.email=t@x', '-c', 'user.name=t', 'commit', '-qm', 'init']);
17
+ fs.writeFileSync(path.join(work, 'full.diff'), '+++ b/src/job.ts\n');
18
+ });
19
+ afterEach(() => { for (const d of [repo, work])
20
+ fs.rmSync(d, { recursive: true, force: true }); });
21
+ /** A model that plays a scripted conversation: each entry is what it answers to the next request. */
22
+ function model(turns) {
23
+ const seen = [];
24
+ let i = 0;
25
+ const fetchImpl = (async (_url, init) => {
26
+ const body = JSON.parse(init.body);
27
+ seen.push(body);
28
+ const turn = turns[i++] ?? { text: 'done' };
29
+ if (turn.status)
30
+ return new Response('nope', { status: turn.status });
31
+ const message = turn.tools
32
+ ? { role: 'assistant', content: null, tool_calls: turn.tools.map((t, k) => ({ id: `c${i}-${k}`, type: 'function', function: { name: t.name, arguments: JSON.stringify(t.args) } })) }
33
+ : { role: 'assistant', content: turn.text };
34
+ return new Response(JSON.stringify({ choices: [{ message }], usage: turn.usage ?? { prompt_tokens: 100, completion_tokens: 10, prompt_tokens_details: { cached_tokens: 60 } } }), { status: 200, headers: { 'content-type': 'application/json' } });
35
+ });
36
+ return { fetchImpl, seen };
37
+ }
38
+ const options = (fetchImpl, extra = {}) => ({ url: 'https://example.test/v1', model: 'some-model', key: 'k', maxTurns: 10, timeoutMs: 30_000, cwd: repo, roots: [repo, work], fetchImpl, ...extra });
39
+ describe('the API judge', () => {
40
+ it('runs the tools the model asks for inside the checkout and the inputs, then returns its answer with the usage and the trace', async () => {
41
+ const { fetchImpl, seen } = model([
42
+ { tools: [{ name: 'read_file', args: { path: path.join(work, 'full.diff') } }, { name: 'read_file', args: { path: 'src/job.ts', start_line: 2, end_line: 2 } }] },
43
+ { tools: [{ name: 'search', args: { pattern: 'return' } }, { name: 'git', args: { args: ['log', '-1', '--format=%s'] } }, { name: 'list_dir', args: { path: 'src' } }] },
44
+ { text: '{"prior_points":[]}', usage: { prompt_tokens: 50, completion_tokens: 5, cost: 0.01 } },
45
+ ]);
46
+ const run = await runApiJudge('review this', options(fetchImpl));
47
+ expect(run.exitCode).toBe(0);
48
+ const answer = JSON.parse(run.stdout);
49
+ expect(answer.result).toBe('{"prior_points":[]}');
50
+ expect(answer.usage).toEqual({ input: 130, cacheRead: 120, cacheWrite: 0, output: 25 });
51
+ expect(answer.cost_usd).toBe(0.01);
52
+ expect(answer.trace.turns).toBe(3);
53
+ expect(answer.trace.calls.map(c => c.tool)).toEqual(['Read', 'Read', 'Grep', 'Bash', 'Glob']);
54
+ const toolResults = (request) => request.messages.filter((m) => m.role === 'tool').map((m) => m.content);
55
+ expect(toolResults(seen[1])).toEqual(['1: +++ b/src/job.ts\n2: ', '2: return 1;']); // turn 1's reads, as the model got them
56
+ expect(toolResults(seen[2]).slice(2)).toEqual(['src/job.ts:2: return 1;', 'init', 'job.ts']); // turn 2's search, git and listing
57
+ expect(seen[0].messages[0].role).toBe('system');
58
+ expect(seen[0].tools.map((t) => t.function.name)).toEqual(['read_file', 'search', 'list_dir', 'git']);
59
+ });
60
+ it('refuses to read outside the roots, a symlink out included, and refuses git that writes or points elsewhere', async () => {
61
+ fs.symlinkSync(os.homedir(), path.join(repo, 'out'));
62
+ const { fetchImpl, seen } = model([
63
+ { tools: [{ name: 'read_file', args: { path: '/etc/hosts' } }, { name: 'read_file', args: { path: 'out/.bashrc' } }, { name: 'git', args: { args: ['log', '--output=/tmp/x'] } }, { name: 'git', args: { args: ['push'] } }, { name: 'nope', args: {} }] },
64
+ { text: 'ok' },
65
+ ]);
66
+ await runApiJudge('p', options(fetchImpl));
67
+ const results = seen[1].messages.filter((m) => m.role === 'tool').map((m) => m.content);
68
+ expect(results).toEqual([
69
+ 'refused: outside the repository and the review inputs', 'refused: outside the repository and the review inputs',
70
+ 'refused: only read-only git commands, without options that write or point elsewhere', 'refused: only read-only git commands, without options that write or point elsewhere',
71
+ 'refused: no tool named nope',
72
+ ]);
73
+ });
74
+ it('fails closed on a turn cap, an HTTP error and a timeout, saying why', async () => {
75
+ const loop = model(Array.from({ length: 5 }, () => ({ tools: [{ name: 'list_dir', args: { path: '.' } }] })));
76
+ expect(await runApiJudge('p', options(loop.fetchImpl, { maxTurns: 3 }))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('no answer within 3 turns') });
77
+ expect(await runApiJudge('p', options(model([{ status: 429 }]).fetchImpl))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('HTTP 429') });
78
+ const slow = (async (_u, init) => new Promise((_r, reject) => init.signal.addEventListener('abort', () => reject(new Error('aborted')))));
79
+ expect(await runApiJudge('p', options(slow, { timeoutMs: 50 }))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('request failed') });
80
+ });
81
+ it('passes the reasoning effort when asked', async () => {
82
+ const { fetchImpl, seen } = model([{ text: 'ok' }]);
83
+ await runApiJudge('p', options(fetchImpl, { reasoning: 'low' }));
84
+ expect(seen[0].reasoning_effort).toBe('low');
85
+ expect(seen[0].model).toBe('some-model');
86
+ });
87
+ });
@@ -31,6 +31,8 @@ export interface ContextInput {
31
31
  router: RouterPolicy | undefined;
32
32
  /** Which of the team's review lessons the judges see (gates.deep.review_lessons): verified by default, all, or off. */
33
33
  lessons?: LessonMode;
34
+ /** The pull request under review: lessons learned only from it are its own reviews, which the judge already reads. */
35
+ pr?: number;
34
36
  /** The previous verdict's panel decisions, and the files changed since it. */
35
37
  previousPanel: PanelItem[] | undefined;
36
38
  touched: Set<string>;
@@ -24,6 +24,9 @@ export const REVIEW_DISMISSALS = path.join('.rigour', 'dismissed-review-items.js
24
24
  const MAX_DOCS = 10;
25
25
  /** Team standards a judge is shown with the lessons about the changed files. */
26
26
  const JUDGE_STANDARDS = 15;
27
+ /** File lessons a judge is shown: on a pull request touching a hundred files, enough for every file, at most this many per file. */
28
+ const JUDGE_FILE_LESSONS = 30;
29
+ const JUDGE_LESSONS_PER_FILE = 3;
27
30
  /** Rules from the repository's own rules files a judge is asked to answer, most relevant first. */
28
31
  const JUDGE_RULES = 15;
29
32
  const MAX_SETTLED = 40;
@@ -78,7 +81,7 @@ export function buildContext(input) {
78
81
  task = undefined;
79
82
  }
80
83
  // A judge reads the whole pull request: more of what the team taught fits than an agent's one question at the stop.
81
- const lessons = input.lessons === 'off' ? [] : lessonsForDiff(input.cwd, input.diff, input.lessons, JUDGE_STANDARDS).map(lessonView);
84
+ const lessons = input.lessons === 'off' ? [] : lessonsForDiff(input.cwd, input.diff, input.lessons, JUDGE_STANDARDS, JUDGE_FILE_LESSONS, JUDGE_LESSONS_PER_FILE, input.pr).map(lessonView);
82
85
  if (lessons.length)
83
86
  sections.push(`## Lessons this team taught on earlier reviews, for what this change touches (context: a lesson never blocks on its own; a finding still needs its quote)\n${lessons.map(l => `- ${describeLesson(l)}`).join('\n')}`);
84
87
  // The repository's own rules, always: the reviewer is the boundary, and what the team wrote is the standard it checks.
@@ -35,6 +35,16 @@ export interface ResolvedReviewer {
35
35
  judges: 2 | 3;
36
36
  escalate: 'always' | 'risk';
37
37
  cross_models: Record<string, string>;
38
+ /** Reasoning effort per reviewer name (codex, api). */
39
+ reasoning: Record<string, 'low' | 'medium' | 'high'>;
40
+ /** The API judge, when the team configured one (review.reviewer.api). */
41
+ api?: {
42
+ url: string;
43
+ model: string;
44
+ key_env: string;
45
+ vendor?: 'anthropic' | 'openai' | 'google' | 'other';
46
+ max_turns: number;
47
+ };
38
48
  /** Environment variables each judge's CLI must not see (the team's, never a person's). */
39
49
  judge_env: Record<string, {
40
50
  unset: string[];
@@ -13,7 +13,7 @@ const RANK = { single: 0, cross: 1, full: 2 };
13
13
  /** How near a layer is to this run: the nearer wins. */
14
14
  const NEAR = { flag: 0, env: 1, user: 2, team: 3 };
15
15
  export function resolveReviewer(config, choice = {}, user = loadSettings().reviewer, env = process.env) {
16
- const team = config.review?.reviewer ?? { enabled: false, on_push: 'background', reviewers: ['claude'], mode: 'single', models: {}, timeout_ms: 15 * 60_000, panel: 'off', mode_required: false, panel_max_items: 20, dismissals: false, judges: 2, escalate: 'always', cross_models: {}, judge_env: {} };
16
+ const team = config.review?.reviewer ?? { enabled: false, on_push: 'background', reviewers: ['claude'], mode: 'single', models: {}, timeout_ms: 15 * 60_000, panel: 'off', mode_required: false, panel_max_items: 20, dismissals: false, judges: 2, escalate: 'always', cross_models: {}, judge_env: {}, reasoning: {} };
17
17
  const refused = [];
18
18
  const envMode = parseMode(env.RIGOUR_REVIEWER_MODE);
19
19
  const envPanel = parseSwitch(env.RIGOUR_REVIEWER_PANEL);
@@ -79,6 +79,8 @@ export function resolveReviewer(config, choice = {}, user = loadSettings().revie
79
79
  escalate: requiredEscalation(team, user, refused),
80
80
  cross_models: team.cross_models,
81
81
  judge_env: team.judge_env ?? {},
82
+ reasoning: team.reasoning ?? {},
83
+ ...(team.api ? { api: team.api } : {}),
82
84
  source: { mode: modeSource, panel: panelSource },
83
85
  required: { mode: team.mode_required, panel: requirePanel },
84
86
  refused,
@@ -224,7 +224,8 @@ export function account(verdict, previousOpen, verify) {
224
224
  for (const r of verdict.rules ?? []) {
225
225
  if (r.status !== 'broken' || !r.rule)
226
226
  continue;
227
- const item = { id: id('repo-rule', r.file, r.id), kind: 'rule', class: 'repo-rule', file: r.file, line: r.line, issue: `breaks a rule this repository wrote for itself (${r.source}): ${r.rule}`, consequence: r.requirement ? 'the team wrote this rule as a requirement' : 'the team wrote this rule as guidance', ...(r.quote ? { quote: r.quote } : {}), evidence: r.evidence, reviewer: r.reviewer };
227
+ // The rule's own words are the issue, so the same point found as a finding reads alike; where it came from is the evidence.
228
+ const item = { id: id('repo-rule', r.file, r.id), kind: 'rule', class: 'repo-rule', file: r.file, line: r.line, issue: r.rule, consequence: r.requirement ? 'the team wrote this rule as a requirement' : 'the team wrote this rule as guidance', ...(r.quote ? { quote: r.quote } : {}), evidence: `breaks a rule this repository wrote for itself (${r.source})${r.evidence ? `: ${r.evidence}` : ''}`, reviewer: r.reviewer };
228
229
  if (r.requirement)
229
230
  add(item);
230
231
  else
@@ -274,8 +275,10 @@ export function account(verdict, previousOpen, verify) {
274
275
  }
275
276
  return { open: onePerRootCause(open), unverified, resolved, answerInReply, notes, advisory: onePerRootCause(advisory) };
276
277
  }
277
- /** How alike two items' words must be to be the same point made in two places. */
278
+ /** How alike two items' words must be to be the same point made in two places; and, on the same lines, to be one point said two ways. */
278
279
  const SAME_POINT = 0.6;
280
+ const SAME_PLACE = 0.3;
281
+ const SAME_LINES = 3;
279
282
  /**
280
283
  * The same point found in several places is one item carrying every location, so a person reads one
281
284
  * line, not one per file. Blocking is unchanged: the item blocks until every location is fixed.
@@ -283,7 +286,9 @@ const SAME_POINT = 0.6;
283
286
  function onePerRootCause(items) {
284
287
  const kept = [];
285
288
  for (const item of items) {
286
- const same = item.kind === 'prior' ? undefined : kept.find(k => k.kind !== 'prior' && k.class === item.class && textSimilarity(k, item) >= SAME_POINT);
289
+ // The same class in the same words anywhere, or any two non-human items on the same lines that read alike (a rule break and the finding it caused).
290
+ const nearby = (k) => !!k.file && k.file === item.file && k.line !== undefined && item.line !== undefined && Math.abs(k.line - item.line) <= SAME_LINES;
291
+ const same = item.kind === 'prior' ? undefined : kept.find(k => k.kind !== 'prior' && ((k.class === item.class && textSimilarity(k, item) >= SAME_POINT) || (nearby(k) && textSimilarity(k, item) >= SAME_PLACE)));
287
292
  if (!same) {
288
293
  kept.push(item);
289
294
  continue;
@@ -9,6 +9,8 @@ export { defaultExec, githubEnv, githubToken, parseJsonArrays, type Exec, type P
9
9
  export { itemLine, type OpenItem } from './reviewer/verdict.js';
10
10
  export type ReviewerOutcome = 'passed' | 'findings' | 'unavailable' | 'skipped';
11
11
  export interface ReviewerOptions {
12
+ /** For tests: what the API judge calls instead of fetch. */
13
+ fetch?: typeof fetch;
12
14
  /** The pull request to read when the checkout is detached (a backtest). */
13
15
  pr?: number;
14
16
  /** ISO time: a review or comment posted from then on is not shown to the reviewer (a backtest). */
@@ -21,7 +21,8 @@
21
21
  import fs from 'fs';
22
22
  import os from 'os';
23
23
  import path from 'path';
24
- import { ADAPTERS, isReviewerName, resolveAdapter, selectReviewers, vendorsOf } from './reviewer/adapters.js';
24
+ import { runApiJudge } from './reviewer/api-judge.js';
25
+ import { ADAPTERS, apiVendor, isReviewerName, resolveAdapter, selectReviewers, vendorsOf } from './reviewer/adapters.js';
25
26
  import { defaultExec, GH_TIMEOUT_MS, githubEnv } from './reviewer/exec.js';
26
27
  import { bodyAsOf, findPullRequest, ghFor, humanReviews, linesChanged, mergesBaseIn, rulesText, sha } from './reviewer/inputs.js';
27
28
  import { mergeImpact } from './reviewer/merge-impact.js';
@@ -76,13 +77,20 @@ async function review(cwd, base, config, exec, progress, options) {
76
77
  const candidates = settings.reviewers.filter(isReviewerName);
77
78
  const installed = new Map();
78
79
  for (const name of candidates) {
80
+ if (name === 'api') {
81
+ // The API judge is installed when the team configured it and the key it names is set.
82
+ if (settings.api && process.env[settings.api.key_env])
83
+ installed.set('api', { binary: 'api', version: settings.api.model });
84
+ continue;
85
+ }
79
86
  const found = await resolveAdapter(ADAPTERS[name], cwd, exec);
80
87
  if (found)
81
88
  installed.set(name, found);
82
89
  }
90
+ const vendorOf = (name) => (name === 'api' ? apiVendor(settings.api) : ADAPTERS[name].vendor);
83
91
  const authors = vendorsOf(await git(['log', '--format=%(trailers:key=Co-Authored-By,valueonly)%(trailers:key=Co-authored-by,valueonly)', `${baseSha}..HEAD`]));
84
92
  const mode = settings.mode;
85
- let reviewers = selectReviewers(candidates, mode, authors, new Set(installed.keys()), settings.judges);
93
+ let reviewers = selectReviewers(candidates, mode, authors, new Set(installed.keys()), settings.judges, vendorOf);
86
94
  if (reviewers.length === 0)
87
95
  return none('unavailable', `no reviewer installed: ${candidates.map(n => ADAPTERS[n].binary).join(', ') || 'review.reviewer.reviewers is empty'}`);
88
96
  modeRecord = modeRan(settings, reviewers, candidates, installed);
@@ -133,7 +141,7 @@ async function review(cwd, base, config, exec, progress, options) {
133
141
  const sincePrevious = previousIsAncestor ? new Set((await git(['diff', '--name-only', `${previous.head}..HEAD`])).split('\n').filter(Boolean)) : new Set();
134
142
  const changedFiles = [...fullDiff.matchAll(/^diff --git a\/.* b\/(.*)$/gm)].map(m => m[1]);
135
143
  const context = buildContext({
136
- cwd, stateRoot, dismissals, diff: fullDiff, router: config.gates.deep?.router, lessons: config.gates.deep?.review_lessons, touched: sincePrevious, checks: options.checks ?? [],
144
+ cwd, stateRoot, dismissals, diff: fullDiff, router: config.gates.deep?.router, lessons: config.gates.deep?.review_lessons, ...(pr ? { pr: pr.number } : {}), touched: sincePrevious, checks: options.checks ?? [],
137
145
  previousPanel: previousIsAncestor ? store.readJson(previous.verdict)?.panel?.items : undefined,
138
146
  docs: await relatedDocs(cwd, changedFiles, exec),
139
147
  });
@@ -199,6 +207,10 @@ async function review(cwd, base, config, exec, progress, options) {
199
207
  if (over)
200
208
  return none(settings.required.panel || settings.required.mode ? 'unavailable' : 'skipped', over, { reviewers, scope, why, pr: pr?.number });
201
209
  const work = fs.mkdtempSync(path.join(os.tmpdir(), 'rigour-reviewer-'));
210
+ // One judge run, by CLI or by API: the same prompt, the same cost accounting, the same trace.
211
+ const runJudge = (name, prompt, model) => name === 'api'
212
+ ? runApiJudge(prompt, { url: settings.api.url, model: settings.api.model, key: process.env[settings.api.key_env] ?? '', maxTurns: settings.api.max_turns, timeoutMs: settings.timeout_ms, cwd, roots: [cwd, work], ...(settings.reasoning[name] ? { reasoning: settings.reasoning[name] } : {}), ...(options.fetch ? { fetchImpl: options.fetch } : {}) })
213
+ : exec(installed.get(name).binary, ADAPTERS[name].args(prompt, model, { reasoning: settings.reasoning[name] }), { cwd, timeoutMs: settings.timeout_ms, unset: judgeUnset(name, settings.judge_env) });
202
214
  try {
203
215
  const file = (name, text) => {
204
216
  const target = path.join(work, name);
@@ -239,7 +251,7 @@ async function review(cwd, base, config, exec, progress, options) {
239
251
  const answers = await Promise.all(reviewers.map(async (name) => {
240
252
  const adapter = ADAPTERS[name];
241
253
  const ask = async () => {
242
- const run = await exec(installed.get(name).binary, adapter.args(prompt, modelFor(name)), { cwd, timeoutMs: settings.timeout_ms, unset: judgeUnset(name, settings.judge_env) });
254
+ const run = await runJudge(name, prompt, modelFor(name));
243
255
  progress(`Rigour reviewer: ${name} finished in ${Math.round((Date.now() - started) / 1000)}s (exit ${run.exitCode})`);
244
256
  const answer = adapter.answer(run.stdout);
245
257
  store.addSpend(1, answer.costUsd); // every run counts against the caps, an answer or not
@@ -293,7 +305,7 @@ async function review(cwd, base, config, exec, progress, options) {
293
305
  },
294
306
  ask: async (judge, asked) => {
295
307
  const name = judge;
296
- const run = await exec(installed.get(name).binary, ADAPTERS[name].args(crossExamPrompt(repoRoot, head.slice(0, 9), diffFile, asked), settings.cross_models[name] ?? modelFor(name)), { cwd, timeoutMs: settings.timeout_ms, unset: judgeUnset(name, settings.judge_env) });
308
+ const run = await runJudge(name, crossExamPrompt(repoRoot, head.slice(0, 9), diffFile, asked), settings.cross_models[name] ?? modelFor(name));
297
309
  const answer = ADAPTERS[name].answer(run.stdout);
298
310
  store.addSpend(1, answer.costUsd);
299
311
  reserved--;
@@ -256,6 +256,33 @@ describe('the reviewer', () => {
256
256
  expect(again.cached).toBe(true);
257
257
  expect(again.record?.integrity).toBe(first.record?.integrity);
258
258
  });
259
+ it('reviews through the API judge when the team configured one and its key is set, with the same prompt and accounting', async () => {
260
+ const seen = seenNow();
261
+ const calls = [];
262
+ const fetchImpl = (async (_url, init) => {
263
+ const body = JSON.parse(init.body);
264
+ calls.push(body);
265
+ const last = body.messages.at(-1);
266
+ const message = last.role === 'user'
267
+ ? { role: 'assistant', content: null, tool_calls: [{ id: 't1', type: 'function', function: { name: 'read_file', arguments: JSON.stringify({ path: /(\S+full\.diff)/.exec(last.content)[1] }) } }] }
268
+ : { role: 'assistant', content: JSON.stringify({ ...EMPTY, findings: [{ class: 'correctness', file: 'src/job.ts', line: 2, issue: 'returns before the lock', input: 'two runs', consequence: 'two emails', quote: 'return 1;', severity: 'blocking' }] }) };
269
+ return new Response(JSON.stringify({ choices: [{ message }], usage: { prompt_tokens: 100, completion_tokens: 20, cost: 0.05 } }), { status: 200 });
270
+ });
271
+ const apiConfig = ConfigSchema.parse({ version: 1, review: { reviewer: { enabled: true, reviewers: ['api'], api: { url: 'https://example.test/v1', model: 'qwen3-coder', key_env: 'TEST_JUDGE_KEY' }, reasoning: { api: 'low' } } } });
272
+ const without = await runReviewer(repo, 'main', apiConfig, fakes(() => '', seen), () => undefined, { fetch: fetchImpl });
273
+ expect(without).toMatchObject({ outcome: 'unavailable', reason: expect.stringContaining('no reviewer installed') }); // the key is not set
274
+ process.env.TEST_JUDGE_KEY = 'secret';
275
+ try {
276
+ const result = await runReviewer(repo, 'main', apiConfig, fakes(() => '', seen), () => undefined, { fetch: fetchImpl, force: true });
277
+ expect(result).toMatchObject({ outcome: 'findings', reviewers: ['api'], costUsd: 0.1, items: [expect.objectContaining({ issue: 'returns before the lock', reviewer: 'api' })] });
278
+ expect(calls[0].reasoning_effort).toBe('low');
279
+ expect(calls[0].messages[1].content).toContain('full.diff'); // the same prompt a CLI judge gets
280
+ expect(result.record?.judges).toEqual([{ reviewer: 'api', version: 'qwen3-coder', cost_usd: 0.1, turns: 2 }]);
281
+ }
282
+ finally {
283
+ delete process.env.TEST_JUDGE_KEY;
284
+ }
285
+ });
259
286
  it('asks a judge once more after an answer that is not a verdict, and is unavailable only when the second is not one either', async () => {
260
287
  const seen = seenNow();
261
288
  let calls = 0;
@@ -393,7 +420,7 @@ describe('verdicts', () => {
393
420
  return { verdict, ...account(verdict, undefined, verify) };
394
421
  };
395
422
  const broken = judged([{ id: 'r1', status: 'broken', file: 'src/job.ts', line: 2, quote: 'return 1;', evidence: 'no lock before the read' }]);
396
- expect(broken.open.map(i => [i.kind, i.class, i.issue])).toEqual([['rule', 'repo-rule', 'breaks a rule this repository wrote for itself (AGENTS.md): Every job must take the lock before its first read.']]);
423
+ expect(broken.open.map(i => [i.kind, i.class, i.issue, i.evidence])).toEqual([['rule', 'repo-rule', 'Every job must take the lock before its first read.', 'breaks a rule this repository wrote for itself (AGENTS.md): no lock before the read']]);
397
424
  expect(judged([{ id: 'r1', status: 'broken', file: 'src/job.ts', line: 2 }])).toMatchObject({ open: [], unverified: [expect.objectContaining({ kind: 'rule' })] }); // no quote: not shown as a block
398
425
  expect(judged([{ id: 'r2', status: 'broken', file: 'src/job.ts', line: 2, quote: 'return 1;' }])).toMatchObject({ open: [], advisory: [expect.objectContaining({ class: 'repo-rule' })] }); // guidance: shown, never a block
399
426
  expect(judged([{ id: 'r1', status: 'followed' }, { id: 'r1', status: 'not-applicable' }])).toMatchObject({ open: [], notes: [], advisory: [], unverified: [] });
@@ -410,9 +437,14 @@ describe('verdicts', () => {
410
437
  const { open } = account(verdict, undefined, checkoutVerifier(repo));
411
438
  expect(open.map(i => [i.kind, i.class, i.locations ?? []])).toEqual([
412
439
  ['prior', 'prior point', []], ['prior', 'prior point', []],
413
- ['finding', 'production-cost', [{ file: 'a.ts', line: 1 }]], // the same point in another file: one item, both places
414
- ['finding', 'correctness', []], // a different class is a different point
440
+ // The same point in another file, and the same point said as another class on the next line: one item, every place.
441
+ ['finding', 'production-cost', [{ file: 'a.ts', line: 1 }, { file: 'src/job.ts', line: 2 }]],
415
442
  ]);
443
+ // A rule break and the finding it caused, on the same lines and in like words, are one item.
444
+ const twice = account({ ...EMPTY, prior_points: [], findings: [{ class: 'correctness', file: 'src/job.ts', line: 2, issue: 'the raw table name is inlined instead of the JOBS_TABLE constant', input: 'any run', consequence: 'a rename misses it', quote: 'return 1;' }],
445
+ rules: [{ id: 'r', status: 'broken', file: 'src/job.ts', line: 2, quote: 'return 1;', rule: 'Import the JOBS_TABLE constant; do not inline the raw table name again.', source: 'AGENTS.md', requirement: true }] }, undefined, checkoutVerifier(repo));
446
+ expect(twice.open.map(i => i.class)).toEqual(['repo-rule']);
447
+ expect(twice.open[0].locations).toEqual([{ file: 'src/job.ts', line: 2 }]);
416
448
  });
417
449
  it('keeps reads, scans, redundancy and merge impact as notes with stable ids, and answers non-blocking points in the reply', () => {
418
450
  const verdict = {
@@ -94,6 +94,8 @@ export declare function matchLessons(lessons: ReviewLesson[], change: ChangeShap
94
94
  includeCandidates?: boolean;
95
95
  limit?: number;
96
96
  standards?: number;
97
+ perFile?: number;
98
+ excludePr?: number;
97
99
  }): ReviewLesson[];
98
100
  /** RIGOUR_REVIEW_LESSONS points at a lessons file outside the clone (CI, or a team's shared copy). */
99
101
  export declare function lessonsPath(cwd: string): string;
@@ -179,6 +179,9 @@ export function mergeLessons(existing, incoming) {
179
179
  * change; the best-evidenced few follow the file lessons.
180
180
  */
181
181
  export function matchLessons(lessons, change, options = {}) {
182
+ // A lesson whose only evidence is the pull request under review is already in front of the judge as the reviewer's own points.
183
+ if (options.excludePr !== undefined)
184
+ lessons = lessons.filter(l => !l.evidence.length || l.evidence.some(e => e.pr !== options.excludePr));
182
185
  const dirs = new Set(change.files.map(f => path.posix.dirname(f)));
183
186
  const scored = lessons
184
187
  .filter(l => !!l.file && (l.state === 'verified' || (options.includeCandidates && l.state === 'candidate')) && !NOT_CODE.test(l.file))
@@ -199,7 +202,20 @@ export function matchLessons(lessons, change, options = {}) {
199
202
  .sort((a, b) => b.shared - a.shared || b.lesson.evidence.length - a.lesson.evidence.length)
200
203
  .slice(0, options.standards ?? MAX_STANDARDS)
201
204
  .map(s => s.lesson);
202
- return [...scored.slice(0, options.limit ?? 5).map(s => s.lesson), ...standards];
205
+ // On a large change, one file's many lessons must not crowd out another file's only one: a cap per file, then the total.
206
+ const perFile = options.perFile ?? Infinity;
207
+ const taken = [];
208
+ const perFileCount = new Map();
209
+ for (const { lesson } of scored) {
210
+ if (taken.length >= (options.limit ?? 5))
211
+ break;
212
+ const n = perFileCount.get(lesson.file) ?? 0;
213
+ if (n >= perFile)
214
+ continue;
215
+ perFileCount.set(lesson.file, n + 1);
216
+ taken.push(lesson);
217
+ }
218
+ return [...taken, ...standards];
203
219
  }
204
220
  /** RIGOUR_REVIEW_LESSONS points at a lessons file outside the clone (CI, or a team's shared copy). */
205
221
  export function lessonsPath(cwd) {
@@ -29,6 +29,9 @@ describe('repository rules', () => {
29
29
  expect(rules[1]).toMatchObject({ paths: ['src/lib/delivery.ts'], symbols: ['deliverOrder'] });
30
30
  expect(rules[0].paths).toEqual(['migrations/']);
31
31
  expect(rules.map(r => r.requirement)).toEqual([true, true, false]); // "never", "every"; "prefer" is guidance
32
+ // A section that only describes an exception, with no imperative, is guidance; one that ends in an imperative is a requirement.
33
+ const [exceptionOnly, withImperative] = splitRules('AGENTS.md', '- **One narrow exception:** the queue table is still literally named `study_jobs`, not renamed with the feature.\n\n- **One narrow exception:** the queue table is still literally named `study_jobs`. Import the `JOBS_TABLE` constant; do not inline the raw table name again.\n');
34
+ expect([exceptionOnly.requirement, withImperative.requirement]).toEqual([false, true]);
32
35
  expect(rules[0].id).toMatch(/^[0-9a-f]{10}$/);
33
36
  expect(splitRules('AGENTS.md', AGENTS)[0].id).toBe(rules[0].id); // stable across runs
34
37
  });
@@ -95,6 +95,15 @@ describe('lessons', () => {
95
95
  const change = { files: ['src/orders.ts'], symbols: new Set(['insert', 'insertOrder', 'orderId']) };
96
96
  expect(matchLessons(lessons, change).map(l => l.id)).toEqual(['2', '1']); // two specific shared names outrank a same-file lesson with only generic ones
97
97
  expect(matchLessons(lessons, change, { includeCandidates: true }).map(l => l.id)).toEqual(['2', '1', '3']);
98
+ // A lesson learned only from the pull request under review is its own reviews, already in front of the judge.
99
+ const own = { ...base, id: '7', text: 'from this very pull request', file: 'src/orders.ts', state: 'verified', evidence: [{ pr: 42, comment: 'c', author: 'r' }] };
100
+ const also = { ...own, id: '8', text: 'from this and another', evidence: [{ pr: 42, comment: 'c', author: 'r' }, { pr: 3, comment: 'd', author: 'r' }] };
101
+ expect(matchLessons([...lessons, own, also], change, { excludePr: 42 }).map(l => l.id).sort()).toEqual(['1', '2', '8']); // '7' is left out; '8' has evidence elsewhere too
102
+ // One file's many lessons never crowd out another file's only one.
103
+ const many = Array.from({ length: 6 }, (_, i) => ({ ...base, id: `m${i}`, text: `orders lesson ${i}`, file: 'src/orders.ts', state: 'verified', symbols: ['insertOrder', 'orderId'] }));
104
+ const lone = { ...base, id: 'lone', text: 'the only lesson about the manifest', file: 'src/manifest.sha', state: 'verified' };
105
+ const served = matchLessons([...many, lone], { files: ['src/orders.ts', 'src/manifest.sha'], symbols: new Set(['insertOrder', 'orderId']) }, { limit: 4, perFile: 3 });
106
+ expect(served.map(l => l.id)).toEqual(['m0', 'm1', 'm2', 'lone']);
98
107
  });
99
108
  });
100
109
  describe('learning from one pull request as it goes', () => {
@@ -4,8 +4,8 @@ export type LessonMode = 'verified' | 'all' | 'off';
4
4
  export declare const DEFAULT_LESSON_MODE: LessonMode;
5
5
  /** The team's review lessons in play for this mode: none when off, verified ones by default. */
6
6
  export declare function activeLessons(cwd: string, mode?: LessonMode): ReviewLesson[];
7
- /** `standards`: how many team standards may come with the file lessons (a judge reading a whole pull request takes more than an agent's one question). */
8
- export declare function lessonsForDiff(cwd: string, diff: string, mode?: LessonMode, standards?: number): ReviewLesson[];
7
+ /** `standards`, `limit`, `perFile`: how many team standards and file lessons may come, and how many per file (a judge reading a whole pull request takes more than an agent's one question). */
8
+ export declare function lessonsForDiff(cwd: string, diff: string, mode?: LessonMode, standards?: number, limit?: number, perFile?: number, excludePr?: number): ReviewLesson[];
9
9
  /** The points this team rejected that a change touches: what the judges are told is settled. */
10
10
  export declare function rejectedForDiff(cwd: string, diff: string): ReviewLesson[];
11
11
  /** A lesson as a judge or agent sees it, in one place. */