@rigour-labs/core 6.7.6 → 6.7.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/review/backtest-init.d.ts +15 -0
- package/dist/review/backtest-init.js +30 -16
- package/dist/review/backtest-last.d.ts +40 -0
- package/dist/review/backtest-last.js +116 -0
- package/dist/review/backtest-last.test.d.ts +1 -0
- package/dist/review/backtest-last.test.js +109 -0
- package/dist/review/backtest.d.ts +26 -0
- package/dist/review/backtest.js +4 -4
- package/dist/review/reviewer/adapters.d.ts +11 -4
- package/dist/review/reviewer/adapters.js +44 -5
- package/dist/review/reviewer/adapters.test.js +16 -1
- package/dist/review/reviewer/api-judge.d.ts +28 -0
- package/dist/review/reviewer/api-judge.js +161 -0
- package/dist/review/reviewer/api-judge.test.d.ts +1 -0
- package/dist/review/reviewer/api-judge.test.js +87 -0
- package/dist/review/reviewer/context.d.ts +2 -0
- package/dist/review/reviewer/context.js +4 -1
- package/dist/review/reviewer/settings.d.ts +10 -0
- package/dist/review/reviewer/settings.js +3 -1
- package/dist/review/reviewer/verdict.js +8 -3
- package/dist/review/reviewer.d.ts +2 -0
- package/dist/review/reviewer.js +17 -5
- package/dist/review/reviewer.test.js +35 -3
- package/dist/review-learning/lessons.d.ts +2 -0
- package/dist/review-learning/lessons.js +17 -1
- package/dist/review-learning/repo-rules.test.js +3 -0
- package/dist/review-learning/review-learning.test.js +9 -0
- package/dist/review-learning/team-lessons.d.ts +2 -2
- package/dist/review-learning/team-lessons.js +3 -3
- package/dist/templates/universal-config.js +1 -0
- package/dist/types/index.d.ts +74 -0
- package/dist/types/index.js +14 -0
- package/package.json +6 -6
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
import type { RunTrace } from './adapters.js';
|
|
2
|
+
export type Reasoning = 'low' | 'medium' | 'high';
|
|
3
|
+
export interface ApiJudgeOptions {
|
|
4
|
+
/** The API's base URL; `/chat/completions` is appended. */
|
|
5
|
+
url: string;
|
|
6
|
+
model: string;
|
|
7
|
+
key: string;
|
|
8
|
+
maxTurns: number;
|
|
9
|
+
timeoutMs: number;
|
|
10
|
+
cwd: string;
|
|
11
|
+
/** Where the tools may read: the checkout and the review's input folder. */
|
|
12
|
+
roots: string[];
|
|
13
|
+
reasoning?: Reasoning;
|
|
14
|
+
fetchImpl?: typeof fetch;
|
|
15
|
+
}
|
|
16
|
+
export interface ApiJudgeRun {
|
|
17
|
+
exitCode: number;
|
|
18
|
+
stdout: string;
|
|
19
|
+
stderr: string;
|
|
20
|
+
}
|
|
21
|
+
/** What a run's stdout carries on success, for the adapter to read. */
|
|
22
|
+
export interface ApiJudgeAnswer {
|
|
23
|
+
result: string;
|
|
24
|
+
usage: RunTrace['usage'];
|
|
25
|
+
cost_usd?: number;
|
|
26
|
+
trace: RunTrace;
|
|
27
|
+
}
|
|
28
|
+
export declare function runApiJudge(prompt: string, o: ApiJudgeOptions): Promise<ApiJudgeRun>;
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A judge reached through a model API instead of an agent CLI, so any model a team can call works as
|
|
3
|
+
* a reviewer: OpenAI-compatible chat completions with tools (OpenAI, OpenRouter, a local server, other
|
|
4
|
+
* vendors through a gateway). Rigour runs the loop itself: the model asks for a read-only tool, Rigour
|
|
5
|
+
* runs it inside the checkout (and the review's own input folder) and hands the result back, until
|
|
6
|
+
* the model answers. The same prompt, the same evidence contract and the same trace as a CLI judge;
|
|
7
|
+
* the key comes from an environment variable named in rigour.yml, never from the file.
|
|
8
|
+
*/
|
|
9
|
+
import fs from 'fs';
|
|
10
|
+
import path from 'path';
|
|
11
|
+
import { execa } from 'execa';
|
|
12
|
+
const SYSTEM = 'You review code with read-only tools. Read the files the task names with read_file (the review inputs are named by absolute path), search the repository with search, read history with git. When you are done, reply with the final answer the task asks for and nothing else.';
|
|
13
|
+
const MAX_RESULT_CHARS = 60_000;
|
|
14
|
+
const GIT_ALLOWED = new Set(['log', 'show', 'diff', 'blame', 'grep', 'ls-files', 'rev-parse', 'merge-base']);
|
|
15
|
+
/** Git options that write, or point git at another repository. */
|
|
16
|
+
const GIT_REFUSED = /^(--output|-o$|--git-dir|--work-tree|-C$|--exec-path|-c$|--config-env)/;
|
|
17
|
+
const TOOLS = [
|
|
18
|
+
{ type: 'function', function: { name: 'read_file', description: 'Read a file in the repository or the review inputs, or a line range of it.', parameters: { type: 'object', properties: { path: { type: 'string' }, start_line: { type: 'integer' }, end_line: { type: 'integer' } }, required: ['path'] } } },
|
|
19
|
+
{ type: 'function', function: { name: 'search', description: 'Search the tracked files for a regular expression (git grep -n), optionally under one path.', parameters: { type: 'object', properties: { pattern: { type: 'string' }, path: { type: 'string' } }, required: ['pattern'] } } },
|
|
20
|
+
{ type: 'function', function: { name: 'list_dir', description: 'List a directory in the repository or the review inputs.', parameters: { type: 'object', properties: { path: { type: 'string' } }, required: ['path'] } } },
|
|
21
|
+
{ type: 'function', function: { name: 'git', description: 'Run a read-only git command in the repository: log, show, diff, blame, grep, ls-files, rev-parse, merge-base.', parameters: { type: 'object', properties: { args: { type: 'array', items: { type: 'string' } } }, required: ['args'] } } },
|
|
22
|
+
];
|
|
23
|
+
const n = (v) => (typeof v === 'number' && Number.isFinite(v) ? v : 0);
|
|
24
|
+
export async function runApiJudge(prompt, o) {
|
|
25
|
+
const fetchImpl = o.fetchImpl ?? fetch;
|
|
26
|
+
const deadline = Date.now() + o.timeoutMs;
|
|
27
|
+
const messages = [{ role: 'system', content: SYSTEM }, { role: 'user', content: prompt }];
|
|
28
|
+
const usage = { input: 0, cacheRead: 0, cacheWrite: 0, output: 0 };
|
|
29
|
+
const calls = [];
|
|
30
|
+
let cost;
|
|
31
|
+
const fail = (why) => ({ exitCode: 1, stdout: '', stderr: `api judge (${o.model}): ${why}` });
|
|
32
|
+
for (let turn = 1; turn <= o.maxTurns; turn++) {
|
|
33
|
+
const left = deadline - Date.now();
|
|
34
|
+
if (left <= 0)
|
|
35
|
+
return fail(`timed out after ${o.timeoutMs} ms`);
|
|
36
|
+
const controller = new AbortController();
|
|
37
|
+
const timer = setTimeout(() => controller.abort(), left);
|
|
38
|
+
let response;
|
|
39
|
+
try {
|
|
40
|
+
response = await fetchImpl(`${o.url.replace(/\/$/, '')}/chat/completions`, {
|
|
41
|
+
method: 'POST',
|
|
42
|
+
headers: { 'content-type': 'application/json', authorization: `Bearer ${o.key}` },
|
|
43
|
+
body: JSON.stringify({ model: o.model, messages, tools: TOOLS, tool_choice: 'auto', ...(o.reasoning ? { reasoning_effort: o.reasoning } : {}) }),
|
|
44
|
+
signal: controller.signal,
|
|
45
|
+
});
|
|
46
|
+
}
|
|
47
|
+
catch (error) {
|
|
48
|
+
return fail(`request failed: ${error instanceof Error ? error.message : String(error)}`);
|
|
49
|
+
}
|
|
50
|
+
finally {
|
|
51
|
+
clearTimeout(timer);
|
|
52
|
+
}
|
|
53
|
+
if (!response.ok)
|
|
54
|
+
return fail(`HTTP ${response.status}: ${(await response.text()).slice(0, 300)}`);
|
|
55
|
+
let body;
|
|
56
|
+
try {
|
|
57
|
+
body = await response.json();
|
|
58
|
+
}
|
|
59
|
+
catch {
|
|
60
|
+
return fail('the answer was not JSON');
|
|
61
|
+
}
|
|
62
|
+
const u = body.usage ?? {};
|
|
63
|
+
const cached = n(u.prompt_tokens_details?.cached_tokens);
|
|
64
|
+
usage.input += Math.max(0, n(u.prompt_tokens) - cached);
|
|
65
|
+
usage.cacheRead += cached;
|
|
66
|
+
usage.output += n(u.completion_tokens);
|
|
67
|
+
if (typeof u.cost === 'number')
|
|
68
|
+
cost = (cost ?? 0) + u.cost;
|
|
69
|
+
const message = body.choices?.[0]?.message;
|
|
70
|
+
if (!message)
|
|
71
|
+
return fail('no choices in the answer');
|
|
72
|
+
messages.push(message);
|
|
73
|
+
const toolCalls = Array.isArray(message.tool_calls) ? message.tool_calls : [];
|
|
74
|
+
if (toolCalls.length === 0) {
|
|
75
|
+
const answer = { result: String(message.content ?? ''), usage, ...(cost !== undefined ? { cost_usd: cost } : {}), trace: { turns: turn, usage, calls } };
|
|
76
|
+
return { exitCode: 0, stdout: JSON.stringify(answer), stderr: '' };
|
|
77
|
+
}
|
|
78
|
+
for (const call of toolCalls) {
|
|
79
|
+
const ran = await runTool(call, o);
|
|
80
|
+
calls.push({ turn, tool: ran.tool, target: ran.target, resultChars: ran.result.length });
|
|
81
|
+
messages.push({ role: 'tool', tool_call_id: call.id, content: ran.result });
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
return fail(`no answer within ${o.maxTurns} turns`);
|
|
85
|
+
}
|
|
86
|
+
/** One tool call, inside the allowed roots only; the trace names tools as the CLI judges do (Read, Grep, Glob, Bash). */
|
|
87
|
+
async function runTool(call, o) {
|
|
88
|
+
const name = String(call?.function?.name ?? '');
|
|
89
|
+
let args = {};
|
|
90
|
+
try {
|
|
91
|
+
args = JSON.parse(call?.function?.arguments || '{}');
|
|
92
|
+
}
|
|
93
|
+
catch {
|
|
94
|
+
return { tool: name, target: '', result: 'refused: the arguments were not JSON' };
|
|
95
|
+
}
|
|
96
|
+
const clip = (text) => (text.length > MAX_RESULT_CHARS ? `${text.slice(0, MAX_RESULT_CHARS)}\n…[truncated at ${MAX_RESULT_CHARS} characters]` : text);
|
|
97
|
+
if (name === 'read_file') {
|
|
98
|
+
const target = inside(String(args.path ?? ''), o);
|
|
99
|
+
if (!target)
|
|
100
|
+
return { tool: 'Read', target: String(args.path ?? ''), result: 'refused: outside the repository and the review inputs' };
|
|
101
|
+
try {
|
|
102
|
+
const lines = fs.readFileSync(target, 'utf8').split('\n');
|
|
103
|
+
const start = Math.max(1, Number(args.start_line) || 1);
|
|
104
|
+
const end = Math.min(lines.length, Number(args.end_line) || lines.length);
|
|
105
|
+
return { tool: 'Read', target, result: clip(lines.slice(start - 1, end).map((l, i) => `${start + i}: ${l}`).join('\n')) };
|
|
106
|
+
}
|
|
107
|
+
catch (error) {
|
|
108
|
+
return { tool: 'Read', target, result: `error: ${error instanceof Error ? error.message : String(error)}` };
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
if (name === 'list_dir') {
|
|
112
|
+
const target = inside(String(args.path ?? '.'), o);
|
|
113
|
+
if (!target)
|
|
114
|
+
return { tool: 'Glob', target: String(args.path ?? ''), result: 'refused: outside the repository and the review inputs' };
|
|
115
|
+
try {
|
|
116
|
+
return { tool: 'Glob', target, result: clip(fs.readdirSync(target, { withFileTypes: true }).map(e => (e.isDirectory() ? `${e.name}/` : e.name)).join('\n')) };
|
|
117
|
+
}
|
|
118
|
+
catch (error) {
|
|
119
|
+
return { tool: 'Glob', target, result: `error: ${error instanceof Error ? error.message : String(error)}` };
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
if (name === 'search') {
|
|
123
|
+
const pattern = String(args.pattern ?? '');
|
|
124
|
+
const under = args.path ? inside(String(args.path), o) : undefined;
|
|
125
|
+
if (args.path && !under)
|
|
126
|
+
return { tool: 'Grep', target: pattern, result: 'refused: outside the repository' };
|
|
127
|
+
const run = await execa('git', ['grep', '-n', '-I', '-e', pattern, ...(under ? ['--', path.relative(o.cwd, under) || '.'] : [])], { cwd: o.cwd, reject: false, maxBuffer: 16 * 1024 * 1024 });
|
|
128
|
+
return { tool: 'Grep', target: pattern, result: clip(run.stdout || (run.exitCode === 1 ? '(no match)' : run.stderr)) };
|
|
129
|
+
}
|
|
130
|
+
if (name === 'git') {
|
|
131
|
+
const gitArgs = Array.isArray(args.args) ? args.args.map(String) : [];
|
|
132
|
+
if (!GIT_ALLOWED.has(gitArgs[0] ?? '') || gitArgs.some(a => GIT_REFUSED.test(a)))
|
|
133
|
+
return { tool: 'Bash', target: `git ${gitArgs.join(' ')}`, result: 'refused: only read-only git commands, without options that write or point elsewhere' };
|
|
134
|
+
const run = await execa('git', ['--no-pager', ...gitArgs], { cwd: o.cwd, reject: false, maxBuffer: 16 * 1024 * 1024 });
|
|
135
|
+
return { tool: 'Bash', target: `git ${gitArgs.join(' ')}`, result: clip(run.stdout || run.stderr) };
|
|
136
|
+
}
|
|
137
|
+
return { tool: name, target: '', result: `refused: no tool named ${name}` };
|
|
138
|
+
}
|
|
139
|
+
/** The real path of `p` when it lies under one of the roots; undefined otherwise (a symlink out is outside). */
|
|
140
|
+
function inside(p, o) {
|
|
141
|
+
const resolved = path.resolve(o.cwd, p);
|
|
142
|
+
let real;
|
|
143
|
+
try {
|
|
144
|
+
real = fs.realpathSync(resolved);
|
|
145
|
+
}
|
|
146
|
+
catch {
|
|
147
|
+
return undefined;
|
|
148
|
+
}
|
|
149
|
+
for (const root of o.roots) {
|
|
150
|
+
let base;
|
|
151
|
+
try {
|
|
152
|
+
base = fs.realpathSync(root);
|
|
153
|
+
}
|
|
154
|
+
catch {
|
|
155
|
+
continue;
|
|
156
|
+
}
|
|
157
|
+
if (real === base || real.startsWith(base + path.sep))
|
|
158
|
+
return real;
|
|
159
|
+
}
|
|
160
|
+
return undefined;
|
|
161
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
import { execFileSync } from 'child_process';
|
|
2
|
+
import fs from 'fs';
|
|
3
|
+
import os from 'os';
|
|
4
|
+
import path from 'path';
|
|
5
|
+
import { afterEach, beforeEach, describe, expect, it } from 'vitest';
|
|
6
|
+
import { runApiJudge } from './api-judge.js';
|
|
7
|
+
let repo;
|
|
8
|
+
let work;
|
|
9
|
+
beforeEach(() => {
|
|
10
|
+
repo = fs.mkdtempSync(path.join(os.tmpdir(), 'api-judge-'));
|
|
11
|
+
work = fs.mkdtempSync(path.join(os.tmpdir(), 'api-judge-inputs-'));
|
|
12
|
+
execFileSync('git', ['-C', repo, 'init', '-q']);
|
|
13
|
+
fs.mkdirSync(path.join(repo, 'src'));
|
|
14
|
+
fs.writeFileSync(path.join(repo, 'src/job.ts'), 'export function job() {\n return 1;\n}\n');
|
|
15
|
+
execFileSync('git', ['-C', repo, 'add', '-A']);
|
|
16
|
+
execFileSync('git', ['-C', repo, '-c', 'user.email=t@x', '-c', 'user.name=t', 'commit', '-qm', 'init']);
|
|
17
|
+
fs.writeFileSync(path.join(work, 'full.diff'), '+++ b/src/job.ts\n');
|
|
18
|
+
});
|
|
19
|
+
afterEach(() => { for (const d of [repo, work])
|
|
20
|
+
fs.rmSync(d, { recursive: true, force: true }); });
|
|
21
|
+
/** A model that plays a scripted conversation: each entry is what it answers to the next request. */
|
|
22
|
+
function model(turns) {
|
|
23
|
+
const seen = [];
|
|
24
|
+
let i = 0;
|
|
25
|
+
const fetchImpl = (async (_url, init) => {
|
|
26
|
+
const body = JSON.parse(init.body);
|
|
27
|
+
seen.push(body);
|
|
28
|
+
const turn = turns[i++] ?? { text: 'done' };
|
|
29
|
+
if (turn.status)
|
|
30
|
+
return new Response('nope', { status: turn.status });
|
|
31
|
+
const message = turn.tools
|
|
32
|
+
? { role: 'assistant', content: null, tool_calls: turn.tools.map((t, k) => ({ id: `c${i}-${k}`, type: 'function', function: { name: t.name, arguments: JSON.stringify(t.args) } })) }
|
|
33
|
+
: { role: 'assistant', content: turn.text };
|
|
34
|
+
return new Response(JSON.stringify({ choices: [{ message }], usage: turn.usage ?? { prompt_tokens: 100, completion_tokens: 10, prompt_tokens_details: { cached_tokens: 60 } } }), { status: 200, headers: { 'content-type': 'application/json' } });
|
|
35
|
+
});
|
|
36
|
+
return { fetchImpl, seen };
|
|
37
|
+
}
|
|
38
|
+
const options = (fetchImpl, extra = {}) => ({ url: 'https://example.test/v1', model: 'some-model', key: 'k', maxTurns: 10, timeoutMs: 30_000, cwd: repo, roots: [repo, work], fetchImpl, ...extra });
|
|
39
|
+
describe('the API judge', () => {
|
|
40
|
+
it('runs the tools the model asks for inside the checkout and the inputs, then returns its answer with the usage and the trace', async () => {
|
|
41
|
+
const { fetchImpl, seen } = model([
|
|
42
|
+
{ tools: [{ name: 'read_file', args: { path: path.join(work, 'full.diff') } }, { name: 'read_file', args: { path: 'src/job.ts', start_line: 2, end_line: 2 } }] },
|
|
43
|
+
{ tools: [{ name: 'search', args: { pattern: 'return' } }, { name: 'git', args: { args: ['log', '-1', '--format=%s'] } }, { name: 'list_dir', args: { path: 'src' } }] },
|
|
44
|
+
{ text: '{"prior_points":[]}', usage: { prompt_tokens: 50, completion_tokens: 5, cost: 0.01 } },
|
|
45
|
+
]);
|
|
46
|
+
const run = await runApiJudge('review this', options(fetchImpl));
|
|
47
|
+
expect(run.exitCode).toBe(0);
|
|
48
|
+
const answer = JSON.parse(run.stdout);
|
|
49
|
+
expect(answer.result).toBe('{"prior_points":[]}');
|
|
50
|
+
expect(answer.usage).toEqual({ input: 130, cacheRead: 120, cacheWrite: 0, output: 25 });
|
|
51
|
+
expect(answer.cost_usd).toBe(0.01);
|
|
52
|
+
expect(answer.trace.turns).toBe(3);
|
|
53
|
+
expect(answer.trace.calls.map(c => c.tool)).toEqual(['Read', 'Read', 'Grep', 'Bash', 'Glob']);
|
|
54
|
+
const toolResults = (request) => request.messages.filter((m) => m.role === 'tool').map((m) => m.content);
|
|
55
|
+
expect(toolResults(seen[1])).toEqual(['1: +++ b/src/job.ts\n2: ', '2: return 1;']); // turn 1's reads, as the model got them
|
|
56
|
+
expect(toolResults(seen[2]).slice(2)).toEqual(['src/job.ts:2: return 1;', 'init', 'job.ts']); // turn 2's search, git and listing
|
|
57
|
+
expect(seen[0].messages[0].role).toBe('system');
|
|
58
|
+
expect(seen[0].tools.map((t) => t.function.name)).toEqual(['read_file', 'search', 'list_dir', 'git']);
|
|
59
|
+
});
|
|
60
|
+
it('refuses to read outside the roots, a symlink out included, and refuses git that writes or points elsewhere', async () => {
|
|
61
|
+
fs.symlinkSync(os.homedir(), path.join(repo, 'out'));
|
|
62
|
+
const { fetchImpl, seen } = model([
|
|
63
|
+
{ tools: [{ name: 'read_file', args: { path: '/etc/hosts' } }, { name: 'read_file', args: { path: 'out/.bashrc' } }, { name: 'git', args: { args: ['log', '--output=/tmp/x'] } }, { name: 'git', args: { args: ['push'] } }, { name: 'nope', args: {} }] },
|
|
64
|
+
{ text: 'ok' },
|
|
65
|
+
]);
|
|
66
|
+
await runApiJudge('p', options(fetchImpl));
|
|
67
|
+
const results = seen[1].messages.filter((m) => m.role === 'tool').map((m) => m.content);
|
|
68
|
+
expect(results).toEqual([
|
|
69
|
+
'refused: outside the repository and the review inputs', 'refused: outside the repository and the review inputs',
|
|
70
|
+
'refused: only read-only git commands, without options that write or point elsewhere', 'refused: only read-only git commands, without options that write or point elsewhere',
|
|
71
|
+
'refused: no tool named nope',
|
|
72
|
+
]);
|
|
73
|
+
});
|
|
74
|
+
it('fails closed on a turn cap, an HTTP error and a timeout, saying why', async () => {
|
|
75
|
+
const loop = model(Array.from({ length: 5 }, () => ({ tools: [{ name: 'list_dir', args: { path: '.' } }] })));
|
|
76
|
+
expect(await runApiJudge('p', options(loop.fetchImpl, { maxTurns: 3 }))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('no answer within 3 turns') });
|
|
77
|
+
expect(await runApiJudge('p', options(model([{ status: 429 }]).fetchImpl))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('HTTP 429') });
|
|
78
|
+
const slow = (async (_u, init) => new Promise((_r, reject) => init.signal.addEventListener('abort', () => reject(new Error('aborted')))));
|
|
79
|
+
expect(await runApiJudge('p', options(slow, { timeoutMs: 50 }))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('request failed') });
|
|
80
|
+
});
|
|
81
|
+
it('passes the reasoning effort when asked', async () => {
|
|
82
|
+
const { fetchImpl, seen } = model([{ text: 'ok' }]);
|
|
83
|
+
await runApiJudge('p', options(fetchImpl, { reasoning: 'low' }));
|
|
84
|
+
expect(seen[0].reasoning_effort).toBe('low');
|
|
85
|
+
expect(seen[0].model).toBe('some-model');
|
|
86
|
+
});
|
|
87
|
+
});
|
|
@@ -31,6 +31,8 @@ export interface ContextInput {
|
|
|
31
31
|
router: RouterPolicy | undefined;
|
|
32
32
|
/** Which of the team's review lessons the judges see (gates.deep.review_lessons): verified by default, all, or off. */
|
|
33
33
|
lessons?: LessonMode;
|
|
34
|
+
/** The pull request under review: lessons learned only from it are its own reviews, which the judge already reads. */
|
|
35
|
+
pr?: number;
|
|
34
36
|
/** The previous verdict's panel decisions, and the files changed since it. */
|
|
35
37
|
previousPanel: PanelItem[] | undefined;
|
|
36
38
|
touched: Set<string>;
|
|
@@ -24,6 +24,9 @@ export const REVIEW_DISMISSALS = path.join('.rigour', 'dismissed-review-items.js
|
|
|
24
24
|
const MAX_DOCS = 10;
|
|
25
25
|
/** Team standards a judge is shown with the lessons about the changed files. */
|
|
26
26
|
const JUDGE_STANDARDS = 15;
|
|
27
|
+
/** File lessons a judge is shown: on a pull request touching a hundred files, enough for every file, at most this many per file. */
|
|
28
|
+
const JUDGE_FILE_LESSONS = 30;
|
|
29
|
+
const JUDGE_LESSONS_PER_FILE = 3;
|
|
27
30
|
/** Rules from the repository's own rules files a judge is asked to answer, most relevant first. */
|
|
28
31
|
const JUDGE_RULES = 15;
|
|
29
32
|
const MAX_SETTLED = 40;
|
|
@@ -78,7 +81,7 @@ export function buildContext(input) {
|
|
|
78
81
|
task = undefined;
|
|
79
82
|
}
|
|
80
83
|
// A judge reads the whole pull request: more of what the team taught fits than an agent's one question at the stop.
|
|
81
|
-
const lessons = input.lessons === 'off' ? [] : lessonsForDiff(input.cwd, input.diff, input.lessons, JUDGE_STANDARDS).map(lessonView);
|
|
84
|
+
const lessons = input.lessons === 'off' ? [] : lessonsForDiff(input.cwd, input.diff, input.lessons, JUDGE_STANDARDS, JUDGE_FILE_LESSONS, JUDGE_LESSONS_PER_FILE, input.pr).map(lessonView);
|
|
82
85
|
if (lessons.length)
|
|
83
86
|
sections.push(`## Lessons this team taught on earlier reviews, for what this change touches (context: a lesson never blocks on its own; a finding still needs its quote)\n${lessons.map(l => `- ${describeLesson(l)}`).join('\n')}`);
|
|
84
87
|
// The repository's own rules, always: the reviewer is the boundary, and what the team wrote is the standard it checks.
|
|
@@ -35,6 +35,16 @@ export interface ResolvedReviewer {
|
|
|
35
35
|
judges: 2 | 3;
|
|
36
36
|
escalate: 'always' | 'risk';
|
|
37
37
|
cross_models: Record<string, string>;
|
|
38
|
+
/** Reasoning effort per reviewer name (codex, api). */
|
|
39
|
+
reasoning: Record<string, 'low' | 'medium' | 'high'>;
|
|
40
|
+
/** The API judge, when the team configured one (review.reviewer.api). */
|
|
41
|
+
api?: {
|
|
42
|
+
url: string;
|
|
43
|
+
model: string;
|
|
44
|
+
key_env: string;
|
|
45
|
+
vendor?: 'anthropic' | 'openai' | 'google' | 'other';
|
|
46
|
+
max_turns: number;
|
|
47
|
+
};
|
|
38
48
|
/** Environment variables each judge's CLI must not see (the team's, never a person's). */
|
|
39
49
|
judge_env: Record<string, {
|
|
40
50
|
unset: string[];
|
|
@@ -13,7 +13,7 @@ const RANK = { single: 0, cross: 1, full: 2 };
|
|
|
13
13
|
/** How near a layer is to this run: the nearer wins. */
|
|
14
14
|
const NEAR = { flag: 0, env: 1, user: 2, team: 3 };
|
|
15
15
|
export function resolveReviewer(config, choice = {}, user = loadSettings().reviewer, env = process.env) {
|
|
16
|
-
const team = config.review?.reviewer ?? { enabled: false, on_push: 'background', reviewers: ['claude'], mode: 'single', models: {}, timeout_ms: 15 * 60_000, panel: 'off', mode_required: false, panel_max_items: 20, dismissals: false, judges: 2, escalate: 'always', cross_models: {}, judge_env: {} };
|
|
16
|
+
const team = config.review?.reviewer ?? { enabled: false, on_push: 'background', reviewers: ['claude'], mode: 'single', models: {}, timeout_ms: 15 * 60_000, panel: 'off', mode_required: false, panel_max_items: 20, dismissals: false, judges: 2, escalate: 'always', cross_models: {}, judge_env: {}, reasoning: {} };
|
|
17
17
|
const refused = [];
|
|
18
18
|
const envMode = parseMode(env.RIGOUR_REVIEWER_MODE);
|
|
19
19
|
const envPanel = parseSwitch(env.RIGOUR_REVIEWER_PANEL);
|
|
@@ -79,6 +79,8 @@ export function resolveReviewer(config, choice = {}, user = loadSettings().revie
|
|
|
79
79
|
escalate: requiredEscalation(team, user, refused),
|
|
80
80
|
cross_models: team.cross_models,
|
|
81
81
|
judge_env: team.judge_env ?? {},
|
|
82
|
+
reasoning: team.reasoning ?? {},
|
|
83
|
+
...(team.api ? { api: team.api } : {}),
|
|
82
84
|
source: { mode: modeSource, panel: panelSource },
|
|
83
85
|
required: { mode: team.mode_required, panel: requirePanel },
|
|
84
86
|
refused,
|
|
@@ -224,7 +224,8 @@ export function account(verdict, previousOpen, verify) {
|
|
|
224
224
|
for (const r of verdict.rules ?? []) {
|
|
225
225
|
if (r.status !== 'broken' || !r.rule)
|
|
226
226
|
continue;
|
|
227
|
-
|
|
227
|
+
// The rule's own words are the issue, so the same point found as a finding reads alike; where it came from is the evidence.
|
|
228
|
+
const item = { id: id('repo-rule', r.file, r.id), kind: 'rule', class: 'repo-rule', file: r.file, line: r.line, issue: r.rule, consequence: r.requirement ? 'the team wrote this rule as a requirement' : 'the team wrote this rule as guidance', ...(r.quote ? { quote: r.quote } : {}), evidence: `breaks a rule this repository wrote for itself (${r.source})${r.evidence ? `: ${r.evidence}` : ''}`, reviewer: r.reviewer };
|
|
228
229
|
if (r.requirement)
|
|
229
230
|
add(item);
|
|
230
231
|
else
|
|
@@ -274,8 +275,10 @@ export function account(verdict, previousOpen, verify) {
|
|
|
274
275
|
}
|
|
275
276
|
return { open: onePerRootCause(open), unverified, resolved, answerInReply, notes, advisory: onePerRootCause(advisory) };
|
|
276
277
|
}
|
|
277
|
-
/** How alike two items' words must be to be the same point made in two places. */
|
|
278
|
+
/** How alike two items' words must be to be the same point made in two places; and, on the same lines, to be one point said two ways. */
|
|
278
279
|
const SAME_POINT = 0.6;
|
|
280
|
+
const SAME_PLACE = 0.3;
|
|
281
|
+
const SAME_LINES = 3;
|
|
279
282
|
/**
|
|
280
283
|
* The same point found in several places is one item carrying every location, so a person reads one
|
|
281
284
|
* line, not one per file. Blocking is unchanged: the item blocks until every location is fixed.
|
|
@@ -283,7 +286,9 @@ const SAME_POINT = 0.6;
|
|
|
283
286
|
function onePerRootCause(items) {
|
|
284
287
|
const kept = [];
|
|
285
288
|
for (const item of items) {
|
|
286
|
-
|
|
289
|
+
// The same class in the same words anywhere, or any two non-human items on the same lines that read alike (a rule break and the finding it caused).
|
|
290
|
+
const nearby = (k) => !!k.file && k.file === item.file && k.line !== undefined && item.line !== undefined && Math.abs(k.line - item.line) <= SAME_LINES;
|
|
291
|
+
const same = item.kind === 'prior' ? undefined : kept.find(k => k.kind !== 'prior' && ((k.class === item.class && textSimilarity(k, item) >= SAME_POINT) || (nearby(k) && textSimilarity(k, item) >= SAME_PLACE)));
|
|
287
292
|
if (!same) {
|
|
288
293
|
kept.push(item);
|
|
289
294
|
continue;
|
|
@@ -9,6 +9,8 @@ export { defaultExec, githubEnv, githubToken, parseJsonArrays, type Exec, type P
|
|
|
9
9
|
export { itemLine, type OpenItem } from './reviewer/verdict.js';
|
|
10
10
|
export type ReviewerOutcome = 'passed' | 'findings' | 'unavailable' | 'skipped';
|
|
11
11
|
export interface ReviewerOptions {
|
|
12
|
+
/** For tests: what the API judge calls instead of fetch. */
|
|
13
|
+
fetch?: typeof fetch;
|
|
12
14
|
/** The pull request to read when the checkout is detached (a backtest). */
|
|
13
15
|
pr?: number;
|
|
14
16
|
/** ISO time: a review or comment posted from then on is not shown to the reviewer (a backtest). */
|
package/dist/review/reviewer.js
CHANGED
|
@@ -21,7 +21,8 @@
|
|
|
21
21
|
import fs from 'fs';
|
|
22
22
|
import os from 'os';
|
|
23
23
|
import path from 'path';
|
|
24
|
-
import {
|
|
24
|
+
import { runApiJudge } from './reviewer/api-judge.js';
|
|
25
|
+
import { ADAPTERS, apiVendor, isReviewerName, resolveAdapter, selectReviewers, vendorsOf } from './reviewer/adapters.js';
|
|
25
26
|
import { defaultExec, GH_TIMEOUT_MS, githubEnv } from './reviewer/exec.js';
|
|
26
27
|
import { bodyAsOf, findPullRequest, ghFor, humanReviews, linesChanged, mergesBaseIn, rulesText, sha } from './reviewer/inputs.js';
|
|
27
28
|
import { mergeImpact } from './reviewer/merge-impact.js';
|
|
@@ -76,13 +77,20 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
76
77
|
const candidates = settings.reviewers.filter(isReviewerName);
|
|
77
78
|
const installed = new Map();
|
|
78
79
|
for (const name of candidates) {
|
|
80
|
+
if (name === 'api') {
|
|
81
|
+
// The API judge is installed when the team configured it and the key it names is set.
|
|
82
|
+
if (settings.api && process.env[settings.api.key_env])
|
|
83
|
+
installed.set('api', { binary: 'api', version: settings.api.model });
|
|
84
|
+
continue;
|
|
85
|
+
}
|
|
79
86
|
const found = await resolveAdapter(ADAPTERS[name], cwd, exec);
|
|
80
87
|
if (found)
|
|
81
88
|
installed.set(name, found);
|
|
82
89
|
}
|
|
90
|
+
const vendorOf = (name) => (name === 'api' ? apiVendor(settings.api) : ADAPTERS[name].vendor);
|
|
83
91
|
const authors = vendorsOf(await git(['log', '--format=%(trailers:key=Co-Authored-By,valueonly)%(trailers:key=Co-authored-by,valueonly)', `${baseSha}..HEAD`]));
|
|
84
92
|
const mode = settings.mode;
|
|
85
|
-
let reviewers = selectReviewers(candidates, mode, authors, new Set(installed.keys()), settings.judges);
|
|
93
|
+
let reviewers = selectReviewers(candidates, mode, authors, new Set(installed.keys()), settings.judges, vendorOf);
|
|
86
94
|
if (reviewers.length === 0)
|
|
87
95
|
return none('unavailable', `no reviewer installed: ${candidates.map(n => ADAPTERS[n].binary).join(', ') || 'review.reviewer.reviewers is empty'}`);
|
|
88
96
|
modeRecord = modeRan(settings, reviewers, candidates, installed);
|
|
@@ -133,7 +141,7 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
133
141
|
const sincePrevious = previousIsAncestor ? new Set((await git(['diff', '--name-only', `${previous.head}..HEAD`])).split('\n').filter(Boolean)) : new Set();
|
|
134
142
|
const changedFiles = [...fullDiff.matchAll(/^diff --git a\/.* b\/(.*)$/gm)].map(m => m[1]);
|
|
135
143
|
const context = buildContext({
|
|
136
|
-
cwd, stateRoot, dismissals, diff: fullDiff, router: config.gates.deep?.router, lessons: config.gates.deep?.review_lessons, touched: sincePrevious, checks: options.checks ?? [],
|
|
144
|
+
cwd, stateRoot, dismissals, diff: fullDiff, router: config.gates.deep?.router, lessons: config.gates.deep?.review_lessons, ...(pr ? { pr: pr.number } : {}), touched: sincePrevious, checks: options.checks ?? [],
|
|
137
145
|
previousPanel: previousIsAncestor ? store.readJson(previous.verdict)?.panel?.items : undefined,
|
|
138
146
|
docs: await relatedDocs(cwd, changedFiles, exec),
|
|
139
147
|
});
|
|
@@ -199,6 +207,10 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
199
207
|
if (over)
|
|
200
208
|
return none(settings.required.panel || settings.required.mode ? 'unavailable' : 'skipped', over, { reviewers, scope, why, pr: pr?.number });
|
|
201
209
|
const work = fs.mkdtempSync(path.join(os.tmpdir(), 'rigour-reviewer-'));
|
|
210
|
+
// One judge run, by CLI or by API: the same prompt, the same cost accounting, the same trace.
|
|
211
|
+
const runJudge = (name, prompt, model) => name === 'api'
|
|
212
|
+
? runApiJudge(prompt, { url: settings.api.url, model: settings.api.model, key: process.env[settings.api.key_env] ?? '', maxTurns: settings.api.max_turns, timeoutMs: settings.timeout_ms, cwd, roots: [cwd, work], ...(settings.reasoning[name] ? { reasoning: settings.reasoning[name] } : {}), ...(options.fetch ? { fetchImpl: options.fetch } : {}) })
|
|
213
|
+
: exec(installed.get(name).binary, ADAPTERS[name].args(prompt, model, { reasoning: settings.reasoning[name] }), { cwd, timeoutMs: settings.timeout_ms, unset: judgeUnset(name, settings.judge_env) });
|
|
202
214
|
try {
|
|
203
215
|
const file = (name, text) => {
|
|
204
216
|
const target = path.join(work, name);
|
|
@@ -239,7 +251,7 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
239
251
|
const answers = await Promise.all(reviewers.map(async (name) => {
|
|
240
252
|
const adapter = ADAPTERS[name];
|
|
241
253
|
const ask = async () => {
|
|
242
|
-
const run = await
|
|
254
|
+
const run = await runJudge(name, prompt, modelFor(name));
|
|
243
255
|
progress(`Rigour reviewer: ${name} finished in ${Math.round((Date.now() - started) / 1000)}s (exit ${run.exitCode})`);
|
|
244
256
|
const answer = adapter.answer(run.stdout);
|
|
245
257
|
store.addSpend(1, answer.costUsd); // every run counts against the caps, an answer or not
|
|
@@ -293,7 +305,7 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
293
305
|
},
|
|
294
306
|
ask: async (judge, asked) => {
|
|
295
307
|
const name = judge;
|
|
296
|
-
const run = await
|
|
308
|
+
const run = await runJudge(name, crossExamPrompt(repoRoot, head.slice(0, 9), diffFile, asked), settings.cross_models[name] ?? modelFor(name));
|
|
297
309
|
const answer = ADAPTERS[name].answer(run.stdout);
|
|
298
310
|
store.addSpend(1, answer.costUsd);
|
|
299
311
|
reserved--;
|
|
@@ -256,6 +256,33 @@ describe('the reviewer', () => {
|
|
|
256
256
|
expect(again.cached).toBe(true);
|
|
257
257
|
expect(again.record?.integrity).toBe(first.record?.integrity);
|
|
258
258
|
});
|
|
259
|
+
it('reviews through the API judge when the team configured one and its key is set, with the same prompt and accounting', async () => {
|
|
260
|
+
const seen = seenNow();
|
|
261
|
+
const calls = [];
|
|
262
|
+
const fetchImpl = (async (_url, init) => {
|
|
263
|
+
const body = JSON.parse(init.body);
|
|
264
|
+
calls.push(body);
|
|
265
|
+
const last = body.messages.at(-1);
|
|
266
|
+
const message = last.role === 'user'
|
|
267
|
+
? { role: 'assistant', content: null, tool_calls: [{ id: 't1', type: 'function', function: { name: 'read_file', arguments: JSON.stringify({ path: /(\S+full\.diff)/.exec(last.content)[1] }) } }] }
|
|
268
|
+
: { role: 'assistant', content: JSON.stringify({ ...EMPTY, findings: [{ class: 'correctness', file: 'src/job.ts', line: 2, issue: 'returns before the lock', input: 'two runs', consequence: 'two emails', quote: 'return 1;', severity: 'blocking' }] }) };
|
|
269
|
+
return new Response(JSON.stringify({ choices: [{ message }], usage: { prompt_tokens: 100, completion_tokens: 20, cost: 0.05 } }), { status: 200 });
|
|
270
|
+
});
|
|
271
|
+
const apiConfig = ConfigSchema.parse({ version: 1, review: { reviewer: { enabled: true, reviewers: ['api'], api: { url: 'https://example.test/v1', model: 'qwen3-coder', key_env: 'TEST_JUDGE_KEY' }, reasoning: { api: 'low' } } } });
|
|
272
|
+
const without = await runReviewer(repo, 'main', apiConfig, fakes(() => '', seen), () => undefined, { fetch: fetchImpl });
|
|
273
|
+
expect(without).toMatchObject({ outcome: 'unavailable', reason: expect.stringContaining('no reviewer installed') }); // the key is not set
|
|
274
|
+
process.env.TEST_JUDGE_KEY = 'secret';
|
|
275
|
+
try {
|
|
276
|
+
const result = await runReviewer(repo, 'main', apiConfig, fakes(() => '', seen), () => undefined, { fetch: fetchImpl, force: true });
|
|
277
|
+
expect(result).toMatchObject({ outcome: 'findings', reviewers: ['api'], costUsd: 0.1, items: [expect.objectContaining({ issue: 'returns before the lock', reviewer: 'api' })] });
|
|
278
|
+
expect(calls[0].reasoning_effort).toBe('low');
|
|
279
|
+
expect(calls[0].messages[1].content).toContain('full.diff'); // the same prompt a CLI judge gets
|
|
280
|
+
expect(result.record?.judges).toEqual([{ reviewer: 'api', version: 'qwen3-coder', cost_usd: 0.1, turns: 2 }]);
|
|
281
|
+
}
|
|
282
|
+
finally {
|
|
283
|
+
delete process.env.TEST_JUDGE_KEY;
|
|
284
|
+
}
|
|
285
|
+
});
|
|
259
286
|
it('asks a judge once more after an answer that is not a verdict, and is unavailable only when the second is not one either', async () => {
|
|
260
287
|
const seen = seenNow();
|
|
261
288
|
let calls = 0;
|
|
@@ -393,7 +420,7 @@ describe('verdicts', () => {
|
|
|
393
420
|
return { verdict, ...account(verdict, undefined, verify) };
|
|
394
421
|
};
|
|
395
422
|
const broken = judged([{ id: 'r1', status: 'broken', file: 'src/job.ts', line: 2, quote: 'return 1;', evidence: 'no lock before the read' }]);
|
|
396
|
-
expect(broken.open.map(i => [i.kind, i.class, i.issue])).toEqual([['rule', 'repo-rule', 'breaks a rule this repository wrote for itself (AGENTS.md):
|
|
423
|
+
expect(broken.open.map(i => [i.kind, i.class, i.issue, i.evidence])).toEqual([['rule', 'repo-rule', 'Every job must take the lock before its first read.', 'breaks a rule this repository wrote for itself (AGENTS.md): no lock before the read']]);
|
|
397
424
|
expect(judged([{ id: 'r1', status: 'broken', file: 'src/job.ts', line: 2 }])).toMatchObject({ open: [], unverified: [expect.objectContaining({ kind: 'rule' })] }); // no quote: not shown as a block
|
|
398
425
|
expect(judged([{ id: 'r2', status: 'broken', file: 'src/job.ts', line: 2, quote: 'return 1;' }])).toMatchObject({ open: [], advisory: [expect.objectContaining({ class: 'repo-rule' })] }); // guidance: shown, never a block
|
|
399
426
|
expect(judged([{ id: 'r1', status: 'followed' }, { id: 'r1', status: 'not-applicable' }])).toMatchObject({ open: [], notes: [], advisory: [], unverified: [] });
|
|
@@ -410,9 +437,14 @@ describe('verdicts', () => {
|
|
|
410
437
|
const { open } = account(verdict, undefined, checkoutVerifier(repo));
|
|
411
438
|
expect(open.map(i => [i.kind, i.class, i.locations ?? []])).toEqual([
|
|
412
439
|
['prior', 'prior point', []], ['prior', 'prior point', []],
|
|
413
|
-
|
|
414
|
-
['finding', '
|
|
440
|
+
// The same point in another file, and the same point said as another class on the next line: one item, every place.
|
|
441
|
+
['finding', 'production-cost', [{ file: 'a.ts', line: 1 }, { file: 'src/job.ts', line: 2 }]],
|
|
415
442
|
]);
|
|
443
|
+
// A rule break and the finding it caused, on the same lines and in like words, are one item.
|
|
444
|
+
const twice = account({ ...EMPTY, prior_points: [], findings: [{ class: 'correctness', file: 'src/job.ts', line: 2, issue: 'the raw table name is inlined instead of the JOBS_TABLE constant', input: 'any run', consequence: 'a rename misses it', quote: 'return 1;' }],
|
|
445
|
+
rules: [{ id: 'r', status: 'broken', file: 'src/job.ts', line: 2, quote: 'return 1;', rule: 'Import the JOBS_TABLE constant; do not inline the raw table name again.', source: 'AGENTS.md', requirement: true }] }, undefined, checkoutVerifier(repo));
|
|
446
|
+
expect(twice.open.map(i => i.class)).toEqual(['repo-rule']);
|
|
447
|
+
expect(twice.open[0].locations).toEqual([{ file: 'src/job.ts', line: 2 }]);
|
|
416
448
|
});
|
|
417
449
|
it('keeps reads, scans, redundancy and merge impact as notes with stable ids, and answers non-blocking points in the reply', () => {
|
|
418
450
|
const verdict = {
|
|
@@ -94,6 +94,8 @@ export declare function matchLessons(lessons: ReviewLesson[], change: ChangeShap
|
|
|
94
94
|
includeCandidates?: boolean;
|
|
95
95
|
limit?: number;
|
|
96
96
|
standards?: number;
|
|
97
|
+
perFile?: number;
|
|
98
|
+
excludePr?: number;
|
|
97
99
|
}): ReviewLesson[];
|
|
98
100
|
/** RIGOUR_REVIEW_LESSONS points at a lessons file outside the clone (CI, or a team's shared copy). */
|
|
99
101
|
export declare function lessonsPath(cwd: string): string;
|
|
@@ -179,6 +179,9 @@ export function mergeLessons(existing, incoming) {
|
|
|
179
179
|
* change; the best-evidenced few follow the file lessons.
|
|
180
180
|
*/
|
|
181
181
|
export function matchLessons(lessons, change, options = {}) {
|
|
182
|
+
// A lesson whose only evidence is the pull request under review is already in front of the judge as the reviewer's own points.
|
|
183
|
+
if (options.excludePr !== undefined)
|
|
184
|
+
lessons = lessons.filter(l => !l.evidence.length || l.evidence.some(e => e.pr !== options.excludePr));
|
|
182
185
|
const dirs = new Set(change.files.map(f => path.posix.dirname(f)));
|
|
183
186
|
const scored = lessons
|
|
184
187
|
.filter(l => !!l.file && (l.state === 'verified' || (options.includeCandidates && l.state === 'candidate')) && !NOT_CODE.test(l.file))
|
|
@@ -199,7 +202,20 @@ export function matchLessons(lessons, change, options = {}) {
|
|
|
199
202
|
.sort((a, b) => b.shared - a.shared || b.lesson.evidence.length - a.lesson.evidence.length)
|
|
200
203
|
.slice(0, options.standards ?? MAX_STANDARDS)
|
|
201
204
|
.map(s => s.lesson);
|
|
202
|
-
|
|
205
|
+
// On a large change, one file's many lessons must not crowd out another file's only one: a cap per file, then the total.
|
|
206
|
+
const perFile = options.perFile ?? Infinity;
|
|
207
|
+
const taken = [];
|
|
208
|
+
const perFileCount = new Map();
|
|
209
|
+
for (const { lesson } of scored) {
|
|
210
|
+
if (taken.length >= (options.limit ?? 5))
|
|
211
|
+
break;
|
|
212
|
+
const n = perFileCount.get(lesson.file) ?? 0;
|
|
213
|
+
if (n >= perFile)
|
|
214
|
+
continue;
|
|
215
|
+
perFileCount.set(lesson.file, n + 1);
|
|
216
|
+
taken.push(lesson);
|
|
217
|
+
}
|
|
218
|
+
return [...taken, ...standards];
|
|
203
219
|
}
|
|
204
220
|
/** RIGOUR_REVIEW_LESSONS points at a lessons file outside the clone (CI, or a team's shared copy). */
|
|
205
221
|
export function lessonsPath(cwd) {
|
|
@@ -29,6 +29,9 @@ describe('repository rules', () => {
|
|
|
29
29
|
expect(rules[1]).toMatchObject({ paths: ['src/lib/delivery.ts'], symbols: ['deliverOrder'] });
|
|
30
30
|
expect(rules[0].paths).toEqual(['migrations/']);
|
|
31
31
|
expect(rules.map(r => r.requirement)).toEqual([true, true, false]); // "never", "every"; "prefer" is guidance
|
|
32
|
+
// A section that only describes an exception, with no imperative, is guidance; one that ends in an imperative is a requirement.
|
|
33
|
+
const [exceptionOnly, withImperative] = splitRules('AGENTS.md', '- **One narrow exception:** the queue table is still literally named `study_jobs`, not renamed with the feature.\n\n- **One narrow exception:** the queue table is still literally named `study_jobs`. Import the `JOBS_TABLE` constant; do not inline the raw table name again.\n');
|
|
34
|
+
expect([exceptionOnly.requirement, withImperative.requirement]).toEqual([false, true]);
|
|
32
35
|
expect(rules[0].id).toMatch(/^[0-9a-f]{10}$/);
|
|
33
36
|
expect(splitRules('AGENTS.md', AGENTS)[0].id).toBe(rules[0].id); // stable across runs
|
|
34
37
|
});
|
|
@@ -95,6 +95,15 @@ describe('lessons', () => {
|
|
|
95
95
|
const change = { files: ['src/orders.ts'], symbols: new Set(['insert', 'insertOrder', 'orderId']) };
|
|
96
96
|
expect(matchLessons(lessons, change).map(l => l.id)).toEqual(['2', '1']); // two specific shared names outrank a same-file lesson with only generic ones
|
|
97
97
|
expect(matchLessons(lessons, change, { includeCandidates: true }).map(l => l.id)).toEqual(['2', '1', '3']);
|
|
98
|
+
// A lesson learned only from the pull request under review is its own reviews, already in front of the judge.
|
|
99
|
+
const own = { ...base, id: '7', text: 'from this very pull request', file: 'src/orders.ts', state: 'verified', evidence: [{ pr: 42, comment: 'c', author: 'r' }] };
|
|
100
|
+
const also = { ...own, id: '8', text: 'from this and another', evidence: [{ pr: 42, comment: 'c', author: 'r' }, { pr: 3, comment: 'd', author: 'r' }] };
|
|
101
|
+
expect(matchLessons([...lessons, own, also], change, { excludePr: 42 }).map(l => l.id).sort()).toEqual(['1', '2', '8']); // '7' is left out; '8' has evidence elsewhere too
|
|
102
|
+
// One file's many lessons never crowd out another file's only one.
|
|
103
|
+
const many = Array.from({ length: 6 }, (_, i) => ({ ...base, id: `m${i}`, text: `orders lesson ${i}`, file: 'src/orders.ts', state: 'verified', symbols: ['insertOrder', 'orderId'] }));
|
|
104
|
+
const lone = { ...base, id: 'lone', text: 'the only lesson about the manifest', file: 'src/manifest.sha', state: 'verified' };
|
|
105
|
+
const served = matchLessons([...many, lone], { files: ['src/orders.ts', 'src/manifest.sha'], symbols: new Set(['insertOrder', 'orderId']) }, { limit: 4, perFile: 3 });
|
|
106
|
+
expect(served.map(l => l.id)).toEqual(['m0', 'm1', 'm2', 'lone']);
|
|
98
107
|
});
|
|
99
108
|
});
|
|
100
109
|
describe('learning from one pull request as it goes', () => {
|
|
@@ -4,8 +4,8 @@ export type LessonMode = 'verified' | 'all' | 'off';
|
|
|
4
4
|
export declare const DEFAULT_LESSON_MODE: LessonMode;
|
|
5
5
|
/** The team's review lessons in play for this mode: none when off, verified ones by default. */
|
|
6
6
|
export declare function activeLessons(cwd: string, mode?: LessonMode): ReviewLesson[];
|
|
7
|
-
/** `standards`: how many team standards may come
|
|
8
|
-
export declare function lessonsForDiff(cwd: string, diff: string, mode?: LessonMode, standards?: number): ReviewLesson[];
|
|
7
|
+
/** `standards`, `limit`, `perFile`: how many team standards and file lessons may come, and how many per file (a judge reading a whole pull request takes more than an agent's one question). */
|
|
8
|
+
export declare function lessonsForDiff(cwd: string, diff: string, mode?: LessonMode, standards?: number, limit?: number, perFile?: number, excludePr?: number): ReviewLesson[];
|
|
9
9
|
/** The points this team rejected that a change touches: what the judges are told is settled. */
|
|
10
10
|
export declare function rejectedForDiff(cwd: string, diff: string): ReviewLesson[];
|
|
11
11
|
/** A lesson as a judge or agent sees it, in one place. */
|