@rigour-labs/core 6.7.7 → 6.7.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/review/reviewer/api-judge.d.ts +5 -0
- package/dist/review/reviewer/api-judge.js +34 -5
- package/dist/review/reviewer/api-judge.test.js +21 -0
- package/dist/review/reviewer/context.d.ts +2 -0
- package/dist/review/reviewer/context.js +4 -1
- package/dist/review/reviewer/verdict.js +8 -3
- package/dist/review/reviewer.js +23 -5
- package/dist/review/reviewer.test.js +29 -6
- package/dist/review-learning/lessons.d.ts +2 -0
- package/dist/review-learning/lessons.js +17 -1
- package/dist/review-learning/repo-rules.test.js +3 -0
- package/dist/review-learning/review-learning.test.js +9 -0
- package/dist/review-learning/team-lessons.d.ts +2 -2
- package/dist/review-learning/team-lessons.js +3 -3
- package/package.json +6 -6
|
@@ -12,6 +12,11 @@ export interface ApiJudgeOptions {
|
|
|
12
12
|
roots: string[];
|
|
13
13
|
reasoning?: Reasoning;
|
|
14
14
|
fetchImpl?: typeof fetch;
|
|
15
|
+
/** The review's input files, given inline in the first message: a model that never calls a tool still has what it needs. */
|
|
16
|
+
inputs?: Array<{
|
|
17
|
+
path: string;
|
|
18
|
+
text: string;
|
|
19
|
+
}>;
|
|
15
20
|
}
|
|
16
21
|
export interface ApiJudgeRun {
|
|
17
22
|
exitCode: number;
|
|
@@ -11,6 +11,10 @@ import path from 'path';
|
|
|
11
11
|
import { execa } from 'execa';
|
|
12
12
|
const SYSTEM = 'You review code with read-only tools. Read the files the task names with read_file (the review inputs are named by absolute path), search the repository with search, read history with git. When you are done, reply with the final answer the task asks for and nothing else.';
|
|
13
13
|
const MAX_RESULT_CHARS = 60_000;
|
|
14
|
+
/** Output tokens a turn may use, reasoning included: a review verdict is long, and a reasoning model's thinking counts against the default budget. */
|
|
15
|
+
const MAX_OUTPUT_TOKENS = 32_000;
|
|
16
|
+
/** Characters of the inputs given inline in the first message; the rest stays in the files, for the tools. */
|
|
17
|
+
const MAX_INLINE_CHARS = 240_000;
|
|
14
18
|
const GIT_ALLOWED = new Set(['log', 'show', 'diff', 'blame', 'grep', 'ls-files', 'rev-parse', 'merge-base']);
|
|
15
19
|
/** Git options that write, or point git at another repository. */
|
|
16
20
|
const GIT_REFUSED = /^(--output|-o$|--git-dir|--work-tree|-C$|--exec-path|-c$|--config-env)/;
|
|
@@ -24,7 +28,7 @@ const n = (v) => (typeof v === 'number' && Number.isFinite(v) ? v : 0);
|
|
|
24
28
|
export async function runApiJudge(prompt, o) {
|
|
25
29
|
const fetchImpl = o.fetchImpl ?? fetch;
|
|
26
30
|
const deadline = Date.now() + o.timeoutMs;
|
|
27
|
-
const messages = [{ role: 'system', content: SYSTEM }, { role: 'user', content: prompt }];
|
|
31
|
+
const messages = [{ role: 'system', content: SYSTEM }, { role: 'user', content: withInputs(prompt, o.inputs ?? []) }];
|
|
28
32
|
const usage = { input: 0, cacheRead: 0, cacheWrite: 0, output: 0 };
|
|
29
33
|
const calls = [];
|
|
30
34
|
let cost;
|
|
@@ -40,7 +44,7 @@ export async function runApiJudge(prompt, o) {
|
|
|
40
44
|
response = await fetchImpl(`${o.url.replace(/\/$/, '')}/chat/completions`, {
|
|
41
45
|
method: 'POST',
|
|
42
46
|
headers: { 'content-type': 'application/json', authorization: `Bearer ${o.key}` },
|
|
43
|
-
body: JSON.stringify({ model: o.model, messages, tools: TOOLS, tool_choice: 'auto', ...(o.reasoning ? { reasoning_effort: o.reasoning } : {}) }),
|
|
47
|
+
body: JSON.stringify({ model: o.model, messages, tools: TOOLS, tool_choice: 'auto', max_tokens: MAX_OUTPUT_TOKENS, ...(o.reasoning ? { reasoning_effort: o.reasoning } : {}) }),
|
|
44
48
|
signal: controller.signal,
|
|
45
49
|
});
|
|
46
50
|
}
|
|
@@ -66,13 +70,22 @@ export async function runApiJudge(prompt, o) {
|
|
|
66
70
|
usage.output += n(u.completion_tokens);
|
|
67
71
|
if (typeof u.cost === 'number')
|
|
68
72
|
cost = (cost ?? 0) + u.cost;
|
|
69
|
-
const
|
|
73
|
+
const choice = body.choices?.[0];
|
|
74
|
+
// A gateway reports a provider's failure inside the choice, with the choice's finish_reason "error": say what it said.
|
|
75
|
+
if (choice?.error || body.error)
|
|
76
|
+
return fail(`the API reported an error: ${JSON.stringify(choice?.error ?? body.error).slice(0, 300)}`);
|
|
77
|
+
const message = choice?.message;
|
|
70
78
|
if (!message)
|
|
71
79
|
return fail('no choices in the answer');
|
|
72
|
-
|
|
80
|
+
// Echo back what the API needs to continue (reasoning_details carries a reasoning model's chain), not the reasoning prose.
|
|
81
|
+
messages.push({ role: 'assistant', content: message.content ?? null, ...(message.tool_calls ? { tool_calls: message.tool_calls } : {}), ...(message.reasoning_details ? { reasoning_details: message.reasoning_details } : {}) });
|
|
73
82
|
const toolCalls = Array.isArray(message.tool_calls) ? message.tool_calls : [];
|
|
74
83
|
if (toolCalls.length === 0) {
|
|
75
|
-
const
|
|
84
|
+
const text = String(message.content ?? '').trim();
|
|
85
|
+
// An empty final message is not an answer: say why (the output budget ran out, a refusal), never report it as one.
|
|
86
|
+
if (!text)
|
|
87
|
+
return fail(`an empty answer on turn ${turn} (finish_reason ${String(body.choices?.[0]?.finish_reason ?? 'unknown')}${message.refusal ? `, refusal: ${String(message.refusal).slice(0, 120)}` : ''})`);
|
|
88
|
+
const answer = { result: text, usage, ...(cost !== undefined ? { cost_usd: cost } : {}), trace: { turns: turn, usage, calls } };
|
|
76
89
|
return { exitCode: 0, stdout: JSON.stringify(answer), stderr: '' };
|
|
77
90
|
}
|
|
78
91
|
for (const call of toolCalls) {
|
|
@@ -83,6 +96,22 @@ export async function runApiJudge(prompt, o) {
|
|
|
83
96
|
}
|
|
84
97
|
return fail(`no answer within ${o.maxTurns} turns`);
|
|
85
98
|
}
|
|
99
|
+
/**
|
|
100
|
+
* The prompt with the inputs it names appended inline, smallest first so the diff takes what budget is left: a judge
|
|
101
|
+
* is fed, not left to decide whether to read. What does not fit is cut with a note naming the file to read.
|
|
102
|
+
*/
|
|
103
|
+
function withInputs(prompt, inputs) {
|
|
104
|
+
if (inputs.length === 0)
|
|
105
|
+
return prompt;
|
|
106
|
+
const ordered = [...inputs].sort((a, b) => a.text.length - b.text.length);
|
|
107
|
+
let left = MAX_INLINE_CHARS;
|
|
108
|
+
const parts = ordered.map(input => {
|
|
109
|
+
const text = input.text.length > left ? `${input.text.slice(0, Math.max(0, left))}\n…[cut here: read ${input.path} with read_file for the rest]` : input.text;
|
|
110
|
+
left = Math.max(0, left - input.text.length);
|
|
111
|
+
return `### ${input.path}\n${text}`;
|
|
112
|
+
});
|
|
113
|
+
return `${prompt}\n\nThe inputs named above, inline (the files are also there for your tools):\n\n${parts.join('\n\n')}`;
|
|
114
|
+
}
|
|
86
115
|
/** One tool call, inside the allowed roots only; the trace names tools as the CLI judges do (Read, Grep, Glob, Bash). */
|
|
87
116
|
async function runTool(call, o) {
|
|
88
117
|
const name = String(call?.function?.name ?? '');
|
|
@@ -78,6 +78,27 @@ describe('the API judge', () => {
|
|
|
78
78
|
const slow = (async (_u, init) => new Promise((_r, reject) => init.signal.addEventListener('abort', () => reject(new Error('aborted')))));
|
|
79
79
|
expect(await runApiJudge('p', options(slow, { timeoutMs: 50 }))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('request failed') });
|
|
80
80
|
});
|
|
81
|
+
it('fails closed on an empty final message, saying why, and echoes back only what the API needs to continue', async () => {
|
|
82
|
+
const empty = (async () => new Response(JSON.stringify({ choices: [{ message: { role: 'assistant', content: '', reasoning: 'thinking…' }, finish_reason: 'length' }], usage: {} }), { status: 200 }));
|
|
83
|
+
expect(await runApiJudge('p', options(empty))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('an empty answer on turn 1 (finish_reason length)') });
|
|
84
|
+
const providerError = (async () => new Response(JSON.stringify({ choices: [{ message: { role: 'assistant', content: '' }, finish_reason: 'error', error: { message: 'Provider returned error', code: 502 } }], usage: {} }), { status: 200 }));
|
|
85
|
+
expect(await runApiJudge('p', options(providerError))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('the API reported an error: {"message":"Provider returned error","code":502}') });
|
|
86
|
+
const { fetchImpl, seen } = model([{ tools: [{ name: 'list_dir', args: { path: '.' } }] }, { text: 'ok' }]);
|
|
87
|
+
await runApiJudge('p', options(fetchImpl));
|
|
88
|
+
const echoed = seen[1].messages.find((m) => m.role === 'assistant');
|
|
89
|
+
expect(Object.keys(echoed).sort()).toEqual(['content', 'role', 'tool_calls']); // no reasoning prose sent back
|
|
90
|
+
expect(seen[0].max_tokens).toBe(32000);
|
|
91
|
+
});
|
|
92
|
+
it('gives the judge its inputs inline, smallest first, the diff cut with a note when the budget runs out', async () => {
|
|
93
|
+
const { fetchImpl, seen } = model([{ text: 'ok' }]);
|
|
94
|
+
const inputs = [{ path: '/w/full.diff', text: 'x'.repeat(300_000) }, { path: '/w/pr-description.md', text: 'the description' }, { path: '/w/previous-reviews.md', text: 'the reviews' }];
|
|
95
|
+
await runApiJudge('review this', options(fetchImpl, { inputs }));
|
|
96
|
+
const first = seen[0].messages[1].content;
|
|
97
|
+
expect(first.startsWith('review this\n\nThe inputs named above, inline')).toBe(true);
|
|
98
|
+
expect(first.indexOf('### /w/previous-reviews.md')).toBeLessThan(first.indexOf('### /w/full.diff')); // smallest first
|
|
99
|
+
expect(first).toContain('…[cut here: read /w/full.diff with read_file for the rest]');
|
|
100
|
+
expect(first.length).toBeLessThan(241_000);
|
|
101
|
+
});
|
|
81
102
|
it('passes the reasoning effort when asked', async () => {
|
|
82
103
|
const { fetchImpl, seen } = model([{ text: 'ok' }]);
|
|
83
104
|
await runApiJudge('p', options(fetchImpl, { reasoning: 'low' }));
|
|
@@ -31,6 +31,8 @@ export interface ContextInput {
|
|
|
31
31
|
router: RouterPolicy | undefined;
|
|
32
32
|
/** Which of the team's review lessons the judges see (gates.deep.review_lessons): verified by default, all, or off. */
|
|
33
33
|
lessons?: LessonMode;
|
|
34
|
+
/** The pull request under review: lessons learned only from it are its own reviews, which the judge already reads. */
|
|
35
|
+
pr?: number;
|
|
34
36
|
/** The previous verdict's panel decisions, and the files changed since it. */
|
|
35
37
|
previousPanel: PanelItem[] | undefined;
|
|
36
38
|
touched: Set<string>;
|
|
@@ -24,6 +24,9 @@ export const REVIEW_DISMISSALS = path.join('.rigour', 'dismissed-review-items.js
|
|
|
24
24
|
const MAX_DOCS = 10;
|
|
25
25
|
/** Team standards a judge is shown with the lessons about the changed files. */
|
|
26
26
|
const JUDGE_STANDARDS = 15;
|
|
27
|
+
/** File lessons a judge is shown: on a pull request touching a hundred files, enough for every file, at most this many per file. */
|
|
28
|
+
const JUDGE_FILE_LESSONS = 30;
|
|
29
|
+
const JUDGE_LESSONS_PER_FILE = 3;
|
|
27
30
|
/** Rules from the repository's own rules files a judge is asked to answer, most relevant first. */
|
|
28
31
|
const JUDGE_RULES = 15;
|
|
29
32
|
const MAX_SETTLED = 40;
|
|
@@ -78,7 +81,7 @@ export function buildContext(input) {
|
|
|
78
81
|
task = undefined;
|
|
79
82
|
}
|
|
80
83
|
// A judge reads the whole pull request: more of what the team taught fits than an agent's one question at the stop.
|
|
81
|
-
const lessons = input.lessons === 'off' ? [] : lessonsForDiff(input.cwd, input.diff, input.lessons, JUDGE_STANDARDS).map(lessonView);
|
|
84
|
+
const lessons = input.lessons === 'off' ? [] : lessonsForDiff(input.cwd, input.diff, input.lessons, JUDGE_STANDARDS, JUDGE_FILE_LESSONS, JUDGE_LESSONS_PER_FILE, input.pr).map(lessonView);
|
|
82
85
|
if (lessons.length)
|
|
83
86
|
sections.push(`## Lessons this team taught on earlier reviews, for what this change touches (context: a lesson never blocks on its own; a finding still needs its quote)\n${lessons.map(l => `- ${describeLesson(l)}`).join('\n')}`);
|
|
84
87
|
// The repository's own rules, always: the reviewer is the boundary, and what the team wrote is the standard it checks.
|
|
@@ -224,7 +224,8 @@ export function account(verdict, previousOpen, verify) {
|
|
|
224
224
|
for (const r of verdict.rules ?? []) {
|
|
225
225
|
if (r.status !== 'broken' || !r.rule)
|
|
226
226
|
continue;
|
|
227
|
-
|
|
227
|
+
// The rule's own words are the issue, so the same point found as a finding reads alike; where it came from is the evidence.
|
|
228
|
+
const item = { id: id('repo-rule', r.file, r.id), kind: 'rule', class: 'repo-rule', file: r.file, line: r.line, issue: r.rule, consequence: r.requirement ? 'the team wrote this rule as a requirement' : 'the team wrote this rule as guidance', ...(r.quote ? { quote: r.quote } : {}), evidence: `breaks a rule this repository wrote for itself (${r.source})${r.evidence ? `: ${r.evidence}` : ''}`, reviewer: r.reviewer };
|
|
228
229
|
if (r.requirement)
|
|
229
230
|
add(item);
|
|
230
231
|
else
|
|
@@ -274,8 +275,10 @@ export function account(verdict, previousOpen, verify) {
|
|
|
274
275
|
}
|
|
275
276
|
return { open: onePerRootCause(open), unverified, resolved, answerInReply, notes, advisory: onePerRootCause(advisory) };
|
|
276
277
|
}
|
|
277
|
-
/** How alike two items' words must be to be the same point made in two places. */
|
|
278
|
+
/** How alike two items' words must be to be the same point made in two places; and, on the same lines, to be one point said two ways. */
|
|
278
279
|
const SAME_POINT = 0.6;
|
|
280
|
+
const SAME_PLACE = 0.3;
|
|
281
|
+
const SAME_LINES = 3;
|
|
279
282
|
/**
|
|
280
283
|
* The same point found in several places is one item carrying every location, so a person reads one
|
|
281
284
|
* line, not one per file. Blocking is unchanged: the item blocks until every location is fixed.
|
|
@@ -283,7 +286,9 @@ const SAME_POINT = 0.6;
|
|
|
283
286
|
function onePerRootCause(items) {
|
|
284
287
|
const kept = [];
|
|
285
288
|
for (const item of items) {
|
|
286
|
-
|
|
289
|
+
// The same class in the same words anywhere, or any two non-human items on the same lines that read alike (a rule break and the finding it caused).
|
|
290
|
+
const nearby = (k) => !!k.file && k.file === item.file && k.line !== undefined && item.line !== undefined && Math.abs(k.line - item.line) <= SAME_LINES;
|
|
291
|
+
const same = item.kind === 'prior' ? undefined : kept.find(k => k.kind !== 'prior' && ((k.class === item.class && textSimilarity(k, item) >= SAME_POINT) || (nearby(k) && textSimilarity(k, item) >= SAME_PLACE)));
|
|
287
292
|
if (!same) {
|
|
288
293
|
kept.push(item);
|
|
289
294
|
continue;
|
package/dist/review/reviewer.js
CHANGED
|
@@ -141,7 +141,7 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
141
141
|
const sincePrevious = previousIsAncestor ? new Set((await git(['diff', '--name-only', `${previous.head}..HEAD`])).split('\n').filter(Boolean)) : new Set();
|
|
142
142
|
const changedFiles = [...fullDiff.matchAll(/^diff --git a\/.* b\/(.*)$/gm)].map(m => m[1]);
|
|
143
143
|
const context = buildContext({
|
|
144
|
-
cwd, stateRoot, dismissals, diff: fullDiff, router: config.gates.deep?.router, lessons: config.gates.deep?.review_lessons, touched: sincePrevious, checks: options.checks ?? [],
|
|
144
|
+
cwd, stateRoot, dismissals, diff: fullDiff, router: config.gates.deep?.router, lessons: config.gates.deep?.review_lessons, ...(pr ? { pr: pr.number } : {}), touched: sincePrevious, checks: options.checks ?? [],
|
|
145
145
|
previousPanel: previousIsAncestor ? store.readJson(previous.verdict)?.panel?.items : undefined,
|
|
146
146
|
docs: await relatedDocs(cwd, changedFiles, exec),
|
|
147
147
|
});
|
|
@@ -208,8 +208,9 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
208
208
|
return none(settings.required.panel || settings.required.mode ? 'unavailable' : 'skipped', over, { reviewers, scope, why, pr: pr?.number });
|
|
209
209
|
const work = fs.mkdtempSync(path.join(os.tmpdir(), 'rigour-reviewer-'));
|
|
210
210
|
// One judge run, by CLI or by API: the same prompt, the same cost accounting, the same trace.
|
|
211
|
+
let inlineInputs = [];
|
|
211
212
|
const runJudge = (name, prompt, model) => name === 'api'
|
|
212
|
-
? runApiJudge(prompt, { url: settings.api.url, model: settings.api.model, key: process.env[settings.api.key_env] ?? '', maxTurns: settings.api.max_turns, timeoutMs: settings.timeout_ms, cwd, roots: [cwd, work], ...(settings.reasoning[name] ? { reasoning: settings.reasoning[name] } : {}), ...(options.fetch ? { fetchImpl: options.fetch } : {}) })
|
|
213
|
+
? runApiJudge(prompt, { url: settings.api.url, model: settings.api.model, key: process.env[settings.api.key_env] ?? '', maxTurns: settings.api.max_turns, timeoutMs: settings.timeout_ms, cwd, roots: [cwd, work], inputs: inlineInputs, ...(settings.reasoning[name] ? { reasoning: settings.reasoning[name] } : {}), ...(options.fetch ? { fetchImpl: options.fetch } : {}) })
|
|
213
214
|
: exec(installed.get(name).binary, ADAPTERS[name].args(prompt, model, { reasoning: settings.reasoning[name] }), { cwd, timeoutMs: settings.timeout_ms, unset: judgeUnset(name, settings.judge_env) });
|
|
214
215
|
try {
|
|
215
216
|
const file = (name, text) => {
|
|
@@ -223,6 +224,7 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
223
224
|
const diffFile = file('full.diff', fullDiff);
|
|
224
225
|
const contextFile = file('team-knowledge.md', context.text);
|
|
225
226
|
const hintsFile = file('hints.txt', options.hints?.trim() || 'none\n');
|
|
227
|
+
inlineInputs = [[reviewsFile, reviews.markdown], [prBodyFile, body], [diffstatFile, await git(['diff', '--stat', `${baseSha}...HEAD`])], [diffFile, fullDiff], [contextFile, context.text], [hintsFile, options.hints?.trim() || 'none\n']].map(([p, text]) => ({ path: p, text }));
|
|
226
228
|
let delta = '';
|
|
227
229
|
// A reviewer must report on the human reviews, unless every point was settled by the previous verdict and is carried.
|
|
228
230
|
let needsPriorPoints = reviews.count > 0;
|
|
@@ -258,13 +260,29 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
258
260
|
return { run, answer, verdict: run.exitCode === 0 || answer.text.trim() ? parseVerdict(answer.text, needsPriorPoints, name, answer) : undefined };
|
|
259
261
|
};
|
|
260
262
|
let first = await ask();
|
|
261
|
-
//
|
|
262
|
-
if (first.verdict
|
|
263
|
-
progress(`Rigour reviewer: ${name} gave no valid verdict; asking once more`);
|
|
263
|
+
// No verdict, whether a malformed answer or a run that died, is a slip, not a decision: asked once more, inside the caps.
|
|
264
|
+
if ((!first.verdict || 'error' in first.verdict) && !overBudget(store.spend(), settings, 1)) {
|
|
265
|
+
progress(`Rigour reviewer: ${name} gave no ${first.verdict ? 'valid verdict' : 'answer'}; asking once more`);
|
|
264
266
|
first = await ask();
|
|
265
267
|
}
|
|
266
268
|
return first.verdict ?? { error: `${name}: no answer (exit ${first.run.exitCode}): ${first.run.stderr.trim().slice(-200)}` };
|
|
267
269
|
}));
|
|
270
|
+
// A judge that gives nothing is replaced by the next one installed, so the boundary stays up: a review ends unavailable only when every judge failed.
|
|
271
|
+
const spare = candidates.filter(c => installed.has(c) && !reviewers.includes(c));
|
|
272
|
+
for (let i = 0; i < answers.length; i++) {
|
|
273
|
+
let answer = answers[i];
|
|
274
|
+
while ('error' in answer && spare.length && !overBudget(store.spend(), settings, 1)) {
|
|
275
|
+
const next = spare.shift();
|
|
276
|
+
progress(`Rigour reviewer: ${reviewers[i]} gave no verdict (${answer.error}); ${next} judges instead`);
|
|
277
|
+
modeRecord = { ...modeRecord, degraded: `${modeRecord.degraded ? `${modeRecord.degraded}; ` : ''}${reviewers[i]} gave no verdict, ${next} judged instead` };
|
|
278
|
+
reviewers[i] = next;
|
|
279
|
+
const run = await runJudge(next, prompt, modelFor(next));
|
|
280
|
+
const got = ADAPTERS[next].answer(run.stdout);
|
|
281
|
+
store.addSpend(1, got.costUsd);
|
|
282
|
+
answer = run.exitCode === 0 || got.text.trim() ? parseVerdict(got.text, needsPriorPoints, next, got) : { error: `${next}: no answer (exit ${run.exitCode}): ${run.stderr.trim().slice(-200)}` };
|
|
283
|
+
}
|
|
284
|
+
answers[i] = answer;
|
|
285
|
+
}
|
|
268
286
|
const failed = answers.find(a => 'error' in a);
|
|
269
287
|
if (failed && 'error' in failed)
|
|
270
288
|
return none('unavailable', failed.error, { reviewers, scope, why, pr: pr?.number });
|
|
@@ -170,7 +170,7 @@ describe('the reviewer', () => {
|
|
|
170
170
|
expect(hidden.outcome).toBe('passed');
|
|
171
171
|
const shown = await runReviewer(repo, 'main', config, fakes(() => JSON.stringify({ ...EMPTY, prior_points: [] }), seen), () => undefined, { pr: 42, reviewsBefore: '2026-10-04', force: true });
|
|
172
172
|
expect(seen.files['previous-reviews.md']).toContain('Review by senior');
|
|
173
|
-
expect(shown).toMatchObject({ outcome: 'unavailable', reason: '
|
|
173
|
+
expect(shown).toMatchObject({ outcome: 'unavailable', reason: expect.stringContaining('did not report on the human reviews') });
|
|
174
174
|
});
|
|
175
175
|
it('for a backtest, gives the description as it read at the review, never a later edit', async () => {
|
|
176
176
|
const versions = { lastEditedAt: '2026-10-06T00:00:00Z', body: 'today: refunds are issued by the nightly job', userContentEdits: { totalCount: 3, nodes: [
|
|
@@ -200,7 +200,7 @@ describe('the reviewer', () => {
|
|
|
200
200
|
});
|
|
201
201
|
it('never passes without a verdict: a crash, a malformed answer, an unreadable pull request or no installed reviewer', async () => {
|
|
202
202
|
const crashed = await runReviewer(repo, 'main', config, fakes(() => ({ exitCode: 1, stdout: '', stderr: 'API error' }), seenNow()), () => undefined);
|
|
203
|
-
expect(crashed).toMatchObject({ outcome: 'unavailable', reason: expect.stringContaining('
|
|
203
|
+
expect(crashed).toMatchObject({ outcome: 'unavailable', reason: expect.stringContaining('cursor: no answer (exit 1)'), mode: { degraded: expect.stringContaining('claude gave no verdict, cursor judged instead') } }); // asked twice, then the spare judge, which failed too
|
|
204
204
|
const prose = await runReviewer(repo, 'main', config, fakes(() => 'Looks good to me!', seenNow()), () => undefined);
|
|
205
205
|
expect(prose).toMatchObject({ outcome: 'unavailable', reason: expect.stringContaining('no valid verdict') });
|
|
206
206
|
const broken = async (command, args, options) => command === 'gh' && args[0] === 'pr' ? { exitCode: 1, stdout: '', stderr: 'HTTP 500' } : fakes(() => '', seenNow())(command, args, options);
|
|
@@ -283,16 +283,34 @@ describe('the reviewer', () => {
|
|
|
283
283
|
delete process.env.TEST_JUDGE_KEY;
|
|
284
284
|
}
|
|
285
285
|
});
|
|
286
|
+
it('replaces a judge that gives nothing with the next one installed, and says so', async () => {
|
|
287
|
+
const seen = seenNow();
|
|
288
|
+
const silent = (async () => new Response(JSON.stringify({ choices: [{ message: { role: 'assistant', content: '' }, finish_reason: 'stop' }], usage: { prompt_tokens: 5, completion_tokens: 0 } }), { status: 200 }));
|
|
289
|
+
const twoJudges = ConfigSchema.parse({ version: 1, review: { reviewer: { enabled: true, reviewers: ['api', 'claude'], api: { url: 'https://example.test/v1', model: 'silent-model', key_env: 'TEST_JUDGE_KEY' } } } });
|
|
290
|
+
process.env.TEST_JUDGE_KEY = 'secret';
|
|
291
|
+
try {
|
|
292
|
+
const result = await runReviewer(repo, 'main', twoJudges, fakes(() => JSON.stringify(EMPTY), seen), () => undefined, { fetch: silent, force: true });
|
|
293
|
+
expect(result).toMatchObject({ outcome: 'passed', reviewers: ['claude'], mode: { degraded: expect.stringContaining('api gave no verdict, claude judged instead') } });
|
|
294
|
+
expect(seen.prompts).toHaveLength(1); // claude ran once, after the api judge's two empty answers
|
|
295
|
+
}
|
|
296
|
+
finally {
|
|
297
|
+
delete process.env.TEST_JUDGE_KEY;
|
|
298
|
+
}
|
|
299
|
+
});
|
|
286
300
|
it('asks a judge once more after an answer that is not a verdict, and is unavailable only when the second is not one either', async () => {
|
|
287
301
|
const seen = seenNow();
|
|
288
302
|
let calls = 0;
|
|
289
303
|
const slipOnce = await runReviewer(repo, 'main', config, fakes(() => (++calls === 1 ? '{"prior_points":[], "findings":[{"class"' : JSON.stringify(EMPTY)), seen), () => undefined, { force: true });
|
|
290
304
|
expect(slipOnce.outcome).toBe('passed');
|
|
291
305
|
expect(seen.prompts).toHaveLength(2);
|
|
306
|
+
let crashes = 0;
|
|
307
|
+
const crashOnce = seenNow();
|
|
308
|
+
const recovered = await runReviewer(repo, 'main', config, fakes(() => (++crashes === 1 ? { exitCode: 1, stdout: '', stderr: 'API error' } : JSON.stringify(EMPTY)), crashOnce), () => undefined, { force: true });
|
|
309
|
+
expect(recovered.outcome).toBe('passed'); // a run that died is asked once more too
|
|
292
310
|
const twice = seenNow();
|
|
293
311
|
const slipTwice = await runReviewer(repo, 'main', config, fakes(() => 'not json', twice), () => undefined, { force: true });
|
|
294
312
|
expect(slipTwice).toMatchObject({ outcome: 'unavailable', reason: expect.stringContaining('no valid verdict') });
|
|
295
|
-
expect(twice.prompts).toHaveLength(
|
|
313
|
+
expect(twice.prompts).toHaveLength(3); // once more, then the spare judge once: never a loop
|
|
296
314
|
});
|
|
297
315
|
it('says what was asked and that nothing ran when a review ends early, with why a judge is missing', async () => {
|
|
298
316
|
const seen = { ...seenNow(), installed: ['claude'] }; // cursor is listed but not installed
|
|
@@ -420,7 +438,7 @@ describe('verdicts', () => {
|
|
|
420
438
|
return { verdict, ...account(verdict, undefined, verify) };
|
|
421
439
|
};
|
|
422
440
|
const broken = judged([{ id: 'r1', status: 'broken', file: 'src/job.ts', line: 2, quote: 'return 1;', evidence: 'no lock before the read' }]);
|
|
423
|
-
expect(broken.open.map(i => [i.kind, i.class, i.issue])).toEqual([['rule', 'repo-rule', 'breaks a rule this repository wrote for itself (AGENTS.md):
|
|
441
|
+
expect(broken.open.map(i => [i.kind, i.class, i.issue, i.evidence])).toEqual([['rule', 'repo-rule', 'Every job must take the lock before its first read.', 'breaks a rule this repository wrote for itself (AGENTS.md): no lock before the read']]);
|
|
424
442
|
expect(judged([{ id: 'r1', status: 'broken', file: 'src/job.ts', line: 2 }])).toMatchObject({ open: [], unverified: [expect.objectContaining({ kind: 'rule' })] }); // no quote: not shown as a block
|
|
425
443
|
expect(judged([{ id: 'r2', status: 'broken', file: 'src/job.ts', line: 2, quote: 'return 1;' }])).toMatchObject({ open: [], advisory: [expect.objectContaining({ class: 'repo-rule' })] }); // guidance: shown, never a block
|
|
426
444
|
expect(judged([{ id: 'r1', status: 'followed' }, { id: 'r1', status: 'not-applicable' }])).toMatchObject({ open: [], notes: [], advisory: [], unverified: [] });
|
|
@@ -437,9 +455,14 @@ describe('verdicts', () => {
|
|
|
437
455
|
const { open } = account(verdict, undefined, checkoutVerifier(repo));
|
|
438
456
|
expect(open.map(i => [i.kind, i.class, i.locations ?? []])).toEqual([
|
|
439
457
|
['prior', 'prior point', []], ['prior', 'prior point', []],
|
|
440
|
-
|
|
441
|
-
['finding', '
|
|
458
|
+
// The same point in another file, and the same point said as another class on the next line: one item, every place.
|
|
459
|
+
['finding', 'production-cost', [{ file: 'a.ts', line: 1 }, { file: 'src/job.ts', line: 2 }]],
|
|
442
460
|
]);
|
|
461
|
+
// A rule break and the finding it caused, on the same lines and in like words, are one item.
|
|
462
|
+
const twice = account({ ...EMPTY, prior_points: [], findings: [{ class: 'correctness', file: 'src/job.ts', line: 2, issue: 'the raw table name is inlined instead of the JOBS_TABLE constant', input: 'any run', consequence: 'a rename misses it', quote: 'return 1;' }],
|
|
463
|
+
rules: [{ id: 'r', status: 'broken', file: 'src/job.ts', line: 2, quote: 'return 1;', rule: 'Import the JOBS_TABLE constant; do not inline the raw table name again.', source: 'AGENTS.md', requirement: true }] }, undefined, checkoutVerifier(repo));
|
|
464
|
+
expect(twice.open.map(i => i.class)).toEqual(['repo-rule']);
|
|
465
|
+
expect(twice.open[0].locations).toEqual([{ file: 'src/job.ts', line: 2 }]);
|
|
443
466
|
});
|
|
444
467
|
it('keeps reads, scans, redundancy and merge impact as notes with stable ids, and answers non-blocking points in the reply', () => {
|
|
445
468
|
const verdict = {
|
|
@@ -94,6 +94,8 @@ export declare function matchLessons(lessons: ReviewLesson[], change: ChangeShap
|
|
|
94
94
|
includeCandidates?: boolean;
|
|
95
95
|
limit?: number;
|
|
96
96
|
standards?: number;
|
|
97
|
+
perFile?: number;
|
|
98
|
+
excludePr?: number;
|
|
97
99
|
}): ReviewLesson[];
|
|
98
100
|
/** RIGOUR_REVIEW_LESSONS points at a lessons file outside the clone (CI, or a team's shared copy). */
|
|
99
101
|
export declare function lessonsPath(cwd: string): string;
|
|
@@ -179,6 +179,9 @@ export function mergeLessons(existing, incoming) {
|
|
|
179
179
|
* change; the best-evidenced few follow the file lessons.
|
|
180
180
|
*/
|
|
181
181
|
export function matchLessons(lessons, change, options = {}) {
|
|
182
|
+
// A lesson whose only evidence is the pull request under review is already in front of the judge as the reviewer's own points.
|
|
183
|
+
if (options.excludePr !== undefined)
|
|
184
|
+
lessons = lessons.filter(l => !l.evidence.length || l.evidence.some(e => e.pr !== options.excludePr));
|
|
182
185
|
const dirs = new Set(change.files.map(f => path.posix.dirname(f)));
|
|
183
186
|
const scored = lessons
|
|
184
187
|
.filter(l => !!l.file && (l.state === 'verified' || (options.includeCandidates && l.state === 'candidate')) && !NOT_CODE.test(l.file))
|
|
@@ -199,7 +202,20 @@ export function matchLessons(lessons, change, options = {}) {
|
|
|
199
202
|
.sort((a, b) => b.shared - a.shared || b.lesson.evidence.length - a.lesson.evidence.length)
|
|
200
203
|
.slice(0, options.standards ?? MAX_STANDARDS)
|
|
201
204
|
.map(s => s.lesson);
|
|
202
|
-
|
|
205
|
+
// On a large change, one file's many lessons must not crowd out another file's only one: a cap per file, then the total.
|
|
206
|
+
const perFile = options.perFile ?? Infinity;
|
|
207
|
+
const taken = [];
|
|
208
|
+
const perFileCount = new Map();
|
|
209
|
+
for (const { lesson } of scored) {
|
|
210
|
+
if (taken.length >= (options.limit ?? 5))
|
|
211
|
+
break;
|
|
212
|
+
const n = perFileCount.get(lesson.file) ?? 0;
|
|
213
|
+
if (n >= perFile)
|
|
214
|
+
continue;
|
|
215
|
+
perFileCount.set(lesson.file, n + 1);
|
|
216
|
+
taken.push(lesson);
|
|
217
|
+
}
|
|
218
|
+
return [...taken, ...standards];
|
|
203
219
|
}
|
|
204
220
|
/** RIGOUR_REVIEW_LESSONS points at a lessons file outside the clone (CI, or a team's shared copy). */
|
|
205
221
|
export function lessonsPath(cwd) {
|
|
@@ -29,6 +29,9 @@ describe('repository rules', () => {
|
|
|
29
29
|
expect(rules[1]).toMatchObject({ paths: ['src/lib/delivery.ts'], symbols: ['deliverOrder'] });
|
|
30
30
|
expect(rules[0].paths).toEqual(['migrations/']);
|
|
31
31
|
expect(rules.map(r => r.requirement)).toEqual([true, true, false]); // "never", "every"; "prefer" is guidance
|
|
32
|
+
// A section that only describes an exception, with no imperative, is guidance; one that ends in an imperative is a requirement.
|
|
33
|
+
const [exceptionOnly, withImperative] = splitRules('AGENTS.md', '- **One narrow exception:** the queue table is still literally named `study_jobs`, not renamed with the feature.\n\n- **One narrow exception:** the queue table is still literally named `study_jobs`. Import the `JOBS_TABLE` constant; do not inline the raw table name again.\n');
|
|
34
|
+
expect([exceptionOnly.requirement, withImperative.requirement]).toEqual([false, true]);
|
|
32
35
|
expect(rules[0].id).toMatch(/^[0-9a-f]{10}$/);
|
|
33
36
|
expect(splitRules('AGENTS.md', AGENTS)[0].id).toBe(rules[0].id); // stable across runs
|
|
34
37
|
});
|
|
@@ -95,6 +95,15 @@ describe('lessons', () => {
|
|
|
95
95
|
const change = { files: ['src/orders.ts'], symbols: new Set(['insert', 'insertOrder', 'orderId']) };
|
|
96
96
|
expect(matchLessons(lessons, change).map(l => l.id)).toEqual(['2', '1']); // two specific shared names outrank a same-file lesson with only generic ones
|
|
97
97
|
expect(matchLessons(lessons, change, { includeCandidates: true }).map(l => l.id)).toEqual(['2', '1', '3']);
|
|
98
|
+
// A lesson learned only from the pull request under review is its own reviews, already in front of the judge.
|
|
99
|
+
const own = { ...base, id: '7', text: 'from this very pull request', file: 'src/orders.ts', state: 'verified', evidence: [{ pr: 42, comment: 'c', author: 'r' }] };
|
|
100
|
+
const also = { ...own, id: '8', text: 'from this and another', evidence: [{ pr: 42, comment: 'c', author: 'r' }, { pr: 3, comment: 'd', author: 'r' }] };
|
|
101
|
+
expect(matchLessons([...lessons, own, also], change, { excludePr: 42 }).map(l => l.id).sort()).toEqual(['1', '2', '8']); // '7' is left out; '8' has evidence elsewhere too
|
|
102
|
+
// One file's many lessons never crowd out another file's only one.
|
|
103
|
+
const many = Array.from({ length: 6 }, (_, i) => ({ ...base, id: `m${i}`, text: `orders lesson ${i}`, file: 'src/orders.ts', state: 'verified', symbols: ['insertOrder', 'orderId'] }));
|
|
104
|
+
const lone = { ...base, id: 'lone', text: 'the only lesson about the manifest', file: 'src/manifest.sha', state: 'verified' };
|
|
105
|
+
const served = matchLessons([...many, lone], { files: ['src/orders.ts', 'src/manifest.sha'], symbols: new Set(['insertOrder', 'orderId']) }, { limit: 4, perFile: 3 });
|
|
106
|
+
expect(served.map(l => l.id)).toEqual(['m0', 'm1', 'm2', 'lone']);
|
|
98
107
|
});
|
|
99
108
|
});
|
|
100
109
|
describe('learning from one pull request as it goes', () => {
|
|
@@ -4,8 +4,8 @@ export type LessonMode = 'verified' | 'all' | 'off';
|
|
|
4
4
|
export declare const DEFAULT_LESSON_MODE: LessonMode;
|
|
5
5
|
/** The team's review lessons in play for this mode: none when off, verified ones by default. */
|
|
6
6
|
export declare function activeLessons(cwd: string, mode?: LessonMode): ReviewLesson[];
|
|
7
|
-
/** `standards`: how many team standards may come
|
|
8
|
-
export declare function lessonsForDiff(cwd: string, diff: string, mode?: LessonMode, standards?: number): ReviewLesson[];
|
|
7
|
+
/** `standards`, `limit`, `perFile`: how many team standards and file lessons may come, and how many per file (a judge reading a whole pull request takes more than an agent's one question). */
|
|
8
|
+
export declare function lessonsForDiff(cwd: string, diff: string, mode?: LessonMode, standards?: number, limit?: number, perFile?: number, excludePr?: number): ReviewLesson[];
|
|
9
9
|
/** The points this team rejected that a change touches: what the judges are told is settled. */
|
|
10
10
|
export declare function rejectedForDiff(cwd: string, diff: string): ReviewLesson[];
|
|
11
11
|
/** A lesson as a judge or agent sees it, in one place. */
|
|
@@ -13,14 +13,14 @@ export function activeLessons(cwd, mode = DEFAULT_LESSON_MODE) {
|
|
|
13
13
|
return [];
|
|
14
14
|
return readLessons(cwd).filter(l => mode === 'all' || l.state === 'verified');
|
|
15
15
|
}
|
|
16
|
-
/** `standards`: how many team standards may come
|
|
17
|
-
export function lessonsForDiff(cwd, diff, mode = DEFAULT_LESSON_MODE, standards) {
|
|
16
|
+
/** `standards`, `limit`, `perFile`: how many team standards and file lessons may come, and how many per file (a judge reading a whole pull request takes more than an agent's one question). */
|
|
17
|
+
export function lessonsForDiff(cwd, diff, mode = DEFAULT_LESSON_MODE, standards, limit, perFile, excludePr) {
|
|
18
18
|
if (mode === 'off')
|
|
19
19
|
return [];
|
|
20
20
|
const lessons = readLessons(cwd);
|
|
21
21
|
if (lessons.length === 0)
|
|
22
22
|
return [];
|
|
23
|
-
return matchLessons(lessons, changeShape(diff), { includeCandidates: mode === 'all', ...(standards !== undefined ? { standards } : {}) });
|
|
23
|
+
return matchLessons(lessons, changeShape(diff), { includeCandidates: mode === 'all', ...(standards !== undefined ? { standards } : {}), ...(limit !== undefined ? { limit } : {}), ...(perFile !== undefined ? { perFile } : {}), ...(excludePr !== undefined ? { excludePr } : {}) });
|
|
24
24
|
}
|
|
25
25
|
/** The points this team rejected that a change touches: what the judges are told is settled. */
|
|
26
26
|
export function rejectedForDiff(cwd, diff) {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@rigour-labs/core",
|
|
3
|
-
"version": "6.7.
|
|
3
|
+
"version": "6.7.9",
|
|
4
4
|
"description": "Rigour's review engine: deterministic gates on changed lines, rules and lessons learned from your team's fixes, and per-check precision from what you fix versus dismiss, across TypeScript, JavaScript, Python, Go, Ruby and C#.",
|
|
5
5
|
"engines": {
|
|
6
6
|
"node": ">=22.13"
|
|
@@ -72,11 +72,11 @@
|
|
|
72
72
|
"@anthropic-ai/sdk": "^0.30.1",
|
|
73
73
|
"pg": "^8.16.3",
|
|
74
74
|
"openai": "^5.23.2",
|
|
75
|
-
"@rigour-labs/brain-darwin-arm64": "6.7.
|
|
76
|
-
"@rigour-labs/brain-
|
|
77
|
-
"@rigour-labs/brain-linux-
|
|
78
|
-
"@rigour-labs/brain-
|
|
79
|
-
"@rigour-labs/brain-
|
|
75
|
+
"@rigour-labs/brain-darwin-arm64": "6.7.9",
|
|
76
|
+
"@rigour-labs/brain-linux-arm64": "6.7.9",
|
|
77
|
+
"@rigour-labs/brain-linux-x64": "6.7.9",
|
|
78
|
+
"@rigour-labs/brain-darwin-x64": "6.7.9",
|
|
79
|
+
"@rigour-labs/brain-win-x64": "6.7.9"
|
|
80
80
|
},
|
|
81
81
|
"devDependencies": {
|
|
82
82
|
"@types/fs-extra": "^11.0.4",
|