@rigour-labs/core 6.7.5 → 6.7.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +2 -0
- package/dist/index.js +2 -0
- package/dist/review/backtest-init.d.ts +15 -0
- package/dist/review/backtest-init.js +30 -16
- package/dist/review/backtest-judges.test.js +1 -1
- package/dist/review/backtest-last.d.ts +40 -0
- package/dist/review/backtest-last.js +116 -0
- package/dist/review/backtest-last.test.d.ts +1 -0
- package/dist/review/backtest-last.test.js +109 -0
- package/dist/review/backtest.d.ts +26 -0
- package/dist/review/backtest.js +5 -5
- package/dist/review/reviewer/adapters.d.ts +11 -4
- package/dist/review/reviewer/adapters.js +44 -5
- package/dist/review/reviewer/adapters.test.js +16 -1
- package/dist/review/reviewer/api-judge.d.ts +28 -0
- package/dist/review/reviewer/api-judge.js +161 -0
- package/dist/review/reviewer/api-judge.test.d.ts +1 -0
- package/dist/review/reviewer/api-judge.test.js +87 -0
- package/dist/review/reviewer/background.js +1 -1
- package/dist/review/reviewer/context.d.ts +3 -1
- package/dist/review/reviewer/context.js +8 -3
- package/dist/review/reviewer/context.test.js +2 -0
- package/dist/review/reviewer/prompt.js +11 -4
- package/dist/review/reviewer/record.d.ts +71 -0
- package/dist/review/reviewer/record.js +69 -0
- package/dist/review/reviewer/record.test.d.ts +1 -0
- package/dist/review/reviewer/record.test.js +32 -0
- package/dist/review/reviewer/settings.d.ts +10 -0
- package/dist/review/reviewer/settings.js +3 -1
- package/dist/review/reviewer/store.d.ts +2 -0
- package/dist/review/reviewer/store.js +4 -0
- package/dist/review/reviewer/usage.test.js +1 -1
- package/dist/review/reviewer/verdict.d.ts +34 -1
- package/dist/review/reviewer/verdict.js +58 -6
- package/dist/review/reviewer.d.ts +15 -0
- package/dist/review/reviewer.js +43 -12
- package/dist/review/reviewer.test.js +89 -3
- package/dist/review-learning/lessons.d.ts +2 -0
- package/dist/review-learning/lessons.js +3 -3
- package/dist/review-learning/repo-rules.d.ts +6 -4
- package/dist/review-learning/repo-rules.js +32 -9
- package/dist/review-learning/repo-rules.test.js +13 -0
- package/dist/templates/universal-config.js +1 -0
- package/dist/types/index.d.ts +74 -0
- package/dist/types/index.js +15 -1
- package/package.json +6 -6
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { describe, expect, it } from 'vitest';
|
|
2
|
-
import { ADAPTERS } from './adapters.js';
|
|
2
|
+
import { ADAPTERS, apiVendor, selectReviewers } from './adapters.js';
|
|
3
3
|
/** A real `codex exec --json` run (codex-cli 0.160.1), the warning's path shortened: the shape the adapter must read. */
|
|
4
4
|
const CODEX = [
|
|
5
5
|
'{"type":"thread.started","thread_id":"01a11477-f9fd-7472-8d31-5863de4cbbdb"}',
|
|
@@ -43,3 +43,18 @@ describe('reading an agent CLI\'s answer', () => {
|
|
|
43
43
|
expect(ADAPTERS.claude.args('p', undefined)).toEqual(expect.arrayContaining(['--output-format', 'stream-json', '--verbose']));
|
|
44
44
|
});
|
|
45
45
|
});
|
|
46
|
+
describe('a judge reached through an API', () => {
|
|
47
|
+
it('is told apart by its model\'s maker, so cross and full modes pair it with a different vendor', () => {
|
|
48
|
+
expect(apiVendor({ model: 'anthropic/claude-sonnet-4.5' })).toBe('anthropic');
|
|
49
|
+
expect(apiVendor({ model: 'gpt-5' })).toBe('openai');
|
|
50
|
+
expect(apiVendor({ model: 'google/gemini-2.5-pro' })).toBe('google');
|
|
51
|
+
expect(apiVendor({ model: 'qwen3-coder' })).toBe('other');
|
|
52
|
+
expect(apiVendor({ model: 'qwen3-coder', vendor: 'openai' })).toBe('openai');
|
|
53
|
+
const vendorOf = (name) => (name === 'api' ? 'openai' : ADAPTERS[name].vendor);
|
|
54
|
+
expect(selectReviewers(['claude', 'api'], 'cross', new Set(['anthropic']), new Set(['claude', 'api']), 2, vendorOf)).toEqual(['api']); // the author's vendor is skipped
|
|
55
|
+
expect(selectReviewers(['claude', 'api'], 'full', new Set(), new Set(['claude', 'api']), 2, vendorOf)).toEqual(['claude', 'api']);
|
|
56
|
+
expect(ADAPTERS.api.answer(JSON.stringify({ result: '{"prior_points":[]}', usage: { input: 10, cacheRead: 5, cacheWrite: 0, output: 3 }, cost_usd: 0.02, trace: { turns: 2, usage: {}, calls: [] } }))).toMatchObject({ text: '{"prior_points":[]}', costUsd: 0.02, tokens: { input: 15, output: 3 }, trace: { turns: 2 } });
|
|
57
|
+
expect(ADAPTERS.codex.args('p', undefined, { reasoning: 'medium' })).toContain('model_reasoning_effort=medium');
|
|
58
|
+
expect(ADAPTERS.codex.args('p', undefined)).toContain('model_reasoning_effort=high');
|
|
59
|
+
});
|
|
60
|
+
});
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
import type { RunTrace } from './adapters.js';
|
|
2
|
+
export type Reasoning = 'low' | 'medium' | 'high';
|
|
3
|
+
export interface ApiJudgeOptions {
|
|
4
|
+
/** The API's base URL; `/chat/completions` is appended. */
|
|
5
|
+
url: string;
|
|
6
|
+
model: string;
|
|
7
|
+
key: string;
|
|
8
|
+
maxTurns: number;
|
|
9
|
+
timeoutMs: number;
|
|
10
|
+
cwd: string;
|
|
11
|
+
/** Where the tools may read: the checkout and the review's input folder. */
|
|
12
|
+
roots: string[];
|
|
13
|
+
reasoning?: Reasoning;
|
|
14
|
+
fetchImpl?: typeof fetch;
|
|
15
|
+
}
|
|
16
|
+
export interface ApiJudgeRun {
|
|
17
|
+
exitCode: number;
|
|
18
|
+
stdout: string;
|
|
19
|
+
stderr: string;
|
|
20
|
+
}
|
|
21
|
+
/** What a run's stdout carries on success, for the adapter to read. */
|
|
22
|
+
export interface ApiJudgeAnswer {
|
|
23
|
+
result: string;
|
|
24
|
+
usage: RunTrace['usage'];
|
|
25
|
+
cost_usd?: number;
|
|
26
|
+
trace: RunTrace;
|
|
27
|
+
}
|
|
28
|
+
export declare function runApiJudge(prompt: string, o: ApiJudgeOptions): Promise<ApiJudgeRun>;
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A judge reached through a model API instead of an agent CLI, so any model a team can call works as
|
|
3
|
+
* a reviewer: OpenAI-compatible chat completions with tools (OpenAI, OpenRouter, a local server, other
|
|
4
|
+
* vendors through a gateway). Rigour runs the loop itself: the model asks for a read-only tool, Rigour
|
|
5
|
+
* runs it inside the checkout (and the review's own input folder) and hands the result back, until
|
|
6
|
+
* the model answers. The same prompt, the same evidence contract and the same trace as a CLI judge;
|
|
7
|
+
* the key comes from an environment variable named in rigour.yml, never from the file.
|
|
8
|
+
*/
|
|
9
|
+
import fs from 'fs';
|
|
10
|
+
import path from 'path';
|
|
11
|
+
import { execa } from 'execa';
|
|
12
|
+
const SYSTEM = 'You review code with read-only tools. Read the files the task names with read_file (the review inputs are named by absolute path), search the repository with search, read history with git. When you are done, reply with the final answer the task asks for and nothing else.';
|
|
13
|
+
const MAX_RESULT_CHARS = 60_000;
|
|
14
|
+
const GIT_ALLOWED = new Set(['log', 'show', 'diff', 'blame', 'grep', 'ls-files', 'rev-parse', 'merge-base']);
|
|
15
|
+
/** Git options that write, or point git at another repository. */
|
|
16
|
+
const GIT_REFUSED = /^(--output|-o$|--git-dir|--work-tree|-C$|--exec-path|-c$|--config-env)/;
|
|
17
|
+
const TOOLS = [
|
|
18
|
+
{ type: 'function', function: { name: 'read_file', description: 'Read a file in the repository or the review inputs, or a line range of it.', parameters: { type: 'object', properties: { path: { type: 'string' }, start_line: { type: 'integer' }, end_line: { type: 'integer' } }, required: ['path'] } } },
|
|
19
|
+
{ type: 'function', function: { name: 'search', description: 'Search the tracked files for a regular expression (git grep -n), optionally under one path.', parameters: { type: 'object', properties: { pattern: { type: 'string' }, path: { type: 'string' } }, required: ['pattern'] } } },
|
|
20
|
+
{ type: 'function', function: { name: 'list_dir', description: 'List a directory in the repository or the review inputs.', parameters: { type: 'object', properties: { path: { type: 'string' } }, required: ['path'] } } },
|
|
21
|
+
{ type: 'function', function: { name: 'git', description: 'Run a read-only git command in the repository: log, show, diff, blame, grep, ls-files, rev-parse, merge-base.', parameters: { type: 'object', properties: { args: { type: 'array', items: { type: 'string' } } }, required: ['args'] } } },
|
|
22
|
+
];
|
|
23
|
+
const n = (v) => (typeof v === 'number' && Number.isFinite(v) ? v : 0);
|
|
24
|
+
export async function runApiJudge(prompt, o) {
|
|
25
|
+
const fetchImpl = o.fetchImpl ?? fetch;
|
|
26
|
+
const deadline = Date.now() + o.timeoutMs;
|
|
27
|
+
const messages = [{ role: 'system', content: SYSTEM }, { role: 'user', content: prompt }];
|
|
28
|
+
const usage = { input: 0, cacheRead: 0, cacheWrite: 0, output: 0 };
|
|
29
|
+
const calls = [];
|
|
30
|
+
let cost;
|
|
31
|
+
const fail = (why) => ({ exitCode: 1, stdout: '', stderr: `api judge (${o.model}): ${why}` });
|
|
32
|
+
for (let turn = 1; turn <= o.maxTurns; turn++) {
|
|
33
|
+
const left = deadline - Date.now();
|
|
34
|
+
if (left <= 0)
|
|
35
|
+
return fail(`timed out after ${o.timeoutMs} ms`);
|
|
36
|
+
const controller = new AbortController();
|
|
37
|
+
const timer = setTimeout(() => controller.abort(), left);
|
|
38
|
+
let response;
|
|
39
|
+
try {
|
|
40
|
+
response = await fetchImpl(`${o.url.replace(/\/$/, '')}/chat/completions`, {
|
|
41
|
+
method: 'POST',
|
|
42
|
+
headers: { 'content-type': 'application/json', authorization: `Bearer ${o.key}` },
|
|
43
|
+
body: JSON.stringify({ model: o.model, messages, tools: TOOLS, tool_choice: 'auto', ...(o.reasoning ? { reasoning_effort: o.reasoning } : {}) }),
|
|
44
|
+
signal: controller.signal,
|
|
45
|
+
});
|
|
46
|
+
}
|
|
47
|
+
catch (error) {
|
|
48
|
+
return fail(`request failed: ${error instanceof Error ? error.message : String(error)}`);
|
|
49
|
+
}
|
|
50
|
+
finally {
|
|
51
|
+
clearTimeout(timer);
|
|
52
|
+
}
|
|
53
|
+
if (!response.ok)
|
|
54
|
+
return fail(`HTTP ${response.status}: ${(await response.text()).slice(0, 300)}`);
|
|
55
|
+
let body;
|
|
56
|
+
try {
|
|
57
|
+
body = await response.json();
|
|
58
|
+
}
|
|
59
|
+
catch {
|
|
60
|
+
return fail('the answer was not JSON');
|
|
61
|
+
}
|
|
62
|
+
const u = body.usage ?? {};
|
|
63
|
+
const cached = n(u.prompt_tokens_details?.cached_tokens);
|
|
64
|
+
usage.input += Math.max(0, n(u.prompt_tokens) - cached);
|
|
65
|
+
usage.cacheRead += cached;
|
|
66
|
+
usage.output += n(u.completion_tokens);
|
|
67
|
+
if (typeof u.cost === 'number')
|
|
68
|
+
cost = (cost ?? 0) + u.cost;
|
|
69
|
+
const message = body.choices?.[0]?.message;
|
|
70
|
+
if (!message)
|
|
71
|
+
return fail('no choices in the answer');
|
|
72
|
+
messages.push(message);
|
|
73
|
+
const toolCalls = Array.isArray(message.tool_calls) ? message.tool_calls : [];
|
|
74
|
+
if (toolCalls.length === 0) {
|
|
75
|
+
const answer = { result: String(message.content ?? ''), usage, ...(cost !== undefined ? { cost_usd: cost } : {}), trace: { turns: turn, usage, calls } };
|
|
76
|
+
return { exitCode: 0, stdout: JSON.stringify(answer), stderr: '' };
|
|
77
|
+
}
|
|
78
|
+
for (const call of toolCalls) {
|
|
79
|
+
const ran = await runTool(call, o);
|
|
80
|
+
calls.push({ turn, tool: ran.tool, target: ran.target, resultChars: ran.result.length });
|
|
81
|
+
messages.push({ role: 'tool', tool_call_id: call.id, content: ran.result });
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
return fail(`no answer within ${o.maxTurns} turns`);
|
|
85
|
+
}
|
|
86
|
+
/** One tool call, inside the allowed roots only; the trace names tools as the CLI judges do (Read, Grep, Glob, Bash). */
|
|
87
|
+
async function runTool(call, o) {
|
|
88
|
+
const name = String(call?.function?.name ?? '');
|
|
89
|
+
let args = {};
|
|
90
|
+
try {
|
|
91
|
+
args = JSON.parse(call?.function?.arguments || '{}');
|
|
92
|
+
}
|
|
93
|
+
catch {
|
|
94
|
+
return { tool: name, target: '', result: 'refused: the arguments were not JSON' };
|
|
95
|
+
}
|
|
96
|
+
const clip = (text) => (text.length > MAX_RESULT_CHARS ? `${text.slice(0, MAX_RESULT_CHARS)}\n…[truncated at ${MAX_RESULT_CHARS} characters]` : text);
|
|
97
|
+
if (name === 'read_file') {
|
|
98
|
+
const target = inside(String(args.path ?? ''), o);
|
|
99
|
+
if (!target)
|
|
100
|
+
return { tool: 'Read', target: String(args.path ?? ''), result: 'refused: outside the repository and the review inputs' };
|
|
101
|
+
try {
|
|
102
|
+
const lines = fs.readFileSync(target, 'utf8').split('\n');
|
|
103
|
+
const start = Math.max(1, Number(args.start_line) || 1);
|
|
104
|
+
const end = Math.min(lines.length, Number(args.end_line) || lines.length);
|
|
105
|
+
return { tool: 'Read', target, result: clip(lines.slice(start - 1, end).map((l, i) => `${start + i}: ${l}`).join('\n')) };
|
|
106
|
+
}
|
|
107
|
+
catch (error) {
|
|
108
|
+
return { tool: 'Read', target, result: `error: ${error instanceof Error ? error.message : String(error)}` };
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
if (name === 'list_dir') {
|
|
112
|
+
const target = inside(String(args.path ?? '.'), o);
|
|
113
|
+
if (!target)
|
|
114
|
+
return { tool: 'Glob', target: String(args.path ?? ''), result: 'refused: outside the repository and the review inputs' };
|
|
115
|
+
try {
|
|
116
|
+
return { tool: 'Glob', target, result: clip(fs.readdirSync(target, { withFileTypes: true }).map(e => (e.isDirectory() ? `${e.name}/` : e.name)).join('\n')) };
|
|
117
|
+
}
|
|
118
|
+
catch (error) {
|
|
119
|
+
return { tool: 'Glob', target, result: `error: ${error instanceof Error ? error.message : String(error)}` };
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
if (name === 'search') {
|
|
123
|
+
const pattern = String(args.pattern ?? '');
|
|
124
|
+
const under = args.path ? inside(String(args.path), o) : undefined;
|
|
125
|
+
if (args.path && !under)
|
|
126
|
+
return { tool: 'Grep', target: pattern, result: 'refused: outside the repository' };
|
|
127
|
+
const run = await execa('git', ['grep', '-n', '-I', '-e', pattern, ...(under ? ['--', path.relative(o.cwd, under) || '.'] : [])], { cwd: o.cwd, reject: false, maxBuffer: 16 * 1024 * 1024 });
|
|
128
|
+
return { tool: 'Grep', target: pattern, result: clip(run.stdout || (run.exitCode === 1 ? '(no match)' : run.stderr)) };
|
|
129
|
+
}
|
|
130
|
+
if (name === 'git') {
|
|
131
|
+
const gitArgs = Array.isArray(args.args) ? args.args.map(String) : [];
|
|
132
|
+
if (!GIT_ALLOWED.has(gitArgs[0] ?? '') || gitArgs.some(a => GIT_REFUSED.test(a)))
|
|
133
|
+
return { tool: 'Bash', target: `git ${gitArgs.join(' ')}`, result: 'refused: only read-only git commands, without options that write or point elsewhere' };
|
|
134
|
+
const run = await execa('git', ['--no-pager', ...gitArgs], { cwd: o.cwd, reject: false, maxBuffer: 16 * 1024 * 1024 });
|
|
135
|
+
return { tool: 'Bash', target: `git ${gitArgs.join(' ')}`, result: clip(run.stdout || run.stderr) };
|
|
136
|
+
}
|
|
137
|
+
return { tool: name, target: '', result: `refused: no tool named ${name}` };
|
|
138
|
+
}
|
|
139
|
+
/** The real path of `p` when it lies under one of the roots; undefined otherwise (a symlink out is outside). */
|
|
140
|
+
function inside(p, o) {
|
|
141
|
+
const resolved = path.resolve(o.cwd, p);
|
|
142
|
+
let real;
|
|
143
|
+
try {
|
|
144
|
+
real = fs.realpathSync(resolved);
|
|
145
|
+
}
|
|
146
|
+
catch {
|
|
147
|
+
return undefined;
|
|
148
|
+
}
|
|
149
|
+
for (const root of o.roots) {
|
|
150
|
+
let base;
|
|
151
|
+
try {
|
|
152
|
+
base = fs.realpathSync(root);
|
|
153
|
+
}
|
|
154
|
+
catch {
|
|
155
|
+
continue;
|
|
156
|
+
}
|
|
157
|
+
if (real === base || real.startsWith(base + path.sep))
|
|
158
|
+
return real;
|
|
159
|
+
}
|
|
160
|
+
return undefined;
|
|
161
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
import { execFileSync } from 'child_process';
|
|
2
|
+
import fs from 'fs';
|
|
3
|
+
import os from 'os';
|
|
4
|
+
import path from 'path';
|
|
5
|
+
import { afterEach, beforeEach, describe, expect, it } from 'vitest';
|
|
6
|
+
import { runApiJudge } from './api-judge.js';
|
|
7
|
+
let repo;
|
|
8
|
+
let work;
|
|
9
|
+
beforeEach(() => {
|
|
10
|
+
repo = fs.mkdtempSync(path.join(os.tmpdir(), 'api-judge-'));
|
|
11
|
+
work = fs.mkdtempSync(path.join(os.tmpdir(), 'api-judge-inputs-'));
|
|
12
|
+
execFileSync('git', ['-C', repo, 'init', '-q']);
|
|
13
|
+
fs.mkdirSync(path.join(repo, 'src'));
|
|
14
|
+
fs.writeFileSync(path.join(repo, 'src/job.ts'), 'export function job() {\n return 1;\n}\n');
|
|
15
|
+
execFileSync('git', ['-C', repo, 'add', '-A']);
|
|
16
|
+
execFileSync('git', ['-C', repo, '-c', 'user.email=t@x', '-c', 'user.name=t', 'commit', '-qm', 'init']);
|
|
17
|
+
fs.writeFileSync(path.join(work, 'full.diff'), '+++ b/src/job.ts\n');
|
|
18
|
+
});
|
|
19
|
+
afterEach(() => { for (const d of [repo, work])
|
|
20
|
+
fs.rmSync(d, { recursive: true, force: true }); });
|
|
21
|
+
/** A model that plays a scripted conversation: each entry is what it answers to the next request. */
|
|
22
|
+
function model(turns) {
|
|
23
|
+
const seen = [];
|
|
24
|
+
let i = 0;
|
|
25
|
+
const fetchImpl = (async (_url, init) => {
|
|
26
|
+
const body = JSON.parse(init.body);
|
|
27
|
+
seen.push(body);
|
|
28
|
+
const turn = turns[i++] ?? { text: 'done' };
|
|
29
|
+
if (turn.status)
|
|
30
|
+
return new Response('nope', { status: turn.status });
|
|
31
|
+
const message = turn.tools
|
|
32
|
+
? { role: 'assistant', content: null, tool_calls: turn.tools.map((t, k) => ({ id: `c${i}-${k}`, type: 'function', function: { name: t.name, arguments: JSON.stringify(t.args) } })) }
|
|
33
|
+
: { role: 'assistant', content: turn.text };
|
|
34
|
+
return new Response(JSON.stringify({ choices: [{ message }], usage: turn.usage ?? { prompt_tokens: 100, completion_tokens: 10, prompt_tokens_details: { cached_tokens: 60 } } }), { status: 200, headers: { 'content-type': 'application/json' } });
|
|
35
|
+
});
|
|
36
|
+
return { fetchImpl, seen };
|
|
37
|
+
}
|
|
38
|
+
const options = (fetchImpl, extra = {}) => ({ url: 'https://example.test/v1', model: 'some-model', key: 'k', maxTurns: 10, timeoutMs: 30_000, cwd: repo, roots: [repo, work], fetchImpl, ...extra });
|
|
39
|
+
describe('the API judge', () => {
|
|
40
|
+
it('runs the tools the model asks for inside the checkout and the inputs, then returns its answer with the usage and the trace', async () => {
|
|
41
|
+
const { fetchImpl, seen } = model([
|
|
42
|
+
{ tools: [{ name: 'read_file', args: { path: path.join(work, 'full.diff') } }, { name: 'read_file', args: { path: 'src/job.ts', start_line: 2, end_line: 2 } }] },
|
|
43
|
+
{ tools: [{ name: 'search', args: { pattern: 'return' } }, { name: 'git', args: { args: ['log', '-1', '--format=%s'] } }, { name: 'list_dir', args: { path: 'src' } }] },
|
|
44
|
+
{ text: '{"prior_points":[]}', usage: { prompt_tokens: 50, completion_tokens: 5, cost: 0.01 } },
|
|
45
|
+
]);
|
|
46
|
+
const run = await runApiJudge('review this', options(fetchImpl));
|
|
47
|
+
expect(run.exitCode).toBe(0);
|
|
48
|
+
const answer = JSON.parse(run.stdout);
|
|
49
|
+
expect(answer.result).toBe('{"prior_points":[]}');
|
|
50
|
+
expect(answer.usage).toEqual({ input: 130, cacheRead: 120, cacheWrite: 0, output: 25 });
|
|
51
|
+
expect(answer.cost_usd).toBe(0.01);
|
|
52
|
+
expect(answer.trace.turns).toBe(3);
|
|
53
|
+
expect(answer.trace.calls.map(c => c.tool)).toEqual(['Read', 'Read', 'Grep', 'Bash', 'Glob']);
|
|
54
|
+
const toolResults = (request) => request.messages.filter((m) => m.role === 'tool').map((m) => m.content);
|
|
55
|
+
expect(toolResults(seen[1])).toEqual(['1: +++ b/src/job.ts\n2: ', '2: return 1;']); // turn 1's reads, as the model got them
|
|
56
|
+
expect(toolResults(seen[2]).slice(2)).toEqual(['src/job.ts:2: return 1;', 'init', 'job.ts']); // turn 2's search, git and listing
|
|
57
|
+
expect(seen[0].messages[0].role).toBe('system');
|
|
58
|
+
expect(seen[0].tools.map((t) => t.function.name)).toEqual(['read_file', 'search', 'list_dir', 'git']);
|
|
59
|
+
});
|
|
60
|
+
it('refuses to read outside the roots, a symlink out included, and refuses git that writes or points elsewhere', async () => {
|
|
61
|
+
fs.symlinkSync(os.homedir(), path.join(repo, 'out'));
|
|
62
|
+
const { fetchImpl, seen } = model([
|
|
63
|
+
{ tools: [{ name: 'read_file', args: { path: '/etc/hosts' } }, { name: 'read_file', args: { path: 'out/.bashrc' } }, { name: 'git', args: { args: ['log', '--output=/tmp/x'] } }, { name: 'git', args: { args: ['push'] } }, { name: 'nope', args: {} }] },
|
|
64
|
+
{ text: 'ok' },
|
|
65
|
+
]);
|
|
66
|
+
await runApiJudge('p', options(fetchImpl));
|
|
67
|
+
const results = seen[1].messages.filter((m) => m.role === 'tool').map((m) => m.content);
|
|
68
|
+
expect(results).toEqual([
|
|
69
|
+
'refused: outside the repository and the review inputs', 'refused: outside the repository and the review inputs',
|
|
70
|
+
'refused: only read-only git commands, without options that write or point elsewhere', 'refused: only read-only git commands, without options that write or point elsewhere',
|
|
71
|
+
'refused: no tool named nope',
|
|
72
|
+
]);
|
|
73
|
+
});
|
|
74
|
+
it('fails closed on a turn cap, an HTTP error and a timeout, saying why', async () => {
|
|
75
|
+
const loop = model(Array.from({ length: 5 }, () => ({ tools: [{ name: 'list_dir', args: { path: '.' } }] })));
|
|
76
|
+
expect(await runApiJudge('p', options(loop.fetchImpl, { maxTurns: 3 }))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('no answer within 3 turns') });
|
|
77
|
+
expect(await runApiJudge('p', options(model([{ status: 429 }]).fetchImpl))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('HTTP 429') });
|
|
78
|
+
const slow = (async (_u, init) => new Promise((_r, reject) => init.signal.addEventListener('abort', () => reject(new Error('aborted')))));
|
|
79
|
+
expect(await runApiJudge('p', options(slow, { timeoutMs: 50 }))).toMatchObject({ exitCode: 1, stderr: expect.stringContaining('request failed') });
|
|
80
|
+
});
|
|
81
|
+
it('passes the reasoning effort when asked', async () => {
|
|
82
|
+
const { fetchImpl, seen } = model([{ text: 'ok' }]);
|
|
83
|
+
await runApiJudge('p', options(fetchImpl, { reasoning: 'low' }));
|
|
84
|
+
expect(seen[0].reasoning_effort).toBe('low');
|
|
85
|
+
expect(seen[0].model).toBe('some-model');
|
|
86
|
+
});
|
|
87
|
+
});
|
|
@@ -44,7 +44,7 @@ export async function backgroundReview(cwd, job, config, exec = defaultExec, log
|
|
|
44
44
|
store.recordAttempt(job.branch, { head: job.head, outcome: 'skipped', reason: skip, at: new Date().toISOString() });
|
|
45
45
|
log(`review of ${job.head.slice(0, 9)}: skipped (${skip})`);
|
|
46
46
|
clearPid(store, job.branch);
|
|
47
|
-
return { outcome: 'skipped', items: [], unverified: [], resolved: [], answerInReply: [], notes: [], disputed: [], dropped: [], dismissed: [], reason: skip, reviewers: [], cached: false };
|
|
47
|
+
return { outcome: 'skipped', items: [], unverified: [], resolved: [], answerInReply: [], notes: [], advisory: [], disputed: [], dropped: [], dismissed: [], reason: skip, reviewers: [], cached: false };
|
|
48
48
|
}
|
|
49
49
|
const worktree = store.worktreeDir(job.head);
|
|
50
50
|
const added = await exec('git', ['worktree', 'add', '--detach', worktree, job.head], { cwd, timeoutMs: 5 * GH_TIMEOUT_MS });
|
|
@@ -2,7 +2,7 @@ import { type LessonMode } from '../../review-learning/team-lessons.js';
|
|
|
2
2
|
import type { RouterPolicy } from '../../deep/router.js';
|
|
3
3
|
import type { PanelItem } from './panel.js';
|
|
4
4
|
import { type Exec } from './exec.js';
|
|
5
|
-
import type { OpenItem } from './verdict.js';
|
|
5
|
+
import type { OpenItem, ServedRule } from './verdict.js';
|
|
6
6
|
export declare const REVIEW_DISMISSALS: string;
|
|
7
7
|
export interface ReviewDismissal {
|
|
8
8
|
id: string;
|
|
@@ -59,6 +59,8 @@ export declare function buildContext(input: ContextInput): {
|
|
|
59
59
|
text: string;
|
|
60
60
|
key: string;
|
|
61
61
|
risky: number | undefined;
|
|
62
|
+
rules: ServedRule[];
|
|
63
|
+
lessons: number;
|
|
62
64
|
};
|
|
63
65
|
/**
|
|
64
66
|
* Tracked Markdown docs that name a changed file by path or by a distinctive file stem: one
|
|
@@ -15,6 +15,7 @@ import path from 'path';
|
|
|
15
15
|
import { createHash } from 'crypto';
|
|
16
16
|
import { buildReviewTask } from '../review-task.js';
|
|
17
17
|
import { describeLesson, lessonsForDiff, lessonView, rejectedForDiff } from '../../review-learning/team-lessons.js';
|
|
18
|
+
import { rulesForDiff } from '../../review-learning/repo-rules.js';
|
|
18
19
|
import { reviewedKeys } from '../ledger.js';
|
|
19
20
|
import { textSimilarity } from './consensus.js';
|
|
20
21
|
import { defaultExec, GH_TIMEOUT_MS } from './exec.js';
|
|
@@ -23,6 +24,8 @@ export const REVIEW_DISMISSALS = path.join('.rigour', 'dismissed-review-items.js
|
|
|
23
24
|
const MAX_DOCS = 10;
|
|
24
25
|
/** Team standards a judge is shown with the lessons about the changed files. */
|
|
25
26
|
const JUDGE_STANDARDS = 15;
|
|
27
|
+
/** Rules from the repository's own rules files a judge is asked to answer, most relevant first. */
|
|
28
|
+
const JUDGE_RULES = 15;
|
|
26
29
|
const MAX_SETTLED = 40;
|
|
27
30
|
export function readReviewDismissals(cwd) {
|
|
28
31
|
try {
|
|
@@ -78,8 +81,10 @@ export function buildContext(input) {
|
|
|
78
81
|
const lessons = input.lessons === 'off' ? [] : lessonsForDiff(input.cwd, input.diff, input.lessons, JUDGE_STANDARDS).map(lessonView);
|
|
79
82
|
if (lessons.length)
|
|
80
83
|
sections.push(`## Lessons this team taught on earlier reviews, for what this change touches (context: a lesson never blocks on its own; a finding still needs its quote)\n${lessons.map(l => `- ${describeLesson(l)}`).join('\n')}`);
|
|
81
|
-
|
|
82
|
-
|
|
84
|
+
// The repository's own rules, always: the reviewer is the boundary, and what the team wrote is the standard it checks.
|
|
85
|
+
const rules = rulesForDiff(input.cwd, input.diff, true, JUDGE_RULES).map((r) => ({ id: r.id, source: r.source, text: r.text, requirement: r.requirement }));
|
|
86
|
+
if (rules.length)
|
|
87
|
+
sections.push(`## Rules this repository wrote for itself that apply to this change (answer every one in rules, by id)\n${rules.map(r => `- [${r.id}] (${r.source}, ${r.requirement ? 'requirement' : 'guidance'}) ${r.text}`).join('\n')}`);
|
|
83
88
|
if (input.checks.length)
|
|
84
89
|
sections.push(`## Already found by Rigour's checks: they block on their own, so do not report them again\n${input.checks.slice(0, MAX_SETTLED).map(c => `- ${c}`).join('\n')}`);
|
|
85
90
|
const rejected = input.lessons === 'off' ? [] : rejectedForDiff(input.cwd, input.diff).map(l => {
|
|
@@ -95,7 +100,7 @@ export function buildContext(input) {
|
|
|
95
100
|
if (docs.length)
|
|
96
101
|
sections.push(`## Docs that describe the changed code (read one when its claim matters to a finding)\n${docs.map(d => `- ${d.doc} (names ${d.names.join(', ')})`).join('\n')}`);
|
|
97
102
|
const text = sections.length ? `# What this team already knows\n\n${sections.join('\n\n')}\n` : 'none\n';
|
|
98
|
-
return { text, key: createHash('sha256').update(text).digest('hex').slice(0, 16), risky: task ? task.items.length + task.alreadyReviewed : undefined };
|
|
103
|
+
return { text, key: createHash('sha256').update(text).digest('hex').slice(0, 16), risky: task ? task.items.length + task.alreadyReviewed : undefined, rules, lessons: lessons.length };
|
|
99
104
|
}
|
|
100
105
|
function where(x) {
|
|
101
106
|
return x.file ? `${x.file}${x.line ? `:${x.line}` : ''}` : '(no file)';
|
|
@@ -35,6 +35,8 @@ describe('what the judges are told the team knows', () => {
|
|
|
35
35
|
expect(told()).not.toContain('Log the save id');
|
|
36
36
|
expect(told('all')).toContain('Log the save id on failure.');
|
|
37
37
|
expect(told('off')).not.toContain('idempotent');
|
|
38
|
+
fs.writeFileSync(path.join(repo, 'AGENTS.md'), '- Every call to `save` in `src/a.ts` must be idempotent on retry.\n');
|
|
39
|
+
expect(told()).toMatch(/## Rules this repository wrote for itself[^\n]*\n- \[[0-9a-f]{10}\] \(AGENTS\.md, requirement\) Every call to `save` in `src\/a\.ts` must be idempotent on retry\./);
|
|
38
40
|
}
|
|
39
41
|
finally {
|
|
40
42
|
fs.rmSync(repo, { recursive: true, force: true });
|
|
@@ -111,12 +111,18 @@ Do these steps in order. Report only what you verified in the code, with file:li
|
|
|
111
111
|
reviewers asked for before; it is not a finding by itself. When the change does repeat it, write
|
|
112
112
|
the finding as for any other miss (input, consequence, quote), naming the lesson in why.
|
|
113
113
|
|
|
114
|
-
10.
|
|
114
|
+
10. Repository rules. For EVERY rule listed under "Rules this repository wrote for itself" in the
|
|
115
|
+
team knowledge file, say in rules, by the rule's id, whether this change follows it, breaks it,
|
|
116
|
+
or does not apply to it. For a break give file, line and quote: the code that breaks it, copied
|
|
117
|
+
exactly. Judge each rule as written, never widened. A rule marked requirement, broken, with its
|
|
118
|
+
quote verified, blocks; guidance broken is shown as a should-fix.
|
|
119
|
+
|
|
120
|
+
11. Review the diff the way the human reviewers do: correctness, production cost, dead code and
|
|
115
121
|
unreferenced exports (a test is not a consumer), code duplicated across sibling routes or
|
|
116
122
|
runners, links or ids built outside the helper that owns them, and the repository's rules.
|
|
117
123
|
|
|
118
|
-
The lists from steps 2-
|
|
119
|
-
own. A miss blocks only when you also put it in findings, with all three of:
|
|
124
|
+
The lists from steps 2-10 are your working notes: people see them, and they never block on their
|
|
125
|
+
own, with one exception: a requirement rule you mark broken, with its quote verified, blocks. A miss blocks only when you also put it in findings, with all three of:
|
|
120
126
|
- input: the concrete input, state or sequence that goes wrong (a user edits, a retry, two runs at once);
|
|
121
127
|
- consequence: what goes wrong for that input, or the cost (reads, calls or memory per what);
|
|
122
128
|
- quote: the code at file:line that does it, copied exactly from the file (one to three lines).
|
|
@@ -155,6 +161,7 @@ Your final message must be ONLY this JSON, starting with { and ending with }, no
|
|
|
155
161
|
"siblings":[{"changed":"file:line","sibling":"file:line","needs_same_change":true|false,"has_it":true|false,"why":"..."}],
|
|
156
162
|
"claims":[{"source":"comment"|"description","claim":"...","file":"<code that contradicts it>","line":0,"holds":true|false,"evidence":"..."}],
|
|
157
163
|
"lessons":[{"lesson":"<the lesson as listed>","applies":true|false,"file":"...","line":0,"evidence":"..."}],
|
|
164
|
+
"rules":[{"id":"<the rule's id as listed>","status":"followed"|"broken"|"not-applicable","file":"...","line":0,"quote":"<when broken: the code that breaks it, copied exactly>","evidence":"..."}],
|
|
158
165
|
"findings":[{"class":"...","severity":"blocking"|"should","file":"...","line":0,"issue":"...","why":"...","input":"...","consequence":"<wrong outcome for that input, or the cost; empty for an opinion>","quote":"<the code at file:line, copied exactly>","absent":"<for a missing call or check: the exact text that is missing>"}],
|
|
159
166
|
"carried":["<delta mode: ids of previous open items that still stand>"],
|
|
160
167
|
"resolved_previous":[{"id":"<delta mode: id of a previous open item now fixed>","evidence":"file:line and the fix"}]}`;
|
|
@@ -162,7 +169,7 @@ Your final message must be ONLY this JSON, starting with { and ending with }, no
|
|
|
162
169
|
export function deltaBlock(previousHead, previousVerdict, previousOpenFile, commitsFile, deltaDiffFile, settledFile) {
|
|
163
170
|
return `- DELTA MODE. The previous verdict on ${previousHead.slice(0, 9)} is at ${previousVerdict}; its open
|
|
164
171
|
items, each with an id, are in ${previousOpenFile}. Only the commits in ${commitsFile} are new;
|
|
165
|
-
their diff is ${deltaDiffFile}. Do steps 1-
|
|
172
|
+
their diff is ${deltaDiffFile}. Do steps 1-11 on that diff and on every file an open item names.
|
|
166
173
|
Then, for EVERY previous open item: put its id in "carried" if it still stands, or in
|
|
167
174
|
"resolved_previous" with the fix quoted at file:line. An id you leave out is treated as still open.
|
|
168
175
|
Human points the previous verdict resolved, whose files these commits do not touch, are in
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
import type { Accounting, OpenItem, Verdict } from './verdict.js';
|
|
2
|
+
export interface ReviewRecord {
|
|
3
|
+
version: 1;
|
|
4
|
+
head: string;
|
|
5
|
+
base: string;
|
|
6
|
+
scope: 'full' | 'delta';
|
|
7
|
+
at: string;
|
|
8
|
+
judges: Array<{
|
|
9
|
+
reviewer: string;
|
|
10
|
+
version?: string;
|
|
11
|
+
model?: string;
|
|
12
|
+
cost_usd?: number;
|
|
13
|
+
turns?: number;
|
|
14
|
+
}>;
|
|
15
|
+
/** Checked by Rigour against the checkout. */
|
|
16
|
+
verified: {
|
|
17
|
+
blocking: OpenItem[];
|
|
18
|
+
should_fix: OpenItem[];
|
|
19
|
+
/** The repository's own rules served to the judge, and its answers. */
|
|
20
|
+
rules: {
|
|
21
|
+
served: number;
|
|
22
|
+
followed: number;
|
|
23
|
+
broken: number;
|
|
24
|
+
not_applicable: number;
|
|
25
|
+
};
|
|
26
|
+
/** The team's lessons served, and how many the judge found the change repeats. */
|
|
27
|
+
lessons: {
|
|
28
|
+
served: number;
|
|
29
|
+
applied: number;
|
|
30
|
+
};
|
|
31
|
+
prior_points: {
|
|
32
|
+
open: number;
|
|
33
|
+
resolved: number;
|
|
34
|
+
answer_in_reply: number;
|
|
35
|
+
};
|
|
36
|
+
unverified: number;
|
|
37
|
+
notes: number;
|
|
38
|
+
disputed: number;
|
|
39
|
+
};
|
|
40
|
+
/** Recorded as reported, not checked by Rigour. */
|
|
41
|
+
reported: {
|
|
42
|
+
human_reviews: number;
|
|
43
|
+
};
|
|
44
|
+
/** Decisions people made. */
|
|
45
|
+
people: {
|
|
46
|
+
dismissed: number;
|
|
47
|
+
};
|
|
48
|
+
/** sha256 of everything above, keys sorted, so a copy can be checked against the original. */
|
|
49
|
+
integrity: string;
|
|
50
|
+
}
|
|
51
|
+
export interface RecordInput {
|
|
52
|
+
head: string;
|
|
53
|
+
base: string;
|
|
54
|
+
scope: 'full' | 'delta';
|
|
55
|
+
verdict: Verdict;
|
|
56
|
+
accounted: Accounting & {
|
|
57
|
+
disputed: OpenItem[];
|
|
58
|
+
dismissed: OpenItem[];
|
|
59
|
+
};
|
|
60
|
+
judges: ReviewRecord['judges'];
|
|
61
|
+
lessonsServed: number;
|
|
62
|
+
humanReviews: number;
|
|
63
|
+
at?: string;
|
|
64
|
+
}
|
|
65
|
+
export declare function buildRecord(input: RecordInput): ReviewRecord;
|
|
66
|
+
/** The hash a record's integrity field must equal; a record whose hash differs was changed after Rigour wrote it. */
|
|
67
|
+
export declare function integrityOf(body: Omit<ReviewRecord, 'integrity'>): string;
|
|
68
|
+
/** Checks that a record's contents hash to its integrity field. */
|
|
69
|
+
export declare function recordIntact(record: ReviewRecord): boolean;
|
|
70
|
+
/** The record as a pull request summary shows it: blocks in full, a few should-fixes, the counts, the judges, the hash. */
|
|
71
|
+
export declare function recordLines(r: ReviewRecord, shouldFixShown?: number): string[];
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The record of one review: what Rigour checked against the checkout, what it could only record as
|
|
3
|
+
* reported, who decided what, and who judged, with an integrity hash so a copy can be checked against
|
|
4
|
+
* the original. Nothing in it rests on a judge's word alone: every blocking or should-fix item here
|
|
5
|
+
* passed the quote check, and every rule and lesson count comes from what Rigour served and the
|
|
6
|
+
* answers it kept. The receipt a pull request carries is built from this.
|
|
7
|
+
*/
|
|
8
|
+
import { createHash } from 'crypto';
|
|
9
|
+
export function buildRecord(input) {
|
|
10
|
+
const rules = input.verdict.rules ?? [];
|
|
11
|
+
const lessons = input.verdict.lessons ?? [];
|
|
12
|
+
const body = {
|
|
13
|
+
version: 1,
|
|
14
|
+
head: input.head,
|
|
15
|
+
base: input.base,
|
|
16
|
+
scope: input.scope,
|
|
17
|
+
at: input.at ?? new Date().toISOString(),
|
|
18
|
+
judges: input.judges,
|
|
19
|
+
verified: {
|
|
20
|
+
blocking: input.accounted.open,
|
|
21
|
+
should_fix: input.accounted.advisory,
|
|
22
|
+
rules: { served: rules.length, followed: rules.filter(r => r.status === 'followed').length, broken: rules.filter(r => r.status === 'broken').length, not_applicable: rules.filter(r => r.status === 'not-applicable').length },
|
|
23
|
+
lessons: { served: input.lessonsServed, applied: lessons.filter(l => l.applies === true).length },
|
|
24
|
+
prior_points: { open: input.accounted.open.filter(i => i.kind === 'prior').length, resolved: input.accounted.resolved.length, answer_in_reply: input.accounted.answerInReply.length },
|
|
25
|
+
unverified: input.accounted.unverified.length,
|
|
26
|
+
notes: input.accounted.notes.length,
|
|
27
|
+
disputed: input.accounted.disputed.length,
|
|
28
|
+
},
|
|
29
|
+
reported: { human_reviews: input.humanReviews },
|
|
30
|
+
people: { dismissed: input.accounted.dismissed.length },
|
|
31
|
+
};
|
|
32
|
+
return { ...body, integrity: integrityOf(body) };
|
|
33
|
+
}
|
|
34
|
+
/** The hash a record's integrity field must equal; a record whose hash differs was changed after Rigour wrote it. */
|
|
35
|
+
export function integrityOf(body) {
|
|
36
|
+
return createHash('sha256').update(canonical(body)).digest('hex');
|
|
37
|
+
}
|
|
38
|
+
/** Checks that a record's contents hash to its integrity field. */
|
|
39
|
+
export function recordIntact(record) {
|
|
40
|
+
const { integrity, ...body } = record;
|
|
41
|
+
return integrityOf(body) === integrity;
|
|
42
|
+
}
|
|
43
|
+
/** JSON with every object's keys sorted and, as JSON itself does, no undefined members: a record read back from disk hashes the same. */
|
|
44
|
+
function canonical(value) {
|
|
45
|
+
if (Array.isArray(value))
|
|
46
|
+
return `[${value.map(v => canonical(v === undefined ? null : v)).join(',')}]`;
|
|
47
|
+
if (value && typeof value === 'object') {
|
|
48
|
+
const object = value;
|
|
49
|
+
return `{${Object.keys(object).filter(k => object[k] !== undefined).sort().map(k => `${JSON.stringify(k)}:${canonical(object[k])}`).join(',')}}`;
|
|
50
|
+
}
|
|
51
|
+
return JSON.stringify(value);
|
|
52
|
+
}
|
|
53
|
+
/** The record as a pull request summary shows it: blocks in full, a few should-fixes, the counts, the judges, the hash. */
|
|
54
|
+
export function recordLines(r, shouldFixShown = 5) {
|
|
55
|
+
const where = (i) => `${i.file ? `\`${i.file}${i.line ? `:${i.line}` : ''}\` ` : ''}${i.issue}${i.locations?.length ? ` (also ${i.locations.map(l => `\`${l.file}${l.line ? `:${l.line}` : ''}\``).join(', ')})` : ''}`;
|
|
56
|
+
const v = r.verified;
|
|
57
|
+
const lines = [`**Review record** · ${v.blocking.length} blocking · ${v.should_fix.length} should-fix · rules ${v.rules.followed} followed, ${v.rules.broken} broken, ${v.rules.not_applicable} not applicable of ${v.rules.served} · lessons ${v.lessons.applied} of ${v.lessons.served} apply · prior points ${v.prior_points.open} open, ${v.prior_points.resolved} resolved`];
|
|
58
|
+
for (const i of v.blocking)
|
|
59
|
+
lines.push(`- **Blocking** ${where(i)}`);
|
|
60
|
+
for (const i of v.should_fix.slice(0, shouldFixShown))
|
|
61
|
+
lines.push(`- Should fix: ${where(i)}`);
|
|
62
|
+
if (v.should_fix.length > shouldFixShown)
|
|
63
|
+
lines.push(`- …and ${v.should_fix.length - shouldFixShown} more should-fix in the record.`);
|
|
64
|
+
const folded = [[v.notes, 'working note'], [v.disputed, 'disputed'], [v.unverified, 'unverified'], [r.people.dismissed, 'dismissed']].filter(([n]) => n > 0);
|
|
65
|
+
if (folded.length)
|
|
66
|
+
lines.push(`Also seen, never blocking: ${folded.map(([n, w]) => `${n} ${w}${n === 1 || w === 'disputed' || w === 'unverified' || w === 'dismissed' ? '' : 's'}`).join(', ')}.`);
|
|
67
|
+
lines.push(`Judged by ${r.judges.map(j => `${j.reviewer}${j.version ? ` ${j.version}` : ''}${j.model ? ` (${j.model})` : ''}${typeof j.cost_usd === 'number' ? ` $${j.cost_usd.toFixed(2)}` : ''}`).join(', ') || 'no judge'} on \`${r.head.slice(0, 9)}\` against \`${r.base.slice(0, 9)}\` (${r.scope}); ${r.reported.human_reviews} human review(s) seen. Integrity \`${r.integrity.slice(0, 16)}\`.`);
|
|
68
|
+
return lines;
|
|
69
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
import { describe, expect, it } from 'vitest';
|
|
2
|
+
import { buildRecord, recordIntact, recordLines } from './record.js';
|
|
3
|
+
const item = (over) => ({ id: 'i1', kind: 'finding', class: 'correctness', file: 'src/job.ts', line: 2, issue: 'returns before the lock', quote: 'return 1;', ...over });
|
|
4
|
+
const input = () => ({
|
|
5
|
+
head: 'abcdef0123456789', base: '0123456789abcdef', scope: 'full',
|
|
6
|
+
verdict: { prior_points: [], redundant: [], reads: [], scans: [], merge_impact: [], findings: [], carried: [], resolved_previous: [],
|
|
7
|
+
rules: [{ id: 'r1', status: 'followed' }, { id: 'r2', status: 'broken' }, { id: 'r3', status: 'not-applicable' }], lessons: [{ lesson: 'bound the window', applies: true }, { lesson: 'keyset', applies: false }] },
|
|
8
|
+
accounted: { open: [item({}), item({ id: 'p1', kind: 'prior', class: 'prior point', issue: 'take the lock' })], advisory: [item({ id: 's1', issue: 'could log the id' })], unverified: [item({ id: 'u1' })], notes: [item({ id: 'n1' }), item({ id: 'n2' })], resolved: [{ item: item({ id: 'old' }), evidence: 'fixed' }], answerInReply: [], disputed: [], dismissed: [item({ id: 'd1' })] },
|
|
9
|
+
judges: [{ reviewer: 'claude', version: '2.1.0', model: 'opus', cost_usd: 1.25, turns: 19 }],
|
|
10
|
+
lessonsServed: 4, humanReviews: 2, at: '2026-10-08T00:00:00Z',
|
|
11
|
+
});
|
|
12
|
+
describe('the review record', () => {
|
|
13
|
+
it('counts what was verified, what was reported and what people decided, and hashes it', () => {
|
|
14
|
+
const record = buildRecord(input());
|
|
15
|
+
expect(record.verified).toMatchObject({ rules: { served: 3, followed: 1, broken: 1, not_applicable: 1 }, lessons: { served: 4, applied: 1 }, prior_points: { open: 1, resolved: 1, answer_in_reply: 0 }, unverified: 1, notes: 2, disputed: 0 });
|
|
16
|
+
expect(record.verified.blocking).toHaveLength(2);
|
|
17
|
+
expect(record.verified.should_fix).toHaveLength(1);
|
|
18
|
+
expect(record).toMatchObject({ reported: { human_reviews: 2 }, people: { dismissed: 1 }, judges: [{ reviewer: 'claude', turns: 19 }] });
|
|
19
|
+
expect(record.integrity).toMatch(/^[0-9a-f]{64}$/);
|
|
20
|
+
expect(buildRecord(input()).integrity).toBe(record.integrity); // the same review hashes the same
|
|
21
|
+
expect(recordIntact(record)).toBe(true);
|
|
22
|
+
expect(recordIntact({ ...record, verified: { ...record.verified, blocking: [] } })).toBe(false); // a block removed after the fact shows
|
|
23
|
+
});
|
|
24
|
+
it('renders for a pull request: blocks in full, a few should-fixes, the counts, the judges and the hash', () => {
|
|
25
|
+
const lines = recordLines(buildRecord(input()), 1);
|
|
26
|
+
expect(lines[0]).toBe('**Review record** · 2 blocking · 1 should-fix · rules 1 followed, 1 broken, 1 not applicable of 3 · lessons 1 of 4 apply · prior points 1 open, 1 resolved');
|
|
27
|
+
expect(lines).toContain('- **Blocking** `src/job.ts:2` returns before the lock');
|
|
28
|
+
expect(lines).toContain('- Should fix: `src/job.ts:2` could log the id');
|
|
29
|
+
expect(lines.at(-2)).toBe('Also seen, never blocking: 2 working notes, 1 unverified, 1 dismissed.');
|
|
30
|
+
expect(lines.at(-1)).toMatch(/^Judged by claude 2\.1\.0 \(opus\) \$1\.25 on `abcdef012` against `012345678` \(full\); 2 human review\(s\) seen\. Integrity `[0-9a-f]{16}`\.$/);
|
|
31
|
+
});
|
|
32
|
+
});
|
|
@@ -35,6 +35,16 @@ export interface ResolvedReviewer {
|
|
|
35
35
|
judges: 2 | 3;
|
|
36
36
|
escalate: 'always' | 'risk';
|
|
37
37
|
cross_models: Record<string, string>;
|
|
38
|
+
/** Reasoning effort per reviewer name (codex, api). */
|
|
39
|
+
reasoning: Record<string, 'low' | 'medium' | 'high'>;
|
|
40
|
+
/** The API judge, when the team configured one (review.reviewer.api). */
|
|
41
|
+
api?: {
|
|
42
|
+
url: string;
|
|
43
|
+
model: string;
|
|
44
|
+
key_env: string;
|
|
45
|
+
vendor?: 'anthropic' | 'openai' | 'google' | 'other';
|
|
46
|
+
max_turns: number;
|
|
47
|
+
};
|
|
38
48
|
/** Environment variables each judge's CLI must not see (the team's, never a person's). */
|
|
39
49
|
judge_env: Record<string, {
|
|
40
50
|
unset: string[];
|