@rigour-labs/core 6.9.0-rc.4 → 6.9.0-rc.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/dist/outcomes/metrics.d.ts +34 -2
  2. package/dist/outcomes/metrics.js +19 -1
  3. package/dist/outcomes/run.js +8 -3
  4. package/dist/review/reviewer/orchestrator.d.ts +51 -0
  5. package/dist/review/reviewer/orchestrator.js +96 -0
  6. package/dist/review/reviewer/settings.js +1 -1
  7. package/dist/review/reviewer/specialists/cleanup.v1.d.ts +9 -0
  8. package/dist/review/reviewer/specialists/cleanup.v1.js +9 -0
  9. package/dist/review/reviewer/specialists/correctness.v1.d.ts +9 -0
  10. package/dist/review/reviewer/specialists/correctness.v1.js +9 -0
  11. package/dist/review/reviewer/specialists/prior-points.v1.d.ts +9 -0
  12. package/dist/review/reviewer/specialists/prior-points.v1.js +9 -0
  13. package/dist/review/reviewer/specialists/production-cost.v1.d.ts +9 -0
  14. package/dist/review/reviewer/specialists/production-cost.v1.js +9 -0
  15. package/dist/review/reviewer/specialists/rules-and-goal.v1.d.ts +9 -0
  16. package/dist/review/reviewer/specialists/rules-and-goal.v1.js +9 -0
  17. package/dist/review/reviewer/store.d.ts +35 -0
  18. package/dist/review/reviewer/store.js +41 -0
  19. package/dist/review/reviewer/triage.d.ts +62 -0
  20. package/dist/review/reviewer/triage.js +147 -0
  21. package/dist/review/reviewer/usage.js +14 -0
  22. package/dist/review/reviewer/verdict.js +24 -7
  23. package/dist/review/reviewer.d.ts +31 -2
  24. package/dist/review/reviewer.js +187 -43
  25. package/dist/settings.d.ts +1 -0
  26. package/dist/switches.d.ts +311 -0
  27. package/dist/switches.js +1 -0
  28. package/dist/templates/universal-config.js +1 -0
  29. package/dist/types/index.d.ts +11 -0
  30. package/dist/types/index.js +5 -0
  31. package/package.json +6 -6
@@ -0,0 +1,147 @@
1
+ /** Prose: its own words are for the rules and the goal, never a correctness pass. Matched by extension only: a `docs/` folder holds code too. */
2
+ const PROSE = /\.(md|mdx|txt|rst|adoc)$/i;
3
+ /**
4
+ * The only files no model reviews: lockfiles, snapshots, source maps, minified bundles and files a generator marks
5
+ * as its own (`__generated__/`, `.generated.`, protobuf output). Everything else gets a correctness pass, whatever
6
+ * its language: a missed skip costs a little, a wrong one leaves code unreviewed.
7
+ */
8
+ const SKIP = [
9
+ /(^|\/)(package-lock\.json|npm-shrinkwrap\.json|pnpm-lock\.yaml|yarn\.lock|Cargo\.lock|poetry\.lock|Pipfile\.lock|uv\.lock|go\.sum|Gemfile\.lock|composer\.lock|bun\.lockb?)$/,
10
+ /\.snap$/, /\.(js|css|d\.ts)\.map$/, /\.min\.(js|css)$/,
11
+ /(^|\/)__generated__\//, /\.generated\.[A-Za-z0-9]+$/, /\.pb\.go$/, /_pb2(_grpc)?\.pyi?$/, /_pb\.(js|ts|d\.ts)$/,
12
+ ];
13
+ /** The most parts a split runs: a change that needs more is one combined pass. */
14
+ export const MAX_PARTS = 3;
15
+ const MIGRATION = /(^|\/)(migrations?|db\/migrate)\/|\.sql$|(^|\/)schema\.prisma$/i;
16
+ /**
17
+ * Lines that read data, per language. Each names a query API, not a word any code uses: `Array.from`, `map.get`,
18
+ * `items.filter` and a `limit` variable are not reads.
19
+ */
20
+ const READS = [
21
+ // SQL, in a .sql file or a string
22
+ /\bselect\s[\s\S]{0,80}?\bfrom\s+[A-Za-z_"`[]|\binsert\s+into\b|\bupdate\s+[A-Za-z_"`.]+\s+set\b|\bdelete\s+from\b|\bcreate\s+(unique\s+)?index\b|\balter\s+table\b|\bcreate\s+table\b|\blimit\s+\d+|\boffset\s+\d+/i,
23
+ // TS/JS: ORMs, query builders, Supabase, fetch
24
+ /\.(query|queryRaw|\$queryRaw|\$executeRaw|findMany|findFirst|findUnique|findOne|findAll|aggregate|groupBy|rpc)\s*\(|\.from\(\s*['"`]|\bfetch\s*\(|\.(range|limit|offset)\s*\(\s*\d/,
25
+ // Python: DB-API, SQLAlchemy, Django
26
+ /\.(execute|executemany|fetchall|fetchone|fetchmany)\s*\(|\bsession\.(query|execute|scalars)\s*\(|\.objects\.(filter|all|get|exclude|raw)\s*\(/,
27
+ // Go: database/sql, sqlx, GORM on a db/tx/conn value
28
+ /\.(Query|QueryRow|QueryContext|QueryRowContext|Exec|ExecContext)\s*\(|\b(db|tx|conn)\.(Get|Select|Find|First|Where)\s*\(|\brows\.Next\s*\(/,
29
+ ];
30
+ /** A loop, and an await within the next few lines of it: one read per turn. */
31
+ const LOOP = /\b(for|while)\b\s*[(\w]|\.(forEach|map|flatMap|reduce)\s*\(\s*async\b|\basync\s+for\b/;
32
+ const AWAIT = /\bawait\b/;
33
+ const LOOP_REACH = 3;
34
+ const DEFINES = /^\s*(?:export\s+)?(?:default\s+)?(?:async\s+)?(?:function\*?|const|let|var|class|interface|type|enum|def|func)\s+(?:\([^)]*\)\s*)?([A-Za-z_$][\w$]*)/;
35
+ const DECLARES = /^\s*(export\s|(async\s+)?function\s|def\s|func\s|class\s)/;
36
+ const COMMENT = /^\s*(\/\/|#|\/\*|\*|<!--)/;
37
+ const IDENTIFIER = /[A-Za-z_$][\w$]{2,}/g;
38
+ /** The diff's hunks, each with its file header. */
39
+ export function parseHunks(diff) {
40
+ const hunks = [];
41
+ for (const block of diff.split(/^(?=diff --git )/m)) {
42
+ const file = /^diff --git a\/.+? b\/(.+)$/m.exec(block)?.[1];
43
+ if (!file)
44
+ continue;
45
+ const headerEnd = block.search(/^@@/m);
46
+ const header = headerEnd >= 0 ? block.slice(0, headerEnd) : block;
47
+ const newFile = /^new file mode|^--- \/dev\/null$/m.test(header);
48
+ const bodies = headerEnd >= 0 ? block.slice(headerEnd).split(/^(?=@@)/m) : [];
49
+ for (const body of bodies) {
50
+ const lines = body.split('\n');
51
+ const added = lines.filter(l => l.startsWith('+') && !l.startsWith('+++')).map(l => l.slice(1));
52
+ const removed = lines.filter(l => l.startsWith('-') && !l.startsWith('---')).map(l => l.slice(1));
53
+ const defines = new Set(added.map(l => DEFINES.exec(l)?.[1]).filter((n) => !!n));
54
+ const references = new Set([...added, ...removed].flatMap(l => l.match(IDENTIFIER) ?? []));
55
+ hunks.push({ file, text: `${header}${body}`, added, removed, newFile, defines, references });
56
+ }
57
+ }
58
+ return hunks;
59
+ }
60
+ /** Per specialist, the hunks it is for (by index); a specialist absent from the map is not needed. Prior points take the whole change. */
61
+ export function triage(hunks, context) {
62
+ const picked = new Map();
63
+ const pick = (id, index) => picked.set(id, [...(picked.get(id) ?? []), index]);
64
+ hunks.forEach((hunk, i) => {
65
+ if (skipped(hunk.file))
66
+ return;
67
+ const prose = PROSE.test(hunk.file);
68
+ const lines = [...hunk.added, ...hunk.removed];
69
+ if (!prose)
70
+ pick('correctness', i);
71
+ if (!prose && (MIGRATION.test(hunk.file) || READS.some(pattern => lines.some(line => pattern.test(line))) || awaitsInLoop(hunk.added)))
72
+ pick('production-cost', i);
73
+ if (!prose && (hunk.removed.length > 0 || hunk.newFile || hunk.added.some(l => DECLARES.test(l))))
74
+ pick('cleanup', i);
75
+ if (prose || hunk.added.some(l => COMMENT.test(l)) || context.rulesAndLessons > 0 || context.goal)
76
+ pick('rules-and-goal', i);
77
+ });
78
+ if (context.humanReviews > 0)
79
+ picked.set('prior-points', hunks.map((_, i) => i).filter(i => !skipped(hunks[i].file)));
80
+ return picked;
81
+ }
82
+ /** Whether no model reviews this file (SKIP). */
83
+ export function skipped(file) {
84
+ return SKIP.some(pattern => pattern.test(file));
85
+ }
86
+ function awaitsInLoop(lines) {
87
+ return lines.some((line, i) => LOOP.test(line) && lines.slice(i, i + LOOP_REACH + 1).some(l => AWAIT.test(l)));
88
+ }
89
+ /**
90
+ * A pass's part of the diff: its hunks, and the hunk defining each name they use, in diff order. Only a name exactly one
91
+ * hunk defines is followed (a local `result` defined in ten files is no one definition), and a definition is added only
92
+ * while the part stays within `limit`.
93
+ */
94
+ function sliceOf(hunks, definedIn, indices, limit) {
95
+ const chosen = new Set(indices);
96
+ let size = indices.reduce((sum, i) => sum + hunks[i].text.length, 0);
97
+ const referenced = new Set(indices.flatMap(i => [...hunks[i].references]));
98
+ for (const name of referenced) {
99
+ const at = definedIn.get(name);
100
+ if (at?.length !== 1 || chosen.has(at[0]) || size + hunks[at[0]].text.length > limit)
101
+ continue;
102
+ chosen.add(at[0]);
103
+ size += hunks[at[0]].text.length;
104
+ }
105
+ return [...chosen].sort((a, b) => a - b);
106
+ }
107
+ /**
108
+ * The passes for what triage picked. A split is by hunk, in diff order: each part takes the next hunks while they stay
109
+ * within `limit`, and runs every specialist that picked any of them. A single hunk over the limit is a part of its own.
110
+ * A change that needs more than MAX_PARTS parts is not split.
111
+ */
112
+ export function planPasses(hunks, picked, order, limit) {
113
+ const ids = order.map(s => s.id).filter(id => picked.has(id));
114
+ if (ids.length === 0)
115
+ return {};
116
+ const definedIn = new Map();
117
+ hunks.forEach((hunk, i) => hunk.defines.forEach(name => definedIn.set(name, [...(definedIn.get(name) ?? []), i])));
118
+ const pickedBy = new Map(ids.map(id => [id, new Set(picked.get(id))]));
119
+ const pass = (indices) => {
120
+ const sliced = sliceOf(hunks, definedIn, indices, limit);
121
+ return { specialists: ids.filter(id => indices.some(i => pickedBy.get(id).has(i))), hunks: indices, sliced, diff: sliced.map(i => hunks[i].text).join('') };
122
+ };
123
+ const union = [...new Set(ids.flatMap(id => picked.get(id)))].sort((a, b) => a - b);
124
+ const combined = pass(union);
125
+ if (combined.diff.length <= limit || union.length === 1)
126
+ return { combined };
127
+ // By the hunks' own size: definitions join a part only within the limit (sliceOf), so they never push it over.
128
+ const parts = [[]];
129
+ let size = 0;
130
+ for (const index of union) {
131
+ const length = hunks[index].text.length;
132
+ if (parts.at(-1).length && size + length > limit) {
133
+ parts.push([]);
134
+ size = 0;
135
+ }
136
+ parts.at(-1).push(index);
137
+ size += length;
138
+ }
139
+ if (parts.length > MAX_PARTS)
140
+ return { combined, needsParts: parts.length };
141
+ return parts.length > 1 ? { combined, split: parts.map(pass) } : { combined };
142
+ }
143
+ /** Characters of the reviewable diff (no lockfiles or generated files) and its changed lines: both modes are measured on this. */
144
+ export function reviewable(hunks) {
145
+ const kept = hunks.filter(h => !skipped(h.file));
146
+ return { chars: kept.reduce((sum, h) => sum + h.text.length, 0), lines: kept.reduce((sum, h) => sum + h.added.length + h.removed.length, 0) };
147
+ }
@@ -24,5 +24,19 @@ export function reviewerUsage(result, trigger) {
24
24
  dismissed: result.dismissed.length,
25
25
  runs: result.runs,
26
26
  cost_bucket: costBucket(result.costUsd),
27
+ ...orchestrated(mode?.specialists),
28
+ };
29
+ }
30
+ /** With the orchestrator: how many parts triage picked, how many passes ran, whether it split, fell back or had nothing to review, and how many passes read beyond their slice. */
31
+ function orchestrated(specialists) {
32
+ if (!specialists)
33
+ return {};
34
+ return {
35
+ parts: specialists.selected.length,
36
+ passes: specialists.passes.length,
37
+ split: specialists.passes.length > 1,
38
+ fallback: !!specialists.fallback,
39
+ nothing_to_review: !!specialists.none,
40
+ beyond_slice: specialists.passes.filter(p => p.readBeyondSlice === true).length,
27
41
  };
28
42
  }
@@ -243,6 +243,17 @@ export function account(verdict, previousOpen, verify, prior = NO_PRIOR_CHECKS)
243
243
  const notes = [];
244
244
  const advisory = [];
245
245
  const seen = new Set();
246
+ // Every item kept, by id: the same item from a second judge or specialist names it too instead of vanishing.
247
+ const kept = new Map();
248
+ const again = (item) => {
249
+ const first = kept.get(item.id);
250
+ if (first)
251
+ first.reviewer = bothReviewers(first.reviewer, item.reviewer);
252
+ };
253
+ const keep = (list, item) => {
254
+ list.push(item);
255
+ kept.set(item.id, item);
256
+ };
246
257
  // The reviewer's own label wins over the judge's reading of it: a judge that calls a blocker a should-fix would demote it silently.
247
258
  const read = verdict.prior_points.map(p => labelled(p, prior.labels ?? []));
248
259
  const points = read.map(r => r.point);
@@ -250,29 +261,29 @@ export function account(verdict, previousOpen, verify, prior = NO_PRIOR_CHECKS)
250
261
  // A should-fix is shown only when the judge could show it: a quote Rigour finds. One that cannot be checked is not a claim worth a person's time.
251
262
  const advise = (item) => {
252
263
  if (seen.has(item.id))
253
- return;
264
+ return void again(item);
254
265
  seen.add(item.id);
255
- (!!item.file && !!item.quote?.trim() && verify(item.file, item.line, item.quote) ? advisory : unverified).push(item);
266
+ keep(!!item.file && !!item.quote?.trim() && verify(item.file, item.line, item.quote) ? advisory : unverified, item);
256
267
  };
257
268
  // A prior point is the human's and needs no file. A finding blocks only when the code it quotes is at the line it
258
269
  // names: any model's claim is checked, never trusted. What the working steps turned up is a note: the reasoning,
259
270
  // shown, and a block only when the judge also makes it a finding it can quote.
260
271
  const add = (item) => {
261
272
  if (seen.has(item.id))
262
- return;
273
+ return void again(item);
263
274
  seen.add(item.id);
264
275
  if (WORKING_NOTES.has(item.kind))
265
- return void notes.push(item);
276
+ return void keep(notes, item);
266
277
  const placed = !!item.file && !!item.quote?.trim() && verify(item.file, item.line, item.quote);
267
278
  if (!placed)
268
- return void unverified.push(item);
279
+ return void keep(unverified, item);
269
280
  // A block is about this change. A finding or rule break in a touched file but on lines the change did not touch is
270
281
  // what the code already had: shown as a note, never a block on this change. A human's point is about the change by
271
282
  // definition. Without the diff (a judge's own items for a panel, a test) nothing is known and nothing is moved.
272
283
  if (item.kind !== 'prior' && prior.changed && (item.line === undefined || !nearChanged(prior.changed, item.file, item.line))) {
273
- return void notes.push({ ...item, evidence: `${item.evidence ? `${item.evidence}; ` : ''}${item.line === undefined ? 'names no line' : 'on a line this change did not touch'}: what the code already had, never a block on this change` });
284
+ return void keep(notes, { ...item, evidence: `${item.evidence ? `${item.evidence}; ` : ''}${item.line === undefined ? 'names no line' : 'on a line this change did not touch'}: what the code already had, never a block on this change` });
274
285
  }
275
- open.push(item);
286
+ keep(open, item);
276
287
  };
277
288
  const answerInReply = [];
278
289
  for (const p of points) {
@@ -441,11 +452,17 @@ function onePerRootCause(items) {
441
452
  kept.push(item);
442
453
  continue;
443
454
  }
455
+ same.reviewer = bothReviewers(same.reviewer, item.reviewer);
444
456
  if (item.file)
445
457
  (same.locations ??= []).push({ file: item.file, ...(item.line ? { line: item.line } : {}) });
446
458
  }
447
459
  return kept;
448
460
  }
461
+ /** Who found an item, each once: `claude:correctness+claude:cleanup`, as the panel tags `claude+codex`. */
462
+ function bothReviewers(a, b) {
463
+ const names = [...new Set([...(a ?? '').split('+'), ...(b ?? '').split('+')].filter(Boolean))];
464
+ return names.length ? names.join('+') : undefined;
465
+ }
449
466
  export function itemLine(item) {
450
467
  const where = item.file ? ` ${item.file}${item.line ? `:${item.line}` : ''}` : '';
451
468
  const by = item.reviewer ? ` (${item.reviewer})` : '';
@@ -35,6 +35,8 @@ export interface ReviewerOptions {
35
35
  branch?: string;
36
36
  /** This run's choice for the goal check (`--goal` / `--no-goal`), the nearest layer of switches.ts. */
37
37
  goal?: boolean;
38
+ /** This run's choice for the orchestrator (`--orchestrator` / `--no-orchestrator`), the nearest layer of switches.ts. */
39
+ orchestrator?: boolean;
38
40
  }
39
41
  export interface ReviewerResult {
40
42
  outcome: ReviewerOutcome;
@@ -64,6 +66,8 @@ export interface ReviewerResult {
64
66
  scope?: 'full' | 'delta';
65
67
  why?: string;
66
68
  costUsd?: number;
69
+ /** What every run of this fresh review reported costing, failed runs included: the number its cost row and its thread event carry. */
70
+ spentUsd?: number;
67
71
  /** The repository's own rules the judge answered, and how. */
68
72
  rules?: {
69
73
  checked: number;
@@ -92,9 +96,34 @@ export interface ReviewerResult {
92
96
  prTitle?: string;
93
97
  }
94
98
  export interface ModeRecord {
95
- asked: 'single' | 'cross' | 'full' | 'panel';
99
+ asked: 'single' | 'cross' | 'full' | 'panel' | 'orchestrator';
96
100
  /** `none` when the review ended before any judge ran (unavailable or skipped); `degraded` or the reason says why. */
97
- ran: 'single' | 'cross' | 'full' | 'panel' | 'none';
101
+ ran: 'single' | 'cross' | 'full' | 'panel' | 'orchestrator' | 'none';
102
+ /**
103
+ * With the orchestrator: the specialists triage picked, the passes that returned and those that did not (by label),
104
+ * each pass (its parts, its slice, whether it read beyond the slice: null without a trace), why the plan is what it
105
+ * is when the change was over the judge's limit or the caps, why it fell back to one judge if it did, and `none`
106
+ * when there was nothing for a model to review.
107
+ */
108
+ specialists?: {
109
+ selected: string[];
110
+ returned: string[];
111
+ missing: string[];
112
+ passes: Array<{
113
+ specialists: string[];
114
+ hunks: number;
115
+ chars: number;
116
+ readBeyondSlice: boolean | null;
117
+ }>;
118
+ /** The most diff one pass could be given, and the judge it came from (orchestrator.ts passLimit). */
119
+ limit: {
120
+ judge: ReviewerName;
121
+ chars: number;
122
+ };
123
+ plan?: string;
124
+ fallback?: string;
125
+ none?: string;
126
+ };
98
127
  source: Source;
99
128
  /** Why fewer judges ran than were asked for. */
100
129
  degraded?: string;
@@ -28,6 +28,8 @@ import { bodyAsOf, findPullRequest, ghFor, humanReviews, linesChanged, mergesBas
28
28
  import { mergeImpact } from './reviewer/merge-impact.js';
29
29
  import { applyPanel, parseAnswers, runPanel } from './reviewer/panel.js';
30
30
  import { crossExamPrompt, deltaBlock, goalStep, mergeBlock, PROMPT_VERSION, renderPrompt } from './reviewer/prompt.js';
31
+ import { BASELINE_MIN_SINGLES, focusBlock, formatLedger, ledger, passLimit, runPasses, SPECIALISTS, SPECIALISTS_KEY, splitNeeds } from './reviewer/orchestrator.js';
32
+ import { MAX_PARTS, parseHunks, planPasses, reviewable, triage } from './reviewer/triage.js';
31
33
  import { modelGoalItems, parseGoal } from '../goal/goal.js';
32
34
  import { resolveSwitch } from '../switches.js';
33
35
  import { resolveReviewer } from './reviewer/settings.js';
@@ -57,10 +59,13 @@ export async function runReviewer(cwd, base, config, exec = defaultExec, progres
57
59
  if (trigger !== 'backtest' && result.outcome !== 'skipped')
58
60
  appendTaskEvent(cwd, {
59
61
  kind: 'review', trigger, outcome: result.outcome, blocking: result.items.length, should_fix: result.advisory.length,
62
+ ...(options.checks ? { checks: options.checks.length } : {}),
60
63
  ...(result.pr ? { pr: result.pr } : {}),
61
64
  ...(result.prTitle ? { pr_title: result.prTitle } : {}),
62
65
  ...(result.lessonsApplied ? { lessons_applied: result.lessonsApplied } : {}),
63
- ...(result.record ? { integrity: result.record.integrity, cost_usd: result.record.judges.reduce((sum, j) => sum + (j.cost_usd ?? 0), 0), judges: result.record.judges.map(j => j.reviewer) } : {}),
66
+ ...(result.record ? { integrity: result.record.integrity, judges: result.record.judges.map(j => j.reviewer) } : {}),
67
+ // Every run this review made, failed ones included: the same dollars as its cost row (the savings ledger).
68
+ ...(result.spentUsd !== undefined ? { cost_usd: result.spentUsd } : {}),
64
69
  });
65
70
  return result;
66
71
  }
@@ -110,6 +115,19 @@ async function review(cwd, base, config, exec, progress, options) {
110
115
  if (reviewers.length < 2 && mode === 'full' && (settings.required.panel || settings.required.mode)) {
111
116
  return none('unavailable', `rigour.yml requires two reviewers from different vendors, and ${modeRecord.degraded}`, { reviewers, mode: modeRecord });
112
117
  }
118
+ // The orchestrator runs the specialists on the first judge. A team floor on the panel or the mode wins over it.
119
+ const orchestrator = resolveSwitch('orchestrator', config, options.orchestrator);
120
+ let orchestrate = orchestrator.enabled;
121
+ if (orchestrator.refused.length)
122
+ modeRecord = { ...modeRecord, refused: [...(modeRecord.refused ?? []), ...orchestrator.refused] };
123
+ if (orchestrate && (settings.required.panel || settings.required.mode)) {
124
+ orchestrate = false;
125
+ modeRecord = { ...modeRecord, refused: [...(modeRecord.refused ?? []), 'orchestrator refused: rigour.yml requires the panel or the mode'] };
126
+ }
127
+ if (orchestrate) {
128
+ reviewers = reviewers.slice(0, 1);
129
+ modeRecord = { ...modeRecord, asked: 'orchestrator', ran: 'orchestrator', source: orchestrator.source };
130
+ }
113
131
  const gh = options.blind ? undefined : ghFor(cwd, exec, await githubEnv(cwd, config.review?.github_account ?? process.env.RIGOUR_GITHUB_ACCOUNT, exec));
114
132
  const found = gh ? await findPullRequest(gh, branch, head, options.pr) : {};
115
133
  if (found.error)
@@ -140,7 +158,7 @@ async function review(cwd, base, config, exec, progress, options) {
140
158
  const goalText = goalItems.map(item => `- [${item.kind}] ${item.text}`).join('\n');
141
159
  const previous = branch !== 'HEAD' ? store.branchState(branch) : undefined;
142
160
  // The same commit, asked again with the same settings and reviews (the background run, then the person): the verdict it already has.
143
- const inputsKey = sha([PROMPT_VERSION, rules, body, goalText, reviews.key, JSON.stringify([settings.mode, settings.panel, settings.judges, settings.escalate, settings.panel_max_items, settings.cross_models, settings.models, candidates]), [...installed].map(([n, i]) => `${n} ${i.version}`).join(';')]);
161
+ const inputsKey = sha([PROMPT_VERSION, rules, body, goalText, reviews.key, JSON.stringify([settings.mode, settings.panel, settings.judges, settings.escalate, settings.panel_max_items, settings.cross_models, settings.models, candidates, orchestrate ? SPECIALISTS_KEY : '']), [...installed].map(([n, i]) => `${n} ${i.version}`).join(';')]);
144
162
  if (!options.force && previous?.head === head && previous.inputsKey === inputsKey && fs.existsSync(store.decidedPath(previous.verdict))) {
145
163
  const verdict = store.readJson(previous.verdict);
146
164
  const decided = store.readJson(store.decidedPath(previous.verdict));
@@ -225,11 +243,66 @@ async function review(cwd, base, config, exec, progress, options) {
225
243
  const verdict = store.readJson(verdictFile);
226
244
  return withRecord(decide(verdict, previousOpen, verify, prior, dismissals), verdict, true);
227
245
  }
228
- // The daily caps, before any judge starts: a cached or reused verdict above cost nothing and never reaches here.
229
- const over = overBudget(store.spend(), settings, reviewers.length);
246
+ // What one judge would be given: the shared input files and the reviewable diff. Both modes' cost rows measure this.
247
+ const hunks = parseHunks(fullDiff);
248
+ const size = reviewable(hunks);
249
+ const shared = reviews.markdown.length + body.length + context.text.length + (options.hints?.trim() || 'none\n').length + (goalText?.length ?? 0);
250
+ const projectedSingle = shared + size.chars;
251
+ // The orchestrator's plan: which specialists the change needs, hunk by hunk, as one combined pass. A change over the
252
+ // judge's limit is split by hunk only when the savings ledger covers the split's extra. Nothing to review: no pass.
253
+ const picked = orchestrate ? triage(hunks, { humanReviews: reviews.count, rulesAndLessons: context.rules.length + context.lessons, goal: goalItems.length > 0 }) : new Map();
254
+ const selected = SPECIALISTS.map(s => s.id).filter(id => picked.has(id));
255
+ let passes = [];
256
+ let planNote;
257
+ const limit = { judge: reviewers[0], chars: passLimit(reviewers[0], settings.timeout_ms) };
258
+ if (orchestrate) {
259
+ const baseline = store.freezeBaseline(BASELINE_MIN_SINGLES);
260
+ const plan = planPasses(hunks, picked, SPECIALISTS, limit.chars);
261
+ passes = plan.combined ? [plan.combined] : [];
262
+ if (plan.needsParts)
263
+ planNote = `over the judge's limit; needs ${plan.needsParts} parts > ${MAX_PARTS}: one pass`;
264
+ if (plan.split) {
265
+ const book = ledger(store.costs(), baseline);
266
+ const needs = splitNeeds(plan.split.reduce((sum, pass) => sum + shared + pass.diff.length, 0), projectedSingle, book.unit, baseline);
267
+ const said = `ledger ${formatLedger(book.credit, book.unit)}, split needs ${formatLedger(needs, book.unit)}`;
268
+ if (book.credit > 0 && book.credit >= needs) {
269
+ passes = plan.split;
270
+ planNote = `over the judge's limit; ${said}: ${plan.split.length} parts`;
271
+ }
272
+ else
273
+ planNote = `over the judge's limit; ${said}: one pass`;
274
+ }
275
+ // The daily caps, before any judge starts: room for one run but not every part is one combined pass.
276
+ if (passes.length > 1 && overBudget(store.spend(), settings, passes.length) && !overBudget(store.spend(), settings, 1)) {
277
+ planNote = `the caps leave one run, not ${passes.length}: one pass`;
278
+ passes = [plan.combined];
279
+ }
280
+ }
281
+ const over = orchestrate && passes.length === 0 ? undefined : overBudget(store.spend(), settings, orchestrate ? passes.length : reviewers.length);
282
+ // A team that requires the reviewer, or the orchestrator, gets no quieter review when the caps are reached: unavailable.
230
283
  if (over)
231
- return none(settings.required.panel || settings.required.mode ? 'unavailable' : 'skipped', over, { reviewers, scope, why, pr: pr?.number });
284
+ return none(settings.required.panel || settings.required.mode || orchestrator.required ? 'unavailable' : 'skipped', over, { reviewers, scope, why, pr: pr?.number });
232
285
  const work = fs.mkdtempSync(path.join(os.tmpdir(), 'rigour-reviewer-'));
286
+ // Every run this review makes, failed ones included: the caps count each, and the review's cost row sums them.
287
+ const tally = { runs: 0, chars: 0, usd: 0 };
288
+ const spent = (usd, chars) => {
289
+ store.addSpend(1, usd);
290
+ tally.runs++;
291
+ tally.chars += chars;
292
+ tally.usd += usd ?? 0;
293
+ };
294
+ const spentUsd = () => Math.round(tally.usd * 10_000) / 10_000;
295
+ // One row per fresh review, for the savings ledger: one judge asked for and run, or orchestrated with everything it ran.
296
+ const recordReviewCost = () => {
297
+ const orchestrated = modeRecord.asked === 'orchestrator';
298
+ if (!orchestrated && !(modeRecord.asked === 'single' && modeRecord.ran === 'single'))
299
+ return;
300
+ store.recordCost({
301
+ at: new Date().toISOString(), mode: orchestrated ? 'orchestrator' : 'single', lines: size.lines, projectedSingleChars: projectedSingle,
302
+ ...(orchestrated ? { projectedChars: passes.reduce((sum, pass) => sum + shared + pass.diff.length, 0) } : {}),
303
+ actualChars: tally.chars, actualUsd: spentUsd(), runs: tally.runs,
304
+ });
305
+ };
233
306
  // One judge run, by CLI or by API: the same prompt, the same cost accounting, the same trace.
234
307
  let inlineInputs = [];
235
308
  const runJudge = (name, prompt, model) => name === 'api'
@@ -274,43 +347,102 @@ async function review(cwd, base, config, exec, progress, options) {
274
347
  const ticker = setInterval(() => progress(`Rigour reviewer: still working (${Math.round((Date.now() - started) / 60_000)} min)`), PROGRESS_EVERY_MS);
275
348
  let parts;
276
349
  try {
277
- const answers = await Promise.all(reviewers.map(async (name) => {
278
- const adapter = ADAPTERS[name];
279
- const ask = async () => {
280
- const run = await runJudge(name, prompt, modelFor(name));
281
- progress(`Rigour reviewer: ${name} finished in ${Math.round((Date.now() - started) / 1000)}s (exit ${run.exitCode})`);
282
- const answer = adapter.answer(run.stdout);
283
- store.addSpend(1, answer.costUsd); // every run counts against the caps, an answer or not
284
- return { run, answer, verdict: run.exitCode === 0 || answer.text.trim() ? parseVerdict(answer.text, needsPriorPoints, name, answer) : undefined };
285
- };
286
- let first = await ask();
287
- // No verdict, whether a malformed answer or a run that died, is a slip, not a decision: asked once more, inside the caps.
288
- if ((!first.verdict || 'error' in first.verdict) && !overBudget(store.spend(), settings, 1)) {
289
- progress(`Rigour reviewer: ${name} gave no ${first.verdict ? 'valid verdict' : 'answer'}; asking once more`);
290
- first = await ask();
350
+ // The single-judge fallback runs once: no retry, no spare judge.
351
+ let once = false;
352
+ if (orchestrate && passes.length === 0) {
353
+ // Nothing for a model to review (only lockfiles, generated files): no run, and that is the verdict.
354
+ parts = [];
355
+ modeRecord = { ...modeRecord, specialists: { selected: [], returned: [], missing: [], passes: [], limit, none: 'nothing for the model reviewer to review' } };
356
+ }
357
+ else if (orchestrate) {
358
+ const judge = reviewers[0];
359
+ const ran = [];
360
+ const run = await runPasses(judge, passes.map((pass, i) => ({ ...pass, file: file(`part-${i + 1}.diff`, pass.diff) })), async (pass) => {
361
+ const assigned = pass.specialists.map(id => SPECIALISTS.find(s => s.id === id));
362
+ const result = await runJudge(judge, `${prompt}${focusBlock(assigned, pass.file)}`, modelFor(judge));
363
+ const answer = ADAPTERS[judge].answer(result.stdout);
364
+ spent(answer.costUsd, shared + pass.diff.length); // every pass counts against the caps and the ledger, an answer or not
365
+ progress(`Rigour reviewer: ${judge} (${pass.specialists.join(', ')}) finished in ${Math.round((Date.now() - started) / 1000)}s (exit ${result.exitCode})`);
366
+ const verdict = result.exitCode === 0 || answer.text.trim()
367
+ ? parseVerdict(answer.text, needsPriorPoints && pass.specialists.includes('prior-points'), judge, answer)
368
+ : { error: `${judge} (${pass.specialists.join(', ')}): no answer (exit ${result.exitCode})` };
369
+ // Whether slicing held: a pass that read the full diff, or read, searched or printed a changed file outside its slice.
370
+ const sliceFiles = new Set(pass.sliced.map(i => hunks[i].file));
371
+ const outside = changedFiles.filter(f => !sliceFiles.has(f));
372
+ const trace = 'verdict' in verdict ? verdict.verdict.trace : undefined;
373
+ const beyond = trace ? trace.calls.some(c => {
374
+ const target = c.target.replace(/\\/g, '/');
375
+ return target.endsWith('/full.diff') || outside.some(f => names(target, f));
376
+ }) : null;
377
+ ran.push({ specialists: pass.specialists, hunks: pass.hunks.length, chars: pass.diff.length, readBeyondSlice: beyond });
378
+ return verdict;
379
+ });
380
+ const specialists = { selected, returned: run.returned, missing: run.missing, passes: ran, limit, ...(planNote ? { plan: planNote } : {}) };
381
+ if (orchestrator.required && run.missing.length) {
382
+ // A required orchestrator reviews every part or gives no verdict: a partial review, or one judge instead, is a quieter one.
383
+ recordReviewCost();
384
+ const fallback = `${run.returned.length} of ${passes.length} passes returned`;
385
+ return none('unavailable', `${fallback}, and rigour.yml requires the orchestrator: every part or no verdict (not reviewed: ${run.missing.join('; ')})`, { reviewers, scope, why, pr: pr?.number, spentUsd: spentUsd(), mode: { ...modeRecord, specialists: { ...specialists, fallback } } });
291
386
  }
292
- return first.verdict ?? { error: `${name}: no answer (exit ${first.run.exitCode}): ${first.run.stderr.trim().slice(-200)}` };
293
- }));
294
- // A judge that gives nothing is replaced by the next one installed, so the boundary stays up: a review ends unavailable only when every judge failed.
295
- const spare = candidates.filter(c => installed.has(c) && !reviewers.includes(c));
296
- for (let i = 0; i < answers.length; i++) {
297
- let answer = answers[i];
298
- while ('error' in answer && spare.length && !overBudget(store.spend(), settings, 1)) {
299
- const next = spare.shift();
300
- progress(`Rigour reviewer: ${reviewers[i]} gave no verdict (${answer.error}); ${next} judges instead`);
301
- modeRecord = { ...modeRecord, degraded: `${modeRecord.degraded ? `${modeRecord.degraded}; ` : ''}${reviewers[i]} gave no verdict, ${next} judged instead` };
302
- reviewers[i] = next;
303
- const run = await runJudge(next, prompt, modelFor(next));
304
- const got = ADAPTERS[next].answer(run.stdout);
305
- store.addSpend(1, got.costUsd);
306
- answer = run.exitCode === 0 || got.text.trim() ? parseVerdict(got.text, needsPriorPoints, next, got) : { error: `${next}: no answer (exit ${run.exitCode}): ${run.stderr.trim().slice(-200)}` };
387
+ if (run.stands) {
388
+ parts = run.parts;
389
+ modeRecord = { ...modeRecord, specialists, ...(run.missing.length ? { degraded: `${modeRecord.degraded ? `${modeRecord.degraded}; ` : ''}not reviewed: ${run.missing.join('; ')} (no verdict)` } : {}) };
390
+ }
391
+ else {
392
+ // Fewer than half of the passes came back: one judge instead, once, only if the caps still allow a run; it counts too.
393
+ const short = overBudget(store.spend(), settings, 1);
394
+ const fallback = `${run.returned.length} of ${passes.length} passes returned`;
395
+ if (short) {
396
+ recordReviewCost();
397
+ return none('unavailable', `${fallback}, and the caps leave no run for one judge: ${short}`, { reviewers, scope, why, pr: pr?.number, spentUsd: spentUsd(), mode: { ...modeRecord, specialists: { ...specialists, fallback } } });
398
+ }
399
+ progress(`Rigour reviewer: ${fallback}; one judge reviews instead`);
400
+ modeRecord = { ...modeRecord, ran: 'single', specialists: { ...specialists, fallback } };
401
+ once = true;
307
402
  }
308
- answers[i] = answer;
309
403
  }
310
- const failed = answers.find(a => 'error' in a);
311
- if (failed && 'error' in failed)
312
- return none('unavailable', failed.error, { reviewers, scope, why, pr: pr?.number });
313
- parts = answers.map(a => a.verdict);
404
+ if (!parts) {
405
+ const answers = await Promise.all(reviewers.map(async (name) => {
406
+ const adapter = ADAPTERS[name];
407
+ const ask = async () => {
408
+ const run = await runJudge(name, prompt, modelFor(name));
409
+ progress(`Rigour reviewer: ${name} finished in ${Math.round((Date.now() - started) / 1000)}s (exit ${run.exitCode})`);
410
+ const answer = adapter.answer(run.stdout);
411
+ spent(answer.costUsd, projectedSingle); // every run counts against the caps, an answer or not
412
+ return { run, answer, verdict: run.exitCode === 0 || answer.text.trim() ? parseVerdict(answer.text, needsPriorPoints, name, answer) : undefined };
413
+ };
414
+ let first = await ask();
415
+ // No verdict, whether a malformed answer or a run that died, is a slip, not a decision: asked once more, inside the caps.
416
+ if (!once && (!first.verdict || 'error' in first.verdict) && !overBudget(store.spend(), settings, 1)) {
417
+ progress(`Rigour reviewer: ${name} gave no ${first.verdict ? 'valid verdict' : 'answer'}; asking once more`);
418
+ first = await ask();
419
+ }
420
+ return first.verdict ?? { error: `${name}: no answer (exit ${first.run.exitCode}): ${first.run.stderr.trim().slice(-200)}` };
421
+ }));
422
+ // A judge that gives nothing is replaced by the next one installed, so the boundary stays up: a review ends unavailable only when every judge failed.
423
+ const spare = candidates.filter(c => installed.has(c) && !reviewers.includes(c));
424
+ for (let i = 0; i < answers.length; i++) {
425
+ let answer = answers[i];
426
+ while (!once && 'error' in answer && spare.length && !overBudget(store.spend(), settings, 1)) {
427
+ const next = spare.shift();
428
+ progress(`Rigour reviewer: ${reviewers[i]} gave no verdict (${answer.error}); ${next} judges instead`);
429
+ modeRecord = { ...modeRecord, degraded: `${modeRecord.degraded ? `${modeRecord.degraded}; ` : ''}${reviewers[i]} gave no verdict, ${next} judged instead` };
430
+ reviewers[i] = next;
431
+ const run = await runJudge(next, prompt, modelFor(next));
432
+ const got = ADAPTERS[next].answer(run.stdout);
433
+ spent(got.costUsd, projectedSingle);
434
+ answer = run.exitCode === 0 || got.text.trim() ? parseVerdict(got.text, needsPriorPoints, next, got) : { error: `${next}: no answer (exit ${run.exitCode}): ${run.stderr.trim().slice(-200)}` };
435
+ }
436
+ answers[i] = answer;
437
+ }
438
+ const failed = answers.find(a => 'error' in a);
439
+ if (failed && 'error' in failed) {
440
+ if (once)
441
+ recordReviewCost();
442
+ return none('unavailable', failed.error, { reviewers, scope, why, pr: pr?.number, spentUsd: spentUsd() });
443
+ }
444
+ parts = answers.map(a => a.verdict);
445
+ }
314
446
  for (const part of parts) {
315
447
  if (part.trace)
316
448
  labelReads(part.trace, work, changedFiles);
@@ -320,7 +452,8 @@ async function review(cwd, base, config, exec, progress, options) {
320
452
  finally {
321
453
  clearInterval(ticker);
322
454
  }
323
- const merged = mergeVerdicts(parts); // one part too: every item is tagged with who found it
455
+ // One part too: every item is tagged with who found it. No part (nothing for a model to review): an empty verdict.
456
+ const merged = parts.length ? mergeVerdicts(parts) : { prior_points: [], redundant: [], reads: [], scans: [], merge_impact: [], findings: [], carried: [], resolved_previous: [], reviewers: [] };
324
457
  const touched = scope === 'delta' ? sincePrevious : new Set();
325
458
  let verdict = scope === 'delta' ? carryResolved(merged, store.readJson(previous.verdict), touched) : merged;
326
459
  if (modeRecord.ran === 'panel' && parts.length > 1) {
@@ -347,9 +480,10 @@ async function review(cwd, base, config, exec, progress, options) {
347
480
  },
348
481
  ask: async (judge, asked) => {
349
482
  const name = judge;
350
- const run = await runJudge(name, crossExamPrompt(repoRoot, head.slice(0, 9), diffFile, asked), settings.cross_models[name] ?? modelFor(name));
483
+ const examPrompt = crossExamPrompt(repoRoot, head.slice(0, 9), diffFile, asked);
484
+ const run = await runJudge(name, examPrompt, settings.cross_models[name] ?? modelFor(name));
351
485
  const answer = ADAPTERS[name].answer(run.stdout);
352
- store.addSpend(1, answer.costUsd);
486
+ spent(answer.costUsd, examPrompt.length);
353
487
  reserved--;
354
488
  cross.push({ reviewer: `${name} cross-exam`, ...(answer.costUsd !== undefined ? { cost_usd: answer.costUsd } : {}), ...(answer.tokens ? { tokens: answer.tokens } : {}) });
355
489
  progress(`Rigour reviewer: ${name} cross-examined ${asked.length} finding(s) (exit ${run.exitCode})`);
@@ -365,9 +499,10 @@ async function review(cwd, base, config, exec, progress, options) {
365
499
  store.writeJson(verdictFile, { ...verdict, inputs: { head, base: baseSha, scope, why, mode: modeRecord, reviewers, versions: reviewerVersions, authors: [...authors], fingerprint, human_reviews: reviews.count, reviews_before: options.reviewsBefore ?? null, since: previous?.head ?? null, at: new Date().toISOString() } });
366
500
  store.writeJson(openFile, accounted.open);
367
501
  store.writeJson(store.decidedPath(verdictFile), accounted);
502
+ recordReviewCost();
368
503
  if (branch !== 'HEAD')
369
504
  store.recordBranch(branch, { head, verdict: verdictFile, mode: scope, rulesHash, reviewsKey: reviews.key, inputsKey });
370
- return withRecord(accounted, verdict, false);
505
+ return { ...withRecord(accounted, verdict, false), spentUsd: spentUsd() };
371
506
  }
372
507
  finally {
373
508
  fs.rmSync(work, { recursive: true, force: true });
@@ -491,3 +626,12 @@ function result(accounted, verdict, reviewers, scope, why, cached, reviews, pr,
491
626
  ...(pr?.title ? { prTitle: pr.title } : {}),
492
627
  };
493
628
  }
629
+ /** Whether a tool call's target (a path, a glob, a shell command) names a repository file. */
630
+ function names(target, file) {
631
+ const at = target.indexOf(file);
632
+ if (at < 0)
633
+ return false;
634
+ const before = target[at - 1];
635
+ const after = target[at + file.length];
636
+ return (before === undefined || /[\s/'"=]/.test(before)) && (after === undefined || /[\s'":)]/.test(after));
637
+ }
@@ -34,6 +34,7 @@ export interface RigourSettings {
34
34
  reviewer?: UserReviewerSettings;
35
35
  goal?: boolean;
36
36
  outcomes?: boolean;
37
+ orchestrator?: boolean;
37
38
  cursor?: {
38
39
  apiKey?: string;
39
40
  };