@rigour-labs/core 6.7.10 → 6.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/dist/brief/briefing.d.ts +59 -0
  2. package/dist/brief/briefing.js +126 -0
  3. package/dist/brief/briefing.test.d.ts +1 -0
  4. package/dist/brief/briefing.test.js +107 -0
  5. package/dist/index.d.ts +2 -0
  6. package/dist/index.js +2 -0
  7. package/dist/review/backtest-init.d.ts +8 -5
  8. package/dist/review/backtest-init.js +27 -9
  9. package/dist/review/backtest-init.test.js +39 -1
  10. package/dist/review/backtest-last.js +3 -1
  11. package/dist/review/reviewer/adapters.d.ts +4 -0
  12. package/dist/review/reviewer/adapters.js +23 -0
  13. package/dist/review/reviewer/adapters.test.js +40 -0
  14. package/dist/review/reviewer/inputs.d.ts +4 -1
  15. package/dist/review/reviewer/inputs.js +36 -2
  16. package/dist/review/reviewer/inputs.test.js +26 -0
  17. package/dist/review/reviewer/record.d.ts +5 -0
  18. package/dist/review/reviewer/record.js +3 -3
  19. package/dist/review/reviewer/record.test.js +8 -0
  20. package/dist/review/reviewer/verdict.d.ts +14 -0
  21. package/dist/review/reviewer/verdict.js +46 -6
  22. package/dist/review/reviewer.d.ts +2 -0
  23. package/dist/review/reviewer.js +24 -5
  24. package/dist/review/reviewer.test.js +93 -6
  25. package/dist/review-learning/repo-rules.d.ts +14 -2
  26. package/dist/review-learning/repo-rules.js +59 -9
  27. package/dist/review-learning/repo-rules.test.js +39 -1
  28. package/dist/task/thread.d.ts +47 -0
  29. package/dist/task/thread.js +234 -0
  30. package/dist/task/thread.test.d.ts +1 -0
  31. package/dist/task/thread.test.js +133 -0
  32. package/dist/templates/universal-config.js +4 -0
  33. package/dist/types/index.d.ts +21 -0
  34. package/dist/types/index.js +7 -0
  35. package/package.json +6 -6
@@ -13,7 +13,7 @@ import { GH_TIMEOUT_MS, parseJsonArrays } from './exec.js';
13
13
  export function ghFor(cwd, exec, env) {
14
14
  return args => exec('gh', args, { cwd, timeoutMs: GH_TIMEOUT_MS, env });
15
15
  }
16
- const PR_FIELDS = 'number,state,isDraft,author,body';
16
+ const PR_FIELDS = 'number,state,isDraft,author,body,title';
17
17
  /**
18
18
  * The description as it read at `at` (a backtest's review time): GitHub keeps every version in
19
19
  * `userContentEdits`, the first being the text at creation. Undefined when it cannot be read, so a
@@ -67,7 +67,7 @@ async function viewPullRequest(gh, selector) {
67
67
  const parsed = JSON.parse(result.stdout);
68
68
  if (!Number.isInteger(parsed.number))
69
69
  return { error: `gh returned no pull request number for ${selector}` };
70
- return { pr: { number: parsed.number, state: String(parsed.state ?? '').toLowerCase(), draft: !!parsed.isDraft, author: String(parsed.author?.login ?? ''), body: String(parsed.body ?? '') } };
70
+ return { pr: { number: parsed.number, state: String(parsed.state ?? '').toLowerCase(), draft: !!parsed.isDraft, author: String(parsed.author?.login ?? ''), body: String(parsed.body ?? ''), ...(parsed.title ? { title: String(parsed.title) } : {}) } };
71
71
  }
72
72
  catch {
73
73
  return { error: `gh returned something other than a pull request for ${selector}: ${result.stdout.slice(0, 120)}` };
@@ -104,10 +104,44 @@ export async function humanReviews(gh, pr, reviewsBefore) {
104
104
  key: [...rounds.map((r) => `${r.id},${r.submitted_at},${r.commit_id},${r.state}`), ...inline.map((c) => `${c.id},${c.updated_at}`)].join('|'),
105
105
  count,
106
106
  approvals,
107
+ labels: rounds.filter(worded).flatMap((r) => severityLabels(String(r.body), String(r.user.login), String(r.submitted_at))),
107
108
  ...(latest ? { label: `${latest.user.login}, ${latest.submitted_at} (${count} review${count === 1 ? '' : 's'})` } : {}),
108
109
  },
109
110
  };
110
111
  }
112
+ /** The severity a heading names: "Blocking", "Should fix", "Nits" and their usual spellings, alone on the line. */
113
+ function headingSeverity(line) {
114
+ const bare = line.replace(/[#*_:`>]/g, ' ').replace(/\(\d+\)/g, ' ').replace(/\s+/g, ' ').trim().toLowerCase();
115
+ if (!bare || bare.split(' ').length > 3)
116
+ return undefined;
117
+ if (/^(blocking|blockers?|must fix|must-fix)$/.test(bare))
118
+ return 'blocking';
119
+ if (/^(should fix|should-fix|should)$/.test(bare))
120
+ return 'should-fix';
121
+ if (/^(nits?|non blocking|non-blocking|minor|optional)$/.test(bare))
122
+ return 'non-blocking';
123
+ return undefined;
124
+ }
125
+ /** Each bullet or numbered line under a severity heading of the review, with that heading's severity. */
126
+ function severityLabels(body, login, at) {
127
+ const out = [];
128
+ let severity;
129
+ for (const line of body.split('\n')) {
130
+ const heading = headingSeverity(line);
131
+ if (heading) {
132
+ severity = heading;
133
+ continue;
134
+ }
135
+ if (/^\s{0,3}#{1,6}\s/.test(line))
136
+ severity = undefined; // another heading ends the section
137
+ // A bullet, a numbered line, or a numbered line set in bold or underline ("**1. The column is NOT NULL.**"), which is
138
+ // how many reviewers title a point before its evidence.
139
+ const item = /^\s*(?:\*\*|__)?\s*(?:[-*+]|\d+[.)])\s+(.+?)\s*(?:\*\*|__)?\s*$/.exec(line);
140
+ if (severity && item)
141
+ out.push({ login, at, severity, text: item[1].replace(/\*\*|__/g, '').trim() });
142
+ }
143
+ return out;
144
+ }
111
145
  /** The repository's rules as the reviewer must read them. */
112
146
  export function rulesText(cwd) {
113
147
  return ['AGENTS.md', 'CLAUDE.md'].map(name => {
@@ -23,4 +23,30 @@ describe('the human reviews a judge reads', () => {
23
23
  const later = await humanReviews(gh, pr, '2026-09-28T15:12:54.000Z');
24
24
  expect(later.reviews.approvals).toHaveLength(1);
25
25
  });
26
+ it('reads the bullets and numbered lines under Blocking / Should fix / Nits headings as that reviewer labelled them, and nothing under another heading', async () => {
27
+ const body = ['Thanks, a few things.', '', '## Blocking', '- The kill switch is read after every query: check it first.', '',
28
+ '**Should fix**', '1. The comment on the window still says daily.', '', 'Nits:', '* Rename tmp to rows.', '', '## Context', '- not a label'].join('\n');
29
+ const labelled = async (args) => ({ exitCode: 0, stdout: JSON.stringify(args[1].endsWith('/reviews') ? [{ ...reviews[0], body }] : []), stderr: '' });
30
+ const { reviews: read } = await humanReviews(labelled, pr, undefined);
31
+ expect(read.labels.map(l => [l.login, l.severity, l.text])).toEqual([
32
+ ['senior', 'blocking', 'The kill switch is read after every query: check it first.'],
33
+ ['senior', 'should-fix', 'The comment on the window still says daily.'],
34
+ ['senior', 'non-blocking', 'Rename tmp to rows.'],
35
+ ]);
36
+ expect(read.labels.every(l => l.at === '2026-09-25T18:09:11Z')).toBe(true);
37
+ });
38
+ it('reads a numbered point set in bold, with its evidence sub-bullets, under each heading', async () => {
39
+ const body = ['## Blocking', '', '**1. The status column is typed nullable but the table declares it NOT NULL.**', '- the reader falls back to an empty string', '- two callers check for null again', '',
40
+ '**2. The retry posts the event twice when the first post times out.**', '', '## Should fix', '', '__1. The window comment says hourly; the job runs daily.__', '', '## Nits', '', '- Rename tmp to rows.'].join('\n');
41
+ const shaped = async (args) => ({ exitCode: 0, stdout: JSON.stringify(args[1].endsWith('/reviews') ? [{ ...reviews[0], body }] : []), stderr: '' });
42
+ const { reviews: read } = await humanReviews(shaped, pr, undefined);
43
+ expect(read.labels.map(l => [l.severity, l.text])).toEqual([
44
+ ['blocking', 'The status column is typed nullable but the table declares it NOT NULL.'],
45
+ ['blocking', 'the reader falls back to an empty string'],
46
+ ['blocking', 'two callers check for null again'],
47
+ ['blocking', 'The retry posts the event twice when the first post times out.'],
48
+ ['should-fix', 'The window comment says hourly; the job runs daily.'],
49
+ ['non-blocking', 'Rename tmp to rows.'],
50
+ ]);
51
+ });
26
52
  });
@@ -5,12 +5,14 @@ export interface ReviewRecord {
5
5
  base: string;
6
6
  scope: 'full' | 'delta';
7
7
  at: string;
8
+ /** `outside_repo`: what the judge also read from the machine's own config (a person's instructions); absent when it read only the repository and Rigour's inputs. */
8
9
  judges: Array<{
9
10
  reviewer: string;
10
11
  version?: string;
11
12
  model?: string;
12
13
  cost_usd?: number;
13
14
  turns?: number;
15
+ outside_repo?: string;
14
16
  }>;
15
17
  /** Checked by Rigour against the checkout. */
16
18
  verified: {
@@ -28,10 +30,13 @@ export interface ReviewRecord {
28
30
  served: number;
29
31
  applied: number;
30
32
  };
33
+ /** `labelled`: points that took the review's own severity heading; `relabelled`: of those, the ones the judge had read otherwise. */
31
34
  prior_points: {
32
35
  open: number;
33
36
  resolved: number;
34
37
  answer_in_reply: number;
38
+ labelled?: number;
39
+ relabelled?: number;
35
40
  };
36
41
  unverified: number;
37
42
  notes: number;
@@ -21,7 +21,7 @@ export function buildRecord(input) {
21
21
  should_fix: input.accounted.advisory,
22
22
  rules: { served: rules.length, followed: rules.filter(r => r.status === 'followed').length, broken: rules.filter(r => r.status === 'broken').length, not_applicable: rules.filter(r => r.status === 'not-applicable').length },
23
23
  lessons: { served: input.lessonsServed, applied: lessons.filter(l => l.applies === true).length },
24
- prior_points: { open: input.accounted.open.filter(i => i.kind === 'prior').length, resolved: input.accounted.resolved.length, answer_in_reply: input.accounted.answerInReply.length },
24
+ prior_points: { open: input.accounted.open.filter(i => i.kind === 'prior').length, resolved: input.accounted.resolved.length, answer_in_reply: input.accounted.answerInReply.length, ...(input.accounted.labels?.served ? { labelled: input.accounted.labels.taken, relabelled: input.accounted.labels.disagreed } : {}) },
25
25
  unverified: input.accounted.unverified.length,
26
26
  notes: input.accounted.notes.length,
27
27
  disputed: input.accounted.disputed.length,
@@ -54,7 +54,7 @@ function canonical(value) {
54
54
  export function recordLines(r, shouldFixShown = 5) {
55
55
  const where = (i) => `${i.file ? `\`${i.file}${i.line ? `:${i.line}` : ''}\` ` : ''}${i.issue}${i.locations?.length ? ` (also ${i.locations.map(l => `\`${l.file}${l.line ? `:${l.line}` : ''}\``).join(', ')})` : ''}`;
56
56
  const v = r.verified;
57
- const lines = [`**Review record** · ${v.blocking.length} blocking · ${v.should_fix.length} should-fix · rules ${v.rules.followed} followed, ${v.rules.broken} broken, ${v.rules.not_applicable} not applicable of ${v.rules.served} · lessons ${v.lessons.applied} of ${v.lessons.served} apply · prior points ${v.prior_points.open} open, ${v.prior_points.resolved} resolved`];
57
+ const lines = [`**Review record** · ${v.blocking.length} blocking · ${v.should_fix.length} should-fix · rules ${v.rules.followed} followed, ${v.rules.broken} broken, ${v.rules.not_applicable} not applicable of ${v.rules.served} · lessons ${v.lessons.applied} of ${v.lessons.served} apply · prior points ${v.prior_points.open} open, ${v.prior_points.resolved} resolved${v.prior_points.labelled !== undefined ? `, ${v.prior_points.labelled} by the review's own label (${v.prior_points.relabelled} relabelled)` : ''}`];
58
58
  for (const i of v.blocking)
59
59
  lines.push(`- **Blocking** ${where(i)}`);
60
60
  for (const i of v.should_fix.slice(0, shouldFixShown))
@@ -64,6 +64,6 @@ export function recordLines(r, shouldFixShown = 5) {
64
64
  const folded = [[v.notes, 'working note'], [v.disputed, 'disputed'], [v.unverified, 'unverified'], [r.people.dismissed, 'dismissed']].filter(([n]) => n > 0);
65
65
  if (folded.length)
66
66
  lines.push(`Also seen, never blocking: ${folded.map(([n, w]) => `${n} ${w}${n === 1 || w === 'disputed' || w === 'unverified' || w === 'dismissed' ? '' : 's'}`).join(', ')}.`);
67
- lines.push(`Judged by ${r.judges.map(j => `${j.reviewer}${j.version ? ` ${j.version}` : ''}${j.model ? ` (${j.model})` : ''}${typeof j.cost_usd === 'number' ? ` $${j.cost_usd.toFixed(2)}` : ''}`).join(', ') || 'no judge'} on \`${r.head.slice(0, 9)}\` against \`${r.base.slice(0, 9)}\` (${r.scope}); ${r.reported.human_reviews} human review(s) seen. Integrity \`${r.integrity.slice(0, 16)}\`.`);
67
+ lines.push(`Judged by ${r.judges.map(j => `${j.reviewer}${j.version ? ` ${j.version}` : ''}${j.model ? ` (${j.model})` : ''}${typeof j.cost_usd === 'number' ? ` $${j.cost_usd.toFixed(2)}` : ''}${j.outside_repo ? ` [${j.outside_repo}]` : ''}`).join(', ') || 'no judge'} on \`${r.head.slice(0, 9)}\` against \`${r.base.slice(0, 9)}\` (${r.scope}); ${r.reported.human_reviews} human review(s) seen. Integrity \`${r.integrity.slice(0, 16)}\`.`);
68
68
  return lines;
69
69
  }
@@ -29,4 +29,12 @@ describe('the review record', () => {
29
29
  expect(lines.at(-2)).toBe('Also seen, never blocking: 2 working notes, 1 unverified, 1 dismissed.');
30
30
  expect(lines.at(-1)).toMatch(/^Judged by claude 2\.1\.0 \(opus\) \$1\.25 on `abcdef012` against `012345678` \(full\); 2 human review\(s\) seen\. Integrity `[0-9a-f]{16}`\.$/);
31
31
  });
32
+ it("counts the prior points that took the review's own severity label, and only when the reviews carry labels", () => {
33
+ const labelled = input();
34
+ labelled.accounted = { ...labelled.accounted, labels: { served: 4, taken: 2, disagreed: 1 } };
35
+ const record = buildRecord(labelled);
36
+ expect(record.verified.prior_points).toMatchObject({ labelled: 2, relabelled: 1 });
37
+ expect(recordLines(record).join('\n')).toContain("2 by the review's own label (1 relabelled)");
38
+ expect(buildRecord(input()).verified.prior_points).not.toHaveProperty('labelled');
39
+ });
32
40
  });
@@ -27,6 +27,14 @@ export interface PriorChecks {
27
27
  approvals: Approval[];
28
28
  inCheckout: (text: string) => string | undefined;
29
29
  changed?: ChangedLines;
30
+ labels?: LabelledPoint[];
31
+ }
32
+ /** A point under a severity heading the reviewer wrote ("Blocking", "Should fix", "Nits"): the reviewer's own label. */
33
+ export interface LabelledPoint {
34
+ login: string;
35
+ at: string;
36
+ severity: NonNullable<PriorPoint['severity']>;
37
+ text: string;
30
38
  }
31
39
  /** The lines a change touched, per file: every line added, and the place of every deletion, in the new file's numbering. */
32
40
  export type ChangedLines = Map<string, Set<number>>;
@@ -252,6 +260,12 @@ export interface Accounting {
252
260
  notes: OpenItem[];
253
261
  /** Should-fixes the judge could show (a verified quote): worth a person's time, never a block. */
254
262
  advisory: OpenItem[];
263
+ /** The reviews' own severity labels: how many there were, how many prior points took one, and how many the judge read otherwise. */
264
+ labels?: {
265
+ served: number;
266
+ taken: number;
267
+ disagreed: number;
268
+ };
255
269
  }
256
270
  /** Everything the reviewers reported blocks; in delta mode, previous open items carry unless resolved with evidence. */
257
271
  export declare function account(verdict: Verdict, previousOpen: OpenItem[] | undefined, verify: Verify, prior?: PriorChecks): Accounting;
@@ -12,6 +12,34 @@ import { createHash } from 'crypto';
12
12
  import fs from 'fs';
13
13
  import path from 'path';
14
14
  import { textSimilarity } from './consensus.js';
15
+ /** How alike a judge's prior point and a labelled line of the review must read to take the reviewer's label. */
16
+ const LABEL_SIMILARITY = 0.4;
17
+ /**
18
+ * The point with the severity its reviewer wrote, when the review labels it; the judge's reading otherwise. The reviewer
19
+ * is matched by login; the review by the date the judge names (judges write a timestamp, a date, or reformat it), and
20
+ * when no review of that reviewer has that date, by that reviewer's latest labelled review. A disagreement is said.
21
+ */
22
+ function labelled(p, labels) {
23
+ const [login, ...rest] = (p.review ?? '').trim().split(/\s+/);
24
+ const day = /\d{4}-\d{2}-\d{2}/.exec(rest.join(' '))?.[0];
25
+ const theirs = login ? labels.filter(l => l.login === login) : [];
26
+ const sameDay = day ? theirs.filter(l => l.at.startsWith(day)) : [];
27
+ const latest = theirs.reduce((max, l) => (l.at > max ? l.at : max), '');
28
+ const candidates = sameDay.length ? sameDay : theirs.filter(l => l.at === latest);
29
+ const asItem = (issue) => ({ id: '', kind: 'prior', class: 'prior point', issue });
30
+ let best;
31
+ for (const label of candidates) {
32
+ const score = textSimilarity(asItem(p.point), asItem(label.text));
33
+ if (score >= LABEL_SIMILARITY && (!best || score > best.score))
34
+ best = { label, score };
35
+ }
36
+ if (!best)
37
+ return { point: p, took: false, disagreed: false };
38
+ if (best.label.severity === p.severity)
39
+ return { point: p, took: true, disagreed: false };
40
+ const note = `the review labels it ${best.label.severity}${p.severity ? `; the judge read ${p.severity}` : ''}`;
41
+ return { point: { ...p, severity: best.label.severity, evidence: p.evidence ? `${p.evidence}; ${note}` : note }, took: true, disagreed: true };
42
+ }
15
43
  const NO_PRIOR_CHECKS = { approvals: [], inCheckout: () => undefined };
16
44
  /** The touched lines of a unified diff (`git diff base...head`). */
17
45
  export function changedLinesOf(diff) {
@@ -214,7 +242,10 @@ export function account(verdict, previousOpen, verify, prior = NO_PRIOR_CHECKS)
214
242
  const notes = [];
215
243
  const advisory = [];
216
244
  const seen = new Set();
217
- const accepted = verdict.prior_points.filter(p => p.severity === 'non-blocking');
245
+ // The reviewer's own label wins over the judge's reading of it: a judge that calls a blocker a should-fix would demote it silently.
246
+ const read = verdict.prior_points.map(p => labelled(p, prior.labels ?? []));
247
+ const points = read.map(r => r.point);
248
+ const accepted = points.filter(p => p.severity === 'non-blocking');
218
249
  // A should-fix is shown only when the judge could show it: a quote Rigour finds. One that cannot be checked is not a claim worth a person's time.
219
250
  const advise = (item) => {
220
251
  if (seen.has(item.id))
@@ -243,7 +274,7 @@ export function account(verdict, previousOpen, verify, prior = NO_PRIOR_CHECKS)
243
274
  open.push(item);
244
275
  };
245
276
  const answerInReply = [];
246
- for (const p of verdict.prior_points) {
277
+ for (const p of points) {
247
278
  if (p.resolved)
248
279
  continue;
249
280
  if (p.severity === 'non-blocking') {
@@ -270,7 +301,11 @@ export function account(verdict, previousOpen, verify, prior = NO_PRIOR_CHECKS)
270
301
  continue;
271
302
  }
272
303
  // Still open only where the judge quotes the code that keeps it open: a point a later commit already fixed never blocks.
273
- add(item);
304
+ // A should-fix keeps the tier its reviewer gave it: shown with its quote, never a block, like a should-fix finding.
305
+ if (p.severity === 'should-fix')
306
+ advise(item);
307
+ else
308
+ add(item);
274
309
  }
275
310
  for (const r of verdict.redundant) {
276
311
  if (r.removed === false)
@@ -374,7 +409,7 @@ export function account(verdict, previousOpen, verify, prior = NO_PRIOR_CHECKS)
374
409
  }
375
410
  }
376
411
  }
377
- return { open: onePerRootCause(open), unverified, resolved, answerInReply, notes, advisory: onePerRootCause(advisory) };
412
+ return { open: onePerRootCause(open), unverified, resolved, answerInReply, notes, advisory: onePerRootCause(advisory), labels: { served: prior.labels?.length ?? 0, taken: read.filter(r => r.took).length, disagreed: read.filter(r => r.disagreed).length } };
378
413
  }
379
414
  /** How alike two items' words must be to be the same point made in two places; and, on the same lines, to be one point said two ways. */
380
415
  const SAME_POINT = 0.6;
@@ -387,9 +422,14 @@ const SAME_LINES = 3;
387
422
  function onePerRootCause(items) {
388
423
  const kept = [];
389
424
  for (const item of items) {
390
- // The same class in the same words anywhere, or any two non-human items on the same lines that read alike (a rule break and the finding it caused).
425
+ // The same class in the same words anywhere, or any two non-human items on the same lines that read alike (a rule break and
426
+ // the finding it caused). A human's point owns its lines: a rule break or finding there in like words is that point found
427
+ // again and folds into it; the human's words stay. One in other words is a different problem and stays its own item, so
428
+ // fixing the human's point does not leave it for the next round. Two human points never merge.
391
429
  const nearby = (k) => !!k.file && k.file === item.file && k.line !== undefined && item.line !== undefined && Math.abs(k.line - item.line) <= SAME_LINES;
392
- const same = item.kind === 'prior' ? undefined : kept.find(k => k.kind !== 'prior' && ((k.class === item.class && textSimilarity(k, item) >= SAME_POINT) || (nearby(k) && textSimilarity(k, item) >= SAME_PLACE)));
430
+ const same = item.kind === 'prior' ? undefined : kept.find(k => k.kind === 'prior'
431
+ ? nearby(k) && textSimilarity(k, item) >= SAME_PLACE
432
+ : (k.class === item.class && textSimilarity(k, item) >= SAME_POINT) || (nearby(k) && textSimilarity(k, item) >= SAME_PLACE));
393
433
  if (!same) {
394
434
  kept.push(item);
395
435
  continue;
@@ -84,6 +84,8 @@ export interface ReviewerResult {
84
84
  /** The latest human review it worked from (`<login>, <date> (<n> reviews)`). */
85
85
  previousReview?: string;
86
86
  pr?: number;
87
+ /** The pull request's title, when one was read. */
88
+ prTitle?: string;
87
89
  }
88
90
  export interface ModeRecord {
89
91
  asked: 'single' | 'cross' | 'full' | 'panel';
@@ -36,6 +36,7 @@ import { buildContext, dismissedAs, readReviewDismissals, relatedDocs } from './
36
36
  import { buildRecord } from './reviewer/record.js';
37
37
  import { account, attachServedRules, changedLinesOf, checkoutSearch, checkoutVerifier, carryResolved, evidenceTouched, mergeVerdicts, parseVerdict } from './reviewer/verdict.js';
38
38
  import { judgeUnset } from './reviewer/judge-env.js';
39
+ import { appendTaskEvent } from '../task/thread.js';
39
40
  export { defaultExec, githubEnv, githubToken, parseJsonArrays } from './reviewer/exec.js';
40
41
  export { itemLine } from './reviewer/verdict.js';
41
42
  /** Findings and no verdict stop a push; a skipped review does not, and is reported as owed. */
@@ -48,7 +49,16 @@ const MAX_DELTA_LINES = 400;
48
49
  /** The reviewer, and one anonymous usage event for what it did (only when the person opted in to telemetry). */
49
50
  export async function runReviewer(cwd, base, config, exec = defaultExec, progress = message => process.stderr.write(`${message}\n`), options = {}) {
50
51
  const result = await review(cwd, base, config, exec, progress, options);
51
- await trackUsage('reviewer_completed', reviewerUsage(result, options.trigger ?? (options.reviewsBefore ? 'backtest' : 'review')));
52
+ const trigger = options.trigger ?? (options.reviewsBefore ? 'backtest' : 'review');
53
+ await trackUsage('reviewer_completed', reviewerUsage(result, trigger));
54
+ // On the task's thread, unless it replays history: a backtest's worktree is not anyone's task.
55
+ if (trigger !== 'backtest' && result.outcome !== 'skipped')
56
+ appendTaskEvent(cwd, {
57
+ kind: 'review', trigger, outcome: result.outcome, blocking: result.items.length, should_fix: result.advisory.length,
58
+ ...(result.pr ? { pr: result.pr } : {}),
59
+ ...(result.prTitle ? { pr_title: result.prTitle } : {}),
60
+ ...(result.record ? { integrity: result.record.integrity, cost_usd: result.record.judges.reduce((sum, j) => sum + (j.cost_usd ?? 0), 0), judges: result.record.judges.map(j => j.reviewer) } : {}),
61
+ });
52
62
  return result;
53
63
  }
54
64
  async function review(cwd, base, config, exec, progress, options) {
@@ -107,7 +117,7 @@ async function review(cwd, base, config, exec, progress, options) {
107
117
  if (skip)
108
118
  return none('skipped', skip, { reviewers, pr: pr?.number });
109
119
  }
110
- let reviews = { markdown: 'none\n', key: '', count: 0, approvals: [] };
120
+ let reviews = { markdown: 'none\n', key: '', count: 0, approvals: [], labels: [] };
111
121
  if (gh && pr) {
112
122
  const read = await humanReviews(gh, pr, options.reviewsBefore);
113
123
  if (read.error)
@@ -187,12 +197,17 @@ async function review(cwd, base, config, exec, progress, options) {
187
197
  const openFile = store.openPath(verdictFile);
188
198
  const previousOpen = scope === 'delta' ? store.readJson(store.openPath(previous.verdict)) ?? [] : undefined;
189
199
  const verify = checkoutVerifier(cwd);
190
- const prior = { approvals: reviews.approvals, inCheckout: checkoutSearch(cwd), changed: changedLinesOf(fullDiff) };
200
+ const prior = { approvals: reviews.approvals, inCheckout: checkoutSearch(cwd), changed: changedLinesOf(fullDiff), labels: reviews.labels };
191
201
  const modelFor = (name) => settings.models[name] ?? (name === 'claude' ? settings.model : undefined);
192
202
  // The record of the review, written beside the verdict once and rebuilt from the same verdict on a cached read.
203
+ // What a judge reads besides the repository and Rigour's inputs, on this machine: on the record, so "the same judge" is a claim it can check.
204
+ const outsideOf = (reviewer) => {
205
+ const outside = ADAPTERS[reviewer]?.outsideRepo?.(os.homedir(), installed.get(reviewer)?.version);
206
+ return outside ? { outside_repo: outside } : {};
207
+ };
193
208
  const withRecord = (accounted, verdict, cached) => {
194
209
  const res = result(accounted, verdict, reviewers, scope, why, cached, reviews, pr, modeRecord, settings.dismissals);
195
- const judges = (verdict.reviewers ?? []).map(r => ({ reviewer: r.reviewer, ...(installed.get(r.reviewer)?.version ? { version: installed.get(r.reviewer).version } : {}), ...(modelFor(r.reviewer) ? { model: modelFor(r.reviewer) } : {}), ...(typeof r.cost_usd === 'number' ? { cost_usd: r.cost_usd } : {}), ...(r.trace?.turns ? { turns: r.trace.turns } : {}) }));
210
+ const judges = (verdict.reviewers ?? []).map(r => ({ reviewer: r.reviewer, ...(installed.get(r.reviewer)?.version ? { version: installed.get(r.reviewer).version } : {}), ...(modelFor(r.reviewer) ? { model: modelFor(r.reviewer) } : {}), ...(typeof r.cost_usd === 'number' ? { cost_usd: r.cost_usd } : {}), ...(r.trace?.turns ? { turns: r.trace.turns } : {}), ...outsideOf(r.reviewer) }));
196
211
  const recordPath = store.recordPath(verdictFile);
197
212
  const record = (cached && store.readJson(recordPath)) || buildRecord({ head, base: baseSha, scope, verdict, accounted, judges, lessonsServed: context.lessons, humanReviews: reviews.count });
198
213
  if (!cached || !fs.existsSync(recordPath))
@@ -212,7 +227,7 @@ async function review(cwd, base, config, exec, progress, options) {
212
227
  let inlineInputs = [];
213
228
  const runJudge = (name, prompt, model) => name === 'api'
214
229
  ? runApiJudge(prompt, { url: settings.api.url, model: settings.api.model, key: process.env[settings.api.key_env] ?? '', maxTurns: settings.api.max_turns, timeoutMs: settings.timeout_ms, cwd, roots: [cwd, work], inputs: inlineInputs, ...(settings.reasoning[name] ? { reasoning: settings.reasoning[name] } : {}), ...(options.fetch ? { fetchImpl: options.fetch } : {}) })
215
- : exec(installed.get(name).binary, ADAPTERS[name].args(prompt, model, { reasoning: settings.reasoning[name] }), { cwd, timeoutMs: settings.timeout_ms, unset: judgeUnset(name, settings.judge_env) });
230
+ : exec(installed.get(name).binary, ADAPTERS[name].args(prompt, model, { reasoning: settings.reasoning[name] }), { cwd, timeoutMs: settings.timeout_ms, unset: judgeUnset(name, settings.judge_env), ...(ADAPTERS[name].env ? { env: ADAPTERS[name].env } : {}) });
216
231
  try {
217
232
  const file = (name, text) => {
218
233
  const target = path.join(work, name);
@@ -336,6 +351,9 @@ async function review(cwd, base, config, exec, progress, options) {
336
351
  verdict = { ...verdict, panel: { judgeItemIds: judgeItems.flat().map(item => item.id), items }, reviewers: [...(verdict.reviewers ?? []), ...cross] };
337
352
  }
338
353
  const accounted = decide(verdict, previousOpen, verify, prior, dismissals);
354
+ // Said every run where the reviews carry labels, so a matcher that takes none is visible rather than silent.
355
+ if (accounted.labels?.served)
356
+ progress(`Rigour reviewer: ${accounted.labels.taken} of ${verdict.prior_points.length} prior point(s) took the review's own severity label (${accounted.labels.served} labelled line(s)); the judge read ${accounted.labels.disagreed} otherwise`);
339
357
  store.writeJson(verdictFile, { ...verdict, inputs: { head, base: baseSha, scope, why, mode: modeRecord, reviewers, versions: reviewerVersions, authors: [...authors], fingerprint, human_reviews: reviews.count, reviews_before: options.reviewsBefore ?? null, since: previous?.head ?? null, at: new Date().toISOString() } });
340
358
  store.writeJson(openFile, accounted.open);
341
359
  store.writeJson(store.decidedPath(verdictFile), accounted);
@@ -462,5 +480,6 @@ function result(accounted, verdict, reviewers, scope, why, cached, reviews, pr,
462
480
  cached,
463
481
  ...(reviews.label ? { previousReview: reviews.label } : {}),
464
482
  ...(pr ? { pr: pr.number } : {}),
483
+ ...(pr?.title ? { prTitle: pr.title } : {}),
465
484
  };
466
485
  }
@@ -9,7 +9,8 @@ import { dismissReviewerFinding } from './reviewer/context.js';
9
9
  import { reviewStatus } from './reviewer/background.js';
10
10
  import { selectReviewers, vendorsOf } from './reviewer/adapters.js';
11
11
  import { account, attachServedRules, carryResolved, changedLinesOf, checkoutSearch, checkoutVerifier, mergeVerdicts, parseVerdict } from './reviewer/verdict.js';
12
- import { recordIntact } from './reviewer/record.js';
12
+ import { recordIntact, recordLines } from './reviewer/record.js';
13
+ import { readThread } from '../task/thread.js';
13
14
  let repo;
14
15
  const config = ConfigSchema.parse({ version: 1, review: { github_account: 'reviewer-account', reviewer: { enabled: true, reviewers: ['claude', 'cursor'] } } });
15
16
  const git = (...args) => execFileSync('git', ['-C', repo, ...args], { encoding: 'utf8' }).trim();
@@ -51,6 +52,7 @@ function fakes(answer, seen, pr = PR) {
51
52
  seen.ran.push(command);
52
53
  (seen.args ??= []).push(args);
53
54
  (seen.unset ??= []).push(options.unset);
55
+ (seen.env ??= []).push(options.env);
54
56
  const name = binary === 'claude' ? 'claude' : binary === 'cursor-agent' ? 'cursor' : 'codex';
55
57
  const prompt = binary === 'claude' ? args[args.indexOf('-p') + 1] : args[args.length - 1];
56
58
  seen.prompts.push(prompt);
@@ -447,18 +449,32 @@ describe('verdicts', () => {
447
449
  expect(unknown.verdict.rules).toEqual([]); // an answer naming no served rule is dropped, whatever it claims
448
450
  expect(unknown.open).toEqual([]);
449
451
  });
450
- it('shows the same point found in several places as one item with every location, and never merges human points', () => {
452
+ it("shows the same point found in several places as one item with every location; a judge item on a human point's lines folds into it, and human points never merge", () => {
451
453
  const scan = (file, line, quote) => ({ class: 'production-cost', file, line, issue: `the ${file.split('/').pop()} scan has no upper bound on updated_at`, input: 'a week of rows', consequence: 'rows read grow with time', quote });
452
454
  const verdict = { ...EMPTY, prior_points: [
453
- { point: 'bound the window', severity: 'blocking', resolved: false, file: 'src/job.ts', line: 1, quote: 'export function job() {' },
455
+ { point: 'the scan has no upper bound on updated_at', severity: 'blocking', resolved: false, file: 'src/job.ts', line: 1, quote: 'export function job() {' },
454
456
  { point: 'bound the window again', severity: 'blocking', resolved: false, file: 'src/job.ts', line: 1, quote: 'export function job() {' },
455
457
  ], findings: [scan('src/job.ts', 1, 'export function job() {'), scan('a.ts', 1, 'export const a = 1;'), { ...scan('src/job.ts', 2, 'return 1;'), class: 'correctness' }] };
456
458
  const { open } = account(verdict, undefined, checkoutVerifier(repo));
457
459
  expect(open.map(i => [i.kind, i.class, i.locations ?? []])).toEqual([
458
- ['prior', 'prior point', []], ['prior', 'prior point', []],
459
- // The same point in another file, and the same point said as another class on the next line: one item, every place.
460
- ['finding', 'production-cost', [{ file: 'a.ts', line: 1 }, { file: 'src/job.ts', line: 2 }]],
460
+ // The judge's scan on the human's lines, in like words, is the human's point found again, and so is the same point said as another class on the next line.
461
+ // The same scan in another file is not on the human's lines: its own item.
462
+ ['prior', 'prior point', [{ file: 'src/job.ts', line: 1 }, { file: 'src/job.ts', line: 2 }]], ['prior', 'prior point', []], ['finding', 'production-cost', []],
461
463
  ]);
464
+ // On other lines than any human point: the same point in another file, and said as another class on the next line, is one item, every place.
465
+ const apart = account({ ...verdict, prior_points: [] }, undefined, checkoutVerifier(repo));
466
+ expect(apart.open.map(i => [i.kind, i.class, i.locations ?? []])).toEqual([['finding', 'production-cost', [{ file: 'a.ts', line: 1 }, { file: 'src/job.ts', line: 2 }]]]);
467
+ // On a human point's lines: a rule break in like words is that point found again; a different rule, or a finding in other
468
+ // words, stays its own item, so fixing the human's point does not leave it for the next round.
469
+ const human = { point: 'null guards on columns the query makes non-null are dead fallbacks', severity: 'blocking', resolved: false, file: 'src/job.ts', line: 2, quote: 'return 1;' };
470
+ const onHuman = account({ ...EMPTY, prior_points: [human],
471
+ findings: [{ class: 'correctness', file: 'src/job.ts', line: 3, issue: 'the retry re-sends the email', input: 'a timeout', consequence: 'two emails', quote: '}' }],
472
+ rules: [
473
+ { id: 'r1', status: 'broken', file: 'src/job.ts', line: 2, quote: 'return 1;', rule: 'No dead fallbacks or null guards on non-null columns.', source: 'AGENTS.md', requirement: true },
474
+ { id: 'r2', status: 'broken', file: 'src/job.ts', line: 3, quote: '}', rule: 'Import the JOBS_TABLE constant; do not inline the raw table name.', source: 'AGENTS.md', requirement: true },
475
+ ] }, undefined, checkoutVerifier(repo));
476
+ expect(onHuman.open.map(i => [i.kind, i.locations ?? []])).toEqual([['prior', [{ file: 'src/job.ts', line: 2 }]], ['rule', []], ['finding', []]]);
477
+ expect(onHuman.open[1].issue).toContain('JOBS_TABLE');
462
478
  // A rule break and the finding it caused, on the same lines and in like words, are one item.
463
479
  const twice = account({ ...EMPTY, prior_points: [], findings: [{ class: 'correctness', file: 'src/job.ts', line: 2, issue: 'the raw table name is inlined instead of the JOBS_TABLE constant', input: 'any run', consequence: 'a rename misses it', quote: 'return 1;' }],
464
480
  rules: [{ id: 'r', status: 'broken', file: 'src/job.ts', line: 2, quote: 'return 1;', rule: 'Import the JOBS_TABLE constant; do not inline the raw table name again.', source: 'AGENTS.md', requirement: true }] }, undefined, checkoutVerifier(repo));
@@ -703,3 +719,74 @@ describe('a block sits on a line the change touched', () => {
703
719
  expect(account(rule('src/player.ts', 1819), undefined, () => true, { approvals: [], inCheckout: () => undefined }).open).toHaveLength(1);
704
720
  });
705
721
  });
722
+ describe("a human's should-fix point", () => {
723
+ it('is shown with its quote and never blocks, like a should-fix finding', () => {
724
+ const verdict = { ...EMPTY, prior_points: [{ point: 'a reopened deck reads as a return every day', review: 'senior 2026-10-01', severity: 'should-fix', resolved: false, file: 'src/job.ts', line: 2, quote: 'return 1;' }] };
725
+ const { open, advisory, unverified } = account(verdict, undefined, checkoutVerifier(repo));
726
+ expect(open).toEqual([]);
727
+ expect(advisory.map(i => [i.kind, i.issue])).toEqual([['prior', 'a reopened deck reads as a return every day']]);
728
+ expect(unverified).toEqual([]);
729
+ const unplaced = account({ ...verdict, prior_points: [{ ...verdict.prior_points[0], quote: 'not in the file' }] }, undefined, checkoutVerifier(repo));
730
+ expect(unplaced.advisory).toEqual([]); // a should-fix the judge cannot show is not worth a person's time
731
+ expect(unplaced.unverified).toHaveLength(1);
732
+ });
733
+ });
734
+ describe("the reviewer's own severity label", () => {
735
+ const at = '2026-10-01T10:00:00Z';
736
+ const labels = [
737
+ { login: 'senior', at, severity: 'blocking', text: 'The kill switch is read after every query: check it first.' },
738
+ { login: 'senior', at, severity: 'should-fix', text: 'The comment on the window still says daily.' },
739
+ ];
740
+ const point = (over) => ({ ...EMPTY, prior_points: [{ point: 'the kill switch is read after every query', review: 'senior 2026-10-01T10:00:00Z', severity: 'should-fix', resolved: false, file: 'src/job.ts', line: 2, quote: 'return 1;', ...over }] });
741
+ it('wins over the judge: a blocker the judge read as a should-fix blocks, and the disagreement is said', () => {
742
+ const { open, advisory } = account(point({}), undefined, checkoutVerifier(repo), { approvals: [], inCheckout: () => undefined, labels });
743
+ expect(advisory).toEqual([]);
744
+ expect(open).toMatchObject([{ kind: 'prior', evidence: 'the review labels it blocking; the judge read should-fix' }]);
745
+ });
746
+ it('matches the review by its date however the judge writes the time, and falls back to the reviewer\'s latest labelled review', () => {
747
+ const later = { login: 'senior', at: '2026-10-03T09:00:00Z', severity: 'should-fix', text: 'The kill switch is read after every query: check it first.' };
748
+ for (const review of ['senior 2026-10-01', 'senior 2026-10-01 10:00', 'senior']) {
749
+ const { open, labels: counted } = account(point({ review }), undefined, checkoutVerifier(repo), { approvals: [], inCheckout: () => undefined, labels });
750
+ expect(open).toHaveLength(1);
751
+ expect(counted).toEqual({ served: 2, taken: 1, disagreed: 1 });
752
+ }
753
+ // No review of that reviewer on the judge's date: the reviewer's latest labelled review decides (here, a should-fix).
754
+ const { open, advisory } = account(point({ review: 'senior 2026-09-30', severity: 'blocking' }), undefined, checkoutVerifier(repo), { approvals: [], inCheckout: () => undefined, labels: [...labels, later] });
755
+ expect([open.length, advisory.length]).toEqual([0, 1]);
756
+ });
757
+ it('applies only to the same reviewer, and only to a point that reads like the labelled line; the counts say so', () => {
758
+ for (const over of [{ review: 'peer 2026-10-01T10:00:00Z' }, { point: 'the email retry sends twice' }]) {
759
+ const { open, advisory, labels: counted } = account(point(over), undefined, checkoutVerifier(repo), { approvals: [], inCheckout: () => undefined, labels });
760
+ expect([open.length, advisory.length]).toEqual([0, 1]); // the judge's should-fix stands
761
+ expect(counted).toEqual({ served: 2, taken: 0, disagreed: 0 });
762
+ }
763
+ });
764
+ });
765
+ describe('the judge Rigour launches', () => {
766
+ it('runs claude with every memory file switched off, and records the isolation as unverified below the version it was verified in', async () => {
767
+ const seen = seenNow();
768
+ const result = await runReviewer(repo, 'main', ConfigSchema.parse({ version: 1, review: { reviewer: { enabled: true, reviewers: ['claude'] } } }), fakes(() => JSON.stringify(EMPTY), seen, null), () => undefined, { trigger: 'review' });
769
+ expect(result.outcome).toBe('passed');
770
+ const claude = seen.ran.findIndex(command => path.basename(command).startsWith('claude'));
771
+ expect(seen.env?.[claude]).toEqual({ CLAUDE_CODE_DISABLE_CLAUDE_MDS: '1', CLAUDE_CODE_DISABLE_AUTO_MEMORY: '1' });
772
+ // The fake reports version 1.0.0, older than the one the switches were verified in: the record says so.
773
+ expect(result.record?.judges.map(j => j.outside_repo)).toEqual(['claude 1.0.0: memory isolation unverified (needs 2.1.285 or later)']);
774
+ expect(recordLines(result.record).join('\n')).toContain('[claude 1.0.0: memory isolation unverified (needs 2.1.285 or later)]');
775
+ const current = seenNow();
776
+ // The installed fake is claude on Unix and claude.cmd on Windows: name both.
777
+ current.versions = { [path.join(bins[0], 'claude')]: '2.1.285 (Claude Code)', [path.join(bins[0], 'claude.cmd')]: '2.1.285 (Claude Code)' };
778
+ const verified = await runReviewer(repo, 'main', ConfigSchema.parse({ version: 1, review: { reviewer: { enabled: true, reviewers: ['claude'] } } }), fakes(() => JSON.stringify(EMPTY), current, null), () => undefined, { trigger: 'review', force: true });
779
+ expect(verified.record?.judges.map(j => j.outside_repo)).toEqual([undefined]);
780
+ });
781
+ });
782
+ describe("the review on the task's thread", () => {
783
+ it('appends each review of a branch to its task, and never a backtest replaying history', async () => {
784
+ const seen = seenNow();
785
+ await runReviewer(repo, 'main', ConfigSchema.parse({ version: 1, review: { reviewer: { enabled: true, reviewers: ['claude'] } } }), fakes(() => JSON.stringify(EMPTY), seen, null), () => undefined, { trigger: 'review' });
786
+ const thread = readThread(repo, 'feature');
787
+ expect(thread?.events.map(e => [e.kind, e.trigger, e.outcome, e.blocking])).toEqual([['review', 'review', 'passed', 0]]);
788
+ expect(thread?.events[0].integrity).toEqual(expect.any(String));
789
+ await runReviewer(repo, 'main', config, fakes(() => JSON.stringify({ ...EMPTY, prior_points: [] }), seenNow()), () => undefined, { pr: 42, reviewsBefore: '2026-10-03', force: true });
790
+ expect(readThread(repo, 'feature')?.events).toHaveLength(1);
791
+ });
792
+ });
@@ -5,14 +5,26 @@ export interface RepoRule {
5
5
  text: string;
6
6
  /** Paths the rule names (files or directories). */
7
7
  paths: string[];
8
+ /** Where the rule applies: the folder of the nested rules file it came from (`services/billing/`); the whole repository when absent. */
9
+ scope?: string;
8
10
  /** Identifiers the rule names. */
9
11
  symbols: string[];
10
12
  /** Worded as a requirement: a break can block. Guidance otherwise: a break is shown. */
11
13
  requirement: boolean;
12
14
  }
15
+ /**
16
+ * Every rule file of the repository: the root ones, the AGENTS.md and CLAUDE.md files in folders below it (outside
17
+ * vendored folders), the rule directories, and every file a rule file imports with an `@path` line (relative to the
18
+ * importing file, inside the repository). A judge does not load these by itself; this is how the repository's rules
19
+ * reach it. A nested file's rules apply to its own folder only, and so do the rules of the files it imports.
20
+ */
13
21
  export declare function readRepoRules(cwd: string): RepoRule[];
14
22
  /** One rule per top-level bullet or paragraph; headings and import lines are not rules. */
15
23
  export declare function splitRules(source: string, text: string): RepoRule[];
16
24
  export declare function rulesSection(rules: RepoRule[]): string;
17
- /** The rules that apply to a diff's changed files and added identifiers, the top `limit`. */
18
- export declare function rulesForDiff(cwd: string, diff: string, enabled?: boolean, limit?: number): RepoRule[];
25
+ /**
26
+ * The rules that apply to a diff's changed files and added identifiers, the top `limit`. `namedOnly`: only rules that
27
+ * name a path or identifier the change touches (a briefing has no code to judge applicability against, so a rule that
28
+ * merely shares words with a large change is noise there).
29
+ */
30
+ export declare function rulesForDiff(cwd: string, diff: string, enabled?: boolean, limit?: number, namedOnly?: boolean): RepoRule[];