@rigour-labs/core 6.9.0-rc.4 → 6.9.0-rc.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/dist/outcomes/metrics.d.ts +41 -2
  2. package/dist/outcomes/metrics.js +20 -1
  3. package/dist/outcomes/run.js +10 -3
  4. package/dist/review/reviewer/orchestrator.d.ts +51 -0
  5. package/dist/review/reviewer/orchestrator.js +96 -0
  6. package/dist/review/reviewer/settings.js +1 -1
  7. package/dist/review/reviewer/specialists/cleanup.v1.d.ts +9 -0
  8. package/dist/review/reviewer/specialists/cleanup.v1.js +9 -0
  9. package/dist/review/reviewer/specialists/correctness.v1.d.ts +9 -0
  10. package/dist/review/reviewer/specialists/correctness.v1.js +9 -0
  11. package/dist/review/reviewer/specialists/prior-points.v1.d.ts +9 -0
  12. package/dist/review/reviewer/specialists/prior-points.v1.js +9 -0
  13. package/dist/review/reviewer/specialists/production-cost.v1.d.ts +9 -0
  14. package/dist/review/reviewer/specialists/production-cost.v1.js +9 -0
  15. package/dist/review/reviewer/specialists/rules-and-goal.v1.d.ts +9 -0
  16. package/dist/review/reviewer/specialists/rules-and-goal.v1.js +9 -0
  17. package/dist/review/reviewer/store.d.ts +35 -0
  18. package/dist/review/reviewer/store.js +41 -0
  19. package/dist/review/reviewer/triage.d.ts +62 -0
  20. package/dist/review/reviewer/triage.js +147 -0
  21. package/dist/review/reviewer/usage.js +14 -0
  22. package/dist/review/reviewer/verdict.js +24 -7
  23. package/dist/review/reviewer.d.ts +31 -2
  24. package/dist/review/reviewer.js +189 -43
  25. package/dist/settings.d.ts +1 -0
  26. package/dist/switches.d.ts +311 -0
  27. package/dist/switches.js +1 -0
  28. package/dist/templates/universal-config.js +1 -0
  29. package/dist/types/index.d.ts +11 -0
  30. package/dist/types/index.js +5 -0
  31. package/package.json +6 -6
@@ -42,6 +42,33 @@ export interface OutcomeMetrics {
42
42
  fixedLater: Share;
43
43
  };
44
44
  };
45
+ /** What the model reviewer added on the settled pull requests a review by Rigour ran on. */
46
+ model: {
47
+ /**
48
+ * The model's findings (blocking and should-fix) out of all findings, the deterministic checks' included, at
49
+ * each pull request's first review that recorded both: before the review's own points changed the code. Rate
50
+ * only from MIN_FOR_RATE such pull requests.
51
+ */
52
+ share: {
53
+ model: number;
54
+ checks: number;
55
+ prs: number;
56
+ rate: number | null;
57
+ reason?: string;
58
+ };
59
+ /**
60
+ * Dollars of every review round on a pull request, summed per pull request, over pull requests whose every review
61
+ * event counts every run. Median only from MIN_FOR_RATE. Pull requests with an earlier event are counted apart in
62
+ * `prsEarlierBasis`, never pooled: their dollars are on another basis.
63
+ */
64
+ costPerPr: {
65
+ prs: number;
66
+ totalUsd: number;
67
+ medianUsd: number | null;
68
+ reason?: string;
69
+ prsEarlierBasis: number;
70
+ };
71
+ };
45
72
  lessons: {
46
73
  /** Candidates waiting on a person: a later fix on their lines, back to candidate, or taken back. */
47
74
  awaitingDecision: number;
@@ -52,5 +79,17 @@ export interface OutcomeMetrics {
52
79
  takenBack: number;
53
80
  };
54
81
  }
55
- /** `reviewed`: the pull requests a review by Rigour ran on (the threads' review events). */
56
- export declare function outcomeMetrics(records: PrOutcome[], lessons: ReviewLesson[], reviewed: Set<number>): OutcomeMetrics;
82
+ /** One pull request's reviews by Rigour, from the threads' review events, oldest first. */
83
+ export interface PrReviews {
84
+ /** The first review that recorded the deterministic checks' count: the model's findings and the checks' then. */
85
+ first?: {
86
+ model: number;
87
+ checks: number;
88
+ };
89
+ /** Every review round's dollars, summed. */
90
+ usd: number;
91
+ /** A review event from before cost_usd counted every run (no `cost_basis: 'runs'`): its dollars are on another basis. */
92
+ earlierBasis: boolean;
93
+ }
94
+ /** `reviewed`: the pull requests a review by Rigour ran on, each with its reviews (the threads' review events). */
95
+ export declare function outcomeMetrics(records: PrOutcome[], lessons: ReviewLesson[], reviewed: Map<number, PrReviews>): OutcomeMetrics;
@@ -1,7 +1,7 @@
1
1
  import { pendingDecision } from '../review-learning/lessons.js';
2
2
  /** Fewer records than this give a count, never a percentage. */
3
3
  const MIN_FOR_RATE = 10;
4
- /** `reviewed`: the pull requests a review by Rigour ran on (the threads' review events). */
4
+ /** `reviewed`: the pull requests a review by Rigour ran on, each with its reviews (the threads' review events). */
5
5
  export function outcomeMetrics(records, lessons, reviewed) {
6
6
  const settled = records.filter(r => r.settled);
7
7
  const fixed = (r) => r.followUps.some(f => f.fix);
@@ -18,6 +18,7 @@ export function outcomeMetrics(records, lessons, reviewed) {
18
18
  reviewed: { prs: inReview.length, fixedLater: share(inReview.filter(fixed).length, inReview.length) },
19
19
  notReviewed: { prs: notInReview.length, fixedLater: share(notInReview.filter(fixed).length, notInReview.length) },
20
20
  },
21
+ model: modelNumbers(inReview.map(r => reviewed.get(r.pr)).filter((x) => !!x)),
21
22
  lessons: {
22
23
  awaitingDecision: lessons.filter(l => pendingDecision(l)).length,
23
24
  promotedFromEvidence: lessons.filter(promotedAfterEvidence).length,
@@ -26,6 +27,24 @@ export function outcomeMetrics(records, lessons, reviewed) {
26
27
  },
27
28
  };
28
29
  }
30
+ function modelNumbers(prs) {
31
+ const firsts = prs.flatMap(p => p.first ? [p.first] : []);
32
+ const model = firsts.reduce((n, f) => n + f.model, 0);
33
+ const checks = firsts.reduce((n, f) => n + f.checks, 0);
34
+ const few = (n) => n < MIN_FOR_RATE ? { reason: `fewer than ${MIN_FOR_RATE} pull requests: a count, not a rate` } : {};
35
+ const usd = prs.filter(p => !p.earlierBasis).map(p => p.usd).sort((a, b) => a - b);
36
+ const mid = usd.length >> 1;
37
+ return {
38
+ share: { model, checks, prs: firsts.length, rate: firsts.length >= MIN_FOR_RATE && model + checks > 0 ? Math.round((model / (model + checks)) * 100) / 100 : null, ...few(firsts.length) },
39
+ costPerPr: {
40
+ prs: usd.length,
41
+ totalUsd: Math.round(usd.reduce((a, b) => a + b, 0) * 100) / 100,
42
+ medianUsd: usd.length >= MIN_FOR_RATE ? Math.round((usd.length % 2 ? usd[mid] : (usd[mid - 1] + usd[mid]) / 2) * 100) / 100 : null,
43
+ ...few(usd.length),
44
+ prsEarlierBasis: prs.filter(p => p.earlierBasis).length,
45
+ },
46
+ };
47
+ }
29
48
  function share(count, of) {
30
49
  return of >= MIN_FOR_RATE
31
50
  ? { count, of, rate: Math.round((count / of) * 100) / 100 }
@@ -44,14 +44,21 @@ function lessonEvidence(cwd, mainRef, demoteAfter) {
44
44
  writeLessons(cwd, lessons);
45
45
  return result;
46
46
  }
47
- /** Per pull request, the lessons a review of it recorded as applying, and every pull request a review by Rigour ran on: from the threads. */
47
+ /** Per pull request, the lessons a review of it recorded as applying, and every pull request a review by Rigour ran on with its rounds' dollars and its first review's findings: from the threads. */
48
48
  function reviews(cwd) {
49
49
  const applied = new Map();
50
- const reviewed = new Set();
50
+ const reviewed = new Map();
51
+ const count = (v) => typeof v === 'number' ? v : 0;
51
52
  for (const e of eventsOfKind(cwd, 'review')) {
52
53
  if (typeof e.pr !== 'number')
53
54
  continue;
54
- reviewed.add(e.pr);
55
+ const pr = reviewed.get(e.pr) ?? { usd: 0, earlierBasis: false };
56
+ pr.usd += count(e.cost_usd);
57
+ if (e.cost_basis !== 'runs')
58
+ pr.earlierBasis = true;
59
+ if (!pr.first && typeof e.checks === 'number')
60
+ pr.first = { model: count(e.blocking) + count(e.should_fix), checks: e.checks };
61
+ reviewed.set(e.pr, pr);
55
62
  if (!Array.isArray(e.lessons_applied))
56
63
  continue;
57
64
  const ids = applied.get(e.pr) ?? new Set();
@@ -0,0 +1,51 @@
1
+ import type { ReviewerName } from './adapters.js';
2
+ import type { CostBaseline, ReviewCost } from './store.js';
3
+ import type { Verdict } from './verdict.js';
4
+ export interface Specialist {
5
+ id: string;
6
+ version: number;
7
+ title: string;
8
+ steps: number[];
9
+ classes: string[];
10
+ focus: string;
11
+ }
12
+ /** The set is fixed in v1: not configurable, so a backtest and a measurement can pin it. */
13
+ export declare const SPECIALISTS: readonly Specialist[];
14
+ /** What changes the orchestrated prompt: part of the verdict's fingerprint, so a single review's verdict is never reused for it. */
15
+ export declare const SPECIALISTS_KEY: string;
16
+ /** The block a pass's prompt ends with: its parts of the review and its slice of the diff; the rest is not its to report. */
17
+ export declare function focusBlock(parts: Specialist[], sliceFile: string): string;
18
+ export interface OrchestratedRun {
19
+ /** The passes' verdicts that came back, each tagged `<judge>:<specialist,specialist>`. */
20
+ parts: Verdict[];
21
+ /** The passes that returned and those that did not, by label (`correctness, cleanup`, or `part 2: correctness`). */
22
+ returned: string[];
23
+ missing: string[];
24
+ /** At least half of the passes returned: the result stands, partial when some are missing. */
25
+ stands: boolean;
26
+ }
27
+ /** Runs every planned pass at once through `ask` (one judge run each, counted by the caller) and keeps what came back. */
28
+ export declare function runPasses<P extends {
29
+ specialists: string[];
30
+ }>(judge: string, passes: P[], ask: (pass: P) => Promise<{
31
+ verdict: Verdict;
32
+ } | {
33
+ error: string;
34
+ }>): Promise<OrchestratedRun>;
35
+ /** Single reviews the frozen baseline needs before it prices a character in dollars; with fewer, the ledger counts characters. */
36
+ export declare const BASELINE_MIN_SINGLES = 5;
37
+ export interface Ledger {
38
+ credit: number;
39
+ unit: 'usd' | 'chars';
40
+ }
41
+ /**
42
+ * The savings ledger: over the last LEDGER_WINDOW orchestrated reviews, what one judge would have been given minus what
43
+ * every run was given. In dollars when the frozen baseline prices a character (a run that reported dollars counts
44
+ * them, one that did not is priced at the baseline); otherwise in characters.
45
+ */
46
+ export declare function ledger(costs: ReviewCost[], baseline: CostBaseline | undefined): Ledger;
47
+ /** What a split needs from the ledger: its projected size minus one pass's (shared inputs + the reviewable diff), in the ledger's unit. */
48
+ export declare function splitNeeds(splitChars: number, singleChars: number, unit: Ledger['unit'], baseline: CostBaseline | undefined): number;
49
+ export declare function formatLedger(amount: number, unit: Ledger['unit']): string;
50
+ /** The most diff one pass is given, from the judge's context window and its timeout. */
51
+ export declare function passLimit(judge: ReviewerName, timeoutMs: number): number;
@@ -0,0 +1,96 @@
1
+ /**
2
+ * The review orchestrator (opt-in, experimental): a router, not a fan-out. Triage (triage.ts) picks, without a model,
3
+ * which of a fixed set of specialists a change needs, hunk by hunk; they run as ONE combined pass at every size, split
4
+ * only when one pass would not fit the judge. The prompt and its input files are built once for the review; a pass adds
5
+ * only its focus block and its slice of the diff. Verdicts merge through the reviewer's own accounting, so the evidence
6
+ * contract, the quote check and what blocks are unchanged; an item two passes both raise is one item naming both.
7
+ *
8
+ * Fails safe: a pass with no verdict is named as not reviewed, never "passed". With at least half of the passes back
9
+ * the result stands; with fewer, the caller falls back to one judge, once, if the caps allow it, else the review is
10
+ * unavailable. A change with nothing for a model to review gets no pass.
11
+ *
12
+ * Cost: a split always costs more than one pass for that review (each pass re-reads the shared inputs), so a split
13
+ * spends only what the router already saved. The savings ledger is, over the last LEDGER_WINDOW orchestrated reviews,
14
+ * what one judge would have been given minus what every run was given. A change over the judge's limit is split by
15
+ * hunk only when the ledger covers the split's extra; an empty ledger (no history) never splits.
16
+ */
17
+ import { createHash } from 'crypto';
18
+ import { CLEANUP_V1 } from './specialists/cleanup.v1.js';
19
+ import { CORRECTNESS_V1 } from './specialists/correctness.v1.js';
20
+ import { PRIOR_POINTS_V1 } from './specialists/prior-points.v1.js';
21
+ import { PRODUCTION_COST_V1 } from './specialists/production-cost.v1.js';
22
+ import { RULES_AND_GOAL_V1 } from './specialists/rules-and-goal.v1.js';
23
+ /** The set is fixed in v1: not configurable, so a backtest and a measurement can pin it. */
24
+ export const SPECIALISTS = [PRIOR_POINTS_V1, CORRECTNESS_V1, PRODUCTION_COST_V1, CLEANUP_V1, RULES_AND_GOAL_V1];
25
+ /** What changes the orchestrated prompt: part of the verdict's fingerprint, so a single review's verdict is never reused for it. */
26
+ export const SPECIALISTS_KEY = createHash('sha256').update(JSON.stringify(SPECIALISTS)).digest('hex').slice(0, 12);
27
+ /** The block a pass's prompt ends with: its parts of the review and its slice of the diff; the rest is not its to report. */
28
+ export function focusBlock(parts, sliceFile) {
29
+ const steps = [...new Set(parts.flatMap(p => p.steps))].sort((a, b) => a - b);
30
+ const classes = [...new Set(parts.flatMap(p => p.classes))];
31
+ const priorPoints = parts.some(p => p.id === 'prior-points');
32
+ return `
33
+
34
+ This review is split by part, and you do these: ${parts.map(p => p.title).join('; ')}. Your part of the diff is in
35
+ ${sliceFile} (the hunks for these parts, and the hunks defining what they use); read the full diff only where a line of
36
+ yours needs it. Do step(s) ${steps.join(', ')} above, fully.${classes.length ? ` Report findings only of class ${classes.join(', ')}.` : ''}${priorPoints ? '' : ' Leave prior_points empty.'}
37
+ ${parts.map(p => p.focus).join(' ')} Leave every list in the JSON that is not your part empty ([]).`;
38
+ }
39
+ /** A pass's name in the record: its specialists, and its part when the change was split. */
40
+ function passLabel(specialists, index, of) {
41
+ return of > 1 ? `part ${index + 1}: ${specialists.join(', ')}` : specialists.join(', ');
42
+ }
43
+ /** Runs every planned pass at once through `ask` (one judge run each, counted by the caller) and keeps what came back. */
44
+ export async function runPasses(judge, passes, ask) {
45
+ const answers = await Promise.all(passes.map(async (pass) => ({ pass, answer: await ask(pass) })));
46
+ const parts = [];
47
+ const returned = [];
48
+ const missing = [];
49
+ answers.forEach(({ pass, answer }, i) => {
50
+ const label = passLabel(pass.specialists, i, passes.length);
51
+ if ('verdict' in answer) {
52
+ parts.push({ ...answer.verdict, reviewer: `${judge}:${pass.specialists.join(',')}` });
53
+ returned.push(label);
54
+ }
55
+ else
56
+ missing.push(label);
57
+ });
58
+ return { parts, returned, missing, stands: returned.length * 2 >= passes.length };
59
+ }
60
+ /** How many recent orchestrated reviews the savings ledger sums: older savings never fund a split. */
61
+ const LEDGER_WINDOW = 20;
62
+ /** Single reviews the frozen baseline needs before it prices a character in dollars; with fewer, the ledger counts characters. */
63
+ export const BASELINE_MIN_SINGLES = 5;
64
+ /**
65
+ * The savings ledger: over the last LEDGER_WINDOW orchestrated reviews, what one judge would have been given minus what
66
+ * every run was given. In dollars when the frozen baseline prices a character (a run that reported dollars counts
67
+ * them, one that did not is priced at the baseline); otherwise in characters.
68
+ */
69
+ export function ledger(costs, baseline) {
70
+ const rate = baseline?.usdPerChar ?? null;
71
+ const recent = costs.filter(c => c.mode === 'orchestrator').slice(-LEDGER_WINDOW);
72
+ if (rate === null)
73
+ return { credit: recent.reduce((sum, c) => sum + c.projectedSingleChars - c.actualChars, 0), unit: 'chars' };
74
+ return { credit: recent.reduce((sum, c) => sum + c.projectedSingleChars * rate - (c.actualUsd > 0 ? c.actualUsd : c.actualChars * rate), 0), unit: 'usd' };
75
+ }
76
+ /** What a split needs from the ledger: its projected size minus one pass's (shared inputs + the reviewable diff), in the ledger's unit. */
77
+ export function splitNeeds(splitChars, singleChars, unit, baseline) {
78
+ const extra = splitChars - singleChars;
79
+ return unit === 'usd' ? extra * baseline.usdPerChar : extra;
80
+ }
81
+ export function formatLedger(amount, unit) {
82
+ return unit === 'usd' ? `$${amount.toFixed(2)}` : `${Math.round(amount)} chars`;
83
+ }
84
+ /**
85
+ * Each judge's context window, in tokens: the defaults its CLI or API runs with. Unverified against every model a team
86
+ * may pick; a pass is held to half of it, so the prompt, the reads and the answer fit beside the diff.
87
+ */
88
+ const CONTEXT_TOKENS = { claude: 200_000, codex: 200_000, cursor: 200_000, api: 128_000 };
89
+ const CHARS_PER_TOKEN = 4;
90
+ const DIFF_SHARE_OF_CONTEXT = 0.5;
91
+ /** Diff a judge can read and reason over per second of its timeout: past this, the run times out before it answers. */
92
+ const DIFF_CHARS_PER_SECOND = 500;
93
+ /** The most diff one pass is given, from the judge's context window and its timeout. */
94
+ export function passLimit(judge, timeoutMs) {
95
+ return Math.floor(Math.min(CONTEXT_TOKENS[judge] * CHARS_PER_TOKEN * DIFF_SHARE_OF_CONTEXT, (timeoutMs / 1000) * DIFF_CHARS_PER_SECOND));
96
+ }
@@ -13,7 +13,7 @@ const RANK = { single: 0, cross: 1, full: 2 };
13
13
  /** How near a layer is to this run: the nearer wins. */
14
14
  const NEAR = { flag: 0, env: 1, user: 2, team: 3 };
15
15
  export function resolveReviewer(config, choice = {}, user = loadSettings().reviewer, env = process.env) {
16
- const team = config.review?.reviewer ?? { enabled: false, on_push: 'background', reviewers: ['claude'], mode: 'single', models: {}, timeout_ms: 15 * 60_000, panel: 'off', mode_required: false, panel_max_items: 20, dismissals: false, judges: 2, escalate: 'always', cross_models: {}, judge_env: {}, reasoning: {} };
16
+ const team = config.review?.reviewer ?? { enabled: false, on_push: 'background', reviewers: ['claude'], mode: 'single', models: {}, timeout_ms: 15 * 60_000, panel: 'off', mode_required: false, panel_max_items: 20, dismissals: false, orchestrator: 'off', judges: 2, escalate: 'always', cross_models: {}, judge_env: {}, reasoning: {} };
17
17
  const refused = [];
18
18
  const envMode = parseMode(env.RIGOUR_REVIEWER_MODE);
19
19
  const envPanel = parseSwitch(env.RIGOUR_REVIEWER_PANEL);
@@ -0,0 +1,9 @@
1
+ /** Specialist v1: what the change leaves behind (prompt step 2 and the dead-code, duplication and helper-bypass parts of 11). */
2
+ export declare const CLEANUP_V1: {
3
+ id: string;
4
+ version: number;
5
+ title: string;
6
+ steps: number[];
7
+ classes: string[];
8
+ focus: string;
9
+ };
@@ -0,0 +1,9 @@
1
+ /** Specialist v1: what the change leaves behind (prompt step 2 and the dead-code, duplication and helper-bypass parts of 11). */
2
+ export const CLEANUP_V1 = {
3
+ id: 'cleanup',
4
+ version: 1,
5
+ title: 'what the change leaves behind: redundancy, dead code, duplication, helpers bypassed',
6
+ steps: [2, 11],
7
+ classes: ['dead-code', 'duplication', 'helper-bypass'],
8
+ focus: 'Find what a fix made unnecessary and whether it was removed, exports nothing consumes (a test is not a consumer), code copied across sibling routes or runners, and links or ids built outside the helper that owns them. Report only those classes.',
9
+ };
@@ -0,0 +1,9 @@
1
+ /** Specialist v1: correctness past the request (prompt steps 5, 6, 7 and the correctness part of 11). */
2
+ export declare const CORRECTNESS_V1: {
3
+ id: string;
4
+ version: number;
5
+ title: string;
6
+ steps: number[];
7
+ classes: string[];
8
+ focus: string;
9
+ };
@@ -0,0 +1,9 @@
1
+ /** Specialist v1: correctness past the request (prompt steps 5, 6, 7 and the correctness part of 11). */
2
+ export const CORRECTNESS_V1 = {
3
+ id: 'correctness',
4
+ version: 1,
5
+ title: 'correctness: merge impact, the journey past the request, sibling parity',
6
+ steps: [5, 6, 7, 11],
7
+ classes: ['correctness'],
8
+ focus: 'Trace what the change does to state that outlives the request: retries, two runs at once, statuses that can move back, keys that change on edit, and every sibling that needs the same change. Report only correctness findings.',
9
+ };
@@ -0,0 +1,9 @@
1
+ /** Specialist v1: every earlier human review point (prompt step 1). A new version is a new file; the record names the version. */
2
+ export declare const PRIOR_POINTS_V1: {
3
+ id: string;
4
+ version: number;
5
+ title: string;
6
+ steps: number[];
7
+ classes: string[];
8
+ focus: string;
9
+ };
@@ -0,0 +1,9 @@
1
+ /** Specialist v1: every earlier human review point (prompt step 1). A new version is a new file; the record names the version. */
2
+ export const PRIOR_POINTS_V1 = {
3
+ id: 'prior-points',
4
+ version: 1,
5
+ title: 'earlier human review points',
6
+ steps: [1],
7
+ classes: [],
8
+ focus: 'Judge every point of every human review as step 1 asks, siblings included. Report findings only when a point is unresolved and you can show it; leave every other list empty.',
9
+ };
@@ -0,0 +1,9 @@
1
+ /** Specialist v1: production cost (prompt steps 3, 4 and the production-cost part of 11). */
2
+ export declare const PRODUCTION_COST_V1: {
3
+ id: string;
4
+ version: number;
5
+ title: string;
6
+ steps: number[];
7
+ classes: string[];
8
+ focus: string;
9
+ };
@@ -0,0 +1,9 @@
1
+ /** Specialist v1: production cost (prompt steps 3, 4 and the production-cost part of 11). */
2
+ export const PRODUCTION_COST_V1 = {
3
+ id: 'production-cost',
4
+ version: 1,
5
+ title: 'production cost: every read and every nested scan',
6
+ steps: [3, 4, 11],
7
+ classes: ['production-cost'],
8
+ focus: 'Trace every read the change adds or alters: rules known before it, a narrower source, bounded windows, keyset paging, the index that serves it, and any collection scanned once per item of another. Report only production-cost findings.',
9
+ };
@@ -0,0 +1,9 @@
1
+ /** Specialist v1: what this team and this pull request asked for (prompt steps 8, 9, 10 and 12). */
2
+ export declare const RULES_AND_GOAL_V1: {
3
+ id: string;
4
+ version: number;
5
+ title: string;
6
+ steps: number[];
7
+ classes: string[];
8
+ focus: string;
9
+ };
@@ -0,0 +1,9 @@
1
+ /** Specialist v1: what this team and this pull request asked for (prompt steps 8, 9, 10 and 12). */
2
+ export const RULES_AND_GOAL_V1 = {
3
+ id: 'rules-and-goal',
4
+ version: 1,
5
+ title: 'claims, team lessons, repository rules and the declared goal',
6
+ steps: [8, 9, 10, 12],
7
+ classes: ['stale-claim', 'repo-rule'],
8
+ focus: 'Check every claim in a touched comment or the description, every team lesson served, every repository rule served (by id), and every declared goal item if the goal step is present. Report only stale-claim and repo-rule findings.',
9
+ };
@@ -21,6 +21,33 @@ export interface BranchState {
21
21
  inputsKey?: string;
22
22
  at: string;
23
23
  }
24
+ /**
25
+ * A finished review's cost, for the orchestrator's savings ledger (orchestrator.ts). Sizes are characters of what the
26
+ * judge is given: `shared` (the input files every run reads, the diff left out) plus the reviewable diff (lockfiles and
27
+ * generated files left out), counted the same way in both modes.
28
+ */
29
+ export interface ReviewCost {
30
+ at: string;
31
+ /** `single`: one judge, asked for and run (never an orchestrator's fallback). `orchestrator`: every run it made, its fallback included. */
32
+ mode: 'single' | 'orchestrator';
33
+ /** Changed lines in the reviewable diff. */
34
+ lines: number;
35
+ /** What one judge would be given for this change: shared inputs + the reviewable diff. */
36
+ projectedSingleChars: number;
37
+ /** What the orchestrator planned: per pass, shared inputs + its slice (0 when no pass). Orchestrator rows only. */
38
+ projectedChars?: number;
39
+ /** What every run was given, a failed pass, a fallback and a retry included. */
40
+ actualChars: number;
41
+ /** What every run reported costing; a judge that reports no dollars adds none. */
42
+ actualUsd: number;
43
+ runs: number;
44
+ }
45
+ /** What one judge cost per character here, frozen once, at the orchestrator's first review: `null` without enough single reviews. */
46
+ export interface CostBaseline {
47
+ at: string;
48
+ usdPerChar: number | null;
49
+ singles: number;
50
+ }
24
51
  export declare class VerdictStore {
25
52
  private readonly dir;
26
53
  private constructor();
@@ -36,6 +63,14 @@ export declare class VerdictStore {
36
63
  spend(day?: string): DaySpend;
37
64
  /** Adds runs and reported dollars to today's log: an append, so the background reviewer and a person's run never lose each other's count. */
38
65
  addSpend(runs: number, usd: number | undefined, day?: string): void;
66
+ /** One line per finished review: how it ran, what one judge would have been given and what every run was (the savings ledger reads it). */
67
+ recordCost(entry: ReviewCost): void;
68
+ /** The last `n` reviews' costs, oldest first; lines that do not parse are skipped. */
69
+ costs(n?: number): ReviewCost[];
70
+ /** The frozen baseline; written by `freezeBaseline` once, and never again. */
71
+ baseline(): CostBaseline | undefined;
72
+ /** Freezes the baseline from the single reviews kept so far, unless it is already frozen; returns the frozen one. */
73
+ freezeBaseline(minSingles: number): CostBaseline;
39
74
  private spendFile;
40
75
  /** The record of the review (record.ts), beside its verdict. */
41
76
  recordPath(verdictPath: string): string;
@@ -12,6 +12,7 @@ import { GH_TIMEOUT_MS } from './exec.js';
12
12
  function localDay(at = new Date()) {
13
13
  return `${at.getFullYear()}-${String(at.getMonth() + 1).padStart(2, '0')}-${String(at.getDate()).padStart(2, '0')}`;
14
14
  }
15
+ const COSTS_KEPT = 200;
15
16
  export class VerdictStore {
16
17
  dir;
17
18
  constructor(dir) {
@@ -74,6 +75,46 @@ export class VerdictStore {
74
75
  fs.mkdirSync(path.join(this.dir, 'spend'), { recursive: true });
75
76
  fs.appendFileSync(this.spendFile(day), `${JSON.stringify({ runs, ...(usd ? { usd } : {}), at: new Date().toISOString() })}\n`);
76
77
  }
78
+ /** One line per finished review: how it ran, what one judge would have been given and what every run was (the savings ledger reads it). */
79
+ recordCost(entry) {
80
+ fs.mkdirSync(this.dir, { recursive: true });
81
+ fs.appendFileSync(path.join(this.dir, 'costs.jsonl'), `${JSON.stringify(entry)}\n`);
82
+ }
83
+ /** The last `n` reviews' costs, oldest first; lines that do not parse are skipped. */
84
+ costs(n = COSTS_KEPT) {
85
+ let text = '';
86
+ try {
87
+ text = fs.readFileSync(path.join(this.dir, 'costs.jsonl'), 'utf8');
88
+ }
89
+ catch {
90
+ return [];
91
+ }
92
+ return text.split('\n').flatMap(line => {
93
+ try {
94
+ const entry = JSON.parse(line);
95
+ return entry && typeof entry.actualChars === 'number' && typeof entry.projectedSingleChars === 'number' ? [entry] : [];
96
+ }
97
+ catch {
98
+ return [];
99
+ }
100
+ }).slice(-n);
101
+ }
102
+ /** The frozen baseline; written by `freezeBaseline` once, and never again. */
103
+ baseline() {
104
+ return this.readJson(path.join(this.dir, 'cost-baseline.json'));
105
+ }
106
+ /** Freezes the baseline from the single reviews kept so far, unless it is already frozen; returns the frozen one. */
107
+ freezeBaseline(minSingles) {
108
+ const frozen = this.baseline();
109
+ if (frozen)
110
+ return frozen;
111
+ const singles = this.costs().filter(c => c.mode === 'single' && c.actualUsd > 0 && c.actualChars > 0);
112
+ const chars = singles.reduce((sum, c) => sum + c.actualChars, 0);
113
+ const usd = singles.reduce((sum, c) => sum + c.actualUsd, 0);
114
+ const baseline = { at: new Date().toISOString(), usdPerChar: singles.length >= minSingles ? usd / chars : null, singles: singles.length };
115
+ this.writeJson(path.join(this.dir, 'cost-baseline.json'), baseline);
116
+ return baseline;
117
+ }
77
118
  spendFile(day) {
78
119
  return path.join(this.dir, 'spend', `${day}.jsonl`);
79
120
  }
@@ -0,0 +1,62 @@
1
+ /**
2
+ * The orchestrator's router: which specialists a change needs, hunk by hunk, decided without a model, and the passes
3
+ * that takes. One combined pass is the plan at every size; a change over the judge's limit (orchestrator.ts passLimit)
4
+ * also gets a split by hunk, each part within the limit, which runs only when the savings ledger covers it. A change
5
+ * with nothing for a model to review (only lockfiles, generated files) gets no pass at all.
6
+ */
7
+ import type { Specialist } from './orchestrator.js';
8
+ export interface Hunk {
9
+ file: string;
10
+ /** The hunk as it appears in the diff, its file header included, so a slice is itself a diff. */
11
+ text: string;
12
+ added: string[];
13
+ removed: string[];
14
+ newFile: boolean;
15
+ /** Names the hunk defines on an added line (function, const, class, type). */
16
+ defines: Set<string>;
17
+ /** Every identifier on its added and removed lines. */
18
+ references: Set<string>;
19
+ }
20
+ /** The most parts a split runs: a change that needs more is one combined pass. */
21
+ export declare const MAX_PARTS = 3;
22
+ /** The diff's hunks, each with its file header. */
23
+ export declare function parseHunks(diff: string): Hunk[];
24
+ export interface TriageContext {
25
+ /** Human reviews on the pull request: the prior-points specialist is for them. */
26
+ humanReviews: number;
27
+ /** Rules and lessons the team's knowledge serves for this change. */
28
+ rulesAndLessons: number;
29
+ /** The goal step has items for a model to judge. */
30
+ goal: boolean;
31
+ }
32
+ /** Per specialist, the hunks it is for (by index); a specialist absent from the map is not needed. Prior points take the whole change. */
33
+ export declare function triage(hunks: Hunk[], context: TriageContext): Map<string, number[]>;
34
+ /** Whether no model reviews this file (SKIP). */
35
+ export declare function skipped(file: string): boolean;
36
+ export interface Pass {
37
+ specialists: string[];
38
+ /** The hunks triage picked for it. */
39
+ hunks: number[];
40
+ /** What it is given: its hunks and the hunks defining what they use. */
41
+ sliced: number[];
42
+ diff: string;
43
+ }
44
+ export interface Plan {
45
+ /** One pass with every picked specialist; undefined when triage picked nothing. */
46
+ combined?: Pass;
47
+ /** Present only when the combined pass is over `limit`: the picked hunks in parts, each within it where one hunk allows. */
48
+ split?: Pass[];
49
+ /** The parts a change over `limit` would need, when that is more than MAX_PARTS: no split, one combined pass. */
50
+ needsParts?: number;
51
+ }
52
+ /**
53
+ * The passes for what triage picked. A split is by hunk, in diff order: each part takes the next hunks while they stay
54
+ * within `limit`, and runs every specialist that picked any of them. A single hunk over the limit is a part of its own.
55
+ * A change that needs more than MAX_PARTS parts is not split.
56
+ */
57
+ export declare function planPasses(hunks: Hunk[], picked: Map<string, number[]>, order: readonly Specialist[], limit: number): Plan;
58
+ /** Characters of the reviewable diff (no lockfiles or generated files) and its changed lines: both modes are measured on this. */
59
+ export declare function reviewable(hunks: Hunk[]): {
60
+ chars: number;
61
+ lines: number;
62
+ };