@rigour-labs/core 6.9.0-rc.3 → 6.9.0-rc.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +3 -2
- package/dist/index.js +2 -2
- package/dist/outcomes/metrics.d.ts +88 -0
- package/dist/outcomes/metrics.js +56 -0
- package/dist/outcomes/run.d.ts +5 -0
- package/dist/outcomes/run.js +26 -6
- package/dist/review/reviewer/orchestrator.d.ts +51 -0
- package/dist/review/reviewer/orchestrator.js +96 -0
- package/dist/review/reviewer/settings.js +1 -1
- package/dist/review/reviewer/specialists/cleanup.v1.d.ts +9 -0
- package/dist/review/reviewer/specialists/cleanup.v1.js +9 -0
- package/dist/review/reviewer/specialists/correctness.v1.d.ts +9 -0
- package/dist/review/reviewer/specialists/correctness.v1.js +9 -0
- package/dist/review/reviewer/specialists/prior-points.v1.d.ts +9 -0
- package/dist/review/reviewer/specialists/prior-points.v1.js +9 -0
- package/dist/review/reviewer/specialists/production-cost.v1.d.ts +9 -0
- package/dist/review/reviewer/specialists/production-cost.v1.js +9 -0
- package/dist/review/reviewer/specialists/rules-and-goal.v1.d.ts +9 -0
- package/dist/review/reviewer/specialists/rules-and-goal.v1.js +9 -0
- package/dist/review/reviewer/store.d.ts +35 -0
- package/dist/review/reviewer/store.js +41 -0
- package/dist/review/reviewer/triage.d.ts +62 -0
- package/dist/review/reviewer/triage.js +147 -0
- package/dist/review/reviewer/usage.js +14 -0
- package/dist/review/reviewer/verdict.js +24 -7
- package/dist/review/reviewer.d.ts +31 -2
- package/dist/review/reviewer.js +187 -43
- package/dist/review-learning/lessons.d.ts +6 -0
- package/dist/review-learning/lessons.js +11 -0
- package/dist/settings.d.ts +1 -0
- package/dist/switches.d.ts +311 -0
- package/dist/switches.js +1 -0
- package/dist/templates/universal-config.js +1 -0
- package/dist/types/index.d.ts +11 -0
- package/dist/types/index.js +5 -0
- package/package.json +6 -6
package/dist/review/reviewer.js
CHANGED
|
@@ -28,6 +28,8 @@ import { bodyAsOf, findPullRequest, ghFor, humanReviews, linesChanged, mergesBas
|
|
|
28
28
|
import { mergeImpact } from './reviewer/merge-impact.js';
|
|
29
29
|
import { applyPanel, parseAnswers, runPanel } from './reviewer/panel.js';
|
|
30
30
|
import { crossExamPrompt, deltaBlock, goalStep, mergeBlock, PROMPT_VERSION, renderPrompt } from './reviewer/prompt.js';
|
|
31
|
+
import { BASELINE_MIN_SINGLES, focusBlock, formatLedger, ledger, passLimit, runPasses, SPECIALISTS, SPECIALISTS_KEY, splitNeeds } from './reviewer/orchestrator.js';
|
|
32
|
+
import { MAX_PARTS, parseHunks, planPasses, reviewable, triage } from './reviewer/triage.js';
|
|
31
33
|
import { modelGoalItems, parseGoal } from '../goal/goal.js';
|
|
32
34
|
import { resolveSwitch } from '../switches.js';
|
|
33
35
|
import { resolveReviewer } from './reviewer/settings.js';
|
|
@@ -57,10 +59,13 @@ export async function runReviewer(cwd, base, config, exec = defaultExec, progres
|
|
|
57
59
|
if (trigger !== 'backtest' && result.outcome !== 'skipped')
|
|
58
60
|
appendTaskEvent(cwd, {
|
|
59
61
|
kind: 'review', trigger, outcome: result.outcome, blocking: result.items.length, should_fix: result.advisory.length,
|
|
62
|
+
...(options.checks ? { checks: options.checks.length } : {}),
|
|
60
63
|
...(result.pr ? { pr: result.pr } : {}),
|
|
61
64
|
...(result.prTitle ? { pr_title: result.prTitle } : {}),
|
|
62
65
|
...(result.lessonsApplied ? { lessons_applied: result.lessonsApplied } : {}),
|
|
63
|
-
...(result.record ? { integrity: result.record.integrity,
|
|
66
|
+
...(result.record ? { integrity: result.record.integrity, judges: result.record.judges.map(j => j.reviewer) } : {}),
|
|
67
|
+
// Every run this review made, failed ones included: the same dollars as its cost row (the savings ledger).
|
|
68
|
+
...(result.spentUsd !== undefined ? { cost_usd: result.spentUsd } : {}),
|
|
64
69
|
});
|
|
65
70
|
return result;
|
|
66
71
|
}
|
|
@@ -110,6 +115,19 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
110
115
|
if (reviewers.length < 2 && mode === 'full' && (settings.required.panel || settings.required.mode)) {
|
|
111
116
|
return none('unavailable', `rigour.yml requires two reviewers from different vendors, and ${modeRecord.degraded}`, { reviewers, mode: modeRecord });
|
|
112
117
|
}
|
|
118
|
+
// The orchestrator runs the specialists on the first judge. A team floor on the panel or the mode wins over it.
|
|
119
|
+
const orchestrator = resolveSwitch('orchestrator', config, options.orchestrator);
|
|
120
|
+
let orchestrate = orchestrator.enabled;
|
|
121
|
+
if (orchestrator.refused.length)
|
|
122
|
+
modeRecord = { ...modeRecord, refused: [...(modeRecord.refused ?? []), ...orchestrator.refused] };
|
|
123
|
+
if (orchestrate && (settings.required.panel || settings.required.mode)) {
|
|
124
|
+
orchestrate = false;
|
|
125
|
+
modeRecord = { ...modeRecord, refused: [...(modeRecord.refused ?? []), 'orchestrator refused: rigour.yml requires the panel or the mode'] };
|
|
126
|
+
}
|
|
127
|
+
if (orchestrate) {
|
|
128
|
+
reviewers = reviewers.slice(0, 1);
|
|
129
|
+
modeRecord = { ...modeRecord, asked: 'orchestrator', ran: 'orchestrator', source: orchestrator.source };
|
|
130
|
+
}
|
|
113
131
|
const gh = options.blind ? undefined : ghFor(cwd, exec, await githubEnv(cwd, config.review?.github_account ?? process.env.RIGOUR_GITHUB_ACCOUNT, exec));
|
|
114
132
|
const found = gh ? await findPullRequest(gh, branch, head, options.pr) : {};
|
|
115
133
|
if (found.error)
|
|
@@ -140,7 +158,7 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
140
158
|
const goalText = goalItems.map(item => `- [${item.kind}] ${item.text}`).join('\n');
|
|
141
159
|
const previous = branch !== 'HEAD' ? store.branchState(branch) : undefined;
|
|
142
160
|
// The same commit, asked again with the same settings and reviews (the background run, then the person): the verdict it already has.
|
|
143
|
-
const inputsKey = sha([PROMPT_VERSION, rules, body, goalText, reviews.key, JSON.stringify([settings.mode, settings.panel, settings.judges, settings.escalate, settings.panel_max_items, settings.cross_models, settings.models, candidates]), [...installed].map(([n, i]) => `${n} ${i.version}`).join(';')]);
|
|
161
|
+
const inputsKey = sha([PROMPT_VERSION, rules, body, goalText, reviews.key, JSON.stringify([settings.mode, settings.panel, settings.judges, settings.escalate, settings.panel_max_items, settings.cross_models, settings.models, candidates, orchestrate ? SPECIALISTS_KEY : '']), [...installed].map(([n, i]) => `${n} ${i.version}`).join(';')]);
|
|
144
162
|
if (!options.force && previous?.head === head && previous.inputsKey === inputsKey && fs.existsSync(store.decidedPath(previous.verdict))) {
|
|
145
163
|
const verdict = store.readJson(previous.verdict);
|
|
146
164
|
const decided = store.readJson(store.decidedPath(previous.verdict));
|
|
@@ -225,11 +243,66 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
225
243
|
const verdict = store.readJson(verdictFile);
|
|
226
244
|
return withRecord(decide(verdict, previousOpen, verify, prior, dismissals), verdict, true);
|
|
227
245
|
}
|
|
228
|
-
//
|
|
229
|
-
const
|
|
246
|
+
// What one judge would be given: the shared input files and the reviewable diff. Both modes' cost rows measure this.
|
|
247
|
+
const hunks = parseHunks(fullDiff);
|
|
248
|
+
const size = reviewable(hunks);
|
|
249
|
+
const shared = reviews.markdown.length + body.length + context.text.length + (options.hints?.trim() || 'none\n').length + (goalText?.length ?? 0);
|
|
250
|
+
const projectedSingle = shared + size.chars;
|
|
251
|
+
// The orchestrator's plan: which specialists the change needs, hunk by hunk, as one combined pass. A change over the
|
|
252
|
+
// judge's limit is split by hunk only when the savings ledger covers the split's extra. Nothing to review: no pass.
|
|
253
|
+
const picked = orchestrate ? triage(hunks, { humanReviews: reviews.count, rulesAndLessons: context.rules.length + context.lessons, goal: goalItems.length > 0 }) : new Map();
|
|
254
|
+
const selected = SPECIALISTS.map(s => s.id).filter(id => picked.has(id));
|
|
255
|
+
let passes = [];
|
|
256
|
+
let planNote;
|
|
257
|
+
const limit = { judge: reviewers[0], chars: passLimit(reviewers[0], settings.timeout_ms) };
|
|
258
|
+
if (orchestrate) {
|
|
259
|
+
const baseline = store.freezeBaseline(BASELINE_MIN_SINGLES);
|
|
260
|
+
const plan = planPasses(hunks, picked, SPECIALISTS, limit.chars);
|
|
261
|
+
passes = plan.combined ? [plan.combined] : [];
|
|
262
|
+
if (plan.needsParts)
|
|
263
|
+
planNote = `over the judge's limit; needs ${plan.needsParts} parts > ${MAX_PARTS}: one pass`;
|
|
264
|
+
if (plan.split) {
|
|
265
|
+
const book = ledger(store.costs(), baseline);
|
|
266
|
+
const needs = splitNeeds(plan.split.reduce((sum, pass) => sum + shared + pass.diff.length, 0), projectedSingle, book.unit, baseline);
|
|
267
|
+
const said = `ledger ${formatLedger(book.credit, book.unit)}, split needs ${formatLedger(needs, book.unit)}`;
|
|
268
|
+
if (book.credit > 0 && book.credit >= needs) {
|
|
269
|
+
passes = plan.split;
|
|
270
|
+
planNote = `over the judge's limit; ${said}: ${plan.split.length} parts`;
|
|
271
|
+
}
|
|
272
|
+
else
|
|
273
|
+
planNote = `over the judge's limit; ${said}: one pass`;
|
|
274
|
+
}
|
|
275
|
+
// The daily caps, before any judge starts: room for one run but not every part is one combined pass.
|
|
276
|
+
if (passes.length > 1 && overBudget(store.spend(), settings, passes.length) && !overBudget(store.spend(), settings, 1)) {
|
|
277
|
+
planNote = `the caps leave one run, not ${passes.length}: one pass`;
|
|
278
|
+
passes = [plan.combined];
|
|
279
|
+
}
|
|
280
|
+
}
|
|
281
|
+
const over = orchestrate && passes.length === 0 ? undefined : overBudget(store.spend(), settings, orchestrate ? passes.length : reviewers.length);
|
|
282
|
+
// A team that requires the reviewer, or the orchestrator, gets no quieter review when the caps are reached: unavailable.
|
|
230
283
|
if (over)
|
|
231
|
-
return none(settings.required.panel || settings.required.mode ? 'unavailable' : 'skipped', over, { reviewers, scope, why, pr: pr?.number });
|
|
284
|
+
return none(settings.required.panel || settings.required.mode || orchestrator.required ? 'unavailable' : 'skipped', over, { reviewers, scope, why, pr: pr?.number });
|
|
232
285
|
const work = fs.mkdtempSync(path.join(os.tmpdir(), 'rigour-reviewer-'));
|
|
286
|
+
// Every run this review makes, failed ones included: the caps count each, and the review's cost row sums them.
|
|
287
|
+
const tally = { runs: 0, chars: 0, usd: 0 };
|
|
288
|
+
const spent = (usd, chars) => {
|
|
289
|
+
store.addSpend(1, usd);
|
|
290
|
+
tally.runs++;
|
|
291
|
+
tally.chars += chars;
|
|
292
|
+
tally.usd += usd ?? 0;
|
|
293
|
+
};
|
|
294
|
+
const spentUsd = () => Math.round(tally.usd * 10_000) / 10_000;
|
|
295
|
+
// One row per fresh review, for the savings ledger: one judge asked for and run, or orchestrated with everything it ran.
|
|
296
|
+
const recordReviewCost = () => {
|
|
297
|
+
const orchestrated = modeRecord.asked === 'orchestrator';
|
|
298
|
+
if (!orchestrated && !(modeRecord.asked === 'single' && modeRecord.ran === 'single'))
|
|
299
|
+
return;
|
|
300
|
+
store.recordCost({
|
|
301
|
+
at: new Date().toISOString(), mode: orchestrated ? 'orchestrator' : 'single', lines: size.lines, projectedSingleChars: projectedSingle,
|
|
302
|
+
...(orchestrated ? { projectedChars: passes.reduce((sum, pass) => sum + shared + pass.diff.length, 0) } : {}),
|
|
303
|
+
actualChars: tally.chars, actualUsd: spentUsd(), runs: tally.runs,
|
|
304
|
+
});
|
|
305
|
+
};
|
|
233
306
|
// One judge run, by CLI or by API: the same prompt, the same cost accounting, the same trace.
|
|
234
307
|
let inlineInputs = [];
|
|
235
308
|
const runJudge = (name, prompt, model) => name === 'api'
|
|
@@ -274,43 +347,102 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
274
347
|
const ticker = setInterval(() => progress(`Rigour reviewer: still working (${Math.round((Date.now() - started) / 60_000)} min)`), PROGRESS_EVERY_MS);
|
|
275
348
|
let parts;
|
|
276
349
|
try {
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
350
|
+
// The single-judge fallback runs once: no retry, no spare judge.
|
|
351
|
+
let once = false;
|
|
352
|
+
if (orchestrate && passes.length === 0) {
|
|
353
|
+
// Nothing for a model to review (only lockfiles, generated files): no run, and that is the verdict.
|
|
354
|
+
parts = [];
|
|
355
|
+
modeRecord = { ...modeRecord, specialists: { selected: [], returned: [], missing: [], passes: [], limit, none: 'nothing for the model reviewer to review' } };
|
|
356
|
+
}
|
|
357
|
+
else if (orchestrate) {
|
|
358
|
+
const judge = reviewers[0];
|
|
359
|
+
const ran = [];
|
|
360
|
+
const run = await runPasses(judge, passes.map((pass, i) => ({ ...pass, file: file(`part-${i + 1}.diff`, pass.diff) })), async (pass) => {
|
|
361
|
+
const assigned = pass.specialists.map(id => SPECIALISTS.find(s => s.id === id));
|
|
362
|
+
const result = await runJudge(judge, `${prompt}${focusBlock(assigned, pass.file)}`, modelFor(judge));
|
|
363
|
+
const answer = ADAPTERS[judge].answer(result.stdout);
|
|
364
|
+
spent(answer.costUsd, shared + pass.diff.length); // every pass counts against the caps and the ledger, an answer or not
|
|
365
|
+
progress(`Rigour reviewer: ${judge} (${pass.specialists.join(', ')}) finished in ${Math.round((Date.now() - started) / 1000)}s (exit ${result.exitCode})`);
|
|
366
|
+
const verdict = result.exitCode === 0 || answer.text.trim()
|
|
367
|
+
? parseVerdict(answer.text, needsPriorPoints && pass.specialists.includes('prior-points'), judge, answer)
|
|
368
|
+
: { error: `${judge} (${pass.specialists.join(', ')}): no answer (exit ${result.exitCode})` };
|
|
369
|
+
// Whether slicing held: a pass that read the full diff, or read, searched or printed a changed file outside its slice.
|
|
370
|
+
const sliceFiles = new Set(pass.sliced.map(i => hunks[i].file));
|
|
371
|
+
const outside = changedFiles.filter(f => !sliceFiles.has(f));
|
|
372
|
+
const trace = 'verdict' in verdict ? verdict.verdict.trace : undefined;
|
|
373
|
+
const beyond = trace ? trace.calls.some(c => {
|
|
374
|
+
const target = c.target.replace(/\\/g, '/');
|
|
375
|
+
return target.endsWith('/full.diff') || outside.some(f => names(target, f));
|
|
376
|
+
}) : null;
|
|
377
|
+
ran.push({ specialists: pass.specialists, hunks: pass.hunks.length, chars: pass.diff.length, readBeyondSlice: beyond });
|
|
378
|
+
return verdict;
|
|
379
|
+
});
|
|
380
|
+
const specialists = { selected, returned: run.returned, missing: run.missing, passes: ran, limit, ...(planNote ? { plan: planNote } : {}) };
|
|
381
|
+
if (orchestrator.required && run.missing.length) {
|
|
382
|
+
// A required orchestrator reviews every part or gives no verdict: a partial review, or one judge instead, is a quieter one.
|
|
383
|
+
recordReviewCost();
|
|
384
|
+
const fallback = `${run.returned.length} of ${passes.length} passes returned`;
|
|
385
|
+
return none('unavailable', `${fallback}, and rigour.yml requires the orchestrator: every part or no verdict (not reviewed: ${run.missing.join('; ')})`, { reviewers, scope, why, pr: pr?.number, spentUsd: spentUsd(), mode: { ...modeRecord, specialists: { ...specialists, fallback } } });
|
|
291
386
|
}
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
const
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
387
|
+
if (run.stands) {
|
|
388
|
+
parts = run.parts;
|
|
389
|
+
modeRecord = { ...modeRecord, specialists, ...(run.missing.length ? { degraded: `${modeRecord.degraded ? `${modeRecord.degraded}; ` : ''}not reviewed: ${run.missing.join('; ')} (no verdict)` } : {}) };
|
|
390
|
+
}
|
|
391
|
+
else {
|
|
392
|
+
// Fewer than half of the passes came back: one judge instead, once, only if the caps still allow a run; it counts too.
|
|
393
|
+
const short = overBudget(store.spend(), settings, 1);
|
|
394
|
+
const fallback = `${run.returned.length} of ${passes.length} passes returned`;
|
|
395
|
+
if (short) {
|
|
396
|
+
recordReviewCost();
|
|
397
|
+
return none('unavailable', `${fallback}, and the caps leave no run for one judge: ${short}`, { reviewers, scope, why, pr: pr?.number, spentUsd: spentUsd(), mode: { ...modeRecord, specialists: { ...specialists, fallback } } });
|
|
398
|
+
}
|
|
399
|
+
progress(`Rigour reviewer: ${fallback}; one judge reviews instead`);
|
|
400
|
+
modeRecord = { ...modeRecord, ran: 'single', specialists: { ...specialists, fallback } };
|
|
401
|
+
once = true;
|
|
307
402
|
}
|
|
308
|
-
answers[i] = answer;
|
|
309
403
|
}
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
404
|
+
if (!parts) {
|
|
405
|
+
const answers = await Promise.all(reviewers.map(async (name) => {
|
|
406
|
+
const adapter = ADAPTERS[name];
|
|
407
|
+
const ask = async () => {
|
|
408
|
+
const run = await runJudge(name, prompt, modelFor(name));
|
|
409
|
+
progress(`Rigour reviewer: ${name} finished in ${Math.round((Date.now() - started) / 1000)}s (exit ${run.exitCode})`);
|
|
410
|
+
const answer = adapter.answer(run.stdout);
|
|
411
|
+
spent(answer.costUsd, projectedSingle); // every run counts against the caps, an answer or not
|
|
412
|
+
return { run, answer, verdict: run.exitCode === 0 || answer.text.trim() ? parseVerdict(answer.text, needsPriorPoints, name, answer) : undefined };
|
|
413
|
+
};
|
|
414
|
+
let first = await ask();
|
|
415
|
+
// No verdict, whether a malformed answer or a run that died, is a slip, not a decision: asked once more, inside the caps.
|
|
416
|
+
if (!once && (!first.verdict || 'error' in first.verdict) && !overBudget(store.spend(), settings, 1)) {
|
|
417
|
+
progress(`Rigour reviewer: ${name} gave no ${first.verdict ? 'valid verdict' : 'answer'}; asking once more`);
|
|
418
|
+
first = await ask();
|
|
419
|
+
}
|
|
420
|
+
return first.verdict ?? { error: `${name}: no answer (exit ${first.run.exitCode}): ${first.run.stderr.trim().slice(-200)}` };
|
|
421
|
+
}));
|
|
422
|
+
// A judge that gives nothing is replaced by the next one installed, so the boundary stays up: a review ends unavailable only when every judge failed.
|
|
423
|
+
const spare = candidates.filter(c => installed.has(c) && !reviewers.includes(c));
|
|
424
|
+
for (let i = 0; i < answers.length; i++) {
|
|
425
|
+
let answer = answers[i];
|
|
426
|
+
while (!once && 'error' in answer && spare.length && !overBudget(store.spend(), settings, 1)) {
|
|
427
|
+
const next = spare.shift();
|
|
428
|
+
progress(`Rigour reviewer: ${reviewers[i]} gave no verdict (${answer.error}); ${next} judges instead`);
|
|
429
|
+
modeRecord = { ...modeRecord, degraded: `${modeRecord.degraded ? `${modeRecord.degraded}; ` : ''}${reviewers[i]} gave no verdict, ${next} judged instead` };
|
|
430
|
+
reviewers[i] = next;
|
|
431
|
+
const run = await runJudge(next, prompt, modelFor(next));
|
|
432
|
+
const got = ADAPTERS[next].answer(run.stdout);
|
|
433
|
+
spent(got.costUsd, projectedSingle);
|
|
434
|
+
answer = run.exitCode === 0 || got.text.trim() ? parseVerdict(got.text, needsPriorPoints, next, got) : { error: `${next}: no answer (exit ${run.exitCode}): ${run.stderr.trim().slice(-200)}` };
|
|
435
|
+
}
|
|
436
|
+
answers[i] = answer;
|
|
437
|
+
}
|
|
438
|
+
const failed = answers.find(a => 'error' in a);
|
|
439
|
+
if (failed && 'error' in failed) {
|
|
440
|
+
if (once)
|
|
441
|
+
recordReviewCost();
|
|
442
|
+
return none('unavailable', failed.error, { reviewers, scope, why, pr: pr?.number, spentUsd: spentUsd() });
|
|
443
|
+
}
|
|
444
|
+
parts = answers.map(a => a.verdict);
|
|
445
|
+
}
|
|
314
446
|
for (const part of parts) {
|
|
315
447
|
if (part.trace)
|
|
316
448
|
labelReads(part.trace, work, changedFiles);
|
|
@@ -320,7 +452,8 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
320
452
|
finally {
|
|
321
453
|
clearInterval(ticker);
|
|
322
454
|
}
|
|
323
|
-
|
|
455
|
+
// One part too: every item is tagged with who found it. No part (nothing for a model to review): an empty verdict.
|
|
456
|
+
const merged = parts.length ? mergeVerdicts(parts) : { prior_points: [], redundant: [], reads: [], scans: [], merge_impact: [], findings: [], carried: [], resolved_previous: [], reviewers: [] };
|
|
324
457
|
const touched = scope === 'delta' ? sincePrevious : new Set();
|
|
325
458
|
let verdict = scope === 'delta' ? carryResolved(merged, store.readJson(previous.verdict), touched) : merged;
|
|
326
459
|
if (modeRecord.ran === 'panel' && parts.length > 1) {
|
|
@@ -347,9 +480,10 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
347
480
|
},
|
|
348
481
|
ask: async (judge, asked) => {
|
|
349
482
|
const name = judge;
|
|
350
|
-
const
|
|
483
|
+
const examPrompt = crossExamPrompt(repoRoot, head.slice(0, 9), diffFile, asked);
|
|
484
|
+
const run = await runJudge(name, examPrompt, settings.cross_models[name] ?? modelFor(name));
|
|
351
485
|
const answer = ADAPTERS[name].answer(run.stdout);
|
|
352
|
-
|
|
486
|
+
spent(answer.costUsd, examPrompt.length);
|
|
353
487
|
reserved--;
|
|
354
488
|
cross.push({ reviewer: `${name} cross-exam`, ...(answer.costUsd !== undefined ? { cost_usd: answer.costUsd } : {}), ...(answer.tokens ? { tokens: answer.tokens } : {}) });
|
|
355
489
|
progress(`Rigour reviewer: ${name} cross-examined ${asked.length} finding(s) (exit ${run.exitCode})`);
|
|
@@ -365,9 +499,10 @@ async function review(cwd, base, config, exec, progress, options) {
|
|
|
365
499
|
store.writeJson(verdictFile, { ...verdict, inputs: { head, base: baseSha, scope, why, mode: modeRecord, reviewers, versions: reviewerVersions, authors: [...authors], fingerprint, human_reviews: reviews.count, reviews_before: options.reviewsBefore ?? null, since: previous?.head ?? null, at: new Date().toISOString() } });
|
|
366
500
|
store.writeJson(openFile, accounted.open);
|
|
367
501
|
store.writeJson(store.decidedPath(verdictFile), accounted);
|
|
502
|
+
recordReviewCost();
|
|
368
503
|
if (branch !== 'HEAD')
|
|
369
504
|
store.recordBranch(branch, { head, verdict: verdictFile, mode: scope, rulesHash, reviewsKey: reviews.key, inputsKey });
|
|
370
|
-
return withRecord(accounted, verdict, false);
|
|
505
|
+
return { ...withRecord(accounted, verdict, false), spentUsd: spentUsd() };
|
|
371
506
|
}
|
|
372
507
|
finally {
|
|
373
508
|
fs.rmSync(work, { recursive: true, force: true });
|
|
@@ -491,3 +626,12 @@ function result(accounted, verdict, reviewers, scope, why, cached, reviews, pr,
|
|
|
491
626
|
...(pr?.title ? { prTitle: pr.title } : {}),
|
|
492
627
|
};
|
|
493
628
|
}
|
|
629
|
+
/** Whether a tool call's target (a path, a glob, a shell command) names a repository file. */
|
|
630
|
+
function names(target, file) {
|
|
631
|
+
const at = target.indexOf(file);
|
|
632
|
+
if (at < 0)
|
|
633
|
+
return false;
|
|
634
|
+
const before = target[at - 1];
|
|
635
|
+
const after = target[at + file.length];
|
|
636
|
+
return (before === undefined || /[\s/'"=]/.test(before)) && (after === undefined || /[\s'":)]/.test(after));
|
|
637
|
+
}
|
|
@@ -109,6 +109,12 @@ export declare function matchLessons(lessons: ReviewLesson[], change: ChangeShap
|
|
|
109
109
|
/** RIGOUR_REVIEW_LESSONS points at a lessons file outside the clone (CI, or a team's shared copy). */
|
|
110
110
|
export declare function lessonsPath(cwd: string): string;
|
|
111
111
|
export declare function readLessons(cwd: string): ReviewLesson[];
|
|
112
|
+
/**
|
|
113
|
+
* What a person is asked to decide on a candidate, from the last of its evidence and decisions: taken back by evidence
|
|
114
|
+
* (`demoted`), back to a candidate when outcomes stopped promoting (`reclassified`), or a later fix on its lines
|
|
115
|
+
* (`lines`); undefined when nothing waits on a person. Studio and the outcome numbers read it, so they never disagree.
|
|
116
|
+
*/
|
|
117
|
+
export declare function pendingDecision(lesson: ReviewLesson): LessonEvidence | undefined;
|
|
112
118
|
export declare function writeLessons(cwd: string, lessons: ReviewLesson[]): void;
|
|
113
119
|
/** A person's decision on a lesson, kept as evidence: accepted makes it a lesson, rejected an anti-lesson. Undefined for an unknown id. */
|
|
114
120
|
export declare function decideLesson(cwd: string, id: string, decision: 'accepted' | 'rejected' | 'dismissed', by: string, why?: string): ReviewLesson | undefined;
|
|
@@ -233,6 +233,17 @@ export function readLessons(cwd) {
|
|
|
233
233
|
return [];
|
|
234
234
|
}
|
|
235
235
|
}
|
|
236
|
+
/**
|
|
237
|
+
* What a person is asked to decide on a candidate, from the last of its evidence and decisions: taken back by evidence
|
|
238
|
+
* (`demoted`), back to a candidate when outcomes stopped promoting (`reclassified`), or a later fix on its lines
|
|
239
|
+
* (`lines`); undefined when nothing waits on a person. Studio and the outcome numbers read it, so they never disagree.
|
|
240
|
+
*/
|
|
241
|
+
export function pendingDecision(lesson) {
|
|
242
|
+
if (lesson.state !== 'candidate')
|
|
243
|
+
return undefined;
|
|
244
|
+
const last = lesson.evidence.filter(e => e.kind === 'demoted' || e.kind === 'lines' || e.kind === 'reclassified' || e.kind === 'accepted' || e.kind === 'rejected' || e.kind === 'dismissed').at(-1);
|
|
245
|
+
return last?.kind === 'demoted' || last?.kind === 'lines' || last?.kind === 'reclassified' ? last : undefined;
|
|
246
|
+
}
|
|
236
247
|
/** Why a lesson an outcome alone had promoted is a candidate again. */
|
|
237
248
|
const RECLASSIFIED = 'promoted by the exact-line rule, which no longer promotes on its own';
|
|
238
249
|
/**
|