lastlight-evals 0.16.0 → 0.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -26,13 +26,22 @@ import { getWorkflow, resolveReviewGitHubClient, runWorkflow, } from "lastlight-
26
26
  import { modelTemplateForRow } from "./phase-models.js";
27
27
  import { startFakeGitHub } from "./fake-github.js";
28
28
  import { appliedRepoConfigKeys, loadRepoConfigFixture, resolveEvalRepoConfig } from "./repo-config.js";
29
- import { seedWorkspace, seedWorkspaceFromGit, seedWorkspacePrReview, prFilesFromGit, isRealSha, injectRepoContext } from "./seed.js";
29
+ import { seedWorkspace, seedWorkspaceFromGit, seedWorkspacePrReview, prFilesFromGit, mergeBaseOf, isRealSha, injectRepoContext, checkoutRound } from "./seed.js";
30
30
  import { collectMetrics, collectMetricsFromFiles, bucketSessionsByPhase, drainSessions, readSessionLog, listSessionFiles, concatJsonl, } from "./metrics.js";
31
31
  import { modelCost } from "./env.js";
32
32
  import { gradeBehavioral, gradeExecution, gradeTriage, gradeReview, gradeInternalRecall, gradeMarkers } from "./grade.js";
33
33
  import { readPipelineStats, persistPipelineArtifacts, internalJudgeInputs, withInternalRecall } from "./review-pipeline-stats.js";
34
- import { prContextPatch } from "./pr-context.js";
34
+ import { writeStoredRunContext } from "./phase-replay-context.js";
35
+ import { caseHeadSha, prContextPatch } from "./pr-context.js";
35
36
  import { resolveFactsBin } from "./paths.js";
37
+ import { coverageSummary, deltaCounts, dispositionCounts, goldMatchedOf, ledgerCounts, planRounds, rollupRereview, } from "./rereview.js";
38
+ import { carryForward, createRoundStore, fileAt, judgeComments, outdatedResolver, readFreshJson, ROUND_ARTIFACTS, roundScratch, scratchCoverage, scratchLedger, snapshotMtimes, UnitsOracle, } from "./rereview-node.js";
39
+ /**
40
+ * How far the fake's clock moves between two review rounds of a chained case,
41
+ * so round k+1's reviews and threads are strictly later than round k's — the
42
+ * order `lastBotReview` and every "latest review" read depends on.
43
+ */
44
+ const ROUND_GAP_MS = 60 * 60 * 1000;
36
45
  const EVAL_ENV_KEYS = [
37
46
  "GITHUB_APP_ID",
38
47
  "GITHUB_APP_INSTALLATION_ID",
@@ -88,6 +97,17 @@ export async function runInstance(inst, opts) {
88
97
  // workspace would be inventing a code path it does not have.
89
98
  const NO_WORKSPACE = new Set(["issue-triage", "dependabot-pr-merge"]);
90
99
  const isCodeFix = !isPrReview && !NO_WORKSPACE.has(workflowName);
100
+ // A chained re-review case (issue #429) — `null` for every case that runs
101
+ // once, which then takes exactly the path it always took. Validated before
102
+ // anything starts: a malformed chain is a case error, not a silent one-round run.
103
+ let rounds = null;
104
+ let roundsError;
105
+ try {
106
+ rounds = isPrReview ? planRounds(inst) : null;
107
+ }
108
+ catch (err) {
109
+ roundsError = err.message;
110
+ }
91
111
  const stateDir = opts.stateDir ?? mkdtempSync(join(tmpdir(), "ll-eval-"));
92
112
  const sessionsDir = join(stateDir, "agent-sessions");
93
113
  // The shim appends per-phase jsonl under <sessionsDir>/projects/<slug>/ and
@@ -115,6 +135,8 @@ export async function runInstance(inst, opts) {
115
135
  // the first end of the `spec` axis. Content here, linkage in the fake.
116
136
  issues: [...(inst.issue ? [inst.issue] : []), ...(inst.pr?.linked_issues ?? [])],
117
137
  pulls: inst.pr ? [inst.pr] : [],
138
+ // A chained case releases seeded discussion round by round (`from_round`).
139
+ chained: !!rounds,
118
140
  // The CI-read tools (`github_list_workflow_runs` / `..._run_jobs` /
119
141
  // `github_get_job_logs`) served from the SAME seed that produces the
120
142
  // prompt's `{{ciSection}}`, so digging into the logs corroborates what the
@@ -123,7 +145,7 @@ export async function runInstance(inst, opts) {
123
145
  ...(inst.pr_state?.ci_jobs?.length
124
146
  ? {
125
147
  actions: {
126
- headSha: inst.pr_state.head_sha ?? "e7a1d09",
148
+ headSha: caseHeadSha(inst),
127
149
  headBranch: inst.pr_state.head_ref,
128
150
  jobs: inst.pr_state.ci_jobs.map((j) => ({
129
151
  name: j.name,
@@ -155,6 +177,8 @@ export async function runInstance(inst, opts) {
155
177
  phases: [],
156
178
  };
157
179
  try {
180
+ if (roundsError)
181
+ throw new Error(roundsError);
158
182
  // 2. Seed the workspace for code-fix (triage needs no repo). A vendored
159
183
  // fixture dir wins; otherwise a git-source case (real base SHA + real
160
184
  // repo) is checked out from the repo-local cache. Either way the agent
@@ -205,276 +229,513 @@ export async function runInstance(inst, opts) {
205
229
  headRef: inst.pr.head_ref,
206
230
  baseCommit: inst.pr.base_commit,
207
231
  headCommit: inst.pr.head_commit,
232
+ ...(rounds ? { roundCommits: rounds.map((r) => r.head_commit) } : {}),
208
233
  repoSubdir,
209
234
  });
210
235
  }
211
236
  // The repo's working dir (the nested subdir when seeded) — where grading and
212
237
  // the diff run. Falls back to the workspace root if nothing was seeded.
213
238
  const repoDir = seed?.workDir ?? join(stateDir, "sandboxes", taskId);
214
- // Serve the PR's changed files at GET /pulls/:n/files (pr-review): computed
215
- // from base..head in the just-seeded workspace, so a review agent that lists
216
- // files via the API gets the real changed set instead of a 404.
217
- //
218
- // KEPT, not discarded: this same set is the SECOND END of every `spec`
219
- // obligation, and in production `resolveSpecContext` reads it from
220
- // `listPullRequestFilePaths` at the dispatch choke point. The eval never
221
- // calls that (it builds the snapshot itself), so without threading it into
222
- // `prContextPatch` below `changedFiles` stays `null`, `buildSpecObligations`
223
- // correctly refuses to emit a one-ended seed, and the whole spec family —
224
- // the one axis nothing else has tried — spends a model call reporting that
225
- // it cannot work. Deriving it here rather than seeding it per case keeps the
226
- // two ends from drifting apart and covers every case for free.
227
- let prFilePaths;
228
- if (isPrReview && inst.pr && seed) {
229
- const files = prFilesFromGit(repoDir, inst.pr.base_commit, inst.pr.head_commit);
230
- fake.setPullFiles(inst.pr.number, files);
231
- prFilePaths = files.map((f) => f.filename);
232
- }
233
- else if (inst.pr?.files?.length) {
234
- // A tier with no checkout (dependency-merge) states its diff in the case
235
- // instead. Same registration, so `GET /pulls/:n/files` and the patch
236
- // `github_get_pull_request_diff` returns come from one source.
237
- fake.setPullFiles(inst.pr.number, inst.pr.files);
238
- prFilePaths = inst.pr.files.map((f) => f.filename);
239
- }
240
- // 2b. Inject synthetic repo-context into the pr-review checkout so the
241
- // reviewing agent reads it — a GENERIC block from the overlay (applies to
242
- // every repo) + a PER-REPO block from the tier dataset. The Pi runtime
243
- // auto-loads AGENTS.md/CLAUDE.md walking up from the agent cwd (= the repo
244
- // dir), so this reaches the model with no prompt change. Faithful to what a
245
- // maintainer could commit, so a kept improvement is a portable "add this to
246
- // your repo" recommendation. Records provenance for inspectability.
247
- if (isPrReview && seed && (opts.injectContext ?? true)) {
248
- const sources = resolveInjectedContext({
249
- overlayDir: opts.overlayDir,
250
- datasetDir: opts.datasetDir,
251
- instanceId: inst.instance_id,
239
+ // Every round of a case runs the same workflow definition.
240
+ const def = getWorkflow(workflowName);
241
+ const trialDir = opts.sessionTrialDir;
242
+ /**
243
+ * One dispatch of the workflow: serve the head's changed files, inject the
244
+ * repo context, build the context from the snapshot, resolve the repo layer
245
+ * and run. A single-round case calls it once; a chained case
246
+ * (`inst.rounds`) once per round head, in order (see below).
247
+ */
248
+ const executeRound = async (round) => {
249
+ // Serve the PR's changed files at GET /pulls/:n/files (pr-review): computed
250
+ // from base..head in the just-seeded workspace, so a review agent that lists
251
+ // files via the API gets the real changed set instead of a 404.
252
+ //
253
+ // KEPT, not discarded: this same set is the SECOND END of every `spec`
254
+ // obligation, and in production `resolveSpecContext` reads it from
255
+ // `listPullRequestFilePaths` at the dispatch choke point. The eval never
256
+ // calls that (it builds the snapshot itself), so without threading it into
257
+ // `prContextPatch` below `changedFiles` stays `null`, `buildSpecObligations`
258
+ // correctly refuses to emit a one-ended seed, and the whole spec family —
259
+ // the one axis nothing else has tried — spends a model call reporting that
260
+ // it cannot work. Deriving it here rather than seeding it per case keeps the
261
+ // two ends from drifting apart and covers every case for free.
262
+ let prFilePaths;
263
+ if (isPrReview && inst.pr && seed) {
264
+ // Against the merge base, as GitHub computes a PR's files: a chained
265
+ // case's earlier heads forked from an older base than the case's (the
266
+ // branch merged or rebased onto main since), and a two-dot diff from
267
+ // the newer base would list main's later changes as the PR's.
268
+ const files = prFilesFromGit(repoDir, mergeBaseOf(repoDir, inst.pr.base_commit, round.head), round.head);
269
+ fake.setPullFiles(inst.pr.number, files);
270
+ prFilePaths = files.map((f) => f.filename);
271
+ }
272
+ else if (inst.pr?.files?.length) {
273
+ // A tier with no checkout (dependency-merge) states its diff in the case
274
+ // instead. Same registration, so `GET /pulls/:n/files` and the patch
275
+ // `github_get_pull_request_diff` returns come from one source.
276
+ fake.setPullFiles(inst.pr.number, inst.pr.files);
277
+ prFilePaths = inst.pr.files.map((f) => f.filename);
278
+ }
279
+ // 2b. Inject synthetic repo-context into the pr-review checkout so the
280
+ // reviewing agent reads it — a GENERIC block from the overlay (applies to
281
+ // every repo) + a PER-REPO block from the tier dataset. The Pi runtime
282
+ // auto-loads AGENTS.md/CLAUDE.md walking up from the agent cwd (= the repo
283
+ // dir), so this reaches the model with no prompt change. Faithful to what a
284
+ // maintainer could commit, so a kept improvement is a portable "add this to
285
+ // your repo" recommendation. Records provenance for inspectability.
286
+ if (isPrReview && seed && (opts.injectContext ?? true)) {
287
+ const sources = resolveInjectedContext({
288
+ overlayDir: opts.overlayDir,
289
+ datasetDir: opts.datasetDir,
290
+ instanceId: inst.instance_id,
291
+ });
292
+ if (sources.length) {
293
+ const combined = sources.map((s) => s.text.trim()).filter(Boolean).join("\n\n");
294
+ if (injectRepoContext(seed.workDir, combined)) {
295
+ result.injectedContext = sources.map((s) => ({
296
+ source: s.source,
297
+ path: s.path,
298
+ bytes: Buffer.byteLength(s.text, "utf8"),
299
+ }));
300
+ }
301
+ }
302
+ }
303
+ // 3. The run context (the workflow definition is resolved once, above).
304
+ const ctx = {
305
+ owner,
306
+ repo: name,
307
+ issueNumber,
308
+ issueTitle: (isPrReview ? inst.pr?.title : inst.issue?.title) ?? inst.instance_id,
309
+ issueBody: (isPrReview ? inst.pr?.body : inst.issue?.body) ?? inst.problem_statement,
310
+ issueLabels: inst.issue?.labels ?? [],
311
+ commentBody: "",
312
+ sender: "eval",
313
+ branch,
314
+ taskId,
315
+ issueDir: `.lastlight/issue-${issueNumber}`,
316
+ bootstrapLabel: "lastlight:bootstrap",
317
+ // pr-review's Context block keys off `prNumber` — the skill goes straight
318
+ // to github_get_pull_request when it's set (buildPhasePrompt dumps every
319
+ // defined ctx field into the "Context:" block). `baseBranch` is what the
320
+ // deterministic `post-review` phase reads to compute the commentable diff
321
+ // (`git diff origin/<baseBranch>...HEAD`) — WITHOUT it every finding is
322
+ // demoted to the review body and the line-anchored inline-comment path
323
+ // (the point of the tier) never fires. Prod sets it from the PR's base ref;
324
+ // the eval must too, or it diverges from what ships.
325
+ ...(isPrReview && inst.pr
326
+ ? { prNumber: inst.pr.number, prTitle: inst.pr.title, baseBranch: inst.pr.base_ref }
327
+ : {}),
328
+ // No prePopulateBranch → the runner never clones from GitHub; the agent
329
+ // works in the dir we seeded above (or an empty dir for triage).
330
+ };
331
+ // 3a. The PR state machine's projection (issues #251, #252).
332
+ //
333
+ // A PR-scoped workflow is dispatched in production, never called: the
334
+ // dispatcher resolves one `PrState` snapshot and `renderContext` projects it
335
+ // into the context. That projection IS what the fix and merge prompts reason
336
+ // with — `{{ciSection}}`, `{{attempt}}`, `{{mayMerge}}`, `{{priorNotes}}`,
337
+ // `{{verifyScript}}` — so running them off a hand-built context measures a
338
+ // workflow production does not have. `./pr-context.ts` builds the snapshot a
339
+ // case seeds and hands it to CORE's projection, unmodified.
340
+ //
341
+ // Gated on the workflow's own `pr_scoped: true` metadata rather than a name
342
+ // list here — the same fact core derives `prScopedWorkflows()` from, so an
343
+ // overlay's forked fix workflow is covered without a change to this file.
344
+ //
345
+ // `pr-review` USED TO BE excluded here, and the exclusion was about scores,
346
+ // not about correctness: pr-review is judge-scored and its numbers are
347
+ // compared across runs and against Martian's leaderboard, so enriching its
348
+ // context would move every historical figure as a side effect of a change
349
+ // that was not about them.
350
+ //
351
+ // The exclusion was LIFTED DELIBERATELY on 2026-08-22. It had made the
352
+ // review evidence pipeline unmeasurable on the only tier its gates are read
353
+ // on: with no `prContextPatch`, core's `renderContext` never runs, the
354
+ // context never gets `analysisEnabled`, and every WP3 phase in
355
+ // `pr-review.yaml` matches `skip_if: "analysisEnabled != true"` and skips.
356
+ // WP0's `{{specObligations}}` was unmeasurable there for the same reason.
357
+ // The choice was between a pipeline that cannot be measured and a baseline
358
+ // that has to be re-run; the baseline is being re-run.
359
+ //
360
+ // THEREFORE: every pr-review number produced BEFORE 2026-08-22 was measured
361
+ // on a different template context and must NOT be compared across that
362
+ // boundary — not in `diff-runs.ts`, not against `2026-08-20_074355`, not
363
+ // against the leaderboard entry that run backed. Re-baseline instead.
364
+ const wantsPrContext = def.pr_scoped === true || !!inst.pr_state;
365
+ if (wantsPrContext) {
366
+ Object.assign(ctx, await prContextPatch({
367
+ repo: `${owner}/${name}`,
368
+ prNumber: inst.pr?.number ?? issueNumber,
369
+ title: inst.pr?.title ?? inst.issue?.title ?? inst.instance_id,
370
+ body: inst.pr?.body ?? inst.issue?.body ?? inst.problem_statement,
371
+ branch,
372
+ seed: round.seed,
373
+ baseRef: inst.pr?.base_ref,
374
+ // The real head — the round's, or the PR's (`caseHeadSha`): never the
375
+ // placeholder when the case names a real commit.
376
+ ...(isRealSha(round.head) ? { headSha: round.head } : {}),
377
+ // A chained case projects the snapshot itself too (`ctx.prState`), as
378
+ // `dispatchWorkflow` does: post-review reads the ledger it was
379
+ // dispatched with off it. Single-round cases keep the old context.
380
+ snapshot: !!rounds,
381
+ // A REAL `GitHubClient` pointed at the fake — the same construction
382
+ // `post-review` already uses against the mock. Core's own
383
+ // `resolveSpecContext` then reads BOTH ends of the spec axis through
384
+ // it, so the eval exercises the production code path (GraphQL
385
+ // `closingIssuesReferences` + `GET /pulls/:n/files`) rather than a
386
+ // harness copy of it. `setPullFiles` above is what the second read
387
+ // hits, so it must already have run — it has.
388
+ github: resolveReviewGitHubClient({ githubApiBaseUrl: fake.url }),
389
+ // Retained as the fallback for a tier with no live client: a case that
390
+ // seeds `pr_state.changed_files` still wins — including seeding `[]`,
391
+ // which asserts "this PR changes nothing" rather than "we could not
392
+ // read it". Those must stay distinguishable (locked decision 6).
393
+ changedFiles: prFilePaths,
394
+ // The arm's own `review:` policy — the overlay's, never gold's. This
395
+ // is the seam that turns the evidence pipeline on for the `wp3` arm
396
+ // and leaves it off for `baseline`, with no per-case special-casing:
397
+ // `baseline/config.yaml` simply declares no `analysis` block.
398
+ review: opts.arm.review,
399
+ }));
400
+ }
401
+ // The arm supplies model selection in one shot: it patches `ctx.models`/
402
+ // `ctx.variants` (config arms — EXACTLY as production's `simple.js`, so phase
403
+ // `model: "{{models.X}}"` templates resolve) and returns the executor model
404
+ // plus the `runWorkflow` `models`/`variants` args. `models` arms leave the
405
+ // context untouched and return just their forced id.
406
+ const prepared = opts.arm.prepare(ctx);
407
+ // 3b. The target repo's `.lastlight/` config layer (issue #180), resolved
408
+ // through core's OWN dispatch-time resolver against the mock — fetch →
409
+ // sanitize → unpack → merge, unmodified. Undefined for a repo with no
410
+ // `.lastlight/`, in which case `runWorkflow` below is called exactly as
411
+ // it was before the feature existed. Never throws: the resolver's whole
412
+ // contract is "warn, drop the bad bits, run anyway".
413
+ const repoRun = await resolveEvalRepoConfig({
414
+ repo: `${owner}/${name}`,
415
+ workflowName,
416
+ client: fake,
417
+ models: prepared.models,
418
+ variants: prepared.variants,
419
+ defaultModel: prepared.model,
420
+ cacheRoot: join(stateDir, "repo-config"),
252
421
  });
253
- if (sources.length) {
254
- const combined = sources.map((s) => s.text.trim()).filter(Boolean).join("\n\n");
255
- if (injectRepoContext(seed.workDir, combined)) {
256
- result.injectedContext = sources.map((s) => ({
257
- source: s.source,
258
- path: s.path,
259
- bytes: Buffer.byteLength(s.text, "utf8"),
260
- }));
422
+ if (repoRun.repoConfig) {
423
+ result.repoLayer = {
424
+ repo: repoRun.repoConfig.repo,
425
+ defaultBranch: repoRun.repoConfig.defaultBranch,
426
+ treeSha: repoRun.repoConfig.treeSha,
427
+ assets: [...repoRun.repoConfig.assets],
428
+ applied: appliedRepoConfigKeys(repoRun.repoConfig),
429
+ warnings: repoRun.repoConfig.warnings.map((w) => `${w.code}: ${w.message}`),
430
+ };
431
+ }
432
+ // The repo opting ITSELF out of this workflow in `.lastlight/lastlight.yml`.
433
+ // Production abandons the dispatch here — no run, no agent call — so the
434
+ // case is `blocked` (a deliberate measured outcome), not an error.
435
+ if (repoRun.refusal) {
436
+ result.blocked = true;
437
+ result.repoLayer = { ...(result.repoLayer ?? { repo: `${owner}/${name}` }), refused: repoRun.refusal };
438
+ result.behavioral = gradeBehavioral(inst.expect_github, fake, { issueNumber, branch });
439
+ result.githubMutations = fake.calls.length;
440
+ return { refused: true };
441
+ }
442
+ const config = {
443
+ sandbox: opts.sandbox ?? "none",
444
+ stateDir,
445
+ sessionsDir: round.sessionsDir,
446
+ // Run the agent inside the pre-seeded `<workspace>/<repo>/` checkout (only
447
+ // when we actually seeded one), matching production's nested layout. Core
448
+ // nests `agentCwd` here without a clone; AGENTS.md/.lastlight-skills stay
449
+ // at the workspace root, siblings outside the repo.
450
+ repoSubdir: seed ? repoSubdir : undefined,
451
+ // `config` arms let core pick per phase (this is only the fallback for
452
+ // phases that resolve to nothing — the merged config's `default`); `models`
453
+ // arms force their one id across every step.
454
+ model: prepared.model,
455
+ githubApiBaseUrl: fake.url,
456
+ // Eval workflows shouldn't reach the network beyond the model + fake GH.
457
+ webSearch: false,
458
+ };
459
+ // Phase windows: `onPhaseStart`/`onPhaseEnd` bracket each phase, and the
460
+ // pair is what makes a phase's duration MEASURED rather than inferred from
461
+ // the next phase's start — which would silently bill the gap between phases
462
+ // (workspace refresh, the `until_bash` container spin-up) to whichever phase
463
+ // happened to precede it.
464
+ //
465
+ // `phaseStarts` additionally backs the FALLBACK attribution rule in
466
+ // `bucketSessionsByPhase`. Sessions now carry their owning phase as a stamp,
467
+ // so the windows are only consulted for jsonl archived before that stamp
468
+ // existed; see that function for why a start-time lookup cannot attribute a
469
+ // fan-out at all.
470
+ const phaseStarts = [];
471
+ const phaseWindows = new Map();
472
+ const callbacks = {
473
+ onPhaseStart: async (phase) => {
474
+ const now = Date.now();
475
+ phaseStarts.push({ phase, start: now });
476
+ // First start wins: a label the engine re-announces (a loop node whose
477
+ // condition-met entry repeats it) must not restart its own clock.
478
+ if (!phaseWindows.has(phase))
479
+ phaseWindows.set(phase, { start: now });
480
+ },
481
+ onPhaseEnd: async (phase) => {
482
+ const w = phaseWindows.get(phase);
483
+ if (w)
484
+ w.end = Date.now();
485
+ },
486
+ };
487
+ const fullFile = trialDir ? join(trialDir, "full.jsonl") : undefined;
488
+ // Flush the consolidated transcript atomically (so a polling dashboard never
489
+ // reads a half-written file): on a timer while running (follow-along), and
490
+ // once at the end. Best-effort — a flush failure must never affect the run.
491
+ const flushFull = () => {
492
+ if (!fullFile || !trialDir)
493
+ return;
494
+ try {
495
+ const log = readSessionLog(round.sessionsDir);
496
+ if (!log)
497
+ return;
498
+ mkdirSync(trialDir, { recursive: true });
499
+ const tmp = `${fullFile}.tmp`;
500
+ writeFileSync(tmp, log);
501
+ renameSync(tmp, fullFile);
261
502
  }
503
+ catch {
504
+ /* best-effort */
505
+ }
506
+ };
507
+ // 4. Run. Empty approvalConfig (7th arg) → every approval gate is disabled.
508
+ // The arm's prepared maps go to args 6 (models) and 9 (variants), matching
509
+ // prod's runWorkflow call; `models` arms leave both undefined so every phase
510
+ // falls back to config.model (one model everywhere). The 10th arg is the
511
+ // repo layer — `undefined` for a repo with no `.lastlight/`, which is the
512
+ // pre-#180 call byte-for-byte. The 5th/8th (store, workflow id) are unset
513
+ // for a single-round case — no db, so every gate is inert — and, for a
514
+ // chained case, the in-memory run store post-review persists the review
515
+ // ledger to (`rereview-node.ts`); `approvalConfig` stays empty either way.
516
+ const flushTimer = fullFile ? setInterval(flushFull, 1000) : undefined;
517
+ let wf;
518
+ try {
519
+ wf = await runWorkflow(def, ctx, config, callbacks, round.store, prepared.models, {}, round.workflowId, prepared.variants, repoRun.repoConfig);
262
520
  }
263
- }
264
- // 3. Real workflow definition + run context.
265
- const def = getWorkflow(workflowName);
266
- const ctx = {
267
- owner,
268
- repo: name,
269
- issueNumber,
270
- issueTitle: (isPrReview ? inst.pr?.title : inst.issue?.title) ?? inst.instance_id,
271
- issueBody: (isPrReview ? inst.pr?.body : inst.issue?.body) ?? inst.problem_statement,
272
- issueLabels: inst.issue?.labels ?? [],
273
- commentBody: "",
274
- sender: "eval",
275
- branch,
276
- taskId,
277
- issueDir: `.lastlight/issue-${issueNumber}`,
278
- bootstrapLabel: "lastlight:bootstrap",
279
- // pr-review's Context block keys off `prNumber` — the skill goes straight
280
- // to github_get_pull_request when it's set (buildPhasePrompt dumps every
281
- // defined ctx field into the "Context:" block). `baseBranch` is what the
282
- // deterministic `post-review` phase reads to compute the commentable diff
283
- // (`git diff origin/<baseBranch>...HEAD`) — WITHOUT it every finding is
284
- // demoted to the review body and the line-anchored inline-comment path
285
- // (the point of the tier) never fires. Prod sets it from the PR's base ref;
286
- // the eval must too, or it diverges from what ships.
287
- ...(isPrReview && inst.pr
288
- ? { prNumber: inst.pr.number, prTitle: inst.pr.title, baseBranch: inst.pr.base_ref }
289
- : {}),
290
- // No prePopulateBranch → the runner never clones from GitHub; the agent
291
- // works in the dir we seeded above (or an empty dir for triage).
521
+ finally {
522
+ if (flushTimer)
523
+ clearInterval(flushTimer);
524
+ }
525
+ return { wf, ctx, prepared, phaseStarts, phaseWindows, flushFull, fullFile };
292
526
  };
293
- // 3a. The PR state machine's projection (issues #251, #252).
294
- //
295
- // A PR-scoped workflow is dispatched in production, never called: the
296
- // dispatcher resolves one `PrState` snapshot and `renderContext` projects it
297
- // into the context. That projection IS what the fix and merge prompts reason
298
- // with — `{{ciSection}}`, `{{attempt}}`, `{{mayMerge}}`, `{{priorNotes}}`,
299
- // `{{verifyScript}}` — so running them off a hand-built context measures a
300
- // workflow production does not have. `./pr-context.ts` builds the snapshot a
301
- // case seeds and hands it to CORE's projection, unmodified.
302
- //
303
- // Gated on the workflow's own `pr_scoped: true` metadata rather than a name
304
- // list here — the same fact core derives `prScopedWorkflows()` from, so an
305
- // overlay's forked fix workflow is covered without a change to this file.
527
+ // ── Chained re-review rounds (issue #429) ──────────────────────────────
306
528
  //
307
- // `pr-review` USED TO BE excluded here, and the exclusion was about scores,
308
- // not about correctness: pr-review is judge-scored and its numbers are
309
- // compared across runs and against Martian's leaderboard, so enriching its
310
- // context would move every historical figure as a side effect of a change
311
- // that was not about them.
312
- //
313
- // The exclusion was LIFTED DELIBERATELY on 2026-08-22. It had made the
314
- // review evidence pipeline unmeasurable on the only tier its gates are read
315
- // on: with no `prContextPatch`, core's `renderContext` never runs, the
316
- // context never gets `analysisEnabled`, and every WP3 phase in
317
- // `pr-review.yaml` matches `skip_if: "analysisEnabled != true"` and skips.
318
- // WP0's `{{specObligations}}` was unmeasurable there for the same reason.
319
- // The choice was between a pipeline that cannot be measured and a baseline
320
- // that has to be re-run; the baseline is being re-run.
321
- //
322
- // THEREFORE: every pr-review number produced BEFORE 2026-08-22 was measured
323
- // on a different template context and must NOT be compared across that
324
- // boundary — not in `diff-runs.ts`, not against `2026-08-20_074355`, not
325
- // against the leaderboard entry that run backed. Re-baseline instead.
326
- const wantsPrContext = def.pr_scoped === true || !!inst.pr_state;
327
- if (wantsPrContext) {
328
- Object.assign(ctx, await prContextPatch({
329
- repo: `${owner}/${name}`,
330
- prNumber: inst.pr?.number ?? issueNumber,
331
- title: inst.pr?.title ?? inst.issue?.title ?? inst.instance_id,
332
- body: inst.pr?.body ?? inst.issue?.body ?? inst.problem_statement,
333
- branch,
334
- seed: inst.pr_state,
335
- baseRef: inst.pr?.base_ref,
336
- // A REAL `GitHubClient` pointed at the fake — the same construction
337
- // `post-review` already uses against the mock. Core's own
338
- // `resolveSpecContext` then reads BOTH ends of the spec axis through
339
- // it, so the eval exercises the production code path (GraphQL
340
- // `closingIssuesReferences` + `GET /pulls/:n/files`) rather than a
341
- // harness copy of it. `setPullFiles` above is what the second read
342
- // hits, so it must already have run — it has.
343
- github: resolveReviewGitHubClient({ githubApiBaseUrl: fake.url }),
344
- // Retained as the fallback for a tier with no live client: a case that
345
- // seeds `pr_state.changed_files` still wins — including seeding `[]`,
346
- // which asserts "this PR changes nothing" rather than "we could not
347
- // read it". Those must stay distinguishable (locked decision 6).
348
- changedFiles: prFilePaths,
349
- // The arm's own `review:` policy — the overlay's, never gold's. This
350
- // is the seam that turns the evidence pipeline on for the `wp3` arm
351
- // and leaves it off for `baseline`, with no per-case special-casing:
352
- // `baseline/config.yaml` simply declares no `analysis` block.
353
- review: opts.arm.review,
354
- }));
529
+ // A case with `rounds` runs the real workflow once per earlier head, in
530
+ // order, before the scored last round below — one fake GitHub (round k's
531
+ // review and threads are what round k+1 reads), one per-PR workspace
532
+ // (checked out to each head, its `.lastlight/pr-review/` carried as a
533
+ // reused production workspace carries it), and one in-memory run store, so
534
+ // post-review persists the review ledger to the round's run scratch and the
535
+ // next round is dispatched with it through core's `deriveReviewLedger`.
536
+ const store = rounds ? createRoundStore() : undefined;
537
+ const roundRecords = [];
538
+ let roundSeed = inst.pr_state;
539
+ let currentHead = rounds ? rounds[0].head_commit : caseHeadSha(inst);
540
+ const prDir = join(repoDir, ".lastlight", "pr-review");
541
+ const heads = rounds?.map((r) => r.head_commit) ?? [];
542
+ const factsBin = rounds ? resolveFactsBin() : null;
543
+ const oracle = rounds && factsBin && inst.pr
544
+ ? new UnitsOracle({ repoDir, base: inst.pr.base_commit, factsBin, root: join(stateDir, "rereview-oracle") })
545
+ : undefined;
546
+ if (rounds && inst.pr && seed) {
547
+ fake.setOutdatedResolver(outdatedResolver(repoDir, () => currentHead));
355
548
  }
356
- // The arm supplies model selection in one shot: it patches `ctx.models`/
357
- // `ctx.variants` (config arms — EXACTLY as production's `simple.js`, so phase
358
- // `model: "{{models.X}}"` templates resolve) and returns the executor model
359
- // plus the `runWorkflow` `models`/`variants` args. `models` arms leave the
360
- // context untouched and return just their forced id.
361
- const prepared = opts.arm.prepare(ctx);
362
- // 3b. The target repo's `.lastlight/` config layer (issue #180), resolved
363
- // through core's OWN dispatch-time resolver against the mock — fetch →
364
- // sanitize → unpack → merge, unmodified. Undefined for a repo with no
365
- // `.lastlight/`, in which case `runWorkflow` below is called exactly as
366
- // it was before the feature existed. Never throws: the resolver's whole
367
- // contract is "warn, drop the bad bits, run anyway".
368
- const repoRun = await resolveEvalRepoConfig({
369
- repo: `${owner}/${name}`,
370
- workflowName,
371
- client: fake,
372
- models: prepared.models,
373
- variants: prepared.variants,
374
- defaultModel: prepared.model,
375
- cacheRoot: join(stateDir, "repo-config"),
376
- });
377
- if (repoRun.repoConfig) {
378
- result.repoLayer = {
379
- repo: repoRun.repoConfig.repo,
380
- defaultBranch: repoRun.repoConfig.defaultBranch,
381
- treeSha: repoRun.repoConfig.treeSha,
382
- assets: [...repoRun.repoConfig.assets],
383
- applied: appliedRepoConfigKeys(repoRun.repoConfig),
384
- warnings: repoRun.repoConfig.warnings.map((w) => `${w.code}: ${w.message}`),
549
+ /**
550
+ * What one finished round measured — everything but the grade and the
551
+ * spend, which the caller adds (the last round's come from the case's own).
552
+ */
553
+ const measureRound = async (k, before, wfOk) => {
554
+ const head = heads[k];
555
+ const reviews = fake.submittedReviews(inst.pr.number);
556
+ const inline = reviews.flatMap((r) => r.comments);
557
+ const disposition = readFreshJson(prDir, "disposition.json", before);
558
+ const scratch = store ? await roundScratch(store, roundWorkflowId(k)) : {};
559
+ const record = {
560
+ round: k + 1,
561
+ headSha: head,
562
+ ...(rounds[k].label ? { label: rounds[k].label } : {}),
563
+ workflowSucceeded: wfOk.success,
564
+ ...(wfOk.error ? { error: wfOk.error } : {}),
565
+ ...(reviews.length ? { event: reviews.at(-1).event } : {}),
566
+ inlinePosted: inline.length,
567
+ ...(disposition?.findings ? dispositionCounts(disposition.findings) : {}),
568
+ costUsd: 0,
569
+ inputTokens: 0,
570
+ cachedTokens: 0,
571
+ outputTokens: 0,
572
+ durationMs: 0,
385
573
  };
386
- }
387
- // The repo opting ITSELF out of this workflow in `.lastlight/lastlight.yml`.
388
- // Production abandons the dispatch here — no run, no agent call — so the
389
- // case is `blocked` (a deliberate measured outcome), not an error.
390
- if (repoRun.refusal) {
391
- result.blocked = true;
392
- result.repoLayer = { ...(result.repoLayer ?? { repo: `${owner}/${name}` }), refused: repoRun.refusal };
393
- result.behavioral = gradeBehavioral(inst.expect_github, fake, { issueNumber, branch });
394
- result.githubMutations = fake.calls.length;
395
- return result;
396
- }
397
- const config = {
398
- sandbox: opts.sandbox ?? "none",
399
- stateDir,
400
- sessionsDir,
401
- // Run the agent inside the pre-seeded `<workspace>/<repo>/` checkout (only
402
- // when we actually seeded one), matching production's nested layout. Core
403
- // nests `agentCwd` here without a clone; AGENTS.md/.lastlight-skills stay
404
- // at the workspace root, siblings outside the repo.
405
- repoSubdir: seed ? repoSubdir : undefined,
406
- // `config` arms let core pick per phase (this is only the fallback for
407
- // phases that resolve to nothing — the merged config's `default`); `models`
408
- // arms force their one id across every step.
409
- model: prepared.model,
410
- githubApiBaseUrl: fake.url,
411
- // Eval workflows shouldn't reach the network beyond the model + fake GH.
412
- webSearch: false,
413
- };
414
- // Phase windows: `onPhaseStart`/`onPhaseEnd` bracket each phase, and the
415
- // pair is what makes a phase's duration MEASURED rather than inferred from
416
- // the next phase's start — which would silently bill the gap between phases
417
- // (workspace refresh, the `until_bash` container spin-up) to whichever phase
418
- // happened to precede it.
419
- //
420
- // `phaseStarts` additionally backs the FALLBACK attribution rule in
421
- // `bucketSessionsByPhase`. Sessions now carry their owning phase as a stamp,
422
- // so the windows are only consulted for jsonl archived before that stamp
423
- // existed; see that function for why a start-time lookup cannot attribute a
424
- // fan-out at all.
425
- const phaseStarts = [];
426
- const phaseWindows = new Map();
427
- const callbacks = {
428
- onPhaseStart: async (phase) => {
429
- const now = Date.now();
430
- phaseStarts.push({ phase, start: now });
431
- // First start wins: a label the engine re-announces (a loop node whose
432
- // condition-met entry repeats it) must not restart its own clock.
433
- if (!phaseWindows.has(phase))
434
- phaseWindows.set(phase, { start: now });
435
- },
436
- onPhaseEnd: async (phase) => {
437
- const w = phaseWindows.get(phase);
438
- if (w)
439
- w.end = Date.now();
440
- },
574
+ const coverage = coverageSummary(scratchCoverage(scratch) ?? readFreshJson(prDir, "review-coverage.json", before));
575
+ if (coverage)
576
+ record.coverage = coverage;
577
+ const ledger = ledgerCounts(scratchLedger(scratch));
578
+ if (ledger)
579
+ record.ledger = ledger;
580
+ // Late discoveries: round ≥ 2 only. The round's OWN units and prior
581
+ // review when its pipeline cut them (what its convergence gate saw);
582
+ // otherwise the harness's oracle cut over the same two heads, so an arm
583
+ // that runs no pipeline is measured on the same instrument.
584
+ if (k >= 1) {
585
+ const ownUnits = readFreshJson(prDir, "units.json", before);
586
+ const ownPrior = readFreshJson(prDir, "prior-review.json", before);
587
+ let units;
588
+ let prior;
589
+ if (ownUnits?.units?.length && ownPrior) {
590
+ units = ownUnits.units;
591
+ prior = ownPrior;
592
+ record.lateDiscoverySource = "units.json";
593
+ }
594
+ else if (oracle) {
595
+ try {
596
+ units = oracle.at(heads, k).units;
597
+ prior = oracle.priorFor(heads, k);
598
+ record.lateDiscoverySource = "oracle";
599
+ }
600
+ catch (err) {
601
+ record.lateDiscoveryUnavailable = `oracle units cut failed: ${err.message.slice(0, 200)}`;
602
+ }
603
+ }
604
+ else {
605
+ record.lateDiscoveryUnavailable = "no units.json/prior-review.json this round and no lastlight-facts binary for the oracle";
606
+ }
607
+ if (units && prior !== undefined) {
608
+ const verdicts = judgeComments({
609
+ comments: inline.map((c) => ({ path: c.path, ...(c.line !== undefined ? { line: c.line } : {}), ...(c.start_line !== undefined ? { start_line: c.start_line } : {}) })),
610
+ prior,
611
+ units,
612
+ fileText: (path) => fileAt(repoDir, head, path),
613
+ });
614
+ record.lateDiscovery = verdicts.filter((v) => v.verdict === "unchanged").length;
615
+ record.lateDiscoveryOf = verdicts.filter((v) => v.verdict !== null).length;
616
+ const delta = deltaCounts(units);
617
+ if (delta)
618
+ record.delta = delta;
619
+ roundVerdicts.set(k, verdicts);
620
+ }
621
+ }
622
+ return { record, reviews, scratch, disposition };
441
623
  };
442
- const trialDir = opts.sessionTrialDir;
443
- const fullFile = trialDir ? join(trialDir, "full.jsonl") : undefined;
444
- // Flush the consolidated transcript atomically (so a polling dashboard never
445
- // reads a half-written file): on a timer while running (follow-along), and
446
- // once at the end. Best-effort — a flush failure must never affect the run.
447
- const flushFull = () => {
448
- if (!fullFile || !trialDir)
449
- return;
624
+ const roundWorkflowId = (k) => `${taskId}:round-${k + 1}`;
625
+ const roundVerdicts = new Map();
626
+ /** A round's evidence, beside its artifacts: what it posted, its dispositions, the ledger it left, the late-discovery verdicts. */
627
+ const archiveRound = (k, m, extra = {}) => {
628
+ if (!trialDir)
629
+ return undefined;
630
+ const dir = join(trialDir, `round-${k + 1}`);
450
631
  try {
451
- const log = readSessionLog(sessionsDir);
452
- if (!log)
453
- return;
454
- mkdirSync(trialDir, { recursive: true });
455
- const tmp = `${fullFile}.tmp`;
456
- writeFileSync(tmp, log);
457
- renameSync(tmp, fullFile);
632
+ mkdirSync(dir, { recursive: true });
633
+ if (extra.sessionsDir) {
634
+ const log = readSessionLog(extra.sessionsDir);
635
+ if (log)
636
+ writeFileSync(join(dir, "full.jsonl"), log);
637
+ persistPipelineArtifacts(repoDir, dir);
638
+ }
639
+ writeFileSync(join(dir, "round.json"), `${JSON.stringify({
640
+ round: k + 1,
641
+ head: heads[k],
642
+ reviews: m.reviews,
643
+ disposition: m.disposition ?? null,
644
+ reviewLedger: scratchLedger(m.scratch) ?? null,
645
+ reviewCoverage: scratchCoverage(m.scratch) ?? null,
646
+ lateDiscovery: roundVerdicts.get(k) ?? null,
647
+ dispatched: roundSeed ?? null,
648
+ }, null, 2)}\n`);
649
+ return `${opts.sessionTrialRel ?? trialDir}/round-${k + 1}`;
458
650
  }
459
651
  catch {
460
- /* best-effort */
652
+ return undefined;
461
653
  }
462
654
  };
463
- // 4. Run. Empty approvalConfig (7th arg) → every approval gate is disabled.
464
- // The arm's prepared maps go to args 6 (models) and 9 (variants), matching
465
- // prod's runWorkflow call; `models` arms leave both undefined so every phase
466
- // falls back to config.model (one model everywhere). The 10th arg is the
467
- // repo layer — `undefined` for a repo with no `.lastlight/`, which is the
468
- // pre-#180 call byte-for-byte.
469
- const flushTimer = fullFile ? setInterval(flushFull, 1000) : undefined;
470
- let wf;
471
- try {
472
- wf = await runWorkflow(def, ctx, config, callbacks, undefined, prepared.models, {}, undefined, prepared.variants, repoRun.repoConfig);
473
- }
474
- finally {
475
- if (flushTimer)
476
- clearInterval(flushTimer);
655
+ if (rounds && inst.pr && seed) {
656
+ for (let k = 0; k < rounds.length - 1; k++) {
657
+ const head = heads[k];
658
+ if (k > 0) {
659
+ checkoutRound(repoDir, seed.branch, head);
660
+ fake.advanceClock(ROUND_GAP_MS);
661
+ }
662
+ currentHead = head;
663
+ fake.setHead(inst.pr.number, head);
664
+ fake.startRound(k + 1);
665
+ const sessionsDirK = join(stateDir, `agent-sessions-round-${k + 1}`);
666
+ mkdirSync(join(sessionsDirK, "projects"), { recursive: true });
667
+ const before = snapshotMtimes(prDir, ROUND_ARTIFACTS);
668
+ const startedAt = Date.now();
669
+ const seedK = { ...(roundSeed ?? {}), head_sha: head };
670
+ roundSeed = seedK;
671
+ const ran = await executeRound({ head, seed: seedK, sessionsDir: sessionsDirK, store, workflowId: roundWorkflowId(k) });
672
+ if ("refused" in ran)
673
+ return result;
674
+ await drainSessions(sessionsDirK);
675
+ const failed = ran.wf.phases.find((p) => !p.success && p.error);
676
+ const m = await measureRound(k, before, {
677
+ success: ran.wf.success,
678
+ ...(failed?.error ? { error: `${failed.phase}: ${failed.error}`.slice(0, 300) } : {}),
679
+ });
680
+ const cost = collectMetrics(sessionsDirK, modelCost(ran.prepared.model));
681
+ Object.assign(m.record, {
682
+ costUsd: cost.costUsd,
683
+ inputTokens: cost.inputTokens,
684
+ cachedTokens: cost.cachedTokens,
685
+ outputTokens: cost.outputTokens,
686
+ durationMs: Date.now() - startedAt,
687
+ });
688
+ // The round's own grade, for cumulative recall — the same judge and
689
+ // gold as the scored round, over the review THIS round posted.
690
+ if (inst.review_gold) {
691
+ const rg = await gradeReview({ gold: inst.review_gold, reviews: m.reviews, beta: opts.judge?.beta, neutralGold: inst.review_gold_neutral });
692
+ const matched = goldMatchedOf(rg.trace);
693
+ if (!rg.error && matched) {
694
+ m.record.goldMatched = matched;
695
+ m.record.postedFindings = rg.posted;
696
+ }
697
+ }
698
+ const rel = archiveRound(k, m, { sessionsDir: sessionsDirK });
699
+ if (rel)
700
+ m.record.artifactRel = rel;
701
+ roundRecords.push(m.record);
702
+ // Recorded as it goes, so a later round that throws still leaves the
703
+ // rounds that ran on the result (re-rolled with the last round below).
704
+ result.rereview = rollupRereview(roundRecords, inst.review_gold?.length);
705
+ // What the next round is dispatched with.
706
+ roundSeed = {
707
+ ...(inst.pr_state ?? {}),
708
+ ...carryForward({
709
+ repoDir,
710
+ baseCommit: inst.pr.base_commit,
711
+ prevHead: head,
712
+ nextHead: heads[k + 1],
713
+ reviews: m.reviews,
714
+ prev: seedK,
715
+ scratch: m.scratch,
716
+ }),
717
+ };
718
+ }
719
+ // The scored round: the PR's own head.
720
+ checkoutRound(repoDir, seed.branch, inst.pr.head_commit);
721
+ fake.advanceClock(ROUND_GAP_MS);
722
+ currentHead = inst.pr.head_commit;
723
+ fake.setHead(inst.pr.number, inst.pr.head_commit);
724
+ fake.startRound(rounds.length);
477
725
  }
726
+ const finalBefore = rounds ? snapshotMtimes(prDir, ROUND_ARTIFACTS) : undefined;
727
+ const finalStartedAt = Date.now();
728
+ const ran = await executeRound({
729
+ // Single-round: the PR's own head (its files are what GET /pulls/:n/files
730
+ // always served), else the snapshot's head for a PR-less case.
731
+ head: inst.pr?.head_commit ?? caseHeadSha(inst),
732
+ seed: roundSeed,
733
+ sessionsDir,
734
+ ...(store ? { store, workflowId: roundWorkflowId(rounds.length - 1) } : {}),
735
+ });
736
+ if ("refused" in ran)
737
+ return result;
738
+ const { wf, ctx, prepared, phaseStarts, phaseWindows, flushFull, fullFile } = ran;
478
739
  result.workflowSucceeded = wf.success;
479
740
  // Record the model each phase resolved to — the arm forced id in `models`
480
741
  // mode, or the per-step model the merged config assigned in `config` mode
@@ -483,7 +744,7 @@ export async function runInstance(inst, opts) {
483
744
  // template lookup goes through modelTemplateForRow (phase-models.ts), which
484
745
  // parses branch rows back to their declaration via core's PhaseRef.
485
746
  result.phases = wf.phases.map((p) => {
486
- const { template, fallbackPhase } = modelTemplateForRow(def.phases, p.phase);
747
+ const { template, fallbackPhase } = modelTemplateForRow(def.phases, p.phase, p.modelTemplate);
487
748
  return {
488
749
  phase: p.phase,
489
750
  success: p.success,
@@ -584,6 +845,12 @@ export async function runInstance(inst, opts) {
584
845
  try {
585
846
  if (persistPipelineArtifacts(repoDir, opts.sessionTrialDir) && opts.sessionTrialRel)
586
847
  result.pipelineArtifactRel = `${opts.sessionTrialRel}/pr-review`;
848
+ // The prompt context a phase REPLAY needs and cannot rebuild from
849
+ // the artifacts: what `select` was told about the PR's earlier
850
+ // conversation and review ledger (`phase-replay-node.ts`'s
851
+ // `readStoredRunContext`). Beside `pr-review/`, never inside it — a
852
+ // replay copies that dir into the checkout the agent works in.
853
+ writeStoredRunContext(opts.sessionTrialDir, ctx);
587
854
  }
588
855
  catch (err) {
589
856
  // Never fail a measured run over its own bookkeeping — but a warning
@@ -740,6 +1007,29 @@ export async function runInstance(inst, opts) {
740
1007
  /* leave sessionTrial unset */
741
1008
  }
742
1009
  }
1010
+ // 5e. The chained case's last round, measured like the others, and the
1011
+ // roll-up. Its grade, spend and artifacts are the case's own (above).
1012
+ if (rounds && finalBefore) {
1013
+ const k = rounds.length - 1;
1014
+ const m = await measureRound(k, finalBefore, { success: wf.success, ...(result.error ? { error: result.error } : {}) });
1015
+ Object.assign(m.record, {
1016
+ costUsd: result.costUsd,
1017
+ inputTokens: result.inputTokens,
1018
+ cachedTokens: result.cachedTokens,
1019
+ outputTokens: result.outputTokens,
1020
+ durationMs: Date.now() - finalStartedAt,
1021
+ });
1022
+ const matched = result.review && !result.error?.startsWith("review judge") ? goldMatchedOf(result.review.trace) : null;
1023
+ if (matched) {
1024
+ m.record.goldMatched = matched;
1025
+ m.record.postedFindings = result.review.posted;
1026
+ }
1027
+ const rel = archiveRound(k, m);
1028
+ if (rel)
1029
+ m.record.artifactRel = rel;
1030
+ roundRecords.push(m.record);
1031
+ result.rereview = rollupRereview(roundRecords, inst.review_gold?.length);
1032
+ }
743
1033
  }
744
1034
  catch (err) {
745
1035
  result.error = err instanceof Error ? err.message : String(err);