lastlight-evals 0.16.0 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +51 -0
- package/dashboard/dist/assets/{index-DSu_AxvO.js → index-BiY2Jj1T.js} +38 -38
- package/dashboard/dist/assets/index-Dzo-4g9I.css +1 -0
- package/dashboard/dist/index.html +2 -2
- package/dist/fake-github.js +111 -20
- package/dist/fake-github.js.map +1 -1
- package/dist/mechanism.test.js +9 -0
- package/dist/mechanism.test.js.map +1 -1
- package/dist/phase-models.js +14 -4
- package/dist/phase-models.js.map +1 -1
- package/dist/phase-replay-context.js +60 -0
- package/dist/phase-replay-context.js.map +1 -0
- package/dist/phase-replay-node.js +12 -3
- package/dist/phase-replay-node.js.map +1 -1
- package/dist/pr-context.js +29 -9
- package/dist/pr-context.js.map +1 -1
- package/dist/rereview-node.js +320 -0
- package/dist/rereview-node.js.map +1 -0
- package/dist/rereview.js +199 -0
- package/dist/rereview.js.map +1 -0
- package/dist/rereview.test.js +512 -0
- package/dist/rereview.test.js.map +1 -0
- package/dist/run-instance.js +544 -254
- package/dist/run-instance.js.map +1 -1
- package/dist/run.js +4 -1
- package/dist/run.js.map +1 -1
- package/dist/schema.js.map +1 -1
- package/dist/seed.js +57 -7
- package/dist/seed.js.map +1 -1
- package/package.json +4 -4
- package/dashboard/dist/assets/index-CcEkcmk_.css +0 -1
package/dist/run-instance.js
CHANGED
|
@@ -26,13 +26,22 @@ import { getWorkflow, resolveReviewGitHubClient, runWorkflow, } from "lastlight-
|
|
|
26
26
|
import { modelTemplateForRow } from "./phase-models.js";
|
|
27
27
|
import { startFakeGitHub } from "./fake-github.js";
|
|
28
28
|
import { appliedRepoConfigKeys, loadRepoConfigFixture, resolveEvalRepoConfig } from "./repo-config.js";
|
|
29
|
-
import { seedWorkspace, seedWorkspaceFromGit, seedWorkspacePrReview, prFilesFromGit, isRealSha, injectRepoContext } from "./seed.js";
|
|
29
|
+
import { seedWorkspace, seedWorkspaceFromGit, seedWorkspacePrReview, prFilesFromGit, mergeBaseOf, isRealSha, injectRepoContext, checkoutRound } from "./seed.js";
|
|
30
30
|
import { collectMetrics, collectMetricsFromFiles, bucketSessionsByPhase, drainSessions, readSessionLog, listSessionFiles, concatJsonl, } from "./metrics.js";
|
|
31
31
|
import { modelCost } from "./env.js";
|
|
32
32
|
import { gradeBehavioral, gradeExecution, gradeTriage, gradeReview, gradeInternalRecall, gradeMarkers } from "./grade.js";
|
|
33
33
|
import { readPipelineStats, persistPipelineArtifacts, internalJudgeInputs, withInternalRecall } from "./review-pipeline-stats.js";
|
|
34
|
-
import {
|
|
34
|
+
import { writeStoredRunContext } from "./phase-replay-context.js";
|
|
35
|
+
import { caseHeadSha, prContextPatch } from "./pr-context.js";
|
|
35
36
|
import { resolveFactsBin } from "./paths.js";
|
|
37
|
+
import { coverageSummary, deltaCounts, dispositionCounts, goldMatchedOf, ledgerCounts, planRounds, rollupRereview, } from "./rereview.js";
|
|
38
|
+
import { carryForward, createRoundStore, fileAt, judgeComments, outdatedResolver, readFreshJson, ROUND_ARTIFACTS, roundScratch, scratchCoverage, scratchLedger, snapshotMtimes, UnitsOracle, } from "./rereview-node.js";
|
|
39
|
+
/**
|
|
40
|
+
* How far the fake's clock moves between two review rounds of a chained case,
|
|
41
|
+
* so round k+1's reviews and threads are strictly later than round k's — the
|
|
42
|
+
* order `lastBotReview` and every "latest review" read depends on.
|
|
43
|
+
*/
|
|
44
|
+
const ROUND_GAP_MS = 60 * 60 * 1000;
|
|
36
45
|
const EVAL_ENV_KEYS = [
|
|
37
46
|
"GITHUB_APP_ID",
|
|
38
47
|
"GITHUB_APP_INSTALLATION_ID",
|
|
@@ -88,6 +97,17 @@ export async function runInstance(inst, opts) {
|
|
|
88
97
|
// workspace would be inventing a code path it does not have.
|
|
89
98
|
const NO_WORKSPACE = new Set(["issue-triage", "dependabot-pr-merge"]);
|
|
90
99
|
const isCodeFix = !isPrReview && !NO_WORKSPACE.has(workflowName);
|
|
100
|
+
// A chained re-review case (issue #429) — `null` for every case that runs
|
|
101
|
+
// once, which then takes exactly the path it always took. Validated before
|
|
102
|
+
// anything starts: a malformed chain is a case error, not a silent one-round run.
|
|
103
|
+
let rounds = null;
|
|
104
|
+
let roundsError;
|
|
105
|
+
try {
|
|
106
|
+
rounds = isPrReview ? planRounds(inst) : null;
|
|
107
|
+
}
|
|
108
|
+
catch (err) {
|
|
109
|
+
roundsError = err.message;
|
|
110
|
+
}
|
|
91
111
|
const stateDir = opts.stateDir ?? mkdtempSync(join(tmpdir(), "ll-eval-"));
|
|
92
112
|
const sessionsDir = join(stateDir, "agent-sessions");
|
|
93
113
|
// The shim appends per-phase jsonl under <sessionsDir>/projects/<slug>/ and
|
|
@@ -115,6 +135,8 @@ export async function runInstance(inst, opts) {
|
|
|
115
135
|
// the first end of the `spec` axis. Content here, linkage in the fake.
|
|
116
136
|
issues: [...(inst.issue ? [inst.issue] : []), ...(inst.pr?.linked_issues ?? [])],
|
|
117
137
|
pulls: inst.pr ? [inst.pr] : [],
|
|
138
|
+
// A chained case releases seeded discussion round by round (`from_round`).
|
|
139
|
+
chained: !!rounds,
|
|
118
140
|
// The CI-read tools (`github_list_workflow_runs` / `..._run_jobs` /
|
|
119
141
|
// `github_get_job_logs`) served from the SAME seed that produces the
|
|
120
142
|
// prompt's `{{ciSection}}`, so digging into the logs corroborates what the
|
|
@@ -123,7 +145,7 @@ export async function runInstance(inst, opts) {
|
|
|
123
145
|
...(inst.pr_state?.ci_jobs?.length
|
|
124
146
|
? {
|
|
125
147
|
actions: {
|
|
126
|
-
headSha: inst
|
|
148
|
+
headSha: caseHeadSha(inst),
|
|
127
149
|
headBranch: inst.pr_state.head_ref,
|
|
128
150
|
jobs: inst.pr_state.ci_jobs.map((j) => ({
|
|
129
151
|
name: j.name,
|
|
@@ -155,6 +177,8 @@ export async function runInstance(inst, opts) {
|
|
|
155
177
|
phases: [],
|
|
156
178
|
};
|
|
157
179
|
try {
|
|
180
|
+
if (roundsError)
|
|
181
|
+
throw new Error(roundsError);
|
|
158
182
|
// 2. Seed the workspace for code-fix (triage needs no repo). A vendored
|
|
159
183
|
// fixture dir wins; otherwise a git-source case (real base SHA + real
|
|
160
184
|
// repo) is checked out from the repo-local cache. Either way the agent
|
|
@@ -205,276 +229,513 @@ export async function runInstance(inst, opts) {
|
|
|
205
229
|
headRef: inst.pr.head_ref,
|
|
206
230
|
baseCommit: inst.pr.base_commit,
|
|
207
231
|
headCommit: inst.pr.head_commit,
|
|
232
|
+
...(rounds ? { roundCommits: rounds.map((r) => r.head_commit) } : {}),
|
|
208
233
|
repoSubdir,
|
|
209
234
|
});
|
|
210
235
|
}
|
|
211
236
|
// The repo's working dir (the nested subdir when seeded) — where grading and
|
|
212
237
|
// the diff run. Falls back to the workspace root if nothing was seeded.
|
|
213
238
|
const repoDir = seed?.workDir ?? join(stateDir, "sandboxes", taskId);
|
|
214
|
-
//
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
//
|
|
235
|
-
//
|
|
236
|
-
//
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
239
|
+
// Every round of a case runs the same workflow definition.
|
|
240
|
+
const def = getWorkflow(workflowName);
|
|
241
|
+
const trialDir = opts.sessionTrialDir;
|
|
242
|
+
/**
|
|
243
|
+
* One dispatch of the workflow: serve the head's changed files, inject the
|
|
244
|
+
* repo context, build the context from the snapshot, resolve the repo layer
|
|
245
|
+
* and run. A single-round case calls it once; a chained case
|
|
246
|
+
* (`inst.rounds`) once per round head, in order (see below).
|
|
247
|
+
*/
|
|
248
|
+
const executeRound = async (round) => {
|
|
249
|
+
// Serve the PR's changed files at GET /pulls/:n/files (pr-review): computed
|
|
250
|
+
// from base..head in the just-seeded workspace, so a review agent that lists
|
|
251
|
+
// files via the API gets the real changed set instead of a 404.
|
|
252
|
+
//
|
|
253
|
+
// KEPT, not discarded: this same set is the SECOND END of every `spec`
|
|
254
|
+
// obligation, and in production `resolveSpecContext` reads it from
|
|
255
|
+
// `listPullRequestFilePaths` at the dispatch choke point. The eval never
|
|
256
|
+
// calls that (it builds the snapshot itself), so without threading it into
|
|
257
|
+
// `prContextPatch` below `changedFiles` stays `null`, `buildSpecObligations`
|
|
258
|
+
// correctly refuses to emit a one-ended seed, and the whole spec family —
|
|
259
|
+
// the one axis nothing else has tried — spends a model call reporting that
|
|
260
|
+
// it cannot work. Deriving it here rather than seeding it per case keeps the
|
|
261
|
+
// two ends from drifting apart and covers every case for free.
|
|
262
|
+
let prFilePaths;
|
|
263
|
+
if (isPrReview && inst.pr && seed) {
|
|
264
|
+
// Against the merge base, as GitHub computes a PR's files: a chained
|
|
265
|
+
// case's earlier heads forked from an older base than the case's (the
|
|
266
|
+
// branch merged or rebased onto main since), and a two-dot diff from
|
|
267
|
+
// the newer base would list main's later changes as the PR's.
|
|
268
|
+
const files = prFilesFromGit(repoDir, mergeBaseOf(repoDir, inst.pr.base_commit, round.head), round.head);
|
|
269
|
+
fake.setPullFiles(inst.pr.number, files);
|
|
270
|
+
prFilePaths = files.map((f) => f.filename);
|
|
271
|
+
}
|
|
272
|
+
else if (inst.pr?.files?.length) {
|
|
273
|
+
// A tier with no checkout (dependency-merge) states its diff in the case
|
|
274
|
+
// instead. Same registration, so `GET /pulls/:n/files` and the patch
|
|
275
|
+
// `github_get_pull_request_diff` returns come from one source.
|
|
276
|
+
fake.setPullFiles(inst.pr.number, inst.pr.files);
|
|
277
|
+
prFilePaths = inst.pr.files.map((f) => f.filename);
|
|
278
|
+
}
|
|
279
|
+
// 2b. Inject synthetic repo-context into the pr-review checkout so the
|
|
280
|
+
// reviewing agent reads it — a GENERIC block from the overlay (applies to
|
|
281
|
+
// every repo) + a PER-REPO block from the tier dataset. The Pi runtime
|
|
282
|
+
// auto-loads AGENTS.md/CLAUDE.md walking up from the agent cwd (= the repo
|
|
283
|
+
// dir), so this reaches the model with no prompt change. Faithful to what a
|
|
284
|
+
// maintainer could commit, so a kept improvement is a portable "add this to
|
|
285
|
+
// your repo" recommendation. Records provenance for inspectability.
|
|
286
|
+
if (isPrReview && seed && (opts.injectContext ?? true)) {
|
|
287
|
+
const sources = resolveInjectedContext({
|
|
288
|
+
overlayDir: opts.overlayDir,
|
|
289
|
+
datasetDir: opts.datasetDir,
|
|
290
|
+
instanceId: inst.instance_id,
|
|
291
|
+
});
|
|
292
|
+
if (sources.length) {
|
|
293
|
+
const combined = sources.map((s) => s.text.trim()).filter(Boolean).join("\n\n");
|
|
294
|
+
if (injectRepoContext(seed.workDir, combined)) {
|
|
295
|
+
result.injectedContext = sources.map((s) => ({
|
|
296
|
+
source: s.source,
|
|
297
|
+
path: s.path,
|
|
298
|
+
bytes: Buffer.byteLength(s.text, "utf8"),
|
|
299
|
+
}));
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
}
|
|
303
|
+
// 3. The run context (the workflow definition is resolved once, above).
|
|
304
|
+
const ctx = {
|
|
305
|
+
owner,
|
|
306
|
+
repo: name,
|
|
307
|
+
issueNumber,
|
|
308
|
+
issueTitle: (isPrReview ? inst.pr?.title : inst.issue?.title) ?? inst.instance_id,
|
|
309
|
+
issueBody: (isPrReview ? inst.pr?.body : inst.issue?.body) ?? inst.problem_statement,
|
|
310
|
+
issueLabels: inst.issue?.labels ?? [],
|
|
311
|
+
commentBody: "",
|
|
312
|
+
sender: "eval",
|
|
313
|
+
branch,
|
|
314
|
+
taskId,
|
|
315
|
+
issueDir: `.lastlight/issue-${issueNumber}`,
|
|
316
|
+
bootstrapLabel: "lastlight:bootstrap",
|
|
317
|
+
// pr-review's Context block keys off `prNumber` — the skill goes straight
|
|
318
|
+
// to github_get_pull_request when it's set (buildPhasePrompt dumps every
|
|
319
|
+
// defined ctx field into the "Context:" block). `baseBranch` is what the
|
|
320
|
+
// deterministic `post-review` phase reads to compute the commentable diff
|
|
321
|
+
// (`git diff origin/<baseBranch>...HEAD`) — WITHOUT it every finding is
|
|
322
|
+
// demoted to the review body and the line-anchored inline-comment path
|
|
323
|
+
// (the point of the tier) never fires. Prod sets it from the PR's base ref;
|
|
324
|
+
// the eval must too, or it diverges from what ships.
|
|
325
|
+
...(isPrReview && inst.pr
|
|
326
|
+
? { prNumber: inst.pr.number, prTitle: inst.pr.title, baseBranch: inst.pr.base_ref }
|
|
327
|
+
: {}),
|
|
328
|
+
// No prePopulateBranch → the runner never clones from GitHub; the agent
|
|
329
|
+
// works in the dir we seeded above (or an empty dir for triage).
|
|
330
|
+
};
|
|
331
|
+
// 3a. The PR state machine's projection (issues #251, #252).
|
|
332
|
+
//
|
|
333
|
+
// A PR-scoped workflow is dispatched in production, never called: the
|
|
334
|
+
// dispatcher resolves one `PrState` snapshot and `renderContext` projects it
|
|
335
|
+
// into the context. That projection IS what the fix and merge prompts reason
|
|
336
|
+
// with — `{{ciSection}}`, `{{attempt}}`, `{{mayMerge}}`, `{{priorNotes}}`,
|
|
337
|
+
// `{{verifyScript}}` — so running them off a hand-built context measures a
|
|
338
|
+
// workflow production does not have. `./pr-context.ts` builds the snapshot a
|
|
339
|
+
// case seeds and hands it to CORE's projection, unmodified.
|
|
340
|
+
//
|
|
341
|
+
// Gated on the workflow's own `pr_scoped: true` metadata rather than a name
|
|
342
|
+
// list here — the same fact core derives `prScopedWorkflows()` from, so an
|
|
343
|
+
// overlay's forked fix workflow is covered without a change to this file.
|
|
344
|
+
//
|
|
345
|
+
// `pr-review` USED TO BE excluded here, and the exclusion was about scores,
|
|
346
|
+
// not about correctness: pr-review is judge-scored and its numbers are
|
|
347
|
+
// compared across runs and against Martian's leaderboard, so enriching its
|
|
348
|
+
// context would move every historical figure as a side effect of a change
|
|
349
|
+
// that was not about them.
|
|
350
|
+
//
|
|
351
|
+
// The exclusion was LIFTED DELIBERATELY on 2026-08-22. It had made the
|
|
352
|
+
// review evidence pipeline unmeasurable on the only tier its gates are read
|
|
353
|
+
// on: with no `prContextPatch`, core's `renderContext` never runs, the
|
|
354
|
+
// context never gets `analysisEnabled`, and every WP3 phase in
|
|
355
|
+
// `pr-review.yaml` matches `skip_if: "analysisEnabled != true"` and skips.
|
|
356
|
+
// WP0's `{{specObligations}}` was unmeasurable there for the same reason.
|
|
357
|
+
// The choice was between a pipeline that cannot be measured and a baseline
|
|
358
|
+
// that has to be re-run; the baseline is being re-run.
|
|
359
|
+
//
|
|
360
|
+
// THEREFORE: every pr-review number produced BEFORE 2026-08-22 was measured
|
|
361
|
+
// on a different template context and must NOT be compared across that
|
|
362
|
+
// boundary — not in `diff-runs.ts`, not against `2026-08-20_074355`, not
|
|
363
|
+
// against the leaderboard entry that run backed. Re-baseline instead.
|
|
364
|
+
const wantsPrContext = def.pr_scoped === true || !!inst.pr_state;
|
|
365
|
+
if (wantsPrContext) {
|
|
366
|
+
Object.assign(ctx, await prContextPatch({
|
|
367
|
+
repo: `${owner}/${name}`,
|
|
368
|
+
prNumber: inst.pr?.number ?? issueNumber,
|
|
369
|
+
title: inst.pr?.title ?? inst.issue?.title ?? inst.instance_id,
|
|
370
|
+
body: inst.pr?.body ?? inst.issue?.body ?? inst.problem_statement,
|
|
371
|
+
branch,
|
|
372
|
+
seed: round.seed,
|
|
373
|
+
baseRef: inst.pr?.base_ref,
|
|
374
|
+
// The real head — the round's, or the PR's (`caseHeadSha`): never the
|
|
375
|
+
// placeholder when the case names a real commit.
|
|
376
|
+
...(isRealSha(round.head) ? { headSha: round.head } : {}),
|
|
377
|
+
// A chained case projects the snapshot itself too (`ctx.prState`), as
|
|
378
|
+
// `dispatchWorkflow` does: post-review reads the ledger it was
|
|
379
|
+
// dispatched with off it. Single-round cases keep the old context.
|
|
380
|
+
snapshot: !!rounds,
|
|
381
|
+
// A REAL `GitHubClient` pointed at the fake — the same construction
|
|
382
|
+
// `post-review` already uses against the mock. Core's own
|
|
383
|
+
// `resolveSpecContext` then reads BOTH ends of the spec axis through
|
|
384
|
+
// it, so the eval exercises the production code path (GraphQL
|
|
385
|
+
// `closingIssuesReferences` + `GET /pulls/:n/files`) rather than a
|
|
386
|
+
// harness copy of it. `setPullFiles` above is what the second read
|
|
387
|
+
// hits, so it must already have run — it has.
|
|
388
|
+
github: resolveReviewGitHubClient({ githubApiBaseUrl: fake.url }),
|
|
389
|
+
// Retained as the fallback for a tier with no live client: a case that
|
|
390
|
+
// seeds `pr_state.changed_files` still wins — including seeding `[]`,
|
|
391
|
+
// which asserts "this PR changes nothing" rather than "we could not
|
|
392
|
+
// read it". Those must stay distinguishable (locked decision 6).
|
|
393
|
+
changedFiles: prFilePaths,
|
|
394
|
+
// The arm's own `review:` policy — the overlay's, never gold's. This
|
|
395
|
+
// is the seam that turns the evidence pipeline on for the `wp3` arm
|
|
396
|
+
// and leaves it off for `baseline`, with no per-case special-casing:
|
|
397
|
+
// `baseline/config.yaml` simply declares no `analysis` block.
|
|
398
|
+
review: opts.arm.review,
|
|
399
|
+
}));
|
|
400
|
+
}
|
|
401
|
+
// The arm supplies model selection in one shot: it patches `ctx.models`/
|
|
402
|
+
// `ctx.variants` (config arms — EXACTLY as production's `simple.js`, so phase
|
|
403
|
+
// `model: "{{models.X}}"` templates resolve) and returns the executor model
|
|
404
|
+
// plus the `runWorkflow` `models`/`variants` args. `models` arms leave the
|
|
405
|
+
// context untouched and return just their forced id.
|
|
406
|
+
const prepared = opts.arm.prepare(ctx);
|
|
407
|
+
// 3b. The target repo's `.lastlight/` config layer (issue #180), resolved
|
|
408
|
+
// through core's OWN dispatch-time resolver against the mock — fetch →
|
|
409
|
+
// sanitize → unpack → merge, unmodified. Undefined for a repo with no
|
|
410
|
+
// `.lastlight/`, in which case `runWorkflow` below is called exactly as
|
|
411
|
+
// it was before the feature existed. Never throws: the resolver's whole
|
|
412
|
+
// contract is "warn, drop the bad bits, run anyway".
|
|
413
|
+
const repoRun = await resolveEvalRepoConfig({
|
|
414
|
+
repo: `${owner}/${name}`,
|
|
415
|
+
workflowName,
|
|
416
|
+
client: fake,
|
|
417
|
+
models: prepared.models,
|
|
418
|
+
variants: prepared.variants,
|
|
419
|
+
defaultModel: prepared.model,
|
|
420
|
+
cacheRoot: join(stateDir, "repo-config"),
|
|
252
421
|
});
|
|
253
|
-
if (
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
})
|
|
422
|
+
if (repoRun.repoConfig) {
|
|
423
|
+
result.repoLayer = {
|
|
424
|
+
repo: repoRun.repoConfig.repo,
|
|
425
|
+
defaultBranch: repoRun.repoConfig.defaultBranch,
|
|
426
|
+
treeSha: repoRun.repoConfig.treeSha,
|
|
427
|
+
assets: [...repoRun.repoConfig.assets],
|
|
428
|
+
applied: appliedRepoConfigKeys(repoRun.repoConfig),
|
|
429
|
+
warnings: repoRun.repoConfig.warnings.map((w) => `${w.code}: ${w.message}`),
|
|
430
|
+
};
|
|
431
|
+
}
|
|
432
|
+
// The repo opting ITSELF out of this workflow in `.lastlight/lastlight.yml`.
|
|
433
|
+
// Production abandons the dispatch here — no run, no agent call — so the
|
|
434
|
+
// case is `blocked` (a deliberate measured outcome), not an error.
|
|
435
|
+
if (repoRun.refusal) {
|
|
436
|
+
result.blocked = true;
|
|
437
|
+
result.repoLayer = { ...(result.repoLayer ?? { repo: `${owner}/${name}` }), refused: repoRun.refusal };
|
|
438
|
+
result.behavioral = gradeBehavioral(inst.expect_github, fake, { issueNumber, branch });
|
|
439
|
+
result.githubMutations = fake.calls.length;
|
|
440
|
+
return { refused: true };
|
|
441
|
+
}
|
|
442
|
+
const config = {
|
|
443
|
+
sandbox: opts.sandbox ?? "none",
|
|
444
|
+
stateDir,
|
|
445
|
+
sessionsDir: round.sessionsDir,
|
|
446
|
+
// Run the agent inside the pre-seeded `<workspace>/<repo>/` checkout (only
|
|
447
|
+
// when we actually seeded one), matching production's nested layout. Core
|
|
448
|
+
// nests `agentCwd` here without a clone; AGENTS.md/.lastlight-skills stay
|
|
449
|
+
// at the workspace root, siblings outside the repo.
|
|
450
|
+
repoSubdir: seed ? repoSubdir : undefined,
|
|
451
|
+
// `config` arms let core pick per phase (this is only the fallback for
|
|
452
|
+
// phases that resolve to nothing — the merged config's `default`); `models`
|
|
453
|
+
// arms force their one id across every step.
|
|
454
|
+
model: prepared.model,
|
|
455
|
+
githubApiBaseUrl: fake.url,
|
|
456
|
+
// Eval workflows shouldn't reach the network beyond the model + fake GH.
|
|
457
|
+
webSearch: false,
|
|
458
|
+
};
|
|
459
|
+
// Phase windows: `onPhaseStart`/`onPhaseEnd` bracket each phase, and the
|
|
460
|
+
// pair is what makes a phase's duration MEASURED rather than inferred from
|
|
461
|
+
// the next phase's start — which would silently bill the gap between phases
|
|
462
|
+
// (workspace refresh, the `until_bash` container spin-up) to whichever phase
|
|
463
|
+
// happened to precede it.
|
|
464
|
+
//
|
|
465
|
+
// `phaseStarts` additionally backs the FALLBACK attribution rule in
|
|
466
|
+
// `bucketSessionsByPhase`. Sessions now carry their owning phase as a stamp,
|
|
467
|
+
// so the windows are only consulted for jsonl archived before that stamp
|
|
468
|
+
// existed; see that function for why a start-time lookup cannot attribute a
|
|
469
|
+
// fan-out at all.
|
|
470
|
+
const phaseStarts = [];
|
|
471
|
+
const phaseWindows = new Map();
|
|
472
|
+
const callbacks = {
|
|
473
|
+
onPhaseStart: async (phase) => {
|
|
474
|
+
const now = Date.now();
|
|
475
|
+
phaseStarts.push({ phase, start: now });
|
|
476
|
+
// First start wins: a label the engine re-announces (a loop node whose
|
|
477
|
+
// condition-met entry repeats it) must not restart its own clock.
|
|
478
|
+
if (!phaseWindows.has(phase))
|
|
479
|
+
phaseWindows.set(phase, { start: now });
|
|
480
|
+
},
|
|
481
|
+
onPhaseEnd: async (phase) => {
|
|
482
|
+
const w = phaseWindows.get(phase);
|
|
483
|
+
if (w)
|
|
484
|
+
w.end = Date.now();
|
|
485
|
+
},
|
|
486
|
+
};
|
|
487
|
+
const fullFile = trialDir ? join(trialDir, "full.jsonl") : undefined;
|
|
488
|
+
// Flush the consolidated transcript atomically (so a polling dashboard never
|
|
489
|
+
// reads a half-written file): on a timer while running (follow-along), and
|
|
490
|
+
// once at the end. Best-effort — a flush failure must never affect the run.
|
|
491
|
+
const flushFull = () => {
|
|
492
|
+
if (!fullFile || !trialDir)
|
|
493
|
+
return;
|
|
494
|
+
try {
|
|
495
|
+
const log = readSessionLog(round.sessionsDir);
|
|
496
|
+
if (!log)
|
|
497
|
+
return;
|
|
498
|
+
mkdirSync(trialDir, { recursive: true });
|
|
499
|
+
const tmp = `${fullFile}.tmp`;
|
|
500
|
+
writeFileSync(tmp, log);
|
|
501
|
+
renameSync(tmp, fullFile);
|
|
261
502
|
}
|
|
503
|
+
catch {
|
|
504
|
+
/* best-effort */
|
|
505
|
+
}
|
|
506
|
+
};
|
|
507
|
+
// 4. Run. Empty approvalConfig (7th arg) → every approval gate is disabled.
|
|
508
|
+
// The arm's prepared maps go to args 6 (models) and 9 (variants), matching
|
|
509
|
+
// prod's runWorkflow call; `models` arms leave both undefined so every phase
|
|
510
|
+
// falls back to config.model (one model everywhere). The 10th arg is the
|
|
511
|
+
// repo layer — `undefined` for a repo with no `.lastlight/`, which is the
|
|
512
|
+
// pre-#180 call byte-for-byte. The 5th/8th (store, workflow id) are unset
|
|
513
|
+
// for a single-round case — no db, so every gate is inert — and, for a
|
|
514
|
+
// chained case, the in-memory run store post-review persists the review
|
|
515
|
+
// ledger to (`rereview-node.ts`); `approvalConfig` stays empty either way.
|
|
516
|
+
const flushTimer = fullFile ? setInterval(flushFull, 1000) : undefined;
|
|
517
|
+
let wf;
|
|
518
|
+
try {
|
|
519
|
+
wf = await runWorkflow(def, ctx, config, callbacks, round.store, prepared.models, {}, round.workflowId, prepared.variants, repoRun.repoConfig);
|
|
262
520
|
}
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
repo: name,
|
|
269
|
-
issueNumber,
|
|
270
|
-
issueTitle: (isPrReview ? inst.pr?.title : inst.issue?.title) ?? inst.instance_id,
|
|
271
|
-
issueBody: (isPrReview ? inst.pr?.body : inst.issue?.body) ?? inst.problem_statement,
|
|
272
|
-
issueLabels: inst.issue?.labels ?? [],
|
|
273
|
-
commentBody: "",
|
|
274
|
-
sender: "eval",
|
|
275
|
-
branch,
|
|
276
|
-
taskId,
|
|
277
|
-
issueDir: `.lastlight/issue-${issueNumber}`,
|
|
278
|
-
bootstrapLabel: "lastlight:bootstrap",
|
|
279
|
-
// pr-review's Context block keys off `prNumber` — the skill goes straight
|
|
280
|
-
// to github_get_pull_request when it's set (buildPhasePrompt dumps every
|
|
281
|
-
// defined ctx field into the "Context:" block). `baseBranch` is what the
|
|
282
|
-
// deterministic `post-review` phase reads to compute the commentable diff
|
|
283
|
-
// (`git diff origin/<baseBranch>...HEAD`) — WITHOUT it every finding is
|
|
284
|
-
// demoted to the review body and the line-anchored inline-comment path
|
|
285
|
-
// (the point of the tier) never fires. Prod sets it from the PR's base ref;
|
|
286
|
-
// the eval must too, or it diverges from what ships.
|
|
287
|
-
...(isPrReview && inst.pr
|
|
288
|
-
? { prNumber: inst.pr.number, prTitle: inst.pr.title, baseBranch: inst.pr.base_ref }
|
|
289
|
-
: {}),
|
|
290
|
-
// No prePopulateBranch → the runner never clones from GitHub; the agent
|
|
291
|
-
// works in the dir we seeded above (or an empty dir for triage).
|
|
521
|
+
finally {
|
|
522
|
+
if (flushTimer)
|
|
523
|
+
clearInterval(flushTimer);
|
|
524
|
+
}
|
|
525
|
+
return { wf, ctx, prepared, phaseStarts, phaseWindows, flushFull, fullFile };
|
|
292
526
|
};
|
|
293
|
-
//
|
|
294
|
-
//
|
|
295
|
-
// A PR-scoped workflow is dispatched in production, never called: the
|
|
296
|
-
// dispatcher resolves one `PrState` snapshot and `renderContext` projects it
|
|
297
|
-
// into the context. That projection IS what the fix and merge prompts reason
|
|
298
|
-
// with — `{{ciSection}}`, `{{attempt}}`, `{{mayMerge}}`, `{{priorNotes}}`,
|
|
299
|
-
// `{{verifyScript}}` — so running them off a hand-built context measures a
|
|
300
|
-
// workflow production does not have. `./pr-context.ts` builds the snapshot a
|
|
301
|
-
// case seeds and hands it to CORE's projection, unmodified.
|
|
302
|
-
//
|
|
303
|
-
// Gated on the workflow's own `pr_scoped: true` metadata rather than a name
|
|
304
|
-
// list here — the same fact core derives `prScopedWorkflows()` from, so an
|
|
305
|
-
// overlay's forked fix workflow is covered without a change to this file.
|
|
527
|
+
// ── Chained re-review rounds (issue #429) ──────────────────────────────
|
|
306
528
|
//
|
|
307
|
-
//
|
|
308
|
-
//
|
|
309
|
-
//
|
|
310
|
-
//
|
|
311
|
-
//
|
|
312
|
-
//
|
|
313
|
-
//
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
const wantsPrContext = def.pr_scoped === true || !!inst.pr_state;
|
|
327
|
-
if (wantsPrContext) {
|
|
328
|
-
Object.assign(ctx, await prContextPatch({
|
|
329
|
-
repo: `${owner}/${name}`,
|
|
330
|
-
prNumber: inst.pr?.number ?? issueNumber,
|
|
331
|
-
title: inst.pr?.title ?? inst.issue?.title ?? inst.instance_id,
|
|
332
|
-
body: inst.pr?.body ?? inst.issue?.body ?? inst.problem_statement,
|
|
333
|
-
branch,
|
|
334
|
-
seed: inst.pr_state,
|
|
335
|
-
baseRef: inst.pr?.base_ref,
|
|
336
|
-
// A REAL `GitHubClient` pointed at the fake — the same construction
|
|
337
|
-
// `post-review` already uses against the mock. Core's own
|
|
338
|
-
// `resolveSpecContext` then reads BOTH ends of the spec axis through
|
|
339
|
-
// it, so the eval exercises the production code path (GraphQL
|
|
340
|
-
// `closingIssuesReferences` + `GET /pulls/:n/files`) rather than a
|
|
341
|
-
// harness copy of it. `setPullFiles` above is what the second read
|
|
342
|
-
// hits, so it must already have run — it has.
|
|
343
|
-
github: resolveReviewGitHubClient({ githubApiBaseUrl: fake.url }),
|
|
344
|
-
// Retained as the fallback for a tier with no live client: a case that
|
|
345
|
-
// seeds `pr_state.changed_files` still wins — including seeding `[]`,
|
|
346
|
-
// which asserts "this PR changes nothing" rather than "we could not
|
|
347
|
-
// read it". Those must stay distinguishable (locked decision 6).
|
|
348
|
-
changedFiles: prFilePaths,
|
|
349
|
-
// The arm's own `review:` policy — the overlay's, never gold's. This
|
|
350
|
-
// is the seam that turns the evidence pipeline on for the `wp3` arm
|
|
351
|
-
// and leaves it off for `baseline`, with no per-case special-casing:
|
|
352
|
-
// `baseline/config.yaml` simply declares no `analysis` block.
|
|
353
|
-
review: opts.arm.review,
|
|
354
|
-
}));
|
|
529
|
+
// A case with `rounds` runs the real workflow once per earlier head, in
|
|
530
|
+
// order, before the scored last round below — one fake GitHub (round k's
|
|
531
|
+
// review and threads are what round k+1 reads), one per-PR workspace
|
|
532
|
+
// (checked out to each head, its `.lastlight/pr-review/` carried as a
|
|
533
|
+
// reused production workspace carries it), and one in-memory run store, so
|
|
534
|
+
// post-review persists the review ledger to the round's run scratch and the
|
|
535
|
+
// next round is dispatched with it through core's `deriveReviewLedger`.
|
|
536
|
+
const store = rounds ? createRoundStore() : undefined;
|
|
537
|
+
const roundRecords = [];
|
|
538
|
+
let roundSeed = inst.pr_state;
|
|
539
|
+
let currentHead = rounds ? rounds[0].head_commit : caseHeadSha(inst);
|
|
540
|
+
const prDir = join(repoDir, ".lastlight", "pr-review");
|
|
541
|
+
const heads = rounds?.map((r) => r.head_commit) ?? [];
|
|
542
|
+
const factsBin = rounds ? resolveFactsBin() : null;
|
|
543
|
+
const oracle = rounds && factsBin && inst.pr
|
|
544
|
+
? new UnitsOracle({ repoDir, base: inst.pr.base_commit, factsBin, root: join(stateDir, "rereview-oracle") })
|
|
545
|
+
: undefined;
|
|
546
|
+
if (rounds && inst.pr && seed) {
|
|
547
|
+
fake.setOutdatedResolver(outdatedResolver(repoDir, () => currentHead));
|
|
355
548
|
}
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
defaultBranch: repoRun.repoConfig.defaultBranch,
|
|
381
|
-
treeSha: repoRun.repoConfig.treeSha,
|
|
382
|
-
assets: [...repoRun.repoConfig.assets],
|
|
383
|
-
applied: appliedRepoConfigKeys(repoRun.repoConfig),
|
|
384
|
-
warnings: repoRun.repoConfig.warnings.map((w) => `${w.code}: ${w.message}`),
|
|
549
|
+
/**
|
|
550
|
+
* What one finished round measured — everything but the grade and the
|
|
551
|
+
* spend, which the caller adds (the last round's come from the case's own).
|
|
552
|
+
*/
|
|
553
|
+
const measureRound = async (k, before, wfOk) => {
|
|
554
|
+
const head = heads[k];
|
|
555
|
+
const reviews = fake.submittedReviews(inst.pr.number);
|
|
556
|
+
const inline = reviews.flatMap((r) => r.comments);
|
|
557
|
+
const disposition = readFreshJson(prDir, "disposition.json", before);
|
|
558
|
+
const scratch = store ? await roundScratch(store, roundWorkflowId(k)) : {};
|
|
559
|
+
const record = {
|
|
560
|
+
round: k + 1,
|
|
561
|
+
headSha: head,
|
|
562
|
+
...(rounds[k].label ? { label: rounds[k].label } : {}),
|
|
563
|
+
workflowSucceeded: wfOk.success,
|
|
564
|
+
...(wfOk.error ? { error: wfOk.error } : {}),
|
|
565
|
+
...(reviews.length ? { event: reviews.at(-1).event } : {}),
|
|
566
|
+
inlinePosted: inline.length,
|
|
567
|
+
...(disposition?.findings ? dispositionCounts(disposition.findings) : {}),
|
|
568
|
+
costUsd: 0,
|
|
569
|
+
inputTokens: 0,
|
|
570
|
+
cachedTokens: 0,
|
|
571
|
+
outputTokens: 0,
|
|
572
|
+
durationMs: 0,
|
|
385
573
|
};
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
},
|
|
436
|
-
onPhaseEnd: async (phase) => {
|
|
437
|
-
const w = phaseWindows.get(phase);
|
|
438
|
-
if (w)
|
|
439
|
-
w.end = Date.now();
|
|
440
|
-
},
|
|
574
|
+
const coverage = coverageSummary(scratchCoverage(scratch) ?? readFreshJson(prDir, "review-coverage.json", before));
|
|
575
|
+
if (coverage)
|
|
576
|
+
record.coverage = coverage;
|
|
577
|
+
const ledger = ledgerCounts(scratchLedger(scratch));
|
|
578
|
+
if (ledger)
|
|
579
|
+
record.ledger = ledger;
|
|
580
|
+
// Late discoveries: round ≥ 2 only. The round's OWN units and prior
|
|
581
|
+
// review when its pipeline cut them (what its convergence gate saw);
|
|
582
|
+
// otherwise the harness's oracle cut over the same two heads, so an arm
|
|
583
|
+
// that runs no pipeline is measured on the same instrument.
|
|
584
|
+
if (k >= 1) {
|
|
585
|
+
const ownUnits = readFreshJson(prDir, "units.json", before);
|
|
586
|
+
const ownPrior = readFreshJson(prDir, "prior-review.json", before);
|
|
587
|
+
let units;
|
|
588
|
+
let prior;
|
|
589
|
+
if (ownUnits?.units?.length && ownPrior) {
|
|
590
|
+
units = ownUnits.units;
|
|
591
|
+
prior = ownPrior;
|
|
592
|
+
record.lateDiscoverySource = "units.json";
|
|
593
|
+
}
|
|
594
|
+
else if (oracle) {
|
|
595
|
+
try {
|
|
596
|
+
units = oracle.at(heads, k).units;
|
|
597
|
+
prior = oracle.priorFor(heads, k);
|
|
598
|
+
record.lateDiscoverySource = "oracle";
|
|
599
|
+
}
|
|
600
|
+
catch (err) {
|
|
601
|
+
record.lateDiscoveryUnavailable = `oracle units cut failed: ${err.message.slice(0, 200)}`;
|
|
602
|
+
}
|
|
603
|
+
}
|
|
604
|
+
else {
|
|
605
|
+
record.lateDiscoveryUnavailable = "no units.json/prior-review.json this round and no lastlight-facts binary for the oracle";
|
|
606
|
+
}
|
|
607
|
+
if (units && prior !== undefined) {
|
|
608
|
+
const verdicts = judgeComments({
|
|
609
|
+
comments: inline.map((c) => ({ path: c.path, ...(c.line !== undefined ? { line: c.line } : {}), ...(c.start_line !== undefined ? { start_line: c.start_line } : {}) })),
|
|
610
|
+
prior,
|
|
611
|
+
units,
|
|
612
|
+
fileText: (path) => fileAt(repoDir, head, path),
|
|
613
|
+
});
|
|
614
|
+
record.lateDiscovery = verdicts.filter((v) => v.verdict === "unchanged").length;
|
|
615
|
+
record.lateDiscoveryOf = verdicts.filter((v) => v.verdict !== null).length;
|
|
616
|
+
const delta = deltaCounts(units);
|
|
617
|
+
if (delta)
|
|
618
|
+
record.delta = delta;
|
|
619
|
+
roundVerdicts.set(k, verdicts);
|
|
620
|
+
}
|
|
621
|
+
}
|
|
622
|
+
return { record, reviews, scratch, disposition };
|
|
441
623
|
};
|
|
442
|
-
const
|
|
443
|
-
const
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
return;
|
|
624
|
+
const roundWorkflowId = (k) => `${taskId}:round-${k + 1}`;
|
|
625
|
+
const roundVerdicts = new Map();
|
|
626
|
+
/** A round's evidence, beside its artifacts: what it posted, its dispositions, the ledger it left, the late-discovery verdicts. */
|
|
627
|
+
const archiveRound = (k, m, extra = {}) => {
|
|
628
|
+
if (!trialDir)
|
|
629
|
+
return undefined;
|
|
630
|
+
const dir = join(trialDir, `round-${k + 1}`);
|
|
450
631
|
try {
|
|
451
|
-
|
|
452
|
-
if (
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
632
|
+
mkdirSync(dir, { recursive: true });
|
|
633
|
+
if (extra.sessionsDir) {
|
|
634
|
+
const log = readSessionLog(extra.sessionsDir);
|
|
635
|
+
if (log)
|
|
636
|
+
writeFileSync(join(dir, "full.jsonl"), log);
|
|
637
|
+
persistPipelineArtifacts(repoDir, dir);
|
|
638
|
+
}
|
|
639
|
+
writeFileSync(join(dir, "round.json"), `${JSON.stringify({
|
|
640
|
+
round: k + 1,
|
|
641
|
+
head: heads[k],
|
|
642
|
+
reviews: m.reviews,
|
|
643
|
+
disposition: m.disposition ?? null,
|
|
644
|
+
reviewLedger: scratchLedger(m.scratch) ?? null,
|
|
645
|
+
reviewCoverage: scratchCoverage(m.scratch) ?? null,
|
|
646
|
+
lateDiscovery: roundVerdicts.get(k) ?? null,
|
|
647
|
+
dispatched: roundSeed ?? null,
|
|
648
|
+
}, null, 2)}\n`);
|
|
649
|
+
return `${opts.sessionTrialRel ?? trialDir}/round-${k + 1}`;
|
|
458
650
|
}
|
|
459
651
|
catch {
|
|
460
|
-
|
|
652
|
+
return undefined;
|
|
461
653
|
}
|
|
462
654
|
};
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
655
|
+
if (rounds && inst.pr && seed) {
|
|
656
|
+
for (let k = 0; k < rounds.length - 1; k++) {
|
|
657
|
+
const head = heads[k];
|
|
658
|
+
if (k > 0) {
|
|
659
|
+
checkoutRound(repoDir, seed.branch, head);
|
|
660
|
+
fake.advanceClock(ROUND_GAP_MS);
|
|
661
|
+
}
|
|
662
|
+
currentHead = head;
|
|
663
|
+
fake.setHead(inst.pr.number, head);
|
|
664
|
+
fake.startRound(k + 1);
|
|
665
|
+
const sessionsDirK = join(stateDir, `agent-sessions-round-${k + 1}`);
|
|
666
|
+
mkdirSync(join(sessionsDirK, "projects"), { recursive: true });
|
|
667
|
+
const before = snapshotMtimes(prDir, ROUND_ARTIFACTS);
|
|
668
|
+
const startedAt = Date.now();
|
|
669
|
+
const seedK = { ...(roundSeed ?? {}), head_sha: head };
|
|
670
|
+
roundSeed = seedK;
|
|
671
|
+
const ran = await executeRound({ head, seed: seedK, sessionsDir: sessionsDirK, store, workflowId: roundWorkflowId(k) });
|
|
672
|
+
if ("refused" in ran)
|
|
673
|
+
return result;
|
|
674
|
+
await drainSessions(sessionsDirK);
|
|
675
|
+
const failed = ran.wf.phases.find((p) => !p.success && p.error);
|
|
676
|
+
const m = await measureRound(k, before, {
|
|
677
|
+
success: ran.wf.success,
|
|
678
|
+
...(failed?.error ? { error: `${failed.phase}: ${failed.error}`.slice(0, 300) } : {}),
|
|
679
|
+
});
|
|
680
|
+
const cost = collectMetrics(sessionsDirK, modelCost(ran.prepared.model));
|
|
681
|
+
Object.assign(m.record, {
|
|
682
|
+
costUsd: cost.costUsd,
|
|
683
|
+
inputTokens: cost.inputTokens,
|
|
684
|
+
cachedTokens: cost.cachedTokens,
|
|
685
|
+
outputTokens: cost.outputTokens,
|
|
686
|
+
durationMs: Date.now() - startedAt,
|
|
687
|
+
});
|
|
688
|
+
// The round's own grade, for cumulative recall — the same judge and
|
|
689
|
+
// gold as the scored round, over the review THIS round posted.
|
|
690
|
+
if (inst.review_gold) {
|
|
691
|
+
const rg = await gradeReview({ gold: inst.review_gold, reviews: m.reviews, beta: opts.judge?.beta, neutralGold: inst.review_gold_neutral });
|
|
692
|
+
const matched = goldMatchedOf(rg.trace);
|
|
693
|
+
if (!rg.error && matched) {
|
|
694
|
+
m.record.goldMatched = matched;
|
|
695
|
+
m.record.postedFindings = rg.posted;
|
|
696
|
+
}
|
|
697
|
+
}
|
|
698
|
+
const rel = archiveRound(k, m, { sessionsDir: sessionsDirK });
|
|
699
|
+
if (rel)
|
|
700
|
+
m.record.artifactRel = rel;
|
|
701
|
+
roundRecords.push(m.record);
|
|
702
|
+
// Recorded as it goes, so a later round that throws still leaves the
|
|
703
|
+
// rounds that ran on the result (re-rolled with the last round below).
|
|
704
|
+
result.rereview = rollupRereview(roundRecords, inst.review_gold?.length);
|
|
705
|
+
// What the next round is dispatched with.
|
|
706
|
+
roundSeed = {
|
|
707
|
+
...(inst.pr_state ?? {}),
|
|
708
|
+
...carryForward({
|
|
709
|
+
repoDir,
|
|
710
|
+
baseCommit: inst.pr.base_commit,
|
|
711
|
+
prevHead: head,
|
|
712
|
+
nextHead: heads[k + 1],
|
|
713
|
+
reviews: m.reviews,
|
|
714
|
+
prev: seedK,
|
|
715
|
+
scratch: m.scratch,
|
|
716
|
+
}),
|
|
717
|
+
};
|
|
718
|
+
}
|
|
719
|
+
// The scored round: the PR's own head.
|
|
720
|
+
checkoutRound(repoDir, seed.branch, inst.pr.head_commit);
|
|
721
|
+
fake.advanceClock(ROUND_GAP_MS);
|
|
722
|
+
currentHead = inst.pr.head_commit;
|
|
723
|
+
fake.setHead(inst.pr.number, inst.pr.head_commit);
|
|
724
|
+
fake.startRound(rounds.length);
|
|
477
725
|
}
|
|
726
|
+
const finalBefore = rounds ? snapshotMtimes(prDir, ROUND_ARTIFACTS) : undefined;
|
|
727
|
+
const finalStartedAt = Date.now();
|
|
728
|
+
const ran = await executeRound({
|
|
729
|
+
// Single-round: the PR's own head (its files are what GET /pulls/:n/files
|
|
730
|
+
// always served), else the snapshot's head for a PR-less case.
|
|
731
|
+
head: inst.pr?.head_commit ?? caseHeadSha(inst),
|
|
732
|
+
seed: roundSeed,
|
|
733
|
+
sessionsDir,
|
|
734
|
+
...(store ? { store, workflowId: roundWorkflowId(rounds.length - 1) } : {}),
|
|
735
|
+
});
|
|
736
|
+
if ("refused" in ran)
|
|
737
|
+
return result;
|
|
738
|
+
const { wf, ctx, prepared, phaseStarts, phaseWindows, flushFull, fullFile } = ran;
|
|
478
739
|
result.workflowSucceeded = wf.success;
|
|
479
740
|
// Record the model each phase resolved to — the arm forced id in `models`
|
|
480
741
|
// mode, or the per-step model the merged config assigned in `config` mode
|
|
@@ -483,7 +744,7 @@ export async function runInstance(inst, opts) {
|
|
|
483
744
|
// template lookup goes through modelTemplateForRow (phase-models.ts), which
|
|
484
745
|
// parses branch rows back to their declaration via core's PhaseRef.
|
|
485
746
|
result.phases = wf.phases.map((p) => {
|
|
486
|
-
const { template, fallbackPhase } = modelTemplateForRow(def.phases, p.phase);
|
|
747
|
+
const { template, fallbackPhase } = modelTemplateForRow(def.phases, p.phase, p.modelTemplate);
|
|
487
748
|
return {
|
|
488
749
|
phase: p.phase,
|
|
489
750
|
success: p.success,
|
|
@@ -584,6 +845,12 @@ export async function runInstance(inst, opts) {
|
|
|
584
845
|
try {
|
|
585
846
|
if (persistPipelineArtifacts(repoDir, opts.sessionTrialDir) && opts.sessionTrialRel)
|
|
586
847
|
result.pipelineArtifactRel = `${opts.sessionTrialRel}/pr-review`;
|
|
848
|
+
// The prompt context a phase REPLAY needs and cannot rebuild from
|
|
849
|
+
// the artifacts: what `select` was told about the PR's earlier
|
|
850
|
+
// conversation and review ledger (`phase-replay-node.ts`'s
|
|
851
|
+
// `readStoredRunContext`). Beside `pr-review/`, never inside it — a
|
|
852
|
+
// replay copies that dir into the checkout the agent works in.
|
|
853
|
+
writeStoredRunContext(opts.sessionTrialDir, ctx);
|
|
587
854
|
}
|
|
588
855
|
catch (err) {
|
|
589
856
|
// Never fail a measured run over its own bookkeeping — but a warning
|
|
@@ -740,6 +1007,29 @@ export async function runInstance(inst, opts) {
|
|
|
740
1007
|
/* leave sessionTrial unset */
|
|
741
1008
|
}
|
|
742
1009
|
}
|
|
1010
|
+
// 5e. The chained case's last round, measured like the others, and the
|
|
1011
|
+
// roll-up. Its grade, spend and artifacts are the case's own (above).
|
|
1012
|
+
if (rounds && finalBefore) {
|
|
1013
|
+
const k = rounds.length - 1;
|
|
1014
|
+
const m = await measureRound(k, finalBefore, { success: wf.success, ...(result.error ? { error: result.error } : {}) });
|
|
1015
|
+
Object.assign(m.record, {
|
|
1016
|
+
costUsd: result.costUsd,
|
|
1017
|
+
inputTokens: result.inputTokens,
|
|
1018
|
+
cachedTokens: result.cachedTokens,
|
|
1019
|
+
outputTokens: result.outputTokens,
|
|
1020
|
+
durationMs: Date.now() - finalStartedAt,
|
|
1021
|
+
});
|
|
1022
|
+
const matched = result.review && !result.error?.startsWith("review judge") ? goldMatchedOf(result.review.trace) : null;
|
|
1023
|
+
if (matched) {
|
|
1024
|
+
m.record.goldMatched = matched;
|
|
1025
|
+
m.record.postedFindings = result.review.posted;
|
|
1026
|
+
}
|
|
1027
|
+
const rel = archiveRound(k, m);
|
|
1028
|
+
if (rel)
|
|
1029
|
+
m.record.artifactRel = rel;
|
|
1030
|
+
roundRecords.push(m.record);
|
|
1031
|
+
result.rereview = rollupRereview(roundRecords, inst.review_gold?.length);
|
|
1032
|
+
}
|
|
743
1033
|
}
|
|
744
1034
|
catch (err) {
|
|
745
1035
|
result.error = err instanceof Error ? err.message : String(err);
|