lastlight-evals 0.1.2 → 0.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/README.md +98 -12
  2. package/dashboard/dist/assets/index-YwScAAxm.js +238 -0
  3. package/dashboard/dist/assets/index-uU9Sj3M4.css +1 -0
  4. package/dashboard/dist/index.html +19 -0
  5. package/dashboard/dist/logo.png +0 -0
  6. package/datasets/code-fix/repos/codefix__date-range-off-by-one/.github/workflows/ci.yml +16 -0
  7. package/datasets/code-fix/repos/codefix__date-range-off-by-one/README.md +7 -1
  8. package/datasets/code-fix/repos/codefix__date-range-off-by-one/package.json +3 -1
  9. package/datasets/code-fix/repos/codefix__date-range-off-by-one/scripts/lint.mjs +25 -0
  10. package/datasets/code-fix/repos/codefix__date-range-off-by-one/test/date-range.test.ts +14 -0
  11. package/datasets/code-fix/repos/codefix__date-range-off-by-one/tsconfig.json +13 -0
  12. package/datasets/code-fix/tests/codefix__date-range-off-by-one/date-range.test.ts +5 -5
  13. package/dist/arm.js +132 -0
  14. package/dist/arm.js.map +1 -0
  15. package/dist/config.js +100 -0
  16. package/dist/config.js.map +1 -0
  17. package/dist/mechanism.test.js +156 -0
  18. package/dist/mechanism.test.js.map +1 -1
  19. package/dist/metrics.js +83 -0
  20. package/dist/metrics.js.map +1 -1
  21. package/dist/paths.js +51 -1
  22. package/dist/paths.js.map +1 -1
  23. package/dist/report.js +112 -3
  24. package/dist/report.js.map +1 -1
  25. package/dist/run-instance.js +136 -15
  26. package/dist/run-instance.js.map +1 -1
  27. package/dist/run.js +350 -118
  28. package/dist/run.js.map +1 -1
  29. package/dist/serve.js +136 -0
  30. package/dist/serve.js.map +1 -0
  31. package/examples/overlay/README.md +30 -0
  32. package/examples/overlay/config.yaml +36 -0
  33. package/examples/overlay-anthropic/config.yaml +27 -0
  34. package/package.json +16 -6
  35. package/dist/html-report.js +0 -325
  36. package/dist/html-report.js.map +0 -1
package/dist/run.js CHANGED
@@ -17,18 +17,19 @@
17
17
  * The deterministic, AI-free plumbing is covered separately by
18
18
  * `evals/mechanism.test.ts` in the normal `npm test` suite.
19
19
  */
20
- import { existsSync, readFileSync } from "node:fs";
20
+ import { existsSync } from "node:fs";
21
21
  import { join } from "node:path";
22
22
  import { spawn } from "node:child_process";
23
23
  import * as p from "@clack/prompts";
24
24
  import chalk from "chalk";
25
25
  import { loadDotEnv, hasProviderKey, evalModels, compareModels, modelLabels, resolveModel, setModelsPath } from "./env.js";
26
- import { runInstance, applyEvalEnv } from "./run-instance.js";
27
- import { summarize, writeArtifacts, aggregateTrials } from "./report.js";
28
- import { writeHtml } from "./html-report.js";
26
+ import { runInstance, applyEvalEnv, slug } from "./run-instance.js";
27
+ import { modelsArm, configArm, releaseOverlayGuard } from "./arm.js";
28
+ import { summarize, writeArtifacts, writeScorecard, aggregateTrials, } from "./report.js";
29
29
  import { bootstrapAssets } from "./bootstrap.js";
30
30
  import { discoverTiers, loadInstances, workflowFor } from "./discovery.js";
31
- import { builtinDatasetsRoot, resultsRoot } from "./paths.js";
31
+ import { builtinDatasetsRoot, tierResultsDir, makeRunId, gitShortSha, resultsRoot, dashboardDistRoot } from "./paths.js";
32
+ import { startServer } from "./serve.js";
32
33
  import { runInit } from "./init.js";
33
34
  /**
34
35
  * A clack spinner in a TTY; in non-TTY (CI / piped / agent) a quiet stub that
@@ -48,9 +49,8 @@ function makeSpinner() {
48
49
  },
49
50
  };
50
51
  }
51
- /** Open a file in the OS default browser (best-effort, never throws). */
52
- function openInBrowser(file) {
53
- const url = `file://${file}`;
52
+ /** Open a URL in the OS default browser (best-effort, never throws). */
53
+ function openInBrowser(url) {
54
54
  const [cmd, args] = process.platform === "darwin"
55
55
  ? ["open", [url]]
56
56
  : process.platform === "win32"
@@ -99,6 +99,12 @@ function fmtMs(ms) {
99
99
  function familyLabel(envKey) {
100
100
  return envKey.replace(/_API_KEY$/i, "").toLowerCase() || "default";
101
101
  }
102
+ /** Attach run metadata to a scorecard (mutate-and-return; `summarize` hands back
103
+ * a fresh object each call, so this is safe). */
104
+ function withMeta(card, meta) {
105
+ card.meta = meta;
106
+ return card;
107
+ }
102
108
  /**
103
109
  * Silence `console.*` for the whole batch (parallel mode). The per-run
104
110
  * `quiet()` swap saves/restores console and would corrupt under concurrent
@@ -116,6 +122,8 @@ function verdictLine(tierName, inst, r) {
116
122
  const head = `${chalk.cyan(tierName)}/${inst.instance_id}`;
117
123
  if (r.error)
118
124
  return `${head} ${chalk.red("harness error")}`;
125
+ if (r.blocked)
126
+ return `${head} ${chalk.yellow("blocked")} ${chalk.dim("(workflow gate)")}`;
119
127
  const count = (pass) => (pass !== undefined && r.trials ? chalk.dim(` ${pass}/${r.trials}`) : "");
120
128
  const parts = [];
121
129
  if (r.resolved !== undefined)
@@ -154,22 +162,93 @@ function strFlag(name) {
154
162
  }
155
163
  return process.env[`EVAL_${name.toUpperCase()}`];
156
164
  }
165
+ /** Collect ALL occurrences of a repeatable string flag (`--name a --name b` or
166
+ * `--name=a`). Used by `--overlay`, which may repeat in `config` mode to compare
167
+ * one arm per overlay. */
168
+ function strFlagAll(name) {
169
+ const out = [];
170
+ const argv = process.argv;
171
+ for (let i = 0; i < argv.length; i++) {
172
+ if (argv[i] === `--${name}` && argv[i + 1] !== undefined) {
173
+ out.push(argv[i + 1]);
174
+ i++;
175
+ continue;
176
+ }
177
+ const m = argv[i].match(new RegExp(`^--${name}=(.+)$`));
178
+ if (m)
179
+ out.push(m[1]);
180
+ }
181
+ return out;
182
+ }
157
183
  /** CLI flags that take a following value (so it isn't read as a tier name). */
158
- const VALUE_FLAGS = new Set(["--runs", "--model", "--models", "--overlay", "--datasets", "--models-file"]);
184
+ const VALUE_FLAGS = new Set(["--runs", "--model", "--models", "--mode", "--overlay", "--datasets", "--models-file"]);
159
185
  async function runEval() {
160
186
  loadDotEnv();
161
187
  p.intro(chalk.bold(`Last Light ${chalk.yellow("·")} eval`));
188
+ // Run type — the comparison axis:
189
+ // `models` — compare N models, each FORCED across every workflow step.
190
+ // `config` — run an overlay's REAL per-step model config (what ships); the
191
+ // arm is the config/overlay, compared across runs (or N overlays
192
+ // side-by-side). `--mode` wins; an explicit `--model`/`--compare`
193
+ // implies `models`; otherwise ask in a TTY (default `models`).
194
+ const modeFlag = strFlag("mode");
195
+ const modelArg = strFlag("model") ?? strFlag("models");
196
+ const compare = process.argv.includes("--compare");
197
+ let runType;
198
+ if (modeFlag === "config")
199
+ runType = "config";
200
+ else if (modeFlag === "models" || modelArg || compare)
201
+ runType = "models";
202
+ else if (process.stdin.isTTY) {
203
+ const picked = await p.select({
204
+ message: "What do you want to eval?",
205
+ options: [
206
+ { value: "models", label: "compare models", hint: "force each model across every workflow step" },
207
+ { value: "config", label: "eval config", hint: "an overlay's real per-step model config (what ships)" },
208
+ ],
209
+ initialValue: "models",
210
+ });
211
+ if (p.isCancel(picked)) {
212
+ p.cancel("aborted");
213
+ return 1;
214
+ }
215
+ runType = picked;
216
+ }
217
+ else
218
+ runType = "models";
162
219
  // Asset roots FIRST — before any getWorkflow/runWorkflow. `--overlay` (or
163
220
  // LASTLIGHT_OVERLAY_DIR) layers a deployment's own workflows/skills over the
164
221
  // built-ins, and also contributes its `evals/datasets/` (see discovery).
165
222
  // With neither set, auto-detect a local `./instance/` overlay checkout — the
166
223
  // Separate layout `init --clone` produces — so a bare run "just works".
224
+ // `--overlay` may REPEAT in config mode (one arm per overlay).
167
225
  const autoInstance = join(process.cwd(), "instance");
168
226
  const autoOverlay = existsSync(join(autoInstance, "config.yaml")) ? autoInstance : undefined;
169
- const overlayDir = strFlag("overlay") ?? process.env.LASTLIGHT_OVERLAY_DIR ?? autoOverlay;
170
- if (overlayDir === autoOverlay && autoOverlay)
227
+ const overlayFlags = strFlagAll("overlay");
228
+ let overlays = overlayFlags.length
229
+ ? overlayFlags
230
+ : [process.env.LASTLIGHT_OVERLAY_DIR ?? autoOverlay].filter((d) => !!d);
231
+ // In a config-mode TTY with no overlay given, ask for one (blank ⇒ core defaults).
232
+ if (runType === "config" && !overlayFlags.length && process.stdin.isTTY) {
233
+ const ans = await p.text({
234
+ message: "Overlay dir whose config.yaml drives per-step models (blank = core defaults):",
235
+ placeholder: overlays[0] ?? autoInstance,
236
+ initialValue: overlays[0] ?? "",
237
+ });
238
+ if (p.isCancel(ans)) {
239
+ p.cancel("aborted");
240
+ return 1;
241
+ }
242
+ const dir = ans.trim();
243
+ overlays = dir ? [dir] : [];
244
+ }
245
+ // The primary overlay wires discovery + the initial asset bootstrap. Config
246
+ // arms re-bootstrap their own overlay before running (the asset root is a
247
+ // process global — see the serial loop below).
248
+ const overlayDir = overlays[0];
249
+ if (overlayDir && overlayDir === autoOverlay)
171
250
  p.log.info(`overlay → ${chalk.cyan("./instance")} ${chalk.dim("(auto-detected)")}`);
172
- bootstrapAssets({ overlayDir });
251
+ const { builtInRoot } = bootstrapAssets({ overlayDir });
173
252
  // A user/overlay can ship its own model registry too: explicit --models-file
174
253
  // wins, else an overlay's `evals/models.json` if present, else the built-in.
175
254
  const overlayModels = overlayDir ? join(overlayDir, "evals", "models.json") : undefined;
@@ -192,7 +271,6 @@ async function runEval() {
192
271
  p.outro(chalk.red("aborted"));
193
272
  return 1;
194
273
  }
195
- const compare = process.argv.includes("--compare");
196
274
  const noOpen = process.argv.includes("--no-open") || !!process.env.CI;
197
275
  const runs = intFlag("runs", 1);
198
276
  // Positional tier names — skip flags AND the values that follow value-flags.
@@ -248,35 +326,54 @@ async function runEval() {
248
326
  if (!discovered.has(t))
249
327
  p.log.warn(`Unknown tier "${t}". Known: ${known.join(", ")}`);
250
328
  }
251
- // Model selection precedence:
252
- // 1. --model / --models (or EVAL_MODEL[S]) — an explicit list, fuzzy-matched
253
- // against models.json; lets you test one model quickly.
254
- // 2. --compare — the full cross-vendor set (key-gated).
255
- // 3. default single model from models.json.
256
- // Each entry carries its provider family (env-key) for parallel grouping.
257
- const modelArg = strFlag("model") ?? strFlag("models");
258
- const mode = modelArg ? "select" : compare ? "compare" : "single";
259
- const entries = modelArg
260
- ? modelArg
261
- .split(",")
262
- .map((s) => s.trim())
263
- .filter(Boolean)
264
- .map((tok) => {
265
- const r = resolveModel(tok);
266
- return { id: r.id, family: r.family };
267
- })
268
- : compare
269
- ? compareModels().map((m) => ({ id: m.id, family: m.envKey ?? m.provider ?? "default" }))
270
- : evalModels().map((id) => ({ id, family: "default" }));
271
- if (!entries.length) {
329
+ // An "arm" is one column of the comparison, behind the `Arm` seam (src/arm.ts):
330
+ // `models` runs build one `modelsArm` per model (forced across every step);
331
+ // `config` runs build one `configArm` per overlay (its merged per-step config
332
+ // drives selection). Both flow through the same work-list → scorecard →
333
+ // dashboard, keyed on the arm's `label`. run.ts owns *which* arms exist (the
334
+ // flag + registry resolution below); the adapters own *how* to build one.
335
+ //
336
+ // The model-selection sub-mode (only meaningful for `models` runs) is shown in
337
+ // the plan note; `config` runs report their arm count instead.
338
+ const mode = runType === "config" ? "config" : modelArg ? "select" : compare ? "compare" : "single";
339
+ let arms;
340
+ if (runType === "config") {
341
+ // One arm per overlay (or a single core-defaults arm when none). `--model`
342
+ // resolves to an id that overrides each merged config's `default` for quick
343
+ // what-if runs (the override is applied inside configArm).
344
+ const configOverlays = overlays.length ? overlays : [overlayDir];
345
+ const defaultOverride = modelArg ? resolveModel(modelArg).id : undefined;
346
+ arms = configOverlays.map((dir) => configArm(builtInRoot, dir, defaultOverride));
347
+ }
348
+ else {
349
+ // Model selection precedence:
350
+ // 1. --model / --models (or EVAL_MODEL[S]) — an explicit list, fuzzy-matched.
351
+ // 2. --compare — the full cross-vendor set (key-gated).
352
+ // 3. default single model from models.json.
353
+ const entries = modelArg
354
+ ? modelArg
355
+ .split(",")
356
+ .map((s) => s.trim())
357
+ .filter(Boolean)
358
+ .map((tok) => {
359
+ const r = resolveModel(tok);
360
+ return { id: r.id, family: r.family };
361
+ })
362
+ : compare
363
+ ? compareModels().map((m) => ({ id: m.id, family: m.envKey ?? m.provider ?? "default" }))
364
+ : evalModels().map((id) => ({ id, family: "default" }));
365
+ arms = entries.map((e) => modelsArm(e.id, e.family));
366
+ }
367
+ if (!arms.length) {
272
368
  p.log.error("No comparison models available — set provider keys (OPENAI_API_KEY / ANTHROPIC_API_KEY /\n" +
273
369
  "FIREWORKS_API_KEY …) for the entries in evals/models.json.");
274
370
  p.outro(chalk.red("aborted"));
275
371
  return 1;
276
372
  }
277
373
  const labels = modelLabels();
278
- // Resolve the work-list up front so we can show deterministic progress.
279
- const work = [];
374
+ // Instances per tier, resolved ONCE — the case set is identical across arms
375
+ // (arms vary only the model selection / assets, never the cases).
376
+ const tierInstances = new Map();
280
377
  for (const tierName of tiers) {
281
378
  const tier = discovered.get(tierName);
282
379
  const instances = loadInstances(tier);
@@ -284,7 +381,15 @@ async function runEval() {
284
381
  p.log.warn(`tier "${tierName}": no instances at ${tier.instancesPath} — skipping`);
285
382
  continue;
286
383
  }
287
- for (const e of entries) {
384
+ tierInstances.set(tierName, instances);
385
+ }
386
+ // Resolve the work-list up front so we can show deterministic progress. Arms
387
+ // are the OUTER loop so a `config` run's per-arm overlay switches at most once
388
+ // (the serial loop re-bootstraps on arm change; see below).
389
+ const work = [];
390
+ for (const arm of arms) {
391
+ for (const [tierName, instances] of tierInstances) {
392
+ const tier = discovered.get(tierName);
288
393
  for (const inst of instances) {
289
394
  // Per-instance workflow wins, else the tier's defaultWorkflow (throws if
290
395
  // neither is set — surfaced as a harness error for that case).
@@ -292,8 +397,7 @@ async function runEval() {
292
397
  tierName,
293
398
  defaultWorkflow: workflowFor(tier, inst),
294
399
  datasetDir: tier.root,
295
- model: e.id,
296
- family: e.family,
400
+ arm,
297
401
  inst,
298
402
  });
299
403
  }
@@ -309,34 +413,84 @@ async function runEval() {
309
413
  // serial with --serial or when there's only one family.
310
414
  const byFamily = new Map();
311
415
  for (const w of work) {
312
- const arr = byFamily.get(w.family);
416
+ const arr = byFamily.get(w.arm.family);
313
417
  if (arr)
314
418
  arr.push(w);
315
419
  else
316
- byFamily.set(w.family, [w]);
420
+ byFamily.set(w.arm.family, [w]);
317
421
  }
318
- const parallel = !process.argv.includes("--serial") && byFamily.size > 1;
319
- const resultsDir = join(resultsRoot(), `${tiers.join("+")}${compare ? "-compare" : ""}`);
320
- const htmlBase = {
321
- models: entries.map((e) => labels[e.id] ?? e.id),
322
- tiers,
323
- labels,
422
+ // `config` runs stay SERIAL: arms may carry distinct overlays and the asset
423
+ // root is a process global, so the serial loop re-bootstraps per arm (below).
424
+ const parallel = runType === "models" && !process.argv.includes("--serial") && byFamily.size > 1;
425
+ // Each tier writes its OWN folder + scorecard (all sharing this run's id), so
426
+ // tiers stay separate in the dashboard instead of collapsing into one combined
427
+ // `<a+b>` entry. A single invocation appears as the same run under each tier it
428
+ // touched. The `-compare` suffix keeps cross-vendor runs on their own trend
429
+ // line, distinct from single-model runs of the same tier. The dashboard server
430
+ // indexes the whole tree; each tier dir is a `tierKey` segment the SPA routes on.
431
+ // `-compare` keeps cross-vendor runs on their own trend line; `-config` keeps
432
+ // config-eval runs on theirs — so the three run shapes never collapse together.
433
+ const gitSha = gitShortSha();
434
+ const tierKeyFor = (tier) => `${tier}${runType === "config" ? "-config" : compare ? "-compare" : ""}`;
435
+ // One shared runId, checked free in the first tier's dir (collisions in the
436
+ // same second across runs are what the suffix guards — rare, one dir suffices).
437
+ const runId = makeRunId(new Date(), gitSha, tierResultsDir(tierKeyFor(tiers[0])));
438
+ const resultsDirFor = (tier) => join(tierResultsDir(tierKeyFor(tier)), runId);
439
+ // Each case's session logs live under `sessions/<id>__<model>/trial-<N>/`
440
+ // (relative to the tier run dir) — the model is in the name since several
441
+ // models share a run dir; per-trial keeps every `--runs N` trial. `full.jsonl`
442
+ // is the consolidated live-followable transcript; `NN-<phase>.jsonl` are the
443
+ // per-phase splits written when the trial finishes.
444
+ const caseRelFor = (instanceId, model) => `sessions/${slug(instanceId)}__${slug(model)}`;
445
+ const trialRelFor = (instanceId, model, trial) => `${caseRelFor(instanceId, model)}/trial-${trial}`;
446
+ // Per-tier run metadata stamped into every scorecard write (the dashboard reads
447
+ // identity, labels, and live state straight off disk). `live`/`progress`/
448
+ // `pending`/`generatedAt` are layered on per write; `tiers` is the single tier.
449
+ const armLabels = arms.map((a) => labels[a.label] ?? a.label);
450
+ const baseMetaFor = (tier) => ({
451
+ runId,
452
+ runType,
453
+ tiers: [tier],
454
+ models: armLabels,
324
455
  runs,
325
- };
456
+ gitSha,
457
+ labels,
458
+ });
459
+ // In `config` runs the axis is the config(s); show the merged per-step model
460
+ // map for a single-arm run so the plan is legible.
461
+ const axisLine = runType === "config"
462
+ ? `${chalk.bold("configs")} ${armLabels.join(", ")}`
463
+ : `${chalk.bold("models")} ${armLabels.join(", ")}`;
464
+ // For a single-arm run, show the per-step model map (config arms) so the plan
465
+ // is legible. `describe()` returns the summary for config arms, undefined for
466
+ // models arms — the reach into the arm's config map stays behind the seam.
467
+ const armSummary = arms.length === 1 ? arms[0].describe() : undefined;
468
+ const phaseMapLine = armSummary ? `\n${chalk.bold("models")} ${chalk.dim(armSummary)}` : "";
326
469
  p.note(`${chalk.bold("mode")} ${mode}${parallel ? chalk.dim(` (parallel · ${byFamily.size} families)`) : ""}\n` +
327
- `${chalk.bold("models")} ${entries.map((e) => labels[e.id] ?? e.id).join(", ")}\n` +
470
+ `${axisLine}${phaseMapLine}\n` +
328
471
  `${chalk.bold("tiers")} ${tiers.join(", ")}\n` +
329
472
  `${chalk.bold("cases")} ${work.length}${runs > 1
330
473
  ? chalk.dim(` × ${runs} trials = ${work.length * runs} runs · worst-case verdict, mean cost`)
331
474
  : ""}`, "plan");
332
475
  // `total` counts individual trials so live progress advances per model call.
333
476
  const total = work.length * runs;
334
- // Open the report immediately (live placeholder) so it fills in as we go.
335
- writeHtml(resultsDir, summarize([]), { ...htmlBase, generatedAt: new Date().toISOString(), live: true, progress: `0/${total}` });
336
- const htmlFile = join(resultsDir, "index.html");
477
+ // Seed an empty live scorecard per tier so the dashboard has something to poll,
478
+ // then start the server and open the SPA deep-linked at this run. The server is
479
+ // skipped entirely when not opening (CI / --no-open) — we only write JSON.
480
+ for (const tier of tiers) {
481
+ writeScorecard(resultsDirFor(tier), withMeta(summarize([]), { ...baseMetaFor(tier), generatedAt: new Date().toISOString(), live: true, progress: `0/${total}` }));
482
+ }
483
+ let server;
337
484
  if (!noOpen) {
338
- openInBrowser(htmlFile);
339
- p.log.info(`Live report → ${chalk.cyan(htmlFile)} ${chalk.dim("(auto-refreshing)")}`);
485
+ try {
486
+ server = await startServer({ resultsRoot: resultsRoot(), dashboardRoot: dashboardDistRoot() });
487
+ const runUrl = `${server.url}/#/${encodeURIComponent(tierKeyFor(tiers[0]))}/${encodeURIComponent(runId)}`;
488
+ openInBrowser(runUrl);
489
+ p.log.info(`Live dashboard → ${chalk.cyan(runUrl)}`);
490
+ }
491
+ catch (err) {
492
+ p.log.warn(`Couldn't start the dashboard server (${err.message}) — writing JSON only.`);
493
+ }
340
494
  }
341
495
  const all = [];
342
496
  let harnessErrors = 0;
@@ -344,44 +498,74 @@ async function runEval() {
344
498
  // Track in-flight cases so the live report can show running / queued rows.
345
499
  const caseKey = (tier, model, id) => `${tier}|${model}|${id}`;
346
500
  const running = new Set();
347
- // writeHtml/summarize/all.push run synchronously to completion inside one
348
- // event-loop turn, so even with concurrent families they never interleave.
501
+ // Current trial number per running case, so the live "follow" link points at
502
+ // the right `trial-<N>/full.jsonl` (matters only for `--runs N>1`).
503
+ const trialOf = new Map();
504
+ // writeScorecard/summarize/all.push run synchronously to completion inside one
505
+ // event-loop turn, so even with concurrent families they never interleave; the
506
+ // temp-file+rename keeps a polling dashboard from reading a half-written file.
349
507
  const refresh = () => {
350
508
  const done = new Set(all.map((r) => caseKey(r.tier ?? "", r.model, r.instance_id)));
351
- const pending = work
352
- .map((w) => ({ w, k: caseKey(w.tierName, w.model, w.inst.instance_id) }))
353
- .filter(({ k }) => !done.has(k))
354
- .map(({ w, k }) => ({
355
- tier: w.tierName,
356
- model: w.model,
357
- instance_id: w.inst.instance_id,
358
- status: (running.has(k) ? "running" : "pending"),
359
- }));
360
- writeHtml(resultsDir, summarize(all), {
361
- ...htmlBase,
362
- generatedAt: new Date().toISOString(),
363
- live: true,
364
- progress: `${completed}/${total}`,
365
- pending,
366
- });
509
+ const now = new Date().toISOString();
510
+ for (const tier of tiers) {
511
+ const tierResults = all.filter((r) => (r.tier ?? "") === tier);
512
+ const pending = work
513
+ .filter((w) => w.tierName === tier)
514
+ .map((w) => ({ w, k: caseKey(w.tierName, w.arm.label, w.inst.instance_id) }))
515
+ .filter(({ k }) => !done.has(k))
516
+ .map(({ w, k }) => ({
517
+ tier: w.tierName,
518
+ model: w.arm.label,
519
+ instance_id: w.inst.instance_id,
520
+ status: running.has(k) ? "running" : "pending",
521
+ // Only a running case has a (live-updating) transcript to follow —
522
+ // point at the current trial's consolidated `full.jsonl`.
523
+ sessionLog: running.has(k)
524
+ ? `${trialRelFor(w.inst.instance_id, w.arm.label, trialOf.get(k) ?? 1)}/full.jsonl`
525
+ : undefined,
526
+ }));
527
+ // Per-tier progress (cases), not the global trial count — each tier's
528
+ // scorecard stands alone, so "0/5" across both tiers was misleading.
529
+ const tierCases = work.filter((w) => w.tierName === tier).length;
530
+ writeScorecard(resultsDirFor(tier), withMeta(summarize(tierResults), {
531
+ ...baseMetaFor(tier),
532
+ generatedAt: now,
533
+ live: true,
534
+ progress: `${tierResults.length}/${tierCases}`,
535
+ pending,
536
+ }));
537
+ }
367
538
  };
368
539
  // Run one case `runs` times and fold the trials into a single result
369
540
  // (worst-case verdict, mean metrics). `onTrial` ticks per model call.
370
541
  const runItem = async (w, onTrial) => {
542
+ const k = caseKey(w.tierName, w.arm.label, w.inst.instance_id);
371
543
  const trials = [];
372
- for (let t = 0; t < runs; t++) {
544
+ for (let t = 1; t <= runs; t++) {
545
+ trialOf.set(k, t); // so the live "follow" link targets this trial
546
+ const trialRel = trialRelFor(w.inst.instance_id, w.arm.label, t);
373
547
  const r = await runInstance(w.inst, {
374
- model: w.model,
548
+ // The arm carries all model selection (forced model / merged config) +
549
+ // the axis label; runInstance calls its prepare()/recordPhaseModel().
550
+ arm: w.arm,
375
551
  datasetDir: w.datasetDir,
376
552
  defaultWorkflow: w.defaultWorkflow,
377
553
  manageEnv: false,
554
+ // Per-trial dir: `full.jsonl` (consolidated, live) + `NN-<phase>.jsonl`.
555
+ sessionTrialDir: join(resultsDirFor(w.tierName), trialRel),
556
+ sessionTrialRel: trialRel,
557
+ trial: t,
378
558
  });
379
559
  r.tier = w.tierName;
380
560
  trials.push(r);
381
561
  completed++;
382
562
  onTrial();
383
563
  }
384
- return aggregateTrials(trials);
564
+ const agg = aggregateTrials(trials);
565
+ // Keep every trial's per-phase sessions on the aggregate (aggregateTrials
566
+ // only carries trial 0's fields through).
567
+ agg.sessions = trials.map((r) => r.sessionTrial).filter((s) => !!s);
568
+ return agg;
385
569
  };
386
570
  // Install the eval's static-token env ONCE for the whole batch so concurrent
387
571
  // runs share one stable baseline (manageEnv:false on every runInstance).
@@ -406,7 +590,7 @@ async function runEval() {
406
590
  try {
407
591
  await Promise.all([...byFamily].map(async ([f, items]) => {
408
592
  for (const w of items) {
409
- const k = caseKey(w.tierName, w.model, w.inst.instance_id);
593
+ const k = caseKey(w.tierName, w.arm.label, w.inst.instance_id);
410
594
  running.add(k);
411
595
  refresh();
412
596
  const result = await runItem(w, () => {
@@ -418,7 +602,7 @@ async function runEval() {
418
602
  all.push(result);
419
603
  if (result.error)
420
604
  harnessErrors++;
421
- const mark = result.error ? chalk.red("✗") : chalk.green("✓");
605
+ const mark = result.error ? chalk.red("✗") : result.blocked ? chalk.yellow("■") : chalk.green("✓");
422
606
  verdicts.push(`${mark} ${chalk.dim(familyLabel(f))} ${verdictLine(w.tierName, w.inst, result)}`);
423
607
  refresh();
424
608
  }
@@ -431,13 +615,25 @@ async function runEval() {
431
615
  p.log.message(verdicts.join("\n"));
432
616
  }
433
617
  else {
434
- // Serial: one spinner per case (updates per trial) + a verdict line.
618
+ // Serial: one spinner per case (updates per trial) + a verdict line. On
619
+ // each arm change `arm.activate()` repoints the (process-global) asset root
620
+ // to that arm's overlay (a no-op for models arms); work is arms-outer, so
621
+ // it fires at most once per arm. `releaseOverlayGuard()` lets the next arm
622
+ // switch overlays — without it the guard treats the switch as a concurrent
623
+ // overlay and throws (ADR 0001).
624
+ let currentArm;
435
625
  for (let i = 0; i < work.length; i++) {
436
626
  const w = work[i];
627
+ if (w.arm !== currentArm) {
628
+ if (currentArm)
629
+ releaseOverlayGuard();
630
+ w.arm.activate();
631
+ currentArm = w.arm;
632
+ }
437
633
  const s = makeSpinner();
438
- const head = `${chalk.dim(`[${i + 1}/${work.length}]`)} ${chalk.cyan(w.tierName)}/${w.inst.instance_id} ${chalk.dim(labels[w.model] ?? w.model)}`;
634
+ const head = `${chalk.dim(`[${i + 1}/${work.length}]`)} ${chalk.cyan(w.tierName)}/${w.inst.instance_id} ${chalk.dim(labels[w.arm.label] ?? w.arm.label)}`;
439
635
  s.start(head);
440
- const k = caseKey(w.tierName, w.model, w.inst.instance_id);
636
+ const k = caseKey(w.tierName, w.arm.label, w.inst.instance_id);
441
637
  running.add(k);
442
638
  refresh();
443
639
  let t = 0;
@@ -451,7 +647,7 @@ async function runEval() {
451
647
  all.push(result);
452
648
  if (result.error)
453
649
  harnessErrors++;
454
- const mark = result.error ? chalk.red("✗") : chalk.green("✓");
650
+ const mark = result.error ? chalk.red("✗") : result.blocked ? chalk.yellow("■") : chalk.green("✓");
455
651
  s.stop(`${chalk.dim(`[${i + 1}/${work.length}]`)} ${mark} ${verdictLine(w.tierName, w.inst, result)}`);
456
652
  if (result.error) {
457
653
  p.log.error(chalk.dim(result.error));
@@ -466,47 +662,73 @@ async function runEval() {
466
662
  finally {
467
663
  restoreEvalEnv();
468
664
  }
469
- // Final, static report + machine artifacts.
470
- const card = summarize(all);
471
- writeArtifacts(resultsDir, card);
472
- const html = writeHtml(resultsDir, card, { ...htmlBase, generatedAt: new Date().toISOString() });
473
- p.log.success(`Scorecard → ${chalk.cyan(html)}`);
474
- p.log.success(`Artifacts → ${chalk.cyan(resultsDir)}/{scorecard.json,predictions.jsonl}`);
665
+ // Final, static scorecard + machine artifacts. The run-level metadata is
666
+ // persisted into scorecard.json so the dashboard can label, order, and (no
667
+ // longer) live-poll the run without re-deriving from the current config.
668
+ const generatedAt = new Date().toISOString();
669
+ for (const tier of tiers) {
670
+ const tierResults = all.filter((r) => (r.tier ?? "") === tier);
671
+ writeArtifacts(resultsDirFor(tier), withMeta(summarize(tierResults), { ...baseMetaFor(tier), generatedAt, live: false }));
672
+ }
673
+ p.log.success(`Artifacts → ${chalk.cyan(tiers.map((t) => resultsDirFor(t)).join("\n "))}\n /{scorecard.json,predictions.jsonl,sessions/}`);
475
674
  const ran = runs > 1 ? `${completed} runs (${all.length} cases × ${runs})` : `${all.length} runs`;
476
- if (harnessErrors > 0) {
477
- p.outro(chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`));
675
+ // Keep the dashboard server alive so the just-finished run stays viewable.
676
+ // Only block in an interactive terminal — piped/non-TTY callers exit cleanly.
677
+ if (server && process.stdout.isTTY) {
678
+ const runUrl = `${server.url}/#/${encodeURIComponent(tierKeyFor(tiers[0]))}/${encodeURIComponent(runId)}`;
679
+ p.log.success(`Dashboard → ${chalk.cyan(runUrl)} ${chalk.dim("(serving · Ctrl-C to stop)")}`);
680
+ const tail = harnessErrors > 0 ? chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`) : chalk.green(`done — ${ran}`);
681
+ p.outro(tail);
682
+ await waitForSigint();
683
+ await server.close();
478
684
  }
479
685
  else {
480
- p.outro(chalk.green(`done — ${ran}, report at ${html}`));
686
+ if (server)
687
+ await server.close();
688
+ if (harnessErrors > 0) {
689
+ p.outro(chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`));
690
+ }
691
+ else {
692
+ p.outro(chalk.green(`done — ${ran}`));
693
+ }
481
694
  }
482
695
  // Non-zero ONLY on harness failure — model quality is the measurement.
483
696
  return harnessErrors > 0 ? 1 : 0;
484
697
  }
698
+ /** Resolve on the first SIGINT (Ctrl-C) so `run`/`serve` can keep a server up
699
+ * for browsing and shut it down cleanly when the user is done. */
700
+ function waitForSigint() {
701
+ return new Promise((resolve) => {
702
+ const onSig = () => {
703
+ process.off("SIGINT", onSig);
704
+ resolve();
705
+ };
706
+ process.on("SIGINT", onSig);
707
+ });
708
+ }
485
709
  /**
486
- * `report <dir>` — re-render `<dir>/index.html` from an existing
487
- * `<dir>/scorecard.json`, no models run. Lets you regenerate the HTML after a
488
- * report-template change (or to re-skin an old run) without paying for a fresh
489
- * eval. Models/tiers/labels are reconstructed from the scorecard itself.
710
+ * `serve` — start the dashboard server over `eval-results/` and open it in the
711
+ * browser to browse every past run (no models run). The same server `run` uses
712
+ * for the live report; blocks until Ctrl-C. There is no HTML to regenerate any
713
+ * more — the SPA reads the JSON directly.
490
714
  */
491
- function runReport(dir) {
492
- if (!dir) {
493
- console.error("usage: lastlight-evals report <results-dir> (a dir holding scorecard.json)");
494
- return 1;
715
+ async function runServe() {
716
+ loadDotEnv();
717
+ const noOpen = process.argv.includes("--no-open");
718
+ const port = intFlag("port", 0) || undefined;
719
+ let server;
720
+ try {
721
+ server = await startServer({ resultsRoot: resultsRoot(), dashboardRoot: dashboardDistRoot(), port });
495
722
  }
496
- const file = join(dir, "scorecard.json");
497
- if (!existsSync(file)) {
498
- console.error(`no scorecard.json in ${dir}`);
723
+ catch (err) {
724
+ console.error(`Couldn't start the dashboard server: ${err.message}`);
499
725
  return 1;
500
726
  }
501
- const card = JSON.parse(readFileSync(file, "utf8"));
502
- const labels = modelLabels();
503
- // Tiers/models come straight off the saved results so the re-render matches
504
- // exactly what that run measured (no dependence on current models.json).
505
- const tiers = [...new Set(card.results.map((r) => r.tier ?? "triage"))];
506
- const models = [...new Set(card.results.map((r) => r.model))].map((m) => labels[m] ?? m);
507
- const runs = Math.max(1, ...card.results.map((r) => r.trials ?? 1));
508
- const html = writeHtml(dir, card, { models, tiers, labels, runs, generatedAt: new Date().toISOString() });
509
- console.log(`Scorecard → ${html}`);
727
+ console.log(`Dashboard → ${server.url} (serving ${resultsRoot()} · Ctrl-C to stop)`);
728
+ if (!noOpen)
729
+ openInBrowser(server.url);
730
+ await waitForSigint();
731
+ await server.close();
510
732
  return 0;
511
733
  }
512
734
  const USAGE = `lastlight-evals — eval harness for Last Light workflows
@@ -514,21 +736,31 @@ const USAGE = `lastlight-evals — eval harness for Last Light workflows
514
736
  Usage:
515
737
  lastlight-evals [run] [tiers...] [options] Run evals (default command)
516
738
  lastlight-evals init [dir] [options] Scaffold an overlay+evals workspace
517
- lastlight-evals report <results-dir> Re-render index.html from scorecard.json
739
+ lastlight-evals serve [options] Browse past runs in the dashboard (no models run)
518
740
 
519
741
  Run options:
520
- --overlay <dir> Layer a deployment's workflows/skills + evals/ over built-ins
521
- --model <m[,m2]> Model(s) to run (fuzzy-matched against models.json)
742
+ --mode <models|config> Comparison axis. models (default): force each --model
743
+ across every step. config: run an overlay's real
744
+ per-step model config (its config.yaml). No flags in a
745
+ TTY ⇒ asks.
746
+ --overlay <dir> Layer a deployment's workflows/skills + evals/ over built-ins.
747
+ Repeatable in --mode config (one arm per overlay).
748
+ --model <m[,m2]> models: model(s) to run (fuzzy-matched against models.json).
749
+ config: override each config's default model.
522
750
  --compare Cross-vendor set (only models whose provider key is present)
523
751
  --runs <n> Repeat each case n× (worst-case verdict, mean metrics)
524
752
  --serial Force serial execution across provider families
525
753
  --datasets <dir> Extra datasets root to discover tiers from
526
754
  --models-file <f> Use an explicit models.json
527
- --no-open Don't open the HTML report (also implied by CI=1)
755
+ --no-open Don't open / auto-serve the dashboard (also implied by CI=1)
756
+
757
+ Serve options:
758
+ --port <n> Preferred port for the dashboard server (default 4319)
759
+ --no-open Start the server but don't open a browser
528
760
 
529
761
  Run \`lastlight-evals init --help\` for init-specific flags.
530
762
  GitHub is mocked end-to-end — no real GitHub token is needed, only a provider key.`;
531
- /** Top-level subcommand dispatcher: `run` (default) | `init` | `report`. */
763
+ /** Top-level subcommand dispatcher: `run` (default) | `init` | `serve`. */
532
764
  async function main() {
533
765
  const sub = process.argv[2];
534
766
  // Top-level help — only when it's not standing in for a `run` tier name.
@@ -540,9 +772,9 @@ async function main() {
540
772
  // `init [dir] [flags]` — scaffold a fresh overlay+evals repo.
541
773
  return runInit(process.argv.slice(3));
542
774
  }
543
- if (sub === "report") {
544
- // `report <dir>` — re-render index.html from a saved scorecard.json.
545
- return runReport(process.argv[3]);
775
+ if (sub === "serve") {
776
+ // `serve` — browse past runs; the live dashboard server, standalone.
777
+ return runServe();
546
778
  }
547
779
  // `run` is the default; allow an explicit leading `run` token too.
548
780
  if (sub === "run")