lastlight-evals 0.1.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/README.md +98 -12
  2. package/dashboard/dist/assets/index-YwScAAxm.js +238 -0
  3. package/dashboard/dist/assets/index-uU9Sj3M4.css +1 -0
  4. package/dashboard/dist/index.html +19 -0
  5. package/dashboard/dist/logo.png +0 -0
  6. package/datasets/code-fix/repos/codefix__date-range-off-by-one/.github/workflows/ci.yml +16 -0
  7. package/datasets/code-fix/repos/codefix__date-range-off-by-one/README.md +7 -1
  8. package/datasets/code-fix/repos/codefix__date-range-off-by-one/package.json +3 -1
  9. package/datasets/code-fix/repos/codefix__date-range-off-by-one/scripts/lint.mjs +25 -0
  10. package/datasets/code-fix/repos/codefix__date-range-off-by-one/test/date-range.test.ts +14 -0
  11. package/datasets/code-fix/repos/codefix__date-range-off-by-one/tsconfig.json +13 -0
  12. package/datasets/code-fix/tests/codefix__date-range-off-by-one/date-range.test.ts +5 -5
  13. package/dist/config.js +100 -0
  14. package/dist/config.js.map +1 -0
  15. package/dist/mechanism.test.js +41 -0
  16. package/dist/mechanism.test.js.map +1 -1
  17. package/dist/metrics.js +83 -0
  18. package/dist/metrics.js.map +1 -1
  19. package/dist/paths.js +51 -1
  20. package/dist/paths.js.map +1 -1
  21. package/dist/report.js +112 -3
  22. package/dist/report.js.map +1 -1
  23. package/dist/run-instance.js +141 -14
  24. package/dist/run-instance.js.map +1 -1
  25. package/dist/run.js +342 -112
  26. package/dist/run.js.map +1 -1
  27. package/dist/serve.js +136 -0
  28. package/dist/serve.js.map +1 -0
  29. package/examples/overlay/README.md +30 -0
  30. package/examples/overlay/config.yaml +36 -0
  31. package/examples/overlay-anthropic/config.yaml +27 -0
  32. package/package.json +16 -6
  33. package/dist/html-report.js +0 -325
  34. package/dist/html-report.js.map +0 -1
package/dist/run.js CHANGED
@@ -17,18 +17,19 @@
17
17
  * The deterministic, AI-free plumbing is covered separately by
18
18
  * `evals/mechanism.test.ts` in the normal `npm test` suite.
19
19
  */
20
- import { existsSync, readFileSync } from "node:fs";
21
- import { join } from "node:path";
20
+ import { existsSync } from "node:fs";
21
+ import { basename, join } from "node:path";
22
22
  import { spawn } from "node:child_process";
23
23
  import * as p from "@clack/prompts";
24
24
  import chalk from "chalk";
25
25
  import { loadDotEnv, hasProviderKey, evalModels, compareModels, modelLabels, resolveModel, setModelsPath } from "./env.js";
26
- import { runInstance, applyEvalEnv } from "./run-instance.js";
27
- import { summarize, writeArtifacts, aggregateTrials } from "./report.js";
28
- import { writeHtml } from "./html-report.js";
26
+ import { runInstance, applyEvalEnv, slug } from "./run-instance.js";
27
+ import { loadMergedConfig } from "./config.js";
28
+ import { summarize, writeArtifacts, writeScorecard, aggregateTrials, } from "./report.js";
29
29
  import { bootstrapAssets } from "./bootstrap.js";
30
30
  import { discoverTiers, loadInstances, workflowFor } from "./discovery.js";
31
- import { builtinDatasetsRoot, resultsRoot } from "./paths.js";
31
+ import { builtinDatasetsRoot, tierResultsDir, makeRunId, gitShortSha, resultsRoot, dashboardDistRoot } from "./paths.js";
32
+ import { startServer } from "./serve.js";
32
33
  import { runInit } from "./init.js";
33
34
  /**
34
35
  * A clack spinner in a TTY; in non-TTY (CI / piped / agent) a quiet stub that
@@ -48,9 +49,8 @@ function makeSpinner() {
48
49
  },
49
50
  };
50
51
  }
51
- /** Open a file in the OS default browser (best-effort, never throws). */
52
- function openInBrowser(file) {
53
- const url = `file://${file}`;
52
+ /** Open a URL in the OS default browser (best-effort, never throws). */
53
+ function openInBrowser(url) {
54
54
  const [cmd, args] = process.platform === "darwin"
55
55
  ? ["open", [url]]
56
56
  : process.platform === "win32"
@@ -99,6 +99,12 @@ function fmtMs(ms) {
99
99
  function familyLabel(envKey) {
100
100
  return envKey.replace(/_API_KEY$/i, "").toLowerCase() || "default";
101
101
  }
102
+ /** Attach run metadata to a scorecard (mutate-and-return; `summarize` hands back
103
+ * a fresh object each call, so this is safe). */
104
+ function withMeta(card, meta) {
105
+ card.meta = meta;
106
+ return card;
107
+ }
102
108
  /**
103
109
  * Silence `console.*` for the whole batch (parallel mode). The per-run
104
110
  * `quiet()` swap saves/restores console and would corrupt under concurrent
@@ -116,6 +122,8 @@ function verdictLine(tierName, inst, r) {
116
122
  const head = `${chalk.cyan(tierName)}/${inst.instance_id}`;
117
123
  if (r.error)
118
124
  return `${head} ${chalk.red("harness error")}`;
125
+ if (r.blocked)
126
+ return `${head} ${chalk.yellow("blocked")} ${chalk.dim("(workflow gate)")}`;
119
127
  const count = (pass) => (pass !== undefined && r.trials ? chalk.dim(` ${pass}/${r.trials}`) : "");
120
128
  const parts = [];
121
129
  if (r.resolved !== undefined)
@@ -154,22 +162,93 @@ function strFlag(name) {
154
162
  }
155
163
  return process.env[`EVAL_${name.toUpperCase()}`];
156
164
  }
165
+ /** Collect ALL occurrences of a repeatable string flag (`--name a --name b` or
166
+ * `--name=a`). Used by `--overlay`, which may repeat in `config` mode to compare
167
+ * one arm per overlay. */
168
+ function strFlagAll(name) {
169
+ const out = [];
170
+ const argv = process.argv;
171
+ for (let i = 0; i < argv.length; i++) {
172
+ if (argv[i] === `--${name}` && argv[i + 1] !== undefined) {
173
+ out.push(argv[i + 1]);
174
+ i++;
175
+ continue;
176
+ }
177
+ const m = argv[i].match(new RegExp(`^--${name}=(.+)$`));
178
+ if (m)
179
+ out.push(m[1]);
180
+ }
181
+ return out;
182
+ }
157
183
  /** CLI flags that take a following value (so it isn't read as a tier name). */
158
- const VALUE_FLAGS = new Set(["--runs", "--model", "--models", "--overlay", "--datasets", "--models-file"]);
184
+ const VALUE_FLAGS = new Set(["--runs", "--model", "--models", "--mode", "--overlay", "--datasets", "--models-file"]);
159
185
  async function runEval() {
160
186
  loadDotEnv();
161
187
  p.intro(chalk.bold(`Last Light ${chalk.yellow("·")} eval`));
188
+ // Run type — the comparison axis:
189
+ // `models` — compare N models, each FORCED across every workflow step.
190
+ // `config` — run an overlay's REAL per-step model config (what ships); the
191
+ // arm is the config/overlay, compared across runs (or N overlays
192
+ // side-by-side). `--mode` wins; an explicit `--model`/`--compare`
193
+ // implies `models`; otherwise ask in a TTY (default `models`).
194
+ const modeFlag = strFlag("mode");
195
+ const modelArg = strFlag("model") ?? strFlag("models");
196
+ const compare = process.argv.includes("--compare");
197
+ let runType;
198
+ if (modeFlag === "config")
199
+ runType = "config";
200
+ else if (modeFlag === "models" || modelArg || compare)
201
+ runType = "models";
202
+ else if (process.stdin.isTTY) {
203
+ const picked = await p.select({
204
+ message: "What do you want to eval?",
205
+ options: [
206
+ { value: "models", label: "compare models", hint: "force each model across every workflow step" },
207
+ { value: "config", label: "eval config", hint: "an overlay's real per-step model config (what ships)" },
208
+ ],
209
+ initialValue: "models",
210
+ });
211
+ if (p.isCancel(picked)) {
212
+ p.cancel("aborted");
213
+ return 1;
214
+ }
215
+ runType = picked;
216
+ }
217
+ else
218
+ runType = "models";
162
219
  // Asset roots FIRST — before any getWorkflow/runWorkflow. `--overlay` (or
163
220
  // LASTLIGHT_OVERLAY_DIR) layers a deployment's own workflows/skills over the
164
221
  // built-ins, and also contributes its `evals/datasets/` (see discovery).
165
222
  // With neither set, auto-detect a local `./instance/` overlay checkout — the
166
223
  // Separate layout `init --clone` produces — so a bare run "just works".
224
+ // `--overlay` may REPEAT in config mode (one arm per overlay).
167
225
  const autoInstance = join(process.cwd(), "instance");
168
226
  const autoOverlay = existsSync(join(autoInstance, "config.yaml")) ? autoInstance : undefined;
169
- const overlayDir = strFlag("overlay") ?? process.env.LASTLIGHT_OVERLAY_DIR ?? autoOverlay;
170
- if (overlayDir === autoOverlay && autoOverlay)
227
+ const overlayFlags = strFlagAll("overlay");
228
+ let overlays = overlayFlags.length
229
+ ? overlayFlags
230
+ : [process.env.LASTLIGHT_OVERLAY_DIR ?? autoOverlay].filter((d) => !!d);
231
+ // In a config-mode TTY with no overlay given, ask for one (blank ⇒ core defaults).
232
+ if (runType === "config" && !overlayFlags.length && process.stdin.isTTY) {
233
+ const ans = await p.text({
234
+ message: "Overlay dir whose config.yaml drives per-step models (blank = core defaults):",
235
+ placeholder: overlays[0] ?? autoInstance,
236
+ initialValue: overlays[0] ?? "",
237
+ });
238
+ if (p.isCancel(ans)) {
239
+ p.cancel("aborted");
240
+ return 1;
241
+ }
242
+ const dir = ans.trim();
243
+ overlays = dir ? [dir] : [];
244
+ }
245
+ // The primary overlay wires discovery + the initial asset bootstrap. Config
246
+ // arms re-bootstrap their own overlay before running (the asset root is a
247
+ // process global — see the serial loop below).
248
+ const overlayDir = overlays[0];
249
+ if (overlayDir && overlayDir === autoOverlay)
171
250
  p.log.info(`overlay → ${chalk.cyan("./instance")} ${chalk.dim("(auto-detected)")}`);
172
- bootstrapAssets({ overlayDir });
251
+ const { builtInRoot } = bootstrapAssets({ overlayDir });
173
252
  // A user/overlay can ship its own model registry too: explicit --models-file
174
253
  // wins, else an overlay's `evals/models.json` if present, else the built-in.
175
254
  const overlayModels = overlayDir ? join(overlayDir, "evals", "models.json") : undefined;
@@ -192,7 +271,6 @@ async function runEval() {
192
271
  p.outro(chalk.red("aborted"));
193
272
  return 1;
194
273
  }
195
- const compare = process.argv.includes("--compare");
196
274
  const noOpen = process.argv.includes("--no-open") || !!process.env.CI;
197
275
  const runs = intFlag("runs", 1);
198
276
  // Positional tier names — skip flags AND the values that follow value-flags.
@@ -248,35 +326,51 @@ async function runEval() {
248
326
  if (!discovered.has(t))
249
327
  p.log.warn(`Unknown tier "${t}". Known: ${known.join(", ")}`);
250
328
  }
251
- // Model selection precedence:
252
- // 1. --model / --models (or EVAL_MODEL[S]) — an explicit list, fuzzy-matched
253
- // against models.json; lets you test one model quickly.
254
- // 2. --compare — the full cross-vendor set (key-gated).
255
- // 3. default single model from models.json.
256
- // Each entry carries its provider family (env-key) for parallel grouping.
257
- const modelArg = strFlag("model") ?? strFlag("models");
258
- const mode = modelArg ? "select" : compare ? "compare" : "single";
259
- const entries = modelArg
260
- ? modelArg
261
- .split(",")
262
- .map((s) => s.trim())
263
- .filter(Boolean)
264
- .map((tok) => {
265
- const r = resolveModel(tok);
266
- return { id: r.id, family: r.family };
267
- })
268
- : compare
269
- ? compareModels().map((m) => ({ id: m.id, family: m.envKey ?? m.provider ?? "default" }))
270
- : evalModels().map((id) => ({ id, family: "default" }));
271
- if (!entries.length) {
329
+ // The model-selection sub-mode (only meaningful for `models` runs); shown in
330
+ // the plan note. `config` runs report their arm count instead.
331
+ const mode = runType === "config" ? "config" : modelArg ? "select" : compare ? "compare" : "single";
332
+ let arms;
333
+ if (runType === "config") {
334
+ // One arm per overlay (or a single core-defaults arm when none). `--model`
335
+ // overrides each merged config's `default` key for quick what-if runs.
336
+ const configOverlays = overlays.length ? overlays : [overlayDir];
337
+ arms = configOverlays.map((dir) => {
338
+ const merged = loadMergedConfig(builtInRoot, dir);
339
+ if (modelArg)
340
+ merged.models.default = resolveModel(modelArg).id;
341
+ const label = dir ? basename(dir) : "config";
342
+ return { label, family: label, modelConfig: merged.models, variantConfig: merged.variants, overlayDir: dir };
343
+ });
344
+ }
345
+ else {
346
+ // Model selection precedence:
347
+ // 1. --model / --models (or EVAL_MODEL[S]) — an explicit list, fuzzy-matched.
348
+ // 2. --compare — the full cross-vendor set (key-gated).
349
+ // 3. default single model from models.json.
350
+ const entries = modelArg
351
+ ? modelArg
352
+ .split(",")
353
+ .map((s) => s.trim())
354
+ .filter(Boolean)
355
+ .map((tok) => {
356
+ const r = resolveModel(tok);
357
+ return { id: r.id, family: r.family };
358
+ })
359
+ : compare
360
+ ? compareModels().map((m) => ({ id: m.id, family: m.envKey ?? m.provider ?? "default" }))
361
+ : evalModels().map((id) => ({ id, family: "default" }));
362
+ arms = entries.map((e) => ({ label: e.id, family: e.family }));
363
+ }
364
+ if (!arms.length) {
272
365
  p.log.error("No comparison models available — set provider keys (OPENAI_API_KEY / ANTHROPIC_API_KEY /\n" +
273
366
  "FIREWORKS_API_KEY …) for the entries in evals/models.json.");
274
367
  p.outro(chalk.red("aborted"));
275
368
  return 1;
276
369
  }
277
370
  const labels = modelLabels();
278
- // Resolve the work-list up front so we can show deterministic progress.
279
- const work = [];
371
+ // Instances per tier, resolved ONCE — the case set is identical across arms
372
+ // (arms vary only the model selection / assets, never the cases).
373
+ const tierInstances = new Map();
280
374
  for (const tierName of tiers) {
281
375
  const tier = discovered.get(tierName);
282
376
  const instances = loadInstances(tier);
@@ -284,7 +378,15 @@ async function runEval() {
284
378
  p.log.warn(`tier "${tierName}": no instances at ${tier.instancesPath} — skipping`);
285
379
  continue;
286
380
  }
287
- for (const e of entries) {
381
+ tierInstances.set(tierName, instances);
382
+ }
383
+ // Resolve the work-list up front so we can show deterministic progress. Arms
384
+ // are the OUTER loop so a `config` run's per-arm overlay switches at most once
385
+ // (the serial loop re-bootstraps on arm change; see below).
386
+ const work = [];
387
+ for (const arm of arms) {
388
+ for (const [tierName, instances] of tierInstances) {
389
+ const tier = discovered.get(tierName);
288
390
  for (const inst of instances) {
289
391
  // Per-instance workflow wins, else the tier's defaultWorkflow (throws if
290
392
  // neither is set — surfaced as a harness error for that case).
@@ -292,8 +394,11 @@ async function runEval() {
292
394
  tierName,
293
395
  defaultWorkflow: workflowFor(tier, inst),
294
396
  datasetDir: tier.root,
295
- model: e.id,
296
- family: e.family,
397
+ model: arm.label,
398
+ family: arm.family,
399
+ modelConfig: arm.modelConfig,
400
+ variantConfig: arm.variantConfig,
401
+ overlayDir: arm.overlayDir,
297
402
  inst,
298
403
  });
299
404
  }
@@ -315,28 +420,78 @@ async function runEval() {
315
420
  else
316
421
  byFamily.set(w.family, [w]);
317
422
  }
318
- const parallel = !process.argv.includes("--serial") && byFamily.size > 1;
319
- const resultsDir = join(resultsRoot(), `${tiers.join("+")}${compare ? "-compare" : ""}`);
320
- const htmlBase = {
321
- models: entries.map((e) => labels[e.id] ?? e.id),
322
- tiers,
323
- labels,
423
+ // `config` runs stay SERIAL: arms may carry distinct overlays and the asset
424
+ // root is a process global, so the serial loop re-bootstraps per arm (below).
425
+ const parallel = runType === "models" && !process.argv.includes("--serial") && byFamily.size > 1;
426
+ // Each tier writes its OWN folder + scorecard (all sharing this run's id), so
427
+ // tiers stay separate in the dashboard instead of collapsing into one combined
428
+ // `<a+b>` entry. A single invocation appears as the same run under each tier it
429
+ // touched. The `-compare` suffix keeps cross-vendor runs on their own trend
430
+ // line, distinct from single-model runs of the same tier. The dashboard server
431
+ // indexes the whole tree; each tier dir is a `tierKey` segment the SPA routes on.
432
+ // `-compare` keeps cross-vendor runs on their own trend line; `-config` keeps
433
+ // config-eval runs on theirs — so the three run shapes never collapse together.
434
+ const gitSha = gitShortSha();
435
+ const tierKeyFor = (tier) => `${tier}${runType === "config" ? "-config" : compare ? "-compare" : ""}`;
436
+ // One shared runId, checked free in the first tier's dir (collisions in the
437
+ // same second across runs are what the suffix guards — rare, one dir suffices).
438
+ const runId = makeRunId(new Date(), gitSha, tierResultsDir(tierKeyFor(tiers[0])));
439
+ const resultsDirFor = (tier) => join(tierResultsDir(tierKeyFor(tier)), runId);
440
+ // Each case's session logs live under `sessions/<id>__<model>/trial-<N>/`
441
+ // (relative to the tier run dir) — the model is in the name since several
442
+ // models share a run dir; per-trial keeps every `--runs N` trial. `full.jsonl`
443
+ // is the consolidated live-followable transcript; `NN-<phase>.jsonl` are the
444
+ // per-phase splits written when the trial finishes.
445
+ const caseRelFor = (instanceId, model) => `sessions/${slug(instanceId)}__${slug(model)}`;
446
+ const trialRelFor = (instanceId, model, trial) => `${caseRelFor(instanceId, model)}/trial-${trial}`;
447
+ // Per-tier run metadata stamped into every scorecard write (the dashboard reads
448
+ // identity, labels, and live state straight off disk). `live`/`progress`/
449
+ // `pending`/`generatedAt` are layered on per write; `tiers` is the single tier.
450
+ const armLabels = arms.map((a) => labels[a.label] ?? a.label);
451
+ const baseMetaFor = (tier) => ({
452
+ runId,
453
+ runType,
454
+ tiers: [tier],
455
+ models: armLabels,
324
456
  runs,
325
- };
457
+ gitSha,
458
+ labels,
459
+ });
460
+ // In `config` runs the axis is the config(s); show the merged per-step model
461
+ // map for a single-arm run so the plan is legible.
462
+ const axisLine = runType === "config"
463
+ ? `${chalk.bold("configs")} ${armLabels.join(", ")}`
464
+ : `${chalk.bold("models")} ${armLabels.join(", ")}`;
465
+ const phaseMapLine = runType === "config" && arms.length === 1 && arms[0].modelConfig
466
+ ? `\n${chalk.bold("models")} ${chalk.dim(Object.entries(arms[0].modelConfig)
467
+ .map(([k, v]) => `${k}→${v}`)
468
+ .join(" "))}`
469
+ : "";
326
470
  p.note(`${chalk.bold("mode")} ${mode}${parallel ? chalk.dim(` (parallel · ${byFamily.size} families)`) : ""}\n` +
327
- `${chalk.bold("models")} ${entries.map((e) => labels[e.id] ?? e.id).join(", ")}\n` +
471
+ `${axisLine}${phaseMapLine}\n` +
328
472
  `${chalk.bold("tiers")} ${tiers.join(", ")}\n` +
329
473
  `${chalk.bold("cases")} ${work.length}${runs > 1
330
474
  ? chalk.dim(` × ${runs} trials = ${work.length * runs} runs · worst-case verdict, mean cost`)
331
475
  : ""}`, "plan");
332
476
  // `total` counts individual trials so live progress advances per model call.
333
477
  const total = work.length * runs;
334
- // Open the report immediately (live placeholder) so it fills in as we go.
335
- writeHtml(resultsDir, summarize([]), { ...htmlBase, generatedAt: new Date().toISOString(), live: true, progress: `0/${total}` });
336
- const htmlFile = join(resultsDir, "index.html");
478
+ // Seed an empty live scorecard per tier so the dashboard has something to poll,
479
+ // then start the server and open the SPA deep-linked at this run. The server is
480
+ // skipped entirely when not opening (CI / --no-open) — we only write JSON.
481
+ for (const tier of tiers) {
482
+ writeScorecard(resultsDirFor(tier), withMeta(summarize([]), { ...baseMetaFor(tier), generatedAt: new Date().toISOString(), live: true, progress: `0/${total}` }));
483
+ }
484
+ let server;
337
485
  if (!noOpen) {
338
- openInBrowser(htmlFile);
339
- p.log.info(`Live report → ${chalk.cyan(htmlFile)} ${chalk.dim("(auto-refreshing)")}`);
486
+ try {
487
+ server = await startServer({ resultsRoot: resultsRoot(), dashboardRoot: dashboardDistRoot() });
488
+ const runUrl = `${server.url}/#/${encodeURIComponent(tierKeyFor(tiers[0]))}/${encodeURIComponent(runId)}`;
489
+ openInBrowser(runUrl);
490
+ p.log.info(`Live dashboard → ${chalk.cyan(runUrl)}`);
491
+ }
492
+ catch (err) {
493
+ p.log.warn(`Couldn't start the dashboard server (${err.message}) — writing JSON only.`);
494
+ }
340
495
  }
341
496
  const all = [];
342
497
  let harnessErrors = 0;
@@ -344,44 +499,76 @@ async function runEval() {
344
499
  // Track in-flight cases so the live report can show running / queued rows.
345
500
  const caseKey = (tier, model, id) => `${tier}|${model}|${id}`;
346
501
  const running = new Set();
347
- // writeHtml/summarize/all.push run synchronously to completion inside one
348
- // event-loop turn, so even with concurrent families they never interleave.
502
+ // Current trial number per running case, so the live "follow" link points at
503
+ // the right `trial-<N>/full.jsonl` (matters only for `--runs N>1`).
504
+ const trialOf = new Map();
505
+ // writeScorecard/summarize/all.push run synchronously to completion inside one
506
+ // event-loop turn, so even with concurrent families they never interleave; the
507
+ // temp-file+rename keeps a polling dashboard from reading a half-written file.
349
508
  const refresh = () => {
350
509
  const done = new Set(all.map((r) => caseKey(r.tier ?? "", r.model, r.instance_id)));
351
- const pending = work
352
- .map((w) => ({ w, k: caseKey(w.tierName, w.model, w.inst.instance_id) }))
353
- .filter(({ k }) => !done.has(k))
354
- .map(({ w, k }) => ({
355
- tier: w.tierName,
356
- model: w.model,
357
- instance_id: w.inst.instance_id,
358
- status: (running.has(k) ? "running" : "pending"),
359
- }));
360
- writeHtml(resultsDir, summarize(all), {
361
- ...htmlBase,
362
- generatedAt: new Date().toISOString(),
363
- live: true,
364
- progress: `${completed}/${total}`,
365
- pending,
366
- });
510
+ const now = new Date().toISOString();
511
+ for (const tier of tiers) {
512
+ const tierResults = all.filter((r) => (r.tier ?? "") === tier);
513
+ const pending = work
514
+ .filter((w) => w.tierName === tier)
515
+ .map((w) => ({ w, k: caseKey(w.tierName, w.model, w.inst.instance_id) }))
516
+ .filter(({ k }) => !done.has(k))
517
+ .map(({ w, k }) => ({
518
+ tier: w.tierName,
519
+ model: w.model,
520
+ instance_id: w.inst.instance_id,
521
+ status: running.has(k) ? "running" : "pending",
522
+ // Only a running case has a (live-updating) transcript to follow —
523
+ // point at the current trial's consolidated `full.jsonl`.
524
+ sessionLog: running.has(k)
525
+ ? `${trialRelFor(w.inst.instance_id, w.model, trialOf.get(k) ?? 1)}/full.jsonl`
526
+ : undefined,
527
+ }));
528
+ // Per-tier progress (cases), not the global trial count — each tier's
529
+ // scorecard stands alone, so "0/5" across both tiers was misleading.
530
+ const tierCases = work.filter((w) => w.tierName === tier).length;
531
+ writeScorecard(resultsDirFor(tier), withMeta(summarize(tierResults), {
532
+ ...baseMetaFor(tier),
533
+ generatedAt: now,
534
+ live: true,
535
+ progress: `${tierResults.length}/${tierCases}`,
536
+ pending,
537
+ }));
538
+ }
367
539
  };
368
540
  // Run one case `runs` times and fold the trials into a single result
369
541
  // (worst-case verdict, mean metrics). `onTrial` ticks per model call.
370
542
  const runItem = async (w, onTrial) => {
543
+ const k = caseKey(w.tierName, w.model, w.inst.instance_id);
371
544
  const trials = [];
372
- for (let t = 0; t < runs; t++) {
545
+ for (let t = 1; t <= runs; t++) {
546
+ trialOf.set(k, t); // so the live "follow" link targets this trial
547
+ const trialRel = trialRelFor(w.inst.instance_id, w.model, t);
373
548
  const r = await runInstance(w.inst, {
374
549
  model: w.model,
550
+ // `config` arms carry the merged per-step maps; `models` arms leave
551
+ // these undefined so the workflow forces `w.model` on every step.
552
+ modelConfig: w.modelConfig,
553
+ variantConfig: w.variantConfig,
375
554
  datasetDir: w.datasetDir,
376
555
  defaultWorkflow: w.defaultWorkflow,
377
556
  manageEnv: false,
557
+ // Per-trial dir: `full.jsonl` (consolidated, live) + `NN-<phase>.jsonl`.
558
+ sessionTrialDir: join(resultsDirFor(w.tierName), trialRel),
559
+ sessionTrialRel: trialRel,
560
+ trial: t,
378
561
  });
379
562
  r.tier = w.tierName;
380
563
  trials.push(r);
381
564
  completed++;
382
565
  onTrial();
383
566
  }
384
- return aggregateTrials(trials);
567
+ const agg = aggregateTrials(trials);
568
+ // Keep every trial's per-phase sessions on the aggregate (aggregateTrials
569
+ // only carries trial 0's fields through).
570
+ agg.sessions = trials.map((r) => r.sessionTrial).filter((s) => !!s);
571
+ return agg;
385
572
  };
386
573
  // Install the eval's static-token env ONCE for the whole batch so concurrent
387
574
  // runs share one stable baseline (manageEnv:false on every runInstance).
@@ -418,7 +605,7 @@ async function runEval() {
418
605
  all.push(result);
419
606
  if (result.error)
420
607
  harnessErrors++;
421
- const mark = result.error ? chalk.red("✗") : chalk.green("✓");
608
+ const mark = result.error ? chalk.red("✗") : result.blocked ? chalk.yellow("■") : chalk.green("✓");
422
609
  verdicts.push(`${mark} ${chalk.dim(familyLabel(f))} ${verdictLine(w.tierName, w.inst, result)}`);
423
610
  refresh();
424
611
  }
@@ -432,8 +619,15 @@ async function runEval() {
432
619
  }
433
620
  else {
434
621
  // Serial: one spinner per case (updates per trial) + a verdict line.
622
+ // `config` runs repoint the (process-global) asset root when the arm's
623
+ // overlay changes — work is arms-outer, so this fires at most once per arm.
624
+ let currentOverlay = overlayDir;
435
625
  for (let i = 0; i < work.length; i++) {
436
626
  const w = work[i];
627
+ if (runType === "config" && w.overlayDir !== currentOverlay) {
628
+ bootstrapAssets({ overlayDir: w.overlayDir });
629
+ currentOverlay = w.overlayDir;
630
+ }
437
631
  const s = makeSpinner();
438
632
  const head = `${chalk.dim(`[${i + 1}/${work.length}]`)} ${chalk.cyan(w.tierName)}/${w.inst.instance_id} ${chalk.dim(labels[w.model] ?? w.model)}`;
439
633
  s.start(head);
@@ -451,7 +645,7 @@ async function runEval() {
451
645
  all.push(result);
452
646
  if (result.error)
453
647
  harnessErrors++;
454
- const mark = result.error ? chalk.red("✗") : chalk.green("✓");
648
+ const mark = result.error ? chalk.red("✗") : result.blocked ? chalk.yellow("■") : chalk.green("✓");
455
649
  s.stop(`${chalk.dim(`[${i + 1}/${work.length}]`)} ${mark} ${verdictLine(w.tierName, w.inst, result)}`);
456
650
  if (result.error) {
457
651
  p.log.error(chalk.dim(result.error));
@@ -466,47 +660,73 @@ async function runEval() {
466
660
  finally {
467
661
  restoreEvalEnv();
468
662
  }
469
- // Final, static report + machine artifacts.
470
- const card = summarize(all);
471
- writeArtifacts(resultsDir, card);
472
- const html = writeHtml(resultsDir, card, { ...htmlBase, generatedAt: new Date().toISOString() });
473
- p.log.success(`Scorecard → ${chalk.cyan(html)}`);
474
- p.log.success(`Artifacts → ${chalk.cyan(resultsDir)}/{scorecard.json,predictions.jsonl}`);
663
+ // Final, static scorecard + machine artifacts. The run-level metadata is
664
+ // persisted into scorecard.json so the dashboard can label, order, and (no
665
+ // longer) live-poll the run without re-deriving from the current config.
666
+ const generatedAt = new Date().toISOString();
667
+ for (const tier of tiers) {
668
+ const tierResults = all.filter((r) => (r.tier ?? "") === tier);
669
+ writeArtifacts(resultsDirFor(tier), withMeta(summarize(tierResults), { ...baseMetaFor(tier), generatedAt, live: false }));
670
+ }
671
+ p.log.success(`Artifacts → ${chalk.cyan(tiers.map((t) => resultsDirFor(t)).join("\n "))}\n /{scorecard.json,predictions.jsonl,sessions/}`);
475
672
  const ran = runs > 1 ? `${completed} runs (${all.length} cases × ${runs})` : `${all.length} runs`;
476
- if (harnessErrors > 0) {
477
- p.outro(chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`));
673
+ // Keep the dashboard server alive so the just-finished run stays viewable.
674
+ // Only block in an interactive terminal — piped/non-TTY callers exit cleanly.
675
+ if (server && process.stdout.isTTY) {
676
+ const runUrl = `${server.url}/#/${encodeURIComponent(tierKeyFor(tiers[0]))}/${encodeURIComponent(runId)}`;
677
+ p.log.success(`Dashboard → ${chalk.cyan(runUrl)} ${chalk.dim("(serving · Ctrl-C to stop)")}`);
678
+ const tail = harnessErrors > 0 ? chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`) : chalk.green(`done — ${ran}`);
679
+ p.outro(tail);
680
+ await waitForSigint();
681
+ await server.close();
478
682
  }
479
683
  else {
480
- p.outro(chalk.green(`done — ${ran}, report at ${html}`));
684
+ if (server)
685
+ await server.close();
686
+ if (harnessErrors > 0) {
687
+ p.outro(chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`));
688
+ }
689
+ else {
690
+ p.outro(chalk.green(`done — ${ran}`));
691
+ }
481
692
  }
482
693
  // Non-zero ONLY on harness failure — model quality is the measurement.
483
694
  return harnessErrors > 0 ? 1 : 0;
484
695
  }
696
+ /** Resolve on the first SIGINT (Ctrl-C) so `run`/`serve` can keep a server up
697
+ * for browsing and shut it down cleanly when the user is done. */
698
+ function waitForSigint() {
699
+ return new Promise((resolve) => {
700
+ const onSig = () => {
701
+ process.off("SIGINT", onSig);
702
+ resolve();
703
+ };
704
+ process.on("SIGINT", onSig);
705
+ });
706
+ }
485
707
  /**
486
- * `report <dir>` — re-render `<dir>/index.html` from an existing
487
- * `<dir>/scorecard.json`, no models run. Lets you regenerate the HTML after a
488
- * report-template change (or to re-skin an old run) without paying for a fresh
489
- * eval. Models/tiers/labels are reconstructed from the scorecard itself.
708
+ * `serve` — start the dashboard server over `eval-results/` and open it in the
709
+ * browser to browse every past run (no models run). The same server `run` uses
710
+ * for the live report; blocks until Ctrl-C. There is no HTML to regenerate any
711
+ * more — the SPA reads the JSON directly.
490
712
  */
491
- function runReport(dir) {
492
- if (!dir) {
493
- console.error("usage: lastlight-evals report <results-dir> (a dir holding scorecard.json)");
494
- return 1;
713
+ async function runServe() {
714
+ loadDotEnv();
715
+ const noOpen = process.argv.includes("--no-open");
716
+ const port = intFlag("port", 0) || undefined;
717
+ let server;
718
+ try {
719
+ server = await startServer({ resultsRoot: resultsRoot(), dashboardRoot: dashboardDistRoot(), port });
495
720
  }
496
- const file = join(dir, "scorecard.json");
497
- if (!existsSync(file)) {
498
- console.error(`no scorecard.json in ${dir}`);
721
+ catch (err) {
722
+ console.error(`Couldn't start the dashboard server: ${err.message}`);
499
723
  return 1;
500
724
  }
501
- const card = JSON.parse(readFileSync(file, "utf8"));
502
- const labels = modelLabels();
503
- // Tiers/models come straight off the saved results so the re-render matches
504
- // exactly what that run measured (no dependence on current models.json).
505
- const tiers = [...new Set(card.results.map((r) => r.tier ?? "triage"))];
506
- const models = [...new Set(card.results.map((r) => r.model))].map((m) => labels[m] ?? m);
507
- const runs = Math.max(1, ...card.results.map((r) => r.trials ?? 1));
508
- const html = writeHtml(dir, card, { models, tiers, labels, runs, generatedAt: new Date().toISOString() });
509
- console.log(`Scorecard → ${html}`);
725
+ console.log(`Dashboard → ${server.url} (serving ${resultsRoot()} · Ctrl-C to stop)`);
726
+ if (!noOpen)
727
+ openInBrowser(server.url);
728
+ await waitForSigint();
729
+ await server.close();
510
730
  return 0;
511
731
  }
512
732
  const USAGE = `lastlight-evals — eval harness for Last Light workflows
@@ -514,21 +734,31 @@ const USAGE = `lastlight-evals — eval harness for Last Light workflows
514
734
  Usage:
515
735
  lastlight-evals [run] [tiers...] [options] Run evals (default command)
516
736
  lastlight-evals init [dir] [options] Scaffold an overlay+evals workspace
517
- lastlight-evals report <results-dir> Re-render index.html from scorecard.json
737
+ lastlight-evals serve [options] Browse past runs in the dashboard (no models run)
518
738
 
519
739
  Run options:
520
- --overlay <dir> Layer a deployment's workflows/skills + evals/ over built-ins
521
- --model <m[,m2]> Model(s) to run (fuzzy-matched against models.json)
740
+ --mode <models|config> Comparison axis. models (default): force each --model
741
+ across every step. config: run an overlay's real
742
+ per-step model config (its config.yaml). No flags in a
743
+ TTY ⇒ asks.
744
+ --overlay <dir> Layer a deployment's workflows/skills + evals/ over built-ins.
745
+ Repeatable in --mode config (one arm per overlay).
746
+ --model <m[,m2]> models: model(s) to run (fuzzy-matched against models.json).
747
+ config: override each config's default model.
522
748
  --compare Cross-vendor set (only models whose provider key is present)
523
749
  --runs <n> Repeat each case n× (worst-case verdict, mean metrics)
524
750
  --serial Force serial execution across provider families
525
751
  --datasets <dir> Extra datasets root to discover tiers from
526
752
  --models-file <f> Use an explicit models.json
527
- --no-open Don't open the HTML report (also implied by CI=1)
753
+ --no-open Don't open / auto-serve the dashboard (also implied by CI=1)
754
+
755
+ Serve options:
756
+ --port <n> Preferred port for the dashboard server (default 4319)
757
+ --no-open Start the server but don't open a browser
528
758
 
529
759
  Run \`lastlight-evals init --help\` for init-specific flags.
530
760
  GitHub is mocked end-to-end — no real GitHub token is needed, only a provider key.`;
531
- /** Top-level subcommand dispatcher: `run` (default) | `init` | `report`. */
761
+ /** Top-level subcommand dispatcher: `run` (default) | `init` | `serve`. */
532
762
  async function main() {
533
763
  const sub = process.argv[2];
534
764
  // Top-level help — only when it's not standing in for a `run` tier name.
@@ -540,9 +770,9 @@ async function main() {
540
770
  // `init [dir] [flags]` — scaffold a fresh overlay+evals repo.
541
771
  return runInit(process.argv.slice(3));
542
772
  }
543
- if (sub === "report") {
544
- // `report <dir>` — re-render index.html from a saved scorecard.json.
545
- return runReport(process.argv[3]);
773
+ if (sub === "serve") {
774
+ // `serve` — browse past runs; the live dashboard server, standalone.
775
+ return runServe();
546
776
  }
547
777
  // `run` is the default; allow an explicit leading `run` token too.
548
778
  if (sub === "run")