lastlight-evals 0.1.1 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/README.md +198 -39
  2. package/dashboard/dist/assets/index-YwScAAxm.js +238 -0
  3. package/dashboard/dist/assets/index-uU9Sj3M4.css +1 -0
  4. package/dashboard/dist/index.html +19 -0
  5. package/dashboard/dist/logo.png +0 -0
  6. package/datasets/code-fix/repos/codefix__date-range-off-by-one/.github/workflows/ci.yml +16 -0
  7. package/datasets/code-fix/repos/codefix__date-range-off-by-one/README.md +7 -1
  8. package/datasets/code-fix/repos/codefix__date-range-off-by-one/package.json +3 -1
  9. package/datasets/code-fix/repos/codefix__date-range-off-by-one/scripts/lint.mjs +25 -0
  10. package/datasets/code-fix/repos/codefix__date-range-off-by-one/test/date-range.test.ts +14 -0
  11. package/datasets/code-fix/repos/codefix__date-range-off-by-one/tsconfig.json +13 -0
  12. package/datasets/code-fix/tests/codefix__date-range-off-by-one/date-range.test.ts +5 -5
  13. package/dist/config.js +100 -0
  14. package/dist/config.js.map +1 -0
  15. package/dist/init.js +229 -32
  16. package/dist/init.js.map +1 -1
  17. package/dist/mechanism.test.js +41 -0
  18. package/dist/mechanism.test.js.map +1 -1
  19. package/dist/metrics.js +91 -3
  20. package/dist/metrics.js.map +1 -1
  21. package/dist/paths.js +51 -1
  22. package/dist/paths.js.map +1 -1
  23. package/dist/report.js +128 -6
  24. package/dist/report.js.map +1 -1
  25. package/dist/run-instance.js +143 -14
  26. package/dist/run-instance.js.map +1 -1
  27. package/dist/run.js +401 -89
  28. package/dist/run.js.map +1 -1
  29. package/dist/serve.js +136 -0
  30. package/dist/serve.js.map +1 -0
  31. package/examples/overlay/README.md +30 -0
  32. package/examples/overlay/config.yaml +36 -0
  33. package/examples/overlay-anthropic/config.yaml +27 -0
  34. package/package.json +16 -6
  35. package/dist/html-report.js +0 -325
  36. package/dist/html-report.js.map +0 -1
package/dist/run.js CHANGED
@@ -18,21 +18,39 @@
18
18
  * `evals/mechanism.test.ts` in the normal `npm test` suite.
19
19
  */
20
20
  import { existsSync } from "node:fs";
21
- import { join } from "node:path";
21
+ import { basename, join } from "node:path";
22
22
  import { spawn } from "node:child_process";
23
23
  import * as p from "@clack/prompts";
24
24
  import chalk from "chalk";
25
25
  import { loadDotEnv, hasProviderKey, evalModels, compareModels, modelLabels, resolveModel, setModelsPath } from "./env.js";
26
- import { runInstance, applyEvalEnv } from "./run-instance.js";
27
- import { summarize, writeArtifacts, aggregateTrials } from "./report.js";
28
- import { writeHtml } from "./html-report.js";
26
+ import { runInstance, applyEvalEnv, slug } from "./run-instance.js";
27
+ import { loadMergedConfig } from "./config.js";
28
+ import { summarize, writeArtifacts, writeScorecard, aggregateTrials, } from "./report.js";
29
29
  import { bootstrapAssets } from "./bootstrap.js";
30
30
  import { discoverTiers, loadInstances, workflowFor } from "./discovery.js";
31
- import { builtinDatasetsRoot, resultsRoot } from "./paths.js";
31
+ import { builtinDatasetsRoot, tierResultsDir, makeRunId, gitShortSha, resultsRoot, dashboardDistRoot } from "./paths.js";
32
+ import { startServer } from "./serve.js";
32
33
  import { runInit } from "./init.js";
33
- /** Open a file in the OS default browser (best-effort, never throws). */
34
- function openInBrowser(file) {
35
- const url = `file://${file}`;
34
+ /**
35
+ * A clack spinner in a TTY; in non-TTY (CI / piped / agent) a quiet stub that
36
+ * drops the animation frames — which redraw dozens of times and shred piped logs
37
+ * — and emits only the final `stop()` line. The plan note already prints what's
38
+ * about to run, so dropping the in-progress frames loses nothing in automation.
39
+ */
40
+ function makeSpinner() {
41
+ if (process.stdout.isTTY)
42
+ return p.spinner();
43
+ return {
44
+ start: () => { },
45
+ message: () => { },
46
+ stop: (msg) => {
47
+ if (msg)
48
+ p.log.message(msg);
49
+ },
50
+ };
51
+ }
52
+ /** Open a URL in the OS default browser (best-effort, never throws). */
53
+ function openInBrowser(url) {
36
54
  const [cmd, args] = process.platform === "darwin"
37
55
  ? ["open", [url]]
38
56
  : process.platform === "win32"
@@ -81,6 +99,12 @@ function fmtMs(ms) {
81
99
  function familyLabel(envKey) {
82
100
  return envKey.replace(/_API_KEY$/i, "").toLowerCase() || "default";
83
101
  }
102
+ /** Attach run metadata to a scorecard (mutate-and-return; `summarize` hands back
103
+ * a fresh object each call, so this is safe). */
104
+ function withMeta(card, meta) {
105
+ card.meta = meta;
106
+ return card;
107
+ }
84
108
  /**
85
109
  * Silence `console.*` for the whole batch (parallel mode). The per-run
86
110
  * `quiet()` swap saves/restores console and would corrupt under concurrent
@@ -98,6 +122,8 @@ function verdictLine(tierName, inst, r) {
98
122
  const head = `${chalk.cyan(tierName)}/${inst.instance_id}`;
99
123
  if (r.error)
100
124
  return `${head} ${chalk.red("harness error")}`;
125
+ if (r.blocked)
126
+ return `${head} ${chalk.yellow("blocked")} ${chalk.dim("(workflow gate)")}`;
101
127
  const count = (pass) => (pass !== undefined && r.trials ? chalk.dim(` ${pass}/${r.trials}`) : "");
102
128
  const parts = [];
103
129
  if (r.resolved !== undefined)
@@ -136,24 +162,104 @@ function strFlag(name) {
136
162
  }
137
163
  return process.env[`EVAL_${name.toUpperCase()}`];
138
164
  }
165
+ /** Collect ALL occurrences of a repeatable string flag (`--name a --name b` or
166
+ * `--name=a`). Used by `--overlay`, which may repeat in `config` mode to compare
167
+ * one arm per overlay. */
168
+ function strFlagAll(name) {
169
+ const out = [];
170
+ const argv = process.argv;
171
+ for (let i = 0; i < argv.length; i++) {
172
+ if (argv[i] === `--${name}` && argv[i + 1] !== undefined) {
173
+ out.push(argv[i + 1]);
174
+ i++;
175
+ continue;
176
+ }
177
+ const m = argv[i].match(new RegExp(`^--${name}=(.+)$`));
178
+ if (m)
179
+ out.push(m[1]);
180
+ }
181
+ return out;
182
+ }
139
183
  /** CLI flags that take a following value (so it isn't read as a tier name). */
140
- const VALUE_FLAGS = new Set(["--runs", "--model", "--models", "--overlay", "--datasets", "--models-file"]);
184
+ const VALUE_FLAGS = new Set(["--runs", "--model", "--models", "--mode", "--overlay", "--datasets", "--models-file"]);
141
185
  async function runEval() {
142
186
  loadDotEnv();
143
187
  p.intro(chalk.bold(`Last Light ${chalk.yellow("·")} eval`));
188
+ // Run type — the comparison axis:
189
+ // `models` — compare N models, each FORCED across every workflow step.
190
+ // `config` — run an overlay's REAL per-step model config (what ships); the
191
+ // arm is the config/overlay, compared across runs (or N overlays
192
+ // side-by-side). `--mode` wins; an explicit `--model`/`--compare`
193
+ // implies `models`; otherwise ask in a TTY (default `models`).
194
+ const modeFlag = strFlag("mode");
195
+ const modelArg = strFlag("model") ?? strFlag("models");
196
+ const compare = process.argv.includes("--compare");
197
+ let runType;
198
+ if (modeFlag === "config")
199
+ runType = "config";
200
+ else if (modeFlag === "models" || modelArg || compare)
201
+ runType = "models";
202
+ else if (process.stdin.isTTY) {
203
+ const picked = await p.select({
204
+ message: "What do you want to eval?",
205
+ options: [
206
+ { value: "models", label: "compare models", hint: "force each model across every workflow step" },
207
+ { value: "config", label: "eval config", hint: "an overlay's real per-step model config (what ships)" },
208
+ ],
209
+ initialValue: "models",
210
+ });
211
+ if (p.isCancel(picked)) {
212
+ p.cancel("aborted");
213
+ return 1;
214
+ }
215
+ runType = picked;
216
+ }
217
+ else
218
+ runType = "models";
144
219
  // Asset roots FIRST — before any getWorkflow/runWorkflow. `--overlay` (or
145
220
  // LASTLIGHT_OVERLAY_DIR) layers a deployment's own workflows/skills over the
146
221
  // built-ins, and also contributes its `evals/datasets/` (see discovery).
147
- const overlayDir = strFlag("overlay") ?? process.env.LASTLIGHT_OVERLAY_DIR;
148
- bootstrapAssets({ overlayDir });
222
+ // With neither set, auto-detect a local `./instance/` overlay checkout — the
223
+ // Separate layout `init --clone` produces — so a bare run "just works".
224
+ // `--overlay` may REPEAT in config mode (one arm per overlay).
225
+ const autoInstance = join(process.cwd(), "instance");
226
+ const autoOverlay = existsSync(join(autoInstance, "config.yaml")) ? autoInstance : undefined;
227
+ const overlayFlags = strFlagAll("overlay");
228
+ let overlays = overlayFlags.length
229
+ ? overlayFlags
230
+ : [process.env.LASTLIGHT_OVERLAY_DIR ?? autoOverlay].filter((d) => !!d);
231
+ // In a config-mode TTY with no overlay given, ask for one (blank ⇒ core defaults).
232
+ if (runType === "config" && !overlayFlags.length && process.stdin.isTTY) {
233
+ const ans = await p.text({
234
+ message: "Overlay dir whose config.yaml drives per-step models (blank = core defaults):",
235
+ placeholder: overlays[0] ?? autoInstance,
236
+ initialValue: overlays[0] ?? "",
237
+ });
238
+ if (p.isCancel(ans)) {
239
+ p.cancel("aborted");
240
+ return 1;
241
+ }
242
+ const dir = ans.trim();
243
+ overlays = dir ? [dir] : [];
244
+ }
245
+ // The primary overlay wires discovery + the initial asset bootstrap. Config
246
+ // arms re-bootstrap their own overlay before running (the asset root is a
247
+ // process global — see the serial loop below).
248
+ const overlayDir = overlays[0];
249
+ if (overlayDir && overlayDir === autoOverlay)
250
+ p.log.info(`overlay → ${chalk.cyan("./instance")} ${chalk.dim("(auto-detected)")}`);
251
+ const { builtInRoot } = bootstrapAssets({ overlayDir });
149
252
  // A user/overlay can ship its own model registry too: explicit --models-file
150
253
  // wins, else an overlay's `evals/models.json` if present, else the built-in.
151
254
  const overlayModels = overlayDir ? join(overlayDir, "evals", "models.json") : undefined;
152
255
  const modelsFile = strFlag("models-file") ?? (overlayModels && existsSync(overlayModels) ? overlayModels : undefined);
153
256
  if (modelsFile)
154
257
  setModelsPath(modelsFile);
155
- // Discover tiers across built-in + user (--datasets) + overlay roots.
156
- const userDatasetsDir = strFlag("datasets") ?? process.env.LASTLIGHT_EVALS_DATASETS;
258
+ // Discover tiers across built-in + user (--datasets) + overlay roots. With no
259
+ // explicit `--datasets`, default to the workspace's own `./evals/datasets`
260
+ // (what `init` seeds) so editing/adding tiers there is picked up automatically.
261
+ const autoDatasets = join(process.cwd(), "evals", "datasets");
262
+ const userDatasetsDir = strFlag("datasets") ?? process.env.LASTLIGHT_EVALS_DATASETS ?? (existsSync(autoDatasets) ? autoDatasets : undefined);
157
263
  const discovered = discoverTiers({
158
264
  builtinRoot: builtinDatasetsRoot(),
159
265
  userDatasetsDir,
@@ -165,7 +271,6 @@ async function runEval() {
165
271
  p.outro(chalk.red("aborted"));
166
272
  return 1;
167
273
  }
168
- const compare = process.argv.includes("--compare");
169
274
  const noOpen = process.argv.includes("--no-open") || !!process.env.CI;
170
275
  const runs = intFlag("runs", 1);
171
276
  // Positional tier names — skip flags AND the values that follow value-flags.
@@ -221,35 +326,51 @@ async function runEval() {
221
326
  if (!discovered.has(t))
222
327
  p.log.warn(`Unknown tier "${t}". Known: ${known.join(", ")}`);
223
328
  }
224
- // Model selection precedence:
225
- // 1. --model / --models (or EVAL_MODEL[S]) — an explicit list, fuzzy-matched
226
- // against models.json; lets you test one model quickly.
227
- // 2. --compare — the full cross-vendor set (key-gated).
228
- // 3. default single model from models.json.
229
- // Each entry carries its provider family (env-key) for parallel grouping.
230
- const modelArg = strFlag("model") ?? strFlag("models");
231
- const mode = modelArg ? "select" : compare ? "compare" : "single";
232
- const entries = modelArg
233
- ? modelArg
234
- .split(",")
235
- .map((s) => s.trim())
236
- .filter(Boolean)
237
- .map((tok) => {
238
- const r = resolveModel(tok);
239
- return { id: r.id, family: r.family };
240
- })
241
- : compare
242
- ? compareModels().map((m) => ({ id: m.id, family: m.envKey ?? m.provider ?? "default" }))
243
- : evalModels().map((id) => ({ id, family: "default" }));
244
- if (!entries.length) {
329
+ // The model-selection sub-mode (only meaningful for `models` runs); shown in
330
+ // the plan note. `config` runs report their arm count instead.
331
+ const mode = runType === "config" ? "config" : modelArg ? "select" : compare ? "compare" : "single";
332
+ let arms;
333
+ if (runType === "config") {
334
+ // One arm per overlay (or a single core-defaults arm when none). `--model`
335
+ // overrides each merged config's `default` key for quick what-if runs.
336
+ const configOverlays = overlays.length ? overlays : [overlayDir];
337
+ arms = configOverlays.map((dir) => {
338
+ const merged = loadMergedConfig(builtInRoot, dir);
339
+ if (modelArg)
340
+ merged.models.default = resolveModel(modelArg).id;
341
+ const label = dir ? basename(dir) : "config";
342
+ return { label, family: label, modelConfig: merged.models, variantConfig: merged.variants, overlayDir: dir };
343
+ });
344
+ }
345
+ else {
346
+ // Model selection precedence:
347
+ // 1. --model / --models (or EVAL_MODEL[S]) — an explicit list, fuzzy-matched.
348
+ // 2. --compare — the full cross-vendor set (key-gated).
349
+ // 3. default single model from models.json.
350
+ const entries = modelArg
351
+ ? modelArg
352
+ .split(",")
353
+ .map((s) => s.trim())
354
+ .filter(Boolean)
355
+ .map((tok) => {
356
+ const r = resolveModel(tok);
357
+ return { id: r.id, family: r.family };
358
+ })
359
+ : compare
360
+ ? compareModels().map((m) => ({ id: m.id, family: m.envKey ?? m.provider ?? "default" }))
361
+ : evalModels().map((id) => ({ id, family: "default" }));
362
+ arms = entries.map((e) => ({ label: e.id, family: e.family }));
363
+ }
364
+ if (!arms.length) {
245
365
  p.log.error("No comparison models available — set provider keys (OPENAI_API_KEY / ANTHROPIC_API_KEY /\n" +
246
366
  "FIREWORKS_API_KEY …) for the entries in evals/models.json.");
247
367
  p.outro(chalk.red("aborted"));
248
368
  return 1;
249
369
  }
250
370
  const labels = modelLabels();
251
- // Resolve the work-list up front so we can show deterministic progress.
252
- const work = [];
371
+ // Instances per tier, resolved ONCE — the case set is identical across arms
372
+ // (arms vary only the model selection / assets, never the cases).
373
+ const tierInstances = new Map();
253
374
  for (const tierName of tiers) {
254
375
  const tier = discovered.get(tierName);
255
376
  const instances = loadInstances(tier);
@@ -257,7 +378,15 @@ async function runEval() {
257
378
  p.log.warn(`tier "${tierName}": no instances at ${tier.instancesPath} — skipping`);
258
379
  continue;
259
380
  }
260
- for (const e of entries) {
381
+ tierInstances.set(tierName, instances);
382
+ }
383
+ // Resolve the work-list up front so we can show deterministic progress. Arms
384
+ // are the OUTER loop so a `config` run's per-arm overlay switches at most once
385
+ // (the serial loop re-bootstraps on arm change; see below).
386
+ const work = [];
387
+ for (const arm of arms) {
388
+ for (const [tierName, instances] of tierInstances) {
389
+ const tier = discovered.get(tierName);
261
390
  for (const inst of instances) {
262
391
  // Per-instance workflow wins, else the tier's defaultWorkflow (throws if
263
392
  // neither is set — surfaced as a harness error for that case).
@@ -265,8 +394,11 @@ async function runEval() {
265
394
  tierName,
266
395
  defaultWorkflow: workflowFor(tier, inst),
267
396
  datasetDir: tier.root,
268
- model: e.id,
269
- family: e.family,
397
+ model: arm.label,
398
+ family: arm.family,
399
+ modelConfig: arm.modelConfig,
400
+ variantConfig: arm.variantConfig,
401
+ overlayDir: arm.overlayDir,
270
402
  inst,
271
403
  });
272
404
  }
@@ -288,28 +420,78 @@ async function runEval() {
288
420
  else
289
421
  byFamily.set(w.family, [w]);
290
422
  }
291
- const parallel = !process.argv.includes("--serial") && byFamily.size > 1;
292
- const resultsDir = join(resultsRoot(), `${tiers.join("+")}${compare ? "-compare" : ""}`);
293
- const htmlBase = {
294
- models: entries.map((e) => labels[e.id] ?? e.id),
295
- tiers,
296
- labels,
423
+ // `config` runs stay SERIAL: arms may carry distinct overlays and the asset
424
+ // root is a process global, so the serial loop re-bootstraps per arm (below).
425
+ const parallel = runType === "models" && !process.argv.includes("--serial") && byFamily.size > 1;
426
+ // Each tier writes its OWN folder + scorecard (all sharing this run's id), so
427
+ // tiers stay separate in the dashboard instead of collapsing into one combined
428
+ // `<a+b>` entry. A single invocation appears as the same run under each tier it
429
+ // touched. The `-compare` suffix keeps cross-vendor runs on their own trend
430
+ // line, distinct from single-model runs of the same tier. The dashboard server
431
+ // indexes the whole tree; each tier dir is a `tierKey` segment the SPA routes on.
432
+ // `-compare` keeps cross-vendor runs on their own trend line; `-config` keeps
433
+ // config-eval runs on theirs — so the three run shapes never collapse together.
434
+ const gitSha = gitShortSha();
435
+ const tierKeyFor = (tier) => `${tier}${runType === "config" ? "-config" : compare ? "-compare" : ""}`;
436
+ // One shared runId, checked free in the first tier's dir (collisions in the
437
+ // same second across runs are what the suffix guards — rare, one dir suffices).
438
+ const runId = makeRunId(new Date(), gitSha, tierResultsDir(tierKeyFor(tiers[0])));
439
+ const resultsDirFor = (tier) => join(tierResultsDir(tierKeyFor(tier)), runId);
440
+ // Each case's session logs live under `sessions/<id>__<model>/trial-<N>/`
441
+ // (relative to the tier run dir) — the model is in the name since several
442
+ // models share a run dir; per-trial keeps every `--runs N` trial. `full.jsonl`
443
+ // is the consolidated live-followable transcript; `NN-<phase>.jsonl` are the
444
+ // per-phase splits written when the trial finishes.
445
+ const caseRelFor = (instanceId, model) => `sessions/${slug(instanceId)}__${slug(model)}`;
446
+ const trialRelFor = (instanceId, model, trial) => `${caseRelFor(instanceId, model)}/trial-${trial}`;
447
+ // Per-tier run metadata stamped into every scorecard write (the dashboard reads
448
+ // identity, labels, and live state straight off disk). `live`/`progress`/
449
+ // `pending`/`generatedAt` are layered on per write; `tiers` is the single tier.
450
+ const armLabels = arms.map((a) => labels[a.label] ?? a.label);
451
+ const baseMetaFor = (tier) => ({
452
+ runId,
453
+ runType,
454
+ tiers: [tier],
455
+ models: armLabels,
297
456
  runs,
298
- };
457
+ gitSha,
458
+ labels,
459
+ });
460
+ // In `config` runs the axis is the config(s); show the merged per-step model
461
+ // map for a single-arm run so the plan is legible.
462
+ const axisLine = runType === "config"
463
+ ? `${chalk.bold("configs")} ${armLabels.join(", ")}`
464
+ : `${chalk.bold("models")} ${armLabels.join(", ")}`;
465
+ const phaseMapLine = runType === "config" && arms.length === 1 && arms[0].modelConfig
466
+ ? `\n${chalk.bold("models")} ${chalk.dim(Object.entries(arms[0].modelConfig)
467
+ .map(([k, v]) => `${k}→${v}`)
468
+ .join(" "))}`
469
+ : "";
299
470
  p.note(`${chalk.bold("mode")} ${mode}${parallel ? chalk.dim(` (parallel · ${byFamily.size} families)`) : ""}\n` +
300
- `${chalk.bold("models")} ${entries.map((e) => labels[e.id] ?? e.id).join(", ")}\n` +
471
+ `${axisLine}${phaseMapLine}\n` +
301
472
  `${chalk.bold("tiers")} ${tiers.join(", ")}\n` +
302
473
  `${chalk.bold("cases")} ${work.length}${runs > 1
303
474
  ? chalk.dim(` × ${runs} trials = ${work.length * runs} runs · worst-case verdict, mean cost`)
304
475
  : ""}`, "plan");
305
476
  // `total` counts individual trials so live progress advances per model call.
306
477
  const total = work.length * runs;
307
- // Open the report immediately (live placeholder) so it fills in as we go.
308
- writeHtml(resultsDir, summarize([]), { ...htmlBase, generatedAt: new Date().toISOString(), live: true, progress: `0/${total}` });
309
- const htmlFile = join(resultsDir, "index.html");
478
+ // Seed an empty live scorecard per tier so the dashboard has something to poll,
479
+ // then start the server and open the SPA deep-linked at this run. The server is
480
+ // skipped entirely when not opening (CI / --no-open) — we only write JSON.
481
+ for (const tier of tiers) {
482
+ writeScorecard(resultsDirFor(tier), withMeta(summarize([]), { ...baseMetaFor(tier), generatedAt: new Date().toISOString(), live: true, progress: `0/${total}` }));
483
+ }
484
+ let server;
310
485
  if (!noOpen) {
311
- openInBrowser(htmlFile);
312
- p.log.info(`Live report → ${chalk.cyan(htmlFile)} ${chalk.dim("(auto-refreshing)")}`);
486
+ try {
487
+ server = await startServer({ resultsRoot: resultsRoot(), dashboardRoot: dashboardDistRoot() });
488
+ const runUrl = `${server.url}/#/${encodeURIComponent(tierKeyFor(tiers[0]))}/${encodeURIComponent(runId)}`;
489
+ openInBrowser(runUrl);
490
+ p.log.info(`Live dashboard → ${chalk.cyan(runUrl)}`);
491
+ }
492
+ catch (err) {
493
+ p.log.warn(`Couldn't start the dashboard server (${err.message}) — writing JSON only.`);
494
+ }
313
495
  }
314
496
  const all = [];
315
497
  let harnessErrors = 0;
@@ -317,44 +499,76 @@ async function runEval() {
317
499
  // Track in-flight cases so the live report can show running / queued rows.
318
500
  const caseKey = (tier, model, id) => `${tier}|${model}|${id}`;
319
501
  const running = new Set();
320
- // writeHtml/summarize/all.push run synchronously to completion inside one
321
- // event-loop turn, so even with concurrent families they never interleave.
502
+ // Current trial number per running case, so the live "follow" link points at
503
+ // the right `trial-<N>/full.jsonl` (matters only for `--runs N>1`).
504
+ const trialOf = new Map();
505
+ // writeScorecard/summarize/all.push run synchronously to completion inside one
506
+ // event-loop turn, so even with concurrent families they never interleave; the
507
+ // temp-file+rename keeps a polling dashboard from reading a half-written file.
322
508
  const refresh = () => {
323
509
  const done = new Set(all.map((r) => caseKey(r.tier ?? "", r.model, r.instance_id)));
324
- const pending = work
325
- .map((w) => ({ w, k: caseKey(w.tierName, w.model, w.inst.instance_id) }))
326
- .filter(({ k }) => !done.has(k))
327
- .map(({ w, k }) => ({
328
- tier: w.tierName,
329
- model: w.model,
330
- instance_id: w.inst.instance_id,
331
- status: (running.has(k) ? "running" : "pending"),
332
- }));
333
- writeHtml(resultsDir, summarize(all), {
334
- ...htmlBase,
335
- generatedAt: new Date().toISOString(),
336
- live: true,
337
- progress: `${completed}/${total}`,
338
- pending,
339
- });
510
+ const now = new Date().toISOString();
511
+ for (const tier of tiers) {
512
+ const tierResults = all.filter((r) => (r.tier ?? "") === tier);
513
+ const pending = work
514
+ .filter((w) => w.tierName === tier)
515
+ .map((w) => ({ w, k: caseKey(w.tierName, w.model, w.inst.instance_id) }))
516
+ .filter(({ k }) => !done.has(k))
517
+ .map(({ w, k }) => ({
518
+ tier: w.tierName,
519
+ model: w.model,
520
+ instance_id: w.inst.instance_id,
521
+ status: running.has(k) ? "running" : "pending",
522
+ // Only a running case has a (live-updating) transcript to follow —
523
+ // point at the current trial's consolidated `full.jsonl`.
524
+ sessionLog: running.has(k)
525
+ ? `${trialRelFor(w.inst.instance_id, w.model, trialOf.get(k) ?? 1)}/full.jsonl`
526
+ : undefined,
527
+ }));
528
+ // Per-tier progress (cases), not the global trial count — each tier's
529
+ // scorecard stands alone, so "0/5" across both tiers was misleading.
530
+ const tierCases = work.filter((w) => w.tierName === tier).length;
531
+ writeScorecard(resultsDirFor(tier), withMeta(summarize(tierResults), {
532
+ ...baseMetaFor(tier),
533
+ generatedAt: now,
534
+ live: true,
535
+ progress: `${tierResults.length}/${tierCases}`,
536
+ pending,
537
+ }));
538
+ }
340
539
  };
341
540
  // Run one case `runs` times and fold the trials into a single result
342
541
  // (worst-case verdict, mean metrics). `onTrial` ticks per model call.
343
542
  const runItem = async (w, onTrial) => {
543
+ const k = caseKey(w.tierName, w.model, w.inst.instance_id);
344
544
  const trials = [];
345
- for (let t = 0; t < runs; t++) {
545
+ for (let t = 1; t <= runs; t++) {
546
+ trialOf.set(k, t); // so the live "follow" link targets this trial
547
+ const trialRel = trialRelFor(w.inst.instance_id, w.model, t);
346
548
  const r = await runInstance(w.inst, {
347
549
  model: w.model,
550
+ // `config` arms carry the merged per-step maps; `models` arms leave
551
+ // these undefined so the workflow forces `w.model` on every step.
552
+ modelConfig: w.modelConfig,
553
+ variantConfig: w.variantConfig,
348
554
  datasetDir: w.datasetDir,
349
555
  defaultWorkflow: w.defaultWorkflow,
350
556
  manageEnv: false,
557
+ // Per-trial dir: `full.jsonl` (consolidated, live) + `NN-<phase>.jsonl`.
558
+ sessionTrialDir: join(resultsDirFor(w.tierName), trialRel),
559
+ sessionTrialRel: trialRel,
560
+ trial: t,
351
561
  });
352
562
  r.tier = w.tierName;
353
563
  trials.push(r);
354
564
  completed++;
355
565
  onTrial();
356
566
  }
357
- return aggregateTrials(trials);
567
+ const agg = aggregateTrials(trials);
568
+ // Keep every trial's per-phase sessions on the aggregate (aggregateTrials
569
+ // only carries trial 0's fields through).
570
+ agg.sessions = trials.map((r) => r.sessionTrial).filter((s) => !!s);
571
+ return agg;
358
572
  };
359
573
  // Install the eval's static-token env ONCE for the whole batch so concurrent
360
574
  // runs share one stable baseline (manageEnv:false on every runInstance).
@@ -372,7 +586,7 @@ async function runEval() {
372
586
  });
373
587
  return `${chalk.dim(`${completed}/${total}`)} ${segs.join(chalk.dim(" · "))}`;
374
588
  };
375
- const s = p.spinner();
589
+ const s = makeSpinner();
376
590
  s.start(status());
377
591
  const restoreConsole = silenceConsole();
378
592
  const verdicts = [];
@@ -391,7 +605,7 @@ async function runEval() {
391
605
  all.push(result);
392
606
  if (result.error)
393
607
  harnessErrors++;
394
- const mark = result.error ? chalk.red("✗") : chalk.green("✓");
608
+ const mark = result.error ? chalk.red("✗") : result.blocked ? chalk.yellow("■") : chalk.green("✓");
395
609
  verdicts.push(`${mark} ${chalk.dim(familyLabel(f))} ${verdictLine(w.tierName, w.inst, result)}`);
396
610
  refresh();
397
611
  }
@@ -405,9 +619,16 @@ async function runEval() {
405
619
  }
406
620
  else {
407
621
  // Serial: one spinner per case (updates per trial) + a verdict line.
622
+ // `config` runs repoint the (process-global) asset root when the arm's
623
+ // overlay changes — work is arms-outer, so this fires at most once per arm.
624
+ let currentOverlay = overlayDir;
408
625
  for (let i = 0; i < work.length; i++) {
409
626
  const w = work[i];
410
- const s = p.spinner();
627
+ if (runType === "config" && w.overlayDir !== currentOverlay) {
628
+ bootstrapAssets({ overlayDir: w.overlayDir });
629
+ currentOverlay = w.overlayDir;
630
+ }
631
+ const s = makeSpinner();
411
632
  const head = `${chalk.dim(`[${i + 1}/${work.length}]`)} ${chalk.cyan(w.tierName)}/${w.inst.instance_id} ${chalk.dim(labels[w.model] ?? w.model)}`;
412
633
  s.start(head);
413
634
  const k = caseKey(w.tierName, w.model, w.inst.instance_id);
@@ -424,7 +645,7 @@ async function runEval() {
424
645
  all.push(result);
425
646
  if (result.error)
426
647
  harnessErrors++;
427
- const mark = result.error ? chalk.red("✗") : chalk.green("✓");
648
+ const mark = result.error ? chalk.red("✗") : result.blocked ? chalk.yellow("■") : chalk.green("✓");
428
649
  s.stop(`${chalk.dim(`[${i + 1}/${work.length}]`)} ${mark} ${verdictLine(w.tierName, w.inst, result)}`);
429
650
  if (result.error) {
430
651
  p.log.error(chalk.dim(result.error));
@@ -439,28 +660,119 @@ async function runEval() {
439
660
  finally {
440
661
  restoreEvalEnv();
441
662
  }
442
- // Final, static report + machine artifacts.
443
- const card = summarize(all);
444
- writeArtifacts(resultsDir, card);
445
- const html = writeHtml(resultsDir, card, { ...htmlBase, generatedAt: new Date().toISOString() });
446
- p.log.success(`Scorecard → ${chalk.cyan(html)}`);
447
- p.log.success(`Artifacts → ${chalk.cyan(resultsDir)}/{scorecard.json,predictions.jsonl}`);
663
+ // Final, static scorecard + machine artifacts. The run-level metadata is
664
+ // persisted into scorecard.json so the dashboard can label, order, and (no
665
+ // longer) live-poll the run without re-deriving from the current config.
666
+ const generatedAt = new Date().toISOString();
667
+ for (const tier of tiers) {
668
+ const tierResults = all.filter((r) => (r.tier ?? "") === tier);
669
+ writeArtifacts(resultsDirFor(tier), withMeta(summarize(tierResults), { ...baseMetaFor(tier), generatedAt, live: false }));
670
+ }
671
+ p.log.success(`Artifacts → ${chalk.cyan(tiers.map((t) => resultsDirFor(t)).join("\n "))}\n /{scorecard.json,predictions.jsonl,sessions/}`);
448
672
  const ran = runs > 1 ? `${completed} runs (${all.length} cases × ${runs})` : `${all.length} runs`;
449
- if (harnessErrors > 0) {
450
- p.outro(chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`));
673
+ // Keep the dashboard server alive so the just-finished run stays viewable.
674
+ // Only block in an interactive terminal — piped/non-TTY callers exit cleanly.
675
+ if (server && process.stdout.isTTY) {
676
+ const runUrl = `${server.url}/#/${encodeURIComponent(tierKeyFor(tiers[0]))}/${encodeURIComponent(runId)}`;
677
+ p.log.success(`Dashboard → ${chalk.cyan(runUrl)} ${chalk.dim("(serving · Ctrl-C to stop)")}`);
678
+ const tail = harnessErrors > 0 ? chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`) : chalk.green(`done — ${ran}`);
679
+ p.outro(tail);
680
+ await waitForSigint();
681
+ await server.close();
451
682
  }
452
683
  else {
453
- p.outro(chalk.green(`done — ${ran}, report at ${html}`));
684
+ if (server)
685
+ await server.close();
686
+ if (harnessErrors > 0) {
687
+ p.outro(chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`));
688
+ }
689
+ else {
690
+ p.outro(chalk.green(`done — ${ran}`));
691
+ }
454
692
  }
455
693
  // Non-zero ONLY on harness failure — model quality is the measurement.
456
694
  return harnessErrors > 0 ? 1 : 0;
457
695
  }
458
- /** Top-level subcommand dispatcher: `run` (default) | `init`. */
696
+ /** Resolve on the first SIGINT (Ctrl-C) so `run`/`serve` can keep a server up
697
+ * for browsing and shut it down cleanly when the user is done. */
698
+ function waitForSigint() {
699
+ return new Promise((resolve) => {
700
+ const onSig = () => {
701
+ process.off("SIGINT", onSig);
702
+ resolve();
703
+ };
704
+ process.on("SIGINT", onSig);
705
+ });
706
+ }
707
+ /**
708
+ * `serve` — start the dashboard server over `eval-results/` and open it in the
709
+ * browser to browse every past run (no models run). The same server `run` uses
710
+ * for the live report; blocks until Ctrl-C. There is no HTML to regenerate any
711
+ * more — the SPA reads the JSON directly.
712
+ */
713
+ async function runServe() {
714
+ loadDotEnv();
715
+ const noOpen = process.argv.includes("--no-open");
716
+ const port = intFlag("port", 0) || undefined;
717
+ let server;
718
+ try {
719
+ server = await startServer({ resultsRoot: resultsRoot(), dashboardRoot: dashboardDistRoot(), port });
720
+ }
721
+ catch (err) {
722
+ console.error(`Couldn't start the dashboard server: ${err.message}`);
723
+ return 1;
724
+ }
725
+ console.log(`Dashboard → ${server.url} (serving ${resultsRoot()} · Ctrl-C to stop)`);
726
+ if (!noOpen)
727
+ openInBrowser(server.url);
728
+ await waitForSigint();
729
+ await server.close();
730
+ return 0;
731
+ }
732
+ const USAGE = `lastlight-evals — eval harness for Last Light workflows
733
+
734
+ Usage:
735
+ lastlight-evals [run] [tiers...] [options] Run evals (default command)
736
+ lastlight-evals init [dir] [options] Scaffold an overlay+evals workspace
737
+ lastlight-evals serve [options] Browse past runs in the dashboard (no models run)
738
+
739
+ Run options:
740
+ --mode <models|config> Comparison axis. models (default): force each --model
741
+ across every step. config: run an overlay's real
742
+ per-step model config (its config.yaml). No flags in a
743
+ TTY ⇒ asks.
744
+ --overlay <dir> Layer a deployment's workflows/skills + evals/ over built-ins.
745
+ Repeatable in --mode config (one arm per overlay).
746
+ --model <m[,m2]> models: model(s) to run (fuzzy-matched against models.json).
747
+ config: override each config's default model.
748
+ --compare Cross-vendor set (only models whose provider key is present)
749
+ --runs <n> Repeat each case n× (worst-case verdict, mean metrics)
750
+ --serial Force serial execution across provider families
751
+ --datasets <dir> Extra datasets root to discover tiers from
752
+ --models-file <f> Use an explicit models.json
753
+ --no-open Don't open / auto-serve the dashboard (also implied by CI=1)
754
+
755
+ Serve options:
756
+ --port <n> Preferred port for the dashboard server (default 4319)
757
+ --no-open Start the server but don't open a browser
758
+
759
+ Run \`lastlight-evals init --help\` for init-specific flags.
760
+ GitHub is mocked end-to-end — no real GitHub token is needed, only a provider key.`;
761
+ /** Top-level subcommand dispatcher: `run` (default) | `init` | `serve`. */
459
762
  async function main() {
460
763
  const sub = process.argv[2];
764
+ // Top-level help — only when it's not standing in for a `run` tier name.
765
+ if (sub === "help" || sub === "--help" || sub === "-h") {
766
+ console.log(USAGE);
767
+ return 0;
768
+ }
461
769
  if (sub === "init") {
462
- // `init [dir]` — scaffold a fresh overlay+evals repo.
463
- return runInit(process.argv[3]);
770
+ // `init [dir] [flags]` — scaffold a fresh overlay+evals repo.
771
+ return runInit(process.argv.slice(3));
772
+ }
773
+ if (sub === "serve") {
774
+ // `serve` — browse past runs; the live dashboard server, standalone.
775
+ return runServe();
464
776
  }
465
777
  // `run` is the default; allow an explicit leading `run` token too.
466
778
  if (sub === "run")