lastlight-evals 0.1.2 → 0.2.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +98 -12
- package/dashboard/dist/assets/index-YwScAAxm.js +238 -0
- package/dashboard/dist/assets/index-uU9Sj3M4.css +1 -0
- package/dashboard/dist/index.html +19 -0
- package/dashboard/dist/logo.png +0 -0
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/.github/workflows/ci.yml +16 -0
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/README.md +7 -1
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/package.json +3 -1
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/scripts/lint.mjs +25 -0
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/test/date-range.test.ts +14 -0
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/tsconfig.json +13 -0
- package/datasets/code-fix/tests/codefix__date-range-off-by-one/date-range.test.ts +5 -5
- package/dist/arm.js +132 -0
- package/dist/arm.js.map +1 -0
- package/dist/config.js +100 -0
- package/dist/config.js.map +1 -0
- package/dist/mechanism.test.js +156 -0
- package/dist/mechanism.test.js.map +1 -1
- package/dist/metrics.js +83 -0
- package/dist/metrics.js.map +1 -1
- package/dist/paths.js +51 -1
- package/dist/paths.js.map +1 -1
- package/dist/report.js +112 -3
- package/dist/report.js.map +1 -1
- package/dist/run-instance.js +136 -15
- package/dist/run-instance.js.map +1 -1
- package/dist/run.js +350 -118
- package/dist/run.js.map +1 -1
- package/dist/serve.js +136 -0
- package/dist/serve.js.map +1 -0
- package/examples/overlay/README.md +30 -0
- package/examples/overlay/config.yaml +36 -0
- package/examples/overlay-anthropic/config.yaml +27 -0
- package/package.json +16 -6
- package/dist/html-report.js +0 -325
- package/dist/html-report.js.map +0 -1
package/dist/run.js
CHANGED
|
@@ -17,18 +17,19 @@
|
|
|
17
17
|
* The deterministic, AI-free plumbing is covered separately by
|
|
18
18
|
* `evals/mechanism.test.ts` in the normal `npm test` suite.
|
|
19
19
|
*/
|
|
20
|
-
import { existsSync
|
|
20
|
+
import { existsSync } from "node:fs";
|
|
21
21
|
import { join } from "node:path";
|
|
22
22
|
import { spawn } from "node:child_process";
|
|
23
23
|
import * as p from "@clack/prompts";
|
|
24
24
|
import chalk from "chalk";
|
|
25
25
|
import { loadDotEnv, hasProviderKey, evalModels, compareModels, modelLabels, resolveModel, setModelsPath } from "./env.js";
|
|
26
|
-
import { runInstance, applyEvalEnv } from "./run-instance.js";
|
|
27
|
-
import {
|
|
28
|
-
import {
|
|
26
|
+
import { runInstance, applyEvalEnv, slug } from "./run-instance.js";
|
|
27
|
+
import { modelsArm, configArm, releaseOverlayGuard } from "./arm.js";
|
|
28
|
+
import { summarize, writeArtifacts, writeScorecard, aggregateTrials, } from "./report.js";
|
|
29
29
|
import { bootstrapAssets } from "./bootstrap.js";
|
|
30
30
|
import { discoverTiers, loadInstances, workflowFor } from "./discovery.js";
|
|
31
|
-
import { builtinDatasetsRoot, resultsRoot } from "./paths.js";
|
|
31
|
+
import { builtinDatasetsRoot, tierResultsDir, makeRunId, gitShortSha, resultsRoot, dashboardDistRoot } from "./paths.js";
|
|
32
|
+
import { startServer } from "./serve.js";
|
|
32
33
|
import { runInit } from "./init.js";
|
|
33
34
|
/**
|
|
34
35
|
* A clack spinner in a TTY; in non-TTY (CI / piped / agent) a quiet stub that
|
|
@@ -48,9 +49,8 @@ function makeSpinner() {
|
|
|
48
49
|
},
|
|
49
50
|
};
|
|
50
51
|
}
|
|
51
|
-
/** Open a
|
|
52
|
-
function openInBrowser(
|
|
53
|
-
const url = `file://${file}`;
|
|
52
|
+
/** Open a URL in the OS default browser (best-effort, never throws). */
|
|
53
|
+
function openInBrowser(url) {
|
|
54
54
|
const [cmd, args] = process.platform === "darwin"
|
|
55
55
|
? ["open", [url]]
|
|
56
56
|
: process.platform === "win32"
|
|
@@ -99,6 +99,12 @@ function fmtMs(ms) {
|
|
|
99
99
|
function familyLabel(envKey) {
|
|
100
100
|
return envKey.replace(/_API_KEY$/i, "").toLowerCase() || "default";
|
|
101
101
|
}
|
|
102
|
+
/** Attach run metadata to a scorecard (mutate-and-return; `summarize` hands back
|
|
103
|
+
* a fresh object each call, so this is safe). */
|
|
104
|
+
function withMeta(card, meta) {
|
|
105
|
+
card.meta = meta;
|
|
106
|
+
return card;
|
|
107
|
+
}
|
|
102
108
|
/**
|
|
103
109
|
* Silence `console.*` for the whole batch (parallel mode). The per-run
|
|
104
110
|
* `quiet()` swap saves/restores console and would corrupt under concurrent
|
|
@@ -116,6 +122,8 @@ function verdictLine(tierName, inst, r) {
|
|
|
116
122
|
const head = `${chalk.cyan(tierName)}/${inst.instance_id}`;
|
|
117
123
|
if (r.error)
|
|
118
124
|
return `${head} ${chalk.red("harness error")}`;
|
|
125
|
+
if (r.blocked)
|
|
126
|
+
return `${head} ${chalk.yellow("blocked")} ${chalk.dim("(workflow gate)")}`;
|
|
119
127
|
const count = (pass) => (pass !== undefined && r.trials ? chalk.dim(` ${pass}/${r.trials}`) : "");
|
|
120
128
|
const parts = [];
|
|
121
129
|
if (r.resolved !== undefined)
|
|
@@ -154,22 +162,93 @@ function strFlag(name) {
|
|
|
154
162
|
}
|
|
155
163
|
return process.env[`EVAL_${name.toUpperCase()}`];
|
|
156
164
|
}
|
|
165
|
+
/** Collect ALL occurrences of a repeatable string flag (`--name a --name b` or
|
|
166
|
+
* `--name=a`). Used by `--overlay`, which may repeat in `config` mode to compare
|
|
167
|
+
* one arm per overlay. */
|
|
168
|
+
function strFlagAll(name) {
|
|
169
|
+
const out = [];
|
|
170
|
+
const argv = process.argv;
|
|
171
|
+
for (let i = 0; i < argv.length; i++) {
|
|
172
|
+
if (argv[i] === `--${name}` && argv[i + 1] !== undefined) {
|
|
173
|
+
out.push(argv[i + 1]);
|
|
174
|
+
i++;
|
|
175
|
+
continue;
|
|
176
|
+
}
|
|
177
|
+
const m = argv[i].match(new RegExp(`^--${name}=(.+)$`));
|
|
178
|
+
if (m)
|
|
179
|
+
out.push(m[1]);
|
|
180
|
+
}
|
|
181
|
+
return out;
|
|
182
|
+
}
|
|
157
183
|
/** CLI flags that take a following value (so it isn't read as a tier name). */
|
|
158
|
-
const VALUE_FLAGS = new Set(["--runs", "--model", "--models", "--overlay", "--datasets", "--models-file"]);
|
|
184
|
+
const VALUE_FLAGS = new Set(["--runs", "--model", "--models", "--mode", "--overlay", "--datasets", "--models-file"]);
|
|
159
185
|
async function runEval() {
|
|
160
186
|
loadDotEnv();
|
|
161
187
|
p.intro(chalk.bold(`Last Light ${chalk.yellow("·")} eval`));
|
|
188
|
+
// Run type — the comparison axis:
|
|
189
|
+
// `models` — compare N models, each FORCED across every workflow step.
|
|
190
|
+
// `config` — run an overlay's REAL per-step model config (what ships); the
|
|
191
|
+
// arm is the config/overlay, compared across runs (or N overlays
|
|
192
|
+
// side-by-side). `--mode` wins; an explicit `--model`/`--compare`
|
|
193
|
+
// implies `models`; otherwise ask in a TTY (default `models`).
|
|
194
|
+
const modeFlag = strFlag("mode");
|
|
195
|
+
const modelArg = strFlag("model") ?? strFlag("models");
|
|
196
|
+
const compare = process.argv.includes("--compare");
|
|
197
|
+
let runType;
|
|
198
|
+
if (modeFlag === "config")
|
|
199
|
+
runType = "config";
|
|
200
|
+
else if (modeFlag === "models" || modelArg || compare)
|
|
201
|
+
runType = "models";
|
|
202
|
+
else if (process.stdin.isTTY) {
|
|
203
|
+
const picked = await p.select({
|
|
204
|
+
message: "What do you want to eval?",
|
|
205
|
+
options: [
|
|
206
|
+
{ value: "models", label: "compare models", hint: "force each model across every workflow step" },
|
|
207
|
+
{ value: "config", label: "eval config", hint: "an overlay's real per-step model config (what ships)" },
|
|
208
|
+
],
|
|
209
|
+
initialValue: "models",
|
|
210
|
+
});
|
|
211
|
+
if (p.isCancel(picked)) {
|
|
212
|
+
p.cancel("aborted");
|
|
213
|
+
return 1;
|
|
214
|
+
}
|
|
215
|
+
runType = picked;
|
|
216
|
+
}
|
|
217
|
+
else
|
|
218
|
+
runType = "models";
|
|
162
219
|
// Asset roots FIRST — before any getWorkflow/runWorkflow. `--overlay` (or
|
|
163
220
|
// LASTLIGHT_OVERLAY_DIR) layers a deployment's own workflows/skills over the
|
|
164
221
|
// built-ins, and also contributes its `evals/datasets/` (see discovery).
|
|
165
222
|
// With neither set, auto-detect a local `./instance/` overlay checkout — the
|
|
166
223
|
// Separate layout `init --clone` produces — so a bare run "just works".
|
|
224
|
+
// `--overlay` may REPEAT in config mode (one arm per overlay).
|
|
167
225
|
const autoInstance = join(process.cwd(), "instance");
|
|
168
226
|
const autoOverlay = existsSync(join(autoInstance, "config.yaml")) ? autoInstance : undefined;
|
|
169
|
-
const
|
|
170
|
-
|
|
227
|
+
const overlayFlags = strFlagAll("overlay");
|
|
228
|
+
let overlays = overlayFlags.length
|
|
229
|
+
? overlayFlags
|
|
230
|
+
: [process.env.LASTLIGHT_OVERLAY_DIR ?? autoOverlay].filter((d) => !!d);
|
|
231
|
+
// In a config-mode TTY with no overlay given, ask for one (blank ⇒ core defaults).
|
|
232
|
+
if (runType === "config" && !overlayFlags.length && process.stdin.isTTY) {
|
|
233
|
+
const ans = await p.text({
|
|
234
|
+
message: "Overlay dir whose config.yaml drives per-step models (blank = core defaults):",
|
|
235
|
+
placeholder: overlays[0] ?? autoInstance,
|
|
236
|
+
initialValue: overlays[0] ?? "",
|
|
237
|
+
});
|
|
238
|
+
if (p.isCancel(ans)) {
|
|
239
|
+
p.cancel("aborted");
|
|
240
|
+
return 1;
|
|
241
|
+
}
|
|
242
|
+
const dir = ans.trim();
|
|
243
|
+
overlays = dir ? [dir] : [];
|
|
244
|
+
}
|
|
245
|
+
// The primary overlay wires discovery + the initial asset bootstrap. Config
|
|
246
|
+
// arms re-bootstrap their own overlay before running (the asset root is a
|
|
247
|
+
// process global — see the serial loop below).
|
|
248
|
+
const overlayDir = overlays[0];
|
|
249
|
+
if (overlayDir && overlayDir === autoOverlay)
|
|
171
250
|
p.log.info(`overlay → ${chalk.cyan("./instance")} ${chalk.dim("(auto-detected)")}`);
|
|
172
|
-
bootstrapAssets({ overlayDir });
|
|
251
|
+
const { builtInRoot } = bootstrapAssets({ overlayDir });
|
|
173
252
|
// A user/overlay can ship its own model registry too: explicit --models-file
|
|
174
253
|
// wins, else an overlay's `evals/models.json` if present, else the built-in.
|
|
175
254
|
const overlayModels = overlayDir ? join(overlayDir, "evals", "models.json") : undefined;
|
|
@@ -192,7 +271,6 @@ async function runEval() {
|
|
|
192
271
|
p.outro(chalk.red("aborted"));
|
|
193
272
|
return 1;
|
|
194
273
|
}
|
|
195
|
-
const compare = process.argv.includes("--compare");
|
|
196
274
|
const noOpen = process.argv.includes("--no-open") || !!process.env.CI;
|
|
197
275
|
const runs = intFlag("runs", 1);
|
|
198
276
|
// Positional tier names — skip flags AND the values that follow value-flags.
|
|
@@ -248,35 +326,54 @@ async function runEval() {
|
|
|
248
326
|
if (!discovered.has(t))
|
|
249
327
|
p.log.warn(`Unknown tier "${t}". Known: ${known.join(", ")}`);
|
|
250
328
|
}
|
|
251
|
-
//
|
|
252
|
-
//
|
|
253
|
-
//
|
|
254
|
-
//
|
|
255
|
-
//
|
|
256
|
-
//
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
329
|
+
// An "arm" is one column of the comparison, behind the `Arm` seam (src/arm.ts):
|
|
330
|
+
// `models` runs build one `modelsArm` per model (forced across every step);
|
|
331
|
+
// `config` runs build one `configArm` per overlay (its merged per-step config
|
|
332
|
+
// drives selection). Both flow through the same work-list → scorecard →
|
|
333
|
+
// dashboard, keyed on the arm's `label`. run.ts owns *which* arms exist (the
|
|
334
|
+
// flag + registry resolution below); the adapters own *how* to build one.
|
|
335
|
+
//
|
|
336
|
+
// The model-selection sub-mode (only meaningful for `models` runs) is shown in
|
|
337
|
+
// the plan note; `config` runs report their arm count instead.
|
|
338
|
+
const mode = runType === "config" ? "config" : modelArg ? "select" : compare ? "compare" : "single";
|
|
339
|
+
let arms;
|
|
340
|
+
if (runType === "config") {
|
|
341
|
+
// One arm per overlay (or a single core-defaults arm when none). `--model`
|
|
342
|
+
// resolves to an id that overrides each merged config's `default` for quick
|
|
343
|
+
// what-if runs (the override is applied inside configArm).
|
|
344
|
+
const configOverlays = overlays.length ? overlays : [overlayDir];
|
|
345
|
+
const defaultOverride = modelArg ? resolveModel(modelArg).id : undefined;
|
|
346
|
+
arms = configOverlays.map((dir) => configArm(builtInRoot, dir, defaultOverride));
|
|
347
|
+
}
|
|
348
|
+
else {
|
|
349
|
+
// Model selection precedence:
|
|
350
|
+
// 1. --model / --models (or EVAL_MODEL[S]) — an explicit list, fuzzy-matched.
|
|
351
|
+
// 2. --compare — the full cross-vendor set (key-gated).
|
|
352
|
+
// 3. default single model from models.json.
|
|
353
|
+
const entries = modelArg
|
|
354
|
+
? modelArg
|
|
355
|
+
.split(",")
|
|
356
|
+
.map((s) => s.trim())
|
|
357
|
+
.filter(Boolean)
|
|
358
|
+
.map((tok) => {
|
|
359
|
+
const r = resolveModel(tok);
|
|
360
|
+
return { id: r.id, family: r.family };
|
|
361
|
+
})
|
|
362
|
+
: compare
|
|
363
|
+
? compareModels().map((m) => ({ id: m.id, family: m.envKey ?? m.provider ?? "default" }))
|
|
364
|
+
: evalModels().map((id) => ({ id, family: "default" }));
|
|
365
|
+
arms = entries.map((e) => modelsArm(e.id, e.family));
|
|
366
|
+
}
|
|
367
|
+
if (!arms.length) {
|
|
272
368
|
p.log.error("No comparison models available — set provider keys (OPENAI_API_KEY / ANTHROPIC_API_KEY /\n" +
|
|
273
369
|
"FIREWORKS_API_KEY …) for the entries in evals/models.json.");
|
|
274
370
|
p.outro(chalk.red("aborted"));
|
|
275
371
|
return 1;
|
|
276
372
|
}
|
|
277
373
|
const labels = modelLabels();
|
|
278
|
-
//
|
|
279
|
-
|
|
374
|
+
// Instances per tier, resolved ONCE — the case set is identical across arms
|
|
375
|
+
// (arms vary only the model selection / assets, never the cases).
|
|
376
|
+
const tierInstances = new Map();
|
|
280
377
|
for (const tierName of tiers) {
|
|
281
378
|
const tier = discovered.get(tierName);
|
|
282
379
|
const instances = loadInstances(tier);
|
|
@@ -284,7 +381,15 @@ async function runEval() {
|
|
|
284
381
|
p.log.warn(`tier "${tierName}": no instances at ${tier.instancesPath} — skipping`);
|
|
285
382
|
continue;
|
|
286
383
|
}
|
|
287
|
-
|
|
384
|
+
tierInstances.set(tierName, instances);
|
|
385
|
+
}
|
|
386
|
+
// Resolve the work-list up front so we can show deterministic progress. Arms
|
|
387
|
+
// are the OUTER loop so a `config` run's per-arm overlay switches at most once
|
|
388
|
+
// (the serial loop re-bootstraps on arm change; see below).
|
|
389
|
+
const work = [];
|
|
390
|
+
for (const arm of arms) {
|
|
391
|
+
for (const [tierName, instances] of tierInstances) {
|
|
392
|
+
const tier = discovered.get(tierName);
|
|
288
393
|
for (const inst of instances) {
|
|
289
394
|
// Per-instance workflow wins, else the tier's defaultWorkflow (throws if
|
|
290
395
|
// neither is set — surfaced as a harness error for that case).
|
|
@@ -292,8 +397,7 @@ async function runEval() {
|
|
|
292
397
|
tierName,
|
|
293
398
|
defaultWorkflow: workflowFor(tier, inst),
|
|
294
399
|
datasetDir: tier.root,
|
|
295
|
-
|
|
296
|
-
family: e.family,
|
|
400
|
+
arm,
|
|
297
401
|
inst,
|
|
298
402
|
});
|
|
299
403
|
}
|
|
@@ -309,34 +413,84 @@ async function runEval() {
|
|
|
309
413
|
// serial with --serial or when there's only one family.
|
|
310
414
|
const byFamily = new Map();
|
|
311
415
|
for (const w of work) {
|
|
312
|
-
const arr = byFamily.get(w.family);
|
|
416
|
+
const arr = byFamily.get(w.arm.family);
|
|
313
417
|
if (arr)
|
|
314
418
|
arr.push(w);
|
|
315
419
|
else
|
|
316
|
-
byFamily.set(w.family, [w]);
|
|
420
|
+
byFamily.set(w.arm.family, [w]);
|
|
317
421
|
}
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
const
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
422
|
+
// `config` runs stay SERIAL: arms may carry distinct overlays and the asset
|
|
423
|
+
// root is a process global, so the serial loop re-bootstraps per arm (below).
|
|
424
|
+
const parallel = runType === "models" && !process.argv.includes("--serial") && byFamily.size > 1;
|
|
425
|
+
// Each tier writes its OWN folder + scorecard (all sharing this run's id), so
|
|
426
|
+
// tiers stay separate in the dashboard instead of collapsing into one combined
|
|
427
|
+
// `<a+b>` entry. A single invocation appears as the same run under each tier it
|
|
428
|
+
// touched. The `-compare` suffix keeps cross-vendor runs on their own trend
|
|
429
|
+
// line, distinct from single-model runs of the same tier. The dashboard server
|
|
430
|
+
// indexes the whole tree; each tier dir is a `tierKey` segment the SPA routes on.
|
|
431
|
+
// `-compare` keeps cross-vendor runs on their own trend line; `-config` keeps
|
|
432
|
+
// config-eval runs on theirs — so the three run shapes never collapse together.
|
|
433
|
+
const gitSha = gitShortSha();
|
|
434
|
+
const tierKeyFor = (tier) => `${tier}${runType === "config" ? "-config" : compare ? "-compare" : ""}`;
|
|
435
|
+
// One shared runId, checked free in the first tier's dir (collisions in the
|
|
436
|
+
// same second across runs are what the suffix guards — rare, one dir suffices).
|
|
437
|
+
const runId = makeRunId(new Date(), gitSha, tierResultsDir(tierKeyFor(tiers[0])));
|
|
438
|
+
const resultsDirFor = (tier) => join(tierResultsDir(tierKeyFor(tier)), runId);
|
|
439
|
+
// Each case's session logs live under `sessions/<id>__<model>/trial-<N>/`
|
|
440
|
+
// (relative to the tier run dir) — the model is in the name since several
|
|
441
|
+
// models share a run dir; per-trial keeps every `--runs N` trial. `full.jsonl`
|
|
442
|
+
// is the consolidated live-followable transcript; `NN-<phase>.jsonl` are the
|
|
443
|
+
// per-phase splits written when the trial finishes.
|
|
444
|
+
const caseRelFor = (instanceId, model) => `sessions/${slug(instanceId)}__${slug(model)}`;
|
|
445
|
+
const trialRelFor = (instanceId, model, trial) => `${caseRelFor(instanceId, model)}/trial-${trial}`;
|
|
446
|
+
// Per-tier run metadata stamped into every scorecard write (the dashboard reads
|
|
447
|
+
// identity, labels, and live state straight off disk). `live`/`progress`/
|
|
448
|
+
// `pending`/`generatedAt` are layered on per write; `tiers` is the single tier.
|
|
449
|
+
const armLabels = arms.map((a) => labels[a.label] ?? a.label);
|
|
450
|
+
const baseMetaFor = (tier) => ({
|
|
451
|
+
runId,
|
|
452
|
+
runType,
|
|
453
|
+
tiers: [tier],
|
|
454
|
+
models: armLabels,
|
|
324
455
|
runs,
|
|
325
|
-
|
|
456
|
+
gitSha,
|
|
457
|
+
labels,
|
|
458
|
+
});
|
|
459
|
+
// In `config` runs the axis is the config(s); show the merged per-step model
|
|
460
|
+
// map for a single-arm run so the plan is legible.
|
|
461
|
+
const axisLine = runType === "config"
|
|
462
|
+
? `${chalk.bold("configs")} ${armLabels.join(", ")}`
|
|
463
|
+
: `${chalk.bold("models")} ${armLabels.join(", ")}`;
|
|
464
|
+
// For a single-arm run, show the per-step model map (config arms) so the plan
|
|
465
|
+
// is legible. `describe()` returns the summary for config arms, undefined for
|
|
466
|
+
// models arms — the reach into the arm's config map stays behind the seam.
|
|
467
|
+
const armSummary = arms.length === 1 ? arms[0].describe() : undefined;
|
|
468
|
+
const phaseMapLine = armSummary ? `\n${chalk.bold("models")} ${chalk.dim(armSummary)}` : "";
|
|
326
469
|
p.note(`${chalk.bold("mode")} ${mode}${parallel ? chalk.dim(` (parallel · ${byFamily.size} families)`) : ""}\n` +
|
|
327
|
-
`${
|
|
470
|
+
`${axisLine}${phaseMapLine}\n` +
|
|
328
471
|
`${chalk.bold("tiers")} ${tiers.join(", ")}\n` +
|
|
329
472
|
`${chalk.bold("cases")} ${work.length}${runs > 1
|
|
330
473
|
? chalk.dim(` × ${runs} trials = ${work.length * runs} runs · worst-case verdict, mean cost`)
|
|
331
474
|
: ""}`, "plan");
|
|
332
475
|
// `total` counts individual trials so live progress advances per model call.
|
|
333
476
|
const total = work.length * runs;
|
|
334
|
-
//
|
|
335
|
-
|
|
336
|
-
|
|
477
|
+
// Seed an empty live scorecard per tier so the dashboard has something to poll,
|
|
478
|
+
// then start the server and open the SPA deep-linked at this run. The server is
|
|
479
|
+
// skipped entirely when not opening (CI / --no-open) — we only write JSON.
|
|
480
|
+
for (const tier of tiers) {
|
|
481
|
+
writeScorecard(resultsDirFor(tier), withMeta(summarize([]), { ...baseMetaFor(tier), generatedAt: new Date().toISOString(), live: true, progress: `0/${total}` }));
|
|
482
|
+
}
|
|
483
|
+
let server;
|
|
337
484
|
if (!noOpen) {
|
|
338
|
-
|
|
339
|
-
|
|
485
|
+
try {
|
|
486
|
+
server = await startServer({ resultsRoot: resultsRoot(), dashboardRoot: dashboardDistRoot() });
|
|
487
|
+
const runUrl = `${server.url}/#/${encodeURIComponent(tierKeyFor(tiers[0]))}/${encodeURIComponent(runId)}`;
|
|
488
|
+
openInBrowser(runUrl);
|
|
489
|
+
p.log.info(`Live dashboard → ${chalk.cyan(runUrl)}`);
|
|
490
|
+
}
|
|
491
|
+
catch (err) {
|
|
492
|
+
p.log.warn(`Couldn't start the dashboard server (${err.message}) — writing JSON only.`);
|
|
493
|
+
}
|
|
340
494
|
}
|
|
341
495
|
const all = [];
|
|
342
496
|
let harnessErrors = 0;
|
|
@@ -344,44 +498,74 @@ async function runEval() {
|
|
|
344
498
|
// Track in-flight cases so the live report can show running / queued rows.
|
|
345
499
|
const caseKey = (tier, model, id) => `${tier}|${model}|${id}`;
|
|
346
500
|
const running = new Set();
|
|
347
|
-
//
|
|
348
|
-
//
|
|
501
|
+
// Current trial number per running case, so the live "follow" link points at
|
|
502
|
+
// the right `trial-<N>/full.jsonl` (matters only for `--runs N>1`).
|
|
503
|
+
const trialOf = new Map();
|
|
504
|
+
// writeScorecard/summarize/all.push run synchronously to completion inside one
|
|
505
|
+
// event-loop turn, so even with concurrent families they never interleave; the
|
|
506
|
+
// temp-file+rename keeps a polling dashboard from reading a half-written file.
|
|
349
507
|
const refresh = () => {
|
|
350
508
|
const done = new Set(all.map((r) => caseKey(r.tier ?? "", r.model, r.instance_id)));
|
|
351
|
-
const
|
|
352
|
-
|
|
353
|
-
.filter((
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
509
|
+
const now = new Date().toISOString();
|
|
510
|
+
for (const tier of tiers) {
|
|
511
|
+
const tierResults = all.filter((r) => (r.tier ?? "") === tier);
|
|
512
|
+
const pending = work
|
|
513
|
+
.filter((w) => w.tierName === tier)
|
|
514
|
+
.map((w) => ({ w, k: caseKey(w.tierName, w.arm.label, w.inst.instance_id) }))
|
|
515
|
+
.filter(({ k }) => !done.has(k))
|
|
516
|
+
.map(({ w, k }) => ({
|
|
517
|
+
tier: w.tierName,
|
|
518
|
+
model: w.arm.label,
|
|
519
|
+
instance_id: w.inst.instance_id,
|
|
520
|
+
status: running.has(k) ? "running" : "pending",
|
|
521
|
+
// Only a running case has a (live-updating) transcript to follow —
|
|
522
|
+
// point at the current trial's consolidated `full.jsonl`.
|
|
523
|
+
sessionLog: running.has(k)
|
|
524
|
+
? `${trialRelFor(w.inst.instance_id, w.arm.label, trialOf.get(k) ?? 1)}/full.jsonl`
|
|
525
|
+
: undefined,
|
|
526
|
+
}));
|
|
527
|
+
// Per-tier progress (cases), not the global trial count — each tier's
|
|
528
|
+
// scorecard stands alone, so "0/5" across both tiers was misleading.
|
|
529
|
+
const tierCases = work.filter((w) => w.tierName === tier).length;
|
|
530
|
+
writeScorecard(resultsDirFor(tier), withMeta(summarize(tierResults), {
|
|
531
|
+
...baseMetaFor(tier),
|
|
532
|
+
generatedAt: now,
|
|
533
|
+
live: true,
|
|
534
|
+
progress: `${tierResults.length}/${tierCases}`,
|
|
535
|
+
pending,
|
|
536
|
+
}));
|
|
537
|
+
}
|
|
367
538
|
};
|
|
368
539
|
// Run one case `runs` times and fold the trials into a single result
|
|
369
540
|
// (worst-case verdict, mean metrics). `onTrial` ticks per model call.
|
|
370
541
|
const runItem = async (w, onTrial) => {
|
|
542
|
+
const k = caseKey(w.tierName, w.arm.label, w.inst.instance_id);
|
|
371
543
|
const trials = [];
|
|
372
|
-
for (let t =
|
|
544
|
+
for (let t = 1; t <= runs; t++) {
|
|
545
|
+
trialOf.set(k, t); // so the live "follow" link targets this trial
|
|
546
|
+
const trialRel = trialRelFor(w.inst.instance_id, w.arm.label, t);
|
|
373
547
|
const r = await runInstance(w.inst, {
|
|
374
|
-
model
|
|
548
|
+
// The arm carries all model selection (forced model / merged config) +
|
|
549
|
+
// the axis label; runInstance calls its prepare()/recordPhaseModel().
|
|
550
|
+
arm: w.arm,
|
|
375
551
|
datasetDir: w.datasetDir,
|
|
376
552
|
defaultWorkflow: w.defaultWorkflow,
|
|
377
553
|
manageEnv: false,
|
|
554
|
+
// Per-trial dir: `full.jsonl` (consolidated, live) + `NN-<phase>.jsonl`.
|
|
555
|
+
sessionTrialDir: join(resultsDirFor(w.tierName), trialRel),
|
|
556
|
+
sessionTrialRel: trialRel,
|
|
557
|
+
trial: t,
|
|
378
558
|
});
|
|
379
559
|
r.tier = w.tierName;
|
|
380
560
|
trials.push(r);
|
|
381
561
|
completed++;
|
|
382
562
|
onTrial();
|
|
383
563
|
}
|
|
384
|
-
|
|
564
|
+
const agg = aggregateTrials(trials);
|
|
565
|
+
// Keep every trial's per-phase sessions on the aggregate (aggregateTrials
|
|
566
|
+
// only carries trial 0's fields through).
|
|
567
|
+
agg.sessions = trials.map((r) => r.sessionTrial).filter((s) => !!s);
|
|
568
|
+
return agg;
|
|
385
569
|
};
|
|
386
570
|
// Install the eval's static-token env ONCE for the whole batch so concurrent
|
|
387
571
|
// runs share one stable baseline (manageEnv:false on every runInstance).
|
|
@@ -406,7 +590,7 @@ async function runEval() {
|
|
|
406
590
|
try {
|
|
407
591
|
await Promise.all([...byFamily].map(async ([f, items]) => {
|
|
408
592
|
for (const w of items) {
|
|
409
|
-
const k = caseKey(w.tierName, w.
|
|
593
|
+
const k = caseKey(w.tierName, w.arm.label, w.inst.instance_id);
|
|
410
594
|
running.add(k);
|
|
411
595
|
refresh();
|
|
412
596
|
const result = await runItem(w, () => {
|
|
@@ -418,7 +602,7 @@ async function runEval() {
|
|
|
418
602
|
all.push(result);
|
|
419
603
|
if (result.error)
|
|
420
604
|
harnessErrors++;
|
|
421
|
-
const mark = result.error ? chalk.red("✗") : chalk.green("✓");
|
|
605
|
+
const mark = result.error ? chalk.red("✗") : result.blocked ? chalk.yellow("■") : chalk.green("✓");
|
|
422
606
|
verdicts.push(`${mark} ${chalk.dim(familyLabel(f))} ${verdictLine(w.tierName, w.inst, result)}`);
|
|
423
607
|
refresh();
|
|
424
608
|
}
|
|
@@ -431,13 +615,25 @@ async function runEval() {
|
|
|
431
615
|
p.log.message(verdicts.join("\n"));
|
|
432
616
|
}
|
|
433
617
|
else {
|
|
434
|
-
// Serial: one spinner per case (updates per trial) + a verdict line.
|
|
618
|
+
// Serial: one spinner per case (updates per trial) + a verdict line. On
|
|
619
|
+
// each arm change `arm.activate()` repoints the (process-global) asset root
|
|
620
|
+
// to that arm's overlay (a no-op for models arms); work is arms-outer, so
|
|
621
|
+
// it fires at most once per arm. `releaseOverlayGuard()` lets the next arm
|
|
622
|
+
// switch overlays — without it the guard treats the switch as a concurrent
|
|
623
|
+
// overlay and throws (ADR 0001).
|
|
624
|
+
let currentArm;
|
|
435
625
|
for (let i = 0; i < work.length; i++) {
|
|
436
626
|
const w = work[i];
|
|
627
|
+
if (w.arm !== currentArm) {
|
|
628
|
+
if (currentArm)
|
|
629
|
+
releaseOverlayGuard();
|
|
630
|
+
w.arm.activate();
|
|
631
|
+
currentArm = w.arm;
|
|
632
|
+
}
|
|
437
633
|
const s = makeSpinner();
|
|
438
|
-
const head = `${chalk.dim(`[${i + 1}/${work.length}]`)} ${chalk.cyan(w.tierName)}/${w.inst.instance_id} ${chalk.dim(labels[w.
|
|
634
|
+
const head = `${chalk.dim(`[${i + 1}/${work.length}]`)} ${chalk.cyan(w.tierName)}/${w.inst.instance_id} ${chalk.dim(labels[w.arm.label] ?? w.arm.label)}`;
|
|
439
635
|
s.start(head);
|
|
440
|
-
const k = caseKey(w.tierName, w.
|
|
636
|
+
const k = caseKey(w.tierName, w.arm.label, w.inst.instance_id);
|
|
441
637
|
running.add(k);
|
|
442
638
|
refresh();
|
|
443
639
|
let t = 0;
|
|
@@ -451,7 +647,7 @@ async function runEval() {
|
|
|
451
647
|
all.push(result);
|
|
452
648
|
if (result.error)
|
|
453
649
|
harnessErrors++;
|
|
454
|
-
const mark = result.error ? chalk.red("✗") : chalk.green("✓");
|
|
650
|
+
const mark = result.error ? chalk.red("✗") : result.blocked ? chalk.yellow("■") : chalk.green("✓");
|
|
455
651
|
s.stop(`${chalk.dim(`[${i + 1}/${work.length}]`)} ${mark} ${verdictLine(w.tierName, w.inst, result)}`);
|
|
456
652
|
if (result.error) {
|
|
457
653
|
p.log.error(chalk.dim(result.error));
|
|
@@ -466,47 +662,73 @@ async function runEval() {
|
|
|
466
662
|
finally {
|
|
467
663
|
restoreEvalEnv();
|
|
468
664
|
}
|
|
469
|
-
// Final, static
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
const
|
|
473
|
-
|
|
474
|
-
|
|
665
|
+
// Final, static scorecard + machine artifacts. The run-level metadata is
|
|
666
|
+
// persisted into scorecard.json so the dashboard can label, order, and (no
|
|
667
|
+
// longer) live-poll the run without re-deriving from the current config.
|
|
668
|
+
const generatedAt = new Date().toISOString();
|
|
669
|
+
for (const tier of tiers) {
|
|
670
|
+
const tierResults = all.filter((r) => (r.tier ?? "") === tier);
|
|
671
|
+
writeArtifacts(resultsDirFor(tier), withMeta(summarize(tierResults), { ...baseMetaFor(tier), generatedAt, live: false }));
|
|
672
|
+
}
|
|
673
|
+
p.log.success(`Artifacts → ${chalk.cyan(tiers.map((t) => resultsDirFor(t)).join("\n "))}\n /{scorecard.json,predictions.jsonl,sessions/}`);
|
|
475
674
|
const ran = runs > 1 ? `${completed} runs (${all.length} cases × ${runs})` : `${all.length} runs`;
|
|
476
|
-
|
|
477
|
-
|
|
675
|
+
// Keep the dashboard server alive so the just-finished run stays viewable.
|
|
676
|
+
// Only block in an interactive terminal — piped/non-TTY callers exit cleanly.
|
|
677
|
+
if (server && process.stdout.isTTY) {
|
|
678
|
+
const runUrl = `${server.url}/#/${encodeURIComponent(tierKeyFor(tiers[0]))}/${encodeURIComponent(runId)}`;
|
|
679
|
+
p.log.success(`Dashboard → ${chalk.cyan(runUrl)} ${chalk.dim("(serving · Ctrl-C to stop)")}`);
|
|
680
|
+
const tail = harnessErrors > 0 ? chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`) : chalk.green(`done — ${ran}`);
|
|
681
|
+
p.outro(tail);
|
|
682
|
+
await waitForSigint();
|
|
683
|
+
await server.close();
|
|
478
684
|
}
|
|
479
685
|
else {
|
|
480
|
-
|
|
686
|
+
if (server)
|
|
687
|
+
await server.close();
|
|
688
|
+
if (harnessErrors > 0) {
|
|
689
|
+
p.outro(chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`));
|
|
690
|
+
}
|
|
691
|
+
else {
|
|
692
|
+
p.outro(chalk.green(`done — ${ran}`));
|
|
693
|
+
}
|
|
481
694
|
}
|
|
482
695
|
// Non-zero ONLY on harness failure — model quality is the measurement.
|
|
483
696
|
return harnessErrors > 0 ? 1 : 0;
|
|
484
697
|
}
|
|
698
|
+
/** Resolve on the first SIGINT (Ctrl-C) so `run`/`serve` can keep a server up
|
|
699
|
+
* for browsing and shut it down cleanly when the user is done. */
|
|
700
|
+
function waitForSigint() {
|
|
701
|
+
return new Promise((resolve) => {
|
|
702
|
+
const onSig = () => {
|
|
703
|
+
process.off("SIGINT", onSig);
|
|
704
|
+
resolve();
|
|
705
|
+
};
|
|
706
|
+
process.on("SIGINT", onSig);
|
|
707
|
+
});
|
|
708
|
+
}
|
|
485
709
|
/**
|
|
486
|
-
* `
|
|
487
|
-
*
|
|
488
|
-
* report
|
|
489
|
-
*
|
|
710
|
+
* `serve` — start the dashboard server over `eval-results/` and open it in the
|
|
711
|
+
* browser to browse every past run (no models run). The same server `run` uses
|
|
712
|
+
* for the live report; blocks until Ctrl-C. There is no HTML to regenerate any
|
|
713
|
+
* more — the SPA reads the JSON directly.
|
|
490
714
|
*/
|
|
491
|
-
function
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
715
|
+
async function runServe() {
|
|
716
|
+
loadDotEnv();
|
|
717
|
+
const noOpen = process.argv.includes("--no-open");
|
|
718
|
+
const port = intFlag("port", 0) || undefined;
|
|
719
|
+
let server;
|
|
720
|
+
try {
|
|
721
|
+
server = await startServer({ resultsRoot: resultsRoot(), dashboardRoot: dashboardDistRoot(), port });
|
|
495
722
|
}
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
console.error(`no scorecard.json in ${dir}`);
|
|
723
|
+
catch (err) {
|
|
724
|
+
console.error(`Couldn't start the dashboard server: ${err.message}`);
|
|
499
725
|
return 1;
|
|
500
726
|
}
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
const models = [...new Set(card.results.map((r) => r.model))].map((m) => labels[m] ?? m);
|
|
507
|
-
const runs = Math.max(1, ...card.results.map((r) => r.trials ?? 1));
|
|
508
|
-
const html = writeHtml(dir, card, { models, tiers, labels, runs, generatedAt: new Date().toISOString() });
|
|
509
|
-
console.log(`Scorecard → ${html}`);
|
|
727
|
+
console.log(`Dashboard → ${server.url} (serving ${resultsRoot()} · Ctrl-C to stop)`);
|
|
728
|
+
if (!noOpen)
|
|
729
|
+
openInBrowser(server.url);
|
|
730
|
+
await waitForSigint();
|
|
731
|
+
await server.close();
|
|
510
732
|
return 0;
|
|
511
733
|
}
|
|
512
734
|
const USAGE = `lastlight-evals — eval harness for Last Light workflows
|
|
@@ -514,21 +736,31 @@ const USAGE = `lastlight-evals — eval harness for Last Light workflows
|
|
|
514
736
|
Usage:
|
|
515
737
|
lastlight-evals [run] [tiers...] [options] Run evals (default command)
|
|
516
738
|
lastlight-evals init [dir] [options] Scaffold an overlay+evals workspace
|
|
517
|
-
lastlight-evals
|
|
739
|
+
lastlight-evals serve [options] Browse past runs in the dashboard (no models run)
|
|
518
740
|
|
|
519
741
|
Run options:
|
|
520
|
-
--
|
|
521
|
-
|
|
742
|
+
--mode <models|config> Comparison axis. models (default): force each --model
|
|
743
|
+
across every step. config: run an overlay's real
|
|
744
|
+
per-step model config (its config.yaml). No flags in a
|
|
745
|
+
TTY ⇒ asks.
|
|
746
|
+
--overlay <dir> Layer a deployment's workflows/skills + evals/ over built-ins.
|
|
747
|
+
Repeatable in --mode config (one arm per overlay).
|
|
748
|
+
--model <m[,m2]> models: model(s) to run (fuzzy-matched against models.json).
|
|
749
|
+
config: override each config's default model.
|
|
522
750
|
--compare Cross-vendor set (only models whose provider key is present)
|
|
523
751
|
--runs <n> Repeat each case n× (worst-case verdict, mean metrics)
|
|
524
752
|
--serial Force serial execution across provider families
|
|
525
753
|
--datasets <dir> Extra datasets root to discover tiers from
|
|
526
754
|
--models-file <f> Use an explicit models.json
|
|
527
|
-
--no-open Don't open the
|
|
755
|
+
--no-open Don't open / auto-serve the dashboard (also implied by CI=1)
|
|
756
|
+
|
|
757
|
+
Serve options:
|
|
758
|
+
--port <n> Preferred port for the dashboard server (default 4319)
|
|
759
|
+
--no-open Start the server but don't open a browser
|
|
528
760
|
|
|
529
761
|
Run \`lastlight-evals init --help\` for init-specific flags.
|
|
530
762
|
GitHub is mocked end-to-end — no real GitHub token is needed, only a provider key.`;
|
|
531
|
-
/** Top-level subcommand dispatcher: `run` (default) | `init` | `
|
|
763
|
+
/** Top-level subcommand dispatcher: `run` (default) | `init` | `serve`. */
|
|
532
764
|
async function main() {
|
|
533
765
|
const sub = process.argv[2];
|
|
534
766
|
// Top-level help — only when it's not standing in for a `run` tier name.
|
|
@@ -540,9 +772,9 @@ async function main() {
|
|
|
540
772
|
// `init [dir] [flags]` — scaffold a fresh overlay+evals repo.
|
|
541
773
|
return runInit(process.argv.slice(3));
|
|
542
774
|
}
|
|
543
|
-
if (sub === "
|
|
544
|
-
// `
|
|
545
|
-
return
|
|
775
|
+
if (sub === "serve") {
|
|
776
|
+
// `serve` — browse past runs; the live dashboard server, standalone.
|
|
777
|
+
return runServe();
|
|
546
778
|
}
|
|
547
779
|
// `run` is the default; allow an explicit leading `run` token too.
|
|
548
780
|
if (sub === "run")
|