lastlight-evals 0.1.1 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +198 -39
- package/dashboard/dist/assets/index-YwScAAxm.js +238 -0
- package/dashboard/dist/assets/index-uU9Sj3M4.css +1 -0
- package/dashboard/dist/index.html +19 -0
- package/dashboard/dist/logo.png +0 -0
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/.github/workflows/ci.yml +16 -0
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/README.md +7 -1
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/package.json +3 -1
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/scripts/lint.mjs +25 -0
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/test/date-range.test.ts +14 -0
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/tsconfig.json +13 -0
- package/datasets/code-fix/tests/codefix__date-range-off-by-one/date-range.test.ts +5 -5
- package/dist/config.js +100 -0
- package/dist/config.js.map +1 -0
- package/dist/init.js +229 -32
- package/dist/init.js.map +1 -1
- package/dist/mechanism.test.js +41 -0
- package/dist/mechanism.test.js.map +1 -1
- package/dist/metrics.js +91 -3
- package/dist/metrics.js.map +1 -1
- package/dist/paths.js +51 -1
- package/dist/paths.js.map +1 -1
- package/dist/report.js +128 -6
- package/dist/report.js.map +1 -1
- package/dist/run-instance.js +143 -14
- package/dist/run-instance.js.map +1 -1
- package/dist/run.js +401 -89
- package/dist/run.js.map +1 -1
- package/dist/serve.js +136 -0
- package/dist/serve.js.map +1 -0
- package/examples/overlay/README.md +30 -0
- package/examples/overlay/config.yaml +36 -0
- package/examples/overlay-anthropic/config.yaml +27 -0
- package/package.json +16 -6
- package/dist/html-report.js +0 -325
- package/dist/html-report.js.map +0 -1
package/dist/run.js
CHANGED
|
@@ -18,21 +18,39 @@
|
|
|
18
18
|
* `evals/mechanism.test.ts` in the normal `npm test` suite.
|
|
19
19
|
*/
|
|
20
20
|
import { existsSync } from "node:fs";
|
|
21
|
-
import { join } from "node:path";
|
|
21
|
+
import { basename, join } from "node:path";
|
|
22
22
|
import { spawn } from "node:child_process";
|
|
23
23
|
import * as p from "@clack/prompts";
|
|
24
24
|
import chalk from "chalk";
|
|
25
25
|
import { loadDotEnv, hasProviderKey, evalModels, compareModels, modelLabels, resolveModel, setModelsPath } from "./env.js";
|
|
26
|
-
import { runInstance, applyEvalEnv } from "./run-instance.js";
|
|
27
|
-
import {
|
|
28
|
-
import {
|
|
26
|
+
import { runInstance, applyEvalEnv, slug } from "./run-instance.js";
|
|
27
|
+
import { loadMergedConfig } from "./config.js";
|
|
28
|
+
import { summarize, writeArtifacts, writeScorecard, aggregateTrials, } from "./report.js";
|
|
29
29
|
import { bootstrapAssets } from "./bootstrap.js";
|
|
30
30
|
import { discoverTiers, loadInstances, workflowFor } from "./discovery.js";
|
|
31
|
-
import { builtinDatasetsRoot, resultsRoot } from "./paths.js";
|
|
31
|
+
import { builtinDatasetsRoot, tierResultsDir, makeRunId, gitShortSha, resultsRoot, dashboardDistRoot } from "./paths.js";
|
|
32
|
+
import { startServer } from "./serve.js";
|
|
32
33
|
import { runInit } from "./init.js";
|
|
33
|
-
/**
|
|
34
|
-
|
|
35
|
-
|
|
34
|
+
/**
|
|
35
|
+
* A clack spinner in a TTY; in non-TTY (CI / piped / agent) a quiet stub that
|
|
36
|
+
* drops the animation frames — which redraw dozens of times and shred piped logs
|
|
37
|
+
* — and emits only the final `stop()` line. The plan note already prints what's
|
|
38
|
+
* about to run, so dropping the in-progress frames loses nothing in automation.
|
|
39
|
+
*/
|
|
40
|
+
function makeSpinner() {
|
|
41
|
+
if (process.stdout.isTTY)
|
|
42
|
+
return p.spinner();
|
|
43
|
+
return {
|
|
44
|
+
start: () => { },
|
|
45
|
+
message: () => { },
|
|
46
|
+
stop: (msg) => {
|
|
47
|
+
if (msg)
|
|
48
|
+
p.log.message(msg);
|
|
49
|
+
},
|
|
50
|
+
};
|
|
51
|
+
}
|
|
52
|
+
/** Open a URL in the OS default browser (best-effort, never throws). */
|
|
53
|
+
function openInBrowser(url) {
|
|
36
54
|
const [cmd, args] = process.platform === "darwin"
|
|
37
55
|
? ["open", [url]]
|
|
38
56
|
: process.platform === "win32"
|
|
@@ -81,6 +99,12 @@ function fmtMs(ms) {
|
|
|
81
99
|
function familyLabel(envKey) {
|
|
82
100
|
return envKey.replace(/_API_KEY$/i, "").toLowerCase() || "default";
|
|
83
101
|
}
|
|
102
|
+
/** Attach run metadata to a scorecard (mutate-and-return; `summarize` hands back
|
|
103
|
+
* a fresh object each call, so this is safe). */
|
|
104
|
+
function withMeta(card, meta) {
|
|
105
|
+
card.meta = meta;
|
|
106
|
+
return card;
|
|
107
|
+
}
|
|
84
108
|
/**
|
|
85
109
|
* Silence `console.*` for the whole batch (parallel mode). The per-run
|
|
86
110
|
* `quiet()` swap saves/restores console and would corrupt under concurrent
|
|
@@ -98,6 +122,8 @@ function verdictLine(tierName, inst, r) {
|
|
|
98
122
|
const head = `${chalk.cyan(tierName)}/${inst.instance_id}`;
|
|
99
123
|
if (r.error)
|
|
100
124
|
return `${head} ${chalk.red("harness error")}`;
|
|
125
|
+
if (r.blocked)
|
|
126
|
+
return `${head} ${chalk.yellow("blocked")} ${chalk.dim("(workflow gate)")}`;
|
|
101
127
|
const count = (pass) => (pass !== undefined && r.trials ? chalk.dim(` ${pass}/${r.trials}`) : "");
|
|
102
128
|
const parts = [];
|
|
103
129
|
if (r.resolved !== undefined)
|
|
@@ -136,24 +162,104 @@ function strFlag(name) {
|
|
|
136
162
|
}
|
|
137
163
|
return process.env[`EVAL_${name.toUpperCase()}`];
|
|
138
164
|
}
|
|
165
|
+
/** Collect ALL occurrences of a repeatable string flag (`--name a --name b` or
|
|
166
|
+
* `--name=a`). Used by `--overlay`, which may repeat in `config` mode to compare
|
|
167
|
+
* one arm per overlay. */
|
|
168
|
+
function strFlagAll(name) {
|
|
169
|
+
const out = [];
|
|
170
|
+
const argv = process.argv;
|
|
171
|
+
for (let i = 0; i < argv.length; i++) {
|
|
172
|
+
if (argv[i] === `--${name}` && argv[i + 1] !== undefined) {
|
|
173
|
+
out.push(argv[i + 1]);
|
|
174
|
+
i++;
|
|
175
|
+
continue;
|
|
176
|
+
}
|
|
177
|
+
const m = argv[i].match(new RegExp(`^--${name}=(.+)$`));
|
|
178
|
+
if (m)
|
|
179
|
+
out.push(m[1]);
|
|
180
|
+
}
|
|
181
|
+
return out;
|
|
182
|
+
}
|
|
139
183
|
/** CLI flags that take a following value (so it isn't read as a tier name). */
|
|
140
|
-
const VALUE_FLAGS = new Set(["--runs", "--model", "--models", "--overlay", "--datasets", "--models-file"]);
|
|
184
|
+
const VALUE_FLAGS = new Set(["--runs", "--model", "--models", "--mode", "--overlay", "--datasets", "--models-file"]);
|
|
141
185
|
async function runEval() {
|
|
142
186
|
loadDotEnv();
|
|
143
187
|
p.intro(chalk.bold(`Last Light ${chalk.yellow("·")} eval`));
|
|
188
|
+
// Run type — the comparison axis:
|
|
189
|
+
// `models` — compare N models, each FORCED across every workflow step.
|
|
190
|
+
// `config` — run an overlay's REAL per-step model config (what ships); the
|
|
191
|
+
// arm is the config/overlay, compared across runs (or N overlays
|
|
192
|
+
// side-by-side). `--mode` wins; an explicit `--model`/`--compare`
|
|
193
|
+
// implies `models`; otherwise ask in a TTY (default `models`).
|
|
194
|
+
const modeFlag = strFlag("mode");
|
|
195
|
+
const modelArg = strFlag("model") ?? strFlag("models");
|
|
196
|
+
const compare = process.argv.includes("--compare");
|
|
197
|
+
let runType;
|
|
198
|
+
if (modeFlag === "config")
|
|
199
|
+
runType = "config";
|
|
200
|
+
else if (modeFlag === "models" || modelArg || compare)
|
|
201
|
+
runType = "models";
|
|
202
|
+
else if (process.stdin.isTTY) {
|
|
203
|
+
const picked = await p.select({
|
|
204
|
+
message: "What do you want to eval?",
|
|
205
|
+
options: [
|
|
206
|
+
{ value: "models", label: "compare models", hint: "force each model across every workflow step" },
|
|
207
|
+
{ value: "config", label: "eval config", hint: "an overlay's real per-step model config (what ships)" },
|
|
208
|
+
],
|
|
209
|
+
initialValue: "models",
|
|
210
|
+
});
|
|
211
|
+
if (p.isCancel(picked)) {
|
|
212
|
+
p.cancel("aborted");
|
|
213
|
+
return 1;
|
|
214
|
+
}
|
|
215
|
+
runType = picked;
|
|
216
|
+
}
|
|
217
|
+
else
|
|
218
|
+
runType = "models";
|
|
144
219
|
// Asset roots FIRST — before any getWorkflow/runWorkflow. `--overlay` (or
|
|
145
220
|
// LASTLIGHT_OVERLAY_DIR) layers a deployment's own workflows/skills over the
|
|
146
221
|
// built-ins, and also contributes its `evals/datasets/` (see discovery).
|
|
147
|
-
|
|
148
|
-
|
|
222
|
+
// With neither set, auto-detect a local `./instance/` overlay checkout — the
|
|
223
|
+
// Separate layout `init --clone` produces — so a bare run "just works".
|
|
224
|
+
// `--overlay` may REPEAT in config mode (one arm per overlay).
|
|
225
|
+
const autoInstance = join(process.cwd(), "instance");
|
|
226
|
+
const autoOverlay = existsSync(join(autoInstance, "config.yaml")) ? autoInstance : undefined;
|
|
227
|
+
const overlayFlags = strFlagAll("overlay");
|
|
228
|
+
let overlays = overlayFlags.length
|
|
229
|
+
? overlayFlags
|
|
230
|
+
: [process.env.LASTLIGHT_OVERLAY_DIR ?? autoOverlay].filter((d) => !!d);
|
|
231
|
+
// In a config-mode TTY with no overlay given, ask for one (blank ⇒ core defaults).
|
|
232
|
+
if (runType === "config" && !overlayFlags.length && process.stdin.isTTY) {
|
|
233
|
+
const ans = await p.text({
|
|
234
|
+
message: "Overlay dir whose config.yaml drives per-step models (blank = core defaults):",
|
|
235
|
+
placeholder: overlays[0] ?? autoInstance,
|
|
236
|
+
initialValue: overlays[0] ?? "",
|
|
237
|
+
});
|
|
238
|
+
if (p.isCancel(ans)) {
|
|
239
|
+
p.cancel("aborted");
|
|
240
|
+
return 1;
|
|
241
|
+
}
|
|
242
|
+
const dir = ans.trim();
|
|
243
|
+
overlays = dir ? [dir] : [];
|
|
244
|
+
}
|
|
245
|
+
// The primary overlay wires discovery + the initial asset bootstrap. Config
|
|
246
|
+
// arms re-bootstrap their own overlay before running (the asset root is a
|
|
247
|
+
// process global — see the serial loop below).
|
|
248
|
+
const overlayDir = overlays[0];
|
|
249
|
+
if (overlayDir && overlayDir === autoOverlay)
|
|
250
|
+
p.log.info(`overlay → ${chalk.cyan("./instance")} ${chalk.dim("(auto-detected)")}`);
|
|
251
|
+
const { builtInRoot } = bootstrapAssets({ overlayDir });
|
|
149
252
|
// A user/overlay can ship its own model registry too: explicit --models-file
|
|
150
253
|
// wins, else an overlay's `evals/models.json` if present, else the built-in.
|
|
151
254
|
const overlayModels = overlayDir ? join(overlayDir, "evals", "models.json") : undefined;
|
|
152
255
|
const modelsFile = strFlag("models-file") ?? (overlayModels && existsSync(overlayModels) ? overlayModels : undefined);
|
|
153
256
|
if (modelsFile)
|
|
154
257
|
setModelsPath(modelsFile);
|
|
155
|
-
// Discover tiers across built-in + user (--datasets) + overlay roots.
|
|
156
|
-
|
|
258
|
+
// Discover tiers across built-in + user (--datasets) + overlay roots. With no
|
|
259
|
+
// explicit `--datasets`, default to the workspace's own `./evals/datasets`
|
|
260
|
+
// (what `init` seeds) so editing/adding tiers there is picked up automatically.
|
|
261
|
+
const autoDatasets = join(process.cwd(), "evals", "datasets");
|
|
262
|
+
const userDatasetsDir = strFlag("datasets") ?? process.env.LASTLIGHT_EVALS_DATASETS ?? (existsSync(autoDatasets) ? autoDatasets : undefined);
|
|
157
263
|
const discovered = discoverTiers({
|
|
158
264
|
builtinRoot: builtinDatasetsRoot(),
|
|
159
265
|
userDatasetsDir,
|
|
@@ -165,7 +271,6 @@ async function runEval() {
|
|
|
165
271
|
p.outro(chalk.red("aborted"));
|
|
166
272
|
return 1;
|
|
167
273
|
}
|
|
168
|
-
const compare = process.argv.includes("--compare");
|
|
169
274
|
const noOpen = process.argv.includes("--no-open") || !!process.env.CI;
|
|
170
275
|
const runs = intFlag("runs", 1);
|
|
171
276
|
// Positional tier names — skip flags AND the values that follow value-flags.
|
|
@@ -221,35 +326,51 @@ async function runEval() {
|
|
|
221
326
|
if (!discovered.has(t))
|
|
222
327
|
p.log.warn(`Unknown tier "${t}". Known: ${known.join(", ")}`);
|
|
223
328
|
}
|
|
224
|
-
//
|
|
225
|
-
//
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
.
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
:
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
329
|
+
// The model-selection sub-mode (only meaningful for `models` runs); shown in
|
|
330
|
+
// the plan note. `config` runs report their arm count instead.
|
|
331
|
+
const mode = runType === "config" ? "config" : modelArg ? "select" : compare ? "compare" : "single";
|
|
332
|
+
let arms;
|
|
333
|
+
if (runType === "config") {
|
|
334
|
+
// One arm per overlay (or a single core-defaults arm when none). `--model`
|
|
335
|
+
// overrides each merged config's `default` key for quick what-if runs.
|
|
336
|
+
const configOverlays = overlays.length ? overlays : [overlayDir];
|
|
337
|
+
arms = configOverlays.map((dir) => {
|
|
338
|
+
const merged = loadMergedConfig(builtInRoot, dir);
|
|
339
|
+
if (modelArg)
|
|
340
|
+
merged.models.default = resolveModel(modelArg).id;
|
|
341
|
+
const label = dir ? basename(dir) : "config";
|
|
342
|
+
return { label, family: label, modelConfig: merged.models, variantConfig: merged.variants, overlayDir: dir };
|
|
343
|
+
});
|
|
344
|
+
}
|
|
345
|
+
else {
|
|
346
|
+
// Model selection precedence:
|
|
347
|
+
// 1. --model / --models (or EVAL_MODEL[S]) — an explicit list, fuzzy-matched.
|
|
348
|
+
// 2. --compare — the full cross-vendor set (key-gated).
|
|
349
|
+
// 3. default single model from models.json.
|
|
350
|
+
const entries = modelArg
|
|
351
|
+
? modelArg
|
|
352
|
+
.split(",")
|
|
353
|
+
.map((s) => s.trim())
|
|
354
|
+
.filter(Boolean)
|
|
355
|
+
.map((tok) => {
|
|
356
|
+
const r = resolveModel(tok);
|
|
357
|
+
return { id: r.id, family: r.family };
|
|
358
|
+
})
|
|
359
|
+
: compare
|
|
360
|
+
? compareModels().map((m) => ({ id: m.id, family: m.envKey ?? m.provider ?? "default" }))
|
|
361
|
+
: evalModels().map((id) => ({ id, family: "default" }));
|
|
362
|
+
arms = entries.map((e) => ({ label: e.id, family: e.family }));
|
|
363
|
+
}
|
|
364
|
+
if (!arms.length) {
|
|
245
365
|
p.log.error("No comparison models available — set provider keys (OPENAI_API_KEY / ANTHROPIC_API_KEY /\n" +
|
|
246
366
|
"FIREWORKS_API_KEY …) for the entries in evals/models.json.");
|
|
247
367
|
p.outro(chalk.red("aborted"));
|
|
248
368
|
return 1;
|
|
249
369
|
}
|
|
250
370
|
const labels = modelLabels();
|
|
251
|
-
//
|
|
252
|
-
|
|
371
|
+
// Instances per tier, resolved ONCE — the case set is identical across arms
|
|
372
|
+
// (arms vary only the model selection / assets, never the cases).
|
|
373
|
+
const tierInstances = new Map();
|
|
253
374
|
for (const tierName of tiers) {
|
|
254
375
|
const tier = discovered.get(tierName);
|
|
255
376
|
const instances = loadInstances(tier);
|
|
@@ -257,7 +378,15 @@ async function runEval() {
|
|
|
257
378
|
p.log.warn(`tier "${tierName}": no instances at ${tier.instancesPath} — skipping`);
|
|
258
379
|
continue;
|
|
259
380
|
}
|
|
260
|
-
|
|
381
|
+
tierInstances.set(tierName, instances);
|
|
382
|
+
}
|
|
383
|
+
// Resolve the work-list up front so we can show deterministic progress. Arms
|
|
384
|
+
// are the OUTER loop so a `config` run's per-arm overlay switches at most once
|
|
385
|
+
// (the serial loop re-bootstraps on arm change; see below).
|
|
386
|
+
const work = [];
|
|
387
|
+
for (const arm of arms) {
|
|
388
|
+
for (const [tierName, instances] of tierInstances) {
|
|
389
|
+
const tier = discovered.get(tierName);
|
|
261
390
|
for (const inst of instances) {
|
|
262
391
|
// Per-instance workflow wins, else the tier's defaultWorkflow (throws if
|
|
263
392
|
// neither is set — surfaced as a harness error for that case).
|
|
@@ -265,8 +394,11 @@ async function runEval() {
|
|
|
265
394
|
tierName,
|
|
266
395
|
defaultWorkflow: workflowFor(tier, inst),
|
|
267
396
|
datasetDir: tier.root,
|
|
268
|
-
model:
|
|
269
|
-
family:
|
|
397
|
+
model: arm.label,
|
|
398
|
+
family: arm.family,
|
|
399
|
+
modelConfig: arm.modelConfig,
|
|
400
|
+
variantConfig: arm.variantConfig,
|
|
401
|
+
overlayDir: arm.overlayDir,
|
|
270
402
|
inst,
|
|
271
403
|
});
|
|
272
404
|
}
|
|
@@ -288,28 +420,78 @@ async function runEval() {
|
|
|
288
420
|
else
|
|
289
421
|
byFamily.set(w.family, [w]);
|
|
290
422
|
}
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
const
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
423
|
+
// `config` runs stay SERIAL: arms may carry distinct overlays and the asset
|
|
424
|
+
// root is a process global, so the serial loop re-bootstraps per arm (below).
|
|
425
|
+
const parallel = runType === "models" && !process.argv.includes("--serial") && byFamily.size > 1;
|
|
426
|
+
// Each tier writes its OWN folder + scorecard (all sharing this run's id), so
|
|
427
|
+
// tiers stay separate in the dashboard instead of collapsing into one combined
|
|
428
|
+
// `<a+b>` entry. A single invocation appears as the same run under each tier it
|
|
429
|
+
// touched. The `-compare` suffix keeps cross-vendor runs on their own trend
|
|
430
|
+
// line, distinct from single-model runs of the same tier. The dashboard server
|
|
431
|
+
// indexes the whole tree; each tier dir is a `tierKey` segment the SPA routes on.
|
|
432
|
+
// `-compare` keeps cross-vendor runs on their own trend line; `-config` keeps
|
|
433
|
+
// config-eval runs on theirs — so the three run shapes never collapse together.
|
|
434
|
+
const gitSha = gitShortSha();
|
|
435
|
+
const tierKeyFor = (tier) => `${tier}${runType === "config" ? "-config" : compare ? "-compare" : ""}`;
|
|
436
|
+
// One shared runId, checked free in the first tier's dir (collisions in the
|
|
437
|
+
// same second across runs are what the suffix guards — rare, one dir suffices).
|
|
438
|
+
const runId = makeRunId(new Date(), gitSha, tierResultsDir(tierKeyFor(tiers[0])));
|
|
439
|
+
const resultsDirFor = (tier) => join(tierResultsDir(tierKeyFor(tier)), runId);
|
|
440
|
+
// Each case's session logs live under `sessions/<id>__<model>/trial-<N>/`
|
|
441
|
+
// (relative to the tier run dir) — the model is in the name since several
|
|
442
|
+
// models share a run dir; per-trial keeps every `--runs N` trial. `full.jsonl`
|
|
443
|
+
// is the consolidated live-followable transcript; `NN-<phase>.jsonl` are the
|
|
444
|
+
// per-phase splits written when the trial finishes.
|
|
445
|
+
const caseRelFor = (instanceId, model) => `sessions/${slug(instanceId)}__${slug(model)}`;
|
|
446
|
+
const trialRelFor = (instanceId, model, trial) => `${caseRelFor(instanceId, model)}/trial-${trial}`;
|
|
447
|
+
// Per-tier run metadata stamped into every scorecard write (the dashboard reads
|
|
448
|
+
// identity, labels, and live state straight off disk). `live`/`progress`/
|
|
449
|
+
// `pending`/`generatedAt` are layered on per write; `tiers` is the single tier.
|
|
450
|
+
const armLabels = arms.map((a) => labels[a.label] ?? a.label);
|
|
451
|
+
const baseMetaFor = (tier) => ({
|
|
452
|
+
runId,
|
|
453
|
+
runType,
|
|
454
|
+
tiers: [tier],
|
|
455
|
+
models: armLabels,
|
|
297
456
|
runs,
|
|
298
|
-
|
|
457
|
+
gitSha,
|
|
458
|
+
labels,
|
|
459
|
+
});
|
|
460
|
+
// In `config` runs the axis is the config(s); show the merged per-step model
|
|
461
|
+
// map for a single-arm run so the plan is legible.
|
|
462
|
+
const axisLine = runType === "config"
|
|
463
|
+
? `${chalk.bold("configs")} ${armLabels.join(", ")}`
|
|
464
|
+
: `${chalk.bold("models")} ${armLabels.join(", ")}`;
|
|
465
|
+
const phaseMapLine = runType === "config" && arms.length === 1 && arms[0].modelConfig
|
|
466
|
+
? `\n${chalk.bold("models")} ${chalk.dim(Object.entries(arms[0].modelConfig)
|
|
467
|
+
.map(([k, v]) => `${k}→${v}`)
|
|
468
|
+
.join(" "))}`
|
|
469
|
+
: "";
|
|
299
470
|
p.note(`${chalk.bold("mode")} ${mode}${parallel ? chalk.dim(` (parallel · ${byFamily.size} families)`) : ""}\n` +
|
|
300
|
-
`${
|
|
471
|
+
`${axisLine}${phaseMapLine}\n` +
|
|
301
472
|
`${chalk.bold("tiers")} ${tiers.join(", ")}\n` +
|
|
302
473
|
`${chalk.bold("cases")} ${work.length}${runs > 1
|
|
303
474
|
? chalk.dim(` × ${runs} trials = ${work.length * runs} runs · worst-case verdict, mean cost`)
|
|
304
475
|
: ""}`, "plan");
|
|
305
476
|
// `total` counts individual trials so live progress advances per model call.
|
|
306
477
|
const total = work.length * runs;
|
|
307
|
-
//
|
|
308
|
-
|
|
309
|
-
|
|
478
|
+
// Seed an empty live scorecard per tier so the dashboard has something to poll,
|
|
479
|
+
// then start the server and open the SPA deep-linked at this run. The server is
|
|
480
|
+
// skipped entirely when not opening (CI / --no-open) — we only write JSON.
|
|
481
|
+
for (const tier of tiers) {
|
|
482
|
+
writeScorecard(resultsDirFor(tier), withMeta(summarize([]), { ...baseMetaFor(tier), generatedAt: new Date().toISOString(), live: true, progress: `0/${total}` }));
|
|
483
|
+
}
|
|
484
|
+
let server;
|
|
310
485
|
if (!noOpen) {
|
|
311
|
-
|
|
312
|
-
|
|
486
|
+
try {
|
|
487
|
+
server = await startServer({ resultsRoot: resultsRoot(), dashboardRoot: dashboardDistRoot() });
|
|
488
|
+
const runUrl = `${server.url}/#/${encodeURIComponent(tierKeyFor(tiers[0]))}/${encodeURIComponent(runId)}`;
|
|
489
|
+
openInBrowser(runUrl);
|
|
490
|
+
p.log.info(`Live dashboard → ${chalk.cyan(runUrl)}`);
|
|
491
|
+
}
|
|
492
|
+
catch (err) {
|
|
493
|
+
p.log.warn(`Couldn't start the dashboard server (${err.message}) — writing JSON only.`);
|
|
494
|
+
}
|
|
313
495
|
}
|
|
314
496
|
const all = [];
|
|
315
497
|
let harnessErrors = 0;
|
|
@@ -317,44 +499,76 @@ async function runEval() {
|
|
|
317
499
|
// Track in-flight cases so the live report can show running / queued rows.
|
|
318
500
|
const caseKey = (tier, model, id) => `${tier}|${model}|${id}`;
|
|
319
501
|
const running = new Set();
|
|
320
|
-
//
|
|
321
|
-
//
|
|
502
|
+
// Current trial number per running case, so the live "follow" link points at
|
|
503
|
+
// the right `trial-<N>/full.jsonl` (matters only for `--runs N>1`).
|
|
504
|
+
const trialOf = new Map();
|
|
505
|
+
// writeScorecard/summarize/all.push run synchronously to completion inside one
|
|
506
|
+
// event-loop turn, so even with concurrent families they never interleave; the
|
|
507
|
+
// temp-file+rename keeps a polling dashboard from reading a half-written file.
|
|
322
508
|
const refresh = () => {
|
|
323
509
|
const done = new Set(all.map((r) => caseKey(r.tier ?? "", r.model, r.instance_id)));
|
|
324
|
-
const
|
|
325
|
-
|
|
326
|
-
.filter((
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
510
|
+
const now = new Date().toISOString();
|
|
511
|
+
for (const tier of tiers) {
|
|
512
|
+
const tierResults = all.filter((r) => (r.tier ?? "") === tier);
|
|
513
|
+
const pending = work
|
|
514
|
+
.filter((w) => w.tierName === tier)
|
|
515
|
+
.map((w) => ({ w, k: caseKey(w.tierName, w.model, w.inst.instance_id) }))
|
|
516
|
+
.filter(({ k }) => !done.has(k))
|
|
517
|
+
.map(({ w, k }) => ({
|
|
518
|
+
tier: w.tierName,
|
|
519
|
+
model: w.model,
|
|
520
|
+
instance_id: w.inst.instance_id,
|
|
521
|
+
status: running.has(k) ? "running" : "pending",
|
|
522
|
+
// Only a running case has a (live-updating) transcript to follow —
|
|
523
|
+
// point at the current trial's consolidated `full.jsonl`.
|
|
524
|
+
sessionLog: running.has(k)
|
|
525
|
+
? `${trialRelFor(w.inst.instance_id, w.model, trialOf.get(k) ?? 1)}/full.jsonl`
|
|
526
|
+
: undefined,
|
|
527
|
+
}));
|
|
528
|
+
// Per-tier progress (cases), not the global trial count — each tier's
|
|
529
|
+
// scorecard stands alone, so "0/5" across both tiers was misleading.
|
|
530
|
+
const tierCases = work.filter((w) => w.tierName === tier).length;
|
|
531
|
+
writeScorecard(resultsDirFor(tier), withMeta(summarize(tierResults), {
|
|
532
|
+
...baseMetaFor(tier),
|
|
533
|
+
generatedAt: now,
|
|
534
|
+
live: true,
|
|
535
|
+
progress: `${tierResults.length}/${tierCases}`,
|
|
536
|
+
pending,
|
|
537
|
+
}));
|
|
538
|
+
}
|
|
340
539
|
};
|
|
341
540
|
// Run one case `runs` times and fold the trials into a single result
|
|
342
541
|
// (worst-case verdict, mean metrics). `onTrial` ticks per model call.
|
|
343
542
|
const runItem = async (w, onTrial) => {
|
|
543
|
+
const k = caseKey(w.tierName, w.model, w.inst.instance_id);
|
|
344
544
|
const trials = [];
|
|
345
|
-
for (let t =
|
|
545
|
+
for (let t = 1; t <= runs; t++) {
|
|
546
|
+
trialOf.set(k, t); // so the live "follow" link targets this trial
|
|
547
|
+
const trialRel = trialRelFor(w.inst.instance_id, w.model, t);
|
|
346
548
|
const r = await runInstance(w.inst, {
|
|
347
549
|
model: w.model,
|
|
550
|
+
// `config` arms carry the merged per-step maps; `models` arms leave
|
|
551
|
+
// these undefined so the workflow forces `w.model` on every step.
|
|
552
|
+
modelConfig: w.modelConfig,
|
|
553
|
+
variantConfig: w.variantConfig,
|
|
348
554
|
datasetDir: w.datasetDir,
|
|
349
555
|
defaultWorkflow: w.defaultWorkflow,
|
|
350
556
|
manageEnv: false,
|
|
557
|
+
// Per-trial dir: `full.jsonl` (consolidated, live) + `NN-<phase>.jsonl`.
|
|
558
|
+
sessionTrialDir: join(resultsDirFor(w.tierName), trialRel),
|
|
559
|
+
sessionTrialRel: trialRel,
|
|
560
|
+
trial: t,
|
|
351
561
|
});
|
|
352
562
|
r.tier = w.tierName;
|
|
353
563
|
trials.push(r);
|
|
354
564
|
completed++;
|
|
355
565
|
onTrial();
|
|
356
566
|
}
|
|
357
|
-
|
|
567
|
+
const agg = aggregateTrials(trials);
|
|
568
|
+
// Keep every trial's per-phase sessions on the aggregate (aggregateTrials
|
|
569
|
+
// only carries trial 0's fields through).
|
|
570
|
+
agg.sessions = trials.map((r) => r.sessionTrial).filter((s) => !!s);
|
|
571
|
+
return agg;
|
|
358
572
|
};
|
|
359
573
|
// Install the eval's static-token env ONCE for the whole batch so concurrent
|
|
360
574
|
// runs share one stable baseline (manageEnv:false on every runInstance).
|
|
@@ -372,7 +586,7 @@ async function runEval() {
|
|
|
372
586
|
});
|
|
373
587
|
return `${chalk.dim(`${completed}/${total}`)} ${segs.join(chalk.dim(" · "))}`;
|
|
374
588
|
};
|
|
375
|
-
const s =
|
|
589
|
+
const s = makeSpinner();
|
|
376
590
|
s.start(status());
|
|
377
591
|
const restoreConsole = silenceConsole();
|
|
378
592
|
const verdicts = [];
|
|
@@ -391,7 +605,7 @@ async function runEval() {
|
|
|
391
605
|
all.push(result);
|
|
392
606
|
if (result.error)
|
|
393
607
|
harnessErrors++;
|
|
394
|
-
const mark = result.error ? chalk.red("✗") : chalk.green("✓");
|
|
608
|
+
const mark = result.error ? chalk.red("✗") : result.blocked ? chalk.yellow("■") : chalk.green("✓");
|
|
395
609
|
verdicts.push(`${mark} ${chalk.dim(familyLabel(f))} ${verdictLine(w.tierName, w.inst, result)}`);
|
|
396
610
|
refresh();
|
|
397
611
|
}
|
|
@@ -405,9 +619,16 @@ async function runEval() {
|
|
|
405
619
|
}
|
|
406
620
|
else {
|
|
407
621
|
// Serial: one spinner per case (updates per trial) + a verdict line.
|
|
622
|
+
// `config` runs repoint the (process-global) asset root when the arm's
|
|
623
|
+
// overlay changes — work is arms-outer, so this fires at most once per arm.
|
|
624
|
+
let currentOverlay = overlayDir;
|
|
408
625
|
for (let i = 0; i < work.length; i++) {
|
|
409
626
|
const w = work[i];
|
|
410
|
-
|
|
627
|
+
if (runType === "config" && w.overlayDir !== currentOverlay) {
|
|
628
|
+
bootstrapAssets({ overlayDir: w.overlayDir });
|
|
629
|
+
currentOverlay = w.overlayDir;
|
|
630
|
+
}
|
|
631
|
+
const s = makeSpinner();
|
|
411
632
|
const head = `${chalk.dim(`[${i + 1}/${work.length}]`)} ${chalk.cyan(w.tierName)}/${w.inst.instance_id} ${chalk.dim(labels[w.model] ?? w.model)}`;
|
|
412
633
|
s.start(head);
|
|
413
634
|
const k = caseKey(w.tierName, w.model, w.inst.instance_id);
|
|
@@ -424,7 +645,7 @@ async function runEval() {
|
|
|
424
645
|
all.push(result);
|
|
425
646
|
if (result.error)
|
|
426
647
|
harnessErrors++;
|
|
427
|
-
const mark = result.error ? chalk.red("✗") : chalk.green("✓");
|
|
648
|
+
const mark = result.error ? chalk.red("✗") : result.blocked ? chalk.yellow("■") : chalk.green("✓");
|
|
428
649
|
s.stop(`${chalk.dim(`[${i + 1}/${work.length}]`)} ${mark} ${verdictLine(w.tierName, w.inst, result)}`);
|
|
429
650
|
if (result.error) {
|
|
430
651
|
p.log.error(chalk.dim(result.error));
|
|
@@ -439,28 +660,119 @@ async function runEval() {
|
|
|
439
660
|
finally {
|
|
440
661
|
restoreEvalEnv();
|
|
441
662
|
}
|
|
442
|
-
// Final, static
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
const
|
|
446
|
-
|
|
447
|
-
|
|
663
|
+
// Final, static scorecard + machine artifacts. The run-level metadata is
|
|
664
|
+
// persisted into scorecard.json so the dashboard can label, order, and (no
|
|
665
|
+
// longer) live-poll the run without re-deriving from the current config.
|
|
666
|
+
const generatedAt = new Date().toISOString();
|
|
667
|
+
for (const tier of tiers) {
|
|
668
|
+
const tierResults = all.filter((r) => (r.tier ?? "") === tier);
|
|
669
|
+
writeArtifacts(resultsDirFor(tier), withMeta(summarize(tierResults), { ...baseMetaFor(tier), generatedAt, live: false }));
|
|
670
|
+
}
|
|
671
|
+
p.log.success(`Artifacts → ${chalk.cyan(tiers.map((t) => resultsDirFor(t)).join("\n "))}\n /{scorecard.json,predictions.jsonl,sessions/}`);
|
|
448
672
|
const ran = runs > 1 ? `${completed} runs (${all.length} cases × ${runs})` : `${all.length} runs`;
|
|
449
|
-
|
|
450
|
-
|
|
673
|
+
// Keep the dashboard server alive so the just-finished run stays viewable.
|
|
674
|
+
// Only block in an interactive terminal — piped/non-TTY callers exit cleanly.
|
|
675
|
+
if (server && process.stdout.isTTY) {
|
|
676
|
+
const runUrl = `${server.url}/#/${encodeURIComponent(tierKeyFor(tiers[0]))}/${encodeURIComponent(runId)}`;
|
|
677
|
+
p.log.success(`Dashboard → ${chalk.cyan(runUrl)} ${chalk.dim("(serving · Ctrl-C to stop)")}`);
|
|
678
|
+
const tail = harnessErrors > 0 ? chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`) : chalk.green(`done — ${ran}`);
|
|
679
|
+
p.outro(tail);
|
|
680
|
+
await waitForSigint();
|
|
681
|
+
await server.close();
|
|
451
682
|
}
|
|
452
683
|
else {
|
|
453
|
-
|
|
684
|
+
if (server)
|
|
685
|
+
await server.close();
|
|
686
|
+
if (harnessErrors > 0) {
|
|
687
|
+
p.outro(chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`));
|
|
688
|
+
}
|
|
689
|
+
else {
|
|
690
|
+
p.outro(chalk.green(`done — ${ran}`));
|
|
691
|
+
}
|
|
454
692
|
}
|
|
455
693
|
// Non-zero ONLY on harness failure — model quality is the measurement.
|
|
456
694
|
return harnessErrors > 0 ? 1 : 0;
|
|
457
695
|
}
|
|
458
|
-
/**
|
|
696
|
+
/** Resolve on the first SIGINT (Ctrl-C) so `run`/`serve` can keep a server up
|
|
697
|
+
* for browsing and shut it down cleanly when the user is done. */
|
|
698
|
+
function waitForSigint() {
|
|
699
|
+
return new Promise((resolve) => {
|
|
700
|
+
const onSig = () => {
|
|
701
|
+
process.off("SIGINT", onSig);
|
|
702
|
+
resolve();
|
|
703
|
+
};
|
|
704
|
+
process.on("SIGINT", onSig);
|
|
705
|
+
});
|
|
706
|
+
}
|
|
707
|
+
/**
|
|
708
|
+
* `serve` — start the dashboard server over `eval-results/` and open it in the
|
|
709
|
+
* browser to browse every past run (no models run). The same server `run` uses
|
|
710
|
+
* for the live report; blocks until Ctrl-C. There is no HTML to regenerate any
|
|
711
|
+
* more — the SPA reads the JSON directly.
|
|
712
|
+
*/
|
|
713
|
+
async function runServe() {
|
|
714
|
+
loadDotEnv();
|
|
715
|
+
const noOpen = process.argv.includes("--no-open");
|
|
716
|
+
const port = intFlag("port", 0) || undefined;
|
|
717
|
+
let server;
|
|
718
|
+
try {
|
|
719
|
+
server = await startServer({ resultsRoot: resultsRoot(), dashboardRoot: dashboardDistRoot(), port });
|
|
720
|
+
}
|
|
721
|
+
catch (err) {
|
|
722
|
+
console.error(`Couldn't start the dashboard server: ${err.message}`);
|
|
723
|
+
return 1;
|
|
724
|
+
}
|
|
725
|
+
console.log(`Dashboard → ${server.url} (serving ${resultsRoot()} · Ctrl-C to stop)`);
|
|
726
|
+
if (!noOpen)
|
|
727
|
+
openInBrowser(server.url);
|
|
728
|
+
await waitForSigint();
|
|
729
|
+
await server.close();
|
|
730
|
+
return 0;
|
|
731
|
+
}
|
|
732
|
+
const USAGE = `lastlight-evals — eval harness for Last Light workflows
|
|
733
|
+
|
|
734
|
+
Usage:
|
|
735
|
+
lastlight-evals [run] [tiers...] [options] Run evals (default command)
|
|
736
|
+
lastlight-evals init [dir] [options] Scaffold an overlay+evals workspace
|
|
737
|
+
lastlight-evals serve [options] Browse past runs in the dashboard (no models run)
|
|
738
|
+
|
|
739
|
+
Run options:
|
|
740
|
+
--mode <models|config> Comparison axis. models (default): force each --model
|
|
741
|
+
across every step. config: run an overlay's real
|
|
742
|
+
per-step model config (its config.yaml). No flags in a
|
|
743
|
+
TTY ⇒ asks.
|
|
744
|
+
--overlay <dir> Layer a deployment's workflows/skills + evals/ over built-ins.
|
|
745
|
+
Repeatable in --mode config (one arm per overlay).
|
|
746
|
+
--model <m[,m2]> models: model(s) to run (fuzzy-matched against models.json).
|
|
747
|
+
config: override each config's default model.
|
|
748
|
+
--compare Cross-vendor set (only models whose provider key is present)
|
|
749
|
+
--runs <n> Repeat each case n× (worst-case verdict, mean metrics)
|
|
750
|
+
--serial Force serial execution across provider families
|
|
751
|
+
--datasets <dir> Extra datasets root to discover tiers from
|
|
752
|
+
--models-file <f> Use an explicit models.json
|
|
753
|
+
--no-open Don't open / auto-serve the dashboard (also implied by CI=1)
|
|
754
|
+
|
|
755
|
+
Serve options:
|
|
756
|
+
--port <n> Preferred port for the dashboard server (default 4319)
|
|
757
|
+
--no-open Start the server but don't open a browser
|
|
758
|
+
|
|
759
|
+
Run \`lastlight-evals init --help\` for init-specific flags.
|
|
760
|
+
GitHub is mocked end-to-end — no real GitHub token is needed, only a provider key.`;
|
|
761
|
+
/** Top-level subcommand dispatcher: `run` (default) | `init` | `serve`. */
|
|
459
762
|
async function main() {
|
|
460
763
|
const sub = process.argv[2];
|
|
764
|
+
// Top-level help — only when it's not standing in for a `run` tier name.
|
|
765
|
+
if (sub === "help" || sub === "--help" || sub === "-h") {
|
|
766
|
+
console.log(USAGE);
|
|
767
|
+
return 0;
|
|
768
|
+
}
|
|
461
769
|
if (sub === "init") {
|
|
462
|
-
// `init [dir]` — scaffold a fresh overlay+evals repo.
|
|
463
|
-
return runInit(process.argv
|
|
770
|
+
// `init [dir] [flags]` — scaffold a fresh overlay+evals repo.
|
|
771
|
+
return runInit(process.argv.slice(3));
|
|
772
|
+
}
|
|
773
|
+
if (sub === "serve") {
|
|
774
|
+
// `serve` — browse past runs; the live dashboard server, standalone.
|
|
775
|
+
return runServe();
|
|
464
776
|
}
|
|
465
777
|
// `run` is the default; allow an explicit leading `run` token too.
|
|
466
778
|
if (sub === "run")
|