lastlight-evals 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +98 -12
- package/dashboard/dist/assets/index-YwScAAxm.js +238 -0
- package/dashboard/dist/assets/index-uU9Sj3M4.css +1 -0
- package/dashboard/dist/index.html +19 -0
- package/dashboard/dist/logo.png +0 -0
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/.github/workflows/ci.yml +16 -0
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/README.md +7 -1
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/package.json +3 -1
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/scripts/lint.mjs +25 -0
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/test/date-range.test.ts +14 -0
- package/datasets/code-fix/repos/codefix__date-range-off-by-one/tsconfig.json +13 -0
- package/datasets/code-fix/tests/codefix__date-range-off-by-one/date-range.test.ts +5 -5
- package/dist/config.js +100 -0
- package/dist/config.js.map +1 -0
- package/dist/mechanism.test.js +41 -0
- package/dist/mechanism.test.js.map +1 -1
- package/dist/metrics.js +83 -0
- package/dist/metrics.js.map +1 -1
- package/dist/paths.js +51 -1
- package/dist/paths.js.map +1 -1
- package/dist/report.js +112 -3
- package/dist/report.js.map +1 -1
- package/dist/run-instance.js +141 -14
- package/dist/run-instance.js.map +1 -1
- package/dist/run.js +342 -112
- package/dist/run.js.map +1 -1
- package/dist/serve.js +136 -0
- package/dist/serve.js.map +1 -0
- package/examples/overlay/README.md +30 -0
- package/examples/overlay/config.yaml +36 -0
- package/examples/overlay-anthropic/config.yaml +27 -0
- package/package.json +16 -6
- package/dist/html-report.js +0 -325
- package/dist/html-report.js.map +0 -1
package/dist/run.js
CHANGED
|
@@ -17,18 +17,19 @@
|
|
|
17
17
|
* The deterministic, AI-free plumbing is covered separately by
|
|
18
18
|
* `evals/mechanism.test.ts` in the normal `npm test` suite.
|
|
19
19
|
*/
|
|
20
|
-
import { existsSync
|
|
21
|
-
import { join } from "node:path";
|
|
20
|
+
import { existsSync } from "node:fs";
|
|
21
|
+
import { basename, join } from "node:path";
|
|
22
22
|
import { spawn } from "node:child_process";
|
|
23
23
|
import * as p from "@clack/prompts";
|
|
24
24
|
import chalk from "chalk";
|
|
25
25
|
import { loadDotEnv, hasProviderKey, evalModels, compareModels, modelLabels, resolveModel, setModelsPath } from "./env.js";
|
|
26
|
-
import { runInstance, applyEvalEnv } from "./run-instance.js";
|
|
27
|
-
import {
|
|
28
|
-
import {
|
|
26
|
+
import { runInstance, applyEvalEnv, slug } from "./run-instance.js";
|
|
27
|
+
import { loadMergedConfig } from "./config.js";
|
|
28
|
+
import { summarize, writeArtifacts, writeScorecard, aggregateTrials, } from "./report.js";
|
|
29
29
|
import { bootstrapAssets } from "./bootstrap.js";
|
|
30
30
|
import { discoverTiers, loadInstances, workflowFor } from "./discovery.js";
|
|
31
|
-
import { builtinDatasetsRoot, resultsRoot } from "./paths.js";
|
|
31
|
+
import { builtinDatasetsRoot, tierResultsDir, makeRunId, gitShortSha, resultsRoot, dashboardDistRoot } from "./paths.js";
|
|
32
|
+
import { startServer } from "./serve.js";
|
|
32
33
|
import { runInit } from "./init.js";
|
|
33
34
|
/**
|
|
34
35
|
* A clack spinner in a TTY; in non-TTY (CI / piped / agent) a quiet stub that
|
|
@@ -48,9 +49,8 @@ function makeSpinner() {
|
|
|
48
49
|
},
|
|
49
50
|
};
|
|
50
51
|
}
|
|
51
|
-
/** Open a
|
|
52
|
-
function openInBrowser(
|
|
53
|
-
const url = `file://${file}`;
|
|
52
|
+
/** Open a URL in the OS default browser (best-effort, never throws). */
|
|
53
|
+
function openInBrowser(url) {
|
|
54
54
|
const [cmd, args] = process.platform === "darwin"
|
|
55
55
|
? ["open", [url]]
|
|
56
56
|
: process.platform === "win32"
|
|
@@ -99,6 +99,12 @@ function fmtMs(ms) {
|
|
|
99
99
|
function familyLabel(envKey) {
|
|
100
100
|
return envKey.replace(/_API_KEY$/i, "").toLowerCase() || "default";
|
|
101
101
|
}
|
|
102
|
+
/** Attach run metadata to a scorecard (mutate-and-return; `summarize` hands back
|
|
103
|
+
* a fresh object each call, so this is safe). */
|
|
104
|
+
function withMeta(card, meta) {
|
|
105
|
+
card.meta = meta;
|
|
106
|
+
return card;
|
|
107
|
+
}
|
|
102
108
|
/**
|
|
103
109
|
* Silence `console.*` for the whole batch (parallel mode). The per-run
|
|
104
110
|
* `quiet()` swap saves/restores console and would corrupt under concurrent
|
|
@@ -116,6 +122,8 @@ function verdictLine(tierName, inst, r) {
|
|
|
116
122
|
const head = `${chalk.cyan(tierName)}/${inst.instance_id}`;
|
|
117
123
|
if (r.error)
|
|
118
124
|
return `${head} ${chalk.red("harness error")}`;
|
|
125
|
+
if (r.blocked)
|
|
126
|
+
return `${head} ${chalk.yellow("blocked")} ${chalk.dim("(workflow gate)")}`;
|
|
119
127
|
const count = (pass) => (pass !== undefined && r.trials ? chalk.dim(` ${pass}/${r.trials}`) : "");
|
|
120
128
|
const parts = [];
|
|
121
129
|
if (r.resolved !== undefined)
|
|
@@ -154,22 +162,93 @@ function strFlag(name) {
|
|
|
154
162
|
}
|
|
155
163
|
return process.env[`EVAL_${name.toUpperCase()}`];
|
|
156
164
|
}
|
|
165
|
+
/** Collect ALL occurrences of a repeatable string flag (`--name a --name b` or
|
|
166
|
+
* `--name=a`). Used by `--overlay`, which may repeat in `config` mode to compare
|
|
167
|
+
* one arm per overlay. */
|
|
168
|
+
function strFlagAll(name) {
|
|
169
|
+
const out = [];
|
|
170
|
+
const argv = process.argv;
|
|
171
|
+
for (let i = 0; i < argv.length; i++) {
|
|
172
|
+
if (argv[i] === `--${name}` && argv[i + 1] !== undefined) {
|
|
173
|
+
out.push(argv[i + 1]);
|
|
174
|
+
i++;
|
|
175
|
+
continue;
|
|
176
|
+
}
|
|
177
|
+
const m = argv[i].match(new RegExp(`^--${name}=(.+)$`));
|
|
178
|
+
if (m)
|
|
179
|
+
out.push(m[1]);
|
|
180
|
+
}
|
|
181
|
+
return out;
|
|
182
|
+
}
|
|
157
183
|
/** CLI flags that take a following value (so it isn't read as a tier name). */
|
|
158
|
-
const VALUE_FLAGS = new Set(["--runs", "--model", "--models", "--overlay", "--datasets", "--models-file"]);
|
|
184
|
+
const VALUE_FLAGS = new Set(["--runs", "--model", "--models", "--mode", "--overlay", "--datasets", "--models-file"]);
|
|
159
185
|
async function runEval() {
|
|
160
186
|
loadDotEnv();
|
|
161
187
|
p.intro(chalk.bold(`Last Light ${chalk.yellow("·")} eval`));
|
|
188
|
+
// Run type — the comparison axis:
|
|
189
|
+
// `models` — compare N models, each FORCED across every workflow step.
|
|
190
|
+
// `config` — run an overlay's REAL per-step model config (what ships); the
|
|
191
|
+
// arm is the config/overlay, compared across runs (or N overlays
|
|
192
|
+
// side-by-side). `--mode` wins; an explicit `--model`/`--compare`
|
|
193
|
+
// implies `models`; otherwise ask in a TTY (default `models`).
|
|
194
|
+
const modeFlag = strFlag("mode");
|
|
195
|
+
const modelArg = strFlag("model") ?? strFlag("models");
|
|
196
|
+
const compare = process.argv.includes("--compare");
|
|
197
|
+
let runType;
|
|
198
|
+
if (modeFlag === "config")
|
|
199
|
+
runType = "config";
|
|
200
|
+
else if (modeFlag === "models" || modelArg || compare)
|
|
201
|
+
runType = "models";
|
|
202
|
+
else if (process.stdin.isTTY) {
|
|
203
|
+
const picked = await p.select({
|
|
204
|
+
message: "What do you want to eval?",
|
|
205
|
+
options: [
|
|
206
|
+
{ value: "models", label: "compare models", hint: "force each model across every workflow step" },
|
|
207
|
+
{ value: "config", label: "eval config", hint: "an overlay's real per-step model config (what ships)" },
|
|
208
|
+
],
|
|
209
|
+
initialValue: "models",
|
|
210
|
+
});
|
|
211
|
+
if (p.isCancel(picked)) {
|
|
212
|
+
p.cancel("aborted");
|
|
213
|
+
return 1;
|
|
214
|
+
}
|
|
215
|
+
runType = picked;
|
|
216
|
+
}
|
|
217
|
+
else
|
|
218
|
+
runType = "models";
|
|
162
219
|
// Asset roots FIRST — before any getWorkflow/runWorkflow. `--overlay` (or
|
|
163
220
|
// LASTLIGHT_OVERLAY_DIR) layers a deployment's own workflows/skills over the
|
|
164
221
|
// built-ins, and also contributes its `evals/datasets/` (see discovery).
|
|
165
222
|
// With neither set, auto-detect a local `./instance/` overlay checkout — the
|
|
166
223
|
// Separate layout `init --clone` produces — so a bare run "just works".
|
|
224
|
+
// `--overlay` may REPEAT in config mode (one arm per overlay).
|
|
167
225
|
const autoInstance = join(process.cwd(), "instance");
|
|
168
226
|
const autoOverlay = existsSync(join(autoInstance, "config.yaml")) ? autoInstance : undefined;
|
|
169
|
-
const
|
|
170
|
-
|
|
227
|
+
const overlayFlags = strFlagAll("overlay");
|
|
228
|
+
let overlays = overlayFlags.length
|
|
229
|
+
? overlayFlags
|
|
230
|
+
: [process.env.LASTLIGHT_OVERLAY_DIR ?? autoOverlay].filter((d) => !!d);
|
|
231
|
+
// In a config-mode TTY with no overlay given, ask for one (blank ⇒ core defaults).
|
|
232
|
+
if (runType === "config" && !overlayFlags.length && process.stdin.isTTY) {
|
|
233
|
+
const ans = await p.text({
|
|
234
|
+
message: "Overlay dir whose config.yaml drives per-step models (blank = core defaults):",
|
|
235
|
+
placeholder: overlays[0] ?? autoInstance,
|
|
236
|
+
initialValue: overlays[0] ?? "",
|
|
237
|
+
});
|
|
238
|
+
if (p.isCancel(ans)) {
|
|
239
|
+
p.cancel("aborted");
|
|
240
|
+
return 1;
|
|
241
|
+
}
|
|
242
|
+
const dir = ans.trim();
|
|
243
|
+
overlays = dir ? [dir] : [];
|
|
244
|
+
}
|
|
245
|
+
// The primary overlay wires discovery + the initial asset bootstrap. Config
|
|
246
|
+
// arms re-bootstrap their own overlay before running (the asset root is a
|
|
247
|
+
// process global — see the serial loop below).
|
|
248
|
+
const overlayDir = overlays[0];
|
|
249
|
+
if (overlayDir && overlayDir === autoOverlay)
|
|
171
250
|
p.log.info(`overlay → ${chalk.cyan("./instance")} ${chalk.dim("(auto-detected)")}`);
|
|
172
|
-
bootstrapAssets({ overlayDir });
|
|
251
|
+
const { builtInRoot } = bootstrapAssets({ overlayDir });
|
|
173
252
|
// A user/overlay can ship its own model registry too: explicit --models-file
|
|
174
253
|
// wins, else an overlay's `evals/models.json` if present, else the built-in.
|
|
175
254
|
const overlayModels = overlayDir ? join(overlayDir, "evals", "models.json") : undefined;
|
|
@@ -192,7 +271,6 @@ async function runEval() {
|
|
|
192
271
|
p.outro(chalk.red("aborted"));
|
|
193
272
|
return 1;
|
|
194
273
|
}
|
|
195
|
-
const compare = process.argv.includes("--compare");
|
|
196
274
|
const noOpen = process.argv.includes("--no-open") || !!process.env.CI;
|
|
197
275
|
const runs = intFlag("runs", 1);
|
|
198
276
|
// Positional tier names — skip flags AND the values that follow value-flags.
|
|
@@ -248,35 +326,51 @@ async function runEval() {
|
|
|
248
326
|
if (!discovered.has(t))
|
|
249
327
|
p.log.warn(`Unknown tier "${t}". Known: ${known.join(", ")}`);
|
|
250
328
|
}
|
|
251
|
-
//
|
|
252
|
-
//
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
.
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
:
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
329
|
+
// The model-selection sub-mode (only meaningful for `models` runs); shown in
|
|
330
|
+
// the plan note. `config` runs report their arm count instead.
|
|
331
|
+
const mode = runType === "config" ? "config" : modelArg ? "select" : compare ? "compare" : "single";
|
|
332
|
+
let arms;
|
|
333
|
+
if (runType === "config") {
|
|
334
|
+
// One arm per overlay (or a single core-defaults arm when none). `--model`
|
|
335
|
+
// overrides each merged config's `default` key for quick what-if runs.
|
|
336
|
+
const configOverlays = overlays.length ? overlays : [overlayDir];
|
|
337
|
+
arms = configOverlays.map((dir) => {
|
|
338
|
+
const merged = loadMergedConfig(builtInRoot, dir);
|
|
339
|
+
if (modelArg)
|
|
340
|
+
merged.models.default = resolveModel(modelArg).id;
|
|
341
|
+
const label = dir ? basename(dir) : "config";
|
|
342
|
+
return { label, family: label, modelConfig: merged.models, variantConfig: merged.variants, overlayDir: dir };
|
|
343
|
+
});
|
|
344
|
+
}
|
|
345
|
+
else {
|
|
346
|
+
// Model selection precedence:
|
|
347
|
+
// 1. --model / --models (or EVAL_MODEL[S]) — an explicit list, fuzzy-matched.
|
|
348
|
+
// 2. --compare — the full cross-vendor set (key-gated).
|
|
349
|
+
// 3. default single model from models.json.
|
|
350
|
+
const entries = modelArg
|
|
351
|
+
? modelArg
|
|
352
|
+
.split(",")
|
|
353
|
+
.map((s) => s.trim())
|
|
354
|
+
.filter(Boolean)
|
|
355
|
+
.map((tok) => {
|
|
356
|
+
const r = resolveModel(tok);
|
|
357
|
+
return { id: r.id, family: r.family };
|
|
358
|
+
})
|
|
359
|
+
: compare
|
|
360
|
+
? compareModels().map((m) => ({ id: m.id, family: m.envKey ?? m.provider ?? "default" }))
|
|
361
|
+
: evalModels().map((id) => ({ id, family: "default" }));
|
|
362
|
+
arms = entries.map((e) => ({ label: e.id, family: e.family }));
|
|
363
|
+
}
|
|
364
|
+
if (!arms.length) {
|
|
272
365
|
p.log.error("No comparison models available — set provider keys (OPENAI_API_KEY / ANTHROPIC_API_KEY /\n" +
|
|
273
366
|
"FIREWORKS_API_KEY …) for the entries in evals/models.json.");
|
|
274
367
|
p.outro(chalk.red("aborted"));
|
|
275
368
|
return 1;
|
|
276
369
|
}
|
|
277
370
|
const labels = modelLabels();
|
|
278
|
-
//
|
|
279
|
-
|
|
371
|
+
// Instances per tier, resolved ONCE — the case set is identical across arms
|
|
372
|
+
// (arms vary only the model selection / assets, never the cases).
|
|
373
|
+
const tierInstances = new Map();
|
|
280
374
|
for (const tierName of tiers) {
|
|
281
375
|
const tier = discovered.get(tierName);
|
|
282
376
|
const instances = loadInstances(tier);
|
|
@@ -284,7 +378,15 @@ async function runEval() {
|
|
|
284
378
|
p.log.warn(`tier "${tierName}": no instances at ${tier.instancesPath} — skipping`);
|
|
285
379
|
continue;
|
|
286
380
|
}
|
|
287
|
-
|
|
381
|
+
tierInstances.set(tierName, instances);
|
|
382
|
+
}
|
|
383
|
+
// Resolve the work-list up front so we can show deterministic progress. Arms
|
|
384
|
+
// are the OUTER loop so a `config` run's per-arm overlay switches at most once
|
|
385
|
+
// (the serial loop re-bootstraps on arm change; see below).
|
|
386
|
+
const work = [];
|
|
387
|
+
for (const arm of arms) {
|
|
388
|
+
for (const [tierName, instances] of tierInstances) {
|
|
389
|
+
const tier = discovered.get(tierName);
|
|
288
390
|
for (const inst of instances) {
|
|
289
391
|
// Per-instance workflow wins, else the tier's defaultWorkflow (throws if
|
|
290
392
|
// neither is set — surfaced as a harness error for that case).
|
|
@@ -292,8 +394,11 @@ async function runEval() {
|
|
|
292
394
|
tierName,
|
|
293
395
|
defaultWorkflow: workflowFor(tier, inst),
|
|
294
396
|
datasetDir: tier.root,
|
|
295
|
-
model:
|
|
296
|
-
family:
|
|
397
|
+
model: arm.label,
|
|
398
|
+
family: arm.family,
|
|
399
|
+
modelConfig: arm.modelConfig,
|
|
400
|
+
variantConfig: arm.variantConfig,
|
|
401
|
+
overlayDir: arm.overlayDir,
|
|
297
402
|
inst,
|
|
298
403
|
});
|
|
299
404
|
}
|
|
@@ -315,28 +420,78 @@ async function runEval() {
|
|
|
315
420
|
else
|
|
316
421
|
byFamily.set(w.family, [w]);
|
|
317
422
|
}
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
const
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
423
|
+
// `config` runs stay SERIAL: arms may carry distinct overlays and the asset
|
|
424
|
+
// root is a process global, so the serial loop re-bootstraps per arm (below).
|
|
425
|
+
const parallel = runType === "models" && !process.argv.includes("--serial") && byFamily.size > 1;
|
|
426
|
+
// Each tier writes its OWN folder + scorecard (all sharing this run's id), so
|
|
427
|
+
// tiers stay separate in the dashboard instead of collapsing into one combined
|
|
428
|
+
// `<a+b>` entry. A single invocation appears as the same run under each tier it
|
|
429
|
+
// touched. The `-compare` suffix keeps cross-vendor runs on their own trend
|
|
430
|
+
// line, distinct from single-model runs of the same tier. The dashboard server
|
|
431
|
+
// indexes the whole tree; each tier dir is a `tierKey` segment the SPA routes on.
|
|
432
|
+
// `-compare` keeps cross-vendor runs on their own trend line; `-config` keeps
|
|
433
|
+
// config-eval runs on theirs — so the three run shapes never collapse together.
|
|
434
|
+
const gitSha = gitShortSha();
|
|
435
|
+
const tierKeyFor = (tier) => `${tier}${runType === "config" ? "-config" : compare ? "-compare" : ""}`;
|
|
436
|
+
// One shared runId, checked free in the first tier's dir (collisions in the
|
|
437
|
+
// same second across runs are what the suffix guards — rare, one dir suffices).
|
|
438
|
+
const runId = makeRunId(new Date(), gitSha, tierResultsDir(tierKeyFor(tiers[0])));
|
|
439
|
+
const resultsDirFor = (tier) => join(tierResultsDir(tierKeyFor(tier)), runId);
|
|
440
|
+
// Each case's session logs live under `sessions/<id>__<model>/trial-<N>/`
|
|
441
|
+
// (relative to the tier run dir) — the model is in the name since several
|
|
442
|
+
// models share a run dir; per-trial keeps every `--runs N` trial. `full.jsonl`
|
|
443
|
+
// is the consolidated live-followable transcript; `NN-<phase>.jsonl` are the
|
|
444
|
+
// per-phase splits written when the trial finishes.
|
|
445
|
+
const caseRelFor = (instanceId, model) => `sessions/${slug(instanceId)}__${slug(model)}`;
|
|
446
|
+
const trialRelFor = (instanceId, model, trial) => `${caseRelFor(instanceId, model)}/trial-${trial}`;
|
|
447
|
+
// Per-tier run metadata stamped into every scorecard write (the dashboard reads
|
|
448
|
+
// identity, labels, and live state straight off disk). `live`/`progress`/
|
|
449
|
+
// `pending`/`generatedAt` are layered on per write; `tiers` is the single tier.
|
|
450
|
+
const armLabels = arms.map((a) => labels[a.label] ?? a.label);
|
|
451
|
+
const baseMetaFor = (tier) => ({
|
|
452
|
+
runId,
|
|
453
|
+
runType,
|
|
454
|
+
tiers: [tier],
|
|
455
|
+
models: armLabels,
|
|
324
456
|
runs,
|
|
325
|
-
|
|
457
|
+
gitSha,
|
|
458
|
+
labels,
|
|
459
|
+
});
|
|
460
|
+
// In `config` runs the axis is the config(s); show the merged per-step model
|
|
461
|
+
// map for a single-arm run so the plan is legible.
|
|
462
|
+
const axisLine = runType === "config"
|
|
463
|
+
? `${chalk.bold("configs")} ${armLabels.join(", ")}`
|
|
464
|
+
: `${chalk.bold("models")} ${armLabels.join(", ")}`;
|
|
465
|
+
const phaseMapLine = runType === "config" && arms.length === 1 && arms[0].modelConfig
|
|
466
|
+
? `\n${chalk.bold("models")} ${chalk.dim(Object.entries(arms[0].modelConfig)
|
|
467
|
+
.map(([k, v]) => `${k}→${v}`)
|
|
468
|
+
.join(" "))}`
|
|
469
|
+
: "";
|
|
326
470
|
p.note(`${chalk.bold("mode")} ${mode}${parallel ? chalk.dim(` (parallel · ${byFamily.size} families)`) : ""}\n` +
|
|
327
|
-
`${
|
|
471
|
+
`${axisLine}${phaseMapLine}\n` +
|
|
328
472
|
`${chalk.bold("tiers")} ${tiers.join(", ")}\n` +
|
|
329
473
|
`${chalk.bold("cases")} ${work.length}${runs > 1
|
|
330
474
|
? chalk.dim(` × ${runs} trials = ${work.length * runs} runs · worst-case verdict, mean cost`)
|
|
331
475
|
: ""}`, "plan");
|
|
332
476
|
// `total` counts individual trials so live progress advances per model call.
|
|
333
477
|
const total = work.length * runs;
|
|
334
|
-
//
|
|
335
|
-
|
|
336
|
-
|
|
478
|
+
// Seed an empty live scorecard per tier so the dashboard has something to poll,
|
|
479
|
+
// then start the server and open the SPA deep-linked at this run. The server is
|
|
480
|
+
// skipped entirely when not opening (CI / --no-open) — we only write JSON.
|
|
481
|
+
for (const tier of tiers) {
|
|
482
|
+
writeScorecard(resultsDirFor(tier), withMeta(summarize([]), { ...baseMetaFor(tier), generatedAt: new Date().toISOString(), live: true, progress: `0/${total}` }));
|
|
483
|
+
}
|
|
484
|
+
let server;
|
|
337
485
|
if (!noOpen) {
|
|
338
|
-
|
|
339
|
-
|
|
486
|
+
try {
|
|
487
|
+
server = await startServer({ resultsRoot: resultsRoot(), dashboardRoot: dashboardDistRoot() });
|
|
488
|
+
const runUrl = `${server.url}/#/${encodeURIComponent(tierKeyFor(tiers[0]))}/${encodeURIComponent(runId)}`;
|
|
489
|
+
openInBrowser(runUrl);
|
|
490
|
+
p.log.info(`Live dashboard → ${chalk.cyan(runUrl)}`);
|
|
491
|
+
}
|
|
492
|
+
catch (err) {
|
|
493
|
+
p.log.warn(`Couldn't start the dashboard server (${err.message}) — writing JSON only.`);
|
|
494
|
+
}
|
|
340
495
|
}
|
|
341
496
|
const all = [];
|
|
342
497
|
let harnessErrors = 0;
|
|
@@ -344,44 +499,76 @@ async function runEval() {
|
|
|
344
499
|
// Track in-flight cases so the live report can show running / queued rows.
|
|
345
500
|
const caseKey = (tier, model, id) => `${tier}|${model}|${id}`;
|
|
346
501
|
const running = new Set();
|
|
347
|
-
//
|
|
348
|
-
//
|
|
502
|
+
// Current trial number per running case, so the live "follow" link points at
|
|
503
|
+
// the right `trial-<N>/full.jsonl` (matters only for `--runs N>1`).
|
|
504
|
+
const trialOf = new Map();
|
|
505
|
+
// writeScorecard/summarize/all.push run synchronously to completion inside one
|
|
506
|
+
// event-loop turn, so even with concurrent families they never interleave; the
|
|
507
|
+
// temp-file+rename keeps a polling dashboard from reading a half-written file.
|
|
349
508
|
const refresh = () => {
|
|
350
509
|
const done = new Set(all.map((r) => caseKey(r.tier ?? "", r.model, r.instance_id)));
|
|
351
|
-
const
|
|
352
|
-
|
|
353
|
-
.filter((
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
510
|
+
const now = new Date().toISOString();
|
|
511
|
+
for (const tier of tiers) {
|
|
512
|
+
const tierResults = all.filter((r) => (r.tier ?? "") === tier);
|
|
513
|
+
const pending = work
|
|
514
|
+
.filter((w) => w.tierName === tier)
|
|
515
|
+
.map((w) => ({ w, k: caseKey(w.tierName, w.model, w.inst.instance_id) }))
|
|
516
|
+
.filter(({ k }) => !done.has(k))
|
|
517
|
+
.map(({ w, k }) => ({
|
|
518
|
+
tier: w.tierName,
|
|
519
|
+
model: w.model,
|
|
520
|
+
instance_id: w.inst.instance_id,
|
|
521
|
+
status: running.has(k) ? "running" : "pending",
|
|
522
|
+
// Only a running case has a (live-updating) transcript to follow —
|
|
523
|
+
// point at the current trial's consolidated `full.jsonl`.
|
|
524
|
+
sessionLog: running.has(k)
|
|
525
|
+
? `${trialRelFor(w.inst.instance_id, w.model, trialOf.get(k) ?? 1)}/full.jsonl`
|
|
526
|
+
: undefined,
|
|
527
|
+
}));
|
|
528
|
+
// Per-tier progress (cases), not the global trial count — each tier's
|
|
529
|
+
// scorecard stands alone, so "0/5" across both tiers was misleading.
|
|
530
|
+
const tierCases = work.filter((w) => w.tierName === tier).length;
|
|
531
|
+
writeScorecard(resultsDirFor(tier), withMeta(summarize(tierResults), {
|
|
532
|
+
...baseMetaFor(tier),
|
|
533
|
+
generatedAt: now,
|
|
534
|
+
live: true,
|
|
535
|
+
progress: `${tierResults.length}/${tierCases}`,
|
|
536
|
+
pending,
|
|
537
|
+
}));
|
|
538
|
+
}
|
|
367
539
|
};
|
|
368
540
|
// Run one case `runs` times and fold the trials into a single result
|
|
369
541
|
// (worst-case verdict, mean metrics). `onTrial` ticks per model call.
|
|
370
542
|
const runItem = async (w, onTrial) => {
|
|
543
|
+
const k = caseKey(w.tierName, w.model, w.inst.instance_id);
|
|
371
544
|
const trials = [];
|
|
372
|
-
for (let t =
|
|
545
|
+
for (let t = 1; t <= runs; t++) {
|
|
546
|
+
trialOf.set(k, t); // so the live "follow" link targets this trial
|
|
547
|
+
const trialRel = trialRelFor(w.inst.instance_id, w.model, t);
|
|
373
548
|
const r = await runInstance(w.inst, {
|
|
374
549
|
model: w.model,
|
|
550
|
+
// `config` arms carry the merged per-step maps; `models` arms leave
|
|
551
|
+
// these undefined so the workflow forces `w.model` on every step.
|
|
552
|
+
modelConfig: w.modelConfig,
|
|
553
|
+
variantConfig: w.variantConfig,
|
|
375
554
|
datasetDir: w.datasetDir,
|
|
376
555
|
defaultWorkflow: w.defaultWorkflow,
|
|
377
556
|
manageEnv: false,
|
|
557
|
+
// Per-trial dir: `full.jsonl` (consolidated, live) + `NN-<phase>.jsonl`.
|
|
558
|
+
sessionTrialDir: join(resultsDirFor(w.tierName), trialRel),
|
|
559
|
+
sessionTrialRel: trialRel,
|
|
560
|
+
trial: t,
|
|
378
561
|
});
|
|
379
562
|
r.tier = w.tierName;
|
|
380
563
|
trials.push(r);
|
|
381
564
|
completed++;
|
|
382
565
|
onTrial();
|
|
383
566
|
}
|
|
384
|
-
|
|
567
|
+
const agg = aggregateTrials(trials);
|
|
568
|
+
// Keep every trial's per-phase sessions on the aggregate (aggregateTrials
|
|
569
|
+
// only carries trial 0's fields through).
|
|
570
|
+
agg.sessions = trials.map((r) => r.sessionTrial).filter((s) => !!s);
|
|
571
|
+
return agg;
|
|
385
572
|
};
|
|
386
573
|
// Install the eval's static-token env ONCE for the whole batch so concurrent
|
|
387
574
|
// runs share one stable baseline (manageEnv:false on every runInstance).
|
|
@@ -418,7 +605,7 @@ async function runEval() {
|
|
|
418
605
|
all.push(result);
|
|
419
606
|
if (result.error)
|
|
420
607
|
harnessErrors++;
|
|
421
|
-
const mark = result.error ? chalk.red("✗") : chalk.green("✓");
|
|
608
|
+
const mark = result.error ? chalk.red("✗") : result.blocked ? chalk.yellow("■") : chalk.green("✓");
|
|
422
609
|
verdicts.push(`${mark} ${chalk.dim(familyLabel(f))} ${verdictLine(w.tierName, w.inst, result)}`);
|
|
423
610
|
refresh();
|
|
424
611
|
}
|
|
@@ -432,8 +619,15 @@ async function runEval() {
|
|
|
432
619
|
}
|
|
433
620
|
else {
|
|
434
621
|
// Serial: one spinner per case (updates per trial) + a verdict line.
|
|
622
|
+
// `config` runs repoint the (process-global) asset root when the arm's
|
|
623
|
+
// overlay changes — work is arms-outer, so this fires at most once per arm.
|
|
624
|
+
let currentOverlay = overlayDir;
|
|
435
625
|
for (let i = 0; i < work.length; i++) {
|
|
436
626
|
const w = work[i];
|
|
627
|
+
if (runType === "config" && w.overlayDir !== currentOverlay) {
|
|
628
|
+
bootstrapAssets({ overlayDir: w.overlayDir });
|
|
629
|
+
currentOverlay = w.overlayDir;
|
|
630
|
+
}
|
|
437
631
|
const s = makeSpinner();
|
|
438
632
|
const head = `${chalk.dim(`[${i + 1}/${work.length}]`)} ${chalk.cyan(w.tierName)}/${w.inst.instance_id} ${chalk.dim(labels[w.model] ?? w.model)}`;
|
|
439
633
|
s.start(head);
|
|
@@ -451,7 +645,7 @@ async function runEval() {
|
|
|
451
645
|
all.push(result);
|
|
452
646
|
if (result.error)
|
|
453
647
|
harnessErrors++;
|
|
454
|
-
const mark = result.error ? chalk.red("✗") : chalk.green("✓");
|
|
648
|
+
const mark = result.error ? chalk.red("✗") : result.blocked ? chalk.yellow("■") : chalk.green("✓");
|
|
455
649
|
s.stop(`${chalk.dim(`[${i + 1}/${work.length}]`)} ${mark} ${verdictLine(w.tierName, w.inst, result)}`);
|
|
456
650
|
if (result.error) {
|
|
457
651
|
p.log.error(chalk.dim(result.error));
|
|
@@ -466,47 +660,73 @@ async function runEval() {
|
|
|
466
660
|
finally {
|
|
467
661
|
restoreEvalEnv();
|
|
468
662
|
}
|
|
469
|
-
// Final, static
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
const
|
|
473
|
-
|
|
474
|
-
|
|
663
|
+
// Final, static scorecard + machine artifacts. The run-level metadata is
|
|
664
|
+
// persisted into scorecard.json so the dashboard can label, order, and (no
|
|
665
|
+
// longer) live-poll the run without re-deriving from the current config.
|
|
666
|
+
const generatedAt = new Date().toISOString();
|
|
667
|
+
for (const tier of tiers) {
|
|
668
|
+
const tierResults = all.filter((r) => (r.tier ?? "") === tier);
|
|
669
|
+
writeArtifacts(resultsDirFor(tier), withMeta(summarize(tierResults), { ...baseMetaFor(tier), generatedAt, live: false }));
|
|
670
|
+
}
|
|
671
|
+
p.log.success(`Artifacts → ${chalk.cyan(tiers.map((t) => resultsDirFor(t)).join("\n "))}\n /{scorecard.json,predictions.jsonl,sessions/}`);
|
|
475
672
|
const ran = runs > 1 ? `${completed} runs (${all.length} cases × ${runs})` : `${all.length} runs`;
|
|
476
|
-
|
|
477
|
-
|
|
673
|
+
// Keep the dashboard server alive so the just-finished run stays viewable.
|
|
674
|
+
// Only block in an interactive terminal — piped/non-TTY callers exit cleanly.
|
|
675
|
+
if (server && process.stdout.isTTY) {
|
|
676
|
+
const runUrl = `${server.url}/#/${encodeURIComponent(tierKeyFor(tiers[0]))}/${encodeURIComponent(runId)}`;
|
|
677
|
+
p.log.success(`Dashboard → ${chalk.cyan(runUrl)} ${chalk.dim("(serving · Ctrl-C to stop)")}`);
|
|
678
|
+
const tail = harnessErrors > 0 ? chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`) : chalk.green(`done — ${ran}`);
|
|
679
|
+
p.outro(tail);
|
|
680
|
+
await waitForSigint();
|
|
681
|
+
await server.close();
|
|
478
682
|
}
|
|
479
683
|
else {
|
|
480
|
-
|
|
684
|
+
if (server)
|
|
685
|
+
await server.close();
|
|
686
|
+
if (harnessErrors > 0) {
|
|
687
|
+
p.outro(chalk.yellow(`done — ${ran}, ${harnessErrors} harness error${harnessErrors === 1 ? "" : "s"} (see above)`));
|
|
688
|
+
}
|
|
689
|
+
else {
|
|
690
|
+
p.outro(chalk.green(`done — ${ran}`));
|
|
691
|
+
}
|
|
481
692
|
}
|
|
482
693
|
// Non-zero ONLY on harness failure — model quality is the measurement.
|
|
483
694
|
return harnessErrors > 0 ? 1 : 0;
|
|
484
695
|
}
|
|
696
|
+
/** Resolve on the first SIGINT (Ctrl-C) so `run`/`serve` can keep a server up
|
|
697
|
+
* for browsing and shut it down cleanly when the user is done. */
|
|
698
|
+
function waitForSigint() {
|
|
699
|
+
return new Promise((resolve) => {
|
|
700
|
+
const onSig = () => {
|
|
701
|
+
process.off("SIGINT", onSig);
|
|
702
|
+
resolve();
|
|
703
|
+
};
|
|
704
|
+
process.on("SIGINT", onSig);
|
|
705
|
+
});
|
|
706
|
+
}
|
|
485
707
|
/**
|
|
486
|
-
* `
|
|
487
|
-
*
|
|
488
|
-
* report
|
|
489
|
-
*
|
|
708
|
+
* `serve` — start the dashboard server over `eval-results/` and open it in the
|
|
709
|
+
* browser to browse every past run (no models run). The same server `run` uses
|
|
710
|
+
* for the live report; blocks until Ctrl-C. There is no HTML to regenerate any
|
|
711
|
+
* more — the SPA reads the JSON directly.
|
|
490
712
|
*/
|
|
491
|
-
function
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
713
|
+
async function runServe() {
|
|
714
|
+
loadDotEnv();
|
|
715
|
+
const noOpen = process.argv.includes("--no-open");
|
|
716
|
+
const port = intFlag("port", 0) || undefined;
|
|
717
|
+
let server;
|
|
718
|
+
try {
|
|
719
|
+
server = await startServer({ resultsRoot: resultsRoot(), dashboardRoot: dashboardDistRoot(), port });
|
|
495
720
|
}
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
console.error(`no scorecard.json in ${dir}`);
|
|
721
|
+
catch (err) {
|
|
722
|
+
console.error(`Couldn't start the dashboard server: ${err.message}`);
|
|
499
723
|
return 1;
|
|
500
724
|
}
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
const models = [...new Set(card.results.map((r) => r.model))].map((m) => labels[m] ?? m);
|
|
507
|
-
const runs = Math.max(1, ...card.results.map((r) => r.trials ?? 1));
|
|
508
|
-
const html = writeHtml(dir, card, { models, tiers, labels, runs, generatedAt: new Date().toISOString() });
|
|
509
|
-
console.log(`Scorecard → ${html}`);
|
|
725
|
+
console.log(`Dashboard → ${server.url} (serving ${resultsRoot()} · Ctrl-C to stop)`);
|
|
726
|
+
if (!noOpen)
|
|
727
|
+
openInBrowser(server.url);
|
|
728
|
+
await waitForSigint();
|
|
729
|
+
await server.close();
|
|
510
730
|
return 0;
|
|
511
731
|
}
|
|
512
732
|
const USAGE = `lastlight-evals — eval harness for Last Light workflows
|
|
@@ -514,21 +734,31 @@ const USAGE = `lastlight-evals — eval harness for Last Light workflows
|
|
|
514
734
|
Usage:
|
|
515
735
|
lastlight-evals [run] [tiers...] [options] Run evals (default command)
|
|
516
736
|
lastlight-evals init [dir] [options] Scaffold an overlay+evals workspace
|
|
517
|
-
lastlight-evals
|
|
737
|
+
lastlight-evals serve [options] Browse past runs in the dashboard (no models run)
|
|
518
738
|
|
|
519
739
|
Run options:
|
|
520
|
-
--
|
|
521
|
-
|
|
740
|
+
--mode <models|config> Comparison axis. models (default): force each --model
|
|
741
|
+
across every step. config: run an overlay's real
|
|
742
|
+
per-step model config (its config.yaml). No flags in a
|
|
743
|
+
TTY ⇒ asks.
|
|
744
|
+
--overlay <dir> Layer a deployment's workflows/skills + evals/ over built-ins.
|
|
745
|
+
Repeatable in --mode config (one arm per overlay).
|
|
746
|
+
--model <m[,m2]> models: model(s) to run (fuzzy-matched against models.json).
|
|
747
|
+
config: override each config's default model.
|
|
522
748
|
--compare Cross-vendor set (only models whose provider key is present)
|
|
523
749
|
--runs <n> Repeat each case n× (worst-case verdict, mean metrics)
|
|
524
750
|
--serial Force serial execution across provider families
|
|
525
751
|
--datasets <dir> Extra datasets root to discover tiers from
|
|
526
752
|
--models-file <f> Use an explicit models.json
|
|
527
|
-
--no-open Don't open the
|
|
753
|
+
--no-open Don't open / auto-serve the dashboard (also implied by CI=1)
|
|
754
|
+
|
|
755
|
+
Serve options:
|
|
756
|
+
--port <n> Preferred port for the dashboard server (default 4319)
|
|
757
|
+
--no-open Start the server but don't open a browser
|
|
528
758
|
|
|
529
759
|
Run \`lastlight-evals init --help\` for init-specific flags.
|
|
530
760
|
GitHub is mocked end-to-end — no real GitHub token is needed, only a provider key.`;
|
|
531
|
-
/** Top-level subcommand dispatcher: `run` (default) | `init` | `
|
|
761
|
+
/** Top-level subcommand dispatcher: `run` (default) | `init` | `serve`. */
|
|
532
762
|
async function main() {
|
|
533
763
|
const sub = process.argv[2];
|
|
534
764
|
// Top-level help — only when it's not standing in for a `run` tier name.
|
|
@@ -540,9 +770,9 @@ async function main() {
|
|
|
540
770
|
// `init [dir] [flags]` — scaffold a fresh overlay+evals repo.
|
|
541
771
|
return runInit(process.argv.slice(3));
|
|
542
772
|
}
|
|
543
|
-
if (sub === "
|
|
544
|
-
// `
|
|
545
|
-
return
|
|
773
|
+
if (sub === "serve") {
|
|
774
|
+
// `serve` — browse past runs; the live dashboard server, standalone.
|
|
775
|
+
return runServe();
|
|
546
776
|
}
|
|
547
777
|
// `run` is the default; allow an explicit leading `run` token too.
|
|
548
778
|
if (sub === "run")
|