@prismatic-io/lux 0.0.2-preview.14 → 0.0.2-preview.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/answerers/claude-code/index.d.ts +14 -2
- package/lib/answerers/claude-code/index.d.ts.map +1 -1
- package/lib/answerers/claude-code/index.js +3 -3
- package/lib/answerers/claude-code/index.js.map +1 -1
- package/lib/answerers/persona/index.d.ts +28 -2
- package/lib/answerers/persona/index.d.ts.map +1 -1
- package/lib/answerers/persona/index.js +3 -3
- package/lib/answerers/persona/index.js.map +1 -1
- package/lib/answerers/terminal/index.d.ts +8 -2
- package/lib/answerers/terminal/index.d.ts.map +1 -1
- package/lib/answerers/terminal/index.js +3 -2
- package/lib/answerers/terminal/index.js.map +1 -1
- package/lib/assertions/core/resource-checks.d.ts.map +1 -1
- package/lib/assertions/core/resource-checks.js +3 -5
- package/lib/assertions/core/resource-checks.js.map +1 -1
- package/lib/assertions/rubric/index.d.ts.map +1 -1
- package/lib/assertions/rubric/index.js +7 -1
- package/lib/assertions/rubric/index.js.map +1 -1
- package/lib/assertions/rubric/internal.d.ts +16 -5
- package/lib/assertions/rubric/internal.d.ts.map +1 -1
- package/lib/assertions/rubric/internal.js +10 -6
- package/lib/assertions/rubric/internal.js.map +1 -1
- package/lib/authoring.d.ts +32 -2
- package/lib/authoring.d.ts.map +1 -1
- package/lib/authoring.js +33 -1
- package/lib/authoring.js.map +1 -1
- package/lib/cli/bin.js +2 -13
- package/lib/cli/bin.js.map +1 -1
- package/lib/cli/command-runtime.d.ts +3 -3
- package/lib/cli/command-runtime.d.ts.map +1 -1
- package/lib/cli/command-runtime.js +17 -6
- package/lib/cli/command-runtime.js.map +1 -1
- package/lib/cli/init-templates.d.ts +11 -0
- package/lib/cli/init-templates.d.ts.map +1 -0
- package/lib/cli/init-templates.js +207 -0
- package/lib/cli/init-templates.js.map +1 -0
- package/lib/cli/init.d.ts +7 -0
- package/lib/cli/init.d.ts.map +1 -1
- package/lib/cli/init.js +23 -173
- package/lib/cli/init.js.map +1 -1
- package/lib/cli/output-schemas.d.ts +1399 -0
- package/lib/cli/output-schemas.d.ts.map +1 -0
- package/lib/cli/output-schemas.js +91 -0
- package/lib/cli/output-schemas.js.map +1 -0
- package/lib/cli/program.d.ts +135 -4
- package/lib/cli/program.d.ts.map +1 -1
- package/lib/cli/program.js +725 -520
- package/lib/cli/program.js.map +1 -1
- package/lib/cli/render/reporter.d.ts.map +1 -1
- package/lib/cli/render/reporter.js +24 -4
- package/lib/cli/render/reporter.js.map +1 -1
- package/lib/cli/run-options.d.ts +0 -4
- package/lib/cli/run-options.d.ts.map +1 -1
- package/lib/cli/run-options.js +8 -19
- package/lib/cli/run-options.js.map +1 -1
- package/lib/cli/view.d.ts +5 -3
- package/lib/cli/view.d.ts.map +1 -1
- package/lib/cli/view.js +120 -78
- package/lib/cli/view.js.map +1 -1
- package/lib/core/annotation.d.ts +6 -2
- package/lib/core/annotation.d.ts.map +1 -1
- package/lib/core/annotation.js +2 -1
- package/lib/core/annotation.js.map +1 -1
- package/lib/core/answerer.d.ts +12 -0
- package/lib/core/answerer.d.ts.map +1 -1
- package/lib/core/answerer.js +2 -0
- package/lib/core/answerer.js.map +1 -1
- package/lib/core/assertion.d.ts +3 -0
- package/lib/core/assertion.d.ts.map +1 -1
- package/lib/core/assertion.js +3 -0
- package/lib/core/assertion.js.map +1 -1
- package/lib/core/case.d.ts +24 -2
- package/lib/core/case.d.ts.map +1 -1
- package/lib/core/case.js +5 -1
- package/lib/core/case.js.map +1 -1
- package/lib/core/driver.d.ts +40 -0
- package/lib/core/driver.d.ts.map +1 -1
- package/lib/core/driver.js +12 -0
- package/lib/core/driver.js.map +1 -1
- package/lib/core/environment.d.ts +18 -0
- package/lib/core/environment.d.ts.map +1 -0
- package/lib/core/environment.js +21 -0
- package/lib/core/environment.js.map +1 -0
- package/lib/core/evaluation-clusters.d.ts +17 -0
- package/lib/core/evaluation-clusters.d.ts.map +1 -0
- package/lib/core/evaluation-clusters.js +23 -0
- package/lib/core/evaluation-clusters.js.map +1 -0
- package/lib/core/experiment.d.ts +11 -0
- package/lib/core/experiment.d.ts.map +1 -1
- package/lib/core/experiment.js +31 -17
- package/lib/core/experiment.js.map +1 -1
- package/lib/core/file-tree.d.ts +7 -0
- package/lib/core/file-tree.d.ts.map +1 -0
- package/lib/core/file-tree.js +35 -0
- package/lib/core/file-tree.js.map +1 -0
- package/lib/core/index.d.ts +1 -0
- package/lib/core/index.d.ts.map +1 -1
- package/lib/core/index.js +1 -0
- package/lib/core/index.js.map +1 -1
- package/lib/core/lifecycle-fixtures.d.ts.map +1 -1
- package/lib/core/lifecycle-fixtures.js +4 -2
- package/lib/core/lifecycle-fixtures.js.map +1 -1
- package/lib/core/platform-process.d.ts +5 -1
- package/lib/core/platform-process.d.ts.map +1 -1
- package/lib/core/platform-process.js +48 -10
- package/lib/core/platform-process.js.map +1 -1
- package/lib/core/run.d.ts +93 -1
- package/lib/core/run.d.ts.map +1 -1
- package/lib/core/run.js +16 -0
- package/lib/core/run.js.map +1 -1
- package/lib/core/usage.d.ts +5 -0
- package/lib/core/usage.d.ts.map +1 -1
- package/lib/core/usage.js +3 -0
- package/lib/core/usage.js.map +1 -1
- package/lib/drivers/claude-code/index.d.ts +65 -2
- package/lib/drivers/claude-code/index.d.ts.map +1 -1
- package/lib/drivers/claude-code/index.js +30 -3
- package/lib/drivers/claude-code/index.js.map +1 -1
- package/lib/drivers/codex/app-events.d.ts.map +1 -1
- package/lib/drivers/codex/app-events.js +41 -30
- package/lib/drivers/codex/app-events.js.map +1 -1
- package/lib/drivers/codex/exec-events.d.ts +4 -0
- package/lib/drivers/codex/exec-events.d.ts.map +1 -0
- package/lib/drivers/codex/exec-events.js +230 -0
- package/lib/drivers/codex/exec-events.js.map +1 -0
- package/lib/drivers/codex/index.d.ts +76 -4
- package/lib/drivers/codex/index.d.ts.map +1 -1
- package/lib/drivers/codex/index.js +35 -232
- package/lib/drivers/codex/index.js.map +1 -1
- package/lib/drivers/cursor/acp-transport.d.ts +38 -0
- package/lib/drivers/cursor/acp-transport.d.ts.map +1 -0
- package/lib/drivers/cursor/acp-transport.js +152 -0
- package/lib/drivers/cursor/acp-transport.js.map +1 -0
- package/lib/drivers/cursor/config.d.ts +22 -0
- package/lib/drivers/cursor/config.d.ts.map +1 -0
- package/lib/drivers/cursor/config.js +26 -0
- package/lib/drivers/cursor/config.js.map +1 -0
- package/lib/drivers/cursor/events.d.ts +13 -0
- package/lib/drivers/cursor/events.d.ts.map +1 -0
- package/lib/drivers/cursor/events.js +73 -0
- package/lib/drivers/cursor/events.js.map +1 -0
- package/lib/drivers/cursor/index.d.ts +37 -0
- package/lib/drivers/cursor/index.d.ts.map +1 -0
- package/lib/drivers/cursor/index.js +276 -0
- package/lib/drivers/cursor/index.js.map +1 -0
- package/lib/drivers/cursor/interaction.d.ts +7 -0
- package/lib/drivers/cursor/interaction.d.ts.map +1 -0
- package/lib/drivers/cursor/interaction.js +66 -0
- package/lib/drivers/cursor/interaction.js.map +1 -0
- package/lib/drivers/mcp/index.d.ts +35 -2
- package/lib/drivers/mcp/index.d.ts.map +1 -1
- package/lib/drivers/mcp/index.js +16 -3
- package/lib/drivers/mcp/index.js.map +1 -1
- package/lib/drivers/shared/experiment-behavior.d.ts +13 -0
- package/lib/drivers/shared/experiment-behavior.d.ts.map +1 -0
- package/lib/drivers/shared/experiment-behavior.js +197 -0
- package/lib/drivers/shared/experiment-behavior.js.map +1 -0
- package/lib/drivers/shared/experiment-policy.d.ts +4 -0
- package/lib/drivers/shared/experiment-policy.d.ts.map +1 -0
- package/lib/drivers/shared/experiment-policy.js +143 -0
- package/lib/drivers/shared/experiment-policy.js.map +1 -0
- package/lib/drivers/subprocess/index.d.ts +37 -2
- package/lib/drivers/subprocess/index.d.ts.map +1 -1
- package/lib/drivers/subprocess/index.js +46 -6
- package/lib/drivers/subprocess/index.js.map +1 -1
- package/lib/index.d.ts +6 -5
- package/lib/index.d.ts.map +1 -1
- package/lib/index.js +4 -3
- package/lib/index.js.map +1 -1
- package/lib/orchestrator/annotation-loader.d.ts +6 -2
- package/lib/orchestrator/annotation-loader.d.ts.map +1 -1
- package/lib/orchestrator/annotation-loader.js +8 -0
- package/lib/orchestrator/annotation-loader.js.map +1 -1
- package/lib/orchestrator/annotation-store.d.ts.map +1 -1
- package/lib/orchestrator/annotation-store.js +16 -2
- package/lib/orchestrator/annotation-store.js.map +1 -1
- package/lib/orchestrator/campaign-lifecycle.d.ts +1 -0
- package/lib/orchestrator/campaign-lifecycle.d.ts.map +1 -1
- package/lib/orchestrator/campaign-lifecycle.js +4 -1
- package/lib/orchestrator/campaign-lifecycle.js.map +1 -1
- package/lib/orchestrator/compare.d.ts +4 -0
- package/lib/orchestrator/compare.d.ts.map +1 -1
- package/lib/orchestrator/compare.js +13 -1
- package/lib/orchestrator/compare.js.map +1 -1
- package/lib/orchestrator/comparison-identity.d.ts.map +1 -1
- package/lib/orchestrator/comparison-identity.js +1 -0
- package/lib/orchestrator/comparison-identity.js.map +1 -1
- package/lib/orchestrator/config.d.ts +42 -0
- package/lib/orchestrator/config.d.ts.map +1 -1
- package/lib/orchestrator/config.js +6 -2
- package/lib/orchestrator/config.js.map +1 -1
- package/lib/orchestrator/discover.d.ts +1 -0
- package/lib/orchestrator/discover.d.ts.map +1 -1
- package/lib/orchestrator/discover.js +7 -1
- package/lib/orchestrator/discover.js.map +1 -1
- package/lib/orchestrator/doctor.d.ts.map +1 -1
- package/lib/orchestrator/doctor.js +34 -26
- package/lib/orchestrator/doctor.js.map +1 -1
- package/lib/orchestrator/driver-capabilities.d.ts +3 -0
- package/lib/orchestrator/driver-capabilities.d.ts.map +1 -0
- package/lib/orchestrator/driver-capabilities.js +10 -0
- package/lib/orchestrator/driver-capabilities.js.map +1 -0
- package/lib/orchestrator/driver-identity.d.ts +5 -0
- package/lib/orchestrator/driver-identity.d.ts.map +1 -0
- package/lib/orchestrator/driver-identity.js +41 -0
- package/lib/orchestrator/driver-identity.js.map +1 -0
- package/lib/orchestrator/experiment-context.d.ts +10 -0
- package/lib/orchestrator/experiment-context.d.ts.map +1 -1
- package/lib/orchestrator/experiment-contracts.d.ts +10 -0
- package/lib/orchestrator/experiment-contracts.d.ts.map +1 -1
- package/lib/orchestrator/experiment-corpus.d.ts.map +1 -1
- package/lib/orchestrator/experiment-corpus.js +17 -19
- package/lib/orchestrator/experiment-corpus.js.map +1 -1
- package/lib/orchestrator/experiment-effects.d.ts +33 -0
- package/lib/orchestrator/experiment-effects.d.ts.map +1 -0
- package/lib/orchestrator/experiment-effects.js +44 -0
- package/lib/orchestrator/experiment-effects.js.map +1 -0
- package/lib/orchestrator/experiment-evaluation.d.ts.map +1 -1
- package/lib/orchestrator/experiment-evaluation.js +36 -3
- package/lib/orchestrator/experiment-evaluation.js.map +1 -1
- package/lib/orchestrator/experiment-identity.d.ts.map +1 -1
- package/lib/orchestrator/experiment-identity.js +18 -29
- package/lib/orchestrator/experiment-identity.js.map +1 -1
- package/lib/orchestrator/experiment-loader.d.ts +7 -2
- package/lib/orchestrator/experiment-loader.d.ts.map +1 -1
- package/lib/orchestrator/experiment-report.d.ts +70 -0
- package/lib/orchestrator/experiment-report.d.ts.map +1 -1
- package/lib/orchestrator/experiment-report.js +9 -1
- package/lib/orchestrator/experiment-report.js.map +1 -1
- package/lib/orchestrator/experiment-runs.d.ts +22 -1
- package/lib/orchestrator/experiment-runs.d.ts.map +1 -1
- package/lib/orchestrator/experiment-runs.js +13 -5
- package/lib/orchestrator/experiment-runs.js.map +1 -1
- package/lib/orchestrator/experiment-runtime-identity.d.ts +0 -16
- package/lib/orchestrator/experiment-runtime-identity.d.ts.map +1 -1
- package/lib/orchestrator/experiment-runtime-identity.js +5 -220
- package/lib/orchestrator/experiment-runtime-identity.js.map +1 -1
- package/lib/orchestrator/experiment-selection.d.ts.map +1 -1
- package/lib/orchestrator/experiment-selection.js +52 -0
- package/lib/orchestrator/experiment-selection.js.map +1 -1
- package/lib/orchestrator/experiment.d.ts +0 -2
- package/lib/orchestrator/experiment.d.ts.map +1 -1
- package/lib/orchestrator/experiment.js +23 -15
- package/lib/orchestrator/experiment.js.map +1 -1
- package/lib/orchestrator/fs-read.d.ts +1 -6
- package/lib/orchestrator/fs-read.d.ts.map +1 -1
- package/lib/orchestrator/fs-read.js +2 -32
- package/lib/orchestrator/fs-read.js.map +1 -1
- package/lib/orchestrator/grader-experiment.d.ts +17 -1
- package/lib/orchestrator/grader-experiment.d.ts.map +1 -1
- package/lib/orchestrator/grader-experiment.js +52 -7
- package/lib/orchestrator/grader-experiment.js.map +1 -1
- package/lib/orchestrator/grading.d.ts.map +1 -1
- package/lib/orchestrator/grading.js +13 -6
- package/lib/orchestrator/grading.js.map +1 -1
- package/lib/orchestrator/list-runs.d.ts +1 -0
- package/lib/orchestrator/list-runs.d.ts.map +1 -1
- package/lib/orchestrator/list-runs.js +27 -6
- package/lib/orchestrator/list-runs.js.map +1 -1
- package/lib/orchestrator/optimization-lifecycle.d.ts.map +1 -1
- package/lib/orchestrator/optimization-lifecycle.js +3 -1
- package/lib/orchestrator/optimization-lifecycle.js.map +1 -1
- package/lib/orchestrator/orchestrator.d.ts +6 -1
- package/lib/orchestrator/orchestrator.d.ts.map +1 -1
- package/lib/orchestrator/orchestrator.js +17 -143
- package/lib/orchestrator/orchestrator.js.map +1 -1
- package/lib/orchestrator/paired-effects.d.ts +34 -0
- package/lib/orchestrator/paired-effects.d.ts.map +1 -0
- package/lib/orchestrator/paired-effects.js +158 -0
- package/lib/orchestrator/paired-effects.js.map +1 -0
- package/lib/orchestrator/provenance.d.ts +5 -1
- package/lib/orchestrator/provenance.d.ts.map +1 -1
- package/lib/orchestrator/provenance.js +16 -3
- package/lib/orchestrator/provenance.js.map +1 -1
- package/lib/orchestrator/regrade.d.ts +1 -0
- package/lib/orchestrator/regrade.d.ts.map +1 -1
- package/lib/orchestrator/run-execution.d.ts +2 -0
- package/lib/orchestrator/run-execution.d.ts.map +1 -1
- package/lib/orchestrator/run-execution.js +18 -3
- package/lib/orchestrator/run-execution.js.map +1 -1
- package/lib/orchestrator/run-lifecycle.d.ts +1 -0
- package/lib/orchestrator/run-lifecycle.d.ts.map +1 -1
- package/lib/orchestrator/run-lifecycle.js +3 -2
- package/lib/orchestrator/run-lifecycle.js.map +1 -1
- package/lib/orchestrator/suite-runs.d.ts +3 -1
- package/lib/orchestrator/suite-runs.d.ts.map +1 -1
- package/lib/orchestrator/suite-runs.js +12 -2
- package/lib/orchestrator/suite-runs.js.map +1 -1
- package/lib/orchestrator/suite-summary.d.ts.map +1 -1
- package/lib/orchestrator/suite-summary.js +2 -1
- package/lib/orchestrator/suite-summary.js.map +1 -1
- package/lib/previewer/server.d.ts +6 -0
- package/lib/previewer/server.d.ts.map +1 -1
- package/lib/previewer/server.js +21 -5
- package/lib/previewer/server.js.map +1 -1
- package/package.json +2 -2
- package/skills/lux/references/cli.md +5 -2
- package/skills/lux/references/eval-authoring.md +0 -7
- package/skills/lux-answerer/SKILL.md +1 -1
- package/src/answerers/claude-code/index.ts +3 -3
- package/src/answerers/persona/index.ts +3 -3
- package/src/answerers/terminal/index.ts +4 -9
- package/src/assertions/core/resource-checks.ts +3 -5
- package/src/assertions/rubric/index.ts +7 -1
- package/src/assertions/rubric/internal.ts +13 -8
- package/src/authoring.ts +331 -2
- package/src/cli/bin.ts +2 -12
- package/src/cli/command-runtime.ts +20 -9
- package/src/cli/init-templates.ts +223 -0
- package/src/cli/init.ts +38 -184
- package/src/cli/output-schemas.ts +106 -0
- package/src/cli/program.ts +621 -548
- package/src/cli/render/reporter.ts +25 -4
- package/src/cli/run-options.ts +8 -25
- package/src/cli/view.ts +45 -84
- package/src/core/annotation.ts +2 -1
- package/src/core/answerer.ts +9 -0
- package/src/core/assertion.ts +3 -0
- package/src/core/case.ts +5 -1
- package/src/core/driver.ts +38 -0
- package/src/core/environment.ts +22 -0
- package/src/core/evaluation-clusters.ts +35 -0
- package/src/core/experiment.ts +60 -23
- package/src/core/file-tree.ts +35 -0
- package/src/core/index.ts +1 -0
- package/src/core/lifecycle-fixtures.ts +4 -2
- package/src/core/platform-process.ts +54 -6
- package/src/core/run.ts +20 -0
- package/src/core/usage.ts +8 -0
- package/src/drivers/claude-code/index.ts +34 -3
- package/src/drivers/codex/app-events.ts +43 -33
- package/src/drivers/codex/exec-events.ts +258 -0
- package/src/drivers/codex/index.ts +37 -262
- package/src/drivers/cursor/README.md +57 -0
- package/src/drivers/cursor/acp-transport.ts +182 -0
- package/src/drivers/cursor/config.ts +29 -0
- package/src/drivers/cursor/events.ts +83 -0
- package/src/drivers/cursor/index.ts +316 -0
- package/src/drivers/cursor/interaction.ts +90 -0
- package/src/drivers/mcp/index.ts +16 -3
- package/src/drivers/shared/experiment-behavior.ts +214 -0
- package/src/drivers/shared/experiment-policy.ts +195 -0
- package/src/drivers/subprocess/index.ts +45 -6
- package/src/index.ts +16 -5
- package/src/orchestrator/annotation-loader.ts +10 -0
- package/src/orchestrator/annotation-store.ts +24 -3
- package/src/orchestrator/campaign-lifecycle.ts +5 -1
- package/src/orchestrator/compare.ts +16 -1
- package/src/orchestrator/comparison-identity.ts +1 -0
- package/src/orchestrator/config.ts +6 -2
- package/src/orchestrator/discover.ts +8 -1
- package/src/orchestrator/doctor.ts +40 -22
- package/src/orchestrator/driver-capabilities.ts +16 -0
- package/src/orchestrator/driver-identity.ts +59 -0
- package/src/orchestrator/experiment-corpus.ts +23 -28
- package/src/orchestrator/experiment-effects.ts +61 -0
- package/src/orchestrator/experiment-evaluation.ts +45 -5
- package/src/orchestrator/experiment-identity.ts +27 -36
- package/src/orchestrator/experiment-report.ts +21 -1
- package/src/orchestrator/experiment-runs.ts +11 -4
- package/src/orchestrator/experiment-runtime-identity.ts +5 -243
- package/src/orchestrator/experiment-selection.ts +77 -0
- package/src/orchestrator/experiment.ts +26 -24
- package/src/orchestrator/fs-read.ts +2 -32
- package/src/orchestrator/grader-experiment.ts +72 -8
- package/src/orchestrator/grading.ts +15 -6
- package/src/orchestrator/list-runs.ts +29 -6
- package/src/orchestrator/optimization-lifecycle.ts +4 -1
- package/src/orchestrator/orchestrator.ts +26 -183
- package/src/orchestrator/paired-effects.ts +209 -0
- package/src/orchestrator/provenance.ts +23 -3
- package/src/orchestrator/run-execution.ts +18 -3
- package/src/orchestrator/run-lifecycle.ts +4 -2
- package/src/orchestrator/suite-runs.ts +10 -1
- package/src/orchestrator/suite-summary.ts +2 -1
- package/src/previewer/server.ts +27 -5
- package/viewer/app.js +27 -6
- package/viewer/detail.js +22 -5
- package/lib/answerers/scripted/index.d.ts +0 -18
- package/lib/answerers/scripted/index.d.ts.map +0 -1
- package/lib/answerers/scripted/index.js +0 -63
- package/lib/answerers/scripted/index.js.map +0 -1
- package/lib/cli/skills.d.ts +0 -38
- package/lib/cli/skills.d.ts.map +0 -1
- package/lib/cli/skills.js +0 -84
- package/lib/cli/skills.js.map +0 -1
- package/src/answerers/scripted/index.ts +0 -78
- package/src/cli/skills.ts +0 -129
package/src/cli/program.ts
CHANGED
|
@@ -3,7 +3,8 @@ import { existsSync } from "node:fs";
|
|
|
3
3
|
import { lstat, mkdir, realpath, rename, rm, stat, writeFile } from "node:fs/promises";
|
|
4
4
|
import { createRequire } from "node:module";
|
|
5
5
|
import { basename, dirname, extname, join, relative, resolve, sep } from "node:path";
|
|
6
|
-
import {
|
|
6
|
+
import { fileURLToPath } from "node:url";
|
|
7
|
+
import { Cli, Errors, z } from "incur";
|
|
7
8
|
import { createRubricAssertion } from "../assertions/rubric/index.js";
|
|
8
9
|
import {
|
|
9
10
|
type Candidate,
|
|
@@ -17,14 +18,18 @@ import {
|
|
|
17
18
|
compareRuns,
|
|
18
19
|
compareSuiteRuns,
|
|
19
20
|
type DiscoveredCase,
|
|
21
|
+
DoctorReportSchema,
|
|
20
22
|
discoverCases,
|
|
21
23
|
doctorProject,
|
|
24
|
+
ExperimentCampaignPlanSchema,
|
|
25
|
+
ExperimentDecisionReportSchema,
|
|
26
|
+
ExperimentRunSchema,
|
|
22
27
|
experimentCampaignMode,
|
|
23
28
|
formatComparisonText,
|
|
24
29
|
formatDoctorReport,
|
|
25
30
|
formatExperimentDecisionMarkdown,
|
|
26
|
-
formatExperimentRun,
|
|
27
31
|
formatSuiteComparisonText,
|
|
32
|
+
GradingSnapshotSchema,
|
|
28
33
|
hasRegression,
|
|
29
34
|
hasSuiteRegression,
|
|
30
35
|
loadEvalCase,
|
|
@@ -55,500 +60,565 @@ import {
|
|
|
55
60
|
failWith,
|
|
56
61
|
relativePathWithin,
|
|
57
62
|
} from "./command-runtime.js";
|
|
58
|
-
import {
|
|
63
|
+
import { initProject } from "./init.js";
|
|
64
|
+
import { CompareOutputSchema, ViewOutputSchema } from "./output-schemas.js";
|
|
59
65
|
import { ConsoleReporter } from "./render/reporter.js";
|
|
60
66
|
import {
|
|
61
|
-
collectTags,
|
|
62
|
-
collectValues,
|
|
63
67
|
parseAnnotationAssertion,
|
|
64
|
-
parsePositiveInt,
|
|
65
|
-
parseUnitInterval,
|
|
66
68
|
type RunOptions,
|
|
67
69
|
resolveAnswererOverride,
|
|
68
70
|
resolveConcurrency,
|
|
69
71
|
} from "./run-options.js";
|
|
70
|
-
import {
|
|
71
|
-
type BundledSkillName,
|
|
72
|
-
formatSkillsInstall,
|
|
73
|
-
formatSkillsList,
|
|
74
|
-
installSkills,
|
|
75
|
-
type SkillsPlatform,
|
|
76
|
-
} from "./skills.js";
|
|
77
|
-
import { parsePreviewPort, viewCommand } from "./view.js";
|
|
72
|
+
import { type ViewOptions, viewCommand, webViewCommand } from "./view.js";
|
|
78
73
|
|
|
79
74
|
const requireFromHere = createRequire(import.meta.url);
|
|
80
75
|
// From lib/cli/program.js (or src/cli/program.ts under test) to the package root.
|
|
81
76
|
const PKG_VERSION = (requireFromHere("../../package.json") as { version: string }).version;
|
|
82
77
|
|
|
83
|
-
const
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
.
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
.
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
78
|
+
const nonEmptyString = z.string().min(1);
|
|
79
|
+
const positiveInteger = z.coerce.number().int().positive();
|
|
80
|
+
const unitInterval = z.coerce.number().min(0).max(1);
|
|
81
|
+
const tagList = z
|
|
82
|
+
.array(nonEmptyString)
|
|
83
|
+
.default([])
|
|
84
|
+
.describe("Repeatable list; each value may be comma-separated");
|
|
85
|
+
const normalizeTags = (values: string[]): string[] =>
|
|
86
|
+
values.flatMap((value) => value.split(",").map((tag) => tag.trim())).filter(Boolean);
|
|
87
|
+
const initOutput = z.object({
|
|
88
|
+
created: z.array(z.string()),
|
|
89
|
+
updated: z.array(z.string()),
|
|
90
|
+
skipped: z.array(z.object({ path: z.string(), reason: z.string() })),
|
|
91
|
+
experiment: z.boolean(),
|
|
92
|
+
});
|
|
93
|
+
const runOutput = z.union([
|
|
94
|
+
z.object({
|
|
95
|
+
caseCount: z.number().int(),
|
|
96
|
+
cases: z.array(z.object({ path: z.string(), id: z.string() })),
|
|
97
|
+
}),
|
|
98
|
+
z.object({
|
|
99
|
+
caseCount: z.number().int(),
|
|
100
|
+
tags: z.array(z.object({ tag: z.string(), count: z.number().int() })),
|
|
101
|
+
}),
|
|
102
|
+
z.object({
|
|
103
|
+
suiteDir: z.string(),
|
|
104
|
+
results: z.array(
|
|
105
|
+
z.object({
|
|
106
|
+
caseId: z.string(),
|
|
107
|
+
runDir: z.string().nullable(),
|
|
108
|
+
exitReason: z.string(),
|
|
109
|
+
casePassed: z.boolean().nullable(),
|
|
110
|
+
caseScore: z.number().nullable(),
|
|
111
|
+
assertionsPassed: z.number().int(),
|
|
112
|
+
assertionsTotal: z.number().int(),
|
|
113
|
+
}),
|
|
114
|
+
),
|
|
115
|
+
}),
|
|
116
|
+
]);
|
|
117
|
+
const experimentOutput = z.union([
|
|
118
|
+
ExperimentCampaignPlanSchema,
|
|
119
|
+
ExperimentRunSchema.extend({ experimentDir: z.string() }),
|
|
120
|
+
]);
|
|
121
|
+
const gradingOutput = GradingSnapshotSchema.extend({ gradingPath: z.string() });
|
|
122
|
+
const reportOutput = z.union([
|
|
123
|
+
ExperimentDecisionReportSchema,
|
|
124
|
+
z.object({ path: z.string(), report: ExperimentDecisionReportSchema }),
|
|
125
|
+
]);
|
|
126
|
+
|
|
127
|
+
const campaignArgs = z.object({ campaign: nonEmptyString.describe("Experiment campaign path") });
|
|
128
|
+
const campaignOptions = z.object({
|
|
129
|
+
runsRoot: nonEmptyString.optional().describe("Run directory root overriding lux.config.ts"),
|
|
130
|
+
resume: nonEmptyString.optional().describe("Interrupted experiment directory to resume"),
|
|
131
|
+
plan: z.boolean().optional().describe("Validate and estimate without calling models"),
|
|
132
|
+
});
|
|
133
|
+
|
|
134
|
+
export const buildCli = () =>
|
|
135
|
+
Cli.create("lux", {
|
|
136
|
+
description:
|
|
137
|
+
"Coding-agent evaluation and improvement with deterministic assertions and human-in-the-loop runs",
|
|
138
|
+
version: PKG_VERSION,
|
|
139
|
+
sync: {
|
|
140
|
+
include: [fileURLToPath(new URL("../../skills/*", import.meta.url))],
|
|
141
|
+
depth: 2,
|
|
142
|
+
suggestions: [
|
|
143
|
+
"Use Lux to inspect the eval cases in this project",
|
|
144
|
+
"Use Lux to diagnose the latest failed eval run",
|
|
145
|
+
"Use Lux to plan an evidence-based improvement experiment",
|
|
146
|
+
],
|
|
147
|
+
},
|
|
148
|
+
mcp: {
|
|
149
|
+
title: "Lux coding-agent evaluation",
|
|
150
|
+
instructions:
|
|
151
|
+
"Inspect and plan before running model-spending commands. Apply only a verified promoted experiment champion.",
|
|
152
|
+
},
|
|
153
|
+
})
|
|
154
|
+
.command("init", {
|
|
155
|
+
description: "Scaffold a new eval project in the current directory",
|
|
156
|
+
options: z.object({
|
|
157
|
+
force: z.boolean().optional().describe("Overwrite existing files"),
|
|
158
|
+
experiment: z.boolean().optional().describe("Also scaffold an optimization campaign"),
|
|
159
|
+
driver: z
|
|
160
|
+
.enum(["subprocess", "claude-code", "codex", "cursor"])
|
|
161
|
+
.optional()
|
|
162
|
+
.describe("Starter driver (default: deterministic subprocess)"),
|
|
163
|
+
model: nonEmptyString.optional().describe("Explicit subject model; required for Cursor"),
|
|
164
|
+
}),
|
|
165
|
+
output: initOutput,
|
|
166
|
+
destructive: true,
|
|
167
|
+
mcp: { annotations: { destructiveHint: true, openWorldHint: false } },
|
|
168
|
+
examples: [
|
|
169
|
+
{ description: "Scaffold a basic eval project" },
|
|
170
|
+
{ options: { experiment: true }, description: "Scaffold an optimization project" },
|
|
171
|
+
],
|
|
172
|
+
run: async (c) =>
|
|
173
|
+
c.ok(await initCommand(c.options), { cta: { commands: ["doctor", "run --list"] } }),
|
|
174
|
+
})
|
|
175
|
+
.command("doctor", {
|
|
176
|
+
description: "Preflight project configuration, CLIs, cases, campaigns, and budgets",
|
|
177
|
+
args: z.object({ campaigns: z.array(z.string()).default([]).describe("Campaign paths") }),
|
|
178
|
+
output: DoctorReportSchema,
|
|
179
|
+
mcp: {
|
|
180
|
+
annotations: { readOnlyHint: true, destructiveHint: false, openWorldHint: false },
|
|
181
|
+
},
|
|
182
|
+
run: async (c) =>
|
|
183
|
+
c.ok(await doctorCommand(c.args.campaigns), { cta: { commands: ["run --list"] } }),
|
|
184
|
+
})
|
|
185
|
+
.command("run", {
|
|
186
|
+
description: "Discover and run cases, filtered by name, path, or tag",
|
|
187
|
+
args: z.object({ filters: z.array(z.string()).default([]).describe("Case names or paths") }),
|
|
188
|
+
options: z.object({
|
|
189
|
+
tag: tagList.describe("Required case tag; repeatable and comma-separated"),
|
|
190
|
+
loop: positiveInteger.optional().describe("Run each matched eval this many times"),
|
|
191
|
+
concurrency: positiveInteger.optional().describe("Maximum cases running concurrently"),
|
|
192
|
+
list: z.boolean().optional().describe("List matching cases without running"),
|
|
193
|
+
listTags: z.boolean().optional().describe("List tags across matching cases"),
|
|
194
|
+
interactive: z.boolean().optional().describe("Answer agent questions in this terminal"),
|
|
195
|
+
claudeAnswerer: z.boolean().optional().describe("Use Claude Code as the answerer"),
|
|
196
|
+
skipJudge: z.boolean().optional().describe("Skip rubric LLM judge assertions"),
|
|
197
|
+
runsRoot: nonEmptyString.optional().describe("Run directory root"),
|
|
198
|
+
profile: nonEmptyString.optional().describe("Named driver and answerer profile"),
|
|
199
|
+
subjectRoot: nonEmptyString.optional().describe("Root bound into subjectPath values"),
|
|
200
|
+
verbose: z.boolean().optional().describe("Include phase timings and run directories"),
|
|
201
|
+
}),
|
|
202
|
+
alias: { tag: "t", concurrency: "j", interactive: "i", verbose: "v" },
|
|
203
|
+
output: runOutput,
|
|
204
|
+
mcp: { annotations: { readOnlyHint: false, destructiveHint: false, openWorldHint: true } },
|
|
205
|
+
examples: [
|
|
206
|
+
{ options: { list: true, tag: [] }, description: "List all discovered cases" },
|
|
207
|
+
{
|
|
208
|
+
args: { filters: ["smoke"] },
|
|
209
|
+
options: { tag: [] },
|
|
210
|
+
description: "Run matching cases",
|
|
122
211
|
},
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
}
|
|
130
|
-
const cwd = process.cwd();
|
|
131
|
-
const installed = await installSkills({
|
|
132
|
-
cwd,
|
|
133
|
-
platform: platform as SkillsPlatform,
|
|
134
|
-
project: options.project ?? false,
|
|
135
|
-
...(options.dir ? { dir: options.dir } : {}),
|
|
136
|
-
...(options.name ? { name: options.name as BundledSkillName } : {}),
|
|
212
|
+
],
|
|
213
|
+
run: async (c) => {
|
|
214
|
+
const result = await runCommand(c.args.filters, c.options as RunOptions, !c.agent);
|
|
215
|
+
const suiteDir = "suiteDir" in result ? result.suiteDir : undefined;
|
|
216
|
+
return c.ok(result, {
|
|
217
|
+
cta: { commands: suiteDir ? [{ command: "view", args: { runDir: suiteDir } }] : [] },
|
|
137
218
|
});
|
|
138
|
-
process.stdout.write(`${formatSkillsInstall(installed, cwd)}\n`);
|
|
139
219
|
},
|
|
140
|
-
)
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
220
|
+
})
|
|
221
|
+
.command("experiment", {
|
|
222
|
+
description: "Evaluate controlled source variants across cases and model profiles",
|
|
223
|
+
args: campaignArgs,
|
|
224
|
+
options: campaignOptions,
|
|
225
|
+
output: experimentOutput,
|
|
226
|
+
mcp: { annotations: { readOnlyHint: false, destructiveHint: false, openWorldHint: true } },
|
|
227
|
+
run: async (c) => experimentCommand("experiment", c.args.campaign, c.options),
|
|
228
|
+
})
|
|
229
|
+
.command("optimize", {
|
|
230
|
+
description: "Evolve candidates from grading feedback and promote on held-out cases",
|
|
231
|
+
args: campaignArgs,
|
|
232
|
+
options: campaignOptions,
|
|
233
|
+
output: experimentOutput,
|
|
234
|
+
destructive: true,
|
|
235
|
+
mcp: { annotations: { readOnlyHint: false, destructiveHint: true, openWorldHint: true } },
|
|
236
|
+
run: async (c) => experimentCommand("optimize", c.args.campaign, c.options),
|
|
237
|
+
})
|
|
238
|
+
.command("grade", {
|
|
239
|
+
description: "Re-run current graders against a persisted run",
|
|
240
|
+
args: z.object({ runDir: nonEmptyString.describe("Persisted run directory") }),
|
|
241
|
+
options: z.object({ case: nonEmptyString.optional().describe("Current case definition") }),
|
|
242
|
+
output: gradingOutput,
|
|
243
|
+
mcp: { annotations: { readOnlyHint: false, destructiveHint: false, openWorldHint: true } },
|
|
244
|
+
run: async (c) => gradeCommand(c.args.runDir, c.options),
|
|
245
|
+
})
|
|
246
|
+
.command("annotate", {
|
|
247
|
+
description: "Add or update a human grader-alignment label",
|
|
248
|
+
args: z.object({ runDir: nonEmptyString.describe("Persisted run directory") }),
|
|
249
|
+
options: z.object({
|
|
250
|
+
out: nonEmptyString.describe("JSON annotation dataset to create or update"),
|
|
251
|
+
rubric: nonEmptyString.describe("Candidate-relative declarative rubric path"),
|
|
252
|
+
id: nonEmptyString.optional().describe("Stable annotation ID"),
|
|
253
|
+
label: z.enum(["pass", "fail"]).optional().describe("Expected overall verdict"),
|
|
254
|
+
score: unitInterval.optional().describe("Expected score from zero to one"),
|
|
255
|
+
assertion: z
|
|
256
|
+
.array(nonEmptyString)
|
|
257
|
+
.default([])
|
|
258
|
+
.describe("Assertion label id=pass|fail[:score]"),
|
|
259
|
+
split: z.enum(["train", "validation", "test"]).optional().describe("Dataset split"),
|
|
260
|
+
tag: tagList.describe("Selection tag; repeatable and comma-separated"),
|
|
261
|
+
feedback: nonEmptyString.optional().describe("Human rationale for mismatches"),
|
|
262
|
+
annotator: nonEmptyString.optional().describe("Label author"),
|
|
263
|
+
reviewer: nonEmptyString.optional().describe("Independent reviewer"),
|
|
264
|
+
source: z.enum(["human", "synthetic"]).default("human").describe("Label provenance"),
|
|
265
|
+
}),
|
|
266
|
+
output: z.object({ id: z.string(), path: z.string(), labels: z.number().int() }),
|
|
267
|
+
destructive: true,
|
|
268
|
+
mcp: {
|
|
269
|
+
annotations: {
|
|
270
|
+
readOnlyHint: false,
|
|
271
|
+
destructiveHint: false,
|
|
272
|
+
idempotentHint: true,
|
|
273
|
+
openWorldHint: false,
|
|
274
|
+
},
|
|
275
|
+
},
|
|
276
|
+
run: async (c) => annotateCommand(c.args.runDir, c.options),
|
|
277
|
+
})
|
|
278
|
+
.command("apply", {
|
|
279
|
+
description: "Apply the verified promoted experiment champion",
|
|
280
|
+
args: z.object({
|
|
281
|
+
experimentDir: nonEmptyString.describe("Experiment directory"),
|
|
282
|
+
candidateId: nonEmptyString.optional().describe("Expected promoted candidate ID"),
|
|
283
|
+
}),
|
|
284
|
+
output: z.object({
|
|
285
|
+
candidateId: z.string(),
|
|
286
|
+
subjectRoot: z.string(),
|
|
287
|
+
sourceBytes: z.number(),
|
|
288
|
+
}),
|
|
289
|
+
destructive: true,
|
|
290
|
+
mcp: {
|
|
291
|
+
annotations: { readOnlyHint: false, destructiveHint: true, openWorldHint: false },
|
|
292
|
+
},
|
|
293
|
+
run: async (c) => applyCommand(c.args.experimentDir, c.args.candidateId),
|
|
294
|
+
})
|
|
295
|
+
.command("report", {
|
|
296
|
+
description: "Render a PR-ready report from a verified experiment record",
|
|
297
|
+
args: z.object({ experimentDir: nonEmptyString.describe("Experiment directory") }),
|
|
298
|
+
options: z.object({
|
|
299
|
+
out: nonEmptyString.optional().describe("Write Markdown to this path"),
|
|
300
|
+
force: z.boolean().optional().describe("Overwrite an existing output file"),
|
|
301
|
+
}),
|
|
302
|
+
output: reportOutput,
|
|
303
|
+
mcp: { annotations: { readOnlyHint: false, destructiveHint: false, openWorldHint: false } },
|
|
304
|
+
run: async (c) => reportCommand(c.args.experimentDir, c.options),
|
|
305
|
+
})
|
|
306
|
+
.command("view", {
|
|
307
|
+
description: "List past runs or show one run in detail",
|
|
308
|
+
args: z.object({ runDir: nonEmptyString.optional().describe("Run or suite directory") }),
|
|
309
|
+
options: z.object({
|
|
310
|
+
runsRoot: nonEmptyString.optional().describe("Run directory root"),
|
|
311
|
+
case: nonEmptyString.optional().describe("Filter lists to one case slug"),
|
|
312
|
+
suites: z.boolean().optional().describe("List logical suite runs"),
|
|
313
|
+
experiments: z.boolean().optional().describe("List experiment runs"),
|
|
314
|
+
web: z.boolean().optional().describe("Open the Lux Review web app"),
|
|
315
|
+
port: z.coerce.number().int().min(0).max(65535).optional().describe("Review server port"),
|
|
316
|
+
open: z.boolean().default(true).describe("Open a browser in web mode"),
|
|
317
|
+
}),
|
|
318
|
+
output: ViewOutputSchema,
|
|
319
|
+
mcp: { annotations: { readOnlyHint: true, destructiveHint: false, openWorldHint: false } },
|
|
320
|
+
run: (c) => {
|
|
321
|
+
if (c.options.web) {
|
|
322
|
+
if (c.formatExplicit && c.format !== "jsonl") {
|
|
323
|
+
throw new Errors.ParseError({
|
|
324
|
+
message: "--web requires streaming output; omit --format or use --format jsonl",
|
|
325
|
+
});
|
|
326
|
+
}
|
|
327
|
+
return webViewCommand(c.args.runDir, c.options as ViewOptions);
|
|
328
|
+
}
|
|
329
|
+
return viewCommand(c.args.runDir, c.options as ViewOptions).then((result) =>
|
|
330
|
+
ViewOutputSchema.parse(result),
|
|
331
|
+
);
|
|
332
|
+
},
|
|
333
|
+
})
|
|
334
|
+
.command("compare", {
|
|
335
|
+
description: "Diff two run or suite directories and fail on regression",
|
|
336
|
+
args: z.object({
|
|
337
|
+
runA: nonEmptyString.describe("Baseline run"),
|
|
338
|
+
runB: nonEmptyString.describe("Current run"),
|
|
339
|
+
}),
|
|
340
|
+
options: z.object({
|
|
341
|
+
passRateTolerance: unitInterval.optional().describe("Allowed per-case pass-rate decrease"),
|
|
342
|
+
}),
|
|
343
|
+
output: CompareOutputSchema,
|
|
344
|
+
mcp: { annotations: { readOnlyHint: true, destructiveHint: false, openWorldHint: false } },
|
|
345
|
+
run: async (c) =>
|
|
346
|
+
CompareOutputSchema.parse(await compareCommand(c.args.runA, c.args.runB, c.options)),
|
|
154
347
|
});
|
|
155
348
|
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
.description("discover and run cases; filter by name/path substring or --tag")
|
|
159
|
-
.option(
|
|
160
|
-
"-t, --tag <tag>",
|
|
161
|
-
"only cases whose meta.tags include this tag (repeatable, comma-separated)",
|
|
162
|
-
collectTags,
|
|
163
|
-
[],
|
|
164
|
-
)
|
|
165
|
-
.option("--loop <n>", "run each matched eval n times", parsePositiveInt)
|
|
166
|
-
.option(
|
|
167
|
-
"-j, --concurrency <n>",
|
|
168
|
-
"run up to n cases at once (overrides lux.config.ts)",
|
|
169
|
-
parsePositiveInt,
|
|
170
|
-
)
|
|
171
|
-
.option("--list", "print the matched cases and exit without running them")
|
|
172
|
-
.option("--list-tags", "print the tags across the matched cases and exit")
|
|
173
|
-
.option("-i, --interactive", "answer agent questions yourself in this terminal (HITL)")
|
|
174
|
-
.option(
|
|
175
|
-
"--claude-answerer",
|
|
176
|
-
"delegate questions through the Claude plugin's lux-answerer skill",
|
|
177
|
-
)
|
|
178
|
-
.option("--skip-judge", "skip rubric LLM judge assertions")
|
|
179
|
-
.option("--runs-root <path>", "run-dir root (overrides lux.config.ts)")
|
|
180
|
-
.option("--profile <id>", "named driver/answerer profile from lux.config.ts")
|
|
181
|
-
.option("--subject-root <path>", "root bound into subjectPath() values")
|
|
182
|
-
.option("-v, --verbose", "add phase timings and the run dir to every case")
|
|
183
|
-
.action(async (filters: string[], options: RunOptions) => {
|
|
184
|
-
await runCommand(filters, options);
|
|
185
|
-
});
|
|
349
|
+
const initCommand = async (options: Omit<Parameters<typeof initProject>[0], "cwd">) =>
|
|
350
|
+
initProject({ cwd: process.cwd(), ...options });
|
|
186
351
|
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
options: { runsRoot?: string; resume?: string; json?: boolean; plan?: boolean },
|
|
198
|
-
) => {
|
|
199
|
-
await experimentCommand("experiment", campaign, options);
|
|
200
|
-
},
|
|
201
|
-
);
|
|
352
|
+
const doctorCommand = async (campaigns: string[]) => {
|
|
353
|
+
const report = await doctorProject({ cwd: process.cwd(), campaigns });
|
|
354
|
+
if (!report.ok) {
|
|
355
|
+
throw new Errors.IncurError({
|
|
356
|
+
code: "PREFLIGHT_FAILED",
|
|
357
|
+
message: formatDoctorReport(report),
|
|
358
|
+
exitCode: 1,
|
|
359
|
+
});
|
|
360
|
+
}
|
|
361
|
+
return report;
|
|
202
362
|
};
|
|
203
363
|
|
|
204
|
-
const
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
const report = result.snapshot.report;
|
|
223
|
-
let gradingStatus = "ungraded";
|
|
224
|
-
if (report.casePassed === true) gradingStatus = "PASS";
|
|
225
|
-
if (report.casePassed === false) gradingStatus = "FAIL";
|
|
226
|
-
let output: string;
|
|
227
|
-
if (options.json) {
|
|
228
|
-
output = `${JSON.stringify({ gradingPath: result.path, ...result.snapshot }, null, 2)}\n`;
|
|
229
|
-
} else {
|
|
230
|
-
const score = report.caseScore === null ? "—" : `${(report.caseScore * 100).toFixed(0)}%`;
|
|
231
|
-
output = [
|
|
232
|
-
`grading: ${gradingStatus}`,
|
|
233
|
-
`score: ${score}`,
|
|
234
|
-
`checks: ${report.assertions.filter((assertion) => assertion.passed).length}/${report.assertions.length}`,
|
|
235
|
-
`saved: ${result.path}`,
|
|
236
|
-
"",
|
|
237
|
-
].join("\n");
|
|
238
|
-
}
|
|
239
|
-
process.stdout.write(output);
|
|
240
|
-
if (controller.signal.aborted) failWith(SIGINT_EXIT_CODE);
|
|
241
|
-
else if (report.casePassed !== true) failWith(1);
|
|
364
|
+
const gradeCommand = async (runDir: string, options: { case?: string | undefined }) => {
|
|
365
|
+
const directory = resolve(runDir);
|
|
366
|
+
if (!(await ensureDirectory("grade", directory))) failWith(1);
|
|
367
|
+
const config = await loadLuxConfig(process.cwd());
|
|
368
|
+
using controller = abortOnInterrupt();
|
|
369
|
+
const result = await regradeRun({
|
|
370
|
+
runDir: directory,
|
|
371
|
+
registry: buildRegistryFromConfig(config),
|
|
372
|
+
graderContext: config.harness ?? null,
|
|
373
|
+
...(options.case ? { case: await loadEvalCase(resolve(options.case)) } : {}),
|
|
374
|
+
abortSignal: controller.signal,
|
|
375
|
+
});
|
|
376
|
+
if (controller.signal.aborted) failWith(SIGINT_EXIT_CODE);
|
|
377
|
+
if (result.snapshot.report.casePassed !== true) {
|
|
378
|
+
throw new Errors.IncurError({
|
|
379
|
+
code: "GRADING_FAILED",
|
|
380
|
+
message: `The persisted run did not pass regrading. Grading saved to ${result.path}.`,
|
|
381
|
+
exitCode: 1,
|
|
242
382
|
});
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
.command("annotate <runDir>")
|
|
246
|
-
.description("add or update a human grader-alignment label for a persisted run")
|
|
247
|
-
.requiredOption("--out <path>", "JSON annotation dataset to create or update")
|
|
248
|
-
.requiredOption("--rubric <path>", "candidate-relative declarative .json rubric path")
|
|
249
|
-
.option("--id <id>", "stable annotation id")
|
|
250
|
-
.option("--label <pass|fail>", "expected overall case verdict")
|
|
251
|
-
.option("--score <n>", "expected overall case score from 0 to 1", parseUnitInterval)
|
|
252
|
-
.option(
|
|
253
|
-
"--assertion <id=pass|fail[:score]>",
|
|
254
|
-
"expected stable assertion verdict and optional score (repeatable)",
|
|
255
|
-
collectValues,
|
|
256
|
-
[],
|
|
257
|
-
)
|
|
258
|
-
.option("--split <train|validation|test>", "dataset split, stored as a tag")
|
|
259
|
-
.option("--tag <tag>", "extra selection tag (repeatable, comma-separated)", collectTags, [])
|
|
260
|
-
.option("--feedback <text>", "human rationale shown to the rubric optimizer on mismatch")
|
|
261
|
-
.option("--annotator <id>", "person or system that authored the label")
|
|
262
|
-
.option("--reviewer <id>", "independent reviewer of the label")
|
|
263
|
-
.option("--source <human|synthetic>", "label provenance", "human")
|
|
264
|
-
.action(
|
|
265
|
-
async (
|
|
266
|
-
runDir: string,
|
|
267
|
-
options: {
|
|
268
|
-
out: string;
|
|
269
|
-
rubric: string;
|
|
270
|
-
id?: string;
|
|
271
|
-
label?: string;
|
|
272
|
-
score?: number;
|
|
273
|
-
assertion: string[];
|
|
274
|
-
split?: string;
|
|
275
|
-
tag: string[];
|
|
276
|
-
feedback?: string;
|
|
277
|
-
annotator?: string;
|
|
278
|
-
reviewer?: string;
|
|
279
|
-
source: string;
|
|
280
|
-
},
|
|
281
|
-
) => {
|
|
282
|
-
const directory = resolve(runDir);
|
|
283
|
-
if (!(await ensureDirectory("annotate", directory))) return;
|
|
284
|
-
if (options.label && options.label !== "pass" && options.label !== "fail") {
|
|
285
|
-
throw new InvalidArgumentError("--label must be pass or fail");
|
|
286
|
-
}
|
|
287
|
-
if (options.split && !["train", "validation", "test"].includes(options.split)) {
|
|
288
|
-
throw new InvalidArgumentError("--split must be train, validation, or test");
|
|
289
|
-
}
|
|
290
|
-
if (options.source !== "human" && options.source !== "synthetic") {
|
|
291
|
-
throw new InvalidArgumentError("--source must be human or synthetic");
|
|
292
|
-
}
|
|
293
|
-
const parsedAssertions = options.assertion.map(parseAnnotationAssertion);
|
|
294
|
-
const assertionIds = parsedAssertions.map(([id]) => id);
|
|
295
|
-
if (new Set(assertionIds).size !== assertionIds.length) {
|
|
296
|
-
throw new InvalidArgumentError("--assertion ids must not be repeated");
|
|
297
|
-
}
|
|
298
|
-
const assertions = Object.fromEntries(parsedAssertions);
|
|
299
|
-
if (
|
|
300
|
-
options.label === undefined &&
|
|
301
|
-
options.score === undefined &&
|
|
302
|
-
Object.keys(assertions).length === 0
|
|
303
|
-
) {
|
|
304
|
-
throw new Error("lux annotate: provide --label, --score, or at least one --assertion");
|
|
305
|
-
}
|
|
306
|
-
const out = resolve(options.out);
|
|
307
|
-
const run = await loadFrozenRegradableRun(directory);
|
|
308
|
-
try {
|
|
309
|
-
const rubric = await loadRubricDefinition(
|
|
310
|
-
await resolveAuthoredFile(options.rubric, "rubric"),
|
|
311
|
-
);
|
|
312
|
-
if (rubric.id !== run.case.id) {
|
|
313
|
-
throw new Error(
|
|
314
|
-
`lux annotate: rubric id '${rubric.id}' does not match persisted case '${run.case.id}'`,
|
|
315
|
-
);
|
|
316
|
-
}
|
|
317
|
-
const rubricIds = new Set(rubric.assertions.map((assertion) => assertion.id));
|
|
318
|
-
const unknownIds = assertionIds.filter((assertionId) => !rubricIds.has(assertionId));
|
|
319
|
-
if (unknownIds.length > 0) {
|
|
320
|
-
throw new Error(`lux annotate: unknown rubric assertion ids: ${unknownIds.join(", ")}`);
|
|
321
|
-
}
|
|
322
|
-
const tags = [...(options.split ? [options.split] : []), ...options.tag].filter(
|
|
323
|
-
(tag, index, all) => all.indexOf(tag) === index,
|
|
324
|
-
);
|
|
325
|
-
const caseId = run.case.id;
|
|
326
|
-
const portableRunDir = (relative(dirname(out), directory) || ".").split(sep).join("/");
|
|
327
|
-
const id =
|
|
328
|
-
options.id ??
|
|
329
|
-
`${caseId}-${basename(directory)}-${valueHash(portableRunDir).slice(0, 10)}`.replace(
|
|
330
|
-
/[^a-zA-Z0-9_-]+/g,
|
|
331
|
-
"-",
|
|
332
|
-
);
|
|
333
|
-
const annotation = GraderAnnotationSchema.parse({
|
|
334
|
-
id,
|
|
335
|
-
runDir: portableRunDir,
|
|
336
|
-
rubricPath: options.rubric,
|
|
337
|
-
tags,
|
|
338
|
-
source: options.source,
|
|
339
|
-
...(options.annotator ? { annotator: options.annotator } : {}),
|
|
340
|
-
labeledAt: new Date().toISOString(),
|
|
341
|
-
...(options.reviewer ? { reviewer: options.reviewer } : {}),
|
|
342
|
-
expected: {
|
|
343
|
-
...(options.label ? { casePassed: options.label === "pass" } : {}),
|
|
344
|
-
...(options.score !== undefined ? { caseScore: options.score } : {}),
|
|
345
|
-
assertions,
|
|
346
|
-
},
|
|
347
|
-
...(options.feedback ? { feedback: options.feedback } : {}),
|
|
348
|
-
});
|
|
349
|
-
const dataset = await upsertGraderAnnotation(out, annotation);
|
|
350
|
-
process.stdout.write(
|
|
351
|
-
`annotated ${annotation.id} in ${out} (${dataset.annotations.length} labels)\n`,
|
|
352
|
-
);
|
|
353
|
-
} finally {
|
|
354
|
-
await releaseFrozenRegradableRun(run);
|
|
355
|
-
}
|
|
356
|
-
},
|
|
357
|
-
);
|
|
383
|
+
}
|
|
384
|
+
return { gradingPath: result.path, ...result.snapshot };
|
|
358
385
|
};
|
|
359
386
|
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
},
|
|
375
|
-
);
|
|
387
|
+
type AnnotationOptions = {
|
|
388
|
+
out: string;
|
|
389
|
+
rubric: string;
|
|
390
|
+
id?: string | undefined;
|
|
391
|
+
label?: "pass" | "fail" | undefined;
|
|
392
|
+
score?: number | undefined;
|
|
393
|
+
assertion: string[];
|
|
394
|
+
split?: "train" | "validation" | "test" | undefined;
|
|
395
|
+
tag: string[];
|
|
396
|
+
feedback?: string | undefined;
|
|
397
|
+
annotator?: string | undefined;
|
|
398
|
+
reviewer?: string | undefined;
|
|
399
|
+
source: "human" | "synthetic";
|
|
400
|
+
};
|
|
376
401
|
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
return;
|
|
394
|
-
}
|
|
395
|
-
const championId = experiment.championCandidateId;
|
|
396
|
-
if (!championId) {
|
|
397
|
-
process.stderr.write("lux apply: passed experiment has no promoted champion\n");
|
|
398
|
-
failWith(1);
|
|
399
|
-
return;
|
|
400
|
-
}
|
|
401
|
-
if (candidateId && candidateId !== championId) {
|
|
402
|
-
process.stderr.write(
|
|
403
|
-
`lux apply: candidate '${candidateId}' is not the promoted champion '${championId}'\n`,
|
|
404
|
-
);
|
|
405
|
-
failWith(1);
|
|
406
|
-
return;
|
|
407
|
-
}
|
|
408
|
-
let candidate: Candidate | undefined;
|
|
409
|
-
try {
|
|
410
|
-
const candidates = await verifyExperimentRunIntegrity(directory, experiment);
|
|
411
|
-
verifyExperimentDecisionIntegrity(experiment);
|
|
412
|
-
candidate = candidates.find((entry) => entry.id === championId);
|
|
413
|
-
if (!candidate) {
|
|
414
|
-
throw new Error(`champion candidate '${championId}' not found`);
|
|
415
|
-
}
|
|
416
|
-
} catch (error) {
|
|
417
|
-
const message = error instanceof Error ? error.message : String(error);
|
|
418
|
-
process.stderr.write(`lux apply: experiment integrity check failed: ${message}\n`);
|
|
419
|
-
failWith(1);
|
|
420
|
-
return;
|
|
421
|
-
}
|
|
422
|
-
await applyCandidate(experiment.subjectRoot, experiment.campaignSnapshot.subject, candidate);
|
|
423
|
-
process.stdout.write(
|
|
424
|
-
`applied candidate ${candidate.id} to ${experiment.subjectRoot} (${candidate.sourceBytes} bytes)\n`,
|
|
425
|
-
);
|
|
402
|
+
const annotateCommand = async (runDir: string, options: AnnotationOptions) => {
|
|
403
|
+
const directory = resolve(runDir);
|
|
404
|
+
if (!(await ensureDirectory("annotate", directory))) failWith(1);
|
|
405
|
+
const parsedAssertions = options.assertion.map(parseAnnotationAssertion);
|
|
406
|
+
const assertionIds = parsedAssertions.map(([id]) => id);
|
|
407
|
+
if (new Set(assertionIds).size !== assertionIds.length) {
|
|
408
|
+
throw new Errors.ParseError({ message: "--assertion IDs must not be repeated" });
|
|
409
|
+
}
|
|
410
|
+
const assertions = Object.fromEntries(parsedAssertions);
|
|
411
|
+
if (
|
|
412
|
+
options.label === undefined &&
|
|
413
|
+
options.score === undefined &&
|
|
414
|
+
Object.keys(assertions).length === 0
|
|
415
|
+
) {
|
|
416
|
+
throw new Errors.ParseError({
|
|
417
|
+
message: "Provide --label, --score, or at least one --assertion",
|
|
426
418
|
});
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
.
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
if (outputStats && !options.force) {
|
|
467
|
-
throw new Error(`lux report: output already exists (use --force): ${out}`);
|
|
468
|
-
}
|
|
469
|
-
await mkdir(dirname(out), { recursive: true });
|
|
470
|
-
const physicalParent = await realpath(dirname(out));
|
|
471
|
-
const finalPhysicalOut = resolve(physicalParent, basename(out));
|
|
472
|
-
assertReportDestination(
|
|
473
|
-
finalPhysicalOut,
|
|
474
|
-
physicalDirectory,
|
|
475
|
-
physicalSubjectRoot,
|
|
476
|
-
experiment.campaignSnapshot.subject.exclude,
|
|
477
|
-
);
|
|
478
|
-
await writeReportSafely(finalPhysicalOut, rendered, options.force ?? false);
|
|
479
|
-
process.stdout.write(`wrote experiment report to ${out}\n`);
|
|
419
|
+
}
|
|
420
|
+
const out = resolve(options.out);
|
|
421
|
+
const run = await loadFrozenRegradableRun(directory);
|
|
422
|
+
try {
|
|
423
|
+
const rubric = await loadRubricDefinition(await resolveAuthoredFile(options.rubric, "rubric"));
|
|
424
|
+
if (rubric.id !== run.case.id) {
|
|
425
|
+
throw new Error(`Rubric ID '${rubric.id}' does not match persisted case '${run.case.id}'.`);
|
|
426
|
+
}
|
|
427
|
+
const rubricIds = new Set(rubric.assertions.map((assertion) => assertion.id));
|
|
428
|
+
const unknownIds = assertionIds.filter((assertionId) => !rubricIds.has(assertionId));
|
|
429
|
+
if (unknownIds.length > 0) {
|
|
430
|
+
throw new Errors.ParseError({
|
|
431
|
+
message: `Unknown rubric assertion IDs: ${unknownIds.join(", ")}`,
|
|
432
|
+
});
|
|
433
|
+
}
|
|
434
|
+
const annotationTags = [
|
|
435
|
+
...(options.split ? [options.split] : []),
|
|
436
|
+
...normalizeTags(options.tag),
|
|
437
|
+
].filter((tag, index, all) => all.indexOf(tag) === index);
|
|
438
|
+
const portableRunDir = (relative(dirname(out), directory) || ".").split(sep).join("/");
|
|
439
|
+
const id =
|
|
440
|
+
options.id ??
|
|
441
|
+
`${run.case.id}-${basename(directory)}-${valueHash(portableRunDir).slice(0, 10)}`.replace(
|
|
442
|
+
/[^a-zA-Z0-9_-]+/g,
|
|
443
|
+
"-",
|
|
444
|
+
);
|
|
445
|
+
const annotation = GraderAnnotationSchema.parse({
|
|
446
|
+
id,
|
|
447
|
+
runDir: portableRunDir,
|
|
448
|
+
rubricPath: options.rubric,
|
|
449
|
+
tags: annotationTags,
|
|
450
|
+
source: options.source,
|
|
451
|
+
...(options.annotator ? { annotator: options.annotator } : {}),
|
|
452
|
+
labeledAt: new Date().toISOString(),
|
|
453
|
+
...(options.reviewer ? { reviewer: options.reviewer } : {}),
|
|
454
|
+
expected: {
|
|
455
|
+
...(options.label ? { casePassed: options.label === "pass" } : {}),
|
|
456
|
+
...(options.score !== undefined ? { caseScore: options.score } : {}),
|
|
457
|
+
assertions,
|
|
480
458
|
},
|
|
481
|
-
|
|
459
|
+
...(options.feedback ? { feedback: options.feedback } : {}),
|
|
460
|
+
});
|
|
461
|
+
const dataset = await upsertGraderAnnotation(out, annotation);
|
|
462
|
+
return { id: annotation.id, path: out, labels: dataset.annotations.length };
|
|
463
|
+
} finally {
|
|
464
|
+
await releaseFrozenRegradableRun(run);
|
|
465
|
+
}
|
|
466
|
+
};
|
|
482
467
|
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
.
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
if (Boolean(suiteA) !== Boolean(suiteB)) {
|
|
514
|
-
process.stderr.write("lux compare: both paths must be runs or both must be suites\n");
|
|
515
|
-
failWith(1);
|
|
516
|
-
return;
|
|
517
|
-
}
|
|
518
|
-
if (suiteA && suiteB) {
|
|
519
|
-
const comparison = await compareSuiteRuns(pathA, pathB, {
|
|
520
|
-
...(options.passRateTolerance !== undefined
|
|
521
|
-
? { passRateTolerance: options.passRateTolerance }
|
|
522
|
-
: {}),
|
|
523
|
-
});
|
|
524
|
-
process.stdout.write(`${formatSuiteComparisonText(comparison)}\n`);
|
|
525
|
-
if (hasSuiteRegression(comparison)) failWith(1);
|
|
526
|
-
return;
|
|
527
|
-
}
|
|
528
|
-
const cmp = await compareRuns({ pathA, pathB });
|
|
529
|
-
process.stdout.write(`${formatComparisonText(cmp)}\n`);
|
|
530
|
-
if (hasRegression(cmp)) failWith(1);
|
|
468
|
+
const applyCommand = async (experimentDir: string, candidateId?: string) => {
|
|
469
|
+
const directory = resolve(experimentDir);
|
|
470
|
+
const experiment = await loadExperimentRun(directory);
|
|
471
|
+
if (!experiment) {
|
|
472
|
+
throw new Errors.IncurError({
|
|
473
|
+
code: "EXPERIMENT_NOT_FOUND",
|
|
474
|
+
message: `Experiment manifest not found at ${directory}.`,
|
|
475
|
+
exitCode: 1,
|
|
476
|
+
});
|
|
477
|
+
}
|
|
478
|
+
if (experiment.status !== "passed") {
|
|
479
|
+
throw new Errors.IncurError({
|
|
480
|
+
code: "EXPERIMENT_NOT_PROMOTED",
|
|
481
|
+
message: `Experiment status is '${experiment.status}', not 'passed'.`,
|
|
482
|
+
exitCode: 1,
|
|
483
|
+
});
|
|
484
|
+
}
|
|
485
|
+
const championId = experiment.championCandidateId;
|
|
486
|
+
if (!championId) {
|
|
487
|
+
throw new Errors.IncurError({
|
|
488
|
+
code: "CHAMPION_NOT_FOUND",
|
|
489
|
+
message: "The passed experiment has no promoted champion.",
|
|
490
|
+
exitCode: 1,
|
|
491
|
+
});
|
|
492
|
+
}
|
|
493
|
+
if (candidateId && candidateId !== championId) {
|
|
494
|
+
throw new Errors.IncurError({
|
|
495
|
+
code: "CANDIDATE_NOT_PROMOTED",
|
|
496
|
+
message: `Candidate '${candidateId}' is not the promoted champion '${championId}'.`,
|
|
497
|
+
exitCode: 1,
|
|
531
498
|
});
|
|
499
|
+
}
|
|
500
|
+
let candidate: Candidate | undefined;
|
|
501
|
+
try {
|
|
502
|
+
const candidates = await verifyExperimentRunIntegrity(directory, experiment);
|
|
503
|
+
verifyExperimentDecisionIntegrity(experiment);
|
|
504
|
+
candidate = candidates.find((entry) => entry.id === championId);
|
|
505
|
+
if (!candidate) throw new Error(`Champion candidate '${championId}' not found.`);
|
|
506
|
+
} catch (error) {
|
|
507
|
+
throw new Errors.IncurError({
|
|
508
|
+
code: "EXPERIMENT_INTEGRITY_FAILED",
|
|
509
|
+
message: error instanceof Error ? error.message : String(error),
|
|
510
|
+
exitCode: 1,
|
|
511
|
+
});
|
|
512
|
+
}
|
|
513
|
+
await applyCandidate(experiment.subjectRoot, experiment.campaignSnapshot.subject, candidate);
|
|
514
|
+
return {
|
|
515
|
+
candidateId: candidate.id,
|
|
516
|
+
subjectRoot: experiment.subjectRoot,
|
|
517
|
+
sourceBytes: candidate.sourceBytes,
|
|
518
|
+
};
|
|
532
519
|
};
|
|
533
520
|
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
521
|
+
const reportCommand = async (
|
|
522
|
+
experimentDir: string,
|
|
523
|
+
options: { out?: string | undefined; force?: boolean | undefined },
|
|
524
|
+
) => {
|
|
525
|
+
const directory = resolve(experimentDir);
|
|
526
|
+
if (!(await ensureDirectory("report", directory))) failWith(1);
|
|
527
|
+
const snapshot = await loadVerifiedExperimentDecisionSnapshot(directory);
|
|
528
|
+
const { experiment, report } = snapshot;
|
|
529
|
+
if (!options.out) return report;
|
|
530
|
+
const out = resolve(options.out);
|
|
531
|
+
const outputStats = await lstat(out).catch((error: NodeJS.ErrnoException) => {
|
|
532
|
+
if (error.code === "ENOENT") return null;
|
|
533
|
+
throw error;
|
|
534
|
+
});
|
|
535
|
+
if (outputStats && (outputStats.isSymbolicLink() || !outputStats.isFile())) {
|
|
536
|
+
throw new Error("An existing --out must be a regular, non-symlink file.");
|
|
537
|
+
}
|
|
538
|
+
const [physicalDirectory, physicalSubjectRoot, physicalOut] = await Promise.all([
|
|
539
|
+
realpath(directory),
|
|
540
|
+
realpath(experiment.subjectRoot),
|
|
541
|
+
physicalDestinationPath(out),
|
|
542
|
+
]);
|
|
543
|
+
assertReportDestination(
|
|
544
|
+
physicalOut,
|
|
545
|
+
physicalDirectory,
|
|
546
|
+
physicalSubjectRoot,
|
|
547
|
+
experiment.campaignSnapshot.subject.exclude,
|
|
548
|
+
);
|
|
549
|
+
if (outputStats && !options.force) {
|
|
550
|
+
throw new Error(`Output already exists (use --force): ${out}`);
|
|
551
|
+
}
|
|
552
|
+
await mkdir(dirname(out), { recursive: true });
|
|
553
|
+
const physicalParent = await realpath(dirname(out));
|
|
554
|
+
const finalPhysicalOut = resolve(physicalParent, basename(out));
|
|
555
|
+
assertReportDestination(
|
|
556
|
+
finalPhysicalOut,
|
|
557
|
+
physicalDirectory,
|
|
558
|
+
physicalSubjectRoot,
|
|
559
|
+
experiment.campaignSnapshot.subject.exclude,
|
|
560
|
+
);
|
|
561
|
+
await writeReportSafely(
|
|
562
|
+
finalPhysicalOut,
|
|
563
|
+
`${formatExperimentDecisionMarkdown(report)}\n`,
|
|
564
|
+
options.force ?? false,
|
|
565
|
+
);
|
|
566
|
+
return { path: out, report };
|
|
567
|
+
};
|
|
568
|
+
|
|
569
|
+
const compareCommand = async (
|
|
570
|
+
runA: string,
|
|
571
|
+
runB: string,
|
|
572
|
+
options: { passRateTolerance?: number | undefined },
|
|
573
|
+
) => {
|
|
574
|
+
const pathA = resolve(runA);
|
|
575
|
+
const pathB = resolve(runB);
|
|
576
|
+
const invalid = (
|
|
577
|
+
await Promise.all(
|
|
578
|
+
[pathA, pathB].map(async (path) =>
|
|
579
|
+
existsSync(path) && (await stat(path)).isDirectory() ? undefined : path,
|
|
580
|
+
),
|
|
540
581
|
)
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
582
|
+
).filter((path): path is string => path !== undefined);
|
|
583
|
+
if (invalid.length > 0) {
|
|
584
|
+
throw new Errors.IncurError({
|
|
585
|
+
code: "RUN_DIRECTORY_NOT_FOUND",
|
|
586
|
+
message: invalid.map((path) => `${path} is not a directory`).join("; "),
|
|
587
|
+
exitCode: 1,
|
|
588
|
+
});
|
|
589
|
+
}
|
|
590
|
+
const [suiteA, suiteB] = await Promise.all([loadSuiteRun(pathA), loadSuiteRun(pathB)]);
|
|
591
|
+
if (Boolean(suiteA) !== Boolean(suiteB)) {
|
|
592
|
+
throw new Errors.IncurError({
|
|
593
|
+
code: "INCOMPATIBLE_RUN_TYPES",
|
|
594
|
+
message: "Both paths must be runs or both must be suites.",
|
|
595
|
+
exitCode: 1,
|
|
596
|
+
});
|
|
597
|
+
}
|
|
598
|
+
if (suiteA && suiteB) {
|
|
599
|
+
const comparison = await compareSuiteRuns(pathA, pathB, {
|
|
600
|
+
...(options.passRateTolerance !== undefined
|
|
601
|
+
? { passRateTolerance: options.passRateTolerance }
|
|
602
|
+
: {}),
|
|
603
|
+
});
|
|
604
|
+
if (hasSuiteRegression(comparison)) {
|
|
605
|
+
throw new Errors.IncurError({
|
|
606
|
+
code: "REGRESSION_DETECTED",
|
|
607
|
+
message: formatSuiteComparisonText(comparison),
|
|
608
|
+
exitCode: 1,
|
|
609
|
+
});
|
|
610
|
+
}
|
|
611
|
+
return comparison;
|
|
612
|
+
}
|
|
613
|
+
const comparison = await compareRuns({ pathA, pathB });
|
|
614
|
+
if (hasRegression(comparison)) {
|
|
615
|
+
throw new Errors.IncurError({
|
|
616
|
+
code: "REGRESSION_DETECTED",
|
|
617
|
+
message: formatComparisonText(comparison),
|
|
618
|
+
exitCode: 1,
|
|
619
|
+
});
|
|
620
|
+
}
|
|
621
|
+
return comparison;
|
|
552
622
|
};
|
|
553
623
|
|
|
554
624
|
const caseIdOf = (casePath: string): string => basename(casePath, extname(casePath));
|
|
@@ -708,15 +778,19 @@ const expandLoop = (base: SuiteItem[], loop: number): SuiteItem[] =>
|
|
|
708
778
|
const experimentCommand = async (
|
|
709
779
|
mode: "experiment" | "optimize",
|
|
710
780
|
campaignPath: string,
|
|
711
|
-
options: {
|
|
712
|
-
|
|
781
|
+
options: {
|
|
782
|
+
runsRoot?: string | undefined;
|
|
783
|
+
resume?: string | undefined;
|
|
784
|
+
plan?: boolean | undefined;
|
|
785
|
+
},
|
|
786
|
+
) => {
|
|
713
787
|
const cwd = process.cwd();
|
|
714
788
|
const config = await loadLuxConfig(cwd);
|
|
715
789
|
const campaign = await resolveExperimentCampaignPath(config.rootDir, campaignPath);
|
|
716
790
|
const loaded = await loadExperimentCampaign(campaign);
|
|
717
791
|
experimentCampaignMode(campaign, loaded.campaign, mode);
|
|
718
792
|
const registry = buildRegistryFromConfig(config);
|
|
719
|
-
|
|
793
|
+
using controller = abortOnInterrupt();
|
|
720
794
|
const execution = {
|
|
721
795
|
config,
|
|
722
796
|
registry,
|
|
@@ -727,64 +801,52 @@ const experimentCommand = async (
|
|
|
727
801
|
};
|
|
728
802
|
if (options.plan) {
|
|
729
803
|
const plan = await planExperimentCampaign(execution, mode);
|
|
730
|
-
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
`campaign: ${plan.campaignId}`,
|
|
737
|
-
`mode: ${plan.mode}`,
|
|
738
|
-
`evaluation: ${plan.evaluationKind}`,
|
|
739
|
-
`profiles: ${plan.profiles.join(", ")}`,
|
|
740
|
-
`splits: ${plan.splits.map((split) => `${split.split}=${split.examples}`).join(", ")}`,
|
|
741
|
-
`components: ${plan.mutableComponents.length}`,
|
|
742
|
-
`candidates: ${plan.projected.candidates}`,
|
|
743
|
-
`metric calls: ${plan.projected.metricCalls}/${plan.budget.maxMetricCalls}`,
|
|
744
|
-
`judge calls: ${plan.projected.judgeCalls}/${plan.budget.maxJudgeCalls}`,
|
|
745
|
-
`proposal calls:${plan.projected.proposalCalls}/${plan.budget.maxProposalCalls}`,
|
|
746
|
-
`unpriced calls: observed at runtime (cap ${plan.budget.maxUnpricedModelCalls})`,
|
|
747
|
-
`projected work within budget: ${projectedStatus}`,
|
|
748
|
-
...plan.warnings.map((warning) => `warning: ${warning}`),
|
|
749
|
-
"",
|
|
750
|
-
].join("\n");
|
|
804
|
+
if (!plan.fitsProjectedCallBudgets) {
|
|
805
|
+
throw new Errors.IncurError({
|
|
806
|
+
code: "EXPERIMENT_BUDGET_EXCEEDED",
|
|
807
|
+
message: "The projected experiment work exceeds its configured call budgets.",
|
|
808
|
+
exitCode: 1,
|
|
809
|
+
});
|
|
751
810
|
}
|
|
752
|
-
|
|
753
|
-
if (!plan.fitsProjectedCallBudgets) failWith(1);
|
|
754
|
-
return;
|
|
811
|
+
return plan;
|
|
755
812
|
}
|
|
756
813
|
const result =
|
|
757
814
|
mode === "optimize"
|
|
758
815
|
? await optimizeExperimentCampaign(execution)
|
|
759
816
|
: await runExperimentCampaign(execution);
|
|
760
|
-
process.stdout.write(
|
|
761
|
-
options.json
|
|
762
|
-
? `${JSON.stringify({ experimentDir: result.experimentDir, ...result.manifest }, null, 2)}\n`
|
|
763
|
-
: `${formatExperimentRun(result.experimentDir, result.manifest)}\n`,
|
|
764
|
-
);
|
|
765
817
|
if (controller.signal.aborted) failWith(SIGINT_EXIT_CODE);
|
|
766
|
-
|
|
818
|
+
if (result.manifest.status !== "passed") {
|
|
819
|
+
throw new Errors.IncurError({
|
|
820
|
+
code: "EXPERIMENT_FAILED",
|
|
821
|
+
message: `Experiment ${result.manifest.experimentId} finished with status '${result.manifest.status}'.`,
|
|
822
|
+
exitCode: 1,
|
|
823
|
+
});
|
|
824
|
+
}
|
|
825
|
+
return { experimentDir: result.experimentDir, ...result.manifest };
|
|
767
826
|
};
|
|
768
827
|
|
|
769
|
-
const runCommand = async (filters: string[], options: RunOptions)
|
|
828
|
+
const runCommand = async (filters: string[], options: RunOptions, human = false) => {
|
|
770
829
|
if (options.interactive && options.claudeAnswerer) {
|
|
771
|
-
throw new
|
|
830
|
+
throw new Errors.ParseError({
|
|
831
|
+
message: "--interactive and --claude-answerer are mutually exclusive",
|
|
832
|
+
});
|
|
772
833
|
}
|
|
773
834
|
const cwd = process.cwd();
|
|
774
835
|
const config = await loadLuxConfig(cwd);
|
|
775
836
|
const profile = options.profile ? config.profiles?.[options.profile] : undefined;
|
|
776
837
|
if (options.profile && !profile) {
|
|
777
838
|
const available = Object.keys(config.profiles ?? {}).sort();
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
|
|
781
|
-
|
|
782
|
-
|
|
839
|
+
throw new Errors.IncurError({
|
|
840
|
+
code: "PROFILE_NOT_FOUND",
|
|
841
|
+
message: `Unknown profile '${options.profile}'${available.length > 0 ? `; available: ${available.join(", ")}` : ""}.`,
|
|
842
|
+
exitCode: 1,
|
|
843
|
+
});
|
|
783
844
|
}
|
|
784
845
|
const harness = profile?.harness ?? config.harness;
|
|
785
846
|
const registry = buildRegistryFromConfig(config).withAssertion(createRubricAssertion(harness));
|
|
786
847
|
const defaultDriver = profile?.driver ?? config.defaultDriver;
|
|
787
848
|
const orchestrator = new Orchestrator({
|
|
849
|
+
environment: profile?.environment ?? config.environment,
|
|
788
850
|
registry,
|
|
789
851
|
defaultAnswerer: profile?.answerer ?? config.defaultAnswerer,
|
|
790
852
|
...(defaultDriver ? { defaultDriver } : {}),
|
|
@@ -802,23 +864,26 @@ const runCommand = async (filters: string[], options: RunOptions): Promise<void>
|
|
|
802
864
|
// broken config, not an empty suite — never mask it behind "no cases
|
|
803
865
|
// matched" or a lucky explicit-file run.
|
|
804
866
|
if (config.casesRoot && !existsSync(config.casesRoot)) {
|
|
805
|
-
|
|
806
|
-
|
|
807
|
-
|
|
867
|
+
throw new Errors.IncurError({
|
|
868
|
+
code: "CASES_ROOT_NOT_FOUND",
|
|
869
|
+
message: `casesRoot does not exist: ${config.casesRoot}`,
|
|
870
|
+
exitCode: 1,
|
|
871
|
+
});
|
|
808
872
|
}
|
|
809
873
|
|
|
810
874
|
const { explicit, patterns } = partitionFilters(filters, cwd);
|
|
811
|
-
const tags = options.tag ?? [];
|
|
875
|
+
const tags = normalizeTags(options.tag ?? []);
|
|
812
876
|
const hasFilter = explicit.length > 0 || patterns.length > 0 || tags.length > 0;
|
|
813
877
|
|
|
814
878
|
// Explicitly named files need no discovery, so they run even without a
|
|
815
879
|
// project; anything else would need a tree walk with nowhere safe to walk.
|
|
816
880
|
if (config.configPath === null && !existsSync(root) && explicit.length === 0) {
|
|
817
|
-
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
881
|
+
throw new Errors.IncurError({
|
|
882
|
+
code: "PROJECT_NOT_INITIALIZED",
|
|
883
|
+
message: `No lux.config.ts or cases/ directory in ${cwd}.`,
|
|
884
|
+
hint: "Run `lux init` or name a case file directly.",
|
|
885
|
+
exitCode: 1,
|
|
886
|
+
});
|
|
822
887
|
}
|
|
823
888
|
|
|
824
889
|
const { cases, problems } = existsSync(root)
|
|
@@ -833,35 +898,32 @@ const runCommand = async (filters: string[], options: RunOptions): Promise<void>
|
|
|
833
898
|
// An empty selection fails every mode the same way — running, `--list`, and
|
|
834
899
|
// `--list-tags` all exit 1, so a typo'd filter can never read as a clean pass.
|
|
835
900
|
if (selected.length === 0) {
|
|
836
|
-
|
|
837
|
-
|
|
838
|
-
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
|
|
842
|
-
|
|
901
|
+
throw new Errors.IncurError({
|
|
902
|
+
code: "NO_CASES_MATCHED",
|
|
903
|
+
message: hasFilter
|
|
904
|
+
? "No cases matched the given filters."
|
|
905
|
+
: `No cases were discovered under ${root}.`,
|
|
906
|
+
exitCode: 1,
|
|
907
|
+
});
|
|
843
908
|
}
|
|
844
909
|
|
|
845
910
|
// List the tag vocabulary of the selected cases — "which --tag values exist?".
|
|
846
911
|
if (options.listTags) {
|
|
847
912
|
const counts = new Map<string, number>();
|
|
848
913
|
for (const c of selected) for (const t of c.tags) counts.set(t, (counts.get(t) ?? 0) + 1);
|
|
849
|
-
|
|
850
|
-
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
return;
|
|
914
|
+
return {
|
|
915
|
+
caseCount: selected.length,
|
|
916
|
+
tags: [...counts]
|
|
917
|
+
.sort((a, b) => a[0].localeCompare(b[0]))
|
|
918
|
+
.map(([tag, count]) => ({ tag, count })),
|
|
919
|
+
};
|
|
856
920
|
}
|
|
857
921
|
|
|
858
922
|
const base: SuiteItem[] = selected.map((c) => ({ path: c.path, id: c.id }));
|
|
859
923
|
|
|
860
924
|
// Collect-only: answer "what would run?" without spawning a single agent.
|
|
861
925
|
if (options.list) {
|
|
862
|
-
|
|
863
|
-
process.stderr.write(`lux: ${base.length} case${base.length === 1 ? "" : "s"} matched\n`);
|
|
864
|
-
return;
|
|
926
|
+
return { caseCount: base.length, cases: base };
|
|
865
927
|
}
|
|
866
928
|
|
|
867
929
|
const loop = options.loop ?? 1;
|
|
@@ -870,9 +932,9 @@ const runCommand = async (filters: string[], options: RunOptions): Promise<void>
|
|
|
870
932
|
const { concurrency, note } = resolveConcurrency(options, config.concurrency);
|
|
871
933
|
if (note) process.stderr.write(note);
|
|
872
934
|
|
|
873
|
-
const reporter = new ConsoleReporter({ verbose: options.verbose ?? false });
|
|
874
|
-
|
|
875
|
-
reporter
|
|
935
|
+
const reporter = human ? new ConsoleReporter({ verbose: options.verbose ?? false }) : undefined;
|
|
936
|
+
using controller = abortOnInterrupt();
|
|
937
|
+
reporter?.begin({
|
|
876
938
|
title: suiteTitle(patterns, tags, explicit, loop),
|
|
877
939
|
caseIds: items.map((i) => i.id),
|
|
878
940
|
runsRoot,
|
|
@@ -886,28 +948,39 @@ const runCommand = async (filters: string[], options: RunOptions): Promise<void>
|
|
|
886
948
|
fixturesRoot: config.fixturesRoot,
|
|
887
949
|
answererOverride: resolveAnswererOverride(options),
|
|
888
950
|
skipJudge: options.skipJudge,
|
|
889
|
-
observer: reporter,
|
|
951
|
+
...(reporter ? { observer: reporter } : {}),
|
|
890
952
|
abortSignal: controller.signal,
|
|
891
953
|
concurrency,
|
|
892
954
|
subjectRoot,
|
|
893
955
|
profileId: options.profile,
|
|
894
956
|
});
|
|
895
957
|
results = suite.results;
|
|
896
|
-
reporter
|
|
958
|
+
reporter?.setSuiteDir(suite.suiteDir);
|
|
959
|
+
if (controller.signal.aborted) failWith(SIGINT_EXIT_CODE);
|
|
960
|
+
if (results.some(isFailure)) {
|
|
961
|
+
throw new Errors.IncurError({
|
|
962
|
+
code: "SUITE_FAILED",
|
|
963
|
+
message: `${results.filter(isFailure).length} of ${results.length} case attempts failed. Suite: ${suite.suiteDir}`,
|
|
964
|
+
exitCode: 1,
|
|
965
|
+
});
|
|
966
|
+
}
|
|
967
|
+
return {
|
|
968
|
+
suiteDir: suite.suiteDir,
|
|
969
|
+
results: results.map((result) => ({
|
|
970
|
+
caseId: result.caseId,
|
|
971
|
+
runDir: result.runDir ?? null,
|
|
972
|
+
exitReason: result.exitReason,
|
|
973
|
+
casePassed: result.casePassed,
|
|
974
|
+
caseScore: result.result?.grading?.caseScore ?? null,
|
|
975
|
+
assertionsPassed:
|
|
976
|
+
result.result?.grading?.assertions.filter((assertion) => assertion.passed).length ?? 0,
|
|
977
|
+
assertionsTotal: result.result?.grading?.assertions.length ?? 0,
|
|
978
|
+
})),
|
|
979
|
+
};
|
|
897
980
|
} finally {
|
|
898
|
-
reporter
|
|
981
|
+
reporter?.finish();
|
|
899
982
|
}
|
|
900
|
-
|
|
901
|
-
if (controller.signal.aborted) failWith(SIGINT_EXIT_CODE);
|
|
902
|
-
else if (results.some(isFailure)) failWith(1);
|
|
903
983
|
};
|
|
904
984
|
|
|
905
985
|
// Internal exports for testing
|
|
906
|
-
export {
|
|
907
|
-
expandLoop,
|
|
908
|
-
isFailure,
|
|
909
|
-
parsePositiveInt,
|
|
910
|
-
partitionFilters,
|
|
911
|
-
resolveConcurrency,
|
|
912
|
-
selectSuite,
|
|
913
|
-
};
|
|
986
|
+
export { expandLoop, isFailure, partitionFilters, resolveConcurrency, selectSuite };
|