@prismatic-io/lux 0.0.2-preview.14 → 0.0.2-preview.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (388) hide show
  1. package/lib/answerers/claude-code/index.d.ts +14 -2
  2. package/lib/answerers/claude-code/index.d.ts.map +1 -1
  3. package/lib/answerers/claude-code/index.js +3 -3
  4. package/lib/answerers/claude-code/index.js.map +1 -1
  5. package/lib/answerers/persona/index.d.ts +28 -2
  6. package/lib/answerers/persona/index.d.ts.map +1 -1
  7. package/lib/answerers/persona/index.js +3 -3
  8. package/lib/answerers/persona/index.js.map +1 -1
  9. package/lib/answerers/terminal/index.d.ts +8 -2
  10. package/lib/answerers/terminal/index.d.ts.map +1 -1
  11. package/lib/answerers/terminal/index.js +3 -2
  12. package/lib/answerers/terminal/index.js.map +1 -1
  13. package/lib/assertions/core/resource-checks.d.ts.map +1 -1
  14. package/lib/assertions/core/resource-checks.js +3 -5
  15. package/lib/assertions/core/resource-checks.js.map +1 -1
  16. package/lib/assertions/rubric/index.d.ts.map +1 -1
  17. package/lib/assertions/rubric/index.js +7 -1
  18. package/lib/assertions/rubric/index.js.map +1 -1
  19. package/lib/assertions/rubric/internal.d.ts +16 -5
  20. package/lib/assertions/rubric/internal.d.ts.map +1 -1
  21. package/lib/assertions/rubric/internal.js +10 -6
  22. package/lib/assertions/rubric/internal.js.map +1 -1
  23. package/lib/authoring.d.ts +32 -2
  24. package/lib/authoring.d.ts.map +1 -1
  25. package/lib/authoring.js +33 -1
  26. package/lib/authoring.js.map +1 -1
  27. package/lib/cli/bin.js +2 -13
  28. package/lib/cli/bin.js.map +1 -1
  29. package/lib/cli/command-runtime.d.ts +3 -3
  30. package/lib/cli/command-runtime.d.ts.map +1 -1
  31. package/lib/cli/command-runtime.js +17 -6
  32. package/lib/cli/command-runtime.js.map +1 -1
  33. package/lib/cli/init-templates.d.ts +11 -0
  34. package/lib/cli/init-templates.d.ts.map +1 -0
  35. package/lib/cli/init-templates.js +207 -0
  36. package/lib/cli/init-templates.js.map +1 -0
  37. package/lib/cli/init.d.ts +7 -0
  38. package/lib/cli/init.d.ts.map +1 -1
  39. package/lib/cli/init.js +23 -173
  40. package/lib/cli/init.js.map +1 -1
  41. package/lib/cli/output-schemas.d.ts +1399 -0
  42. package/lib/cli/output-schemas.d.ts.map +1 -0
  43. package/lib/cli/output-schemas.js +91 -0
  44. package/lib/cli/output-schemas.js.map +1 -0
  45. package/lib/cli/program.d.ts +135 -4
  46. package/lib/cli/program.d.ts.map +1 -1
  47. package/lib/cli/program.js +725 -520
  48. package/lib/cli/program.js.map +1 -1
  49. package/lib/cli/render/reporter.d.ts.map +1 -1
  50. package/lib/cli/render/reporter.js +24 -4
  51. package/lib/cli/render/reporter.js.map +1 -1
  52. package/lib/cli/run-options.d.ts +0 -4
  53. package/lib/cli/run-options.d.ts.map +1 -1
  54. package/lib/cli/run-options.js +8 -19
  55. package/lib/cli/run-options.js.map +1 -1
  56. package/lib/cli/view.d.ts +5 -3
  57. package/lib/cli/view.d.ts.map +1 -1
  58. package/lib/cli/view.js +120 -78
  59. package/lib/cli/view.js.map +1 -1
  60. package/lib/core/annotation.d.ts +6 -2
  61. package/lib/core/annotation.d.ts.map +1 -1
  62. package/lib/core/annotation.js +2 -1
  63. package/lib/core/annotation.js.map +1 -1
  64. package/lib/core/answerer.d.ts +12 -0
  65. package/lib/core/answerer.d.ts.map +1 -1
  66. package/lib/core/answerer.js +2 -0
  67. package/lib/core/answerer.js.map +1 -1
  68. package/lib/core/assertion.d.ts +3 -0
  69. package/lib/core/assertion.d.ts.map +1 -1
  70. package/lib/core/assertion.js +3 -0
  71. package/lib/core/assertion.js.map +1 -1
  72. package/lib/core/case.d.ts +24 -2
  73. package/lib/core/case.d.ts.map +1 -1
  74. package/lib/core/case.js +5 -1
  75. package/lib/core/case.js.map +1 -1
  76. package/lib/core/driver.d.ts +40 -0
  77. package/lib/core/driver.d.ts.map +1 -1
  78. package/lib/core/driver.js +12 -0
  79. package/lib/core/driver.js.map +1 -1
  80. package/lib/core/environment.d.ts +18 -0
  81. package/lib/core/environment.d.ts.map +1 -0
  82. package/lib/core/environment.js +21 -0
  83. package/lib/core/environment.js.map +1 -0
  84. package/lib/core/evaluation-clusters.d.ts +17 -0
  85. package/lib/core/evaluation-clusters.d.ts.map +1 -0
  86. package/lib/core/evaluation-clusters.js +23 -0
  87. package/lib/core/evaluation-clusters.js.map +1 -0
  88. package/lib/core/experiment.d.ts +11 -0
  89. package/lib/core/experiment.d.ts.map +1 -1
  90. package/lib/core/experiment.js +31 -17
  91. package/lib/core/experiment.js.map +1 -1
  92. package/lib/core/file-tree.d.ts +7 -0
  93. package/lib/core/file-tree.d.ts.map +1 -0
  94. package/lib/core/file-tree.js +35 -0
  95. package/lib/core/file-tree.js.map +1 -0
  96. package/lib/core/index.d.ts +1 -0
  97. package/lib/core/index.d.ts.map +1 -1
  98. package/lib/core/index.js +1 -0
  99. package/lib/core/index.js.map +1 -1
  100. package/lib/core/lifecycle-fixtures.d.ts.map +1 -1
  101. package/lib/core/lifecycle-fixtures.js +4 -2
  102. package/lib/core/lifecycle-fixtures.js.map +1 -1
  103. package/lib/core/platform-process.d.ts +5 -1
  104. package/lib/core/platform-process.d.ts.map +1 -1
  105. package/lib/core/platform-process.js +48 -10
  106. package/lib/core/platform-process.js.map +1 -1
  107. package/lib/core/run.d.ts +93 -1
  108. package/lib/core/run.d.ts.map +1 -1
  109. package/lib/core/run.js +16 -0
  110. package/lib/core/run.js.map +1 -1
  111. package/lib/core/usage.d.ts +5 -0
  112. package/lib/core/usage.d.ts.map +1 -1
  113. package/lib/core/usage.js +3 -0
  114. package/lib/core/usage.js.map +1 -1
  115. package/lib/drivers/claude-code/index.d.ts +65 -2
  116. package/lib/drivers/claude-code/index.d.ts.map +1 -1
  117. package/lib/drivers/claude-code/index.js +30 -3
  118. package/lib/drivers/claude-code/index.js.map +1 -1
  119. package/lib/drivers/codex/app-events.d.ts.map +1 -1
  120. package/lib/drivers/codex/app-events.js +41 -30
  121. package/lib/drivers/codex/app-events.js.map +1 -1
  122. package/lib/drivers/codex/exec-events.d.ts +4 -0
  123. package/lib/drivers/codex/exec-events.d.ts.map +1 -0
  124. package/lib/drivers/codex/exec-events.js +230 -0
  125. package/lib/drivers/codex/exec-events.js.map +1 -0
  126. package/lib/drivers/codex/index.d.ts +76 -4
  127. package/lib/drivers/codex/index.d.ts.map +1 -1
  128. package/lib/drivers/codex/index.js +35 -232
  129. package/lib/drivers/codex/index.js.map +1 -1
  130. package/lib/drivers/cursor/acp-transport.d.ts +38 -0
  131. package/lib/drivers/cursor/acp-transport.d.ts.map +1 -0
  132. package/lib/drivers/cursor/acp-transport.js +152 -0
  133. package/lib/drivers/cursor/acp-transport.js.map +1 -0
  134. package/lib/drivers/cursor/config.d.ts +22 -0
  135. package/lib/drivers/cursor/config.d.ts.map +1 -0
  136. package/lib/drivers/cursor/config.js +26 -0
  137. package/lib/drivers/cursor/config.js.map +1 -0
  138. package/lib/drivers/cursor/events.d.ts +13 -0
  139. package/lib/drivers/cursor/events.d.ts.map +1 -0
  140. package/lib/drivers/cursor/events.js +73 -0
  141. package/lib/drivers/cursor/events.js.map +1 -0
  142. package/lib/drivers/cursor/index.d.ts +37 -0
  143. package/lib/drivers/cursor/index.d.ts.map +1 -0
  144. package/lib/drivers/cursor/index.js +276 -0
  145. package/lib/drivers/cursor/index.js.map +1 -0
  146. package/lib/drivers/cursor/interaction.d.ts +7 -0
  147. package/lib/drivers/cursor/interaction.d.ts.map +1 -0
  148. package/lib/drivers/cursor/interaction.js +66 -0
  149. package/lib/drivers/cursor/interaction.js.map +1 -0
  150. package/lib/drivers/mcp/index.d.ts +35 -2
  151. package/lib/drivers/mcp/index.d.ts.map +1 -1
  152. package/lib/drivers/mcp/index.js +16 -3
  153. package/lib/drivers/mcp/index.js.map +1 -1
  154. package/lib/drivers/shared/experiment-behavior.d.ts +13 -0
  155. package/lib/drivers/shared/experiment-behavior.d.ts.map +1 -0
  156. package/lib/drivers/shared/experiment-behavior.js +197 -0
  157. package/lib/drivers/shared/experiment-behavior.js.map +1 -0
  158. package/lib/drivers/shared/experiment-policy.d.ts +4 -0
  159. package/lib/drivers/shared/experiment-policy.d.ts.map +1 -0
  160. package/lib/drivers/shared/experiment-policy.js +143 -0
  161. package/lib/drivers/shared/experiment-policy.js.map +1 -0
  162. package/lib/drivers/subprocess/index.d.ts +37 -2
  163. package/lib/drivers/subprocess/index.d.ts.map +1 -1
  164. package/lib/drivers/subprocess/index.js +46 -6
  165. package/lib/drivers/subprocess/index.js.map +1 -1
  166. package/lib/index.d.ts +6 -5
  167. package/lib/index.d.ts.map +1 -1
  168. package/lib/index.js +4 -3
  169. package/lib/index.js.map +1 -1
  170. package/lib/orchestrator/annotation-loader.d.ts +6 -2
  171. package/lib/orchestrator/annotation-loader.d.ts.map +1 -1
  172. package/lib/orchestrator/annotation-loader.js +8 -0
  173. package/lib/orchestrator/annotation-loader.js.map +1 -1
  174. package/lib/orchestrator/annotation-store.d.ts.map +1 -1
  175. package/lib/orchestrator/annotation-store.js +16 -2
  176. package/lib/orchestrator/annotation-store.js.map +1 -1
  177. package/lib/orchestrator/campaign-lifecycle.d.ts +1 -0
  178. package/lib/orchestrator/campaign-lifecycle.d.ts.map +1 -1
  179. package/lib/orchestrator/campaign-lifecycle.js +4 -1
  180. package/lib/orchestrator/campaign-lifecycle.js.map +1 -1
  181. package/lib/orchestrator/compare.d.ts +4 -0
  182. package/lib/orchestrator/compare.d.ts.map +1 -1
  183. package/lib/orchestrator/compare.js +13 -1
  184. package/lib/orchestrator/compare.js.map +1 -1
  185. package/lib/orchestrator/comparison-identity.d.ts.map +1 -1
  186. package/lib/orchestrator/comparison-identity.js +1 -0
  187. package/lib/orchestrator/comparison-identity.js.map +1 -1
  188. package/lib/orchestrator/config.d.ts +42 -0
  189. package/lib/orchestrator/config.d.ts.map +1 -1
  190. package/lib/orchestrator/config.js +6 -2
  191. package/lib/orchestrator/config.js.map +1 -1
  192. package/lib/orchestrator/discover.d.ts +1 -0
  193. package/lib/orchestrator/discover.d.ts.map +1 -1
  194. package/lib/orchestrator/discover.js +7 -1
  195. package/lib/orchestrator/discover.js.map +1 -1
  196. package/lib/orchestrator/doctor.d.ts.map +1 -1
  197. package/lib/orchestrator/doctor.js +34 -26
  198. package/lib/orchestrator/doctor.js.map +1 -1
  199. package/lib/orchestrator/driver-capabilities.d.ts +3 -0
  200. package/lib/orchestrator/driver-capabilities.d.ts.map +1 -0
  201. package/lib/orchestrator/driver-capabilities.js +10 -0
  202. package/lib/orchestrator/driver-capabilities.js.map +1 -0
  203. package/lib/orchestrator/driver-identity.d.ts +5 -0
  204. package/lib/orchestrator/driver-identity.d.ts.map +1 -0
  205. package/lib/orchestrator/driver-identity.js +41 -0
  206. package/lib/orchestrator/driver-identity.js.map +1 -0
  207. package/lib/orchestrator/experiment-context.d.ts +10 -0
  208. package/lib/orchestrator/experiment-context.d.ts.map +1 -1
  209. package/lib/orchestrator/experiment-contracts.d.ts +10 -0
  210. package/lib/orchestrator/experiment-contracts.d.ts.map +1 -1
  211. package/lib/orchestrator/experiment-corpus.d.ts.map +1 -1
  212. package/lib/orchestrator/experiment-corpus.js +17 -19
  213. package/lib/orchestrator/experiment-corpus.js.map +1 -1
  214. package/lib/orchestrator/experiment-effects.d.ts +33 -0
  215. package/lib/orchestrator/experiment-effects.d.ts.map +1 -0
  216. package/lib/orchestrator/experiment-effects.js +44 -0
  217. package/lib/orchestrator/experiment-effects.js.map +1 -0
  218. package/lib/orchestrator/experiment-evaluation.d.ts.map +1 -1
  219. package/lib/orchestrator/experiment-evaluation.js +36 -3
  220. package/lib/orchestrator/experiment-evaluation.js.map +1 -1
  221. package/lib/orchestrator/experiment-identity.d.ts.map +1 -1
  222. package/lib/orchestrator/experiment-identity.js +18 -29
  223. package/lib/orchestrator/experiment-identity.js.map +1 -1
  224. package/lib/orchestrator/experiment-loader.d.ts +7 -2
  225. package/lib/orchestrator/experiment-loader.d.ts.map +1 -1
  226. package/lib/orchestrator/experiment-report.d.ts +70 -0
  227. package/lib/orchestrator/experiment-report.d.ts.map +1 -1
  228. package/lib/orchestrator/experiment-report.js +9 -1
  229. package/lib/orchestrator/experiment-report.js.map +1 -1
  230. package/lib/orchestrator/experiment-runs.d.ts +22 -1
  231. package/lib/orchestrator/experiment-runs.d.ts.map +1 -1
  232. package/lib/orchestrator/experiment-runs.js +13 -5
  233. package/lib/orchestrator/experiment-runs.js.map +1 -1
  234. package/lib/orchestrator/experiment-runtime-identity.d.ts +0 -16
  235. package/lib/orchestrator/experiment-runtime-identity.d.ts.map +1 -1
  236. package/lib/orchestrator/experiment-runtime-identity.js +5 -220
  237. package/lib/orchestrator/experiment-runtime-identity.js.map +1 -1
  238. package/lib/orchestrator/experiment-selection.d.ts.map +1 -1
  239. package/lib/orchestrator/experiment-selection.js +52 -0
  240. package/lib/orchestrator/experiment-selection.js.map +1 -1
  241. package/lib/orchestrator/experiment.d.ts +0 -2
  242. package/lib/orchestrator/experiment.d.ts.map +1 -1
  243. package/lib/orchestrator/experiment.js +23 -15
  244. package/lib/orchestrator/experiment.js.map +1 -1
  245. package/lib/orchestrator/fs-read.d.ts +1 -6
  246. package/lib/orchestrator/fs-read.d.ts.map +1 -1
  247. package/lib/orchestrator/fs-read.js +2 -32
  248. package/lib/orchestrator/fs-read.js.map +1 -1
  249. package/lib/orchestrator/grader-experiment.d.ts +17 -1
  250. package/lib/orchestrator/grader-experiment.d.ts.map +1 -1
  251. package/lib/orchestrator/grader-experiment.js +52 -7
  252. package/lib/orchestrator/grader-experiment.js.map +1 -1
  253. package/lib/orchestrator/grading.d.ts.map +1 -1
  254. package/lib/orchestrator/grading.js +13 -6
  255. package/lib/orchestrator/grading.js.map +1 -1
  256. package/lib/orchestrator/list-runs.d.ts +1 -0
  257. package/lib/orchestrator/list-runs.d.ts.map +1 -1
  258. package/lib/orchestrator/list-runs.js +27 -6
  259. package/lib/orchestrator/list-runs.js.map +1 -1
  260. package/lib/orchestrator/optimization-lifecycle.d.ts.map +1 -1
  261. package/lib/orchestrator/optimization-lifecycle.js +3 -1
  262. package/lib/orchestrator/optimization-lifecycle.js.map +1 -1
  263. package/lib/orchestrator/orchestrator.d.ts +6 -1
  264. package/lib/orchestrator/orchestrator.d.ts.map +1 -1
  265. package/lib/orchestrator/orchestrator.js +17 -143
  266. package/lib/orchestrator/orchestrator.js.map +1 -1
  267. package/lib/orchestrator/paired-effects.d.ts +34 -0
  268. package/lib/orchestrator/paired-effects.d.ts.map +1 -0
  269. package/lib/orchestrator/paired-effects.js +158 -0
  270. package/lib/orchestrator/paired-effects.js.map +1 -0
  271. package/lib/orchestrator/provenance.d.ts +5 -1
  272. package/lib/orchestrator/provenance.d.ts.map +1 -1
  273. package/lib/orchestrator/provenance.js +16 -3
  274. package/lib/orchestrator/provenance.js.map +1 -1
  275. package/lib/orchestrator/regrade.d.ts +1 -0
  276. package/lib/orchestrator/regrade.d.ts.map +1 -1
  277. package/lib/orchestrator/run-execution.d.ts +2 -0
  278. package/lib/orchestrator/run-execution.d.ts.map +1 -1
  279. package/lib/orchestrator/run-execution.js +18 -3
  280. package/lib/orchestrator/run-execution.js.map +1 -1
  281. package/lib/orchestrator/run-lifecycle.d.ts +1 -0
  282. package/lib/orchestrator/run-lifecycle.d.ts.map +1 -1
  283. package/lib/orchestrator/run-lifecycle.js +3 -2
  284. package/lib/orchestrator/run-lifecycle.js.map +1 -1
  285. package/lib/orchestrator/suite-runs.d.ts +3 -1
  286. package/lib/orchestrator/suite-runs.d.ts.map +1 -1
  287. package/lib/orchestrator/suite-runs.js +12 -2
  288. package/lib/orchestrator/suite-runs.js.map +1 -1
  289. package/lib/orchestrator/suite-summary.d.ts.map +1 -1
  290. package/lib/orchestrator/suite-summary.js +2 -1
  291. package/lib/orchestrator/suite-summary.js.map +1 -1
  292. package/lib/previewer/server.d.ts +6 -0
  293. package/lib/previewer/server.d.ts.map +1 -1
  294. package/lib/previewer/server.js +21 -5
  295. package/lib/previewer/server.js.map +1 -1
  296. package/package.json +2 -2
  297. package/skills/lux/references/cli.md +5 -2
  298. package/skills/lux/references/eval-authoring.md +0 -7
  299. package/skills/lux-answerer/SKILL.md +1 -1
  300. package/src/answerers/claude-code/index.ts +3 -3
  301. package/src/answerers/persona/index.ts +3 -3
  302. package/src/answerers/terminal/index.ts +4 -9
  303. package/src/assertions/core/resource-checks.ts +3 -5
  304. package/src/assertions/rubric/index.ts +7 -1
  305. package/src/assertions/rubric/internal.ts +13 -8
  306. package/src/authoring.ts +331 -2
  307. package/src/cli/bin.ts +2 -12
  308. package/src/cli/command-runtime.ts +20 -9
  309. package/src/cli/init-templates.ts +223 -0
  310. package/src/cli/init.ts +38 -184
  311. package/src/cli/output-schemas.ts +106 -0
  312. package/src/cli/program.ts +621 -548
  313. package/src/cli/render/reporter.ts +25 -4
  314. package/src/cli/run-options.ts +8 -25
  315. package/src/cli/view.ts +45 -84
  316. package/src/core/annotation.ts +2 -1
  317. package/src/core/answerer.ts +9 -0
  318. package/src/core/assertion.ts +3 -0
  319. package/src/core/case.ts +5 -1
  320. package/src/core/driver.ts +38 -0
  321. package/src/core/environment.ts +22 -0
  322. package/src/core/evaluation-clusters.ts +35 -0
  323. package/src/core/experiment.ts +60 -23
  324. package/src/core/file-tree.ts +35 -0
  325. package/src/core/index.ts +1 -0
  326. package/src/core/lifecycle-fixtures.ts +4 -2
  327. package/src/core/platform-process.ts +54 -6
  328. package/src/core/run.ts +20 -0
  329. package/src/core/usage.ts +8 -0
  330. package/src/drivers/claude-code/index.ts +34 -3
  331. package/src/drivers/codex/app-events.ts +43 -33
  332. package/src/drivers/codex/exec-events.ts +258 -0
  333. package/src/drivers/codex/index.ts +37 -262
  334. package/src/drivers/cursor/README.md +57 -0
  335. package/src/drivers/cursor/acp-transport.ts +182 -0
  336. package/src/drivers/cursor/config.ts +29 -0
  337. package/src/drivers/cursor/events.ts +83 -0
  338. package/src/drivers/cursor/index.ts +316 -0
  339. package/src/drivers/cursor/interaction.ts +90 -0
  340. package/src/drivers/mcp/index.ts +16 -3
  341. package/src/drivers/shared/experiment-behavior.ts +214 -0
  342. package/src/drivers/shared/experiment-policy.ts +195 -0
  343. package/src/drivers/subprocess/index.ts +45 -6
  344. package/src/index.ts +16 -5
  345. package/src/orchestrator/annotation-loader.ts +10 -0
  346. package/src/orchestrator/annotation-store.ts +24 -3
  347. package/src/orchestrator/campaign-lifecycle.ts +5 -1
  348. package/src/orchestrator/compare.ts +16 -1
  349. package/src/orchestrator/comparison-identity.ts +1 -0
  350. package/src/orchestrator/config.ts +6 -2
  351. package/src/orchestrator/discover.ts +8 -1
  352. package/src/orchestrator/doctor.ts +40 -22
  353. package/src/orchestrator/driver-capabilities.ts +16 -0
  354. package/src/orchestrator/driver-identity.ts +59 -0
  355. package/src/orchestrator/experiment-corpus.ts +23 -28
  356. package/src/orchestrator/experiment-effects.ts +61 -0
  357. package/src/orchestrator/experiment-evaluation.ts +45 -5
  358. package/src/orchestrator/experiment-identity.ts +27 -36
  359. package/src/orchestrator/experiment-report.ts +21 -1
  360. package/src/orchestrator/experiment-runs.ts +11 -4
  361. package/src/orchestrator/experiment-runtime-identity.ts +5 -243
  362. package/src/orchestrator/experiment-selection.ts +77 -0
  363. package/src/orchestrator/experiment.ts +26 -24
  364. package/src/orchestrator/fs-read.ts +2 -32
  365. package/src/orchestrator/grader-experiment.ts +72 -8
  366. package/src/orchestrator/grading.ts +15 -6
  367. package/src/orchestrator/list-runs.ts +29 -6
  368. package/src/orchestrator/optimization-lifecycle.ts +4 -1
  369. package/src/orchestrator/orchestrator.ts +26 -183
  370. package/src/orchestrator/paired-effects.ts +209 -0
  371. package/src/orchestrator/provenance.ts +23 -3
  372. package/src/orchestrator/run-execution.ts +18 -3
  373. package/src/orchestrator/run-lifecycle.ts +4 -2
  374. package/src/orchestrator/suite-runs.ts +10 -1
  375. package/src/orchestrator/suite-summary.ts +2 -1
  376. package/src/previewer/server.ts +27 -5
  377. package/viewer/app.js +27 -6
  378. package/viewer/detail.js +22 -5
  379. package/lib/answerers/scripted/index.d.ts +0 -18
  380. package/lib/answerers/scripted/index.d.ts.map +0 -1
  381. package/lib/answerers/scripted/index.js +0 -63
  382. package/lib/answerers/scripted/index.js.map +0 -1
  383. package/lib/cli/skills.d.ts +0 -38
  384. package/lib/cli/skills.d.ts.map +0 -1
  385. package/lib/cli/skills.js +0 -84
  386. package/lib/cli/skills.js.map +0 -1
  387. package/src/answerers/scripted/index.ts +0 -78
  388. package/src/cli/skills.ts +0 -129
@@ -3,7 +3,8 @@ import { existsSync } from "node:fs";
3
3
  import { lstat, mkdir, realpath, rename, rm, stat, writeFile } from "node:fs/promises";
4
4
  import { createRequire } from "node:module";
5
5
  import { basename, dirname, extname, join, relative, resolve, sep } from "node:path";
6
- import { Command, InvalidArgumentError } from "commander";
6
+ import { fileURLToPath } from "node:url";
7
+ import { Cli, Errors, z } from "incur";
7
8
  import { createRubricAssertion } from "../assertions/rubric/index.js";
8
9
  import {
9
10
  type Candidate,
@@ -17,14 +18,18 @@ import {
17
18
  compareRuns,
18
19
  compareSuiteRuns,
19
20
  type DiscoveredCase,
21
+ DoctorReportSchema,
20
22
  discoverCases,
21
23
  doctorProject,
24
+ ExperimentCampaignPlanSchema,
25
+ ExperimentDecisionReportSchema,
26
+ ExperimentRunSchema,
22
27
  experimentCampaignMode,
23
28
  formatComparisonText,
24
29
  formatDoctorReport,
25
30
  formatExperimentDecisionMarkdown,
26
- formatExperimentRun,
27
31
  formatSuiteComparisonText,
32
+ GradingSnapshotSchema,
28
33
  hasRegression,
29
34
  hasSuiteRegression,
30
35
  loadEvalCase,
@@ -55,500 +60,565 @@ import {
55
60
  failWith,
56
61
  relativePathWithin,
57
62
  } from "./command-runtime.js";
58
- import { formatInitReport, initProject } from "./init.js";
63
+ import { initProject } from "./init.js";
64
+ import { CompareOutputSchema, ViewOutputSchema } from "./output-schemas.js";
59
65
  import { ConsoleReporter } from "./render/reporter.js";
60
66
  import {
61
- collectTags,
62
- collectValues,
63
67
  parseAnnotationAssertion,
64
- parsePositiveInt,
65
- parseUnitInterval,
66
68
  type RunOptions,
67
69
  resolveAnswererOverride,
68
70
  resolveConcurrency,
69
71
  } from "./run-options.js";
70
- import {
71
- type BundledSkillName,
72
- formatSkillsInstall,
73
- formatSkillsList,
74
- installSkills,
75
- type SkillsPlatform,
76
- } from "./skills.js";
77
- import { parsePreviewPort, viewCommand } from "./view.js";
72
+ import { type ViewOptions, viewCommand, webViewCommand } from "./view.js";
78
73
 
79
74
  const requireFromHere = createRequire(import.meta.url);
80
75
  // From lib/cli/program.js (or src/cli/program.ts under test) to the package root.
81
76
  const PKG_VERSION = (requireFromHere("../../package.json") as { version: string }).version;
82
77
 
83
- const registerProjectCommands = (program: Command): void => {
84
- program
85
- .command("init")
86
- .description("scaffold a new evals project in the current directory")
87
- .option("--force", "overwrite existing files")
88
- .option("--experiment", "also scaffold a prompt optimization campaign and split cases")
89
- .action(async (options: { force?: boolean; experiment?: boolean }) => {
90
- const cwd = process.cwd();
91
- const result = await initProject({
92
- cwd,
93
- force: options.force ?? false,
94
- experiment: options.experiment ?? false,
95
- });
96
- process.stdout.write(`${formatInitReport(result, cwd)}\n`);
97
- });
98
-
99
- const skills = program.command("skills").description("manage bundled agent skills");
100
-
101
- skills
102
- .command("list")
103
- .description("list bundled skills and supported platforms")
104
- .action(() => {
105
- process.stdout.write(`${formatSkillsList()}\n`);
106
- });
107
-
108
- skills
109
- .command("install")
110
- .description("install the Lux eval-authoring skill for Claude Code or Codex")
111
- .argument("<platform>", "claude, codex, or all")
112
- .option("--project", "install into the current project instead of the user profile")
113
- .option("--dir <path>", "install under an explicit skills directory")
114
- .option("--name <name>", "install one bundled skill")
115
- .action(
116
- async (
117
- platform: string,
118
- options: {
119
- project?: boolean;
120
- dir?: string;
121
- name?: string;
78
+ const nonEmptyString = z.string().min(1);
79
+ const positiveInteger = z.coerce.number().int().positive();
80
+ const unitInterval = z.coerce.number().min(0).max(1);
81
+ const tagList = z
82
+ .array(nonEmptyString)
83
+ .default([])
84
+ .describe("Repeatable list; each value may be comma-separated");
85
+ const normalizeTags = (values: string[]): string[] =>
86
+ values.flatMap((value) => value.split(",").map((tag) => tag.trim())).filter(Boolean);
87
+ const initOutput = z.object({
88
+ created: z.array(z.string()),
89
+ updated: z.array(z.string()),
90
+ skipped: z.array(z.object({ path: z.string(), reason: z.string() })),
91
+ experiment: z.boolean(),
92
+ });
93
+ const runOutput = z.union([
94
+ z.object({
95
+ caseCount: z.number().int(),
96
+ cases: z.array(z.object({ path: z.string(), id: z.string() })),
97
+ }),
98
+ z.object({
99
+ caseCount: z.number().int(),
100
+ tags: z.array(z.object({ tag: z.string(), count: z.number().int() })),
101
+ }),
102
+ z.object({
103
+ suiteDir: z.string(),
104
+ results: z.array(
105
+ z.object({
106
+ caseId: z.string(),
107
+ runDir: z.string().nullable(),
108
+ exitReason: z.string(),
109
+ casePassed: z.boolean().nullable(),
110
+ caseScore: z.number().nullable(),
111
+ assertionsPassed: z.number().int(),
112
+ assertionsTotal: z.number().int(),
113
+ }),
114
+ ),
115
+ }),
116
+ ]);
117
+ const experimentOutput = z.union([
118
+ ExperimentCampaignPlanSchema,
119
+ ExperimentRunSchema.extend({ experimentDir: z.string() }),
120
+ ]);
121
+ const gradingOutput = GradingSnapshotSchema.extend({ gradingPath: z.string() });
122
+ const reportOutput = z.union([
123
+ ExperimentDecisionReportSchema,
124
+ z.object({ path: z.string(), report: ExperimentDecisionReportSchema }),
125
+ ]);
126
+
127
+ const campaignArgs = z.object({ campaign: nonEmptyString.describe("Experiment campaign path") });
128
+ const campaignOptions = z.object({
129
+ runsRoot: nonEmptyString.optional().describe("Run directory root overriding lux.config.ts"),
130
+ resume: nonEmptyString.optional().describe("Interrupted experiment directory to resume"),
131
+ plan: z.boolean().optional().describe("Validate and estimate without calling models"),
132
+ });
133
+
134
+ export const buildCli = () =>
135
+ Cli.create("lux", {
136
+ description:
137
+ "Coding-agent evaluation and improvement with deterministic assertions and human-in-the-loop runs",
138
+ version: PKG_VERSION,
139
+ sync: {
140
+ include: [fileURLToPath(new URL("../../skills/*", import.meta.url))],
141
+ depth: 2,
142
+ suggestions: [
143
+ "Use Lux to inspect the eval cases in this project",
144
+ "Use Lux to diagnose the latest failed eval run",
145
+ "Use Lux to plan an evidence-based improvement experiment",
146
+ ],
147
+ },
148
+ mcp: {
149
+ title: "Lux coding-agent evaluation",
150
+ instructions:
151
+ "Inspect and plan before running model-spending commands. Apply only a verified promoted experiment champion.",
152
+ },
153
+ })
154
+ .command("init", {
155
+ description: "Scaffold a new eval project in the current directory",
156
+ options: z.object({
157
+ force: z.boolean().optional().describe("Overwrite existing files"),
158
+ experiment: z.boolean().optional().describe("Also scaffold an optimization campaign"),
159
+ driver: z
160
+ .enum(["subprocess", "claude-code", "codex", "cursor"])
161
+ .optional()
162
+ .describe("Starter driver (default: deterministic subprocess)"),
163
+ model: nonEmptyString.optional().describe("Explicit subject model; required for Cursor"),
164
+ }),
165
+ output: initOutput,
166
+ destructive: true,
167
+ mcp: { annotations: { destructiveHint: true, openWorldHint: false } },
168
+ examples: [
169
+ { description: "Scaffold a basic eval project" },
170
+ { options: { experiment: true }, description: "Scaffold an optimization project" },
171
+ ],
172
+ run: async (c) =>
173
+ c.ok(await initCommand(c.options), { cta: { commands: ["doctor", "run --list"] } }),
174
+ })
175
+ .command("doctor", {
176
+ description: "Preflight project configuration, CLIs, cases, campaigns, and budgets",
177
+ args: z.object({ campaigns: z.array(z.string()).default([]).describe("Campaign paths") }),
178
+ output: DoctorReportSchema,
179
+ mcp: {
180
+ annotations: { readOnlyHint: true, destructiveHint: false, openWorldHint: false },
181
+ },
182
+ run: async (c) =>
183
+ c.ok(await doctorCommand(c.args.campaigns), { cta: { commands: ["run --list"] } }),
184
+ })
185
+ .command("run", {
186
+ description: "Discover and run cases, filtered by name, path, or tag",
187
+ args: z.object({ filters: z.array(z.string()).default([]).describe("Case names or paths") }),
188
+ options: z.object({
189
+ tag: tagList.describe("Required case tag; repeatable and comma-separated"),
190
+ loop: positiveInteger.optional().describe("Run each matched eval this many times"),
191
+ concurrency: positiveInteger.optional().describe("Maximum cases running concurrently"),
192
+ list: z.boolean().optional().describe("List matching cases without running"),
193
+ listTags: z.boolean().optional().describe("List tags across matching cases"),
194
+ interactive: z.boolean().optional().describe("Answer agent questions in this terminal"),
195
+ claudeAnswerer: z.boolean().optional().describe("Use Claude Code as the answerer"),
196
+ skipJudge: z.boolean().optional().describe("Skip rubric LLM judge assertions"),
197
+ runsRoot: nonEmptyString.optional().describe("Run directory root"),
198
+ profile: nonEmptyString.optional().describe("Named driver and answerer profile"),
199
+ subjectRoot: nonEmptyString.optional().describe("Root bound into subjectPath values"),
200
+ verbose: z.boolean().optional().describe("Include phase timings and run directories"),
201
+ }),
202
+ alias: { tag: "t", concurrency: "j", interactive: "i", verbose: "v" },
203
+ output: runOutput,
204
+ mcp: { annotations: { readOnlyHint: false, destructiveHint: false, openWorldHint: true } },
205
+ examples: [
206
+ { options: { list: true, tag: [] }, description: "List all discovered cases" },
207
+ {
208
+ args: { filters: ["smoke"] },
209
+ options: { tag: [] },
210
+ description: "Run matching cases",
122
211
  },
123
- ) => {
124
- if (!["claude", "claude-code", "codex", "all"].includes(platform)) {
125
- throw new InvalidArgumentError("platform must be claude, codex, or all");
126
- }
127
- if (options.name && !["lux", "lux-answerer"].includes(options.name)) {
128
- throw new InvalidArgumentError("skill name must be lux or lux-answerer");
129
- }
130
- const cwd = process.cwd();
131
- const installed = await installSkills({
132
- cwd,
133
- platform: platform as SkillsPlatform,
134
- project: options.project ?? false,
135
- ...(options.dir ? { dir: options.dir } : {}),
136
- ...(options.name ? { name: options.name as BundledSkillName } : {}),
212
+ ],
213
+ run: async (c) => {
214
+ const result = await runCommand(c.args.filters, c.options as RunOptions, !c.agent);
215
+ const suiteDir = "suiteDir" in result ? result.suiteDir : undefined;
216
+ return c.ok(result, {
217
+ cta: { commands: suiteDir ? [{ command: "view", args: { runDir: suiteDir } }] : [] },
137
218
  });
138
- process.stdout.write(`${formatSkillsInstall(installed, cwd)}\n`);
139
219
  },
140
- );
141
- };
142
-
143
- const registerExperimentCommands = (program: Command): void => {
144
- program
145
- .command("doctor [campaigns...]")
146
- .description("preflight project configuration, CLIs, cases, campaigns, and budgets")
147
- .option("--json", "emit machine-readable JSON")
148
- .action(async (campaigns: string[], options: { json?: boolean }) => {
149
- const report = await doctorProject({ cwd: process.cwd(), campaigns });
150
- process.stdout.write(
151
- options.json ? `${JSON.stringify(report, null, 2)}\n` : `${formatDoctorReport(report)}\n`,
152
- );
153
- if (!report.ok) failWith(1);
220
+ })
221
+ .command("experiment", {
222
+ description: "Evaluate controlled source variants across cases and model profiles",
223
+ args: campaignArgs,
224
+ options: campaignOptions,
225
+ output: experimentOutput,
226
+ mcp: { annotations: { readOnlyHint: false, destructiveHint: false, openWorldHint: true } },
227
+ run: async (c) => experimentCommand("experiment", c.args.campaign, c.options),
228
+ })
229
+ .command("optimize", {
230
+ description: "Evolve candidates from grading feedback and promote on held-out cases",
231
+ args: campaignArgs,
232
+ options: campaignOptions,
233
+ output: experimentOutput,
234
+ destructive: true,
235
+ mcp: { annotations: { readOnlyHint: false, destructiveHint: true, openWorldHint: true } },
236
+ run: async (c) => experimentCommand("optimize", c.args.campaign, c.options),
237
+ })
238
+ .command("grade", {
239
+ description: "Re-run current graders against a persisted run",
240
+ args: z.object({ runDir: nonEmptyString.describe("Persisted run directory") }),
241
+ options: z.object({ case: nonEmptyString.optional().describe("Current case definition") }),
242
+ output: gradingOutput,
243
+ mcp: { annotations: { readOnlyHint: false, destructiveHint: false, openWorldHint: true } },
244
+ run: async (c) => gradeCommand(c.args.runDir, c.options),
245
+ })
246
+ .command("annotate", {
247
+ description: "Add or update a human grader-alignment label",
248
+ args: z.object({ runDir: nonEmptyString.describe("Persisted run directory") }),
249
+ options: z.object({
250
+ out: nonEmptyString.describe("JSON annotation dataset to create or update"),
251
+ rubric: nonEmptyString.describe("Candidate-relative declarative rubric path"),
252
+ id: nonEmptyString.optional().describe("Stable annotation ID"),
253
+ label: z.enum(["pass", "fail"]).optional().describe("Expected overall verdict"),
254
+ score: unitInterval.optional().describe("Expected score from zero to one"),
255
+ assertion: z
256
+ .array(nonEmptyString)
257
+ .default([])
258
+ .describe("Assertion label id=pass|fail[:score]"),
259
+ split: z.enum(["train", "validation", "test"]).optional().describe("Dataset split"),
260
+ tag: tagList.describe("Selection tag; repeatable and comma-separated"),
261
+ feedback: nonEmptyString.optional().describe("Human rationale for mismatches"),
262
+ annotator: nonEmptyString.optional().describe("Label author"),
263
+ reviewer: nonEmptyString.optional().describe("Independent reviewer"),
264
+ source: z.enum(["human", "synthetic"]).default("human").describe("Label provenance"),
265
+ }),
266
+ output: z.object({ id: z.string(), path: z.string(), labels: z.number().int() }),
267
+ destructive: true,
268
+ mcp: {
269
+ annotations: {
270
+ readOnlyHint: false,
271
+ destructiveHint: false,
272
+ idempotentHint: true,
273
+ openWorldHint: false,
274
+ },
275
+ },
276
+ run: async (c) => annotateCommand(c.args.runDir, c.options),
277
+ })
278
+ .command("apply", {
279
+ description: "Apply the verified promoted experiment champion",
280
+ args: z.object({
281
+ experimentDir: nonEmptyString.describe("Experiment directory"),
282
+ candidateId: nonEmptyString.optional().describe("Expected promoted candidate ID"),
283
+ }),
284
+ output: z.object({
285
+ candidateId: z.string(),
286
+ subjectRoot: z.string(),
287
+ sourceBytes: z.number(),
288
+ }),
289
+ destructive: true,
290
+ mcp: {
291
+ annotations: { readOnlyHint: false, destructiveHint: true, openWorldHint: false },
292
+ },
293
+ run: async (c) => applyCommand(c.args.experimentDir, c.args.candidateId),
294
+ })
295
+ .command("report", {
296
+ description: "Render a PR-ready report from a verified experiment record",
297
+ args: z.object({ experimentDir: nonEmptyString.describe("Experiment directory") }),
298
+ options: z.object({
299
+ out: nonEmptyString.optional().describe("Write Markdown to this path"),
300
+ force: z.boolean().optional().describe("Overwrite an existing output file"),
301
+ }),
302
+ output: reportOutput,
303
+ mcp: { annotations: { readOnlyHint: false, destructiveHint: false, openWorldHint: false } },
304
+ run: async (c) => reportCommand(c.args.experimentDir, c.options),
305
+ })
306
+ .command("view", {
307
+ description: "List past runs or show one run in detail",
308
+ args: z.object({ runDir: nonEmptyString.optional().describe("Run or suite directory") }),
309
+ options: z.object({
310
+ runsRoot: nonEmptyString.optional().describe("Run directory root"),
311
+ case: nonEmptyString.optional().describe("Filter lists to one case slug"),
312
+ suites: z.boolean().optional().describe("List logical suite runs"),
313
+ experiments: z.boolean().optional().describe("List experiment runs"),
314
+ web: z.boolean().optional().describe("Open the Lux Review web app"),
315
+ port: z.coerce.number().int().min(0).max(65535).optional().describe("Review server port"),
316
+ open: z.boolean().default(true).describe("Open a browser in web mode"),
317
+ }),
318
+ output: ViewOutputSchema,
319
+ mcp: { annotations: { readOnlyHint: true, destructiveHint: false, openWorldHint: false } },
320
+ run: (c) => {
321
+ if (c.options.web) {
322
+ if (c.formatExplicit && c.format !== "jsonl") {
323
+ throw new Errors.ParseError({
324
+ message: "--web requires streaming output; omit --format or use --format jsonl",
325
+ });
326
+ }
327
+ return webViewCommand(c.args.runDir, c.options as ViewOptions);
328
+ }
329
+ return viewCommand(c.args.runDir, c.options as ViewOptions).then((result) =>
330
+ ViewOutputSchema.parse(result),
331
+ );
332
+ },
333
+ })
334
+ .command("compare", {
335
+ description: "Diff two run or suite directories and fail on regression",
336
+ args: z.object({
337
+ runA: nonEmptyString.describe("Baseline run"),
338
+ runB: nonEmptyString.describe("Current run"),
339
+ }),
340
+ options: z.object({
341
+ passRateTolerance: unitInterval.optional().describe("Allowed per-case pass-rate decrease"),
342
+ }),
343
+ output: CompareOutputSchema,
344
+ mcp: { annotations: { readOnlyHint: true, destructiveHint: false, openWorldHint: false } },
345
+ run: async (c) =>
346
+ CompareOutputSchema.parse(await compareCommand(c.args.runA, c.args.runB, c.options)),
154
347
  });
155
348
 
156
- program
157
- .command("run [filters...]")
158
- .description("discover and run cases; filter by name/path substring or --tag")
159
- .option(
160
- "-t, --tag <tag>",
161
- "only cases whose meta.tags include this tag (repeatable, comma-separated)",
162
- collectTags,
163
- [],
164
- )
165
- .option("--loop <n>", "run each matched eval n times", parsePositiveInt)
166
- .option(
167
- "-j, --concurrency <n>",
168
- "run up to n cases at once (overrides lux.config.ts)",
169
- parsePositiveInt,
170
- )
171
- .option("--list", "print the matched cases and exit without running them")
172
- .option("--list-tags", "print the tags across the matched cases and exit")
173
- .option("-i, --interactive", "answer agent questions yourself in this terminal (HITL)")
174
- .option(
175
- "--claude-answerer",
176
- "delegate questions through the Claude plugin's lux-answerer skill",
177
- )
178
- .option("--skip-judge", "skip rubric LLM judge assertions")
179
- .option("--runs-root <path>", "run-dir root (overrides lux.config.ts)")
180
- .option("--profile <id>", "named driver/answerer profile from lux.config.ts")
181
- .option("--subject-root <path>", "root bound into subjectPath() values")
182
- .option("-v, --verbose", "add phase timings and the run dir to every case")
183
- .action(async (filters: string[], options: RunOptions) => {
184
- await runCommand(filters, options);
185
- });
349
+ const initCommand = async (options: Omit<Parameters<typeof initProject>[0], "cwd">) =>
350
+ initProject({ cwd: process.cwd(), ...options });
186
351
 
187
- program
188
- .command("experiment <campaign>")
189
- .description("evaluate controlled source variants across cases and model profiles")
190
- .option("--runs-root <path>", "run-dir root (overrides lux.config.ts)")
191
- .option("--resume <experiment-dir>", "resume an interrupted experiment ledger")
192
- .option("--plan", "validate and estimate the campaign without calling models")
193
- .option("--json", "emit machine-readable JSON")
194
- .action(
195
- async (
196
- campaign: string,
197
- options: { runsRoot?: string; resume?: string; json?: boolean; plan?: boolean },
198
- ) => {
199
- await experimentCommand("experiment", campaign, options);
200
- },
201
- );
352
+ const doctorCommand = async (campaigns: string[]) => {
353
+ const report = await doctorProject({ cwd: process.cwd(), campaigns });
354
+ if (!report.ok) {
355
+ throw new Errors.IncurError({
356
+ code: "PREFLIGHT_FAILED",
357
+ message: formatDoctorReport(report),
358
+ exitCode: 1,
359
+ });
360
+ }
361
+ return report;
202
362
  };
203
363
 
204
- const registerCandidateCommands = (program: Command): void => {
205
- program
206
- .command("grade <runDir>")
207
- .description("re-run current graders against a persisted run without rerunning the agent")
208
- .option("--case <path>", "grade with a current case definition instead of frozen case.json")
209
- .option("--json", "emit machine-readable JSON")
210
- .action(async (runDir: string, options: { case?: string; json?: boolean }) => {
211
- const directory = resolve(runDir);
212
- if (!(await ensureDirectory("grade", directory))) return;
213
- const config = await loadLuxConfig(process.cwd());
214
- const controller = abortOnInterrupt();
215
- const result = await regradeRun({
216
- runDir: directory,
217
- registry: buildRegistryFromConfig(config),
218
- graderContext: config.harness ?? null,
219
- ...(options.case ? { case: await loadEvalCase(resolve(options.case)) } : {}),
220
- abortSignal: controller.signal,
221
- });
222
- const report = result.snapshot.report;
223
- let gradingStatus = "ungraded";
224
- if (report.casePassed === true) gradingStatus = "PASS";
225
- if (report.casePassed === false) gradingStatus = "FAIL";
226
- let output: string;
227
- if (options.json) {
228
- output = `${JSON.stringify({ gradingPath: result.path, ...result.snapshot }, null, 2)}\n`;
229
- } else {
230
- const score = report.caseScore === null ? "—" : `${(report.caseScore * 100).toFixed(0)}%`;
231
- output = [
232
- `grading: ${gradingStatus}`,
233
- `score: ${score}`,
234
- `checks: ${report.assertions.filter((assertion) => assertion.passed).length}/${report.assertions.length}`,
235
- `saved: ${result.path}`,
236
- "",
237
- ].join("\n");
238
- }
239
- process.stdout.write(output);
240
- if (controller.signal.aborted) failWith(SIGINT_EXIT_CODE);
241
- else if (report.casePassed !== true) failWith(1);
364
+ const gradeCommand = async (runDir: string, options: { case?: string | undefined }) => {
365
+ const directory = resolve(runDir);
366
+ if (!(await ensureDirectory("grade", directory))) failWith(1);
367
+ const config = await loadLuxConfig(process.cwd());
368
+ using controller = abortOnInterrupt();
369
+ const result = await regradeRun({
370
+ runDir: directory,
371
+ registry: buildRegistryFromConfig(config),
372
+ graderContext: config.harness ?? null,
373
+ ...(options.case ? { case: await loadEvalCase(resolve(options.case)) } : {}),
374
+ abortSignal: controller.signal,
375
+ });
376
+ if (controller.signal.aborted) failWith(SIGINT_EXIT_CODE);
377
+ if (result.snapshot.report.casePassed !== true) {
378
+ throw new Errors.IncurError({
379
+ code: "GRADING_FAILED",
380
+ message: `The persisted run did not pass regrading. Grading saved to ${result.path}.`,
381
+ exitCode: 1,
242
382
  });
243
-
244
- program
245
- .command("annotate <runDir>")
246
- .description("add or update a human grader-alignment label for a persisted run")
247
- .requiredOption("--out <path>", "JSON annotation dataset to create or update")
248
- .requiredOption("--rubric <path>", "candidate-relative declarative .json rubric path")
249
- .option("--id <id>", "stable annotation id")
250
- .option("--label <pass|fail>", "expected overall case verdict")
251
- .option("--score <n>", "expected overall case score from 0 to 1", parseUnitInterval)
252
- .option(
253
- "--assertion <id=pass|fail[:score]>",
254
- "expected stable assertion verdict and optional score (repeatable)",
255
- collectValues,
256
- [],
257
- )
258
- .option("--split <train|validation|test>", "dataset split, stored as a tag")
259
- .option("--tag <tag>", "extra selection tag (repeatable, comma-separated)", collectTags, [])
260
- .option("--feedback <text>", "human rationale shown to the rubric optimizer on mismatch")
261
- .option("--annotator <id>", "person or system that authored the label")
262
- .option("--reviewer <id>", "independent reviewer of the label")
263
- .option("--source <human|synthetic>", "label provenance", "human")
264
- .action(
265
- async (
266
- runDir: string,
267
- options: {
268
- out: string;
269
- rubric: string;
270
- id?: string;
271
- label?: string;
272
- score?: number;
273
- assertion: string[];
274
- split?: string;
275
- tag: string[];
276
- feedback?: string;
277
- annotator?: string;
278
- reviewer?: string;
279
- source: string;
280
- },
281
- ) => {
282
- const directory = resolve(runDir);
283
- if (!(await ensureDirectory("annotate", directory))) return;
284
- if (options.label && options.label !== "pass" && options.label !== "fail") {
285
- throw new InvalidArgumentError("--label must be pass or fail");
286
- }
287
- if (options.split && !["train", "validation", "test"].includes(options.split)) {
288
- throw new InvalidArgumentError("--split must be train, validation, or test");
289
- }
290
- if (options.source !== "human" && options.source !== "synthetic") {
291
- throw new InvalidArgumentError("--source must be human or synthetic");
292
- }
293
- const parsedAssertions = options.assertion.map(parseAnnotationAssertion);
294
- const assertionIds = parsedAssertions.map(([id]) => id);
295
- if (new Set(assertionIds).size !== assertionIds.length) {
296
- throw new InvalidArgumentError("--assertion ids must not be repeated");
297
- }
298
- const assertions = Object.fromEntries(parsedAssertions);
299
- if (
300
- options.label === undefined &&
301
- options.score === undefined &&
302
- Object.keys(assertions).length === 0
303
- ) {
304
- throw new Error("lux annotate: provide --label, --score, or at least one --assertion");
305
- }
306
- const out = resolve(options.out);
307
- const run = await loadFrozenRegradableRun(directory);
308
- try {
309
- const rubric = await loadRubricDefinition(
310
- await resolveAuthoredFile(options.rubric, "rubric"),
311
- );
312
- if (rubric.id !== run.case.id) {
313
- throw new Error(
314
- `lux annotate: rubric id '${rubric.id}' does not match persisted case '${run.case.id}'`,
315
- );
316
- }
317
- const rubricIds = new Set(rubric.assertions.map((assertion) => assertion.id));
318
- const unknownIds = assertionIds.filter((assertionId) => !rubricIds.has(assertionId));
319
- if (unknownIds.length > 0) {
320
- throw new Error(`lux annotate: unknown rubric assertion ids: ${unknownIds.join(", ")}`);
321
- }
322
- const tags = [...(options.split ? [options.split] : []), ...options.tag].filter(
323
- (tag, index, all) => all.indexOf(tag) === index,
324
- );
325
- const caseId = run.case.id;
326
- const portableRunDir = (relative(dirname(out), directory) || ".").split(sep).join("/");
327
- const id =
328
- options.id ??
329
- `${caseId}-${basename(directory)}-${valueHash(portableRunDir).slice(0, 10)}`.replace(
330
- /[^a-zA-Z0-9_-]+/g,
331
- "-",
332
- );
333
- const annotation = GraderAnnotationSchema.parse({
334
- id,
335
- runDir: portableRunDir,
336
- rubricPath: options.rubric,
337
- tags,
338
- source: options.source,
339
- ...(options.annotator ? { annotator: options.annotator } : {}),
340
- labeledAt: new Date().toISOString(),
341
- ...(options.reviewer ? { reviewer: options.reviewer } : {}),
342
- expected: {
343
- ...(options.label ? { casePassed: options.label === "pass" } : {}),
344
- ...(options.score !== undefined ? { caseScore: options.score } : {}),
345
- assertions,
346
- },
347
- ...(options.feedback ? { feedback: options.feedback } : {}),
348
- });
349
- const dataset = await upsertGraderAnnotation(out, annotation);
350
- process.stdout.write(
351
- `annotated ${annotation.id} in ${out} (${dataset.annotations.length} labels)\n`,
352
- );
353
- } finally {
354
- await releaseFrozenRegradableRun(run);
355
- }
356
- },
357
- );
383
+ }
384
+ return { gradingPath: result.path, ...result.snapshot };
358
385
  };
359
386
 
360
- const registerReportingCommands = (program: Command): void => {
361
- program
362
- .command("optimize <campaign>")
363
- .description("evolve source candidates from grading feedback and promote on held-out cases")
364
- .option("--runs-root <path>", "run-dir root (overrides lux.config.ts)")
365
- .option("--resume <experiment-dir>", "resume an interrupted optimization ledger")
366
- .option("--plan", "validate and estimate the campaign without calling models")
367
- .option("--json", "emit machine-readable JSON")
368
- .action(
369
- async (
370
- campaign: string,
371
- options: { runsRoot?: string; resume?: string; json?: boolean; plan?: boolean },
372
- ) => {
373
- await experimentCommand("optimize", campaign, options);
374
- },
375
- );
387
+ type AnnotationOptions = {
388
+ out: string;
389
+ rubric: string;
390
+ id?: string | undefined;
391
+ label?: "pass" | "fail" | undefined;
392
+ score?: number | undefined;
393
+ assertion: string[];
394
+ split?: "train" | "validation" | "test" | undefined;
395
+ tag: string[];
396
+ feedback?: string | undefined;
397
+ annotator?: string | undefined;
398
+ reviewer?: string | undefined;
399
+ source: "human" | "synthetic";
400
+ };
376
401
 
377
- program
378
- .command("apply <experimentDir> [candidateId]")
379
- .description("apply the promoted experiment champion after verifying the source base hash")
380
- .action(async (experimentDir: string, candidateId?: string) => {
381
- const directory = resolve(experimentDir);
382
- const experiment = await loadExperimentRun(directory);
383
- if (!experiment) {
384
- process.stderr.write(`lux apply: experiment manifest not found at ${directory}\n`);
385
- failWith(1);
386
- return;
387
- }
388
- if (experiment.status !== "passed") {
389
- process.stderr.write(
390
- `lux apply: experiment status is '${experiment.status}', not 'passed'\n`,
391
- );
392
- failWith(1);
393
- return;
394
- }
395
- const championId = experiment.championCandidateId;
396
- if (!championId) {
397
- process.stderr.write("lux apply: passed experiment has no promoted champion\n");
398
- failWith(1);
399
- return;
400
- }
401
- if (candidateId && candidateId !== championId) {
402
- process.stderr.write(
403
- `lux apply: candidate '${candidateId}' is not the promoted champion '${championId}'\n`,
404
- );
405
- failWith(1);
406
- return;
407
- }
408
- let candidate: Candidate | undefined;
409
- try {
410
- const candidates = await verifyExperimentRunIntegrity(directory, experiment);
411
- verifyExperimentDecisionIntegrity(experiment);
412
- candidate = candidates.find((entry) => entry.id === championId);
413
- if (!candidate) {
414
- throw new Error(`champion candidate '${championId}' not found`);
415
- }
416
- } catch (error) {
417
- const message = error instanceof Error ? error.message : String(error);
418
- process.stderr.write(`lux apply: experiment integrity check failed: ${message}\n`);
419
- failWith(1);
420
- return;
421
- }
422
- await applyCandidate(experiment.subjectRoot, experiment.campaignSnapshot.subject, candidate);
423
- process.stdout.write(
424
- `applied candidate ${candidate.id} to ${experiment.subjectRoot} (${candidate.sourceBytes} bytes)\n`,
425
- );
402
+ const annotateCommand = async (runDir: string, options: AnnotationOptions) => {
403
+ const directory = resolve(runDir);
404
+ if (!(await ensureDirectory("annotate", directory))) failWith(1);
405
+ const parsedAssertions = options.assertion.map(parseAnnotationAssertion);
406
+ const assertionIds = parsedAssertions.map(([id]) => id);
407
+ if (new Set(assertionIds).size !== assertionIds.length) {
408
+ throw new Errors.ParseError({ message: "--assertion IDs must not be repeated" });
409
+ }
410
+ const assertions = Object.fromEntries(parsedAssertions);
411
+ if (
412
+ options.label === undefined &&
413
+ options.score === undefined &&
414
+ Object.keys(assertions).length === 0
415
+ ) {
416
+ throw new Errors.ParseError({
417
+ message: "Provide --label, --score, or at least one --assertion",
426
418
  });
427
-
428
- program
429
- .command("report <experimentDir>")
430
- .description("render a PR-ready promotion report from a verified experiment record")
431
- .option("--json", "emit the structured decision report as JSON")
432
- .option("--out <path>", "write the report to a file instead of stdout")
433
- .option("--force", "overwrite an existing output file")
434
- .action(
435
- async (experimentDir: string, options: { json?: boolean; out?: string; force?: boolean }) => {
436
- const directory = resolve(experimentDir);
437
- if (!(await ensureDirectory("report", directory))) return;
438
- const snapshot = await loadVerifiedExperimentDecisionSnapshot(directory);
439
- const { experiment, report } = snapshot;
440
- const rendered = options.json
441
- ? `${JSON.stringify(report, null, 2)}\n`
442
- : `${formatExperimentDecisionMarkdown(report)}\n`;
443
- if (!options.out) {
444
- process.stdout.write(rendered);
445
- return;
446
- }
447
- const out = resolve(options.out);
448
- const outputStats = await lstat(out).catch((error: NodeJS.ErrnoException) => {
449
- if (error.code === "ENOENT") return null;
450
- throw error;
451
- });
452
- if (outputStats && (outputStats.isSymbolicLink() || !outputStats.isFile())) {
453
- throw new Error("lux report: an existing --out must be a regular, non-symlink file");
454
- }
455
- const [physicalDirectory, physicalSubjectRoot, physicalOut] = await Promise.all([
456
- realpath(directory),
457
- realpath(experiment.subjectRoot),
458
- physicalDestinationPath(out),
459
- ]);
460
- assertReportDestination(
461
- physicalOut,
462
- physicalDirectory,
463
- physicalSubjectRoot,
464
- experiment.campaignSnapshot.subject.exclude,
465
- );
466
- if (outputStats && !options.force) {
467
- throw new Error(`lux report: output already exists (use --force): ${out}`);
468
- }
469
- await mkdir(dirname(out), { recursive: true });
470
- const physicalParent = await realpath(dirname(out));
471
- const finalPhysicalOut = resolve(physicalParent, basename(out));
472
- assertReportDestination(
473
- finalPhysicalOut,
474
- physicalDirectory,
475
- physicalSubjectRoot,
476
- experiment.campaignSnapshot.subject.exclude,
477
- );
478
- await writeReportSafely(finalPhysicalOut, rendered, options.force ?? false);
479
- process.stdout.write(`wrote experiment report to ${out}\n`);
419
+ }
420
+ const out = resolve(options.out);
421
+ const run = await loadFrozenRegradableRun(directory);
422
+ try {
423
+ const rubric = await loadRubricDefinition(await resolveAuthoredFile(options.rubric, "rubric"));
424
+ if (rubric.id !== run.case.id) {
425
+ throw new Error(`Rubric ID '${rubric.id}' does not match persisted case '${run.case.id}'.`);
426
+ }
427
+ const rubricIds = new Set(rubric.assertions.map((assertion) => assertion.id));
428
+ const unknownIds = assertionIds.filter((assertionId) => !rubricIds.has(assertionId));
429
+ if (unknownIds.length > 0) {
430
+ throw new Errors.ParseError({
431
+ message: `Unknown rubric assertion IDs: ${unknownIds.join(", ")}`,
432
+ });
433
+ }
434
+ const annotationTags = [
435
+ ...(options.split ? [options.split] : []),
436
+ ...normalizeTags(options.tag),
437
+ ].filter((tag, index, all) => all.indexOf(tag) === index);
438
+ const portableRunDir = (relative(dirname(out), directory) || ".").split(sep).join("/");
439
+ const id =
440
+ options.id ??
441
+ `${run.case.id}-${basename(directory)}-${valueHash(portableRunDir).slice(0, 10)}`.replace(
442
+ /[^a-zA-Z0-9_-]+/g,
443
+ "-",
444
+ );
445
+ const annotation = GraderAnnotationSchema.parse({
446
+ id,
447
+ runDir: portableRunDir,
448
+ rubricPath: options.rubric,
449
+ tags: annotationTags,
450
+ source: options.source,
451
+ ...(options.annotator ? { annotator: options.annotator } : {}),
452
+ labeledAt: new Date().toISOString(),
453
+ ...(options.reviewer ? { reviewer: options.reviewer } : {}),
454
+ expected: {
455
+ ...(options.label ? { casePassed: options.label === "pass" } : {}),
456
+ ...(options.score !== undefined ? { caseScore: options.score } : {}),
457
+ assertions,
480
458
  },
481
- );
459
+ ...(options.feedback ? { feedback: options.feedback } : {}),
460
+ });
461
+ const dataset = await upsertGraderAnnotation(out, annotation);
462
+ return { id: annotation.id, path: out, labels: dataset.annotations.length };
463
+ } finally {
464
+ await releaseFrozenRegradableRun(run);
465
+ }
466
+ };
482
467
 
483
- program
484
- .command("view [runDir]")
485
- .description("list past runs, or show one run's detail when given a runDir")
486
- .option("--runs-root <path>", "run-dir root (defaults to lux.config.ts)")
487
- .option("--case <id>", "filter the run list to one case slug (list mode only)")
488
- .option("--suites", "list logical suite runs instead of individual runs")
489
- .option("--experiments", "list experiment runs instead of individual runs")
490
- .option("--web", "open the local Lux Review web app")
491
- .option("--port <n>", "listen on a specific local port in web mode", parsePreviewPort)
492
- .option("--no-open", "start web mode without opening a browser")
493
- .option("--json", "emit machine-readable JSON")
494
- .action(viewCommand);
495
-
496
- program
497
- .command("compare <runA> <runB>")
498
- .description("diff two run or suite directories")
499
- .option(
500
- "--pass-rate-tolerance <fraction>",
501
- "allowed per-case pass-rate decrease for suite comparison",
502
- parseUnitInterval,
503
- )
504
- .action(async (runA: string, runB: string, options: { passRateTolerance?: number }) => {
505
- const pathA = resolve(runA);
506
- const pathB = resolve(runB);
507
- // Validate both up front so one typo'd path in CI cannot read as
508
- // "0 regressed"; report every bad path, not just the first.
509
- const okA = await ensureDirectory("compare", pathA);
510
- const okB = await ensureDirectory("compare", pathB);
511
- if (!okA || !okB) return;
512
- const [suiteA, suiteB] = await Promise.all([loadSuiteRun(pathA), loadSuiteRun(pathB)]);
513
- if (Boolean(suiteA) !== Boolean(suiteB)) {
514
- process.stderr.write("lux compare: both paths must be runs or both must be suites\n");
515
- failWith(1);
516
- return;
517
- }
518
- if (suiteA && suiteB) {
519
- const comparison = await compareSuiteRuns(pathA, pathB, {
520
- ...(options.passRateTolerance !== undefined
521
- ? { passRateTolerance: options.passRateTolerance }
522
- : {}),
523
- });
524
- process.stdout.write(`${formatSuiteComparisonText(comparison)}\n`);
525
- if (hasSuiteRegression(comparison)) failWith(1);
526
- return;
527
- }
528
- const cmp = await compareRuns({ pathA, pathB });
529
- process.stdout.write(`${formatComparisonText(cmp)}\n`);
530
- if (hasRegression(cmp)) failWith(1);
468
+ const applyCommand = async (experimentDir: string, candidateId?: string) => {
469
+ const directory = resolve(experimentDir);
470
+ const experiment = await loadExperimentRun(directory);
471
+ if (!experiment) {
472
+ throw new Errors.IncurError({
473
+ code: "EXPERIMENT_NOT_FOUND",
474
+ message: `Experiment manifest not found at ${directory}.`,
475
+ exitCode: 1,
476
+ });
477
+ }
478
+ if (experiment.status !== "passed") {
479
+ throw new Errors.IncurError({
480
+ code: "EXPERIMENT_NOT_PROMOTED",
481
+ message: `Experiment status is '${experiment.status}', not 'passed'.`,
482
+ exitCode: 1,
483
+ });
484
+ }
485
+ const championId = experiment.championCandidateId;
486
+ if (!championId) {
487
+ throw new Errors.IncurError({
488
+ code: "CHAMPION_NOT_FOUND",
489
+ message: "The passed experiment has no promoted champion.",
490
+ exitCode: 1,
491
+ });
492
+ }
493
+ if (candidateId && candidateId !== championId) {
494
+ throw new Errors.IncurError({
495
+ code: "CANDIDATE_NOT_PROMOTED",
496
+ message: `Candidate '${candidateId}' is not the promoted champion '${championId}'.`,
497
+ exitCode: 1,
531
498
  });
499
+ }
500
+ let candidate: Candidate | undefined;
501
+ try {
502
+ const candidates = await verifyExperimentRunIntegrity(directory, experiment);
503
+ verifyExperimentDecisionIntegrity(experiment);
504
+ candidate = candidates.find((entry) => entry.id === championId);
505
+ if (!candidate) throw new Error(`Champion candidate '${championId}' not found.`);
506
+ } catch (error) {
507
+ throw new Errors.IncurError({
508
+ code: "EXPERIMENT_INTEGRITY_FAILED",
509
+ message: error instanceof Error ? error.message : String(error),
510
+ exitCode: 1,
511
+ });
512
+ }
513
+ await applyCandidate(experiment.subjectRoot, experiment.campaignSnapshot.subject, candidate);
514
+ return {
515
+ candidateId: candidate.id,
516
+ subjectRoot: experiment.subjectRoot,
517
+ sourceBytes: candidate.sourceBytes,
518
+ };
532
519
  };
533
520
 
534
- export const buildProgram = (): Command => {
535
- const program = new Command();
536
- program
537
- .name("lux")
538
- .description(
539
- "Coding-agent evaluation and improvement with deterministic assertions and human-in-the-loop runs",
521
+ const reportCommand = async (
522
+ experimentDir: string,
523
+ options: { out?: string | undefined; force?: boolean | undefined },
524
+ ) => {
525
+ const directory = resolve(experimentDir);
526
+ if (!(await ensureDirectory("report", directory))) failWith(1);
527
+ const snapshot = await loadVerifiedExperimentDecisionSnapshot(directory);
528
+ const { experiment, report } = snapshot;
529
+ if (!options.out) return report;
530
+ const out = resolve(options.out);
531
+ const outputStats = await lstat(out).catch((error: NodeJS.ErrnoException) => {
532
+ if (error.code === "ENOENT") return null;
533
+ throw error;
534
+ });
535
+ if (outputStats && (outputStats.isSymbolicLink() || !outputStats.isFile())) {
536
+ throw new Error("An existing --out must be a regular, non-symlink file.");
537
+ }
538
+ const [physicalDirectory, physicalSubjectRoot, physicalOut] = await Promise.all([
539
+ realpath(directory),
540
+ realpath(experiment.subjectRoot),
541
+ physicalDestinationPath(out),
542
+ ]);
543
+ assertReportDestination(
544
+ physicalOut,
545
+ physicalDirectory,
546
+ physicalSubjectRoot,
547
+ experiment.campaignSnapshot.subject.exclude,
548
+ );
549
+ if (outputStats && !options.force) {
550
+ throw new Error(`Output already exists (use --force): ${out}`);
551
+ }
552
+ await mkdir(dirname(out), { recursive: true });
553
+ const physicalParent = await realpath(dirname(out));
554
+ const finalPhysicalOut = resolve(physicalParent, basename(out));
555
+ assertReportDestination(
556
+ finalPhysicalOut,
557
+ physicalDirectory,
558
+ physicalSubjectRoot,
559
+ experiment.campaignSnapshot.subject.exclude,
560
+ );
561
+ await writeReportSafely(
562
+ finalPhysicalOut,
563
+ `${formatExperimentDecisionMarkdown(report)}\n`,
564
+ options.force ?? false,
565
+ );
566
+ return { path: out, report };
567
+ };
568
+
569
+ const compareCommand = async (
570
+ runA: string,
571
+ runB: string,
572
+ options: { passRateTolerance?: number | undefined },
573
+ ) => {
574
+ const pathA = resolve(runA);
575
+ const pathB = resolve(runB);
576
+ const invalid = (
577
+ await Promise.all(
578
+ [pathA, pathB].map(async (path) =>
579
+ existsSync(path) && (await stat(path)).isDirectory() ? undefined : path,
580
+ ),
540
581
  )
541
- .version(PKG_VERSION)
542
- .addHelpText(
543
- "after",
544
- "\nAgent skills:\n $ lux skills list\n $ lux skills install codex --project\n $ lux skills install claude --project",
545
- );
546
- registerProjectCommands(program);
547
- registerExperimentCommands(program);
548
- registerCandidateCommands(program);
549
- registerReportingCommands(program);
550
- program.action(() => program.outputHelp());
551
- return program;
582
+ ).filter((path): path is string => path !== undefined);
583
+ if (invalid.length > 0) {
584
+ throw new Errors.IncurError({
585
+ code: "RUN_DIRECTORY_NOT_FOUND",
586
+ message: invalid.map((path) => `${path} is not a directory`).join("; "),
587
+ exitCode: 1,
588
+ });
589
+ }
590
+ const [suiteA, suiteB] = await Promise.all([loadSuiteRun(pathA), loadSuiteRun(pathB)]);
591
+ if (Boolean(suiteA) !== Boolean(suiteB)) {
592
+ throw new Errors.IncurError({
593
+ code: "INCOMPATIBLE_RUN_TYPES",
594
+ message: "Both paths must be runs or both must be suites.",
595
+ exitCode: 1,
596
+ });
597
+ }
598
+ if (suiteA && suiteB) {
599
+ const comparison = await compareSuiteRuns(pathA, pathB, {
600
+ ...(options.passRateTolerance !== undefined
601
+ ? { passRateTolerance: options.passRateTolerance }
602
+ : {}),
603
+ });
604
+ if (hasSuiteRegression(comparison)) {
605
+ throw new Errors.IncurError({
606
+ code: "REGRESSION_DETECTED",
607
+ message: formatSuiteComparisonText(comparison),
608
+ exitCode: 1,
609
+ });
610
+ }
611
+ return comparison;
612
+ }
613
+ const comparison = await compareRuns({ pathA, pathB });
614
+ if (hasRegression(comparison)) {
615
+ throw new Errors.IncurError({
616
+ code: "REGRESSION_DETECTED",
617
+ message: formatComparisonText(comparison),
618
+ exitCode: 1,
619
+ });
620
+ }
621
+ return comparison;
552
622
  };
553
623
 
554
624
  const caseIdOf = (casePath: string): string => basename(casePath, extname(casePath));
@@ -708,15 +778,19 @@ const expandLoop = (base: SuiteItem[], loop: number): SuiteItem[] =>
708
778
  const experimentCommand = async (
709
779
  mode: "experiment" | "optimize",
710
780
  campaignPath: string,
711
- options: { runsRoot?: string; resume?: string; json?: boolean; plan?: boolean },
712
- ): Promise<void> => {
781
+ options: {
782
+ runsRoot?: string | undefined;
783
+ resume?: string | undefined;
784
+ plan?: boolean | undefined;
785
+ },
786
+ ) => {
713
787
  const cwd = process.cwd();
714
788
  const config = await loadLuxConfig(cwd);
715
789
  const campaign = await resolveExperimentCampaignPath(config.rootDir, campaignPath);
716
790
  const loaded = await loadExperimentCampaign(campaign);
717
791
  experimentCampaignMode(campaign, loaded.campaign, mode);
718
792
  const registry = buildRegistryFromConfig(config);
719
- const controller = abortOnInterrupt();
793
+ using controller = abortOnInterrupt();
720
794
  const execution = {
721
795
  config,
722
796
  registry,
@@ -727,64 +801,52 @@ const experimentCommand = async (
727
801
  };
728
802
  if (options.plan) {
729
803
  const plan = await planExperimentCampaign(execution, mode);
730
- let output: string;
731
- if (options.json) {
732
- output = `${JSON.stringify(plan, null, 2)}\n`;
733
- } else {
734
- const projectedStatus = plan.fitsProjectedCallBudgets ? "yes" : "no";
735
- output = [
736
- `campaign: ${plan.campaignId}`,
737
- `mode: ${plan.mode}`,
738
- `evaluation: ${plan.evaluationKind}`,
739
- `profiles: ${plan.profiles.join(", ")}`,
740
- `splits: ${plan.splits.map((split) => `${split.split}=${split.examples}`).join(", ")}`,
741
- `components: ${plan.mutableComponents.length}`,
742
- `candidates: ${plan.projected.candidates}`,
743
- `metric calls: ${plan.projected.metricCalls}/${plan.budget.maxMetricCalls}`,
744
- `judge calls: ${plan.projected.judgeCalls}/${plan.budget.maxJudgeCalls}`,
745
- `proposal calls:${plan.projected.proposalCalls}/${plan.budget.maxProposalCalls}`,
746
- `unpriced calls: observed at runtime (cap ${plan.budget.maxUnpricedModelCalls})`,
747
- `projected work within budget: ${projectedStatus}`,
748
- ...plan.warnings.map((warning) => `warning: ${warning}`),
749
- "",
750
- ].join("\n");
804
+ if (!plan.fitsProjectedCallBudgets) {
805
+ throw new Errors.IncurError({
806
+ code: "EXPERIMENT_BUDGET_EXCEEDED",
807
+ message: "The projected experiment work exceeds its configured call budgets.",
808
+ exitCode: 1,
809
+ });
751
810
  }
752
- process.stdout.write(output);
753
- if (!plan.fitsProjectedCallBudgets) failWith(1);
754
- return;
811
+ return plan;
755
812
  }
756
813
  const result =
757
814
  mode === "optimize"
758
815
  ? await optimizeExperimentCampaign(execution)
759
816
  : await runExperimentCampaign(execution);
760
- process.stdout.write(
761
- options.json
762
- ? `${JSON.stringify({ experimentDir: result.experimentDir, ...result.manifest }, null, 2)}\n`
763
- : `${formatExperimentRun(result.experimentDir, result.manifest)}\n`,
764
- );
765
817
  if (controller.signal.aborted) failWith(SIGINT_EXIT_CODE);
766
- else if (result.manifest.status !== "passed") failWith(1);
818
+ if (result.manifest.status !== "passed") {
819
+ throw new Errors.IncurError({
820
+ code: "EXPERIMENT_FAILED",
821
+ message: `Experiment ${result.manifest.experimentId} finished with status '${result.manifest.status}'.`,
822
+ exitCode: 1,
823
+ });
824
+ }
825
+ return { experimentDir: result.experimentDir, ...result.manifest };
767
826
  };
768
827
 
769
- const runCommand = async (filters: string[], options: RunOptions): Promise<void> => {
828
+ const runCommand = async (filters: string[], options: RunOptions, human = false) => {
770
829
  if (options.interactive && options.claudeAnswerer) {
771
- throw new Error("lux run: --interactive and --claude-answerer are mutually exclusive");
830
+ throw new Errors.ParseError({
831
+ message: "--interactive and --claude-answerer are mutually exclusive",
832
+ });
772
833
  }
773
834
  const cwd = process.cwd();
774
835
  const config = await loadLuxConfig(cwd);
775
836
  const profile = options.profile ? config.profiles?.[options.profile] : undefined;
776
837
  if (options.profile && !profile) {
777
838
  const available = Object.keys(config.profiles ?? {}).sort();
778
- process.stderr.write(
779
- `lux: unknown profile '${options.profile}'${available.length > 0 ? `; available: ${available.join(", ")}` : ""}\n`,
780
- );
781
- failWith(1);
782
- return;
839
+ throw new Errors.IncurError({
840
+ code: "PROFILE_NOT_FOUND",
841
+ message: `Unknown profile '${options.profile}'${available.length > 0 ? `; available: ${available.join(", ")}` : ""}.`,
842
+ exitCode: 1,
843
+ });
783
844
  }
784
845
  const harness = profile?.harness ?? config.harness;
785
846
  const registry = buildRegistryFromConfig(config).withAssertion(createRubricAssertion(harness));
786
847
  const defaultDriver = profile?.driver ?? config.defaultDriver;
787
848
  const orchestrator = new Orchestrator({
849
+ environment: profile?.environment ?? config.environment,
788
850
  registry,
789
851
  defaultAnswerer: profile?.answerer ?? config.defaultAnswerer,
790
852
  ...(defaultDriver ? { defaultDriver } : {}),
@@ -802,23 +864,26 @@ const runCommand = async (filters: string[], options: RunOptions): Promise<void>
802
864
  // broken config, not an empty suite — never mask it behind "no cases
803
865
  // matched" or a lucky explicit-file run.
804
866
  if (config.casesRoot && !existsSync(config.casesRoot)) {
805
- process.stderr.write(`lux: casesRoot does not exist: ${config.casesRoot}\n`);
806
- failWith(1);
807
- return;
867
+ throw new Errors.IncurError({
868
+ code: "CASES_ROOT_NOT_FOUND",
869
+ message: `casesRoot does not exist: ${config.casesRoot}`,
870
+ exitCode: 1,
871
+ });
808
872
  }
809
873
 
810
874
  const { explicit, patterns } = partitionFilters(filters, cwd);
811
- const tags = options.tag ?? [];
875
+ const tags = normalizeTags(options.tag ?? []);
812
876
  const hasFilter = explicit.length > 0 || patterns.length > 0 || tags.length > 0;
813
877
 
814
878
  // Explicitly named files need no discovery, so they run even without a
815
879
  // project; anything else would need a tree walk with nowhere safe to walk.
816
880
  if (config.configPath === null && !existsSync(root) && explicit.length === 0) {
817
- process.stderr.write(
818
- `lux: no lux.config.ts or cases/ directory in ${cwd} — run \`lux init\` to scaffold a project, or name a case file directly\n`,
819
- );
820
- failWith(1);
821
- return;
881
+ throw new Errors.IncurError({
882
+ code: "PROJECT_NOT_INITIALIZED",
883
+ message: `No lux.config.ts or cases/ directory in ${cwd}.`,
884
+ hint: "Run `lux init` or name a case file directly.",
885
+ exitCode: 1,
886
+ });
822
887
  }
823
888
 
824
889
  const { cases, problems } = existsSync(root)
@@ -833,35 +898,32 @@ const runCommand = async (filters: string[], options: RunOptions): Promise<void>
833
898
  // An empty selection fails every mode the same way — running, `--list`, and
834
899
  // `--list-tags` all exit 1, so a typo'd filter can never read as a clean pass.
835
900
  if (selected.length === 0) {
836
- process.stderr.write(
837
- hasFilter
838
- ? "lux: no cases matched the given filters\n"
839
- : `lux: no cases discovered under ${root}\n`,
840
- );
841
- failWith(1);
842
- return;
901
+ throw new Errors.IncurError({
902
+ code: "NO_CASES_MATCHED",
903
+ message: hasFilter
904
+ ? "No cases matched the given filters."
905
+ : `No cases were discovered under ${root}.`,
906
+ exitCode: 1,
907
+ });
843
908
  }
844
909
 
845
910
  // List the tag vocabulary of the selected cases — "which --tag values exist?".
846
911
  if (options.listTags) {
847
912
  const counts = new Map<string, number>();
848
913
  for (const c of selected) for (const t of c.tags) counts.set(t, (counts.get(t) ?? 0) + 1);
849
- for (const [tag, n] of [...counts].sort((a, b) => a[0].localeCompare(b[0]))) {
850
- process.stdout.write(`${tag}\t${n}\n`);
851
- }
852
- process.stderr.write(
853
- `lux: ${counts.size} tag${counts.size === 1 ? "" : "s"} across ${selected.length} cases\n`,
854
- );
855
- return;
914
+ return {
915
+ caseCount: selected.length,
916
+ tags: [...counts]
917
+ .sort((a, b) => a[0].localeCompare(b[0]))
918
+ .map(([tag, count]) => ({ tag, count })),
919
+ };
856
920
  }
857
921
 
858
922
  const base: SuiteItem[] = selected.map((c) => ({ path: c.path, id: c.id }));
859
923
 
860
924
  // Collect-only: answer "what would run?" without spawning a single agent.
861
925
  if (options.list) {
862
- for (const item of base) process.stdout.write(`${item.id}\n`);
863
- process.stderr.write(`lux: ${base.length} case${base.length === 1 ? "" : "s"} matched\n`);
864
- return;
926
+ return { caseCount: base.length, cases: base };
865
927
  }
866
928
 
867
929
  const loop = options.loop ?? 1;
@@ -870,9 +932,9 @@ const runCommand = async (filters: string[], options: RunOptions): Promise<void>
870
932
  const { concurrency, note } = resolveConcurrency(options, config.concurrency);
871
933
  if (note) process.stderr.write(note);
872
934
 
873
- const reporter = new ConsoleReporter({ verbose: options.verbose ?? false });
874
- const controller = abortOnInterrupt();
875
- reporter.begin({
935
+ const reporter = human ? new ConsoleReporter({ verbose: options.verbose ?? false }) : undefined;
936
+ using controller = abortOnInterrupt();
937
+ reporter?.begin({
876
938
  title: suiteTitle(patterns, tags, explicit, loop),
877
939
  caseIds: items.map((i) => i.id),
878
940
  runsRoot,
@@ -886,28 +948,39 @@ const runCommand = async (filters: string[], options: RunOptions): Promise<void>
886
948
  fixturesRoot: config.fixturesRoot,
887
949
  answererOverride: resolveAnswererOverride(options),
888
950
  skipJudge: options.skipJudge,
889
- observer: reporter,
951
+ ...(reporter ? { observer: reporter } : {}),
890
952
  abortSignal: controller.signal,
891
953
  concurrency,
892
954
  subjectRoot,
893
955
  profileId: options.profile,
894
956
  });
895
957
  results = suite.results;
896
- reporter.setSuiteDir(suite.suiteDir);
958
+ reporter?.setSuiteDir(suite.suiteDir);
959
+ if (controller.signal.aborted) failWith(SIGINT_EXIT_CODE);
960
+ if (results.some(isFailure)) {
961
+ throw new Errors.IncurError({
962
+ code: "SUITE_FAILED",
963
+ message: `${results.filter(isFailure).length} of ${results.length} case attempts failed. Suite: ${suite.suiteDir}`,
964
+ exitCode: 1,
965
+ });
966
+ }
967
+ return {
968
+ suiteDir: suite.suiteDir,
969
+ results: results.map((result) => ({
970
+ caseId: result.caseId,
971
+ runDir: result.runDir ?? null,
972
+ exitReason: result.exitReason,
973
+ casePassed: result.casePassed,
974
+ caseScore: result.result?.grading?.caseScore ?? null,
975
+ assertionsPassed:
976
+ result.result?.grading?.assertions.filter((assertion) => assertion.passed).length ?? 0,
977
+ assertionsTotal: result.result?.grading?.assertions.length ?? 0,
978
+ })),
979
+ };
897
980
  } finally {
898
- reporter.finish();
981
+ reporter?.finish();
899
982
  }
900
-
901
- if (controller.signal.aborted) failWith(SIGINT_EXIT_CODE);
902
- else if (results.some(isFailure)) failWith(1);
903
983
  };
904
984
 
905
985
  // Internal exports for testing
906
- export {
907
- expandLoop,
908
- isFailure,
909
- parsePositiveInt,
910
- partitionFilters,
911
- resolveConcurrency,
912
- selectSuite,
913
- };
986
+ export { expandLoop, isFailure, partitionFilters, resolveConcurrency, selectSuite };