@prismatic-io/lux 0.0.2-preview.14 → 0.0.2-preview.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (388) hide show
  1. package/lib/answerers/claude-code/index.d.ts +14 -2
  2. package/lib/answerers/claude-code/index.d.ts.map +1 -1
  3. package/lib/answerers/claude-code/index.js +3 -3
  4. package/lib/answerers/claude-code/index.js.map +1 -1
  5. package/lib/answerers/persona/index.d.ts +28 -2
  6. package/lib/answerers/persona/index.d.ts.map +1 -1
  7. package/lib/answerers/persona/index.js +3 -3
  8. package/lib/answerers/persona/index.js.map +1 -1
  9. package/lib/answerers/terminal/index.d.ts +8 -2
  10. package/lib/answerers/terminal/index.d.ts.map +1 -1
  11. package/lib/answerers/terminal/index.js +3 -2
  12. package/lib/answerers/terminal/index.js.map +1 -1
  13. package/lib/assertions/core/resource-checks.d.ts.map +1 -1
  14. package/lib/assertions/core/resource-checks.js +3 -5
  15. package/lib/assertions/core/resource-checks.js.map +1 -1
  16. package/lib/assertions/rubric/index.d.ts.map +1 -1
  17. package/lib/assertions/rubric/index.js +7 -1
  18. package/lib/assertions/rubric/index.js.map +1 -1
  19. package/lib/assertions/rubric/internal.d.ts +16 -5
  20. package/lib/assertions/rubric/internal.d.ts.map +1 -1
  21. package/lib/assertions/rubric/internal.js +10 -6
  22. package/lib/assertions/rubric/internal.js.map +1 -1
  23. package/lib/authoring.d.ts +32 -2
  24. package/lib/authoring.d.ts.map +1 -1
  25. package/lib/authoring.js +33 -1
  26. package/lib/authoring.js.map +1 -1
  27. package/lib/cli/bin.js +2 -13
  28. package/lib/cli/bin.js.map +1 -1
  29. package/lib/cli/command-runtime.d.ts +3 -3
  30. package/lib/cli/command-runtime.d.ts.map +1 -1
  31. package/lib/cli/command-runtime.js +17 -6
  32. package/lib/cli/command-runtime.js.map +1 -1
  33. package/lib/cli/init-templates.d.ts +11 -0
  34. package/lib/cli/init-templates.d.ts.map +1 -0
  35. package/lib/cli/init-templates.js +207 -0
  36. package/lib/cli/init-templates.js.map +1 -0
  37. package/lib/cli/init.d.ts +7 -0
  38. package/lib/cli/init.d.ts.map +1 -1
  39. package/lib/cli/init.js +23 -173
  40. package/lib/cli/init.js.map +1 -1
  41. package/lib/cli/output-schemas.d.ts +1399 -0
  42. package/lib/cli/output-schemas.d.ts.map +1 -0
  43. package/lib/cli/output-schemas.js +91 -0
  44. package/lib/cli/output-schemas.js.map +1 -0
  45. package/lib/cli/program.d.ts +135 -4
  46. package/lib/cli/program.d.ts.map +1 -1
  47. package/lib/cli/program.js +725 -520
  48. package/lib/cli/program.js.map +1 -1
  49. package/lib/cli/render/reporter.d.ts.map +1 -1
  50. package/lib/cli/render/reporter.js +24 -4
  51. package/lib/cli/render/reporter.js.map +1 -1
  52. package/lib/cli/run-options.d.ts +0 -4
  53. package/lib/cli/run-options.d.ts.map +1 -1
  54. package/lib/cli/run-options.js +8 -19
  55. package/lib/cli/run-options.js.map +1 -1
  56. package/lib/cli/view.d.ts +5 -3
  57. package/lib/cli/view.d.ts.map +1 -1
  58. package/lib/cli/view.js +120 -78
  59. package/lib/cli/view.js.map +1 -1
  60. package/lib/core/annotation.d.ts +6 -2
  61. package/lib/core/annotation.d.ts.map +1 -1
  62. package/lib/core/annotation.js +2 -1
  63. package/lib/core/annotation.js.map +1 -1
  64. package/lib/core/answerer.d.ts +12 -0
  65. package/lib/core/answerer.d.ts.map +1 -1
  66. package/lib/core/answerer.js +2 -0
  67. package/lib/core/answerer.js.map +1 -1
  68. package/lib/core/assertion.d.ts +3 -0
  69. package/lib/core/assertion.d.ts.map +1 -1
  70. package/lib/core/assertion.js +3 -0
  71. package/lib/core/assertion.js.map +1 -1
  72. package/lib/core/case.d.ts +24 -2
  73. package/lib/core/case.d.ts.map +1 -1
  74. package/lib/core/case.js +5 -1
  75. package/lib/core/case.js.map +1 -1
  76. package/lib/core/driver.d.ts +40 -0
  77. package/lib/core/driver.d.ts.map +1 -1
  78. package/lib/core/driver.js +12 -0
  79. package/lib/core/driver.js.map +1 -1
  80. package/lib/core/environment.d.ts +18 -0
  81. package/lib/core/environment.d.ts.map +1 -0
  82. package/lib/core/environment.js +21 -0
  83. package/lib/core/environment.js.map +1 -0
  84. package/lib/core/evaluation-clusters.d.ts +17 -0
  85. package/lib/core/evaluation-clusters.d.ts.map +1 -0
  86. package/lib/core/evaluation-clusters.js +23 -0
  87. package/lib/core/evaluation-clusters.js.map +1 -0
  88. package/lib/core/experiment.d.ts +11 -0
  89. package/lib/core/experiment.d.ts.map +1 -1
  90. package/lib/core/experiment.js +31 -17
  91. package/lib/core/experiment.js.map +1 -1
  92. package/lib/core/file-tree.d.ts +7 -0
  93. package/lib/core/file-tree.d.ts.map +1 -0
  94. package/lib/core/file-tree.js +35 -0
  95. package/lib/core/file-tree.js.map +1 -0
  96. package/lib/core/index.d.ts +1 -0
  97. package/lib/core/index.d.ts.map +1 -1
  98. package/lib/core/index.js +1 -0
  99. package/lib/core/index.js.map +1 -1
  100. package/lib/core/lifecycle-fixtures.d.ts.map +1 -1
  101. package/lib/core/lifecycle-fixtures.js +4 -2
  102. package/lib/core/lifecycle-fixtures.js.map +1 -1
  103. package/lib/core/platform-process.d.ts +5 -1
  104. package/lib/core/platform-process.d.ts.map +1 -1
  105. package/lib/core/platform-process.js +48 -10
  106. package/lib/core/platform-process.js.map +1 -1
  107. package/lib/core/run.d.ts +93 -1
  108. package/lib/core/run.d.ts.map +1 -1
  109. package/lib/core/run.js +16 -0
  110. package/lib/core/run.js.map +1 -1
  111. package/lib/core/usage.d.ts +5 -0
  112. package/lib/core/usage.d.ts.map +1 -1
  113. package/lib/core/usage.js +3 -0
  114. package/lib/core/usage.js.map +1 -1
  115. package/lib/drivers/claude-code/index.d.ts +65 -2
  116. package/lib/drivers/claude-code/index.d.ts.map +1 -1
  117. package/lib/drivers/claude-code/index.js +30 -3
  118. package/lib/drivers/claude-code/index.js.map +1 -1
  119. package/lib/drivers/codex/app-events.d.ts.map +1 -1
  120. package/lib/drivers/codex/app-events.js +41 -30
  121. package/lib/drivers/codex/app-events.js.map +1 -1
  122. package/lib/drivers/codex/exec-events.d.ts +4 -0
  123. package/lib/drivers/codex/exec-events.d.ts.map +1 -0
  124. package/lib/drivers/codex/exec-events.js +230 -0
  125. package/lib/drivers/codex/exec-events.js.map +1 -0
  126. package/lib/drivers/codex/index.d.ts +76 -4
  127. package/lib/drivers/codex/index.d.ts.map +1 -1
  128. package/lib/drivers/codex/index.js +35 -232
  129. package/lib/drivers/codex/index.js.map +1 -1
  130. package/lib/drivers/cursor/acp-transport.d.ts +38 -0
  131. package/lib/drivers/cursor/acp-transport.d.ts.map +1 -0
  132. package/lib/drivers/cursor/acp-transport.js +152 -0
  133. package/lib/drivers/cursor/acp-transport.js.map +1 -0
  134. package/lib/drivers/cursor/config.d.ts +22 -0
  135. package/lib/drivers/cursor/config.d.ts.map +1 -0
  136. package/lib/drivers/cursor/config.js +26 -0
  137. package/lib/drivers/cursor/config.js.map +1 -0
  138. package/lib/drivers/cursor/events.d.ts +13 -0
  139. package/lib/drivers/cursor/events.d.ts.map +1 -0
  140. package/lib/drivers/cursor/events.js +73 -0
  141. package/lib/drivers/cursor/events.js.map +1 -0
  142. package/lib/drivers/cursor/index.d.ts +37 -0
  143. package/lib/drivers/cursor/index.d.ts.map +1 -0
  144. package/lib/drivers/cursor/index.js +276 -0
  145. package/lib/drivers/cursor/index.js.map +1 -0
  146. package/lib/drivers/cursor/interaction.d.ts +7 -0
  147. package/lib/drivers/cursor/interaction.d.ts.map +1 -0
  148. package/lib/drivers/cursor/interaction.js +66 -0
  149. package/lib/drivers/cursor/interaction.js.map +1 -0
  150. package/lib/drivers/mcp/index.d.ts +35 -2
  151. package/lib/drivers/mcp/index.d.ts.map +1 -1
  152. package/lib/drivers/mcp/index.js +16 -3
  153. package/lib/drivers/mcp/index.js.map +1 -1
  154. package/lib/drivers/shared/experiment-behavior.d.ts +13 -0
  155. package/lib/drivers/shared/experiment-behavior.d.ts.map +1 -0
  156. package/lib/drivers/shared/experiment-behavior.js +197 -0
  157. package/lib/drivers/shared/experiment-behavior.js.map +1 -0
  158. package/lib/drivers/shared/experiment-policy.d.ts +4 -0
  159. package/lib/drivers/shared/experiment-policy.d.ts.map +1 -0
  160. package/lib/drivers/shared/experiment-policy.js +143 -0
  161. package/lib/drivers/shared/experiment-policy.js.map +1 -0
  162. package/lib/drivers/subprocess/index.d.ts +37 -2
  163. package/lib/drivers/subprocess/index.d.ts.map +1 -1
  164. package/lib/drivers/subprocess/index.js +46 -6
  165. package/lib/drivers/subprocess/index.js.map +1 -1
  166. package/lib/index.d.ts +6 -5
  167. package/lib/index.d.ts.map +1 -1
  168. package/lib/index.js +4 -3
  169. package/lib/index.js.map +1 -1
  170. package/lib/orchestrator/annotation-loader.d.ts +6 -2
  171. package/lib/orchestrator/annotation-loader.d.ts.map +1 -1
  172. package/lib/orchestrator/annotation-loader.js +8 -0
  173. package/lib/orchestrator/annotation-loader.js.map +1 -1
  174. package/lib/orchestrator/annotation-store.d.ts.map +1 -1
  175. package/lib/orchestrator/annotation-store.js +16 -2
  176. package/lib/orchestrator/annotation-store.js.map +1 -1
  177. package/lib/orchestrator/campaign-lifecycle.d.ts +1 -0
  178. package/lib/orchestrator/campaign-lifecycle.d.ts.map +1 -1
  179. package/lib/orchestrator/campaign-lifecycle.js +4 -1
  180. package/lib/orchestrator/campaign-lifecycle.js.map +1 -1
  181. package/lib/orchestrator/compare.d.ts +4 -0
  182. package/lib/orchestrator/compare.d.ts.map +1 -1
  183. package/lib/orchestrator/compare.js +13 -1
  184. package/lib/orchestrator/compare.js.map +1 -1
  185. package/lib/orchestrator/comparison-identity.d.ts.map +1 -1
  186. package/lib/orchestrator/comparison-identity.js +1 -0
  187. package/lib/orchestrator/comparison-identity.js.map +1 -1
  188. package/lib/orchestrator/config.d.ts +42 -0
  189. package/lib/orchestrator/config.d.ts.map +1 -1
  190. package/lib/orchestrator/config.js +6 -2
  191. package/lib/orchestrator/config.js.map +1 -1
  192. package/lib/orchestrator/discover.d.ts +1 -0
  193. package/lib/orchestrator/discover.d.ts.map +1 -1
  194. package/lib/orchestrator/discover.js +7 -1
  195. package/lib/orchestrator/discover.js.map +1 -1
  196. package/lib/orchestrator/doctor.d.ts.map +1 -1
  197. package/lib/orchestrator/doctor.js +34 -26
  198. package/lib/orchestrator/doctor.js.map +1 -1
  199. package/lib/orchestrator/driver-capabilities.d.ts +3 -0
  200. package/lib/orchestrator/driver-capabilities.d.ts.map +1 -0
  201. package/lib/orchestrator/driver-capabilities.js +10 -0
  202. package/lib/orchestrator/driver-capabilities.js.map +1 -0
  203. package/lib/orchestrator/driver-identity.d.ts +5 -0
  204. package/lib/orchestrator/driver-identity.d.ts.map +1 -0
  205. package/lib/orchestrator/driver-identity.js +41 -0
  206. package/lib/orchestrator/driver-identity.js.map +1 -0
  207. package/lib/orchestrator/experiment-context.d.ts +10 -0
  208. package/lib/orchestrator/experiment-context.d.ts.map +1 -1
  209. package/lib/orchestrator/experiment-contracts.d.ts +10 -0
  210. package/lib/orchestrator/experiment-contracts.d.ts.map +1 -1
  211. package/lib/orchestrator/experiment-corpus.d.ts.map +1 -1
  212. package/lib/orchestrator/experiment-corpus.js +17 -19
  213. package/lib/orchestrator/experiment-corpus.js.map +1 -1
  214. package/lib/orchestrator/experiment-effects.d.ts +33 -0
  215. package/lib/orchestrator/experiment-effects.d.ts.map +1 -0
  216. package/lib/orchestrator/experiment-effects.js +44 -0
  217. package/lib/orchestrator/experiment-effects.js.map +1 -0
  218. package/lib/orchestrator/experiment-evaluation.d.ts.map +1 -1
  219. package/lib/orchestrator/experiment-evaluation.js +36 -3
  220. package/lib/orchestrator/experiment-evaluation.js.map +1 -1
  221. package/lib/orchestrator/experiment-identity.d.ts.map +1 -1
  222. package/lib/orchestrator/experiment-identity.js +18 -29
  223. package/lib/orchestrator/experiment-identity.js.map +1 -1
  224. package/lib/orchestrator/experiment-loader.d.ts +7 -2
  225. package/lib/orchestrator/experiment-loader.d.ts.map +1 -1
  226. package/lib/orchestrator/experiment-report.d.ts +70 -0
  227. package/lib/orchestrator/experiment-report.d.ts.map +1 -1
  228. package/lib/orchestrator/experiment-report.js +9 -1
  229. package/lib/orchestrator/experiment-report.js.map +1 -1
  230. package/lib/orchestrator/experiment-runs.d.ts +22 -1
  231. package/lib/orchestrator/experiment-runs.d.ts.map +1 -1
  232. package/lib/orchestrator/experiment-runs.js +13 -5
  233. package/lib/orchestrator/experiment-runs.js.map +1 -1
  234. package/lib/orchestrator/experiment-runtime-identity.d.ts +0 -16
  235. package/lib/orchestrator/experiment-runtime-identity.d.ts.map +1 -1
  236. package/lib/orchestrator/experiment-runtime-identity.js +5 -220
  237. package/lib/orchestrator/experiment-runtime-identity.js.map +1 -1
  238. package/lib/orchestrator/experiment-selection.d.ts.map +1 -1
  239. package/lib/orchestrator/experiment-selection.js +52 -0
  240. package/lib/orchestrator/experiment-selection.js.map +1 -1
  241. package/lib/orchestrator/experiment.d.ts +0 -2
  242. package/lib/orchestrator/experiment.d.ts.map +1 -1
  243. package/lib/orchestrator/experiment.js +23 -15
  244. package/lib/orchestrator/experiment.js.map +1 -1
  245. package/lib/orchestrator/fs-read.d.ts +1 -6
  246. package/lib/orchestrator/fs-read.d.ts.map +1 -1
  247. package/lib/orchestrator/fs-read.js +2 -32
  248. package/lib/orchestrator/fs-read.js.map +1 -1
  249. package/lib/orchestrator/grader-experiment.d.ts +17 -1
  250. package/lib/orchestrator/grader-experiment.d.ts.map +1 -1
  251. package/lib/orchestrator/grader-experiment.js +52 -7
  252. package/lib/orchestrator/grader-experiment.js.map +1 -1
  253. package/lib/orchestrator/grading.d.ts.map +1 -1
  254. package/lib/orchestrator/grading.js +13 -6
  255. package/lib/orchestrator/grading.js.map +1 -1
  256. package/lib/orchestrator/list-runs.d.ts +1 -0
  257. package/lib/orchestrator/list-runs.d.ts.map +1 -1
  258. package/lib/orchestrator/list-runs.js +27 -6
  259. package/lib/orchestrator/list-runs.js.map +1 -1
  260. package/lib/orchestrator/optimization-lifecycle.d.ts.map +1 -1
  261. package/lib/orchestrator/optimization-lifecycle.js +3 -1
  262. package/lib/orchestrator/optimization-lifecycle.js.map +1 -1
  263. package/lib/orchestrator/orchestrator.d.ts +6 -1
  264. package/lib/orchestrator/orchestrator.d.ts.map +1 -1
  265. package/lib/orchestrator/orchestrator.js +17 -143
  266. package/lib/orchestrator/orchestrator.js.map +1 -1
  267. package/lib/orchestrator/paired-effects.d.ts +34 -0
  268. package/lib/orchestrator/paired-effects.d.ts.map +1 -0
  269. package/lib/orchestrator/paired-effects.js +158 -0
  270. package/lib/orchestrator/paired-effects.js.map +1 -0
  271. package/lib/orchestrator/provenance.d.ts +5 -1
  272. package/lib/orchestrator/provenance.d.ts.map +1 -1
  273. package/lib/orchestrator/provenance.js +16 -3
  274. package/lib/orchestrator/provenance.js.map +1 -1
  275. package/lib/orchestrator/regrade.d.ts +1 -0
  276. package/lib/orchestrator/regrade.d.ts.map +1 -1
  277. package/lib/orchestrator/run-execution.d.ts +2 -0
  278. package/lib/orchestrator/run-execution.d.ts.map +1 -1
  279. package/lib/orchestrator/run-execution.js +18 -3
  280. package/lib/orchestrator/run-execution.js.map +1 -1
  281. package/lib/orchestrator/run-lifecycle.d.ts +1 -0
  282. package/lib/orchestrator/run-lifecycle.d.ts.map +1 -1
  283. package/lib/orchestrator/run-lifecycle.js +3 -2
  284. package/lib/orchestrator/run-lifecycle.js.map +1 -1
  285. package/lib/orchestrator/suite-runs.d.ts +3 -1
  286. package/lib/orchestrator/suite-runs.d.ts.map +1 -1
  287. package/lib/orchestrator/suite-runs.js +12 -2
  288. package/lib/orchestrator/suite-runs.js.map +1 -1
  289. package/lib/orchestrator/suite-summary.d.ts.map +1 -1
  290. package/lib/orchestrator/suite-summary.js +2 -1
  291. package/lib/orchestrator/suite-summary.js.map +1 -1
  292. package/lib/previewer/server.d.ts +6 -0
  293. package/lib/previewer/server.d.ts.map +1 -1
  294. package/lib/previewer/server.js +21 -5
  295. package/lib/previewer/server.js.map +1 -1
  296. package/package.json +2 -2
  297. package/skills/lux/references/cli.md +5 -2
  298. package/skills/lux/references/eval-authoring.md +0 -7
  299. package/skills/lux-answerer/SKILL.md +1 -1
  300. package/src/answerers/claude-code/index.ts +3 -3
  301. package/src/answerers/persona/index.ts +3 -3
  302. package/src/answerers/terminal/index.ts +4 -9
  303. package/src/assertions/core/resource-checks.ts +3 -5
  304. package/src/assertions/rubric/index.ts +7 -1
  305. package/src/assertions/rubric/internal.ts +13 -8
  306. package/src/authoring.ts +331 -2
  307. package/src/cli/bin.ts +2 -12
  308. package/src/cli/command-runtime.ts +20 -9
  309. package/src/cli/init-templates.ts +223 -0
  310. package/src/cli/init.ts +38 -184
  311. package/src/cli/output-schemas.ts +106 -0
  312. package/src/cli/program.ts +621 -548
  313. package/src/cli/render/reporter.ts +25 -4
  314. package/src/cli/run-options.ts +8 -25
  315. package/src/cli/view.ts +45 -84
  316. package/src/core/annotation.ts +2 -1
  317. package/src/core/answerer.ts +9 -0
  318. package/src/core/assertion.ts +3 -0
  319. package/src/core/case.ts +5 -1
  320. package/src/core/driver.ts +38 -0
  321. package/src/core/environment.ts +22 -0
  322. package/src/core/evaluation-clusters.ts +35 -0
  323. package/src/core/experiment.ts +60 -23
  324. package/src/core/file-tree.ts +35 -0
  325. package/src/core/index.ts +1 -0
  326. package/src/core/lifecycle-fixtures.ts +4 -2
  327. package/src/core/platform-process.ts +54 -6
  328. package/src/core/run.ts +20 -0
  329. package/src/core/usage.ts +8 -0
  330. package/src/drivers/claude-code/index.ts +34 -3
  331. package/src/drivers/codex/app-events.ts +43 -33
  332. package/src/drivers/codex/exec-events.ts +258 -0
  333. package/src/drivers/codex/index.ts +37 -262
  334. package/src/drivers/cursor/README.md +57 -0
  335. package/src/drivers/cursor/acp-transport.ts +182 -0
  336. package/src/drivers/cursor/config.ts +29 -0
  337. package/src/drivers/cursor/events.ts +83 -0
  338. package/src/drivers/cursor/index.ts +316 -0
  339. package/src/drivers/cursor/interaction.ts +90 -0
  340. package/src/drivers/mcp/index.ts +16 -3
  341. package/src/drivers/shared/experiment-behavior.ts +214 -0
  342. package/src/drivers/shared/experiment-policy.ts +195 -0
  343. package/src/drivers/subprocess/index.ts +45 -6
  344. package/src/index.ts +16 -5
  345. package/src/orchestrator/annotation-loader.ts +10 -0
  346. package/src/orchestrator/annotation-store.ts +24 -3
  347. package/src/orchestrator/campaign-lifecycle.ts +5 -1
  348. package/src/orchestrator/compare.ts +16 -1
  349. package/src/orchestrator/comparison-identity.ts +1 -0
  350. package/src/orchestrator/config.ts +6 -2
  351. package/src/orchestrator/discover.ts +8 -1
  352. package/src/orchestrator/doctor.ts +40 -22
  353. package/src/orchestrator/driver-capabilities.ts +16 -0
  354. package/src/orchestrator/driver-identity.ts +59 -0
  355. package/src/orchestrator/experiment-corpus.ts +23 -28
  356. package/src/orchestrator/experiment-effects.ts +61 -0
  357. package/src/orchestrator/experiment-evaluation.ts +45 -5
  358. package/src/orchestrator/experiment-identity.ts +27 -36
  359. package/src/orchestrator/experiment-report.ts +21 -1
  360. package/src/orchestrator/experiment-runs.ts +11 -4
  361. package/src/orchestrator/experiment-runtime-identity.ts +5 -243
  362. package/src/orchestrator/experiment-selection.ts +77 -0
  363. package/src/orchestrator/experiment.ts +26 -24
  364. package/src/orchestrator/fs-read.ts +2 -32
  365. package/src/orchestrator/grader-experiment.ts +72 -8
  366. package/src/orchestrator/grading.ts +15 -6
  367. package/src/orchestrator/list-runs.ts +29 -6
  368. package/src/orchestrator/optimization-lifecycle.ts +4 -1
  369. package/src/orchestrator/orchestrator.ts +26 -183
  370. package/src/orchestrator/paired-effects.ts +209 -0
  371. package/src/orchestrator/provenance.ts +23 -3
  372. package/src/orchestrator/run-execution.ts +18 -3
  373. package/src/orchestrator/run-lifecycle.ts +4 -2
  374. package/src/orchestrator/suite-runs.ts +10 -1
  375. package/src/orchestrator/suite-summary.ts +2 -1
  376. package/src/previewer/server.ts +27 -5
  377. package/viewer/app.js +27 -6
  378. package/viewer/detail.js +22 -5
  379. package/lib/answerers/scripted/index.d.ts +0 -18
  380. package/lib/answerers/scripted/index.d.ts.map +0 -1
  381. package/lib/answerers/scripted/index.js +0 -63
  382. package/lib/answerers/scripted/index.js.map +0 -1
  383. package/lib/cli/skills.d.ts +0 -38
  384. package/lib/cli/skills.d.ts.map +0 -1
  385. package/lib/cli/skills.js +0 -84
  386. package/lib/cli/skills.js.map +0 -1
  387. package/src/answerers/scripted/index.ts +0 -78
  388. package/src/cli/skills.ts +0 -129
@@ -1,120 +1,351 @@
1
+ var __addDisposableResource = (this && this.__addDisposableResource) || function (env, value, async) {
2
+ if (value !== null && value !== void 0) {
3
+ if (typeof value !== "object" && typeof value !== "function") throw new TypeError("Object expected.");
4
+ var dispose, inner;
5
+ if (async) {
6
+ if (!Symbol.asyncDispose) throw new TypeError("Symbol.asyncDispose is not defined.");
7
+ dispose = value[Symbol.asyncDispose];
8
+ }
9
+ if (dispose === void 0) {
10
+ if (!Symbol.dispose) throw new TypeError("Symbol.dispose is not defined.");
11
+ dispose = value[Symbol.dispose];
12
+ if (async) inner = dispose;
13
+ }
14
+ if (typeof dispose !== "function") throw new TypeError("Object not disposable.");
15
+ if (inner) dispose = function() { try { inner.call(this); } catch (e) { return Promise.reject(e); } };
16
+ env.stack.push({ value: value, dispose: dispose, async: async });
17
+ }
18
+ else if (async) {
19
+ env.stack.push({ async: true });
20
+ }
21
+ return value;
22
+ };
23
+ var __disposeResources = (this && this.__disposeResources) || (function (SuppressedError) {
24
+ return function (env) {
25
+ function fail(e) {
26
+ env.error = env.hasError ? new SuppressedError(e, env.error, "An error was suppressed during disposal.") : e;
27
+ env.hasError = true;
28
+ }
29
+ var r, s = 0;
30
+ function next() {
31
+ while (r = env.stack.pop()) {
32
+ try {
33
+ if (!r.async && s === 1) return s = 0, env.stack.push(r), Promise.resolve().then(next);
34
+ if (r.dispose) {
35
+ var result = r.dispose.call(r.value);
36
+ if (r.async) return s |= 2, Promise.resolve(result).then(next, function(e) { fail(e); return next(); });
37
+ }
38
+ else s |= 1;
39
+ }
40
+ catch (e) {
41
+ fail(e);
42
+ }
43
+ }
44
+ if (s === 1) return env.hasError ? Promise.reject(env.error) : Promise.resolve();
45
+ if (env.hasError) throw env.error;
46
+ }
47
+ return next();
48
+ };
49
+ })(typeof SuppressedError === "function" ? SuppressedError : function (error, suppressed, message) {
50
+ var e = new Error(message);
51
+ return e.name = "SuppressedError", e.error = error, e.suppressed = suppressed, e;
52
+ });
1
53
  import { randomUUID } from "node:crypto";
2
54
  import { existsSync } from "node:fs";
3
55
  import { lstat, mkdir, realpath, rename, rm, stat, writeFile } from "node:fs/promises";
4
56
  import { createRequire } from "node:module";
5
57
  import { basename, dirname, extname, join, relative, resolve, sep } from "node:path";
6
- import { Command, InvalidArgumentError } from "commander";
58
+ import { fileURLToPath } from "node:url";
59
+ import { Cli, Errors, z } from "incur";
7
60
  import { createRubricAssertion } from "../assertions/rubric/index.js";
8
61
  import { GraderAnnotationSchema, matchesAnyGlob, valueHash, } from "../core/index.js";
9
- import { applyCandidate, buildRegistryFromConfig, compareRuns, compareSuiteRuns, discoverCases, doctorProject, experimentCampaignMode, formatComparisonText, formatDoctorReport, formatExperimentDecisionMarkdown, formatExperimentRun, formatSuiteComparisonText, hasRegression, hasSuiteRegression, loadEvalCase, loadExperimentCampaign, loadExperimentRun, loadFrozenRegradableRun, loadLuxConfig, loadRubricDefinition, loadSuiteRun, loadVerifiedExperimentDecisionSnapshot, Orchestrator, optimizeExperimentCampaign, planExperimentCampaign, regradeRun, releaseFrozenRegradableRun, resolveExperimentCampaignPath, runExperimentCampaign, selectCases, upsertGraderAnnotation, verifyExperimentDecisionIntegrity, verifyExperimentRunIntegrity, } from "../orchestrator/index.js";
62
+ import { applyCandidate, buildRegistryFromConfig, compareRuns, compareSuiteRuns, DoctorReportSchema, discoverCases, doctorProject, ExperimentCampaignPlanSchema, ExperimentDecisionReportSchema, ExperimentRunSchema, experimentCampaignMode, formatComparisonText, formatDoctorReport, formatExperimentDecisionMarkdown, formatSuiteComparisonText, GradingSnapshotSchema, hasRegression, hasSuiteRegression, loadEvalCase, loadExperimentCampaign, loadExperimentRun, loadFrozenRegradableRun, loadLuxConfig, loadRubricDefinition, loadSuiteRun, loadVerifiedExperimentDecisionSnapshot, Orchestrator, optimizeExperimentCampaign, planExperimentCampaign, regradeRun, releaseFrozenRegradableRun, resolveExperimentCampaignPath, runExperimentCampaign, selectCases, upsertGraderAnnotation, verifyExperimentDecisionIntegrity, verifyExperimentRunIntegrity, } from "../orchestrator/index.js";
10
63
  import { abortOnInterrupt, ensureDirectory, failWith, relativePathWithin, } from "./command-runtime.js";
11
- import { formatInitReport, initProject } from "./init.js";
64
+ import { initProject } from "./init.js";
65
+ import { CompareOutputSchema, ViewOutputSchema } from "./output-schemas.js";
12
66
  import { ConsoleReporter } from "./render/reporter.js";
13
- import { collectTags, collectValues, parseAnnotationAssertion, parsePositiveInt, parseUnitInterval, resolveAnswererOverride, resolveConcurrency, } from "./run-options.js";
14
- import { formatSkillsInstall, formatSkillsList, installSkills, } from "./skills.js";
15
- import { parsePreviewPort, viewCommand } from "./view.js";
67
+ import { parseAnnotationAssertion, resolveAnswererOverride, resolveConcurrency, } from "./run-options.js";
68
+ import { viewCommand, webViewCommand } from "./view.js";
16
69
  const requireFromHere = createRequire(import.meta.url);
17
70
  // From lib/cli/program.js (or src/cli/program.ts under test) to the package root.
18
71
  const PKG_VERSION = requireFromHere("../../package.json").version;
19
- const registerProjectCommands = (program) => {
20
- program
21
- .command("init")
22
- .description("scaffold a new evals project in the current directory")
23
- .option("--force", "overwrite existing files")
24
- .option("--experiment", "also scaffold a prompt optimization campaign and split cases")
25
- .action(async (options) => {
26
- const cwd = process.cwd();
27
- const result = await initProject({
28
- cwd,
29
- force: options.force ?? false,
30
- experiment: options.experiment ?? false,
72
+ const nonEmptyString = z.string().min(1);
73
+ const positiveInteger = z.coerce.number().int().positive();
74
+ const unitInterval = z.coerce.number().min(0).max(1);
75
+ const tagList = z
76
+ .array(nonEmptyString)
77
+ .default([])
78
+ .describe("Repeatable list; each value may be comma-separated");
79
+ const normalizeTags = (values) => values.flatMap((value) => value.split(",").map((tag) => tag.trim())).filter(Boolean);
80
+ const initOutput = z.object({
81
+ created: z.array(z.string()),
82
+ updated: z.array(z.string()),
83
+ skipped: z.array(z.object({ path: z.string(), reason: z.string() })),
84
+ experiment: z.boolean(),
85
+ });
86
+ const runOutput = z.union([
87
+ z.object({
88
+ caseCount: z.number().int(),
89
+ cases: z.array(z.object({ path: z.string(), id: z.string() })),
90
+ }),
91
+ z.object({
92
+ caseCount: z.number().int(),
93
+ tags: z.array(z.object({ tag: z.string(), count: z.number().int() })),
94
+ }),
95
+ z.object({
96
+ suiteDir: z.string(),
97
+ results: z.array(z.object({
98
+ caseId: z.string(),
99
+ runDir: z.string().nullable(),
100
+ exitReason: z.string(),
101
+ casePassed: z.boolean().nullable(),
102
+ caseScore: z.number().nullable(),
103
+ assertionsPassed: z.number().int(),
104
+ assertionsTotal: z.number().int(),
105
+ })),
106
+ }),
107
+ ]);
108
+ const experimentOutput = z.union([
109
+ ExperimentCampaignPlanSchema,
110
+ ExperimentRunSchema.extend({ experimentDir: z.string() }),
111
+ ]);
112
+ const gradingOutput = GradingSnapshotSchema.extend({ gradingPath: z.string() });
113
+ const reportOutput = z.union([
114
+ ExperimentDecisionReportSchema,
115
+ z.object({ path: z.string(), report: ExperimentDecisionReportSchema }),
116
+ ]);
117
+ const campaignArgs = z.object({ campaign: nonEmptyString.describe("Experiment campaign path") });
118
+ const campaignOptions = z.object({
119
+ runsRoot: nonEmptyString.optional().describe("Run directory root overriding lux.config.ts"),
120
+ resume: nonEmptyString.optional().describe("Interrupted experiment directory to resume"),
121
+ plan: z.boolean().optional().describe("Validate and estimate without calling models"),
122
+ });
123
+ export const buildCli = () => Cli.create("lux", {
124
+ description: "Coding-agent evaluation and improvement with deterministic assertions and human-in-the-loop runs",
125
+ version: PKG_VERSION,
126
+ sync: {
127
+ include: [fileURLToPath(new URL("../../skills/*", import.meta.url))],
128
+ depth: 2,
129
+ suggestions: [
130
+ "Use Lux to inspect the eval cases in this project",
131
+ "Use Lux to diagnose the latest failed eval run",
132
+ "Use Lux to plan an evidence-based improvement experiment",
133
+ ],
134
+ },
135
+ mcp: {
136
+ title: "Lux coding-agent evaluation",
137
+ instructions: "Inspect and plan before running model-spending commands. Apply only a verified promoted experiment champion.",
138
+ },
139
+ })
140
+ .command("init", {
141
+ description: "Scaffold a new eval project in the current directory",
142
+ options: z.object({
143
+ force: z.boolean().optional().describe("Overwrite existing files"),
144
+ experiment: z.boolean().optional().describe("Also scaffold an optimization campaign"),
145
+ driver: z
146
+ .enum(["subprocess", "claude-code", "codex", "cursor"])
147
+ .optional()
148
+ .describe("Starter driver (default: deterministic subprocess)"),
149
+ model: nonEmptyString.optional().describe("Explicit subject model; required for Cursor"),
150
+ }),
151
+ output: initOutput,
152
+ destructive: true,
153
+ mcp: { annotations: { destructiveHint: true, openWorldHint: false } },
154
+ examples: [
155
+ { description: "Scaffold a basic eval project" },
156
+ { options: { experiment: true }, description: "Scaffold an optimization project" },
157
+ ],
158
+ run: async (c) => c.ok(await initCommand(c.options), { cta: { commands: ["doctor", "run --list"] } }),
159
+ })
160
+ .command("doctor", {
161
+ description: "Preflight project configuration, CLIs, cases, campaigns, and budgets",
162
+ args: z.object({ campaigns: z.array(z.string()).default([]).describe("Campaign paths") }),
163
+ output: DoctorReportSchema,
164
+ mcp: {
165
+ annotations: { readOnlyHint: true, destructiveHint: false, openWorldHint: false },
166
+ },
167
+ run: async (c) => c.ok(await doctorCommand(c.args.campaigns), { cta: { commands: ["run --list"] } }),
168
+ })
169
+ .command("run", {
170
+ description: "Discover and run cases, filtered by name, path, or tag",
171
+ args: z.object({ filters: z.array(z.string()).default([]).describe("Case names or paths") }),
172
+ options: z.object({
173
+ tag: tagList.describe("Required case tag; repeatable and comma-separated"),
174
+ loop: positiveInteger.optional().describe("Run each matched eval this many times"),
175
+ concurrency: positiveInteger.optional().describe("Maximum cases running concurrently"),
176
+ list: z.boolean().optional().describe("List matching cases without running"),
177
+ listTags: z.boolean().optional().describe("List tags across matching cases"),
178
+ interactive: z.boolean().optional().describe("Answer agent questions in this terminal"),
179
+ claudeAnswerer: z.boolean().optional().describe("Use Claude Code as the answerer"),
180
+ skipJudge: z.boolean().optional().describe("Skip rubric LLM judge assertions"),
181
+ runsRoot: nonEmptyString.optional().describe("Run directory root"),
182
+ profile: nonEmptyString.optional().describe("Named driver and answerer profile"),
183
+ subjectRoot: nonEmptyString.optional().describe("Root bound into subjectPath values"),
184
+ verbose: z.boolean().optional().describe("Include phase timings and run directories"),
185
+ }),
186
+ alias: { tag: "t", concurrency: "j", interactive: "i", verbose: "v" },
187
+ output: runOutput,
188
+ mcp: { annotations: { readOnlyHint: false, destructiveHint: false, openWorldHint: true } },
189
+ examples: [
190
+ { options: { list: true, tag: [] }, description: "List all discovered cases" },
191
+ {
192
+ args: { filters: ["smoke"] },
193
+ options: { tag: [] },
194
+ description: "Run matching cases",
195
+ },
196
+ ],
197
+ run: async (c) => {
198
+ const result = await runCommand(c.args.filters, c.options, !c.agent);
199
+ const suiteDir = "suiteDir" in result ? result.suiteDir : undefined;
200
+ return c.ok(result, {
201
+ cta: { commands: suiteDir ? [{ command: "view", args: { runDir: suiteDir } }] : [] },
31
202
  });
32
- process.stdout.write(`${formatInitReport(result, cwd)}\n`);
33
- });
34
- const skills = program.command("skills").description("manage bundled agent skills");
35
- skills
36
- .command("list")
37
- .description("list bundled skills and supported platforms")
38
- .action(() => {
39
- process.stdout.write(`${formatSkillsList()}\n`);
40
- });
41
- skills
42
- .command("install")
43
- .description("install the Lux eval-authoring skill for Claude Code or Codex")
44
- .argument("<platform>", "claude, codex, or all")
45
- .option("--project", "install into the current project instead of the user profile")
46
- .option("--dir <path>", "install under an explicit skills directory")
47
- .option("--name <name>", "install one bundled skill")
48
- .action(async (platform, options) => {
49
- if (!["claude", "claude-code", "codex", "all"].includes(platform)) {
50
- throw new InvalidArgumentError("platform must be claude, codex, or all");
51
- }
52
- if (options.name && !["lux", "lux-answerer"].includes(options.name)) {
53
- throw new InvalidArgumentError("skill name must be lux or lux-answerer");
203
+ },
204
+ })
205
+ .command("experiment", {
206
+ description: "Evaluate controlled source variants across cases and model profiles",
207
+ args: campaignArgs,
208
+ options: campaignOptions,
209
+ output: experimentOutput,
210
+ mcp: { annotations: { readOnlyHint: false, destructiveHint: false, openWorldHint: true } },
211
+ run: async (c) => experimentCommand("experiment", c.args.campaign, c.options),
212
+ })
213
+ .command("optimize", {
214
+ description: "Evolve candidates from grading feedback and promote on held-out cases",
215
+ args: campaignArgs,
216
+ options: campaignOptions,
217
+ output: experimentOutput,
218
+ destructive: true,
219
+ mcp: { annotations: { readOnlyHint: false, destructiveHint: true, openWorldHint: true } },
220
+ run: async (c) => experimentCommand("optimize", c.args.campaign, c.options),
221
+ })
222
+ .command("grade", {
223
+ description: "Re-run current graders against a persisted run",
224
+ args: z.object({ runDir: nonEmptyString.describe("Persisted run directory") }),
225
+ options: z.object({ case: nonEmptyString.optional().describe("Current case definition") }),
226
+ output: gradingOutput,
227
+ mcp: { annotations: { readOnlyHint: false, destructiveHint: false, openWorldHint: true } },
228
+ run: async (c) => gradeCommand(c.args.runDir, c.options),
229
+ })
230
+ .command("annotate", {
231
+ description: "Add or update a human grader-alignment label",
232
+ args: z.object({ runDir: nonEmptyString.describe("Persisted run directory") }),
233
+ options: z.object({
234
+ out: nonEmptyString.describe("JSON annotation dataset to create or update"),
235
+ rubric: nonEmptyString.describe("Candidate-relative declarative rubric path"),
236
+ id: nonEmptyString.optional().describe("Stable annotation ID"),
237
+ label: z.enum(["pass", "fail"]).optional().describe("Expected overall verdict"),
238
+ score: unitInterval.optional().describe("Expected score from zero to one"),
239
+ assertion: z
240
+ .array(nonEmptyString)
241
+ .default([])
242
+ .describe("Assertion label id=pass|fail[:score]"),
243
+ split: z.enum(["train", "validation", "test"]).optional().describe("Dataset split"),
244
+ tag: tagList.describe("Selection tag; repeatable and comma-separated"),
245
+ feedback: nonEmptyString.optional().describe("Human rationale for mismatches"),
246
+ annotator: nonEmptyString.optional().describe("Label author"),
247
+ reviewer: nonEmptyString.optional().describe("Independent reviewer"),
248
+ source: z.enum(["human", "synthetic"]).default("human").describe("Label provenance"),
249
+ }),
250
+ output: z.object({ id: z.string(), path: z.string(), labels: z.number().int() }),
251
+ destructive: true,
252
+ mcp: {
253
+ annotations: {
254
+ readOnlyHint: false,
255
+ destructiveHint: false,
256
+ idempotentHint: true,
257
+ openWorldHint: false,
258
+ },
259
+ },
260
+ run: async (c) => annotateCommand(c.args.runDir, c.options),
261
+ })
262
+ .command("apply", {
263
+ description: "Apply the verified promoted experiment champion",
264
+ args: z.object({
265
+ experimentDir: nonEmptyString.describe("Experiment directory"),
266
+ candidateId: nonEmptyString.optional().describe("Expected promoted candidate ID"),
267
+ }),
268
+ output: z.object({
269
+ candidateId: z.string(),
270
+ subjectRoot: z.string(),
271
+ sourceBytes: z.number(),
272
+ }),
273
+ destructive: true,
274
+ mcp: {
275
+ annotations: { readOnlyHint: false, destructiveHint: true, openWorldHint: false },
276
+ },
277
+ run: async (c) => applyCommand(c.args.experimentDir, c.args.candidateId),
278
+ })
279
+ .command("report", {
280
+ description: "Render a PR-ready report from a verified experiment record",
281
+ args: z.object({ experimentDir: nonEmptyString.describe("Experiment directory") }),
282
+ options: z.object({
283
+ out: nonEmptyString.optional().describe("Write Markdown to this path"),
284
+ force: z.boolean().optional().describe("Overwrite an existing output file"),
285
+ }),
286
+ output: reportOutput,
287
+ mcp: { annotations: { readOnlyHint: false, destructiveHint: false, openWorldHint: false } },
288
+ run: async (c) => reportCommand(c.args.experimentDir, c.options),
289
+ })
290
+ .command("view", {
291
+ description: "List past runs or show one run in detail",
292
+ args: z.object({ runDir: nonEmptyString.optional().describe("Run or suite directory") }),
293
+ options: z.object({
294
+ runsRoot: nonEmptyString.optional().describe("Run directory root"),
295
+ case: nonEmptyString.optional().describe("Filter lists to one case slug"),
296
+ suites: z.boolean().optional().describe("List logical suite runs"),
297
+ experiments: z.boolean().optional().describe("List experiment runs"),
298
+ web: z.boolean().optional().describe("Open the Lux Review web app"),
299
+ port: z.coerce.number().int().min(0).max(65535).optional().describe("Review server port"),
300
+ open: z.boolean().default(true).describe("Open a browser in web mode"),
301
+ }),
302
+ output: ViewOutputSchema,
303
+ mcp: { annotations: { readOnlyHint: true, destructiveHint: false, openWorldHint: false } },
304
+ run: (c) => {
305
+ if (c.options.web) {
306
+ if (c.formatExplicit && c.format !== "jsonl") {
307
+ throw new Errors.ParseError({
308
+ message: "--web requires streaming output; omit --format or use --format jsonl",
309
+ });
310
+ }
311
+ return webViewCommand(c.args.runDir, c.options);
54
312
  }
55
- const cwd = process.cwd();
56
- const installed = await installSkills({
57
- cwd,
58
- platform: platform,
59
- project: options.project ?? false,
60
- ...(options.dir ? { dir: options.dir } : {}),
61
- ...(options.name ? { name: options.name } : {}),
313
+ return viewCommand(c.args.runDir, c.options).then((result) => ViewOutputSchema.parse(result));
314
+ },
315
+ })
316
+ .command("compare", {
317
+ description: "Diff two run or suite directories and fail on regression",
318
+ args: z.object({
319
+ runA: nonEmptyString.describe("Baseline run"),
320
+ runB: nonEmptyString.describe("Current run"),
321
+ }),
322
+ options: z.object({
323
+ passRateTolerance: unitInterval.optional().describe("Allowed per-case pass-rate decrease"),
324
+ }),
325
+ output: CompareOutputSchema,
326
+ mcp: { annotations: { readOnlyHint: true, destructiveHint: false, openWorldHint: false } },
327
+ run: async (c) => CompareOutputSchema.parse(await compareCommand(c.args.runA, c.args.runB, c.options)),
328
+ });
329
+ const initCommand = async (options) => initProject({ cwd: process.cwd(), ...options });
330
+ const doctorCommand = async (campaigns) => {
331
+ const report = await doctorProject({ cwd: process.cwd(), campaigns });
332
+ if (!report.ok) {
333
+ throw new Errors.IncurError({
334
+ code: "PREFLIGHT_FAILED",
335
+ message: formatDoctorReport(report),
336
+ exitCode: 1,
62
337
  });
63
- process.stdout.write(`${formatSkillsInstall(installed, cwd)}\n`);
64
- });
65
- };
66
- const registerExperimentCommands = (program) => {
67
- program
68
- .command("doctor [campaigns...]")
69
- .description("preflight project configuration, CLIs, cases, campaigns, and budgets")
70
- .option("--json", "emit machine-readable JSON")
71
- .action(async (campaigns, options) => {
72
- const report = await doctorProject({ cwd: process.cwd(), campaigns });
73
- process.stdout.write(options.json ? `${JSON.stringify(report, null, 2)}\n` : `${formatDoctorReport(report)}\n`);
74
- if (!report.ok)
75
- failWith(1);
76
- });
77
- program
78
- .command("run [filters...]")
79
- .description("discover and run cases; filter by name/path substring or --tag")
80
- .option("-t, --tag <tag>", "only cases whose meta.tags include this tag (repeatable, comma-separated)", collectTags, [])
81
- .option("--loop <n>", "run each matched eval n times", parsePositiveInt)
82
- .option("-j, --concurrency <n>", "run up to n cases at once (overrides lux.config.ts)", parsePositiveInt)
83
- .option("--list", "print the matched cases and exit without running them")
84
- .option("--list-tags", "print the tags across the matched cases and exit")
85
- .option("-i, --interactive", "answer agent questions yourself in this terminal (HITL)")
86
- .option("--claude-answerer", "delegate questions through the Claude plugin's lux-answerer skill")
87
- .option("--skip-judge", "skip rubric LLM judge assertions")
88
- .option("--runs-root <path>", "run-dir root (overrides lux.config.ts)")
89
- .option("--profile <id>", "named driver/answerer profile from lux.config.ts")
90
- .option("--subject-root <path>", "root bound into subjectPath() values")
91
- .option("-v, --verbose", "add phase timings and the run dir to every case")
92
- .action(async (filters, options) => {
93
- await runCommand(filters, options);
94
- });
95
- program
96
- .command("experiment <campaign>")
97
- .description("evaluate controlled source variants across cases and model profiles")
98
- .option("--runs-root <path>", "run-dir root (overrides lux.config.ts)")
99
- .option("--resume <experiment-dir>", "resume an interrupted experiment ledger")
100
- .option("--plan", "validate and estimate the campaign without calling models")
101
- .option("--json", "emit machine-readable JSON")
102
- .action(async (campaign, options) => {
103
- await experimentCommand("experiment", campaign, options);
104
- });
338
+ }
339
+ return report;
105
340
  };
106
- const registerCandidateCommands = (program) => {
107
- program
108
- .command("grade <runDir>")
109
- .description("re-run current graders against a persisted run without rerunning the agent")
110
- .option("--case <path>", "grade with a current case definition instead of frozen case.json")
111
- .option("--json", "emit machine-readable JSON")
112
- .action(async (runDir, options) => {
341
+ const gradeCommand = async (runDir, options) => {
342
+ const env_1 = { stack: [], error: void 0, hasError: false };
343
+ try {
113
344
  const directory = resolve(runDir);
114
345
  if (!(await ensureDirectory("grade", directory)))
115
- return;
346
+ failWith(1);
116
347
  const config = await loadLuxConfig(process.cwd());
117
- const controller = abortOnInterrupt();
348
+ const controller = __addDisposableResource(env_1, abortOnInterrupt(), false);
118
349
  const result = await regradeRun({
119
350
  runDir: directory,
120
351
  registry: buildRegistryFromConfig(config),
@@ -122,273 +353,216 @@ const registerCandidateCommands = (program) => {
122
353
  ...(options.case ? { case: await loadEvalCase(resolve(options.case)) } : {}),
123
354
  abortSignal: controller.signal,
124
355
  });
125
- const report = result.snapshot.report;
126
- let gradingStatus = "ungraded";
127
- if (report.casePassed === true)
128
- gradingStatus = "PASS";
129
- if (report.casePassed === false)
130
- gradingStatus = "FAIL";
131
- let output;
132
- if (options.json) {
133
- output = `${JSON.stringify({ gradingPath: result.path, ...result.snapshot }, null, 2)}\n`;
134
- }
135
- else {
136
- const score = report.caseScore === null ? "—" : `${(report.caseScore * 100).toFixed(0)}%`;
137
- output = [
138
- `grading: ${gradingStatus}`,
139
- `score: ${score}`,
140
- `checks: ${report.assertions.filter((assertion) => assertion.passed).length}/${report.assertions.length}`,
141
- `saved: ${result.path}`,
142
- "",
143
- ].join("\n");
144
- }
145
- process.stdout.write(output);
146
356
  if (controller.signal.aborted)
147
357
  failWith(SIGINT_EXIT_CODE);
148
- else if (report.casePassed !== true)
149
- failWith(1);
150
- });
151
- program
152
- .command("annotate <runDir>")
153
- .description("add or update a human grader-alignment label for a persisted run")
154
- .requiredOption("--out <path>", "JSON annotation dataset to create or update")
155
- .requiredOption("--rubric <path>", "candidate-relative declarative .json rubric path")
156
- .option("--id <id>", "stable annotation id")
157
- .option("--label <pass|fail>", "expected overall case verdict")
158
- .option("--score <n>", "expected overall case score from 0 to 1", parseUnitInterval)
159
- .option("--assertion <id=pass|fail[:score]>", "expected stable assertion verdict and optional score (repeatable)", collectValues, [])
160
- .option("--split <train|validation|test>", "dataset split, stored as a tag")
161
- .option("--tag <tag>", "extra selection tag (repeatable, comma-separated)", collectTags, [])
162
- .option("--feedback <text>", "human rationale shown to the rubric optimizer on mismatch")
163
- .option("--annotator <id>", "person or system that authored the label")
164
- .option("--reviewer <id>", "independent reviewer of the label")
165
- .option("--source <human|synthetic>", "label provenance", "human")
166
- .action(async (runDir, options) => {
167
- const directory = resolve(runDir);
168
- if (!(await ensureDirectory("annotate", directory)))
169
- return;
170
- if (options.label && options.label !== "pass" && options.label !== "fail") {
171
- throw new InvalidArgumentError("--label must be pass or fail");
172
- }
173
- if (options.split && !["train", "validation", "test"].includes(options.split)) {
174
- throw new InvalidArgumentError("--split must be train, validation, or test");
175
- }
176
- if (options.source !== "human" && options.source !== "synthetic") {
177
- throw new InvalidArgumentError("--source must be human or synthetic");
178
- }
179
- const parsedAssertions = options.assertion.map(parseAnnotationAssertion);
180
- const assertionIds = parsedAssertions.map(([id]) => id);
181
- if (new Set(assertionIds).size !== assertionIds.length) {
182
- throw new InvalidArgumentError("--assertion ids must not be repeated");
183
- }
184
- const assertions = Object.fromEntries(parsedAssertions);
185
- if (options.label === undefined &&
186
- options.score === undefined &&
187
- Object.keys(assertions).length === 0) {
188
- throw new Error("lux annotate: provide --label, --score, or at least one --assertion");
189
- }
190
- const out = resolve(options.out);
191
- const run = await loadFrozenRegradableRun(directory);
192
- try {
193
- const rubric = await loadRubricDefinition(await resolveAuthoredFile(options.rubric, "rubric"));
194
- if (rubric.id !== run.case.id) {
195
- throw new Error(`lux annotate: rubric id '${rubric.id}' does not match persisted case '${run.case.id}'`);
196
- }
197
- const rubricIds = new Set(rubric.assertions.map((assertion) => assertion.id));
198
- const unknownIds = assertionIds.filter((assertionId) => !rubricIds.has(assertionId));
199
- if (unknownIds.length > 0) {
200
- throw new Error(`lux annotate: unknown rubric assertion ids: ${unknownIds.join(", ")}`);
201
- }
202
- const tags = [...(options.split ? [options.split] : []), ...options.tag].filter((tag, index, all) => all.indexOf(tag) === index);
203
- const caseId = run.case.id;
204
- const portableRunDir = (relative(dirname(out), directory) || ".").split(sep).join("/");
205
- const id = options.id ??
206
- `${caseId}-${basename(directory)}-${valueHash(portableRunDir).slice(0, 10)}`.replace(/[^a-zA-Z0-9_-]+/g, "-");
207
- const annotation = GraderAnnotationSchema.parse({
208
- id,
209
- runDir: portableRunDir,
210
- rubricPath: options.rubric,
211
- tags,
212
- source: options.source,
213
- ...(options.annotator ? { annotator: options.annotator } : {}),
214
- labeledAt: new Date().toISOString(),
215
- ...(options.reviewer ? { reviewer: options.reviewer } : {}),
216
- expected: {
217
- ...(options.label ? { casePassed: options.label === "pass" } : {}),
218
- ...(options.score !== undefined ? { caseScore: options.score } : {}),
219
- assertions,
220
- },
221
- ...(options.feedback ? { feedback: options.feedback } : {}),
358
+ if (result.snapshot.report.casePassed !== true) {
359
+ throw new Errors.IncurError({
360
+ code: "GRADING_FAILED",
361
+ message: `The persisted run did not pass regrading. Grading saved to ${result.path}.`,
362
+ exitCode: 1,
222
363
  });
223
- const dataset = await upsertGraderAnnotation(out, annotation);
224
- process.stdout.write(`annotated ${annotation.id} in ${out} (${dataset.annotations.length} labels)\n`);
225
- }
226
- finally {
227
- await releaseFrozenRegradableRun(run);
228
364
  }
229
- });
365
+ return { gradingPath: result.path, ...result.snapshot };
366
+ }
367
+ catch (e_1) {
368
+ env_1.error = e_1;
369
+ env_1.hasError = true;
370
+ }
371
+ finally {
372
+ __disposeResources(env_1);
373
+ }
230
374
  };
231
- const registerReportingCommands = (program) => {
232
- program
233
- .command("optimize <campaign>")
234
- .description("evolve source candidates from grading feedback and promote on held-out cases")
235
- .option("--runs-root <path>", "run-dir root (overrides lux.config.ts)")
236
- .option("--resume <experiment-dir>", "resume an interrupted optimization ledger")
237
- .option("--plan", "validate and estimate the campaign without calling models")
238
- .option("--json", "emit machine-readable JSON")
239
- .action(async (campaign, options) => {
240
- await experimentCommand("optimize", campaign, options);
241
- });
242
- program
243
- .command("apply <experimentDir> [candidateId]")
244
- .description("apply the promoted experiment champion after verifying the source base hash")
245
- .action(async (experimentDir, candidateId) => {
246
- const directory = resolve(experimentDir);
247
- const experiment = await loadExperimentRun(directory);
248
- if (!experiment) {
249
- process.stderr.write(`lux apply: experiment manifest not found at ${directory}\n`);
250
- failWith(1);
251
- return;
252
- }
253
- if (experiment.status !== "passed") {
254
- process.stderr.write(`lux apply: experiment status is '${experiment.status}', not 'passed'\n`);
255
- failWith(1);
256
- return;
257
- }
258
- const championId = experiment.championCandidateId;
259
- if (!championId) {
260
- process.stderr.write("lux apply: passed experiment has no promoted champion\n");
261
- failWith(1);
262
- return;
263
- }
264
- if (candidateId && candidateId !== championId) {
265
- process.stderr.write(`lux apply: candidate '${candidateId}' is not the promoted champion '${championId}'\n`);
266
- failWith(1);
267
- return;
268
- }
269
- let candidate;
270
- try {
271
- const candidates = await verifyExperimentRunIntegrity(directory, experiment);
272
- verifyExperimentDecisionIntegrity(experiment);
273
- candidate = candidates.find((entry) => entry.id === championId);
274
- if (!candidate) {
275
- throw new Error(`champion candidate '${championId}' not found`);
276
- }
277
- }
278
- catch (error) {
279
- const message = error instanceof Error ? error.message : String(error);
280
- process.stderr.write(`lux apply: experiment integrity check failed: ${message}\n`);
281
- failWith(1);
282
- return;
283
- }
284
- await applyCandidate(experiment.subjectRoot, experiment.campaignSnapshot.subject, candidate);
285
- process.stdout.write(`applied candidate ${candidate.id} to ${experiment.subjectRoot} (${candidate.sourceBytes} bytes)\n`);
286
- });
287
- program
288
- .command("report <experimentDir>")
289
- .description("render a PR-ready promotion report from a verified experiment record")
290
- .option("--json", "emit the structured decision report as JSON")
291
- .option("--out <path>", "write the report to a file instead of stdout")
292
- .option("--force", "overwrite an existing output file")
293
- .action(async (experimentDir, options) => {
294
- const directory = resolve(experimentDir);
295
- if (!(await ensureDirectory("report", directory)))
296
- return;
297
- const snapshot = await loadVerifiedExperimentDecisionSnapshot(directory);
298
- const { experiment, report } = snapshot;
299
- const rendered = options.json
300
- ? `${JSON.stringify(report, null, 2)}\n`
301
- : `${formatExperimentDecisionMarkdown(report)}\n`;
302
- if (!options.out) {
303
- process.stdout.write(rendered);
304
- return;
305
- }
306
- const out = resolve(options.out);
307
- const outputStats = await lstat(out).catch((error) => {
308
- if (error.code === "ENOENT")
309
- return null;
310
- throw error;
375
+ const annotateCommand = async (runDir, options) => {
376
+ const directory = resolve(runDir);
377
+ if (!(await ensureDirectory("annotate", directory)))
378
+ failWith(1);
379
+ const parsedAssertions = options.assertion.map(parseAnnotationAssertion);
380
+ const assertionIds = parsedAssertions.map(([id]) => id);
381
+ if (new Set(assertionIds).size !== assertionIds.length) {
382
+ throw new Errors.ParseError({ message: "--assertion IDs must not be repeated" });
383
+ }
384
+ const assertions = Object.fromEntries(parsedAssertions);
385
+ if (options.label === undefined &&
386
+ options.score === undefined &&
387
+ Object.keys(assertions).length === 0) {
388
+ throw new Errors.ParseError({
389
+ message: "Provide --label, --score, or at least one --assertion",
311
390
  });
312
- if (outputStats && (outputStats.isSymbolicLink() || !outputStats.isFile())) {
313
- throw new Error("lux report: an existing --out must be a regular, non-symlink file");
314
- }
315
- const [physicalDirectory, physicalSubjectRoot, physicalOut] = await Promise.all([
316
- realpath(directory),
317
- realpath(experiment.subjectRoot),
318
- physicalDestinationPath(out),
319
- ]);
320
- assertReportDestination(physicalOut, physicalDirectory, physicalSubjectRoot, experiment.campaignSnapshot.subject.exclude);
321
- if (outputStats && !options.force) {
322
- throw new Error(`lux report: output already exists (use --force): ${out}`);
323
- }
324
- await mkdir(dirname(out), { recursive: true });
325
- const physicalParent = await realpath(dirname(out));
326
- const finalPhysicalOut = resolve(physicalParent, basename(out));
327
- assertReportDestination(finalPhysicalOut, physicalDirectory, physicalSubjectRoot, experiment.campaignSnapshot.subject.exclude);
328
- await writeReportSafely(finalPhysicalOut, rendered, options.force ?? false);
329
- process.stdout.write(`wrote experiment report to ${out}\n`);
330
- });
331
- program
332
- .command("view [runDir]")
333
- .description("list past runs, or show one run's detail when given a runDir")
334
- .option("--runs-root <path>", "run-dir root (defaults to lux.config.ts)")
335
- .option("--case <id>", "filter the run list to one case slug (list mode only)")
336
- .option("--suites", "list logical suite runs instead of individual runs")
337
- .option("--experiments", "list experiment runs instead of individual runs")
338
- .option("--web", "open the local Lux Review web app")
339
- .option("--port <n>", "listen on a specific local port in web mode", parsePreviewPort)
340
- .option("--no-open", "start web mode without opening a browser")
341
- .option("--json", "emit machine-readable JSON")
342
- .action(viewCommand);
343
- program
344
- .command("compare <runA> <runB>")
345
- .description("diff two run or suite directories")
346
- .option("--pass-rate-tolerance <fraction>", "allowed per-case pass-rate decrease for suite comparison", parseUnitInterval)
347
- .action(async (runA, runB, options) => {
348
- const pathA = resolve(runA);
349
- const pathB = resolve(runB);
350
- // Validate both up front so one typo'd path in CI cannot read as
351
- // "0 regressed"; report every bad path, not just the first.
352
- const okA = await ensureDirectory("compare", pathA);
353
- const okB = await ensureDirectory("compare", pathB);
354
- if (!okA || !okB)
355
- return;
356
- const [suiteA, suiteB] = await Promise.all([loadSuiteRun(pathA), loadSuiteRun(pathB)]);
357
- if (Boolean(suiteA) !== Boolean(suiteB)) {
358
- process.stderr.write("lux compare: both paths must be runs or both must be suites\n");
359
- failWith(1);
360
- return;
391
+ }
392
+ const out = resolve(options.out);
393
+ const run = await loadFrozenRegradableRun(directory);
394
+ try {
395
+ const rubric = await loadRubricDefinition(await resolveAuthoredFile(options.rubric, "rubric"));
396
+ if (rubric.id !== run.case.id) {
397
+ throw new Error(`Rubric ID '${rubric.id}' does not match persisted case '${run.case.id}'.`);
361
398
  }
362
- if (suiteA && suiteB) {
363
- const comparison = await compareSuiteRuns(pathA, pathB, {
364
- ...(options.passRateTolerance !== undefined
365
- ? { passRateTolerance: options.passRateTolerance }
366
- : {}),
399
+ const rubricIds = new Set(rubric.assertions.map((assertion) => assertion.id));
400
+ const unknownIds = assertionIds.filter((assertionId) => !rubricIds.has(assertionId));
401
+ if (unknownIds.length > 0) {
402
+ throw new Errors.ParseError({
403
+ message: `Unknown rubric assertion IDs: ${unknownIds.join(", ")}`,
367
404
  });
368
- process.stdout.write(`${formatSuiteComparisonText(comparison)}\n`);
369
- if (hasSuiteRegression(comparison))
370
- failWith(1);
371
- return;
372
405
  }
373
- const cmp = await compareRuns({ pathA, pathB });
374
- process.stdout.write(`${formatComparisonText(cmp)}\n`);
375
- if (hasRegression(cmp))
376
- failWith(1);
406
+ const annotationTags = [
407
+ ...(options.split ? [options.split] : []),
408
+ ...normalizeTags(options.tag),
409
+ ].filter((tag, index, all) => all.indexOf(tag) === index);
410
+ const portableRunDir = (relative(dirname(out), directory) || ".").split(sep).join("/");
411
+ const id = options.id ??
412
+ `${run.case.id}-${basename(directory)}-${valueHash(portableRunDir).slice(0, 10)}`.replace(/[^a-zA-Z0-9_-]+/g, "-");
413
+ const annotation = GraderAnnotationSchema.parse({
414
+ id,
415
+ runDir: portableRunDir,
416
+ rubricPath: options.rubric,
417
+ tags: annotationTags,
418
+ source: options.source,
419
+ ...(options.annotator ? { annotator: options.annotator } : {}),
420
+ labeledAt: new Date().toISOString(),
421
+ ...(options.reviewer ? { reviewer: options.reviewer } : {}),
422
+ expected: {
423
+ ...(options.label ? { casePassed: options.label === "pass" } : {}),
424
+ ...(options.score !== undefined ? { caseScore: options.score } : {}),
425
+ assertions,
426
+ },
427
+ ...(options.feedback ? { feedback: options.feedback } : {}),
428
+ });
429
+ const dataset = await upsertGraderAnnotation(out, annotation);
430
+ return { id: annotation.id, path: out, labels: dataset.annotations.length };
431
+ }
432
+ finally {
433
+ await releaseFrozenRegradableRun(run);
434
+ }
435
+ };
436
+ const applyCommand = async (experimentDir, candidateId) => {
437
+ const directory = resolve(experimentDir);
438
+ const experiment = await loadExperimentRun(directory);
439
+ if (!experiment) {
440
+ throw new Errors.IncurError({
441
+ code: "EXPERIMENT_NOT_FOUND",
442
+ message: `Experiment manifest not found at ${directory}.`,
443
+ exitCode: 1,
444
+ });
445
+ }
446
+ if (experiment.status !== "passed") {
447
+ throw new Errors.IncurError({
448
+ code: "EXPERIMENT_NOT_PROMOTED",
449
+ message: `Experiment status is '${experiment.status}', not 'passed'.`,
450
+ exitCode: 1,
451
+ });
452
+ }
453
+ const championId = experiment.championCandidateId;
454
+ if (!championId) {
455
+ throw new Errors.IncurError({
456
+ code: "CHAMPION_NOT_FOUND",
457
+ message: "The passed experiment has no promoted champion.",
458
+ exitCode: 1,
459
+ });
460
+ }
461
+ if (candidateId && candidateId !== championId) {
462
+ throw new Errors.IncurError({
463
+ code: "CANDIDATE_NOT_PROMOTED",
464
+ message: `Candidate '${candidateId}' is not the promoted champion '${championId}'.`,
465
+ exitCode: 1,
466
+ });
467
+ }
468
+ let candidate;
469
+ try {
470
+ const candidates = await verifyExperimentRunIntegrity(directory, experiment);
471
+ verifyExperimentDecisionIntegrity(experiment);
472
+ candidate = candidates.find((entry) => entry.id === championId);
473
+ if (!candidate)
474
+ throw new Error(`Champion candidate '${championId}' not found.`);
475
+ }
476
+ catch (error) {
477
+ throw new Errors.IncurError({
478
+ code: "EXPERIMENT_INTEGRITY_FAILED",
479
+ message: error instanceof Error ? error.message : String(error),
480
+ exitCode: 1,
481
+ });
482
+ }
483
+ await applyCandidate(experiment.subjectRoot, experiment.campaignSnapshot.subject, candidate);
484
+ return {
485
+ candidateId: candidate.id,
486
+ subjectRoot: experiment.subjectRoot,
487
+ sourceBytes: candidate.sourceBytes,
488
+ };
489
+ };
490
+ const reportCommand = async (experimentDir, options) => {
491
+ const directory = resolve(experimentDir);
492
+ if (!(await ensureDirectory("report", directory)))
493
+ failWith(1);
494
+ const snapshot = await loadVerifiedExperimentDecisionSnapshot(directory);
495
+ const { experiment, report } = snapshot;
496
+ if (!options.out)
497
+ return report;
498
+ const out = resolve(options.out);
499
+ const outputStats = await lstat(out).catch((error) => {
500
+ if (error.code === "ENOENT")
501
+ return null;
502
+ throw error;
377
503
  });
504
+ if (outputStats && (outputStats.isSymbolicLink() || !outputStats.isFile())) {
505
+ throw new Error("An existing --out must be a regular, non-symlink file.");
506
+ }
507
+ const [physicalDirectory, physicalSubjectRoot, physicalOut] = await Promise.all([
508
+ realpath(directory),
509
+ realpath(experiment.subjectRoot),
510
+ physicalDestinationPath(out),
511
+ ]);
512
+ assertReportDestination(physicalOut, physicalDirectory, physicalSubjectRoot, experiment.campaignSnapshot.subject.exclude);
513
+ if (outputStats && !options.force) {
514
+ throw new Error(`Output already exists (use --force): ${out}`);
515
+ }
516
+ await mkdir(dirname(out), { recursive: true });
517
+ const physicalParent = await realpath(dirname(out));
518
+ const finalPhysicalOut = resolve(physicalParent, basename(out));
519
+ assertReportDestination(finalPhysicalOut, physicalDirectory, physicalSubjectRoot, experiment.campaignSnapshot.subject.exclude);
520
+ await writeReportSafely(finalPhysicalOut, `${formatExperimentDecisionMarkdown(report)}\n`, options.force ?? false);
521
+ return { path: out, report };
378
522
  };
379
- export const buildProgram = () => {
380
- const program = new Command();
381
- program
382
- .name("lux")
383
- .description("Coding-agent evaluation and improvement with deterministic assertions and human-in-the-loop runs")
384
- .version(PKG_VERSION)
385
- .addHelpText("after", "\nAgent skills:\n $ lux skills list\n $ lux skills install codex --project\n $ lux skills install claude --project");
386
- registerProjectCommands(program);
387
- registerExperimentCommands(program);
388
- registerCandidateCommands(program);
389
- registerReportingCommands(program);
390
- program.action(() => program.outputHelp());
391
- return program;
523
+ const compareCommand = async (runA, runB, options) => {
524
+ const pathA = resolve(runA);
525
+ const pathB = resolve(runB);
526
+ const invalid = (await Promise.all([pathA, pathB].map(async (path) => existsSync(path) && (await stat(path)).isDirectory() ? undefined : path))).filter((path) => path !== undefined);
527
+ if (invalid.length > 0) {
528
+ throw new Errors.IncurError({
529
+ code: "RUN_DIRECTORY_NOT_FOUND",
530
+ message: invalid.map((path) => `${path} is not a directory`).join("; "),
531
+ exitCode: 1,
532
+ });
533
+ }
534
+ const [suiteA, suiteB] = await Promise.all([loadSuiteRun(pathA), loadSuiteRun(pathB)]);
535
+ if (Boolean(suiteA) !== Boolean(suiteB)) {
536
+ throw new Errors.IncurError({
537
+ code: "INCOMPATIBLE_RUN_TYPES",
538
+ message: "Both paths must be runs or both must be suites.",
539
+ exitCode: 1,
540
+ });
541
+ }
542
+ if (suiteA && suiteB) {
543
+ const comparison = await compareSuiteRuns(pathA, pathB, {
544
+ ...(options.passRateTolerance !== undefined
545
+ ? { passRateTolerance: options.passRateTolerance }
546
+ : {}),
547
+ });
548
+ if (hasSuiteRegression(comparison)) {
549
+ throw new Errors.IncurError({
550
+ code: "REGRESSION_DETECTED",
551
+ message: formatSuiteComparisonText(comparison),
552
+ exitCode: 1,
553
+ });
554
+ }
555
+ return comparison;
556
+ }
557
+ const comparison = await compareRuns({ pathA, pathB });
558
+ if (hasRegression(comparison)) {
559
+ throw new Errors.IncurError({
560
+ code: "REGRESSION_DETECTED",
561
+ message: formatComparisonText(comparison),
562
+ exitCode: 1,
563
+ });
564
+ }
565
+ return comparison;
392
566
  };
393
567
  const caseIdOf = (casePath) => basename(casePath, extname(casePath));
394
568
  const SIGINT_EXIT_CODE = 130;
@@ -520,183 +694,214 @@ const expandLoop = (base, loop) => loop > 1
520
694
  })))
521
695
  : base;
522
696
  const experimentCommand = async (mode, campaignPath, options) => {
523
- const cwd = process.cwd();
524
- const config = await loadLuxConfig(cwd);
525
- const campaign = await resolveExperimentCampaignPath(config.rootDir, campaignPath);
526
- const loaded = await loadExperimentCampaign(campaign);
527
- experimentCampaignMode(campaign, loaded.campaign, mode);
528
- const registry = buildRegistryFromConfig(config);
529
- const controller = abortOnInterrupt();
530
- const execution = {
531
- config,
532
- registry,
533
- loaded,
534
- ...(options.runsRoot ? { runsRoot: resolve(options.runsRoot) } : {}),
535
- ...(options.resume ? { resumeFrom: resolve(options.resume) } : {}),
536
- abortSignal: controller.signal,
537
- };
538
- if (options.plan) {
539
- const plan = await planExperimentCampaign(execution, mode);
540
- let output;
541
- if (options.json) {
542
- output = `${JSON.stringify(plan, null, 2)}\n`;
697
+ const env_2 = { stack: [], error: void 0, hasError: false };
698
+ try {
699
+ const cwd = process.cwd();
700
+ const config = await loadLuxConfig(cwd);
701
+ const campaign = await resolveExperimentCampaignPath(config.rootDir, campaignPath);
702
+ const loaded = await loadExperimentCampaign(campaign);
703
+ experimentCampaignMode(campaign, loaded.campaign, mode);
704
+ const registry = buildRegistryFromConfig(config);
705
+ const controller = __addDisposableResource(env_2, abortOnInterrupt(), false);
706
+ const execution = {
707
+ config,
708
+ registry,
709
+ loaded,
710
+ ...(options.runsRoot ? { runsRoot: resolve(options.runsRoot) } : {}),
711
+ ...(options.resume ? { resumeFrom: resolve(options.resume) } : {}),
712
+ abortSignal: controller.signal,
713
+ };
714
+ if (options.plan) {
715
+ const plan = await planExperimentCampaign(execution, mode);
716
+ if (!plan.fitsProjectedCallBudgets) {
717
+ throw new Errors.IncurError({
718
+ code: "EXPERIMENT_BUDGET_EXCEEDED",
719
+ message: "The projected experiment work exceeds its configured call budgets.",
720
+ exitCode: 1,
721
+ });
722
+ }
723
+ return plan;
543
724
  }
544
- else {
545
- const projectedStatus = plan.fitsProjectedCallBudgets ? "yes" : "no";
546
- output = [
547
- `campaign: ${plan.campaignId}`,
548
- `mode: ${plan.mode}`,
549
- `evaluation: ${plan.evaluationKind}`,
550
- `profiles: ${plan.profiles.join(", ")}`,
551
- `splits: ${plan.splits.map((split) => `${split.split}=${split.examples}`).join(", ")}`,
552
- `components: ${plan.mutableComponents.length}`,
553
- `candidates: ${plan.projected.candidates}`,
554
- `metric calls: ${plan.projected.metricCalls}/${plan.budget.maxMetricCalls}`,
555
- `judge calls: ${plan.projected.judgeCalls}/${plan.budget.maxJudgeCalls}`,
556
- `proposal calls:${plan.projected.proposalCalls}/${plan.budget.maxProposalCalls}`,
557
- `unpriced calls: observed at runtime (cap ${plan.budget.maxUnpricedModelCalls})`,
558
- `projected work within budget: ${projectedStatus}`,
559
- ...plan.warnings.map((warning) => `warning: ${warning}`),
560
- "",
561
- ].join("\n");
725
+ const result = mode === "optimize"
726
+ ? await optimizeExperimentCampaign(execution)
727
+ : await runExperimentCampaign(execution);
728
+ if (controller.signal.aborted)
729
+ failWith(SIGINT_EXIT_CODE);
730
+ if (result.manifest.status !== "passed") {
731
+ throw new Errors.IncurError({
732
+ code: "EXPERIMENT_FAILED",
733
+ message: `Experiment ${result.manifest.experimentId} finished with status '${result.manifest.status}'.`,
734
+ exitCode: 1,
735
+ });
562
736
  }
563
- process.stdout.write(output);
564
- if (!plan.fitsProjectedCallBudgets)
565
- failWith(1);
566
- return;
737
+ return { experimentDir: result.experimentDir, ...result.manifest };
567
738
  }
568
- const result = mode === "optimize"
569
- ? await optimizeExperimentCampaign(execution)
570
- : await runExperimentCampaign(execution);
571
- process.stdout.write(options.json
572
- ? `${JSON.stringify({ experimentDir: result.experimentDir, ...result.manifest }, null, 2)}\n`
573
- : `${formatExperimentRun(result.experimentDir, result.manifest)}\n`);
574
- if (controller.signal.aborted)
575
- failWith(SIGINT_EXIT_CODE);
576
- else if (result.manifest.status !== "passed")
577
- failWith(1);
578
- };
579
- const runCommand = async (filters, options) => {
580
- if (options.interactive && options.claudeAnswerer) {
581
- throw new Error("lux run: --interactive and --claude-answerer are mutually exclusive");
582
- }
583
- const cwd = process.cwd();
584
- const config = await loadLuxConfig(cwd);
585
- const profile = options.profile ? config.profiles?.[options.profile] : undefined;
586
- if (options.profile && !profile) {
587
- const available = Object.keys(config.profiles ?? {}).sort();
588
- process.stderr.write(`lux: unknown profile '${options.profile}'${available.length > 0 ? `; available: ${available.join(", ")}` : ""}\n`);
589
- failWith(1);
590
- return;
739
+ catch (e_2) {
740
+ env_2.error = e_2;
741
+ env_2.hasError = true;
591
742
  }
592
- const harness = profile?.harness ?? config.harness;
593
- const registry = buildRegistryFromConfig(config).withAssertion(createRubricAssertion(harness));
594
- const defaultDriver = profile?.driver ?? config.defaultDriver;
595
- const orchestrator = new Orchestrator({
596
- registry,
597
- defaultAnswerer: profile?.answerer ?? config.defaultAnswerer,
598
- ...(defaultDriver ? { defaultDriver } : {}),
599
- ...(harness ? { harness } : {}),
600
- lifecycleFixtures: config.lifecycleFixtures,
601
- });
602
- const runsRoot = options.runsRoot ? resolve(options.runsRoot) : config.runsRoot;
603
- const subjectRoot = options.subjectRoot ? resolve(options.subjectRoot) : config.subjectRoot;
604
- // A config file is the user's opt-in to tree-wide discovery. Without one,
605
- // stay scoped to the `cases/` scaffold: discovery imports every .ts it
606
- // visits, and running module-level side effects across an arbitrary cwd
607
- // must never be the default.
608
- const root = config.casesRoot ?? (config.configPath ? config.rootDir : join(cwd, "cases"));
609
- // A configured casesRoot is a path the user declared; its absence is a
610
- // broken config, not an empty suite — never mask it behind "no cases
611
- // matched" or a lucky explicit-file run.
612
- if (config.casesRoot && !existsSync(config.casesRoot)) {
613
- process.stderr.write(`lux: casesRoot does not exist: ${config.casesRoot}\n`);
614
- failWith(1);
615
- return;
616
- }
617
- const { explicit, patterns } = partitionFilters(filters, cwd);
618
- const tags = options.tag ?? [];
619
- const hasFilter = explicit.length > 0 || patterns.length > 0 || tags.length > 0;
620
- // Explicitly named files need no discovery, so they run even without a
621
- // project; anything else would need a tree walk with nowhere safe to walk.
622
- if (config.configPath === null && !existsSync(root) && explicit.length === 0) {
623
- process.stderr.write(`lux: no lux.config.ts or cases/ directory in ${cwd} — run \`lux init\` to scaffold a project, or name a case file directly\n`);
624
- failWith(1);
625
- return;
626
- }
627
- const { cases, problems } = existsSync(root)
628
- ? await discoverCases(root)
629
- : { cases: [], problems: [] };
630
- for (const problem of problems) {
631
- process.stderr.write(`lux: skipped ${problem.path}: ${firstLine(problem.reason)}\n`);
632
- }
633
- const selected = selectSuite(cases, { explicit, patterns }, tags);
634
- // An empty selection fails every mode the same way — running, `--list`, and
635
- // `--list-tags` all exit 1, so a typo'd filter can never read as a clean pass.
636
- if (selected.length === 0) {
637
- process.stderr.write(hasFilter
638
- ? "lux: no cases matched the given filters\n"
639
- : `lux: no cases discovered under ${root}\n`);
640
- failWith(1);
641
- return;
642
- }
643
- // List the tag vocabulary of the selected cases — "which --tag values exist?".
644
- if (options.listTags) {
645
- const counts = new Map();
646
- for (const c of selected)
647
- for (const t of c.tags)
648
- counts.set(t, (counts.get(t) ?? 0) + 1);
649
- for (const [tag, n] of [...counts].sort((a, b) => a[0].localeCompare(b[0]))) {
650
- process.stdout.write(`${tag}\t${n}\n`);
651
- }
652
- process.stderr.write(`lux: ${counts.size} tag${counts.size === 1 ? "" : "s"} across ${selected.length} cases\n`);
653
- return;
654
- }
655
- const base = selected.map((c) => ({ path: c.path, id: c.id }));
656
- // Collect-only: answer "what would run?" without spawning a single agent.
657
- if (options.list) {
658
- for (const item of base)
659
- process.stdout.write(`${item.id}\n`);
660
- process.stderr.write(`lux: ${base.length} case${base.length === 1 ? "" : "s"} matched\n`);
661
- return;
743
+ finally {
744
+ __disposeResources(env_2);
662
745
  }
663
- const loop = options.loop ?? 1;
664
- const items = expandLoop(base, loop);
665
- const { concurrency, note } = resolveConcurrency(options, config.concurrency);
666
- if (note)
667
- process.stderr.write(note);
668
- const reporter = new ConsoleReporter({ verbose: options.verbose ?? false });
669
- const controller = abortOnInterrupt();
670
- reporter.begin({
671
- title: suiteTitle(patterns, tags, explicit, loop),
672
- caseIds: items.map((i) => i.id),
673
- runsRoot,
674
- loop,
675
- });
676
- let results = [];
746
+ };
747
+ const runCommand = async (filters, options, human = false) => {
748
+ const env_3 = { stack: [], error: void 0, hasError: false };
677
749
  try {
678
- const suite = await orchestrator.runSuite(items, {
750
+ if (options.interactive && options.claudeAnswerer) {
751
+ throw new Errors.ParseError({
752
+ message: "--interactive and --claude-answerer are mutually exclusive",
753
+ });
754
+ }
755
+ const cwd = process.cwd();
756
+ const config = await loadLuxConfig(cwd);
757
+ const profile = options.profile ? config.profiles?.[options.profile] : undefined;
758
+ if (options.profile && !profile) {
759
+ const available = Object.keys(config.profiles ?? {}).sort();
760
+ throw new Errors.IncurError({
761
+ code: "PROFILE_NOT_FOUND",
762
+ message: `Unknown profile '${options.profile}'${available.length > 0 ? `; available: ${available.join(", ")}` : ""}.`,
763
+ exitCode: 1,
764
+ });
765
+ }
766
+ const harness = profile?.harness ?? config.harness;
767
+ const registry = buildRegistryFromConfig(config).withAssertion(createRubricAssertion(harness));
768
+ const defaultDriver = profile?.driver ?? config.defaultDriver;
769
+ const orchestrator = new Orchestrator({
770
+ environment: profile?.environment ?? config.environment,
771
+ registry,
772
+ defaultAnswerer: profile?.answerer ?? config.defaultAnswerer,
773
+ ...(defaultDriver ? { defaultDriver } : {}),
774
+ ...(harness ? { harness } : {}),
775
+ lifecycleFixtures: config.lifecycleFixtures,
776
+ });
777
+ const runsRoot = options.runsRoot ? resolve(options.runsRoot) : config.runsRoot;
778
+ const subjectRoot = options.subjectRoot ? resolve(options.subjectRoot) : config.subjectRoot;
779
+ // A config file is the user's opt-in to tree-wide discovery. Without one,
780
+ // stay scoped to the `cases/` scaffold: discovery imports every .ts it
781
+ // visits, and running module-level side effects across an arbitrary cwd
782
+ // must never be the default.
783
+ const root = config.casesRoot ?? (config.configPath ? config.rootDir : join(cwd, "cases"));
784
+ // A configured casesRoot is a path the user declared; its absence is a
785
+ // broken config, not an empty suite — never mask it behind "no cases
786
+ // matched" or a lucky explicit-file run.
787
+ if (config.casesRoot && !existsSync(config.casesRoot)) {
788
+ throw new Errors.IncurError({
789
+ code: "CASES_ROOT_NOT_FOUND",
790
+ message: `casesRoot does not exist: ${config.casesRoot}`,
791
+ exitCode: 1,
792
+ });
793
+ }
794
+ const { explicit, patterns } = partitionFilters(filters, cwd);
795
+ const tags = normalizeTags(options.tag ?? []);
796
+ const hasFilter = explicit.length > 0 || patterns.length > 0 || tags.length > 0;
797
+ // Explicitly named files need no discovery, so they run even without a
798
+ // project; anything else would need a tree walk with nowhere safe to walk.
799
+ if (config.configPath === null && !existsSync(root) && explicit.length === 0) {
800
+ throw new Errors.IncurError({
801
+ code: "PROJECT_NOT_INITIALIZED",
802
+ message: `No lux.config.ts or cases/ directory in ${cwd}.`,
803
+ hint: "Run `lux init` or name a case file directly.",
804
+ exitCode: 1,
805
+ });
806
+ }
807
+ const { cases, problems } = existsSync(root)
808
+ ? await discoverCases(root)
809
+ : { cases: [], problems: [] };
810
+ for (const problem of problems) {
811
+ process.stderr.write(`lux: skipped ${problem.path}: ${firstLine(problem.reason)}\n`);
812
+ }
813
+ const selected = selectSuite(cases, { explicit, patterns }, tags);
814
+ // An empty selection fails every mode the same way — running, `--list`, and
815
+ // `--list-tags` all exit 1, so a typo'd filter can never read as a clean pass.
816
+ if (selected.length === 0) {
817
+ throw new Errors.IncurError({
818
+ code: "NO_CASES_MATCHED",
819
+ message: hasFilter
820
+ ? "No cases matched the given filters."
821
+ : `No cases were discovered under ${root}.`,
822
+ exitCode: 1,
823
+ });
824
+ }
825
+ // List the tag vocabulary of the selected cases — "which --tag values exist?".
826
+ if (options.listTags) {
827
+ const counts = new Map();
828
+ for (const c of selected)
829
+ for (const t of c.tags)
830
+ counts.set(t, (counts.get(t) ?? 0) + 1);
831
+ return {
832
+ caseCount: selected.length,
833
+ tags: [...counts]
834
+ .sort((a, b) => a[0].localeCompare(b[0]))
835
+ .map(([tag, count]) => ({ tag, count })),
836
+ };
837
+ }
838
+ const base = selected.map((c) => ({ path: c.path, id: c.id }));
839
+ // Collect-only: answer "what would run?" without spawning a single agent.
840
+ if (options.list) {
841
+ return { caseCount: base.length, cases: base };
842
+ }
843
+ const loop = options.loop ?? 1;
844
+ const items = expandLoop(base, loop);
845
+ const { concurrency, note } = resolveConcurrency(options, config.concurrency);
846
+ if (note)
847
+ process.stderr.write(note);
848
+ const reporter = human ? new ConsoleReporter({ verbose: options.verbose ?? false }) : undefined;
849
+ const controller = __addDisposableResource(env_3, abortOnInterrupt(), false);
850
+ reporter?.begin({
851
+ title: suiteTitle(patterns, tags, explicit, loop),
852
+ caseIds: items.map((i) => i.id),
679
853
  runsRoot,
680
- fixturesRoot: config.fixturesRoot,
681
- answererOverride: resolveAnswererOverride(options),
682
- skipJudge: options.skipJudge,
683
- observer: reporter,
684
- abortSignal: controller.signal,
685
- concurrency,
686
- subjectRoot,
687
- profileId: options.profile,
854
+ loop,
688
855
  });
689
- results = suite.results;
690
- reporter.setSuiteDir(suite.suiteDir);
856
+ let results = [];
857
+ try {
858
+ const suite = await orchestrator.runSuite(items, {
859
+ runsRoot,
860
+ fixturesRoot: config.fixturesRoot,
861
+ answererOverride: resolveAnswererOverride(options),
862
+ skipJudge: options.skipJudge,
863
+ ...(reporter ? { observer: reporter } : {}),
864
+ abortSignal: controller.signal,
865
+ concurrency,
866
+ subjectRoot,
867
+ profileId: options.profile,
868
+ });
869
+ results = suite.results;
870
+ reporter?.setSuiteDir(suite.suiteDir);
871
+ if (controller.signal.aborted)
872
+ failWith(SIGINT_EXIT_CODE);
873
+ if (results.some(isFailure)) {
874
+ throw new Errors.IncurError({
875
+ code: "SUITE_FAILED",
876
+ message: `${results.filter(isFailure).length} of ${results.length} case attempts failed. Suite: ${suite.suiteDir}`,
877
+ exitCode: 1,
878
+ });
879
+ }
880
+ return {
881
+ suiteDir: suite.suiteDir,
882
+ results: results.map((result) => ({
883
+ caseId: result.caseId,
884
+ runDir: result.runDir ?? null,
885
+ exitReason: result.exitReason,
886
+ casePassed: result.casePassed,
887
+ caseScore: result.result?.grading?.caseScore ?? null,
888
+ assertionsPassed: result.result?.grading?.assertions.filter((assertion) => assertion.passed).length ?? 0,
889
+ assertionsTotal: result.result?.grading?.assertions.length ?? 0,
890
+ })),
891
+ };
892
+ }
893
+ finally {
894
+ reporter?.finish();
895
+ }
896
+ }
897
+ catch (e_3) {
898
+ env_3.error = e_3;
899
+ env_3.hasError = true;
691
900
  }
692
901
  finally {
693
- reporter.finish();
902
+ __disposeResources(env_3);
694
903
  }
695
- if (controller.signal.aborted)
696
- failWith(SIGINT_EXIT_CODE);
697
- else if (results.some(isFailure))
698
- failWith(1);
699
904
  };
700
905
  // Internal exports for testing
701
- export { expandLoop, isFailure, parsePositiveInt, partitionFilters, resolveConcurrency, selectSuite, };
906
+ export { expandLoop, isFailure, partitionFilters, resolveConcurrency, selectSuite };
702
907
  //# sourceMappingURL=program.js.map