@velum-labs/routekit-eval-setup 1.4.0 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (341) hide show
  1. package/dist/adapters/authoring-responses-request.d.ts +5 -0
  2. package/dist/adapters/authoring-responses-request.js +38 -0
  3. package/dist/adapters/evaluation-evidence-freshness.d.ts +7 -0
  4. package/dist/adapters/evaluation-evidence-freshness.js +76 -0
  5. package/dist/adapters/git-task-history.d.ts +67 -0
  6. package/dist/adapters/git-task-history.js +171 -0
  7. package/dist/adapters/integrated-repository-history.d.ts +21 -0
  8. package/dist/adapters/integrated-repository-history.js +175 -0
  9. package/dist/adapters/repository-command-diagnostic.d.ts +8 -0
  10. package/dist/adapters/repository-command-diagnostic.js +46 -0
  11. package/dist/adapters/repository-command-evidence.d.ts +13 -0
  12. package/dist/adapters/repository-command-evidence.js +102 -0
  13. package/dist/adapters/repository-command-runner.d.ts +238 -0
  14. package/dist/adapters/repository-command-runner.js +1483 -0
  15. package/dist/adapters/repository-import-context.d.ts +47 -0
  16. package/dist/adapters/repository-import-context.js +469 -0
  17. package/dist/adapters/repository-node-test-reporter.d.ts +3 -0
  18. package/dist/adapters/repository-node-test-reporter.js +27 -0
  19. package/dist/adapters/repository-review-evidence.d.ts +39 -0
  20. package/dist/adapters/repository-review-evidence.js +632 -0
  21. package/dist/adapters/repository-seed-selection.d.ts +7 -0
  22. package/dist/adapters/repository-seed-selection.js +79 -0
  23. package/dist/adapters/repository-solution-edits.d.ts +49 -0
  24. package/dist/adapters/repository-solution-edits.js +136 -0
  25. package/dist/adapters/repository-vitest-phase-adapter.d.ts +8 -0
  26. package/dist/adapters/repository-vitest-phase-adapter.js +310 -0
  27. package/dist/adapters/repository-vitest-reporter.d.ts +24 -0
  28. package/dist/adapters/repository-vitest-reporter.js +314 -0
  29. package/dist/adapters/strict-authoring-schema.d.ts +5 -0
  30. package/dist/adapters/strict-authoring-schema.js +158 -0
  31. package/dist/adapters/test-discovery.d.ts +30 -0
  32. package/dist/adapters/test-discovery.js +124 -0
  33. package/dist/adapters/typescript-repository-index.d.ts +51 -0
  34. package/dist/adapters/typescript-repository-index.js +226 -0
  35. package/dist/agentic-capabilities-protocol.d.ts +1373 -0
  36. package/dist/agentic-capabilities-protocol.js +786 -0
  37. package/dist/case-checkpoint-store.d.ts +29 -0
  38. package/dist/case-checkpoint-store.js +133 -0
  39. package/dist/case-pipeline-protocol-v2.d.ts +184 -0
  40. package/dist/case-pipeline-protocol-v2.js +193 -0
  41. package/dist/case-pipeline-protocol.d.ts +2626 -0
  42. package/dist/case-pipeline-protocol.js +371 -0
  43. package/dist/effect-api.d.ts +74 -12
  44. package/dist/effect-api.js +56 -7
  45. package/dist/errors.d.ts +31 -0
  46. package/dist/errors.js +10 -0
  47. package/dist/eval-capability-execution-envelope.d.ts +64 -0
  48. package/dist/eval-capability-execution-envelope.js +98 -0
  49. package/dist/eval-capability-policy.d.ts +90 -0
  50. package/dist/eval-capability-policy.js +107 -0
  51. package/dist/eval-event-log.d.ts +51 -0
  52. package/dist/eval-event-log.js +69 -10
  53. package/dist/evaluation-authoring-policy.d.ts +18 -0
  54. package/dist/evaluation-authoring-policy.js +19 -0
  55. package/dist/evaluation-authoring-validation.d.ts +22 -0
  56. package/dist/evaluation-authoring-validation.js +72 -0
  57. package/dist/evaluation-evidence.d.ts +20 -0
  58. package/dist/evaluation-evidence.js +319 -0
  59. package/dist/evaluation-grader-calibration-protocol.d.ts +108 -0
  60. package/dist/evaluation-grader-calibration-protocol.js +80 -0
  61. package/dist/evaluation-grader-calibration.d.ts +18 -0
  62. package/dist/evaluation-grader-calibration.js +334 -0
  63. package/dist/evaluation-grading-policy.d.ts +24 -0
  64. package/dist/evaluation-grading-policy.js +54 -0
  65. package/dist/evaluation-proposal-policy.d.ts +4 -0
  66. package/dist/evaluation-proposal-policy.js +91 -0
  67. package/dist/evaluation-source-retrieval.d.ts +68 -0
  68. package/dist/evaluation-source-retrieval.js +513 -0
  69. package/dist/evaluation-structure-policy.d.ts +29 -0
  70. package/dist/evaluation-structure-policy.js +138 -0
  71. package/dist/index.d.ts +124 -19
  72. package/dist/index.js +69 -12
  73. package/dist/inspection.js +2 -3
  74. package/dist/project-artifacts.d.ts +2 -2
  75. package/dist/project-artifacts.js +26 -121
  76. package/dist/project-authoring.d.ts +66 -5
  77. package/dist/project-authoring.js +783 -109
  78. package/dist/project-contracts.d.ts +292 -82
  79. package/dist/project-contracts.js +65 -7
  80. package/dist/project-store.js +2 -1
  81. package/dist/project-workflow.d.ts +3 -3
  82. package/dist/project-workflow.js +116 -18
  83. package/dist/repository-adversary-protocol.d.ts +64 -0
  84. package/dist/repository-adversary-protocol.js +105 -0
  85. package/dist/repository-behavior-protocol.d.ts +188 -0
  86. package/dist/repository-behavior-protocol.js +202 -0
  87. package/dist/repository-benchmark-protocol.d.ts +487 -0
  88. package/dist/repository-benchmark-protocol.js +96 -0
  89. package/dist/repository-execution-protocol.d.ts +150 -0
  90. package/dist/repository-execution-protocol.js +38 -0
  91. package/dist/repository-fixture-instructions.d.ts +3 -0
  92. package/dist/repository-fixture-instructions.js +91 -0
  93. package/dist/repository-fixture-protocol.d.ts +79 -0
  94. package/dist/repository-fixture-protocol.js +79 -0
  95. package/dist/repository-foundry-plan-protocol.d.ts +118 -0
  96. package/dist/repository-foundry-plan-protocol.js +296 -0
  97. package/dist/repository-foundry-progress-protocol.d.ts +52 -0
  98. package/dist/repository-foundry-progress-protocol.js +52 -0
  99. package/dist/repository-improvement-protocol.d.ts +100 -0
  100. package/dist/repository-improvement-protocol.js +106 -0
  101. package/dist/repository-language-model-protocol.d.ts +43 -0
  102. package/dist/repository-language-model-protocol.js +146 -0
  103. package/dist/repository-oracle-coverage-protocol.d.ts +18 -0
  104. package/dist/repository-oracle-coverage-protocol.js +39 -0
  105. package/dist/repository-oracle-execution-binding.d.ts +27 -0
  106. package/dist/repository-oracle-execution-binding.js +59 -0
  107. package/dist/repository-oracle-protocol.d.ts +230 -0
  108. package/dist/repository-oracle-protocol.js +156 -0
  109. package/dist/repository-oracle-scope-policy.d.ts +22 -0
  110. package/dist/repository-oracle-scope-policy.js +92 -0
  111. package/dist/repository-quality-policy.d.ts +15 -0
  112. package/dist/repository-quality-policy.js +357 -0
  113. package/dist/repository-routing-benchmark-protocol.d.ts +176 -0
  114. package/dist/repository-routing-benchmark-protocol.js +103 -0
  115. package/dist/repository-routing-model-protocol.d.ts +36 -0
  116. package/dist/repository-routing-model-protocol.js +89 -0
  117. package/dist/repository-routing-plan-protocol.d.ts +112 -0
  118. package/dist/repository-routing-plan-protocol.js +58 -0
  119. package/dist/repository-routing-quality-policy.d.ts +9 -0
  120. package/dist/repository-routing-quality-policy.js +191 -0
  121. package/dist/repository-seed-qualification-progress-protocol.d.ts +205 -0
  122. package/dist/repository-seed-qualification-progress-protocol.js +28 -0
  123. package/dist/repository-semantic-calibration-protocol.d.ts +768 -0
  124. package/dist/repository-semantic-calibration-protocol.js +276 -0
  125. package/dist/repository-semantic-calibration.d.ts +163 -0
  126. package/dist/repository-semantic-calibration.js +581 -0
  127. package/dist/repository-specification-contract-facts-protocol.d.ts +224 -0
  128. package/dist/repository-specification-contract-facts-protocol.js +276 -0
  129. package/dist/repository-specification-critique-protocol.d.ts +189 -0
  130. package/dist/repository-specification-critique-protocol.js +103 -0
  131. package/dist/repository-task-family-protocol.d.ts +24 -0
  132. package/dist/repository-task-family-protocol.js +37 -0
  133. package/dist/repository-task-seed-protocol.d.ts +384 -0
  134. package/dist/repository-task-seed-protocol.js +236 -0
  135. package/dist/repository-trajectory-protocol.d.ts +20 -0
  136. package/dist/repository-trajectory-protocol.js +42 -0
  137. package/dist/service.js +1 -1
  138. package/dist/services/adversary/service.d.ts +64 -0
  139. package/dist/services/adversary/service.js +330 -0
  140. package/dist/services/benchmark-compiler/service.d.ts +450 -0
  141. package/dist/services/benchmark-compiler/service.js +9 -0
  142. package/dist/services/budgeted-model/service.d.ts +118 -0
  143. package/dist/services/budgeted-model/service.js +460 -0
  144. package/dist/services/case-authoring/service.d.ts +163 -0
  145. package/dist/services/case-authoring/service.js +1456 -0
  146. package/dist/services/case-finalization/service.d.ts +283 -0
  147. package/dist/services/case-finalization/service.js +370 -0
  148. package/dist/services/case-generation/service.d.ts +619 -0
  149. package/dist/services/case-generation/service.js +2628 -0
  150. package/dist/services/case-pipeline/service.d.ts +31 -0
  151. package/dist/services/case-pipeline/service.js +485 -0
  152. package/dist/services/case-pipeline-v2/service.d.ts +70 -0
  153. package/dist/services/case-pipeline-v2/service.js +477 -0
  154. package/dist/services/command-observability/service.d.ts +13 -0
  155. package/dist/services/command-observability/service.js +3 -0
  156. package/dist/services/dimension-labeling/service.d.ts +77 -0
  157. package/dist/services/dimension-labeling/service.js +188 -0
  158. package/dist/services/eval-candidate/service.d.ts +208 -0
  159. package/dist/services/eval-candidate/service.js +64 -0
  160. package/dist/services/eval-capabilities/service.d.ts +183 -0
  161. package/dist/services/eval-capabilities/service.js +1433 -0
  162. package/dist/services/eval-environment/service.d.ts +173 -0
  163. package/dist/services/eval-environment/service.js +127 -0
  164. package/dist/services/evidence-reconstruction/service.d.ts +36 -0
  165. package/dist/services/evidence-reconstruction/service.js +145 -0
  166. package/dist/services/fixture-builder/service.d.ts +62 -0
  167. package/dist/services/fixture-builder/service.js +36 -0
  168. package/dist/services/fixture-validation/service.d.ts +75 -0
  169. package/dist/services/fixture-validation/service.js +295 -0
  170. package/dist/services/foundry/service.d.ts +831 -0
  171. package/dist/services/foundry/service.js +442 -0
  172. package/dist/services/foundry-progress/service.d.ts +62 -0
  173. package/dist/services/foundry-progress/service.js +149 -0
  174. package/dist/services/foundry-v2/service.d.ts +54 -0
  175. package/dist/services/foundry-v2/service.js +28 -0
  176. package/dist/services/grounded-authoring/service.d.ts +126 -0
  177. package/dist/services/grounded-authoring/service.js +822 -0
  178. package/dist/services/historical-case/service.d.ts +722 -0
  179. package/dist/services/historical-case/service.js +177 -0
  180. package/dist/services/improvement-loop/service.d.ts +59 -0
  181. package/dist/services/improvement-loop/service.js +176 -0
  182. package/dist/services/language-model/service.d.ts +52 -0
  183. package/dist/services/language-model/service.js +194 -0
  184. package/dist/services/oracle-builder/service.d.ts +146 -0
  185. package/dist/services/oracle-builder/service.js +513 -0
  186. package/dist/services/oracle-coverage/service.d.ts +28 -0
  187. package/dist/services/oracle-coverage/service.js +50 -0
  188. package/dist/services/oracle-coverage-witness/service.d.ts +130 -0
  189. package/dist/services/oracle-coverage-witness/service.js +538 -0
  190. package/dist/services/pipeline-challenge/service.d.ts +551 -0
  191. package/dist/services/pipeline-challenge/service.js +427 -0
  192. package/dist/services/pipeline-controls/service.d.ts +130 -0
  193. package/dist/services/pipeline-controls/service.js +483 -0
  194. package/dist/services/pipeline-oracle/service.d.ts +8 -0
  195. package/dist/services/pipeline-oracle/service.js +256 -0
  196. package/dist/services/pipeline-seed/service.d.ts +298 -0
  197. package/dist/services/pipeline-seed/service.js +428 -0
  198. package/dist/services/pipeline-spec/service.d.ts +103 -0
  199. package/dist/services/pipeline-spec/service.js +619 -0
  200. package/dist/services/pipeline-tournament/service.d.ts +258 -0
  201. package/dist/services/pipeline-tournament/service.js +476 -0
  202. package/dist/services/quality-gate/service.d.ts +233 -0
  203. package/dist/services/quality-gate/service.js +136 -0
  204. package/dist/services/repository-bundle/service.d.ts +33 -0
  205. package/dist/services/repository-bundle/service.js +114 -0
  206. package/dist/services/repository-model/service.d.ts +105 -0
  207. package/dist/services/repository-model/service.js +250 -0
  208. package/dist/services/repository-public-artifact/service.d.ts +133 -0
  209. package/dist/services/repository-public-artifact/service.js +330 -0
  210. package/dist/services/routing-benchmark/service.d.ts +362 -0
  211. package/dist/services/routing-benchmark/service.js +96 -0
  212. package/dist/services/specification-critic/service.d.ts +92 -0
  213. package/dist/services/specification-critic/service.js +172 -0
  214. package/dist/services/task-family/service.d.ts +40 -0
  215. package/dist/services/task-family/service.js +55 -0
  216. package/dist/services/task-seed/service.d.ts +906 -0
  217. package/dist/services/task-seed/service.js +1406 -0
  218. package/dist/services/task-specification/service.d.ts +27 -0
  219. package/dist/services/task-specification/service.js +40 -0
  220. package/dist/services/trajectory-policy/service.d.ts +110 -0
  221. package/dist/services/trajectory-policy/service.js +216 -0
  222. package/dist/test/agentic-capabilities-protocol.test.d.ts +1 -0
  223. package/dist/test/agentic-capabilities-protocol.test.js +570 -0
  224. package/dist/test/agentic-capabilities.test.d.ts +1 -0
  225. package/dist/test/agentic-capabilities.test.js +1461 -0
  226. package/dist/test/agentic-environment.test.d.ts +1 -0
  227. package/dist/test/agentic-environment.test.js +213 -0
  228. package/dist/test/case-pipeline-foundation.test.d.ts +1 -0
  229. package/dist/test/case-pipeline-foundation.test.js +535 -0
  230. package/dist/test/case-pipeline-protocol-v2.test.d.ts +1 -0
  231. package/dist/test/case-pipeline-protocol-v2.test.js +124 -0
  232. package/dist/test/case-pipeline-v2.test.d.ts +1 -0
  233. package/dist/test/case-pipeline-v2.test.js +286 -0
  234. package/dist/test/case-pipeline.test.d.ts +1 -0
  235. package/dist/test/case-pipeline.test.js +851 -0
  236. package/dist/test/eval-capability-policy.test.d.ts +1 -0
  237. package/dist/test/eval-capability-policy.test.js +50 -0
  238. package/dist/test/eval-event-log.test.js +34 -1
  239. package/dist/test/evaluation-evidence-freshness.test.d.ts +1 -0
  240. package/dist/test/evaluation-evidence-freshness.test.js +44 -0
  241. package/dist/test/evaluation-evidence.test.d.ts +1 -0
  242. package/dist/test/evaluation-evidence.test.js +230 -0
  243. package/dist/test/evaluation-grader-calibration.test.d.ts +1 -0
  244. package/dist/test/evaluation-grader-calibration.test.js +373 -0
  245. package/dist/test/evaluation-proposal-digest.test.d.ts +1 -0
  246. package/dist/test/evaluation-proposal-digest.test.js +187 -0
  247. package/dist/test/evaluation-source-retrieval.test.d.ts +1 -0
  248. package/dist/test/evaluation-source-retrieval.test.js +237 -0
  249. package/dist/test/evaluation-structure-policy.test.d.ts +1 -0
  250. package/dist/test/evaluation-structure-policy.test.js +196 -0
  251. package/dist/test/fixtures/repository-resource-panel.d.ts +39 -0
  252. package/dist/test/fixtures/repository-resource-panel.js +111 -0
  253. package/dist/test/fixtures/vitest-boundary-panel.d.ts +84 -0
  254. package/dist/test/fixtures/vitest-boundary-panel.js +120 -0
  255. package/dist/test/fixtures/vitest-phase-panel.d.ts +135 -0
  256. package/dist/test/fixtures/vitest-phase-panel.js +213 -0
  257. package/dist/test/fixtures/vitest-reporter-results.d.ts +76 -0
  258. package/dist/test/fixtures/vitest-reporter-results.js +94 -0
  259. package/dist/test/grounded-authoring.test.d.ts +1 -0
  260. package/dist/test/grounded-authoring.test.js +565 -0
  261. package/dist/test/integrated-repository-history.test.d.ts +1 -0
  262. package/dist/test/integrated-repository-history.test.js +227 -0
  263. package/dist/test/project-authoring.test.js +593 -43
  264. package/dist/test/project-workflow.test.js +395 -32
  265. package/dist/test/repository-authoring-artifacts.test.d.ts +1 -0
  266. package/dist/test/repository-authoring-artifacts.test.js +185 -0
  267. package/dist/test/repository-bundle.test.d.ts +1 -0
  268. package/dist/test/repository-bundle.test.js +52 -0
  269. package/dist/test/repository-case-generation.test.d.ts +1 -0
  270. package/dist/test/repository-case-generation.test.js +3465 -0
  271. package/dist/test/repository-command-diagnostic.test.d.ts +1 -0
  272. package/dist/test/repository-command-diagnostic.test.js +55 -0
  273. package/dist/test/repository-command-signals.test.d.ts +1 -0
  274. package/dist/test/repository-command-signals.test.js +124 -0
  275. package/dist/test/repository-fixture-scope-coverage.test.d.ts +1 -0
  276. package/dist/test/repository-fixture-scope-coverage.test.js +127 -0
  277. package/dist/test/repository-fixture-validation.test.d.ts +1 -0
  278. package/dist/test/repository-fixture-validation.test.js +362 -0
  279. package/dist/test/repository-foundry-progress.test.d.ts +1 -0
  280. package/dist/test/repository-foundry-progress.test.js +110 -0
  281. package/dist/test/repository-foundry-quality.test.d.ts +1 -0
  282. package/dist/test/repository-foundry-quality.test.js +1138 -0
  283. package/dist/test/repository-import-context.test.d.ts +1 -0
  284. package/dist/test/repository-import-context.test.js +354 -0
  285. package/dist/test/repository-model-authoring.test.d.ts +1 -0
  286. package/dist/test/repository-model-authoring.test.js +544 -0
  287. package/dist/test/repository-model.test.d.ts +1 -0
  288. package/dist/test/repository-model.test.js +2195 -0
  289. package/dist/test/repository-node-test-reporter.test.d.ts +1 -0
  290. package/dist/test/repository-node-test-reporter.test.js +104 -0
  291. package/dist/test/repository-oracle-concurrency.test.d.ts +1 -0
  292. package/dist/test/repository-oracle-concurrency.test.js +542 -0
  293. package/dist/test/repository-oracle-coverage-witness.test.d.ts +1 -0
  294. package/dist/test/repository-oracle-coverage-witness.test.js +511 -0
  295. package/dist/test/repository-oracle-coverage.test.d.ts +1 -0
  296. package/dist/test/repository-oracle-coverage.test.js +168 -0
  297. package/dist/test/repository-oracle-evidence.test.d.ts +1 -0
  298. package/dist/test/repository-oracle-evidence.test.js +185 -0
  299. package/dist/test/repository-oracle-plan.test.d.ts +1 -0
  300. package/dist/test/repository-oracle-plan.test.js +176 -0
  301. package/dist/test/repository-overlay-isolation.test.d.ts +1 -0
  302. package/dist/test/repository-overlay-isolation.test.js +85 -0
  303. package/dist/test/repository-preparation-cache.test.d.ts +1 -0
  304. package/dist/test/repository-preparation-cache.test.js +414 -0
  305. package/dist/test/repository-public-artifact.test.d.ts +1 -0
  306. package/dist/test/repository-public-artifact.test.js +273 -0
  307. package/dist/test/repository-qualification-diagnostics.test.d.ts +1 -0
  308. package/dist/test/repository-qualification-diagnostics.test.js +524 -0
  309. package/dist/test/repository-reference-authoring.test.d.ts +1 -0
  310. package/dist/test/repository-reference-authoring.test.js +1633 -0
  311. package/dist/test/repository-review-evidence-v2.test.d.ts +1 -0
  312. package/dist/test/repository-review-evidence-v2.test.js +183 -0
  313. package/dist/test/repository-review-evidence.test.d.ts +1 -0
  314. package/dist/test/repository-review-evidence.test.js +124 -0
  315. package/dist/test/repository-seed-exclusions.test.d.ts +1 -0
  316. package/dist/test/repository-seed-exclusions.test.js +96 -0
  317. package/dist/test/repository-seed-selection.test.d.ts +1 -0
  318. package/dist/test/repository-seed-selection.test.js +504 -0
  319. package/dist/test/repository-semantic-calibration.test.d.ts +1 -0
  320. package/dist/test/repository-semantic-calibration.test.js +688 -0
  321. package/dist/test/repository-solution-edits.test.d.ts +1 -0
  322. package/dist/test/repository-solution-edits.test.js +377 -0
  323. package/dist/test/repository-specification-budget.test.d.ts +1 -0
  324. package/dist/test/repository-specification-budget.test.js +171 -0
  325. package/dist/test/repository-specification-contract-checkpoint.test.d.ts +1 -0
  326. package/dist/test/repository-specification-contract-checkpoint.test.js +228 -0
  327. package/dist/test/repository-specification-contract-facts.test.d.ts +1 -0
  328. package/dist/test/repository-specification-contract-facts.test.js +177 -0
  329. package/dist/test/repository-trajectory-authoring.test.d.ts +1 -0
  330. package/dist/test/repository-trajectory-authoring.test.js +176 -0
  331. package/dist/test/repository-valid-control-plan.test.d.ts +1 -0
  332. package/dist/test/repository-valid-control-plan.test.js +45 -0
  333. package/dist/test/repository-vitest-phase.test.d.ts +1 -0
  334. package/dist/test/repository-vitest-phase.test.js +848 -0
  335. package/dist/test/repository-vitest-reporter.test.d.ts +1 -0
  336. package/dist/test/repository-vitest-reporter.test.js +158 -0
  337. package/dist/test/repository-workspace-build.test.d.ts +1 -0
  338. package/dist/test/repository-workspace-build.test.js +160 -0
  339. package/dist/test/strict-authoring-schema.test.d.ts +1 -0
  340. package/dist/test/strict-authoring-schema.test.js +169 -0
  341. package/package.json +48 -7
@@ -0,0 +1,64 @@
1
+ import type { AgenticCapabilityNameV1 } from "./agentic-capabilities-protocol.js";
2
+ import { type EvalCapabilityScientificPhaseV1 } from "./eval-capability-policy.js";
3
+ export type EvalCapabilityScientificReserveV1 = Readonly<{
4
+ calls: number;
5
+ wallTimeMs: number;
6
+ attempts: number;
7
+ }>;
8
+ export type EvalCapabilityExecutionBudgetV1 = Readonly<{
9
+ calls: number;
10
+ wallTimeMs: number;
11
+ }>;
12
+ /**
13
+ * Time kept outside scientific execution for checkpoint/archive persistence
14
+ * and the host's durable capability-completion receipt.
15
+ */
16
+ export declare const EVAL_CAPABILITY_DURABLE_FINALIZATION_RESERVE_MS_V1 = 30000;
17
+ export type EvalCapabilityExecutionEnvelopeDecisionV1 = Readonly<{
18
+ version: 1;
19
+ capability: AgenticCapabilityNameV1;
20
+ phase: EvalCapabilityScientificPhaseV1 | null;
21
+ available: EvalCapabilityExecutionBudgetV1;
22
+ requestedMaximum: EvalCapabilityExecutionBudgetV1;
23
+ downstreamReserve: EvalCapabilityScientificReserveV1;
24
+ finalizationReserve: Readonly<{
25
+ wallTimeMs: number;
26
+ }>;
27
+ minimumRequired: EvalCapabilityExecutionBudgetV1;
28
+ availableForCapability: EvalCapabilityExecutionBudgetV1;
29
+ }> & (Readonly<{
30
+ admitted: true;
31
+ envelope: Readonly<{
32
+ maximumModelCalls: number;
33
+ maximumWallTimeMs: number;
34
+ }>;
35
+ }> | Readonly<{
36
+ admitted: false;
37
+ dimension: "calls" | "wall_time";
38
+ envelope: null;
39
+ }>);
40
+ /**
41
+ * The immutable downstream reserve for a scientific capability. This is the
42
+ * single package-owned source used by admission and execution; campaign-size
43
+ * ratios must not weaken these fixed phase reserves.
44
+ */
45
+ export declare const evalCapabilityScientificDownstreamReserveV1: (capability: AgenticCapabilityNameV1) => Readonly<{
46
+ phase: EvalCapabilityScientificPhaseV1 | null;
47
+ reserve: EvalCapabilityScientificReserveV1;
48
+ }>;
49
+ /**
50
+ * Computes the positive execution slice that fits after preserving the fixed
51
+ * downstream reserve and durable finalization time. Requested maxima are
52
+ * ceilings, not all-or-nothing reservations: a smaller positive slice is
53
+ * admitted whenever one fits.
54
+ *
55
+ * The structured decision retains every value needed to explain or persist the
56
+ * decision without reconstructing it from prose.
57
+ */
58
+ export declare const evalCapabilityExecutionEnvelopeV1: (input: {
59
+ readonly capability: AgenticCapabilityNameV1;
60
+ readonly requestedMaximumModelCalls?: number;
61
+ readonly requestedMaximumWallMs?: number;
62
+ readonly campaignModelCallsRemaining: number;
63
+ readonly campaignWallTimeRemainingMs: number;
64
+ }) => EvalCapabilityExecutionEnvelopeDecisionV1;
@@ -0,0 +1,98 @@
1
+ import { EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1, evalCapabilityScientificMaximumModelCallsV1, evalCapabilityScientificMaximumWallTimeMsV1, evalCapabilityScientificPhaseV1 } from "./eval-capability-policy.js";
2
+ /**
3
+ * Time kept outside scientific execution for checkpoint/archive persistence
4
+ * and the host's durable capability-completion receipt.
5
+ */
6
+ export const EVAL_CAPABILITY_DURABLE_FINALIZATION_RESERVE_MS_V1 = 30_000;
7
+ const positiveInteger = (value, fallback) => Number.isSafeInteger(value) && value > 0 ? value : fallback;
8
+ const availableInteger = (value) => Number.isSafeInteger(value) && value > 0 ? value : 0;
9
+ /**
10
+ * The immutable downstream reserve for a scientific capability. This is the
11
+ * single package-owned source used by admission and execution; campaign-size
12
+ * ratios must not weaken these fixed phase reserves.
13
+ */
14
+ export const evalCapabilityScientificDownstreamReserveV1 = (capability) => {
15
+ const phase = evalCapabilityScientificPhaseV1(capability);
16
+ const reserve = phase === null
17
+ ? { calls: 0, wallTimeMs: 0, attempts: 0 }
18
+ : EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1[phase].downstream;
19
+ return { phase, reserve };
20
+ };
21
+ /**
22
+ * Computes the positive execution slice that fits after preserving the fixed
23
+ * downstream reserve and durable finalization time. Requested maxima are
24
+ * ceilings, not all-or-nothing reservations: a smaller positive slice is
25
+ * admitted whenever one fits.
26
+ *
27
+ * The structured decision retains every value needed to explain or persist the
28
+ * decision without reconstructing it from prose.
29
+ */
30
+ export const evalCapabilityExecutionEnvelopeV1 = (input) => {
31
+ const { phase, reserve } = evalCapabilityScientificDownstreamReserveV1(input.capability);
32
+ const scientific = phase !== null;
33
+ const finalizationReserveWallTimeMs = scientific
34
+ ? EVAL_CAPABILITY_DURABLE_FINALIZATION_RESERVE_MS_V1
35
+ : 0;
36
+ const available = {
37
+ calls: availableInteger(input.campaignModelCallsRemaining),
38
+ wallTimeMs: availableInteger(input.campaignWallTimeRemainingMs)
39
+ };
40
+ const requestedMaximum = {
41
+ calls: Math.min(scientific
42
+ ? evalCapabilityScientificMaximumModelCallsV1(input.capability)
43
+ : 12, positiveInteger(input.requestedMaximumModelCalls ?? 6, 6)),
44
+ wallTimeMs: Math.min(scientific
45
+ ? evalCapabilityScientificMaximumWallTimeMsV1(input.capability)
46
+ : 300_000, positiveInteger(input.requestedMaximumWallMs ?? 120_000, 120_000))
47
+ };
48
+ const minimumRequired = {
49
+ calls: reserve.calls + 1,
50
+ wallTimeMs: reserve.wallTimeMs + finalizationReserveWallTimeMs + 1
51
+ };
52
+ const availableForCapability = {
53
+ calls: Math.max(0, available.calls - reserve.calls),
54
+ wallTimeMs: Math.max(0, available.wallTimeMs -
55
+ reserve.wallTimeMs -
56
+ finalizationReserveWallTimeMs)
57
+ };
58
+ const common = {
59
+ version: 1,
60
+ capability: input.capability,
61
+ phase,
62
+ available,
63
+ requestedMaximum,
64
+ downstreamReserve: reserve,
65
+ finalizationReserve: {
66
+ wallTimeMs: finalizationReserveWallTimeMs
67
+ },
68
+ minimumRequired,
69
+ availableForCapability
70
+ };
71
+ // Read-only/control capabilities remain reachable after scientific exhaustion
72
+ // so callers can inspect progress and finish the campaign. They receive the
73
+ // legacy minimal positive execution fence and no downstream reserve.
74
+ if (!scientific) {
75
+ return {
76
+ ...common,
77
+ admitted: true,
78
+ envelope: {
79
+ maximumModelCalls: Math.max(1, Math.min(requestedMaximum.calls, available.calls)),
80
+ maximumWallTimeMs: Math.max(1, Math.min(requestedMaximum.wallTimeMs, available.wallTimeMs))
81
+ }
82
+ };
83
+ }
84
+ if (availableForCapability.calls < 1) {
85
+ return { ...common, admitted: false, dimension: "calls", envelope: null };
86
+ }
87
+ if (availableForCapability.wallTimeMs < 1) {
88
+ return { ...common, admitted: false, dimension: "wall_time", envelope: null };
89
+ }
90
+ return {
91
+ ...common,
92
+ admitted: true,
93
+ envelope: {
94
+ maximumModelCalls: Math.min(requestedMaximum.calls, availableForCapability.calls),
95
+ maximumWallTimeMs: Math.min(requestedMaximum.wallTimeMs, availableForCapability.wallTimeMs)
96
+ }
97
+ };
98
+ };
@@ -0,0 +1,90 @@
1
+ import type { AgenticCapabilityNameV1, AgenticProgressFingerprintV1, AgenticScientificProgressReceiptV1 } from "./agentic-capabilities-protocol.js";
2
+ /**
3
+ * A scheduling slice that only advances verified durable work is not another
4
+ * scientific attempt. Charging it makes sufficiently slow but productive
5
+ * cases impossible, regardless of their remaining campaign budget.
6
+ * New decisions, resets, legacy/unbound receipts, and unchanged work still
7
+ * consume attempts. Calls, spend, wall time and the no-progress circuit are
8
+ * independent and are never refunded.
9
+ */
10
+ export declare const evalCapabilityScientificAttemptCostV1: (receipt: Pick<AgenticScientificProgressReceiptV1, "previousFingerprint" | "resultingFingerprint" | "inputRevision" | "outputRevision" | "gatesCompleted" | "gatesInvalidated" | "sameFingerprintCount">) => 0 | 1;
11
+ /**
12
+ * No-op calls to other tools cannot reset a capability's stall counter. A
13
+ * genuine accepted-state change does reset it, even if a later repair returns
14
+ * to the same state. Worker admission and Cloud receipt validation share this
15
+ * rule; the global previous-fingerprint chain is verified independently.
16
+ * History must be ordered oldest first and scoped to one candidate revision.
17
+ */
18
+ export declare const evalCapabilityScientificProgressPredecessorV1: (input: {
19
+ readonly capability: AgenticCapabilityNameV1;
20
+ readonly fingerprint: AgenticProgressFingerprintV1;
21
+ readonly history: readonly AgenticScientificProgressReceiptV1[];
22
+ }) => AgenticScientificProgressReceiptV1 | undefined;
23
+ /**
24
+ * One scientific capability is allowed enough time to finish a complete
25
+ * repository-grounded role chain (analysis, visible authoring, and private
26
+ * review) under normal provider tail latency. Cloud admission, the sandbox
27
+ * executor, and the native tool import this dependency-light policy module so
28
+ * loading an OpenCode tool never initializes the server-side foundry runtime.
29
+ */
30
+ export declare const EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_MODEL_CALLS_V1 = 12;
31
+ export declare const EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_WALL_TIME_MS_V1: number;
32
+ /**
33
+ * Versioned capability slices are derived from the maximum number of native
34
+ * foundry turns that must remain contiguous to preserve tool state. Most
35
+ * capabilities fit one base slice. Oracle authoring and repair own longer
36
+ * tool-using conversations and therefore receive multiple contiguous slices
37
+ * instead of repeatedly restarting after the base ceiling.
38
+ */
39
+ export declare const EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1: {
40
+ readonly author_oracle: {
41
+ readonly modelCallSlices: 3;
42
+ readonly wallTimeSlices: 3;
43
+ };
44
+ readonly repair_oracle: {
45
+ readonly modelCallSlices: 2;
46
+ readonly wallTimeSlices: 2;
47
+ };
48
+ };
49
+ export declare const evalCapabilityScientificMaximumModelCallsV1: (capability: AgenticCapabilityNameV1) => number;
50
+ export declare const evalCapabilityScientificMaximumWallTimeMsV1: (capability: AgenticCapabilityNameV1) => number;
51
+ export declare const EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1: {
52
+ readonly specification: {
53
+ readonly capabilities: readonly ["draft_specification"];
54
+ readonly downstream: {
55
+ readonly calls: 24;
56
+ readonly wallTimeMs: 4500000;
57
+ readonly attempts: 6;
58
+ };
59
+ readonly maximumAttempts: 8;
60
+ };
61
+ readonly environment: {
62
+ readonly capabilities: readonly ["plan_environment", "probe_fixture", "qualify_reference"];
63
+ readonly downstream: {
64
+ readonly calls: 16;
65
+ readonly wallTimeMs: 3600000;
66
+ readonly attempts: 4;
67
+ };
68
+ readonly maximumAttempts: 8;
69
+ };
70
+ readonly oracle: {
71
+ readonly capabilities: readonly ["author_oracle", "generate_controls", "evaluate_oracle", "repair_oracle"];
72
+ readonly downstream: {
73
+ readonly calls: 8;
74
+ readonly wallTimeMs: 1800000;
75
+ readonly attempts: 2;
76
+ };
77
+ readonly maximumAttempts: 12;
78
+ };
79
+ readonly admission: {
80
+ readonly capabilities: readonly ["freeze_candidate", "run_held_out_challenge", "request_admission"];
81
+ readonly downstream: {
82
+ readonly calls: 0;
83
+ readonly wallTimeMs: 0;
84
+ readonly attempts: 0;
85
+ };
86
+ readonly maximumAttempts: 6;
87
+ };
88
+ };
89
+ export type EvalCapabilityScientificPhaseV1 = keyof typeof EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1;
90
+ export declare const evalCapabilityScientificPhaseV1: (capability: AgenticCapabilityNameV1) => EvalCapabilityScientificPhaseV1 | null;
@@ -0,0 +1,107 @@
1
+ /**
2
+ * A scheduling slice that only advances verified durable work is not another
3
+ * scientific attempt. Charging it makes sufficiently slow but productive
4
+ * cases impossible, regardless of their remaining campaign budget.
5
+ * New decisions, resets, legacy/unbound receipts, and unchanged work still
6
+ * consume attempts. Calls, spend, wall time and the no-progress circuit are
7
+ * independent and are never refunded.
8
+ */
9
+ export const evalCapabilityScientificAttemptCostV1 = (receipt) => {
10
+ const previous = receipt.previousFingerprint;
11
+ const next = receipt.resultingFingerprint;
12
+ return previous !== null &&
13
+ receipt.inputRevision !== null &&
14
+ receipt.inputRevision === receipt.outputRevision &&
15
+ receipt.sameFingerprintCount === 0 &&
16
+ receipt.gatesCompleted.length === 0 &&
17
+ receipt.gatesInvalidated.length === 0 &&
18
+ next.continuationProgressDigest != null &&
19
+ next.continuationProgressDigest !== (previous.continuationProgressDigest ?? null) &&
20
+ next.completedGateDigest === previous.completedGateDigest &&
21
+ next.environmentDigest === previous.environmentDigest
22
+ ? 0
23
+ : 1;
24
+ };
25
+ /**
26
+ * No-op calls to other tools cannot reset a capability's stall counter. A
27
+ * genuine accepted-state change does reset it, even if a later repair returns
28
+ * to the same state. Worker admission and Cloud receipt validation share this
29
+ * rule; the global previous-fingerprint chain is verified independently.
30
+ * History must be ordered oldest first and scoped to one candidate revision.
31
+ */
32
+ export const evalCapabilityScientificProgressPredecessorV1 = (input) => {
33
+ for (let index = input.history.length - 1; index >= 0; index -= 1) {
34
+ const receipt = input.history[index];
35
+ const fingerprint = receipt.resultingFingerprint;
36
+ if (fingerprint.acceptedEvidenceDigest !== input.fingerprint.acceptedEvidenceDigest ||
37
+ fingerprint.completedGateDigest !== input.fingerprint.completedGateDigest ||
38
+ fingerprint.environmentDigest !== input.fingerprint.environmentDigest ||
39
+ (fingerprint.continuationProgressDigest ?? null) !==
40
+ (input.fingerprint.continuationProgressDigest ?? null))
41
+ return undefined;
42
+ if (receipt.capability === input.capability)
43
+ return receipt;
44
+ }
45
+ return undefined;
46
+ };
47
+ /**
48
+ * One scientific capability is allowed enough time to finish a complete
49
+ * repository-grounded role chain (analysis, visible authoring, and private
50
+ * review) under normal provider tail latency. Cloud admission, the sandbox
51
+ * executor, and the native tool import this dependency-light policy module so
52
+ * loading an OpenCode tool never initializes the server-side foundry runtime.
53
+ */
54
+ export const EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_MODEL_CALLS_V1 = 12;
55
+ export const EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_WALL_TIME_MS_V1 = 10 * 60_000;
56
+ /**
57
+ * Versioned capability slices are derived from the maximum number of native
58
+ * foundry turns that must remain contiguous to preserve tool state. Most
59
+ * capabilities fit one base slice. Oracle authoring and repair own longer
60
+ * tool-using conversations and therefore receive multiple contiguous slices
61
+ * instead of repeatedly restarting after the base ceiling.
62
+ */
63
+ export const EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1 = {
64
+ author_oracle: { modelCallSlices: 3, wallTimeSlices: 3 },
65
+ repair_oracle: { modelCallSlices: 2, wallTimeSlices: 2 }
66
+ };
67
+ export const evalCapabilityScientificMaximumModelCallsV1 = (capability) => EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_MODEL_CALLS_V1 *
68
+ (capability in EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1
69
+ ? EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1[capability].modelCallSlices
70
+ : 1);
71
+ export const evalCapabilityScientificMaximumWallTimeMsV1 = (capability) => EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_WALL_TIME_MS_V1 *
72
+ (capability in EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1
73
+ ? EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1[capability].wallTimeSlices
74
+ : 1);
75
+ export const EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1 = {
76
+ specification: {
77
+ capabilities: ["draft_specification"],
78
+ // This is the minimum protected envelope for environment qualification,
79
+ // oracle/control construction, held-out challenge, and admission. It is not
80
+ // a per-turn timeout. A smaller campaign can start useful work but cannot
81
+ // complete the mandatory scientific path.
82
+ downstream: { calls: 24, wallTimeMs: 4_500_000, attempts: 6 },
83
+ maximumAttempts: 8
84
+ },
85
+ environment: {
86
+ capabilities: ["plan_environment", "probe_fixture", "qualify_reference"],
87
+ downstream: { calls: 16, wallTimeMs: 3_600_000, attempts: 4 },
88
+ maximumAttempts: 8
89
+ },
90
+ oracle: {
91
+ capabilities: ["author_oracle", "generate_controls", "evaluate_oracle", "repair_oracle"],
92
+ downstream: { calls: 8, wallTimeMs: 1_800_000, attempts: 2 },
93
+ maximumAttempts: 12
94
+ },
95
+ admission: {
96
+ capabilities: ["freeze_candidate", "run_held_out_challenge", "request_admission"],
97
+ downstream: { calls: 0, wallTimeMs: 0, attempts: 0 },
98
+ maximumAttempts: 6
99
+ }
100
+ };
101
+ export const evalCapabilityScientificPhaseV1 = (capability) => {
102
+ for (const [phase, policy] of Object.entries(EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1)) {
103
+ if (policy.capabilities.includes(capability))
104
+ return phase;
105
+ }
106
+ return null;
107
+ };
@@ -1,4 +1,6 @@
1
1
  import { Schema } from "effect";
2
+ export declare const EvalRunFailurePhase: Schema.Literals<readonly ["target", "classifier", "suite-inspection", "comparison", "activation", "accounting", "completion"]>;
3
+ export type EvalRunFailurePhase = typeof EvalRunFailurePhase.Type;
2
4
  export declare const EvalNamedEvent: Schema.Union<readonly [Schema.Struct<{
3
5
  readonly name: Schema.Literal<"plan">;
4
6
  readonly at: Schema.String;
@@ -18,6 +20,30 @@ export declare const EvalNamedEvent: Schema.Union<readonly [Schema.Struct<{
18
20
  readonly caseId: Schema.String;
19
21
  readonly sequence: Schema.Finite;
20
22
  readonly name: Schema.Literal<"judge#">;
23
+ }>, Schema.Struct<{
24
+ readonly name: Schema.Literal<"failure-detail">;
25
+ readonly at: Schema.String;
26
+ readonly runId: Schema.String;
27
+ readonly phase: Schema.Literals<readonly ["target", "classifier", "suite-inspection", "comparison", "activation", "accounting", "completion"]>;
28
+ readonly reason: Schema.String;
29
+ readonly retryable: Schema.Boolean;
30
+ readonly callId: Schema.optionalKey<Schema.String>;
31
+ readonly reservationId: Schema.optionalKey<Schema.String>;
32
+ readonly accounting: Schema.Struct<{
33
+ readonly expectedCalls: Schema.Finite;
34
+ readonly observedCalls: Schema.Finite;
35
+ readonly knownInputTokens: Schema.Finite;
36
+ readonly knownOutputTokens: Schema.Finite;
37
+ readonly unknownTokenMeasurements: Schema.Finite;
38
+ readonly knownPricedSubtotalUsd: Schema.Finite;
39
+ readonly unpricedCalls: Schema.Finite;
40
+ readonly subscriptionUnitCalls: Schema.Finite;
41
+ }>;
42
+ readonly cleanup: Schema.Struct<{
43
+ readonly sessionOpened: Schema.Boolean;
44
+ readonly sessionClosed: Schema.Boolean;
45
+ readonly detail: Schema.optionalKey<Schema.String>;
46
+ }>;
21
47
  }>, Schema.Struct<{
22
48
  readonly at: Schema.String;
23
49
  readonly operationId: Schema.String;
@@ -51,6 +77,30 @@ export declare const EvalNamedEventLog: Schema.$Array<Schema.Union<readonly [Sch
51
77
  readonly caseId: Schema.String;
52
78
  readonly sequence: Schema.Finite;
53
79
  readonly name: Schema.Literal<"judge#">;
80
+ }>, Schema.Struct<{
81
+ readonly name: Schema.Literal<"failure-detail">;
82
+ readonly at: Schema.String;
83
+ readonly runId: Schema.String;
84
+ readonly phase: Schema.Literals<readonly ["target", "classifier", "suite-inspection", "comparison", "activation", "accounting", "completion"]>;
85
+ readonly reason: Schema.String;
86
+ readonly retryable: Schema.Boolean;
87
+ readonly callId: Schema.optionalKey<Schema.String>;
88
+ readonly reservationId: Schema.optionalKey<Schema.String>;
89
+ readonly accounting: Schema.Struct<{
90
+ readonly expectedCalls: Schema.Finite;
91
+ readonly observedCalls: Schema.Finite;
92
+ readonly knownInputTokens: Schema.Finite;
93
+ readonly knownOutputTokens: Schema.Finite;
94
+ readonly unknownTokenMeasurements: Schema.Finite;
95
+ readonly knownPricedSubtotalUsd: Schema.Finite;
96
+ readonly unpricedCalls: Schema.Finite;
97
+ readonly subscriptionUnitCalls: Schema.Finite;
98
+ }>;
99
+ readonly cleanup: Schema.Struct<{
100
+ readonly sessionOpened: Schema.Boolean;
101
+ readonly sessionClosed: Schema.Boolean;
102
+ readonly detail: Schema.optionalKey<Schema.String>;
103
+ }>;
54
104
  }>, Schema.Struct<{
55
105
  readonly at: Schema.String;
56
106
  readonly operationId: Schema.String;
@@ -86,4 +136,5 @@ export type EvalEventLogPresentation = {
86
136
  export declare function presentEvalEventLog(log: EvalNamedEventLog, options?: {
87
137
  readonly failHoles?: boolean;
88
138
  }): EvalEventLogPresentation;
139
+ export declare function assertEvalRunCanFail(log: EvalNamedEventLog): void;
89
140
  export declare function assertEvalRunCanComplete(log: EvalNamedEventLog): void;
@@ -1,6 +1,10 @@
1
1
  import { Schema } from "effect";
2
2
  const NonNegativeInteger = Schema.Finite.pipe(Schema.check(Schema.makeFilter((value) => value >= 0 && Number.isInteger(value) ? undefined : "value must be a non-negative integer")));
3
3
  const PositiveInteger = NonNegativeInteger.pipe(Schema.check(Schema.makeFilter((value) => value >= 1 ? undefined : "value must be a positive integer")));
4
+ const NonNegativeFinite = Schema.Finite.pipe(Schema.check(Schema.makeFilter((value) => value >= 0 ? undefined : "value must be a non-negative finite number")));
5
+ const BoundedFailureReason = Schema.String.pipe(Schema.check(Schema.makeFilter((value) => value.length >= 1 && value.length <= 2_000
6
+ ? undefined
7
+ : "failure reason must contain between 1 and 2000 characters")));
4
8
  const RunPlanEvent = Schema.Struct({
5
9
  name: Schema.Literal("plan"),
6
10
  at: Schema.String,
@@ -22,6 +26,40 @@ const RunJudgeEvent = Schema.Struct({
22
26
  name: Schema.Literal("judge#"),
23
27
  ...RunCallEventFields
24
28
  });
29
+ export const EvalRunFailurePhase = Schema.Literals([
30
+ "target",
31
+ "classifier",
32
+ "suite-inspection",
33
+ "comparison",
34
+ "activation",
35
+ "accounting",
36
+ "completion"
37
+ ]);
38
+ const RunFailureDetailEvent = Schema.Struct({
39
+ name: Schema.Literal("failure-detail"),
40
+ at: Schema.String,
41
+ runId: Schema.String,
42
+ phase: EvalRunFailurePhase,
43
+ reason: BoundedFailureReason,
44
+ retryable: Schema.Boolean,
45
+ callId: Schema.optionalKey(Schema.String),
46
+ reservationId: Schema.optionalKey(Schema.String),
47
+ accounting: Schema.Struct({
48
+ expectedCalls: NonNegativeInteger,
49
+ observedCalls: NonNegativeInteger,
50
+ knownInputTokens: NonNegativeInteger,
51
+ knownOutputTokens: NonNegativeInteger,
52
+ unknownTokenMeasurements: NonNegativeInteger,
53
+ knownPricedSubtotalUsd: NonNegativeFinite,
54
+ unpricedCalls: NonNegativeInteger,
55
+ subscriptionUnitCalls: NonNegativeInteger
56
+ }),
57
+ cleanup: Schema.Struct({
58
+ sessionOpened: Schema.Boolean,
59
+ sessionClosed: Schema.Boolean,
60
+ detail: Schema.optionalKey(Schema.String)
61
+ })
62
+ });
25
63
  const AuthoringEventFields = {
26
64
  at: Schema.String,
27
65
  operationId: Schema.String
@@ -42,6 +80,7 @@ export const EvalNamedEvent = Schema.Union([
42
80
  RunPlanEvent,
43
81
  RunCandidateEvent,
44
82
  RunJudgeEvent,
83
+ RunFailureDetailEvent,
45
84
  AuthoringBasisEvent,
46
85
  AuthoringDimensionsEvent,
47
86
  AuthoringActivationEvent
@@ -125,6 +164,9 @@ export function presentEvalEventLog(log, options = {}) {
125
164
  const cursors = resumeEvalCaseCursors(log);
126
165
  const failHoles = options.failHoles ?? true;
127
166
  const lastEvent = log[log.length - 1];
167
+ const failureLine = lastEvent?.name === "failure-detail"
168
+ ? `run fail ${lastEvent.phase}: ${lastEvent.reason}`
169
+ : undefined;
128
170
  const authoringLine = lastEvent?.name === "basis" ||
129
171
  lastEvent?.name === "dimensions" ||
130
172
  lastEvent?.name === "activation"
@@ -137,23 +179,40 @@ export function presentEvalEventLog(log, options = {}) {
137
179
  : `fail ${label} missing judge#${String(cursor.hole.sequence)}`}`;
138
180
  });
139
181
  return {
140
- failed: failHoles && cursors.some((cursor) => cursor.hole !== undefined),
141
- lines: cursors.length === 0
142
- ? authoringLine === undefined
143
- ? plan === undefined
144
- ? []
145
- : [`run ${evalEventLabel(plan)}`]
146
- : [authoringLine]
147
- : authoringLine === undefined
148
- ? cursorLines
149
- : [...cursorLines, authoringLine]
182
+ failed: failureLine !== undefined ||
183
+ (failHoles && cursors.some((cursor) => cursor.hole !== undefined)),
184
+ lines: failureLine !== undefined
185
+ ? [...cursorLines, failureLine]
186
+ : cursors.length === 0
187
+ ? authoringLine === undefined
188
+ ? plan === undefined
189
+ ? []
190
+ : [`run ${evalEventLabel(plan)}`]
191
+ : [authoringLine]
192
+ : authoringLine === undefined
193
+ ? cursorLines
194
+ : [...cursorLines, authoringLine]
150
195
  };
151
196
  }
197
+ export function assertEvalRunCanFail(log) {
198
+ const plan = lastPlanEvent(log);
199
+ if (plan === undefined) {
200
+ throw new Error("failed eval run cannot terminate without its plan event");
201
+ }
202
+ const terminal = log[log.length - 1];
203
+ if (terminal?.name !== "failure-detail" || terminal.runId !== plan.runId) {
204
+ throw new Error("failed eval run must end with a matching terminal failure-detail event");
205
+ }
206
+ }
152
207
  export function assertEvalRunCanComplete(log) {
153
208
  const plan = lastPlanEvent(log);
154
209
  if (plan === undefined || plan.plannedCalls < 1) {
155
210
  throw new Error("eval run cannot complete without a plan containing at least one call");
156
211
  }
212
+ const failure = log.find((event) => event.name === "failure-detail" && event.runId === plan.runId);
213
+ if (failure !== undefined) {
214
+ throw new Error(`eval run cannot complete after terminal failure in phase ${JSON.stringify(failure.phase)}`);
215
+ }
157
216
  const hole = resumeEvalCaseCursors(log).find((cursor) => cursor.hole !== undefined);
158
217
  if (hole?.hole !== undefined) {
159
218
  throw new Error(`eval case ${JSON.stringify(hole.dimensionId)}/${JSON.stringify(hole.caseId)} cannot complete after ${evalEventLabel(hole.hole)} without judge#${String(hole.hole.sequence)}`);
@@ -0,0 +1,18 @@
1
+ /**
2
+ * Provisional transport limits for compiler-backed evaluation authoring.
3
+ *
4
+ * These values bound model input/output and persisted candidate context; they
5
+ * are not semantic-quality thresholds. Live preflight and later empirical
6
+ * calibration must determine whether they produce coherent, useful cases.
7
+ */
8
+ export declare const EVAL_AUTHORING_EVIDENCE_BLOCK_BYTES = 3000;
9
+ export declare const EVAL_AUTHORING_CASE_EVIDENCE_BYTES = 12000;
10
+ /**
11
+ * Maximum compiler blocks exposed as one mechanically safe author selection.
12
+ *
13
+ * A block may consume the complete block-byte allowance, so this cardinality
14
+ * guarantees that selecting every exposed block cannot exceed the case context
15
+ * allowance. The exact byte check remains authoritative because smaller blocks
16
+ * do not make every same-cardinality combination equivalent.
17
+ */
18
+ export declare const EVAL_AUTHORING_CASE_EVIDENCE_BLOCKS: number;
@@ -0,0 +1,19 @@
1
+ /**
2
+ * Provisional transport limits for compiler-backed evaluation authoring.
3
+ *
4
+ * These values bound model input/output and persisted candidate context; they
5
+ * are not semantic-quality thresholds. Live preflight and later empirical
6
+ * calibration must determine whether they produce coherent, useful cases.
7
+ */
8
+ export const EVAL_AUTHORING_EVIDENCE_BLOCK_BYTES = 3_000;
9
+ export const EVAL_AUTHORING_CASE_EVIDENCE_BYTES = 12_000;
10
+ /**
11
+ * Maximum compiler blocks exposed as one mechanically safe author selection.
12
+ *
13
+ * A block may consume the complete block-byte allowance, so this cardinality
14
+ * guarantees that selecting every exposed block cannot exceed the case context
15
+ * allowance. The exact byte check remains authoritative because smaller blocks
16
+ * do not make every same-cardinality combination equivalent.
17
+ */
18
+ export const EVAL_AUTHORING_CASE_EVIDENCE_BLOCKS = Math.floor(EVAL_AUTHORING_CASE_EVIDENCE_BYTES /
19
+ EVAL_AUTHORING_EVIDENCE_BLOCK_BYTES);
@@ -0,0 +1,22 @@
1
+ export type EvalAuthoringValidationCategory = "schema-constraint" | "schema-decode" | "case-shape" | "criterion-topology" | "control-topology" | "evidence-selection";
2
+ export declare class EvalAuthoringValidationIssue extends Error {
3
+ readonly category: EvalAuthoringValidationCategory;
4
+ readonly path: string;
5
+ constructor(input: {
6
+ readonly category: EvalAuthoringValidationCategory;
7
+ readonly path: string;
8
+ readonly detail: string;
9
+ readonly cause?: unknown;
10
+ });
11
+ }
12
+ export declare const evalAuthoringValidationDiagnostic: (cause: unknown) => {
13
+ readonly category: EvalAuthoringValidationCategory;
14
+ readonly path: string;
15
+ readonly detail: string;
16
+ };
17
+ export declare const prefixEvalAuthoringValidationIssue: (cause: unknown, prefix: string) => EvalAuthoringValidationIssue;
18
+ export declare const renderEvalAuthoringValidationFailure: (input: {
19
+ readonly scope: string;
20
+ readonly cause: unknown;
21
+ readonly callId?: string;
22
+ }) => string;