@velum-labs/routekit-eval-setup 1.3.2 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (342) hide show
  1. package/dist/adapters/authoring-responses-request.d.ts +5 -0
  2. package/dist/adapters/authoring-responses-request.js +38 -0
  3. package/dist/adapters/evaluation-evidence-freshness.d.ts +7 -0
  4. package/dist/adapters/evaluation-evidence-freshness.js +76 -0
  5. package/dist/adapters/git-task-history.d.ts +67 -0
  6. package/dist/adapters/git-task-history.js +171 -0
  7. package/dist/adapters/integrated-repository-history.d.ts +21 -0
  8. package/dist/adapters/integrated-repository-history.js +175 -0
  9. package/dist/adapters/repository-command-diagnostic.d.ts +8 -0
  10. package/dist/adapters/repository-command-diagnostic.js +46 -0
  11. package/dist/adapters/repository-command-evidence.d.ts +13 -0
  12. package/dist/adapters/repository-command-evidence.js +102 -0
  13. package/dist/adapters/repository-command-runner.d.ts +238 -0
  14. package/dist/adapters/repository-command-runner.js +1483 -0
  15. package/dist/adapters/repository-import-context.d.ts +47 -0
  16. package/dist/adapters/repository-import-context.js +469 -0
  17. package/dist/adapters/repository-node-test-reporter.d.ts +3 -0
  18. package/dist/adapters/repository-node-test-reporter.js +27 -0
  19. package/dist/adapters/repository-review-evidence.d.ts +39 -0
  20. package/dist/adapters/repository-review-evidence.js +632 -0
  21. package/dist/adapters/repository-seed-selection.d.ts +7 -0
  22. package/dist/adapters/repository-seed-selection.js +79 -0
  23. package/dist/adapters/repository-solution-edits.d.ts +49 -0
  24. package/dist/adapters/repository-solution-edits.js +136 -0
  25. package/dist/adapters/repository-vitest-phase-adapter.d.ts +8 -0
  26. package/dist/adapters/repository-vitest-phase-adapter.js +310 -0
  27. package/dist/adapters/repository-vitest-reporter.d.ts +24 -0
  28. package/dist/adapters/repository-vitest-reporter.js +314 -0
  29. package/dist/adapters/strict-authoring-schema.d.ts +5 -0
  30. package/dist/adapters/strict-authoring-schema.js +158 -0
  31. package/dist/adapters/test-discovery.d.ts +30 -0
  32. package/dist/adapters/test-discovery.js +124 -0
  33. package/dist/adapters/typescript-repository-index.d.ts +51 -0
  34. package/dist/adapters/typescript-repository-index.js +226 -0
  35. package/dist/agentic-capabilities-protocol.d.ts +1373 -0
  36. package/dist/agentic-capabilities-protocol.js +786 -0
  37. package/dist/case-checkpoint-store.d.ts +29 -0
  38. package/dist/case-checkpoint-store.js +133 -0
  39. package/dist/case-pipeline-protocol-v2.d.ts +184 -0
  40. package/dist/case-pipeline-protocol-v2.js +193 -0
  41. package/dist/case-pipeline-protocol.d.ts +2626 -0
  42. package/dist/case-pipeline-protocol.js +371 -0
  43. package/dist/effect-api.d.ts +74 -10
  44. package/dist/effect-api.js +56 -6
  45. package/dist/errors.d.ts +31 -0
  46. package/dist/errors.js +10 -0
  47. package/dist/eval-capability-execution-envelope.d.ts +64 -0
  48. package/dist/eval-capability-execution-envelope.js +98 -0
  49. package/dist/eval-capability-policy.d.ts +90 -0
  50. package/dist/eval-capability-policy.js +107 -0
  51. package/dist/eval-event-log.d.ts +140 -0
  52. package/dist/eval-event-log.js +220 -0
  53. package/dist/evaluation-authoring-policy.d.ts +18 -0
  54. package/dist/evaluation-authoring-policy.js +19 -0
  55. package/dist/evaluation-authoring-validation.d.ts +22 -0
  56. package/dist/evaluation-authoring-validation.js +72 -0
  57. package/dist/evaluation-evidence.d.ts +20 -0
  58. package/dist/evaluation-evidence.js +319 -0
  59. package/dist/evaluation-grader-calibration-protocol.d.ts +108 -0
  60. package/dist/evaluation-grader-calibration-protocol.js +80 -0
  61. package/dist/evaluation-grader-calibration.d.ts +18 -0
  62. package/dist/evaluation-grader-calibration.js +334 -0
  63. package/dist/evaluation-grading-policy.d.ts +24 -0
  64. package/dist/evaluation-grading-policy.js +54 -0
  65. package/dist/evaluation-proposal-policy.d.ts +4 -0
  66. package/dist/evaluation-proposal-policy.js +91 -0
  67. package/dist/evaluation-source-retrieval.d.ts +68 -0
  68. package/dist/evaluation-source-retrieval.js +513 -0
  69. package/dist/evaluation-structure-policy.d.ts +29 -0
  70. package/dist/evaluation-structure-policy.js +138 -0
  71. package/dist/index.d.ts +124 -17
  72. package/dist/index.js +69 -11
  73. package/dist/inspection.js +2 -3
  74. package/dist/project-artifacts.d.ts +7 -2
  75. package/dist/project-artifacts.js +49 -136
  76. package/dist/project-authoring.d.ts +66 -5
  77. package/dist/project-authoring.js +783 -109
  78. package/dist/project-contracts.d.ts +419 -84
  79. package/dist/project-contracts.js +160 -52
  80. package/dist/project-store.js +2 -1
  81. package/dist/project-workflow.d.ts +5 -4
  82. package/dist/project-workflow.js +154 -35
  83. package/dist/repository-adversary-protocol.d.ts +64 -0
  84. package/dist/repository-adversary-protocol.js +105 -0
  85. package/dist/repository-behavior-protocol.d.ts +188 -0
  86. package/dist/repository-behavior-protocol.js +202 -0
  87. package/dist/repository-benchmark-protocol.d.ts +487 -0
  88. package/dist/repository-benchmark-protocol.js +96 -0
  89. package/dist/repository-execution-protocol.d.ts +150 -0
  90. package/dist/repository-execution-protocol.js +38 -0
  91. package/dist/repository-fixture-instructions.d.ts +3 -0
  92. package/dist/repository-fixture-instructions.js +91 -0
  93. package/dist/repository-fixture-protocol.d.ts +79 -0
  94. package/dist/repository-fixture-protocol.js +79 -0
  95. package/dist/repository-foundry-plan-protocol.d.ts +118 -0
  96. package/dist/repository-foundry-plan-protocol.js +296 -0
  97. package/dist/repository-foundry-progress-protocol.d.ts +52 -0
  98. package/dist/repository-foundry-progress-protocol.js +52 -0
  99. package/dist/repository-improvement-protocol.d.ts +100 -0
  100. package/dist/repository-improvement-protocol.js +106 -0
  101. package/dist/repository-language-model-protocol.d.ts +43 -0
  102. package/dist/repository-language-model-protocol.js +146 -0
  103. package/dist/repository-oracle-coverage-protocol.d.ts +18 -0
  104. package/dist/repository-oracle-coverage-protocol.js +39 -0
  105. package/dist/repository-oracle-execution-binding.d.ts +27 -0
  106. package/dist/repository-oracle-execution-binding.js +59 -0
  107. package/dist/repository-oracle-protocol.d.ts +230 -0
  108. package/dist/repository-oracle-protocol.js +156 -0
  109. package/dist/repository-oracle-scope-policy.d.ts +22 -0
  110. package/dist/repository-oracle-scope-policy.js +92 -0
  111. package/dist/repository-quality-policy.d.ts +15 -0
  112. package/dist/repository-quality-policy.js +357 -0
  113. package/dist/repository-routing-benchmark-protocol.d.ts +176 -0
  114. package/dist/repository-routing-benchmark-protocol.js +103 -0
  115. package/dist/repository-routing-model-protocol.d.ts +36 -0
  116. package/dist/repository-routing-model-protocol.js +89 -0
  117. package/dist/repository-routing-plan-protocol.d.ts +112 -0
  118. package/dist/repository-routing-plan-protocol.js +58 -0
  119. package/dist/repository-routing-quality-policy.d.ts +9 -0
  120. package/dist/repository-routing-quality-policy.js +191 -0
  121. package/dist/repository-seed-qualification-progress-protocol.d.ts +205 -0
  122. package/dist/repository-seed-qualification-progress-protocol.js +28 -0
  123. package/dist/repository-semantic-calibration-protocol.d.ts +768 -0
  124. package/dist/repository-semantic-calibration-protocol.js +276 -0
  125. package/dist/repository-semantic-calibration.d.ts +163 -0
  126. package/dist/repository-semantic-calibration.js +581 -0
  127. package/dist/repository-specification-contract-facts-protocol.d.ts +224 -0
  128. package/dist/repository-specification-contract-facts-protocol.js +276 -0
  129. package/dist/repository-specification-critique-protocol.d.ts +189 -0
  130. package/dist/repository-specification-critique-protocol.js +103 -0
  131. package/dist/repository-task-family-protocol.d.ts +24 -0
  132. package/dist/repository-task-family-protocol.js +37 -0
  133. package/dist/repository-task-seed-protocol.d.ts +384 -0
  134. package/dist/repository-task-seed-protocol.js +236 -0
  135. package/dist/repository-trajectory-protocol.d.ts +20 -0
  136. package/dist/repository-trajectory-protocol.js +42 -0
  137. package/dist/service.js +1 -1
  138. package/dist/services/adversary/service.d.ts +64 -0
  139. package/dist/services/adversary/service.js +330 -0
  140. package/dist/services/benchmark-compiler/service.d.ts +450 -0
  141. package/dist/services/benchmark-compiler/service.js +9 -0
  142. package/dist/services/budgeted-model/service.d.ts +118 -0
  143. package/dist/services/budgeted-model/service.js +460 -0
  144. package/dist/services/case-authoring/service.d.ts +163 -0
  145. package/dist/services/case-authoring/service.js +1456 -0
  146. package/dist/services/case-finalization/service.d.ts +283 -0
  147. package/dist/services/case-finalization/service.js +370 -0
  148. package/dist/services/case-generation/service.d.ts +619 -0
  149. package/dist/services/case-generation/service.js +2628 -0
  150. package/dist/services/case-pipeline/service.d.ts +31 -0
  151. package/dist/services/case-pipeline/service.js +485 -0
  152. package/dist/services/case-pipeline-v2/service.d.ts +70 -0
  153. package/dist/services/case-pipeline-v2/service.js +477 -0
  154. package/dist/services/command-observability/service.d.ts +13 -0
  155. package/dist/services/command-observability/service.js +3 -0
  156. package/dist/services/dimension-labeling/service.d.ts +77 -0
  157. package/dist/services/dimension-labeling/service.js +188 -0
  158. package/dist/services/eval-candidate/service.d.ts +208 -0
  159. package/dist/services/eval-candidate/service.js +64 -0
  160. package/dist/services/eval-capabilities/service.d.ts +183 -0
  161. package/dist/services/eval-capabilities/service.js +1433 -0
  162. package/dist/services/eval-environment/service.d.ts +173 -0
  163. package/dist/services/eval-environment/service.js +127 -0
  164. package/dist/services/evidence-reconstruction/service.d.ts +36 -0
  165. package/dist/services/evidence-reconstruction/service.js +145 -0
  166. package/dist/services/fixture-builder/service.d.ts +62 -0
  167. package/dist/services/fixture-builder/service.js +36 -0
  168. package/dist/services/fixture-validation/service.d.ts +75 -0
  169. package/dist/services/fixture-validation/service.js +295 -0
  170. package/dist/services/foundry/service.d.ts +831 -0
  171. package/dist/services/foundry/service.js +442 -0
  172. package/dist/services/foundry-progress/service.d.ts +62 -0
  173. package/dist/services/foundry-progress/service.js +149 -0
  174. package/dist/services/foundry-v2/service.d.ts +54 -0
  175. package/dist/services/foundry-v2/service.js +28 -0
  176. package/dist/services/grounded-authoring/service.d.ts +126 -0
  177. package/dist/services/grounded-authoring/service.js +822 -0
  178. package/dist/services/historical-case/service.d.ts +722 -0
  179. package/dist/services/historical-case/service.js +177 -0
  180. package/dist/services/improvement-loop/service.d.ts +59 -0
  181. package/dist/services/improvement-loop/service.js +176 -0
  182. package/dist/services/language-model/service.d.ts +52 -0
  183. package/dist/services/language-model/service.js +194 -0
  184. package/dist/services/oracle-builder/service.d.ts +146 -0
  185. package/dist/services/oracle-builder/service.js +513 -0
  186. package/dist/services/oracle-coverage/service.d.ts +28 -0
  187. package/dist/services/oracle-coverage/service.js +50 -0
  188. package/dist/services/oracle-coverage-witness/service.d.ts +130 -0
  189. package/dist/services/oracle-coverage-witness/service.js +538 -0
  190. package/dist/services/pipeline-challenge/service.d.ts +551 -0
  191. package/dist/services/pipeline-challenge/service.js +427 -0
  192. package/dist/services/pipeline-controls/service.d.ts +130 -0
  193. package/dist/services/pipeline-controls/service.js +483 -0
  194. package/dist/services/pipeline-oracle/service.d.ts +8 -0
  195. package/dist/services/pipeline-oracle/service.js +256 -0
  196. package/dist/services/pipeline-seed/service.d.ts +298 -0
  197. package/dist/services/pipeline-seed/service.js +428 -0
  198. package/dist/services/pipeline-spec/service.d.ts +103 -0
  199. package/dist/services/pipeline-spec/service.js +619 -0
  200. package/dist/services/pipeline-tournament/service.d.ts +258 -0
  201. package/dist/services/pipeline-tournament/service.js +476 -0
  202. package/dist/services/quality-gate/service.d.ts +233 -0
  203. package/dist/services/quality-gate/service.js +136 -0
  204. package/dist/services/repository-bundle/service.d.ts +33 -0
  205. package/dist/services/repository-bundle/service.js +114 -0
  206. package/dist/services/repository-model/service.d.ts +105 -0
  207. package/dist/services/repository-model/service.js +250 -0
  208. package/dist/services/repository-public-artifact/service.d.ts +133 -0
  209. package/dist/services/repository-public-artifact/service.js +330 -0
  210. package/dist/services/routing-benchmark/service.d.ts +362 -0
  211. package/dist/services/routing-benchmark/service.js +96 -0
  212. package/dist/services/specification-critic/service.d.ts +92 -0
  213. package/dist/services/specification-critic/service.js +172 -0
  214. package/dist/services/task-family/service.d.ts +40 -0
  215. package/dist/services/task-family/service.js +55 -0
  216. package/dist/services/task-seed/service.d.ts +906 -0
  217. package/dist/services/task-seed/service.js +1406 -0
  218. package/dist/services/task-specification/service.d.ts +27 -0
  219. package/dist/services/task-specification/service.js +40 -0
  220. package/dist/services/trajectory-policy/service.d.ts +110 -0
  221. package/dist/services/trajectory-policy/service.js +216 -0
  222. package/dist/test/agentic-capabilities-protocol.test.d.ts +1 -0
  223. package/dist/test/agentic-capabilities-protocol.test.js +570 -0
  224. package/dist/test/agentic-capabilities.test.d.ts +1 -0
  225. package/dist/test/agentic-capabilities.test.js +1461 -0
  226. package/dist/test/agentic-environment.test.d.ts +1 -0
  227. package/dist/test/agentic-environment.test.js +213 -0
  228. package/dist/test/case-pipeline-foundation.test.d.ts +1 -0
  229. package/dist/test/case-pipeline-foundation.test.js +535 -0
  230. package/dist/test/case-pipeline-protocol-v2.test.d.ts +1 -0
  231. package/dist/test/case-pipeline-protocol-v2.test.js +124 -0
  232. package/dist/test/case-pipeline-v2.test.d.ts +1 -0
  233. package/dist/test/case-pipeline-v2.test.js +286 -0
  234. package/dist/test/case-pipeline.test.d.ts +1 -0
  235. package/dist/test/case-pipeline.test.js +851 -0
  236. package/dist/test/eval-capability-policy.test.d.ts +1 -0
  237. package/dist/test/eval-capability-policy.test.js +50 -0
  238. package/dist/test/eval-event-log.test.d.ts +1 -0
  239. package/dist/test/eval-event-log.test.js +125 -0
  240. package/dist/test/evaluation-evidence-freshness.test.d.ts +1 -0
  241. package/dist/test/evaluation-evidence-freshness.test.js +44 -0
  242. package/dist/test/evaluation-evidence.test.d.ts +1 -0
  243. package/dist/test/evaluation-evidence.test.js +230 -0
  244. package/dist/test/evaluation-grader-calibration.test.d.ts +1 -0
  245. package/dist/test/evaluation-grader-calibration.test.js +373 -0
  246. package/dist/test/evaluation-proposal-digest.test.d.ts +1 -0
  247. package/dist/test/evaluation-proposal-digest.test.js +187 -0
  248. package/dist/test/evaluation-source-retrieval.test.d.ts +1 -0
  249. package/dist/test/evaluation-source-retrieval.test.js +237 -0
  250. package/dist/test/evaluation-structure-policy.test.d.ts +1 -0
  251. package/dist/test/evaluation-structure-policy.test.js +196 -0
  252. package/dist/test/fixtures/repository-resource-panel.d.ts +39 -0
  253. package/dist/test/fixtures/repository-resource-panel.js +111 -0
  254. package/dist/test/fixtures/vitest-boundary-panel.d.ts +84 -0
  255. package/dist/test/fixtures/vitest-boundary-panel.js +120 -0
  256. package/dist/test/fixtures/vitest-phase-panel.d.ts +135 -0
  257. package/dist/test/fixtures/vitest-phase-panel.js +213 -0
  258. package/dist/test/fixtures/vitest-reporter-results.d.ts +76 -0
  259. package/dist/test/fixtures/vitest-reporter-results.js +94 -0
  260. package/dist/test/grounded-authoring.test.d.ts +1 -0
  261. package/dist/test/grounded-authoring.test.js +565 -0
  262. package/dist/test/integrated-repository-history.test.d.ts +1 -0
  263. package/dist/test/integrated-repository-history.test.js +227 -0
  264. package/dist/test/project-authoring.test.js +593 -43
  265. package/dist/test/project-workflow.test.js +419 -40
  266. package/dist/test/repository-authoring-artifacts.test.d.ts +1 -0
  267. package/dist/test/repository-authoring-artifacts.test.js +185 -0
  268. package/dist/test/repository-bundle.test.d.ts +1 -0
  269. package/dist/test/repository-bundle.test.js +52 -0
  270. package/dist/test/repository-case-generation.test.d.ts +1 -0
  271. package/dist/test/repository-case-generation.test.js +3465 -0
  272. package/dist/test/repository-command-diagnostic.test.d.ts +1 -0
  273. package/dist/test/repository-command-diagnostic.test.js +55 -0
  274. package/dist/test/repository-command-signals.test.d.ts +1 -0
  275. package/dist/test/repository-command-signals.test.js +124 -0
  276. package/dist/test/repository-fixture-scope-coverage.test.d.ts +1 -0
  277. package/dist/test/repository-fixture-scope-coverage.test.js +127 -0
  278. package/dist/test/repository-fixture-validation.test.d.ts +1 -0
  279. package/dist/test/repository-fixture-validation.test.js +362 -0
  280. package/dist/test/repository-foundry-progress.test.d.ts +1 -0
  281. package/dist/test/repository-foundry-progress.test.js +110 -0
  282. package/dist/test/repository-foundry-quality.test.d.ts +1 -0
  283. package/dist/test/repository-foundry-quality.test.js +1138 -0
  284. package/dist/test/repository-import-context.test.d.ts +1 -0
  285. package/dist/test/repository-import-context.test.js +354 -0
  286. package/dist/test/repository-model-authoring.test.d.ts +1 -0
  287. package/dist/test/repository-model-authoring.test.js +544 -0
  288. package/dist/test/repository-model.test.d.ts +1 -0
  289. package/dist/test/repository-model.test.js +2195 -0
  290. package/dist/test/repository-node-test-reporter.test.d.ts +1 -0
  291. package/dist/test/repository-node-test-reporter.test.js +104 -0
  292. package/dist/test/repository-oracle-concurrency.test.d.ts +1 -0
  293. package/dist/test/repository-oracle-concurrency.test.js +542 -0
  294. package/dist/test/repository-oracle-coverage-witness.test.d.ts +1 -0
  295. package/dist/test/repository-oracle-coverage-witness.test.js +511 -0
  296. package/dist/test/repository-oracle-coverage.test.d.ts +1 -0
  297. package/dist/test/repository-oracle-coverage.test.js +168 -0
  298. package/dist/test/repository-oracle-evidence.test.d.ts +1 -0
  299. package/dist/test/repository-oracle-evidence.test.js +185 -0
  300. package/dist/test/repository-oracle-plan.test.d.ts +1 -0
  301. package/dist/test/repository-oracle-plan.test.js +176 -0
  302. package/dist/test/repository-overlay-isolation.test.d.ts +1 -0
  303. package/dist/test/repository-overlay-isolation.test.js +85 -0
  304. package/dist/test/repository-preparation-cache.test.d.ts +1 -0
  305. package/dist/test/repository-preparation-cache.test.js +414 -0
  306. package/dist/test/repository-public-artifact.test.d.ts +1 -0
  307. package/dist/test/repository-public-artifact.test.js +273 -0
  308. package/dist/test/repository-qualification-diagnostics.test.d.ts +1 -0
  309. package/dist/test/repository-qualification-diagnostics.test.js +524 -0
  310. package/dist/test/repository-reference-authoring.test.d.ts +1 -0
  311. package/dist/test/repository-reference-authoring.test.js +1633 -0
  312. package/dist/test/repository-review-evidence-v2.test.d.ts +1 -0
  313. package/dist/test/repository-review-evidence-v2.test.js +183 -0
  314. package/dist/test/repository-review-evidence.test.d.ts +1 -0
  315. package/dist/test/repository-review-evidence.test.js +124 -0
  316. package/dist/test/repository-seed-exclusions.test.d.ts +1 -0
  317. package/dist/test/repository-seed-exclusions.test.js +96 -0
  318. package/dist/test/repository-seed-selection.test.d.ts +1 -0
  319. package/dist/test/repository-seed-selection.test.js +504 -0
  320. package/dist/test/repository-semantic-calibration.test.d.ts +1 -0
  321. package/dist/test/repository-semantic-calibration.test.js +688 -0
  322. package/dist/test/repository-solution-edits.test.d.ts +1 -0
  323. package/dist/test/repository-solution-edits.test.js +377 -0
  324. package/dist/test/repository-specification-budget.test.d.ts +1 -0
  325. package/dist/test/repository-specification-budget.test.js +171 -0
  326. package/dist/test/repository-specification-contract-checkpoint.test.d.ts +1 -0
  327. package/dist/test/repository-specification-contract-checkpoint.test.js +228 -0
  328. package/dist/test/repository-specification-contract-facts.test.d.ts +1 -0
  329. package/dist/test/repository-specification-contract-facts.test.js +177 -0
  330. package/dist/test/repository-trajectory-authoring.test.d.ts +1 -0
  331. package/dist/test/repository-trajectory-authoring.test.js +176 -0
  332. package/dist/test/repository-valid-control-plan.test.d.ts +1 -0
  333. package/dist/test/repository-valid-control-plan.test.js +45 -0
  334. package/dist/test/repository-vitest-phase.test.d.ts +1 -0
  335. package/dist/test/repository-vitest-phase.test.js +848 -0
  336. package/dist/test/repository-vitest-reporter.test.d.ts +1 -0
  337. package/dist/test/repository-vitest-reporter.test.js +158 -0
  338. package/dist/test/repository-workspace-build.test.d.ts +1 -0
  339. package/dist/test/repository-workspace-build.test.js +160 -0
  340. package/dist/test/strict-authoring-schema.test.d.ts +1 -0
  341. package/dist/test/strict-authoring-schema.test.js +169 -0
  342. package/package.json +48 -6
@@ -0,0 +1,334 @@
1
+ import { createHash } from "node:crypto";
2
+ import { Schema } from "effect";
3
+ import { EVAL_GRADER_JUDGE_PROTOCOL_VERSION, EVAL_GRADER_MINIMUM_SCORE, EVAL_GRADER_SYSTEM_PROMPT, renderEvaluationCandidatePrompt } from "./evaluation-grading-policy.js";
4
+ import { EVAL_GRADER_CALIBRATION_POLICY_VERSION, EvalGraderCalibrationObservation as CalibrationObservationSchema, EvalGraderCalibrationPlan as CalibrationPlanSchema, EvalGraderCalibrationReport as CalibrationReportSchema } from "./evaluation-grader-calibration-protocol.js";
5
+ import { assertEvaluationStructure, candidateCaseProjection, hiddenOracleProjection } from "./evaluation-structure-policy.js";
6
+ import { assertEvaluationProposal } from "./evaluation-proposal-policy.js";
7
+ export { EVAL_GRADER_CALIBRATION_POLICY_VERSION, EvalGraderCalibrationControl, EvalGraderCalibrationObservation, EvalGraderCalibrationPlan, EvalGraderCalibrationReport, EvalGraderCalibrationResult } from "./evaluation-grader-calibration-protocol.js";
8
+ const sha256 = (value) => createHash("sha256").update(value).digest("hex");
9
+ const stableCompare = (left, right) => left < right ? -1 : left > right ? 1 : 0;
10
+ const controlSemantic = (control) => ({
11
+ suiteKind: control.suiteKind,
12
+ suiteId: control.suiteId,
13
+ caseId: control.caseId,
14
+ authoredControlId: control.authoredControlId,
15
+ kind: control.kind,
16
+ targetedCriterionIds: control.targetedCriterionIds,
17
+ expectedOutcome: control.expectedOutcome,
18
+ prompt: control.prompt,
19
+ criteria: control.criteria,
20
+ response: control.response
21
+ });
22
+ const compileControl = (control) => {
23
+ const controlDigest = sha256(JSON.stringify(controlSemantic(control)));
24
+ return {
25
+ ...control,
26
+ id: `cal_${controlDigest}`,
27
+ controlDigest
28
+ };
29
+ };
30
+ const controlsForCase = (testCase, suiteKind, suiteId) => {
31
+ const candidate = candidateCaseProjection(testCase);
32
+ const oracle = hiddenOracleProjection(testCase);
33
+ const common = {
34
+ suiteKind,
35
+ suiteId,
36
+ caseId: testCase.id,
37
+ prompt: renderEvaluationCandidatePrompt(candidate),
38
+ criteria: oracle.rubric
39
+ };
40
+ return [
41
+ ...oracle.positiveControls.map((control) => compileControl({
42
+ ...common,
43
+ authoredControlId: control.id,
44
+ kind: "positive",
45
+ targetedCriterionIds: [],
46
+ expectedOutcome: "accept",
47
+ response: control.response
48
+ })),
49
+ ...oracle.negativeControls.map((control) => compileControl({
50
+ ...common,
51
+ authoredControlId: control.id,
52
+ kind: control.kind,
53
+ targetedCriterionIds: [...control.targetedCriterionIds].sort(stableCompare),
54
+ expectedOutcome: "reject",
55
+ response: control.response
56
+ }))
57
+ ];
58
+ };
59
+ const planSemantic = (plan) => ({
60
+ version: plan.version,
61
+ policyVersion: plan.policyVersion,
62
+ evaluationDigest: plan.evaluationDigest,
63
+ judgeProtocolVersion: plan.judgeProtocolVersion,
64
+ judgeModel: plan.judgeModel,
65
+ judgeReasoningEffort: plan.judgeReasoningEffort,
66
+ judgeSystemPrompt: plan.judgeSystemPrompt,
67
+ minimumJudgeScore: plan.minimumJudgeScore,
68
+ callCount: plan.callCount,
69
+ controls: plan.controls
70
+ });
71
+ const expectedPlanDigest = (plan) => sha256(JSON.stringify(planSemantic(plan)));
72
+ const reportSemantic = (report) => ({
73
+ version: report.version,
74
+ policyVersion: report.policyVersion,
75
+ planDigest: report.planDigest,
76
+ evaluationDigest: report.evaluationDigest,
77
+ judgeProtocolVersion: report.judgeProtocolVersion,
78
+ judgeModel: report.judgeModel,
79
+ judgeReasoningEffort: report.judgeReasoningEffort,
80
+ judgeSystemPrompt: report.judgeSystemPrompt,
81
+ minimumJudgeScore: report.minimumJudgeScore,
82
+ startedAt: report.startedAt,
83
+ finishedAt: report.finishedAt,
84
+ callCount: report.callCount,
85
+ acceptedCount: report.acceptedCount,
86
+ rejectedCount: report.rejectedCount,
87
+ matchedCount: report.matchedCount,
88
+ passed: report.passed,
89
+ results: report.results
90
+ });
91
+ export const evaluationGraderCalibrationReportDigest = (report) => sha256(JSON.stringify(reportSemantic(report)));
92
+ const assertNonEmpty = (value, label) => {
93
+ if (value.trim().length === 0) {
94
+ throw new Error(`${label} must not be empty`);
95
+ }
96
+ };
97
+ const assertPlan = (plan) => {
98
+ Schema.decodeSync(CalibrationPlanSchema)(plan);
99
+ if (plan.policyVersion !== EVAL_GRADER_CALIBRATION_POLICY_VERSION ||
100
+ plan.judgeProtocolVersion !== EVAL_GRADER_JUDGE_PROTOCOL_VERSION) {
101
+ throw new Error("grader calibration plan uses an unsupported policy version");
102
+ }
103
+ if (plan.minimumJudgeScore !== EVAL_GRADER_MINIMUM_SCORE ||
104
+ plan.judgeSystemPrompt !== EVAL_GRADER_SYSTEM_PROMPT) {
105
+ throw new Error("grader calibration plan does not bind the current judge policy");
106
+ }
107
+ assertNonEmpty(plan.evaluationDigest, "evaluation digest");
108
+ assertNonEmpty(plan.judgeModel, "grader calibration judge model");
109
+ if (plan.callCount !== plan.controls.length || plan.callCount < 1) {
110
+ throw new Error("grader calibration plan call count is inconsistent");
111
+ }
112
+ if (new Set(plan.controls.map((control) => control.id)).size !==
113
+ plan.controls.length) {
114
+ throw new Error("grader calibration control ids must be unique");
115
+ }
116
+ for (const control of plan.controls) {
117
+ assertNonEmpty(control.suiteId, "grader calibration suite id");
118
+ assertNonEmpty(control.caseId, "grader calibration case id");
119
+ assertNonEmpty(control.authoredControlId, "authored control id");
120
+ assertNonEmpty(control.criteria, "grader calibration criteria");
121
+ assertNonEmpty(control.prompt, "grader calibration prompt");
122
+ assertNonEmpty(control.response, "grader calibration response");
123
+ const canonicalTargetIds = [...new Set(control.targetedCriterionIds)].sort(stableCompare);
124
+ if ((control.kind === "positive" &&
125
+ (control.expectedOutcome !== "accept" ||
126
+ control.targetedCriterionIds.length !== 0)) ||
127
+ (control.kind !== "positive" &&
128
+ (control.expectedOutcome !== "reject" ||
129
+ control.targetedCriterionIds.length === 0 ||
130
+ control.targetedCriterionIds.length !== canonicalTargetIds.length ||
131
+ control.targetedCriterionIds.some((criterionId, index) => criterionId !== canonicalTargetIds[index])))) {
132
+ throw new Error(`grader calibration control ${JSON.stringify(control.id)} has inconsistent semantics`);
133
+ }
134
+ const digest = sha256(JSON.stringify(controlSemantic(control)));
135
+ if (control.controlDigest !== digest ||
136
+ control.id !== `cal_${digest}`) {
137
+ throw new Error(`grader calibration control ${JSON.stringify(control.id)} digest is stale`);
138
+ }
139
+ }
140
+ const { planDigest: _planDigest, ...withoutDigest } = plan;
141
+ if (plan.planDigest !== expectedPlanDigest(withoutDigest)) {
142
+ throw new Error("grader calibration plan digest is stale");
143
+ }
144
+ };
145
+ const isCanonicalIsoInstant = (value) => {
146
+ const timestamp = Date.parse(value);
147
+ return (Number.isFinite(timestamp) && new Date(timestamp).toISOString() === value);
148
+ };
149
+ const assertReportMatchesPlan = (plan, report) => {
150
+ if (report.planDigest !== plan.planDigest ||
151
+ report.evaluationDigest !== plan.evaluationDigest ||
152
+ report.judgeProtocolVersion !== plan.judgeProtocolVersion ||
153
+ report.judgeModel !== plan.judgeModel ||
154
+ report.judgeReasoningEffort !== plan.judgeReasoningEffort ||
155
+ report.judgeSystemPrompt !== plan.judgeSystemPrompt ||
156
+ report.minimumJudgeScore !== plan.minimumJudgeScore) {
157
+ throw new Error("grader calibration report does not match its expected plan");
158
+ }
159
+ };
160
+ export function compileEvaluationGraderCalibrationPlan(input) {
161
+ const { proposal } = input;
162
+ assertEvaluationProposal(proposal);
163
+ assertEvaluationStructure(input);
164
+ const controls = [
165
+ ...proposal.suites.flatMap((suite) => suite.cases.flatMap((testCase) => controlsForCase(testCase, "dimension", suite.dimensionId))),
166
+ ...proposal.compositionSuite.cases.flatMap((testCase) => controlsForCase(testCase, "composition", "composition"))
167
+ ].sort((left, right) => stableCompare([
168
+ left.suiteKind,
169
+ left.suiteId,
170
+ left.caseId,
171
+ left.kind,
172
+ left.authoredControlId,
173
+ left.id
174
+ ].join("\u0000"), [
175
+ right.suiteKind,
176
+ right.suiteId,
177
+ right.caseId,
178
+ right.kind,
179
+ right.authoredControlId,
180
+ right.id
181
+ ].join("\u0000")));
182
+ if (controls.length === 0) {
183
+ throw new Error("grader calibration requires at least one authored control");
184
+ }
185
+ const withoutDigest = {
186
+ version: 1,
187
+ policyVersion: EVAL_GRADER_CALIBRATION_POLICY_VERSION,
188
+ evaluationDigest: proposal.evaluationDigest,
189
+ judgeProtocolVersion: EVAL_GRADER_JUDGE_PROTOCOL_VERSION,
190
+ judgeModel: proposal.judgeModel,
191
+ // The project protocol does not currently collect a judge effort. Null
192
+ // explicitly binds the absence of an override instead of pretending the
193
+ // provider-resolved default is known.
194
+ judgeReasoningEffort: null,
195
+ judgeSystemPrompt: EVAL_GRADER_SYSTEM_PROMPT,
196
+ minimumJudgeScore: EVAL_GRADER_MINIMUM_SCORE,
197
+ callCount: controls.length,
198
+ controls
199
+ };
200
+ const plan = {
201
+ ...withoutDigest,
202
+ planDigest: expectedPlanDigest(withoutDigest)
203
+ };
204
+ assertPlan(plan);
205
+ return plan;
206
+ }
207
+ export function compileEvaluationGraderCalibrationReport(input) {
208
+ assertPlan(input.plan);
209
+ if (!isCanonicalIsoInstant(input.startedAt) ||
210
+ !isCanonicalIsoInstant(input.finishedAt) ||
211
+ Date.parse(input.finishedAt) < Date.parse(input.startedAt)) {
212
+ throw new Error("grader calibration report timestamps are invalid");
213
+ }
214
+ if (input.observations.length !== input.plan.controls.length) {
215
+ throw new Error("grader calibration observations are incomplete");
216
+ }
217
+ const controls = new Map(input.plan.controls.map((control) => [control.id, control]));
218
+ const seenControls = new Set();
219
+ const seenCalls = new Set();
220
+ const results = input.observations.map((observation) => {
221
+ Schema.decodeSync(CalibrationObservationSchema)(observation);
222
+ const control = controls.get(observation.controlId);
223
+ if (control === undefined) {
224
+ throw new Error(`grader calibration contains unknown control ${JSON.stringify(observation.controlId)}`);
225
+ }
226
+ if (seenControls.has(observation.controlId)) {
227
+ throw new Error(`grader calibration contains duplicate control ${JSON.stringify(observation.controlId)}`);
228
+ }
229
+ seenControls.add(observation.controlId);
230
+ if (observation.judgeModel !== input.plan.judgeModel) {
231
+ throw new Error("grader calibration observation uses the wrong judge model");
232
+ }
233
+ assertNonEmpty(observation.callId, "grader calibration call id");
234
+ if (seenCalls.has(observation.callId)) {
235
+ throw new Error("grader calibration observations reuse a judge call id");
236
+ }
237
+ seenCalls.add(observation.callId);
238
+ assertNonEmpty(observation.reason, "grader calibration verdict reason");
239
+ const accepted = observation.pass &&
240
+ observation.score >= input.plan.minimumJudgeScore;
241
+ const observedOutcome = accepted ? "accept" : "reject";
242
+ return {
243
+ ...observation,
244
+ expectedOutcome: control.expectedOutcome,
245
+ observedOutcome,
246
+ matchedExpectation: observedOutcome === control.expectedOutcome
247
+ };
248
+ });
249
+ if (seenControls.size !== controls.size) {
250
+ throw new Error("grader calibration observations do not cover every control");
251
+ }
252
+ results.sort((left, right) => stableCompare(left.controlId, right.controlId));
253
+ const acceptedCount = results.filter((result) => result.observedOutcome === "accept").length;
254
+ const rejectedCount = results.length - acceptedCount;
255
+ const matchedCount = results.filter((result) => result.matchedExpectation).length;
256
+ const withoutDigest = {
257
+ version: 1,
258
+ policyVersion: EVAL_GRADER_CALIBRATION_POLICY_VERSION,
259
+ planDigest: input.plan.planDigest,
260
+ evaluationDigest: input.plan.evaluationDigest,
261
+ judgeProtocolVersion: input.plan.judgeProtocolVersion,
262
+ judgeModel: input.plan.judgeModel,
263
+ judgeReasoningEffort: input.plan.judgeReasoningEffort,
264
+ judgeSystemPrompt: input.plan.judgeSystemPrompt,
265
+ minimumJudgeScore: input.plan.minimumJudgeScore,
266
+ startedAt: input.startedAt,
267
+ finishedAt: input.finishedAt,
268
+ callCount: results.length,
269
+ acceptedCount,
270
+ rejectedCount,
271
+ matchedCount,
272
+ passed: matchedCount === results.length,
273
+ results
274
+ };
275
+ return {
276
+ ...withoutDigest,
277
+ reportDigest: evaluationGraderCalibrationReportDigest(withoutDigest)
278
+ };
279
+ }
280
+ export function assertEvaluationGraderCalibrationPassed(input) {
281
+ assertPlan(input.plan);
282
+ const { report } = input;
283
+ Schema.decodeSync(CalibrationReportSchema)(report);
284
+ assertReportMatchesPlan(input.plan, report);
285
+ if (!isCanonicalIsoInstant(report.startedAt) ||
286
+ !isCanonicalIsoInstant(report.finishedAt) ||
287
+ Date.parse(report.finishedAt) < Date.parse(report.startedAt)) {
288
+ throw new Error("grader calibration report timestamps are invalid");
289
+ }
290
+ const { reportDigest: _reportDigest, ...withoutDigest } = report;
291
+ if (report.reportDigest !==
292
+ evaluationGraderCalibrationReportDigest(withoutDigest)) {
293
+ throw new Error("grader calibration report digest is stale");
294
+ }
295
+ if (report.policyVersion !== EVAL_GRADER_CALIBRATION_POLICY_VERSION ||
296
+ report.callCount !== report.results.length ||
297
+ report.callCount !== input.plan.callCount ||
298
+ new Set(report.results.map((result) => result.controlId)).size !==
299
+ report.results.length ||
300
+ new Set(report.results.map((result) => result.callId)).size !==
301
+ report.results.length ||
302
+ report.matchedCount !==
303
+ report.results.filter((result) => result.matchedExpectation).length ||
304
+ report.acceptedCount !==
305
+ report.results.filter((result) => result.observedOutcome === "accept")
306
+ .length ||
307
+ report.rejectedCount !==
308
+ report.results.filter((result) => result.observedOutcome === "reject")
309
+ .length) {
310
+ throw new Error("grader calibration report summary is inconsistent");
311
+ }
312
+ const controls = new Map(input.plan.controls.map((control) => [control.id, control]));
313
+ if (report.results.some((result) => {
314
+ const control = controls.get(result.controlId);
315
+ return (control === undefined ||
316
+ result.judgeModel !== input.plan.judgeModel ||
317
+ result.callId.trim().length === 0 ||
318
+ result.reason.trim().length === 0 ||
319
+ result.expectedOutcome !== control.expectedOutcome ||
320
+ result.observedOutcome !==
321
+ (result.pass && result.score >= input.plan.minimumJudgeScore
322
+ ? "accept"
323
+ : "reject") ||
324
+ result.matchedExpectation !==
325
+ (result.observedOutcome === result.expectedOutcome));
326
+ })) {
327
+ throw new Error("grader calibration report results are inconsistent");
328
+ }
329
+ if (!report.passed ||
330
+ report.matchedCount !== report.callCount ||
331
+ report.results.some((result) => !result.matchedExpectation)) {
332
+ throw new Error("grader calibration did not satisfy every control");
333
+ }
334
+ }
@@ -0,0 +1,24 @@
1
+ export declare const EVAL_GRADER_JUDGE_PROTOCOL_VERSION: 1;
2
+ /**
3
+ * The current generated qualification suite's inherited judge acceptance
4
+ * threshold. Calibration binds this exact behavior; it does not claim that the
5
+ * threshold is empirically optimal.
6
+ */
7
+ export declare const EVAL_GRADER_MINIMUM_SCORE = 0.8;
8
+ /**
9
+ * Pin the judge instruction bytes in calibration plans rather than depending
10
+ * on an SDK default that can change independently of a reviewed plan.
11
+ */
12
+ export declare const EVAL_GRADER_SYSTEM_PROMPT: string;
13
+ export declare const EVAL_CANDIDATE_RESPONSE_REQUIREMENTS: readonly ["Answer every part of the request directly and completely.", "When translating protocols, emit the complete target-protocol envelope and terminal event; do not merely reframe or forward source objects.", "When asked for a body, bytes, frames, or code, include the concrete output rather than only describing it."];
14
+ export type EvalCandidatePromptInput = {
15
+ readonly prompt: string;
16
+ readonly context: string;
17
+ };
18
+ export declare function renderEvaluationCandidatePrompt(candidateCase: EvalCandidatePromptInput): string;
19
+ /**
20
+ * Render the generated-suite statement from the same policy constants as the
21
+ * pure compiler renderer. This remains source generation because materialized
22
+ * suites execute outside eval-setup and cannot import its private modules.
23
+ */
24
+ export declare function renderEvaluationCandidatePromptSource(candidateExpression?: string): string;
@@ -0,0 +1,54 @@
1
+ export const EVAL_GRADER_JUDGE_PROTOCOL_VERSION = 1;
2
+ /**
3
+ * The current generated qualification suite's inherited judge acceptance
4
+ * threshold. Calibration binds this exact behavior; it does not claim that the
5
+ * threshold is empirically optimal.
6
+ */
7
+ export const EVAL_GRADER_MINIMUM_SCORE = 0.8;
8
+ /**
9
+ * Pin the judge instruction bytes in calibration plans rather than depending
10
+ * on an SDK default that can change independently of a reviewed plan.
11
+ */
12
+ export const EVAL_GRADER_SYSTEM_PROMPT = [
13
+ "You are a strict evaluator for AI agent outputs.",
14
+ "Grade the candidate against the criteria. Be conservative: only pass when the criteria are clearly met.",
15
+ "Return your verdict as the structured JSON object you were asked for — nothing else."
16
+ ].join("\n");
17
+ export const EVAL_CANDIDATE_RESPONSE_REQUIREMENTS = [
18
+ "Answer every part of the request directly and completely.",
19
+ "When translating protocols, emit the complete target-protocol envelope and terminal event; do not merely reframe or forward source objects.",
20
+ "When asked for a body, bytes, frames, or code, include the concrete output rather than only describing it."
21
+ ];
22
+ const candidatePromptParts = (candidateCase) => [
23
+ candidateCase.prompt,
24
+ "",
25
+ "Response requirements:",
26
+ ...EVAL_CANDIDATE_RESPONSE_REQUIREMENTS.map((requirement) => `- ${requirement}`),
27
+ "",
28
+ "Reference material:",
29
+ "-----",
30
+ candidateCase.context,
31
+ "-----"
32
+ ];
33
+ export function renderEvaluationCandidatePrompt(candidateCase) {
34
+ return candidatePromptParts(candidateCase).join("\n");
35
+ }
36
+ /**
37
+ * Render the generated-suite statement from the same policy constants as the
38
+ * pure compiler renderer. This remains source generation because materialized
39
+ * suites execute outside eval-setup and cannot import its private modules.
40
+ */
41
+ export function renderEvaluationCandidatePromptSource(candidateExpression = "candidateCase") {
42
+ return `const responseRequirements = ${JSON.stringify(EVAL_CANDIDATE_RESPONSE_REQUIREMENTS)};
43
+ const prompt = [
44
+ ${candidateExpression}.prompt,
45
+ "",
46
+ "Response requirements:",
47
+ ...responseRequirements.map((requirement) => \`- \${requirement}\`),
48
+ "",
49
+ "Reference material:",
50
+ "-----",
51
+ ${candidateExpression}.context,
52
+ "-----"
53
+ ].join("\\n");`;
54
+ }
@@ -0,0 +1,4 @@
1
+ import type { EvalEvaluationProposal } from "./project-contracts.js";
2
+ export declare const evaluationProposalPayload: (proposal: EvalEvaluationProposal) => Omit<EvalEvaluationProposal, "evaluationDigest">;
3
+ export declare function evaluationProposalDigest(proposal: Omit<EvalEvaluationProposal, "evaluationDigest">): string;
4
+ export declare function assertEvaluationProposal(proposal: EvalEvaluationProposal): void;
@@ -0,0 +1,91 @@
1
+ import { createHash } from "node:crypto";
2
+ const digest = (value) => createHash("sha256").update(JSON.stringify(value)).digest("hex");
3
+ export const evaluationProposalPayload = (proposal) => ({
4
+ version: proposal.version,
5
+ basisDigest: proposal.basisDigest,
6
+ candidateModels: proposal.candidateModels,
7
+ ...(proposal.candidateReasoningEfforts === undefined
8
+ ? {}
9
+ : { candidateReasoningEfforts: proposal.candidateReasoningEfforts }),
10
+ judgeModel: proposal.judgeModel,
11
+ ...(proposal.evidenceSources === undefined
12
+ ? {}
13
+ : { evidenceSources: proposal.evidenceSources }),
14
+ suites: proposal.suites,
15
+ decompositionBenchmark: proposal.decompositionBenchmark,
16
+ compositionSuite: proposal.compositionSuite
17
+ });
18
+ export function evaluationProposalDigest(proposal) {
19
+ return digest(proposal);
20
+ }
21
+ const assertDimensionSuite = (suite) => {
22
+ if (suite.maximumOutputTokens < 1) {
23
+ throw new Error(`dimension suite ${JSON.stringify(suite.dimensionId)} has no output allowance`);
24
+ }
25
+ if (suite.cases.length < 5) {
26
+ throw new Error(`dimension suite ${JSON.stringify(suite.dimensionId)} must contain at least five cases`);
27
+ }
28
+ const ids = new Set();
29
+ for (const testCase of suite.cases) {
30
+ if (testCase.id.trim().length === 0 ||
31
+ testCase.prompt.trim().length === 0 ||
32
+ testCase.rubric.trim().length === 0) {
33
+ throw new Error(`dimension suite ${JSON.stringify(suite.dimensionId)} contains an incomplete case`);
34
+ }
35
+ if (ids.has(testCase.id)) {
36
+ throw new Error(`dimension suite ${JSON.stringify(suite.dimensionId)} contains duplicate case ${JSON.stringify(testCase.id)}`);
37
+ }
38
+ ids.add(testCase.id);
39
+ }
40
+ };
41
+ export function assertEvaluationProposal(proposal) {
42
+ if (proposal.evaluationDigest !==
43
+ evaluationProposalDigest(evaluationProposalPayload(proposal))) {
44
+ throw new Error("evaluation proposal digest does not match its contents");
45
+ }
46
+ if (proposal.candidateModels.length < 2) {
47
+ throw new Error("evaluation proposal requires at least two candidate models");
48
+ }
49
+ if (new Set(proposal.candidateModels).size !== proposal.candidateModels.length) {
50
+ throw new Error("evaluation proposal candidate models must be unique");
51
+ }
52
+ if (Object.keys(proposal.candidateReasoningEfforts ?? {}).some((model) => !proposal.candidateModels.includes(model))) {
53
+ throw new Error("evaluation proposal reasoning efforts must belong to candidate models");
54
+ }
55
+ if (proposal.judgeModel.trim().length === 0) {
56
+ throw new Error("evaluation proposal judge must be explicit");
57
+ }
58
+ const dimensions = new Set();
59
+ for (const suite of proposal.suites) {
60
+ if (dimensions.has(suite.dimensionId)) {
61
+ throw new Error(`duplicate dimension suite ${JSON.stringify(suite.dimensionId)}`);
62
+ }
63
+ dimensions.add(suite.dimensionId);
64
+ assertDimensionSuite(suite);
65
+ }
66
+ if (proposal.decompositionBenchmark.maximumVectorL1Error < 0 ||
67
+ proposal.decompositionBenchmark.maximumVectorL1Error > 2 ||
68
+ proposal.decompositionBenchmark.cases.length < 5) {
69
+ throw new Error("decomposition benchmark must define a reviewed threshold and at least five cases");
70
+ }
71
+ if (proposal.compositionSuite.maximumOutputTokens < 1 ||
72
+ proposal.compositionSuite.minimumWinnerScoreGap < 0 ||
73
+ proposal.compositionSuite.minimumWinnerScoreGap > 1 ||
74
+ proposal.compositionSuite.minimumWinnerAgreement < 0 ||
75
+ proposal.compositionSuite.minimumWinnerAgreement > 1 ||
76
+ proposal.compositionSuite.cases.length < 5) {
77
+ throw new Error("composition benchmark must define reviewed thresholds and at least five cases");
78
+ }
79
+ for (const [label, cases] of [
80
+ ["decomposition", proposal.decompositionBenchmark.cases],
81
+ ["composition", proposal.compositionSuite.cases]
82
+ ]) {
83
+ const ids = new Set();
84
+ for (const testCase of cases) {
85
+ if (testCase.id.trim().length === 0 || ids.has(testCase.id)) {
86
+ throw new Error(`${label} benchmark contains an invalid or duplicate case id`);
87
+ }
88
+ ids.add(testCase.id);
89
+ }
90
+ }
91
+ }
@@ -0,0 +1,68 @@
1
+ import type { RoutingBasis } from "@velum-labs/routekit-eval-contracts";
2
+ import type { EvalEvidenceBlock, EvalEvidenceSource } from "./project-contracts.js";
3
+ export declare const EVAL_SOURCE_RETRIEVAL_INDEX_FILES = 512;
4
+ export declare const EVAL_SOURCE_RETRIEVAL_INDEX_BYTES: number;
5
+ export declare const EVAL_SOURCE_RETRIEVAL_FILE_BYTES: number;
6
+ export declare const EVAL_SOURCE_RETRIEVAL_PACKET_BLOCKS = 20;
7
+ export declare const EVAL_SOURCE_RETRIEVAL_BLOCKS_PER_SOURCE = 2;
8
+ export type EvalSourceRetrievalCandidate = {
9
+ readonly path: string;
10
+ readonly sizeBytes: number;
11
+ readonly sourceClass: "implementation" | "test" | "protocol" | "documentation" | "operational";
12
+ readonly explicitReason?: string;
13
+ readonly pathScore?: number;
14
+ };
15
+ export type EvalSourceRetrievalDocument = EvalSourceRetrievalCandidate & {
16
+ readonly content: string;
17
+ };
18
+ export type EvalSourceRetrievalSelection = EvalSourceRetrievalDocument & {
19
+ readonly inclusionReason: string;
20
+ };
21
+ export type EvalEvidenceRetrievalDocument = EvalSourceRetrievalCandidate & {
22
+ readonly evidenceSource: EvalEvidenceSource;
23
+ readonly pathScore?: number;
24
+ };
25
+ export type EvalEvidenceRetrievalSelection = {
26
+ readonly source: EvalSourceRetrievalCandidate;
27
+ readonly blocks: readonly EvalEvidenceBlock[];
28
+ readonly sizeBytes: number;
29
+ readonly inclusionReason: string;
30
+ };
31
+ /**
32
+ * Tokenizes identifiers and prose without asking a model to infer source
33
+ * relevance. CamelCase boundaries are retained so `providerBackend` can match
34
+ * the reviewed positive scope `provider backend`.
35
+ */
36
+ export declare const evaluationSourceTokens: (value: string) => readonly string[];
37
+ /**
38
+ * Produces a bounded content-index plan. Positive path matches identify likely
39
+ * ownership roots, then generic filenames inside those roots may compete by
40
+ * content. This avoids reading an unbounded repository or relying on inventory
41
+ * order.
42
+ */
43
+ export declare function planEvaluationSourceRetrieval(input: {
44
+ readonly candidates: readonly EvalSourceRetrievalCandidate[];
45
+ readonly workloadDescription?: string;
46
+ readonly targetDimensions?: RoutingBasis["dimensions"];
47
+ readonly maximumFiles?: number;
48
+ readonly maximumBytes?: number;
49
+ }): readonly EvalSourceRetrievalCandidate[];
50
+ export declare function selectEvaluationSourcePacket(input: {
51
+ readonly documents: readonly EvalSourceRetrievalDocument[];
52
+ readonly workloadDescription?: string;
53
+ readonly targetDimensions?: RoutingBasis["dimensions"];
54
+ readonly maximumFiles: number;
55
+ readonly maximumBytes: number;
56
+ }): readonly EvalSourceRetrievalSelection[];
57
+ /**
58
+ * Selects compiler-owned evidence blocks rather than whole files. Packet
59
+ * assembly proceeds in per-source rounds so one large source cannot consume
60
+ * the complete authoring budget before complementary owners can compete.
61
+ */
62
+ export declare function selectEvaluationEvidencePacket(input: {
63
+ readonly documents: readonly EvalEvidenceRetrievalDocument[];
64
+ readonly targetDimensions: RoutingBasis["dimensions"];
65
+ readonly maximumBlocks?: number;
66
+ readonly maximumBytes: number;
67
+ readonly maximumBlocksPerSource?: number;
68
+ }): readonly EvalEvidenceRetrievalSelection[];