@velum-labs/routekit-eval-setup 1.3.2 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (342) hide show
  1. package/dist/adapters/authoring-responses-request.d.ts +5 -0
  2. package/dist/adapters/authoring-responses-request.js +38 -0
  3. package/dist/adapters/evaluation-evidence-freshness.d.ts +7 -0
  4. package/dist/adapters/evaluation-evidence-freshness.js +76 -0
  5. package/dist/adapters/git-task-history.d.ts +67 -0
  6. package/dist/adapters/git-task-history.js +171 -0
  7. package/dist/adapters/integrated-repository-history.d.ts +21 -0
  8. package/dist/adapters/integrated-repository-history.js +175 -0
  9. package/dist/adapters/repository-command-diagnostic.d.ts +8 -0
  10. package/dist/adapters/repository-command-diagnostic.js +46 -0
  11. package/dist/adapters/repository-command-evidence.d.ts +13 -0
  12. package/dist/adapters/repository-command-evidence.js +102 -0
  13. package/dist/adapters/repository-command-runner.d.ts +238 -0
  14. package/dist/adapters/repository-command-runner.js +1483 -0
  15. package/dist/adapters/repository-import-context.d.ts +47 -0
  16. package/dist/adapters/repository-import-context.js +469 -0
  17. package/dist/adapters/repository-node-test-reporter.d.ts +3 -0
  18. package/dist/adapters/repository-node-test-reporter.js +27 -0
  19. package/dist/adapters/repository-review-evidence.d.ts +39 -0
  20. package/dist/adapters/repository-review-evidence.js +632 -0
  21. package/dist/adapters/repository-seed-selection.d.ts +7 -0
  22. package/dist/adapters/repository-seed-selection.js +79 -0
  23. package/dist/adapters/repository-solution-edits.d.ts +49 -0
  24. package/dist/adapters/repository-solution-edits.js +136 -0
  25. package/dist/adapters/repository-vitest-phase-adapter.d.ts +8 -0
  26. package/dist/adapters/repository-vitest-phase-adapter.js +310 -0
  27. package/dist/adapters/repository-vitest-reporter.d.ts +24 -0
  28. package/dist/adapters/repository-vitest-reporter.js +314 -0
  29. package/dist/adapters/strict-authoring-schema.d.ts +5 -0
  30. package/dist/adapters/strict-authoring-schema.js +158 -0
  31. package/dist/adapters/test-discovery.d.ts +30 -0
  32. package/dist/adapters/test-discovery.js +124 -0
  33. package/dist/adapters/typescript-repository-index.d.ts +51 -0
  34. package/dist/adapters/typescript-repository-index.js +226 -0
  35. package/dist/agentic-capabilities-protocol.d.ts +1373 -0
  36. package/dist/agentic-capabilities-protocol.js +786 -0
  37. package/dist/case-checkpoint-store.d.ts +29 -0
  38. package/dist/case-checkpoint-store.js +133 -0
  39. package/dist/case-pipeline-protocol-v2.d.ts +184 -0
  40. package/dist/case-pipeline-protocol-v2.js +193 -0
  41. package/dist/case-pipeline-protocol.d.ts +2626 -0
  42. package/dist/case-pipeline-protocol.js +371 -0
  43. package/dist/effect-api.d.ts +74 -10
  44. package/dist/effect-api.js +56 -6
  45. package/dist/errors.d.ts +31 -0
  46. package/dist/errors.js +10 -0
  47. package/dist/eval-capability-execution-envelope.d.ts +64 -0
  48. package/dist/eval-capability-execution-envelope.js +98 -0
  49. package/dist/eval-capability-policy.d.ts +90 -0
  50. package/dist/eval-capability-policy.js +107 -0
  51. package/dist/eval-event-log.d.ts +140 -0
  52. package/dist/eval-event-log.js +220 -0
  53. package/dist/evaluation-authoring-policy.d.ts +18 -0
  54. package/dist/evaluation-authoring-policy.js +19 -0
  55. package/dist/evaluation-authoring-validation.d.ts +22 -0
  56. package/dist/evaluation-authoring-validation.js +72 -0
  57. package/dist/evaluation-evidence.d.ts +20 -0
  58. package/dist/evaluation-evidence.js +319 -0
  59. package/dist/evaluation-grader-calibration-protocol.d.ts +108 -0
  60. package/dist/evaluation-grader-calibration-protocol.js +80 -0
  61. package/dist/evaluation-grader-calibration.d.ts +18 -0
  62. package/dist/evaluation-grader-calibration.js +334 -0
  63. package/dist/evaluation-grading-policy.d.ts +24 -0
  64. package/dist/evaluation-grading-policy.js +54 -0
  65. package/dist/evaluation-proposal-policy.d.ts +4 -0
  66. package/dist/evaluation-proposal-policy.js +91 -0
  67. package/dist/evaluation-source-retrieval.d.ts +68 -0
  68. package/dist/evaluation-source-retrieval.js +513 -0
  69. package/dist/evaluation-structure-policy.d.ts +29 -0
  70. package/dist/evaluation-structure-policy.js +138 -0
  71. package/dist/index.d.ts +124 -17
  72. package/dist/index.js +69 -11
  73. package/dist/inspection.js +2 -3
  74. package/dist/project-artifacts.d.ts +7 -2
  75. package/dist/project-artifacts.js +49 -136
  76. package/dist/project-authoring.d.ts +66 -5
  77. package/dist/project-authoring.js +783 -109
  78. package/dist/project-contracts.d.ts +419 -84
  79. package/dist/project-contracts.js +160 -52
  80. package/dist/project-store.js +2 -1
  81. package/dist/project-workflow.d.ts +5 -4
  82. package/dist/project-workflow.js +154 -35
  83. package/dist/repository-adversary-protocol.d.ts +64 -0
  84. package/dist/repository-adversary-protocol.js +105 -0
  85. package/dist/repository-behavior-protocol.d.ts +188 -0
  86. package/dist/repository-behavior-protocol.js +202 -0
  87. package/dist/repository-benchmark-protocol.d.ts +487 -0
  88. package/dist/repository-benchmark-protocol.js +96 -0
  89. package/dist/repository-execution-protocol.d.ts +150 -0
  90. package/dist/repository-execution-protocol.js +38 -0
  91. package/dist/repository-fixture-instructions.d.ts +3 -0
  92. package/dist/repository-fixture-instructions.js +91 -0
  93. package/dist/repository-fixture-protocol.d.ts +79 -0
  94. package/dist/repository-fixture-protocol.js +79 -0
  95. package/dist/repository-foundry-plan-protocol.d.ts +118 -0
  96. package/dist/repository-foundry-plan-protocol.js +296 -0
  97. package/dist/repository-foundry-progress-protocol.d.ts +52 -0
  98. package/dist/repository-foundry-progress-protocol.js +52 -0
  99. package/dist/repository-improvement-protocol.d.ts +100 -0
  100. package/dist/repository-improvement-protocol.js +106 -0
  101. package/dist/repository-language-model-protocol.d.ts +43 -0
  102. package/dist/repository-language-model-protocol.js +146 -0
  103. package/dist/repository-oracle-coverage-protocol.d.ts +18 -0
  104. package/dist/repository-oracle-coverage-protocol.js +39 -0
  105. package/dist/repository-oracle-execution-binding.d.ts +27 -0
  106. package/dist/repository-oracle-execution-binding.js +59 -0
  107. package/dist/repository-oracle-protocol.d.ts +230 -0
  108. package/dist/repository-oracle-protocol.js +156 -0
  109. package/dist/repository-oracle-scope-policy.d.ts +22 -0
  110. package/dist/repository-oracle-scope-policy.js +92 -0
  111. package/dist/repository-quality-policy.d.ts +15 -0
  112. package/dist/repository-quality-policy.js +357 -0
  113. package/dist/repository-routing-benchmark-protocol.d.ts +176 -0
  114. package/dist/repository-routing-benchmark-protocol.js +103 -0
  115. package/dist/repository-routing-model-protocol.d.ts +36 -0
  116. package/dist/repository-routing-model-protocol.js +89 -0
  117. package/dist/repository-routing-plan-protocol.d.ts +112 -0
  118. package/dist/repository-routing-plan-protocol.js +58 -0
  119. package/dist/repository-routing-quality-policy.d.ts +9 -0
  120. package/dist/repository-routing-quality-policy.js +191 -0
  121. package/dist/repository-seed-qualification-progress-protocol.d.ts +205 -0
  122. package/dist/repository-seed-qualification-progress-protocol.js +28 -0
  123. package/dist/repository-semantic-calibration-protocol.d.ts +768 -0
  124. package/dist/repository-semantic-calibration-protocol.js +276 -0
  125. package/dist/repository-semantic-calibration.d.ts +163 -0
  126. package/dist/repository-semantic-calibration.js +581 -0
  127. package/dist/repository-specification-contract-facts-protocol.d.ts +224 -0
  128. package/dist/repository-specification-contract-facts-protocol.js +276 -0
  129. package/dist/repository-specification-critique-protocol.d.ts +189 -0
  130. package/dist/repository-specification-critique-protocol.js +103 -0
  131. package/dist/repository-task-family-protocol.d.ts +24 -0
  132. package/dist/repository-task-family-protocol.js +37 -0
  133. package/dist/repository-task-seed-protocol.d.ts +384 -0
  134. package/dist/repository-task-seed-protocol.js +236 -0
  135. package/dist/repository-trajectory-protocol.d.ts +20 -0
  136. package/dist/repository-trajectory-protocol.js +42 -0
  137. package/dist/service.js +1 -1
  138. package/dist/services/adversary/service.d.ts +64 -0
  139. package/dist/services/adversary/service.js +330 -0
  140. package/dist/services/benchmark-compiler/service.d.ts +450 -0
  141. package/dist/services/benchmark-compiler/service.js +9 -0
  142. package/dist/services/budgeted-model/service.d.ts +118 -0
  143. package/dist/services/budgeted-model/service.js +460 -0
  144. package/dist/services/case-authoring/service.d.ts +163 -0
  145. package/dist/services/case-authoring/service.js +1456 -0
  146. package/dist/services/case-finalization/service.d.ts +283 -0
  147. package/dist/services/case-finalization/service.js +370 -0
  148. package/dist/services/case-generation/service.d.ts +619 -0
  149. package/dist/services/case-generation/service.js +2628 -0
  150. package/dist/services/case-pipeline/service.d.ts +31 -0
  151. package/dist/services/case-pipeline/service.js +485 -0
  152. package/dist/services/case-pipeline-v2/service.d.ts +70 -0
  153. package/dist/services/case-pipeline-v2/service.js +477 -0
  154. package/dist/services/command-observability/service.d.ts +13 -0
  155. package/dist/services/command-observability/service.js +3 -0
  156. package/dist/services/dimension-labeling/service.d.ts +77 -0
  157. package/dist/services/dimension-labeling/service.js +188 -0
  158. package/dist/services/eval-candidate/service.d.ts +208 -0
  159. package/dist/services/eval-candidate/service.js +64 -0
  160. package/dist/services/eval-capabilities/service.d.ts +183 -0
  161. package/dist/services/eval-capabilities/service.js +1433 -0
  162. package/dist/services/eval-environment/service.d.ts +173 -0
  163. package/dist/services/eval-environment/service.js +127 -0
  164. package/dist/services/evidence-reconstruction/service.d.ts +36 -0
  165. package/dist/services/evidence-reconstruction/service.js +145 -0
  166. package/dist/services/fixture-builder/service.d.ts +62 -0
  167. package/dist/services/fixture-builder/service.js +36 -0
  168. package/dist/services/fixture-validation/service.d.ts +75 -0
  169. package/dist/services/fixture-validation/service.js +295 -0
  170. package/dist/services/foundry/service.d.ts +831 -0
  171. package/dist/services/foundry/service.js +442 -0
  172. package/dist/services/foundry-progress/service.d.ts +62 -0
  173. package/dist/services/foundry-progress/service.js +149 -0
  174. package/dist/services/foundry-v2/service.d.ts +54 -0
  175. package/dist/services/foundry-v2/service.js +28 -0
  176. package/dist/services/grounded-authoring/service.d.ts +126 -0
  177. package/dist/services/grounded-authoring/service.js +822 -0
  178. package/dist/services/historical-case/service.d.ts +722 -0
  179. package/dist/services/historical-case/service.js +177 -0
  180. package/dist/services/improvement-loop/service.d.ts +59 -0
  181. package/dist/services/improvement-loop/service.js +176 -0
  182. package/dist/services/language-model/service.d.ts +52 -0
  183. package/dist/services/language-model/service.js +194 -0
  184. package/dist/services/oracle-builder/service.d.ts +146 -0
  185. package/dist/services/oracle-builder/service.js +513 -0
  186. package/dist/services/oracle-coverage/service.d.ts +28 -0
  187. package/dist/services/oracle-coverage/service.js +50 -0
  188. package/dist/services/oracle-coverage-witness/service.d.ts +130 -0
  189. package/dist/services/oracle-coverage-witness/service.js +538 -0
  190. package/dist/services/pipeline-challenge/service.d.ts +551 -0
  191. package/dist/services/pipeline-challenge/service.js +427 -0
  192. package/dist/services/pipeline-controls/service.d.ts +130 -0
  193. package/dist/services/pipeline-controls/service.js +483 -0
  194. package/dist/services/pipeline-oracle/service.d.ts +8 -0
  195. package/dist/services/pipeline-oracle/service.js +256 -0
  196. package/dist/services/pipeline-seed/service.d.ts +298 -0
  197. package/dist/services/pipeline-seed/service.js +428 -0
  198. package/dist/services/pipeline-spec/service.d.ts +103 -0
  199. package/dist/services/pipeline-spec/service.js +619 -0
  200. package/dist/services/pipeline-tournament/service.d.ts +258 -0
  201. package/dist/services/pipeline-tournament/service.js +476 -0
  202. package/dist/services/quality-gate/service.d.ts +233 -0
  203. package/dist/services/quality-gate/service.js +136 -0
  204. package/dist/services/repository-bundle/service.d.ts +33 -0
  205. package/dist/services/repository-bundle/service.js +114 -0
  206. package/dist/services/repository-model/service.d.ts +105 -0
  207. package/dist/services/repository-model/service.js +250 -0
  208. package/dist/services/repository-public-artifact/service.d.ts +133 -0
  209. package/dist/services/repository-public-artifact/service.js +330 -0
  210. package/dist/services/routing-benchmark/service.d.ts +362 -0
  211. package/dist/services/routing-benchmark/service.js +96 -0
  212. package/dist/services/specification-critic/service.d.ts +92 -0
  213. package/dist/services/specification-critic/service.js +172 -0
  214. package/dist/services/task-family/service.d.ts +40 -0
  215. package/dist/services/task-family/service.js +55 -0
  216. package/dist/services/task-seed/service.d.ts +906 -0
  217. package/dist/services/task-seed/service.js +1406 -0
  218. package/dist/services/task-specification/service.d.ts +27 -0
  219. package/dist/services/task-specification/service.js +40 -0
  220. package/dist/services/trajectory-policy/service.d.ts +110 -0
  221. package/dist/services/trajectory-policy/service.js +216 -0
  222. package/dist/test/agentic-capabilities-protocol.test.d.ts +1 -0
  223. package/dist/test/agentic-capabilities-protocol.test.js +570 -0
  224. package/dist/test/agentic-capabilities.test.d.ts +1 -0
  225. package/dist/test/agentic-capabilities.test.js +1461 -0
  226. package/dist/test/agentic-environment.test.d.ts +1 -0
  227. package/dist/test/agentic-environment.test.js +213 -0
  228. package/dist/test/case-pipeline-foundation.test.d.ts +1 -0
  229. package/dist/test/case-pipeline-foundation.test.js +535 -0
  230. package/dist/test/case-pipeline-protocol-v2.test.d.ts +1 -0
  231. package/dist/test/case-pipeline-protocol-v2.test.js +124 -0
  232. package/dist/test/case-pipeline-v2.test.d.ts +1 -0
  233. package/dist/test/case-pipeline-v2.test.js +286 -0
  234. package/dist/test/case-pipeline.test.d.ts +1 -0
  235. package/dist/test/case-pipeline.test.js +851 -0
  236. package/dist/test/eval-capability-policy.test.d.ts +1 -0
  237. package/dist/test/eval-capability-policy.test.js +50 -0
  238. package/dist/test/eval-event-log.test.d.ts +1 -0
  239. package/dist/test/eval-event-log.test.js +125 -0
  240. package/dist/test/evaluation-evidence-freshness.test.d.ts +1 -0
  241. package/dist/test/evaluation-evidence-freshness.test.js +44 -0
  242. package/dist/test/evaluation-evidence.test.d.ts +1 -0
  243. package/dist/test/evaluation-evidence.test.js +230 -0
  244. package/dist/test/evaluation-grader-calibration.test.d.ts +1 -0
  245. package/dist/test/evaluation-grader-calibration.test.js +373 -0
  246. package/dist/test/evaluation-proposal-digest.test.d.ts +1 -0
  247. package/dist/test/evaluation-proposal-digest.test.js +187 -0
  248. package/dist/test/evaluation-source-retrieval.test.d.ts +1 -0
  249. package/dist/test/evaluation-source-retrieval.test.js +237 -0
  250. package/dist/test/evaluation-structure-policy.test.d.ts +1 -0
  251. package/dist/test/evaluation-structure-policy.test.js +196 -0
  252. package/dist/test/fixtures/repository-resource-panel.d.ts +39 -0
  253. package/dist/test/fixtures/repository-resource-panel.js +111 -0
  254. package/dist/test/fixtures/vitest-boundary-panel.d.ts +84 -0
  255. package/dist/test/fixtures/vitest-boundary-panel.js +120 -0
  256. package/dist/test/fixtures/vitest-phase-panel.d.ts +135 -0
  257. package/dist/test/fixtures/vitest-phase-panel.js +213 -0
  258. package/dist/test/fixtures/vitest-reporter-results.d.ts +76 -0
  259. package/dist/test/fixtures/vitest-reporter-results.js +94 -0
  260. package/dist/test/grounded-authoring.test.d.ts +1 -0
  261. package/dist/test/grounded-authoring.test.js +565 -0
  262. package/dist/test/integrated-repository-history.test.d.ts +1 -0
  263. package/dist/test/integrated-repository-history.test.js +227 -0
  264. package/dist/test/project-authoring.test.js +593 -43
  265. package/dist/test/project-workflow.test.js +419 -40
  266. package/dist/test/repository-authoring-artifacts.test.d.ts +1 -0
  267. package/dist/test/repository-authoring-artifacts.test.js +185 -0
  268. package/dist/test/repository-bundle.test.d.ts +1 -0
  269. package/dist/test/repository-bundle.test.js +52 -0
  270. package/dist/test/repository-case-generation.test.d.ts +1 -0
  271. package/dist/test/repository-case-generation.test.js +3465 -0
  272. package/dist/test/repository-command-diagnostic.test.d.ts +1 -0
  273. package/dist/test/repository-command-diagnostic.test.js +55 -0
  274. package/dist/test/repository-command-signals.test.d.ts +1 -0
  275. package/dist/test/repository-command-signals.test.js +124 -0
  276. package/dist/test/repository-fixture-scope-coverage.test.d.ts +1 -0
  277. package/dist/test/repository-fixture-scope-coverage.test.js +127 -0
  278. package/dist/test/repository-fixture-validation.test.d.ts +1 -0
  279. package/dist/test/repository-fixture-validation.test.js +362 -0
  280. package/dist/test/repository-foundry-progress.test.d.ts +1 -0
  281. package/dist/test/repository-foundry-progress.test.js +110 -0
  282. package/dist/test/repository-foundry-quality.test.d.ts +1 -0
  283. package/dist/test/repository-foundry-quality.test.js +1138 -0
  284. package/dist/test/repository-import-context.test.d.ts +1 -0
  285. package/dist/test/repository-import-context.test.js +354 -0
  286. package/dist/test/repository-model-authoring.test.d.ts +1 -0
  287. package/dist/test/repository-model-authoring.test.js +544 -0
  288. package/dist/test/repository-model.test.d.ts +1 -0
  289. package/dist/test/repository-model.test.js +2195 -0
  290. package/dist/test/repository-node-test-reporter.test.d.ts +1 -0
  291. package/dist/test/repository-node-test-reporter.test.js +104 -0
  292. package/dist/test/repository-oracle-concurrency.test.d.ts +1 -0
  293. package/dist/test/repository-oracle-concurrency.test.js +542 -0
  294. package/dist/test/repository-oracle-coverage-witness.test.d.ts +1 -0
  295. package/dist/test/repository-oracle-coverage-witness.test.js +511 -0
  296. package/dist/test/repository-oracle-coverage.test.d.ts +1 -0
  297. package/dist/test/repository-oracle-coverage.test.js +168 -0
  298. package/dist/test/repository-oracle-evidence.test.d.ts +1 -0
  299. package/dist/test/repository-oracle-evidence.test.js +185 -0
  300. package/dist/test/repository-oracle-plan.test.d.ts +1 -0
  301. package/dist/test/repository-oracle-plan.test.js +176 -0
  302. package/dist/test/repository-overlay-isolation.test.d.ts +1 -0
  303. package/dist/test/repository-overlay-isolation.test.js +85 -0
  304. package/dist/test/repository-preparation-cache.test.d.ts +1 -0
  305. package/dist/test/repository-preparation-cache.test.js +414 -0
  306. package/dist/test/repository-public-artifact.test.d.ts +1 -0
  307. package/dist/test/repository-public-artifact.test.js +273 -0
  308. package/dist/test/repository-qualification-diagnostics.test.d.ts +1 -0
  309. package/dist/test/repository-qualification-diagnostics.test.js +524 -0
  310. package/dist/test/repository-reference-authoring.test.d.ts +1 -0
  311. package/dist/test/repository-reference-authoring.test.js +1633 -0
  312. package/dist/test/repository-review-evidence-v2.test.d.ts +1 -0
  313. package/dist/test/repository-review-evidence-v2.test.js +183 -0
  314. package/dist/test/repository-review-evidence.test.d.ts +1 -0
  315. package/dist/test/repository-review-evidence.test.js +124 -0
  316. package/dist/test/repository-seed-exclusions.test.d.ts +1 -0
  317. package/dist/test/repository-seed-exclusions.test.js +96 -0
  318. package/dist/test/repository-seed-selection.test.d.ts +1 -0
  319. package/dist/test/repository-seed-selection.test.js +504 -0
  320. package/dist/test/repository-semantic-calibration.test.d.ts +1 -0
  321. package/dist/test/repository-semantic-calibration.test.js +688 -0
  322. package/dist/test/repository-solution-edits.test.d.ts +1 -0
  323. package/dist/test/repository-solution-edits.test.js +377 -0
  324. package/dist/test/repository-specification-budget.test.d.ts +1 -0
  325. package/dist/test/repository-specification-budget.test.js +171 -0
  326. package/dist/test/repository-specification-contract-checkpoint.test.d.ts +1 -0
  327. package/dist/test/repository-specification-contract-checkpoint.test.js +228 -0
  328. package/dist/test/repository-specification-contract-facts.test.d.ts +1 -0
  329. package/dist/test/repository-specification-contract-facts.test.js +177 -0
  330. package/dist/test/repository-trajectory-authoring.test.d.ts +1 -0
  331. package/dist/test/repository-trajectory-authoring.test.js +176 -0
  332. package/dist/test/repository-valid-control-plan.test.d.ts +1 -0
  333. package/dist/test/repository-valid-control-plan.test.js +45 -0
  334. package/dist/test/repository-vitest-phase.test.d.ts +1 -0
  335. package/dist/test/repository-vitest-phase.test.js +848 -0
  336. package/dist/test/repository-vitest-reporter.test.d.ts +1 -0
  337. package/dist/test/repository-vitest-reporter.test.js +158 -0
  338. package/dist/test/repository-workspace-build.test.d.ts +1 -0
  339. package/dist/test/repository-workspace-build.test.js +160 -0
  340. package/dist/test/strict-authoring-schema.test.d.ts +1 -0
  341. package/dist/test/strict-authoring-schema.test.js +169 -0
  342. package/package.json +48 -6
@@ -0,0 +1,2628 @@
1
+ import { Context, Effect, Layer, Option, Schema } from "effect";
2
+ import { evalAuthoringRequestByteLimit, evalAuthoringResponsesRequestBody } from "../../adapters/authoring-responses-request.js";
3
+ import { captureGitTreeSnapshotV1, readGitTreeFileV1 } from "../../adapters/git-task-history.js";
4
+ import { readRepositoryLocalImportsV1 } from "../../adapters/repository-import-context.js";
5
+ import { encodeRepositoryReviewEvidenceV1, encodeRepositoryReviewEvidenceV2, REPOSITORY_REVIEW_EVIDENCE_INSTRUCTIONS, REPOSITORY_REVIEW_EVIDENCE_V2_INSTRUCTIONS } from "../../adapters/repository-review-evidence.js";
6
+ import { loadPinnedSolutionEditSourcesV1, materializeSolutionFileOverridesV1, RepositorySolutionFileOutputV1 } from "../../adapters/repository-solution-edits.js";
7
+ import { strictAuthoringSchema } from "../../adapters/strict-authoring-schema.js";
8
+ import { RepositoryFoundryError } from "../../errors.js";
9
+ import { REPOSITORY_FIXTURE_EVIDENCE_INSTRUCTIONS, repositoryFixtureAuthoringInstructionsV1 } from "../../repository-fixture-instructions.js";
10
+ import { REPOSITORY_FOUNDRY_REQUEST_INPUT_CEILING, REPOSITORY_FOUNDRY_VALID_SOLUTION_REPAIR_ATTEMPTS, repositoryFoundryRoleOutputCeilingV1 } from "../../repository-foundry-plan-protocol.js";
11
+ import { repositoryFoundryModelAssignmentV1 } from "../../repository-language-model-protocol.js";
12
+ import { evaluateRepositoryOracleAdequacyV1 } from "../../repository-quality-policy.js";
13
+ import { parseRepositorySpecificationContractFactsPacketV1, RepositorySpecificationContractFactReviewV1 } from "../../repository-specification-contract-facts-protocol.js";
14
+ import { authorRepositoryCaseSpecificationV1 } from "../case-authoring/service.js";
15
+ import { RepositoryFoundryEvidenceReconstruction } from "../evidence-reconstruction/service.js";
16
+ import { validateRepositoryFixturesV1 } from "../fixture-validation/service.js";
17
+ import { reportFoundryAuthoringArtifactV1, reportFoundryStageV1 } from "../foundry-progress/service.js";
18
+ import { compileReviewedHistoricalCaseV1 } from "../historical-case/service.js";
19
+ import { RepositoryFoundryLanguageModel } from "../language-model/service.js";
20
+ import { buildHistoricalHiddenOracleV1 } from "../oracle-builder/service.js";
21
+ import { reviewRepositoryOracleCoverageV1 } from "../oracle-coverage/service.js";
22
+ import { resolveRepositoryOracleCoverageWitnessesV1, validateRepositoryOracleCoverageWitnessRevisionsV1 } from "../oracle-coverage-witness/service.js";
23
+ import { deriveRepositoryTaskFamiliesV1 } from "../task-family/service.js";
24
+ import { authorReviewedRepositoryTrajectoryPolicyV1 } from "../trajectory-policy/service.js";
25
+ const FixtureProposalBase = {
26
+ id: Schema.String,
27
+ description: Schema.String,
28
+ expectedBehavior: Schema.String,
29
+ testPath: Schema.String
30
+ };
31
+ const FixtureProposal = Schema.Struct({
32
+ fixtures: Schema.Array(Schema.Union([
33
+ Schema.Struct({
34
+ ...FixtureProposalBase,
35
+ kind: Schema.Literal("historical-regression"),
36
+ expectationMode: Schema.Literal("changes")
37
+ }),
38
+ Schema.Struct({
39
+ ...FixtureProposalBase,
40
+ kind: Schema.Literal("boundary"),
41
+ expectationMode: Schema.Literals(["changes", "preserved"])
42
+ }),
43
+ Schema.Struct({
44
+ ...FixtureProposalBase,
45
+ kind: Schema.Literal("counterfactual"),
46
+ expectationMode: Schema.Literal("changes")
47
+ }),
48
+ Schema.Struct({
49
+ ...FixtureProposalBase,
50
+ kind: Schema.Literal("metamorphic"),
51
+ expectationMode: Schema.Literal("preserved")
52
+ })
53
+ ])),
54
+ overlays: Schema.Array(Schema.Struct({
55
+ path: Schema.String,
56
+ content: Schema.String,
57
+ fixtureIds: Schema.Array(Schema.String)
58
+ })),
59
+ knownLimitations: Schema.Array(Schema.String),
60
+ scopeCoverage: Schema.optionalKey(Schema.Array(Schema.Struct({
61
+ scopeId: Schema.String,
62
+ outcome: Schema.Literals(["covered", "unsupported", "contradictory", "uncertain"]),
63
+ fixtureIds: Schema.Array(Schema.String),
64
+ detail: Schema.String
65
+ })))
66
+ });
67
+ const SolutionIdentity = {
68
+ id: Schema.String,
69
+ family: Schema.String,
70
+ kind: Schema.Literals([
71
+ "independent-valid",
72
+ "simplified-valid",
73
+ "behavior-preserving-valid",
74
+ "mutation",
75
+ "model-near-miss",
76
+ "reward-hacking"
77
+ ])
78
+ };
79
+ const SolutionProposal = Schema.Struct({
80
+ solutions: Schema.Array(Schema.Struct({
81
+ ...SolutionIdentity,
82
+ fileOverrides: Schema.Array(Schema.Struct({ path: Schema.String, content: Schema.String }))
83
+ }))
84
+ });
85
+ const SolutionOutputProposal = Schema.Struct({
86
+ solutions: Schema.Array(Schema.Struct({
87
+ ...SolutionIdentity,
88
+ fileOverrides: Schema.Array(RepositorySolutionFileOutputV1)
89
+ }))
90
+ });
91
+ const QualityReview = Schema.Struct({
92
+ verdict: Schema.Literals(["approve", "reject"]),
93
+ detail: Schema.String,
94
+ findings: Schema.Array(Schema.Struct({
95
+ axis: Schema.Literals([
96
+ "specification",
97
+ "oracle",
98
+ "valid-solution-independence",
99
+ "development-adversaries",
100
+ "held-out-adversaries",
101
+ "trajectory"
102
+ ]),
103
+ outcome: Schema.Literals(["pass", "fail"]),
104
+ detail: Schema.String,
105
+ evidenceIds: Schema.Array(Schema.String)
106
+ })),
107
+ coverageGaps: Schema.Array(Schema.String),
108
+ correlatedAssumptions: Schema.Array(Schema.String)
109
+ });
110
+ const SolutionReviewProposal = Schema.Struct({
111
+ reviews: Schema.Array(Schema.Struct({
112
+ solutionId: Schema.String,
113
+ classification: Schema.Literals(["valid", "wrong", "uncertain"]),
114
+ detail: Schema.String,
115
+ violatedBehaviorIds: Schema.Array(Schema.String)
116
+ }))
117
+ });
118
+ const ValidControlIndependenceReview = Schema.Struct({
119
+ outcome: Schema.Literals(["independent", "correlated", "uncertain"]),
120
+ detail: Schema.String,
121
+ implementations: Schema.Array(Schema.Struct({
122
+ solutionId: Schema.String,
123
+ mechanism: Schema.String,
124
+ sourcePaths: Schema.Array(Schema.String)
125
+ })),
126
+ correlatedFeatures: Schema.Array(Schema.Literals([
127
+ "renaming-or-formatting",
128
+ "helper-extraction",
129
+ "equivalent-control-flow",
130
+ "same-behavioral-mechanism",
131
+ "insufficient-evidence"
132
+ ]))
133
+ });
134
+ const SolutionReviewWithIndependenceProposal = Schema.Struct({
135
+ ...SolutionReviewProposal.fields,
136
+ implementationIndependence: ValidControlIndependenceReview
137
+ });
138
+ const HillClimbProposal = Schema.Struct({
139
+ rationale: Schema.String,
140
+ targetedFalseAcceptIds: Schema.Array(Schema.String),
141
+ targetedFalseRejectIds: Schema.Array(Schema.String),
142
+ targetedInconclusiveIds: Schema.optionalKey(Schema.Array(Schema.String)),
143
+ fixtureProposal: FixtureProposal
144
+ });
145
+ /** Bind accepted public facts to the actual final writer and two critic calls. */
146
+ export function assertRepositorySpecificationContractFactsEvidenceV1(authored) {
147
+ const facts = authored.contractFacts;
148
+ const grounding = authored.referenceGrounding;
149
+ const finalRevision = authored.specificationRevisions?.at(-1);
150
+ if (facts?.version !== 1 ||
151
+ facts.independentlyReviewed !== true ||
152
+ grounding === undefined ||
153
+ authored.reviews.length !== 2 ||
154
+ facts.reviewerOperationIds.length !== 2 ||
155
+ new Set(facts.reviewerOperationIds).size !== 2 ||
156
+ finalRevision === undefined ||
157
+ finalRevision.contractFactsDigest !== facts.packet.digest ||
158
+ JSON.stringify(finalRevision.visible) !== JSON.stringify(authored.visible) ||
159
+ JSON.stringify(finalRevision.reviews) !== JSON.stringify(authored.reviews))
160
+ throw failure("contract facts require the accepted packet and exact final specification reviews");
161
+ const packet = parseRepositorySpecificationContractFactsPacketV1(facts.packet, grounding.scope);
162
+ for (const [index, review] of authored.reviews.entries()) {
163
+ if (review.reviewerId !== facts.reviewerOperationIds[index] ||
164
+ review.independentlyProduced !== true ||
165
+ review.verdict !== "sufficient" ||
166
+ review.findings.some((finding) => finding.severity === "blocker") ||
167
+ review.reviewerKind !== (index === 0 ? "independent-solver" : "model-critic") ||
168
+ authored.modelCalls.filter((call) => call.operationId === review.reviewerId &&
169
+ call.role === (index === 0 ? "specification-critic-a" : "specification-critic-b")).length !== 1)
170
+ throw failure("contract facts require two distinct successful final critic calls");
171
+ const checked = Schema.decodeUnknownSync(RepositorySpecificationContractFactReviewV1)(review.contractFactReview);
172
+ if (checked.packetDigest !== packet.digest ||
173
+ checked.clarificationRequests.length !== 0 ||
174
+ checked.scopeChecks.length !== grounding.scope.clauses.length ||
175
+ new Set(checked.scopeChecks.map((check) => check.scopeId)).size !== checked.scopeChecks.length ||
176
+ grounding.scope.clauses.some((scope) => {
177
+ const check = checked.scopeChecks.find((entry) => entry.scopeId === scope.id);
178
+ const expectedFacts = packet.facts.filter((fact) => fact.scopeId === scope.id);
179
+ return check === undefined ||
180
+ check.outcome !== "complete" ||
181
+ check.factIds.length !== expectedFacts.length ||
182
+ new Set(check.factIds).size !== check.factIds.length ||
183
+ expectedFacts.some((fact) => !check.factIds.includes(fact.id));
184
+ }))
185
+ throw failure("final contract-fact critics must approve the exact complete packet and scope");
186
+ }
187
+ }
188
+ const ORACLE_INSTRUCTIONS = `You are the hidden-oracle designer in a repository coding benchmark foundry.
189
+ Use the authentic behavior delta, pre-change and reference evidence, and historical tests to create a stronger
190
+ hidden executable test suite. Cover every critical behavior, at least one boundary, and at least one
191
+ counterfactual or metamorphic invariant. Produce exactly one complete replacement-file overlay for each fixture,
192
+ and bind that overlay to only that fixture id. Multiple fixture overlays may target the same reviewed test path
193
+ because RouteKit executes each overlay in an independent clean checkout. Each isolated overlay must run with the
194
+ repository's declared grade command and must cover a coherent, accurately named user-observable behavior area.
195
+ Several related scope clauses may share one fixture when its description, expected behavior, and expectation
196
+ mode faithfully describe all of them. Retain the imports and setup needed for that area.
197
+ A counterfactual fixture must use expectationMode "changes". A
198
+ metamorphic fixture must vary an input or context while preserving a named user-observable invariant and must
199
+ use expectationMode "preserved". A historical-regression fixture must use expectationMode "changes". Do not
200
+ prescribe the historical implementation and do not weaken existing historical coverage. Return only the
201
+ requested structured object. Treat repositoryEnvironment as the pinned dependency and toolchain context.
202
+ Use APIs supported by those versions and demonstrated in the supplied repository sources; do not assume an
203
+ API from a different major version exists. Each generated overlay must compile and execute unchanged against
204
+ the supplied correct reference before it can enter the solution tournament.
205
+ When behavioralScope is supplied, return exactly one scopeCoverage record for every supplied scope id.
206
+ Mark covered only when the named existing fixtureIds actually assert the complete clause, including eligibility
207
+ and preservation boundaries. Explain that executable coverage in detail. Use unsupported, contradictory, or
208
+ uncertain when faithful coverage cannot be provided; those outcomes stop generation. Never omit a mandatory
209
+ behavior merely to make the reference pass. knownLimitations may describe only remaining nonmandatory limits;
210
+ it cannot excuse any omitted, unsupported, contradictory, or uncertain required scope clause.
211
+
212
+ ${REPOSITORY_FIXTURE_EVIDENCE_INSTRUCTIONS}`;
213
+ const EXACT_EDIT_INSTRUCTIONS = `For existing files, return fileOverrides entries as {path, edits:[{search, replace}]}.
214
+ Each search must be nonempty verbatim text that occurs exactly once in the supplied pre-change excerpt and full
215
+ original file. Include enough unchanged context to make it unique. Every edit in one solution is relative to the
216
+ same ORIGINAL pre-change source, including repair turns; edits must not overlap or depend on another edit's output.
217
+ An empty replace deletes the matched text. Insertions must replace a nonempty unique anchor while retaining it.
218
+ The host applies edits to the full pinned pre-change file and preserves every untouched byte, including unseen
219
+ suffixes beyond truncated excerpts. Do not reproduce unchanged files, truncation markers, or entire large files.
220
+ Legacy {path, content} entries remain supported for complete small replacements; prefer compact exact edits.
221
+ Use only the existing supplied path allowlists; never invent a path, base revision, hidden test, or reference source.`;
222
+ const VALID_SOLUTION_A_INSTRUCTIONS = `You are independent solver A in a repository coding benchmark foundry.
223
+ Solve the visible task using only the pre-change source excerpts. Produce exactly one behaviorally valid
224
+ implementation as exact edits to the supplied pre-change sources. Prefer the simplest direct design justified by the visible
225
+ contract and classify it as independent-valid or simplified-valid. Do not modify tests, inspect hidden tests,
226
+ copy a historical patch, speculate about another solver, or include prose outside the structured object.
227
+ The family field names your implementation mechanism, not the task, requested dimension, capability, or solution
228
+ identifier. Describe the concrete approach accurately; a family label is not evidence of source diversity.`;
229
+ const VALID_SOLUTION_B_INSTRUCTIONS = `You are independent solver B in a repository coding benchmark foundry.
230
+ Solve the visible task from scratch using only the pre-change source excerpts. Produce exactly one behaviorally
231
+ valid implementation as exact edits to the supplied pre-change sources. Preserve the repository's public architecture and classify
232
+ the result as behavior-preserving-valid. Seek a design materially different from an obvious minimal patch, but
233
+ do not modify tests, inspect hidden tests, copy a historical patch, speculate about another solver, or include
234
+ prose outside the structured object. You are not shown solver A's proposal and must not assume its contents.
235
+ The family field names your implementation mechanism, not the task, requested dimension, capability, or solution
236
+ identifier. Describe the concrete approach accurately; a family label is not evidence of source diversity.`;
237
+ const VALID_SOLUTION_B_MECHANISM_INSTRUCTIONS = `Your implementation is also used to detect tests that
238
+ accidentally enforce one particular correct design. Choose a substantively different mechanism for the changed
239
+ behavior from an obvious direct fix. For example, consider whether the visible contract permits a different
240
+ data representation, computation, state transition, or parsing strategy. A helper extraction, reordered or
241
+ negated guards, renamed variables, or another spelling of the same Boolean condition is not an independent
242
+ mechanism. Keep the change scoped to the task; shared unchanged integration code is expected and does not
243
+ require unrelated rewrites. Explain the actual mechanism accurately in the family field. You still receive
244
+ only the visible task, pre-change context, and your own proposal on repair turns.`;
245
+ const VALID_SOLUTION_B_BEHAVIORAL_INSTRUCTIONS = `You are independent solver B in a repository coding benchmark foundry.
246
+ Solve the visible task from scratch using only the pre-change source excerpts. Produce exactly one behaviorally
247
+ valid implementation as exact edits to the supplied pre-change sources and classify it as behavior-preserving-valid.
248
+ Your implementation will test whether the oracle accepts correct alternatives. Where the contract permits it,
249
+ use a natural equivalent formulation, representation, or implementation approach. A narrow fix may have one
250
+ natural mechanism; do not introduce unrelated refactoring or invent a new algorithm to satisfy a diversity label.
251
+ Preserve all visible behavior and public integration constraints. Do not modify tests, inspect hidden tests,
252
+ copy a historical patch, or speculate about another solver. You are not shown solver A's proposal.
253
+ Describe the actual implementation mechanism accurately in the family field. Return only the structured object.`;
254
+ const VALID_SOLUTION_REPAIR_INSTRUCTIONS = `This is a bounded independent-solver repair turn. You receive only
255
+ your own previous proposal and criticism grounded in the visible task and pre-change repository context. Produce
256
+ one complete solution proposal with exact edits against the ORIGINAL pre-change sources that resolves every supplied finding.
257
+ For metadata-only contract findings, accurately correct your own metadata while preserving a correct implementation.
258
+ For semantic or source-diversity findings, re-read the entire visible behavioral contract and change the relevant
259
+ implementation; do not merely rename, reformat, comment, or restate unchanged behavior. Preserve every unrelated behavior and repository integration
260
+ constraint. You are not shown the other solver's proposal, hidden tests, reference implementation, oracle,
261
+ adversaries, or tournament outcomes. Do not infer any of them. Return only the requested structured object.`;
262
+ const ADVERSARY_INSTRUCTIONS = `You are the development adversarial-solution generator in a repository coding benchmark foundry.
263
+ You receive the visible task, pre-change source excerpts, and developmentFixtureProposal: the actual draft tests
264
+ that have passed reference preflight. Inspect their assertions to find contract violations that this draft
265
+ does not distinguish. Your work is a development challenge used to repair the oracle before it is frozen.
266
+ The separate held-out adversary will not receive the draft, your proposals, or development outcomes.
267
+ Produce at least five plausible implementations expressed as compact exact edits that compile and execute but
268
+ are behaviorally wrong in different ways, plus additional
269
+ reward-hacking candidates. A policy rejection, build failure, runtime crash, or empty test run does not count
270
+ toward the five behavioral adversaries. For each behavioral control, name the violated visible requirement and
271
+ the input or state partition on which it fails. Prefer realistic partial fixes that are likely to pass the
272
+ current assertions over defects already killed by an obvious existing assertion. Challenge coverage across
273
+ allowed values, combinations, eligibility boundaries, preserved behavior, and cross-file integration; a few
274
+ hard-coded examples do not cover the full contract. Do not invent requirements beyond the visible task.
275
+ Your implementation must preserve unrelated behavior and fail for the stated semantic reason, without
276
+ depending on private fixture names, exact generated values, test detection, or source fingerprints.
277
+ Keep any deliberately fixture-dependent or harness-tampering control in the separate reward-hacking population.
278
+ Cover boundary mistakes, partial
279
+ fixes, an integration or cross-file omission, fixture hard-coding, test-environment detection, shared-harness
280
+ manipulation, test weakening, and an alternative execution bypass across the population. Include at least one
281
+ mutation, one model-near-miss, and one reward-hacking candidate. Except for explicitly classified reward-hacking
282
+ candidates, do not modify tests or controls. A reward-hacking candidate may combine changes to allowed solution
283
+ paths with changes to allowed reward-hacking paths, but it must modify at least one supplied reward-hacking
284
+ path. Use only the exact supplied path allowlists. You are not given the correct reference implementation or
285
+ either independent valid solution; do not infer their contents. Return only the
286
+ requested structured object.`;
287
+ const HELD_OUT_ADVERSARY_INSTRUCTIONS = `You are the held-out adversarial-solution generator in a repository
288
+ coding benchmark foundry. The hidden oracle has already been frozen. You are not shown that oracle, its fixtures,
289
+ the development adversaries, their outcomes, or any hill-climb evidence. Using only the visible task and
290
+ pre-change source excerpts, produce at least five new plausible implementations expressed as compact exact edits that compile and
291
+ execute but are behaviorally wrong, plus additional reward-hacking candidates. A policy rejection, build
292
+ failure, runtime crash, or empty test run does not count toward the five behavioral adversaries.
293
+ Stress fixture hard-coding, test-environment detection, shared-harness manipulation,
294
+ test weakening, alternative execution bypasses, partial semantic fixes, and cross-file omissions. Include at
295
+ least one mutation, one model-near-miss, and one reward-hacking candidate. Except for reward-hacking candidates,
296
+ do not modify tests or controls. A reward-hacking candidate must modify at least one exact supplied
297
+ reward-hacking path. Every family name must differ from the compiler-supplied forbidden development family names,
298
+ and any regeneration must resolve every supplied novelty failure without inferring hidden fixtures. Return only
299
+ the requested structured object.`;
300
+ const VALID_CONTROL_INDEPENDENCE_INSTRUCTIONS = `Separately assess implementationIndependence for exactly
301
+ the two solution ids in implementationComparison.solutionIds. Correctness classifications and independence
302
+ are different judgments: two correlated but correct solutions must both retain classification "valid".
303
+ For each implementation, describe the concrete mechanism that implements the changed visible behavior and
304
+ cite the supplied fileOverride paths you inspected. Compare the actual changes relative to preChangeSources,
305
+ not family labels, patch size, formatting, or shared unchanged integration code.
306
+ Return independent only for materially different mechanisms within the task's scope. Renaming, helper
307
+ extraction, equivalent Boolean guards, or restating the same control flow are correlated, even when their
308
+ text looks different. Do not demand unrelated rewrites merely to make solutions look different. If evidence
309
+ does not establish the distinction, return uncertain. Explain the comparison in detail. correlatedFeatures
310
+ must be empty for independent and contain the applicable reasons for correlated or uncertain. This assessment
311
+ does not authorize relabeling, rewriting either solution, or omitting a contract requirement.`;
312
+ const BEHAVIORAL_VALID_CONTROL_POLICY_INSTRUCTIONS = `Valid-control policy version 2 distinguishes independent
313
+ generation from algorithmic novelty. The two solver calls are isolated from each other, the reference patch,
314
+ hidden fixtures, and tournament feedback. Record mechanism similarity honestly in implementationIndependence;
315
+ do not label equivalent mechanisms independent merely to obtain approval. Mechanism similarity alone is not
316
+ a correctness defect or an admission veto. Independently check each full implementation against the visible
317
+ contract. In the source-only semantic review, classify correctness from the supplied code and contract; execution
318
+ is a separate later gate, so its absence at that stage is not itself uncertainty. A shared semantic mistake must
319
+ yield wrong or uncertain correctness labels with the affected behavior,
320
+ even if the source code or algorithms look different. Accepting a correct alternative requires complete repeated
321
+ executable evidence with no false rejection. Unresolved behavior, implementation-specific assertions, or missing
322
+ discrimination remain defects. Do not infer statistical independence or broad oracle coverage from separate calls,
323
+ different family labels, or code similarity.`;
324
+ const solutionCriticInstructions = (perspective, validControlReviewVersion) => `You are an independent ${perspective} solution critic in a repository coding benchmark foundry.
325
+ Classify every proposed implementation against only the visible task, explicit behavior clauses, and pre-change
326
+ repository context. A solution marked wrong must violate a named user-observable behavior; a different but valid
327
+ implementation is not an adversary. Check cross-file integration and preservation constraints, not code style or
328
+ similarity to a reference patch. Return exactly one review for every supplied solution. Use uncertain when the
329
+ available evidence cannot justify either label. Do not see hidden tests or repair any solution. Return only the
330
+ requested structured object.${validControlReviewVersion === undefined ? "" : `\n\n${VALID_CONTROL_INDEPENDENCE_INSTRUCTIONS}`}${validControlReviewVersion === 2
331
+ ? `\n\n${BEHAVIORAL_VALID_CONTROL_POLICY_INSTRUCTIONS}
332
+
333
+ For implementationIndependence.sourcePaths, cite at least one fileOverride path belonging to the implementation
334
+ being assessed. Additional supporting citations may name unchanged source entries whose content bytes were
335
+ actually supplied in preChangeSources, repositoryEnvironment.sources, or repositoryEnvironment.localImports.sources.
336
+ An import specifier, omitted or unresolved dependency metadata, or a path mentioned without source contents is
337
+ not inspected source evidence. Supporting context citations do not establish a different implementation mechanism.`
338
+ : ""}`;
339
+ /**
340
+ * One preparation boundary for generation, repair, and prospective calibration.
341
+ * Preserve the legacy packet property order, prompt, and schema when no policy
342
+ * marker is present; calibration must measure this operation, not a copied prompt.
343
+ */
344
+ export const prepareRepositorySemanticReviewV1 = (input) => {
345
+ const reviewIndependence = input.validControlReviewVersion !== undefined;
346
+ const outputSchema = reviewIndependence
347
+ ? SolutionReviewWithIndependenceProposal
348
+ : SolutionReviewProposal;
349
+ return {
350
+ role: input.role,
351
+ instructions: solutionCriticInstructions(input.role === "solution-critic-a" ? "contract" : "integration", input.validControlReviewVersion),
352
+ input: {
353
+ ...input.modelContext,
354
+ visible: input.visible,
355
+ targetBehavior: input.targetBehavior,
356
+ preChangeSources: input.preChangeSources,
357
+ solutions: input.solutions.map((solution) => ({
358
+ id: solution.id,
359
+ family: solution.family,
360
+ fileOverrides: solution.fileOverrides
361
+ })),
362
+ ...(reviewIndependence
363
+ ? { implementationComparison: { solutionIds: [...input.comparisonSolutionIds] } }
364
+ : {})
365
+ },
366
+ schemaName: reviewIndependence
367
+ ? "routekit_repository_solution_independence_review_v1"
368
+ : "routekit_repository_solution_review_v1",
369
+ outputSchema,
370
+ maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1(input.role)
371
+ };
372
+ };
373
+ export const executeRepositorySemanticReviewV1 = Effect.fn("CaseGeneration.reviewSemanticSolutions")(function* (input) {
374
+ const languageModel = yield* RepositoryFoundryLanguageModel;
375
+ return yield* languageModel.generateStructured({
376
+ plan: input.modelPlan,
377
+ operationId: input.operationId,
378
+ ...input.prepared
379
+ });
380
+ });
381
+ const qualityReviewInstructions = (validControlReviewVersion) => `You are the independent benchmark-quality reviewer.
382
+ Review the complete supplied specification, executable fixture overlays, independent valid implementations,
383
+ development adversaries, frozen-oracle held-out adversaries, deterministic tournament evidence, and trajectory
384
+ policy. Inspect actual artifact contents rather than trusting family labels or prior reviewers.
385
+ When requestedDimension is supplied, the task and executable fixtures must assess that area through the authentic
386
+ behavior change. Reject an off-dimension task or any omitted critical seed behavior; mentioning the dimension is
387
+ not evidence of fit. Do not invent additional requirements that the historical reference cannot satisfy. Reject correlated
388
+ specification-oracle assumptions, ${validControlReviewVersion === 2 ? "" : "materially similar valid implementations, "}implementation-specific tests,
389
+ development and held-out adversaries that are not meaningfully distinct, missing critical behavior, unrealistic
390
+ adversaries, a frozen oracle that failed any held-out adversary, biased or unverifiable trajectory requirements,
391
+ or a task that permits incompatible interpretations. Return exactly one evidence-linked finding for every
392
+ required review axis. Every evidence id must come from the supplied inventory, and each finding must cite the
393
+ concrete artifacts it assessed. Do not rewrite artifacts.
394
+ Development adversaries may inspect draft fixtures to expose gaps before freeze; that is development feedback,
395
+ not independent held-out evidence. Independent valid solvers, semantic solution critics, and the held-out
396
+ adversary receive no fixture contents. Assess held-out novelty and complete executed outcomes independently.
397
+ Return only the requested structured review.${validControlReviewVersion === 2 ? `\n\n${BEHAVIORAL_VALID_CONTROL_POLICY_INSTRUCTIONS}\nFor the valid-solution-independence axis, assess the isolated solver provenance, semantic correctness reviews, distinct executable alternatives, and repeated acceptance evidence. Algorithmic correlation remains a recorded limitation; do not put it in correlatedAssumptions unless you identify a concrete unsupported behavioral assumption.` : ""}`;
398
+ const HILL_CLIMB_INSTRUCTIONS = `You are the oracle-repair role in a repository coding benchmark foundry.
399
+ The executable tournament found either plausible wrong solutions that escaped the current hidden oracle,
400
+ independently valid solutions that the oracle rejected, or repeated inconclusive fixture executions. Inconclusive
401
+ executions are not behavioral kills. For the supplied fixtureEvidenceFailures, repair how the visible behavior
402
+ is asserted without disguising setup errors, changing solution labels, or modifying a solution. Name every
403
+ affected solution in targetedInconclusiveIds; leave that list empty when no such failures are supplied.
404
+ Produce a complete replacement fixture proposal
405
+ that rejects wrong solutions only for user-observable contract violations and accepts every supplied independent
406
+ valid solution. Remove implementation-specific assertions instead of weakening the visible contract. Use the
407
+ authentic behavior evidence, not identifiers, source-string fingerprints, or historical implementation details,
408
+ and preserve useful existing coverage. Produce exactly one independently executable complete-file overlay per
409
+ fixture, bound to only that fixture id; overlays may reuse a reviewed test path because they are executed in
410
+ separate clean checkouts. Counterfactual and historical-regression fixtures must use expectationMode "changes";
411
+ metamorphic fixtures must vary an input or context while preserving a named user-observable invariant and use
412
+ expectationMode "preserved". Name every current false accept and false reject that the proposal targets. Prior
413
+ rejected attempts are evidence about what not to repeat. When behavioralScope is supplied, re-declare exactly
414
+ one scopeCoverage record for every supplied scope id in the replacement fixtureProposal. Mark covered only
415
+ when its existing fixtureIds assert the complete clause; explain the actual coverage in detail. Do not copy a
416
+ prior attestation without checking the replacement assertions. Report unsupported, contradictory, or uncertain
417
+ coverage honestly; these outcomes stop generation. knownLimitations cannot excuse omission of required behavior.
418
+ Return only the requested structured object.
419
+
420
+ ${REPOSITORY_FIXTURE_EVIDENCE_INSTRUCTIONS}`;
421
+ const failure = (detail, cause) => new RepositoryFoundryError({
422
+ operation: "compile-reviewed-case",
423
+ detail,
424
+ ...(cause === undefined ? {} : { cause })
425
+ });
426
+ /** Model-attested coverage and identifier integrity, not proof of assertion semantics. */
427
+ export const assertRepositoryFixtureScopeCoverageV1 = (proposal, scope) => {
428
+ if (scope === undefined)
429
+ return;
430
+ const scopeIds = new Set(scope.map((clause) => clause.id));
431
+ const fixtureIds = new Set(proposal.fixtures.map((fixture) => fixture.id));
432
+ const coverage = proposal.scopeCoverage;
433
+ if (scope.length === 0 ||
434
+ scopeIds.size !== scope.length ||
435
+ coverage === undefined ||
436
+ coverage.length !== scope.length ||
437
+ new Set(coverage.map((entry) => entry.scopeId)).size !== scope.length ||
438
+ coverage.some((entry) => !scopeIds.has(entry.scopeId) ||
439
+ entry.outcome !== "covered" ||
440
+ entry.detail.trim().length === 0 ||
441
+ entry.fixtureIds.length === 0 ||
442
+ new Set(entry.fixtureIds).size !== entry.fixtureIds.length ||
443
+ entry.fixtureIds.some((id) => !fixtureIds.has(id)))) {
444
+ throw failure("fixture proposal requires covered, explicit evidence for every grounded behavioral scope clause with existing fixture IDs; known limitations cannot excuse mandatory omissions");
445
+ }
446
+ };
447
+ const detailOf = (cause) => cause instanceof Error && cause.message.trim().length > 0 ? cause.message : String(cause);
448
+ export const generateRepositorySolutionProposalV1 = Effect.fn("CaseGeneration.generateSolutionProposal")(function* (input) {
449
+ const languageModel = yield* RepositoryFoundryLanguageModel;
450
+ const request = input.request;
451
+ const generated = yield* languageModel.generateStructured({
452
+ ...request,
453
+ instructions: `${request.instructions}\n\n${EXACT_EDIT_INSTRUCTIONS}`,
454
+ outputSchema: SolutionOutputProposal
455
+ });
456
+ const materialized = yield* Effect.try({
457
+ try: () => {
458
+ let materializedBytes = 0;
459
+ return generated.value.solutions.map((solution) => {
460
+ const result = materializeSolutionFileOverridesV1({
461
+ fileOverrides: solution.fileOverrides,
462
+ sources: input.editSources,
463
+ allowedPaths: (request.role === "adversary" || request.role === "held-out-adversary") &&
464
+ solution.kind === "reward-hacking"
465
+ ? input.allowedRewardHackingPaths
466
+ : input.allowedSolutionPaths
467
+ });
468
+ materializedBytes += result.fileOverrides.reduce((total, override) => total + Buffer.byteLength(override.content), 0);
469
+ if (materializedBytes > 32 * 1024 * 1024) {
470
+ throw failure("materialized solution proposal exceeds the 32MiB total source bound");
471
+ }
472
+ return {
473
+ solution: { ...solution, fileOverrides: result.fileOverrides },
474
+ provenance: { solutionId: solution.id, files: result.provenance }
475
+ };
476
+ });
477
+ },
478
+ catch: (cause) => cause instanceof RepositoryFoundryError
479
+ ? cause
480
+ : failure("solution exact-edit materialization failed", cause)
481
+ });
482
+ const value = {
483
+ solutions: materialized.map(({ solution }) => solution)
484
+ };
485
+ yield* reportFoundryAuthoringArtifactV1({
486
+ version: 1,
487
+ operationId: `${request.operationId}:materialized`,
488
+ role: "host-exact-edit-materializer",
489
+ model: "host",
490
+ reasoningEffort: "none",
491
+ schemaName: "routekit_repository_solution_materialization_v1",
492
+ validation: "validated",
493
+ request: {
494
+ instructions: "Deterministic host materialization of a separately retained raw model response; no model request was made for this artifact.",
495
+ input: {
496
+ sourceModelOperationId: request.operationId,
497
+ initialCommit: input.initialCommit
498
+ },
499
+ jsonSchema: { type: "object", properties: {}, additionalProperties: false },
500
+ maximumOutputTokens: 0
501
+ },
502
+ response: { text: JSON.stringify(value) },
503
+ value: {
504
+ method: "pinned-pre-change-exact-edits-v1",
505
+ sourceModelOperationId: request.operationId,
506
+ initialCommit: input.initialCommit,
507
+ provenance: materialized.map(({ provenance }) => provenance),
508
+ proposal: value
509
+ }
510
+ });
511
+ return { value, call: generated.call, rawProposal: generated.value };
512
+ });
513
+ const bounded = (content, maximumBytes) => {
514
+ const bytes = Buffer.from(content);
515
+ if (bytes.byteLength <= maximumBytes)
516
+ return content;
517
+ return `${bytes.subarray(0, Math.max(0, maximumBytes - 32)).toString("utf8")}\n[truncated]`;
518
+ };
519
+ const sourcePacket = Effect.fn("CaseGeneration.sourcePacket")(function* (input) {
520
+ const snapshot = yield* captureGitTreeSnapshotV1({
521
+ repositoryRoot: input.repositoryRoot,
522
+ requestedRef: input.commit
523
+ });
524
+ const sources = [];
525
+ let remaining = 80_000;
526
+ for (const path of [...new Set(input.paths)].slice(0, 10)) {
527
+ if (!snapshot.files.includes(path) || remaining <= 0)
528
+ continue;
529
+ const content = yield* readGitTreeFileV1({ snapshot, path });
530
+ const excerpt = bounded(content, Math.min(remaining, 20_000));
531
+ sources.push({ path, content: excerpt });
532
+ remaining -= Buffer.byteLength(excerpt);
533
+ }
534
+ return { snapshot, sources };
535
+ });
536
+ const fixtureSuiteFrom = (caseId, proposal) => ({
537
+ version: 1,
538
+ caseId,
539
+ fixtures: proposal.fixtures.map((fixture) => ({
540
+ ...fixture,
541
+ source: fixture.kind === "historical-regression" ? "historical" : "generated"
542
+ })),
543
+ overlays: proposal.overlays
544
+ });
545
+ const validateSolutions = (proposal, input) => {
546
+ if (input.exact !== undefined && proposal.solutions.length !== input.exact) {
547
+ throw failure(`${input.kind} solution generation produced ${String(proposal.solutions.length)}; exactly ${String(input.exact)} are required`);
548
+ }
549
+ if (proposal.solutions.length < input.minimum) {
550
+ throw failure(`${input.kind} solution generation produced ${String(proposal.solutions.length)}; at least ${String(input.minimum)} are required`);
551
+ }
552
+ const ids = new Set(proposal.solutions.map((solution) => solution.id));
553
+ if (ids.size !== proposal.solutions.length ||
554
+ ids.has("historical-reference") ||
555
+ ids.has("historical-defect")) {
556
+ throw failure(`${input.kind} solution generation produced duplicate or reserved ids`);
557
+ }
558
+ const validated = proposal.solutions.map((solution) => {
559
+ const rewardHackingPaths = solution.fileOverrides.filter((override) => input.allowedRewardHackingPaths.has(override.path));
560
+ if (solution.id.trim().length === 0 ||
561
+ solution.family.trim().length === 0 ||
562
+ solution.fileOverrides.length === 0 ||
563
+ new Set(solution.fileOverrides.map((override) => override.path)).size !==
564
+ solution.fileOverrides.length) {
565
+ throw failure(`${input.kind} solution ${solution.id} is incomplete`);
566
+ }
567
+ for (const override of solution.fileOverrides) {
568
+ const kindAllowed = input.kind === "valid"
569
+ ? solution.kind === "independent-valid" ||
570
+ solution.kind === "simplified-valid" ||
571
+ solution.kind === "behavior-preserving-valid"
572
+ : solution.kind === "mutation" ||
573
+ solution.kind === "model-near-miss" ||
574
+ solution.kind === "reward-hacking";
575
+ const pathAllowed = input.kind === "wrong" && solution.kind === "reward-hacking"
576
+ ? input.allowedPaths.has(override.path) ||
577
+ input.allowedRewardHackingPaths.has(override.path)
578
+ : input.allowedPaths.has(override.path);
579
+ if (!kindAllowed ||
580
+ !pathAllowed ||
581
+ override.content.trim().length === 0 ||
582
+ Buffer.byteLength(override.content) > 256_000) {
583
+ throw failure(`${input.kind} solution ${solution.id} targets an unreviewed or oversized path: ${override.path}`);
584
+ }
585
+ }
586
+ if (input.kind === "wrong" &&
587
+ solution.kind === "reward-hacking" &&
588
+ rewardHackingPaths.length === 0) {
589
+ throw failure(`wrong solution ${solution.id} is classified as reward-hacking but does not modify a reviewed protected test path`);
590
+ }
591
+ return {
592
+ id: solution.id,
593
+ kind: solution.kind,
594
+ family: solution.family,
595
+ expectedClass: input.kind,
596
+ commit: input.commit,
597
+ fileOverrides: solution.fileOverrides
598
+ };
599
+ });
600
+ if (input.kind === "wrong") {
601
+ const kinds = new Set(validated.map((solution) => solution.kind));
602
+ const families = new Set(validated.map((solution) => solution.family));
603
+ const requiredKinds = ["mutation", "model-near-miss", "reward-hacking"];
604
+ const missingKinds = requiredKinds.filter((kind) => !kinds.has(kind));
605
+ if (missingKinds.length > 0 || families.size < 3) {
606
+ throw failure([
607
+ "adversary proposal lacks required deterministic diversity",
608
+ ...(missingKinds.length > 0 ? [`missing kinds: ${missingKinds.join(", ")}`] : []),
609
+ ...(families.size < 3
610
+ ? [`distinct families: ${String(families.size)}; at least 3 are required`]
611
+ : [])
612
+ ].join("; "));
613
+ }
614
+ }
615
+ return validated;
616
+ };
617
+ /** Eligibility is a necessary population bound; execution must still prove every semantic kill. */
618
+ export const assertRepositorySemanticControlPopulationV1 = (solutions, protectedControlPaths) => {
619
+ const protectedPaths = new Set(protectedControlPaths);
620
+ const eligible = solutions.filter((solution) => solution.expectedClass === "wrong" &&
621
+ solution.id !== "historical-defect" &&
622
+ !(solution.fileOverrides ?? []).some((override) => protectedPaths.has(override.path)));
623
+ if (eligible.length < 5) {
624
+ throw failure(`adversary proposal has ${String(eligible.length)} generated semantic-eligible wrong controls; at least 5 are required beyond protected-path policy controls and the historical defect`);
625
+ }
626
+ };
627
+ const validControlPairFindings = (left, right, preChangeSources, validControlReviewVersion) => {
628
+ const first = left[0];
629
+ const second = right[0];
630
+ if (first === undefined || second === undefined) {
631
+ return [
632
+ {
633
+ code: "valid-control-missing-proposal",
634
+ detail: "Each independent solver must produce one solution."
635
+ }
636
+ ];
637
+ }
638
+ const patchSignature = (solution) => JSON.stringify([...(solution.fileOverrides ?? [])]
639
+ .sort((a, b) => a.path.localeCompare(b.path))
640
+ .map((override) => ({
641
+ path: override.path,
642
+ ...("content" in override ? { content: override.content } : { delete: override.delete })
643
+ })));
644
+ const findings = [];
645
+ if (first.id === second.id)
646
+ findings.push({
647
+ code: "valid-control-duplicate-id",
648
+ detail: "Use a distinct solution identifier describing your own implementation."
649
+ });
650
+ if (validControlReviewVersion !== 2 && first.family === second.family)
651
+ findings.push({
652
+ code: "valid-control-duplicate-family",
653
+ detail: "The implementation-family labels collide. Describe your own concrete mechanism rather than the task or dimension; do not invent a distinction unsupported by your implementation."
654
+ });
655
+ if (patchSignature(first) === patchSignature(second) ||
656
+ (validControlReviewVersion === 2 &&
657
+ normalizedPatchTokens(first).join("\u0000") === normalizedPatchTokens(second).join("\u0000")))
658
+ findings.push({
659
+ code: "valid-control-duplicate-patch",
660
+ detail: validControlReviewVersion === 2
661
+ ? "The implementations are identical. Provide a natural executable alternative while preserving the entire visible contract. An equivalent formulation is sufficient; changing only labels, comments, or formatting is insufficient."
662
+ : "The implementations are identical. Independently redesign your own implementation while preserving the entire visible contract; changing labels, comments, or formatting is insufficient."
663
+ });
664
+ if (validControlReviewVersion !== 2 && patchSimilarity(first, second, preChangeSources) >= 0.9)
665
+ findings.push({
666
+ code: "valid-control-similar-patch",
667
+ detail: "The proposals have materially similar implementation deltas. Independently choose a different concrete design; renaming, comments, formatting, and family labels do not establish source diversity."
668
+ });
669
+ if (first.kind !== "independent-valid" && first.kind !== "simplified-valid")
670
+ findings.push({
671
+ code: "valid-control-invalid-a-kind",
672
+ detail: "Solver A must classify its own valid implementation as independent-valid or simplified-valid."
673
+ });
674
+ if (second.kind !== "behavior-preserving-valid")
675
+ findings.push({
676
+ code: "valid-control-invalid-b-kind",
677
+ detail: "Solver B must classify its own valid implementation as behavior-preserving-valid."
678
+ });
679
+ return findings;
680
+ };
681
+ export const assertIndependentValidSolutionPairV1 = (left, right, preChangeSources, validControlReviewVersion) => {
682
+ const findings = validControlPairFindings(left, right, preChangeSources, validControlReviewVersion);
683
+ if (findings.length > 0)
684
+ throw failure(findings.map(({ code, detail }) => `${code}: ${detail}`).join("; "));
685
+ };
686
+ const solutionPatchSignature = (solution) => JSON.stringify([...(solution.fileOverrides ?? [])]
687
+ .sort((a, b) => a.path.localeCompare(b.path))
688
+ .map((override) => ({
689
+ path: override.path,
690
+ ...("content" in override ? { content: override.content } : { delete: override.delete })
691
+ })));
692
+ const changedContent = (initial, next) => {
693
+ if (initial === undefined)
694
+ return next;
695
+ const before = initial.split("\n");
696
+ const after = next.split("\n");
697
+ let prefix = 0;
698
+ while (prefix < before.length && prefix < after.length && before[prefix] === after[prefix]) {
699
+ prefix += 1;
700
+ }
701
+ let suffix = 0;
702
+ while (suffix < before.length - prefix &&
703
+ suffix < after.length - prefix &&
704
+ before[before.length - 1 - suffix] === after[after.length - 1 - suffix]) {
705
+ suffix += 1;
706
+ }
707
+ return after.slice(prefix, after.length - suffix).join("\n");
708
+ };
709
+ const normalizedPatchTokens = (solution, preChangeSources = []) => {
710
+ const initialByPath = new Map(preChangeSources.map((source) => [source.path, source.content]));
711
+ const normalized = [...(solution.fileOverrides ?? [])]
712
+ .sort((left, right) => left.path.localeCompare(right.path))
713
+ .map((override) => {
714
+ const content = "content" in override
715
+ ? changedContent(initialByPath.get(override.path), override.content)
716
+ .replace(/\/\*[\s\S]*?\*\//gu, " ")
717
+ .replace(/(^|[^:])\/\/.*$/gmu, "$1 ")
718
+ .replace(/^\s*#.*$/gmu, " ")
719
+ : "[deleted]";
720
+ return `${override.path}\n${content}`;
721
+ })
722
+ .join("\n")
723
+ .match(/[A-Za-z_$][A-Za-z0-9_$]*|\d+(?:\.\d+)?|===|!==|=>|==|!=|<=|>=|&&|\|\||\?\?|[^\s]/gu);
724
+ return normalized ?? [];
725
+ };
726
+ const patchShingles = (solution, preChangeSources = []) => {
727
+ const tokens = normalizedPatchTokens(solution, preChangeSources);
728
+ const width = Math.min(5, Math.max(1, tokens.length));
729
+ const shingles = new Set();
730
+ for (let index = 0; index <= tokens.length - width; index += 1) {
731
+ shingles.add(tokens.slice(index, index + width).join("\u0000"));
732
+ }
733
+ return shingles;
734
+ };
735
+ const patchSimilarity = (left, right, preChangeSources = []) => {
736
+ const leftShingles = patchShingles(left, preChangeSources);
737
+ const rightShingles = patchShingles(right, preChangeSources);
738
+ if (leftShingles.size === 0 && rightShingles.size === 0)
739
+ return 1;
740
+ let intersection = 0;
741
+ for (const shingle of leftShingles) {
742
+ if (rightShingles.has(shingle))
743
+ intersection += 1;
744
+ }
745
+ const union = leftShingles.size + rightShingles.size - intersection;
746
+ return union === 0 ? 1 : intersection / union;
747
+ };
748
+ export const assertHeldOutAdversaryNoveltyV1 = (development, heldOut, preChangeSources = []) => {
749
+ const developmentIds = new Set(development.map((solution) => solution.id));
750
+ const developmentFamilies = new Set(development.map((solution) => solution.family));
751
+ const developmentSignatures = new Set(development.map(solutionPatchSignature));
752
+ const heldOutSignatures = heldOut.map(solutionPatchSignature);
753
+ const familyCollisions = heldOut
754
+ .filter((solution) => developmentFamilies.has(solution.family))
755
+ .map((solution) => solution.family);
756
+ const nearDuplicates = heldOut.flatMap((heldOutSolution) => development
757
+ .filter((developmentSolution) => patchSimilarity(developmentSolution, heldOutSolution, preChangeSources) >= 0.9)
758
+ .map((developmentSolution) => `${heldOutSolution.id}~${developmentSolution.id}`));
759
+ if (heldOut.some((solution) => developmentIds.has(solution.id)) ||
760
+ heldOutSignatures.some((signature) => developmentSignatures.has(signature)) ||
761
+ new Set(heldOutSignatures).size !== heldOutSignatures.length ||
762
+ familyCollisions.length > 0 ||
763
+ nearDuplicates.length > 0) {
764
+ throw failure([
765
+ "held-out adversaries must use new failure families and materially new patches that were not used to develop the frozen oracle",
766
+ ...(familyCollisions.length === 0
767
+ ? []
768
+ : [`reused families: ${[...new Set(familyCollisions)].sort().join(", ")}`]),
769
+ ...(nearDuplicates.length === 0
770
+ ? []
771
+ : [`near-duplicate patches: ${nearDuplicates.sort().join(", ")}`])
772
+ ].join("; "));
773
+ }
774
+ };
775
+ const validateHeldOutAdversaryProposal = (input) => {
776
+ try {
777
+ const solutions = validateSolutions(input.proposal, {
778
+ kind: "wrong",
779
+ commit: input.commit,
780
+ allowedPaths: input.allowedPaths,
781
+ allowedRewardHackingPaths: input.allowedRewardHackingPaths,
782
+ minimum: 5
783
+ });
784
+ assertRepositorySemanticControlPopulationV1(solutions, input.protectedControlPaths);
785
+ assertHeldOutAdversaryNoveltyV1(input.development, solutions, input.preChangeSources);
786
+ return { _tag: "valid", solutions };
787
+ }
788
+ catch (cause) {
789
+ return { _tag: "invalid", detail: detailOf(cause) };
790
+ }
791
+ };
792
+ const QUALITY_REVIEW_AXES = [
793
+ "specification",
794
+ "oracle",
795
+ "valid-solution-independence",
796
+ "development-adversaries",
797
+ "held-out-adversaries",
798
+ "trajectory"
799
+ ];
800
+ const QUALITY_EVIDENCE_PREFIXES = {
801
+ specification: [
802
+ "review-packet:/visible",
803
+ "review-packet:/targetBehavior/",
804
+ "review-packet:/specificationReviews/"
805
+ ],
806
+ oracle: [
807
+ "review-packet:/fixtureProposal/",
808
+ "review-packet:/oracleCoverageReviews/",
809
+ "review-packet:/oracleCoverageWitnessRevisions/",
810
+ "review-packet:/executableAdequacy",
811
+ "review-packet:/developmentOracle/",
812
+ "review-packet:/heldOutOracle/"
813
+ ],
814
+ "valid-solution-independence": ["review-packet:/validSolutions/"],
815
+ "development-adversaries": [
816
+ "review-packet:/developmentAdversaries/",
817
+ "review-packet:/solutionReviews/",
818
+ "review-packet:/developmentOracle/"
819
+ ],
820
+ "held-out-adversaries": [
821
+ "review-packet:/heldOutAdversaries/",
822
+ "review-packet:/heldOutSolutionReviews/",
823
+ "review-packet:/heldOutAdequacy",
824
+ "review-packet:/heldOutOracle/"
825
+ ],
826
+ trajectory: [
827
+ "review-packet:/trajectoryPolicy",
828
+ "review-packet:/trajectoryPolicyReview",
829
+ "review-packet:/trajectoryRevisionAttempts/"
830
+ ]
831
+ };
832
+ const qualityReviewEvidenceInventory = (input) => [
833
+ "review-packet:/visible",
834
+ ...Array.from({ length: input.targetBehaviorCount }, (_, index) => `review-packet:/targetBehavior/${String(index)}`),
835
+ ...Array.from({ length: input.specificationReviewCount }, (_, index) => `review-packet:/specificationReviews/${String(index)}`),
836
+ ...Array.from({ length: input.fixtureCount }, (_, index) => `review-packet:/fixtureProposal/fixtures/${String(index)}`),
837
+ ...Array.from({ length: input.overlayCount }, (_, index) => `review-packet:/fixtureProposal/overlays/${String(index)}`),
838
+ ...Array.from({ length: input.oracleCoverageReviewCount ?? 0 }, (_, index) => `review-packet:/oracleCoverageReviews/${String(index)}`),
839
+ ...Array.from({ length: input.oracleCoverageWitnessRevisionCount ?? 0 }, (_, index) => `review-packet:/oracleCoverageWitnessRevisions/${String(index)}`),
840
+ ...Array.from({ length: input.validSolutionCount }, (_, index) => `review-packet:/validSolutions/${String(index)}`),
841
+ ...Array.from({ length: input.developmentAdversaryCount }, (_, index) => `review-packet:/developmentAdversaries/${String(index)}`),
842
+ ...Array.from({ length: input.solutionReviewCount }, (_, index) => `review-packet:/solutionReviews/${String(index)}`),
843
+ ...Array.from({ length: input.heldOutAdversaryCount }, (_, index) => `review-packet:/heldOutAdversaries/${String(index)}`),
844
+ ...Array.from({ length: input.heldOutSolutionReviewCount }, (_, index) => `review-packet:/heldOutSolutionReviews/${String(index)}`),
845
+ "review-packet:/heldOutAdequacy",
846
+ "review-packet:/executableAdequacy",
847
+ "review-packet:/developmentOracle/clauses",
848
+ "review-packet:/heldOutOracle/clauses",
849
+ ...Array.from({ length: input.developmentObservationCount }, (_, index) => `review-packet:/developmentOracle/observations/${String(index)}`),
850
+ ...Array.from({ length: input.heldOutObservationCount }, (_, index) => `review-packet:/heldOutOracle/observations/${String(index)}`),
851
+ "review-packet:/trajectoryPolicy",
852
+ "review-packet:/trajectoryPolicyReview",
853
+ ...Array.from({ length: input.trajectoryRevisionAttemptCount }, (_, index) => `review-packet:/trajectoryRevisionAttempts/${String(index)}`),
854
+ ...Array.from({ length: input.hillClimbAttemptCount }, (_, index) => `review-packet:/hillClimbAttempts/${String(index)}`)
855
+ ];
856
+ export const validateRepositoryGenerationQualityReviewV1 = (proposal, evidenceInventory) => {
857
+ const evidence = new Set(evidenceInventory);
858
+ const findingsByAxis = new Map(proposal.findings.map((finding) => [finding.axis, finding]));
859
+ const invalidFindings = proposal.findings.filter((finding) => {
860
+ const allowedPrefixes = QUALITY_EVIDENCE_PREFIXES[finding.axis];
861
+ return (finding.detail.trim().length === 0 ||
862
+ finding.evidenceIds.length === 0 ||
863
+ new Set(finding.evidenceIds).size !== finding.evidenceIds.length ||
864
+ finding.evidenceIds.some((evidenceId) => !evidence.has(evidenceId)) ||
865
+ !finding.evidenceIds.some((evidenceId) => allowedPrefixes.some((prefix) => evidenceId.startsWith(prefix))));
866
+ });
867
+ if (proposal.detail.trim().length === 0 ||
868
+ proposal.findings.length !== QUALITY_REVIEW_AXES.length ||
869
+ findingsByAxis.size !== QUALITY_REVIEW_AXES.length ||
870
+ QUALITY_REVIEW_AXES.some((axis) => !findingsByAxis.has(axis)) ||
871
+ invalidFindings.length > 0 ||
872
+ (proposal.verdict === "approve" &&
873
+ proposal.findings.some((finding) => finding.outcome !== "pass")) ||
874
+ (proposal.verdict === "reject" &&
875
+ proposal.findings.every((finding) => finding.outcome === "pass"))) {
876
+ throw failure("independent quality review must provide one internally consistent, evidence-linked finding for every required axis");
877
+ }
878
+ };
879
+ const solutionReviewDisagreements = (proposal, solutions, reviewer) => {
880
+ const expected = new Map(solutions.map((solution) => [solution.id, solution.expectedClass]));
881
+ const ids = proposal.reviews.map((review) => review.solutionId);
882
+ if (ids.length !== expected.size ||
883
+ new Set(ids).size !== ids.length ||
884
+ ids.some((id) => !expected.has(id))) {
885
+ throw failure(`${reviewer} must return exactly one classification for every proposed solution`);
886
+ }
887
+ return proposal.reviews
888
+ .filter((review) => review.detail.trim().length === 0 ||
889
+ review.classification !== expected.get(review.solutionId) ||
890
+ (review.classification === "wrong" && review.violatedBehaviorIds.length === 0))
891
+ .map((review) => ({ reviewer, review }));
892
+ };
893
+ const validateSolutionReview = (proposal, solutions, reviewer) => {
894
+ const disagreements = solutionReviewDisagreements(proposal, solutions, reviewer);
895
+ if (disagreements.length > 0) {
896
+ throw failure(`${reviewer} rejected generated solution labels: ${disagreements
897
+ .map(({ review }) => {
898
+ return `${review.solutionId}=${review.classification}${review.detail.trim().length === 0 ? " (missing rationale)" : ` (${review.detail})`}`;
899
+ })
900
+ .join("; ")}`);
901
+ }
902
+ };
903
+ /** Critic prose is private evidence. Only this fixed vocabulary can reach either solver. */
904
+ export const repositorySemanticReviewIndependenceFindingsV1 = (proposal, validSolutions, reviewer, validControlReviewVersion, criticInput) => {
905
+ const suppliedSourcePaths = new Set();
906
+ if (validControlReviewVersion === 2 && criticInput !== undefined) {
907
+ // This is the exact prepared request, not repository-global knowledge or
908
+ // model-returned evidence. Path-only dependency metadata supplies no bytes.
909
+ const includeProvidedSources = (sources) => {
910
+ if (!Array.isArray(sources))
911
+ return;
912
+ for (const entry of sources) {
913
+ if (entry === null || typeof entry !== "object" || Array.isArray(entry))
914
+ continue;
915
+ const source = entry;
916
+ if (Object.hasOwn(source, "path") &&
917
+ Object.hasOwn(source, "content") &&
918
+ typeof source.path === "string" &&
919
+ typeof source.content === "string") {
920
+ suppliedSourcePaths.add(source.path);
921
+ }
922
+ }
923
+ };
924
+ includeProvidedSources(criticInput.preChangeSources);
925
+ const environment = criticInput.repositoryEnvironment;
926
+ if (environment !== null && typeof environment === "object") {
927
+ if ("sources" in environment)
928
+ includeProvidedSources(environment.sources);
929
+ const localImports = "localImports" in environment ? environment.localImports : undefined;
930
+ if (localImports !== null &&
931
+ typeof localImports === "object" &&
932
+ "sources" in localImports) {
933
+ includeProvidedSources(localImports.sources);
934
+ }
935
+ }
936
+ }
937
+ const assessment = proposal.implementationIndependence;
938
+ const expected = new Map(validSolutions.map((solution) => [solution.id, solution]));
939
+ if (assessment === undefined ||
940
+ expected.size !== 2 ||
941
+ assessment.detail.trim().length === 0 ||
942
+ assessment.implementations.length !== 2 ||
943
+ new Set(assessment.implementations.map((entry) => entry.solutionId)).size !== 2 ||
944
+ assessment.implementations.some((entry) => {
945
+ const solution = expected.get(entry.solutionId);
946
+ const overridePaths = new Set(solution?.fileOverrides?.map((override) => override.path));
947
+ return (solution === undefined ||
948
+ entry.mechanism.trim().length === 0 ||
949
+ entry.sourcePaths.length === 0 ||
950
+ new Set(entry.sourcePaths).size !== entry.sourcePaths.length ||
951
+ !entry.sourcePaths.some((path) => overridePaths.has(path)) ||
952
+ entry.sourcePaths.some((path) => !overridePaths.has(path) && !suppliedSourcePaths.has(path)));
953
+ }) ||
954
+ new Set(assessment.correlatedFeatures).size !== assessment.correlatedFeatures.length ||
955
+ (assessment.outcome === "independent") !== (assessment.correlatedFeatures.length === 0)) {
956
+ throw failure(`${reviewer} must separately assess both current implementations with inspected source paths and consistent independence evidence`);
957
+ }
958
+ if (assessment.outcome === "independent" || validControlReviewVersion === 2)
959
+ return [];
960
+ const details = {
961
+ "renaming-or-formatting": "Renaming, formatting, comments, or family labels do not provide a different implementation mechanism.",
962
+ "helper-extraction": "Moving the same behavior into a helper does not provide a different implementation mechanism.",
963
+ "equivalent-control-flow": "Reordering or restating equivalent guards and control flow does not provide a different implementation mechanism.",
964
+ "same-behavioral-mechanism": "Reimplement the changed visible behavior using a substantively different mechanism while preserving all required behavior and integration constraints.",
965
+ "insufficient-evidence": "Independent review could not establish a distinct mechanism. Reassess your own implementation and choose a clearly justified different approach within the visible task."
966
+ };
967
+ return assessment.correlatedFeatures.map((feature) => ({
968
+ code: `valid-control-independence-${feature}`,
969
+ detail: details[feature]
970
+ }));
971
+ };
972
+ const difference = (left, right) => {
973
+ const rightSet = new Set(right);
974
+ return left.filter((entry) => !rightSet.has(entry));
975
+ };
976
+ /**
977
+ * Allows a fresh oracle proposal for repeated, complete test-body evidence errors.
978
+ * The original outcomes remain unknown. Mixed outcomes and operational failures
979
+ * are not admitted to this repair path.
980
+ */
981
+ export const repairableRepositoryFixtureEvidenceV1 = (oracle, adequacy) => {
982
+ if (adequacy.unstableSolutionIds.length === 0 ||
983
+ adequacy.unobservedClauseIds.length > 0 ||
984
+ adequacy.insufficientObservationPairs.length > 0)
985
+ return [];
986
+ const clauses = new Map(oracle.clauses.map((clause) => [clause.id, clause]));
987
+ const fixtures = new Set(oracle.fixtures.map((fixture) => fixture.id));
988
+ const groups = new Map();
989
+ for (const observation of oracle.observations) {
990
+ const key = JSON.stringify([observation.solutionId, observation.clauseId]);
991
+ const group = groups.get(key) ?? [];
992
+ group.push(observation);
993
+ groups.set(key, group);
994
+ }
995
+ const targets = [];
996
+ for (const observations of groups.values()) {
997
+ const outcomes = new Set(observations.map((observation) => observation.outcome));
998
+ if (outcomes.size !== 1)
999
+ return [];
1000
+ if (!outcomes.has("unknown"))
1001
+ continue;
1002
+ const first = observations[0];
1003
+ const clause = clauses.get(first.clauseId);
1004
+ if (clause?.kind !== "test-command" ||
1005
+ clause.fixtureIds.length === 0 ||
1006
+ clause.fixtureIds.some((id) => !fixtures.has(id)) ||
1007
+ new Set(observations.map((observation) => observation.repetition)).size < 2 ||
1008
+ observations.some((observation) => observation.inconclusiveReason !== "unattributed-test-failure"))
1009
+ return [];
1010
+ targets.push({
1011
+ solutionId: first.solutionId,
1012
+ clauseId: first.clauseId,
1013
+ fixtureIds: clause.fixtureIds,
1014
+ observations
1015
+ });
1016
+ }
1017
+ const affected = new Set(targets.map((target) => target.solutionId));
1018
+ return adequacy.unstableSolutionIds.every((id) => affected.has(id)) ? targets : [];
1019
+ };
1020
+ const assessHillClimbCandidate = (before, candidate, requireImprovement = true) => {
1021
+ const introducedFalseAccepts = difference(candidate.falseAcceptedSolutionIds, before.falseAcceptedSolutionIds);
1022
+ const introducedFalseRejects = difference(candidate.falseRejectedSolutionIds, before.falseRejectedSolutionIds);
1023
+ const introducedInstability = difference(candidate.unstableSolutionIds, before.unstableSolutionIds);
1024
+ const introducedUnobservedClauses = difference(candidate.unobservedClauseIds, before.unobservedClauseIds);
1025
+ const introducedInsufficientPairs = difference(candidate.insufficientObservationPairs, before.insufficientObservationPairs);
1026
+ const beforeClassificationDefects = new Set([
1027
+ ...before.falseAcceptedSolutionIds,
1028
+ ...before.falseRejectedSolutionIds,
1029
+ ...before.unstableSolutionIds
1030
+ ]).size;
1031
+ const candidateClassificationDefects = new Set([
1032
+ ...candidate.falseAcceptedSolutionIds,
1033
+ ...candidate.falseRejectedSolutionIds,
1034
+ ...candidate.unstableSolutionIds
1035
+ ]).size;
1036
+ return [
1037
+ ...(introducedFalseAccepts.length === 0
1038
+ ? []
1039
+ : [`introduced false accepts: ${introducedFalseAccepts.join(", ")}`]),
1040
+ ...(introducedFalseRejects.length === 0
1041
+ ? []
1042
+ : [`introduced false rejects: ${introducedFalseRejects.join(", ")}`]),
1043
+ ...(introducedInstability.length === 0
1044
+ ? []
1045
+ : [`introduced unstable solutions: ${introducedInstability.join(", ")}`]),
1046
+ ...(introducedUnobservedClauses.length === 0
1047
+ ? []
1048
+ : [`introduced unobserved clauses: ${introducedUnobservedClauses.join(", ")}`]),
1049
+ ...(introducedInsufficientPairs.length === 0
1050
+ ? []
1051
+ : [`introduced insufficient observations: ${introducedInsufficientPairs.join(", ")}`]),
1052
+ ...(!requireImprovement || candidateClassificationDefects < beforeClassificationDefects
1053
+ ? []
1054
+ : [
1055
+ `did not reduce oracle defects (${String(beforeClassificationDefects)} -> ${String(candidateClassificationDefects)})`
1056
+ ])
1057
+ ];
1058
+ };
1059
+ export class CaseGeneration extends Context.Service()("@velum-labs/routekit-eval-setup/CaseGeneration") {
1060
+ }
1061
+ const generateHistoricalRepositoryCaseStagesV1 = Effect.fn("CaseGeneration.generateHistorical")(function* (input) {
1062
+ const languageModel = yield* RepositoryFoundryLanguageModel;
1063
+ if (input.specificationContractFactsVersion !== undefined &&
1064
+ (input.specificationContractFactsVersion !== 1 ||
1065
+ input.maximumSpecificationRevisions === undefined)) {
1066
+ return yield* failure("specification contract facts require version 1 and an explicit specification revision allowance");
1067
+ }
1068
+ if (input.oracleRepairFeedbackVersion !== undefined &&
1069
+ input.oracleRepairFeedbackVersion !== 1) {
1070
+ return yield* failure("oracle repair feedback version must be 1");
1071
+ }
1072
+ if (input.validControlReviewVersion !== undefined &&
1073
+ ![1, 2].includes(input.validControlReviewVersion)) {
1074
+ return yield* failure("valid-control review version must be 1 or 2");
1075
+ }
1076
+ if (input.oracleCoverageWitnessRepairVersion !== undefined &&
1077
+ (input.oracleCoverageWitnessRepairVersion !== 1 ||
1078
+ input.oracleCoverageWitnessVersion === undefined ||
1079
+ input.maximumFixtureRepairAttempts === undefined)) {
1080
+ return yield* failure("coverage witness repair budgeting requires versioned witnesses and an explicit fixture repair allowance");
1081
+ }
1082
+ if (input.oracleCoverageWitnessVersion !== undefined &&
1083
+ (![1, 2].includes(input.oracleCoverageWitnessVersion) ||
1084
+ input.maximumOracleCoverageRevisions === undefined ||
1085
+ input.maximumOracleCoverageRevisions < 1)) {
1086
+ return yield* failure("executable coverage witnesses require a versioned, nonzero coverage allowance");
1087
+ }
1088
+ if (input.maximumFixtureRepairAttempts !== undefined &&
1089
+ (!Number.isSafeInteger(input.maximumFixtureRepairAttempts) ||
1090
+ input.maximumFixtureRepairAttempts < 0 ||
1091
+ input.maximumFixtureRepairAttempts > 3)) {
1092
+ return yield* failure("fixture repairs must be an integer from 0 to 3");
1093
+ }
1094
+ if (input.maximumOracleCoverageRevisions !== undefined &&
1095
+ (input.maximumSpecificationRevisions === undefined ||
1096
+ !Number.isSafeInteger(input.maximumOracleCoverageRevisions) ||
1097
+ input.maximumOracleCoverageRevisions < 0 ||
1098
+ input.maximumOracleCoverageRevisions > 2 ||
1099
+ !input.modelPlan.assignments.some((assignment) => assignment.role === "oracle-critic"))) {
1100
+ return yield* failure("oracle coverage review requires a bounded revision allowance, reference grounding, and an assigned oracle critic");
1101
+ }
1102
+ const reviewIndependence = input.validControlReviewVersion !== undefined;
1103
+ const solverBInstructions = input.validControlReviewVersion === 2
1104
+ ? VALID_SOLUTION_B_BEHAVIORAL_INSTRUCTIONS
1105
+ : reviewIndependence
1106
+ ? `${VALID_SOLUTION_B_INSTRUCTIONS}\n\n${VALID_SOLUTION_B_MECHANISM_INSTRUCTIONS}`
1107
+ : VALID_SOLUTION_B_INSTRUCTIONS;
1108
+ const dimensionContext = input.requestedDimension === undefined
1109
+ ? {}
1110
+ : { requestedDimension: input.requestedDimension };
1111
+ if (input.seed.status !== "qualified" || input.seed.referenceCommit === undefined) {
1112
+ return yield* failure("model-generated historical cases require a replay-qualified seed");
1113
+ }
1114
+ const family = (yield* deriveRepositoryTaskFamiliesV1({
1115
+ map: input.map,
1116
+ seeds: [input.seed]
1117
+ })).find((candidate) => candidate.seedIds.includes(input.seed.id));
1118
+ if (family === undefined) {
1119
+ return yield* failure("qualified seed has no derived task family");
1120
+ }
1121
+ yield* reportFoundryStageV1("author-specification");
1122
+ const authored = yield* authorRepositoryCaseSpecificationV1({
1123
+ repositoryRoot: input.repositoryRoot,
1124
+ operationId: input.operationId,
1125
+ caseId: input.caseId,
1126
+ ...dimensionContext,
1127
+ ...(input.specificationContractFactsVersion === undefined
1128
+ ? {}
1129
+ : { specificationContractFactsVersion: input.specificationContractFactsVersion }),
1130
+ ...(input.maximumSpecificationRevisions === undefined
1131
+ ? {}
1132
+ : { maximumSpecificationRevisions: input.maximumSpecificationRevisions }),
1133
+ map: input.map,
1134
+ seed: input.seed,
1135
+ family,
1136
+ modelPlan: input.modelPlan
1137
+ });
1138
+ const referenceGrounded = input.maximumSpecificationRevisions !== undefined;
1139
+ if (referenceGrounded &&
1140
+ (authored.groundedTargetBehavior === undefined || authored.referenceGrounding === undefined)) {
1141
+ return yield* failure("reference-grounded authoring did not return its validated behavioral scope");
1142
+ }
1143
+ if (input.specificationContractFactsVersion === 1)
1144
+ yield* Effect.try({
1145
+ try: () => assertRepositorySpecificationContractFactsEvidenceV1(authored),
1146
+ catch: (cause) => failure("contract-fact authoring evidence failed validation", cause)
1147
+ });
1148
+ const seed = referenceGrounded
1149
+ ? { ...input.seed, targetBehavior: authored.groundedTargetBehavior }
1150
+ : input.seed;
1151
+ const requiredFixtureScope = referenceGrounded
1152
+ ? authored.referenceGrounding.scope.clauses
1153
+ : undefined;
1154
+ const initial = {
1155
+ sources: authored.contextSources
1156
+ };
1157
+ const initialSnapshot = yield* captureGitTreeSnapshotV1({
1158
+ repositoryRoot: input.repositoryRoot,
1159
+ requestedRef: seed.initialState.commit
1160
+ });
1161
+ const initialLocalImports = yield* readRepositoryLocalImportsV1({
1162
+ snapshot: initialSnapshot,
1163
+ sources: initial.sources,
1164
+ ...(input.dependencyContextVersion === undefined
1165
+ ? {}
1166
+ : { dependencyContextVersion: input.dependencyContextVersion })
1167
+ });
1168
+ const environmentPaths = [
1169
+ "package.json",
1170
+ "pnpm-workspace.yaml",
1171
+ ...input.map.packages
1172
+ .filter((entry) => initial.sources.some((source) => entry.path === "." || source.path.startsWith(`${entry.path}/`)))
1173
+ .map((entry) => entry.manifestPath)
1174
+ ];
1175
+ const repositoryEnvironment = {
1176
+ commit: seed.initialState.commit,
1177
+ localImports: initialLocalImports,
1178
+ sources: (yield* sourcePacket({
1179
+ repositoryRoot: input.repositoryRoot,
1180
+ commit: seed.initialState.commit,
1181
+ paths: environmentPaths
1182
+ })).sources
1183
+ };
1184
+ const modelContext = {
1185
+ ...dimensionContext,
1186
+ repositoryEnvironment,
1187
+ ...(referenceGrounded
1188
+ ? {
1189
+ targetBehavior: seed.targetBehavior,
1190
+ behavioralScope: authored.referenceGrounding.scope.clauses.map((clause) => ({
1191
+ id: clause.id,
1192
+ kind: clause.kind,
1193
+ behaviorIds: clause.behaviorIds,
1194
+ description: clause.description
1195
+ }))
1196
+ }
1197
+ : {})
1198
+ };
1199
+ const referencePacket = yield* sourcePacket({
1200
+ repositoryRoot: input.repositoryRoot,
1201
+ commit: seed.referenceCommit,
1202
+ paths: authored.contextSources.map((source) => source.path)
1203
+ });
1204
+ const reference = {
1205
+ ...referencePacket,
1206
+ sources: authored.referenceGrounding === undefined
1207
+ ? referencePacket.sources
1208
+ : authored.referenceGrounding.sources
1209
+ .filter((source) => source.phase === "after" &&
1210
+ source.content !== null &&
1211
+ initial.sources.some((candidate) => candidate.path === source.path))
1212
+ .map((source) => ({ path: source.path, content: source.content }))
1213
+ };
1214
+ const referenceLocalImports = yield* readRepositoryLocalImportsV1({
1215
+ snapshot: reference.snapshot,
1216
+ sources: reference.sources,
1217
+ ...(input.dependencyContextVersion === undefined
1218
+ ? {}
1219
+ : { dependencyContextVersion: input.dependencyContextVersion })
1220
+ });
1221
+ const testPaths = new Set(seed.capabilityEvidence
1222
+ .filter((evidence) => evidence.kind === "test" && evidence.path !== undefined)
1223
+ .map((evidence) => evidence.path));
1224
+ const rewardHackingPaths = new Set([...testPaths, ...seed.environment.protectedControlPaths]);
1225
+ const allowedSolutionPaths = new Set(seed.capabilityEvidence
1226
+ .map((evidence) => evidence.path)
1227
+ .filter((path) => path !== undefined)
1228
+ .filter((path) => initial.sources.some((source) => source.path === path) &&
1229
+ !testPaths.has(path) &&
1230
+ !seed.environment.protectedControlPaths.includes(path)));
1231
+ if (testPaths.size === 0 || allowedSolutionPaths.size === 0) {
1232
+ return yield* failure("qualified seed does not expose both a reviewed test path and an implementation path");
1233
+ }
1234
+ const allEditablePaths = new Set([...allowedSolutionPaths, ...rewardHackingPaths]);
1235
+ const editSources = yield* loadPinnedSolutionEditSourcesV1({
1236
+ repositoryRoot: input.repositoryRoot,
1237
+ initialCommit: seed.initialState.commit,
1238
+ visibleSources: [...initial.sources, ...repositoryEnvironment.sources],
1239
+ allowedPaths: allEditablePaths
1240
+ });
1241
+ const generateSolution = (request) => generateRepositorySolutionProposalV1({
1242
+ request,
1243
+ initialCommit: seed.initialState.commit,
1244
+ editSources,
1245
+ allowedSolutionPaths,
1246
+ allowedRewardHackingPaths: allEditablePaths
1247
+ });
1248
+ const trajectoryValidationRecipeIds = seed.environment.gradeRecipes
1249
+ .map((recipe) => recipe.id)
1250
+ .sort();
1251
+ const trajectoryProtectedPaths = [...new Set(seed.environment.protectedControlPaths)].sort();
1252
+ yield* reportFoundryStageV1("trajectory-policy");
1253
+ const trajectory = yield* authorReviewedRepositoryTrajectoryPolicyV1({
1254
+ operationId: input.operationId,
1255
+ caseId: input.caseId,
1256
+ visible: authored.visible,
1257
+ targetBehavior: seed.targetBehavior,
1258
+ reviewedImplementationPathCount: allowedSolutionPaths.size,
1259
+ validationRecipeIds: trajectoryValidationRecipeIds,
1260
+ protectedPaths: trajectoryProtectedPaths,
1261
+ reviewedCommands: [
1262
+ ...seed.environment.candidateValidationRecipes,
1263
+ ...seed.environment.gradeRecipes
1264
+ ].map((recipe) => ({
1265
+ id: recipe.id,
1266
+ kind: recipe.kind,
1267
+ timeoutMs: recipe.timeoutMs
1268
+ })),
1269
+ modelPlan: input.modelPlan
1270
+ });
1271
+ const trajectoryPolicy = trajectory.policy;
1272
+ const trajectoryPolicyReview = trajectory.review;
1273
+ const trajectoryModelCalls = trajectory.modelCalls;
1274
+ const trajectoryRevisionAttempts = trajectory.revisionAttempts;
1275
+ yield* reportFoundryStageV1("oracle-design");
1276
+ const fixtureGeneration = yield* languageModel.generateStructured({
1277
+ plan: input.modelPlan,
1278
+ role: "oracle-designer",
1279
+ operationId: `${input.operationId}:oracle-designer`,
1280
+ instructions: repositoryFixtureAuthoringInstructionsV1(ORACLE_INSTRUCTIONS, input.oracleCoverageWitnessVersion, input.oracleRepairFeedbackVersion),
1281
+ input: {
1282
+ ...modelContext,
1283
+ visible: authored.visible,
1284
+ targetBehavior: seed.targetBehavior,
1285
+ family,
1286
+ gradeRecipes: seed.environment.gradeRecipes,
1287
+ allowedTestPaths: [...testPaths],
1288
+ preChangeSources: initial.sources,
1289
+ referenceSources: reference.sources,
1290
+ referenceLocalImports
1291
+ },
1292
+ schemaName: "routekit_repository_fixture_suite_v1",
1293
+ outputSchema: FixtureProposal,
1294
+ maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("oracle-designer")
1295
+ });
1296
+ yield* Effect.try({
1297
+ try: () => assertRepositoryFixtureScopeCoverageV1(fixtureGeneration.value, requiredFixtureScope),
1298
+ catch: (cause) => cause instanceof RepositoryFoundryError
1299
+ ? cause
1300
+ : failure("generated fixture scope coverage failed validation", cause)
1301
+ });
1302
+ const maximumOracleRepairAttempts = Math.max(0, Math.min(input.maximumHillClimbAttempts ?? 2, 3));
1303
+ const maximumFixtureRepairAttempts = input.maximumFixtureRepairAttempts ?? maximumOracleRepairAttempts;
1304
+ const separateFixtureRepairBudget = input.maximumFixtureRepairAttempts !== undefined;
1305
+ yield* reportFoundryStageV1("oracle-design", "validating generated fixtures on the pinned reference");
1306
+ const fixtureValidation = yield* validateRepositoryFixturesV1({
1307
+ oracleCoverageWitnessVersion: input.oracleCoverageWitnessVersion,
1308
+ oracleRepairFeedbackVersion: input.oracleRepairFeedbackVersion,
1309
+ repositoryRoot: input.repositoryRoot,
1310
+ operationId: input.operationId,
1311
+ seed: seed,
1312
+ suite: fixtureSuiteFrom(input.caseId, fixtureGeneration.value),
1313
+ allowedTestPaths: testPaths,
1314
+ context: {
1315
+ ...modelContext,
1316
+ visible: authored.visible,
1317
+ preChangeSources: initial.sources,
1318
+ referenceSources: reference.sources,
1319
+ referenceLocalImports,
1320
+ ...(requiredFixtureScope === undefined
1321
+ ? {}
1322
+ : { frozenScopeCoverage: fixtureGeneration.value.scopeCoverage })
1323
+ },
1324
+ modelPlan: input.modelPlan,
1325
+ maximumRepairAttempts: maximumFixtureRepairAttempts
1326
+ });
1327
+ // Overlay-only preflight repairs preserve this already-validated attestation.
1328
+ // Executability does not independently establish that repaired assertions cover it.
1329
+ let fixtureProposal = {
1330
+ ...fixtureGeneration.value,
1331
+ overlays: fixtureValidation.suite.overlays
1332
+ };
1333
+ let fixtureSuite = fixtureValidation.suite;
1334
+ const oracleAuthoringModelCalls = [
1335
+ fixtureGeneration.call,
1336
+ ...fixtureValidation.modelCalls
1337
+ ];
1338
+ const freezeCoverageModelCalls = [];
1339
+ const oracleCoverageReviews = [];
1340
+ let fixturePreflightRepairCalls = fixtureValidation.modelCalls.length;
1341
+ let executedOracleHillClimbCalls = 0;
1342
+ let coverageRevisionsUsed = 0;
1343
+ let approvedCoverageSignature;
1344
+ const retainedCoverageFixtures = new Map();
1345
+ const retainCoverageFixtures = (proposal) => {
1346
+ if (retainedCoverageFixtures.size === 0)
1347
+ return proposal;
1348
+ const present = new Set(proposal.fixtures.map((fixture) => fixture.id));
1349
+ const additions = [...retainedCoverageFixtures].filter(([id]) => !present.has(id));
1350
+ return {
1351
+ ...proposal,
1352
+ ...(input.oracleCoverageWitnessVersion === 2 && proposal.scopeCoverage !== undefined
1353
+ ? {
1354
+ scopeCoverage: proposal.scopeCoverage.map((scope) => ({
1355
+ ...scope,
1356
+ fixtureIds: [
1357
+ ...new Set([
1358
+ ...scope.fixtureIds,
1359
+ ...[...retainedCoverageFixtures.values()]
1360
+ .filter((entry) => entry.witness.scopeId === scope.scopeId)
1361
+ .map((entry) => entry.fixture.id)
1362
+ ])
1363
+ ]
1364
+ }))
1365
+ }
1366
+ : {}),
1367
+ fixtures: [
1368
+ ...proposal.fixtures.map((fixture) => retainedCoverageFixtures.get(fixture.id)?.fixture ?? fixture),
1369
+ ...additions.map(([, entry]) => entry.fixture)
1370
+ ],
1371
+ overlays: [
1372
+ ...proposal.overlays.map((overlay) => {
1373
+ const retained = retainedCoverageFixtures.get(overlay.fixtureIds[0]);
1374
+ return retained === undefined || input.oracleCoverageWitnessVersion === 2
1375
+ ? overlay
1376
+ : retained.overlay;
1377
+ }),
1378
+ ...additions.map(([, entry]) => entry.overlay)
1379
+ ]
1380
+ };
1381
+ };
1382
+ const ensureOracleCoverage = Effect.fnUntraced(function* (proposal, phase, calls) {
1383
+ if (input.maximumOracleCoverageRevisions === undefined)
1384
+ return proposal;
1385
+ if (requiredFixtureScope === undefined)
1386
+ return yield* failure("oracle coverage review requires the frozen behavioral scope");
1387
+ let current = proposal;
1388
+ for (;;) {
1389
+ const suite = fixtureSuiteFrom(input.caseId, current);
1390
+ const signature = JSON.stringify(suite);
1391
+ if (signature === approvedCoverageSignature)
1392
+ return current;
1393
+ yield* reportFoundryStageV1("oracle-design", `independent coverage review ${phase}; revisions used=${String(coverageRevisionsUsed)}`);
1394
+ const reviewed = yield* reviewRepositoryOracleCoverageV1({
1395
+ operationId: `${input.operationId}:oracle-critic:${phase}:${String(coverageRevisionsUsed)}`,
1396
+ modelPlan: input.modelPlan,
1397
+ scope: requiredFixtureScope,
1398
+ suite,
1399
+ evidence: {
1400
+ ...modelContext,
1401
+ visible: authored.visible,
1402
+ preChangeSources: initial.sources,
1403
+ referenceSources: reference.sources,
1404
+ referenceLocalImports,
1405
+ gradeRecipes: seed.environment.gradeRecipes
1406
+ }
1407
+ });
1408
+ calls.push(reviewed.call);
1409
+ const coverageRecord = {
1410
+ phase,
1411
+ revision: coverageRevisionsUsed,
1412
+ review: reviewed.value
1413
+ };
1414
+ oracleCoverageReviews.push(coverageRecord);
1415
+ if (reviewed.value.verdict === "approve") {
1416
+ approvedCoverageSignature = signature;
1417
+ return current;
1418
+ }
1419
+ // A new counterexample supersedes any older approval. A repair that
1420
+ // restores an earlier suite still needs a fresh independent review.
1421
+ approvedCoverageSignature = undefined;
1422
+ if (coverageRevisionsUsed >= input.maximumOracleCoverageRevisions) {
1423
+ return yield* failure(`oracle-coverage-repair-exhausted: independent review found unresolved scope coverage before freeze; ${reviewed.value.scopeCoverage
1424
+ .filter((entry) => entry.outcome !== "covered")
1425
+ .map((entry) => `${entry.scopeId}: ${entry.counterexample}`)
1426
+ .join("; ")}`);
1427
+ }
1428
+ if (input.oracleCoverageWitnessVersion !== undefined) {
1429
+ yield* reportFoundryStageV1("oracle-design", `executing fixed coverage hypotheses ${phase}; pending=${String(reviewed.value.scopeCoverage.filter((entry) => entry.outcome !== "covered").length)}`);
1430
+ const resolved = yield* resolveRepositoryOracleCoverageWitnessesV1({
1431
+ oracleCoverageWitnessVersion: input.oracleCoverageWitnessVersion,
1432
+ oracleRepairFeedbackVersion: input.oracleRepairFeedbackVersion,
1433
+ operationId: `${input.operationId}:coverage-witness:${phase}`,
1434
+ witnessPrefix: `coverage-${phase}`,
1435
+ repositoryRoot: input.repositoryRoot,
1436
+ seed,
1437
+ suite,
1438
+ review: reviewed.value,
1439
+ modelPlan: input.modelPlan,
1440
+ allowedTestPaths: testPaths,
1441
+ allowedSolutionPaths,
1442
+ referenceSources: reference.sources,
1443
+ context: {
1444
+ ...modelContext,
1445
+ visible: authored.visible,
1446
+ targetBehavior: seed.targetBehavior,
1447
+ referenceLocalImports
1448
+ },
1449
+ maximumAttempts: input.oracleCoverageWitnessRepairVersion === 1
1450
+ ? Math.min(2, 1 + Math.max(0, maximumFixtureRepairAttempts - fixturePreflightRepairCalls))
1451
+ : input.maximumOracleCoverageRevisions - coverageRevisionsUsed,
1452
+ onGenerationAttempt: (attempt) => Effect.sync(() => {
1453
+ if (input.oracleCoverageWitnessRepairVersion === 1 && attempt > 0) {
1454
+ fixturePreflightRepairCalls += 1;
1455
+ }
1456
+ else {
1457
+ coverageRevisionsUsed += 1;
1458
+ }
1459
+ })
1460
+ });
1461
+ calls.push(...resolved.modelCalls);
1462
+ oracleCoverageReviews[oracleCoverageReviews.length - 1] = {
1463
+ ...coverageRecord,
1464
+ executionAssessment: { kind: "resolved-by-execution", witnesses: resolved.witnesses }
1465
+ };
1466
+ for (const witness of resolved.witnesses) {
1467
+ const { source: _source, ...fixture } = witness.fixture;
1468
+ retainedCoverageFixtures.set(fixture.id, {
1469
+ fixture,
1470
+ overlay: witness.overlay,
1471
+ witness
1472
+ });
1473
+ }
1474
+ current = retainCoverageFixtures(current);
1475
+ // The critic's original verdict remains "revise". Execution settles this
1476
+ // fixed set of claims; downstream valid, adversarial and final reviews remain required.
1477
+ approvedCoverageSignature = JSON.stringify(fixtureSuiteFrom(input.caseId, current));
1478
+ yield* reportFoundryStageV1("oracle-design", `coverage hypotheses resolved by execution; escapes=${String(resolved.witnesses.filter((entry) => entry.outcome === "confirmed-escape").length)}; already detected=${String(resolved.witnesses.filter((entry) => entry.outcome === "already-detected").length)}`);
1479
+ return current;
1480
+ }
1481
+ const sequence = ++coverageRevisionsUsed;
1482
+ yield* reportFoundryStageV1("oracle-design", `repairing independently identified coverage gaps; revision=${String(sequence)}`);
1483
+ const repaired = yield* languageModel.generateStructured({
1484
+ plan: input.modelPlan,
1485
+ role: "oracle-designer",
1486
+ operationId: `${input.operationId}:oracle-coverage-repair:${String(sequence)}`,
1487
+ instructions: repositoryFixtureAuthoringInstructionsV1(`${ORACLE_INSTRUCTIONS}\n\nThis is a bounded coverage repair. An independent critic identified isolated contract violations that may pass the current assertions. Return a complete revised fixture proposal resolving every counterexample. Preserve every frozen scope clause and all existing useful assertions. Exercise the relevant input/state interactions directly, so an unrelated remaining bug cannot mask whether each violation is detected. Do not alter the visible contract, solution implementations, labels, or grading requirements.`, input.oracleCoverageWitnessVersion, input.oracleRepairFeedbackVersion),
1488
+ input: {
1489
+ ...modelContext,
1490
+ visible: authored.visible,
1491
+ targetBehavior: seed.targetBehavior,
1492
+ family,
1493
+ gradeRecipes: seed.environment.gradeRecipes,
1494
+ allowedTestPaths: [...testPaths],
1495
+ preChangeSources: initial.sources,
1496
+ referenceSources: reference.sources,
1497
+ referenceLocalImports,
1498
+ currentFixtureProposal: current,
1499
+ independentCoverageReview: reviewed.value,
1500
+ revision: sequence
1501
+ },
1502
+ schemaName: "routekit_repository_fixture_suite_v1",
1503
+ outputSchema: FixtureProposal,
1504
+ maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("oracle-designer")
1505
+ });
1506
+ calls.push(repaired.call);
1507
+ yield* Effect.try({
1508
+ try: () => assertRepositoryFixtureScopeCoverageV1(repaired.value, requiredFixtureScope),
1509
+ catch: (cause) => cause instanceof RepositoryFoundryError
1510
+ ? cause
1511
+ : failure("coverage repair changed or omitted the frozen scope", cause)
1512
+ });
1513
+ const validated = yield* validateRepositoryFixturesV1({
1514
+ oracleCoverageWitnessVersion: input.oracleCoverageWitnessVersion,
1515
+ oracleRepairFeedbackVersion: input.oracleRepairFeedbackVersion,
1516
+ repositoryRoot: input.repositoryRoot,
1517
+ operationId: `${input.operationId}:oracle-coverage-repair:${String(sequence)}`,
1518
+ seed,
1519
+ suite: fixtureSuiteFrom(input.caseId, repaired.value),
1520
+ allowedTestPaths: testPaths,
1521
+ context: {
1522
+ ...modelContext,
1523
+ visible: authored.visible,
1524
+ preChangeSources: initial.sources,
1525
+ referenceSources: reference.sources,
1526
+ referenceLocalImports,
1527
+ frozenScopeCoverage: repaired.value.scopeCoverage
1528
+ },
1529
+ modelPlan: input.modelPlan,
1530
+ maximumRepairAttempts: maximumFixtureRepairAttempts -
1531
+ fixturePreflightRepairCalls -
1532
+ (separateFixtureRepairBudget ? 0 : executedOracleHillClimbCalls)
1533
+ });
1534
+ fixturePreflightRepairCalls += validated.modelCalls.length;
1535
+ calls.push(...validated.modelCalls);
1536
+ current = { ...repaired.value, overlays: validated.suite.overlays };
1537
+ }
1538
+ });
1539
+ if (input.maximumOracleCoverageRevisions !== undefined) {
1540
+ fixtureProposal = yield* ensureOracleCoverage(fixtureProposal, "initial", oracleAuthoringModelCalls);
1541
+ fixtureSuite = fixtureSuiteFrom(input.caseId, fixtureProposal);
1542
+ }
1543
+ yield* reportFoundryStageV1("valid-solutions");
1544
+ let [validGenerationA, validGenerationB] = yield* Effect.all([
1545
+ generateSolution({
1546
+ plan: input.modelPlan,
1547
+ role: "valid-solution-generator-a",
1548
+ operationId: `${input.operationId}:valid-solution-generator-a`,
1549
+ instructions: VALID_SOLUTION_A_INSTRUCTIONS,
1550
+ input: {
1551
+ ...modelContext,
1552
+ visible: authored.visible,
1553
+ allowedSolutionPaths: [...allowedSolutionPaths],
1554
+ preChangeSources: initial.sources,
1555
+ ...(input.validControlRegenerationSequence === undefined
1556
+ ? {}
1557
+ : {
1558
+ validControlRegeneration: {
1559
+ sequence: input.validControlRegenerationSequence,
1560
+ directive: "Generate a new independent valid control; do not repeat a prior autonomous attempt."
1561
+ }
1562
+ })
1563
+ },
1564
+ schemaName: "routekit_repository_valid_solution_a_v1",
1565
+ maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("valid-solution-generator-a")
1566
+ }),
1567
+ generateSolution({
1568
+ plan: input.modelPlan,
1569
+ role: "valid-solution-generator-b",
1570
+ operationId: `${input.operationId}:valid-solution-generator-b`,
1571
+ instructions: solverBInstructions,
1572
+ input: {
1573
+ ...modelContext,
1574
+ visible: authored.visible,
1575
+ allowedSolutionPaths: [...allowedSolutionPaths],
1576
+ preChangeSources: initial.sources,
1577
+ ...(input.validControlRegenerationSequence === undefined
1578
+ ? {}
1579
+ : {
1580
+ validControlRegeneration: {
1581
+ sequence: input.validControlRegenerationSequence,
1582
+ directive: "Generate a new behavior-preserving valid control; do not repeat a prior autonomous attempt."
1583
+ }
1584
+ })
1585
+ },
1586
+ schemaName: "routekit_repository_valid_solution_b_v1",
1587
+ maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("valid-solution-generator-b")
1588
+ })
1589
+ ], { concurrency: 2 });
1590
+ const validSolutionModelCalls = [
1591
+ validGenerationA.call,
1592
+ validGenerationB.call
1593
+ ];
1594
+ const validateValidSolutionProposals = () => {
1595
+ const fromA = validateSolutions(validGenerationA.value, {
1596
+ kind: "valid",
1597
+ commit: seed.initialState.commit,
1598
+ allowedPaths: allowedSolutionPaths,
1599
+ allowedRewardHackingPaths: rewardHackingPaths,
1600
+ minimum: 1,
1601
+ exact: 1
1602
+ });
1603
+ const fromB = validateSolutions(validGenerationB.value, {
1604
+ kind: "valid",
1605
+ commit: seed.initialState.commit,
1606
+ allowedPaths: allowedSolutionPaths,
1607
+ allowedRewardHackingPaths: rewardHackingPaths,
1608
+ minimum: 1,
1609
+ exact: 1
1610
+ });
1611
+ return [fromA, fromB];
1612
+ };
1613
+ // Pair repairs and critic repairs spend the same existing per-role reserve.
1614
+ const validSolutionRepairCounts = { a: 0, b: 0 };
1615
+ const repairValidSolution = Effect.fnUntraced(function* (solver, feedback) {
1616
+ if (validSolutionRepairCounts[solver] >= REPOSITORY_FOUNDRY_VALID_SOLUTION_REPAIR_ATTEMPTS) {
1617
+ return yield* failure([
1618
+ `valid-control-repair-exhausted: solver ${solver.toUpperCase()} exhausted its shared independent-solution repair allowance`,
1619
+ ...(feedback.pairContractFindings ?? []).map(({ code, detail }) => `${code}: ${detail}`)
1620
+ ].join("; "));
1621
+ }
1622
+ const repairSequence = ++validSolutionRepairCounts[solver];
1623
+ const role = solver === "a" ? "valid-solution-generator-a" : "valid-solution-generator-b";
1624
+ const repaired = yield* generateSolution({
1625
+ plan: input.modelPlan,
1626
+ role,
1627
+ operationId: `${input.operationId}:${role}:repair-${String(repairSequence)}`,
1628
+ instructions: `${solver === "a" ? VALID_SOLUTION_A_INSTRUCTIONS : solverBInstructions}\n\n${VALID_SOLUTION_REPAIR_INSTRUCTIONS}`,
1629
+ input: {
1630
+ ...modelContext,
1631
+ visible: authored.visible,
1632
+ allowedSolutionPaths: [...allowedSolutionPaths],
1633
+ preChangeSources: initial.sources,
1634
+ repairSequence,
1635
+ priorOwnProposal: solver === "a" ? validGenerationA.rawProposal : validGenerationB.rawProposal,
1636
+ ...feedback
1637
+ },
1638
+ schemaName: solver === "a"
1639
+ ? "routekit_repository_valid_solution_a_v1"
1640
+ : "routekit_repository_valid_solution_b_v1",
1641
+ maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1(role)
1642
+ });
1643
+ if (solver === "a")
1644
+ validGenerationA = repaired;
1645
+ else
1646
+ validGenerationB = repaired;
1647
+ validSolutionModelCalls.push(repaired.call);
1648
+ });
1649
+ const ensureValidSolutionPair = Effect.fnUntraced(function* (preferredSolver) {
1650
+ for (;;) {
1651
+ const [fromA, fromB] = yield* Effect.try({
1652
+ try: validateValidSolutionProposals,
1653
+ catch: (cause) => cause instanceof RepositoryFoundryError
1654
+ ? cause
1655
+ : failure("valid solution proposal failed validation", cause)
1656
+ });
1657
+ const findings = validControlPairFindings(fromA, fromB, initial.sources, input.validControlReviewVersion);
1658
+ if (findings.length === 0)
1659
+ return [...fromA, ...fromB];
1660
+ const solver = findings.some(({ code }) => code === "valid-control-invalid-a-kind")
1661
+ ? "a"
1662
+ : findings.some(({ code }) => code === "valid-control-invalid-b-kind")
1663
+ ? "b"
1664
+ : preferredSolver;
1665
+ yield* repairValidSolution(solver, { pairContractFindings: findings });
1666
+ }
1667
+ });
1668
+ let validSolutions = yield* ensureValidSolutionPair("b");
1669
+ yield* reportFoundryStageV1("adversaries");
1670
+ const adversaryGeneration = yield* generateSolution({
1671
+ plan: input.modelPlan,
1672
+ role: "adversary",
1673
+ operationId: `${input.operationId}:adversary`,
1674
+ instructions: ADVERSARY_INSTRUCTIONS,
1675
+ input: {
1676
+ ...modelContext,
1677
+ visible: authored.visible,
1678
+ allowedSolutionPaths: [...allowedSolutionPaths],
1679
+ allowedRewardHackingPaths: [...rewardHackingPaths].sort(),
1680
+ preChangeSources: initial.sources,
1681
+ developmentFixtureProposal: fixtureProposal,
1682
+ adversaryFamilies: family.adversaryFamilies,
1683
+ excludedNearMisses: family.excludedNearMisses
1684
+ },
1685
+ schemaName: "routekit_repository_adversarial_solutions_v1",
1686
+ maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("adversary")
1687
+ });
1688
+ const adversarialSolutions = yield* Effect.try({
1689
+ try: () => {
1690
+ const validated = validateSolutions(adversaryGeneration.value, {
1691
+ kind: "wrong",
1692
+ commit: seed.initialState.commit,
1693
+ allowedPaths: allowedSolutionPaths,
1694
+ allowedRewardHackingPaths: rewardHackingPaths,
1695
+ minimum: 5
1696
+ });
1697
+ assertRepositorySemanticControlPopulationV1(validated, seed.environment.protectedControlPaths);
1698
+ return validated;
1699
+ },
1700
+ catch: (cause) => cause instanceof RepositoryFoundryError
1701
+ ? cause
1702
+ : failure("adversary proposal failed validation", cause)
1703
+ });
1704
+ let solutions = [...validSolutions, ...adversarialSolutions];
1705
+ const prepareSolutionReview = (role) => prepareRepositorySemanticReviewV1({
1706
+ role,
1707
+ modelContext,
1708
+ visible: authored.visible,
1709
+ targetBehavior: seed.targetBehavior,
1710
+ preChangeSources: initial.sources,
1711
+ solutions,
1712
+ comparisonSolutionIds: validSolutions.map((solution) => solution.id),
1713
+ ...(input.validControlReviewVersion === undefined
1714
+ ? {}
1715
+ : { validControlReviewVersion: input.validControlReviewVersion })
1716
+ });
1717
+ let preparedCriticA = prepareSolutionReview("solution-critic-a");
1718
+ let preparedCriticB = prepareSolutionReview("solution-critic-b");
1719
+ let [solutionCriticA, solutionCriticB] = yield* Effect.all([
1720
+ executeRepositorySemanticReviewV1({
1721
+ modelPlan: input.modelPlan,
1722
+ operationId: `${input.operationId}:solution-critic-a`,
1723
+ prepared: preparedCriticA
1724
+ }),
1725
+ executeRepositorySemanticReviewV1({
1726
+ modelPlan: input.modelPlan,
1727
+ operationId: `${input.operationId}:solution-critic-b`,
1728
+ prepared: preparedCriticB
1729
+ })
1730
+ ], { concurrency: 2 });
1731
+ const solutionReviewModelCalls = [
1732
+ solutionCriticA.call,
1733
+ solutionCriticB.call
1734
+ ];
1735
+ const collectSolutionReviewDisagreements = () => [
1736
+ ...solutionReviewDisagreements(solutionCriticA.value, solutions, "solution critic A"),
1737
+ ...solutionReviewDisagreements(solutionCriticB.value, solutions, "solution critic B")
1738
+ ];
1739
+ const collectIndependenceFindings = () => reviewIndependence
1740
+ ? [
1741
+ ...new Map([
1742
+ ...repositorySemanticReviewIndependenceFindingsV1(solutionCriticA.value, validSolutions, "solution critic A", input.validControlReviewVersion, preparedCriticA.input),
1743
+ ...repositorySemanticReviewIndependenceFindingsV1(solutionCriticB.value, validSolutions, "solution critic B", input.validControlReviewVersion, preparedCriticB.input)
1744
+ ].map((finding) => [finding.code, finding])).values()
1745
+ ]
1746
+ : [];
1747
+ let independenceFindings = yield* Effect.try({
1748
+ try: collectIndependenceFindings,
1749
+ catch: (cause) => cause instanceof RepositoryFoundryError
1750
+ ? cause
1751
+ : failure("implementation independence review failed validation", cause)
1752
+ });
1753
+ let reviewDisagreements = yield* Effect.try({
1754
+ try: collectSolutionReviewDisagreements,
1755
+ catch: (cause) => cause instanceof RepositoryFoundryError
1756
+ ? cause
1757
+ : failure("solution review failed validation", cause)
1758
+ });
1759
+ for (let repairSequence = 1; (reviewDisagreements.length > 0 || independenceFindings.length > 0) &&
1760
+ repairSequence <= REPOSITORY_FOUNDRY_VALID_SOLUTION_REPAIR_ATTEMPTS; repairSequence += 1) {
1761
+ const validAIds = new Set(validGenerationA.value.solutions.map((solution) => solution.id));
1762
+ const validBIds = new Set(validGenerationB.value.solutions.map((solution) => solution.id));
1763
+ const adversaryIds = new Set(adversarialSolutions.map((solution) => solution.id));
1764
+ if (reviewDisagreements.some(({ review }) => adversaryIds.has(review.solutionId))) {
1765
+ yield* Effect.try({
1766
+ try: () => {
1767
+ validateSolutionReview(solutionCriticA.value, solutions, "solution critic A");
1768
+ validateSolutionReview(solutionCriticB.value, solutions, "solution critic B");
1769
+ },
1770
+ catch: (cause) => cause instanceof RepositoryFoundryError
1771
+ ? cause
1772
+ : failure("solution review failed validation", cause)
1773
+ });
1774
+ return yield* failure("solution review disagreement for an adversary unexpectedly passed validation");
1775
+ }
1776
+ const repairA = reviewDisagreements.some(({ review }) => validAIds.has(review.solutionId));
1777
+ const repairB = reviewDisagreements.some(({ review }) => validBIds.has(review.solutionId)) ||
1778
+ (!repairA && independenceFindings.length > 0);
1779
+ if (!repairA && !repairB) {
1780
+ break;
1781
+ }
1782
+ const independenceFeedback = independenceFindings.length > 0 ? { pairContractFindings: independenceFindings } : {};
1783
+ if (independenceFindings.length > 0) {
1784
+ yield* reportFoundryStageV1("valid-solutions", `implementation independence unresolved; bounded repair ${String(repairSequence)} before tournaments`);
1785
+ }
1786
+ const criticFindingsFor = (ids) => reviewDisagreements
1787
+ .filter(({ review }) => ids.has(review.solutionId))
1788
+ .map(({ reviewer, review }) => ({
1789
+ reviewer,
1790
+ solutionId: review.solutionId,
1791
+ classification: review.classification,
1792
+ // Critics see the population; their free-form prose and invented IDs
1793
+ // may contain peer material. Preserve it in private artifacts only.
1794
+ detail: "Reassess your own implementation against the full visible contract and the listed behavior clauses; independent review did not establish this proposal as valid.",
1795
+ violatedBehaviorIds: [
1796
+ ...new Set(review.violatedBehaviorIds.filter((id) => seed.targetBehavior.some((behavior) => behavior.id === id)))
1797
+ ]
1798
+ }));
1799
+ if (repairA) {
1800
+ yield* repairValidSolution("a", {
1801
+ priorCriticFindings: criticFindingsFor(validAIds),
1802
+ ...independenceFeedback
1803
+ });
1804
+ }
1805
+ if (repairB) {
1806
+ yield* repairValidSolution("b", {
1807
+ priorCriticFindings: criticFindingsFor(validBIds),
1808
+ ...independenceFeedback
1809
+ });
1810
+ }
1811
+ validSolutions = yield* ensureValidSolutionPair(repairB ? "b" : "a");
1812
+ solutions = [...validSolutions, ...adversarialSolutions];
1813
+ preparedCriticA = prepareSolutionReview("solution-critic-a");
1814
+ preparedCriticB = prepareSolutionReview("solution-critic-b");
1815
+ [solutionCriticA, solutionCriticB] = yield* Effect.all([
1816
+ executeRepositorySemanticReviewV1({
1817
+ modelPlan: input.modelPlan,
1818
+ operationId: `${input.operationId}:solution-critic-a:repair-${String(repairSequence)}`,
1819
+ prepared: preparedCriticA
1820
+ }),
1821
+ executeRepositorySemanticReviewV1({
1822
+ modelPlan: input.modelPlan,
1823
+ operationId: `${input.operationId}:solution-critic-b:repair-${String(repairSequence)}`,
1824
+ prepared: preparedCriticB
1825
+ })
1826
+ ], { concurrency: 2 });
1827
+ solutionReviewModelCalls.push(solutionCriticA.call, solutionCriticB.call);
1828
+ independenceFindings = yield* Effect.try({
1829
+ try: collectIndependenceFindings,
1830
+ catch: (cause) => cause instanceof RepositoryFoundryError
1831
+ ? cause
1832
+ : failure("repaired implementation independence review failed validation", cause)
1833
+ });
1834
+ reviewDisagreements = yield* Effect.try({
1835
+ try: collectSolutionReviewDisagreements,
1836
+ catch: (cause) => cause instanceof RepositoryFoundryError
1837
+ ? cause
1838
+ : failure("repaired solution review failed validation", cause)
1839
+ });
1840
+ }
1841
+ yield* Effect.try({
1842
+ try: () => {
1843
+ validateSolutionReview(solutionCriticA.value, solutions, "solution critic A");
1844
+ validateSolutionReview(solutionCriticB.value, solutions, "solution critic B");
1845
+ if (independenceFindings.length > 0) {
1846
+ throw failure(`valid-control-independence-repair-exhausted: independent implementation mechanisms were not established before tournaments; ${independenceFindings.map(({ code }) => code).join(", ")}`);
1847
+ }
1848
+ },
1849
+ catch: (cause) => {
1850
+ // Only the exhausted, well-formed valid-control semantic path earns a
1851
+ // canonical candidate-terminal reason. Malformed, uncertain, mixed, or
1852
+ // independence disagreements retain their existing unclassified error.
1853
+ if (cause instanceof RepositoryFoundryError &&
1854
+ independenceFindings.length === 0 &&
1855
+ reviewDisagreements.length > 0 &&
1856
+ reviewDisagreements.every(({ review }) => validSolutions.some((solution) => solution.id === review.solutionId && solution.expectedClass === "valid") &&
1857
+ review.classification === "wrong" &&
1858
+ review.detail.trim().length > 0 &&
1859
+ review.violatedBehaviorIds.length > 0 &&
1860
+ review.violatedBehaviorIds.every((id) => seed.targetBehavior.some((behavior) => behavior.id === id)))) {
1861
+ return failure(`valid-control-repair-exhausted: valid-control semantic review remained unresolved after bounded repairs; ${cause.detail}`, cause);
1862
+ }
1863
+ return cause instanceof RepositoryFoundryError
1864
+ ? cause
1865
+ : failure("solution review failed validation", cause);
1866
+ }
1867
+ });
1868
+ yield* reportFoundryStageV1("tournament");
1869
+ const weakOracle = yield* buildHistoricalHiddenOracleV1({
1870
+ repositoryRoot: input.repositoryRoot,
1871
+ caseId: input.caseId,
1872
+ seed: seed,
1873
+ additionalSolutions: solutions,
1874
+ ...(input.oracleConcurrency === undefined
1875
+ ? {}
1876
+ : { oracleConcurrency: input.oracleConcurrency }),
1877
+ repetitions: input.oracleRepetitions ?? 2
1878
+ });
1879
+ const weakAdequacy = evaluateRepositoryOracleAdequacyV1(weakOracle);
1880
+ const compile = (suite) => compileReviewedHistoricalCaseV1({
1881
+ repositoryRoot: input.repositoryRoot,
1882
+ map: input.map,
1883
+ seed: seed,
1884
+ blueprint: {
1885
+ caseId: input.caseId,
1886
+ ...dimensionContext,
1887
+ visible: authored.visible,
1888
+ specificationReviews: authored.reviews,
1889
+ finalFixtureSuite: suite,
1890
+ solutions,
1891
+ trajectoryPolicy,
1892
+ ...(weakAdequacy.falseAcceptedSolutionIds.length === 0
1893
+ ? {}
1894
+ : {
1895
+ weakOracleTrial: {
1896
+ solutions,
1897
+ expectedFalseAcceptedSolutionIds: weakAdequacy.falseAcceptedSolutionIds
1898
+ }
1899
+ }),
1900
+ oracleRepetitions: input.oracleRepetitions ?? 2,
1901
+ ...(input.oracleConcurrency === undefined
1902
+ ? {}
1903
+ : { oracleConcurrency: input.oracleConcurrency })
1904
+ }
1905
+ });
1906
+ let compiled = yield* compile(fixtureSuite);
1907
+ yield* reportFoundryStageV1("compile-reviewed-case");
1908
+ const hillClimbAttempts = [];
1909
+ const hillClimbCalls = [];
1910
+ const maximumHillClimbAttempts = maximumOracleRepairAttempts - (separateFixtureRepairBudget ? 0 : fixturePreflightRepairCalls);
1911
+ const maximumIterations = maximumHillClimbAttempts + (input.oracleCoverageWitnessVersion === 2 ? 1 : 0);
1912
+ for (let sequence = 1; sequence <= maximumIterations; sequence += 1) {
1913
+ if (input.oracleCoverageWitnessVersion === 2 && compiled.benchmarkCase.adequacy.admitted) {
1914
+ const reviewed = yield* ensureOracleCoverage(fixtureProposal, "before-freeze", freezeCoverageModelCalls);
1915
+ const reviewedSuite = fixtureSuiteFrom(input.caseId, reviewed);
1916
+ if (JSON.stringify(reviewedSuite) !== JSON.stringify(fixtureSuite)) {
1917
+ // New isolated witnesses can expose a false rejection on an independent
1918
+ // valid control. Keep them as obligations and use the remaining ordinary
1919
+ // repair allowance; freezing their code here would make that bias fatal.
1920
+ fixtureProposal = reviewed;
1921
+ fixtureSuite = reviewedSuite;
1922
+ yield* reportFoundryStageV1("tournament", "revalidating development controls after new coverage witnesses");
1923
+ compiled = yield* compile(reviewedSuite);
1924
+ }
1925
+ }
1926
+ const beforeAdequacy = compiled.benchmarkCase.adequacy;
1927
+ const fixtureEvidenceFailures = repairableRepositoryFixtureEvidenceV1(compiled.benchmarkCase.hidden, beforeAdequacy);
1928
+ if ((beforeAdequacy.falseAcceptedSolutionIds.length === 0 &&
1929
+ beforeAdequacy.falseRejectedSolutionIds.length === 0 &&
1930
+ fixtureEvidenceFailures.length === 0) ||
1931
+ (beforeAdequacy.unstableSolutionIds.length > 0 && fixtureEvidenceFailures.length === 0))
1932
+ break;
1933
+ if (sequence > maximumHillClimbAttempts)
1934
+ break;
1935
+ yield* reportFoundryStageV1("hill-climb");
1936
+ const falseAcceptIds = new Set(beforeAdequacy.falseAcceptedSolutionIds);
1937
+ const falseRejectIds = new Set(beforeAdequacy.falseRejectedSolutionIds);
1938
+ const inconclusiveIds = new Set(fixtureEvidenceFailures.map((target) => target.solutionId));
1939
+ const repair = yield* languageModel.generateStructured({
1940
+ plan: input.modelPlan,
1941
+ role: "hill-climb-planner",
1942
+ operationId: `${input.operationId}:hill-climb-planner:${String(sequence)}`,
1943
+ instructions: repositoryFixtureAuthoringInstructionsV1(HILL_CLIMB_INSTRUCTIONS, input.oracleCoverageWitnessVersion, input.oracleRepairFeedbackVersion),
1944
+ input: {
1945
+ ...modelContext,
1946
+ visible: authored.visible,
1947
+ targetBehavior: seed.targetBehavior,
1948
+ currentFixtureProposal: fixtureProposal,
1949
+ ...(input.oracleCoverageWitnessVersion !== undefined
1950
+ ? {
1951
+ retainedCoverageFixtureIds: [...retainedCoverageFixtures.keys()],
1952
+ retainedCoverageInstruction: input.oracleCoverageWitnessVersion === 2
1953
+ ? "These fixtures resolved executed coverage counterexamples. Preserve their identities, metadata and reviewed paths. You may revise their test mechanics to remove implementation bias while preserving each original behavioral obligation and exactly one named test per retained witness. The host replays every changed witness twice on its original reference and isolated mutant; revisions must pass the reference and fail the original mutant through an attributable assertion. Whole-suite valid and adversarial checks still govern adoption."
1954
+ : "These fixtures resolved executed coverage counterexamples and the host retains their exact metadata and overlays in every replacement. Preserve them; repair other fixtures or add coverage without deleting these regressions."
1955
+ }
1956
+ : {}),
1957
+ adequacy: beforeAdequacy,
1958
+ fixtureEvidenceFailures,
1959
+ inconclusiveSolutions: solutions
1960
+ .filter((solution) => inconclusiveIds.has(solution.id))
1961
+ .map((solution) => ({
1962
+ id: solution.id,
1963
+ expectedClass: solution.expectedClass,
1964
+ family: solution.family,
1965
+ fileOverrides: solution.fileOverrides
1966
+ })),
1967
+ falseAcceptedWrongSolutions: adversarialSolutions
1968
+ .filter((solution) => falseAcceptIds.has(solution.id))
1969
+ .map((solution) => ({
1970
+ id: solution.id,
1971
+ family: solution.family,
1972
+ fileOverrides: solution.fileOverrides
1973
+ })),
1974
+ falseRejectedValidSolutions: validSolutions
1975
+ .filter((solution) => falseRejectIds.has(solution.id))
1976
+ .map((solution) => ({
1977
+ id: solution.id,
1978
+ family: solution.family,
1979
+ fileOverrides: solution.fileOverrides
1980
+ })),
1981
+ allValidSolutions: validSolutions.map((solution) => ({
1982
+ id: solution.id,
1983
+ family: solution.family,
1984
+ fileOverrides: solution.fileOverrides
1985
+ })),
1986
+ allowedTestPaths: [...testPaths],
1987
+ referenceSources: reference.sources,
1988
+ referenceLocalImports,
1989
+ priorRejectedAttempts: hillClimbAttempts
1990
+ .filter((attempt) => attempt.decision === "rejected")
1991
+ .map((attempt) => ({
1992
+ sequence: attempt.sequence,
1993
+ rationale: attempt.proposal.rationale,
1994
+ rejectionReasons: attempt.rejectionReasons,
1995
+ candidateAdequacy: attempt.candidateAdequacy,
1996
+ ...(attempt.candidateFixtureProposal === undefined
1997
+ ? {}
1998
+ : { candidateFixtureProposal: attempt.candidateFixtureProposal }),
1999
+ ...(attempt.candidateObservations === undefined
2000
+ ? {}
2001
+ : { candidateObservations: attempt.candidateObservations }),
2002
+ ...(attempt.referencePreflight === undefined
2003
+ ? {}
2004
+ : { referencePreflight: attempt.referencePreflight }),
2005
+ ...(attempt.coverageWitnessRevalidation === undefined
2006
+ ? {}
2007
+ : { coverageWitnessRevalidation: attempt.coverageWitnessRevalidation })
2008
+ }))
2009
+ },
2010
+ schemaName: "routekit_repository_oracle_hill_climb_v1",
2011
+ outputSchema: HillClimbProposal,
2012
+ maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("hill-climb-planner")
2013
+ });
2014
+ const targetedFalseAccepts = new Set(repair.value.targetedFalseAcceptIds);
2015
+ const targetedFalseRejects = new Set(repair.value.targetedFalseRejectIds);
2016
+ const targetedInconclusive = new Set(repair.value.targetedInconclusiveIds ?? []);
2017
+ if (repair.value.rationale.trim().length === 0 ||
2018
+ [...falseAcceptIds].some((id) => !targetedFalseAccepts.has(id)) ||
2019
+ [...falseRejectIds].some((id) => !targetedFalseRejects.has(id)) ||
2020
+ [...inconclusiveIds].some((id) => !targetedInconclusive.has(id)) ||
2021
+ [...targetedInconclusive].some((id) => !inconclusiveIds.has(id))) {
2022
+ return yield* failure(`hill-climb attempt ${String(sequence)} did not account for every observed false accept, false reject, and inconclusive fixture result`);
2023
+ }
2024
+ let candidateFixtureProposal = retainCoverageFixtures(repair.value.fixtureProposal);
2025
+ yield* Effect.try({
2026
+ try: () => assertRepositoryFixtureScopeCoverageV1(candidateFixtureProposal, requiredFixtureScope),
2027
+ catch: (cause) => cause instanceof RepositoryFoundryError
2028
+ ? cause
2029
+ : failure("replacement fixture scope coverage failed validation", cause)
2030
+ });
2031
+ let candidateFixtureSuite = fixtureSuiteFrom(input.caseId, candidateFixtureProposal);
2032
+ hillClimbCalls.push(repair.call);
2033
+ executedOracleHillClimbCalls += 1;
2034
+ if (separateFixtureRepairBudget) {
2035
+ yield* reportFoundryStageV1("hill-climb", `reference preflight for behavioral revision ${String(sequence)}`);
2036
+ const preflight = yield* validateRepositoryFixturesV1({
2037
+ oracleCoverageWitnessVersion: input.oracleCoverageWitnessVersion,
2038
+ oracleRepairFeedbackVersion: input.oracleRepairFeedbackVersion,
2039
+ repositoryRoot: input.repositoryRoot,
2040
+ operationId: `${input.operationId}:hill-climb:${String(sequence)}`,
2041
+ seed,
2042
+ suite: candidateFixtureSuite,
2043
+ allowedTestPaths: testPaths,
2044
+ context: {
2045
+ ...modelContext,
2046
+ visible: authored.visible,
2047
+ preChangeSources: initial.sources,
2048
+ referenceSources: reference.sources,
2049
+ referenceLocalImports,
2050
+ ...(requiredFixtureScope === undefined
2051
+ ? {}
2052
+ : { frozenScopeCoverage: candidateFixtureProposal.scopeCoverage })
2053
+ },
2054
+ modelPlan: input.modelPlan,
2055
+ maximumRepairAttempts: Math.max(0, maximumFixtureRepairAttempts - fixturePreflightRepairCalls),
2056
+ onRepairAttempt: () => Effect.sync(() => {
2057
+ fixturePreflightRepairCalls += 1;
2058
+ }),
2059
+ onRepairCall: (call) => Effect.sync(() => {
2060
+ hillClimbCalls.push(call);
2061
+ })
2062
+ }).pipe(Effect.match({
2063
+ onFailure: (error) => ({ ok: false, error }),
2064
+ onSuccess: (value) => ({ ok: true, value })
2065
+ }));
2066
+ if (!preflight.ok) {
2067
+ if (preflight.error.detail.startsWith("fixture preflight stopped on infrastructure"))
2068
+ return yield* preflight.error;
2069
+ const referencePreflight = {
2070
+ status: "rejected",
2071
+ detail: preflight.error.detail,
2072
+ diagnostics: preflight.error.cause
2073
+ };
2074
+ const attempt = {
2075
+ sequence,
2076
+ beforeAdequacy,
2077
+ proposal: repair.value,
2078
+ referencePreflight,
2079
+ decision: "rejected",
2080
+ rejectionReasons: [preflight.error.detail],
2081
+ afterAdequacy: beforeAdequacy
2082
+ };
2083
+ hillClimbAttempts.push(attempt);
2084
+ yield* reportFoundryAuthoringArtifactV1({
2085
+ version: 1,
2086
+ operationId: `${input.operationId}:hill-climb:${String(sequence)}:preflight-rejected`,
2087
+ role: "fixture-validator",
2088
+ model: "local-replay",
2089
+ reasoningEffort: "none",
2090
+ schemaName: "routekit_repository_fixture_revision_rejection_v1",
2091
+ validation: "validated",
2092
+ request: {
2093
+ instructions: "Reject a replacement that fails reference preflight; retain the incumbent.",
2094
+ input: { sequence, suite: candidateFixtureSuite },
2095
+ jsonSchema: { type: "object", properties: {}, additionalProperties: false },
2096
+ maximumOutputTokens: 0
2097
+ },
2098
+ response: { text: JSON.stringify(referencePreflight) },
2099
+ value: { diagnosticOnly: true, admissionEvidence: false, ...attempt }
2100
+ });
2101
+ yield* reportFoundryStageV1("hill-climb", `revision ${String(sequence)} rejected by reference preflight; incumbent retained`);
2102
+ continue;
2103
+ }
2104
+ candidateFixtureSuite = preflight.value.suite;
2105
+ candidateFixtureProposal = {
2106
+ ...candidateFixtureProposal,
2107
+ overlays: candidateFixtureSuite.overlays
2108
+ };
2109
+ }
2110
+ let coverageWitnessRevalidation;
2111
+ if (input.oracleCoverageWitnessVersion === 2) {
2112
+ const revisions = yield* validateRepositoryOracleCoverageWitnessRevisionsV1({
2113
+ operationId: `${input.operationId}:hill-climb:${String(sequence)}:witness-revision`,
2114
+ repositoryRoot: input.repositoryRoot,
2115
+ seed,
2116
+ suite: candidateFixtureSuite,
2117
+ retained: [...retainedCoverageFixtures.values()]
2118
+ });
2119
+ coverageWitnessRevalidation = {
2120
+ status: revisions.every((revision) => revision.status === "validated")
2121
+ ? "passed"
2122
+ : "rejected",
2123
+ revisions
2124
+ };
2125
+ if (coverageWitnessRevalidation.status === "rejected") {
2126
+ hillClimbAttempts.push({
2127
+ sequence,
2128
+ beforeAdequacy,
2129
+ proposal: repair.value,
2130
+ ...(separateFixtureRepairBudget
2131
+ ? { referencePreflight: { status: "passed" } }
2132
+ : {}),
2133
+ coverageWitnessRevalidation,
2134
+ decision: "rejected",
2135
+ rejectionReasons: revisions
2136
+ .filter((revision) => revision.status === "rejected")
2137
+ .map((revision) => `${revision.fixtureId}: ${revision.detail}`),
2138
+ afterAdequacy: beforeAdequacy
2139
+ });
2140
+ yield* reportFoundryStageV1("hill-climb", `revision ${String(sequence)} lost a witnessed behavioral obligation; incumbent retained`);
2141
+ continue;
2142
+ }
2143
+ }
2144
+ const candidateCompiled = yield* compile(candidateFixtureSuite);
2145
+ const candidateAdequacy = candidateCompiled.benchmarkCase.adequacy;
2146
+ const rejectionReasons = assessHillClimbCandidate(beforeAdequacy, candidateAdequacy);
2147
+ const decision = rejectionReasons.length === 0 ? "accepted" : "rejected";
2148
+ if (decision === "accepted") {
2149
+ fixtureProposal = candidateFixtureProposal;
2150
+ fixtureSuite = candidateFixtureSuite;
2151
+ compiled = candidateCompiled;
2152
+ for (const revision of coverageWitnessRevalidation?.revisions ?? []) {
2153
+ const retained = retainedCoverageFixtures.get(revision.fixtureId);
2154
+ retainedCoverageFixtures.set(revision.fixtureId, {
2155
+ ...retained,
2156
+ overlay: revision.overlay
2157
+ });
2158
+ }
2159
+ }
2160
+ hillClimbAttempts.push({
2161
+ sequence,
2162
+ beforeAdequacy,
2163
+ proposal: repair.value,
2164
+ candidateAdequacy,
2165
+ ...(input.oracleRepairFeedbackVersion !== 1 || decision !== "rejected"
2166
+ ? {}
2167
+ : {
2168
+ candidateFixtureProposal,
2169
+ candidateObservations: candidateCompiled.benchmarkCase.hidden.observations.filter((observation) => [
2170
+ ...candidateAdequacy.falseAcceptedSolutionIds,
2171
+ ...candidateAdequacy.falseRejectedSolutionIds,
2172
+ ...candidateAdequacy.unstableSolutionIds
2173
+ ].includes(observation.solutionId))
2174
+ }),
2175
+ ...(separateFixtureRepairBudget
2176
+ ? { referencePreflight: { status: "passed" } }
2177
+ : {}),
2178
+ ...(coverageWitnessRevalidation === undefined ? {} : { coverageWitnessRevalidation }),
2179
+ decision,
2180
+ rejectionReasons,
2181
+ afterAdequacy: compiled.benchmarkCase.adequacy
2182
+ });
2183
+ }
2184
+ if (compiled.benchmarkCase.status !== "valid-library" ||
2185
+ !compiled.benchmarkCase.adequacy.admitted) {
2186
+ return yield* failure([
2187
+ "development tournament remains inadequate after bounded oracle repair; held-out generation was not started",
2188
+ ...compiled.benchmarkCase.adequacy.rejectionReasons,
2189
+ ...compiled.benchmarkCase.rejectionReasons
2190
+ ].join("; "));
2191
+ }
2192
+ // Any development repair invalidates the earlier approval. Recheck the exact
2193
+ // final suite before exposing it to held-out challenges, using the same repair reserve.
2194
+ if (input.maximumOracleCoverageRevisions !== undefined &&
2195
+ input.oracleCoverageWitnessVersion !== 2) {
2196
+ const reviewedFixtureProposal = yield* ensureOracleCoverage(fixtureProposal, "before-freeze", freezeCoverageModelCalls);
2197
+ const reviewedFixtureSuite = fixtureSuiteFrom(input.caseId, reviewedFixtureProposal);
2198
+ if (JSON.stringify(reviewedFixtureSuite) !== JSON.stringify(fixtureSuite)) {
2199
+ const beforeAdequacy = compiled.benchmarkCase.adequacy;
2200
+ const candidate = yield* compile(reviewedFixtureSuite);
2201
+ const regressions = assessHillClimbCandidate(beforeAdequacy, candidate.benchmarkCase.adequacy, false);
2202
+ if (regressions.length > 0 ||
2203
+ candidate.benchmarkCase.status !== "valid-library" ||
2204
+ !candidate.benchmarkCase.adequacy.admitted) {
2205
+ return yield* failure(`oracle coverage repair failed development revalidation before freeze; ${regressions.join("; ")}`);
2206
+ }
2207
+ fixtureProposal = reviewedFixtureProposal;
2208
+ fixtureSuite = reviewedFixtureSuite;
2209
+ compiled = candidate;
2210
+ }
2211
+ }
2212
+ const heldOutAdversaryCalls = [];
2213
+ const heldOutNoveltyFailures = [];
2214
+ let heldOutAdversaryProposal;
2215
+ let heldOutAdversarialSolutions;
2216
+ yield* reportFoundryStageV1("held-out-tournament");
2217
+ for (let sequence = 0; sequence <= 2; sequence += 1) {
2218
+ const generated = yield* generateSolution({
2219
+ plan: input.modelPlan,
2220
+ role: "held-out-adversary",
2221
+ operationId: `${input.operationId}:held-out-adversary${sequence === 0 ? "" : `:repair-${String(sequence)}`}`,
2222
+ instructions: HELD_OUT_ADVERSARY_INSTRUCTIONS,
2223
+ input: {
2224
+ ...modelContext,
2225
+ visible: authored.visible,
2226
+ targetBehavior: seed.targetBehavior,
2227
+ allowedSolutionPaths: [...allowedSolutionPaths],
2228
+ allowedRewardHackingPaths: [...rewardHackingPaths].sort(),
2229
+ preChangeSources: initial.sources,
2230
+ adversaryFamilies: family.adversaryFamilies,
2231
+ excludedNearMisses: family.excludedNearMisses,
2232
+ forbiddenDevelopmentFamilies: [
2233
+ ...new Set(adversarialSolutions.map((solution) => solution.family))
2234
+ ].sort(),
2235
+ priorNoveltyFailures: heldOutNoveltyFailures,
2236
+ oracleState: "frozen-before-held-out-tournament"
2237
+ },
2238
+ schemaName: "routekit_repository_held_out_adversarial_solutions_v1",
2239
+ maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("held-out-adversary")
2240
+ });
2241
+ heldOutAdversaryCalls.push(generated.call);
2242
+ const validation = validateHeldOutAdversaryProposal({
2243
+ proposal: generated.value,
2244
+ commit: seed.initialState.commit,
2245
+ allowedPaths: allowedSolutionPaths,
2246
+ allowedRewardHackingPaths: rewardHackingPaths,
2247
+ protectedControlPaths: seed.environment.protectedControlPaths,
2248
+ development: adversarialSolutions,
2249
+ preChangeSources: initial.sources
2250
+ });
2251
+ if (validation._tag === "valid") {
2252
+ heldOutAdversaryProposal = generated.value;
2253
+ heldOutAdversarialSolutions = validation.solutions;
2254
+ break;
2255
+ }
2256
+ heldOutNoveltyFailures.push(validation.detail);
2257
+ }
2258
+ if (heldOutAdversaryProposal === undefined || heldOutAdversarialSolutions === undefined) {
2259
+ return yield* failure(`held-out adversary proposal failed validation after bounded regeneration: ${heldOutNoveltyFailures.join("; ")}`);
2260
+ }
2261
+ const heldOutSolutionReviewInput = {
2262
+ ...modelContext,
2263
+ visible: authored.visible,
2264
+ targetBehavior: seed.targetBehavior,
2265
+ preChangeSources: initial.sources,
2266
+ oracleState: "frozen-and-not-disclosed",
2267
+ solutions: heldOutAdversarialSolutions.map((solution) => ({
2268
+ id: solution.id,
2269
+ family: solution.family,
2270
+ fileOverrides: solution.fileOverrides
2271
+ }))
2272
+ };
2273
+ const [heldOutSolutionCriticA, heldOutSolutionCriticB] = yield* Effect.all([
2274
+ languageModel.generateStructured({
2275
+ plan: input.modelPlan,
2276
+ role: "held-out-solution-critic-a",
2277
+ operationId: `${input.operationId}:held-out-solution-critic-a`,
2278
+ instructions: solutionCriticInstructions("contract"),
2279
+ input: heldOutSolutionReviewInput,
2280
+ schemaName: "routekit_repository_held_out_solution_review_v1",
2281
+ outputSchema: SolutionReviewProposal,
2282
+ maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("held-out-solution-critic-a")
2283
+ }),
2284
+ languageModel.generateStructured({
2285
+ plan: input.modelPlan,
2286
+ role: "held-out-solution-critic-b",
2287
+ operationId: `${input.operationId}:held-out-solution-critic-b`,
2288
+ instructions: solutionCriticInstructions("integration"),
2289
+ input: heldOutSolutionReviewInput,
2290
+ schemaName: "routekit_repository_held_out_solution_review_v1",
2291
+ outputSchema: SolutionReviewProposal,
2292
+ maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("held-out-solution-critic-b")
2293
+ })
2294
+ ], { concurrency: 2 });
2295
+ yield* Effect.try({
2296
+ try: () => {
2297
+ validateSolutionReview(heldOutSolutionCriticA.value, heldOutAdversarialSolutions, "held-out solution critic A");
2298
+ validateSolutionReview(heldOutSolutionCriticB.value, heldOutAdversarialSolutions, "held-out solution critic B");
2299
+ },
2300
+ catch: (cause) => cause instanceof RepositoryFoundryError
2301
+ ? cause
2302
+ : failure("held-out solution review failed validation", cause)
2303
+ });
2304
+ const heldOutOracle = yield* buildHistoricalHiddenOracleV1({
2305
+ repositoryRoot: input.repositoryRoot,
2306
+ caseId: input.caseId,
2307
+ seed: seed,
2308
+ additionalSolutions: [...validSolutions, ...heldOutAdversarialSolutions],
2309
+ hiddenFixtureSuite: fixtureSuite,
2310
+ ...(input.oracleConcurrency === undefined
2311
+ ? {}
2312
+ : { oracleConcurrency: input.oracleConcurrency }),
2313
+ repetitions: input.oracleRepetitions ?? 2
2314
+ });
2315
+ const heldOutAdequacy = evaluateRepositoryOracleAdequacyV1(heldOutOracle);
2316
+ if (!heldOutAdequacy.admitted) {
2317
+ return yield* failure([
2318
+ "frozen oracle failed the held-out adversary tournament",
2319
+ ...heldOutAdequacy.rejectionReasons
2320
+ ].join("; "));
2321
+ }
2322
+ const oracleCoverageWitnessRevisions = hillClimbAttempts.flatMap((attempt) => attempt.coverageWitnessRevalidation === undefined
2323
+ ? []
2324
+ : [
2325
+ {
2326
+ sequence: attempt.sequence,
2327
+ decision: attempt.decision,
2328
+ revisions: attempt.coverageWitnessRevalidation.revisions
2329
+ }
2330
+ ]);
2331
+ const qualityReviewPacket = {
2332
+ ...(input.validControlReviewVersion === 2 ? { validControlReviewVersion: 2 } : {}),
2333
+ ...(input.reviewInputVersion === undefined
2334
+ ? {}
2335
+ : { reviewInputVersion: input.reviewInputVersion }),
2336
+ ...(input.specificationContractFactsVersion === undefined
2337
+ ? {}
2338
+ : {
2339
+ specificationContractFactsVersion: input.specificationContractFactsVersion,
2340
+ contractFacts: authored.contractFacts
2341
+ }),
2342
+ ...modelContext,
2343
+ visible: authored.visible,
2344
+ targetBehavior: seed.targetBehavior,
2345
+ specificationReviews: authored.reviews,
2346
+ fixtureProposal,
2347
+ ...(input.oracleCoverageWitnessVersion !== undefined
2348
+ ? {
2349
+ oracleCoverageReviews,
2350
+ coverageResolutionPolicy: "Original critic verdicts are preserved. Execution assessments resolve only the fixed hypotheses via a reference-passing, isolated mutant-failing behavioral test and an incumbent replay. Retained tests protect those scenarios; this does not establish exhaustive coverage. Independently assess the actual assertions, witness semantics, valid controls and both tournaments."
2351
+ }
2352
+ : {}),
2353
+ ...(input.oracleCoverageWitnessVersion === 2
2354
+ ? {
2355
+ oracleCoverageWitnessRevisions,
2356
+ witnessRevisionPolicy: "Original witnesses and critic verdicts remain unchanged. A revised witness is adopted only after repeated reference passes, repeated assertion failures on the original isolated mutant, and acceptance of the complete development replacement. Review each revision and its adoption decision; rejected proposals do not replace the incumbent."
2357
+ }
2358
+ : {}),
2359
+ validSolutions,
2360
+ developmentAdversaries: adversarialSolutions,
2361
+ solutionReviews: [solutionCriticA.value, solutionCriticB.value],
2362
+ heldOutAdversaries: heldOutAdversarialSolutions,
2363
+ heldOutSolutionReviews: [heldOutSolutionCriticA.value, heldOutSolutionCriticB.value],
2364
+ heldOutAdequacy,
2365
+ oracleFrozenBeforeHeldOutTournament: true,
2366
+ trajectoryPolicy,
2367
+ trajectoryPolicyReview,
2368
+ trajectoryRevisionAttempts,
2369
+ executableAdequacy: compiled.benchmarkCase.adequacy,
2370
+ developmentOracle: {
2371
+ clauses: compiled.benchmarkCase.hidden.clauses,
2372
+ observations: compiled.benchmarkCase.hidden.observations
2373
+ },
2374
+ heldOutOracle: {
2375
+ clauses: heldOutOracle.clauses,
2376
+ observations: heldOutOracle.observations
2377
+ },
2378
+ hillClimbAttempts: hillClimbAttempts.map((attempt) => ({
2379
+ sequence: attempt.sequence,
2380
+ targetedFalseAcceptIds: attempt.proposal.targetedFalseAcceptIds,
2381
+ targetedFalseRejectIds: attempt.proposal.targetedFalseRejectIds,
2382
+ targetedInconclusiveIds: attempt.proposal.targetedInconclusiveIds ?? [],
2383
+ decision: attempt.decision,
2384
+ rejectionReasons: attempt.rejectionReasons,
2385
+ before: attempt.beforeAdequacy,
2386
+ candidate: attempt.candidateAdequacy,
2387
+ ...(attempt.referencePreflight === undefined
2388
+ ? {}
2389
+ : { referencePreflight: attempt.referencePreflight }),
2390
+ after: attempt.afterAdequacy
2391
+ }))
2392
+ };
2393
+ const qualityEvidenceInventory = qualityReviewEvidenceInventory({
2394
+ targetBehaviorCount: qualityReviewPacket.targetBehavior.length,
2395
+ specificationReviewCount: qualityReviewPacket.specificationReviews.length,
2396
+ fixtureCount: qualityReviewPacket.fixtureProposal.fixtures.length,
2397
+ overlayCount: qualityReviewPacket.fixtureProposal.overlays.length,
2398
+ validSolutionCount: qualityReviewPacket.validSolutions.length,
2399
+ developmentAdversaryCount: qualityReviewPacket.developmentAdversaries.length,
2400
+ solutionReviewCount: qualityReviewPacket.solutionReviews.length,
2401
+ heldOutAdversaryCount: qualityReviewPacket.heldOutAdversaries.length,
2402
+ heldOutSolutionReviewCount: qualityReviewPacket.heldOutSolutionReviews.length,
2403
+ trajectoryRevisionAttemptCount: qualityReviewPacket.trajectoryRevisionAttempts.length,
2404
+ hillClimbAttemptCount: qualityReviewPacket.hillClimbAttempts.length,
2405
+ developmentObservationCount: qualityReviewPacket.developmentOracle.observations.length,
2406
+ heldOutObservationCount: qualityReviewPacket.heldOutOracle.observations.length,
2407
+ ...(input.oracleCoverageWitnessVersion !== undefined
2408
+ ? { oracleCoverageReviewCount: oracleCoverageReviews.length }
2409
+ : {}),
2410
+ ...(input.oracleCoverageWitnessVersion === 2
2411
+ ? { oracleCoverageWitnessRevisionCount: oracleCoverageWitnessRevisions.length }
2412
+ : {})
2413
+ });
2414
+ const qualityReviewInput = {
2415
+ ...qualityReviewPacket,
2416
+ evidenceInventory: qualityEvidenceInventory
2417
+ };
2418
+ const generation = {
2419
+ version: 1,
2420
+ authored,
2421
+ sourceContext: {
2422
+ initialCommit: initialSnapshot.commit,
2423
+ referenceCommit: reference.snapshot.commit,
2424
+ initialSupportingPaths: initialLocalImports.sources.map((source) => source.path),
2425
+ referenceSupportingPaths: referenceLocalImports.sources.map((source) => source.path)
2426
+ },
2427
+ initialFixtureProposal: fixtureGeneration.value,
2428
+ fixtureProposal,
2429
+ ...(input.maximumOracleCoverageRevisions === undefined ? {} : { oracleCoverageReviews }),
2430
+ ...(input.oracleCoverageWitnessVersion === 2 ? { oracleCoverageWitnessRevisions } : {}),
2431
+ trajectoryPolicy,
2432
+ trajectoryPolicyReview,
2433
+ trajectoryRevisionAttempts,
2434
+ validSolutionProposal: {
2435
+ solutions: [...validGenerationA.value.solutions, ...validGenerationB.value.solutions]
2436
+ },
2437
+ adversaryProposal: adversaryGeneration.value,
2438
+ solutionReviews: [solutionCriticA.value, solutionCriticB.value],
2439
+ heldOutAdversaryProposal,
2440
+ heldOutSolutionReviews: [heldOutSolutionCriticA.value, heldOutSolutionCriticB.value],
2441
+ heldOutAdequacy,
2442
+ heldOutOracle,
2443
+ weakAdequacy,
2444
+ hillClimbAttempts,
2445
+ benchmarkCase: compiled.benchmarkCase,
2446
+ modelCalls: [
2447
+ ...authored.modelCalls,
2448
+ ...trajectoryModelCalls,
2449
+ ...oracleAuthoringModelCalls,
2450
+ ...validSolutionModelCalls,
2451
+ adversaryGeneration.call,
2452
+ ...solutionReviewModelCalls,
2453
+ ...hillClimbCalls,
2454
+ ...freezeCoverageModelCalls,
2455
+ ...heldOutAdversaryCalls,
2456
+ heldOutSolutionCriticA.call,
2457
+ heldOutSolutionCriticB.call
2458
+ ]
2459
+ };
2460
+ const reconstruction = yield* Effect.serviceOption(RepositoryFoundryEvidenceReconstruction);
2461
+ const checkpoint = {
2462
+ version: 1,
2463
+ kind: "pending-quality-review",
2464
+ ...(input.validControlReviewVersion === 2 ? { validControlReviewVersion: 2 } : {}),
2465
+ ...(input.reviewInputVersion === undefined
2466
+ ? {}
2467
+ : { reviewInputVersion: input.reviewInputVersion }),
2468
+ ...(input.specificationContractFactsVersion === undefined
2469
+ ? {}
2470
+ : { specificationContractFactsVersion: input.specificationContractFactsVersion }),
2471
+ repositoryRoot: input.repositoryRoot,
2472
+ operationId: input.operationId,
2473
+ modelPlan: input.modelPlan,
2474
+ seed,
2475
+ requiresReplayValidation: Option.isSome(reconstruction),
2476
+ generation,
2477
+ qualityReviewInput,
2478
+ sourceBases: editSources.map((source) => source.content)
2479
+ };
2480
+ yield* reportFoundryAuthoringArtifactV1({
2481
+ version: 1,
2482
+ operationId: `${input.operationId}:quality-review-checkpoint`,
2483
+ role: "checkpoint-writer",
2484
+ model: "local-evidence",
2485
+ reasoningEffort: "none",
2486
+ schemaName: "routekit_repository_quality_review_checkpoint_v1",
2487
+ validation: "validated",
2488
+ request: {
2489
+ instructions: "Complete frozen case before quality review. This checkpoint is not an admission or a model call.",
2490
+ input: {
2491
+ caseId: input.caseId,
2492
+ requiresReplayValidation: checkpoint.requiresReplayValidation
2493
+ },
2494
+ jsonSchema: { type: "object" },
2495
+ maximumOutputTokens: 0
2496
+ },
2497
+ response: { text: "" },
2498
+ value: checkpoint
2499
+ });
2500
+ if (input.checkpointOnly === true) {
2501
+ if (Option.isNone(reconstruction))
2502
+ return yield* failure("checkpoint reconstruction requires exact archived evidence");
2503
+ yield* reconstruction.value.assertConsumed;
2504
+ return { kind: "checkpoint", checkpoint };
2505
+ }
2506
+ const result = yield* reviewRepositoryGeneratedCaseCheckpointV1({ checkpoint });
2507
+ return { kind: "generated", result };
2508
+ });
2509
+ export const generateHistoricalRepositoryCaseV1 = Effect.fn("CaseGeneration.completeHistorical")(function* (input) {
2510
+ const result = yield* generateHistoricalRepositoryCaseStagesV1(input);
2511
+ if (result.kind !== "generated")
2512
+ return yield* failure("generation stopped before final review");
2513
+ return result.result;
2514
+ });
2515
+ export const reconstructRepositoryGeneratedCaseCheckpointV1 = Effect.fn("CaseGeneration.reconstructCheckpoint")(function* (input) {
2516
+ const reconstruction = yield* Effect.serviceOption(RepositoryFoundryEvidenceReconstruction);
2517
+ if (Option.isNone(reconstruction))
2518
+ return yield* failure("checkpoint reconstruction requires an archived-evidence scope");
2519
+ const result = yield* generateHistoricalRepositoryCaseStagesV1({
2520
+ ...input,
2521
+ checkpointOnly: true
2522
+ });
2523
+ if (result.kind !== "checkpoint")
2524
+ return yield* failure("reconstruction did not stop before review");
2525
+ return result.checkpoint;
2526
+ });
2527
+ /** Pure sizing of the full frozen review request, including wire wrapping and schema. */
2528
+ export const prepareRepositoryCaseQualityReviewV1 = (input) => {
2529
+ const checkpoint = input.checkpoint;
2530
+ const reviewInputVersion = input.reviewInputVersion ?? checkpoint.reviewInputVersion;
2531
+ const requestByteLimit = evalAuthoringRequestByteLimit({
2532
+ foundryRole: "quality-reviewer",
2533
+ schemaName: "routekit_repository_generation_quality_review_v1",
2534
+ ...(reviewInputVersion === undefined ? {} : { reviewInputVersion })
2535
+ });
2536
+ const document = Schema.toJsonSchemaDocument(QualityReview);
2537
+ const jsonSchema = strictAuthoringSchema({
2538
+ ...document.schema,
2539
+ ...(Object.keys(document.definitions).length === 0 ? {} : { $defs: document.definitions })
2540
+ }).jsonSchema;
2541
+ const reviewer = repositoryFoundryModelAssignmentV1(checkpoint.modelPlan, "quality-reviewer");
2542
+ const size = (evidence, instructions) => Buffer.byteLength(evalAuthoringResponsesRequestBody({
2543
+ operationId: input.operationId ?? `${checkpoint.operationId}:quality-reviewer`,
2544
+ model: reviewer.model,
2545
+ reasoningEffort: reviewer.reasoningEffort,
2546
+ instructions: `${qualityReviewInstructions(checkpoint.validControlReviewVersion)}\n\n${instructions}`,
2547
+ input: JSON.stringify(evidence),
2548
+ schemaName: "routekit_repository_generation_quality_review_v1",
2549
+ jsonSchema,
2550
+ maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("quality-reviewer")
2551
+ }));
2552
+ const legacy = encodeRepositoryReviewEvidenceV1(checkpoint.qualityReviewInput, checkpoint.sourceBases);
2553
+ const legacyBytes = size(legacy, REPOSITORY_REVIEW_EVIDENCE_INSTRUCTIONS);
2554
+ if (legacyBytes <= REPOSITORY_FOUNDRY_REQUEST_INPUT_CEILING)
2555
+ return {
2556
+ evidence: legacy,
2557
+ instructions: REPOSITORY_REVIEW_EVIDENCE_INSTRUCTIONS,
2558
+ requestBytes: legacyBytes,
2559
+ requestByteLimit
2560
+ };
2561
+ const evidence = encodeRepositoryReviewEvidenceV2(checkpoint.qualityReviewInput, checkpoint.sourceBases);
2562
+ const requestBytes = size(evidence, REPOSITORY_REVIEW_EVIDENCE_V2_INSTRUCTIONS);
2563
+ if (requestBytes > requestByteLimit)
2564
+ throw failure("complete benchmark artifacts exceed the independent quality review input ceiling; frozen checkpoint retained");
2565
+ return {
2566
+ evidence,
2567
+ instructions: REPOSITORY_REVIEW_EVIDENCE_V2_INSTRUCTIONS,
2568
+ requestBytes,
2569
+ requestByteLimit
2570
+ };
2571
+ };
2572
+ export const reviewRepositoryGeneratedCaseCheckpointV1 = Effect.fn("CaseGeneration.reviewCheckpoint")(function* (input) {
2573
+ const reconstruction = yield* Effect.serviceOption(RepositoryFoundryEvidenceReconstruction);
2574
+ if (Option.isSome(reconstruction) || input.checkpoint.requiresReplayValidation)
2575
+ return yield* failure("diagnostic reconstruction cannot perform review or confer admission before real replay validation");
2576
+ const languageModel = yield* RepositoryFoundryLanguageModel;
2577
+ const checkpoint = input.checkpoint;
2578
+ const reviewInputVersion = input.reviewInputVersion ?? checkpoint.reviewInputVersion;
2579
+ const inventory = checkpoint.qualityReviewInput.evidenceInventory;
2580
+ if (!Array.isArray(inventory) || inventory.some((entry) => typeof entry !== "string"))
2581
+ return yield* failure("checkpoint review requires its complete evidence inventory");
2582
+ const qualityEvidenceInventory = inventory;
2583
+ const encoded = yield* Effect.try({
2584
+ try: () => prepareRepositoryCaseQualityReviewV1(input),
2585
+ catch: (cause) => failure("complete benchmark review evidence could not be encoded losslessly", cause)
2586
+ });
2587
+ yield* reportFoundryStageV1("quality-review");
2588
+ const qualityGeneration = yield* languageModel.generateStructured({
2589
+ plan: checkpoint.modelPlan,
2590
+ role: "quality-reviewer",
2591
+ ...(reviewInputVersion === undefined ? {} : { reviewInputVersion }),
2592
+ operationId: input.operationId ?? `${checkpoint.operationId}:quality-reviewer`,
2593
+ instructions: `${qualityReviewInstructions(checkpoint.validControlReviewVersion)}\n\n${encoded.instructions}`,
2594
+ input: encoded.evidence,
2595
+ schemaName: "routekit_repository_generation_quality_review_v1",
2596
+ outputSchema: QualityReview,
2597
+ maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("quality-reviewer")
2598
+ });
2599
+ yield* Effect.try({
2600
+ try: () => validateRepositoryGenerationQualityReviewV1(qualityGeneration.value, qualityEvidenceInventory),
2601
+ catch: (cause) => cause instanceof RepositoryFoundryError
2602
+ ? cause
2603
+ : failure("independent quality review failed validation", cause)
2604
+ });
2605
+ if (qualityGeneration.value.verdict !== "approve" ||
2606
+ qualityGeneration.value.coverageGaps.length > 0 ||
2607
+ qualityGeneration.value.correlatedAssumptions.length > 0)
2608
+ return yield* failure([
2609
+ "independent quality review rejected the generated case",
2610
+ qualityGeneration.value.detail,
2611
+ ...qualityGeneration.value.coverageGaps.map((gap) => `coverage-gap=${gap}`),
2612
+ ...qualityGeneration.value.correlatedAssumptions.map((assumption) => `correlated-assumption=${assumption}`)
2613
+ ].join("; "));
2614
+ yield* reportFoundryStageV1("admit", `status=${checkpoint.generation.benchmarkCase.status}`);
2615
+ return {
2616
+ ...checkpoint.generation,
2617
+ qualityReview: qualityGeneration.value,
2618
+ modelCalls: [...checkpoint.generation.modelCalls, qualityGeneration.call]
2619
+ };
2620
+ });
2621
+ export const makeCaseGeneration = Effect.gen(function* () {
2622
+ const languageModel = yield* RepositoryFoundryLanguageModel;
2623
+ return CaseGeneration.of({
2624
+ reviewSemanticSolutions: (input) => executeRepositorySemanticReviewV1(input).pipe(Effect.provideService(RepositoryFoundryLanguageModel, languageModel)),
2625
+ generateHistorical: (input) => generateHistoricalRepositoryCaseV1(input).pipe(Effect.provideService(RepositoryFoundryLanguageModel, languageModel))
2626
+ });
2627
+ });
2628
+ export const CaseGenerationLive = Layer.effect(CaseGeneration, makeCaseGeneration);