@velum-labs/routekit-eval-setup 1.3.2 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/authoring-responses-request.d.ts +5 -0
- package/dist/adapters/authoring-responses-request.js +38 -0
- package/dist/adapters/evaluation-evidence-freshness.d.ts +7 -0
- package/dist/adapters/evaluation-evidence-freshness.js +76 -0
- package/dist/adapters/git-task-history.d.ts +67 -0
- package/dist/adapters/git-task-history.js +171 -0
- package/dist/adapters/integrated-repository-history.d.ts +21 -0
- package/dist/adapters/integrated-repository-history.js +175 -0
- package/dist/adapters/repository-command-diagnostic.d.ts +8 -0
- package/dist/adapters/repository-command-diagnostic.js +46 -0
- package/dist/adapters/repository-command-evidence.d.ts +13 -0
- package/dist/adapters/repository-command-evidence.js +102 -0
- package/dist/adapters/repository-command-runner.d.ts +238 -0
- package/dist/adapters/repository-command-runner.js +1483 -0
- package/dist/adapters/repository-import-context.d.ts +47 -0
- package/dist/adapters/repository-import-context.js +469 -0
- package/dist/adapters/repository-node-test-reporter.d.ts +3 -0
- package/dist/adapters/repository-node-test-reporter.js +27 -0
- package/dist/adapters/repository-review-evidence.d.ts +39 -0
- package/dist/adapters/repository-review-evidence.js +632 -0
- package/dist/adapters/repository-seed-selection.d.ts +7 -0
- package/dist/adapters/repository-seed-selection.js +79 -0
- package/dist/adapters/repository-solution-edits.d.ts +49 -0
- package/dist/adapters/repository-solution-edits.js +136 -0
- package/dist/adapters/repository-vitest-phase-adapter.d.ts +8 -0
- package/dist/adapters/repository-vitest-phase-adapter.js +310 -0
- package/dist/adapters/repository-vitest-reporter.d.ts +24 -0
- package/dist/adapters/repository-vitest-reporter.js +314 -0
- package/dist/adapters/strict-authoring-schema.d.ts +5 -0
- package/dist/adapters/strict-authoring-schema.js +158 -0
- package/dist/adapters/test-discovery.d.ts +30 -0
- package/dist/adapters/test-discovery.js +124 -0
- package/dist/adapters/typescript-repository-index.d.ts +51 -0
- package/dist/adapters/typescript-repository-index.js +226 -0
- package/dist/agentic-capabilities-protocol.d.ts +1373 -0
- package/dist/agentic-capabilities-protocol.js +786 -0
- package/dist/case-checkpoint-store.d.ts +29 -0
- package/dist/case-checkpoint-store.js +133 -0
- package/dist/case-pipeline-protocol-v2.d.ts +184 -0
- package/dist/case-pipeline-protocol-v2.js +193 -0
- package/dist/case-pipeline-protocol.d.ts +2626 -0
- package/dist/case-pipeline-protocol.js +371 -0
- package/dist/effect-api.d.ts +74 -10
- package/dist/effect-api.js +56 -6
- package/dist/errors.d.ts +31 -0
- package/dist/errors.js +10 -0
- package/dist/eval-capability-execution-envelope.d.ts +64 -0
- package/dist/eval-capability-execution-envelope.js +98 -0
- package/dist/eval-capability-policy.d.ts +90 -0
- package/dist/eval-capability-policy.js +107 -0
- package/dist/eval-event-log.d.ts +140 -0
- package/dist/eval-event-log.js +220 -0
- package/dist/evaluation-authoring-policy.d.ts +18 -0
- package/dist/evaluation-authoring-policy.js +19 -0
- package/dist/evaluation-authoring-validation.d.ts +22 -0
- package/dist/evaluation-authoring-validation.js +72 -0
- package/dist/evaluation-evidence.d.ts +20 -0
- package/dist/evaluation-evidence.js +319 -0
- package/dist/evaluation-grader-calibration-protocol.d.ts +108 -0
- package/dist/evaluation-grader-calibration-protocol.js +80 -0
- package/dist/evaluation-grader-calibration.d.ts +18 -0
- package/dist/evaluation-grader-calibration.js +334 -0
- package/dist/evaluation-grading-policy.d.ts +24 -0
- package/dist/evaluation-grading-policy.js +54 -0
- package/dist/evaluation-proposal-policy.d.ts +4 -0
- package/dist/evaluation-proposal-policy.js +91 -0
- package/dist/evaluation-source-retrieval.d.ts +68 -0
- package/dist/evaluation-source-retrieval.js +513 -0
- package/dist/evaluation-structure-policy.d.ts +29 -0
- package/dist/evaluation-structure-policy.js +138 -0
- package/dist/index.d.ts +124 -17
- package/dist/index.js +69 -11
- package/dist/inspection.js +2 -3
- package/dist/project-artifacts.d.ts +7 -2
- package/dist/project-artifacts.js +49 -136
- package/dist/project-authoring.d.ts +66 -5
- package/dist/project-authoring.js +783 -109
- package/dist/project-contracts.d.ts +419 -84
- package/dist/project-contracts.js +160 -52
- package/dist/project-store.js +2 -1
- package/dist/project-workflow.d.ts +5 -4
- package/dist/project-workflow.js +154 -35
- package/dist/repository-adversary-protocol.d.ts +64 -0
- package/dist/repository-adversary-protocol.js +105 -0
- package/dist/repository-behavior-protocol.d.ts +188 -0
- package/dist/repository-behavior-protocol.js +202 -0
- package/dist/repository-benchmark-protocol.d.ts +487 -0
- package/dist/repository-benchmark-protocol.js +96 -0
- package/dist/repository-execution-protocol.d.ts +150 -0
- package/dist/repository-execution-protocol.js +38 -0
- package/dist/repository-fixture-instructions.d.ts +3 -0
- package/dist/repository-fixture-instructions.js +91 -0
- package/dist/repository-fixture-protocol.d.ts +79 -0
- package/dist/repository-fixture-protocol.js +79 -0
- package/dist/repository-foundry-plan-protocol.d.ts +118 -0
- package/dist/repository-foundry-plan-protocol.js +296 -0
- package/dist/repository-foundry-progress-protocol.d.ts +52 -0
- package/dist/repository-foundry-progress-protocol.js +52 -0
- package/dist/repository-improvement-protocol.d.ts +100 -0
- package/dist/repository-improvement-protocol.js +106 -0
- package/dist/repository-language-model-protocol.d.ts +43 -0
- package/dist/repository-language-model-protocol.js +146 -0
- package/dist/repository-oracle-coverage-protocol.d.ts +18 -0
- package/dist/repository-oracle-coverage-protocol.js +39 -0
- package/dist/repository-oracle-execution-binding.d.ts +27 -0
- package/dist/repository-oracle-execution-binding.js +59 -0
- package/dist/repository-oracle-protocol.d.ts +230 -0
- package/dist/repository-oracle-protocol.js +156 -0
- package/dist/repository-oracle-scope-policy.d.ts +22 -0
- package/dist/repository-oracle-scope-policy.js +92 -0
- package/dist/repository-quality-policy.d.ts +15 -0
- package/dist/repository-quality-policy.js +357 -0
- package/dist/repository-routing-benchmark-protocol.d.ts +176 -0
- package/dist/repository-routing-benchmark-protocol.js +103 -0
- package/dist/repository-routing-model-protocol.d.ts +36 -0
- package/dist/repository-routing-model-protocol.js +89 -0
- package/dist/repository-routing-plan-protocol.d.ts +112 -0
- package/dist/repository-routing-plan-protocol.js +58 -0
- package/dist/repository-routing-quality-policy.d.ts +9 -0
- package/dist/repository-routing-quality-policy.js +191 -0
- package/dist/repository-seed-qualification-progress-protocol.d.ts +205 -0
- package/dist/repository-seed-qualification-progress-protocol.js +28 -0
- package/dist/repository-semantic-calibration-protocol.d.ts +768 -0
- package/dist/repository-semantic-calibration-protocol.js +276 -0
- package/dist/repository-semantic-calibration.d.ts +163 -0
- package/dist/repository-semantic-calibration.js +581 -0
- package/dist/repository-specification-contract-facts-protocol.d.ts +224 -0
- package/dist/repository-specification-contract-facts-protocol.js +276 -0
- package/dist/repository-specification-critique-protocol.d.ts +189 -0
- package/dist/repository-specification-critique-protocol.js +103 -0
- package/dist/repository-task-family-protocol.d.ts +24 -0
- package/dist/repository-task-family-protocol.js +37 -0
- package/dist/repository-task-seed-protocol.d.ts +384 -0
- package/dist/repository-task-seed-protocol.js +236 -0
- package/dist/repository-trajectory-protocol.d.ts +20 -0
- package/dist/repository-trajectory-protocol.js +42 -0
- package/dist/service.js +1 -1
- package/dist/services/adversary/service.d.ts +64 -0
- package/dist/services/adversary/service.js +330 -0
- package/dist/services/benchmark-compiler/service.d.ts +450 -0
- package/dist/services/benchmark-compiler/service.js +9 -0
- package/dist/services/budgeted-model/service.d.ts +118 -0
- package/dist/services/budgeted-model/service.js +460 -0
- package/dist/services/case-authoring/service.d.ts +163 -0
- package/dist/services/case-authoring/service.js +1456 -0
- package/dist/services/case-finalization/service.d.ts +283 -0
- package/dist/services/case-finalization/service.js +370 -0
- package/dist/services/case-generation/service.d.ts +619 -0
- package/dist/services/case-generation/service.js +2628 -0
- package/dist/services/case-pipeline/service.d.ts +31 -0
- package/dist/services/case-pipeline/service.js +485 -0
- package/dist/services/case-pipeline-v2/service.d.ts +70 -0
- package/dist/services/case-pipeline-v2/service.js +477 -0
- package/dist/services/command-observability/service.d.ts +13 -0
- package/dist/services/command-observability/service.js +3 -0
- package/dist/services/dimension-labeling/service.d.ts +77 -0
- package/dist/services/dimension-labeling/service.js +188 -0
- package/dist/services/eval-candidate/service.d.ts +208 -0
- package/dist/services/eval-candidate/service.js +64 -0
- package/dist/services/eval-capabilities/service.d.ts +183 -0
- package/dist/services/eval-capabilities/service.js +1433 -0
- package/dist/services/eval-environment/service.d.ts +173 -0
- package/dist/services/eval-environment/service.js +127 -0
- package/dist/services/evidence-reconstruction/service.d.ts +36 -0
- package/dist/services/evidence-reconstruction/service.js +145 -0
- package/dist/services/fixture-builder/service.d.ts +62 -0
- package/dist/services/fixture-builder/service.js +36 -0
- package/dist/services/fixture-validation/service.d.ts +75 -0
- package/dist/services/fixture-validation/service.js +295 -0
- package/dist/services/foundry/service.d.ts +831 -0
- package/dist/services/foundry/service.js +442 -0
- package/dist/services/foundry-progress/service.d.ts +62 -0
- package/dist/services/foundry-progress/service.js +149 -0
- package/dist/services/foundry-v2/service.d.ts +54 -0
- package/dist/services/foundry-v2/service.js +28 -0
- package/dist/services/grounded-authoring/service.d.ts +126 -0
- package/dist/services/grounded-authoring/service.js +822 -0
- package/dist/services/historical-case/service.d.ts +722 -0
- package/dist/services/historical-case/service.js +177 -0
- package/dist/services/improvement-loop/service.d.ts +59 -0
- package/dist/services/improvement-loop/service.js +176 -0
- package/dist/services/language-model/service.d.ts +52 -0
- package/dist/services/language-model/service.js +194 -0
- package/dist/services/oracle-builder/service.d.ts +146 -0
- package/dist/services/oracle-builder/service.js +513 -0
- package/dist/services/oracle-coverage/service.d.ts +28 -0
- package/dist/services/oracle-coverage/service.js +50 -0
- package/dist/services/oracle-coverage-witness/service.d.ts +130 -0
- package/dist/services/oracle-coverage-witness/service.js +538 -0
- package/dist/services/pipeline-challenge/service.d.ts +551 -0
- package/dist/services/pipeline-challenge/service.js +427 -0
- package/dist/services/pipeline-controls/service.d.ts +130 -0
- package/dist/services/pipeline-controls/service.js +483 -0
- package/dist/services/pipeline-oracle/service.d.ts +8 -0
- package/dist/services/pipeline-oracle/service.js +256 -0
- package/dist/services/pipeline-seed/service.d.ts +298 -0
- package/dist/services/pipeline-seed/service.js +428 -0
- package/dist/services/pipeline-spec/service.d.ts +103 -0
- package/dist/services/pipeline-spec/service.js +619 -0
- package/dist/services/pipeline-tournament/service.d.ts +258 -0
- package/dist/services/pipeline-tournament/service.js +476 -0
- package/dist/services/quality-gate/service.d.ts +233 -0
- package/dist/services/quality-gate/service.js +136 -0
- package/dist/services/repository-bundle/service.d.ts +33 -0
- package/dist/services/repository-bundle/service.js +114 -0
- package/dist/services/repository-model/service.d.ts +105 -0
- package/dist/services/repository-model/service.js +250 -0
- package/dist/services/repository-public-artifact/service.d.ts +133 -0
- package/dist/services/repository-public-artifact/service.js +330 -0
- package/dist/services/routing-benchmark/service.d.ts +362 -0
- package/dist/services/routing-benchmark/service.js +96 -0
- package/dist/services/specification-critic/service.d.ts +92 -0
- package/dist/services/specification-critic/service.js +172 -0
- package/dist/services/task-family/service.d.ts +40 -0
- package/dist/services/task-family/service.js +55 -0
- package/dist/services/task-seed/service.d.ts +906 -0
- package/dist/services/task-seed/service.js +1406 -0
- package/dist/services/task-specification/service.d.ts +27 -0
- package/dist/services/task-specification/service.js +40 -0
- package/dist/services/trajectory-policy/service.d.ts +110 -0
- package/dist/services/trajectory-policy/service.js +216 -0
- package/dist/test/agentic-capabilities-protocol.test.d.ts +1 -0
- package/dist/test/agentic-capabilities-protocol.test.js +570 -0
- package/dist/test/agentic-capabilities.test.d.ts +1 -0
- package/dist/test/agentic-capabilities.test.js +1461 -0
- package/dist/test/agentic-environment.test.d.ts +1 -0
- package/dist/test/agentic-environment.test.js +213 -0
- package/dist/test/case-pipeline-foundation.test.d.ts +1 -0
- package/dist/test/case-pipeline-foundation.test.js +535 -0
- package/dist/test/case-pipeline-protocol-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-protocol-v2.test.js +124 -0
- package/dist/test/case-pipeline-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-v2.test.js +286 -0
- package/dist/test/case-pipeline.test.d.ts +1 -0
- package/dist/test/case-pipeline.test.js +851 -0
- package/dist/test/eval-capability-policy.test.d.ts +1 -0
- package/dist/test/eval-capability-policy.test.js +50 -0
- package/dist/test/eval-event-log.test.d.ts +1 -0
- package/dist/test/eval-event-log.test.js +125 -0
- package/dist/test/evaluation-evidence-freshness.test.d.ts +1 -0
- package/dist/test/evaluation-evidence-freshness.test.js +44 -0
- package/dist/test/evaluation-evidence.test.d.ts +1 -0
- package/dist/test/evaluation-evidence.test.js +230 -0
- package/dist/test/evaluation-grader-calibration.test.d.ts +1 -0
- package/dist/test/evaluation-grader-calibration.test.js +373 -0
- package/dist/test/evaluation-proposal-digest.test.d.ts +1 -0
- package/dist/test/evaluation-proposal-digest.test.js +187 -0
- package/dist/test/evaluation-source-retrieval.test.d.ts +1 -0
- package/dist/test/evaluation-source-retrieval.test.js +237 -0
- package/dist/test/evaluation-structure-policy.test.d.ts +1 -0
- package/dist/test/evaluation-structure-policy.test.js +196 -0
- package/dist/test/fixtures/repository-resource-panel.d.ts +39 -0
- package/dist/test/fixtures/repository-resource-panel.js +111 -0
- package/dist/test/fixtures/vitest-boundary-panel.d.ts +84 -0
- package/dist/test/fixtures/vitest-boundary-panel.js +120 -0
- package/dist/test/fixtures/vitest-phase-panel.d.ts +135 -0
- package/dist/test/fixtures/vitest-phase-panel.js +213 -0
- package/dist/test/fixtures/vitest-reporter-results.d.ts +76 -0
- package/dist/test/fixtures/vitest-reporter-results.js +94 -0
- package/dist/test/grounded-authoring.test.d.ts +1 -0
- package/dist/test/grounded-authoring.test.js +565 -0
- package/dist/test/integrated-repository-history.test.d.ts +1 -0
- package/dist/test/integrated-repository-history.test.js +227 -0
- package/dist/test/project-authoring.test.js +593 -43
- package/dist/test/project-workflow.test.js +419 -40
- package/dist/test/repository-authoring-artifacts.test.d.ts +1 -0
- package/dist/test/repository-authoring-artifacts.test.js +185 -0
- package/dist/test/repository-bundle.test.d.ts +1 -0
- package/dist/test/repository-bundle.test.js +52 -0
- package/dist/test/repository-case-generation.test.d.ts +1 -0
- package/dist/test/repository-case-generation.test.js +3465 -0
- package/dist/test/repository-command-diagnostic.test.d.ts +1 -0
- package/dist/test/repository-command-diagnostic.test.js +55 -0
- package/dist/test/repository-command-signals.test.d.ts +1 -0
- package/dist/test/repository-command-signals.test.js +124 -0
- package/dist/test/repository-fixture-scope-coverage.test.d.ts +1 -0
- package/dist/test/repository-fixture-scope-coverage.test.js +127 -0
- package/dist/test/repository-fixture-validation.test.d.ts +1 -0
- package/dist/test/repository-fixture-validation.test.js +362 -0
- package/dist/test/repository-foundry-progress.test.d.ts +1 -0
- package/dist/test/repository-foundry-progress.test.js +110 -0
- package/dist/test/repository-foundry-quality.test.d.ts +1 -0
- package/dist/test/repository-foundry-quality.test.js +1138 -0
- package/dist/test/repository-import-context.test.d.ts +1 -0
- package/dist/test/repository-import-context.test.js +354 -0
- package/dist/test/repository-model-authoring.test.d.ts +1 -0
- package/dist/test/repository-model-authoring.test.js +544 -0
- package/dist/test/repository-model.test.d.ts +1 -0
- package/dist/test/repository-model.test.js +2195 -0
- package/dist/test/repository-node-test-reporter.test.d.ts +1 -0
- package/dist/test/repository-node-test-reporter.test.js +104 -0
- package/dist/test/repository-oracle-concurrency.test.d.ts +1 -0
- package/dist/test/repository-oracle-concurrency.test.js +542 -0
- package/dist/test/repository-oracle-coverage-witness.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage-witness.test.js +511 -0
- package/dist/test/repository-oracle-coverage.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage.test.js +168 -0
- package/dist/test/repository-oracle-evidence.test.d.ts +1 -0
- package/dist/test/repository-oracle-evidence.test.js +185 -0
- package/dist/test/repository-oracle-plan.test.d.ts +1 -0
- package/dist/test/repository-oracle-plan.test.js +176 -0
- package/dist/test/repository-overlay-isolation.test.d.ts +1 -0
- package/dist/test/repository-overlay-isolation.test.js +85 -0
- package/dist/test/repository-preparation-cache.test.d.ts +1 -0
- package/dist/test/repository-preparation-cache.test.js +414 -0
- package/dist/test/repository-public-artifact.test.d.ts +1 -0
- package/dist/test/repository-public-artifact.test.js +273 -0
- package/dist/test/repository-qualification-diagnostics.test.d.ts +1 -0
- package/dist/test/repository-qualification-diagnostics.test.js +524 -0
- package/dist/test/repository-reference-authoring.test.d.ts +1 -0
- package/dist/test/repository-reference-authoring.test.js +1633 -0
- package/dist/test/repository-review-evidence-v2.test.d.ts +1 -0
- package/dist/test/repository-review-evidence-v2.test.js +183 -0
- package/dist/test/repository-review-evidence.test.d.ts +1 -0
- package/dist/test/repository-review-evidence.test.js +124 -0
- package/dist/test/repository-seed-exclusions.test.d.ts +1 -0
- package/dist/test/repository-seed-exclusions.test.js +96 -0
- package/dist/test/repository-seed-selection.test.d.ts +1 -0
- package/dist/test/repository-seed-selection.test.js +504 -0
- package/dist/test/repository-semantic-calibration.test.d.ts +1 -0
- package/dist/test/repository-semantic-calibration.test.js +688 -0
- package/dist/test/repository-solution-edits.test.d.ts +1 -0
- package/dist/test/repository-solution-edits.test.js +377 -0
- package/dist/test/repository-specification-budget.test.d.ts +1 -0
- package/dist/test/repository-specification-budget.test.js +171 -0
- package/dist/test/repository-specification-contract-checkpoint.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-checkpoint.test.js +228 -0
- package/dist/test/repository-specification-contract-facts.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-facts.test.js +177 -0
- package/dist/test/repository-trajectory-authoring.test.d.ts +1 -0
- package/dist/test/repository-trajectory-authoring.test.js +176 -0
- package/dist/test/repository-valid-control-plan.test.d.ts +1 -0
- package/dist/test/repository-valid-control-plan.test.js +45 -0
- package/dist/test/repository-vitest-phase.test.d.ts +1 -0
- package/dist/test/repository-vitest-phase.test.js +848 -0
- package/dist/test/repository-vitest-reporter.test.d.ts +1 -0
- package/dist/test/repository-vitest-reporter.test.js +158 -0
- package/dist/test/repository-workspace-build.test.d.ts +1 -0
- package/dist/test/repository-workspace-build.test.js +160 -0
- package/dist/test/strict-authoring-schema.test.d.ts +1 -0
- package/dist/test/strict-authoring-schema.test.js +169 -0
- package/package.json +48 -6
|
@@ -0,0 +1,2628 @@
|
|
|
1
|
+
import { Context, Effect, Layer, Option, Schema } from "effect";
|
|
2
|
+
import { evalAuthoringRequestByteLimit, evalAuthoringResponsesRequestBody } from "../../adapters/authoring-responses-request.js";
|
|
3
|
+
import { captureGitTreeSnapshotV1, readGitTreeFileV1 } from "../../adapters/git-task-history.js";
|
|
4
|
+
import { readRepositoryLocalImportsV1 } from "../../adapters/repository-import-context.js";
|
|
5
|
+
import { encodeRepositoryReviewEvidenceV1, encodeRepositoryReviewEvidenceV2, REPOSITORY_REVIEW_EVIDENCE_INSTRUCTIONS, REPOSITORY_REVIEW_EVIDENCE_V2_INSTRUCTIONS } from "../../adapters/repository-review-evidence.js";
|
|
6
|
+
import { loadPinnedSolutionEditSourcesV1, materializeSolutionFileOverridesV1, RepositorySolutionFileOutputV1 } from "../../adapters/repository-solution-edits.js";
|
|
7
|
+
import { strictAuthoringSchema } from "../../adapters/strict-authoring-schema.js";
|
|
8
|
+
import { RepositoryFoundryError } from "../../errors.js";
|
|
9
|
+
import { REPOSITORY_FIXTURE_EVIDENCE_INSTRUCTIONS, repositoryFixtureAuthoringInstructionsV1 } from "../../repository-fixture-instructions.js";
|
|
10
|
+
import { REPOSITORY_FOUNDRY_REQUEST_INPUT_CEILING, REPOSITORY_FOUNDRY_VALID_SOLUTION_REPAIR_ATTEMPTS, repositoryFoundryRoleOutputCeilingV1 } from "../../repository-foundry-plan-protocol.js";
|
|
11
|
+
import { repositoryFoundryModelAssignmentV1 } from "../../repository-language-model-protocol.js";
|
|
12
|
+
import { evaluateRepositoryOracleAdequacyV1 } from "../../repository-quality-policy.js";
|
|
13
|
+
import { parseRepositorySpecificationContractFactsPacketV1, RepositorySpecificationContractFactReviewV1 } from "../../repository-specification-contract-facts-protocol.js";
|
|
14
|
+
import { authorRepositoryCaseSpecificationV1 } from "../case-authoring/service.js";
|
|
15
|
+
import { RepositoryFoundryEvidenceReconstruction } from "../evidence-reconstruction/service.js";
|
|
16
|
+
import { validateRepositoryFixturesV1 } from "../fixture-validation/service.js";
|
|
17
|
+
import { reportFoundryAuthoringArtifactV1, reportFoundryStageV1 } from "../foundry-progress/service.js";
|
|
18
|
+
import { compileReviewedHistoricalCaseV1 } from "../historical-case/service.js";
|
|
19
|
+
import { RepositoryFoundryLanguageModel } from "../language-model/service.js";
|
|
20
|
+
import { buildHistoricalHiddenOracleV1 } from "../oracle-builder/service.js";
|
|
21
|
+
import { reviewRepositoryOracleCoverageV1 } from "../oracle-coverage/service.js";
|
|
22
|
+
import { resolveRepositoryOracleCoverageWitnessesV1, validateRepositoryOracleCoverageWitnessRevisionsV1 } from "../oracle-coverage-witness/service.js";
|
|
23
|
+
import { deriveRepositoryTaskFamiliesV1 } from "../task-family/service.js";
|
|
24
|
+
import { authorReviewedRepositoryTrajectoryPolicyV1 } from "../trajectory-policy/service.js";
|
|
25
|
+
const FixtureProposalBase = {
|
|
26
|
+
id: Schema.String,
|
|
27
|
+
description: Schema.String,
|
|
28
|
+
expectedBehavior: Schema.String,
|
|
29
|
+
testPath: Schema.String
|
|
30
|
+
};
|
|
31
|
+
const FixtureProposal = Schema.Struct({
|
|
32
|
+
fixtures: Schema.Array(Schema.Union([
|
|
33
|
+
Schema.Struct({
|
|
34
|
+
...FixtureProposalBase,
|
|
35
|
+
kind: Schema.Literal("historical-regression"),
|
|
36
|
+
expectationMode: Schema.Literal("changes")
|
|
37
|
+
}),
|
|
38
|
+
Schema.Struct({
|
|
39
|
+
...FixtureProposalBase,
|
|
40
|
+
kind: Schema.Literal("boundary"),
|
|
41
|
+
expectationMode: Schema.Literals(["changes", "preserved"])
|
|
42
|
+
}),
|
|
43
|
+
Schema.Struct({
|
|
44
|
+
...FixtureProposalBase,
|
|
45
|
+
kind: Schema.Literal("counterfactual"),
|
|
46
|
+
expectationMode: Schema.Literal("changes")
|
|
47
|
+
}),
|
|
48
|
+
Schema.Struct({
|
|
49
|
+
...FixtureProposalBase,
|
|
50
|
+
kind: Schema.Literal("metamorphic"),
|
|
51
|
+
expectationMode: Schema.Literal("preserved")
|
|
52
|
+
})
|
|
53
|
+
])),
|
|
54
|
+
overlays: Schema.Array(Schema.Struct({
|
|
55
|
+
path: Schema.String,
|
|
56
|
+
content: Schema.String,
|
|
57
|
+
fixtureIds: Schema.Array(Schema.String)
|
|
58
|
+
})),
|
|
59
|
+
knownLimitations: Schema.Array(Schema.String),
|
|
60
|
+
scopeCoverage: Schema.optionalKey(Schema.Array(Schema.Struct({
|
|
61
|
+
scopeId: Schema.String,
|
|
62
|
+
outcome: Schema.Literals(["covered", "unsupported", "contradictory", "uncertain"]),
|
|
63
|
+
fixtureIds: Schema.Array(Schema.String),
|
|
64
|
+
detail: Schema.String
|
|
65
|
+
})))
|
|
66
|
+
});
|
|
67
|
+
const SolutionIdentity = {
|
|
68
|
+
id: Schema.String,
|
|
69
|
+
family: Schema.String,
|
|
70
|
+
kind: Schema.Literals([
|
|
71
|
+
"independent-valid",
|
|
72
|
+
"simplified-valid",
|
|
73
|
+
"behavior-preserving-valid",
|
|
74
|
+
"mutation",
|
|
75
|
+
"model-near-miss",
|
|
76
|
+
"reward-hacking"
|
|
77
|
+
])
|
|
78
|
+
};
|
|
79
|
+
const SolutionProposal = Schema.Struct({
|
|
80
|
+
solutions: Schema.Array(Schema.Struct({
|
|
81
|
+
...SolutionIdentity,
|
|
82
|
+
fileOverrides: Schema.Array(Schema.Struct({ path: Schema.String, content: Schema.String }))
|
|
83
|
+
}))
|
|
84
|
+
});
|
|
85
|
+
const SolutionOutputProposal = Schema.Struct({
|
|
86
|
+
solutions: Schema.Array(Schema.Struct({
|
|
87
|
+
...SolutionIdentity,
|
|
88
|
+
fileOverrides: Schema.Array(RepositorySolutionFileOutputV1)
|
|
89
|
+
}))
|
|
90
|
+
});
|
|
91
|
+
const QualityReview = Schema.Struct({
|
|
92
|
+
verdict: Schema.Literals(["approve", "reject"]),
|
|
93
|
+
detail: Schema.String,
|
|
94
|
+
findings: Schema.Array(Schema.Struct({
|
|
95
|
+
axis: Schema.Literals([
|
|
96
|
+
"specification",
|
|
97
|
+
"oracle",
|
|
98
|
+
"valid-solution-independence",
|
|
99
|
+
"development-adversaries",
|
|
100
|
+
"held-out-adversaries",
|
|
101
|
+
"trajectory"
|
|
102
|
+
]),
|
|
103
|
+
outcome: Schema.Literals(["pass", "fail"]),
|
|
104
|
+
detail: Schema.String,
|
|
105
|
+
evidenceIds: Schema.Array(Schema.String)
|
|
106
|
+
})),
|
|
107
|
+
coverageGaps: Schema.Array(Schema.String),
|
|
108
|
+
correlatedAssumptions: Schema.Array(Schema.String)
|
|
109
|
+
});
|
|
110
|
+
const SolutionReviewProposal = Schema.Struct({
|
|
111
|
+
reviews: Schema.Array(Schema.Struct({
|
|
112
|
+
solutionId: Schema.String,
|
|
113
|
+
classification: Schema.Literals(["valid", "wrong", "uncertain"]),
|
|
114
|
+
detail: Schema.String,
|
|
115
|
+
violatedBehaviorIds: Schema.Array(Schema.String)
|
|
116
|
+
}))
|
|
117
|
+
});
|
|
118
|
+
const ValidControlIndependenceReview = Schema.Struct({
|
|
119
|
+
outcome: Schema.Literals(["independent", "correlated", "uncertain"]),
|
|
120
|
+
detail: Schema.String,
|
|
121
|
+
implementations: Schema.Array(Schema.Struct({
|
|
122
|
+
solutionId: Schema.String,
|
|
123
|
+
mechanism: Schema.String,
|
|
124
|
+
sourcePaths: Schema.Array(Schema.String)
|
|
125
|
+
})),
|
|
126
|
+
correlatedFeatures: Schema.Array(Schema.Literals([
|
|
127
|
+
"renaming-or-formatting",
|
|
128
|
+
"helper-extraction",
|
|
129
|
+
"equivalent-control-flow",
|
|
130
|
+
"same-behavioral-mechanism",
|
|
131
|
+
"insufficient-evidence"
|
|
132
|
+
]))
|
|
133
|
+
});
|
|
134
|
+
const SolutionReviewWithIndependenceProposal = Schema.Struct({
|
|
135
|
+
...SolutionReviewProposal.fields,
|
|
136
|
+
implementationIndependence: ValidControlIndependenceReview
|
|
137
|
+
});
|
|
138
|
+
const HillClimbProposal = Schema.Struct({
|
|
139
|
+
rationale: Schema.String,
|
|
140
|
+
targetedFalseAcceptIds: Schema.Array(Schema.String),
|
|
141
|
+
targetedFalseRejectIds: Schema.Array(Schema.String),
|
|
142
|
+
targetedInconclusiveIds: Schema.optionalKey(Schema.Array(Schema.String)),
|
|
143
|
+
fixtureProposal: FixtureProposal
|
|
144
|
+
});
|
|
145
|
+
/** Bind accepted public facts to the actual final writer and two critic calls. */
|
|
146
|
+
export function assertRepositorySpecificationContractFactsEvidenceV1(authored) {
|
|
147
|
+
const facts = authored.contractFacts;
|
|
148
|
+
const grounding = authored.referenceGrounding;
|
|
149
|
+
const finalRevision = authored.specificationRevisions?.at(-1);
|
|
150
|
+
if (facts?.version !== 1 ||
|
|
151
|
+
facts.independentlyReviewed !== true ||
|
|
152
|
+
grounding === undefined ||
|
|
153
|
+
authored.reviews.length !== 2 ||
|
|
154
|
+
facts.reviewerOperationIds.length !== 2 ||
|
|
155
|
+
new Set(facts.reviewerOperationIds).size !== 2 ||
|
|
156
|
+
finalRevision === undefined ||
|
|
157
|
+
finalRevision.contractFactsDigest !== facts.packet.digest ||
|
|
158
|
+
JSON.stringify(finalRevision.visible) !== JSON.stringify(authored.visible) ||
|
|
159
|
+
JSON.stringify(finalRevision.reviews) !== JSON.stringify(authored.reviews))
|
|
160
|
+
throw failure("contract facts require the accepted packet and exact final specification reviews");
|
|
161
|
+
const packet = parseRepositorySpecificationContractFactsPacketV1(facts.packet, grounding.scope);
|
|
162
|
+
for (const [index, review] of authored.reviews.entries()) {
|
|
163
|
+
if (review.reviewerId !== facts.reviewerOperationIds[index] ||
|
|
164
|
+
review.independentlyProduced !== true ||
|
|
165
|
+
review.verdict !== "sufficient" ||
|
|
166
|
+
review.findings.some((finding) => finding.severity === "blocker") ||
|
|
167
|
+
review.reviewerKind !== (index === 0 ? "independent-solver" : "model-critic") ||
|
|
168
|
+
authored.modelCalls.filter((call) => call.operationId === review.reviewerId &&
|
|
169
|
+
call.role === (index === 0 ? "specification-critic-a" : "specification-critic-b")).length !== 1)
|
|
170
|
+
throw failure("contract facts require two distinct successful final critic calls");
|
|
171
|
+
const checked = Schema.decodeUnknownSync(RepositorySpecificationContractFactReviewV1)(review.contractFactReview);
|
|
172
|
+
if (checked.packetDigest !== packet.digest ||
|
|
173
|
+
checked.clarificationRequests.length !== 0 ||
|
|
174
|
+
checked.scopeChecks.length !== grounding.scope.clauses.length ||
|
|
175
|
+
new Set(checked.scopeChecks.map((check) => check.scopeId)).size !== checked.scopeChecks.length ||
|
|
176
|
+
grounding.scope.clauses.some((scope) => {
|
|
177
|
+
const check = checked.scopeChecks.find((entry) => entry.scopeId === scope.id);
|
|
178
|
+
const expectedFacts = packet.facts.filter((fact) => fact.scopeId === scope.id);
|
|
179
|
+
return check === undefined ||
|
|
180
|
+
check.outcome !== "complete" ||
|
|
181
|
+
check.factIds.length !== expectedFacts.length ||
|
|
182
|
+
new Set(check.factIds).size !== check.factIds.length ||
|
|
183
|
+
expectedFacts.some((fact) => !check.factIds.includes(fact.id));
|
|
184
|
+
}))
|
|
185
|
+
throw failure("final contract-fact critics must approve the exact complete packet and scope");
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
const ORACLE_INSTRUCTIONS = `You are the hidden-oracle designer in a repository coding benchmark foundry.
|
|
189
|
+
Use the authentic behavior delta, pre-change and reference evidence, and historical tests to create a stronger
|
|
190
|
+
hidden executable test suite. Cover every critical behavior, at least one boundary, and at least one
|
|
191
|
+
counterfactual or metamorphic invariant. Produce exactly one complete replacement-file overlay for each fixture,
|
|
192
|
+
and bind that overlay to only that fixture id. Multiple fixture overlays may target the same reviewed test path
|
|
193
|
+
because RouteKit executes each overlay in an independent clean checkout. Each isolated overlay must run with the
|
|
194
|
+
repository's declared grade command and must cover a coherent, accurately named user-observable behavior area.
|
|
195
|
+
Several related scope clauses may share one fixture when its description, expected behavior, and expectation
|
|
196
|
+
mode faithfully describe all of them. Retain the imports and setup needed for that area.
|
|
197
|
+
A counterfactual fixture must use expectationMode "changes". A
|
|
198
|
+
metamorphic fixture must vary an input or context while preserving a named user-observable invariant and must
|
|
199
|
+
use expectationMode "preserved". A historical-regression fixture must use expectationMode "changes". Do not
|
|
200
|
+
prescribe the historical implementation and do not weaken existing historical coverage. Return only the
|
|
201
|
+
requested structured object. Treat repositoryEnvironment as the pinned dependency and toolchain context.
|
|
202
|
+
Use APIs supported by those versions and demonstrated in the supplied repository sources; do not assume an
|
|
203
|
+
API from a different major version exists. Each generated overlay must compile and execute unchanged against
|
|
204
|
+
the supplied correct reference before it can enter the solution tournament.
|
|
205
|
+
When behavioralScope is supplied, return exactly one scopeCoverage record for every supplied scope id.
|
|
206
|
+
Mark covered only when the named existing fixtureIds actually assert the complete clause, including eligibility
|
|
207
|
+
and preservation boundaries. Explain that executable coverage in detail. Use unsupported, contradictory, or
|
|
208
|
+
uncertain when faithful coverage cannot be provided; those outcomes stop generation. Never omit a mandatory
|
|
209
|
+
behavior merely to make the reference pass. knownLimitations may describe only remaining nonmandatory limits;
|
|
210
|
+
it cannot excuse any omitted, unsupported, contradictory, or uncertain required scope clause.
|
|
211
|
+
|
|
212
|
+
${REPOSITORY_FIXTURE_EVIDENCE_INSTRUCTIONS}`;
|
|
213
|
+
const EXACT_EDIT_INSTRUCTIONS = `For existing files, return fileOverrides entries as {path, edits:[{search, replace}]}.
|
|
214
|
+
Each search must be nonempty verbatim text that occurs exactly once in the supplied pre-change excerpt and full
|
|
215
|
+
original file. Include enough unchanged context to make it unique. Every edit in one solution is relative to the
|
|
216
|
+
same ORIGINAL pre-change source, including repair turns; edits must not overlap or depend on another edit's output.
|
|
217
|
+
An empty replace deletes the matched text. Insertions must replace a nonempty unique anchor while retaining it.
|
|
218
|
+
The host applies edits to the full pinned pre-change file and preserves every untouched byte, including unseen
|
|
219
|
+
suffixes beyond truncated excerpts. Do not reproduce unchanged files, truncation markers, or entire large files.
|
|
220
|
+
Legacy {path, content} entries remain supported for complete small replacements; prefer compact exact edits.
|
|
221
|
+
Use only the existing supplied path allowlists; never invent a path, base revision, hidden test, or reference source.`;
|
|
222
|
+
const VALID_SOLUTION_A_INSTRUCTIONS = `You are independent solver A in a repository coding benchmark foundry.
|
|
223
|
+
Solve the visible task using only the pre-change source excerpts. Produce exactly one behaviorally valid
|
|
224
|
+
implementation as exact edits to the supplied pre-change sources. Prefer the simplest direct design justified by the visible
|
|
225
|
+
contract and classify it as independent-valid or simplified-valid. Do not modify tests, inspect hidden tests,
|
|
226
|
+
copy a historical patch, speculate about another solver, or include prose outside the structured object.
|
|
227
|
+
The family field names your implementation mechanism, not the task, requested dimension, capability, or solution
|
|
228
|
+
identifier. Describe the concrete approach accurately; a family label is not evidence of source diversity.`;
|
|
229
|
+
const VALID_SOLUTION_B_INSTRUCTIONS = `You are independent solver B in a repository coding benchmark foundry.
|
|
230
|
+
Solve the visible task from scratch using only the pre-change source excerpts. Produce exactly one behaviorally
|
|
231
|
+
valid implementation as exact edits to the supplied pre-change sources. Preserve the repository's public architecture and classify
|
|
232
|
+
the result as behavior-preserving-valid. Seek a design materially different from an obvious minimal patch, but
|
|
233
|
+
do not modify tests, inspect hidden tests, copy a historical patch, speculate about another solver, or include
|
|
234
|
+
prose outside the structured object. You are not shown solver A's proposal and must not assume its contents.
|
|
235
|
+
The family field names your implementation mechanism, not the task, requested dimension, capability, or solution
|
|
236
|
+
identifier. Describe the concrete approach accurately; a family label is not evidence of source diversity.`;
|
|
237
|
+
const VALID_SOLUTION_B_MECHANISM_INSTRUCTIONS = `Your implementation is also used to detect tests that
|
|
238
|
+
accidentally enforce one particular correct design. Choose a substantively different mechanism for the changed
|
|
239
|
+
behavior from an obvious direct fix. For example, consider whether the visible contract permits a different
|
|
240
|
+
data representation, computation, state transition, or parsing strategy. A helper extraction, reordered or
|
|
241
|
+
negated guards, renamed variables, or another spelling of the same Boolean condition is not an independent
|
|
242
|
+
mechanism. Keep the change scoped to the task; shared unchanged integration code is expected and does not
|
|
243
|
+
require unrelated rewrites. Explain the actual mechanism accurately in the family field. You still receive
|
|
244
|
+
only the visible task, pre-change context, and your own proposal on repair turns.`;
|
|
245
|
+
const VALID_SOLUTION_B_BEHAVIORAL_INSTRUCTIONS = `You are independent solver B in a repository coding benchmark foundry.
|
|
246
|
+
Solve the visible task from scratch using only the pre-change source excerpts. Produce exactly one behaviorally
|
|
247
|
+
valid implementation as exact edits to the supplied pre-change sources and classify it as behavior-preserving-valid.
|
|
248
|
+
Your implementation will test whether the oracle accepts correct alternatives. Where the contract permits it,
|
|
249
|
+
use a natural equivalent formulation, representation, or implementation approach. A narrow fix may have one
|
|
250
|
+
natural mechanism; do not introduce unrelated refactoring or invent a new algorithm to satisfy a diversity label.
|
|
251
|
+
Preserve all visible behavior and public integration constraints. Do not modify tests, inspect hidden tests,
|
|
252
|
+
copy a historical patch, or speculate about another solver. You are not shown solver A's proposal.
|
|
253
|
+
Describe the actual implementation mechanism accurately in the family field. Return only the structured object.`;
|
|
254
|
+
const VALID_SOLUTION_REPAIR_INSTRUCTIONS = `This is a bounded independent-solver repair turn. You receive only
|
|
255
|
+
your own previous proposal and criticism grounded in the visible task and pre-change repository context. Produce
|
|
256
|
+
one complete solution proposal with exact edits against the ORIGINAL pre-change sources that resolves every supplied finding.
|
|
257
|
+
For metadata-only contract findings, accurately correct your own metadata while preserving a correct implementation.
|
|
258
|
+
For semantic or source-diversity findings, re-read the entire visible behavioral contract and change the relevant
|
|
259
|
+
implementation; do not merely rename, reformat, comment, or restate unchanged behavior. Preserve every unrelated behavior and repository integration
|
|
260
|
+
constraint. You are not shown the other solver's proposal, hidden tests, reference implementation, oracle,
|
|
261
|
+
adversaries, or tournament outcomes. Do not infer any of them. Return only the requested structured object.`;
|
|
262
|
+
const ADVERSARY_INSTRUCTIONS = `You are the development adversarial-solution generator in a repository coding benchmark foundry.
|
|
263
|
+
You receive the visible task, pre-change source excerpts, and developmentFixtureProposal: the actual draft tests
|
|
264
|
+
that have passed reference preflight. Inspect their assertions to find contract violations that this draft
|
|
265
|
+
does not distinguish. Your work is a development challenge used to repair the oracle before it is frozen.
|
|
266
|
+
The separate held-out adversary will not receive the draft, your proposals, or development outcomes.
|
|
267
|
+
Produce at least five plausible implementations expressed as compact exact edits that compile and execute but
|
|
268
|
+
are behaviorally wrong in different ways, plus additional
|
|
269
|
+
reward-hacking candidates. A policy rejection, build failure, runtime crash, or empty test run does not count
|
|
270
|
+
toward the five behavioral adversaries. For each behavioral control, name the violated visible requirement and
|
|
271
|
+
the input or state partition on which it fails. Prefer realistic partial fixes that are likely to pass the
|
|
272
|
+
current assertions over defects already killed by an obvious existing assertion. Challenge coverage across
|
|
273
|
+
allowed values, combinations, eligibility boundaries, preserved behavior, and cross-file integration; a few
|
|
274
|
+
hard-coded examples do not cover the full contract. Do not invent requirements beyond the visible task.
|
|
275
|
+
Your implementation must preserve unrelated behavior and fail for the stated semantic reason, without
|
|
276
|
+
depending on private fixture names, exact generated values, test detection, or source fingerprints.
|
|
277
|
+
Keep any deliberately fixture-dependent or harness-tampering control in the separate reward-hacking population.
|
|
278
|
+
Cover boundary mistakes, partial
|
|
279
|
+
fixes, an integration or cross-file omission, fixture hard-coding, test-environment detection, shared-harness
|
|
280
|
+
manipulation, test weakening, and an alternative execution bypass across the population. Include at least one
|
|
281
|
+
mutation, one model-near-miss, and one reward-hacking candidate. Except for explicitly classified reward-hacking
|
|
282
|
+
candidates, do not modify tests or controls. A reward-hacking candidate may combine changes to allowed solution
|
|
283
|
+
paths with changes to allowed reward-hacking paths, but it must modify at least one supplied reward-hacking
|
|
284
|
+
path. Use only the exact supplied path allowlists. You are not given the correct reference implementation or
|
|
285
|
+
either independent valid solution; do not infer their contents. Return only the
|
|
286
|
+
requested structured object.`;
|
|
287
|
+
const HELD_OUT_ADVERSARY_INSTRUCTIONS = `You are the held-out adversarial-solution generator in a repository
|
|
288
|
+
coding benchmark foundry. The hidden oracle has already been frozen. You are not shown that oracle, its fixtures,
|
|
289
|
+
the development adversaries, their outcomes, or any hill-climb evidence. Using only the visible task and
|
|
290
|
+
pre-change source excerpts, produce at least five new plausible implementations expressed as compact exact edits that compile and
|
|
291
|
+
execute but are behaviorally wrong, plus additional reward-hacking candidates. A policy rejection, build
|
|
292
|
+
failure, runtime crash, or empty test run does not count toward the five behavioral adversaries.
|
|
293
|
+
Stress fixture hard-coding, test-environment detection, shared-harness manipulation,
|
|
294
|
+
test weakening, alternative execution bypasses, partial semantic fixes, and cross-file omissions. Include at
|
|
295
|
+
least one mutation, one model-near-miss, and one reward-hacking candidate. Except for reward-hacking candidates,
|
|
296
|
+
do not modify tests or controls. A reward-hacking candidate must modify at least one exact supplied
|
|
297
|
+
reward-hacking path. Every family name must differ from the compiler-supplied forbidden development family names,
|
|
298
|
+
and any regeneration must resolve every supplied novelty failure without inferring hidden fixtures. Return only
|
|
299
|
+
the requested structured object.`;
|
|
300
|
+
const VALID_CONTROL_INDEPENDENCE_INSTRUCTIONS = `Separately assess implementationIndependence for exactly
|
|
301
|
+
the two solution ids in implementationComparison.solutionIds. Correctness classifications and independence
|
|
302
|
+
are different judgments: two correlated but correct solutions must both retain classification "valid".
|
|
303
|
+
For each implementation, describe the concrete mechanism that implements the changed visible behavior and
|
|
304
|
+
cite the supplied fileOverride paths you inspected. Compare the actual changes relative to preChangeSources,
|
|
305
|
+
not family labels, patch size, formatting, or shared unchanged integration code.
|
|
306
|
+
Return independent only for materially different mechanisms within the task's scope. Renaming, helper
|
|
307
|
+
extraction, equivalent Boolean guards, or restating the same control flow are correlated, even when their
|
|
308
|
+
text looks different. Do not demand unrelated rewrites merely to make solutions look different. If evidence
|
|
309
|
+
does not establish the distinction, return uncertain. Explain the comparison in detail. correlatedFeatures
|
|
310
|
+
must be empty for independent and contain the applicable reasons for correlated or uncertain. This assessment
|
|
311
|
+
does not authorize relabeling, rewriting either solution, or omitting a contract requirement.`;
|
|
312
|
+
const BEHAVIORAL_VALID_CONTROL_POLICY_INSTRUCTIONS = `Valid-control policy version 2 distinguishes independent
|
|
313
|
+
generation from algorithmic novelty. The two solver calls are isolated from each other, the reference patch,
|
|
314
|
+
hidden fixtures, and tournament feedback. Record mechanism similarity honestly in implementationIndependence;
|
|
315
|
+
do not label equivalent mechanisms independent merely to obtain approval. Mechanism similarity alone is not
|
|
316
|
+
a correctness defect or an admission veto. Independently check each full implementation against the visible
|
|
317
|
+
contract. In the source-only semantic review, classify correctness from the supplied code and contract; execution
|
|
318
|
+
is a separate later gate, so its absence at that stage is not itself uncertainty. A shared semantic mistake must
|
|
319
|
+
yield wrong or uncertain correctness labels with the affected behavior,
|
|
320
|
+
even if the source code or algorithms look different. Accepting a correct alternative requires complete repeated
|
|
321
|
+
executable evidence with no false rejection. Unresolved behavior, implementation-specific assertions, or missing
|
|
322
|
+
discrimination remain defects. Do not infer statistical independence or broad oracle coverage from separate calls,
|
|
323
|
+
different family labels, or code similarity.`;
|
|
324
|
+
const solutionCriticInstructions = (perspective, validControlReviewVersion) => `You are an independent ${perspective} solution critic in a repository coding benchmark foundry.
|
|
325
|
+
Classify every proposed implementation against only the visible task, explicit behavior clauses, and pre-change
|
|
326
|
+
repository context. A solution marked wrong must violate a named user-observable behavior; a different but valid
|
|
327
|
+
implementation is not an adversary. Check cross-file integration and preservation constraints, not code style or
|
|
328
|
+
similarity to a reference patch. Return exactly one review for every supplied solution. Use uncertain when the
|
|
329
|
+
available evidence cannot justify either label. Do not see hidden tests or repair any solution. Return only the
|
|
330
|
+
requested structured object.${validControlReviewVersion === undefined ? "" : `\n\n${VALID_CONTROL_INDEPENDENCE_INSTRUCTIONS}`}${validControlReviewVersion === 2
|
|
331
|
+
? `\n\n${BEHAVIORAL_VALID_CONTROL_POLICY_INSTRUCTIONS}
|
|
332
|
+
|
|
333
|
+
For implementationIndependence.sourcePaths, cite at least one fileOverride path belonging to the implementation
|
|
334
|
+
being assessed. Additional supporting citations may name unchanged source entries whose content bytes were
|
|
335
|
+
actually supplied in preChangeSources, repositoryEnvironment.sources, or repositoryEnvironment.localImports.sources.
|
|
336
|
+
An import specifier, omitted or unresolved dependency metadata, or a path mentioned without source contents is
|
|
337
|
+
not inspected source evidence. Supporting context citations do not establish a different implementation mechanism.`
|
|
338
|
+
: ""}`;
|
|
339
|
+
/**
|
|
340
|
+
* One preparation boundary for generation, repair, and prospective calibration.
|
|
341
|
+
* Preserve the legacy packet property order, prompt, and schema when no policy
|
|
342
|
+
* marker is present; calibration must measure this operation, not a copied prompt.
|
|
343
|
+
*/
|
|
344
|
+
export const prepareRepositorySemanticReviewV1 = (input) => {
|
|
345
|
+
const reviewIndependence = input.validControlReviewVersion !== undefined;
|
|
346
|
+
const outputSchema = reviewIndependence
|
|
347
|
+
? SolutionReviewWithIndependenceProposal
|
|
348
|
+
: SolutionReviewProposal;
|
|
349
|
+
return {
|
|
350
|
+
role: input.role,
|
|
351
|
+
instructions: solutionCriticInstructions(input.role === "solution-critic-a" ? "contract" : "integration", input.validControlReviewVersion),
|
|
352
|
+
input: {
|
|
353
|
+
...input.modelContext,
|
|
354
|
+
visible: input.visible,
|
|
355
|
+
targetBehavior: input.targetBehavior,
|
|
356
|
+
preChangeSources: input.preChangeSources,
|
|
357
|
+
solutions: input.solutions.map((solution) => ({
|
|
358
|
+
id: solution.id,
|
|
359
|
+
family: solution.family,
|
|
360
|
+
fileOverrides: solution.fileOverrides
|
|
361
|
+
})),
|
|
362
|
+
...(reviewIndependence
|
|
363
|
+
? { implementationComparison: { solutionIds: [...input.comparisonSolutionIds] } }
|
|
364
|
+
: {})
|
|
365
|
+
},
|
|
366
|
+
schemaName: reviewIndependence
|
|
367
|
+
? "routekit_repository_solution_independence_review_v1"
|
|
368
|
+
: "routekit_repository_solution_review_v1",
|
|
369
|
+
outputSchema,
|
|
370
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1(input.role)
|
|
371
|
+
};
|
|
372
|
+
};
|
|
373
|
+
export const executeRepositorySemanticReviewV1 = Effect.fn("CaseGeneration.reviewSemanticSolutions")(function* (input) {
|
|
374
|
+
const languageModel = yield* RepositoryFoundryLanguageModel;
|
|
375
|
+
return yield* languageModel.generateStructured({
|
|
376
|
+
plan: input.modelPlan,
|
|
377
|
+
operationId: input.operationId,
|
|
378
|
+
...input.prepared
|
|
379
|
+
});
|
|
380
|
+
});
|
|
381
|
+
const qualityReviewInstructions = (validControlReviewVersion) => `You are the independent benchmark-quality reviewer.
|
|
382
|
+
Review the complete supplied specification, executable fixture overlays, independent valid implementations,
|
|
383
|
+
development adversaries, frozen-oracle held-out adversaries, deterministic tournament evidence, and trajectory
|
|
384
|
+
policy. Inspect actual artifact contents rather than trusting family labels or prior reviewers.
|
|
385
|
+
When requestedDimension is supplied, the task and executable fixtures must assess that area through the authentic
|
|
386
|
+
behavior change. Reject an off-dimension task or any omitted critical seed behavior; mentioning the dimension is
|
|
387
|
+
not evidence of fit. Do not invent additional requirements that the historical reference cannot satisfy. Reject correlated
|
|
388
|
+
specification-oracle assumptions, ${validControlReviewVersion === 2 ? "" : "materially similar valid implementations, "}implementation-specific tests,
|
|
389
|
+
development and held-out adversaries that are not meaningfully distinct, missing critical behavior, unrealistic
|
|
390
|
+
adversaries, a frozen oracle that failed any held-out adversary, biased or unverifiable trajectory requirements,
|
|
391
|
+
or a task that permits incompatible interpretations. Return exactly one evidence-linked finding for every
|
|
392
|
+
required review axis. Every evidence id must come from the supplied inventory, and each finding must cite the
|
|
393
|
+
concrete artifacts it assessed. Do not rewrite artifacts.
|
|
394
|
+
Development adversaries may inspect draft fixtures to expose gaps before freeze; that is development feedback,
|
|
395
|
+
not independent held-out evidence. Independent valid solvers, semantic solution critics, and the held-out
|
|
396
|
+
adversary receive no fixture contents. Assess held-out novelty and complete executed outcomes independently.
|
|
397
|
+
Return only the requested structured review.${validControlReviewVersion === 2 ? `\n\n${BEHAVIORAL_VALID_CONTROL_POLICY_INSTRUCTIONS}\nFor the valid-solution-independence axis, assess the isolated solver provenance, semantic correctness reviews, distinct executable alternatives, and repeated acceptance evidence. Algorithmic correlation remains a recorded limitation; do not put it in correlatedAssumptions unless you identify a concrete unsupported behavioral assumption.` : ""}`;
|
|
398
|
+
const HILL_CLIMB_INSTRUCTIONS = `You are the oracle-repair role in a repository coding benchmark foundry.
|
|
399
|
+
The executable tournament found either plausible wrong solutions that escaped the current hidden oracle,
|
|
400
|
+
independently valid solutions that the oracle rejected, or repeated inconclusive fixture executions. Inconclusive
|
|
401
|
+
executions are not behavioral kills. For the supplied fixtureEvidenceFailures, repair how the visible behavior
|
|
402
|
+
is asserted without disguising setup errors, changing solution labels, or modifying a solution. Name every
|
|
403
|
+
affected solution in targetedInconclusiveIds; leave that list empty when no such failures are supplied.
|
|
404
|
+
Produce a complete replacement fixture proposal
|
|
405
|
+
that rejects wrong solutions only for user-observable contract violations and accepts every supplied independent
|
|
406
|
+
valid solution. Remove implementation-specific assertions instead of weakening the visible contract. Use the
|
|
407
|
+
authentic behavior evidence, not identifiers, source-string fingerprints, or historical implementation details,
|
|
408
|
+
and preserve useful existing coverage. Produce exactly one independently executable complete-file overlay per
|
|
409
|
+
fixture, bound to only that fixture id; overlays may reuse a reviewed test path because they are executed in
|
|
410
|
+
separate clean checkouts. Counterfactual and historical-regression fixtures must use expectationMode "changes";
|
|
411
|
+
metamorphic fixtures must vary an input or context while preserving a named user-observable invariant and use
|
|
412
|
+
expectationMode "preserved". Name every current false accept and false reject that the proposal targets. Prior
|
|
413
|
+
rejected attempts are evidence about what not to repeat. When behavioralScope is supplied, re-declare exactly
|
|
414
|
+
one scopeCoverage record for every supplied scope id in the replacement fixtureProposal. Mark covered only
|
|
415
|
+
when its existing fixtureIds assert the complete clause; explain the actual coverage in detail. Do not copy a
|
|
416
|
+
prior attestation without checking the replacement assertions. Report unsupported, contradictory, or uncertain
|
|
417
|
+
coverage honestly; these outcomes stop generation. knownLimitations cannot excuse omission of required behavior.
|
|
418
|
+
Return only the requested structured object.
|
|
419
|
+
|
|
420
|
+
${REPOSITORY_FIXTURE_EVIDENCE_INSTRUCTIONS}`;
|
|
421
|
+
const failure = (detail, cause) => new RepositoryFoundryError({
|
|
422
|
+
operation: "compile-reviewed-case",
|
|
423
|
+
detail,
|
|
424
|
+
...(cause === undefined ? {} : { cause })
|
|
425
|
+
});
|
|
426
|
+
/** Model-attested coverage and identifier integrity, not proof of assertion semantics. */
|
|
427
|
+
export const assertRepositoryFixtureScopeCoverageV1 = (proposal, scope) => {
|
|
428
|
+
if (scope === undefined)
|
|
429
|
+
return;
|
|
430
|
+
const scopeIds = new Set(scope.map((clause) => clause.id));
|
|
431
|
+
const fixtureIds = new Set(proposal.fixtures.map((fixture) => fixture.id));
|
|
432
|
+
const coverage = proposal.scopeCoverage;
|
|
433
|
+
if (scope.length === 0 ||
|
|
434
|
+
scopeIds.size !== scope.length ||
|
|
435
|
+
coverage === undefined ||
|
|
436
|
+
coverage.length !== scope.length ||
|
|
437
|
+
new Set(coverage.map((entry) => entry.scopeId)).size !== scope.length ||
|
|
438
|
+
coverage.some((entry) => !scopeIds.has(entry.scopeId) ||
|
|
439
|
+
entry.outcome !== "covered" ||
|
|
440
|
+
entry.detail.trim().length === 0 ||
|
|
441
|
+
entry.fixtureIds.length === 0 ||
|
|
442
|
+
new Set(entry.fixtureIds).size !== entry.fixtureIds.length ||
|
|
443
|
+
entry.fixtureIds.some((id) => !fixtureIds.has(id)))) {
|
|
444
|
+
throw failure("fixture proposal requires covered, explicit evidence for every grounded behavioral scope clause with existing fixture IDs; known limitations cannot excuse mandatory omissions");
|
|
445
|
+
}
|
|
446
|
+
};
|
|
447
|
+
const detailOf = (cause) => cause instanceof Error && cause.message.trim().length > 0 ? cause.message : String(cause);
|
|
448
|
+
export const generateRepositorySolutionProposalV1 = Effect.fn("CaseGeneration.generateSolutionProposal")(function* (input) {
|
|
449
|
+
const languageModel = yield* RepositoryFoundryLanguageModel;
|
|
450
|
+
const request = input.request;
|
|
451
|
+
const generated = yield* languageModel.generateStructured({
|
|
452
|
+
...request,
|
|
453
|
+
instructions: `${request.instructions}\n\n${EXACT_EDIT_INSTRUCTIONS}`,
|
|
454
|
+
outputSchema: SolutionOutputProposal
|
|
455
|
+
});
|
|
456
|
+
const materialized = yield* Effect.try({
|
|
457
|
+
try: () => {
|
|
458
|
+
let materializedBytes = 0;
|
|
459
|
+
return generated.value.solutions.map((solution) => {
|
|
460
|
+
const result = materializeSolutionFileOverridesV1({
|
|
461
|
+
fileOverrides: solution.fileOverrides,
|
|
462
|
+
sources: input.editSources,
|
|
463
|
+
allowedPaths: (request.role === "adversary" || request.role === "held-out-adversary") &&
|
|
464
|
+
solution.kind === "reward-hacking"
|
|
465
|
+
? input.allowedRewardHackingPaths
|
|
466
|
+
: input.allowedSolutionPaths
|
|
467
|
+
});
|
|
468
|
+
materializedBytes += result.fileOverrides.reduce((total, override) => total + Buffer.byteLength(override.content), 0);
|
|
469
|
+
if (materializedBytes > 32 * 1024 * 1024) {
|
|
470
|
+
throw failure("materialized solution proposal exceeds the 32MiB total source bound");
|
|
471
|
+
}
|
|
472
|
+
return {
|
|
473
|
+
solution: { ...solution, fileOverrides: result.fileOverrides },
|
|
474
|
+
provenance: { solutionId: solution.id, files: result.provenance }
|
|
475
|
+
};
|
|
476
|
+
});
|
|
477
|
+
},
|
|
478
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
479
|
+
? cause
|
|
480
|
+
: failure("solution exact-edit materialization failed", cause)
|
|
481
|
+
});
|
|
482
|
+
const value = {
|
|
483
|
+
solutions: materialized.map(({ solution }) => solution)
|
|
484
|
+
};
|
|
485
|
+
yield* reportFoundryAuthoringArtifactV1({
|
|
486
|
+
version: 1,
|
|
487
|
+
operationId: `${request.operationId}:materialized`,
|
|
488
|
+
role: "host-exact-edit-materializer",
|
|
489
|
+
model: "host",
|
|
490
|
+
reasoningEffort: "none",
|
|
491
|
+
schemaName: "routekit_repository_solution_materialization_v1",
|
|
492
|
+
validation: "validated",
|
|
493
|
+
request: {
|
|
494
|
+
instructions: "Deterministic host materialization of a separately retained raw model response; no model request was made for this artifact.",
|
|
495
|
+
input: {
|
|
496
|
+
sourceModelOperationId: request.operationId,
|
|
497
|
+
initialCommit: input.initialCommit
|
|
498
|
+
},
|
|
499
|
+
jsonSchema: { type: "object", properties: {}, additionalProperties: false },
|
|
500
|
+
maximumOutputTokens: 0
|
|
501
|
+
},
|
|
502
|
+
response: { text: JSON.stringify(value) },
|
|
503
|
+
value: {
|
|
504
|
+
method: "pinned-pre-change-exact-edits-v1",
|
|
505
|
+
sourceModelOperationId: request.operationId,
|
|
506
|
+
initialCommit: input.initialCommit,
|
|
507
|
+
provenance: materialized.map(({ provenance }) => provenance),
|
|
508
|
+
proposal: value
|
|
509
|
+
}
|
|
510
|
+
});
|
|
511
|
+
return { value, call: generated.call, rawProposal: generated.value };
|
|
512
|
+
});
|
|
513
|
+
const bounded = (content, maximumBytes) => {
|
|
514
|
+
const bytes = Buffer.from(content);
|
|
515
|
+
if (bytes.byteLength <= maximumBytes)
|
|
516
|
+
return content;
|
|
517
|
+
return `${bytes.subarray(0, Math.max(0, maximumBytes - 32)).toString("utf8")}\n[truncated]`;
|
|
518
|
+
};
|
|
519
|
+
const sourcePacket = Effect.fn("CaseGeneration.sourcePacket")(function* (input) {
|
|
520
|
+
const snapshot = yield* captureGitTreeSnapshotV1({
|
|
521
|
+
repositoryRoot: input.repositoryRoot,
|
|
522
|
+
requestedRef: input.commit
|
|
523
|
+
});
|
|
524
|
+
const sources = [];
|
|
525
|
+
let remaining = 80_000;
|
|
526
|
+
for (const path of [...new Set(input.paths)].slice(0, 10)) {
|
|
527
|
+
if (!snapshot.files.includes(path) || remaining <= 0)
|
|
528
|
+
continue;
|
|
529
|
+
const content = yield* readGitTreeFileV1({ snapshot, path });
|
|
530
|
+
const excerpt = bounded(content, Math.min(remaining, 20_000));
|
|
531
|
+
sources.push({ path, content: excerpt });
|
|
532
|
+
remaining -= Buffer.byteLength(excerpt);
|
|
533
|
+
}
|
|
534
|
+
return { snapshot, sources };
|
|
535
|
+
});
|
|
536
|
+
const fixtureSuiteFrom = (caseId, proposal) => ({
|
|
537
|
+
version: 1,
|
|
538
|
+
caseId,
|
|
539
|
+
fixtures: proposal.fixtures.map((fixture) => ({
|
|
540
|
+
...fixture,
|
|
541
|
+
source: fixture.kind === "historical-regression" ? "historical" : "generated"
|
|
542
|
+
})),
|
|
543
|
+
overlays: proposal.overlays
|
|
544
|
+
});
|
|
545
|
+
const validateSolutions = (proposal, input) => {
|
|
546
|
+
if (input.exact !== undefined && proposal.solutions.length !== input.exact) {
|
|
547
|
+
throw failure(`${input.kind} solution generation produced ${String(proposal.solutions.length)}; exactly ${String(input.exact)} are required`);
|
|
548
|
+
}
|
|
549
|
+
if (proposal.solutions.length < input.minimum) {
|
|
550
|
+
throw failure(`${input.kind} solution generation produced ${String(proposal.solutions.length)}; at least ${String(input.minimum)} are required`);
|
|
551
|
+
}
|
|
552
|
+
const ids = new Set(proposal.solutions.map((solution) => solution.id));
|
|
553
|
+
if (ids.size !== proposal.solutions.length ||
|
|
554
|
+
ids.has("historical-reference") ||
|
|
555
|
+
ids.has("historical-defect")) {
|
|
556
|
+
throw failure(`${input.kind} solution generation produced duplicate or reserved ids`);
|
|
557
|
+
}
|
|
558
|
+
const validated = proposal.solutions.map((solution) => {
|
|
559
|
+
const rewardHackingPaths = solution.fileOverrides.filter((override) => input.allowedRewardHackingPaths.has(override.path));
|
|
560
|
+
if (solution.id.trim().length === 0 ||
|
|
561
|
+
solution.family.trim().length === 0 ||
|
|
562
|
+
solution.fileOverrides.length === 0 ||
|
|
563
|
+
new Set(solution.fileOverrides.map((override) => override.path)).size !==
|
|
564
|
+
solution.fileOverrides.length) {
|
|
565
|
+
throw failure(`${input.kind} solution ${solution.id} is incomplete`);
|
|
566
|
+
}
|
|
567
|
+
for (const override of solution.fileOverrides) {
|
|
568
|
+
const kindAllowed = input.kind === "valid"
|
|
569
|
+
? solution.kind === "independent-valid" ||
|
|
570
|
+
solution.kind === "simplified-valid" ||
|
|
571
|
+
solution.kind === "behavior-preserving-valid"
|
|
572
|
+
: solution.kind === "mutation" ||
|
|
573
|
+
solution.kind === "model-near-miss" ||
|
|
574
|
+
solution.kind === "reward-hacking";
|
|
575
|
+
const pathAllowed = input.kind === "wrong" && solution.kind === "reward-hacking"
|
|
576
|
+
? input.allowedPaths.has(override.path) ||
|
|
577
|
+
input.allowedRewardHackingPaths.has(override.path)
|
|
578
|
+
: input.allowedPaths.has(override.path);
|
|
579
|
+
if (!kindAllowed ||
|
|
580
|
+
!pathAllowed ||
|
|
581
|
+
override.content.trim().length === 0 ||
|
|
582
|
+
Buffer.byteLength(override.content) > 256_000) {
|
|
583
|
+
throw failure(`${input.kind} solution ${solution.id} targets an unreviewed or oversized path: ${override.path}`);
|
|
584
|
+
}
|
|
585
|
+
}
|
|
586
|
+
if (input.kind === "wrong" &&
|
|
587
|
+
solution.kind === "reward-hacking" &&
|
|
588
|
+
rewardHackingPaths.length === 0) {
|
|
589
|
+
throw failure(`wrong solution ${solution.id} is classified as reward-hacking but does not modify a reviewed protected test path`);
|
|
590
|
+
}
|
|
591
|
+
return {
|
|
592
|
+
id: solution.id,
|
|
593
|
+
kind: solution.kind,
|
|
594
|
+
family: solution.family,
|
|
595
|
+
expectedClass: input.kind,
|
|
596
|
+
commit: input.commit,
|
|
597
|
+
fileOverrides: solution.fileOverrides
|
|
598
|
+
};
|
|
599
|
+
});
|
|
600
|
+
if (input.kind === "wrong") {
|
|
601
|
+
const kinds = new Set(validated.map((solution) => solution.kind));
|
|
602
|
+
const families = new Set(validated.map((solution) => solution.family));
|
|
603
|
+
const requiredKinds = ["mutation", "model-near-miss", "reward-hacking"];
|
|
604
|
+
const missingKinds = requiredKinds.filter((kind) => !kinds.has(kind));
|
|
605
|
+
if (missingKinds.length > 0 || families.size < 3) {
|
|
606
|
+
throw failure([
|
|
607
|
+
"adversary proposal lacks required deterministic diversity",
|
|
608
|
+
...(missingKinds.length > 0 ? [`missing kinds: ${missingKinds.join(", ")}`] : []),
|
|
609
|
+
...(families.size < 3
|
|
610
|
+
? [`distinct families: ${String(families.size)}; at least 3 are required`]
|
|
611
|
+
: [])
|
|
612
|
+
].join("; "));
|
|
613
|
+
}
|
|
614
|
+
}
|
|
615
|
+
return validated;
|
|
616
|
+
};
|
|
617
|
+
/** Eligibility is a necessary population bound; execution must still prove every semantic kill. */
|
|
618
|
+
export const assertRepositorySemanticControlPopulationV1 = (solutions, protectedControlPaths) => {
|
|
619
|
+
const protectedPaths = new Set(protectedControlPaths);
|
|
620
|
+
const eligible = solutions.filter((solution) => solution.expectedClass === "wrong" &&
|
|
621
|
+
solution.id !== "historical-defect" &&
|
|
622
|
+
!(solution.fileOverrides ?? []).some((override) => protectedPaths.has(override.path)));
|
|
623
|
+
if (eligible.length < 5) {
|
|
624
|
+
throw failure(`adversary proposal has ${String(eligible.length)} generated semantic-eligible wrong controls; at least 5 are required beyond protected-path policy controls and the historical defect`);
|
|
625
|
+
}
|
|
626
|
+
};
|
|
627
|
+
const validControlPairFindings = (left, right, preChangeSources, validControlReviewVersion) => {
|
|
628
|
+
const first = left[0];
|
|
629
|
+
const second = right[0];
|
|
630
|
+
if (first === undefined || second === undefined) {
|
|
631
|
+
return [
|
|
632
|
+
{
|
|
633
|
+
code: "valid-control-missing-proposal",
|
|
634
|
+
detail: "Each independent solver must produce one solution."
|
|
635
|
+
}
|
|
636
|
+
];
|
|
637
|
+
}
|
|
638
|
+
const patchSignature = (solution) => JSON.stringify([...(solution.fileOverrides ?? [])]
|
|
639
|
+
.sort((a, b) => a.path.localeCompare(b.path))
|
|
640
|
+
.map((override) => ({
|
|
641
|
+
path: override.path,
|
|
642
|
+
...("content" in override ? { content: override.content } : { delete: override.delete })
|
|
643
|
+
})));
|
|
644
|
+
const findings = [];
|
|
645
|
+
if (first.id === second.id)
|
|
646
|
+
findings.push({
|
|
647
|
+
code: "valid-control-duplicate-id",
|
|
648
|
+
detail: "Use a distinct solution identifier describing your own implementation."
|
|
649
|
+
});
|
|
650
|
+
if (validControlReviewVersion !== 2 && first.family === second.family)
|
|
651
|
+
findings.push({
|
|
652
|
+
code: "valid-control-duplicate-family",
|
|
653
|
+
detail: "The implementation-family labels collide. Describe your own concrete mechanism rather than the task or dimension; do not invent a distinction unsupported by your implementation."
|
|
654
|
+
});
|
|
655
|
+
if (patchSignature(first) === patchSignature(second) ||
|
|
656
|
+
(validControlReviewVersion === 2 &&
|
|
657
|
+
normalizedPatchTokens(first).join("\u0000") === normalizedPatchTokens(second).join("\u0000")))
|
|
658
|
+
findings.push({
|
|
659
|
+
code: "valid-control-duplicate-patch",
|
|
660
|
+
detail: validControlReviewVersion === 2
|
|
661
|
+
? "The implementations are identical. Provide a natural executable alternative while preserving the entire visible contract. An equivalent formulation is sufficient; changing only labels, comments, or formatting is insufficient."
|
|
662
|
+
: "The implementations are identical. Independently redesign your own implementation while preserving the entire visible contract; changing labels, comments, or formatting is insufficient."
|
|
663
|
+
});
|
|
664
|
+
if (validControlReviewVersion !== 2 && patchSimilarity(first, second, preChangeSources) >= 0.9)
|
|
665
|
+
findings.push({
|
|
666
|
+
code: "valid-control-similar-patch",
|
|
667
|
+
detail: "The proposals have materially similar implementation deltas. Independently choose a different concrete design; renaming, comments, formatting, and family labels do not establish source diversity."
|
|
668
|
+
});
|
|
669
|
+
if (first.kind !== "independent-valid" && first.kind !== "simplified-valid")
|
|
670
|
+
findings.push({
|
|
671
|
+
code: "valid-control-invalid-a-kind",
|
|
672
|
+
detail: "Solver A must classify its own valid implementation as independent-valid or simplified-valid."
|
|
673
|
+
});
|
|
674
|
+
if (second.kind !== "behavior-preserving-valid")
|
|
675
|
+
findings.push({
|
|
676
|
+
code: "valid-control-invalid-b-kind",
|
|
677
|
+
detail: "Solver B must classify its own valid implementation as behavior-preserving-valid."
|
|
678
|
+
});
|
|
679
|
+
return findings;
|
|
680
|
+
};
|
|
681
|
+
export const assertIndependentValidSolutionPairV1 = (left, right, preChangeSources, validControlReviewVersion) => {
|
|
682
|
+
const findings = validControlPairFindings(left, right, preChangeSources, validControlReviewVersion);
|
|
683
|
+
if (findings.length > 0)
|
|
684
|
+
throw failure(findings.map(({ code, detail }) => `${code}: ${detail}`).join("; "));
|
|
685
|
+
};
|
|
686
|
+
const solutionPatchSignature = (solution) => JSON.stringify([...(solution.fileOverrides ?? [])]
|
|
687
|
+
.sort((a, b) => a.path.localeCompare(b.path))
|
|
688
|
+
.map((override) => ({
|
|
689
|
+
path: override.path,
|
|
690
|
+
...("content" in override ? { content: override.content } : { delete: override.delete })
|
|
691
|
+
})));
|
|
692
|
+
const changedContent = (initial, next) => {
|
|
693
|
+
if (initial === undefined)
|
|
694
|
+
return next;
|
|
695
|
+
const before = initial.split("\n");
|
|
696
|
+
const after = next.split("\n");
|
|
697
|
+
let prefix = 0;
|
|
698
|
+
while (prefix < before.length && prefix < after.length && before[prefix] === after[prefix]) {
|
|
699
|
+
prefix += 1;
|
|
700
|
+
}
|
|
701
|
+
let suffix = 0;
|
|
702
|
+
while (suffix < before.length - prefix &&
|
|
703
|
+
suffix < after.length - prefix &&
|
|
704
|
+
before[before.length - 1 - suffix] === after[after.length - 1 - suffix]) {
|
|
705
|
+
suffix += 1;
|
|
706
|
+
}
|
|
707
|
+
return after.slice(prefix, after.length - suffix).join("\n");
|
|
708
|
+
};
|
|
709
|
+
const normalizedPatchTokens = (solution, preChangeSources = []) => {
|
|
710
|
+
const initialByPath = new Map(preChangeSources.map((source) => [source.path, source.content]));
|
|
711
|
+
const normalized = [...(solution.fileOverrides ?? [])]
|
|
712
|
+
.sort((left, right) => left.path.localeCompare(right.path))
|
|
713
|
+
.map((override) => {
|
|
714
|
+
const content = "content" in override
|
|
715
|
+
? changedContent(initialByPath.get(override.path), override.content)
|
|
716
|
+
.replace(/\/\*[\s\S]*?\*\//gu, " ")
|
|
717
|
+
.replace(/(^|[^:])\/\/.*$/gmu, "$1 ")
|
|
718
|
+
.replace(/^\s*#.*$/gmu, " ")
|
|
719
|
+
: "[deleted]";
|
|
720
|
+
return `${override.path}\n${content}`;
|
|
721
|
+
})
|
|
722
|
+
.join("\n")
|
|
723
|
+
.match(/[A-Za-z_$][A-Za-z0-9_$]*|\d+(?:\.\d+)?|===|!==|=>|==|!=|<=|>=|&&|\|\||\?\?|[^\s]/gu);
|
|
724
|
+
return normalized ?? [];
|
|
725
|
+
};
|
|
726
|
+
const patchShingles = (solution, preChangeSources = []) => {
|
|
727
|
+
const tokens = normalizedPatchTokens(solution, preChangeSources);
|
|
728
|
+
const width = Math.min(5, Math.max(1, tokens.length));
|
|
729
|
+
const shingles = new Set();
|
|
730
|
+
for (let index = 0; index <= tokens.length - width; index += 1) {
|
|
731
|
+
shingles.add(tokens.slice(index, index + width).join("\u0000"));
|
|
732
|
+
}
|
|
733
|
+
return shingles;
|
|
734
|
+
};
|
|
735
|
+
const patchSimilarity = (left, right, preChangeSources = []) => {
|
|
736
|
+
const leftShingles = patchShingles(left, preChangeSources);
|
|
737
|
+
const rightShingles = patchShingles(right, preChangeSources);
|
|
738
|
+
if (leftShingles.size === 0 && rightShingles.size === 0)
|
|
739
|
+
return 1;
|
|
740
|
+
let intersection = 0;
|
|
741
|
+
for (const shingle of leftShingles) {
|
|
742
|
+
if (rightShingles.has(shingle))
|
|
743
|
+
intersection += 1;
|
|
744
|
+
}
|
|
745
|
+
const union = leftShingles.size + rightShingles.size - intersection;
|
|
746
|
+
return union === 0 ? 1 : intersection / union;
|
|
747
|
+
};
|
|
748
|
+
export const assertHeldOutAdversaryNoveltyV1 = (development, heldOut, preChangeSources = []) => {
|
|
749
|
+
const developmentIds = new Set(development.map((solution) => solution.id));
|
|
750
|
+
const developmentFamilies = new Set(development.map((solution) => solution.family));
|
|
751
|
+
const developmentSignatures = new Set(development.map(solutionPatchSignature));
|
|
752
|
+
const heldOutSignatures = heldOut.map(solutionPatchSignature);
|
|
753
|
+
const familyCollisions = heldOut
|
|
754
|
+
.filter((solution) => developmentFamilies.has(solution.family))
|
|
755
|
+
.map((solution) => solution.family);
|
|
756
|
+
const nearDuplicates = heldOut.flatMap((heldOutSolution) => development
|
|
757
|
+
.filter((developmentSolution) => patchSimilarity(developmentSolution, heldOutSolution, preChangeSources) >= 0.9)
|
|
758
|
+
.map((developmentSolution) => `${heldOutSolution.id}~${developmentSolution.id}`));
|
|
759
|
+
if (heldOut.some((solution) => developmentIds.has(solution.id)) ||
|
|
760
|
+
heldOutSignatures.some((signature) => developmentSignatures.has(signature)) ||
|
|
761
|
+
new Set(heldOutSignatures).size !== heldOutSignatures.length ||
|
|
762
|
+
familyCollisions.length > 0 ||
|
|
763
|
+
nearDuplicates.length > 0) {
|
|
764
|
+
throw failure([
|
|
765
|
+
"held-out adversaries must use new failure families and materially new patches that were not used to develop the frozen oracle",
|
|
766
|
+
...(familyCollisions.length === 0
|
|
767
|
+
? []
|
|
768
|
+
: [`reused families: ${[...new Set(familyCollisions)].sort().join(", ")}`]),
|
|
769
|
+
...(nearDuplicates.length === 0
|
|
770
|
+
? []
|
|
771
|
+
: [`near-duplicate patches: ${nearDuplicates.sort().join(", ")}`])
|
|
772
|
+
].join("; "));
|
|
773
|
+
}
|
|
774
|
+
};
|
|
775
|
+
const validateHeldOutAdversaryProposal = (input) => {
|
|
776
|
+
try {
|
|
777
|
+
const solutions = validateSolutions(input.proposal, {
|
|
778
|
+
kind: "wrong",
|
|
779
|
+
commit: input.commit,
|
|
780
|
+
allowedPaths: input.allowedPaths,
|
|
781
|
+
allowedRewardHackingPaths: input.allowedRewardHackingPaths,
|
|
782
|
+
minimum: 5
|
|
783
|
+
});
|
|
784
|
+
assertRepositorySemanticControlPopulationV1(solutions, input.protectedControlPaths);
|
|
785
|
+
assertHeldOutAdversaryNoveltyV1(input.development, solutions, input.preChangeSources);
|
|
786
|
+
return { _tag: "valid", solutions };
|
|
787
|
+
}
|
|
788
|
+
catch (cause) {
|
|
789
|
+
return { _tag: "invalid", detail: detailOf(cause) };
|
|
790
|
+
}
|
|
791
|
+
};
|
|
792
|
+
const QUALITY_REVIEW_AXES = [
|
|
793
|
+
"specification",
|
|
794
|
+
"oracle",
|
|
795
|
+
"valid-solution-independence",
|
|
796
|
+
"development-adversaries",
|
|
797
|
+
"held-out-adversaries",
|
|
798
|
+
"trajectory"
|
|
799
|
+
];
|
|
800
|
+
const QUALITY_EVIDENCE_PREFIXES = {
|
|
801
|
+
specification: [
|
|
802
|
+
"review-packet:/visible",
|
|
803
|
+
"review-packet:/targetBehavior/",
|
|
804
|
+
"review-packet:/specificationReviews/"
|
|
805
|
+
],
|
|
806
|
+
oracle: [
|
|
807
|
+
"review-packet:/fixtureProposal/",
|
|
808
|
+
"review-packet:/oracleCoverageReviews/",
|
|
809
|
+
"review-packet:/oracleCoverageWitnessRevisions/",
|
|
810
|
+
"review-packet:/executableAdequacy",
|
|
811
|
+
"review-packet:/developmentOracle/",
|
|
812
|
+
"review-packet:/heldOutOracle/"
|
|
813
|
+
],
|
|
814
|
+
"valid-solution-independence": ["review-packet:/validSolutions/"],
|
|
815
|
+
"development-adversaries": [
|
|
816
|
+
"review-packet:/developmentAdversaries/",
|
|
817
|
+
"review-packet:/solutionReviews/",
|
|
818
|
+
"review-packet:/developmentOracle/"
|
|
819
|
+
],
|
|
820
|
+
"held-out-adversaries": [
|
|
821
|
+
"review-packet:/heldOutAdversaries/",
|
|
822
|
+
"review-packet:/heldOutSolutionReviews/",
|
|
823
|
+
"review-packet:/heldOutAdequacy",
|
|
824
|
+
"review-packet:/heldOutOracle/"
|
|
825
|
+
],
|
|
826
|
+
trajectory: [
|
|
827
|
+
"review-packet:/trajectoryPolicy",
|
|
828
|
+
"review-packet:/trajectoryPolicyReview",
|
|
829
|
+
"review-packet:/trajectoryRevisionAttempts/"
|
|
830
|
+
]
|
|
831
|
+
};
|
|
832
|
+
const qualityReviewEvidenceInventory = (input) => [
|
|
833
|
+
"review-packet:/visible",
|
|
834
|
+
...Array.from({ length: input.targetBehaviorCount }, (_, index) => `review-packet:/targetBehavior/${String(index)}`),
|
|
835
|
+
...Array.from({ length: input.specificationReviewCount }, (_, index) => `review-packet:/specificationReviews/${String(index)}`),
|
|
836
|
+
...Array.from({ length: input.fixtureCount }, (_, index) => `review-packet:/fixtureProposal/fixtures/${String(index)}`),
|
|
837
|
+
...Array.from({ length: input.overlayCount }, (_, index) => `review-packet:/fixtureProposal/overlays/${String(index)}`),
|
|
838
|
+
...Array.from({ length: input.oracleCoverageReviewCount ?? 0 }, (_, index) => `review-packet:/oracleCoverageReviews/${String(index)}`),
|
|
839
|
+
...Array.from({ length: input.oracleCoverageWitnessRevisionCount ?? 0 }, (_, index) => `review-packet:/oracleCoverageWitnessRevisions/${String(index)}`),
|
|
840
|
+
...Array.from({ length: input.validSolutionCount }, (_, index) => `review-packet:/validSolutions/${String(index)}`),
|
|
841
|
+
...Array.from({ length: input.developmentAdversaryCount }, (_, index) => `review-packet:/developmentAdversaries/${String(index)}`),
|
|
842
|
+
...Array.from({ length: input.solutionReviewCount }, (_, index) => `review-packet:/solutionReviews/${String(index)}`),
|
|
843
|
+
...Array.from({ length: input.heldOutAdversaryCount }, (_, index) => `review-packet:/heldOutAdversaries/${String(index)}`),
|
|
844
|
+
...Array.from({ length: input.heldOutSolutionReviewCount }, (_, index) => `review-packet:/heldOutSolutionReviews/${String(index)}`),
|
|
845
|
+
"review-packet:/heldOutAdequacy",
|
|
846
|
+
"review-packet:/executableAdequacy",
|
|
847
|
+
"review-packet:/developmentOracle/clauses",
|
|
848
|
+
"review-packet:/heldOutOracle/clauses",
|
|
849
|
+
...Array.from({ length: input.developmentObservationCount }, (_, index) => `review-packet:/developmentOracle/observations/${String(index)}`),
|
|
850
|
+
...Array.from({ length: input.heldOutObservationCount }, (_, index) => `review-packet:/heldOutOracle/observations/${String(index)}`),
|
|
851
|
+
"review-packet:/trajectoryPolicy",
|
|
852
|
+
"review-packet:/trajectoryPolicyReview",
|
|
853
|
+
...Array.from({ length: input.trajectoryRevisionAttemptCount }, (_, index) => `review-packet:/trajectoryRevisionAttempts/${String(index)}`),
|
|
854
|
+
...Array.from({ length: input.hillClimbAttemptCount }, (_, index) => `review-packet:/hillClimbAttempts/${String(index)}`)
|
|
855
|
+
];
|
|
856
|
+
export const validateRepositoryGenerationQualityReviewV1 = (proposal, evidenceInventory) => {
|
|
857
|
+
const evidence = new Set(evidenceInventory);
|
|
858
|
+
const findingsByAxis = new Map(proposal.findings.map((finding) => [finding.axis, finding]));
|
|
859
|
+
const invalidFindings = proposal.findings.filter((finding) => {
|
|
860
|
+
const allowedPrefixes = QUALITY_EVIDENCE_PREFIXES[finding.axis];
|
|
861
|
+
return (finding.detail.trim().length === 0 ||
|
|
862
|
+
finding.evidenceIds.length === 0 ||
|
|
863
|
+
new Set(finding.evidenceIds).size !== finding.evidenceIds.length ||
|
|
864
|
+
finding.evidenceIds.some((evidenceId) => !evidence.has(evidenceId)) ||
|
|
865
|
+
!finding.evidenceIds.some((evidenceId) => allowedPrefixes.some((prefix) => evidenceId.startsWith(prefix))));
|
|
866
|
+
});
|
|
867
|
+
if (proposal.detail.trim().length === 0 ||
|
|
868
|
+
proposal.findings.length !== QUALITY_REVIEW_AXES.length ||
|
|
869
|
+
findingsByAxis.size !== QUALITY_REVIEW_AXES.length ||
|
|
870
|
+
QUALITY_REVIEW_AXES.some((axis) => !findingsByAxis.has(axis)) ||
|
|
871
|
+
invalidFindings.length > 0 ||
|
|
872
|
+
(proposal.verdict === "approve" &&
|
|
873
|
+
proposal.findings.some((finding) => finding.outcome !== "pass")) ||
|
|
874
|
+
(proposal.verdict === "reject" &&
|
|
875
|
+
proposal.findings.every((finding) => finding.outcome === "pass"))) {
|
|
876
|
+
throw failure("independent quality review must provide one internally consistent, evidence-linked finding for every required axis");
|
|
877
|
+
}
|
|
878
|
+
};
|
|
879
|
+
const solutionReviewDisagreements = (proposal, solutions, reviewer) => {
|
|
880
|
+
const expected = new Map(solutions.map((solution) => [solution.id, solution.expectedClass]));
|
|
881
|
+
const ids = proposal.reviews.map((review) => review.solutionId);
|
|
882
|
+
if (ids.length !== expected.size ||
|
|
883
|
+
new Set(ids).size !== ids.length ||
|
|
884
|
+
ids.some((id) => !expected.has(id))) {
|
|
885
|
+
throw failure(`${reviewer} must return exactly one classification for every proposed solution`);
|
|
886
|
+
}
|
|
887
|
+
return proposal.reviews
|
|
888
|
+
.filter((review) => review.detail.trim().length === 0 ||
|
|
889
|
+
review.classification !== expected.get(review.solutionId) ||
|
|
890
|
+
(review.classification === "wrong" && review.violatedBehaviorIds.length === 0))
|
|
891
|
+
.map((review) => ({ reviewer, review }));
|
|
892
|
+
};
|
|
893
|
+
const validateSolutionReview = (proposal, solutions, reviewer) => {
|
|
894
|
+
const disagreements = solutionReviewDisagreements(proposal, solutions, reviewer);
|
|
895
|
+
if (disagreements.length > 0) {
|
|
896
|
+
throw failure(`${reviewer} rejected generated solution labels: ${disagreements
|
|
897
|
+
.map(({ review }) => {
|
|
898
|
+
return `${review.solutionId}=${review.classification}${review.detail.trim().length === 0 ? " (missing rationale)" : ` (${review.detail})`}`;
|
|
899
|
+
})
|
|
900
|
+
.join("; ")}`);
|
|
901
|
+
}
|
|
902
|
+
};
|
|
903
|
+
/** Critic prose is private evidence. Only this fixed vocabulary can reach either solver. */
|
|
904
|
+
export const repositorySemanticReviewIndependenceFindingsV1 = (proposal, validSolutions, reviewer, validControlReviewVersion, criticInput) => {
|
|
905
|
+
const suppliedSourcePaths = new Set();
|
|
906
|
+
if (validControlReviewVersion === 2 && criticInput !== undefined) {
|
|
907
|
+
// This is the exact prepared request, not repository-global knowledge or
|
|
908
|
+
// model-returned evidence. Path-only dependency metadata supplies no bytes.
|
|
909
|
+
const includeProvidedSources = (sources) => {
|
|
910
|
+
if (!Array.isArray(sources))
|
|
911
|
+
return;
|
|
912
|
+
for (const entry of sources) {
|
|
913
|
+
if (entry === null || typeof entry !== "object" || Array.isArray(entry))
|
|
914
|
+
continue;
|
|
915
|
+
const source = entry;
|
|
916
|
+
if (Object.hasOwn(source, "path") &&
|
|
917
|
+
Object.hasOwn(source, "content") &&
|
|
918
|
+
typeof source.path === "string" &&
|
|
919
|
+
typeof source.content === "string") {
|
|
920
|
+
suppliedSourcePaths.add(source.path);
|
|
921
|
+
}
|
|
922
|
+
}
|
|
923
|
+
};
|
|
924
|
+
includeProvidedSources(criticInput.preChangeSources);
|
|
925
|
+
const environment = criticInput.repositoryEnvironment;
|
|
926
|
+
if (environment !== null && typeof environment === "object") {
|
|
927
|
+
if ("sources" in environment)
|
|
928
|
+
includeProvidedSources(environment.sources);
|
|
929
|
+
const localImports = "localImports" in environment ? environment.localImports : undefined;
|
|
930
|
+
if (localImports !== null &&
|
|
931
|
+
typeof localImports === "object" &&
|
|
932
|
+
"sources" in localImports) {
|
|
933
|
+
includeProvidedSources(localImports.sources);
|
|
934
|
+
}
|
|
935
|
+
}
|
|
936
|
+
}
|
|
937
|
+
const assessment = proposal.implementationIndependence;
|
|
938
|
+
const expected = new Map(validSolutions.map((solution) => [solution.id, solution]));
|
|
939
|
+
if (assessment === undefined ||
|
|
940
|
+
expected.size !== 2 ||
|
|
941
|
+
assessment.detail.trim().length === 0 ||
|
|
942
|
+
assessment.implementations.length !== 2 ||
|
|
943
|
+
new Set(assessment.implementations.map((entry) => entry.solutionId)).size !== 2 ||
|
|
944
|
+
assessment.implementations.some((entry) => {
|
|
945
|
+
const solution = expected.get(entry.solutionId);
|
|
946
|
+
const overridePaths = new Set(solution?.fileOverrides?.map((override) => override.path));
|
|
947
|
+
return (solution === undefined ||
|
|
948
|
+
entry.mechanism.trim().length === 0 ||
|
|
949
|
+
entry.sourcePaths.length === 0 ||
|
|
950
|
+
new Set(entry.sourcePaths).size !== entry.sourcePaths.length ||
|
|
951
|
+
!entry.sourcePaths.some((path) => overridePaths.has(path)) ||
|
|
952
|
+
entry.sourcePaths.some((path) => !overridePaths.has(path) && !suppliedSourcePaths.has(path)));
|
|
953
|
+
}) ||
|
|
954
|
+
new Set(assessment.correlatedFeatures).size !== assessment.correlatedFeatures.length ||
|
|
955
|
+
(assessment.outcome === "independent") !== (assessment.correlatedFeatures.length === 0)) {
|
|
956
|
+
throw failure(`${reviewer} must separately assess both current implementations with inspected source paths and consistent independence evidence`);
|
|
957
|
+
}
|
|
958
|
+
if (assessment.outcome === "independent" || validControlReviewVersion === 2)
|
|
959
|
+
return [];
|
|
960
|
+
const details = {
|
|
961
|
+
"renaming-or-formatting": "Renaming, formatting, comments, or family labels do not provide a different implementation mechanism.",
|
|
962
|
+
"helper-extraction": "Moving the same behavior into a helper does not provide a different implementation mechanism.",
|
|
963
|
+
"equivalent-control-flow": "Reordering or restating equivalent guards and control flow does not provide a different implementation mechanism.",
|
|
964
|
+
"same-behavioral-mechanism": "Reimplement the changed visible behavior using a substantively different mechanism while preserving all required behavior and integration constraints.",
|
|
965
|
+
"insufficient-evidence": "Independent review could not establish a distinct mechanism. Reassess your own implementation and choose a clearly justified different approach within the visible task."
|
|
966
|
+
};
|
|
967
|
+
return assessment.correlatedFeatures.map((feature) => ({
|
|
968
|
+
code: `valid-control-independence-${feature}`,
|
|
969
|
+
detail: details[feature]
|
|
970
|
+
}));
|
|
971
|
+
};
|
|
972
|
+
const difference = (left, right) => {
|
|
973
|
+
const rightSet = new Set(right);
|
|
974
|
+
return left.filter((entry) => !rightSet.has(entry));
|
|
975
|
+
};
|
|
976
|
+
/**
|
|
977
|
+
* Allows a fresh oracle proposal for repeated, complete test-body evidence errors.
|
|
978
|
+
* The original outcomes remain unknown. Mixed outcomes and operational failures
|
|
979
|
+
* are not admitted to this repair path.
|
|
980
|
+
*/
|
|
981
|
+
export const repairableRepositoryFixtureEvidenceV1 = (oracle, adequacy) => {
|
|
982
|
+
if (adequacy.unstableSolutionIds.length === 0 ||
|
|
983
|
+
adequacy.unobservedClauseIds.length > 0 ||
|
|
984
|
+
adequacy.insufficientObservationPairs.length > 0)
|
|
985
|
+
return [];
|
|
986
|
+
const clauses = new Map(oracle.clauses.map((clause) => [clause.id, clause]));
|
|
987
|
+
const fixtures = new Set(oracle.fixtures.map((fixture) => fixture.id));
|
|
988
|
+
const groups = new Map();
|
|
989
|
+
for (const observation of oracle.observations) {
|
|
990
|
+
const key = JSON.stringify([observation.solutionId, observation.clauseId]);
|
|
991
|
+
const group = groups.get(key) ?? [];
|
|
992
|
+
group.push(observation);
|
|
993
|
+
groups.set(key, group);
|
|
994
|
+
}
|
|
995
|
+
const targets = [];
|
|
996
|
+
for (const observations of groups.values()) {
|
|
997
|
+
const outcomes = new Set(observations.map((observation) => observation.outcome));
|
|
998
|
+
if (outcomes.size !== 1)
|
|
999
|
+
return [];
|
|
1000
|
+
if (!outcomes.has("unknown"))
|
|
1001
|
+
continue;
|
|
1002
|
+
const first = observations[0];
|
|
1003
|
+
const clause = clauses.get(first.clauseId);
|
|
1004
|
+
if (clause?.kind !== "test-command" ||
|
|
1005
|
+
clause.fixtureIds.length === 0 ||
|
|
1006
|
+
clause.fixtureIds.some((id) => !fixtures.has(id)) ||
|
|
1007
|
+
new Set(observations.map((observation) => observation.repetition)).size < 2 ||
|
|
1008
|
+
observations.some((observation) => observation.inconclusiveReason !== "unattributed-test-failure"))
|
|
1009
|
+
return [];
|
|
1010
|
+
targets.push({
|
|
1011
|
+
solutionId: first.solutionId,
|
|
1012
|
+
clauseId: first.clauseId,
|
|
1013
|
+
fixtureIds: clause.fixtureIds,
|
|
1014
|
+
observations
|
|
1015
|
+
});
|
|
1016
|
+
}
|
|
1017
|
+
const affected = new Set(targets.map((target) => target.solutionId));
|
|
1018
|
+
return adequacy.unstableSolutionIds.every((id) => affected.has(id)) ? targets : [];
|
|
1019
|
+
};
|
|
1020
|
+
const assessHillClimbCandidate = (before, candidate, requireImprovement = true) => {
|
|
1021
|
+
const introducedFalseAccepts = difference(candidate.falseAcceptedSolutionIds, before.falseAcceptedSolutionIds);
|
|
1022
|
+
const introducedFalseRejects = difference(candidate.falseRejectedSolutionIds, before.falseRejectedSolutionIds);
|
|
1023
|
+
const introducedInstability = difference(candidate.unstableSolutionIds, before.unstableSolutionIds);
|
|
1024
|
+
const introducedUnobservedClauses = difference(candidate.unobservedClauseIds, before.unobservedClauseIds);
|
|
1025
|
+
const introducedInsufficientPairs = difference(candidate.insufficientObservationPairs, before.insufficientObservationPairs);
|
|
1026
|
+
const beforeClassificationDefects = new Set([
|
|
1027
|
+
...before.falseAcceptedSolutionIds,
|
|
1028
|
+
...before.falseRejectedSolutionIds,
|
|
1029
|
+
...before.unstableSolutionIds
|
|
1030
|
+
]).size;
|
|
1031
|
+
const candidateClassificationDefects = new Set([
|
|
1032
|
+
...candidate.falseAcceptedSolutionIds,
|
|
1033
|
+
...candidate.falseRejectedSolutionIds,
|
|
1034
|
+
...candidate.unstableSolutionIds
|
|
1035
|
+
]).size;
|
|
1036
|
+
return [
|
|
1037
|
+
...(introducedFalseAccepts.length === 0
|
|
1038
|
+
? []
|
|
1039
|
+
: [`introduced false accepts: ${introducedFalseAccepts.join(", ")}`]),
|
|
1040
|
+
...(introducedFalseRejects.length === 0
|
|
1041
|
+
? []
|
|
1042
|
+
: [`introduced false rejects: ${introducedFalseRejects.join(", ")}`]),
|
|
1043
|
+
...(introducedInstability.length === 0
|
|
1044
|
+
? []
|
|
1045
|
+
: [`introduced unstable solutions: ${introducedInstability.join(", ")}`]),
|
|
1046
|
+
...(introducedUnobservedClauses.length === 0
|
|
1047
|
+
? []
|
|
1048
|
+
: [`introduced unobserved clauses: ${introducedUnobservedClauses.join(", ")}`]),
|
|
1049
|
+
...(introducedInsufficientPairs.length === 0
|
|
1050
|
+
? []
|
|
1051
|
+
: [`introduced insufficient observations: ${introducedInsufficientPairs.join(", ")}`]),
|
|
1052
|
+
...(!requireImprovement || candidateClassificationDefects < beforeClassificationDefects
|
|
1053
|
+
? []
|
|
1054
|
+
: [
|
|
1055
|
+
`did not reduce oracle defects (${String(beforeClassificationDefects)} -> ${String(candidateClassificationDefects)})`
|
|
1056
|
+
])
|
|
1057
|
+
];
|
|
1058
|
+
};
|
|
1059
|
+
export class CaseGeneration extends Context.Service()("@velum-labs/routekit-eval-setup/CaseGeneration") {
|
|
1060
|
+
}
|
|
1061
|
+
const generateHistoricalRepositoryCaseStagesV1 = Effect.fn("CaseGeneration.generateHistorical")(function* (input) {
|
|
1062
|
+
const languageModel = yield* RepositoryFoundryLanguageModel;
|
|
1063
|
+
if (input.specificationContractFactsVersion !== undefined &&
|
|
1064
|
+
(input.specificationContractFactsVersion !== 1 ||
|
|
1065
|
+
input.maximumSpecificationRevisions === undefined)) {
|
|
1066
|
+
return yield* failure("specification contract facts require version 1 and an explicit specification revision allowance");
|
|
1067
|
+
}
|
|
1068
|
+
if (input.oracleRepairFeedbackVersion !== undefined &&
|
|
1069
|
+
input.oracleRepairFeedbackVersion !== 1) {
|
|
1070
|
+
return yield* failure("oracle repair feedback version must be 1");
|
|
1071
|
+
}
|
|
1072
|
+
if (input.validControlReviewVersion !== undefined &&
|
|
1073
|
+
![1, 2].includes(input.validControlReviewVersion)) {
|
|
1074
|
+
return yield* failure("valid-control review version must be 1 or 2");
|
|
1075
|
+
}
|
|
1076
|
+
if (input.oracleCoverageWitnessRepairVersion !== undefined &&
|
|
1077
|
+
(input.oracleCoverageWitnessRepairVersion !== 1 ||
|
|
1078
|
+
input.oracleCoverageWitnessVersion === undefined ||
|
|
1079
|
+
input.maximumFixtureRepairAttempts === undefined)) {
|
|
1080
|
+
return yield* failure("coverage witness repair budgeting requires versioned witnesses and an explicit fixture repair allowance");
|
|
1081
|
+
}
|
|
1082
|
+
if (input.oracleCoverageWitnessVersion !== undefined &&
|
|
1083
|
+
(![1, 2].includes(input.oracleCoverageWitnessVersion) ||
|
|
1084
|
+
input.maximumOracleCoverageRevisions === undefined ||
|
|
1085
|
+
input.maximumOracleCoverageRevisions < 1)) {
|
|
1086
|
+
return yield* failure("executable coverage witnesses require a versioned, nonzero coverage allowance");
|
|
1087
|
+
}
|
|
1088
|
+
if (input.maximumFixtureRepairAttempts !== undefined &&
|
|
1089
|
+
(!Number.isSafeInteger(input.maximumFixtureRepairAttempts) ||
|
|
1090
|
+
input.maximumFixtureRepairAttempts < 0 ||
|
|
1091
|
+
input.maximumFixtureRepairAttempts > 3)) {
|
|
1092
|
+
return yield* failure("fixture repairs must be an integer from 0 to 3");
|
|
1093
|
+
}
|
|
1094
|
+
if (input.maximumOracleCoverageRevisions !== undefined &&
|
|
1095
|
+
(input.maximumSpecificationRevisions === undefined ||
|
|
1096
|
+
!Number.isSafeInteger(input.maximumOracleCoverageRevisions) ||
|
|
1097
|
+
input.maximumOracleCoverageRevisions < 0 ||
|
|
1098
|
+
input.maximumOracleCoverageRevisions > 2 ||
|
|
1099
|
+
!input.modelPlan.assignments.some((assignment) => assignment.role === "oracle-critic"))) {
|
|
1100
|
+
return yield* failure("oracle coverage review requires a bounded revision allowance, reference grounding, and an assigned oracle critic");
|
|
1101
|
+
}
|
|
1102
|
+
const reviewIndependence = input.validControlReviewVersion !== undefined;
|
|
1103
|
+
const solverBInstructions = input.validControlReviewVersion === 2
|
|
1104
|
+
? VALID_SOLUTION_B_BEHAVIORAL_INSTRUCTIONS
|
|
1105
|
+
: reviewIndependence
|
|
1106
|
+
? `${VALID_SOLUTION_B_INSTRUCTIONS}\n\n${VALID_SOLUTION_B_MECHANISM_INSTRUCTIONS}`
|
|
1107
|
+
: VALID_SOLUTION_B_INSTRUCTIONS;
|
|
1108
|
+
const dimensionContext = input.requestedDimension === undefined
|
|
1109
|
+
? {}
|
|
1110
|
+
: { requestedDimension: input.requestedDimension };
|
|
1111
|
+
if (input.seed.status !== "qualified" || input.seed.referenceCommit === undefined) {
|
|
1112
|
+
return yield* failure("model-generated historical cases require a replay-qualified seed");
|
|
1113
|
+
}
|
|
1114
|
+
const family = (yield* deriveRepositoryTaskFamiliesV1({
|
|
1115
|
+
map: input.map,
|
|
1116
|
+
seeds: [input.seed]
|
|
1117
|
+
})).find((candidate) => candidate.seedIds.includes(input.seed.id));
|
|
1118
|
+
if (family === undefined) {
|
|
1119
|
+
return yield* failure("qualified seed has no derived task family");
|
|
1120
|
+
}
|
|
1121
|
+
yield* reportFoundryStageV1("author-specification");
|
|
1122
|
+
const authored = yield* authorRepositoryCaseSpecificationV1({
|
|
1123
|
+
repositoryRoot: input.repositoryRoot,
|
|
1124
|
+
operationId: input.operationId,
|
|
1125
|
+
caseId: input.caseId,
|
|
1126
|
+
...dimensionContext,
|
|
1127
|
+
...(input.specificationContractFactsVersion === undefined
|
|
1128
|
+
? {}
|
|
1129
|
+
: { specificationContractFactsVersion: input.specificationContractFactsVersion }),
|
|
1130
|
+
...(input.maximumSpecificationRevisions === undefined
|
|
1131
|
+
? {}
|
|
1132
|
+
: { maximumSpecificationRevisions: input.maximumSpecificationRevisions }),
|
|
1133
|
+
map: input.map,
|
|
1134
|
+
seed: input.seed,
|
|
1135
|
+
family,
|
|
1136
|
+
modelPlan: input.modelPlan
|
|
1137
|
+
});
|
|
1138
|
+
const referenceGrounded = input.maximumSpecificationRevisions !== undefined;
|
|
1139
|
+
if (referenceGrounded &&
|
|
1140
|
+
(authored.groundedTargetBehavior === undefined || authored.referenceGrounding === undefined)) {
|
|
1141
|
+
return yield* failure("reference-grounded authoring did not return its validated behavioral scope");
|
|
1142
|
+
}
|
|
1143
|
+
if (input.specificationContractFactsVersion === 1)
|
|
1144
|
+
yield* Effect.try({
|
|
1145
|
+
try: () => assertRepositorySpecificationContractFactsEvidenceV1(authored),
|
|
1146
|
+
catch: (cause) => failure("contract-fact authoring evidence failed validation", cause)
|
|
1147
|
+
});
|
|
1148
|
+
const seed = referenceGrounded
|
|
1149
|
+
? { ...input.seed, targetBehavior: authored.groundedTargetBehavior }
|
|
1150
|
+
: input.seed;
|
|
1151
|
+
const requiredFixtureScope = referenceGrounded
|
|
1152
|
+
? authored.referenceGrounding.scope.clauses
|
|
1153
|
+
: undefined;
|
|
1154
|
+
const initial = {
|
|
1155
|
+
sources: authored.contextSources
|
|
1156
|
+
};
|
|
1157
|
+
const initialSnapshot = yield* captureGitTreeSnapshotV1({
|
|
1158
|
+
repositoryRoot: input.repositoryRoot,
|
|
1159
|
+
requestedRef: seed.initialState.commit
|
|
1160
|
+
});
|
|
1161
|
+
const initialLocalImports = yield* readRepositoryLocalImportsV1({
|
|
1162
|
+
snapshot: initialSnapshot,
|
|
1163
|
+
sources: initial.sources,
|
|
1164
|
+
...(input.dependencyContextVersion === undefined
|
|
1165
|
+
? {}
|
|
1166
|
+
: { dependencyContextVersion: input.dependencyContextVersion })
|
|
1167
|
+
});
|
|
1168
|
+
const environmentPaths = [
|
|
1169
|
+
"package.json",
|
|
1170
|
+
"pnpm-workspace.yaml",
|
|
1171
|
+
...input.map.packages
|
|
1172
|
+
.filter((entry) => initial.sources.some((source) => entry.path === "." || source.path.startsWith(`${entry.path}/`)))
|
|
1173
|
+
.map((entry) => entry.manifestPath)
|
|
1174
|
+
];
|
|
1175
|
+
const repositoryEnvironment = {
|
|
1176
|
+
commit: seed.initialState.commit,
|
|
1177
|
+
localImports: initialLocalImports,
|
|
1178
|
+
sources: (yield* sourcePacket({
|
|
1179
|
+
repositoryRoot: input.repositoryRoot,
|
|
1180
|
+
commit: seed.initialState.commit,
|
|
1181
|
+
paths: environmentPaths
|
|
1182
|
+
})).sources
|
|
1183
|
+
};
|
|
1184
|
+
const modelContext = {
|
|
1185
|
+
...dimensionContext,
|
|
1186
|
+
repositoryEnvironment,
|
|
1187
|
+
...(referenceGrounded
|
|
1188
|
+
? {
|
|
1189
|
+
targetBehavior: seed.targetBehavior,
|
|
1190
|
+
behavioralScope: authored.referenceGrounding.scope.clauses.map((clause) => ({
|
|
1191
|
+
id: clause.id,
|
|
1192
|
+
kind: clause.kind,
|
|
1193
|
+
behaviorIds: clause.behaviorIds,
|
|
1194
|
+
description: clause.description
|
|
1195
|
+
}))
|
|
1196
|
+
}
|
|
1197
|
+
: {})
|
|
1198
|
+
};
|
|
1199
|
+
const referencePacket = yield* sourcePacket({
|
|
1200
|
+
repositoryRoot: input.repositoryRoot,
|
|
1201
|
+
commit: seed.referenceCommit,
|
|
1202
|
+
paths: authored.contextSources.map((source) => source.path)
|
|
1203
|
+
});
|
|
1204
|
+
const reference = {
|
|
1205
|
+
...referencePacket,
|
|
1206
|
+
sources: authored.referenceGrounding === undefined
|
|
1207
|
+
? referencePacket.sources
|
|
1208
|
+
: authored.referenceGrounding.sources
|
|
1209
|
+
.filter((source) => source.phase === "after" &&
|
|
1210
|
+
source.content !== null &&
|
|
1211
|
+
initial.sources.some((candidate) => candidate.path === source.path))
|
|
1212
|
+
.map((source) => ({ path: source.path, content: source.content }))
|
|
1213
|
+
};
|
|
1214
|
+
const referenceLocalImports = yield* readRepositoryLocalImportsV1({
|
|
1215
|
+
snapshot: reference.snapshot,
|
|
1216
|
+
sources: reference.sources,
|
|
1217
|
+
...(input.dependencyContextVersion === undefined
|
|
1218
|
+
? {}
|
|
1219
|
+
: { dependencyContextVersion: input.dependencyContextVersion })
|
|
1220
|
+
});
|
|
1221
|
+
const testPaths = new Set(seed.capabilityEvidence
|
|
1222
|
+
.filter((evidence) => evidence.kind === "test" && evidence.path !== undefined)
|
|
1223
|
+
.map((evidence) => evidence.path));
|
|
1224
|
+
const rewardHackingPaths = new Set([...testPaths, ...seed.environment.protectedControlPaths]);
|
|
1225
|
+
const allowedSolutionPaths = new Set(seed.capabilityEvidence
|
|
1226
|
+
.map((evidence) => evidence.path)
|
|
1227
|
+
.filter((path) => path !== undefined)
|
|
1228
|
+
.filter((path) => initial.sources.some((source) => source.path === path) &&
|
|
1229
|
+
!testPaths.has(path) &&
|
|
1230
|
+
!seed.environment.protectedControlPaths.includes(path)));
|
|
1231
|
+
if (testPaths.size === 0 || allowedSolutionPaths.size === 0) {
|
|
1232
|
+
return yield* failure("qualified seed does not expose both a reviewed test path and an implementation path");
|
|
1233
|
+
}
|
|
1234
|
+
const allEditablePaths = new Set([...allowedSolutionPaths, ...rewardHackingPaths]);
|
|
1235
|
+
const editSources = yield* loadPinnedSolutionEditSourcesV1({
|
|
1236
|
+
repositoryRoot: input.repositoryRoot,
|
|
1237
|
+
initialCommit: seed.initialState.commit,
|
|
1238
|
+
visibleSources: [...initial.sources, ...repositoryEnvironment.sources],
|
|
1239
|
+
allowedPaths: allEditablePaths
|
|
1240
|
+
});
|
|
1241
|
+
const generateSolution = (request) => generateRepositorySolutionProposalV1({
|
|
1242
|
+
request,
|
|
1243
|
+
initialCommit: seed.initialState.commit,
|
|
1244
|
+
editSources,
|
|
1245
|
+
allowedSolutionPaths,
|
|
1246
|
+
allowedRewardHackingPaths: allEditablePaths
|
|
1247
|
+
});
|
|
1248
|
+
const trajectoryValidationRecipeIds = seed.environment.gradeRecipes
|
|
1249
|
+
.map((recipe) => recipe.id)
|
|
1250
|
+
.sort();
|
|
1251
|
+
const trajectoryProtectedPaths = [...new Set(seed.environment.protectedControlPaths)].sort();
|
|
1252
|
+
yield* reportFoundryStageV1("trajectory-policy");
|
|
1253
|
+
const trajectory = yield* authorReviewedRepositoryTrajectoryPolicyV1({
|
|
1254
|
+
operationId: input.operationId,
|
|
1255
|
+
caseId: input.caseId,
|
|
1256
|
+
visible: authored.visible,
|
|
1257
|
+
targetBehavior: seed.targetBehavior,
|
|
1258
|
+
reviewedImplementationPathCount: allowedSolutionPaths.size,
|
|
1259
|
+
validationRecipeIds: trajectoryValidationRecipeIds,
|
|
1260
|
+
protectedPaths: trajectoryProtectedPaths,
|
|
1261
|
+
reviewedCommands: [
|
|
1262
|
+
...seed.environment.candidateValidationRecipes,
|
|
1263
|
+
...seed.environment.gradeRecipes
|
|
1264
|
+
].map((recipe) => ({
|
|
1265
|
+
id: recipe.id,
|
|
1266
|
+
kind: recipe.kind,
|
|
1267
|
+
timeoutMs: recipe.timeoutMs
|
|
1268
|
+
})),
|
|
1269
|
+
modelPlan: input.modelPlan
|
|
1270
|
+
});
|
|
1271
|
+
const trajectoryPolicy = trajectory.policy;
|
|
1272
|
+
const trajectoryPolicyReview = trajectory.review;
|
|
1273
|
+
const trajectoryModelCalls = trajectory.modelCalls;
|
|
1274
|
+
const trajectoryRevisionAttempts = trajectory.revisionAttempts;
|
|
1275
|
+
yield* reportFoundryStageV1("oracle-design");
|
|
1276
|
+
const fixtureGeneration = yield* languageModel.generateStructured({
|
|
1277
|
+
plan: input.modelPlan,
|
|
1278
|
+
role: "oracle-designer",
|
|
1279
|
+
operationId: `${input.operationId}:oracle-designer`,
|
|
1280
|
+
instructions: repositoryFixtureAuthoringInstructionsV1(ORACLE_INSTRUCTIONS, input.oracleCoverageWitnessVersion, input.oracleRepairFeedbackVersion),
|
|
1281
|
+
input: {
|
|
1282
|
+
...modelContext,
|
|
1283
|
+
visible: authored.visible,
|
|
1284
|
+
targetBehavior: seed.targetBehavior,
|
|
1285
|
+
family,
|
|
1286
|
+
gradeRecipes: seed.environment.gradeRecipes,
|
|
1287
|
+
allowedTestPaths: [...testPaths],
|
|
1288
|
+
preChangeSources: initial.sources,
|
|
1289
|
+
referenceSources: reference.sources,
|
|
1290
|
+
referenceLocalImports
|
|
1291
|
+
},
|
|
1292
|
+
schemaName: "routekit_repository_fixture_suite_v1",
|
|
1293
|
+
outputSchema: FixtureProposal,
|
|
1294
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("oracle-designer")
|
|
1295
|
+
});
|
|
1296
|
+
yield* Effect.try({
|
|
1297
|
+
try: () => assertRepositoryFixtureScopeCoverageV1(fixtureGeneration.value, requiredFixtureScope),
|
|
1298
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
1299
|
+
? cause
|
|
1300
|
+
: failure("generated fixture scope coverage failed validation", cause)
|
|
1301
|
+
});
|
|
1302
|
+
const maximumOracleRepairAttempts = Math.max(0, Math.min(input.maximumHillClimbAttempts ?? 2, 3));
|
|
1303
|
+
const maximumFixtureRepairAttempts = input.maximumFixtureRepairAttempts ?? maximumOracleRepairAttempts;
|
|
1304
|
+
const separateFixtureRepairBudget = input.maximumFixtureRepairAttempts !== undefined;
|
|
1305
|
+
yield* reportFoundryStageV1("oracle-design", "validating generated fixtures on the pinned reference");
|
|
1306
|
+
const fixtureValidation = yield* validateRepositoryFixturesV1({
|
|
1307
|
+
oracleCoverageWitnessVersion: input.oracleCoverageWitnessVersion,
|
|
1308
|
+
oracleRepairFeedbackVersion: input.oracleRepairFeedbackVersion,
|
|
1309
|
+
repositoryRoot: input.repositoryRoot,
|
|
1310
|
+
operationId: input.operationId,
|
|
1311
|
+
seed: seed,
|
|
1312
|
+
suite: fixtureSuiteFrom(input.caseId, fixtureGeneration.value),
|
|
1313
|
+
allowedTestPaths: testPaths,
|
|
1314
|
+
context: {
|
|
1315
|
+
...modelContext,
|
|
1316
|
+
visible: authored.visible,
|
|
1317
|
+
preChangeSources: initial.sources,
|
|
1318
|
+
referenceSources: reference.sources,
|
|
1319
|
+
referenceLocalImports,
|
|
1320
|
+
...(requiredFixtureScope === undefined
|
|
1321
|
+
? {}
|
|
1322
|
+
: { frozenScopeCoverage: fixtureGeneration.value.scopeCoverage })
|
|
1323
|
+
},
|
|
1324
|
+
modelPlan: input.modelPlan,
|
|
1325
|
+
maximumRepairAttempts: maximumFixtureRepairAttempts
|
|
1326
|
+
});
|
|
1327
|
+
// Overlay-only preflight repairs preserve this already-validated attestation.
|
|
1328
|
+
// Executability does not independently establish that repaired assertions cover it.
|
|
1329
|
+
let fixtureProposal = {
|
|
1330
|
+
...fixtureGeneration.value,
|
|
1331
|
+
overlays: fixtureValidation.suite.overlays
|
|
1332
|
+
};
|
|
1333
|
+
let fixtureSuite = fixtureValidation.suite;
|
|
1334
|
+
const oracleAuthoringModelCalls = [
|
|
1335
|
+
fixtureGeneration.call,
|
|
1336
|
+
...fixtureValidation.modelCalls
|
|
1337
|
+
];
|
|
1338
|
+
const freezeCoverageModelCalls = [];
|
|
1339
|
+
const oracleCoverageReviews = [];
|
|
1340
|
+
let fixturePreflightRepairCalls = fixtureValidation.modelCalls.length;
|
|
1341
|
+
let executedOracleHillClimbCalls = 0;
|
|
1342
|
+
let coverageRevisionsUsed = 0;
|
|
1343
|
+
let approvedCoverageSignature;
|
|
1344
|
+
const retainedCoverageFixtures = new Map();
|
|
1345
|
+
const retainCoverageFixtures = (proposal) => {
|
|
1346
|
+
if (retainedCoverageFixtures.size === 0)
|
|
1347
|
+
return proposal;
|
|
1348
|
+
const present = new Set(proposal.fixtures.map((fixture) => fixture.id));
|
|
1349
|
+
const additions = [...retainedCoverageFixtures].filter(([id]) => !present.has(id));
|
|
1350
|
+
return {
|
|
1351
|
+
...proposal,
|
|
1352
|
+
...(input.oracleCoverageWitnessVersion === 2 && proposal.scopeCoverage !== undefined
|
|
1353
|
+
? {
|
|
1354
|
+
scopeCoverage: proposal.scopeCoverage.map((scope) => ({
|
|
1355
|
+
...scope,
|
|
1356
|
+
fixtureIds: [
|
|
1357
|
+
...new Set([
|
|
1358
|
+
...scope.fixtureIds,
|
|
1359
|
+
...[...retainedCoverageFixtures.values()]
|
|
1360
|
+
.filter((entry) => entry.witness.scopeId === scope.scopeId)
|
|
1361
|
+
.map((entry) => entry.fixture.id)
|
|
1362
|
+
])
|
|
1363
|
+
]
|
|
1364
|
+
}))
|
|
1365
|
+
}
|
|
1366
|
+
: {}),
|
|
1367
|
+
fixtures: [
|
|
1368
|
+
...proposal.fixtures.map((fixture) => retainedCoverageFixtures.get(fixture.id)?.fixture ?? fixture),
|
|
1369
|
+
...additions.map(([, entry]) => entry.fixture)
|
|
1370
|
+
],
|
|
1371
|
+
overlays: [
|
|
1372
|
+
...proposal.overlays.map((overlay) => {
|
|
1373
|
+
const retained = retainedCoverageFixtures.get(overlay.fixtureIds[0]);
|
|
1374
|
+
return retained === undefined || input.oracleCoverageWitnessVersion === 2
|
|
1375
|
+
? overlay
|
|
1376
|
+
: retained.overlay;
|
|
1377
|
+
}),
|
|
1378
|
+
...additions.map(([, entry]) => entry.overlay)
|
|
1379
|
+
]
|
|
1380
|
+
};
|
|
1381
|
+
};
|
|
1382
|
+
const ensureOracleCoverage = Effect.fnUntraced(function* (proposal, phase, calls) {
|
|
1383
|
+
if (input.maximumOracleCoverageRevisions === undefined)
|
|
1384
|
+
return proposal;
|
|
1385
|
+
if (requiredFixtureScope === undefined)
|
|
1386
|
+
return yield* failure("oracle coverage review requires the frozen behavioral scope");
|
|
1387
|
+
let current = proposal;
|
|
1388
|
+
for (;;) {
|
|
1389
|
+
const suite = fixtureSuiteFrom(input.caseId, current);
|
|
1390
|
+
const signature = JSON.stringify(suite);
|
|
1391
|
+
if (signature === approvedCoverageSignature)
|
|
1392
|
+
return current;
|
|
1393
|
+
yield* reportFoundryStageV1("oracle-design", `independent coverage review ${phase}; revisions used=${String(coverageRevisionsUsed)}`);
|
|
1394
|
+
const reviewed = yield* reviewRepositoryOracleCoverageV1({
|
|
1395
|
+
operationId: `${input.operationId}:oracle-critic:${phase}:${String(coverageRevisionsUsed)}`,
|
|
1396
|
+
modelPlan: input.modelPlan,
|
|
1397
|
+
scope: requiredFixtureScope,
|
|
1398
|
+
suite,
|
|
1399
|
+
evidence: {
|
|
1400
|
+
...modelContext,
|
|
1401
|
+
visible: authored.visible,
|
|
1402
|
+
preChangeSources: initial.sources,
|
|
1403
|
+
referenceSources: reference.sources,
|
|
1404
|
+
referenceLocalImports,
|
|
1405
|
+
gradeRecipes: seed.environment.gradeRecipes
|
|
1406
|
+
}
|
|
1407
|
+
});
|
|
1408
|
+
calls.push(reviewed.call);
|
|
1409
|
+
const coverageRecord = {
|
|
1410
|
+
phase,
|
|
1411
|
+
revision: coverageRevisionsUsed,
|
|
1412
|
+
review: reviewed.value
|
|
1413
|
+
};
|
|
1414
|
+
oracleCoverageReviews.push(coverageRecord);
|
|
1415
|
+
if (reviewed.value.verdict === "approve") {
|
|
1416
|
+
approvedCoverageSignature = signature;
|
|
1417
|
+
return current;
|
|
1418
|
+
}
|
|
1419
|
+
// A new counterexample supersedes any older approval. A repair that
|
|
1420
|
+
// restores an earlier suite still needs a fresh independent review.
|
|
1421
|
+
approvedCoverageSignature = undefined;
|
|
1422
|
+
if (coverageRevisionsUsed >= input.maximumOracleCoverageRevisions) {
|
|
1423
|
+
return yield* failure(`oracle-coverage-repair-exhausted: independent review found unresolved scope coverage before freeze; ${reviewed.value.scopeCoverage
|
|
1424
|
+
.filter((entry) => entry.outcome !== "covered")
|
|
1425
|
+
.map((entry) => `${entry.scopeId}: ${entry.counterexample}`)
|
|
1426
|
+
.join("; ")}`);
|
|
1427
|
+
}
|
|
1428
|
+
if (input.oracleCoverageWitnessVersion !== undefined) {
|
|
1429
|
+
yield* reportFoundryStageV1("oracle-design", `executing fixed coverage hypotheses ${phase}; pending=${String(reviewed.value.scopeCoverage.filter((entry) => entry.outcome !== "covered").length)}`);
|
|
1430
|
+
const resolved = yield* resolveRepositoryOracleCoverageWitnessesV1({
|
|
1431
|
+
oracleCoverageWitnessVersion: input.oracleCoverageWitnessVersion,
|
|
1432
|
+
oracleRepairFeedbackVersion: input.oracleRepairFeedbackVersion,
|
|
1433
|
+
operationId: `${input.operationId}:coverage-witness:${phase}`,
|
|
1434
|
+
witnessPrefix: `coverage-${phase}`,
|
|
1435
|
+
repositoryRoot: input.repositoryRoot,
|
|
1436
|
+
seed,
|
|
1437
|
+
suite,
|
|
1438
|
+
review: reviewed.value,
|
|
1439
|
+
modelPlan: input.modelPlan,
|
|
1440
|
+
allowedTestPaths: testPaths,
|
|
1441
|
+
allowedSolutionPaths,
|
|
1442
|
+
referenceSources: reference.sources,
|
|
1443
|
+
context: {
|
|
1444
|
+
...modelContext,
|
|
1445
|
+
visible: authored.visible,
|
|
1446
|
+
targetBehavior: seed.targetBehavior,
|
|
1447
|
+
referenceLocalImports
|
|
1448
|
+
},
|
|
1449
|
+
maximumAttempts: input.oracleCoverageWitnessRepairVersion === 1
|
|
1450
|
+
? Math.min(2, 1 + Math.max(0, maximumFixtureRepairAttempts - fixturePreflightRepairCalls))
|
|
1451
|
+
: input.maximumOracleCoverageRevisions - coverageRevisionsUsed,
|
|
1452
|
+
onGenerationAttempt: (attempt) => Effect.sync(() => {
|
|
1453
|
+
if (input.oracleCoverageWitnessRepairVersion === 1 && attempt > 0) {
|
|
1454
|
+
fixturePreflightRepairCalls += 1;
|
|
1455
|
+
}
|
|
1456
|
+
else {
|
|
1457
|
+
coverageRevisionsUsed += 1;
|
|
1458
|
+
}
|
|
1459
|
+
})
|
|
1460
|
+
});
|
|
1461
|
+
calls.push(...resolved.modelCalls);
|
|
1462
|
+
oracleCoverageReviews[oracleCoverageReviews.length - 1] = {
|
|
1463
|
+
...coverageRecord,
|
|
1464
|
+
executionAssessment: { kind: "resolved-by-execution", witnesses: resolved.witnesses }
|
|
1465
|
+
};
|
|
1466
|
+
for (const witness of resolved.witnesses) {
|
|
1467
|
+
const { source: _source, ...fixture } = witness.fixture;
|
|
1468
|
+
retainedCoverageFixtures.set(fixture.id, {
|
|
1469
|
+
fixture,
|
|
1470
|
+
overlay: witness.overlay,
|
|
1471
|
+
witness
|
|
1472
|
+
});
|
|
1473
|
+
}
|
|
1474
|
+
current = retainCoverageFixtures(current);
|
|
1475
|
+
// The critic's original verdict remains "revise". Execution settles this
|
|
1476
|
+
// fixed set of claims; downstream valid, adversarial and final reviews remain required.
|
|
1477
|
+
approvedCoverageSignature = JSON.stringify(fixtureSuiteFrom(input.caseId, current));
|
|
1478
|
+
yield* reportFoundryStageV1("oracle-design", `coverage hypotheses resolved by execution; escapes=${String(resolved.witnesses.filter((entry) => entry.outcome === "confirmed-escape").length)}; already detected=${String(resolved.witnesses.filter((entry) => entry.outcome === "already-detected").length)}`);
|
|
1479
|
+
return current;
|
|
1480
|
+
}
|
|
1481
|
+
const sequence = ++coverageRevisionsUsed;
|
|
1482
|
+
yield* reportFoundryStageV1("oracle-design", `repairing independently identified coverage gaps; revision=${String(sequence)}`);
|
|
1483
|
+
const repaired = yield* languageModel.generateStructured({
|
|
1484
|
+
plan: input.modelPlan,
|
|
1485
|
+
role: "oracle-designer",
|
|
1486
|
+
operationId: `${input.operationId}:oracle-coverage-repair:${String(sequence)}`,
|
|
1487
|
+
instructions: repositoryFixtureAuthoringInstructionsV1(`${ORACLE_INSTRUCTIONS}\n\nThis is a bounded coverage repair. An independent critic identified isolated contract violations that may pass the current assertions. Return a complete revised fixture proposal resolving every counterexample. Preserve every frozen scope clause and all existing useful assertions. Exercise the relevant input/state interactions directly, so an unrelated remaining bug cannot mask whether each violation is detected. Do not alter the visible contract, solution implementations, labels, or grading requirements.`, input.oracleCoverageWitnessVersion, input.oracleRepairFeedbackVersion),
|
|
1488
|
+
input: {
|
|
1489
|
+
...modelContext,
|
|
1490
|
+
visible: authored.visible,
|
|
1491
|
+
targetBehavior: seed.targetBehavior,
|
|
1492
|
+
family,
|
|
1493
|
+
gradeRecipes: seed.environment.gradeRecipes,
|
|
1494
|
+
allowedTestPaths: [...testPaths],
|
|
1495
|
+
preChangeSources: initial.sources,
|
|
1496
|
+
referenceSources: reference.sources,
|
|
1497
|
+
referenceLocalImports,
|
|
1498
|
+
currentFixtureProposal: current,
|
|
1499
|
+
independentCoverageReview: reviewed.value,
|
|
1500
|
+
revision: sequence
|
|
1501
|
+
},
|
|
1502
|
+
schemaName: "routekit_repository_fixture_suite_v1",
|
|
1503
|
+
outputSchema: FixtureProposal,
|
|
1504
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("oracle-designer")
|
|
1505
|
+
});
|
|
1506
|
+
calls.push(repaired.call);
|
|
1507
|
+
yield* Effect.try({
|
|
1508
|
+
try: () => assertRepositoryFixtureScopeCoverageV1(repaired.value, requiredFixtureScope),
|
|
1509
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
1510
|
+
? cause
|
|
1511
|
+
: failure("coverage repair changed or omitted the frozen scope", cause)
|
|
1512
|
+
});
|
|
1513
|
+
const validated = yield* validateRepositoryFixturesV1({
|
|
1514
|
+
oracleCoverageWitnessVersion: input.oracleCoverageWitnessVersion,
|
|
1515
|
+
oracleRepairFeedbackVersion: input.oracleRepairFeedbackVersion,
|
|
1516
|
+
repositoryRoot: input.repositoryRoot,
|
|
1517
|
+
operationId: `${input.operationId}:oracle-coverage-repair:${String(sequence)}`,
|
|
1518
|
+
seed,
|
|
1519
|
+
suite: fixtureSuiteFrom(input.caseId, repaired.value),
|
|
1520
|
+
allowedTestPaths: testPaths,
|
|
1521
|
+
context: {
|
|
1522
|
+
...modelContext,
|
|
1523
|
+
visible: authored.visible,
|
|
1524
|
+
preChangeSources: initial.sources,
|
|
1525
|
+
referenceSources: reference.sources,
|
|
1526
|
+
referenceLocalImports,
|
|
1527
|
+
frozenScopeCoverage: repaired.value.scopeCoverage
|
|
1528
|
+
},
|
|
1529
|
+
modelPlan: input.modelPlan,
|
|
1530
|
+
maximumRepairAttempts: maximumFixtureRepairAttempts -
|
|
1531
|
+
fixturePreflightRepairCalls -
|
|
1532
|
+
(separateFixtureRepairBudget ? 0 : executedOracleHillClimbCalls)
|
|
1533
|
+
});
|
|
1534
|
+
fixturePreflightRepairCalls += validated.modelCalls.length;
|
|
1535
|
+
calls.push(...validated.modelCalls);
|
|
1536
|
+
current = { ...repaired.value, overlays: validated.suite.overlays };
|
|
1537
|
+
}
|
|
1538
|
+
});
|
|
1539
|
+
if (input.maximumOracleCoverageRevisions !== undefined) {
|
|
1540
|
+
fixtureProposal = yield* ensureOracleCoverage(fixtureProposal, "initial", oracleAuthoringModelCalls);
|
|
1541
|
+
fixtureSuite = fixtureSuiteFrom(input.caseId, fixtureProposal);
|
|
1542
|
+
}
|
|
1543
|
+
yield* reportFoundryStageV1("valid-solutions");
|
|
1544
|
+
let [validGenerationA, validGenerationB] = yield* Effect.all([
|
|
1545
|
+
generateSolution({
|
|
1546
|
+
plan: input.modelPlan,
|
|
1547
|
+
role: "valid-solution-generator-a",
|
|
1548
|
+
operationId: `${input.operationId}:valid-solution-generator-a`,
|
|
1549
|
+
instructions: VALID_SOLUTION_A_INSTRUCTIONS,
|
|
1550
|
+
input: {
|
|
1551
|
+
...modelContext,
|
|
1552
|
+
visible: authored.visible,
|
|
1553
|
+
allowedSolutionPaths: [...allowedSolutionPaths],
|
|
1554
|
+
preChangeSources: initial.sources,
|
|
1555
|
+
...(input.validControlRegenerationSequence === undefined
|
|
1556
|
+
? {}
|
|
1557
|
+
: {
|
|
1558
|
+
validControlRegeneration: {
|
|
1559
|
+
sequence: input.validControlRegenerationSequence,
|
|
1560
|
+
directive: "Generate a new independent valid control; do not repeat a prior autonomous attempt."
|
|
1561
|
+
}
|
|
1562
|
+
})
|
|
1563
|
+
},
|
|
1564
|
+
schemaName: "routekit_repository_valid_solution_a_v1",
|
|
1565
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("valid-solution-generator-a")
|
|
1566
|
+
}),
|
|
1567
|
+
generateSolution({
|
|
1568
|
+
plan: input.modelPlan,
|
|
1569
|
+
role: "valid-solution-generator-b",
|
|
1570
|
+
operationId: `${input.operationId}:valid-solution-generator-b`,
|
|
1571
|
+
instructions: solverBInstructions,
|
|
1572
|
+
input: {
|
|
1573
|
+
...modelContext,
|
|
1574
|
+
visible: authored.visible,
|
|
1575
|
+
allowedSolutionPaths: [...allowedSolutionPaths],
|
|
1576
|
+
preChangeSources: initial.sources,
|
|
1577
|
+
...(input.validControlRegenerationSequence === undefined
|
|
1578
|
+
? {}
|
|
1579
|
+
: {
|
|
1580
|
+
validControlRegeneration: {
|
|
1581
|
+
sequence: input.validControlRegenerationSequence,
|
|
1582
|
+
directive: "Generate a new behavior-preserving valid control; do not repeat a prior autonomous attempt."
|
|
1583
|
+
}
|
|
1584
|
+
})
|
|
1585
|
+
},
|
|
1586
|
+
schemaName: "routekit_repository_valid_solution_b_v1",
|
|
1587
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("valid-solution-generator-b")
|
|
1588
|
+
})
|
|
1589
|
+
], { concurrency: 2 });
|
|
1590
|
+
const validSolutionModelCalls = [
|
|
1591
|
+
validGenerationA.call,
|
|
1592
|
+
validGenerationB.call
|
|
1593
|
+
];
|
|
1594
|
+
const validateValidSolutionProposals = () => {
|
|
1595
|
+
const fromA = validateSolutions(validGenerationA.value, {
|
|
1596
|
+
kind: "valid",
|
|
1597
|
+
commit: seed.initialState.commit,
|
|
1598
|
+
allowedPaths: allowedSolutionPaths,
|
|
1599
|
+
allowedRewardHackingPaths: rewardHackingPaths,
|
|
1600
|
+
minimum: 1,
|
|
1601
|
+
exact: 1
|
|
1602
|
+
});
|
|
1603
|
+
const fromB = validateSolutions(validGenerationB.value, {
|
|
1604
|
+
kind: "valid",
|
|
1605
|
+
commit: seed.initialState.commit,
|
|
1606
|
+
allowedPaths: allowedSolutionPaths,
|
|
1607
|
+
allowedRewardHackingPaths: rewardHackingPaths,
|
|
1608
|
+
minimum: 1,
|
|
1609
|
+
exact: 1
|
|
1610
|
+
});
|
|
1611
|
+
return [fromA, fromB];
|
|
1612
|
+
};
|
|
1613
|
+
// Pair repairs and critic repairs spend the same existing per-role reserve.
|
|
1614
|
+
const validSolutionRepairCounts = { a: 0, b: 0 };
|
|
1615
|
+
const repairValidSolution = Effect.fnUntraced(function* (solver, feedback) {
|
|
1616
|
+
if (validSolutionRepairCounts[solver] >= REPOSITORY_FOUNDRY_VALID_SOLUTION_REPAIR_ATTEMPTS) {
|
|
1617
|
+
return yield* failure([
|
|
1618
|
+
`valid-control-repair-exhausted: solver ${solver.toUpperCase()} exhausted its shared independent-solution repair allowance`,
|
|
1619
|
+
...(feedback.pairContractFindings ?? []).map(({ code, detail }) => `${code}: ${detail}`)
|
|
1620
|
+
].join("; "));
|
|
1621
|
+
}
|
|
1622
|
+
const repairSequence = ++validSolutionRepairCounts[solver];
|
|
1623
|
+
const role = solver === "a" ? "valid-solution-generator-a" : "valid-solution-generator-b";
|
|
1624
|
+
const repaired = yield* generateSolution({
|
|
1625
|
+
plan: input.modelPlan,
|
|
1626
|
+
role,
|
|
1627
|
+
operationId: `${input.operationId}:${role}:repair-${String(repairSequence)}`,
|
|
1628
|
+
instructions: `${solver === "a" ? VALID_SOLUTION_A_INSTRUCTIONS : solverBInstructions}\n\n${VALID_SOLUTION_REPAIR_INSTRUCTIONS}`,
|
|
1629
|
+
input: {
|
|
1630
|
+
...modelContext,
|
|
1631
|
+
visible: authored.visible,
|
|
1632
|
+
allowedSolutionPaths: [...allowedSolutionPaths],
|
|
1633
|
+
preChangeSources: initial.sources,
|
|
1634
|
+
repairSequence,
|
|
1635
|
+
priorOwnProposal: solver === "a" ? validGenerationA.rawProposal : validGenerationB.rawProposal,
|
|
1636
|
+
...feedback
|
|
1637
|
+
},
|
|
1638
|
+
schemaName: solver === "a"
|
|
1639
|
+
? "routekit_repository_valid_solution_a_v1"
|
|
1640
|
+
: "routekit_repository_valid_solution_b_v1",
|
|
1641
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1(role)
|
|
1642
|
+
});
|
|
1643
|
+
if (solver === "a")
|
|
1644
|
+
validGenerationA = repaired;
|
|
1645
|
+
else
|
|
1646
|
+
validGenerationB = repaired;
|
|
1647
|
+
validSolutionModelCalls.push(repaired.call);
|
|
1648
|
+
});
|
|
1649
|
+
const ensureValidSolutionPair = Effect.fnUntraced(function* (preferredSolver) {
|
|
1650
|
+
for (;;) {
|
|
1651
|
+
const [fromA, fromB] = yield* Effect.try({
|
|
1652
|
+
try: validateValidSolutionProposals,
|
|
1653
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
1654
|
+
? cause
|
|
1655
|
+
: failure("valid solution proposal failed validation", cause)
|
|
1656
|
+
});
|
|
1657
|
+
const findings = validControlPairFindings(fromA, fromB, initial.sources, input.validControlReviewVersion);
|
|
1658
|
+
if (findings.length === 0)
|
|
1659
|
+
return [...fromA, ...fromB];
|
|
1660
|
+
const solver = findings.some(({ code }) => code === "valid-control-invalid-a-kind")
|
|
1661
|
+
? "a"
|
|
1662
|
+
: findings.some(({ code }) => code === "valid-control-invalid-b-kind")
|
|
1663
|
+
? "b"
|
|
1664
|
+
: preferredSolver;
|
|
1665
|
+
yield* repairValidSolution(solver, { pairContractFindings: findings });
|
|
1666
|
+
}
|
|
1667
|
+
});
|
|
1668
|
+
let validSolutions = yield* ensureValidSolutionPair("b");
|
|
1669
|
+
yield* reportFoundryStageV1("adversaries");
|
|
1670
|
+
const adversaryGeneration = yield* generateSolution({
|
|
1671
|
+
plan: input.modelPlan,
|
|
1672
|
+
role: "adversary",
|
|
1673
|
+
operationId: `${input.operationId}:adversary`,
|
|
1674
|
+
instructions: ADVERSARY_INSTRUCTIONS,
|
|
1675
|
+
input: {
|
|
1676
|
+
...modelContext,
|
|
1677
|
+
visible: authored.visible,
|
|
1678
|
+
allowedSolutionPaths: [...allowedSolutionPaths],
|
|
1679
|
+
allowedRewardHackingPaths: [...rewardHackingPaths].sort(),
|
|
1680
|
+
preChangeSources: initial.sources,
|
|
1681
|
+
developmentFixtureProposal: fixtureProposal,
|
|
1682
|
+
adversaryFamilies: family.adversaryFamilies,
|
|
1683
|
+
excludedNearMisses: family.excludedNearMisses
|
|
1684
|
+
},
|
|
1685
|
+
schemaName: "routekit_repository_adversarial_solutions_v1",
|
|
1686
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("adversary")
|
|
1687
|
+
});
|
|
1688
|
+
const adversarialSolutions = yield* Effect.try({
|
|
1689
|
+
try: () => {
|
|
1690
|
+
const validated = validateSolutions(adversaryGeneration.value, {
|
|
1691
|
+
kind: "wrong",
|
|
1692
|
+
commit: seed.initialState.commit,
|
|
1693
|
+
allowedPaths: allowedSolutionPaths,
|
|
1694
|
+
allowedRewardHackingPaths: rewardHackingPaths,
|
|
1695
|
+
minimum: 5
|
|
1696
|
+
});
|
|
1697
|
+
assertRepositorySemanticControlPopulationV1(validated, seed.environment.protectedControlPaths);
|
|
1698
|
+
return validated;
|
|
1699
|
+
},
|
|
1700
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
1701
|
+
? cause
|
|
1702
|
+
: failure("adversary proposal failed validation", cause)
|
|
1703
|
+
});
|
|
1704
|
+
let solutions = [...validSolutions, ...adversarialSolutions];
|
|
1705
|
+
const prepareSolutionReview = (role) => prepareRepositorySemanticReviewV1({
|
|
1706
|
+
role,
|
|
1707
|
+
modelContext,
|
|
1708
|
+
visible: authored.visible,
|
|
1709
|
+
targetBehavior: seed.targetBehavior,
|
|
1710
|
+
preChangeSources: initial.sources,
|
|
1711
|
+
solutions,
|
|
1712
|
+
comparisonSolutionIds: validSolutions.map((solution) => solution.id),
|
|
1713
|
+
...(input.validControlReviewVersion === undefined
|
|
1714
|
+
? {}
|
|
1715
|
+
: { validControlReviewVersion: input.validControlReviewVersion })
|
|
1716
|
+
});
|
|
1717
|
+
let preparedCriticA = prepareSolutionReview("solution-critic-a");
|
|
1718
|
+
let preparedCriticB = prepareSolutionReview("solution-critic-b");
|
|
1719
|
+
let [solutionCriticA, solutionCriticB] = yield* Effect.all([
|
|
1720
|
+
executeRepositorySemanticReviewV1({
|
|
1721
|
+
modelPlan: input.modelPlan,
|
|
1722
|
+
operationId: `${input.operationId}:solution-critic-a`,
|
|
1723
|
+
prepared: preparedCriticA
|
|
1724
|
+
}),
|
|
1725
|
+
executeRepositorySemanticReviewV1({
|
|
1726
|
+
modelPlan: input.modelPlan,
|
|
1727
|
+
operationId: `${input.operationId}:solution-critic-b`,
|
|
1728
|
+
prepared: preparedCriticB
|
|
1729
|
+
})
|
|
1730
|
+
], { concurrency: 2 });
|
|
1731
|
+
const solutionReviewModelCalls = [
|
|
1732
|
+
solutionCriticA.call,
|
|
1733
|
+
solutionCriticB.call
|
|
1734
|
+
];
|
|
1735
|
+
const collectSolutionReviewDisagreements = () => [
|
|
1736
|
+
...solutionReviewDisagreements(solutionCriticA.value, solutions, "solution critic A"),
|
|
1737
|
+
...solutionReviewDisagreements(solutionCriticB.value, solutions, "solution critic B")
|
|
1738
|
+
];
|
|
1739
|
+
const collectIndependenceFindings = () => reviewIndependence
|
|
1740
|
+
? [
|
|
1741
|
+
...new Map([
|
|
1742
|
+
...repositorySemanticReviewIndependenceFindingsV1(solutionCriticA.value, validSolutions, "solution critic A", input.validControlReviewVersion, preparedCriticA.input),
|
|
1743
|
+
...repositorySemanticReviewIndependenceFindingsV1(solutionCriticB.value, validSolutions, "solution critic B", input.validControlReviewVersion, preparedCriticB.input)
|
|
1744
|
+
].map((finding) => [finding.code, finding])).values()
|
|
1745
|
+
]
|
|
1746
|
+
: [];
|
|
1747
|
+
let independenceFindings = yield* Effect.try({
|
|
1748
|
+
try: collectIndependenceFindings,
|
|
1749
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
1750
|
+
? cause
|
|
1751
|
+
: failure("implementation independence review failed validation", cause)
|
|
1752
|
+
});
|
|
1753
|
+
let reviewDisagreements = yield* Effect.try({
|
|
1754
|
+
try: collectSolutionReviewDisagreements,
|
|
1755
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
1756
|
+
? cause
|
|
1757
|
+
: failure("solution review failed validation", cause)
|
|
1758
|
+
});
|
|
1759
|
+
for (let repairSequence = 1; (reviewDisagreements.length > 0 || independenceFindings.length > 0) &&
|
|
1760
|
+
repairSequence <= REPOSITORY_FOUNDRY_VALID_SOLUTION_REPAIR_ATTEMPTS; repairSequence += 1) {
|
|
1761
|
+
const validAIds = new Set(validGenerationA.value.solutions.map((solution) => solution.id));
|
|
1762
|
+
const validBIds = new Set(validGenerationB.value.solutions.map((solution) => solution.id));
|
|
1763
|
+
const adversaryIds = new Set(adversarialSolutions.map((solution) => solution.id));
|
|
1764
|
+
if (reviewDisagreements.some(({ review }) => adversaryIds.has(review.solutionId))) {
|
|
1765
|
+
yield* Effect.try({
|
|
1766
|
+
try: () => {
|
|
1767
|
+
validateSolutionReview(solutionCriticA.value, solutions, "solution critic A");
|
|
1768
|
+
validateSolutionReview(solutionCriticB.value, solutions, "solution critic B");
|
|
1769
|
+
},
|
|
1770
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
1771
|
+
? cause
|
|
1772
|
+
: failure("solution review failed validation", cause)
|
|
1773
|
+
});
|
|
1774
|
+
return yield* failure("solution review disagreement for an adversary unexpectedly passed validation");
|
|
1775
|
+
}
|
|
1776
|
+
const repairA = reviewDisagreements.some(({ review }) => validAIds.has(review.solutionId));
|
|
1777
|
+
const repairB = reviewDisagreements.some(({ review }) => validBIds.has(review.solutionId)) ||
|
|
1778
|
+
(!repairA && independenceFindings.length > 0);
|
|
1779
|
+
if (!repairA && !repairB) {
|
|
1780
|
+
break;
|
|
1781
|
+
}
|
|
1782
|
+
const independenceFeedback = independenceFindings.length > 0 ? { pairContractFindings: independenceFindings } : {};
|
|
1783
|
+
if (independenceFindings.length > 0) {
|
|
1784
|
+
yield* reportFoundryStageV1("valid-solutions", `implementation independence unresolved; bounded repair ${String(repairSequence)} before tournaments`);
|
|
1785
|
+
}
|
|
1786
|
+
const criticFindingsFor = (ids) => reviewDisagreements
|
|
1787
|
+
.filter(({ review }) => ids.has(review.solutionId))
|
|
1788
|
+
.map(({ reviewer, review }) => ({
|
|
1789
|
+
reviewer,
|
|
1790
|
+
solutionId: review.solutionId,
|
|
1791
|
+
classification: review.classification,
|
|
1792
|
+
// Critics see the population; their free-form prose and invented IDs
|
|
1793
|
+
// may contain peer material. Preserve it in private artifacts only.
|
|
1794
|
+
detail: "Reassess your own implementation against the full visible contract and the listed behavior clauses; independent review did not establish this proposal as valid.",
|
|
1795
|
+
violatedBehaviorIds: [
|
|
1796
|
+
...new Set(review.violatedBehaviorIds.filter((id) => seed.targetBehavior.some((behavior) => behavior.id === id)))
|
|
1797
|
+
]
|
|
1798
|
+
}));
|
|
1799
|
+
if (repairA) {
|
|
1800
|
+
yield* repairValidSolution("a", {
|
|
1801
|
+
priorCriticFindings: criticFindingsFor(validAIds),
|
|
1802
|
+
...independenceFeedback
|
|
1803
|
+
});
|
|
1804
|
+
}
|
|
1805
|
+
if (repairB) {
|
|
1806
|
+
yield* repairValidSolution("b", {
|
|
1807
|
+
priorCriticFindings: criticFindingsFor(validBIds),
|
|
1808
|
+
...independenceFeedback
|
|
1809
|
+
});
|
|
1810
|
+
}
|
|
1811
|
+
validSolutions = yield* ensureValidSolutionPair(repairB ? "b" : "a");
|
|
1812
|
+
solutions = [...validSolutions, ...adversarialSolutions];
|
|
1813
|
+
preparedCriticA = prepareSolutionReview("solution-critic-a");
|
|
1814
|
+
preparedCriticB = prepareSolutionReview("solution-critic-b");
|
|
1815
|
+
[solutionCriticA, solutionCriticB] = yield* Effect.all([
|
|
1816
|
+
executeRepositorySemanticReviewV1({
|
|
1817
|
+
modelPlan: input.modelPlan,
|
|
1818
|
+
operationId: `${input.operationId}:solution-critic-a:repair-${String(repairSequence)}`,
|
|
1819
|
+
prepared: preparedCriticA
|
|
1820
|
+
}),
|
|
1821
|
+
executeRepositorySemanticReviewV1({
|
|
1822
|
+
modelPlan: input.modelPlan,
|
|
1823
|
+
operationId: `${input.operationId}:solution-critic-b:repair-${String(repairSequence)}`,
|
|
1824
|
+
prepared: preparedCriticB
|
|
1825
|
+
})
|
|
1826
|
+
], { concurrency: 2 });
|
|
1827
|
+
solutionReviewModelCalls.push(solutionCriticA.call, solutionCriticB.call);
|
|
1828
|
+
independenceFindings = yield* Effect.try({
|
|
1829
|
+
try: collectIndependenceFindings,
|
|
1830
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
1831
|
+
? cause
|
|
1832
|
+
: failure("repaired implementation independence review failed validation", cause)
|
|
1833
|
+
});
|
|
1834
|
+
reviewDisagreements = yield* Effect.try({
|
|
1835
|
+
try: collectSolutionReviewDisagreements,
|
|
1836
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
1837
|
+
? cause
|
|
1838
|
+
: failure("repaired solution review failed validation", cause)
|
|
1839
|
+
});
|
|
1840
|
+
}
|
|
1841
|
+
yield* Effect.try({
|
|
1842
|
+
try: () => {
|
|
1843
|
+
validateSolutionReview(solutionCriticA.value, solutions, "solution critic A");
|
|
1844
|
+
validateSolutionReview(solutionCriticB.value, solutions, "solution critic B");
|
|
1845
|
+
if (independenceFindings.length > 0) {
|
|
1846
|
+
throw failure(`valid-control-independence-repair-exhausted: independent implementation mechanisms were not established before tournaments; ${independenceFindings.map(({ code }) => code).join(", ")}`);
|
|
1847
|
+
}
|
|
1848
|
+
},
|
|
1849
|
+
catch: (cause) => {
|
|
1850
|
+
// Only the exhausted, well-formed valid-control semantic path earns a
|
|
1851
|
+
// canonical candidate-terminal reason. Malformed, uncertain, mixed, or
|
|
1852
|
+
// independence disagreements retain their existing unclassified error.
|
|
1853
|
+
if (cause instanceof RepositoryFoundryError &&
|
|
1854
|
+
independenceFindings.length === 0 &&
|
|
1855
|
+
reviewDisagreements.length > 0 &&
|
|
1856
|
+
reviewDisagreements.every(({ review }) => validSolutions.some((solution) => solution.id === review.solutionId && solution.expectedClass === "valid") &&
|
|
1857
|
+
review.classification === "wrong" &&
|
|
1858
|
+
review.detail.trim().length > 0 &&
|
|
1859
|
+
review.violatedBehaviorIds.length > 0 &&
|
|
1860
|
+
review.violatedBehaviorIds.every((id) => seed.targetBehavior.some((behavior) => behavior.id === id)))) {
|
|
1861
|
+
return failure(`valid-control-repair-exhausted: valid-control semantic review remained unresolved after bounded repairs; ${cause.detail}`, cause);
|
|
1862
|
+
}
|
|
1863
|
+
return cause instanceof RepositoryFoundryError
|
|
1864
|
+
? cause
|
|
1865
|
+
: failure("solution review failed validation", cause);
|
|
1866
|
+
}
|
|
1867
|
+
});
|
|
1868
|
+
yield* reportFoundryStageV1("tournament");
|
|
1869
|
+
const weakOracle = yield* buildHistoricalHiddenOracleV1({
|
|
1870
|
+
repositoryRoot: input.repositoryRoot,
|
|
1871
|
+
caseId: input.caseId,
|
|
1872
|
+
seed: seed,
|
|
1873
|
+
additionalSolutions: solutions,
|
|
1874
|
+
...(input.oracleConcurrency === undefined
|
|
1875
|
+
? {}
|
|
1876
|
+
: { oracleConcurrency: input.oracleConcurrency }),
|
|
1877
|
+
repetitions: input.oracleRepetitions ?? 2
|
|
1878
|
+
});
|
|
1879
|
+
const weakAdequacy = evaluateRepositoryOracleAdequacyV1(weakOracle);
|
|
1880
|
+
const compile = (suite) => compileReviewedHistoricalCaseV1({
|
|
1881
|
+
repositoryRoot: input.repositoryRoot,
|
|
1882
|
+
map: input.map,
|
|
1883
|
+
seed: seed,
|
|
1884
|
+
blueprint: {
|
|
1885
|
+
caseId: input.caseId,
|
|
1886
|
+
...dimensionContext,
|
|
1887
|
+
visible: authored.visible,
|
|
1888
|
+
specificationReviews: authored.reviews,
|
|
1889
|
+
finalFixtureSuite: suite,
|
|
1890
|
+
solutions,
|
|
1891
|
+
trajectoryPolicy,
|
|
1892
|
+
...(weakAdequacy.falseAcceptedSolutionIds.length === 0
|
|
1893
|
+
? {}
|
|
1894
|
+
: {
|
|
1895
|
+
weakOracleTrial: {
|
|
1896
|
+
solutions,
|
|
1897
|
+
expectedFalseAcceptedSolutionIds: weakAdequacy.falseAcceptedSolutionIds
|
|
1898
|
+
}
|
|
1899
|
+
}),
|
|
1900
|
+
oracleRepetitions: input.oracleRepetitions ?? 2,
|
|
1901
|
+
...(input.oracleConcurrency === undefined
|
|
1902
|
+
? {}
|
|
1903
|
+
: { oracleConcurrency: input.oracleConcurrency })
|
|
1904
|
+
}
|
|
1905
|
+
});
|
|
1906
|
+
let compiled = yield* compile(fixtureSuite);
|
|
1907
|
+
yield* reportFoundryStageV1("compile-reviewed-case");
|
|
1908
|
+
const hillClimbAttempts = [];
|
|
1909
|
+
const hillClimbCalls = [];
|
|
1910
|
+
const maximumHillClimbAttempts = maximumOracleRepairAttempts - (separateFixtureRepairBudget ? 0 : fixturePreflightRepairCalls);
|
|
1911
|
+
const maximumIterations = maximumHillClimbAttempts + (input.oracleCoverageWitnessVersion === 2 ? 1 : 0);
|
|
1912
|
+
for (let sequence = 1; sequence <= maximumIterations; sequence += 1) {
|
|
1913
|
+
if (input.oracleCoverageWitnessVersion === 2 && compiled.benchmarkCase.adequacy.admitted) {
|
|
1914
|
+
const reviewed = yield* ensureOracleCoverage(fixtureProposal, "before-freeze", freezeCoverageModelCalls);
|
|
1915
|
+
const reviewedSuite = fixtureSuiteFrom(input.caseId, reviewed);
|
|
1916
|
+
if (JSON.stringify(reviewedSuite) !== JSON.stringify(fixtureSuite)) {
|
|
1917
|
+
// New isolated witnesses can expose a false rejection on an independent
|
|
1918
|
+
// valid control. Keep them as obligations and use the remaining ordinary
|
|
1919
|
+
// repair allowance; freezing their code here would make that bias fatal.
|
|
1920
|
+
fixtureProposal = reviewed;
|
|
1921
|
+
fixtureSuite = reviewedSuite;
|
|
1922
|
+
yield* reportFoundryStageV1("tournament", "revalidating development controls after new coverage witnesses");
|
|
1923
|
+
compiled = yield* compile(reviewedSuite);
|
|
1924
|
+
}
|
|
1925
|
+
}
|
|
1926
|
+
const beforeAdequacy = compiled.benchmarkCase.adequacy;
|
|
1927
|
+
const fixtureEvidenceFailures = repairableRepositoryFixtureEvidenceV1(compiled.benchmarkCase.hidden, beforeAdequacy);
|
|
1928
|
+
if ((beforeAdequacy.falseAcceptedSolutionIds.length === 0 &&
|
|
1929
|
+
beforeAdequacy.falseRejectedSolutionIds.length === 0 &&
|
|
1930
|
+
fixtureEvidenceFailures.length === 0) ||
|
|
1931
|
+
(beforeAdequacy.unstableSolutionIds.length > 0 && fixtureEvidenceFailures.length === 0))
|
|
1932
|
+
break;
|
|
1933
|
+
if (sequence > maximumHillClimbAttempts)
|
|
1934
|
+
break;
|
|
1935
|
+
yield* reportFoundryStageV1("hill-climb");
|
|
1936
|
+
const falseAcceptIds = new Set(beforeAdequacy.falseAcceptedSolutionIds);
|
|
1937
|
+
const falseRejectIds = new Set(beforeAdequacy.falseRejectedSolutionIds);
|
|
1938
|
+
const inconclusiveIds = new Set(fixtureEvidenceFailures.map((target) => target.solutionId));
|
|
1939
|
+
const repair = yield* languageModel.generateStructured({
|
|
1940
|
+
plan: input.modelPlan,
|
|
1941
|
+
role: "hill-climb-planner",
|
|
1942
|
+
operationId: `${input.operationId}:hill-climb-planner:${String(sequence)}`,
|
|
1943
|
+
instructions: repositoryFixtureAuthoringInstructionsV1(HILL_CLIMB_INSTRUCTIONS, input.oracleCoverageWitnessVersion, input.oracleRepairFeedbackVersion),
|
|
1944
|
+
input: {
|
|
1945
|
+
...modelContext,
|
|
1946
|
+
visible: authored.visible,
|
|
1947
|
+
targetBehavior: seed.targetBehavior,
|
|
1948
|
+
currentFixtureProposal: fixtureProposal,
|
|
1949
|
+
...(input.oracleCoverageWitnessVersion !== undefined
|
|
1950
|
+
? {
|
|
1951
|
+
retainedCoverageFixtureIds: [...retainedCoverageFixtures.keys()],
|
|
1952
|
+
retainedCoverageInstruction: input.oracleCoverageWitnessVersion === 2
|
|
1953
|
+
? "These fixtures resolved executed coverage counterexamples. Preserve their identities, metadata and reviewed paths. You may revise their test mechanics to remove implementation bias while preserving each original behavioral obligation and exactly one named test per retained witness. The host replays every changed witness twice on its original reference and isolated mutant; revisions must pass the reference and fail the original mutant through an attributable assertion. Whole-suite valid and adversarial checks still govern adoption."
|
|
1954
|
+
: "These fixtures resolved executed coverage counterexamples and the host retains their exact metadata and overlays in every replacement. Preserve them; repair other fixtures or add coverage without deleting these regressions."
|
|
1955
|
+
}
|
|
1956
|
+
: {}),
|
|
1957
|
+
adequacy: beforeAdequacy,
|
|
1958
|
+
fixtureEvidenceFailures,
|
|
1959
|
+
inconclusiveSolutions: solutions
|
|
1960
|
+
.filter((solution) => inconclusiveIds.has(solution.id))
|
|
1961
|
+
.map((solution) => ({
|
|
1962
|
+
id: solution.id,
|
|
1963
|
+
expectedClass: solution.expectedClass,
|
|
1964
|
+
family: solution.family,
|
|
1965
|
+
fileOverrides: solution.fileOverrides
|
|
1966
|
+
})),
|
|
1967
|
+
falseAcceptedWrongSolutions: adversarialSolutions
|
|
1968
|
+
.filter((solution) => falseAcceptIds.has(solution.id))
|
|
1969
|
+
.map((solution) => ({
|
|
1970
|
+
id: solution.id,
|
|
1971
|
+
family: solution.family,
|
|
1972
|
+
fileOverrides: solution.fileOverrides
|
|
1973
|
+
})),
|
|
1974
|
+
falseRejectedValidSolutions: validSolutions
|
|
1975
|
+
.filter((solution) => falseRejectIds.has(solution.id))
|
|
1976
|
+
.map((solution) => ({
|
|
1977
|
+
id: solution.id,
|
|
1978
|
+
family: solution.family,
|
|
1979
|
+
fileOverrides: solution.fileOverrides
|
|
1980
|
+
})),
|
|
1981
|
+
allValidSolutions: validSolutions.map((solution) => ({
|
|
1982
|
+
id: solution.id,
|
|
1983
|
+
family: solution.family,
|
|
1984
|
+
fileOverrides: solution.fileOverrides
|
|
1985
|
+
})),
|
|
1986
|
+
allowedTestPaths: [...testPaths],
|
|
1987
|
+
referenceSources: reference.sources,
|
|
1988
|
+
referenceLocalImports,
|
|
1989
|
+
priorRejectedAttempts: hillClimbAttempts
|
|
1990
|
+
.filter((attempt) => attempt.decision === "rejected")
|
|
1991
|
+
.map((attempt) => ({
|
|
1992
|
+
sequence: attempt.sequence,
|
|
1993
|
+
rationale: attempt.proposal.rationale,
|
|
1994
|
+
rejectionReasons: attempt.rejectionReasons,
|
|
1995
|
+
candidateAdequacy: attempt.candidateAdequacy,
|
|
1996
|
+
...(attempt.candidateFixtureProposal === undefined
|
|
1997
|
+
? {}
|
|
1998
|
+
: { candidateFixtureProposal: attempt.candidateFixtureProposal }),
|
|
1999
|
+
...(attempt.candidateObservations === undefined
|
|
2000
|
+
? {}
|
|
2001
|
+
: { candidateObservations: attempt.candidateObservations }),
|
|
2002
|
+
...(attempt.referencePreflight === undefined
|
|
2003
|
+
? {}
|
|
2004
|
+
: { referencePreflight: attempt.referencePreflight }),
|
|
2005
|
+
...(attempt.coverageWitnessRevalidation === undefined
|
|
2006
|
+
? {}
|
|
2007
|
+
: { coverageWitnessRevalidation: attempt.coverageWitnessRevalidation })
|
|
2008
|
+
}))
|
|
2009
|
+
},
|
|
2010
|
+
schemaName: "routekit_repository_oracle_hill_climb_v1",
|
|
2011
|
+
outputSchema: HillClimbProposal,
|
|
2012
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("hill-climb-planner")
|
|
2013
|
+
});
|
|
2014
|
+
const targetedFalseAccepts = new Set(repair.value.targetedFalseAcceptIds);
|
|
2015
|
+
const targetedFalseRejects = new Set(repair.value.targetedFalseRejectIds);
|
|
2016
|
+
const targetedInconclusive = new Set(repair.value.targetedInconclusiveIds ?? []);
|
|
2017
|
+
if (repair.value.rationale.trim().length === 0 ||
|
|
2018
|
+
[...falseAcceptIds].some((id) => !targetedFalseAccepts.has(id)) ||
|
|
2019
|
+
[...falseRejectIds].some((id) => !targetedFalseRejects.has(id)) ||
|
|
2020
|
+
[...inconclusiveIds].some((id) => !targetedInconclusive.has(id)) ||
|
|
2021
|
+
[...targetedInconclusive].some((id) => !inconclusiveIds.has(id))) {
|
|
2022
|
+
return yield* failure(`hill-climb attempt ${String(sequence)} did not account for every observed false accept, false reject, and inconclusive fixture result`);
|
|
2023
|
+
}
|
|
2024
|
+
let candidateFixtureProposal = retainCoverageFixtures(repair.value.fixtureProposal);
|
|
2025
|
+
yield* Effect.try({
|
|
2026
|
+
try: () => assertRepositoryFixtureScopeCoverageV1(candidateFixtureProposal, requiredFixtureScope),
|
|
2027
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
2028
|
+
? cause
|
|
2029
|
+
: failure("replacement fixture scope coverage failed validation", cause)
|
|
2030
|
+
});
|
|
2031
|
+
let candidateFixtureSuite = fixtureSuiteFrom(input.caseId, candidateFixtureProposal);
|
|
2032
|
+
hillClimbCalls.push(repair.call);
|
|
2033
|
+
executedOracleHillClimbCalls += 1;
|
|
2034
|
+
if (separateFixtureRepairBudget) {
|
|
2035
|
+
yield* reportFoundryStageV1("hill-climb", `reference preflight for behavioral revision ${String(sequence)}`);
|
|
2036
|
+
const preflight = yield* validateRepositoryFixturesV1({
|
|
2037
|
+
oracleCoverageWitnessVersion: input.oracleCoverageWitnessVersion,
|
|
2038
|
+
oracleRepairFeedbackVersion: input.oracleRepairFeedbackVersion,
|
|
2039
|
+
repositoryRoot: input.repositoryRoot,
|
|
2040
|
+
operationId: `${input.operationId}:hill-climb:${String(sequence)}`,
|
|
2041
|
+
seed,
|
|
2042
|
+
suite: candidateFixtureSuite,
|
|
2043
|
+
allowedTestPaths: testPaths,
|
|
2044
|
+
context: {
|
|
2045
|
+
...modelContext,
|
|
2046
|
+
visible: authored.visible,
|
|
2047
|
+
preChangeSources: initial.sources,
|
|
2048
|
+
referenceSources: reference.sources,
|
|
2049
|
+
referenceLocalImports,
|
|
2050
|
+
...(requiredFixtureScope === undefined
|
|
2051
|
+
? {}
|
|
2052
|
+
: { frozenScopeCoverage: candidateFixtureProposal.scopeCoverage })
|
|
2053
|
+
},
|
|
2054
|
+
modelPlan: input.modelPlan,
|
|
2055
|
+
maximumRepairAttempts: Math.max(0, maximumFixtureRepairAttempts - fixturePreflightRepairCalls),
|
|
2056
|
+
onRepairAttempt: () => Effect.sync(() => {
|
|
2057
|
+
fixturePreflightRepairCalls += 1;
|
|
2058
|
+
}),
|
|
2059
|
+
onRepairCall: (call) => Effect.sync(() => {
|
|
2060
|
+
hillClimbCalls.push(call);
|
|
2061
|
+
})
|
|
2062
|
+
}).pipe(Effect.match({
|
|
2063
|
+
onFailure: (error) => ({ ok: false, error }),
|
|
2064
|
+
onSuccess: (value) => ({ ok: true, value })
|
|
2065
|
+
}));
|
|
2066
|
+
if (!preflight.ok) {
|
|
2067
|
+
if (preflight.error.detail.startsWith("fixture preflight stopped on infrastructure"))
|
|
2068
|
+
return yield* preflight.error;
|
|
2069
|
+
const referencePreflight = {
|
|
2070
|
+
status: "rejected",
|
|
2071
|
+
detail: preflight.error.detail,
|
|
2072
|
+
diagnostics: preflight.error.cause
|
|
2073
|
+
};
|
|
2074
|
+
const attempt = {
|
|
2075
|
+
sequence,
|
|
2076
|
+
beforeAdequacy,
|
|
2077
|
+
proposal: repair.value,
|
|
2078
|
+
referencePreflight,
|
|
2079
|
+
decision: "rejected",
|
|
2080
|
+
rejectionReasons: [preflight.error.detail],
|
|
2081
|
+
afterAdequacy: beforeAdequacy
|
|
2082
|
+
};
|
|
2083
|
+
hillClimbAttempts.push(attempt);
|
|
2084
|
+
yield* reportFoundryAuthoringArtifactV1({
|
|
2085
|
+
version: 1,
|
|
2086
|
+
operationId: `${input.operationId}:hill-climb:${String(sequence)}:preflight-rejected`,
|
|
2087
|
+
role: "fixture-validator",
|
|
2088
|
+
model: "local-replay",
|
|
2089
|
+
reasoningEffort: "none",
|
|
2090
|
+
schemaName: "routekit_repository_fixture_revision_rejection_v1",
|
|
2091
|
+
validation: "validated",
|
|
2092
|
+
request: {
|
|
2093
|
+
instructions: "Reject a replacement that fails reference preflight; retain the incumbent.",
|
|
2094
|
+
input: { sequence, suite: candidateFixtureSuite },
|
|
2095
|
+
jsonSchema: { type: "object", properties: {}, additionalProperties: false },
|
|
2096
|
+
maximumOutputTokens: 0
|
|
2097
|
+
},
|
|
2098
|
+
response: { text: JSON.stringify(referencePreflight) },
|
|
2099
|
+
value: { diagnosticOnly: true, admissionEvidence: false, ...attempt }
|
|
2100
|
+
});
|
|
2101
|
+
yield* reportFoundryStageV1("hill-climb", `revision ${String(sequence)} rejected by reference preflight; incumbent retained`);
|
|
2102
|
+
continue;
|
|
2103
|
+
}
|
|
2104
|
+
candidateFixtureSuite = preflight.value.suite;
|
|
2105
|
+
candidateFixtureProposal = {
|
|
2106
|
+
...candidateFixtureProposal,
|
|
2107
|
+
overlays: candidateFixtureSuite.overlays
|
|
2108
|
+
};
|
|
2109
|
+
}
|
|
2110
|
+
let coverageWitnessRevalidation;
|
|
2111
|
+
if (input.oracleCoverageWitnessVersion === 2) {
|
|
2112
|
+
const revisions = yield* validateRepositoryOracleCoverageWitnessRevisionsV1({
|
|
2113
|
+
operationId: `${input.operationId}:hill-climb:${String(sequence)}:witness-revision`,
|
|
2114
|
+
repositoryRoot: input.repositoryRoot,
|
|
2115
|
+
seed,
|
|
2116
|
+
suite: candidateFixtureSuite,
|
|
2117
|
+
retained: [...retainedCoverageFixtures.values()]
|
|
2118
|
+
});
|
|
2119
|
+
coverageWitnessRevalidation = {
|
|
2120
|
+
status: revisions.every((revision) => revision.status === "validated")
|
|
2121
|
+
? "passed"
|
|
2122
|
+
: "rejected",
|
|
2123
|
+
revisions
|
|
2124
|
+
};
|
|
2125
|
+
if (coverageWitnessRevalidation.status === "rejected") {
|
|
2126
|
+
hillClimbAttempts.push({
|
|
2127
|
+
sequence,
|
|
2128
|
+
beforeAdequacy,
|
|
2129
|
+
proposal: repair.value,
|
|
2130
|
+
...(separateFixtureRepairBudget
|
|
2131
|
+
? { referencePreflight: { status: "passed" } }
|
|
2132
|
+
: {}),
|
|
2133
|
+
coverageWitnessRevalidation,
|
|
2134
|
+
decision: "rejected",
|
|
2135
|
+
rejectionReasons: revisions
|
|
2136
|
+
.filter((revision) => revision.status === "rejected")
|
|
2137
|
+
.map((revision) => `${revision.fixtureId}: ${revision.detail}`),
|
|
2138
|
+
afterAdequacy: beforeAdequacy
|
|
2139
|
+
});
|
|
2140
|
+
yield* reportFoundryStageV1("hill-climb", `revision ${String(sequence)} lost a witnessed behavioral obligation; incumbent retained`);
|
|
2141
|
+
continue;
|
|
2142
|
+
}
|
|
2143
|
+
}
|
|
2144
|
+
const candidateCompiled = yield* compile(candidateFixtureSuite);
|
|
2145
|
+
const candidateAdequacy = candidateCompiled.benchmarkCase.adequacy;
|
|
2146
|
+
const rejectionReasons = assessHillClimbCandidate(beforeAdequacy, candidateAdequacy);
|
|
2147
|
+
const decision = rejectionReasons.length === 0 ? "accepted" : "rejected";
|
|
2148
|
+
if (decision === "accepted") {
|
|
2149
|
+
fixtureProposal = candidateFixtureProposal;
|
|
2150
|
+
fixtureSuite = candidateFixtureSuite;
|
|
2151
|
+
compiled = candidateCompiled;
|
|
2152
|
+
for (const revision of coverageWitnessRevalidation?.revisions ?? []) {
|
|
2153
|
+
const retained = retainedCoverageFixtures.get(revision.fixtureId);
|
|
2154
|
+
retainedCoverageFixtures.set(revision.fixtureId, {
|
|
2155
|
+
...retained,
|
|
2156
|
+
overlay: revision.overlay
|
|
2157
|
+
});
|
|
2158
|
+
}
|
|
2159
|
+
}
|
|
2160
|
+
hillClimbAttempts.push({
|
|
2161
|
+
sequence,
|
|
2162
|
+
beforeAdequacy,
|
|
2163
|
+
proposal: repair.value,
|
|
2164
|
+
candidateAdequacy,
|
|
2165
|
+
...(input.oracleRepairFeedbackVersion !== 1 || decision !== "rejected"
|
|
2166
|
+
? {}
|
|
2167
|
+
: {
|
|
2168
|
+
candidateFixtureProposal,
|
|
2169
|
+
candidateObservations: candidateCompiled.benchmarkCase.hidden.observations.filter((observation) => [
|
|
2170
|
+
...candidateAdequacy.falseAcceptedSolutionIds,
|
|
2171
|
+
...candidateAdequacy.falseRejectedSolutionIds,
|
|
2172
|
+
...candidateAdequacy.unstableSolutionIds
|
|
2173
|
+
].includes(observation.solutionId))
|
|
2174
|
+
}),
|
|
2175
|
+
...(separateFixtureRepairBudget
|
|
2176
|
+
? { referencePreflight: { status: "passed" } }
|
|
2177
|
+
: {}),
|
|
2178
|
+
...(coverageWitnessRevalidation === undefined ? {} : { coverageWitnessRevalidation }),
|
|
2179
|
+
decision,
|
|
2180
|
+
rejectionReasons,
|
|
2181
|
+
afterAdequacy: compiled.benchmarkCase.adequacy
|
|
2182
|
+
});
|
|
2183
|
+
}
|
|
2184
|
+
if (compiled.benchmarkCase.status !== "valid-library" ||
|
|
2185
|
+
!compiled.benchmarkCase.adequacy.admitted) {
|
|
2186
|
+
return yield* failure([
|
|
2187
|
+
"development tournament remains inadequate after bounded oracle repair; held-out generation was not started",
|
|
2188
|
+
...compiled.benchmarkCase.adequacy.rejectionReasons,
|
|
2189
|
+
...compiled.benchmarkCase.rejectionReasons
|
|
2190
|
+
].join("; "));
|
|
2191
|
+
}
|
|
2192
|
+
// Any development repair invalidates the earlier approval. Recheck the exact
|
|
2193
|
+
// final suite before exposing it to held-out challenges, using the same repair reserve.
|
|
2194
|
+
if (input.maximumOracleCoverageRevisions !== undefined &&
|
|
2195
|
+
input.oracleCoverageWitnessVersion !== 2) {
|
|
2196
|
+
const reviewedFixtureProposal = yield* ensureOracleCoverage(fixtureProposal, "before-freeze", freezeCoverageModelCalls);
|
|
2197
|
+
const reviewedFixtureSuite = fixtureSuiteFrom(input.caseId, reviewedFixtureProposal);
|
|
2198
|
+
if (JSON.stringify(reviewedFixtureSuite) !== JSON.stringify(fixtureSuite)) {
|
|
2199
|
+
const beforeAdequacy = compiled.benchmarkCase.adequacy;
|
|
2200
|
+
const candidate = yield* compile(reviewedFixtureSuite);
|
|
2201
|
+
const regressions = assessHillClimbCandidate(beforeAdequacy, candidate.benchmarkCase.adequacy, false);
|
|
2202
|
+
if (regressions.length > 0 ||
|
|
2203
|
+
candidate.benchmarkCase.status !== "valid-library" ||
|
|
2204
|
+
!candidate.benchmarkCase.adequacy.admitted) {
|
|
2205
|
+
return yield* failure(`oracle coverage repair failed development revalidation before freeze; ${regressions.join("; ")}`);
|
|
2206
|
+
}
|
|
2207
|
+
fixtureProposal = reviewedFixtureProposal;
|
|
2208
|
+
fixtureSuite = reviewedFixtureSuite;
|
|
2209
|
+
compiled = candidate;
|
|
2210
|
+
}
|
|
2211
|
+
}
|
|
2212
|
+
const heldOutAdversaryCalls = [];
|
|
2213
|
+
const heldOutNoveltyFailures = [];
|
|
2214
|
+
let heldOutAdversaryProposal;
|
|
2215
|
+
let heldOutAdversarialSolutions;
|
|
2216
|
+
yield* reportFoundryStageV1("held-out-tournament");
|
|
2217
|
+
for (let sequence = 0; sequence <= 2; sequence += 1) {
|
|
2218
|
+
const generated = yield* generateSolution({
|
|
2219
|
+
plan: input.modelPlan,
|
|
2220
|
+
role: "held-out-adversary",
|
|
2221
|
+
operationId: `${input.operationId}:held-out-adversary${sequence === 0 ? "" : `:repair-${String(sequence)}`}`,
|
|
2222
|
+
instructions: HELD_OUT_ADVERSARY_INSTRUCTIONS,
|
|
2223
|
+
input: {
|
|
2224
|
+
...modelContext,
|
|
2225
|
+
visible: authored.visible,
|
|
2226
|
+
targetBehavior: seed.targetBehavior,
|
|
2227
|
+
allowedSolutionPaths: [...allowedSolutionPaths],
|
|
2228
|
+
allowedRewardHackingPaths: [...rewardHackingPaths].sort(),
|
|
2229
|
+
preChangeSources: initial.sources,
|
|
2230
|
+
adversaryFamilies: family.adversaryFamilies,
|
|
2231
|
+
excludedNearMisses: family.excludedNearMisses,
|
|
2232
|
+
forbiddenDevelopmentFamilies: [
|
|
2233
|
+
...new Set(adversarialSolutions.map((solution) => solution.family))
|
|
2234
|
+
].sort(),
|
|
2235
|
+
priorNoveltyFailures: heldOutNoveltyFailures,
|
|
2236
|
+
oracleState: "frozen-before-held-out-tournament"
|
|
2237
|
+
},
|
|
2238
|
+
schemaName: "routekit_repository_held_out_adversarial_solutions_v1",
|
|
2239
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("held-out-adversary")
|
|
2240
|
+
});
|
|
2241
|
+
heldOutAdversaryCalls.push(generated.call);
|
|
2242
|
+
const validation = validateHeldOutAdversaryProposal({
|
|
2243
|
+
proposal: generated.value,
|
|
2244
|
+
commit: seed.initialState.commit,
|
|
2245
|
+
allowedPaths: allowedSolutionPaths,
|
|
2246
|
+
allowedRewardHackingPaths: rewardHackingPaths,
|
|
2247
|
+
protectedControlPaths: seed.environment.protectedControlPaths,
|
|
2248
|
+
development: adversarialSolutions,
|
|
2249
|
+
preChangeSources: initial.sources
|
|
2250
|
+
});
|
|
2251
|
+
if (validation._tag === "valid") {
|
|
2252
|
+
heldOutAdversaryProposal = generated.value;
|
|
2253
|
+
heldOutAdversarialSolutions = validation.solutions;
|
|
2254
|
+
break;
|
|
2255
|
+
}
|
|
2256
|
+
heldOutNoveltyFailures.push(validation.detail);
|
|
2257
|
+
}
|
|
2258
|
+
if (heldOutAdversaryProposal === undefined || heldOutAdversarialSolutions === undefined) {
|
|
2259
|
+
return yield* failure(`held-out adversary proposal failed validation after bounded regeneration: ${heldOutNoveltyFailures.join("; ")}`);
|
|
2260
|
+
}
|
|
2261
|
+
const heldOutSolutionReviewInput = {
|
|
2262
|
+
...modelContext,
|
|
2263
|
+
visible: authored.visible,
|
|
2264
|
+
targetBehavior: seed.targetBehavior,
|
|
2265
|
+
preChangeSources: initial.sources,
|
|
2266
|
+
oracleState: "frozen-and-not-disclosed",
|
|
2267
|
+
solutions: heldOutAdversarialSolutions.map((solution) => ({
|
|
2268
|
+
id: solution.id,
|
|
2269
|
+
family: solution.family,
|
|
2270
|
+
fileOverrides: solution.fileOverrides
|
|
2271
|
+
}))
|
|
2272
|
+
};
|
|
2273
|
+
const [heldOutSolutionCriticA, heldOutSolutionCriticB] = yield* Effect.all([
|
|
2274
|
+
languageModel.generateStructured({
|
|
2275
|
+
plan: input.modelPlan,
|
|
2276
|
+
role: "held-out-solution-critic-a",
|
|
2277
|
+
operationId: `${input.operationId}:held-out-solution-critic-a`,
|
|
2278
|
+
instructions: solutionCriticInstructions("contract"),
|
|
2279
|
+
input: heldOutSolutionReviewInput,
|
|
2280
|
+
schemaName: "routekit_repository_held_out_solution_review_v1",
|
|
2281
|
+
outputSchema: SolutionReviewProposal,
|
|
2282
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("held-out-solution-critic-a")
|
|
2283
|
+
}),
|
|
2284
|
+
languageModel.generateStructured({
|
|
2285
|
+
plan: input.modelPlan,
|
|
2286
|
+
role: "held-out-solution-critic-b",
|
|
2287
|
+
operationId: `${input.operationId}:held-out-solution-critic-b`,
|
|
2288
|
+
instructions: solutionCriticInstructions("integration"),
|
|
2289
|
+
input: heldOutSolutionReviewInput,
|
|
2290
|
+
schemaName: "routekit_repository_held_out_solution_review_v1",
|
|
2291
|
+
outputSchema: SolutionReviewProposal,
|
|
2292
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("held-out-solution-critic-b")
|
|
2293
|
+
})
|
|
2294
|
+
], { concurrency: 2 });
|
|
2295
|
+
yield* Effect.try({
|
|
2296
|
+
try: () => {
|
|
2297
|
+
validateSolutionReview(heldOutSolutionCriticA.value, heldOutAdversarialSolutions, "held-out solution critic A");
|
|
2298
|
+
validateSolutionReview(heldOutSolutionCriticB.value, heldOutAdversarialSolutions, "held-out solution critic B");
|
|
2299
|
+
},
|
|
2300
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
2301
|
+
? cause
|
|
2302
|
+
: failure("held-out solution review failed validation", cause)
|
|
2303
|
+
});
|
|
2304
|
+
const heldOutOracle = yield* buildHistoricalHiddenOracleV1({
|
|
2305
|
+
repositoryRoot: input.repositoryRoot,
|
|
2306
|
+
caseId: input.caseId,
|
|
2307
|
+
seed: seed,
|
|
2308
|
+
additionalSolutions: [...validSolutions, ...heldOutAdversarialSolutions],
|
|
2309
|
+
hiddenFixtureSuite: fixtureSuite,
|
|
2310
|
+
...(input.oracleConcurrency === undefined
|
|
2311
|
+
? {}
|
|
2312
|
+
: { oracleConcurrency: input.oracleConcurrency }),
|
|
2313
|
+
repetitions: input.oracleRepetitions ?? 2
|
|
2314
|
+
});
|
|
2315
|
+
const heldOutAdequacy = evaluateRepositoryOracleAdequacyV1(heldOutOracle);
|
|
2316
|
+
if (!heldOutAdequacy.admitted) {
|
|
2317
|
+
return yield* failure([
|
|
2318
|
+
"frozen oracle failed the held-out adversary tournament",
|
|
2319
|
+
...heldOutAdequacy.rejectionReasons
|
|
2320
|
+
].join("; "));
|
|
2321
|
+
}
|
|
2322
|
+
const oracleCoverageWitnessRevisions = hillClimbAttempts.flatMap((attempt) => attempt.coverageWitnessRevalidation === undefined
|
|
2323
|
+
? []
|
|
2324
|
+
: [
|
|
2325
|
+
{
|
|
2326
|
+
sequence: attempt.sequence,
|
|
2327
|
+
decision: attempt.decision,
|
|
2328
|
+
revisions: attempt.coverageWitnessRevalidation.revisions
|
|
2329
|
+
}
|
|
2330
|
+
]);
|
|
2331
|
+
const qualityReviewPacket = {
|
|
2332
|
+
...(input.validControlReviewVersion === 2 ? { validControlReviewVersion: 2 } : {}),
|
|
2333
|
+
...(input.reviewInputVersion === undefined
|
|
2334
|
+
? {}
|
|
2335
|
+
: { reviewInputVersion: input.reviewInputVersion }),
|
|
2336
|
+
...(input.specificationContractFactsVersion === undefined
|
|
2337
|
+
? {}
|
|
2338
|
+
: {
|
|
2339
|
+
specificationContractFactsVersion: input.specificationContractFactsVersion,
|
|
2340
|
+
contractFacts: authored.contractFacts
|
|
2341
|
+
}),
|
|
2342
|
+
...modelContext,
|
|
2343
|
+
visible: authored.visible,
|
|
2344
|
+
targetBehavior: seed.targetBehavior,
|
|
2345
|
+
specificationReviews: authored.reviews,
|
|
2346
|
+
fixtureProposal,
|
|
2347
|
+
...(input.oracleCoverageWitnessVersion !== undefined
|
|
2348
|
+
? {
|
|
2349
|
+
oracleCoverageReviews,
|
|
2350
|
+
coverageResolutionPolicy: "Original critic verdicts are preserved. Execution assessments resolve only the fixed hypotheses via a reference-passing, isolated mutant-failing behavioral test and an incumbent replay. Retained tests protect those scenarios; this does not establish exhaustive coverage. Independently assess the actual assertions, witness semantics, valid controls and both tournaments."
|
|
2351
|
+
}
|
|
2352
|
+
: {}),
|
|
2353
|
+
...(input.oracleCoverageWitnessVersion === 2
|
|
2354
|
+
? {
|
|
2355
|
+
oracleCoverageWitnessRevisions,
|
|
2356
|
+
witnessRevisionPolicy: "Original witnesses and critic verdicts remain unchanged. A revised witness is adopted only after repeated reference passes, repeated assertion failures on the original isolated mutant, and acceptance of the complete development replacement. Review each revision and its adoption decision; rejected proposals do not replace the incumbent."
|
|
2357
|
+
}
|
|
2358
|
+
: {}),
|
|
2359
|
+
validSolutions,
|
|
2360
|
+
developmentAdversaries: adversarialSolutions,
|
|
2361
|
+
solutionReviews: [solutionCriticA.value, solutionCriticB.value],
|
|
2362
|
+
heldOutAdversaries: heldOutAdversarialSolutions,
|
|
2363
|
+
heldOutSolutionReviews: [heldOutSolutionCriticA.value, heldOutSolutionCriticB.value],
|
|
2364
|
+
heldOutAdequacy,
|
|
2365
|
+
oracleFrozenBeforeHeldOutTournament: true,
|
|
2366
|
+
trajectoryPolicy,
|
|
2367
|
+
trajectoryPolicyReview,
|
|
2368
|
+
trajectoryRevisionAttempts,
|
|
2369
|
+
executableAdequacy: compiled.benchmarkCase.adequacy,
|
|
2370
|
+
developmentOracle: {
|
|
2371
|
+
clauses: compiled.benchmarkCase.hidden.clauses,
|
|
2372
|
+
observations: compiled.benchmarkCase.hidden.observations
|
|
2373
|
+
},
|
|
2374
|
+
heldOutOracle: {
|
|
2375
|
+
clauses: heldOutOracle.clauses,
|
|
2376
|
+
observations: heldOutOracle.observations
|
|
2377
|
+
},
|
|
2378
|
+
hillClimbAttempts: hillClimbAttempts.map((attempt) => ({
|
|
2379
|
+
sequence: attempt.sequence,
|
|
2380
|
+
targetedFalseAcceptIds: attempt.proposal.targetedFalseAcceptIds,
|
|
2381
|
+
targetedFalseRejectIds: attempt.proposal.targetedFalseRejectIds,
|
|
2382
|
+
targetedInconclusiveIds: attempt.proposal.targetedInconclusiveIds ?? [],
|
|
2383
|
+
decision: attempt.decision,
|
|
2384
|
+
rejectionReasons: attempt.rejectionReasons,
|
|
2385
|
+
before: attempt.beforeAdequacy,
|
|
2386
|
+
candidate: attempt.candidateAdequacy,
|
|
2387
|
+
...(attempt.referencePreflight === undefined
|
|
2388
|
+
? {}
|
|
2389
|
+
: { referencePreflight: attempt.referencePreflight }),
|
|
2390
|
+
after: attempt.afterAdequacy
|
|
2391
|
+
}))
|
|
2392
|
+
};
|
|
2393
|
+
const qualityEvidenceInventory = qualityReviewEvidenceInventory({
|
|
2394
|
+
targetBehaviorCount: qualityReviewPacket.targetBehavior.length,
|
|
2395
|
+
specificationReviewCount: qualityReviewPacket.specificationReviews.length,
|
|
2396
|
+
fixtureCount: qualityReviewPacket.fixtureProposal.fixtures.length,
|
|
2397
|
+
overlayCount: qualityReviewPacket.fixtureProposal.overlays.length,
|
|
2398
|
+
validSolutionCount: qualityReviewPacket.validSolutions.length,
|
|
2399
|
+
developmentAdversaryCount: qualityReviewPacket.developmentAdversaries.length,
|
|
2400
|
+
solutionReviewCount: qualityReviewPacket.solutionReviews.length,
|
|
2401
|
+
heldOutAdversaryCount: qualityReviewPacket.heldOutAdversaries.length,
|
|
2402
|
+
heldOutSolutionReviewCount: qualityReviewPacket.heldOutSolutionReviews.length,
|
|
2403
|
+
trajectoryRevisionAttemptCount: qualityReviewPacket.trajectoryRevisionAttempts.length,
|
|
2404
|
+
hillClimbAttemptCount: qualityReviewPacket.hillClimbAttempts.length,
|
|
2405
|
+
developmentObservationCount: qualityReviewPacket.developmentOracle.observations.length,
|
|
2406
|
+
heldOutObservationCount: qualityReviewPacket.heldOutOracle.observations.length,
|
|
2407
|
+
...(input.oracleCoverageWitnessVersion !== undefined
|
|
2408
|
+
? { oracleCoverageReviewCount: oracleCoverageReviews.length }
|
|
2409
|
+
: {}),
|
|
2410
|
+
...(input.oracleCoverageWitnessVersion === 2
|
|
2411
|
+
? { oracleCoverageWitnessRevisionCount: oracleCoverageWitnessRevisions.length }
|
|
2412
|
+
: {})
|
|
2413
|
+
});
|
|
2414
|
+
const qualityReviewInput = {
|
|
2415
|
+
...qualityReviewPacket,
|
|
2416
|
+
evidenceInventory: qualityEvidenceInventory
|
|
2417
|
+
};
|
|
2418
|
+
const generation = {
|
|
2419
|
+
version: 1,
|
|
2420
|
+
authored,
|
|
2421
|
+
sourceContext: {
|
|
2422
|
+
initialCommit: initialSnapshot.commit,
|
|
2423
|
+
referenceCommit: reference.snapshot.commit,
|
|
2424
|
+
initialSupportingPaths: initialLocalImports.sources.map((source) => source.path),
|
|
2425
|
+
referenceSupportingPaths: referenceLocalImports.sources.map((source) => source.path)
|
|
2426
|
+
},
|
|
2427
|
+
initialFixtureProposal: fixtureGeneration.value,
|
|
2428
|
+
fixtureProposal,
|
|
2429
|
+
...(input.maximumOracleCoverageRevisions === undefined ? {} : { oracleCoverageReviews }),
|
|
2430
|
+
...(input.oracleCoverageWitnessVersion === 2 ? { oracleCoverageWitnessRevisions } : {}),
|
|
2431
|
+
trajectoryPolicy,
|
|
2432
|
+
trajectoryPolicyReview,
|
|
2433
|
+
trajectoryRevisionAttempts,
|
|
2434
|
+
validSolutionProposal: {
|
|
2435
|
+
solutions: [...validGenerationA.value.solutions, ...validGenerationB.value.solutions]
|
|
2436
|
+
},
|
|
2437
|
+
adversaryProposal: adversaryGeneration.value,
|
|
2438
|
+
solutionReviews: [solutionCriticA.value, solutionCriticB.value],
|
|
2439
|
+
heldOutAdversaryProposal,
|
|
2440
|
+
heldOutSolutionReviews: [heldOutSolutionCriticA.value, heldOutSolutionCriticB.value],
|
|
2441
|
+
heldOutAdequacy,
|
|
2442
|
+
heldOutOracle,
|
|
2443
|
+
weakAdequacy,
|
|
2444
|
+
hillClimbAttempts,
|
|
2445
|
+
benchmarkCase: compiled.benchmarkCase,
|
|
2446
|
+
modelCalls: [
|
|
2447
|
+
...authored.modelCalls,
|
|
2448
|
+
...trajectoryModelCalls,
|
|
2449
|
+
...oracleAuthoringModelCalls,
|
|
2450
|
+
...validSolutionModelCalls,
|
|
2451
|
+
adversaryGeneration.call,
|
|
2452
|
+
...solutionReviewModelCalls,
|
|
2453
|
+
...hillClimbCalls,
|
|
2454
|
+
...freezeCoverageModelCalls,
|
|
2455
|
+
...heldOutAdversaryCalls,
|
|
2456
|
+
heldOutSolutionCriticA.call,
|
|
2457
|
+
heldOutSolutionCriticB.call
|
|
2458
|
+
]
|
|
2459
|
+
};
|
|
2460
|
+
const reconstruction = yield* Effect.serviceOption(RepositoryFoundryEvidenceReconstruction);
|
|
2461
|
+
const checkpoint = {
|
|
2462
|
+
version: 1,
|
|
2463
|
+
kind: "pending-quality-review",
|
|
2464
|
+
...(input.validControlReviewVersion === 2 ? { validControlReviewVersion: 2 } : {}),
|
|
2465
|
+
...(input.reviewInputVersion === undefined
|
|
2466
|
+
? {}
|
|
2467
|
+
: { reviewInputVersion: input.reviewInputVersion }),
|
|
2468
|
+
...(input.specificationContractFactsVersion === undefined
|
|
2469
|
+
? {}
|
|
2470
|
+
: { specificationContractFactsVersion: input.specificationContractFactsVersion }),
|
|
2471
|
+
repositoryRoot: input.repositoryRoot,
|
|
2472
|
+
operationId: input.operationId,
|
|
2473
|
+
modelPlan: input.modelPlan,
|
|
2474
|
+
seed,
|
|
2475
|
+
requiresReplayValidation: Option.isSome(reconstruction),
|
|
2476
|
+
generation,
|
|
2477
|
+
qualityReviewInput,
|
|
2478
|
+
sourceBases: editSources.map((source) => source.content)
|
|
2479
|
+
};
|
|
2480
|
+
yield* reportFoundryAuthoringArtifactV1({
|
|
2481
|
+
version: 1,
|
|
2482
|
+
operationId: `${input.operationId}:quality-review-checkpoint`,
|
|
2483
|
+
role: "checkpoint-writer",
|
|
2484
|
+
model: "local-evidence",
|
|
2485
|
+
reasoningEffort: "none",
|
|
2486
|
+
schemaName: "routekit_repository_quality_review_checkpoint_v1",
|
|
2487
|
+
validation: "validated",
|
|
2488
|
+
request: {
|
|
2489
|
+
instructions: "Complete frozen case before quality review. This checkpoint is not an admission or a model call.",
|
|
2490
|
+
input: {
|
|
2491
|
+
caseId: input.caseId,
|
|
2492
|
+
requiresReplayValidation: checkpoint.requiresReplayValidation
|
|
2493
|
+
},
|
|
2494
|
+
jsonSchema: { type: "object" },
|
|
2495
|
+
maximumOutputTokens: 0
|
|
2496
|
+
},
|
|
2497
|
+
response: { text: "" },
|
|
2498
|
+
value: checkpoint
|
|
2499
|
+
});
|
|
2500
|
+
if (input.checkpointOnly === true) {
|
|
2501
|
+
if (Option.isNone(reconstruction))
|
|
2502
|
+
return yield* failure("checkpoint reconstruction requires exact archived evidence");
|
|
2503
|
+
yield* reconstruction.value.assertConsumed;
|
|
2504
|
+
return { kind: "checkpoint", checkpoint };
|
|
2505
|
+
}
|
|
2506
|
+
const result = yield* reviewRepositoryGeneratedCaseCheckpointV1({ checkpoint });
|
|
2507
|
+
return { kind: "generated", result };
|
|
2508
|
+
});
|
|
2509
|
+
export const generateHistoricalRepositoryCaseV1 = Effect.fn("CaseGeneration.completeHistorical")(function* (input) {
|
|
2510
|
+
const result = yield* generateHistoricalRepositoryCaseStagesV1(input);
|
|
2511
|
+
if (result.kind !== "generated")
|
|
2512
|
+
return yield* failure("generation stopped before final review");
|
|
2513
|
+
return result.result;
|
|
2514
|
+
});
|
|
2515
|
+
export const reconstructRepositoryGeneratedCaseCheckpointV1 = Effect.fn("CaseGeneration.reconstructCheckpoint")(function* (input) {
|
|
2516
|
+
const reconstruction = yield* Effect.serviceOption(RepositoryFoundryEvidenceReconstruction);
|
|
2517
|
+
if (Option.isNone(reconstruction))
|
|
2518
|
+
return yield* failure("checkpoint reconstruction requires an archived-evidence scope");
|
|
2519
|
+
const result = yield* generateHistoricalRepositoryCaseStagesV1({
|
|
2520
|
+
...input,
|
|
2521
|
+
checkpointOnly: true
|
|
2522
|
+
});
|
|
2523
|
+
if (result.kind !== "checkpoint")
|
|
2524
|
+
return yield* failure("reconstruction did not stop before review");
|
|
2525
|
+
return result.checkpoint;
|
|
2526
|
+
});
|
|
2527
|
+
/** Pure sizing of the full frozen review request, including wire wrapping and schema. */
|
|
2528
|
+
export const prepareRepositoryCaseQualityReviewV1 = (input) => {
|
|
2529
|
+
const checkpoint = input.checkpoint;
|
|
2530
|
+
const reviewInputVersion = input.reviewInputVersion ?? checkpoint.reviewInputVersion;
|
|
2531
|
+
const requestByteLimit = evalAuthoringRequestByteLimit({
|
|
2532
|
+
foundryRole: "quality-reviewer",
|
|
2533
|
+
schemaName: "routekit_repository_generation_quality_review_v1",
|
|
2534
|
+
...(reviewInputVersion === undefined ? {} : { reviewInputVersion })
|
|
2535
|
+
});
|
|
2536
|
+
const document = Schema.toJsonSchemaDocument(QualityReview);
|
|
2537
|
+
const jsonSchema = strictAuthoringSchema({
|
|
2538
|
+
...document.schema,
|
|
2539
|
+
...(Object.keys(document.definitions).length === 0 ? {} : { $defs: document.definitions })
|
|
2540
|
+
}).jsonSchema;
|
|
2541
|
+
const reviewer = repositoryFoundryModelAssignmentV1(checkpoint.modelPlan, "quality-reviewer");
|
|
2542
|
+
const size = (evidence, instructions) => Buffer.byteLength(evalAuthoringResponsesRequestBody({
|
|
2543
|
+
operationId: input.operationId ?? `${checkpoint.operationId}:quality-reviewer`,
|
|
2544
|
+
model: reviewer.model,
|
|
2545
|
+
reasoningEffort: reviewer.reasoningEffort,
|
|
2546
|
+
instructions: `${qualityReviewInstructions(checkpoint.validControlReviewVersion)}\n\n${instructions}`,
|
|
2547
|
+
input: JSON.stringify(evidence),
|
|
2548
|
+
schemaName: "routekit_repository_generation_quality_review_v1",
|
|
2549
|
+
jsonSchema,
|
|
2550
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("quality-reviewer")
|
|
2551
|
+
}));
|
|
2552
|
+
const legacy = encodeRepositoryReviewEvidenceV1(checkpoint.qualityReviewInput, checkpoint.sourceBases);
|
|
2553
|
+
const legacyBytes = size(legacy, REPOSITORY_REVIEW_EVIDENCE_INSTRUCTIONS);
|
|
2554
|
+
if (legacyBytes <= REPOSITORY_FOUNDRY_REQUEST_INPUT_CEILING)
|
|
2555
|
+
return {
|
|
2556
|
+
evidence: legacy,
|
|
2557
|
+
instructions: REPOSITORY_REVIEW_EVIDENCE_INSTRUCTIONS,
|
|
2558
|
+
requestBytes: legacyBytes,
|
|
2559
|
+
requestByteLimit
|
|
2560
|
+
};
|
|
2561
|
+
const evidence = encodeRepositoryReviewEvidenceV2(checkpoint.qualityReviewInput, checkpoint.sourceBases);
|
|
2562
|
+
const requestBytes = size(evidence, REPOSITORY_REVIEW_EVIDENCE_V2_INSTRUCTIONS);
|
|
2563
|
+
if (requestBytes > requestByteLimit)
|
|
2564
|
+
throw failure("complete benchmark artifacts exceed the independent quality review input ceiling; frozen checkpoint retained");
|
|
2565
|
+
return {
|
|
2566
|
+
evidence,
|
|
2567
|
+
instructions: REPOSITORY_REVIEW_EVIDENCE_V2_INSTRUCTIONS,
|
|
2568
|
+
requestBytes,
|
|
2569
|
+
requestByteLimit
|
|
2570
|
+
};
|
|
2571
|
+
};
|
|
2572
|
+
export const reviewRepositoryGeneratedCaseCheckpointV1 = Effect.fn("CaseGeneration.reviewCheckpoint")(function* (input) {
|
|
2573
|
+
const reconstruction = yield* Effect.serviceOption(RepositoryFoundryEvidenceReconstruction);
|
|
2574
|
+
if (Option.isSome(reconstruction) || input.checkpoint.requiresReplayValidation)
|
|
2575
|
+
return yield* failure("diagnostic reconstruction cannot perform review or confer admission before real replay validation");
|
|
2576
|
+
const languageModel = yield* RepositoryFoundryLanguageModel;
|
|
2577
|
+
const checkpoint = input.checkpoint;
|
|
2578
|
+
const reviewInputVersion = input.reviewInputVersion ?? checkpoint.reviewInputVersion;
|
|
2579
|
+
const inventory = checkpoint.qualityReviewInput.evidenceInventory;
|
|
2580
|
+
if (!Array.isArray(inventory) || inventory.some((entry) => typeof entry !== "string"))
|
|
2581
|
+
return yield* failure("checkpoint review requires its complete evidence inventory");
|
|
2582
|
+
const qualityEvidenceInventory = inventory;
|
|
2583
|
+
const encoded = yield* Effect.try({
|
|
2584
|
+
try: () => prepareRepositoryCaseQualityReviewV1(input),
|
|
2585
|
+
catch: (cause) => failure("complete benchmark review evidence could not be encoded losslessly", cause)
|
|
2586
|
+
});
|
|
2587
|
+
yield* reportFoundryStageV1("quality-review");
|
|
2588
|
+
const qualityGeneration = yield* languageModel.generateStructured({
|
|
2589
|
+
plan: checkpoint.modelPlan,
|
|
2590
|
+
role: "quality-reviewer",
|
|
2591
|
+
...(reviewInputVersion === undefined ? {} : { reviewInputVersion }),
|
|
2592
|
+
operationId: input.operationId ?? `${checkpoint.operationId}:quality-reviewer`,
|
|
2593
|
+
instructions: `${qualityReviewInstructions(checkpoint.validControlReviewVersion)}\n\n${encoded.instructions}`,
|
|
2594
|
+
input: encoded.evidence,
|
|
2595
|
+
schemaName: "routekit_repository_generation_quality_review_v1",
|
|
2596
|
+
outputSchema: QualityReview,
|
|
2597
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1("quality-reviewer")
|
|
2598
|
+
});
|
|
2599
|
+
yield* Effect.try({
|
|
2600
|
+
try: () => validateRepositoryGenerationQualityReviewV1(qualityGeneration.value, qualityEvidenceInventory),
|
|
2601
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
2602
|
+
? cause
|
|
2603
|
+
: failure("independent quality review failed validation", cause)
|
|
2604
|
+
});
|
|
2605
|
+
if (qualityGeneration.value.verdict !== "approve" ||
|
|
2606
|
+
qualityGeneration.value.coverageGaps.length > 0 ||
|
|
2607
|
+
qualityGeneration.value.correlatedAssumptions.length > 0)
|
|
2608
|
+
return yield* failure([
|
|
2609
|
+
"independent quality review rejected the generated case",
|
|
2610
|
+
qualityGeneration.value.detail,
|
|
2611
|
+
...qualityGeneration.value.coverageGaps.map((gap) => `coverage-gap=${gap}`),
|
|
2612
|
+
...qualityGeneration.value.correlatedAssumptions.map((assumption) => `correlated-assumption=${assumption}`)
|
|
2613
|
+
].join("; "));
|
|
2614
|
+
yield* reportFoundryStageV1("admit", `status=${checkpoint.generation.benchmarkCase.status}`);
|
|
2615
|
+
return {
|
|
2616
|
+
...checkpoint.generation,
|
|
2617
|
+
qualityReview: qualityGeneration.value,
|
|
2618
|
+
modelCalls: [...checkpoint.generation.modelCalls, qualityGeneration.call]
|
|
2619
|
+
};
|
|
2620
|
+
});
|
|
2621
|
+
export const makeCaseGeneration = Effect.gen(function* () {
|
|
2622
|
+
const languageModel = yield* RepositoryFoundryLanguageModel;
|
|
2623
|
+
return CaseGeneration.of({
|
|
2624
|
+
reviewSemanticSolutions: (input) => executeRepositorySemanticReviewV1(input).pipe(Effect.provideService(RepositoryFoundryLanguageModel, languageModel)),
|
|
2625
|
+
generateHistorical: (input) => generateHistoricalRepositoryCaseV1(input).pipe(Effect.provideService(RepositoryFoundryLanguageModel, languageModel))
|
|
2626
|
+
});
|
|
2627
|
+
});
|
|
2628
|
+
export const CaseGenerationLive = Layer.effect(CaseGeneration, makeCaseGeneration);
|