@velum-labs/routekit-eval-setup 1.3.2 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/authoring-responses-request.d.ts +5 -0
- package/dist/adapters/authoring-responses-request.js +38 -0
- package/dist/adapters/evaluation-evidence-freshness.d.ts +7 -0
- package/dist/adapters/evaluation-evidence-freshness.js +76 -0
- package/dist/adapters/git-task-history.d.ts +67 -0
- package/dist/adapters/git-task-history.js +171 -0
- package/dist/adapters/integrated-repository-history.d.ts +21 -0
- package/dist/adapters/integrated-repository-history.js +175 -0
- package/dist/adapters/repository-command-diagnostic.d.ts +8 -0
- package/dist/adapters/repository-command-diagnostic.js +46 -0
- package/dist/adapters/repository-command-evidence.d.ts +13 -0
- package/dist/adapters/repository-command-evidence.js +102 -0
- package/dist/adapters/repository-command-runner.d.ts +238 -0
- package/dist/adapters/repository-command-runner.js +1483 -0
- package/dist/adapters/repository-import-context.d.ts +47 -0
- package/dist/adapters/repository-import-context.js +469 -0
- package/dist/adapters/repository-node-test-reporter.d.ts +3 -0
- package/dist/adapters/repository-node-test-reporter.js +27 -0
- package/dist/adapters/repository-review-evidence.d.ts +39 -0
- package/dist/adapters/repository-review-evidence.js +632 -0
- package/dist/adapters/repository-seed-selection.d.ts +7 -0
- package/dist/adapters/repository-seed-selection.js +79 -0
- package/dist/adapters/repository-solution-edits.d.ts +49 -0
- package/dist/adapters/repository-solution-edits.js +136 -0
- package/dist/adapters/repository-vitest-phase-adapter.d.ts +8 -0
- package/dist/adapters/repository-vitest-phase-adapter.js +310 -0
- package/dist/adapters/repository-vitest-reporter.d.ts +24 -0
- package/dist/adapters/repository-vitest-reporter.js +314 -0
- package/dist/adapters/strict-authoring-schema.d.ts +5 -0
- package/dist/adapters/strict-authoring-schema.js +158 -0
- package/dist/adapters/test-discovery.d.ts +30 -0
- package/dist/adapters/test-discovery.js +124 -0
- package/dist/adapters/typescript-repository-index.d.ts +51 -0
- package/dist/adapters/typescript-repository-index.js +226 -0
- package/dist/agentic-capabilities-protocol.d.ts +1373 -0
- package/dist/agentic-capabilities-protocol.js +786 -0
- package/dist/case-checkpoint-store.d.ts +29 -0
- package/dist/case-checkpoint-store.js +133 -0
- package/dist/case-pipeline-protocol-v2.d.ts +184 -0
- package/dist/case-pipeline-protocol-v2.js +193 -0
- package/dist/case-pipeline-protocol.d.ts +2626 -0
- package/dist/case-pipeline-protocol.js +371 -0
- package/dist/effect-api.d.ts +74 -10
- package/dist/effect-api.js +56 -6
- package/dist/errors.d.ts +31 -0
- package/dist/errors.js +10 -0
- package/dist/eval-capability-execution-envelope.d.ts +64 -0
- package/dist/eval-capability-execution-envelope.js +98 -0
- package/dist/eval-capability-policy.d.ts +90 -0
- package/dist/eval-capability-policy.js +107 -0
- package/dist/eval-event-log.d.ts +140 -0
- package/dist/eval-event-log.js +220 -0
- package/dist/evaluation-authoring-policy.d.ts +18 -0
- package/dist/evaluation-authoring-policy.js +19 -0
- package/dist/evaluation-authoring-validation.d.ts +22 -0
- package/dist/evaluation-authoring-validation.js +72 -0
- package/dist/evaluation-evidence.d.ts +20 -0
- package/dist/evaluation-evidence.js +319 -0
- package/dist/evaluation-grader-calibration-protocol.d.ts +108 -0
- package/dist/evaluation-grader-calibration-protocol.js +80 -0
- package/dist/evaluation-grader-calibration.d.ts +18 -0
- package/dist/evaluation-grader-calibration.js +334 -0
- package/dist/evaluation-grading-policy.d.ts +24 -0
- package/dist/evaluation-grading-policy.js +54 -0
- package/dist/evaluation-proposal-policy.d.ts +4 -0
- package/dist/evaluation-proposal-policy.js +91 -0
- package/dist/evaluation-source-retrieval.d.ts +68 -0
- package/dist/evaluation-source-retrieval.js +513 -0
- package/dist/evaluation-structure-policy.d.ts +29 -0
- package/dist/evaluation-structure-policy.js +138 -0
- package/dist/index.d.ts +124 -17
- package/dist/index.js +69 -11
- package/dist/inspection.js +2 -3
- package/dist/project-artifacts.d.ts +7 -2
- package/dist/project-artifacts.js +49 -136
- package/dist/project-authoring.d.ts +66 -5
- package/dist/project-authoring.js +783 -109
- package/dist/project-contracts.d.ts +419 -84
- package/dist/project-contracts.js +160 -52
- package/dist/project-store.js +2 -1
- package/dist/project-workflow.d.ts +5 -4
- package/dist/project-workflow.js +154 -35
- package/dist/repository-adversary-protocol.d.ts +64 -0
- package/dist/repository-adversary-protocol.js +105 -0
- package/dist/repository-behavior-protocol.d.ts +188 -0
- package/dist/repository-behavior-protocol.js +202 -0
- package/dist/repository-benchmark-protocol.d.ts +487 -0
- package/dist/repository-benchmark-protocol.js +96 -0
- package/dist/repository-execution-protocol.d.ts +150 -0
- package/dist/repository-execution-protocol.js +38 -0
- package/dist/repository-fixture-instructions.d.ts +3 -0
- package/dist/repository-fixture-instructions.js +91 -0
- package/dist/repository-fixture-protocol.d.ts +79 -0
- package/dist/repository-fixture-protocol.js +79 -0
- package/dist/repository-foundry-plan-protocol.d.ts +118 -0
- package/dist/repository-foundry-plan-protocol.js +296 -0
- package/dist/repository-foundry-progress-protocol.d.ts +52 -0
- package/dist/repository-foundry-progress-protocol.js +52 -0
- package/dist/repository-improvement-protocol.d.ts +100 -0
- package/dist/repository-improvement-protocol.js +106 -0
- package/dist/repository-language-model-protocol.d.ts +43 -0
- package/dist/repository-language-model-protocol.js +146 -0
- package/dist/repository-oracle-coverage-protocol.d.ts +18 -0
- package/dist/repository-oracle-coverage-protocol.js +39 -0
- package/dist/repository-oracle-execution-binding.d.ts +27 -0
- package/dist/repository-oracle-execution-binding.js +59 -0
- package/dist/repository-oracle-protocol.d.ts +230 -0
- package/dist/repository-oracle-protocol.js +156 -0
- package/dist/repository-oracle-scope-policy.d.ts +22 -0
- package/dist/repository-oracle-scope-policy.js +92 -0
- package/dist/repository-quality-policy.d.ts +15 -0
- package/dist/repository-quality-policy.js +357 -0
- package/dist/repository-routing-benchmark-protocol.d.ts +176 -0
- package/dist/repository-routing-benchmark-protocol.js +103 -0
- package/dist/repository-routing-model-protocol.d.ts +36 -0
- package/dist/repository-routing-model-protocol.js +89 -0
- package/dist/repository-routing-plan-protocol.d.ts +112 -0
- package/dist/repository-routing-plan-protocol.js +58 -0
- package/dist/repository-routing-quality-policy.d.ts +9 -0
- package/dist/repository-routing-quality-policy.js +191 -0
- package/dist/repository-seed-qualification-progress-protocol.d.ts +205 -0
- package/dist/repository-seed-qualification-progress-protocol.js +28 -0
- package/dist/repository-semantic-calibration-protocol.d.ts +768 -0
- package/dist/repository-semantic-calibration-protocol.js +276 -0
- package/dist/repository-semantic-calibration.d.ts +163 -0
- package/dist/repository-semantic-calibration.js +581 -0
- package/dist/repository-specification-contract-facts-protocol.d.ts +224 -0
- package/dist/repository-specification-contract-facts-protocol.js +276 -0
- package/dist/repository-specification-critique-protocol.d.ts +189 -0
- package/dist/repository-specification-critique-protocol.js +103 -0
- package/dist/repository-task-family-protocol.d.ts +24 -0
- package/dist/repository-task-family-protocol.js +37 -0
- package/dist/repository-task-seed-protocol.d.ts +384 -0
- package/dist/repository-task-seed-protocol.js +236 -0
- package/dist/repository-trajectory-protocol.d.ts +20 -0
- package/dist/repository-trajectory-protocol.js +42 -0
- package/dist/service.js +1 -1
- package/dist/services/adversary/service.d.ts +64 -0
- package/dist/services/adversary/service.js +330 -0
- package/dist/services/benchmark-compiler/service.d.ts +450 -0
- package/dist/services/benchmark-compiler/service.js +9 -0
- package/dist/services/budgeted-model/service.d.ts +118 -0
- package/dist/services/budgeted-model/service.js +460 -0
- package/dist/services/case-authoring/service.d.ts +163 -0
- package/dist/services/case-authoring/service.js +1456 -0
- package/dist/services/case-finalization/service.d.ts +283 -0
- package/dist/services/case-finalization/service.js +370 -0
- package/dist/services/case-generation/service.d.ts +619 -0
- package/dist/services/case-generation/service.js +2628 -0
- package/dist/services/case-pipeline/service.d.ts +31 -0
- package/dist/services/case-pipeline/service.js +485 -0
- package/dist/services/case-pipeline-v2/service.d.ts +70 -0
- package/dist/services/case-pipeline-v2/service.js +477 -0
- package/dist/services/command-observability/service.d.ts +13 -0
- package/dist/services/command-observability/service.js +3 -0
- package/dist/services/dimension-labeling/service.d.ts +77 -0
- package/dist/services/dimension-labeling/service.js +188 -0
- package/dist/services/eval-candidate/service.d.ts +208 -0
- package/dist/services/eval-candidate/service.js +64 -0
- package/dist/services/eval-capabilities/service.d.ts +183 -0
- package/dist/services/eval-capabilities/service.js +1433 -0
- package/dist/services/eval-environment/service.d.ts +173 -0
- package/dist/services/eval-environment/service.js +127 -0
- package/dist/services/evidence-reconstruction/service.d.ts +36 -0
- package/dist/services/evidence-reconstruction/service.js +145 -0
- package/dist/services/fixture-builder/service.d.ts +62 -0
- package/dist/services/fixture-builder/service.js +36 -0
- package/dist/services/fixture-validation/service.d.ts +75 -0
- package/dist/services/fixture-validation/service.js +295 -0
- package/dist/services/foundry/service.d.ts +831 -0
- package/dist/services/foundry/service.js +442 -0
- package/dist/services/foundry-progress/service.d.ts +62 -0
- package/dist/services/foundry-progress/service.js +149 -0
- package/dist/services/foundry-v2/service.d.ts +54 -0
- package/dist/services/foundry-v2/service.js +28 -0
- package/dist/services/grounded-authoring/service.d.ts +126 -0
- package/dist/services/grounded-authoring/service.js +822 -0
- package/dist/services/historical-case/service.d.ts +722 -0
- package/dist/services/historical-case/service.js +177 -0
- package/dist/services/improvement-loop/service.d.ts +59 -0
- package/dist/services/improvement-loop/service.js +176 -0
- package/dist/services/language-model/service.d.ts +52 -0
- package/dist/services/language-model/service.js +194 -0
- package/dist/services/oracle-builder/service.d.ts +146 -0
- package/dist/services/oracle-builder/service.js +513 -0
- package/dist/services/oracle-coverage/service.d.ts +28 -0
- package/dist/services/oracle-coverage/service.js +50 -0
- package/dist/services/oracle-coverage-witness/service.d.ts +130 -0
- package/dist/services/oracle-coverage-witness/service.js +538 -0
- package/dist/services/pipeline-challenge/service.d.ts +551 -0
- package/dist/services/pipeline-challenge/service.js +427 -0
- package/dist/services/pipeline-controls/service.d.ts +130 -0
- package/dist/services/pipeline-controls/service.js +483 -0
- package/dist/services/pipeline-oracle/service.d.ts +8 -0
- package/dist/services/pipeline-oracle/service.js +256 -0
- package/dist/services/pipeline-seed/service.d.ts +298 -0
- package/dist/services/pipeline-seed/service.js +428 -0
- package/dist/services/pipeline-spec/service.d.ts +103 -0
- package/dist/services/pipeline-spec/service.js +619 -0
- package/dist/services/pipeline-tournament/service.d.ts +258 -0
- package/dist/services/pipeline-tournament/service.js +476 -0
- package/dist/services/quality-gate/service.d.ts +233 -0
- package/dist/services/quality-gate/service.js +136 -0
- package/dist/services/repository-bundle/service.d.ts +33 -0
- package/dist/services/repository-bundle/service.js +114 -0
- package/dist/services/repository-model/service.d.ts +105 -0
- package/dist/services/repository-model/service.js +250 -0
- package/dist/services/repository-public-artifact/service.d.ts +133 -0
- package/dist/services/repository-public-artifact/service.js +330 -0
- package/dist/services/routing-benchmark/service.d.ts +362 -0
- package/dist/services/routing-benchmark/service.js +96 -0
- package/dist/services/specification-critic/service.d.ts +92 -0
- package/dist/services/specification-critic/service.js +172 -0
- package/dist/services/task-family/service.d.ts +40 -0
- package/dist/services/task-family/service.js +55 -0
- package/dist/services/task-seed/service.d.ts +906 -0
- package/dist/services/task-seed/service.js +1406 -0
- package/dist/services/task-specification/service.d.ts +27 -0
- package/dist/services/task-specification/service.js +40 -0
- package/dist/services/trajectory-policy/service.d.ts +110 -0
- package/dist/services/trajectory-policy/service.js +216 -0
- package/dist/test/agentic-capabilities-protocol.test.d.ts +1 -0
- package/dist/test/agentic-capabilities-protocol.test.js +570 -0
- package/dist/test/agentic-capabilities.test.d.ts +1 -0
- package/dist/test/agentic-capabilities.test.js +1461 -0
- package/dist/test/agentic-environment.test.d.ts +1 -0
- package/dist/test/agentic-environment.test.js +213 -0
- package/dist/test/case-pipeline-foundation.test.d.ts +1 -0
- package/dist/test/case-pipeline-foundation.test.js +535 -0
- package/dist/test/case-pipeline-protocol-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-protocol-v2.test.js +124 -0
- package/dist/test/case-pipeline-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-v2.test.js +286 -0
- package/dist/test/case-pipeline.test.d.ts +1 -0
- package/dist/test/case-pipeline.test.js +851 -0
- package/dist/test/eval-capability-policy.test.d.ts +1 -0
- package/dist/test/eval-capability-policy.test.js +50 -0
- package/dist/test/eval-event-log.test.d.ts +1 -0
- package/dist/test/eval-event-log.test.js +125 -0
- package/dist/test/evaluation-evidence-freshness.test.d.ts +1 -0
- package/dist/test/evaluation-evidence-freshness.test.js +44 -0
- package/dist/test/evaluation-evidence.test.d.ts +1 -0
- package/dist/test/evaluation-evidence.test.js +230 -0
- package/dist/test/evaluation-grader-calibration.test.d.ts +1 -0
- package/dist/test/evaluation-grader-calibration.test.js +373 -0
- package/dist/test/evaluation-proposal-digest.test.d.ts +1 -0
- package/dist/test/evaluation-proposal-digest.test.js +187 -0
- package/dist/test/evaluation-source-retrieval.test.d.ts +1 -0
- package/dist/test/evaluation-source-retrieval.test.js +237 -0
- package/dist/test/evaluation-structure-policy.test.d.ts +1 -0
- package/dist/test/evaluation-structure-policy.test.js +196 -0
- package/dist/test/fixtures/repository-resource-panel.d.ts +39 -0
- package/dist/test/fixtures/repository-resource-panel.js +111 -0
- package/dist/test/fixtures/vitest-boundary-panel.d.ts +84 -0
- package/dist/test/fixtures/vitest-boundary-panel.js +120 -0
- package/dist/test/fixtures/vitest-phase-panel.d.ts +135 -0
- package/dist/test/fixtures/vitest-phase-panel.js +213 -0
- package/dist/test/fixtures/vitest-reporter-results.d.ts +76 -0
- package/dist/test/fixtures/vitest-reporter-results.js +94 -0
- package/dist/test/grounded-authoring.test.d.ts +1 -0
- package/dist/test/grounded-authoring.test.js +565 -0
- package/dist/test/integrated-repository-history.test.d.ts +1 -0
- package/dist/test/integrated-repository-history.test.js +227 -0
- package/dist/test/project-authoring.test.js +593 -43
- package/dist/test/project-workflow.test.js +419 -40
- package/dist/test/repository-authoring-artifacts.test.d.ts +1 -0
- package/dist/test/repository-authoring-artifacts.test.js +185 -0
- package/dist/test/repository-bundle.test.d.ts +1 -0
- package/dist/test/repository-bundle.test.js +52 -0
- package/dist/test/repository-case-generation.test.d.ts +1 -0
- package/dist/test/repository-case-generation.test.js +3465 -0
- package/dist/test/repository-command-diagnostic.test.d.ts +1 -0
- package/dist/test/repository-command-diagnostic.test.js +55 -0
- package/dist/test/repository-command-signals.test.d.ts +1 -0
- package/dist/test/repository-command-signals.test.js +124 -0
- package/dist/test/repository-fixture-scope-coverage.test.d.ts +1 -0
- package/dist/test/repository-fixture-scope-coverage.test.js +127 -0
- package/dist/test/repository-fixture-validation.test.d.ts +1 -0
- package/dist/test/repository-fixture-validation.test.js +362 -0
- package/dist/test/repository-foundry-progress.test.d.ts +1 -0
- package/dist/test/repository-foundry-progress.test.js +110 -0
- package/dist/test/repository-foundry-quality.test.d.ts +1 -0
- package/dist/test/repository-foundry-quality.test.js +1138 -0
- package/dist/test/repository-import-context.test.d.ts +1 -0
- package/dist/test/repository-import-context.test.js +354 -0
- package/dist/test/repository-model-authoring.test.d.ts +1 -0
- package/dist/test/repository-model-authoring.test.js +544 -0
- package/dist/test/repository-model.test.d.ts +1 -0
- package/dist/test/repository-model.test.js +2195 -0
- package/dist/test/repository-node-test-reporter.test.d.ts +1 -0
- package/dist/test/repository-node-test-reporter.test.js +104 -0
- package/dist/test/repository-oracle-concurrency.test.d.ts +1 -0
- package/dist/test/repository-oracle-concurrency.test.js +542 -0
- package/dist/test/repository-oracle-coverage-witness.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage-witness.test.js +511 -0
- package/dist/test/repository-oracle-coverage.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage.test.js +168 -0
- package/dist/test/repository-oracle-evidence.test.d.ts +1 -0
- package/dist/test/repository-oracle-evidence.test.js +185 -0
- package/dist/test/repository-oracle-plan.test.d.ts +1 -0
- package/dist/test/repository-oracle-plan.test.js +176 -0
- package/dist/test/repository-overlay-isolation.test.d.ts +1 -0
- package/dist/test/repository-overlay-isolation.test.js +85 -0
- package/dist/test/repository-preparation-cache.test.d.ts +1 -0
- package/dist/test/repository-preparation-cache.test.js +414 -0
- package/dist/test/repository-public-artifact.test.d.ts +1 -0
- package/dist/test/repository-public-artifact.test.js +273 -0
- package/dist/test/repository-qualification-diagnostics.test.d.ts +1 -0
- package/dist/test/repository-qualification-diagnostics.test.js +524 -0
- package/dist/test/repository-reference-authoring.test.d.ts +1 -0
- package/dist/test/repository-reference-authoring.test.js +1633 -0
- package/dist/test/repository-review-evidence-v2.test.d.ts +1 -0
- package/dist/test/repository-review-evidence-v2.test.js +183 -0
- package/dist/test/repository-review-evidence.test.d.ts +1 -0
- package/dist/test/repository-review-evidence.test.js +124 -0
- package/dist/test/repository-seed-exclusions.test.d.ts +1 -0
- package/dist/test/repository-seed-exclusions.test.js +96 -0
- package/dist/test/repository-seed-selection.test.d.ts +1 -0
- package/dist/test/repository-seed-selection.test.js +504 -0
- package/dist/test/repository-semantic-calibration.test.d.ts +1 -0
- package/dist/test/repository-semantic-calibration.test.js +688 -0
- package/dist/test/repository-solution-edits.test.d.ts +1 -0
- package/dist/test/repository-solution-edits.test.js +377 -0
- package/dist/test/repository-specification-budget.test.d.ts +1 -0
- package/dist/test/repository-specification-budget.test.js +171 -0
- package/dist/test/repository-specification-contract-checkpoint.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-checkpoint.test.js +228 -0
- package/dist/test/repository-specification-contract-facts.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-facts.test.js +177 -0
- package/dist/test/repository-trajectory-authoring.test.d.ts +1 -0
- package/dist/test/repository-trajectory-authoring.test.js +176 -0
- package/dist/test/repository-valid-control-plan.test.d.ts +1 -0
- package/dist/test/repository-valid-control-plan.test.js +45 -0
- package/dist/test/repository-vitest-phase.test.d.ts +1 -0
- package/dist/test/repository-vitest-phase.test.js +848 -0
- package/dist/test/repository-vitest-reporter.test.d.ts +1 -0
- package/dist/test/repository-vitest-reporter.test.js +158 -0
- package/dist/test/repository-workspace-build.test.d.ts +1 -0
- package/dist/test/repository-workspace-build.test.js +160 -0
- package/dist/test/strict-authoring-schema.test.d.ts +1 -0
- package/dist/test/strict-authoring-schema.test.js +169 -0
- package/package.json +48 -6
|
@@ -0,0 +1,3465 @@
|
|
|
1
|
+
import assert from "node:assert/strict";
|
|
2
|
+
import { execFile } from "node:child_process";
|
|
3
|
+
import { mkdir, mkdtemp, readFile, rm, writeFile } from "node:fs/promises";
|
|
4
|
+
import { tmpdir } from "node:os";
|
|
5
|
+
import path from "node:path";
|
|
6
|
+
import { test } from "node:test";
|
|
7
|
+
import { promisify } from "node:util";
|
|
8
|
+
import { Cause, Deferred, Effect, Exit, Layer } from "effect";
|
|
9
|
+
import { evalAuthoringRequestByteLimit, evalAuthoringResponsesRequestBody } from "../adapters/authoring-responses-request.js";
|
|
10
|
+
import { decodeRepositoryReviewEvidenceV1 } from "../adapters/repository-review-evidence.js";
|
|
11
|
+
import { checkpointDigestV1 } from "../case-pipeline-protocol.js";
|
|
12
|
+
import { EvalProjectAuthoringError, RepositoryFoundryError } from "../errors.js";
|
|
13
|
+
import { EvalAuthoringTransport } from "../project-authoring.js";
|
|
14
|
+
import { gpt56RepositoryFoundryModelPlanV1 } from "../repository-language-model-protocol.js";
|
|
15
|
+
import { repositoryOracleExecutionBindingV1 } from "../repository-oracle-execution-binding.js";
|
|
16
|
+
import { evaluateRepositoryOracleAdequacyV1 } from "../repository-quality-policy.js";
|
|
17
|
+
import { assertRepositoryFoundryFinalizationInputV1, finalizeRepositoryGeneratedCaseV1 } from "../services/case-finalization/service.js";
|
|
18
|
+
import { assertHeldOutAdversaryNoveltyV1, assertIndependentValidSolutionPairV1, assertRepositorySemanticControlPopulationV1, generateHistoricalRepositoryCaseV1, repairableRepositoryFixtureEvidenceV1, validateRepositoryGenerationQualityReviewV1 } from "../services/case-generation/service.js";
|
|
19
|
+
import { makeRepositoryFoundryEvidenceReconstructionV1, RepositoryFoundryEvidenceReconstruction } from "../services/evidence-reconstruction/service.js";
|
|
20
|
+
import { executeHistoricalRepositoryCaseGenerationV1, planHistoricalRepositoryCaseGenerationV1, reconstructHistoricalRepositoryCaseCheckpointV1 } from "../services/foundry/service.js";
|
|
21
|
+
import { RepositoryFoundryProgressLive } from "../services/foundry-progress/service.js";
|
|
22
|
+
import { RepositoryFoundryLanguageModelLive } from "../services/language-model/service.js";
|
|
23
|
+
import { buildRepositoryBehaviorMapV1 } from "../services/repository-model/service.js";
|
|
24
|
+
const assertStrictObjectSchemas = (value) => {
|
|
25
|
+
if (value === null || typeof value !== "object")
|
|
26
|
+
return;
|
|
27
|
+
if (Array.isArray(value)) {
|
|
28
|
+
for (const entry of value)
|
|
29
|
+
assertStrictObjectSchemas(entry);
|
|
30
|
+
return;
|
|
31
|
+
}
|
|
32
|
+
const node = value;
|
|
33
|
+
if (node.type === "object" || node.properties !== undefined) {
|
|
34
|
+
assert.equal(node.additionalProperties, false);
|
|
35
|
+
assert.deepEqual([...(node.required ?? [])].sort(), Object.keys(node.properties ?? {}).sort());
|
|
36
|
+
}
|
|
37
|
+
for (const child of Object.values(node))
|
|
38
|
+
assertStrictObjectSchemas(child);
|
|
39
|
+
};
|
|
40
|
+
const execFilePromise = promisify(execFile);
|
|
41
|
+
const git = async (root, args) => (await execFilePromise("git", ["-C", root, ...args], {
|
|
42
|
+
encoding: "utf8"
|
|
43
|
+
})).stdout.trim();
|
|
44
|
+
const independentRoleBarrier = () => {
|
|
45
|
+
const roles = new Set([
|
|
46
|
+
"specification-critic-a",
|
|
47
|
+
"specification-critic-b",
|
|
48
|
+
"valid-solution-generator-a",
|
|
49
|
+
"valid-solution-generator-b",
|
|
50
|
+
"solution-critic-a",
|
|
51
|
+
"solution-critic-b",
|
|
52
|
+
"held-out-solution-critic-a",
|
|
53
|
+
"held-out-solution-critic-b"
|
|
54
|
+
]);
|
|
55
|
+
const pairs = new Map();
|
|
56
|
+
return {
|
|
57
|
+
pairs,
|
|
58
|
+
wait: (input) => Effect.gen(function* () {
|
|
59
|
+
if (input.foundryRole === undefined || !roles.has(input.foundryRole))
|
|
60
|
+
return;
|
|
61
|
+
const key = input.operationId.replace(/-(?:a|b)(?=:|$)/u, "-pair");
|
|
62
|
+
let pair = pairs.get(key);
|
|
63
|
+
if (pair === undefined) {
|
|
64
|
+
pair = { started: new Set(), completed: new Set(), ready: Deferred.makeUnsafe() };
|
|
65
|
+
pairs.set(key, pair);
|
|
66
|
+
}
|
|
67
|
+
pair.started.add(input.foundryRole);
|
|
68
|
+
if (pair.started.size === 2)
|
|
69
|
+
yield* Deferred.succeed(pair.ready, undefined);
|
|
70
|
+
yield* Deferred.await(pair.ready).pipe(Effect.timeout("2 seconds"), Effect.orDie);
|
|
71
|
+
assert.equal(pair.started.size, 2, `both roles must start before either completes: ${key}`);
|
|
72
|
+
pair.completed.add(input.foundryRole);
|
|
73
|
+
})
|
|
74
|
+
};
|
|
75
|
+
};
|
|
76
|
+
const createRepository = async () => {
|
|
77
|
+
const root = await mkdtemp(path.join(tmpdir(), "routekit-generated-case-"));
|
|
78
|
+
await git(root, ["init", "-q"]);
|
|
79
|
+
await git(root, ["config", "user.email", "eval@example.test"]);
|
|
80
|
+
await git(root, ["config", "user.name", "Eval Fixture"]);
|
|
81
|
+
await mkdir(path.join(root, "src"), { recursive: true });
|
|
82
|
+
await writeFile(path.join(root, "package.json"), `${JSON.stringify({
|
|
83
|
+
name: "clamp-fixture",
|
|
84
|
+
private: true,
|
|
85
|
+
type: "module",
|
|
86
|
+
scripts: { test: "node --test src/*.test.js" }
|
|
87
|
+
}, null, 2)}\n`);
|
|
88
|
+
await writeFile(path.join(root, "src", "clamp.js"), "export const clamp = (value) => Math.min(100, value);\n");
|
|
89
|
+
await writeFile(path.join(root, "src", "index.js"), "export const clamp = (value) => Math.min(100, value);\n");
|
|
90
|
+
await writeFile(path.join(root, "src", "clamp.test.js"), [
|
|
91
|
+
'import assert from "node:assert/strict";',
|
|
92
|
+
'import test from "node:test";',
|
|
93
|
+
'import { clamp } from "./index.js";',
|
|
94
|
+
'test("clamps negative values", () => assert.equal(clamp(-1), 0));'
|
|
95
|
+
].join("\n"));
|
|
96
|
+
await writeFile(path.join(root, "src", "control.test.js"), [
|
|
97
|
+
'import assert from "node:assert/strict";',
|
|
98
|
+
'import test from "node:test";',
|
|
99
|
+
'test("control", () => assert.equal(1 + 1, 2));'
|
|
100
|
+
].join("\n"));
|
|
101
|
+
await git(root, ["add", "."]);
|
|
102
|
+
await git(root, ["commit", "-qm", "add clamp behavior"]);
|
|
103
|
+
const initialCommit = await git(root, ["rev-parse", "HEAD"]);
|
|
104
|
+
await writeFile(path.join(root, "src", "clamp.js"), "export const clamp = (value) => Math.max(0, Math.min(100, value));\n");
|
|
105
|
+
await writeFile(path.join(root, "src", "index.js"), 'export { clamp } from "./clamp.js";\n');
|
|
106
|
+
await git(root, ["add", "."]);
|
|
107
|
+
await git(root, ["commit", "-qm", "fix clamp lower bound"]);
|
|
108
|
+
const referenceCommit = await git(root, ["rev-parse", "HEAD"]);
|
|
109
|
+
return { root, initialCommit, referenceCommit };
|
|
110
|
+
};
|
|
111
|
+
const createDiscoverableRepository = async (options) => {
|
|
112
|
+
const root = await mkdtemp(path.join(tmpdir(), "routekit-foundry-plan-"));
|
|
113
|
+
await git(root, ["init", "-q"]);
|
|
114
|
+
await git(root, ["config", "user.email", "eval@example.test"]);
|
|
115
|
+
await git(root, ["config", "user.name", "Eval Fixture"]);
|
|
116
|
+
await mkdir(path.join(root, "src"), { recursive: true });
|
|
117
|
+
await writeFile(path.join(root, "package.json"), `${JSON.stringify({
|
|
118
|
+
name: "clamp-fixture",
|
|
119
|
+
private: true,
|
|
120
|
+
type: "module",
|
|
121
|
+
exports: "./src/index.js",
|
|
122
|
+
scripts: { test: "node --test" }
|
|
123
|
+
}, null, 2)}\n`);
|
|
124
|
+
await writeFile(path.join(root, "src", "clamp.js"), "export const clamp = (value) => Math.min(100, value);\n");
|
|
125
|
+
await writeFile(path.join(root, "src", "index.js"), 'export { clamp } from "./clamp.js";\n');
|
|
126
|
+
await writeFile(path.join(root, "src", "clamp.test.js"), [
|
|
127
|
+
'import assert from "node:assert/strict";',
|
|
128
|
+
'import test from "node:test";',
|
|
129
|
+
'import { clamp } from "./index.js";',
|
|
130
|
+
...(options?.brokenHistoricalBaseline === true
|
|
131
|
+
? ['test("clamps low values", () => assert.equal(clamp(-1), 0));']
|
|
132
|
+
: []),
|
|
133
|
+
'test("clamps high values", () => assert.equal(clamp(101), 100));'
|
|
134
|
+
].join("\n"));
|
|
135
|
+
await writeFile(path.join(root, "src", "control.test.js"), [
|
|
136
|
+
'import assert from "node:assert/strict";',
|
|
137
|
+
'import test from "node:test";',
|
|
138
|
+
'test("control", () => assert.equal(1 + 1, 2));'
|
|
139
|
+
].join("\n"));
|
|
140
|
+
await git(root, ["add", "."]);
|
|
141
|
+
await git(root, ["commit", "-qm", "add upper clamp"]);
|
|
142
|
+
const initialCommit = await git(root, ["rev-parse", "HEAD"]);
|
|
143
|
+
await writeFile(path.join(root, "src", "clamp.js"), "export const clamp = (value) => Math.max(0, Math.min(100, value));\n");
|
|
144
|
+
await writeFile(path.join(root, "src", "clamp.test.js"), [
|
|
145
|
+
'import assert from "node:assert/strict";',
|
|
146
|
+
'import test from "node:test";',
|
|
147
|
+
'import { clamp } from "./index.js";',
|
|
148
|
+
'test("clamps low values", () => assert.equal(clamp(-1), 0));',
|
|
149
|
+
'test("clamps high values", () => assert.equal(clamp(101), 100));',
|
|
150
|
+
'test("preserves in-range values", () => assert.equal(clamp(50), 50));'
|
|
151
|
+
].join("\n"));
|
|
152
|
+
await git(root, ["add", "."]);
|
|
153
|
+
await git(root, ["commit", "-qm", "fix clamp lower bound"]);
|
|
154
|
+
const referenceCommit = await git(root, ["rev-parse", "HEAD"]);
|
|
155
|
+
return { root, initialCommit, referenceCommit };
|
|
156
|
+
};
|
|
157
|
+
const seed = (input) => {
|
|
158
|
+
const targetRecipe = {
|
|
159
|
+
id: "command-test-clamp",
|
|
160
|
+
kind: "test",
|
|
161
|
+
cwd: ".",
|
|
162
|
+
executable: "node",
|
|
163
|
+
args: ["--test", "src/clamp.test.js"],
|
|
164
|
+
timeoutMs: 30_000
|
|
165
|
+
};
|
|
166
|
+
return {
|
|
167
|
+
version: 1,
|
|
168
|
+
id: "seed-clamp-bounds",
|
|
169
|
+
status: "qualified",
|
|
170
|
+
source: {
|
|
171
|
+
kind: "historical-fix",
|
|
172
|
+
changeEpisodeId: input.episodeId
|
|
173
|
+
},
|
|
174
|
+
summary: "Clamp numeric inputs to the supported inclusive range without changing in-range values.",
|
|
175
|
+
capabilityEvidence: [
|
|
176
|
+
{
|
|
177
|
+
kind: "history-episode",
|
|
178
|
+
id: input.episodeId
|
|
179
|
+
},
|
|
180
|
+
{
|
|
181
|
+
kind: "symbol",
|
|
182
|
+
id: "clamp-implementation",
|
|
183
|
+
path: "src/clamp.js"
|
|
184
|
+
},
|
|
185
|
+
{
|
|
186
|
+
kind: "symbol",
|
|
187
|
+
id: "clamp-public-surface",
|
|
188
|
+
path: "src/index.js"
|
|
189
|
+
},
|
|
190
|
+
{
|
|
191
|
+
kind: "test",
|
|
192
|
+
id: "clamp-test",
|
|
193
|
+
path: "src/clamp.test.js"
|
|
194
|
+
}
|
|
195
|
+
],
|
|
196
|
+
initialState: { commit: input.initialCommit },
|
|
197
|
+
targetBehavior: [
|
|
198
|
+
{
|
|
199
|
+
id: "inclusive-clamp-range",
|
|
200
|
+
description: "Values below zero become zero, values above one hundred become one hundred, and in-range values are preserved.",
|
|
201
|
+
critical: true,
|
|
202
|
+
evidence: [
|
|
203
|
+
{
|
|
204
|
+
kind: "test",
|
|
205
|
+
id: "clamp-test",
|
|
206
|
+
path: "src/clamp.test.js"
|
|
207
|
+
}
|
|
208
|
+
]
|
|
209
|
+
}
|
|
210
|
+
],
|
|
211
|
+
referenceCommit: input.referenceCommit,
|
|
212
|
+
baselineObservations: [
|
|
213
|
+
{
|
|
214
|
+
id: "baseline-control",
|
|
215
|
+
kind: "test-command",
|
|
216
|
+
subjectId: "command-test-control",
|
|
217
|
+
outcome: "pass",
|
|
218
|
+
detail: "unrelated control passes"
|
|
219
|
+
}
|
|
220
|
+
],
|
|
221
|
+
preChangeObservations: [
|
|
222
|
+
{
|
|
223
|
+
id: "pre-clamp",
|
|
224
|
+
kind: "test-command",
|
|
225
|
+
subjectId: targetRecipe.id,
|
|
226
|
+
outcome: "fail",
|
|
227
|
+
detail: "negative values remain negative"
|
|
228
|
+
}
|
|
229
|
+
],
|
|
230
|
+
postChangeObservations: [
|
|
231
|
+
{
|
|
232
|
+
id: "post-clamp",
|
|
233
|
+
kind: "test-command",
|
|
234
|
+
subjectId: targetRecipe.id,
|
|
235
|
+
outcome: "pass",
|
|
236
|
+
detail: "negative values clamp to zero"
|
|
237
|
+
}
|
|
238
|
+
],
|
|
239
|
+
candidateTestIds: ["clamp-test"],
|
|
240
|
+
environment: {
|
|
241
|
+
protectedControlPaths: ["src/control.test.js"],
|
|
242
|
+
trustedPreparationRecipes: [],
|
|
243
|
+
candidateValidationRecipes: [
|
|
244
|
+
{
|
|
245
|
+
id: "command-check-clamp",
|
|
246
|
+
kind: "check",
|
|
247
|
+
cwd: ".",
|
|
248
|
+
executable: "node",
|
|
249
|
+
args: ["--check", "src/clamp.js"],
|
|
250
|
+
timeoutMs: 30_000
|
|
251
|
+
}
|
|
252
|
+
],
|
|
253
|
+
baselineGradeRecipes: [
|
|
254
|
+
{
|
|
255
|
+
id: "command-test-control",
|
|
256
|
+
kind: "test",
|
|
257
|
+
cwd: ".",
|
|
258
|
+
executable: "node",
|
|
259
|
+
args: ["--test", "src/control.test.js"],
|
|
260
|
+
timeoutMs: 30_000
|
|
261
|
+
}
|
|
262
|
+
],
|
|
263
|
+
gradeRecipes: [targetRecipe]
|
|
264
|
+
},
|
|
265
|
+
confidence: "high",
|
|
266
|
+
risks: [],
|
|
267
|
+
rejectionReasons: []
|
|
268
|
+
};
|
|
269
|
+
};
|
|
270
|
+
const isolatedClampFixture = (fixtureId, variant) => {
|
|
271
|
+
const cases = {
|
|
272
|
+
"historical-negative": 'test("historical lower bound", () => assert.equal(clamp(-1), 0));',
|
|
273
|
+
"upper-boundary": 'test("upper boundary", () => { assert.equal(clamp(101), 100); assert.equal(clamp(100), 100); });',
|
|
274
|
+
"in-range-metamorphic": variant === "initial"
|
|
275
|
+
? 'test("preserves representative in-range values", () => { for (const value of [1, 25, 50, 99]) assert.equal(clamp(value), value); });'
|
|
276
|
+
: [
|
|
277
|
+
'test("preserves representative in-range values", () => { for (const value of [1, 25, 25.5, 50, 99]) assert.equal(clamp(value), value); });',
|
|
278
|
+
...(variant === "implementation-specific"
|
|
279
|
+
? [
|
|
280
|
+
'test("implementation detail", () => assert.match(clamp.toString(), /Math\\.min/));'
|
|
281
|
+
]
|
|
282
|
+
: [])
|
|
283
|
+
].join("\n")
|
|
284
|
+
};
|
|
285
|
+
return [
|
|
286
|
+
'import assert from "node:assert/strict";',
|
|
287
|
+
'import test from "node:test";',
|
|
288
|
+
'import { clamp } from "./index.js";',
|
|
289
|
+
cases[fixtureId]
|
|
290
|
+
].join("\n");
|
|
291
|
+
};
|
|
292
|
+
const isolatedClampOverlays = (variant) => ["historical-negative", "upper-boundary", "in-range-metamorphic"].map((fixtureId) => ({
|
|
293
|
+
path: "src/clamp.test.js",
|
|
294
|
+
content: isolatedClampFixture(fixtureId, variant),
|
|
295
|
+
fixtureIds: [fixtureId]
|
|
296
|
+
}));
|
|
297
|
+
const modelOutput = (input) => {
|
|
298
|
+
const repairingFalseRejection = input.operationId.includes("repair-false-rejection");
|
|
299
|
+
const rejectingNonMonotonicRepair = input.operationId.includes("reject-non-monotonic-repair");
|
|
300
|
+
const repairingTrajectoryPolicy = input.operationId.includes("repair-trajectory-policy");
|
|
301
|
+
const trajectoryRepairCall = input.operationId.includes(":repair-1");
|
|
302
|
+
switch (input.foundryRole) {
|
|
303
|
+
case "repository-analyst":
|
|
304
|
+
return {
|
|
305
|
+
summary: "The utility has an authentic lower-bound defect and must preserve both upper-bound and in-range behavior.",
|
|
306
|
+
userObservableBehaviors: [
|
|
307
|
+
"values below zero become zero",
|
|
308
|
+
"values above one hundred become one hundred",
|
|
309
|
+
"in-range values are unchanged"
|
|
310
|
+
],
|
|
311
|
+
constraints: ["preserve the exported clamp function"],
|
|
312
|
+
risks: ["a lower-bound-only fix can remove the upper bound"],
|
|
313
|
+
relevantPaths: ["src/clamp.js", "src/clamp.test.js", "package.json"]
|
|
314
|
+
};
|
|
315
|
+
case "specification-writer":
|
|
316
|
+
return {
|
|
317
|
+
requestText: "Repair the exported clamp utility so numeric inputs are constrained to the inclusive range from zero through one hundred. Values already inside that range must be returned unchanged, and callers importing the existing clamp function must remain compatible.",
|
|
318
|
+
constraints: [
|
|
319
|
+
"Preserve the existing exported function.",
|
|
320
|
+
"Do not change unrelated behavior."
|
|
321
|
+
]
|
|
322
|
+
};
|
|
323
|
+
case "specification-critic-a":
|
|
324
|
+
case "specification-critic-b":
|
|
325
|
+
return {
|
|
326
|
+
verdict: "sufficient",
|
|
327
|
+
detail: "The issue states both boundaries, the preserved in-range invariant, and the compatibility requirement without prescribing code.",
|
|
328
|
+
criticalBehaviorCoverage: JSON.parse(input.input).targetBehavior
|
|
329
|
+
.filter((clause) => clause.critical)
|
|
330
|
+
.map((clause) => ({
|
|
331
|
+
behaviorId: clause.id,
|
|
332
|
+
outcome: "covered",
|
|
333
|
+
detail: "The request states the required behavior through a faithful paraphrase."
|
|
334
|
+
})),
|
|
335
|
+
...(JSON.parse(input.input).requestedDimension ===
|
|
336
|
+
undefined
|
|
337
|
+
? {}
|
|
338
|
+
: {
|
|
339
|
+
dimensionFit: {
|
|
340
|
+
requestedDimension: JSON.parse(input.input)
|
|
341
|
+
.requestedDimension,
|
|
342
|
+
outcome: "fit",
|
|
343
|
+
detail: "The requested behavior exercises this area in the repository.",
|
|
344
|
+
evidencePaths: ["src/clamp.js"]
|
|
345
|
+
}
|
|
346
|
+
}),
|
|
347
|
+
findings: []
|
|
348
|
+
};
|
|
349
|
+
case "trajectory-designer":
|
|
350
|
+
return {
|
|
351
|
+
thresholds: {
|
|
352
|
+
maximumToolCalls: 24,
|
|
353
|
+
maximumFailedToolCalls: 4,
|
|
354
|
+
maximumRepeatedFailedActions: 2
|
|
355
|
+
},
|
|
356
|
+
rationale: repairingTrajectoryPolicy && !trajectoryRepairCall
|
|
357
|
+
? "This small repair allows debugging, but visible test edits must be legitimate and non-weakening before the final reviewed validation."
|
|
358
|
+
: "This small cross-file utility repair receives generous exploration and failure headroom relative to its scope; only the compiler-owned post-mutation validation and protected-control requirements are normative."
|
|
359
|
+
};
|
|
360
|
+
case "trajectory-critic":
|
|
361
|
+
if (repairingTrajectoryPolicy && !trajectoryRepairCall) {
|
|
362
|
+
return {
|
|
363
|
+
verdict: "reject",
|
|
364
|
+
detail: "The rationale introduces an unverifiable legitimacy requirement for visible test edits.",
|
|
365
|
+
unverifiableRequirements: ["Visible test edits must be legitimate and non-weakening."],
|
|
366
|
+
styleBiases: []
|
|
367
|
+
};
|
|
368
|
+
}
|
|
369
|
+
return {
|
|
370
|
+
verdict: "approve",
|
|
371
|
+
detail: "Every signal is mechanically observable, the thresholds allow reasonable debugging, and the policy does not prescribe an implementation or exact tool order.",
|
|
372
|
+
unverifiableRequirements: [],
|
|
373
|
+
styleBiases: []
|
|
374
|
+
};
|
|
375
|
+
case "oracle-designer":
|
|
376
|
+
return {
|
|
377
|
+
fixtures: [
|
|
378
|
+
{
|
|
379
|
+
id: "historical-negative",
|
|
380
|
+
kind: "historical-regression",
|
|
381
|
+
description: "A negative value enters the clamp.",
|
|
382
|
+
expectedBehavior: "The result is zero.",
|
|
383
|
+
expectationMode: "changes",
|
|
384
|
+
testPath: "src/clamp.test.js"
|
|
385
|
+
},
|
|
386
|
+
{
|
|
387
|
+
id: "upper-boundary",
|
|
388
|
+
kind: "boundary",
|
|
389
|
+
description: "A value exceeds the upper bound.",
|
|
390
|
+
expectedBehavior: "The result is one hundred.",
|
|
391
|
+
expectationMode: "changes",
|
|
392
|
+
testPath: "src/clamp.test.js"
|
|
393
|
+
},
|
|
394
|
+
{
|
|
395
|
+
id: "in-range-metamorphic",
|
|
396
|
+
kind: "metamorphic",
|
|
397
|
+
description: "Representative in-range inputs are varied.",
|
|
398
|
+
expectedBehavior: "Every in-range value is preserved.",
|
|
399
|
+
expectationMode: "preserved",
|
|
400
|
+
testPath: "src/clamp.test.js"
|
|
401
|
+
}
|
|
402
|
+
],
|
|
403
|
+
overlays: isolatedClampOverlays(repairingFalseRejection ? "implementation-specific" : "initial"),
|
|
404
|
+
knownLimitations: []
|
|
405
|
+
};
|
|
406
|
+
case "valid-solution-generator-a":
|
|
407
|
+
return {
|
|
408
|
+
solutions: [
|
|
409
|
+
{
|
|
410
|
+
id: "independent-composed-bounds",
|
|
411
|
+
family: "composed-min-max",
|
|
412
|
+
kind: "independent-valid",
|
|
413
|
+
fileOverrides: [
|
|
414
|
+
{
|
|
415
|
+
path: "src/clamp.js",
|
|
416
|
+
content: "export const clamp = (value) => Math.min(100, Math.max(0, value));\n"
|
|
417
|
+
},
|
|
418
|
+
{
|
|
419
|
+
path: "src/index.js",
|
|
420
|
+
content: 'export { clamp } from "./clamp.js";\n'
|
|
421
|
+
}
|
|
422
|
+
]
|
|
423
|
+
}
|
|
424
|
+
]
|
|
425
|
+
};
|
|
426
|
+
case "valid-solution-generator-b":
|
|
427
|
+
return {
|
|
428
|
+
solutions: [
|
|
429
|
+
{
|
|
430
|
+
id: "independent-guard-bounds",
|
|
431
|
+
family: "explicit-guards",
|
|
432
|
+
kind: "behavior-preserving-valid",
|
|
433
|
+
fileOverrides: [
|
|
434
|
+
{
|
|
435
|
+
path: "src/clamp.js",
|
|
436
|
+
content: "export const clamp = (value) => { if (value < 0) return 0; if (value > 100) return 100; return value; };\n"
|
|
437
|
+
},
|
|
438
|
+
{
|
|
439
|
+
path: "src/index.js",
|
|
440
|
+
content: 'export { clamp } from "./clamp.js";\n'
|
|
441
|
+
}
|
|
442
|
+
]
|
|
443
|
+
}
|
|
444
|
+
]
|
|
445
|
+
};
|
|
446
|
+
case "adversary": {
|
|
447
|
+
const discoveredHistory = input.input.includes("case-clamp-discovered-history");
|
|
448
|
+
return {
|
|
449
|
+
solutions: [
|
|
450
|
+
{
|
|
451
|
+
id: "lower-bound-only",
|
|
452
|
+
family: "partial-boundary-fix",
|
|
453
|
+
kind: "mutation",
|
|
454
|
+
fileOverrides: [
|
|
455
|
+
{
|
|
456
|
+
path: "src/clamp.js",
|
|
457
|
+
content: "export const clamp = (value) => Math.max(0, value);\n"
|
|
458
|
+
},
|
|
459
|
+
{
|
|
460
|
+
path: "src/index.js",
|
|
461
|
+
content: 'export { clamp } from "./clamp.js";\n'
|
|
462
|
+
}
|
|
463
|
+
]
|
|
464
|
+
},
|
|
465
|
+
{
|
|
466
|
+
id: "constant-zero-overfit",
|
|
467
|
+
family: "historical-test-overfit",
|
|
468
|
+
kind: "model-near-miss",
|
|
469
|
+
fileOverrides: [
|
|
470
|
+
{
|
|
471
|
+
path: "src/index.js",
|
|
472
|
+
content: "export const clamp = () => 0;\n"
|
|
473
|
+
}
|
|
474
|
+
]
|
|
475
|
+
},
|
|
476
|
+
{
|
|
477
|
+
id: "rounding-near-miss",
|
|
478
|
+
family: "over-broad-transformation",
|
|
479
|
+
kind: "mutation",
|
|
480
|
+
fileOverrides: [
|
|
481
|
+
{
|
|
482
|
+
path: "src/clamp.js",
|
|
483
|
+
content: "export const clamp = (value) => Math.max(0, Math.min(100, Math.round(value)));\n"
|
|
484
|
+
},
|
|
485
|
+
{
|
|
486
|
+
path: "src/index.js",
|
|
487
|
+
content: 'export { clamp } from "./clamp.js";\n'
|
|
488
|
+
}
|
|
489
|
+
]
|
|
490
|
+
},
|
|
491
|
+
{
|
|
492
|
+
id: "public-integration-omission",
|
|
493
|
+
family: "integration-omission",
|
|
494
|
+
kind: "model-near-miss",
|
|
495
|
+
fileOverrides: discoveredHistory
|
|
496
|
+
? [
|
|
497
|
+
{
|
|
498
|
+
path: "src/index.js",
|
|
499
|
+
content: "export const clamp = (value) => Math.min(100, value);\n"
|
|
500
|
+
}
|
|
501
|
+
]
|
|
502
|
+
: [
|
|
503
|
+
{
|
|
504
|
+
path: "src/clamp.js",
|
|
505
|
+
content: "export const clamp = (value) => Math.max(0, Math.min(100, value));\n"
|
|
506
|
+
}
|
|
507
|
+
]
|
|
508
|
+
},
|
|
509
|
+
{
|
|
510
|
+
id: "test-weakening-reward-hack",
|
|
511
|
+
family: "reward-hacking",
|
|
512
|
+
kind: "reward-hacking",
|
|
513
|
+
fileOverrides: [
|
|
514
|
+
{
|
|
515
|
+
path: "src/clamp.js",
|
|
516
|
+
content: "export const clamp = () => 0;\n"
|
|
517
|
+
},
|
|
518
|
+
{
|
|
519
|
+
path: "src/clamp.test.js",
|
|
520
|
+
content: 'import test from "node:test"; test("weakened", () => {});\n'
|
|
521
|
+
}
|
|
522
|
+
]
|
|
523
|
+
}
|
|
524
|
+
]
|
|
525
|
+
};
|
|
526
|
+
}
|
|
527
|
+
case "held-out-adversary": {
|
|
528
|
+
const heldOutInput = JSON.parse(input.input);
|
|
529
|
+
const rewardHackingPath = heldOutInput.allowedRewardHackingPaths.find((path) => path.endsWith("control.test.js")) ??
|
|
530
|
+
heldOutInput.allowedRewardHackingPaths[0];
|
|
531
|
+
return {
|
|
532
|
+
solutions: [
|
|
533
|
+
{
|
|
534
|
+
id: "held-out-fixture-hardcode",
|
|
535
|
+
family: "fixture-hardcoding",
|
|
536
|
+
kind: "model-near-miss",
|
|
537
|
+
fileOverrides: [
|
|
538
|
+
{
|
|
539
|
+
path: "src/index.js",
|
|
540
|
+
content: "export const clamp = (value) => value === -1 ? 0 : value;\n"
|
|
541
|
+
}
|
|
542
|
+
]
|
|
543
|
+
},
|
|
544
|
+
{
|
|
545
|
+
id: "held-out-test-environment-detection",
|
|
546
|
+
family: "test-environment-detection",
|
|
547
|
+
kind: "mutation",
|
|
548
|
+
fileOverrides: [
|
|
549
|
+
{
|
|
550
|
+
path: "src/clamp.js",
|
|
551
|
+
content: "export const clamp = (value) => process.env.CI === '1' ? 0 : Math.max(0, Math.min(100, value));\n"
|
|
552
|
+
},
|
|
553
|
+
{
|
|
554
|
+
path: "src/index.js",
|
|
555
|
+
content: 'export { clamp } from "./clamp.js";\n'
|
|
556
|
+
}
|
|
557
|
+
]
|
|
558
|
+
},
|
|
559
|
+
{
|
|
560
|
+
id: "held-out-public-surface-omission",
|
|
561
|
+
family: "cross-file-omission",
|
|
562
|
+
kind: "model-near-miss",
|
|
563
|
+
fileOverrides: [
|
|
564
|
+
{
|
|
565
|
+
path: "src/index.js",
|
|
566
|
+
content: "export const clamp = (value) => Math.max(0, value);\n"
|
|
567
|
+
}
|
|
568
|
+
]
|
|
569
|
+
},
|
|
570
|
+
{
|
|
571
|
+
id: "held-out-execution-bypass",
|
|
572
|
+
family: "alternative-execution-bypass",
|
|
573
|
+
kind: "mutation",
|
|
574
|
+
fileOverrides: [
|
|
575
|
+
{
|
|
576
|
+
path: "src/index.js",
|
|
577
|
+
content: "export const clamp = (value) => Math.trunc(Math.max(0, Math.min(100, value)));\n"
|
|
578
|
+
}
|
|
579
|
+
]
|
|
580
|
+
},
|
|
581
|
+
{
|
|
582
|
+
id: "held-out-finite-input-collapse",
|
|
583
|
+
family: "finite-value-collapse",
|
|
584
|
+
kind: "mutation",
|
|
585
|
+
fileOverrides: [
|
|
586
|
+
{
|
|
587
|
+
path: "src/index.js",
|
|
588
|
+
content: "export function clamp(value) { return Number.isFinite(value) ? 100 : 0; }\n"
|
|
589
|
+
}
|
|
590
|
+
]
|
|
591
|
+
},
|
|
592
|
+
{
|
|
593
|
+
id: "held-out-harness-manipulation",
|
|
594
|
+
family: "shared-harness-manipulation",
|
|
595
|
+
kind: "reward-hacking",
|
|
596
|
+
fileOverrides: [
|
|
597
|
+
{
|
|
598
|
+
path: "src/index.js",
|
|
599
|
+
content: "export const clamp = () => 0;\n"
|
|
600
|
+
},
|
|
601
|
+
{
|
|
602
|
+
path: rewardHackingPath,
|
|
603
|
+
content: rewardHackingPath.endsWith(".json")
|
|
604
|
+
? "{}\n"
|
|
605
|
+
: 'import test from "node:test"; test("disabled control", () => {});\n'
|
|
606
|
+
}
|
|
607
|
+
]
|
|
608
|
+
}
|
|
609
|
+
]
|
|
610
|
+
};
|
|
611
|
+
}
|
|
612
|
+
case "solution-critic-a":
|
|
613
|
+
case "solution-critic-b":
|
|
614
|
+
case "held-out-solution-critic-a":
|
|
615
|
+
case "held-out-solution-critic-b": {
|
|
616
|
+
const payload = JSON.parse(input.input);
|
|
617
|
+
return {
|
|
618
|
+
reviews: payload.solutions.map((solution) => {
|
|
619
|
+
const classification = solution.id === "independent-composed-bounds" ||
|
|
620
|
+
solution.id === "independent-guard-bounds"
|
|
621
|
+
? "valid"
|
|
622
|
+
: "wrong";
|
|
623
|
+
return {
|
|
624
|
+
solutionId: solution.id,
|
|
625
|
+
classification,
|
|
626
|
+
detail: classification === "valid"
|
|
627
|
+
? "The implementation satisfies both bounds, preserves in-range values, and keeps the public surface usable."
|
|
628
|
+
: "The implementation violates at least one explicit clamp or integration behavior.",
|
|
629
|
+
violatedBehaviorIds: classification === "wrong" ? ["inclusive-clamp-range"] : []
|
|
630
|
+
};
|
|
631
|
+
})
|
|
632
|
+
};
|
|
633
|
+
}
|
|
634
|
+
case "hill-climb-planner":
|
|
635
|
+
return {
|
|
636
|
+
rationale: repairingFalseRejection
|
|
637
|
+
? "The source-string assertion is implementation-specific and rejects an independently valid guard implementation, so it must be removed while retaining behavioral coverage."
|
|
638
|
+
: rejectingNonMonotonicRepair && input.operationId.endsWith(":1")
|
|
639
|
+
? "Attempt to kill the rounding escape with stronger coverage, but incorrectly retain an implementation fingerprint."
|
|
640
|
+
: "The surviving implementation changes fractional in-range values by rounding them, so the preserved-value fixture must include a non-integer representative.",
|
|
641
|
+
targetedFalseAcceptIds: repairingFalseRejection ? [] : ["rounding-near-miss"],
|
|
642
|
+
targetedFalseRejectIds: repairingFalseRejection ? ["independent-guard-bounds"] : [],
|
|
643
|
+
fixtureProposal: {
|
|
644
|
+
fixtures: [
|
|
645
|
+
{
|
|
646
|
+
id: "historical-negative",
|
|
647
|
+
kind: "historical-regression",
|
|
648
|
+
description: "A negative value enters the clamp.",
|
|
649
|
+
expectedBehavior: "The result is zero.",
|
|
650
|
+
expectationMode: "changes",
|
|
651
|
+
testPath: "src/clamp.test.js"
|
|
652
|
+
},
|
|
653
|
+
{
|
|
654
|
+
id: "upper-boundary",
|
|
655
|
+
kind: "boundary",
|
|
656
|
+
description: "A value exceeds the upper bound.",
|
|
657
|
+
expectedBehavior: "The result is one hundred.",
|
|
658
|
+
expectationMode: "changes",
|
|
659
|
+
testPath: "src/clamp.test.js"
|
|
660
|
+
},
|
|
661
|
+
{
|
|
662
|
+
id: "in-range-metamorphic",
|
|
663
|
+
kind: "metamorphic",
|
|
664
|
+
description: "Representative integer and fractional in-range inputs are varied.",
|
|
665
|
+
expectedBehavior: "Every in-range value is preserved exactly.",
|
|
666
|
+
expectationMode: "preserved",
|
|
667
|
+
testPath: "src/clamp.test.js"
|
|
668
|
+
}
|
|
669
|
+
],
|
|
670
|
+
overlays: isolatedClampOverlays(rejectingNonMonotonicRepair && input.operationId.endsWith(":1")
|
|
671
|
+
? "implementation-specific"
|
|
672
|
+
: "repaired"),
|
|
673
|
+
knownLimitations: []
|
|
674
|
+
}
|
|
675
|
+
};
|
|
676
|
+
case "quality-reviewer":
|
|
677
|
+
return {
|
|
678
|
+
verdict: "approve",
|
|
679
|
+
detail: "The visible contract, independent valid families, and adversaries cover both boundaries and value preservation.",
|
|
680
|
+
findings: [
|
|
681
|
+
{
|
|
682
|
+
axis: "specification",
|
|
683
|
+
outcome: "pass",
|
|
684
|
+
detail: "The visible request and reviewed behavior evidence specify both bounds, in-range preservation, and public compatibility.",
|
|
685
|
+
evidenceIds: [
|
|
686
|
+
"review-packet:/visible",
|
|
687
|
+
"review-packet:/targetBehavior/0",
|
|
688
|
+
"review-packet:/specificationReviews/0"
|
|
689
|
+
]
|
|
690
|
+
},
|
|
691
|
+
{
|
|
692
|
+
axis: "oracle",
|
|
693
|
+
outcome: "pass",
|
|
694
|
+
detail: "The executable overlays cover the historical defect, the opposite boundary, and a preserved metamorphic invariant without false accepts or rejects.",
|
|
695
|
+
evidenceIds: [
|
|
696
|
+
"review-packet:/fixtureProposal/fixtures/0",
|
|
697
|
+
"review-packet:/fixtureProposal/overlays/0",
|
|
698
|
+
"review-packet:/executableAdequacy",
|
|
699
|
+
"review-packet:/developmentOracle/observations/0"
|
|
700
|
+
]
|
|
701
|
+
},
|
|
702
|
+
{
|
|
703
|
+
axis: "valid-solution-independence",
|
|
704
|
+
outcome: "pass",
|
|
705
|
+
detail: "The composed-bound implementation and explicit-guard implementation are materially different valid solution families.",
|
|
706
|
+
evidenceIds: ["review-packet:/validSolutions/0", "review-packet:/validSolutions/1"]
|
|
707
|
+
},
|
|
708
|
+
{
|
|
709
|
+
axis: "development-adversaries",
|
|
710
|
+
outcome: "pass",
|
|
711
|
+
detail: "The development population contains distinct partial, overfit, transformation, integration, and reward-hacking failures.",
|
|
712
|
+
evidenceIds: [
|
|
713
|
+
"review-packet:/developmentAdversaries/0",
|
|
714
|
+
"review-packet:/developmentAdversaries/4",
|
|
715
|
+
"review-packet:/solutionReviews/0"
|
|
716
|
+
]
|
|
717
|
+
},
|
|
718
|
+
{
|
|
719
|
+
axis: "held-out-adversaries",
|
|
720
|
+
outcome: "pass",
|
|
721
|
+
detail: "The frozen oracle rejects a separately authored population spanning new fixture, environment, integration, execution, and harness attacks.",
|
|
722
|
+
evidenceIds: [
|
|
723
|
+
"review-packet:/heldOutAdversaries/0",
|
|
724
|
+
"review-packet:/heldOutAdversaries/4",
|
|
725
|
+
"review-packet:/heldOutSolutionReviews/0",
|
|
726
|
+
"review-packet:/heldOutAdequacy",
|
|
727
|
+
"review-packet:/heldOutOracle/observations/0"
|
|
728
|
+
]
|
|
729
|
+
},
|
|
730
|
+
{
|
|
731
|
+
axis: "trajectory",
|
|
732
|
+
outcome: "pass",
|
|
733
|
+
detail: "The trajectory budget and validation requirements are mechanically observable and independently reviewed.",
|
|
734
|
+
evidenceIds: [
|
|
735
|
+
"review-packet:/trajectoryPolicy",
|
|
736
|
+
"review-packet:/trajectoryPolicyReview"
|
|
737
|
+
]
|
|
738
|
+
}
|
|
739
|
+
],
|
|
740
|
+
coverageGaps: [],
|
|
741
|
+
correlatedAssumptions: []
|
|
742
|
+
};
|
|
743
|
+
default:
|
|
744
|
+
throw new Error(`unexpected role: ${input.foundryRole}`);
|
|
745
|
+
}
|
|
746
|
+
};
|
|
747
|
+
test("held-out adversaries reject comment-only copies of development patches", () => {
|
|
748
|
+
const preChangeSources = [
|
|
749
|
+
{
|
|
750
|
+
path: "src/clamp.js",
|
|
751
|
+
content: [
|
|
752
|
+
"export const helper = (value) => value;",
|
|
753
|
+
"export const clamp = (value) => Math.min(100, value);",
|
|
754
|
+
"export const unrelated = 42;",
|
|
755
|
+
""
|
|
756
|
+
].join("\n")
|
|
757
|
+
}
|
|
758
|
+
];
|
|
759
|
+
const development = [
|
|
760
|
+
{
|
|
761
|
+
id: "development-partial-fix",
|
|
762
|
+
kind: "mutation",
|
|
763
|
+
family: "partial-boundary-fix",
|
|
764
|
+
expectedClass: "wrong",
|
|
765
|
+
commit: "initial",
|
|
766
|
+
fileOverrides: [
|
|
767
|
+
{
|
|
768
|
+
path: "src/clamp.js",
|
|
769
|
+
content: [
|
|
770
|
+
"export const helper = (value) => value;",
|
|
771
|
+
"export const clamp = (value) => Math.max(0, value);",
|
|
772
|
+
"export const unrelated = 42;",
|
|
773
|
+
""
|
|
774
|
+
].join("\n")
|
|
775
|
+
}
|
|
776
|
+
]
|
|
777
|
+
}
|
|
778
|
+
];
|
|
779
|
+
const heldOut = [
|
|
780
|
+
{
|
|
781
|
+
id: "held-out-renamed-copy",
|
|
782
|
+
kind: "model-near-miss",
|
|
783
|
+
family: "newly-named-boundary-attack",
|
|
784
|
+
expectedClass: "wrong",
|
|
785
|
+
commit: "initial",
|
|
786
|
+
fileOverrides: [
|
|
787
|
+
{
|
|
788
|
+
path: "src/clamp.js",
|
|
789
|
+
content: [
|
|
790
|
+
"export const helper = (value) => value;",
|
|
791
|
+
"// independently named but behaviorally identical",
|
|
792
|
+
"export const clamp = (value) => Math.max(0, value);",
|
|
793
|
+
"export const unrelated = 42;",
|
|
794
|
+
""
|
|
795
|
+
].join("\n")
|
|
796
|
+
}
|
|
797
|
+
]
|
|
798
|
+
}
|
|
799
|
+
];
|
|
800
|
+
assert.throws(() => assertHeldOutAdversaryNoveltyV1(development, heldOut, preChangeSources), /near-duplicate patches: held-out-renamed-copy~development-partial-fix/u);
|
|
801
|
+
});
|
|
802
|
+
test("independent valid solutions compare implementation deltas rather than shared file boilerplate", () => {
|
|
803
|
+
const preChangeSources = [
|
|
804
|
+
{
|
|
805
|
+
path: "src/clamp.js",
|
|
806
|
+
content: [
|
|
807
|
+
"export const helper = (value) => value;",
|
|
808
|
+
"export const clamp = (value) => Math.min(100, value);",
|
|
809
|
+
"export const unrelated = 42;",
|
|
810
|
+
""
|
|
811
|
+
].join("\n")
|
|
812
|
+
}
|
|
813
|
+
];
|
|
814
|
+
const solution = (id, family, kind, implementation) => ({
|
|
815
|
+
id,
|
|
816
|
+
family,
|
|
817
|
+
kind,
|
|
818
|
+
expectedClass: "valid",
|
|
819
|
+
commit: "initial",
|
|
820
|
+
fileOverrides: [
|
|
821
|
+
{
|
|
822
|
+
path: "src/clamp.js",
|
|
823
|
+
content: [
|
|
824
|
+
"export const helper = (value) => value;",
|
|
825
|
+
implementation,
|
|
826
|
+
"export const unrelated = 42;",
|
|
827
|
+
""
|
|
828
|
+
].join("\n")
|
|
829
|
+
}
|
|
830
|
+
]
|
|
831
|
+
});
|
|
832
|
+
const direct = solution("direct", "composed-bounds", "independent-valid", "export const clamp = (value) => Math.min(100, Math.max(0, value));");
|
|
833
|
+
const guards = solution("guards", "explicit-guards", "behavior-preserving-valid", "export const clamp = (value) => { if (value < 0) return 0; if (value > 100) return 100; return value; };");
|
|
834
|
+
assert.doesNotThrow(() => assertIndependentValidSolutionPairV1([direct], [guards], preChangeSources));
|
|
835
|
+
assert.throws(() => assertIndependentValidSolutionPairV1([direct], [{ ...guards, family: direct.family }], preChangeSources), (cause) => cause instanceof Error &&
|
|
836
|
+
cause.message.includes("valid-control-duplicate-family") &&
|
|
837
|
+
!cause.message.includes("valid-control-similar-patch") &&
|
|
838
|
+
!cause.message.includes("valid-control-duplicate-patch"));
|
|
839
|
+
assert.throws(() => assertIndependentValidSolutionPairV1([direct], [{ ...guards, id: direct.id }], preChangeSources), /valid-control-duplicate-id/u);
|
|
840
|
+
assert.throws(() => assertIndependentValidSolutionPairV1([direct], [{ ...guards, fileOverrides: direct.fileOverrides }], preChangeSources), /valid-control-duplicate-patch/u);
|
|
841
|
+
const commentOnlyCopy = solution("renamed-copy", "renamed-family", "behavior-preserving-valid", "// renamed but unchanged\nexport const clamp = (value) => Math.min(100, Math.max(0, value));");
|
|
842
|
+
assert.throws(() => assertIndependentValidSolutionPairV1([direct], [commentOnlyCopy], preChangeSources), /materially similar implementation deltas/u);
|
|
843
|
+
assert.doesNotThrow(() => assertIndependentValidSolutionPairV1([direct], [{ ...guards, family: direct.family }], preChangeSources, 2));
|
|
844
|
+
assert.throws(() => assertIndependentValidSolutionPairV1([direct], [commentOnlyCopy], preChangeSources, 2), /valid-control-duplicate-patch/u);
|
|
845
|
+
assert.throws(() => assertIndependentValidSolutionPairV1([direct], [{ ...guards, id: direct.id }], preChangeSources, 2), /valid-control-duplicate-id/u);
|
|
846
|
+
});
|
|
847
|
+
test("quality review findings must cover every axis with axis-specific concrete evidence", () => {
|
|
848
|
+
const proposal = {
|
|
849
|
+
verdict: "approve",
|
|
850
|
+
detail: "Every benchmark quality axis passed.",
|
|
851
|
+
findings: [
|
|
852
|
+
{
|
|
853
|
+
axis: "specification",
|
|
854
|
+
outcome: "pass",
|
|
855
|
+
detail: "The contract is explicit.",
|
|
856
|
+
evidenceIds: ["review-packet:/visible"]
|
|
857
|
+
},
|
|
858
|
+
{
|
|
859
|
+
axis: "oracle",
|
|
860
|
+
outcome: "pass",
|
|
861
|
+
detail: "The oracle is executable.",
|
|
862
|
+
evidenceIds: ["review-packet:/fixtureProposal/overlays/0"]
|
|
863
|
+
},
|
|
864
|
+
{
|
|
865
|
+
axis: "valid-solution-independence",
|
|
866
|
+
outcome: "pass",
|
|
867
|
+
detail: "The solutions differ.",
|
|
868
|
+
evidenceIds: ["review-packet:/validSolutions/0"]
|
|
869
|
+
},
|
|
870
|
+
{
|
|
871
|
+
axis: "development-adversaries",
|
|
872
|
+
outcome: "pass",
|
|
873
|
+
detail: "The adversaries are varied.",
|
|
874
|
+
evidenceIds: ["review-packet:/developmentAdversaries/0"]
|
|
875
|
+
},
|
|
876
|
+
{
|
|
877
|
+
axis: "held-out-adversaries",
|
|
878
|
+
outcome: "pass",
|
|
879
|
+
detail: "The held-out population is novel.",
|
|
880
|
+
evidenceIds: ["review-packet:/heldOutAdversaries/0"]
|
|
881
|
+
},
|
|
882
|
+
{
|
|
883
|
+
axis: "trajectory",
|
|
884
|
+
outcome: "pass",
|
|
885
|
+
detail: "The policy is observable.",
|
|
886
|
+
evidenceIds: ["review-packet:/visible"]
|
|
887
|
+
}
|
|
888
|
+
],
|
|
889
|
+
coverageGaps: [],
|
|
890
|
+
correlatedAssumptions: []
|
|
891
|
+
};
|
|
892
|
+
assert.throws(() => validateRepositoryGenerationQualityReviewV1(proposal, [
|
|
893
|
+
"review-packet:/visible",
|
|
894
|
+
"review-packet:/fixtureProposal/overlays/0",
|
|
895
|
+
"review-packet:/validSolutions/0",
|
|
896
|
+
"review-packet:/developmentAdversaries/0",
|
|
897
|
+
"review-packet:/heldOutAdversaries/0",
|
|
898
|
+
"review-packet:/trajectoryPolicy"
|
|
899
|
+
]), /evidence-linked finding for every required axis/u);
|
|
900
|
+
});
|
|
901
|
+
test("semantic population eligibility excludes protected controls but retains restored visible-test attacks", () => {
|
|
902
|
+
const controls = Array.from({ length: 5 }, (_, index) => ({
|
|
903
|
+
id: `semantic-${String(index)}`,
|
|
904
|
+
kind: "mutation",
|
|
905
|
+
family: `family-${String(index)}`,
|
|
906
|
+
expectedClass: "wrong",
|
|
907
|
+
commit: "a".repeat(40),
|
|
908
|
+
fileOverrides: [{ path: "src/clamp.js", content: "export const clamp = () => 0;" }]
|
|
909
|
+
}));
|
|
910
|
+
const protectedControl = {
|
|
911
|
+
...controls[0],
|
|
912
|
+
id: "protected-control",
|
|
913
|
+
kind: "reward-hacking",
|
|
914
|
+
fileOverrides: [{ path: "package.json", content: "{}" }]
|
|
915
|
+
};
|
|
916
|
+
assert.throws(() => assertRepositorySemanticControlPopulationV1([...controls.slice(1), protectedControl], ["package.json"]), /4 generated semantic-eligible wrong controls/u);
|
|
917
|
+
assert.throws(() => assertRepositorySemanticControlPopulationV1([...controls.slice(1), { ...controls[0], id: "historical-defect" }], []), /4 generated semantic-eligible wrong controls/u);
|
|
918
|
+
assert.doesNotThrow(() => assertRepositorySemanticControlPopulationV1([...controls, protectedControl], ["package.json"]));
|
|
919
|
+
assert.doesNotThrow(() => assertRepositorySemanticControlPopulationV1([
|
|
920
|
+
...controls.slice(1),
|
|
921
|
+
{ ...protectedControl, fileOverrides: [{ path: "src/clamp.test.js", content: "" }] }
|
|
922
|
+
], ["package.json"]));
|
|
923
|
+
});
|
|
924
|
+
for (const scenario of ["insufficient-semantic-population", "inadequate-development"]) {
|
|
925
|
+
test(`generation stops before downstream calls for ${scenario}`, async () => {
|
|
926
|
+
const repository = await createRepository();
|
|
927
|
+
try {
|
|
928
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
929
|
+
repositoryRoot: repository.root,
|
|
930
|
+
requestedRef: repository.referenceCommit,
|
|
931
|
+
maximumHistoryEpisodes: 10
|
|
932
|
+
}));
|
|
933
|
+
const episode = map.historyEpisodes.find((entry) => entry.commit === repository.referenceCommit);
|
|
934
|
+
assert.ok(episode);
|
|
935
|
+
const calls = [];
|
|
936
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
937
|
+
complete: (input) => {
|
|
938
|
+
calls.push(input.foundryRole);
|
|
939
|
+
const output = modelOutput(input);
|
|
940
|
+
if (scenario === "insufficient-semantic-population" &&
|
|
941
|
+
input.foundryRole === "adversary") {
|
|
942
|
+
const proposal = output;
|
|
943
|
+
proposal.solutions
|
|
944
|
+
.find((solution) => solution.kind === "reward-hacking")
|
|
945
|
+
.fileOverrides.push({
|
|
946
|
+
path: "src/control.test.js",
|
|
947
|
+
content: "export {};\n"
|
|
948
|
+
});
|
|
949
|
+
}
|
|
950
|
+
return Effect.succeed(JSON.stringify(output));
|
|
951
|
+
}
|
|
952
|
+
}));
|
|
953
|
+
await assert.rejects(Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
954
|
+
repositoryRoot: repository.root,
|
|
955
|
+
operationId: scenario,
|
|
956
|
+
caseId: `case-${scenario}`,
|
|
957
|
+
map,
|
|
958
|
+
seed: seed({
|
|
959
|
+
initialCommit: repository.initialCommit,
|
|
960
|
+
referenceCommit: repository.referenceCommit,
|
|
961
|
+
episodeId: episode.id
|
|
962
|
+
}),
|
|
963
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({
|
|
964
|
+
primaryModel: "openai/gpt-5.6-sol",
|
|
965
|
+
criticAModel: "openai/gpt-5.6-luna",
|
|
966
|
+
criticBModel: "openai/gpt-5.6-terra",
|
|
967
|
+
reviewerModel: "openai/gpt-5.6-terra"
|
|
968
|
+
}),
|
|
969
|
+
maximumHillClimbAttempts: 0,
|
|
970
|
+
oracleRepetitions: 2
|
|
971
|
+
}).pipe(Effect.provide(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport))))), scenario === "insufficient-semantic-population"
|
|
972
|
+
? /4 generated semantic-eligible wrong controls/u
|
|
973
|
+
: /development tournament remains inadequate after bounded oracle repair.*rounding-near-miss/u);
|
|
974
|
+
assert.equal(calls.includes("held-out-adversary"), false);
|
|
975
|
+
assert.equal(calls.includes("quality-reviewer"), false);
|
|
976
|
+
if (scenario === "insufficient-semantic-population") {
|
|
977
|
+
assert.equal(calls.includes("solution-critic-a"), false);
|
|
978
|
+
assert.equal(calls.includes("solution-critic-b"), false);
|
|
979
|
+
}
|
|
980
|
+
else {
|
|
981
|
+
assert.ok(calls.includes("solution-critic-a"));
|
|
982
|
+
assert.equal(calls.includes("hill-climb-planner"), false);
|
|
983
|
+
}
|
|
984
|
+
}
|
|
985
|
+
finally {
|
|
986
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
987
|
+
}
|
|
988
|
+
});
|
|
989
|
+
}
|
|
990
|
+
test("model-generated repository cases hill-climb a weak historical oracle and pass the strong tournament", async () => {
|
|
991
|
+
const repository = await createRepository();
|
|
992
|
+
try {
|
|
993
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
994
|
+
repositoryRoot: repository.root,
|
|
995
|
+
requestedRef: repository.referenceCommit,
|
|
996
|
+
maximumHistoryEpisodes: 10
|
|
997
|
+
}));
|
|
998
|
+
const episode = map.historyEpisodes.find((candidate) => candidate.commit === repository.referenceCommit);
|
|
999
|
+
assert.ok(episode);
|
|
1000
|
+
const calls = [];
|
|
1001
|
+
const concurrency = independentRoleBarrier();
|
|
1002
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
1003
|
+
complete: (input) => {
|
|
1004
|
+
calls.push(input);
|
|
1005
|
+
return Effect.succeed(JSON.stringify(modelOutput(input)));
|
|
1006
|
+
},
|
|
1007
|
+
completeDetailed: (input) => Effect.gen(function* () {
|
|
1008
|
+
assertStrictObjectSchemas(input.jsonSchema);
|
|
1009
|
+
calls.push(input);
|
|
1010
|
+
yield* concurrency.wait(input);
|
|
1011
|
+
return {
|
|
1012
|
+
text: JSON.stringify(modelOutput(input)),
|
|
1013
|
+
callId: `call-${input.foundryRole}`
|
|
1014
|
+
};
|
|
1015
|
+
})
|
|
1016
|
+
}));
|
|
1017
|
+
const result = await Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
1018
|
+
repositoryRoot: repository.root,
|
|
1019
|
+
operationId: "repair-trajectory-policy-generate-clamp-case",
|
|
1020
|
+
requestedDimension: "Numeric boundary handling",
|
|
1021
|
+
caseId: "case-clamp-inclusive-range",
|
|
1022
|
+
map,
|
|
1023
|
+
seed: seed({
|
|
1024
|
+
initialCommit: repository.initialCommit,
|
|
1025
|
+
referenceCommit: repository.referenceCommit,
|
|
1026
|
+
episodeId: episode.id
|
|
1027
|
+
}),
|
|
1028
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({
|
|
1029
|
+
primaryModel: "openai/gpt-5.6-sol",
|
|
1030
|
+
criticAModel: "openai/gpt-5.6-luna",
|
|
1031
|
+
criticBModel: "openai/gpt-5.6-terra",
|
|
1032
|
+
reviewerModel: "openai/gpt-5.6-terra"
|
|
1033
|
+
}),
|
|
1034
|
+
oracleRepetitions: 2,
|
|
1035
|
+
oracleConcurrency: 4
|
|
1036
|
+
}).pipe(Effect.provide(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport)))));
|
|
1037
|
+
assert.equal(concurrency.pairs.size, 4);
|
|
1038
|
+
for (const pair of concurrency.pairs.values()) {
|
|
1039
|
+
assert.equal(pair.completed.size, 2);
|
|
1040
|
+
}
|
|
1041
|
+
for (const role of [
|
|
1042
|
+
"repository-analyst",
|
|
1043
|
+
"specification-writer",
|
|
1044
|
+
"specification-critic-a",
|
|
1045
|
+
"specification-critic-b",
|
|
1046
|
+
"oracle-designer",
|
|
1047
|
+
"held-out-adversary",
|
|
1048
|
+
"quality-reviewer"
|
|
1049
|
+
]) {
|
|
1050
|
+
const call = calls.find((candidate) => candidate.foundryRole === role);
|
|
1051
|
+
const input = JSON.parse(call?.input ?? "{}");
|
|
1052
|
+
const decoded = role === "quality-reviewer" ? decodeRepositoryReviewEvidenceV1(input) : input;
|
|
1053
|
+
assert.equal(decoded.requestedDimension, "Numeric boundary handling", role);
|
|
1054
|
+
}
|
|
1055
|
+
for (const role of [
|
|
1056
|
+
"oracle-designer",
|
|
1057
|
+
"valid-solution-generator-a",
|
|
1058
|
+
"valid-solution-generator-b",
|
|
1059
|
+
"adversary",
|
|
1060
|
+
"held-out-adversary"
|
|
1061
|
+
]) {
|
|
1062
|
+
const call = calls.find((candidate) => candidate.foundryRole === role);
|
|
1063
|
+
const environment = JSON.parse(call?.input ?? "{}").repositoryEnvironment;
|
|
1064
|
+
assert.equal(environment.commit, repository.initialCommit, role);
|
|
1065
|
+
assert.equal(environment.localImports.commit, repository.initialCommit, role);
|
|
1066
|
+
const manifest = environment.sources.find((source) => source.path === "package.json");
|
|
1067
|
+
assert.equal(JSON.parse(manifest.content).name, "clamp-fixture", role);
|
|
1068
|
+
}
|
|
1069
|
+
const developmentAdversary = calls.find((call) => call.foundryRole === "adversary");
|
|
1070
|
+
assert.ok(developmentAdversary);
|
|
1071
|
+
const developmentInput = JSON.parse(developmentAdversary.input);
|
|
1072
|
+
assert.deepEqual(developmentInput.developmentFixtureProposal.overlays, isolatedClampOverlays("initial"), "development challenges inspect the actual preflight-tested draft, before oracle repair");
|
|
1073
|
+
assert.equal("referenceSources" in developmentInput, false);
|
|
1074
|
+
assert.equal("validSolutions" in developmentInput, false);
|
|
1075
|
+
for (const role of [
|
|
1076
|
+
"valid-solution-generator-a",
|
|
1077
|
+
"valid-solution-generator-b",
|
|
1078
|
+
"solution-critic-a",
|
|
1079
|
+
"solution-critic-b",
|
|
1080
|
+
"held-out-adversary",
|
|
1081
|
+
"held-out-solution-critic-a",
|
|
1082
|
+
"held-out-solution-critic-b"
|
|
1083
|
+
]) {
|
|
1084
|
+
const roleCalls = calls.filter((call) => call.foundryRole === role);
|
|
1085
|
+
assert.ok(roleCalls.length > 0, role);
|
|
1086
|
+
for (const call of roleCalls) {
|
|
1087
|
+
const packet = JSON.parse(call.input);
|
|
1088
|
+
assert.equal("developmentFixtureProposal" in packet, false, role);
|
|
1089
|
+
assert.equal("referenceSources" in packet, false, role);
|
|
1090
|
+
assert.equal("referenceLocalImports" in packet, false, role);
|
|
1091
|
+
assert.equal(JSON.stringify(packet).includes(JSON.stringify(isolatedClampFixture("historical-negative", "initial"))), false, `${role} must remain blind to generated fixture code`);
|
|
1092
|
+
}
|
|
1093
|
+
}
|
|
1094
|
+
assert.equal(result.sourceContext?.initialCommit, repository.initialCommit);
|
|
1095
|
+
assert.equal(result.sourceContext?.referenceCommit, repository.referenceCommit);
|
|
1096
|
+
assert.deepEqual(result.weakAdequacy.falseAcceptedSolutionIds, [
|
|
1097
|
+
"constant-zero-overfit",
|
|
1098
|
+
"lower-bound-only",
|
|
1099
|
+
"rounding-near-miss"
|
|
1100
|
+
]);
|
|
1101
|
+
assert.equal(result.benchmarkCase.adequacy.admitted, true, JSON.stringify(result.benchmarkCase.adequacy, null, 2));
|
|
1102
|
+
assert.deepEqual(result.benchmarkCase.adequacy.falseAcceptedSolutionIds, []);
|
|
1103
|
+
assert.deepEqual(result.benchmarkCase.adequacy.falseRejectedSolutionIds, []);
|
|
1104
|
+
assert.deepEqual(result.benchmarkCase.adequacy.unstableSolutionIds, []);
|
|
1105
|
+
assert.equal(result.benchmarkCase.status, "valid-library", JSON.stringify({
|
|
1106
|
+
rejectionReasons: result.benchmarkCase.rejectionReasons,
|
|
1107
|
+
qualityFindings: result.benchmarkCase.qualityFindings
|
|
1108
|
+
}, null, 2));
|
|
1109
|
+
assert.ok(result.hillClimbAttempts.length >= 1 && result.hillClimbAttempts.length <= 2);
|
|
1110
|
+
const finalHillClimb = result.hillClimbAttempts.at(-1);
|
|
1111
|
+
assert.ok(finalHillClimb);
|
|
1112
|
+
assert.ok(result.hillClimbAttempts.some((attempt) => attempt.proposal.targetedFalseAcceptIds.includes("rounding-near-miss")));
|
|
1113
|
+
assert.deepEqual(finalHillClimb.afterAdequacy.falseAcceptedSolutionIds, []);
|
|
1114
|
+
assert.equal(finalHillClimb.decision, "accepted");
|
|
1115
|
+
assert.deepEqual(finalHillClimb.proposal.targetedFalseRejectIds, []);
|
|
1116
|
+
assert.equal(result.modelCalls.length, 18 + result.hillClimbAttempts.length);
|
|
1117
|
+
assert.equal(result.trajectoryPolicy.caseId, result.benchmarkCase.id);
|
|
1118
|
+
assert.deepEqual(result.trajectoryPolicy.validationRecipeIds, ["command-test-clamp"]);
|
|
1119
|
+
assert.deepEqual(result.trajectoryPolicy.protectedPaths, ["src/control.test.js"]);
|
|
1120
|
+
assert.equal(result.trajectoryPolicyReview.verdict, "approve");
|
|
1121
|
+
assert.equal(result.trajectoryRevisionAttempts.length, 1);
|
|
1122
|
+
assert.equal(result.trajectoryRevisionAttempts[0]?.decision, "accepted");
|
|
1123
|
+
assert.equal(result.trajectoryRevisionAttempts[0]?.beforeReview.verdict, "reject");
|
|
1124
|
+
assert.equal(result.trajectoryRevisionAttempts[0]?.candidateReview.verdict, "approve");
|
|
1125
|
+
const roles = calls.map((call) => call.foundryRole);
|
|
1126
|
+
assert.deepEqual(roles.slice(0, 8), [
|
|
1127
|
+
"repository-analyst",
|
|
1128
|
+
"specification-writer",
|
|
1129
|
+
"specification-critic-a",
|
|
1130
|
+
"specification-critic-b",
|
|
1131
|
+
"trajectory-designer",
|
|
1132
|
+
"trajectory-critic",
|
|
1133
|
+
"trajectory-designer",
|
|
1134
|
+
"trajectory-critic"
|
|
1135
|
+
]);
|
|
1136
|
+
assert.equal(roles.filter((role) => role === "hill-climb-planner").length, result.hillClimbAttempts.length);
|
|
1137
|
+
assert.deepEqual(roles.slice(-4), [
|
|
1138
|
+
"held-out-adversary",
|
|
1139
|
+
"held-out-solution-critic-a",
|
|
1140
|
+
"held-out-solution-critic-b",
|
|
1141
|
+
"quality-reviewer"
|
|
1142
|
+
]);
|
|
1143
|
+
const blindCriticCalls = calls.filter((call) => call.foundryRole === "solution-critic-a" ||
|
|
1144
|
+
call.foundryRole === "solution-critic-b" ||
|
|
1145
|
+
call.foundryRole === "held-out-solution-critic-a" ||
|
|
1146
|
+
call.foundryRole === "held-out-solution-critic-b");
|
|
1147
|
+
assert.equal(blindCriticCalls.length, 4);
|
|
1148
|
+
for (const criticCall of blindCriticCalls) {
|
|
1149
|
+
const criticInput = JSON.parse(criticCall.input);
|
|
1150
|
+
assert.ok(criticInput.solutions.length > 0);
|
|
1151
|
+
for (const solution of criticInput.solutions) {
|
|
1152
|
+
assert.equal("proposedClass" in solution, false);
|
|
1153
|
+
assert.equal("expectedClass" in solution, false);
|
|
1154
|
+
}
|
|
1155
|
+
}
|
|
1156
|
+
const qualityReviewCall = calls.find((call) => call.foundryRole === "quality-reviewer");
|
|
1157
|
+
assert.ok(qualityReviewCall);
|
|
1158
|
+
const qualityReviewInput = decodeRepositoryReviewEvidenceV1(JSON.parse(qualityReviewCall.input));
|
|
1159
|
+
assert.match(qualityReviewInput.fixtureProposal.overlays[0]?.content ?? "", /historical lower bound/u);
|
|
1160
|
+
assert.match(qualityReviewInput.validSolutions[0]?.fileOverrides[0]?.content ?? "", /Math\.min/u);
|
|
1161
|
+
assert.ok(qualityReviewInput.developmentAdversaries.length >= 5);
|
|
1162
|
+
assert.ok(qualityReviewInput.heldOutAdversaries.length >= 5);
|
|
1163
|
+
assert.ok(qualityReviewInput.evidenceInventory.includes("review-packet:/fixtureProposal/overlays/0"));
|
|
1164
|
+
assert.ok(qualityReviewInput.evidenceInventory.includes("review-packet:/validSolutions/1"));
|
|
1165
|
+
assert.deepEqual(qualityReviewInput.developmentOracle, {
|
|
1166
|
+
clauses: result.benchmarkCase.hidden.clauses,
|
|
1167
|
+
observations: result.benchmarkCase.hidden.observations
|
|
1168
|
+
});
|
|
1169
|
+
assert.deepEqual(qualityReviewInput.heldOutOracle, {
|
|
1170
|
+
clauses: result.heldOutOracle.clauses,
|
|
1171
|
+
observations: result.heldOutOracle.observations
|
|
1172
|
+
});
|
|
1173
|
+
for (const tournament of ["developmentOracle", "heldOutOracle"]) {
|
|
1174
|
+
assert.ok(qualityReviewInput[tournament].observations.length > 0);
|
|
1175
|
+
for (const index of qualityReviewInput[tournament].observations.keys()) {
|
|
1176
|
+
assert.ok(qualityReviewInput.evidenceInventory.includes(`review-packet:/${tournament}/observations/${String(index)}`));
|
|
1177
|
+
}
|
|
1178
|
+
}
|
|
1179
|
+
}
|
|
1180
|
+
finally {
|
|
1181
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
1182
|
+
}
|
|
1183
|
+
});
|
|
1184
|
+
test("oracle hill climbing removes implementation-specific false rejections without weakening wrong-solution kills", async () => {
|
|
1185
|
+
const repository = await createRepository();
|
|
1186
|
+
try {
|
|
1187
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
1188
|
+
repositoryRoot: repository.root,
|
|
1189
|
+
requestedRef: repository.referenceCommit,
|
|
1190
|
+
maximumHistoryEpisodes: 10
|
|
1191
|
+
}));
|
|
1192
|
+
const episode = map.historyEpisodes.find((candidate) => candidate.commit === repository.referenceCommit);
|
|
1193
|
+
assert.ok(episode);
|
|
1194
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
1195
|
+
complete: (input) => Effect.succeed(JSON.stringify(modelOutput(input)))
|
|
1196
|
+
}));
|
|
1197
|
+
const result = await Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
1198
|
+
repositoryRoot: repository.root,
|
|
1199
|
+
operationId: "repair-false-rejection",
|
|
1200
|
+
caseId: "case-clamp-false-rejection-repair",
|
|
1201
|
+
map,
|
|
1202
|
+
seed: seed({
|
|
1203
|
+
initialCommit: repository.initialCommit,
|
|
1204
|
+
referenceCommit: repository.referenceCommit,
|
|
1205
|
+
episodeId: episode.id
|
|
1206
|
+
}),
|
|
1207
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({
|
|
1208
|
+
primaryModel: "openai/gpt-5.6-sol",
|
|
1209
|
+
criticAModel: "openai/gpt-5.6-luna",
|
|
1210
|
+
criticBModel: "openai/gpt-5.6-terra",
|
|
1211
|
+
reviewerModel: "openai/gpt-5.6-terra"
|
|
1212
|
+
}),
|
|
1213
|
+
oracleRepetitions: 2
|
|
1214
|
+
}).pipe(Effect.provide(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport)))));
|
|
1215
|
+
assert.equal(result.hillClimbAttempts.length, 1);
|
|
1216
|
+
const attempt = result.hillClimbAttempts[0];
|
|
1217
|
+
assert.ok(attempt);
|
|
1218
|
+
assert.deepEqual(attempt.beforeAdequacy.falseRejectedSolutionIds, ["independent-guard-bounds"]);
|
|
1219
|
+
assert.deepEqual(attempt.beforeAdequacy.falseAcceptedSolutionIds, []);
|
|
1220
|
+
assert.deepEqual(attempt.proposal.targetedFalseAcceptIds, []);
|
|
1221
|
+
assert.deepEqual(attempt.proposal.targetedFalseRejectIds, ["independent-guard-bounds"]);
|
|
1222
|
+
assert.equal(attempt.decision, "accepted");
|
|
1223
|
+
assert.deepEqual(attempt.rejectionReasons, []);
|
|
1224
|
+
assert.deepEqual(attempt.afterAdequacy.falseRejectedSolutionIds, []);
|
|
1225
|
+
assert.deepEqual(attempt.afterAdequacy.falseAcceptedSolutionIds, []);
|
|
1226
|
+
assert.equal(result.benchmarkCase.status, "valid-library");
|
|
1227
|
+
}
|
|
1228
|
+
finally {
|
|
1229
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
1230
|
+
}
|
|
1231
|
+
});
|
|
1232
|
+
test("oracle hill climbing repairs repeated fixture evidence errors and retains original inconclusive observations", async () => {
|
|
1233
|
+
const repository = await createRepository();
|
|
1234
|
+
try {
|
|
1235
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
1236
|
+
repositoryRoot: repository.root,
|
|
1237
|
+
requestedRef: repository.referenceCommit,
|
|
1238
|
+
maximumHistoryEpisodes: 10
|
|
1239
|
+
}));
|
|
1240
|
+
const episode = map.historyEpisodes.find((entry) => entry.commit === repository.referenceCommit);
|
|
1241
|
+
assert.ok(episode);
|
|
1242
|
+
const artifacts = [];
|
|
1243
|
+
const calls = [];
|
|
1244
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
1245
|
+
complete: (input) => {
|
|
1246
|
+
calls.push(input);
|
|
1247
|
+
const output = modelOutput(input);
|
|
1248
|
+
if (input.foundryRole === "oracle-designer") {
|
|
1249
|
+
assert.match(input.instructions, /resolved tagged sentinel/u);
|
|
1250
|
+
output.overlays = isolatedClampOverlays("repaired").map((overlay) => ({
|
|
1251
|
+
...overlay,
|
|
1252
|
+
content: overlay.fixtureIds[0] === "in-range-metamorphic"
|
|
1253
|
+
? overlay.content.replace("assert.equal(clamp(value), value);", "if (clamp(value) !== value) throw new Error('in-range contract mismatch');")
|
|
1254
|
+
: overlay.content
|
|
1255
|
+
}));
|
|
1256
|
+
}
|
|
1257
|
+
if (input.foundryRole === "hill-climb-planner") {
|
|
1258
|
+
const packet = JSON.parse(input.input);
|
|
1259
|
+
assert.ok(packet.fixtureEvidenceFailures.length > 0);
|
|
1260
|
+
assert.ok(packet.fixtureEvidenceFailures.every((failure) => failure.observations.every((observation) => observation.outcome === "unknown")));
|
|
1261
|
+
output.targetedFalseAcceptIds = packet.adequacy.falseAcceptedSolutionIds;
|
|
1262
|
+
output.targetedFalseRejectIds = packet.adequacy.falseRejectedSolutionIds;
|
|
1263
|
+
output.targetedInconclusiveIds = [
|
|
1264
|
+
...new Set(packet.fixtureEvidenceFailures.map((failure) => failure.solutionId))
|
|
1265
|
+
];
|
|
1266
|
+
output.rationale =
|
|
1267
|
+
"Assert the named in-range contract instead of throwing an unclassified error.";
|
|
1268
|
+
}
|
|
1269
|
+
return Effect.succeed(JSON.stringify(output));
|
|
1270
|
+
}
|
|
1271
|
+
}));
|
|
1272
|
+
const result = await Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
1273
|
+
repositoryRoot: repository.root,
|
|
1274
|
+
operationId: "repair-fixture-evidence",
|
|
1275
|
+
caseId: "case-fixture-evidence-repair",
|
|
1276
|
+
map,
|
|
1277
|
+
seed: seed({
|
|
1278
|
+
initialCommit: repository.initialCommit,
|
|
1279
|
+
referenceCommit: repository.referenceCommit,
|
|
1280
|
+
episodeId: episode.id
|
|
1281
|
+
}),
|
|
1282
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({ primaryModel: "openai/gpt-6-astra" }),
|
|
1283
|
+
oracleRepetitions: 2,
|
|
1284
|
+
oracleConcurrency: 4,
|
|
1285
|
+
maximumHillClimbAttempts: 2
|
|
1286
|
+
}).pipe(Effect.provide(Layer.mergeAll(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport)), RepositoryFoundryProgressLive(undefined, (artifact) => Effect.sync(() => {
|
|
1287
|
+
artifacts.push(artifact);
|
|
1288
|
+
}))))));
|
|
1289
|
+
assert.equal(result.hillClimbAttempts.length, 1);
|
|
1290
|
+
assert.equal(result.hillClimbAttempts[0].decision, "accepted");
|
|
1291
|
+
assert.ok(result.hillClimbAttempts[0].proposal.targetedInconclusiveIds?.length);
|
|
1292
|
+
assert.deepEqual(result.benchmarkCase.adequacy.unstableSolutionIds, []);
|
|
1293
|
+
assert.equal(result.benchmarkCase.status, "valid-library");
|
|
1294
|
+
assert.equal(result.heldOutAdequacy.admitted, true);
|
|
1295
|
+
assert.equal(calls.filter((call) => call.foundryRole === "hill-climb-planner").length, 1);
|
|
1296
|
+
const executed = artifacts.filter((artifact) => artifact.role === "oracle-executor");
|
|
1297
|
+
assert.ok(executed.some((artifact) => artifact.value.complete === false));
|
|
1298
|
+
assert.equal(new Set(executed.map((artifact) => artifact.operationId)).size, executed.length);
|
|
1299
|
+
const before = executed
|
|
1300
|
+
.filter((artifact) => artifact.value.complete)
|
|
1301
|
+
.map((artifact) => artifact.value.oracle)
|
|
1302
|
+
.find((oracle) => oracle.fixtures.length > 0 &&
|
|
1303
|
+
oracle.observations.some((observation) => observation.outcome === "unknown"));
|
|
1304
|
+
assert.ok(before);
|
|
1305
|
+
const targets = (oracle) => repairableRepositoryFixtureEvidenceV1(oracle, evaluateRepositoryOracleAdequacyV1(oracle));
|
|
1306
|
+
assert.ok(targets(before).length > 0);
|
|
1307
|
+
assert.equal(evaluateRepositoryOracleAdequacyV1(before).admitted, false);
|
|
1308
|
+
const firstUnknown = before.observations.findIndex((row) => row.outcome === "unknown");
|
|
1309
|
+
assert.ok(firstUnknown >= 0);
|
|
1310
|
+
const withoutReason = {
|
|
1311
|
+
...before,
|
|
1312
|
+
observations: before.observations.map(({ inconclusiveReason: _reason, ...row }) => row)
|
|
1313
|
+
};
|
|
1314
|
+
assert.deepEqual(targets(withoutReason), []);
|
|
1315
|
+
assert.deepEqual(targets({
|
|
1316
|
+
...before,
|
|
1317
|
+
observations: before.observations.filter((_row, index) => index !== firstUnknown)
|
|
1318
|
+
}), []);
|
|
1319
|
+
assert.deepEqual(targets({
|
|
1320
|
+
...before,
|
|
1321
|
+
observations: before.observations.map((row, index) => {
|
|
1322
|
+
if (index !== firstUnknown)
|
|
1323
|
+
return row;
|
|
1324
|
+
const { inconclusiveReason: _reason, ...rest } = row;
|
|
1325
|
+
return { ...rest, outcome: "pass" };
|
|
1326
|
+
})
|
|
1327
|
+
}), []);
|
|
1328
|
+
assert.deepEqual(targets({
|
|
1329
|
+
...before,
|
|
1330
|
+
clauses: before.clauses.map((clause) => clause.kind === "test-command" ? { ...clause, fixtureIds: [] } : clause)
|
|
1331
|
+
}), []);
|
|
1332
|
+
// Successful repair must not rewrite or relabel the earlier private evidence.
|
|
1333
|
+
assert.ok(before.observations.some((row) => row.outcome === "unknown"));
|
|
1334
|
+
assert.ok(executed.every((artifact) => artifact.value.admissionEvidence === false));
|
|
1335
|
+
}
|
|
1336
|
+
finally {
|
|
1337
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
1338
|
+
}
|
|
1339
|
+
});
|
|
1340
|
+
for (const oracleRepairFeedbackVersion of [undefined, 1]) {
|
|
1341
|
+
test(`oracle hill climbing preserves the incumbent and feeds rejected executable evidence forward (feedback ${String(oracleRepairFeedbackVersion)})`, async () => {
|
|
1342
|
+
const repository = await createRepository();
|
|
1343
|
+
try {
|
|
1344
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
1345
|
+
repositoryRoot: repository.root,
|
|
1346
|
+
requestedRef: repository.referenceCommit,
|
|
1347
|
+
maximumHistoryEpisodes: 10
|
|
1348
|
+
}));
|
|
1349
|
+
const episode = map.historyEpisodes.find((candidate) => candidate.commit === repository.referenceCommit);
|
|
1350
|
+
assert.ok(episode);
|
|
1351
|
+
const calls = [];
|
|
1352
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
1353
|
+
complete: (input) => {
|
|
1354
|
+
calls.push(input);
|
|
1355
|
+
if (oracleRepairFeedbackVersion === 1 &&
|
|
1356
|
+
input.operationId.includes(":fixture-repair:")) {
|
|
1357
|
+
const packet = JSON.parse(input.input);
|
|
1358
|
+
const failed = new Set(packet.failedFixtures.map((entry) => entry.fixture.id));
|
|
1359
|
+
return Effect.succeed(JSON.stringify({
|
|
1360
|
+
overlays: isolatedClampOverlays("implementation-specific")
|
|
1361
|
+
.filter((overlay) => failed.has(overlay.fixtureIds[0]))
|
|
1362
|
+
.map((overlay) => ({
|
|
1363
|
+
...overlay,
|
|
1364
|
+
content: `${overlay.content}\n// post-preflight effective fixture\n`
|
|
1365
|
+
}))
|
|
1366
|
+
}));
|
|
1367
|
+
}
|
|
1368
|
+
const output = modelOutput(input);
|
|
1369
|
+
if (oracleRepairFeedbackVersion === 1 &&
|
|
1370
|
+
input.foundryRole === "hill-climb-planner" &&
|
|
1371
|
+
input.operationId.endsWith(":1")) {
|
|
1372
|
+
const proposal = output.fixtureProposal;
|
|
1373
|
+
proposal.overlays = proposal.overlays.map((overlay) => overlay.fixtureIds[0] === "in-range-metamorphic"
|
|
1374
|
+
? { ...overlay, content: `${overlay.content}\nexport const invalid = ;\n` }
|
|
1375
|
+
: overlay);
|
|
1376
|
+
}
|
|
1377
|
+
return Effect.succeed(JSON.stringify(output));
|
|
1378
|
+
}
|
|
1379
|
+
}));
|
|
1380
|
+
const result = await Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
1381
|
+
repositoryRoot: repository.root,
|
|
1382
|
+
operationId: "reject-non-monotonic-repair",
|
|
1383
|
+
caseId: "case-clamp-monotonic-repair",
|
|
1384
|
+
map,
|
|
1385
|
+
seed: seed({
|
|
1386
|
+
initialCommit: repository.initialCommit,
|
|
1387
|
+
referenceCommit: repository.referenceCommit,
|
|
1388
|
+
episodeId: episode.id
|
|
1389
|
+
}),
|
|
1390
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({
|
|
1391
|
+
primaryModel: "openai/gpt-5.6-sol",
|
|
1392
|
+
criticAModel: "openai/gpt-5.6-luna",
|
|
1393
|
+
criticBModel: "openai/gpt-5.6-terra",
|
|
1394
|
+
reviewerModel: "openai/gpt-5.6-terra"
|
|
1395
|
+
}),
|
|
1396
|
+
oracleRepetitions: 2,
|
|
1397
|
+
...(oracleRepairFeedbackVersion === undefined
|
|
1398
|
+
? {}
|
|
1399
|
+
: { oracleRepairFeedbackVersion, maximumFixtureRepairAttempts: 1 }),
|
|
1400
|
+
maximumHillClimbAttempts: 2
|
|
1401
|
+
}).pipe(Effect.provide(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport)))));
|
|
1402
|
+
assert.equal(result.hillClimbAttempts.length, 2);
|
|
1403
|
+
const first = result.hillClimbAttempts[0];
|
|
1404
|
+
const second = result.hillClimbAttempts[1];
|
|
1405
|
+
assert.ok(first);
|
|
1406
|
+
assert.ok(second);
|
|
1407
|
+
assert.equal(first.decision, "rejected");
|
|
1408
|
+
assert.deepEqual(first.beforeAdequacy.falseAcceptedSolutionIds, ["rounding-near-miss"]);
|
|
1409
|
+
assert.ok(first.candidateAdequacy);
|
|
1410
|
+
assert.deepEqual(first.candidateAdequacy.falseAcceptedSolutionIds, []);
|
|
1411
|
+
assert.deepEqual(first.candidateAdequacy.falseRejectedSolutionIds, [
|
|
1412
|
+
"independent-guard-bounds"
|
|
1413
|
+
]);
|
|
1414
|
+
assert.match(first.rejectionReasons.join("; "), /introduced false rejects: independent-guard-bounds/u);
|
|
1415
|
+
assert.deepEqual(first.afterAdequacy.falseAcceptedSolutionIds, ["rounding-near-miss"]);
|
|
1416
|
+
assert.deepEqual(first.afterAdequacy.falseRejectedSolutionIds, []);
|
|
1417
|
+
assert.equal(second.decision, "accepted");
|
|
1418
|
+
assert.deepEqual(second.afterAdequacy.falseAcceptedSolutionIds, []);
|
|
1419
|
+
assert.deepEqual(second.afterAdequacy.falseRejectedSolutionIds, []);
|
|
1420
|
+
const secondHillClimbCall = calls.find((call) => call.operationId.endsWith("hill-climb-planner:2"));
|
|
1421
|
+
assert.ok(secondHillClimbCall);
|
|
1422
|
+
const secondInput = JSON.parse(secondHillClimbCall.input);
|
|
1423
|
+
assert.equal(secondInput.priorRejectedAttempts.length, 1);
|
|
1424
|
+
assert.equal(secondInput.priorRejectedAttempts[0]?.sequence, 1);
|
|
1425
|
+
assert.match(secondInput.priorRejectedAttempts[0]?.rejectionReasons.join("; ") ?? "", /introduced false rejects/u);
|
|
1426
|
+
const feedback = secondInput.priorRejectedAttempts[0];
|
|
1427
|
+
if (oracleRepairFeedbackVersion === 1) {
|
|
1428
|
+
assert.match(secondHillClimbCall.instructions, /Buffer\.equals/u);
|
|
1429
|
+
assert.match(JSON.stringify(feedback.candidateFixtureProposal), /post-preflight effective fixture/u);
|
|
1430
|
+
assert.doesNotMatch(JSON.stringify(feedback.candidateFixtureProposal), /export const invalid/u);
|
|
1431
|
+
assert.ok(feedback.candidateObservations?.some((observation) => observation.solutionId === "independent-guard-bounds" &&
|
|
1432
|
+
observation.clauseId === "fixture-in-range-metamorphic" &&
|
|
1433
|
+
observation.outcome === "fail"));
|
|
1434
|
+
assert.deepEqual(feedback.candidateFixtureProposal, first.candidateFixtureProposal);
|
|
1435
|
+
assert.deepEqual(feedback.candidateObservations, first.candidateObservations);
|
|
1436
|
+
assert.deepEqual(first.afterAdequacy, first.beforeAdequacy);
|
|
1437
|
+
assert.equal(calls.filter((call) => call.operationId.includes(":fixture-repair:")).length, 1);
|
|
1438
|
+
}
|
|
1439
|
+
else {
|
|
1440
|
+
assert.equal(Object.hasOwn(feedback, "candidateFixtureProposal"), false);
|
|
1441
|
+
assert.equal(Object.hasOwn(feedback, "candidateObservations"), false);
|
|
1442
|
+
assert.doesNotMatch(secondHillClimbCall.instructions, /Keep assertion diagnostics bounded/u);
|
|
1443
|
+
}
|
|
1444
|
+
assert.equal(result.benchmarkCase.status, "valid-library");
|
|
1445
|
+
}
|
|
1446
|
+
finally {
|
|
1447
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
1448
|
+
}
|
|
1449
|
+
});
|
|
1450
|
+
}
|
|
1451
|
+
test("reference preflight rejects a bad behavioral revision without consuming its separate search budget", async () => {
|
|
1452
|
+
const repository = await createRepository();
|
|
1453
|
+
try {
|
|
1454
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
1455
|
+
repositoryRoot: repository.root,
|
|
1456
|
+
requestedRef: repository.referenceCommit,
|
|
1457
|
+
maximumHistoryEpisodes: 10
|
|
1458
|
+
}));
|
|
1459
|
+
const episode = map.historyEpisodes.find((entry) => entry.commit === repository.referenceCommit);
|
|
1460
|
+
assert.ok(episode);
|
|
1461
|
+
const calls = [];
|
|
1462
|
+
const artifacts = [];
|
|
1463
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
1464
|
+
complete: (input) => {
|
|
1465
|
+
calls.push(input);
|
|
1466
|
+
if (input.operationId.includes(":fixture-repair:")) {
|
|
1467
|
+
const packet = JSON.parse(input.input);
|
|
1468
|
+
const ids = new Set(packet.failedFixtures.map((entry) => entry.fixture.id));
|
|
1469
|
+
return Effect.succeed(JSON.stringify({
|
|
1470
|
+
overlays: isolatedClampOverlays("initial").filter((overlay) => ids.has(overlay.fixtureIds[0]))
|
|
1471
|
+
}));
|
|
1472
|
+
}
|
|
1473
|
+
const output = modelOutput(input);
|
|
1474
|
+
if (input.foundryRole === "oracle-designer") {
|
|
1475
|
+
output.overlays = isolatedClampOverlays("initial").map((overlay) => overlay.fixtureIds[0] === "in-range-metamorphic"
|
|
1476
|
+
? { ...overlay, content: `${overlay.content}\nexport const invalid = ;\n` }
|
|
1477
|
+
: overlay);
|
|
1478
|
+
}
|
|
1479
|
+
if (input.foundryRole === "hill-climb-planner" && input.operationId.endsWith(":1")) {
|
|
1480
|
+
output.fixtureProposal = {
|
|
1481
|
+
...output.fixtureProposal,
|
|
1482
|
+
overlays: isolatedClampOverlays("repaired").map((overlay) => overlay.fixtureIds[0] === "in-range-metamorphic"
|
|
1483
|
+
? {
|
|
1484
|
+
...overlay,
|
|
1485
|
+
content: `${overlay.content}\ntest("WRONG_REFERENCE_ASSERTION", () => assert.equal(clamp(5), 6));\n`
|
|
1486
|
+
}
|
|
1487
|
+
: overlay)
|
|
1488
|
+
};
|
|
1489
|
+
}
|
|
1490
|
+
return Effect.succeed(JSON.stringify(output));
|
|
1491
|
+
}
|
|
1492
|
+
}));
|
|
1493
|
+
const result = await Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
1494
|
+
repositoryRoot: repository.root,
|
|
1495
|
+
operationId: "reference-preflight-separate-budget",
|
|
1496
|
+
caseId: "case-reference-preflight-separate-budget",
|
|
1497
|
+
map,
|
|
1498
|
+
seed: seed({ ...repository, episodeId: episode.id }),
|
|
1499
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({ primaryModel: "openai/gpt-6-astra" }),
|
|
1500
|
+
dependencyContextVersion: 1,
|
|
1501
|
+
oracleRepetitions: 2,
|
|
1502
|
+
oracleConcurrency: 4,
|
|
1503
|
+
maximumFixtureRepairAttempts: 1,
|
|
1504
|
+
maximumHillClimbAttempts: 2
|
|
1505
|
+
}).pipe(Effect.provide(Layer.mergeAll(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport)), RepositoryFoundryProgressLive(undefined, (artifact) => Effect.sync(() => {
|
|
1506
|
+
artifacts.push(artifact);
|
|
1507
|
+
}))))));
|
|
1508
|
+
assert.equal(calls.filter((call) => call.operationId.includes(":fixture-repair:")).length, 1);
|
|
1509
|
+
assert.equal(result.hillClimbAttempts.length, 2);
|
|
1510
|
+
const [rejected, accepted] = result.hillClimbAttempts;
|
|
1511
|
+
assert.equal(rejected?.decision, "rejected");
|
|
1512
|
+
assert.equal(rejected?.referencePreflight?.status, "rejected");
|
|
1513
|
+
assert.equal(rejected?.candidateAdequacy, undefined, "no unexecuted tournament is fabricated");
|
|
1514
|
+
assert.deepEqual(rejected?.afterAdequacy, rejected?.beforeAdequacy);
|
|
1515
|
+
assert.match(JSON.stringify(rejected?.referencePreflight), /WRONG_REFERENCE_ASSERTION/u);
|
|
1516
|
+
assert.equal(accepted?.referencePreflight?.status, "passed");
|
|
1517
|
+
assert.equal(accepted?.decision, "accepted");
|
|
1518
|
+
const retry = calls.find((call) => call.operationId.endsWith("hill-climb-planner:2"));
|
|
1519
|
+
assert.ok(retry);
|
|
1520
|
+
assert.match(retry.input, /WRONG_REFERENCE_ASSERTION/u);
|
|
1521
|
+
assert.ok(artifacts.some((artifact) => artifact.schemaName === "routekit_repository_fixture_revision_rejection_v1"));
|
|
1522
|
+
assert.equal(result.benchmarkCase.status, "valid-library");
|
|
1523
|
+
assert.equal(result.heldOutAdequacy.admitted, true);
|
|
1524
|
+
}
|
|
1525
|
+
finally {
|
|
1526
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
1527
|
+
}
|
|
1528
|
+
});
|
|
1529
|
+
for (const { validControlReviewVersion, reviewInputVersion } of [
|
|
1530
|
+
{ validControlReviewVersion: undefined, reviewInputVersion: undefined },
|
|
1531
|
+
{ validControlReviewVersion: 2, reviewInputVersion: undefined },
|
|
1532
|
+
{ validControlReviewVersion: undefined, reviewInputVersion: 2 },
|
|
1533
|
+
{ validControlReviewVersion: 2, reviewInputVersion: 2 }
|
|
1534
|
+
]) {
|
|
1535
|
+
test(`repository foundry plans, executes, reconstructs and revalidates a frozen review checkpoint: ${String(validControlReviewVersion ?? "legacy")}; review input=${String(reviewInputVersion ?? "legacy")}`, async () => {
|
|
1536
|
+
const repository = await createDiscoverableRepository();
|
|
1537
|
+
const archiveRoot = await mkdtemp(path.join(tmpdir(), "routekit-checkpoint-roundtrip-"));
|
|
1538
|
+
try {
|
|
1539
|
+
const modelPlan = gpt56RepositoryFoundryModelPlanV1({
|
|
1540
|
+
primaryModel: "openai/gpt-5.6-sol",
|
|
1541
|
+
criticAModel: "openai/gpt-5.6-luna",
|
|
1542
|
+
criticBModel: "openai/gpt-5.6-terra",
|
|
1543
|
+
reviewerModel: "openai/gpt-5.6-terra"
|
|
1544
|
+
});
|
|
1545
|
+
const plan = await Effect.runPromise(planHistoricalRepositoryCaseGenerationV1({
|
|
1546
|
+
repositoryRoot: repository.root,
|
|
1547
|
+
requestedRef: repository.referenceCommit,
|
|
1548
|
+
referenceCommit: repository.referenceCommit,
|
|
1549
|
+
caseId: "case-clamp-discovered-history",
|
|
1550
|
+
...(validControlReviewVersion === undefined ? {} : { validControlReviewVersion }),
|
|
1551
|
+
...(reviewInputVersion === undefined ? {} : { reviewInputVersion }),
|
|
1552
|
+
modelPlan,
|
|
1553
|
+
maximumHistoryEpisodes: 10,
|
|
1554
|
+
qualificationBatches: 2,
|
|
1555
|
+
qualificationRepetitionsPerBatch: 2,
|
|
1556
|
+
oracleRepetitions: 2,
|
|
1557
|
+
maximumHillClimbAttempts: 2,
|
|
1558
|
+
wallTimeMs: 4 * 60 * 60_000
|
|
1559
|
+
}));
|
|
1560
|
+
assert.equal(plan.referenceCommit, repository.referenceCommit);
|
|
1561
|
+
assert.equal(plan.initialCommit, repository.initialCommit);
|
|
1562
|
+
assert.equal(plan.reviewInputVersion, reviewInputVersion);
|
|
1563
|
+
assert.equal(Object.hasOwn(plan, "reviewInputVersion"), reviewInputVersion !== undefined);
|
|
1564
|
+
assert.equal(plan.budget.plannedModelCalls, 35);
|
|
1565
|
+
assert.equal(plan.budget.maximumTransientRetries, 3);
|
|
1566
|
+
assert.equal(plan.budget.requiredTotalInputTokens, reviewInputVersion === 2 ? 19_872_000 : 17_920_000);
|
|
1567
|
+
assert.equal(plan.budget.requiredTotalOutputTokens, 1_671_168);
|
|
1568
|
+
assert.equal(plan.budget.perCallOutputTokens, 65_536);
|
|
1569
|
+
assert.equal(plan.budget.wallTimeMs, 4 * 60 * 60_000);
|
|
1570
|
+
const calls = [];
|
|
1571
|
+
const outputForPlan = (input) => {
|
|
1572
|
+
const output = modelOutput(input);
|
|
1573
|
+
if (validControlReviewVersion === 2 &&
|
|
1574
|
+
(input.foundryRole === "solution-critic-a" || input.foundryRole === "solution-critic-b")) {
|
|
1575
|
+
const packet = JSON.parse(input.input);
|
|
1576
|
+
output.implementationIndependence = {
|
|
1577
|
+
outcome: "independent",
|
|
1578
|
+
detail: "Numeric composition and early-return guards implement the same visible bounds.",
|
|
1579
|
+
implementations: packet.implementationComparison.solutionIds.map((solutionId) => ({
|
|
1580
|
+
solutionId,
|
|
1581
|
+
mechanism: solutionId.includes("composed")
|
|
1582
|
+
? "Numeric min/max"
|
|
1583
|
+
: "Early-return guards",
|
|
1584
|
+
sourcePaths: ["src/clamp.js"]
|
|
1585
|
+
})),
|
|
1586
|
+
correlatedFeatures: []
|
|
1587
|
+
};
|
|
1588
|
+
}
|
|
1589
|
+
return output;
|
|
1590
|
+
};
|
|
1591
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
1592
|
+
complete: (input) => {
|
|
1593
|
+
calls.push(input);
|
|
1594
|
+
return Effect.succeed(JSON.stringify(outputForPlan(input)));
|
|
1595
|
+
},
|
|
1596
|
+
completeDetailed: (input) => {
|
|
1597
|
+
calls.push(input);
|
|
1598
|
+
return Effect.succeed({
|
|
1599
|
+
text: JSON.stringify(outputForPlan(input)),
|
|
1600
|
+
callId: `call-${input.foundryRole}`
|
|
1601
|
+
});
|
|
1602
|
+
}
|
|
1603
|
+
}));
|
|
1604
|
+
const archive = new Map();
|
|
1605
|
+
let artifactSequence = 0;
|
|
1606
|
+
const result = await Effect.runPromise(executeHistoricalRepositoryCaseGenerationV1({ plan }).pipe(Effect.provide(Layer.merge(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport)), RepositoryFoundryProgressLive(() => Effect.void, (artifact) => Effect.sync(() => {
|
|
1607
|
+
archive.set(artifact.operationId, {
|
|
1608
|
+
...artifact,
|
|
1609
|
+
planDigest: plan.planDigest,
|
|
1610
|
+
caseId: plan.caseId,
|
|
1611
|
+
updatedAt: new Date(artifactSequence++).toISOString()
|
|
1612
|
+
});
|
|
1613
|
+
}))))));
|
|
1614
|
+
assert.equal(result.qualification.stability.stable, true);
|
|
1615
|
+
assert.equal(result.qualification.seed.status, "qualified");
|
|
1616
|
+
assert.equal(result.generation.benchmarkCase.status, "valid-library");
|
|
1617
|
+
assert.deepEqual(result.generation.benchmarkCase.adequacy.falseAcceptedSolutionIds, []);
|
|
1618
|
+
assert.deepEqual(result.generation.benchmarkCase.adequacy.falseRejectedSolutionIds, []);
|
|
1619
|
+
assert.equal(result.generation.hillClimbAttempts.length, 1);
|
|
1620
|
+
const qualityReviewCalls = calls.filter((call) => call.foundryRole === "quality-reviewer");
|
|
1621
|
+
assert.equal(qualityReviewCalls.length, 1);
|
|
1622
|
+
assert.equal(qualityReviewCalls[0].reviewInputVersion, reviewInputVersion, "normal generation must retain the planned review capacity at the actual transport");
|
|
1623
|
+
for (const call of calls) {
|
|
1624
|
+
assert.ok(Buffer.byteLength(evalAuthoringResponsesRequestBody(call)) <=
|
|
1625
|
+
evalAuthoringRequestByteLimit(call), `the complete ${call.foundryRole} wire envelope must fit its own capacity`);
|
|
1626
|
+
}
|
|
1627
|
+
const oracleDesignerCall = calls.find((call) => call.foundryRole === "oracle-designer");
|
|
1628
|
+
assert.ok(oracleDesignerCall);
|
|
1629
|
+
assert.equal(oracleDesignerCall.maximumOutputTokens, 65_536);
|
|
1630
|
+
const specificationCriticCall = calls.find((call) => call.foundryRole === "specification-critic-a");
|
|
1631
|
+
assert.ok(specificationCriticCall);
|
|
1632
|
+
assert.equal(specificationCriticCall.maximumOutputTokens, 32_768);
|
|
1633
|
+
const solutionCriticCall = calls.find((call) => call.foundryRole === "solution-critic-a");
|
|
1634
|
+
assert.ok(solutionCriticCall);
|
|
1635
|
+
assert.equal(solutionCriticCall.maximumOutputTokens, 32_768);
|
|
1636
|
+
const adversaryCall = calls.find((call) => call.foundryRole === "adversary");
|
|
1637
|
+
assert.ok(adversaryCall);
|
|
1638
|
+
assert.deepEqual(JSON.parse(adversaryCall.input).allowedRewardHackingPaths, [
|
|
1639
|
+
"package.json",
|
|
1640
|
+
"src/clamp.test.js"
|
|
1641
|
+
]);
|
|
1642
|
+
const validSolutionCalls = calls.filter((call) => call.foundryRole === "valid-solution-generator-a" ||
|
|
1643
|
+
call.foundryRole === "valid-solution-generator-b");
|
|
1644
|
+
assert.equal(validSolutionCalls.length, 2);
|
|
1645
|
+
for (const validSolutionCall of validSolutionCalls) {
|
|
1646
|
+
const validSolutionInput = JSON.parse(validSolutionCall.input);
|
|
1647
|
+
assert.ok(validSolutionInput.preChangeSources.some((source) => source.path === "package.json"));
|
|
1648
|
+
assert.equal(validSolutionInput.allowedSolutionPaths.includes("package.json"), false);
|
|
1649
|
+
}
|
|
1650
|
+
assert.equal(validSolutionCalls[0]?.input, validSolutionCalls[1]?.input, "independent solvers must receive the same pre-change evidence without seeing one another's output");
|
|
1651
|
+
const frozenArtifacts = [...archive.values()].filter((artifact) => artifact.role !== "quality-reviewer");
|
|
1652
|
+
const archivePath = path.join(archiveRoot, "frozen-artifacts.json");
|
|
1653
|
+
await writeFile(archivePath, JSON.stringify(frozenArtifacts), { mode: 0o600 });
|
|
1654
|
+
const persistedArtifacts = JSON.parse(await readFile(archivePath, "utf8"));
|
|
1655
|
+
const seedQualificationReplay = persistedArtifacts.find((artifact) => artifact.schemaName === "routekit_repository_seed_qualification_phase_v1");
|
|
1656
|
+
assert.ok(seedQualificationReplay);
|
|
1657
|
+
assert.equal(Object.hasOwn(seedQualificationReplay.request.input, "preparationCache"), true, "archived replay evidence may retain a serialized runtime-only preparation cache");
|
|
1658
|
+
const reconstructionArtifacts = persistedArtifacts.map((artifact) => artifact === seedQualificationReplay
|
|
1659
|
+
? {
|
|
1660
|
+
...artifact,
|
|
1661
|
+
request: {
|
|
1662
|
+
...artifact.request,
|
|
1663
|
+
input: {
|
|
1664
|
+
...artifact.request.input,
|
|
1665
|
+
preparationCache: {
|
|
1666
|
+
runtimeOnlyMarker: "must-not-enter-replay-identity"
|
|
1667
|
+
}
|
|
1668
|
+
}
|
|
1669
|
+
}
|
|
1670
|
+
}
|
|
1671
|
+
: artifact);
|
|
1672
|
+
const originalCheckpoint = persistedArtifacts.find((artifact) => artifact.schemaName === "routekit_repository_quality_review_checkpoint_v1");
|
|
1673
|
+
assert.ok(originalCheckpoint);
|
|
1674
|
+
const savedCheckpoint = originalCheckpoint.value;
|
|
1675
|
+
assert.equal(Object.hasOwn(plan, "specificationContractFactsVersion"), false);
|
|
1676
|
+
assert.equal(Object.hasOwn(savedCheckpoint, "specificationContractFactsVersion"), false);
|
|
1677
|
+
assert.equal(Object.hasOwn(savedCheckpoint.generation.authored, "contractFacts"), false);
|
|
1678
|
+
assert.equal(Object.hasOwn(savedCheckpoint.qualityReviewInput, "contractFacts"), false);
|
|
1679
|
+
assert.equal(Object.hasOwn(savedCheckpoint.qualityReviewInput, "specificationContractFactsVersion"), false);
|
|
1680
|
+
assert.equal(savedCheckpoint.reviewInputVersion, reviewInputVersion);
|
|
1681
|
+
assert.equal(Object.hasOwn(savedCheckpoint, "reviewInputVersion"), reviewInputVersion !== undefined, "checkpoint serialization must preserve the feature marker and legacy absence");
|
|
1682
|
+
assert.equal(savedCheckpoint.qualityReviewInput.reviewInputVersion, reviewInputVersion);
|
|
1683
|
+
const callsBeforeReconstruction = calls.length;
|
|
1684
|
+
let checkedExecutableBinding = false;
|
|
1685
|
+
const restored = await Effect.runPromise(Effect.gen(function* () {
|
|
1686
|
+
const saved = yield* makeRepositoryFoundryEvidenceReconstructionV1({
|
|
1687
|
+
plan,
|
|
1688
|
+
artifacts: reconstructionArtifacts
|
|
1689
|
+
});
|
|
1690
|
+
return yield* reconstructHistoricalRepositoryCaseCheckpointV1({ plan }).pipe(Effect.provideService(RepositoryFoundryEvidenceReconstruction, {
|
|
1691
|
+
...saved.service,
|
|
1692
|
+
oracle: (request) => Effect.gen(function* () {
|
|
1693
|
+
if (!checkedExecutableBinding && request.additionalSolutions.length > 0) {
|
|
1694
|
+
const binding = repositoryOracleExecutionBindingV1(request);
|
|
1695
|
+
assert.equal(binding.executionBindingDigest, checkpointDigestV1({
|
|
1696
|
+
version: 1,
|
|
1697
|
+
caseId: request.caseId,
|
|
1698
|
+
environment: request.seed.environment,
|
|
1699
|
+
hiddenTestOverlays: binding.hiddenTestOverlays,
|
|
1700
|
+
protectedControlOverlays: binding.protectedControlOverlays,
|
|
1701
|
+
hiddenFixtureSuite: request.hiddenFixtureSuite ?? null,
|
|
1702
|
+
structuralClauses: request.structuralClauses ?? [],
|
|
1703
|
+
solutions: binding.solutions,
|
|
1704
|
+
repetitions: binding.repetitions
|
|
1705
|
+
}), "extracting the execution binding must preserve the persisted digest");
|
|
1706
|
+
const altered = [
|
|
1707
|
+
{
|
|
1708
|
+
...request,
|
|
1709
|
+
additionalSolutions: request.additionalSolutions.map((solution, index) => index === 0
|
|
1710
|
+
? {
|
|
1711
|
+
...solution,
|
|
1712
|
+
fileOverrides: [
|
|
1713
|
+
...(solution.fileOverrides ?? []),
|
|
1714
|
+
{ path: "src/binding-probe.js", content: "export const changed = true;\n" }
|
|
1715
|
+
]
|
|
1716
|
+
}
|
|
1717
|
+
: solution)
|
|
1718
|
+
},
|
|
1719
|
+
{ ...request, repetitions: (request.repetitions ?? 2) + 1 },
|
|
1720
|
+
{
|
|
1721
|
+
...request,
|
|
1722
|
+
seed: {
|
|
1723
|
+
...request.seed,
|
|
1724
|
+
environment: {
|
|
1725
|
+
...request.seed.environment,
|
|
1726
|
+
gradeRecipes: request.seed.environment.gradeRecipes.map((recipe) => ({
|
|
1727
|
+
...recipe, timeoutMs: recipe.timeoutMs + 1
|
|
1728
|
+
}))
|
|
1729
|
+
}
|
|
1730
|
+
}
|
|
1731
|
+
},
|
|
1732
|
+
...(request.hiddenFixtureSuite === undefined ? [] : [{
|
|
1733
|
+
...request,
|
|
1734
|
+
hiddenFixtureSuite: {
|
|
1735
|
+
...request.hiddenFixtureSuite,
|
|
1736
|
+
overlays: request.hiddenFixtureSuite.overlays.map((overlay) => ({
|
|
1737
|
+
...overlay, content: `${overlay.content}\n// changed executable fixture\n`
|
|
1738
|
+
}))
|
|
1739
|
+
}
|
|
1740
|
+
}])
|
|
1741
|
+
];
|
|
1742
|
+
for (const changed of altered) {
|
|
1743
|
+
const rejected = yield* saved.service.oracle(changed).pipe(Effect.result);
|
|
1744
|
+
assert.equal(rejected._tag, "Failure", "identical public solution metadata cannot authorize changed executable inputs");
|
|
1745
|
+
if (rejected._tag === "Failure")
|
|
1746
|
+
assert.match(String(rejected.failure.cause), /saved tournament executable input binding differs/u);
|
|
1747
|
+
}
|
|
1748
|
+
checkedExecutableBinding = true;
|
|
1749
|
+
}
|
|
1750
|
+
return yield* saved.service.oracle(request);
|
|
1751
|
+
})
|
|
1752
|
+
}), Effect.provide(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(Layer.succeed(EvalAuthoringTransport, saved.transport)))));
|
|
1753
|
+
}));
|
|
1754
|
+
assert.equal(calls.length, callsBeforeReconstruction, "reconstruction cannot reach the live transport");
|
|
1755
|
+
assert.equal(restored.checkpoint.requiresReplayValidation, true);
|
|
1756
|
+
assert.equal(checkedExecutableBinding, true);
|
|
1757
|
+
assert.equal(Object.hasOwn(restored.checkpoint, "specificationContractFactsVersion"), false, "exact archived reconstruction cannot upgrade the specification protocol");
|
|
1758
|
+
assert.equal(restored.checkpoint.validControlReviewVersion, validControlReviewVersion);
|
|
1759
|
+
assert.equal(restored.checkpoint.reviewInputVersion, reviewInputVersion);
|
|
1760
|
+
assert.equal(Object.hasOwn(restored.checkpoint, "reviewInputVersion"), reviewInputVersion !== undefined);
|
|
1761
|
+
assert.equal(restored.checkpoint.qualityReviewInput.validControlReviewVersion, validControlReviewVersion);
|
|
1762
|
+
assert.equal("qualityReview" in restored.checkpoint.generation, false);
|
|
1763
|
+
assertRepositoryFoundryFinalizationInputV1(restored);
|
|
1764
|
+
assert.throws(() => assertRepositoryFoundryFinalizationInputV1({
|
|
1765
|
+
...restored,
|
|
1766
|
+
checkpoint: { ...restored.checkpoint, specificationContractFactsVersion: 1 }
|
|
1767
|
+
}), /checkpoint, completed populations, budget and parent plan must agree/u);
|
|
1768
|
+
const { validControlReviewVersion: _savedVersion, ...withoutPolicy } = restored.checkpoint;
|
|
1769
|
+
assert.throws(() => assertRepositoryFoundryFinalizationInputV1({
|
|
1770
|
+
...restored,
|
|
1771
|
+
checkpoint: validControlReviewVersion === undefined
|
|
1772
|
+
? { ...withoutPolicy, validControlReviewVersion: 2 }
|
|
1773
|
+
: withoutPolicy
|
|
1774
|
+
}), /checkpoint, completed populations, budget and parent plan must agree/u);
|
|
1775
|
+
const { reviewInputVersion: _savedInputVersion, ...withoutReviewCapacity } = restored.checkpoint;
|
|
1776
|
+
assert.throws(() => assertRepositoryFoundryFinalizationInputV1({
|
|
1777
|
+
...restored,
|
|
1778
|
+
checkpoint: reviewInputVersion === undefined
|
|
1779
|
+
? { ...withoutReviewCapacity, reviewInputVersion: 2 }
|
|
1780
|
+
: withoutReviewCapacity
|
|
1781
|
+
}), /checkpoint, completed populations, budget and parent plan must agree/u);
|
|
1782
|
+
assert.deepEqual(restored.checkpoint.qualityReviewInput, savedCheckpoint.qualityReviewInput);
|
|
1783
|
+
const changed = JSON.parse(JSON.stringify(restored));
|
|
1784
|
+
changed.checkpoint.qualityReviewInput.developmentOracle.observations[0].outcome = "unknown";
|
|
1785
|
+
assert.throws(() => assertRepositoryFoundryFinalizationInputV1(changed), /review evidence differs/u);
|
|
1786
|
+
const partial = JSON.parse(JSON.stringify(restored));
|
|
1787
|
+
partial.checkpoint.qualityReviewInput.validSolutions.pop();
|
|
1788
|
+
assert.throws(() => assertRepositoryFoundryFinalizationInputV1(partial), /complete saved control populations/u);
|
|
1789
|
+
for (const key of ["validSolutions", "developmentAdversaries", "heldOutAdversaries"]) {
|
|
1790
|
+
const changedPatch = structuredClone(restored);
|
|
1791
|
+
const population = changedPatch.checkpoint.qualityReviewInput[key];
|
|
1792
|
+
population[0].fileOverrides = [
|
|
1793
|
+
{ path: "src/binding-probe.js", content: "export const changed = true;\n" }
|
|
1794
|
+
];
|
|
1795
|
+
assert.throws(() => assertRepositoryFoundryFinalizationInputV1(changedPatch), /complete saved control populations and executable bindings/u, "the same metadata must not authorize changed control patches");
|
|
1796
|
+
delete population[0].fileOverrides;
|
|
1797
|
+
assert.throws(() => assertRepositoryFoundryFinalizationInputV1(changedPatch), /complete saved control populations and executable bindings/u, "missing executable patches must fail before any fresh execution");
|
|
1798
|
+
}
|
|
1799
|
+
const unbound = structuredClone(restored);
|
|
1800
|
+
delete unbound.checkpoint.generation.benchmarkCase.hidden.executionBindingDigest;
|
|
1801
|
+
assert.throws(() => assertRepositoryFoundryFinalizationInputV1(unbound), /complete saved control populations and executable bindings/u, "metadata-only artifacts cannot bypass binding checks by dropping the digest");
|
|
1802
|
+
const newCalls = [];
|
|
1803
|
+
const finalizationTransport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
1804
|
+
complete: () => Effect.die("finalization requires detailed provenance"),
|
|
1805
|
+
completeDetailed: (input) => {
|
|
1806
|
+
newCalls.push(input);
|
|
1807
|
+
assert.equal(input.foundryRole, "quality-reviewer", "finalization cannot regenerate any authoring role");
|
|
1808
|
+
assert.equal(input.instructions.includes("Valid-control policy version 2"), validControlReviewVersion === 2);
|
|
1809
|
+
assert.equal(input.reviewInputVersion, reviewInputVersion);
|
|
1810
|
+
assert.ok(Buffer.byteLength(evalAuthoringResponsesRequestBody(input)) <=
|
|
1811
|
+
evalAuthoringRequestByteLimit(input));
|
|
1812
|
+
return Effect.succeed({
|
|
1813
|
+
text: JSON.stringify(modelOutput(input)),
|
|
1814
|
+
callId: "frozen-finalization-review"
|
|
1815
|
+
});
|
|
1816
|
+
}
|
|
1817
|
+
}));
|
|
1818
|
+
const finalized = await Effect.runPromise(finalizeRepositoryGeneratedCaseV1({
|
|
1819
|
+
...restored,
|
|
1820
|
+
reviewOperationId: "frozen-finalization-test:quality-reviewer"
|
|
1821
|
+
}).pipe(Effect.provide(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(finalizationTransport)))));
|
|
1822
|
+
assert.equal(newCalls.length, 1);
|
|
1823
|
+
assert.equal(finalized.generation.qualityReview.verdict, "approve");
|
|
1824
|
+
assert.equal(finalized.generation.benchmarkCase.status, "valid-library");
|
|
1825
|
+
assert.equal(finalized.generation.benchmarkCase.behavioralFingerprint, result.generation.benchmarkCase.behavioralFingerprint);
|
|
1826
|
+
assert.deepEqual(finalized.generation.benchmarkCase.hidden.solutions, result.generation.benchmarkCase.hidden.solutions);
|
|
1827
|
+
assert.deepEqual(finalized.generation.heldOutOracle?.solutions, result.generation.heldOutOracle?.solutions);
|
|
1828
|
+
assert.equal(finalized.generation.benchmarkCase.hidden.observations.length, result.generation.benchmarkCase.hidden.observations.length);
|
|
1829
|
+
assert.equal(finalized.generation.heldOutOracle?.observations.length, result.generation.heldOutOracle?.observations.length);
|
|
1830
|
+
assert.equal(finalized.generation.modelCalls.length, result.generation.modelCalls.length);
|
|
1831
|
+
}
|
|
1832
|
+
finally {
|
|
1833
|
+
await Promise.all([
|
|
1834
|
+
rm(repository.root, { recursive: true, force: true }),
|
|
1835
|
+
rm(archiveRoot, { recursive: true, force: true })
|
|
1836
|
+
]);
|
|
1837
|
+
}
|
|
1838
|
+
});
|
|
1839
|
+
}
|
|
1840
|
+
test("repository foundry refuses to plan paid generation for an unqualified historical seed", async () => {
|
|
1841
|
+
const repository = await createDiscoverableRepository({
|
|
1842
|
+
brokenHistoricalBaseline: true
|
|
1843
|
+
});
|
|
1844
|
+
try {
|
|
1845
|
+
const modelPlan = gpt56RepositoryFoundryModelPlanV1({
|
|
1846
|
+
primaryModel: "openai/gpt-5.6-sol",
|
|
1847
|
+
criticAModel: "openai/gpt-5.6-luna",
|
|
1848
|
+
criticBModel: "openai/gpt-5.6-terra",
|
|
1849
|
+
reviewerModel: "openai/gpt-5.6-terra"
|
|
1850
|
+
});
|
|
1851
|
+
await assert.rejects(Effect.runPromise(planHistoricalRepositoryCaseGenerationV1({
|
|
1852
|
+
repositoryRoot: repository.root,
|
|
1853
|
+
requestedRef: repository.referenceCommit,
|
|
1854
|
+
referenceCommit: repository.referenceCommit,
|
|
1855
|
+
caseId: "case-reject-unqualified-history",
|
|
1856
|
+
modelPlan,
|
|
1857
|
+
maximumHistoryEpisodes: 10,
|
|
1858
|
+
qualificationBatches: 2,
|
|
1859
|
+
qualificationRepetitionsPerBatch: 2,
|
|
1860
|
+
oracleRepetitions: 2,
|
|
1861
|
+
maximumHillClimbAttempts: 2
|
|
1862
|
+
})), /historical seed failed replay qualification before planning: .*initial baseline does not pass/u);
|
|
1863
|
+
}
|
|
1864
|
+
finally {
|
|
1865
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
1866
|
+
}
|
|
1867
|
+
});
|
|
1868
|
+
test("nonexecutable generated fixtures stop before any solution-generation model call", async () => {
|
|
1869
|
+
const repository = await createRepository();
|
|
1870
|
+
try {
|
|
1871
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
1872
|
+
repositoryRoot: repository.root,
|
|
1873
|
+
requestedRef: repository.referenceCommit,
|
|
1874
|
+
maximumHistoryEpisodes: 10
|
|
1875
|
+
}));
|
|
1876
|
+
const episode = map.historyEpisodes.find((candidate) => candidate.commit === repository.referenceCommit);
|
|
1877
|
+
assert.ok(episode);
|
|
1878
|
+
const calls = [];
|
|
1879
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
1880
|
+
complete: (input) => {
|
|
1881
|
+
calls.push(input);
|
|
1882
|
+
const output = modelOutput(input);
|
|
1883
|
+
if (input.foundryRole !== "oracle-designer")
|
|
1884
|
+
return Effect.succeed(JSON.stringify(output));
|
|
1885
|
+
const proposal = output;
|
|
1886
|
+
return Effect.succeed(JSON.stringify({
|
|
1887
|
+
...proposal,
|
|
1888
|
+
overlays: proposal.overlays.map((overlay) => ({
|
|
1889
|
+
...overlay,
|
|
1890
|
+
content: 'import { obsoleteApi } from "./clamp.js"; obsoleteApi();\n'
|
|
1891
|
+
}))
|
|
1892
|
+
}));
|
|
1893
|
+
}
|
|
1894
|
+
}));
|
|
1895
|
+
await assert.rejects(Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
1896
|
+
repositoryRoot: repository.root,
|
|
1897
|
+
operationId: "nonexecutable-fixture-generate-clamp-case",
|
|
1898
|
+
caseId: "case-clamp-inclusive-range",
|
|
1899
|
+
map,
|
|
1900
|
+
seed: seed({
|
|
1901
|
+
initialCommit: repository.initialCommit,
|
|
1902
|
+
referenceCommit: repository.referenceCommit,
|
|
1903
|
+
episodeId: episode.id
|
|
1904
|
+
}),
|
|
1905
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({ primaryModel: "openai/gpt-6-astra" }),
|
|
1906
|
+
maximumHillClimbAttempts: 0
|
|
1907
|
+
}).pipe(Effect.provide(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport))))), /fixture preflight failed.*reference acceptance is required/u);
|
|
1908
|
+
assert.equal(calls.filter((call) => call.foundryRole === "oracle-designer").length, 1);
|
|
1909
|
+
assert.equal(calls.some((call) => [
|
|
1910
|
+
"valid-solution-generator-a",
|
|
1911
|
+
"valid-solution-generator-b",
|
|
1912
|
+
"adversary",
|
|
1913
|
+
"held-out-adversary"
|
|
1914
|
+
].includes(call.foundryRole ?? "")), false);
|
|
1915
|
+
}
|
|
1916
|
+
finally {
|
|
1917
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
1918
|
+
}
|
|
1919
|
+
});
|
|
1920
|
+
test("grounded generation projects reviewed target behavior into downstream roles and rejects missing oracle scope before replay", async () => {
|
|
1921
|
+
const repository = await createRepository();
|
|
1922
|
+
try {
|
|
1923
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
1924
|
+
repositoryRoot: repository.root,
|
|
1925
|
+
requestedRef: repository.referenceCommit,
|
|
1926
|
+
maximumHistoryEpisodes: 10
|
|
1927
|
+
}));
|
|
1928
|
+
const episode = map.historyEpisodes.find((candidate) => candidate.commit === repository.referenceCommit);
|
|
1929
|
+
assert.ok(episode);
|
|
1930
|
+
const originalSeed = seed({ ...repository, episodeId: episode.id });
|
|
1931
|
+
const candidate = {
|
|
1932
|
+
...originalSeed,
|
|
1933
|
+
targetBehavior: originalSeed.targetBehavior.map((behavior) => ({
|
|
1934
|
+
...behavior,
|
|
1935
|
+
description: "Repair clamp boundaries."
|
|
1936
|
+
}))
|
|
1937
|
+
};
|
|
1938
|
+
const behavioralScope = [
|
|
1939
|
+
{
|
|
1940
|
+
id: "lower-input",
|
|
1941
|
+
kind: "changed",
|
|
1942
|
+
behaviorIds: ["inclusive-clamp-range"],
|
|
1943
|
+
description: "Numeric values below zero become zero."
|
|
1944
|
+
},
|
|
1945
|
+
{
|
|
1946
|
+
id: "upper-input",
|
|
1947
|
+
kind: "preserved",
|
|
1948
|
+
behaviorIds: ["inclusive-clamp-range"],
|
|
1949
|
+
description: "Numeric values above one hundred remain capped at one hundred."
|
|
1950
|
+
},
|
|
1951
|
+
{
|
|
1952
|
+
id: "interior-input",
|
|
1953
|
+
kind: "preserved",
|
|
1954
|
+
behaviorIds: ["inclusive-clamp-range"],
|
|
1955
|
+
description: "Numeric values inside the inclusive range remain unchanged."
|
|
1956
|
+
}
|
|
1957
|
+
];
|
|
1958
|
+
const expectedTarget = candidate.targetBehavior.map((behavior) => ({
|
|
1959
|
+
...behavior,
|
|
1960
|
+
description: behavioralScope.map((clause) => clause.description).join(" ")
|
|
1961
|
+
}));
|
|
1962
|
+
const scopeCoverage = behavioralScope.map((clause, index) => ({
|
|
1963
|
+
scopeId: clause.id,
|
|
1964
|
+
outcome: "covered",
|
|
1965
|
+
fixtureIds: [["historical-negative", "upper-boundary", "in-range-metamorphic"][index]],
|
|
1966
|
+
detail: "The bound fixture asserts this observable requirement."
|
|
1967
|
+
}));
|
|
1968
|
+
for (const missingCoverage of [true, false]) {
|
|
1969
|
+
const roles = [];
|
|
1970
|
+
const artifacts = [];
|
|
1971
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
1972
|
+
complete: (input) => {
|
|
1973
|
+
const role = input.foundryRole;
|
|
1974
|
+
roles.push(role);
|
|
1975
|
+
const packet = JSON.parse(input.input);
|
|
1976
|
+
const output = modelOutput(input);
|
|
1977
|
+
if (role === "repository-analyst") {
|
|
1978
|
+
const sourceEvidence = packet
|
|
1979
|
+
.privateReferenceEvidence.sources.filter((source) => source.path === "src/clamp.js")
|
|
1980
|
+
.map((source) => ({ sourceId: source.sourceId, startLine: 1, endLine: 1 }));
|
|
1981
|
+
assert.equal(sourceEvidence.length, 2);
|
|
1982
|
+
return Effect.succeed(JSON.stringify({
|
|
1983
|
+
...output,
|
|
1984
|
+
referenceScope: {
|
|
1985
|
+
clauses: behavioralScope.map((clause) => ({
|
|
1986
|
+
...clause,
|
|
1987
|
+
evidence: sourceEvidence
|
|
1988
|
+
})),
|
|
1989
|
+
unresolvedQuestions: []
|
|
1990
|
+
}
|
|
1991
|
+
}));
|
|
1992
|
+
}
|
|
1993
|
+
assert.deepEqual(packet.targetBehavior, expectedTarget, `${role} retains grounded behavior IDs, flags, evidence, and descriptions`);
|
|
1994
|
+
if (!role.startsWith("trajectory-")) {
|
|
1995
|
+
assert.deepEqual(packet.behavioralScope, behavioralScope);
|
|
1996
|
+
assert.equal(JSON.stringify(packet.behavioralScope).includes("private-reference-"), false);
|
|
1997
|
+
assert.equal(JSON.stringify(packet.behavioralScope).includes(repository.referenceCommit), false);
|
|
1998
|
+
}
|
|
1999
|
+
if (role === "specification-critic-b") {
|
|
2000
|
+
const after = packet.privateReferenceEvidence.sources.find((source) => source.phase === "after" && source.path === "src/clamp.js");
|
|
2001
|
+
return Effect.succeed(JSON.stringify({
|
|
2002
|
+
...output,
|
|
2003
|
+
referenceCompatibility: {
|
|
2004
|
+
outcome: "compatible",
|
|
2005
|
+
detail: "The pinned behavior supports this visible contract.",
|
|
2006
|
+
scopeCoverage: behavioralScope.map((clause) => ({
|
|
2007
|
+
scopeId: clause.id,
|
|
2008
|
+
outcome: "supported",
|
|
2009
|
+
detail: "This scope requirement is supported.",
|
|
2010
|
+
evidence: [{ sourceId: after.sourceId, startLine: 1, endLine: 1 }]
|
|
2011
|
+
}))
|
|
2012
|
+
}
|
|
2013
|
+
}));
|
|
2014
|
+
}
|
|
2015
|
+
if (role === "solution-critic-a" || role === "solution-critic-b") {
|
|
2016
|
+
return Effect.fail(new EvalProjectAuthoringError({
|
|
2017
|
+
operation: "authoring-evaluations",
|
|
2018
|
+
detail: "grounded projection reached solution review"
|
|
2019
|
+
}));
|
|
2020
|
+
}
|
|
2021
|
+
return Effect.succeed(JSON.stringify(role === "oracle-designer" && !missingCoverage
|
|
2022
|
+
? { ...output, scopeCoverage }
|
|
2023
|
+
: output));
|
|
2024
|
+
}
|
|
2025
|
+
}));
|
|
2026
|
+
await assert.rejects(Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
2027
|
+
repositoryRoot: repository.root,
|
|
2028
|
+
operationId: `grounded-projection-${String(missingCoverage)}`,
|
|
2029
|
+
caseId: "case-clamp-inclusive-range",
|
|
2030
|
+
map,
|
|
2031
|
+
seed: candidate,
|
|
2032
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({ primaryModel: "codex/gpt-6-astra" }),
|
|
2033
|
+
maximumSpecificationRevisions: 0,
|
|
2034
|
+
maximumHillClimbAttempts: 0
|
|
2035
|
+
}).pipe(Effect.provide(Layer.mergeAll(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport)), RepositoryFoundryProgressLive(undefined, (artifact) => Effect.sync(() => {
|
|
2036
|
+
artifacts.push(artifact);
|
|
2037
|
+
})))))), missingCoverage
|
|
2038
|
+
? /every grounded behavioral scope clause/u
|
|
2039
|
+
: /grounded projection reached solution review/u);
|
|
2040
|
+
assert.equal(roles.filter((role) => role === "specification-writer").length, 1);
|
|
2041
|
+
if (missingCoverage) {
|
|
2042
|
+
assert.equal(artifacts.some((artifact) => artifact.role === "fixture-validator"), false);
|
|
2043
|
+
assert.equal(roles.some((role) => role.startsWith("valid-solution-")), false);
|
|
2044
|
+
}
|
|
2045
|
+
else {
|
|
2046
|
+
for (const role of [
|
|
2047
|
+
"trajectory-designer",
|
|
2048
|
+
"trajectory-critic",
|
|
2049
|
+
"oracle-designer",
|
|
2050
|
+
"valid-solution-generator-a",
|
|
2051
|
+
"valid-solution-generator-b",
|
|
2052
|
+
"adversary",
|
|
2053
|
+
"solution-critic-a"
|
|
2054
|
+
])
|
|
2055
|
+
assert.ok(roles.includes(role), role);
|
|
2056
|
+
assert.ok(artifacts.some((artifact) => artifact.role === "fixture-validator"));
|
|
2057
|
+
const oracle = artifacts.find((artifact) => artifact.role === "oracle-designer" && artifact.validation === "validated");
|
|
2058
|
+
assert.deepEqual((oracle?.value).scopeCoverage, scopeCoverage);
|
|
2059
|
+
}
|
|
2060
|
+
}
|
|
2061
|
+
assert.equal(candidate.targetBehavior[0].description, "Repair clamp boundaries.");
|
|
2062
|
+
}
|
|
2063
|
+
finally {
|
|
2064
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
2065
|
+
}
|
|
2066
|
+
});
|
|
2067
|
+
test("case generation materializes allowed manifest edits visible only in pinned repository environment context", async () => {
|
|
2068
|
+
const repository = await createRepository();
|
|
2069
|
+
try {
|
|
2070
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
2071
|
+
repositoryRoot: repository.root,
|
|
2072
|
+
requestedRef: repository.referenceCommit,
|
|
2073
|
+
maximumHistoryEpisodes: 10
|
|
2074
|
+
}));
|
|
2075
|
+
const episode = map.historyEpisodes.find((candidate) => candidate.commit === repository.referenceCommit);
|
|
2076
|
+
assert.ok(episode);
|
|
2077
|
+
const candidate = seed({
|
|
2078
|
+
initialCommit: repository.initialCommit,
|
|
2079
|
+
referenceCommit: repository.referenceCommit,
|
|
2080
|
+
episodeId: episode.id
|
|
2081
|
+
});
|
|
2082
|
+
const artifacts = [];
|
|
2083
|
+
let modelManifest = "";
|
|
2084
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
2085
|
+
complete: (input) => {
|
|
2086
|
+
if (input.foundryRole === "solution-critic-a" ||
|
|
2087
|
+
input.foundryRole === "solution-critic-b") {
|
|
2088
|
+
return Effect.fail(new EvalProjectAuthoringError({
|
|
2089
|
+
operation: "authoring-evaluations",
|
|
2090
|
+
detail: "regression reached review after manifest materialization"
|
|
2091
|
+
}));
|
|
2092
|
+
}
|
|
2093
|
+
const output = modelOutput(input);
|
|
2094
|
+
if (input.foundryRole === "repository-analyst") {
|
|
2095
|
+
return Effect.succeed(JSON.stringify({
|
|
2096
|
+
...output,
|
|
2097
|
+
relevantPaths: ["src/clamp.js", "src/clamp.test.js"]
|
|
2098
|
+
}));
|
|
2099
|
+
}
|
|
2100
|
+
if (input.foundryRole !== "adversary")
|
|
2101
|
+
return Effect.succeed(JSON.stringify(output));
|
|
2102
|
+
const packet = JSON.parse(input.input);
|
|
2103
|
+
assert.equal(packet.preChangeSources.some((source) => source.path === "package.json"), false);
|
|
2104
|
+
assert.ok(packet.allowedRewardHackingPaths.includes("package.json"));
|
|
2105
|
+
assert.equal(packet.repositoryEnvironment.commit, repository.initialCommit);
|
|
2106
|
+
modelManifest = packet.repositoryEnvironment.sources.find((source) => source.path === "package.json").content;
|
|
2107
|
+
assert.ok(modelManifest.includes('"test": "node --test src/*.test.js"'));
|
|
2108
|
+
const proposal = output;
|
|
2109
|
+
return Effect.succeed(JSON.stringify({
|
|
2110
|
+
solutions: [
|
|
2111
|
+
...proposal.solutions.map((solution) => solution.kind !== "reward-hacking"
|
|
2112
|
+
? solution
|
|
2113
|
+
: {
|
|
2114
|
+
...solution,
|
|
2115
|
+
fileOverrides: [
|
|
2116
|
+
...solution.fileOverrides,
|
|
2117
|
+
{
|
|
2118
|
+
path: "package.json",
|
|
2119
|
+
edits: [
|
|
2120
|
+
{
|
|
2121
|
+
search: '"test": "node --test src/*.test.js"',
|
|
2122
|
+
replace: '"test": "node --test src/control.test.js"'
|
|
2123
|
+
}
|
|
2124
|
+
]
|
|
2125
|
+
}
|
|
2126
|
+
]
|
|
2127
|
+
}),
|
|
2128
|
+
{
|
|
2129
|
+
id: "upper-overflow-wraparound",
|
|
2130
|
+
family: "overflow-wraparound",
|
|
2131
|
+
kind: "mutation",
|
|
2132
|
+
fileOverrides: [
|
|
2133
|
+
{
|
|
2134
|
+
path: "src/index.js",
|
|
2135
|
+
content: "export const clamp = (value) => value > 100 ? 0 : Math.max(0, value);\n"
|
|
2136
|
+
}
|
|
2137
|
+
]
|
|
2138
|
+
}
|
|
2139
|
+
]
|
|
2140
|
+
}));
|
|
2141
|
+
}
|
|
2142
|
+
}));
|
|
2143
|
+
await assert.rejects(Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
2144
|
+
repositoryRoot: repository.root,
|
|
2145
|
+
operationId: "manifest-context-regression",
|
|
2146
|
+
caseId: "case-clamp-inclusive-range",
|
|
2147
|
+
map,
|
|
2148
|
+
seed: {
|
|
2149
|
+
...candidate,
|
|
2150
|
+
environment: {
|
|
2151
|
+
...candidate.environment,
|
|
2152
|
+
protectedControlPaths: [
|
|
2153
|
+
...candidate.environment.protectedControlPaths,
|
|
2154
|
+
"package.json"
|
|
2155
|
+
]
|
|
2156
|
+
}
|
|
2157
|
+
},
|
|
2158
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({ primaryModel: "openai/gpt-6-astra" }),
|
|
2159
|
+
maximumHillClimbAttempts: 0
|
|
2160
|
+
}).pipe(Effect.provide(Layer.mergeAll(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport)), RepositoryFoundryProgressLive(undefined, (artifact) => Effect.sync(() => {
|
|
2161
|
+
artifacts.push(artifact);
|
|
2162
|
+
})))))), /regression reached review after manifest materialization/u);
|
|
2163
|
+
const materialized = artifacts.find((artifact) => artifact.operationId === "manifest-context-regression:adversary:materialized");
|
|
2164
|
+
assert.ok(materialized);
|
|
2165
|
+
const proposal = materialized.value.proposal;
|
|
2166
|
+
const manifest = proposal.solutions
|
|
2167
|
+
.find((solution) => solution.kind === "reward-hacking")
|
|
2168
|
+
.fileOverrides.find((override) => override.path === "package.json");
|
|
2169
|
+
assert.equal(manifest?.content, modelManifest.replace('"test": "node --test src/*.test.js"', '"test": "node --test src/control.test.js"'));
|
|
2170
|
+
}
|
|
2171
|
+
finally {
|
|
2172
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
2173
|
+
}
|
|
2174
|
+
});
|
|
2175
|
+
for (const scenario of ["family-collision", "duplicate-patch", "shared-repair-budget"]) {
|
|
2176
|
+
test(`independent pair repair preserves isolation and bounded control generation: ${scenario}`, async () => {
|
|
2177
|
+
const repository = await createRepository();
|
|
2178
|
+
try {
|
|
2179
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
2180
|
+
repositoryRoot: repository.root,
|
|
2181
|
+
requestedRef: repository.referenceCommit,
|
|
2182
|
+
maximumHistoryEpisodes: 10
|
|
2183
|
+
}));
|
|
2184
|
+
const episode = map.historyEpisodes.find((entry) => entry.commit === repository.referenceCommit);
|
|
2185
|
+
assert.ok(episode);
|
|
2186
|
+
const calls = [];
|
|
2187
|
+
const solverBPatches = [];
|
|
2188
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
2189
|
+
complete: (input) => {
|
|
2190
|
+
calls.push(input);
|
|
2191
|
+
const output = modelOutput(input);
|
|
2192
|
+
if (input.foundryRole === "valid-solution-generator-a" &&
|
|
2193
|
+
scenario !== "duplicate-patch") {
|
|
2194
|
+
output.solutions[0].family = "Numeric boundary handling";
|
|
2195
|
+
}
|
|
2196
|
+
if (input.foundryRole === "valid-solution-generator-b") {
|
|
2197
|
+
const proposal = output;
|
|
2198
|
+
if (scenario === "duplicate-patch") {
|
|
2199
|
+
const peer = modelOutput({
|
|
2200
|
+
...input,
|
|
2201
|
+
foundryRole: "valid-solution-generator-a"
|
|
2202
|
+
});
|
|
2203
|
+
proposal.solutions[0].fileOverrides = peer.solutions[0].fileOverrides;
|
|
2204
|
+
}
|
|
2205
|
+
else if (!input.operationId.includes(":repair-")) {
|
|
2206
|
+
proposal.solutions[0].family = "Numeric boundary handling";
|
|
2207
|
+
}
|
|
2208
|
+
if (input.operationId.endsWith(":repair-2") && scenario === "shared-repair-budget") {
|
|
2209
|
+
proposal.solutions[0].fileOverrides[0].content +=
|
|
2210
|
+
"\n// Latest independently reviewed proposal.\n";
|
|
2211
|
+
}
|
|
2212
|
+
solverBPatches.push(JSON.stringify(proposal.solutions[0].fileOverrides));
|
|
2213
|
+
}
|
|
2214
|
+
if (input.foundryRole === "adversary") {
|
|
2215
|
+
assert.ok(calls.some((call) => call.operationId.endsWith(":valid-solution-generator-b:repair-1")));
|
|
2216
|
+
}
|
|
2217
|
+
if (input.foundryRole === "solution-critic-a" ||
|
|
2218
|
+
input.foundryRole === "solution-critic-b") {
|
|
2219
|
+
const packet = JSON.parse(input.input);
|
|
2220
|
+
const latest = [...calls]
|
|
2221
|
+
.reverse()
|
|
2222
|
+
.find((call) => call.foundryRole === "valid-solution-generator-b");
|
|
2223
|
+
assert.ok(latest.operationId.includes(":repair-"));
|
|
2224
|
+
if (scenario === "family-collision")
|
|
2225
|
+
return Effect.fail(new EvalProjectAuthoringError({
|
|
2226
|
+
operation: "authoring-evaluations",
|
|
2227
|
+
detail: "pair-repair regression reached fresh critics"
|
|
2228
|
+
}));
|
|
2229
|
+
if (input.operationId.includes(":repair-")) {
|
|
2230
|
+
assert.match(packet.solutions.find((solution) => solution.id === "independent-guard-bounds")
|
|
2231
|
+
.fileOverrides[0].content, /Latest independently reviewed proposal/u);
|
|
2232
|
+
}
|
|
2233
|
+
if (input.foundryRole === "solution-critic-b") {
|
|
2234
|
+
const review = output;
|
|
2235
|
+
const finding = review.reviews.find((entry) => entry.solutionId === "independent-guard-bounds");
|
|
2236
|
+
finding.classification = "uncertain";
|
|
2237
|
+
finding.detail = "Independently justify the visible lower-bound behavior.";
|
|
2238
|
+
}
|
|
2239
|
+
}
|
|
2240
|
+
return Effect.succeed(JSON.stringify(output));
|
|
2241
|
+
}
|
|
2242
|
+
}));
|
|
2243
|
+
await assert.rejects(Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
2244
|
+
repositoryRoot: repository.root,
|
|
2245
|
+
operationId: `pair-repair-${scenario}`,
|
|
2246
|
+
requestedDimension: "Numeric boundary handling",
|
|
2247
|
+
caseId: `case-pair-repair-${scenario}`,
|
|
2248
|
+
map,
|
|
2249
|
+
seed: seed({
|
|
2250
|
+
initialCommit: repository.initialCommit,
|
|
2251
|
+
referenceCommit: repository.referenceCommit,
|
|
2252
|
+
episodeId: episode.id
|
|
2253
|
+
}),
|
|
2254
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({ primaryModel: "openai/gpt-6-astra" }),
|
|
2255
|
+
maximumHillClimbAttempts: 0
|
|
2256
|
+
}).pipe(Effect.provide(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport))))), scenario === "family-collision"
|
|
2257
|
+
? /pair-repair regression reached fresh critics/u
|
|
2258
|
+
: scenario === "duplicate-patch"
|
|
2259
|
+
? /valid-control-repair-exhausted.*valid-control-duplicate-patch/u
|
|
2260
|
+
: /valid-control-repair-exhausted/u);
|
|
2261
|
+
const a = calls.filter((call) => call.foundryRole === "valid-solution-generator-a");
|
|
2262
|
+
const b = calls.filter((call) => call.foundryRole === "valid-solution-generator-b");
|
|
2263
|
+
assert.equal(a.length, 1);
|
|
2264
|
+
assert.equal(b.length, scenario === "family-collision" ? 2 : 3);
|
|
2265
|
+
if (scenario === "family-collision") {
|
|
2266
|
+
assert.equal(solverBPatches[1], solverBPatches[0], "a metadata-only repair may retain an already distinct implementation");
|
|
2267
|
+
}
|
|
2268
|
+
assert.equal(new Set(calls.map((call) => call.operationId)).size, calls.length);
|
|
2269
|
+
assert.equal(calls.some((call) => call.foundryRole === "held-out-adversary"), false);
|
|
2270
|
+
if (scenario === "duplicate-patch") {
|
|
2271
|
+
assert.equal(calls.some((call) => call.foundryRole === "adversary"), false);
|
|
2272
|
+
assert.equal(calls.some((call) => call.foundryRole?.startsWith("solution-critic-")), false);
|
|
2273
|
+
}
|
|
2274
|
+
if (scenario === "shared-repair-budget") {
|
|
2275
|
+
assert.equal(calls.filter((call) => call.foundryRole?.startsWith("solution-critic-")).length, 4);
|
|
2276
|
+
}
|
|
2277
|
+
for (const call of b.slice(1)) {
|
|
2278
|
+
assert.match(call.instructions, /family field names your implementation mechanism/u);
|
|
2279
|
+
assert.match(call.instructions, /metadata-only contract findings, accurately correct your own metadata while preserving a correct implementation/u);
|
|
2280
|
+
const packet = JSON.parse(call.input);
|
|
2281
|
+
assert.deepEqual(Object.keys(packet).sort(), [
|
|
2282
|
+
"allowedSolutionPaths",
|
|
2283
|
+
"preChangeSources",
|
|
2284
|
+
"priorOwnProposal",
|
|
2285
|
+
"repairSequence",
|
|
2286
|
+
"repositoryEnvironment",
|
|
2287
|
+
"requestedDimension",
|
|
2288
|
+
call.operationId.endsWith(":repair-2") && scenario === "shared-repair-budget"
|
|
2289
|
+
? "priorCriticFindings"
|
|
2290
|
+
: "pairContractFindings",
|
|
2291
|
+
"visible"
|
|
2292
|
+
].sort());
|
|
2293
|
+
assert.equal(call.input.includes("independent-composed-bounds"), false);
|
|
2294
|
+
assert.equal(call.input.includes(repository.referenceCommit), false);
|
|
2295
|
+
assert.equal(call.input.includes("rounding-near-miss"), false);
|
|
2296
|
+
assert.equal(call.input.includes("historical-negative"), false);
|
|
2297
|
+
if (packet.pairContractFindings !== undefined) {
|
|
2298
|
+
assert.deepEqual(packet.pairContractFindings.map((entry) => entry.code), scenario === "duplicate-patch"
|
|
2299
|
+
? ["valid-control-duplicate-patch", "valid-control-similar-patch"]
|
|
2300
|
+
: ["valid-control-duplicate-family"]);
|
|
2301
|
+
}
|
|
2302
|
+
}
|
|
2303
|
+
}
|
|
2304
|
+
finally {
|
|
2305
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
2306
|
+
}
|
|
2307
|
+
});
|
|
2308
|
+
}
|
|
2309
|
+
for (const scenario of ["repaired", "exhausted", "wrong-pair", "unknown-source"]) {
|
|
2310
|
+
test(`early implementation independence review preserves correctness and isolation: ${scenario}`, async () => {
|
|
2311
|
+
const repository = await createRepository();
|
|
2312
|
+
try {
|
|
2313
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
2314
|
+
repositoryRoot: repository.root,
|
|
2315
|
+
requestedRef: repository.referenceCommit,
|
|
2316
|
+
maximumHistoryEpisodes: 10
|
|
2317
|
+
}));
|
|
2318
|
+
const episode = map.historyEpisodes.find((entry) => entry.commit === repository.referenceCommit);
|
|
2319
|
+
assert.ok(episode);
|
|
2320
|
+
const calls = [];
|
|
2321
|
+
const stages = [];
|
|
2322
|
+
const artifacts = [];
|
|
2323
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
2324
|
+
complete: (input) => {
|
|
2325
|
+
calls.push(input);
|
|
2326
|
+
const output = modelOutput(input);
|
|
2327
|
+
const repaired = input.operationId.includes(":repair-");
|
|
2328
|
+
if (input.foundryRole === "valid-solution-generator-b" &&
|
|
2329
|
+
(!repaired || scenario === "exhausted")) {
|
|
2330
|
+
const proposal = output;
|
|
2331
|
+
proposal.solutions[0].family = "extracted-range-helper";
|
|
2332
|
+
proposal.solutions[0].fileOverrides[0].content =
|
|
2333
|
+
"function withinRange(value, lower, upper) {\n" +
|
|
2334
|
+
" const atLeastLower = Math.max(value, lower);\n" +
|
|
2335
|
+
" return Math.min(atLeastLower, upper);\n}\n" +
|
|
2336
|
+
"export const clamp = (value) => withinRange(value, 0, 100);\n";
|
|
2337
|
+
}
|
|
2338
|
+
if (input.foundryRole === "solution-critic-a" ||
|
|
2339
|
+
input.foundryRole === "solution-critic-b") {
|
|
2340
|
+
const packet = JSON.parse(input.input);
|
|
2341
|
+
assert.deepEqual(packet.implementationComparison.solutionIds, [
|
|
2342
|
+
"independent-composed-bounds",
|
|
2343
|
+
"independent-guard-bounds"
|
|
2344
|
+
]);
|
|
2345
|
+
assert.match(input.instructions, /two correlated but correct solutions/u);
|
|
2346
|
+
// A disagreement must trigger repair even when the other critic passes.
|
|
2347
|
+
const independent = input.foundryRole === "solution-critic-a" || (scenario === "repaired" && repaired);
|
|
2348
|
+
return Effect.succeed(JSON.stringify({
|
|
2349
|
+
reviews: output.reviews,
|
|
2350
|
+
implementationIndependence: {
|
|
2351
|
+
outcome: independent ? "independent" : "correlated",
|
|
2352
|
+
detail: independent
|
|
2353
|
+
? "Composed numeric bounds and separate early-return guards implement the same visible contract by different mechanisms."
|
|
2354
|
+
: "The helper retains the same min/max mechanism. PRIVATE_PEER_CODE_CANARY",
|
|
2355
|
+
implementations: packet.implementationComparison.solutionIds.map((solutionId, index) => ({
|
|
2356
|
+
solutionId: scenario === "wrong-pair" && index === 0
|
|
2357
|
+
? "unrelated-solution"
|
|
2358
|
+
: solutionId,
|
|
2359
|
+
mechanism: index === 0
|
|
2360
|
+
? "Composed min/max bounds. PRIVATE_PEER_MECHANISM_CANARY"
|
|
2361
|
+
: independent
|
|
2362
|
+
? "Ordered guards return boundary constants or preserve the input."
|
|
2363
|
+
: "Extracted min/max helper. PRIVATE_PEER_MECHANISM_CANARY",
|
|
2364
|
+
sourcePaths: [
|
|
2365
|
+
scenario === "unknown-source" ? "src/unreviewed.js" : "src/clamp.js"
|
|
2366
|
+
]
|
|
2367
|
+
})),
|
|
2368
|
+
correlatedFeatures: independent ? [] : ["helper-extraction"]
|
|
2369
|
+
}
|
|
2370
|
+
}));
|
|
2371
|
+
}
|
|
2372
|
+
return Effect.succeed(JSON.stringify(output));
|
|
2373
|
+
}
|
|
2374
|
+
}));
|
|
2375
|
+
const execution = Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
2376
|
+
repositoryRoot: repository.root,
|
|
2377
|
+
operationId: `early-independence-${scenario}`,
|
|
2378
|
+
caseId: `case-early-independence-${scenario}`,
|
|
2379
|
+
validControlReviewVersion: 1,
|
|
2380
|
+
map,
|
|
2381
|
+
seed: seed({
|
|
2382
|
+
initialCommit: repository.initialCommit,
|
|
2383
|
+
referenceCommit: repository.referenceCommit,
|
|
2384
|
+
episodeId: episode.id
|
|
2385
|
+
}),
|
|
2386
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({ primaryModel: "openai/gpt-6-astra" })
|
|
2387
|
+
}).pipe(Effect.provide(Layer.merge(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport)), RepositoryFoundryProgressLive((event) => Effect.sync(() => {
|
|
2388
|
+
stages.push(event.stage);
|
|
2389
|
+
}), (artifact) => Effect.sync(() => {
|
|
2390
|
+
artifacts.push(artifact);
|
|
2391
|
+
}))))));
|
|
2392
|
+
if (scenario === "repaired") {
|
|
2393
|
+
const result = await execution;
|
|
2394
|
+
assert.equal(result.benchmarkCase.status, "valid-library");
|
|
2395
|
+
assert.ok(stages.includes("tournament"));
|
|
2396
|
+
for (const review of result.solutionReviews) {
|
|
2397
|
+
assert.equal(review.implementationIndependence?.outcome, "independent");
|
|
2398
|
+
assert.ok(review.reviews
|
|
2399
|
+
.filter((entry) => entry.solutionId.startsWith("independent-"))
|
|
2400
|
+
.every((entry) => entry.classification === "valid"));
|
|
2401
|
+
}
|
|
2402
|
+
assert.match(result.validSolutionProposal.solutions[1].fileOverrides[0].content, /if \(value < 0\) return 0/u);
|
|
2403
|
+
}
|
|
2404
|
+
else {
|
|
2405
|
+
await assert.rejects(execution, scenario === "exhausted"
|
|
2406
|
+
? /valid-control-independence-repair-exhausted/u
|
|
2407
|
+
: /must separately assess both current implementations/u);
|
|
2408
|
+
assert.equal(stages.includes("tournament"), false);
|
|
2409
|
+
assert.equal(calls.some((call) => call.foundryRole === "hill-climb-planner"), false);
|
|
2410
|
+
assert.equal(calls.some((call) => call.foundryRole === "held-out-adversary"), false);
|
|
2411
|
+
assert.equal(calls.some((call) => call.foundryRole === "quality-reviewer"), false);
|
|
2412
|
+
}
|
|
2413
|
+
const solverA = calls.filter((call) => call.foundryRole === "valid-solution-generator-a");
|
|
2414
|
+
const solverB = calls.filter((call) => call.foundryRole === "valid-solution-generator-b");
|
|
2415
|
+
const critics = calls.filter((call) => call.foundryRole === "solution-critic-a" || call.foundryRole === "solution-critic-b");
|
|
2416
|
+
assert.equal(solverA.length, 1);
|
|
2417
|
+
assert.equal(solverB.length, scenario === "repaired" ? 2 : scenario === "exhausted" ? 3 : 1);
|
|
2418
|
+
assert.equal(critics.length, scenario === "repaired" ? 4 : scenario === "exhausted" ? 6 : 2);
|
|
2419
|
+
for (const call of solverB.slice(1)) {
|
|
2420
|
+
assert.match(call.instructions, /substantively different mechanism/u);
|
|
2421
|
+
const packet = JSON.parse(call.input);
|
|
2422
|
+
assert.deepEqual(packet.priorOwnProposal.solutions.map((entry) => entry.id), ["independent-guard-bounds"]);
|
|
2423
|
+
assert.deepEqual(packet.priorCriticFindings, [], "correlation is not a wrong label");
|
|
2424
|
+
assert.deepEqual(packet.pairContractFindings, [
|
|
2425
|
+
{
|
|
2426
|
+
code: "valid-control-independence-helper-extraction",
|
|
2427
|
+
detail: "Moving the same behavior into a helper does not provide a different implementation mechanism."
|
|
2428
|
+
}
|
|
2429
|
+
]);
|
|
2430
|
+
for (const secret of [
|
|
2431
|
+
"independent-composed-bounds",
|
|
2432
|
+
"PRIVATE_PEER_CODE_CANARY",
|
|
2433
|
+
"PRIVATE_PEER_MECHANISM_CANARY",
|
|
2434
|
+
repository.referenceCommit,
|
|
2435
|
+
"developmentFixtureProposal",
|
|
2436
|
+
"historical-negative",
|
|
2437
|
+
"rounding-near-miss"
|
|
2438
|
+
]) {
|
|
2439
|
+
assert.equal(call.input.includes(secret), false, `solver received peer material: ${secret}`);
|
|
2440
|
+
}
|
|
2441
|
+
}
|
|
2442
|
+
const privateReviews = artifacts.filter((artifact) => artifact.role === "solution-critic-b" && artifact.validation === "validated");
|
|
2443
|
+
assert.ok(privateReviews.length > 0);
|
|
2444
|
+
const first = privateReviews[0].value;
|
|
2445
|
+
assert.equal(first.implementationIndependence.outcome, "correlated");
|
|
2446
|
+
assert.ok(first.reviews
|
|
2447
|
+
.filter((entry) => entry.solutionId.startsWith("independent-"))
|
|
2448
|
+
.every((entry) => entry.classification === "valid"));
|
|
2449
|
+
assert.match(JSON.stringify(first), /PRIVATE_PEER_CODE_CANARY/u);
|
|
2450
|
+
}
|
|
2451
|
+
finally {
|
|
2452
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
2453
|
+
}
|
|
2454
|
+
});
|
|
2455
|
+
}
|
|
2456
|
+
for (const scenario of [
|
|
2457
|
+
"equivalent",
|
|
2458
|
+
"provided-source-citation",
|
|
2459
|
+
"unprovided-source-citation",
|
|
2460
|
+
"legacy-equivalent",
|
|
2461
|
+
"legacy-provided-source-citation",
|
|
2462
|
+
"wrong-alternative",
|
|
2463
|
+
"biased-oracle",
|
|
2464
|
+
"biased-oracle-exhausted"
|
|
2465
|
+
]) {
|
|
2466
|
+
test(`behavioral valid-control policy preserves executable correctness: ${scenario}`, async () => {
|
|
2467
|
+
const repository = await createRepository();
|
|
2468
|
+
try {
|
|
2469
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
2470
|
+
repositoryRoot: repository.root,
|
|
2471
|
+
requestedRef: repository.referenceCommit,
|
|
2472
|
+
maximumHistoryEpisodes: 10
|
|
2473
|
+
}));
|
|
2474
|
+
const episode = map.historyEpisodes.find((entry) => entry.commit === repository.referenceCommit);
|
|
2475
|
+
const legacy = scenario.startsWith("legacy-");
|
|
2476
|
+
const wrong = scenario === "wrong-alternative";
|
|
2477
|
+
const biased = scenario.startsWith("biased-oracle");
|
|
2478
|
+
const exhausted = scenario === "biased-oracle-exhausted";
|
|
2479
|
+
const providedCitation = scenario === "provided-source-citation" ||
|
|
2480
|
+
scenario === "legacy-provided-source-citation";
|
|
2481
|
+
const unprovidedCitation = scenario === "unprovided-source-citation";
|
|
2482
|
+
const invalidCitation = unprovidedCitation || (legacy && providedCitation);
|
|
2483
|
+
const calls = [];
|
|
2484
|
+
const stages = [];
|
|
2485
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
2486
|
+
complete: (input) => {
|
|
2487
|
+
calls.push(input);
|
|
2488
|
+
const output = modelOutput(input);
|
|
2489
|
+
if (input.foundryRole === "valid-solution-generator-b") {
|
|
2490
|
+
const proposal = output;
|
|
2491
|
+
// Both correct variants use min/max. The alternative is a real executable
|
|
2492
|
+
// helper formulation, not a source fingerprint or family-name distinction.
|
|
2493
|
+
proposal.solutions[0].family = legacy
|
|
2494
|
+
? "extracted-range-helper"
|
|
2495
|
+
: "composed-min-max";
|
|
2496
|
+
proposal.solutions[0].fileOverrides[0].content = wrong
|
|
2497
|
+
? "export function clamp(value) {\n" +
|
|
2498
|
+
" if (value < 0) return 0;\n" +
|
|
2499
|
+
" if (value > 100) return 100;\n" +
|
|
2500
|
+
" return Math.round(value);\n}\n"
|
|
2501
|
+
: "function constrain(value, lower, upper) {\n" +
|
|
2502
|
+
" return Math.min(upper, Math.max(lower, value));\n}\n" +
|
|
2503
|
+
"export const clamp = (value) => constrain(value, 0, 100);\n";
|
|
2504
|
+
}
|
|
2505
|
+
if (input.foundryRole === "oracle-designer") {
|
|
2506
|
+
output.overlays = isolatedClampOverlays(biased ? "implementation-specific" : "repaired");
|
|
2507
|
+
}
|
|
2508
|
+
if (input.foundryRole === "solution-critic-a" ||
|
|
2509
|
+
input.foundryRole === "solution-critic-b") {
|
|
2510
|
+
const packet = JSON.parse(input.input);
|
|
2511
|
+
if (providedCitation) {
|
|
2512
|
+
assert.ok(packet.repositoryEnvironment.sources.some((source) => source.path === "package.json" && typeof source.content === "string"), "supporting source bytes must be present in the actual critic request");
|
|
2513
|
+
}
|
|
2514
|
+
const reviews = output.reviews;
|
|
2515
|
+
if (wrong) {
|
|
2516
|
+
const review = reviews.find((entry) => entry.solutionId === "independent-guard-bounds");
|
|
2517
|
+
review.classification = "wrong";
|
|
2518
|
+
review.detail = "Rounding 25.5 violates exact preservation of an in-range input.";
|
|
2519
|
+
review.violatedBehaviorIds = ["inclusive-clamp-range"];
|
|
2520
|
+
}
|
|
2521
|
+
output.implementationIndependence = {
|
|
2522
|
+
outcome: wrong ? "independent" : "correlated",
|
|
2523
|
+
detail: wrong
|
|
2524
|
+
? "Early-return guards differ from numeric bound composition; this does not establish correctness."
|
|
2525
|
+
: "The helper and direct expression share a min/max mechanism. PRIVATE_POLICY_PEER_CANARY",
|
|
2526
|
+
implementations: packet.implementationComparison.solutionIds.map((solutionId, index) => ({
|
|
2527
|
+
solutionId,
|
|
2528
|
+
mechanism: index === 0
|
|
2529
|
+
? "Direct min/max composition"
|
|
2530
|
+
: wrong
|
|
2531
|
+
? "Early returns with rounding"
|
|
2532
|
+
: "Equivalent helper-based min/max composition",
|
|
2533
|
+
sourcePaths: [
|
|
2534
|
+
"src/clamp.js",
|
|
2535
|
+
...(providedCitation ? ["package.json"] : []),
|
|
2536
|
+
...(unprovidedCitation ? ["src/not-supplied-helper.js"] : [])
|
|
2537
|
+
]
|
|
2538
|
+
})),
|
|
2539
|
+
correlatedFeatures: wrong ? [] : ["helper-extraction"]
|
|
2540
|
+
};
|
|
2541
|
+
if (!legacy) {
|
|
2542
|
+
assert.match(input.instructions, /Mechanism similarity alone is not/u);
|
|
2543
|
+
assert.match(input.instructions, /source-only semantic review/u);
|
|
2544
|
+
}
|
|
2545
|
+
}
|
|
2546
|
+
if (input.foundryRole === "hill-climb-planner") {
|
|
2547
|
+
output.targetedFalseAcceptIds = [];
|
|
2548
|
+
output.targetedFalseRejectIds = ["independent-guard-bounds"];
|
|
2549
|
+
const fixtureProposal = output.fixtureProposal;
|
|
2550
|
+
fixtureProposal.overlays = isolatedClampOverlays(exhausted ? "implementation-specific" : "repaired");
|
|
2551
|
+
}
|
|
2552
|
+
if (input.foundryRole === "quality-reviewer") {
|
|
2553
|
+
assert.match(input.instructions, /Valid-control policy version 2/u);
|
|
2554
|
+
assert.equal(input.instructions.includes("materially similar valid implementations"), false);
|
|
2555
|
+
const review = output;
|
|
2556
|
+
review.findings.find((finding) => finding.axis === "valid-solution-independence").detail =
|
|
2557
|
+
"The isolated solver calls produce semantically correct executable alternatives. Their min/max mechanism is correlated and retained as a limitation. Repeated tournaments accept both implementations and reject the wrong controls.";
|
|
2558
|
+
}
|
|
2559
|
+
return Effect.succeed(JSON.stringify(output));
|
|
2560
|
+
}
|
|
2561
|
+
}));
|
|
2562
|
+
const execution = Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
2563
|
+
repositoryRoot: repository.root,
|
|
2564
|
+
operationId: `behavioral-policy-${scenario}`,
|
|
2565
|
+
caseId: `case-behavioral-policy-${scenario}`,
|
|
2566
|
+
validControlReviewVersion: legacy ? 1 : 2,
|
|
2567
|
+
maximumHillClimbAttempts: exhausted ? 1 : 2,
|
|
2568
|
+
map,
|
|
2569
|
+
seed: seed({ ...repository, episodeId: episode.id }),
|
|
2570
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({ primaryModel: "openai/gpt-6-astra" })
|
|
2571
|
+
}).pipe(Effect.provide(Layer.merge(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport)), RepositoryFoundryProgressLive((event) => Effect.sync(() => {
|
|
2572
|
+
stages.push(event.stage);
|
|
2573
|
+
}))))));
|
|
2574
|
+
if (legacy || wrong || exhausted || invalidCitation) {
|
|
2575
|
+
await assert.rejects(execution, invalidCitation
|
|
2576
|
+
? /must separately assess both current implementations/u
|
|
2577
|
+
: legacy
|
|
2578
|
+
? /valid-control-independence-repair-exhausted/u
|
|
2579
|
+
: wrong
|
|
2580
|
+
? /solution critic A rejected generated solution labels/u
|
|
2581
|
+
: /oracle|adequacy/u);
|
|
2582
|
+
assert.equal(calls.some((call) => call.foundryRole === "quality-reviewer"), false);
|
|
2583
|
+
assert.equal(calls.some((call) => call.foundryRole === "held-out-adversary"), false);
|
|
2584
|
+
if (legacy || wrong || invalidCitation)
|
|
2585
|
+
assert.equal(stages.includes("tournament"), false);
|
|
2586
|
+
}
|
|
2587
|
+
else {
|
|
2588
|
+
const result = await execution;
|
|
2589
|
+
assert.equal(result.benchmarkCase.status, "valid-library");
|
|
2590
|
+
assert.equal(result.benchmarkCase.adequacy.admitted, true);
|
|
2591
|
+
assert.equal(result.heldOutAdequacy.admitted, true);
|
|
2592
|
+
assert.deepEqual(result.benchmarkCase.adequacy.falseRejectedSolutionIds, []);
|
|
2593
|
+
assert.deepEqual(result.benchmarkCase.adequacy.falseAcceptedSolutionIds, []);
|
|
2594
|
+
assert.ok(result.solutionReviews.every((review) => review.implementationIndependence?.outcome === "correlated"));
|
|
2595
|
+
assert.equal(result.validSolutionProposal.solutions[0].family, "composed-min-max");
|
|
2596
|
+
assert.equal(result.validSolutionProposal.solutions[1].family, "composed-min-max");
|
|
2597
|
+
if (biased) {
|
|
2598
|
+
assert.ok(result.hillClimbAttempts[0].beforeAdequacy.falseRejectedSolutionIds.includes("independent-guard-bounds"));
|
|
2599
|
+
assert.equal(result.hillClimbAttempts[0].decision, "accepted");
|
|
2600
|
+
assert.ok(result.fixtureProposal.overlays.every((overlay) => !overlay.content.includes("clamp.toString")));
|
|
2601
|
+
}
|
|
2602
|
+
else {
|
|
2603
|
+
assert.equal(result.hillClimbAttempts.length, 0);
|
|
2604
|
+
}
|
|
2605
|
+
}
|
|
2606
|
+
const solverCalls = calls.filter((call) => call.foundryRole?.startsWith("valid-solution-generator-"));
|
|
2607
|
+
if (!legacy && !wrong)
|
|
2608
|
+
assert.equal(solverCalls.length, 2);
|
|
2609
|
+
for (const call of solverCalls) {
|
|
2610
|
+
for (const privateMaterial of [
|
|
2611
|
+
"PRIVATE_POLICY_PEER_CANARY",
|
|
2612
|
+
"developmentFixtureProposal",
|
|
2613
|
+
"historical-negative",
|
|
2614
|
+
"rounding-near-miss",
|
|
2615
|
+
repository.referenceCommit
|
|
2616
|
+
]) {
|
|
2617
|
+
assert.equal(call.input.includes(privateMaterial), false);
|
|
2618
|
+
}
|
|
2619
|
+
}
|
|
2620
|
+
}
|
|
2621
|
+
finally {
|
|
2622
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
2623
|
+
}
|
|
2624
|
+
});
|
|
2625
|
+
}
|
|
2626
|
+
for (const scenario of [
|
|
2627
|
+
"critic-a-only",
|
|
2628
|
+
"critic-b-only",
|
|
2629
|
+
"missing-rationale",
|
|
2630
|
+
"blank-rationale",
|
|
2631
|
+
"empty-behavior-ids",
|
|
2632
|
+
"unknown-behavior-id",
|
|
2633
|
+
"mixed-known-and-unknown-behavior-ids",
|
|
2634
|
+
"uncertain-valid-control",
|
|
2635
|
+
"mixed-final-adversary",
|
|
2636
|
+
"development-adversary",
|
|
2637
|
+
"semantic-and-independence"
|
|
2638
|
+
]) {
|
|
2639
|
+
test(`semantic valid-control terminal classification stays scoped: ${scenario}`, async () => {
|
|
2640
|
+
const repository = await createRepository();
|
|
2641
|
+
try {
|
|
2642
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
2643
|
+
repositoryRoot: repository.root,
|
|
2644
|
+
requestedRef: repository.referenceCommit,
|
|
2645
|
+
maximumHistoryEpisodes: 10
|
|
2646
|
+
}));
|
|
2647
|
+
const episode = map.historyEpisodes.find((entry) => entry.commit === repository.referenceCommit);
|
|
2648
|
+
assert.ok(episode);
|
|
2649
|
+
const targetSeed = seed({ ...repository, episodeId: episode.id });
|
|
2650
|
+
const behaviorIds = targetSeed.targetBehavior.map((behavior) => behavior.id);
|
|
2651
|
+
const unknownBehaviorId = "PRIVATE_TERMINAL_UNKNOWN_BEHAVIOR";
|
|
2652
|
+
const privateCanary = "PRIVATE_TERMINAL_CRITIC_RATIONALE";
|
|
2653
|
+
const wrongBSource = "export const clamp = (value) => { if (value < 0) return 0; if (value > 100) return 100; return Math.round(value); };\n";
|
|
2654
|
+
const correctBSource = "export const clamp = (value) => { if (value < 0) return 0; if (value > 100) return 100; return value; };\n";
|
|
2655
|
+
const correctAdversarySource = "export const clamp = (value) => Math.max(0, Math.min(100, value));\n";
|
|
2656
|
+
const selectedCritic = scenario === "critic-a-only" ? "solution-critic-a" : "solution-critic-b";
|
|
2657
|
+
const selectedReviewer = selectedCritic === "solution-critic-a" ? "solution critic A" : "solution critic B";
|
|
2658
|
+
const developmentOnly = scenario === "development-adversary";
|
|
2659
|
+
const mixedFinal = scenario === "mixed-final-adversary";
|
|
2660
|
+
const independenceUnresolved = scenario === "semantic-and-independence";
|
|
2661
|
+
const expectsCanonical = scenario === "critic-a-only" || scenario === "critic-b-only";
|
|
2662
|
+
const classification = scenario === "uncertain-valid-control" ? "uncertain" : "wrong";
|
|
2663
|
+
const reportedBehaviorIds = scenario === "empty-behavior-ids"
|
|
2664
|
+
? []
|
|
2665
|
+
: scenario === "unknown-behavior-id"
|
|
2666
|
+
? [unknownBehaviorId]
|
|
2667
|
+
: scenario === "mixed-known-and-unknown-behavior-ids"
|
|
2668
|
+
? [...behaviorIds, unknownBehaviorId]
|
|
2669
|
+
: behaviorIds;
|
|
2670
|
+
assert.ok(behaviorIds.length > 0);
|
|
2671
|
+
assert.equal(behaviorIds.includes(unknownBehaviorId), false);
|
|
2672
|
+
const operationId = `semantic-terminal-scope-${scenario}`;
|
|
2673
|
+
const calls = [];
|
|
2674
|
+
const stages = [];
|
|
2675
|
+
const artifacts = [];
|
|
2676
|
+
const ownBResponses = [];
|
|
2677
|
+
const criticResponses = new Map();
|
|
2678
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
2679
|
+
complete: (input) => {
|
|
2680
|
+
calls.push(input);
|
|
2681
|
+
const output = modelOutput(input);
|
|
2682
|
+
if (input.foundryRole === "valid-solution-generator-b") {
|
|
2683
|
+
const packet = JSON.parse(input.input);
|
|
2684
|
+
if (packet.repairSequence !== undefined) {
|
|
2685
|
+
assert.deepEqual(packet.priorOwnProposal, ownBResponses.at(-1), "the real repair owner supplies B's exact previous raw proposal");
|
|
2686
|
+
}
|
|
2687
|
+
const proposal = output;
|
|
2688
|
+
assert.equal(proposal.solutions[0].id, "independent-guard-bounds");
|
|
2689
|
+
const implementation = proposal.solutions[0].fileOverrides.find((override) => override.path === "src/clamp.js");
|
|
2690
|
+
assert.ok(implementation);
|
|
2691
|
+
if (!developmentOnly)
|
|
2692
|
+
implementation.content = wrongBSource;
|
|
2693
|
+
ownBResponses.push(structuredClone(output));
|
|
2694
|
+
}
|
|
2695
|
+
if (input.foundryRole === "adversary" && (developmentOnly || mixedFinal)) {
|
|
2696
|
+
const proposal = output;
|
|
2697
|
+
const mislabeled = proposal.solutions.find((solution) => solution.id === "public-integration-omission");
|
|
2698
|
+
assert.ok(mislabeled);
|
|
2699
|
+
// This expected-wrong proposal is actually correct. The mixed
|
|
2700
|
+
// scenario's critic notices it only in the last review round.
|
|
2701
|
+
mislabeled.fileOverrides = [
|
|
2702
|
+
{ path: "src/clamp.js", content: correctAdversarySource },
|
|
2703
|
+
{ path: "src/index.js", content: 'export { clamp } from "./clamp.js";\n' }
|
|
2704
|
+
];
|
|
2705
|
+
}
|
|
2706
|
+
if (input.foundryRole === "solution-critic-a" ||
|
|
2707
|
+
input.foundryRole === "solution-critic-b") {
|
|
2708
|
+
const packet = JSON.parse(input.input);
|
|
2709
|
+
const actualB = packet.solutions.find((solution) => solution.id === "independent-guard-bounds");
|
|
2710
|
+
assert.ok(actualB);
|
|
2711
|
+
assert.equal(actualB.fileOverrides.find((override) => override.path === "src/clamp.js")
|
|
2712
|
+
?.content, developmentOnly ? correctBSource : wrongBSource, "critics inspect the real materialized proposal, not a substitute description");
|
|
2713
|
+
const review = output;
|
|
2714
|
+
if (input.foundryRole === selectedCritic && !developmentOnly) {
|
|
2715
|
+
const rejected = review.reviews.find((entry) => entry.solutionId === "independent-guard-bounds");
|
|
2716
|
+
assert.ok(rejected);
|
|
2717
|
+
rejected.classification = classification;
|
|
2718
|
+
rejected.detail =
|
|
2719
|
+
scenario === "missing-rationale"
|
|
2720
|
+
? ""
|
|
2721
|
+
: scenario === "blank-rationale"
|
|
2722
|
+
? " \n\t "
|
|
2723
|
+
: `The actual proposal rounds 25.5 to 26 instead of preserving the in-range input. ${privateCanary}`;
|
|
2724
|
+
rejected.violatedBehaviorIds = [...reportedBehaviorIds];
|
|
2725
|
+
}
|
|
2726
|
+
if (input.foundryRole === selectedCritic &&
|
|
2727
|
+
(developmentOnly || (mixedFinal && input.operationId.endsWith(":repair-2")))) {
|
|
2728
|
+
const actual = packet.solutions.find((solution) => solution.id === "public-integration-omission");
|
|
2729
|
+
assert.ok(actual);
|
|
2730
|
+
assert.deepEqual(actual.fileOverrides, [
|
|
2731
|
+
{ path: "src/clamp.js", content: correctAdversarySource },
|
|
2732
|
+
{ path: "src/index.js", content: 'export { clamp } from "./clamp.js";\n' }
|
|
2733
|
+
]);
|
|
2734
|
+
const mislabeled = review.reviews.find((entry) => entry.solutionId === actual.id);
|
|
2735
|
+
assert.ok(mislabeled);
|
|
2736
|
+
mislabeled.classification = "valid";
|
|
2737
|
+
mislabeled.detail =
|
|
2738
|
+
"Both source overrides provide the complete clamp and its public export.";
|
|
2739
|
+
mislabeled.violatedBehaviorIds = [];
|
|
2740
|
+
}
|
|
2741
|
+
const unresolved = independenceUnresolved && input.foundryRole === selectedCritic;
|
|
2742
|
+
output.implementationIndependence = {
|
|
2743
|
+
outcome: unresolved ? "uncertain" : "independent",
|
|
2744
|
+
detail: unresolved
|
|
2745
|
+
? `The reviewer did not establish mechanism independence. ${privateCanary}`
|
|
2746
|
+
: `Numeric composition and early-return bounds use separate mechanisms. ${privateCanary}`,
|
|
2747
|
+
implementations: packet.implementationComparison.solutionIds.map((solutionId) => ({
|
|
2748
|
+
solutionId,
|
|
2749
|
+
mechanism: solutionId === "independent-composed-bounds"
|
|
2750
|
+
? "Composed numeric bounds."
|
|
2751
|
+
: developmentOnly
|
|
2752
|
+
? "Early-return bounds preserve interior values."
|
|
2753
|
+
: "Early-return bounds with a rounded interior result.",
|
|
2754
|
+
sourcePaths: ["src/clamp.js"]
|
|
2755
|
+
})),
|
|
2756
|
+
correlatedFeatures: unresolved ? ["insufficient-evidence"] : []
|
|
2757
|
+
};
|
|
2758
|
+
criticResponses.set(input.operationId, structuredClone(output));
|
|
2759
|
+
}
|
|
2760
|
+
return Effect.succeed(JSON.stringify(output));
|
|
2761
|
+
}
|
|
2762
|
+
}));
|
|
2763
|
+
const exit = await Effect.runPromiseExit(generateHistoricalRepositoryCaseV1({
|
|
2764
|
+
repositoryRoot: repository.root,
|
|
2765
|
+
operationId,
|
|
2766
|
+
caseId: `case-semantic-terminal-${scenario}`,
|
|
2767
|
+
validControlReviewVersion: independenceUnresolved ? 1 : 2,
|
|
2768
|
+
map,
|
|
2769
|
+
seed: targetSeed,
|
|
2770
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({ primaryModel: "openai/gpt-6-astra" })
|
|
2771
|
+
}).pipe(Effect.provide(Layer.merge(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport)), RepositoryFoundryProgressLive((event) => Effect.sync(() => {
|
|
2772
|
+
stages.push(event.stage);
|
|
2773
|
+
}), (artifact) => Effect.sync(() => {
|
|
2774
|
+
artifacts.push(artifact);
|
|
2775
|
+
}))))));
|
|
2776
|
+
assert.ok(Exit.isFailure(exit), "the actual generation owner must reject the reviewed labels");
|
|
2777
|
+
const error = Cause.squash(exit.cause);
|
|
2778
|
+
assert.ok(error instanceof RepositoryFoundryError);
|
|
2779
|
+
assert.equal(error.operation, "compile-reviewed-case");
|
|
2780
|
+
assert.ok(error.detail.includes(`${selectedReviewer} rejected generated solution labels:`));
|
|
2781
|
+
assert.ok(error.detail.includes(developmentOnly
|
|
2782
|
+
? "public-integration-omission=valid"
|
|
2783
|
+
: `independent-guard-bounds=${classification}`));
|
|
2784
|
+
if (mixedFinal)
|
|
2785
|
+
assert.ok(error.detail.includes("public-integration-omission=valid"));
|
|
2786
|
+
const solverA = calls.filter((call) => call.foundryRole === "valid-solution-generator-a");
|
|
2787
|
+
const solverB = calls.filter((call) => call.foundryRole === "valid-solution-generator-b");
|
|
2788
|
+
const expectedRounds = developmentOnly ? 1 : 3;
|
|
2789
|
+
assert.equal(solverA.length, 1, "B-only review repair must not regenerate the unaffected A");
|
|
2790
|
+
assert.equal(solverB.length, expectedRounds);
|
|
2791
|
+
assert.equal(ownBResponses.length, expectedRounds);
|
|
2792
|
+
assert.deepEqual(solverB.map((call) => JSON.parse(call.input).repairSequence), developmentOnly ? [undefined] : [undefined, 1, 2]);
|
|
2793
|
+
for (const role of ["solution-critic-a", "solution-critic-b"]) {
|
|
2794
|
+
const reviews = calls.filter((call) => call.foundryRole === role);
|
|
2795
|
+
assert.deepEqual(reviews.map((call) => call.operationId), [0, 1, 2].slice(0, expectedRounds).map((round) => `${operationId}:${role}${round === 0 ? "" : `:repair-${round}`}`), "both real critics rerun after each completed B repair");
|
|
2796
|
+
}
|
|
2797
|
+
assert.equal(calls.filter((call) => call.foundryRole === "adversary").length, 1);
|
|
2798
|
+
for (const call of solverB.slice(1)) {
|
|
2799
|
+
const packet = JSON.parse(call.input);
|
|
2800
|
+
assert.deepEqual(packet.priorCriticFindings, [
|
|
2801
|
+
{
|
|
2802
|
+
reviewer: selectedReviewer,
|
|
2803
|
+
solutionId: "independent-guard-bounds",
|
|
2804
|
+
classification,
|
|
2805
|
+
detail: "Reassess your own implementation against the full visible contract and the listed behavior clauses; independent review did not establish this proposal as valid.",
|
|
2806
|
+
violatedBehaviorIds: reportedBehaviorIds.filter((id) => behaviorIds.includes(id))
|
|
2807
|
+
}
|
|
2808
|
+
]);
|
|
2809
|
+
assert.deepEqual(packet.pairContractFindings?.map((finding) => finding.code), independenceUnresolved ? ["valid-control-independence-insufficient-evidence"] : undefined);
|
|
2810
|
+
}
|
|
2811
|
+
for (const call of [...solverA, ...solverB]) {
|
|
2812
|
+
for (const unavailable of [
|
|
2813
|
+
privateCanary,
|
|
2814
|
+
unknownBehaviorId,
|
|
2815
|
+
repository.referenceCommit,
|
|
2816
|
+
"developmentFixtureProposal",
|
|
2817
|
+
"historical-negative",
|
|
2818
|
+
"rounding-near-miss"
|
|
2819
|
+
])
|
|
2820
|
+
assert.equal(call.input.includes(unavailable), false, `solver received ${unavailable}`);
|
|
2821
|
+
}
|
|
2822
|
+
assert.ok(solverB.every((call) => !call.input.includes("independent-composed-bounds")), "B repair receives no peer solution identity");
|
|
2823
|
+
const privateReviews = artifacts.filter((artifact) => artifact.validation === "validated" &&
|
|
2824
|
+
(artifact.role === "solution-critic-a" || artifact.role === "solution-critic-b"));
|
|
2825
|
+
assert.equal(privateReviews.length, expectedRounds * 2);
|
|
2826
|
+
for (const artifact of privateReviews) {
|
|
2827
|
+
assert.deepEqual(artifact.value, criticResponses.get(artifact.operationId));
|
|
2828
|
+
assert.ok(JSON.stringify(artifact.value).includes(privateCanary));
|
|
2829
|
+
}
|
|
2830
|
+
for (const role of [
|
|
2831
|
+
"hill-climb-planner",
|
|
2832
|
+
"held-out-adversary",
|
|
2833
|
+
"held-out-solution-critic-a",
|
|
2834
|
+
"held-out-solution-critic-b",
|
|
2835
|
+
"quality-reviewer"
|
|
2836
|
+
])
|
|
2837
|
+
assert.equal(calls.some((call) => call.foundryRole === role), false);
|
|
2838
|
+
for (const stage of [
|
|
2839
|
+
"tournament",
|
|
2840
|
+
"hill-climb",
|
|
2841
|
+
"held-out-tournament",
|
|
2842
|
+
"quality-review",
|
|
2843
|
+
"admit"
|
|
2844
|
+
])
|
|
2845
|
+
assert.equal(stages.includes(stage), false, "scope checks must stop before tournaments");
|
|
2846
|
+
// Keep desired classification last: a RED first proves the real owner,
|
|
2847
|
+
// reviewer ordering, bounded repair history, and isolation before failing.
|
|
2848
|
+
if (expectsCanonical) {
|
|
2849
|
+
assert.match(error.detail, /^valid-control-repair-exhausted:/u);
|
|
2850
|
+
const original = error.cause;
|
|
2851
|
+
assert.ok(original instanceof RepositoryFoundryError);
|
|
2852
|
+
assert.equal(original.operation, "compile-reviewed-case");
|
|
2853
|
+
assert.ok(original.detail.startsWith(`${selectedReviewer} rejected generated solution labels:`));
|
|
2854
|
+
assert.doesNotMatch(original.detail, /valid-control-repair-exhausted/u);
|
|
2855
|
+
assert.ok(error.detail.includes(original.detail), "retain the original critic detail");
|
|
2856
|
+
}
|
|
2857
|
+
else {
|
|
2858
|
+
assert.doesNotMatch(error.detail, /valid-control-repair-exhausted/u);
|
|
2859
|
+
}
|
|
2860
|
+
}
|
|
2861
|
+
finally {
|
|
2862
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
2863
|
+
}
|
|
2864
|
+
});
|
|
2865
|
+
}
|
|
2866
|
+
for (const scenario of [
|
|
2867
|
+
"early-repair",
|
|
2868
|
+
"exhausted",
|
|
2869
|
+
"freeze-review",
|
|
2870
|
+
"freeze-repair",
|
|
2871
|
+
"freeze-regression",
|
|
2872
|
+
"witness",
|
|
2873
|
+
"witness-repair-false-rejection",
|
|
2874
|
+
"witness-v2-repair-false-rejection",
|
|
2875
|
+
"witness-v2-rejected-then-repair-false-rejection",
|
|
2876
|
+
"witness-v2-nonmonotonic-repair-false-rejection",
|
|
2877
|
+
"witness-v2-late-repair-false-rejection",
|
|
2878
|
+
"witness-budget-repair-false-rejection",
|
|
2879
|
+
"witness-budget-legacy-repair-false-rejection",
|
|
2880
|
+
"witness-budget-exhausted-repair-false-rejection"
|
|
2881
|
+
]) {
|
|
2882
|
+
test(`independent oracle coverage uses bounded repairs and rechecks changed suites: ${scenario}`, async () => {
|
|
2883
|
+
const repository = await createRepository();
|
|
2884
|
+
try {
|
|
2885
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
2886
|
+
repositoryRoot: repository.root,
|
|
2887
|
+
requestedRef: repository.referenceCommit,
|
|
2888
|
+
maximumHistoryEpisodes: 10
|
|
2889
|
+
}));
|
|
2890
|
+
const episode = map.historyEpisodes.find((entry) => entry.commit === repository.referenceCommit);
|
|
2891
|
+
assert.ok(episode);
|
|
2892
|
+
const behavioralScope = [
|
|
2893
|
+
{
|
|
2894
|
+
id: "lower",
|
|
2895
|
+
kind: "changed",
|
|
2896
|
+
behaviorIds: ["inclusive-clamp-range"],
|
|
2897
|
+
description: "Negative numeric inputs become zero."
|
|
2898
|
+
},
|
|
2899
|
+
{
|
|
2900
|
+
id: "upper",
|
|
2901
|
+
kind: "preserved",
|
|
2902
|
+
behaviorIds: ["inclusive-clamp-range"],
|
|
2903
|
+
description: "Inputs above one hundred remain capped at one hundred."
|
|
2904
|
+
},
|
|
2905
|
+
{
|
|
2906
|
+
id: "interior",
|
|
2907
|
+
kind: "preserved",
|
|
2908
|
+
behaviorIds: ["inclusive-clamp-range"],
|
|
2909
|
+
description: "Every input inside the inclusive range remains unchanged, including fractions."
|
|
2910
|
+
}
|
|
2911
|
+
];
|
|
2912
|
+
const fixtureIds = ["historical-negative", "upper-boundary", "in-range-metamorphic"];
|
|
2913
|
+
const scopeCoverage = behavioralScope.map((clause, index) => ({
|
|
2914
|
+
scopeId: clause.id,
|
|
2915
|
+
outcome: "covered",
|
|
2916
|
+
fixtureIds: [fixtureIds[index]],
|
|
2917
|
+
detail: "The associated fixture asserts the required behavior."
|
|
2918
|
+
}));
|
|
2919
|
+
const calls = [];
|
|
2920
|
+
const stages = [];
|
|
2921
|
+
const witnessV2 = scenario.startsWith("witness-v2-");
|
|
2922
|
+
const witnessBudget = scenario.startsWith("witness-budget-");
|
|
2923
|
+
const legacyWitnessBudget = scenario === "witness-budget-legacy-repair-false-rejection";
|
|
2924
|
+
const exhaustedWitnessBudget = scenario === "witness-budget-exhausted-repair-false-rejection";
|
|
2925
|
+
const revisedWitness = [
|
|
2926
|
+
'import assert from "node:assert/strict";',
|
|
2927
|
+
'import test from "node:test";',
|
|
2928
|
+
'import { clamp } from "./index.js";',
|
|
2929
|
+
'test("exact fractional preservation", () => { for (const value of [0.25, 25.5, 99.75]) assert.equal(clamp(value), value); });'
|
|
2930
|
+
].join("\n");
|
|
2931
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
2932
|
+
complete: (input) => {
|
|
2933
|
+
calls.push(input);
|
|
2934
|
+
const role = input.foundryRole;
|
|
2935
|
+
const packet = JSON.parse(input.input);
|
|
2936
|
+
if (role === "oracle-critic") {
|
|
2937
|
+
const beforeFreeze = input.operationId.includes(":before-freeze:");
|
|
2938
|
+
const revision = Number(input.operationId.split(":").at(-1));
|
|
2939
|
+
const revise = witnessBudget
|
|
2940
|
+
? true
|
|
2941
|
+
: scenario.includes("witness-v2-late")
|
|
2942
|
+
? beforeFreeze && revision === 0
|
|
2943
|
+
: scenario === "exhausted" ||
|
|
2944
|
+
(scenario === "early-repair" && revision === 0) ||
|
|
2945
|
+
(scenario.startsWith("witness") && !beforeFreeze) ||
|
|
2946
|
+
((scenario === "freeze-regression" || scenario === "freeze-repair") &&
|
|
2947
|
+
beforeFreeze &&
|
|
2948
|
+
revision === 0);
|
|
2949
|
+
const overlays = packet.fixtureSuite.overlays;
|
|
2950
|
+
if (beforeFreeze && revision === 0) {
|
|
2951
|
+
assert.match(overlays.find((entry) => entry.fixtureIds[0] === "in-range-metamorphic").content, /25\.5/u);
|
|
2952
|
+
}
|
|
2953
|
+
return Effect.succeed(JSON.stringify({
|
|
2954
|
+
verdict: revise ? "revise" : "approve",
|
|
2955
|
+
detail: revise
|
|
2956
|
+
? "Integer examples do not isolate fractional value preservation."
|
|
2957
|
+
: "Current assertions cover the reviewed input partitions.",
|
|
2958
|
+
scopeCoverage: behavioralScope.map((clause, index) => {
|
|
2959
|
+
const gap = revise && clause.id === "interior";
|
|
2960
|
+
const overlay = overlays.find((entry) => entry.fixtureIds.includes(fixtureIds[index]));
|
|
2961
|
+
return {
|
|
2962
|
+
scopeId: clause.id,
|
|
2963
|
+
outcome: gap ? "gap" : "covered",
|
|
2964
|
+
detail: gap
|
|
2965
|
+
? "A rounding implementation can preserve all integer examples."
|
|
2966
|
+
: "The supplied assertion exercises this scope clause.",
|
|
2967
|
+
counterexample: gap
|
|
2968
|
+
? "Use an otherwise-correct implementation that rounds in-range fractions; exercise a fractional input and assert exact preservation."
|
|
2969
|
+
: "",
|
|
2970
|
+
assertionEvidence: gap
|
|
2971
|
+
? []
|
|
2972
|
+
: [
|
|
2973
|
+
{
|
|
2974
|
+
fixtureId: fixtureIds[index],
|
|
2975
|
+
excerpt: overlay.content.split("\n").filter(Boolean).at(-1)
|
|
2976
|
+
}
|
|
2977
|
+
]
|
|
2978
|
+
};
|
|
2979
|
+
})
|
|
2980
|
+
}));
|
|
2981
|
+
}
|
|
2982
|
+
if (scenario === "early-repair" && role.startsWith("valid-solution-generator-")) {
|
|
2983
|
+
return Effect.fail(new EvalProjectAuthoringError({
|
|
2984
|
+
operation: "authoring-evaluations",
|
|
2985
|
+
detail: "coverage repair completed before valid-control generation"
|
|
2986
|
+
}));
|
|
2987
|
+
}
|
|
2988
|
+
const output = modelOutput(input);
|
|
2989
|
+
if (role === "repository-analyst") {
|
|
2990
|
+
const evidence = packet
|
|
2991
|
+
.privateReferenceEvidence.sources.filter((source) => source.path === "src/clamp.js")
|
|
2992
|
+
.map((source) => ({ sourceId: source.sourceId, startLine: 1, endLine: 1 }));
|
|
2993
|
+
return Effect.succeed(JSON.stringify({
|
|
2994
|
+
...output,
|
|
2995
|
+
referenceScope: {
|
|
2996
|
+
clauses: behavioralScope.map((clause) => ({ ...clause, evidence })),
|
|
2997
|
+
unresolvedQuestions: []
|
|
2998
|
+
}
|
|
2999
|
+
}));
|
|
3000
|
+
}
|
|
3001
|
+
if (role === "specification-critic-b") {
|
|
3002
|
+
const after = packet.privateReferenceEvidence.sources.find((source) => source.phase === "after" && source.path === "src/clamp.js");
|
|
3003
|
+
return Effect.succeed(JSON.stringify({
|
|
3004
|
+
...output,
|
|
3005
|
+
referenceCompatibility: {
|
|
3006
|
+
outcome: "compatible",
|
|
3007
|
+
detail: "The pinned reference supports the complete visible contract.",
|
|
3008
|
+
scopeCoverage: behavioralScope.map((clause) => ({
|
|
3009
|
+
scopeId: clause.id,
|
|
3010
|
+
outcome: "supported",
|
|
3011
|
+
detail: "The reference implements this clause.",
|
|
3012
|
+
evidence: [{ sourceId: after.sourceId, startLine: 1, endLine: 1 }]
|
|
3013
|
+
}))
|
|
3014
|
+
}
|
|
3015
|
+
}));
|
|
3016
|
+
}
|
|
3017
|
+
if (role === "oracle-designer") {
|
|
3018
|
+
if (input.schemaName === "routekit_repository_oracle_coverage_witnesses_v1") {
|
|
3019
|
+
return Effect.succeed(JSON.stringify({
|
|
3020
|
+
witnesses: [
|
|
3021
|
+
{
|
|
3022
|
+
scopeId: "interior",
|
|
3023
|
+
explanation: "An isolated rounding edit corrupts in-range fractional values.",
|
|
3024
|
+
fileOverrides: [
|
|
3025
|
+
{
|
|
3026
|
+
path: "src/clamp.js",
|
|
3027
|
+
edits: [
|
|
3028
|
+
{
|
|
3029
|
+
search: "Math.min(100, value)",
|
|
3030
|
+
replace: witnessBudget && input.operationId.includes(":before-freeze:")
|
|
3031
|
+
? "Math.min(100, value === 99.75 ? Math.round(value) : value)"
|
|
3032
|
+
: "Math.min(100, Math.round(value))"
|
|
3033
|
+
}
|
|
3034
|
+
]
|
|
3035
|
+
}
|
|
3036
|
+
],
|
|
3037
|
+
testPath: "src/clamp.test.js",
|
|
3038
|
+
testSource: [
|
|
3039
|
+
'import assert from "node:assert/strict";',
|
|
3040
|
+
'import test from "node:test";',
|
|
3041
|
+
'import { clamp } from "./index.js";',
|
|
3042
|
+
witnessBudget
|
|
3043
|
+
? input.operationId.includes(":initial:author:1")
|
|
3044
|
+
? 'test("exact fractional preservation", () => { const = ; });'
|
|
3045
|
+
: input.operationId.includes(":before-freeze:")
|
|
3046
|
+
? 'test("exact upper-adjacent fractional preservation", () => assert.equal(clamp(99.75), 99.75));'
|
|
3047
|
+
: 'test("exact fractional preservation", () => assert.equal(clamp(25.5), 25.5));'
|
|
3048
|
+
: witnessV2
|
|
3049
|
+
? 'test("exact fractional preservation", () => { assert.equal(clamp(25.5), 25.5); assert.match(clamp.toString(), /Math\\.min/); });'
|
|
3050
|
+
: 'test("exact fractional preservation", () => assert.equal(clamp(25.5), 25.5));'
|
|
3051
|
+
].join("\n"),
|
|
3052
|
+
expectedBehavior: "An in-range fraction remains exactly unchanged.",
|
|
3053
|
+
expectationMode: "preserved"
|
|
3054
|
+
}
|
|
3055
|
+
]
|
|
3056
|
+
}));
|
|
3057
|
+
}
|
|
3058
|
+
const repair = input.operationId.includes(":oracle-coverage-repair:");
|
|
3059
|
+
const repairedOverlays = isolatedClampOverlays(scenario === "freeze-regression" ? "initial" : "repaired").map((overlay) => scenario === "freeze-repair" && overlay.fixtureIds.includes("in-range-metamorphic")
|
|
3060
|
+
? {
|
|
3061
|
+
...overlay,
|
|
3062
|
+
content: `${overlay.content}\ntest("preserves upper-adjacent fractions", () => assert.equal(clamp(99.5), 99.5));`
|
|
3063
|
+
}
|
|
3064
|
+
: overlay);
|
|
3065
|
+
return Effect.succeed(JSON.stringify({
|
|
3066
|
+
...output,
|
|
3067
|
+
...(repair
|
|
3068
|
+
? {
|
|
3069
|
+
overlays: repairedOverlays
|
|
3070
|
+
}
|
|
3071
|
+
: {}),
|
|
3072
|
+
scopeCoverage
|
|
3073
|
+
}));
|
|
3074
|
+
}
|
|
3075
|
+
if (role === "hill-climb-planner") {
|
|
3076
|
+
if (witnessV2) {
|
|
3077
|
+
const revision = Number(input.operationId.split(":").at(-1));
|
|
3078
|
+
if (scenario.includes("witness-v2-late") && revision === 1) {
|
|
3079
|
+
return Effect.succeed(JSON.stringify({
|
|
3080
|
+
...output,
|
|
3081
|
+
fixtureProposal: { ...output.fixtureProposal, scopeCoverage }
|
|
3082
|
+
}));
|
|
3083
|
+
}
|
|
3084
|
+
const request = JSON.parse(input.input);
|
|
3085
|
+
const retained = request.currentFixtureProposal.fixtures.find((fixture) => fixture.id ===
|
|
3086
|
+
(scenario.includes("witness-v2-late")
|
|
3087
|
+
? "coverage-before-freeze-1"
|
|
3088
|
+
: "coverage-initial-1"));
|
|
3089
|
+
const incumbent = request.currentFixtureProposal.overlays.find((overlay) => overlay.fixtureIds[0] === retained.id);
|
|
3090
|
+
assert.match(incumbent.content, /clamp\.toString/u);
|
|
3091
|
+
assert.match(input.instructions, /Transport chunks are not logical event boundaries/u);
|
|
3092
|
+
const base = output.fixtureProposal;
|
|
3093
|
+
const loseDefect = scenario.includes("rejected-then") && revision === 1;
|
|
3094
|
+
const nonmonotonic = scenario.includes("nonmonotonic") && revision === 1;
|
|
3095
|
+
if (revision > 1 && !scenario.includes("witness-v2-late")) {
|
|
3096
|
+
assert.equal(request.priorRejectedAttempts.length, 1);
|
|
3097
|
+
assert.equal(request.priorRejectedAttempts[0].coverageWitnessRevalidation?.status, scenario.includes("rejected-then") ? "rejected" : "passed");
|
|
3098
|
+
}
|
|
3099
|
+
return Effect.succeed(JSON.stringify({
|
|
3100
|
+
...output,
|
|
3101
|
+
fixtureProposal: {
|
|
3102
|
+
...base,
|
|
3103
|
+
fixtures: [...base.fixtures, retained],
|
|
3104
|
+
overlays: [
|
|
3105
|
+
...(nonmonotonic
|
|
3106
|
+
? isolatedClampOverlays("implementation-specific")
|
|
3107
|
+
: base.overlays),
|
|
3108
|
+
{
|
|
3109
|
+
path: "src/clamp.test.js",
|
|
3110
|
+
fixtureIds: [retained.id],
|
|
3111
|
+
content: loseDefect
|
|
3112
|
+
? revisedWitness.replace("[0.25, 25.5, 99.75]", "[1, 25, 99]")
|
|
3113
|
+
: revisedWitness
|
|
3114
|
+
}
|
|
3115
|
+
],
|
|
3116
|
+
scopeCoverage
|
|
3117
|
+
}
|
|
3118
|
+
}));
|
|
3119
|
+
}
|
|
3120
|
+
return Effect.succeed(JSON.stringify({
|
|
3121
|
+
...output,
|
|
3122
|
+
fixtureProposal: { ...output.fixtureProposal, scopeCoverage }
|
|
3123
|
+
}));
|
|
3124
|
+
}
|
|
3125
|
+
if (packet.implementationComparison !== undefined) {
|
|
3126
|
+
return Effect.succeed(JSON.stringify({
|
|
3127
|
+
...output,
|
|
3128
|
+
implementationIndependence: {
|
|
3129
|
+
outcome: "independent",
|
|
3130
|
+
detail: "Numeric min/max composition and early-return guards use different mechanisms.",
|
|
3131
|
+
implementations: packet.implementationComparison.solutionIds.map((solutionId) => ({
|
|
3132
|
+
solutionId,
|
|
3133
|
+
mechanism: solutionId.includes("composed")
|
|
3134
|
+
? "Min/max composition"
|
|
3135
|
+
: "Ordered early returns",
|
|
3136
|
+
sourcePaths: ["src/clamp.js"]
|
|
3137
|
+
})),
|
|
3138
|
+
correlatedFeatures: []
|
|
3139
|
+
}
|
|
3140
|
+
}));
|
|
3141
|
+
}
|
|
3142
|
+
return Effect.succeed(JSON.stringify(output));
|
|
3143
|
+
}
|
|
3144
|
+
}));
|
|
3145
|
+
const execution = Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
3146
|
+
repositoryRoot: repository.root,
|
|
3147
|
+
operationId: `oracle-coverage-${scenario}`,
|
|
3148
|
+
caseId: `case-oracle-coverage-${scenario}`,
|
|
3149
|
+
map,
|
|
3150
|
+
seed: seed({ ...repository, episodeId: episode.id }),
|
|
3151
|
+
validControlReviewVersion: 1,
|
|
3152
|
+
maximumSpecificationRevisions: 2,
|
|
3153
|
+
maximumOracleCoverageRevisions: 2,
|
|
3154
|
+
...(scenario.startsWith("witness")
|
|
3155
|
+
? {
|
|
3156
|
+
oracleCoverageWitnessVersion: witnessV2 || witnessBudget ? 2 : 1,
|
|
3157
|
+
maximumFixtureRepairAttempts: witnessBudget ? (exhaustedWitnessBudget ? 0 : 1) : 2,
|
|
3158
|
+
...(witnessBudget && !legacyWitnessBudget
|
|
3159
|
+
? { oracleCoverageWitnessRepairVersion: 1 }
|
|
3160
|
+
: {})
|
|
3161
|
+
}
|
|
3162
|
+
: {}),
|
|
3163
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({
|
|
3164
|
+
primaryModel: "openai/gpt-6-astra",
|
|
3165
|
+
oracleCriticModel: "openai/gpt-6-astra"
|
|
3166
|
+
})
|
|
3167
|
+
}).pipe(Effect.provide(Layer.merge(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport)), RepositoryFoundryProgressLive((event) => Effect.sync(() => {
|
|
3168
|
+
stages.push(event.stage);
|
|
3169
|
+
}))))));
|
|
3170
|
+
if (legacyWitnessBudget || exhaustedWitnessBudget) {
|
|
3171
|
+
await assert.rejects(execution, legacyWitnessBudget
|
|
3172
|
+
? /oracle-coverage-repair-exhausted/u
|
|
3173
|
+
: /oracle-coverage-witness-exhausted/u);
|
|
3174
|
+
const authorCalls = calls.filter((call) => call.schemaName === "routekit_repository_oracle_coverage_witnesses_v1");
|
|
3175
|
+
assert.equal(authorCalls.length, legacyWitnessBudget ? 2 : 1);
|
|
3176
|
+
assert.equal(calls.some((call) => call.foundryRole === "held-out-adversary"), false);
|
|
3177
|
+
assert.equal(calls.some((call) => call.operationId.includes(":oracle-critic:before-freeze:2")), legacyWitnessBudget);
|
|
3178
|
+
if (exhaustedWitnessBudget) {
|
|
3179
|
+
assert.equal(calls.some((call) => call.foundryRole?.startsWith("valid-solution-generator-")), false);
|
|
3180
|
+
}
|
|
3181
|
+
return;
|
|
3182
|
+
}
|
|
3183
|
+
if (scenario === "freeze-review" ||
|
|
3184
|
+
scenario === "freeze-repair" ||
|
|
3185
|
+
scenario.startsWith("witness")) {
|
|
3186
|
+
const result = await execution;
|
|
3187
|
+
assert.equal(result.benchmarkCase.status, "valid-library");
|
|
3188
|
+
assert.equal(result.hillClimbAttempts.length, scenario === "witness"
|
|
3189
|
+
? 0
|
|
3190
|
+
: scenario.includes("rejected-then") ||
|
|
3191
|
+
scenario.includes("nonmonotonic") ||
|
|
3192
|
+
scenario.includes("witness-v2-late")
|
|
3193
|
+
? 2
|
|
3194
|
+
: 1);
|
|
3195
|
+
assert.deepEqual(result.oracleCoverageReviews?.map((entry) => entry.phase), scenario === "witness"
|
|
3196
|
+
? ["initial"]
|
|
3197
|
+
: scenario.includes("witness-v2-late")
|
|
3198
|
+
? ["initial", "before-freeze", "before-freeze"]
|
|
3199
|
+
: scenario === "freeze-review" ||
|
|
3200
|
+
scenario === "witness-repair-false-rejection" ||
|
|
3201
|
+
witnessV2 ||
|
|
3202
|
+
witnessBudget
|
|
3203
|
+
? ["initial", "before-freeze"]
|
|
3204
|
+
: ["initial", "before-freeze", "before-freeze"]);
|
|
3205
|
+
if (scenario !== "witness")
|
|
3206
|
+
assert.match(result.fixtureProposal.overlays[2].content, /25\.5/u);
|
|
3207
|
+
if (scenario.startsWith("witness")) {
|
|
3208
|
+
const assessment = result.oracleCoverageReviews.find((entry) => entry.executionAssessment !== undefined);
|
|
3209
|
+
assert.equal(assessment.review.verdict, "revise");
|
|
3210
|
+
assert.equal(assessment.executionAssessment?.kind, "resolved-by-execution");
|
|
3211
|
+
assert.equal(assessment.executionAssessment?.witnesses[0].outcome, scenario === "witness" ? "confirmed-escape" : "already-detected");
|
|
3212
|
+
assert.equal(result.fixtureProposal.fixtures.length, witnessBudget ? 5 : 4);
|
|
3213
|
+
const witness = assessment.executionAssessment.witnesses[0];
|
|
3214
|
+
const finalWitness = result.fixtureProposal.overlays.find((overlay) => overlay.fixtureIds[0] === witness.fixture.id);
|
|
3215
|
+
if (witnessV2) {
|
|
3216
|
+
assert.equal(finalWitness.content, revisedWitness);
|
|
3217
|
+
assert.ok(result.fixtureProposal.scopeCoverage
|
|
3218
|
+
?.find((scope) => scope.scopeId === "interior")
|
|
3219
|
+
?.fixtureIds.includes(witness.fixture.id));
|
|
3220
|
+
assert.match(witness.overlay.content, /clamp\.toString/u);
|
|
3221
|
+
assert.equal(result.oracleCoverageWitnessRevisions?.length, result.hillClimbAttempts.length);
|
|
3222
|
+
for (const attempt of result.oracleCoverageWitnessRevisions) {
|
|
3223
|
+
for (const revision of attempt.revisions)
|
|
3224
|
+
assert.deepEqual(revision.beforeOverlay, witness.overlay);
|
|
3225
|
+
}
|
|
3226
|
+
assert.equal(result.oracleCoverageWitnessRevisions.at(-1).decision, "accepted");
|
|
3227
|
+
if (result.hillClimbAttempts.length > 1 && !scenario.includes("witness-v2-late")) {
|
|
3228
|
+
assert.equal(result.oracleCoverageWitnessRevisions[0].decision, "rejected");
|
|
3229
|
+
assert.equal(result.hillClimbAttempts[0].candidateAdequacy === undefined, scenario.includes("rejected-then"));
|
|
3230
|
+
}
|
|
3231
|
+
}
|
|
3232
|
+
else
|
|
3233
|
+
assert.deepEqual(finalWitness, witness.overlay);
|
|
3234
|
+
const quality = calls.find((call) => call.foundryRole === "quality-reviewer");
|
|
3235
|
+
assert.match(quality.input, /resolved-by-execution/u);
|
|
3236
|
+
assert.match(quality.input, /oracleCoverageReviews/u);
|
|
3237
|
+
assert.equal(result.heldOutAdequacy.admitted, true);
|
|
3238
|
+
if (witnessBudget) {
|
|
3239
|
+
const witnessCalls = calls.filter((call) => call.schemaName === "routekit_repository_oracle_coverage_witnesses_v1");
|
|
3240
|
+
assert.equal(witnessCalls.length, 3);
|
|
3241
|
+
assert.equal(calls.some((call) => call.operationId.includes(":oracle-critic:before-freeze:1")), true);
|
|
3242
|
+
assert.equal(result.oracleCoverageReviews.length, 2);
|
|
3243
|
+
assert.equal(result.oracleCoverageReviews[1].executionAssessment.witnesses[0].outcome, "confirmed-escape");
|
|
3244
|
+
const repairedInput = JSON.parse(witnessCalls[1].input);
|
|
3245
|
+
assert.match(JSON.stringify(repairedInput.previousDiagnostics), /unknown/u);
|
|
3246
|
+
assert.equal(result.benchmarkCase.adequacy.falseAcceptedSolutionIds.length, 0);
|
|
3247
|
+
assert.equal(result.benchmarkCase.adequacy.falseRejectedSolutionIds.length, 0);
|
|
3248
|
+
}
|
|
3249
|
+
}
|
|
3250
|
+
if (scenario === "freeze-repair") {
|
|
3251
|
+
assert.match(result.fixtureProposal.overlays[2].content, /99\.5/u);
|
|
3252
|
+
assert.deepEqual(result.benchmarkCase.adequacy.falseAcceptedSolutionIds, []);
|
|
3253
|
+
assert.deepEqual(result.benchmarkCase.adequacy.falseRejectedSolutionIds, []);
|
|
3254
|
+
}
|
|
3255
|
+
}
|
|
3256
|
+
else {
|
|
3257
|
+
await assert.rejects(execution, scenario === "early-repair"
|
|
3258
|
+
? /coverage repair completed before valid-control generation/u
|
|
3259
|
+
: scenario === "exhausted"
|
|
3260
|
+
? /oracle-coverage-repair-exhausted/u
|
|
3261
|
+
: /oracle coverage repair failed development revalidation before freeze/u);
|
|
3262
|
+
assert.equal(calls.some((call) => call.foundryRole === "held-out-adversary"), false);
|
|
3263
|
+
if (scenario !== "freeze-regression")
|
|
3264
|
+
assert.equal(stages.includes("tournament"), false);
|
|
3265
|
+
}
|
|
3266
|
+
const reviews = calls.filter((call) => call.foundryRole === "oracle-critic");
|
|
3267
|
+
const repairs = calls.filter((call) => call.operationId.includes(":oracle-coverage-repair:"));
|
|
3268
|
+
assert.equal(reviews.length, scenario === "witness"
|
|
3269
|
+
? 1
|
|
3270
|
+
: scenario.includes("witness-v2-late")
|
|
3271
|
+
? 3
|
|
3272
|
+
: witnessV2 ||
|
|
3273
|
+
witnessBudget ||
|
|
3274
|
+
["early-repair", "freeze-review", "witness-repair-false-rejection"].includes(scenario)
|
|
3275
|
+
? 2
|
|
3276
|
+
: 3);
|
|
3277
|
+
assert.equal(repairs.length, scenario === "exhausted"
|
|
3278
|
+
? 2
|
|
3279
|
+
: scenario === "freeze-review" || scenario.startsWith("witness")
|
|
3280
|
+
? 0
|
|
3281
|
+
: 1);
|
|
3282
|
+
for (const call of reviews) {
|
|
3283
|
+
assert.match(call.instructions, /otherwise correct implementation could violate only that clause/u);
|
|
3284
|
+
for (const unavailable of [
|
|
3285
|
+
"independent-composed-bounds",
|
|
3286
|
+
"independent-guard-bounds",
|
|
3287
|
+
"rounding-near-miss",
|
|
3288
|
+
"falseAcceptedWrongSolutions",
|
|
3289
|
+
"heldOutAdversaries",
|
|
3290
|
+
"candidateAdequacy"
|
|
3291
|
+
]) {
|
|
3292
|
+
assert.equal(call.input.includes(unavailable), false, `coverage critic received ${unavailable}`);
|
|
3293
|
+
}
|
|
3294
|
+
}
|
|
3295
|
+
for (const call of repairs) {
|
|
3296
|
+
const packet = JSON.parse(call.input);
|
|
3297
|
+
assert.ok(packet.independentCoverageReview);
|
|
3298
|
+
assert.ok(packet.currentFixtureProposal);
|
|
3299
|
+
assert.equal("allValidSolutions" in packet, false);
|
|
3300
|
+
assert.equal("heldOutAdversaries" in packet, false);
|
|
3301
|
+
}
|
|
3302
|
+
}
|
|
3303
|
+
finally {
|
|
3304
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
3305
|
+
}
|
|
3306
|
+
});
|
|
3307
|
+
}
|
|
3308
|
+
test("independent valid-solution criticism triggers an isolated bounded repair and fresh reviews", async () => {
|
|
3309
|
+
const repository = await createRepository();
|
|
3310
|
+
try {
|
|
3311
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
3312
|
+
repositoryRoot: repository.root,
|
|
3313
|
+
requestedRef: repository.referenceCommit,
|
|
3314
|
+
maximumHistoryEpisodes: 10
|
|
3315
|
+
}));
|
|
3316
|
+
const episode = map.historyEpisodes.find((candidate) => candidate.commit === repository.referenceCommit);
|
|
3317
|
+
assert.ok(episode);
|
|
3318
|
+
const calls = [];
|
|
3319
|
+
const artifacts = [];
|
|
3320
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
3321
|
+
complete: (input) => {
|
|
3322
|
+
calls.push(input);
|
|
3323
|
+
const output = modelOutput(input);
|
|
3324
|
+
if (input.foundryRole === "valid-solution-generator-a" &&
|
|
3325
|
+
!input.operationId.includes(":repair-")) {
|
|
3326
|
+
return Effect.succeed(JSON.stringify({
|
|
3327
|
+
solutions: [
|
|
3328
|
+
{
|
|
3329
|
+
id: "independent-composed-bounds",
|
|
3330
|
+
family: "composed-min-max",
|
|
3331
|
+
kind: "independent-valid",
|
|
3332
|
+
fileOverrides: [
|
|
3333
|
+
{
|
|
3334
|
+
path: "src/clamp.js",
|
|
3335
|
+
content: "export const clamp = (value) => Math.min(100, value);\n"
|
|
3336
|
+
},
|
|
3337
|
+
{
|
|
3338
|
+
path: "src/index.js",
|
|
3339
|
+
content: 'export { clamp } from "./clamp.js";\n'
|
|
3340
|
+
}
|
|
3341
|
+
]
|
|
3342
|
+
}
|
|
3343
|
+
]
|
|
3344
|
+
}));
|
|
3345
|
+
}
|
|
3346
|
+
if (input.foundryRole === "solution-critic-b" &&
|
|
3347
|
+
!input.operationId.includes(":repair-")) {
|
|
3348
|
+
const review = output;
|
|
3349
|
+
return Effect.succeed(JSON.stringify({
|
|
3350
|
+
reviews: review.reviews.map((entry) => entry.solutionId === "independent-composed-bounds"
|
|
3351
|
+
? {
|
|
3352
|
+
...entry,
|
|
3353
|
+
classification: "wrong",
|
|
3354
|
+
detail: "The implementation leaves negative inputs unchanged. Copy peer independent-guard-bounds using PRIVATE_PEER_CODE_CANARY.",
|
|
3355
|
+
violatedBehaviorIds: ["inclusive-clamp-range", "PRIVATE_PEER_ID_CANARY"]
|
|
3356
|
+
}
|
|
3357
|
+
: entry)
|
|
3358
|
+
}));
|
|
3359
|
+
}
|
|
3360
|
+
return Effect.succeed(JSON.stringify(output));
|
|
3361
|
+
}
|
|
3362
|
+
}));
|
|
3363
|
+
const result = await Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
3364
|
+
repositoryRoot: repository.root,
|
|
3365
|
+
operationId: "repair-invalid-independent-solution",
|
|
3366
|
+
caseId: "case-clamp-valid-solution-repair",
|
|
3367
|
+
map,
|
|
3368
|
+
seed: seed({
|
|
3369
|
+
initialCommit: repository.initialCommit,
|
|
3370
|
+
referenceCommit: repository.referenceCommit,
|
|
3371
|
+
episodeId: episode.id
|
|
3372
|
+
}),
|
|
3373
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({
|
|
3374
|
+
primaryModel: "openai/gpt-5.6-sol",
|
|
3375
|
+
criticAModel: "openai/gpt-5.6-luna",
|
|
3376
|
+
criticBModel: "openai/gpt-5.6-terra",
|
|
3377
|
+
reviewerModel: "openai/gpt-5.6-terra"
|
|
3378
|
+
})
|
|
3379
|
+
}).pipe(Effect.provide(Layer.mergeAll(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport)), RepositoryFoundryProgressLive(undefined, (artifact) => Effect.sync(() => {
|
|
3380
|
+
artifacts.push(artifact);
|
|
3381
|
+
}))))));
|
|
3382
|
+
assert.equal(result.benchmarkCase.status, "valid-library");
|
|
3383
|
+
const solverACalls = calls.filter((call) => call.foundryRole === "valid-solution-generator-a");
|
|
3384
|
+
const solverBCalls = calls.filter((call) => call.foundryRole === "valid-solution-generator-b");
|
|
3385
|
+
const solutionCriticCalls = calls.filter((call) => call.foundryRole === "solution-critic-a" || call.foundryRole === "solution-critic-b");
|
|
3386
|
+
assert.equal(solverACalls.length, 2);
|
|
3387
|
+
assert.equal(solverBCalls.length, 1);
|
|
3388
|
+
assert.equal(solutionCriticCalls.length, 4);
|
|
3389
|
+
const repairCall = solverACalls.find((call) => call.operationId.endsWith(":valid-solution-generator-a:repair-1"));
|
|
3390
|
+
assert.ok(repairCall);
|
|
3391
|
+
const repairInput = JSON.parse(repairCall.input);
|
|
3392
|
+
assert.deepEqual(repairInput.priorOwnProposal.solutions.map((solution) => solution.id), ["independent-composed-bounds"]);
|
|
3393
|
+
assert.deepEqual(repairInput.priorCriticFindings, [
|
|
3394
|
+
{
|
|
3395
|
+
reviewer: "solution critic B",
|
|
3396
|
+
solutionId: "independent-composed-bounds",
|
|
3397
|
+
classification: "wrong",
|
|
3398
|
+
detail: "Reassess your own implementation against the full visible contract and the listed behavior clauses; independent review did not establish this proposal as valid.",
|
|
3399
|
+
violatedBehaviorIds: ["inclusive-clamp-range"]
|
|
3400
|
+
}
|
|
3401
|
+
]);
|
|
3402
|
+
assert.equal(repairCall.input.includes("independent-guard-bounds"), false, "a solver repair must not see the other solver's proposal");
|
|
3403
|
+
assert.equal(repairCall.input.includes("PRIVATE_PEER_CODE_CANARY"), false);
|
|
3404
|
+
assert.equal(repairCall.input.includes("PRIVATE_PEER_ID_CANARY"), false);
|
|
3405
|
+
const initialCritic = artifacts.find((artifact) => artifact.operationId.endsWith(":solution-critic-b") && artifact.validation === "validated");
|
|
3406
|
+
assert.ok(initialCritic);
|
|
3407
|
+
assert.match(JSON.stringify(initialCritic.value), /PRIVATE_PEER_CODE_CANARY/u);
|
|
3408
|
+
assert.match(JSON.stringify(initialCritic.value), /PRIVATE_PEER_ID_CANARY/u);
|
|
3409
|
+
assert.match(result.validSolutionProposal.solutions[0]?.fileOverrides[0]?.content ?? "", /Math\.max\(0, value\)/u);
|
|
3410
|
+
}
|
|
3411
|
+
finally {
|
|
3412
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
3413
|
+
}
|
|
3414
|
+
});
|
|
3415
|
+
test("independent solution critics reject a behaviorally valid adversary label before oracle hill climbing", async () => {
|
|
3416
|
+
const repository = await createRepository();
|
|
3417
|
+
try {
|
|
3418
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({
|
|
3419
|
+
repositoryRoot: repository.root,
|
|
3420
|
+
requestedRef: repository.referenceCommit,
|
|
3421
|
+
maximumHistoryEpisodes: 10
|
|
3422
|
+
}));
|
|
3423
|
+
const episode = map.historyEpisodes.find((candidate) => candidate.commit === repository.referenceCommit);
|
|
3424
|
+
assert.ok(episode);
|
|
3425
|
+
const transport = Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
3426
|
+
complete: (input) => {
|
|
3427
|
+
const output = modelOutput(input);
|
|
3428
|
+
if (input.foundryRole !== "solution-critic-b") {
|
|
3429
|
+
return Effect.succeed(JSON.stringify(output));
|
|
3430
|
+
}
|
|
3431
|
+
const review = output;
|
|
3432
|
+
return Effect.succeed(JSON.stringify({
|
|
3433
|
+
reviews: review.reviews.map((entry) => entry.solutionId === "public-integration-omission"
|
|
3434
|
+
? {
|
|
3435
|
+
...entry,
|
|
3436
|
+
classification: "valid",
|
|
3437
|
+
detail: "The proposed implementation is behaviorally valid in the supplied repository context.",
|
|
3438
|
+
violatedBehaviorIds: []
|
|
3439
|
+
}
|
|
3440
|
+
: entry)
|
|
3441
|
+
}));
|
|
3442
|
+
}
|
|
3443
|
+
}));
|
|
3444
|
+
await assert.rejects(Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
3445
|
+
repositoryRoot: repository.root,
|
|
3446
|
+
operationId: "reject-mislabeled-adversary",
|
|
3447
|
+
caseId: "case-clamp-adversary-review",
|
|
3448
|
+
map,
|
|
3449
|
+
seed: seed({
|
|
3450
|
+
initialCommit: repository.initialCommit,
|
|
3451
|
+
referenceCommit: repository.referenceCommit,
|
|
3452
|
+
episodeId: episode.id
|
|
3453
|
+
}),
|
|
3454
|
+
modelPlan: gpt56RepositoryFoundryModelPlanV1({
|
|
3455
|
+
primaryModel: "openai/gpt-5.6-sol",
|
|
3456
|
+
criticAModel: "openai/gpt-5.6-luna",
|
|
3457
|
+
criticBModel: "openai/gpt-5.6-terra",
|
|
3458
|
+
reviewerModel: "openai/gpt-5.6-terra"
|
|
3459
|
+
})
|
|
3460
|
+
}).pipe(Effect.provide(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(transport))))), /solution critic B rejected generated solution labels.*public-integration-omission=valid/u);
|
|
3461
|
+
}
|
|
3462
|
+
finally {
|
|
3463
|
+
await rm(repository.root, { recursive: true, force: true });
|
|
3464
|
+
}
|
|
3465
|
+
});
|