@velum-labs/routekit-eval-setup 1.3.2 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/authoring-responses-request.d.ts +5 -0
- package/dist/adapters/authoring-responses-request.js +38 -0
- package/dist/adapters/evaluation-evidence-freshness.d.ts +7 -0
- package/dist/adapters/evaluation-evidence-freshness.js +76 -0
- package/dist/adapters/git-task-history.d.ts +67 -0
- package/dist/adapters/git-task-history.js +171 -0
- package/dist/adapters/integrated-repository-history.d.ts +21 -0
- package/dist/adapters/integrated-repository-history.js +175 -0
- package/dist/adapters/repository-command-diagnostic.d.ts +8 -0
- package/dist/adapters/repository-command-diagnostic.js +46 -0
- package/dist/adapters/repository-command-evidence.d.ts +13 -0
- package/dist/adapters/repository-command-evidence.js +102 -0
- package/dist/adapters/repository-command-runner.d.ts +238 -0
- package/dist/adapters/repository-command-runner.js +1483 -0
- package/dist/adapters/repository-import-context.d.ts +47 -0
- package/dist/adapters/repository-import-context.js +469 -0
- package/dist/adapters/repository-node-test-reporter.d.ts +3 -0
- package/dist/adapters/repository-node-test-reporter.js +27 -0
- package/dist/adapters/repository-review-evidence.d.ts +39 -0
- package/dist/adapters/repository-review-evidence.js +632 -0
- package/dist/adapters/repository-seed-selection.d.ts +7 -0
- package/dist/adapters/repository-seed-selection.js +79 -0
- package/dist/adapters/repository-solution-edits.d.ts +49 -0
- package/dist/adapters/repository-solution-edits.js +136 -0
- package/dist/adapters/repository-vitest-phase-adapter.d.ts +8 -0
- package/dist/adapters/repository-vitest-phase-adapter.js +310 -0
- package/dist/adapters/repository-vitest-reporter.d.ts +24 -0
- package/dist/adapters/repository-vitest-reporter.js +314 -0
- package/dist/adapters/strict-authoring-schema.d.ts +5 -0
- package/dist/adapters/strict-authoring-schema.js +158 -0
- package/dist/adapters/test-discovery.d.ts +30 -0
- package/dist/adapters/test-discovery.js +124 -0
- package/dist/adapters/typescript-repository-index.d.ts +51 -0
- package/dist/adapters/typescript-repository-index.js +226 -0
- package/dist/agentic-capabilities-protocol.d.ts +1373 -0
- package/dist/agentic-capabilities-protocol.js +786 -0
- package/dist/case-checkpoint-store.d.ts +29 -0
- package/dist/case-checkpoint-store.js +133 -0
- package/dist/case-pipeline-protocol-v2.d.ts +184 -0
- package/dist/case-pipeline-protocol-v2.js +193 -0
- package/dist/case-pipeline-protocol.d.ts +2626 -0
- package/dist/case-pipeline-protocol.js +371 -0
- package/dist/effect-api.d.ts +74 -10
- package/dist/effect-api.js +56 -6
- package/dist/errors.d.ts +31 -0
- package/dist/errors.js +10 -0
- package/dist/eval-capability-execution-envelope.d.ts +64 -0
- package/dist/eval-capability-execution-envelope.js +98 -0
- package/dist/eval-capability-policy.d.ts +90 -0
- package/dist/eval-capability-policy.js +107 -0
- package/dist/eval-event-log.d.ts +140 -0
- package/dist/eval-event-log.js +220 -0
- package/dist/evaluation-authoring-policy.d.ts +18 -0
- package/dist/evaluation-authoring-policy.js +19 -0
- package/dist/evaluation-authoring-validation.d.ts +22 -0
- package/dist/evaluation-authoring-validation.js +72 -0
- package/dist/evaluation-evidence.d.ts +20 -0
- package/dist/evaluation-evidence.js +319 -0
- package/dist/evaluation-grader-calibration-protocol.d.ts +108 -0
- package/dist/evaluation-grader-calibration-protocol.js +80 -0
- package/dist/evaluation-grader-calibration.d.ts +18 -0
- package/dist/evaluation-grader-calibration.js +334 -0
- package/dist/evaluation-grading-policy.d.ts +24 -0
- package/dist/evaluation-grading-policy.js +54 -0
- package/dist/evaluation-proposal-policy.d.ts +4 -0
- package/dist/evaluation-proposal-policy.js +91 -0
- package/dist/evaluation-source-retrieval.d.ts +68 -0
- package/dist/evaluation-source-retrieval.js +513 -0
- package/dist/evaluation-structure-policy.d.ts +29 -0
- package/dist/evaluation-structure-policy.js +138 -0
- package/dist/index.d.ts +124 -17
- package/dist/index.js +69 -11
- package/dist/inspection.js +2 -3
- package/dist/project-artifacts.d.ts +7 -2
- package/dist/project-artifacts.js +49 -136
- package/dist/project-authoring.d.ts +66 -5
- package/dist/project-authoring.js +783 -109
- package/dist/project-contracts.d.ts +419 -84
- package/dist/project-contracts.js +160 -52
- package/dist/project-store.js +2 -1
- package/dist/project-workflow.d.ts +5 -4
- package/dist/project-workflow.js +154 -35
- package/dist/repository-adversary-protocol.d.ts +64 -0
- package/dist/repository-adversary-protocol.js +105 -0
- package/dist/repository-behavior-protocol.d.ts +188 -0
- package/dist/repository-behavior-protocol.js +202 -0
- package/dist/repository-benchmark-protocol.d.ts +487 -0
- package/dist/repository-benchmark-protocol.js +96 -0
- package/dist/repository-execution-protocol.d.ts +150 -0
- package/dist/repository-execution-protocol.js +38 -0
- package/dist/repository-fixture-instructions.d.ts +3 -0
- package/dist/repository-fixture-instructions.js +91 -0
- package/dist/repository-fixture-protocol.d.ts +79 -0
- package/dist/repository-fixture-protocol.js +79 -0
- package/dist/repository-foundry-plan-protocol.d.ts +118 -0
- package/dist/repository-foundry-plan-protocol.js +296 -0
- package/dist/repository-foundry-progress-protocol.d.ts +52 -0
- package/dist/repository-foundry-progress-protocol.js +52 -0
- package/dist/repository-improvement-protocol.d.ts +100 -0
- package/dist/repository-improvement-protocol.js +106 -0
- package/dist/repository-language-model-protocol.d.ts +43 -0
- package/dist/repository-language-model-protocol.js +146 -0
- package/dist/repository-oracle-coverage-protocol.d.ts +18 -0
- package/dist/repository-oracle-coverage-protocol.js +39 -0
- package/dist/repository-oracle-execution-binding.d.ts +27 -0
- package/dist/repository-oracle-execution-binding.js +59 -0
- package/dist/repository-oracle-protocol.d.ts +230 -0
- package/dist/repository-oracle-protocol.js +156 -0
- package/dist/repository-oracle-scope-policy.d.ts +22 -0
- package/dist/repository-oracle-scope-policy.js +92 -0
- package/dist/repository-quality-policy.d.ts +15 -0
- package/dist/repository-quality-policy.js +357 -0
- package/dist/repository-routing-benchmark-protocol.d.ts +176 -0
- package/dist/repository-routing-benchmark-protocol.js +103 -0
- package/dist/repository-routing-model-protocol.d.ts +36 -0
- package/dist/repository-routing-model-protocol.js +89 -0
- package/dist/repository-routing-plan-protocol.d.ts +112 -0
- package/dist/repository-routing-plan-protocol.js +58 -0
- package/dist/repository-routing-quality-policy.d.ts +9 -0
- package/dist/repository-routing-quality-policy.js +191 -0
- package/dist/repository-seed-qualification-progress-protocol.d.ts +205 -0
- package/dist/repository-seed-qualification-progress-protocol.js +28 -0
- package/dist/repository-semantic-calibration-protocol.d.ts +768 -0
- package/dist/repository-semantic-calibration-protocol.js +276 -0
- package/dist/repository-semantic-calibration.d.ts +163 -0
- package/dist/repository-semantic-calibration.js +581 -0
- package/dist/repository-specification-contract-facts-protocol.d.ts +224 -0
- package/dist/repository-specification-contract-facts-protocol.js +276 -0
- package/dist/repository-specification-critique-protocol.d.ts +189 -0
- package/dist/repository-specification-critique-protocol.js +103 -0
- package/dist/repository-task-family-protocol.d.ts +24 -0
- package/dist/repository-task-family-protocol.js +37 -0
- package/dist/repository-task-seed-protocol.d.ts +384 -0
- package/dist/repository-task-seed-protocol.js +236 -0
- package/dist/repository-trajectory-protocol.d.ts +20 -0
- package/dist/repository-trajectory-protocol.js +42 -0
- package/dist/service.js +1 -1
- package/dist/services/adversary/service.d.ts +64 -0
- package/dist/services/adversary/service.js +330 -0
- package/dist/services/benchmark-compiler/service.d.ts +450 -0
- package/dist/services/benchmark-compiler/service.js +9 -0
- package/dist/services/budgeted-model/service.d.ts +118 -0
- package/dist/services/budgeted-model/service.js +460 -0
- package/dist/services/case-authoring/service.d.ts +163 -0
- package/dist/services/case-authoring/service.js +1456 -0
- package/dist/services/case-finalization/service.d.ts +283 -0
- package/dist/services/case-finalization/service.js +370 -0
- package/dist/services/case-generation/service.d.ts +619 -0
- package/dist/services/case-generation/service.js +2628 -0
- package/dist/services/case-pipeline/service.d.ts +31 -0
- package/dist/services/case-pipeline/service.js +485 -0
- package/dist/services/case-pipeline-v2/service.d.ts +70 -0
- package/dist/services/case-pipeline-v2/service.js +477 -0
- package/dist/services/command-observability/service.d.ts +13 -0
- package/dist/services/command-observability/service.js +3 -0
- package/dist/services/dimension-labeling/service.d.ts +77 -0
- package/dist/services/dimension-labeling/service.js +188 -0
- package/dist/services/eval-candidate/service.d.ts +208 -0
- package/dist/services/eval-candidate/service.js +64 -0
- package/dist/services/eval-capabilities/service.d.ts +183 -0
- package/dist/services/eval-capabilities/service.js +1433 -0
- package/dist/services/eval-environment/service.d.ts +173 -0
- package/dist/services/eval-environment/service.js +127 -0
- package/dist/services/evidence-reconstruction/service.d.ts +36 -0
- package/dist/services/evidence-reconstruction/service.js +145 -0
- package/dist/services/fixture-builder/service.d.ts +62 -0
- package/dist/services/fixture-builder/service.js +36 -0
- package/dist/services/fixture-validation/service.d.ts +75 -0
- package/dist/services/fixture-validation/service.js +295 -0
- package/dist/services/foundry/service.d.ts +831 -0
- package/dist/services/foundry/service.js +442 -0
- package/dist/services/foundry-progress/service.d.ts +62 -0
- package/dist/services/foundry-progress/service.js +149 -0
- package/dist/services/foundry-v2/service.d.ts +54 -0
- package/dist/services/foundry-v2/service.js +28 -0
- package/dist/services/grounded-authoring/service.d.ts +126 -0
- package/dist/services/grounded-authoring/service.js +822 -0
- package/dist/services/historical-case/service.d.ts +722 -0
- package/dist/services/historical-case/service.js +177 -0
- package/dist/services/improvement-loop/service.d.ts +59 -0
- package/dist/services/improvement-loop/service.js +176 -0
- package/dist/services/language-model/service.d.ts +52 -0
- package/dist/services/language-model/service.js +194 -0
- package/dist/services/oracle-builder/service.d.ts +146 -0
- package/dist/services/oracle-builder/service.js +513 -0
- package/dist/services/oracle-coverage/service.d.ts +28 -0
- package/dist/services/oracle-coverage/service.js +50 -0
- package/dist/services/oracle-coverage-witness/service.d.ts +130 -0
- package/dist/services/oracle-coverage-witness/service.js +538 -0
- package/dist/services/pipeline-challenge/service.d.ts +551 -0
- package/dist/services/pipeline-challenge/service.js +427 -0
- package/dist/services/pipeline-controls/service.d.ts +130 -0
- package/dist/services/pipeline-controls/service.js +483 -0
- package/dist/services/pipeline-oracle/service.d.ts +8 -0
- package/dist/services/pipeline-oracle/service.js +256 -0
- package/dist/services/pipeline-seed/service.d.ts +298 -0
- package/dist/services/pipeline-seed/service.js +428 -0
- package/dist/services/pipeline-spec/service.d.ts +103 -0
- package/dist/services/pipeline-spec/service.js +619 -0
- package/dist/services/pipeline-tournament/service.d.ts +258 -0
- package/dist/services/pipeline-tournament/service.js +476 -0
- package/dist/services/quality-gate/service.d.ts +233 -0
- package/dist/services/quality-gate/service.js +136 -0
- package/dist/services/repository-bundle/service.d.ts +33 -0
- package/dist/services/repository-bundle/service.js +114 -0
- package/dist/services/repository-model/service.d.ts +105 -0
- package/dist/services/repository-model/service.js +250 -0
- package/dist/services/repository-public-artifact/service.d.ts +133 -0
- package/dist/services/repository-public-artifact/service.js +330 -0
- package/dist/services/routing-benchmark/service.d.ts +362 -0
- package/dist/services/routing-benchmark/service.js +96 -0
- package/dist/services/specification-critic/service.d.ts +92 -0
- package/dist/services/specification-critic/service.js +172 -0
- package/dist/services/task-family/service.d.ts +40 -0
- package/dist/services/task-family/service.js +55 -0
- package/dist/services/task-seed/service.d.ts +906 -0
- package/dist/services/task-seed/service.js +1406 -0
- package/dist/services/task-specification/service.d.ts +27 -0
- package/dist/services/task-specification/service.js +40 -0
- package/dist/services/trajectory-policy/service.d.ts +110 -0
- package/dist/services/trajectory-policy/service.js +216 -0
- package/dist/test/agentic-capabilities-protocol.test.d.ts +1 -0
- package/dist/test/agentic-capabilities-protocol.test.js +570 -0
- package/dist/test/agentic-capabilities.test.d.ts +1 -0
- package/dist/test/agentic-capabilities.test.js +1461 -0
- package/dist/test/agentic-environment.test.d.ts +1 -0
- package/dist/test/agentic-environment.test.js +213 -0
- package/dist/test/case-pipeline-foundation.test.d.ts +1 -0
- package/dist/test/case-pipeline-foundation.test.js +535 -0
- package/dist/test/case-pipeline-protocol-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-protocol-v2.test.js +124 -0
- package/dist/test/case-pipeline-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-v2.test.js +286 -0
- package/dist/test/case-pipeline.test.d.ts +1 -0
- package/dist/test/case-pipeline.test.js +851 -0
- package/dist/test/eval-capability-policy.test.d.ts +1 -0
- package/dist/test/eval-capability-policy.test.js +50 -0
- package/dist/test/eval-event-log.test.d.ts +1 -0
- package/dist/test/eval-event-log.test.js +125 -0
- package/dist/test/evaluation-evidence-freshness.test.d.ts +1 -0
- package/dist/test/evaluation-evidence-freshness.test.js +44 -0
- package/dist/test/evaluation-evidence.test.d.ts +1 -0
- package/dist/test/evaluation-evidence.test.js +230 -0
- package/dist/test/evaluation-grader-calibration.test.d.ts +1 -0
- package/dist/test/evaluation-grader-calibration.test.js +373 -0
- package/dist/test/evaluation-proposal-digest.test.d.ts +1 -0
- package/dist/test/evaluation-proposal-digest.test.js +187 -0
- package/dist/test/evaluation-source-retrieval.test.d.ts +1 -0
- package/dist/test/evaluation-source-retrieval.test.js +237 -0
- package/dist/test/evaluation-structure-policy.test.d.ts +1 -0
- package/dist/test/evaluation-structure-policy.test.js +196 -0
- package/dist/test/fixtures/repository-resource-panel.d.ts +39 -0
- package/dist/test/fixtures/repository-resource-panel.js +111 -0
- package/dist/test/fixtures/vitest-boundary-panel.d.ts +84 -0
- package/dist/test/fixtures/vitest-boundary-panel.js +120 -0
- package/dist/test/fixtures/vitest-phase-panel.d.ts +135 -0
- package/dist/test/fixtures/vitest-phase-panel.js +213 -0
- package/dist/test/fixtures/vitest-reporter-results.d.ts +76 -0
- package/dist/test/fixtures/vitest-reporter-results.js +94 -0
- package/dist/test/grounded-authoring.test.d.ts +1 -0
- package/dist/test/grounded-authoring.test.js +565 -0
- package/dist/test/integrated-repository-history.test.d.ts +1 -0
- package/dist/test/integrated-repository-history.test.js +227 -0
- package/dist/test/project-authoring.test.js +593 -43
- package/dist/test/project-workflow.test.js +419 -40
- package/dist/test/repository-authoring-artifacts.test.d.ts +1 -0
- package/dist/test/repository-authoring-artifacts.test.js +185 -0
- package/dist/test/repository-bundle.test.d.ts +1 -0
- package/dist/test/repository-bundle.test.js +52 -0
- package/dist/test/repository-case-generation.test.d.ts +1 -0
- package/dist/test/repository-case-generation.test.js +3465 -0
- package/dist/test/repository-command-diagnostic.test.d.ts +1 -0
- package/dist/test/repository-command-diagnostic.test.js +55 -0
- package/dist/test/repository-command-signals.test.d.ts +1 -0
- package/dist/test/repository-command-signals.test.js +124 -0
- package/dist/test/repository-fixture-scope-coverage.test.d.ts +1 -0
- package/dist/test/repository-fixture-scope-coverage.test.js +127 -0
- package/dist/test/repository-fixture-validation.test.d.ts +1 -0
- package/dist/test/repository-fixture-validation.test.js +362 -0
- package/dist/test/repository-foundry-progress.test.d.ts +1 -0
- package/dist/test/repository-foundry-progress.test.js +110 -0
- package/dist/test/repository-foundry-quality.test.d.ts +1 -0
- package/dist/test/repository-foundry-quality.test.js +1138 -0
- package/dist/test/repository-import-context.test.d.ts +1 -0
- package/dist/test/repository-import-context.test.js +354 -0
- package/dist/test/repository-model-authoring.test.d.ts +1 -0
- package/dist/test/repository-model-authoring.test.js +544 -0
- package/dist/test/repository-model.test.d.ts +1 -0
- package/dist/test/repository-model.test.js +2195 -0
- package/dist/test/repository-node-test-reporter.test.d.ts +1 -0
- package/dist/test/repository-node-test-reporter.test.js +104 -0
- package/dist/test/repository-oracle-concurrency.test.d.ts +1 -0
- package/dist/test/repository-oracle-concurrency.test.js +542 -0
- package/dist/test/repository-oracle-coverage-witness.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage-witness.test.js +511 -0
- package/dist/test/repository-oracle-coverage.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage.test.js +168 -0
- package/dist/test/repository-oracle-evidence.test.d.ts +1 -0
- package/dist/test/repository-oracle-evidence.test.js +185 -0
- package/dist/test/repository-oracle-plan.test.d.ts +1 -0
- package/dist/test/repository-oracle-plan.test.js +176 -0
- package/dist/test/repository-overlay-isolation.test.d.ts +1 -0
- package/dist/test/repository-overlay-isolation.test.js +85 -0
- package/dist/test/repository-preparation-cache.test.d.ts +1 -0
- package/dist/test/repository-preparation-cache.test.js +414 -0
- package/dist/test/repository-public-artifact.test.d.ts +1 -0
- package/dist/test/repository-public-artifact.test.js +273 -0
- package/dist/test/repository-qualification-diagnostics.test.d.ts +1 -0
- package/dist/test/repository-qualification-diagnostics.test.js +524 -0
- package/dist/test/repository-reference-authoring.test.d.ts +1 -0
- package/dist/test/repository-reference-authoring.test.js +1633 -0
- package/dist/test/repository-review-evidence-v2.test.d.ts +1 -0
- package/dist/test/repository-review-evidence-v2.test.js +183 -0
- package/dist/test/repository-review-evidence.test.d.ts +1 -0
- package/dist/test/repository-review-evidence.test.js +124 -0
- package/dist/test/repository-seed-exclusions.test.d.ts +1 -0
- package/dist/test/repository-seed-exclusions.test.js +96 -0
- package/dist/test/repository-seed-selection.test.d.ts +1 -0
- package/dist/test/repository-seed-selection.test.js +504 -0
- package/dist/test/repository-semantic-calibration.test.d.ts +1 -0
- package/dist/test/repository-semantic-calibration.test.js +688 -0
- package/dist/test/repository-solution-edits.test.d.ts +1 -0
- package/dist/test/repository-solution-edits.test.js +377 -0
- package/dist/test/repository-specification-budget.test.d.ts +1 -0
- package/dist/test/repository-specification-budget.test.js +171 -0
- package/dist/test/repository-specification-contract-checkpoint.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-checkpoint.test.js +228 -0
- package/dist/test/repository-specification-contract-facts.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-facts.test.js +177 -0
- package/dist/test/repository-trajectory-authoring.test.d.ts +1 -0
- package/dist/test/repository-trajectory-authoring.test.js +176 -0
- package/dist/test/repository-valid-control-plan.test.d.ts +1 -0
- package/dist/test/repository-valid-control-plan.test.js +45 -0
- package/dist/test/repository-vitest-phase.test.d.ts +1 -0
- package/dist/test/repository-vitest-phase.test.js +848 -0
- package/dist/test/repository-vitest-reporter.test.d.ts +1 -0
- package/dist/test/repository-vitest-reporter.test.js +158 -0
- package/dist/test/repository-workspace-build.test.d.ts +1 -0
- package/dist/test/repository-workspace-build.test.js +160 -0
- package/dist/test/strict-authoring-schema.test.d.ts +1 -0
- package/dist/test/strict-authoring-schema.test.js +169 -0
- package/package.json +48 -6
|
@@ -0,0 +1,483 @@
|
|
|
1
|
+
import { Clock, Effect, Schema } from "effect";
|
|
2
|
+
import { repositoryCommandTestEvidenceV1 } from "../../adapters/repository-command-evidence.js";
|
|
3
|
+
import { replayRepositoryCommandsV1 } from "../../adapters/repository-command-runner.js";
|
|
4
|
+
import { loadPinnedSolutionEditSourcesV1, materializeSolutionFileOverridesV1, RepositorySolutionFileOutputV1 } from "../../adapters/repository-solution-edits.js";
|
|
5
|
+
import { checkpointDigestV1, pipelineStageToProgressStageV1, SolutionCritiqueV1 } from "../../case-pipeline-protocol.js";
|
|
6
|
+
import { RepositoryFoundryError } from "../../errors.js";
|
|
7
|
+
import { repositoryFoundryRoleOutputCeilingV1 } from "../../repository-foundry-plan-protocol.js";
|
|
8
|
+
import { repositoryFoundryModelAssignmentV1 } from "../../repository-language-model-protocol.js";
|
|
9
|
+
import { generateWithFeedbackV1, normalizePipelineFailureV1 } from "../budgeted-model/service.js";
|
|
10
|
+
import { reportFoundryStageV1 } from "../foundry-progress/service.js";
|
|
11
|
+
import { RepositoryFoundryLanguageModel } from "../language-model/service.js";
|
|
12
|
+
const STAGE = "controls";
|
|
13
|
+
const REQUIRED_VALID_CONTROLS = 1;
|
|
14
|
+
const REQUIRED_WRONG_CONTROLS = 1;
|
|
15
|
+
const CONTROL_CRITIC_ROLES = ["solution-critic-a", "solution-critic-b"];
|
|
16
|
+
const SolutionSubmissionV1 = Schema.Struct({
|
|
17
|
+
id: Schema.String,
|
|
18
|
+
family: Schema.String,
|
|
19
|
+
kind: Schema.Literals([
|
|
20
|
+
"independent-valid",
|
|
21
|
+
"simplified-valid",
|
|
22
|
+
"behavior-preserving-valid",
|
|
23
|
+
"mutation",
|
|
24
|
+
"model-near-miss"
|
|
25
|
+
]),
|
|
26
|
+
fileOverrides: Schema.Array(RepositorySolutionFileOutputV1)
|
|
27
|
+
});
|
|
28
|
+
const SolutionCritiqueSubmissionV1 = Schema.Struct({
|
|
29
|
+
critiques: Schema.Array(SolutionCritiqueV1)
|
|
30
|
+
});
|
|
31
|
+
const EXACT_EDIT_INSTRUCTIONS = `Return edits against the ORIGINAL pinned pre-change source. For an existing
|
|
32
|
+
file, prefer {path, edits:[{search, replace}]}. Every search must be nonempty verbatim text that occurs exactly
|
|
33
|
+
once in both the supplied complete source and the pinned file, include enough unchanged context to be unique,
|
|
34
|
+
and refer to the original source rather than another edit's output. Edits in one file must not overlap. Preserve
|
|
35
|
+
all bytes outside the requested replacements. A complete small replacement may use {path, content}. Do not
|
|
36
|
+
modify tests, manifests, lockfiles, generated controls, or any path outside allowedSolutionPaths. Return one
|
|
37
|
+
structured solution submission. Invalid structure, unmatched anchors, protected paths, failed materialization,
|
|
38
|
+
failed validation commands, or failed isolated execution are host findings: revise your own prior submission
|
|
39
|
+
and submit again.`;
|
|
40
|
+
const VALID_A_INSTRUCTIONS = `You are blind solver A for a repository coding benchmark. Solve only the visible
|
|
41
|
+
task from the complete dimension contract, grounded public scope, and pinned pre-change sources. Produce one
|
|
42
|
+
behaviorally correct implementation. Preserve every stated boundary and unrelated behavior. You cannot see the
|
|
43
|
+
reference change, hidden fixtures, oracle, another solver, or any tournament result. Do not infer or request
|
|
44
|
+
those materials. The family field is a concise description of your own implementation mechanism and is never
|
|
45
|
+
used as correctness evidence.`;
|
|
46
|
+
const ADVERSARY_INSTRUCTIONS = `You are a blind adversarial solver for a repository coding benchmark. Using
|
|
47
|
+
only the visible task, complete dimension contract, grounded public scope, and pinned pre-change sources,
|
|
48
|
+
produce one plausible implementation that builds and passes the repository's ordinary sanity commands but is
|
|
49
|
+
behaviorally wrong for one explicitly stated visible requirement. Choose kind mutation or model-near-miss.
|
|
50
|
+
Preserve unrelated behavior. Do not modify tests or protected controls. You cannot see the reference change,
|
|
51
|
+
hidden fixtures, oracle, valid solvers, or tournament outcomes. Do not rely on test detection, fixture names,
|
|
52
|
+
source fingerprints, process timing, or harness manipulation. The family field describes the concrete defect
|
|
53
|
+
mechanism and is not compared with any other family.`;
|
|
54
|
+
const CONTROL_CRITIC_INSTRUCTIONS = `You are an independent fixture-blind correctness critic for repository
|
|
55
|
+
benchmark controls. Classify every supplied implementation against only the visible task, complete dimension
|
|
56
|
+
contract, grounded public scope, target behaviors, and pinned pre-change sources. Inspect the actual complete
|
|
57
|
+
file overrides. You are not shown proposal labels, the reference change, hidden fixtures, oracle, another
|
|
58
|
+
critic, or tournament outcomes. A different implementation mechanism is not a correctness defect. Classify
|
|
59
|
+
wrong only when the implementation violates a named user-observable target behavior; list every such behavior
|
|
60
|
+
id. Use uncertain when the supplied source and contract do not justify valid or wrong. Return exactly one
|
|
61
|
+
critique for every supplied solution. The host records this review as advisory development adjudication;
|
|
62
|
+
execution, not your opinion, remains the admission gate.`;
|
|
63
|
+
const failure = (detail, cause) => new RepositoryFoundryError({
|
|
64
|
+
operation: "run-pipeline-stage",
|
|
65
|
+
detail,
|
|
66
|
+
...(cause === undefined ? {} : { cause })
|
|
67
|
+
});
|
|
68
|
+
const describe = (cause) => cause instanceof Error && cause.message.trim().length > 0 ? cause.message : String(cause);
|
|
69
|
+
const assignmentOf = (input) => Effect.try({
|
|
70
|
+
try: () => repositoryFoundryModelAssignmentV1(input.plan, input.role),
|
|
71
|
+
catch: (cause) => failure(`could not resolve ${input.role} model assignment`, cause)
|
|
72
|
+
});
|
|
73
|
+
const solutionPathsOf = (input) => {
|
|
74
|
+
const protectedPaths = new Set(input.seed.environment.protectedControlPaths);
|
|
75
|
+
const testPaths = new Set(input.seed.capabilityEvidence
|
|
76
|
+
.filter((evidence) => evidence.kind === "test" && evidence.path !== undefined)
|
|
77
|
+
.map((evidence) => evidence.path));
|
|
78
|
+
return new Set(input.contextSources
|
|
79
|
+
.map((source) => source.path)
|
|
80
|
+
.filter((path) => !protectedPaths.has(path) && !testPaths.has(path)));
|
|
81
|
+
};
|
|
82
|
+
const submissionFindings = (input) => {
|
|
83
|
+
const findings = [];
|
|
84
|
+
const expectedKind = input.expected === "valid"
|
|
85
|
+
? input.submission.kind === "independent-valid" ||
|
|
86
|
+
input.submission.kind === "simplified-valid" ||
|
|
87
|
+
input.submission.kind === "behavior-preserving-valid"
|
|
88
|
+
: input.submission.kind === "mutation" || input.submission.kind === "model-near-miss";
|
|
89
|
+
if (input.submission.id.trim().length === 0) {
|
|
90
|
+
findings.push({ kind: "contract", message: "solution id must be non-empty" });
|
|
91
|
+
}
|
|
92
|
+
else if (input.submission.id === "historical-reference" ||
|
|
93
|
+
input.submission.id === "historical-defect" ||
|
|
94
|
+
input.existingIds.has(input.submission.id)) {
|
|
95
|
+
findings.push({
|
|
96
|
+
kind: "contract",
|
|
97
|
+
message: `solution id ${JSON.stringify(input.submission.id)} is reserved or already accepted`
|
|
98
|
+
});
|
|
99
|
+
}
|
|
100
|
+
if (input.submission.family.trim().length === 0) {
|
|
101
|
+
findings.push({ kind: "contract", message: "solution family must describe its mechanism" });
|
|
102
|
+
}
|
|
103
|
+
if (!expectedKind) {
|
|
104
|
+
findings.push({
|
|
105
|
+
kind: "contract",
|
|
106
|
+
message: input.expected === "valid"
|
|
107
|
+
? "a valid solver must use independent-valid, simplified-valid, or behavior-preserving-valid"
|
|
108
|
+
: "the blind adversary must use mutation or model-near-miss"
|
|
109
|
+
});
|
|
110
|
+
}
|
|
111
|
+
if (input.submission.fileOverrides.length === 0) {
|
|
112
|
+
findings.push({ kind: "contract", message: "solution must contain at least one source edit" });
|
|
113
|
+
}
|
|
114
|
+
const seen = new Set();
|
|
115
|
+
for (const override of input.submission.fileOverrides) {
|
|
116
|
+
if (seen.has(override.path)) {
|
|
117
|
+
findings.push({
|
|
118
|
+
kind: "materialization",
|
|
119
|
+
path: override.path,
|
|
120
|
+
message: "solution contains duplicate overrides for one path"
|
|
121
|
+
});
|
|
122
|
+
}
|
|
123
|
+
seen.add(override.path);
|
|
124
|
+
if (input.protectedPaths.has(override.path)) {
|
|
125
|
+
findings.push({
|
|
126
|
+
kind: "contract",
|
|
127
|
+
path: override.path,
|
|
128
|
+
message: "solution modifies a protected repository control"
|
|
129
|
+
});
|
|
130
|
+
}
|
|
131
|
+
else if (!input.allowedPaths.has(override.path)) {
|
|
132
|
+
findings.push({
|
|
133
|
+
kind: "contract",
|
|
134
|
+
path: override.path,
|
|
135
|
+
message: "solution modifies a path that was not supplied to this blind role"
|
|
136
|
+
});
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
return findings;
|
|
140
|
+
};
|
|
141
|
+
const commandDetail = (result) => [
|
|
142
|
+
`recipe=${result.recipeId}`,
|
|
143
|
+
`stage=${result.stage}`,
|
|
144
|
+
`exit=${String(result.exitCode)}`,
|
|
145
|
+
`timeout=${String(result.timedOut)}`,
|
|
146
|
+
`gradeExecuted=${String(result.gradeExecuted)}`,
|
|
147
|
+
`stdout=${result.stdout.slice(-2_000)}`,
|
|
148
|
+
`stderr=${result.stderr.slice(-2_000)}`
|
|
149
|
+
].join("; ");
|
|
150
|
+
const executionFindings = (input) => {
|
|
151
|
+
const findings = [];
|
|
152
|
+
for (const result of input.results) {
|
|
153
|
+
const preparationFailure = result.trustedPreparationResults.find((candidate) => candidate.exitCode !== 0 || candidate.timedOut);
|
|
154
|
+
if (preparationFailure !== undefined) {
|
|
155
|
+
findings.push({
|
|
156
|
+
kind: "execution",
|
|
157
|
+
message: `trusted preparation did not complete: recipe=${preparationFailure.recipeId}; exit=${String(preparationFailure.exitCode)}; timeout=${String(preparationFailure.timedOut)}`
|
|
158
|
+
});
|
|
159
|
+
}
|
|
160
|
+
for (const validation of result.candidateValidationResults) {
|
|
161
|
+
if (validation.exitCode !== 0 || validation.timedOut) {
|
|
162
|
+
findings.push({
|
|
163
|
+
kind: "compile",
|
|
164
|
+
message: `candidate validation failed: recipe=${validation.recipeId}; exit=${String(validation.exitCode)}; timeout=${String(validation.timedOut)}; stdout=${validation.stdout.slice(-2_000)}; stderr=${validation.stderr.slice(-2_000)}`
|
|
165
|
+
});
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
const evidence = repositoryCommandTestEvidenceV1(result, result.recipeId);
|
|
169
|
+
if (evidence.outcome !== "pass") {
|
|
170
|
+
findings.push({
|
|
171
|
+
kind: "execution",
|
|
172
|
+
message: `${commandDetail(result)}; evidence=${evidence.detail}`
|
|
173
|
+
});
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
const observed = new Set(input.results.filter((result) => result.gradeExecuted).map((result) => result.recipeId));
|
|
177
|
+
for (const recipeId of input.expectedRecipeIds) {
|
|
178
|
+
if (!observed.has(recipeId)) {
|
|
179
|
+
findings.push({
|
|
180
|
+
kind: "execution",
|
|
181
|
+
message: `isolated sanity execution did not reach reviewed recipe ${recipeId}`
|
|
182
|
+
});
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
return findings;
|
|
186
|
+
};
|
|
187
|
+
const materializeSubmission = (input) => Effect.try({
|
|
188
|
+
try: () => {
|
|
189
|
+
const materialized = materializeSolutionFileOverridesV1({
|
|
190
|
+
fileOverrides: input.submission.fileOverrides,
|
|
191
|
+
sources: input.sources,
|
|
192
|
+
allowedPaths: input.allowedPaths
|
|
193
|
+
});
|
|
194
|
+
return {
|
|
195
|
+
id: input.submission.id,
|
|
196
|
+
family: input.submission.family,
|
|
197
|
+
kind: input.submission.kind,
|
|
198
|
+
expectedClass: input.expected,
|
|
199
|
+
commit: input.initialCommit,
|
|
200
|
+
fileOverrides: materialized.fileOverrides
|
|
201
|
+
};
|
|
202
|
+
},
|
|
203
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
204
|
+
? cause
|
|
205
|
+
: failure("solution exact-edit materialization failed", cause)
|
|
206
|
+
});
|
|
207
|
+
const checkpointOf = (input) => ({
|
|
208
|
+
version: 1,
|
|
209
|
+
stage: STAGE,
|
|
210
|
+
parentDigest: input.parentDigest,
|
|
211
|
+
createdAt: input.createdAt,
|
|
212
|
+
valid: [...input.valid],
|
|
213
|
+
wrong: [...input.wrong],
|
|
214
|
+
critiques: [...input.critiques],
|
|
215
|
+
regenerations: input.regenerations
|
|
216
|
+
});
|
|
217
|
+
const critiqueFindings = (input) => {
|
|
218
|
+
const findings = [];
|
|
219
|
+
const ids = input.critiques.map(({ solutionId }) => solutionId);
|
|
220
|
+
if (ids.length !== input.solutionIds.size ||
|
|
221
|
+
new Set(ids).size !== ids.length ||
|
|
222
|
+
ids.some((id) => !input.solutionIds.has(id))) {
|
|
223
|
+
findings.push({
|
|
224
|
+
kind: "contract",
|
|
225
|
+
message: "critic must return exactly one classification for every supplied solution"
|
|
226
|
+
});
|
|
227
|
+
}
|
|
228
|
+
for (const critique of input.critiques) {
|
|
229
|
+
if (critique.detail.trim().length === 0) {
|
|
230
|
+
findings.push({
|
|
231
|
+
kind: "critique",
|
|
232
|
+
message: `critic rationale is empty for ${JSON.stringify(critique.solutionId)}`
|
|
233
|
+
});
|
|
234
|
+
}
|
|
235
|
+
if (critique.classification === "wrong" &&
|
|
236
|
+
(critique.violatedBehaviorIds.length === 0 ||
|
|
237
|
+
critique.violatedBehaviorIds.some((id) => !input.behaviorIds.has(id)))) {
|
|
238
|
+
findings.push({
|
|
239
|
+
kind: "critique",
|
|
240
|
+
message: `wrong classification for ${JSON.stringify(critique.solutionId)} must cite supplied target behavior ids`
|
|
241
|
+
});
|
|
242
|
+
}
|
|
243
|
+
if (critique.classification === "valid" && critique.violatedBehaviorIds.length > 0) {
|
|
244
|
+
findings.push({
|
|
245
|
+
kind: "critique",
|
|
246
|
+
message: `valid classification for ${JSON.stringify(critique.solutionId)} cannot cite violated behaviors`
|
|
247
|
+
});
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
return findings;
|
|
251
|
+
};
|
|
252
|
+
/**
|
|
253
|
+
* Run fixture-blind correctness review only when execution has false-rejected
|
|
254
|
+
* a generated valid control. Callers may supply prior completed rounds when
|
|
255
|
+
* resuming or when an older controls checkpoint already contains them.
|
|
256
|
+
*/
|
|
257
|
+
export const runPipelineControlCorrectnessCriticsV1 = Effect.fnUntraced(function* (input) {
|
|
258
|
+
const solutionIds = new Set(input.solutions.map(({ id }) => id));
|
|
259
|
+
const behaviorIds = new Set(input.seed.targetBehavior.map(({ id }) => id));
|
|
260
|
+
const critiques = (input.priorCritiques ?? []).filter(({ solutionId }) => solutionIds.has(solutionId));
|
|
261
|
+
let regenerations = 0;
|
|
262
|
+
const completedCriticRounds = input.solutions.length === 0 ? 0 : Math.floor(critiques.length / input.solutions.length);
|
|
263
|
+
const languageModel = yield* RepositoryFoundryLanguageModel;
|
|
264
|
+
for (let criticIndex = completedCriticRounds; criticIndex < CONTROL_CRITIC_ROLES.length; criticIndex += 1) {
|
|
265
|
+
const role = CONTROL_CRITIC_ROLES[criticIndex];
|
|
266
|
+
yield* reportFoundryStageV1(pipelineStageToProgressStageV1(STAGE), `fixture-blind correctness adjudication ${String(criticIndex + 1)} of ${String(CONTROL_CRITIC_ROLES.length)}`);
|
|
267
|
+
const assignment = yield* assignmentOf({ plan: input.context.modelPlan, role });
|
|
268
|
+
const review = yield* generateWithFeedbackV1({
|
|
269
|
+
...(input.workCheckpointStore === undefined ? {} : {
|
|
270
|
+
checkpoint: { store: input.workCheckpointStore, binding: input.seed }
|
|
271
|
+
}),
|
|
272
|
+
languageModel,
|
|
273
|
+
assignment,
|
|
274
|
+
operationId: `${input.context.caseId}:pipeline:controls:${role}`,
|
|
275
|
+
instructions: CONTROL_CRITIC_INSTRUCTIONS,
|
|
276
|
+
input: {
|
|
277
|
+
dimension: input.context.dimension,
|
|
278
|
+
visibleTask: input.spec.visible,
|
|
279
|
+
analysis: input.spec.analysis,
|
|
280
|
+
groundedScope: input.spec.scope,
|
|
281
|
+
targetBehavior: input.seed.targetBehavior.map(({ id, description, critical }) => ({
|
|
282
|
+
id,
|
|
283
|
+
description,
|
|
284
|
+
critical
|
|
285
|
+
})),
|
|
286
|
+
preChangeSources: input.spec.contextSources,
|
|
287
|
+
solutions: input.solutions.map(({ id, fileOverrides }) => ({
|
|
288
|
+
id,
|
|
289
|
+
fileOverrides
|
|
290
|
+
}))
|
|
291
|
+
},
|
|
292
|
+
schemaName: "routekit_pipeline_control_correctness_review_v1",
|
|
293
|
+
outputSchema: SolutionCritiqueSubmissionV1,
|
|
294
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1(role),
|
|
295
|
+
maxAttempts: Number.MAX_SAFE_INTEGER,
|
|
296
|
+
validate: ({ critiques: proposed }) => Effect.succeed(critiqueFindings({
|
|
297
|
+
critiques: proposed,
|
|
298
|
+
solutionIds,
|
|
299
|
+
behaviorIds
|
|
300
|
+
}))
|
|
301
|
+
});
|
|
302
|
+
critiques.push(...review.value.critiques);
|
|
303
|
+
regenerations += review.attempts - 1;
|
|
304
|
+
if (input.onRound !== undefined)
|
|
305
|
+
yield* input.onRound(critiques, regenerations);
|
|
306
|
+
}
|
|
307
|
+
return { critiques, regenerations };
|
|
308
|
+
});
|
|
309
|
+
/**
|
|
310
|
+
* Generate fixture-blind valid and wrong controls. Each proposal is exact-edit
|
|
311
|
+
* materialized and run in an isolated checkout before it is accepted. Model
|
|
312
|
+
* labels remain advisory; the tournament stage performs semantic adjudication.
|
|
313
|
+
*/
|
|
314
|
+
export const runPipelineControlsCapabilityV1 = (input) => Effect.gen(function* () {
|
|
315
|
+
const seedCheckpoint = input.checkpoints.seed;
|
|
316
|
+
const spec = input.checkpoints.spec;
|
|
317
|
+
const oracle = input.checkpoints.oracle;
|
|
318
|
+
if (seedCheckpoint === undefined || spec === undefined || oracle === undefined) {
|
|
319
|
+
return yield* failure("controls requires accepted seed, specification, and oracle checkpoints");
|
|
320
|
+
}
|
|
321
|
+
const seed = seedCheckpoint.seed;
|
|
322
|
+
if (seed.status !== "qualified" || seed.referenceCommit === undefined) {
|
|
323
|
+
return yield* failure("controls requires a replay-qualified pinned reference");
|
|
324
|
+
}
|
|
325
|
+
const parentDigest = checkpointDigestV1(oracle);
|
|
326
|
+
const resumed = input.checkpoints.controls?.parentDigest === parentDigest
|
|
327
|
+
? input.checkpoints.controls
|
|
328
|
+
: undefined;
|
|
329
|
+
const createdAt = resumed?.createdAt ?? new Date(yield* Clock.currentTimeMillis).toISOString();
|
|
330
|
+
const valid = [...(resumed?.valid ?? [])];
|
|
331
|
+
const wrong = [...(resumed?.wrong ?? [])];
|
|
332
|
+
const critiques = [...(resumed?.critiques ?? [])];
|
|
333
|
+
let regenerations = resumed?.regenerations ?? 0;
|
|
334
|
+
const languageModel = yield* RepositoryFoundryLanguageModel;
|
|
335
|
+
const allowedPaths = solutionPathsOf({ seed, contextSources: spec.contextSources });
|
|
336
|
+
if (allowedPaths.size === 0) {
|
|
337
|
+
return yield* failure("specification exposes no unprotected implementation source to blind solvers");
|
|
338
|
+
}
|
|
339
|
+
const protectedPaths = new Set(seed.environment.protectedControlPaths);
|
|
340
|
+
const editSources = yield* loadPinnedSolutionEditSourcesV1({
|
|
341
|
+
repositoryRoot: input.context.repositoryRoot,
|
|
342
|
+
initialCommit: seed.initialState.commit,
|
|
343
|
+
visibleSources: spec.contextSources,
|
|
344
|
+
allowedPaths
|
|
345
|
+
});
|
|
346
|
+
if (editSources.length === 0) {
|
|
347
|
+
return yield* failure("no complete pinned pre-change edit source could be materialized");
|
|
348
|
+
}
|
|
349
|
+
// Blind controls run on the initial tree, without reference-test overlays.
|
|
350
|
+
// Strong grade recipes may name files introduced by the historical fix.
|
|
351
|
+
// Only the qualified original baseline is an executable, non-leaking sanity
|
|
352
|
+
// contract here; hidden behavior is measured later by the tournament.
|
|
353
|
+
const sanityRecipes = seed.environment.baselineGradeRecipes;
|
|
354
|
+
if (sanityRecipes.length === 0)
|
|
355
|
+
return yield* failure("blind controls require qualified initial baseline recipes");
|
|
356
|
+
const expectedRecipeIds = new Set(sanityRecipes.map((recipe) => recipe.id));
|
|
357
|
+
const generateControl = Effect.fnUntraced(function* (request) {
|
|
358
|
+
const assignment = yield* assignmentOf({ plan: input.context.modelPlan, role: request.role });
|
|
359
|
+
let accepted;
|
|
360
|
+
const generated = yield* generateWithFeedbackV1({
|
|
361
|
+
...(input.workCheckpointStore === undefined ? {} : {
|
|
362
|
+
checkpoint: {
|
|
363
|
+
store: input.workCheckpointStore,
|
|
364
|
+
binding: {
|
|
365
|
+
parentDigest,
|
|
366
|
+
seed,
|
|
367
|
+
editSources,
|
|
368
|
+
expected: request.expected,
|
|
369
|
+
existingIds: [...valid, ...wrong].map(({ id }) => id).sort()
|
|
370
|
+
}
|
|
371
|
+
}
|
|
372
|
+
}),
|
|
373
|
+
languageModel,
|
|
374
|
+
assignment,
|
|
375
|
+
operationId: `${input.context.caseId}:pipeline:controls:${request.role}:${String(request.sequence)}${input.replacementRound === undefined ? "" : `:replacement-${String(input.replacementRound)}`}`,
|
|
376
|
+
instructions: `${request.instructions}\n\n${EXACT_EDIT_INSTRUCTIONS}`,
|
|
377
|
+
input: {
|
|
378
|
+
dimension: input.context.dimension,
|
|
379
|
+
visibleTask: spec.visible,
|
|
380
|
+
analysis: spec.analysis,
|
|
381
|
+
groundedScope: spec.scope,
|
|
382
|
+
targetBehavior: seed.targetBehavior.map(({ id, description, critical }) => ({
|
|
383
|
+
id,
|
|
384
|
+
description,
|
|
385
|
+
critical
|
|
386
|
+
})),
|
|
387
|
+
repositoryEnvironment: {
|
|
388
|
+
packageManager: seed.environment.packageManager,
|
|
389
|
+
candidateValidationRecipes: seed.environment.candidateValidationRecipes,
|
|
390
|
+
sanityRecipes
|
|
391
|
+
},
|
|
392
|
+
allowedSolutionPaths: [...allowedPaths].sort(),
|
|
393
|
+
preChangeSources: spec.contextSources
|
|
394
|
+
},
|
|
395
|
+
schemaName: `routekit_pipeline_${request.role.replaceAll("-", "_")}_v1`,
|
|
396
|
+
outputSchema: SolutionSubmissionV1,
|
|
397
|
+
maximumOutputTokens: repositoryFoundryRoleOutputCeilingV1(request.role),
|
|
398
|
+
maxAttempts: Number.MAX_SAFE_INTEGER,
|
|
399
|
+
validate: (submission) => Effect.gen(function* () {
|
|
400
|
+
const findings = submissionFindings({
|
|
401
|
+
submission,
|
|
402
|
+
expected: request.expected,
|
|
403
|
+
existingIds: new Set([...valid, ...wrong].map((solution) => solution.id)),
|
|
404
|
+
allowedPaths,
|
|
405
|
+
protectedPaths
|
|
406
|
+
});
|
|
407
|
+
if (findings.length > 0)
|
|
408
|
+
return findings;
|
|
409
|
+
const materialized = yield* Effect.result(materializeSubmission({
|
|
410
|
+
submission,
|
|
411
|
+
expected: request.expected,
|
|
412
|
+
initialCommit: seed.initialState.commit,
|
|
413
|
+
sources: editSources,
|
|
414
|
+
allowedPaths
|
|
415
|
+
}));
|
|
416
|
+
if (materialized._tag === "Failure") {
|
|
417
|
+
return [
|
|
418
|
+
{ kind: "materialization", message: describe(materialized.failure) }
|
|
419
|
+
];
|
|
420
|
+
}
|
|
421
|
+
const execution = yield* Effect.result(replayRepositoryCommandsV1({
|
|
422
|
+
repositoryRoot: input.context.repositoryRoot,
|
|
423
|
+
commit: seed.initialState.commit,
|
|
424
|
+
recipes: sanityRecipes,
|
|
425
|
+
trustedPreparationRecipes: seed.environment.trustedPreparationRecipes,
|
|
426
|
+
candidateValidationRecipes: seed.environment.candidateValidationRecipes,
|
|
427
|
+
fileOverrides: materialized.success.fileOverrides,
|
|
428
|
+
repetitions: 1
|
|
429
|
+
}));
|
|
430
|
+
if (execution._tag === "Failure") {
|
|
431
|
+
return [
|
|
432
|
+
{ kind: "execution", message: describe(execution.failure) }
|
|
433
|
+
];
|
|
434
|
+
}
|
|
435
|
+
const observedFindings = executionFindings({
|
|
436
|
+
results: execution.success,
|
|
437
|
+
expectedRecipeIds
|
|
438
|
+
});
|
|
439
|
+
if (observedFindings.length === 0)
|
|
440
|
+
accepted = materialized.success;
|
|
441
|
+
return observedFindings;
|
|
442
|
+
})
|
|
443
|
+
});
|
|
444
|
+
regenerations += generated.attempts - 1;
|
|
445
|
+
if (accepted === undefined) {
|
|
446
|
+
return yield* failure(`accepted ${request.role} response had no materialized execution`);
|
|
447
|
+
}
|
|
448
|
+
return accepted;
|
|
449
|
+
});
|
|
450
|
+
let added = 0;
|
|
451
|
+
while (input.kind !== "wrong" &&
|
|
452
|
+
valid.length < REQUIRED_VALID_CONTROLS &&
|
|
453
|
+
added < (input.maximumNewControls ?? Number.MAX_SAFE_INTEGER)) {
|
|
454
|
+
const role = "valid-solution-generator-a";
|
|
455
|
+
yield* reportFoundryStageV1(pipelineStageToProgressStageV1(STAGE), `blind valid control ${String(valid.length + 1)} of ${String(REQUIRED_VALID_CONTROLS)}`);
|
|
456
|
+
valid.push(yield* generateControl({
|
|
457
|
+
role,
|
|
458
|
+
expected: "valid",
|
|
459
|
+
instructions: VALID_A_INSTRUCTIONS,
|
|
460
|
+
sequence: valid.length + 1
|
|
461
|
+
}));
|
|
462
|
+
added += 1;
|
|
463
|
+
yield* input.checkpoint(checkpointOf({ parentDigest, createdAt, valid, wrong, critiques, regenerations }));
|
|
464
|
+
}
|
|
465
|
+
while (input.kind !== "valid" &&
|
|
466
|
+
wrong.length < REQUIRED_WRONG_CONTROLS &&
|
|
467
|
+
added < (input.maximumNewControls ?? Number.MAX_SAFE_INTEGER)) {
|
|
468
|
+
yield* reportFoundryStageV1(pipelineStageToProgressStageV1(STAGE), `blind wrong control ${String(wrong.length + 1)} of ${String(REQUIRED_WRONG_CONTROLS)}`);
|
|
469
|
+
wrong.push(yield* generateControl({
|
|
470
|
+
role: "adversary",
|
|
471
|
+
expected: "wrong",
|
|
472
|
+
instructions: ADVERSARY_INSTRUCTIONS,
|
|
473
|
+
sequence: wrong.length + 1
|
|
474
|
+
}));
|
|
475
|
+
added += 1;
|
|
476
|
+
yield* input.checkpoint(checkpointOf({ parentDigest, createdAt, valid, wrong, critiques, regenerations }));
|
|
477
|
+
}
|
|
478
|
+
yield* reportFoundryStageV1(pipelineStageToProgressStageV1(STAGE), `controls ready; valid=${String(valid.length)}; wrong=${String(wrong.length)}; advisoryCritiques=${String(critiques.length)}`);
|
|
479
|
+
return checkpointOf({ parentDigest, createdAt, valid, wrong, critiques, regenerations });
|
|
480
|
+
}).pipe(Effect.mapError((error) => normalizePipelineFailureV1(error)));
|
|
481
|
+
/** Short name used by the orchestrator. */
|
|
482
|
+
export const runPipelineControlsStageV1 = (input) => runPipelineControlsCapabilityV1(input);
|
|
483
|
+
export const runPipelineControlsV1 = runPipelineControlsStageV1;
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
import { type OracleCheckpointV1, type PipelineStageRunV1 } from "../../case-pipeline-protocol.js";
|
|
2
|
+
/**
|
|
3
|
+
* Authors and reference-preflights the initial oracle. Proactive coverage
|
|
4
|
+
* criticism and speculative witness generation are intentionally deferred to
|
|
5
|
+
* the tournament, where actual correct and wrong solutions provide measured
|
|
6
|
+
* escapes. Only executed observations can later establish admission evidence.
|
|
7
|
+
*/
|
|
8
|
+
export declare const runPipelineOracleStageV1: PipelineStageRunV1<OracleCheckpointV1>;
|