@velum-labs/routekit-eval-setup 1.3.2 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/authoring-responses-request.d.ts +5 -0
- package/dist/adapters/authoring-responses-request.js +38 -0
- package/dist/adapters/evaluation-evidence-freshness.d.ts +7 -0
- package/dist/adapters/evaluation-evidence-freshness.js +76 -0
- package/dist/adapters/git-task-history.d.ts +67 -0
- package/dist/adapters/git-task-history.js +171 -0
- package/dist/adapters/integrated-repository-history.d.ts +21 -0
- package/dist/adapters/integrated-repository-history.js +175 -0
- package/dist/adapters/repository-command-diagnostic.d.ts +8 -0
- package/dist/adapters/repository-command-diagnostic.js +46 -0
- package/dist/adapters/repository-command-evidence.d.ts +13 -0
- package/dist/adapters/repository-command-evidence.js +102 -0
- package/dist/adapters/repository-command-runner.d.ts +238 -0
- package/dist/adapters/repository-command-runner.js +1483 -0
- package/dist/adapters/repository-import-context.d.ts +47 -0
- package/dist/adapters/repository-import-context.js +469 -0
- package/dist/adapters/repository-node-test-reporter.d.ts +3 -0
- package/dist/adapters/repository-node-test-reporter.js +27 -0
- package/dist/adapters/repository-review-evidence.d.ts +39 -0
- package/dist/adapters/repository-review-evidence.js +632 -0
- package/dist/adapters/repository-seed-selection.d.ts +7 -0
- package/dist/adapters/repository-seed-selection.js +79 -0
- package/dist/adapters/repository-solution-edits.d.ts +49 -0
- package/dist/adapters/repository-solution-edits.js +136 -0
- package/dist/adapters/repository-vitest-phase-adapter.d.ts +8 -0
- package/dist/adapters/repository-vitest-phase-adapter.js +310 -0
- package/dist/adapters/repository-vitest-reporter.d.ts +24 -0
- package/dist/adapters/repository-vitest-reporter.js +314 -0
- package/dist/adapters/strict-authoring-schema.d.ts +5 -0
- package/dist/adapters/strict-authoring-schema.js +158 -0
- package/dist/adapters/test-discovery.d.ts +30 -0
- package/dist/adapters/test-discovery.js +124 -0
- package/dist/adapters/typescript-repository-index.d.ts +51 -0
- package/dist/adapters/typescript-repository-index.js +226 -0
- package/dist/agentic-capabilities-protocol.d.ts +1373 -0
- package/dist/agentic-capabilities-protocol.js +786 -0
- package/dist/case-checkpoint-store.d.ts +29 -0
- package/dist/case-checkpoint-store.js +133 -0
- package/dist/case-pipeline-protocol-v2.d.ts +184 -0
- package/dist/case-pipeline-protocol-v2.js +193 -0
- package/dist/case-pipeline-protocol.d.ts +2626 -0
- package/dist/case-pipeline-protocol.js +371 -0
- package/dist/effect-api.d.ts +74 -10
- package/dist/effect-api.js +56 -6
- package/dist/errors.d.ts +31 -0
- package/dist/errors.js +10 -0
- package/dist/eval-capability-execution-envelope.d.ts +64 -0
- package/dist/eval-capability-execution-envelope.js +98 -0
- package/dist/eval-capability-policy.d.ts +90 -0
- package/dist/eval-capability-policy.js +107 -0
- package/dist/eval-event-log.d.ts +140 -0
- package/dist/eval-event-log.js +220 -0
- package/dist/evaluation-authoring-policy.d.ts +18 -0
- package/dist/evaluation-authoring-policy.js +19 -0
- package/dist/evaluation-authoring-validation.d.ts +22 -0
- package/dist/evaluation-authoring-validation.js +72 -0
- package/dist/evaluation-evidence.d.ts +20 -0
- package/dist/evaluation-evidence.js +319 -0
- package/dist/evaluation-grader-calibration-protocol.d.ts +108 -0
- package/dist/evaluation-grader-calibration-protocol.js +80 -0
- package/dist/evaluation-grader-calibration.d.ts +18 -0
- package/dist/evaluation-grader-calibration.js +334 -0
- package/dist/evaluation-grading-policy.d.ts +24 -0
- package/dist/evaluation-grading-policy.js +54 -0
- package/dist/evaluation-proposal-policy.d.ts +4 -0
- package/dist/evaluation-proposal-policy.js +91 -0
- package/dist/evaluation-source-retrieval.d.ts +68 -0
- package/dist/evaluation-source-retrieval.js +513 -0
- package/dist/evaluation-structure-policy.d.ts +29 -0
- package/dist/evaluation-structure-policy.js +138 -0
- package/dist/index.d.ts +124 -17
- package/dist/index.js +69 -11
- package/dist/inspection.js +2 -3
- package/dist/project-artifacts.d.ts +7 -2
- package/dist/project-artifacts.js +49 -136
- package/dist/project-authoring.d.ts +66 -5
- package/dist/project-authoring.js +783 -109
- package/dist/project-contracts.d.ts +419 -84
- package/dist/project-contracts.js +160 -52
- package/dist/project-store.js +2 -1
- package/dist/project-workflow.d.ts +5 -4
- package/dist/project-workflow.js +154 -35
- package/dist/repository-adversary-protocol.d.ts +64 -0
- package/dist/repository-adversary-protocol.js +105 -0
- package/dist/repository-behavior-protocol.d.ts +188 -0
- package/dist/repository-behavior-protocol.js +202 -0
- package/dist/repository-benchmark-protocol.d.ts +487 -0
- package/dist/repository-benchmark-protocol.js +96 -0
- package/dist/repository-execution-protocol.d.ts +150 -0
- package/dist/repository-execution-protocol.js +38 -0
- package/dist/repository-fixture-instructions.d.ts +3 -0
- package/dist/repository-fixture-instructions.js +91 -0
- package/dist/repository-fixture-protocol.d.ts +79 -0
- package/dist/repository-fixture-protocol.js +79 -0
- package/dist/repository-foundry-plan-protocol.d.ts +118 -0
- package/dist/repository-foundry-plan-protocol.js +296 -0
- package/dist/repository-foundry-progress-protocol.d.ts +52 -0
- package/dist/repository-foundry-progress-protocol.js +52 -0
- package/dist/repository-improvement-protocol.d.ts +100 -0
- package/dist/repository-improvement-protocol.js +106 -0
- package/dist/repository-language-model-protocol.d.ts +43 -0
- package/dist/repository-language-model-protocol.js +146 -0
- package/dist/repository-oracle-coverage-protocol.d.ts +18 -0
- package/dist/repository-oracle-coverage-protocol.js +39 -0
- package/dist/repository-oracle-execution-binding.d.ts +27 -0
- package/dist/repository-oracle-execution-binding.js +59 -0
- package/dist/repository-oracle-protocol.d.ts +230 -0
- package/dist/repository-oracle-protocol.js +156 -0
- package/dist/repository-oracle-scope-policy.d.ts +22 -0
- package/dist/repository-oracle-scope-policy.js +92 -0
- package/dist/repository-quality-policy.d.ts +15 -0
- package/dist/repository-quality-policy.js +357 -0
- package/dist/repository-routing-benchmark-protocol.d.ts +176 -0
- package/dist/repository-routing-benchmark-protocol.js +103 -0
- package/dist/repository-routing-model-protocol.d.ts +36 -0
- package/dist/repository-routing-model-protocol.js +89 -0
- package/dist/repository-routing-plan-protocol.d.ts +112 -0
- package/dist/repository-routing-plan-protocol.js +58 -0
- package/dist/repository-routing-quality-policy.d.ts +9 -0
- package/dist/repository-routing-quality-policy.js +191 -0
- package/dist/repository-seed-qualification-progress-protocol.d.ts +205 -0
- package/dist/repository-seed-qualification-progress-protocol.js +28 -0
- package/dist/repository-semantic-calibration-protocol.d.ts +768 -0
- package/dist/repository-semantic-calibration-protocol.js +276 -0
- package/dist/repository-semantic-calibration.d.ts +163 -0
- package/dist/repository-semantic-calibration.js +581 -0
- package/dist/repository-specification-contract-facts-protocol.d.ts +224 -0
- package/dist/repository-specification-contract-facts-protocol.js +276 -0
- package/dist/repository-specification-critique-protocol.d.ts +189 -0
- package/dist/repository-specification-critique-protocol.js +103 -0
- package/dist/repository-task-family-protocol.d.ts +24 -0
- package/dist/repository-task-family-protocol.js +37 -0
- package/dist/repository-task-seed-protocol.d.ts +384 -0
- package/dist/repository-task-seed-protocol.js +236 -0
- package/dist/repository-trajectory-protocol.d.ts +20 -0
- package/dist/repository-trajectory-protocol.js +42 -0
- package/dist/service.js +1 -1
- package/dist/services/adversary/service.d.ts +64 -0
- package/dist/services/adversary/service.js +330 -0
- package/dist/services/benchmark-compiler/service.d.ts +450 -0
- package/dist/services/benchmark-compiler/service.js +9 -0
- package/dist/services/budgeted-model/service.d.ts +118 -0
- package/dist/services/budgeted-model/service.js +460 -0
- package/dist/services/case-authoring/service.d.ts +163 -0
- package/dist/services/case-authoring/service.js +1456 -0
- package/dist/services/case-finalization/service.d.ts +283 -0
- package/dist/services/case-finalization/service.js +370 -0
- package/dist/services/case-generation/service.d.ts +619 -0
- package/dist/services/case-generation/service.js +2628 -0
- package/dist/services/case-pipeline/service.d.ts +31 -0
- package/dist/services/case-pipeline/service.js +485 -0
- package/dist/services/case-pipeline-v2/service.d.ts +70 -0
- package/dist/services/case-pipeline-v2/service.js +477 -0
- package/dist/services/command-observability/service.d.ts +13 -0
- package/dist/services/command-observability/service.js +3 -0
- package/dist/services/dimension-labeling/service.d.ts +77 -0
- package/dist/services/dimension-labeling/service.js +188 -0
- package/dist/services/eval-candidate/service.d.ts +208 -0
- package/dist/services/eval-candidate/service.js +64 -0
- package/dist/services/eval-capabilities/service.d.ts +183 -0
- package/dist/services/eval-capabilities/service.js +1433 -0
- package/dist/services/eval-environment/service.d.ts +173 -0
- package/dist/services/eval-environment/service.js +127 -0
- package/dist/services/evidence-reconstruction/service.d.ts +36 -0
- package/dist/services/evidence-reconstruction/service.js +145 -0
- package/dist/services/fixture-builder/service.d.ts +62 -0
- package/dist/services/fixture-builder/service.js +36 -0
- package/dist/services/fixture-validation/service.d.ts +75 -0
- package/dist/services/fixture-validation/service.js +295 -0
- package/dist/services/foundry/service.d.ts +831 -0
- package/dist/services/foundry/service.js +442 -0
- package/dist/services/foundry-progress/service.d.ts +62 -0
- package/dist/services/foundry-progress/service.js +149 -0
- package/dist/services/foundry-v2/service.d.ts +54 -0
- package/dist/services/foundry-v2/service.js +28 -0
- package/dist/services/grounded-authoring/service.d.ts +126 -0
- package/dist/services/grounded-authoring/service.js +822 -0
- package/dist/services/historical-case/service.d.ts +722 -0
- package/dist/services/historical-case/service.js +177 -0
- package/dist/services/improvement-loop/service.d.ts +59 -0
- package/dist/services/improvement-loop/service.js +176 -0
- package/dist/services/language-model/service.d.ts +52 -0
- package/dist/services/language-model/service.js +194 -0
- package/dist/services/oracle-builder/service.d.ts +146 -0
- package/dist/services/oracle-builder/service.js +513 -0
- package/dist/services/oracle-coverage/service.d.ts +28 -0
- package/dist/services/oracle-coverage/service.js +50 -0
- package/dist/services/oracle-coverage-witness/service.d.ts +130 -0
- package/dist/services/oracle-coverage-witness/service.js +538 -0
- package/dist/services/pipeline-challenge/service.d.ts +551 -0
- package/dist/services/pipeline-challenge/service.js +427 -0
- package/dist/services/pipeline-controls/service.d.ts +130 -0
- package/dist/services/pipeline-controls/service.js +483 -0
- package/dist/services/pipeline-oracle/service.d.ts +8 -0
- package/dist/services/pipeline-oracle/service.js +256 -0
- package/dist/services/pipeline-seed/service.d.ts +298 -0
- package/dist/services/pipeline-seed/service.js +428 -0
- package/dist/services/pipeline-spec/service.d.ts +103 -0
- package/dist/services/pipeline-spec/service.js +619 -0
- package/dist/services/pipeline-tournament/service.d.ts +258 -0
- package/dist/services/pipeline-tournament/service.js +476 -0
- package/dist/services/quality-gate/service.d.ts +233 -0
- package/dist/services/quality-gate/service.js +136 -0
- package/dist/services/repository-bundle/service.d.ts +33 -0
- package/dist/services/repository-bundle/service.js +114 -0
- package/dist/services/repository-model/service.d.ts +105 -0
- package/dist/services/repository-model/service.js +250 -0
- package/dist/services/repository-public-artifact/service.d.ts +133 -0
- package/dist/services/repository-public-artifact/service.js +330 -0
- package/dist/services/routing-benchmark/service.d.ts +362 -0
- package/dist/services/routing-benchmark/service.js +96 -0
- package/dist/services/specification-critic/service.d.ts +92 -0
- package/dist/services/specification-critic/service.js +172 -0
- package/dist/services/task-family/service.d.ts +40 -0
- package/dist/services/task-family/service.js +55 -0
- package/dist/services/task-seed/service.d.ts +906 -0
- package/dist/services/task-seed/service.js +1406 -0
- package/dist/services/task-specification/service.d.ts +27 -0
- package/dist/services/task-specification/service.js +40 -0
- package/dist/services/trajectory-policy/service.d.ts +110 -0
- package/dist/services/trajectory-policy/service.js +216 -0
- package/dist/test/agentic-capabilities-protocol.test.d.ts +1 -0
- package/dist/test/agentic-capabilities-protocol.test.js +570 -0
- package/dist/test/agentic-capabilities.test.d.ts +1 -0
- package/dist/test/agentic-capabilities.test.js +1461 -0
- package/dist/test/agentic-environment.test.d.ts +1 -0
- package/dist/test/agentic-environment.test.js +213 -0
- package/dist/test/case-pipeline-foundation.test.d.ts +1 -0
- package/dist/test/case-pipeline-foundation.test.js +535 -0
- package/dist/test/case-pipeline-protocol-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-protocol-v2.test.js +124 -0
- package/dist/test/case-pipeline-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-v2.test.js +286 -0
- package/dist/test/case-pipeline.test.d.ts +1 -0
- package/dist/test/case-pipeline.test.js +851 -0
- package/dist/test/eval-capability-policy.test.d.ts +1 -0
- package/dist/test/eval-capability-policy.test.js +50 -0
- package/dist/test/eval-event-log.test.d.ts +1 -0
- package/dist/test/eval-event-log.test.js +125 -0
- package/dist/test/evaluation-evidence-freshness.test.d.ts +1 -0
- package/dist/test/evaluation-evidence-freshness.test.js +44 -0
- package/dist/test/evaluation-evidence.test.d.ts +1 -0
- package/dist/test/evaluation-evidence.test.js +230 -0
- package/dist/test/evaluation-grader-calibration.test.d.ts +1 -0
- package/dist/test/evaluation-grader-calibration.test.js +373 -0
- package/dist/test/evaluation-proposal-digest.test.d.ts +1 -0
- package/dist/test/evaluation-proposal-digest.test.js +187 -0
- package/dist/test/evaluation-source-retrieval.test.d.ts +1 -0
- package/dist/test/evaluation-source-retrieval.test.js +237 -0
- package/dist/test/evaluation-structure-policy.test.d.ts +1 -0
- package/dist/test/evaluation-structure-policy.test.js +196 -0
- package/dist/test/fixtures/repository-resource-panel.d.ts +39 -0
- package/dist/test/fixtures/repository-resource-panel.js +111 -0
- package/dist/test/fixtures/vitest-boundary-panel.d.ts +84 -0
- package/dist/test/fixtures/vitest-boundary-panel.js +120 -0
- package/dist/test/fixtures/vitest-phase-panel.d.ts +135 -0
- package/dist/test/fixtures/vitest-phase-panel.js +213 -0
- package/dist/test/fixtures/vitest-reporter-results.d.ts +76 -0
- package/dist/test/fixtures/vitest-reporter-results.js +94 -0
- package/dist/test/grounded-authoring.test.d.ts +1 -0
- package/dist/test/grounded-authoring.test.js +565 -0
- package/dist/test/integrated-repository-history.test.d.ts +1 -0
- package/dist/test/integrated-repository-history.test.js +227 -0
- package/dist/test/project-authoring.test.js +593 -43
- package/dist/test/project-workflow.test.js +419 -40
- package/dist/test/repository-authoring-artifacts.test.d.ts +1 -0
- package/dist/test/repository-authoring-artifacts.test.js +185 -0
- package/dist/test/repository-bundle.test.d.ts +1 -0
- package/dist/test/repository-bundle.test.js +52 -0
- package/dist/test/repository-case-generation.test.d.ts +1 -0
- package/dist/test/repository-case-generation.test.js +3465 -0
- package/dist/test/repository-command-diagnostic.test.d.ts +1 -0
- package/dist/test/repository-command-diagnostic.test.js +55 -0
- package/dist/test/repository-command-signals.test.d.ts +1 -0
- package/dist/test/repository-command-signals.test.js +124 -0
- package/dist/test/repository-fixture-scope-coverage.test.d.ts +1 -0
- package/dist/test/repository-fixture-scope-coverage.test.js +127 -0
- package/dist/test/repository-fixture-validation.test.d.ts +1 -0
- package/dist/test/repository-fixture-validation.test.js +362 -0
- package/dist/test/repository-foundry-progress.test.d.ts +1 -0
- package/dist/test/repository-foundry-progress.test.js +110 -0
- package/dist/test/repository-foundry-quality.test.d.ts +1 -0
- package/dist/test/repository-foundry-quality.test.js +1138 -0
- package/dist/test/repository-import-context.test.d.ts +1 -0
- package/dist/test/repository-import-context.test.js +354 -0
- package/dist/test/repository-model-authoring.test.d.ts +1 -0
- package/dist/test/repository-model-authoring.test.js +544 -0
- package/dist/test/repository-model.test.d.ts +1 -0
- package/dist/test/repository-model.test.js +2195 -0
- package/dist/test/repository-node-test-reporter.test.d.ts +1 -0
- package/dist/test/repository-node-test-reporter.test.js +104 -0
- package/dist/test/repository-oracle-concurrency.test.d.ts +1 -0
- package/dist/test/repository-oracle-concurrency.test.js +542 -0
- package/dist/test/repository-oracle-coverage-witness.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage-witness.test.js +511 -0
- package/dist/test/repository-oracle-coverage.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage.test.js +168 -0
- package/dist/test/repository-oracle-evidence.test.d.ts +1 -0
- package/dist/test/repository-oracle-evidence.test.js +185 -0
- package/dist/test/repository-oracle-plan.test.d.ts +1 -0
- package/dist/test/repository-oracle-plan.test.js +176 -0
- package/dist/test/repository-overlay-isolation.test.d.ts +1 -0
- package/dist/test/repository-overlay-isolation.test.js +85 -0
- package/dist/test/repository-preparation-cache.test.d.ts +1 -0
- package/dist/test/repository-preparation-cache.test.js +414 -0
- package/dist/test/repository-public-artifact.test.d.ts +1 -0
- package/dist/test/repository-public-artifact.test.js +273 -0
- package/dist/test/repository-qualification-diagnostics.test.d.ts +1 -0
- package/dist/test/repository-qualification-diagnostics.test.js +524 -0
- package/dist/test/repository-reference-authoring.test.d.ts +1 -0
- package/dist/test/repository-reference-authoring.test.js +1633 -0
- package/dist/test/repository-review-evidence-v2.test.d.ts +1 -0
- package/dist/test/repository-review-evidence-v2.test.js +183 -0
- package/dist/test/repository-review-evidence.test.d.ts +1 -0
- package/dist/test/repository-review-evidence.test.js +124 -0
- package/dist/test/repository-seed-exclusions.test.d.ts +1 -0
- package/dist/test/repository-seed-exclusions.test.js +96 -0
- package/dist/test/repository-seed-selection.test.d.ts +1 -0
- package/dist/test/repository-seed-selection.test.js +504 -0
- package/dist/test/repository-semantic-calibration.test.d.ts +1 -0
- package/dist/test/repository-semantic-calibration.test.js +688 -0
- package/dist/test/repository-solution-edits.test.d.ts +1 -0
- package/dist/test/repository-solution-edits.test.js +377 -0
- package/dist/test/repository-specification-budget.test.d.ts +1 -0
- package/dist/test/repository-specification-budget.test.js +171 -0
- package/dist/test/repository-specification-contract-checkpoint.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-checkpoint.test.js +228 -0
- package/dist/test/repository-specification-contract-facts.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-facts.test.js +177 -0
- package/dist/test/repository-trajectory-authoring.test.d.ts +1 -0
- package/dist/test/repository-trajectory-authoring.test.js +176 -0
- package/dist/test/repository-valid-control-plan.test.d.ts +1 -0
- package/dist/test/repository-valid-control-plan.test.js +45 -0
- package/dist/test/repository-vitest-phase.test.d.ts +1 -0
- package/dist/test/repository-vitest-phase.test.js +848 -0
- package/dist/test/repository-vitest-reporter.test.d.ts +1 -0
- package/dist/test/repository-vitest-reporter.test.js +158 -0
- package/dist/test/repository-workspace-build.test.d.ts +1 -0
- package/dist/test/repository-workspace-build.test.js +160 -0
- package/dist/test/strict-authoring-schema.test.d.ts +1 -0
- package/dist/test/strict-authoring-schema.test.js +169 -0
- package/package.json +48 -6
|
@@ -0,0 +1,1433 @@
|
|
|
1
|
+
import { Clock, Effect, Option, Schema, Semaphore } from "effect";
|
|
2
|
+
import { evalCapabilityExecutionEnvelopeV1 } from "../../eval-capability-execution-envelope.js";
|
|
3
|
+
import { EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1, evalCapabilityScientificAttemptCostV1, evalCapabilityScientificPhaseV1, evalCapabilityScientificProgressPredecessorV1 } from "../../eval-capability-policy.js";
|
|
4
|
+
import { agenticDigestV1, decodeAgenticCapabilityRequestV1, decodeAgenticCapabilityResultV1, decodeAgenticQualityScorecardV1, decodeAgenticScientificProgressReceiptV1, makeAgenticProgressFingerprintV1 } from "../../agentic-capabilities-protocol.js";
|
|
5
|
+
import { replayRepositoryCommandsV1 } from "../../adapters/repository-command-runner.js";
|
|
6
|
+
import { repositoryCommandProbeOutcomeV1 } from "../../adapters/repository-command-evidence.js";
|
|
7
|
+
import { RepositoryHiddenFixtureSuiteV1 } from "../../repository-fixture-protocol.js";
|
|
8
|
+
import { makeCaseCheckpointStoreV1 } from "../../case-checkpoint-store.js";
|
|
9
|
+
import { checkpointDigestV1, PIPELINE_CHECKPOINT_SCHEMAS, PIPELINE_STAGES, RepositoryExecutableSolutionSchemaV1 } from "../../case-pipeline-protocol.js";
|
|
10
|
+
import { RepositoryFoundryError } from "../../errors.js";
|
|
11
|
+
import { RepositoryHiddenOracleV1 } from "../../repository-oracle-protocol.js";
|
|
12
|
+
import { evaluateRepositoryOracleScopeCoverageV1 } from "../../repository-oracle-scope-policy.js";
|
|
13
|
+
import { repositoryOracleSolutionObservationsCompleteV1 } from "../../repository-quality-policy.js";
|
|
14
|
+
import { RepositorySpecificationReviewV1 } from "../../repository-specification-critique-protocol.js";
|
|
15
|
+
import { assertRepositoryTaskSeedV1 } from "../../repository-task-seed-protocol.js";
|
|
16
|
+
import { RepositorySeedQualificationProgressV1 } from "../../repository-seed-qualification-progress-protocol.js";
|
|
17
|
+
import { BudgetedLanguageModelLive, budgetExhaustedErrorOfV1, makePipelineStateAccessorV1 } from "../budgeted-model/service.js";
|
|
18
|
+
import { draftEvalCandidateV1 } from "../eval-candidate/service.js";
|
|
19
|
+
import { planEvalEnvironmentV1 } from "../eval-environment/service.js";
|
|
20
|
+
import { RepositoryFoundryLanguageModel } from "../language-model/service.js";
|
|
21
|
+
import { runPipelineAdmitStageV1, runPipelineChallengeCapabilityV1 } from "../pipeline-challenge/service.js";
|
|
22
|
+
import { runPipelineControlsCapabilityV1 } from "../pipeline-controls/service.js";
|
|
23
|
+
import { runPipelineOracleStageV1 } from "../pipeline-oracle/service.js";
|
|
24
|
+
import { runPipelineTournamentCapabilityV1 } from "../pipeline-tournament/service.js";
|
|
25
|
+
import { qualifyHistoricalTaskSeedAcrossBatchesResumableV1, repositorySeedQualificationProgressSummaryV1 } from "../task-seed/service.js";
|
|
26
|
+
export { EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_MODEL_CALLS_V1, EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_WALL_TIME_MS_V1, EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1, evalCapabilityScientificPhaseV1 } from "../../eval-capability-policy.js";
|
|
27
|
+
const failure = (detail, cause) => new RepositoryFoundryError({
|
|
28
|
+
operation: "run-pipeline-stage",
|
|
29
|
+
detail: `${detail}${cause instanceof RepositoryFoundryError ? `: ${cause.detail}` : cause instanceof Error ? `: ${cause.message}` : ""}`,
|
|
30
|
+
...(cause === undefined ? {} : { cause })
|
|
31
|
+
});
|
|
32
|
+
const REFERENCE_QUALIFICATION_WORK_CHECKPOINT = "reference-qualification";
|
|
33
|
+
const GROUNDED_WORK_INDEX = "grounded-continuations";
|
|
34
|
+
const GroundedWorkIndexV1 = Schema.Array(Schema.Struct({
|
|
35
|
+
name: Schema.String,
|
|
36
|
+
contentDigest: Schema.String,
|
|
37
|
+
stage: Schema.Literals(PIPELINE_STAGES)
|
|
38
|
+
}));
|
|
39
|
+
const PendingHeldOutCheckpointV1 = Schema.Struct({
|
|
40
|
+
frozen: Schema.String,
|
|
41
|
+
agenticPendingHeldOut: RepositoryExecutableSolutionSchemaV1,
|
|
42
|
+
agenticPendingOracle: Schema.optionalKey(RepositoryHiddenOracleV1)
|
|
43
|
+
});
|
|
44
|
+
export const evalCapabilityArtifactHandleV1 = (stage, value) => `${stage}-${checkpointDigestV1(value)}`;
|
|
45
|
+
const acceptedSpec = (checkpoint) => {
|
|
46
|
+
const review = Schema.decodeUnknownOption(RepositorySpecificationReviewV1)(checkpoint?.reviews.at(-1));
|
|
47
|
+
return (Option.isSome(review) &&
|
|
48
|
+
review.value.verdict === "sufficient" &&
|
|
49
|
+
review.value.dimensionFit?.outcome === "fit" &&
|
|
50
|
+
review.value.referenceCompatibility?.outcome === "compatible" &&
|
|
51
|
+
(review.value.criticalBehaviorCoverage?.every((entry) => entry.outcome === "covered") ??
|
|
52
|
+
false) &&
|
|
53
|
+
!review.value.findings.some((entry) => entry.severity === "blocker"));
|
|
54
|
+
};
|
|
55
|
+
const CANDIDATE_ALLOWED_CAPABILITIES = [
|
|
56
|
+
"get_progress",
|
|
57
|
+
"draft_specification",
|
|
58
|
+
"plan_environment",
|
|
59
|
+
"qualify_reference",
|
|
60
|
+
"author_oracle",
|
|
61
|
+
"generate_controls",
|
|
62
|
+
"evaluate_oracle",
|
|
63
|
+
"repair_oracle",
|
|
64
|
+
"freeze_candidate",
|
|
65
|
+
"run_held_out_challenge",
|
|
66
|
+
"request_admission",
|
|
67
|
+
"reject_candidate"
|
|
68
|
+
];
|
|
69
|
+
const measureSolutions = (oracle, expected, evidence) => {
|
|
70
|
+
const solutions = oracle?.solutions.filter((solution) => solution.expectedClass === expected) ?? [];
|
|
71
|
+
let passed = 0;
|
|
72
|
+
let failed = 0;
|
|
73
|
+
let missing = solutions.length === 0 ? 1 : 0;
|
|
74
|
+
for (const solution of solutions) {
|
|
75
|
+
const observations = oracle.observations.filter((row) => row.solutionId === solution.id);
|
|
76
|
+
if (!repositoryOracleSolutionObservationsCompleteV1(oracle, solution.id)) {
|
|
77
|
+
missing += 1;
|
|
78
|
+
continue;
|
|
79
|
+
}
|
|
80
|
+
const discriminated = expected === "valid"
|
|
81
|
+
? observations.every((row) => row.outcome === "pass")
|
|
82
|
+
: observations.some((row) => row.outcome === "fail");
|
|
83
|
+
if (discriminated)
|
|
84
|
+
passed += 1;
|
|
85
|
+
else
|
|
86
|
+
failed += 1;
|
|
87
|
+
}
|
|
88
|
+
return { passed, failed, missing, evidence: solutions.length === 0 ? [] : evidence };
|
|
89
|
+
};
|
|
90
|
+
/** Measurements cite executed checkpoint evidence. Draft/model assertions never count as observed quality. */
|
|
91
|
+
export const evalCandidateQualityScorecardV1 = (input) => {
|
|
92
|
+
const c = input.checkpoints;
|
|
93
|
+
const handle = (stage) => c[stage] === undefined ? [] : [evalCapabilityArtifactHandleV1(stage, c[stage])];
|
|
94
|
+
const reference = c.seed === undefined
|
|
95
|
+
? { passed: 0, failed: 0, missing: 1, evidence: [] }
|
|
96
|
+
: {
|
|
97
|
+
passed: c.seed.qualification.stability.stable && c.seed.seed.status === "qualified"
|
|
98
|
+
? c.seed.qualification.stability.batches.length
|
|
99
|
+
: 0,
|
|
100
|
+
failed: c.seed.seed.status === "rejected" ? 1 : 0,
|
|
101
|
+
missing: c.seed.seed.status === "qualified" ? 0 : 1,
|
|
102
|
+
evidence: handle("seed")
|
|
103
|
+
};
|
|
104
|
+
const validControls = measureSolutions(c.tournament?.oracle, "valid", handle("tournament"));
|
|
105
|
+
const wrongControls = measureSolutions(c.tournament?.oracle, "wrong", handle("tournament"));
|
|
106
|
+
const heldOut = measureSolutions(c.challenge?.oracle, "wrong", handle("challenge"));
|
|
107
|
+
const admitted = c.admit?.verdict === "admitted";
|
|
108
|
+
return decodeAgenticQualityScorecardV1({
|
|
109
|
+
version: 1,
|
|
110
|
+
candidate: input.candidate,
|
|
111
|
+
revision: input.revision,
|
|
112
|
+
reference,
|
|
113
|
+
validControls,
|
|
114
|
+
wrongControls,
|
|
115
|
+
heldOut,
|
|
116
|
+
// Scope coverage is executed by the existing admission service, not inferred from model self-reports.
|
|
117
|
+
coverage: (c.spec?.scope ?? []).map((scope) => ({
|
|
118
|
+
cell: scope.id,
|
|
119
|
+
observed: admitted,
|
|
120
|
+
evidence: admitted ? handle("admit") : []
|
|
121
|
+
})),
|
|
122
|
+
disposition: admitted
|
|
123
|
+
? "admitted"
|
|
124
|
+
: (input.disposition ??
|
|
125
|
+
(c.admit?.verdict === "rejected"
|
|
126
|
+
? "rejected"
|
|
127
|
+
: c.seed?.seed.status === "qualified"
|
|
128
|
+
? "qualified"
|
|
129
|
+
: "draft")),
|
|
130
|
+
reasons: [...new Set([...(input.reasons ?? []), ...(c.admit?.reasons ?? [])])].slice(0, 32),
|
|
131
|
+
missingEvidence: [
|
|
132
|
+
reference.missing > 0 ? "reference" : undefined,
|
|
133
|
+
validControls.missing > 0 ? "valid_controls" : undefined,
|
|
134
|
+
wrongControls.missing > 0 ? "wrong_controls" : undefined,
|
|
135
|
+
heldOut.missing > 0 ? "held_out" : undefined
|
|
136
|
+
].filter((entry) => entry !== undefined),
|
|
137
|
+
usage: {
|
|
138
|
+
elapsedMs: Math.max(0, Math.trunc(input.elapsedMs)),
|
|
139
|
+
receivedCalls: input.receivedCalls,
|
|
140
|
+
spendKnownUsd: input.spendKnownUsd ?? "0",
|
|
141
|
+
spendUnknown: input.spendUnknown ?? input.receivedCalls > 0
|
|
142
|
+
},
|
|
143
|
+
admission: admitted ? evalCapabilityArtifactHandleV1("admit", c.admit) : null
|
|
144
|
+
});
|
|
145
|
+
};
|
|
146
|
+
/**
|
|
147
|
+
* One scientific capability is allowed enough time to finish a complete
|
|
148
|
+
* repository-grounded role chain (analysis, visible authoring, and private
|
|
149
|
+
* review) under normal provider tail latency. Cloud admission and the sandbox
|
|
150
|
+
* executor import this same value so the host never reserves a shorter slice
|
|
151
|
+
* than the worker can use.
|
|
152
|
+
*/
|
|
153
|
+
const scientificRoleForCapability = (capability) => {
|
|
154
|
+
switch (capability) {
|
|
155
|
+
case "draft_specification":
|
|
156
|
+
return "specification-author";
|
|
157
|
+
case "plan_environment":
|
|
158
|
+
case "probe_fixture":
|
|
159
|
+
case "qualify_reference":
|
|
160
|
+
return "seed-author";
|
|
161
|
+
case "author_oracle":
|
|
162
|
+
return "oracle-author";
|
|
163
|
+
case "generate_controls":
|
|
164
|
+
return "control-author";
|
|
165
|
+
case "evaluate_oracle":
|
|
166
|
+
case "repair_oracle":
|
|
167
|
+
return "oracle-critic";
|
|
168
|
+
case "run_held_out_challenge":
|
|
169
|
+
return "adversary-author";
|
|
170
|
+
case "request_admission":
|
|
171
|
+
return "admission-reviewer";
|
|
172
|
+
default:
|
|
173
|
+
return "eval-orchestrator";
|
|
174
|
+
}
|
|
175
|
+
};
|
|
176
|
+
const scientificStageForCapability = (capability) => {
|
|
177
|
+
switch (capability) {
|
|
178
|
+
case "draft_specification":
|
|
179
|
+
return "spec";
|
|
180
|
+
case "plan_environment":
|
|
181
|
+
case "probe_fixture":
|
|
182
|
+
case "qualify_reference":
|
|
183
|
+
return "seed";
|
|
184
|
+
case "author_oracle":
|
|
185
|
+
return "oracle";
|
|
186
|
+
case "generate_controls":
|
|
187
|
+
return "controls";
|
|
188
|
+
case "evaluate_oracle":
|
|
189
|
+
case "repair_oracle":
|
|
190
|
+
return "tournament";
|
|
191
|
+
case "run_held_out_challenge":
|
|
192
|
+
return "challenge";
|
|
193
|
+
case "request_admission":
|
|
194
|
+
return "admit";
|
|
195
|
+
default:
|
|
196
|
+
return null;
|
|
197
|
+
}
|
|
198
|
+
};
|
|
199
|
+
const sameProgressFingerprint = (left, right) => left.acceptedEvidenceDigest === right.acceptedEvidenceDigest &&
|
|
200
|
+
left.completedGateDigest === right.completedGateDigest &&
|
|
201
|
+
left.environmentDigest === right.environmentDigest &&
|
|
202
|
+
(left.continuationProgressDigest ?? null) ===
|
|
203
|
+
(right.continuationProgressDigest ?? null);
|
|
204
|
+
const NON_SCIENTIFIC_DIGEST_KEYS = new Set([
|
|
205
|
+
"callId",
|
|
206
|
+
"completedAt",
|
|
207
|
+
"createdAt",
|
|
208
|
+
"durationMs",
|
|
209
|
+
"elapsedMs",
|
|
210
|
+
"observedAt",
|
|
211
|
+
"operationId",
|
|
212
|
+
"requestId",
|
|
213
|
+
"startedAt",
|
|
214
|
+
"updatedAt"
|
|
215
|
+
]);
|
|
216
|
+
export const acceptedScientificDigestV1 = (value) => {
|
|
217
|
+
const normalize = (current) => {
|
|
218
|
+
if (Array.isArray(current))
|
|
219
|
+
return current.map(normalize);
|
|
220
|
+
if (typeof current !== "object" || current === null)
|
|
221
|
+
return current;
|
|
222
|
+
return Object.fromEntries(Object.entries(current)
|
|
223
|
+
.filter(([key]) => !NON_SCIENTIFIC_DIGEST_KEYS.has(key))
|
|
224
|
+
.map(([key, nested]) => [key, normalize(nested)]));
|
|
225
|
+
};
|
|
226
|
+
return agenticDigestV1(normalize(value));
|
|
227
|
+
};
|
|
228
|
+
/**
|
|
229
|
+
* One candidate's trusted scientific runtime. The caller resolves campaign/source/model authority;
|
|
230
|
+
* tool JSON supplies handles, never context, checkpoint maps, or provider identities.
|
|
231
|
+
* The supplied store owns persistence; immutable artifact snapshots precede each mutable head.
|
|
232
|
+
*/
|
|
233
|
+
export const makeEvalCapabilitiesV1 = Effect.fn("EvalCapabilities.make")(function* (input) {
|
|
234
|
+
const inner = yield* RepositoryFoundryLanguageModel;
|
|
235
|
+
const store = input.store ?? makeCaseCheckpointStoreV1({ caseDirectory: input.context.caseDirectory });
|
|
236
|
+
const restoredGroundedWork = yield* store.readWorkCheckpoint(GROUNDED_WORK_INDEX, GroundedWorkIndexV1);
|
|
237
|
+
const groundedWork = new Map(Option.isSome(restoredGroundedWork)
|
|
238
|
+
? restoredGroundedWork.value.map((entry) => [entry.name, entry])
|
|
239
|
+
: []);
|
|
240
|
+
const groundedContinuationValue = () => ({
|
|
241
|
+
version: 1,
|
|
242
|
+
candidateRevision: input.proposal.candidate.revision,
|
|
243
|
+
checkpoints: [...groundedWork.values()].sort((left, right) => left.name.localeCompare(right.name))
|
|
244
|
+
});
|
|
245
|
+
const groundedArtifact = () => {
|
|
246
|
+
if (groundedWork.size === 0)
|
|
247
|
+
return undefined;
|
|
248
|
+
const contentDigest = acceptedScientificDigestV1(groundedContinuationValue());
|
|
249
|
+
return {
|
|
250
|
+
handle: `continuation-progress-grounded-${contentDigest}`,
|
|
251
|
+
contentDigest,
|
|
252
|
+
kind: "continuation_progress"
|
|
253
|
+
};
|
|
254
|
+
};
|
|
255
|
+
const saveGroundedIndex = () => store.writeWorkCheckpoint(GROUNDED_WORK_INDEX, [...groundedWork.values()]);
|
|
256
|
+
const workPersistence = yield* Semaphore.make(1);
|
|
257
|
+
const workCheckpointStoreFor = (stage) => ({
|
|
258
|
+
...store,
|
|
259
|
+
writeWorkCheckpoint: (name, value) => workPersistence.withPermit(Effect.gen(function* () {
|
|
260
|
+
yield* store.writeWorkCheckpoint(name, value);
|
|
261
|
+
if (!name.startsWith("grounded-authoring-") &&
|
|
262
|
+
!name.startsWith("model-feedback-") &&
|
|
263
|
+
!name.startsWith("pipeline-execution-"))
|
|
264
|
+
return;
|
|
265
|
+
groundedWork.set(name, { name, stage, contentDigest: acceptedScientificDigestV1(value) });
|
|
266
|
+
yield* saveGroundedIndex();
|
|
267
|
+
})).pipe(Effect.uninterruptible)
|
|
268
|
+
});
|
|
269
|
+
let prepared = input.environment;
|
|
270
|
+
const retainedQualificationProgress = yield* store.readWorkCheckpoint(REFERENCE_QUALIFICATION_WORK_CHECKPOINT, RepositorySeedQualificationProgressV1);
|
|
271
|
+
let qualificationProgress = Option.isSome(retainedQualificationProgress)
|
|
272
|
+
? retainedQualificationProgress.value
|
|
273
|
+
: undefined;
|
|
274
|
+
const candidate = input.proposal.candidate;
|
|
275
|
+
const acceptedScientificArtifacts = new Map((input.acceptedScientificArtifacts ?? []).map((artifact) => [
|
|
276
|
+
artifact.handle,
|
|
277
|
+
artifact
|
|
278
|
+
]));
|
|
279
|
+
const initialProgressReceipts = [...(input.progressReceipts ?? [])];
|
|
280
|
+
let checkpoints = {};
|
|
281
|
+
const tournamentReady = () => checkpoints.tournament?.adequacy.admitted === true &&
|
|
282
|
+
checkpoints.spec !== undefined &&
|
|
283
|
+
checkpoints.oracle !== undefined &&
|
|
284
|
+
evaluateRepositoryOracleScopeCoverageV1({
|
|
285
|
+
spec: checkpoints.spec, oracleCheckpoint: checkpoints.oracle,
|
|
286
|
+
oracle: checkpoints.tournament.oracle,
|
|
287
|
+
repetitions: input.context.oracleRepetitions, label: "development"
|
|
288
|
+
}).reasons.length === 0;
|
|
289
|
+
let pendingChallenge;
|
|
290
|
+
for (const stage of PIPELINE_STAGES) {
|
|
291
|
+
if (stage === "challenge") {
|
|
292
|
+
const value = yield* store.readStage(stage, Schema.Union([PIPELINE_CHECKPOINT_SCHEMAS.challenge, PendingHeldOutCheckpointV1]));
|
|
293
|
+
if (Option.isSome(value)) {
|
|
294
|
+
if ("stage" in value.value)
|
|
295
|
+
checkpoints = { ...checkpoints, challenge: value.value };
|
|
296
|
+
else
|
|
297
|
+
pendingChallenge = value.value;
|
|
298
|
+
}
|
|
299
|
+
continue;
|
|
300
|
+
}
|
|
301
|
+
const value = yield* store.readStage(stage, PIPELINE_CHECKPOINT_SCHEMAS[stage]);
|
|
302
|
+
if (Option.isSome(value))
|
|
303
|
+
checkpoints = { ...checkpoints, [stage]: value.value };
|
|
304
|
+
}
|
|
305
|
+
if (checkpoints.seed !== undefined) {
|
|
306
|
+
yield* Effect.try({
|
|
307
|
+
try: () => assertRepositoryTaskSeedV1(checkpoints.seed.seed),
|
|
308
|
+
catch: (cause) => new RepositoryFoundryError({
|
|
309
|
+
operation: "extract-task-seeds",
|
|
310
|
+
classification: "environment_blocked",
|
|
311
|
+
detail: "retained seed checkpoint does not satisfy the executed behavioral evidence contract; select another candidate or repair the grading environment",
|
|
312
|
+
cause
|
|
313
|
+
})
|
|
314
|
+
});
|
|
315
|
+
}
|
|
316
|
+
const existingState = yield* store.readState;
|
|
317
|
+
const startedAt = yield* Clock.currentTimeMillis;
|
|
318
|
+
const initialState = Option.isSome(existingState)
|
|
319
|
+
? existingState.value
|
|
320
|
+
: {
|
|
321
|
+
version: 1,
|
|
322
|
+
caseId: input.context.caseId,
|
|
323
|
+
repositoryRoot: input.context.repositoryRoot,
|
|
324
|
+
dimension: input.context.dimension,
|
|
325
|
+
model: input.context.primaryModel,
|
|
326
|
+
identityDigest: candidate.revision,
|
|
327
|
+
stage: "seed",
|
|
328
|
+
status: "running",
|
|
329
|
+
startedAt: new Date(startedAt).toISOString(),
|
|
330
|
+
updatedAt: new Date(startedAt).toISOString(),
|
|
331
|
+
modelCalls: 0,
|
|
332
|
+
elapsedMs: 0,
|
|
333
|
+
resumeCount: 0,
|
|
334
|
+
folds: 0
|
|
335
|
+
};
|
|
336
|
+
if (initialState.identityDigest !== candidate.revision ||
|
|
337
|
+
initialState.caseId !== input.context.caseId)
|
|
338
|
+
return yield* failure("candidate store belongs to a different immutable proposal");
|
|
339
|
+
const state = yield* makePipelineStateAccessorV1(initialState);
|
|
340
|
+
const persist = (stage, value) => Effect.gen(function* () {
|
|
341
|
+
// Normalize before hashing, exactly like legacy checkpoint persistence.
|
|
342
|
+
// Runtime oracle solutions may carry execution-only fields that the persisted
|
|
343
|
+
// schema drops; hashing those would make every restored parent handle stale.
|
|
344
|
+
const checkpoint = yield* Schema.decodeUnknownEffect(PIPELINE_CHECKPOINT_SCHEMAS[stage])(value).pipe(Effect.mapError((cause) => failure("invalid scientific checkpoint", cause)));
|
|
345
|
+
const contentDigest = checkpointDigestV1(checkpoint);
|
|
346
|
+
const artifact = {
|
|
347
|
+
handle: evalCapabilityArtifactHandleV1(stage, checkpoint),
|
|
348
|
+
stage,
|
|
349
|
+
contentDigest,
|
|
350
|
+
checkpoint
|
|
351
|
+
};
|
|
352
|
+
yield* store.writeCall(`artifact-${contentDigest}`, 1, {
|
|
353
|
+
request: { stage },
|
|
354
|
+
response: checkpoint
|
|
355
|
+
});
|
|
356
|
+
if (input.onArtifact !== undefined)
|
|
357
|
+
yield* input.onArtifact(artifact);
|
|
358
|
+
yield* store.writeStage(stage, checkpoint);
|
|
359
|
+
checkpoints = { ...checkpoints, [stage]: checkpoint };
|
|
360
|
+
// A completed stage supersedes its authoring continuation. Keep the
|
|
361
|
+
// durable work for crash recovery, but never advertise it as another gate.
|
|
362
|
+
const complete = stage === "spec"
|
|
363
|
+
? acceptedSpec(checkpoints.spec)
|
|
364
|
+
: stage === "oracle" ||
|
|
365
|
+
(stage === "controls" &&
|
|
366
|
+
(checkpoints.controls?.valid.length ?? 0) > 0 &&
|
|
367
|
+
(checkpoints.controls?.wrong.length ?? 0) > 0) ||
|
|
368
|
+
stage === "challenge" ||
|
|
369
|
+
(stage === "tournament" && tournamentReady());
|
|
370
|
+
if (complete) {
|
|
371
|
+
for (const [name, entry] of groundedWork) {
|
|
372
|
+
if (entry.stage === stage)
|
|
373
|
+
groundedWork.delete(name);
|
|
374
|
+
}
|
|
375
|
+
yield* saveGroundedIndex();
|
|
376
|
+
}
|
|
377
|
+
acceptedScientificArtifacts.set(artifact.handle, {
|
|
378
|
+
handle: artifact.handle,
|
|
379
|
+
contentDigest: acceptedScientificDigestV1(checkpoint),
|
|
380
|
+
kind: stage
|
|
381
|
+
});
|
|
382
|
+
});
|
|
383
|
+
const stageInput = (stage) => ({
|
|
384
|
+
context: input.context,
|
|
385
|
+
map: input.map,
|
|
386
|
+
checkpoints,
|
|
387
|
+
workCheckpointStore: workCheckpointStoreFor(stage),
|
|
388
|
+
checkpoint: (value) => persist(stage, value).pipe(Effect.orDie)
|
|
389
|
+
});
|
|
390
|
+
const getHandle = (stage) => checkpoints[stage] === undefined
|
|
391
|
+
? undefined
|
|
392
|
+
: evalCapabilityArtifactHandleV1(stage, checkpoints[stage]);
|
|
393
|
+
const scientificInputs = () => ({
|
|
394
|
+
sourceDigest: candidate.sourceDigest,
|
|
395
|
+
environmentDigest: agenticDigestV1(prepared?.plan ?? checkpoints.seed?.seed.environment ?? null),
|
|
396
|
+
specificationDigest: checkpointDigestV1(checkpoints.spec ?? null),
|
|
397
|
+
oracleDigest: checkpointDigestV1(checkpoints.oracle ?? null),
|
|
398
|
+
controlsDigest: checkpointDigestV1(checkpoints.controls ?? null),
|
|
399
|
+
policyDigest: agenticDigestV1({
|
|
400
|
+
repetitions: input.context.oracleRepetitions,
|
|
401
|
+
foldLimit: input.context.foldLimit
|
|
402
|
+
}),
|
|
403
|
+
modelPlanDigest: agenticDigestV1(input.context.modelPlan)
|
|
404
|
+
});
|
|
405
|
+
const frozenHandle = () => `frozen-${agenticDigestV1({ candidate: candidate.revision, inputs: scientificInputs(), tournament: checkpointDigestV1(checkpoints.tournament ?? null) })}`;
|
|
406
|
+
const qualificationBindingDigest = (environment = prepared) => environment === undefined
|
|
407
|
+
? undefined
|
|
408
|
+
: agenticDigestV1({
|
|
409
|
+
version: 2,
|
|
410
|
+
candidateRevision: candidate.revision,
|
|
411
|
+
sourceDigest: candidate.sourceDigest,
|
|
412
|
+
initialCommit: input.proposal.seed.initialState.commit,
|
|
413
|
+
referenceCommit: input.proposal.seed.referenceCommit,
|
|
414
|
+
environment: environment.environment,
|
|
415
|
+
executionMode: environment.plan.executionMode,
|
|
416
|
+
runtime: environment.plan.runtime,
|
|
417
|
+
batchesRequired: 2,
|
|
418
|
+
repetitionsPerBatch: 2
|
|
419
|
+
});
|
|
420
|
+
const qualificationProgressArtifact = () => {
|
|
421
|
+
const bindingDigest = qualificationBindingDigest();
|
|
422
|
+
if (qualificationProgress === undefined ||
|
|
423
|
+
bindingDigest === undefined ||
|
|
424
|
+
qualificationProgress.bindingDigest !== bindingDigest)
|
|
425
|
+
return undefined;
|
|
426
|
+
const contentDigest = acceptedScientificDigestV1(qualificationProgress);
|
|
427
|
+
return {
|
|
428
|
+
handle: `qualification-progress-${contentDigest}`,
|
|
429
|
+
contentDigest,
|
|
430
|
+
kind: "qualification_progress"
|
|
431
|
+
};
|
|
432
|
+
};
|
|
433
|
+
const persistQualificationProgress = (progress) => {
|
|
434
|
+
const handle = `qualification-progress-${acceptedScientificDigestV1(progress)}`;
|
|
435
|
+
return store
|
|
436
|
+
.writeWorkCheckpoint(REFERENCE_QUALIFICATION_WORK_CHECKPOINT, progress)
|
|
437
|
+
.pipe(Effect.andThen(store.writeCall(handle, 1, {
|
|
438
|
+
request: {
|
|
439
|
+
tool: "qualify_reference",
|
|
440
|
+
candidateRevision: candidate.revision
|
|
441
|
+
},
|
|
442
|
+
response: progress
|
|
443
|
+
})), Effect.tap(() => Effect.sync(() => {
|
|
444
|
+
qualificationProgress = progress;
|
|
445
|
+
})));
|
|
446
|
+
};
|
|
447
|
+
const challengeContinuationValue = (pending) => ({
|
|
448
|
+
version: 1,
|
|
449
|
+
capability: "run_held_out_challenge",
|
|
450
|
+
candidateRevision: candidate.revision,
|
|
451
|
+
bindingDigest: agenticDigestV1(pending.frozen),
|
|
452
|
+
checkpoint: pending
|
|
453
|
+
});
|
|
454
|
+
const challengeProgressArtifact = () => {
|
|
455
|
+
if (pendingChallenge === undefined ||
|
|
456
|
+
checkpoints.challenge !== undefined ||
|
|
457
|
+
pendingChallenge.frozen !== frozenHandle())
|
|
458
|
+
return undefined;
|
|
459
|
+
const contentDigest = acceptedScientificDigestV1(challengeContinuationValue(pendingChallenge));
|
|
460
|
+
return {
|
|
461
|
+
handle: `continuation-progress-challenge-${contentDigest}`,
|
|
462
|
+
contentDigest,
|
|
463
|
+
kind: "continuation_progress"
|
|
464
|
+
};
|
|
465
|
+
};
|
|
466
|
+
const persistChallengeProgress = (pending) => Effect.gen(function* () {
|
|
467
|
+
// Normalize before both persistence and hashing so restoration yields
|
|
468
|
+
// exactly the same accepted continuation, including metadata-only oracle solutions.
|
|
469
|
+
const checkpoint = yield* Schema.decodeUnknownEffect(PendingHeldOutCheckpointV1)(pending).pipe(Effect.mapError((cause) => failure("invalid held-out continuation", cause)));
|
|
470
|
+
const value = challengeContinuationValue(checkpoint);
|
|
471
|
+
const handle = `continuation-progress-challenge-${acceptedScientificDigestV1(value)}`;
|
|
472
|
+
yield* store.writeCall(handle, 1, {
|
|
473
|
+
request: { tool: "run_held_out_challenge", candidateRevision: candidate.revision },
|
|
474
|
+
response: value
|
|
475
|
+
});
|
|
476
|
+
yield* store.writeStage("challenge", checkpoint);
|
|
477
|
+
pendingChallenge = checkpoint;
|
|
478
|
+
}).pipe(Effect.uninterruptible, Effect.orDie);
|
|
479
|
+
const completedScientificGates = () => {
|
|
480
|
+
const gates = [];
|
|
481
|
+
if (prepared !== undefined)
|
|
482
|
+
gates.push("environment_planned");
|
|
483
|
+
if (acceptedSpec(checkpoints.spec))
|
|
484
|
+
gates.push("specification_accepted");
|
|
485
|
+
if (checkpoints.seed?.seed.status === "qualified")
|
|
486
|
+
gates.push("reference_qualified");
|
|
487
|
+
if (checkpoints.oracle !== undefined)
|
|
488
|
+
gates.push("oracle_authored");
|
|
489
|
+
if ((checkpoints.controls?.valid.length ?? 0) > 0)
|
|
490
|
+
gates.push("valid_controls_generated");
|
|
491
|
+
if ((checkpoints.controls?.wrong.length ?? 0) > 0)
|
|
492
|
+
gates.push("wrong_controls_generated");
|
|
493
|
+
if (tournamentReady())
|
|
494
|
+
gates.push("oracle_evaluated");
|
|
495
|
+
if (acceptedScientificArtifacts.has(frozenHandle()))
|
|
496
|
+
gates.push("candidate_frozen");
|
|
497
|
+
if (checkpoints.challenge?.verdict === "passed")
|
|
498
|
+
gates.push("held_out_challenge_passed");
|
|
499
|
+
if (checkpoints.admit?.verdict === "admitted")
|
|
500
|
+
gates.push("candidate_admitted");
|
|
501
|
+
return gates;
|
|
502
|
+
};
|
|
503
|
+
const currentScientificArtifacts = () => {
|
|
504
|
+
const current = PIPELINE_STAGES.flatMap((stage) => {
|
|
505
|
+
const checkpoint = checkpoints[stage];
|
|
506
|
+
if (checkpoint === undefined)
|
|
507
|
+
return [];
|
|
508
|
+
return [
|
|
509
|
+
{
|
|
510
|
+
handle: evalCapabilityArtifactHandleV1(stage, checkpoint),
|
|
511
|
+
contentDigest: acceptedScientificDigestV1(checkpoint),
|
|
512
|
+
kind: stage
|
|
513
|
+
}
|
|
514
|
+
];
|
|
515
|
+
});
|
|
516
|
+
if (prepared !== undefined) {
|
|
517
|
+
current.push({
|
|
518
|
+
handle: prepared.plan.planId,
|
|
519
|
+
contentDigest: acceptedScientificDigestV1(prepared),
|
|
520
|
+
kind: "environment"
|
|
521
|
+
});
|
|
522
|
+
}
|
|
523
|
+
const continuation = qualificationProgressArtifact();
|
|
524
|
+
if (continuation !== undefined)
|
|
525
|
+
current.push(continuation);
|
|
526
|
+
const challengeContinuation = challengeProgressArtifact();
|
|
527
|
+
if (challengeContinuation !== undefined)
|
|
528
|
+
current.push(challengeContinuation);
|
|
529
|
+
const groundedContinuation = groundedArtifact();
|
|
530
|
+
if (groundedContinuation !== undefined)
|
|
531
|
+
current.push(groundedContinuation);
|
|
532
|
+
for (const artifact of acceptedScientificArtifacts.values()) {
|
|
533
|
+
if (artifact.kind === "observation" ||
|
|
534
|
+
(artifact.kind === "frozen_candidate" && artifact.handle === frozenHandle())) {
|
|
535
|
+
current.push(artifact);
|
|
536
|
+
}
|
|
537
|
+
}
|
|
538
|
+
return [
|
|
539
|
+
...new Map(current.map((artifact) => [
|
|
540
|
+
`${artifact.kind}:${artifact.contentDigest}`,
|
|
541
|
+
artifact
|
|
542
|
+
])).values()
|
|
543
|
+
].sort((left, right) => `${left.kind}:${left.contentDigest}`.localeCompare(`${right.kind}:${right.contentDigest}`));
|
|
544
|
+
};
|
|
545
|
+
const scientificFingerprint = () => makeAgenticProgressFingerprintV1({
|
|
546
|
+
acceptedEvidence: currentScientificArtifacts(),
|
|
547
|
+
completedGates: completedScientificGates(),
|
|
548
|
+
environmentDigest: prepared === undefined ? null : acceptedScientificDigestV1(prepared.plan)
|
|
549
|
+
});
|
|
550
|
+
const requireHandle = (received, expected, label) => {
|
|
551
|
+
if (expected === undefined || received !== expected)
|
|
552
|
+
throw failure(`unaccepted or stale ${label} handle`);
|
|
553
|
+
};
|
|
554
|
+
const readyToFreeze = () => checkpoints.seed?.seed.status === "qualified" &&
|
|
555
|
+
acceptedSpec(checkpoints.spec) &&
|
|
556
|
+
checkpoints.oracle !== undefined &&
|
|
557
|
+
checkpoints.controls !== undefined &&
|
|
558
|
+
checkpoints.controls.valid.length > 0 &&
|
|
559
|
+
checkpoints.controls.wrong.length > 0 &&
|
|
560
|
+
tournamentReady();
|
|
561
|
+
const assertReady = () => {
|
|
562
|
+
if (!readyToFreeze())
|
|
563
|
+
throw failure("freeze requires executed reference, reviewed specification, controls and an adequate tournament");
|
|
564
|
+
const seed = checkpoints.seed;
|
|
565
|
+
if (seed === undefined)
|
|
566
|
+
throw failure("freeze requires a qualified seed checkpoint");
|
|
567
|
+
assertRepositoryTaskSeedV1(seed.seed);
|
|
568
|
+
for (const [parent, child] of [
|
|
569
|
+
["seed", "spec"],
|
|
570
|
+
["spec", "oracle"],
|
|
571
|
+
["oracle", "controls"],
|
|
572
|
+
["controls", "tournament"]
|
|
573
|
+
]) {
|
|
574
|
+
if (checkpoints[child].parentDigest !== checkpointDigestV1(checkpoints[parent]))
|
|
575
|
+
throw failure(`stale ${child} evidence does not bind the current ${parent}`);
|
|
576
|
+
}
|
|
577
|
+
};
|
|
578
|
+
/**
|
|
579
|
+
* One prerequisite registry drives both capability advertisement and
|
|
580
|
+
* execution admission. Argument-handle validation remains capability-local,
|
|
581
|
+
* but the host can no longer advertise a transition that the same runtime
|
|
582
|
+
* will deterministically reject for missing scientific state.
|
|
583
|
+
*/
|
|
584
|
+
const hasCapabilityPrerequisites = (capability) => {
|
|
585
|
+
const specificationAccepted = acceptedSpec(checkpoints.spec);
|
|
586
|
+
const referenceQualified = checkpoints.seed?.seed.status === "qualified";
|
|
587
|
+
const environmentPlanned = prepared !== undefined;
|
|
588
|
+
const oracleAuthored = checkpoints.oracle !== undefined;
|
|
589
|
+
const controlsGenerated = checkpoints.controls !== undefined &&
|
|
590
|
+
(checkpoints.controls.valid.length > 0 ||
|
|
591
|
+
checkpoints.controls.wrong.length > 0);
|
|
592
|
+
const challengeBindsCurrentTournament = checkpoints.challenge !== undefined &&
|
|
593
|
+
checkpoints.challenge.parentDigest ===
|
|
594
|
+
checkpointDigestV1(checkpoints.tournament ?? null);
|
|
595
|
+
// A persisted challenge is itself durable proof that the current
|
|
596
|
+
// tournament was frozen. This keeps replay/admission available after a
|
|
597
|
+
// process restart even when the caller does not rehydrate call artifacts.
|
|
598
|
+
const frozen = acceptedScientificArtifacts.has(frozenHandle()) ||
|
|
599
|
+
challengeBindsCurrentTournament;
|
|
600
|
+
switch (capability) {
|
|
601
|
+
case "get_progress":
|
|
602
|
+
case "reject_candidate":
|
|
603
|
+
case "finish_campaign":
|
|
604
|
+
return true;
|
|
605
|
+
case "draft_specification":
|
|
606
|
+
return !referenceQualified;
|
|
607
|
+
case "plan_environment":
|
|
608
|
+
return !referenceQualified;
|
|
609
|
+
case "probe_fixture":
|
|
610
|
+
return environmentPlanned && !referenceQualified;
|
|
611
|
+
case "qualify_reference":
|
|
612
|
+
return specificationAccepted && environmentPlanned && !referenceQualified;
|
|
613
|
+
case "author_oracle":
|
|
614
|
+
return specificationAccepted && referenceQualified;
|
|
615
|
+
case "generate_controls":
|
|
616
|
+
return (specificationAccepted &&
|
|
617
|
+
referenceQualified &&
|
|
618
|
+
environmentPlanned &&
|
|
619
|
+
oracleAuthored &&
|
|
620
|
+
checkpoints.challenge === undefined);
|
|
621
|
+
case "evaluate_oracle":
|
|
622
|
+
case "repair_oracle":
|
|
623
|
+
return oracleAuthored && controlsGenerated;
|
|
624
|
+
case "freeze_candidate":
|
|
625
|
+
return readyToFreeze();
|
|
626
|
+
case "run_held_out_challenge":
|
|
627
|
+
return readyToFreeze() && frozen;
|
|
628
|
+
case "request_admission":
|
|
629
|
+
return (readyToFreeze() &&
|
|
630
|
+
frozen &&
|
|
631
|
+
checkpoints.challenge?.verdict === "passed");
|
|
632
|
+
default:
|
|
633
|
+
return false;
|
|
634
|
+
}
|
|
635
|
+
};
|
|
636
|
+
const currentScorecard = (disposition, reasons) => Effect.gen(function* () {
|
|
637
|
+
const current = yield* state.get;
|
|
638
|
+
const scorecard = yield* Effect.try({
|
|
639
|
+
try: () => evalCandidateQualityScorecardV1({
|
|
640
|
+
candidate: candidate.candidate,
|
|
641
|
+
revision: candidate.revision,
|
|
642
|
+
checkpoints,
|
|
643
|
+
disposition,
|
|
644
|
+
reasons,
|
|
645
|
+
elapsedMs: current.elapsedMs,
|
|
646
|
+
receivedCalls: current.modelCalls
|
|
647
|
+
}),
|
|
648
|
+
catch: (cause) => failure("could not encode candidate quality evidence", cause)
|
|
649
|
+
});
|
|
650
|
+
yield* store.writeCall(`scorecard-${agenticDigestV1(scorecard)}`, 1, {
|
|
651
|
+
request: { candidate: candidate.candidate },
|
|
652
|
+
response: scorecard
|
|
653
|
+
});
|
|
654
|
+
if (input.onScorecard !== undefined)
|
|
655
|
+
yield* input.onScorecard(scorecard);
|
|
656
|
+
return scorecard;
|
|
657
|
+
});
|
|
658
|
+
const invoke = Effect.fn("EvalCapabilities.invoke")(function* (invocation) {
|
|
659
|
+
const request = yield* Effect.try({
|
|
660
|
+
try: () => decodeAgenticCapabilityRequestV1(invocation.request),
|
|
661
|
+
catch: (cause) => failure("invalid capability request", cause)
|
|
662
|
+
});
|
|
663
|
+
const before = yield* state.get;
|
|
664
|
+
const invocationStartedAt = yield* Clock.currentTimeMillis;
|
|
665
|
+
const scientificPhase = evalCapabilityScientificPhaseV1(request.tool);
|
|
666
|
+
const phasePolicy = scientificPhase === null
|
|
667
|
+
? null
|
|
668
|
+
: EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1[scientificPhase];
|
|
669
|
+
const beforeArtifacts = currentScientificArtifacts();
|
|
670
|
+
const beforeArtifactHandles = new Set(beforeArtifacts.map((artifact) => artifact.handle));
|
|
671
|
+
const beforeGates = completedScientificGates();
|
|
672
|
+
const beforeFingerprint = scientificFingerprint();
|
|
673
|
+
const progressReceipts = [
|
|
674
|
+
...(invocation.authoritativeProgressReceipts ?? initialProgressReceipts)
|
|
675
|
+
].sort((left, right) => left.recordedAt.localeCompare(right.recordedAt));
|
|
676
|
+
const latestProgressReceipt = evalCapabilityScientificProgressPredecessorV1({
|
|
677
|
+
capability: request.tool,
|
|
678
|
+
fingerprint: beforeFingerprint,
|
|
679
|
+
history: progressReceipts
|
|
680
|
+
});
|
|
681
|
+
const repeatedUnchangedCapability = latestProgressReceipt !== undefined &&
|
|
682
|
+
latestProgressReceipt.capability === request.tool &&
|
|
683
|
+
latestProgressReceipt.sameFingerprintCount >= 2 &&
|
|
684
|
+
sameProgressFingerprint(latestProgressReceipt.resultingFingerprint, beforeFingerprint);
|
|
685
|
+
const phaseReceipts = scientificPhase === null
|
|
686
|
+
? []
|
|
687
|
+
: progressReceipts.filter((receipt) => receipt.phase === scientificPhase);
|
|
688
|
+
const phaseAttemptsConsumed = phaseReceipts.reduce((total, receipt) => total + evalCapabilityScientificAttemptCostV1(receipt), 0);
|
|
689
|
+
const parsedDeadline = invocation.deadlineAt === undefined ? undefined : Date.parse(invocation.deadlineAt);
|
|
690
|
+
if (parsedDeadline !== undefined && !Number.isFinite(parsedDeadline))
|
|
691
|
+
return yield* failure("invalid capability deadline");
|
|
692
|
+
const campaignWallTimeRemainingAtStart = Math.max(0, parsedDeadline === undefined
|
|
693
|
+
? input.context.budget.maxWallMs - before.elapsedMs
|
|
694
|
+
: parsedDeadline - invocationStartedAt);
|
|
695
|
+
const envelopeDecision = evalCapabilityExecutionEnvelopeV1({
|
|
696
|
+
capability: request.tool,
|
|
697
|
+
...(invocation.maximumModelCalls === undefined
|
|
698
|
+
? {}
|
|
699
|
+
: { requestedMaximumModelCalls: invocation.maximumModelCalls }),
|
|
700
|
+
...(invocation.maximumWallMs === undefined
|
|
701
|
+
? {}
|
|
702
|
+
: { requestedMaximumWallMs: invocation.maximumWallMs }),
|
|
703
|
+
campaignModelCallsRemaining: input.context.budget.maxModelCalls - before.modelCalls,
|
|
704
|
+
campaignWallTimeRemainingMs: campaignWallTimeRemainingAtStart
|
|
705
|
+
});
|
|
706
|
+
const downstreamCalls = envelopeDecision.downstreamReserve.calls;
|
|
707
|
+
const downstreamWallTimeMs = envelopeDecision.downstreamReserve.wallTimeMs;
|
|
708
|
+
const finalizationWallTimeMs = envelopeDecision.finalizationReserve.wallTimeMs;
|
|
709
|
+
const phaseAttemptsAvailable = phasePolicy === null
|
|
710
|
+
? Number.MAX_SAFE_INTEGER
|
|
711
|
+
: Math.max(0, phasePolicy.maximumAttempts - phaseAttemptsConsumed);
|
|
712
|
+
const maximumCalls = envelopeDecision.admitted
|
|
713
|
+
? envelopeDecision.envelope.maximumModelCalls
|
|
714
|
+
: 1;
|
|
715
|
+
const maximumWallMs = envelopeDecision.admitted
|
|
716
|
+
? envelopeDecision.envelope.maximumWallTimeMs
|
|
717
|
+
: 1;
|
|
718
|
+
const budget = {
|
|
719
|
+
maxModelCalls: Math.min(input.context.budget.maxModelCalls, before.modelCalls + maximumCalls),
|
|
720
|
+
maxWallMs: Math.min(input.context.budget.maxWallMs, before.elapsedMs + maximumWallMs)
|
|
721
|
+
};
|
|
722
|
+
const accepted = [];
|
|
723
|
+
let gradeExecuted;
|
|
724
|
+
let detail = "Capability completed; inspect accepted evidence before advancing.";
|
|
725
|
+
let disposition;
|
|
726
|
+
let outcome = "completed";
|
|
727
|
+
let yieldedAtSliceBoundary = false;
|
|
728
|
+
let completedLogicalAttempt = false;
|
|
729
|
+
const execute = Effect.gen(function* () {
|
|
730
|
+
if (repeatedUnchangedCapability) {
|
|
731
|
+
outcome = "scientific_rejection";
|
|
732
|
+
disposition = "rejected";
|
|
733
|
+
detail =
|
|
734
|
+
"Scientific no-progress circuit opened after two unchanged results for this capability. The third unchanged attempt was rejected before model, repository, or grader work, preserving downstream reserves.";
|
|
735
|
+
return;
|
|
736
|
+
}
|
|
737
|
+
if (!envelopeDecision.admitted ||
|
|
738
|
+
(scientificPhase !== null && phaseAttemptsAvailable <= 0)) {
|
|
739
|
+
outcome = "budget_exhausted";
|
|
740
|
+
detail = scientificPhase === null
|
|
741
|
+
? "The capability cannot fit within the remaining campaign budget."
|
|
742
|
+
: "The host-owned scientific phase budget cannot admit this operation without consuming the mandatory downstream reserve.";
|
|
743
|
+
return;
|
|
744
|
+
}
|
|
745
|
+
yield* Effect.try({
|
|
746
|
+
try: () => {
|
|
747
|
+
if ("candidate" in request.args && request.args.candidate !== undefined)
|
|
748
|
+
requireHandle(request.args.candidate, candidate.candidate, "candidate");
|
|
749
|
+
},
|
|
750
|
+
catch: (cause) => cause instanceof RepositoryFoundryError
|
|
751
|
+
? cause
|
|
752
|
+
: failure("candidate handle validation failed", cause)
|
|
753
|
+
});
|
|
754
|
+
if ((before.status === "admitted" || before.status === "rejected") &&
|
|
755
|
+
request.tool !== "get_progress" &&
|
|
756
|
+
request.tool !== "finish_campaign")
|
|
757
|
+
return yield* failure("terminal candidate is immutable; inspect it or start a new candidate revision");
|
|
758
|
+
if (scientificPhase !== null &&
|
|
759
|
+
(before.modelCalls >= input.context.budget.maxModelCalls ||
|
|
760
|
+
campaignWallTimeRemainingAtStart <= 0)) {
|
|
761
|
+
outcome = "budget_exhausted";
|
|
762
|
+
return;
|
|
763
|
+
}
|
|
764
|
+
if (!hasCapabilityPrerequisites(request.tool)) {
|
|
765
|
+
return yield* failure(request.tool === "freeze_candidate"
|
|
766
|
+
? "candidate is not ready to freeze"
|
|
767
|
+
: request.tool === "qualify_reference"
|
|
768
|
+
? "reference qualification requires an accepted grounded specification review"
|
|
769
|
+
: `${request.tool} prerequisites are not satisfied by the accepted candidate state`);
|
|
770
|
+
}
|
|
771
|
+
switch (request.tool) {
|
|
772
|
+
case "draft_specification": {
|
|
773
|
+
if (checkpoints.seed?.seed.status === "qualified") {
|
|
774
|
+
outcome = "progress";
|
|
775
|
+
detail =
|
|
776
|
+
"Reference already qualified; use the retained source specification or start a new candidate revision.";
|
|
777
|
+
break;
|
|
778
|
+
}
|
|
779
|
+
const spec = yield* draftEvalCandidateV1({
|
|
780
|
+
proposal: input.proposal,
|
|
781
|
+
context: input.context,
|
|
782
|
+
map: input.map,
|
|
783
|
+
previous: checkpoints.spec,
|
|
784
|
+
workCheckpointStore: workCheckpointStoreFor("spec"),
|
|
785
|
+
checkpoint: (value) => persist("spec", value).pipe(Effect.orDie)
|
|
786
|
+
});
|
|
787
|
+
accepted.push(evalCapabilityArtifactHandleV1("spec", spec));
|
|
788
|
+
if (!acceptedSpec(spec)) {
|
|
789
|
+
outcome = "progress";
|
|
790
|
+
detail =
|
|
791
|
+
"Source-grounded draft retained with review findings; revise before qualification.";
|
|
792
|
+
}
|
|
793
|
+
break;
|
|
794
|
+
}
|
|
795
|
+
case "plan_environment": {
|
|
796
|
+
if (checkpoints.seed?.seed.status === "qualified")
|
|
797
|
+
return yield* failure("qualified environments are immutable; create a new candidate revision to change preparation");
|
|
798
|
+
const planned = yield* planEvalEnvironmentV1({
|
|
799
|
+
repositoryRoot: input.context.repositoryRoot,
|
|
800
|
+
candidate: candidate.candidate,
|
|
801
|
+
revision: candidate.revision,
|
|
802
|
+
sourceDigest: candidate.sourceDigest,
|
|
803
|
+
seed: input.proposal.seed,
|
|
804
|
+
recipeIds: request.args.recipeIds,
|
|
805
|
+
executionMode: request.args.executionMode,
|
|
806
|
+
requiredArtifacts: request.args.requiredArtifacts,
|
|
807
|
+
runtime: input.runtime,
|
|
808
|
+
justification: request.args.justification
|
|
809
|
+
});
|
|
810
|
+
// Explanatory rewrites are not environment repairs. Preserve the
|
|
811
|
+
// accepted handle and failed-unit evidence when executable inputs
|
|
812
|
+
// are unchanged; only a real plan change may restart qualification.
|
|
813
|
+
prepared =
|
|
814
|
+
prepared === undefined ||
|
|
815
|
+
qualificationBindingDigest(planned) !== qualificationBindingDigest() ||
|
|
816
|
+
agenticDigestV1(planned.plan.requiredArtifacts) !==
|
|
817
|
+
agenticDigestV1(prepared.plan.requiredArtifacts)
|
|
818
|
+
? planned
|
|
819
|
+
: prepared;
|
|
820
|
+
if (qualificationProgress !== undefined &&
|
|
821
|
+
qualificationProgress.bindingDigest !== qualificationBindingDigest()) {
|
|
822
|
+
yield* store.deleteWorkCheckpoint(REFERENCE_QUALIFICATION_WORK_CHECKPOINT);
|
|
823
|
+
qualificationProgress = undefined;
|
|
824
|
+
}
|
|
825
|
+
yield* store.writeCall(prepared.plan.planId, 1, { request, response: prepared });
|
|
826
|
+
acceptedScientificArtifacts.set(prepared.plan.planId, {
|
|
827
|
+
handle: prepared.plan.planId,
|
|
828
|
+
contentDigest: acceptedScientificDigestV1(prepared),
|
|
829
|
+
kind: "environment"
|
|
830
|
+
});
|
|
831
|
+
if (input.onEnvironment !== undefined)
|
|
832
|
+
yield* input.onEnvironment(prepared);
|
|
833
|
+
accepted.push(prepared.plan.planId);
|
|
834
|
+
break;
|
|
835
|
+
}
|
|
836
|
+
case "probe_fixture": {
|
|
837
|
+
if (prepared === undefined) {
|
|
838
|
+
outcome = "environment_blocked";
|
|
839
|
+
gradeExecuted = false;
|
|
840
|
+
detail = "Select an explicit environment plan before repository execution.";
|
|
841
|
+
break;
|
|
842
|
+
}
|
|
843
|
+
yield* Effect.try({
|
|
844
|
+
try: () => requireHandle(request.args.environment, prepared.plan.planId, "environment"),
|
|
845
|
+
catch: (cause) => failure("invalid environment handle", cause)
|
|
846
|
+
});
|
|
847
|
+
if (input.resolveFixtureDraft === undefined)
|
|
848
|
+
return yield* failure("fixture draft handle resolver is unavailable; no grade was executed");
|
|
849
|
+
const suite = yield* input.resolveFixtureDraft(request.args.fixtureDraft).pipe(Effect.flatMap((value) => Schema.decodeUnknownEffect(RepositoryHiddenFixtureSuiteV1)(value)), Effect.mapError((cause) => failure("invalid fixture draft", cause)));
|
|
850
|
+
// A diagnostic probe materializes one workspace. Helpers and test files
|
|
851
|
+
// must coexist; this is not the independent-overlay scientific admission gate.
|
|
852
|
+
if (suite.overlays.length === 0 ||
|
|
853
|
+
new Set(suite.overlays.map((overlay) => overlay.path)).size !== suite.overlays.length)
|
|
854
|
+
return yield* failure("a fixture probe requires a nonempty workspace with distinct overlay paths");
|
|
855
|
+
const results = yield* replayRepositoryCommandsV1({
|
|
856
|
+
repositoryRoot: input.context.repositoryRoot,
|
|
857
|
+
commit: input.proposal.seed.referenceCommit ?? input.proposal.seed.initialState.commit,
|
|
858
|
+
recipes: prepared.environment.gradeRecipes,
|
|
859
|
+
trustedPreparationRecipes: prepared.environment.trustedPreparationRecipes,
|
|
860
|
+
candidateValidationRecipes: prepared.environment.candidateValidationRecipes,
|
|
861
|
+
contentOverlays: suite.overlays.map(({ path, content }) => ({ path, content })),
|
|
862
|
+
repetitions: 1
|
|
863
|
+
});
|
|
864
|
+
outcome = repositoryCommandProbeOutcomeV1(results);
|
|
865
|
+
gradeExecuted = outcome !== "environment_blocked";
|
|
866
|
+
const probe = { suiteDigest: agenticDigestV1(suite), results, gradeExecuted, outcome };
|
|
867
|
+
const handle = `probe-${agenticDigestV1(probe)}`;
|
|
868
|
+
yield* store.writeCall(handle, 1, { request, response: probe });
|
|
869
|
+
acceptedScientificArtifacts.set(handle, {
|
|
870
|
+
handle,
|
|
871
|
+
contentDigest: acceptedScientificDigestV1(probe),
|
|
872
|
+
kind: "observation"
|
|
873
|
+
});
|
|
874
|
+
accepted.push(handle);
|
|
875
|
+
break;
|
|
876
|
+
}
|
|
877
|
+
case "qualify_reference": {
|
|
878
|
+
if (!acceptedSpec(checkpoints.spec))
|
|
879
|
+
return yield* failure("reference qualification requires an accepted grounded specification review");
|
|
880
|
+
if (prepared === undefined) {
|
|
881
|
+
outcome = "environment_blocked";
|
|
882
|
+
gradeExecuted = false;
|
|
883
|
+
detail = "Plan the environment before qualification.";
|
|
884
|
+
break;
|
|
885
|
+
}
|
|
886
|
+
yield* Effect.try({
|
|
887
|
+
try: () => requireHandle(request.args.environment, prepared.plan.planId, "environment"),
|
|
888
|
+
catch: (cause) => failure("invalid environment handle", cause)
|
|
889
|
+
});
|
|
890
|
+
if (checkpoints.seed?.seed.status !== "qualified") {
|
|
891
|
+
const bindingDigest = qualificationBindingDigest();
|
|
892
|
+
if (qualificationProgress !== undefined &&
|
|
893
|
+
qualificationProgress.bindingDigest !== bindingDigest) {
|
|
894
|
+
yield* store.deleteWorkCheckpoint(REFERENCE_QUALIFICATION_WORK_CHECKPOINT);
|
|
895
|
+
qualificationProgress = undefined;
|
|
896
|
+
}
|
|
897
|
+
const resumed = yield* qualifyHistoricalTaskSeedAcrossBatchesResumableV1({
|
|
898
|
+
repositoryRoot: input.context.repositoryRoot,
|
|
899
|
+
seed: { ...input.proposal.seed, environment: prepared.environment },
|
|
900
|
+
bindingDigest,
|
|
901
|
+
...(qualificationProgress === undefined
|
|
902
|
+
? {}
|
|
903
|
+
: { progress: qualificationProgress }),
|
|
904
|
+
checkpoint: persistQualificationProgress,
|
|
905
|
+
batches: 2,
|
|
906
|
+
repetitionsPerBatch: 2
|
|
907
|
+
});
|
|
908
|
+
const qualification = resumed.qualification;
|
|
909
|
+
const checkpoint = {
|
|
910
|
+
version: 1,
|
|
911
|
+
stage: "seed",
|
|
912
|
+
parentDigest: checkpointDigestV1(input.context),
|
|
913
|
+
createdAt: new Date(yield* Clock.currentTimeMillis).toISOString(),
|
|
914
|
+
seed: qualification.seed,
|
|
915
|
+
qualification,
|
|
916
|
+
repository: {
|
|
917
|
+
root: input.context.repositoryRoot,
|
|
918
|
+
commit: input.map.repository.commit,
|
|
919
|
+
tree: input.map.repository.tree
|
|
920
|
+
},
|
|
921
|
+
selection: {
|
|
922
|
+
candidates: [
|
|
923
|
+
{
|
|
924
|
+
commit: input.proposal.seed.referenceCommit,
|
|
925
|
+
subject: input.proposal.seed.summary,
|
|
926
|
+
score: 0,
|
|
927
|
+
rationale: input.proposal.rationale
|
|
928
|
+
}
|
|
929
|
+
],
|
|
930
|
+
selectedCommit: input.proposal.seed.referenceCommit,
|
|
931
|
+
rejected: []
|
|
932
|
+
}
|
|
933
|
+
};
|
|
934
|
+
yield* persist("seed", checkpoint);
|
|
935
|
+
yield* store.deleteWorkCheckpoint(REFERENCE_QUALIFICATION_WORK_CHECKPOINT);
|
|
936
|
+
qualificationProgress = undefined;
|
|
937
|
+
if (checkpoints.spec !== undefined)
|
|
938
|
+
yield* persist("spec", {
|
|
939
|
+
...checkpoints.spec,
|
|
940
|
+
parentDigest: checkpointDigestV1(checkpoints.seed)
|
|
941
|
+
});
|
|
942
|
+
}
|
|
943
|
+
gradeExecuted = true;
|
|
944
|
+
if (checkpoints.seed?.seed.status !== "qualified") {
|
|
945
|
+
const seed = checkpoints.seed?.seed;
|
|
946
|
+
const environmentBlocked = [
|
|
947
|
+
...(seed?.baselineObservations ?? []),
|
|
948
|
+
...(seed?.postChangeObservations ?? [])
|
|
949
|
+
].some((row) => row.detail.includes("gradeExecuted=false"));
|
|
950
|
+
outcome = environmentBlocked ? "environment_blocked" : "scientific_rejection";
|
|
951
|
+
gradeExecuted = !environmentBlocked;
|
|
952
|
+
disposition = environmentBlocked ? "environment_blocked" : "rejected";
|
|
953
|
+
detail = seed?.rejectionReasons.join("; ") ?? "Reference qualification did not pass.";
|
|
954
|
+
}
|
|
955
|
+
accepted.push(getHandle("seed"));
|
|
956
|
+
if (getHandle("spec") !== undefined)
|
|
957
|
+
accepted.push(getHandle("spec"));
|
|
958
|
+
break;
|
|
959
|
+
}
|
|
960
|
+
case "author_oracle": {
|
|
961
|
+
yield* Effect.try({
|
|
962
|
+
try: () => {
|
|
963
|
+
requireHandle(request.args.specification, getHandle("spec"), "specification");
|
|
964
|
+
if (!acceptedSpec(checkpoints.spec))
|
|
965
|
+
throw failure("oracle authoring requires an accepted grounded specification review");
|
|
966
|
+
},
|
|
967
|
+
catch: (cause) => failure("specification prerequisite failed", cause)
|
|
968
|
+
});
|
|
969
|
+
const value = yield* runPipelineOracleStageV1(stageInput("oracle"));
|
|
970
|
+
yield* persist("oracle", value);
|
|
971
|
+
accepted.push(getHandle("oracle"));
|
|
972
|
+
gradeExecuted = true;
|
|
973
|
+
break;
|
|
974
|
+
}
|
|
975
|
+
case "generate_controls": {
|
|
976
|
+
yield* Effect.try({
|
|
977
|
+
try: () => {
|
|
978
|
+
requireHandle(request.args.specification, getHandle("spec"), "specification");
|
|
979
|
+
requireHandle(request.args.environment, prepared?.plan.planId, "environment");
|
|
980
|
+
if (checkpoints.challenge !== undefined)
|
|
981
|
+
throw failure("repair the challenged candidate before changing controls");
|
|
982
|
+
},
|
|
983
|
+
catch: (cause) => failure("control prerequisites failed", cause)
|
|
984
|
+
});
|
|
985
|
+
const value = yield* runPipelineControlsCapabilityV1({
|
|
986
|
+
...stageInput("controls"),
|
|
987
|
+
kind: request.args.kind,
|
|
988
|
+
maximumNewControls: 1
|
|
989
|
+
});
|
|
990
|
+
yield* persist("controls", value);
|
|
991
|
+
accepted.push(getHandle("controls"));
|
|
992
|
+
detail =
|
|
993
|
+
`Accepted cumulative controls checkpoint; valid=${String(value.valid.length)}; wrong=${String(value.wrong.length)}. ` +
|
|
994
|
+
"This returned handle supersedes all previous controls snapshots. For evaluate_oracle, pass only this latest handle in the controls list.";
|
|
995
|
+
gradeExecuted = true;
|
|
996
|
+
break;
|
|
997
|
+
}
|
|
998
|
+
case "evaluate_oracle":
|
|
999
|
+
case "repair_oracle": {
|
|
1000
|
+
yield* Effect.try({
|
|
1001
|
+
try: () => requireHandle(request.args.oracle, getHandle("oracle"), "oracle"),
|
|
1002
|
+
catch: (cause) => failure("oracle prerequisite failed", cause)
|
|
1003
|
+
});
|
|
1004
|
+
if (request.tool === "evaluate_oracle")
|
|
1005
|
+
yield* Effect.try({
|
|
1006
|
+
try: () => {
|
|
1007
|
+
for (const handle of request.args.controls)
|
|
1008
|
+
requireHandle(handle, getHandle("controls"), "controls");
|
|
1009
|
+
},
|
|
1010
|
+
catch: (cause) => failure("control prerequisite failed; use only the latest cumulative controls checkpoint returned by generate_controls, not superseded snapshots", cause)
|
|
1011
|
+
});
|
|
1012
|
+
if (request.tool === "repair_oracle" && checkpoints.challenge !== undefined) {
|
|
1013
|
+
const challenged = checkpoints.challenge;
|
|
1014
|
+
if (challenged.verdict === "passed")
|
|
1015
|
+
return yield* failure("passed held-out evidence must not be used for repair");
|
|
1016
|
+
if (challenged.fold >= input.context.foldLimit || checkpoints.controls === undefined) {
|
|
1017
|
+
outcome = "scientific_rejection";
|
|
1018
|
+
disposition = "rejected";
|
|
1019
|
+
detail = "Fresh challenge failed at the fold limit.";
|
|
1020
|
+
break;
|
|
1021
|
+
}
|
|
1022
|
+
const controls = checkpoints.controls;
|
|
1023
|
+
// Consume held-out evidence as development evidence before invalidating it.
|
|
1024
|
+
const nextState = { ...before, folds: Math.max(before.folds, challenged.fold + 1) };
|
|
1025
|
+
yield* state.set(nextState);
|
|
1026
|
+
yield* store.writeState(nextState);
|
|
1027
|
+
yield* persist("controls", {
|
|
1028
|
+
...controls,
|
|
1029
|
+
wrong: [
|
|
1030
|
+
...controls.wrong,
|
|
1031
|
+
...challenged.heldOut.filter((held) => !controls.wrong.some((wrong) => wrong.id === held.id))
|
|
1032
|
+
]
|
|
1033
|
+
});
|
|
1034
|
+
yield* store.deleteFrom("tournament");
|
|
1035
|
+
checkpoints = {
|
|
1036
|
+
seed: checkpoints.seed,
|
|
1037
|
+
spec: checkpoints.spec,
|
|
1038
|
+
oracle: checkpoints.oracle,
|
|
1039
|
+
controls: checkpoints.controls
|
|
1040
|
+
};
|
|
1041
|
+
}
|
|
1042
|
+
const value = yield* runPipelineTournamentCapabilityV1({
|
|
1043
|
+
...stageInput("tournament"),
|
|
1044
|
+
maximumIterations: request.tool === "evaluate_oracle" ? 0 : 1
|
|
1045
|
+
});
|
|
1046
|
+
yield* persist("tournament", value);
|
|
1047
|
+
accepted.push(getHandle("tournament"));
|
|
1048
|
+
gradeExecuted = true;
|
|
1049
|
+
if (!value.adequacy.admitted) {
|
|
1050
|
+
outcome = "progress";
|
|
1051
|
+
detail =
|
|
1052
|
+
value.adequacy.rejectionReasons.join("; ") ||
|
|
1053
|
+
"Tournament returned measured repair findings.";
|
|
1054
|
+
}
|
|
1055
|
+
break;
|
|
1056
|
+
}
|
|
1057
|
+
case "freeze_candidate": {
|
|
1058
|
+
yield* Effect.try({
|
|
1059
|
+
try: () => {
|
|
1060
|
+
requireHandle(request.args.revision, candidate.revision, "candidate revision");
|
|
1061
|
+
assertReady();
|
|
1062
|
+
},
|
|
1063
|
+
catch: (cause) => failure("candidate is not ready to freeze", cause)
|
|
1064
|
+
});
|
|
1065
|
+
const handle = frozenHandle();
|
|
1066
|
+
const frozen = {
|
|
1067
|
+
candidate,
|
|
1068
|
+
inputs: scientificInputs(),
|
|
1069
|
+
tournament: getHandle("tournament")
|
|
1070
|
+
};
|
|
1071
|
+
yield* store.writeCall(handle, 1, {
|
|
1072
|
+
request,
|
|
1073
|
+
response: frozen
|
|
1074
|
+
});
|
|
1075
|
+
acceptedScientificArtifacts.set(handle, {
|
|
1076
|
+
handle,
|
|
1077
|
+
contentDigest: acceptedScientificDigestV1(frozen),
|
|
1078
|
+
kind: "frozen_candidate"
|
|
1079
|
+
});
|
|
1080
|
+
accepted.push(handle);
|
|
1081
|
+
break;
|
|
1082
|
+
}
|
|
1083
|
+
case "run_held_out_challenge": {
|
|
1084
|
+
yield* Effect.try({
|
|
1085
|
+
try: () => {
|
|
1086
|
+
assertReady();
|
|
1087
|
+
requireHandle(request.args.frozenCandidate, frozenHandle(), "frozen candidate");
|
|
1088
|
+
},
|
|
1089
|
+
catch: (cause) => failure("challenge prerequisite failed", cause)
|
|
1090
|
+
});
|
|
1091
|
+
if (checkpoints.challenge === undefined) {
|
|
1092
|
+
// Changed frozen inputs invalidate partial evidence just as they
|
|
1093
|
+
// invalidate its accepted-progress handle. Never execute stale work.
|
|
1094
|
+
const pending = pendingChallenge?.frozen === frozenHandle()
|
|
1095
|
+
? pendingChallenge
|
|
1096
|
+
: undefined;
|
|
1097
|
+
const fold = (yield* state.get).folds;
|
|
1098
|
+
const value = yield* runPipelineChallengeCapabilityV1({
|
|
1099
|
+
...stageInput("challenge"),
|
|
1100
|
+
checkpoints: {
|
|
1101
|
+
...checkpoints,
|
|
1102
|
+
challenge: { fold }
|
|
1103
|
+
},
|
|
1104
|
+
materialized: pending?.agenticPendingHeldOut,
|
|
1105
|
+
...(pending?.agenticPendingOracle !== undefined
|
|
1106
|
+
? { resumeOracle: pending.agenticPendingOracle }
|
|
1107
|
+
: {}),
|
|
1108
|
+
onMaterialized: (solution) => persistChallengeProgress({
|
|
1109
|
+
frozen: frozenHandle(),
|
|
1110
|
+
agenticPendingHeldOut: solution
|
|
1111
|
+
}),
|
|
1112
|
+
onOracleProgress: (solution, oracle) => persistChallengeProgress({
|
|
1113
|
+
frozen: frozenHandle(),
|
|
1114
|
+
agenticPendingHeldOut: solution,
|
|
1115
|
+
agenticPendingOracle: oracle
|
|
1116
|
+
})
|
|
1117
|
+
});
|
|
1118
|
+
yield* persist("challenge", value);
|
|
1119
|
+
pendingChallenge = undefined;
|
|
1120
|
+
}
|
|
1121
|
+
accepted.push(getHandle("challenge"));
|
|
1122
|
+
gradeExecuted = true;
|
|
1123
|
+
if (checkpoints.challenge?.verdict !== "passed") {
|
|
1124
|
+
outcome = "scientific_rejection";
|
|
1125
|
+
detail =
|
|
1126
|
+
"Frozen held-out challenge failed. Repeating the tool cannot select another challenge; repair consumes this evidence and requires a new freeze.";
|
|
1127
|
+
}
|
|
1128
|
+
break;
|
|
1129
|
+
}
|
|
1130
|
+
case "request_admission": {
|
|
1131
|
+
yield* Effect.try({
|
|
1132
|
+
try: () => {
|
|
1133
|
+
assertReady();
|
|
1134
|
+
requireHandle(request.args.frozenCandidate, frozenHandle(), "frozen candidate");
|
|
1135
|
+
requireHandle(request.args.qualification, getHandle("challenge"), "held-out qualification");
|
|
1136
|
+
if (checkpoints.challenge?.parentDigest !== checkpointDigestV1(checkpoints.tournament))
|
|
1137
|
+
throw failure("held-out challenge does not bind current frozen tournament");
|
|
1138
|
+
},
|
|
1139
|
+
catch: (cause) => failure("admission prerequisite failed", cause)
|
|
1140
|
+
});
|
|
1141
|
+
const value = yield* runPipelineAdmitStageV1(stageInput("admit"));
|
|
1142
|
+
yield* persist("admit", value);
|
|
1143
|
+
accepted.push(getHandle("admit"));
|
|
1144
|
+
gradeExecuted = true;
|
|
1145
|
+
if (value.verdict === "rejected") {
|
|
1146
|
+
outcome = "scientific_rejection";
|
|
1147
|
+
disposition = "rejected";
|
|
1148
|
+
detail = value.reasons.join("; ");
|
|
1149
|
+
}
|
|
1150
|
+
break;
|
|
1151
|
+
}
|
|
1152
|
+
case "get_progress":
|
|
1153
|
+
accepted.push(...PIPELINE_STAGES.flatMap((stage) => getHandle(stage) === undefined ? [] : [getHandle(stage)]));
|
|
1154
|
+
break;
|
|
1155
|
+
case "reject_candidate":
|
|
1156
|
+
outcome = "scientific_rejection";
|
|
1157
|
+
disposition = "rejected";
|
|
1158
|
+
detail = request.args.reason;
|
|
1159
|
+
break;
|
|
1160
|
+
case "finish_campaign":
|
|
1161
|
+
detail = request.args.reason;
|
|
1162
|
+
break;
|
|
1163
|
+
default:
|
|
1164
|
+
return yield* failure(`${request.tool} is a repository/session capability, not a scientific candidate operation`);
|
|
1165
|
+
}
|
|
1166
|
+
});
|
|
1167
|
+
const result = yield* execute.pipe(Effect.provide(BudgetedLanguageModelLive({ store, state, budget, startedAtMs: invocationStartedAt, inner })), Effect.timeoutOption(maximumWallMs), Effect.result);
|
|
1168
|
+
if (result._tag === "Failure") {
|
|
1169
|
+
const environmentBlocked = result.failure instanceof RepositoryFoundryError &&
|
|
1170
|
+
result.failure.classification === "environment_blocked";
|
|
1171
|
+
outcome = environmentBlocked
|
|
1172
|
+
? "environment_blocked"
|
|
1173
|
+
: budgetExhaustedErrorOfV1(result.failure) === undefined
|
|
1174
|
+
? "progress"
|
|
1175
|
+
: "budget_exhausted";
|
|
1176
|
+
if (environmentBlocked)
|
|
1177
|
+
gradeExecuted = false;
|
|
1178
|
+
detail =
|
|
1179
|
+
result.failure instanceof RepositoryFoundryError
|
|
1180
|
+
? result.failure.detail
|
|
1181
|
+
: String(result.failure);
|
|
1182
|
+
}
|
|
1183
|
+
else if (Option.isNone(result.success)) {
|
|
1184
|
+
yieldedAtSliceBoundary = true;
|
|
1185
|
+
outcome = "progress";
|
|
1186
|
+
const summary = request.tool === "qualify_reference" && qualificationProgress !== undefined
|
|
1187
|
+
? repositorySeedQualificationProgressSummaryV1(qualificationProgress)
|
|
1188
|
+
: undefined;
|
|
1189
|
+
detail =
|
|
1190
|
+
summary === undefined
|
|
1191
|
+
? "Capability wall-time bound reached; retained checkpoints and command diagnostics are available for continuation."
|
|
1192
|
+
: `Capability wall-time bound reached after checkpointing reference qualification progress: completedBatches=${String(summary.completedBatches)}/${String(summary.batchesRequired)}; activeUnits=${String(summary.completedUnits)}/${String(summary.unitsPerBatch)}. Continuation resumes at the next incomplete repetition.`;
|
|
1193
|
+
}
|
|
1194
|
+
else {
|
|
1195
|
+
completedLogicalAttempt = true;
|
|
1196
|
+
}
|
|
1197
|
+
const retainedQualificationArtifact = qualificationProgressArtifact();
|
|
1198
|
+
if (request.tool === "qualify_reference" &&
|
|
1199
|
+
retainedQualificationArtifact !== undefined &&
|
|
1200
|
+
!beforeArtifactHandles.has(retainedQualificationArtifact.handle) &&
|
|
1201
|
+
!accepted.includes(retainedQualificationArtifact.handle)) {
|
|
1202
|
+
accepted.push(retainedQualificationArtifact.handle);
|
|
1203
|
+
}
|
|
1204
|
+
const retainedChallengeArtifact = challengeProgressArtifact();
|
|
1205
|
+
if (request.tool === "run_held_out_challenge" &&
|
|
1206
|
+
retainedChallengeArtifact !== undefined &&
|
|
1207
|
+
!beforeArtifactHandles.has(retainedChallengeArtifact.handle) &&
|
|
1208
|
+
!accepted.includes(retainedChallengeArtifact.handle)) {
|
|
1209
|
+
accepted.push(retainedChallengeArtifact.handle);
|
|
1210
|
+
}
|
|
1211
|
+
const after = yield* state.get;
|
|
1212
|
+
const invocationCompletedAt = yield* Clock.currentTimeMillis;
|
|
1213
|
+
const elapsedMs = before.elapsedMs + Math.max(0, invocationCompletedAt - invocationStartedAt);
|
|
1214
|
+
const campaignWallTimeRemainingAfter = Math.max(0, parsedDeadline === undefined
|
|
1215
|
+
? input.context.budget.maxWallMs - elapsedMs
|
|
1216
|
+
: parsedDeadline - invocationCompletedAt);
|
|
1217
|
+
if (outcome === "budget_exhausted" &&
|
|
1218
|
+
after.modelCalls < input.context.budget.maxModelCalls &&
|
|
1219
|
+
campaignWallTimeRemainingAfter > 0) {
|
|
1220
|
+
yieldedAtSliceBoundary = true;
|
|
1221
|
+
outcome = "progress";
|
|
1222
|
+
detail = `Capability slice completed without consuming the campaign budget: ${detail}`;
|
|
1223
|
+
}
|
|
1224
|
+
if (outcome === "environment_blocked")
|
|
1225
|
+
disposition = "environment_blocked";
|
|
1226
|
+
if (outcome === "budget_exhausted")
|
|
1227
|
+
disposition = "exhausted";
|
|
1228
|
+
const scientificStage = scientificStageForCapability(request.tool);
|
|
1229
|
+
if (scientificStage !== null && completedLogicalAttempt && !yieldedAtSliceBoundary) {
|
|
1230
|
+
const completedWork = [...groundedWork.values()].filter((entry) => entry.stage === scientificStage);
|
|
1231
|
+
if (completedWork.length > 0) {
|
|
1232
|
+
yield* workPersistence.withPermit(Effect.gen(function* () {
|
|
1233
|
+
for (const entry of completedWork) {
|
|
1234
|
+
groundedWork.delete(entry.name);
|
|
1235
|
+
yield* store.deleteWorkCheckpoint(entry.name);
|
|
1236
|
+
}
|
|
1237
|
+
yield* saveGroundedIndex();
|
|
1238
|
+
})).pipe(Effect.uninterruptible);
|
|
1239
|
+
}
|
|
1240
|
+
}
|
|
1241
|
+
else if (scientificStage !== null && yieldedAtSliceBoundary) {
|
|
1242
|
+
const provisionalFingerprint = scientificFingerprint();
|
|
1243
|
+
if (!sameProgressFingerprint(beforeFingerprint, provisionalFingerprint)) {
|
|
1244
|
+
const name = `pipeline-execution-capability-slice-${request.tool}`;
|
|
1245
|
+
const value = {
|
|
1246
|
+
version: 1,
|
|
1247
|
+
capability: request.tool,
|
|
1248
|
+
candidateRevision: candidate.revision,
|
|
1249
|
+
fingerprint: provisionalFingerprint
|
|
1250
|
+
};
|
|
1251
|
+
yield* workPersistence.withPermit(Effect.gen(function* () {
|
|
1252
|
+
yield* store.writeWorkCheckpoint(name, value);
|
|
1253
|
+
groundedWork.set(name, {
|
|
1254
|
+
name,
|
|
1255
|
+
stage: scientificStage,
|
|
1256
|
+
contentDigest: acceptedScientificDigestV1(value)
|
|
1257
|
+
});
|
|
1258
|
+
yield* saveGroundedIndex();
|
|
1259
|
+
})).pipe(Effect.uninterruptible);
|
|
1260
|
+
}
|
|
1261
|
+
}
|
|
1262
|
+
// One immutable manifest represents all durable work, irrespective of the
|
|
1263
|
+
// number of fixture units. Finalize the slice marker first so the returned
|
|
1264
|
+
// handle and receipt delta always name the same authoritative manifest.
|
|
1265
|
+
const groundedContinuation = groundedArtifact();
|
|
1266
|
+
if (groundedContinuation !== undefined &&
|
|
1267
|
+
!beforeArtifactHandles.has(groundedContinuation.handle)) {
|
|
1268
|
+
for (const [name, { contentDigest }] of groundedWork) {
|
|
1269
|
+
const work = yield* store.readWorkCheckpoint(name, Schema.Unknown);
|
|
1270
|
+
if (Option.isNone(work) || acceptedScientificDigestV1(work.value) !== contentDigest)
|
|
1271
|
+
return yield* failure("grounded continuation does not bind its durable work checkpoint");
|
|
1272
|
+
}
|
|
1273
|
+
yield* store.writeCall(groundedContinuation.handle, 1, {
|
|
1274
|
+
request: { candidateRevision: candidate.revision },
|
|
1275
|
+
response: groundedContinuationValue()
|
|
1276
|
+
});
|
|
1277
|
+
accepted.push(groundedContinuation.handle);
|
|
1278
|
+
}
|
|
1279
|
+
const afterArtifacts = currentScientificArtifacts();
|
|
1280
|
+
const afterArtifactHandles = new Set(afterArtifacts.map((artifact) => artifact.handle));
|
|
1281
|
+
const afterGates = completedScientificGates();
|
|
1282
|
+
const resultingFingerprint = scientificFingerprint();
|
|
1283
|
+
const sameFingerprintCount = sameProgressFingerprint(beforeFingerprint, resultingFingerprint)
|
|
1284
|
+
? (latestProgressReceipt !== undefined &&
|
|
1285
|
+
latestProgressReceipt.capability === request.tool &&
|
|
1286
|
+
sameProgressFingerprint(latestProgressReceipt.resultingFingerprint, beforeFingerprint)
|
|
1287
|
+
? latestProgressReceipt.sameFingerprintCount
|
|
1288
|
+
: 0) + 1
|
|
1289
|
+
: 0;
|
|
1290
|
+
const attemptCost = evalCapabilityScientificAttemptCostV1({
|
|
1291
|
+
previousFingerprint: beforeFingerprint,
|
|
1292
|
+
resultingFingerprint,
|
|
1293
|
+
inputRevision: candidate.revision,
|
|
1294
|
+
outputRevision: candidate.revision,
|
|
1295
|
+
gatesCompleted: afterGates.filter((gate) => !beforeGates.includes(gate)),
|
|
1296
|
+
gatesInvalidated: beforeGates.filter((gate) => !afterGates.includes(gate)),
|
|
1297
|
+
sameFingerprintCount
|
|
1298
|
+
});
|
|
1299
|
+
if (scientificPhase !== null &&
|
|
1300
|
+
sameFingerprintCount >= 3 &&
|
|
1301
|
+
(outcome === "completed" || outcome === "progress")) {
|
|
1302
|
+
outcome = "scientific_rejection";
|
|
1303
|
+
disposition = "rejected";
|
|
1304
|
+
detail =
|
|
1305
|
+
"Scientific no-progress circuit opened after three unchanged accepted-state results. No further repeat of this capability is admissible for the candidate.";
|
|
1306
|
+
}
|
|
1307
|
+
const terminal = checkpoints.admit?.verdict === "admitted"
|
|
1308
|
+
? "admitted"
|
|
1309
|
+
: disposition === "rejected"
|
|
1310
|
+
? "rejected"
|
|
1311
|
+
: undefined;
|
|
1312
|
+
const settledState = {
|
|
1313
|
+
...after,
|
|
1314
|
+
elapsedMs,
|
|
1315
|
+
...(terminal === undefined ? {} : { status: terminal, stage: "done" })
|
|
1316
|
+
};
|
|
1317
|
+
yield* state.set(settledState);
|
|
1318
|
+
yield* store.writeState(settledState);
|
|
1319
|
+
const receipt = scientificPhase === null
|
|
1320
|
+
? undefined
|
|
1321
|
+
: yield* Effect.try({
|
|
1322
|
+
try: () => {
|
|
1323
|
+
const recordedAt = new Date(invocationStartedAt + Math.max(0, elapsedMs - before.elapsedMs)).toISOString();
|
|
1324
|
+
const remainingCalls = Math.max(0, input.context.budget.maxModelCalls - after.modelCalls);
|
|
1325
|
+
const remainingWallTimeMs = campaignWallTimeRemainingAfter;
|
|
1326
|
+
const value = {
|
|
1327
|
+
version: 1,
|
|
1328
|
+
operationId: invocation.operationId,
|
|
1329
|
+
capability: request.tool,
|
|
1330
|
+
scientificRole: scientificRoleForCapability(request.tool),
|
|
1331
|
+
phase: scientificPhase,
|
|
1332
|
+
inputRevision: candidate.revision,
|
|
1333
|
+
outputRevision: candidate.revision,
|
|
1334
|
+
acceptedEvidenceDigest: resultingFingerprint.acceptedEvidenceDigest,
|
|
1335
|
+
previousFingerprint: beforeFingerprint,
|
|
1336
|
+
resultingFingerprint,
|
|
1337
|
+
acceptedEvidenceAdded: afterArtifacts
|
|
1338
|
+
.filter((artifact) => !beforeArtifactHandles.has(artifact.handle))
|
|
1339
|
+
.map((artifact) => artifact.handle),
|
|
1340
|
+
acceptedEvidenceRemoved: beforeArtifacts
|
|
1341
|
+
.filter((artifact) => !afterArtifactHandles.has(artifact.handle))
|
|
1342
|
+
.map((artifact) => artifact.handle),
|
|
1343
|
+
gatesCompleted: afterGates.filter((gate) => !beforeGates.includes(gate)),
|
|
1344
|
+
gatesInvalidated: beforeGates.filter((gate) => !afterGates.includes(gate)),
|
|
1345
|
+
sameFingerprintCount,
|
|
1346
|
+
consumption: {
|
|
1347
|
+
wallTimeMs: Math.max(0, elapsedMs - before.elapsedMs),
|
|
1348
|
+
receivedCalls: Math.max(0, after.modelCalls - before.modelCalls),
|
|
1349
|
+
spendKnownUsd: "0",
|
|
1350
|
+
spendUnknown: after.modelCalls > before.modelCalls
|
|
1351
|
+
},
|
|
1352
|
+
remaining: {
|
|
1353
|
+
phaseCalls: Math.max(0, remainingCalls - downstreamCalls),
|
|
1354
|
+
phaseSpendUsd: null,
|
|
1355
|
+
phaseWallTimeMs: Math.max(0, remainingWallTimeMs -
|
|
1356
|
+
downstreamWallTimeMs -
|
|
1357
|
+
finalizationWallTimeMs),
|
|
1358
|
+
phaseAttempts: Math.max(0, (phasePolicy?.maximumAttempts ?? 0) - phaseAttemptsConsumed - attemptCost),
|
|
1359
|
+
downstreamCalls: Math.min(downstreamCalls, remainingCalls),
|
|
1360
|
+
downstreamSpendUsd: null,
|
|
1361
|
+
downstreamWallTimeMs: Math.min(downstreamWallTimeMs, Math.max(0, remainingWallTimeMs - finalizationWallTimeMs)),
|
|
1362
|
+
downstreamAttempts: phasePolicy?.downstream.attempts ?? 0
|
|
1363
|
+
},
|
|
1364
|
+
recordedAt
|
|
1365
|
+
};
|
|
1366
|
+
return decodeAgenticScientificProgressReceiptV1({
|
|
1367
|
+
...value,
|
|
1368
|
+
receiptId: `progress-${agenticDigestV1(value)}`
|
|
1369
|
+
});
|
|
1370
|
+
},
|
|
1371
|
+
catch: (cause) => failure("could not encode scientific progress receipt", cause)
|
|
1372
|
+
});
|
|
1373
|
+
if (receipt !== undefined) {
|
|
1374
|
+
// This receipt remains provisional until Cloud accepts the enclosing
|
|
1375
|
+
// capability completion. The next invocation supplies the host-owned
|
|
1376
|
+
// accepted chain, so an interrupted or rejected completion can never
|
|
1377
|
+
// advance local scientific sequencing or enter a recovery checkpoint.
|
|
1378
|
+
if (invocation.authoritativeProgressReceipts === undefined) {
|
|
1379
|
+
// Preserve the standalone package contract for callers that do not
|
|
1380
|
+
// have a remote authority. Production passes an explicit array on
|
|
1381
|
+
// every invocation, including the empty authoritative chain.
|
|
1382
|
+
initialProgressReceipts.push(receipt);
|
|
1383
|
+
yield* store.writeCall(receipt.receiptId, 1, {
|
|
1384
|
+
request: { operationId: invocation.operationId },
|
|
1385
|
+
response: receipt
|
|
1386
|
+
});
|
|
1387
|
+
if (input.onProgressReceipt !== undefined) {
|
|
1388
|
+
yield* input.onProgressReceipt(receipt);
|
|
1389
|
+
}
|
|
1390
|
+
}
|
|
1391
|
+
}
|
|
1392
|
+
yield* currentScorecard(disposition, [detail.slice(0, 4096)]);
|
|
1393
|
+
const diagnostics = `diagnostics-${agenticDigestV1({ operation: invocation.operationId, detail, accepted })}`;
|
|
1394
|
+
yield* store.writeCall(diagnostics, 1, { request, response: { outcome, detail, accepted } });
|
|
1395
|
+
const allowedNext = terminal !== undefined
|
|
1396
|
+
? ["get_progress", "list_candidates", "read_diagnostics", "finish_campaign"]
|
|
1397
|
+
: outcome === "budget_exhausted"
|
|
1398
|
+
? ["get_progress", "finish_campaign"]
|
|
1399
|
+
: outcome === "environment_blocked" && request.tool === "qualify_reference"
|
|
1400
|
+
? ["get_progress", "read_diagnostics", "plan_environment", "reject_candidate"]
|
|
1401
|
+
: outcome === "scientific_rejection" && sameFingerprintCount >= 3
|
|
1402
|
+
? ["get_progress", "reject_candidate", "finish_campaign"]
|
|
1403
|
+
: CANDIDATE_ALLOWED_CAPABILITIES.filter((capability) => (outcome !== "environment_blocked" || capability !== request.tool) &&
|
|
1404
|
+
(sameFingerprintCount < 2 || capability !== request.tool) &&
|
|
1405
|
+
hasCapabilityPrerequisites(capability));
|
|
1406
|
+
return yield* Effect.try({
|
|
1407
|
+
try: () => decodeAgenticCapabilityResultV1({
|
|
1408
|
+
outcome,
|
|
1409
|
+
operation: invocation.operationId,
|
|
1410
|
+
inputDigest: agenticDigestV1(request),
|
|
1411
|
+
candidateRevision: candidate.revision,
|
|
1412
|
+
accepted,
|
|
1413
|
+
evidenceSummary: detail.slice(0, 7000),
|
|
1414
|
+
...(gradeExecuted === undefined ? {} : { gradeExecuted }),
|
|
1415
|
+
diagnostics,
|
|
1416
|
+
...(receipt === undefined ? {} : { progressReceipt: receipt }),
|
|
1417
|
+
usage: {
|
|
1418
|
+
receivedCalls: after.modelCalls - before.modelCalls,
|
|
1419
|
+
spendKnownUsd: "0",
|
|
1420
|
+
spendUnknown: after.modelCalls > before.modelCalls
|
|
1421
|
+
},
|
|
1422
|
+
allowedNext
|
|
1423
|
+
}),
|
|
1424
|
+
catch: (cause) => failure("could not encode bounded capability result", cause)
|
|
1425
|
+
});
|
|
1426
|
+
});
|
|
1427
|
+
return {
|
|
1428
|
+
invoke,
|
|
1429
|
+
scorecard: currentScorecard(),
|
|
1430
|
+
environment: () => prepared,
|
|
1431
|
+
handles: () => Object.fromEntries(PIPELINE_STAGES.map((stage) => [stage, getHandle(stage)]))
|
|
1432
|
+
};
|
|
1433
|
+
});
|