@velum-labs/routekit-eval-setup 1.4.0 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/authoring-responses-request.d.ts +5 -0
- package/dist/adapters/authoring-responses-request.js +38 -0
- package/dist/adapters/evaluation-evidence-freshness.d.ts +7 -0
- package/dist/adapters/evaluation-evidence-freshness.js +76 -0
- package/dist/adapters/git-task-history.d.ts +67 -0
- package/dist/adapters/git-task-history.js +171 -0
- package/dist/adapters/integrated-repository-history.d.ts +21 -0
- package/dist/adapters/integrated-repository-history.js +175 -0
- package/dist/adapters/repository-command-diagnostic.d.ts +8 -0
- package/dist/adapters/repository-command-diagnostic.js +46 -0
- package/dist/adapters/repository-command-evidence.d.ts +13 -0
- package/dist/adapters/repository-command-evidence.js +102 -0
- package/dist/adapters/repository-command-runner.d.ts +238 -0
- package/dist/adapters/repository-command-runner.js +1483 -0
- package/dist/adapters/repository-import-context.d.ts +47 -0
- package/dist/adapters/repository-import-context.js +469 -0
- package/dist/adapters/repository-node-test-reporter.d.ts +3 -0
- package/dist/adapters/repository-node-test-reporter.js +27 -0
- package/dist/adapters/repository-review-evidence.d.ts +39 -0
- package/dist/adapters/repository-review-evidence.js +632 -0
- package/dist/adapters/repository-seed-selection.d.ts +7 -0
- package/dist/adapters/repository-seed-selection.js +79 -0
- package/dist/adapters/repository-solution-edits.d.ts +49 -0
- package/dist/adapters/repository-solution-edits.js +136 -0
- package/dist/adapters/repository-vitest-phase-adapter.d.ts +8 -0
- package/dist/adapters/repository-vitest-phase-adapter.js +310 -0
- package/dist/adapters/repository-vitest-reporter.d.ts +24 -0
- package/dist/adapters/repository-vitest-reporter.js +314 -0
- package/dist/adapters/strict-authoring-schema.d.ts +5 -0
- package/dist/adapters/strict-authoring-schema.js +158 -0
- package/dist/adapters/test-discovery.d.ts +30 -0
- package/dist/adapters/test-discovery.js +124 -0
- package/dist/adapters/typescript-repository-index.d.ts +51 -0
- package/dist/adapters/typescript-repository-index.js +226 -0
- package/dist/agentic-capabilities-protocol.d.ts +1373 -0
- package/dist/agentic-capabilities-protocol.js +786 -0
- package/dist/case-checkpoint-store.d.ts +29 -0
- package/dist/case-checkpoint-store.js +133 -0
- package/dist/case-pipeline-protocol-v2.d.ts +184 -0
- package/dist/case-pipeline-protocol-v2.js +193 -0
- package/dist/case-pipeline-protocol.d.ts +2626 -0
- package/dist/case-pipeline-protocol.js +371 -0
- package/dist/effect-api.d.ts +74 -12
- package/dist/effect-api.js +56 -7
- package/dist/errors.d.ts +31 -0
- package/dist/errors.js +10 -0
- package/dist/eval-capability-execution-envelope.d.ts +64 -0
- package/dist/eval-capability-execution-envelope.js +98 -0
- package/dist/eval-capability-policy.d.ts +90 -0
- package/dist/eval-capability-policy.js +107 -0
- package/dist/eval-event-log.d.ts +51 -0
- package/dist/eval-event-log.js +69 -10
- package/dist/evaluation-authoring-policy.d.ts +18 -0
- package/dist/evaluation-authoring-policy.js +19 -0
- package/dist/evaluation-authoring-validation.d.ts +22 -0
- package/dist/evaluation-authoring-validation.js +72 -0
- package/dist/evaluation-evidence.d.ts +20 -0
- package/dist/evaluation-evidence.js +319 -0
- package/dist/evaluation-grader-calibration-protocol.d.ts +108 -0
- package/dist/evaluation-grader-calibration-protocol.js +80 -0
- package/dist/evaluation-grader-calibration.d.ts +18 -0
- package/dist/evaluation-grader-calibration.js +334 -0
- package/dist/evaluation-grading-policy.d.ts +24 -0
- package/dist/evaluation-grading-policy.js +54 -0
- package/dist/evaluation-proposal-policy.d.ts +4 -0
- package/dist/evaluation-proposal-policy.js +91 -0
- package/dist/evaluation-source-retrieval.d.ts +68 -0
- package/dist/evaluation-source-retrieval.js +513 -0
- package/dist/evaluation-structure-policy.d.ts +29 -0
- package/dist/evaluation-structure-policy.js +138 -0
- package/dist/index.d.ts +124 -19
- package/dist/index.js +69 -12
- package/dist/inspection.js +2 -3
- package/dist/project-artifacts.d.ts +2 -2
- package/dist/project-artifacts.js +26 -121
- package/dist/project-authoring.d.ts +66 -5
- package/dist/project-authoring.js +783 -109
- package/dist/project-contracts.d.ts +292 -82
- package/dist/project-contracts.js +65 -7
- package/dist/project-store.js +2 -1
- package/dist/project-workflow.d.ts +3 -3
- package/dist/project-workflow.js +116 -18
- package/dist/repository-adversary-protocol.d.ts +64 -0
- package/dist/repository-adversary-protocol.js +105 -0
- package/dist/repository-behavior-protocol.d.ts +188 -0
- package/dist/repository-behavior-protocol.js +202 -0
- package/dist/repository-benchmark-protocol.d.ts +487 -0
- package/dist/repository-benchmark-protocol.js +96 -0
- package/dist/repository-execution-protocol.d.ts +150 -0
- package/dist/repository-execution-protocol.js +38 -0
- package/dist/repository-fixture-instructions.d.ts +3 -0
- package/dist/repository-fixture-instructions.js +91 -0
- package/dist/repository-fixture-protocol.d.ts +79 -0
- package/dist/repository-fixture-protocol.js +79 -0
- package/dist/repository-foundry-plan-protocol.d.ts +118 -0
- package/dist/repository-foundry-plan-protocol.js +296 -0
- package/dist/repository-foundry-progress-protocol.d.ts +52 -0
- package/dist/repository-foundry-progress-protocol.js +52 -0
- package/dist/repository-improvement-protocol.d.ts +100 -0
- package/dist/repository-improvement-protocol.js +106 -0
- package/dist/repository-language-model-protocol.d.ts +43 -0
- package/dist/repository-language-model-protocol.js +146 -0
- package/dist/repository-oracle-coverage-protocol.d.ts +18 -0
- package/dist/repository-oracle-coverage-protocol.js +39 -0
- package/dist/repository-oracle-execution-binding.d.ts +27 -0
- package/dist/repository-oracle-execution-binding.js +59 -0
- package/dist/repository-oracle-protocol.d.ts +230 -0
- package/dist/repository-oracle-protocol.js +156 -0
- package/dist/repository-oracle-scope-policy.d.ts +22 -0
- package/dist/repository-oracle-scope-policy.js +92 -0
- package/dist/repository-quality-policy.d.ts +15 -0
- package/dist/repository-quality-policy.js +357 -0
- package/dist/repository-routing-benchmark-protocol.d.ts +176 -0
- package/dist/repository-routing-benchmark-protocol.js +103 -0
- package/dist/repository-routing-model-protocol.d.ts +36 -0
- package/dist/repository-routing-model-protocol.js +89 -0
- package/dist/repository-routing-plan-protocol.d.ts +112 -0
- package/dist/repository-routing-plan-protocol.js +58 -0
- package/dist/repository-routing-quality-policy.d.ts +9 -0
- package/dist/repository-routing-quality-policy.js +191 -0
- package/dist/repository-seed-qualification-progress-protocol.d.ts +205 -0
- package/dist/repository-seed-qualification-progress-protocol.js +28 -0
- package/dist/repository-semantic-calibration-protocol.d.ts +768 -0
- package/dist/repository-semantic-calibration-protocol.js +276 -0
- package/dist/repository-semantic-calibration.d.ts +163 -0
- package/dist/repository-semantic-calibration.js +581 -0
- package/dist/repository-specification-contract-facts-protocol.d.ts +224 -0
- package/dist/repository-specification-contract-facts-protocol.js +276 -0
- package/dist/repository-specification-critique-protocol.d.ts +189 -0
- package/dist/repository-specification-critique-protocol.js +103 -0
- package/dist/repository-task-family-protocol.d.ts +24 -0
- package/dist/repository-task-family-protocol.js +37 -0
- package/dist/repository-task-seed-protocol.d.ts +384 -0
- package/dist/repository-task-seed-protocol.js +236 -0
- package/dist/repository-trajectory-protocol.d.ts +20 -0
- package/dist/repository-trajectory-protocol.js +42 -0
- package/dist/service.js +1 -1
- package/dist/services/adversary/service.d.ts +64 -0
- package/dist/services/adversary/service.js +330 -0
- package/dist/services/benchmark-compiler/service.d.ts +450 -0
- package/dist/services/benchmark-compiler/service.js +9 -0
- package/dist/services/budgeted-model/service.d.ts +118 -0
- package/dist/services/budgeted-model/service.js +460 -0
- package/dist/services/case-authoring/service.d.ts +163 -0
- package/dist/services/case-authoring/service.js +1456 -0
- package/dist/services/case-finalization/service.d.ts +283 -0
- package/dist/services/case-finalization/service.js +370 -0
- package/dist/services/case-generation/service.d.ts +619 -0
- package/dist/services/case-generation/service.js +2628 -0
- package/dist/services/case-pipeline/service.d.ts +31 -0
- package/dist/services/case-pipeline/service.js +485 -0
- package/dist/services/case-pipeline-v2/service.d.ts +70 -0
- package/dist/services/case-pipeline-v2/service.js +477 -0
- package/dist/services/command-observability/service.d.ts +13 -0
- package/dist/services/command-observability/service.js +3 -0
- package/dist/services/dimension-labeling/service.d.ts +77 -0
- package/dist/services/dimension-labeling/service.js +188 -0
- package/dist/services/eval-candidate/service.d.ts +208 -0
- package/dist/services/eval-candidate/service.js +64 -0
- package/dist/services/eval-capabilities/service.d.ts +183 -0
- package/dist/services/eval-capabilities/service.js +1433 -0
- package/dist/services/eval-environment/service.d.ts +173 -0
- package/dist/services/eval-environment/service.js +127 -0
- package/dist/services/evidence-reconstruction/service.d.ts +36 -0
- package/dist/services/evidence-reconstruction/service.js +145 -0
- package/dist/services/fixture-builder/service.d.ts +62 -0
- package/dist/services/fixture-builder/service.js +36 -0
- package/dist/services/fixture-validation/service.d.ts +75 -0
- package/dist/services/fixture-validation/service.js +295 -0
- package/dist/services/foundry/service.d.ts +831 -0
- package/dist/services/foundry/service.js +442 -0
- package/dist/services/foundry-progress/service.d.ts +62 -0
- package/dist/services/foundry-progress/service.js +149 -0
- package/dist/services/foundry-v2/service.d.ts +54 -0
- package/dist/services/foundry-v2/service.js +28 -0
- package/dist/services/grounded-authoring/service.d.ts +126 -0
- package/dist/services/grounded-authoring/service.js +822 -0
- package/dist/services/historical-case/service.d.ts +722 -0
- package/dist/services/historical-case/service.js +177 -0
- package/dist/services/improvement-loop/service.d.ts +59 -0
- package/dist/services/improvement-loop/service.js +176 -0
- package/dist/services/language-model/service.d.ts +52 -0
- package/dist/services/language-model/service.js +194 -0
- package/dist/services/oracle-builder/service.d.ts +146 -0
- package/dist/services/oracle-builder/service.js +513 -0
- package/dist/services/oracle-coverage/service.d.ts +28 -0
- package/dist/services/oracle-coverage/service.js +50 -0
- package/dist/services/oracle-coverage-witness/service.d.ts +130 -0
- package/dist/services/oracle-coverage-witness/service.js +538 -0
- package/dist/services/pipeline-challenge/service.d.ts +551 -0
- package/dist/services/pipeline-challenge/service.js +427 -0
- package/dist/services/pipeline-controls/service.d.ts +130 -0
- package/dist/services/pipeline-controls/service.js +483 -0
- package/dist/services/pipeline-oracle/service.d.ts +8 -0
- package/dist/services/pipeline-oracle/service.js +256 -0
- package/dist/services/pipeline-seed/service.d.ts +298 -0
- package/dist/services/pipeline-seed/service.js +428 -0
- package/dist/services/pipeline-spec/service.d.ts +103 -0
- package/dist/services/pipeline-spec/service.js +619 -0
- package/dist/services/pipeline-tournament/service.d.ts +258 -0
- package/dist/services/pipeline-tournament/service.js +476 -0
- package/dist/services/quality-gate/service.d.ts +233 -0
- package/dist/services/quality-gate/service.js +136 -0
- package/dist/services/repository-bundle/service.d.ts +33 -0
- package/dist/services/repository-bundle/service.js +114 -0
- package/dist/services/repository-model/service.d.ts +105 -0
- package/dist/services/repository-model/service.js +250 -0
- package/dist/services/repository-public-artifact/service.d.ts +133 -0
- package/dist/services/repository-public-artifact/service.js +330 -0
- package/dist/services/routing-benchmark/service.d.ts +362 -0
- package/dist/services/routing-benchmark/service.js +96 -0
- package/dist/services/specification-critic/service.d.ts +92 -0
- package/dist/services/specification-critic/service.js +172 -0
- package/dist/services/task-family/service.d.ts +40 -0
- package/dist/services/task-family/service.js +55 -0
- package/dist/services/task-seed/service.d.ts +906 -0
- package/dist/services/task-seed/service.js +1406 -0
- package/dist/services/task-specification/service.d.ts +27 -0
- package/dist/services/task-specification/service.js +40 -0
- package/dist/services/trajectory-policy/service.d.ts +110 -0
- package/dist/services/trajectory-policy/service.js +216 -0
- package/dist/test/agentic-capabilities-protocol.test.d.ts +1 -0
- package/dist/test/agentic-capabilities-protocol.test.js +570 -0
- package/dist/test/agentic-capabilities.test.d.ts +1 -0
- package/dist/test/agentic-capabilities.test.js +1461 -0
- package/dist/test/agentic-environment.test.d.ts +1 -0
- package/dist/test/agentic-environment.test.js +213 -0
- package/dist/test/case-pipeline-foundation.test.d.ts +1 -0
- package/dist/test/case-pipeline-foundation.test.js +535 -0
- package/dist/test/case-pipeline-protocol-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-protocol-v2.test.js +124 -0
- package/dist/test/case-pipeline-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-v2.test.js +286 -0
- package/dist/test/case-pipeline.test.d.ts +1 -0
- package/dist/test/case-pipeline.test.js +851 -0
- package/dist/test/eval-capability-policy.test.d.ts +1 -0
- package/dist/test/eval-capability-policy.test.js +50 -0
- package/dist/test/eval-event-log.test.js +34 -1
- package/dist/test/evaluation-evidence-freshness.test.d.ts +1 -0
- package/dist/test/evaluation-evidence-freshness.test.js +44 -0
- package/dist/test/evaluation-evidence.test.d.ts +1 -0
- package/dist/test/evaluation-evidence.test.js +230 -0
- package/dist/test/evaluation-grader-calibration.test.d.ts +1 -0
- package/dist/test/evaluation-grader-calibration.test.js +373 -0
- package/dist/test/evaluation-proposal-digest.test.d.ts +1 -0
- package/dist/test/evaluation-proposal-digest.test.js +187 -0
- package/dist/test/evaluation-source-retrieval.test.d.ts +1 -0
- package/dist/test/evaluation-source-retrieval.test.js +237 -0
- package/dist/test/evaluation-structure-policy.test.d.ts +1 -0
- package/dist/test/evaluation-structure-policy.test.js +196 -0
- package/dist/test/fixtures/repository-resource-panel.d.ts +39 -0
- package/dist/test/fixtures/repository-resource-panel.js +111 -0
- package/dist/test/fixtures/vitest-boundary-panel.d.ts +84 -0
- package/dist/test/fixtures/vitest-boundary-panel.js +120 -0
- package/dist/test/fixtures/vitest-phase-panel.d.ts +135 -0
- package/dist/test/fixtures/vitest-phase-panel.js +213 -0
- package/dist/test/fixtures/vitest-reporter-results.d.ts +76 -0
- package/dist/test/fixtures/vitest-reporter-results.js +94 -0
- package/dist/test/grounded-authoring.test.d.ts +1 -0
- package/dist/test/grounded-authoring.test.js +565 -0
- package/dist/test/integrated-repository-history.test.d.ts +1 -0
- package/dist/test/integrated-repository-history.test.js +227 -0
- package/dist/test/project-authoring.test.js +593 -43
- package/dist/test/project-workflow.test.js +395 -32
- package/dist/test/repository-authoring-artifacts.test.d.ts +1 -0
- package/dist/test/repository-authoring-artifacts.test.js +185 -0
- package/dist/test/repository-bundle.test.d.ts +1 -0
- package/dist/test/repository-bundle.test.js +52 -0
- package/dist/test/repository-case-generation.test.d.ts +1 -0
- package/dist/test/repository-case-generation.test.js +3465 -0
- package/dist/test/repository-command-diagnostic.test.d.ts +1 -0
- package/dist/test/repository-command-diagnostic.test.js +55 -0
- package/dist/test/repository-command-signals.test.d.ts +1 -0
- package/dist/test/repository-command-signals.test.js +124 -0
- package/dist/test/repository-fixture-scope-coverage.test.d.ts +1 -0
- package/dist/test/repository-fixture-scope-coverage.test.js +127 -0
- package/dist/test/repository-fixture-validation.test.d.ts +1 -0
- package/dist/test/repository-fixture-validation.test.js +362 -0
- package/dist/test/repository-foundry-progress.test.d.ts +1 -0
- package/dist/test/repository-foundry-progress.test.js +110 -0
- package/dist/test/repository-foundry-quality.test.d.ts +1 -0
- package/dist/test/repository-foundry-quality.test.js +1138 -0
- package/dist/test/repository-import-context.test.d.ts +1 -0
- package/dist/test/repository-import-context.test.js +354 -0
- package/dist/test/repository-model-authoring.test.d.ts +1 -0
- package/dist/test/repository-model-authoring.test.js +544 -0
- package/dist/test/repository-model.test.d.ts +1 -0
- package/dist/test/repository-model.test.js +2195 -0
- package/dist/test/repository-node-test-reporter.test.d.ts +1 -0
- package/dist/test/repository-node-test-reporter.test.js +104 -0
- package/dist/test/repository-oracle-concurrency.test.d.ts +1 -0
- package/dist/test/repository-oracle-concurrency.test.js +542 -0
- package/dist/test/repository-oracle-coverage-witness.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage-witness.test.js +511 -0
- package/dist/test/repository-oracle-coverage.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage.test.js +168 -0
- package/dist/test/repository-oracle-evidence.test.d.ts +1 -0
- package/dist/test/repository-oracle-evidence.test.js +185 -0
- package/dist/test/repository-oracle-plan.test.d.ts +1 -0
- package/dist/test/repository-oracle-plan.test.js +176 -0
- package/dist/test/repository-overlay-isolation.test.d.ts +1 -0
- package/dist/test/repository-overlay-isolation.test.js +85 -0
- package/dist/test/repository-preparation-cache.test.d.ts +1 -0
- package/dist/test/repository-preparation-cache.test.js +414 -0
- package/dist/test/repository-public-artifact.test.d.ts +1 -0
- package/dist/test/repository-public-artifact.test.js +273 -0
- package/dist/test/repository-qualification-diagnostics.test.d.ts +1 -0
- package/dist/test/repository-qualification-diagnostics.test.js +524 -0
- package/dist/test/repository-reference-authoring.test.d.ts +1 -0
- package/dist/test/repository-reference-authoring.test.js +1633 -0
- package/dist/test/repository-review-evidence-v2.test.d.ts +1 -0
- package/dist/test/repository-review-evidence-v2.test.js +183 -0
- package/dist/test/repository-review-evidence.test.d.ts +1 -0
- package/dist/test/repository-review-evidence.test.js +124 -0
- package/dist/test/repository-seed-exclusions.test.d.ts +1 -0
- package/dist/test/repository-seed-exclusions.test.js +96 -0
- package/dist/test/repository-seed-selection.test.d.ts +1 -0
- package/dist/test/repository-seed-selection.test.js +504 -0
- package/dist/test/repository-semantic-calibration.test.d.ts +1 -0
- package/dist/test/repository-semantic-calibration.test.js +688 -0
- package/dist/test/repository-solution-edits.test.d.ts +1 -0
- package/dist/test/repository-solution-edits.test.js +377 -0
- package/dist/test/repository-specification-budget.test.d.ts +1 -0
- package/dist/test/repository-specification-budget.test.js +171 -0
- package/dist/test/repository-specification-contract-checkpoint.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-checkpoint.test.js +228 -0
- package/dist/test/repository-specification-contract-facts.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-facts.test.js +177 -0
- package/dist/test/repository-trajectory-authoring.test.d.ts +1 -0
- package/dist/test/repository-trajectory-authoring.test.js +176 -0
- package/dist/test/repository-valid-control-plan.test.d.ts +1 -0
- package/dist/test/repository-valid-control-plan.test.js +45 -0
- package/dist/test/repository-vitest-phase.test.d.ts +1 -0
- package/dist/test/repository-vitest-phase.test.js +848 -0
- package/dist/test/repository-vitest-reporter.test.d.ts +1 -0
- package/dist/test/repository-vitest-reporter.test.js +158 -0
- package/dist/test/repository-workspace-build.test.d.ts +1 -0
- package/dist/test/repository-workspace-build.test.js +160 -0
- package/dist/test/strict-authoring-schema.test.d.ts +1 -0
- package/dist/test/strict-authoring-schema.test.js +169 -0
- package/package.json +48 -7
|
@@ -0,0 +1,1461 @@
|
|
|
1
|
+
import assert from "node:assert/strict";
|
|
2
|
+
import { execFile } from "node:child_process";
|
|
3
|
+
import { mkdir, mkdtemp, readFile, readdir, rm, writeFile } from "node:fs/promises";
|
|
4
|
+
import { tmpdir } from "node:os";
|
|
5
|
+
import { join } from "node:path";
|
|
6
|
+
import { test } from "node:test";
|
|
7
|
+
import { promisify } from "node:util";
|
|
8
|
+
import { Effect, Schema } from "effect";
|
|
9
|
+
import { replayRepositoryCommandsV1, RepositoryCommandExecutionPort } from "../adapters/repository-command-runner.js";
|
|
10
|
+
import { makeCaseCheckpointStoreV1 } from "../case-checkpoint-store.js";
|
|
11
|
+
import { makePipelineCaseContextV1 } from "../case-pipeline-protocol.js";
|
|
12
|
+
import { RepositoryFoundryError } from "../errors.js";
|
|
13
|
+
import { EVAL_CAPABILITY_DURABLE_FINALIZATION_RESERVE_MS_V1, evalCapabilityExecutionEnvelopeV1 } from "../eval-capability-execution-envelope.js";
|
|
14
|
+
import { EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_MODEL_CALLS_V1, EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_WALL_TIME_MS_V1 } from "../eval-capability-policy.js";
|
|
15
|
+
import { discoverEvalCandidatesV1 } from "../services/eval-candidate/service.js";
|
|
16
|
+
import { acceptedScientificDigestV1, evalCandidateQualityScorecardV1, makeEvalCapabilitiesV1 } from "../services/eval-capabilities/service.js";
|
|
17
|
+
import { RepositoryCommandObservability } from "../services/command-observability/service.js";
|
|
18
|
+
import { RepositoryFoundryLanguageModel } from "../services/language-model/service.js";
|
|
19
|
+
import { buildRepositoryBehaviorMapV1 } from "../services/repository-model/service.js";
|
|
20
|
+
import { runPipelineOracleStageV1 } from "../services/pipeline-oracle/service.js";
|
|
21
|
+
import { runPipelineControlsCapabilityV1 } from "../services/pipeline-controls/service.js";
|
|
22
|
+
const exec = promisify(execFile);
|
|
23
|
+
const runtime = {
|
|
24
|
+
name: "node",
|
|
25
|
+
version: process.versions.node,
|
|
26
|
+
platform: process.platform,
|
|
27
|
+
architecture: process.arch
|
|
28
|
+
};
|
|
29
|
+
const dimension = {
|
|
30
|
+
id: "clamp",
|
|
31
|
+
description: "Clamp lower bound without changing the upper bound.",
|
|
32
|
+
includes: ["behavioral regression"],
|
|
33
|
+
excludes: ["formatting"]
|
|
34
|
+
};
|
|
35
|
+
const sourcePath = "packages/main/src/index.js";
|
|
36
|
+
async function fixture(t, maximumModelCalls = 30, regressionTestPath = "packages/main/src/index.test.js") {
|
|
37
|
+
const root = await mkdtemp(join(tmpdir(), "agentic-capabilities-"));
|
|
38
|
+
const stateRoot = await mkdtemp(join(tmpdir(), "agentic-capabilities-state-"));
|
|
39
|
+
t.after(async () => {
|
|
40
|
+
await rm(root, { recursive: true, force: true });
|
|
41
|
+
await rm(stateRoot, { recursive: true, force: true });
|
|
42
|
+
});
|
|
43
|
+
const git = async (...args) => (await exec("git", ["-C", root, ...args])).stdout.trim();
|
|
44
|
+
await git("init", "-q");
|
|
45
|
+
await git("config", "user.email", "eval@example.test");
|
|
46
|
+
await git("config", "user.name", "Eval Fixture");
|
|
47
|
+
await mkdir(join(root, "packages/main/src"), { recursive: true });
|
|
48
|
+
await writeFile(join(root, "package.json"), JSON.stringify({ name: "fixture", private: true, workspaces: ["packages/*"] }));
|
|
49
|
+
await writeFile(join(root, "packages/main/package.json"), JSON.stringify({
|
|
50
|
+
name: "fixture-main",
|
|
51
|
+
type: "module",
|
|
52
|
+
exports: "./src/index.js",
|
|
53
|
+
scripts: { test: "node --test" }
|
|
54
|
+
}));
|
|
55
|
+
await writeFile(join(root, sourcePath), "export const clamp = (value) => Math.min(100, value);\n");
|
|
56
|
+
await writeFile(join(root, "packages/main/src/index.test.js"), 'import assert from "node:assert/strict"; import test from "node:test"; import { clamp } from "./index.js"; test("upper", () => assert.equal(clamp(101), 100));\n');
|
|
57
|
+
await git("add", ".");
|
|
58
|
+
await git("commit", "-qm", "feat: add upper clamp");
|
|
59
|
+
await writeFile(join(root, sourcePath), "export const clamp = (value) => Math.max(0, Math.min(100, value));\n");
|
|
60
|
+
await writeFile(join(root, regressionTestPath), 'import assert from "node:assert/strict"; import test from "node:test"; import { clamp } from "./index.js"; test("upper", () => assert.equal(clamp(101), 100)); test("lower", () => assert.equal(clamp(-1), 0));\n');
|
|
61
|
+
await git("add", ".");
|
|
62
|
+
await git("commit", "-qm", "fix: clamp lower bound");
|
|
63
|
+
const map = await Effect.runPromise(buildRepositoryBehaviorMapV1({ repositoryRoot: root, requestedRef: "HEAD" }));
|
|
64
|
+
const proposals = await Effect.runPromise(discoverEvalCandidatesV1({ map, coverageCell: "clamp", query: "clamp" }));
|
|
65
|
+
assert.equal(proposals.candidates.length, 1);
|
|
66
|
+
const proposal = proposals.candidates[0];
|
|
67
|
+
const context = makePipelineCaseContextV1({
|
|
68
|
+
repositoryRoot: root,
|
|
69
|
+
caseDirectory: stateRoot,
|
|
70
|
+
caseId: "agentic-clamp",
|
|
71
|
+
dimension,
|
|
72
|
+
primaryModel: "openai/fixture",
|
|
73
|
+
budget: { maxModelCalls: maximumModelCalls, maxWallMs: 90 * 60_000 },
|
|
74
|
+
oracleRepetitions: 2,
|
|
75
|
+
oracleConcurrency: 1
|
|
76
|
+
});
|
|
77
|
+
return { root, stateRoot, map, proposal, context };
|
|
78
|
+
}
|
|
79
|
+
function fixtureSuite(testPath = "packages/main/src/index.test.js") {
|
|
80
|
+
const entries = [
|
|
81
|
+
{
|
|
82
|
+
id: "negative",
|
|
83
|
+
kind: "historical-regression",
|
|
84
|
+
expectationMode: "changes",
|
|
85
|
+
assertion: "assert.equal(clamp(-1), 0)"
|
|
86
|
+
},
|
|
87
|
+
{
|
|
88
|
+
id: "upper",
|
|
89
|
+
kind: "boundary",
|
|
90
|
+
expectationMode: "preserved",
|
|
91
|
+
assertion: "assert.equal(clamp(101), 100)"
|
|
92
|
+
},
|
|
93
|
+
{
|
|
94
|
+
id: "fraction",
|
|
95
|
+
kind: "metamorphic",
|
|
96
|
+
expectationMode: "preserved",
|
|
97
|
+
assertion: "assert.equal(clamp(25.5), 25.5)"
|
|
98
|
+
}
|
|
99
|
+
];
|
|
100
|
+
return {
|
|
101
|
+
version: 1,
|
|
102
|
+
caseId: "agentic-clamp",
|
|
103
|
+
fixtures: entries.map(({ assertion: _assertion, ...entry }) => ({
|
|
104
|
+
...entry,
|
|
105
|
+
description: entry.id,
|
|
106
|
+
expectedBehavior: entry.id,
|
|
107
|
+
source: "generated",
|
|
108
|
+
testPath
|
|
109
|
+
})),
|
|
110
|
+
overlays: entries.map((entry) => ({
|
|
111
|
+
path: testPath,
|
|
112
|
+
fixtureIds: [entry.id],
|
|
113
|
+
content: `import assert from "node:assert/strict"; import test from "node:test"; import { clamp } from "./index.js"; test("${entry.id}", () => { ${entry.assertion}; });`
|
|
114
|
+
}))
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
function model(heldOutEscapes = false, specificationAccepted = true, regressionTestPath = "packages/main/src/index.test.js") {
|
|
118
|
+
const calls = [];
|
|
119
|
+
const inner = {
|
|
120
|
+
generateAssignedStructured: (request) => Effect.gen(function* () {
|
|
121
|
+
calls.push(request.schemaName);
|
|
122
|
+
const input = request.input;
|
|
123
|
+
const task = input.task;
|
|
124
|
+
let value;
|
|
125
|
+
if (request.schemaName === "routekit_pipeline_spec_analysis_v1")
|
|
126
|
+
value = {
|
|
127
|
+
action: "submit",
|
|
128
|
+
result: {
|
|
129
|
+
summary: "Clamp lower inputs at zero while preserving upper behavior.",
|
|
130
|
+
userObservableBehaviors: ["negative returns zero", "upper returns 100"],
|
|
131
|
+
constraints: ["preserve export"],
|
|
132
|
+
risks: [],
|
|
133
|
+
relevantPaths: [sourcePath],
|
|
134
|
+
scope: ["changed", "preserved"].map((kind) => ({
|
|
135
|
+
id: kind,
|
|
136
|
+
kind,
|
|
137
|
+
behaviorIds: [task.seed.targetBehavior[0].id],
|
|
138
|
+
description: kind === "changed"
|
|
139
|
+
? "Negative inputs return zero."
|
|
140
|
+
: "Upper cap and exact in-range values remain preserved.",
|
|
141
|
+
evidence: ["initial", "reference"].map((commit) => ({
|
|
142
|
+
commit,
|
|
143
|
+
path: sourcePath,
|
|
144
|
+
startLine: 1,
|
|
145
|
+
endLine: 1
|
|
146
|
+
}))
|
|
147
|
+
}))
|
|
148
|
+
}
|
|
149
|
+
};
|
|
150
|
+
else if (request.schemaName === "routekit_pipeline_visible_task_v1")
|
|
151
|
+
value = {
|
|
152
|
+
requestText: "Clamp negative inputs to zero, preserve upper cap at one hundred and in-range values.",
|
|
153
|
+
constraints: ["Preserve public export."]
|
|
154
|
+
};
|
|
155
|
+
else if (request.schemaName === "routekit_pipeline_specification_review_v1")
|
|
156
|
+
value = {
|
|
157
|
+
action: "submit",
|
|
158
|
+
result: {
|
|
159
|
+
verdict: specificationAccepted ? "sufficient" : "insufficient",
|
|
160
|
+
detail: specificationAccepted
|
|
161
|
+
? "All grounded behavior is specified."
|
|
162
|
+
: "The visible task still needs a grounded revision.",
|
|
163
|
+
findings: [],
|
|
164
|
+
criticalBehaviorCoverage: task
|
|
165
|
+
.targetBehavior.filter((behavior) => behavior.critical)
|
|
166
|
+
.map((behavior) => ({
|
|
167
|
+
behaviorId: behavior.id,
|
|
168
|
+
outcome: "covered",
|
|
169
|
+
detail: "Stated explicitly."
|
|
170
|
+
})),
|
|
171
|
+
dimensionFit: {
|
|
172
|
+
requestedDimension: dimension.id,
|
|
173
|
+
outcome: "fit",
|
|
174
|
+
detail: "Behavior repair.",
|
|
175
|
+
evidencePaths: [sourcePath]
|
|
176
|
+
},
|
|
177
|
+
referenceCompatibility: {
|
|
178
|
+
outcome: "compatible",
|
|
179
|
+
detail: "Reference satisfies scope.",
|
|
180
|
+
scopeCoverage: task.scope.map((scope) => ({
|
|
181
|
+
scopeId: scope.id,
|
|
182
|
+
outcome: "supported",
|
|
183
|
+
detail: "Pinned implementation supports it.",
|
|
184
|
+
evidence: [
|
|
185
|
+
{
|
|
186
|
+
sourceId: task.sourceCatalog.find((source) => source.path === sourcePath && source.phase === "reference").sourceId,
|
|
187
|
+
startLine: 1,
|
|
188
|
+
endLine: 1
|
|
189
|
+
}
|
|
190
|
+
]
|
|
191
|
+
}))
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
};
|
|
195
|
+
else if (request.schemaName === "routekit_pipeline_oracle_submission_v1")
|
|
196
|
+
value = {
|
|
197
|
+
action: "submit",
|
|
198
|
+
result: {
|
|
199
|
+
suite: fixtureSuite(regressionTestPath),
|
|
200
|
+
scopeCoverage: [
|
|
201
|
+
{
|
|
202
|
+
scopeId: "changed",
|
|
203
|
+
outcome: "covered",
|
|
204
|
+
fixtureIds: ["negative"],
|
|
205
|
+
detail: "Lower clamp assertion."
|
|
206
|
+
},
|
|
207
|
+
{
|
|
208
|
+
scopeId: "preserved",
|
|
209
|
+
outcome: "covered",
|
|
210
|
+
fixtureIds: ["upper", "fraction"],
|
|
211
|
+
detail: "Upper cap and exact fractional values asserted."
|
|
212
|
+
}
|
|
213
|
+
]
|
|
214
|
+
}
|
|
215
|
+
};
|
|
216
|
+
else if (request.schemaName === "routekit_pipeline_valid_solution_generator_a_v1")
|
|
217
|
+
value = {
|
|
218
|
+
id: "solver-a",
|
|
219
|
+
family: "composition",
|
|
220
|
+
kind: "independent-valid",
|
|
221
|
+
fileOverrides: [
|
|
222
|
+
{
|
|
223
|
+
path: sourcePath,
|
|
224
|
+
content: "export const clamp = (value) => Math.min(100, Math.max(0, value));\n"
|
|
225
|
+
}
|
|
226
|
+
]
|
|
227
|
+
};
|
|
228
|
+
else if (request.schemaName === "routekit_pipeline_adversary_v1")
|
|
229
|
+
value = {
|
|
230
|
+
id: "wrong-rounding",
|
|
231
|
+
family: "rounding",
|
|
232
|
+
kind: "mutation",
|
|
233
|
+
fileOverrides: [
|
|
234
|
+
{
|
|
235
|
+
path: sourcePath,
|
|
236
|
+
content: "export const clamp = (value) => Math.min(100, Math.max(0, Math.round(value)));\n"
|
|
237
|
+
}
|
|
238
|
+
]
|
|
239
|
+
};
|
|
240
|
+
else if (request.schemaName === "routekit_pipeline_held_out_adversary_v1")
|
|
241
|
+
value = {
|
|
242
|
+
solutions: [
|
|
243
|
+
{
|
|
244
|
+
id: "heldout-truncation",
|
|
245
|
+
family: "truncation",
|
|
246
|
+
kind: "model-near-miss",
|
|
247
|
+
fileOverrides: [
|
|
248
|
+
{
|
|
249
|
+
path: sourcePath,
|
|
250
|
+
content: heldOutEscapes
|
|
251
|
+
? "export const clamp = (value) => Math.min(100, Math.max(0, value));\n"
|
|
252
|
+
: "export const clamp = (value) => Math.min(100, Math.max(0, Math.trunc(value)));\n"
|
|
253
|
+
}
|
|
254
|
+
]
|
|
255
|
+
}
|
|
256
|
+
]
|
|
257
|
+
};
|
|
258
|
+
else
|
|
259
|
+
return yield* new RepositoryFoundryError({
|
|
260
|
+
operation: "invoke-language-model",
|
|
261
|
+
detail: `unexpected model request ${request.schemaName}`
|
|
262
|
+
});
|
|
263
|
+
const decoded = yield* Schema.decodeUnknownEffect(request.outputSchema)(typeof value === "object" && value !== null && "action" in value ? { turn: value } : value).pipe(Effect.mapError((cause) => new RepositoryFoundryError({
|
|
264
|
+
operation: "invoke-language-model",
|
|
265
|
+
detail: `fixture response invalid ${String(cause)}`,
|
|
266
|
+
cause
|
|
267
|
+
})));
|
|
268
|
+
return {
|
|
269
|
+
value: decoded,
|
|
270
|
+
call: { operationId: request.operationId, ...request.assignment }
|
|
271
|
+
};
|
|
272
|
+
}),
|
|
273
|
+
generateStructured: () => Effect.fail(new RepositoryFoundryError({
|
|
274
|
+
operation: "invoke-language-model",
|
|
275
|
+
detail: "unplanned structured call"
|
|
276
|
+
}))
|
|
277
|
+
};
|
|
278
|
+
return { inner, calls };
|
|
279
|
+
}
|
|
280
|
+
async function harness(t, heldOutEscapes = false, maximumModelCalls = 30, regressionTestPath = "packages/main/src/index.test.js") {
|
|
281
|
+
const repo = await fixture(t, maximumModelCalls, regressionTestPath);
|
|
282
|
+
const scripted = model(heldOutEscapes, true, regressionTestPath);
|
|
283
|
+
const observed = [];
|
|
284
|
+
const capabilities = await Effect.runPromise(makeEvalCapabilitiesV1({ ...repo, runtime }).pipe(Effect.provideService(RepositoryFoundryLanguageModel, scripted.inner)));
|
|
285
|
+
const invoke = (request) => capabilities.invoke(request).pipe(Effect.provideService(RepositoryCommandObservability, {
|
|
286
|
+
observe: (event) => {
|
|
287
|
+
observed.push(event.kind);
|
|
288
|
+
}
|
|
289
|
+
}));
|
|
290
|
+
return { ...repo, scripted, capabilities, invoke, observed };
|
|
291
|
+
}
|
|
292
|
+
test("scientific execution uses the shared fixed downstream reserve and ten-minute ceiling", () => {
|
|
293
|
+
const decision = evalCapabilityExecutionEnvelopeV1({
|
|
294
|
+
capability: "draft_specification",
|
|
295
|
+
requestedMaximumModelCalls: EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_MODEL_CALLS_V1,
|
|
296
|
+
requestedMaximumWallMs: EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_WALL_TIME_MS_V1,
|
|
297
|
+
campaignModelCallsRemaining: 120,
|
|
298
|
+
campaignWallTimeRemainingMs: 90 * 60_000
|
|
299
|
+
});
|
|
300
|
+
assert.equal(decision.admitted, true);
|
|
301
|
+
assert.deepEqual(decision.downstreamReserve, {
|
|
302
|
+
calls: 24,
|
|
303
|
+
wallTimeMs: 75 * 60_000,
|
|
304
|
+
attempts: 6
|
|
305
|
+
});
|
|
306
|
+
assert.deepEqual(decision.minimumRequired, {
|
|
307
|
+
calls: 25,
|
|
308
|
+
wallTimeMs: 75 * 60_000 +
|
|
309
|
+
EVAL_CAPABILITY_DURABLE_FINALIZATION_RESERVE_MS_V1 +
|
|
310
|
+
1
|
|
311
|
+
});
|
|
312
|
+
assert.deepEqual(decision.finalizationReserve, { wallTimeMs: 30_000 });
|
|
313
|
+
assert.deepEqual(decision.availableForCapability, {
|
|
314
|
+
calls: 96,
|
|
315
|
+
wallTimeMs: 870_000
|
|
316
|
+
});
|
|
317
|
+
assert.deepEqual(decision.envelope, {
|
|
318
|
+
maximumModelCalls: EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_MODEL_CALLS_V1,
|
|
319
|
+
maximumWallTimeMs: EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_WALL_TIME_MS_V1
|
|
320
|
+
});
|
|
321
|
+
});
|
|
322
|
+
test("the rounded incident budget admits a shorter positive slice instead of requiring the maximum", () => {
|
|
323
|
+
const decision = evalCapabilityExecutionEnvelopeV1({
|
|
324
|
+
capability: "draft_specification",
|
|
325
|
+
requestedMaximumModelCalls: 12,
|
|
326
|
+
requestedMaximumWallMs: 10 * 60_000,
|
|
327
|
+
campaignModelCallsRemaining: 104,
|
|
328
|
+
campaignWallTimeRemainingMs: 82 * 60_000
|
|
329
|
+
});
|
|
330
|
+
assert.equal(decision.admitted, true);
|
|
331
|
+
assert.deepEqual(decision.requestedMaximum, { calls: 12, wallTimeMs: 600_000 });
|
|
332
|
+
assert.deepEqual(decision.downstreamReserve, {
|
|
333
|
+
calls: 24,
|
|
334
|
+
wallTimeMs: 4_500_000,
|
|
335
|
+
attempts: 6
|
|
336
|
+
});
|
|
337
|
+
assert.deepEqual(decision.finalizationReserve, { wallTimeMs: 30_000 });
|
|
338
|
+
assert.deepEqual(decision.availableForCapability, {
|
|
339
|
+
calls: 80,
|
|
340
|
+
wallTimeMs: 390_000
|
|
341
|
+
});
|
|
342
|
+
assert.deepEqual(decision.envelope, {
|
|
343
|
+
maximumModelCalls: 12,
|
|
344
|
+
maximumWallTimeMs: 390_000
|
|
345
|
+
});
|
|
346
|
+
});
|
|
347
|
+
test("the exact incident capsule keeps durable finalization outside its adaptive slice", () => {
|
|
348
|
+
const decision = evalCapabilityExecutionEnvelopeV1({
|
|
349
|
+
capability: "draft_specification",
|
|
350
|
+
requestedMaximumModelCalls: 12,
|
|
351
|
+
requestedMaximumWallMs: 600_000,
|
|
352
|
+
campaignModelCallsRemaining: 104,
|
|
353
|
+
campaignWallTimeRemainingMs: 5_016_723
|
|
354
|
+
});
|
|
355
|
+
assert.equal(decision.admitted, true);
|
|
356
|
+
assert.deepEqual(decision.available, {
|
|
357
|
+
calls: 104,
|
|
358
|
+
wallTimeMs: 5_016_723
|
|
359
|
+
});
|
|
360
|
+
assert.deepEqual(decision.downstreamReserve, {
|
|
361
|
+
calls: 24,
|
|
362
|
+
wallTimeMs: 4_500_000,
|
|
363
|
+
attempts: 6
|
|
364
|
+
});
|
|
365
|
+
assert.deepEqual(decision.finalizationReserve, { wallTimeMs: 30_000 });
|
|
366
|
+
assert.deepEqual(decision.availableForCapability, {
|
|
367
|
+
calls: 80,
|
|
368
|
+
wallTimeMs: 486_723
|
|
369
|
+
});
|
|
370
|
+
assert.deepEqual(decision.envelope, {
|
|
371
|
+
maximumModelCalls: 12,
|
|
372
|
+
maximumWallTimeMs: 486_723
|
|
373
|
+
});
|
|
374
|
+
});
|
|
375
|
+
test("scientific execution denies on calls when no positive call slice remains", () => {
|
|
376
|
+
const decision = evalCapabilityExecutionEnvelopeV1({
|
|
377
|
+
capability: "draft_specification",
|
|
378
|
+
requestedMaximumModelCalls: 12,
|
|
379
|
+
requestedMaximumWallMs: 600_000,
|
|
380
|
+
campaignModelCallsRemaining: 24,
|
|
381
|
+
campaignWallTimeRemainingMs: 5_400_000
|
|
382
|
+
});
|
|
383
|
+
assert.equal(decision.admitted, false);
|
|
384
|
+
assert.equal(decision.admitted ? null : decision.dimension, "calls");
|
|
385
|
+
assert.equal(decision.envelope, null);
|
|
386
|
+
assert.deepEqual(decision.availableForCapability, {
|
|
387
|
+
calls: 0,
|
|
388
|
+
wallTimeMs: 870_000
|
|
389
|
+
});
|
|
390
|
+
});
|
|
391
|
+
test("scientific execution denies on wall time when only downstream and finalization reserves remain", () => {
|
|
392
|
+
const decision = evalCapabilityExecutionEnvelopeV1({
|
|
393
|
+
capability: "draft_specification",
|
|
394
|
+
requestedMaximumModelCalls: 12,
|
|
395
|
+
requestedMaximumWallMs: 600_000,
|
|
396
|
+
campaignModelCallsRemaining: 104,
|
|
397
|
+
campaignWallTimeRemainingMs: 4_530_000
|
|
398
|
+
});
|
|
399
|
+
assert.equal(decision.admitted, false);
|
|
400
|
+
assert.equal(decision.admitted ? null : decision.dimension, "wall_time");
|
|
401
|
+
assert.equal(decision.envelope, null);
|
|
402
|
+
assert.deepEqual(decision.availableForCapability, {
|
|
403
|
+
calls: 80,
|
|
404
|
+
wallTimeMs: 0
|
|
405
|
+
});
|
|
406
|
+
});
|
|
407
|
+
test("control capabilities remain reachable after scientific budget exhaustion", () => {
|
|
408
|
+
const decision = evalCapabilityExecutionEnvelopeV1({
|
|
409
|
+
capability: "finish_campaign",
|
|
410
|
+
campaignModelCallsRemaining: 0,
|
|
411
|
+
campaignWallTimeRemainingMs: 0
|
|
412
|
+
});
|
|
413
|
+
assert.equal(decision.admitted, true);
|
|
414
|
+
assert.deepEqual(decision.downstreamReserve, { calls: 0, wallTimeMs: 0, attempts: 0 });
|
|
415
|
+
assert.deepEqual(decision.finalizationReserve, { wallTimeMs: 0 });
|
|
416
|
+
assert.deepEqual(decision.envelope, { maximumModelCalls: 1, maximumWallTimeMs: 1 });
|
|
417
|
+
});
|
|
418
|
+
test("finish_campaign executes after the candidate runtime has exhausted calls and wall time", async (t) => {
|
|
419
|
+
const repo = await fixture(t, 30);
|
|
420
|
+
const store = makeCaseCheckpointStoreV1({ caseDirectory: repo.context.caseDirectory });
|
|
421
|
+
const now = new Date().toISOString();
|
|
422
|
+
await Effect.runPromise(store.writeState({
|
|
423
|
+
version: 1,
|
|
424
|
+
caseId: repo.context.caseId,
|
|
425
|
+
repositoryRoot: repo.context.repositoryRoot,
|
|
426
|
+
dimension: repo.context.dimension,
|
|
427
|
+
model: repo.context.primaryModel,
|
|
428
|
+
identityDigest: repo.proposal.candidate.revision,
|
|
429
|
+
stage: "seed",
|
|
430
|
+
status: "running",
|
|
431
|
+
startedAt: now,
|
|
432
|
+
updatedAt: now,
|
|
433
|
+
modelCalls: repo.context.budget.maxModelCalls,
|
|
434
|
+
elapsedMs: repo.context.budget.maxWallMs,
|
|
435
|
+
resumeCount: 0,
|
|
436
|
+
folds: 0
|
|
437
|
+
}));
|
|
438
|
+
const scripted = model();
|
|
439
|
+
const capabilities = await Effect.runPromise(makeEvalCapabilitiesV1({ ...repo, store, runtime }).pipe(Effect.provideService(RepositoryFoundryLanguageModel, scripted.inner)));
|
|
440
|
+
const result = await Effect.runPromise(capabilities.invoke({
|
|
441
|
+
operationId: "finish-after-exhaustion",
|
|
442
|
+
deadlineAt: new Date(Date.now() - 1_000).toISOString(),
|
|
443
|
+
request: { tool: "finish_campaign", args: { reason: "Budget is exhausted." } }
|
|
444
|
+
}));
|
|
445
|
+
assert.equal(result.outcome, "completed", result.evidenceSummary);
|
|
446
|
+
assert.deepEqual(result.allowedNext, [
|
|
447
|
+
"get_progress",
|
|
448
|
+
"draft_specification",
|
|
449
|
+
"plan_environment",
|
|
450
|
+
"reject_candidate"
|
|
451
|
+
]);
|
|
452
|
+
assert.equal(scripted.calls.length, 0);
|
|
453
|
+
});
|
|
454
|
+
test("terminal candidate results recommend only inspect, reselection, and finish capabilities", async (t) => {
|
|
455
|
+
const h = await harness(t);
|
|
456
|
+
const candidate = h.proposal.candidate.candidate;
|
|
457
|
+
const rejected = await Effect.runPromise(h.invoke({
|
|
458
|
+
operationId: "reject-terminal-candidate",
|
|
459
|
+
request: {
|
|
460
|
+
tool: "reject_candidate",
|
|
461
|
+
args: {
|
|
462
|
+
candidate,
|
|
463
|
+
reason: "Candidate is conclusively unsuitable.",
|
|
464
|
+
evidence: []
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
}));
|
|
468
|
+
assert.equal(rejected.outcome, "scientific_rejection");
|
|
469
|
+
assert.deepEqual(rejected.allowedNext, [
|
|
470
|
+
"get_progress",
|
|
471
|
+
"list_candidates",
|
|
472
|
+
"read_diagnostics",
|
|
473
|
+
"finish_campaign"
|
|
474
|
+
]);
|
|
475
|
+
assert.ok(!rejected.allowedNext.includes("reject_candidate"));
|
|
476
|
+
assert.ok(!rejected.allowedNext.includes("request_admission"));
|
|
477
|
+
});
|
|
478
|
+
test("source-only grounded draft returns immutable artifacts and honest missing quality measurements", async (t) => {
|
|
479
|
+
const h = await harness(t);
|
|
480
|
+
const drafted = await Effect.runPromise(h.invoke({
|
|
481
|
+
operationId: "draft-1",
|
|
482
|
+
request: { tool: "draft_specification", args: { candidate: h.proposal.candidate.candidate } }
|
|
483
|
+
}));
|
|
484
|
+
assert.equal(drafted.outcome, "completed", drafted.evidenceSummary);
|
|
485
|
+
assert.equal(drafted.usage.receivedCalls, 3);
|
|
486
|
+
assert.equal(drafted.accepted.length, 1);
|
|
487
|
+
assert.equal(h.scripted.calls.length, 3);
|
|
488
|
+
assert.deepEqual(h.observed, []);
|
|
489
|
+
const scorecard = await Effect.runPromise(h.capabilities.scorecard);
|
|
490
|
+
assert.equal(scorecard.disposition, "draft");
|
|
491
|
+
assert.equal(scorecard.reference.missing, 1);
|
|
492
|
+
assert.equal(scorecard.usage.receivedCalls, 3);
|
|
493
|
+
assert.equal(scorecard.usage.spendUnknown, true);
|
|
494
|
+
assert.ok((await readdir(join(h.stateRoot, "calls"))).some((path) => path.startsWith("artifact-")));
|
|
495
|
+
const frozen = await Effect.runPromise(h.invoke({
|
|
496
|
+
operationId: "freeze-without-evidence",
|
|
497
|
+
request: {
|
|
498
|
+
tool: "freeze_candidate",
|
|
499
|
+
args: { candidate: h.proposal.candidate.candidate, revision: h.proposal.candidate.revision }
|
|
500
|
+
}
|
|
501
|
+
}));
|
|
502
|
+
assert.equal(frozen.outcome, "progress");
|
|
503
|
+
assert.deepEqual(frozen.accepted, []);
|
|
504
|
+
assert.match(frozen.evidenceSummary, /not ready to freeze/u);
|
|
505
|
+
});
|
|
506
|
+
test("scientific progress projects wall time from the absolute invocation deadline", async (t) => {
|
|
507
|
+
const h = await harness(t);
|
|
508
|
+
const deadlineAt = new Date(Date.now() + 78 * 60_000).toISOString();
|
|
509
|
+
const result = await Effect.runPromise(h.invoke({
|
|
510
|
+
operationId: "absolute-deadline-draft",
|
|
511
|
+
deadlineAt,
|
|
512
|
+
request: {
|
|
513
|
+
tool: "draft_specification",
|
|
514
|
+
args: { candidate: h.proposal.candidate.candidate }
|
|
515
|
+
}
|
|
516
|
+
}));
|
|
517
|
+
assert.equal(result.outcome, "completed", result.evidenceSummary);
|
|
518
|
+
assert.equal(result.progressReceipt?.remaining.downstreamWallTimeMs, 75 * 60_000);
|
|
519
|
+
assert.ok((result.progressReceipt?.remaining.phaseWallTimeMs ?? 0) > 140_000);
|
|
520
|
+
assert.ok((result.progressReceipt?.remaining.phaseWallTimeMs ?? Infinity) <= 150_000);
|
|
521
|
+
});
|
|
522
|
+
test("host progress authority ignores provisional receipts until the host returns them", async (t) => {
|
|
523
|
+
const h = await harness(t, false, 60);
|
|
524
|
+
const request = {
|
|
525
|
+
tool: "draft_specification",
|
|
526
|
+
args: { candidate: h.proposal.candidate.candidate }
|
|
527
|
+
};
|
|
528
|
+
const first = await Effect.runPromise(h.invoke({
|
|
529
|
+
operationId: "authority-first",
|
|
530
|
+
request,
|
|
531
|
+
authoritativeProgressReceipts: []
|
|
532
|
+
}));
|
|
533
|
+
assert.ok(first.progressReceipt);
|
|
534
|
+
assert.equal(first.progressReceipt.sameFingerprintCount, 0);
|
|
535
|
+
assert.equal((await readdir(join(h.stateRoot, "calls"))).some((path) => path.startsWith(first.progressReceipt.receiptId)), false, "a host-controlled provisional receipt must not enter the local checkpoint store");
|
|
536
|
+
const ignored = await Effect.runPromise(h.invoke({
|
|
537
|
+
operationId: "authority-provisional-ignored",
|
|
538
|
+
request,
|
|
539
|
+
authoritativeProgressReceipts: []
|
|
540
|
+
}));
|
|
541
|
+
assert.ok(ignored.progressReceipt);
|
|
542
|
+
assert.equal(ignored.progressReceipt.remaining.phaseAttempts, first.progressReceipt.remaining.phaseAttempts, "an unaccepted local receipt cannot consume an authoritative phase attempt");
|
|
543
|
+
const accepted = await Effect.runPromise(h.invoke({
|
|
544
|
+
operationId: "authority-host-accepted",
|
|
545
|
+
request,
|
|
546
|
+
authoritativeProgressReceipts: [first.progressReceipt]
|
|
547
|
+
}));
|
|
548
|
+
assert.ok(accepted.progressReceipt);
|
|
549
|
+
assert.equal(accepted.progressReceipt.remaining.phaseAttempts, first.progressReceipt.remaining.phaseAttempts - 1, "the exact receipt advances the chain only after the host returns it");
|
|
550
|
+
assert.equal(accepted.progressReceipt.sameFingerprintCount, 1);
|
|
551
|
+
assert.deepEqual(accepted.progressReceipt.previousFingerprint, first.progressReceipt.resultingFingerprint);
|
|
552
|
+
});
|
|
553
|
+
test("an invocation yields at its call budget without exhausting the whole campaign", async (t) => {
|
|
554
|
+
const h = await harness(t);
|
|
555
|
+
const result = await Effect.runPromise(h.invoke({
|
|
556
|
+
operationId: "one-call-slice",
|
|
557
|
+
maximumModelCalls: 1,
|
|
558
|
+
request: { tool: "draft_specification", args: { candidate: h.proposal.candidate.candidate } }
|
|
559
|
+
}));
|
|
560
|
+
assert.equal(result.outcome, "progress");
|
|
561
|
+
assert.equal(result.usage.receivedCalls, 1);
|
|
562
|
+
assert.equal(result.progressReceipt?.remaining.phaseAttempts, 8);
|
|
563
|
+
assert.match(result.evidenceSummary, /without consuming the campaign budget/u);
|
|
564
|
+
const continued = await Effect.runPromise(h.invoke({
|
|
565
|
+
operationId: "continued-slice",
|
|
566
|
+
request: { tool: "draft_specification", args: { candidate: h.proposal.candidate.candidate } }
|
|
567
|
+
}));
|
|
568
|
+
assert.equal(continued.outcome, "completed", continued.evidenceSummary);
|
|
569
|
+
assert.equal(continued.usage.receivedCalls, 2, "completed grounded analysis is not purchased again");
|
|
570
|
+
assert.equal((await Effect.runPromise(h.capabilities.scorecard)).usage.receivedCalls, 3);
|
|
571
|
+
});
|
|
572
|
+
test("three unchanged specification receipts open the scientific circuit without spending the downstream reserve", async (t) => {
|
|
573
|
+
const h = await harness(t, false, 60);
|
|
574
|
+
const request = {
|
|
575
|
+
tool: "draft_specification",
|
|
576
|
+
args: { candidate: h.proposal.candidate.candidate }
|
|
577
|
+
};
|
|
578
|
+
const results = [];
|
|
579
|
+
for (let index = 0; index < 3; index += 1) {
|
|
580
|
+
results.push(await Effect.runPromise(h.invoke({
|
|
581
|
+
operationId: `unchanged-draft-${String(index + 1)}`,
|
|
582
|
+
request
|
|
583
|
+
})));
|
|
584
|
+
}
|
|
585
|
+
const last = results[2].progressReceipt;
|
|
586
|
+
const unrelatedNoOp = {
|
|
587
|
+
...last,
|
|
588
|
+
receiptId: "unrelated-no-op",
|
|
589
|
+
operationId: "unrelated-no-op",
|
|
590
|
+
capability: "plan_environment",
|
|
591
|
+
phase: "environment",
|
|
592
|
+
previousFingerprint: last.resultingFingerprint,
|
|
593
|
+
sameFingerprintCount: 1,
|
|
594
|
+
recordedAt: new Date(Date.parse(last.recordedAt) + 1).toISOString()
|
|
595
|
+
};
|
|
596
|
+
const replayed = await Effect.runPromise(makeEvalCapabilitiesV1({
|
|
597
|
+
context: h.context,
|
|
598
|
+
map: h.map,
|
|
599
|
+
proposal: h.proposal,
|
|
600
|
+
runtime,
|
|
601
|
+
progressReceipts: [...results.flatMap((result) => result.progressReceipt === undefined ? [] : [result.progressReceipt]), unrelatedNoOp]
|
|
602
|
+
}).pipe(Effect.provideService(RepositoryFoundryLanguageModel, h.scripted.inner)));
|
|
603
|
+
results.push(await Effect.runPromise(replayed.invoke({ operationId: "unchanged-draft-4", request })));
|
|
604
|
+
assert.equal(results[0].progressReceipt?.sameFingerprintCount, 0);
|
|
605
|
+
assert.equal(results[1].progressReceipt?.sameFingerprintCount, 1);
|
|
606
|
+
assert.equal(results[2].progressReceipt?.sameFingerprintCount, 2);
|
|
607
|
+
assert.ok(!results[2].allowedNext.includes("draft_specification"));
|
|
608
|
+
assert.equal(results[3].outcome, "scientific_rejection");
|
|
609
|
+
assert.equal(results[3].progressReceipt?.sameFingerprintCount, 3);
|
|
610
|
+
assert.deepEqual(results[3].progressReceipt?.previousFingerprint, results[2].progressReceipt?.resultingFingerprint);
|
|
611
|
+
assert.equal(results[3].usage.receivedCalls, 0);
|
|
612
|
+
assert.equal(results[3].progressReceipt?.remaining.downstreamCalls, 24);
|
|
613
|
+
assert.ok((results[3].progressReceipt?.remaining.phaseCalls ?? 0) > 0, "the circuit opens before consuming the remaining phase allowance");
|
|
614
|
+
assert.equal(h.scripted.calls.length, 3, "cached analysis, writer and critic are free; unchanged repeats perform no model call");
|
|
615
|
+
});
|
|
616
|
+
test("grade timeout is an environment blocker, and an explanatory replan cannot erase it", async (t) => {
|
|
617
|
+
const h = await harness(t);
|
|
618
|
+
const candidate = h.proposal.candidate.candidate;
|
|
619
|
+
await Effect.runPromise(h.invoke({
|
|
620
|
+
operationId: "timeout-spec", request: { tool: "draft_specification", args: { candidate } }
|
|
621
|
+
}));
|
|
622
|
+
const plan = {
|
|
623
|
+
tool: "plan_environment",
|
|
624
|
+
args: {
|
|
625
|
+
candidate, recipeIds: h.proposal.seed.environment.gradeRecipes.map((recipe) => recipe.id),
|
|
626
|
+
executionMode: "source", targetRoots: ["packages/main"], requiredArtifacts: [],
|
|
627
|
+
justification: "Execute the reviewed source."
|
|
628
|
+
}
|
|
629
|
+
};
|
|
630
|
+
const planned = await Effect.runPromise(h.invoke({ operationId: "timeout-plan", request: plan }));
|
|
631
|
+
const environment = planned.accepted[0];
|
|
632
|
+
let commands = 0;
|
|
633
|
+
const qualify = (operationId) => Effect.runPromise(h.invoke({
|
|
634
|
+
operationId, request: { tool: "qualify_reference", args: { candidate, environment } }
|
|
635
|
+
}).pipe(Effect.provideService(RepositoryCommandExecutionPort, {
|
|
636
|
+
run: () => Effect.sync(() => {
|
|
637
|
+
commands += 1;
|
|
638
|
+
return { exitCode: null, timedOut: true, stdout: "", stderr: "" };
|
|
639
|
+
})
|
|
640
|
+
})));
|
|
641
|
+
const blocked = await qualify("timeout-first");
|
|
642
|
+
assert.equal(blocked.outcome, "environment_blocked");
|
|
643
|
+
assert.equal(blocked.gradeExecuted, false);
|
|
644
|
+
assert.ok(!blocked.allowedNext.includes("qualify_reference"));
|
|
645
|
+
assert.ok(!blocked.allowedNext.includes("draft_specification"));
|
|
646
|
+
assert.ok(blocked.allowedNext.includes("plan_environment"));
|
|
647
|
+
assert.ok(!blocked.progressReceipt?.gatesCompleted.includes("reference_qualified"));
|
|
648
|
+
const completedCommands = commands;
|
|
649
|
+
assert.ok(completedCommands > 0);
|
|
650
|
+
const replanned = await Effect.runPromise(h.invoke({
|
|
651
|
+
operationId: "timeout-no-op-plan",
|
|
652
|
+
request: { ...plan, args: { ...plan.args, justification: "A different explanation only." } }
|
|
653
|
+
}));
|
|
654
|
+
assert.deepEqual(replanned.accepted, planned.accepted);
|
|
655
|
+
assert.deepEqual(replanned.progressReceipt?.acceptedEvidenceAdded, []);
|
|
656
|
+
assert.deepEqual(replanned.progressReceipt?.acceptedEvidenceRemoved, []);
|
|
657
|
+
assert.equal((await qualify("timeout-retained")).outcome, "environment_blocked");
|
|
658
|
+
assert.equal(commands, completedCommands);
|
|
659
|
+
});
|
|
660
|
+
test("a caller cannot inject checkpoint or provider authority into a tool request", async (t) => {
|
|
661
|
+
const h = await harness(t);
|
|
662
|
+
await assert.rejects(Effect.runPromise(h.invoke({
|
|
663
|
+
operationId: "invalid",
|
|
664
|
+
request: {
|
|
665
|
+
tool: "get_progress",
|
|
666
|
+
args: { checkpoints: { admit: "passed" }, model: "auto" }
|
|
667
|
+
}
|
|
668
|
+
})), /invalid capability request/u);
|
|
669
|
+
assert.deepEqual(h.scripted.calls, []);
|
|
670
|
+
});
|
|
671
|
+
test("injected execution port runs the real isolated checkout and overlays without credentials", async (t) => {
|
|
672
|
+
const h = await fixture(t);
|
|
673
|
+
const calls = [];
|
|
674
|
+
const results = await Effect.runPromise(replayRepositoryCommandsV1({
|
|
675
|
+
repositoryRoot: h.root,
|
|
676
|
+
commit: h.map.repository.commit,
|
|
677
|
+
recipes: [
|
|
678
|
+
{
|
|
679
|
+
id: "read-real-overlay",
|
|
680
|
+
kind: "custom",
|
|
681
|
+
executable: "node",
|
|
682
|
+
cwd: ".",
|
|
683
|
+
args: [
|
|
684
|
+
"-e",
|
|
685
|
+
"process.stdout.write(require('node:fs').readFileSync('diagnostic.txt','utf8'))"
|
|
686
|
+
],
|
|
687
|
+
timeoutMs: 5_000
|
|
688
|
+
}
|
|
689
|
+
],
|
|
690
|
+
fileOverrides: [{ path: "diagnostic.txt", content: "真实🔎 diagnostic overlay" }],
|
|
691
|
+
repetitions: 1
|
|
692
|
+
}).pipe(Effect.provideService(RepositoryCommandExecutionPort, {
|
|
693
|
+
run: (input) => Effect.tryPromise({
|
|
694
|
+
try: async () => {
|
|
695
|
+
calls.push(input);
|
|
696
|
+
const result = await exec(input.executable, [...input.args], {
|
|
697
|
+
cwd: input.cwd,
|
|
698
|
+
env: input.environment,
|
|
699
|
+
timeout: input.timeoutMs
|
|
700
|
+
});
|
|
701
|
+
return { exitCode: 0, stdout: result.stdout, stderr: result.stderr };
|
|
702
|
+
},
|
|
703
|
+
catch: (cause) => new RepositoryFoundryError({
|
|
704
|
+
operation: "run-command",
|
|
705
|
+
detail: "injected platform command failed",
|
|
706
|
+
cause
|
|
707
|
+
})
|
|
708
|
+
})
|
|
709
|
+
})));
|
|
710
|
+
assert.equal(calls.length, 1);
|
|
711
|
+
assert.notEqual(calls[0].cwd, h.root);
|
|
712
|
+
assert.equal(calls[0].environment.PLANETSCALE_PASSWORD, undefined);
|
|
713
|
+
assert.equal(results[0].stdout, "真实🔎 diagnostic overlay");
|
|
714
|
+
assert.equal(results[0].gradeExecuted, true);
|
|
715
|
+
await assert.rejects(readFile(join(h.root, "diagnostic.txt")), /ENOENT/u);
|
|
716
|
+
});
|
|
717
|
+
test("scorecards refuse fabricated admission without executed reference, controls and held-out evidence", () => {
|
|
718
|
+
assert.throws(() => evalCandidateQualityScorecardV1({
|
|
719
|
+
candidate: "candidate",
|
|
720
|
+
revision: "revision",
|
|
721
|
+
checkpoints: {
|
|
722
|
+
admit: { verdict: "admitted", reasons: [] }
|
|
723
|
+
},
|
|
724
|
+
elapsedMs: 1,
|
|
725
|
+
receivedCalls: 0
|
|
726
|
+
}), /admission requires/u);
|
|
727
|
+
});
|
|
728
|
+
test("typed scientific resource exhaustion is an environment blockage, never progress", async (t) => {
|
|
729
|
+
const h = await harness(t);
|
|
730
|
+
const candidate = h.proposal.candidate.candidate;
|
|
731
|
+
await Effect.runPromise(h.invoke({
|
|
732
|
+
operationId: "resource-specification",
|
|
733
|
+
request: { tool: "draft_specification", args: { candidate } }
|
|
734
|
+
}));
|
|
735
|
+
const planned = await Effect.runPromise(h.invoke({
|
|
736
|
+
operationId: "resource-plan",
|
|
737
|
+
request: {
|
|
738
|
+
tool: "plan_environment",
|
|
739
|
+
args: {
|
|
740
|
+
candidate,
|
|
741
|
+
recipeIds: h.proposal.seed.environment.gradeRecipes.map((recipe) => recipe.id),
|
|
742
|
+
executionMode: "source",
|
|
743
|
+
targetRoots: ["packages/main"],
|
|
744
|
+
requiredArtifacts: [],
|
|
745
|
+
justification: "Exercise typed command resource classification."
|
|
746
|
+
}
|
|
747
|
+
}
|
|
748
|
+
}));
|
|
749
|
+
const environment = planned.accepted[0];
|
|
750
|
+
const result = await Effect.runPromise(h.invoke({
|
|
751
|
+
operationId: "resource-qualification",
|
|
752
|
+
request: {
|
|
753
|
+
tool: "qualify_reference",
|
|
754
|
+
args: { candidate, environment }
|
|
755
|
+
}
|
|
756
|
+
}).pipe(Effect.provideService(RepositoryCommandExecutionPort, {
|
|
757
|
+
run: () => Effect.fail(new RepositoryFoundryError({
|
|
758
|
+
operation: "run-command",
|
|
759
|
+
detail: "scientific_command_resource_exhausted",
|
|
760
|
+
classification: "environment_blocked",
|
|
761
|
+
cause: { code: "scientific_command_resource_exhausted" }
|
|
762
|
+
}))
|
|
763
|
+
})));
|
|
764
|
+
assert.equal(result.outcome, "environment_blocked");
|
|
765
|
+
assert.equal(result.gradeExecuted, false);
|
|
766
|
+
assert.equal(result.evidenceSummary, "scientific_command_resource_exhausted");
|
|
767
|
+
});
|
|
768
|
+
test("reference qualification cannot run or retain evidence before specification acceptance", async (t) => {
|
|
769
|
+
const repo = await fixture(t);
|
|
770
|
+
const scripted = model(false, false);
|
|
771
|
+
const capabilities = await Effect.runPromise(makeEvalCapabilitiesV1({ ...repo, runtime }).pipe(Effect.provideService(RepositoryFoundryLanguageModel, scripted.inner)));
|
|
772
|
+
const candidate = repo.proposal.candidate.candidate;
|
|
773
|
+
const drafted = await Effect.runPromise(capabilities.invoke({
|
|
774
|
+
operationId: "unaccepted-specification",
|
|
775
|
+
request: { tool: "draft_specification", args: { candidate } }
|
|
776
|
+
}));
|
|
777
|
+
assert.equal(drafted.outcome, "progress", drafted.evidenceSummary);
|
|
778
|
+
const specification = capabilities.handles().spec;
|
|
779
|
+
assert.ok(specification, "the reviewable draft must remain retained");
|
|
780
|
+
const planned = await Effect.runPromise(capabilities.invoke({
|
|
781
|
+
operationId: "unaccepted-specification-plan",
|
|
782
|
+
request: {
|
|
783
|
+
tool: "plan_environment",
|
|
784
|
+
args: {
|
|
785
|
+
candidate,
|
|
786
|
+
recipeIds: repo.proposal.seed.environment.gradeRecipes.map((recipe) => recipe.id),
|
|
787
|
+
executionMode: "source",
|
|
788
|
+
targetRoots: ["packages/main"],
|
|
789
|
+
requiredArtifacts: [],
|
|
790
|
+
justification: "Prove reference execution is ordered after specification acceptance."
|
|
791
|
+
}
|
|
792
|
+
}
|
|
793
|
+
}));
|
|
794
|
+
let gradeExecutions = 0;
|
|
795
|
+
const blocked = await Effect.runPromise(capabilities
|
|
796
|
+
.invoke({
|
|
797
|
+
operationId: "unaccepted-specification-qualification",
|
|
798
|
+
request: {
|
|
799
|
+
tool: "qualify_reference",
|
|
800
|
+
args: { candidate, environment: planned.accepted[0] }
|
|
801
|
+
}
|
|
802
|
+
})
|
|
803
|
+
.pipe(Effect.provideService(RepositoryCommandExecutionPort, {
|
|
804
|
+
run: () => {
|
|
805
|
+
gradeExecutions += 1;
|
|
806
|
+
return Effect.succeed({ exitCode: 0, stdout: "", stderr: "" });
|
|
807
|
+
}
|
|
808
|
+
})));
|
|
809
|
+
assert.equal(blocked.outcome, "progress");
|
|
810
|
+
assert.match(blocked.evidenceSummary, /reference qualification requires an accepted grounded specification review/u);
|
|
811
|
+
assert.equal(gradeExecutions, 0);
|
|
812
|
+
assert.equal(capabilities.handles().spec, specification);
|
|
813
|
+
assert.equal(capabilities.handles().seed, undefined);
|
|
814
|
+
assert.ok(blocked.allowedNext.includes("draft_specification"));
|
|
815
|
+
assert.equal(blocked.allowedNext.includes("qualify_reference"), false);
|
|
816
|
+
assert.equal(blocked.allowedNext.includes("author_oracle"), false);
|
|
817
|
+
assert.equal(blocked.allowedNext.includes("generate_controls"), false);
|
|
818
|
+
assert.equal(blocked.allowedNext.includes("request_admission"), false);
|
|
819
|
+
assert.equal((await readdir(join(repo.stateRoot, "work")).catch((error) => error.code === "ENOENT" ? [] : Promise.reject(error))).includes("reference-qualification.json"), false);
|
|
820
|
+
const revised = await Effect.runPromise(capabilities.invoke({
|
|
821
|
+
operationId: "revise-unaccepted-specification",
|
|
822
|
+
request: { tool: "draft_specification", args: { candidate } }
|
|
823
|
+
}));
|
|
824
|
+
assert.equal(revised.outcome, "progress", revised.evidenceSummary);
|
|
825
|
+
});
|
|
826
|
+
test("qualification slice progress is materialized as immutable accepted evidence", async (t) => {
|
|
827
|
+
const repo = await fixture(t);
|
|
828
|
+
const scripted = model();
|
|
829
|
+
const baseStore = makeCaseCheckpointStoreV1({ caseDirectory: repo.stateRoot });
|
|
830
|
+
let qualificationCheckpointed = false;
|
|
831
|
+
const store = {
|
|
832
|
+
...baseStore,
|
|
833
|
+
writeWorkCheckpoint: (name, value) => baseStore.writeWorkCheckpoint(name, value).pipe(Effect.tap(() => Effect.sync(() => {
|
|
834
|
+
if (name === "reference-qualification")
|
|
835
|
+
qualificationCheckpointed = true;
|
|
836
|
+
})))
|
|
837
|
+
};
|
|
838
|
+
const capabilities = await Effect.runPromise(makeEvalCapabilitiesV1({ ...repo, runtime, store }).pipe(Effect.provideService(RepositoryFoundryLanguageModel, scripted.inner)));
|
|
839
|
+
const candidate = repo.proposal.candidate.candidate;
|
|
840
|
+
await Effect.runPromise(capabilities.invoke({
|
|
841
|
+
operationId: "progress-specification",
|
|
842
|
+
request: { tool: "draft_specification", args: { candidate } }
|
|
843
|
+
}));
|
|
844
|
+
const planned = await Effect.runPromise(capabilities.invoke({
|
|
845
|
+
operationId: "progress-plan",
|
|
846
|
+
request: {
|
|
847
|
+
tool: "plan_environment",
|
|
848
|
+
args: {
|
|
849
|
+
candidate,
|
|
850
|
+
recipeIds: repo.proposal.seed.environment.gradeRecipes.map((recipe) => recipe.id),
|
|
851
|
+
executionMode: "source",
|
|
852
|
+
targetRoots: ["packages/main"],
|
|
853
|
+
requiredArtifacts: [],
|
|
854
|
+
justification: "Exercise durable qualification progress handoff."
|
|
855
|
+
}
|
|
856
|
+
}
|
|
857
|
+
}));
|
|
858
|
+
const result = await Effect.runPromise(capabilities.invoke({
|
|
859
|
+
operationId: "progress-qualification",
|
|
860
|
+
maximumWallMs: 2_000,
|
|
861
|
+
request: {
|
|
862
|
+
tool: "qualify_reference",
|
|
863
|
+
args: { candidate, environment: planned.accepted[0] }
|
|
864
|
+
}
|
|
865
|
+
}).pipe(Effect.provideService(RepositoryCommandExecutionPort, {
|
|
866
|
+
run: (input) => qualificationCheckpointed
|
|
867
|
+
? Effect.never
|
|
868
|
+
: Effect.succeed({
|
|
869
|
+
exitCode: 0,
|
|
870
|
+
stdout: input.args.includes("--test")
|
|
871
|
+
? "TAP version 13\n# Subtest: baseline\nok 1 - baseline\n# tests 1\n# pass 1\n# fail 0\n# cancelled 0\n# skipped 0\n# todo 0\n"
|
|
872
|
+
: "",
|
|
873
|
+
stderr: ""
|
|
874
|
+
})
|
|
875
|
+
})));
|
|
876
|
+
assert.equal(result.outcome, "progress", result.evidenceSummary);
|
|
877
|
+
const handle = result.accepted.find((accepted) => accepted.startsWith("qualification-progress-"));
|
|
878
|
+
assert.ok(handle, "the bounded result must accept its retained qualification progress");
|
|
879
|
+
assert.ok(result.progressReceipt?.acceptedEvidenceAdded.includes(handle), "Cloud reconstructs accepted continuation state from receipt deltas, not result.accepted");
|
|
880
|
+
assert.ok(result.progressReceipt?.resultingFingerprint.continuationProgressDigest);
|
|
881
|
+
const call = JSON.parse(await readFile(join(repo.stateRoot, "calls", `${handle}.attempt-1.json`), "utf8"));
|
|
882
|
+
assert.equal(call.request.tool, "qualify_reference");
|
|
883
|
+
assert.equal(call.request.candidateRevision, repo.proposal.candidate.revision);
|
|
884
|
+
assert.ok(call.response, "the immutable accepted artifact must contain retained progress");
|
|
885
|
+
});
|
|
886
|
+
test("oracle authoring continuation survives a capability process restart without claiming a scientific gate", async (t) => {
|
|
887
|
+
const h = await harness(t);
|
|
888
|
+
const candidate = h.proposal.candidate.candidate;
|
|
889
|
+
const invoke = async (request) => {
|
|
890
|
+
const result = await Effect.runPromise(h.invoke({
|
|
891
|
+
operationId: `grounded-prepare-${request.tool}`, maximumWallMs: 120_000, request
|
|
892
|
+
}));
|
|
893
|
+
assert.equal(result.outcome, "completed", result.evidenceSummary);
|
|
894
|
+
return result;
|
|
895
|
+
};
|
|
896
|
+
await invoke({ tool: "draft_specification", args: { candidate } });
|
|
897
|
+
const plan = await invoke({
|
|
898
|
+
tool: "plan_environment",
|
|
899
|
+
args: {
|
|
900
|
+
candidate, recipeIds: h.proposal.seed.environment.gradeRecipes.map((recipe) => recipe.id),
|
|
901
|
+
executionMode: "source", targetRoots: ["packages/main"], requiredArtifacts: [],
|
|
902
|
+
justification: "Exercise durable authoring continuation."
|
|
903
|
+
}
|
|
904
|
+
});
|
|
905
|
+
await invoke({ tool: "qualify_reference", args: { candidate, environment: plan.accepted[0] } });
|
|
906
|
+
const specification = h.capabilities.handles().spec;
|
|
907
|
+
const store = makeCaseCheckpointStoreV1({ caseDirectory: h.stateRoot });
|
|
908
|
+
const priorUnits = [];
|
|
909
|
+
for (let index = 0; index < 40; index += 1) {
|
|
910
|
+
const name = `grounded-authoring-fixture-retained-${String(index)}`;
|
|
911
|
+
const value = { version: 1, bindingDigest: "a".repeat(64), results: [], ordinal: index };
|
|
912
|
+
await Effect.runPromise(store.writeWorkCheckpoint(name, value));
|
|
913
|
+
priorUnits.push({ name, stage: "oracle", contentDigest: acceptedScientificDigestV1(value) });
|
|
914
|
+
}
|
|
915
|
+
await Effect.runPromise(store.writeWorkCheckpoint("grounded-continuations", priorUnits));
|
|
916
|
+
const interruptedRuntime = await Effect.runPromise(makeEvalCapabilitiesV1({
|
|
917
|
+
...h, runtime, environment: h.capabilities.environment()
|
|
918
|
+
}).pipe(Effect.provideService(RepositoryFoundryLanguageModel, h.scripted.inner)));
|
|
919
|
+
const partial = await Effect.runPromise(interruptedRuntime.invoke({
|
|
920
|
+
operationId: "grounded-interrupted", maximumWallMs: 1_000,
|
|
921
|
+
request: { tool: "author_oracle", args: { candidate, specification } }
|
|
922
|
+
}).pipe(Effect.provideService(RepositoryCommandExecutionPort, { run: () => Effect.never })));
|
|
923
|
+
assert.equal(partial.outcome, "progress", partial.evidenceSummary);
|
|
924
|
+
assert.equal(h.capabilities.handles().oracle, undefined);
|
|
925
|
+
const continuation = partial.accepted.find((handle) => handle.startsWith("continuation-progress-grounded-"));
|
|
926
|
+
assert.ok(continuation);
|
|
927
|
+
assert.equal(partial.accepted.length, 1, "one manifest, not an unbounded handle per fixture");
|
|
928
|
+
const manifest = JSON.parse(await readFile(join(h.stateRoot, "calls", `${continuation}.attempt-1.json`), "utf8")).response;
|
|
929
|
+
assert.equal(manifest.checkpoints.length, 42, "all retained units plus one bounded slice marker bind one continuation");
|
|
930
|
+
assert.ok(partial.progressReceipt?.acceptedEvidenceAdded.includes(continuation));
|
|
931
|
+
assert.equal(partial.progressReceipt?.sameFingerprintCount, 0);
|
|
932
|
+
assert.equal(partial.progressReceipt?.remaining.phaseAttempts, 12, "a productive authoring slice does not consume a new scientific attempt");
|
|
933
|
+
assert.equal(partial.progressReceipt?.gatesCompleted.includes("oracle_authored"), false);
|
|
934
|
+
assert.equal(partial.progressReceipt?.previousFingerprint?.acceptedEvidenceDigest, partial.progressReceipt?.resultingFingerprint.acceptedEvidenceDigest, "retained work is not completed scientific evidence");
|
|
935
|
+
const receivedCalls = h.scripted.calls.length;
|
|
936
|
+
const environment = JSON.parse(await readFile(join(h.stateRoot, "calls", `${plan.accepted[0]}.attempt-1.json`), "utf8")).response;
|
|
937
|
+
const restored = await Effect.runPromise(makeEvalCapabilitiesV1({
|
|
938
|
+
...h, runtime, environment,
|
|
939
|
+
progressReceipts: partial.progressReceipt === undefined ? [] : [partial.progressReceipt]
|
|
940
|
+
}).pipe(Effect.provideService(RepositoryFoundryLanguageModel, h.scripted.inner)));
|
|
941
|
+
const completed = await Effect.runPromise(restored.invoke({
|
|
942
|
+
operationId: "grounded-resumed", maximumWallMs: 120_000,
|
|
943
|
+
request: { tool: "author_oracle", args: { candidate, specification } }
|
|
944
|
+
}));
|
|
945
|
+
assert.equal(completed.outcome, "completed", completed.evidenceSummary);
|
|
946
|
+
assert.equal(h.scripted.calls.length, receivedCalls, "resume pending host replay, not the model");
|
|
947
|
+
assert.ok(restored.handles().oracle);
|
|
948
|
+
assert.ok(completed.progressReceipt?.gatesCompleted.includes("oracle_authored"));
|
|
949
|
+
assert.ok(completed.progressReceipt?.acceptedEvidenceRemoved.includes(continuation));
|
|
950
|
+
assert.deepEqual(completed.progressReceipt?.previousFingerprint, partial.progressReceipt?.resultingFingerprint);
|
|
951
|
+
});
|
|
952
|
+
test("blind control response survives a slice before validation without claiming control admission", async (t) => {
|
|
953
|
+
const h = await harness(t);
|
|
954
|
+
const candidate = h.proposal.candidate.candidate;
|
|
955
|
+
const invoke = async (request) => {
|
|
956
|
+
const result = await Effect.runPromise(h.invoke({
|
|
957
|
+
operationId: `feedback-prepare-${request.tool}`, maximumWallMs: 120_000, request
|
|
958
|
+
}));
|
|
959
|
+
assert.equal(result.outcome, "completed", result.evidenceSummary);
|
|
960
|
+
return result;
|
|
961
|
+
};
|
|
962
|
+
await invoke({ tool: "draft_specification", args: { candidate } });
|
|
963
|
+
const plan = await invoke({
|
|
964
|
+
tool: "plan_environment",
|
|
965
|
+
args: {
|
|
966
|
+
candidate, recipeIds: h.proposal.seed.environment.gradeRecipes.map((recipe) => recipe.id),
|
|
967
|
+
executionMode: "source", targetRoots: ["packages/main"], requiredArtifacts: [],
|
|
968
|
+
justification: "Retain blind control authoring across its execution slice."
|
|
969
|
+
}
|
|
970
|
+
});
|
|
971
|
+
const environment = plan.accepted[0];
|
|
972
|
+
await invoke({ tool: "qualify_reference", args: { candidate, environment } });
|
|
973
|
+
const specification = h.capabilities.handles().spec;
|
|
974
|
+
await invoke({ tool: "author_oracle", args: { candidate, specification } });
|
|
975
|
+
const request = {
|
|
976
|
+
tool: "generate_controls",
|
|
977
|
+
args: { candidate, specification, environment, kind: "valid" }
|
|
978
|
+
};
|
|
979
|
+
const partial = await Effect.runPromise(h.invoke({
|
|
980
|
+
operationId: "feedback-interrupted", maximumWallMs: 1_000, request
|
|
981
|
+
}).pipe(Effect.provideService(RepositoryCommandExecutionPort, { run: () => Effect.never })));
|
|
982
|
+
assert.equal(partial.outcome, "progress", partial.evidenceSummary);
|
|
983
|
+
assert.equal(h.capabilities.handles().controls, undefined);
|
|
984
|
+
assert.equal(partial.progressReceipt?.gatesCompleted.includes("valid_controls_generated"), false);
|
|
985
|
+
assert.equal(partial.accepted.length, 1);
|
|
986
|
+
assert.ok(partial.progressReceipt?.resultingFingerprint.continuationProgressDigest);
|
|
987
|
+
const calls = h.scripted.calls.length;
|
|
988
|
+
const restored = await Effect.runPromise(makeEvalCapabilitiesV1({
|
|
989
|
+
...h, runtime, environment: h.capabilities.environment(),
|
|
990
|
+
progressReceipts: partial.progressReceipt === undefined ? [] : [partial.progressReceipt]
|
|
991
|
+
}).pipe(Effect.provideService(RepositoryFoundryLanguageModel, h.scripted.inner)));
|
|
992
|
+
const completed = await Effect.runPromise(restored.invoke({
|
|
993
|
+
operationId: "feedback-resumed", maximumWallMs: 120_000, request
|
|
994
|
+
}));
|
|
995
|
+
assert.equal(completed.outcome, "completed", completed.evidenceSummary);
|
|
996
|
+
assert.equal(h.scripted.calls.length, calls, "resume executes the received control, not the model");
|
|
997
|
+
assert.ok(completed.progressReceipt?.gatesCompleted.includes("valid_controls_generated"));
|
|
998
|
+
assert.equal(completed.progressReceipt?.gatesCompleted.includes("wrong_controls_generated"), false);
|
|
999
|
+
const controls = JSON.parse(await readFile(join(h.stateRoot, "controls.json"), "utf8"));
|
|
1000
|
+
const checkpoints = {
|
|
1001
|
+
seed: JSON.parse(await readFile(join(h.stateRoot, "seed.json"), "utf8")),
|
|
1002
|
+
spec: JSON.parse(await readFile(join(h.stateRoot, "spec.json"), "utf8")),
|
|
1003
|
+
oracle: JSON.parse(await readFile(join(h.stateRoot, "oracle.json"), "utf8")),
|
|
1004
|
+
controls: { ...controls, valid: [], wrong: [] }
|
|
1005
|
+
};
|
|
1006
|
+
for (const [replacementRound, expectedNewCalls] of [[1, 1], [1, 1], [2, 2]]) {
|
|
1007
|
+
await Effect.runPromise(runPipelineControlsCapabilityV1({
|
|
1008
|
+
context: h.context, map: h.map, checkpoints, kind: "valid", replacementRound,
|
|
1009
|
+
workCheckpointStore: makeCaseCheckpointStoreV1({ caseDirectory: h.stateRoot }),
|
|
1010
|
+
checkpoint: () => Effect.void
|
|
1011
|
+
}).pipe(Effect.provideService(RepositoryFoundryLanguageModel, h.scripted.inner)));
|
|
1012
|
+
assert.equal(h.scripted.calls.length, calls + expectedNewCalls, "same replacement round resumes; a new scientific rejection requires a fresh proposal");
|
|
1013
|
+
}
|
|
1014
|
+
});
|
|
1015
|
+
test("individual capability chain admits only after real reference, controls, tournament and fresh held-out execution", async (t) => {
|
|
1016
|
+
const h = await harness(t);
|
|
1017
|
+
let operation = 0;
|
|
1018
|
+
const invoke = async (request) => {
|
|
1019
|
+
const result = await Effect.runPromise(h.invoke({ operationId: `chain-${String(++operation)}`, maximumWallMs: 120_000, request }));
|
|
1020
|
+
assert.equal(result.outcome, "completed", `${request.tool}: ${result.evidenceSummary}`);
|
|
1021
|
+
return result;
|
|
1022
|
+
};
|
|
1023
|
+
const candidate = h.proposal.candidate.candidate;
|
|
1024
|
+
await invoke({ tool: "draft_specification", args: { candidate } });
|
|
1025
|
+
const planned = await invoke({
|
|
1026
|
+
tool: "plan_environment",
|
|
1027
|
+
args: {
|
|
1028
|
+
candidate,
|
|
1029
|
+
recipeIds: h.proposal.seed.environment.gradeRecipes.map((recipe) => recipe.id),
|
|
1030
|
+
executionMode: "source",
|
|
1031
|
+
targetRoots: ["packages/main"],
|
|
1032
|
+
requiredArtifacts: [],
|
|
1033
|
+
justification: "Execute source recipe without compilation."
|
|
1034
|
+
}
|
|
1035
|
+
});
|
|
1036
|
+
const environment = planned.accepted[0];
|
|
1037
|
+
const qualified = await invoke({ tool: "qualify_reference", args: { candidate, environment } });
|
|
1038
|
+
assert.equal(qualified.gradeExecuted, true);
|
|
1039
|
+
const specification = h.capabilities.handles().spec;
|
|
1040
|
+
const authored = await invoke({ tool: "author_oracle", args: { candidate, specification } });
|
|
1041
|
+
const oracleHandle = h.capabilities.handles().oracle;
|
|
1042
|
+
const oracleCheckpoint = await readFile(join(h.stateRoot, "oracle.json"), "utf8");
|
|
1043
|
+
const oracleAuthorCalls = h.scripted.calls.length;
|
|
1044
|
+
for (let repetition = 1; repetition <= 2; repetition += 1) {
|
|
1045
|
+
const resumed = await invoke({ tool: "author_oracle", args: { candidate, specification } });
|
|
1046
|
+
assert.equal(h.capabilities.handles().oracle, oracleHandle);
|
|
1047
|
+
assert.equal(await readFile(join(h.stateRoot, "oracle.json"), "utf8"), oracleCheckpoint);
|
|
1048
|
+
assert.deepEqual(resumed.progressReceipt?.resultingFingerprint, authored.progressReceipt?.resultingFingerprint);
|
|
1049
|
+
assert.deepEqual(resumed.progressReceipt?.acceptedEvidenceAdded, []);
|
|
1050
|
+
assert.deepEqual(resumed.progressReceipt?.acceptedEvidenceRemoved, []);
|
|
1051
|
+
assert.equal(resumed.progressReceipt?.sameFingerprintCount, repetition);
|
|
1052
|
+
assert.equal(h.scripted.calls.length, oracleAuthorCalls);
|
|
1053
|
+
}
|
|
1054
|
+
const legacyCheckpoints = {
|
|
1055
|
+
seed: JSON.parse(await readFile(join(h.stateRoot, "seed.json"), "utf8")),
|
|
1056
|
+
spec: JSON.parse(await readFile(join(h.stateRoot, "spec.json"), "utf8")),
|
|
1057
|
+
oracle: { ...JSON.parse(oracleCheckpoint), coverageRounds: 1 }
|
|
1058
|
+
};
|
|
1059
|
+
let migrations = 0;
|
|
1060
|
+
const migrate = (checkpoints) => Effect.runPromise(runPipelineOracleStageV1({
|
|
1061
|
+
context: h.context, map: h.map, checkpoints,
|
|
1062
|
+
checkpoint: () => Effect.sync(() => { migrations += 1; })
|
|
1063
|
+
}).pipe(Effect.provideService(RepositoryFoundryLanguageModel, h.scripted.inner)));
|
|
1064
|
+
const migrated = await migrate(legacyCheckpoints);
|
|
1065
|
+
assert.equal(migrated.coverageRounds, 0);
|
|
1066
|
+
assert.deepEqual(migrated.scopeCoverage, JSON.parse(oracleCheckpoint).scopeCoverage);
|
|
1067
|
+
assert.equal(await migrate({ ...legacyCheckpoints, oracle: migrated }), migrated);
|
|
1068
|
+
assert.equal(migrations, 1, "legacy coverage normalization happens exactly once");
|
|
1069
|
+
await invoke({
|
|
1070
|
+
tool: "generate_controls",
|
|
1071
|
+
args: { candidate, specification, environment, kind: "valid" }
|
|
1072
|
+
});
|
|
1073
|
+
const previousControls = h.capabilities.handles().controls;
|
|
1074
|
+
const cumulative = await invoke({
|
|
1075
|
+
tool: "generate_controls",
|
|
1076
|
+
args: { candidate, specification, environment, kind: "wrong" }
|
|
1077
|
+
});
|
|
1078
|
+
assert.match(cumulative.evidenceSummary, /cumulative controls checkpoint; valid=1; wrong=1/u);
|
|
1079
|
+
assert.match(cumulative.evidenceSummary, /supersedes all previous controls snapshots/u);
|
|
1080
|
+
const callsBeforeStaleSnapshot = h.scripted.calls.length;
|
|
1081
|
+
const stale = await Effect.runPromise(h.invoke({
|
|
1082
|
+
operationId: "chain-stale-controls",
|
|
1083
|
+
request: {
|
|
1084
|
+
tool: "evaluate_oracle",
|
|
1085
|
+
args: {
|
|
1086
|
+
candidate, oracle: h.capabilities.handles().oracle,
|
|
1087
|
+
controls: [previousControls, h.capabilities.handles().controls]
|
|
1088
|
+
}
|
|
1089
|
+
}
|
|
1090
|
+
}));
|
|
1091
|
+
assert.equal(stale.outcome, "progress");
|
|
1092
|
+
assert.match(stale.evidenceSummary, /use only the latest cumulative controls checkpoint/u);
|
|
1093
|
+
assert.match(stale.evidenceSummary, /unaccepted or stale controls handle/u);
|
|
1094
|
+
assert.equal(h.scripted.calls.length, callsBeforeStaleSnapshot, "stale snapshots are rejected before any additional model or scientific work");
|
|
1095
|
+
assert.equal(h.capabilities.handles().tournament, undefined);
|
|
1096
|
+
const evaluated = await invoke({
|
|
1097
|
+
tool: "evaluate_oracle",
|
|
1098
|
+
args: {
|
|
1099
|
+
candidate,
|
|
1100
|
+
oracle: h.capabilities.handles().oracle,
|
|
1101
|
+
controls: [h.capabilities.handles().controls]
|
|
1102
|
+
}
|
|
1103
|
+
});
|
|
1104
|
+
assert.ok(evaluated.allowedNext.includes("freeze_candidate"));
|
|
1105
|
+
assert.equal(evaluated.allowedNext.includes("run_held_out_challenge"), false);
|
|
1106
|
+
const tournamentPath = join(h.stateRoot, "tournament.json");
|
|
1107
|
+
const retainedTournament = await readFile(tournamentPath, "utf8");
|
|
1108
|
+
const legacyTournament = JSON.parse(retainedTournament);
|
|
1109
|
+
await writeFile(tournamentPath, JSON.stringify({
|
|
1110
|
+
...legacyTournament,
|
|
1111
|
+
oracle: {
|
|
1112
|
+
...legacyTournament.oracle,
|
|
1113
|
+
observations: legacyTournament.oracle.observations.filter((row) => !(row.solutionId === "historical-defect" && row.clauseId === "fixture-negative"))
|
|
1114
|
+
}
|
|
1115
|
+
}));
|
|
1116
|
+
try {
|
|
1117
|
+
const restoredLegacy = await Effect.runPromise(makeEvalCapabilitiesV1({
|
|
1118
|
+
...h, runtime, environment: h.capabilities.environment()
|
|
1119
|
+
}).pipe(Effect.provideService(RepositoryFoundryLanguageModel, h.scripted.inner)));
|
|
1120
|
+
const legacyProgress = await Effect.runPromise(restoredLegacy.invoke({
|
|
1121
|
+
operationId: "legacy-scope-progress", request: { tool: "get_progress", args: { candidate } }
|
|
1122
|
+
}));
|
|
1123
|
+
assert.equal(legacyProgress.allowedNext.includes("freeze_candidate"), false, "a legacy adequate flag cannot replace missing executed scope evidence");
|
|
1124
|
+
assert.equal(legacyProgress.allowedNext.includes("run_held_out_challenge"), false);
|
|
1125
|
+
}
|
|
1126
|
+
finally {
|
|
1127
|
+
await writeFile(tournamentPath, retainedTournament);
|
|
1128
|
+
}
|
|
1129
|
+
const frozen = await invoke({
|
|
1130
|
+
tool: "freeze_candidate",
|
|
1131
|
+
args: { candidate, revision: h.proposal.candidate.revision }
|
|
1132
|
+
});
|
|
1133
|
+
const frozenCandidate = frozen.accepted[0];
|
|
1134
|
+
assert.ok(frozen.allowedNext.includes("run_held_out_challenge"));
|
|
1135
|
+
const frozenPayload = JSON.parse(await readFile(join(h.stateRoot, "calls", `${frozenCandidate}.attempt-1.json`), "utf8"));
|
|
1136
|
+
const baseStore = makeCaseCheckpointStoreV1({ caseDirectory: h.stateRoot });
|
|
1137
|
+
let oracleCheckpointed = false;
|
|
1138
|
+
const store = {
|
|
1139
|
+
...baseStore,
|
|
1140
|
+
writeStage: (...args) => baseStore.writeStage(...args).pipe(Effect.tap(() => Effect.sync(() => {
|
|
1141
|
+
if (args[0] === "challenge" && typeof args[1] === "object" &&
|
|
1142
|
+
args[1] !== null && "agenticPendingOracle" in args[1])
|
|
1143
|
+
oracleCheckpointed = true;
|
|
1144
|
+
})))
|
|
1145
|
+
};
|
|
1146
|
+
const restoreChallenge = () => Effect.runPromise(makeEvalCapabilitiesV1({
|
|
1147
|
+
...h, runtime, store, environment: h.capabilities.environment(),
|
|
1148
|
+
acceptedScientificArtifacts: [{
|
|
1149
|
+
handle: frozenCandidate,
|
|
1150
|
+
kind: "frozen_candidate",
|
|
1151
|
+
contentDigest: acceptedScientificDigestV1(frozenPayload.response)
|
|
1152
|
+
}]
|
|
1153
|
+
}).pipe(Effect.provideService(RepositoryFoundryLanguageModel, h.scripted.inner)));
|
|
1154
|
+
const sliced = await restoreChallenge();
|
|
1155
|
+
const partial = await Effect.runPromise(sliced.invoke({
|
|
1156
|
+
operationId: "held-out-first-slice", maximumWallMs: 3_000,
|
|
1157
|
+
request: { tool: "run_held_out_challenge", args: { frozenCandidate } }
|
|
1158
|
+
}).pipe(Effect.provideService(RepositoryCommandExecutionPort, {
|
|
1159
|
+
run: (command) => oracleCheckpointed
|
|
1160
|
+
? Effect.never
|
|
1161
|
+
: Effect.tryPromise({
|
|
1162
|
+
try: async () => {
|
|
1163
|
+
const output = await exec(command.executable, [...command.args], {
|
|
1164
|
+
cwd: command.cwd,
|
|
1165
|
+
env: command.environment,
|
|
1166
|
+
timeout: command.timeoutMs
|
|
1167
|
+
});
|
|
1168
|
+
return { exitCode: 0, stdout: output.stdout, stderr: output.stderr };
|
|
1169
|
+
},
|
|
1170
|
+
catch: (cause) => new RepositoryFoundryError({
|
|
1171
|
+
operation: "run-command", detail: "fixture reference command failed", cause
|
|
1172
|
+
})
|
|
1173
|
+
})
|
|
1174
|
+
})));
|
|
1175
|
+
assert.equal(partial.outcome, "progress", partial.evidenceSummary);
|
|
1176
|
+
assert.ok(oracleCheckpointed, "first completed solution must be saved before interruption");
|
|
1177
|
+
const continuation = partial.accepted.find((handle) => handle.startsWith("continuation-progress-challenge-"));
|
|
1178
|
+
assert.ok(continuation);
|
|
1179
|
+
assert.ok(partial.progressReceipt?.acceptedEvidenceAdded.includes(continuation));
|
|
1180
|
+
assert.equal(partial.progressReceipt?.sameFingerprintCount, 0);
|
|
1181
|
+
assert.ok(partial.progressReceipt?.resultingFingerprint.continuationProgressDigest);
|
|
1182
|
+
assert.equal(partial.progressReceipt?.gatesCompleted.includes("held_out_challenge_passed"), false);
|
|
1183
|
+
const saved = await readFile(join(h.stateRoot, "challenge.json"), "utf8");
|
|
1184
|
+
const authoredCalls = h.scripted.calls.length;
|
|
1185
|
+
const interrupted = await restoreChallenge();
|
|
1186
|
+
const noNewWork = await Effect.runPromise(interrupted.invoke({
|
|
1187
|
+
operationId: "held-out-interrupted-resume", maximumWallMs: 200,
|
|
1188
|
+
authoritativeProgressReceipts: [partial.progressReceipt],
|
|
1189
|
+
request: { tool: "run_held_out_challenge", args: { frozenCandidate } }
|
|
1190
|
+
}).pipe(Effect.provideService(RepositoryCommandExecutionPort, {
|
|
1191
|
+
run: () => Effect.never
|
|
1192
|
+
})));
|
|
1193
|
+
assert.equal(noNewWork.outcome, "progress", noNewWork.evidenceSummary);
|
|
1194
|
+
assert.equal(await readFile(join(h.stateRoot, "challenge.json"), "utf8"), saved, "resumed materialization must not erase an already checkpointed oracle");
|
|
1195
|
+
assert.deepEqual(noNewWork.progressReceipt?.previousFingerprint, partial.progressReceipt?.resultingFingerprint, "serialized continuation identity is stable");
|
|
1196
|
+
assert.equal(h.scripted.calls.length, authoredCalls, "resume must never resample the adversary");
|
|
1197
|
+
const resumed = await restoreChallenge();
|
|
1198
|
+
const challenge = await Effect.runPromise(resumed.invoke({
|
|
1199
|
+
operationId: "held-out-complete",
|
|
1200
|
+
authoritativeProgressReceipts: [partial.progressReceipt, noNewWork.progressReceipt],
|
|
1201
|
+
request: { tool: "run_held_out_challenge", args: { frozenCandidate } }
|
|
1202
|
+
}));
|
|
1203
|
+
assert.equal(challenge.outcome, "completed", challenge.evidenceSummary);
|
|
1204
|
+
assert.ok(challenge.progressReceipt?.acceptedEvidenceRemoved.includes(continuation));
|
|
1205
|
+
assert.equal(challenge.progressReceipt?.resultingFingerprint.continuationProgressDigest, null);
|
|
1206
|
+
assert.ok(challenge.progressReceipt?.gatesCompleted.includes("held_out_challenge_passed"));
|
|
1207
|
+
const calls = h.scripted.calls.length;
|
|
1208
|
+
const restored = await Effect.runPromise(makeEvalCapabilitiesV1({ ...h, runtime, environment: h.capabilities.environment() }).pipe(Effect.provideService(RepositoryFoundryLanguageModel, h.scripted.inner)));
|
|
1209
|
+
const replay = await Effect.runPromise(restored.invoke({
|
|
1210
|
+
operationId: "restored-challenge",
|
|
1211
|
+
request: { tool: "run_held_out_challenge", args: { frozenCandidate } }
|
|
1212
|
+
}));
|
|
1213
|
+
assert.equal(replay.outcome, "completed", replay.evidenceSummary);
|
|
1214
|
+
assert.deepEqual(replay.accepted, challenge.accepted);
|
|
1215
|
+
assert.equal(h.scripted.calls.length, calls, "same frozen challenge is reused, never resampled");
|
|
1216
|
+
const admitted = await Effect.runPromise(restored.invoke({
|
|
1217
|
+
operationId: "restored-admission",
|
|
1218
|
+
request: {
|
|
1219
|
+
tool: "request_admission",
|
|
1220
|
+
args: { frozenCandidate, qualification: challenge.accepted[0] }
|
|
1221
|
+
}
|
|
1222
|
+
}));
|
|
1223
|
+
assert.equal(admitted.outcome, "completed", admitted.evidenceSummary);
|
|
1224
|
+
const scorecard = await Effect.runPromise(restored.scorecard);
|
|
1225
|
+
assert.equal(scorecard.disposition, "admitted");
|
|
1226
|
+
assert.ok(scorecard.reference.passed > 0);
|
|
1227
|
+
assert.ok(scorecard.validControls.passed > 0);
|
|
1228
|
+
assert.ok(scorecard.wrongControls.passed > 0);
|
|
1229
|
+
assert.ok(scorecard.heldOut.passed > 0);
|
|
1230
|
+
assert.deepEqual(scorecard.missingEvidence, []);
|
|
1231
|
+
assert.ok(h.observed.includes("started") && h.observed.includes("completed"));
|
|
1232
|
+
});
|
|
1233
|
+
test("a diagnostic multi-file fixture probe materializes helper and test overlays together", async (t) => {
|
|
1234
|
+
const repo = await fixture(t);
|
|
1235
|
+
const scripted = model();
|
|
1236
|
+
const suite = {
|
|
1237
|
+
version: 1,
|
|
1238
|
+
caseId: repo.context.caseId,
|
|
1239
|
+
fixtures: [
|
|
1240
|
+
{
|
|
1241
|
+
id: "helper-probe",
|
|
1242
|
+
kind: "metamorphic",
|
|
1243
|
+
expectationMode: "preserved",
|
|
1244
|
+
description: "Uses helper fixture.",
|
|
1245
|
+
expectedBehavior: "Helper and repository result agree.",
|
|
1246
|
+
source: "generated",
|
|
1247
|
+
testPath: "packages/main/src/index.test.js"
|
|
1248
|
+
}
|
|
1249
|
+
],
|
|
1250
|
+
overlays: [
|
|
1251
|
+
{
|
|
1252
|
+
path: "packages/main/src/helper.js",
|
|
1253
|
+
content: "export const expected = 25.5;",
|
|
1254
|
+
fixtureIds: []
|
|
1255
|
+
},
|
|
1256
|
+
{
|
|
1257
|
+
path: "packages/main/src/index.test.js",
|
|
1258
|
+
content: 'import assert from "node:assert/strict"; import test from "node:test"; import {clamp} from "./index.js"; import {expected} from "./helper.js"; test("uses helper", () => assert.equal(clamp(expected), expected));',
|
|
1259
|
+
fixtureIds: ["helper-probe"]
|
|
1260
|
+
}
|
|
1261
|
+
]
|
|
1262
|
+
};
|
|
1263
|
+
const capabilities = await Effect.runPromise(makeEvalCapabilitiesV1({
|
|
1264
|
+
...repo,
|
|
1265
|
+
runtime,
|
|
1266
|
+
resolveFixtureDraft: (handle) => handle === "helper-fixture"
|
|
1267
|
+
? Effect.succeed(suite)
|
|
1268
|
+
: Effect.fail(new RepositoryFoundryError({
|
|
1269
|
+
operation: "run-pipeline-stage",
|
|
1270
|
+
detail: "unknown fixture"
|
|
1271
|
+
}))
|
|
1272
|
+
}).pipe(Effect.provideService(RepositoryFoundryLanguageModel, scripted.inner)));
|
|
1273
|
+
const candidate = repo.proposal.candidate.candidate;
|
|
1274
|
+
const plan = await Effect.runPromise(capabilities.invoke({
|
|
1275
|
+
operationId: "multi-plan",
|
|
1276
|
+
request: {
|
|
1277
|
+
tool: "plan_environment",
|
|
1278
|
+
args: {
|
|
1279
|
+
candidate,
|
|
1280
|
+
recipeIds: repo.proposal.seed.environment.gradeRecipes.map((recipe) => recipe.id),
|
|
1281
|
+
executionMode: "source",
|
|
1282
|
+
targetRoots: ["packages/main"],
|
|
1283
|
+
requiredArtifacts: [],
|
|
1284
|
+
justification: "Probe helper and test as one workspace."
|
|
1285
|
+
}
|
|
1286
|
+
}
|
|
1287
|
+
}));
|
|
1288
|
+
assert.equal(plan.outcome, "completed", plan.evidenceSummary);
|
|
1289
|
+
const probed = await Effect.runPromise(capabilities.invoke({
|
|
1290
|
+
operationId: "multi-probe",
|
|
1291
|
+
request: {
|
|
1292
|
+
tool: "probe_fixture",
|
|
1293
|
+
args: { candidate, environment: plan.accepted[0], fixtureDraft: "helper-fixture" }
|
|
1294
|
+
}
|
|
1295
|
+
}));
|
|
1296
|
+
assert.equal(probed.outcome, "completed", probed.evidenceSummary);
|
|
1297
|
+
assert.equal(probed.gradeExecuted, true);
|
|
1298
|
+
assert.deepEqual(scripted.calls, []);
|
|
1299
|
+
});
|
|
1300
|
+
test("a failed frozen challenge is not resampled and repair consumes it as development evidence", async (t) => {
|
|
1301
|
+
const h = await harness(t, true);
|
|
1302
|
+
let operation = 0;
|
|
1303
|
+
const invoke = (request) => Effect.runPromise(h.invoke({ operationId: `fold-${String(++operation)}`, request }));
|
|
1304
|
+
const candidate = h.proposal.candidate.candidate;
|
|
1305
|
+
await invoke({ tool: "draft_specification", args: { candidate } });
|
|
1306
|
+
const planned = await invoke({
|
|
1307
|
+
tool: "plan_environment",
|
|
1308
|
+
args: {
|
|
1309
|
+
candidate,
|
|
1310
|
+
recipeIds: h.proposal.seed.environment.gradeRecipes.map((recipe) => recipe.id),
|
|
1311
|
+
executionMode: "source",
|
|
1312
|
+
targetRoots: ["packages/main"],
|
|
1313
|
+
requiredArtifacts: [],
|
|
1314
|
+
justification: "Source regression recipe."
|
|
1315
|
+
}
|
|
1316
|
+
});
|
|
1317
|
+
const environment = planned.accepted[0];
|
|
1318
|
+
await invoke({ tool: "qualify_reference", args: { candidate, environment } });
|
|
1319
|
+
const specification = h.capabilities.handles().spec;
|
|
1320
|
+
await invoke({ tool: "author_oracle", args: { candidate, specification } });
|
|
1321
|
+
await invoke({
|
|
1322
|
+
tool: "generate_controls",
|
|
1323
|
+
args: { candidate, specification, environment, kind: "valid" }
|
|
1324
|
+
});
|
|
1325
|
+
await invoke({
|
|
1326
|
+
tool: "generate_controls",
|
|
1327
|
+
args: { candidate, specification, environment, kind: "wrong" }
|
|
1328
|
+
});
|
|
1329
|
+
await invoke({
|
|
1330
|
+
tool: "evaluate_oracle",
|
|
1331
|
+
args: {
|
|
1332
|
+
candidate,
|
|
1333
|
+
oracle: h.capabilities.handles().oracle,
|
|
1334
|
+
controls: [h.capabilities.handles().controls]
|
|
1335
|
+
}
|
|
1336
|
+
});
|
|
1337
|
+
const frozen = await invoke({
|
|
1338
|
+
tool: "freeze_candidate",
|
|
1339
|
+
args: { candidate, revision: h.proposal.candidate.revision }
|
|
1340
|
+
});
|
|
1341
|
+
assert.equal(frozen.outcome, "completed", frozen.evidenceSummary);
|
|
1342
|
+
const frozenCandidate = frozen.accepted[0];
|
|
1343
|
+
const failed = await invoke({ tool: "run_held_out_challenge", args: { frozenCandidate } });
|
|
1344
|
+
assert.equal(failed.outcome, "scientific_rejection", failed.evidenceSummary);
|
|
1345
|
+
const calls = h.scripted.calls.length;
|
|
1346
|
+
const retried = await invoke({ tool: "run_held_out_challenge", args: { frozenCandidate } });
|
|
1347
|
+
assert.equal(retried.outcome, "scientific_rejection");
|
|
1348
|
+
assert.deepEqual(retried.accepted, failed.accepted);
|
|
1349
|
+
assert.equal(h.scripted.calls.length, calls);
|
|
1350
|
+
const repaired = await invoke({
|
|
1351
|
+
tool: "repair_oracle",
|
|
1352
|
+
args: { candidate, oracle: h.capabilities.handles().oracle, findings: failed.diagnostics }
|
|
1353
|
+
});
|
|
1354
|
+
assert.equal(repaired.outcome, "progress", "the valid disguised adversary is a measured repair finding, not admission");
|
|
1355
|
+
const store = makeCaseCheckpointStoreV1({ caseDirectory: h.stateRoot });
|
|
1356
|
+
const controls = JSON.parse(await readFile(store.stagePath("controls"), "utf8"));
|
|
1357
|
+
assert.ok(controls.wrong.some((control) => control.id === "heldout-truncation"));
|
|
1358
|
+
await assert.rejects(readFile(store.stagePath("challenge")), /ENOENT/u);
|
|
1359
|
+
const old = await invoke({ tool: "run_held_out_challenge", args: { frozenCandidate } });
|
|
1360
|
+
assert.equal(old.outcome, "progress");
|
|
1361
|
+
assert.deepEqual(old.accepted, []);
|
|
1362
|
+
});
|
|
1363
|
+
test("restoring a seed never treats a legacy compile-only verdict as reference qualification", async (t) => {
|
|
1364
|
+
const h = await harness(t);
|
|
1365
|
+
const candidate = h.proposal.candidate.candidate;
|
|
1366
|
+
const invoke = async (request) => {
|
|
1367
|
+
const result = await Effect.runPromise(h.invoke({
|
|
1368
|
+
operationId: `legacy-seed-${request.tool}`, maximumWallMs: 120_000, request
|
|
1369
|
+
}));
|
|
1370
|
+
assert.equal(result.outcome, "completed", result.evidenceSummary);
|
|
1371
|
+
return result;
|
|
1372
|
+
};
|
|
1373
|
+
await invoke({ tool: "draft_specification", args: { candidate } });
|
|
1374
|
+
const planned = await invoke({
|
|
1375
|
+
tool: "plan_environment",
|
|
1376
|
+
args: {
|
|
1377
|
+
candidate,
|
|
1378
|
+
recipeIds: h.proposal.seed.environment.gradeRecipes.map((recipe) => recipe.id),
|
|
1379
|
+
executionMode: "source", targetRoots: ["packages/main"], requiredArtifacts: [],
|
|
1380
|
+
justification: "Execute the real source recipe."
|
|
1381
|
+
}
|
|
1382
|
+
});
|
|
1383
|
+
await invoke({ tool: "qualify_reference", args: { candidate, environment: planned.accepted[0] } });
|
|
1384
|
+
const path = join(h.stateRoot, "seed.json");
|
|
1385
|
+
const retained = JSON.parse(await readFile(path, "utf8"));
|
|
1386
|
+
await writeFile(path, JSON.stringify({
|
|
1387
|
+
...retained,
|
|
1388
|
+
seed: {
|
|
1389
|
+
...retained.seed,
|
|
1390
|
+
preChangeObservations: retained.seed.preChangeObservations.map((row) => ({
|
|
1391
|
+
...row, detail: "exit=2; stage=candidate-validation; gradeExecuted=false"
|
|
1392
|
+
}))
|
|
1393
|
+
}
|
|
1394
|
+
}));
|
|
1395
|
+
const callsBeforeRestore = h.scripted.calls.length;
|
|
1396
|
+
await assert.rejects(Effect.runPromise(makeEvalCapabilitiesV1({ ...h, runtime, environment: h.capabilities.environment() }).pipe(Effect.provideService(RepositoryFoundryLanguageModel, h.scripted.inner))), /retained seed checkpoint does not satisfy the executed behavioral evidence contract/u);
|
|
1397
|
+
assert.equal(h.scripted.calls.length, callsBeforeRestore);
|
|
1398
|
+
});
|
|
1399
|
+
for (const interruptDiscovery of [false, true]) {
|
|
1400
|
+
test(`blind controls use initial baseline when regression tests were added (empty discovery=${String(interruptDiscovery)})`, async (t) => {
|
|
1401
|
+
const h = await harness(t, false, 30, "packages/main/src/regression.test.js");
|
|
1402
|
+
const candidate = h.proposal.candidate.candidate;
|
|
1403
|
+
let operation = 0;
|
|
1404
|
+
const invoke = async (request) => {
|
|
1405
|
+
const result = await Effect.runPromise(h.invoke({
|
|
1406
|
+
operationId: `new-test-${String(++operation)}`, maximumWallMs: 120_000, request
|
|
1407
|
+
}));
|
|
1408
|
+
assert.equal(result.outcome, "completed", `${request.tool}: ${result.evidenceSummary}`);
|
|
1409
|
+
return result;
|
|
1410
|
+
};
|
|
1411
|
+
await invoke({ tool: "draft_specification", args: { candidate } });
|
|
1412
|
+
const plan = await invoke({
|
|
1413
|
+
tool: "plan_environment",
|
|
1414
|
+
args: {
|
|
1415
|
+
candidate,
|
|
1416
|
+
recipeIds: h.proposal.seed.environment.gradeRecipes.map(({ id }) => id),
|
|
1417
|
+
executionMode: "source", targetRoots: ["packages/main"], requiredArtifacts: [],
|
|
1418
|
+
justification: "Execute real baseline and newly added regression separately."
|
|
1419
|
+
}
|
|
1420
|
+
});
|
|
1421
|
+
const environment = plan.accepted[0];
|
|
1422
|
+
const prepared = h.capabilities.environment();
|
|
1423
|
+
assert.ok(prepared.environment.gradeRecipes[0].args.includes("src/regression.test.js"));
|
|
1424
|
+
assert.ok(prepared.environment.baselineGradeRecipes[0].args.includes("src/index.test.js"));
|
|
1425
|
+
await invoke({ tool: "qualify_reference", args: { candidate, environment } });
|
|
1426
|
+
const specification = h.capabilities.handles().spec;
|
|
1427
|
+
await invoke({ tool: "author_oracle", args: { candidate, specification } });
|
|
1428
|
+
let commands = 0;
|
|
1429
|
+
let falseSuccesses = 0;
|
|
1430
|
+
for (const kind of ["valid", "wrong"]) {
|
|
1431
|
+
const result = await Effect.runPromise(h.invoke({
|
|
1432
|
+
operationId: `controls-${kind}`, maximumWallMs: 120_000,
|
|
1433
|
+
request: { tool: "generate_controls", args: { candidate, specification, environment, kind } }
|
|
1434
|
+
}).pipe(Effect.provideService(RepositoryCommandExecutionPort, {
|
|
1435
|
+
run: (input) => Effect.gen(function* () {
|
|
1436
|
+
commands += 1;
|
|
1437
|
+
assert.ok(input.args.includes("src/index.test.js"));
|
|
1438
|
+
assert.ok(!input.args.includes("src/regression.test.js"), "blind sanity cannot require the absent reference test");
|
|
1439
|
+
if (interruptDiscovery && commands === 1) {
|
|
1440
|
+
falseSuccesses += 1;
|
|
1441
|
+
return { exitCode: 0, stdout: "TAP version 13\n1..0\n# tests 0\n# pass 0\n# fail 0\n# cancelled 0\n# skipped 0\n# todo 0\n", stderr: "" };
|
|
1442
|
+
}
|
|
1443
|
+
const result = yield* Effect.tryPromise({
|
|
1444
|
+
try: () => exec(input.executable, [...input.args], {
|
|
1445
|
+
cwd: input.cwd, env: input.environment, timeout: input.timeoutMs
|
|
1446
|
+
}),
|
|
1447
|
+
catch: (cause) => new RepositoryFoundryError({ operation: "run-command", detail: "fixture sanity command failed", cause })
|
|
1448
|
+
});
|
|
1449
|
+
return { exitCode: 0, stdout: result.stdout, stderr: result.stderr };
|
|
1450
|
+
})
|
|
1451
|
+
})));
|
|
1452
|
+
assert.equal(result.outcome, "completed", result.evidenceSummary);
|
|
1453
|
+
}
|
|
1454
|
+
assert.equal(commands, interruptDiscovery ? 3 : 2, "empty discovery must return feedback and require an actual passing replay");
|
|
1455
|
+
assert.equal(falseSuccesses, interruptDiscovery ? 1 : 0);
|
|
1456
|
+
const checkpoint = JSON.parse(await readFile(join(h.stateRoot, "controls.json"), "utf8"));
|
|
1457
|
+
assert.equal(checkpoint.valid.length, 1);
|
|
1458
|
+
assert.equal(checkpoint.wrong.length, 1);
|
|
1459
|
+
assert.equal(checkpoint.regenerations, interruptDiscovery ? 1 : 0);
|
|
1460
|
+
});
|
|
1461
|
+
}
|