@velum-labs/routekit-eval-setup 1.3.2 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/authoring-responses-request.d.ts +5 -0
- package/dist/adapters/authoring-responses-request.js +38 -0
- package/dist/adapters/evaluation-evidence-freshness.d.ts +7 -0
- package/dist/adapters/evaluation-evidence-freshness.js +76 -0
- package/dist/adapters/git-task-history.d.ts +67 -0
- package/dist/adapters/git-task-history.js +171 -0
- package/dist/adapters/integrated-repository-history.d.ts +21 -0
- package/dist/adapters/integrated-repository-history.js +175 -0
- package/dist/adapters/repository-command-diagnostic.d.ts +8 -0
- package/dist/adapters/repository-command-diagnostic.js +46 -0
- package/dist/adapters/repository-command-evidence.d.ts +13 -0
- package/dist/adapters/repository-command-evidence.js +102 -0
- package/dist/adapters/repository-command-runner.d.ts +238 -0
- package/dist/adapters/repository-command-runner.js +1483 -0
- package/dist/adapters/repository-import-context.d.ts +47 -0
- package/dist/adapters/repository-import-context.js +469 -0
- package/dist/adapters/repository-node-test-reporter.d.ts +3 -0
- package/dist/adapters/repository-node-test-reporter.js +27 -0
- package/dist/adapters/repository-review-evidence.d.ts +39 -0
- package/dist/adapters/repository-review-evidence.js +632 -0
- package/dist/adapters/repository-seed-selection.d.ts +7 -0
- package/dist/adapters/repository-seed-selection.js +79 -0
- package/dist/adapters/repository-solution-edits.d.ts +49 -0
- package/dist/adapters/repository-solution-edits.js +136 -0
- package/dist/adapters/repository-vitest-phase-adapter.d.ts +8 -0
- package/dist/adapters/repository-vitest-phase-adapter.js +310 -0
- package/dist/adapters/repository-vitest-reporter.d.ts +24 -0
- package/dist/adapters/repository-vitest-reporter.js +314 -0
- package/dist/adapters/strict-authoring-schema.d.ts +5 -0
- package/dist/adapters/strict-authoring-schema.js +158 -0
- package/dist/adapters/test-discovery.d.ts +30 -0
- package/dist/adapters/test-discovery.js +124 -0
- package/dist/adapters/typescript-repository-index.d.ts +51 -0
- package/dist/adapters/typescript-repository-index.js +226 -0
- package/dist/agentic-capabilities-protocol.d.ts +1373 -0
- package/dist/agentic-capabilities-protocol.js +786 -0
- package/dist/case-checkpoint-store.d.ts +29 -0
- package/dist/case-checkpoint-store.js +133 -0
- package/dist/case-pipeline-protocol-v2.d.ts +184 -0
- package/dist/case-pipeline-protocol-v2.js +193 -0
- package/dist/case-pipeline-protocol.d.ts +2626 -0
- package/dist/case-pipeline-protocol.js +371 -0
- package/dist/effect-api.d.ts +74 -10
- package/dist/effect-api.js +56 -6
- package/dist/errors.d.ts +31 -0
- package/dist/errors.js +10 -0
- package/dist/eval-capability-execution-envelope.d.ts +64 -0
- package/dist/eval-capability-execution-envelope.js +98 -0
- package/dist/eval-capability-policy.d.ts +90 -0
- package/dist/eval-capability-policy.js +107 -0
- package/dist/eval-event-log.d.ts +140 -0
- package/dist/eval-event-log.js +220 -0
- package/dist/evaluation-authoring-policy.d.ts +18 -0
- package/dist/evaluation-authoring-policy.js +19 -0
- package/dist/evaluation-authoring-validation.d.ts +22 -0
- package/dist/evaluation-authoring-validation.js +72 -0
- package/dist/evaluation-evidence.d.ts +20 -0
- package/dist/evaluation-evidence.js +319 -0
- package/dist/evaluation-grader-calibration-protocol.d.ts +108 -0
- package/dist/evaluation-grader-calibration-protocol.js +80 -0
- package/dist/evaluation-grader-calibration.d.ts +18 -0
- package/dist/evaluation-grader-calibration.js +334 -0
- package/dist/evaluation-grading-policy.d.ts +24 -0
- package/dist/evaluation-grading-policy.js +54 -0
- package/dist/evaluation-proposal-policy.d.ts +4 -0
- package/dist/evaluation-proposal-policy.js +91 -0
- package/dist/evaluation-source-retrieval.d.ts +68 -0
- package/dist/evaluation-source-retrieval.js +513 -0
- package/dist/evaluation-structure-policy.d.ts +29 -0
- package/dist/evaluation-structure-policy.js +138 -0
- package/dist/index.d.ts +124 -17
- package/dist/index.js +69 -11
- package/dist/inspection.js +2 -3
- package/dist/project-artifacts.d.ts +7 -2
- package/dist/project-artifacts.js +49 -136
- package/dist/project-authoring.d.ts +66 -5
- package/dist/project-authoring.js +783 -109
- package/dist/project-contracts.d.ts +419 -84
- package/dist/project-contracts.js +160 -52
- package/dist/project-store.js +2 -1
- package/dist/project-workflow.d.ts +5 -4
- package/dist/project-workflow.js +154 -35
- package/dist/repository-adversary-protocol.d.ts +64 -0
- package/dist/repository-adversary-protocol.js +105 -0
- package/dist/repository-behavior-protocol.d.ts +188 -0
- package/dist/repository-behavior-protocol.js +202 -0
- package/dist/repository-benchmark-protocol.d.ts +487 -0
- package/dist/repository-benchmark-protocol.js +96 -0
- package/dist/repository-execution-protocol.d.ts +150 -0
- package/dist/repository-execution-protocol.js +38 -0
- package/dist/repository-fixture-instructions.d.ts +3 -0
- package/dist/repository-fixture-instructions.js +91 -0
- package/dist/repository-fixture-protocol.d.ts +79 -0
- package/dist/repository-fixture-protocol.js +79 -0
- package/dist/repository-foundry-plan-protocol.d.ts +118 -0
- package/dist/repository-foundry-plan-protocol.js +296 -0
- package/dist/repository-foundry-progress-protocol.d.ts +52 -0
- package/dist/repository-foundry-progress-protocol.js +52 -0
- package/dist/repository-improvement-protocol.d.ts +100 -0
- package/dist/repository-improvement-protocol.js +106 -0
- package/dist/repository-language-model-protocol.d.ts +43 -0
- package/dist/repository-language-model-protocol.js +146 -0
- package/dist/repository-oracle-coverage-protocol.d.ts +18 -0
- package/dist/repository-oracle-coverage-protocol.js +39 -0
- package/dist/repository-oracle-execution-binding.d.ts +27 -0
- package/dist/repository-oracle-execution-binding.js +59 -0
- package/dist/repository-oracle-protocol.d.ts +230 -0
- package/dist/repository-oracle-protocol.js +156 -0
- package/dist/repository-oracle-scope-policy.d.ts +22 -0
- package/dist/repository-oracle-scope-policy.js +92 -0
- package/dist/repository-quality-policy.d.ts +15 -0
- package/dist/repository-quality-policy.js +357 -0
- package/dist/repository-routing-benchmark-protocol.d.ts +176 -0
- package/dist/repository-routing-benchmark-protocol.js +103 -0
- package/dist/repository-routing-model-protocol.d.ts +36 -0
- package/dist/repository-routing-model-protocol.js +89 -0
- package/dist/repository-routing-plan-protocol.d.ts +112 -0
- package/dist/repository-routing-plan-protocol.js +58 -0
- package/dist/repository-routing-quality-policy.d.ts +9 -0
- package/dist/repository-routing-quality-policy.js +191 -0
- package/dist/repository-seed-qualification-progress-protocol.d.ts +205 -0
- package/dist/repository-seed-qualification-progress-protocol.js +28 -0
- package/dist/repository-semantic-calibration-protocol.d.ts +768 -0
- package/dist/repository-semantic-calibration-protocol.js +276 -0
- package/dist/repository-semantic-calibration.d.ts +163 -0
- package/dist/repository-semantic-calibration.js +581 -0
- package/dist/repository-specification-contract-facts-protocol.d.ts +224 -0
- package/dist/repository-specification-contract-facts-protocol.js +276 -0
- package/dist/repository-specification-critique-protocol.d.ts +189 -0
- package/dist/repository-specification-critique-protocol.js +103 -0
- package/dist/repository-task-family-protocol.d.ts +24 -0
- package/dist/repository-task-family-protocol.js +37 -0
- package/dist/repository-task-seed-protocol.d.ts +384 -0
- package/dist/repository-task-seed-protocol.js +236 -0
- package/dist/repository-trajectory-protocol.d.ts +20 -0
- package/dist/repository-trajectory-protocol.js +42 -0
- package/dist/service.js +1 -1
- package/dist/services/adversary/service.d.ts +64 -0
- package/dist/services/adversary/service.js +330 -0
- package/dist/services/benchmark-compiler/service.d.ts +450 -0
- package/dist/services/benchmark-compiler/service.js +9 -0
- package/dist/services/budgeted-model/service.d.ts +118 -0
- package/dist/services/budgeted-model/service.js +460 -0
- package/dist/services/case-authoring/service.d.ts +163 -0
- package/dist/services/case-authoring/service.js +1456 -0
- package/dist/services/case-finalization/service.d.ts +283 -0
- package/dist/services/case-finalization/service.js +370 -0
- package/dist/services/case-generation/service.d.ts +619 -0
- package/dist/services/case-generation/service.js +2628 -0
- package/dist/services/case-pipeline/service.d.ts +31 -0
- package/dist/services/case-pipeline/service.js +485 -0
- package/dist/services/case-pipeline-v2/service.d.ts +70 -0
- package/dist/services/case-pipeline-v2/service.js +477 -0
- package/dist/services/command-observability/service.d.ts +13 -0
- package/dist/services/command-observability/service.js +3 -0
- package/dist/services/dimension-labeling/service.d.ts +77 -0
- package/dist/services/dimension-labeling/service.js +188 -0
- package/dist/services/eval-candidate/service.d.ts +208 -0
- package/dist/services/eval-candidate/service.js +64 -0
- package/dist/services/eval-capabilities/service.d.ts +183 -0
- package/dist/services/eval-capabilities/service.js +1433 -0
- package/dist/services/eval-environment/service.d.ts +173 -0
- package/dist/services/eval-environment/service.js +127 -0
- package/dist/services/evidence-reconstruction/service.d.ts +36 -0
- package/dist/services/evidence-reconstruction/service.js +145 -0
- package/dist/services/fixture-builder/service.d.ts +62 -0
- package/dist/services/fixture-builder/service.js +36 -0
- package/dist/services/fixture-validation/service.d.ts +75 -0
- package/dist/services/fixture-validation/service.js +295 -0
- package/dist/services/foundry/service.d.ts +831 -0
- package/dist/services/foundry/service.js +442 -0
- package/dist/services/foundry-progress/service.d.ts +62 -0
- package/dist/services/foundry-progress/service.js +149 -0
- package/dist/services/foundry-v2/service.d.ts +54 -0
- package/dist/services/foundry-v2/service.js +28 -0
- package/dist/services/grounded-authoring/service.d.ts +126 -0
- package/dist/services/grounded-authoring/service.js +822 -0
- package/dist/services/historical-case/service.d.ts +722 -0
- package/dist/services/historical-case/service.js +177 -0
- package/dist/services/improvement-loop/service.d.ts +59 -0
- package/dist/services/improvement-loop/service.js +176 -0
- package/dist/services/language-model/service.d.ts +52 -0
- package/dist/services/language-model/service.js +194 -0
- package/dist/services/oracle-builder/service.d.ts +146 -0
- package/dist/services/oracle-builder/service.js +513 -0
- package/dist/services/oracle-coverage/service.d.ts +28 -0
- package/dist/services/oracle-coverage/service.js +50 -0
- package/dist/services/oracle-coverage-witness/service.d.ts +130 -0
- package/dist/services/oracle-coverage-witness/service.js +538 -0
- package/dist/services/pipeline-challenge/service.d.ts +551 -0
- package/dist/services/pipeline-challenge/service.js +427 -0
- package/dist/services/pipeline-controls/service.d.ts +130 -0
- package/dist/services/pipeline-controls/service.js +483 -0
- package/dist/services/pipeline-oracle/service.d.ts +8 -0
- package/dist/services/pipeline-oracle/service.js +256 -0
- package/dist/services/pipeline-seed/service.d.ts +298 -0
- package/dist/services/pipeline-seed/service.js +428 -0
- package/dist/services/pipeline-spec/service.d.ts +103 -0
- package/dist/services/pipeline-spec/service.js +619 -0
- package/dist/services/pipeline-tournament/service.d.ts +258 -0
- package/dist/services/pipeline-tournament/service.js +476 -0
- package/dist/services/quality-gate/service.d.ts +233 -0
- package/dist/services/quality-gate/service.js +136 -0
- package/dist/services/repository-bundle/service.d.ts +33 -0
- package/dist/services/repository-bundle/service.js +114 -0
- package/dist/services/repository-model/service.d.ts +105 -0
- package/dist/services/repository-model/service.js +250 -0
- package/dist/services/repository-public-artifact/service.d.ts +133 -0
- package/dist/services/repository-public-artifact/service.js +330 -0
- package/dist/services/routing-benchmark/service.d.ts +362 -0
- package/dist/services/routing-benchmark/service.js +96 -0
- package/dist/services/specification-critic/service.d.ts +92 -0
- package/dist/services/specification-critic/service.js +172 -0
- package/dist/services/task-family/service.d.ts +40 -0
- package/dist/services/task-family/service.js +55 -0
- package/dist/services/task-seed/service.d.ts +906 -0
- package/dist/services/task-seed/service.js +1406 -0
- package/dist/services/task-specification/service.d.ts +27 -0
- package/dist/services/task-specification/service.js +40 -0
- package/dist/services/trajectory-policy/service.d.ts +110 -0
- package/dist/services/trajectory-policy/service.js +216 -0
- package/dist/test/agentic-capabilities-protocol.test.d.ts +1 -0
- package/dist/test/agentic-capabilities-protocol.test.js +570 -0
- package/dist/test/agentic-capabilities.test.d.ts +1 -0
- package/dist/test/agentic-capabilities.test.js +1461 -0
- package/dist/test/agentic-environment.test.d.ts +1 -0
- package/dist/test/agentic-environment.test.js +213 -0
- package/dist/test/case-pipeline-foundation.test.d.ts +1 -0
- package/dist/test/case-pipeline-foundation.test.js +535 -0
- package/dist/test/case-pipeline-protocol-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-protocol-v2.test.js +124 -0
- package/dist/test/case-pipeline-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-v2.test.js +286 -0
- package/dist/test/case-pipeline.test.d.ts +1 -0
- package/dist/test/case-pipeline.test.js +851 -0
- package/dist/test/eval-capability-policy.test.d.ts +1 -0
- package/dist/test/eval-capability-policy.test.js +50 -0
- package/dist/test/eval-event-log.test.d.ts +1 -0
- package/dist/test/eval-event-log.test.js +125 -0
- package/dist/test/evaluation-evidence-freshness.test.d.ts +1 -0
- package/dist/test/evaluation-evidence-freshness.test.js +44 -0
- package/dist/test/evaluation-evidence.test.d.ts +1 -0
- package/dist/test/evaluation-evidence.test.js +230 -0
- package/dist/test/evaluation-grader-calibration.test.d.ts +1 -0
- package/dist/test/evaluation-grader-calibration.test.js +373 -0
- package/dist/test/evaluation-proposal-digest.test.d.ts +1 -0
- package/dist/test/evaluation-proposal-digest.test.js +187 -0
- package/dist/test/evaluation-source-retrieval.test.d.ts +1 -0
- package/dist/test/evaluation-source-retrieval.test.js +237 -0
- package/dist/test/evaluation-structure-policy.test.d.ts +1 -0
- package/dist/test/evaluation-structure-policy.test.js +196 -0
- package/dist/test/fixtures/repository-resource-panel.d.ts +39 -0
- package/dist/test/fixtures/repository-resource-panel.js +111 -0
- package/dist/test/fixtures/vitest-boundary-panel.d.ts +84 -0
- package/dist/test/fixtures/vitest-boundary-panel.js +120 -0
- package/dist/test/fixtures/vitest-phase-panel.d.ts +135 -0
- package/dist/test/fixtures/vitest-phase-panel.js +213 -0
- package/dist/test/fixtures/vitest-reporter-results.d.ts +76 -0
- package/dist/test/fixtures/vitest-reporter-results.js +94 -0
- package/dist/test/grounded-authoring.test.d.ts +1 -0
- package/dist/test/grounded-authoring.test.js +565 -0
- package/dist/test/integrated-repository-history.test.d.ts +1 -0
- package/dist/test/integrated-repository-history.test.js +227 -0
- package/dist/test/project-authoring.test.js +593 -43
- package/dist/test/project-workflow.test.js +419 -40
- package/dist/test/repository-authoring-artifacts.test.d.ts +1 -0
- package/dist/test/repository-authoring-artifacts.test.js +185 -0
- package/dist/test/repository-bundle.test.d.ts +1 -0
- package/dist/test/repository-bundle.test.js +52 -0
- package/dist/test/repository-case-generation.test.d.ts +1 -0
- package/dist/test/repository-case-generation.test.js +3465 -0
- package/dist/test/repository-command-diagnostic.test.d.ts +1 -0
- package/dist/test/repository-command-diagnostic.test.js +55 -0
- package/dist/test/repository-command-signals.test.d.ts +1 -0
- package/dist/test/repository-command-signals.test.js +124 -0
- package/dist/test/repository-fixture-scope-coverage.test.d.ts +1 -0
- package/dist/test/repository-fixture-scope-coverage.test.js +127 -0
- package/dist/test/repository-fixture-validation.test.d.ts +1 -0
- package/dist/test/repository-fixture-validation.test.js +362 -0
- package/dist/test/repository-foundry-progress.test.d.ts +1 -0
- package/dist/test/repository-foundry-progress.test.js +110 -0
- package/dist/test/repository-foundry-quality.test.d.ts +1 -0
- package/dist/test/repository-foundry-quality.test.js +1138 -0
- package/dist/test/repository-import-context.test.d.ts +1 -0
- package/dist/test/repository-import-context.test.js +354 -0
- package/dist/test/repository-model-authoring.test.d.ts +1 -0
- package/dist/test/repository-model-authoring.test.js +544 -0
- package/dist/test/repository-model.test.d.ts +1 -0
- package/dist/test/repository-model.test.js +2195 -0
- package/dist/test/repository-node-test-reporter.test.d.ts +1 -0
- package/dist/test/repository-node-test-reporter.test.js +104 -0
- package/dist/test/repository-oracle-concurrency.test.d.ts +1 -0
- package/dist/test/repository-oracle-concurrency.test.js +542 -0
- package/dist/test/repository-oracle-coverage-witness.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage-witness.test.js +511 -0
- package/dist/test/repository-oracle-coverage.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage.test.js +168 -0
- package/dist/test/repository-oracle-evidence.test.d.ts +1 -0
- package/dist/test/repository-oracle-evidence.test.js +185 -0
- package/dist/test/repository-oracle-plan.test.d.ts +1 -0
- package/dist/test/repository-oracle-plan.test.js +176 -0
- package/dist/test/repository-overlay-isolation.test.d.ts +1 -0
- package/dist/test/repository-overlay-isolation.test.js +85 -0
- package/dist/test/repository-preparation-cache.test.d.ts +1 -0
- package/dist/test/repository-preparation-cache.test.js +414 -0
- package/dist/test/repository-public-artifact.test.d.ts +1 -0
- package/dist/test/repository-public-artifact.test.js +273 -0
- package/dist/test/repository-qualification-diagnostics.test.d.ts +1 -0
- package/dist/test/repository-qualification-diagnostics.test.js +524 -0
- package/dist/test/repository-reference-authoring.test.d.ts +1 -0
- package/dist/test/repository-reference-authoring.test.js +1633 -0
- package/dist/test/repository-review-evidence-v2.test.d.ts +1 -0
- package/dist/test/repository-review-evidence-v2.test.js +183 -0
- package/dist/test/repository-review-evidence.test.d.ts +1 -0
- package/dist/test/repository-review-evidence.test.js +124 -0
- package/dist/test/repository-seed-exclusions.test.d.ts +1 -0
- package/dist/test/repository-seed-exclusions.test.js +96 -0
- package/dist/test/repository-seed-selection.test.d.ts +1 -0
- package/dist/test/repository-seed-selection.test.js +504 -0
- package/dist/test/repository-semantic-calibration.test.d.ts +1 -0
- package/dist/test/repository-semantic-calibration.test.js +688 -0
- package/dist/test/repository-solution-edits.test.d.ts +1 -0
- package/dist/test/repository-solution-edits.test.js +377 -0
- package/dist/test/repository-specification-budget.test.d.ts +1 -0
- package/dist/test/repository-specification-budget.test.js +171 -0
- package/dist/test/repository-specification-contract-checkpoint.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-checkpoint.test.js +228 -0
- package/dist/test/repository-specification-contract-facts.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-facts.test.js +177 -0
- package/dist/test/repository-trajectory-authoring.test.d.ts +1 -0
- package/dist/test/repository-trajectory-authoring.test.js +176 -0
- package/dist/test/repository-valid-control-plan.test.d.ts +1 -0
- package/dist/test/repository-valid-control-plan.test.js +45 -0
- package/dist/test/repository-vitest-phase.test.d.ts +1 -0
- package/dist/test/repository-vitest-phase.test.js +848 -0
- package/dist/test/repository-vitest-reporter.test.d.ts +1 -0
- package/dist/test/repository-vitest-reporter.test.js +158 -0
- package/dist/test/repository-workspace-build.test.d.ts +1 -0
- package/dist/test/repository-workspace-build.test.js +160 -0
- package/dist/test/strict-authoring-schema.test.d.ts +1 -0
- package/dist/test/strict-authoring-schema.test.js +169 -0
- package/package.json +48 -6
|
@@ -0,0 +1,1633 @@
|
|
|
1
|
+
import assert from "node:assert/strict";
|
|
2
|
+
import { execFile } from "node:child_process";
|
|
3
|
+
import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises";
|
|
4
|
+
import { tmpdir } from "node:os";
|
|
5
|
+
import path from "node:path";
|
|
6
|
+
import { test } from "node:test";
|
|
7
|
+
import { promisify } from "node:util";
|
|
8
|
+
import { runInNewContext } from "node:vm";
|
|
9
|
+
import { Effect, Layer } from "effect";
|
|
10
|
+
import { RepositoryFoundryError } from "../errors.js";
|
|
11
|
+
import { EvalAuthoringTransport } from "../project-authoring.js";
|
|
12
|
+
import { repositoryFoundryModelPlanV1 } from "../repository-language-model-protocol.js";
|
|
13
|
+
import { parseRepositorySpecificationContractFactsPacketV1, renderRepositorySpecificationContractFactV1 } from "../repository-specification-contract-facts-protocol.js";
|
|
14
|
+
import { authorRepositoryCaseSpecificationV1 } from "../services/case-authoring/service.js";
|
|
15
|
+
import { generateHistoricalRepositoryCaseV1 } from "../services/case-generation/service.js";
|
|
16
|
+
import { RepositoryFoundryProgressLive } from "../services/foundry-progress/service.js";
|
|
17
|
+
import { RepositoryFoundryLanguageModelLive } from "../services/language-model/service.js";
|
|
18
|
+
const exec = promisify(execFile);
|
|
19
|
+
const privateCode = 'const PRIVATE_AFTER_IMPLEMENTATION_CANARY = "not candidate-facing code";';
|
|
20
|
+
const privateTest = 'const PRIVATE_HISTORICAL_TEST_CANARY = "not candidate-facing test content";';
|
|
21
|
+
const changed = "Normalize event-stream responses to JSON for non-streaming requests.";
|
|
22
|
+
const preserved = "Keep text/plain responses unchanged, even if their body resembles an event stream.";
|
|
23
|
+
const broadRequest = "Repair non-streaming response normalization so both event-stream and text/plain responses containing event-shaped bodies become JSON. Preserve status codes and ordinary JSON responses, and keep public response metadata compatible.";
|
|
24
|
+
const narrowRequest = `Repair non-streaming response normalization. ${changed} ${preserved} Preserve response status codes, ordinary JSON responses, and public response metadata.`;
|
|
25
|
+
const wireDimension = "wire-protocol-compatibility: Correct external request, response, streaming, and error behavior across the gateway's supported wire surfaces.";
|
|
26
|
+
const routingChanged = "Reject incomplete internal routing activations instead of silently selecting a fallback target.";
|
|
27
|
+
const routingPreserved = "Preserve the selected target for complete internal routing activations.";
|
|
28
|
+
const routingRequest = `Repair internal activation selection. ${routingChanged} ${routingPreserved} Keep this change confined to the in-process policy API, without changing external HTTP request, response, streaming, or error behavior.`;
|
|
29
|
+
const dimensionMismatchDetail = "This seed changes only internal activation selection, not an external request, response, stream, or error contract. Rewriting the task as a wire-protocol change would invent unsupported scope.";
|
|
30
|
+
const git = async (root, args) => (await exec("git", ["-C", root, ...args])).stdout.trim();
|
|
31
|
+
const fixture = async (oversized = false, domain = "wire") => {
|
|
32
|
+
const behavior = domain === "routing"
|
|
33
|
+
? {
|
|
34
|
+
sourcePath: "src/routing-policy.js",
|
|
35
|
+
testPath: "src/routing-policy.test.js",
|
|
36
|
+
id: "complete-routing-activation",
|
|
37
|
+
summary: "Reject incomplete internal routing activations",
|
|
38
|
+
changed: routingChanged,
|
|
39
|
+
preserved: routingPreserved,
|
|
40
|
+
changedScopeId: "reject-incomplete-activation",
|
|
41
|
+
preservedScopeId: "preserve-complete-target",
|
|
42
|
+
changedLine: 3,
|
|
43
|
+
preservedLine: 4,
|
|
44
|
+
beforeSource: 'export function resolveRoute(activation) {\n return activation?.target ?? "fallback";\n}\n',
|
|
45
|
+
afterSource: 'export function resolveRoute(activation) {\n if (!activation?.target) throw new Error("incomplete routing activation");\n return activation.target;\n}\n',
|
|
46
|
+
requestText: routingRequest
|
|
47
|
+
}
|
|
48
|
+
: {
|
|
49
|
+
sourcePath: "src/wire.js",
|
|
50
|
+
testPath: "src/wire.test.js",
|
|
51
|
+
id: "sse-json",
|
|
52
|
+
summary: "Normalize non-streaming responses",
|
|
53
|
+
changed,
|
|
54
|
+
preserved,
|
|
55
|
+
changedScopeId: "normalize-event-stream",
|
|
56
|
+
preservedScopeId: "preserve-text-plain",
|
|
57
|
+
changedLine: 4,
|
|
58
|
+
preservedLine: 3,
|
|
59
|
+
beforeSource: "export function normalize(type, body) {\n return body;\n}\n",
|
|
60
|
+
afterSource: 'export function normalize(type, body) {\n if (type === "text/plain") return body;\n if (type === "text/event-stream") return JSON.stringify({ value: body });\n return body;\n}\n',
|
|
61
|
+
requestText: narrowRequest
|
|
62
|
+
};
|
|
63
|
+
const root = await mkdtemp(path.join(tmpdir(), "routekit-reference-scope-"));
|
|
64
|
+
await git(root, ["init", "-q"]);
|
|
65
|
+
await git(root, ["config", "user.email", "eval@example.test"]);
|
|
66
|
+
await git(root, ["config", "user.name", "Reference Scope Fixture"]);
|
|
67
|
+
await mkdir(path.join(root, "src"));
|
|
68
|
+
await writeFile(path.join(root, "package.json"), '{"name":"reference-fixture","type":"module"}\n');
|
|
69
|
+
await writeFile(path.join(root, "src/a-context.js"), "export const unchangedContext = true;\n");
|
|
70
|
+
await writeFile(path.join(root, "src/empty.js"), "");
|
|
71
|
+
await writeFile(path.join(root, behavior.sourcePath), behavior.beforeSource);
|
|
72
|
+
await writeFile(path.join(root, behavior.testPath), `export const assertion = ${JSON.stringify(behavior.preserved)};\n`);
|
|
73
|
+
await git(root, ["add", "."]);
|
|
74
|
+
await git(root, ["commit", "-qm", "initial"]);
|
|
75
|
+
const initialCommit = await git(root, ["rev-parse", "HEAD"]);
|
|
76
|
+
await writeFile(path.join(root, behavior.sourcePath), `${privateCode}\n${behavior.afterSource}${oversized ? `//${"x".repeat(97_000)}\n` : ""}`);
|
|
77
|
+
await writeFile(path.join(root, behavior.testPath), `${privateTest}\nexport const assertion = ${JSON.stringify(`${behavior.changed} ${behavior.preserved}`)};\n`);
|
|
78
|
+
await writeFile(path.join(root, "src/added.js"), "export const addedContext = true;\n");
|
|
79
|
+
await git(root, ["add", "."]);
|
|
80
|
+
await git(root, ["commit", "-qm", behavior.summary]);
|
|
81
|
+
const referenceCommit = await git(root, ["rev-parse", "HEAD"]);
|
|
82
|
+
const seed = {
|
|
83
|
+
version: 1,
|
|
84
|
+
id: `seed-${domain}`,
|
|
85
|
+
status: "qualified",
|
|
86
|
+
source: { kind: "historical-fix", changeEpisodeId: `episode-${domain}` },
|
|
87
|
+
summary: behavior.summary,
|
|
88
|
+
initialState: { commit: initialCommit },
|
|
89
|
+
referenceCommit,
|
|
90
|
+
capabilityEvidence: [
|
|
91
|
+
{ kind: "symbol", id: domain, path: behavior.sourcePath },
|
|
92
|
+
{ kind: "test", id: `${domain}-test`, path: behavior.testPath }
|
|
93
|
+
],
|
|
94
|
+
targetBehavior: [
|
|
95
|
+
{
|
|
96
|
+
id: behavior.id,
|
|
97
|
+
description: domain === "wire" ? "Normalize all event-shaped responses to JSON." : behavior.changed,
|
|
98
|
+
critical: true,
|
|
99
|
+
evidence: [{ kind: "test", id: `${domain}-test`, path: behavior.testPath }]
|
|
100
|
+
}
|
|
101
|
+
],
|
|
102
|
+
baselineObservations: [],
|
|
103
|
+
preChangeObservations: [],
|
|
104
|
+
postChangeObservations: [],
|
|
105
|
+
candidateTestIds: [`${domain}-test`],
|
|
106
|
+
environment: {
|
|
107
|
+
protectedControlPaths: [behavior.testPath],
|
|
108
|
+
trustedPreparationRecipes: [],
|
|
109
|
+
candidateValidationRecipes: [],
|
|
110
|
+
baselineGradeRecipes: [],
|
|
111
|
+
gradeRecipes: []
|
|
112
|
+
},
|
|
113
|
+
confidence: "high",
|
|
114
|
+
risks: [],
|
|
115
|
+
rejectionReasons: []
|
|
116
|
+
};
|
|
117
|
+
const map = {
|
|
118
|
+
version: 1,
|
|
119
|
+
repository: { root, commit: initialCommit, requestedRef: initialCommit, tree: "fixture" },
|
|
120
|
+
packages: [
|
|
121
|
+
{
|
|
122
|
+
id: "package-fixture",
|
|
123
|
+
name: "reference-fixture",
|
|
124
|
+
path: ".",
|
|
125
|
+
manifestPath: "package.json",
|
|
126
|
+
dependencyPackageIds: [],
|
|
127
|
+
publicExports: ["."]
|
|
128
|
+
}
|
|
129
|
+
],
|
|
130
|
+
publicSurfaces: [
|
|
131
|
+
{
|
|
132
|
+
id: "surface-context",
|
|
133
|
+
packageId: "package-fixture",
|
|
134
|
+
path: "src/a-context.js",
|
|
135
|
+
kind: "package-export",
|
|
136
|
+
symbols: ["unchangedContext"]
|
|
137
|
+
}
|
|
138
|
+
],
|
|
139
|
+
protocols: [],
|
|
140
|
+
commands: [],
|
|
141
|
+
tests: [],
|
|
142
|
+
testBindings: [],
|
|
143
|
+
workloadSignals: [],
|
|
144
|
+
blindSpots: [],
|
|
145
|
+
historyEpisodes: [
|
|
146
|
+
{
|
|
147
|
+
id: `episode-${domain}`,
|
|
148
|
+
commit: referenceCommit,
|
|
149
|
+
parentCommit: initialCommit,
|
|
150
|
+
occurredAt: "2026-09-05T00:00:00Z",
|
|
151
|
+
subject: behavior.summary,
|
|
152
|
+
body: "",
|
|
153
|
+
changedPaths: [
|
|
154
|
+
{ path: behavior.sourcePath, kind: "implementation", status: "modified" },
|
|
155
|
+
{ path: behavior.testPath, kind: "test", status: "modified" }
|
|
156
|
+
],
|
|
157
|
+
testPaths: [behavior.testPath],
|
|
158
|
+
implementationPaths: [behavior.sourcePath],
|
|
159
|
+
protocolPaths: [],
|
|
160
|
+
behaviorChanging: true,
|
|
161
|
+
seedEligible: true
|
|
162
|
+
}
|
|
163
|
+
]
|
|
164
|
+
};
|
|
165
|
+
const family = {
|
|
166
|
+
version: 1,
|
|
167
|
+
id: `family-${domain}`,
|
|
168
|
+
capabilityId: domain === "wire" ? "response-normalization" : "internal-routing-policy",
|
|
169
|
+
jobStatement: behavior.summary,
|
|
170
|
+
lane: "repository-agent",
|
|
171
|
+
seedIds: [seed.id],
|
|
172
|
+
inputStateFacets: [domain === "wire" ? "response content type" : "activation completeness"],
|
|
173
|
+
requestedChangeFacets: [behavior.changed],
|
|
174
|
+
observableFacets: [behavior.changed, behavior.preserved],
|
|
175
|
+
admissibleSolutionPolicy: "behavior-equivalent",
|
|
176
|
+
fixtureAxes: [domain === "wire" ? "content type" : "activation completeness"],
|
|
177
|
+
adversaryFamilies: [domain === "wire" ? "overbroad normalization" : "silent fallback"],
|
|
178
|
+
excludedNearMisses: [],
|
|
179
|
+
repositoryCoverage: seed.capabilityEvidence,
|
|
180
|
+
confidence: "high"
|
|
181
|
+
};
|
|
182
|
+
return { root, initialCommit, referenceCommit, seed, map, family, domain, behavior };
|
|
183
|
+
};
|
|
184
|
+
const evidence = (packet, phase, line = 1, sourcePath = "src/wire.js") => {
|
|
185
|
+
const source = packet.privateReferenceEvidence?.sources.find((entry) => entry.phase === phase && entry.path === sourcePath);
|
|
186
|
+
assert.ok(source?.content);
|
|
187
|
+
return { sourceId: source.sourceId, startLine: line, endLine: line };
|
|
188
|
+
};
|
|
189
|
+
const fixtureContractFacts = (f, packet, incomplete) => ({
|
|
190
|
+
version: 1,
|
|
191
|
+
facts: [
|
|
192
|
+
f.domain === "wire" && !incomplete
|
|
193
|
+
? {
|
|
194
|
+
id: "normalization-policy",
|
|
195
|
+
scopeId: f.behavior.changedScopeId,
|
|
196
|
+
kind: "closed-set",
|
|
197
|
+
subject: "response content types selected for normalization",
|
|
198
|
+
appliesWhen: "the request is non-streaming",
|
|
199
|
+
members: ["text/event-stream"],
|
|
200
|
+
outsideBehavior: "Preserve the original response body.",
|
|
201
|
+
evidence: [evidence(packet, "after", f.behavior.changedLine, f.behavior.sourcePath)]
|
|
202
|
+
}
|
|
203
|
+
: {
|
|
204
|
+
id: "normalization-policy",
|
|
205
|
+
scopeId: f.behavior.changedScopeId,
|
|
206
|
+
kind: "behavior",
|
|
207
|
+
statement: incomplete ? "Apply the appropriate response policy." : f.behavior.changed,
|
|
208
|
+
evidence: [evidence(packet, "after", f.behavior.changedLine, f.behavior.sourcePath)]
|
|
209
|
+
},
|
|
210
|
+
{
|
|
211
|
+
id: "preservation-policy",
|
|
212
|
+
scopeId: f.behavior.preservedScopeId,
|
|
213
|
+
kind: "behavior",
|
|
214
|
+
statement: f.behavior.preserved,
|
|
215
|
+
evidence: [evidence(packet, "after", f.behavior.preservedLine, f.behavior.sourcePath)]
|
|
216
|
+
}
|
|
217
|
+
]
|
|
218
|
+
});
|
|
219
|
+
const markedFixtureResponse = (f, options, request, packet, original) => {
|
|
220
|
+
if (request.foundryRole === "repository-analyst") {
|
|
221
|
+
const clarification = request.schemaName === "routekit_repository_contract_fact_clarification_v1";
|
|
222
|
+
const proposal = fixtureContractFacts(f, packet, !clarification && options.initialContractFacts === "incomplete");
|
|
223
|
+
if (clarification) {
|
|
224
|
+
assert.ok(packet.clarificationRequests?.length);
|
|
225
|
+
return {
|
|
226
|
+
...proposal,
|
|
227
|
+
resolutions: packet.clarificationRequests.map((question) => ({
|
|
228
|
+
requestId: question.requestId,
|
|
229
|
+
outcome: "answered",
|
|
230
|
+
factIds: proposal.facts
|
|
231
|
+
.filter((fact) => fact.scopeId === question.scopeId)
|
|
232
|
+
.map((fact) => fact.id),
|
|
233
|
+
detail: "The complete task-local input boundary is established by the pinned evidence."
|
|
234
|
+
}))
|
|
235
|
+
};
|
|
236
|
+
}
|
|
237
|
+
return { ...original, contractFacts: proposal };
|
|
238
|
+
}
|
|
239
|
+
if (!request.foundryRole?.startsWith("specification-critic-"))
|
|
240
|
+
return original;
|
|
241
|
+
assert.ok(packet.contractFacts);
|
|
242
|
+
const reference = request.foundryRole === "specification-critic-b";
|
|
243
|
+
const policy = packet.contractFacts.facts.find((fact) => fact.id === "normalization-policy");
|
|
244
|
+
let complete = f.domain === "routing" || policy?.kind === "closed-set";
|
|
245
|
+
if (reference && f.domain === "wire" && policy?.kind === "closed-set") {
|
|
246
|
+
const after = packet.privateReferenceEvidence?.sources.find((source) => source.phase === "after" && source.path === f.behavior.sourcePath);
|
|
247
|
+
assert.ok(after?.content);
|
|
248
|
+
// Execute only this test's trusted tiny fixture. The verdict is derived from
|
|
249
|
+
// actual supplied pinned behavior, not a pre-labelled sufficient response.
|
|
250
|
+
assert.ok(after.content.includes(f.behavior.afterSource));
|
|
251
|
+
const normalize = runInNewContext(`${after.content.replace(/\bexport\s+/gu, "")}\nnormalize;`, Object.create(null), { timeout: 1_000 });
|
|
252
|
+
const probeTypes = [
|
|
253
|
+
...new Set([
|
|
254
|
+
...Array.from(after.content.matchAll(/type === "([^"]+)"/gu), (match) => match[1]),
|
|
255
|
+
...policy.members,
|
|
256
|
+
"application/json",
|
|
257
|
+
"outside-declared-types"
|
|
258
|
+
])
|
|
259
|
+
];
|
|
260
|
+
const preservedBody = "bounded-fixture-response";
|
|
261
|
+
const actualNormalizedTypes = probeTypes.filter((type) => normalize(type, preservedBody) !== preservedBody);
|
|
262
|
+
complete =
|
|
263
|
+
policy.members.length === actualNormalizedTypes.length &&
|
|
264
|
+
policy.members.every((member) => actualNormalizedTypes.includes(member)) &&
|
|
265
|
+
policy.outsideBehavior === "Preserve the original response body.";
|
|
266
|
+
}
|
|
267
|
+
const scopeChecks = packet.behavioralScope.map((scope) => ({
|
|
268
|
+
scopeId: scope.id,
|
|
269
|
+
factIds: packet
|
|
270
|
+
.contractFacts.facts.filter((fact) => fact.scopeId === scope.id)
|
|
271
|
+
.map((fact) => fact.id),
|
|
272
|
+
outcome: scope.id === f.behavior.changedScopeId && !complete
|
|
273
|
+
? "missing-facts"
|
|
274
|
+
: "complete",
|
|
275
|
+
detail: scope.id === f.behavior.changedScopeId && !complete
|
|
276
|
+
? "The packet does not establish the complete task-local input boundary and outside behavior."
|
|
277
|
+
: "The complete local behavior is represented by the supplied facts and visible constraints.",
|
|
278
|
+
evidence: reference
|
|
279
|
+
? [
|
|
280
|
+
evidence(packet, "after", scope.id === f.behavior.changedScopeId
|
|
281
|
+
? f.behavior.changedLine
|
|
282
|
+
: f.behavior.preservedLine, f.behavior.sourcePath)
|
|
283
|
+
]
|
|
284
|
+
: []
|
|
285
|
+
}));
|
|
286
|
+
const contractFactReview = {
|
|
287
|
+
packetDigest: packet.contractFacts.digest,
|
|
288
|
+
scopeChecks,
|
|
289
|
+
clarificationRequests: scopeChecks
|
|
290
|
+
.filter((check) => check.outcome !== "complete")
|
|
291
|
+
.map((check) => ({
|
|
292
|
+
scopeId: check.scopeId,
|
|
293
|
+
factIds: check.factIds,
|
|
294
|
+
reason: "missing-boundary",
|
|
295
|
+
question: "Which complete input set is selected, and what happens outside that set?"
|
|
296
|
+
}))
|
|
297
|
+
};
|
|
298
|
+
// The ordinary sufficient self-label intentionally remains. The owner must
|
|
299
|
+
// independently enforce the structured fact check rather than trust it.
|
|
300
|
+
return { ...original, contractFactReview };
|
|
301
|
+
};
|
|
302
|
+
const authoringTransport = (f, options, calls) => {
|
|
303
|
+
let writes = 0;
|
|
304
|
+
const complete = (request) => {
|
|
305
|
+
calls.push(request);
|
|
306
|
+
const packet = JSON.parse(request.input);
|
|
307
|
+
let value;
|
|
308
|
+
if (request.foundryRole === "repository-analyst") {
|
|
309
|
+
assert.ok(packet.privateReferenceEvidence?.sources.some((source) => source.content?.includes(privateCode)));
|
|
310
|
+
assert.ok(packet.privateReferenceEvidence?.sources.some((source) => source.content?.includes(privateTest)));
|
|
311
|
+
value = {
|
|
312
|
+
summary: f.behavior.summary,
|
|
313
|
+
userObservableBehaviors: [f.behavior.changed, f.behavior.preserved],
|
|
314
|
+
constraints: [f.behavior.preserved],
|
|
315
|
+
risks: [],
|
|
316
|
+
relevantPaths: [f.behavior.sourcePath, f.behavior.testPath],
|
|
317
|
+
referenceScope: {
|
|
318
|
+
clauses: [
|
|
319
|
+
{
|
|
320
|
+
id: f.behavior.changedScopeId,
|
|
321
|
+
kind: "changed",
|
|
322
|
+
behaviorIds: [f.behavior.id],
|
|
323
|
+
description: f.behavior.changed,
|
|
324
|
+
evidence: [
|
|
325
|
+
evidence(packet, "before", 2, f.behavior.sourcePath),
|
|
326
|
+
evidence(packet, "after", f.behavior.changedLine, f.behavior.sourcePath)
|
|
327
|
+
]
|
|
328
|
+
},
|
|
329
|
+
{
|
|
330
|
+
id: f.behavior.preservedScopeId,
|
|
331
|
+
kind: "preserved",
|
|
332
|
+
behaviorIds: [f.behavior.id],
|
|
333
|
+
description: f.behavior.preserved,
|
|
334
|
+
evidence: [
|
|
335
|
+
evidence(packet, "before", 2, f.behavior.sourcePath),
|
|
336
|
+
evidence(packet, "after", f.behavior.preservedLine, f.behavior.sourcePath)
|
|
337
|
+
]
|
|
338
|
+
}
|
|
339
|
+
],
|
|
340
|
+
unresolvedQuestions: []
|
|
341
|
+
}
|
|
342
|
+
};
|
|
343
|
+
}
|
|
344
|
+
else if (request.foundryRole === "specification-writer") {
|
|
345
|
+
writes += 1;
|
|
346
|
+
value = {
|
|
347
|
+
requestText: f.domain === "wire" && (options.alwaysBroad || (writes === 1 && !options.alwaysNarrow))
|
|
348
|
+
? broadRequest
|
|
349
|
+
: f.behavior.requestText,
|
|
350
|
+
constraints: ["Preserve unrelated public behavior."]
|
|
351
|
+
};
|
|
352
|
+
}
|
|
353
|
+
else {
|
|
354
|
+
assert.ok(request.foundryRole === "specification-critic-a" ||
|
|
355
|
+
request.foundryRole === "specification-critic-b", `Unexpected downstream generation call: ${request.foundryRole}`);
|
|
356
|
+
const contradictory = packet.visible?.requestText === broadRequest;
|
|
357
|
+
value = {
|
|
358
|
+
// Deliberately self-label as sufficient: the host must also check the
|
|
359
|
+
// independently produced structured reference verdict and all evidence.
|
|
360
|
+
verdict: "sufficient",
|
|
361
|
+
detail: "The visible request is otherwise actionable.",
|
|
362
|
+
criticalBehaviorCoverage: packet.targetBehavior
|
|
363
|
+
.filter((entry) => entry.critical)
|
|
364
|
+
.map((entry) => ({
|
|
365
|
+
behaviorId: entry.id,
|
|
366
|
+
outcome: "covered",
|
|
367
|
+
detail: "The visible request states this behavior."
|
|
368
|
+
})),
|
|
369
|
+
findings: []
|
|
370
|
+
};
|
|
371
|
+
if (packet.requestedDimension !== undefined) {
|
|
372
|
+
value.dimensionFit = {
|
|
373
|
+
requestedDimension: packet.requestedDimension,
|
|
374
|
+
outcome: f.domain === "wire" ? "fit" : "mismatch",
|
|
375
|
+
detail: f.domain === "wire"
|
|
376
|
+
? "The authentic delta changes response content-type handling and external response bodies."
|
|
377
|
+
: dimensionMismatchDetail,
|
|
378
|
+
evidencePaths: [f.behavior.sourcePath, f.behavior.testPath]
|
|
379
|
+
};
|
|
380
|
+
}
|
|
381
|
+
if (request.foundryRole === "specification-critic-b") {
|
|
382
|
+
const source = packet.privateReferenceEvidence?.sources.find((entry) => entry.phase === "after" && entry.path === f.behavior.sourcePath);
|
|
383
|
+
assert.ok(source?.content?.includes(f.behavior.afterSource));
|
|
384
|
+
value.referenceCompatibility = {
|
|
385
|
+
outcome: contradictory ? "contradictory" : "compatible",
|
|
386
|
+
detail: contradictory
|
|
387
|
+
? "text/plain must remain unchanged, because it is excluded from normalization."
|
|
388
|
+
: "Every requirement respects the authentic change and preservation boundaries.",
|
|
389
|
+
scopeCoverage: packet.behavioralScope?.map((clause) => ({
|
|
390
|
+
scopeId: clause.id,
|
|
391
|
+
outcome: contradictory && clause.id === f.behavior.preservedScopeId
|
|
392
|
+
? "contradictory"
|
|
393
|
+
: "supported",
|
|
394
|
+
detail: clause.id === f.behavior.preservedScopeId ? f.behavior.preserved : f.behavior.changed,
|
|
395
|
+
evidence: [
|
|
396
|
+
evidence(packet, "after", clause.id === f.behavior.preservedScopeId
|
|
397
|
+
? f.behavior.preservedLine
|
|
398
|
+
: f.behavior.changedLine, f.behavior.sourcePath)
|
|
399
|
+
]
|
|
400
|
+
}))
|
|
401
|
+
};
|
|
402
|
+
}
|
|
403
|
+
}
|
|
404
|
+
if (options.specificationContractFactsVersion === 1) {
|
|
405
|
+
value = markedFixtureResponse(f, options, request, packet, value);
|
|
406
|
+
}
|
|
407
|
+
options.mutate?.(request.foundryRole ?? "", value, packet);
|
|
408
|
+
return JSON.stringify(value);
|
|
409
|
+
};
|
|
410
|
+
return Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
411
|
+
complete: (request) => Effect.sync(() => complete(request))
|
|
412
|
+
}));
|
|
413
|
+
};
|
|
414
|
+
const run = (f, options, calls) => Effect.runPromise(authorRepositoryCaseSpecificationV1({
|
|
415
|
+
repositoryRoot: f.root,
|
|
416
|
+
operationId: "reference-scope-test",
|
|
417
|
+
caseId: "case-sse",
|
|
418
|
+
map: f.map,
|
|
419
|
+
seed: f.seed,
|
|
420
|
+
family: f.family,
|
|
421
|
+
maximumSpecificationRevisions: options.maximum,
|
|
422
|
+
...(options.specificationContractFactsVersion === undefined
|
|
423
|
+
? {}
|
|
424
|
+
: { specificationContractFactsVersion: options.specificationContractFactsVersion }),
|
|
425
|
+
...(options.requestedDimension === undefined
|
|
426
|
+
? {}
|
|
427
|
+
: { requestedDimension: options.requestedDimension }),
|
|
428
|
+
modelPlan: repositoryFoundryModelPlanV1({ primaryModel: "codex/gpt-6-astra" })
|
|
429
|
+
}).pipe(Effect.provide(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(authoringTransport(f, options, calls))))));
|
|
430
|
+
const assertPrivate = (calls, f) => {
|
|
431
|
+
for (const call of calls.filter((entry) => entry.foundryRole === "specification-writer" || entry.foundryRole === "specification-critic-a")) {
|
|
432
|
+
for (const canary of [
|
|
433
|
+
f.referenceCommit,
|
|
434
|
+
privateCode,
|
|
435
|
+
privateTest,
|
|
436
|
+
"private-reference-after-",
|
|
437
|
+
"private-reference-question-",
|
|
438
|
+
"private-contract-fact-question-",
|
|
439
|
+
"pendingQuestions",
|
|
440
|
+
"privateReferenceEvidence",
|
|
441
|
+
"privateContractFacts",
|
|
442
|
+
'"referenceScope"'
|
|
443
|
+
]) {
|
|
444
|
+
assert.equal(call.input.includes(canary), false, `${call.foundryRole} must not receive ${canary}`);
|
|
445
|
+
}
|
|
446
|
+
}
|
|
447
|
+
};
|
|
448
|
+
const assertGenerationRejection = (error, code, completedRevisions) => {
|
|
449
|
+
assert.ok(error instanceof RepositoryFoundryError);
|
|
450
|
+
assert.equal(error.operation, "select-generation-context");
|
|
451
|
+
assert.deepEqual(error.generationRejection, { code, completedRevisions });
|
|
452
|
+
return true;
|
|
453
|
+
};
|
|
454
|
+
test("a genuinely matching wire seed passes the same pinned dimension in one independent review round", async () => {
|
|
455
|
+
const f = await fixture();
|
|
456
|
+
const calls = [];
|
|
457
|
+
try {
|
|
458
|
+
const result = await run(f, { maximum: 2, alwaysNarrow: true, requestedDimension: wireDimension }, calls);
|
|
459
|
+
assert.equal(result.modelCalls.length, 4);
|
|
460
|
+
assert.equal(result.specificationRevisions?.length, 1);
|
|
461
|
+
assert.deepEqual(result.specificationRevisions?.[0]?.rejectionReasons, []);
|
|
462
|
+
assert.ok(result.reviews.every((review) => review.dimensionFit?.outcome === "fit"));
|
|
463
|
+
assert.ok(calls.every((call) => JSON.parse(call.input).requestedDimension === wireDimension));
|
|
464
|
+
assertPrivate(calls, f);
|
|
465
|
+
}
|
|
466
|
+
finally {
|
|
467
|
+
await rm(f.root, { recursive: true, force: true });
|
|
468
|
+
}
|
|
469
|
+
});
|
|
470
|
+
test("marked authoring clarifies missing facts once, renders them, and obtains two fresh independent reviews", async () => {
|
|
471
|
+
const f = await fixture();
|
|
472
|
+
const calls = [];
|
|
473
|
+
try {
|
|
474
|
+
const result = await run(f, {
|
|
475
|
+
maximum: 1,
|
|
476
|
+
specificationContractFactsVersion: 1,
|
|
477
|
+
initialContractFacts: "incomplete",
|
|
478
|
+
alwaysNarrow: true
|
|
479
|
+
}, calls);
|
|
480
|
+
assert.equal(calls.length, 8);
|
|
481
|
+
assert.deepEqual(calls.map((call) => call.foundryRole), [
|
|
482
|
+
"repository-analyst",
|
|
483
|
+
"specification-writer",
|
|
484
|
+
"specification-critic-a",
|
|
485
|
+
"specification-critic-b",
|
|
486
|
+
"repository-analyst",
|
|
487
|
+
"specification-writer",
|
|
488
|
+
"specification-critic-a",
|
|
489
|
+
"specification-critic-b"
|
|
490
|
+
]);
|
|
491
|
+
assert.equal(calls[0].schemaName, "routekit_repository_analysis_contract_facts_v1");
|
|
492
|
+
assert.equal(calls[4].schemaName, "routekit_repository_contract_fact_clarification_v1");
|
|
493
|
+
const firstWriter = JSON.parse(calls[1].input);
|
|
494
|
+
const firstReferenceCritic = JSON.parse(calls[3].input);
|
|
495
|
+
const clarification = JSON.parse(calls[4].input);
|
|
496
|
+
const finalWriter = JSON.parse(calls[5].input);
|
|
497
|
+
assert.deepEqual(clarification.contractFacts, firstWriter.contractFacts);
|
|
498
|
+
assert.deepEqual(clarification.privateContractFacts, firstReferenceCritic.privateContractFacts);
|
|
499
|
+
assert.deepEqual(clarification.privateReferenceEvidence?.sources, firstReferenceCritic.privateReferenceEvidence?.sources);
|
|
500
|
+
assert.equal(clarification.clarificationRequests?.length, 1);
|
|
501
|
+
assert.ok(calls[4].instructions.includes("copy unaffected facts and their evidence"));
|
|
502
|
+
assert.notEqual(finalWriter.contractFacts?.digest, firstWriter.contractFacts?.digest);
|
|
503
|
+
assert.ok(result.contractFacts);
|
|
504
|
+
assert.equal(result.contractFacts.independentlyReviewed, true);
|
|
505
|
+
assert.deepEqual(result.contractFacts.packet, finalWriter.contractFacts);
|
|
506
|
+
assert.deepEqual(result.contractFacts.reviewerOperationIds, result.reviews.map((review) => review.reviewerId));
|
|
507
|
+
assert.deepEqual(result.contractFacts.history.map((entry) => entry.sequence), [0, 1]);
|
|
508
|
+
assert.equal(result.contractFacts.history[1].analystOperationId, calls[4].operationId);
|
|
509
|
+
assert.deepEqual(result.contractFacts.history[1].clarificationRequestIds, clarification.clarificationRequests.map((question) => question.requestId));
|
|
510
|
+
assert.deepEqual(result.specificationRevisions?.map((revision) => revision.rejectionReasons.length > 0), [true, false]);
|
|
511
|
+
assert.equal(result.specificationRevisions?.at(-1)?.contractFactsDigest, result.contractFacts.packet.digest);
|
|
512
|
+
for (const review of result.reviews) {
|
|
513
|
+
assert.equal(review.contractFactReview?.packetDigest, result.contractFacts.packet.digest);
|
|
514
|
+
assert.ok(review.contractFactReview?.scopeChecks.every((check) => check.outcome === "complete"));
|
|
515
|
+
assert.deepEqual(review.contractFactReview?.clarificationRequests, []);
|
|
516
|
+
}
|
|
517
|
+
assert.ok(result.visible.lane === "repository-agent");
|
|
518
|
+
for (const fact of result.contractFacts.packet.facts) {
|
|
519
|
+
assert.ok(result.visible.constraints.includes(renderRepositorySpecificationContractFactV1(fact)));
|
|
520
|
+
}
|
|
521
|
+
assert.deepEqual(parseRepositorySpecificationContractFactsPacketV1(result.contractFacts.packet, result.referenceGrounding.scope), result.contractFacts.packet);
|
|
522
|
+
assert.equal(new Set(calls.map((call) => call.operationId)).size, 8);
|
|
523
|
+
assertPrivate(calls, f);
|
|
524
|
+
}
|
|
525
|
+
finally {
|
|
526
|
+
await rm(f.root, { recursive: true, force: true });
|
|
527
|
+
}
|
|
528
|
+
});
|
|
529
|
+
test("a false fact with valid citations cannot pass the independent marked reference check", async () => {
|
|
530
|
+
const f = await fixture();
|
|
531
|
+
const calls = [];
|
|
532
|
+
try {
|
|
533
|
+
await assert.rejects(run(f, {
|
|
534
|
+
maximum: 0,
|
|
535
|
+
specificationContractFactsVersion: 1,
|
|
536
|
+
alwaysNarrow: true,
|
|
537
|
+
mutate: (role, output) => {
|
|
538
|
+
if (role !== "repository-analyst")
|
|
539
|
+
return;
|
|
540
|
+
const proposal = output.contractFacts;
|
|
541
|
+
assert.ok(proposal.facts[0].evidence.length > 0);
|
|
542
|
+
proposal.facts[0].members = ["text/plain"];
|
|
543
|
+
}
|
|
544
|
+
}, calls), (error) => assertGenerationRejection(error, "specification-review-exhausted", 0));
|
|
545
|
+
assert.equal(calls.length, 4);
|
|
546
|
+
const reference = JSON.parse(calls[3].input);
|
|
547
|
+
assert.equal(reference.contractFacts?.facts[0]?.kind, "closed-set");
|
|
548
|
+
assert.ok(reference.privateContractFacts?.facts[0]?.evidence.length);
|
|
549
|
+
assertPrivate(calls, f);
|
|
550
|
+
}
|
|
551
|
+
finally {
|
|
552
|
+
await rm(f.root, { recursive: true, force: true });
|
|
553
|
+
}
|
|
554
|
+
});
|
|
555
|
+
test("stale fact review digests fail identity validation rather than becoming a semantic repair", async () => {
|
|
556
|
+
const f = await fixture();
|
|
557
|
+
const calls = [];
|
|
558
|
+
try {
|
|
559
|
+
await assert.rejects(run(f, {
|
|
560
|
+
maximum: 2,
|
|
561
|
+
specificationContractFactsVersion: 1,
|
|
562
|
+
alwaysNarrow: true,
|
|
563
|
+
mutate: (role, output) => {
|
|
564
|
+
if (role === "specification-critic-b")
|
|
565
|
+
output.contractFactReview.packetDigest = "0".repeat(64);
|
|
566
|
+
}
|
|
567
|
+
}, calls), (error) => {
|
|
568
|
+
assert.ok(error instanceof RepositoryFoundryError);
|
|
569
|
+
assert.equal(error.operation, "select-generation-context");
|
|
570
|
+
assert.equal(error.generationRejection, undefined);
|
|
571
|
+
assert.match(error.detail, /stale identity/u);
|
|
572
|
+
return true;
|
|
573
|
+
});
|
|
574
|
+
assert.equal(calls.length, 4);
|
|
575
|
+
assertPrivate(calls, f);
|
|
576
|
+
}
|
|
577
|
+
finally {
|
|
578
|
+
await rm(f.root, { recursive: true, force: true });
|
|
579
|
+
}
|
|
580
|
+
});
|
|
581
|
+
test("marked zero allowance cannot clarify missing facts and exhausted revisions stay within twelve calls", async () => {
|
|
582
|
+
const f = await fixture();
|
|
583
|
+
try {
|
|
584
|
+
for (const maximum of [0, 2]) {
|
|
585
|
+
const calls = [];
|
|
586
|
+
await assert.rejects(run(f, {
|
|
587
|
+
maximum,
|
|
588
|
+
specificationContractFactsVersion: 1,
|
|
589
|
+
initialContractFacts: "incomplete",
|
|
590
|
+
alwaysNarrow: true,
|
|
591
|
+
mutate: (role, output, packet) => {
|
|
592
|
+
if (role !== "repository-analyst" || !packet.clarificationRequests)
|
|
593
|
+
return;
|
|
594
|
+
output.facts = fixtureContractFacts(f, packet, true).facts;
|
|
595
|
+
}
|
|
596
|
+
}, calls), (error) => assertGenerationRejection(error, "specification-review-exhausted", maximum));
|
|
597
|
+
assert.equal(calls.length, 4 + maximum * 4);
|
|
598
|
+
assert.equal(calls.filter((call) => call.schemaName === "routekit_repository_contract_fact_clarification_v1").length, maximum);
|
|
599
|
+
assert.equal(new Set(calls.map((call) => call.operationId)).size, calls.length);
|
|
600
|
+
assertPrivate(calls, f);
|
|
601
|
+
}
|
|
602
|
+
}
|
|
603
|
+
finally {
|
|
604
|
+
await rm(f.root, { recursive: true, force: true });
|
|
605
|
+
}
|
|
606
|
+
});
|
|
607
|
+
test("writer-only marked repair does not spend another analyst call or change the accepted packet", async () => {
|
|
608
|
+
const f = await fixture();
|
|
609
|
+
const calls = [];
|
|
610
|
+
try {
|
|
611
|
+
const result = await run(f, { maximum: 1, specificationContractFactsVersion: 1 }, calls);
|
|
612
|
+
assert.equal(calls.length, 7);
|
|
613
|
+
assert.equal(calls.filter((call) => call.foundryRole === "repository-analyst").length, 1);
|
|
614
|
+
assert.ok(result.contractFacts);
|
|
615
|
+
assert.equal(result.contractFacts.history[0].packetDigest, result.contractFacts.history[1].packetDigest);
|
|
616
|
+
assert.deepEqual(result.contractFacts.history[1].clarificationRequestIds, []);
|
|
617
|
+
assertPrivate(calls, f);
|
|
618
|
+
}
|
|
619
|
+
finally {
|
|
620
|
+
await rm(f.root, { recursive: true, force: true });
|
|
621
|
+
}
|
|
622
|
+
});
|
|
623
|
+
test("unresolved or unrelated-scope clarification fails before another writer without false exhausted credit", async () => {
|
|
624
|
+
const f = await fixture();
|
|
625
|
+
try {
|
|
626
|
+
for (const mode of ["unresolved", "unrelated-scope", "missing-resolution"]) {
|
|
627
|
+
const calls = [];
|
|
628
|
+
await assert.rejects(run(f, {
|
|
629
|
+
maximum: 2,
|
|
630
|
+
specificationContractFactsVersion: 1,
|
|
631
|
+
initialContractFacts: "incomplete",
|
|
632
|
+
alwaysNarrow: true,
|
|
633
|
+
mutate: (role, output, packet) => {
|
|
634
|
+
if (role !== "repository-analyst" || !packet.clarificationRequests)
|
|
635
|
+
return;
|
|
636
|
+
if (mode === "unresolved")
|
|
637
|
+
output.resolutions[0].outcome = "unresolved";
|
|
638
|
+
else if (mode === "missing-resolution")
|
|
639
|
+
output.resolutions = [];
|
|
640
|
+
else
|
|
641
|
+
output.facts[1].statement =
|
|
642
|
+
"Change previously preserved responses.";
|
|
643
|
+
}
|
|
644
|
+
}, calls), (error) => {
|
|
645
|
+
assert.ok(error instanceof RepositoryFoundryError);
|
|
646
|
+
assert.equal(error.operation, "select-generation-context");
|
|
647
|
+
assert.equal(error.generationRejection, undefined);
|
|
648
|
+
assert.match(error.detail, /clarification/u);
|
|
649
|
+
return true;
|
|
650
|
+
});
|
|
651
|
+
assert.equal(calls.length, 5, mode);
|
|
652
|
+
assertPrivate(calls, f);
|
|
653
|
+
}
|
|
654
|
+
}
|
|
655
|
+
finally {
|
|
656
|
+
await rm(f.root, { recursive: true, force: true });
|
|
657
|
+
}
|
|
658
|
+
});
|
|
659
|
+
test("marked fact privacy and strict excess-field failures do not expose a writer to private proposals", async () => {
|
|
660
|
+
const f = await fixture();
|
|
661
|
+
try {
|
|
662
|
+
for (const mode of ["private-code", "extra-field"]) {
|
|
663
|
+
const calls = [];
|
|
664
|
+
await assert.rejects(run(f, {
|
|
665
|
+
maximum: 1,
|
|
666
|
+
specificationContractFactsVersion: 1,
|
|
667
|
+
alwaysNarrow: true,
|
|
668
|
+
mutate: (role, output) => {
|
|
669
|
+
if (role !== "repository-analyst")
|
|
670
|
+
return;
|
|
671
|
+
const facts = output.contractFacts.facts;
|
|
672
|
+
if (mode === "extra-field")
|
|
673
|
+
facts[0].privateCode = privateCode;
|
|
674
|
+
else
|
|
675
|
+
facts[1].statement = privateCode;
|
|
676
|
+
}
|
|
677
|
+
}, calls), (error) => {
|
|
678
|
+
assert.ok(error instanceof RepositoryFoundryError);
|
|
679
|
+
assert.equal(error.generationRejection, undefined);
|
|
680
|
+
assert.equal(error.detail.includes(privateCode), false);
|
|
681
|
+
return true;
|
|
682
|
+
});
|
|
683
|
+
assert.equal(calls.length, 1, mode);
|
|
684
|
+
}
|
|
685
|
+
}
|
|
686
|
+
finally {
|
|
687
|
+
await rm(f.root, { recursive: true, force: true });
|
|
688
|
+
}
|
|
689
|
+
});
|
|
690
|
+
test("marked real case generation preserves the grounded no-fit terminal before downstream work", async () => {
|
|
691
|
+
const f = await fixture(false, "routing");
|
|
692
|
+
const calls = [];
|
|
693
|
+
const stages = [];
|
|
694
|
+
try {
|
|
695
|
+
await assert.rejects(Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
696
|
+
repositoryRoot: f.root,
|
|
697
|
+
operationId: "marked-no-fit-generation",
|
|
698
|
+
caseId: "marked-routing-case",
|
|
699
|
+
requestedDimension: wireDimension,
|
|
700
|
+
specificationContractFactsVersion: 1,
|
|
701
|
+
maximumSpecificationRevisions: 2,
|
|
702
|
+
map: f.map,
|
|
703
|
+
seed: f.seed,
|
|
704
|
+
modelPlan: repositoryFoundryModelPlanV1({ primaryModel: "codex/gpt-6-astra" })
|
|
705
|
+
}).pipe(Effect.provide(Layer.merge(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(authoringTransport(f, {
|
|
706
|
+
maximum: 2,
|
|
707
|
+
specificationContractFactsVersion: 1,
|
|
708
|
+
requestedDimension: wireDimension
|
|
709
|
+
}, calls))), RepositoryFoundryProgressLive((event) => Effect.sync(() => {
|
|
710
|
+
if (event.kind === "stage")
|
|
711
|
+
stages.push(event.stage);
|
|
712
|
+
})))))), (error) => assertGenerationRejection(error, "specification-dimension-mismatch", 0));
|
|
713
|
+
assert.deepEqual(stages, ["author-specification"]);
|
|
714
|
+
assert.equal(calls.length, 4);
|
|
715
|
+
assertPrivate(calls, f);
|
|
716
|
+
}
|
|
717
|
+
finally {
|
|
718
|
+
await rm(f.root, { recursive: true, force: true });
|
|
719
|
+
}
|
|
720
|
+
});
|
|
721
|
+
test("two validated internal-routing mismatches stop after four calls even with unresolved ancillary reference questions", async () => {
|
|
722
|
+
const f = await fixture(false, "routing");
|
|
723
|
+
try {
|
|
724
|
+
for (const referenceOutcome of ["compatible", "uncertain"]) {
|
|
725
|
+
const calls = [];
|
|
726
|
+
await assert.rejects(run(f, {
|
|
727
|
+
maximum: 2,
|
|
728
|
+
requestedDimension: wireDimension,
|
|
729
|
+
mutate: (role, output, packet) => {
|
|
730
|
+
if (referenceOutcome !== "uncertain")
|
|
731
|
+
return;
|
|
732
|
+
if (role === "repository-analyst")
|
|
733
|
+
setQuestions(output, ["Which outer HTTP endpoint exposes this internal policy?"]);
|
|
734
|
+
if (role === "specification-critic-b") {
|
|
735
|
+
const compatibility = output.referenceCompatibility;
|
|
736
|
+
compatibility.outcome = "uncertain";
|
|
737
|
+
setResolutions(output, packet.privateReferenceEvidence.pendingQuestions.map((question) => ({
|
|
738
|
+
questionId: question.id,
|
|
739
|
+
outcome: "unresolved",
|
|
740
|
+
detail: "The outer HTTP call sites are not established by this local delta.",
|
|
741
|
+
affectedScopeIds: [],
|
|
742
|
+
evidence: []
|
|
743
|
+
})));
|
|
744
|
+
output.dimensionFit.detail =
|
|
745
|
+
`${dimensionMismatchDetail} ${privateCode} ${f.referenceCommit} private-reference-after-0`;
|
|
746
|
+
}
|
|
747
|
+
}
|
|
748
|
+
}, calls), (error) => {
|
|
749
|
+
assertGenerationRejection(error, "specification-dimension-mismatch", 0);
|
|
750
|
+
for (const canary of [
|
|
751
|
+
privateCode,
|
|
752
|
+
privateTest,
|
|
753
|
+
f.referenceCommit,
|
|
754
|
+
"private-reference-after-",
|
|
755
|
+
"private-reference-question-"
|
|
756
|
+
])
|
|
757
|
+
assert.equal(error.message.includes(canary), false);
|
|
758
|
+
return true;
|
|
759
|
+
});
|
|
760
|
+
assert.equal(calls.length, 4, referenceOutcome);
|
|
761
|
+
assert.deepEqual(calls.map((call) => call.foundryRole), [
|
|
762
|
+
"repository-analyst",
|
|
763
|
+
"specification-writer",
|
|
764
|
+
"specification-critic-a",
|
|
765
|
+
"specification-critic-b"
|
|
766
|
+
]);
|
|
767
|
+
assert.equal(new Set(calls.map((call) => call.operationId)).size, 4);
|
|
768
|
+
assert.ok(calls.every((call) => JSON.parse(call.input).requestedDimension === wireDimension));
|
|
769
|
+
assertPrivate(calls, f);
|
|
770
|
+
}
|
|
771
|
+
}
|
|
772
|
+
finally {
|
|
773
|
+
await rm(f.root, { recursive: true, force: true });
|
|
774
|
+
}
|
|
775
|
+
});
|
|
776
|
+
test("uncertain and mixed dimension judgments remain repairable and cannot short-circuit admission", async () => {
|
|
777
|
+
const f = await fixture();
|
|
778
|
+
try {
|
|
779
|
+
for (const firstOutcomes of [
|
|
780
|
+
["uncertain", "uncertain"],
|
|
781
|
+
["fit", "mismatch"],
|
|
782
|
+
["mismatch", "uncertain"]
|
|
783
|
+
]) {
|
|
784
|
+
let writes = 0;
|
|
785
|
+
const calls = [];
|
|
786
|
+
const result = await run(f, {
|
|
787
|
+
maximum: 2,
|
|
788
|
+
alwaysNarrow: true,
|
|
789
|
+
requestedDimension: wireDimension,
|
|
790
|
+
mutate: (role, output) => {
|
|
791
|
+
if (role === "specification-writer")
|
|
792
|
+
writes += 1;
|
|
793
|
+
if (writes !== 1 || !role.startsWith("specification-critic-"))
|
|
794
|
+
return;
|
|
795
|
+
const fit = output.dimensionFit;
|
|
796
|
+
fit.outcome = firstOutcomes[role === "specification-critic-a" ? 0 : 1];
|
|
797
|
+
fit.detail = "Clarify which externally observable response behavior this task changes.";
|
|
798
|
+
}
|
|
799
|
+
}, calls);
|
|
800
|
+
assert.equal(result.modelCalls.length, 7, firstOutcomes.join("/"));
|
|
801
|
+
assert.deepEqual(result.specificationRevisions?.map((revision) => revision.rejectionReasons.length > 0), [true, false]);
|
|
802
|
+
assert.ok(result.reviews.every((review) => review.dimensionFit?.outcome === "fit"));
|
|
803
|
+
assertPrivate(calls, f);
|
|
804
|
+
}
|
|
805
|
+
}
|
|
806
|
+
finally {
|
|
807
|
+
await rm(f.root, { recursive: true, force: true });
|
|
808
|
+
}
|
|
809
|
+
});
|
|
810
|
+
test("a later definite no-fit decision records only the completed host revision sequence", async () => {
|
|
811
|
+
const f = await fixture(false, "routing");
|
|
812
|
+
try {
|
|
813
|
+
for (const completedRevisions of [1, 2]) {
|
|
814
|
+
let writes = 0;
|
|
815
|
+
const calls = [];
|
|
816
|
+
await assert.rejects(run(f, {
|
|
817
|
+
maximum: 2,
|
|
818
|
+
requestedDimension: wireDimension,
|
|
819
|
+
mutate: (role, output) => {
|
|
820
|
+
if (role === "specification-writer")
|
|
821
|
+
writes += 1;
|
|
822
|
+
if (role !== "specification-critic-a" || writes > completedRevisions)
|
|
823
|
+
return;
|
|
824
|
+
output.dimensionFit.outcome = "uncertain";
|
|
825
|
+
}
|
|
826
|
+
}, calls), (error) => assertGenerationRejection(error, "specification-dimension-mismatch", completedRevisions));
|
|
827
|
+
assert.equal(calls.length, 4 + 3 * completedRevisions);
|
|
828
|
+
assert.equal(calls.filter((call) => call.foundryRole === "specification-writer").length, completedRevisions + 1);
|
|
829
|
+
assertPrivate(calls, f);
|
|
830
|
+
}
|
|
831
|
+
}
|
|
832
|
+
finally {
|
|
833
|
+
await rm(f.root, { recursive: true, force: true });
|
|
834
|
+
}
|
|
835
|
+
});
|
|
836
|
+
test("revision feedback includes dimension details but never exposes private reference material", async () => {
|
|
837
|
+
const f = await fixture();
|
|
838
|
+
const calls = [];
|
|
839
|
+
const publicFeedback = "State the content-type eligibility boundary as external response behavior.";
|
|
840
|
+
let writes = 0;
|
|
841
|
+
try {
|
|
842
|
+
const result = await run(f, {
|
|
843
|
+
maximum: 2,
|
|
844
|
+
alwaysNarrow: true,
|
|
845
|
+
requestedDimension: wireDimension,
|
|
846
|
+
mutate: (role, output) => {
|
|
847
|
+
if (role === "specification-writer")
|
|
848
|
+
writes += 1;
|
|
849
|
+
if (writes !== 1 || !role.startsWith("specification-critic-"))
|
|
850
|
+
return;
|
|
851
|
+
const fit = output.dimensionFit;
|
|
852
|
+
fit.outcome = "uncertain";
|
|
853
|
+
fit.detail =
|
|
854
|
+
role === "specification-critic-a"
|
|
855
|
+
? publicFeedback
|
|
856
|
+
: `${privateCode} ${privateTest} ${f.referenceCommit} private-reference-after-0 private-reference-question-1`;
|
|
857
|
+
}
|
|
858
|
+
}, calls);
|
|
859
|
+
assert.equal(result.modelCalls.length, 7);
|
|
860
|
+
const revision = calls.find((call) => call.foundryRole === "specification-writer" &&
|
|
861
|
+
call.operationId.includes(":specification-revision:1:"));
|
|
862
|
+
assert.ok(revision);
|
|
863
|
+
const feedback = JSON.parse(revision.input)
|
|
864
|
+
.behavioralReviewFeedback;
|
|
865
|
+
assert.ok(feedback.includes(publicFeedback));
|
|
866
|
+
assert.ok(result.specificationRevisions?.[0]?.reviews[1]?.dimensionFit?.detail.includes(privateCode));
|
|
867
|
+
assertPrivate(calls, f);
|
|
868
|
+
}
|
|
869
|
+
finally {
|
|
870
|
+
await rm(f.root, { recursive: true, force: true });
|
|
871
|
+
}
|
|
872
|
+
});
|
|
873
|
+
test("no-fit requires complete exact dimension, critical behavior, and grounded scope evidence from both critics", async () => {
|
|
874
|
+
const f = await fixture(false, "routing");
|
|
875
|
+
try {
|
|
876
|
+
for (const scenario of [
|
|
877
|
+
"missing-dimension",
|
|
878
|
+
"second-critic-missing-dimension",
|
|
879
|
+
"wrong-dimension",
|
|
880
|
+
"second-critic-wrong-dimension",
|
|
881
|
+
"blank-dimension-detail",
|
|
882
|
+
"missing-paths",
|
|
883
|
+
"unknown-path",
|
|
884
|
+
"blank-path",
|
|
885
|
+
"annotated-path",
|
|
886
|
+
"second-critic-annotated-path",
|
|
887
|
+
"after-only-path",
|
|
888
|
+
"no-critical-behavior",
|
|
889
|
+
"missing-critical-coverage",
|
|
890
|
+
"duplicate-critical-coverage",
|
|
891
|
+
"unknown-critical-behavior",
|
|
892
|
+
"uncovered-critical-behavior",
|
|
893
|
+
"blank-critical-detail",
|
|
894
|
+
"missing-reference-review",
|
|
895
|
+
"contradictory-reference",
|
|
896
|
+
"blank-reference-detail",
|
|
897
|
+
"missing-scope",
|
|
898
|
+
"partial-scope",
|
|
899
|
+
"extra-scope",
|
|
900
|
+
"duplicate-scope",
|
|
901
|
+
"unknown-scope",
|
|
902
|
+
"uncertain-scope",
|
|
903
|
+
"contradictory-scope",
|
|
904
|
+
"blank-scope-detail",
|
|
905
|
+
"missing-after-evidence",
|
|
906
|
+
"before-only-evidence",
|
|
907
|
+
"unknown-source",
|
|
908
|
+
"invalid-line",
|
|
909
|
+
"fractional-line"
|
|
910
|
+
]) {
|
|
911
|
+
const calls = [];
|
|
912
|
+
const selected = scenario === "no-critical-behavior"
|
|
913
|
+
? {
|
|
914
|
+
...f,
|
|
915
|
+
seed: {
|
|
916
|
+
...f.seed,
|
|
917
|
+
targetBehavior: f.seed.targetBehavior.map((behavior) => ({
|
|
918
|
+
...behavior,
|
|
919
|
+
critical: false
|
|
920
|
+
}))
|
|
921
|
+
}
|
|
922
|
+
}
|
|
923
|
+
: f;
|
|
924
|
+
await assert.rejects(run(selected, {
|
|
925
|
+
maximum: 2,
|
|
926
|
+
requestedDimension: wireDimension,
|
|
927
|
+
mutate: (role, output) => {
|
|
928
|
+
if (!role.startsWith("specification-critic-"))
|
|
929
|
+
return;
|
|
930
|
+
const fit = output.dimensionFit;
|
|
931
|
+
const critical = output.criticalBehaviorCoverage;
|
|
932
|
+
if (role === "specification-critic-a") {
|
|
933
|
+
if (scenario === "missing-dimension")
|
|
934
|
+
delete output.dimensionFit;
|
|
935
|
+
if (scenario === "wrong-dimension")
|
|
936
|
+
fit.requestedDimension = "internal-routing";
|
|
937
|
+
if (scenario === "blank-dimension-detail")
|
|
938
|
+
fit.detail = " ";
|
|
939
|
+
if (scenario === "missing-paths")
|
|
940
|
+
fit.evidencePaths = [];
|
|
941
|
+
if (scenario === "unknown-path")
|
|
942
|
+
fit.evidencePaths = ["src/missing.js"];
|
|
943
|
+
if (scenario === "blank-path")
|
|
944
|
+
fit.evidencePaths = [" "];
|
|
945
|
+
if (scenario === "annotated-path")
|
|
946
|
+
fit.evidencePaths = [`${f.behavior.sourcePath} (internal policy only)`];
|
|
947
|
+
if (scenario === "after-only-path")
|
|
948
|
+
fit.evidencePaths = ["src/added.js"];
|
|
949
|
+
if (scenario === "missing-critical-coverage")
|
|
950
|
+
critical.length = 0;
|
|
951
|
+
if (scenario === "duplicate-critical-coverage")
|
|
952
|
+
critical.push({ ...critical[0] });
|
|
953
|
+
if (scenario === "unknown-critical-behavior")
|
|
954
|
+
critical[0].behaviorId = "unknown";
|
|
955
|
+
if (scenario === "uncovered-critical-behavior")
|
|
956
|
+
critical[0].outcome = "uncertain";
|
|
957
|
+
if (scenario === "blank-critical-detail")
|
|
958
|
+
critical[0].detail = " ";
|
|
959
|
+
}
|
|
960
|
+
if (role !== "specification-critic-b")
|
|
961
|
+
return;
|
|
962
|
+
if (scenario === "second-critic-missing-dimension")
|
|
963
|
+
delete output.dimensionFit;
|
|
964
|
+
if (scenario === "second-critic-wrong-dimension")
|
|
965
|
+
fit.requestedDimension = `${wireDimension} `;
|
|
966
|
+
if (scenario === "second-critic-annotated-path")
|
|
967
|
+
fit.evidencePaths = [`${f.behavior.testPath}:1-2`];
|
|
968
|
+
if (scenario === "missing-reference-review") {
|
|
969
|
+
delete output.referenceCompatibility;
|
|
970
|
+
return;
|
|
971
|
+
}
|
|
972
|
+
const compatibility = output.referenceCompatibility;
|
|
973
|
+
const coverage = compatibility.scopeCoverage;
|
|
974
|
+
if (scenario === "contradictory-reference")
|
|
975
|
+
compatibility.outcome = "contradictory";
|
|
976
|
+
if (scenario === "blank-reference-detail")
|
|
977
|
+
compatibility.detail = " ";
|
|
978
|
+
if (scenario === "missing-scope")
|
|
979
|
+
coverage.length = 0;
|
|
980
|
+
if (scenario === "partial-scope")
|
|
981
|
+
coverage.pop();
|
|
982
|
+
if (scenario === "extra-scope")
|
|
983
|
+
coverage.push({ ...coverage[0], scopeId: "extra" });
|
|
984
|
+
if (scenario === "duplicate-scope")
|
|
985
|
+
coverage[1] = coverage[0];
|
|
986
|
+
if (scenario === "unknown-scope")
|
|
987
|
+
coverage[0].scopeId = "unknown";
|
|
988
|
+
if (scenario === "uncertain-scope")
|
|
989
|
+
coverage[0].outcome = "uncertain";
|
|
990
|
+
if (scenario === "contradictory-scope")
|
|
991
|
+
coverage[0].outcome = "contradictory";
|
|
992
|
+
if (scenario === "blank-scope-detail")
|
|
993
|
+
coverage[0].detail = " ";
|
|
994
|
+
if (scenario === "missing-after-evidence")
|
|
995
|
+
coverage[0].evidence = [];
|
|
996
|
+
if (scenario === "before-only-evidence")
|
|
997
|
+
coverage[0].evidence[0].sourceId = coverage[0].evidence[0].sourceId.replace("after", "before");
|
|
998
|
+
if (scenario === "unknown-source")
|
|
999
|
+
coverage[0].evidence[0].sourceId = "private-reference-after-unknown";
|
|
1000
|
+
if (scenario === "invalid-line")
|
|
1001
|
+
coverage[0].evidence[0].endLine = 99999;
|
|
1002
|
+
if (scenario === "fractional-line")
|
|
1003
|
+
coverage[0].evidence[0].startLine = 1.5;
|
|
1004
|
+
}
|
|
1005
|
+
}, calls), (error) => {
|
|
1006
|
+
assertGenerationRejection(error, "specification-review-exhausted", 2);
|
|
1007
|
+
return true;
|
|
1008
|
+
}, scenario);
|
|
1009
|
+
assert.equal(calls.length, 10, `${scenario} must preserve all bounded repair rounds`);
|
|
1010
|
+
assertPrivate(calls, f);
|
|
1011
|
+
}
|
|
1012
|
+
}
|
|
1013
|
+
finally {
|
|
1014
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1015
|
+
}
|
|
1016
|
+
});
|
|
1017
|
+
test("schema-malformed critic judgments never become semantic generation rejections", async () => {
|
|
1018
|
+
const f = await fixture(false, "routing");
|
|
1019
|
+
try {
|
|
1020
|
+
for (const scenario of ["invalid-dimension-outcome", "missing-required-coverage"]) {
|
|
1021
|
+
const calls = [];
|
|
1022
|
+
await assert.rejects(run(f, {
|
|
1023
|
+
maximum: 2,
|
|
1024
|
+
requestedDimension: wireDimension,
|
|
1025
|
+
mutate: (role, output) => {
|
|
1026
|
+
if (role !== "specification-critic-b")
|
|
1027
|
+
return;
|
|
1028
|
+
if (scenario === "invalid-dimension-outcome")
|
|
1029
|
+
output.dimensionFit.outcome = "definitely-no-fit";
|
|
1030
|
+
else
|
|
1031
|
+
delete output.criticalBehaviorCoverage;
|
|
1032
|
+
}
|
|
1033
|
+
}, calls), (error) => {
|
|
1034
|
+
assert.ok(error instanceof RepositoryFoundryError);
|
|
1035
|
+
assert.equal(error.operation, "invoke-language-model");
|
|
1036
|
+
assert.equal(error.modelResponseFailure, "schema");
|
|
1037
|
+
assert.equal(error.generationRejection, undefined);
|
|
1038
|
+
return true;
|
|
1039
|
+
});
|
|
1040
|
+
assert.equal(calls.length, 4, scenario);
|
|
1041
|
+
assertPrivate(calls, f);
|
|
1042
|
+
}
|
|
1043
|
+
}
|
|
1044
|
+
finally {
|
|
1045
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1046
|
+
}
|
|
1047
|
+
});
|
|
1048
|
+
test("unrequested dimensions cannot manufacture an early target-selection rejection", async () => {
|
|
1049
|
+
const f = await fixture(false, "routing");
|
|
1050
|
+
const calls = [];
|
|
1051
|
+
try {
|
|
1052
|
+
const result = await run(f, {
|
|
1053
|
+
maximum: 2,
|
|
1054
|
+
mutate: (role, output) => {
|
|
1055
|
+
if (!role.startsWith("specification-critic-"))
|
|
1056
|
+
return;
|
|
1057
|
+
output.dimensionFit = {
|
|
1058
|
+
requestedDimension: wireDimension,
|
|
1059
|
+
outcome: "mismatch",
|
|
1060
|
+
detail: dimensionMismatchDetail,
|
|
1061
|
+
evidencePaths: [f.behavior.sourcePath]
|
|
1062
|
+
};
|
|
1063
|
+
}
|
|
1064
|
+
}, calls);
|
|
1065
|
+
assert.equal(result.modelCalls.length, 4);
|
|
1066
|
+
assertPrivate(calls, f);
|
|
1067
|
+
}
|
|
1068
|
+
finally {
|
|
1069
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1070
|
+
}
|
|
1071
|
+
});
|
|
1072
|
+
test("real case generation stops at the no-fit authoring terminal before solver, oracle, review, or admission work", async () => {
|
|
1073
|
+
const f = await fixture(false, "routing");
|
|
1074
|
+
const calls = [];
|
|
1075
|
+
const stages = [];
|
|
1076
|
+
try {
|
|
1077
|
+
await assert.rejects(Effect.runPromise(generateHistoricalRepositoryCaseV1({
|
|
1078
|
+
repositoryRoot: f.root,
|
|
1079
|
+
operationId: "dimension-no-fit-generation-test",
|
|
1080
|
+
caseId: "case-internal-routing",
|
|
1081
|
+
requestedDimension: wireDimension,
|
|
1082
|
+
maximumSpecificationRevisions: 2,
|
|
1083
|
+
map: f.map,
|
|
1084
|
+
seed: f.seed,
|
|
1085
|
+
modelPlan: repositoryFoundryModelPlanV1({ primaryModel: "codex/gpt-6-astra" })
|
|
1086
|
+
}).pipe(Effect.provide(Layer.merge(RepositoryFoundryLanguageModelLive.pipe(Layer.provide(authoringTransport(f, { maximum: 2, requestedDimension: wireDimension }, calls))), RepositoryFoundryProgressLive((event) => Effect.sync(() => {
|
|
1087
|
+
if (event.kind === "stage")
|
|
1088
|
+
stages.push(event.stage);
|
|
1089
|
+
})))))), (error) => assertGenerationRejection(error, "specification-dimension-mismatch", 0));
|
|
1090
|
+
assert.deepEqual(stages, ["author-specification"]);
|
|
1091
|
+
assert.deepEqual(calls.map((call) => call.foundryRole), [
|
|
1092
|
+
"repository-analyst",
|
|
1093
|
+
"specification-writer",
|
|
1094
|
+
"specification-critic-a",
|
|
1095
|
+
"specification-critic-b"
|
|
1096
|
+
]);
|
|
1097
|
+
assertPrivate(calls, f);
|
|
1098
|
+
}
|
|
1099
|
+
finally {
|
|
1100
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1101
|
+
}
|
|
1102
|
+
});
|
|
1103
|
+
test("grounded authoring rejects text/plain expansion and revises with two fresh critics within the pinned budget", async () => {
|
|
1104
|
+
const f = await fixture();
|
|
1105
|
+
const calls = [];
|
|
1106
|
+
try {
|
|
1107
|
+
const result = await run(f, { maximum: 2 }, calls);
|
|
1108
|
+
assert.ok(result.visible.lane === "repository-agent");
|
|
1109
|
+
assert.equal(result.visible.requestText, narrowRequest);
|
|
1110
|
+
assert.equal(result.modelCalls.length, 7);
|
|
1111
|
+
assert.deepEqual(result.specificationRevisions?.map((entry) => entry.rejectionReasons.length > 0), [true, false]);
|
|
1112
|
+
assert.equal(result.referenceGrounding?.referenceCommit, f.referenceCommit);
|
|
1113
|
+
assert.deepEqual(result.groundedTargetBehavior?.map(({ id, critical, evidence: refs }) => ({
|
|
1114
|
+
id,
|
|
1115
|
+
critical,
|
|
1116
|
+
evidence: refs
|
|
1117
|
+
})), f.seed.targetBehavior.map(({ id, critical, evidence: refs }) => ({
|
|
1118
|
+
id,
|
|
1119
|
+
critical,
|
|
1120
|
+
evidence: refs
|
|
1121
|
+
})));
|
|
1122
|
+
assert.equal(result.groundedTargetBehavior?.[0]?.description, `${changed} ${preserved}`);
|
|
1123
|
+
assert.equal(new Set(result.modelCalls.map((call) => call.operationId)).size, 7);
|
|
1124
|
+
assert.ok(result.modelCalls
|
|
1125
|
+
.slice(4)
|
|
1126
|
+
.every((call) => call.operationId.includes(":specification-revision:1:")));
|
|
1127
|
+
assertPrivate(calls, f);
|
|
1128
|
+
const solverProjection = JSON.stringify({
|
|
1129
|
+
visible: result.visible,
|
|
1130
|
+
initialSources: result.contextSources,
|
|
1131
|
+
targetBehavior: result.groundedTargetBehavior
|
|
1132
|
+
});
|
|
1133
|
+
for (const canary of [f.referenceCommit, privateCode, privateTest, "private-reference-after-"])
|
|
1134
|
+
assert.equal(solverProjection.includes(canary), false);
|
|
1135
|
+
}
|
|
1136
|
+
finally {
|
|
1137
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1138
|
+
}
|
|
1139
|
+
});
|
|
1140
|
+
test("explicit zero still grounds and rejects contradictory text/plain behavior without revision calls", async () => {
|
|
1141
|
+
const f = await fixture();
|
|
1142
|
+
const calls = [];
|
|
1143
|
+
try {
|
|
1144
|
+
await assert.rejects(run(f, { maximum: 0 }, calls), (error) => {
|
|
1145
|
+
assert.match(error.message, /failed independent review after 0 bounded revisions/u);
|
|
1146
|
+
return assertGenerationRejection(error, "specification-review-exhausted", 0);
|
|
1147
|
+
});
|
|
1148
|
+
assert.equal(calls.length, 4);
|
|
1149
|
+
assertPrivate(calls, f);
|
|
1150
|
+
}
|
|
1151
|
+
finally {
|
|
1152
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1153
|
+
}
|
|
1154
|
+
});
|
|
1155
|
+
test("exhausted specification repair never exceeds one analyst and three bounded writer/critic rounds", async () => {
|
|
1156
|
+
const f = await fixture();
|
|
1157
|
+
const calls = [];
|
|
1158
|
+
try {
|
|
1159
|
+
await assert.rejects(run(f, { maximum: 2, alwaysBroad: true }, calls), (error) => {
|
|
1160
|
+
assert.match(error.message, /failed independent review after 2 bounded revisions/u);
|
|
1161
|
+
return assertGenerationRejection(error, "specification-review-exhausted", 2);
|
|
1162
|
+
});
|
|
1163
|
+
assert.equal(calls.length, 10);
|
|
1164
|
+
assert.equal(calls.filter((entry) => entry.foundryRole === "repository-analyst").length, 1);
|
|
1165
|
+
assertPrivate(calls, f);
|
|
1166
|
+
}
|
|
1167
|
+
finally {
|
|
1168
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1169
|
+
}
|
|
1170
|
+
});
|
|
1171
|
+
test("grounding fails closed for malformed scope and invalid or incomplete source evidence", async () => {
|
|
1172
|
+
const f = await fixture();
|
|
1173
|
+
try {
|
|
1174
|
+
for (const scenario of [
|
|
1175
|
+
"blank-question",
|
|
1176
|
+
"missing-before",
|
|
1177
|
+
"missing-after",
|
|
1178
|
+
"unknown-source",
|
|
1179
|
+
"invalid-line",
|
|
1180
|
+
"missing-behavior"
|
|
1181
|
+
]) {
|
|
1182
|
+
const calls = [];
|
|
1183
|
+
await assert.rejects(run(f, {
|
|
1184
|
+
maximum: 2,
|
|
1185
|
+
mutate: (role, output) => {
|
|
1186
|
+
if (role !== "repository-analyst")
|
|
1187
|
+
return;
|
|
1188
|
+
const scope = output.referenceScope;
|
|
1189
|
+
if (scenario === "blank-question")
|
|
1190
|
+
scope.unresolvedQuestions = [" "];
|
|
1191
|
+
const clause = scope.clauses[0];
|
|
1192
|
+
if (scenario === "missing-before")
|
|
1193
|
+
clause.evidence = clause.evidence.filter((entry) => !entry.sourceId.includes("before"));
|
|
1194
|
+
if (scenario === "missing-after")
|
|
1195
|
+
clause.evidence = clause.evidence.filter((entry) => !entry.sourceId.includes("after"));
|
|
1196
|
+
if (scenario === "unknown-source")
|
|
1197
|
+
clause.evidence[0].sourceId = "unbound-source";
|
|
1198
|
+
if (scenario === "invalid-line")
|
|
1199
|
+
clause.evidence[0].endLine = 99999;
|
|
1200
|
+
if (scenario === "missing-behavior")
|
|
1201
|
+
clause.behaviorIds = ["invented-behavior"];
|
|
1202
|
+
}
|
|
1203
|
+
}, calls), /scope is incomplete, malformed, or lacks valid pinned source line evidence/u);
|
|
1204
|
+
assert.equal(calls.length, 1, scenario);
|
|
1205
|
+
}
|
|
1206
|
+
}
|
|
1207
|
+
finally {
|
|
1208
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1209
|
+
}
|
|
1210
|
+
});
|
|
1211
|
+
test("sufficient critic labels cannot override unknown, missing, or unbound reference compatibility", async () => {
|
|
1212
|
+
const f = await fixture();
|
|
1213
|
+
try {
|
|
1214
|
+
for (const scenario of [
|
|
1215
|
+
"missing",
|
|
1216
|
+
"uncertain",
|
|
1217
|
+
"missing-scope",
|
|
1218
|
+
"duplicate-scope",
|
|
1219
|
+
"before-only",
|
|
1220
|
+
"fractional-line"
|
|
1221
|
+
]) {
|
|
1222
|
+
const calls = [];
|
|
1223
|
+
await assert.rejects(run(f, {
|
|
1224
|
+
maximum: 0,
|
|
1225
|
+
alwaysNarrow: true,
|
|
1226
|
+
mutate: (role, output) => {
|
|
1227
|
+
if (role !== "specification-critic-b")
|
|
1228
|
+
return;
|
|
1229
|
+
const review = output.referenceCompatibility;
|
|
1230
|
+
if (scenario === "missing")
|
|
1231
|
+
delete output.referenceCompatibility;
|
|
1232
|
+
if (scenario === "uncertain")
|
|
1233
|
+
review.outcome = "uncertain";
|
|
1234
|
+
if (scenario === "missing-scope")
|
|
1235
|
+
review.scopeCoverage.pop();
|
|
1236
|
+
if (scenario === "duplicate-scope")
|
|
1237
|
+
review.scopeCoverage[1] = review.scopeCoverage[0];
|
|
1238
|
+
if (scenario === "before-only")
|
|
1239
|
+
review.scopeCoverage[0].evidence[0].sourceId =
|
|
1240
|
+
review.scopeCoverage[0].evidence[0].sourceId.replace("after", "before");
|
|
1241
|
+
if (scenario === "fractional-line")
|
|
1242
|
+
review.scopeCoverage[0].evidence[0].startLine = 1.5;
|
|
1243
|
+
}
|
|
1244
|
+
}, calls), /failed independent review/u);
|
|
1245
|
+
assert.equal(calls.length, 4, scenario);
|
|
1246
|
+
}
|
|
1247
|
+
}
|
|
1248
|
+
finally {
|
|
1249
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1250
|
+
}
|
|
1251
|
+
});
|
|
1252
|
+
test("complete snapshot limits reject before the first model call and never silently truncate critical evidence", async () => {
|
|
1253
|
+
const f = await fixture(true);
|
|
1254
|
+
const calls = [];
|
|
1255
|
+
try {
|
|
1256
|
+
await assert.rejects(run(f, { maximum: 2 }, calls), /complete reference-grounding sources exceed/u);
|
|
1257
|
+
assert.equal(calls.length, 0);
|
|
1258
|
+
}
|
|
1259
|
+
finally {
|
|
1260
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1261
|
+
}
|
|
1262
|
+
});
|
|
1263
|
+
test("private analyst prose cannot smuggle reference code or identifiers across the writer boundary", async () => {
|
|
1264
|
+
const f = await fixture();
|
|
1265
|
+
try {
|
|
1266
|
+
for (const canary of [privateCode, f.referenceCommit, "private-reference-after-0"]) {
|
|
1267
|
+
const calls = [];
|
|
1268
|
+
await assert.rejects(run(f, {
|
|
1269
|
+
maximum: 2,
|
|
1270
|
+
mutate: (role, output) => {
|
|
1271
|
+
if (role === "repository-analyst")
|
|
1272
|
+
output.summary = canary;
|
|
1273
|
+
}
|
|
1274
|
+
}, calls), /private reference material leaked into the analyst/u);
|
|
1275
|
+
assert.equal(calls.length, 1);
|
|
1276
|
+
}
|
|
1277
|
+
}
|
|
1278
|
+
finally {
|
|
1279
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1280
|
+
}
|
|
1281
|
+
});
|
|
1282
|
+
test("analyst line references stay bound to their original files when selected context sorts before them", async () => {
|
|
1283
|
+
const f = await fixture();
|
|
1284
|
+
const calls = [];
|
|
1285
|
+
try {
|
|
1286
|
+
const result = await run(f, {
|
|
1287
|
+
maximum: 0,
|
|
1288
|
+
alwaysNarrow: true,
|
|
1289
|
+
mutate: (role, output) => {
|
|
1290
|
+
if (role === "repository-analyst")
|
|
1291
|
+
output.relevantPaths = ["src/wire.js", "src/wire.test.js", "src/a-context.js"];
|
|
1292
|
+
}
|
|
1293
|
+
}, calls);
|
|
1294
|
+
const grounding = result.referenceGrounding;
|
|
1295
|
+
assert.ok(result.contextSources.some((source) => source.path === "src/a-context.js"));
|
|
1296
|
+
assert.equal(new Set(grounding.sources.map((source) => source.sourceId)).size, grounding.sources.length);
|
|
1297
|
+
for (const reference of grounding.scope.clauses.flatMap((clause) => clause.evidence)) {
|
|
1298
|
+
const source = grounding.sources.find((entry) => entry.sourceId === reference.sourceId);
|
|
1299
|
+
assert.equal(source?.path, "src/wire.js");
|
|
1300
|
+
}
|
|
1301
|
+
assertPrivate(calls, f);
|
|
1302
|
+
}
|
|
1303
|
+
finally {
|
|
1304
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1305
|
+
}
|
|
1306
|
+
});
|
|
1307
|
+
test("empty files and explicit absent snapshots cannot serve as behavioral line evidence", async () => {
|
|
1308
|
+
const f = await fixture();
|
|
1309
|
+
try {
|
|
1310
|
+
for (const file of ["src/empty.js", "src/added.js"]) {
|
|
1311
|
+
const withSource = {
|
|
1312
|
+
...f,
|
|
1313
|
+
seed: {
|
|
1314
|
+
...f.seed,
|
|
1315
|
+
capabilityEvidence: [
|
|
1316
|
+
...f.seed.capabilityEvidence,
|
|
1317
|
+
{ kind: "symbol", id: "additional-source", path: file }
|
|
1318
|
+
]
|
|
1319
|
+
}
|
|
1320
|
+
};
|
|
1321
|
+
const calls = [];
|
|
1322
|
+
await assert.rejects(run(withSource, {
|
|
1323
|
+
maximum: 0,
|
|
1324
|
+
mutate: (role, output, packet) => {
|
|
1325
|
+
if (role !== "repository-analyst")
|
|
1326
|
+
return;
|
|
1327
|
+
const source = packet.privateReferenceEvidence.sources.find((entry) => entry.path === file && entry.phase === "before");
|
|
1328
|
+
assert.equal(source.content, file === "src/empty.js" ? "" : null);
|
|
1329
|
+
const scope = output.referenceScope;
|
|
1330
|
+
scope.clauses[0].evidence[0] = {
|
|
1331
|
+
sourceId: source.sourceId,
|
|
1332
|
+
startLine: 1,
|
|
1333
|
+
endLine: 1
|
|
1334
|
+
};
|
|
1335
|
+
}
|
|
1336
|
+
}, calls), /scope is incomplete, malformed, or lacks valid pinned source line evidence/u);
|
|
1337
|
+
assert.equal(calls.length, 1);
|
|
1338
|
+
}
|
|
1339
|
+
}
|
|
1340
|
+
finally {
|
|
1341
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1342
|
+
}
|
|
1343
|
+
});
|
|
1344
|
+
test("private compatibility findings remain in private history but are removed from writer revision feedback", async () => {
|
|
1345
|
+
const f = await fixture();
|
|
1346
|
+
const calls = [];
|
|
1347
|
+
try {
|
|
1348
|
+
const result = await run(f, {
|
|
1349
|
+
maximum: 1,
|
|
1350
|
+
mutate: (role, output, packet) => {
|
|
1351
|
+
if (role === "specification-critic-b" && packet.visible?.requestText === broadRequest) {
|
|
1352
|
+
output.detail = `${privateCode} ${f.referenceCommit}`;
|
|
1353
|
+
output.findings = [
|
|
1354
|
+
{
|
|
1355
|
+
axis: "unsupported-assumption",
|
|
1356
|
+
severity: "blocker",
|
|
1357
|
+
detail: `Use private-reference-after-0: ${privateTest}`
|
|
1358
|
+
}
|
|
1359
|
+
];
|
|
1360
|
+
}
|
|
1361
|
+
}
|
|
1362
|
+
}, calls);
|
|
1363
|
+
assert.ok(result.visible.lane === "repository-agent");
|
|
1364
|
+
assert.equal(result.visible.requestText, narrowRequest);
|
|
1365
|
+
assert.ok(result.specificationRevisions?.[0]?.reviews[1]?.detail.includes(privateCode));
|
|
1366
|
+
const revision = calls.find((call) => call.foundryRole === "specification-writer" &&
|
|
1367
|
+
call.operationId.includes(":specification-revision:1:"));
|
|
1368
|
+
assert.ok(revision?.input.includes("behavioralReviewFeedback"));
|
|
1369
|
+
assertPrivate(calls, f);
|
|
1370
|
+
}
|
|
1371
|
+
finally {
|
|
1372
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1373
|
+
}
|
|
1374
|
+
});
|
|
1375
|
+
// Frozen, small calibration cases based on the v5 unsupported media-type request
|
|
1376
|
+
// and the two broader integration unknowns returned by the v6 analyst. Model
|
|
1377
|
+
// responses are synthetic here; these tests establish the host's gate behavior.
|
|
1378
|
+
const integrationQuestions = [
|
|
1379
|
+
"Which downstream HTTP status and error body are exposed when newly eligible untyped responses fail buffering or SSE parsing? The available evidence establishes normalization failure but not endpoint-level error translation.",
|
|
1380
|
+
"Do all native Responses entry points reach this normalization flow? The available evidence establishes the finalization behavior but does not include its provider and endpoint call sites."
|
|
1381
|
+
];
|
|
1382
|
+
const narrowExclusion = "This task covers the shared normalizer; endpoint HTTP error translation and auditing every provider call site are outside this change.";
|
|
1383
|
+
const setQuestions = (output, questions) => {
|
|
1384
|
+
output.referenceScope.unresolvedQuestions = [...questions];
|
|
1385
|
+
};
|
|
1386
|
+
const excludedQuestions = (packet) => (packet.privateReferenceEvidence?.pendingQuestions ?? []).map((question) => ({
|
|
1387
|
+
questionId: question.id,
|
|
1388
|
+
outcome: "not-required",
|
|
1389
|
+
detail: "The shared normalizer's required behavior is determined independently of this broader integration question.",
|
|
1390
|
+
affectedScopeIds: [],
|
|
1391
|
+
evidence: [],
|
|
1392
|
+
visibleScopeExclusion: {
|
|
1393
|
+
quotation: narrowExclusion,
|
|
1394
|
+
rationale: "The visible task explicitly excludes the outer endpoint error mapping and a repository-wide provider call-site audit, while retaining all required normalization and preservation behavior."
|
|
1395
|
+
}
|
|
1396
|
+
}));
|
|
1397
|
+
const setResolutions = (output, resolutions) => {
|
|
1398
|
+
output.referenceCompatibility.questionResolutions =
|
|
1399
|
+
resolutions;
|
|
1400
|
+
};
|
|
1401
|
+
test("paired calibration rejects v5 text/plain expansion but accepts narrow v6 scope with justified outside-scope questions", async () => {
|
|
1402
|
+
const f = await fixture();
|
|
1403
|
+
try {
|
|
1404
|
+
for (const overbroad of [true, false]) {
|
|
1405
|
+
const calls = [];
|
|
1406
|
+
const result = run(f, {
|
|
1407
|
+
maximum: 0,
|
|
1408
|
+
alwaysBroad: overbroad,
|
|
1409
|
+
alwaysNarrow: !overbroad,
|
|
1410
|
+
mutate: (role, output, packet) => {
|
|
1411
|
+
if (role === "repository-analyst")
|
|
1412
|
+
setQuestions(output, integrationQuestions);
|
|
1413
|
+
if (role === "specification-writer")
|
|
1414
|
+
output.constraints = [narrowExclusion];
|
|
1415
|
+
if (role === "specification-critic-b")
|
|
1416
|
+
setResolutions(output, excludedQuestions(packet));
|
|
1417
|
+
}
|
|
1418
|
+
}, calls);
|
|
1419
|
+
if (overbroad) {
|
|
1420
|
+
await assert.rejects(result, /failed independent review/u);
|
|
1421
|
+
}
|
|
1422
|
+
else {
|
|
1423
|
+
const accepted = await result;
|
|
1424
|
+
assert.ok(accepted.visible.lane === "repository-agent");
|
|
1425
|
+
assert.equal(accepted.visible.requestText, narrowRequest);
|
|
1426
|
+
assert.equal(accepted.reviews[1]?.referenceCompatibility?.questionResolutions?.length, 2);
|
|
1427
|
+
assert.equal(accepted.modelCalls.length, 4);
|
|
1428
|
+
}
|
|
1429
|
+
assert.equal(calls.length, 4, "questions use the existing writer and critic calls");
|
|
1430
|
+
assertPrivate(calls, f);
|
|
1431
|
+
for (const call of calls.filter((entry) => entry.foundryRole === "specification-writer" ||
|
|
1432
|
+
entry.foundryRole === "specification-critic-a")) {
|
|
1433
|
+
for (const question of integrationQuestions)
|
|
1434
|
+
assert.equal(call.input.includes(question), false);
|
|
1435
|
+
}
|
|
1436
|
+
}
|
|
1437
|
+
}
|
|
1438
|
+
finally {
|
|
1439
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1440
|
+
}
|
|
1441
|
+
});
|
|
1442
|
+
test("required MIME uncertainty must be answered with pinned evidence and cannot be dismissed as outside scope", async () => {
|
|
1443
|
+
const f = await fixture();
|
|
1444
|
+
try {
|
|
1445
|
+
for (const disposition of ["unresolved", "not-required", "answered"]) {
|
|
1446
|
+
const calls = [];
|
|
1447
|
+
const result = run(f, {
|
|
1448
|
+
maximum: 0,
|
|
1449
|
+
alwaysNarrow: true,
|
|
1450
|
+
mutate: (role, output, packet) => {
|
|
1451
|
+
if (role === "repository-analyst")
|
|
1452
|
+
setQuestions(output, ["Must text/plain containing SSE remain unchanged?"]);
|
|
1453
|
+
if (role === "specification-writer")
|
|
1454
|
+
output.constraints = [narrowExclusion];
|
|
1455
|
+
if (role === "specification-critic-b") {
|
|
1456
|
+
setResolutions(output, [
|
|
1457
|
+
{
|
|
1458
|
+
questionId: packet.privateReferenceEvidence.pendingQuestions[0].id,
|
|
1459
|
+
outcome: disposition,
|
|
1460
|
+
detail: disposition === "answered"
|
|
1461
|
+
? "The media-type eligibility predicate bypasses text/plain, which the visible task explicitly preserves."
|
|
1462
|
+
: "MIME eligibility affects a required preservation obligation.",
|
|
1463
|
+
affectedScopeIds: ["preserve-text-plain"],
|
|
1464
|
+
evidence: disposition === "answered" ? [evidence(packet, "after", 3)] : [],
|
|
1465
|
+
...(disposition === "not-required"
|
|
1466
|
+
? {
|
|
1467
|
+
visibleScopeExclusion: {
|
|
1468
|
+
quotation: narrowExclusion,
|
|
1469
|
+
rationale: "Treat this uncertainty as outside scope."
|
|
1470
|
+
}
|
|
1471
|
+
}
|
|
1472
|
+
: {})
|
|
1473
|
+
}
|
|
1474
|
+
]);
|
|
1475
|
+
}
|
|
1476
|
+
}
|
|
1477
|
+
}, calls);
|
|
1478
|
+
if (disposition === "answered")
|
|
1479
|
+
assert.equal((await result).modelCalls.length, 4);
|
|
1480
|
+
else
|
|
1481
|
+
await assert.rejects(result, /questions require complete evidence-backed answers or justified visible-scope exclusions/u);
|
|
1482
|
+
assert.equal(calls.length, 4);
|
|
1483
|
+
}
|
|
1484
|
+
}
|
|
1485
|
+
finally {
|
|
1486
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1487
|
+
}
|
|
1488
|
+
});
|
|
1489
|
+
test("a writer cannot exclude mandatory MIME behavior to silence a pending question", async () => {
|
|
1490
|
+
const f = await fixture();
|
|
1491
|
+
const calls = [];
|
|
1492
|
+
try {
|
|
1493
|
+
await assert.rejects(run(f, {
|
|
1494
|
+
maximum: 0,
|
|
1495
|
+
alwaysNarrow: true,
|
|
1496
|
+
mutate: (role, output, packet) => {
|
|
1497
|
+
if (role === "repository-analyst")
|
|
1498
|
+
setQuestions(output, ["Must text/plain containing SSE remain unchanged?"]);
|
|
1499
|
+
if (role === "specification-writer") {
|
|
1500
|
+
output.requestText = `Repair response normalization. ${changed} Preserve response metadata. MIME eligibility and text/plain preservation are outside this task.`;
|
|
1501
|
+
output.constraints = [narrowExclusion];
|
|
1502
|
+
}
|
|
1503
|
+
if (role === "specification-critic-b") {
|
|
1504
|
+
// Even a complete not-required record must not override the independent
|
|
1505
|
+
// finding that the visible task drops a mandatory grounded scope clause.
|
|
1506
|
+
setResolutions(output, excludedQuestions(packet));
|
|
1507
|
+
const compatibility = output.referenceCompatibility;
|
|
1508
|
+
compatibility.scopeCoverage.find((entry) => entry.scopeId === "preserve-text-plain").outcome = "contradictory";
|
|
1509
|
+
}
|
|
1510
|
+
}
|
|
1511
|
+
}, calls), /visible requirements are contradictory, uncertain, or lack complete pinned reference compatibility evidence/u);
|
|
1512
|
+
assert.equal(calls.length, 4);
|
|
1513
|
+
}
|
|
1514
|
+
finally {
|
|
1515
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1516
|
+
}
|
|
1517
|
+
});
|
|
1518
|
+
test("missing, duplicate, invented, stale, and malformed private question resolutions fail closed", async () => {
|
|
1519
|
+
const f = await fixture();
|
|
1520
|
+
try {
|
|
1521
|
+
for (const scenario of [
|
|
1522
|
+
"missing",
|
|
1523
|
+
"duplicate",
|
|
1524
|
+
"invented",
|
|
1525
|
+
"extra",
|
|
1526
|
+
"unresolved",
|
|
1527
|
+
"blank-detail",
|
|
1528
|
+
"blank-quote",
|
|
1529
|
+
"stale-quote",
|
|
1530
|
+
"joined-fields",
|
|
1531
|
+
"blank-rationale",
|
|
1532
|
+
"unknown-scope",
|
|
1533
|
+
"duplicate-scope",
|
|
1534
|
+
"answered-no-evidence",
|
|
1535
|
+
"answered-bad-evidence"
|
|
1536
|
+
]) {
|
|
1537
|
+
const calls = [];
|
|
1538
|
+
await assert.rejects(run(f, {
|
|
1539
|
+
maximum: 0,
|
|
1540
|
+
alwaysNarrow: true,
|
|
1541
|
+
mutate: (role, output, packet) => {
|
|
1542
|
+
if (role === "repository-analyst")
|
|
1543
|
+
setQuestions(output, integrationQuestions);
|
|
1544
|
+
if (role === "specification-writer")
|
|
1545
|
+
output.constraints = [narrowExclusion];
|
|
1546
|
+
if (role !== "specification-critic-b")
|
|
1547
|
+
return;
|
|
1548
|
+
const resolutions = excludedQuestions(packet);
|
|
1549
|
+
const first = resolutions[0];
|
|
1550
|
+
if (scenario === "missing")
|
|
1551
|
+
resolutions.pop();
|
|
1552
|
+
if (scenario === "duplicate")
|
|
1553
|
+
resolutions[1] = first;
|
|
1554
|
+
if (scenario === "invented")
|
|
1555
|
+
first.questionId = "private-reference-question-999";
|
|
1556
|
+
if (scenario === "extra")
|
|
1557
|
+
resolutions.push({ ...first, questionId: "private-reference-question-999" });
|
|
1558
|
+
if (scenario === "unresolved")
|
|
1559
|
+
first.outcome = "unresolved";
|
|
1560
|
+
if (scenario === "blank-detail")
|
|
1561
|
+
first.detail = " ";
|
|
1562
|
+
if (scenario === "blank-quote")
|
|
1563
|
+
first.visibleScopeExclusion.quotation = " ";
|
|
1564
|
+
if (scenario === "stale-quote")
|
|
1565
|
+
first.visibleScopeExclusion.quotation =
|
|
1566
|
+
"An exclusion stated only by a previous draft.";
|
|
1567
|
+
if (scenario === "joined-fields")
|
|
1568
|
+
first.visibleScopeExclusion.quotation = `${packet.visible.requestText}\n${packet.visible.constraints[0]}`;
|
|
1569
|
+
if (scenario === "blank-rationale")
|
|
1570
|
+
first.visibleScopeExclusion.rationale = " ";
|
|
1571
|
+
if (scenario === "unknown-scope")
|
|
1572
|
+
first.affectedScopeIds = ["unknown-scope"];
|
|
1573
|
+
if (scenario === "duplicate-scope")
|
|
1574
|
+
first.affectedScopeIds = ["preserve-text-plain", "preserve-text-plain"];
|
|
1575
|
+
if (scenario === "answered-no-evidence" || scenario === "answered-bad-evidence") {
|
|
1576
|
+
first.outcome = "answered";
|
|
1577
|
+
first.affectedScopeIds = ["preserve-text-plain"];
|
|
1578
|
+
delete first.visibleScopeExclusion;
|
|
1579
|
+
first.evidence =
|
|
1580
|
+
scenario === "answered-no-evidence"
|
|
1581
|
+
? []
|
|
1582
|
+
: [{ ...evidence(packet, "after", 3), endLine: 9999 }];
|
|
1583
|
+
}
|
|
1584
|
+
setResolutions(output, resolutions);
|
|
1585
|
+
}
|
|
1586
|
+
}, calls), /questions require complete evidence-backed answers or justified visible-scope exclusions/u);
|
|
1587
|
+
assert.equal(calls.length, 4, scenario);
|
|
1588
|
+
}
|
|
1589
|
+
}
|
|
1590
|
+
finally {
|
|
1591
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1592
|
+
}
|
|
1593
|
+
});
|
|
1594
|
+
test("question IDs remain frozen across bounded revisions and private resolution feedback never reaches the writer", async () => {
|
|
1595
|
+
const f = await fixture();
|
|
1596
|
+
const calls = [];
|
|
1597
|
+
const reviewedIds = [];
|
|
1598
|
+
try {
|
|
1599
|
+
const result = await run(f, {
|
|
1600
|
+
maximum: 1,
|
|
1601
|
+
alwaysNarrow: true,
|
|
1602
|
+
mutate: (role, output, packet) => {
|
|
1603
|
+
if (role === "repository-analyst")
|
|
1604
|
+
setQuestions(output, integrationQuestions);
|
|
1605
|
+
if (role === "specification-writer")
|
|
1606
|
+
output.constraints = [narrowExclusion];
|
|
1607
|
+
if (role !== "specification-critic-b")
|
|
1608
|
+
return;
|
|
1609
|
+
const resolutions = excludedQuestions(packet);
|
|
1610
|
+
reviewedIds.push(resolutions.map((entry) => entry.questionId));
|
|
1611
|
+
if (reviewedIds.length === 1) {
|
|
1612
|
+
resolutions[0].outcome = "unresolved";
|
|
1613
|
+
resolutions[0].detail = `private-reference-question-1 ${privateCode} ${f.referenceCommit}`;
|
|
1614
|
+
}
|
|
1615
|
+
setResolutions(output, resolutions);
|
|
1616
|
+
}
|
|
1617
|
+
}, calls);
|
|
1618
|
+
assert.equal(result.modelCalls.length, 7);
|
|
1619
|
+
assert.deepEqual(reviewedIds, [
|
|
1620
|
+
["private-reference-question-1", "private-reference-question-2"],
|
|
1621
|
+
["private-reference-question-1", "private-reference-question-2"]
|
|
1622
|
+
]);
|
|
1623
|
+
assert.deepEqual(result.specificationRevisions?.map((entry) => entry.rejectionReasons.length > 0), [true, false]);
|
|
1624
|
+
assertPrivate(calls, f);
|
|
1625
|
+
const revision = calls.find((call) => call.foundryRole === "specification-writer" &&
|
|
1626
|
+
call.operationId.includes(":specification-revision:1:"));
|
|
1627
|
+
assert.ok(revision.input.includes("behavioralReviewFeedback"));
|
|
1628
|
+
assert.equal(revision.input.includes("private-reference-question-"), false);
|
|
1629
|
+
}
|
|
1630
|
+
finally {
|
|
1631
|
+
await rm(f.root, { recursive: true, force: true });
|
|
1632
|
+
}
|
|
1633
|
+
});
|