@velum-labs/routekit-eval-setup 1.3.2 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/authoring-responses-request.d.ts +5 -0
- package/dist/adapters/authoring-responses-request.js +38 -0
- package/dist/adapters/evaluation-evidence-freshness.d.ts +7 -0
- package/dist/adapters/evaluation-evidence-freshness.js +76 -0
- package/dist/adapters/git-task-history.d.ts +67 -0
- package/dist/adapters/git-task-history.js +171 -0
- package/dist/adapters/integrated-repository-history.d.ts +21 -0
- package/dist/adapters/integrated-repository-history.js +175 -0
- package/dist/adapters/repository-command-diagnostic.d.ts +8 -0
- package/dist/adapters/repository-command-diagnostic.js +46 -0
- package/dist/adapters/repository-command-evidence.d.ts +13 -0
- package/dist/adapters/repository-command-evidence.js +102 -0
- package/dist/adapters/repository-command-runner.d.ts +238 -0
- package/dist/adapters/repository-command-runner.js +1483 -0
- package/dist/adapters/repository-import-context.d.ts +47 -0
- package/dist/adapters/repository-import-context.js +469 -0
- package/dist/adapters/repository-node-test-reporter.d.ts +3 -0
- package/dist/adapters/repository-node-test-reporter.js +27 -0
- package/dist/adapters/repository-review-evidence.d.ts +39 -0
- package/dist/adapters/repository-review-evidence.js +632 -0
- package/dist/adapters/repository-seed-selection.d.ts +7 -0
- package/dist/adapters/repository-seed-selection.js +79 -0
- package/dist/adapters/repository-solution-edits.d.ts +49 -0
- package/dist/adapters/repository-solution-edits.js +136 -0
- package/dist/adapters/repository-vitest-phase-adapter.d.ts +8 -0
- package/dist/adapters/repository-vitest-phase-adapter.js +310 -0
- package/dist/adapters/repository-vitest-reporter.d.ts +24 -0
- package/dist/adapters/repository-vitest-reporter.js +314 -0
- package/dist/adapters/strict-authoring-schema.d.ts +5 -0
- package/dist/adapters/strict-authoring-schema.js +158 -0
- package/dist/adapters/test-discovery.d.ts +30 -0
- package/dist/adapters/test-discovery.js +124 -0
- package/dist/adapters/typescript-repository-index.d.ts +51 -0
- package/dist/adapters/typescript-repository-index.js +226 -0
- package/dist/agentic-capabilities-protocol.d.ts +1373 -0
- package/dist/agentic-capabilities-protocol.js +786 -0
- package/dist/case-checkpoint-store.d.ts +29 -0
- package/dist/case-checkpoint-store.js +133 -0
- package/dist/case-pipeline-protocol-v2.d.ts +184 -0
- package/dist/case-pipeline-protocol-v2.js +193 -0
- package/dist/case-pipeline-protocol.d.ts +2626 -0
- package/dist/case-pipeline-protocol.js +371 -0
- package/dist/effect-api.d.ts +74 -10
- package/dist/effect-api.js +56 -6
- package/dist/errors.d.ts +31 -0
- package/dist/errors.js +10 -0
- package/dist/eval-capability-execution-envelope.d.ts +64 -0
- package/dist/eval-capability-execution-envelope.js +98 -0
- package/dist/eval-capability-policy.d.ts +90 -0
- package/dist/eval-capability-policy.js +107 -0
- package/dist/eval-event-log.d.ts +140 -0
- package/dist/eval-event-log.js +220 -0
- package/dist/evaluation-authoring-policy.d.ts +18 -0
- package/dist/evaluation-authoring-policy.js +19 -0
- package/dist/evaluation-authoring-validation.d.ts +22 -0
- package/dist/evaluation-authoring-validation.js +72 -0
- package/dist/evaluation-evidence.d.ts +20 -0
- package/dist/evaluation-evidence.js +319 -0
- package/dist/evaluation-grader-calibration-protocol.d.ts +108 -0
- package/dist/evaluation-grader-calibration-protocol.js +80 -0
- package/dist/evaluation-grader-calibration.d.ts +18 -0
- package/dist/evaluation-grader-calibration.js +334 -0
- package/dist/evaluation-grading-policy.d.ts +24 -0
- package/dist/evaluation-grading-policy.js +54 -0
- package/dist/evaluation-proposal-policy.d.ts +4 -0
- package/dist/evaluation-proposal-policy.js +91 -0
- package/dist/evaluation-source-retrieval.d.ts +68 -0
- package/dist/evaluation-source-retrieval.js +513 -0
- package/dist/evaluation-structure-policy.d.ts +29 -0
- package/dist/evaluation-structure-policy.js +138 -0
- package/dist/index.d.ts +124 -17
- package/dist/index.js +69 -11
- package/dist/inspection.js +2 -3
- package/dist/project-artifacts.d.ts +7 -2
- package/dist/project-artifacts.js +49 -136
- package/dist/project-authoring.d.ts +66 -5
- package/dist/project-authoring.js +783 -109
- package/dist/project-contracts.d.ts +419 -84
- package/dist/project-contracts.js +160 -52
- package/dist/project-store.js +2 -1
- package/dist/project-workflow.d.ts +5 -4
- package/dist/project-workflow.js +154 -35
- package/dist/repository-adversary-protocol.d.ts +64 -0
- package/dist/repository-adversary-protocol.js +105 -0
- package/dist/repository-behavior-protocol.d.ts +188 -0
- package/dist/repository-behavior-protocol.js +202 -0
- package/dist/repository-benchmark-protocol.d.ts +487 -0
- package/dist/repository-benchmark-protocol.js +96 -0
- package/dist/repository-execution-protocol.d.ts +150 -0
- package/dist/repository-execution-protocol.js +38 -0
- package/dist/repository-fixture-instructions.d.ts +3 -0
- package/dist/repository-fixture-instructions.js +91 -0
- package/dist/repository-fixture-protocol.d.ts +79 -0
- package/dist/repository-fixture-protocol.js +79 -0
- package/dist/repository-foundry-plan-protocol.d.ts +118 -0
- package/dist/repository-foundry-plan-protocol.js +296 -0
- package/dist/repository-foundry-progress-protocol.d.ts +52 -0
- package/dist/repository-foundry-progress-protocol.js +52 -0
- package/dist/repository-improvement-protocol.d.ts +100 -0
- package/dist/repository-improvement-protocol.js +106 -0
- package/dist/repository-language-model-protocol.d.ts +43 -0
- package/dist/repository-language-model-protocol.js +146 -0
- package/dist/repository-oracle-coverage-protocol.d.ts +18 -0
- package/dist/repository-oracle-coverage-protocol.js +39 -0
- package/dist/repository-oracle-execution-binding.d.ts +27 -0
- package/dist/repository-oracle-execution-binding.js +59 -0
- package/dist/repository-oracle-protocol.d.ts +230 -0
- package/dist/repository-oracle-protocol.js +156 -0
- package/dist/repository-oracle-scope-policy.d.ts +22 -0
- package/dist/repository-oracle-scope-policy.js +92 -0
- package/dist/repository-quality-policy.d.ts +15 -0
- package/dist/repository-quality-policy.js +357 -0
- package/dist/repository-routing-benchmark-protocol.d.ts +176 -0
- package/dist/repository-routing-benchmark-protocol.js +103 -0
- package/dist/repository-routing-model-protocol.d.ts +36 -0
- package/dist/repository-routing-model-protocol.js +89 -0
- package/dist/repository-routing-plan-protocol.d.ts +112 -0
- package/dist/repository-routing-plan-protocol.js +58 -0
- package/dist/repository-routing-quality-policy.d.ts +9 -0
- package/dist/repository-routing-quality-policy.js +191 -0
- package/dist/repository-seed-qualification-progress-protocol.d.ts +205 -0
- package/dist/repository-seed-qualification-progress-protocol.js +28 -0
- package/dist/repository-semantic-calibration-protocol.d.ts +768 -0
- package/dist/repository-semantic-calibration-protocol.js +276 -0
- package/dist/repository-semantic-calibration.d.ts +163 -0
- package/dist/repository-semantic-calibration.js +581 -0
- package/dist/repository-specification-contract-facts-protocol.d.ts +224 -0
- package/dist/repository-specification-contract-facts-protocol.js +276 -0
- package/dist/repository-specification-critique-protocol.d.ts +189 -0
- package/dist/repository-specification-critique-protocol.js +103 -0
- package/dist/repository-task-family-protocol.d.ts +24 -0
- package/dist/repository-task-family-protocol.js +37 -0
- package/dist/repository-task-seed-protocol.d.ts +384 -0
- package/dist/repository-task-seed-protocol.js +236 -0
- package/dist/repository-trajectory-protocol.d.ts +20 -0
- package/dist/repository-trajectory-protocol.js +42 -0
- package/dist/service.js +1 -1
- package/dist/services/adversary/service.d.ts +64 -0
- package/dist/services/adversary/service.js +330 -0
- package/dist/services/benchmark-compiler/service.d.ts +450 -0
- package/dist/services/benchmark-compiler/service.js +9 -0
- package/dist/services/budgeted-model/service.d.ts +118 -0
- package/dist/services/budgeted-model/service.js +460 -0
- package/dist/services/case-authoring/service.d.ts +163 -0
- package/dist/services/case-authoring/service.js +1456 -0
- package/dist/services/case-finalization/service.d.ts +283 -0
- package/dist/services/case-finalization/service.js +370 -0
- package/dist/services/case-generation/service.d.ts +619 -0
- package/dist/services/case-generation/service.js +2628 -0
- package/dist/services/case-pipeline/service.d.ts +31 -0
- package/dist/services/case-pipeline/service.js +485 -0
- package/dist/services/case-pipeline-v2/service.d.ts +70 -0
- package/dist/services/case-pipeline-v2/service.js +477 -0
- package/dist/services/command-observability/service.d.ts +13 -0
- package/dist/services/command-observability/service.js +3 -0
- package/dist/services/dimension-labeling/service.d.ts +77 -0
- package/dist/services/dimension-labeling/service.js +188 -0
- package/dist/services/eval-candidate/service.d.ts +208 -0
- package/dist/services/eval-candidate/service.js +64 -0
- package/dist/services/eval-capabilities/service.d.ts +183 -0
- package/dist/services/eval-capabilities/service.js +1433 -0
- package/dist/services/eval-environment/service.d.ts +173 -0
- package/dist/services/eval-environment/service.js +127 -0
- package/dist/services/evidence-reconstruction/service.d.ts +36 -0
- package/dist/services/evidence-reconstruction/service.js +145 -0
- package/dist/services/fixture-builder/service.d.ts +62 -0
- package/dist/services/fixture-builder/service.js +36 -0
- package/dist/services/fixture-validation/service.d.ts +75 -0
- package/dist/services/fixture-validation/service.js +295 -0
- package/dist/services/foundry/service.d.ts +831 -0
- package/dist/services/foundry/service.js +442 -0
- package/dist/services/foundry-progress/service.d.ts +62 -0
- package/dist/services/foundry-progress/service.js +149 -0
- package/dist/services/foundry-v2/service.d.ts +54 -0
- package/dist/services/foundry-v2/service.js +28 -0
- package/dist/services/grounded-authoring/service.d.ts +126 -0
- package/dist/services/grounded-authoring/service.js +822 -0
- package/dist/services/historical-case/service.d.ts +722 -0
- package/dist/services/historical-case/service.js +177 -0
- package/dist/services/improvement-loop/service.d.ts +59 -0
- package/dist/services/improvement-loop/service.js +176 -0
- package/dist/services/language-model/service.d.ts +52 -0
- package/dist/services/language-model/service.js +194 -0
- package/dist/services/oracle-builder/service.d.ts +146 -0
- package/dist/services/oracle-builder/service.js +513 -0
- package/dist/services/oracle-coverage/service.d.ts +28 -0
- package/dist/services/oracle-coverage/service.js +50 -0
- package/dist/services/oracle-coverage-witness/service.d.ts +130 -0
- package/dist/services/oracle-coverage-witness/service.js +538 -0
- package/dist/services/pipeline-challenge/service.d.ts +551 -0
- package/dist/services/pipeline-challenge/service.js +427 -0
- package/dist/services/pipeline-controls/service.d.ts +130 -0
- package/dist/services/pipeline-controls/service.js +483 -0
- package/dist/services/pipeline-oracle/service.d.ts +8 -0
- package/dist/services/pipeline-oracle/service.js +256 -0
- package/dist/services/pipeline-seed/service.d.ts +298 -0
- package/dist/services/pipeline-seed/service.js +428 -0
- package/dist/services/pipeline-spec/service.d.ts +103 -0
- package/dist/services/pipeline-spec/service.js +619 -0
- package/dist/services/pipeline-tournament/service.d.ts +258 -0
- package/dist/services/pipeline-tournament/service.js +476 -0
- package/dist/services/quality-gate/service.d.ts +233 -0
- package/dist/services/quality-gate/service.js +136 -0
- package/dist/services/repository-bundle/service.d.ts +33 -0
- package/dist/services/repository-bundle/service.js +114 -0
- package/dist/services/repository-model/service.d.ts +105 -0
- package/dist/services/repository-model/service.js +250 -0
- package/dist/services/repository-public-artifact/service.d.ts +133 -0
- package/dist/services/repository-public-artifact/service.js +330 -0
- package/dist/services/routing-benchmark/service.d.ts +362 -0
- package/dist/services/routing-benchmark/service.js +96 -0
- package/dist/services/specification-critic/service.d.ts +92 -0
- package/dist/services/specification-critic/service.js +172 -0
- package/dist/services/task-family/service.d.ts +40 -0
- package/dist/services/task-family/service.js +55 -0
- package/dist/services/task-seed/service.d.ts +906 -0
- package/dist/services/task-seed/service.js +1406 -0
- package/dist/services/task-specification/service.d.ts +27 -0
- package/dist/services/task-specification/service.js +40 -0
- package/dist/services/trajectory-policy/service.d.ts +110 -0
- package/dist/services/trajectory-policy/service.js +216 -0
- package/dist/test/agentic-capabilities-protocol.test.d.ts +1 -0
- package/dist/test/agentic-capabilities-protocol.test.js +570 -0
- package/dist/test/agentic-capabilities.test.d.ts +1 -0
- package/dist/test/agentic-capabilities.test.js +1461 -0
- package/dist/test/agentic-environment.test.d.ts +1 -0
- package/dist/test/agentic-environment.test.js +213 -0
- package/dist/test/case-pipeline-foundation.test.d.ts +1 -0
- package/dist/test/case-pipeline-foundation.test.js +535 -0
- package/dist/test/case-pipeline-protocol-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-protocol-v2.test.js +124 -0
- package/dist/test/case-pipeline-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-v2.test.js +286 -0
- package/dist/test/case-pipeline.test.d.ts +1 -0
- package/dist/test/case-pipeline.test.js +851 -0
- package/dist/test/eval-capability-policy.test.d.ts +1 -0
- package/dist/test/eval-capability-policy.test.js +50 -0
- package/dist/test/eval-event-log.test.d.ts +1 -0
- package/dist/test/eval-event-log.test.js +125 -0
- package/dist/test/evaluation-evidence-freshness.test.d.ts +1 -0
- package/dist/test/evaluation-evidence-freshness.test.js +44 -0
- package/dist/test/evaluation-evidence.test.d.ts +1 -0
- package/dist/test/evaluation-evidence.test.js +230 -0
- package/dist/test/evaluation-grader-calibration.test.d.ts +1 -0
- package/dist/test/evaluation-grader-calibration.test.js +373 -0
- package/dist/test/evaluation-proposal-digest.test.d.ts +1 -0
- package/dist/test/evaluation-proposal-digest.test.js +187 -0
- package/dist/test/evaluation-source-retrieval.test.d.ts +1 -0
- package/dist/test/evaluation-source-retrieval.test.js +237 -0
- package/dist/test/evaluation-structure-policy.test.d.ts +1 -0
- package/dist/test/evaluation-structure-policy.test.js +196 -0
- package/dist/test/fixtures/repository-resource-panel.d.ts +39 -0
- package/dist/test/fixtures/repository-resource-panel.js +111 -0
- package/dist/test/fixtures/vitest-boundary-panel.d.ts +84 -0
- package/dist/test/fixtures/vitest-boundary-panel.js +120 -0
- package/dist/test/fixtures/vitest-phase-panel.d.ts +135 -0
- package/dist/test/fixtures/vitest-phase-panel.js +213 -0
- package/dist/test/fixtures/vitest-reporter-results.d.ts +76 -0
- package/dist/test/fixtures/vitest-reporter-results.js +94 -0
- package/dist/test/grounded-authoring.test.d.ts +1 -0
- package/dist/test/grounded-authoring.test.js +565 -0
- package/dist/test/integrated-repository-history.test.d.ts +1 -0
- package/dist/test/integrated-repository-history.test.js +227 -0
- package/dist/test/project-authoring.test.js +593 -43
- package/dist/test/project-workflow.test.js +419 -40
- package/dist/test/repository-authoring-artifacts.test.d.ts +1 -0
- package/dist/test/repository-authoring-artifacts.test.js +185 -0
- package/dist/test/repository-bundle.test.d.ts +1 -0
- package/dist/test/repository-bundle.test.js +52 -0
- package/dist/test/repository-case-generation.test.d.ts +1 -0
- package/dist/test/repository-case-generation.test.js +3465 -0
- package/dist/test/repository-command-diagnostic.test.d.ts +1 -0
- package/dist/test/repository-command-diagnostic.test.js +55 -0
- package/dist/test/repository-command-signals.test.d.ts +1 -0
- package/dist/test/repository-command-signals.test.js +124 -0
- package/dist/test/repository-fixture-scope-coverage.test.d.ts +1 -0
- package/dist/test/repository-fixture-scope-coverage.test.js +127 -0
- package/dist/test/repository-fixture-validation.test.d.ts +1 -0
- package/dist/test/repository-fixture-validation.test.js +362 -0
- package/dist/test/repository-foundry-progress.test.d.ts +1 -0
- package/dist/test/repository-foundry-progress.test.js +110 -0
- package/dist/test/repository-foundry-quality.test.d.ts +1 -0
- package/dist/test/repository-foundry-quality.test.js +1138 -0
- package/dist/test/repository-import-context.test.d.ts +1 -0
- package/dist/test/repository-import-context.test.js +354 -0
- package/dist/test/repository-model-authoring.test.d.ts +1 -0
- package/dist/test/repository-model-authoring.test.js +544 -0
- package/dist/test/repository-model.test.d.ts +1 -0
- package/dist/test/repository-model.test.js +2195 -0
- package/dist/test/repository-node-test-reporter.test.d.ts +1 -0
- package/dist/test/repository-node-test-reporter.test.js +104 -0
- package/dist/test/repository-oracle-concurrency.test.d.ts +1 -0
- package/dist/test/repository-oracle-concurrency.test.js +542 -0
- package/dist/test/repository-oracle-coverage-witness.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage-witness.test.js +511 -0
- package/dist/test/repository-oracle-coverage.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage.test.js +168 -0
- package/dist/test/repository-oracle-evidence.test.d.ts +1 -0
- package/dist/test/repository-oracle-evidence.test.js +185 -0
- package/dist/test/repository-oracle-plan.test.d.ts +1 -0
- package/dist/test/repository-oracle-plan.test.js +176 -0
- package/dist/test/repository-overlay-isolation.test.d.ts +1 -0
- package/dist/test/repository-overlay-isolation.test.js +85 -0
- package/dist/test/repository-preparation-cache.test.d.ts +1 -0
- package/dist/test/repository-preparation-cache.test.js +414 -0
- package/dist/test/repository-public-artifact.test.d.ts +1 -0
- package/dist/test/repository-public-artifact.test.js +273 -0
- package/dist/test/repository-qualification-diagnostics.test.d.ts +1 -0
- package/dist/test/repository-qualification-diagnostics.test.js +524 -0
- package/dist/test/repository-reference-authoring.test.d.ts +1 -0
- package/dist/test/repository-reference-authoring.test.js +1633 -0
- package/dist/test/repository-review-evidence-v2.test.d.ts +1 -0
- package/dist/test/repository-review-evidence-v2.test.js +183 -0
- package/dist/test/repository-review-evidence.test.d.ts +1 -0
- package/dist/test/repository-review-evidence.test.js +124 -0
- package/dist/test/repository-seed-exclusions.test.d.ts +1 -0
- package/dist/test/repository-seed-exclusions.test.js +96 -0
- package/dist/test/repository-seed-selection.test.d.ts +1 -0
- package/dist/test/repository-seed-selection.test.js +504 -0
- package/dist/test/repository-semantic-calibration.test.d.ts +1 -0
- package/dist/test/repository-semantic-calibration.test.js +688 -0
- package/dist/test/repository-solution-edits.test.d.ts +1 -0
- package/dist/test/repository-solution-edits.test.js +377 -0
- package/dist/test/repository-specification-budget.test.d.ts +1 -0
- package/dist/test/repository-specification-budget.test.js +171 -0
- package/dist/test/repository-specification-contract-checkpoint.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-checkpoint.test.js +228 -0
- package/dist/test/repository-specification-contract-facts.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-facts.test.js +177 -0
- package/dist/test/repository-trajectory-authoring.test.d.ts +1 -0
- package/dist/test/repository-trajectory-authoring.test.js +176 -0
- package/dist/test/repository-valid-control-plan.test.d.ts +1 -0
- package/dist/test/repository-valid-control-plan.test.js +45 -0
- package/dist/test/repository-vitest-phase.test.d.ts +1 -0
- package/dist/test/repository-vitest-phase.test.js +848 -0
- package/dist/test/repository-vitest-reporter.test.d.ts +1 -0
- package/dist/test/repository-vitest-reporter.test.js +158 -0
- package/dist/test/repository-workspace-build.test.d.ts +1 -0
- package/dist/test/repository-workspace-build.test.js +160 -0
- package/dist/test/strict-authoring-schema.test.d.ts +1 -0
- package/dist/test/strict-authoring-schema.test.js +169 -0
- package/package.json +48 -6
|
@@ -0,0 +1,581 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
import { Effect, Schema } from "effect";
|
|
3
|
+
import { strictAuthoringSchema } from "./adapters/strict-authoring-schema.js";
|
|
4
|
+
import { RepositoryFoundryError } from "./errors.js";
|
|
5
|
+
import { REPOSITORY_FOUNDRY_REQUEST_INPUT_CEILING } from "./repository-foundry-plan-protocol.js";
|
|
6
|
+
import { assertRepositoryFoundryModelPlanV1, RepositoryFoundryModelPlanV1, repositoryFoundryModelAssignmentV1 } from "./repository-language-model-protocol.js";
|
|
7
|
+
import { REPOSITORY_SEMANTIC_CALIBRATION_MAXIMUM_EXAMPLES, REPOSITORY_SEMANTIC_CALIBRATION_MAXIMUM_REPETITIONS, RepositorySemanticCalibrationExecutableCheckV1, RepositorySemanticCalibrationManifestV1, RepositorySemanticCalibrationObservationV1, RepositorySemanticCalibrationPlanV1, RepositorySemanticCalibrationReportV1 } from "./repository-semantic-calibration-protocol.js";
|
|
8
|
+
import { executeRepositorySemanticReviewV1, prepareRepositorySemanticReviewV1, repositorySemanticReviewIndependenceFindingsV1 } from "./services/case-generation/service.js";
|
|
9
|
+
const roles = ["solution-critic-a", "solution-critic-b"];
|
|
10
|
+
const sha256 = (text) => createHash("sha256").update(text).digest("hex");
|
|
11
|
+
const canonicalJson = (value) => {
|
|
12
|
+
if (Array.isArray(value))
|
|
13
|
+
return `[${value.map(canonicalJson).join(",")}]`;
|
|
14
|
+
if (value !== null && typeof value === "object") {
|
|
15
|
+
const record = value;
|
|
16
|
+
return `{${Object.keys(record)
|
|
17
|
+
.sort()
|
|
18
|
+
.filter((key) => record[key] !== undefined)
|
|
19
|
+
.map((key) => `${JSON.stringify(key)}:${canonicalJson(record[key])}`)
|
|
20
|
+
.join(",")}}`;
|
|
21
|
+
}
|
|
22
|
+
const encoded = JSON.stringify(value);
|
|
23
|
+
if (encoded === undefined)
|
|
24
|
+
throw new Error("calibration requires JSON data");
|
|
25
|
+
return encoded;
|
|
26
|
+
};
|
|
27
|
+
const digestOf = (value) => sha256(canonicalJson(value));
|
|
28
|
+
const requireValue = (condition, message) => {
|
|
29
|
+
if (!condition)
|
|
30
|
+
throw new Error(`semantic calibration: ${message}`);
|
|
31
|
+
};
|
|
32
|
+
const nonempty = (value) => value.trim().length > 0;
|
|
33
|
+
const digest = (value) => /^[a-f0-9]{64}$/u.test(value);
|
|
34
|
+
const identifier = (value) => /^[A-Za-z0-9][A-Za-z0-9._:-]{0,159}$/u.test(value);
|
|
35
|
+
const unique = (values) => values.every(nonempty) && new Set(values).size === values.length;
|
|
36
|
+
const sameIds = (left, right) => unique(left) && left.length === right.length && left.every((id) => right.includes(id));
|
|
37
|
+
const sourcePath = (value) => nonempty(value) &&
|
|
38
|
+
!value.includes("\\") &&
|
|
39
|
+
!value.includes("\0") &&
|
|
40
|
+
value.split("/").every((part) => !["", ".", "..", ".git"].includes(part));
|
|
41
|
+
const decode = (schema, value) => Schema.decodeUnknownSync(schema, { onExcessProperty: "error" })(value);
|
|
42
|
+
const freeze = (value) => {
|
|
43
|
+
if (value !== null && typeof value === "object") {
|
|
44
|
+
for (const entry of Object.values(value))
|
|
45
|
+
freeze(entry);
|
|
46
|
+
Object.freeze(value);
|
|
47
|
+
}
|
|
48
|
+
return value;
|
|
49
|
+
};
|
|
50
|
+
const detached = (value) => JSON.parse(JSON.stringify(value));
|
|
51
|
+
export const repositorySemanticCalibrationInputDigestV1 = (input) => digestOf(input);
|
|
52
|
+
export const repositorySemanticCalibrationExampleDigestV1 = (example) => digestOf({
|
|
53
|
+
id: example.id,
|
|
54
|
+
split: example.split,
|
|
55
|
+
repository: example.repository,
|
|
56
|
+
input: example.input,
|
|
57
|
+
adjudication: example.adjudication
|
|
58
|
+
});
|
|
59
|
+
const validateExample = (example) => {
|
|
60
|
+
const input = example.input;
|
|
61
|
+
requireValue(identifier(example.id), "example id must be a bounded identifier");
|
|
62
|
+
requireValue(nonempty(example.repository.id), "repository identity is missing");
|
|
63
|
+
requireValue(/^(?:[a-f0-9]{40}|[a-f0-9]{64})$/u.test(example.repository.commit), "repository commit must be an exact Git object id");
|
|
64
|
+
requireValue(input.visible.lane === "repository-agent" &&
|
|
65
|
+
input.visible.checkoutCommit === example.repository.commit &&
|
|
66
|
+
input.repositoryEnvironment.commit === example.repository.commit &&
|
|
67
|
+
input.repositoryEnvironment.localImports.commit === example.repository.commit, "development source and visible task must bind one exact repository commit");
|
|
68
|
+
requireValue(input.visible.lane === "repository-agent" && nonempty(input.visible.requestText), "a nonempty visible repository task is required");
|
|
69
|
+
requireValue(input.targetBehavior.length > 0 &&
|
|
70
|
+
unique(input.targetBehavior.map((clause) => clause.id)) &&
|
|
71
|
+
input.targetBehavior.every((clause) => nonempty(clause.description)), "explicit unique behavior clauses are required");
|
|
72
|
+
const behaviorIds = input.targetBehavior.map((clause) => clause.id);
|
|
73
|
+
const sources = [
|
|
74
|
+
...input.preChangeSources,
|
|
75
|
+
...input.repositoryEnvironment.sources,
|
|
76
|
+
...input.repositoryEnvironment.localImports.sources
|
|
77
|
+
];
|
|
78
|
+
requireValue(input.preChangeSources.length > 0 && sources.length <= 128, "development source context must be nonempty and bounded");
|
|
79
|
+
const sourceDigests = new Map();
|
|
80
|
+
for (const source of sources) {
|
|
81
|
+
requireValue(sourcePath(source.path), "source path must be canonical and repository-relative");
|
|
82
|
+
const sourceDigest = sha256(source.content);
|
|
83
|
+
requireValue(!sourceDigests.has(source.path) || sourceDigests.get(source.path) === sourceDigest, "the same pinned source path has conflicting contents");
|
|
84
|
+
sourceDigests.set(source.path, sourceDigest);
|
|
85
|
+
}
|
|
86
|
+
requireValue(sameIds(example.repository.sourceDigests.map((entry) => entry.path), [...sourceDigests.keys()]), "source digest inventory must cover the exact supplied sources");
|
|
87
|
+
for (const entry of example.repository.sourceDigests)
|
|
88
|
+
requireValue(digest(entry.sha256) && sourceDigests.get(entry.path) === entry.sha256, "pinned source digest mismatch");
|
|
89
|
+
requireValue(input.solutions.length >= 2 &&
|
|
90
|
+
input.solutions.length <= 32 &&
|
|
91
|
+
unique(input.solutions.map((solution) => solution.id)) &&
|
|
92
|
+
input.solutions.every((solution) => identifier(solution.id) && nonempty(solution.family)), "each example requires two to 32 uniquely identified development controls");
|
|
93
|
+
for (const solution of input.solutions) {
|
|
94
|
+
requireValue(solution.fileOverrides.length > 0 &&
|
|
95
|
+
unique(solution.fileOverrides.map((source) => source.path)) &&
|
|
96
|
+
solution.fileOverrides.every((source) => sourcePath(source.path)), "control overrides must be complete, uniquely named repository-relative files");
|
|
97
|
+
}
|
|
98
|
+
requireValue(input.comparisonSolutionIds.length === 2 &&
|
|
99
|
+
unique(input.comparisonSolutionIds) &&
|
|
100
|
+
input.comparisonSolutionIds.every((id) => input.solutions.some((solution) => solution.id === id)), "comparison membership must identify exactly two supplied controls independently of their labels");
|
|
101
|
+
if (input.behavioralScope !== undefined)
|
|
102
|
+
requireValue(unique(input.behavioralScope.map((clause) => clause.id)) &&
|
|
103
|
+
input.behavioralScope.every((clause) => nonempty(clause.description) &&
|
|
104
|
+
clause.behaviorIds.length > 0 &&
|
|
105
|
+
clause.behaviorIds.every((id) => behaviorIds.includes(id))), "behavioral scope must reference the supplied visible behavior");
|
|
106
|
+
const adjudication = example.adjudication;
|
|
107
|
+
requireValue(adjudication.inputDigest === repositorySemanticCalibrationInputDigestV1(input), "adjudication does not bind the exact development input");
|
|
108
|
+
requireValue(adjudication.adjudicatorIds.length > 0 &&
|
|
109
|
+
unique(adjudication.adjudicatorIds) &&
|
|
110
|
+
adjudication.generatorIdentities.length > 0 &&
|
|
111
|
+
unique(adjudication.generatorIdentities) &&
|
|
112
|
+
adjudication.adjudicatorIds.every((id) => !adjudication.generatorIdentities.includes(id)), "adjudication requires explicit identities separate from the example generators");
|
|
113
|
+
requireValue(Number.isFinite(Date.parse(adjudication.completedAt)), "adjudication completion timestamp is invalid");
|
|
114
|
+
requireValue(adjudication.evidence.length > 0 && unique(adjudication.evidence.map((entry) => entry.id)), "independent adjudication evidence is required");
|
|
115
|
+
const executableChecks = new Map();
|
|
116
|
+
for (const evidence of adjudication.evidence) {
|
|
117
|
+
requireValue(nonempty(evidence.content) &&
|
|
118
|
+
digest(evidence.sha256) &&
|
|
119
|
+
sha256(evidence.content) === evidence.sha256, "adjudication evidence digest mismatch");
|
|
120
|
+
if (evidence.kind === "executable-development-check") {
|
|
121
|
+
requireValue(nonempty(evidence.checkerSource) &&
|
|
122
|
+
digest(evidence.checkerSourceSha256) &&
|
|
123
|
+
sha256(evidence.checkerSource) === evidence.checkerSourceSha256, "independent checker source digest mismatch");
|
|
124
|
+
const check = decode(RepositorySemanticCalibrationExecutableCheckV1, JSON.parse(evidence.content));
|
|
125
|
+
requireValue(check.inputDigest === adjudication.inputDigest &&
|
|
126
|
+
check.checkerSourceDigest === evidence.checkerSourceSha256 &&
|
|
127
|
+
adjudication.adjudicatorIds.includes(check.checkerId) &&
|
|
128
|
+
unique(check.checks.map((entry) => entry.solutionId)) &&
|
|
129
|
+
check.checks.every((entry) => entry.repetitions >= 2 &&
|
|
130
|
+
input.solutions.some((solution) => solution.id === entry.solutionId)), "executable adjudication must positively confirm pinned controls with repeated independent checks");
|
|
131
|
+
executableChecks.set(evidence.id, check);
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
requireValue(sameIds(adjudication.labels.map((label) => label.solutionId), input.solutions.map((solution) => solution.id)), "every control requires one adjudicated label");
|
|
135
|
+
for (const label of adjudication.labels) {
|
|
136
|
+
requireValue(nonempty(label.rationale) &&
|
|
137
|
+
label.evidenceIds.length > 0 &&
|
|
138
|
+
unique(label.evidenceIds) &&
|
|
139
|
+
label.evidenceIds.every((id) => adjudication.evidence.some((entry) => entry.id === id)) &&
|
|
140
|
+
unique(label.violatedBehaviorIds) &&
|
|
141
|
+
label.violatedBehaviorIds.every((id) => behaviorIds.includes(id)) &&
|
|
142
|
+
(label.expectedClass === "wrong"
|
|
143
|
+
? label.violatedBehaviorIds.length > 0
|
|
144
|
+
: label.violatedBehaviorIds.length === 0), "labels require consistent visible behavior and explicit independent evidence");
|
|
145
|
+
if (adjudication.kind === "independent-human-development-adjudication") {
|
|
146
|
+
requireValue(label.evidenceIds.some((id) => adjudication.evidence.some((entry) => entry.id === id && entry.kind === "human-review")), "human adjudication requires human-review evidence for every expected label");
|
|
147
|
+
}
|
|
148
|
+
else {
|
|
149
|
+
requireValue(label.evidenceIds.some((id) => executableChecks
|
|
150
|
+
.get(id)
|
|
151
|
+
?.checks.some((check) => check.solutionId === label.solutionId &&
|
|
152
|
+
check.classification === label.expectedClass &&
|
|
153
|
+
sameIds(check.violatedBehaviorIds, label.violatedBehaviorIds)) === true), "every executable-adjudicated label requires a matching positive independent check");
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
requireValue(example.exampleDigest === repositorySemanticCalibrationExampleDigestV1(example), "example digest is stale");
|
|
157
|
+
};
|
|
158
|
+
/** Adjudications and their evidence are deliberately absent from this projection. */
|
|
159
|
+
const prepareCritic = (example, role, version) => prepareRepositorySemanticReviewV1({
|
|
160
|
+
role,
|
|
161
|
+
modelContext: {
|
|
162
|
+
...(example.input.requestedDimension === undefined
|
|
163
|
+
? {}
|
|
164
|
+
: { requestedDimension: example.input.requestedDimension }),
|
|
165
|
+
repositoryEnvironment: example.input.repositoryEnvironment,
|
|
166
|
+
...(example.input.behavioralScope === undefined
|
|
167
|
+
? {}
|
|
168
|
+
: {
|
|
169
|
+
targetBehavior: example.input.targetBehavior,
|
|
170
|
+
behavioralScope: example.input.behavioralScope
|
|
171
|
+
})
|
|
172
|
+
},
|
|
173
|
+
visible: example.input.visible,
|
|
174
|
+
targetBehavior: example.input.targetBehavior,
|
|
175
|
+
preChangeSources: example.input.preChangeSources,
|
|
176
|
+
solutions: example.input.solutions,
|
|
177
|
+
comparisonSolutionIds: example.input.comparisonSolutionIds,
|
|
178
|
+
...(version === null ? {} : { validControlReviewVersion: version })
|
|
179
|
+
});
|
|
180
|
+
export const compileRepositorySemanticCalibrationPlanV1 = (input) => {
|
|
181
|
+
requireValue(Buffer.byteLength(JSON.stringify(input.manifest)) <= 8 * 1024 * 1024, "development manifest exceeds the bounded private input size");
|
|
182
|
+
const manifest = decode(RepositorySemanticCalibrationManifestV1, detached(input.manifest));
|
|
183
|
+
const modelPlan = decode(RepositoryFoundryModelPlanV1, detached(input.modelPlan));
|
|
184
|
+
assertRepositoryFoundryModelPlanV1(modelPlan);
|
|
185
|
+
requireValue([null, 1, 2].includes(input.validControlReviewVersion), "valid-control review policy must be explicitly frozen");
|
|
186
|
+
requireValue(Number.isSafeInteger(input.repetitions) &&
|
|
187
|
+
input.repetitions >= 2 &&
|
|
188
|
+
input.repetitions <= REPOSITORY_SEMANTIC_CALIBRATION_MAXIMUM_REPETITIONS, "repetitions must be between two and five");
|
|
189
|
+
requireValue(manifest.examples.length > 0 &&
|
|
190
|
+
manifest.examples.length <= REPOSITORY_SEMANTIC_CALIBRATION_MAXIMUM_EXAMPLES &&
|
|
191
|
+
unique(manifest.examples.map((example) => example.id)) &&
|
|
192
|
+
unique(manifest.examples.map((example) => example.exampleDigest)), "development examples must be nonempty, bounded, and unique");
|
|
193
|
+
for (const example of manifest.examples)
|
|
194
|
+
validateExample(example);
|
|
195
|
+
const labels = manifest.examples.flatMap((example) => example.adjudication.labels);
|
|
196
|
+
requireValue(manifest.thresholds.minimumScenarios > 0 &&
|
|
197
|
+
manifest.thresholds.minimumValidControls > 0 &&
|
|
198
|
+
manifest.thresholds.minimumWrongControls > 0 &&
|
|
199
|
+
manifest.examples.length >= manifest.thresholds.minimumScenarios &&
|
|
200
|
+
labels.filter((label) => label.expectedClass === "valid").length >=
|
|
201
|
+
manifest.thresholds.minimumValidControls &&
|
|
202
|
+
labels.filter((label) => label.expectedClass === "wrong").length >=
|
|
203
|
+
manifest.thresholds.minimumWrongControls, "the declared minimum development population must be present before calibration");
|
|
204
|
+
const criticModels = roles.map((role) => repositoryFoundryModelAssignmentV1(modelPlan, role).model);
|
|
205
|
+
for (const example of manifest.examples)
|
|
206
|
+
requireValue(example.adjudication.adjudicatorIds.every((id) => !criticModels.includes(id)), "the calibrated model cannot supply its own expected-label adjudication");
|
|
207
|
+
const preparedCritics = manifest.examples.flatMap((example) => roles.map((role) => {
|
|
208
|
+
const prepared = prepareCritic(example, role, input.validControlReviewVersion);
|
|
209
|
+
const document = Schema.toJsonSchemaDocument(prepared.outputSchema);
|
|
210
|
+
const schema = strictAuthoringSchema(Object.keys(document.definitions).length === 0
|
|
211
|
+
? document.schema
|
|
212
|
+
: { ...document.schema, $defs: document.definitions }).jsonSchema;
|
|
213
|
+
return {
|
|
214
|
+
exampleId: example.id,
|
|
215
|
+
role,
|
|
216
|
+
model: repositoryFoundryModelAssignmentV1(modelPlan, role).model,
|
|
217
|
+
reasoningEffort: repositoryFoundryModelAssignmentV1(modelPlan, role).reasoningEffort,
|
|
218
|
+
inputDigest: sha256(JSON.stringify(prepared.input)),
|
|
219
|
+
instructionsDigest: sha256(prepared.instructions),
|
|
220
|
+
schemaName: prepared.schemaName,
|
|
221
|
+
schemaDigest: digestOf(schema),
|
|
222
|
+
maximumOutputTokens: prepared.maximumOutputTokens
|
|
223
|
+
};
|
|
224
|
+
}));
|
|
225
|
+
const callCount = preparedCritics.length * input.repetitions;
|
|
226
|
+
const body = {
|
|
227
|
+
version: 1,
|
|
228
|
+
policyVersion: 1,
|
|
229
|
+
purpose: "development-semantic-calibration",
|
|
230
|
+
admissionCredit: false,
|
|
231
|
+
manifest,
|
|
232
|
+
manifestDigest: digestOf(manifest),
|
|
233
|
+
modelPlan,
|
|
234
|
+
validControlReviewVersion: input.validControlReviewVersion,
|
|
235
|
+
repetitions: input.repetitions,
|
|
236
|
+
callCount,
|
|
237
|
+
maximumInputBytesPerCall: REPOSITORY_FOUNDRY_REQUEST_INPUT_CEILING,
|
|
238
|
+
requiredTotalInputTokens: callCount * REPOSITORY_FOUNDRY_REQUEST_INPUT_CEILING,
|
|
239
|
+
requiredTotalOutputTokens: input.repetitions *
|
|
240
|
+
preparedCritics.reduce((sum, critic) => sum + critic.maximumOutputTokens, 0),
|
|
241
|
+
maximumOutputTokensPerCall: Math.max(...preparedCritics.map((critic) => critic.maximumOutputTokens)),
|
|
242
|
+
preparedCritics
|
|
243
|
+
};
|
|
244
|
+
return freeze({ ...body, planDigest: digestOf(body) });
|
|
245
|
+
};
|
|
246
|
+
const checkedPlan = (value) => {
|
|
247
|
+
const candidate = decode(RepositorySemanticCalibrationPlanV1, value);
|
|
248
|
+
const expected = compileRepositorySemanticCalibrationPlanV1(candidate);
|
|
249
|
+
requireValue(canonicalJson(candidate) === canonicalJson(expected), "plan digest, prepared critic policy, models, inputs, or bounds are stale");
|
|
250
|
+
return expected;
|
|
251
|
+
};
|
|
252
|
+
export function assertRepositorySemanticCalibrationPlanV1(value) {
|
|
253
|
+
checkedPlan(value);
|
|
254
|
+
}
|
|
255
|
+
const predictionFailure = (example, reviews) => {
|
|
256
|
+
if (!sameIds(reviews.map((review) => review.solutionId), example.input.solutions.map((solution) => solution.id)))
|
|
257
|
+
return "review-population";
|
|
258
|
+
const behaviorIds = example.input.targetBehavior.map((clause) => clause.id);
|
|
259
|
+
if (reviews.some((review) => !unique(review.violatedBehaviorIds) ||
|
|
260
|
+
review.violatedBehaviorIds.some((id) => !behaviorIds.includes(id)) ||
|
|
261
|
+
(review.classification === "wrong" && review.violatedBehaviorIds.length === 0) ||
|
|
262
|
+
(review.classification === "valid" && review.violatedBehaviorIds.length !== 0)))
|
|
263
|
+
return "behavior-reference";
|
|
264
|
+
return undefined;
|
|
265
|
+
};
|
|
266
|
+
const reviewOutcome = (example, value, version, criticInput, callId) => {
|
|
267
|
+
const predictionProblem = predictionFailure(example, value.reviews);
|
|
268
|
+
if (predictionProblem !== undefined)
|
|
269
|
+
return { kind: "malformed", reason: predictionProblem };
|
|
270
|
+
if (value.reviews.some((review) => !nonempty(review.detail)))
|
|
271
|
+
return { kind: "malformed", reason: "review-rationale" };
|
|
272
|
+
const independence = value.implementationIndependence;
|
|
273
|
+
if (version !== null) {
|
|
274
|
+
try {
|
|
275
|
+
// The same production predicate validates ids, source references, duplicate
|
|
276
|
+
// features, and outcome consistency. Returned policy findings are NOT an
|
|
277
|
+
// adjudication of mechanism truth or benchmark admission.
|
|
278
|
+
repositorySemanticReviewIndependenceFindingsV1(value, example.input.comparisonSolutionIds.map((id) => example.input.solutions.find((solution) => solution.id === id)), "development calibration critic", version, criticInput);
|
|
279
|
+
}
|
|
280
|
+
catch {
|
|
281
|
+
return { kind: "malformed", reason: "independence-review" };
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
else if (independence !== undefined) {
|
|
285
|
+
return { kind: "malformed", reason: "independence-review" };
|
|
286
|
+
}
|
|
287
|
+
return {
|
|
288
|
+
kind: "classified",
|
|
289
|
+
reviews: value.reviews.map((review) => ({
|
|
290
|
+
solutionId: review.solutionId,
|
|
291
|
+
classification: review.classification,
|
|
292
|
+
violatedBehaviorIds: [...review.violatedBehaviorIds]
|
|
293
|
+
})),
|
|
294
|
+
...(independence === undefined ? {} : { independence: independence.outcome }),
|
|
295
|
+
...(callId === undefined ? {} : { callId })
|
|
296
|
+
};
|
|
297
|
+
};
|
|
298
|
+
const scheduled = (plan, operationId) => {
|
|
299
|
+
requireValue(identifier(operationId), "operation id must be a bounded identifier");
|
|
300
|
+
return plan.manifest.examples.flatMap((example) => Array.from({ length: plan.repetitions }, (_, index) => roles.map((role) => {
|
|
301
|
+
const binding = plan.preparedCritics.find((entry) => entry.exampleId === example.id && entry.role === role);
|
|
302
|
+
return {
|
|
303
|
+
exampleId: example.id,
|
|
304
|
+
exampleDigest: example.exampleDigest,
|
|
305
|
+
role,
|
|
306
|
+
repetition: index + 1,
|
|
307
|
+
operationId: `${operationId}:${plan.planDigest}:${example.id}:repeat-${String(index + 1)}:${role}`,
|
|
308
|
+
model: binding.model,
|
|
309
|
+
reasoningEffort: binding.reasoningEffort,
|
|
310
|
+
inputDigest: binding.inputDigest,
|
|
311
|
+
instructionsDigest: binding.instructionsDigest,
|
|
312
|
+
schemaDigest: binding.schemaDigest
|
|
313
|
+
};
|
|
314
|
+
})).flat());
|
|
315
|
+
};
|
|
316
|
+
const observationKey = (value) => `${value.exampleId}\0${value.role}\0${String(value.repetition)}`;
|
|
317
|
+
const countObservations = (plan, observations) => {
|
|
318
|
+
const counts = {
|
|
319
|
+
scheduledCalls: observations.length,
|
|
320
|
+
classifiedCalls: 0,
|
|
321
|
+
malformedCalls: 0,
|
|
322
|
+
executionFailureCalls: 0,
|
|
323
|
+
notRunCalls: 0,
|
|
324
|
+
scheduledControls: 0,
|
|
325
|
+
unscoredControls: 0,
|
|
326
|
+
validAsValid: 0,
|
|
327
|
+
validAsWrong: 0,
|
|
328
|
+
validAsUncertain: 0,
|
|
329
|
+
wrongAsValid: 0,
|
|
330
|
+
wrongAsWrong: 0,
|
|
331
|
+
wrongAsUncertain: 0,
|
|
332
|
+
independenceIndependent: 0,
|
|
333
|
+
independenceCorrelated: 0,
|
|
334
|
+
independenceUncertain: 0,
|
|
335
|
+
repeatabilityGroups: 0,
|
|
336
|
+
stableClassificationGroups: 0,
|
|
337
|
+
changedClassificationGroups: 0,
|
|
338
|
+
incompleteClassificationGroups: 0
|
|
339
|
+
};
|
|
340
|
+
const cells = {
|
|
341
|
+
valid: { valid: "validAsValid", wrong: "validAsWrong", uncertain: "validAsUncertain" },
|
|
342
|
+
wrong: { valid: "wrongAsValid", wrong: "wrongAsWrong", uncertain: "wrongAsUncertain" }
|
|
343
|
+
};
|
|
344
|
+
for (const observation of observations) {
|
|
345
|
+
const example = plan.manifest.examples.find((entry) => entry.id === observation.exampleId);
|
|
346
|
+
counts.scheduledControls += example.input.solutions.length;
|
|
347
|
+
const outcome = observation.outcome;
|
|
348
|
+
if (outcome.kind !== "classified") {
|
|
349
|
+
counts.unscoredControls += example.input.solutions.length;
|
|
350
|
+
if (outcome.kind === "malformed")
|
|
351
|
+
counts.malformedCalls += 1;
|
|
352
|
+
else if (outcome.kind === "execution-failure")
|
|
353
|
+
counts.executionFailureCalls += 1;
|
|
354
|
+
else
|
|
355
|
+
counts.notRunCalls += 1;
|
|
356
|
+
continue;
|
|
357
|
+
}
|
|
358
|
+
counts.classifiedCalls += 1;
|
|
359
|
+
for (const review of outcome.reviews) {
|
|
360
|
+
const expected = example.adjudication.labels.find((label) => label.solutionId === review.solutionId).expectedClass;
|
|
361
|
+
counts[cells[expected][review.classification]] += 1;
|
|
362
|
+
}
|
|
363
|
+
if (outcome.independence === "independent")
|
|
364
|
+
counts.independenceIndependent += 1;
|
|
365
|
+
if (outcome.independence === "correlated")
|
|
366
|
+
counts.independenceCorrelated += 1;
|
|
367
|
+
if (outcome.independence === "uncertain")
|
|
368
|
+
counts.independenceUncertain += 1;
|
|
369
|
+
}
|
|
370
|
+
for (const example of plan.manifest.examples)
|
|
371
|
+
for (const role of roles) {
|
|
372
|
+
const repeated = observations.filter((entry) => entry.exampleId === example.id && entry.role === role);
|
|
373
|
+
if (repeated.length === 0)
|
|
374
|
+
continue;
|
|
375
|
+
for (const solution of example.input.solutions) {
|
|
376
|
+
counts.repeatabilityGroups += 1;
|
|
377
|
+
const predictions = repeated.flatMap((entry) => entry.outcome.kind === "classified"
|
|
378
|
+
? entry.outcome.reviews
|
|
379
|
+
.filter((review) => review.solutionId === solution.id)
|
|
380
|
+
.map((review) => review.classification)
|
|
381
|
+
: []);
|
|
382
|
+
if (predictions.length !== plan.repetitions)
|
|
383
|
+
counts.incompleteClassificationGroups += 1;
|
|
384
|
+
else if (new Set(predictions).size === 1)
|
|
385
|
+
counts.stableClassificationGroups += 1;
|
|
386
|
+
else
|
|
387
|
+
counts.changedClassificationGroups += 1;
|
|
388
|
+
}
|
|
389
|
+
}
|
|
390
|
+
return counts;
|
|
391
|
+
};
|
|
392
|
+
const limitations = [
|
|
393
|
+
"Development measurement only: no held-out generalization, benchmark admission, or routing credit.",
|
|
394
|
+
"Evidence and checker digests bind supplied adjudication provenance; artifact origin and actual execution require independent verification.",
|
|
395
|
+
"Manifest thresholds are predeclared and copied unchanged; this report does not manufacture a pass.",
|
|
396
|
+
"Repeatability measures classifications, not rationales; stable uncertainty is not semantic correctness.",
|
|
397
|
+
"Mechanism-independence accuracy and full valid-control admission are not established by executable correctness labels.",
|
|
398
|
+
"Separate role calls or similar mechanisms do not establish statistical independence.",
|
|
399
|
+
"This report is not exposure authority; the CLI must reconcile and close its strict upstream execution evidence."
|
|
400
|
+
];
|
|
401
|
+
export const compileRepositorySemanticCalibrationReportV1 = (input) => {
|
|
402
|
+
const plan = checkedPlan(input.plan);
|
|
403
|
+
const observations = decode(Schema.Array(RepositorySemanticCalibrationObservationV1), detached(input.observations));
|
|
404
|
+
const expected = scheduled(plan, input.operationId);
|
|
405
|
+
const byKey = new Map(observations.map((entry) => [observationKey(entry), entry]));
|
|
406
|
+
requireValue(observations.length === expected.length && byKey.size === observations.length, "report must retain exactly one observation for every scheduled critic call");
|
|
407
|
+
const ordered = expected.map((metadata) => {
|
|
408
|
+
const observation = byKey.get(observationKey(metadata));
|
|
409
|
+
requireValue(observation !== undefined, "report is missing a scheduled observation");
|
|
410
|
+
const { outcome, ...actual } = observation;
|
|
411
|
+
requireValue(canonicalJson(actual) === canonicalJson(metadata), "observation does not bind the exact example, role, repetition, model, or prepared policy");
|
|
412
|
+
if (outcome.kind === "classified") {
|
|
413
|
+
const example = plan.manifest.examples.find((entry) => entry.id === metadata.exampleId);
|
|
414
|
+
requireValue(predictionFailure(example, outcome.reviews) === undefined, "classified observation has malformed control predictions");
|
|
415
|
+
requireValue(plan.validControlReviewVersion === null
|
|
416
|
+
? outcome.independence === undefined
|
|
417
|
+
: outcome.independence !== undefined, "classified observation uses the wrong independence policy");
|
|
418
|
+
requireValue(outcome.callId === undefined || nonempty(outcome.callId), "a returned call id must be nonempty");
|
|
419
|
+
}
|
|
420
|
+
return observation;
|
|
421
|
+
});
|
|
422
|
+
let executionStopped = false;
|
|
423
|
+
for (let index = 0; index < ordered.length; index += roles.length) {
|
|
424
|
+
const pair = ordered.slice(index, index + roles.length);
|
|
425
|
+
requireValue(executionStopped
|
|
426
|
+
? pair.every((entry) => entry.outcome.kind === "not-run")
|
|
427
|
+
: pair.every((entry) => entry.outcome.kind !== "not-run"), "not-run observations must exactly follow the first execution-failure pair");
|
|
428
|
+
executionStopped ||= pair.some((entry) => entry.outcome.kind === "execution-failure");
|
|
429
|
+
}
|
|
430
|
+
const callIds = ordered.flatMap((entry) => entry.outcome.kind === "classified" && entry.outcome.callId !== undefined
|
|
431
|
+
? [entry.outcome.callId]
|
|
432
|
+
: []);
|
|
433
|
+
requireValue(unique(callIds), "returned critic call ids cannot be reused");
|
|
434
|
+
const controls = plan.manifest.examples.flatMap((example) => example.adjudication.labels.map((label) => ({
|
|
435
|
+
exampleId: example.id,
|
|
436
|
+
exampleDigest: example.exampleDigest,
|
|
437
|
+
solutionId: label.solutionId,
|
|
438
|
+
controlDigest: digestOf({
|
|
439
|
+
repository: example.repository.id,
|
|
440
|
+
commit: example.repository.commit,
|
|
441
|
+
inputDigest: example.adjudication.inputDigest,
|
|
442
|
+
solution: example.input.solutions.find((solution) => solution.id === label.solutionId)
|
|
443
|
+
}),
|
|
444
|
+
expectedClass: label.expectedClass,
|
|
445
|
+
evidenceDigests: label.evidenceIds.map((id) => example.adjudication.evidence.find((entry) => entry.id === id).sha256)
|
|
446
|
+
})));
|
|
447
|
+
const validControls = controls.filter((entry) => entry.expectedClass === "valid").length;
|
|
448
|
+
const wrongControls = controls.length - validControls;
|
|
449
|
+
const body = {
|
|
450
|
+
version: 1,
|
|
451
|
+
policyVersion: 1,
|
|
452
|
+
purpose: "development-semantic-calibration",
|
|
453
|
+
admissionCredit: false,
|
|
454
|
+
planDigest: plan.planDigest,
|
|
455
|
+
manifestDigest: plan.manifestDigest,
|
|
456
|
+
operationId: input.operationId,
|
|
457
|
+
coverage: {
|
|
458
|
+
semanticCorrectness: "development-only",
|
|
459
|
+
mechanismIndependence: "unadjudicated",
|
|
460
|
+
fullValidControlAcceptance: "not-established"
|
|
461
|
+
},
|
|
462
|
+
thresholds: plan.manifest.thresholds,
|
|
463
|
+
denominators: {
|
|
464
|
+
scenarios: plan.manifest.examples.length,
|
|
465
|
+
models: new Set(plan.preparedCritics.map((entry) => entry.model)).size,
|
|
466
|
+
critics: roles.length,
|
|
467
|
+
repetitions: plan.repetitions,
|
|
468
|
+
distinctControls: controls.length,
|
|
469
|
+
distinctValidControls: validControls,
|
|
470
|
+
distinctWrongControls: wrongControls,
|
|
471
|
+
scheduledCalls: plan.callCount,
|
|
472
|
+
scheduledControlClassifications: controls.length * plan.repetitions * roles.length,
|
|
473
|
+
scheduledValidClassifications: validControls * plan.repetitions * roles.length,
|
|
474
|
+
scheduledWrongClassifications: wrongControls * plan.repetitions * roles.length
|
|
475
|
+
},
|
|
476
|
+
controls,
|
|
477
|
+
observations: ordered,
|
|
478
|
+
counts: countObservations(plan, ordered),
|
|
479
|
+
byRole: roles.map((role) => ({
|
|
480
|
+
role,
|
|
481
|
+
counts: countObservations(plan, ordered.filter((entry) => entry.role === role))
|
|
482
|
+
})),
|
|
483
|
+
limitations: [...limitations]
|
|
484
|
+
};
|
|
485
|
+
return freeze({ ...body, reportDigest: digestOf(body) });
|
|
486
|
+
};
|
|
487
|
+
export function assertRepositorySemanticCalibrationReportV1(input) {
|
|
488
|
+
const report = decode(RepositorySemanticCalibrationReportV1, input.report);
|
|
489
|
+
const expected = compileRepositorySemanticCalibrationReportV1({
|
|
490
|
+
plan: input.plan,
|
|
491
|
+
operationId: report.operationId,
|
|
492
|
+
observations: report.observations
|
|
493
|
+
});
|
|
494
|
+
requireValue(canonicalJson(report) === canonicalJson(expected), "report digest, counts, labels, provenance, thresholds, or denominators are inconsistent");
|
|
495
|
+
}
|
|
496
|
+
/**
|
|
497
|
+
* Composition only: the CLI owns immutable execution claims, strict submission
|
|
498
|
+
* authority, wall time, durable observation writes, and final evidence closure.
|
|
499
|
+
* No retries, repairs, generated labels, admission, or direct provider transport.
|
|
500
|
+
*/
|
|
501
|
+
export const executeRepositorySemanticCalibrationPlanV1 = Effect.fn("CaseGeneration.calibrateDevelopmentSemanticReview")(function* (input) {
|
|
502
|
+
const plan = yield* Effect.try({
|
|
503
|
+
try: () => checkedPlan(input.plan),
|
|
504
|
+
catch: (cause) => new RepositoryFoundryError({
|
|
505
|
+
operation: "plan-generation",
|
|
506
|
+
detail: "development semantic calibration plan failed validation",
|
|
507
|
+
cause
|
|
508
|
+
})
|
|
509
|
+
});
|
|
510
|
+
const schedule = yield* Effect.try({
|
|
511
|
+
try: () => scheduled(plan, input.operationId),
|
|
512
|
+
catch: (cause) => new RepositoryFoundryError({
|
|
513
|
+
operation: "plan-generation",
|
|
514
|
+
detail: "development semantic calibration operation identity is invalid",
|
|
515
|
+
cause
|
|
516
|
+
})
|
|
517
|
+
});
|
|
518
|
+
const observations = [];
|
|
519
|
+
let executionStopped = false;
|
|
520
|
+
for (let index = 0; index < schedule.length; index += roles.length) {
|
|
521
|
+
const pair = yield* Effect.all(schedule.slice(index, index + roles.length).map((metadata) => Effect.gen(function* () {
|
|
522
|
+
let outcome;
|
|
523
|
+
if (executionStopped) {
|
|
524
|
+
outcome = { kind: "not-run", reason: "prior-execution-failure" };
|
|
525
|
+
}
|
|
526
|
+
else {
|
|
527
|
+
const example = plan.manifest.examples.find((entry) => entry.id === metadata.exampleId);
|
|
528
|
+
const prepared = prepareCritic(example, metadata.role, plan.validControlReviewVersion);
|
|
529
|
+
outcome = yield* executeRepositorySemanticReviewV1({
|
|
530
|
+
modelPlan: plan.modelPlan,
|
|
531
|
+
operationId: metadata.operationId,
|
|
532
|
+
prepared
|
|
533
|
+
}).pipe(Effect.matchEffect({
|
|
534
|
+
// Required artifact/progress I/O is not a model observation. The
|
|
535
|
+
// existing LM boundary owns invoke-language-model errors; failures
|
|
536
|
+
// from other owners must retain their original typed cause.
|
|
537
|
+
onFailure: (error) => error.operation !== "invoke-language-model"
|
|
538
|
+
? Effect.fail(error)
|
|
539
|
+
: Effect.succeed(error.modelResponseFailure === undefined
|
|
540
|
+
? { kind: "execution-failure", errorOperation: error.operation }
|
|
541
|
+
: { kind: "malformed", reason: error.modelResponseFailure }),
|
|
542
|
+
onSuccess: (generation) => Effect.sync(() => {
|
|
543
|
+
if (generation.call.operationId !== metadata.operationId ||
|
|
544
|
+
generation.call.role !== metadata.role ||
|
|
545
|
+
generation.call.model !== metadata.model ||
|
|
546
|
+
generation.call.reasoningEffort !== metadata.reasoningEffort)
|
|
547
|
+
return { kind: "execution-failure", errorOperation: "invoke-language-model" };
|
|
548
|
+
let value;
|
|
549
|
+
try {
|
|
550
|
+
value = decode(prepared.outputSchema, generation.value);
|
|
551
|
+
}
|
|
552
|
+
catch {
|
|
553
|
+
return { kind: "malformed", reason: "schema" };
|
|
554
|
+
}
|
|
555
|
+
return reviewOutcome(example, value, plan.validControlReviewVersion, prepared.input, generation.call.callId);
|
|
556
|
+
})
|
|
557
|
+
}));
|
|
558
|
+
}
|
|
559
|
+
const observation = freeze({ ...metadata, outcome });
|
|
560
|
+
// Deliberately outside model-error classification. Required I/O failure
|
|
561
|
+
// propagates with its original typed cause and cannot earn report credit.
|
|
562
|
+
if (input.recordObservation !== undefined)
|
|
563
|
+
yield* input.recordObservation(observation);
|
|
564
|
+
return observation;
|
|
565
|
+
})), { concurrency: roles.length });
|
|
566
|
+
observations.push(...pair);
|
|
567
|
+
executionStopped ||= pair.some((entry) => entry.outcome.kind === "execution-failure");
|
|
568
|
+
}
|
|
569
|
+
return yield* Effect.try({
|
|
570
|
+
try: () => compileRepositorySemanticCalibrationReportV1({
|
|
571
|
+
plan,
|
|
572
|
+
operationId: input.operationId,
|
|
573
|
+
observations
|
|
574
|
+
}),
|
|
575
|
+
catch: (cause) => new RepositoryFoundryError({
|
|
576
|
+
operation: "execute-generation",
|
|
577
|
+
detail: "development semantic calibration report failed validation",
|
|
578
|
+
cause
|
|
579
|
+
})
|
|
580
|
+
});
|
|
581
|
+
});
|