@velum-labs/routekit-eval-setup 1.3.2 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/authoring-responses-request.d.ts +5 -0
- package/dist/adapters/authoring-responses-request.js +38 -0
- package/dist/adapters/evaluation-evidence-freshness.d.ts +7 -0
- package/dist/adapters/evaluation-evidence-freshness.js +76 -0
- package/dist/adapters/git-task-history.d.ts +67 -0
- package/dist/adapters/git-task-history.js +171 -0
- package/dist/adapters/integrated-repository-history.d.ts +21 -0
- package/dist/adapters/integrated-repository-history.js +175 -0
- package/dist/adapters/repository-command-diagnostic.d.ts +8 -0
- package/dist/adapters/repository-command-diagnostic.js +46 -0
- package/dist/adapters/repository-command-evidence.d.ts +13 -0
- package/dist/adapters/repository-command-evidence.js +102 -0
- package/dist/adapters/repository-command-runner.d.ts +238 -0
- package/dist/adapters/repository-command-runner.js +1483 -0
- package/dist/adapters/repository-import-context.d.ts +47 -0
- package/dist/adapters/repository-import-context.js +469 -0
- package/dist/adapters/repository-node-test-reporter.d.ts +3 -0
- package/dist/adapters/repository-node-test-reporter.js +27 -0
- package/dist/adapters/repository-review-evidence.d.ts +39 -0
- package/dist/adapters/repository-review-evidence.js +632 -0
- package/dist/adapters/repository-seed-selection.d.ts +7 -0
- package/dist/adapters/repository-seed-selection.js +79 -0
- package/dist/adapters/repository-solution-edits.d.ts +49 -0
- package/dist/adapters/repository-solution-edits.js +136 -0
- package/dist/adapters/repository-vitest-phase-adapter.d.ts +8 -0
- package/dist/adapters/repository-vitest-phase-adapter.js +310 -0
- package/dist/adapters/repository-vitest-reporter.d.ts +24 -0
- package/dist/adapters/repository-vitest-reporter.js +314 -0
- package/dist/adapters/strict-authoring-schema.d.ts +5 -0
- package/dist/adapters/strict-authoring-schema.js +158 -0
- package/dist/adapters/test-discovery.d.ts +30 -0
- package/dist/adapters/test-discovery.js +124 -0
- package/dist/adapters/typescript-repository-index.d.ts +51 -0
- package/dist/adapters/typescript-repository-index.js +226 -0
- package/dist/agentic-capabilities-protocol.d.ts +1373 -0
- package/dist/agentic-capabilities-protocol.js +786 -0
- package/dist/case-checkpoint-store.d.ts +29 -0
- package/dist/case-checkpoint-store.js +133 -0
- package/dist/case-pipeline-protocol-v2.d.ts +184 -0
- package/dist/case-pipeline-protocol-v2.js +193 -0
- package/dist/case-pipeline-protocol.d.ts +2626 -0
- package/dist/case-pipeline-protocol.js +371 -0
- package/dist/effect-api.d.ts +74 -10
- package/dist/effect-api.js +56 -6
- package/dist/errors.d.ts +31 -0
- package/dist/errors.js +10 -0
- package/dist/eval-capability-execution-envelope.d.ts +64 -0
- package/dist/eval-capability-execution-envelope.js +98 -0
- package/dist/eval-capability-policy.d.ts +90 -0
- package/dist/eval-capability-policy.js +107 -0
- package/dist/eval-event-log.d.ts +140 -0
- package/dist/eval-event-log.js +220 -0
- package/dist/evaluation-authoring-policy.d.ts +18 -0
- package/dist/evaluation-authoring-policy.js +19 -0
- package/dist/evaluation-authoring-validation.d.ts +22 -0
- package/dist/evaluation-authoring-validation.js +72 -0
- package/dist/evaluation-evidence.d.ts +20 -0
- package/dist/evaluation-evidence.js +319 -0
- package/dist/evaluation-grader-calibration-protocol.d.ts +108 -0
- package/dist/evaluation-grader-calibration-protocol.js +80 -0
- package/dist/evaluation-grader-calibration.d.ts +18 -0
- package/dist/evaluation-grader-calibration.js +334 -0
- package/dist/evaluation-grading-policy.d.ts +24 -0
- package/dist/evaluation-grading-policy.js +54 -0
- package/dist/evaluation-proposal-policy.d.ts +4 -0
- package/dist/evaluation-proposal-policy.js +91 -0
- package/dist/evaluation-source-retrieval.d.ts +68 -0
- package/dist/evaluation-source-retrieval.js +513 -0
- package/dist/evaluation-structure-policy.d.ts +29 -0
- package/dist/evaluation-structure-policy.js +138 -0
- package/dist/index.d.ts +124 -17
- package/dist/index.js +69 -11
- package/dist/inspection.js +2 -3
- package/dist/project-artifacts.d.ts +7 -2
- package/dist/project-artifacts.js +49 -136
- package/dist/project-authoring.d.ts +66 -5
- package/dist/project-authoring.js +783 -109
- package/dist/project-contracts.d.ts +419 -84
- package/dist/project-contracts.js +160 -52
- package/dist/project-store.js +2 -1
- package/dist/project-workflow.d.ts +5 -4
- package/dist/project-workflow.js +154 -35
- package/dist/repository-adversary-protocol.d.ts +64 -0
- package/dist/repository-adversary-protocol.js +105 -0
- package/dist/repository-behavior-protocol.d.ts +188 -0
- package/dist/repository-behavior-protocol.js +202 -0
- package/dist/repository-benchmark-protocol.d.ts +487 -0
- package/dist/repository-benchmark-protocol.js +96 -0
- package/dist/repository-execution-protocol.d.ts +150 -0
- package/dist/repository-execution-protocol.js +38 -0
- package/dist/repository-fixture-instructions.d.ts +3 -0
- package/dist/repository-fixture-instructions.js +91 -0
- package/dist/repository-fixture-protocol.d.ts +79 -0
- package/dist/repository-fixture-protocol.js +79 -0
- package/dist/repository-foundry-plan-protocol.d.ts +118 -0
- package/dist/repository-foundry-plan-protocol.js +296 -0
- package/dist/repository-foundry-progress-protocol.d.ts +52 -0
- package/dist/repository-foundry-progress-protocol.js +52 -0
- package/dist/repository-improvement-protocol.d.ts +100 -0
- package/dist/repository-improvement-protocol.js +106 -0
- package/dist/repository-language-model-protocol.d.ts +43 -0
- package/dist/repository-language-model-protocol.js +146 -0
- package/dist/repository-oracle-coverage-protocol.d.ts +18 -0
- package/dist/repository-oracle-coverage-protocol.js +39 -0
- package/dist/repository-oracle-execution-binding.d.ts +27 -0
- package/dist/repository-oracle-execution-binding.js +59 -0
- package/dist/repository-oracle-protocol.d.ts +230 -0
- package/dist/repository-oracle-protocol.js +156 -0
- package/dist/repository-oracle-scope-policy.d.ts +22 -0
- package/dist/repository-oracle-scope-policy.js +92 -0
- package/dist/repository-quality-policy.d.ts +15 -0
- package/dist/repository-quality-policy.js +357 -0
- package/dist/repository-routing-benchmark-protocol.d.ts +176 -0
- package/dist/repository-routing-benchmark-protocol.js +103 -0
- package/dist/repository-routing-model-protocol.d.ts +36 -0
- package/dist/repository-routing-model-protocol.js +89 -0
- package/dist/repository-routing-plan-protocol.d.ts +112 -0
- package/dist/repository-routing-plan-protocol.js +58 -0
- package/dist/repository-routing-quality-policy.d.ts +9 -0
- package/dist/repository-routing-quality-policy.js +191 -0
- package/dist/repository-seed-qualification-progress-protocol.d.ts +205 -0
- package/dist/repository-seed-qualification-progress-protocol.js +28 -0
- package/dist/repository-semantic-calibration-protocol.d.ts +768 -0
- package/dist/repository-semantic-calibration-protocol.js +276 -0
- package/dist/repository-semantic-calibration.d.ts +163 -0
- package/dist/repository-semantic-calibration.js +581 -0
- package/dist/repository-specification-contract-facts-protocol.d.ts +224 -0
- package/dist/repository-specification-contract-facts-protocol.js +276 -0
- package/dist/repository-specification-critique-protocol.d.ts +189 -0
- package/dist/repository-specification-critique-protocol.js +103 -0
- package/dist/repository-task-family-protocol.d.ts +24 -0
- package/dist/repository-task-family-protocol.js +37 -0
- package/dist/repository-task-seed-protocol.d.ts +384 -0
- package/dist/repository-task-seed-protocol.js +236 -0
- package/dist/repository-trajectory-protocol.d.ts +20 -0
- package/dist/repository-trajectory-protocol.js +42 -0
- package/dist/service.js +1 -1
- package/dist/services/adversary/service.d.ts +64 -0
- package/dist/services/adversary/service.js +330 -0
- package/dist/services/benchmark-compiler/service.d.ts +450 -0
- package/dist/services/benchmark-compiler/service.js +9 -0
- package/dist/services/budgeted-model/service.d.ts +118 -0
- package/dist/services/budgeted-model/service.js +460 -0
- package/dist/services/case-authoring/service.d.ts +163 -0
- package/dist/services/case-authoring/service.js +1456 -0
- package/dist/services/case-finalization/service.d.ts +283 -0
- package/dist/services/case-finalization/service.js +370 -0
- package/dist/services/case-generation/service.d.ts +619 -0
- package/dist/services/case-generation/service.js +2628 -0
- package/dist/services/case-pipeline/service.d.ts +31 -0
- package/dist/services/case-pipeline/service.js +485 -0
- package/dist/services/case-pipeline-v2/service.d.ts +70 -0
- package/dist/services/case-pipeline-v2/service.js +477 -0
- package/dist/services/command-observability/service.d.ts +13 -0
- package/dist/services/command-observability/service.js +3 -0
- package/dist/services/dimension-labeling/service.d.ts +77 -0
- package/dist/services/dimension-labeling/service.js +188 -0
- package/dist/services/eval-candidate/service.d.ts +208 -0
- package/dist/services/eval-candidate/service.js +64 -0
- package/dist/services/eval-capabilities/service.d.ts +183 -0
- package/dist/services/eval-capabilities/service.js +1433 -0
- package/dist/services/eval-environment/service.d.ts +173 -0
- package/dist/services/eval-environment/service.js +127 -0
- package/dist/services/evidence-reconstruction/service.d.ts +36 -0
- package/dist/services/evidence-reconstruction/service.js +145 -0
- package/dist/services/fixture-builder/service.d.ts +62 -0
- package/dist/services/fixture-builder/service.js +36 -0
- package/dist/services/fixture-validation/service.d.ts +75 -0
- package/dist/services/fixture-validation/service.js +295 -0
- package/dist/services/foundry/service.d.ts +831 -0
- package/dist/services/foundry/service.js +442 -0
- package/dist/services/foundry-progress/service.d.ts +62 -0
- package/dist/services/foundry-progress/service.js +149 -0
- package/dist/services/foundry-v2/service.d.ts +54 -0
- package/dist/services/foundry-v2/service.js +28 -0
- package/dist/services/grounded-authoring/service.d.ts +126 -0
- package/dist/services/grounded-authoring/service.js +822 -0
- package/dist/services/historical-case/service.d.ts +722 -0
- package/dist/services/historical-case/service.js +177 -0
- package/dist/services/improvement-loop/service.d.ts +59 -0
- package/dist/services/improvement-loop/service.js +176 -0
- package/dist/services/language-model/service.d.ts +52 -0
- package/dist/services/language-model/service.js +194 -0
- package/dist/services/oracle-builder/service.d.ts +146 -0
- package/dist/services/oracle-builder/service.js +513 -0
- package/dist/services/oracle-coverage/service.d.ts +28 -0
- package/dist/services/oracle-coverage/service.js +50 -0
- package/dist/services/oracle-coverage-witness/service.d.ts +130 -0
- package/dist/services/oracle-coverage-witness/service.js +538 -0
- package/dist/services/pipeline-challenge/service.d.ts +551 -0
- package/dist/services/pipeline-challenge/service.js +427 -0
- package/dist/services/pipeline-controls/service.d.ts +130 -0
- package/dist/services/pipeline-controls/service.js +483 -0
- package/dist/services/pipeline-oracle/service.d.ts +8 -0
- package/dist/services/pipeline-oracle/service.js +256 -0
- package/dist/services/pipeline-seed/service.d.ts +298 -0
- package/dist/services/pipeline-seed/service.js +428 -0
- package/dist/services/pipeline-spec/service.d.ts +103 -0
- package/dist/services/pipeline-spec/service.js +619 -0
- package/dist/services/pipeline-tournament/service.d.ts +258 -0
- package/dist/services/pipeline-tournament/service.js +476 -0
- package/dist/services/quality-gate/service.d.ts +233 -0
- package/dist/services/quality-gate/service.js +136 -0
- package/dist/services/repository-bundle/service.d.ts +33 -0
- package/dist/services/repository-bundle/service.js +114 -0
- package/dist/services/repository-model/service.d.ts +105 -0
- package/dist/services/repository-model/service.js +250 -0
- package/dist/services/repository-public-artifact/service.d.ts +133 -0
- package/dist/services/repository-public-artifact/service.js +330 -0
- package/dist/services/routing-benchmark/service.d.ts +362 -0
- package/dist/services/routing-benchmark/service.js +96 -0
- package/dist/services/specification-critic/service.d.ts +92 -0
- package/dist/services/specification-critic/service.js +172 -0
- package/dist/services/task-family/service.d.ts +40 -0
- package/dist/services/task-family/service.js +55 -0
- package/dist/services/task-seed/service.d.ts +906 -0
- package/dist/services/task-seed/service.js +1406 -0
- package/dist/services/task-specification/service.d.ts +27 -0
- package/dist/services/task-specification/service.js +40 -0
- package/dist/services/trajectory-policy/service.d.ts +110 -0
- package/dist/services/trajectory-policy/service.js +216 -0
- package/dist/test/agentic-capabilities-protocol.test.d.ts +1 -0
- package/dist/test/agentic-capabilities-protocol.test.js +570 -0
- package/dist/test/agentic-capabilities.test.d.ts +1 -0
- package/dist/test/agentic-capabilities.test.js +1461 -0
- package/dist/test/agentic-environment.test.d.ts +1 -0
- package/dist/test/agentic-environment.test.js +213 -0
- package/dist/test/case-pipeline-foundation.test.d.ts +1 -0
- package/dist/test/case-pipeline-foundation.test.js +535 -0
- package/dist/test/case-pipeline-protocol-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-protocol-v2.test.js +124 -0
- package/dist/test/case-pipeline-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-v2.test.js +286 -0
- package/dist/test/case-pipeline.test.d.ts +1 -0
- package/dist/test/case-pipeline.test.js +851 -0
- package/dist/test/eval-capability-policy.test.d.ts +1 -0
- package/dist/test/eval-capability-policy.test.js +50 -0
- package/dist/test/eval-event-log.test.d.ts +1 -0
- package/dist/test/eval-event-log.test.js +125 -0
- package/dist/test/evaluation-evidence-freshness.test.d.ts +1 -0
- package/dist/test/evaluation-evidence-freshness.test.js +44 -0
- package/dist/test/evaluation-evidence.test.d.ts +1 -0
- package/dist/test/evaluation-evidence.test.js +230 -0
- package/dist/test/evaluation-grader-calibration.test.d.ts +1 -0
- package/dist/test/evaluation-grader-calibration.test.js +373 -0
- package/dist/test/evaluation-proposal-digest.test.d.ts +1 -0
- package/dist/test/evaluation-proposal-digest.test.js +187 -0
- package/dist/test/evaluation-source-retrieval.test.d.ts +1 -0
- package/dist/test/evaluation-source-retrieval.test.js +237 -0
- package/dist/test/evaluation-structure-policy.test.d.ts +1 -0
- package/dist/test/evaluation-structure-policy.test.js +196 -0
- package/dist/test/fixtures/repository-resource-panel.d.ts +39 -0
- package/dist/test/fixtures/repository-resource-panel.js +111 -0
- package/dist/test/fixtures/vitest-boundary-panel.d.ts +84 -0
- package/dist/test/fixtures/vitest-boundary-panel.js +120 -0
- package/dist/test/fixtures/vitest-phase-panel.d.ts +135 -0
- package/dist/test/fixtures/vitest-phase-panel.js +213 -0
- package/dist/test/fixtures/vitest-reporter-results.d.ts +76 -0
- package/dist/test/fixtures/vitest-reporter-results.js +94 -0
- package/dist/test/grounded-authoring.test.d.ts +1 -0
- package/dist/test/grounded-authoring.test.js +565 -0
- package/dist/test/integrated-repository-history.test.d.ts +1 -0
- package/dist/test/integrated-repository-history.test.js +227 -0
- package/dist/test/project-authoring.test.js +593 -43
- package/dist/test/project-workflow.test.js +419 -40
- package/dist/test/repository-authoring-artifacts.test.d.ts +1 -0
- package/dist/test/repository-authoring-artifacts.test.js +185 -0
- package/dist/test/repository-bundle.test.d.ts +1 -0
- package/dist/test/repository-bundle.test.js +52 -0
- package/dist/test/repository-case-generation.test.d.ts +1 -0
- package/dist/test/repository-case-generation.test.js +3465 -0
- package/dist/test/repository-command-diagnostic.test.d.ts +1 -0
- package/dist/test/repository-command-diagnostic.test.js +55 -0
- package/dist/test/repository-command-signals.test.d.ts +1 -0
- package/dist/test/repository-command-signals.test.js +124 -0
- package/dist/test/repository-fixture-scope-coverage.test.d.ts +1 -0
- package/dist/test/repository-fixture-scope-coverage.test.js +127 -0
- package/dist/test/repository-fixture-validation.test.d.ts +1 -0
- package/dist/test/repository-fixture-validation.test.js +362 -0
- package/dist/test/repository-foundry-progress.test.d.ts +1 -0
- package/dist/test/repository-foundry-progress.test.js +110 -0
- package/dist/test/repository-foundry-quality.test.d.ts +1 -0
- package/dist/test/repository-foundry-quality.test.js +1138 -0
- package/dist/test/repository-import-context.test.d.ts +1 -0
- package/dist/test/repository-import-context.test.js +354 -0
- package/dist/test/repository-model-authoring.test.d.ts +1 -0
- package/dist/test/repository-model-authoring.test.js +544 -0
- package/dist/test/repository-model.test.d.ts +1 -0
- package/dist/test/repository-model.test.js +2195 -0
- package/dist/test/repository-node-test-reporter.test.d.ts +1 -0
- package/dist/test/repository-node-test-reporter.test.js +104 -0
- package/dist/test/repository-oracle-concurrency.test.d.ts +1 -0
- package/dist/test/repository-oracle-concurrency.test.js +542 -0
- package/dist/test/repository-oracle-coverage-witness.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage-witness.test.js +511 -0
- package/dist/test/repository-oracle-coverage.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage.test.js +168 -0
- package/dist/test/repository-oracle-evidence.test.d.ts +1 -0
- package/dist/test/repository-oracle-evidence.test.js +185 -0
- package/dist/test/repository-oracle-plan.test.d.ts +1 -0
- package/dist/test/repository-oracle-plan.test.js +176 -0
- package/dist/test/repository-overlay-isolation.test.d.ts +1 -0
- package/dist/test/repository-overlay-isolation.test.js +85 -0
- package/dist/test/repository-preparation-cache.test.d.ts +1 -0
- package/dist/test/repository-preparation-cache.test.js +414 -0
- package/dist/test/repository-public-artifact.test.d.ts +1 -0
- package/dist/test/repository-public-artifact.test.js +273 -0
- package/dist/test/repository-qualification-diagnostics.test.d.ts +1 -0
- package/dist/test/repository-qualification-diagnostics.test.js +524 -0
- package/dist/test/repository-reference-authoring.test.d.ts +1 -0
- package/dist/test/repository-reference-authoring.test.js +1633 -0
- package/dist/test/repository-review-evidence-v2.test.d.ts +1 -0
- package/dist/test/repository-review-evidence-v2.test.js +183 -0
- package/dist/test/repository-review-evidence.test.d.ts +1 -0
- package/dist/test/repository-review-evidence.test.js +124 -0
- package/dist/test/repository-seed-exclusions.test.d.ts +1 -0
- package/dist/test/repository-seed-exclusions.test.js +96 -0
- package/dist/test/repository-seed-selection.test.d.ts +1 -0
- package/dist/test/repository-seed-selection.test.js +504 -0
- package/dist/test/repository-semantic-calibration.test.d.ts +1 -0
- package/dist/test/repository-semantic-calibration.test.js +688 -0
- package/dist/test/repository-solution-edits.test.d.ts +1 -0
- package/dist/test/repository-solution-edits.test.js +377 -0
- package/dist/test/repository-specification-budget.test.d.ts +1 -0
- package/dist/test/repository-specification-budget.test.js +171 -0
- package/dist/test/repository-specification-contract-checkpoint.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-checkpoint.test.js +228 -0
- package/dist/test/repository-specification-contract-facts.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-facts.test.js +177 -0
- package/dist/test/repository-trajectory-authoring.test.d.ts +1 -0
- package/dist/test/repository-trajectory-authoring.test.js +176 -0
- package/dist/test/repository-valid-control-plan.test.d.ts +1 -0
- package/dist/test/repository-valid-control-plan.test.js +45 -0
- package/dist/test/repository-vitest-phase.test.d.ts +1 -0
- package/dist/test/repository-vitest-phase.test.js +848 -0
- package/dist/test/repository-vitest-reporter.test.d.ts +1 -0
- package/dist/test/repository-vitest-reporter.test.js +158 -0
- package/dist/test/repository-workspace-build.test.d.ts +1 -0
- package/dist/test/repository-workspace-build.test.js +160 -0
- package/dist/test/strict-authoring-schema.test.d.ts +1 -0
- package/dist/test/strict-authoring-schema.test.js +169 -0
- package/package.json +48 -6
|
@@ -1,20 +1,21 @@
|
|
|
1
1
|
import assert from "node:assert/strict";
|
|
2
|
-
import { mkdtemp, rm, symlink, writeFile } from "node:fs/promises";
|
|
2
|
+
import { mkdir, mkdtemp, readFile, rm, symlink, writeFile } from "node:fs/promises";
|
|
3
3
|
import os from "node:os";
|
|
4
4
|
import path from "node:path";
|
|
5
5
|
import test from "node:test";
|
|
6
6
|
import { layer as NodeServicesLayer } from "@effect/platform-node/NodeServices";
|
|
7
7
|
import { Effect, Layer } from "effect";
|
|
8
8
|
import { EvalProjectAuthoringError } from "../errors.js";
|
|
9
|
-
import {
|
|
9
|
+
import { EVAL_AUTHORING_CASE_EVIDENCE_BLOCKS, EVAL_AUTHORING_CASE_EVIDENCE_BYTES, EVAL_AUTHORING_EVIDENCE_BLOCK_BYTES } from "../evaluation-authoring-policy.js";
|
|
10
|
+
import { EVAL_AUTHORING_CASES_PER_DIMENSION, EVAL_AUTHORING_EVALUATION_OUTPUT_TOKENS, EVAL_AUTHORING_PREFLIGHT_OUTPUT_TOKENS, EVAL_AUTHORING_REQUEST_BYTES, EVAL_AUTHORING_SOURCE_BYTES, EVAL_AUTHORING_SOURCE_FILES, EvalAuthoringTransport, EvalProjectAuthor, EvalProjectAuthorLive, readProjectAuthoringSources, selectProjectAuthoringEvidencePacket, selectProjectAuthoringSourceFiles } from "../project-authoring.js";
|
|
10
11
|
import { EVAL_PROJECT_VERSION } from "../project-contracts.js";
|
|
11
12
|
const withRepository = async (use) => {
|
|
12
13
|
const root = await mkdtemp(path.join(os.tmpdir(), "routekit-author-sources-"));
|
|
13
14
|
const outside = await mkdtemp(path.join(os.tmpdir(), "routekit-author-outside-"));
|
|
14
15
|
try {
|
|
15
|
-
await writeFile(path.join(root, "source.md"), "
|
|
16
|
+
await writeFile(path.join(root, "source.md"), "A terminal event must be last; later output is invalid and unsupported.\n");
|
|
16
17
|
await writeFile(path.join(outside, "secret.md"), "outside\n");
|
|
17
|
-
await use({ root, outside });
|
|
18
|
+
return await use({ root, outside });
|
|
18
19
|
}
|
|
19
20
|
finally {
|
|
20
21
|
await Promise.all([
|
|
@@ -24,6 +25,7 @@ const withRepository = async (use) => {
|
|
|
24
25
|
}
|
|
25
26
|
};
|
|
26
27
|
const read = (input) => Effect.runPromise(readProjectAuthoringSources(input).pipe(Effect.provide(NodeServicesLayer)));
|
|
28
|
+
const select = (input) => Effect.runPromise(selectProjectAuthoringSourceFiles(input).pipe(Effect.provide(NodeServicesLayer)));
|
|
27
29
|
const configuration = {
|
|
28
30
|
workloadDescription: "Route production requests across separable workload dimensions.",
|
|
29
31
|
candidateModels: ["openai/candidate-a", "openai/candidate-b"],
|
|
@@ -60,17 +62,63 @@ const basis = {
|
|
|
60
62
|
basisDigest: "basis-digest",
|
|
61
63
|
dimensions: dimensions(5)
|
|
62
64
|
};
|
|
63
|
-
const
|
|
65
|
+
const promptFor = (index) => {
|
|
66
|
+
switch (index % 5) {
|
|
67
|
+
case 0:
|
|
68
|
+
return `Transform the captured request for scenario ${String(index + 1)}.`;
|
|
69
|
+
case 1:
|
|
70
|
+
return `Locate the fault in trace scenario ${String(index + 1)}.`;
|
|
71
|
+
case 2:
|
|
72
|
+
return `Construct the expected artifact for scenario ${String(index + 1)}.`;
|
|
73
|
+
case 3:
|
|
74
|
+
return `Analyze the boundary behavior in scenario ${String(index + 1)}.`;
|
|
75
|
+
default:
|
|
76
|
+
return `Explain the operational outcome for scenario ${String(index + 1)}.`;
|
|
77
|
+
}
|
|
78
|
+
};
|
|
79
|
+
const authoredCase = (index, evidenceId) => ({
|
|
64
80
|
id: `dimension-case-${String(index + 1)}`,
|
|
65
|
-
prompt:
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
81
|
+
prompt: promptFor(index),
|
|
82
|
+
evidenceIds: [evidenceId],
|
|
83
|
+
criteria: [
|
|
84
|
+
{
|
|
85
|
+
id: `criterion-${String(index + 1)}`,
|
|
86
|
+
description: `Derive the observable repository behavior for scenario ${String(index + 1)}.`,
|
|
87
|
+
importance: "critical"
|
|
88
|
+
}
|
|
89
|
+
],
|
|
90
|
+
positiveControls: [
|
|
91
|
+
{
|
|
92
|
+
id: `positive-${String(index + 1)}`,
|
|
93
|
+
response: `The repository evidence supports the expected scenario ${String(index + 1)} behavior.`
|
|
94
|
+
}
|
|
95
|
+
],
|
|
96
|
+
negativeControls: [
|
|
97
|
+
{
|
|
98
|
+
id: `missing-${String(index + 1)}`,
|
|
99
|
+
response: "The response discusses the topic but omits the required behavior.",
|
|
100
|
+
targetedCriterionIds: [`criterion-${String(index + 1)}`],
|
|
101
|
+
kind: "missing-criterion"
|
|
102
|
+
},
|
|
103
|
+
{
|
|
104
|
+
id: `forbidden-${String(index + 1)}`,
|
|
105
|
+
response: "The response invents behavior contradicted by the repository evidence.",
|
|
106
|
+
targetedCriterionIds: [`criterion-${String(index + 1)}`],
|
|
107
|
+
kind: "forbidden-behavior"
|
|
108
|
+
},
|
|
109
|
+
{
|
|
110
|
+
id: `non-responsive-${String(index + 1)}`,
|
|
111
|
+
response: "This is a polished overview that never answers the concrete request.",
|
|
112
|
+
targetedCriterionIds: [`criterion-${String(index + 1)}`],
|
|
113
|
+
kind: "non-responsive"
|
|
114
|
+
}
|
|
115
|
+
]
|
|
116
|
+
});
|
|
117
|
+
const authoredSuite = (evidenceId, dimensionId = basis.dimensions[0].id) => ({
|
|
70
118
|
version: EVAL_PROJECT_VERSION,
|
|
71
119
|
dimensionId,
|
|
72
120
|
maximumOutputTokens: 256,
|
|
73
|
-
cases:
|
|
121
|
+
cases: Array.from({ length: EVAL_AUTHORING_CASES_PER_DIMENSION }, (_, index) => authoredCase(index, evidenceId))
|
|
74
122
|
});
|
|
75
123
|
const decompositionWeights = () => basis.dimensions.map((dimension) => ({
|
|
76
124
|
dimensionId: dimension.id,
|
|
@@ -87,15 +135,13 @@ const decompositionBenchmark = () => ({
|
|
|
87
135
|
}
|
|
88
136
|
}))
|
|
89
137
|
});
|
|
90
|
-
const
|
|
138
|
+
const authoredCompositionSuite = (evidenceId) => ({
|
|
91
139
|
maximumOutputTokens: 256,
|
|
92
140
|
minimumWinnerScoreGap: 0.05,
|
|
93
141
|
minimumWinnerAgreement: 0.8,
|
|
94
142
|
cases: Array.from({ length: EVAL_AUTHORING_CASES_PER_DIMENSION }, (_, index) => ({
|
|
143
|
+
...authoredCase(index, evidenceId),
|
|
95
144
|
id: `composition-case-${String(index + 1)}`,
|
|
96
|
-
prompt: `Compose production response ${String(index + 1)}.`,
|
|
97
|
-
context: "Reference context.",
|
|
98
|
-
rubric: "State the expected composed production behavior.",
|
|
99
145
|
decomposition: {
|
|
100
146
|
weights: decompositionWeights(),
|
|
101
147
|
unknownWeight: 0
|
|
@@ -109,37 +155,242 @@ const compositionSuite = () => ({
|
|
|
109
155
|
}
|
|
110
156
|
}))
|
|
111
157
|
});
|
|
112
|
-
const
|
|
158
|
+
const authoredPreflight = (packets, replace) => ({
|
|
159
|
+
cases: packets.map((packet, index) => {
|
|
160
|
+
const evidenceId = packet.evidenceSources[0].blocks[0].id;
|
|
161
|
+
return {
|
|
162
|
+
dimensionId: packet.dimension.id,
|
|
163
|
+
case: replace?.({ packet, index, evidenceId }) ??
|
|
164
|
+
authoredCase(index, evidenceId)
|
|
165
|
+
};
|
|
166
|
+
})
|
|
167
|
+
});
|
|
168
|
+
const authoredEvaluationCompletion = (input, outputs, requests) => {
|
|
169
|
+
requests.push(input);
|
|
170
|
+
const authoringInput = JSON.parse(input.input);
|
|
171
|
+
const evidenceId = authoringInput.evidenceSources?.[0]?.blocks?.[0]?.id;
|
|
172
|
+
if (input.schemaName !== "routekit_decomposition_benchmark" &&
|
|
173
|
+
input.schemaName !== "routekit_evaluation_authoring_preflight") {
|
|
174
|
+
assert.ok(evidenceId);
|
|
175
|
+
}
|
|
176
|
+
const withCallId = (text) => ({
|
|
177
|
+
text,
|
|
178
|
+
...(outputs.callId === undefined ? {} : { callId: outputs.callId })
|
|
179
|
+
});
|
|
180
|
+
if (input.schemaName === "routekit_evaluation_authoring_preflight") {
|
|
181
|
+
const packets = (authoringInput.packets ?? []).map((packet) => {
|
|
182
|
+
const dimensionId = packet.dimension?.id;
|
|
183
|
+
const packetEvidenceId = packet.evidenceSources?.[0]?.blocks?.[0]?.id;
|
|
184
|
+
assert.ok(dimensionId);
|
|
185
|
+
assert.ok(packetEvidenceId);
|
|
186
|
+
return {
|
|
187
|
+
dimension: { id: dimensionId },
|
|
188
|
+
evidenceSources: [{ blocks: [{ id: packetEvidenceId }] }]
|
|
189
|
+
};
|
|
190
|
+
});
|
|
191
|
+
return withCallId(JSON.stringify(typeof outputs.preflight === "function"
|
|
192
|
+
? outputs.preflight(packets)
|
|
193
|
+
: outputs.preflight ?? authoredPreflight(packets)));
|
|
194
|
+
}
|
|
195
|
+
if (input.schemaName === "routekit_dimension_suite") {
|
|
196
|
+
const dimension = basis.dimensions.find((candidate) => input.operationId.endsWith(`:${candidate.id}`));
|
|
197
|
+
const dimensionId = dimension?.id ?? basis.dimensions[0].id;
|
|
198
|
+
return withCallId(JSON.stringify(typeof outputs.suite === "function"
|
|
199
|
+
? outputs.suite(evidenceId, dimensionId)
|
|
200
|
+
: outputs.suite ?? authoredSuite(evidenceId, dimensionId)));
|
|
201
|
+
}
|
|
202
|
+
if (input.schemaName === "routekit_decomposition_benchmark") {
|
|
203
|
+
return withCallId(JSON.stringify(outputs.decomposition ?? decompositionBenchmark()));
|
|
204
|
+
}
|
|
205
|
+
return withCallId(JSON.stringify(typeof outputs.composition === "function"
|
|
206
|
+
? outputs.composition(evidenceId)
|
|
207
|
+
: outputs.composition ?? authoredCompositionSuite(evidenceId)));
|
|
208
|
+
};
|
|
209
|
+
const proposeEvaluations = (root, outputs, requests = [], authoringBasis = basis, sourceInventory = ["source.md"]) => Effect.runPromise(Effect.gen(function* () {
|
|
113
210
|
const author = yield* EvalProjectAuthor;
|
|
114
211
|
return yield* author.proposeEvaluations({
|
|
115
212
|
operationId: "eng-833",
|
|
116
213
|
repositoryRoot: root,
|
|
117
|
-
sourceInventory
|
|
214
|
+
sourceInventory,
|
|
118
215
|
configuration,
|
|
119
|
-
basis
|
|
216
|
+
basis: authoringBasis
|
|
120
217
|
});
|
|
121
218
|
}).pipe(Effect.provide(EvalProjectAuthorLive.pipe(Layer.provide(Layer.succeed(EvalAuthoringTransport, EvalAuthoringTransport.of({
|
|
122
|
-
complete: (input) => Effect.sync(() =>
|
|
123
|
-
|
|
124
|
-
if (input.schemaName === "routekit_dimension_suite") {
|
|
125
|
-
const dimension = basis.dimensions.find((candidate) => input.operationId.endsWith(`:${candidate.id}`));
|
|
126
|
-
return JSON.stringify(outputs.suite ?? dimensionSuite(dimension?.id ?? basis.dimensions[0].id));
|
|
127
|
-
}
|
|
128
|
-
if (input.schemaName === "routekit_decomposition_benchmark") {
|
|
129
|
-
return JSON.stringify(outputs.decomposition ?? decompositionBenchmark());
|
|
130
|
-
}
|
|
131
|
-
return JSON.stringify(outputs.composition ?? compositionSuite());
|
|
132
|
-
})
|
|
219
|
+
complete: (input) => Effect.sync(() => authoredEvaluationCompletion(input, outputs, requests).text),
|
|
220
|
+
completeDetailed: (input) => Effect.sync(() => authoredEvaluationCompletion(input, outputs, requests))
|
|
133
221
|
}))), Layer.provide(NodeServicesLayer)))));
|
|
134
|
-
test("evaluation authoring
|
|
222
|
+
test("evaluation authoring compiles opaque evidence selections and hidden controls", async () => {
|
|
135
223
|
const requests = [];
|
|
136
|
-
await withRepository(async ({ root }) => {
|
|
137
|
-
|
|
224
|
+
const proposal = await withRepository(async ({ root }) => {
|
|
225
|
+
return proposeEvaluations(root, {}, requests);
|
|
138
226
|
});
|
|
139
227
|
const suiteRequest = requests.find((request) => request.schemaName === "routekit_dimension_suite");
|
|
140
228
|
assert.ok(suiteRequest);
|
|
141
|
-
assert.match(suiteRequest.instructions,
|
|
142
|
-
assert.match(suiteRequest.instructions, /
|
|
229
|
+
assert.match(suiteRequest.instructions, new RegExp(`Select at most ${String(EVAL_AUTHORING_CASE_EVIDENCE_BLOCKS)} evidenceIds per case`, "u"));
|
|
230
|
+
assert.match(suiteRequest.instructions, /Do not produce paths, line ranges, excerpts/u);
|
|
231
|
+
assert.match(suiteRequest.instructions, /hidden criteria and controls/u);
|
|
232
|
+
const suiteCaseSchema = suiteRequest.jsonSchema.properties.cases.items.properties;
|
|
233
|
+
assert.equal(Object.hasOwn(suiteCaseSchema, "context"), false);
|
|
234
|
+
assert.equal(Object.hasOwn(suiteCaseSchema, "rubric"), false);
|
|
235
|
+
const suiteInput = JSON.parse(suiteRequest.input);
|
|
236
|
+
const suppliedEvidenceIds = suiteInput.evidenceSources.flatMap((source) => source.blocks.map((block) => block.id));
|
|
237
|
+
assert.deepEqual(suiteCaseSchema.evidenceIds, {
|
|
238
|
+
type: "array",
|
|
239
|
+
minItems: 1,
|
|
240
|
+
maxItems: EVAL_AUTHORING_CASE_EVIDENCE_BLOCKS,
|
|
241
|
+
items: { type: "string", enum: suppliedEvidenceIds }
|
|
242
|
+
});
|
|
243
|
+
const firstCase = proposal.suites[0].cases[0];
|
|
244
|
+
assert.equal(firstCase.evidenceIds?.length, 1);
|
|
245
|
+
assert.match(firstCase.evidenceIds[0], /^ev_[0-9a-f]{24}$/u);
|
|
246
|
+
assert.match(firstCase.context ?? "", /must be last/u);
|
|
247
|
+
assert.match(firstCase.context ?? "", /invalid and unsupported/u);
|
|
248
|
+
assert.match(firstCase.rubric, /Critical criterion criterion-1/u);
|
|
249
|
+
assert.equal(firstCase.criteria?.[0]?.gradingMode, "semantic");
|
|
250
|
+
assert.deepEqual(firstCase.negativeControls?.map((control) => control.kind), ["missing-criterion", "forbidden-behavior", "non-responsive"]);
|
|
251
|
+
assert.ok(proposal.evidenceSources !== undefined);
|
|
252
|
+
});
|
|
253
|
+
test("evaluation authoring preflight rejects unknown evidence before full-suite spending", async () => {
|
|
254
|
+
const requests = [];
|
|
255
|
+
await withRepository(async ({ root }) => {
|
|
256
|
+
await assert.rejects(proposeEvaluations(root, {
|
|
257
|
+
callId: "model_call_cross_dimension_preflight",
|
|
258
|
+
preflight: (packets) => authoredPreflight(packets, ({ index, evidenceId }) => authoredCase(index, index === 0 ? "ev_unknown" : evidenceId))
|
|
259
|
+
}, requests), (error) => {
|
|
260
|
+
assert.ok(error instanceof EvalProjectAuthoringError);
|
|
261
|
+
assert.match(error.detail, /cross-dimension authoring preflight failed validation/u);
|
|
262
|
+
assert.match(error.detail, /call id model_call_cross_dimension_preflight/u);
|
|
263
|
+
assert.match(error.detail, /category evidence-selection/u);
|
|
264
|
+
assert.match(error.detail, /path \$\.cases\[0\]\.case\.evidenceIds/u);
|
|
265
|
+
assert.match(error.detail, /unknown evidence id/u);
|
|
266
|
+
return true;
|
|
267
|
+
});
|
|
268
|
+
});
|
|
269
|
+
assert.deepEqual(requests.map((request) => request.schemaName), ["routekit_evaluation_authoring_preflight"]);
|
|
270
|
+
});
|
|
271
|
+
test("evaluation authoring preflight rejects malformed grader controls", async () => {
|
|
272
|
+
const requests = [];
|
|
273
|
+
await withRepository(async ({ root }) => {
|
|
274
|
+
await assert.rejects(proposeEvaluations(root, {
|
|
275
|
+
preflight: (packets) => authoredPreflight(packets, ({ index, evidenceId }) => {
|
|
276
|
+
const testCase = authoredCase(index, evidenceId);
|
|
277
|
+
return index === packets.length - 1
|
|
278
|
+
? {
|
|
279
|
+
...testCase,
|
|
280
|
+
negativeControls: [testCase.negativeControls[0]]
|
|
281
|
+
}
|
|
282
|
+
: testCase;
|
|
283
|
+
})
|
|
284
|
+
}, requests), (error) => {
|
|
285
|
+
assert.ok(error instanceof EvalProjectAuthoringError);
|
|
286
|
+
assert.match(error.detail, /cross-dimension authoring preflight failed validation/u);
|
|
287
|
+
assert.match(error.detail, /path \$\.cases\[4\]\.case\.negativeControls/u);
|
|
288
|
+
assert.match(error.detail, /requires a forbidden-behavior negative control/u);
|
|
289
|
+
return true;
|
|
290
|
+
});
|
|
291
|
+
});
|
|
292
|
+
assert.deepEqual(requests.map((request) => request.schemaName), ["routekit_evaluation_authoring_preflight"]);
|
|
293
|
+
});
|
|
294
|
+
test("cross-dimension preflight requires every supplied dimension exactly once", async () => {
|
|
295
|
+
await withRepository(async ({ root }) => {
|
|
296
|
+
for (const scenario of [
|
|
297
|
+
{
|
|
298
|
+
label: "missing",
|
|
299
|
+
output: (packets) => ({
|
|
300
|
+
cases: authoredPreflight(packets).cases.slice(1)
|
|
301
|
+
}),
|
|
302
|
+
expected: /exactly one case per supplied dimension/u
|
|
303
|
+
},
|
|
304
|
+
{
|
|
305
|
+
label: "duplicate",
|
|
306
|
+
output: (packets) => {
|
|
307
|
+
const authored = authoredPreflight(packets);
|
|
308
|
+
return {
|
|
309
|
+
cases: [
|
|
310
|
+
authored.cases[0],
|
|
311
|
+
{
|
|
312
|
+
...authored.cases[1],
|
|
313
|
+
dimensionId: authored.cases[0].dimensionId
|
|
314
|
+
},
|
|
315
|
+
...authored.cases.slice(2)
|
|
316
|
+
]
|
|
317
|
+
};
|
|
318
|
+
},
|
|
319
|
+
expected: /must be unique/u
|
|
320
|
+
},
|
|
321
|
+
{
|
|
322
|
+
label: "unknown",
|
|
323
|
+
output: (packets) => {
|
|
324
|
+
const authored = authoredPreflight(packets);
|
|
325
|
+
return {
|
|
326
|
+
cases: [
|
|
327
|
+
{
|
|
328
|
+
...authored.cases[0],
|
|
329
|
+
dimensionId: "dimension-unknown"
|
|
330
|
+
},
|
|
331
|
+
...authored.cases.slice(1)
|
|
332
|
+
]
|
|
333
|
+
};
|
|
334
|
+
},
|
|
335
|
+
expected: /must identify a supplied routing dimension/u
|
|
336
|
+
}
|
|
337
|
+
]) {
|
|
338
|
+
const requests = [];
|
|
339
|
+
await assert.rejects(proposeEvaluations(root, { preflight: scenario.output }, requests), (error) => {
|
|
340
|
+
assert.ok(error instanceof EvalProjectAuthoringError);
|
|
341
|
+
assert.match(error.detail, scenario.expected, scenario.label);
|
|
342
|
+
return true;
|
|
343
|
+
});
|
|
344
|
+
assert.deepEqual(requests.map((request) => request.schemaName), ["routekit_evaluation_authoring_preflight"]);
|
|
345
|
+
}
|
|
346
|
+
});
|
|
347
|
+
});
|
|
348
|
+
test("cross-dimension preflight rejects evidence from another dimension packet", async () => {
|
|
349
|
+
const requests = [];
|
|
350
|
+
await withRepository(async ({ root }) => {
|
|
351
|
+
const separatedDimensions = [
|
|
352
|
+
"alphaowner",
|
|
353
|
+
"bravoowner",
|
|
354
|
+
"charlieowner",
|
|
355
|
+
"deltaowner",
|
|
356
|
+
"echoowner"
|
|
357
|
+
].map((term, index) => ({
|
|
358
|
+
...dimensions(5)[index],
|
|
359
|
+
id: `${term}-dimension`,
|
|
360
|
+
description: `Production behavior owned by ${term}.`,
|
|
361
|
+
includes: [term]
|
|
362
|
+
}));
|
|
363
|
+
const separatedBasis = {
|
|
364
|
+
version: 2,
|
|
365
|
+
basisDigest: "separated-basis-digest",
|
|
366
|
+
dimensions: separatedDimensions
|
|
367
|
+
};
|
|
368
|
+
const sourceInventory = separatedDimensions.map((dimension) => `${dimension.id}.md`);
|
|
369
|
+
await Promise.all(separatedDimensions.map((dimension, index) => writeFile(path.join(root, sourceInventory[index]), `${dimension.id}\n`)));
|
|
370
|
+
await assert.rejects(proposeEvaluations(root, {
|
|
371
|
+
preflight: (packets) => {
|
|
372
|
+
const authored = authoredPreflight(packets);
|
|
373
|
+
const foreignEvidenceId = packets[1].evidenceSources[0].blocks[0].id;
|
|
374
|
+
return {
|
|
375
|
+
cases: authored.cases.map((item, index) => index === 0
|
|
376
|
+
? {
|
|
377
|
+
...item,
|
|
378
|
+
case: {
|
|
379
|
+
...item.case,
|
|
380
|
+
evidenceIds: [foreignEvidenceId]
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
: item)
|
|
384
|
+
};
|
|
385
|
+
}
|
|
386
|
+
}, requests, separatedBasis, sourceInventory), (error) => {
|
|
387
|
+
assert.ok(error instanceof EvalProjectAuthoringError);
|
|
388
|
+
assert.match(error.detail, /category evidence-selection/u);
|
|
389
|
+
assert.match(error.detail, /unknown evidence id/u);
|
|
390
|
+
return true;
|
|
391
|
+
});
|
|
392
|
+
});
|
|
393
|
+
assert.deepEqual(requests.map((request) => request.schemaName), ["routekit_evaluation_authoring_preflight"]);
|
|
143
394
|
});
|
|
144
395
|
const collectSchemaNumberKeyword = (value, keyword) => {
|
|
145
396
|
if (typeof value !== "object" || value === null)
|
|
@@ -159,7 +410,12 @@ test("authoring reads an exact discovered regular source", async () => {
|
|
|
159
410
|
repositoryRoot: root,
|
|
160
411
|
selectedFiles: ["source.md"],
|
|
161
412
|
sourceInventory: ["source.md"]
|
|
162
|
-
}), [
|
|
413
|
+
}), [
|
|
414
|
+
{
|
|
415
|
+
path: "source.md",
|
|
416
|
+
content: "A terminal event must be last; later output is invalid and unsupported.\n"
|
|
417
|
+
}
|
|
418
|
+
]);
|
|
163
419
|
});
|
|
164
420
|
});
|
|
165
421
|
test("authoring rejects traversal and absolute selected paths", async () => {
|
|
@@ -196,6 +452,221 @@ test("authoring rejects a selected file absent from discovery inventory", async
|
|
|
196
452
|
}), /not in the bounded discovery inventory/u);
|
|
197
453
|
});
|
|
198
454
|
});
|
|
455
|
+
test("authoring preserves selected-source file and byte caps", async () => {
|
|
456
|
+
await withRepository(async ({ root }) => {
|
|
457
|
+
const files = Array.from({ length: EVAL_AUTHORING_SOURCE_FILES + 1 }, (_, index) => `src/file-${String(index).padStart(2, "0")}.ts`);
|
|
458
|
+
await mkdir(path.join(root, "src"), { recursive: true });
|
|
459
|
+
await Promise.all(files.map((relative) => writeFile(path.join(root, relative), "export {};\n")));
|
|
460
|
+
await assert.rejects(read({
|
|
461
|
+
repositoryRoot: root,
|
|
462
|
+
selectedFiles: files,
|
|
463
|
+
sourceInventory: files
|
|
464
|
+
}), /select between 1 and 64 discovered source files/u);
|
|
465
|
+
await writeFile(path.join(root, "src/large-a.ts"), "a".repeat(30_001));
|
|
466
|
+
await writeFile(path.join(root, "src/large-b.ts"), "b".repeat(30_000));
|
|
467
|
+
await assert.rejects(read({
|
|
468
|
+
repositoryRoot: root,
|
|
469
|
+
selectedFiles: ["src/large-a.ts", "src/large-b.ts"],
|
|
470
|
+
sourceInventory: ["src/large-a.ts", "src/large-b.ts"]
|
|
471
|
+
}), /exceed the 60000 byte authoring bound/u);
|
|
472
|
+
});
|
|
473
|
+
});
|
|
474
|
+
test("authoring source selection is independent of inventory order", async () => {
|
|
475
|
+
await withRepository(async ({ root }) => {
|
|
476
|
+
const sources = [
|
|
477
|
+
["src/router.ts", "export const routeProductionRequest = () => \"routed\";\n"],
|
|
478
|
+
["test/router.test.ts", "test(\"routes a production request\", () => {});\n"],
|
|
479
|
+
["src/protocol/request-contract.ts", "export type RouteRequest = { model: string };\n"],
|
|
480
|
+
["docs/routing.md", "# Production routing\n"],
|
|
481
|
+
["scripts/verify-routing.ts", "export const verifyRouting = true;\n"]
|
|
482
|
+
];
|
|
483
|
+
for (const [relative, content] of sources) {
|
|
484
|
+
await mkdir(path.dirname(path.join(root, relative)), { recursive: true });
|
|
485
|
+
await writeFile(path.join(root, relative), content);
|
|
486
|
+
}
|
|
487
|
+
const inventory = sources.map(([relative]) => relative);
|
|
488
|
+
const selectionInput = {
|
|
489
|
+
repositoryRoot: root,
|
|
490
|
+
workloadDescription: "Route production requests across provider protocols.",
|
|
491
|
+
targetDimensions: basis.dimensions
|
|
492
|
+
};
|
|
493
|
+
const expected = await select({ ...selectionInput, sourceInventory: inventory });
|
|
494
|
+
const shuffled = await select({
|
|
495
|
+
...selectionInput,
|
|
496
|
+
sourceInventory: [inventory[3], inventory[0], inventory[4], inventory[1], inventory[2]]
|
|
497
|
+
});
|
|
498
|
+
assert.deepEqual(shuffled, expected);
|
|
499
|
+
assert.deepEqual([...expected].sort(), expected);
|
|
500
|
+
});
|
|
501
|
+
});
|
|
502
|
+
test("authoring drops leading session metadata and keeps relevant evidence within bounds", async () => {
|
|
503
|
+
await withRepository(async ({ root }) => {
|
|
504
|
+
const metadata = Array.from({ length: 20 }, (_, index) => `.ori/logs/sessions/session-${String(index).padStart(2, "0")}/metadata.json`);
|
|
505
|
+
const relevant = [
|
|
506
|
+
"packages/gateway/src/routing/provider-protocol.ts",
|
|
507
|
+
"packages/gateway/src/test/provider-protocol.test.ts",
|
|
508
|
+
"packages/contracts/src/provider-protocol-contract.ts",
|
|
509
|
+
"docs/provider-routing.md",
|
|
510
|
+
"scripts/verify-provider-routing.ts"
|
|
511
|
+
];
|
|
512
|
+
for (const relative of metadata) {
|
|
513
|
+
await mkdir(path.dirname(path.join(root, relative)), { recursive: true });
|
|
514
|
+
await writeFile(path.join(root, relative), "m".repeat(8_000));
|
|
515
|
+
}
|
|
516
|
+
for (const relative of relevant) {
|
|
517
|
+
await mkdir(path.dirname(path.join(root, relative)), { recursive: true });
|
|
518
|
+
await writeFile(path.join(root, relative), `${relative}\n${"r".repeat(9_000)}`);
|
|
519
|
+
}
|
|
520
|
+
const selected = await select({
|
|
521
|
+
repositoryRoot: root,
|
|
522
|
+
sourceInventory: [...metadata, ...relevant],
|
|
523
|
+
workloadDescription: "Route provider protocol requests.",
|
|
524
|
+
targetDimensions: [
|
|
525
|
+
{
|
|
526
|
+
id: "provider-protocol",
|
|
527
|
+
description: "Translate provider protocol request and response envelopes",
|
|
528
|
+
includes: ["provider protocol routing"],
|
|
529
|
+
excludes: ["unrelated requests"],
|
|
530
|
+
inScopeRequest: "Translate a provider protocol request.",
|
|
531
|
+
nearMissRequest: "Handle an unrelated request."
|
|
532
|
+
}
|
|
533
|
+
]
|
|
534
|
+
});
|
|
535
|
+
const selectedBytes = (await Promise.all(selected.map(async (relative) => Buffer.byteLength(await readFile(path.join(root, relative), "utf8"))))).reduce((total, bytes) => total + bytes, 0);
|
|
536
|
+
assert.ok(relevant.every((relative) => selected.includes(relative)));
|
|
537
|
+
assert.ok(selected.every((relative) => !relative.startsWith(".ori/")));
|
|
538
|
+
assert.ok(selected.length <= EVAL_AUTHORING_SOURCE_FILES);
|
|
539
|
+
assert.ok(selectedBytes <= EVAL_AUTHORING_SOURCE_BYTES);
|
|
540
|
+
});
|
|
541
|
+
});
|
|
542
|
+
test("authoring keeps implementation paths containing session, cache, or log segments", async () => {
|
|
543
|
+
await withRepository(async ({ root }) => {
|
|
544
|
+
const implementation = [
|
|
545
|
+
"packages/gateway/src/session/handler.ts",
|
|
546
|
+
"packages/gateway/src/cache/store.ts",
|
|
547
|
+
"packages/gateway/src/log/writer.ts"
|
|
548
|
+
];
|
|
549
|
+
for (const relative of implementation) {
|
|
550
|
+
await mkdir(path.dirname(path.join(root, relative)), { recursive: true });
|
|
551
|
+
await writeFile(path.join(root, relative), `export const source = ${JSON.stringify(relative)};\n`);
|
|
552
|
+
}
|
|
553
|
+
const selected = await select({
|
|
554
|
+
repositoryRoot: root,
|
|
555
|
+
sourceInventory: implementation,
|
|
556
|
+
workloadDescription: "Route gateway session requests with cached logging state."
|
|
557
|
+
});
|
|
558
|
+
assert.deepEqual(selected, [...implementation].sort());
|
|
559
|
+
});
|
|
560
|
+
});
|
|
561
|
+
test("authoring records a versioned selected-source manifest", async () => {
|
|
562
|
+
await withRepository(async ({ root }) => {
|
|
563
|
+
const sourceInventory = [
|
|
564
|
+
"src/router.ts",
|
|
565
|
+
"test/router.test.ts",
|
|
566
|
+
"src/protocol/request-contract.ts",
|
|
567
|
+
"docs/routing.md",
|
|
568
|
+
"scripts/verify-routing.ts"
|
|
569
|
+
];
|
|
570
|
+
for (const relative of sourceInventory) {
|
|
571
|
+
await mkdir(path.dirname(path.join(root, relative)), { recursive: true });
|
|
572
|
+
await writeFile(path.join(root, relative), `${relative}\n`);
|
|
573
|
+
}
|
|
574
|
+
const requests = [];
|
|
575
|
+
await proposeDimensions(root, { dimensions: dimensions(5) }, requests, sourceInventory);
|
|
576
|
+
const manifest = JSON.parse(await readFile(path.join(root, ".routekit/evals/authoring-sources.dimensions.v1.json"), "utf8"));
|
|
577
|
+
assert.equal(manifest.version, 1);
|
|
578
|
+
assert.equal(manifest.purpose, "dimensions");
|
|
579
|
+
assert.ok(manifest.packets[0].sources.length > 0);
|
|
580
|
+
assert.ok(manifest.packets[0].sources.every((source) => typeof source.path === "string" &&
|
|
581
|
+
typeof source.sizeBytes === "number" &&
|
|
582
|
+
typeof source.sourceClass === "string" &&
|
|
583
|
+
Array.isArray(source.targetDimensions) &&
|
|
584
|
+
typeof source.inclusionReason === "string"));
|
|
585
|
+
assert.ok(manifest.packets[0].sources.every((source) => JSON.stringify(Object.keys(source).sort()) ===
|
|
586
|
+
JSON.stringify([
|
|
587
|
+
"inclusionReason",
|
|
588
|
+
"path",
|
|
589
|
+
"sizeBytes",
|
|
590
|
+
"sourceClass",
|
|
591
|
+
"targetDimensions"
|
|
592
|
+
])));
|
|
593
|
+
});
|
|
594
|
+
});
|
|
595
|
+
test("authoring only admits excluded artifacts with an explicit recorded reason", async () => {
|
|
596
|
+
await withRepository(async ({ root }) => {
|
|
597
|
+
const metadata = ".ori/logs/sessions/session-01/metadata.json";
|
|
598
|
+
await mkdir(path.dirname(path.join(root, metadata)), { recursive: true });
|
|
599
|
+
await writeFile(path.join(root, metadata), "session evidence\n");
|
|
600
|
+
assert.deepEqual(await select({
|
|
601
|
+
repositoryRoot: root,
|
|
602
|
+
sourceInventory: [metadata, "source.md"],
|
|
603
|
+
workloadDescription: "Route production requests."
|
|
604
|
+
}), ["source.md"]);
|
|
605
|
+
assert.deepEqual(await Effect.runPromise(selectProjectAuthoringSourceFiles({
|
|
606
|
+
repositoryRoot: root,
|
|
607
|
+
sourceInventory: [metadata, "source.md"],
|
|
608
|
+
workloadDescription: "Route production requests.",
|
|
609
|
+
explicitInclusions: [
|
|
610
|
+
{ path: metadata, reason: "incident-specific operational evidence" }
|
|
611
|
+
]
|
|
612
|
+
}).pipe(Effect.provide(NodeServicesLayer))), [metadata, "source.md"]);
|
|
613
|
+
await assert.rejects(Effect.runPromise(selectProjectAuthoringSourceFiles({
|
|
614
|
+
repositoryRoot: root,
|
|
615
|
+
sourceInventory: [metadata, "source.md"],
|
|
616
|
+
workloadDescription: "Route production requests.",
|
|
617
|
+
explicitInclusions: [{ path: metadata, reason: " " }]
|
|
618
|
+
}).pipe(Effect.provide(NodeServicesLayer))), /requires a recorded inclusion reason/u);
|
|
619
|
+
});
|
|
620
|
+
});
|
|
621
|
+
test("evaluation evidence packets select compiler blocks instead of whole files", async () => {
|
|
622
|
+
await withRepository(async ({ root }) => {
|
|
623
|
+
const relevant = "packages/gateway/src/providers/openai-backend.ts";
|
|
624
|
+
const broad = "packages/gateway/src/test/router.test.ts";
|
|
625
|
+
const historical = "docs/evidence/eval-routing/old/published-routing.json";
|
|
626
|
+
for (const [relative, content] of [
|
|
627
|
+
[
|
|
628
|
+
relevant,
|
|
629
|
+
[
|
|
630
|
+
"export const executeProviderModel = true;\n",
|
|
631
|
+
"\n",
|
|
632
|
+
"export const decodeProviderResponse = true;\n",
|
|
633
|
+
"\n",
|
|
634
|
+
"export const unrelated = true;\n"
|
|
635
|
+
].join("")
|
|
636
|
+
],
|
|
637
|
+
[
|
|
638
|
+
broad,
|
|
639
|
+
Array.from({ length: 80 }, (_, index) => `test provider request response routing ${String(index)};\n`).join("")
|
|
640
|
+
],
|
|
641
|
+
[historical, '{"provider":"historical routing result"}\n']
|
|
642
|
+
]) {
|
|
643
|
+
await mkdir(path.dirname(path.join(root, relative)), {
|
|
644
|
+
recursive: true
|
|
645
|
+
});
|
|
646
|
+
await writeFile(path.join(root, relative), content);
|
|
647
|
+
}
|
|
648
|
+
const result = await Effect.runPromise(selectProjectAuthoringEvidencePacket({
|
|
649
|
+
repositoryRoot: root,
|
|
650
|
+
sourceInventory: [relevant, broad, historical],
|
|
651
|
+
targetDimensions: [
|
|
652
|
+
{
|
|
653
|
+
id: "provider-model-execution",
|
|
654
|
+
description: "Execute explicit provider model requests",
|
|
655
|
+
includes: [
|
|
656
|
+
"provider backend request execution",
|
|
657
|
+
"provider response decoding"
|
|
658
|
+
],
|
|
659
|
+
excludes: ["semantic routing"]
|
|
660
|
+
}
|
|
661
|
+
]
|
|
662
|
+
}).pipe(Effect.provide(NodeServicesLayer)));
|
|
663
|
+
assert.ok(result.evidenceSources.some((source) => source.path === relevant));
|
|
664
|
+
assert.ok(result.evidenceSources.every((source) => source.path !== historical));
|
|
665
|
+
assert.ok(result.evidenceSources.every((source) => source.sourceBlockCount !== undefined &&
|
|
666
|
+
source.blocks.length <= source.sourceBlockCount));
|
|
667
|
+
assert.ok(result.packet.sources.reduce((total, source) => total + source.sizeBytes, 0) <= EVAL_AUTHORING_SOURCE_BYTES);
|
|
668
|
+
});
|
|
669
|
+
});
|
|
199
670
|
test("dimension authoring sends an Anthropic-compatible structured output schema", async () => {
|
|
200
671
|
await withRepository(async ({ root }) => {
|
|
201
672
|
const requests = [];
|
|
@@ -322,7 +793,10 @@ test("evaluation authoring enforces Anthropic-deferred bounds after parsing", as
|
|
|
322
793
|
await withRepository(async ({ root }) => {
|
|
323
794
|
for (const maximumOutputTokens of [0, 16_385]) {
|
|
324
795
|
await assert.rejects(proposeEvaluations(root, {
|
|
325
|
-
suite:
|
|
796
|
+
suite: (evidenceId, dimensionId) => ({
|
|
797
|
+
...authoredSuite(evidenceId, dimensionId),
|
|
798
|
+
maximumOutputTokens
|
|
799
|
+
})
|
|
326
800
|
}), (error) => {
|
|
327
801
|
assert.ok(error instanceof EvalProjectAuthoringError);
|
|
328
802
|
assert.match(error.detail, /suite .* failed validation/u);
|
|
@@ -331,9 +805,30 @@ test("evaluation authoring enforces Anthropic-deferred bounds after parsing", as
|
|
|
331
805
|
});
|
|
332
806
|
}
|
|
333
807
|
await assert.rejects(proposeEvaluations(root, {
|
|
334
|
-
suite: {
|
|
335
|
-
|
|
336
|
-
|
|
808
|
+
suite: (evidenceId, dimensionId) => {
|
|
809
|
+
const suite = authoredSuite(evidenceId, dimensionId);
|
|
810
|
+
return {
|
|
811
|
+
...suite,
|
|
812
|
+
cases: suite.cases.map((testCase, index) => index === 0
|
|
813
|
+
? {
|
|
814
|
+
...testCase,
|
|
815
|
+
evidenceIds: Array.from({ length: EVAL_AUTHORING_CASE_EVIDENCE_BLOCKS + 1 }, () => evidenceId)
|
|
816
|
+
}
|
|
817
|
+
: testCase)
|
|
818
|
+
};
|
|
819
|
+
}
|
|
820
|
+
}), (error) => {
|
|
821
|
+
assert.ok(error instanceof EvalProjectAuthoringError);
|
|
822
|
+
assert.match(String(error.cause), new RegExp(`evidenceIds.*must contain at most ${String(EVAL_AUTHORING_CASE_EVIDENCE_BLOCKS)} items`, "u"));
|
|
823
|
+
return true;
|
|
824
|
+
});
|
|
825
|
+
await assert.rejects(proposeEvaluations(root, {
|
|
826
|
+
suite: (evidenceId, dimensionId) => {
|
|
827
|
+
const suite = authoredSuite(evidenceId, dimensionId);
|
|
828
|
+
return {
|
|
829
|
+
...suite,
|
|
830
|
+
cases: suite.cases.slice(0, EVAL_AUTHORING_CASES_PER_DIMENSION - 1)
|
|
831
|
+
};
|
|
337
832
|
}
|
|
338
833
|
}), (error) => {
|
|
339
834
|
assert.ok(error instanceof EvalProjectAuthoringError);
|
|
@@ -349,7 +844,10 @@ test("evaluation authoring enforces Anthropic-deferred bounds after parsing", as
|
|
|
349
844
|
return true;
|
|
350
845
|
});
|
|
351
846
|
await assert.rejects(proposeEvaluations(root, {
|
|
352
|
-
composition:
|
|
847
|
+
composition: (evidenceId) => ({
|
|
848
|
+
...authoredCompositionSuite(evidenceId),
|
|
849
|
+
minimumWinnerAgreement: 2
|
|
850
|
+
})
|
|
353
851
|
}), (error) => {
|
|
354
852
|
assert.ok(error instanceof EvalProjectAuthoringError);
|
|
355
853
|
assert.equal(error.detail, "composition benchmark failed validation");
|
|
@@ -358,14 +856,66 @@ test("evaluation authoring enforces Anthropic-deferred bounds after parsing", as
|
|
|
358
856
|
});
|
|
359
857
|
});
|
|
360
858
|
});
|
|
859
|
+
test("evaluation authoring exposes bounded call-linked validation diagnostics", async () => {
|
|
860
|
+
await withRepository(async ({ root }) => {
|
|
861
|
+
await assert.rejects(proposeEvaluations(root, {
|
|
862
|
+
callId: "model_call_authoring_validation",
|
|
863
|
+
suite: (evidenceId, dimensionId) => {
|
|
864
|
+
const suite = authoredSuite(evidenceId, dimensionId);
|
|
865
|
+
return {
|
|
866
|
+
...suite,
|
|
867
|
+
cases: suite.cases.map((testCase, index) => index === 0
|
|
868
|
+
? {
|
|
869
|
+
...testCase,
|
|
870
|
+
criteria: testCase.criteria.map((criterion) => ({
|
|
871
|
+
...criterion,
|
|
872
|
+
importance: "supporting"
|
|
873
|
+
}))
|
|
874
|
+
}
|
|
875
|
+
: testCase)
|
|
876
|
+
};
|
|
877
|
+
}
|
|
878
|
+
}), (error) => {
|
|
879
|
+
assert.ok(error instanceof EvalProjectAuthoringError);
|
|
880
|
+
assert.match(error.detail, /call id model_call_authoring_validation/u);
|
|
881
|
+
assert.match(error.detail, /category criterion-topology/u);
|
|
882
|
+
assert.match(error.detail, /path \$\.cases\[0\]\.criteria/u);
|
|
883
|
+
assert.match(error.detail, /requires at least one critical criterion/u);
|
|
884
|
+
assert.doesNotMatch(error.detail, /Known-good response/u);
|
|
885
|
+
return true;
|
|
886
|
+
});
|
|
887
|
+
});
|
|
888
|
+
});
|
|
361
889
|
test("evaluation authoring budgets enough output for all twenty requested cases", async () => {
|
|
362
890
|
await withRepository(async ({ root }) => {
|
|
891
|
+
assert.equal(EVAL_AUTHORING_CASE_EVIDENCE_BLOCKS *
|
|
892
|
+
EVAL_AUTHORING_EVIDENCE_BLOCK_BYTES, EVAL_AUTHORING_CASE_EVIDENCE_BYTES);
|
|
893
|
+
assert.ok((EVAL_AUTHORING_CASE_EVIDENCE_BLOCKS + 1) *
|
|
894
|
+
EVAL_AUTHORING_EVIDENCE_BLOCK_BYTES >
|
|
895
|
+
EVAL_AUTHORING_CASE_EVIDENCE_BYTES);
|
|
363
896
|
const requests = [];
|
|
364
897
|
const proposal = await proposeEvaluations(root, {}, requests);
|
|
365
898
|
assert.equal(proposal.suites.length, basis.dimensions.length);
|
|
366
|
-
assert.equal(requests.length, basis.dimensions.length +
|
|
367
|
-
|
|
368
|
-
assert.
|
|
899
|
+
assert.equal(requests.length, basis.dimensions.length + 3);
|
|
900
|
+
const preflightRequests = requests.filter((request) => request.schemaName === "routekit_evaluation_authoring_preflight");
|
|
901
|
+
assert.equal(preflightRequests.length, 1);
|
|
902
|
+
const preflight = preflightRequests[0];
|
|
903
|
+
assert.equal(preflight.schemaName, "routekit_evaluation_authoring_preflight");
|
|
904
|
+
assert.equal(preflight.maximumOutputTokens, EVAL_AUTHORING_PREFLIGHT_OUTPUT_TOKENS);
|
|
905
|
+
assert.ok(Buffer.byteLength(preflight.input, "utf8") <
|
|
906
|
+
EVAL_AUTHORING_REQUEST_BYTES);
|
|
907
|
+
const preflightInput = JSON.parse(preflight.input);
|
|
908
|
+
assert.deepEqual(preflightInput.packets.map((packet) => packet.dimension.id), basis.dimensions.map((dimension) => dimension.id));
|
|
909
|
+
assert.ok(preflightInput.packets.every((packet) => packet.evidenceSources.reduce((total, source) => total + source.blocks.length, 0) <= EVAL_AUTHORING_CASE_EVIDENCE_BLOCKS));
|
|
910
|
+
const preflightCasesSchema = preflight.jsonSchema.properties.cases;
|
|
911
|
+
assert.equal(preflightCasesSchema.minItems, basis.dimensions.length);
|
|
912
|
+
assert.equal(preflightCasesSchema.maxItems, basis.dimensions.length);
|
|
913
|
+
assert.deepEqual(preflightCasesSchema.items.anyOf.map((variant) => variant.properties.dimensionId.enum[0]), basis.dimensions.map((dimension) => dimension.id));
|
|
914
|
+
assert.deepEqual(preflightCasesSchema.items.anyOf.map((variant) => variant.properties.case.properties.evidenceIds.items.enum), preflightInput.packets.map((packet) => packet.evidenceSources.flatMap((source) => source.blocks.map((block) => block.id))));
|
|
915
|
+
assert.ok(requests.slice(1).every((request) => request.maximumOutputTokens === EVAL_AUTHORING_EVALUATION_OUTPUT_TOKENS));
|
|
916
|
+
assert.ok(requests
|
|
917
|
+
.filter((request) => request.schemaName !== "routekit_evaluation_authoring_preflight")
|
|
918
|
+
.every((request) => request.instructions.includes("exactly 20") &&
|
|
369
919
|
request.instructions.includes("Keep each case concise")));
|
|
370
920
|
});
|
|
371
921
|
});
|