@velum-labs/routekit-eval-setup 1.4.0 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/authoring-responses-request.d.ts +5 -0
- package/dist/adapters/authoring-responses-request.js +38 -0
- package/dist/adapters/evaluation-evidence-freshness.d.ts +7 -0
- package/dist/adapters/evaluation-evidence-freshness.js +76 -0
- package/dist/adapters/git-task-history.d.ts +67 -0
- package/dist/adapters/git-task-history.js +171 -0
- package/dist/adapters/integrated-repository-history.d.ts +21 -0
- package/dist/adapters/integrated-repository-history.js +175 -0
- package/dist/adapters/repository-command-diagnostic.d.ts +8 -0
- package/dist/adapters/repository-command-diagnostic.js +46 -0
- package/dist/adapters/repository-command-evidence.d.ts +13 -0
- package/dist/adapters/repository-command-evidence.js +102 -0
- package/dist/adapters/repository-command-runner.d.ts +238 -0
- package/dist/adapters/repository-command-runner.js +1483 -0
- package/dist/adapters/repository-import-context.d.ts +47 -0
- package/dist/adapters/repository-import-context.js +469 -0
- package/dist/adapters/repository-node-test-reporter.d.ts +3 -0
- package/dist/adapters/repository-node-test-reporter.js +27 -0
- package/dist/adapters/repository-review-evidence.d.ts +39 -0
- package/dist/adapters/repository-review-evidence.js +632 -0
- package/dist/adapters/repository-seed-selection.d.ts +7 -0
- package/dist/adapters/repository-seed-selection.js +79 -0
- package/dist/adapters/repository-solution-edits.d.ts +49 -0
- package/dist/adapters/repository-solution-edits.js +136 -0
- package/dist/adapters/repository-vitest-phase-adapter.d.ts +8 -0
- package/dist/adapters/repository-vitest-phase-adapter.js +310 -0
- package/dist/adapters/repository-vitest-reporter.d.ts +24 -0
- package/dist/adapters/repository-vitest-reporter.js +314 -0
- package/dist/adapters/strict-authoring-schema.d.ts +5 -0
- package/dist/adapters/strict-authoring-schema.js +158 -0
- package/dist/adapters/test-discovery.d.ts +30 -0
- package/dist/adapters/test-discovery.js +124 -0
- package/dist/adapters/typescript-repository-index.d.ts +51 -0
- package/dist/adapters/typescript-repository-index.js +226 -0
- package/dist/agentic-capabilities-protocol.d.ts +1373 -0
- package/dist/agentic-capabilities-protocol.js +786 -0
- package/dist/case-checkpoint-store.d.ts +29 -0
- package/dist/case-checkpoint-store.js +133 -0
- package/dist/case-pipeline-protocol-v2.d.ts +184 -0
- package/dist/case-pipeline-protocol-v2.js +193 -0
- package/dist/case-pipeline-protocol.d.ts +2626 -0
- package/dist/case-pipeline-protocol.js +371 -0
- package/dist/effect-api.d.ts +74 -12
- package/dist/effect-api.js +56 -7
- package/dist/errors.d.ts +31 -0
- package/dist/errors.js +10 -0
- package/dist/eval-capability-execution-envelope.d.ts +64 -0
- package/dist/eval-capability-execution-envelope.js +98 -0
- package/dist/eval-capability-policy.d.ts +90 -0
- package/dist/eval-capability-policy.js +107 -0
- package/dist/eval-event-log.d.ts +51 -0
- package/dist/eval-event-log.js +69 -10
- package/dist/evaluation-authoring-policy.d.ts +18 -0
- package/dist/evaluation-authoring-policy.js +19 -0
- package/dist/evaluation-authoring-validation.d.ts +22 -0
- package/dist/evaluation-authoring-validation.js +72 -0
- package/dist/evaluation-evidence.d.ts +20 -0
- package/dist/evaluation-evidence.js +319 -0
- package/dist/evaluation-grader-calibration-protocol.d.ts +108 -0
- package/dist/evaluation-grader-calibration-protocol.js +80 -0
- package/dist/evaluation-grader-calibration.d.ts +18 -0
- package/dist/evaluation-grader-calibration.js +334 -0
- package/dist/evaluation-grading-policy.d.ts +24 -0
- package/dist/evaluation-grading-policy.js +54 -0
- package/dist/evaluation-proposal-policy.d.ts +4 -0
- package/dist/evaluation-proposal-policy.js +91 -0
- package/dist/evaluation-source-retrieval.d.ts +68 -0
- package/dist/evaluation-source-retrieval.js +513 -0
- package/dist/evaluation-structure-policy.d.ts +29 -0
- package/dist/evaluation-structure-policy.js +138 -0
- package/dist/index.d.ts +124 -19
- package/dist/index.js +69 -12
- package/dist/inspection.js +2 -3
- package/dist/project-artifacts.d.ts +2 -2
- package/dist/project-artifacts.js +26 -121
- package/dist/project-authoring.d.ts +66 -5
- package/dist/project-authoring.js +783 -109
- package/dist/project-contracts.d.ts +292 -82
- package/dist/project-contracts.js +65 -7
- package/dist/project-store.js +2 -1
- package/dist/project-workflow.d.ts +3 -3
- package/dist/project-workflow.js +116 -18
- package/dist/repository-adversary-protocol.d.ts +64 -0
- package/dist/repository-adversary-protocol.js +105 -0
- package/dist/repository-behavior-protocol.d.ts +188 -0
- package/dist/repository-behavior-protocol.js +202 -0
- package/dist/repository-benchmark-protocol.d.ts +487 -0
- package/dist/repository-benchmark-protocol.js +96 -0
- package/dist/repository-execution-protocol.d.ts +150 -0
- package/dist/repository-execution-protocol.js +38 -0
- package/dist/repository-fixture-instructions.d.ts +3 -0
- package/dist/repository-fixture-instructions.js +91 -0
- package/dist/repository-fixture-protocol.d.ts +79 -0
- package/dist/repository-fixture-protocol.js +79 -0
- package/dist/repository-foundry-plan-protocol.d.ts +118 -0
- package/dist/repository-foundry-plan-protocol.js +296 -0
- package/dist/repository-foundry-progress-protocol.d.ts +52 -0
- package/dist/repository-foundry-progress-protocol.js +52 -0
- package/dist/repository-improvement-protocol.d.ts +100 -0
- package/dist/repository-improvement-protocol.js +106 -0
- package/dist/repository-language-model-protocol.d.ts +43 -0
- package/dist/repository-language-model-protocol.js +146 -0
- package/dist/repository-oracle-coverage-protocol.d.ts +18 -0
- package/dist/repository-oracle-coverage-protocol.js +39 -0
- package/dist/repository-oracle-execution-binding.d.ts +27 -0
- package/dist/repository-oracle-execution-binding.js +59 -0
- package/dist/repository-oracle-protocol.d.ts +230 -0
- package/dist/repository-oracle-protocol.js +156 -0
- package/dist/repository-oracle-scope-policy.d.ts +22 -0
- package/dist/repository-oracle-scope-policy.js +92 -0
- package/dist/repository-quality-policy.d.ts +15 -0
- package/dist/repository-quality-policy.js +357 -0
- package/dist/repository-routing-benchmark-protocol.d.ts +176 -0
- package/dist/repository-routing-benchmark-protocol.js +103 -0
- package/dist/repository-routing-model-protocol.d.ts +36 -0
- package/dist/repository-routing-model-protocol.js +89 -0
- package/dist/repository-routing-plan-protocol.d.ts +112 -0
- package/dist/repository-routing-plan-protocol.js +58 -0
- package/dist/repository-routing-quality-policy.d.ts +9 -0
- package/dist/repository-routing-quality-policy.js +191 -0
- package/dist/repository-seed-qualification-progress-protocol.d.ts +205 -0
- package/dist/repository-seed-qualification-progress-protocol.js +28 -0
- package/dist/repository-semantic-calibration-protocol.d.ts +768 -0
- package/dist/repository-semantic-calibration-protocol.js +276 -0
- package/dist/repository-semantic-calibration.d.ts +163 -0
- package/dist/repository-semantic-calibration.js +581 -0
- package/dist/repository-specification-contract-facts-protocol.d.ts +224 -0
- package/dist/repository-specification-contract-facts-protocol.js +276 -0
- package/dist/repository-specification-critique-protocol.d.ts +189 -0
- package/dist/repository-specification-critique-protocol.js +103 -0
- package/dist/repository-task-family-protocol.d.ts +24 -0
- package/dist/repository-task-family-protocol.js +37 -0
- package/dist/repository-task-seed-protocol.d.ts +384 -0
- package/dist/repository-task-seed-protocol.js +236 -0
- package/dist/repository-trajectory-protocol.d.ts +20 -0
- package/dist/repository-trajectory-protocol.js +42 -0
- package/dist/service.js +1 -1
- package/dist/services/adversary/service.d.ts +64 -0
- package/dist/services/adversary/service.js +330 -0
- package/dist/services/benchmark-compiler/service.d.ts +450 -0
- package/dist/services/benchmark-compiler/service.js +9 -0
- package/dist/services/budgeted-model/service.d.ts +118 -0
- package/dist/services/budgeted-model/service.js +460 -0
- package/dist/services/case-authoring/service.d.ts +163 -0
- package/dist/services/case-authoring/service.js +1456 -0
- package/dist/services/case-finalization/service.d.ts +283 -0
- package/dist/services/case-finalization/service.js +370 -0
- package/dist/services/case-generation/service.d.ts +619 -0
- package/dist/services/case-generation/service.js +2628 -0
- package/dist/services/case-pipeline/service.d.ts +31 -0
- package/dist/services/case-pipeline/service.js +485 -0
- package/dist/services/case-pipeline-v2/service.d.ts +70 -0
- package/dist/services/case-pipeline-v2/service.js +477 -0
- package/dist/services/command-observability/service.d.ts +13 -0
- package/dist/services/command-observability/service.js +3 -0
- package/dist/services/dimension-labeling/service.d.ts +77 -0
- package/dist/services/dimension-labeling/service.js +188 -0
- package/dist/services/eval-candidate/service.d.ts +208 -0
- package/dist/services/eval-candidate/service.js +64 -0
- package/dist/services/eval-capabilities/service.d.ts +183 -0
- package/dist/services/eval-capabilities/service.js +1433 -0
- package/dist/services/eval-environment/service.d.ts +173 -0
- package/dist/services/eval-environment/service.js +127 -0
- package/dist/services/evidence-reconstruction/service.d.ts +36 -0
- package/dist/services/evidence-reconstruction/service.js +145 -0
- package/dist/services/fixture-builder/service.d.ts +62 -0
- package/dist/services/fixture-builder/service.js +36 -0
- package/dist/services/fixture-validation/service.d.ts +75 -0
- package/dist/services/fixture-validation/service.js +295 -0
- package/dist/services/foundry/service.d.ts +831 -0
- package/dist/services/foundry/service.js +442 -0
- package/dist/services/foundry-progress/service.d.ts +62 -0
- package/dist/services/foundry-progress/service.js +149 -0
- package/dist/services/foundry-v2/service.d.ts +54 -0
- package/dist/services/foundry-v2/service.js +28 -0
- package/dist/services/grounded-authoring/service.d.ts +126 -0
- package/dist/services/grounded-authoring/service.js +822 -0
- package/dist/services/historical-case/service.d.ts +722 -0
- package/dist/services/historical-case/service.js +177 -0
- package/dist/services/improvement-loop/service.d.ts +59 -0
- package/dist/services/improvement-loop/service.js +176 -0
- package/dist/services/language-model/service.d.ts +52 -0
- package/dist/services/language-model/service.js +194 -0
- package/dist/services/oracle-builder/service.d.ts +146 -0
- package/dist/services/oracle-builder/service.js +513 -0
- package/dist/services/oracle-coverage/service.d.ts +28 -0
- package/dist/services/oracle-coverage/service.js +50 -0
- package/dist/services/oracle-coverage-witness/service.d.ts +130 -0
- package/dist/services/oracle-coverage-witness/service.js +538 -0
- package/dist/services/pipeline-challenge/service.d.ts +551 -0
- package/dist/services/pipeline-challenge/service.js +427 -0
- package/dist/services/pipeline-controls/service.d.ts +130 -0
- package/dist/services/pipeline-controls/service.js +483 -0
- package/dist/services/pipeline-oracle/service.d.ts +8 -0
- package/dist/services/pipeline-oracle/service.js +256 -0
- package/dist/services/pipeline-seed/service.d.ts +298 -0
- package/dist/services/pipeline-seed/service.js +428 -0
- package/dist/services/pipeline-spec/service.d.ts +103 -0
- package/dist/services/pipeline-spec/service.js +619 -0
- package/dist/services/pipeline-tournament/service.d.ts +258 -0
- package/dist/services/pipeline-tournament/service.js +476 -0
- package/dist/services/quality-gate/service.d.ts +233 -0
- package/dist/services/quality-gate/service.js +136 -0
- package/dist/services/repository-bundle/service.d.ts +33 -0
- package/dist/services/repository-bundle/service.js +114 -0
- package/dist/services/repository-model/service.d.ts +105 -0
- package/dist/services/repository-model/service.js +250 -0
- package/dist/services/repository-public-artifact/service.d.ts +133 -0
- package/dist/services/repository-public-artifact/service.js +330 -0
- package/dist/services/routing-benchmark/service.d.ts +362 -0
- package/dist/services/routing-benchmark/service.js +96 -0
- package/dist/services/specification-critic/service.d.ts +92 -0
- package/dist/services/specification-critic/service.js +172 -0
- package/dist/services/task-family/service.d.ts +40 -0
- package/dist/services/task-family/service.js +55 -0
- package/dist/services/task-seed/service.d.ts +906 -0
- package/dist/services/task-seed/service.js +1406 -0
- package/dist/services/task-specification/service.d.ts +27 -0
- package/dist/services/task-specification/service.js +40 -0
- package/dist/services/trajectory-policy/service.d.ts +110 -0
- package/dist/services/trajectory-policy/service.js +216 -0
- package/dist/test/agentic-capabilities-protocol.test.d.ts +1 -0
- package/dist/test/agentic-capabilities-protocol.test.js +570 -0
- package/dist/test/agentic-capabilities.test.d.ts +1 -0
- package/dist/test/agentic-capabilities.test.js +1461 -0
- package/dist/test/agentic-environment.test.d.ts +1 -0
- package/dist/test/agentic-environment.test.js +213 -0
- package/dist/test/case-pipeline-foundation.test.d.ts +1 -0
- package/dist/test/case-pipeline-foundation.test.js +535 -0
- package/dist/test/case-pipeline-protocol-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-protocol-v2.test.js +124 -0
- package/dist/test/case-pipeline-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-v2.test.js +286 -0
- package/dist/test/case-pipeline.test.d.ts +1 -0
- package/dist/test/case-pipeline.test.js +851 -0
- package/dist/test/eval-capability-policy.test.d.ts +1 -0
- package/dist/test/eval-capability-policy.test.js +50 -0
- package/dist/test/eval-event-log.test.js +34 -1
- package/dist/test/evaluation-evidence-freshness.test.d.ts +1 -0
- package/dist/test/evaluation-evidence-freshness.test.js +44 -0
- package/dist/test/evaluation-evidence.test.d.ts +1 -0
- package/dist/test/evaluation-evidence.test.js +230 -0
- package/dist/test/evaluation-grader-calibration.test.d.ts +1 -0
- package/dist/test/evaluation-grader-calibration.test.js +373 -0
- package/dist/test/evaluation-proposal-digest.test.d.ts +1 -0
- package/dist/test/evaluation-proposal-digest.test.js +187 -0
- package/dist/test/evaluation-source-retrieval.test.d.ts +1 -0
- package/dist/test/evaluation-source-retrieval.test.js +237 -0
- package/dist/test/evaluation-structure-policy.test.d.ts +1 -0
- package/dist/test/evaluation-structure-policy.test.js +196 -0
- package/dist/test/fixtures/repository-resource-panel.d.ts +39 -0
- package/dist/test/fixtures/repository-resource-panel.js +111 -0
- package/dist/test/fixtures/vitest-boundary-panel.d.ts +84 -0
- package/dist/test/fixtures/vitest-boundary-panel.js +120 -0
- package/dist/test/fixtures/vitest-phase-panel.d.ts +135 -0
- package/dist/test/fixtures/vitest-phase-panel.js +213 -0
- package/dist/test/fixtures/vitest-reporter-results.d.ts +76 -0
- package/dist/test/fixtures/vitest-reporter-results.js +94 -0
- package/dist/test/grounded-authoring.test.d.ts +1 -0
- package/dist/test/grounded-authoring.test.js +565 -0
- package/dist/test/integrated-repository-history.test.d.ts +1 -0
- package/dist/test/integrated-repository-history.test.js +227 -0
- package/dist/test/project-authoring.test.js +593 -43
- package/dist/test/project-workflow.test.js +395 -32
- package/dist/test/repository-authoring-artifacts.test.d.ts +1 -0
- package/dist/test/repository-authoring-artifacts.test.js +185 -0
- package/dist/test/repository-bundle.test.d.ts +1 -0
- package/dist/test/repository-bundle.test.js +52 -0
- package/dist/test/repository-case-generation.test.d.ts +1 -0
- package/dist/test/repository-case-generation.test.js +3465 -0
- package/dist/test/repository-command-diagnostic.test.d.ts +1 -0
- package/dist/test/repository-command-diagnostic.test.js +55 -0
- package/dist/test/repository-command-signals.test.d.ts +1 -0
- package/dist/test/repository-command-signals.test.js +124 -0
- package/dist/test/repository-fixture-scope-coverage.test.d.ts +1 -0
- package/dist/test/repository-fixture-scope-coverage.test.js +127 -0
- package/dist/test/repository-fixture-validation.test.d.ts +1 -0
- package/dist/test/repository-fixture-validation.test.js +362 -0
- package/dist/test/repository-foundry-progress.test.d.ts +1 -0
- package/dist/test/repository-foundry-progress.test.js +110 -0
- package/dist/test/repository-foundry-quality.test.d.ts +1 -0
- package/dist/test/repository-foundry-quality.test.js +1138 -0
- package/dist/test/repository-import-context.test.d.ts +1 -0
- package/dist/test/repository-import-context.test.js +354 -0
- package/dist/test/repository-model-authoring.test.d.ts +1 -0
- package/dist/test/repository-model-authoring.test.js +544 -0
- package/dist/test/repository-model.test.d.ts +1 -0
- package/dist/test/repository-model.test.js +2195 -0
- package/dist/test/repository-node-test-reporter.test.d.ts +1 -0
- package/dist/test/repository-node-test-reporter.test.js +104 -0
- package/dist/test/repository-oracle-concurrency.test.d.ts +1 -0
- package/dist/test/repository-oracle-concurrency.test.js +542 -0
- package/dist/test/repository-oracle-coverage-witness.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage-witness.test.js +511 -0
- package/dist/test/repository-oracle-coverage.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage.test.js +168 -0
- package/dist/test/repository-oracle-evidence.test.d.ts +1 -0
- package/dist/test/repository-oracle-evidence.test.js +185 -0
- package/dist/test/repository-oracle-plan.test.d.ts +1 -0
- package/dist/test/repository-oracle-plan.test.js +176 -0
- package/dist/test/repository-overlay-isolation.test.d.ts +1 -0
- package/dist/test/repository-overlay-isolation.test.js +85 -0
- package/dist/test/repository-preparation-cache.test.d.ts +1 -0
- package/dist/test/repository-preparation-cache.test.js +414 -0
- package/dist/test/repository-public-artifact.test.d.ts +1 -0
- package/dist/test/repository-public-artifact.test.js +273 -0
- package/dist/test/repository-qualification-diagnostics.test.d.ts +1 -0
- package/dist/test/repository-qualification-diagnostics.test.js +524 -0
- package/dist/test/repository-reference-authoring.test.d.ts +1 -0
- package/dist/test/repository-reference-authoring.test.js +1633 -0
- package/dist/test/repository-review-evidence-v2.test.d.ts +1 -0
- package/dist/test/repository-review-evidence-v2.test.js +183 -0
- package/dist/test/repository-review-evidence.test.d.ts +1 -0
- package/dist/test/repository-review-evidence.test.js +124 -0
- package/dist/test/repository-seed-exclusions.test.d.ts +1 -0
- package/dist/test/repository-seed-exclusions.test.js +96 -0
- package/dist/test/repository-seed-selection.test.d.ts +1 -0
- package/dist/test/repository-seed-selection.test.js +504 -0
- package/dist/test/repository-semantic-calibration.test.d.ts +1 -0
- package/dist/test/repository-semantic-calibration.test.js +688 -0
- package/dist/test/repository-solution-edits.test.d.ts +1 -0
- package/dist/test/repository-solution-edits.test.js +377 -0
- package/dist/test/repository-specification-budget.test.d.ts +1 -0
- package/dist/test/repository-specification-budget.test.js +171 -0
- package/dist/test/repository-specification-contract-checkpoint.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-checkpoint.test.js +228 -0
- package/dist/test/repository-specification-contract-facts.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-facts.test.js +177 -0
- package/dist/test/repository-trajectory-authoring.test.d.ts +1 -0
- package/dist/test/repository-trajectory-authoring.test.js +176 -0
- package/dist/test/repository-valid-control-plan.test.d.ts +1 -0
- package/dist/test/repository-valid-control-plan.test.js +45 -0
- package/dist/test/repository-vitest-phase.test.d.ts +1 -0
- package/dist/test/repository-vitest-phase.test.js +848 -0
- package/dist/test/repository-vitest-reporter.test.d.ts +1 -0
- package/dist/test/repository-vitest-reporter.test.js +158 -0
- package/dist/test/repository-workspace-build.test.d.ts +1 -0
- package/dist/test/repository-workspace-build.test.js +160 -0
- package/dist/test/strict-authoring-schema.test.d.ts +1 -0
- package/dist/test/strict-authoring-schema.test.js +169 -0
- package/package.json +48 -7
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import type { AgenticCapabilityNameV1 } from "./agentic-capabilities-protocol.js";
|
|
2
|
+
import { type EvalCapabilityScientificPhaseV1 } from "./eval-capability-policy.js";
|
|
3
|
+
export type EvalCapabilityScientificReserveV1 = Readonly<{
|
|
4
|
+
calls: number;
|
|
5
|
+
wallTimeMs: number;
|
|
6
|
+
attempts: number;
|
|
7
|
+
}>;
|
|
8
|
+
export type EvalCapabilityExecutionBudgetV1 = Readonly<{
|
|
9
|
+
calls: number;
|
|
10
|
+
wallTimeMs: number;
|
|
11
|
+
}>;
|
|
12
|
+
/**
|
|
13
|
+
* Time kept outside scientific execution for checkpoint/archive persistence
|
|
14
|
+
* and the host's durable capability-completion receipt.
|
|
15
|
+
*/
|
|
16
|
+
export declare const EVAL_CAPABILITY_DURABLE_FINALIZATION_RESERVE_MS_V1 = 30000;
|
|
17
|
+
export type EvalCapabilityExecutionEnvelopeDecisionV1 = Readonly<{
|
|
18
|
+
version: 1;
|
|
19
|
+
capability: AgenticCapabilityNameV1;
|
|
20
|
+
phase: EvalCapabilityScientificPhaseV1 | null;
|
|
21
|
+
available: EvalCapabilityExecutionBudgetV1;
|
|
22
|
+
requestedMaximum: EvalCapabilityExecutionBudgetV1;
|
|
23
|
+
downstreamReserve: EvalCapabilityScientificReserveV1;
|
|
24
|
+
finalizationReserve: Readonly<{
|
|
25
|
+
wallTimeMs: number;
|
|
26
|
+
}>;
|
|
27
|
+
minimumRequired: EvalCapabilityExecutionBudgetV1;
|
|
28
|
+
availableForCapability: EvalCapabilityExecutionBudgetV1;
|
|
29
|
+
}> & (Readonly<{
|
|
30
|
+
admitted: true;
|
|
31
|
+
envelope: Readonly<{
|
|
32
|
+
maximumModelCalls: number;
|
|
33
|
+
maximumWallTimeMs: number;
|
|
34
|
+
}>;
|
|
35
|
+
}> | Readonly<{
|
|
36
|
+
admitted: false;
|
|
37
|
+
dimension: "calls" | "wall_time";
|
|
38
|
+
envelope: null;
|
|
39
|
+
}>);
|
|
40
|
+
/**
|
|
41
|
+
* The immutable downstream reserve for a scientific capability. This is the
|
|
42
|
+
* single package-owned source used by admission and execution; campaign-size
|
|
43
|
+
* ratios must not weaken these fixed phase reserves.
|
|
44
|
+
*/
|
|
45
|
+
export declare const evalCapabilityScientificDownstreamReserveV1: (capability: AgenticCapabilityNameV1) => Readonly<{
|
|
46
|
+
phase: EvalCapabilityScientificPhaseV1 | null;
|
|
47
|
+
reserve: EvalCapabilityScientificReserveV1;
|
|
48
|
+
}>;
|
|
49
|
+
/**
|
|
50
|
+
* Computes the positive execution slice that fits after preserving the fixed
|
|
51
|
+
* downstream reserve and durable finalization time. Requested maxima are
|
|
52
|
+
* ceilings, not all-or-nothing reservations: a smaller positive slice is
|
|
53
|
+
* admitted whenever one fits.
|
|
54
|
+
*
|
|
55
|
+
* The structured decision retains every value needed to explain or persist the
|
|
56
|
+
* decision without reconstructing it from prose.
|
|
57
|
+
*/
|
|
58
|
+
export declare const evalCapabilityExecutionEnvelopeV1: (input: {
|
|
59
|
+
readonly capability: AgenticCapabilityNameV1;
|
|
60
|
+
readonly requestedMaximumModelCalls?: number;
|
|
61
|
+
readonly requestedMaximumWallMs?: number;
|
|
62
|
+
readonly campaignModelCallsRemaining: number;
|
|
63
|
+
readonly campaignWallTimeRemainingMs: number;
|
|
64
|
+
}) => EvalCapabilityExecutionEnvelopeDecisionV1;
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
import { EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1, evalCapabilityScientificMaximumModelCallsV1, evalCapabilityScientificMaximumWallTimeMsV1, evalCapabilityScientificPhaseV1 } from "./eval-capability-policy.js";
|
|
2
|
+
/**
|
|
3
|
+
* Time kept outside scientific execution for checkpoint/archive persistence
|
|
4
|
+
* and the host's durable capability-completion receipt.
|
|
5
|
+
*/
|
|
6
|
+
export const EVAL_CAPABILITY_DURABLE_FINALIZATION_RESERVE_MS_V1 = 30_000;
|
|
7
|
+
const positiveInteger = (value, fallback) => Number.isSafeInteger(value) && value > 0 ? value : fallback;
|
|
8
|
+
const availableInteger = (value) => Number.isSafeInteger(value) && value > 0 ? value : 0;
|
|
9
|
+
/**
|
|
10
|
+
* The immutable downstream reserve for a scientific capability. This is the
|
|
11
|
+
* single package-owned source used by admission and execution; campaign-size
|
|
12
|
+
* ratios must not weaken these fixed phase reserves.
|
|
13
|
+
*/
|
|
14
|
+
export const evalCapabilityScientificDownstreamReserveV1 = (capability) => {
|
|
15
|
+
const phase = evalCapabilityScientificPhaseV1(capability);
|
|
16
|
+
const reserve = phase === null
|
|
17
|
+
? { calls: 0, wallTimeMs: 0, attempts: 0 }
|
|
18
|
+
: EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1[phase].downstream;
|
|
19
|
+
return { phase, reserve };
|
|
20
|
+
};
|
|
21
|
+
/**
|
|
22
|
+
* Computes the positive execution slice that fits after preserving the fixed
|
|
23
|
+
* downstream reserve and durable finalization time. Requested maxima are
|
|
24
|
+
* ceilings, not all-or-nothing reservations: a smaller positive slice is
|
|
25
|
+
* admitted whenever one fits.
|
|
26
|
+
*
|
|
27
|
+
* The structured decision retains every value needed to explain or persist the
|
|
28
|
+
* decision without reconstructing it from prose.
|
|
29
|
+
*/
|
|
30
|
+
export const evalCapabilityExecutionEnvelopeV1 = (input) => {
|
|
31
|
+
const { phase, reserve } = evalCapabilityScientificDownstreamReserveV1(input.capability);
|
|
32
|
+
const scientific = phase !== null;
|
|
33
|
+
const finalizationReserveWallTimeMs = scientific
|
|
34
|
+
? EVAL_CAPABILITY_DURABLE_FINALIZATION_RESERVE_MS_V1
|
|
35
|
+
: 0;
|
|
36
|
+
const available = {
|
|
37
|
+
calls: availableInteger(input.campaignModelCallsRemaining),
|
|
38
|
+
wallTimeMs: availableInteger(input.campaignWallTimeRemainingMs)
|
|
39
|
+
};
|
|
40
|
+
const requestedMaximum = {
|
|
41
|
+
calls: Math.min(scientific
|
|
42
|
+
? evalCapabilityScientificMaximumModelCallsV1(input.capability)
|
|
43
|
+
: 12, positiveInteger(input.requestedMaximumModelCalls ?? 6, 6)),
|
|
44
|
+
wallTimeMs: Math.min(scientific
|
|
45
|
+
? evalCapabilityScientificMaximumWallTimeMsV1(input.capability)
|
|
46
|
+
: 300_000, positiveInteger(input.requestedMaximumWallMs ?? 120_000, 120_000))
|
|
47
|
+
};
|
|
48
|
+
const minimumRequired = {
|
|
49
|
+
calls: reserve.calls + 1,
|
|
50
|
+
wallTimeMs: reserve.wallTimeMs + finalizationReserveWallTimeMs + 1
|
|
51
|
+
};
|
|
52
|
+
const availableForCapability = {
|
|
53
|
+
calls: Math.max(0, available.calls - reserve.calls),
|
|
54
|
+
wallTimeMs: Math.max(0, available.wallTimeMs -
|
|
55
|
+
reserve.wallTimeMs -
|
|
56
|
+
finalizationReserveWallTimeMs)
|
|
57
|
+
};
|
|
58
|
+
const common = {
|
|
59
|
+
version: 1,
|
|
60
|
+
capability: input.capability,
|
|
61
|
+
phase,
|
|
62
|
+
available,
|
|
63
|
+
requestedMaximum,
|
|
64
|
+
downstreamReserve: reserve,
|
|
65
|
+
finalizationReserve: {
|
|
66
|
+
wallTimeMs: finalizationReserveWallTimeMs
|
|
67
|
+
},
|
|
68
|
+
minimumRequired,
|
|
69
|
+
availableForCapability
|
|
70
|
+
};
|
|
71
|
+
// Read-only/control capabilities remain reachable after scientific exhaustion
|
|
72
|
+
// so callers can inspect progress and finish the campaign. They receive the
|
|
73
|
+
// legacy minimal positive execution fence and no downstream reserve.
|
|
74
|
+
if (!scientific) {
|
|
75
|
+
return {
|
|
76
|
+
...common,
|
|
77
|
+
admitted: true,
|
|
78
|
+
envelope: {
|
|
79
|
+
maximumModelCalls: Math.max(1, Math.min(requestedMaximum.calls, available.calls)),
|
|
80
|
+
maximumWallTimeMs: Math.max(1, Math.min(requestedMaximum.wallTimeMs, available.wallTimeMs))
|
|
81
|
+
}
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
if (availableForCapability.calls < 1) {
|
|
85
|
+
return { ...common, admitted: false, dimension: "calls", envelope: null };
|
|
86
|
+
}
|
|
87
|
+
if (availableForCapability.wallTimeMs < 1) {
|
|
88
|
+
return { ...common, admitted: false, dimension: "wall_time", envelope: null };
|
|
89
|
+
}
|
|
90
|
+
return {
|
|
91
|
+
...common,
|
|
92
|
+
admitted: true,
|
|
93
|
+
envelope: {
|
|
94
|
+
maximumModelCalls: Math.min(requestedMaximum.calls, availableForCapability.calls),
|
|
95
|
+
maximumWallTimeMs: Math.min(requestedMaximum.wallTimeMs, availableForCapability.wallTimeMs)
|
|
96
|
+
}
|
|
97
|
+
};
|
|
98
|
+
};
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
import type { AgenticCapabilityNameV1, AgenticProgressFingerprintV1, AgenticScientificProgressReceiptV1 } from "./agentic-capabilities-protocol.js";
|
|
2
|
+
/**
|
|
3
|
+
* A scheduling slice that only advances verified durable work is not another
|
|
4
|
+
* scientific attempt. Charging it makes sufficiently slow but productive
|
|
5
|
+
* cases impossible, regardless of their remaining campaign budget.
|
|
6
|
+
* New decisions, resets, legacy/unbound receipts, and unchanged work still
|
|
7
|
+
* consume attempts. Calls, spend, wall time and the no-progress circuit are
|
|
8
|
+
* independent and are never refunded.
|
|
9
|
+
*/
|
|
10
|
+
export declare const evalCapabilityScientificAttemptCostV1: (receipt: Pick<AgenticScientificProgressReceiptV1, "previousFingerprint" | "resultingFingerprint" | "inputRevision" | "outputRevision" | "gatesCompleted" | "gatesInvalidated" | "sameFingerprintCount">) => 0 | 1;
|
|
11
|
+
/**
|
|
12
|
+
* No-op calls to other tools cannot reset a capability's stall counter. A
|
|
13
|
+
* genuine accepted-state change does reset it, even if a later repair returns
|
|
14
|
+
* to the same state. Worker admission and Cloud receipt validation share this
|
|
15
|
+
* rule; the global previous-fingerprint chain is verified independently.
|
|
16
|
+
* History must be ordered oldest first and scoped to one candidate revision.
|
|
17
|
+
*/
|
|
18
|
+
export declare const evalCapabilityScientificProgressPredecessorV1: (input: {
|
|
19
|
+
readonly capability: AgenticCapabilityNameV1;
|
|
20
|
+
readonly fingerprint: AgenticProgressFingerprintV1;
|
|
21
|
+
readonly history: readonly AgenticScientificProgressReceiptV1[];
|
|
22
|
+
}) => AgenticScientificProgressReceiptV1 | undefined;
|
|
23
|
+
/**
|
|
24
|
+
* One scientific capability is allowed enough time to finish a complete
|
|
25
|
+
* repository-grounded role chain (analysis, visible authoring, and private
|
|
26
|
+
* review) under normal provider tail latency. Cloud admission, the sandbox
|
|
27
|
+
* executor, and the native tool import this dependency-light policy module so
|
|
28
|
+
* loading an OpenCode tool never initializes the server-side foundry runtime.
|
|
29
|
+
*/
|
|
30
|
+
export declare const EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_MODEL_CALLS_V1 = 12;
|
|
31
|
+
export declare const EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_WALL_TIME_MS_V1: number;
|
|
32
|
+
/**
|
|
33
|
+
* Versioned capability slices are derived from the maximum number of native
|
|
34
|
+
* foundry turns that must remain contiguous to preserve tool state. Most
|
|
35
|
+
* capabilities fit one base slice. Oracle authoring and repair own longer
|
|
36
|
+
* tool-using conversations and therefore receive multiple contiguous slices
|
|
37
|
+
* instead of repeatedly restarting after the base ceiling.
|
|
38
|
+
*/
|
|
39
|
+
export declare const EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1: {
|
|
40
|
+
readonly author_oracle: {
|
|
41
|
+
readonly modelCallSlices: 3;
|
|
42
|
+
readonly wallTimeSlices: 3;
|
|
43
|
+
};
|
|
44
|
+
readonly repair_oracle: {
|
|
45
|
+
readonly modelCallSlices: 2;
|
|
46
|
+
readonly wallTimeSlices: 2;
|
|
47
|
+
};
|
|
48
|
+
};
|
|
49
|
+
export declare const evalCapabilityScientificMaximumModelCallsV1: (capability: AgenticCapabilityNameV1) => number;
|
|
50
|
+
export declare const evalCapabilityScientificMaximumWallTimeMsV1: (capability: AgenticCapabilityNameV1) => number;
|
|
51
|
+
export declare const EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1: {
|
|
52
|
+
readonly specification: {
|
|
53
|
+
readonly capabilities: readonly ["draft_specification"];
|
|
54
|
+
readonly downstream: {
|
|
55
|
+
readonly calls: 24;
|
|
56
|
+
readonly wallTimeMs: 4500000;
|
|
57
|
+
readonly attempts: 6;
|
|
58
|
+
};
|
|
59
|
+
readonly maximumAttempts: 8;
|
|
60
|
+
};
|
|
61
|
+
readonly environment: {
|
|
62
|
+
readonly capabilities: readonly ["plan_environment", "probe_fixture", "qualify_reference"];
|
|
63
|
+
readonly downstream: {
|
|
64
|
+
readonly calls: 16;
|
|
65
|
+
readonly wallTimeMs: 3600000;
|
|
66
|
+
readonly attempts: 4;
|
|
67
|
+
};
|
|
68
|
+
readonly maximumAttempts: 8;
|
|
69
|
+
};
|
|
70
|
+
readonly oracle: {
|
|
71
|
+
readonly capabilities: readonly ["author_oracle", "generate_controls", "evaluate_oracle", "repair_oracle"];
|
|
72
|
+
readonly downstream: {
|
|
73
|
+
readonly calls: 8;
|
|
74
|
+
readonly wallTimeMs: 1800000;
|
|
75
|
+
readonly attempts: 2;
|
|
76
|
+
};
|
|
77
|
+
readonly maximumAttempts: 12;
|
|
78
|
+
};
|
|
79
|
+
readonly admission: {
|
|
80
|
+
readonly capabilities: readonly ["freeze_candidate", "run_held_out_challenge", "request_admission"];
|
|
81
|
+
readonly downstream: {
|
|
82
|
+
readonly calls: 0;
|
|
83
|
+
readonly wallTimeMs: 0;
|
|
84
|
+
readonly attempts: 0;
|
|
85
|
+
};
|
|
86
|
+
readonly maximumAttempts: 6;
|
|
87
|
+
};
|
|
88
|
+
};
|
|
89
|
+
export type EvalCapabilityScientificPhaseV1 = keyof typeof EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1;
|
|
90
|
+
export declare const evalCapabilityScientificPhaseV1: (capability: AgenticCapabilityNameV1) => EvalCapabilityScientificPhaseV1 | null;
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A scheduling slice that only advances verified durable work is not another
|
|
3
|
+
* scientific attempt. Charging it makes sufficiently slow but productive
|
|
4
|
+
* cases impossible, regardless of their remaining campaign budget.
|
|
5
|
+
* New decisions, resets, legacy/unbound receipts, and unchanged work still
|
|
6
|
+
* consume attempts. Calls, spend, wall time and the no-progress circuit are
|
|
7
|
+
* independent and are never refunded.
|
|
8
|
+
*/
|
|
9
|
+
export const evalCapabilityScientificAttemptCostV1 = (receipt) => {
|
|
10
|
+
const previous = receipt.previousFingerprint;
|
|
11
|
+
const next = receipt.resultingFingerprint;
|
|
12
|
+
return previous !== null &&
|
|
13
|
+
receipt.inputRevision !== null &&
|
|
14
|
+
receipt.inputRevision === receipt.outputRevision &&
|
|
15
|
+
receipt.sameFingerprintCount === 0 &&
|
|
16
|
+
receipt.gatesCompleted.length === 0 &&
|
|
17
|
+
receipt.gatesInvalidated.length === 0 &&
|
|
18
|
+
next.continuationProgressDigest != null &&
|
|
19
|
+
next.continuationProgressDigest !== (previous.continuationProgressDigest ?? null) &&
|
|
20
|
+
next.completedGateDigest === previous.completedGateDigest &&
|
|
21
|
+
next.environmentDigest === previous.environmentDigest
|
|
22
|
+
? 0
|
|
23
|
+
: 1;
|
|
24
|
+
};
|
|
25
|
+
/**
|
|
26
|
+
* No-op calls to other tools cannot reset a capability's stall counter. A
|
|
27
|
+
* genuine accepted-state change does reset it, even if a later repair returns
|
|
28
|
+
* to the same state. Worker admission and Cloud receipt validation share this
|
|
29
|
+
* rule; the global previous-fingerprint chain is verified independently.
|
|
30
|
+
* History must be ordered oldest first and scoped to one candidate revision.
|
|
31
|
+
*/
|
|
32
|
+
export const evalCapabilityScientificProgressPredecessorV1 = (input) => {
|
|
33
|
+
for (let index = input.history.length - 1; index >= 0; index -= 1) {
|
|
34
|
+
const receipt = input.history[index];
|
|
35
|
+
const fingerprint = receipt.resultingFingerprint;
|
|
36
|
+
if (fingerprint.acceptedEvidenceDigest !== input.fingerprint.acceptedEvidenceDigest ||
|
|
37
|
+
fingerprint.completedGateDigest !== input.fingerprint.completedGateDigest ||
|
|
38
|
+
fingerprint.environmentDigest !== input.fingerprint.environmentDigest ||
|
|
39
|
+
(fingerprint.continuationProgressDigest ?? null) !==
|
|
40
|
+
(input.fingerprint.continuationProgressDigest ?? null))
|
|
41
|
+
return undefined;
|
|
42
|
+
if (receipt.capability === input.capability)
|
|
43
|
+
return receipt;
|
|
44
|
+
}
|
|
45
|
+
return undefined;
|
|
46
|
+
};
|
|
47
|
+
/**
|
|
48
|
+
* One scientific capability is allowed enough time to finish a complete
|
|
49
|
+
* repository-grounded role chain (analysis, visible authoring, and private
|
|
50
|
+
* review) under normal provider tail latency. Cloud admission, the sandbox
|
|
51
|
+
* executor, and the native tool import this dependency-light policy module so
|
|
52
|
+
* loading an OpenCode tool never initializes the server-side foundry runtime.
|
|
53
|
+
*/
|
|
54
|
+
export const EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_MODEL_CALLS_V1 = 12;
|
|
55
|
+
export const EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_WALL_TIME_MS_V1 = 10 * 60_000;
|
|
56
|
+
/**
|
|
57
|
+
* Versioned capability slices are derived from the maximum number of native
|
|
58
|
+
* foundry turns that must remain contiguous to preserve tool state. Most
|
|
59
|
+
* capabilities fit one base slice. Oracle authoring and repair own longer
|
|
60
|
+
* tool-using conversations and therefore receive multiple contiguous slices
|
|
61
|
+
* instead of repeatedly restarting after the base ceiling.
|
|
62
|
+
*/
|
|
63
|
+
export const EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1 = {
|
|
64
|
+
author_oracle: { modelCallSlices: 3, wallTimeSlices: 3 },
|
|
65
|
+
repair_oracle: { modelCallSlices: 2, wallTimeSlices: 2 }
|
|
66
|
+
};
|
|
67
|
+
export const evalCapabilityScientificMaximumModelCallsV1 = (capability) => EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_MODEL_CALLS_V1 *
|
|
68
|
+
(capability in EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1
|
|
69
|
+
? EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1[capability].modelCallSlices
|
|
70
|
+
: 1);
|
|
71
|
+
export const evalCapabilityScientificMaximumWallTimeMsV1 = (capability) => EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_WALL_TIME_MS_V1 *
|
|
72
|
+
(capability in EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1
|
|
73
|
+
? EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1[capability].wallTimeSlices
|
|
74
|
+
: 1);
|
|
75
|
+
export const EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1 = {
|
|
76
|
+
specification: {
|
|
77
|
+
capabilities: ["draft_specification"],
|
|
78
|
+
// This is the minimum protected envelope for environment qualification,
|
|
79
|
+
// oracle/control construction, held-out challenge, and admission. It is not
|
|
80
|
+
// a per-turn timeout. A smaller campaign can start useful work but cannot
|
|
81
|
+
// complete the mandatory scientific path.
|
|
82
|
+
downstream: { calls: 24, wallTimeMs: 4_500_000, attempts: 6 },
|
|
83
|
+
maximumAttempts: 8
|
|
84
|
+
},
|
|
85
|
+
environment: {
|
|
86
|
+
capabilities: ["plan_environment", "probe_fixture", "qualify_reference"],
|
|
87
|
+
downstream: { calls: 16, wallTimeMs: 3_600_000, attempts: 4 },
|
|
88
|
+
maximumAttempts: 8
|
|
89
|
+
},
|
|
90
|
+
oracle: {
|
|
91
|
+
capabilities: ["author_oracle", "generate_controls", "evaluate_oracle", "repair_oracle"],
|
|
92
|
+
downstream: { calls: 8, wallTimeMs: 1_800_000, attempts: 2 },
|
|
93
|
+
maximumAttempts: 12
|
|
94
|
+
},
|
|
95
|
+
admission: {
|
|
96
|
+
capabilities: ["freeze_candidate", "run_held_out_challenge", "request_admission"],
|
|
97
|
+
downstream: { calls: 0, wallTimeMs: 0, attempts: 0 },
|
|
98
|
+
maximumAttempts: 6
|
|
99
|
+
}
|
|
100
|
+
};
|
|
101
|
+
export const evalCapabilityScientificPhaseV1 = (capability) => {
|
|
102
|
+
for (const [phase, policy] of Object.entries(EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1)) {
|
|
103
|
+
if (policy.capabilities.includes(capability))
|
|
104
|
+
return phase;
|
|
105
|
+
}
|
|
106
|
+
return null;
|
|
107
|
+
};
|
package/dist/eval-event-log.d.ts
CHANGED
|
@@ -1,4 +1,6 @@
|
|
|
1
1
|
import { Schema } from "effect";
|
|
2
|
+
export declare const EvalRunFailurePhase: Schema.Literals<readonly ["target", "classifier", "suite-inspection", "comparison", "activation", "accounting", "completion"]>;
|
|
3
|
+
export type EvalRunFailurePhase = typeof EvalRunFailurePhase.Type;
|
|
2
4
|
export declare const EvalNamedEvent: Schema.Union<readonly [Schema.Struct<{
|
|
3
5
|
readonly name: Schema.Literal<"plan">;
|
|
4
6
|
readonly at: Schema.String;
|
|
@@ -18,6 +20,30 @@ export declare const EvalNamedEvent: Schema.Union<readonly [Schema.Struct<{
|
|
|
18
20
|
readonly caseId: Schema.String;
|
|
19
21
|
readonly sequence: Schema.Finite;
|
|
20
22
|
readonly name: Schema.Literal<"judge#">;
|
|
23
|
+
}>, Schema.Struct<{
|
|
24
|
+
readonly name: Schema.Literal<"failure-detail">;
|
|
25
|
+
readonly at: Schema.String;
|
|
26
|
+
readonly runId: Schema.String;
|
|
27
|
+
readonly phase: Schema.Literals<readonly ["target", "classifier", "suite-inspection", "comparison", "activation", "accounting", "completion"]>;
|
|
28
|
+
readonly reason: Schema.String;
|
|
29
|
+
readonly retryable: Schema.Boolean;
|
|
30
|
+
readonly callId: Schema.optionalKey<Schema.String>;
|
|
31
|
+
readonly reservationId: Schema.optionalKey<Schema.String>;
|
|
32
|
+
readonly accounting: Schema.Struct<{
|
|
33
|
+
readonly expectedCalls: Schema.Finite;
|
|
34
|
+
readonly observedCalls: Schema.Finite;
|
|
35
|
+
readonly knownInputTokens: Schema.Finite;
|
|
36
|
+
readonly knownOutputTokens: Schema.Finite;
|
|
37
|
+
readonly unknownTokenMeasurements: Schema.Finite;
|
|
38
|
+
readonly knownPricedSubtotalUsd: Schema.Finite;
|
|
39
|
+
readonly unpricedCalls: Schema.Finite;
|
|
40
|
+
readonly subscriptionUnitCalls: Schema.Finite;
|
|
41
|
+
}>;
|
|
42
|
+
readonly cleanup: Schema.Struct<{
|
|
43
|
+
readonly sessionOpened: Schema.Boolean;
|
|
44
|
+
readonly sessionClosed: Schema.Boolean;
|
|
45
|
+
readonly detail: Schema.optionalKey<Schema.String>;
|
|
46
|
+
}>;
|
|
21
47
|
}>, Schema.Struct<{
|
|
22
48
|
readonly at: Schema.String;
|
|
23
49
|
readonly operationId: Schema.String;
|
|
@@ -51,6 +77,30 @@ export declare const EvalNamedEventLog: Schema.$Array<Schema.Union<readonly [Sch
|
|
|
51
77
|
readonly caseId: Schema.String;
|
|
52
78
|
readonly sequence: Schema.Finite;
|
|
53
79
|
readonly name: Schema.Literal<"judge#">;
|
|
80
|
+
}>, Schema.Struct<{
|
|
81
|
+
readonly name: Schema.Literal<"failure-detail">;
|
|
82
|
+
readonly at: Schema.String;
|
|
83
|
+
readonly runId: Schema.String;
|
|
84
|
+
readonly phase: Schema.Literals<readonly ["target", "classifier", "suite-inspection", "comparison", "activation", "accounting", "completion"]>;
|
|
85
|
+
readonly reason: Schema.String;
|
|
86
|
+
readonly retryable: Schema.Boolean;
|
|
87
|
+
readonly callId: Schema.optionalKey<Schema.String>;
|
|
88
|
+
readonly reservationId: Schema.optionalKey<Schema.String>;
|
|
89
|
+
readonly accounting: Schema.Struct<{
|
|
90
|
+
readonly expectedCalls: Schema.Finite;
|
|
91
|
+
readonly observedCalls: Schema.Finite;
|
|
92
|
+
readonly knownInputTokens: Schema.Finite;
|
|
93
|
+
readonly knownOutputTokens: Schema.Finite;
|
|
94
|
+
readonly unknownTokenMeasurements: Schema.Finite;
|
|
95
|
+
readonly knownPricedSubtotalUsd: Schema.Finite;
|
|
96
|
+
readonly unpricedCalls: Schema.Finite;
|
|
97
|
+
readonly subscriptionUnitCalls: Schema.Finite;
|
|
98
|
+
}>;
|
|
99
|
+
readonly cleanup: Schema.Struct<{
|
|
100
|
+
readonly sessionOpened: Schema.Boolean;
|
|
101
|
+
readonly sessionClosed: Schema.Boolean;
|
|
102
|
+
readonly detail: Schema.optionalKey<Schema.String>;
|
|
103
|
+
}>;
|
|
54
104
|
}>, Schema.Struct<{
|
|
55
105
|
readonly at: Schema.String;
|
|
56
106
|
readonly operationId: Schema.String;
|
|
@@ -86,4 +136,5 @@ export type EvalEventLogPresentation = {
|
|
|
86
136
|
export declare function presentEvalEventLog(log: EvalNamedEventLog, options?: {
|
|
87
137
|
readonly failHoles?: boolean;
|
|
88
138
|
}): EvalEventLogPresentation;
|
|
139
|
+
export declare function assertEvalRunCanFail(log: EvalNamedEventLog): void;
|
|
89
140
|
export declare function assertEvalRunCanComplete(log: EvalNamedEventLog): void;
|
package/dist/eval-event-log.js
CHANGED
|
@@ -1,6 +1,10 @@
|
|
|
1
1
|
import { Schema } from "effect";
|
|
2
2
|
const NonNegativeInteger = Schema.Finite.pipe(Schema.check(Schema.makeFilter((value) => value >= 0 && Number.isInteger(value) ? undefined : "value must be a non-negative integer")));
|
|
3
3
|
const PositiveInteger = NonNegativeInteger.pipe(Schema.check(Schema.makeFilter((value) => value >= 1 ? undefined : "value must be a positive integer")));
|
|
4
|
+
const NonNegativeFinite = Schema.Finite.pipe(Schema.check(Schema.makeFilter((value) => value >= 0 ? undefined : "value must be a non-negative finite number")));
|
|
5
|
+
const BoundedFailureReason = Schema.String.pipe(Schema.check(Schema.makeFilter((value) => value.length >= 1 && value.length <= 2_000
|
|
6
|
+
? undefined
|
|
7
|
+
: "failure reason must contain between 1 and 2000 characters")));
|
|
4
8
|
const RunPlanEvent = Schema.Struct({
|
|
5
9
|
name: Schema.Literal("plan"),
|
|
6
10
|
at: Schema.String,
|
|
@@ -22,6 +26,40 @@ const RunJudgeEvent = Schema.Struct({
|
|
|
22
26
|
name: Schema.Literal("judge#"),
|
|
23
27
|
...RunCallEventFields
|
|
24
28
|
});
|
|
29
|
+
export const EvalRunFailurePhase = Schema.Literals([
|
|
30
|
+
"target",
|
|
31
|
+
"classifier",
|
|
32
|
+
"suite-inspection",
|
|
33
|
+
"comparison",
|
|
34
|
+
"activation",
|
|
35
|
+
"accounting",
|
|
36
|
+
"completion"
|
|
37
|
+
]);
|
|
38
|
+
const RunFailureDetailEvent = Schema.Struct({
|
|
39
|
+
name: Schema.Literal("failure-detail"),
|
|
40
|
+
at: Schema.String,
|
|
41
|
+
runId: Schema.String,
|
|
42
|
+
phase: EvalRunFailurePhase,
|
|
43
|
+
reason: BoundedFailureReason,
|
|
44
|
+
retryable: Schema.Boolean,
|
|
45
|
+
callId: Schema.optionalKey(Schema.String),
|
|
46
|
+
reservationId: Schema.optionalKey(Schema.String),
|
|
47
|
+
accounting: Schema.Struct({
|
|
48
|
+
expectedCalls: NonNegativeInteger,
|
|
49
|
+
observedCalls: NonNegativeInteger,
|
|
50
|
+
knownInputTokens: NonNegativeInteger,
|
|
51
|
+
knownOutputTokens: NonNegativeInteger,
|
|
52
|
+
unknownTokenMeasurements: NonNegativeInteger,
|
|
53
|
+
knownPricedSubtotalUsd: NonNegativeFinite,
|
|
54
|
+
unpricedCalls: NonNegativeInteger,
|
|
55
|
+
subscriptionUnitCalls: NonNegativeInteger
|
|
56
|
+
}),
|
|
57
|
+
cleanup: Schema.Struct({
|
|
58
|
+
sessionOpened: Schema.Boolean,
|
|
59
|
+
sessionClosed: Schema.Boolean,
|
|
60
|
+
detail: Schema.optionalKey(Schema.String)
|
|
61
|
+
})
|
|
62
|
+
});
|
|
25
63
|
const AuthoringEventFields = {
|
|
26
64
|
at: Schema.String,
|
|
27
65
|
operationId: Schema.String
|
|
@@ -42,6 +80,7 @@ export const EvalNamedEvent = Schema.Union([
|
|
|
42
80
|
RunPlanEvent,
|
|
43
81
|
RunCandidateEvent,
|
|
44
82
|
RunJudgeEvent,
|
|
83
|
+
RunFailureDetailEvent,
|
|
45
84
|
AuthoringBasisEvent,
|
|
46
85
|
AuthoringDimensionsEvent,
|
|
47
86
|
AuthoringActivationEvent
|
|
@@ -125,6 +164,9 @@ export function presentEvalEventLog(log, options = {}) {
|
|
|
125
164
|
const cursors = resumeEvalCaseCursors(log);
|
|
126
165
|
const failHoles = options.failHoles ?? true;
|
|
127
166
|
const lastEvent = log[log.length - 1];
|
|
167
|
+
const failureLine = lastEvent?.name === "failure-detail"
|
|
168
|
+
? `run fail ${lastEvent.phase}: ${lastEvent.reason}`
|
|
169
|
+
: undefined;
|
|
128
170
|
const authoringLine = lastEvent?.name === "basis" ||
|
|
129
171
|
lastEvent?.name === "dimensions" ||
|
|
130
172
|
lastEvent?.name === "activation"
|
|
@@ -137,23 +179,40 @@ export function presentEvalEventLog(log, options = {}) {
|
|
|
137
179
|
: `fail ${label} missing judge#${String(cursor.hole.sequence)}`}`;
|
|
138
180
|
});
|
|
139
181
|
return {
|
|
140
|
-
failed:
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
182
|
+
failed: failureLine !== undefined ||
|
|
183
|
+
(failHoles && cursors.some((cursor) => cursor.hole !== undefined)),
|
|
184
|
+
lines: failureLine !== undefined
|
|
185
|
+
? [...cursorLines, failureLine]
|
|
186
|
+
: cursors.length === 0
|
|
187
|
+
? authoringLine === undefined
|
|
188
|
+
? plan === undefined
|
|
189
|
+
? []
|
|
190
|
+
: [`run ${evalEventLabel(plan)}`]
|
|
191
|
+
: [authoringLine]
|
|
192
|
+
: authoringLine === undefined
|
|
193
|
+
? cursorLines
|
|
194
|
+
: [...cursorLines, authoringLine]
|
|
150
195
|
};
|
|
151
196
|
}
|
|
197
|
+
export function assertEvalRunCanFail(log) {
|
|
198
|
+
const plan = lastPlanEvent(log);
|
|
199
|
+
if (plan === undefined) {
|
|
200
|
+
throw new Error("failed eval run cannot terminate without its plan event");
|
|
201
|
+
}
|
|
202
|
+
const terminal = log[log.length - 1];
|
|
203
|
+
if (terminal?.name !== "failure-detail" || terminal.runId !== plan.runId) {
|
|
204
|
+
throw new Error("failed eval run must end with a matching terminal failure-detail event");
|
|
205
|
+
}
|
|
206
|
+
}
|
|
152
207
|
export function assertEvalRunCanComplete(log) {
|
|
153
208
|
const plan = lastPlanEvent(log);
|
|
154
209
|
if (plan === undefined || plan.plannedCalls < 1) {
|
|
155
210
|
throw new Error("eval run cannot complete without a plan containing at least one call");
|
|
156
211
|
}
|
|
212
|
+
const failure = log.find((event) => event.name === "failure-detail" && event.runId === plan.runId);
|
|
213
|
+
if (failure !== undefined) {
|
|
214
|
+
throw new Error(`eval run cannot complete after terminal failure in phase ${JSON.stringify(failure.phase)}`);
|
|
215
|
+
}
|
|
157
216
|
const hole = resumeEvalCaseCursors(log).find((cursor) => cursor.hole !== undefined);
|
|
158
217
|
if (hole?.hole !== undefined) {
|
|
159
218
|
throw new Error(`eval case ${JSON.stringify(hole.dimensionId)}/${JSON.stringify(hole.caseId)} cannot complete after ${evalEventLabel(hole.hole)} without judge#${String(hole.hole.sequence)}`);
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Provisional transport limits for compiler-backed evaluation authoring.
|
|
3
|
+
*
|
|
4
|
+
* These values bound model input/output and persisted candidate context; they
|
|
5
|
+
* are not semantic-quality thresholds. Live preflight and later empirical
|
|
6
|
+
* calibration must determine whether they produce coherent, useful cases.
|
|
7
|
+
*/
|
|
8
|
+
export declare const EVAL_AUTHORING_EVIDENCE_BLOCK_BYTES = 3000;
|
|
9
|
+
export declare const EVAL_AUTHORING_CASE_EVIDENCE_BYTES = 12000;
|
|
10
|
+
/**
|
|
11
|
+
* Maximum compiler blocks exposed as one mechanically safe author selection.
|
|
12
|
+
*
|
|
13
|
+
* A block may consume the complete block-byte allowance, so this cardinality
|
|
14
|
+
* guarantees that selecting every exposed block cannot exceed the case context
|
|
15
|
+
* allowance. The exact byte check remains authoritative because smaller blocks
|
|
16
|
+
* do not make every same-cardinality combination equivalent.
|
|
17
|
+
*/
|
|
18
|
+
export declare const EVAL_AUTHORING_CASE_EVIDENCE_BLOCKS: number;
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Provisional transport limits for compiler-backed evaluation authoring.
|
|
3
|
+
*
|
|
4
|
+
* These values bound model input/output and persisted candidate context; they
|
|
5
|
+
* are not semantic-quality thresholds. Live preflight and later empirical
|
|
6
|
+
* calibration must determine whether they produce coherent, useful cases.
|
|
7
|
+
*/
|
|
8
|
+
export const EVAL_AUTHORING_EVIDENCE_BLOCK_BYTES = 3_000;
|
|
9
|
+
export const EVAL_AUTHORING_CASE_EVIDENCE_BYTES = 12_000;
|
|
10
|
+
/**
|
|
11
|
+
* Maximum compiler blocks exposed as one mechanically safe author selection.
|
|
12
|
+
*
|
|
13
|
+
* A block may consume the complete block-byte allowance, so this cardinality
|
|
14
|
+
* guarantees that selecting every exposed block cannot exceed the case context
|
|
15
|
+
* allowance. The exact byte check remains authoritative because smaller blocks
|
|
16
|
+
* do not make every same-cardinality combination equivalent.
|
|
17
|
+
*/
|
|
18
|
+
export const EVAL_AUTHORING_CASE_EVIDENCE_BLOCKS = Math.floor(EVAL_AUTHORING_CASE_EVIDENCE_BYTES /
|
|
19
|
+
EVAL_AUTHORING_EVIDENCE_BLOCK_BYTES);
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
export type EvalAuthoringValidationCategory = "schema-constraint" | "schema-decode" | "case-shape" | "criterion-topology" | "control-topology" | "evidence-selection";
|
|
2
|
+
export declare class EvalAuthoringValidationIssue extends Error {
|
|
3
|
+
readonly category: EvalAuthoringValidationCategory;
|
|
4
|
+
readonly path: string;
|
|
5
|
+
constructor(input: {
|
|
6
|
+
readonly category: EvalAuthoringValidationCategory;
|
|
7
|
+
readonly path: string;
|
|
8
|
+
readonly detail: string;
|
|
9
|
+
readonly cause?: unknown;
|
|
10
|
+
});
|
|
11
|
+
}
|
|
12
|
+
export declare const evalAuthoringValidationDiagnostic: (cause: unknown) => {
|
|
13
|
+
readonly category: EvalAuthoringValidationCategory;
|
|
14
|
+
readonly path: string;
|
|
15
|
+
readonly detail: string;
|
|
16
|
+
};
|
|
17
|
+
export declare const prefixEvalAuthoringValidationIssue: (cause: unknown, prefix: string) => EvalAuthoringValidationIssue;
|
|
18
|
+
export declare const renderEvalAuthoringValidationFailure: (input: {
|
|
19
|
+
readonly scope: string;
|
|
20
|
+
readonly cause: unknown;
|
|
21
|
+
readonly callId?: string;
|
|
22
|
+
}) => string;
|