@velum-labs/routekit-eval-setup 1.3.2 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/authoring-responses-request.d.ts +5 -0
- package/dist/adapters/authoring-responses-request.js +38 -0
- package/dist/adapters/evaluation-evidence-freshness.d.ts +7 -0
- package/dist/adapters/evaluation-evidence-freshness.js +76 -0
- package/dist/adapters/git-task-history.d.ts +67 -0
- package/dist/adapters/git-task-history.js +171 -0
- package/dist/adapters/integrated-repository-history.d.ts +21 -0
- package/dist/adapters/integrated-repository-history.js +175 -0
- package/dist/adapters/repository-command-diagnostic.d.ts +8 -0
- package/dist/adapters/repository-command-diagnostic.js +46 -0
- package/dist/adapters/repository-command-evidence.d.ts +13 -0
- package/dist/adapters/repository-command-evidence.js +102 -0
- package/dist/adapters/repository-command-runner.d.ts +238 -0
- package/dist/adapters/repository-command-runner.js +1483 -0
- package/dist/adapters/repository-import-context.d.ts +47 -0
- package/dist/adapters/repository-import-context.js +469 -0
- package/dist/adapters/repository-node-test-reporter.d.ts +3 -0
- package/dist/adapters/repository-node-test-reporter.js +27 -0
- package/dist/adapters/repository-review-evidence.d.ts +39 -0
- package/dist/adapters/repository-review-evidence.js +632 -0
- package/dist/adapters/repository-seed-selection.d.ts +7 -0
- package/dist/adapters/repository-seed-selection.js +79 -0
- package/dist/adapters/repository-solution-edits.d.ts +49 -0
- package/dist/adapters/repository-solution-edits.js +136 -0
- package/dist/adapters/repository-vitest-phase-adapter.d.ts +8 -0
- package/dist/adapters/repository-vitest-phase-adapter.js +310 -0
- package/dist/adapters/repository-vitest-reporter.d.ts +24 -0
- package/dist/adapters/repository-vitest-reporter.js +314 -0
- package/dist/adapters/strict-authoring-schema.d.ts +5 -0
- package/dist/adapters/strict-authoring-schema.js +158 -0
- package/dist/adapters/test-discovery.d.ts +30 -0
- package/dist/adapters/test-discovery.js +124 -0
- package/dist/adapters/typescript-repository-index.d.ts +51 -0
- package/dist/adapters/typescript-repository-index.js +226 -0
- package/dist/agentic-capabilities-protocol.d.ts +1373 -0
- package/dist/agentic-capabilities-protocol.js +786 -0
- package/dist/case-checkpoint-store.d.ts +29 -0
- package/dist/case-checkpoint-store.js +133 -0
- package/dist/case-pipeline-protocol-v2.d.ts +184 -0
- package/dist/case-pipeline-protocol-v2.js +193 -0
- package/dist/case-pipeline-protocol.d.ts +2626 -0
- package/dist/case-pipeline-protocol.js +371 -0
- package/dist/effect-api.d.ts +74 -10
- package/dist/effect-api.js +56 -6
- package/dist/errors.d.ts +31 -0
- package/dist/errors.js +10 -0
- package/dist/eval-capability-execution-envelope.d.ts +64 -0
- package/dist/eval-capability-execution-envelope.js +98 -0
- package/dist/eval-capability-policy.d.ts +90 -0
- package/dist/eval-capability-policy.js +107 -0
- package/dist/eval-event-log.d.ts +140 -0
- package/dist/eval-event-log.js +220 -0
- package/dist/evaluation-authoring-policy.d.ts +18 -0
- package/dist/evaluation-authoring-policy.js +19 -0
- package/dist/evaluation-authoring-validation.d.ts +22 -0
- package/dist/evaluation-authoring-validation.js +72 -0
- package/dist/evaluation-evidence.d.ts +20 -0
- package/dist/evaluation-evidence.js +319 -0
- package/dist/evaluation-grader-calibration-protocol.d.ts +108 -0
- package/dist/evaluation-grader-calibration-protocol.js +80 -0
- package/dist/evaluation-grader-calibration.d.ts +18 -0
- package/dist/evaluation-grader-calibration.js +334 -0
- package/dist/evaluation-grading-policy.d.ts +24 -0
- package/dist/evaluation-grading-policy.js +54 -0
- package/dist/evaluation-proposal-policy.d.ts +4 -0
- package/dist/evaluation-proposal-policy.js +91 -0
- package/dist/evaluation-source-retrieval.d.ts +68 -0
- package/dist/evaluation-source-retrieval.js +513 -0
- package/dist/evaluation-structure-policy.d.ts +29 -0
- package/dist/evaluation-structure-policy.js +138 -0
- package/dist/index.d.ts +124 -17
- package/dist/index.js +69 -11
- package/dist/inspection.js +2 -3
- package/dist/project-artifacts.d.ts +7 -2
- package/dist/project-artifacts.js +49 -136
- package/dist/project-authoring.d.ts +66 -5
- package/dist/project-authoring.js +783 -109
- package/dist/project-contracts.d.ts +419 -84
- package/dist/project-contracts.js +160 -52
- package/dist/project-store.js +2 -1
- package/dist/project-workflow.d.ts +5 -4
- package/dist/project-workflow.js +154 -35
- package/dist/repository-adversary-protocol.d.ts +64 -0
- package/dist/repository-adversary-protocol.js +105 -0
- package/dist/repository-behavior-protocol.d.ts +188 -0
- package/dist/repository-behavior-protocol.js +202 -0
- package/dist/repository-benchmark-protocol.d.ts +487 -0
- package/dist/repository-benchmark-protocol.js +96 -0
- package/dist/repository-execution-protocol.d.ts +150 -0
- package/dist/repository-execution-protocol.js +38 -0
- package/dist/repository-fixture-instructions.d.ts +3 -0
- package/dist/repository-fixture-instructions.js +91 -0
- package/dist/repository-fixture-protocol.d.ts +79 -0
- package/dist/repository-fixture-protocol.js +79 -0
- package/dist/repository-foundry-plan-protocol.d.ts +118 -0
- package/dist/repository-foundry-plan-protocol.js +296 -0
- package/dist/repository-foundry-progress-protocol.d.ts +52 -0
- package/dist/repository-foundry-progress-protocol.js +52 -0
- package/dist/repository-improvement-protocol.d.ts +100 -0
- package/dist/repository-improvement-protocol.js +106 -0
- package/dist/repository-language-model-protocol.d.ts +43 -0
- package/dist/repository-language-model-protocol.js +146 -0
- package/dist/repository-oracle-coverage-protocol.d.ts +18 -0
- package/dist/repository-oracle-coverage-protocol.js +39 -0
- package/dist/repository-oracle-execution-binding.d.ts +27 -0
- package/dist/repository-oracle-execution-binding.js +59 -0
- package/dist/repository-oracle-protocol.d.ts +230 -0
- package/dist/repository-oracle-protocol.js +156 -0
- package/dist/repository-oracle-scope-policy.d.ts +22 -0
- package/dist/repository-oracle-scope-policy.js +92 -0
- package/dist/repository-quality-policy.d.ts +15 -0
- package/dist/repository-quality-policy.js +357 -0
- package/dist/repository-routing-benchmark-protocol.d.ts +176 -0
- package/dist/repository-routing-benchmark-protocol.js +103 -0
- package/dist/repository-routing-model-protocol.d.ts +36 -0
- package/dist/repository-routing-model-protocol.js +89 -0
- package/dist/repository-routing-plan-protocol.d.ts +112 -0
- package/dist/repository-routing-plan-protocol.js +58 -0
- package/dist/repository-routing-quality-policy.d.ts +9 -0
- package/dist/repository-routing-quality-policy.js +191 -0
- package/dist/repository-seed-qualification-progress-protocol.d.ts +205 -0
- package/dist/repository-seed-qualification-progress-protocol.js +28 -0
- package/dist/repository-semantic-calibration-protocol.d.ts +768 -0
- package/dist/repository-semantic-calibration-protocol.js +276 -0
- package/dist/repository-semantic-calibration.d.ts +163 -0
- package/dist/repository-semantic-calibration.js +581 -0
- package/dist/repository-specification-contract-facts-protocol.d.ts +224 -0
- package/dist/repository-specification-contract-facts-protocol.js +276 -0
- package/dist/repository-specification-critique-protocol.d.ts +189 -0
- package/dist/repository-specification-critique-protocol.js +103 -0
- package/dist/repository-task-family-protocol.d.ts +24 -0
- package/dist/repository-task-family-protocol.js +37 -0
- package/dist/repository-task-seed-protocol.d.ts +384 -0
- package/dist/repository-task-seed-protocol.js +236 -0
- package/dist/repository-trajectory-protocol.d.ts +20 -0
- package/dist/repository-trajectory-protocol.js +42 -0
- package/dist/service.js +1 -1
- package/dist/services/adversary/service.d.ts +64 -0
- package/dist/services/adversary/service.js +330 -0
- package/dist/services/benchmark-compiler/service.d.ts +450 -0
- package/dist/services/benchmark-compiler/service.js +9 -0
- package/dist/services/budgeted-model/service.d.ts +118 -0
- package/dist/services/budgeted-model/service.js +460 -0
- package/dist/services/case-authoring/service.d.ts +163 -0
- package/dist/services/case-authoring/service.js +1456 -0
- package/dist/services/case-finalization/service.d.ts +283 -0
- package/dist/services/case-finalization/service.js +370 -0
- package/dist/services/case-generation/service.d.ts +619 -0
- package/dist/services/case-generation/service.js +2628 -0
- package/dist/services/case-pipeline/service.d.ts +31 -0
- package/dist/services/case-pipeline/service.js +485 -0
- package/dist/services/case-pipeline-v2/service.d.ts +70 -0
- package/dist/services/case-pipeline-v2/service.js +477 -0
- package/dist/services/command-observability/service.d.ts +13 -0
- package/dist/services/command-observability/service.js +3 -0
- package/dist/services/dimension-labeling/service.d.ts +77 -0
- package/dist/services/dimension-labeling/service.js +188 -0
- package/dist/services/eval-candidate/service.d.ts +208 -0
- package/dist/services/eval-candidate/service.js +64 -0
- package/dist/services/eval-capabilities/service.d.ts +183 -0
- package/dist/services/eval-capabilities/service.js +1433 -0
- package/dist/services/eval-environment/service.d.ts +173 -0
- package/dist/services/eval-environment/service.js +127 -0
- package/dist/services/evidence-reconstruction/service.d.ts +36 -0
- package/dist/services/evidence-reconstruction/service.js +145 -0
- package/dist/services/fixture-builder/service.d.ts +62 -0
- package/dist/services/fixture-builder/service.js +36 -0
- package/dist/services/fixture-validation/service.d.ts +75 -0
- package/dist/services/fixture-validation/service.js +295 -0
- package/dist/services/foundry/service.d.ts +831 -0
- package/dist/services/foundry/service.js +442 -0
- package/dist/services/foundry-progress/service.d.ts +62 -0
- package/dist/services/foundry-progress/service.js +149 -0
- package/dist/services/foundry-v2/service.d.ts +54 -0
- package/dist/services/foundry-v2/service.js +28 -0
- package/dist/services/grounded-authoring/service.d.ts +126 -0
- package/dist/services/grounded-authoring/service.js +822 -0
- package/dist/services/historical-case/service.d.ts +722 -0
- package/dist/services/historical-case/service.js +177 -0
- package/dist/services/improvement-loop/service.d.ts +59 -0
- package/dist/services/improvement-loop/service.js +176 -0
- package/dist/services/language-model/service.d.ts +52 -0
- package/dist/services/language-model/service.js +194 -0
- package/dist/services/oracle-builder/service.d.ts +146 -0
- package/dist/services/oracle-builder/service.js +513 -0
- package/dist/services/oracle-coverage/service.d.ts +28 -0
- package/dist/services/oracle-coverage/service.js +50 -0
- package/dist/services/oracle-coverage-witness/service.d.ts +130 -0
- package/dist/services/oracle-coverage-witness/service.js +538 -0
- package/dist/services/pipeline-challenge/service.d.ts +551 -0
- package/dist/services/pipeline-challenge/service.js +427 -0
- package/dist/services/pipeline-controls/service.d.ts +130 -0
- package/dist/services/pipeline-controls/service.js +483 -0
- package/dist/services/pipeline-oracle/service.d.ts +8 -0
- package/dist/services/pipeline-oracle/service.js +256 -0
- package/dist/services/pipeline-seed/service.d.ts +298 -0
- package/dist/services/pipeline-seed/service.js +428 -0
- package/dist/services/pipeline-spec/service.d.ts +103 -0
- package/dist/services/pipeline-spec/service.js +619 -0
- package/dist/services/pipeline-tournament/service.d.ts +258 -0
- package/dist/services/pipeline-tournament/service.js +476 -0
- package/dist/services/quality-gate/service.d.ts +233 -0
- package/dist/services/quality-gate/service.js +136 -0
- package/dist/services/repository-bundle/service.d.ts +33 -0
- package/dist/services/repository-bundle/service.js +114 -0
- package/dist/services/repository-model/service.d.ts +105 -0
- package/dist/services/repository-model/service.js +250 -0
- package/dist/services/repository-public-artifact/service.d.ts +133 -0
- package/dist/services/repository-public-artifact/service.js +330 -0
- package/dist/services/routing-benchmark/service.d.ts +362 -0
- package/dist/services/routing-benchmark/service.js +96 -0
- package/dist/services/specification-critic/service.d.ts +92 -0
- package/dist/services/specification-critic/service.js +172 -0
- package/dist/services/task-family/service.d.ts +40 -0
- package/dist/services/task-family/service.js +55 -0
- package/dist/services/task-seed/service.d.ts +906 -0
- package/dist/services/task-seed/service.js +1406 -0
- package/dist/services/task-specification/service.d.ts +27 -0
- package/dist/services/task-specification/service.js +40 -0
- package/dist/services/trajectory-policy/service.d.ts +110 -0
- package/dist/services/trajectory-policy/service.js +216 -0
- package/dist/test/agentic-capabilities-protocol.test.d.ts +1 -0
- package/dist/test/agentic-capabilities-protocol.test.js +570 -0
- package/dist/test/agentic-capabilities.test.d.ts +1 -0
- package/dist/test/agentic-capabilities.test.js +1461 -0
- package/dist/test/agentic-environment.test.d.ts +1 -0
- package/dist/test/agentic-environment.test.js +213 -0
- package/dist/test/case-pipeline-foundation.test.d.ts +1 -0
- package/dist/test/case-pipeline-foundation.test.js +535 -0
- package/dist/test/case-pipeline-protocol-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-protocol-v2.test.js +124 -0
- package/dist/test/case-pipeline-v2.test.d.ts +1 -0
- package/dist/test/case-pipeline-v2.test.js +286 -0
- package/dist/test/case-pipeline.test.d.ts +1 -0
- package/dist/test/case-pipeline.test.js +851 -0
- package/dist/test/eval-capability-policy.test.d.ts +1 -0
- package/dist/test/eval-capability-policy.test.js +50 -0
- package/dist/test/eval-event-log.test.d.ts +1 -0
- package/dist/test/eval-event-log.test.js +125 -0
- package/dist/test/evaluation-evidence-freshness.test.d.ts +1 -0
- package/dist/test/evaluation-evidence-freshness.test.js +44 -0
- package/dist/test/evaluation-evidence.test.d.ts +1 -0
- package/dist/test/evaluation-evidence.test.js +230 -0
- package/dist/test/evaluation-grader-calibration.test.d.ts +1 -0
- package/dist/test/evaluation-grader-calibration.test.js +373 -0
- package/dist/test/evaluation-proposal-digest.test.d.ts +1 -0
- package/dist/test/evaluation-proposal-digest.test.js +187 -0
- package/dist/test/evaluation-source-retrieval.test.d.ts +1 -0
- package/dist/test/evaluation-source-retrieval.test.js +237 -0
- package/dist/test/evaluation-structure-policy.test.d.ts +1 -0
- package/dist/test/evaluation-structure-policy.test.js +196 -0
- package/dist/test/fixtures/repository-resource-panel.d.ts +39 -0
- package/dist/test/fixtures/repository-resource-panel.js +111 -0
- package/dist/test/fixtures/vitest-boundary-panel.d.ts +84 -0
- package/dist/test/fixtures/vitest-boundary-panel.js +120 -0
- package/dist/test/fixtures/vitest-phase-panel.d.ts +135 -0
- package/dist/test/fixtures/vitest-phase-panel.js +213 -0
- package/dist/test/fixtures/vitest-reporter-results.d.ts +76 -0
- package/dist/test/fixtures/vitest-reporter-results.js +94 -0
- package/dist/test/grounded-authoring.test.d.ts +1 -0
- package/dist/test/grounded-authoring.test.js +565 -0
- package/dist/test/integrated-repository-history.test.d.ts +1 -0
- package/dist/test/integrated-repository-history.test.js +227 -0
- package/dist/test/project-authoring.test.js +593 -43
- package/dist/test/project-workflow.test.js +419 -40
- package/dist/test/repository-authoring-artifacts.test.d.ts +1 -0
- package/dist/test/repository-authoring-artifacts.test.js +185 -0
- package/dist/test/repository-bundle.test.d.ts +1 -0
- package/dist/test/repository-bundle.test.js +52 -0
- package/dist/test/repository-case-generation.test.d.ts +1 -0
- package/dist/test/repository-case-generation.test.js +3465 -0
- package/dist/test/repository-command-diagnostic.test.d.ts +1 -0
- package/dist/test/repository-command-diagnostic.test.js +55 -0
- package/dist/test/repository-command-signals.test.d.ts +1 -0
- package/dist/test/repository-command-signals.test.js +124 -0
- package/dist/test/repository-fixture-scope-coverage.test.d.ts +1 -0
- package/dist/test/repository-fixture-scope-coverage.test.js +127 -0
- package/dist/test/repository-fixture-validation.test.d.ts +1 -0
- package/dist/test/repository-fixture-validation.test.js +362 -0
- package/dist/test/repository-foundry-progress.test.d.ts +1 -0
- package/dist/test/repository-foundry-progress.test.js +110 -0
- package/dist/test/repository-foundry-quality.test.d.ts +1 -0
- package/dist/test/repository-foundry-quality.test.js +1138 -0
- package/dist/test/repository-import-context.test.d.ts +1 -0
- package/dist/test/repository-import-context.test.js +354 -0
- package/dist/test/repository-model-authoring.test.d.ts +1 -0
- package/dist/test/repository-model-authoring.test.js +544 -0
- package/dist/test/repository-model.test.d.ts +1 -0
- package/dist/test/repository-model.test.js +2195 -0
- package/dist/test/repository-node-test-reporter.test.d.ts +1 -0
- package/dist/test/repository-node-test-reporter.test.js +104 -0
- package/dist/test/repository-oracle-concurrency.test.d.ts +1 -0
- package/dist/test/repository-oracle-concurrency.test.js +542 -0
- package/dist/test/repository-oracle-coverage-witness.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage-witness.test.js +511 -0
- package/dist/test/repository-oracle-coverage.test.d.ts +1 -0
- package/dist/test/repository-oracle-coverage.test.js +168 -0
- package/dist/test/repository-oracle-evidence.test.d.ts +1 -0
- package/dist/test/repository-oracle-evidence.test.js +185 -0
- package/dist/test/repository-oracle-plan.test.d.ts +1 -0
- package/dist/test/repository-oracle-plan.test.js +176 -0
- package/dist/test/repository-overlay-isolation.test.d.ts +1 -0
- package/dist/test/repository-overlay-isolation.test.js +85 -0
- package/dist/test/repository-preparation-cache.test.d.ts +1 -0
- package/dist/test/repository-preparation-cache.test.js +414 -0
- package/dist/test/repository-public-artifact.test.d.ts +1 -0
- package/dist/test/repository-public-artifact.test.js +273 -0
- package/dist/test/repository-qualification-diagnostics.test.d.ts +1 -0
- package/dist/test/repository-qualification-diagnostics.test.js +524 -0
- package/dist/test/repository-reference-authoring.test.d.ts +1 -0
- package/dist/test/repository-reference-authoring.test.js +1633 -0
- package/dist/test/repository-review-evidence-v2.test.d.ts +1 -0
- package/dist/test/repository-review-evidence-v2.test.js +183 -0
- package/dist/test/repository-review-evidence.test.d.ts +1 -0
- package/dist/test/repository-review-evidence.test.js +124 -0
- package/dist/test/repository-seed-exclusions.test.d.ts +1 -0
- package/dist/test/repository-seed-exclusions.test.js +96 -0
- package/dist/test/repository-seed-selection.test.d.ts +1 -0
- package/dist/test/repository-seed-selection.test.js +504 -0
- package/dist/test/repository-semantic-calibration.test.d.ts +1 -0
- package/dist/test/repository-semantic-calibration.test.js +688 -0
- package/dist/test/repository-solution-edits.test.d.ts +1 -0
- package/dist/test/repository-solution-edits.test.js +377 -0
- package/dist/test/repository-specification-budget.test.d.ts +1 -0
- package/dist/test/repository-specification-budget.test.js +171 -0
- package/dist/test/repository-specification-contract-checkpoint.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-checkpoint.test.js +228 -0
- package/dist/test/repository-specification-contract-facts.test.d.ts +1 -0
- package/dist/test/repository-specification-contract-facts.test.js +177 -0
- package/dist/test/repository-trajectory-authoring.test.d.ts +1 -0
- package/dist/test/repository-trajectory-authoring.test.js +176 -0
- package/dist/test/repository-valid-control-plan.test.d.ts +1 -0
- package/dist/test/repository-valid-control-plan.test.js +45 -0
- package/dist/test/repository-vitest-phase.test.d.ts +1 -0
- package/dist/test/repository-vitest-phase.test.js +848 -0
- package/dist/test/repository-vitest-reporter.test.d.ts +1 -0
- package/dist/test/repository-vitest-reporter.test.js +158 -0
- package/dist/test/repository-workspace-build.test.d.ts +1 -0
- package/dist/test/repository-workspace-build.test.js +160 -0
- package/dist/test/strict-authoring-schema.test.d.ts +1 -0
- package/dist/test/strict-authoring-schema.test.js +169 -0
- package/package.json +48 -6
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import type { AgenticCapabilityNameV1 } from "./agentic-capabilities-protocol.js";
|
|
2
|
+
import { type EvalCapabilityScientificPhaseV1 } from "./eval-capability-policy.js";
|
|
3
|
+
export type EvalCapabilityScientificReserveV1 = Readonly<{
|
|
4
|
+
calls: number;
|
|
5
|
+
wallTimeMs: number;
|
|
6
|
+
attempts: number;
|
|
7
|
+
}>;
|
|
8
|
+
export type EvalCapabilityExecutionBudgetV1 = Readonly<{
|
|
9
|
+
calls: number;
|
|
10
|
+
wallTimeMs: number;
|
|
11
|
+
}>;
|
|
12
|
+
/**
|
|
13
|
+
* Time kept outside scientific execution for checkpoint/archive persistence
|
|
14
|
+
* and the host's durable capability-completion receipt.
|
|
15
|
+
*/
|
|
16
|
+
export declare const EVAL_CAPABILITY_DURABLE_FINALIZATION_RESERVE_MS_V1 = 30000;
|
|
17
|
+
export type EvalCapabilityExecutionEnvelopeDecisionV1 = Readonly<{
|
|
18
|
+
version: 1;
|
|
19
|
+
capability: AgenticCapabilityNameV1;
|
|
20
|
+
phase: EvalCapabilityScientificPhaseV1 | null;
|
|
21
|
+
available: EvalCapabilityExecutionBudgetV1;
|
|
22
|
+
requestedMaximum: EvalCapabilityExecutionBudgetV1;
|
|
23
|
+
downstreamReserve: EvalCapabilityScientificReserveV1;
|
|
24
|
+
finalizationReserve: Readonly<{
|
|
25
|
+
wallTimeMs: number;
|
|
26
|
+
}>;
|
|
27
|
+
minimumRequired: EvalCapabilityExecutionBudgetV1;
|
|
28
|
+
availableForCapability: EvalCapabilityExecutionBudgetV1;
|
|
29
|
+
}> & (Readonly<{
|
|
30
|
+
admitted: true;
|
|
31
|
+
envelope: Readonly<{
|
|
32
|
+
maximumModelCalls: number;
|
|
33
|
+
maximumWallTimeMs: number;
|
|
34
|
+
}>;
|
|
35
|
+
}> | Readonly<{
|
|
36
|
+
admitted: false;
|
|
37
|
+
dimension: "calls" | "wall_time";
|
|
38
|
+
envelope: null;
|
|
39
|
+
}>);
|
|
40
|
+
/**
|
|
41
|
+
* The immutable downstream reserve for a scientific capability. This is the
|
|
42
|
+
* single package-owned source used by admission and execution; campaign-size
|
|
43
|
+
* ratios must not weaken these fixed phase reserves.
|
|
44
|
+
*/
|
|
45
|
+
export declare const evalCapabilityScientificDownstreamReserveV1: (capability: AgenticCapabilityNameV1) => Readonly<{
|
|
46
|
+
phase: EvalCapabilityScientificPhaseV1 | null;
|
|
47
|
+
reserve: EvalCapabilityScientificReserveV1;
|
|
48
|
+
}>;
|
|
49
|
+
/**
|
|
50
|
+
* Computes the positive execution slice that fits after preserving the fixed
|
|
51
|
+
* downstream reserve and durable finalization time. Requested maxima are
|
|
52
|
+
* ceilings, not all-or-nothing reservations: a smaller positive slice is
|
|
53
|
+
* admitted whenever one fits.
|
|
54
|
+
*
|
|
55
|
+
* The structured decision retains every value needed to explain or persist the
|
|
56
|
+
* decision without reconstructing it from prose.
|
|
57
|
+
*/
|
|
58
|
+
export declare const evalCapabilityExecutionEnvelopeV1: (input: {
|
|
59
|
+
readonly capability: AgenticCapabilityNameV1;
|
|
60
|
+
readonly requestedMaximumModelCalls?: number;
|
|
61
|
+
readonly requestedMaximumWallMs?: number;
|
|
62
|
+
readonly campaignModelCallsRemaining: number;
|
|
63
|
+
readonly campaignWallTimeRemainingMs: number;
|
|
64
|
+
}) => EvalCapabilityExecutionEnvelopeDecisionV1;
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
import { EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1, evalCapabilityScientificMaximumModelCallsV1, evalCapabilityScientificMaximumWallTimeMsV1, evalCapabilityScientificPhaseV1 } from "./eval-capability-policy.js";
|
|
2
|
+
/**
|
|
3
|
+
* Time kept outside scientific execution for checkpoint/archive persistence
|
|
4
|
+
* and the host's durable capability-completion receipt.
|
|
5
|
+
*/
|
|
6
|
+
export const EVAL_CAPABILITY_DURABLE_FINALIZATION_RESERVE_MS_V1 = 30_000;
|
|
7
|
+
const positiveInteger = (value, fallback) => Number.isSafeInteger(value) && value > 0 ? value : fallback;
|
|
8
|
+
const availableInteger = (value) => Number.isSafeInteger(value) && value > 0 ? value : 0;
|
|
9
|
+
/**
|
|
10
|
+
* The immutable downstream reserve for a scientific capability. This is the
|
|
11
|
+
* single package-owned source used by admission and execution; campaign-size
|
|
12
|
+
* ratios must not weaken these fixed phase reserves.
|
|
13
|
+
*/
|
|
14
|
+
export const evalCapabilityScientificDownstreamReserveV1 = (capability) => {
|
|
15
|
+
const phase = evalCapabilityScientificPhaseV1(capability);
|
|
16
|
+
const reserve = phase === null
|
|
17
|
+
? { calls: 0, wallTimeMs: 0, attempts: 0 }
|
|
18
|
+
: EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1[phase].downstream;
|
|
19
|
+
return { phase, reserve };
|
|
20
|
+
};
|
|
21
|
+
/**
|
|
22
|
+
* Computes the positive execution slice that fits after preserving the fixed
|
|
23
|
+
* downstream reserve and durable finalization time. Requested maxima are
|
|
24
|
+
* ceilings, not all-or-nothing reservations: a smaller positive slice is
|
|
25
|
+
* admitted whenever one fits.
|
|
26
|
+
*
|
|
27
|
+
* The structured decision retains every value needed to explain or persist the
|
|
28
|
+
* decision without reconstructing it from prose.
|
|
29
|
+
*/
|
|
30
|
+
export const evalCapabilityExecutionEnvelopeV1 = (input) => {
|
|
31
|
+
const { phase, reserve } = evalCapabilityScientificDownstreamReserveV1(input.capability);
|
|
32
|
+
const scientific = phase !== null;
|
|
33
|
+
const finalizationReserveWallTimeMs = scientific
|
|
34
|
+
? EVAL_CAPABILITY_DURABLE_FINALIZATION_RESERVE_MS_V1
|
|
35
|
+
: 0;
|
|
36
|
+
const available = {
|
|
37
|
+
calls: availableInteger(input.campaignModelCallsRemaining),
|
|
38
|
+
wallTimeMs: availableInteger(input.campaignWallTimeRemainingMs)
|
|
39
|
+
};
|
|
40
|
+
const requestedMaximum = {
|
|
41
|
+
calls: Math.min(scientific
|
|
42
|
+
? evalCapabilityScientificMaximumModelCallsV1(input.capability)
|
|
43
|
+
: 12, positiveInteger(input.requestedMaximumModelCalls ?? 6, 6)),
|
|
44
|
+
wallTimeMs: Math.min(scientific
|
|
45
|
+
? evalCapabilityScientificMaximumWallTimeMsV1(input.capability)
|
|
46
|
+
: 300_000, positiveInteger(input.requestedMaximumWallMs ?? 120_000, 120_000))
|
|
47
|
+
};
|
|
48
|
+
const minimumRequired = {
|
|
49
|
+
calls: reserve.calls + 1,
|
|
50
|
+
wallTimeMs: reserve.wallTimeMs + finalizationReserveWallTimeMs + 1
|
|
51
|
+
};
|
|
52
|
+
const availableForCapability = {
|
|
53
|
+
calls: Math.max(0, available.calls - reserve.calls),
|
|
54
|
+
wallTimeMs: Math.max(0, available.wallTimeMs -
|
|
55
|
+
reserve.wallTimeMs -
|
|
56
|
+
finalizationReserveWallTimeMs)
|
|
57
|
+
};
|
|
58
|
+
const common = {
|
|
59
|
+
version: 1,
|
|
60
|
+
capability: input.capability,
|
|
61
|
+
phase,
|
|
62
|
+
available,
|
|
63
|
+
requestedMaximum,
|
|
64
|
+
downstreamReserve: reserve,
|
|
65
|
+
finalizationReserve: {
|
|
66
|
+
wallTimeMs: finalizationReserveWallTimeMs
|
|
67
|
+
},
|
|
68
|
+
minimumRequired,
|
|
69
|
+
availableForCapability
|
|
70
|
+
};
|
|
71
|
+
// Read-only/control capabilities remain reachable after scientific exhaustion
|
|
72
|
+
// so callers can inspect progress and finish the campaign. They receive the
|
|
73
|
+
// legacy minimal positive execution fence and no downstream reserve.
|
|
74
|
+
if (!scientific) {
|
|
75
|
+
return {
|
|
76
|
+
...common,
|
|
77
|
+
admitted: true,
|
|
78
|
+
envelope: {
|
|
79
|
+
maximumModelCalls: Math.max(1, Math.min(requestedMaximum.calls, available.calls)),
|
|
80
|
+
maximumWallTimeMs: Math.max(1, Math.min(requestedMaximum.wallTimeMs, available.wallTimeMs))
|
|
81
|
+
}
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
if (availableForCapability.calls < 1) {
|
|
85
|
+
return { ...common, admitted: false, dimension: "calls", envelope: null };
|
|
86
|
+
}
|
|
87
|
+
if (availableForCapability.wallTimeMs < 1) {
|
|
88
|
+
return { ...common, admitted: false, dimension: "wall_time", envelope: null };
|
|
89
|
+
}
|
|
90
|
+
return {
|
|
91
|
+
...common,
|
|
92
|
+
admitted: true,
|
|
93
|
+
envelope: {
|
|
94
|
+
maximumModelCalls: Math.min(requestedMaximum.calls, availableForCapability.calls),
|
|
95
|
+
maximumWallTimeMs: Math.min(requestedMaximum.wallTimeMs, availableForCapability.wallTimeMs)
|
|
96
|
+
}
|
|
97
|
+
};
|
|
98
|
+
};
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
import type { AgenticCapabilityNameV1, AgenticProgressFingerprintV1, AgenticScientificProgressReceiptV1 } from "./agentic-capabilities-protocol.js";
|
|
2
|
+
/**
|
|
3
|
+
* A scheduling slice that only advances verified durable work is not another
|
|
4
|
+
* scientific attempt. Charging it makes sufficiently slow but productive
|
|
5
|
+
* cases impossible, regardless of their remaining campaign budget.
|
|
6
|
+
* New decisions, resets, legacy/unbound receipts, and unchanged work still
|
|
7
|
+
* consume attempts. Calls, spend, wall time and the no-progress circuit are
|
|
8
|
+
* independent and are never refunded.
|
|
9
|
+
*/
|
|
10
|
+
export declare const evalCapabilityScientificAttemptCostV1: (receipt: Pick<AgenticScientificProgressReceiptV1, "previousFingerprint" | "resultingFingerprint" | "inputRevision" | "outputRevision" | "gatesCompleted" | "gatesInvalidated" | "sameFingerprintCount">) => 0 | 1;
|
|
11
|
+
/**
|
|
12
|
+
* No-op calls to other tools cannot reset a capability's stall counter. A
|
|
13
|
+
* genuine accepted-state change does reset it, even if a later repair returns
|
|
14
|
+
* to the same state. Worker admission and Cloud receipt validation share this
|
|
15
|
+
* rule; the global previous-fingerprint chain is verified independently.
|
|
16
|
+
* History must be ordered oldest first and scoped to one candidate revision.
|
|
17
|
+
*/
|
|
18
|
+
export declare const evalCapabilityScientificProgressPredecessorV1: (input: {
|
|
19
|
+
readonly capability: AgenticCapabilityNameV1;
|
|
20
|
+
readonly fingerprint: AgenticProgressFingerprintV1;
|
|
21
|
+
readonly history: readonly AgenticScientificProgressReceiptV1[];
|
|
22
|
+
}) => AgenticScientificProgressReceiptV1 | undefined;
|
|
23
|
+
/**
|
|
24
|
+
* One scientific capability is allowed enough time to finish a complete
|
|
25
|
+
* repository-grounded role chain (analysis, visible authoring, and private
|
|
26
|
+
* review) under normal provider tail latency. Cloud admission, the sandbox
|
|
27
|
+
* executor, and the native tool import this dependency-light policy module so
|
|
28
|
+
* loading an OpenCode tool never initializes the server-side foundry runtime.
|
|
29
|
+
*/
|
|
30
|
+
export declare const EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_MODEL_CALLS_V1 = 12;
|
|
31
|
+
export declare const EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_WALL_TIME_MS_V1: number;
|
|
32
|
+
/**
|
|
33
|
+
* Versioned capability slices are derived from the maximum number of native
|
|
34
|
+
* foundry turns that must remain contiguous to preserve tool state. Most
|
|
35
|
+
* capabilities fit one base slice. Oracle authoring and repair own longer
|
|
36
|
+
* tool-using conversations and therefore receive multiple contiguous slices
|
|
37
|
+
* instead of repeatedly restarting after the base ceiling.
|
|
38
|
+
*/
|
|
39
|
+
export declare const EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1: {
|
|
40
|
+
readonly author_oracle: {
|
|
41
|
+
readonly modelCallSlices: 3;
|
|
42
|
+
readonly wallTimeSlices: 3;
|
|
43
|
+
};
|
|
44
|
+
readonly repair_oracle: {
|
|
45
|
+
readonly modelCallSlices: 2;
|
|
46
|
+
readonly wallTimeSlices: 2;
|
|
47
|
+
};
|
|
48
|
+
};
|
|
49
|
+
export declare const evalCapabilityScientificMaximumModelCallsV1: (capability: AgenticCapabilityNameV1) => number;
|
|
50
|
+
export declare const evalCapabilityScientificMaximumWallTimeMsV1: (capability: AgenticCapabilityNameV1) => number;
|
|
51
|
+
export declare const EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1: {
|
|
52
|
+
readonly specification: {
|
|
53
|
+
readonly capabilities: readonly ["draft_specification"];
|
|
54
|
+
readonly downstream: {
|
|
55
|
+
readonly calls: 24;
|
|
56
|
+
readonly wallTimeMs: 4500000;
|
|
57
|
+
readonly attempts: 6;
|
|
58
|
+
};
|
|
59
|
+
readonly maximumAttempts: 8;
|
|
60
|
+
};
|
|
61
|
+
readonly environment: {
|
|
62
|
+
readonly capabilities: readonly ["plan_environment", "probe_fixture", "qualify_reference"];
|
|
63
|
+
readonly downstream: {
|
|
64
|
+
readonly calls: 16;
|
|
65
|
+
readonly wallTimeMs: 3600000;
|
|
66
|
+
readonly attempts: 4;
|
|
67
|
+
};
|
|
68
|
+
readonly maximumAttempts: 8;
|
|
69
|
+
};
|
|
70
|
+
readonly oracle: {
|
|
71
|
+
readonly capabilities: readonly ["author_oracle", "generate_controls", "evaluate_oracle", "repair_oracle"];
|
|
72
|
+
readonly downstream: {
|
|
73
|
+
readonly calls: 8;
|
|
74
|
+
readonly wallTimeMs: 1800000;
|
|
75
|
+
readonly attempts: 2;
|
|
76
|
+
};
|
|
77
|
+
readonly maximumAttempts: 12;
|
|
78
|
+
};
|
|
79
|
+
readonly admission: {
|
|
80
|
+
readonly capabilities: readonly ["freeze_candidate", "run_held_out_challenge", "request_admission"];
|
|
81
|
+
readonly downstream: {
|
|
82
|
+
readonly calls: 0;
|
|
83
|
+
readonly wallTimeMs: 0;
|
|
84
|
+
readonly attempts: 0;
|
|
85
|
+
};
|
|
86
|
+
readonly maximumAttempts: 6;
|
|
87
|
+
};
|
|
88
|
+
};
|
|
89
|
+
export type EvalCapabilityScientificPhaseV1 = keyof typeof EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1;
|
|
90
|
+
export declare const evalCapabilityScientificPhaseV1: (capability: AgenticCapabilityNameV1) => EvalCapabilityScientificPhaseV1 | null;
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A scheduling slice that only advances verified durable work is not another
|
|
3
|
+
* scientific attempt. Charging it makes sufficiently slow but productive
|
|
4
|
+
* cases impossible, regardless of their remaining campaign budget.
|
|
5
|
+
* New decisions, resets, legacy/unbound receipts, and unchanged work still
|
|
6
|
+
* consume attempts. Calls, spend, wall time and the no-progress circuit are
|
|
7
|
+
* independent and are never refunded.
|
|
8
|
+
*/
|
|
9
|
+
export const evalCapabilityScientificAttemptCostV1 = (receipt) => {
|
|
10
|
+
const previous = receipt.previousFingerprint;
|
|
11
|
+
const next = receipt.resultingFingerprint;
|
|
12
|
+
return previous !== null &&
|
|
13
|
+
receipt.inputRevision !== null &&
|
|
14
|
+
receipt.inputRevision === receipt.outputRevision &&
|
|
15
|
+
receipt.sameFingerprintCount === 0 &&
|
|
16
|
+
receipt.gatesCompleted.length === 0 &&
|
|
17
|
+
receipt.gatesInvalidated.length === 0 &&
|
|
18
|
+
next.continuationProgressDigest != null &&
|
|
19
|
+
next.continuationProgressDigest !== (previous.continuationProgressDigest ?? null) &&
|
|
20
|
+
next.completedGateDigest === previous.completedGateDigest &&
|
|
21
|
+
next.environmentDigest === previous.environmentDigest
|
|
22
|
+
? 0
|
|
23
|
+
: 1;
|
|
24
|
+
};
|
|
25
|
+
/**
|
|
26
|
+
* No-op calls to other tools cannot reset a capability's stall counter. A
|
|
27
|
+
* genuine accepted-state change does reset it, even if a later repair returns
|
|
28
|
+
* to the same state. Worker admission and Cloud receipt validation share this
|
|
29
|
+
* rule; the global previous-fingerprint chain is verified independently.
|
|
30
|
+
* History must be ordered oldest first and scoped to one candidate revision.
|
|
31
|
+
*/
|
|
32
|
+
export const evalCapabilityScientificProgressPredecessorV1 = (input) => {
|
|
33
|
+
for (let index = input.history.length - 1; index >= 0; index -= 1) {
|
|
34
|
+
const receipt = input.history[index];
|
|
35
|
+
const fingerprint = receipt.resultingFingerprint;
|
|
36
|
+
if (fingerprint.acceptedEvidenceDigest !== input.fingerprint.acceptedEvidenceDigest ||
|
|
37
|
+
fingerprint.completedGateDigest !== input.fingerprint.completedGateDigest ||
|
|
38
|
+
fingerprint.environmentDigest !== input.fingerprint.environmentDigest ||
|
|
39
|
+
(fingerprint.continuationProgressDigest ?? null) !==
|
|
40
|
+
(input.fingerprint.continuationProgressDigest ?? null))
|
|
41
|
+
return undefined;
|
|
42
|
+
if (receipt.capability === input.capability)
|
|
43
|
+
return receipt;
|
|
44
|
+
}
|
|
45
|
+
return undefined;
|
|
46
|
+
};
|
|
47
|
+
/**
|
|
48
|
+
* One scientific capability is allowed enough time to finish a complete
|
|
49
|
+
* repository-grounded role chain (analysis, visible authoring, and private
|
|
50
|
+
* review) under normal provider tail latency. Cloud admission, the sandbox
|
|
51
|
+
* executor, and the native tool import this dependency-light policy module so
|
|
52
|
+
* loading an OpenCode tool never initializes the server-side foundry runtime.
|
|
53
|
+
*/
|
|
54
|
+
export const EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_MODEL_CALLS_V1 = 12;
|
|
55
|
+
export const EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_WALL_TIME_MS_V1 = 10 * 60_000;
|
|
56
|
+
/**
|
|
57
|
+
* Versioned capability slices are derived from the maximum number of native
|
|
58
|
+
* foundry turns that must remain contiguous to preserve tool state. Most
|
|
59
|
+
* capabilities fit one base slice. Oracle authoring and repair own longer
|
|
60
|
+
* tool-using conversations and therefore receive multiple contiguous slices
|
|
61
|
+
* instead of repeatedly restarting after the base ceiling.
|
|
62
|
+
*/
|
|
63
|
+
export const EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1 = {
|
|
64
|
+
author_oracle: { modelCallSlices: 3, wallTimeSlices: 3 },
|
|
65
|
+
repair_oracle: { modelCallSlices: 2, wallTimeSlices: 2 }
|
|
66
|
+
};
|
|
67
|
+
export const evalCapabilityScientificMaximumModelCallsV1 = (capability) => EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_MODEL_CALLS_V1 *
|
|
68
|
+
(capability in EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1
|
|
69
|
+
? EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1[capability].modelCallSlices
|
|
70
|
+
: 1);
|
|
71
|
+
export const evalCapabilityScientificMaximumWallTimeMsV1 = (capability) => EVAL_CAPABILITY_SCIENTIFIC_MAXIMUM_WALL_TIME_MS_V1 *
|
|
72
|
+
(capability in EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1
|
|
73
|
+
? EVAL_CAPABILITY_SCIENTIFIC_SLICE_PLAN_V1[capability].wallTimeSlices
|
|
74
|
+
: 1);
|
|
75
|
+
export const EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1 = {
|
|
76
|
+
specification: {
|
|
77
|
+
capabilities: ["draft_specification"],
|
|
78
|
+
// This is the minimum protected envelope for environment qualification,
|
|
79
|
+
// oracle/control construction, held-out challenge, and admission. It is not
|
|
80
|
+
// a per-turn timeout. A smaller campaign can start useful work but cannot
|
|
81
|
+
// complete the mandatory scientific path.
|
|
82
|
+
downstream: { calls: 24, wallTimeMs: 4_500_000, attempts: 6 },
|
|
83
|
+
maximumAttempts: 8
|
|
84
|
+
},
|
|
85
|
+
environment: {
|
|
86
|
+
capabilities: ["plan_environment", "probe_fixture", "qualify_reference"],
|
|
87
|
+
downstream: { calls: 16, wallTimeMs: 3_600_000, attempts: 4 },
|
|
88
|
+
maximumAttempts: 8
|
|
89
|
+
},
|
|
90
|
+
oracle: {
|
|
91
|
+
capabilities: ["author_oracle", "generate_controls", "evaluate_oracle", "repair_oracle"],
|
|
92
|
+
downstream: { calls: 8, wallTimeMs: 1_800_000, attempts: 2 },
|
|
93
|
+
maximumAttempts: 12
|
|
94
|
+
},
|
|
95
|
+
admission: {
|
|
96
|
+
capabilities: ["freeze_candidate", "run_held_out_challenge", "request_admission"],
|
|
97
|
+
downstream: { calls: 0, wallTimeMs: 0, attempts: 0 },
|
|
98
|
+
maximumAttempts: 6
|
|
99
|
+
}
|
|
100
|
+
};
|
|
101
|
+
export const evalCapabilityScientificPhaseV1 = (capability) => {
|
|
102
|
+
for (const [phase, policy] of Object.entries(EVAL_CAPABILITY_SCIENTIFIC_PHASE_POLICY_V1)) {
|
|
103
|
+
if (policy.capabilities.includes(capability))
|
|
104
|
+
return phase;
|
|
105
|
+
}
|
|
106
|
+
return null;
|
|
107
|
+
};
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
import { Schema } from "effect";
|
|
2
|
+
export declare const EvalRunFailurePhase: Schema.Literals<readonly ["target", "classifier", "suite-inspection", "comparison", "activation", "accounting", "completion"]>;
|
|
3
|
+
export type EvalRunFailurePhase = typeof EvalRunFailurePhase.Type;
|
|
4
|
+
export declare const EvalNamedEvent: Schema.Union<readonly [Schema.Struct<{
|
|
5
|
+
readonly name: Schema.Literal<"plan">;
|
|
6
|
+
readonly at: Schema.String;
|
|
7
|
+
readonly runId: Schema.String;
|
|
8
|
+
readonly plannedCalls: Schema.Finite;
|
|
9
|
+
}>, Schema.Struct<{
|
|
10
|
+
readonly at: Schema.String;
|
|
11
|
+
readonly runId: Schema.String;
|
|
12
|
+
readonly dimensionId: Schema.String;
|
|
13
|
+
readonly caseId: Schema.String;
|
|
14
|
+
readonly sequence: Schema.Finite;
|
|
15
|
+
readonly name: Schema.Literal<"candidate#">;
|
|
16
|
+
}>, Schema.Struct<{
|
|
17
|
+
readonly at: Schema.String;
|
|
18
|
+
readonly runId: Schema.String;
|
|
19
|
+
readonly dimensionId: Schema.String;
|
|
20
|
+
readonly caseId: Schema.String;
|
|
21
|
+
readonly sequence: Schema.Finite;
|
|
22
|
+
readonly name: Schema.Literal<"judge#">;
|
|
23
|
+
}>, Schema.Struct<{
|
|
24
|
+
readonly name: Schema.Literal<"failure-detail">;
|
|
25
|
+
readonly at: Schema.String;
|
|
26
|
+
readonly runId: Schema.String;
|
|
27
|
+
readonly phase: Schema.Literals<readonly ["target", "classifier", "suite-inspection", "comparison", "activation", "accounting", "completion"]>;
|
|
28
|
+
readonly reason: Schema.String;
|
|
29
|
+
readonly retryable: Schema.Boolean;
|
|
30
|
+
readonly callId: Schema.optionalKey<Schema.String>;
|
|
31
|
+
readonly reservationId: Schema.optionalKey<Schema.String>;
|
|
32
|
+
readonly accounting: Schema.Struct<{
|
|
33
|
+
readonly expectedCalls: Schema.Finite;
|
|
34
|
+
readonly observedCalls: Schema.Finite;
|
|
35
|
+
readonly knownInputTokens: Schema.Finite;
|
|
36
|
+
readonly knownOutputTokens: Schema.Finite;
|
|
37
|
+
readonly unknownTokenMeasurements: Schema.Finite;
|
|
38
|
+
readonly knownPricedSubtotalUsd: Schema.Finite;
|
|
39
|
+
readonly unpricedCalls: Schema.Finite;
|
|
40
|
+
readonly subscriptionUnitCalls: Schema.Finite;
|
|
41
|
+
}>;
|
|
42
|
+
readonly cleanup: Schema.Struct<{
|
|
43
|
+
readonly sessionOpened: Schema.Boolean;
|
|
44
|
+
readonly sessionClosed: Schema.Boolean;
|
|
45
|
+
readonly detail: Schema.optionalKey<Schema.String>;
|
|
46
|
+
}>;
|
|
47
|
+
}>, Schema.Struct<{
|
|
48
|
+
readonly at: Schema.String;
|
|
49
|
+
readonly operationId: Schema.String;
|
|
50
|
+
readonly name: Schema.Literal<"basis">;
|
|
51
|
+
}>, Schema.Struct<{
|
|
52
|
+
readonly at: Schema.String;
|
|
53
|
+
readonly operationId: Schema.String;
|
|
54
|
+
readonly name: Schema.Literal<"dimensions">;
|
|
55
|
+
}>, Schema.Struct<{
|
|
56
|
+
readonly at: Schema.String;
|
|
57
|
+
readonly operationId: Schema.String;
|
|
58
|
+
readonly name: Schema.Literal<"activation">;
|
|
59
|
+
}>]>;
|
|
60
|
+
export type EvalNamedEvent = typeof EvalNamedEvent.Type;
|
|
61
|
+
export declare const EvalNamedEventLog: Schema.$Array<Schema.Union<readonly [Schema.Struct<{
|
|
62
|
+
readonly name: Schema.Literal<"plan">;
|
|
63
|
+
readonly at: Schema.String;
|
|
64
|
+
readonly runId: Schema.String;
|
|
65
|
+
readonly plannedCalls: Schema.Finite;
|
|
66
|
+
}>, Schema.Struct<{
|
|
67
|
+
readonly at: Schema.String;
|
|
68
|
+
readonly runId: Schema.String;
|
|
69
|
+
readonly dimensionId: Schema.String;
|
|
70
|
+
readonly caseId: Schema.String;
|
|
71
|
+
readonly sequence: Schema.Finite;
|
|
72
|
+
readonly name: Schema.Literal<"candidate#">;
|
|
73
|
+
}>, Schema.Struct<{
|
|
74
|
+
readonly at: Schema.String;
|
|
75
|
+
readonly runId: Schema.String;
|
|
76
|
+
readonly dimensionId: Schema.String;
|
|
77
|
+
readonly caseId: Schema.String;
|
|
78
|
+
readonly sequence: Schema.Finite;
|
|
79
|
+
readonly name: Schema.Literal<"judge#">;
|
|
80
|
+
}>, Schema.Struct<{
|
|
81
|
+
readonly name: Schema.Literal<"failure-detail">;
|
|
82
|
+
readonly at: Schema.String;
|
|
83
|
+
readonly runId: Schema.String;
|
|
84
|
+
readonly phase: Schema.Literals<readonly ["target", "classifier", "suite-inspection", "comparison", "activation", "accounting", "completion"]>;
|
|
85
|
+
readonly reason: Schema.String;
|
|
86
|
+
readonly retryable: Schema.Boolean;
|
|
87
|
+
readonly callId: Schema.optionalKey<Schema.String>;
|
|
88
|
+
readonly reservationId: Schema.optionalKey<Schema.String>;
|
|
89
|
+
readonly accounting: Schema.Struct<{
|
|
90
|
+
readonly expectedCalls: Schema.Finite;
|
|
91
|
+
readonly observedCalls: Schema.Finite;
|
|
92
|
+
readonly knownInputTokens: Schema.Finite;
|
|
93
|
+
readonly knownOutputTokens: Schema.Finite;
|
|
94
|
+
readonly unknownTokenMeasurements: Schema.Finite;
|
|
95
|
+
readonly knownPricedSubtotalUsd: Schema.Finite;
|
|
96
|
+
readonly unpricedCalls: Schema.Finite;
|
|
97
|
+
readonly subscriptionUnitCalls: Schema.Finite;
|
|
98
|
+
}>;
|
|
99
|
+
readonly cleanup: Schema.Struct<{
|
|
100
|
+
readonly sessionOpened: Schema.Boolean;
|
|
101
|
+
readonly sessionClosed: Schema.Boolean;
|
|
102
|
+
readonly detail: Schema.optionalKey<Schema.String>;
|
|
103
|
+
}>;
|
|
104
|
+
}>, Schema.Struct<{
|
|
105
|
+
readonly at: Schema.String;
|
|
106
|
+
readonly operationId: Schema.String;
|
|
107
|
+
readonly name: Schema.Literal<"basis">;
|
|
108
|
+
}>, Schema.Struct<{
|
|
109
|
+
readonly at: Schema.String;
|
|
110
|
+
readonly operationId: Schema.String;
|
|
111
|
+
readonly name: Schema.Literal<"dimensions">;
|
|
112
|
+
}>, Schema.Struct<{
|
|
113
|
+
readonly at: Schema.String;
|
|
114
|
+
readonly operationId: Schema.String;
|
|
115
|
+
readonly name: Schema.Literal<"activation">;
|
|
116
|
+
}>]>>;
|
|
117
|
+
export type EvalNamedEventLog = typeof EvalNamedEventLog.Type;
|
|
118
|
+
export type EvalRunCallEvent = Extract<EvalNamedEvent, {
|
|
119
|
+
readonly name: "candidate#" | "judge#";
|
|
120
|
+
}>;
|
|
121
|
+
export type EvalCaseCursor = {
|
|
122
|
+
readonly dimensionId: string;
|
|
123
|
+
readonly caseId: string;
|
|
124
|
+
readonly completePairs: number;
|
|
125
|
+
readonly lastEvent?: EvalRunCallEvent;
|
|
126
|
+
readonly hole?: Extract<EvalRunCallEvent, {
|
|
127
|
+
readonly name: "candidate#";
|
|
128
|
+
}>;
|
|
129
|
+
};
|
|
130
|
+
export declare function resumeEvalCaseCursors(log: EvalNamedEventLog): readonly EvalCaseCursor[];
|
|
131
|
+
export declare function evalEventLabel(event: EvalNamedEvent): string;
|
|
132
|
+
export type EvalEventLogPresentation = {
|
|
133
|
+
readonly failed: boolean;
|
|
134
|
+
readonly lines: readonly string[];
|
|
135
|
+
};
|
|
136
|
+
export declare function presentEvalEventLog(log: EvalNamedEventLog, options?: {
|
|
137
|
+
readonly failHoles?: boolean;
|
|
138
|
+
}): EvalEventLogPresentation;
|
|
139
|
+
export declare function assertEvalRunCanFail(log: EvalNamedEventLog): void;
|
|
140
|
+
export declare function assertEvalRunCanComplete(log: EvalNamedEventLog): void;
|