@tangle-network/agent-eval 0.86.0 → 0.89.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/http.d.ts +3 -3
- package/dist/adapters/langchain.d.ts +3 -3
- package/dist/adapters/otel.d.ts +6 -6
- package/dist/adversarial-DIVcDoI_.d.ts +88 -0
- package/dist/analyst/index.d.ts +11 -10
- package/dist/analyst/index.js +13 -8
- package/dist/analyst/index.js.map +1 -1
- package/dist/analyze-runs-DwCEkpO_.d.ts +81 -0
- package/dist/belief-state/index.d.ts +4 -4
- package/dist/belief-state/index.js +1 -1
- package/dist/benchmarks/index.d.ts +3 -3
- package/dist/campaign/index.d.ts +165 -18
- package/dist/campaign/index.js +289 -14
- package/dist/campaign/index.js.map +1 -1
- package/dist/chunk-45EEMHTC.js +35 -0
- package/dist/chunk-45EEMHTC.js.map +1 -0
- package/dist/{chunk-FZWAFVAA.js → chunk-4FBZZIYD.js} +2 -2
- package/dist/{chunk-YV7J7X5N.js → chunk-5HRORJQY.js} +22 -12
- package/dist/chunk-5HRORJQY.js.map +1 -0
- package/dist/{chunk-OTYQPHPL.js → chunk-6SOJM3VR.js} +5 -5
- package/dist/chunk-BOD4O7OF.js +40 -0
- package/dist/chunk-BOD4O7OF.js.map +1 -0
- package/dist/{chunk-Z7VFTS2J.js → chunk-CY6U5S3X.js} +2 -2
- package/dist/{chunk-VIDQF3F5.js → chunk-D3V5B42D.js} +5 -34
- package/dist/chunk-D3V5B42D.js.map +1 -0
- package/dist/{chunk-YGYXHNAQ.js → chunk-FIUKOSWI.js} +21 -8
- package/dist/chunk-FIUKOSWI.js.map +1 -0
- package/dist/{chunk-WJL2NJXN.js → chunk-GSH6QNNS.js} +2 -2
- package/dist/{chunk-RBNA5AZT.js → chunk-L3JOU6XM.js} +2 -2
- package/dist/{chunk-IDVBLYCY.js → chunk-LMZQ2Z4U.js} +56 -2
- package/dist/{chunk-IDVBLYCY.js.map → chunk-LMZQ2Z4U.js.map} +1 -1
- package/dist/{chunk-VUINJM5M.js → chunk-QAY5UIJO.js} +2 -193
- package/dist/chunk-QAY5UIJO.js.map +1 -0
- package/dist/{chunk-P2J6SOXT.js → chunk-QG2OVF2D.js} +5 -3
- package/dist/{chunk-P2J6SOXT.js.map → chunk-QG2OVF2D.js.map} +1 -1
- package/dist/chunk-REVYNR6C.js +100 -0
- package/dist/chunk-REVYNR6C.js.map +1 -0
- package/dist/{chunk-ZZ2HOPME.js → chunk-TWS7AZEY.js} +2 -2
- package/dist/chunk-UHMJT4T7.js +200 -0
- package/dist/chunk-UHMJT4T7.js.map +1 -0
- package/dist/chunk-UMMZHCPB.js +190 -0
- package/dist/chunk-UMMZHCPB.js.map +1 -0
- package/dist/chunk-VZSRQ272.js +149 -0
- package/dist/chunk-VZSRQ272.js.map +1 -0
- package/dist/{chunk-L5G7OUKD.js → chunk-XY4DDNEG.js} +8 -190
- package/dist/chunk-XY4DDNEG.js.map +1 -0
- package/dist/chunk-Y47J2LJ3.js +859 -0
- package/dist/chunk-Y47J2LJ3.js.map +1 -0
- package/dist/{chunk-BABOZOSN.js → chunk-ZFIBGEOL.js} +3 -3
- package/dist/chunk-ZFIBGEOL.js.map +1 -0
- package/dist/{code-agent-session-BRXmavYv.d.ts → code-agent-session-BO8nCnv3.d.ts} +1 -1
- package/dist/contract/index.d.ts +24 -95
- package/dist/contract/index.js +16 -755
- package/dist/contract/index.js.map +1 -1
- package/dist/{control-GeE8OhpN.d.ts → control-_Qb7skHX.d.ts} +2 -2
- package/dist/control.d.ts +5 -5
- package/dist/corpus-BoR-041R.d.ts +560 -0
- package/dist/cost-ledger-DuSqlw5B.d.ts +113 -0
- package/dist/counterfactual-Dwibr5IW.d.ts +85 -0
- package/dist/{dataset-B2kL-fSM.d.ts → dataset-BbGkaN2I.d.ts} +1 -1
- package/dist/{registry-DrEQ3Luj.d.ts → default-registry-zoGHUQEH.d.ts} +29 -2
- package/dist/diagnose.d.ts +251 -0
- package/dist/diagnose.js +381 -0
- package/dist/diagnose.js.map +1 -0
- package/dist/{errors-Dwqw-T_m.d.ts → errors-CzMUYo7b.d.ts} +1 -1
- package/dist/{feedback-trajectory-B3rErRsh.d.ts → feedback-trajectory-D9OVLrg9.d.ts} +1 -1
- package/dist/fuzz.d.ts +484 -0
- package/dist/fuzz.js +613 -0
- package/dist/fuzz.js.map +1 -0
- package/dist/governance/index.d.ts +4 -4
- package/dist/hosted/index.d.ts +6 -6
- package/dist/{index-DE3RXAXD.d.ts → index-Bx3gZ8xl.d.ts} +1 -1
- package/dist/index.d.ts +717 -455
- package/dist/index.js +1590 -793
- package/dist/index.js.map +1 -1
- package/dist/{insight-report-3ADTfClO.d.ts → insight-report-BBwvOh6x.d.ts} +2 -2
- package/dist/{integrity-CJzrpUua.d.ts → integrity-VJ9A7aST.d.ts} +1 -1
- package/dist/{judge-calibration-DilmB3Ml.d.ts → judge-calibration-0p2QcWNE.d.ts} +1 -1
- package/dist/{kind-factory-CVecZZG_.d.ts → kind-factory-5b7xXXOr.d.ts} +2 -2
- package/dist/{llm-client-CuUg2Mn3.d.ts → llm-client-BeEcAokY.d.ts} +1 -1
- package/dist/matrix/index.d.ts +2 -2
- package/dist/meta-eval/index.d.ts +177 -3
- package/dist/meta-eval/index.js +260 -1
- package/dist/meta-eval/index.js.map +1 -1
- package/dist/{multi-layer-verifier-DlWCXuxL.d.ts → multi-layer-verifier-DUZXrPDA.d.ts} +7 -1
- package/dist/multishot/index.d.ts +25 -11
- package/dist/multishot/index.js +36 -7
- package/dist/multishot/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/pipelines/index.js +2 -2
- package/dist/{agent-profile-D0PBIWlV.d.ts → pre-registration-DELOEJ8v.d.ts} +144 -4
- package/dist/{provenance-DPpNIOJD.d.ts → provenance-LnqRT0sS.d.ts} +5 -5
- package/dist/{red-team-DW9Ca_tj.d.ts → red-team-BXHil6c8.d.ts} +1 -1
- package/dist/{release-report-hlNtD12q.d.ts → release-report-euXIV_Sk.d.ts} +3 -3
- package/dist/reporting.d.ts +8 -8
- package/dist/reporting.js +3 -3
- package/dist/{researcher-BLPHBbNV.d.ts → researcher-DE6Gpnb4.d.ts} +4 -4
- package/dist/rl.d.ts +194 -656
- package/dist/rl.js +236 -154
- package/dist/rl.js.map +1 -1
- package/dist/{rubric-predictive-validity-CnEl9Jc8.d.ts → rubric-predictive-validity-Cy_W-hWZ.d.ts} +1 -1
- package/dist/{run-campaign-4Y5V5CN3.js → run-campaign-RDGAM5KJ.js} +3 -3
- package/dist/{run-improvement-loop-CNqQckTj.d.ts → run-improvement-loop-5z_l5zDz.d.ts} +2 -2
- package/dist/{run-record-De9VarXR.d.ts → run-record-e7vj1uZQ.d.ts} +1 -1
- package/dist/{runtime-trajectory-BLRiaifm.d.ts → runtime-trajectory-BDgfGZSr.d.ts} +1 -1
- package/dist/{semantic-concept-judge-DIEgr_6v.d.ts → semantic-concept-judge-Dn8Z6KEG.d.ts} +5 -31
- package/dist/series-convergence-D5OWMBg6.d.ts +33 -0
- package/dist/{statistics-CnC1FMbx.d.ts → statistics-C7PozGrZ.d.ts} +71 -2
- package/dist/{summary-report-Db0dDSWP.d.ts → summary-report-DGmUucwQ.d.ts} +1 -1
- package/dist/traces.d.ts +3 -3
- package/dist/traces.js +8 -6
- package/dist/{types-Cu3u_x59.d.ts → types-2VVIL04s.d.ts} +2 -2
- package/dist/{types-D7lLRYe9.d.ts → types-BU-7W85F.d.ts} +21 -1
- package/dist/{types-CqPax19X.d.ts → types-mn5Aqk7x.d.ts} +1 -1
- package/dist/{verdict-CeEgtjyI.d.ts → verdict-C9MlYujm.d.ts} +3 -0
- package/dist/wire/index.d.ts +3 -3
- package/dist/workflow/index.d.ts +12 -11
- package/dist/workflow/index.js +1 -1
- package/package.json +11 -1
- package/dist/chunk-BABOZOSN.js.map +0 -1
- package/dist/chunk-L5G7OUKD.js.map +0 -1
- package/dist/chunk-SHTXZ4O2.js +0 -113
- package/dist/chunk-SHTXZ4O2.js.map +0 -1
- package/dist/chunk-VIDQF3F5.js.map +0 -1
- package/dist/chunk-VUINJM5M.js.map +0 -1
- package/dist/chunk-YGYXHNAQ.js.map +0 -1
- package/dist/chunk-YV7J7X5N.js.map +0 -1
- /package/dist/{chunk-FZWAFVAA.js.map → chunk-4FBZZIYD.js.map} +0 -0
- /package/dist/{chunk-OTYQPHPL.js.map → chunk-6SOJM3VR.js.map} +0 -0
- /package/dist/{chunk-Z7VFTS2J.js.map → chunk-CY6U5S3X.js.map} +0 -0
- /package/dist/{chunk-WJL2NJXN.js.map → chunk-GSH6QNNS.js.map} +0 -0
- /package/dist/{chunk-RBNA5AZT.js.map → chunk-L3JOU6XM.js.map} +0 -0
- /package/dist/{chunk-ZZ2HOPME.js.map → chunk-TWS7AZEY.js.map} +0 -0
- /package/dist/{run-campaign-4Y5V5CN3.js.map → run-campaign-RDGAM5KJ.js.map} +0 -0
package/dist/index.d.ts
CHANGED
|
@@ -1,30 +1,32 @@
|
|
|
1
|
-
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-
|
|
2
|
-
import { R as RunRecord, a as RunSplitTag } from './run-record-
|
|
3
|
-
export { e as AGENT_PROFILE_KINDS, A as AgentProfileCell, d as AgentProfileCellInput, f as AgentProfileCellSchemaVersion, g as AgentProfileCellValidationError, h as AgentProfileDimensionValue, i as AgentProfileHarness, j as AgentProfileJson, k as AgentProfileKind, l as AgentProfileSource, m as AgentProfileSourceInput, J as JudgeScoresRecord, c as RunJudgeMetadata, n as RunOutcome, o as RunRecordValidationError, b as RunTokenUsage, S as SandboxAgentProfileLike, p as agentProfileCellHashMaterial, q as agentProfileCellKey, r as assertRunAgentProfileCell, s as buildAgentProfileCell, t as buildSandboxAgentProfileCell, u as groupRunsByAgentProfileCell, v as isRunRecord, w as parseRunRecordSafe, x as requireAgentProfileCell, y as roundTripRunRecord, z as toAgentProfileJson, B as validateAgentProfileCell, C as validateRunRecord, D as verifyAgentProfileCell } from './run-record-
|
|
4
|
-
export { B as BehavioralMetrics,
|
|
5
|
-
import { l as ChatRequest, p as CreateChatClientOpts } from './types-
|
|
6
|
-
export { A as Analyst, a as AnalystContext, g as AnalystCost, c as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-
|
|
7
|
-
export {
|
|
8
|
-
export {
|
|
1
|
+
export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-_Qb7skHX.js';
|
|
2
|
+
import { R as RunRecord, a as RunSplitTag } from './run-record-e7vj1uZQ.js';
|
|
3
|
+
export { e as AGENT_PROFILE_KINDS, A as AgentProfileCell, d as AgentProfileCellInput, f as AgentProfileCellSchemaVersion, g as AgentProfileCellValidationError, h as AgentProfileDimensionValue, i as AgentProfileHarness, j as AgentProfileJson, k as AgentProfileKind, l as AgentProfileSource, m as AgentProfileSourceInput, J as JudgeScoresRecord, c as RunJudgeMetadata, n as RunOutcome, o as RunRecordValidationError, b as RunTokenUsage, S as SandboxAgentProfileLike, p as agentProfileCellHashMaterial, q as agentProfileCellKey, r as assertRunAgentProfileCell, s as buildAgentProfileCell, t as buildSandboxAgentProfileCell, u as groupRunsByAgentProfileCell, v as isRunRecord, w as parseRunRecordSafe, x as requireAgentProfileCell, y as roundTripRunRecord, z as toAgentProfileJson, B as validateAgentProfileCell, C as validateRunRecord, D as verifyAgentProfileCell } from './run-record-e7vj1uZQ.js';
|
|
4
|
+
export { B as BehavioralMetrics, x as ConceptComplexity, y as ConceptFinding, z as ConceptSpec, A as ConceptWeightStrategy, C as CreateAnalystAiConfig, E as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, e as FindingSubject, f as FindingSubjectKind, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, G as SEMANTIC_CONCEPT_JUDGE_VERSION, l as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, H as SemanticConceptJudgeResult, m as SkillUsageAnalyst, J as SuboptimalCode, L as SuboptimalSignal, M as computeTraceMetrics, r as createAnalystAi, N as createSemanticConceptJudge, s as defaultIsMaterial, t as diffFindings, O as runSemanticConceptJudge } from './semantic-concept-judge-Dn8Z6KEG.js';
|
|
5
|
+
import { l as ChatRequest, p as CreateChatClientOpts } from './types-2VVIL04s.js';
|
|
6
|
+
export { A as Analyst, a as AnalystContext, g as AnalystCost, c as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-2VVIL04s.js';
|
|
7
|
+
export { a as AnalystHooks, A as AnalystRegistry, c as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, b as buildDefaultAnalystRegistry } from './default-registry-zoGHUQEH.js';
|
|
8
|
+
export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-5b7xXXOr.js';
|
|
9
9
|
import { TCloud } from '@tangle-network/tcloud';
|
|
10
10
|
import { B as BenchmarkRunnerConfig, S as Scenario, c as BenchmarkReport, P as ProductClientConfig, C as CheckResult, T as TestResult, d as PersonaConfig, D as DriverResult, e as DriverState, b as JudgeFn, f as CollectedArtifacts, g as ScenarioResult, h as TurnMetrics, i as ScenarioFile, j as CompletionCriterion } from './types-Croy5h7V.js';
|
|
11
11
|
export { A as ArtifactCheck, k as ArtifactResult, E as EvalResult, F as FeedbackPattern, l as JudgeConfig, a as JudgeInput, m as JudgeRubric, J as JudgeScore, n as PersonaRigor, R as RouteMap, o as RubricDimension, p as Turn, q as TurnResult } from './types-Croy5h7V.js';
|
|
12
12
|
export { c as ControlActionFailureMode, d as ControlActionOutcome, e as ControlBudget, f as ControlContext, g as ControlDecision, C as ControlEvalResult, a as ControlRunResult, h as ControlRuntimeConfig, i as ControlRuntimeError, j as ControlSeverity, b as ControlStep, k as ControlStopPolicies, S as StopDecision, l as allCriticalPassed, o as objectiveEval, r as runAgentControlLoop, s as stopOnNoProgress, m as stopOnRepeatedAction, n as subjectiveEval } from './control-runtime-DuFBYg7A.js';
|
|
13
|
-
import { A as AgentEvalError } from './errors-
|
|
14
|
-
export {
|
|
15
|
-
import { b as FeedbackLabel, F as FeedbackTrajectoryStore, a as FeedbackTrajectory } from './feedback-trajectory-
|
|
16
|
-
export { c as FeedbackArtifactType, d as FeedbackAttempt, e as FeedbackLabelKind, f as FeedbackLabelSource, g as FeedbackOptimizerRow, h as FeedbackOutcome, i as FeedbackReplayAdapter, j as FeedbackReplayResult, k as FeedbackSeverity, l as FeedbackSplitPolicy, m as FeedbackTask, n as FeedbackTrajectoryFilter, o as FileSystemFeedbackTrajectoryStore, I as InMemoryFeedbackTrajectoryStore, P as PreferenceMemoryEntry, p as ProposedSideEffect, q as assignFeedbackSplit, r as controlRunToFeedbackTrajectory, s as createFeedbackTrajectory, t as feedbackTrajectoriesToDatasetScenarios, u as feedbackTrajectoriesToOptimizerRows, v as feedbackTrajectoryToDatasetScenario, w as feedbackTrajectoryToOptimizerRow, x as parseFeedbackTrajectoriesJsonl, y as renderPreferenceMemoryMarkdown, z as replayFeedbackTrajectories, A as replayFeedbackTrajectory, B as serializeFeedbackTrajectoriesJsonl, C as summarizePreferenceMemory, D as withAssignedFeedbackSplit } from './feedback-trajectory-
|
|
17
|
-
import { A as AgentProfile$1 } from './
|
|
18
|
-
export { c as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict,
|
|
13
|
+
import { A as AgentEvalError, J as JudgeError, a as ConfigError } from './errors-CzMUYo7b.js';
|
|
14
|
+
export { b as AgentEvalErrorCode, C as CaptureIntegrityError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-CzMUYo7b.js';
|
|
15
|
+
import { b as FeedbackLabel, F as FeedbackTrajectoryStore, a as FeedbackTrajectory } from './feedback-trajectory-D9OVLrg9.js';
|
|
16
|
+
export { c as FeedbackArtifactType, d as FeedbackAttempt, e as FeedbackLabelKind, f as FeedbackLabelSource, g as FeedbackOptimizerRow, h as FeedbackOutcome, i as FeedbackReplayAdapter, j as FeedbackReplayResult, k as FeedbackSeverity, l as FeedbackSplitPolicy, m as FeedbackTask, n as FeedbackTrajectoryFilter, o as FileSystemFeedbackTrajectoryStore, I as InMemoryFeedbackTrajectoryStore, P as PreferenceMemoryEntry, p as ProposedSideEffect, q as assignFeedbackSplit, r as controlRunToFeedbackTrajectory, s as createFeedbackTrajectory, t as feedbackTrajectoriesToDatasetScenarios, u as feedbackTrajectoriesToOptimizerRows, v as feedbackTrajectoryToDatasetScenario, w as feedbackTrajectoryToOptimizerRow, x as parseFeedbackTrajectoriesJsonl, y as renderPreferenceMemoryMarkdown, z as replayFeedbackTrajectories, A as replayFeedbackTrajectory, B as serializeFeedbackTrajectoriesJsonl, C as summarizePreferenceMemory, D as withAssignedFeedbackSplit } from './feedback-trajectory-D9OVLrg9.js';
|
|
17
|
+
import { b as CorrectnessChecker, A as AgentProfile$1 } from './pre-registration-DELOEJ8v.js';
|
|
18
|
+
export { c as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, g as HypothesisResult, L as LlmCorrectnessCheckerOpts, h as ProducedProposal, P as ProducedState, i as ProposalEventLike, j as RequirementCheck, R as RuntimeEventLike, k as SatisfiedBy, S as SignedManifest, l as SignedManifestAlgo, T as TaskGold, m as ToolCallEventLike, V as ValidationContext, n as ValidationIssue, o as ValidationResult, p as agentProfileHash, q as assertRealBackend, r as byteLengthRange, s as canonicalize, t as completionVerdict, u as composeValidators, v as containsAll, w as createLlmCorrectnessChecker, x as createTokenRecallChecker, y as evaluateHypothesis, z as extractProducedState, D as hashJson, E as jsonHasKeys, F as parseCorrectnessResponse, G as regexMatch, I as signManifest, J as summarizeBackendIntegrity, K as verifyCompletion, M as verifyManifest } from './pre-registration-DELOEJ8v.js';
|
|
19
19
|
export { DataAcquisitionPlan, KnowledgeAcquisitionMode, KnowledgeBundle, KnowledgeFallbackPolicy, KnowledgeFreshness, KnowledgeImportance, KnowledgeReadinessReport, KnowledgeRecommendedAction, KnowledgeRequirement, KnowledgeRequirementCategory, KnowledgeResponsibleSurface, KnowledgeSensitivity, ScoreKnowledgeReadinessOptions, UserQuestion, acquisitionPlansForKnowledgeGaps, blockingKnowledgeEval, knowledgeReadinessTracePayload, scoreKnowledgeReadiness, userQuestionsForKnowledgeGaps } from './knowledge/index.js';
|
|
20
|
-
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-
|
|
21
|
-
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-
|
|
22
|
-
export {
|
|
20
|
+
import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-euXIV_Sk.js';
|
|
21
|
+
export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-euXIV_Sk.js';
|
|
22
|
+
export { c as CliffsMagnitude, d as CorpusAgreementOptions, e as CorpusAgreementPerDimension, C as CorpusAgreementReport, f as CorpusScoreRecord, g as EProcess, h as EProcessOptions, E as EProcessState, i as EProcessStep, P as PairedBootstrapOptions, a as PairedBootstrapResult, W as WeightedCompositeInput, j as WeightedCompositeResult, b as benjaminiHochberg, k as bonferroni, l as cliffsDelta, m as cohensD, n as confidenceInterval, o as corpusInterRaterAgreement, q as corpusInterRaterAgreementFromJudgeScores, r as eProcess, s as interRaterReliability, t as interpretCliffs, u as mannWhitneyU, v as mulberry32, x as normalizeScores, p as pairedBootstrap, y as pairedMde, z as pairedTTest, A as partialCredit, B as requiredSampleSize, D as weightedComposite, F as weightedMean, w as wilcoxonSignedRank } from './statistics-C7PozGrZ.js';
|
|
23
23
|
import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
|
|
24
24
|
export { c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
|
|
25
25
|
import { OtelExporter, OtelExportConfig } from './traces.js';
|
|
26
26
|
export { CaptureFetchContext, CaptureFetchOptions, ExportableSpan, ExtractedUsage, FlattenOtlpOptions, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SpanNotFoundError, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete } from './traces.js';
|
|
27
27
|
export { D as DEFAULT_TRACE_ANALYST_BUDGETS, b as DatasetOverview, E as ErrorCluster, Q as QueryTracesPage, S as SearchSpanResult, c as SearchTraceResult, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, T as TraceAnalysisStore, f as TraceAnalystByteBudgets, g as TraceAnalystFilters, a as TraceAnalystSpan, h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, j as TraceAnalystTraceSummary, V as ViewSpansResult, k as ViewTraceOversized, l as ViewTraceResult } from './store-C1YxJDEK.js';
|
|
28
|
+
import { a as JudgeConfig, S as Scenario$1, G as Gate, J as JudgeScore } from './types-BU-7W85F.js';
|
|
29
|
+
import { A as AnalyzeRunsOptions } from './analyze-runs-DwCEkpO_.js';
|
|
28
30
|
import { S as SteeringBundle } from './harness-optimizer-EnEnQPsr.js';
|
|
29
31
|
export { D as DEFAULT_HARNESS_OBJECTIVES, H as HarnessAdapter, a as HarnessExperimentConfig, b as HarnessExperimentResult, c as HarnessIntervention, d as HarnessRunRequest, e as HarnessRunResult, f as HarnessScenario, g as HarnessSelection, h as HarnessVariant, i as HarnessVariantReport, M as MeasurementPolicy, j as SteeringDelta, k as SteeringRolePrompt, W as WorkflowTopology, m as mergeSteeringBundle, r as renderSteeringText, l as runHarnessExperiment, s as selectHarnessVariant, n as summarizeHarnessResults } from './harness-optimizer-EnEnQPsr.js';
|
|
30
32
|
import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult } from './test-graded-scenario-BdVaPyHT.js';
|
|
@@ -33,7 +35,7 @@ import { b as RunScoreWeights, R as RunScore } from './run-critic-BAIjX99r.js';
|
|
|
33
35
|
export { D as DEFAULT_RUN_SCORE_WEIGHTS, c as RunCritic, d as RunCriticOptions, a as RunTrace, e as aggregateRunScore, f as clamp01 } from './run-critic-BAIjX99r.js';
|
|
34
36
|
import { T as TraceEmitter } from './emitter-DEZwY14K.js';
|
|
35
37
|
export { R as RunCompleteHook, a as RunCompleteHookContext, S as SpanHandle, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-DEZwY14K.js';
|
|
36
|
-
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-
|
|
38
|
+
export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-VJ9A7aST.js';
|
|
37
39
|
export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-CqTxMwDw.js';
|
|
38
40
|
export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, P as ProviderRedactor, c as RawProviderDirection, d as RawProviderEvent, R as RawProviderSink, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
|
|
39
41
|
export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionReport, R as RedactionRule, r as redactString, c as redactValue } from './redact-B40YG2M_.js';
|
|
@@ -42,31 +44,35 @@ export { A as Artifact, E as EventKind, i as FAILURE_CLASSES, F as FailureClass,
|
|
|
42
44
|
import { T as TraceStore, R as RunFilter } from './store-CKUAgsJz.js';
|
|
43
45
|
export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, S as SpanFilter } from './store-CKUAgsJz.js';
|
|
44
46
|
export { D as DEFAULT_FAILURE_RULES, b as FailureClassification, c as FailureContext, d as FailureRule, e as classifyFailure } from './failure-cluster-CL7IVgkJ.js';
|
|
45
|
-
export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-
|
|
47
|
+
export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-BDgfGZSr.js';
|
|
46
48
|
import { a as BaselineReport } from './baseline-DE36-Np7.js';
|
|
47
49
|
export { B as BaselineOptions, M as MetricSamples, b as MetricVerdict, T as ToolStats, d as ToolUseMetrics, e as ToolUseOptions, f as compareToBaseline, c as computeToolUseMetrics, i as iqr, w as welchsTTest } from './baseline-DE36-Np7.js';
|
|
48
|
-
import {
|
|
50
|
+
import { a as TrajectoryStep, T as Trajectory } from './trajectory-GEdXJCL5.js';
|
|
49
51
|
export { b as buildTrajectory } from './trajectory-GEdXJCL5.js';
|
|
52
|
+
import { b as ChannelRollup, C as CostLedger } from './cost-ledger-DuSqlw5B.js';
|
|
53
|
+
export { a as CostChannel, c as CostLedgerEntry, d as CostLedgerSummary, e as CostResult, f as CostUsage, g as costForUsage, m as modelPriceKey } from './cost-ledger-DuSqlw5B.js';
|
|
50
54
|
export { D as Direction, O as Objective, P as ParetoResult, c as crowdingDistance, d as dominates, p as paretoFrontier, a as paretoFrontierWithCrowding, s as scalarScore } from './pareto-E-pembql.js';
|
|
51
|
-
export {
|
|
52
|
-
import {
|
|
53
|
-
|
|
54
|
-
export {
|
|
55
|
-
export {
|
|
55
|
+
export { S as SeriesConvergenceOptions, a as SeriesConvergenceResult, b as analyzeSeries } from './series-convergence-D5OWMBg6.js';
|
|
56
|
+
import { D as DefaultVerdict } from './verdict-C9MlYujm.js';
|
|
57
|
+
import { a as DatasetScenario, b as Dataset } from './dataset-BbGkaN2I.js';
|
|
58
|
+
export { d as DatasetDifficulty, c as DatasetManifest, e as DatasetProvenance, D as DatasetSplit, H as HoldoutLockedError, S as SliceOptions, h as hashScenarios } from './dataset-BbGkaN2I.js';
|
|
59
|
+
export { a as CalibrationResult, c as CandidateScore, C as ContinuousAgreement, d as ContinuousAgreementOptions, b as ContinuousCalibrationResult, G as GoldenItem, P as PositionalBiasResult, S as SelfPreferenceResult, V as VerbosityBiasResult, e as calibrateJudge, f as calibrateJudgeContinuous, g as continuousAgreement, p as positionalBias, s as selfPreference, v as verbosityBias } from './judge-calibration-0p2QcWNE.js';
|
|
60
|
+
export { D as DEFAULT_RED_TEAM_CORPUS, R as RedTeamCase, a as RedTeamCategory, b as RedTeamFinding, c as RedTeamPayload, d as RedTeamReport, r as redTeamDataset, e as redTeamReport, s as scoreRedTeamOutput, t as toolNamesForRun } from './red-team-BXHil6c8.js';
|
|
61
|
+
export { c as CounterfactualContext, C as CounterfactualMutation, d as CounterfactualResult, b as CounterfactualRunner, a as attributeCounterfactuals, r as runCounterfactual } from './counterfactual-Dwibr5IW.js';
|
|
56
62
|
import { a as PrmGrader } from './rubric-BOfxn4ja.js';
|
|
57
63
|
export { EuRiskClass, GovernanceContext, GovernanceFinding, GovernanceReport, UseCaseSignals, classifyEuAiRisk, euAiActReport, nistAiRmfReport, renderMarkdown, soc2Report, summarize } from './governance/index.js';
|
|
58
|
-
import { b as Layer, S as Severity, L as LayerResult, c as VerifyContext } from './multi-layer-verifier-
|
|
59
|
-
export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-
|
|
60
|
-
import { L as LlmClientOptions } from './llm-client-
|
|
61
|
-
export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-
|
|
62
|
-
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-
|
|
63
|
-
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-
|
|
64
|
-
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-
|
|
64
|
+
import { b as Layer, S as Severity, L as LayerResult, c as VerifyContext } from './multi-layer-verifier-DUZXrPDA.js';
|
|
65
|
+
export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-DUZXrPDA.js';
|
|
66
|
+
import { L as LlmClientOptions } from './llm-client-BeEcAokY.js';
|
|
67
|
+
export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-BeEcAokY.js';
|
|
68
|
+
export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-Bx3gZ8xl.js';
|
|
69
|
+
export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-DE6Gpnb4.js';
|
|
70
|
+
export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-DGmUucwQ.js';
|
|
65
71
|
export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
|
|
66
|
-
import {
|
|
67
|
-
import { e as GepaDriverConstraints, a as RunImprovementLoopResult } from './run-improvement-loop-CNqQckTj.js';
|
|
72
|
+
import { e as GepaDriverConstraints, a as RunImprovementLoopResult } from './run-improvement-loop-5z_l5zDz.js';
|
|
68
73
|
import '@ax-llm/ax';
|
|
69
74
|
import 'zod';
|
|
75
|
+
import './insight-report-BBwvOh6x.js';
|
|
70
76
|
import './outcome-store-rnXLEqSn.js';
|
|
71
77
|
|
|
72
78
|
/**
|
|
@@ -533,33 +539,76 @@ declare class CrossFamilyError extends Error {
|
|
|
533
539
|
*/
|
|
534
540
|
declare function assertCrossFamily(models: string[], opts?: AssertCrossFamilyOptions): JudgeFamily[];
|
|
535
541
|
|
|
542
|
+
/**
|
|
543
|
+
* A judge's LLM response could not be parsed into scored dimensions.
|
|
544
|
+
* Thrown instead of fabricating a `{ dimension: 'parse_error', score: 0 }`
|
|
545
|
+
* row — a synthetic zero is indistinguishable from a real low score
|
|
546
|
+
* downstream. Carries the raw response for forensics. Callers (executor,
|
|
547
|
+
* ensemble wrappers) catch this per-judge and record a failed judge.
|
|
548
|
+
*/
|
|
549
|
+
declare class JudgeParseError extends JudgeError {
|
|
550
|
+
/** Name of the judge whose response failed to parse. */
|
|
551
|
+
readonly judgeName: string;
|
|
552
|
+
/** The raw (truncated) model response that failed to parse. */
|
|
553
|
+
readonly raw: string;
|
|
554
|
+
constructor(judgeName: string, raw: string, options?: {
|
|
555
|
+
cause?: unknown;
|
|
556
|
+
});
|
|
557
|
+
}
|
|
536
558
|
/**
|
|
537
559
|
* Create a domain expert judge with a configurable domain.
|
|
538
560
|
*
|
|
539
561
|
* The judge evaluates professional accuracy and depth.
|
|
562
|
+
*
|
|
563
|
+
* @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
|
|
564
|
+
* Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
|
|
565
|
+
* multi-model panels via `ensembleJudge` (src/judge-panel.ts) — which are
|
|
566
|
+
* pluggable, fail-loud, and drive the campaign/improvement-loop engines.
|
|
540
567
|
*/
|
|
541
568
|
declare function createDomainExpertJudge(domain: string): JudgeFn;
|
|
542
569
|
/**
|
|
543
570
|
* Code execution judge — evaluates whether code blocks are valid and runnable.
|
|
571
|
+
*
|
|
572
|
+
* @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
|
|
573
|
+
* Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
|
|
574
|
+
* multi-model panels via `ensembleJudge` (src/judge-panel.ts).
|
|
544
575
|
*/
|
|
545
576
|
declare const codeExecutionJudge: JudgeFn;
|
|
546
577
|
/**
|
|
547
578
|
* Coherence judge — evaluates multi-turn consistency and progression.
|
|
579
|
+
*
|
|
580
|
+
* @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
|
|
581
|
+
* Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
|
|
582
|
+
* multi-model panels via `ensembleJudge` (src/judge-panel.ts).
|
|
548
583
|
*/
|
|
549
584
|
declare const coherenceJudge: JudgeFn;
|
|
550
585
|
/**
|
|
551
586
|
* Adversarial judge — red-teams agent responses.
|
|
587
|
+
*
|
|
588
|
+
* @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
|
|
589
|
+
* Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
|
|
590
|
+
* multi-model panels via `ensembleJudge` (src/judge-panel.ts).
|
|
552
591
|
*/
|
|
553
592
|
declare const adversarialJudge: JudgeFn;
|
|
554
593
|
/**
|
|
555
594
|
* Create a custom judge with a fully custom prompt.
|
|
595
|
+
*
|
|
596
|
+
* @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
|
|
597
|
+
* Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
|
|
598
|
+
* multi-model panels via `ensembleJudge` (src/judge-panel.ts).
|
|
556
599
|
*/
|
|
557
600
|
declare function createCustomJudge(name: string, systemPrompt: string, opts?: {
|
|
558
601
|
model?: string;
|
|
559
602
|
temperature?: number;
|
|
560
603
|
maxTokens?: number;
|
|
561
604
|
}): JudgeFn;
|
|
562
|
-
/**
|
|
605
|
+
/**
|
|
606
|
+
* Default judge set (domain must be provided for domain expert)
|
|
607
|
+
*
|
|
608
|
+
* @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
|
|
609
|
+
* Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
|
|
610
|
+
* multi-model panels via `ensembleJudge` (src/judge-panel.ts).
|
|
611
|
+
*/
|
|
563
612
|
declare function defaultJudges(domain: string): JudgeFn[];
|
|
564
613
|
|
|
565
614
|
interface LiveProofArtifact {
|
|
@@ -980,6 +1029,59 @@ declare class DualAgentBench {
|
|
|
980
1029
|
run(config: DualAgentBenchConfig): Promise<DualAgentReport>;
|
|
981
1030
|
}
|
|
982
1031
|
|
|
1032
|
+
/**
|
|
1033
|
+
* Eval primitives as agent tools — `makeEvalTools` packages the substrate's
|
|
1034
|
+
* judge / completion / analysis entry points as JSON-Schema tool definitions
|
|
1035
|
+
* an LLM agent loop can call. The host closes over the live config (judge
|
|
1036
|
+
* panels, the correctness checker, analyst registry); the agent passes only
|
|
1037
|
+
* data over the wire.
|
|
1038
|
+
*
|
|
1039
|
+
* Three tools, each present only when its config section is supplied:
|
|
1040
|
+
* - `run_judges` — score an artifact with the configured `JudgeConfig`s
|
|
1041
|
+
* - `verify_completion` — gold-spec requirement check → `CompletionVerdict`
|
|
1042
|
+
* - `analyze_runs` — `RunRecord[]` (inline or from a file) → `InsightReport`
|
|
1043
|
+
*
|
|
1044
|
+
* `toOpenAiTool` converts a definition to the OpenAI function-tool wire shape.
|
|
1045
|
+
*/
|
|
1046
|
+
|
|
1047
|
+
/** One agent-callable tool. `parameters` is a JSON Schema (draft-07+) object. */
|
|
1048
|
+
interface EvalToolDef {
|
|
1049
|
+
name: string;
|
|
1050
|
+
description: string;
|
|
1051
|
+
parameters: Record<string, unknown>;
|
|
1052
|
+
handler: (args: unknown, ctx?: {
|
|
1053
|
+
signal?: AbortSignal;
|
|
1054
|
+
}) => Promise<unknown>;
|
|
1055
|
+
}
|
|
1056
|
+
interface MakeEvalToolsConfig {
|
|
1057
|
+
/** Judges available to `run_judges`. Omit to exclude the tool. */
|
|
1058
|
+
judges?: Array<JudgeConfig<unknown>>;
|
|
1059
|
+
/** Host-side correctness checker for `verify_completion` (the third
|
|
1060
|
+
* `verifyCompletion` argument — a function, so it cannot cross the wire).
|
|
1061
|
+
* Omit to exclude the tool. */
|
|
1062
|
+
completion?: {
|
|
1063
|
+
checkCorrectness: CorrectnessChecker;
|
|
1064
|
+
};
|
|
1065
|
+
/** `analyzeRuns` options minus `runs` (runs arrive as tool args, inline or
|
|
1066
|
+
* via `path`). Omit to exclude the tool. */
|
|
1067
|
+
analyze?: Omit<AnalyzeRunsOptions, 'runs'>;
|
|
1068
|
+
}
|
|
1069
|
+
/** OpenAI function-tool wire shape for an `EvalToolDef`. */
|
|
1070
|
+
declare function toOpenAiTool(def: EvalToolDef): {
|
|
1071
|
+
type: 'function';
|
|
1072
|
+
function: {
|
|
1073
|
+
name: string;
|
|
1074
|
+
description: string;
|
|
1075
|
+
parameters: Record<string, unknown>;
|
|
1076
|
+
};
|
|
1077
|
+
};
|
|
1078
|
+
/**
|
|
1079
|
+
* Build the eval toolset for the supplied config. Only sections present in
|
|
1080
|
+
* `cfg` produce tools, so the agent's tool list mirrors what the host
|
|
1081
|
+
* actually wired. Handlers fail loud on malformed args — no silent defaults.
|
|
1082
|
+
*/
|
|
1083
|
+
declare function makeEvalTools(cfg: MakeEvalToolsConfig): EvalToolDef[];
|
|
1084
|
+
|
|
983
1085
|
/**
|
|
984
1086
|
* Judge-ensemble reducer — folds N independent judge verdicts on the same
|
|
985
1087
|
* artifact into one aggregate score.
|
|
@@ -1006,6 +1108,12 @@ interface JudgeVerdict<D extends string = string> {
|
|
|
1006
1108
|
rationale?: string;
|
|
1007
1109
|
/** Optional reported cost — summed across ALL verdicts (failed included). */
|
|
1008
1110
|
costUsd?: number;
|
|
1111
|
+
/** Optional per-dimension reasoning/evidence. Carried through to
|
|
1112
|
+
* `EnsembleAggregate.verdicts` verbatim — never folded into the math. */
|
|
1113
|
+
detail?: Partial<Record<D, {
|
|
1114
|
+
reasoning?: string;
|
|
1115
|
+
evidence?: string;
|
|
1116
|
+
}>>;
|
|
1009
1117
|
}
|
|
1010
1118
|
/** The aggregated ensemble result. */
|
|
1011
1119
|
interface EnsembleAggregate<D extends string = string> {
|
|
@@ -1023,6 +1131,9 @@ interface EnsembleAggregate<D extends string = string> {
|
|
|
1023
1131
|
costUsd: number;
|
|
1024
1132
|
/** First non-empty survivor rationale, or `'llm-judge'`. */
|
|
1025
1133
|
rationale: string;
|
|
1134
|
+
/** The input verdicts, verbatim — drill-down to raw scores, `detail`
|
|
1135
|
+
* reasoning/evidence, and per-verdict cost without re-running judges. */
|
|
1136
|
+
verdicts: JudgeVerdict<D>[];
|
|
1026
1137
|
}
|
|
1027
1138
|
/**
|
|
1028
1139
|
* Reduce per-judge verdicts to one aggregate. Generic over the rubric: pass the
|
|
@@ -1037,6 +1148,139 @@ interface EnsembleAggregate<D extends string = string> {
|
|
|
1037
1148
|
*/
|
|
1038
1149
|
declare function aggregateJudgeVerdicts<D extends string>(verdicts: readonly JudgeVerdict<D>[], dimensionKeys: readonly D[], weights?: Partial<Record<D, number>>): EnsembleAggregate<D>;
|
|
1039
1150
|
|
|
1151
|
+
/**
|
|
1152
|
+
* Wrap a single judge LLM call with retry, optional fallback-model
|
|
1153
|
+
* rotation, exponential backoff, and a typed `JudgeRetryOutcome`. Callers
|
|
1154
|
+
* MUST inspect `succeeded` before using `value`; on failure the library
|
|
1155
|
+
* returns `value: null` rather than substituting a default, so a judge
|
|
1156
|
+
* abort cannot silently corrupt a downstream composite.
|
|
1157
|
+
*
|
|
1158
|
+
* Reporting contract: callers ship `TrialResult.judgeSucceeded = succeeded`
|
|
1159
|
+
* and `TrialResult.judgeAttempts = attempts` so `aggregateTrialsByMode`
|
|
1160
|
+
* with `mode: 'exclude-failed'` drops the trial.
|
|
1161
|
+
*/
|
|
1162
|
+
/** Retry policy for judge LLM calls. */
|
|
1163
|
+
interface JudgeRetryPolicy {
|
|
1164
|
+
/** Max attempts per model. Default 3 (one initial + two retries). */
|
|
1165
|
+
maxAttempts?: number;
|
|
1166
|
+
/** Per-attempt timeout in ms. Default 300_000. */
|
|
1167
|
+
timeoutMs?: number;
|
|
1168
|
+
/**
|
|
1169
|
+
* Models to try, in order. The first model is the primary; subsequent
|
|
1170
|
+
* models are fallbacks invoked only when ALL retries on the previous
|
|
1171
|
+
* model have been exhausted. Example: `['claude-code/sonnet', 'kimi-code/k2p6']`
|
|
1172
|
+
* runs claude-code up to maxAttempts times, then falls back to kimi.
|
|
1173
|
+
* If omitted, the caller's judge function controls model selection and
|
|
1174
|
+
* the retries apply to that single model.
|
|
1175
|
+
*/
|
|
1176
|
+
models?: readonly string[];
|
|
1177
|
+
/** Exponential backoff function, default `attempt → min(500 * 2^attempt, 16_000)`. */
|
|
1178
|
+
backoffMs?: (attempt: number) => number;
|
|
1179
|
+
/**
|
|
1180
|
+
* Predicate deciding whether an error should trigger a retry. Defaults to
|
|
1181
|
+
* `isTransientLlmError` — the package-wide classifier shared with
|
|
1182
|
+
* `callLlm` — which retries aborts/timeouts, network faults, HTTP/2
|
|
1183
|
+
* transport faults, and any `LlmCallError` with status in {429,502,503,504}.
|
|
1184
|
+
* JSON-parse and schema-rejection errors are NOT retriable (the model
|
|
1185
|
+
* needs prompt adjustment, not another shot).
|
|
1186
|
+
*/
|
|
1187
|
+
isRetryable?: (err: unknown) => boolean;
|
|
1188
|
+
}
|
|
1189
|
+
/** Outcome of a wrapped judge invocation. */
|
|
1190
|
+
interface JudgeRetryOutcome<T> {
|
|
1191
|
+
/** The judge's returned value when `succeeded === true`. */
|
|
1192
|
+
value: T | null;
|
|
1193
|
+
/** True iff one of the attempts completed without throwing. */
|
|
1194
|
+
succeeded: boolean;
|
|
1195
|
+
/** Total attempts made across all models. */
|
|
1196
|
+
attempts: number;
|
|
1197
|
+
/** Which model the successful attempt used (when succeeded). */
|
|
1198
|
+
modelUsed?: string;
|
|
1199
|
+
/** Last error captured when `succeeded === false`. */
|
|
1200
|
+
error?: Error;
|
|
1201
|
+
/** Per-attempt error log for forensics. */
|
|
1202
|
+
attemptErrors: Array<{
|
|
1203
|
+
attempt: number;
|
|
1204
|
+
model: string;
|
|
1205
|
+
error: string;
|
|
1206
|
+
}>;
|
|
1207
|
+
}
|
|
1208
|
+
/**
|
|
1209
|
+
* Wrap a judge call with retry + fallback-model + typed outcome semantics.
|
|
1210
|
+
*
|
|
1211
|
+
* The `judgeFn` signature is `(model: string, signal: AbortSignal) => Promise<T>`.
|
|
1212
|
+
* The signal will be aborted at `timeoutMs`. Callers should pass the signal
|
|
1213
|
+
* to their underlying fetch/SDK call so the abort actually fires.
|
|
1214
|
+
*
|
|
1215
|
+
* Returns a typed outcome — callers MUST inspect `succeeded` before using
|
|
1216
|
+
* `value`. The library refuses to default to a silent zero score because a
|
|
1217
|
+
* synthetic zero is indistinguishable from a real low score downstream.
|
|
1218
|
+
*/
|
|
1219
|
+
declare function withJudgeRetry<T>(judgeFn: (model: string, signal: AbortSignal) => Promise<T>, policy?: JudgeRetryPolicy): Promise<JudgeRetryOutcome<T>>;
|
|
1220
|
+
|
|
1221
|
+
/**
|
|
1222
|
+
* Multi-model judge panel — `ensembleJudge` builds a campaign `JudgeConfig`
|
|
1223
|
+
* that fans one artifact out to K judge models and reduces their verdicts
|
|
1224
|
+
* through `aggregateJudgeVerdicts` (src/judge-ensemble.ts).
|
|
1225
|
+
*
|
|
1226
|
+
* The panel is the fail-loud composition of the substrate's existing judge
|
|
1227
|
+
* primitives:
|
|
1228
|
+
* - `assertCrossFamily` (construction-time) — a single-family panel is
|
|
1229
|
+
* correlated bias, not independent signal.
|
|
1230
|
+
* - `withJudgeRetry` (per model, opt-in) — transient-fault retry with a
|
|
1231
|
+
* typed outcome; a judge that exhausts retries is recorded as failed,
|
|
1232
|
+
* never folded into a zero.
|
|
1233
|
+
* - `aggregateJudgeVerdicts` — the pure reducer; throws when EVERY judge
|
|
1234
|
+
* failed so a silent zero can't reach the gate.
|
|
1235
|
+
*
|
|
1236
|
+
* The returned `JudgeScore` is on the campaign [0,1] scale and carries the
|
|
1237
|
+
* ensemble extras (`maxDisagreement`, `failedJudges`, `perJudge`) declared
|
|
1238
|
+
* on the canonical `JudgeScore` in src/campaign/types.ts.
|
|
1239
|
+
*/
|
|
1240
|
+
|
|
1241
|
+
interface EnsembleJudgeOptions<D extends string> {
|
|
1242
|
+
/** Judge name — becomes the returned `JudgeConfig.name`. */
|
|
1243
|
+
name: string;
|
|
1244
|
+
/** Rubric dimensions every model scores. Keys of the verdict's `perDimension`. */
|
|
1245
|
+
dimensions: D[];
|
|
1246
|
+
/** Judge model ids — one `scoreWith` call per entry. List a model twice to
|
|
1247
|
+
* sample it twice (votes are suffix-keyed `model#2` so none overwrite). */
|
|
1248
|
+
models: string[];
|
|
1249
|
+
/**
|
|
1250
|
+
* Score the artifact with one model. Throw (or reject) on failure — the
|
|
1251
|
+
* panel records that model as a failed judge; it is never folded into a
|
|
1252
|
+
* zero. Verdict scores are clamped to [0,1] by the reducer.
|
|
1253
|
+
*/
|
|
1254
|
+
scoreWith: (model: string, input: {
|
|
1255
|
+
artifact: unknown;
|
|
1256
|
+
scenario?: unknown;
|
|
1257
|
+
}) => Promise<JudgeVerdict<D>>;
|
|
1258
|
+
/**
|
|
1259
|
+
* Per-model retry policy, applied via `withJudgeRetry`. The panel's
|
|
1260
|
+
* `models` list drives the fan-out, so `retry.models` (the fallback
|
|
1261
|
+
* rotation) is overridden to each panel model in turn.
|
|
1262
|
+
*/
|
|
1263
|
+
retry?: JudgeRetryPolicy;
|
|
1264
|
+
/** Enforce `assertCrossFamily` over `models` at construction. Default true.
|
|
1265
|
+
* Opt out only for deliberate single-family panels (e.g. self-consistency
|
|
1266
|
+
* sampling of one model). */
|
|
1267
|
+
crossFamily?: boolean;
|
|
1268
|
+
/** Composite weights forwarded to `aggregateJudgeVerdicts`: a partial map
|
|
1269
|
+
* selects AND weights exactly the named dimensions. Omit for uniform. */
|
|
1270
|
+
weights?: Partial<Record<D, number>>;
|
|
1271
|
+
}
|
|
1272
|
+
/**
|
|
1273
|
+
* Build a campaign-shaped `JudgeConfig` whose `score()` runs every panel
|
|
1274
|
+
* model in parallel and reduces the surviving verdicts to one canonical
|
|
1275
|
+
* `JudgeScore` in [0,1].
|
|
1276
|
+
*
|
|
1277
|
+
* Failure semantics: a model whose `scoreWith` throws (or exhausts `retry`)
|
|
1278
|
+
* lands in `failedJudges` and is excluded from the means. When EVERY model
|
|
1279
|
+
* fails, `aggregateJudgeVerdicts` throws — the campaign engine records a
|
|
1280
|
+
* failed cell instead of averaging a fabricated zero.
|
|
1281
|
+
*/
|
|
1282
|
+
declare function ensembleJudge<D extends string>(opts: EnsembleJudgeOptions<D>): JudgeConfig<unknown>;
|
|
1283
|
+
|
|
1040
1284
|
type SandboxJudgeKind = 'compiler' | 'test' | 'linter' | 'security';
|
|
1041
1285
|
interface SandboxJudgeSpec {
|
|
1042
1286
|
id: string;
|
|
@@ -1275,118 +1519,6 @@ declare class BudgetGuard {
|
|
|
1275
1519
|
get state(): Record<keyof BudgetSpec, number>;
|
|
1276
1520
|
}
|
|
1277
1521
|
|
|
1278
|
-
/**
|
|
1279
|
-
* CostLedger — per-run token + USD accounting with an explicit `costUnknown`
|
|
1280
|
-
* axis, folded over the substrate's pricing resolver.
|
|
1281
|
-
*
|
|
1282
|
-
* `estimateCost` already resolves a model id to a price (exact table, then
|
|
1283
|
-
* family regex) and warns-once on a miss, but it returns 0 for an unpriced
|
|
1284
|
-
* model — indistinguishable downstream from a genuinely free run. Four
|
|
1285
|
-
* consumers re-wrap it to surface that distinction (physim's `costForUsage` /
|
|
1286
|
-
* `modelPriceKey` is the cleanest), and to bucket spend by "channel" (the
|
|
1287
|
-
* logical role of the call: agent / judge / verifier / …) so a dashboard can
|
|
1288
|
-
* answer "how much did judging cost vs the agent itself?".
|
|
1289
|
-
*
|
|
1290
|
-
* This is the canonical version. `modelPriceKey` exposes the resolver's verdict
|
|
1291
|
-
* as a stable key (or null). `CostLedger` folds usage records into per-channel
|
|
1292
|
-
* and total rollups, tracks `unpricedModels` so a $0 is never mistaken for a
|
|
1293
|
-
* measured zero, and computes cost-per-completed-task.
|
|
1294
|
-
*/
|
|
1295
|
-
/** Logical role of an LLM call. Free-form union — consumers add their own
|
|
1296
|
-
* channels; the rollup keys on whatever string is supplied. */
|
|
1297
|
-
type CostChannel = 'agent' | 'judge' | 'verifier' | 'analyst' | 'driver' | (string & {});
|
|
1298
|
-
interface CostUsage {
|
|
1299
|
-
inputTokens: number;
|
|
1300
|
-
outputTokens: number;
|
|
1301
|
-
cachedTokens?: number;
|
|
1302
|
-
}
|
|
1303
|
-
/**
|
|
1304
|
-
* Resolve a model id to the stable pricing key the substrate's `MODEL_PRICING`
|
|
1305
|
-
* / family resolver would use, or null when the id is unpriced. A non-null
|
|
1306
|
-
* return means `estimateCost` will produce a real number for this id; null
|
|
1307
|
-
* means any cost computed is `costUnknown` and the 0 must not aggregate as a
|
|
1308
|
-
* measured cost.
|
|
1309
|
-
*/
|
|
1310
|
-
declare function modelPriceKey(model: string): string | null;
|
|
1311
|
-
interface CostResult {
|
|
1312
|
-
costUsd: number;
|
|
1313
|
-
/** True when `model` has no pricing — the 0 is "not priced", NOT "free". */
|
|
1314
|
-
costUnknown: boolean;
|
|
1315
|
-
}
|
|
1316
|
-
/**
|
|
1317
|
-
* Cost for one usage record. Resolves pricing via the substrate resolver and
|
|
1318
|
-
* flags `costUnknown` when the model is unpriced so the 0 is observable rather
|
|
1319
|
-
* than silently emitted as a measured cost. Cached tokens are billed at the
|
|
1320
|
-
* model's input rate when present (no separate cache-discount table — callers
|
|
1321
|
-
* that need provider-specific cache pricing supply `actualCostUsd` upstream).
|
|
1322
|
-
*/
|
|
1323
|
-
declare function costForUsage(model: string, usage: CostUsage): CostResult;
|
|
1324
|
-
interface CostLedgerEntry extends CostUsage {
|
|
1325
|
-
model: string;
|
|
1326
|
-
channel: CostChannel;
|
|
1327
|
-
costUsd: number;
|
|
1328
|
-
costUnknown: boolean;
|
|
1329
|
-
/** Override the estimate with an observed provider cost. */
|
|
1330
|
-
actualCostUsd?: number;
|
|
1331
|
-
/** Free-form tags (scenario id, variant id, round, …). */
|
|
1332
|
-
tags?: Record<string, string>;
|
|
1333
|
-
timestamp: number;
|
|
1334
|
-
}
|
|
1335
|
-
interface ChannelRollup {
|
|
1336
|
-
channel: CostChannel;
|
|
1337
|
-
calls: number;
|
|
1338
|
-
inputTokens: number;
|
|
1339
|
-
outputTokens: number;
|
|
1340
|
-
cachedTokens: number;
|
|
1341
|
-
costUsd: number;
|
|
1342
|
-
/** Calls whose model was unpriced (their costUsd is 0-but-unknown). */
|
|
1343
|
-
unpricedCalls: number;
|
|
1344
|
-
}
|
|
1345
|
-
interface CostLedgerSummary {
|
|
1346
|
-
totalCalls: number;
|
|
1347
|
-
inputTokens: number;
|
|
1348
|
-
outputTokens: number;
|
|
1349
|
-
cachedTokens: number;
|
|
1350
|
-
totalCostUsd: number;
|
|
1351
|
-
/** Per-channel breakdown, sorted by channel name. */
|
|
1352
|
-
byChannel: ChannelRollup[];
|
|
1353
|
-
/** Distinct unpriced model ids seen — non-empty means totalCostUsd is a
|
|
1354
|
-
* lower bound (some calls priced to an unknown 0). */
|
|
1355
|
-
unpricedModels: string[];
|
|
1356
|
-
/** True when no unpriced model was charged — totalCostUsd is then exact. */
|
|
1357
|
-
fullyPriced: boolean;
|
|
1358
|
-
}
|
|
1359
|
-
/**
|
|
1360
|
-
* Append-only ledger of LLM spend for a single run. Record each call with its
|
|
1361
|
-
* channel; read per-channel and total rollups plus the unpriced-model set.
|
|
1362
|
-
* Pure accounting — no I/O. The `markCompleted` / `costPerCompletedTask` pair
|
|
1363
|
-
* answers "dollars per finished task", the metric every optimizer's
|
|
1364
|
-
* quality-vs-cost tradeoff needs.
|
|
1365
|
-
*/
|
|
1366
|
-
declare class CostLedger {
|
|
1367
|
-
private readonly entries;
|
|
1368
|
-
private completedTasks;
|
|
1369
|
-
/**
|
|
1370
|
-
* Record one LLM call. The cost is computed from pricing unless
|
|
1371
|
-
* `actualCostUsd` is supplied (a finite observed cost from the provider
|
|
1372
|
-
* response), in which case `costUnknown` is false regardless of pricing.
|
|
1373
|
-
*/
|
|
1374
|
-
record(input: {
|
|
1375
|
-
model: string;
|
|
1376
|
-
channel: CostChannel;
|
|
1377
|
-
usage: CostUsage;
|
|
1378
|
-
actualCostUsd?: number;
|
|
1379
|
-
tags?: Record<string, string>;
|
|
1380
|
-
timestamp?: number;
|
|
1381
|
-
}): CostLedgerEntry;
|
|
1382
|
-
/** Increment the completed-task counter (used for cost-per-completed-task). */
|
|
1383
|
-
markCompleted(count?: number): void;
|
|
1384
|
-
list(): CostLedgerEntry[];
|
|
1385
|
-
summary(): CostLedgerSummary;
|
|
1386
|
-
/** Total spend divided by completed tasks; null when nothing completed. */
|
|
1387
|
-
costPerCompletedTask(): number | null;
|
|
1388
|
-
}
|
|
1389
|
-
|
|
1390
1522
|
/**
|
|
1391
1523
|
* Cost tracker — token + USD accounting per scenario and per run.
|
|
1392
1524
|
*
|
|
@@ -2107,38 +2239,6 @@ declare function diffScorecard(scorecard: Scorecard, opts?: DiffScorecardOptions
|
|
|
2107
2239
|
*/
|
|
2108
2240
|
declare function formatScorecardDiff(diff: ScorecardDiff): string;
|
|
2109
2241
|
|
|
2110
|
-
/**
|
|
2111
|
-
* Series convergence — detects whether a sequence of scalar measurements
|
|
2112
|
-
* is stabilizing, drifting, or noisy.
|
|
2113
|
-
*
|
|
2114
|
-
* Lifted from ADC convergence.ts. The per-turn `ConvergenceTracker` is
|
|
2115
|
-
* about progress *within* a single run; this module is about drift
|
|
2116
|
-
* *across* runs (e.g. "are my nightly eval scores stabilizing?").
|
|
2117
|
-
*
|
|
2118
|
-
* Three signals:
|
|
2119
|
-
* - stabilized: last K values have low variance (< epsilon) — done
|
|
2120
|
-
* - drifting: recent trend is monotonic and beyond noise — regressing or improving
|
|
2121
|
-
* - noisy: neither — keep iterating, but flag as untrustworthy for gating
|
|
2122
|
-
*/
|
|
2123
|
-
interface SeriesConvergenceOptions {
|
|
2124
|
-
/** Window size for "recent" analysis (default 5). */
|
|
2125
|
-
window?: number;
|
|
2126
|
-
/** Coefficient-of-variation threshold below which the window is stabilized (default 0.05 = 5%). */
|
|
2127
|
-
stableCv?: number;
|
|
2128
|
-
/** Minimum monotone run length to call drift (default 3). */
|
|
2129
|
-
driftRun?: number;
|
|
2130
|
-
}
|
|
2131
|
-
interface SeriesConvergenceResult {
|
|
2132
|
-
state: 'stabilized' | 'drifting-up' | 'drifting-down' | 'noisy' | 'insufficient-data';
|
|
2133
|
-
windowMean: number;
|
|
2134
|
-
windowCv: number;
|
|
2135
|
-
/** Longest monotonic run at the tail of the series (positive for up, negative for down). */
|
|
2136
|
-
tailRun: number;
|
|
2137
|
-
/** True when n ≥ window AND windowCv ≤ stableCv. */
|
|
2138
|
-
stable: boolean;
|
|
2139
|
-
}
|
|
2140
|
-
declare function analyzeSeries(values: number[], options?: SeriesConvergenceOptions): SeriesConvergenceResult;
|
|
2141
|
-
|
|
2142
2242
|
/**
|
|
2143
2243
|
* SLO gates — quantified pass/fail primitives beyond score thresholds.
|
|
2144
2244
|
*
|
|
@@ -2338,6 +2438,185 @@ interface UiFinding {
|
|
|
2338
2438
|
createdAt?: string;
|
|
2339
2439
|
}
|
|
2340
2440
|
|
|
2441
|
+
/**
|
|
2442
|
+
* Trace contracts — finite-trace temporal assertions over span sequences.
|
|
2443
|
+
*
|
|
2444
|
+
* Five LTLf operators over one ordered span sequence — `always(p)`,
|
|
2445
|
+
* `never(p)`, `eventually(p)`, `precedes(a, b)`, `neverUnless(p, prior)` —
|
|
2446
|
+
* deterministic and judge-free. No nesting: each rule is one operator over
|
|
2447
|
+
* flat `SpanPredicate`s; compose richer checks with multiple rules.
|
|
2448
|
+
*
|
|
2449
|
+
* A built `TraceContract` is a serializable plain object (RegExp matchers
|
|
2450
|
+
* are normalized to `SerializedRegex`), so ONE contract definition is
|
|
2451
|
+
* dual-use:
|
|
2452
|
+
*
|
|
2453
|
+
* - recorded eval traces — `evaluateTraceContract(contract, await
|
|
2454
|
+
* store.spans({ runId }))`, or via the behavior DSL:
|
|
2455
|
+
* `expectAgent(store, runId).toSatisfyContract(contract)`.
|
|
2456
|
+
* - the production OTLP stream — `ExportableSpan`s flattened by
|
|
2457
|
+
* `trace/otel-bridge` satisfy `ContractSpan` structurally. A production
|
|
2458
|
+
* monitor implements `OtelExporter`, buffers `exportSpan` payloads per
|
|
2459
|
+
* trace, and runs `checkTraceContracts(buffer, contracts)` on flush:
|
|
2460
|
+
*
|
|
2461
|
+
* const buffer: ContractSpan[] = []
|
|
2462
|
+
* const monitor: OtelExporter = {
|
|
2463
|
+
* exportSpan: (s) => { buffer.push(s) },
|
|
2464
|
+
* flush: async () => {
|
|
2465
|
+
* const { allValid, verdicts } = checkTraceContracts(buffer, contracts)
|
|
2466
|
+
* if (!allValid) alert(verdicts)
|
|
2467
|
+
* },
|
|
2468
|
+
* shutdown: async () => {},
|
|
2469
|
+
* }
|
|
2470
|
+
* const store = createOtelTracingStore(inner, monitor, runId)
|
|
2471
|
+
*
|
|
2472
|
+
* `custom` predicate functions are the one non-serializable escape hatch:
|
|
2473
|
+
* the builder stamps `requiresCustom: true` (which DOES survive JSON) so a
|
|
2474
|
+
* deserialized contract that lost its function fails loud at evaluation
|
|
2475
|
+
* instead of silently weakening.
|
|
2476
|
+
*
|
|
2477
|
+
* Naming: the root barrel exports ci-gate's threshold-contract
|
|
2478
|
+
* `evaluateContract`, so the evaluators here are `evaluateTraceContract` /
|
|
2479
|
+
* `checkTraceContracts`.
|
|
2480
|
+
*/
|
|
2481
|
+
|
|
2482
|
+
/**
|
|
2483
|
+
* Minimal structural span the checker reads. Both the eval-side `Span`
|
|
2484
|
+
* (trace/schema) and the OTLP-flattened `ExportableSpan` (trace/otel-export)
|
|
2485
|
+
* satisfy it; any other producer only needs these fields.
|
|
2486
|
+
*/
|
|
2487
|
+
interface ContractSpan {
|
|
2488
|
+
spanId?: string;
|
|
2489
|
+
name?: string;
|
|
2490
|
+
kind?: string;
|
|
2491
|
+
startedAt?: number;
|
|
2492
|
+
status?: string;
|
|
2493
|
+
error?: string;
|
|
2494
|
+
/** Typed field on eval-side ToolSpans; OTLP flattenings drop it (see
|
|
2495
|
+
* `tool` matching order in {@link matchSpan}). */
|
|
2496
|
+
toolName?: string;
|
|
2497
|
+
attributes?: Record<string, unknown>;
|
|
2498
|
+
}
|
|
2499
|
+
/** JSON-safe RegExp form — what the builder normalizes RegExp matchers to. */
|
|
2500
|
+
interface SerializedRegex {
|
|
2501
|
+
$regex: string;
|
|
2502
|
+
flags: string;
|
|
2503
|
+
}
|
|
2504
|
+
type TextMatcher = string | RegExp | SerializedRegex;
|
|
2505
|
+
/**
|
|
2506
|
+
* Proposition over one span. All specified fields must match (AND).
|
|
2507
|
+
* At least one field is required — an empty predicate would match every
|
|
2508
|
+
* span and is rejected.
|
|
2509
|
+
*
|
|
2510
|
+
* `tool` resolution order covers both span shapes: `span.toolName` (typed
|
|
2511
|
+
* ToolSpan) → `attributes['tool.name']` / `attributes['toolName']`
|
|
2512
|
+
* (OTLP-flat attribute conventions) → `span.name` when `kind === 'tool'`
|
|
2513
|
+
* (otel-bridge's `ExportableSpan`, which drops `toolName`).
|
|
2514
|
+
*
|
|
2515
|
+
* `attr` values match by strict equality, or regex-test when the value is a
|
|
2516
|
+
* RegExp/SerializedRegex and the attribute is a string. Structured attribute
|
|
2517
|
+
* values need `custom`.
|
|
2518
|
+
*/
|
|
2519
|
+
interface SpanPredicate {
|
|
2520
|
+
name?: TextMatcher;
|
|
2521
|
+
tool?: TextMatcher;
|
|
2522
|
+
attr?: Record<string, unknown>;
|
|
2523
|
+
custom?: (span: ContractSpan) => boolean;
|
|
2524
|
+
/** Stamped by the builder when `custom` is present. Survives JSON while
|
|
2525
|
+
* the function does not, so evaluation of a deserialized contract throws
|
|
2526
|
+
* instead of silently dropping the check. */
|
|
2527
|
+
requiresCustom?: true;
|
|
2528
|
+
}
|
|
2529
|
+
type ContractRuleKind = 'always' | 'never' | 'eventually' | 'precedes' | 'neverUnless';
|
|
2530
|
+
interface ContractRule {
|
|
2531
|
+
kind: ContractRuleKind;
|
|
2532
|
+
/** Unique within the contract — keys the per-rule score. */
|
|
2533
|
+
label: string;
|
|
2534
|
+
/** Subject predicate for always / never / eventually / neverUnless. */
|
|
2535
|
+
p?: SpanPredicate;
|
|
2536
|
+
/** precedes: the required precondition. */
|
|
2537
|
+
a?: SpanPredicate;
|
|
2538
|
+
/** precedes: the guarded match — every b-match needs an earlier a-match. */
|
|
2539
|
+
b?: SpanPredicate;
|
|
2540
|
+
/** neverUnless: the authorizing earlier match. */
|
|
2541
|
+
prior?: SpanPredicate;
|
|
2542
|
+
}
|
|
2543
|
+
/** Serializable plain object — `traceContract(name)....build()` output. */
|
|
2544
|
+
interface TraceContract {
|
|
2545
|
+
name: string;
|
|
2546
|
+
rules: ContractRule[];
|
|
2547
|
+
}
|
|
2548
|
+
interface ContractViolation {
|
|
2549
|
+
rule: string;
|
|
2550
|
+
spanId?: string;
|
|
2551
|
+
detail: string;
|
|
2552
|
+
}
|
|
2553
|
+
interface ContractVerdict extends DefaultVerdict {
|
|
2554
|
+
/** Contract name — keys this verdict in multi-contract reports. */
|
|
2555
|
+
contract: string;
|
|
2556
|
+
valid: boolean;
|
|
2557
|
+
/** Fraction of rules passing, in [0, 1]. */
|
|
2558
|
+
score: number;
|
|
2559
|
+
/** Per-rule 0|1 keyed by rule label. */
|
|
2560
|
+
scores: Record<string, number>;
|
|
2561
|
+
violations: ContractViolation[];
|
|
2562
|
+
}
|
|
2563
|
+
interface ContractCheckResult {
|
|
2564
|
+
verdicts: ContractVerdict[];
|
|
2565
|
+
allValid: boolean;
|
|
2566
|
+
}
|
|
2567
|
+
/** Test one span against one predicate. All specified fields must match. */
|
|
2568
|
+
declare function matchSpan(span: ContractSpan, predicate: SpanPredicate): boolean;
|
|
2569
|
+
declare class TraceContractBuilder {
|
|
2570
|
+
private readonly name;
|
|
2571
|
+
private readonly rules;
|
|
2572
|
+
constructor(name: string);
|
|
2573
|
+
/** Every span in the trace must satisfy `p`. */
|
|
2574
|
+
always(p: SpanPredicate, label?: string): this;
|
|
2575
|
+
/** No span in the trace may satisfy `p`. */
|
|
2576
|
+
never(p: SpanPredicate, label?: string): this;
|
|
2577
|
+
/** At least one span in the trace must satisfy `p`. */
|
|
2578
|
+
eventually(p: SpanPredicate, label?: string): this;
|
|
2579
|
+
/** Every `b`-match must have a strictly earlier `a`-match. */
|
|
2580
|
+
precedes(a: SpanPredicate, b: SpanPredicate, label?: string): this;
|
|
2581
|
+
/** Every `p`-match is a violation unless a strictly earlier `prior`-match exists. */
|
|
2582
|
+
neverUnless(p: SpanPredicate, prior: SpanPredicate, label?: string): this;
|
|
2583
|
+
build(): TraceContract;
|
|
2584
|
+
private add;
|
|
2585
|
+
}
|
|
2586
|
+
declare function traceContract(name: string): TraceContractBuilder;
|
|
2587
|
+
/**
|
|
2588
|
+
* Evaluate one contract over a span sequence. Pure and synchronous — works
|
|
2589
|
+
* on `Span[]` from a TraceStore, `ExportableSpan[]` from the otel-bridge
|
|
2590
|
+
* flattening, or any array satisfying `ContractSpan`.
|
|
2591
|
+
*/
|
|
2592
|
+
declare function evaluateTraceContract(contract: TraceContract, spans: readonly ContractSpan[]): ContractVerdict;
|
|
2593
|
+
/**
|
|
2594
|
+
* Evaluate many contracts over one span sequence. Throws on an empty
|
|
2595
|
+
* contract list — `allValid: true` over zero contracts is a silent pass.
|
|
2596
|
+
*/
|
|
2597
|
+
declare function checkTraceContracts(spans: readonly ContractSpan[], contracts: readonly TraceContract[]): ContractCheckResult;
|
|
2598
|
+
interface ContractJudgeOptions<TArtifact, TScenario extends Scenario$1 = Scenario$1> {
|
|
2599
|
+
/**
|
|
2600
|
+
* Project the span sequence out of a cell's artifact. `JudgeConfig.score`
|
|
2601
|
+
* receives only `{ artifact, scenario, signal }` (src/campaign/types.ts) —
|
|
2602
|
+
* spans are NOT reachable generically — so the consumer supplies this
|
|
2603
|
+
* explicit extraction (e.g. dispatch writes spans into the artifact, or
|
|
2604
|
+
* closes over a per-cell TraceStore read).
|
|
2605
|
+
*/
|
|
2606
|
+
spans: (input: {
|
|
2607
|
+
artifact: TArtifact;
|
|
2608
|
+
scenario: TScenario;
|
|
2609
|
+
}) => readonly ContractSpan[];
|
|
2610
|
+
/** Judge name in campaign reports. Default 'trace-contracts'. */
|
|
2611
|
+
name?: string;
|
|
2612
|
+
}
|
|
2613
|
+
/**
|
|
2614
|
+
* Adapt trace contracts to a campaign `JudgeConfig`. One judge dimension per
|
|
2615
|
+
* contract (key = contract name, value = its rule-pass fraction); composite
|
|
2616
|
+
* is the mean across contracts. Deterministic — no LLM call.
|
|
2617
|
+
*/
|
|
2618
|
+
declare function contractJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1>(contracts: readonly TraceContract[], opts: ContractJudgeOptions<TArtifact, TScenario>): JudgeConfig<TArtifact, TScenario>;
|
|
2619
|
+
|
|
2341
2620
|
/**
|
|
2342
2621
|
* Behavior DSL — pytest-style assertions over a run's trajectory.
|
|
2343
2622
|
*
|
|
@@ -2376,6 +2655,9 @@ declare class BehaviorAssertion {
|
|
|
2376
2655
|
toolCalls?: number;
|
|
2377
2656
|
llmTurns?: number;
|
|
2378
2657
|
}): Expectation;
|
|
2658
|
+
/** Evaluate a finite-trace temporal contract (`traceContract(...)`) over
|
|
2659
|
+
* this run's span sequence. See `trace-contracts.ts` for the operators. */
|
|
2660
|
+
toSatisfyContract(contract: TraceContract): Expectation;
|
|
2379
2661
|
toNeverCall(toolName: string): Expectation;
|
|
2380
2662
|
}
|
|
2381
2663
|
declare class CallExpectation implements Expectation {
|
|
@@ -2816,86 +3098,6 @@ declare function promptBisect(options: {
|
|
|
2816
3098
|
offendingParagraphIndex?: number;
|
|
2817
3099
|
}>;
|
|
2818
3100
|
|
|
2819
|
-
/**
|
|
2820
|
-
* Counterfactual replay — "what would have happened if we'd changed
|
|
2821
|
-
* exactly one thing at turn N?"
|
|
2822
|
-
*
|
|
2823
|
-
* The framework does NOT drive the agent — it sets up the replay
|
|
2824
|
-
* context (prior spans, prior state, mutation spec) and records the
|
|
2825
|
-
* resulting divergence. Consumers supply an `executeFrom(ctx)` callback
|
|
2826
|
-
* that runs their agent starting from turn N with the mutation applied.
|
|
2827
|
-
*
|
|
2828
|
-
* Counterfactual runs are recorded as a new Run with `layer='meta'` and
|
|
2829
|
-
* `parentRunId = originalRunId`, so downstream diff + correlation
|
|
2830
|
-
* pipelines see them natively.
|
|
2831
|
-
*/
|
|
2832
|
-
|
|
2833
|
-
type CounterfactualMutation = {
|
|
2834
|
-
kind: 'swap-model';
|
|
2835
|
-
at: number;
|
|
2836
|
-
newModel: string;
|
|
2837
|
-
} | {
|
|
2838
|
-
kind: 'swap-tool-result';
|
|
2839
|
-
at: number;
|
|
2840
|
-
newResult: unknown;
|
|
2841
|
-
} | {
|
|
2842
|
-
kind: 'truncate-after';
|
|
2843
|
-
at: number;
|
|
2844
|
-
} | {
|
|
2845
|
-
kind: 'inject-system-message';
|
|
2846
|
-
at: number;
|
|
2847
|
-
content: string;
|
|
2848
|
-
} | {
|
|
2849
|
-
kind: 'custom';
|
|
2850
|
-
at: number;
|
|
2851
|
-
describe: string;
|
|
2852
|
-
apply: (step: TrajectoryStep) => TrajectoryStep;
|
|
2853
|
-
};
|
|
2854
|
-
interface CounterfactualContext {
|
|
2855
|
-
originalRunId: string;
|
|
2856
|
-
originalTrajectory: Trajectory;
|
|
2857
|
-
/** Steps up to (but not including) the mutation point — the prefix the
|
|
2858
|
-
* replayed agent inherits as its prior conversation/tool history. */
|
|
2859
|
-
prefix: TrajectoryStep[];
|
|
2860
|
-
mutation: CounterfactualMutation;
|
|
2861
|
-
/** Pre-applied mutation on the step at `mutation.at`. Consumers use this
|
|
2862
|
-
* as the FIRST step the replayed agent emits (they decide whether to
|
|
2863
|
-
* re-emit it or continue from there). */
|
|
2864
|
-
mutatedStep: TrajectoryStep;
|
|
2865
|
-
}
|
|
2866
|
-
interface CounterfactualResult {
|
|
2867
|
-
counterfactualRunId: string;
|
|
2868
|
-
originalRunId: string;
|
|
2869
|
-
mutation: CounterfactualMutation;
|
|
2870
|
-
/** Structured delta summary — caller can extend via scoring. */
|
|
2871
|
-
delta: {
|
|
2872
|
-
originalOutcomeScore: number | null;
|
|
2873
|
-
counterfactualOutcomeScore: number | null;
|
|
2874
|
-
deltaScore: number | null;
|
|
2875
|
-
};
|
|
2876
|
-
}
|
|
2877
|
-
interface CounterfactualRunner {
|
|
2878
|
-
/**
|
|
2879
|
-
* Execute the agent from `ctx.prefix` with the mutation applied.
|
|
2880
|
-
* MUST emit spans into the provided emitter so they become part of
|
|
2881
|
-
* the counterfactual run. MUST call emitter.endRun() with a verdict.
|
|
2882
|
-
*/
|
|
2883
|
-
executeFrom: (ctx: CounterfactualContext, emitter: TraceEmitter) => Promise<void>;
|
|
2884
|
-
}
|
|
2885
|
-
declare function runCounterfactual(store: TraceStore, originalRunId: string, mutation: CounterfactualMutation, runner: CounterfactualRunner): Promise<CounterfactualResult>;
|
|
2886
|
-
/**
|
|
2887
|
-
* Aggregate a batch of counterfactuals into a simple attribution table:
|
|
2888
|
-
* which mutation kinds move outcomes most? (Useful when you run a grid
|
|
2889
|
-
* over the same trajectory — swap-model at every llm span, swap-tool
|
|
2890
|
-
* at every tool span — and want a ranked summary.)
|
|
2891
|
-
*/
|
|
2892
|
-
declare function attributeCounterfactuals(results: CounterfactualResult[]): Array<{
|
|
2893
|
-
mutationKind: CounterfactualMutation['kind'];
|
|
2894
|
-
n: number;
|
|
2895
|
-
meanAbsDelta: number;
|
|
2896
|
-
meanSignedDelta: number;
|
|
2897
|
-
}>;
|
|
2898
|
-
|
|
2899
3101
|
/**
|
|
2900
3102
|
* Full cross-trace diff — align two trajectories step-by-step, report
|
|
2901
3103
|
* per-step score deltas, attribute a variant's total outcome lead to
|
|
@@ -2951,131 +3153,6 @@ interface CrossTraceDiffOptions {
|
|
|
2951
3153
|
}
|
|
2952
3154
|
declare function crossTraceDiff(store: TraceStore, runA: string, runB: string, options?: CrossTraceDiffOptions): Promise<CrossTraceDiff>;
|
|
2953
3155
|
|
|
2954
|
-
/**
|
|
2955
|
-
* Pre-registered hypotheses — declare what you're testing BEFORE the
|
|
2956
|
-
* run, check it AFTER. Prevents p-hacking, optional stopping, and the
|
|
2957
|
-
* "we ran until it looked good" failure mode.
|
|
2958
|
-
*
|
|
2959
|
-
* Manifest is a plain JSON-friendly object. Sign it with a content hash
|
|
2960
|
-
* + timestamp; the registered record becomes immutable. Post-run,
|
|
2961
|
-
* evaluate the manifest against observed results — the library refuses
|
|
2962
|
-
* to let you re-interpret a different metric as the declared one.
|
|
2963
|
-
*/
|
|
2964
|
-
interface HypothesisManifest {
|
|
2965
|
-
id: string;
|
|
2966
|
-
/** Human prose — goes into the audit trail. */
|
|
2967
|
-
hypothesis: string;
|
|
2968
|
-
/** Metric the hypothesis claims to move. */
|
|
2969
|
-
metric: string;
|
|
2970
|
-
/** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */
|
|
2971
|
-
direction: 'increase' | 'decrease';
|
|
2972
|
-
/** Minimum effect size to count (same units as the metric). */
|
|
2973
|
-
minEffect: number;
|
|
2974
|
-
/** Alpha threshold. */
|
|
2975
|
-
alpha: number;
|
|
2976
|
-
/** Target statistical power at which sample size was pre-computed. */
|
|
2977
|
-
power: number;
|
|
2978
|
-
/** Declared N per arm before running. */
|
|
2979
|
-
preRegisteredN: number;
|
|
2980
|
-
/** ISO8601 timestamp the manifest was registered. */
|
|
2981
|
-
registeredAt: string;
|
|
2982
|
-
/** Optional identifiers to tie into the trace corpus. */
|
|
2983
|
-
baselineLabel?: string;
|
|
2984
|
-
candidateLabel?: string;
|
|
2985
|
-
}
|
|
2986
|
-
/**
|
|
2987
|
-
* Identifier for the hashing scheme used to produce `contentHash`.
|
|
2988
|
-
*
|
|
2989
|
-
* `'sha256-content'` — sha256 hex over the canonicalized manifest with
|
|
2990
|
-
* the `contentHash` and `algo` fields stripped. Held as a string union
|
|
2991
|
-
* so future schemes can be added without breaking parsers; SignedManifest
|
|
2992
|
-
* values without `algo` deserialize cleanly because the field is optional.
|
|
2993
|
-
*/
|
|
2994
|
-
type SignedManifestAlgo = 'sha256-content';
|
|
2995
|
-
interface SignedManifest extends HypothesisManifest {
|
|
2996
|
-
/** sha256 hex of canonicalized manifest (everything except contentHash and algo). */
|
|
2997
|
-
contentHash: string;
|
|
2998
|
-
/**
|
|
2999
|
-
* Algorithm string describing how `contentHash` was produced.
|
|
3000
|
-
*
|
|
3001
|
-
* Optional on the type so serialized manifests without it still parse,
|
|
3002
|
-
* but ALWAYS populated by {@link signManifest}. Consumers that want to
|
|
3003
|
-
* enforce a known algorithm should reject manifests where this field
|
|
3004
|
-
* is missing or unrecognized.
|
|
3005
|
-
*/
|
|
3006
|
-
algo?: SignedManifestAlgo;
|
|
3007
|
-
}
|
|
3008
|
-
interface HypothesisResult {
|
|
3009
|
-
manifest: SignedManifest;
|
|
3010
|
-
observedN: number;
|
|
3011
|
-
observedEffect: number;
|
|
3012
|
-
observedPValue: number;
|
|
3013
|
-
/** True iff the observed effect hits the pre-declared direction with
|
|
3014
|
-
* magnitude ≥ minEffect AND p < alpha. */
|
|
3015
|
-
confirmed: boolean;
|
|
3016
|
-
/** Enumerated reasons the hypothesis was rejected (each a machine-tag). */
|
|
3017
|
-
rejectionReasons: Array<'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'>;
|
|
3018
|
-
notes?: string;
|
|
3019
|
-
}
|
|
3020
|
-
/**
|
|
3021
|
-
* Deterministic JSON canonicalization — sort object keys recursively.
|
|
3022
|
-
*
|
|
3023
|
-
* Two semantically-equal objects produce byte-identical canonicalized output;
|
|
3024
|
-
* this is what makes a content-hash stable across encoders, key insertion
|
|
3025
|
-
* orders, and runtime versions. Exported for any consumer that needs the same
|
|
3026
|
-
* canonicalization guarantee outside the manifest-signing path (e.g., signing
|
|
3027
|
-
* an artifact bundle, hashing a dataset version, etc.).
|
|
3028
|
-
*/
|
|
3029
|
-
declare function canonicalize(v: unknown): unknown;
|
|
3030
|
-
/**
|
|
3031
|
-
* SHA-256 hex (full 64 chars) over the canonicalized JSON encoding of `obj`.
|
|
3032
|
-
*
|
|
3033
|
-
* The same primitive `signManifest` and `verifyManifest` are built on, exposed
|
|
3034
|
-
* directly so consumers signing arbitrary structured content (artifact bundles,
|
|
3035
|
-
* production packets, dataset manifests, etc.) don't have to re-derive
|
|
3036
|
-
* canonicalize+sha256 from scratch.
|
|
3037
|
-
*
|
|
3038
|
-
* Stable across:
|
|
3039
|
-
* - object key insertion order (canonicalization sorts keys recursively)
|
|
3040
|
-
* - encoder choice (UTF-8 via TextEncoder, fixed)
|
|
3041
|
-
* - runtime (uses the Web Crypto subtle digest, present in Node ≥18 and browsers)
|
|
3042
|
-
*
|
|
3043
|
-
* Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,
|
|
3044
|
-
* which takes a string input and returns a truncated 12-char prompt id.
|
|
3045
|
-
* Use `hashJson` when you mean "canonicalize then hash."
|
|
3046
|
-
*
|
|
3047
|
-
* @example
|
|
3048
|
-
* const hash = await hashJson({ id: '1', kind: 'spec' })
|
|
3049
|
-
* // 'a3f1...' (64 hex chars)
|
|
3050
|
-
*/
|
|
3051
|
-
declare function hashJson<T>(obj: T): Promise<string>;
|
|
3052
|
-
/**
|
|
3053
|
-
* Sign a manifest with a SHA-256 content hash.
|
|
3054
|
-
*
|
|
3055
|
-
* The hash covers the canonicalized manifest with the `contentHash`
|
|
3056
|
-
* and `algo` fields stripped; this lets verifiers re-sign the rest and
|
|
3057
|
-
* compare. Returned manifest always carries `algo: 'sha256-content'`
|
|
3058
|
-
* so downstream consumers can identify the scheme; manifests without
|
|
3059
|
-
* `algo` still verify because it is stripped before hashing on both sides.
|
|
3060
|
-
*/
|
|
3061
|
-
declare function signManifest(m: HypothesisManifest): Promise<SignedManifest>;
|
|
3062
|
-
/**
|
|
3063
|
-
* Verify that a signed manifest has not been tampered with.
|
|
3064
|
-
*
|
|
3065
|
-
* Strips `contentHash` and `algo` before re-signing so manifests without
|
|
3066
|
-
* `algo` verify identically to ones that carry it.
|
|
3067
|
-
*/
|
|
3068
|
-
declare function verifyManifest(m: SignedManifest): Promise<boolean>;
|
|
3069
|
-
/**
|
|
3070
|
-
* Evaluate a pre-registered hypothesis against observed results.
|
|
3071
|
-
* Mechanical — no re-interpretation permitted.
|
|
3072
|
-
*/
|
|
3073
|
-
declare function evaluateHypothesis(manifest: SignedManifest, observed: {
|
|
3074
|
-
n: number;
|
|
3075
|
-
effect: number;
|
|
3076
|
-
pValue: number;
|
|
3077
|
-
}): Promise<HypothesisResult>;
|
|
3078
|
-
|
|
3079
3156
|
/**
|
|
3080
3157
|
* Active learning — agent-as-scenario-author.
|
|
3081
3158
|
*
|
|
@@ -4502,76 +4579,6 @@ declare function precision<T>(goldens: GoldenSpec[], candidates: T[], options?:
|
|
|
4502
4579
|
text?: (candidate: T) => string;
|
|
4503
4580
|
}): number;
|
|
4504
4581
|
|
|
4505
|
-
/**
|
|
4506
|
-
* Wrap a single judge LLM call with retry, optional fallback-model
|
|
4507
|
-
* rotation, exponential backoff, and a typed `JudgeRetryOutcome`. Callers
|
|
4508
|
-
* MUST inspect `succeeded` before using `value`; on failure the library
|
|
4509
|
-
* returns `value: null` rather than substituting a default, so a judge
|
|
4510
|
-
* abort cannot silently corrupt a downstream composite.
|
|
4511
|
-
*
|
|
4512
|
-
* Reporting contract: callers ship `TrialResult.judgeSucceeded = succeeded`
|
|
4513
|
-
* and `TrialResult.judgeAttempts = attempts` so `aggregateTrialsByMode`
|
|
4514
|
-
* with `mode: 'exclude-failed'` drops the trial.
|
|
4515
|
-
*/
|
|
4516
|
-
/** Retry policy for judge LLM calls. */
|
|
4517
|
-
interface JudgeRetryPolicy {
|
|
4518
|
-
/** Max attempts per model. Default 3 (one initial + two retries). */
|
|
4519
|
-
maxAttempts?: number;
|
|
4520
|
-
/** Per-attempt timeout in ms. Default 300_000. */
|
|
4521
|
-
timeoutMs?: number;
|
|
4522
|
-
/**
|
|
4523
|
-
* Models to try, in order. The first model is the primary; subsequent
|
|
4524
|
-
* models are fallbacks invoked only when ALL retries on the previous
|
|
4525
|
-
* model have been exhausted. Example: `['claude-code/sonnet', 'kimi-code/k2p6']`
|
|
4526
|
-
* runs claude-code up to maxAttempts times, then falls back to kimi.
|
|
4527
|
-
* If omitted, the caller's judge function controls model selection and
|
|
4528
|
-
* the retries apply to that single model.
|
|
4529
|
-
*/
|
|
4530
|
-
models?: readonly string[];
|
|
4531
|
-
/** Exponential backoff function, default `attempt → min(500 * 2^attempt, 16_000)`. */
|
|
4532
|
-
backoffMs?: (attempt: number) => number;
|
|
4533
|
-
/**
|
|
4534
|
-
* Predicate deciding whether an error should trigger a retry. Defaults to
|
|
4535
|
-
* `isTransientLlmError` — the package-wide classifier shared with
|
|
4536
|
-
* `callLlm` — which retries aborts/timeouts, network faults, HTTP/2
|
|
4537
|
-
* transport faults, and any `LlmCallError` with status in {429,502,503,504}.
|
|
4538
|
-
* JSON-parse and schema-rejection errors are NOT retriable (the model
|
|
4539
|
-
* needs prompt adjustment, not another shot).
|
|
4540
|
-
*/
|
|
4541
|
-
isRetryable?: (err: unknown) => boolean;
|
|
4542
|
-
}
|
|
4543
|
-
/** Outcome of a wrapped judge invocation. */
|
|
4544
|
-
interface JudgeRetryOutcome<T> {
|
|
4545
|
-
/** The judge's returned value when `succeeded === true`. */
|
|
4546
|
-
value: T | null;
|
|
4547
|
-
/** True iff one of the attempts completed without throwing. */
|
|
4548
|
-
succeeded: boolean;
|
|
4549
|
-
/** Total attempts made across all models. */
|
|
4550
|
-
attempts: number;
|
|
4551
|
-
/** Which model the successful attempt used (when succeeded). */
|
|
4552
|
-
modelUsed?: string;
|
|
4553
|
-
/** Last error captured when `succeeded === false`. */
|
|
4554
|
-
error?: Error;
|
|
4555
|
-
/** Per-attempt error log for forensics. */
|
|
4556
|
-
attemptErrors: Array<{
|
|
4557
|
-
attempt: number;
|
|
4558
|
-
model: string;
|
|
4559
|
-
error: string;
|
|
4560
|
-
}>;
|
|
4561
|
-
}
|
|
4562
|
-
/**
|
|
4563
|
-
* Wrap a judge call with retry + fallback-model + typed outcome semantics.
|
|
4564
|
-
*
|
|
4565
|
-
* The `judgeFn` signature is `(model: string, signal: AbortSignal) => Promise<T>`.
|
|
4566
|
-
* The signal will be aborted at `timeoutMs`. Callers should pass the signal
|
|
4567
|
-
* to their underlying fetch/SDK call so the abort actually fires.
|
|
4568
|
-
*
|
|
4569
|
-
* Returns a typed outcome — callers MUST inspect `succeeded` before using
|
|
4570
|
-
* `value`. The library refuses to default to a silent zero score because a
|
|
4571
|
-
* synthetic zero is indistinguishable from a real low score downstream.
|
|
4572
|
-
*/
|
|
4573
|
-
declare function withJudgeRetry<T>(judgeFn: (model: string, signal: AbortSignal) => Promise<T>, policy?: JudgeRetryPolicy): Promise<JudgeRetryOutcome<T>>;
|
|
4574
|
-
|
|
4575
4582
|
/**
|
|
4576
4583
|
* LockedJsonlAppender — mutex-serialized JSONL append helper for arbitrary
|
|
4577
4584
|
* payloads. The reference-replay store does the same thing for typed
|
|
@@ -5268,4 +5275,259 @@ declare namespace index {
|
|
|
5268
5275
|
export { type index_AgentProfile as AgentProfile, type index_AgentProfileSection as AgentProfileSection, index_BASELINE_ROLES as BASELINE_ROLES, type index_BaselineRoleKey as BaselineRoleKey, type index_ProfileSkill as ProfileSkill, index_applyDomainPatch as applyDomainPatch, index_baselineProfile as baselineProfile, index_baselineProfileFromRole as baselineProfileFromRole, index_engineerRole as engineerRole, index_generalistRole as generalistRole, index_prodProfile as prodProfile, index_profileToSurface as profileToSurface, index_renderProfile as renderProfile, index_researcherRole as researcherRole, index_sectionHash as sectionHash };
|
|
5269
5276
|
}
|
|
5270
5277
|
|
|
5271
|
-
export { type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, AgentProfile$1 as AgentProfile, type AgreementResult, type AlignmentOp, AnalyzeTracesInput, AnalyzeTracesOptions, AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, type BackendDescriptor, BaselineReport, BehaviorAssertion, BenchmarkReport, BenchmarkRunner, BenchmarkRunnerConfig, type BisectOptions, type BisectResult, type BisectStep, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, BudgetSpec, type BuildAgreementJudgeOptions, CallExpectation, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CausalAttributionReport, type CellVerdict, type ChannelRollup, ChatRequest, CheckResult, CollectedArtifacts, type CommandRunner, type CompareLabels, CompletionCriterion, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContractMetric, type ContractReport, ConvergenceTracker, type CostChannel, type CostEntry, CostLedger, type CostLedgerEntry, type CostLedgerSummary, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, DEFAULT_AGENT_SLOS, DEFAULT_FINDERS, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, Dataset, DatasetScenario, type DecideNextUserTurnOpts, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, DriverResult, DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type ErrorCountPattern, EvalTraceStore, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentProvenance, type ExperimentRep, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type FactorContribution, type FactorialCell, FeedbackLabel, FeedbackTrajectory, FeedbackTrajectoryStore, type FieldAgreementSpec, type FileChange, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenSeverity, type GoldenSpec, HarnessConfig, type HeldOutPartition, HoldoutAuditor, type HttpGithubClientOptions, type HypothesisManifest, type HypothesisResult, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, type JudgeFamily, type JudgeFleetOptions, JudgeFn, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, JudgeRunner, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, Layer, LayerResult, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmClientOptions, LlmSpan, LockedJsonlAppender, MODEL_PRICING, type MatchResult, type MatcherResult, type MergeOptions, MetricsCollector, type ModelPreflight, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, type MultiToolchainLayerConfig, type Mutator, Mutex, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParseStudentLabel, type PartitionHeldOutOptions, PersonaConfig, type Playbook, type PlaybookEntry, type PoolSlot, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreflightModelsOptions, type PreflightOutcome, ProductClient, ProductClientConfig, type PromptHandle, PromptRegistry, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProvenanceReader, type RecordRunsOptions, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, ReleaseConfidenceScorecard, ReleaseConfidenceThresholds, type RenderStudentPrompt, type RepoRef, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RobustnessResult, Run, type RunCommandInput, type RunCommandResult, type RunDistillationOptions, type RunDistillationResult, RunFilter, RunRecord, type RunRecordBackend, type RunRecordFilter, RunScore, RunScoreWeights, RunSplitTag, SandboxDriver, SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type ScanOptions, Scenario, type ScenarioCost, ScenarioFile, ScenarioRegistry, ScenarioResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SeriesConvergenceOptions, type SeriesConvergenceResult, Severity, type SignedManifest, type SignedManifestAlgo, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SplitGoldOptions, SteeringBundle, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type StepAttribution, type SynthesisReason, type SynthesisTarget, TestResult, type ThresholdContract, TokenCounter, type TokenSpec, TraceEmitter, TraceStore, type TracedAnalystOptions, type TracedJudgeOptions, Trajectory, TrajectoryStep, type TrialTrace, TurnMetrics, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, VerifyContext, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, adversarialJudge, aggregateJudgeVerdicts, aggregatePrReviewScore, analyzeAntiSlop, analyzeSeries, appendScorecard, assertCrossFamily, assertModelsServed, assertSingleBackend, assignHeldOutTag, attributeCounterfactuals, bisect, buildAgreementJudge, buildDriverSystemPrompt, buildReflectionPrompt, buildReviewerPrompt, canaryLeakView, canonicalize, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, codeExecutionJudge, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, compareReferenceReplay, compilerJudge, computeExperimentStats, costForUsage, createAntiSlopJudge, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createIntentMatchJudge, createSandboxPool, crossTraceDiff, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultJudges, defaultParseStudentLabel, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, deployGateLayer, diffScorecard, discoverPersonas, distillPlaybook, estimateCost, estimateTokens, evaluateContract, evaluateHypothesis, evaluateOracles, executeScenario, expectAgent, exportRewardModel, extractAssetUrls, extractErrorCount, fieldAgreement, fileContains, fileExists, fileExperimentStore, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findSkipCountsAsPass, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, ghCliClient, gitProvenanceReader, precision as goldenPrecision, hashContent, hashJson, hashToUnit, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryRunRecordBackend, isModelPriced, isOtelConfigured, jsonShape, jsonlReferenceReplayStore, jsonlRunRecordBackend, judgeFamily, keyPreserved, linterJudge, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, matchGoldens, mergeLayerResults, modelDescriptionBits, modelPriceKey, multiToolchainLayer, notBlocked, paraphraseRobustness, paraphraseRobustnessScenarios, parseGoldJsonl, parseReflectionResponse, partitionHeldOut, passOrthogonality, pixelDeltaRatio, politenessPrefixMutator, preflightModels, printDriverSummary, index as profile, promptBisect, proposeAutomatedPullRequest, proposeSynthesisTargets, recordRuns, recordRunsToScorecard, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, replayScorerOverCorpus, replayTraceThroughJudge, resetLockedAppendersForTesting, resolveModelPricing, rowCount, rowWhere, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runDistillation, runE2EWorkflow, runExpectations, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runReferenceReplay, runScore, runSelfPlay, scanForMuffledGates, scoreContinuity, scorePrReviewComments, scorePrReviewSource, scoreReferenceReplay, securityJudge, sentenceReorderMutator, signManifest, splitGold, statusAdvanced, summarizePrReviewBenchmark, testJudge, textInSnapshot, toLangfuseEnvelope, toPrometheusText, traceJudge, traceJudgeEnsemble, tracedAnalyzeTraces, typoMutator, urlContains, verifyManifest, visualDiff, viteDeployRunner, weightedRecall, whitespaceCollapseMutator, withJudgeRetry, withOtelPipeline, wranglerDeployRunner };
|
|
5278
|
+
/**
|
|
5279
|
+
* Program cost report — a thin projection over `CostLedger.summary()` that
|
|
5280
|
+
* adds the per-model rollup the summary lacks, plus `attachCostToReport`, the
|
|
5281
|
+
* one way every artifact (capsules, campaign results, diagnose reports) gets
|
|
5282
|
+
* its cost stamp.
|
|
5283
|
+
*
|
|
5284
|
+
* Honesty contract carried through from the ledger: `total.unknownEntries`
|
|
5285
|
+
* and `perModel[].unpriced` surface the costUnknown axis — a $0 from an
|
|
5286
|
+
* unpriced model is a lower bound, never a measured zero.
|
|
5287
|
+
*/
|
|
5288
|
+
|
|
5289
|
+
interface ModelCostRollup {
|
|
5290
|
+
model: string;
|
|
5291
|
+
usd: number;
|
|
5292
|
+
entries: number;
|
|
5293
|
+
/** ≥1 entry for this model was costUnknown — `usd` is a lower bound. An
|
|
5294
|
+
* `actualCostUsd` override clears the flag for that entry (the dollars are
|
|
5295
|
+
* observed, even when the model has no pricing). */
|
|
5296
|
+
unpriced: boolean;
|
|
5297
|
+
}
|
|
5298
|
+
interface CostReport {
|
|
5299
|
+
/** Per-channel breakdown — `CostLedgerSummary.byChannel` verbatim. */
|
|
5300
|
+
perChannel: ChannelRollup[];
|
|
5301
|
+
total: {
|
|
5302
|
+
usd: number;
|
|
5303
|
+
/** Entries whose cost was unknown — non-zero means `usd` is a lower bound. */
|
|
5304
|
+
unknownEntries: number;
|
|
5305
|
+
};
|
|
5306
|
+
/** Per-model spend, sorted by model id. */
|
|
5307
|
+
perModel: ModelCostRollup[];
|
|
5308
|
+
}
|
|
5309
|
+
/** Project a ledger into the program cost report. Pure — no I/O, no clock. */
|
|
5310
|
+
declare function costReport(ledger: CostLedger): CostReport;
|
|
5311
|
+
/**
|
|
5312
|
+
* Stamp a report-shaped object with its cost projection under the `cost` key.
|
|
5313
|
+
* Generic so capsules, campaign results, and diagnose reports all stamp the
|
|
5314
|
+
* same way. Throws when the report already carries a `cost` key — silently
|
|
5315
|
+
* overwriting an existing stamp would corrupt the artifact's provenance.
|
|
5316
|
+
*/
|
|
5317
|
+
declare function attachCostToReport<R extends object>(report: R, ledger: CostLedger): R & {
|
|
5318
|
+
cost: CostReport;
|
|
5319
|
+
};
|
|
5320
|
+
|
|
5321
|
+
/**
|
|
5322
|
+
* ModelSeats — the program's model seating chart.
|
|
5323
|
+
*
|
|
5324
|
+
* One object names which model fills each role in an eval program: the worker
|
|
5325
|
+
* under evaluation, the judge panel, the analyst, the reflection/driver model,
|
|
5326
|
+
* and the verifier. Re-tiering an entire program (economy ↔ frontier) is one
|
|
5327
|
+
* swapped object instead of a hunt through call sites.
|
|
5328
|
+
*
|
|
5329
|
+
* Wiring points — consumers thread seats; this module implements none of them
|
|
5330
|
+
* (those files belong to other surfaces):
|
|
5331
|
+
* - `judges` → `ensembleJudge({ models: seats.judges, … })` (src/judge-panel.ts)
|
|
5332
|
+
* and the `JudgeConfig`s handed to `makeEvalTools({ judges })`
|
|
5333
|
+
* (src/eval-tools.ts).
|
|
5334
|
+
* - `reflection` → `selfImprove({ llm: { model: seats.reflection } })` — the
|
|
5335
|
+
* `gepaDriver` reflection model (src/contract/self-improve.ts);
|
|
5336
|
+
* same seat for any custom `ImprovementDriver`'s LLM.
|
|
5337
|
+
* - `worker` → the dispatch model the agent itself calls — the model an
|
|
5338
|
+
* `AgentProfile` declares.
|
|
5339
|
+
* - `analyst` → the LLM behind `analyzeRuns` / analyst-registry kinds.
|
|
5340
|
+
* - `verifier` → completion-verifier / objective-checker model.
|
|
5341
|
+
* - campaign cells thread `judges` + driver models the same way; that wiring
|
|
5342
|
+
* lands with the campaign surface, not here.
|
|
5343
|
+
*
|
|
5344
|
+
* `resolveSeat` is the only read path: an unset seat with no explicit fallback
|
|
5345
|
+
* throws — a model id is a budget decision, never a silent default.
|
|
5346
|
+
*/
|
|
5347
|
+
|
|
5348
|
+
interface ModelSeats {
|
|
5349
|
+
/** The model under evaluation — what the agent itself dispatches with. */
|
|
5350
|
+
worker?: string;
|
|
5351
|
+
/** Judge-panel model ids — thread into `ensembleJudge({ models })`. */
|
|
5352
|
+
judges?: string[];
|
|
5353
|
+
/** Analyst model — `analyzeRuns` / analyst-registry LLM calls. */
|
|
5354
|
+
analyst?: string;
|
|
5355
|
+
/** Reflection/driver model — `gepaDriver` mutation proposals. */
|
|
5356
|
+
reflection?: string;
|
|
5357
|
+
/** Verifier model — completion/objective checking. */
|
|
5358
|
+
verifier?: string;
|
|
5359
|
+
}
|
|
5360
|
+
type SeatName = keyof ModelSeats;
|
|
5361
|
+
type SeatPresetName = keyof typeof seatPresets;
|
|
5362
|
+
/**
|
|
5363
|
+
* Tier presets — plain data, swap or spread freely.
|
|
5364
|
+
*
|
|
5365
|
+
* `economy` uses the fleet-policy ids: every id resolves through the
|
|
5366
|
+
* substrate's family pricing (no costUnknown axis) and the judge trio spans
|
|
5367
|
+
* three provider families (moonshot / deepseek / openai), so it passes
|
|
5368
|
+
* `assertCrossFamily` as-is.
|
|
5369
|
+
*
|
|
5370
|
+
* `frontier` is deliberately EMPTY: entitled frontier ids vary per router
|
|
5371
|
+
* account, and a hardcoded claude/gpt-5 id 401s on keys that lack it. Supply
|
|
5372
|
+
* your own: `{ ...seatPresets.frontier, worker: '<your-frontier-id>', … }` —
|
|
5373
|
+
* `resolveSeat` throws on every seat you haven't filled.
|
|
5374
|
+
*/
|
|
5375
|
+
declare const seatPresets: Record<'economy' | 'frontier', ModelSeats>;
|
|
5376
|
+
/** Thrown by `resolveSeat` when a seat is unset and no fallback was given. */
|
|
5377
|
+
declare class SeatUnsetError extends ConfigError {
|
|
5378
|
+
readonly seat: SeatName;
|
|
5379
|
+
constructor(seat: SeatName);
|
|
5380
|
+
}
|
|
5381
|
+
/**
|
|
5382
|
+
* Read one seat. Blank strings and empty arrays count as unset (env-var
|
|
5383
|
+
* plumbing produces them); malformed values (non-string seat, non-array or
|
|
5384
|
+
* blank-entry `judges`) throw `ValidationError`. When the seat is unset, an
|
|
5385
|
+
* explicit `fallback` is returned (`[fallback]` for `judges` — a one-model
|
|
5386
|
+
* panel); without one, `SeatUnsetError`.
|
|
5387
|
+
*/
|
|
5388
|
+
declare function resolveSeat(seats: ModelSeats, seat: 'judges', fallback?: string): string[];
|
|
5389
|
+
declare function resolveSeat(seats: ModelSeats, seat: Exclude<SeatName, 'judges'>, fallback?: string): string;
|
|
5390
|
+
declare function resolveSeat(seats: ModelSeats, seat: SeatName, fallback?: string): string | string[];
|
|
5391
|
+
|
|
5392
|
+
/**
|
|
5393
|
+
* Reproducibility attestation for any serializable report object.
|
|
5394
|
+
*
|
|
5395
|
+
* `attest()` binds a report to its content address (sha-256 over canonical
|
|
5396
|
+
* JSON) plus the provenance needed to reproduce it: model versions, seeds,
|
|
5397
|
+
* price-table hash, code SHA, inputs hash. `verifyAttestation()` recomputes
|
|
5398
|
+
* the address and answers "is this the exact report that provenance
|
|
5399
|
+
* describes?" — any single-field tamper changes the hash.
|
|
5400
|
+
*
|
|
5401
|
+
* Layering: content-addressing is the substrate's job; cryptographic SIGNING
|
|
5402
|
+
* (who vouches for the attestation, key management, transparency logs) is the
|
|
5403
|
+
* consumer's layer on top. An `AttestedReport` is a stable byte-identical
|
|
5404
|
+
* payload a consumer can sign — the substrate never holds keys.
|
|
5405
|
+
*
|
|
5406
|
+
* Generic by design: the report parameter is ANY value `canonicalJson`
|
|
5407
|
+
* accepts (campaign results, fuzz capsules, scorecards, cost ledgers). Do not
|
|
5408
|
+
* couple this module to a specific report schema.
|
|
5409
|
+
*/
|
|
5410
|
+
/** Hash scheme identifier carried by every attestation. A verifier rejects
|
|
5411
|
+
* unknown algorithms instead of guessing. */
|
|
5412
|
+
declare const ATTESTATION_ALGORITHM: "sha256/canonical-json";
|
|
5413
|
+
interface AttestationProvenance {
|
|
5414
|
+
/** Every model involved in producing the report, name → version/id. */
|
|
5415
|
+
modelVersions: Record<string, string>;
|
|
5416
|
+
/** RNG seeds the run was driven by, when seeded. */
|
|
5417
|
+
seeds?: number[];
|
|
5418
|
+
/** Content hash of the price table used for cost figures — cost numbers
|
|
5419
|
+
* are only reproducible against the same prices. */
|
|
5420
|
+
priceTableHash?: string;
|
|
5421
|
+
/** Git SHA of the code that produced the report. */
|
|
5422
|
+
codeSha: string;
|
|
5423
|
+
/** Content hash of the input set (scenarios, dataset manifest, ...). */
|
|
5424
|
+
inputsHash?: string;
|
|
5425
|
+
/** ISO-8601 timestamp, caller-supplied — the substrate stays clock-free
|
|
5426
|
+
* so attestation is deterministic and testable. */
|
|
5427
|
+
createdAt: string;
|
|
5428
|
+
}
|
|
5429
|
+
interface AttestedReport {
|
|
5430
|
+
/** Hex sha-256 over the canonical JSON of the report. */
|
|
5431
|
+
reportHash: string;
|
|
5432
|
+
provenance: AttestationProvenance;
|
|
5433
|
+
algorithm: typeof ATTESTATION_ALGORITHM;
|
|
5434
|
+
}
|
|
5435
|
+
interface AttestationVerification {
|
|
5436
|
+
valid: boolean;
|
|
5437
|
+
/** Populated iff `valid` is false — names the exact mismatch. */
|
|
5438
|
+
reason?: string;
|
|
5439
|
+
}
|
|
5440
|
+
/**
|
|
5441
|
+
* Content-address a report and bind it to its provenance. Throws (via
|
|
5442
|
+
* `canonicalJson`) if the report contains undefined / function / symbol /
|
|
5443
|
+
* non-finite numbers — a report that cannot be unambiguously serialized
|
|
5444
|
+
* cannot be attested.
|
|
5445
|
+
*/
|
|
5446
|
+
declare function attest(report: unknown, provenance: AttestationProvenance): AttestedReport;
|
|
5447
|
+
/**
|
|
5448
|
+
* Verify a report against its attestation. Returns a typed outcome rather
|
|
5449
|
+
* than throwing: an unverifiable report (e.g. one that no longer
|
|
5450
|
+
* canonicalizes) is a verification failure with the cause in `reason`, not a
|
|
5451
|
+
* crash — verifiers run in pipelines that must record WHY, not die.
|
|
5452
|
+
*/
|
|
5453
|
+
declare function verifyAttestation(report: unknown, attested: AttestedReport): AttestationVerification;
|
|
5454
|
+
|
|
5455
|
+
/**
|
|
5456
|
+
* Content-addressed judge-verdict caching.
|
|
5457
|
+
*
|
|
5458
|
+
* LAW: cache JUDGE VERDICTS only — judging the same artifact with the same
|
|
5459
|
+
* judge+rubric is pure. NEVER cache agent rollouts. (A router that cached
|
|
5460
|
+
* identical fanout prompts silently destroyed best-of-N diversity; rollout
|
|
5461
|
+
* caching reintroduces that failure class. Judging has no diversity to
|
|
5462
|
+
* destroy — same artifact + same rubric ⇒ same verdict is the desired
|
|
5463
|
+
* property, not a bug.)
|
|
5464
|
+
*
|
|
5465
|
+
* The cache key is a sha-256 over the canonical JSON of everything that can
|
|
5466
|
+
* change a verdict: the artifact content, the scenario id, the judge name,
|
|
5467
|
+
* the full dimension list (key + description — the description IS the rubric
|
|
5468
|
+
* text shown to the judge), and a caller-supplied `judgeVersion`.
|
|
5469
|
+
* `judgeVersion` is REQUIRED: a judge whose prompt/model/ensemble changes
|
|
5470
|
+
* without a version bump would otherwise silently serve stale verdicts.
|
|
5471
|
+
*
|
|
5472
|
+
* Strict canonicalization (`canonicalJson`) throws on undefined / function /
|
|
5473
|
+
* symbol / non-finite numbers — an artifact that cannot be unambiguously
|
|
5474
|
+
* serialized cannot be content-addressed, and coercing it would let two
|
|
5475
|
+
* different artifacts collide on one key.
|
|
5476
|
+
*/
|
|
5477
|
+
|
|
5478
|
+
/**
|
|
5479
|
+
* Stable JSON stringify: object keys sorted recursively, so two semantically
|
|
5480
|
+
* equal values produce byte-identical output regardless of key insertion
|
|
5481
|
+
* order. Throws on undefined / function / symbol / NaN / ±Infinity / bigint /
|
|
5482
|
+
* Map / Set — anything JSON.stringify would coerce or drop silently.
|
|
5483
|
+
*
|
|
5484
|
+
* Distinct from `pre-registration.ts`'s `canonicalize`/`hashJson`, which are
|
|
5485
|
+
* permissive (coercion allowed) and async (web-crypto). Use THIS pair when a
|
|
5486
|
+
* hash collision or silent coercion would corrupt a cache key or attestation.
|
|
5487
|
+
*/
|
|
5488
|
+
declare function canonicalJson(value: unknown): string;
|
|
5489
|
+
/** Hex sha-256 over `canonicalJson(value)`. The content address used by the
|
|
5490
|
+
* verdict cache and report attestation. */
|
|
5491
|
+
declare function contentHash(value: unknown): string;
|
|
5492
|
+
/** Pluggable verdict store. Sync or async on both legs — `cachedJudge`
|
|
5493
|
+
* awaits the results either way. */
|
|
5494
|
+
interface VerdictCacheStore {
|
|
5495
|
+
get(key: string): Promise<JudgeScore | undefined> | JudgeScore | undefined;
|
|
5496
|
+
set(key: string, score: JudgeScore): Promise<void> | void;
|
|
5497
|
+
}
|
|
5498
|
+
/** Process-local Map-backed store. */
|
|
5499
|
+
declare function inMemoryVerdictCache(): VerdictCacheStore;
|
|
5500
|
+
/**
|
|
5501
|
+
* JSONL-file-backed store: the full file is loaded into an in-memory index at
|
|
5502
|
+
* construction; every `set` appends one line synchronously (durable before
|
|
5503
|
+
* the verdict is returned). A corrupt or malformed line throws at load with
|
|
5504
|
+
* file:line — a skipped line would silently re-judge (cost) or, worse, mask
|
|
5505
|
+
* a half-written file that needs operator attention.
|
|
5506
|
+
*/
|
|
5507
|
+
declare function fileVerdictCache(path: string): VerdictCacheStore;
|
|
5508
|
+
interface VerdictCacheStats {
|
|
5509
|
+
hits: number;
|
|
5510
|
+
misses: number;
|
|
5511
|
+
}
|
|
5512
|
+
interface CachedJudgeOptions {
|
|
5513
|
+
/** REQUIRED — part of the cache key. Bump on any change to the judge's
|
|
5514
|
+
* prompt, model, ensemble, or scoring logic; silent judge upgrades must
|
|
5515
|
+
* never serve stale verdicts. */
|
|
5516
|
+
judgeVersion: string;
|
|
5517
|
+
}
|
|
5518
|
+
/** The wrapped judge: same `JudgeConfig` seam, plus hit/miss observability. */
|
|
5519
|
+
type CachedJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1> = JudgeConfig<TArtifact, TScenario> & {
|
|
5520
|
+
stats(): VerdictCacheStats;
|
|
5521
|
+
};
|
|
5522
|
+
/**
|
|
5523
|
+
* Wrap a `JudgeConfig` so repeat judgments of the same artifact are served
|
|
5524
|
+
* from the store instead of re-invoking `score()`. The wrapper is generic
|
|
5525
|
+
* over the judge's own type parameters and preserves `appliesTo` — it is a
|
|
5526
|
+
* drop-in replacement anywhere a `JudgeConfig` is accepted.
|
|
5527
|
+
*
|
|
5528
|
+
* A judge that throws is NOT cached: the error propagates and the next
|
|
5529
|
+
* attempt re-judges (caching a failure would pin a transient outage forever).
|
|
5530
|
+
*/
|
|
5531
|
+
declare function cachedJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1>(judge: JudgeConfig<TArtifact, TScenario>, store: VerdictCacheStore, options: CachedJudgeOptions): CachedJudge<TArtifact, TScenario>;
|
|
5532
|
+
|
|
5533
|
+
export { ATTESTATION_ALGORITHM, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, AgentProfile$1 as AgentProfile, type AgreementResult, type AlignmentOp, AnalyzeTracesInput, AnalyzeTracesOptions, AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, type BackendDescriptor, BaselineReport, BehaviorAssertion, BenchmarkReport, BenchmarkRunner, BenchmarkRunnerConfig, type BisectOptions, type BisectResult, type BisectStep, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, BudgetSpec, type BuildAgreementJudgeOptions, type CachedJudge, type CachedJudgeOptions, CallExpectation, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CausalAttributionReport, type CellVerdict, ChannelRollup, ChatRequest, CheckResult, CollectedArtifacts, type CommandRunner, type CompareLabels, CompletionCriterion, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, ConvergenceTracker, CorrectnessChecker, type CostEntry, CostLedger, type CostReport, type CostSummary, CostTracker, CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, DEFAULT_AGENT_SLOS, DEFAULT_FINDERS, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, Dataset, DatasetScenario, type DecideNextUserTurnOpts, DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, DriverResult, DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCountPattern, type EvalToolDef, EvalTraceStore, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentProvenance, type ExperimentRep, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type FactorContribution, type FactorialCell, FeedbackLabel, FeedbackTrajectory, FeedbackTrajectoryStore, type FieldAgreementSpec, type FileChange, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenSeverity, type GoldenSpec, HarnessConfig, type HeldOutPartition, HoldoutAuditor, type HttpGithubClientOptions, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, JudgeError, type JudgeFamily, type JudgeFleetOptions, JudgeFn, JudgeParseError, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, JudgeRunner, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, Layer, LayerResult, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmClientOptions, LlmSpan, LockedJsonlAppender, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatcherResult, type MergeOptions, MetricsCollector, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, type MultiToolchainLayerConfig, type Mutator, Mutex, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParseStudentLabel, type PartitionHeldOutOptions, PersonaConfig, type Playbook, type PlaybookEntry, type PoolSlot, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreflightModelsOptions, type PreflightOutcome, ProductClient, ProductClientConfig, type PromptHandle, PromptRegistry, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProvenanceReader, type RecordRunsOptions, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, ReleaseConfidenceScorecard, ReleaseConfidenceThresholds, type RenderStudentPrompt, type RepoRef, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RobustnessResult, Run, type RunCommandInput, type RunCommandResult, type RunDistillationOptions, type RunDistillationResult, RunFilter, RunRecord, type RunRecordBackend, type RunRecordFilter, RunScore, RunScoreWeights, RunSplitTag, SandboxDriver, SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type ScanOptions, Scenario, type ScenarioCost, ScenarioFile, ScenarioRegistry, ScenarioResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SerializedRegex, Severity, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SpanPredicate, type SplitGoldOptions, SteeringBundle, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type StepAttribution, type SynthesisReason, type SynthesisTarget, TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type TraceContract, TraceContractBuilder, TraceEmitter, TraceStore, type TracedAnalystOptions, type TracedJudgeOptions, Trajectory, TrajectoryStep, type TrialTrace, TurnMetrics, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type VerdictCacheStats, type VerdictCacheStore, VerifyContext, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, adversarialJudge, aggregateJudgeVerdicts, aggregatePrReviewScore, analyzeAntiSlop, appendScorecard, assertCrossFamily, assertModelsServed, assertSingleBackend, assignHeldOutTag, attachCostToReport, attest, bisect, buildAgreementJudge, buildDriverSystemPrompt, buildReflectionPrompt, buildReviewerPrompt, cachedJudge, canaryLeakView, canonicalJson, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, codeExecutionJudge, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, compareReferenceReplay, compilerJudge, computeExperimentStats, contentHash, contractJudge, costReport, createAntiSlopJudge, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createIntentMatchJudge, createSandboxPool, crossTraceDiff, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultJudges, defaultParseStudentLabel, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, deployGateLayer, diffScorecard, discoverPersonas, distillPlaybook, ensembleJudge, estimateCost, estimateTokens, evaluateContract, evaluateOracles, evaluateTraceContract, executeScenario, expectAgent, exportRewardModel, extractAssetUrls, extractErrorCount, fieldAgreement, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findSkipCountsAsPass, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, ghCliClient, gitProvenanceReader, precision as goldenPrecision, hashContent, hashToUnit, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryRunRecordBackend, inMemoryVerdictCache, isModelPriced, isOtelConfigured, jsonShape, jsonlReferenceReplayStore, jsonlRunRecordBackend, judgeFamily, keyPreserved, linterJudge, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, matchGoldens, matchSpan, mergeLayerResults, modelDescriptionBits, multiToolchainLayer, notBlocked, paraphraseRobustness, paraphraseRobustnessScenarios, parseGoldJsonl, parseReflectionResponse, partitionHeldOut, passOrthogonality, pixelDeltaRatio, politenessPrefixMutator, preflightModels, printDriverSummary, index as profile, promptBisect, proposeAutomatedPullRequest, proposeSynthesisTargets, recordRuns, recordRunsToScorecard, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, replayScorerOverCorpus, replayTraceThroughJudge, resetLockedAppendersForTesting, resolveModelPricing, resolveSeat, rowCount, rowWhere, runAssertions, runBehavioralCanaries, runCanaries, runDistillation, runE2EWorkflow, runExpectations, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runReferenceReplay, runScore, runSelfPlay, scanForMuffledGates, scoreContinuity, scorePrReviewComments, scorePrReviewSource, scoreReferenceReplay, seatPresets, securityJudge, sentenceReorderMutator, splitGold, statusAdvanced, summarizePrReviewBenchmark, testJudge, textInSnapshot, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, traceContract, traceJudge, traceJudgeEnsemble, tracedAnalyzeTraces, typoMutator, urlContains, verifyAttestation, visualDiff, viteDeployRunner, weightedRecall, whitespaceCollapseMutator, withJudgeRetry, withOtelPipeline, wranglerDeployRunner };
|