@tangle-network/agent-eval 0.86.0 → 0.89.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (134) hide show
  1. package/dist/adapters/http.d.ts +3 -3
  2. package/dist/adapters/langchain.d.ts +3 -3
  3. package/dist/adapters/otel.d.ts +6 -6
  4. package/dist/adversarial-DIVcDoI_.d.ts +88 -0
  5. package/dist/analyst/index.d.ts +11 -10
  6. package/dist/analyst/index.js +13 -8
  7. package/dist/analyst/index.js.map +1 -1
  8. package/dist/analyze-runs-DwCEkpO_.d.ts +81 -0
  9. package/dist/belief-state/index.d.ts +4 -4
  10. package/dist/belief-state/index.js +1 -1
  11. package/dist/benchmarks/index.d.ts +3 -3
  12. package/dist/campaign/index.d.ts +165 -18
  13. package/dist/campaign/index.js +289 -14
  14. package/dist/campaign/index.js.map +1 -1
  15. package/dist/chunk-45EEMHTC.js +35 -0
  16. package/dist/chunk-45EEMHTC.js.map +1 -0
  17. package/dist/{chunk-FZWAFVAA.js → chunk-4FBZZIYD.js} +2 -2
  18. package/dist/{chunk-YV7J7X5N.js → chunk-5HRORJQY.js} +22 -12
  19. package/dist/chunk-5HRORJQY.js.map +1 -0
  20. package/dist/{chunk-OTYQPHPL.js → chunk-6SOJM3VR.js} +5 -5
  21. package/dist/chunk-BOD4O7OF.js +40 -0
  22. package/dist/chunk-BOD4O7OF.js.map +1 -0
  23. package/dist/{chunk-Z7VFTS2J.js → chunk-CY6U5S3X.js} +2 -2
  24. package/dist/{chunk-VIDQF3F5.js → chunk-D3V5B42D.js} +5 -34
  25. package/dist/chunk-D3V5B42D.js.map +1 -0
  26. package/dist/{chunk-YGYXHNAQ.js → chunk-FIUKOSWI.js} +21 -8
  27. package/dist/chunk-FIUKOSWI.js.map +1 -0
  28. package/dist/{chunk-WJL2NJXN.js → chunk-GSH6QNNS.js} +2 -2
  29. package/dist/{chunk-RBNA5AZT.js → chunk-L3JOU6XM.js} +2 -2
  30. package/dist/{chunk-IDVBLYCY.js → chunk-LMZQ2Z4U.js} +56 -2
  31. package/dist/{chunk-IDVBLYCY.js.map → chunk-LMZQ2Z4U.js.map} +1 -1
  32. package/dist/{chunk-VUINJM5M.js → chunk-QAY5UIJO.js} +2 -193
  33. package/dist/chunk-QAY5UIJO.js.map +1 -0
  34. package/dist/{chunk-P2J6SOXT.js → chunk-QG2OVF2D.js} +5 -3
  35. package/dist/{chunk-P2J6SOXT.js.map → chunk-QG2OVF2D.js.map} +1 -1
  36. package/dist/chunk-REVYNR6C.js +100 -0
  37. package/dist/chunk-REVYNR6C.js.map +1 -0
  38. package/dist/{chunk-ZZ2HOPME.js → chunk-TWS7AZEY.js} +2 -2
  39. package/dist/chunk-UHMJT4T7.js +200 -0
  40. package/dist/chunk-UHMJT4T7.js.map +1 -0
  41. package/dist/chunk-UMMZHCPB.js +190 -0
  42. package/dist/chunk-UMMZHCPB.js.map +1 -0
  43. package/dist/chunk-VZSRQ272.js +149 -0
  44. package/dist/chunk-VZSRQ272.js.map +1 -0
  45. package/dist/{chunk-L5G7OUKD.js → chunk-XY4DDNEG.js} +8 -190
  46. package/dist/chunk-XY4DDNEG.js.map +1 -0
  47. package/dist/chunk-Y47J2LJ3.js +859 -0
  48. package/dist/chunk-Y47J2LJ3.js.map +1 -0
  49. package/dist/{chunk-BABOZOSN.js → chunk-ZFIBGEOL.js} +3 -3
  50. package/dist/chunk-ZFIBGEOL.js.map +1 -0
  51. package/dist/{code-agent-session-BRXmavYv.d.ts → code-agent-session-BO8nCnv3.d.ts} +1 -1
  52. package/dist/contract/index.d.ts +24 -95
  53. package/dist/contract/index.js +16 -755
  54. package/dist/contract/index.js.map +1 -1
  55. package/dist/{control-GeE8OhpN.d.ts → control-_Qb7skHX.d.ts} +2 -2
  56. package/dist/control.d.ts +5 -5
  57. package/dist/corpus-BoR-041R.d.ts +560 -0
  58. package/dist/cost-ledger-DuSqlw5B.d.ts +113 -0
  59. package/dist/counterfactual-Dwibr5IW.d.ts +85 -0
  60. package/dist/{dataset-B2kL-fSM.d.ts → dataset-BbGkaN2I.d.ts} +1 -1
  61. package/dist/{registry-DrEQ3Luj.d.ts → default-registry-zoGHUQEH.d.ts} +29 -2
  62. package/dist/diagnose.d.ts +251 -0
  63. package/dist/diagnose.js +381 -0
  64. package/dist/diagnose.js.map +1 -0
  65. package/dist/{errors-Dwqw-T_m.d.ts → errors-CzMUYo7b.d.ts} +1 -1
  66. package/dist/{feedback-trajectory-B3rErRsh.d.ts → feedback-trajectory-D9OVLrg9.d.ts} +1 -1
  67. package/dist/fuzz.d.ts +484 -0
  68. package/dist/fuzz.js +613 -0
  69. package/dist/fuzz.js.map +1 -0
  70. package/dist/governance/index.d.ts +4 -4
  71. package/dist/hosted/index.d.ts +6 -6
  72. package/dist/{index-DE3RXAXD.d.ts → index-Bx3gZ8xl.d.ts} +1 -1
  73. package/dist/index.d.ts +717 -455
  74. package/dist/index.js +1590 -793
  75. package/dist/index.js.map +1 -1
  76. package/dist/{insight-report-3ADTfClO.d.ts → insight-report-BBwvOh6x.d.ts} +2 -2
  77. package/dist/{integrity-CJzrpUua.d.ts → integrity-VJ9A7aST.d.ts} +1 -1
  78. package/dist/{judge-calibration-DilmB3Ml.d.ts → judge-calibration-0p2QcWNE.d.ts} +1 -1
  79. package/dist/{kind-factory-CVecZZG_.d.ts → kind-factory-5b7xXXOr.d.ts} +2 -2
  80. package/dist/{llm-client-CuUg2Mn3.d.ts → llm-client-BeEcAokY.d.ts} +1 -1
  81. package/dist/matrix/index.d.ts +2 -2
  82. package/dist/meta-eval/index.d.ts +177 -3
  83. package/dist/meta-eval/index.js +260 -1
  84. package/dist/meta-eval/index.js.map +1 -1
  85. package/dist/{multi-layer-verifier-DlWCXuxL.d.ts → multi-layer-verifier-DUZXrPDA.d.ts} +7 -1
  86. package/dist/multishot/index.d.ts +25 -11
  87. package/dist/multishot/index.js +36 -7
  88. package/dist/multishot/index.js.map +1 -1
  89. package/dist/openapi.json +1 -1
  90. package/dist/pipelines/index.js +2 -2
  91. package/dist/{agent-profile-D0PBIWlV.d.ts → pre-registration-DELOEJ8v.d.ts} +144 -4
  92. package/dist/{provenance-DPpNIOJD.d.ts → provenance-LnqRT0sS.d.ts} +5 -5
  93. package/dist/{red-team-DW9Ca_tj.d.ts → red-team-BXHil6c8.d.ts} +1 -1
  94. package/dist/{release-report-hlNtD12q.d.ts → release-report-euXIV_Sk.d.ts} +3 -3
  95. package/dist/reporting.d.ts +8 -8
  96. package/dist/reporting.js +3 -3
  97. package/dist/{researcher-BLPHBbNV.d.ts → researcher-DE6Gpnb4.d.ts} +4 -4
  98. package/dist/rl.d.ts +194 -656
  99. package/dist/rl.js +236 -154
  100. package/dist/rl.js.map +1 -1
  101. package/dist/{rubric-predictive-validity-CnEl9Jc8.d.ts → rubric-predictive-validity-Cy_W-hWZ.d.ts} +1 -1
  102. package/dist/{run-campaign-4Y5V5CN3.js → run-campaign-RDGAM5KJ.js} +3 -3
  103. package/dist/{run-improvement-loop-CNqQckTj.d.ts → run-improvement-loop-5z_l5zDz.d.ts} +2 -2
  104. package/dist/{run-record-De9VarXR.d.ts → run-record-e7vj1uZQ.d.ts} +1 -1
  105. package/dist/{runtime-trajectory-BLRiaifm.d.ts → runtime-trajectory-BDgfGZSr.d.ts} +1 -1
  106. package/dist/{semantic-concept-judge-DIEgr_6v.d.ts → semantic-concept-judge-Dn8Z6KEG.d.ts} +5 -31
  107. package/dist/series-convergence-D5OWMBg6.d.ts +33 -0
  108. package/dist/{statistics-CnC1FMbx.d.ts → statistics-C7PozGrZ.d.ts} +71 -2
  109. package/dist/{summary-report-Db0dDSWP.d.ts → summary-report-DGmUucwQ.d.ts} +1 -1
  110. package/dist/traces.d.ts +3 -3
  111. package/dist/traces.js +8 -6
  112. package/dist/{types-Cu3u_x59.d.ts → types-2VVIL04s.d.ts} +2 -2
  113. package/dist/{types-D7lLRYe9.d.ts → types-BU-7W85F.d.ts} +21 -1
  114. package/dist/{types-CqPax19X.d.ts → types-mn5Aqk7x.d.ts} +1 -1
  115. package/dist/{verdict-CeEgtjyI.d.ts → verdict-C9MlYujm.d.ts} +3 -0
  116. package/dist/wire/index.d.ts +3 -3
  117. package/dist/workflow/index.d.ts +12 -11
  118. package/dist/workflow/index.js +1 -1
  119. package/package.json +11 -1
  120. package/dist/chunk-BABOZOSN.js.map +0 -1
  121. package/dist/chunk-L5G7OUKD.js.map +0 -1
  122. package/dist/chunk-SHTXZ4O2.js +0 -113
  123. package/dist/chunk-SHTXZ4O2.js.map +0 -1
  124. package/dist/chunk-VIDQF3F5.js.map +0 -1
  125. package/dist/chunk-VUINJM5M.js.map +0 -1
  126. package/dist/chunk-YGYXHNAQ.js.map +0 -1
  127. package/dist/chunk-YV7J7X5N.js.map +0 -1
  128. /package/dist/{chunk-FZWAFVAA.js.map → chunk-4FBZZIYD.js.map} +0 -0
  129. /package/dist/{chunk-OTYQPHPL.js.map → chunk-6SOJM3VR.js.map} +0 -0
  130. /package/dist/{chunk-Z7VFTS2J.js.map → chunk-CY6U5S3X.js.map} +0 -0
  131. /package/dist/{chunk-WJL2NJXN.js.map → chunk-GSH6QNNS.js.map} +0 -0
  132. /package/dist/{chunk-RBNA5AZT.js.map → chunk-L3JOU6XM.js.map} +0 -0
  133. /package/dist/{chunk-ZZ2HOPME.js.map → chunk-TWS7AZEY.js.map} +0 -0
  134. /package/dist/{run-campaign-4Y5V5CN3.js.map → run-campaign-RDGAM5KJ.js.map} +0 -0
package/dist/index.d.ts CHANGED
@@ -1,30 +1,32 @@
1
- export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-GeE8OhpN.js';
2
- import { R as RunRecord, a as RunSplitTag } from './run-record-De9VarXR.js';
3
- export { e as AGENT_PROFILE_KINDS, A as AgentProfileCell, d as AgentProfileCellInput, f as AgentProfileCellSchemaVersion, g as AgentProfileCellValidationError, h as AgentProfileDimensionValue, i as AgentProfileHarness, j as AgentProfileJson, k as AgentProfileKind, l as AgentProfileSource, m as AgentProfileSourceInput, J as JudgeScoresRecord, c as RunJudgeMetadata, n as RunOutcome, o as RunRecordValidationError, b as RunTokenUsage, S as SandboxAgentProfileLike, p as agentProfileCellHashMaterial, q as agentProfileCellKey, r as assertRunAgentProfileCell, s as buildAgentProfileCell, t as buildSandboxAgentProfileCell, u as groupRunsByAgentProfileCell, v as isRunRecord, w as parseRunRecordSafe, x as requireAgentProfileCell, y as roundTripRunRecord, z as toAgentProfileJson, B as validateAgentProfileCell, C as validateRunRecord, D as verifyAgentProfileCell } from './run-record-De9VarXR.js';
4
- export { B as BehavioralMetrics, z as ConceptComplexity, A as ConceptFinding, E as ConceptSpec, G as ConceptWeightStrategy, C as CreateAnalystAiConfig, H as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DefaultAnalystRegistryOptions, c as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, f as FindingSubject, g as FindingSubjectKind, i as FindingsDiff, j as FindingsStore, I as IMPROVEMENT_KIND_SPEC, k as KNOWLEDGE_GAP_KIND_SPEC, l as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, J as SEMANTIC_CONCEPT_JUDGE_VERSION, m as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, L as SemanticConceptJudgeResult, n as SkillUsageAnalyst, M as SuboptimalCode, N as SuboptimalSignal, r as buildDefaultAnalystRegistry, O as computeTraceMetrics, t as createAnalystAi, Q as createSemanticConceptJudge, u as defaultIsMaterial, v as diffFindings, R as runSemanticConceptJudge } from './semantic-concept-judge-DIEgr_6v.js';
5
- import { l as ChatRequest, p as CreateChatClientOpts } from './types-Cu3u_x59.js';
6
- export { A as Analyst, a as AnalystContext, g as AnalystCost, c as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-Cu3u_x59.js';
7
- export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, T as TraceAnalystGolden, c as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-CVecZZG_.js';
8
- export { A as AnalystHooks, a as AnalystRegistry, b as AnalystRegistryOptions, B as BudgetPolicy, R as RegistryRunOpts } from './registry-DrEQ3Luj.js';
1
+ export { A as ActionExecutionPolicy, a as ActionPolicyDecision, C as ControlRunToRunRecordOptions, L as LlmJsonCall, b as LlmReviewerConfig, P as ProposeFn, c as ProposeInput, d as ProposeOutput, e as ProposeReviewConfig, f as ProposeReviewControlAction, g as ProposeReviewControlConfig, h as ProposeReviewControlResult, i as ProposeReviewControlState, j as ProposeReviewReport, k as ProposeReviewShot, R as Review, l as ReviewFn, m as ReviewInput, n as ReviewMemoryEntry, o as ReviewMemoryStore, p as RunEvidenceMetadata, V as Verification, q as VerifyFn, r as controlFailureClassFromVerification, s as controlRunToRunRecord, t as createLlmReviewer, u as evaluateActionPolicy, v as inMemoryReviewStore, w as jsonlReviewStore, x as runProposeReview, y as runProposeReviewAsControlLoop, z as scoreFromEvals } from './control-_Qb7skHX.js';
2
+ import { R as RunRecord, a as RunSplitTag } from './run-record-e7vj1uZQ.js';
3
+ export { e as AGENT_PROFILE_KINDS, A as AgentProfileCell, d as AgentProfileCellInput, f as AgentProfileCellSchemaVersion, g as AgentProfileCellValidationError, h as AgentProfileDimensionValue, i as AgentProfileHarness, j as AgentProfileJson, k as AgentProfileKind, l as AgentProfileSource, m as AgentProfileSourceInput, J as JudgeScoresRecord, c as RunJudgeMetadata, n as RunOutcome, o as RunRecordValidationError, b as RunTokenUsage, S as SandboxAgentProfileLike, p as agentProfileCellHashMaterial, q as agentProfileCellKey, r as assertRunAgentProfileCell, s as buildAgentProfileCell, t as buildSandboxAgentProfileCell, u as groupRunsByAgentProfileCell, v as isRunRecord, w as parseRunRecordSafe, x as requireAgentProfileCell, y as roundTripRunRecord, z as toAgentProfileJson, B as validateAgentProfileCell, C as validateRunRecord, D as verifyAgentProfileCell } from './run-record-e7vj1uZQ.js';
4
+ export { B as BehavioralMetrics, x as ConceptComplexity, y as ConceptFinding, z as ConceptSpec, A as ConceptWeightStrategy, C as CreateAnalystAiConfig, E as DEFAULT_COMPLEXITY_WEIGHTS, D as DEFAULT_TRACE_ANALYST_KINDS, b as DiffPolicy, F as FAILURE_MODE_KIND_SPEC, e as FindingSubject, f as FindingSubjectKind, h as FindingsDiff, i as FindingsStore, I as IMPROVEMENT_KIND_SPEC, j as KNOWLEDGE_GAP_KIND_SPEC, k as KNOWLEDGE_POISONING_KIND_SPEC, P as PersistedFinding, G as SEMANTIC_CONCEPT_JUDGE_VERSION, l as SKILL_USAGE_ANALYST, a as SemanticConceptJudgeInput, S as SemanticConceptJudgeOptions, H as SemanticConceptJudgeResult, m as SkillUsageAnalyst, J as SuboptimalCode, L as SuboptimalSignal, M as computeTraceMetrics, r as createAnalystAi, N as createSemanticConceptJudge, s as defaultIsMaterial, t as diffFindings, O as runSemanticConceptJudge } from './semantic-concept-judge-Dn8Z6KEG.js';
5
+ import { l as ChatRequest, p as CreateChatClientOpts } from './types-2VVIL04s.js';
6
+ export { A as Analyst, a as AnalystContext, g as AnalystCost, c as AnalystFinding, i as AnalystInputKind, j as AnalystRequirements, f as AnalystRunEvent, e as AnalystRunInputs, d as AnalystRunResult, b as AnalystRunSummary, h as AnalystSeverity, k as ChatCallOpts, C as ChatClient, m as ChatResponse, n as ChatTransport, o as CliBridgeTransportOpts, D as DirectProviderTransportOpts, E as EvidenceRef, M as MockTransportOpts, R as RouterTransportOpts, S as SandboxSdkTransportOpts, q as computeFindingId, r as createChatClient, s as makeFinding } from './types-2VVIL04s.js';
7
+ export { a as AnalystHooks, A as AnalystRegistry, c as AnalystRegistryOptions, B as BudgetPolicy, D as DefaultAnalystRegistryOptions, R as RegistryRunOpts, b as buildDefaultAnalystRegistry } from './default-registry-zoGHUQEH.js';
8
+ export { C as CreateTraceAnalystKindOpts, a as RawAnalystFinding, c as TraceAnalystGolden, T as TraceAnalystKindSpec, d as createTraceAnalystKind, r as renderPriorFindings } from './kind-factory-5b7xXXOr.js';
9
9
  import { TCloud } from '@tangle-network/tcloud';
10
10
  import { B as BenchmarkRunnerConfig, S as Scenario, c as BenchmarkReport, P as ProductClientConfig, C as CheckResult, T as TestResult, d as PersonaConfig, D as DriverResult, e as DriverState, b as JudgeFn, f as CollectedArtifacts, g as ScenarioResult, h as TurnMetrics, i as ScenarioFile, j as CompletionCriterion } from './types-Croy5h7V.js';
11
11
  export { A as ArtifactCheck, k as ArtifactResult, E as EvalResult, F as FeedbackPattern, l as JudgeConfig, a as JudgeInput, m as JudgeRubric, J as JudgeScore, n as PersonaRigor, R as RouteMap, o as RubricDimension, p as Turn, q as TurnResult } from './types-Croy5h7V.js';
12
12
  export { c as ControlActionFailureMode, d as ControlActionOutcome, e as ControlBudget, f as ControlContext, g as ControlDecision, C as ControlEvalResult, a as ControlRunResult, h as ControlRuntimeConfig, i as ControlRuntimeError, j as ControlSeverity, b as ControlStep, k as ControlStopPolicies, S as StopDecision, l as allCriticalPassed, o as objectiveEval, r as runAgentControlLoop, s as stopOnNoProgress, m as stopOnRepeatedAction, n as subjectiveEval } from './control-runtime-DuFBYg7A.js';
13
- import { A as AgentEvalError } from './errors-Dwqw-T_m.js';
14
- export { a as AgentEvalErrorCode, C as CaptureIntegrityError, b as ConfigError, J as JudgeError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-Dwqw-T_m.js';
15
- import { b as FeedbackLabel, F as FeedbackTrajectoryStore, a as FeedbackTrajectory } from './feedback-trajectory-B3rErRsh.js';
16
- export { c as FeedbackArtifactType, d as FeedbackAttempt, e as FeedbackLabelKind, f as FeedbackLabelSource, g as FeedbackOptimizerRow, h as FeedbackOutcome, i as FeedbackReplayAdapter, j as FeedbackReplayResult, k as FeedbackSeverity, l as FeedbackSplitPolicy, m as FeedbackTask, n as FeedbackTrajectoryFilter, o as FileSystemFeedbackTrajectoryStore, I as InMemoryFeedbackTrajectoryStore, P as PreferenceMemoryEntry, p as ProposedSideEffect, q as assignFeedbackSplit, r as controlRunToFeedbackTrajectory, s as createFeedbackTrajectory, t as feedbackTrajectoriesToDatasetScenarios, u as feedbackTrajectoriesToOptimizerRows, v as feedbackTrajectoryToDatasetScenario, w as feedbackTrajectoryToOptimizerRow, x as parseFeedbackTrajectoriesJsonl, y as renderPreferenceMemoryMarkdown, z as replayFeedbackTrajectories, A as replayFeedbackTrajectory, B as serializeFeedbackTrajectoriesJsonl, C as summarizePreferenceMemory, D as withAssignedFeedbackSplit } from './feedback-trajectory-B3rErRsh.js';
17
- import { A as AgentProfile$1 } from './agent-profile-D0PBIWlV.js';
18
- export { c as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, b as CorrectnessChecker, L as LlmCorrectnessCheckerOpts, g as ProducedProposal, P as ProducedState, h as ProposalEventLike, i as RequirementCheck, R as RuntimeEventLike, S as SatisfiedBy, T as TaskGold, j as ToolCallEventLike, V as ValidationContext, k as ValidationIssue, l as ValidationResult, m as agentProfileHash, n as assertRealBackend, o as byteLengthRange, p as composeValidators, q as containsAll, r as createLlmCorrectnessChecker, s as createTokenRecallChecker, t as extractProducedState, u as jsonHasKeys, v as parseCorrectnessResponse, w as regexMatch, x as summarizeBackendIntegrity, y as verifyCompletion } from './agent-profile-D0PBIWlV.js';
13
+ import { A as AgentEvalError, J as JudgeError, a as ConfigError } from './errors-CzMUYo7b.js';
14
+ export { b as AgentEvalErrorCode, C as CaptureIntegrityError, N as NotFoundError, R as ReplayError, V as ValidationError, c as VerificationError } from './errors-CzMUYo7b.js';
15
+ import { b as FeedbackLabel, F as FeedbackTrajectoryStore, a as FeedbackTrajectory } from './feedback-trajectory-D9OVLrg9.js';
16
+ export { c as FeedbackArtifactType, d as FeedbackAttempt, e as FeedbackLabelKind, f as FeedbackLabelSource, g as FeedbackOptimizerRow, h as FeedbackOutcome, i as FeedbackReplayAdapter, j as FeedbackReplayResult, k as FeedbackSeverity, l as FeedbackSplitPolicy, m as FeedbackTask, n as FeedbackTrajectoryFilter, o as FileSystemFeedbackTrajectoryStore, I as InMemoryFeedbackTrajectoryStore, P as PreferenceMemoryEntry, p as ProposedSideEffect, q as assignFeedbackSplit, r as controlRunToFeedbackTrajectory, s as createFeedbackTrajectory, t as feedbackTrajectoriesToDatasetScenarios, u as feedbackTrajectoriesToOptimizerRows, v as feedbackTrajectoryToDatasetScenario, w as feedbackTrajectoryToOptimizerRow, x as parseFeedbackTrajectoriesJsonl, y as renderPreferenceMemoryMarkdown, z as replayFeedbackTrajectories, A as replayFeedbackTrajectory, B as serializeFeedbackTrajectoriesJsonl, C as summarizePreferenceMemory, D as withAssignedFeedbackSplit } from './feedback-trajectory-D9OVLrg9.js';
17
+ import { b as CorrectnessChecker, A as AgentProfile$1 } from './pre-registration-DELOEJ8v.js';
18
+ export { c as ArtifactCheckArtifact, d as ArtifactEventLike, e as ArtifactValidator, f as BackendIntegrityError, B as BackendIntegrityReport, C as CompletionRequirement, a as CompletionVerdict, H as HypothesisManifest, g as HypothesisResult, L as LlmCorrectnessCheckerOpts, h as ProducedProposal, P as ProducedState, i as ProposalEventLike, j as RequirementCheck, R as RuntimeEventLike, k as SatisfiedBy, S as SignedManifest, l as SignedManifestAlgo, T as TaskGold, m as ToolCallEventLike, V as ValidationContext, n as ValidationIssue, o as ValidationResult, p as agentProfileHash, q as assertRealBackend, r as byteLengthRange, s as canonicalize, t as completionVerdict, u as composeValidators, v as containsAll, w as createLlmCorrectnessChecker, x as createTokenRecallChecker, y as evaluateHypothesis, z as extractProducedState, D as hashJson, E as jsonHasKeys, F as parseCorrectnessResponse, G as regexMatch, I as signManifest, J as summarizeBackendIntegrity, K as verifyCompletion, M as verifyManifest } from './pre-registration-DELOEJ8v.js';
19
19
  export { DataAcquisitionPlan, KnowledgeAcquisitionMode, KnowledgeBundle, KnowledgeFallbackPolicy, KnowledgeFreshness, KnowledgeImportance, KnowledgeReadinessReport, KnowledgeRecommendedAction, KnowledgeRequirement, KnowledgeRequirementCategory, KnowledgeResponsibleSurface, KnowledgeSensitivity, ScoreKnowledgeReadinessOptions, UserQuestion, acquisitionPlansForKnowledgeGaps, blockingKnowledgeEval, knowledgeReadinessTracePayload, scoreKnowledgeReadiness, userQuestionsForKnowledgeGaps } from './knowledge/index.js';
20
- import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-hlNtD12q.js';
21
- export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-hlNtD12q.js';
22
- export { C as CliffsMagnitude, c as CorpusAgreementOptions, d as CorpusAgreementPerDimension, e as CorpusAgreementReport, f as CorpusScoreRecord, P as PairedBootstrapOptions, a as PairedBootstrapResult, W as WeightedCompositeInput, g as WeightedCompositeResult, b as benjaminiHochberg, h as bonferroni, i as cliffsDelta, j as cohensD, k as confidenceInterval, l as corpusInterRaterAgreement, m as corpusInterRaterAgreementFromJudgeScores, n as interRaterReliability, o as interpretCliffs, q as mannWhitneyU, r as normalizeScores, p as pairedBootstrap, s as pairedMde, t as pairedTTest, u as partialCredit, v as requiredSampleSize, x as weightedComposite, y as weightedMean, w as wilcoxonSignedRank } from './statistics-CnC1FMbx.js';
20
+ import { h as ReleaseConfidenceThresholds, f as ReleaseConfidenceScorecard } from './release-report-euXIV_Sk.js';
21
+ export { A as ActionableSideInfo, o as AsiSeverity, B as BootstrapOptions, a as BootstrapResult, J as JudgeReplayGateArgs, R as ReleaseConfidenceAxis, b as ReleaseConfidenceAxisName, c as ReleaseConfidenceInput, d as ReleaseConfidenceIssue, e as ReleaseConfidenceMetrics, g as ReleaseConfidenceStatus, i as ReleaseTraceEvidence, j as RenderReleaseReportOptions, V as Verdict, k as assertReleaseConfidence, l as bootstrapCi, m as evaluateReleaseConfidence, n as judgeReplayGate, r as renderReleaseReport } from './release-report-euXIV_Sk.js';
22
+ export { c as CliffsMagnitude, d as CorpusAgreementOptions, e as CorpusAgreementPerDimension, C as CorpusAgreementReport, f as CorpusScoreRecord, g as EProcess, h as EProcessOptions, E as EProcessState, i as EProcessStep, P as PairedBootstrapOptions, a as PairedBootstrapResult, W as WeightedCompositeInput, j as WeightedCompositeResult, b as benjaminiHochberg, k as bonferroni, l as cliffsDelta, m as cohensD, n as confidenceInterval, o as corpusInterRaterAgreement, q as corpusInterRaterAgreementFromJudgeScores, r as eProcess, s as interRaterReliability, t as interpretCliffs, u as mannWhitneyU, v as mulberry32, x as normalizeScores, p as pairedBootstrap, y as pairedMde, z as pairedTTest, A as partialCredit, B as requiredSampleSize, D as weightedComposite, F as weightedMean, w as wilcoxonSignedRank } from './statistics-C7PozGrZ.js';
23
23
  import { a as AnalyzeTracesInput, A as AnalyzeTracesOptions, b as AnalyzeTracesResult } from './analyst-C8HHvfJp.js';
24
24
  export { c as AnalyzeTracesTurnSnapshot, d as analyzeTraces } from './analyst-C8HHvfJp.js';
25
25
  import { OtelExporter, OtelExportConfig } from './traces.js';
26
26
  export { CaptureFetchContext, CaptureFetchOptions, ExportableSpan, ExtractedUsage, FlattenOtlpOptions, OTEL_AGENT_EVAL_SCOPE, OtlpExport, OtlpFileTraceStore, OtlpFileTraceStoreOptions, OtlpFlatLine, OtlpResourceSpans, OtlpSpan, OtlpToRunRecordsOptions, OtlpTraceRunRecord, ProjectedOtlpSpan, ReplayCache, ReplayCacheEntry, ReplayCacheMissError, ReplayCacheStats, ReplayFetchOptions, SpanNotFoundError, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_SUBAGENT_DESCRIPTION, TraceAggregate, TraceAnalystHookOptions, TraceFileMissingError, TraceInsightContext, TraceInsightFinding, TraceInsightPanelRole, TraceInsightPromptInput, TraceInsightQualityGate, TraceInsightQuestion, TraceInsightReadiness, TraceInsightSuite, TraceInsightTask, TraceNotFoundError, TraceStoreSource, TraceStoreToOtlpOptions, TracesToOtlpResult, asNumber, asString, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, captureFetchToRawSink, convertTraceStoresToOtlp, createOtelExporter, createOtelTracingStore, createReplayFetch, defaultTraceInsightPanel, describeTraceInsightScope, domainEvidencePattern, exportRunAsOtlp, extractOtlpAttributes, extractUsage, extractUsageFromResponse, extractUsageFromSse, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, inferDomainKeywords, inferOtlpKind, iterateRawCalls, otelRunCompleteHook, otlpToRunRecords, otlpToTraceRunRecords, planTraceInsightQuestions, projectOtlpFlatLine, readOtlpStatus, scoreTraceInsightReadiness, stringField, tokenizeDomainWords, traceAnalystFunctionGroup, traceAnalystOnRunComplete } from './traces.js';
27
27
  export { D as DEFAULT_TRACE_ANALYST_BUDGETS, b as DatasetOverview, E as ErrorCluster, Q as QueryTracesPage, S as SearchSpanResult, c as SearchTraceResult, d as SpanMatchRecord, e as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, T as TraceAnalysisStore, f as TraceAnalystByteBudgets, g as TraceAnalystFilters, a as TraceAnalystSpan, h as TraceAnalystSpanKind, i as TraceAnalystSpanStatus, j as TraceAnalystTraceSummary, V as ViewSpansResult, k as ViewTraceOversized, l as ViewTraceResult } from './store-C1YxJDEK.js';
28
+ import { a as JudgeConfig, S as Scenario$1, G as Gate, J as JudgeScore } from './types-BU-7W85F.js';
29
+ import { A as AnalyzeRunsOptions } from './analyze-runs-DwCEkpO_.js';
28
30
  import { S as SteeringBundle } from './harness-optimizer-EnEnQPsr.js';
29
31
  export { D as DEFAULT_HARNESS_OBJECTIVES, H as HarnessAdapter, a as HarnessExperimentConfig, b as HarnessExperimentResult, c as HarnessIntervention, d as HarnessRunRequest, e as HarnessRunResult, f as HarnessScenario, g as HarnessSelection, h as HarnessVariant, i as HarnessVariantReport, M as MeasurementPolicy, j as SteeringDelta, k as SteeringRolePrompt, W as WorkflowTopology, m as mergeSteeringBundle, r as renderSteeringText, l as runHarnessExperiment, s as selectHarnessVariant, n as summarizeHarnessResults } from './harness-optimizer-EnEnQPsr.js';
30
32
  import { S as SandboxDriver, H as HarnessConfig, a as SandboxHarnessResult } from './test-graded-scenario-BdVaPyHT.js';
@@ -33,7 +35,7 @@ import { b as RunScoreWeights, R as RunScore } from './run-critic-BAIjX99r.js';
33
35
  export { D as DEFAULT_RUN_SCORE_WEIGHTS, c as RunCritic, d as RunCriticOptions, a as RunTrace, e as aggregateRunScore, f as clamp01 } from './run-critic-BAIjX99r.js';
34
36
  import { T as TraceEmitter } from './emitter-DEZwY14K.js';
35
37
  export { R as RunCompleteHook, a as RunCompleteHookContext, S as SpanHandle, b as TraceEmitterOptions, l as llmSpanFromProvider } from './emitter-DEZwY14K.js';
36
- export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-CJzrpUua.js';
38
+ export { b as RunIntegrityError, R as RunIntegrityExpectations, c as RunIntegrityIssue, d as RunIntegrityIssueCode, a as RunIntegrityReport, e as assertRunCaptured, t as throwIfRunIncomplete } from './integrity-VJ9A7aST.js';
37
39
  export { a as aggregateLlm, b as argHash, g as groupBy, j as judgeSpans, l as llmSpans, r as runFailureClass, c as runsForScenario, t as toolSpans } from './query-CqTxMwDw.js';
38
40
  export { F as FileSystemRawProviderSink, a as FileSystemRawProviderSinkOptions, I as InMemoryRawProviderSink, b as InMemoryRawProviderSinkOptions, N as NoopRawProviderSink, P as ProviderRedactor, c as RawProviderDirection, d as RawProviderEvent, R as RawProviderSink, e as RawProviderSinkFilter, f as defaultProviderRedactor, p as providerFromBaseUrl } from './raw-provider-sink-C46HDghv.js';
39
41
  export { D as DEFAULT_REDACTION_RULES, b as REDACTION_VERSION, a as RedactionReport, R as RedactionRule, r as redactString, c as redactValue } from './redact-B40YG2M_.js';
@@ -42,31 +44,35 @@ export { A as Artifact, E as EventKind, i as FAILURE_CLASSES, F as FailureClass,
42
44
  import { T as TraceStore, R as RunFilter } from './store-CKUAgsJz.js';
43
45
  export { E as EventFilter, F as FileSystemTraceStore, a as FileSystemTraceStoreOptions, I as InMemoryTraceStore, S as SpanFilter } from './store-CKUAgsJz.js';
44
46
  export { D as DEFAULT_FAILURE_RULES, b as FailureClassification, c as FailureContext, d as FailureRule, e as classifyFailure } from './failure-cluster-CL7IVgkJ.js';
45
- export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-BLRiaifm.js';
47
+ export { P as ProjectRuntimeTrajectoryEvidenceOptions, a as RuntimeTrajectoryEvidenceProjection, b as RuntimeTrajectoryEvidenceSummary, c as RuntimeTrajectoryHookEvent, R as RuntimeTrajectoryRecord, d as RuntimeTrajectoryRunRecord, p as parseRuntimeTrajectoryHookEvent, e as projectRuntimeTrajectoryEvidence } from './runtime-trajectory-BDgfGZSr.js';
46
48
  import { a as BaselineReport } from './baseline-DE36-Np7.js';
47
49
  export { B as BaselineOptions, M as MetricSamples, b as MetricVerdict, T as ToolStats, d as ToolUseMetrics, e as ToolUseOptions, f as compareToBaseline, c as computeToolUseMetrics, i as iqr, w as welchsTTest } from './baseline-DE36-Np7.js';
48
- import { T as Trajectory, a as TrajectoryStep } from './trajectory-GEdXJCL5.js';
50
+ import { a as TrajectoryStep, T as Trajectory } from './trajectory-GEdXJCL5.js';
49
51
  export { b as buildTrajectory } from './trajectory-GEdXJCL5.js';
52
+ import { b as ChannelRollup, C as CostLedger } from './cost-ledger-DuSqlw5B.js';
53
+ export { a as CostChannel, c as CostLedgerEntry, d as CostLedgerSummary, e as CostResult, f as CostUsage, g as costForUsage, m as modelPriceKey } from './cost-ledger-DuSqlw5B.js';
50
54
  export { D as Direction, O as Objective, P as ParetoResult, c as crowdingDistance, d as dominates, p as paretoFrontier, a as paretoFrontierWithCrowding, s as scalarScore } from './pareto-E-pembql.js';
51
- export { D as DefaultVerdict } from './verdict-CeEgtjyI.js';
52
- import { a as DatasetScenario, b as Dataset } from './dataset-B2kL-fSM.js';
53
- export { d as DatasetDifficulty, c as DatasetManifest, e as DatasetProvenance, D as DatasetSplit, H as HoldoutLockedError, S as SliceOptions, h as hashScenarios } from './dataset-B2kL-fSM.js';
54
- export { b as CalibrationResult, c as CandidateScore, a as ContinuousAgreement, C as ContinuousAgreementOptions, d as ContinuousCalibrationResult, G as GoldenItem, P as PositionalBiasResult, S as SelfPreferenceResult, V as VerbosityBiasResult, e as calibrateJudge, f as calibrateJudgeContinuous, g as continuousAgreement, p as positionalBias, s as selfPreference, v as verbosityBias } from './judge-calibration-DilmB3Ml.js';
55
- export { D as DEFAULT_RED_TEAM_CORPUS, R as RedTeamCase, a as RedTeamCategory, b as RedTeamFinding, c as RedTeamPayload, d as RedTeamReport, r as redTeamDataset, e as redTeamReport, s as scoreRedTeamOutput, t as toolNamesForRun } from './red-team-DW9Ca_tj.js';
55
+ export { S as SeriesConvergenceOptions, a as SeriesConvergenceResult, b as analyzeSeries } from './series-convergence-D5OWMBg6.js';
56
+ import { D as DefaultVerdict } from './verdict-C9MlYujm.js';
57
+ import { a as DatasetScenario, b as Dataset } from './dataset-BbGkaN2I.js';
58
+ export { d as DatasetDifficulty, c as DatasetManifest, e as DatasetProvenance, D as DatasetSplit, H as HoldoutLockedError, S as SliceOptions, h as hashScenarios } from './dataset-BbGkaN2I.js';
59
+ export { a as CalibrationResult, c as CandidateScore, C as ContinuousAgreement, d as ContinuousAgreementOptions, b as ContinuousCalibrationResult, G as GoldenItem, P as PositionalBiasResult, S as SelfPreferenceResult, V as VerbosityBiasResult, e as calibrateJudge, f as calibrateJudgeContinuous, g as continuousAgreement, p as positionalBias, s as selfPreference, v as verbosityBias } from './judge-calibration-0p2QcWNE.js';
60
+ export { D as DEFAULT_RED_TEAM_CORPUS, R as RedTeamCase, a as RedTeamCategory, b as RedTeamFinding, c as RedTeamPayload, d as RedTeamReport, r as redTeamDataset, e as redTeamReport, s as scoreRedTeamOutput, t as toolNamesForRun } from './red-team-BXHil6c8.js';
61
+ export { c as CounterfactualContext, C as CounterfactualMutation, d as CounterfactualResult, b as CounterfactualRunner, a as attributeCounterfactuals, r as runCounterfactual } from './counterfactual-Dwibr5IW.js';
56
62
  import { a as PrmGrader } from './rubric-BOfxn4ja.js';
57
63
  export { EuRiskClass, GovernanceContext, GovernanceFinding, GovernanceReport, UseCaseSignals, classifyEuAiRisk, euAiActReport, nistAiRmfReport, renderMarkdown, soc2Report, summarize } from './governance/index.js';
58
- import { b as Layer, S as Severity, L as LayerResult, c as VerifyContext } from './multi-layer-verifier-DlWCXuxL.js';
59
- export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-DlWCXuxL.js';
60
- import { L as LlmClientOptions } from './llm-client-CuUg2Mn3.js';
61
- export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-CuUg2Mn3.js';
62
- export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-DE3RXAXD.js';
63
- export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-BLPHBbNV.js';
64
- export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-Db0dDSWP.js';
64
+ import { b as Layer, S as Severity, L as LayerResult, c as VerifyContext } from './multi-layer-verifier-DUZXrPDA.js';
65
+ export { F as Finding, d as LayerStatus, M as MultiLayerVerifier, a as VerificationReport, V as VerifyOptions, g as gradeSemanticStatus } from './multi-layer-verifier-DUZXrPDA.js';
66
+ import { L as LlmClientOptions } from './llm-client-BeEcAokY.js';
67
+ export { d as LlmCallError, b as LlmCallRequest, c as LlmCallResult, e as LlmClient, f as LlmMessage, g as LlmRouteAssertionError, a as LlmRouteRequirements, h as LlmUsage, i as assertLlmRoute, j as backoffMs, k as callLlm, l as callLlmJson, m as isTransientLlmError, p as probeLlm, s as stripFencedJson } from './llm-client-BeEcAokY.js';
68
+ export { B as BENCHMARK_SPLIT_SEED, a as BenchmarkAdapter, b as BenchmarkDatasetItem, c as BenchmarkEvaluation, d as benchmarkDeterministicSplit, i as benchmarks } from './index-Bx3gZ8xl.js';
69
+ export { C as CallbackResearcher, d as CallbackResearcherOptions, e as CampaignFactoryParams, f as CampaignIntegrityPolicy, g as CampaignRunContext, h as CampaignRunOutcome, i as CampaignRunner, j as CampaignScenario, k as CampaignVariant, c as EvalCampaignOptions, b as EvalCampaignResult, E as ExperimentPlan, a as ExperimentResult, l as FailedRun, F as FailureMode, N as NoopResearcher, R as Researcher, S as SteeringChange, r as runEvalCampaign } from './researcher-DE6Gpnb4.js';
70
+ export { G as GainDistributionBin, a as GainDistributionFigureSpec, b as GainDistributionOptions, m as GateDecision, n as GateEvidence, H as HeldOutGate, o as HeldOutGateConfig, q as HeldOutGateRejectionCode, P as ParetoFigureSpec, c as ParetoPoint, R as RESEARCH_REPORT_HARD_PAIR_FLOOR, d as ResearchReport, e as ResearchReportCandidate, f as ResearchReportDecision, g as ResearchReportMethodology, h as ResearchReportOptions, i as ResearchReportRecommendation, S as SummaryTable, j as SummaryTableOptions, k as SummaryTableRow, l as gainHistogram, p as paretoChart, r as researchReport, s as summaryTable } from './summary-report-DGmUucwQ.js';
65
71
  export { I as InterimReleaseConfidence, a as InterimReleaseConfidenceInput, P as PairedEvalueOptions, b as PairedEvalueSequence, c as PairedEvalueStep, S as SequentialDecision, e as evaluateInterimReleaseConfidence, p as pairedEvalueSequence } from './sequential-5iSVfzl2.js';
66
- import { S as Scenario$1, a as JudgeConfig, G as Gate } from './types-D7lLRYe9.js';
67
- import { e as GepaDriverConstraints, a as RunImprovementLoopResult } from './run-improvement-loop-CNqQckTj.js';
72
+ import { e as GepaDriverConstraints, a as RunImprovementLoopResult } from './run-improvement-loop-5z_l5zDz.js';
68
73
  import '@ax-llm/ax';
69
74
  import 'zod';
75
+ import './insight-report-BBwvOh6x.js';
70
76
  import './outcome-store-rnXLEqSn.js';
71
77
 
72
78
  /**
@@ -533,33 +539,76 @@ declare class CrossFamilyError extends Error {
533
539
  */
534
540
  declare function assertCrossFamily(models: string[], opts?: AssertCrossFamilyOptions): JudgeFamily[];
535
541
 
542
+ /**
543
+ * A judge's LLM response could not be parsed into scored dimensions.
544
+ * Thrown instead of fabricating a `{ dimension: 'parse_error', score: 0 }`
545
+ * row — a synthetic zero is indistinguishable from a real low score
546
+ * downstream. Carries the raw response for forensics. Callers (executor,
547
+ * ensemble wrappers) catch this per-judge and record a failed judge.
548
+ */
549
+ declare class JudgeParseError extends JudgeError {
550
+ /** Name of the judge whose response failed to parse. */
551
+ readonly judgeName: string;
552
+ /** The raw (truncated) model response that failed to parse. */
553
+ readonly raw: string;
554
+ constructor(judgeName: string, raw: string, options?: {
555
+ cause?: unknown;
556
+ });
557
+ }
536
558
  /**
537
559
  * Create a domain expert judge with a configurable domain.
538
560
  *
539
561
  * The judge evaluates professional accuracy and depth.
562
+ *
563
+ * @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
564
+ * Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
565
+ * multi-model panels via `ensembleJudge` (src/judge-panel.ts) — which are
566
+ * pluggable, fail-loud, and drive the campaign/improvement-loop engines.
540
567
  */
541
568
  declare function createDomainExpertJudge(domain: string): JudgeFn;
542
569
  /**
543
570
  * Code execution judge — evaluates whether code blocks are valid and runnable.
571
+ *
572
+ * @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
573
+ * Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
574
+ * multi-model panels via `ensembleJudge` (src/judge-panel.ts).
544
575
  */
545
576
  declare const codeExecutionJudge: JudgeFn;
546
577
  /**
547
578
  * Coherence judge — evaluates multi-turn consistency and progression.
579
+ *
580
+ * @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
581
+ * Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
582
+ * multi-model panels via `ensembleJudge` (src/judge-panel.ts).
548
583
  */
549
584
  declare const coherenceJudge: JudgeFn;
550
585
  /**
551
586
  * Adversarial judge — red-teams agent responses.
587
+ *
588
+ * @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
589
+ * Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
590
+ * multi-model panels via `ensembleJudge` (src/judge-panel.ts).
552
591
  */
553
592
  declare const adversarialJudge: JudgeFn;
554
593
  /**
555
594
  * Create a custom judge with a fully custom prompt.
595
+ *
596
+ * @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
597
+ * Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
598
+ * multi-model panels via `ensembleJudge` (src/judge-panel.ts).
556
599
  */
557
600
  declare function createCustomJudge(name: string, systemPrompt: string, opts?: {
558
601
  model?: string;
559
602
  temperature?: number;
560
603
  maxTokens?: number;
561
604
  }): JudgeFn;
562
- /** Default judge set (domain must be provided for domain expert) */
605
+ /**
606
+ * Default judge set (domain must be provided for domain expert)
607
+ *
608
+ * @deprecated Legacy `JudgeFn` factory tied to the fixed gpt-4o prompt shape.
609
+ * Build judges as campaign `JudgeConfig`s (src/campaign/types.ts) — or
610
+ * multi-model panels via `ensembleJudge` (src/judge-panel.ts).
611
+ */
563
612
  declare function defaultJudges(domain: string): JudgeFn[];
564
613
 
565
614
  interface LiveProofArtifact {
@@ -980,6 +1029,59 @@ declare class DualAgentBench {
980
1029
  run(config: DualAgentBenchConfig): Promise<DualAgentReport>;
981
1030
  }
982
1031
 
1032
+ /**
1033
+ * Eval primitives as agent tools — `makeEvalTools` packages the substrate's
1034
+ * judge / completion / analysis entry points as JSON-Schema tool definitions
1035
+ * an LLM agent loop can call. The host closes over the live config (judge
1036
+ * panels, the correctness checker, analyst registry); the agent passes only
1037
+ * data over the wire.
1038
+ *
1039
+ * Three tools, each present only when its config section is supplied:
1040
+ * - `run_judges` — score an artifact with the configured `JudgeConfig`s
1041
+ * - `verify_completion` — gold-spec requirement check → `CompletionVerdict`
1042
+ * - `analyze_runs` — `RunRecord[]` (inline or from a file) → `InsightReport`
1043
+ *
1044
+ * `toOpenAiTool` converts a definition to the OpenAI function-tool wire shape.
1045
+ */
1046
+
1047
+ /** One agent-callable tool. `parameters` is a JSON Schema (draft-07+) object. */
1048
+ interface EvalToolDef {
1049
+ name: string;
1050
+ description: string;
1051
+ parameters: Record<string, unknown>;
1052
+ handler: (args: unknown, ctx?: {
1053
+ signal?: AbortSignal;
1054
+ }) => Promise<unknown>;
1055
+ }
1056
+ interface MakeEvalToolsConfig {
1057
+ /** Judges available to `run_judges`. Omit to exclude the tool. */
1058
+ judges?: Array<JudgeConfig<unknown>>;
1059
+ /** Host-side correctness checker for `verify_completion` (the third
1060
+ * `verifyCompletion` argument — a function, so it cannot cross the wire).
1061
+ * Omit to exclude the tool. */
1062
+ completion?: {
1063
+ checkCorrectness: CorrectnessChecker;
1064
+ };
1065
+ /** `analyzeRuns` options minus `runs` (runs arrive as tool args, inline or
1066
+ * via `path`). Omit to exclude the tool. */
1067
+ analyze?: Omit<AnalyzeRunsOptions, 'runs'>;
1068
+ }
1069
+ /** OpenAI function-tool wire shape for an `EvalToolDef`. */
1070
+ declare function toOpenAiTool(def: EvalToolDef): {
1071
+ type: 'function';
1072
+ function: {
1073
+ name: string;
1074
+ description: string;
1075
+ parameters: Record<string, unknown>;
1076
+ };
1077
+ };
1078
+ /**
1079
+ * Build the eval toolset for the supplied config. Only sections present in
1080
+ * `cfg` produce tools, so the agent's tool list mirrors what the host
1081
+ * actually wired. Handlers fail loud on malformed args — no silent defaults.
1082
+ */
1083
+ declare function makeEvalTools(cfg: MakeEvalToolsConfig): EvalToolDef[];
1084
+
983
1085
  /**
984
1086
  * Judge-ensemble reducer — folds N independent judge verdicts on the same
985
1087
  * artifact into one aggregate score.
@@ -1006,6 +1108,12 @@ interface JudgeVerdict<D extends string = string> {
1006
1108
  rationale?: string;
1007
1109
  /** Optional reported cost — summed across ALL verdicts (failed included). */
1008
1110
  costUsd?: number;
1111
+ /** Optional per-dimension reasoning/evidence. Carried through to
1112
+ * `EnsembleAggregate.verdicts` verbatim — never folded into the math. */
1113
+ detail?: Partial<Record<D, {
1114
+ reasoning?: string;
1115
+ evidence?: string;
1116
+ }>>;
1009
1117
  }
1010
1118
  /** The aggregated ensemble result. */
1011
1119
  interface EnsembleAggregate<D extends string = string> {
@@ -1023,6 +1131,9 @@ interface EnsembleAggregate<D extends string = string> {
1023
1131
  costUsd: number;
1024
1132
  /** First non-empty survivor rationale, or `'llm-judge'`. */
1025
1133
  rationale: string;
1134
+ /** The input verdicts, verbatim — drill-down to raw scores, `detail`
1135
+ * reasoning/evidence, and per-verdict cost without re-running judges. */
1136
+ verdicts: JudgeVerdict<D>[];
1026
1137
  }
1027
1138
  /**
1028
1139
  * Reduce per-judge verdicts to one aggregate. Generic over the rubric: pass the
@@ -1037,6 +1148,139 @@ interface EnsembleAggregate<D extends string = string> {
1037
1148
  */
1038
1149
  declare function aggregateJudgeVerdicts<D extends string>(verdicts: readonly JudgeVerdict<D>[], dimensionKeys: readonly D[], weights?: Partial<Record<D, number>>): EnsembleAggregate<D>;
1039
1150
 
1151
+ /**
1152
+ * Wrap a single judge LLM call with retry, optional fallback-model
1153
+ * rotation, exponential backoff, and a typed `JudgeRetryOutcome`. Callers
1154
+ * MUST inspect `succeeded` before using `value`; on failure the library
1155
+ * returns `value: null` rather than substituting a default, so a judge
1156
+ * abort cannot silently corrupt a downstream composite.
1157
+ *
1158
+ * Reporting contract: callers ship `TrialResult.judgeSucceeded = succeeded`
1159
+ * and `TrialResult.judgeAttempts = attempts` so `aggregateTrialsByMode`
1160
+ * with `mode: 'exclude-failed'` drops the trial.
1161
+ */
1162
+ /** Retry policy for judge LLM calls. */
1163
+ interface JudgeRetryPolicy {
1164
+ /** Max attempts per model. Default 3 (one initial + two retries). */
1165
+ maxAttempts?: number;
1166
+ /** Per-attempt timeout in ms. Default 300_000. */
1167
+ timeoutMs?: number;
1168
+ /**
1169
+ * Models to try, in order. The first model is the primary; subsequent
1170
+ * models are fallbacks invoked only when ALL retries on the previous
1171
+ * model have been exhausted. Example: `['claude-code/sonnet', 'kimi-code/k2p6']`
1172
+ * runs claude-code up to maxAttempts times, then falls back to kimi.
1173
+ * If omitted, the caller's judge function controls model selection and
1174
+ * the retries apply to that single model.
1175
+ */
1176
+ models?: readonly string[];
1177
+ /** Exponential backoff function, default `attempt → min(500 * 2^attempt, 16_000)`. */
1178
+ backoffMs?: (attempt: number) => number;
1179
+ /**
1180
+ * Predicate deciding whether an error should trigger a retry. Defaults to
1181
+ * `isTransientLlmError` — the package-wide classifier shared with
1182
+ * `callLlm` — which retries aborts/timeouts, network faults, HTTP/2
1183
+ * transport faults, and any `LlmCallError` with status in {429,502,503,504}.
1184
+ * JSON-parse and schema-rejection errors are NOT retriable (the model
1185
+ * needs prompt adjustment, not another shot).
1186
+ */
1187
+ isRetryable?: (err: unknown) => boolean;
1188
+ }
1189
+ /** Outcome of a wrapped judge invocation. */
1190
+ interface JudgeRetryOutcome<T> {
1191
+ /** The judge's returned value when `succeeded === true`. */
1192
+ value: T | null;
1193
+ /** True iff one of the attempts completed without throwing. */
1194
+ succeeded: boolean;
1195
+ /** Total attempts made across all models. */
1196
+ attempts: number;
1197
+ /** Which model the successful attempt used (when succeeded). */
1198
+ modelUsed?: string;
1199
+ /** Last error captured when `succeeded === false`. */
1200
+ error?: Error;
1201
+ /** Per-attempt error log for forensics. */
1202
+ attemptErrors: Array<{
1203
+ attempt: number;
1204
+ model: string;
1205
+ error: string;
1206
+ }>;
1207
+ }
1208
+ /**
1209
+ * Wrap a judge call with retry + fallback-model + typed outcome semantics.
1210
+ *
1211
+ * The `judgeFn` signature is `(model: string, signal: AbortSignal) => Promise<T>`.
1212
+ * The signal will be aborted at `timeoutMs`. Callers should pass the signal
1213
+ * to their underlying fetch/SDK call so the abort actually fires.
1214
+ *
1215
+ * Returns a typed outcome — callers MUST inspect `succeeded` before using
1216
+ * `value`. The library refuses to default to a silent zero score because a
1217
+ * synthetic zero is indistinguishable from a real low score downstream.
1218
+ */
1219
+ declare function withJudgeRetry<T>(judgeFn: (model: string, signal: AbortSignal) => Promise<T>, policy?: JudgeRetryPolicy): Promise<JudgeRetryOutcome<T>>;
1220
+
1221
+ /**
1222
+ * Multi-model judge panel — `ensembleJudge` builds a campaign `JudgeConfig`
1223
+ * that fans one artifact out to K judge models and reduces their verdicts
1224
+ * through `aggregateJudgeVerdicts` (src/judge-ensemble.ts).
1225
+ *
1226
+ * The panel is the fail-loud composition of the substrate's existing judge
1227
+ * primitives:
1228
+ * - `assertCrossFamily` (construction-time) — a single-family panel is
1229
+ * correlated bias, not independent signal.
1230
+ * - `withJudgeRetry` (per model, opt-in) — transient-fault retry with a
1231
+ * typed outcome; a judge that exhausts retries is recorded as failed,
1232
+ * never folded into a zero.
1233
+ * - `aggregateJudgeVerdicts` — the pure reducer; throws when EVERY judge
1234
+ * failed so a silent zero can't reach the gate.
1235
+ *
1236
+ * The returned `JudgeScore` is on the campaign [0,1] scale and carries the
1237
+ * ensemble extras (`maxDisagreement`, `failedJudges`, `perJudge`) declared
1238
+ * on the canonical `JudgeScore` in src/campaign/types.ts.
1239
+ */
1240
+
1241
+ interface EnsembleJudgeOptions<D extends string> {
1242
+ /** Judge name — becomes the returned `JudgeConfig.name`. */
1243
+ name: string;
1244
+ /** Rubric dimensions every model scores. Keys of the verdict's `perDimension`. */
1245
+ dimensions: D[];
1246
+ /** Judge model ids — one `scoreWith` call per entry. List a model twice to
1247
+ * sample it twice (votes are suffix-keyed `model#2` so none overwrite). */
1248
+ models: string[];
1249
+ /**
1250
+ * Score the artifact with one model. Throw (or reject) on failure — the
1251
+ * panel records that model as a failed judge; it is never folded into a
1252
+ * zero. Verdict scores are clamped to [0,1] by the reducer.
1253
+ */
1254
+ scoreWith: (model: string, input: {
1255
+ artifact: unknown;
1256
+ scenario?: unknown;
1257
+ }) => Promise<JudgeVerdict<D>>;
1258
+ /**
1259
+ * Per-model retry policy, applied via `withJudgeRetry`. The panel's
1260
+ * `models` list drives the fan-out, so `retry.models` (the fallback
1261
+ * rotation) is overridden to each panel model in turn.
1262
+ */
1263
+ retry?: JudgeRetryPolicy;
1264
+ /** Enforce `assertCrossFamily` over `models` at construction. Default true.
1265
+ * Opt out only for deliberate single-family panels (e.g. self-consistency
1266
+ * sampling of one model). */
1267
+ crossFamily?: boolean;
1268
+ /** Composite weights forwarded to `aggregateJudgeVerdicts`: a partial map
1269
+ * selects AND weights exactly the named dimensions. Omit for uniform. */
1270
+ weights?: Partial<Record<D, number>>;
1271
+ }
1272
+ /**
1273
+ * Build a campaign-shaped `JudgeConfig` whose `score()` runs every panel
1274
+ * model in parallel and reduces the surviving verdicts to one canonical
1275
+ * `JudgeScore` in [0,1].
1276
+ *
1277
+ * Failure semantics: a model whose `scoreWith` throws (or exhausts `retry`)
1278
+ * lands in `failedJudges` and is excluded from the means. When EVERY model
1279
+ * fails, `aggregateJudgeVerdicts` throws — the campaign engine records a
1280
+ * failed cell instead of averaging a fabricated zero.
1281
+ */
1282
+ declare function ensembleJudge<D extends string>(opts: EnsembleJudgeOptions<D>): JudgeConfig<unknown>;
1283
+
1040
1284
  type SandboxJudgeKind = 'compiler' | 'test' | 'linter' | 'security';
1041
1285
  interface SandboxJudgeSpec {
1042
1286
  id: string;
@@ -1275,118 +1519,6 @@ declare class BudgetGuard {
1275
1519
  get state(): Record<keyof BudgetSpec, number>;
1276
1520
  }
1277
1521
 
1278
- /**
1279
- * CostLedger — per-run token + USD accounting with an explicit `costUnknown`
1280
- * axis, folded over the substrate's pricing resolver.
1281
- *
1282
- * `estimateCost` already resolves a model id to a price (exact table, then
1283
- * family regex) and warns-once on a miss, but it returns 0 for an unpriced
1284
- * model — indistinguishable downstream from a genuinely free run. Four
1285
- * consumers re-wrap it to surface that distinction (physim's `costForUsage` /
1286
- * `modelPriceKey` is the cleanest), and to bucket spend by "channel" (the
1287
- * logical role of the call: agent / judge / verifier / …) so a dashboard can
1288
- * answer "how much did judging cost vs the agent itself?".
1289
- *
1290
- * This is the canonical version. `modelPriceKey` exposes the resolver's verdict
1291
- * as a stable key (or null). `CostLedger` folds usage records into per-channel
1292
- * and total rollups, tracks `unpricedModels` so a $0 is never mistaken for a
1293
- * measured zero, and computes cost-per-completed-task.
1294
- */
1295
- /** Logical role of an LLM call. Free-form union — consumers add their own
1296
- * channels; the rollup keys on whatever string is supplied. */
1297
- type CostChannel = 'agent' | 'judge' | 'verifier' | 'analyst' | 'driver' | (string & {});
1298
- interface CostUsage {
1299
- inputTokens: number;
1300
- outputTokens: number;
1301
- cachedTokens?: number;
1302
- }
1303
- /**
1304
- * Resolve a model id to the stable pricing key the substrate's `MODEL_PRICING`
1305
- * / family resolver would use, or null when the id is unpriced. A non-null
1306
- * return means `estimateCost` will produce a real number for this id; null
1307
- * means any cost computed is `costUnknown` and the 0 must not aggregate as a
1308
- * measured cost.
1309
- */
1310
- declare function modelPriceKey(model: string): string | null;
1311
- interface CostResult {
1312
- costUsd: number;
1313
- /** True when `model` has no pricing — the 0 is "not priced", NOT "free". */
1314
- costUnknown: boolean;
1315
- }
1316
- /**
1317
- * Cost for one usage record. Resolves pricing via the substrate resolver and
1318
- * flags `costUnknown` when the model is unpriced so the 0 is observable rather
1319
- * than silently emitted as a measured cost. Cached tokens are billed at the
1320
- * model's input rate when present (no separate cache-discount table — callers
1321
- * that need provider-specific cache pricing supply `actualCostUsd` upstream).
1322
- */
1323
- declare function costForUsage(model: string, usage: CostUsage): CostResult;
1324
- interface CostLedgerEntry extends CostUsage {
1325
- model: string;
1326
- channel: CostChannel;
1327
- costUsd: number;
1328
- costUnknown: boolean;
1329
- /** Override the estimate with an observed provider cost. */
1330
- actualCostUsd?: number;
1331
- /** Free-form tags (scenario id, variant id, round, …). */
1332
- tags?: Record<string, string>;
1333
- timestamp: number;
1334
- }
1335
- interface ChannelRollup {
1336
- channel: CostChannel;
1337
- calls: number;
1338
- inputTokens: number;
1339
- outputTokens: number;
1340
- cachedTokens: number;
1341
- costUsd: number;
1342
- /** Calls whose model was unpriced (their costUsd is 0-but-unknown). */
1343
- unpricedCalls: number;
1344
- }
1345
- interface CostLedgerSummary {
1346
- totalCalls: number;
1347
- inputTokens: number;
1348
- outputTokens: number;
1349
- cachedTokens: number;
1350
- totalCostUsd: number;
1351
- /** Per-channel breakdown, sorted by channel name. */
1352
- byChannel: ChannelRollup[];
1353
- /** Distinct unpriced model ids seen — non-empty means totalCostUsd is a
1354
- * lower bound (some calls priced to an unknown 0). */
1355
- unpricedModels: string[];
1356
- /** True when no unpriced model was charged — totalCostUsd is then exact. */
1357
- fullyPriced: boolean;
1358
- }
1359
- /**
1360
- * Append-only ledger of LLM spend for a single run. Record each call with its
1361
- * channel; read per-channel and total rollups plus the unpriced-model set.
1362
- * Pure accounting — no I/O. The `markCompleted` / `costPerCompletedTask` pair
1363
- * answers "dollars per finished task", the metric every optimizer's
1364
- * quality-vs-cost tradeoff needs.
1365
- */
1366
- declare class CostLedger {
1367
- private readonly entries;
1368
- private completedTasks;
1369
- /**
1370
- * Record one LLM call. The cost is computed from pricing unless
1371
- * `actualCostUsd` is supplied (a finite observed cost from the provider
1372
- * response), in which case `costUnknown` is false regardless of pricing.
1373
- */
1374
- record(input: {
1375
- model: string;
1376
- channel: CostChannel;
1377
- usage: CostUsage;
1378
- actualCostUsd?: number;
1379
- tags?: Record<string, string>;
1380
- timestamp?: number;
1381
- }): CostLedgerEntry;
1382
- /** Increment the completed-task counter (used for cost-per-completed-task). */
1383
- markCompleted(count?: number): void;
1384
- list(): CostLedgerEntry[];
1385
- summary(): CostLedgerSummary;
1386
- /** Total spend divided by completed tasks; null when nothing completed. */
1387
- costPerCompletedTask(): number | null;
1388
- }
1389
-
1390
1522
  /**
1391
1523
  * Cost tracker — token + USD accounting per scenario and per run.
1392
1524
  *
@@ -2107,38 +2239,6 @@ declare function diffScorecard(scorecard: Scorecard, opts?: DiffScorecardOptions
2107
2239
  */
2108
2240
  declare function formatScorecardDiff(diff: ScorecardDiff): string;
2109
2241
 
2110
- /**
2111
- * Series convergence — detects whether a sequence of scalar measurements
2112
- * is stabilizing, drifting, or noisy.
2113
- *
2114
- * Lifted from ADC convergence.ts. The per-turn `ConvergenceTracker` is
2115
- * about progress *within* a single run; this module is about drift
2116
- * *across* runs (e.g. "are my nightly eval scores stabilizing?").
2117
- *
2118
- * Three signals:
2119
- * - stabilized: last K values have low variance (< epsilon) — done
2120
- * - drifting: recent trend is monotonic and beyond noise — regressing or improving
2121
- * - noisy: neither — keep iterating, but flag as untrustworthy for gating
2122
- */
2123
- interface SeriesConvergenceOptions {
2124
- /** Window size for "recent" analysis (default 5). */
2125
- window?: number;
2126
- /** Coefficient-of-variation threshold below which the window is stabilized (default 0.05 = 5%). */
2127
- stableCv?: number;
2128
- /** Minimum monotone run length to call drift (default 3). */
2129
- driftRun?: number;
2130
- }
2131
- interface SeriesConvergenceResult {
2132
- state: 'stabilized' | 'drifting-up' | 'drifting-down' | 'noisy' | 'insufficient-data';
2133
- windowMean: number;
2134
- windowCv: number;
2135
- /** Longest monotonic run at the tail of the series (positive for up, negative for down). */
2136
- tailRun: number;
2137
- /** True when n ≥ window AND windowCv ≤ stableCv. */
2138
- stable: boolean;
2139
- }
2140
- declare function analyzeSeries(values: number[], options?: SeriesConvergenceOptions): SeriesConvergenceResult;
2141
-
2142
2242
  /**
2143
2243
  * SLO gates — quantified pass/fail primitives beyond score thresholds.
2144
2244
  *
@@ -2338,6 +2438,185 @@ interface UiFinding {
2338
2438
  createdAt?: string;
2339
2439
  }
2340
2440
 
2441
+ /**
2442
+ * Trace contracts — finite-trace temporal assertions over span sequences.
2443
+ *
2444
+ * Five LTLf operators over one ordered span sequence — `always(p)`,
2445
+ * `never(p)`, `eventually(p)`, `precedes(a, b)`, `neverUnless(p, prior)` —
2446
+ * deterministic and judge-free. No nesting: each rule is one operator over
2447
+ * flat `SpanPredicate`s; compose richer checks with multiple rules.
2448
+ *
2449
+ * A built `TraceContract` is a serializable plain object (RegExp matchers
2450
+ * are normalized to `SerializedRegex`), so ONE contract definition is
2451
+ * dual-use:
2452
+ *
2453
+ * - recorded eval traces — `evaluateTraceContract(contract, await
2454
+ * store.spans({ runId }))`, or via the behavior DSL:
2455
+ * `expectAgent(store, runId).toSatisfyContract(contract)`.
2456
+ * - the production OTLP stream — `ExportableSpan`s flattened by
2457
+ * `trace/otel-bridge` satisfy `ContractSpan` structurally. A production
2458
+ * monitor implements `OtelExporter`, buffers `exportSpan` payloads per
2459
+ * trace, and runs `checkTraceContracts(buffer, contracts)` on flush:
2460
+ *
2461
+ * const buffer: ContractSpan[] = []
2462
+ * const monitor: OtelExporter = {
2463
+ * exportSpan: (s) => { buffer.push(s) },
2464
+ * flush: async () => {
2465
+ * const { allValid, verdicts } = checkTraceContracts(buffer, contracts)
2466
+ * if (!allValid) alert(verdicts)
2467
+ * },
2468
+ * shutdown: async () => {},
2469
+ * }
2470
+ * const store = createOtelTracingStore(inner, monitor, runId)
2471
+ *
2472
+ * `custom` predicate functions are the one non-serializable escape hatch:
2473
+ * the builder stamps `requiresCustom: true` (which DOES survive JSON) so a
2474
+ * deserialized contract that lost its function fails loud at evaluation
2475
+ * instead of silently weakening.
2476
+ *
2477
+ * Naming: the root barrel exports ci-gate's threshold-contract
2478
+ * `evaluateContract`, so the evaluators here are `evaluateTraceContract` /
2479
+ * `checkTraceContracts`.
2480
+ */
2481
+
2482
+ /**
2483
+ * Minimal structural span the checker reads. Both the eval-side `Span`
2484
+ * (trace/schema) and the OTLP-flattened `ExportableSpan` (trace/otel-export)
2485
+ * satisfy it; any other producer only needs these fields.
2486
+ */
2487
+ interface ContractSpan {
2488
+ spanId?: string;
2489
+ name?: string;
2490
+ kind?: string;
2491
+ startedAt?: number;
2492
+ status?: string;
2493
+ error?: string;
2494
+ /** Typed field on eval-side ToolSpans; OTLP flattenings drop it (see
2495
+ * `tool` matching order in {@link matchSpan}). */
2496
+ toolName?: string;
2497
+ attributes?: Record<string, unknown>;
2498
+ }
2499
+ /** JSON-safe RegExp form — what the builder normalizes RegExp matchers to. */
2500
+ interface SerializedRegex {
2501
+ $regex: string;
2502
+ flags: string;
2503
+ }
2504
+ type TextMatcher = string | RegExp | SerializedRegex;
2505
+ /**
2506
+ * Proposition over one span. All specified fields must match (AND).
2507
+ * At least one field is required — an empty predicate would match every
2508
+ * span and is rejected.
2509
+ *
2510
+ * `tool` resolution order covers both span shapes: `span.toolName` (typed
2511
+ * ToolSpan) → `attributes['tool.name']` / `attributes['toolName']`
2512
+ * (OTLP-flat attribute conventions) → `span.name` when `kind === 'tool'`
2513
+ * (otel-bridge's `ExportableSpan`, which drops `toolName`).
2514
+ *
2515
+ * `attr` values match by strict equality, or regex-test when the value is a
2516
+ * RegExp/SerializedRegex and the attribute is a string. Structured attribute
2517
+ * values need `custom`.
2518
+ */
2519
+ interface SpanPredicate {
2520
+ name?: TextMatcher;
2521
+ tool?: TextMatcher;
2522
+ attr?: Record<string, unknown>;
2523
+ custom?: (span: ContractSpan) => boolean;
2524
+ /** Stamped by the builder when `custom` is present. Survives JSON while
2525
+ * the function does not, so evaluation of a deserialized contract throws
2526
+ * instead of silently dropping the check. */
2527
+ requiresCustom?: true;
2528
+ }
2529
+ type ContractRuleKind = 'always' | 'never' | 'eventually' | 'precedes' | 'neverUnless';
2530
+ interface ContractRule {
2531
+ kind: ContractRuleKind;
2532
+ /** Unique within the contract — keys the per-rule score. */
2533
+ label: string;
2534
+ /** Subject predicate for always / never / eventually / neverUnless. */
2535
+ p?: SpanPredicate;
2536
+ /** precedes: the required precondition. */
2537
+ a?: SpanPredicate;
2538
+ /** precedes: the guarded match — every b-match needs an earlier a-match. */
2539
+ b?: SpanPredicate;
2540
+ /** neverUnless: the authorizing earlier match. */
2541
+ prior?: SpanPredicate;
2542
+ }
2543
+ /** Serializable plain object — `traceContract(name)....build()` output. */
2544
+ interface TraceContract {
2545
+ name: string;
2546
+ rules: ContractRule[];
2547
+ }
2548
+ interface ContractViolation {
2549
+ rule: string;
2550
+ spanId?: string;
2551
+ detail: string;
2552
+ }
2553
+ interface ContractVerdict extends DefaultVerdict {
2554
+ /** Contract name — keys this verdict in multi-contract reports. */
2555
+ contract: string;
2556
+ valid: boolean;
2557
+ /** Fraction of rules passing, in [0, 1]. */
2558
+ score: number;
2559
+ /** Per-rule 0|1 keyed by rule label. */
2560
+ scores: Record<string, number>;
2561
+ violations: ContractViolation[];
2562
+ }
2563
+ interface ContractCheckResult {
2564
+ verdicts: ContractVerdict[];
2565
+ allValid: boolean;
2566
+ }
2567
+ /** Test one span against one predicate. All specified fields must match. */
2568
+ declare function matchSpan(span: ContractSpan, predicate: SpanPredicate): boolean;
2569
+ declare class TraceContractBuilder {
2570
+ private readonly name;
2571
+ private readonly rules;
2572
+ constructor(name: string);
2573
+ /** Every span in the trace must satisfy `p`. */
2574
+ always(p: SpanPredicate, label?: string): this;
2575
+ /** No span in the trace may satisfy `p`. */
2576
+ never(p: SpanPredicate, label?: string): this;
2577
+ /** At least one span in the trace must satisfy `p`. */
2578
+ eventually(p: SpanPredicate, label?: string): this;
2579
+ /** Every `b`-match must have a strictly earlier `a`-match. */
2580
+ precedes(a: SpanPredicate, b: SpanPredicate, label?: string): this;
2581
+ /** Every `p`-match is a violation unless a strictly earlier `prior`-match exists. */
2582
+ neverUnless(p: SpanPredicate, prior: SpanPredicate, label?: string): this;
2583
+ build(): TraceContract;
2584
+ private add;
2585
+ }
2586
+ declare function traceContract(name: string): TraceContractBuilder;
2587
+ /**
2588
+ * Evaluate one contract over a span sequence. Pure and synchronous — works
2589
+ * on `Span[]` from a TraceStore, `ExportableSpan[]` from the otel-bridge
2590
+ * flattening, or any array satisfying `ContractSpan`.
2591
+ */
2592
+ declare function evaluateTraceContract(contract: TraceContract, spans: readonly ContractSpan[]): ContractVerdict;
2593
+ /**
2594
+ * Evaluate many contracts over one span sequence. Throws on an empty
2595
+ * contract list — `allValid: true` over zero contracts is a silent pass.
2596
+ */
2597
+ declare function checkTraceContracts(spans: readonly ContractSpan[], contracts: readonly TraceContract[]): ContractCheckResult;
2598
+ interface ContractJudgeOptions<TArtifact, TScenario extends Scenario$1 = Scenario$1> {
2599
+ /**
2600
+ * Project the span sequence out of a cell's artifact. `JudgeConfig.score`
2601
+ * receives only `{ artifact, scenario, signal }` (src/campaign/types.ts) —
2602
+ * spans are NOT reachable generically — so the consumer supplies this
2603
+ * explicit extraction (e.g. dispatch writes spans into the artifact, or
2604
+ * closes over a per-cell TraceStore read).
2605
+ */
2606
+ spans: (input: {
2607
+ artifact: TArtifact;
2608
+ scenario: TScenario;
2609
+ }) => readonly ContractSpan[];
2610
+ /** Judge name in campaign reports. Default 'trace-contracts'. */
2611
+ name?: string;
2612
+ }
2613
+ /**
2614
+ * Adapt trace contracts to a campaign `JudgeConfig`. One judge dimension per
2615
+ * contract (key = contract name, value = its rule-pass fraction); composite
2616
+ * is the mean across contracts. Deterministic — no LLM call.
2617
+ */
2618
+ declare function contractJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1>(contracts: readonly TraceContract[], opts: ContractJudgeOptions<TArtifact, TScenario>): JudgeConfig<TArtifact, TScenario>;
2619
+
2341
2620
  /**
2342
2621
  * Behavior DSL — pytest-style assertions over a run's trajectory.
2343
2622
  *
@@ -2376,6 +2655,9 @@ declare class BehaviorAssertion {
2376
2655
  toolCalls?: number;
2377
2656
  llmTurns?: number;
2378
2657
  }): Expectation;
2658
+ /** Evaluate a finite-trace temporal contract (`traceContract(...)`) over
2659
+ * this run's span sequence. See `trace-contracts.ts` for the operators. */
2660
+ toSatisfyContract(contract: TraceContract): Expectation;
2379
2661
  toNeverCall(toolName: string): Expectation;
2380
2662
  }
2381
2663
  declare class CallExpectation implements Expectation {
@@ -2816,86 +3098,6 @@ declare function promptBisect(options: {
2816
3098
  offendingParagraphIndex?: number;
2817
3099
  }>;
2818
3100
 
2819
- /**
2820
- * Counterfactual replay — "what would have happened if we'd changed
2821
- * exactly one thing at turn N?"
2822
- *
2823
- * The framework does NOT drive the agent — it sets up the replay
2824
- * context (prior spans, prior state, mutation spec) and records the
2825
- * resulting divergence. Consumers supply an `executeFrom(ctx)` callback
2826
- * that runs their agent starting from turn N with the mutation applied.
2827
- *
2828
- * Counterfactual runs are recorded as a new Run with `layer='meta'` and
2829
- * `parentRunId = originalRunId`, so downstream diff + correlation
2830
- * pipelines see them natively.
2831
- */
2832
-
2833
- type CounterfactualMutation = {
2834
- kind: 'swap-model';
2835
- at: number;
2836
- newModel: string;
2837
- } | {
2838
- kind: 'swap-tool-result';
2839
- at: number;
2840
- newResult: unknown;
2841
- } | {
2842
- kind: 'truncate-after';
2843
- at: number;
2844
- } | {
2845
- kind: 'inject-system-message';
2846
- at: number;
2847
- content: string;
2848
- } | {
2849
- kind: 'custom';
2850
- at: number;
2851
- describe: string;
2852
- apply: (step: TrajectoryStep) => TrajectoryStep;
2853
- };
2854
- interface CounterfactualContext {
2855
- originalRunId: string;
2856
- originalTrajectory: Trajectory;
2857
- /** Steps up to (but not including) the mutation point — the prefix the
2858
- * replayed agent inherits as its prior conversation/tool history. */
2859
- prefix: TrajectoryStep[];
2860
- mutation: CounterfactualMutation;
2861
- /** Pre-applied mutation on the step at `mutation.at`. Consumers use this
2862
- * as the FIRST step the replayed agent emits (they decide whether to
2863
- * re-emit it or continue from there). */
2864
- mutatedStep: TrajectoryStep;
2865
- }
2866
- interface CounterfactualResult {
2867
- counterfactualRunId: string;
2868
- originalRunId: string;
2869
- mutation: CounterfactualMutation;
2870
- /** Structured delta summary — caller can extend via scoring. */
2871
- delta: {
2872
- originalOutcomeScore: number | null;
2873
- counterfactualOutcomeScore: number | null;
2874
- deltaScore: number | null;
2875
- };
2876
- }
2877
- interface CounterfactualRunner {
2878
- /**
2879
- * Execute the agent from `ctx.prefix` with the mutation applied.
2880
- * MUST emit spans into the provided emitter so they become part of
2881
- * the counterfactual run. MUST call emitter.endRun() with a verdict.
2882
- */
2883
- executeFrom: (ctx: CounterfactualContext, emitter: TraceEmitter) => Promise<void>;
2884
- }
2885
- declare function runCounterfactual(store: TraceStore, originalRunId: string, mutation: CounterfactualMutation, runner: CounterfactualRunner): Promise<CounterfactualResult>;
2886
- /**
2887
- * Aggregate a batch of counterfactuals into a simple attribution table:
2888
- * which mutation kinds move outcomes most? (Useful when you run a grid
2889
- * over the same trajectory — swap-model at every llm span, swap-tool
2890
- * at every tool span — and want a ranked summary.)
2891
- */
2892
- declare function attributeCounterfactuals(results: CounterfactualResult[]): Array<{
2893
- mutationKind: CounterfactualMutation['kind'];
2894
- n: number;
2895
- meanAbsDelta: number;
2896
- meanSignedDelta: number;
2897
- }>;
2898
-
2899
3101
  /**
2900
3102
  * Full cross-trace diff — align two trajectories step-by-step, report
2901
3103
  * per-step score deltas, attribute a variant's total outcome lead to
@@ -2951,131 +3153,6 @@ interface CrossTraceDiffOptions {
2951
3153
  }
2952
3154
  declare function crossTraceDiff(store: TraceStore, runA: string, runB: string, options?: CrossTraceDiffOptions): Promise<CrossTraceDiff>;
2953
3155
 
2954
- /**
2955
- * Pre-registered hypotheses — declare what you're testing BEFORE the
2956
- * run, check it AFTER. Prevents p-hacking, optional stopping, and the
2957
- * "we ran until it looked good" failure mode.
2958
- *
2959
- * Manifest is a plain JSON-friendly object. Sign it with a content hash
2960
- * + timestamp; the registered record becomes immutable. Post-run,
2961
- * evaluate the manifest against observed results — the library refuses
2962
- * to let you re-interpret a different metric as the declared one.
2963
- */
2964
- interface HypothesisManifest {
2965
- id: string;
2966
- /** Human prose — goes into the audit trail. */
2967
- hypothesis: string;
2968
- /** Metric the hypothesis claims to move. */
2969
- metric: string;
2970
- /** 'increase' = candidate should score higher than baseline; 'decrease' = lower. */
2971
- direction: 'increase' | 'decrease';
2972
- /** Minimum effect size to count (same units as the metric). */
2973
- minEffect: number;
2974
- /** Alpha threshold. */
2975
- alpha: number;
2976
- /** Target statistical power at which sample size was pre-computed. */
2977
- power: number;
2978
- /** Declared N per arm before running. */
2979
- preRegisteredN: number;
2980
- /** ISO8601 timestamp the manifest was registered. */
2981
- registeredAt: string;
2982
- /** Optional identifiers to tie into the trace corpus. */
2983
- baselineLabel?: string;
2984
- candidateLabel?: string;
2985
- }
2986
- /**
2987
- * Identifier for the hashing scheme used to produce `contentHash`.
2988
- *
2989
- * `'sha256-content'` — sha256 hex over the canonicalized manifest with
2990
- * the `contentHash` and `algo` fields stripped. Held as a string union
2991
- * so future schemes can be added without breaking parsers; SignedManifest
2992
- * values without `algo` deserialize cleanly because the field is optional.
2993
- */
2994
- type SignedManifestAlgo = 'sha256-content';
2995
- interface SignedManifest extends HypothesisManifest {
2996
- /** sha256 hex of canonicalized manifest (everything except contentHash and algo). */
2997
- contentHash: string;
2998
- /**
2999
- * Algorithm string describing how `contentHash` was produced.
3000
- *
3001
- * Optional on the type so serialized manifests without it still parse,
3002
- * but ALWAYS populated by {@link signManifest}. Consumers that want to
3003
- * enforce a known algorithm should reject manifests where this field
3004
- * is missing or unrecognized.
3005
- */
3006
- algo?: SignedManifestAlgo;
3007
- }
3008
- interface HypothesisResult {
3009
- manifest: SignedManifest;
3010
- observedN: number;
3011
- observedEffect: number;
3012
- observedPValue: number;
3013
- /** True iff the observed effect hits the pre-declared direction with
3014
- * magnitude ≥ minEffect AND p < alpha. */
3015
- confirmed: boolean;
3016
- /** Enumerated reasons the hypothesis was rejected (each a machine-tag). */
3017
- rejectionReasons: Array<'wrong_direction' | 'effect_too_small' | 'not_significant' | 'undersampled'>;
3018
- notes?: string;
3019
- }
3020
- /**
3021
- * Deterministic JSON canonicalization — sort object keys recursively.
3022
- *
3023
- * Two semantically-equal objects produce byte-identical canonicalized output;
3024
- * this is what makes a content-hash stable across encoders, key insertion
3025
- * orders, and runtime versions. Exported for any consumer that needs the same
3026
- * canonicalization guarantee outside the manifest-signing path (e.g., signing
3027
- * an artifact bundle, hashing a dataset version, etc.).
3028
- */
3029
- declare function canonicalize(v: unknown): unknown;
3030
- /**
3031
- * SHA-256 hex (full 64 chars) over the canonicalized JSON encoding of `obj`.
3032
- *
3033
- * The same primitive `signManifest` and `verifyManifest` are built on, exposed
3034
- * directly so consumers signing arbitrary structured content (artifact bundles,
3035
- * production packets, dataset manifests, etc.) don't have to re-derive
3036
- * canonicalize+sha256 from scratch.
3037
- *
3038
- * Stable across:
3039
- * - object key insertion order (canonicalization sorts keys recursively)
3040
- * - encoder choice (UTF-8 via TextEncoder, fixed)
3041
- * - runtime (uses the Web Crypto subtle digest, present in Node ≥18 and browsers)
3042
- *
3043
- * Named `hashJson` to disambiguate from `prompt-registry.ts`'s `hashContent`,
3044
- * which takes a string input and returns a truncated 12-char prompt id.
3045
- * Use `hashJson` when you mean "canonicalize then hash."
3046
- *
3047
- * @example
3048
- * const hash = await hashJson({ id: '1', kind: 'spec' })
3049
- * // 'a3f1...' (64 hex chars)
3050
- */
3051
- declare function hashJson<T>(obj: T): Promise<string>;
3052
- /**
3053
- * Sign a manifest with a SHA-256 content hash.
3054
- *
3055
- * The hash covers the canonicalized manifest with the `contentHash`
3056
- * and `algo` fields stripped; this lets verifiers re-sign the rest and
3057
- * compare. Returned manifest always carries `algo: 'sha256-content'`
3058
- * so downstream consumers can identify the scheme; manifests without
3059
- * `algo` still verify because it is stripped before hashing on both sides.
3060
- */
3061
- declare function signManifest(m: HypothesisManifest): Promise<SignedManifest>;
3062
- /**
3063
- * Verify that a signed manifest has not been tampered with.
3064
- *
3065
- * Strips `contentHash` and `algo` before re-signing so manifests without
3066
- * `algo` verify identically to ones that carry it.
3067
- */
3068
- declare function verifyManifest(m: SignedManifest): Promise<boolean>;
3069
- /**
3070
- * Evaluate a pre-registered hypothesis against observed results.
3071
- * Mechanical — no re-interpretation permitted.
3072
- */
3073
- declare function evaluateHypothesis(manifest: SignedManifest, observed: {
3074
- n: number;
3075
- effect: number;
3076
- pValue: number;
3077
- }): Promise<HypothesisResult>;
3078
-
3079
3156
  /**
3080
3157
  * Active learning — agent-as-scenario-author.
3081
3158
  *
@@ -4502,76 +4579,6 @@ declare function precision<T>(goldens: GoldenSpec[], candidates: T[], options?:
4502
4579
  text?: (candidate: T) => string;
4503
4580
  }): number;
4504
4581
 
4505
- /**
4506
- * Wrap a single judge LLM call with retry, optional fallback-model
4507
- * rotation, exponential backoff, and a typed `JudgeRetryOutcome`. Callers
4508
- * MUST inspect `succeeded` before using `value`; on failure the library
4509
- * returns `value: null` rather than substituting a default, so a judge
4510
- * abort cannot silently corrupt a downstream composite.
4511
- *
4512
- * Reporting contract: callers ship `TrialResult.judgeSucceeded = succeeded`
4513
- * and `TrialResult.judgeAttempts = attempts` so `aggregateTrialsByMode`
4514
- * with `mode: 'exclude-failed'` drops the trial.
4515
- */
4516
- /** Retry policy for judge LLM calls. */
4517
- interface JudgeRetryPolicy {
4518
- /** Max attempts per model. Default 3 (one initial + two retries). */
4519
- maxAttempts?: number;
4520
- /** Per-attempt timeout in ms. Default 300_000. */
4521
- timeoutMs?: number;
4522
- /**
4523
- * Models to try, in order. The first model is the primary; subsequent
4524
- * models are fallbacks invoked only when ALL retries on the previous
4525
- * model have been exhausted. Example: `['claude-code/sonnet', 'kimi-code/k2p6']`
4526
- * runs claude-code up to maxAttempts times, then falls back to kimi.
4527
- * If omitted, the caller's judge function controls model selection and
4528
- * the retries apply to that single model.
4529
- */
4530
- models?: readonly string[];
4531
- /** Exponential backoff function, default `attempt → min(500 * 2^attempt, 16_000)`. */
4532
- backoffMs?: (attempt: number) => number;
4533
- /**
4534
- * Predicate deciding whether an error should trigger a retry. Defaults to
4535
- * `isTransientLlmError` — the package-wide classifier shared with
4536
- * `callLlm` — which retries aborts/timeouts, network faults, HTTP/2
4537
- * transport faults, and any `LlmCallError` with status in {429,502,503,504}.
4538
- * JSON-parse and schema-rejection errors are NOT retriable (the model
4539
- * needs prompt adjustment, not another shot).
4540
- */
4541
- isRetryable?: (err: unknown) => boolean;
4542
- }
4543
- /** Outcome of a wrapped judge invocation. */
4544
- interface JudgeRetryOutcome<T> {
4545
- /** The judge's returned value when `succeeded === true`. */
4546
- value: T | null;
4547
- /** True iff one of the attempts completed without throwing. */
4548
- succeeded: boolean;
4549
- /** Total attempts made across all models. */
4550
- attempts: number;
4551
- /** Which model the successful attempt used (when succeeded). */
4552
- modelUsed?: string;
4553
- /** Last error captured when `succeeded === false`. */
4554
- error?: Error;
4555
- /** Per-attempt error log for forensics. */
4556
- attemptErrors: Array<{
4557
- attempt: number;
4558
- model: string;
4559
- error: string;
4560
- }>;
4561
- }
4562
- /**
4563
- * Wrap a judge call with retry + fallback-model + typed outcome semantics.
4564
- *
4565
- * The `judgeFn` signature is `(model: string, signal: AbortSignal) => Promise<T>`.
4566
- * The signal will be aborted at `timeoutMs`. Callers should pass the signal
4567
- * to their underlying fetch/SDK call so the abort actually fires.
4568
- *
4569
- * Returns a typed outcome — callers MUST inspect `succeeded` before using
4570
- * `value`. The library refuses to default to a silent zero score because a
4571
- * synthetic zero is indistinguishable from a real low score downstream.
4572
- */
4573
- declare function withJudgeRetry<T>(judgeFn: (model: string, signal: AbortSignal) => Promise<T>, policy?: JudgeRetryPolicy): Promise<JudgeRetryOutcome<T>>;
4574
-
4575
4582
  /**
4576
4583
  * LockedJsonlAppender — mutex-serialized JSONL append helper for arbitrary
4577
4584
  * payloads. The reference-replay store does the same thing for typed
@@ -5268,4 +5275,259 @@ declare namespace index {
5268
5275
  export { type index_AgentProfile as AgentProfile, type index_AgentProfileSection as AgentProfileSection, index_BASELINE_ROLES as BASELINE_ROLES, type index_BaselineRoleKey as BaselineRoleKey, type index_ProfileSkill as ProfileSkill, index_applyDomainPatch as applyDomainPatch, index_baselineProfile as baselineProfile, index_baselineProfileFromRole as baselineProfileFromRole, index_engineerRole as engineerRole, index_generalistRole as generalistRole, index_prodProfile as prodProfile, index_profileToSurface as profileToSurface, index_renderProfile as renderProfile, index_researcherRole as researcherRole, index_sectionHash as sectionHash };
5269
5276
  }
5270
5277
 
5271
- export { type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, AgentProfile$1 as AgentProfile, type AgreementResult, type AlignmentOp, AnalyzeTracesInput, AnalyzeTracesOptions, AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, type BackendDescriptor, BaselineReport, BehaviorAssertion, BenchmarkReport, BenchmarkRunner, BenchmarkRunnerConfig, type BisectOptions, type BisectResult, type BisectStep, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, BudgetSpec, type BuildAgreementJudgeOptions, CallExpectation, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CausalAttributionReport, type CellVerdict, type ChannelRollup, ChatRequest, CheckResult, CollectedArtifacts, type CommandRunner, type CompareLabels, CompletionCriterion, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContractMetric, type ContractReport, ConvergenceTracker, type CostChannel, type CostEntry, CostLedger, type CostLedgerEntry, type CostLedgerSummary, type CostResult, type CostSummary, CostTracker, type CostUsage, type CounterfactualContext, type CounterfactualMutation, type CounterfactualResult, type CounterfactualRunner, CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, DEFAULT_AGENT_SLOS, DEFAULT_FINDERS, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, Dataset, DatasetScenario, type DecideNextUserTurnOpts, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, DriverResult, DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type ErrorCountPattern, EvalTraceStore, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentProvenance, type ExperimentRep, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type FactorContribution, type FactorialCell, FeedbackLabel, FeedbackTrajectory, FeedbackTrajectoryStore, type FieldAgreementSpec, type FileChange, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenSeverity, type GoldenSpec, HarnessConfig, type HeldOutPartition, HoldoutAuditor, type HttpGithubClientOptions, type HypothesisManifest, type HypothesisResult, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, type JudgeFamily, type JudgeFleetOptions, JudgeFn, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, JudgeRunner, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, Layer, LayerResult, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmClientOptions, LlmSpan, LockedJsonlAppender, MODEL_PRICING, type MatchResult, type MatcherResult, type MergeOptions, MetricsCollector, type ModelPreflight, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, type MultiToolchainLayerConfig, type Mutator, Mutex, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParseStudentLabel, type PartitionHeldOutOptions, PersonaConfig, type Playbook, type PlaybookEntry, type PoolSlot, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreflightModelsOptions, type PreflightOutcome, ProductClient, ProductClientConfig, type PromptHandle, PromptRegistry, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProvenanceReader, type RecordRunsOptions, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, ReleaseConfidenceScorecard, ReleaseConfidenceThresholds, type RenderStudentPrompt, type RepoRef, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RobustnessResult, Run, type RunCommandInput, type RunCommandResult, type RunDistillationOptions, type RunDistillationResult, RunFilter, RunRecord, type RunRecordBackend, type RunRecordFilter, RunScore, RunScoreWeights, RunSplitTag, SandboxDriver, SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type ScanOptions, Scenario, type ScenarioCost, ScenarioFile, ScenarioRegistry, ScenarioResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SeriesConvergenceOptions, type SeriesConvergenceResult, Severity, type SignedManifest, type SignedManifestAlgo, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SplitGoldOptions, SteeringBundle, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type StepAttribution, type SynthesisReason, type SynthesisTarget, TestResult, type ThresholdContract, TokenCounter, type TokenSpec, TraceEmitter, TraceStore, type TracedAnalystOptions, type TracedJudgeOptions, Trajectory, TrajectoryStep, type TrialTrace, TurnMetrics, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, VerifyContext, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, adversarialJudge, aggregateJudgeVerdicts, aggregatePrReviewScore, analyzeAntiSlop, analyzeSeries, appendScorecard, assertCrossFamily, assertModelsServed, assertSingleBackend, assignHeldOutTag, attributeCounterfactuals, bisect, buildAgreementJudge, buildDriverSystemPrompt, buildReflectionPrompt, buildReviewerPrompt, canaryLeakView, canonicalize, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, codeExecutionJudge, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, compareReferenceReplay, compilerJudge, computeExperimentStats, costForUsage, createAntiSlopJudge, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createIntentMatchJudge, createSandboxPool, crossTraceDiff, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultJudges, defaultParseStudentLabel, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, deployGateLayer, diffScorecard, discoverPersonas, distillPlaybook, estimateCost, estimateTokens, evaluateContract, evaluateHypothesis, evaluateOracles, executeScenario, expectAgent, exportRewardModel, extractAssetUrls, extractErrorCount, fieldAgreement, fileContains, fileExists, fileExperimentStore, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findSkipCountsAsPass, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, ghCliClient, gitProvenanceReader, precision as goldenPrecision, hashContent, hashJson, hashToUnit, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryRunRecordBackend, isModelPriced, isOtelConfigured, jsonShape, jsonlReferenceReplayStore, jsonlRunRecordBackend, judgeFamily, keyPreserved, linterJudge, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, matchGoldens, mergeLayerResults, modelDescriptionBits, modelPriceKey, multiToolchainLayer, notBlocked, paraphraseRobustness, paraphraseRobustnessScenarios, parseGoldJsonl, parseReflectionResponse, partitionHeldOut, passOrthogonality, pixelDeltaRatio, politenessPrefixMutator, preflightModels, printDriverSummary, index as profile, promptBisect, proposeAutomatedPullRequest, proposeSynthesisTargets, recordRuns, recordRunsToScorecard, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, replayScorerOverCorpus, replayTraceThroughJudge, resetLockedAppendersForTesting, resolveModelPricing, rowCount, rowWhere, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runDistillation, runE2EWorkflow, runExpectations, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runReferenceReplay, runScore, runSelfPlay, scanForMuffledGates, scoreContinuity, scorePrReviewComments, scorePrReviewSource, scoreReferenceReplay, securityJudge, sentenceReorderMutator, signManifest, splitGold, statusAdvanced, summarizePrReviewBenchmark, testJudge, textInSnapshot, toLangfuseEnvelope, toPrometheusText, traceJudge, traceJudgeEnsemble, tracedAnalyzeTraces, typoMutator, urlContains, verifyManifest, visualDiff, viteDeployRunner, weightedRecall, whitespaceCollapseMutator, withJudgeRetry, withOtelPipeline, wranglerDeployRunner };
5278
+ /**
5279
+ * Program cost report — a thin projection over `CostLedger.summary()` that
5280
+ * adds the per-model rollup the summary lacks, plus `attachCostToReport`, the
5281
+ * one way every artifact (capsules, campaign results, diagnose reports) gets
5282
+ * its cost stamp.
5283
+ *
5284
+ * Honesty contract carried through from the ledger: `total.unknownEntries`
5285
+ * and `perModel[].unpriced` surface the costUnknown axis — a $0 from an
5286
+ * unpriced model is a lower bound, never a measured zero.
5287
+ */
5288
+
5289
+ interface ModelCostRollup {
5290
+ model: string;
5291
+ usd: number;
5292
+ entries: number;
5293
+ /** ≥1 entry for this model was costUnknown — `usd` is a lower bound. An
5294
+ * `actualCostUsd` override clears the flag for that entry (the dollars are
5295
+ * observed, even when the model has no pricing). */
5296
+ unpriced: boolean;
5297
+ }
5298
+ interface CostReport {
5299
+ /** Per-channel breakdown — `CostLedgerSummary.byChannel` verbatim. */
5300
+ perChannel: ChannelRollup[];
5301
+ total: {
5302
+ usd: number;
5303
+ /** Entries whose cost was unknown — non-zero means `usd` is a lower bound. */
5304
+ unknownEntries: number;
5305
+ };
5306
+ /** Per-model spend, sorted by model id. */
5307
+ perModel: ModelCostRollup[];
5308
+ }
5309
+ /** Project a ledger into the program cost report. Pure — no I/O, no clock. */
5310
+ declare function costReport(ledger: CostLedger): CostReport;
5311
+ /**
5312
+ * Stamp a report-shaped object with its cost projection under the `cost` key.
5313
+ * Generic so capsules, campaign results, and diagnose reports all stamp the
5314
+ * same way. Throws when the report already carries a `cost` key — silently
5315
+ * overwriting an existing stamp would corrupt the artifact's provenance.
5316
+ */
5317
+ declare function attachCostToReport<R extends object>(report: R, ledger: CostLedger): R & {
5318
+ cost: CostReport;
5319
+ };
5320
+
5321
+ /**
5322
+ * ModelSeats — the program's model seating chart.
5323
+ *
5324
+ * One object names which model fills each role in an eval program: the worker
5325
+ * under evaluation, the judge panel, the analyst, the reflection/driver model,
5326
+ * and the verifier. Re-tiering an entire program (economy ↔ frontier) is one
5327
+ * swapped object instead of a hunt through call sites.
5328
+ *
5329
+ * Wiring points — consumers thread seats; this module implements none of them
5330
+ * (those files belong to other surfaces):
5331
+ * - `judges` → `ensembleJudge({ models: seats.judges, … })` (src/judge-panel.ts)
5332
+ * and the `JudgeConfig`s handed to `makeEvalTools({ judges })`
5333
+ * (src/eval-tools.ts).
5334
+ * - `reflection` → `selfImprove({ llm: { model: seats.reflection } })` — the
5335
+ * `gepaDriver` reflection model (src/contract/self-improve.ts);
5336
+ * same seat for any custom `ImprovementDriver`'s LLM.
5337
+ * - `worker` → the dispatch model the agent itself calls — the model an
5338
+ * `AgentProfile` declares.
5339
+ * - `analyst` → the LLM behind `analyzeRuns` / analyst-registry kinds.
5340
+ * - `verifier` → completion-verifier / objective-checker model.
5341
+ * - campaign cells thread `judges` + driver models the same way; that wiring
5342
+ * lands with the campaign surface, not here.
5343
+ *
5344
+ * `resolveSeat` is the only read path: an unset seat with no explicit fallback
5345
+ * throws — a model id is a budget decision, never a silent default.
5346
+ */
5347
+
5348
+ interface ModelSeats {
5349
+ /** The model under evaluation — what the agent itself dispatches with. */
5350
+ worker?: string;
5351
+ /** Judge-panel model ids — thread into `ensembleJudge({ models })`. */
5352
+ judges?: string[];
5353
+ /** Analyst model — `analyzeRuns` / analyst-registry LLM calls. */
5354
+ analyst?: string;
5355
+ /** Reflection/driver model — `gepaDriver` mutation proposals. */
5356
+ reflection?: string;
5357
+ /** Verifier model — completion/objective checking. */
5358
+ verifier?: string;
5359
+ }
5360
+ type SeatName = keyof ModelSeats;
5361
+ type SeatPresetName = keyof typeof seatPresets;
5362
+ /**
5363
+ * Tier presets — plain data, swap or spread freely.
5364
+ *
5365
+ * `economy` uses the fleet-policy ids: every id resolves through the
5366
+ * substrate's family pricing (no costUnknown axis) and the judge trio spans
5367
+ * three provider families (moonshot / deepseek / openai), so it passes
5368
+ * `assertCrossFamily` as-is.
5369
+ *
5370
+ * `frontier` is deliberately EMPTY: entitled frontier ids vary per router
5371
+ * account, and a hardcoded claude/gpt-5 id 401s on keys that lack it. Supply
5372
+ * your own: `{ ...seatPresets.frontier, worker: '<your-frontier-id>', … }` —
5373
+ * `resolveSeat` throws on every seat you haven't filled.
5374
+ */
5375
+ declare const seatPresets: Record<'economy' | 'frontier', ModelSeats>;
5376
+ /** Thrown by `resolveSeat` when a seat is unset and no fallback was given. */
5377
+ declare class SeatUnsetError extends ConfigError {
5378
+ readonly seat: SeatName;
5379
+ constructor(seat: SeatName);
5380
+ }
5381
+ /**
5382
+ * Read one seat. Blank strings and empty arrays count as unset (env-var
5383
+ * plumbing produces them); malformed values (non-string seat, non-array or
5384
+ * blank-entry `judges`) throw `ValidationError`. When the seat is unset, an
5385
+ * explicit `fallback` is returned (`[fallback]` for `judges` — a one-model
5386
+ * panel); without one, `SeatUnsetError`.
5387
+ */
5388
+ declare function resolveSeat(seats: ModelSeats, seat: 'judges', fallback?: string): string[];
5389
+ declare function resolveSeat(seats: ModelSeats, seat: Exclude<SeatName, 'judges'>, fallback?: string): string;
5390
+ declare function resolveSeat(seats: ModelSeats, seat: SeatName, fallback?: string): string | string[];
5391
+
5392
+ /**
5393
+ * Reproducibility attestation for any serializable report object.
5394
+ *
5395
+ * `attest()` binds a report to its content address (sha-256 over canonical
5396
+ * JSON) plus the provenance needed to reproduce it: model versions, seeds,
5397
+ * price-table hash, code SHA, inputs hash. `verifyAttestation()` recomputes
5398
+ * the address and answers "is this the exact report that provenance
5399
+ * describes?" — any single-field tamper changes the hash.
5400
+ *
5401
+ * Layering: content-addressing is the substrate's job; cryptographic SIGNING
5402
+ * (who vouches for the attestation, key management, transparency logs) is the
5403
+ * consumer's layer on top. An `AttestedReport` is a stable byte-identical
5404
+ * payload a consumer can sign — the substrate never holds keys.
5405
+ *
5406
+ * Generic by design: the report parameter is ANY value `canonicalJson`
5407
+ * accepts (campaign results, fuzz capsules, scorecards, cost ledgers). Do not
5408
+ * couple this module to a specific report schema.
5409
+ */
5410
+ /** Hash scheme identifier carried by every attestation. A verifier rejects
5411
+ * unknown algorithms instead of guessing. */
5412
+ declare const ATTESTATION_ALGORITHM: "sha256/canonical-json";
5413
+ interface AttestationProvenance {
5414
+ /** Every model involved in producing the report, name → version/id. */
5415
+ modelVersions: Record<string, string>;
5416
+ /** RNG seeds the run was driven by, when seeded. */
5417
+ seeds?: number[];
5418
+ /** Content hash of the price table used for cost figures — cost numbers
5419
+ * are only reproducible against the same prices. */
5420
+ priceTableHash?: string;
5421
+ /** Git SHA of the code that produced the report. */
5422
+ codeSha: string;
5423
+ /** Content hash of the input set (scenarios, dataset manifest, ...). */
5424
+ inputsHash?: string;
5425
+ /** ISO-8601 timestamp, caller-supplied — the substrate stays clock-free
5426
+ * so attestation is deterministic and testable. */
5427
+ createdAt: string;
5428
+ }
5429
+ interface AttestedReport {
5430
+ /** Hex sha-256 over the canonical JSON of the report. */
5431
+ reportHash: string;
5432
+ provenance: AttestationProvenance;
5433
+ algorithm: typeof ATTESTATION_ALGORITHM;
5434
+ }
5435
+ interface AttestationVerification {
5436
+ valid: boolean;
5437
+ /** Populated iff `valid` is false — names the exact mismatch. */
5438
+ reason?: string;
5439
+ }
5440
+ /**
5441
+ * Content-address a report and bind it to its provenance. Throws (via
5442
+ * `canonicalJson`) if the report contains undefined / function / symbol /
5443
+ * non-finite numbers — a report that cannot be unambiguously serialized
5444
+ * cannot be attested.
5445
+ */
5446
+ declare function attest(report: unknown, provenance: AttestationProvenance): AttestedReport;
5447
+ /**
5448
+ * Verify a report against its attestation. Returns a typed outcome rather
5449
+ * than throwing: an unverifiable report (e.g. one that no longer
5450
+ * canonicalizes) is a verification failure with the cause in `reason`, not a
5451
+ * crash — verifiers run in pipelines that must record WHY, not die.
5452
+ */
5453
+ declare function verifyAttestation(report: unknown, attested: AttestedReport): AttestationVerification;
5454
+
5455
+ /**
5456
+ * Content-addressed judge-verdict caching.
5457
+ *
5458
+ * LAW: cache JUDGE VERDICTS only — judging the same artifact with the same
5459
+ * judge+rubric is pure. NEVER cache agent rollouts. (A router that cached
5460
+ * identical fanout prompts silently destroyed best-of-N diversity; rollout
5461
+ * caching reintroduces that failure class. Judging has no diversity to
5462
+ * destroy — same artifact + same rubric ⇒ same verdict is the desired
5463
+ * property, not a bug.)
5464
+ *
5465
+ * The cache key is a sha-256 over the canonical JSON of everything that can
5466
+ * change a verdict: the artifact content, the scenario id, the judge name,
5467
+ * the full dimension list (key + description — the description IS the rubric
5468
+ * text shown to the judge), and a caller-supplied `judgeVersion`.
5469
+ * `judgeVersion` is REQUIRED: a judge whose prompt/model/ensemble changes
5470
+ * without a version bump would otherwise silently serve stale verdicts.
5471
+ *
5472
+ * Strict canonicalization (`canonicalJson`) throws on undefined / function /
5473
+ * symbol / non-finite numbers — an artifact that cannot be unambiguously
5474
+ * serialized cannot be content-addressed, and coercing it would let two
5475
+ * different artifacts collide on one key.
5476
+ */
5477
+
5478
+ /**
5479
+ * Stable JSON stringify: object keys sorted recursively, so two semantically
5480
+ * equal values produce byte-identical output regardless of key insertion
5481
+ * order. Throws on undefined / function / symbol / NaN / ±Infinity / bigint /
5482
+ * Map / Set — anything JSON.stringify would coerce or drop silently.
5483
+ *
5484
+ * Distinct from `pre-registration.ts`'s `canonicalize`/`hashJson`, which are
5485
+ * permissive (coercion allowed) and async (web-crypto). Use THIS pair when a
5486
+ * hash collision or silent coercion would corrupt a cache key or attestation.
5487
+ */
5488
+ declare function canonicalJson(value: unknown): string;
5489
+ /** Hex sha-256 over `canonicalJson(value)`. The content address used by the
5490
+ * verdict cache and report attestation. */
5491
+ declare function contentHash(value: unknown): string;
5492
+ /** Pluggable verdict store. Sync or async on both legs — `cachedJudge`
5493
+ * awaits the results either way. */
5494
+ interface VerdictCacheStore {
5495
+ get(key: string): Promise<JudgeScore | undefined> | JudgeScore | undefined;
5496
+ set(key: string, score: JudgeScore): Promise<void> | void;
5497
+ }
5498
+ /** Process-local Map-backed store. */
5499
+ declare function inMemoryVerdictCache(): VerdictCacheStore;
5500
+ /**
5501
+ * JSONL-file-backed store: the full file is loaded into an in-memory index at
5502
+ * construction; every `set` appends one line synchronously (durable before
5503
+ * the verdict is returned). A corrupt or malformed line throws at load with
5504
+ * file:line — a skipped line would silently re-judge (cost) or, worse, mask
5505
+ * a half-written file that needs operator attention.
5506
+ */
5507
+ declare function fileVerdictCache(path: string): VerdictCacheStore;
5508
+ interface VerdictCacheStats {
5509
+ hits: number;
5510
+ misses: number;
5511
+ }
5512
+ interface CachedJudgeOptions {
5513
+ /** REQUIRED — part of the cache key. Bump on any change to the judge's
5514
+ * prompt, model, ensemble, or scoring logic; silent judge upgrades must
5515
+ * never serve stale verdicts. */
5516
+ judgeVersion: string;
5517
+ }
5518
+ /** The wrapped judge: same `JudgeConfig` seam, plus hit/miss observability. */
5519
+ type CachedJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1> = JudgeConfig<TArtifact, TScenario> & {
5520
+ stats(): VerdictCacheStats;
5521
+ };
5522
+ /**
5523
+ * Wrap a `JudgeConfig` so repeat judgments of the same artifact are served
5524
+ * from the store instead of re-invoking `score()`. The wrapper is generic
5525
+ * over the judge's own type parameters and preserves `appliesTo` — it is a
5526
+ * drop-in replacement anywhere a `JudgeConfig` is accepted.
5527
+ *
5528
+ * A judge that throws is NOT cached: the error propagates and the next
5529
+ * attempt re-judges (caching a failure would pin a transient outage forever).
5530
+ */
5531
+ declare function cachedJudge<TArtifact, TScenario extends Scenario$1 = Scenario$1>(judge: JudgeConfig<TArtifact, TScenario>, store: VerdictCacheStore, options: CachedJudgeOptions): CachedJudge<TArtifact, TScenario>;
5532
+
5533
+ export { ATTESTATION_ALGORITHM, type ActiveLearningOptions, type AdapterRun, AgentDriver, type AgentDriverConfig, AgentEvalError, AgentProfile$1 as AgentProfile, type AgreementResult, type AlignmentOp, AnalyzeTracesInput, AnalyzeTracesOptions, AnalyzeTracesResult, type AntiSlopConfig, type AntiSlopIssue, type AntiSlopReport, type AssertCrossFamilyOptions, type AssertSingleBackendOptions, type AttestationProvenance, type AttestationVerification, type AttestedReport, type AutoPrClient, AxGepaSteeringOptimizer, type AxSteeringOptimizerConfig, type BackendDescriptor, BaselineReport, BehaviorAssertion, BenchmarkReport, BenchmarkRunner, BenchmarkRunnerConfig, type BisectOptions, type BisectResult, type BisectStep, BudgetBreachError, BudgetGuard, BudgetLedgerEntry, BudgetSpec, type BuildAgreementJudgeOptions, type CachedJudge, type CachedJudgeOptions, CallExpectation, type CanaryAlert, type CanaryKind, type CanaryLeak, type CanaryOptions, type CanaryReport, type CanarySeverity, type CandidateComparison, type CandidateScenario, type CausalAttributionReport, type CellVerdict, ChannelRollup, ChatRequest, CheckResult, CollectedArtifacts, type CommandRunner, type CompareLabels, CompletionCriterion, ConfigError, type ContinuityCheck, type ContinuityCheckResult, type ContinuityReport, type ContinuitySnapshotPair, type ContractCheckResult, type ContractJudgeOptions, type ContractMetric, type ContractReport, type ContractRule, type ContractRuleKind, type ContractSpan, type ContractVerdict, type ContractViolation, ConvergenceTracker, CorrectnessChecker, type CostEntry, CostLedger, type CostReport, type CostSummary, CostTracker, CreateChatClientOpts, type CreateDefaultReviewerOptions, type CreateExperimentInput, type CreateSandboxPoolOpts, CrossFamilyError, type CrossTraceDiff, type CrossTraceDiffOptions, DEFAULT_AGENT_SLOS, DEFAULT_FINDERS, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, Dataset, DatasetScenario, type DecideNextUserTurnOpts, DefaultVerdict, type DeployFamily, type DeployGateLayerInput, type DeployRunResult, type DeployRunner, type DescriptionLengthCandidate, type DescriptionLengthConfig, type DescriptionLengthDecision, type DescriptionLengthEvidence, DescriptionLengthGate, type DescriptionLengthRejectionCode, type DiffScorecardOptions, type DirEntry, type DiscoverPersonasOptions, type DiscoveredPersona, DriverResult, DriverState, DualAgentBench, type DualAgentBenchConfig, type DualAgentReport, type DualAgentRound, type DualAgentScenario, type DualAgentScenarioResult, ERROR_COUNT_PATTERNS, type EnsembleAggregate, type EnsembleJudgeOptions, type ErrorCountPattern, type EvalToolDef, EvalTraceStore, type EvolutionRound, type ExecutorConfig, type Expectation, type Experiment, type ExperimentProvenance, type ExperimentRep, type ExperimentStats, type ExperimentStore, ExperimentTracker, type ExperimentTrackerOptions, type ExperimentVerdict, type ExportedRewardModel, type ExtractOptions, type ExtractResult, type FactorContribution, type FactorialCell, FeedbackLabel, FeedbackTrajectory, FeedbackTrajectoryStore, type FieldAgreementSpec, type FileChange, type FlowAction, type FlowLayerEnv, type FlowLayerFactoryInput, type FlowRunner, type FlowRunnerStepResult, type FlowSpec, type FlowStep, type GhCliClientOptions, type GoldScenario, type GoldSplit, type GoldenSeverity, type GoldenSpec, HarnessConfig, type HeldOutPartition, HoldoutAuditor, type HttpGithubClientOptions, INTENT_MATCH_JUDGE_VERSION, type ImageData, type ImprovementThresholds, type ImprovementVerdictResult, InMemoryWorkspaceInspector, type InferenceScorer, type InspectorContext, type IntentMatchInput, type IntentMatchOptions, type IntentMatchResult, type InteractionContribution, JudgeError, type JudgeFamily, type JudgeFleetOptions, JudgeFn, JudgeParseError, type JudgeReplayResult, type JudgeRetryOutcome, type JudgeRetryPolicy, JudgeRunner, type JudgeVerdict, type KeywordConceptSpec, type KeywordCoverageFinding, type KeywordCoverageOptions, type KeywordCoverageResult, type LangfuseEnvelope, type LangfuseGeneration, type LangfuseScore, Layer, LayerResult, type LiveProofArtifact, type LiveProofConfig, type LiveProofContext, type LiveProofResult, LlmClientOptions, LlmSpan, LockedJsonlAppender, MODEL_PRICING, type MakeEvalToolsConfig, type MatchResult, type MatcherResult, type MergeOptions, MetricsCollector, type ModelCostRollup, type ModelPreflight, type ModelSeats, ModelsUnreachableError, type MuffledFinder, type MuffledFinding, type MultiToolchainLayerConfig, type Mutator, Mutex, type Oracle, type OracleObservation, type OracleReport, type OracleResult, type OrthogonalityInput, type OrthogonalityResult, OtelExportConfig, OtelExporter, type OtelPipelineHandle, type OtelPipelineOptions, PairwiseSteeringOptimizer, type ParaphraseRobustnessScenarioInput, type ParaphraseRobustnessScenarioResult, type ParseStudentLabel, type PartitionHeldOutOptions, PersonaConfig, type Playbook, type PlaybookEntry, type PoolSlot, type PrReviewAuditCase, type PrReviewBenchmarkSummary, type PrReviewComment, type PrReviewMatchedFinding, type PrReviewOutcome, type PrReviewReferenceFinding, type PrReviewScore, type PrReviewScoreWeights, type PrReviewSeverity, type PrReviewSource, type PreflightModelsOptions, type PreflightOutcome, ProductClient, ProductClientConfig, type PromptHandle, PromptRegistry, type ProposeAutomatedPullRequestInput, type ProposeAutomatedPullRequestResult, type ProvenanceReader, type RecordRunsOptions, type ReferenceMatchResult, type ReferenceReplayAdapter, type ReferenceReplayAdapterFn, type ReferenceReplayAdapterLike, type ReferenceReplayAggregate, type ReferenceReplayCandidate, type ReferenceReplayCase, type ReferenceReplayCaseRun, type ReferenceReplayExecutionScenario, type ReferenceReplayItem, type ReferenceReplayMatch, type ReferenceReplayMatchStrategy, type ReferenceReplayMatcher, type ReferenceReplayPromotionDecision, type ReferenceReplayPromotionPolicy, type ReferenceReplayRun, type ReferenceReplayRunContext, type ReferenceReplayRunOptions, type ReferenceReplayRunStore, type ReferenceReplayScenario, type ReferenceReplayScenarioScore, type ReferenceReplayScore, type ReferenceReplayScoreOptions, type ReferenceReplaySplit, type ReferenceReplaySplitComparison, type ReferenceReplaySteeringRowsOptions, type ReflectionContext, type ReflectionProposal, ReleaseConfidenceScorecard, ReleaseConfidenceThresholds, type RenderStudentPrompt, type RepoRef, type ReviewerMemoryEntry, type ReviewerOutput, type ReviewerPromptInput, type ReviewerSoftFailDefaults, type ReviewerVerificationSummary, type RobustnessResult, Run, type RunCommandInput, type RunCommandResult, type RunDistillationOptions, type RunDistillationResult, RunFilter, RunRecord, type RunRecordBackend, type RunRecordFilter, RunScore, RunScoreWeights, RunSplitTag, SandboxDriver, SandboxHarnessResult, type SandboxJudgeKind, type SandboxJudgeResult, type SandboxJudgeSpec, type SandboxPool, type ScanOptions, Scenario, type ScenarioCost, ScenarioFile, ScenarioRegistry, ScenarioResult, type Scorecard, type ScorecardCell, type ScorecardCellDiff, type ScorecardDiff, type ScorecardEntry, type ScorecardLogLine, type ScoredTarget, type SeatName, type SeatPresetName, SeatUnsetError, type SelfPlayOptions, type SelfPlayProposer, type SelfPlayScorer, type SerializedRegex, Severity, type SingleBackendDivergence, SingleBackendError, type SingleBackendField, type SingleBackendReport, type Slo, type SloCheckResult, type SloComparator, type SloReport, type SloSeverity, type SlopCategory, type SlotFactory, type SpanPredicate, type SplitGoldOptions, SteeringBundle, type SteeringOptimizationResult, type SteeringOptimizationRow, type SteeringOptimizationSelector, type SteeringOptimizerBackend, type SteeringOptimizerConfig, type StepAttribution, type SynthesisReason, type SynthesisTarget, TestResult, type TextMatcher, type ThresholdContract, TokenCounter, type TokenSpec, type TraceContract, TraceContractBuilder, TraceEmitter, TraceStore, type TracedAnalystOptions, type TracedJudgeOptions, Trajectory, TrajectoryStep, type TrialTrace, TurnMetrics, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, type UiFinding, type UiFindingScreenshot, type UiFindingSeverity, type UiLens, type VerdictCacheStats, type VerdictCacheStore, VerifyContext, type VisualDiffOptions, type VisualDiffResult, type ViteDeployRunnerInput, type WorkspaceAssertion, type WorkspaceAssertionResult, type WorkspaceInspector, type WorkspaceSnapshot, type WranglerDeployRunnerInput, adversarialJudge, aggregateJudgeVerdicts, aggregatePrReviewScore, analyzeAntiSlop, appendScorecard, assertCrossFamily, assertModelsServed, assertSingleBackend, assignHeldOutTag, attachCostToReport, attest, bisect, buildAgreementJudge, buildDriverSystemPrompt, buildReflectionPrompt, buildReviewerPrompt, cachedJudge, canaryLeakView, canonicalJson, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, codeExecutionJudge, coherenceJudge, collectionPreserved, commentsForSource, commitBisect, compareReferenceReplay, compilerJudge, computeExperimentStats, contentHash, contractJudge, costReport, createAntiSlopJudge, createCustomJudge, createDefaultReviewer, createDomainExpertJudge, createIntentMatchJudge, createSandboxPool, crossTraceDiff, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultJudges, defaultParseStudentLabel, defaultReferenceReplayMatcher, defaultRenderStudentPrompt, deployGateLayer, diffScorecard, discoverPersonas, distillPlaybook, ensembleJudge, estimateCost, estimateTokens, evaluateContract, evaluateOracles, evaluateTraceContract, executeScenario, expectAgent, exportRewardModel, extractAssetUrls, extractErrorCount, fieldAgreement, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findSkipCountsAsPass, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, ghCliClient, gitProvenanceReader, precision as goldenPrecision, hashContent, hashToUnit, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryRunRecordBackend, inMemoryVerdictCache, isModelPriced, isOtelConfigured, jsonShape, jsonlReferenceReplayStore, jsonlRunRecordBackend, judgeFamily, keyPreserved, linterJudge, loadGoldScenarios, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, matchGoldens, matchSpan, mergeLayerResults, modelDescriptionBits, multiToolchainLayer, notBlocked, paraphraseRobustness, paraphraseRobustnessScenarios, parseGoldJsonl, parseReflectionResponse, partitionHeldOut, passOrthogonality, pixelDeltaRatio, politenessPrefixMutator, preflightModels, printDriverSummary, index as profile, promptBisect, proposeAutomatedPullRequest, proposeSynthesisTargets, recordRuns, recordRunsToScorecard, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatches, renderMarkdownReport, renderPlaybookMarkdown, replayScorerOverCorpus, replayTraceThroughJudge, resetLockedAppendersForTesting, resolveModelPricing, resolveSeat, rowCount, rowWhere, runAssertions, runBehavioralCanaries, runCanaries, runDistillation, runE2EWorkflow, runExpectations, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runReferenceReplay, runScore, runSelfPlay, scanForMuffledGates, scoreContinuity, scorePrReviewComments, scorePrReviewSource, scoreReferenceReplay, seatPresets, securityJudge, sentenceReorderMutator, splitGold, statusAdvanced, summarizePrReviewBenchmark, testJudge, textInSnapshot, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, traceContract, traceJudge, traceJudgeEnsemble, tracedAnalyzeTraces, typoMutator, urlContains, verifyAttestation, visualDiff, viteDeployRunner, weightedRecall, whitespaceCollapseMutator, withJudgeRetry, withOtelPipeline, wranglerDeployRunner };