@tangle-network/agent-eval 0.133.2 → 0.134.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/CHANGELOG.md +218 -0
  2. package/dist/analyst/index.d.ts +11 -35
  3. package/dist/analyst/index.d.ts.map +1 -1
  4. package/dist/analyst/index.js +4 -53
  5. package/dist/analyst/index.js.map +1 -1
  6. package/dist/{analyze-runs-BmX-h_yn.d.ts → analyze-runs-DMo3Lb_y.d.ts} +4 -4
  7. package/dist/{analyze-runs-BmX-h_yn.d.ts.map → analyze-runs-DMo3Lb_y.d.ts.map} +1 -1
  8. package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
  9. package/dist/analyze-runs-qk8op0tN.js.map +1 -0
  10. package/dist/baseline-BaPxoROc.js +149 -0
  11. package/dist/baseline-BaPxoROc.js.map +1 -0
  12. package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
  13. package/dist/baseline-D_fT6277.d.ts.map +1 -0
  14. package/dist/benchmarks/index.d.ts +1 -1
  15. package/dist/benchmarks/index.js +1 -1
  16. package/dist/{benchmarks-CJr1H1_a.js → benchmarks-v5piCeDl.js} +3 -3
  17. package/dist/{benchmarks-CJr1H1_a.js.map → benchmarks-v5piCeDl.js.map} +1 -1
  18. package/dist/builder-eval/index.js +1 -1
  19. package/dist/campaign/index.d.ts +5 -4
  20. package/dist/campaign/index.js +4 -3
  21. package/dist/{campaign-BJjn1rhw.js → campaign-DEC_7DLn.js} +12 -6
  22. package/dist/{campaign-BJjn1rhw.js.map → campaign-DEC_7DLn.js.map} +1 -1
  23. package/dist/{client-COvaLoQG.d.ts → client-BIyh1RCr.d.ts} +29 -15
  24. package/dist/client-BIyh1RCr.d.ts.map +1 -0
  25. package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
  26. package/dist/client-LIuo-KPv.js.map +1 -0
  27. package/dist/contract/index.d.ts +12 -11
  28. package/dist/contract/index.d.ts.map +1 -1
  29. package/dist/contract/index.js +12 -18
  30. package/dist/contract/index.js.map +1 -1
  31. package/dist/{default-registry-Cl3pHo4n.d.ts → default-registry-Brxr728w.d.ts} +4 -268
  32. package/dist/default-registry-Brxr728w.d.ts.map +1 -0
  33. package/dist/{default-registry-D3T9XbuY.js → default-registry-IjYs7T8l.js} +4 -61
  34. package/dist/default-registry-IjYs7T8l.js.map +1 -0
  35. package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
  36. package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
  37. package/dist/hosted/index.d.ts +2 -2
  38. package/dist/hosted/index.d.ts.map +1 -1
  39. package/dist/hosted/index.js +1 -1
  40. package/dist/{index-C7Wue8R6.d.ts → index-BoJNQR6n.d.ts} +29 -11
  41. package/dist/index-BoJNQR6n.d.ts.map +1 -0
  42. package/dist/{index-BREtv3ZZ.d.ts → index-C21xKtxu.d.ts} +4 -4
  43. package/dist/{index-BREtv3ZZ.d.ts.map → index-C21xKtxu.d.ts.map} +1 -1
  44. package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
  45. package/dist/index-DSC51roc2.d.ts.map +1 -0
  46. package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
  47. package/dist/index-DuhJaaiH.d.ts.map +1 -0
  48. package/dist/index.d.ts +60 -13
  49. package/dist/index.d.ts.map +1 -1
  50. package/dist/index.js +134 -27
  51. package/dist/index.js.map +1 -1
  52. package/dist/ledger-core/index.d.ts +2 -2
  53. package/dist/ledger-core/index.js +2 -2
  54. package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
  55. package/dist/ledger-core-DAKFKRzi.js.map +1 -0
  56. package/dist/matrix/index.d.ts +1 -1
  57. package/dist/meta-eval/index.d.ts +1 -1
  58. package/dist/meta-eval/index.js +2 -2
  59. package/dist/multishot/index.d.ts +2 -2
  60. package/dist/openapi.json +1 -1
  61. package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
  62. package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
  63. package/dist/pipelines/index.d.ts +1 -1
  64. package/dist/pipelines/index.js +3 -2
  65. package/dist/pipelines/index.js.map +1 -1
  66. package/dist/proposal-findings-DCawte-y.js +164 -0
  67. package/dist/proposal-findings-DCawte-y.js.map +1 -0
  68. package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
  69. package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
  70. package/dist/{release-report-CjHWa8Ia.d.ts → release-report-CuULWKyk.d.ts} +2 -2
  71. package/dist/{release-report-CjHWa8Ia.d.ts.map → release-report-CuULWKyk.d.ts.map} +1 -1
  72. package/dist/reporting.d.ts +3 -3
  73. package/dist/reporting.js +4 -4
  74. package/dist/{researcher-CbSKhK8z.d.ts → researcher-DVtruQ9U.d.ts} +2 -2
  75. package/dist/{researcher-CbSKhK8z.d.ts.map → researcher-DVtruQ9U.d.ts.map} +1 -1
  76. package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
  77. package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
  78. package/dist/rl.d.ts +2 -2
  79. package/dist/rl.d.ts.map +1 -1
  80. package/dist/rl.js +18 -5
  81. package/dist/rl.js.map +1 -1
  82. package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
  83. package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
  84. package/dist/{semantic-concept-judge-BypLt6Fw.js → semantic-concept-judge-C0P1VTXD.js} +2 -3
  85. package/dist/{semantic-concept-judge-BypLt6Fw.js.map → semantic-concept-judge-C0P1VTXD.js.map} +1 -1
  86. package/dist/{skill-usage-BaaxFSJR.d.ts → skill-usage-BDQVPIG1.d.ts} +3 -2
  87. package/dist/skill-usage-BDQVPIG1.d.ts.map +1 -0
  88. package/dist/{skillopt-optimization-method-CF6a327Q.js → skillopt-optimization-method-BY6vKLJB.js} +169 -53
  89. package/dist/skillopt-optimization-method-BY6vKLJB.js.map +1 -0
  90. package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts → skillopt-optimization-method-DJ3l4w8W.d.ts} +20 -18
  91. package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +1 -0
  92. package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
  93. package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
  94. package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
  95. package/dist/statistics-RwRNu2__.js.map +1 -0
  96. package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
  97. package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
  98. package/dist/{summary-report-CFnQgNfg.d.ts → summary-report-DGp0-_XO.d.ts} +61 -4
  99. package/dist/summary-report-DGp0-_XO.d.ts.map +1 -0
  100. package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
  101. package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
  102. package/dist/types-DVjczBM9.d.ts +276 -0
  103. package/dist/types-DVjczBM9.d.ts.map +1 -0
  104. package/dist/{types-BokuXvOG.d.ts → types-DiWLru6Z.d.ts} +20 -37
  105. package/dist/types-DiWLru6Z.d.ts.map +1 -0
  106. package/docs/campaign-proposers.md +5 -0
  107. package/docs/design/statistics-decisions.md +271 -0
  108. package/docs/design.md +1 -0
  109. package/docs/insight-report.md +1 -1
  110. package/docs/research-report-methodology.md +4 -1
  111. package/package.json +2 -1
  112. package/dist/analyze-runs-B-afTpCv.js.map +0 -1
  113. package/dist/baseline-DcX5hQDv.js.map +0 -1
  114. package/dist/baseline-hG3K85h4.d.ts.map +0 -1
  115. package/dist/client-COvaLoQG.d.ts.map +0 -1
  116. package/dist/client-CYzbdJOZ.js.map +0 -1
  117. package/dist/default-registry-Cl3pHo4n.d.ts.map +0 -1
  118. package/dist/default-registry-D3T9XbuY.js.map +0 -1
  119. package/dist/index-C7Wue8R6.d.ts.map +0 -1
  120. package/dist/index-DSC51roc.d.ts.map +0 -1
  121. package/dist/index-nhIYz9hn.d.ts.map +0 -1
  122. package/dist/ledger-core-CPZfcrC2.js.map +0 -1
  123. package/dist/run-score-iEEAWiBY.js +0 -41
  124. package/dist/run-score-iEEAWiBY.js.map +0 -1
  125. package/dist/skill-usage-BaaxFSJR.d.ts.map +0 -1
  126. package/dist/skillopt-optimization-method-CF6a327Q.js.map +0 -1
  127. package/dist/skillopt-optimization-method-wHF5xsUv.d.ts.map +0 -1
  128. package/dist/statistics-DWM_AyLe.js.map +0 -1
  129. package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
  130. package/dist/summary-report-CFnQgNfg.d.ts.map +0 -1
  131. package/dist/types-BokuXvOG.d.ts.map +0 -1
package/dist/index.js CHANGED
@@ -2,23 +2,23 @@ import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
2
2
  import { _ as observeAll, a as jsonlReviewStore, c as scoreFromEvals, d as runAgentControlLoop, f as stopOnNoProgress, g as noProgressDetector, h as errorStreakDetector, i as inMemoryReviewStore, l as allCriticalPassed, m as subjectiveEval, n as runProposeReviewAsControlLoop, o as runProposeReview, p as stopOnRepeatedAction, r as createLlmReviewer, s as controlRunToRunRecord, t as controlFailureClassFromVerification, u as objectiveEval, v as repeatedActionDetector, y as evaluateActionPolicy } from "./propose-review-control-SQ-n9-We.js";
3
3
  import { a as NotFoundError, c as VerificationError, i as JudgeError, n as CaptureIntegrityError, o as ReplayError, r as ConfigError, s as ValidationError, t as AgentEvalError } from "./errors-8YnH8WlF.js";
4
4
  import { _ as verifyManifest, a as assertRunAgentProfileCell, c as groupRunsByAgentProfileCell, d as validateAgentProfileCell, f as verifyAgentProfileCell, g as signManifest, h as hashJson, i as agentProfileCellKey, l as requireAgentProfileCell, m as evaluateHypothesis, n as AgentProfileCellValidationError, o as buildAgentInterfaceProfileCell, p as canonicalize, r as agentProfileCellHashMaterial, s as buildAgentProfileCell, t as AGENT_PROFILE_KINDS, u as toAgentProfileJson } from "./agent-profile-cell-OhuTee9n.js";
5
- import { F as makeFinding, I as computeTraceMetrics, L as createChatClient, P as computeFindingId, R as createAnalystAi, a as KNOWLEDGE_GAP_KIND_SPEC, d as renderUpstreamFindings, i as KNOWLEDGE_POISONING_KIND_SPEC, l as createTraceAnalystKind, n as AnalystRegistry, o as IMPROVEMENT_KIND_SPEC, r as DEFAULT_TRACE_ANALYST_KINDS, s as FAILURE_MODE_KIND_SPEC, t as buildDefaultAnalystRegistry, u as renderPriorFindings } from "./default-registry-D3T9XbuY.js";
5
+ import { F as createChatClient, I as createAnalystAi, P as computeTraceMetrics, a as KNOWLEDGE_GAP_KIND_SPEC, d as renderUpstreamFindings, i as KNOWLEDGE_POISONING_KIND_SPEC, l as createTraceAnalystKind, n as AnalystRegistry, o as IMPROVEMENT_KIND_SPEC, r as DEFAULT_TRACE_ANALYST_KINDS, s as FAILURE_MODE_KIND_SPEC, t as buildDefaultAnalystRegistry, u as renderPriorFindings } from "./default-registry-IjYs7T8l.js";
6
6
  import { _ as resolveModelPricing, a as CostLedgerPersistenceError, c as costForTokenPricing, d as MODEL_PRICING, f as MetricsCollector, g as isModelPriced, h as estimateTokens, i as CostLedger, l as costForUsage, m as estimateCost, n as CostCallConflictError, o as CostReceiptCaptureError, p as TokenCounter, r as CostCeilingReachedError, s as CostReservationExceededError, t as CostAccountingIncompleteError, u as modelPriceKey } from "./cost-ledger-BrJxbrMy.js";
7
7
  import { a as providerFromBaseUrl, i as defaultProviderRedactor, n as InMemoryRawProviderSink, r as NoopRawProviderSink, t as FileSystemRawProviderSink } from "./raw-provider-sink-BQd7mzyT.js";
8
8
  import { a as assertLlmRoute, c as callLlmJson, d as isTransientLlmError, f as maximumChargeForLlmRequest, i as LlmRouteAssertionError, l as costReceiptFromLlm, m as stripFencedJson, n as LlmClient, o as backoffMs, p as probeLlm, r as LlmResponseError, s as callLlm, t as LlmCallError, u as costReceiptFromLlmError } from "./llm-client-ClPW-dWB.js";
9
9
  import { INPUT_VALUE, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, OPENINFERENCE_SPAN_KIND, OUTPUT_VALUE, RUN_COST_ATTR_KEYS, SPAN_KIND_ATTR_KEYS, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, applyLlmSpanOtlpAttributes, asNumber, contextInputTokens, firstNumberAttr } from "./trace-attributes.js";
10
10
  import { S as traceSpanKindToOpenInferenceKind, a as SpanNotFoundError, b as classifyOtlpSpanRole, c as DEFAULT_TRACE_ANALYST_BUDGETS, f as extractOtlpAttributes, g as readOtlpStatus, h as projectOtlpFlatLine, i as OtlpFileTraceStore, l as TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, m as inferOtlpKind, n as traceAnalystFunctionGroup, o as TraceFileMissingError, p as firstStringAttr, s as TraceNotFoundError, t as buildTraceAnalystTools, u as asString, v as stringField, x as isOtlpModelCall, y as applyToolSpanOtlpAttributes } from "./tools-BmuN627J.js";
11
- import { n as aggregateRunScore, r as clamp01, t as DEFAULT_RUN_SCORE_WEIGHTS } from "./run-score-iEEAWiBY.js";
11
+ import { a as clamp01, c as makeFinding, i as aggregateRunScore, l as makeProposalFinding, r as DEFAULT_RUN_SCORE_WEIGHTS, s as computeFindingId } from "./proposal-findings-DCawte-y.js";
12
12
  import { n as mapConcurrent, t as Mutex } from "./concurrency-MUjT7VjM.js";
13
- import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-BypLt6Fw.js";
14
- import { At as dominates, B as surfaceContentHash, Bt as JudgeParseError, Ct as llmJudge, D as runCanaries, Dt as fileVerdictCache, Et as contentHash, Ft as BackendIntegrityError, G as DEFAULT_MUTATION_PRIMITIVES, It as assertRealAgentReceipts, K as buildReflectionPrompt, Lt as assertRealBackend, Mt as paretoFrontierWithCrowding, Nt as scalarScore, Ot as inMemoryVerdictCache, Rt as summarizeAgentReceiptIntegrity, St as hashScenarios, Tt as canonicalJson, _t as redTeamReport, bt as Dataset, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, gt as redTeamDataset, ht as DEFAULT_RED_TEAM_CORPUS, jt as paretoFrontier, kt as crowdingDistance, mt as runReferenceEquivalenceJudge, pt as createReferenceEquivalenceJudge, q as parseReflectionResponse, vt as scoreRedTeamOutput, wt as cachedJudge, xt as HoldoutLockedError, yt as toolNamesForRun, zt as summarizeBackendIntegrity } from "./skillopt-optimization-method-CF6a327Q.js";
15
- import { A as spearmanR, B as positionalBias, C as pairedTTest, D as ranks, E as pearsonR, H as verbosityBias, L as calibrateJudge, M as weightedMean, N as wilcoxonSignedRank, O as requiredPairedSampleSize, P as wilson, R as calibrateJudgeContinuous, S as pairedSignTest, T as passAtK, V as selfPreference, _ as normalizeScores, a as confidenceInterval, b as pairedMde, c as eProcess, d as interpretCliffs, f as mannWhitneyU, g as mulberry32, h as mcnemarRequiredN, i as cohensD, j as weightedComposite, k as requiredSampleSize, l as holm, m as mcnemarPower, n as bonferroni, o as corpusInterRaterAgreement, p as mcnemar, r as cliffsDelta, s as corpusInterRaterAgreementFromJudgeScores, t as benjaminiHochberg, u as interRaterReliability, v as pairedBootstrap, w as partialCredit, x as pairedRiskDifference, y as pairedCohensDz, z as continuousAgreement } from "./statistics-DWM_AyLe.js";
16
- import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-6XItKzd1.js";
13
+ import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-C0P1VTXD.js";
14
+ import { At as dominates, B as surfaceContentHash, Bt as summarizeAgentReceiptIntegrity, Ct as llmJudge, D as runCanaries, Dt as fileVerdictCache, Et as contentHash, Ft as minimumPairsForPairedDeltaTest, G as DEFAULT_MUTATION_PRIMITIVES, Ht as JudgeParseError, It as pairedDeltaTest, K as buildReflectionPrompt, Lt as BackendIntegrityError, Mt as paretoFrontierWithCrowding, Nt as scalarScore, Ot as inMemoryVerdictCache, Rt as assertRealAgentReceipts, St as hashScenarios, Tt as canonicalJson, Vt as summarizeBackendIntegrity, _t as redTeamReport, bt as Dataset, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, gt as redTeamDataset, ht as DEFAULT_RED_TEAM_CORPUS, jt as paretoFrontier, kt as crowdingDistance, mt as runReferenceEquivalenceJudge, pt as createReferenceEquivalenceJudge, q as parseReflectionResponse, vt as scoreRedTeamOutput, wt as cachedJudge, xt as HoldoutLockedError, yt as toolNamesForRun, zt as assertRealBackend } from "./skillopt-optimization-method-BY6vKLJB.js";
15
+ import { A as passAtK, B as studentTCdf, C as pairedBootstrap, D as pairedSignTest, E as pairedRiskDifference, F as spearmanR, G as continuousAgreement, H as normalCdf, I as weightedComposite, J as verbosityBias, K as positionalBias, L as weightedMean, M as ranks, N as requiredPairedSampleSize, O as pairedTTest, P as requiredSampleSize, R as wilcoxonSignedRank, S as normalizeScores, T as pairedMde, U as calibrateJudge, W as calibrateJudgeContinuous, _ as mannWhitneyU, a as WILCOXON_EXACT_MAX_N, b as mcnemarRequiredN, c as cliffsDelta, d as corpusInterRaterAgreement, f as corpusInterRaterAgreementFromJudgeScores, g as interpretCliffs, h as interRaterReliability, i as MANN_WHITNEY_EXACT_MAX_WORK, j as pearsonR, k as partialCredit, l as cohensD, m as holm, n as DEFAULT_PERMUTATIONS, o as benjaminiHochberg, p as eProcess, q as selfPreference, r as MANN_WHITNEY_EXACT_MAX_STATES, s as bonferroni, t as BOOTSTRAP_GATE_MIN_N, u as confidenceInterval, v as mcnemar, w as pairedCohensDz, x as mulberry32, y as mcnemarPower, z as wilson } from "./statistics-RwRNu2__.js";
16
+ import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-CA_8pN01.js";
17
17
  import { n as llmSpanFromProvider, t as TraceEmitter } from "./emitter-CPBAhxum.js";
18
18
  import { a as scoreOrigin, n as observedScore, o as trainingReward, r as observedSplitScore, s as trainingScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
19
19
  import { a as isRetrievalSpan, i as isLlmSpan, n as TRACE_SCHEMA_VERSION, o as isSandboxSpan, r as isJudgeSpan, s as isToolSpan, t as FAILURE_CLASSES } from "./schema-CRhEY1SO.js";
20
20
  import { a as roundTripRunRecord, i as parseRunRecordSafe, n as isRunRecord, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord, t as RunRecordValidationError } from "./run-record-BIwU2wdV.js";
21
- import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-wuilQkvK.js";
21
+ import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-BVZBmRZp.js";
22
22
  import { c as assertMintedLines, f as isRolloutLine, i as ROLLOUT_SCHEMA, m as validateRolloutLine, p as isTrainableSplit, s as assertMinted, u as assertRolloutLine } from "./schema-C6DW4ZHR.js";
23
23
  import { i as toRewardRows, r as toJsonl, s as toSftRows } from "./exporters-q9iL-2Jf.js";
24
24
  import { a as toHarborTrajectories, i as relabelImportedSplit, n as HARBOR_IMPORT_GAP, o as toHarborTrajectory, r as fromHarborTrajectory, t as ATIF_SCHEMA_VERSION } from "./rollout-CreDz__7.js";
@@ -28,17 +28,18 @@ import { D as showMeasured, E as isUnavailable, S as rollupSupervisorRuns, T as
28
28
  import { n as TRACE_ANALYST_ACTOR_DESCRIPTION, r as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, t as analyzeTraces } from "./analyst-LsnNpSkm.js";
29
29
  import { C as planTraceInsightQuestions, E as traceAnalystOnRunComplete, S as inferDomainKeywords, T as tokenizeDomainWords, _ as buildTraceInsightContext, a as convertTraceStoresToOtlp, b as describeTraceInsightScope, c as otelRunCompleteHook, d as captureFetchToRawSink, f as otlpRowsToRunRecords, g as flattenOtlpExportToNdjson, h as otlpToTraceRunRecords, i as iterateRawCalls, l as OTEL_AGENT_EVAL_SCOPE, m as otlpToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as otlpRowsToTraceRunRecords, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as buildTraceInsightPrompt, w as scoreTraceInsightReadiness, x as domainEvidencePattern, y as defaultTraceInsightPanel } from "./replay-CJfGLdx4.js";
30
30
  import { n as extractUsageFromResponse, r as extractUsageFromSse, t as extractUsage } from "./extract-usage-2j25whHw.js";
31
- import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign-BJjn1rhw.js";
31
+ import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign-DEC_7DLn.js";
32
+ import { n as iqr, r as welchsTTest, t as compareToBaseline } from "./baseline-BaPxoROc.js";
32
33
  import { a as judgeSpans, c as runsForScenario, i as hasCapturedToolArgs, l as toolSpans, n as argHash, o as llmSpans, r as groupBy, s as runFailureClass, t as aggregateLlm } from "./query-Di7eEQ79.js";
33
- import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-B-afTpCv.js";
34
- import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-Ci17nIdU.js";
34
+ import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-qk8op0tN.js";
35
+ import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-BxtossFi.js";
35
36
  import { a as composeParsers, c as vitestTestParser, i as SubprocessSandboxDriver, n as DockerSandboxDriver, o as jestTestParser, r as SandboxHarness, s as pytestTestParser, t as runTestGradedScenario } from "./test-graded-scenario-BsqWLmPt.js";
36
37
  import { a as InMemoryTraceStore, i as FileSystemTraceStore, n as assertRunCaptured, r as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-BzRbCHzi.js";
37
38
  import { i as redactValue, n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
38
- import { a as DEFAULT_RULES, i as computeToolUseMetrics, n as iqr, o as classifyFailure, r as welchsTTest, t as compareToBaseline } from "./baseline-DcX5hQDv.js";
39
+ import { n as DEFAULT_RULES, r as classifyFailure, t as computeToolUseMetrics } from "./tool-use-metrics-DEGMKycK.js";
39
40
  import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
40
- import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-CJr1H1_a.js";
41
- import { t as runEvalCampaign } from "./eval-campaign-DXhpZghy.js";
41
+ import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-v5piCeDl.js";
42
+ import { t as runEvalCampaign } from "./eval-campaign-CvPcvqXC.js";
42
43
  import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
43
44
  import { AxGEPA, AxSignature, ax } from "@ax-llm/ax";
44
45
  import { accessSync, appendFileSync, constants, cpSync, existsSync, mkdirSync, promises, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
@@ -6004,7 +6005,7 @@ function diffScorecard(scorecard, opts = {}) {
6004
6005
  d = cohensD(baseline.scores, current.scores);
6005
6006
  const t = welchsTTest(baseline.scores, current.scores);
6006
6007
  p = Number.isFinite(t.p) ? t.p : null;
6007
- verdict = Math.abs(d) >= minEffect && p !== null && p <= maxP ? delta > 0 ? "improved" : "regressed" : "flat";
6008
+ verdict = (d === null || Math.abs(d) >= minEffect) && p !== null && p <= maxP ? delta > 0 ? "improved" : "regressed" : "flat";
6008
6009
  } else verdict = Math.abs(delta) >= minDelta ? delta > 0 ? "improved" : "regressed" : "flat";
6009
6010
  cells.push({
6010
6011
  ...base,
@@ -6816,7 +6817,7 @@ async function evaluateContract(store, contract) {
6816
6817
  if (!decl) continue;
6817
6818
  if (metric.verdict === "regressed") {
6818
6819
  const magnitude = Math.abs(metric.delta);
6819
- if (decl.maxRegression === void 0 || magnitude > decl.maxRegression) breaches.push(`metric "${metric.metric}" regressed by ${metric.delta.toFixed(4)} (d=${metric.cohensD.toFixed(2)}, p=${metric.welchP.toExponential(2)})`);
6820
+ if (decl.maxRegression === void 0 || magnitude > decl.maxRegression) breaches.push(`metric "${metric.metric}" regressed by ${metric.delta.toFixed(4)} (d=${formatEffectSize(metric.cohensD)}, p=${metric.welchP.toExponential(2)})`);
6820
6821
  }
6821
6822
  }
6822
6823
  if (sloReport) for (const r of sloReport.criticalBreaches) breaches.push(`SLO "${r.slo.id}" breached: ${r.detail}`);
@@ -6844,7 +6845,7 @@ function renderMarkdownReport(reports) {
6844
6845
  lines.push("");
6845
6846
  lines.push("| metric | baseline | candidate | Δ | Cohen d | p | verdict |");
6846
6847
  lines.push("|---|---|---|---|---|---|---|");
6847
- for (const m of r.baselineReport.metrics) lines.push(`| ${m.metric} | ${m.baselineMean.toFixed(4)} | ${m.candidateMean.toFixed(4)} | ${m.delta.toFixed(4)} | ${m.cohensD.toFixed(2)} | ${m.welchP.toExponential(2)} | ${m.verdict} |`);
6848
+ for (const m of r.baselineReport.metrics) lines.push(`| ${m.metric} | ${m.baselineMean.toFixed(4)} | ${m.candidateMean.toFixed(4)} | ${m.delta.toFixed(4)} | ${formatEffectSize(m.cohensD)} | ${m.welchP.toExponential(2)} | ${m.verdict} |`);
6848
6849
  }
6849
6850
  if (r.sloReport && r.sloReport.results.length > 0) {
6850
6851
  lines.push("");
@@ -6882,6 +6883,11 @@ function average(xs) {
6882
6883
  if (xs.length === 0) return 0;
6883
6884
  return xs.reduce((a, b) => a + b, 0) / xs.length;
6884
6885
  }
6886
+ /** Cohen's d is null when both samples are constant with unequal means — an
6887
+ * unbounded effect, which must not be rendered as a number. */
6888
+ function formatEffectSize(d) {
6889
+ return d === null ? "unbounded (zero pooled spread)" : d.toFixed(2);
6890
+ }
6885
6891
  async function extractAll(runs, extract, store) {
6886
6892
  const out = [];
6887
6893
  for (const r of runs) {
@@ -10083,8 +10089,11 @@ function precision(goldens, candidates, options = {}) {
10083
10089
  * The optimizer's best-guess is one thing; what we should actually
10084
10090
  * ship is another. The gate is the line between them.
10085
10091
  *
10086
- * A candidate is promoted iff ALL three pass:
10092
+ * A candidate is promoted iff ALL FOUR pass:
10087
10093
  *
10094
+ * 0. **Coverage**: on BOTH splits, the fraction of DEALT work items that
10095
+ * carry a real score on BOTH arms is at least `minCoverage`
10096
+ * (default 1 — all of them). See "Missing scores are missing data".
10088
10097
  * 1. **Productive runs**: the candidate has at least
10089
10098
  * `minProductiveRuns` paired observations on items where BOTH
10090
10099
  * candidate and baseline produced a real (non-silent) score.
@@ -10097,6 +10106,36 @@ function precision(goldens, candidates, options = {}) {
10097
10106
  * "Better on search, worse on holdout" is the canonical
10098
10107
  * overfit pattern; this catches it.
10099
10108
  *
10109
+ * ## Missing scores are missing data
10110
+ *
10111
+ * Gates 1-3 all read numbers off the items where BOTH arms produced a finite
10112
+ * score. Without gate 0 that set is a SILENT FILTER: an item the candidate
10113
+ * crashed on, timed out on, or never wrote a row for simply left the
10114
+ * comparison, and the verdict was computed over whatever survived. Measured on
10115
+ * the pre-0.134 gate: 26 held-out items, a candidate that produced no score at
10116
+ * all on 20 of them, promoted at every threshold from −0.05 to +0.30 on the 6
10117
+ * it answered (`unpairedBaselineRuns: 20` sat in the evidence, read by
10118
+ * nothing); the control — the same 20 failures scored as the 0 they earned —
10119
+ * was correctly refused at a mean paired delta of −0.3808. An agent that fails
10120
+ * 77% of its tasks was promoted, and every guarantee gates 1-3 provide sat
10121
+ * behind that filter.
10122
+ *
10123
+ * The gate does NOT impute a value for those items. It cannot: it does not know
10124
+ * the failure value of the caller's metric (0 on [0,1]? 0 on 0-100? a
10125
+ * lower-is-better latency?), and guessing one is exactly the silent assumption
10126
+ * this gate exists to remove. A caller who DOES know it writes it onto the
10127
+ * record before calling — that is the caller's decision to make and to be
10128
+ * accountable for, and it then travels in the run corpus where an auditor can
10129
+ * see it. What the gate does instead is REFUSE, and report the full coverage
10130
+ * accounting (`GateEvidence.holdoutCoverage` / `searchCoverage`) so the caller
10131
+ * can see exactly which items went dark and on which arm.
10132
+ *
10133
+ * The denominator is MEASURED, never assumed: it is the set of work items the
10134
+ * comparison was DEALT, which is what `pairRunRecords` reports when it is given
10135
+ * every row of a split rather than only the scored ones. There is no list of
10136
+ * "expected" scenarios to keep in sync and nothing to trust — an item counts as
10137
+ * dealt because a row for it exists on at least one arm.
10138
+ *
10100
10139
  * The decision carries a machine-readable `rejectionCode` plus an
10101
10140
  * `evidence` block with every number the gate looked at, so the
10102
10141
  * downstream researcher / paper / dashboard can re-derive the
@@ -10122,13 +10161,16 @@ var HeldOutGate = class {
10122
10161
  resamples;
10123
10162
  seed;
10124
10163
  costPerTaskCeiling;
10164
+ minCoverage;
10125
10165
  constructor(config) {
10126
10166
  if (!config.baselineKey) throw new Error("HeldOutGate: baselineKey is required");
10127
- this.minProductiveRuns = config.minProductiveRuns ?? 3;
10167
+ if (config.minCoverage !== void 0 && !(Number.isFinite(config.minCoverage) && config.minCoverage >= 0 && config.minCoverage <= 1)) throw new Error(`HeldOutGate: minCoverage must be a finite fraction in [0, 1], got ${config.minCoverage}`);
10168
+ this.minCoverage = config.minCoverage ?? 1;
10169
+ this.confidence = config.confidence ?? .95;
10170
+ this.minProductiveRuns = Math.max(config.minProductiveRuns ?? 3, minimumPairsForPairedDeltaTest(this.confidence));
10128
10171
  this.pairedDeltaThreshold = config.pairedDeltaThreshold ?? 0;
10129
10172
  this.overfitGapThreshold = config.overfitGapThreshold ?? .15;
10130
10173
  this.baselineKey = config.baselineKey;
10131
- this.confidence = config.confidence ?? .95;
10132
10174
  this.resamples = config.bootstrapResamples ?? 2e3;
10133
10175
  this.seed = config.seed;
10134
10176
  if (config.costPerTaskCeiling !== void 0 && !(Number.isFinite(config.costPerTaskCeiling) && config.costPerTaskCeiling > 0)) throw new Error("HeldOutGate: costPerTaskCeiling must be a positive finite number");
@@ -10150,6 +10192,8 @@ var HeldOutGate = class {
10150
10192
  const baselineHoldout = scoredRuns(honestBaseline, "holdout");
10151
10193
  const searchPairing = pairRunRecords(baselineSearch, candidateSearch);
10152
10194
  const holdoutPairing = pairRunRecords(baselineHoldout, candidateHoldout);
10195
+ const searchCoverage = splitCoverage(honestBaseline, honestCandidate, "search");
10196
+ const holdoutCoverage = splitCoverage(honestBaseline, honestCandidate, "holdout");
10153
10197
  const splitScoreOf = (run, split) => observedSplitScore(run, split);
10154
10198
  const beforeSearch = searchPairing.pairs.map((pair) => splitScoreOf(pair.baseline, "search"));
10155
10199
  const afterSearch = searchPairing.pairs.map((pair) => splitScoreOf(pair.treatment, "search"));
@@ -10162,8 +10206,8 @@ var HeldOutGate = class {
10162
10206
  const baselineHoldoutMean = meanOrNull(beforeHoldout);
10163
10207
  const overfitGap = diffOrNull(candidateSearchMean, candidateHoldoutMean);
10164
10208
  const baselineOverfitGap = diffOrNull(baselineSearchMean, baselineHoldoutMean);
10165
- const medianCandidateCost = completeCostMedian(candidate);
10166
- const medianBaselineCost = completeCostMedian(baseline);
10209
+ const medianCandidateCost = completeCostMedian([...searchPairing.pairs.map((pair) => pair.treatment), ...holdoutPairing.pairs.map((pair) => pair.treatment)]);
10210
+ const medianBaselineCost = completeCostMedian([...searchPairing.pairs.map((pair) => pair.baseline), ...holdoutPairing.pairs.map((pair) => pair.baseline)]);
10167
10211
  const commonEvidence = {
10168
10212
  productiveRuns,
10169
10213
  unpairedCandidateRuns: holdoutPairing.unpairedTreatment.length,
@@ -10174,7 +10218,9 @@ var HeldOutGate = class {
10174
10218
  baselineOverfitGap,
10175
10219
  medianCandidateCost,
10176
10220
  medianBaselineCost,
10177
- realnessGatedRuns
10221
+ realnessGatedRuns,
10222
+ searchCoverage,
10223
+ holdoutCoverage
10178
10224
  };
10179
10225
  const missingSplitScores = [
10180
10226
  candidateSearch.length === 0 ? "candidate search" : null,
@@ -10195,6 +10241,20 @@ var HeldOutGate = class {
10195
10241
  reason: `missing_split_scores: ${missingSplitScores.join(", ")} score evidence is absent`,
10196
10242
  rejectionCode: "missing_split_scores"
10197
10243
  };
10244
+ const shortfalls = [["holdout", holdoutCoverage], ["search", searchCoverage]].filter(([, cov]) => cov.coverage < this.minCoverage);
10245
+ if (shortfalls.length > 0) return {
10246
+ promote: false,
10247
+ candidateId,
10248
+ baselineId,
10249
+ evidence: {
10250
+ ...commonEvidence,
10251
+ medianPairedDelta: productiveRuns > 0 ? medianDelta(beforeHoldout, afterHoldout) : null,
10252
+ pairedCI: null,
10253
+ pairedPValue: null
10254
+ },
10255
+ reason: `incomplete_coverage: ${shortfalls.map(([split, cov]) => describeCoverage(split, cov)).join("; ")} — required ${fmt(this.minCoverage)} of every dealt item scored on both arms. A missing score is missing data, not an absent item: the gate will not decide on the subset that answered`,
10256
+ rejectionCode: "incomplete_coverage"
10257
+ };
10198
10258
  if (productiveRuns < this.minProductiveRuns) return {
10199
10259
  promote: false,
10200
10260
  candidateId,
@@ -10209,12 +10269,15 @@ var HeldOutGate = class {
10209
10269
  rejectionCode: "few_runs"
10210
10270
  };
10211
10271
  if (overfitGap === null || baselineOverfitGap === null) throw new Error("HeldOutGate: complete split scores did not produce overfit gaps");
10212
- const ci = pairedBootstrap(beforeHoldout, afterHoldout, {
10272
+ const deltaTest = pairedDeltaTest(beforeHoldout, afterHoldout, {
10213
10273
  confidence: this.confidence,
10214
10274
  resamples: this.resamples,
10215
10275
  statistic: "median",
10216
- seed: this.seed
10276
+ seed: this.seed,
10277
+ threshold: this.pairedDeltaThreshold,
10278
+ minPairs: this.minProductiveRuns
10217
10279
  });
10280
+ const ci = deltaTest.bootstrap;
10218
10281
  const wilcoxon = wilcoxonSignedRank(beforeHoldout, afterHoldout);
10219
10282
  const evidence = {
10220
10283
  ...commonEvidence,
@@ -10233,12 +10296,12 @@ var HeldOutGate = class {
10233
10296
  reason: `indeterminate_delta: paired holdout median CI=[${fmt(ci.low)}, ${fmt(ci.high)}] carries no direction`,
10234
10297
  rejectionCode: "indeterminate_delta"
10235
10298
  };
10236
- if (!(ci.low > this.pairedDeltaThreshold)) return {
10299
+ if (!deltaTest.significant) return {
10237
10300
  promote: false,
10238
10301
  candidateId,
10239
10302
  baselineId,
10240
10303
  evidence,
10241
- reason: `negative_delta: paired holdout median Δ=${fmt(ci.median)} CI=[${fmt(ci.low)}, ${fmt(ci.high)}] does not clear threshold ${fmt(this.pairedDeltaThreshold)}`,
10304
+ reason: `negative_delta: paired holdout median Δ=${fmt(ci.median)} ${deltaTest.method === "bootstrap-ci" ? `CI=[${fmt(ci.low)}, ${fmt(ci.high)}]` : `exact one-sided p=${fmt(deltaTest.pValue ?? 1)}`} does not clear threshold ${fmt(this.pairedDeltaThreshold)}`,
10242
10305
  rejectionCode: "negative_delta"
10243
10306
  };
10244
10307
  if (overfitGap > baselineOverfitGap + this.overfitGapThreshold) return {
@@ -10295,6 +10358,50 @@ function scoredRuns(runs, split) {
10295
10358
  return typeof v === "number" && Number.isFinite(v);
10296
10359
  });
10297
10360
  }
10361
+ /** Every row tagged to one split, scored or not — the work the split was
10362
+ * DEALT, as opposed to the work it managed to measure. */
10363
+ function dealtRuns(runs, split) {
10364
+ return runs.filter((run) => run.splitTag === split);
10365
+ }
10366
+ function hasFiniteScore(run, split) {
10367
+ const v = observedSplitScore(run, split);
10368
+ return typeof v === "number" && Number.isFinite(v);
10369
+ }
10370
+ /**
10371
+ * Partition one split's dealt work into answered / unscored / one-armed.
10372
+ *
10373
+ * Pairs the UNFILTERED rows with the same `pairRunRecords` the decision uses,
10374
+ * so the denominator comes out of the same identity rules as the numerator and
10375
+ * cannot drift from it. Deliberately a SECOND pairing rather than a widening of
10376
+ * the decision's own: the decision keeps pairing exactly the rows it always
10377
+ * paired, so this check can only ever add a refusal, never move a verdict that
10378
+ * was already being made on a complete set.
10379
+ */
10380
+ function splitCoverage(baseline, candidate, split) {
10381
+ const pairing = pairRunRecords(dealtRuns(baseline, split), dealtRuns(candidate, split));
10382
+ let answered = 0;
10383
+ for (const pair of pairing.pairs) if (hasFiniteScore(pair.baseline, split) && hasFiniteScore(pair.treatment, split)) answered += 1;
10384
+ const unscoredPairs = pairing.pairs.length - answered;
10385
+ const candidateOnly = pairing.unpairedTreatment.length;
10386
+ const baselineOnly = pairing.unpairedBaseline.length;
10387
+ const dealt = pairing.pairs.length + candidateOnly + baselineOnly;
10388
+ return {
10389
+ dealt,
10390
+ answered,
10391
+ unscoredPairs,
10392
+ candidateOnly,
10393
+ baselineOnly,
10394
+ coverage: dealt === 0 ? 0 : answered / dealt
10395
+ };
10396
+ }
10397
+ function describeCoverage(split, cov) {
10398
+ const dark = [
10399
+ cov.unscoredPairs > 0 ? `${cov.unscoredPairs} ran on both arms but produced no score` : null,
10400
+ cov.baselineOnly > 0 ? `${cov.baselineOnly} only the baseline produced a row for` : null,
10401
+ cov.candidateOnly > 0 ? `${cov.candidateOnly} only the candidate produced a row for` : null
10402
+ ].filter((part) => part !== null);
10403
+ return `${split} scored ${cov.answered} of ${cov.dealt} dealt items (coverage ${fmt(cov.coverage)})` + (dark.length > 0 ? ` — ${dark.join(", ")}` : "");
10404
+ }
10298
10405
  function meanOrNull(xs) {
10299
10406
  if (xs.length === 0) return null;
10300
10407
  return xs.reduce((s, x) => s + x, 0) / xs.length;
@@ -12114,6 +12221,6 @@ function assertProductBenchmarkRun(runDir) {
12114
12221
  return report;
12115
12222
  }
12116
12223
  //#endregion
12117
- export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, AxGepaSteeringOptimizer, BENCHMARK_SPLIT_SEED, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CallbackResearcher, CaptureIntegrityError, ConfigError, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MODEL_PRICING, MetricsCollector, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, TraceContractBuilder, TraceEmitter, TraceFileMissingError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createDefaultReviewer, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBootstrap, pairedCohensDz, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
12224
+ export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, AxGepaSteeringOptimizer, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CallbackResearcher, CaptureIntegrityError, ConfigError, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, TraceContractBuilder, TraceEmitter, TraceFileMissingError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createDefaultReviewer, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
12118
12225
 
12119
12226
  //# sourceMappingURL=index.js.map