@tangle-network/agent-eval 0.135.1 → 0.135.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. package/CHANGELOG.md +17 -0
  2. package/dist/{analyze-runs-DMo3Lb_y.d.ts → analyze-runs-Cda5Xkj1.d.ts} +3 -3
  3. package/dist/{analyze-runs-DMo3Lb_y.d.ts.map → analyze-runs-Cda5Xkj1.d.ts.map} +1 -1
  4. package/dist/{analyze-runs-qk8op0tN.js → analyze-runs-jjCmF8pU.js} +14 -7
  5. package/dist/analyze-runs-jjCmF8pU.js.map +1 -0
  6. package/dist/{baseline-BaPxoROc.js → baseline-BUeFcgrn.js} +2 -2
  7. package/dist/{baseline-BaPxoROc.js.map → baseline-BUeFcgrn.js.map} +1 -1
  8. package/dist/benchmarks/index.d.ts +1 -1
  9. package/dist/benchmarks/index.js +1 -1
  10. package/dist/{benchmarks-DwFrCrCS.js → benchmarks-Mtu251Jz.js} +3 -3
  11. package/dist/{benchmarks-DwFrCrCS.js.map → benchmarks-Mtu251Jz.js.map} +1 -1
  12. package/dist/builder-eval/index.js +1 -1
  13. package/dist/campaign/index.d.ts +2 -2
  14. package/dist/campaign/index.js +2 -2
  15. package/dist/{campaign-B0oyngTs.js → campaign-RVIqtJh0.js} +5 -5
  16. package/dist/{campaign-B0oyngTs.js.map → campaign-RVIqtJh0.js.map} +1 -1
  17. package/dist/{client-BIyh1RCr.d.ts → client-DcvgkaZi.d.ts} +13 -4
  18. package/dist/client-DcvgkaZi.d.ts.map +1 -0
  19. package/dist/contract/index.d.ts +3 -3
  20. package/dist/contract/index.d.ts.map +1 -1
  21. package/dist/contract/index.js +17 -5
  22. package/dist/contract/index.js.map +1 -1
  23. package/dist/{eval-campaign-aMG73hvX.js → eval-campaign-Cc8WZJ6b.js} +2 -2
  24. package/dist/{eval-campaign-aMG73hvX.js.map → eval-campaign-Cc8WZJ6b.js.map} +1 -1
  25. package/dist/hosted/index.d.ts +1 -1
  26. package/dist/{index-BoJNQR6n.d.ts → index-B4Fjfo5U.d.ts} +93 -15
  27. package/dist/index-B4Fjfo5U.d.ts.map +1 -0
  28. package/dist/{index-C21xKtxu.d.ts → index-CQsJcqch.d.ts} +3 -3
  29. package/dist/{index-C21xKtxu.d.ts.map → index-CQsJcqch.d.ts.map} +1 -1
  30. package/dist/{index-DSC51roc2.d.ts → index-DSC51roc.d.ts} +1 -1
  31. package/dist/index-DSC51roc.d.ts.map +1 -0
  32. package/dist/index.d.ts +41 -10
  33. package/dist/index.d.ts.map +1 -1
  34. package/dist/index.js +131 -53
  35. package/dist/index.js.map +1 -1
  36. package/dist/matrix/index.d.ts +1 -1
  37. package/dist/meta-eval/index.d.ts +1 -2
  38. package/dist/meta-eval/index.d.ts.map +1 -1
  39. package/dist/meta-eval/index.js +2 -2
  40. package/dist/multishot/index.d.ts +1 -1
  41. package/dist/openapi.json +1 -1
  42. package/dist/{paired-arms-CA_8pN01.js → paired-arms-BbFKrAU-.js} +2 -2
  43. package/dist/{paired-arms-CA_8pN01.js.map → paired-arms-BbFKrAU-.js.map} +1 -1
  44. package/dist/pipelines/index.js +2 -2
  45. package/dist/{release-report-BVZBmRZp.js → release-report-DooPguBc.js} +4 -3
  46. package/dist/{release-report-BVZBmRZp.js.map → release-report-DooPguBc.js.map} +1 -1
  47. package/dist/{release-report-CuULWKyk.d.ts → release-report-DpBxGGI1.d.ts} +2 -2
  48. package/dist/{release-report-CuULWKyk.d.ts.map → release-report-DpBxGGI1.d.ts.map} +1 -1
  49. package/dist/reporting.d.ts +3 -3
  50. package/dist/reporting.js +4 -4
  51. package/dist/{researcher-DVtruQ9U.d.ts → researcher-Doo95b50.d.ts} +2 -2
  52. package/dist/{researcher-DVtruQ9U.d.ts.map → researcher-Doo95b50.d.ts.map} +1 -1
  53. package/dist/{reward-hacking-DCdRK9TY.js → reward-hacking-a-kYs0-i.js} +2 -2
  54. package/dist/{reward-hacking-DCdRK9TY.js.map → reward-hacking-a-kYs0-i.js.map} +1 -1
  55. package/dist/rl.d.ts +45 -3
  56. package/dist/rl.d.ts.map +1 -1
  57. package/dist/rl.js +108 -22
  58. package/dist/rl.js.map +1 -1
  59. package/dist/{rubric-predictive-validity-D6Q6n9oq.js → rubric-predictive-validity-BJf-8ejY.js} +2 -2
  60. package/dist/{rubric-predictive-validity-D6Q6n9oq.js.map → rubric-predictive-validity-BJf-8ejY.js.map} +1 -1
  61. package/dist/{skillopt-optimization-method-C9yGg-4C.js → skillopt-optimization-method-0UmPD6aP.js} +340 -54
  62. package/dist/skillopt-optimization-method-0UmPD6aP.js.map +1 -0
  63. package/dist/{skillopt-optimization-method-DJ3l4w8W.d.ts → skillopt-optimization-method-CwSYkv35.d.ts} +39 -9
  64. package/dist/skillopt-optimization-method-CwSYkv35.d.ts.map +1 -0
  65. package/dist/{statistics-D_4Snl-5.d.ts → statistics-CKOqre5S.d.ts} +329 -3
  66. package/dist/statistics-CKOqre5S.d.ts.map +1 -0
  67. package/dist/{statistics-RwRNu2__.js → statistics-CnGCLLqc.js} +315 -2
  68. package/dist/statistics-CnGCLLqc.js.map +1 -0
  69. package/dist/{summary-report-BxtossFi.js → summary-report-BEk8OFLs.js} +11 -6
  70. package/dist/summary-report-BEk8OFLs.js.map +1 -0
  71. package/dist/{summary-report-DGp0-_XO.d.ts → summary-report-CPMINBqs.d.ts} +182 -7
  72. package/dist/summary-report-CPMINBqs.d.ts.map +1 -0
  73. package/package.json +1 -1
  74. package/dist/analyze-runs-qk8op0tN.js.map +0 -1
  75. package/dist/client-BIyh1RCr.d.ts.map +0 -1
  76. package/dist/index-BoJNQR6n.d.ts.map +0 -1
  77. package/dist/index-DSC51roc2.d.ts.map +0 -1
  78. package/dist/judge-calibration-DFtEMlde.d.ts +0 -146
  79. package/dist/judge-calibration-DFtEMlde.d.ts.map +0 -1
  80. package/dist/skillopt-optimization-method-C9yGg-4C.js.map +0 -1
  81. package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +0 -1
  82. package/dist/statistics-D_4Snl-5.d.ts.map +0 -1
  83. package/dist/statistics-RwRNu2__.js.map +0 -1
  84. package/dist/summary-report-BxtossFi.js.map +0 -1
  85. package/dist/summary-report-DGp0-_XO.d.ts.map +0 -1
package/dist/index.js CHANGED
@@ -12,14 +12,14 @@ import { S as traceSpanKindToOpenInferenceKind, a as SpanNotFoundError, b as cla
12
12
  import { a as clamp01, c as makeFinding, i as aggregateRunScore, l as makeProposalFinding, r as DEFAULT_RUN_SCORE_WEIGHTS, s as computeFindingId } from "./proposal-findings-DCawte-y.js";
13
13
  import { n as mapConcurrent, t as Mutex } from "./concurrency-MUjT7VjM.js";
14
14
  import { a as RunCritic, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, p as LockedJsonlAppender, r as createSemanticConceptJudge, s as SkillUsageAnalyst, t as DEFAULT_COMPLEXITY_WEIGHTS, u as FindingsStore } from "./semantic-concept-judge-Btozx3Vc.js";
15
- import { At as dominates, B as surfaceContentHash, Bt as summarizeAgentReceiptIntegrity, Ct as llmJudge, D as runCanaries, Dt as fileVerdictCache, Et as contentHash, Ft as minimumPairsForPairedDeltaTest, G as DEFAULT_MUTATION_PRIMITIVES, Ht as JudgeParseError, It as pairedDeltaTest, K as buildReflectionPrompt, Lt as BackendIntegrityError, Mt as paretoFrontierWithCrowding, Nt as scalarScore, Ot as inMemoryVerdictCache, Rt as assertRealAgentReceipts, St as hashScenarios, Tt as canonicalJson, Vt as summarizeBackendIntegrity, _t as redTeamReport, bt as Dataset, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, gt as redTeamDataset, ht as DEFAULT_RED_TEAM_CORPUS, jt as paretoFrontier, kt as crowdingDistance, mt as runReferenceEquivalenceJudge, pt as createReferenceEquivalenceJudge, q as parseReflectionResponse, vt as scoreRedTeamOutput, wt as cachedJudge, xt as HoldoutLockedError, yt as toolNamesForRun, zt as assertRealBackend } from "./skillopt-optimization-method-C9yGg-4C.js";
16
- import { A as passAtK, B as studentTCdf, C as pairedBootstrap, D as pairedSignTest, E as pairedRiskDifference, F as spearmanR, G as continuousAgreement, H as normalCdf, I as weightedComposite, J as verbosityBias, K as positionalBias, L as weightedMean, M as ranks, N as requiredPairedSampleSize, O as pairedTTest, P as requiredSampleSize, R as wilcoxonSignedRank, S as normalizeScores, T as pairedMde, U as calibrateJudge, W as calibrateJudgeContinuous, _ as mannWhitneyU, a as WILCOXON_EXACT_MAX_N, b as mcnemarRequiredN, c as cliffsDelta, d as corpusInterRaterAgreement, f as corpusInterRaterAgreementFromJudgeScores, g as interpretCliffs, h as interRaterReliability, i as MANN_WHITNEY_EXACT_MAX_WORK, j as pearsonR, k as partialCredit, l as cohensD, m as holm, n as DEFAULT_PERMUTATIONS, o as benjaminiHochberg, p as eProcess, q as selfPreference, r as MANN_WHITNEY_EXACT_MAX_STATES, s as bonferroni, t as BOOTSTRAP_GATE_MIN_N, u as confidenceInterval, v as mcnemar, w as pairedCohensDz, x as mulberry32, y as mcnemarPower, z as wilson } from "./statistics-RwRNu2__.js";
17
- import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-CA_8pN01.js";
15
+ import { At as dominates, B as surfaceContentHash, Bt as assertRealAgentReceipts, Ct as llmJudge, D as runCanaries, Dt as fileVerdictCache, Et as contentHash, Ft as decidePairedPromotion, G as DEFAULT_MUTATION_PRIMITIVES, Ht as summarizeAgentReceiptIntegrity, It as pairedDecisionShape, K as buildReflectionPrompt, Lt as minimumPairsForPairedDeltaTest, Mt as paretoFrontierWithCrowding, Nt as scalarScore, Ot as inMemoryVerdictCache, Rt as pairedDeltaTest, St as hashScenarios, Tt as canonicalJson, Ut as summarizeBackendIntegrity, Vt as assertRealBackend, Wt as JudgeParseError, _t as redTeamReport, bt as Dataset, dt as REFERENCE_EQUIVALENCE_INPUT_LIMITS, ft as REFERENCE_EQUIVALENCE_JUDGE_VERSION, gt as redTeamDataset, ht as DEFAULT_RED_TEAM_CORPUS, jt as paretoFrontier, kt as crowdingDistance, mt as runReferenceEquivalenceJudge, pt as createReferenceEquivalenceJudge, q as parseReflectionResponse, vt as scoreRedTeamOutput, wt as cachedJudge, xt as HoldoutLockedError, yt as toolNamesForRun, zt as BackendIntegrityError } from "./skillopt-optimization-method-0UmPD6aP.js";
16
+ import { $ as selfPreference, A as pairedRiskDifference, B as requiredSampleSize, C as mulberry32, D as pairedCohensDz, E as pairedBootstrap, F as partialCredit, G as wilson, H as weightedComposite, I as passAtK, J as normalCdf, K as studentTCdf, L as pearsonR, M as pairedRiskDifferenceScore, N as pairedSignTest, O as pairedDeltaTieFraction, P as pairedTTest, Q as positionalBias, R as ranks, S as mcnemarRequiredN, T as pairedBinaryScale, U as weightedMean, V as spearmanR, W as wilcoxonSignedRank, X as calibrateJudgeContinuous, Y as calibrateJudge, Z as continuousAgreement, _ as interpretCliffs, a as MANN_WHITNEY_EXACT_MAX_WORK, b as mcnemar, c as bonferroni, d as confidenceInterval, et as verbosityBias, f as corpusInterRaterAgreement, g as interRaterReliability, h as holm, i as MANN_WHITNEY_EXACT_MAX_STATES, j as pairedRiskDifferenceExact, k as pairedMde, l as cliffsDelta, m as eProcess, n as DECISION_PAIRED_DELTA_STATISTIC, o as WILCOXON_EXACT_MAX_N, p as corpusInterRaterAgreementFromJudgeScores, r as DEFAULT_PERMUTATIONS, s as benjaminiHochberg, t as BOOTSTRAP_GATE_MIN_N, u as cohensD, v as isBinaryOutcomeVector, w as normalizeScores, x as mcnemarPower, y as mannWhitneyU, z as requiredPairedSampleSize } from "./statistics-CnGCLLqc.js";
17
+ import { n as pairArms, r as pairRunRecords, t as comparePairedArms } from "./paired-arms-BbFKrAU-.js";
18
18
  import { n as llmSpanFromProvider, t as TraceEmitter } from "./emitter-CPBAhxum.js";
19
19
  import { a as scoreOrigin, n as observedScore, o as trainingReward, r as observedSplitScore, s as trainingScore, t as isRealnessGated } from "./reward-nw2xZGZG.js";
20
20
  import { a as isRetrievalSpan, i as isLlmSpan, n as TRACE_SCHEMA_VERSION, o as isSandboxSpan, r as isJudgeSpan, s as isToolSpan, t as FAILURE_CLASSES } from "./schema-CRhEY1SO.js";
21
21
  import { a as roundTripRunRecord, i as parseRunRecordSafe, n as isRunRecord, o as runTaskScore, r as modelHasSnapshot, s as validateRunRecord, t as RunRecordValidationError } from "./run-record-BIwU2wdV.js";
22
- import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-BVZBmRZp.js";
22
+ import { a as evaluateReleaseConfidence, i as assertReleaseConfidence, n as bootstrapCi, r as judgeReplayGate, t as renderReleaseReport } from "./release-report-DooPguBc.js";
23
23
  import { c as assertMintedLines, f as isRolloutLine, i as ROLLOUT_SCHEMA, m as validateRolloutLine, p as isTrainableSplit, s as assertMinted, u as assertRolloutLine } from "./schema-C6DW4ZHR.js";
24
24
  import { i as toRewardRows, r as toJsonl, s as toSftRows } from "./exporters-q9iL-2Jf.js";
25
25
  import { a as toHarborTrajectories, i as relabelImportedSplit, n as HARBOR_IMPORT_GAP, o as toHarborTrajectory, r as fromHarborTrajectory, t as ATIF_SCHEMA_VERSION } from "./rollout-DLSUIWLu.js";
@@ -29,18 +29,18 @@ import { D as showMeasured, E as isUnavailable, S as rollupSupervisorRuns, T as
29
29
  import { n as TRACE_ANALYST_ACTOR_DESCRIPTION, r as TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, t as analyzeTraces } from "./analyst-LsnNpSkm.js";
30
30
  import { C as planTraceInsightQuestions, E as traceAnalystOnRunComplete, S as inferDomainKeywords, T as tokenizeDomainWords, _ as buildTraceInsightContext, a as convertTraceStoresToOtlp, b as describeTraceInsightScope, c as otelRunCompleteHook, d as captureFetchToRawSink, f as otlpRowsToRunRecords, g as flattenOtlpExportToNdjson, h as otlpToTraceRunRecords, i as iterateRawCalls, l as OTEL_AGENT_EVAL_SCOPE, m as otlpToRunRecords, n as ReplayCacheMissError, o as createOtelExporter, p as otlpRowsToTraceRunRecords, r as createReplayFetch, s as createOtelTracingStore, t as ReplayCache, u as exportRunAsOtlp, v as buildTraceInsightPrompt, w as scoreTraceInsightReadiness, x as domainEvidencePattern, y as defaultTraceInsightPanel } from "./replay-CJfGLdx4.js";
31
31
  import { n as extractUsageFromResponse, r as extractUsageFromSse, t as extractUsage } from "./extract-usage-2j25whHw.js";
32
- import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign-B0oyngTs.js";
33
- import { n as iqr, r as welchsTTest, t as compareToBaseline } from "./baseline-BaPxoROc.js";
32
+ import { B as harnessAxisOf, F as HARNESS_NATIVE_MODEL, G as parseCorrectnessResponse, H as completionVerdict, I as agentProfileHash, K as verifyCompletion, L as agentProfileId, P as CODING_HARNESSES, R as agentProfileModelId, U as createLlmCorrectnessChecker, V as extractProducedState, W as createTokenRecallChecker, z as expandProfileAxes } from "./campaign-RVIqtJh0.js";
33
+ import { n as iqr, r as welchsTTest, t as compareToBaseline } from "./baseline-BUeFcgrn.js";
34
34
  import { a as judgeSpans, c as runsForScenario, i as hasCapturedToolArgs, l as toolSpans, n as argHash, o as llmSpans, r as groupBy, s as runFailureClass, t as aggregateLlm } from "./query-Di7eEQ79.js";
35
- import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-qk8op0tN.js";
36
- import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-BxtossFi.js";
35
+ import { a as checkBehavioralCanary, i as canaryLeakView, o as checkCanaries, r as HoldoutAuditor, s as runBehavioralCanaries, t as analyzeRuns } from "./analyze-runs-jjCmF8pU.js";
36
+ import { a as summaryTable, i as researchReport, n as gainHistogram, r as paretoChart, t as RESEARCH_REPORT_HARD_PAIR_FLOOR } from "./summary-report-BEk8OFLs.js";
37
37
  import { a as composeParsers, c as vitestTestParser, i as SubprocessSandboxDriver, n as DockerSandboxDriver, o as jestTestParser, r as SandboxHarness, s as pytestTestParser, t as runTestGradedScenario } from "./test-graded-scenario-BsqWLmPt.js";
38
38
  import { a as InMemoryTraceStore, i as FileSystemTraceStore, n as assertRunCaptured, r as throwIfRunIncomplete, t as RunIntegrityError } from "./integrity-BzRbCHzi.js";
39
39
  import { i as redactValue, n as REDACTION_VERSION, r as redactString, t as DEFAULT_REDACTION_RULES } from "./redact-7Aq1ukl-.js";
40
40
  import { n as DEFAULT_RULES, r as classifyFailure, t as computeToolUseMetrics } from "./tool-use-metrics-DEGMKycK.js";
41
41
  import { t as analyzeSeries } from "./series-convergence-CjO2QdRW.js";
42
- import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-DwFrCrCS.js";
43
- import { t as runEvalCampaign } from "./eval-campaign-aMG73hvX.js";
42
+ import { _ as deterministicSplit, g as BENCHMARK_SPLIT_SEED, t as benchmarks_exports } from "./benchmarks-Mtu251Jz.js";
43
+ import { t as runEvalCampaign } from "./eval-campaign-Cc8WZJ6b.js";
44
44
  import { n as pairedEvalueSequence, t as evaluateInterimReleaseConfidence } from "./sequential-Br0mAPHA.js";
45
45
  import { AxGEPA, AxSignature, ax } from "@ax-llm/ax";
46
46
  import { accessSync, appendFileSync, constants, cpSync, existsSync, mkdirSync, promises, readFileSync, readdirSync, statSync, writeFileSync } from "node:fs";
@@ -10170,9 +10170,10 @@ function precision(goldens, candidates, options = {}) {
10170
10170
  * 1. **Productive runs**: the candidate has at least
10171
10171
  * `minProductiveRuns` paired observations on items where BOTH
10172
10172
  * candidate and baseline produced a real (non-silent) score.
10173
- * 2. **Paired delta**: the lower bound of the bootstrap CI on the
10174
- * median per-item delta (candidate − baseline) on the HOLDOUT
10175
- * split is strictly greater than `pairedDeltaThreshold`.
10173
+ * 2. **Paired delta**: the lower bound of the CI on the per-item
10174
+ * holdout delta (candidate − baseline) is strictly greater than
10175
+ * `pairedDeltaThreshold`. WHICH paired statistic carries that CI
10176
+ * depends on the outcome's shape — see "Choosing the statistic".
10176
10177
  * 3. **Overfit gap**: the candidate's gap between search-split
10177
10178
  * score and holdout-split score is no worse (more positive)
10178
10179
  * than the baseline's gap by more than `overfitGapThreshold`.
@@ -10209,13 +10210,69 @@ function precision(goldens, candidates, options = {}) {
10209
10210
  * "expected" scenarios to keep in sync and nothing to trust — an item counts as
10210
10211
  * dealt because a row for it exists on at least one arm.
10211
10212
  *
10213
+ * ## Choosing the statistic
10214
+ *
10215
+ * The median paired delta is the robust default, but it goes BLIND wherever
10216
+ * ties dominate: with more than half the pairs tied its sample value is exactly
10217
+ * 0 whatever the shift on the rest, and the whole bootstrap CI collapses to
10218
+ * [0, 0]. `low > threshold` is then "no" forever at a non-negative threshold
10219
+ * and "yes" forever at a negative one — the gate refuses real gains AND
10220
+ * promotes real regressions. Measured: 76 held-out items with 15 candidate
10221
+ * wins, 5 candidate losses and 56 ties is a real +13.2pp lift whose median
10222
+ * paired delta is 0; the median-CI gate refused it at every non-negative
10223
+ * threshold while PROMOTING the mirror-image −13.2pp regression at any
10224
+ * negative one.
10225
+ *
10226
+ * So the statistic is picked by the shape of the data, in three cases:
10227
+ *
10228
+ * 1. **Two-point (pass/fail) outcomes on ANY encoding** — every value on both
10229
+ * arms is either 0 or one common positive level, so {0,1} and {0,100} and
10230
+ * {0,5} all qualify (see `pairedBinaryScale`). The estimand is the change in
10231
+ * success rate, rescaled into the caller's native units. Two statistics
10232
+ * answer about it and BOTH must agree before anything is promoted:
10233
+ * `pairedRiskDifferenceScore` (Tango's score interval) supplies the
10234
+ * interval, and McNemar's exact test — carried on
10235
+ * `pairedRiskDifferenceExact` — holds a veto at any non-negative threshold.
10236
+ * The score interval is the one that decides because it is valid at a
10237
+ * NONZERO margin, which is the only thing a noninferiority threshold asks
10238
+ * about; the exact test's own Clopper-Pearson interval is conditional on the
10239
+ * discordant count and is therefore a valid interval only at margin 0.
10240
+ * McNemar's b/c/p ship as evidence.
10241
+ * 2. **Everything else** — the bootstrap CI of the MEAN paired delta. The mean
10242
+ * is the estimator that answers the gate's own question ("by how much did
10243
+ * the score move") in the caller's units, and it is well defined on every
10244
+ * shape the median loses: tie-dominated vectors, low-cardinality lattices
10245
+ * (block scores in {⅔, 1} from averaging pass/fail leaves; integer 0-100
10246
+ * judge dimensions), and mixed arms. Measured: 26 blocks of 3 pass/fail
10247
+ * leaves carrying a real +12.8pp lift, only 23% of pairs tied, give a
10248
+ * median CI of [0, 0.333] — a lower bound of exactly 0 refuses a real lift
10249
+ * at threshold 0, which is the original bug one lattice over. There is
10250
+ * therefore no tie-fraction cutoff: any cutoff leaves the lattice case open
10251
+ * on the far side of it. `heldoutSignificance` has defaulted to the mean
10252
+ * since #316 for the same reason.
10253
+ *
10254
+ * Set `deltaStatistic: 'median'` to force the median on every input — the
10255
+ * pre-0.134 behaviour, kept for callers who want outlier robustness and accept
10256
+ * the blindness that comes with it.
10257
+ *
10258
+ * ## Fail closed
10259
+ *
10260
+ * A ZERO-WIDTH CI carries no information about how far the estimate could be
10261
+ * wrong, wherever it sits. At [0, 0] it cannot tell a gain from a regression
10262
+ * and it clears every negative threshold. Away from zero it fails the opposite
10263
+ * way: n identical positive deltas give [g, g], which clears threshold 0 on no
10264
+ * spread at all. Both are an absence of evidence, so the gate refuses with
10265
+ * `indeterminate_delta` rather than answering "promote".
10266
+ *
10212
10267
  * The decision carries a machine-readable `rejectionCode` plus an
10213
10268
  * `evidence` block with every number the gate looked at, so the
10214
10269
  * downstream researcher / paper / dashboard can re-derive the
10215
10270
  * verdict without re-running.
10216
10271
  *
10217
10272
  * See also:
10218
- * - `src/statistics.ts` for `pairedBootstrap` + `wilcoxonSignedRank`
10273
+ * - `src/statistics.ts` for `pairedBootstrap` + `wilcoxonSignedRank`, and
10274
+ * `pairedRiskDifferenceScore` / `pairedRiskDifferenceExact` (+
10275
+ * `pairedBinaryScale`) for the binary path
10219
10276
  * - `src/run-record.ts` for the input row schema
10220
10277
  * - `src/reference-replay.ts` for the older, reference-replay-
10221
10278
  * specific promotion path (still useful for replay-style evals).
@@ -10235,6 +10292,7 @@ var HeldOutGate = class {
10235
10292
  seed;
10236
10293
  costPerTaskCeiling;
10237
10294
  minCoverage;
10295
+ deltaStatistic;
10238
10296
  constructor(config) {
10239
10297
  if (!config.baselineKey) throw new Error("HeldOutGate: baselineKey is required");
10240
10298
  if (config.minCoverage !== void 0 && !(Number.isFinite(config.minCoverage) && config.minCoverage >= 0 && config.minCoverage <= 1)) throw new Error(`HeldOutGate: minCoverage must be a finite fraction in [0, 1], got ${config.minCoverage}`);
@@ -10246,6 +10304,7 @@ var HeldOutGate = class {
10246
10304
  this.baselineKey = config.baselineKey;
10247
10305
  this.resamples = config.bootstrapResamples ?? 2e3;
10248
10306
  this.seed = config.seed;
10307
+ this.deltaStatistic = config.deltaStatistic ?? "auto";
10249
10308
  if (config.costPerTaskCeiling !== void 0 && !(Number.isFinite(config.costPerTaskCeiling) && config.costPerTaskCeiling > 0)) throw new Error("HeldOutGate: costPerTaskCeiling must be a positive finite number");
10250
10309
  this.costPerTaskCeiling = config.costPerTaskCeiling;
10251
10310
  }
@@ -10295,6 +10354,21 @@ var HeldOutGate = class {
10295
10354
  searchCoverage,
10296
10355
  holdoutCoverage
10297
10356
  };
10357
+ /** Evidence for the early exits, where no CI was computed at all. The
10358
+ * statistic that WOULD have decided is still reported so a caller can see
10359
+ * which branch their data lands in without producing a verdict. */
10360
+ const earlyShape = this.chooseShape(beforeHoldout, afterHoldout);
10361
+ const noCiEvidence = {
10362
+ ...commonEvidence,
10363
+ medianPairedDelta: productiveRuns > 0 ? medianDelta(beforeHoldout, afterHoldout) : null,
10364
+ deltaStatistic: earlyShape.statistic,
10365
+ decidingDelta: null,
10366
+ pairedCI: null,
10367
+ pairedPValue: null,
10368
+ mcnemar: null,
10369
+ binaryScale: earlyShape.binaryScale,
10370
+ tieFraction: earlyShape.tieFraction
10371
+ };
10298
10372
  const missingSplitScores = [
10299
10373
  candidateSearch.length === 0 ? "candidate search" : null,
10300
10374
  candidateHoldout.length === 0 ? "candidate holdout" : null,
@@ -10305,12 +10379,7 @@ var HeldOutGate = class {
10305
10379
  promote: false,
10306
10380
  candidateId,
10307
10381
  baselineId,
10308
- evidence: {
10309
- ...commonEvidence,
10310
- medianPairedDelta: productiveRuns > 0 ? medianDelta(beforeHoldout, afterHoldout) : null,
10311
- pairedCI: null,
10312
- pairedPValue: null
10313
- },
10382
+ evidence: noCiEvidence,
10314
10383
  reason: `missing_split_scores: ${missingSplitScores.join(", ")} score evidence is absent`,
10315
10384
  rejectionCode: "missing_split_scores"
10316
10385
  };
@@ -10319,12 +10388,7 @@ var HeldOutGate = class {
10319
10388
  promote: false,
10320
10389
  candidateId,
10321
10390
  baselineId,
10322
- evidence: {
10323
- ...commonEvidence,
10324
- medianPairedDelta: productiveRuns > 0 ? medianDelta(beforeHoldout, afterHoldout) : null,
10325
- pairedCI: null,
10326
- pairedPValue: null
10327
- },
10391
+ evidence: noCiEvidence,
10328
10392
  reason: `incomplete_coverage: ${shortfalls.map(([split, cov]) => describeCoverage(split, cov)).join("; ")} — required ${fmt(this.minCoverage)} of every dealt item scored on both arms. A missing score is missing data, not an absent item: the gate will not decide on the subset that answered`,
10329
10393
  rejectionCode: "incomplete_coverage"
10330
10394
  };
@@ -10332,51 +10396,56 @@ var HeldOutGate = class {
10332
10396
  promote: false,
10333
10397
  candidateId,
10334
10398
  baselineId,
10335
- evidence: {
10336
- ...commonEvidence,
10337
- medianPairedDelta: productiveRuns > 0 ? medianDelta(beforeHoldout, afterHoldout) : null,
10338
- pairedCI: null,
10339
- pairedPValue: null
10340
- },
10399
+ evidence: noCiEvidence,
10341
10400
  reason: `few_runs: ${productiveRuns} paired holdout observation(s) < min ${this.minProductiveRuns}`,
10342
10401
  rejectionCode: "few_runs"
10343
10402
  };
10344
10403
  if (overfitGap === null || baselineOverfitGap === null) throw new Error("HeldOutGate: complete split scores did not produce overfit gaps");
10345
- const deltaTest = pairedDeltaTest(beforeHoldout, afterHoldout, {
10404
+ const decision = decidePairedPromotion(beforeHoldout, afterHoldout, {
10405
+ threshold: this.pairedDeltaThreshold,
10346
10406
  confidence: this.confidence,
10347
10407
  resamples: this.resamples,
10348
- statistic: "median",
10349
10408
  seed: this.seed,
10350
- threshold: this.pairedDeltaThreshold,
10351
- minPairs: this.minProductiveRuns
10409
+ minPairs: this.minProductiveRuns,
10410
+ statistic: this.deltaStatistic === "median" ? "median" : "mean"
10352
10411
  });
10353
- const ci = deltaTest.bootstrap;
10412
+ const literalMedian = medianDelta(beforeHoldout, afterHoldout);
10354
10413
  const wilcoxon = wilcoxonSignedRank(beforeHoldout, afterHoldout);
10414
+ const { delta: decidingDelta, low, high, label: deltaLabel } = decision;
10355
10415
  const evidence = {
10356
10416
  ...commonEvidence,
10357
- medianPairedDelta: ci.median,
10417
+ medianPairedDelta: literalMedian,
10418
+ deltaStatistic: decision.statistic,
10419
+ decidingDelta,
10358
10420
  pairedCI: {
10359
- low: ci.low,
10360
- high: ci.high
10421
+ low,
10422
+ high
10361
10423
  },
10362
- pairedPValue: wilcoxon.p
10424
+ pairedPValue: wilcoxon.p,
10425
+ mcnemar: decision.mcnemar,
10426
+ binaryScale: decision.binaryScale,
10427
+ tieFraction: decision.tieFraction
10363
10428
  };
10364
- if (!Number.isFinite(ci.low) || !Number.isFinite(ci.high) || ci.low === 0 && ci.high === 0) return {
10429
+ if (decision.indeterminate) return {
10365
10430
  promote: false,
10366
10431
  candidateId,
10367
10432
  baselineId,
10368
10433
  evidence,
10369
- reason: `indeterminate_delta: paired holdout median CI=[${fmt(ci.low)}, ${fmt(ci.high)}] carries no direction`,
10434
+ reason: `indeterminate_delta: ${decision.indeterminateCause}, so the paired holdout CI is [${fmt(low)}, ${fmt(high)}] and carries no direction — it cannot clear threshold ${fmt(this.pairedDeltaThreshold)} on evidence`,
10370
10435
  rejectionCode: "indeterminate_delta"
10371
10436
  };
10372
- if (!deltaTest.significant) return {
10373
- promote: false,
10374
- candidateId,
10375
- baselineId,
10376
- evidence,
10377
- reason: `negative_delta: paired holdout median Δ=${fmt(ci.median)} ${deltaTest.method === "bootstrap-ci" ? `CI=[${fmt(ci.low)}, ${fmt(ci.high)}]` : `exact one-sided p=${fmt(deltaTest.pValue ?? 1)}`} does not clear threshold ${fmt(this.pairedDeltaThreshold)}`,
10378
- rejectionCode: "negative_delta"
10379
- };
10437
+ const exactTestVetoes = decision.exactTestVetoes;
10438
+ if (!decision.clearsThreshold || exactTestVetoes) {
10439
+ const detail = exactTestVetoes ? ` McNemar exact p=${decision.mcnemar?.pValue.toExponential(2)} does not reject at α=${fmt(1 - this.confidence)}.` : decision.methodDetail;
10440
+ return {
10441
+ promote: false,
10442
+ candidateId,
10443
+ baselineId,
10444
+ evidence,
10445
+ reason: `negative_delta: paired holdout ${deltaLabel} Δ=${fmt(decidingDelta)} CI=[${fmt(low)}, ${fmt(high)}] does not clear threshold ${fmt(this.pairedDeltaThreshold)}.${detail}`,
10446
+ rejectionCode: "negative_delta"
10447
+ };
10448
+ }
10380
10449
  if (overfitGap > baselineOverfitGap + this.overfitGapThreshold) return {
10381
10450
  promote: false,
10382
10451
  candidateId,
@@ -10406,10 +10475,19 @@ var HeldOutGate = class {
10406
10475
  candidateId,
10407
10476
  baselineId,
10408
10477
  evidence,
10409
- reason: `promote: paired holdout median Δ=${fmt(ci.median)} CI=[${fmt(ci.low)}, ${fmt(ci.high)}] over ${productiveRuns} pairs; overfit gap candidate=${fmt(overfitGap)} vs baseline=${fmt(baselineOverfitGap)}; median cost candidate=$${fmt(medianCandidateCost)} vs baseline=$${fmt(medianBaselineCost)}`,
10478
+ reason: `promote: paired holdout ${deltaLabel} Δ=${fmt(decidingDelta)} CI=[${fmt(low)}, ${fmt(high)}] over ${productiveRuns} pairs; overfit gap candidate=${fmt(overfitGap)} vs baseline=${fmt(baselineOverfitGap)}; median cost candidate=$${fmt(medianCandidateCost)} vs baseline=$${fmt(medianBaselineCost)}`,
10410
10479
  rejectionCode: null
10411
10480
  };
10412
10481
  }
10482
+ /**
10483
+ * Which paired statistic decides this data, and the shape facts behind it —
10484
+ * for the early exits, where no interval is computed. Delegates to
10485
+ * {@link pairedDecisionShape} so the reported shape can never disagree with
10486
+ * the shape the decision actually routes on.
10487
+ */
10488
+ chooseShape(before, after) {
10489
+ return pairedDecisionShape(before, after, this.deltaStatistic === "median" ? "median" : "mean");
10490
+ }
10413
10491
  };
10414
10492
  function inferCandidateId(candidate, baselineKey) {
10415
10493
  for (const run of candidate) if (run.candidateId && run.candidateId !== baselineKey) return run.candidateId;
@@ -12294,6 +12372,6 @@ function assertProductBenchmarkRun(runDir) {
12294
12372
  return report;
12295
12373
  }
12296
12374
  //#endregion
12297
- export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, AxGepaSteeringOptimizer, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CallbackResearcher, CaptureIntegrityError, ConfigError, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, TraceContractBuilder, TraceEmitter, TraceFileMissingError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createDefaultReviewer, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBootstrap, pairedCohensDz, pairedDeltaTest, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
12375
+ export { AGENT_PROFILE_KINDS, ATIF_SCHEMA_VERSION, ATTESTATION_ALGORITHM, AgentDriver, AgentEvalError, AgentProfileCellValidationError, AnalystRegistry, AxGepaSteeringOptimizer, BENCHMARK_SPLIT_SEED, BOOTSTRAP_GATE_MIN_N, BackendIntegrityError, BenchmarkRunner, BudgetBreachError, BudgetGuard, CODING_HARNESSES, CallbackResearcher, CaptureIntegrityError, ConfigError, ConvergenceTracker, CostAccountingIncompleteError, CostCallConflictError, CostCeilingReachedError, CostLedger, CostLedgerPersistenceError, CostReceiptCaptureError, CostReservationExceededError, CostTracker, CrossFamilyError, DECISION_PAIRED_DELTA_STATISTIC, DEFAULT_AGENT_SLOS, DEFAULT_COMPLEXITY_WEIGHTS, DEFAULT_RULES as DEFAULT_FAILURE_RULES, DEFAULT_FINDERS, DEFAULT_HARNESS_OBJECTIVES, DEFAULT_MUTATION_PRIMITIVES, DEFAULT_MUTATORS, DEFAULT_PERMUTATIONS, DEFAULT_PR_REVIEW_SCORE_WEIGHTS, DEFAULT_REDACTION_RULES, DEFAULT_RED_TEAM_CORPUS, DEFAULT_RUN_SCORE_WEIGHTS, DEFAULT_SEVERITY_WEIGHTS, DEFAULT_TRACE_ANALYST_BUDGETS, DEFAULT_TRACE_ANALYST_KINDS, Dataset, DescriptionLengthGate, DockerSandboxDriver, DualAgentBench, ERROR_COUNT_PATTERNS, EvalTraceStore, ExperimentTracker, FAILURE_CLASSES, FAILURE_MODE_KIND_SPEC, FileSystemFeedbackTrajectoryStore, FileSystemRawProviderSink, FileSystemTraceStore, FindingsStore, HARBOR_IMPORT_GAP, HARNESS_NATIVE_MODEL, HeldOutGate, HoldoutAuditor, HoldoutLockedError, IMPROVEMENT_KIND_SPEC, INPUT_VALUE, INTENT_MATCH_JUDGE_VERSION, InMemoryFeedbackTrajectoryStore, InMemoryRawProviderSink, InMemoryTraceStore, InMemoryWorkspaceInspector, JudgeError, JudgeParseError, JudgeRunner, KNOWLEDGE_GAP_KIND_SPEC, KNOWLEDGE_POISONING_KIND_SPEC, LLM_CACHED_TOKENS, LLM_CACHED_TOKEN_ATTR_KEYS, LLM_CACHE_WRITE_TOKENS, LLM_CACHE_WRITE_TOKEN_ATTR_KEYS, LLM_CONTEXT_TOKENS, LLM_COST_ATTR_KEYS, LLM_COST_USD, LLM_INPUT_TOKENS, LLM_INPUT_TOKEN_ATTR_KEYS, LLM_MODEL_ATTR_KEYS, LLM_MODEL_NAME, LLM_OUTPUT_TOKENS, LLM_OUTPUT_TOKEN_ATTR_KEYS, LLM_REASONING_TOKENS, LLM_REASONING_TOKEN_ATTR_KEYS, LlmCallError, LlmClient, LlmResponseError, LlmRouteAssertionError, LockedJsonlAppender, MANN_WHITNEY_EXACT_MAX_STATES, MANN_WHITNEY_EXACT_MAX_WORK, MODEL_PRICING, MetricsCollector, ModelsUnreachableError, MultiLayerVerifier, Mutex, NoopRawProviderSink, NoopResearcher, NotFoundError, OPENINFERENCE_SPAN_KIND, OTEL_AGENT_EVAL_SCOPE, OUTPUT_VALUE, OtlpFileTraceStore, PairwiseSteeringOptimizer, ProductClient, PromptRegistry, REDACTION_VERSION, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, RESEARCH_REPORT_HARD_PAIR_FLOOR, ROLLOUT_SCHEMA, RUN_COST_ATTR_KEYS, ReplayCache, ReplayCacheMissError, ReplayError, RunCritic, RunIntegrityError, RunRecordValidationError, SEMANTIC_CONCEPT_JUDGE_VERSION, SKILL_USAGE_ANALYST, SPAN_KIND_ATTR_KEYS, SUPERVISOR_RUN_SCHEMA, SandboxHarness, ScenarioRegistry, SeatUnsetError, SingleBackendError, SkillUsageAnalyst, SpanNotFoundError, SubprocessSandboxDriver, TOOL_ARGS_CAPTURED, TOOL_LATENCY_MS, TOOL_NAME, TOOL_NAME_ATTR_KEYS, TRACE_ANALYST_ACTOR_DESCRIPTION, TRACE_ANALYST_ACTOR_DESCRIPTION_VERSION, TRACE_ANALYST_TRUNCATION_MARKER_PREFIX, TRACE_SCHEMA_VERSION, TokenCounter, TraceContractBuilder, TraceEmitter, TraceFileMissingError, TraceNotFoundError, UI_FINDING_SEVERITIES, UI_LENSES, UNIVERSAL_FINDERS, ValidationError, VerificationError, WILCOXON_EXACT_MAX_N, acquisitionPlansForKnowledgeGaps, agentProfileCellHashMaterial, agentProfileCellKey, agentProfileHash, agentProfileId, agentProfileModelId, agentVisibleFields, aggregateJudgeVerdicts, aggregateLlm, aggregatePrReviewScore, aggregateRunScore, allCriticalPassed, analyzeAntiSlop, analyzeSeries, analyzeSupervisorRun, analyzeSupervisorRunSources, analyzeTraces, appendScorecard, applyLlmSpanOtlpAttributes, applyToolSpanOtlpAttributes, argHash, asNumber, asString, assertCapabilityHeadroom, assertCrossFamily, assertLlmRoute, assertMinted, assertMintedLines, assertModelsServed, assertNoHiddenLeak, assertProductBenchmarkRun, assertRealAgentReceipts, assertRealBackend, assertReleaseConfidence, assertRolloutLine, assertRunAgentProfileCell, assertRunCaptured, assertSingleBackend, assignFeedbackSplit, assignHeldOutTag, attachCostToReport, attest, attributeCounterfactuals, backoffMs, deterministicSplit as benchmarkDeterministicSplit, benchmarks_exports as benchmarks, benjaminiHochberg, bisect, blendHeldout, blockingKnowledgeEval, bonferroni, bootstrapCi, buildAgentInterfaceProfileCell, buildAgentProfileCell, buildDefaultAnalystRegistry, buildDriverSystemPrompt, buildProductBenchmarkManifest, buildReflectionPrompt, buildReviewerPrompt, buildTraceAnalystTools, buildTraceInsightContext, buildTraceInsightPrompt, buildTrajectory, buildWorkerDriverSystemPrompt, byteLengthRange, cachedJudge, calibrateJudge, calibrateJudgeContinuous, callLlm, callLlmJson, canaryLeakView, canonicalJson, canonicalize, capabilityHeadroom, captureFetchToRawSink, causalAttribution, checkBehavioralCanary, checkCanaries, checkSlos, checkTraceContracts, clamp01, classifyFailure, classifyOtlpSpanRole, classifyTreatment, claudeCodeSupervisorRunReader, cliffsDelta, clusteredPairedBinary, cohensD, collectionPreserved, commentsForSource, commitBisect, comparePairedArms, compareReferenceReplay, compareToBaseline, compilerJudge, completionVerdict, composeParsers, composeValidators, computeExperimentStats, computeFindingId, computeToolUseMetrics, computeTraceMetrics, confidenceInterval, containsAll, contentHash, contextInputTokens, continuousAgreement, contractJudge, controlFailureClassFromVerification, controlRunToFeedbackTrajectory, controlRunToRunRecord, convertTraceStoresToOtlp, corpusInterRaterAgreement, corpusInterRaterAgreementFromJudgeScores, costForTokenPricing, costForUsage, costReceiptFromLlm, costReceiptFromLlmError, costReport, createAnalystAi, createAntiSlopJudge, createChatClient, createDefaultReviewer, createFeedbackTrajectory, createIntentMatchJudge, createLlmCorrectnessChecker, createLlmReviewer, createOtelExporter, createOtelTracingStore, createReferenceEquivalenceJudge, createReplayFetch, createSandboxPool, createSemanticConceptJudge, createTokenRecallChecker, createTraceAnalystKind, crossTraceDiff, crowdingDistance, dataDescriptionBits, decideNextUserTurn, decidePairedPromotion, decideReferenceReplayPromotion, decideReferenceReplayRunPromotion, defaultBlendWeights, defaultIsMaterial, defaultProviderRedactor, defaultReferenceReplayMatcher, defaultTraceInsightPanel, deployGateLayer, describeTraceInsightScope, diffFindings, diffScorecard, discoverPersonas, distillPlaybook, domainEvidencePattern, dominates, eProcess, ensembleJudge, errorStreakDetector, estimateCost, estimateTokens, evaluateActionPolicy, evaluateContract, evaluateHypothesis, evaluateInterimReleaseConfidence, evaluateOracles, evaluateReleaseConfidence, evaluateTraceContract, executeScenario, expandProfileAxes, expectAgent, exportProductBenchmark, exportProductBenchmarkRuns, exportRewardModel, exportRunAsOtlp, extractAssetUrls, extractErrorCount, extractOtlpAttributes, extractProducedState, extractUsage, extractUsageFromResponse, extractUsageFromSse, feedbackTrajectoriesToDatasetScenarios, feedbackTrajectoriesToOptimizerRows, feedbackTrajectoryToDatasetScenario, feedbackTrajectoryToOptimizerRow, fileContains, fileExists, fileExperimentStore, fileVerdictCache, findAutoMatchNoExpectation, findConstructorCwdDropped, findFallbackToPass, findLiteralTruePass, findProductBenchmarkArtifacts, findSkipCountsAsPass, firstNumberAttr, firstStringAttr, flattenOtlpExportToNdjson, flowLayer, fnv1a32, formatBenchmarkReport, formatDriverReport, formatFindings, formatScorecardDiff, fromHarborTrajectory, gainHistogram, gateTreatmentApplied, gateTreatmentFromMetrics, gateTreatmentFromSpans, gateTreatmentFromToolSpans, ghCliClient, gitProvenanceReader, precision as goldenPrecision, gradeOnHidden, gradeSemanticStatus, groupBy, groupRunsByAgentProfileCell, harnessAxisOf, hasCapturedToolArgs, hashContent, hashJson, hashScenarios, hashToUnit, hiddenGrade, holm, htmlContainsElement, httpGithubClient, improvementVerdict, inMemoryExperimentStore, inMemoryReferenceReplayStore, inMemoryReviewStore, inMemoryRunRecordBackend, inMemoryVerdictCache, inferDomainKeywords, inferOtlpKind, interRaterReliability, interpretCliffs, iqr, isBinaryOutcomeVector, isHiddenDestination, isJudgeSpan, isLlmSpan, isModelPriced, isOtelConfigured, isOtlpModelCall, isRealnessGated, isRetrievalSpan, isRolloutLine, isRunRecord, isSandboxSpan, isToolSpan, isTrainableSplit, isTransientLlmError, isUnavailable, iterateRawCalls, jestTestParser, jsonHasKeys, jsonShape, jsonlReferenceReplayStore, jsonlReviewStore, jsonlRunRecordBackend, judgeFamily, judgeReplayGate, judgeSpans, keyPreserved, knowledgeReadinessTracePayload, leaderboard, linterJudge, llmJudge, llmSpanFromProvider, llmSpans, loadScorecard, loadScorerFromGrader, localCommandRunner, lowercaseMutator, makeEvalTools, makeFinding, makeProposalFinding, mannWhitneyU, mapConcurrent, matchGoldens, matchSpan, maximumChargeForLlmRequest, mcnemar, mcnemarPower, mcnemarRequiredN, mergeLayerResults, mergeSteeringBundle, minimumPairsForPairedDeltaTest, mintRolloutRows, modelDescriptionBits, modelHasSnapshot, modelPriceKey, mulberry32, multiToolchainLayer, noProgressDetector, normalCdf, normalizeScores, notBlocked, objectiveEval, observeAll, observedScore, observedSplitScore, otelRunCompleteHook, otlpRowsToRunRecords, otlpRowsToTraceRunRecords, otlpToRunRecords, otlpToTraceRunRecords, pairArms, pairRunRecords, pairedBinaryScale, pairedBootstrap, pairedCohensDz, pairedDecisionShape, pairedDeltaTest, pairedDeltaTieFraction, pairedEvalueSequence, pairedMde, pairedRiskDifference, pairedRiskDifferenceExact, pairedRiskDifferenceScore, pairedSignTest, pairedTTest, paraphraseRobustness, paraphraseRobustnessScenarios, paretoChart, paretoFrontier, paretoFrontierWithCrowding, parseCorrectnessResponse, parseFeedbackTrajectoriesJsonl, parseReflectionResponse, parseRunRecordSafe, parseRuntimeTrajectoryHookEvent, partialCredit, partitionHeldOut, passAtK, passOrthogonality, pearsonR, pixelDeltaRatio, planTraceInsightQuestions, politenessPrefixMutator, positionalBias, preflightModels, printDriverSummary, probeLlm, productBenchmarkIntegrityFailures, productBenchmarkMutableSurfaces, productBenchmarkRepoIdentity, productBenchmarkSplits, profile_exports as profile, projectOtlpFlatLine, projectRuntimeTrajectoryEvidence, promptBisect, proposeSynthesisTargets, providerFromBaseUrl, pytestTestParser, ranks, readClaudeCodeSupervisorRun, readOtlpStatus, readProductBenchmarkManifest, readProductBenchmarkRecords, recordRuns, recordRunsToScorecard, redTeamDataset, redTeamReport, redactString, redactValue, referenceReplayRunsToSteeringRows, referenceReplayScenarioToRunScore, regexMatch, regexMatches, relabelImportedSplit, renderMarkdownReport, renderPlaybookMarkdown, renderPreferenceMemoryMarkdown, renderPriorFindings, renderReleaseReport, renderSteeringText, renderSupervisorRunHeadline, renderSupervisorRunMarkdown, renderUpstreamFindings, repeatedActionDetector, replayFeedbackTrajectories, replayFeedbackTrajectory, replayScorerOverCorpus, replayTraceThroughJudge, requireAgentProfileCell, requiredPairedSampleSize, requiredSampleSize, researchReport, resolveModelPricing, resolveSeat, rollupSupervisorRuns, roundTripRunRecord, routeFields, rowCount, rowWhere, runAgentControlLoop, runAssertions, runBehavioralCanaries, runCanaries, runCounterfactual, runE2EWorkflow, runEvalCampaign, runExpectations, runFailureClass, runHarnessExperiment, runIntentMatchJudge, runJudgeFleet, runKeywordCoverageJudge, runKeywordCoverageJudgeUrl, runLiveProof, runProposeReview, runProposeReviewAsControlLoop, runRecordToProductBenchmarkRecord, runReferenceEquivalenceJudge, runReferenceReplay, runScore, runSelfPlay, runSemanticConceptJudge, runTaskScore, runTestGradedScenario, runsForScenario, scalarScore, scanForMuffledGates, scoreContinuity, scoreFromEvals, scoreKnowledgeReadiness, scoreOrigin, scorePrReviewComments, scorePrReviewSource, scoreRedTeamOutput, scoreReferenceReplay, scoreTraceInsightReadiness, seatPresets, securityJudge, selectHarnessVariant, selfPreference, sentenceReorderMutator, serializeFeedbackTrajectoriesJsonl, showMeasured, signManifest, spearmanR, statusAdvanced, stopOnNoProgress, stopOnRepeatedAction, stringField, stripFencedJson, studentTCdf, subjectiveEval, summarizeAgentReceiptIntegrity, summarizeBackendIntegrity, summarizeHarnessResults, summarizePrReviewBenchmark, summarizePreferenceMemory, summaryTable, supervisorRunRolloutLines, testJudge, textInSnapshot, throwIfRunIncomplete, toAgentProfileJson, toHarborTrajectories, toHarborTrajectory, toJsonl, toLangfuseEnvelope, toOpenAiTool, toPrometheusText, toRewardRows, toSftRows, tokenizeDomainWords, toolNamesForRun, toolSpans, traceAnalystFunctionGroup, traceAnalystOnRunComplete, traceContract, traceJudge, traceJudgeEnsemble, traceSpanKindToOpenInferenceKind, tracedAnalyzeTraces, trainingReward, trainingScore, typoMutator, unmintableReasons, urlContains, userQuestionsForKnowledgeGaps, validateAgentProfileCell, validateProductBenchmarkManifest, validateProductBenchmarkRecord, validateProductBenchmarkRun, validateRolloutLine, validateRunRecord, verbosityBias, verifyAgentProfileCell, verifyAttestation, verifyCompletion, verifyManifest, visualDiff, viteDeployRunner, vitestTestParser, weightedComposite, weightedMean, weightedRecall, welchsTTest, whitespaceCollapseMutator, wilcoxonSignedRank, wilson, withAssignedFeedbackSplit, withHeldoutBlend, withJudgeRetry, withOtelPipeline, wranglerDeployRunner, writeSupervisorRunReport };
12298
12376
 
12299
12377
  //# sourceMappingURL=index.js.map