@tangle-network/agent-eval 0.135.0 → 0.135.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/CHANGELOG.md +24 -0
  2. package/dist/analyst/index.js +3 -3
  3. package/dist/{analyze-runs-DMo3Lb_y.d.ts → analyze-runs-Cda5Xkj1.d.ts} +3 -3
  4. package/dist/{analyze-runs-DMo3Lb_y.d.ts.map → analyze-runs-Cda5Xkj1.d.ts.map} +1 -1
  5. package/dist/{analyze-runs-qk8op0tN.js → analyze-runs-jjCmF8pU.js} +14 -7
  6. package/dist/analyze-runs-jjCmF8pU.js.map +1 -0
  7. package/dist/{baseline-BaPxoROc.js → baseline-BUeFcgrn.js} +2 -2
  8. package/dist/{baseline-BaPxoROc.js.map → baseline-BUeFcgrn.js.map} +1 -1
  9. package/dist/benchmarks/index.d.ts +1 -1
  10. package/dist/benchmarks/index.js +1 -1
  11. package/dist/{benchmarks-Dw1Wv_JQ.js → benchmarks-Mtu251Jz.js} +3 -3
  12. package/dist/{benchmarks-Dw1Wv_JQ.js.map → benchmarks-Mtu251Jz.js.map} +1 -1
  13. package/dist/builder-eval/index.js +1 -1
  14. package/dist/campaign/index.d.ts +2 -2
  15. package/dist/campaign/index.js +2 -2
  16. package/dist/{campaign-B1c1T0kv.js → campaign-RVIqtJh0.js} +7 -7
  17. package/dist/{campaign-B1c1T0kv.js.map → campaign-RVIqtJh0.js.map} +1 -1
  18. package/dist/cli.js +1 -1
  19. package/dist/{client-BIyh1RCr.d.ts → client-DcvgkaZi.d.ts} +13 -4
  20. package/dist/client-DcvgkaZi.d.ts.map +1 -0
  21. package/dist/contract/index.d.ts +3 -3
  22. package/dist/contract/index.d.ts.map +1 -1
  23. package/dist/contract/index.js +19 -7
  24. package/dist/contract/index.js.map +1 -1
  25. package/dist/{cost-ledger-ZAa_P4r0.js → cost-ledger-DHAjwNj7.js} +6 -2
  26. package/dist/{cost-ledger-ZAa_P4r0.js.map → cost-ledger-DHAjwNj7.js.map} +1 -1
  27. package/dist/{default-registry-CFUZyNeZ.js → default-registry-BAhV-lbE.js} +3 -3
  28. package/dist/{default-registry-CFUZyNeZ.js.map → default-registry-BAhV-lbE.js.map} +1 -1
  29. package/dist/{eval-campaign-CHFxPTVl.js → eval-campaign-Cc8WZJ6b.js} +3 -3
  30. package/dist/{eval-campaign-CHFxPTVl.js.map → eval-campaign-Cc8WZJ6b.js.map} +1 -1
  31. package/dist/fuzz.js +1 -1
  32. package/dist/hosted/index.d.ts +1 -1
  33. package/dist/{index-BoJNQR6n.d.ts → index-B4Fjfo5U.d.ts} +93 -15
  34. package/dist/index-B4Fjfo5U.d.ts.map +1 -0
  35. package/dist/{index-C21xKtxu.d.ts → index-CQsJcqch.d.ts} +3 -3
  36. package/dist/{index-C21xKtxu.d.ts.map → index-CQsJcqch.d.ts.map} +1 -1
  37. package/dist/{index-DSC51roc2.d.ts → index-DSC51roc.d.ts} +1 -1
  38. package/dist/index-DSC51roc.d.ts.map +1 -0
  39. package/dist/index.d.ts +41 -10
  40. package/dist/index.d.ts.map +1 -1
  41. package/dist/index.js +135 -57
  42. package/dist/index.js.map +1 -1
  43. package/dist/{llm-client-BNcP4v08.js → llm-client-DHx8pzyJ.js} +2 -2
  44. package/dist/{llm-client-BNcP4v08.js.map → llm-client-DHx8pzyJ.js.map} +1 -1
  45. package/dist/matrix/index.d.ts +1 -1
  46. package/dist/meta-eval/index.d.ts +1 -2
  47. package/dist/meta-eval/index.d.ts.map +1 -1
  48. package/dist/meta-eval/index.js +2 -2
  49. package/dist/multishot/index.d.ts +1 -1
  50. package/dist/openapi.json +1 -1
  51. package/dist/{paired-arms-CA_8pN01.js → paired-arms-BbFKrAU-.js} +2 -2
  52. package/dist/{paired-arms-CA_8pN01.js.map → paired-arms-BbFKrAU-.js.map} +1 -1
  53. package/dist/pipelines/index.js +2 -2
  54. package/dist/{release-report-BVZBmRZp.js → release-report-DooPguBc.js} +4 -3
  55. package/dist/{release-report-BVZBmRZp.js.map → release-report-DooPguBc.js.map} +1 -1
  56. package/dist/{release-report-CuULWKyk.d.ts → release-report-DpBxGGI1.d.ts} +2 -2
  57. package/dist/{release-report-CuULWKyk.d.ts.map → release-report-DpBxGGI1.d.ts.map} +1 -1
  58. package/dist/reporting.d.ts +3 -3
  59. package/dist/reporting.js +4 -4
  60. package/dist/{researcher-DVtruQ9U.d.ts → researcher-Doo95b50.d.ts} +2 -2
  61. package/dist/{researcher-DVtruQ9U.d.ts.map → researcher-Doo95b50.d.ts.map} +1 -1
  62. package/dist/{reward-hacking-DCdRK9TY.js → reward-hacking-a-kYs0-i.js} +2 -2
  63. package/dist/{reward-hacking-DCdRK9TY.js.map → reward-hacking-a-kYs0-i.js.map} +1 -1
  64. package/dist/rl.d.ts +45 -3
  65. package/dist/rl.d.ts.map +1 -1
  66. package/dist/rl.js +108 -22
  67. package/dist/rl.js.map +1 -1
  68. package/dist/{rubric-predictive-validity-D6Q6n9oq.js → rubric-predictive-validity-BJf-8ejY.js} +2 -2
  69. package/dist/{rubric-predictive-validity-D6Q6n9oq.js.map → rubric-predictive-validity-BJf-8ejY.js.map} +1 -1
  70. package/dist/{semantic-concept-judge-Ca--u10C.js → semantic-concept-judge-Btozx3Vc.js} +3 -3
  71. package/dist/{semantic-concept-judge-Ca--u10C.js.map → semantic-concept-judge-Btozx3Vc.js.map} +1 -1
  72. package/dist/{server-Dc_lsOYd.js → server-Bz3WQJs6.js} +3 -3
  73. package/dist/{server-Dc_lsOYd.js.map → server-Bz3WQJs6.js.map} +1 -1
  74. package/dist/{skillopt-optimization-method-Cl4XPkLC.js → skillopt-optimization-method-0UmPD6aP.js} +342 -56
  75. package/dist/skillopt-optimization-method-0UmPD6aP.js.map +1 -0
  76. package/dist/{skillopt-optimization-method-DJ3l4w8W.d.ts → skillopt-optimization-method-CwSYkv35.d.ts} +39 -9
  77. package/dist/skillopt-optimization-method-CwSYkv35.d.ts.map +1 -0
  78. package/dist/{statistics-D_4Snl-5.d.ts → statistics-CKOqre5S.d.ts} +329 -3
  79. package/dist/statistics-CKOqre5S.d.ts.map +1 -0
  80. package/dist/{statistics-RwRNu2__.js → statistics-CnGCLLqc.js} +315 -2
  81. package/dist/statistics-CnGCLLqc.js.map +1 -0
  82. package/dist/{summary-report-BxtossFi.js → summary-report-BEk8OFLs.js} +11 -6
  83. package/dist/summary-report-BEk8OFLs.js.map +1 -0
  84. package/dist/{summary-report-DGp0-_XO.d.ts → summary-report-CPMINBqs.d.ts} +182 -7
  85. package/dist/summary-report-CPMINBqs.d.ts.map +1 -0
  86. package/dist/wire/index.js +1 -1
  87. package/package.json +1 -1
  88. package/dist/analyze-runs-qk8op0tN.js.map +0 -1
  89. package/dist/client-BIyh1RCr.d.ts.map +0 -1
  90. package/dist/index-BoJNQR6n.d.ts.map +0 -1
  91. package/dist/index-DSC51roc2.d.ts.map +0 -1
  92. package/dist/judge-calibration-DFtEMlde.d.ts +0 -146
  93. package/dist/judge-calibration-DFtEMlde.d.ts.map +0 -1
  94. package/dist/skillopt-optimization-method-Cl4XPkLC.js.map +0 -1
  95. package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +0 -1
  96. package/dist/statistics-D_4Snl-5.d.ts.map +0 -1
  97. package/dist/statistics-RwRNu2__.js.map +0 -1
  98. package/dist/summary-report-BxtossFi.js.map +0 -1
  99. package/dist/summary-report-DGp0-_XO.d.ts.map +0 -1
@@ -5,10 +5,11 @@ import { n as LlmCallMetadata } from "./llm-client-BiK4HW0u.js";
5
5
  import { A as ChatClient } from "./types-Cc3qbqzj.js";
6
6
  import { m as ProposalFinding } from "./types-DVjczBM9.js";
7
7
  import { C as JudgeDimension, H as SurfaceProposer, N as ParetoParent, R as Scenario, S as JudgeConfig, _ as GateDecision, a as CampaignResult, b as GenerationRecord, c as CampaignTraceWriter, d as DispatchContext, f as DispatchFn, g as GateContribution, h as GateContext, i as CampaignCostMeter, j as MutableSurface, k as LabeledScenarioStore, o as CampaignScenarioIdentity, p as Gate, r as CampaignCellResult, v as GateResult, y as GenerationCandidate } from "./types-DiWLru6Z.js";
8
- import { _ as PairedBootstrapResult } from "./statistics-D_4Snl-5.js";
8
+ import { y as PairedBootstrapResult } from "./statistics-CKOqre5S.js";
9
9
  import { a as DatasetScenario, t as Dataset } from "./dataset-BvtnC8Dc.js";
10
10
  import { t as DetectRewardHackingInput } from "./reward-hacking-D-QqXvg-.js";
11
- import { g as TraceSpanEvent, t as HostedClient } from "./client-BIyh1RCr.js";
11
+ import { A as PairedMcNemarEvidence, D as PairedDecisionMethod, k as PairedDecisionStatistic } from "./summary-report-CPMINBqs.js";
12
+ import { g as TraceSpanEvent, t as HostedClient } from "./client-DcvgkaZi.js";
12
13
  import { z } from "zod";
13
14
  //#region src/llm-judge.d.ts
14
15
  /** A rubric dimension as a bare key or the full `{ key, description }` shape. A
@@ -891,8 +892,10 @@ interface HeldOutGateOptions<TScenario extends Scenario = Scenario> {
891
892
  bootstrapSeed?: number;
892
893
  }
893
894
  /**
894
- * Composable held-out gate: ships only when the PAIRED bootstrap CI lower bound
895
- * of the candidate-minus-baseline composite delta clears `deltaThreshold`.
895
+ * Composable held-out gate: ships only when the lower bound of the DECIDING
896
+ * paired interval on the candidate-minus-baseline composite delta clears
897
+ * `deltaThreshold` — Tango's score interval on a pass/fail holdout, the mean
898
+ * bootstrap otherwise. See {@link decidePairedPromotion}.
896
899
  */
897
900
  declare function heldOutGate<TArtifact, TScenario extends Scenario>(options: HeldOutGateOptions<TScenario>): Gate<TArtifact, TScenario>;
898
901
  //#endregion
@@ -1072,8 +1075,9 @@ interface PromotionObjective {
1072
1075
  * scale. Default 0 (⇒ "confidently better"). */
1073
1076
  gainThreshold?: number;
1074
1077
  /** A floor breach (regression) is declared when the good-direction CI lower
1075
- * bound is below −floorTolerance. When omitted it auto-scales off observed
1076
- * magnitudes (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */
1078
+ * bound is below −floorTolerance, or when the exact small-sample test proves
1079
+ * a drop past it. When omitted it auto-scales off observed magnitudes
1080
+ * (0.05 on [0,1], 5 on 0-100), matching `dimensionRegressions`. */
1077
1081
  floorTolerance?: number;
1078
1082
  }
1079
1083
  /** Per-axis verdict from the good-direction paired bootstrap. */
@@ -1083,12 +1087,35 @@ interface AxisEvidence {
1083
1087
  source: ObjectiveSource;
1084
1088
  direction: Direction;
1085
1089
  /** Paired bootstrap on the GOOD-DIRECTION delta (oriented by `direction`):
1086
- * a positive value means the candidate is better on this axis. */
1090
+ * a positive value means the candidate is better on this axis.
1091
+ *
1092
+ * DIAGNOSTIC on a pass/fail axis: there the verdict is decided on Tango's
1093
+ * score interval instead, because a percentile bootstrap over a three-atom
1094
+ * delta lattice is not a valid interval at the nonzero margin `floorTolerance`
1095
+ * and `gainThreshold` create. `ci` carries the interval that decided. */
1087
1096
  bootstrap: PairedBootstrapResult;
1097
+ /** Which paired statistic `bootstrap.low`/`.high` bracket. `'mean'` unless the
1098
+ * caller asked for the median — on a pass/fail axis the median and its whole
1099
+ * CI are pinned at 0 by tie domination and can see neither a gain nor a
1100
+ * regression. `bootstrap.median` still carries the median point estimate. */
1101
+ bootstrapStatistic: 'median' | 'mean';
1102
+ /** The interval the axis verdict was actually decided on, good-direction and
1103
+ * in the axis's native units. */
1104
+ ci: {
1105
+ low: number;
1106
+ high: number;
1107
+ };
1108
+ /** Which estimator produced `ci`. */
1109
+ decisionStatistic: PairedDecisionStatistic;
1110
+ /** McNemar's exact evidence on a pass/fail axis; null otherwise. */
1111
+ mcnemar: PairedMcNemarEvidence | null;
1112
+ /** `ci` has zero width — no evidence in either direction, so the axis is
1113
+ * neither improved nor regressed however the point estimate sits. */
1114
+ indeterminate: boolean;
1088
1115
  /** Paired observations contributing to this axis. */
1089
1116
  n: number;
1090
1117
  minimumRequired: number;
1091
- decisionMethod: 'bootstrap-ci' | 'exact-sign';
1118
+ decisionMethod: PairedDecisionMethod;
1092
1119
  gainThreshold: number;
1093
1120
  floorTolerance: number;
1094
1121
  verdict: AxisVerdict;
@@ -1121,6 +1148,9 @@ interface BuildEvidenceVectorOptions {
1121
1148
  resamples?: number;
1122
1149
  /** Fixed bootstrap seed for a deterministic, reproducible verdict. Default 1337. */
1123
1150
  seed?: number;
1151
+ /** Paired statistic every axis CI is computed on. Default `'mean'` — see
1152
+ * {@link DECISION_PAIRED_DELTA_STATISTIC} for why the median is not. */
1153
+ statistic?: 'mean' | 'median';
1124
1154
  }
1125
1155
  /**
1126
1156
  * The Evidence Bus. For each objective, pair candidate vs baseline by full
@@ -1739,4 +1769,4 @@ interface SkillOptOptimizationMethodConfig<TScenario extends Scenario, TArtifact
1739
1769
  declare function skillOptOptimizationMethod<TScenario extends Scenario, TArtifact>(config: SkillOptOptimizationMethodConfig<TScenario, TArtifact>): OptimizationMethod<TScenario, TArtifact>;
1740
1770
  //#endregion
1741
1771
  export { Objective as $, optimizationTokenUsageFromSummary as $t, RunEvalOptions as A, CanarySeverity as At, OptimizerModelBudget as B, ComparisonCost as Bt, RunImprovementLoopOptions as C, ReferenceEquivalenceJudgeResult as Cn, scoreRedTeamOutput as Ct, RunOptimizationOptions as D, LlmJudgeDimension as Dn, CanaryKind as Dt, PremeasuredOptimizationBaseline as E, runReferenceEquivalenceJudge as En, CanaryEvaluation as Et, GepaOptimizationMethodConfig as F, ExternalTextOptimizerContext as Ft, ObjectiveSource as G, OptimizationMethodProvenance as Gt, AxisVerdict as H, OptimizationMethodComparison as Ht, GepaOptimizationRecipe as I, ExternalTextOptimizerResult as It, PromotionPolicy as J, OptimizationMethodScore as Jt, ParetoSignificanceGateOptions as K, OptimizationMethodResult as Kt, GepaRunnerCommand as L, ExternalOptimizationExample as Lt, GepaAdaptiveEngineRun as M, composeGate as Mt, GepaEngineOptions as N, externalTextOptimizationMethod as Nt, RunOptimizationResult as O, LlmJudgeOptions as On, CanaryOptions as Ot, GepaEngineRun as P, ExternalTextOptimizationMethodConfig as Pt, Direction as Q, costFromLedgerSummary as Qt, gepaOptimizationMethod as R, ExternalTextEvaluationResponse as Rt, verifyLoopProvenanceRecord as S, ReferenceEquivalenceJudgeOptions as Sn, redTeamReport as St, runImprovementLoop as T, createReferenceEquivalenceJudge as Tn, CanaryAlert as Tt, BuildEvidenceVectorOptions as U, OptimizationMethodInput as Ut, AxisEvidence as V, OptimizationMethod as Vt, EvidenceVector as W, OptimizationMethodPairwise as Wt, paretoPolicy as X, OptimizationTokenUsage as Xt, buildEvidenceVector as Y, OptimizationPackageSource as Yt, paretoSignificanceGate as Z, compareOptimizationMethods as Zt, emitLoopProvenance as _, OpenAutoPrResult as _n, RedTeamCategory as _t, BuildLoopProvenanceArgs as a, planCampaignRun as an, scalarScore as at, provenanceRecordPath as b, REFERENCE_EQUIVALENCE_JUDGE_VERSION as bn, RedTeamReport as bt, LoopProvenanceArgsFromResult as c, createRunCostLedger as cn, powerPreflight as ct, LoopProvenanceEvidence as d, assertCampaignDesign as dn, DefaultProductionGateCheck as dt, CampaignCellFailureReceipt as en, ParetoResult as et, LoopProvenanceOptimizationMethod as f, assertCampaignSplitIdentity as fn, DefaultProductionGateOptions as ft, canonicalDigest as g, OpenAutoPrOptions as gn, RedTeamCase as gt, campaignMeasurementDigest as h, campaignSplitDigestFromIdentities as hn, DEFAULT_RED_TEAM_CORPUS as ht, skillOptOptimizationMethod as i, RunCampaignOptions as in, paretoFrontierWithCrowding as it, runEval as j, runCanaries as jt, runOptimization as k, llmJudge as kn, CanaryReport as kt, LoopProvenanceBackend as l, fsCampaignStorage as ln, HeldOutGateOptions as lt, buildLoopProvenanceRecord as m, campaignSplitDigest as mn, defaultProductionGate as mt, SkillOptRunnerCommand as n, CampaignRunPlanCell as nn, dominates as nt, EmitLoopProvenanceArgs as o, runCampaign as on, PowerPreflight as ot, LoopProvenanceRecord as p, campaignScenarioIdentity as pn, DefaultProductionRewardHackingOptions as pt, PromotionObjective as q, OptimizationMethodRunOptions as qt, SkillOptTrainerConfig as r, PlanCampaignRunOptions as rn, paretoFrontier as rt, EmitLoopProvenanceResult as s, CampaignStorage as sn, PowerPreflightOptions as st, SkillOptOptimizationMethodConfig as t, CampaignRunPlan as tn, crowdingDistance as tt, LoopProvenanceCandidate as u, inMemoryCampaignStorage as un, heldOutGate as ut, loopProvenanceArgsFromResult as v, openAutoPr as vn, RedTeamFinding as vt, RunImprovementLoopResult as w, ReferenceEquivalenceScenario as wn, toolNamesForRun as wt, provenanceSpansPath as x, ReferenceEquivalenceJudgeInput as xn, redTeamDataset as xt, loopProvenanceSpans as y, REFERENCE_EQUIVALENCE_INPUT_LIMITS as yn, RedTeamPayload as yt, OpenAICompatibleOptimizerModel as z, CompareOptimizationMethodsOptions as zt };
1742
- //# sourceMappingURL=skillopt-optimization-method-DJ3l4w8W.d.ts.map
1772
+ //# sourceMappingURL=skillopt-optimization-method-CwSYkv35.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"skillopt-optimization-method-CwSYkv35.d.ts","names":[],"sources":["../src/llm-judge.ts","../src/reference-equivalence-judge.ts","../src/campaign/auto-pr.ts","../src/campaign/coverage.ts","../src/campaign/external-optimizer-contracts.ts","../src/campaign/storage.ts","../src/campaign/run-campaign.ts","../src/campaign/presets/compare-optimization-methods.ts","../src/campaign/external-text-evaluation.ts","../src/campaign/external-text-optimization-contract.ts","../src/campaign/external-text-optimization.ts","../src/campaign/gates/compose.ts","../src/canary.ts","../src/red-team.ts","../src/campaign/gates/default-production-gate.ts","../src/campaign/gates/heldout-gate.ts","../src/campaign/gates/power-preflight.ts","../src/pareto.ts","../src/campaign/gates/promotion-policy.ts","../src/campaign/optimizer-model.ts","../src/campaign/gepa-optimization-method.ts","../src/campaign/presets/run-eval.ts","../src/campaign/presets/run-optimization.ts","../src/campaign/presets/run-improvement-loop.ts","../src/campaign/provenance.ts","../src/campaign/skillopt-optimization-method.ts"],"mappings":";;;;;;;;;;;;;;;;KA6CY,6BAA6B;UAExB,gBAAgB,WAAW,kBAAkB,WAAW;;;;EAIvE,MAAM;;;EAGN,aAAa;;EAEb;;EAEA;EACA;EACA;;;EAGA,UAAU;;;;;EAKV;;EAEA,aAAa,UAAU;;;EAGvB,cAAc;IAAS,UAAU;IAAW,UAAU;;;EAEtD,aAAa;EACb;IAAmB;IAAc,QAAQ,EAAE;;;;;;;;;;;;;iBAoB7B,SAAS,qBAAqB,kBAAkB,WAAW,UACzE,cACA,gBACA,MAAM,gBAAgB,WAAW,aAChC,YAAY,WAAW;;;cC7Fb;cAEA;WACX;WACA;WACA;;UAGe,qCAAqC;EACpD;EACA;;UAGe;EACf;EACA;EACA;;UAGe;;EAEf,MAAM;;EAEN;;EAEA,SAAS;;EAET,aAAa;;UAGE,wCAAwC;EACvD;EACA;EACA;EACA;;;iBA0Bc,gCACd,SAAS,mCACR,oBAAoB;;iBA2CD,6BACpB,OAAO,gCACP,SAAS,mCACR,QAAQ;;;UCjGM,kBAAkB,WAAW,kBAAkB;;EAE9D,QAAQ,eAAe,WAAW;;;EAGlC,MAAM;;;EAGN;;EAEA;EACA;;EAEA;;EAEA;;;EAGA;;EAEA,UAAU;IAAqB;IAAgB;IAAgB;;;UAGhD;EACf;EACA;EACA;EACA;;;;;iBAMc,WAAW,WAAW,kBAAkB,UACtD,SAAS,kBAAkB,WAAW,aACrC;;;;iBC1Ca,qBAAqB,kBAAkB,UACrD,oBAAoB,aACpB;;iBA2Bc,yBAAyB,kBAAkB,UACzD,UAAU,YACT,2BAA2B,KAAK;;iBAUnB,kCACd,oBAAoB,4BACpB;;iBAgBc,oBAAoB,kBAAkB,UACpD,oBAAoB,aACpB;;iBAOc,4BACd,oBAAoB,4BACpB,cACA;;;UC7Ee;EACf;EACA;EACA,MAAM,OAAO;;KAGH;KAEA,iCAAiC;UAE5B;EACf,WAAW;EACX;;UAUe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,SAAS;;EAET;;;;;;;;;;;;;;;;;;;UCrBe;;EAEf,UAAU;;EAEV,OAAO;;EAEP,KAAK;;EAEL,MAAM,cAAc,kBAAkB;;;EAGtC,OAAO,cAAc,iBAAiB;;;;;;;;;iBAUxB,qBAAqB;;;;iBA0CrB,2BAA2B;;iBAmC3B,oBAAoB;EAClC,SAAS;EACT;EACA;IACE;;;UC/Ea,mBAAmB,kBAAkB,UAAU;EAC9D,WAAW;EACX,UAAU,WAAW,WAAW;;EAEhC,SAAS;;;;;;EAMT;EACA,SAAS,YAAY,WAAW;;EAEhC;;;EAGA;;;EAGA;;;;EAIA,eAAe;EACf;EACA;;EAEA;;;EAGA,aAAa;;EAEb;;EAEA,WAAW,SAAS;;EAEpB;;;;;;;;EAQA;;;;;;;;;;EAUA;;;;;EAKA;;;;EAIA;;;EAGA;;;;EAIA;;;;;;;;;;;EAWA;;EAEA,YAAY;;EAEZ,oBAAoB,gBAAgB,gBAAgB;;;;;;EAMpD,UAAU;;;;;;;;;;;;EAYV,iBAAiB;IACf,UAAU;IACV;IACA;;;;;;;UAQa,2BAA2B;EAC1C;EACA;EACA;EACA;IACE;IACA;IACA;MACE;MACA;MACA;;;EAGJ,MAAM,mBAAmB;EACzB,MAAM;;;;;iBAMc,YAAY,kBAAkB,UAAU,WAC5D,MAAM,mBAAmB,WAAW,aACnC,QAAQ,eAAe,WAAW;UAyjBpB;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf;EACA;EACA;EACA;EACA;EACA,OAAO;;UAGQ,uBAAuB,kBAAkB,UAAU;EAClE,WAAW;EACX,WAAW,WAAW,WAAW;EACjC;EACA,SAAS,YAAY,WAAW;EAChC;EACA;EACA;EACA;;EAEA;EACA,UAAU;;;;;;iBAOI,gBAAgB,kBAAkB,UAAU,WAC1D,MAAM,uBAAuB,WAAW,aACvC;;;;KCxvBS,6BAA6B,kBAAkB,UAAU,aAAa,KAChF,mBAAmB,WAAW;;UAKf;EACf;EACA;EACA;;UAGe;EACf;;EAEA;EACA;EACA;EACA;EACA;;EAEA;;UAGe;EACf;EACA;;UAGe;EACf;EACA;;UAGe;;EAEf;;EAEA;;EAEA;EACA;;EAEA;EACA;EACA;;UAGe;;EAEf,QAAQ;;EAER,SAAS;;EAET,UAAU;;EAEV,SAAS;;EAET;EACA;;EAEA;EACA;EACA;EACA;EACA,aAAa;;;UAIE,wBAAwB,kBAAkB,UAAU;;WAE1D,iBAAiB;;WAEjB,yBAAyB;;WAEzB,6BAA6B;;WAE7B,sBACP,SAAS,gBACT,UAAU,WACV,KAAK,oBACF,QAAQ;;WAEJ,iBAAiB,YAAY,WAAW;;WAExC;WACA;;WAEA,YAAY,SAAS,6BAA6B,WAAW;;WAE7D,YAAY;;UAGN;;EAEf,eAAe;;EAEf,MAAM;;EAEN;;EAEA,aAAa;;;UAIE,mBAAmB,kBAAkB,WAAW,UAAU;;EAEzE;EACA,WACE,OAAO,wBAAwB,WAAW,eACvC,QAAQ;;UAGE;EACf;;EAEA;;EAEA;;EAEA;;;EAGA;IAAU;IAAa;;;EAEvB,kBAAkB;;EAElB;;EAEA,aAAa;;EAEb,gBAAgB;IACd;IACA;IACA;IACA;;EAEF,eAAe;;EAEf;;UAGe;;EAEf;EACA;;EAEA;EACA;EACA;;EAEA;;UAGe;;EAEf,QAAQ;EACR,MAAM;;EAEN,UAAU;EACV;;EAEA,kBAAkB;;EAElB,UAAU;;EAEV,WAAW;;EAEX;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;UAGe,kCAAkC,kBAAkB,UAAU,mBACrE,KAAK,mBAAmB,WAAW;EAC3C,SAAS,mBAAmB,WAAW;EACvC,iBAAiB;;EAEjB,gBAAgB;;EAEhB,oBAAoB;;EAEpB,eAAe;;EAEf,sBACE,SAAS,gBACT,UAAU,WACV,KAAK,oBACF,QAAQ;EACb,QAAQ,YAAY,WAAW;;;EAG/B;;EAEA,yBAAyB,6BAA6B,WAAW;;EAEjE;;;EAGA;;EAEA;;;;;iBAMoB,2BAA2B,kBAAkB,UAAU,WAC3E,MAAM,kCAAkC,WAAW,aAClD,QAAQ;;iBAosBK,sBAAsB,SAAS,oBAAoB;;iBAWnD,kCACd,SAAS,mBACT,mBAAmB,gBAClB;;;UCx7Bc;EACf;EACA;;UAGe;EACf;EACA;IACE;IACA,YAAY;IACZ;IACA;;;;;UCba;WACN;WACA;WACA;WACA;WACA;WACA,eAAe;WACf,mBAAmB;WACnB,uBAAuB;WACvB;WACA;;WAEA;WACA;WACA;WACA,QAAQ;;WAER,MAAM;WACN,WACP,SAFa,kCAGV,QAAQ;;UAGE;EACf,eAAe;EACf;EACA;IACM;;IACA;;IACA;IAAkB;;;;;;;;;UAqBT,qCACf,kBAAkB,UAClB;EAEA;EACA,QAAQ,KAAK;EACb;EACA;EACA;EACA;;EAEA;;EAEA;;EAEA,SAAS;EACT;EACA;EACA,oBAAoB,UAAU;EAC9B,oBAAoB,UAAU,WAAW,UAAU;EACnD,MAAM,SAAS,iCAAiC,QAAQ;;;;;;;;;;iBCP1C,+BAA+B,kBAAkB,UAAU,WACzE,QAAQ,qCAAqC,WAAW,aACvD,mBAAmB,WAAW;;;;;;iBCjEjB,YAAY,qBAAqB,kBAAkB,WAAW,aACzE,OAAO,MAAM,KAAK,WAAW,cAC/B,KAAK,WAAW;;;KCkBP;KAEA;UAEK;EACf,MAAM;EACN,UAAU;EACV;;;EAGA,UAAU;;UAGK;EACf,QAAQ;;EAER,QAAQ,OAAO;;EAEf,aAAa;;UAGE;EACf,MAAM;EACN;EACA;EACA;;UAGe;;;;;;;;;;EAUf;IACE;IACA;;IAEA;;;;;;;;;;;;;EAcF;IACE;IACA;IACA;IACA;;;;;;;;;;EAWF;IACE,WAAW,KAAK;IAChB;IACA;IACA;IACA;;;;;;;;iBASY,YAAY,MAAM,aAAa,OAAM,gBAAqB;;;KClG9D;UAUK;EACf,UAAU;;EAEV;;;;;EAKA;;EAEA;;EAEA;;UAGe,oBAAoB;EACnC,SAAS;;UAGM;EACf;EACA,UAAU;EACV;EACA;EACA;;UAGe;EACf,UAAU;EACV,oBAAoB,OAAO;EAC3B;;;cA0DW,yBAAyB;iBA6FtB,eAAe,aAAY,gBAAqB;;;;;iBAkBhD,mBACd,gBACA,qBACA,QAAQ,cACP;;iBA4Fa,cAAc,UAAU,mBAAmB;;;;;iBAqBrC,gBAAgB,OAAO,YAAY,gBAAgB;;;KC9T7D;KAOA,wCAAwC,KAClD;EAGA,SAAS,YAAY;;UAGN;;;EAGf,kBAAkB;;;;;EAKlB;;EAEA;;EAEA;;EAEA;;;;EAIA;;;;EAIA;;;;;EAKA;;;;EAIA;;;EAGA;;;;EAIA,iBAAiB;;;EAGjB,aAAa;;EAEb,gBAAgB;;EAEhB,SAAS;;;;;EAKT,iBAAiB;;;;;iBAMH,sBAAsB,WAAW,kBAAkB,UACjE,SAAS,+BACR,KAAK,WAAW;;;UC/EF,mBAAmB,kBAAkB,WAAW;EAC/D,WAAW;;;EAGX;;EAEA;;;;EAIA;;EAEA;;EAEA;;;;;;;;iBASc,YAAY,WAAW,kBAAkB,UACvD,SAAS,mBAAmB,aAC3B,KAAK,WAAW;;;;;;;;;;;;;;;;;;;;;;;;;;UCrBF;;EAEf;;;EAGA;;EAEA;;EAEA;;;;;;;EAOA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;;EAIA;;;EAGA;EACA;EACA;;;EAGA;;EAEA;;;;;;iBAec,eAAe,MAAM,wBAAwB;;;;;;;;;;;;;;;;;;KClEjD;UAEK,UAAU;;EAEzB;EACA,WAAW;EACX,QAAQ,WAAW;;UAGJ,aAAa;EAC5B,UAAU;EACV,WAAW;;EAEX,cAAc;IAAQ,WAAW;IAAG,WAAW;;;;iBAIjC,UAAU,GAAG,GAAG,GAAG,GAAG,GAAG,YAAY,UAAU;;;;;;iBAmB/C,eAAe,GAAG,YAAY,KAAK,YAAY,UAAU,OAAO,aAAa;;;;;;;;;;iBA4B7E,YAAY,GAC1B,YAAY,KACZ,YAAY,UAAU,MACtB;EAAW,UAAU,QAAQ;IAC5B;EAAQ,WAAW;EAAG;;;;;;;;;;;;;;iBAyCT,iBAAiB,GAC/B,YAAY,KACZ,YAAY,UAAU,OACrB;EAAQ,WAAW;EAAG;;;;;;;iBA6BT,2BAA2B,GACzC,YAAY,KACZ,YAAY,UAAU,OACrB;EAAQ,WAAW;EAAG;;;;;;KCtHb;EAAoB;;EAAwB;EAAmB;;UAE1D;;EAEf;EACA,QAAQ;;;EAGR,WAAW;;;;EAIX;;;;;EAKA;;;KAIU;UAEK;EACf;EACA,QAAQ;EACR,WAAW;;;;;;;;EAQX,WAAW;;;;;EAKX;;;EAGA;IAAM;IAAa;;;EAEnB,mBAAmB;;EAEnB,SAAS;;;EAGT;;EAEA;EACA;EACA,gBAAgB;EAChB;EACA;EACA,SAAS;;UAGM;;EAEf,MAAM;;;EAGN;;;EAGA;IAAQ;IAAmB;;;;;;KAMjB,mBAAmB,IAAI,mBAAmB;UAErC;;;;EAIf;;EAEA;;EAEA;;EAEA;;;EAGA;;;;;;;;iBASc,oBAAoB,WAAW,kBAAkB,UAC/D,KAAK,YAAY,WAAW,YAC5B,YAAY,sBACZ,OAAM,6BACL;;;;;;;;cAwIU,cAAc;UAiFV,sCAAsC;;EAErD,YAAY;;;EAGZ,SAAS;;EAET;;;;;;;iBAQc,uBAAuB,qBAAqB,kBAAkB,WAAW,UACvF,SAAS,gCACR,KAAK,WAAW;;;KCpXP,uBAAuB;;UAGlB;EACf;EACA;EACA;EACA,QAAQ;;;;;UC6CO;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;;;EAKA,eAAe;;;UAIA,sBAAsB;;EAErC;;;KAIU,wBAAwB;;;;;;;KAQxB;EAEN;EACA,KAAK;;EAGL;EACA,eAAe;;EAGf;EACA,eAAe;;EAEf;;EAEA;EACA;EACA;EACA;EACA;EACA;EACA;;EAGA;EACA,eAAe;EACf;;EAGA;EACA,eAAe;EACf;;EAGA;EACA,kBAAkB;EAClB,cAAc;EACd;;;KAIM,oBAAoB;UAEf,6BAA6B,kBAAkB,UAAU;;EAExE;;EAEA,QAAQ;;EAER;;EAEA;;EAEA;;;;;EAKA;;EAEA;;EAEA;;EAEA;;;;;;EAMA,YAAY;;;;;;EAMZ,oBAAoB,UAAU;;EAE9B,oBAAoB,UAAU,WAAW,UAAU;;EAEnD,SAAS;;;;;EAKT;EACA,SAAS;;;;;;;;;;iBAWK,uBAAuB,kBAAkB,UAAU,WACjE,QAAQ,6BAA6B,WAAW,aAC/C,mBAAmB,WAAW;;;UClLhB,eAAe,kBAAkB,UAAU,mBAClD,KAAK,mBAAmB,WAAW;EAC3C;;;;;iBAMoB,QAAQ,kBAAkB,UAAU,WACxD,MAAM,eAAe,WAAW,aAC/B,QAAQ,eAAe,WAAW;;;UC0BpB,gCAAgC,WAAW,kBAAkB;;EAE5E;;EAEA,UAAU,eAAe,WAAW;;UAGrB,2BAA2B,kBAAkB,UAAU,mBAC9D,KAAK,mBAAmB,WAAW;;EAE3C,iBAAiB;;;;;;;;;EASjB,sBAAsB,gCAAgC,WAAW;;EAEjE,sBACE,SAAS,gBACT,UAAU,WACV,KAAK,WAAW,mBAAmB,WAAW,+BAC3C,QAAQ;;EAEb,UAAU,gBAAgB;EAC1B;EACA;;;EAGA;;;EAGA;;EAEA,WAAW,cAAc;;;;;;;;;;;EAWzB,qBAAqB;IACnB;IACA;IACA,YAAY;MACV;MACA,UAAU,eAAe,WAAW;MACpC;;IAEF,SAAS;;IAET,aAAa;IACb;QACI,QAAQ,cAAc;;;;;;;;;;;;;;;;;;EAkB5B,oBAAoB,UAAU,eAAe,WAAW;;KAG9C,uBACV,kBAAkB,UAClB,aACE,2BAA2B,WAAW;UAEzB,sBAAsB,WAAW,kBAAkB;EAClE,aAAa;IACX,QAAQ;IACR,UAAU;MACR;MACA,SAAS;MACT,UAAU,eAAe,WAAW;;;;EAIxC,iBAAiB;EACjB,eAAe;EACf;;;;EAIA;;;;EAIA;EACA,kBAAkB,eAAe,WAAW;;EAE5C,MAAM;;;;;;EAMN,gBAAgB;;;;;iBAMI,gBAAgB,kBAAkB,UAAU,WAChE,MAAM,uBAAuB,WAAW,aACvC,QAAQ,sBAAsB,WAAW;;;KCpJhC,0BACV,kBAAkB,UAClB,aACE,uBAAuB,WAAW;;;EAGpC,kBAAkB;;;;;;;;;EASlB;;;;EAIA,MAAM,KAAK,WAAW;;;;;EAKtB;;EAEA;EACA;;;;;;;;EAQA,cAAc,eAAe,gBAAgB,iBAAiB,mBAAmB;;UAGlE,yBAAyB,WAAW,kBAAkB,kBAC7D,sBAAsB,WAAW;EACzC,mBAAmB,eAAe,WAAW;EAC7C,iBAAiB,eAAe,WAAW;EAC3C,uBAAuB,eAAe,WAAW;EACjD,qBAAqB;EACrB,YAAY,QAAQ,WAAW,KAAK,WAAW;;;;EAI/C;;;;;EAKA;EACA,WAAW,kBAAkB;;;;;iBAMT,mBAAmB,kBAAkB,UAAU,WACnE,MAAM,0BAA0B,WAAW,aAC1C,QAAQ,yBAAyB,WAAW;;;UC1B9B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;;EAGA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,UAAU,YAAY;;EAEtB;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA;EACA;EACA;EACA;;UAGe;EACf;IACE;IACA;;EAEF;IACE;IACA;IACA;IACA;MACE;MACA;MACA;MACA;;;EAGJ;;UAGe;EACf;EACA,MAAM;EACN;EACA,aAAa;;;;;;;UAQE;EACf;;EAEA;EACA;EACA;EACA;;EAEA;EACA;;EAEA;EACA;;EAEA;;EAEA,YAAY;;EAEZ,qBAAqB;;EAErB,UAAU;;EAEV;;EAEA;IACE,UAAU;IACV;IACA;IACA,mBAAmB;;;;;;EAMrB;;EAEA;;EAEA;;;EAGA;;EAEA,SAAS;EACT;EACA;;UAGe,wBAAwB,WAAW,kBAAkB;EACpE;EACA;EACA;EACA,iBAAiB;EACjB,eAAe;EACf;EACA;;EAEA,wBAAwB,eAAe,WAAW;;EAElD,aAAa;IACX;IACA,YAAY;IACZ;;;IAGA,UAAU;MACR;MACA,SAAS;MACT,UAAU,eAAe,WAAW;;;EAGxC,MAAM;;;;EAIN;EACA,mBAAmB,eAAe,WAAW;EAC7C,iBAAiB,eAAe,WAAW;EAC3C,qBAAqB;EACrB,uBAAuB,eAAe,WAAW;;EAEjD,cAAc,cAAc;EAC5B;EACA;EACA,qBAAqB;;UAGN,6BAA6B,WAAW,kBAAkB;EACzE;EACA;EACA;EACA,iBAAiB;EACjB,QAAQ,yBAAyB,WAAW;EAC5C,cAAc,cAAc;EAC5B;EACA;;;iBAIc,6BAA6B,WAAW,kBAAkB,UACxE,OAAO,6BAA6B,WAAW,aAC9C,wBAAwB,WAAW;;iBA4CtB,0BAA0B,WAAW,kBAAkB,UACrE,MAAM,wBAAwB,WAAW,aACxC;;iBAoRa,0BAA0B,WAAW,kBAAkB,UACrE,UAAU,eAAe,WAAW;;iBAoCtB,2BAA2B,QAAQ,uBAAuB;;iBAa1D,gBAAgB;;;;;;;;;;;;iBA2KhB,oBACd,QAAQ,sBACR;EAAQ;IACP;;iBAiJa,qBAAqB;;;;iBAMrB,oBAAoB;UAInB;EACf,QAAQ;EACR,OAAO;;EAEP;EACA;;UAGe,uBAAuB,WAAW,kBAAkB,kBAC3D,wBAAwB,WAAW;;EAE3C,SAAS;;;EAGT,eAAe;;;;;;;;;;;;iBA+EK,mBAAmB,WAAW,kBAAkB,UACpE,MAAM,uBAAuB,WAAW,aACvC,QAAQ;;;UC/8BM;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;;;;;EAKA,YAAY;;KAGF,wBAAwB;UAEnB,iCAAiC,kBAAkB,UAAU;EAC5E;;EAEA;EACA;;EAEA;EACA,SAAS;;;;;EAKT,WAAW;;EAEX;;EAEA;EACA;;EAEA;EACA;EACA,oBAAoB,UAAU;EAC9B,oBAAoB,UAAU,WAAW,UAAU;EACnD,SAAS;EACT,SAAS;;;iBAIK,2BAA2B,kBAAkB,UAAU,WACrE,QAAQ,iCAAiC,WAAW,aACnD,mBAAmB,WAAW"}
@@ -1,5 +1,148 @@
1
1
  import { g as JudgeScore } from "./types-Cc3qbqzj.js";
2
- import { i as ContinuousAgreementOptions, r as ContinuousAgreement } from "./judge-calibration-DFtEMlde.js";
2
+ //#region src/judge-calibration.d.ts
3
+ /**
4
+ * Judge calibration — measure judge quality against human gold + bias.
5
+ *
6
+ * Workflow:
7
+ * 1. Build a golden set: {itemId, humanScore}[].
8
+ * 2. Run candidate judges; each produces {itemId, score}.
9
+ * 3. `calibrateJudge(golden, candidate)` reports κ + Pearson + MAE.
10
+ * 4. `calibrateJudgeContinuous(golden, candidate)` adds quadratic-weighted
11
+ * κ over the un-rounded [0,1] scores plus ICC(2,1), Pearson, Spearman,
12
+ * and bootstrap CIs — use this for fine-grained judges where rounding
13
+ * to int discards information (e.g. 0.78 vs 0.81 both round to 1 and
14
+ * look "perfectly agreed" to integer κ).
15
+ * 5. Run bias probes (positional, verbosity, self-preference) to
16
+ * detect systematic score inflation.
17
+ * 6. For N≥2 judges on the same items, `continuousAgreement(scores)`
18
+ * reports ICC(2,1) + κ_w + Pearson + Spearman with bootstrap CIs.
19
+ *
20
+ * Returns actionable diagnostics, not a single number. Consumers then
21
+ * decide whether to trust the judge, retrain it, or add a tie-breaker.
22
+ */
23
+ interface GoldenItem {
24
+ itemId: string;
25
+ humanScore: number;
26
+ /** Optional group used for per-group bias audits (e.g. model-of-output family). */
27
+ group?: string;
28
+ }
29
+ interface CandidateScore {
30
+ itemId: string;
31
+ score: number;
32
+ /** Optional — enables positional-bias analysis (did order matter?). */
33
+ positionOfAInput?: 'first' | 'second';
34
+ }
35
+ interface CalibrationResult {
36
+ n: number;
37
+ pearson: number;
38
+ /** Cohen's κ with quadratic weights over integer-rounded scores. */
39
+ kappa: number;
40
+ /** Mean absolute error vs human. */
41
+ mae: number;
42
+ /** Worst-5 miscalibrations (largest |judge - human|). */
43
+ worstItems: Array<{
44
+ itemId: string;
45
+ judge: number;
46
+ human: number;
47
+ delta: number;
48
+ }>;
49
+ }
50
+ /**
51
+ * Measure judge quality against human gold labels: computes Cohen's κ, Pearson correlation, and MAE over matched item ids.
52
+ */
53
+ declare function calibrateJudge(golden: GoldenItem[], candidate: CandidateScore[]): CalibrationResult;
54
+ interface PositionalBiasResult {
55
+ /**
56
+ * Score delta (first-position - second-position) averaged across items
57
+ * presented in both positions. Non-zero = positional bias.
58
+ */
59
+ avgDelta: number;
60
+ n: number;
61
+ }
62
+ /**
63
+ * Feed the same items to the judge twice with A/B swapped and pass all
64
+ * results here. Items that don't appear in both positions are ignored.
65
+ */
66
+ declare function positionalBias(scores: CandidateScore[]): PositionalBiasResult;
67
+ interface VerbosityBiasResult {
68
+ /** Pearson correlation between output length and score. Strong positive = verbosity bias. */
69
+ pearson: number;
70
+ n: number;
71
+ }
72
+ declare function verbosityBias(samples: Array<{
73
+ outputLen: number;
74
+ score: number;
75
+ }>): VerbosityBiasResult;
76
+ interface SelfPreferenceResult {
77
+ /** Mean judge score when judge's family matches output's family. */
78
+ inFamilyMean: number;
79
+ outOfFamilyMean: number;
80
+ deltaMean: number;
81
+ n: number;
82
+ }
83
+ /**
84
+ * Pass the same scenarios scored with judge-model X grading outputs from
85
+ * model X (in-family) and model Y (out-of-family). Non-zero delta
86
+ * indicates self-preference.
87
+ */
88
+ declare function selfPreference(samples: Array<{
89
+ score: number;
90
+ inFamily: boolean;
91
+ }>): SelfPreferenceResult;
92
+ interface ContinuousAgreement {
93
+ /** Cohen's κ_w with quadratic weights, computed on raw [0,1] scores. */
94
+ weightedKappa: number;
95
+ /** ICC(2,1): two-way random effects, absolute agreement, single rater. */
96
+ icc: number;
97
+ /** Pearson product-moment correlation (averaged over rater pairs if N>2). */
98
+ pearson: number;
99
+ /** Spearman rank correlation (averaged over rater pairs if N>2). */
100
+ spearman: number;
101
+ /** 95% bootstrap percentile CIs over items. */
102
+ ci: {
103
+ icc: [number, number];
104
+ weightedKappa: [number, number];
105
+ };
106
+ /** Number of complete items (no NaN across raters). */
107
+ n: number;
108
+ /** Number of raters. */
109
+ raters: number;
110
+ }
111
+ interface ContinuousAgreementOptions {
112
+ /** Bootstrap iterations. Default 1000. Set to 0 to skip CIs (CI = [NaN, NaN]). */
113
+ bootstrap?: number;
114
+ /** κ weighting scheme. Default 'quadratic'. */
115
+ weights?: 'linear' | 'quadratic';
116
+ /** PRNG seed for reproducible bootstrap. Default 0xC0FFEE. */
117
+ seed?: number;
118
+ /** Confidence level for percentile CI. Default 0.95. */
119
+ ciLevel?: number;
120
+ }
121
+ /**
122
+ * Inter-rater agreement on continuous (typically [0,1]) scores.
123
+ *
124
+ * `scores` has shape [n_items][n_raters]. Rows with any non-finite entry
125
+ * are dropped. Returns NaN metrics if fewer than 2 raters or 2 complete
126
+ * items remain.
127
+ */
128
+ declare function continuousAgreement(scores: number[][], opts?: ContinuousAgreementOptions): ContinuousAgreement;
129
+ interface ContinuousCalibrationResult extends CalibrationResult {
130
+ /** Cohen's κ_w computed on raw (un-rounded) scores. */
131
+ weightedKappaContinuous: number;
132
+ /** ICC(2,1) treating golden + candidate as two raters. */
133
+ icc: number;
134
+ spearman: number;
135
+ ci: {
136
+ icc: [number, number];
137
+ weightedKappa: [number, number];
138
+ };
139
+ }
140
+ /**
141
+ * Extends `calibrateJudge` with continuous-value agreement metrics while
142
+ * retaining its base calibration summary.
143
+ */
144
+ declare function calibrateJudgeContinuous(golden: GoldenItem[], candidate: CandidateScore[], opts?: ContinuousAgreementOptions): ContinuousCalibrationResult;
145
+ //#endregion
3
146
  //#region src/statistics.d.ts
4
147
  /** Identity: dimensions already follow "higher = better" by prompt convention
5
148
  * (inverted dims like hallucination are scored 10 = best at the source). */
@@ -503,6 +646,32 @@ interface ProportionInterval {
503
646
  * proportion. `n = 0 ⇒ {0, 0, 0}`.
504
647
  */
505
648
  declare function wilson(successes: number, n: number, confidence?: number): ProportionInterval;
649
+ /**
650
+ * Are these per-item outcomes binary (every value exactly 0 or 1)?
651
+ *
652
+ * The discriminator a promotion gate needs before choosing a paired statistic.
653
+ * On binary outcomes the paired delta vector lives in {-1, 0, +1} and is
654
+ * normally dominated by zeros (both arms solve, or both arms miss, most items),
655
+ * so its MEDIAN is pinned at exactly 0 no matter how large the real shift in
656
+ * success rate is — and a bootstrap CI on that median collapses to [0, 0].
657
+ * A gate keying on `ci.low > threshold` is then structurally unable to see
658
+ * either a gain or a regression. Detect this shape and switch to the
659
+ * paired-binary estimators ({@link mcnemar}, {@link pairedRiskDifference})
660
+ * instead of silently answering "no" forever.
661
+ *
662
+ * Empty input is NOT binary: there is no evidence of the outcome's shape, and
663
+ * defaulting an empty vector into the binary branch would pick a statistic on
664
+ * no data at all.
665
+ *
666
+ * NOT the right discriminator for a gate. It recognises the literal {0, 1}
667
+ * encoding and nothing else, so a pass/fail dimension emitted on 0-100 — which
668
+ * judges in this codebase do routinely — reads as non-binary, and a single
669
+ * partial-credit score in an otherwise pass/fail vector flips it to false while
670
+ * leaving the median just as blind. Gates want {@link pairedBinaryScale} (any
671
+ * two-point encoding). This predicate remains for callers that specifically
672
+ * mean "literally 0/1".
673
+ */
674
+ declare function isBinaryOutcomeVector(values: ArrayLike<number>): boolean;
506
675
  /** Result of a McNemar paired-binary significance test. */
507
676
  interface McNemarResult {
508
677
  /** Total paired observations. */
@@ -555,8 +724,165 @@ interface RiskDifferenceResult {
555
724
  * the discordant counts, not the independent-samples formula (which overstates
556
725
  * the interval by ignoring the pairing). Inputs are paired 0/1 (or boolean)
557
726
  * arrays, control first. Throws on unequal lengths.
727
+ *
728
+ * REPORTING ONLY — do NOT decide a promotion on this interval. The CI is a Wald
729
+ * normal approximation, which badly UNDERCOVERS when only a handful of pairs are
730
+ * discordant: at n = 3 with b = 2, c = 0 it returns [0.133, 1.000], excluding 0,
731
+ * while McNemar's exact test on the same data gives p = 0.50. A gate keying on
732
+ * `lower > 0` would promote noise. Use {@link pairedRiskDifferenceExact}, whose
733
+ * interval is dual to the exact test by construction, for any decision.
558
734
  */
559
735
  declare function pairedRiskDifference(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): RiskDifferenceResult;
736
+ /** A paired binary effect size with an EXACT interval and the exact test that
737
+ * bounds it — one object so a caller cannot read the estimate without the
738
+ * significance it is entitled to. */
739
+ interface ExactRiskDifferenceResult {
740
+ /** Total paired observations. */
741
+ n: number;
742
+ /** Discordant pairs: treatment-win count. */
743
+ b: number;
744
+ /** Discordant pairs: control-win count. */
745
+ c: number;
746
+ /** Discordant pairs (b + c) — the only ones carrying information. */
747
+ nDiscordant: number;
748
+ /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
749
+ riskDifference: number;
750
+ /** Exact conditional CI lower bound. 0 when there are no discordant pairs. */
751
+ lower: number;
752
+ /** Exact conditional CI upper bound. 0 when there are no discordant pairs. */
753
+ upper: number;
754
+ /** Confidence level used. */
755
+ confidence: number;
756
+ /** McNemar's exact two-sided p-value on the same discordant counts. */
757
+ pValue: number;
758
+ }
759
+ /**
760
+ * Paired risk difference with the EXACT CONDITIONAL interval — the estimator a
761
+ * promotion gate may decide on.
762
+ *
763
+ * Conditional on the number of discordant pairs m = b + c, the treatment-win
764
+ * count b is Binomial(m, π) with π = P(treatment wins | discordant), and the
765
+ * risk difference is an exact reparameterisation: RD = (2π − 1)·m/n. So a
766
+ * Clopper-Pearson exact interval for π maps straight onto RD. This buys the
767
+ * property the Wald interval in {@link pairedRiskDifference} does not have:
768
+ *
769
+ * **`lower > 0` ⟺ McNemar's exact test rejects at α = 1 − confidence.**
770
+ *
771
+ * Clopper-Pearson excludes π = 0.5 exactly when the two-sided exact binomial
772
+ * test of π = 0.5 rejects, and that test IS {@link mcnemar}'s p-value — so the
773
+ * interval and the test can never disagree, and a gate keyed on `lower` cannot
774
+ * promote what the exact test refuses. The exact p is returned in the same
775
+ * object so the two are impossible to compute apart.
776
+ *
777
+ * The interval is conservative (exact intervals over-cover; conditioning on m
778
+ * discards the concordant pairs' information about m itself). That is the
779
+ * correct direction for a promotion gate: it refuses more often, never less.
780
+ *
781
+ * With m = 0 there are no discordant pairs and π is not identified: the result
782
+ * is the degenerate [0, 0] with p = 1. That is NOT evidence of equivalence —
783
+ * callers must treat a zero-width interval as "cannot decide", not as "no
784
+ * difference". Inputs are paired 0/1 (or boolean) arrays, control first.
785
+ * Throws on unequal lengths.
786
+ */
787
+ declare function pairedRiskDifferenceExact(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): ExactRiskDifferenceResult;
788
+ /** A paired binary effect size with an interval that is valid at a NONZERO
789
+ * margin — the estimator a noninferiority decision may be made on. */
790
+ interface ScoreRiskDifferenceResult {
791
+ /** Total paired observations. */
792
+ n: number;
793
+ /** Discordant pairs: treatment-win count. */
794
+ b: number;
795
+ /** Discordant pairs: control-win count. */
796
+ c: number;
797
+ /** Discordant pairs (b + c). */
798
+ nDiscordant: number;
799
+ /** Paired risk difference p(treatment) − p(control) = (b − c) / n. */
800
+ riskDifference: number;
801
+ /** Score-interval lower bound on the population risk difference. */
802
+ lower: number;
803
+ /** Score-interval upper bound on the population risk difference. */
804
+ upper: number;
805
+ /** Confidence level used. */
806
+ confidence: number;
807
+ }
808
+ /**
809
+ * Paired risk difference with TANGO'S (1998) SCORE INTERVAL — the estimator a
810
+ * promotion gate may decide on **at a nonzero margin**.
811
+ *
812
+ * {@link pairedRiskDifferenceExact} conditions on the observed discordant count
813
+ * `m = b + c`, builds a Clopper-Pearson interval for the win share among those
814
+ * `m` pairs, and multiplies by the observed `m/n`. That is exact for testing
815
+ * RD = 0 — it is dual to McNemar — but it is NOT a confidence interval for the
816
+ * population risk difference at a nonzero margin, because the sampling
817
+ * variability of `m/n` itself is discarded. The gap is not academic: with the
818
+ * production caller's `pairedDeltaThreshold: -0.05`, a process whose true risk
819
+ * difference sits exactly on that margin clears a nominal-95 % `lower > margin`
820
+ * check 24.75 % of the time at n = 40 and 43.95 % at n = 76 (2000 replicates
821
+ * each) when the conditional interval decides.
822
+ *
823
+ * Tango's interval inverts the score test of RD = delta, which estimates the
824
+ * nuisance loss rate under each hypothesised delta instead of fixing it at the
825
+ * observed value, so `m` contributes its own uncertainty. It is the method
826
+ * `ratesci::scorepairci` uses for paired risk-difference noninferiority, and it
827
+ * is not conditional, so it stays valid as the margin moves away from zero.
828
+ *
829
+ * The bounds are found by bisecting `tangoScore(delta) = ±z` — the score is
830
+ * monotone decreasing in delta, so each crossing is unique. Inputs are paired
831
+ * 0/1 (or boolean) arrays, control first. Throws on unequal lengths.
832
+ */
833
+ declare function pairedRiskDifferenceScore(control: ArrayLike<number | boolean>, treatment: ArrayLike<number | boolean>, confidence?: number): ScoreRiskDifferenceResult;
834
+ /**
835
+ * The common positive level `s` such that EVERY value across both paired arms is
836
+ * exactly 0 or `s` — i.e. the outcome is pass/fail, whatever encoding it arrived
837
+ * in. Returns null when the outcomes are not two-point, when the two arms use
838
+ * different levels, or when no positive value was observed at all (all-zero
839
+ * arms: the level is not identified, and there is nothing to decide anyway).
840
+ *
841
+ * This is the scale-aware successor to {@link isBinaryOutcomeVector}, which only
842
+ * recognises literal {0, 1}. Judges in this codebase emit dimensions on 0-100 as
843
+ * well as [0,1] (see `detectScale` in `campaign/gates/statistical-heldout.ts`),
844
+ * so a pass/fail dimension routinely arrives as {0, 100} and a {0,1}-only test
845
+ * silently sends it down the median path that cannot see it. Any positive level
846
+ * is accepted, not just 1 and 100: for a two-point {0, s} outcome the mean paired
847
+ * delta is exactly s·(b − c)/n, so the binary estimators apply after dividing by
848
+ * s and rescaling the result back into the caller's native units.
849
+ *
850
+ * Non-finite values ⇒ null: an unusable outcome must not be classified as a
851
+ * clean pass/fail shape.
852
+ */
853
+ declare function pairedBinaryScale(before: ArrayLike<number>, after: ArrayLike<number>): number | null;
854
+ /** Fraction of paired observations whose delta is an exact tie (|after − before|
855
+ * < 1e-9). Throws on unequal sample sizes; 0 pairs ⇒ 0. */
856
+ declare function pairedDeltaTieFraction(before: ArrayLike<number>, after: ArrayLike<number>): number;
857
+ /**
858
+ * The paired-delta statistic a DECISION is computed on, package-wide.
859
+ *
860
+ * The mean paired delta is the estimator that answers the question a promotion
861
+ * gate asks — "by how much did the candidate move the score" — in the caller's
862
+ * own units, and it equals the aggregate lift everyone quotes. The MEDIAN
863
+ * answers a different question and loses the answer to this one in every regime
864
+ * eval data actually lands in:
865
+ * - TWO-POINT (pass/fail) outcomes on any encoding: the delta vector lives in
866
+ * {−s, 0, +s} dominated by zeros, so the median and its whole bootstrap CI
867
+ * are pinned at exactly 0 however large the shift. (Decide these on
868
+ * {@link pairedRiskDifferenceExact} instead — same estimand, exact interval.)
869
+ * - TIE-DOMINATED outcomes: at half the pairs tied the sample median is 0 by
870
+ * construction, and `ci.low > threshold` then answers "no" forever at a
871
+ * non-negative threshold and "yes" forever at a negative one.
872
+ * - LOW-CARDINALITY outcomes, even well below half ties: judge dimensions on
873
+ * integer 0-100, and block scores like {⅔, 1} from averaging pass/fail
874
+ * leaves, put the median on a coarse lattice whose bootstrap percentiles
875
+ * land on atoms. Measured: 26 blocks of 3 pass/fail leaves carrying a real
876
+ * +12.8pp lift, only 23% of pairs tied, gives a median CI of [0, 0.333] —
877
+ * lower bound exactly 0, so a gate at threshold 0 refuses a real lift.
878
+ * That last case is why there is no tie-fraction threshold here: any cutoff on
879
+ * ties leaves the lattice case open on the other side of it.
880
+ *
881
+ * `heldoutSignificance` has defaulted to the mean since #316 for the same
882
+ * reason. The median remains available per call site for callers who
883
+ * specifically want outlier robustness and accept the blindness.
884
+ */
885
+ declare const DECISION_PAIRED_DELTA_STATISTIC: 'mean';
560
886
  /**
561
887
  * Unbiased pass@k for code generation (Chen et al. 2021, "Evaluating Large
562
888
  * Language Models Trained on Code"). Given `n` independent samples for one
@@ -638,5 +964,5 @@ declare function eProcess(opts?: EProcessOptions): EProcess;
638
964
  * stream, including 0. */
639
965
  declare function mulberry32(seed: number): () => number;
640
966
  //#endregion
641
- export { partialCredit as $, benjaminiHochberg as A, interpretCliffs as B, RankTestOptions as C, WeightedCompositeInput as D, WILCOXON_EXACT_MAX_N as E, corpusInterRaterAgreement as F, mulberry32 as G, mcnemar as H, corpusInterRaterAgreementFromJudgeScores as I, pairedCohensDz as J, normalizeScores as K, eProcess as L, cliffsDelta as M, cohensD as N, WeightedCompositeResult as O, confidenceInterval as P, pairedTTest as Q, holm as R, RankTestMethodRequest as S, SignTestAlternative as T, mcnemarPower as U, mannWhitneyU as V, mcnemarRequiredN as W, pairedRiskDifference as X, pairedMde as Y, pairedSignTest as Z, PairedBootstrapResult as _, CorpusAgreementReport as a, spearmanR as at, ProportionInterval as b, EProcess as c, wilcoxonSignedRank as ct, EProcessStep as d, passAtK as et, MANN_WHITNEY_EXACT_MAX_STATES as f, PairedBootstrapOptions as g, McNemarResult as h, CorpusAgreementPerDimension as i, requiredSampleSize as it, bonferroni as j, WilcoxonSignedRankResult as k, EProcessOptions as l, wilson as lt, MannWhitneyResult as m, CliffsMagnitude as n, ranks as nt, CorpusScoreRecord as o, weightedComposite as ot, MANN_WHITNEY_EXACT_MAX_WORK as p, pairedBootstrap as q, CorpusAgreementOptions as r, requiredPairedSampleSize as rt, DEFAULT_PERMUTATIONS as s, weightedMean as st, BOOTSTRAP_GATE_MIN_N as t, pearsonR as tt, EProcessState as u, PairedSignTestResult as v, RiskDifferenceResult as w, RankTestMethod as x, PairedTTestResult as y, interRaterReliability as z };
642
- //# sourceMappingURL=statistics-D_4Snl-5.d.ts.map
967
+ export { pairedCohensDz as $, WeightedCompositeInput as A, positionalBias as At, eProcess as B, RankTestMethod as C, GoldenItem as Ct, ScoreRiskDifferenceResult as D, calibrateJudge as Dt, RiskDifferenceResult as E, VerbosityBiasResult as Et, cliffsDelta as F, mannWhitneyU as G, interRaterReliability as H, cohensD as I, mcnemarRequiredN as J, mcnemar as K, confidenceInterval as L, WilcoxonSignedRankResult as M, verbosityBias as Mt, benjaminiHochberg as N, SignTestAlternative as O, calibrateJudgeContinuous as Ot, bonferroni as P, pairedBootstrap as Q, corpusInterRaterAgreement as R, ProportionInterval as S, ContinuousCalibrationResult as St, RankTestOptions as T, SelfPreferenceResult as Tt, interpretCliffs as U, holm as V, isBinaryOutcomeVector as W, normalizeScores as X, mulberry32 as Y, pairedBinaryScale as Z, McNemarResult as _, wilson as _t, CorpusAgreementReport as a, pairedSignTest as at, PairedSignTestResult as b, ContinuousAgreement as bt, DEFAULT_PERMUTATIONS as c, passAtK as ct, EProcessState as d, requiredPairedSampleSize as dt, pairedDeltaTieFraction as et, EProcessStep as f, requiredSampleSize as ft, MannWhitneyResult as g, wilcoxonSignedRank as gt, MANN_WHITNEY_EXACT_MAX_WORK as h, weightedMean as ht, CorpusAgreementPerDimension as i, pairedRiskDifferenceScore as it, WeightedCompositeResult as j, selfPreference as jt, WILCOXON_EXACT_MAX_N as k, continuousAgreement as kt, EProcess as l, pearsonR as lt, MANN_WHITNEY_EXACT_MAX_STATES as m, weightedComposite as mt, CliffsMagnitude as n, pairedRiskDifference as nt, CorpusScoreRecord as o, pairedTTest as ot, ExactRiskDifferenceResult as p, spearmanR as pt, mcnemarPower as q, CorpusAgreementOptions as r, pairedRiskDifferenceExact as rt, DECISION_PAIRED_DELTA_STATISTIC as s, partialCredit as st, BOOTSTRAP_GATE_MIN_N as t, pairedMde as tt, EProcessOptions as u, ranks as ut, PairedBootstrapOptions as v, CalibrationResult as vt, RankTestMethodRequest as w, PositionalBiasResult as wt, PairedTTestResult as x, ContinuousAgreementOptions as xt, PairedBootstrapResult as y, CandidateScore as yt, corpusInterRaterAgreementFromJudgeScores as z };
968
+ //# sourceMappingURL=statistics-CKOqre5S.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"statistics-CKOqre5S.d.ts","names":[],"sources":["../src/judge-calibration.ts","../src/statistics.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;UAuBiB;EACf;EACA;;EAEA;;UAGe;EACf;EACA;;EAEA;;UAGe;EACf;EACA;;EAEA;;EAEA;;EAEA,YAAY;IAAQ;IAAgB;IAAe;IAAe;;;;;;iBAMpD,eACd,QAAQ,cACR,WAAW,mBACV;UA0Bc;;;;;EAKf;EACA;;;;;;iBAOc,eAAe,QAAQ,mBAAmB;UAgBzC;;EAEf;EACA;;iBAGc,cACd,SAAS;EAAQ;EAAmB;KACnC;UAYc;;EAEf;EACA;EACA;EACA;;;;;;;iBAQc,eACd,SAAS;EAAQ;EAAe;KAC/B;UA8Ec;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;IACE;IACA;;;EAGF;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA;;EAEA;;;;;;;;;iBAUc,oBACd,oBACA,OAAM,6BACL;UAqEc,oCAAoC;;EAEnD;;EAEA;EACA;EACA;IACE;IACA;;;;;;;iBAQY,yBACd,QAAQ,cACR,WAAW,kBACX,OAAM,6BACL;;;;;cCnVU,kBAAe,QAAY,iBAAe;;iBAGvC,aAAa;EAAU;EAAe;;;;;;;;;;iBAoBtC,mBACd,kBACA,qBACA;EAAQ;EAAe;;EACpB;EAAc;EAAe;;;;;;;;;;;;;;;;;;;iBAiDlB,sBAAsB,aAAa;;KAkFvC;;;;;KAMA;UAEK;;EAEf,SAAS;;EAET;;;EAGA;;;cAIW;;cAEA;;cAEA;;cAEA;UAEI;;EAEf;;EAEA;;EAEA;;EAEA,QAAQ;;EAER;;;;;;;;;;;;;;iBAec,aACd,aACA,aACA,OAAM,kBACL;;iBAwFa,cAAc,iBAAiB;UAK9B;;EAEf;EACA;;EAEA;;;;;;;;;;;;;;;;iBAiBc,YAAY,kBAAkB,kBAAkB;UAyB/C;;;EAGf;;EAEA;;EAEA,QAAQ;;EAER;;EAEA;;;;;;;;;;;;;;;iBAgBc,mBACd,kBACA,iBACA,OAAM,kBACL;;;;;;;;;;;;;iBAmFa,QAAQ,aAAa;;;;;;;;;iBAqBrB,eAAe,kBAAkB;KAoBrC;;;;;;;;;;;iBAYI,YAAY,kBAAkB;;;;;;iBAkB9B,gBAAgB,gBAAgB;;;;;iBAuBhC,MAAM;;;;;;iBAmBN,SAAS,aAAa;;;;;iBAuBtB,UAAU,aAAa;UAKtB;;EAEf,MAAM;;;;;EAKN,SAAS;;EAET;;UAGe;EACf;EACA;;;;;;;;;;;iBAYc,kBAAkB,OAAO,yBAAyB;UA4CjD;;EAEf;;EAEA;;EAEA;;EAEA;;UAGe,oCAAoC;EACnD;;EAEA;;EAEA;;UAGe;;EAEf,cAAc;;EAEd;;EAEA;;EAEA;;EAEA;;UAGe,+BAA+B;;;;;;EAM9C;;;;;;EAMA;;;;;;;;;;;;;;;;;;iBAmBc,0BACd,SAAS,qBACT,OAAM,yBACL;;;;;;;;;iBA0Ha,yCACd,aAAa;EAAQ;EAAgB,QAAQ;IAC7C,OAAM,yBACL;;;;;;;iBA8Ba,mBAAmB;EACjC;EACA;EACA;EACA;;;;;;;;;;;iBAsBc,yBAAyB;EACvC;EACA;EACA;EACA;;;;;;;;iBAkBc,UAAU;EACxB;EACA;EACA;EACA;;;;;;;;;;;;;;;;iBAyBc,iBAAiB;EAC/B;EACA;EACA;EACA;EACA;;;;;;;iBAyBc,aAAa;EAC3B;EACA;EACA;EACA;EACA;;;;;;;;;iBAwBc,WACd,4BACA;EACG;EAAoB;;;;;;;;;;iBAgBT,KACd,4BACA;EACG;EAAoB;;;;;;;;;iBA4BT,kBACd,4BACA;EACG;EAAmB;;UAuCP;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;cAaW;UAEI;;EAEf;;EAEA;;EAEA;;;EAGA;;;;;;;;;;;;iBAac,gBACd,kBACA,iBACA,OAAM,yBACL;;KA0DS;;UAGK;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA,aAAa;;EAEb;;;;;;;;;;;;;;;iBAgBc,eACd,gCACA,aAAa,sBACZ;;UAgDc;;EAEf;;EAEA;;EAEA;;;;;;;;;iBAUc,OAAO,mBAAmB,WAAW,sBAAoB;;;;;;;;;;;;;;;;;;;;;;;;;;iBA2CzD,sBAAsB,QAAQ;;UAU7B;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;iBAec,QACd,SAAS,6BACT,WAAW,8BACV;;UAmBc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;iBAkBc,qBACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;UAkCc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA+Bc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;UAyEc;;EAEf;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;EAEA;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8Dc,0BACd,SAAS,6BACT,WAAW,6BACX,sBACC;;;;;;;;;;;;;;;;;;;;iBA6Ea,kBACd,QAAQ,mBACR,OAAO;;;iBAkBO,uBACd,QAAQ,mBACR,OAAO;;;;;;;;;;;;;;;;;;;;;;;;;;;;;cA4CI;;;;;;;;;;;iBAYG,QAAQ,WAAW,WAAW;UAgE7B;;;EAGf;;;;EAIA;;;;EAIA;;UAGe;;EAEf;;EAEA;;EAEA;;UAGe,sBAAsB;EACrC;EACA;EACA;;EAEA;;EAEA;;UAGe;;;EAGf,OAAO,YAAY;EACnB,SAAS;;;;;;;;;;;;;;;;;;;;;;;;;;;;;iBA8BK,SAAS,OAAM,kBAAuB;;;;;iBAsatC,WAAW"}