@tangle-network/agent-eval 0.133.2 → 0.133.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. package/CHANGELOG.md +156 -0
  2. package/dist/{analyze-runs-BmX-h_yn.d.ts → analyze-runs-BClW9OSe.d.ts} +3 -3
  3. package/dist/{analyze-runs-BmX-h_yn.d.ts.map → analyze-runs-BClW9OSe.d.ts.map} +1 -1
  4. package/dist/{analyze-runs-B-afTpCv.js → analyze-runs-qk8op0tN.js} +63 -42
  5. package/dist/analyze-runs-qk8op0tN.js.map +1 -0
  6. package/dist/baseline-BaPxoROc.js +149 -0
  7. package/dist/baseline-BaPxoROc.js.map +1 -0
  8. package/dist/{baseline-hG3K85h4.d.ts → baseline-D_fT6277.d.ts} +43 -11
  9. package/dist/baseline-D_fT6277.d.ts.map +1 -0
  10. package/dist/benchmarks/index.d.ts +1 -1
  11. package/dist/benchmarks/index.js +1 -1
  12. package/dist/{benchmarks-CJr1H1_a.js → benchmarks-BP9sgMia.js} +3 -3
  13. package/dist/{benchmarks-CJr1H1_a.js.map → benchmarks-BP9sgMia.js.map} +1 -1
  14. package/dist/builder-eval/index.js +1 -1
  15. package/dist/campaign/index.d.ts +2 -2
  16. package/dist/campaign/index.js +2 -2
  17. package/dist/{campaign-BJjn1rhw.js → campaign--V4ffEKR.js} +12 -6
  18. package/dist/{campaign-BJjn1rhw.js.map → campaign--V4ffEKR.js.map} +1 -1
  19. package/dist/{client-COvaLoQG.d.ts → client-Du7B81wW.d.ts} +28 -14
  20. package/dist/client-Du7B81wW.d.ts.map +1 -0
  21. package/dist/{client-CYzbdJOZ.js → client-LIuo-KPv.js} +19 -7
  22. package/dist/client-LIuo-KPv.js.map +1 -0
  23. package/dist/contract/index.d.ts +3 -3
  24. package/dist/contract/index.d.ts.map +1 -1
  25. package/dist/contract/index.js +9 -8
  26. package/dist/contract/index.js.map +1 -1
  27. package/dist/{eval-campaign-DXhpZghy.js → eval-campaign-CvPcvqXC.js} +2 -2
  28. package/dist/{eval-campaign-DXhpZghy.js.map → eval-campaign-CvPcvqXC.js.map} +1 -1
  29. package/dist/hosted/index.d.ts +1 -1
  30. package/dist/hosted/index.d.ts.map +1 -1
  31. package/dist/hosted/index.js +1 -1
  32. package/dist/{index-BREtv3ZZ.d.ts → index-B5MNN1f1.d.ts} +3 -3
  33. package/dist/{index-BREtv3ZZ.d.ts.map → index-B5MNN1f1.d.ts.map} +1 -1
  34. package/dist/{index-C7Wue8R6.d.ts → index-DOqvIJ8I.d.ts} +27 -10
  35. package/dist/index-DOqvIJ8I.d.ts.map +1 -0
  36. package/dist/{index-DSC51roc.d.ts → index-DSC51roc2.d.ts} +1 -1
  37. package/dist/index-DSC51roc2.d.ts.map +1 -0
  38. package/dist/{index-nhIYz9hn.d.ts → index-DuhJaaiH.d.ts} +68 -7
  39. package/dist/index-DuhJaaiH.d.ts.map +1 -0
  40. package/dist/index.d.ts +56 -10
  41. package/dist/index.d.ts.map +1 -1
  42. package/dist/index.js +29 -20
  43. package/dist/index.js.map +1 -1
  44. package/dist/ledger-core/index.d.ts +2 -2
  45. package/dist/ledger-core/index.js +2 -2
  46. package/dist/{ledger-core-CPZfcrC2.js → ledger-core-DAKFKRzi.js} +136 -18
  47. package/dist/ledger-core-DAKFKRzi.js.map +1 -0
  48. package/dist/matrix/index.d.ts +1 -1
  49. package/dist/meta-eval/index.d.ts +1 -1
  50. package/dist/meta-eval/index.js +2 -2
  51. package/dist/multishot/index.d.ts +1 -1
  52. package/dist/openapi.json +1 -1
  53. package/dist/{paired-arms-6XItKzd1.js → paired-arms-CA_8pN01.js} +2 -2
  54. package/dist/{paired-arms-6XItKzd1.js.map → paired-arms-CA_8pN01.js.map} +1 -1
  55. package/dist/pipelines/index.d.ts +1 -1
  56. package/dist/pipelines/index.js +3 -2
  57. package/dist/pipelines/index.js.map +1 -1
  58. package/dist/{release-report-wuilQkvK.js → release-report-BVZBmRZp.js} +2 -2
  59. package/dist/{release-report-wuilQkvK.js.map → release-report-BVZBmRZp.js.map} +1 -1
  60. package/dist/{release-report-CjHWa8Ia.d.ts → release-report-DKBtegGt.d.ts} +2 -2
  61. package/dist/{release-report-CjHWa8Ia.d.ts.map → release-report-DKBtegGt.d.ts.map} +1 -1
  62. package/dist/reporting.d.ts +3 -3
  63. package/dist/reporting.js +4 -4
  64. package/dist/{researcher-CbSKhK8z.d.ts → researcher-BtD5U1Up.d.ts} +2 -2
  65. package/dist/{researcher-CbSKhK8z.d.ts.map → researcher-BtD5U1Up.d.ts.map} +1 -1
  66. package/dist/{reward-hacking-Dl2UBzej.js → reward-hacking-DCdRK9TY.js} +2 -2
  67. package/dist/{reward-hacking-Dl2UBzej.js.map → reward-hacking-DCdRK9TY.js.map} +1 -1
  68. package/dist/rl.d.ts +1 -1
  69. package/dist/rl.js +4 -4
  70. package/dist/{rubric-predictive-validity-QG7ydk0s.js → rubric-predictive-validity-D6Q6n9oq.js} +2 -2
  71. package/dist/{rubric-predictive-validity-QG7ydk0s.js.map → rubric-predictive-validity-D6Q6n9oq.js.map} +1 -1
  72. package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts → skillopt-optimization-method-Dxr8pdZd.d.ts} +12 -7
  73. package/dist/{skillopt-optimization-method-wHF5xsUv.d.ts.map → skillopt-optimization-method-Dxr8pdZd.d.ts.map} +1 -1
  74. package/dist/{skillopt-optimization-method-CF6a327Q.js → skillopt-optimization-method-vvJ4bMNI.js} +119 -24
  75. package/dist/skillopt-optimization-method-vvJ4bMNI.js.map +1 -0
  76. package/dist/{statistics-DbvkkDPa.d.ts → statistics-D_4Snl-5.d.ts} +158 -30
  77. package/dist/statistics-D_4Snl-5.d.ts.map +1 -0
  78. package/dist/{statistics-DWM_AyLe.js → statistics-RwRNu2__.js} +546 -98
  79. package/dist/statistics-RwRNu2__.js.map +1 -0
  80. package/dist/{summary-report-Ci17nIdU.js → summary-report-BxtossFi.js} +3 -3
  81. package/dist/{summary-report-Ci17nIdU.js.map → summary-report-BxtossFi.js.map} +1 -1
  82. package/dist/{summary-report-CFnQgNfg.d.ts → summary-report-DyOhItws.d.ts} +3 -2
  83. package/dist/summary-report-DyOhItws.d.ts.map +1 -0
  84. package/dist/{baseline-DcX5hQDv.js → tool-use-metrics-DEGMKycK.js} +2 -114
  85. package/dist/tool-use-metrics-DEGMKycK.js.map +1 -0
  86. package/docs/design/statistics-decisions.md +271 -0
  87. package/docs/design.md +1 -0
  88. package/docs/insight-report.md +1 -1
  89. package/docs/research-report-methodology.md +4 -1
  90. package/package.json +2 -1
  91. package/dist/analyze-runs-B-afTpCv.js.map +0 -1
  92. package/dist/baseline-DcX5hQDv.js.map +0 -1
  93. package/dist/baseline-hG3K85h4.d.ts.map +0 -1
  94. package/dist/client-COvaLoQG.d.ts.map +0 -1
  95. package/dist/client-CYzbdJOZ.js.map +0 -1
  96. package/dist/index-C7Wue8R6.d.ts.map +0 -1
  97. package/dist/index-DSC51roc.d.ts.map +0 -1
  98. package/dist/index-nhIYz9hn.d.ts.map +0 -1
  99. package/dist/ledger-core-CPZfcrC2.js.map +0 -1
  100. package/dist/skillopt-optimization-method-CF6a327Q.js.map +0 -1
  101. package/dist/statistics-DWM_AyLe.js.map +0 -1
  102. package/dist/statistics-DbvkkDPa.d.ts.map +0 -1
  103. package/dist/summary-report-CFnQgNfg.d.ts.map +0 -1
@@ -0,0 +1,271 @@
1
+ # Statistics: adoption decisions
2
+
3
+ This is the standing decision record for `src/statistics.ts` and the estimators layered on it.
4
+ It answers three questions: which statistic is trustworthy, where each number comes from, and what a caller is allowed to ask for at 3–10 repetitions per arm.
5
+ The API itself is documented in [`concepts.md`](../concepts.md); this file records *why* each statistic is kept, fixed, or refused.
6
+
7
+ ## Verdict
8
+
9
+ Keep the statistics in-house, fix them here, and take no statistics package as a runtime dependency.
10
+
11
+ Of the 38 exported statistics, 17 were correct, 15 were defective, and 6 are correct mathematics applied in a regime where they mislead.
12
+ The defects concentrated in exactly the functions a promotion gate reads: the paired rank tests, the bootstrap interval, and the multiple-comparison boundary.
13
+ All 15 are repaired; the 6 regime limits are now declared in the result rather than left for a caller to infer.
14
+ The largest single defect is a standard-normal CDF that mixed the arguments of the Abramowitz–Stegun error-function approximation, giving up to `3.7189e-2` absolute CDF error at `x = 0.567` where a correct implementation is bounded by `7.5e-8`.
15
+ That one line made every affected p-value too small by 26–36 % relative, so the module's real type-I error rate was 6.53 % at a nominal 5 %.
16
+
17
+ The reason to stay in-house is not preference.
18
+ Every statistic a JavaScript library gets right is one this module already gets right, and every statistic this module gets wrong at 3–10 reps is one no JavaScript library gets right either.
19
+ A survey of 13 installed packages, tested by execution rather than documentation, found no npm package that computes an exact rank test under ties, Cliff's delta, or a paired bootstrap interval.
20
+
21
+ ## How the numbers below were produced
22
+
23
+ Every measurement in this document was produced by executing the code, not by reading it.
24
+
25
+ The pristine baseline is `origin/main` blob `md5 15ca1ed9fbdd10215b5d0b682d99946c`, extracted with `git archive` and bundled with `esbuild` so that concurrent edits to the working tree could not move it.
26
+ The comparison build is the same bundle taken from the working tree.
27
+ References are `scipy 1.13.1` (`stats.norm`, `stats.t`, `stats.mannwhitneyu`, `stats.wilcoxon`, `stats.binomtest`, `stats.pearsonr`, `stats.spearmanr`) and, for the exact conditional null, an independent enumeration of every split of the observed data that agreed with scipy's permutation test to four decimals on 9 of 9 two-sample cases.
28
+ Monte Carlo figures are 4,000 trials per cell with a seeded generator unless stated otherwise.
29
+
30
+ Where a number in this document differs from one recorded elsewhere, the difference is almost always which build was measured.
31
+ The same call `mannWhitneyU([1,2,3],[4,5,6]).p` returns `0.03769147` on `origin/main` and `0.04953448` once the CDF is repaired; both are correctly reported values of different code.
32
+
33
+ ## Decision key
34
+
35
+ - **keep-and-fix** — the statistic belongs in this package, the implementation is wrong or incomplete, and the fix is ours to write.
36
+ - **keep-with-CI-oracle** — the implementation is correct and stays as-is, pinned against a scipy-generated golden fixture so it cannot silently regress.
37
+ - **replace-with-library** — a maintained package does this better than we do.
38
+ - **delete** — the statistic should not be callable.
39
+
40
+ No statistic in this module is marked **replace-with-library**.
41
+ That is the survey's conclusion, not an oversight, and the argument is in [Dependency verdict](#dependency-verdict).
42
+
43
+ ## Defective
44
+
45
+ These 15 produced a wrong number, hung, or fabricated a verdict.
46
+ All are repaired, each with a regression test that fails on the pristine blob and passes after.
47
+
48
+ | Statistic | Status | Decision | Measured evidence | Regime caveat |
49
+ | --- | --- | --- | --- | --- |
50
+ | `normalCdf` | defective, **fixed** | keep-and-fix | `t = 1/(1+p·|x|)` used the unscaled argument while the exponential used `exp(-x²/2) = exp(-z²)` with `z = x/√2`; max abs CDF error `3.7189e-2` at `x = 0.567`. Repaired to evaluate both halves at `z = |x|/√2`; worst deviation from `scipy.stats.norm.cdf` now `4e-10` at `x = 1.6` and `3.4e-8` at `x = 0.565`, inside the A&S 7.1.26 bound of `7.5e-8`. | None once fixed. |
51
+ | `normalCdf` duplicate in `src/baseline.ts` | defective, **fixed** | keep-and-fix | A byte-identical second copy of the same defect reached production verdicts through `welchsTTest → studentTCdf`. Deleted with its duplicated `studentTCdf`/`incompleteBeta`/`lnGamma`; the shared math now lives in `src/math/normal.ts`, `src/math/student-t.ts` and `src/math/special-functions.ts`. A `df = 298` case moved from `p = 0.010887` to the true Student-t `0.015150`. | Now one implementation repo-wide: `grep 0.3275911` returns a single site. |
52
+ | `mcnemarPower` | defective, **fixed** | keep-and-fix | Reached the normal distribution through the broken forward CDF while its documented inverse `mcnemarRequiredN` reached it through `zQuantile`, so the two were not inverses. `mcnemarPower({p10:0.2, p01:0.1, nPairs:200})` returned `0.773437` against a true `0.736522`; it now returns `0.736522`. | Power was **overstated** above 0.5 and understated below it, so pre-registered N read off this function was too small. |
53
+ | `studentTCdf` `df > 100` branch | defective, **deleted** | delete | The branch short-circuited to `normalCdf`, which changed a two-sided `df = 102, t = 1.98` result from the true `0.050398` to `0.047703` and flipped a 5% decision. | Every finite degree of freedom now uses the regularized incomplete beta. |
54
+ | `studentTCdf` / `incompleteBeta` at `df ≤ 100` | defective, **fixed** | keep-and-fix | The continued fraction omits the mandatory symmetry branch (`x < (a+1)/(a+b+2)`, else `1 − I(1−x, b, a)`), so it is evaluated outside its convergence domain for small `|t|` and collapses as `x → 1`. `studentTCdf(0.005, 100)` returns `0.89152130` against a true `0.50198972`, an error of `0.3895`. Through `pairedTTest` at `df = 7`: `t = 0.001` reports `p = 0.15130173` against a true `0.99923002`, and `t = 1e-6` reports `p = 0.00015251`. There is a discontinuity at zero — `t = 1e-8` returns exactly `p = 1.0` because `x ≥ 1` short-circuits. | A perfectly null paired result reports `p < 0.05`. The materially wrong band is `|t| ≲ 0.02` at `df = 2–10`, widening to `|t| ≲ 0.058` near `df = 100`; outside it `pairedTTest` matched `scipy.stats.ttest_rel` to `≤ 1e-6` relative across 166 cases. |
55
+ | `interRaterReliability` | defective, **fixed** | keep-and-fix | The grouping loop iterates judge-major and opens a new item bucket whenever the last holds `judgeScores.length` entries, so each bucket collects consecutive scores from the *same* judge. It measures within-judge spread, not between-judge agreement, and is anti-correlated with the truth. Two identical judges scoring `[0,100]` return **`−0.500000`** where the true α is `+1.0`; two maximally disagreeing judges (`[0,0]` versus `[100,100]`) return **`+1.000000`** where the true α is `−0.5`. Perfect agreement on three items returns `−0.250000`, on four items `+0.650000`. | The result is also unstable in the item count, because bucket boundaries depend on how the item count divides by the judge count. Zero test coverage repo-wide; consumed by `src/pipelines/judge-agreement.ts:73`. |
56
+ | `mannWhitneyU` (NaN input) | defective, **fixed** | keep-and-fix | The tie-grouping loop advances via `while (j < len && combined[j].v === combined[i].v) j++` then `i = j`. `NaN === NaN` is false, so `j` never leaves `i` and the process spins forever. `mannWhitneyU([1,2,3,4,5,6],[NaN,2,3,4,5,6])` did not return within 8 s and blocked the event loop hard enough that a 4 s `setTimeout` never fired. | A single NaN score hangs a campaign rather than failing it. `ranks` is safe only because it uses `i = j + 1`. |
57
+ | `wilcoxonSignedRank` (NaN input) | defective, **fixed** | keep-and-fix | The same construct at the absolute-rank grouping loop, same hang. | Reachable from `held-out-gate.ts:278`, so a NaN in a holdout set hangs the gate. |
58
+ | `mannWhitneyU` (tie and continuity correction) | defective, **fixed** | keep-and-fix | The variance uses the no-ties formula `n₁n₂(n₁+n₂+1)/12` and there is no continuity correction. `[1,2,3]` versus `[4,5,6]` (zero ties) and `[0,0,0]` versus `[1,1,1]` (maximal ties) both return the byte-identical `p = 0.04953448`. scipy's tie-and-continuity-corrected asymptotic separates them at `0.080856` and `0.046854`. | Both omissions push p downward, so they compound the CDF defect rather than offsetting it. `u` is returned as `min(u1,u2)`, which discards the direction of the effect. |
59
+ | `wilcoxonSignedRank` (tie and continuity correction) | defective, **fixed** | keep-and-fix | The variance uses `n(n+1)(2n+1)/24` with no tie term `−(1/48)Σ(t³−t)` and no continuity correction, despite the function computing average ranks for ties. | Same compounding direction. The statistic convention also differs from scipy without being documented: this returns `W⁺`, scipy returns `min(W⁺, W⁻)`. |
60
+ | `wilcoxonSignedRank` (`n < 6` branch) | defective, **fixed** | **branch deleted** | Line 219 hard-returns `{w: 0, p: 1}` below six non-zero differences, with no exception, no flag, and nothing in the result to distinguish it from a measured null. A clean `+0.5` on all five of five pairs returns `p = 1` where the exact answer is `0.0625`. Because zero deltas are dropped first, ties silently push `n` under the threshold: ten pairs with five zero deltas also returned `p = 1`. | This is the single most consequential open defect at 3–10 reps, and it violates the package's own *no fallbacks, fail loud* rule. It is a false-negative generator precisely in the target regime. |
61
+ | `bonferroni` | defective, **fixed** | keep-and-fix | Adjusted values are correct (`min(1, p·k)` matched R `p.adjust` on 8 case sets), but the rejection boundary uses strict `<` where the rule is `p ≤ α/k`. With `p = [0.0125]×4` and `α = 0.05` it returns `significant = [false,false,false,false]`; `holm` on the identical input returns `[true,true,true,true]`. It also has no input validation, unlike `holm`: `bonferroni([-0.1, 0.2], 0.05)` returns `{adjusted: [-0.2, 0.4], significant: [true, false]}` — a negative p-value declared significant. | Two corrections in one module disagree at their shared boundary. |
62
+ | `benjaminiHochberg` | defective, **fixed** | keep-and-fix | q-values are correct (matched R `p.adjust('BH')` on 8 case sets including the 15-value worked example and ties), same two rule defects. `benjaminiHochberg([0.05, 0.05], 0.05)` returns `significant = [false, false]` where BH rejects iff `q ≤ α`. `benjaminiHochberg([-0.1, 0.2])` returns `qValues = [-0.2, 0.2], significant = [true, false]`. | Drives `rl/contamination.ts:162` and `summary-report.ts:159`. |
63
+ | `pairedTTest` (zero-variance branch) | defective, **fixed** | keep-and-fix | Line 199 returns `p = 0` when the standard error is zero. `pairedTTest([0, 0, 0.5], [0.5, 0.5, 1])` — a constant `+0.5` shift on three pairs — returns `{t: null, df: 2, p: 0}`. That is absolute certainty from three observations where the exact signed-rank floor at `n = 3` is `0.25`, and the object is internally inconsistent (`t` serialized as `null` alongside `p = 0`). | `pairedCohensDz` handles the identical condition correctly by returning `null` and documents why. The module should answer the degenerate case once, the same way, everywhere. |
64
+ | `cohensD` (degenerate branches) | defective, **fixed** | keep-and-fix | Returns a silent `0` twice: when either sample has fewer than two observations, and when the pooled standard deviation is zero. `cohensD([1,1,1],[2,2,2])` returns `0` — "no effect" for a maximal, zero-variance separation. | Third distinct answer to the same degenerate condition, after `pairedTTest`'s `p = 0` and `pairedCohensDz`'s `null`. Only `pairedCohensDz` is right. |
65
+ | `mulberry32` | defective, **fixed** | keep-and-fix | `let s = seed \| 0 \|\| 0x9e3779b9` collapses seed `0` to the golden-ratio constant. `mulberry32(0)` and `mulberry32(0x9e3779b9 \| 0)` both emit `[0.3588899802, 0.1059032613, 0.6752904793]`. | Seed `0` is a common default, so two runs a caller believes are independent replicates are the same run. |
66
+ | `makeRng` / bootstrap seeding | defective, **fixed** | keep-and-fix | `makeRng` returns raw `Math.random` when `opts.seed` is undefined. Two back-to-back `confidenceInterval` calls on identical input returned `lower = 0.30000000000000004` and `lower = 0.3`, `upper = 0.6749999999999999` and `upper = 0.6499999999999999`. Seeding works when supplied (`seed: 7` reproduced exactly). | This contradicts `mulberry32`'s own docstring, which states that a seed is required because unseeded randomness in gate verdicts is non-reproducible by construction. `src/contract/analyze-runs.ts:908` is the live call site that passes no seed; the gates in `held-out-gate.ts`, `promotion-policy.ts`, `statistical-heldout.ts`, `measured-comparison.ts` and `summary-report.ts` all thread one through. |
67
+
68
+ ## Correct mathematics, misleading regime
69
+
70
+ These 5 compute what they claim.
71
+ They mislead at 3–10 repetitions because the asymptotic assumption does not hold there, and no change to the implementation fixes that.
72
+
73
+ | Statistic | Status | Decision | Measured evidence | Regime caveat |
74
+ | --- | --- | --- | --- | --- |
75
+ | `mannWhitneyU` small-n | approximation limit, **exact path added** | keep-and-fix (add exact path) | The normal approximation was applied unconditionally, including `n = 1`, with no switchover threshold. At three per group the exact minimum attainable two-sided p is `0.100000`, yet complete separation returned `p = 0.04953448`. The default now selects exact computation from the dynamic program's actual state and work, so balanced 12+12 and imbalanced 1+24 designs are both exact. Above that limit, the automatic permutation seed is invariant to observation order and group order; the prior seed changed a two-sided result from `0.04960` to `0.04420` when the groups were swapped. | Repairing the CDF did not fix this; only an exact test does. Discreteness is the binding constraint, and tie-aware `pFloor` now reports it on every result. |
76
+ | `wilcoxonSignedRank` small-n | approximation limit, **exact path added** | keep-and-fix (add exact path) | Where the approximation ran it was anti-conservative at `n = 6–7` (minimum attainable `p = 0.0209` against an exact `0.0312` at `n = 6`) and conservative from `n ≥ 8`. The default is now exact by sign-flip enumeration at `n ≤ 20`. | The `n < 6` hard return is gone. `pFloor = 2^(1−n)` is reported on every result, so a design that cannot reach alpha says so. |
77
+ | `confidenceInterval` / `pairedBootstrap` | approximation limit, **floor declared** | keep-and-fix (add an n floor) | Percentile bootstrap, correctly constructed. The gate-relevant quantity is `P(low > 0)` under a true null against a nominal 2.5 %. Measured over 4,000 seeded trials on a continuous null: **13.53 % at `n = 3`**, 3.33 % at `n = 5`, 4.45 % at `n = 8`, 3.52 % at `n = 10`, 3.10 % at `n = 20` for the median statistic; 13.85 %, 7.95 %, 5.80 %, 4.90 %, 3.80 % for the mean statistic. On a five-value discrete grid resembling judge scores: 6.02 % at `n = 3` (median), 6.68 % (mean), still 3.43 % at `n = 10` (mean). | This is an intrinsic limit of bootstrapping three points, not an implementation error — scipy's BCa on the same `n = 3` data gives 16.0 %. The defect is the docstring, which states that `low > threshold` means the gain is real at the confidence level. At `n = 3` that claim is wrong by more than 5×. Consumed by `promotion-policy.ts:133`, `held-out-gate.ts:272`, `statistical-heldout.ts:174`, `measured-comparison.ts:995`, `analyze-runs.ts:908`. `pairedBootstrap` now returns `gateEligible`, false below `BOOTSTRAP_GATE_MIN_N = 20`. |
78
+ | `pairedRiskDifference` | approximation limit | keep-and-fix | The point estimate and the variance formula both matched the closed form exactly on 6 configurations, but the interval is Wald: empirical coverage against a nominal 95 % is 73.80 % at `n = 5`, 86.48 % at `n = 10`, 93.80 % at `n = 20`, 94.73 % at `n = 200`. With no discordant pairs the variance is exactly zero, so `pairedRiskDifference([0,0,0],[1,1,1])` returns `riskDifference = 1, lower = 1, upper = 1` — a zero-width 95 % interval asserting certainty from three observations, and ten concordant pairs return `[0, 0]`. | The module already ships `wilson` and its own comment block argues against exactly this Wald approach for proportions. A Wilson-style or Tango score interval is the standard fix. |
79
+ | `requiredSampleSize`, `requiredPairedSampleSize`, `pairedMde` | approximation limit, **documented** | keep-with-CI-oracle | All three match their stated normal-approximation formulas exactly against `scipy.stats.norm.ppf` closed forms, and are numerically unchanged by the CDF fix because they route through `zQuantile` rather than the forward CDF (`requiredSampleSize({effect: 0.5}) = 63` before and after). They use normal quantiles with no t correction, so they understate required n where n is small: `requiredPairedSampleSize({effect: 0.5})` returns **32** against an exact t-based **34**, and `{effect: 0.8}` returns **13** against **15**. | A 6–13 % shortfall, precisely in the range a caller consults to decide whether 3–10 reps suffice. Both docstrings now say "treat as a lower bound". |
80
+
81
+ ## Correct
82
+
83
+ These 17 matched their references with zero mismatches and stay as they are, pinned against a golden fixture so they cannot regress silently.
84
+
85
+ | Statistic | Decision | Measured evidence |
86
+ | --- | --- | --- |
87
+ | `pairedTTest` (normal range) | keep-with-CI-oracle | Across 166 cases the t statistic matched `scipy.stats.ttest_rel` to `≤ 1.5e-15` and p to `≤ 1e-6` relative in all but 4. Type-I under a true null over 20,000 replicates: 5.20 % at `n = 3`, 4.74 % at `n = 4`, 4.97 % at `n = 6`, 5.17 % at `n = 8`, 5.03 % at `n = 10` — correctly calibrated at this package's actual sample sizes. |
88
+ | `mcnemar` | keep-with-CI-oracle | Exact two-sided binomial on discordant pairs; matched `scipy.stats.binomtest` two-sided to `< 1e-9` on all 13 `(b,c)` configurations including `(0,0)`, `(10,0)`, `(100,70)`. A `b = 1, c = 5` case returns `pValue = 0.21875`. Log-space accumulation via `lnGamma` keeps it stable at large counts. |
89
+ | `pairedSignTest` | keep-with-CI-oracle | Exact one-sided binomial; matched `scipy.stats.binomtest(..., alternative='greater')` to `< 1e-12` on all 9 cases. `pairedSignTest([0.5,0.5,0.5], 'greater')` returns `0.125`. Ties are excluded from the denominator and still reported; `alternative` must be passed explicitly and an invalid value throws, which prevents post-hoc direction selection. |
90
+ | `wilson` | keep-with-CI-oracle | Matched the closed form to `< 1e-8` on all 13 `(successes, n)` pairs including `0/1`, `10/10`, `1/1000`, `0/0`. Correctly asymmetric at the boundaries: `wilson(0, 10)` returns `[0, 0.27753280]`. |
91
+ | `passAtK` | keep-with-CI-oracle | Chen et al. 2021 unbiased estimator, exhaustively verified for every `(n, c, k)` with `n = 1..6` against exact integer `math.comb`: zero mismatches. Stable at scale (`n=1000, c=3, k=100` agreed to `1.11e-16`). `passAtK(10, 3, 5) = 0.9166666667`. |
92
+ | `corpusInterRaterAgreement` | keep-with-CI-oracle | The ICC(2,1) it surfaces matched a hand-derived two-way random-effects ANOVA reference to `< 1e-9` on 5 matrices, including the inverted case at `−1.959459`. This is a genuinely different and correct computation from `interRaterReliability`: it pivots to a proper items × judges matrix and delegates to `continuousAgreement`. Its fail-loud contract behaves — empty input, fewer than two judges, fewer than two common items, duplicate records, and absent dimensions all throw `ValidationError`. |
93
+ | `eProcess` | keep-with-CI-oracle | Betting test-martingale. Empirical type-I over 4,000 sequences of 200 observations at `α = 0.05`: 2.40 % at the null boundary, 0.00 % in the null interior, 1.33 % on continuous uniform — all inside Ville's bound. Power at `E[x] = 0.7` is 99.775 %. The predictability invariant holds: the first update leaves wealth at exactly 1. |
94
+ | `holm` | keep-with-CI-oracle | Matched `statsmodels.multipletests` to `1e-9` on a 6-value reference vector, and uses `≤` at the boundary, which is the correct rule. It also validates both `alpha` and the p range, which `bonferroni` does not. |
95
+ | `zQuantile` | keep-with-CI-oracle | Acklam inverse-normal, structurally independent of `normalCdf`. This independence is why the sample-size functions were untouched by the CDF defect, and why `mcnemarPower`'s failure to invert `mcnemarRequiredN` was a valid detector of it. |
96
+ | `mcnemarRequiredN` | keep-with-CI-oracle | Matched the Lachin closed form exactly on all 4 parameter sets: `234`, `77`, `155`, `Infinity`. Unchanged by the CDF fix. Round-trip against the repaired `mcnemarPower` now holds across 16 configurations: power at `requiredN` meets the target with overshoot `≤ 0.0049`, and power at `requiredN − 1` is below target in all 16. |
97
+ | `ranks` | keep-with-CI-oracle | Correct average-rank-with-ties. `ranks([3,1,1,2])` returns `[4, 1.5, 1.5, 3]`, matching `scipy.stats.rankdata`. Safe against NaN input because it advances with `i = j + 1`. |
98
+ | `pearsonR` | keep-with-CI-oracle | Matched `scipy.stats.pearsonr` to 8 decimals (`0.99061012` on an 8-point case). |
99
+ | `spearmanR` | keep-with-CI-oracle | Matched `scipy.stats.spearmanr` to 8 decimals (`0.99402980` on the same case). |
100
+ | `cliffsDelta` | keep-with-CI-oracle | Matched a hand-computed `(#gt − #lt)/(n₁n₂)` reference exactly (`0.312500`). Note the orientation is *after over before*, the opposite sign to the textbook `δ` written over `(a, b)`; the docstring states this and the parameter names `(before, after)` carry it. |
101
+ | `pairedCohensDz` | keep-with-CI-oracle | Matched `mean(d)/sd(d)` exactly (`2.184070`). It is the only function in the module that handles the zero-variance degenerate case correctly, returning `null` with a docstring explaining that the standardized effect is undefined rather than an arbitrarily large finite number. Treat it as the reference behaviour the other degenerate branches should adopt. |
102
+ | `weightedMean`, `partialCredit`, `weightedComposite`, `interpretCliffs`, `normalizeScores` | keep-with-CI-oracle | Arithmetic and thresholding, verified by direct evaluation (`weightedMean` of `[{1,w2},{4,w1}]` is `2.000000`; `partialCredit(3,4)` is `0.750000`). |
103
+ | `welchsTTest`, `compareToBaseline` (`src/baseline.ts`) | keep-with-CI-oracle, **covered** | Correct, and previously unguarded: no test file imported either, which is the structural reason a duplicated broken CDF survived here. It is exported publicly and it gates improved / regressed / stable verdicts. Now pinned against `scipy.stats.ttest_ind(equal_var=False)` in the oracle fixture and exercised through `compareToBaseline` in `tests/statistics.test.ts`. |
104
+
105
+ ## Public surface change
106
+
107
+ `normalCdf` and `studentTCdf` were private on `origin/main` and are now exported with explicit accuracy contracts.
108
+ `baseline.ts` owns the single Welch implementation, and `contract/analyze-runs.ts` consumes that result instead of carrying another normal-approximation copy.
109
+ That is the right trade: one implementation with one accuracy contract beats three copies of which two were wrong.
110
+ It also means both functions are now public API and owe callers a stated accuracy bound.
111
+ `normalCdf` is documented at `7.5e-8` absolute.
112
+ `studentTCdf` is exact to the incomplete beta's own precision at every finite degree of freedom and matches `scipy.stats.t.cdf` to `1e-9` across the pinned oracle cases; the residual floor near `t = 0` is float64 cancellation in `x = df/(df + t²)`, not the approximation.
113
+
114
+ ## Dependency verdict
115
+
116
+ **Take no statistics package as a runtime dependency.**
117
+ **Use scipy as a CI oracle.**
118
+ **Implement the exact small-n rank tests here, because nothing else does.**
119
+
120
+ ### What we must implement ourselves
121
+
122
+ Four things the survey found no correct implementation of in npm, at any package, under ties:
123
+
124
+ | Needed | Library that does it correctly |
125
+ | --- | --- |
126
+ | Exact two-sample rank test under ties | none |
127
+ | Exact paired signed-rank test under ties | none |
128
+ | Cliff's delta | no package exists in npm at all |
129
+ | Paired bootstrap confidence interval | none |
130
+
131
+ Those four rows are the entire bottleneck, and they are the reason this decision is not close.
132
+ `lib-r-math.js` comes closest — it is the only source of the exact two-sample rank-sum null distribution in JavaScript, and it reproduces every exact floor we care about (`pwilcox(0, 3, 3) = 0.100000`, `psignrank` at `n = 5` gives `0.062500`, at `n = 8` gives `0.007813`) at a cost of 2 packages and 1.2 MB.
133
+ But its signature is `(q, m, n)` with no tie vector, so ties are structurally unrepresentable, exactly as in R where `wilcox.test` warns and falls back.
134
+ It ships distributions, not tests, so the test wrapper, the tie conditioning, and the effect sizes remain ours regardless.
135
+
136
+ The cost of writing them ourselves is small and was measured, not estimated.
137
+ Enumerating every split of the observed data takes 0.1 ms at 3 versus 3 (20 splits), **3.7 ms at 10 versus 10** (184,756 splits), and 33 ms at 12 versus 12 (2,704,156 splits).
138
+ Paired sign-flip enumeration is `2ⁿ`: 1,024 at `n = 10`, about 1 M at `n = 20`.
139
+ The entire stated 3–10 repetition regime runs exact in under 4 ms.
140
+ This is roughly 150 lines, not a research project.
141
+
142
+ ### What we must not take as a runtime dependency
143
+
144
+ The libraries that are correct are correct at things this module already gets right.
145
+
146
+ `@stdlib/stats-padjust` matches `statsmodels.multipletests` to `1e-9` on `holm`, `bh`, and `bonferroni`, and so does this module's own `holm`, `benjaminiHochberg`, and `bonferroni` on the same frozen reference vectors.
147
+ It was removed even as a development dependency because it bought zero additional coverage for 190 locked packages.
148
+ `@stdlib/stats-ranks` and `jstat.rank` both do average-rank-with-ties correctly, and so does `ranks` here.
149
+ `@sipemu/anofox-statistics`'s count-based exact tests are correct (`fisherExact([[3,0],[0,3]]) = 0.09999999999999992`, `binomTest(3,3,0.5) = 0.25000000000000006`, `mcnemarExact = 0.21875000000000008`), and so are `mcnemar` and `pairedSignTest` here.
150
+
151
+ Meanwhile the libraries that cover the missing statistics are wrong in this regime, sometimes worse than the incumbent:
152
+
153
+ - `@stdlib/stats-kruskal-test` matches `scipy.kruskal` to `1e-9` but is anti-conservative against the exact conditional truth in 7 of 8 cases: `p = 0.0495` against an exact `0.1000` at 3 versus 3, and `p = 0.4945` against an exact `1.0000` on a binary grid — off by `0.5055`. It returns a silent `NaN` on all-tied input rather than throwing. Measured cost: 286 packages, 11,997,061 bytes, 4,033 files.
154
+ - `@sipemu/anofox-statistics`'s `mannWhitneyU` returns `p_value = 0` for two **identical** samples, verified on four inputs including `x = y = [0.5,0.5,0.5]`, where the truth is `p = 1.0`. Its documented `exact` flag is a silent no-op under ties: all five tied cases returned byte-identical p for `exact: true` and `exact: false`.
155
+ - `@stdlib/stats-wilcoxon` has a genuine exact path but silently switches to the normal approximation when the differences contain ties, with nothing in the result to signal it — the `method` string is identical either way. Holding the statistic constant at `W = 15, n = 5`, untied input returns the exact `0.062500` and tied input returns `0.053337`, a p-value below the exact floor and therefore one that cannot exist at `n = 5`.
156
+ - `simple-statistics`'s `wilcoxonRankSum` returns the wrong rank sum on 5 of 9 regime cases; every case with two or more tie groups is corrupted. The tie accumulator is reset only in the single-element branch, so a multi-element group's average spans from the previous group's start. Upstream PR #809 carries a fix and a regression test, opened 2026-07-08, still open with no maintainer response.
157
+ - `mann-whitney-utest` and `@tainakanchu/mann-whitney-utest` (byte-identical source) compare a U statistic against a z-score in their significance decision, so the comparison is dimensionally meaningless. Complete separation at 3 versus 3 reports `significant = true` where no result at that size can reach `α = 0.05`.
158
+ - `jstat` (last published 2022-11-21) has no Mann-Whitney, no Wilcoxon, no Kruskal, no p-adjust. `science.js` (last published 2015-08-20) has zero hypothesis tests.
159
+
160
+ Dependency weight is a real cost, not a stylistic one.
161
+ This package's entire runtime dependency set is 7 packages.
162
+ Adding 190–286 for statistics we already compute correctly would be inherited by every consumer of `@tangle-network/agent-eval`.
163
+
164
+ ### What scipy is for
165
+
166
+ scipy is the CI oracle and never ships.
167
+
168
+ Pin scipy-generated golden values as a JSON fixture under `tests/`, regenerated by a checked-in script, and assert every **keep-with-CI-oracle** statistic against it.
169
+ That is `scripts/generate-statistics-oracle.py` → `tests/fixtures/statistics-oracle.json`, asserted by `tests/statistics-oracle.test.ts`: 154 cases across 22 statistics, each carrying its own tolerance (`7.5e-8` where A&S bounds it, `1e-8` where Acklam's inverse normal does, `1e-12` elsewhere).
170
+ scipy 1.13.1 was the reference for every number in this document and it catches exactly the class of defect found here: a hand-rolled approximation that is plausible on inspection and wrong by `3.7e-2`.
171
+ `lib-r-math.js` is a **devDependency only**, cross-checking the untied exact null distributions in `tests/statistics-library-crosscheck.test.ts`.
172
+ The three correction functions are independently covered by the statsmodels-generated fixture.
173
+ Neither reference implementation is imported by `src/`.
174
+ `fast-check` is already a devDependency, so the invariants that no fixture can express — a p-value never below the attainable floor, monotonicity of p in the statistic, an exact and an asymptotic path agreeing as `n` grows — belong there.
175
+
176
+ ## Exact versus asymptotic policy
177
+
178
+ The governing fact is combinatorial, not numerical.
179
+ At 3 versus 3 there are only 20 possible splits, so the attainable two-sided p-grid is `{0.1, 0.2, …}` and `0.05` is unreachable.
180
+ `[1,2,3]` versus `[4,5,6]` and `[0,0,0]` versus `[1,1,1]` both have an exact `p = 0.1000`, while scipy's tie-corrected asymptotic gives `0.080856` and `0.046854` — **adding the tie correction makes the answer worse.**
181
+ No better approximation reaches the right answer here; only an exact test does.
182
+
183
+ ### Switchover thresholds
184
+
185
+ | Test | Exact by enumeration | Seeded Monte Carlo permutation | Asymptotic |
186
+ | --- | --- | --- | --- |
187
+ | Two-sample rank (`mannWhitneyU`) | dynamic program up to 8,192 cells and 250,000 transitions; includes 12 v 12 and 1 v 24 | above that, default 100,000 permutations | never the default; only on explicit request above the exact work limits |
188
+ | Paired signed-rank (`wilcoxonSignedRank`) | `n ≤ 20` — `2²⁰ = 1,048,576` sign flips | above that, default 100,000 sign flips | same |
189
+ | Paired sign test (`pairedSignTest`) | always exact — binomial, already correct | — | never |
190
+ | McNemar (`mcnemar`) | always exact — binomial, already correct | — | never |
191
+ | Paired mean difference (`pairedTTest`) | — | — | valid from `n ≥ 3`, now that the `incompleteBeta` symmetry branch is in place |
192
+ | Bootstrap interval (`confidenceInterval`, `pairedBootstrap`) | — | — | **not a valid gate below `n = 20`** |
193
+
194
+ The bootstrap row is the strongest recommendation here and the one most likely to be resisted.
195
+ Measured `P(low > 0)` under a true null against a nominal 2.5 % never reaches nominal in the tested range: 13.53 % at `n = 3`, 3.52 % at `n = 10`, 3.10 % at `n = 20` for the median statistic, and 13.85 % / 4.90 % / 3.80 % for the mean.
196
+ Below `n = 20` the bootstrap interval should be reported as descriptive spread and must not be the leg a promotion turns on; the exact sign test or exact signed-rank should carry the decision instead.
197
+
198
+ ### What to do when the caller asks for a misleading number
199
+
200
+ Refuse, loudly, in the package's own idiom.
201
+
202
+ Every rank test takes `method: 'exact' | 'asymptotic' | 'auto'`, defaulting to `'auto'`.
203
+ `'auto'` selects exact whenever the design is inside the enumeration threshold, Monte Carlo permutation above it, and asymptotic never.
204
+ An explicit `method: 'asymptotic'` inside the exact-feasible range **throws a `ValidationError`** naming the smallest attainable p at that design and the exact work limits.
205
+ An explicit `method: 'exact'` ABOVE the threshold throws too, rather than enumerating a distribution whose cost is unbounded.
206
+
207
+ This is a refusal, not a warning, and the reason is the package's own doctrine.
208
+ A warning on `stderr` does not reach the JSON a gate reads, does not reach a CI log a human skims, and does not survive serialization into a run record.
209
+ An anti-conservative p that a gate silently believes is exactly the class of silent fallback that *no fallbacks, fail loud* exists to prevent, and the `wilcoxonSignedRank` `n < 6` branch was the proof: it returned `p = 1` for real effects from v0.1.0 to 0.133.0 and nothing downstream could tell.
210
+
211
+ The cost of the refusal is real and is stated here rather than discovered later: an explicit-asymptotic caller at 3 v 3 who was reading `0.0495` now gets an exception, and the same design read through the default now reports `0.1000`.
212
+ A historical verdict that turned on the difference was never valid.
213
+
214
+ Two supporting requirements make the refusal usable rather than merely obstructive.
215
+
216
+ Every rank-test result carries `method` and `pFloor` — the method actually used and the smallest attainable p at that design — so a downstream gate can see the discreteness rather than infer it.
217
+ This is the field `@stdlib/stats-wilcoxon` omits, which is why its silent exact-to-approximate switch is undetectable by a caller.
218
+ `pairedBootstrap` carries the same signal as `gateEligible`.
219
+
220
+ Still to do: every gate that consumes a rank test should state its minimum n at construction and fail its own precondition check when the data is smaller, rather than accepting whatever the test returns.
221
+ A gate that cannot reach its alpha at the n it was handed should report *underpowered*, which is a true statement about the experiment, not *not significant*, which is a false statement about the effect.
222
+ `pFloor` and `gateEligible` make that check expressible.
223
+ Promotion paths now use `pairedDeltaTest`: an exact one-sided sign test carries decisions from 6 through 19 pairs, and the bootstrap interval carries them from 20 onward.
224
+
225
+ ## Consumer notice
226
+
227
+ Every published version from **0.1.0 (2026-04-20)** through **0.133.0 (2026-07-27)** reported p-values that are too small from the functions listed below.
228
+ The defect entered at the initial commit (`7d5032b`, `src/statistics.ts:747`) and was present in every release since.
229
+
230
+ The normal CDF itself was corrected in 0.133.1; every other row below was still open at that release.
231
+
232
+ ### Which functions, and by how much
233
+
234
+ | Function | Path to the defect | Direction | Measured |
235
+ | --- | --- | --- | --- |
236
+ | `mannWhitneyU` | `normalCdf`, then the asymptotic path itself | p too small | `[1,2,3]` v `[4,5,6]`: reported `0.03769147`. Repairing the CDF alone gives `0.04953448` (1.31×); the shipped exact answer is `0.10000000` (2.65×), and `0.05` is unreachable at 3 v 3 in the first place. `[1..5]` v `[10..14]`: reported `0.00671001`, shipped exact `0.00793651` (1.18×). |
237
+ | `wilcoxonSignedRank` | `normalCdf`, then the asymptotic path itself | p too small, or fabricated as 1 | `n = 8` constant shift: reported `0.00873623`, CDF-repaired `0.01171872`, shipped exact `0.00781250`. Below six non-zero differences the old code returned `p = 1` regardless of the data: a clean 5-of-5 shift reported `1.0` where the exact answer is `0.0625`. |
238
+ | `pairedTTest` at `df > 100` | deleted `studentTCdf` normal shortcut | p too small | A `df = 298` case: reported `0.010887`, correct Student-t `0.015150`. |
239
+ | `welchsTTest`, `compareToBaseline` | duplicate `normalCdf` in `baseline.ts` | p too small | Same `df = 298` case, same shift. This drives the improved / regressed / stable verdict at `baseline.ts:97`. |
240
+ | `mcnemarPower` | `normalCdf` | power **overstated** above 0.5, understated below | `{p10: 0.2, p01: 0.1, nPairs: 200}`: reported `0.773437`, correct `0.736522`. At `n = 80`: `0.9368` against `0.9197`. At `n = 20`: `0.3294` against a correct `0.3627`. |
241
+ | `pairedTTest` at `df ≤ 100`, small `\|t\|` | `incompleteBeta` — **fixed** | p wildly too small near `t = 0` | `df = 7, t = 0.001`: reported `0.15130173`, correct `0.99923002`. `df = 100, t = 1e-6`: reported `0.00004411`, correct ≈ `1.0`. |
242
+ | `interRaterReliability` | grouping loop — **fixed** | sign inverted | Identical judges return `−0.500000`; maximally disagreeing judges return `+1.000000`. |
243
+
244
+ `requiredSampleSize`, `requiredPairedSampleSize`, `pairedMde`, `mcnemarRequiredN`, `mcnemar`, `pairedSignTest`, `wilson`, `passAtK`, `corpusInterRaterAgreement`, `eProcess`, `holm`, `ranks`, `pearsonR`, `spearmanR`, `cliffsDelta`, and `pairedCohensDz` are **unaffected** — verified numerically identical before and after (`requiredSampleSize({effect: 0.5}) = 63`, `mcnemarRequiredN({p10: 0.2, p01: 0.1}) = 234` both ways).
245
+
246
+ ### How to re-check a decision you already made
247
+
248
+ The defect is monotone in `|z|`, so the affected band is exact and narrow.
249
+
250
+ The broken code crossed `p = 0.05` at `|z| = 1.843031` instead of the correct `1.959964`, and at that true critical value it reported `p = 0.038053`.
251
+ Therefore:
252
+
253
+ - **Any recorded p in `[0.038053, 0.050000)` from an affected function crossed a 5 % gate that it should not have crossed.**
254
+ - At `α = 0.01` the band is `[0.007443, 0.010000)`.
255
+ - At `α = 0.10` the band is `[0.077398, 0.100000)`.
256
+ - A recorded p below `0.038053` was significant at 5 % either way; a recorded p at or above `0.05` was not significant either way. Neither needs re-checking.
257
+
258
+ The practical size of the error: the module's real type-I error rate was **6.53 % at a nominal 5 %** and **1.34 % at a nominal 1 %**, confirmed by 20,000 null replicates at `n = 8` per group where `mannWhitneyU` rejected at 6.47 % as shipped against 4.86 % with the same U and z fed a correct CDF.
259
+
260
+ Three further cautions for anyone auditing an old verdict.
261
+
262
+ A promotion that turned on a `pairedBootstrap` `low > 0` check at fewer than 10 pairs was never valid at the stated confidence, independent of this defect — the measured false-positive rate is 13.53 % at `n = 3` against a nominal 2.5 %.
263
+
264
+ A `wilcoxonSignedRank` leg that reported `p = 1` on fewer than six non-zero differences measured nothing.
265
+ It is not evidence of no effect, and re-running it on this release returns a real exact p.
266
+ Note that exact ties are dropped before ranking, so ten pairs with five tied deltas also fell into that branch.
267
+
268
+ Any bootstrap interval recorded from a release at or before 0.133.0 through `analyze-runs.ts` is not reproducible: that call site passed no seed and `makeRng` fell back to `Math.random`.
269
+ Re-running it against the old release will not give the same interval. From this release the seed is derived from the data when the caller supplies none, so it is reproducible either way — but an interval recorded earlier cannot be reconstructed.
270
+
271
+ Mirror this section into `CHANGELOG.md` at the release that carries the fix, with the affected version range and the re-check bands, so a consumer who never reads this file still gets the notice.
package/docs/design.md CHANGED
@@ -65,5 +65,6 @@ They are not adoption reference:
65
65
 
66
66
  - [`building-doctrine.md`](./building-doctrine.md): conventions our agents follow when consuming this package (reachable model defaults, probe-before-debug, experiment integrity checklist)
67
67
  - [`design/loop-taxonomy.md`](./design/loop-taxonomy.md): the internal vocabulary for execution drivers, workers, measurements, and proposers
68
+ - [`design/statistics-decisions.md`](./design/statistics-decisions.md): per-statistic trust status, the no-runtime-dependency verdict, and the exact-versus-asymptotic policy at 3–10 repetitions
68
69
  - [`research-report-methodology.md`](./research-report-methodology.md): the evidence standard our own research reports are held to
69
70
  - [`.claude/skills/agent-eval/SKILL.md`](../.claude/skills/agent-eval/SKILL.md): directives for LLM agents writing integration code, encoding bug classes we have already shipped and fixed once
@@ -247,7 +247,7 @@ Populated when baseline + candidate candidates are present (auto-detected from t
247
247
  "candidateMean": 0.65,
248
248
  "delta": 0.07,
249
249
  "ci95": [0.04, 0.10], // bootstrap CI on the delta
250
- "pValue": 0.0008, // paired t-test
250
+ "pValue": 0.0008, // paired t-test; null when the delta is a non-zero constant
251
251
  "n": 40, // paired observations
252
252
  "unpairedBaselineRuns": 2,
253
253
  "unpairedCandidateRuns": 1,
@@ -62,9 +62,10 @@ In order: first match wins:
62
62
  |---|---|---|
63
63
  | Marginal CI on score mean | `confidenceInterval` | `statistics.ts` |
64
64
  | Paired Cohen's dz vs comparator | `pairedCohensDz` | `statistics.ts` |
65
- | Wilcoxon signed-rank (paired) | `wilcoxonSignedRank` | `statistics.ts` |
65
+ | Wilcoxon signed-rank (paired), exact at n ≤ 20 | `wilcoxonSignedRank` | `statistics.ts` |
66
66
  | BH-FDR q-values | `benjaminiHochberg` | `statistics.ts` |
67
67
  | Paired bootstrap CI on median delta | `pairedBootstrap` | `statistics.ts` |
68
+ | Smallest p a rank-test design can produce | `pFloor` on the result | `statistics.ts` |
68
69
  | Bayesian-bootstrap Pr(Δ>0), Pr(Δ∈ROPE) | `bayesianBootstrapMeanSamples` | `summary-report.ts` (private) |
69
70
  | Minimum detectable paired effect | `pairedMde` | `statistics.ts` |
70
71
  | Run fingerprint | `hashJson(canonicalize(...))` | `pre-registration.ts` |
@@ -73,6 +74,8 @@ The Pr(Δ>0) and Pr(Δ∈ROPE) summaries use Rubin's Bayesian bootstrap.
73
74
  Each posterior draw assigns the observed paired deltas Dirichlet(1, ..., 1) weights, implemented as normalized independent Exponential(1) draws.
74
75
  The posterior summaries apply to the **mean** delta.
75
76
  The separate frequentist bootstrap CI applies to the **median** delta because the median is more robust to heavy-tailed agent scores.
77
+ It is descriptive spread below `BOOTSTRAP_GATE_MIN_N = 20` pairs, which the result reports as `gateEligible: false`; the exact signed-rank or sign test carries the decision there.
78
+ See [`design/statistics-decisions.md`](./design/statistics-decisions.md) for the measured false-positive rates and the exact-versus-asymptotic policy.
76
79
 
77
80
  ## MDE
78
81
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tangle-network/agent-eval",
3
- "version": "0.133.2",
3
+ "version": "0.133.3",
4
4
  "description": "Evaluate and improve AI agents from runs, traces, judges, and feedback. Compare candidates, cluster failures, measure lift, and gate releases.",
5
5
  "homepage": "https://github.com/tangle-network/agent-eval#readme",
6
6
  "repository": {
@@ -180,6 +180,7 @@
180
180
  "esbuild": "^0.28.1",
181
181
  "fast-check": "^4.9.0",
182
182
  "husky": "^9.1.7",
183
+ "lib-r-math.js": "^3.0.2",
183
184
  "lint-staged": "^17.2.0",
184
185
  "openapi3-ts": "^4.6.0",
185
186
  "oxc-parser": "^0.141.0",
@@ -1 +0,0 @@
1
- {"version":3,"file":"analyze-runs-B-afTpCv.js","names":[],"sources":["../src/contamination-guard.ts","../src/contract/analyze-runs.ts"],"sourcesContent":["/**\n * ContaminationGuard — ensures held-out scenarios don't leak into\n * training/prompt paths, and flags model memorization.\n *\n * Three probes:\n * 1. `checkCanaries(output, scenario)` — if the scenario carries a\n * canary token, it MUST NOT appear in the agent's output.\n * Canaries are strings that are statistically impossible to\n * reconstruct from the scenario description alone — so if they\n * echo back, the model memorized them.\n * 2. `canaryLeakView(store)` — cross-corpus view of every run whose\n * output contained a canary, with the offending scenario + run.\n * 3. `HoldoutAuditor` — wraps a Dataset and emits a structured error\n * on any code path that reads holdout scenarios but doesn't flag\n * `purpose: 'evaluation'`. Keeps engineers honest.\n */\n\nimport type { DatasetScenario } from './dataset'\nimport { llmSpans } from './trace/query'\nimport type { TraceStore } from './trace/store'\n\nexport interface CanaryLeak {\n scenarioId: string\n canary: string\n runId?: string\n evidence: string\n}\n\nexport function checkCanaries(output: string, scenarios: DatasetScenario[]): CanaryLeak[] {\n const leaks: CanaryLeak[] = []\n for (const s of scenarios) {\n if (!s.canary) continue\n if (output.includes(s.canary)) {\n leaks.push({ scenarioId: s.id, canary: s.canary, evidence: excerpt(output, s.canary) })\n }\n }\n return leaks\n}\n\n/**\n * Behavioral canary: tests the AGENT, not the eval grep.\n *\n * - `forbiddenPattern` PRESENT in output ⇒ AGENT EMITTED BAD CONTENT ⇒ LEAK.\n * - `forbiddenPattern` ABSENT from output ⇒ AGENT HELD THE LINE ⇒ PASS.\n *\n * Use when running known-bad-prompt scenarios against the agent under\n * test and you want to know if the agent misbehaved. The classical\n * {@link checkCanaries} / {@link import('./canary').runCanaries | runCanaries}\n * test whether the eval check fires when the bad output is forced\n * into the eval flow — different question, different answer.\n *\n * Pattern resolution order (first match wins):\n * 1. `scenario.forbiddenPattern` — if it parses as `/body/flags`,\n * treated as a regex; otherwise a literal substring.\n * 2. `scenario.canary` — literal substring fallback so the helper\n * works on existing scenario fixtures.\n *\n * Returns `null` when nothing forbidden was found OR the scenario\n * declared no pattern.\n */\nexport function checkBehavioralCanary(\n output: string,\n scenario: DatasetScenario,\n): CanaryLeak | null {\n const pattern = scenario.forbiddenPattern ?? scenario.canary\n if (!pattern) return null\n const hit = matchForbidden(output, pattern)\n if (!hit) return null\n return {\n scenarioId: scenario.id,\n canary: pattern,\n evidence: excerpt(output, hit),\n }\n}\n\n/**\n * Behavioral canary over many (scenario, output) pairs. Sibling to\n * {@link import('./canary').runCanaries | runCanaries} — same idea\n * (run-many → report) but the question being answered is \"did the\n * AGENT misbehave?\" rather than \"did the EVAL grep fire?\".\n *\n * Returns one `CanaryLeak` per pair where the agent's output\n * contained its scenario's `forbiddenPattern` (or `canary` fallback).\n */\nexport function runBehavioralCanaries(\n cases: Array<{ scenario: DatasetScenario; output: string; runId?: string }>,\n): CanaryLeak[] {\n const leaks: CanaryLeak[] = []\n for (const c of cases) {\n const leak = checkBehavioralCanary(c.output, c.scenario)\n if (leak) leaks.push({ ...leak, runId: c.runId ?? leak.runId })\n }\n return leaks\n}\n\n/**\n * Resolve a forbidden-pattern string to the matched substring inside\n * `output`. `/body/flags` notation is interpreted as a regex; anything\n * else is a literal substring.\n */\nfunction matchForbidden(output: string, pattern: string): string | null {\n const re = tryParseRegex(pattern)\n if (re) {\n const m = output.match(re)\n return m && m[0].length > 0 ? m[0] : null\n }\n return output.includes(pattern) ? pattern : null\n}\n\nfunction tryParseRegex(pattern: string): RegExp | null {\n if (pattern.length < 2 || pattern[0] !== '/') return null\n const last = pattern.lastIndexOf('/')\n if (last <= 0) return null\n const body = pattern.slice(1, last)\n const flags = pattern.slice(last + 1)\n if (!/^[gimsuy]*$/.test(flags)) return null\n try {\n return new RegExp(body, flags)\n } catch {\n return null\n }\n}\n\n/**\n * Scan the LLM-output history in a corpus; returns every case where a\n * canary from a known scenario appeared in agent output. Pass the full\n * set of scenarios whose canaries you care about (typically the whole\n * held-out slice).\n */\nexport async function canaryLeakView(\n store: TraceStore,\n scenarios: DatasetScenario[],\n): Promise<CanaryLeak[]> {\n const targets = scenarios.filter((s) => !!s.canary)\n if (targets.length === 0) return []\n const spans = await llmSpans(store)\n const leaks: CanaryLeak[] = []\n for (const span of spans) {\n const output = span.output ?? ''\n for (const s of targets) {\n if (s.canary && output.includes(s.canary)) {\n leaks.push({\n scenarioId: s.id,\n canary: s.canary,\n runId: span.runId,\n evidence: excerpt(output, s.canary),\n })\n }\n }\n }\n return leaks\n}\n\nexport class HoldoutAuditor {\n private scenarios: DatasetScenario[]\n private accessLog: Array<{ scenarioId: string; purpose: string; at: number }> = []\n\n constructor(scenarios: DatasetScenario[]) {\n this.scenarios = scenarios\n }\n\n /** Retrieve a holdout scenario for a declared purpose. Non-'evaluation' throws. */\n get(scenarioId: string, purpose: 'evaluation' | 'debugging'): DatasetScenario {\n if (purpose !== 'evaluation' && purpose !== 'debugging') {\n throw new Error(\n `HoldoutAuditor.get: purpose must be 'evaluation' or 'debugging', got ${purpose}`,\n )\n }\n const s = this.scenarios.find((x) => x.id === scenarioId)\n if (!s) throw new Error(`holdout scenario \"${scenarioId}\" not found`)\n this.accessLog.push({ scenarioId, purpose, at: Date.now() })\n return s\n }\n\n getAccessLog(): ReadonlyArray<{ scenarioId: string; purpose: string; at: number }> {\n return this.accessLog\n }\n}\n\nfunction excerpt(source: string, needle: string): string {\n const at = source.indexOf(needle)\n if (at < 0) return ''\n const start = Math.max(0, at - 30)\n const end = Math.min(source.length, at + needle.length + 30)\n return (start > 0 ? '…' : '') + source.slice(start, end) + (end < source.length ? '…' : '')\n}\n","/**\n * # `analyzeRuns()` — turn a set of agent runs into an actionable decision packet.\n *\n * Wires the substrate's statistical, calibration, clustering, Pareto, and\n * release-confidence primitives into one `InsightReport`. Two top-level\n * entry points use this function:\n *\n * - `selfImprove()` calls it on the campaign output to attach a packet\n * to every run.\n * - Consumers with observed `RunRecord[]` (production traces, gold\n * corpora, approve/reject tables) call it directly via `analyzeRuns()`\n * for analysis without a closed loop.\n *\n * Every section is opt-in based on what the input data supports — the\n * function never invents signal. If runs carry no judge scores, `judges`\n * is empty. If there's no baseline/candidate split, `lift` is undefined.\n * If no `analyst` is wired, `failureClusters` is undefined.\n *\n * The `recommendations` array is the human-readable layer; everything\n * else is the evidence backing each recommendation.\n */\n\nimport type { AnalystRegistry } from '../analyst/registry'\nimport type { AnalystFinding } from '../analyst/types'\nimport { checkCanaries } from '../contamination-guard'\nimport type { DatasetScenario } from '../dataset'\nimport { continuousAgreement } from '../judge-calibration'\nimport { normalCdf } from '../math/normal'\nimport { pairRunRecords } from '../paired-arms'\nimport { observedSplitScore } from '../rollout/reward'\nimport {\n type RunRecord,\n type RunTerminalOutcome,\n type RunTokenUsage,\n validateRunRecord,\n} from '../run-record'\nimport {\n pairedBootstrap,\n pairedCohensDz,\n pairedMde,\n pairedTTest,\n pearsonR,\n requiredPairedSampleSize,\n spearmanR,\n} from '../statistics'\nimport { type ParetoFigureSpec, paretoChart } from '../summary-report'\nimport type { FailureClass } from '../trace/schema'\n\nimport type {\n CostProvenanceSummary,\n ExecutionInsight,\n FailureClassTally,\n FailureClusterInsight,\n InsightReport,\n InterRaterInsight,\n JudgeInsight,\n LiftInsight,\n MetricDelta,\n OutcomeCorrelationInsight,\n PriorPeriodComparison,\n Recommendation,\n ScalarDistribution,\n TokenUsageInsight,\n} from './insight-report'\n\n// ── Public API ───────────────────────────────────────────────────────\n\nexport interface AnalyzeRunsOptions {\n /** The runs to analyze. */\n runs: RunRecord[]\n /** Which split to score against when reading composite from RunOutcome.\n * Default: holdout when ANY run has a `holdoutScore`, else search. */\n split?: 'search' | 'holdout' | 'auto'\n /** Pairwise analysis configuration. When both `baselineCandidateId` and\n * `candidateCandidateId` are present, lift is computed on paired\n * (experimentId, scenarioId, seed) identities shared between the two sides.\n * Unmatched rows remain visible in the lift result. */\n baselineCandidateId?: string\n candidateCandidateId?: string\n /** Canary scenarios — checked against every run's raw output for\n * holdout contamination. */\n canaryScenarios?: DatasetScenario[]\n /** Analyst registry for failure clustering. When omitted, the\n * `failureClusters` section is left undefined. */\n analyst?: AnalystRegistry\n /** Downstream outcome metric per run (e.g. engagement rate, approval\n * rate, downstream pass rate). When present, the report includes\n * `outcomeCorrelation` + a simple linear reward model fit. */\n outcomeSignal?: {\n metric: string\n valueByRunId: Record<string, number>\n }\n /** Multi-rater feedback for inter-rater agreement. Each entry is one\n * rater's score for one run. Two or more raters → kappa + disagreement\n * triage list. */\n raterScores?: Array<{ runId: string; rater: string; score: number }>\n /** Number of histogram bins for distributional summaries. Default 12. */\n histogramBins?: number\n /** Decision threshold — the smallest composite lift the caller cares\n * about. Used by the recommendations engine to call ship vs hold.\n * Default 0.02. */\n decisionThreshold?: number\n /** Optional prior-period runs. When set, the report includes\n * `priorPeriodComparison` with per-metric Welch-CI deltas and\n * recommendations fire on statistically significant regressions.\n * The two windows do NOT have to share scenarios — the comparison\n * is two-sample unpaired (the substrate's `lift` field uses paired\n * bootstrap on shared (experimentId, scenarioId, seed) identities; this is the\n * shape for \"this week vs last week\" rather than \"candidate vs\n * baseline within a campaign\"). */\n baselineRuns?: RunRecord[]\n /** Human-readable label for the baseline window, e.g. \"vs prior 7\n * days\", \"vs v3.1 release\". Surfaces in recommendations + UI. */\n baselineLabel?: string\n}\n\nexport interface SummarizeExecutionOptions {\n runs: RunRecord[]\n histogramBins?: number\n}\n\nexport interface ExecutionReport {\n execution: ExecutionInsight\n costProvenance: CostProvenanceSummary\n}\n\n/** Summarize runtime facts without interpreting task quality or promotion readiness. */\nexport function summarizeExecution(opts: SummarizeExecutionOptions): ExecutionReport {\n const runs = opts.runs.map(validateRunRecord)\n const bins = opts.histogramBins ?? 12\n return {\n execution: computeExecutionInsight(runs, bins),\n costProvenance: summarizeCostProvenance(runs),\n }\n}\n\nexport async function analyzeRuns(opts: AnalyzeRunsOptions): Promise<InsightReport> {\n const runs = opts.runs.map(validateRunRecord)\n const bins = opts.histogramBins ?? 12\n const threshold = opts.decisionThreshold ?? 0.02\n const split = resolveSplit(runs, opts.split ?? 'auto')\n\n const compositeWithIds = runs\n .map((r) => ({ runId: r.runId, score: compositeOf(r, split) }))\n .filter((p) => Number.isFinite(p.score))\n const composite = distributionOf(\n compositeWithIds.map((p) => p.score),\n bins,\n compositeWithIds,\n )\n\n const perDimension = computePerDimension(runs, bins)\n const { execution, costProvenance: provenance } = summarizeExecution({\n runs,\n histogramBins: bins,\n })\n const knownCostRuns = runs.filter((run) => run.costProvenance.kind !== 'uncaptured')\n const costs = knownCostRuns.map((r) => r.costUsd).filter(isFiniteNumber)\n const costDist = distributionOf(costs, bins)\n const pareto = paretoChart(knownCostRuns, { split })\n const degraded: { cost?: string; pareto?: string } = {}\n if (provenance.uncaptured.n > 0) {\n degraded.cost = diagnoseCostCoverage(runs, provenance)\n } else if (costs.length === 0 || costs.every((c) => c === 0)) {\n degraded.cost = `all ${runs.length} explicitly observed or estimated USD values are $0`\n }\n if (pareto.points.length < 2) {\n degraded.pareto =\n pareto.points.length === 0\n ? 'no candidates — Pareto unavailable'\n : 'single candidate — Pareto is a single point, not a frontier'\n }\n const costQuality = {\n cost: costDist,\n pareto,\n provenance,\n ...(degraded.cost || degraded.pareto ? { degraded } : {}),\n }\n\n const judges = computeJudgeInsights(runs)\n\n const interRater = opts.raterScores ? computeInterRater(opts.raterScores) : undefined\n\n const lift = computeLift(runs, opts.baselineCandidateId, opts.candidateCandidateId, split)\n\n const failureClusters = opts.analyst\n ? await computeFailureClusters(runs, opts.analyst, split)\n : undefined\n\n const failureClasses = computeFailureClasses(runs, split)\n\n const contamination = opts.canaryScenarios\n ? computeContamination(runs, opts.canaryScenarios)\n : undefined\n\n const outcomeCorrelation = opts.outcomeSignal\n ? computeOutcomeCorrelation(runs, opts.outcomeSignal, split)\n : undefined\n\n const release = buildReleaseScorecard(composite, lift, contamination)\n\n const priorPeriodComparison = opts.baselineRuns\n ? computePriorPeriodComparison(runs, opts.baselineRuns, split, opts.baselineLabel)\n : undefined\n\n const recommendations = buildRecommendations({\n composite,\n judges,\n interRater,\n lift,\n failureClusters,\n failureClasses,\n contamination,\n outcomeCorrelation,\n priorPeriodComparison,\n threshold,\n })\n\n return {\n n: runs.length,\n execution,\n composite,\n perDimension,\n costQuality,\n judges,\n interRater,\n lift,\n failureClusters,\n contamination,\n outcomeCorrelation,\n release,\n ...(failureClasses ? { failureClasses } : {}),\n ...(priorPeriodComparison ? { priorPeriodComparison } : {}),\n recommendations,\n }\n}\n\nfunction computeExecutionInsight(runs: RunRecord[], bins: number): ExecutionInsight {\n const aggregateRows = runs.flatMap((run) => {\n const usage = aggregateTokenUsage(run)\n return usage ? [{ usage, costUsd: finiteRaw(run, 'aggregate_cost_usd') }] : []\n })\n const aggregateCosts = aggregateRows.flatMap((row) =>\n row.costUsd !== undefined ? [row.costUsd] : [],\n )\n const modelCounts = new Map<string, number>()\n let executionErrorRuns = 0\n let executionErrorEvents = 0\n let errorReportingRuns = 0\n let errorSpanEvents = 0\n let errorSpanReportingRuns = 0\n const terminalOutcomes: Record<RunTerminalOutcome, number> = {\n succeeded: 0,\n failed: 0,\n cancelled: 0,\n incomplete: 0,\n unknown: 0,\n }\n const errorsByTerminalOutcome: ExecutionInsight['executionErrors']['byTerminalOutcome'] = {\n succeeded: { withErrors: 0, withoutErrors: 0, unreported: 0 },\n failed: { withErrors: 0, withoutErrors: 0, unreported: 0 },\n cancelled: { withErrors: 0, withoutErrors: 0, unreported: 0 },\n incomplete: { withErrors: 0, withoutErrors: 0, unreported: 0 },\n unknown: { withErrors: 0, withoutErrors: 0, unreported: 0 },\n }\n let modelCallRuns = 0\n let modelCallEvents = 0\n let modelCallReportingRuns = 0\n\n for (const run of runs) {\n modelCounts.set(run.model, (modelCounts.get(run.model) ?? 0) + 1)\n const terminalOutcome = run.terminalOutcome\n terminalOutcomes[terminalOutcome] += 1\n const modelCalls = nonNegativeCountRaw(run, 'llm_span_count')\n if (modelCalls !== undefined) {\n modelCallEvents += modelCalls\n modelCallReportingRuns += 1\n }\n const usage = run.tokenUsage\n if (\n (modelCalls ?? 0) > 0 ||\n usage.input > 0 ||\n usage.output > 0 ||\n (usage.cached ?? 0) > 0 ||\n (usage.cacheWrite ?? 0) > 0\n ) {\n modelCallRuns += 1\n }\n const errorEvents = reportedExecutionErrorEvents(run)\n if (errorEvents !== undefined) {\n executionErrorEvents += errorEvents\n errorReportingRuns += 1\n if (errorEvents > 0) {\n executionErrorRuns += 1\n errorsByTerminalOutcome[terminalOutcome].withErrors += 1\n } else errorsByTerminalOutcome[terminalOutcome].withoutErrors += 1\n } else errorsByTerminalOutcome[terminalOutcome].unreported += 1\n const reportedErrorSpans = nonNegativeCountRaw(run, 'error_span_count')\n if (reportedErrorSpans !== undefined) {\n errorSpanEvents += reportedErrorSpans\n errorSpanReportingRuns += 1\n }\n }\n\n return {\n durationMs: distributionOf(\n runs.map((run) => run.wallMs),\n bins,\n ),\n queueMs: distributionOf(\n runs.filter((run) => run.queueMs !== undefined).map((run) => run.queueMs!),\n bins,\n ),\n tokenUsage: summarizeTokenUsage(\n runs.map((run) => run.tokenUsage),\n bins,\n ),\n aggregateUsage: {\n runs: aggregateRows.length,\n tokenUsage: summarizeTokenUsage(\n aggregateRows.map((row) => row.usage),\n bins,\n ),\n costUsd: distributionOf(aggregateCosts, bins),\n totalCostUsd: aggregateCosts.reduce((total, value) => total + value, 0),\n },\n models: [...modelCounts.entries()]\n .map(([model, count]) => ({ model, runs: count }))\n .sort((left, right) => right.runs - left.runs || left.model.localeCompare(right.model)),\n modelCalls: {\n runs: modelCallRuns,\n events: modelCallEvents,\n reportingRuns: modelCallReportingRuns,\n },\n executionErrors: {\n runs: executionErrorRuns,\n fraction: errorReportingRuns > 0 ? executionErrorRuns / errorReportingRuns : null,\n events: executionErrorEvents,\n reportingRuns: errorReportingRuns,\n errorSpanEvents,\n errorSpanReportingRuns,\n byTerminalOutcome: errorsByTerminalOutcome,\n },\n terminalOutcomes,\n }\n}\n\nfunction reportedExecutionErrorEvents(run: RunRecord): number | undefined {\n return nonNegativeCountRaw(run, 'execution_error_count')\n}\n\nfunction nonNegativeCountRaw(run: RunRecord, key: string): number | undefined {\n const value = finiteRaw(run, key)\n return value !== undefined && Number.isInteger(value) && value >= 0 ? value : undefined\n}\n\nfunction summarizeTokenUsage(usages: RunTokenUsage[], bins: number): TokenUsageInsight {\n const reasoning = usages.flatMap((usage) =>\n usage.reasoning !== undefined ? [usage.reasoning] : [],\n )\n const cached = usages.flatMap((usage) => (usage.cached !== undefined ? [usage.cached] : []))\n const cacheWrite = usages.flatMap((usage) =>\n usage.cacheWrite !== undefined ? [usage.cacheWrite] : [],\n )\n return {\n input: distributionOf(\n usages.map((usage) => usage.input),\n bins,\n ),\n output: distributionOf(\n usages.map((usage) => usage.output),\n bins,\n ),\n reasoning: distributionOf(reasoning, bins),\n cached: distributionOf(cached, bins),\n cacheWrite: distributionOf(cacheWrite, bins),\n totals: {\n input: usages.reduce((total, usage) => total + usage.input, 0),\n output: usages.reduce((total, usage) => total + usage.output, 0),\n reasoning: reasoning.reduce((total, value) => total + value, 0),\n cached: cached.reduce((total, value) => total + value, 0),\n cacheWrite: cacheWrite.reduce((total, value) => total + value, 0),\n },\n }\n}\n\nfunction aggregateTokenUsage(run: RunRecord): RunTokenUsage | undefined {\n const input = finiteRaw(run, 'aggregate_prompt_tokens')\n const output = finiteRaw(run, 'aggregate_completion_tokens')\n const reasoning = finiteRaw(run, 'aggregate_reasoning_tokens')\n const cached = finiteRaw(run, 'aggregate_cached_tokens')\n const cacheWrite = finiteRaw(run, 'aggregate_cache_write_tokens')\n if (\n input === undefined &&\n output === undefined &&\n reasoning === undefined &&\n cached === undefined &&\n cacheWrite === undefined\n )\n return undefined\n return {\n input: input ?? 0,\n output: output ?? 0,\n ...(reasoning !== undefined ? { reasoning } : {}),\n ...(cached !== undefined ? { cached } : {}),\n ...(cacheWrite !== undefined ? { cacheWrite } : {}),\n }\n}\n\nfunction finiteRaw(run: RunRecord, key: string): number | undefined {\n const value = run.outcome.raw[key]\n return typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : undefined\n}\n\nfunction summarizeCostProvenance(runs: RunRecord[]): CostProvenanceSummary {\n const summary: CostProvenanceSummary = {\n observed: { n: 0, totalUsd: 0 },\n estimated: { n: 0, totalUsd: 0 },\n uncaptured: { n: 0 },\n knownFraction: 0,\n }\n for (const run of runs) {\n const cost = run.costProvenance\n if (cost.kind === 'uncaptured') {\n summary.uncaptured.n += 1\n } else {\n summary[cost.kind].n += 1\n summary[cost.kind].totalUsd += cost.usd\n }\n }\n const known = summary.observed.n + summary.estimated.n\n summary.knownFraction = runs.length > 0 ? known / runs.length : 0\n return summary\n}\n\nfunction diagnoseCostCoverage(runs: RunRecord[], provenance: CostProvenanceSummary): string {\n const uncaptured = provenance.uncaptured.n\n const known = provenance.observed.n + provenance.estimated.n\n if (uncaptured === runs.length) {\n return `USD cost uncaptured for all ${runs.length} runs — no observed or estimated USD values; token and wall-time metrics remain available.`\n }\n return `USD cost uncaptured for ${uncaptured}/${runs.length} runs; excluded those rows from cost statistics (${known}/${runs.length} retained: ${provenance.observed.n} observed, ${provenance.estimated.n} estimated).`\n}\n\n/**\n * Model-free task-failure tally.\n *\n * Explicit non-success classes are task-failure evidence.\n * A low task score without a class is counted as `unknown`.\n */\nfunction computeFailureClasses(\n runs: RunRecord[],\n split: 'search' | 'holdout',\n): FailureClassTally[] | undefined {\n const counts = new Map<FailureClass, number>()\n for (const r of runs) {\n if (!isTaskFailure(r, split)) continue\n const key =\n r.failureClass !== undefined && r.failureClass !== 'success' ? r.failureClass : 'unknown'\n counts.set(key, (counts.get(key) ?? 0) + 1)\n }\n if (counts.size === 0) return undefined\n const n = runs.length\n return [...counts.entries()]\n .map(([failureClass, count]) => ({\n failureClass,\n count,\n share: n > 0 ? count / n : 0,\n }))\n .sort((a, b) => b.count - a.count || a.failureClass.localeCompare(b.failureClass))\n}\n\n// ── Prior-period comparison ─────────────────────────────────────────\n\n/** Direction of the metric — does \"higher current\" mean better or worse?\n * Composite + judge dimensions: higher is better. Cost + duration: lower\n * is better. The recommendations engine flips the sign before judging\n * regressed vs improved. */\ntype MetricDirection = 'higher-is-better' | 'lower-is-better'\n\nfunction computePriorPeriodComparison(\n current: RunRecord[],\n baseline: RunRecord[],\n split: 'search' | 'holdout',\n windowLabel: string | undefined,\n): PriorPeriodComparison | undefined {\n if (current.length === 0 || baseline.length === 0) return undefined\n\n const metrics: Record<string, MetricDelta> = {}\n const directions: Record<string, MetricDirection> = {}\n\n const compositeCurrent = current\n .map((r) => compositeOf(r, split))\n .filter(Number.isFinite) as number[]\n const compositeBaseline = baseline\n .map((r) => compositeOf(r, split))\n .filter(Number.isFinite) as number[]\n if (compositeCurrent.length > 0 && compositeBaseline.length > 0) {\n metrics.composite = welchCompare(compositeBaseline, compositeCurrent)\n directions.composite = 'higher-is-better'\n }\n\n const costCurrent = knownCostValues(current)\n const costBaseline = knownCostValues(baseline)\n if (costCurrent.length > 0 && costBaseline.length > 0) {\n metrics.cost = welchCompare(costBaseline, costCurrent)\n directions.cost = 'lower-is-better'\n }\n\n const durCurrent = current.map((r) => r.wallMs).filter(Number.isFinite)\n const durBaseline = baseline.map((r) => r.wallMs).filter(Number.isFinite)\n if (durCurrent.length > 0 && durBaseline.length > 0) {\n metrics.duration = welchCompare(durBaseline, durCurrent)\n directions.duration = 'lower-is-better'\n }\n\n const tokCurrent = current\n .map((r) => (r.tokenUsage.input ?? 0) + (r.tokenUsage.output ?? 0))\n .filter(Number.isFinite)\n const tokBaseline = baseline\n .map((r) => (r.tokenUsage.input ?? 0) + (r.tokenUsage.output ?? 0))\n .filter(Number.isFinite)\n if (tokCurrent.length > 0 && tokBaseline.length > 0) {\n metrics.tokenUsage = welchCompare(tokBaseline, tokCurrent)\n directions.tokenUsage = 'lower-is-better'\n }\n\n // Per-dimension judge comparisons — only for dimensions present in BOTH\n // windows. We use perDimMean since per-judge nesting is finicky for\n // two-sample comparisons across different judge configurations.\n const dimsCurrent = collectPerDimension(current)\n const dimsBaseline = collectPerDimension(baseline)\n for (const dim of Object.keys(dimsCurrent)) {\n const b = dimsBaseline[dim]\n const c = dimsCurrent[dim]\n if (!b || b.length === 0 || !c || c.length === 0) continue\n metrics[`dim.${dim}`] = welchCompare(b, c)\n directions[`dim.${dim}`] = 'higher-is-better'\n }\n\n const regressedMetrics: string[] = []\n const improvedMetrics: string[] = []\n for (const [name, delta] of Object.entries(metrics)) {\n if (!delta.significant) continue\n const dir = directions[name] ?? 'higher-is-better'\n const better = dir === 'higher-is-better' ? delta.delta > 0 : delta.delta < 0\n if (better) improvedMetrics.push(name)\n else regressedMetrics.push(name)\n }\n\n return {\n baselineN: baseline.length,\n currentN: current.length,\n ...(windowLabel ? { windowLabel } : {}),\n metrics,\n regressedMetrics,\n improvedMetrics,\n }\n}\n\nfunction knownCostValues(runs: RunRecord[]): number[] {\n return runs\n .filter((run) => run.costProvenance.kind !== 'uncaptured')\n .map((run) => run.costUsd)\n .filter(isFiniteNumber)\n}\n\nfunction isFiniteNumber(value: unknown): value is number {\n return typeof value === 'number' && Number.isFinite(value)\n}\n\n/** Collect per-dimension values across runs (from outcome.judgeScores.perDimMean). */\nfunction collectPerDimension(runs: RunRecord[]): Record<string, number[]> {\n const out: Record<string, number[]> = {}\n for (const r of runs) {\n const perDim = r.outcome.judgeScores?.perDimMean\n if (!perDim) continue\n for (const [dim, value] of Object.entries(perDim)) {\n if (!Number.isFinite(value)) continue\n if (!out[dim]) out[dim] = []\n out[dim].push(value as number)\n }\n }\n return out\n}\n\n/** Two-sample Welch comparison: unequal-variance t-test + CI on the delta\n * + Cohen's d (pooled stddev). Significance = p < 0.05 AND |d| >= 0.2. */\nfunction welchCompare(baseline: number[], current: number[]): MetricDelta {\n const baselineMean = mean(baseline)\n const currentMean = mean(current)\n const baselineVar = sampleVariance(baseline, baselineMean)\n const currentVar = sampleVariance(current, currentMean)\n const baselineN = baseline.length\n const currentN = current.length\n const delta = currentMean - baselineMean\n\n // Welch standard error\n const se = Math.sqrt(baselineVar / baselineN + currentVar / currentN)\n // For 95% CI we use z=1.96 (large-n approximation). Customers running\n // analyzeRuns will typically have n >= 30; the t-correction is\n // negligible vs the practical noise floor.\n const halfWidth = 1.96 * (se > 0 ? se : 0)\n const ci95: [number, number] = [delta - halfWidth, delta + halfWidth]\n\n // p-value via normal approximation to the t-statistic.\n const t = se > 0 ? delta / se : 0\n const pValue = se > 0 ? 2 * (1 - normalCdf(Math.abs(t))) : 1\n\n // Cohen's d — pooled stddev.\n const pooledStddev = Math.sqrt(\n ((baselineN - 1) * baselineVar + (currentN - 1) * currentVar) /\n Math.max(1, baselineN + currentN - 2),\n )\n const cohensD = pooledStddev > 0 ? delta / pooledStddev : 0\n\n // Significance: BOTH p < 0.05 AND |d| >= 0.2 (small-effect threshold).\n const significant = pValue < 0.05 && Math.abs(cohensD) >= 0.2\n\n return {\n current: currentMean,\n baseline: baselineMean,\n delta,\n ci95,\n pValue,\n cohensD,\n baselineN,\n currentN,\n significant,\n }\n}\n\nfunction sampleVariance(xs: number[], xsMean: number): number {\n if (xs.length < 2) return 0\n let s = 0\n for (const x of xs) s += (x - xsMean) ** 2\n return s / (xs.length - 1)\n}\n\n// ── Composite + split selection ─────────────────────────────────────\n\nfunction resolveSplit(\n runs: RunRecord[],\n pref: 'search' | 'holdout' | 'auto',\n): 'search' | 'holdout' {\n if (pref !== 'auto') return pref\n const hasHoldout = runs.some((r) => Number.isFinite(observedSplitScore(r, 'holdout')))\n return hasHoldout ? 'holdout' : 'search'\n}\n\n/**\n * RAW (`observedSplitScore`): `analyzeRuns` describes what a set of runs\n * reported, and every downstream reader of this composite — distributions,\n * per-candidate summaries, the reward-hacking correlation — needs the ungated\n * number to see an inflated run at all.\n */\nfunction compositeOf(run: RunRecord, split: 'search' | 'holdout'): number {\n // Split-exact, no cross-split fallthrough: answering \"what did this run\n // score on the split I am summarising\" with the other split's number\n // silently mixes populations.\n const score = observedSplitScore(run, split)\n return Number.isFinite(score) ? (score as number) : Number.NaN\n}\n\n// ── Distribution helpers ────────────────────────────────────────────\n\nfunction distributionOf(\n values: number[],\n bins: number,\n withIds?: Array<{ runId: string; score: number }>,\n): ScalarDistribution {\n if (values.length === 0) {\n return {\n n: 0,\n mean: null,\n p50: null,\n p95: null,\n stddev: null,\n min: null,\n max: null,\n histogram: [],\n }\n }\n const sorted = [...values].sort((a, b) => a - b)\n const n = sorted.length\n const mean = sorted.reduce((s, v) => s + v, 0) / n\n const variance = sorted.reduce((s, v) => s + (v - mean) ** 2, 0) / n\n const stddev = Math.sqrt(variance)\n const tailRuns = withIds\n ? [...withIds].sort((a, b) => a.score - b.score).slice(0, Math.min(5, withIds.length))\n : undefined\n return {\n n,\n mean,\n p50: percentile(sorted, 0.5),\n p95: percentile(sorted, 0.95),\n stddev,\n min: sorted[0]!,\n max: sorted[n - 1]!,\n histogram: histogram(sorted, bins),\n ...(tailRuns ? { tailRuns } : {}),\n }\n}\n\nfunction percentile(sorted: number[], q: number): number {\n if (sorted.length === 0) return 0\n if (sorted.length === 1) return sorted[0]!\n const idx = (sorted.length - 1) * q\n const lo = Math.floor(idx)\n const hi = Math.ceil(idx)\n if (lo === hi) return sorted[lo]!\n const w = idx - lo\n return sorted[lo]! * (1 - w) + sorted[hi]! * w\n}\n\n/** Even-width histogram over the value range. Returns inclusive-lo /\n * exclusive-hi bins (closed on right for the last bin) compatible with\n * the substrate's `GainDistributionBin` shape. */\nfunction histogram(sorted: number[], bins: number): ScalarDistribution['histogram'] {\n if (sorted.length === 0 || bins < 1) return []\n const min = sorted[0]!\n const max = sorted[sorted.length - 1]!\n if (min === max) return [{ lo: min, hi: max, count: sorted.length }]\n const width = (max - min) / bins\n const out: ScalarDistribution['histogram'] = []\n for (let i = 0; i < bins; i++) {\n const lo = min + i * width\n const hi = i === bins - 1 ? max : lo + width\n out.push({ lo, hi, count: 0 })\n }\n for (const v of sorted) {\n const idx = Math.min(bins - 1, Math.floor((v - min) / width))\n out[idx]!.count++\n }\n return out\n}\n\nfunction computePerDimension(runs: RunRecord[], bins: number): Record<string, ScalarDistribution> {\n // JudgeScoresRecord pre-aggregates `perDimMean` (mean across judges per\n // dimension). We collect those means across runs to produce a per-dim\n // distribution at the corpus level. Consumers who want per-judge\n // dimension values reach into `perJudge[judgeId][dim]` themselves.\n const byDim = new Map<string, number[]>()\n for (const run of runs) {\n const scores = run.outcome.judgeScores\n if (!scores) continue\n for (const [dim, value] of Object.entries(scores.perDimMean ?? {})) {\n if (!Number.isFinite(value)) continue\n const arr = byDim.get(dim) ?? []\n arr.push(value)\n byDim.set(dim, arr)\n }\n }\n const out: Record<string, ScalarDistribution> = {}\n for (const [dim, values] of byDim) out[dim] = distributionOf(values, bins)\n return out\n}\n\n// ── Judge insights ──────────────────────────────────────────────────\n\nfunction computeJudgeInsights(runs: RunRecord[]): Record<string, JudgeInsight> {\n // Each judge's per-run mean is the average of its per-dimension scores\n // for that run. We aggregate those means across all runs each judge\n // scored — giving consumers a \"this judge's typical verdict\" reading.\n const out: Record<string, JudgeInsight> = {}\n const byJudge = new Map<string, number[]>()\n for (const run of runs) {\n const scores = run.outcome.judgeScores\n if (!scores?.perJudge) continue\n for (const [judgeId, dims] of Object.entries(scores.perJudge)) {\n const dimValues = Object.values(dims).filter(Number.isFinite) as number[]\n if (dimValues.length === 0) continue\n const judgeMean = dimValues.reduce((s, v) => s + v, 0) / dimValues.length\n const arr = byJudge.get(judgeId) ?? []\n arr.push(judgeMean)\n byJudge.set(judgeId, arr)\n }\n }\n for (const [judgeId, values] of byJudge) {\n out[judgeId] = {\n n: values.length,\n meanScore: values.reduce((s, v) => s + v, 0) / values.length,\n }\n }\n return out\n}\n\n// ── Inter-rater agreement ───────────────────────────────────────────\n\nfunction computeInterRater(\n ratings: Array<{ runId: string; rater: string; score: number }>,\n): InterRaterInsight | undefined {\n const byRun = new Map<string, Array<{ rater: string; score: number }>>()\n for (const r of ratings) {\n if (!Number.isFinite(r.score)) continue\n const list = byRun.get(r.runId) ?? []\n list.push({ rater: r.rater, score: r.score })\n byRun.set(r.runId, list)\n }\n const raters = new Set(ratings.map((r) => r.rater))\n const jointlyRated: string[] = []\n for (const [runId, ratersForRun] of byRun) {\n const seen = new Set(ratersForRun.map((r) => r.rater))\n let all = true\n for (const r of raters) if (!seen.has(r)) all = false\n if (all) jointlyRated.push(runId)\n }\n if (raters.size < 2 || jointlyRated.length === 0) return undefined\n\n const raterList = [...raters].sort()\n const perPair: Record<string, number> = {}\n for (let i = 0; i < raterList.length; i++) {\n for (let j = i + 1; j < raterList.length; j++) {\n const a = raterList[i]!\n const b = raterList[j]!\n const aScores: number[] = []\n const bScores: number[] = []\n for (const runId of jointlyRated) {\n const ratersForRun = byRun.get(runId)!\n const sa = ratersForRun.find((r) => r.rater === a)?.score\n const sb = ratersForRun.find((r) => r.rater === b)?.score\n if (sa !== undefined && sb !== undefined) {\n aScores.push(sa)\n bScores.push(sb)\n }\n }\n const agreement = continuousAgreement(\n aScores.map((score, index) => [score, bScores[index]!]),\n { bootstrap: 0 },\n )\n perPair[`${a}::${b}`] = agreement.weightedKappa\n }\n }\n const matrix = jointlyRated.map((runId) => {\n const ratingsByRater = new Map(byRun.get(runId)!.map((rating) => [rating.rater, rating.score]))\n return raterList.map((rater) => ratingsByRater.get(rater)!)\n })\n const agreement = continuousAgreement(matrix, { bootstrap: 0 })\n\n const disagreementCases = jointlyRated\n .map((runId) => {\n const ratersForRun = byRun.get(runId)!\n const scores = ratersForRun.map((r) => r.score)\n const range = Math.max(...scores) - Math.min(...scores)\n return { runId, ratings: ratersForRun, range }\n })\n .sort((a, b) => b.range - a.range)\n .slice(0, 20)\n\n return {\n raters: raters.size,\n jointlyRated: jointlyRated.length,\n kappa: Number.isFinite(agreement.weightedKappa) ? agreement.weightedKappa : 0,\n icc: agreement.icc,\n pearson: agreement.pearson,\n spearman: agreement.spearman,\n perPair,\n disagreementCases,\n }\n}\n\n// ── Lift ────────────────────────────────────────────────────────────\n\nfunction computeLift(\n runs: RunRecord[],\n baselineId: string | undefined,\n candidateId: string | undefined,\n split: 'search' | 'holdout',\n): LiftInsight | undefined {\n let bId = baselineId\n let cId = candidateId\n if (!bId || !cId) {\n // Auto-detect: when exactly two distinct candidateIds appear, treat the\n // lower-mean side as baseline.\n const ids = [...new Set(runs.map((r) => r.candidateId))]\n if (ids.length !== 2) return undefined\n const [idA, idB] = ids as [string, string]\n const scoresA = finiteCompositeScores(\n runs.filter((run) => run.candidateId === idA),\n split,\n )\n const scoresB = finiteCompositeScores(\n runs.filter((run) => run.candidateId === idB),\n split,\n )\n if (scoresA.length === 0 || scoresB.length === 0) return undefined\n const meanA = mean(scoresA)\n const meanB = mean(scoresB)\n bId = meanA <= meanB ? idA : idB\n cId = meanA <= meanB ? idB : idA\n }\n\n const baseline = runs.filter((r) => r.candidateId === bId)\n const candidate = runs.filter((r) => r.candidateId === cId)\n if (baseline.length === 0 || candidate.length === 0) return undefined\n\n const scoredBaseline = baseline.filter((run) => Number.isFinite(compositeOf(run, split)))\n const scoredCandidate = candidate.filter((run) => Number.isFinite(compositeOf(run, split)))\n const pairing = pairRunRecords(scoredBaseline, scoredCandidate)\n const pairedBaseline = pairing.pairs.map((pair) => compositeOf(pair.baseline, split))\n const pairedCandidate = pairing.pairs.map((pair) => compositeOf(pair.treatment, split))\n if (pairedBaseline.length === 0) return undefined\n\n const baselineMean = mean(pairedBaseline)\n const candidateMean = mean(pairedCandidate)\n const delta = candidateMean - baselineMean\n\n const bootstrap = pairedBootstrap(pairedBaseline, pairedCandidate, {\n confidence: 0.95,\n resamples: 2000,\n statistic: 'mean',\n })\n const tTest = pairedTTest(pairedBaseline, pairedCandidate)\n const d = pairedCohensDz(pairedBaseline, pairedCandidate)\n const mde = pairedMde({ nPaired: pairedBaseline.length, power: 0.8, alpha: 0.05 })\n const requiredN =\n d === null || d === 0\n ? null\n : requiredPairedSampleSize({\n effect: Math.abs(d),\n power: 0.8,\n alpha: 0.05,\n })\n\n return {\n baselineMean,\n candidateMean,\n delta,\n ci95: [bootstrap.low, bootstrap.high],\n pValue: tTest.p,\n n: pairedBaseline.length,\n unpairedBaseline: pairing.unpairedBaseline.length,\n unpairedCandidate: pairing.unpairedTreatment.length,\n cohensD: d,\n mde,\n requiredN,\n }\n}\n\nfunction mean(arr: number[]): number {\n return arr.length === 0 ? 0 : arr.reduce((s, v) => s + v, 0) / arr.length\n}\n\n// ── Failure clustering ──────────────────────────────────────────────\n\nasync function computeFailureClusters(\n runs: RunRecord[],\n analyst: AnalystRegistry,\n split: 'search' | 'holdout',\n): Promise<FailureClusterInsight | undefined> {\n const failed = runs.filter((run) => isTaskFailure(run, split))\n if (failed.length === 0) return { clusters: [], totalFailures: 0 }\n\n const clusters = new Map<string, { exemplars: string[]; share: number }>()\n for (const run of failed) {\n try {\n // AnalystRunInputs routes by field name: run-record analysts read\n // `runRecord`. Any other shape makes every analyst skip with\n // \"missing input\" and the clusters come back silently empty.\n const result = await analyst.run(run.runId, { runRecord: run })\n for (const finding of result.findings as AnalystFinding[]) {\n const key = finding.area || finding.analyst_id || 'unclassified'\n const c = clusters.get(key) ?? { exemplars: [], share: 0 }\n if (c.exemplars.length < 5) c.exemplars.push(run.runId)\n clusters.set(key, c)\n }\n } catch {\n const c = clusters.get('analyst-error') ?? { exemplars: [], share: 0 }\n if (c.exemplars.length < 5) c.exemplars.push(run.runId)\n clusters.set('analyst-error', c)\n }\n }\n const clusterList = [...clusters.entries()].map(([id, c]) => ({\n id,\n name: id,\n share: c.exemplars.length / failed.length,\n exemplars: c.exemplars,\n }))\n clusterList.sort((a, b) => b.share - a.share)\n return { clusters: clusterList, totalFailures: failed.length }\n}\n\nfunction finiteCompositeScores(runs: readonly RunRecord[], split: 'search' | 'holdout'): number[] {\n return runs.map((run) => compositeOf(run, split)).filter(Number.isFinite)\n}\n\nfunction isTaskFailure(run: RunRecord, split: 'search' | 'holdout'): boolean {\n if (run.failureClass !== undefined && run.failureClass !== 'success') return true\n const score = compositeOf(run, split)\n return Number.isFinite(score) && score < 0.5\n}\n\n// ── Contamination ──────────────────────────────────────────────────\n\nfunction computeContamination(\n runs: RunRecord[],\n canaries: DatasetScenario[],\n): InsightReport['contamination'] {\n let leaks = 0\n const details: Array<{ runId: string; canary: string; matched: string }> = []\n for (const run of runs) {\n const output = stringifyOutput(run)\n if (!output) continue\n const leaksHere = checkCanaries(output, canaries)\n for (const leak of leaksHere) {\n leaks++\n details.push({ runId: run.runId, canary: leak.canary, matched: leak.evidence })\n }\n }\n return { leaks, holdoutAuditPassed: leaks === 0, details }\n}\n\nfunction stringifyOutput(run: RunRecord): string | undefined {\n // RunRecord doesn't fix where \"the agent's output\" lives — different\n // consumers stash it differently. We probe the common shapes: the\n // outcome.raw map (numeric only by design — unlikely to contain text),\n // and any string-valued fields tucked under metadata via type casting.\n // Consumers with bespoke shapes pass canaryScenarios only when they\n // know their runs carry a stringifiable surface.\n const metadata = (run as unknown as { metadata?: Record<string, unknown> }).metadata\n if (typeof metadata?.output === 'string') return metadata.output\n if (typeof metadata?.text === 'string') return metadata.text\n return undefined\n}\n\n// ── Outcome correlation + linear reward model ──────────────────────\n\nfunction computeOutcomeCorrelation(\n runs: RunRecord[],\n outcome: { metric: string; valueByRunId: Record<string, number> },\n split: 'search' | 'holdout',\n): OutcomeCorrelationInsight | undefined {\n const xs: number[] = []\n const ys: number[] = []\n for (const run of runs) {\n const y = outcome.valueByRunId[run.runId]\n if (y === undefined || !Number.isFinite(y)) continue\n const x = compositeOf(run, split)\n if (!Number.isFinite(x)) continue\n xs.push(x)\n ys.push(y)\n }\n if (xs.length < 3) return undefined\n\n const p = pearsonR(xs, ys)\n const s = spearmanR(xs, ys)\n const meanX = mean(xs)\n const meanY = mean(ys)\n let num = 0\n let denom = 0\n for (let i = 0; i < xs.length; i++) {\n num += (xs[i]! - meanX) * (ys[i]! - meanY)\n denom += (xs[i]! - meanX) ** 2\n }\n const slope = denom === 0 ? 0 : num / denom\n const intercept = meanY - slope * meanX\n const ssTot = ys.reduce((a, y) => a + (y - meanY) ** 2, 0)\n const ssRes = ys.reduce((a, y, i) => a + (y - (intercept + slope * xs[i]!)) ** 2, 0)\n const r2 = ssTot === 0 ? 0 : 1 - ssRes / ssTot\n\n return {\n metric: outcome.metric,\n n: xs.length,\n pearson: p,\n spearman: s,\n rewardModel: { intercept, slope, r2 },\n }\n}\n\n// ── Release confidence scorecard ───────────────────────────────────\n\nfunction buildReleaseScorecard(\n composite: ScalarDistribution,\n lift: LiftInsight | undefined,\n contamination: InsightReport['contamination'],\n): InsightReport['release'] {\n // Synthesise a minimal scorecard from the rolled-up signal. The\n // substrate's `evaluateReleaseConfidence` primitive consumes a richer\n // input shape that callers can produce by wiring SLO definitions; the\n // shape here is the contract `selfImprove`/`analyzeRuns` consumers\n // receive automatically. They can call `evaluateReleaseConfidence`\n // directly when they want SLO-based axis evaluation.\n const axes: InsightReport['release']['axes'] = []\n const liftPass =\n lift === undefined\n ? ('not_evaluated' as const)\n : lift.ci95[0] > 0\n ? ('pass' as const)\n : lift.delta > 0\n ? ('warn' as const)\n : ('fail' as const)\n axes.push({\n name: 'quality-lift',\n status: liftPass,\n detail: lift\n ? `delta=${lift.delta.toFixed(3)}, CI95=[${lift.ci95[0].toFixed(3)}, ${lift.ci95[1].toFixed(3)}], n=${lift.n}`\n : 'no baseline/candidate pair available',\n })\n const contamPass =\n contamination === undefined\n ? ('not_evaluated' as const)\n : contamination.leaks === 0\n ? ('pass' as const)\n : ('fail' as const)\n axes.push({\n name: 'contamination',\n status: contamPass,\n detail: contamination ? `${contamination.leaks} canary leak(s)` : 'no canaries supplied',\n })\n axes.push(\n composite.n === 0\n ? {\n name: 'composite-distribution',\n status: 'not_evaluated',\n detail: 'no task-quality scores available',\n }\n : {\n name: 'composite-distribution',\n status:\n composite.mean !== null && composite.mean >= 0.5\n ? 'pass'\n : composite.mean !== null && composite.mean >= 0.3\n ? 'warn'\n : 'fail',\n detail:\n composite.mean === null || composite.p50 === null || composite.p95 === null\n ? 'task-quality distribution is internally incomplete'\n : `mean=${composite.mean.toFixed(3)}, p50=${composite.p50.toFixed(3)}, p95=${composite.p95.toFixed(3)} over n=${composite.n}`,\n },\n )\n const status = axes.some((a) => a.status === 'fail')\n ? 'fail'\n : axes.some((a) => a.status === 'warn' || a.status === 'not_evaluated')\n ? 'warn'\n : 'pass'\n return {\n status,\n axes,\n issues: [],\n }\n}\n\n// ── Recommendations engine ─────────────────────────────────────────\n\ninterface RecommendationContext {\n composite: ScalarDistribution\n judges: Record<string, JudgeInsight>\n interRater?: InterRaterInsight\n lift?: LiftInsight\n failureClusters?: FailureClusterInsight\n failureClasses?: FailureClassTally[]\n contamination?: InsightReport['contamination']\n outcomeCorrelation?: OutcomeCorrelationInsight\n priorPeriodComparison?: PriorPeriodComparison\n threshold: number\n}\n\nfunction buildRecommendations(ctx: RecommendationContext): Recommendation[] {\n const out: Recommendation[] = []\n\n // Prior-period regressions — highest customer-impact signal when present.\n // \"Did my last change help?\" with a falsifiable answer.\n if (ctx.priorPeriodComparison) {\n const ppc = ctx.priorPeriodComparison\n const label = ppc.windowLabel ?? 'baseline period'\n for (const name of ppc.regressedMetrics) {\n const d = ppc.metrics[name]\n if (!d) continue\n out.push({\n priority: 'critical',\n kind: 'investigate',\n title: `${name} regressed from ${d.baseline.toFixed(3)} → ${d.current.toFixed(3)} vs ${label}`,\n detail: `Welch CI95 = [${d.ci95[0].toFixed(3)}, ${d.ci95[1].toFixed(3)}], p=${d.pValue.toFixed(4)}, Cohen's d=${d.cohensD.toFixed(2)} (n_current=${d.currentN}, n_baseline=${d.baselineN}). The regression is statistically significant at p<0.05 with at-least-small effect size.`,\n evidencePath: `priorPeriodComparison.metrics.${name}`,\n })\n }\n for (const name of ppc.improvedMetrics) {\n const d = ppc.metrics[name]\n if (!d) continue\n out.push({\n priority: 'low',\n kind: 'ship',\n title: `${name} improved from ${d.baseline.toFixed(3)} → ${d.current.toFixed(3)} vs ${label}`,\n detail: `Welch CI95 = [${d.ci95[0].toFixed(3)}, ${d.ci95[1].toFixed(3)}], p=${d.pValue.toFixed(4)}, Cohen's d=${d.cohensD.toFixed(2)} (n_current=${d.currentN}, n_baseline=${d.baselineN}). Statistically significant improvement worth flagging.`,\n evidencePath: `priorPeriodComparison.metrics.${name}`,\n })\n }\n }\n\n // Composite-distribution branch. Fires when the overall quality signal is\n // poor regardless of lift / contamination / clusters — the customer needs\n // to know they have a problem AND which specific runs to inspect.\n if (\n ctx.composite.n > 0 &&\n ctx.composite.mean !== null &&\n ctx.composite.p50 !== null &&\n ctx.composite.p95 !== null\n ) {\n if (ctx.composite.mean < 0.3) {\n const tail = ctx.composite.tailRuns ?? []\n const names = tail\n .slice(0, 5)\n .map((t) => `${t.runId}=${t.score.toFixed(3)}`)\n .join(', ')\n out.push({\n priority: 'critical',\n kind: 'investigate',\n title: `Composite mean ${ctx.composite.mean.toFixed(3)} is below the 0.3 floor — the agent is broken on this corpus`,\n detail:\n tail.length > 0\n ? `Worst ${tail.length} run${tail.length === 1 ? '' : 's'} to inspect first: ${names}. Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.`\n : `Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.`,\n evidencePath: 'composite.tailRuns',\n })\n } else if (ctx.composite.mean < 0.5) {\n const tail = ctx.composite.tailRuns ?? []\n const names = tail\n .slice(0, 3)\n .map((t) => `${t.runId}=${t.score.toFixed(3)}`)\n .join(', ')\n out.push({\n priority: 'high',\n kind: 'investigate',\n title: `Composite mean ${ctx.composite.mean.toFixed(3)} is below 0.5 — investigate the lower tail before claiming the agent is healthy`,\n detail:\n tail.length > 0\n ? `Worst ${tail.length} run${tail.length === 1 ? '' : 's'}: ${names}. Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.`\n : `Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.`,\n evidencePath: 'composite.tailRuns',\n })\n }\n }\n\n // A healthy-looking mean can hide a group of failed tasks sharing one\n // producer-reported cause. This path does not require an analyst.\n if (ctx.failureClasses && ctx.failureClasses.length > 0) {\n const top = ctx.failureClasses[0]!\n if (top.count >= 3 && top.share >= 0.15) {\n out.push({\n priority: top.share >= 0.25 ? 'high' : 'medium',\n kind: 'investigate',\n title: `'${top.failureClass}' is the dominant failure class — ${top.count} runs (${(top.share * 100).toFixed(0)}% of the corpus)`,\n detail: `The mean composite can look acceptable while one failure class dominates the lower tail. ${top.count} of ${ctx.composite.n} runs failed with '${top.failureClass}'${ctx.failureClasses.length > 1 ? ` (next: '${ctx.failureClasses[1]!.failureClass}' ×${ctx.failureClasses[1]!.count})` : ''}. Fix this cause first.`,\n evidencePath: 'failureClasses',\n })\n }\n }\n\n // Missing-judges branch. The report can't surface per-dimension or\n // calibration signal when `outcome.judgeScores` is empty across the\n // corpus. Tell the customer how to enrich.\n if (Object.keys(ctx.judges).length === 0 && ctx.composite.n > 0) {\n out.push({\n priority: 'medium',\n kind: 'expand-corpus',\n title: 'No judge scores recorded — per-dimension + calibration insights unavailable',\n detail:\n 'Records have no `outcome.judgeScores`. To unlock perDimension, judges, and calibration, attach a Judge run during your eval pass and populate `outcome.judgeScores.perJudge[judgeName][dimension] = score`. See `docs/insight-report.md` for the expected shape.',\n evidencePath: 'judges',\n })\n }\n\n if (ctx.lift) {\n const pairedEffect =\n ctx.lift.cohensD === null ? 'undefined (zero delta variance)' : ctx.lift.cohensD.toFixed(2)\n const requiredRuns =\n ctx.lift.requiredN === null ? 'not estimable' : `~${ctx.lift.requiredN} paired runs`\n const decisive = ctx.lift.ci95[0] > ctx.threshold\n const inconclusive = ctx.lift.ci95[0] <= ctx.threshold && ctx.lift.ci95[1] > ctx.threshold\n if (decisive) {\n out.push({\n priority: 'critical',\n kind: 'ship',\n title: `Ship — lift ${ctx.lift.delta.toFixed(3)} (95% CI ${ctx.lift.ci95[0].toFixed(3)}..${ctx.lift.ci95[1].toFixed(3)})`,\n detail: `Holdout lift exceeds threshold ${ctx.threshold} with 95% bootstrap confidence (n=${ctx.lift.n}, p=${ctx.lift.pValue.toFixed(4)}, paired d=${pairedEffect}).`,\n evidencePath: 'lift',\n })\n } else if (inconclusive) {\n out.push({\n priority: 'high',\n kind: 'expand-corpus',\n title: `Inconclusive — required sample is ${requiredRuns} (have ${ctx.lift.n}) at current effect size`,\n detail: `CI straddles threshold. Current MDE at 80% power is ${ctx.lift.mde.toFixed(3)}; observed delta is ${ctx.lift.delta.toFixed(3)}.`,\n evidencePath: 'lift',\n })\n } else {\n out.push({\n priority: 'critical',\n kind: 'hold',\n title: `Hold — lift CI lower bound ${ctx.lift.ci95[0].toFixed(3)} is at or below threshold ${ctx.threshold}`,\n detail: `Bootstrap CI provides no statistical evidence the candidate is better. Consider tightening the mutation or expanding the holdout.`,\n evidencePath: 'lift',\n })\n }\n }\n\n if (ctx.contamination && ctx.contamination.leaks > 0) {\n out.push({\n priority: 'critical',\n kind: 'fix',\n title: `${ctx.contamination.leaks} canary leak${ctx.contamination.leaks === 1 ? '' : 's'} detected`,\n detail: `Holdout integrity is compromised. The lift number is unreliable until you investigate.`,\n evidencePath: 'contamination',\n })\n }\n\n if (ctx.interRater && ctx.interRater.kappa < 0.5) {\n out.push({\n priority: 'high',\n kind: 'recalibrate',\n title: `Inter-rater weighted kappa ${ctx.interRater.kappa.toFixed(2)} is below 0.5`,\n detail:\n 'Raters disagree on what good looks like. Review the largest disagreement cases and refine the rubric before automating these decisions.',\n evidencePath: 'interRater',\n })\n }\n\n if (ctx.failureClusters && ctx.failureClusters.clusters.length > 0) {\n const top = ctx.failureClusters.clusters[0]!\n out.push({\n priority: 'high',\n kind: 'investigate',\n title: `Top failure cluster: ${top.name} (${(top.share * 100).toFixed(0)}% of failures)`,\n detail: `${ctx.failureClusters.totalFailures} runs failed. The largest cluster groups ${top.exemplars.length} exemplars under '${top.name}'.`,\n evidencePath: 'failureClusters.clusters[0]',\n })\n }\n\n if (ctx.outcomeCorrelation && Math.abs(ctx.outcomeCorrelation.spearman) < 0.3) {\n out.push({\n priority: 'medium',\n kind: 'recalibrate',\n title: `Judge scores decoupled from ${ctx.outcomeCorrelation.metric} (Spearman ρ=${ctx.outcomeCorrelation.spearman.toFixed(2)})`,\n detail: `Your judges score what they were trained to score, but it isn't predicting downstream ${ctx.outcomeCorrelation.metric}. Consider retraining the judge against ${ctx.outcomeCorrelation.metric} as the gold signal.`,\n evidencePath: 'outcomeCorrelation',\n })\n }\n\n return out\n}\n\n// ── Re-export pareto figure spec for hosted-side rendering ─────────\n\nexport type { ParetoFigureSpec }\n"],"mappings":";;;;;;;AA4BA,SAAgB,cAAc,QAAgB,WAA4C;CACxF,MAAM,QAAsB,CAAC;CAC7B,KAAK,MAAM,KAAK,WAAW;EACzB,IAAI,CAAC,EAAE,QAAQ;EACf,IAAI,OAAO,SAAS,EAAE,MAAM,GAC1B,MAAM,KAAK;GAAE,YAAY,EAAE;GAAI,QAAQ,EAAE;GAAQ,UAAU,QAAQ,QAAQ,EAAE,MAAM;EAAE,CAAC;CAE1F;CACA,OAAO;AACT;;;;;;;;;;;;;;;;;;;;;;AAuBA,SAAgB,sBACd,QACA,UACmB;CACnB,MAAM,UAAU,SAAS,oBAAoB,SAAS;CACtD,IAAI,CAAC,SAAS,OAAO;CACrB,MAAM,MAAM,eAAe,QAAQ,OAAO;CAC1C,IAAI,CAAC,KAAK,OAAO;CACjB,OAAO;EACL,YAAY,SAAS;EACrB,QAAQ;EACR,UAAU,QAAQ,QAAQ,GAAG;CAC/B;AACF;;;;;;;;;;AAWA,SAAgB,sBACd,OACc;CACd,MAAM,QAAsB,CAAC;CAC7B,KAAK,MAAM,KAAK,OAAO;EACrB,MAAM,OAAO,sBAAsB,EAAE,QAAQ,EAAE,QAAQ;EACvD,IAAI,MAAM,MAAM,KAAK;GAAE,GAAG;GAAM,OAAO,EAAE,SAAS,KAAK;EAAM,CAAC;CAChE;CACA,OAAO;AACT;;;;;;AAOA,SAAS,eAAe,QAAgB,SAAgC;CACtE,MAAM,KAAK,cAAc,OAAO;CAChC,IAAI,IAAI;EACN,MAAM,IAAI,OAAO,MAAM,EAAE;EACzB,OAAO,KAAK,EAAE,EAAE,CAAC,SAAS,IAAI,EAAE,KAAK;CACvC;CACA,OAAO,OAAO,SAAS,OAAO,IAAI,UAAU;AAC9C;AAEA,SAAS,cAAc,SAAgC;CACrD,IAAI,QAAQ,SAAS,KAAK,QAAQ,OAAO,KAAK,OAAO;CACrD,MAAM,OAAO,QAAQ,YAAY,GAAG;CACpC,IAAI,QAAQ,GAAG,OAAO;CACtB,MAAM,OAAO,QAAQ,MAAM,GAAG,IAAI;CAClC,MAAM,QAAQ,QAAQ,MAAM,OAAO,CAAC;CACpC,IAAI,CAAC,cAAc,KAAK,KAAK,GAAG,OAAO;CACvC,IAAI;EACF,OAAO,IAAI,OAAO,MAAM,KAAK;CAC/B,QAAQ;EACN,OAAO;CACT;AACF;;;;;;;AAQA,eAAsB,eACpB,OACA,WACuB;CACvB,MAAM,UAAU,UAAU,QAAQ,MAAM,CAAC,CAAC,EAAE,MAAM;CAClD,IAAI,QAAQ,WAAW,GAAG,OAAO,CAAC;CAClC,MAAM,QAAQ,MAAM,SAAS,KAAK;CAClC,MAAM,QAAsB,CAAC;CAC7B,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,SAAS,KAAK,UAAU;EAC9B,KAAK,MAAM,KAAK,SACd,IAAI,EAAE,UAAU,OAAO,SAAS,EAAE,MAAM,GACtC,MAAM,KAAK;GACT,YAAY,EAAE;GACd,QAAQ,EAAE;GACV,OAAO,KAAK;GACZ,UAAU,QAAQ,QAAQ,EAAE,MAAM;EACpC,CAAC;CAGP;CACA,OAAO;AACT;AAEA,IAAa,iBAAb,MAA4B;CAC1B;CACA,YAAgF,CAAC;CAEjF,YAAY,WAA8B;EACxC,KAAK,YAAY;CACnB;;CAGA,IAAI,YAAoB,SAAsD;EAC5E,IAAI,YAAY,gBAAgB,YAAY,aAC1C,MAAM,IAAI,MACR,wEAAwE,SAC1E;EAEF,MAAM,IAAI,KAAK,UAAU,MAAM,MAAM,EAAE,OAAO,UAAU;EACxD,IAAI,CAAC,GAAG,MAAM,IAAI,MAAM,qBAAqB,WAAW,YAAY;EACpE,KAAK,UAAU,KAAK;GAAE;GAAY;GAAS,IAAI,KAAK,IAAI;EAAE,CAAC;EAC3D,OAAO;CACT;CAEA,eAAmF;EACjF,OAAO,KAAK;CACd;AACF;AAEA,SAAS,QAAQ,QAAgB,QAAwB;CACvD,MAAM,KAAK,OAAO,QAAQ,MAAM;CAChC,IAAI,KAAK,GAAG,OAAO;CACnB,MAAM,QAAQ,KAAK,IAAI,GAAG,KAAK,EAAE;CACjC,MAAM,MAAM,KAAK,IAAI,OAAO,QAAQ,KAAK,OAAO,SAAS,EAAE;CAC3D,QAAQ,QAAQ,IAAI,MAAM,MAAM,OAAO,MAAM,OAAO,GAAG,KAAK,MAAM,OAAO,SAAS,MAAM;AAC1F;;;;AC1DA,SAAgB,mBAAmB,MAAkD;CACnF,MAAM,OAAO,KAAK,KAAK,IAAI,iBAAiB;CAE5C,OAAO;EACL,WAAW,wBAAwB,MAFxB,KAAK,iBAAiB,EAEY;EAC7C,gBAAgB,wBAAwB,IAAI;CAC9C;AACF;AAEA,eAAsB,YAAY,MAAkD;CAClF,MAAM,OAAO,KAAK,KAAK,IAAI,iBAAiB;CAC5C,MAAM,OAAO,KAAK,iBAAiB;CACnC,MAAM,YAAY,KAAK,qBAAqB;CAC5C,MAAM,QAAQ,aAAa,MAAM,KAAK,SAAS,MAAM;CAErD,MAAM,mBAAmB,KACtB,KAAK,OAAO;EAAE,OAAO,EAAE;EAAO,OAAO,YAAY,GAAG,KAAK;CAAE,EAAE,CAAC,CAC9D,QAAQ,MAAM,OAAO,SAAS,EAAE,KAAK,CAAC;CACzC,MAAM,YAAY,eAChB,iBAAiB,KAAK,MAAM,EAAE,KAAK,GACnC,MACA,gBACF;CAEA,MAAM,eAAe,oBAAoB,MAAM,IAAI;CACnD,MAAM,EAAE,WAAW,gBAAgB,eAAe,mBAAmB;EACnE;EACA,eAAe;CACjB,CAAC;CACD,MAAM,gBAAgB,KAAK,QAAQ,QAAQ,IAAI,eAAe,SAAS,YAAY;CACnF,MAAM,QAAQ,cAAc,KAAK,MAAM,EAAE,OAAO,CAAC,CAAC,OAAO,cAAc;CACvE,MAAM,WAAW,eAAe,OAAO,IAAI;CAC3C,MAAM,SAAS,YAAY,eAAe,EAAE,MAAM,CAAC;CACnD,MAAM,WAA+C,CAAC;CACtD,IAAI,WAAW,WAAW,IAAI,GAC5B,SAAS,OAAO,qBAAqB,MAAM,UAAU;MAChD,IAAI,MAAM,WAAW,KAAK,MAAM,OAAO,MAAM,MAAM,CAAC,GACzD,SAAS,OAAO,OAAO,KAAK,OAAO;CAErC,IAAI,OAAO,OAAO,SAAS,GACzB,SAAS,SACP,OAAO,OAAO,WAAW,IACrB,uCACA;CAER,MAAM,cAAc;EAClB,MAAM;EACN;EACA;EACA,GAAI,SAAS,QAAQ,SAAS,SAAS,EAAE,SAAS,IAAI,CAAC;CACzD;CAEA,MAAM,SAAS,qBAAqB,IAAI;CAExC,MAAM,aAAa,KAAK,cAAc,kBAAkB,KAAK,WAAW,IAAI,KAAA;CAE5E,MAAM,OAAO,YAAY,MAAM,KAAK,qBAAqB,KAAK,sBAAsB,KAAK;CAEzF,MAAM,kBAAkB,KAAK,UACzB,MAAM,uBAAuB,MAAM,KAAK,SAAS,KAAK,IACtD,KAAA;CAEJ,MAAM,iBAAiB,sBAAsB,MAAM,KAAK;CAExD,MAAM,gBAAgB,KAAK,kBACvB,qBAAqB,MAAM,KAAK,eAAe,IAC/C,KAAA;CAEJ,MAAM,qBAAqB,KAAK,gBAC5B,0BAA0B,MAAM,KAAK,eAAe,KAAK,IACzD,KAAA;CAEJ,MAAM,UAAU,sBAAsB,WAAW,MAAM,aAAa;CAEpE,MAAM,wBAAwB,KAAK,eAC/B,6BAA6B,MAAM,KAAK,cAAc,OAAO,KAAK,aAAa,IAC/E,KAAA;CAEJ,MAAM,kBAAkB,qBAAqB;EAC3C;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CACF,CAAC;CAED,OAAO;EACL,GAAG,KAAK;EACR;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,GAAI,iBAAiB,EAAE,eAAe,IAAI,CAAC;EAC3C,GAAI,wBAAwB,EAAE,sBAAsB,IAAI,CAAC;EACzD;CACF;AACF;AAEA,SAAS,wBAAwB,MAAmB,MAAgC;CAClF,MAAM,gBAAgB,KAAK,SAAS,QAAQ;EAC1C,MAAM,QAAQ,oBAAoB,GAAG;EACrC,OAAO,QAAQ,CAAC;GAAE;GAAO,SAAS,UAAU,KAAK,oBAAoB;EAAE,CAAC,IAAI,CAAC;CAC/E,CAAC;CACD,MAAM,iBAAiB,cAAc,SAAS,QAC5C,IAAI,YAAY,KAAA,IAAY,CAAC,IAAI,OAAO,IAAI,CAAC,CAC/C;CACA,MAAM,8BAAc,IAAI,IAAoB;CAC5C,IAAI,qBAAqB;CACzB,IAAI,uBAAuB;CAC3B,IAAI,qBAAqB;CACzB,IAAI,kBAAkB;CACtB,IAAI,yBAAyB;CAC7B,MAAM,mBAAuD;EAC3D,WAAW;EACX,QAAQ;EACR,WAAW;EACX,YAAY;EACZ,SAAS;CACX;CACA,MAAM,0BAAoF;EACxF,WAAW;GAAE,YAAY;GAAG,eAAe;GAAG,YAAY;EAAE;EAC5D,QAAQ;GAAE,YAAY;GAAG,eAAe;GAAG,YAAY;EAAE;EACzD,WAAW;GAAE,YAAY;GAAG,eAAe;GAAG,YAAY;EAAE;EAC5D,YAAY;GAAE,YAAY;GAAG,eAAe;GAAG,YAAY;EAAE;EAC7D,SAAS;GAAE,YAAY;GAAG,eAAe;GAAG,YAAY;EAAE;CAC5D;CACA,IAAI,gBAAgB;CACpB,IAAI,kBAAkB;CACtB,IAAI,yBAAyB;CAE7B,KAAK,MAAM,OAAO,MAAM;EACtB,YAAY,IAAI,IAAI,QAAQ,YAAY,IAAI,IAAI,KAAK,KAAK,KAAK,CAAC;EAChE,MAAM,kBAAkB,IAAI;EAC5B,iBAAiB,oBAAoB;EACrC,MAAM,aAAa,oBAAoB,KAAK,gBAAgB;EAC5D,IAAI,eAAe,KAAA,GAAW;GAC5B,mBAAmB;GACnB,0BAA0B;EAC5B;EACA,MAAM,QAAQ,IAAI;EAClB,KACG,cAAc,KAAK,KACpB,MAAM,QAAQ,KACd,MAAM,SAAS,MACd,MAAM,UAAU,KAAK,MACrB,MAAM,cAAc,KAAK,GAE1B,iBAAiB;EAEnB,MAAM,cAAc,6BAA6B,GAAG;EACpD,IAAI,gBAAgB,KAAA,GAAW;GAC7B,wBAAwB;GACxB,sBAAsB;GACtB,IAAI,cAAc,GAAG;IACnB,sBAAsB;IACtB,wBAAwB,gBAAgB,CAAC,cAAc;GACzD,OAAO,wBAAwB,gBAAgB,CAAC,iBAAiB;EACnE,OAAO,wBAAwB,gBAAgB,CAAC,cAAc;EAC9D,MAAM,qBAAqB,oBAAoB,KAAK,kBAAkB;EACtE,IAAI,uBAAuB,KAAA,GAAW;GACpC,mBAAmB;GACnB,0BAA0B;EAC5B;CACF;CAEA,OAAO;EACL,YAAY,eACV,KAAK,KAAK,QAAQ,IAAI,MAAM,GAC5B,IACF;EACA,SAAS,eACP,KAAK,QAAQ,QAAQ,IAAI,YAAY,KAAA,CAAS,CAAC,CAAC,KAAK,QAAQ,IAAI,OAAQ,GACzE,IACF;EACA,YAAY,oBACV,KAAK,KAAK,QAAQ,IAAI,UAAU,GAChC,IACF;EACA,gBAAgB;GACd,MAAM,cAAc;GACpB,YAAY,oBACV,cAAc,KAAK,QAAQ,IAAI,KAAK,GACpC,IACF;GACA,SAAS,eAAe,gBAAgB,IAAI;GAC5C,cAAc,eAAe,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC;EACxE;EACA,QAAQ,CAAC,GAAG,YAAY,QAAQ,CAAC,CAAC,CAC/B,KAAK,CAAC,OAAO,YAAY;GAAE;GAAO,MAAM;EAAM,EAAE,CAAC,CACjD,MAAM,MAAM,UAAU,MAAM,OAAO,KAAK,QAAQ,KAAK,MAAM,cAAc,MAAM,KAAK,CAAC;EACxF,YAAY;GACV,MAAM;GACN,QAAQ;GACR,eAAe;EACjB;EACA,iBAAiB;GACf,MAAM;GACN,UAAU,qBAAqB,IAAI,qBAAqB,qBAAqB;GAC7E,QAAQ;GACR,eAAe;GACf;GACA;GACA,mBAAmB;EACrB;EACA;CACF;AACF;AAEA,SAAS,6BAA6B,KAAoC;CACxE,OAAO,oBAAoB,KAAK,uBAAuB;AACzD;AAEA,SAAS,oBAAoB,KAAgB,KAAiC;CAC5E,MAAM,QAAQ,UAAU,KAAK,GAAG;CAChC,OAAO,UAAU,KAAA,KAAa,OAAO,UAAU,KAAK,KAAK,SAAS,IAAI,QAAQ,KAAA;AAChF;AAEA,SAAS,oBAAoB,QAAyB,MAAiC;CACrF,MAAM,YAAY,OAAO,SAAS,UAChC,MAAM,cAAc,KAAA,IAAY,CAAC,MAAM,SAAS,IAAI,CAAC,CACvD;CACA,MAAM,SAAS,OAAO,SAAS,UAAW,MAAM,WAAW,KAAA,IAAY,CAAC,MAAM,MAAM,IAAI,CAAC,CAAE;CAC3F,MAAM,aAAa,OAAO,SAAS,UACjC,MAAM,eAAe,KAAA,IAAY,CAAC,MAAM,UAAU,IAAI,CAAC,CACzD;CACA,OAAO;EACL,OAAO,eACL,OAAO,KAAK,UAAU,MAAM,KAAK,GACjC,IACF;EACA,QAAQ,eACN,OAAO,KAAK,UAAU,MAAM,MAAM,GAClC,IACF;EACA,WAAW,eAAe,WAAW,IAAI;EACzC,QAAQ,eAAe,QAAQ,IAAI;EACnC,YAAY,eAAe,YAAY,IAAI;EAC3C,QAAQ;GACN,OAAO,OAAO,QAAQ,OAAO,UAAU,QAAQ,MAAM,OAAO,CAAC;GAC7D,QAAQ,OAAO,QAAQ,OAAO,UAAU,QAAQ,MAAM,QAAQ,CAAC;GAC/D,WAAW,UAAU,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC;GAC9D,QAAQ,OAAO,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC;GACxD,YAAY,WAAW,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC;EAClE;CACF;AACF;AAEA,SAAS,oBAAoB,KAA2C;CACtE,MAAM,QAAQ,UAAU,KAAK,yBAAyB;CACtD,MAAM,SAAS,UAAU,KAAK,6BAA6B;CAC3D,MAAM,YAAY,UAAU,KAAK,4BAA4B;CAC7D,MAAM,SAAS,UAAU,KAAK,yBAAyB;CACvD,MAAM,aAAa,UAAU,KAAK,8BAA8B;CAChE,IACE,UAAU,KAAA,KACV,WAAW,KAAA,KACX,cAAc,KAAA,KACd,WAAW,KAAA,KACX,eAAe,KAAA,GAEf,OAAO,KAAA;CACT,OAAO;EACL,OAAO,SAAS;EAChB,QAAQ,UAAU;EAClB,GAAI,cAAc,KAAA,IAAY,EAAE,UAAU,IAAI,CAAC;EAC/C,GAAI,WAAW,KAAA,IAAY,EAAE,OAAO,IAAI,CAAC;EACzC,GAAI,eAAe,KAAA,IAAY,EAAE,WAAW,IAAI,CAAC;CACnD;AACF;AAEA,SAAS,UAAU,KAAgB,KAAiC;CAClE,MAAM,QAAQ,IAAI,QAAQ,IAAI;CAC9B,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,KAAK,SAAS,IAAI,QAAQ,KAAA;AACrF;AAEA,SAAS,wBAAwB,MAA0C;CACzE,MAAM,UAAiC;EACrC,UAAU;GAAE,GAAG;GAAG,UAAU;EAAE;EAC9B,WAAW;GAAE,GAAG;GAAG,UAAU;EAAE;EAC/B,YAAY,EAAE,GAAG,EAAE;EACnB,eAAe;CACjB;CACA,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,OAAO,IAAI;EACjB,IAAI,KAAK,SAAS,cAChB,QAAQ,WAAW,KAAK;OACnB;GACL,QAAQ,KAAK,KAAK,CAAC,KAAK;GACxB,QAAQ,KAAK,KAAK,CAAC,YAAY,KAAK;EACtC;CACF;CACA,MAAM,QAAQ,QAAQ,SAAS,IAAI,QAAQ,UAAU;CACrD,QAAQ,gBAAgB,KAAK,SAAS,IAAI,QAAQ,KAAK,SAAS;CAChE,OAAO;AACT;AAEA,SAAS,qBAAqB,MAAmB,YAA2C;CAC1F,MAAM,aAAa,WAAW,WAAW;CACzC,MAAM,QAAQ,WAAW,SAAS,IAAI,WAAW,UAAU;CAC3D,IAAI,eAAe,KAAK,QACtB,OAAO,+BAA+B,KAAK,OAAO;CAEpD,OAAO,2BAA2B,WAAW,GAAG,KAAK,OAAO,mDAAmD,MAAM,GAAG,KAAK,OAAO,aAAa,WAAW,SAAS,EAAE,aAAa,WAAW,UAAU,EAAE;AAC7M;;;;;;;AAQA,SAAS,sBACP,MACA,OACiC;CACjC,MAAM,yBAAS,IAAI,IAA0B;CAC7C,KAAK,MAAM,KAAK,MAAM;EACpB,IAAI,CAAC,cAAc,GAAG,KAAK,GAAG;EAC9B,MAAM,MACJ,EAAE,iBAAiB,KAAA,KAAa,EAAE,iBAAiB,YAAY,EAAE,eAAe;EAClF,OAAO,IAAI,MAAM,OAAO,IAAI,GAAG,KAAK,KAAK,CAAC;CAC5C;CACA,IAAI,OAAO,SAAS,GAAG,OAAO,KAAA;CAC9B,MAAM,IAAI,KAAK;CACf,OAAO,CAAC,GAAG,OAAO,QAAQ,CAAC,CAAC,CACzB,KAAK,CAAC,cAAc,YAAY;EAC/B;EACA;EACA,OAAO,IAAI,IAAI,QAAQ,IAAI;CAC7B,EAAE,CAAC,CACF,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,SAAS,EAAE,aAAa,cAAc,EAAE,YAAY,CAAC;AACrF;AAUA,SAAS,6BACP,SACA,UACA,OACA,aACmC;CACnC,IAAI,QAAQ,WAAW,KAAK,SAAS,WAAW,GAAG,OAAO,KAAA;CAE1D,MAAM,UAAuC,CAAC;CAC9C,MAAM,aAA8C,CAAC;CAErD,MAAM,mBAAmB,QACtB,KAAK,MAAM,YAAY,GAAG,KAAK,CAAC,CAAC,CACjC,OAAO,OAAO,QAAQ;CACzB,MAAM,oBAAoB,SACvB,KAAK,MAAM,YAAY,GAAG,KAAK,CAAC,CAAC,CACjC,OAAO,OAAO,QAAQ;CACzB,IAAI,iBAAiB,SAAS,KAAK,kBAAkB,SAAS,GAAG;EAC/D,QAAQ,YAAY,aAAa,mBAAmB,gBAAgB;EACpE,WAAW,YAAY;CACzB;CAEA,MAAM,cAAc,gBAAgB,OAAO;CAC3C,MAAM,eAAe,gBAAgB,QAAQ;CAC7C,IAAI,YAAY,SAAS,KAAK,aAAa,SAAS,GAAG;EACrD,QAAQ,OAAO,aAAa,cAAc,WAAW;EACrD,WAAW,OAAO;CACpB;CAEA,MAAM,aAAa,QAAQ,KAAK,MAAM,EAAE,MAAM,CAAC,CAAC,OAAO,OAAO,QAAQ;CACtE,MAAM,cAAc,SAAS,KAAK,MAAM,EAAE,MAAM,CAAC,CAAC,OAAO,OAAO,QAAQ;CACxE,IAAI,WAAW,SAAS,KAAK,YAAY,SAAS,GAAG;EACnD,QAAQ,WAAW,aAAa,aAAa,UAAU;EACvD,WAAW,WAAW;CACxB;CAEA,MAAM,aAAa,QAChB,KAAK,OAAO,EAAE,WAAW,SAAS,MAAM,EAAE,WAAW,UAAU,EAAE,CAAC,CAClE,OAAO,OAAO,QAAQ;CACzB,MAAM,cAAc,SACjB,KAAK,OAAO,EAAE,WAAW,SAAS,MAAM,EAAE,WAAW,UAAU,EAAE,CAAC,CAClE,OAAO,OAAO,QAAQ;CACzB,IAAI,WAAW,SAAS,KAAK,YAAY,SAAS,GAAG;EACnD,QAAQ,aAAa,aAAa,aAAa,UAAU;EACzD,WAAW,aAAa;CAC1B;CAKA,MAAM,cAAc,oBAAoB,OAAO;CAC/C,MAAM,eAAe,oBAAoB,QAAQ;CACjD,KAAK,MAAM,OAAO,OAAO,KAAK,WAAW,GAAG;EAC1C,MAAM,IAAI,aAAa;EACvB,MAAM,IAAI,YAAY;EACtB,IAAI,CAAC,KAAK,EAAE,WAAW,KAAK,CAAC,KAAK,EAAE,WAAW,GAAG;EAClD,QAAQ,OAAO,SAAS,aAAa,GAAG,CAAC;EACzC,WAAW,OAAO,SAAS;CAC7B;CAEA,MAAM,mBAA6B,CAAC;CACpC,MAAM,kBAA4B,CAAC;CACnC,KAAK,MAAM,CAAC,MAAM,UAAU,OAAO,QAAQ,OAAO,GAAG;EACnD,IAAI,CAAC,MAAM,aAAa;EAGxB,KAFY,WAAW,SAAS,wBACT,qBAAqB,MAAM,QAAQ,IAAI,MAAM,QAAQ,GAChE,gBAAgB,KAAK,IAAI;OAChC,iBAAiB,KAAK,IAAI;CACjC;CAEA,OAAO;EACL,WAAW,SAAS;EACpB,UAAU,QAAQ;EAClB,GAAI,cAAc,EAAE,YAAY,IAAI,CAAC;EACrC;EACA;EACA;CACF;AACF;AAEA,SAAS,gBAAgB,MAA6B;CACpD,OAAO,KACJ,QAAQ,QAAQ,IAAI,eAAe,SAAS,YAAY,CAAC,CACzD,KAAK,QAAQ,IAAI,OAAO,CAAC,CACzB,OAAO,cAAc;AAC1B;AAEA,SAAS,eAAe,OAAiC;CACvD,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK;AAC3D;;AAGA,SAAS,oBAAoB,MAA6C;CACxE,MAAM,MAAgC,CAAC;CACvC,KAAK,MAAM,KAAK,MAAM;EACpB,MAAM,SAAS,EAAE,QAAQ,aAAa;EACtC,IAAI,CAAC,QAAQ;EACb,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,MAAM,GAAG;GACjD,IAAI,CAAC,OAAO,SAAS,KAAK,GAAG;GAC7B,IAAI,CAAC,IAAI,MAAM,IAAI,OAAO,CAAC;GAC3B,IAAI,IAAI,CAAC,KAAK,KAAe;EAC/B;CACF;CACA,OAAO;AACT;;;AAIA,SAAS,aAAa,UAAoB,SAAgC;CACxE,MAAM,eAAe,KAAK,QAAQ;CAClC,MAAM,cAAc,KAAK,OAAO;CAChC,MAAM,cAAc,eAAe,UAAU,YAAY;CACzD,MAAM,aAAa,eAAe,SAAS,WAAW;CACtD,MAAM,YAAY,SAAS;CAC3B,MAAM,WAAW,QAAQ;CACzB,MAAM,QAAQ,cAAc;CAG5B,MAAM,KAAK,KAAK,KAAK,cAAc,YAAY,aAAa,QAAQ;CAIpE,MAAM,YAAY,QAAQ,KAAK,IAAI,KAAK;CACxC,MAAM,OAAyB,CAAC,QAAQ,WAAW,QAAQ,SAAS;CAGpE,MAAM,IAAI,KAAK,IAAI,QAAQ,KAAK;CAChC,MAAM,SAAS,KAAK,IAAI,KAAK,IAAI,UAAU,KAAK,IAAI,CAAC,CAAC,KAAK;CAG3D,MAAM,eAAe,KAAK,OACtB,YAAY,KAAK,eAAe,WAAW,KAAK,cAChD,KAAK,IAAI,GAAG,YAAY,WAAW,CAAC,CACxC;CACA,MAAM,UAAU,eAAe,IAAI,QAAQ,eAAe;CAK1D,OAAO;EACL,SAAS;EACT,UAAU;EACV;EACA;EACA;EACA;EACA;EACA;EACA,aAXkB,SAAS,OAAQ,KAAK,IAAI,OAAO,KAAK;CAY1D;AACF;AAEA,SAAS,eAAe,IAAc,QAAwB;CAC5D,IAAI,GAAG,SAAS,GAAG,OAAO;CAC1B,IAAI,IAAI;CACR,KAAK,MAAM,KAAK,IAAI,MAAM,IAAI,WAAW;CACzC,OAAO,KAAK,GAAG,SAAS;AAC1B;AAIA,SAAS,aACP,MACA,MACsB;CACtB,IAAI,SAAS,QAAQ,OAAO;CAE5B,OADmB,KAAK,MAAM,MAAM,OAAO,SAAS,mBAAmB,GAAG,SAAS,CAAC,CACpE,IAAI,YAAY;AAClC;;;;;;;AAQA,SAAS,YAAY,KAAgB,OAAqC;CAIxE,MAAM,QAAQ,mBAAmB,KAAK,KAAK;CAC3C,OAAO,OAAO,SAAS,KAAK,IAAK,QAAmB;AACtD;AAIA,SAAS,eACP,QACA,MACA,SACoB;CACpB,IAAI,OAAO,WAAW,GACpB,OAAO;EACL,GAAG;EACH,MAAM;EACN,KAAK;EACL,KAAK;EACL,QAAQ;EACR,KAAK;EACL,KAAK;EACL,WAAW,CAAC;CACd;CAEF,MAAM,SAAS,CAAC,GAAG,MAAM,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC/C,MAAM,IAAI,OAAO;CACjB,MAAM,OAAO,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CACjD,MAAM,WAAW,OAAO,QAAQ,GAAG,MAAM,KAAK,IAAI,SAAS,GAAG,CAAC,IAAI;CACnE,MAAM,SAAS,KAAK,KAAK,QAAQ;CACjC,MAAM,WAAW,UACb,CAAC,GAAG,OAAO,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK,CAAC,CAAC,MAAM,GAAG,KAAK,IAAI,GAAG,QAAQ,MAAM,CAAC,IACnF,KAAA;CACJ,OAAO;EACL;EACA;EACA,KAAK,WAAW,QAAQ,EAAG;EAC3B,KAAK,WAAW,QAAQ,GAAI;EAC5B;EACA,KAAK,OAAO;EACZ,KAAK,OAAO,IAAI;EAChB,WAAW,UAAU,QAAQ,IAAI;EACjC,GAAI,WAAW,EAAE,SAAS,IAAI,CAAC;CACjC;AACF;AAEA,SAAS,WAAW,QAAkB,GAAmB;CACvD,IAAI,OAAO,WAAW,GAAG,OAAO;CAChC,IAAI,OAAO,WAAW,GAAG,OAAO,OAAO;CACvC,MAAM,OAAO,OAAO,SAAS,KAAK;CAClC,MAAM,KAAK,KAAK,MAAM,GAAG;CACzB,MAAM,KAAK,KAAK,KAAK,GAAG;CACxB,IAAI,OAAO,IAAI,OAAO,OAAO;CAC7B,MAAM,IAAI,MAAM;CAChB,OAAO,OAAO,OAAQ,IAAI,KAAK,OAAO,MAAO;AAC/C;;;;AAKA,SAAS,UAAU,QAAkB,MAA+C;CAClF,IAAI,OAAO,WAAW,KAAK,OAAO,GAAG,OAAO,CAAC;CAC7C,MAAM,MAAM,OAAO;CACnB,MAAM,MAAM,OAAO,OAAO,SAAS;CACnC,IAAI,QAAQ,KAAK,OAAO,CAAC;EAAE,IAAI;EAAK,IAAI;EAAK,OAAO,OAAO;CAAO,CAAC;CACnE,MAAM,SAAS,MAAM,OAAO;CAC5B,MAAM,MAAuC,CAAC;CAC9C,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,KAAK;EAC7B,MAAM,KAAK,MAAM,IAAI;EACrB,MAAM,KAAK,MAAM,OAAO,IAAI,MAAM,KAAK;EACvC,IAAI,KAAK;GAAE;GAAI;GAAI,OAAO;EAAE,CAAC;CAC/B;CACA,KAAK,MAAM,KAAK,QAAQ;EACtB,MAAM,MAAM,KAAK,IAAI,OAAO,GAAG,KAAK,OAAO,IAAI,OAAO,KAAK,CAAC;EAC5D,IAAI,IAAI,CAAE;CACZ;CACA,OAAO;AACT;AAEA,SAAS,oBAAoB,MAAmB,MAAkD;CAKhG,MAAM,wBAAQ,IAAI,IAAsB;CACxC,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,SAAS,IAAI,QAAQ;EAC3B,IAAI,CAAC,QAAQ;EACb,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,OAAO,cAAc,CAAC,CAAC,GAAG;GAClE,IAAI,CAAC,OAAO,SAAS,KAAK,GAAG;GAC7B,MAAM,MAAM,MAAM,IAAI,GAAG,KAAK,CAAC;GAC/B,IAAI,KAAK,KAAK;GACd,MAAM,IAAI,KAAK,GAAG;EACpB;CACF;CACA,MAAM,MAA0C,CAAC;CACjD,KAAK,MAAM,CAAC,KAAK,WAAW,OAAO,IAAI,OAAO,eAAe,QAAQ,IAAI;CACzE,OAAO;AACT;AAIA,SAAS,qBAAqB,MAAiD;CAI7E,MAAM,MAAoC,CAAC;CAC3C,MAAM,0BAAU,IAAI,IAAsB;CAC1C,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,SAAS,IAAI,QAAQ;EAC3B,IAAI,CAAC,QAAQ,UAAU;EACvB,KAAK,MAAM,CAAC,SAAS,SAAS,OAAO,QAAQ,OAAO,QAAQ,GAAG;GAC7D,MAAM,YAAY,OAAO,OAAO,IAAI,CAAC,CAAC,OAAO,OAAO,QAAQ;GAC5D,IAAI,UAAU,WAAW,GAAG;GAC5B,MAAM,YAAY,UAAU,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,UAAU;GACnE,MAAM,MAAM,QAAQ,IAAI,OAAO,KAAK,CAAC;GACrC,IAAI,KAAK,SAAS;GAClB,QAAQ,IAAI,SAAS,GAAG;EAC1B;CACF;CACA,KAAK,MAAM,CAAC,SAAS,WAAW,SAC9B,IAAI,WAAW;EACb,GAAG,OAAO;EACV,WAAW,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;CACxD;CAEF,OAAO;AACT;AAIA,SAAS,kBACP,SAC+B;CAC/B,MAAM,wBAAQ,IAAI,IAAqD;CACvE,KAAK,MAAM,KAAK,SAAS;EACvB,IAAI,CAAC,OAAO,SAAS,EAAE,KAAK,GAAG;EAC/B,MAAM,OAAO,MAAM,IAAI,EAAE,KAAK,KAAK,CAAC;EACpC,KAAK,KAAK;GAAE,OAAO,EAAE;GAAO,OAAO,EAAE;EAAM,CAAC;EAC5C,MAAM,IAAI,EAAE,OAAO,IAAI;CACzB;CACA,MAAM,SAAS,IAAI,IAAI,QAAQ,KAAK,MAAM,EAAE,KAAK,CAAC;CAClD,MAAM,eAAyB,CAAC;CAChC,KAAK,MAAM,CAAC,OAAO,iBAAiB,OAAO;EACzC,MAAM,OAAO,IAAI,IAAI,aAAa,KAAK,MAAM,EAAE,KAAK,CAAC;EACrD,IAAI,MAAM;EACV,KAAK,MAAM,KAAK,QAAQ,IAAI,CAAC,KAAK,IAAI,CAAC,GAAG,MAAM;EAChD,IAAI,KAAK,aAAa,KAAK,KAAK;CAClC;CACA,IAAI,OAAO,OAAO,KAAK,aAAa,WAAW,GAAG,OAAO,KAAA;CAEzD,MAAM,YAAY,CAAC,GAAG,MAAM,CAAC,CAAC,KAAK;CACnC,MAAM,UAAkC,CAAC;CACzC,KAAK,IAAI,IAAI,GAAG,IAAI,UAAU,QAAQ,KACpC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,UAAU,QAAQ,KAAK;EAC7C,MAAM,IAAI,UAAU;EACpB,MAAM,IAAI,UAAU;EACpB,MAAM,UAAoB,CAAC;EAC3B,MAAM,UAAoB,CAAC;EAC3B,KAAK,MAAM,SAAS,cAAc;GAChC,MAAM,eAAe,MAAM,IAAI,KAAK;GACpC,MAAM,KAAK,aAAa,MAAM,MAAM,EAAE,UAAU,CAAC,CAAC,EAAE;GACpD,MAAM,KAAK,aAAa,MAAM,MAAM,EAAE,UAAU,CAAC,CAAC,EAAE;GACpD,IAAI,OAAO,KAAA,KAAa,OAAO,KAAA,GAAW;IACxC,QAAQ,KAAK,EAAE;IACf,QAAQ,KAAK,EAAE;GACjB;EACF;EACA,MAAM,YAAY,oBAChB,QAAQ,KAAK,OAAO,UAAU,CAAC,OAAO,QAAQ,MAAO,CAAC,GACtD,EAAE,WAAW,EAAE,CACjB;EACA,QAAQ,GAAG,EAAE,IAAI,OAAO,UAAU;CACpC;CAMF,MAAM,YAAY,oBAJH,aAAa,KAAK,UAAU;EACzC,MAAM,iBAAiB,IAAI,IAAI,MAAM,IAAI,KAAK,CAAC,CAAE,KAAK,WAAW,CAAC,OAAO,OAAO,OAAO,KAAK,CAAC,CAAC;EAC9F,OAAO,UAAU,KAAK,UAAU,eAAe,IAAI,KAAK,CAAE;CAC5D,CAC2C,GAAG,EAAE,WAAW,EAAE,CAAC;CAE9D,MAAM,oBAAoB,aACvB,KAAK,UAAU;EACd,MAAM,eAAe,MAAM,IAAI,KAAK;EACpC,MAAM,SAAS,aAAa,KAAK,MAAM,EAAE,KAAK;EAE9C,OAAO;GAAE;GAAO,SAAS;GAAc,OADzB,KAAK,IAAI,GAAG,MAAM,IAAI,KAAK,IAAI,GAAG,MAAM;EACT;CAC/C,CAAC,CAAC,CACD,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK,CAAC,CACjC,MAAM,GAAG,EAAE;CAEd,OAAO;EACL,QAAQ,OAAO;EACf,cAAc,aAAa;EAC3B,OAAO,OAAO,SAAS,UAAU,aAAa,IAAI,UAAU,gBAAgB;EAC5E,KAAK,UAAU;EACf,SAAS,UAAU;EACnB,UAAU,UAAU;EACpB;EACA;CACF;AACF;AAIA,SAAS,YACP,MACA,YACA,aACA,OACyB;CACzB,IAAI,MAAM;CACV,IAAI,MAAM;CACV,IAAI,CAAC,OAAO,CAAC,KAAK;EAGhB,MAAM,MAAM,CAAC,GAAG,IAAI,IAAI,KAAK,KAAK,MAAM,EAAE,WAAW,CAAC,CAAC;EACvD,IAAI,IAAI,WAAW,GAAG,OAAO,KAAA;EAC7B,MAAM,CAAC,KAAK,OAAO;EACnB,MAAM,UAAU,sBACd,KAAK,QAAQ,QAAQ,IAAI,gBAAgB,GAAG,GAC5C,KACF;EACA,MAAM,UAAU,sBACd,KAAK,QAAQ,QAAQ,IAAI,gBAAgB,GAAG,GAC5C,KACF;EACA,IAAI,QAAQ,WAAW,KAAK,QAAQ,WAAW,GAAG,OAAO,KAAA;EACzD,MAAM,QAAQ,KAAK,OAAO;EAC1B,MAAM,QAAQ,KAAK,OAAO;EAC1B,MAAM,SAAS,QAAQ,MAAM;EAC7B,MAAM,SAAS,QAAQ,MAAM;CAC/B;CAEA,MAAM,WAAW,KAAK,QAAQ,MAAM,EAAE,gBAAgB,GAAG;CACzD,MAAM,YAAY,KAAK,QAAQ,MAAM,EAAE,gBAAgB,GAAG;CAC1D,IAAI,SAAS,WAAW,KAAK,UAAU,WAAW,GAAG,OAAO,KAAA;CAI5D,MAAM,UAAU,eAFO,SAAS,QAAQ,QAAQ,OAAO,SAAS,YAAY,KAAK,KAAK,CAAC,CAE3C,GADpB,UAAU,QAAQ,QAAQ,OAAO,SAAS,YAAY,KAAK,KAAK,CAAC,CAC5B,CAAC;CAC9D,MAAM,iBAAiB,QAAQ,MAAM,KAAK,SAAS,YAAY,KAAK,UAAU,KAAK,CAAC;CACpF,MAAM,kBAAkB,QAAQ,MAAM,KAAK,SAAS,YAAY,KAAK,WAAW,KAAK,CAAC;CACtF,IAAI,eAAe,WAAW,GAAG,OAAO,KAAA;CAExC,MAAM,eAAe,KAAK,cAAc;CACxC,MAAM,gBAAgB,KAAK,eAAe;CAC1C,MAAM,QAAQ,gBAAgB;CAE9B,MAAM,YAAY,gBAAgB,gBAAgB,iBAAiB;EACjE,YAAY;EACZ,WAAW;EACX,WAAW;CACb,CAAC;CACD,MAAM,QAAQ,YAAY,gBAAgB,eAAe;CACzD,MAAM,IAAI,eAAe,gBAAgB,eAAe;CACxD,MAAM,MAAM,UAAU;EAAE,SAAS,eAAe;EAAQ,OAAO;EAAK,OAAO;CAAK,CAAC;CACjF,MAAM,YACJ,MAAM,QAAQ,MAAM,IAChB,OACA,yBAAyB;EACvB,QAAQ,KAAK,IAAI,CAAC;EAClB,OAAO;EACP,OAAO;CACT,CAAC;CAEP,OAAO;EACL;EACA;EACA;EACA,MAAM,CAAC,UAAU,KAAK,UAAU,IAAI;EACpC,QAAQ,MAAM;EACd,GAAG,eAAe;EAClB,kBAAkB,QAAQ,iBAAiB;EAC3C,mBAAmB,QAAQ,kBAAkB;EAC7C,SAAS;EACT;EACA;CACF;AACF;AAEA,SAAS,KAAK,KAAuB;CACnC,OAAO,IAAI,WAAW,IAAI,IAAI,IAAI,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,IAAI;AACrE;AAIA,eAAe,uBACb,MACA,SACA,OAC4C;CAC5C,MAAM,SAAS,KAAK,QAAQ,QAAQ,cAAc,KAAK,KAAK,CAAC;CAC7D,IAAI,OAAO,WAAW,GAAG,OAAO;EAAE,UAAU,CAAC;EAAG,eAAe;CAAE;CAEjE,MAAM,2BAAW,IAAI,IAAoD;CACzE,KAAK,MAAM,OAAO,QAChB,IAAI;EAIF,MAAM,SAAS,MAAM,QAAQ,IAAI,IAAI,OAAO,EAAE,WAAW,IAAI,CAAC;EAC9D,KAAK,MAAM,WAAW,OAAO,UAA8B;GACzD,MAAM,MAAM,QAAQ,QAAQ,QAAQ,cAAc;GAClD,MAAM,IAAI,SAAS,IAAI,GAAG,KAAK;IAAE,WAAW,CAAC;IAAG,OAAO;GAAE;GACzD,IAAI,EAAE,UAAU,SAAS,GAAG,EAAE,UAAU,KAAK,IAAI,KAAK;GACtD,SAAS,IAAI,KAAK,CAAC;EACrB;CACF,QAAQ;EACN,MAAM,IAAI,SAAS,IAAI,eAAe,KAAK;GAAE,WAAW,CAAC;GAAG,OAAO;EAAE;EACrE,IAAI,EAAE,UAAU,SAAS,GAAG,EAAE,UAAU,KAAK,IAAI,KAAK;EACtD,SAAS,IAAI,iBAAiB,CAAC;CACjC;CAEF,MAAM,cAAc,CAAC,GAAG,SAAS,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,IAAI,QAAQ;EAC5D;EACA,MAAM;EACN,OAAO,EAAE,UAAU,SAAS,OAAO;EACnC,WAAW,EAAE;CACf,EAAE;CACF,YAAY,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK;CAC5C,OAAO;EAAE,UAAU;EAAa,eAAe,OAAO;CAAO;AAC/D;AAEA,SAAS,sBAAsB,MAA4B,OAAuC;CAChG,OAAO,KAAK,KAAK,QAAQ,YAAY,KAAK,KAAK,CAAC,CAAC,CAAC,OAAO,OAAO,QAAQ;AAC1E;AAEA,SAAS,cAAc,KAAgB,OAAsC;CAC3E,IAAI,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,WAAW,OAAO;CAC7E,MAAM,QAAQ,YAAY,KAAK,KAAK;CACpC,OAAO,OAAO,SAAS,KAAK,KAAK,QAAQ;AAC3C;AAIA,SAAS,qBACP,MACA,UACgC;CAChC,IAAI,QAAQ;CACZ,MAAM,UAAqE,CAAC;CAC5E,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,SAAS,gBAAgB,GAAG;EAClC,IAAI,CAAC,QAAQ;EACb,MAAM,YAAY,cAAc,QAAQ,QAAQ;EAChD,KAAK,MAAM,QAAQ,WAAW;GAC5B;GACA,QAAQ,KAAK;IAAE,OAAO,IAAI;IAAO,QAAQ,KAAK;IAAQ,SAAS,KAAK;GAAS,CAAC;EAChF;CACF;CACA,OAAO;EAAE;EAAO,oBAAoB,UAAU;EAAG;CAAQ;AAC3D;AAEA,SAAS,gBAAgB,KAAoC;CAO3D,MAAM,WAAY,IAA0D;CAC5E,IAAI,OAAO,UAAU,WAAW,UAAU,OAAO,SAAS;CAC1D,IAAI,OAAO,UAAU,SAAS,UAAU,OAAO,SAAS;AAE1D;AAIA,SAAS,0BACP,MACA,SACA,OACuC;CACvC,MAAM,KAAe,CAAC;CACtB,MAAM,KAAe,CAAC;CACtB,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,IAAI,QAAQ,aAAa,IAAI;EACnC,IAAI,MAAM,KAAA,KAAa,CAAC,OAAO,SAAS,CAAC,GAAG;EAC5C,MAAM,IAAI,YAAY,KAAK,KAAK;EAChC,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG;EACzB,GAAG,KAAK,CAAC;EACT,GAAG,KAAK,CAAC;CACX;CACA,IAAI,GAAG,SAAS,GAAG,OAAO,KAAA;CAE1B,MAAM,IAAI,SAAS,IAAI,EAAE;CACzB,MAAM,IAAI,UAAU,IAAI,EAAE;CAC1B,MAAM,QAAQ,KAAK,EAAE;CACrB,MAAM,QAAQ,KAAK,EAAE;CACrB,IAAI,MAAM;CACV,IAAI,QAAQ;CACZ,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,QAAQ,KAAK;EAClC,QAAQ,GAAG,KAAM,UAAU,GAAG,KAAM;EACpC,UAAU,GAAG,KAAM,UAAU;CAC/B;CACA,MAAM,QAAQ,UAAU,IAAI,IAAI,MAAM;CACtC,MAAM,YAAY,QAAQ,QAAQ;CAClC,MAAM,QAAQ,GAAG,QAAQ,GAAG,MAAM,KAAK,IAAI,UAAU,GAAG,CAAC;CACzD,MAAM,QAAQ,GAAG,QAAQ,GAAG,GAAG,MAAM,KAAK,KAAK,YAAY,QAAQ,GAAG,QAAS,GAAG,CAAC;CACnF,MAAM,KAAK,UAAU,IAAI,IAAI,IAAI,QAAQ;CAEzC,OAAO;EACL,QAAQ,QAAQ;EAChB,GAAG,GAAG;EACN,SAAS;EACT,UAAU;EACV,aAAa;GAAE;GAAW;GAAO;EAAG;CACtC;AACF;AAIA,SAAS,sBACP,WACA,MACA,eAC0B;CAO1B,MAAM,OAAyC,CAAC;CAChD,MAAM,WACJ,SAAS,KAAA,IACJ,kBACD,KAAK,KAAK,KAAK,IACZ,SACD,KAAK,QAAQ,IACV,SACA;CACX,KAAK,KAAK;EACR,MAAM;EACN,QAAQ;EACR,QAAQ,OACJ,SAAS,KAAK,MAAM,QAAQ,CAAC,EAAE,UAAU,KAAK,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,IAAI,KAAK,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,OAAO,KAAK,MACzG;CACN,CAAC;CACD,MAAM,aACJ,kBAAkB,KAAA,IACb,kBACD,cAAc,UAAU,IACrB,SACA;CACT,KAAK,KAAK;EACR,MAAM;EACN,QAAQ;EACR,QAAQ,gBAAgB,GAAG,cAAc,MAAM,mBAAmB;CACpE,CAAC;CACD,KAAK,KACH,UAAU,MAAM,IACZ;EACE,MAAM;EACN,QAAQ;EACR,QAAQ;CACV,IACA;EACE,MAAM;EACN,QACE,UAAU,SAAS,QAAQ,UAAU,QAAQ,KACzC,SACA,UAAU,SAAS,QAAQ,UAAU,QAAQ,KAC3C,SACA;EACR,QACE,UAAU,SAAS,QAAQ,UAAU,QAAQ,QAAQ,UAAU,QAAQ,OACnE,uDACA,QAAQ,UAAU,KAAK,QAAQ,CAAC,EAAE,QAAQ,UAAU,IAAI,QAAQ,CAAC,EAAE,QAAQ,UAAU,IAAI,QAAQ,CAAC,EAAE,UAAU,UAAU;CAChI,CACN;CAMA,OAAO;EACL,QANa,KAAK,MAAM,MAAM,EAAE,WAAW,MAAM,IAC/C,SACA,KAAK,MAAM,MAAM,EAAE,WAAW,UAAU,EAAE,WAAW,eAAe,IAClE,SACA;EAGJ;EACA,QAAQ,CAAC;CACX;AACF;AAiBA,SAAS,qBAAqB,KAA8C;CAC1E,MAAM,MAAwB,CAAC;CAI/B,IAAI,IAAI,uBAAuB;EAC7B,MAAM,MAAM,IAAI;EAChB,MAAM,QAAQ,IAAI,eAAe;EACjC,KAAK,MAAM,QAAQ,IAAI,kBAAkB;GACvC,MAAM,IAAI,IAAI,QAAQ;GACtB,IAAI,CAAC,GAAG;GACR,IAAI,KAAK;IACP,UAAU;IACV,MAAM;IACN,OAAO,GAAG,KAAK,kBAAkB,EAAE,SAAS,QAAQ,CAAC,EAAE,KAAK,EAAE,QAAQ,QAAQ,CAAC,EAAE,MAAM;IACvF,QAAQ,iBAAiB,EAAE,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,IAAI,EAAE,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,OAAO,EAAE,OAAO,QAAQ,CAAC,EAAE,cAAc,EAAE,QAAQ,QAAQ,CAAC,EAAE,cAAc,EAAE,SAAS,eAAe,EAAE,UAAU;IACzL,cAAc,iCAAiC;GACjD,CAAC;EACH;EACA,KAAK,MAAM,QAAQ,IAAI,iBAAiB;GACtC,MAAM,IAAI,IAAI,QAAQ;GACtB,IAAI,CAAC,GAAG;GACR,IAAI,KAAK;IACP,UAAU;IACV,MAAM;IACN,OAAO,GAAG,KAAK,iBAAiB,EAAE,SAAS,QAAQ,CAAC,EAAE,KAAK,EAAE,QAAQ,QAAQ,CAAC,EAAE,MAAM;IACtF,QAAQ,iBAAiB,EAAE,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,IAAI,EAAE,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,OAAO,EAAE,OAAO,QAAQ,CAAC,EAAE,cAAc,EAAE,QAAQ,QAAQ,CAAC,EAAE,cAAc,EAAE,SAAS,eAAe,EAAE,UAAU;IACzL,cAAc,iCAAiC;GACjD,CAAC;EACH;CACF;CAKA,IACE,IAAI,UAAU,IAAI,KAClB,IAAI,UAAU,SAAS,QACvB,IAAI,UAAU,QAAQ,QACtB,IAAI,UAAU,QAAQ,MAElB;MAAA,IAAI,UAAU,OAAO,IAAK;GAC5B,MAAM,OAAO,IAAI,UAAU,YAAY,CAAC;GACxC,MAAM,QAAQ,KACX,MAAM,GAAG,CAAC,CAAC,CACX,KAAK,MAAM,GAAG,EAAE,MAAM,GAAG,EAAE,MAAM,QAAQ,CAAC,GAAG,CAAC,CAC9C,KAAK,IAAI;GACZ,IAAI,KAAK;IACP,UAAU;IACV,MAAM;IACN,OAAO,kBAAkB,IAAI,UAAU,KAAK,QAAQ,CAAC,EAAE;IACvD,QACE,KAAK,SAAS,IACV,SAAS,KAAK,OAAO,MAAM,KAAK,WAAW,IAAI,KAAK,IAAI,qBAAqB,MAAM,kBAAkB,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE,QAAQ,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE,KACvK,iBAAiB,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE,QAAQ,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE;IACzF,cAAc;GAChB,CAAC;EACH,OAAO,IAAI,IAAI,UAAU,OAAO,IAAK;GACnC,MAAM,OAAO,IAAI,UAAU,YAAY,CAAC;GACxC,MAAM,QAAQ,KACX,MAAM,GAAG,CAAC,CAAC,CACX,KAAK,MAAM,GAAG,EAAE,MAAM,GAAG,EAAE,MAAM,QAAQ,CAAC,GAAG,CAAC,CAC9C,KAAK,IAAI;GACZ,IAAI,KAAK;IACP,UAAU;IACV,MAAM;IACN,OAAO,kBAAkB,IAAI,UAAU,KAAK,QAAQ,CAAC,EAAE;IACvD,QACE,KAAK,SAAS,IACV,SAAS,KAAK,OAAO,MAAM,KAAK,WAAW,IAAI,KAAK,IAAI,IAAI,MAAM,kBAAkB,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE,QAAQ,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE,KACtJ,iBAAiB,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE,QAAQ,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE;IACzF,cAAc;GAChB,CAAC;EACH;;CAKF,IAAI,IAAI,kBAAkB,IAAI,eAAe,SAAS,GAAG;EACvD,MAAM,MAAM,IAAI,eAAe;EAC/B,IAAI,IAAI,SAAS,KAAK,IAAI,SAAS,KACjC,IAAI,KAAK;GACP,UAAU,IAAI,SAAS,MAAO,SAAS;GACvC,MAAM;GACN,OAAO,IAAI,IAAI,aAAa,oCAAoC,IAAI,MAAM,UAAU,IAAI,QAAQ,IAAA,CAAK,QAAQ,CAAC,EAAE;GAChH,QAAQ,4FAA4F,IAAI,MAAM,MAAM,IAAI,UAAU,EAAE,qBAAqB,IAAI,aAAa,GAAG,IAAI,eAAe,SAAS,IAAI,YAAY,IAAI,eAAe,EAAE,CAAE,aAAa,KAAK,IAAI,eAAe,EAAE,CAAE,MAAM,KAAK,GAAG;GACvS,cAAc;EAChB,CAAC;CAEL;CAKA,IAAI,OAAO,KAAK,IAAI,MAAM,CAAC,CAAC,WAAW,KAAK,IAAI,UAAU,IAAI,GAC5D,IAAI,KAAK;EACP,UAAU;EACV,MAAM;EACN,OAAO;EACP,QACE;EACF,cAAc;CAChB,CAAC;CAGH,IAAI,IAAI,MAAM;EACZ,MAAM,eACJ,IAAI,KAAK,YAAY,OAAO,oCAAoC,IAAI,KAAK,QAAQ,QAAQ,CAAC;EAC5F,MAAM,eACJ,IAAI,KAAK,cAAc,OAAO,kBAAkB,IAAI,IAAI,KAAK,UAAU;EACzE,MAAM,WAAW,IAAI,KAAK,KAAK,KAAK,IAAI;EACxC,MAAM,eAAe,IAAI,KAAK,KAAK,MAAM,IAAI,aAAa,IAAI,KAAK,KAAK,KAAK,IAAI;EACjF,IAAI,UACF,IAAI,KAAK;GACP,UAAU;GACV,MAAM;GACN,OAAO,eAAe,IAAI,KAAK,MAAM,QAAQ,CAAC,EAAE,WAAW,IAAI,KAAK,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,IAAI,IAAI,KAAK,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE;GACvH,QAAQ,kCAAkC,IAAI,UAAU,oCAAoC,IAAI,KAAK,EAAE,MAAM,IAAI,KAAK,OAAO,QAAQ,CAAC,EAAE,aAAa,aAAa;GAClK,cAAc;EAChB,CAAC;OACI,IAAI,cACT,IAAI,KAAK;GACP,UAAU;GACV,MAAM;GACN,OAAO,qCAAqC,aAAa,SAAS,IAAI,KAAK,EAAE;GAC7E,QAAQ,uDAAuD,IAAI,KAAK,IAAI,QAAQ,CAAC,EAAE,sBAAsB,IAAI,KAAK,MAAM,QAAQ,CAAC,EAAE;GACvI,cAAc;EAChB,CAAC;OAED,IAAI,KAAK;GACP,UAAU;GACV,MAAM;GACN,OAAO,8BAA8B,IAAI,KAAK,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,4BAA4B,IAAI;GACjG,QAAQ;GACR,cAAc;EAChB,CAAC;CAEL;CAEA,IAAI,IAAI,iBAAiB,IAAI,cAAc,QAAQ,GACjD,IAAI,KAAK;EACP,UAAU;EACV,MAAM;EACN,OAAO,GAAG,IAAI,cAAc,MAAM,cAAc,IAAI,cAAc,UAAU,IAAI,KAAK,IAAI;EACzF,QAAQ;EACR,cAAc;CAChB,CAAC;CAGH,IAAI,IAAI,cAAc,IAAI,WAAW,QAAQ,IAC3C,IAAI,KAAK;EACP,UAAU;EACV,MAAM;EACN,OAAO,8BAA8B,IAAI,WAAW,MAAM,QAAQ,CAAC,EAAE;EACrE,QACE;EACF,cAAc;CAChB,CAAC;CAGH,IAAI,IAAI,mBAAmB,IAAI,gBAAgB,SAAS,SAAS,GAAG;EAClE,MAAM,MAAM,IAAI,gBAAgB,SAAS;EACzC,IAAI,KAAK;GACP,UAAU;GACV,MAAM;GACN,OAAO,wBAAwB,IAAI,KAAK,KAAK,IAAI,QAAQ,IAAA,CAAK,QAAQ,CAAC,EAAE;GACzE,QAAQ,GAAG,IAAI,gBAAgB,cAAc,2CAA2C,IAAI,UAAU,OAAO,oBAAoB,IAAI,KAAK;GAC1I,cAAc;EAChB,CAAC;CACH;CAEA,IAAI,IAAI,sBAAsB,KAAK,IAAI,IAAI,mBAAmB,QAAQ,IAAI,IACxE,IAAI,KAAK;EACP,UAAU;EACV,MAAM;EACN,OAAO,+BAA+B,IAAI,mBAAmB,OAAO,eAAe,IAAI,mBAAmB,SAAS,QAAQ,CAAC,EAAE;EAC9H,QAAQ,yFAAyF,IAAI,mBAAmB,OAAO,0CAA0C,IAAI,mBAAmB,OAAO;EACvM,cAAc;CAChB,CAAC;CAGH,OAAO;AACT"}