@tangle-network/agent-eval 0.135.0 → 0.135.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. package/CHANGELOG.md +24 -0
  2. package/dist/analyst/index.js +3 -3
  3. package/dist/{analyze-runs-DMo3Lb_y.d.ts → analyze-runs-Cda5Xkj1.d.ts} +3 -3
  4. package/dist/{analyze-runs-DMo3Lb_y.d.ts.map → analyze-runs-Cda5Xkj1.d.ts.map} +1 -1
  5. package/dist/{analyze-runs-qk8op0tN.js → analyze-runs-jjCmF8pU.js} +14 -7
  6. package/dist/analyze-runs-jjCmF8pU.js.map +1 -0
  7. package/dist/{baseline-BaPxoROc.js → baseline-BUeFcgrn.js} +2 -2
  8. package/dist/{baseline-BaPxoROc.js.map → baseline-BUeFcgrn.js.map} +1 -1
  9. package/dist/benchmarks/index.d.ts +1 -1
  10. package/dist/benchmarks/index.js +1 -1
  11. package/dist/{benchmarks-Dw1Wv_JQ.js → benchmarks-Mtu251Jz.js} +3 -3
  12. package/dist/{benchmarks-Dw1Wv_JQ.js.map → benchmarks-Mtu251Jz.js.map} +1 -1
  13. package/dist/builder-eval/index.js +1 -1
  14. package/dist/campaign/index.d.ts +2 -2
  15. package/dist/campaign/index.js +2 -2
  16. package/dist/{campaign-B1c1T0kv.js → campaign-RVIqtJh0.js} +7 -7
  17. package/dist/{campaign-B1c1T0kv.js.map → campaign-RVIqtJh0.js.map} +1 -1
  18. package/dist/cli.js +1 -1
  19. package/dist/{client-BIyh1RCr.d.ts → client-DcvgkaZi.d.ts} +13 -4
  20. package/dist/client-DcvgkaZi.d.ts.map +1 -0
  21. package/dist/contract/index.d.ts +3 -3
  22. package/dist/contract/index.d.ts.map +1 -1
  23. package/dist/contract/index.js +19 -7
  24. package/dist/contract/index.js.map +1 -1
  25. package/dist/{cost-ledger-ZAa_P4r0.js → cost-ledger-DHAjwNj7.js} +6 -2
  26. package/dist/{cost-ledger-ZAa_P4r0.js.map → cost-ledger-DHAjwNj7.js.map} +1 -1
  27. package/dist/{default-registry-CFUZyNeZ.js → default-registry-BAhV-lbE.js} +3 -3
  28. package/dist/{default-registry-CFUZyNeZ.js.map → default-registry-BAhV-lbE.js.map} +1 -1
  29. package/dist/{eval-campaign-CHFxPTVl.js → eval-campaign-Cc8WZJ6b.js} +3 -3
  30. package/dist/{eval-campaign-CHFxPTVl.js.map → eval-campaign-Cc8WZJ6b.js.map} +1 -1
  31. package/dist/fuzz.js +1 -1
  32. package/dist/hosted/index.d.ts +1 -1
  33. package/dist/{index-BoJNQR6n.d.ts → index-B4Fjfo5U.d.ts} +93 -15
  34. package/dist/index-B4Fjfo5U.d.ts.map +1 -0
  35. package/dist/{index-C21xKtxu.d.ts → index-CQsJcqch.d.ts} +3 -3
  36. package/dist/{index-C21xKtxu.d.ts.map → index-CQsJcqch.d.ts.map} +1 -1
  37. package/dist/{index-DSC51roc2.d.ts → index-DSC51roc.d.ts} +1 -1
  38. package/dist/index-DSC51roc.d.ts.map +1 -0
  39. package/dist/index.d.ts +41 -10
  40. package/dist/index.d.ts.map +1 -1
  41. package/dist/index.js +135 -57
  42. package/dist/index.js.map +1 -1
  43. package/dist/{llm-client-BNcP4v08.js → llm-client-DHx8pzyJ.js} +2 -2
  44. package/dist/{llm-client-BNcP4v08.js.map → llm-client-DHx8pzyJ.js.map} +1 -1
  45. package/dist/matrix/index.d.ts +1 -1
  46. package/dist/meta-eval/index.d.ts +1 -2
  47. package/dist/meta-eval/index.d.ts.map +1 -1
  48. package/dist/meta-eval/index.js +2 -2
  49. package/dist/multishot/index.d.ts +1 -1
  50. package/dist/openapi.json +1 -1
  51. package/dist/{paired-arms-CA_8pN01.js → paired-arms-BbFKrAU-.js} +2 -2
  52. package/dist/{paired-arms-CA_8pN01.js.map → paired-arms-BbFKrAU-.js.map} +1 -1
  53. package/dist/pipelines/index.js +2 -2
  54. package/dist/{release-report-BVZBmRZp.js → release-report-DooPguBc.js} +4 -3
  55. package/dist/{release-report-BVZBmRZp.js.map → release-report-DooPguBc.js.map} +1 -1
  56. package/dist/{release-report-CuULWKyk.d.ts → release-report-DpBxGGI1.d.ts} +2 -2
  57. package/dist/{release-report-CuULWKyk.d.ts.map → release-report-DpBxGGI1.d.ts.map} +1 -1
  58. package/dist/reporting.d.ts +3 -3
  59. package/dist/reporting.js +4 -4
  60. package/dist/{researcher-DVtruQ9U.d.ts → researcher-Doo95b50.d.ts} +2 -2
  61. package/dist/{researcher-DVtruQ9U.d.ts.map → researcher-Doo95b50.d.ts.map} +1 -1
  62. package/dist/{reward-hacking-DCdRK9TY.js → reward-hacking-a-kYs0-i.js} +2 -2
  63. package/dist/{reward-hacking-DCdRK9TY.js.map → reward-hacking-a-kYs0-i.js.map} +1 -1
  64. package/dist/rl.d.ts +45 -3
  65. package/dist/rl.d.ts.map +1 -1
  66. package/dist/rl.js +108 -22
  67. package/dist/rl.js.map +1 -1
  68. package/dist/{rubric-predictive-validity-D6Q6n9oq.js → rubric-predictive-validity-BJf-8ejY.js} +2 -2
  69. package/dist/{rubric-predictive-validity-D6Q6n9oq.js.map → rubric-predictive-validity-BJf-8ejY.js.map} +1 -1
  70. package/dist/{semantic-concept-judge-Ca--u10C.js → semantic-concept-judge-Btozx3Vc.js} +3 -3
  71. package/dist/{semantic-concept-judge-Ca--u10C.js.map → semantic-concept-judge-Btozx3Vc.js.map} +1 -1
  72. package/dist/{server-Dc_lsOYd.js → server-Bz3WQJs6.js} +3 -3
  73. package/dist/{server-Dc_lsOYd.js.map → server-Bz3WQJs6.js.map} +1 -1
  74. package/dist/{skillopt-optimization-method-Cl4XPkLC.js → skillopt-optimization-method-0UmPD6aP.js} +342 -56
  75. package/dist/skillopt-optimization-method-0UmPD6aP.js.map +1 -0
  76. package/dist/{skillopt-optimization-method-DJ3l4w8W.d.ts → skillopt-optimization-method-CwSYkv35.d.ts} +39 -9
  77. package/dist/skillopt-optimization-method-CwSYkv35.d.ts.map +1 -0
  78. package/dist/{statistics-D_4Snl-5.d.ts → statistics-CKOqre5S.d.ts} +329 -3
  79. package/dist/statistics-CKOqre5S.d.ts.map +1 -0
  80. package/dist/{statistics-RwRNu2__.js → statistics-CnGCLLqc.js} +315 -2
  81. package/dist/statistics-CnGCLLqc.js.map +1 -0
  82. package/dist/{summary-report-BxtossFi.js → summary-report-BEk8OFLs.js} +11 -6
  83. package/dist/summary-report-BEk8OFLs.js.map +1 -0
  84. package/dist/{summary-report-DGp0-_XO.d.ts → summary-report-CPMINBqs.d.ts} +182 -7
  85. package/dist/summary-report-CPMINBqs.d.ts.map +1 -0
  86. package/dist/wire/index.js +1 -1
  87. package/package.json +1 -1
  88. package/dist/analyze-runs-qk8op0tN.js.map +0 -1
  89. package/dist/client-BIyh1RCr.d.ts.map +0 -1
  90. package/dist/index-BoJNQR6n.d.ts.map +0 -1
  91. package/dist/index-DSC51roc2.d.ts.map +0 -1
  92. package/dist/judge-calibration-DFtEMlde.d.ts +0 -146
  93. package/dist/judge-calibration-DFtEMlde.d.ts.map +0 -1
  94. package/dist/skillopt-optimization-method-Cl4XPkLC.js.map +0 -1
  95. package/dist/skillopt-optimization-method-DJ3l4w8W.d.ts.map +0 -1
  96. package/dist/statistics-D_4Snl-5.d.ts.map +0 -1
  97. package/dist/statistics-RwRNu2__.js.map +0 -1
  98. package/dist/summary-report-BxtossFi.js.map +0 -1
  99. package/dist/summary-report-DGp0-_XO.d.ts.map +0 -1
package/CHANGELOG.md CHANGED
@@ -4,6 +4,30 @@ All notable changes to `@tangle-network/agent-eval` and its sibling `agent-eval-
4
4
 
5
5
  ---
6
6
 
7
+ ## [0.135.2] - 2026-07-29 - correct paired promotion decisions
8
+
9
+ ### Fixed
10
+
11
+ - Paired promotion paths now share one decision function.
12
+ Binary outcomes use an interval that remains valid at the configured margin, plus an exact test.
13
+ Continuous outcomes reject zero-width samples.
14
+ This prevents real pass/fail gains and regressions from hiding at `[0, 0]`, and stops constant samples from being treated as certainty.
15
+ - `runRLCampaign()` now computes interim confidence only when the declared fraction of paired cells has usable scores on both arms.
16
+ Failed runs count toward coverage, and `RLCampaignResult.deltaCoverage` reports every missing or unscored cell.
17
+
18
+ ### Changed
19
+
20
+ - `sequential.minDeltaCoverage` defaults to `1`.
21
+ Set a lower value explicitly when a campaign may accept incomplete paired results.
22
+ Values outside `[0, 1]` throw.
23
+
24
+ ## [0.135.1] - 2026-07-28 - stable estimated-cost receipts
25
+
26
+ ### Fixed
27
+
28
+ - Token-priced receipts now tolerate the machine-precision difference introduced when per-million rates are persisted and replayed as per-thousand rates.
29
+ - Material disagreement between a receipt's estimated cost, token usage, and pricing snapshot still fails validation.
30
+
7
31
  ## [0.135.0] - 2026-07-28 - mint refuses what nobody measured
8
32
 
9
33
  ### Why a MINOR and not a patch
@@ -1,7 +1,7 @@
1
- import { A as parseFindingSubject, C as stripCodeFences, D as FindingSubjectStringSchema, E as FINDING_SUBJECT_SYNTAX, F as createChatClient, I as createAnalystAi, M as behavioralAnalyst, N as deriveEfficiencyFindings, O as KIND_EXPECTED_SUBJECTS, S as coerceToFindingRows, T as FINDING_SUBJECT_KINDS, _ as RawAnalystEvidenceSchema, a as KNOWLEDGE_GAP_KIND_SPEC, b as parseRawFinding, c as buildTraceToolsForGroup, d as renderUpstreamFindings, f as settleUsageReceiptFromCostLedger, g as RAW_FINDING_SCHEMA_PROMPT, h as ANALYST_SEVERITIES, i as KNOWLEDGE_POISONING_KIND_SPEC, j as renderFindingSubject, k as findingSubjectGrammarPromptFor, l as createTraceAnalystKind, m as structureFindings, n as AnalystRegistry, o as IMPROVEMENT_KIND_SPEC, p as validateUsageSettlementTimeout, r as DEFAULT_TRACE_ANALYST_KINDS, s as FAILURE_MODE_KIND_SPEC, t as buildDefaultAnalystRegistry, u as renderPriorFindings, v as RawAnalystFindingSchema, w as FINDING_SUBJECT_GRAMMAR_PROMPT, x as coerceJson, y as evidenceRefsFromRawFinding } from "../default-registry-CFUZyNeZ.js";
2
- import { i as CostLedger } from "../cost-ledger-ZAa_P4r0.js";
1
+ import { A as parseFindingSubject, C as stripCodeFences, D as FindingSubjectStringSchema, E as FINDING_SUBJECT_SYNTAX, F as createChatClient, I as createAnalystAi, M as behavioralAnalyst, N as deriveEfficiencyFindings, O as KIND_EXPECTED_SUBJECTS, S as coerceToFindingRows, T as FINDING_SUBJECT_KINDS, _ as RawAnalystEvidenceSchema, a as KNOWLEDGE_GAP_KIND_SPEC, b as parseRawFinding, c as buildTraceToolsForGroup, d as renderUpstreamFindings, f as settleUsageReceiptFromCostLedger, g as RAW_FINDING_SCHEMA_PROMPT, h as ANALYST_SEVERITIES, i as KNOWLEDGE_POISONING_KIND_SPEC, j as renderFindingSubject, k as findingSubjectGrammarPromptFor, l as createTraceAnalystKind, m as structureFindings, n as AnalystRegistry, o as IMPROVEMENT_KIND_SPEC, p as validateUsageSettlementTimeout, r as DEFAULT_TRACE_ANALYST_KINDS, s as FAILURE_MODE_KIND_SPEC, t as buildDefaultAnalystRegistry, u as renderPriorFindings, v as RawAnalystFindingSchema, w as FINDING_SUBJECT_GRAMMAR_PROMPT, x as coerceJson, y as evidenceRefsFromRawFinding } from "../default-registry-BAhV-lbE.js";
2
+ import { i as CostLedger } from "../cost-ledger-DHAjwNj7.js";
3
3
  import { c as makeFinding, l as makeProposalFinding, n as isProposalFinding, s as computeFindingId, t as assertProposalFindings } from "../proposal-findings-DCawte-y.js";
4
- import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, l as emitSkillUsageFindings, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-Ca--u10C.js";
4
+ import { a as RunCritic, c as buildSkillUsageReport, d as defaultIsMaterial, f as diffFindings, i as runSemanticConceptJudge, l as emitSkillUsageFindings, n as SEMANTIC_CONCEPT_JUDGE_VERSION, o as SKILL_USAGE_ANALYST, s as SkillUsageAnalyst, u as FindingsStore } from "../semantic-concept-judge-Btozx3Vc.js";
5
5
  //#region src/analyst/adapters.ts
6
6
  /**
7
7
  * Adapter factories — lift each existing agent-eval primitive into the
@@ -1,8 +1,8 @@
1
1
  import { a as RunRecord } from "./run-record-DcObtIGh.js";
2
2
  import { i as AnalystRegistry } from "./default-registry-Brxr728w.js";
3
3
  import { a as DatasetScenario } from "./dataset-BvtnC8Dc.js";
4
- import "./summary-report-DGp0-_XO.js";
5
- import { C as InsightReport, b as ExecutionInsight, v as CostProvenanceSummary } from "./client-BIyh1RCr.js";
4
+ import "./summary-report-CPMINBqs.js";
5
+ import { C as InsightReport, b as ExecutionInsight, v as CostProvenanceSummary } from "./client-DcvgkaZi.js";
6
6
  //#region src/contract/analyze-runs.d.ts
7
7
  interface AnalyzeRunsOptions {
8
8
  /** The runs to analyze. */
@@ -69,4 +69,4 @@ declare function summarizeExecution(opts: SummarizeExecutionOptions): ExecutionR
69
69
  declare function analyzeRuns(opts: AnalyzeRunsOptions): Promise<InsightReport>;
70
70
  //#endregion
71
71
  export { summarizeExecution as a, analyzeRuns as i, ExecutionReport as n, SummarizeExecutionOptions as r, AnalyzeRunsOptions as t };
72
- //# sourceMappingURL=analyze-runs-DMo3Lb_y.d.ts.map
72
+ //# sourceMappingURL=analyze-runs-Cda5Xkj1.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"analyze-runs-DMo3Lb_y.d.ts","names":[],"sources":["../src/contract/analyze-runs.ts"],"mappings":";;;;;;UAoEiB;;EAEf,MAAM;;;EAGN;;;;;EAKA;EACA;;;EAGA,kBAAkB;;;EAGlB,UAAU;;;;EAIV;IACE;IACA,cAAc;;;;;EAKhB,cAAc;IAAQ;IAAe;IAAe;;;EAEpD;;;;EAIA;;;;;;;;;EASA,eAAe;;;EAGf;;UAGe;EACf,MAAM;EACN;;UAGe;EACf,WAAW;EACX,gBAAgB;;;iBAIF,mBAAmB,MAAM,4BAA4B;iBAS/C,YAAY,MAAM,qBAAqB,QAAQ"}
1
+ {"version":3,"file":"analyze-runs-Cda5Xkj1.d.ts","names":[],"sources":["../src/contract/analyze-runs.ts"],"mappings":";;;;;;UAoEiB;;EAEf,MAAM;;;EAGN;;;;;EAKA;EACA;;;EAGA,kBAAkB;;;EAGlB,UAAU;;;;EAIV;IACE;IACA,cAAc;;;;;EAKhB,cAAc;IAAQ;IAAe;IAAe;;;EAEpD;;;;EAIA;;;;;;;;;EASA,eAAe;;;EAGf;;UAGe;EACf,MAAM;EACN;;UAGe;EACf,WAAW;EACX,gBAAgB;;;iBAIF,mBAAmB,MAAM,4BAA4B;iBAgB/C,YAAY,MAAM,qBAAqB,QAAQ"}
@@ -1,10 +1,10 @@
1
- import { C as pairedBootstrap, F as spearmanR, G as continuousAgreement, N as requiredPairedSampleSize, O as pairedTTest, T as pairedMde, j as pearsonR, w as pairedCohensDz } from "./statistics-RwRNu2__.js";
2
- import { r as pairRunRecords } from "./paired-arms-CA_8pN01.js";
1
+ import { D as pairedCohensDz, E as pairedBootstrap, L as pearsonR, P as pairedTTest, V as spearmanR, Z as continuousAgreement, k as pairedMde, z as requiredPairedSampleSize } from "./statistics-CnGCLLqc.js";
2
+ import { r as pairRunRecords } from "./paired-arms-BbFKrAU-.js";
3
3
  import { r as observedSplitScore } from "./reward-nw2xZGZG.js";
4
4
  import { s as validateRunRecord } from "./run-record-BIwU2wdV.js";
5
- import { r as welchsTTest } from "./baseline-BaPxoROc.js";
5
+ import { r as welchsTTest } from "./baseline-BUeFcgrn.js";
6
6
  import { o as llmSpans } from "./query-Di7eEQ79.js";
7
- import { r as paretoChart } from "./summary-report-BxtossFi.js";
7
+ import { r as paretoChart } from "./summary-report-BEk8OFLs.js";
8
8
  //#region src/contamination-guard.ts
9
9
  function checkCanaries(output, scenarios) {
10
10
  const leaks = [];
@@ -157,10 +157,17 @@ function summarizeExecution(opts) {
157
157
  costProvenance: summarizeCostProvenance(runs)
158
158
  };
159
159
  }
160
+ /** A bootstrap interval with no spread: every resample landed on the same
161
+ * value, so the interval carries no information about how far the point
162
+ * estimate could be wrong and cannot support a directional claim. */
163
+ function zeroWidth(ci) {
164
+ return !Number.isFinite(ci[0]) || !Number.isFinite(ci[1]) || ci[0] === ci[1];
165
+ }
160
166
  async function analyzeRuns(opts) {
161
167
  const runs = opts.runs.map(validateRunRecord);
162
168
  const bins = opts.histogramBins ?? 12;
163
169
  const threshold = opts.decisionThreshold ?? .02;
170
+ if (!Number.isFinite(threshold)) throw new Error(`analyzeRuns: decisionThreshold must be finite, got ${threshold}`);
164
171
  const split = resolveSplit(runs, opts.split ?? "auto");
165
172
  const compositeWithIds = runs.map((r) => ({
166
173
  runId: r.runId,
@@ -894,7 +901,7 @@ function computeOutcomeCorrelation(runs, outcome, split) {
894
901
  }
895
902
  function buildReleaseScorecard(composite, lift, contamination) {
896
903
  const axes = [];
897
- const liftPass = lift === void 0 ? "not_evaluated" : !lift.decisionEligible ? "not_evaluated" : lift.ci95[0] > 0 ? "pass" : lift.delta > 0 ? "warn" : "fail";
904
+ const liftPass = lift === void 0 ? "not_evaluated" : !lift.decisionEligible ? "not_evaluated" : lift.ci95[0] > 0 && !zeroWidth(lift.ci95) ? "pass" : lift.delta > 0 ? "warn" : "fail";
898
905
  axes.push({
899
906
  name: "quality-lift",
900
907
  status: liftPass,
@@ -1012,7 +1019,7 @@ function buildRecommendations(ctx) {
1012
1019
  const pairedEffect = ctx.lift.cohensD === null ? "undefined (zero delta variance)" : ctx.lift.cohensD.toFixed(2);
1013
1020
  const pairedP = ctx.lift.pValue === null ? "undefined (zero delta variance)" : ctx.lift.pValue.toFixed(4);
1014
1021
  const requiredRuns = ctx.lift.requiredN === null ? "not estimable" : `~${ctx.lift.requiredN} paired runs`;
1015
- const decisive = ctx.lift.ci95[0] > ctx.threshold;
1022
+ const decisive = !zeroWidth(ctx.lift.ci95) && ctx.lift.ci95[0] > ctx.threshold;
1016
1023
  const inconclusive = ctx.lift.ci95[0] <= ctx.threshold && ctx.lift.ci95[1] > ctx.threshold;
1017
1024
  if (decisive) out.push({
1018
1025
  priority: "critical",
@@ -1072,4 +1079,4 @@ function buildRecommendations(ctx) {
1072
1079
  //#endregion
1073
1080
  export { checkBehavioralCanary as a, canaryLeakView as i, summarizeExecution as n, checkCanaries as o, HoldoutAuditor as r, runBehavioralCanaries as s, analyzeRuns as t };
1074
1081
 
1075
- //# sourceMappingURL=analyze-runs-qk8op0tN.js.map
1082
+ //# sourceMappingURL=analyze-runs-jjCmF8pU.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"analyze-runs-jjCmF8pU.js","names":[],"sources":["../src/contamination-guard.ts","../src/contract/analyze-runs.ts"],"sourcesContent":["/**\n * ContaminationGuard — ensures held-out scenarios don't leak into\n * training/prompt paths, and flags model memorization.\n *\n * Three probes:\n * 1. `checkCanaries(output, scenario)` — if the scenario carries a\n * canary token, it MUST NOT appear in the agent's output.\n * Canaries are strings that are statistically impossible to\n * reconstruct from the scenario description alone — so if they\n * echo back, the model memorized them.\n * 2. `canaryLeakView(store)` — cross-corpus view of every run whose\n * output contained a canary, with the offending scenario + run.\n * 3. `HoldoutAuditor` — wraps a Dataset and emits a structured error\n * on any code path that reads holdout scenarios but doesn't flag\n * `purpose: 'evaluation'`. Keeps engineers honest.\n */\n\nimport type { DatasetScenario } from './dataset'\nimport { llmSpans } from './trace/query'\nimport type { TraceStore } from './trace/store'\n\nexport interface CanaryLeak {\n scenarioId: string\n canary: string\n runId?: string\n evidence: string\n}\n\nexport function checkCanaries(output: string, scenarios: DatasetScenario[]): CanaryLeak[] {\n const leaks: CanaryLeak[] = []\n for (const s of scenarios) {\n if (!s.canary) continue\n if (output.includes(s.canary)) {\n leaks.push({ scenarioId: s.id, canary: s.canary, evidence: excerpt(output, s.canary) })\n }\n }\n return leaks\n}\n\n/**\n * Behavioral canary: tests the AGENT, not the eval grep.\n *\n * - `forbiddenPattern` PRESENT in output ⇒ AGENT EMITTED BAD CONTENT ⇒ LEAK.\n * - `forbiddenPattern` ABSENT from output ⇒ AGENT HELD THE LINE ⇒ PASS.\n *\n * Use when running known-bad-prompt scenarios against the agent under\n * test and you want to know if the agent misbehaved. The classical\n * {@link checkCanaries} / {@link import('./canary').runCanaries | runCanaries}\n * test whether the eval check fires when the bad output is forced\n * into the eval flow — different question, different answer.\n *\n * Pattern resolution order (first match wins):\n * 1. `scenario.forbiddenPattern` — if it parses as `/body/flags`,\n * treated as a regex; otherwise a literal substring.\n * 2. `scenario.canary` — literal substring fallback so the helper\n * works on existing scenario fixtures.\n *\n * Returns `null` when nothing forbidden was found OR the scenario\n * declared no pattern.\n */\nexport function checkBehavioralCanary(\n output: string,\n scenario: DatasetScenario,\n): CanaryLeak | null {\n const pattern = scenario.forbiddenPattern ?? scenario.canary\n if (!pattern) return null\n const hit = matchForbidden(output, pattern)\n if (!hit) return null\n return {\n scenarioId: scenario.id,\n canary: pattern,\n evidence: excerpt(output, hit),\n }\n}\n\n/**\n * Behavioral canary over many (scenario, output) pairs. Sibling to\n * {@link import('./canary').runCanaries | runCanaries} — same idea\n * (run-many → report) but the question being answered is \"did the\n * AGENT misbehave?\" rather than \"did the EVAL grep fire?\".\n *\n * Returns one `CanaryLeak` per pair where the agent's output\n * contained its scenario's `forbiddenPattern` (or `canary` fallback).\n */\nexport function runBehavioralCanaries(\n cases: Array<{ scenario: DatasetScenario; output: string; runId?: string }>,\n): CanaryLeak[] {\n const leaks: CanaryLeak[] = []\n for (const c of cases) {\n const leak = checkBehavioralCanary(c.output, c.scenario)\n if (leak) leaks.push({ ...leak, runId: c.runId ?? leak.runId })\n }\n return leaks\n}\n\n/**\n * Resolve a forbidden-pattern string to the matched substring inside\n * `output`. `/body/flags` notation is interpreted as a regex; anything\n * else is a literal substring.\n */\nfunction matchForbidden(output: string, pattern: string): string | null {\n const re = tryParseRegex(pattern)\n if (re) {\n const m = output.match(re)\n return m && m[0].length > 0 ? m[0] : null\n }\n return output.includes(pattern) ? pattern : null\n}\n\nfunction tryParseRegex(pattern: string): RegExp | null {\n if (pattern.length < 2 || pattern[0] !== '/') return null\n const last = pattern.lastIndexOf('/')\n if (last <= 0) return null\n const body = pattern.slice(1, last)\n const flags = pattern.slice(last + 1)\n if (!/^[gimsuy]*$/.test(flags)) return null\n try {\n return new RegExp(body, flags)\n } catch {\n return null\n }\n}\n\n/**\n * Scan the LLM-output history in a corpus; returns every case where a\n * canary from a known scenario appeared in agent output. Pass the full\n * set of scenarios whose canaries you care about (typically the whole\n * held-out slice).\n */\nexport async function canaryLeakView(\n store: TraceStore,\n scenarios: DatasetScenario[],\n): Promise<CanaryLeak[]> {\n const targets = scenarios.filter((s) => !!s.canary)\n if (targets.length === 0) return []\n const spans = await llmSpans(store)\n const leaks: CanaryLeak[] = []\n for (const span of spans) {\n const output = span.output ?? ''\n for (const s of targets) {\n if (s.canary && output.includes(s.canary)) {\n leaks.push({\n scenarioId: s.id,\n canary: s.canary,\n runId: span.runId,\n evidence: excerpt(output, s.canary),\n })\n }\n }\n }\n return leaks\n}\n\nexport class HoldoutAuditor {\n private scenarios: DatasetScenario[]\n private accessLog: Array<{ scenarioId: string; purpose: string; at: number }> = []\n\n constructor(scenarios: DatasetScenario[]) {\n this.scenarios = scenarios\n }\n\n /** Retrieve a holdout scenario for a declared purpose. Non-'evaluation' throws. */\n get(scenarioId: string, purpose: 'evaluation' | 'debugging'): DatasetScenario {\n if (purpose !== 'evaluation' && purpose !== 'debugging') {\n throw new Error(\n `HoldoutAuditor.get: purpose must be 'evaluation' or 'debugging', got ${purpose}`,\n )\n }\n const s = this.scenarios.find((x) => x.id === scenarioId)\n if (!s) throw new Error(`holdout scenario \"${scenarioId}\" not found`)\n this.accessLog.push({ scenarioId, purpose, at: Date.now() })\n return s\n }\n\n getAccessLog(): ReadonlyArray<{ scenarioId: string; purpose: string; at: number }> {\n return this.accessLog\n }\n}\n\nfunction excerpt(source: string, needle: string): string {\n const at = source.indexOf(needle)\n if (at < 0) return ''\n const start = Math.max(0, at - 30)\n const end = Math.min(source.length, at + needle.length + 30)\n return (start > 0 ? '…' : '') + source.slice(start, end) + (end < source.length ? '…' : '')\n}\n","/**\n * # `analyzeRuns()` — turn a set of agent runs into an actionable decision packet.\n *\n * Wires the substrate's statistical, calibration, clustering, Pareto, and\n * release-confidence primitives into one `InsightReport`. Two top-level\n * entry points use this function:\n *\n * - `selfImprove()` calls it on the campaign output to attach a packet\n * to every run.\n * - Consumers with observed `RunRecord[]` (production traces, gold\n * corpora, approve/reject tables) call it directly via `analyzeRuns()`\n * for analysis without a closed loop.\n *\n * Every section is opt-in based on what the input data supports — the\n * function never invents signal. If runs carry no judge scores, `judges`\n * is empty. If there's no baseline/candidate split, `lift` is undefined.\n * If no `analyst` is wired, `failureClusters` is undefined.\n *\n * The `recommendations` array is the human-readable layer; everything\n * else is the evidence backing each recommendation.\n */\n\nimport type { AnalystRegistry } from '../analyst/registry'\nimport type { AnalystFinding } from '../analyst/types'\nimport { welchsTTest } from '../baseline'\nimport { checkCanaries } from '../contamination-guard'\nimport type { DatasetScenario } from '../dataset'\nimport { continuousAgreement } from '../judge-calibration'\nimport { pairRunRecords } from '../paired-arms'\nimport { observedSplitScore } from '../rollout/reward'\nimport {\n type RunRecord,\n type RunTerminalOutcome,\n type RunTokenUsage,\n validateRunRecord,\n} from '../run-record'\nimport {\n BOOTSTRAP_GATE_MIN_N,\n pairedBootstrap,\n pairedCohensDz,\n pairedMde,\n pairedTTest,\n pearsonR,\n requiredPairedSampleSize,\n spearmanR,\n} from '../statistics'\nimport { type ParetoFigureSpec, paretoChart } from '../summary-report'\nimport type { FailureClass } from '../trace/schema'\n\nimport type {\n CostProvenanceSummary,\n ExecutionInsight,\n FailureClassTally,\n FailureClusterInsight,\n InsightReport,\n InterRaterInsight,\n JudgeInsight,\n LiftInsight,\n MetricDelta,\n OutcomeCorrelationInsight,\n PriorPeriodComparison,\n Recommendation,\n ScalarDistribution,\n TokenUsageInsight,\n} from './insight-report'\n\n// ── Public API ───────────────────────────────────────────────────────\n\nexport interface AnalyzeRunsOptions {\n /** The runs to analyze. */\n runs: RunRecord[]\n /** Which split to score against when reading composite from RunOutcome.\n * Default: holdout when ANY run has a `holdoutScore`, else search. */\n split?: 'search' | 'holdout' | 'auto'\n /** Pairwise analysis configuration. When both `baselineCandidateId` and\n * `candidateCandidateId` are present, lift is computed on paired\n * (experimentId, scenarioId, seed) identities shared between the two sides.\n * Unmatched rows remain visible in the lift result. */\n baselineCandidateId?: string\n candidateCandidateId?: string\n /** Canary scenarios — checked against every run's raw output for\n * holdout contamination. */\n canaryScenarios?: DatasetScenario[]\n /** Analyst registry for failure clustering. When omitted, the\n * `failureClusters` section is left undefined. */\n analyst?: AnalystRegistry\n /** Downstream outcome metric per run (e.g. engagement rate, approval\n * rate, downstream pass rate). When present, the report includes\n * `outcomeCorrelation` + a simple linear reward model fit. */\n outcomeSignal?: {\n metric: string\n valueByRunId: Record<string, number>\n }\n /** Multi-rater feedback for inter-rater agreement. Each entry is one\n * rater's score for one run. Two or more raters → kappa + disagreement\n * triage list. */\n raterScores?: Array<{ runId: string; rater: string; score: number }>\n /** Number of histogram bins for distributional summaries. Default 12. */\n histogramBins?: number\n /** Decision threshold — the smallest composite lift the caller cares\n * about. Used by the recommendations engine to call ship vs hold.\n * Default 0.02. */\n decisionThreshold?: number\n /** Optional prior-period runs. When set, the report includes\n * `priorPeriodComparison` with per-metric Welch-CI deltas and\n * recommendations fire on statistically significant regressions.\n * The two windows do NOT have to share scenarios — the comparison\n * is two-sample unpaired (the substrate's `lift` field uses paired\n * bootstrap on shared (experimentId, scenarioId, seed) identities; this is the\n * shape for \"this week vs last week\" rather than \"candidate vs\n * baseline within a campaign\"). */\n baselineRuns?: RunRecord[]\n /** Human-readable label for the baseline window, e.g. \"vs prior 7\n * days\", \"vs v3.1 release\". Surfaces in recommendations + UI. */\n baselineLabel?: string\n}\n\nexport interface SummarizeExecutionOptions {\n runs: RunRecord[]\n histogramBins?: number\n}\n\nexport interface ExecutionReport {\n execution: ExecutionInsight\n costProvenance: CostProvenanceSummary\n}\n\n/** Summarize runtime facts without interpreting task quality or promotion readiness. */\nexport function summarizeExecution(opts: SummarizeExecutionOptions): ExecutionReport {\n const runs = opts.runs.map(validateRunRecord)\n const bins = opts.histogramBins ?? 12\n return {\n execution: computeExecutionInsight(runs, bins),\n costProvenance: summarizeCostProvenance(runs),\n }\n}\n\n/** A bootstrap interval with no spread: every resample landed on the same\n * value, so the interval carries no information about how far the point\n * estimate could be wrong and cannot support a directional claim. */\nfunction zeroWidth(ci: readonly [number, number]): boolean {\n return !Number.isFinite(ci[0]) || !Number.isFinite(ci[1]) || ci[0] === ci[1]\n}\n\nexport async function analyzeRuns(opts: AnalyzeRunsOptions): Promise<InsightReport> {\n const runs = opts.runs.map(validateRunRecord)\n const bins = opts.histogramBins ?? 12\n const threshold = opts.decisionThreshold ?? 0.02\n if (!Number.isFinite(threshold)) {\n throw new Error(`analyzeRuns: decisionThreshold must be finite, got ${threshold}`)\n }\n const split = resolveSplit(runs, opts.split ?? 'auto')\n\n const compositeWithIds = runs\n .map((r) => ({ runId: r.runId, score: compositeOf(r, split) }))\n .filter((p) => Number.isFinite(p.score))\n const composite = distributionOf(\n compositeWithIds.map((p) => p.score),\n bins,\n compositeWithIds,\n )\n\n const perDimension = computePerDimension(runs, bins)\n const { execution, costProvenance: provenance } = summarizeExecution({\n runs,\n histogramBins: bins,\n })\n const knownCostRuns = runs.filter((run) => run.costProvenance.kind !== 'uncaptured')\n const costs = knownCostRuns.map((r) => r.costUsd).filter(isFiniteNumber)\n const costDist = distributionOf(costs, bins)\n const pareto = paretoChart(knownCostRuns, { split })\n const degraded: { cost?: string; pareto?: string } = {}\n if (provenance.uncaptured.n > 0) {\n degraded.cost = diagnoseCostCoverage(runs, provenance)\n } else if (costs.length === 0 || costs.every((c) => c === 0)) {\n degraded.cost = `all ${runs.length} explicitly observed or estimated USD values are $0`\n }\n if (pareto.points.length < 2) {\n degraded.pareto =\n pareto.points.length === 0\n ? 'no candidates — Pareto unavailable'\n : 'single candidate — Pareto is a single point, not a frontier'\n }\n const costQuality = {\n cost: costDist,\n pareto,\n provenance,\n ...(degraded.cost || degraded.pareto ? { degraded } : {}),\n }\n\n const judges = computeJudgeInsights(runs)\n\n const interRater = opts.raterScores ? computeInterRater(opts.raterScores) : undefined\n\n const lift = computeLift(runs, opts.baselineCandidateId, opts.candidateCandidateId, split)\n\n const failureClusters = opts.analyst\n ? await computeFailureClusters(runs, opts.analyst, split)\n : undefined\n\n const failureClasses = computeFailureClasses(runs, split)\n\n const contamination = opts.canaryScenarios\n ? computeContamination(runs, opts.canaryScenarios)\n : undefined\n\n const outcomeCorrelation = opts.outcomeSignal\n ? computeOutcomeCorrelation(runs, opts.outcomeSignal, split)\n : undefined\n\n const release = buildReleaseScorecard(composite, lift, contamination)\n\n const priorPeriodComparison = opts.baselineRuns\n ? computePriorPeriodComparison(runs, opts.baselineRuns, split, opts.baselineLabel)\n : undefined\n\n const recommendations = buildRecommendations({\n composite,\n judges,\n interRater,\n lift,\n failureClusters,\n failureClasses,\n contamination,\n outcomeCorrelation,\n priorPeriodComparison,\n threshold,\n })\n\n return {\n n: runs.length,\n execution,\n composite,\n perDimension,\n costQuality,\n judges,\n interRater,\n lift,\n failureClusters,\n contamination,\n outcomeCorrelation,\n release,\n ...(failureClasses ? { failureClasses } : {}),\n ...(priorPeriodComparison ? { priorPeriodComparison } : {}),\n recommendations,\n }\n}\n\nfunction computeExecutionInsight(runs: RunRecord[], bins: number): ExecutionInsight {\n const aggregateRows = runs.flatMap((run) => {\n const usage = aggregateTokenUsage(run)\n return usage ? [{ usage, costUsd: finiteRaw(run, 'aggregate_cost_usd') }] : []\n })\n const aggregateCosts = aggregateRows.flatMap((row) =>\n row.costUsd !== undefined ? [row.costUsd] : [],\n )\n const modelCounts = new Map<string, number>()\n let executionErrorRuns = 0\n let executionErrorEvents = 0\n let errorReportingRuns = 0\n let errorSpanEvents = 0\n let errorSpanReportingRuns = 0\n const terminalOutcomes: Record<RunTerminalOutcome, number> = {\n succeeded: 0,\n failed: 0,\n cancelled: 0,\n incomplete: 0,\n unknown: 0,\n }\n const errorsByTerminalOutcome: ExecutionInsight['executionErrors']['byTerminalOutcome'] = {\n succeeded: { withErrors: 0, withoutErrors: 0, unreported: 0 },\n failed: { withErrors: 0, withoutErrors: 0, unreported: 0 },\n cancelled: { withErrors: 0, withoutErrors: 0, unreported: 0 },\n incomplete: { withErrors: 0, withoutErrors: 0, unreported: 0 },\n unknown: { withErrors: 0, withoutErrors: 0, unreported: 0 },\n }\n let modelCallRuns = 0\n let modelCallEvents = 0\n let modelCallReportingRuns = 0\n\n for (const run of runs) {\n modelCounts.set(run.model, (modelCounts.get(run.model) ?? 0) + 1)\n const terminalOutcome = run.terminalOutcome\n terminalOutcomes[terminalOutcome] += 1\n const modelCalls = nonNegativeCountRaw(run, 'llm_span_count')\n if (modelCalls !== undefined) {\n modelCallEvents += modelCalls\n modelCallReportingRuns += 1\n }\n const usage = run.tokenUsage\n if (\n (modelCalls ?? 0) > 0 ||\n usage.input > 0 ||\n usage.output > 0 ||\n (usage.cached ?? 0) > 0 ||\n (usage.cacheWrite ?? 0) > 0\n ) {\n modelCallRuns += 1\n }\n const errorEvents = reportedExecutionErrorEvents(run)\n if (errorEvents !== undefined) {\n executionErrorEvents += errorEvents\n errorReportingRuns += 1\n if (errorEvents > 0) {\n executionErrorRuns += 1\n errorsByTerminalOutcome[terminalOutcome].withErrors += 1\n } else errorsByTerminalOutcome[terminalOutcome].withoutErrors += 1\n } else errorsByTerminalOutcome[terminalOutcome].unreported += 1\n const reportedErrorSpans = nonNegativeCountRaw(run, 'error_span_count')\n if (reportedErrorSpans !== undefined) {\n errorSpanEvents += reportedErrorSpans\n errorSpanReportingRuns += 1\n }\n }\n\n return {\n durationMs: distributionOf(\n runs.map((run) => run.wallMs),\n bins,\n ),\n queueMs: distributionOf(\n runs.filter((run) => run.queueMs !== undefined).map((run) => run.queueMs!),\n bins,\n ),\n tokenUsage: summarizeTokenUsage(\n runs.map((run) => run.tokenUsage),\n bins,\n ),\n aggregateUsage: {\n runs: aggregateRows.length,\n tokenUsage: summarizeTokenUsage(\n aggregateRows.map((row) => row.usage),\n bins,\n ),\n costUsd: distributionOf(aggregateCosts, bins),\n totalCostUsd: aggregateCosts.reduce((total, value) => total + value, 0),\n },\n models: [...modelCounts.entries()]\n .map(([model, count]) => ({ model, runs: count }))\n .sort((left, right) => right.runs - left.runs || left.model.localeCompare(right.model)),\n modelCalls: {\n runs: modelCallRuns,\n events: modelCallEvents,\n reportingRuns: modelCallReportingRuns,\n },\n executionErrors: {\n runs: executionErrorRuns,\n fraction: errorReportingRuns > 0 ? executionErrorRuns / errorReportingRuns : null,\n events: executionErrorEvents,\n reportingRuns: errorReportingRuns,\n errorSpanEvents,\n errorSpanReportingRuns,\n byTerminalOutcome: errorsByTerminalOutcome,\n },\n terminalOutcomes,\n }\n}\n\nfunction reportedExecutionErrorEvents(run: RunRecord): number | undefined {\n return nonNegativeCountRaw(run, 'execution_error_count')\n}\n\nfunction nonNegativeCountRaw(run: RunRecord, key: string): number | undefined {\n const value = finiteRaw(run, key)\n return value !== undefined && Number.isInteger(value) && value >= 0 ? value : undefined\n}\n\nfunction summarizeTokenUsage(usages: RunTokenUsage[], bins: number): TokenUsageInsight {\n const reasoning = usages.flatMap((usage) =>\n usage.reasoning !== undefined ? [usage.reasoning] : [],\n )\n const cached = usages.flatMap((usage) => (usage.cached !== undefined ? [usage.cached] : []))\n const cacheWrite = usages.flatMap((usage) =>\n usage.cacheWrite !== undefined ? [usage.cacheWrite] : [],\n )\n return {\n input: distributionOf(\n usages.map((usage) => usage.input),\n bins,\n ),\n output: distributionOf(\n usages.map((usage) => usage.output),\n bins,\n ),\n reasoning: distributionOf(reasoning, bins),\n cached: distributionOf(cached, bins),\n cacheWrite: distributionOf(cacheWrite, bins),\n totals: {\n input: usages.reduce((total, usage) => total + usage.input, 0),\n output: usages.reduce((total, usage) => total + usage.output, 0),\n reasoning: reasoning.reduce((total, value) => total + value, 0),\n cached: cached.reduce((total, value) => total + value, 0),\n cacheWrite: cacheWrite.reduce((total, value) => total + value, 0),\n },\n }\n}\n\nfunction aggregateTokenUsage(run: RunRecord): RunTokenUsage | undefined {\n const input = finiteRaw(run, 'aggregate_prompt_tokens')\n const output = finiteRaw(run, 'aggregate_completion_tokens')\n const reasoning = finiteRaw(run, 'aggregate_reasoning_tokens')\n const cached = finiteRaw(run, 'aggregate_cached_tokens')\n const cacheWrite = finiteRaw(run, 'aggregate_cache_write_tokens')\n if (\n input === undefined &&\n output === undefined &&\n reasoning === undefined &&\n cached === undefined &&\n cacheWrite === undefined\n )\n return undefined\n return {\n input: input ?? 0,\n output: output ?? 0,\n ...(reasoning !== undefined ? { reasoning } : {}),\n ...(cached !== undefined ? { cached } : {}),\n ...(cacheWrite !== undefined ? { cacheWrite } : {}),\n }\n}\n\nfunction finiteRaw(run: RunRecord, key: string): number | undefined {\n const value = run.outcome.raw[key]\n return typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : undefined\n}\n\nfunction summarizeCostProvenance(runs: RunRecord[]): CostProvenanceSummary {\n const summary: CostProvenanceSummary = {\n observed: { n: 0, totalUsd: 0 },\n estimated: { n: 0, totalUsd: 0 },\n uncaptured: { n: 0 },\n knownFraction: 0,\n }\n for (const run of runs) {\n const cost = run.costProvenance\n if (cost.kind === 'uncaptured') {\n summary.uncaptured.n += 1\n } else {\n summary[cost.kind].n += 1\n summary[cost.kind].totalUsd += cost.usd\n }\n }\n const known = summary.observed.n + summary.estimated.n\n summary.knownFraction = runs.length > 0 ? known / runs.length : 0\n return summary\n}\n\nfunction diagnoseCostCoverage(runs: RunRecord[], provenance: CostProvenanceSummary): string {\n const uncaptured = provenance.uncaptured.n\n const known = provenance.observed.n + provenance.estimated.n\n if (uncaptured === runs.length) {\n return `USD cost uncaptured for all ${runs.length} runs — no observed or estimated USD values; token and wall-time metrics remain available.`\n }\n return `USD cost uncaptured for ${uncaptured}/${runs.length} runs; excluded those rows from cost statistics (${known}/${runs.length} retained: ${provenance.observed.n} observed, ${provenance.estimated.n} estimated).`\n}\n\n/**\n * Model-free task-failure tally.\n *\n * Explicit non-success classes are task-failure evidence.\n * A low task score without a class is counted as `unknown`.\n */\nfunction computeFailureClasses(\n runs: RunRecord[],\n split: 'search' | 'holdout',\n): FailureClassTally[] | undefined {\n const counts = new Map<FailureClass, number>()\n for (const r of runs) {\n if (!isTaskFailure(r, split)) continue\n const key =\n r.failureClass !== undefined && r.failureClass !== 'success' ? r.failureClass : 'unknown'\n counts.set(key, (counts.get(key) ?? 0) + 1)\n }\n if (counts.size === 0) return undefined\n const n = runs.length\n return [...counts.entries()]\n .map(([failureClass, count]) => ({\n failureClass,\n count,\n share: n > 0 ? count / n : 0,\n }))\n .sort((a, b) => b.count - a.count || a.failureClass.localeCompare(b.failureClass))\n}\n\n// ── Prior-period comparison ─────────────────────────────────────────\n\n/** Direction of the metric — does \"higher current\" mean better or worse?\n * Composite + judge dimensions: higher is better. Cost + duration: lower\n * is better. The recommendations engine flips the sign before judging\n * regressed vs improved. */\ntype MetricDirection = 'higher-is-better' | 'lower-is-better'\n\nfunction computePriorPeriodComparison(\n current: RunRecord[],\n baseline: RunRecord[],\n split: 'search' | 'holdout',\n windowLabel: string | undefined,\n): PriorPeriodComparison | undefined {\n if (current.length === 0 || baseline.length === 0) return undefined\n\n const metrics: Record<string, MetricDelta> = {}\n const directions: Record<string, MetricDirection> = {}\n\n const compositeCurrent = current\n .map((r) => compositeOf(r, split))\n .filter(Number.isFinite) as number[]\n const compositeBaseline = baseline\n .map((r) => compositeOf(r, split))\n .filter(Number.isFinite) as number[]\n if (compositeCurrent.length > 0 && compositeBaseline.length > 0) {\n metrics.composite = welchCompare(compositeBaseline, compositeCurrent)\n directions.composite = 'higher-is-better'\n }\n\n const costCurrent = knownCostValues(current)\n const costBaseline = knownCostValues(baseline)\n if (costCurrent.length > 0 && costBaseline.length > 0) {\n metrics.cost = welchCompare(costBaseline, costCurrent)\n directions.cost = 'lower-is-better'\n }\n\n const durCurrent = current.map((r) => r.wallMs).filter(Number.isFinite)\n const durBaseline = baseline.map((r) => r.wallMs).filter(Number.isFinite)\n if (durCurrent.length > 0 && durBaseline.length > 0) {\n metrics.duration = welchCompare(durBaseline, durCurrent)\n directions.duration = 'lower-is-better'\n }\n\n const tokCurrent = current\n .map((r) => (r.tokenUsage.input ?? 0) + (r.tokenUsage.output ?? 0))\n .filter(Number.isFinite)\n const tokBaseline = baseline\n .map((r) => (r.tokenUsage.input ?? 0) + (r.tokenUsage.output ?? 0))\n .filter(Number.isFinite)\n if (tokCurrent.length > 0 && tokBaseline.length > 0) {\n metrics.tokenUsage = welchCompare(tokBaseline, tokCurrent)\n directions.tokenUsage = 'lower-is-better'\n }\n\n // Per-dimension judge comparisons — only for dimensions present in BOTH\n // windows. We use perDimMean since per-judge nesting is finicky for\n // two-sample comparisons across different judge configurations.\n const dimsCurrent = collectPerDimension(current)\n const dimsBaseline = collectPerDimension(baseline)\n for (const dim of Object.keys(dimsCurrent)) {\n const b = dimsBaseline[dim]\n const c = dimsCurrent[dim]\n if (!b || b.length === 0 || !c || c.length === 0) continue\n metrics[`dim.${dim}`] = welchCompare(b, c)\n directions[`dim.${dim}`] = 'higher-is-better'\n }\n\n const regressedMetrics: string[] = []\n const improvedMetrics: string[] = []\n const inconclusiveMetrics: string[] = []\n for (const [name, delta] of Object.entries(metrics)) {\n if (delta.status !== 'ok') {\n inconclusiveMetrics.push(name)\n continue\n }\n if (!delta.significant) continue\n const dir = directions[name] ?? 'higher-is-better'\n const better = dir === 'higher-is-better' ? delta.delta > 0 : delta.delta < 0\n if (better) improvedMetrics.push(name)\n else regressedMetrics.push(name)\n }\n\n return {\n baselineN: baseline.length,\n currentN: current.length,\n ...(windowLabel ? { windowLabel } : {}),\n metrics,\n regressedMetrics,\n improvedMetrics,\n inconclusiveMetrics,\n }\n}\n\nfunction knownCostValues(runs: RunRecord[]): number[] {\n return runs\n .filter((run) => run.costProvenance.kind !== 'uncaptured')\n .map((run) => run.costUsd)\n .filter(isFiniteNumber)\n}\n\nfunction isFiniteNumber(value: unknown): value is number {\n return typeof value === 'number' && Number.isFinite(value)\n}\n\n/** Collect per-dimension values across runs (from outcome.judgeScores.perDimMean). */\nfunction collectPerDimension(runs: RunRecord[]): Record<string, number[]> {\n const out: Record<string, number[]> = {}\n for (const r of runs) {\n const perDim = r.outcome.judgeScores?.perDimMean\n if (!perDim) continue\n for (const [dim, value] of Object.entries(perDim)) {\n if (!Number.isFinite(value)) continue\n if (!out[dim]) out[dim] = []\n out[dim].push(value as number)\n }\n }\n return out\n}\n\n/** Adapt the shared two-sample Welch result to the report contract. */\nfunction welchCompare(baseline: number[], current: number[]): MetricDelta {\n const result = welchsTTest(baseline, current)\n const base = {\n current: result.meanB,\n baseline: result.meanA,\n delta: result.delta,\n baselineN: baseline.length,\n currentN: current.length,\n }\n if (result.status !== 'ok') {\n return {\n ...base,\n status: result.status,\n ci95: null,\n pValue: null,\n cohensD: null,\n significant: false,\n }\n }\n return {\n ...base,\n status: 'ok',\n ci95: result.ci95,\n pValue: result.p,\n cohensD: result.cohensD,\n significant: result.p < 0.05 && Math.abs(result.cohensD) >= 0.2,\n }\n}\n\n// ── Composite + split selection ─────────────────────────────────────\n\nfunction resolveSplit(\n runs: RunRecord[],\n pref: 'search' | 'holdout' | 'auto',\n): 'search' | 'holdout' {\n if (pref !== 'auto') return pref\n const hasHoldout = runs.some((r) => Number.isFinite(observedSplitScore(r, 'holdout')))\n return hasHoldout ? 'holdout' : 'search'\n}\n\n/**\n * RAW (`observedSplitScore`): `analyzeRuns` describes what a set of runs\n * reported, and every downstream reader of this composite — distributions,\n * per-candidate summaries, the reward-hacking correlation — needs the ungated\n * number to see an inflated run at all.\n */\nfunction compositeOf(run: RunRecord, split: 'search' | 'holdout'): number {\n // Split-exact, no cross-split fallthrough: answering \"what did this run\n // score on the split I am summarising\" with the other split's number\n // silently mixes populations.\n const score = observedSplitScore(run, split)\n return Number.isFinite(score) ? (score as number) : Number.NaN\n}\n\n// ── Distribution helpers ────────────────────────────────────────────\n\nfunction distributionOf(\n values: number[],\n bins: number,\n withIds?: Array<{ runId: string; score: number }>,\n): ScalarDistribution {\n if (values.length === 0) {\n return {\n n: 0,\n mean: null,\n p50: null,\n p95: null,\n stddev: null,\n min: null,\n max: null,\n histogram: [],\n }\n }\n const sorted = [...values].sort((a, b) => a - b)\n const n = sorted.length\n const mean = sorted.reduce((s, v) => s + v, 0) / n\n const variance = sorted.reduce((s, v) => s + (v - mean) ** 2, 0) / n\n const stddev = Math.sqrt(variance)\n const tailRuns = withIds\n ? [...withIds].sort((a, b) => a.score - b.score).slice(0, Math.min(5, withIds.length))\n : undefined\n return {\n n,\n mean,\n p50: percentile(sorted, 0.5),\n p95: percentile(sorted, 0.95),\n stddev,\n min: sorted[0]!,\n max: sorted[n - 1]!,\n histogram: histogram(sorted, bins),\n ...(tailRuns ? { tailRuns } : {}),\n }\n}\n\nfunction percentile(sorted: number[], q: number): number {\n if (sorted.length === 0) return 0\n if (sorted.length === 1) return sorted[0]!\n const idx = (sorted.length - 1) * q\n const lo = Math.floor(idx)\n const hi = Math.ceil(idx)\n if (lo === hi) return sorted[lo]!\n const w = idx - lo\n return sorted[lo]! * (1 - w) + sorted[hi]! * w\n}\n\n/** Even-width histogram over the value range. Returns inclusive-lo /\n * exclusive-hi bins (closed on right for the last bin) compatible with\n * the substrate's `GainDistributionBin` shape. */\nfunction histogram(sorted: number[], bins: number): ScalarDistribution['histogram'] {\n if (sorted.length === 0 || bins < 1) return []\n const min = sorted[0]!\n const max = sorted[sorted.length - 1]!\n if (min === max) return [{ lo: min, hi: max, count: sorted.length }]\n const width = (max - min) / bins\n const out: ScalarDistribution['histogram'] = []\n for (let i = 0; i < bins; i++) {\n const lo = min + i * width\n const hi = i === bins - 1 ? max : lo + width\n out.push({ lo, hi, count: 0 })\n }\n for (const v of sorted) {\n const idx = Math.min(bins - 1, Math.floor((v - min) / width))\n out[idx]!.count++\n }\n return out\n}\n\nfunction computePerDimension(runs: RunRecord[], bins: number): Record<string, ScalarDistribution> {\n // JudgeScoresRecord pre-aggregates `perDimMean` (mean across judges per\n // dimension). We collect those means across runs to produce a per-dim\n // distribution at the corpus level. Consumers who want per-judge\n // dimension values reach into `perJudge[judgeId][dim]` themselves.\n const byDim = new Map<string, number[]>()\n for (const run of runs) {\n const scores = run.outcome.judgeScores\n if (!scores) continue\n for (const [dim, value] of Object.entries(scores.perDimMean ?? {})) {\n if (!Number.isFinite(value)) continue\n const arr = byDim.get(dim) ?? []\n arr.push(value)\n byDim.set(dim, arr)\n }\n }\n const out: Record<string, ScalarDistribution> = {}\n for (const [dim, values] of byDim) out[dim] = distributionOf(values, bins)\n return out\n}\n\n// ── Judge insights ──────────────────────────────────────────────────\n\nfunction computeJudgeInsights(runs: RunRecord[]): Record<string, JudgeInsight> {\n // Each judge's per-run mean is the average of its per-dimension scores\n // for that run. We aggregate those means across all runs each judge\n // scored — giving consumers a \"this judge's typical verdict\" reading.\n const out: Record<string, JudgeInsight> = {}\n const byJudge = new Map<string, number[]>()\n for (const run of runs) {\n const scores = run.outcome.judgeScores\n if (!scores?.perJudge) continue\n for (const [judgeId, dims] of Object.entries(scores.perJudge)) {\n const dimValues = Object.values(dims).filter(Number.isFinite) as number[]\n if (dimValues.length === 0) continue\n const judgeMean = dimValues.reduce((s, v) => s + v, 0) / dimValues.length\n const arr = byJudge.get(judgeId) ?? []\n arr.push(judgeMean)\n byJudge.set(judgeId, arr)\n }\n }\n for (const [judgeId, values] of byJudge) {\n out[judgeId] = {\n n: values.length,\n meanScore: values.reduce((s, v) => s + v, 0) / values.length,\n }\n }\n return out\n}\n\n// ── Inter-rater agreement ───────────────────────────────────────────\n\nfunction computeInterRater(\n ratings: Array<{ runId: string; rater: string; score: number }>,\n): InterRaterInsight | undefined {\n const byRun = new Map<string, Array<{ rater: string; score: number }>>()\n for (const r of ratings) {\n if (!Number.isFinite(r.score)) continue\n const list = byRun.get(r.runId) ?? []\n list.push({ rater: r.rater, score: r.score })\n byRun.set(r.runId, list)\n }\n const raters = new Set(ratings.map((r) => r.rater))\n const jointlyRated: string[] = []\n for (const [runId, ratersForRun] of byRun) {\n const seen = new Set(ratersForRun.map((r) => r.rater))\n let all = true\n for (const r of raters) if (!seen.has(r)) all = false\n if (all) jointlyRated.push(runId)\n }\n if (raters.size < 2 || jointlyRated.length === 0) return undefined\n\n const raterList = [...raters].sort()\n const perPair: Record<string, number> = {}\n for (let i = 0; i < raterList.length; i++) {\n for (let j = i + 1; j < raterList.length; j++) {\n const a = raterList[i]!\n const b = raterList[j]!\n const aScores: number[] = []\n const bScores: number[] = []\n for (const runId of jointlyRated) {\n const ratersForRun = byRun.get(runId)!\n const sa = ratersForRun.find((r) => r.rater === a)?.score\n const sb = ratersForRun.find((r) => r.rater === b)?.score\n if (sa !== undefined && sb !== undefined) {\n aScores.push(sa)\n bScores.push(sb)\n }\n }\n const agreement = continuousAgreement(\n aScores.map((score, index) => [score, bScores[index]!]),\n { bootstrap: 0 },\n )\n perPair[`${a}::${b}`] = agreement.weightedKappa\n }\n }\n const matrix = jointlyRated.map((runId) => {\n const ratingsByRater = new Map(byRun.get(runId)!.map((rating) => [rating.rater, rating.score]))\n return raterList.map((rater) => ratingsByRater.get(rater)!)\n })\n const agreement = continuousAgreement(matrix, { bootstrap: 0 })\n\n const disagreementCases = jointlyRated\n .map((runId) => {\n const ratersForRun = byRun.get(runId)!\n const scores = ratersForRun.map((r) => r.score)\n const range = Math.max(...scores) - Math.min(...scores)\n return { runId, ratings: ratersForRun, range }\n })\n .sort((a, b) => b.range - a.range)\n .slice(0, 20)\n\n return {\n raters: raters.size,\n jointlyRated: jointlyRated.length,\n kappa: Number.isFinite(agreement.weightedKappa) ? agreement.weightedKappa : 0,\n icc: agreement.icc,\n pearson: agreement.pearson,\n spearman: agreement.spearman,\n perPair,\n disagreementCases,\n }\n}\n\n// ── Lift ────────────────────────────────────────────────────────────\n\nfunction computeLift(\n runs: RunRecord[],\n baselineId: string | undefined,\n candidateId: string | undefined,\n split: 'search' | 'holdout',\n): LiftInsight | undefined {\n let bId = baselineId\n let cId = candidateId\n if (!bId || !cId) {\n // Auto-detect: when exactly two distinct candidateIds appear, treat the\n // lower-mean side as baseline.\n const ids = [...new Set(runs.map((r) => r.candidateId))]\n if (ids.length !== 2) return undefined\n const [idA, idB] = ids as [string, string]\n const scoresA = finiteCompositeScores(\n runs.filter((run) => run.candidateId === idA),\n split,\n )\n const scoresB = finiteCompositeScores(\n runs.filter((run) => run.candidateId === idB),\n split,\n )\n if (scoresA.length === 0 || scoresB.length === 0) return undefined\n const meanA = mean(scoresA)\n const meanB = mean(scoresB)\n bId = meanA <= meanB ? idA : idB\n cId = meanA <= meanB ? idB : idA\n }\n\n const baseline = runs.filter((r) => r.candidateId === bId)\n const candidate = runs.filter((r) => r.candidateId === cId)\n if (baseline.length === 0 || candidate.length === 0) return undefined\n\n const scoredBaseline = baseline.filter((run) => Number.isFinite(compositeOf(run, split)))\n const scoredCandidate = candidate.filter((run) => Number.isFinite(compositeOf(run, split)))\n const pairing = pairRunRecords(scoredBaseline, scoredCandidate)\n const pairedBaseline = pairing.pairs.map((pair) => compositeOf(pair.baseline, split))\n const pairedCandidate = pairing.pairs.map((pair) => compositeOf(pair.treatment, split))\n if (pairedBaseline.length === 0) return undefined\n\n const baselineMean = mean(pairedBaseline)\n const candidateMean = mean(pairedCandidate)\n const delta = candidateMean - baselineMean\n\n const bootstrap = pairedBootstrap(pairedBaseline, pairedCandidate, {\n confidence: 0.95,\n resamples: 2000,\n statistic: 'mean',\n })\n const tTest = pairedTTest(pairedBaseline, pairedCandidate)\n const d = pairedCohensDz(pairedBaseline, pairedCandidate)\n const mde = pairedMde({ nPaired: pairedBaseline.length, power: 0.8, alpha: 0.05 })\n const requiredN =\n d === null || d === 0\n ? null\n : requiredPairedSampleSize({\n effect: Math.abs(d),\n power: 0.8,\n alpha: 0.05,\n })\n\n return {\n baselineMean,\n candidateMean,\n delta,\n ci95: [bootstrap.low, bootstrap.high],\n pValue: tTest.p,\n n: pairedBaseline.length,\n minimumRequired: BOOTSTRAP_GATE_MIN_N,\n decisionEligible: bootstrap.gateEligible,\n unpairedBaseline: pairing.unpairedBaseline.length,\n unpairedCandidate: pairing.unpairedTreatment.length,\n cohensD: d,\n mde,\n requiredN,\n }\n}\n\nfunction mean(arr: number[]): number {\n return arr.length === 0 ? 0 : arr.reduce((s, v) => s + v, 0) / arr.length\n}\n\n// ── Failure clustering ──────────────────────────────────────────────\n\nasync function computeFailureClusters(\n runs: RunRecord[],\n analyst: AnalystRegistry,\n split: 'search' | 'holdout',\n): Promise<FailureClusterInsight | undefined> {\n const failed = runs.filter((run) => isTaskFailure(run, split))\n if (failed.length === 0) return { clusters: [], totalFailures: 0 }\n\n const clusters = new Map<string, { exemplars: string[]; share: number }>()\n for (const run of failed) {\n try {\n // AnalystRunInputs routes by field name: run-record analysts read\n // `runRecord`. Any other shape makes every analyst skip with\n // \"missing input\" and the clusters come back silently empty.\n const result = await analyst.run(run.runId, { runRecord: run })\n for (const finding of result.findings as AnalystFinding[]) {\n const key = finding.area || finding.analyst_id || 'unclassified'\n const c = clusters.get(key) ?? { exemplars: [], share: 0 }\n if (c.exemplars.length < 5) c.exemplars.push(run.runId)\n clusters.set(key, c)\n }\n } catch {\n const c = clusters.get('analyst-error') ?? { exemplars: [], share: 0 }\n if (c.exemplars.length < 5) c.exemplars.push(run.runId)\n clusters.set('analyst-error', c)\n }\n }\n const clusterList = [...clusters.entries()].map(([id, c]) => ({\n id,\n name: id,\n share: c.exemplars.length / failed.length,\n exemplars: c.exemplars,\n }))\n clusterList.sort((a, b) => b.share - a.share)\n return { clusters: clusterList, totalFailures: failed.length }\n}\n\nfunction finiteCompositeScores(runs: readonly RunRecord[], split: 'search' | 'holdout'): number[] {\n return runs.map((run) => compositeOf(run, split)).filter(Number.isFinite)\n}\n\nfunction isTaskFailure(run: RunRecord, split: 'search' | 'holdout'): boolean {\n if (run.failureClass !== undefined && run.failureClass !== 'success') return true\n const score = compositeOf(run, split)\n return Number.isFinite(score) && score < 0.5\n}\n\n// ── Contamination ──────────────────────────────────────────────────\n\nfunction computeContamination(\n runs: RunRecord[],\n canaries: DatasetScenario[],\n): InsightReport['contamination'] {\n let leaks = 0\n const details: Array<{ runId: string; canary: string; matched: string }> = []\n for (const run of runs) {\n const output = stringifyOutput(run)\n if (!output) continue\n const leaksHere = checkCanaries(output, canaries)\n for (const leak of leaksHere) {\n leaks++\n details.push({ runId: run.runId, canary: leak.canary, matched: leak.evidence })\n }\n }\n return { leaks, holdoutAuditPassed: leaks === 0, details }\n}\n\nfunction stringifyOutput(run: RunRecord): string | undefined {\n // RunRecord doesn't fix where \"the agent's output\" lives — different\n // consumers stash it differently. We probe the common shapes: the\n // outcome.raw map (numeric only by design — unlikely to contain text),\n // and any string-valued fields tucked under metadata via type casting.\n // Consumers with bespoke shapes pass canaryScenarios only when they\n // know their runs carry a stringifiable surface.\n const metadata = (run as unknown as { metadata?: Record<string, unknown> }).metadata\n if (typeof metadata?.output === 'string') return metadata.output\n if (typeof metadata?.text === 'string') return metadata.text\n return undefined\n}\n\n// ── Outcome correlation + linear reward model ──────────────────────\n\nfunction computeOutcomeCorrelation(\n runs: RunRecord[],\n outcome: { metric: string; valueByRunId: Record<string, number> },\n split: 'search' | 'holdout',\n): OutcomeCorrelationInsight | undefined {\n const xs: number[] = []\n const ys: number[] = []\n for (const run of runs) {\n const y = outcome.valueByRunId[run.runId]\n if (y === undefined || !Number.isFinite(y)) continue\n const x = compositeOf(run, split)\n if (!Number.isFinite(x)) continue\n xs.push(x)\n ys.push(y)\n }\n if (xs.length < 3) return undefined\n\n const p = pearsonR(xs, ys)\n const s = spearmanR(xs, ys)\n const meanX = mean(xs)\n const meanY = mean(ys)\n let num = 0\n let denom = 0\n for (let i = 0; i < xs.length; i++) {\n num += (xs[i]! - meanX) * (ys[i]! - meanY)\n denom += (xs[i]! - meanX) ** 2\n }\n const slope = denom === 0 ? 0 : num / denom\n const intercept = meanY - slope * meanX\n const ssTot = ys.reduce((a, y) => a + (y - meanY) ** 2, 0)\n const ssRes = ys.reduce((a, y, i) => a + (y - (intercept + slope * xs[i]!)) ** 2, 0)\n const r2 = ssTot === 0 ? 0 : 1 - ssRes / ssTot\n\n return {\n metric: outcome.metric,\n n: xs.length,\n pearson: p,\n spearman: s,\n rewardModel: { intercept, slope, r2 },\n }\n}\n\n// ── Release confidence scorecard ───────────────────────────────────\n\nfunction buildReleaseScorecard(\n composite: ScalarDistribution,\n lift: LiftInsight | undefined,\n contamination: InsightReport['contamination'],\n): InsightReport['release'] {\n // Synthesise a minimal scorecard from the rolled-up signal. The\n // substrate's `evaluateReleaseConfidence` primitive consumes a richer\n // input shape that callers can produce by wiring SLO definitions; the\n // shape here is the contract `selfImprove`/`analyzeRuns` consumers\n // receive automatically. They can call `evaluateReleaseConfidence`\n // directly when they want SLO-based axis evaluation.\n const axes: InsightReport['release']['axes'] = []\n const liftPass =\n lift === undefined\n ? ('not_evaluated' as const)\n : !lift.decisionEligible\n ? ('not_evaluated' as const)\n : lift.ci95[0] > 0 && !zeroWidth(lift.ci95)\n ? ('pass' as const)\n : lift.delta > 0\n ? ('warn' as const)\n : ('fail' as const)\n axes.push({\n name: 'quality-lift',\n status: liftPass,\n detail: lift\n ? `delta=${lift.delta.toFixed(3)}, CI95=[${lift.ci95[0].toFixed(3)}, ${lift.ci95[1].toFixed(3)}], n=${lift.n}${lift.decisionEligible ? '' : ` (descriptive only; ${lift.minimumRequired} required)`}`\n : 'no baseline/candidate pair available',\n })\n const contamPass =\n contamination === undefined\n ? ('not_evaluated' as const)\n : contamination.leaks === 0\n ? ('pass' as const)\n : ('fail' as const)\n axes.push({\n name: 'contamination',\n status: contamPass,\n detail: contamination ? `${contamination.leaks} canary leak(s)` : 'no canaries supplied',\n })\n axes.push(\n composite.n === 0\n ? {\n name: 'composite-distribution',\n status: 'not_evaluated',\n detail: 'no task-quality scores available',\n }\n : {\n name: 'composite-distribution',\n status:\n composite.mean !== null && composite.mean >= 0.5\n ? 'pass'\n : composite.mean !== null && composite.mean >= 0.3\n ? 'warn'\n : 'fail',\n detail:\n composite.mean === null || composite.p50 === null || composite.p95 === null\n ? 'task-quality distribution is internally incomplete'\n : `mean=${composite.mean.toFixed(3)}, p50=${composite.p50.toFixed(3)}, p95=${composite.p95.toFixed(3)} over n=${composite.n}`,\n },\n )\n const status = axes.some((a) => a.status === 'fail')\n ? 'fail'\n : axes.some((a) => a.status === 'warn' || a.status === 'not_evaluated')\n ? 'warn'\n : 'pass'\n return {\n status,\n axes,\n issues: [],\n }\n}\n\n// ── Recommendations engine ─────────────────────────────────────────\n\ninterface RecommendationContext {\n composite: ScalarDistribution\n judges: Record<string, JudgeInsight>\n interRater?: InterRaterInsight\n lift?: LiftInsight\n failureClusters?: FailureClusterInsight\n failureClasses?: FailureClassTally[]\n contamination?: InsightReport['contamination']\n outcomeCorrelation?: OutcomeCorrelationInsight\n priorPeriodComparison?: PriorPeriodComparison\n threshold: number\n}\n\nfunction buildRecommendations(ctx: RecommendationContext): Recommendation[] {\n const out: Recommendation[] = []\n\n // Prior-period regressions — highest customer-impact signal when present.\n // \"Did my last change help?\" with a falsifiable answer.\n if (ctx.priorPeriodComparison) {\n const ppc = ctx.priorPeriodComparison\n const label = ppc.windowLabel ?? 'baseline period'\n for (const name of ppc.regressedMetrics) {\n const d = ppc.metrics[name]\n if (d?.status !== 'ok') continue\n out.push({\n priority: 'critical',\n kind: 'investigate',\n title: `${name} regressed from ${d.baseline.toFixed(3)} → ${d.current.toFixed(3)} vs ${label}`,\n detail: `Welch CI95 = [${d.ci95[0].toFixed(3)}, ${d.ci95[1].toFixed(3)}], p=${d.pValue.toFixed(4)}, Cohen's d=${d.cohensD.toFixed(2)} (n_current=${d.currentN}, n_baseline=${d.baselineN}). The regression is statistically significant at p<0.05 with at-least-small effect size.`,\n evidencePath: `priorPeriodComparison.metrics.${name}`,\n })\n }\n for (const name of ppc.improvedMetrics) {\n const d = ppc.metrics[name]\n if (d?.status !== 'ok') continue\n out.push({\n priority: 'low',\n kind: 'ship',\n title: `${name} improved from ${d.baseline.toFixed(3)} → ${d.current.toFixed(3)} vs ${label}`,\n detail: `Welch CI95 = [${d.ci95[0].toFixed(3)}, ${d.ci95[1].toFixed(3)}], p=${d.pValue.toFixed(4)}, Cohen's d=${d.cohensD.toFixed(2)} (n_current=${d.currentN}, n_baseline=${d.baselineN}). Statistically significant improvement worth flagging.`,\n evidencePath: `priorPeriodComparison.metrics.${name}`,\n })\n }\n for (const name of ppc.inconclusiveMetrics) {\n const d = ppc.metrics[name]\n if (!d || d.status === 'ok' || d.delta === 0) continue\n const reason =\n d.status === 'zero-variance'\n ? 'both periods have zero observed variance'\n : 'one or both periods have fewer than two observations'\n out.push({\n priority: 'high',\n kind: 'investigate',\n title: `${name} changed from ${d.baseline.toFixed(3)} → ${d.current.toFixed(3)} vs ${label}; inference unavailable`,\n detail: `Observed delta ${d.delta.toFixed(3)} across n_current=${d.currentN} and n_baseline=${d.baselineN}, but ${reason}. The report does not fabricate a p-value, confidence interval, or effect size; inspect independence and data capture before acting.`,\n evidencePath: `priorPeriodComparison.metrics.${name}`,\n })\n }\n }\n\n // Composite-distribution branch. Fires when the overall quality signal is\n // poor regardless of lift / contamination / clusters — the customer needs\n // to know they have a problem AND which specific runs to inspect.\n if (\n ctx.composite.n > 0 &&\n ctx.composite.mean !== null &&\n ctx.composite.p50 !== null &&\n ctx.composite.p95 !== null\n ) {\n if (ctx.composite.mean < 0.3) {\n const tail = ctx.composite.tailRuns ?? []\n const names = tail\n .slice(0, 5)\n .map((t) => `${t.runId}=${t.score.toFixed(3)}`)\n .join(', ')\n out.push({\n priority: 'critical',\n kind: 'investigate',\n title: `Composite mean ${ctx.composite.mean.toFixed(3)} is below the 0.3 floor — the agent is broken on this corpus`,\n detail:\n tail.length > 0\n ? `Worst ${tail.length} run${tail.length === 1 ? '' : 's'} to inspect first: ${names}. Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.`\n : `Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.`,\n evidencePath: 'composite.tailRuns',\n })\n } else if (ctx.composite.mean < 0.5) {\n const tail = ctx.composite.tailRuns ?? []\n const names = tail\n .slice(0, 3)\n .map((t) => `${t.runId}=${t.score.toFixed(3)}`)\n .join(', ')\n out.push({\n priority: 'high',\n kind: 'investigate',\n title: `Composite mean ${ctx.composite.mean.toFixed(3)} is below 0.5 — investigate the lower tail before claiming the agent is healthy`,\n detail:\n tail.length > 0\n ? `Worst ${tail.length} run${tail.length === 1 ? '' : 's'}: ${names}. Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.`\n : `Histogram p50=${ctx.composite.p50.toFixed(3)}, p95=${ctx.composite.p95.toFixed(3)}.`,\n evidencePath: 'composite.tailRuns',\n })\n }\n }\n\n // A healthy-looking mean can hide a group of failed tasks sharing one\n // producer-reported cause. This path does not require an analyst.\n if (ctx.failureClasses && ctx.failureClasses.length > 0) {\n const top = ctx.failureClasses[0]!\n if (top.count >= 3 && top.share >= 0.15) {\n out.push({\n priority: top.share >= 0.25 ? 'high' : 'medium',\n kind: 'investigate',\n title: `'${top.failureClass}' is the dominant failure class — ${top.count} runs (${(top.share * 100).toFixed(0)}% of the corpus)`,\n detail: `The mean composite can look acceptable while one failure class dominates the lower tail. ${top.count} of ${ctx.composite.n} runs failed with '${top.failureClass}'${ctx.failureClasses.length > 1 ? ` (next: '${ctx.failureClasses[1]!.failureClass}' ×${ctx.failureClasses[1]!.count})` : ''}. Fix this cause first.`,\n evidencePath: 'failureClasses',\n })\n }\n }\n\n // Missing-judges branch. The report can't surface per-dimension or\n // calibration signal when `outcome.judgeScores` is empty across the\n // corpus. Tell the customer how to enrich.\n if (Object.keys(ctx.judges).length === 0 && ctx.composite.n > 0) {\n out.push({\n priority: 'medium',\n kind: 'expand-corpus',\n title: 'No judge scores recorded — per-dimension + calibration insights unavailable',\n detail:\n 'Records have no `outcome.judgeScores`. To unlock perDimension, judges, and calibration, attach a Judge run during your eval pass and populate `outcome.judgeScores.perJudge[judgeName][dimension] = score`. See `docs/insight-report.md` for the expected shape.',\n evidencePath: 'judges',\n })\n }\n\n if (ctx.lift) {\n if (!ctx.lift.decisionEligible) {\n out.push({\n priority: 'high',\n kind: 'expand-corpus',\n title: `Inconclusive — ${ctx.lift.n} paired runs; ${ctx.lift.minimumRequired} required`,\n detail: `The bootstrap interval is descriptive below ${ctx.lift.minimumRequired} paired observations and cannot support a ship decision.`,\n evidencePath: 'lift',\n })\n } else {\n const pairedEffect =\n ctx.lift.cohensD === null ? 'undefined (zero delta variance)' : ctx.lift.cohensD.toFixed(2)\n const pairedP =\n ctx.lift.pValue === null ? 'undefined (zero delta variance)' : ctx.lift.pValue.toFixed(4)\n const requiredRuns =\n ctx.lift.requiredN === null ? 'not estimable' : `~${ctx.lift.requiredN} paired runs`\n // A ZERO-WIDTH interval never reads as \"ship\": n identical paired deltas\n // make every resample identical, so `[g, g]` clears any threshold below g\n // and `[0, 0]` clears any negative `decisionThreshold`, on no spread at\n // all. It falls through to the inconclusive/hold arms, which is where a\n // sample carrying no information about its own error belongs.\n const decisive = !zeroWidth(ctx.lift.ci95) && ctx.lift.ci95[0] > ctx.threshold\n const inconclusive = ctx.lift.ci95[0] <= ctx.threshold && ctx.lift.ci95[1] > ctx.threshold\n if (decisive) {\n out.push({\n priority: 'critical',\n kind: 'ship',\n title: `Ship — lift ${ctx.lift.delta.toFixed(3)} (95% CI ${ctx.lift.ci95[0].toFixed(3)}..${ctx.lift.ci95[1].toFixed(3)})`,\n detail: `Holdout lift exceeds threshold ${ctx.threshold} with 95% bootstrap confidence (n=${ctx.lift.n}, p=${pairedP}, paired d=${pairedEffect}).`,\n evidencePath: 'lift',\n })\n } else if (inconclusive) {\n out.push({\n priority: 'high',\n kind: 'expand-corpus',\n title: `Inconclusive — required sample is ${requiredRuns} (have ${ctx.lift.n}) at current effect size`,\n detail: `CI straddles threshold. Current MDE at 80% power is ${ctx.lift.mde.toFixed(3)}; observed delta is ${ctx.lift.delta.toFixed(3)}.`,\n evidencePath: 'lift',\n })\n } else {\n out.push({\n priority: 'critical',\n kind: 'hold',\n title: `Hold — lift CI lower bound ${ctx.lift.ci95[0].toFixed(3)} is at or below threshold ${ctx.threshold}`,\n detail: `Bootstrap CI provides no statistical evidence the candidate is better. Consider tightening the mutation or expanding the holdout.`,\n evidencePath: 'lift',\n })\n }\n }\n }\n\n if (ctx.contamination && ctx.contamination.leaks > 0) {\n out.push({\n priority: 'critical',\n kind: 'fix',\n title: `${ctx.contamination.leaks} canary leak${ctx.contamination.leaks === 1 ? '' : 's'} detected`,\n detail: `Holdout integrity is compromised. The lift number is unreliable until you investigate.`,\n evidencePath: 'contamination',\n })\n }\n\n if (ctx.interRater && ctx.interRater.kappa < 0.5) {\n out.push({\n priority: 'high',\n kind: 'recalibrate',\n title: `Inter-rater weighted kappa ${ctx.interRater.kappa.toFixed(2)} is below 0.5`,\n detail:\n 'Raters disagree on what good looks like. Review the largest disagreement cases and refine the rubric before automating these decisions.',\n evidencePath: 'interRater',\n })\n }\n\n if (ctx.failureClusters && ctx.failureClusters.clusters.length > 0) {\n const top = ctx.failureClusters.clusters[0]!\n out.push({\n priority: 'high',\n kind: 'investigate',\n title: `Top failure cluster: ${top.name} (${(top.share * 100).toFixed(0)}% of failures)`,\n detail: `${ctx.failureClusters.totalFailures} runs failed. The largest cluster groups ${top.exemplars.length} exemplars under '${top.name}'.`,\n evidencePath: 'failureClusters.clusters[0]',\n })\n }\n\n if (ctx.outcomeCorrelation && Math.abs(ctx.outcomeCorrelation.spearman) < 0.3) {\n out.push({\n priority: 'medium',\n kind: 'recalibrate',\n title: `Judge scores decoupled from ${ctx.outcomeCorrelation.metric} (Spearman ρ=${ctx.outcomeCorrelation.spearman.toFixed(2)})`,\n detail: `Your judges score what they were trained to score, but it isn't predicting downstream ${ctx.outcomeCorrelation.metric}. Consider retraining the judge against ${ctx.outcomeCorrelation.metric} as the gold signal.`,\n evidencePath: 'outcomeCorrelation',\n })\n }\n\n return out\n}\n\n// ── Re-export pareto figure spec for hosted-side rendering ─────────\n\nexport type { ParetoFigureSpec }\n"],"mappings":";;;;;;;;AA4BA,SAAgB,cAAc,QAAgB,WAA4C;CACxF,MAAM,QAAsB,CAAC;CAC7B,KAAK,MAAM,KAAK,WAAW;EACzB,IAAI,CAAC,EAAE,QAAQ;EACf,IAAI,OAAO,SAAS,EAAE,MAAM,GAC1B,MAAM,KAAK;GAAE,YAAY,EAAE;GAAI,QAAQ,EAAE;GAAQ,UAAU,QAAQ,QAAQ,EAAE,MAAM;EAAE,CAAC;CAE1F;CACA,OAAO;AACT;;;;;;;;;;;;;;;;;;;;;;AAuBA,SAAgB,sBACd,QACA,UACmB;CACnB,MAAM,UAAU,SAAS,oBAAoB,SAAS;CACtD,IAAI,CAAC,SAAS,OAAO;CACrB,MAAM,MAAM,eAAe,QAAQ,OAAO;CAC1C,IAAI,CAAC,KAAK,OAAO;CACjB,OAAO;EACL,YAAY,SAAS;EACrB,QAAQ;EACR,UAAU,QAAQ,QAAQ,GAAG;CAC/B;AACF;;;;;;;;;;AAWA,SAAgB,sBACd,OACc;CACd,MAAM,QAAsB,CAAC;CAC7B,KAAK,MAAM,KAAK,OAAO;EACrB,MAAM,OAAO,sBAAsB,EAAE,QAAQ,EAAE,QAAQ;EACvD,IAAI,MAAM,MAAM,KAAK;GAAE,GAAG;GAAM,OAAO,EAAE,SAAS,KAAK;EAAM,CAAC;CAChE;CACA,OAAO;AACT;;;;;;AAOA,SAAS,eAAe,QAAgB,SAAgC;CACtE,MAAM,KAAK,cAAc,OAAO;CAChC,IAAI,IAAI;EACN,MAAM,IAAI,OAAO,MAAM,EAAE;EACzB,OAAO,KAAK,EAAE,EAAE,CAAC,SAAS,IAAI,EAAE,KAAK;CACvC;CACA,OAAO,OAAO,SAAS,OAAO,IAAI,UAAU;AAC9C;AAEA,SAAS,cAAc,SAAgC;CACrD,IAAI,QAAQ,SAAS,KAAK,QAAQ,OAAO,KAAK,OAAO;CACrD,MAAM,OAAO,QAAQ,YAAY,GAAG;CACpC,IAAI,QAAQ,GAAG,OAAO;CACtB,MAAM,OAAO,QAAQ,MAAM,GAAG,IAAI;CAClC,MAAM,QAAQ,QAAQ,MAAM,OAAO,CAAC;CACpC,IAAI,CAAC,cAAc,KAAK,KAAK,GAAG,OAAO;CACvC,IAAI;EACF,OAAO,IAAI,OAAO,MAAM,KAAK;CAC/B,QAAQ;EACN,OAAO;CACT;AACF;;;;;;;AAQA,eAAsB,eACpB,OACA,WACuB;CACvB,MAAM,UAAU,UAAU,QAAQ,MAAM,CAAC,CAAC,EAAE,MAAM;CAClD,IAAI,QAAQ,WAAW,GAAG,OAAO,CAAC;CAClC,MAAM,QAAQ,MAAM,SAAS,KAAK;CAClC,MAAM,QAAsB,CAAC;CAC7B,KAAK,MAAM,QAAQ,OAAO;EACxB,MAAM,SAAS,KAAK,UAAU;EAC9B,KAAK,MAAM,KAAK,SACd,IAAI,EAAE,UAAU,OAAO,SAAS,EAAE,MAAM,GACtC,MAAM,KAAK;GACT,YAAY,EAAE;GACd,QAAQ,EAAE;GACV,OAAO,KAAK;GACZ,UAAU,QAAQ,QAAQ,EAAE,MAAM;EACpC,CAAC;CAGP;CACA,OAAO;AACT;AAEA,IAAa,iBAAb,MAA4B;CAC1B;CACA,YAAgF,CAAC;CAEjF,YAAY,WAA8B;EACxC,KAAK,YAAY;CACnB;;CAGA,IAAI,YAAoB,SAAsD;EAC5E,IAAI,YAAY,gBAAgB,YAAY,aAC1C,MAAM,IAAI,MACR,wEAAwE,SAC1E;EAEF,MAAM,IAAI,KAAK,UAAU,MAAM,MAAM,EAAE,OAAO,UAAU;EACxD,IAAI,CAAC,GAAG,MAAM,IAAI,MAAM,qBAAqB,WAAW,YAAY;EACpE,KAAK,UAAU,KAAK;GAAE;GAAY;GAAS,IAAI,KAAK,IAAI;EAAE,CAAC;EAC3D,OAAO;CACT;CAEA,eAAmF;EACjF,OAAO,KAAK;CACd;AACF;AAEA,SAAS,QAAQ,QAAgB,QAAwB;CACvD,MAAM,KAAK,OAAO,QAAQ,MAAM;CAChC,IAAI,KAAK,GAAG,OAAO;CACnB,MAAM,QAAQ,KAAK,IAAI,GAAG,KAAK,EAAE;CACjC,MAAM,MAAM,KAAK,IAAI,OAAO,QAAQ,KAAK,OAAO,SAAS,EAAE;CAC3D,QAAQ,QAAQ,IAAI,MAAM,MAAM,OAAO,MAAM,OAAO,GAAG,KAAK,MAAM,OAAO,SAAS,MAAM;AAC1F;;;;ACzDA,SAAgB,mBAAmB,MAAkD;CACnF,MAAM,OAAO,KAAK,KAAK,IAAI,iBAAiB;CAE5C,OAAO;EACL,WAAW,wBAAwB,MAFxB,KAAK,iBAAiB,EAEY;EAC7C,gBAAgB,wBAAwB,IAAI;CAC9C;AACF;;;;AAKA,SAAS,UAAU,IAAwC;CACzD,OAAO,CAAC,OAAO,SAAS,GAAG,EAAE,KAAK,CAAC,OAAO,SAAS,GAAG,EAAE,KAAK,GAAG,OAAO,GAAG;AAC5E;AAEA,eAAsB,YAAY,MAAkD;CAClF,MAAM,OAAO,KAAK,KAAK,IAAI,iBAAiB;CAC5C,MAAM,OAAO,KAAK,iBAAiB;CACnC,MAAM,YAAY,KAAK,qBAAqB;CAC5C,IAAI,CAAC,OAAO,SAAS,SAAS,GAC5B,MAAM,IAAI,MAAM,sDAAsD,WAAW;CAEnF,MAAM,QAAQ,aAAa,MAAM,KAAK,SAAS,MAAM;CAErD,MAAM,mBAAmB,KACtB,KAAK,OAAO;EAAE,OAAO,EAAE;EAAO,OAAO,YAAY,GAAG,KAAK;CAAE,EAAE,CAAC,CAC9D,QAAQ,MAAM,OAAO,SAAS,EAAE,KAAK,CAAC;CACzC,MAAM,YAAY,eAChB,iBAAiB,KAAK,MAAM,EAAE,KAAK,GACnC,MACA,gBACF;CAEA,MAAM,eAAe,oBAAoB,MAAM,IAAI;CACnD,MAAM,EAAE,WAAW,gBAAgB,eAAe,mBAAmB;EACnE;EACA,eAAe;CACjB,CAAC;CACD,MAAM,gBAAgB,KAAK,QAAQ,QAAQ,IAAI,eAAe,SAAS,YAAY;CACnF,MAAM,QAAQ,cAAc,KAAK,MAAM,EAAE,OAAO,CAAC,CAAC,OAAO,cAAc;CACvE,MAAM,WAAW,eAAe,OAAO,IAAI;CAC3C,MAAM,SAAS,YAAY,eAAe,EAAE,MAAM,CAAC;CACnD,MAAM,WAA+C,CAAC;CACtD,IAAI,WAAW,WAAW,IAAI,GAC5B,SAAS,OAAO,qBAAqB,MAAM,UAAU;MAChD,IAAI,MAAM,WAAW,KAAK,MAAM,OAAO,MAAM,MAAM,CAAC,GACzD,SAAS,OAAO,OAAO,KAAK,OAAO;CAErC,IAAI,OAAO,OAAO,SAAS,GACzB,SAAS,SACP,OAAO,OAAO,WAAW,IACrB,uCACA;CAER,MAAM,cAAc;EAClB,MAAM;EACN;EACA;EACA,GAAI,SAAS,QAAQ,SAAS,SAAS,EAAE,SAAS,IAAI,CAAC;CACzD;CAEA,MAAM,SAAS,qBAAqB,IAAI;CAExC,MAAM,aAAa,KAAK,cAAc,kBAAkB,KAAK,WAAW,IAAI,KAAA;CAE5E,MAAM,OAAO,YAAY,MAAM,KAAK,qBAAqB,KAAK,sBAAsB,KAAK;CAEzF,MAAM,kBAAkB,KAAK,UACzB,MAAM,uBAAuB,MAAM,KAAK,SAAS,KAAK,IACtD,KAAA;CAEJ,MAAM,iBAAiB,sBAAsB,MAAM,KAAK;CAExD,MAAM,gBAAgB,KAAK,kBACvB,qBAAqB,MAAM,KAAK,eAAe,IAC/C,KAAA;CAEJ,MAAM,qBAAqB,KAAK,gBAC5B,0BAA0B,MAAM,KAAK,eAAe,KAAK,IACzD,KAAA;CAEJ,MAAM,UAAU,sBAAsB,WAAW,MAAM,aAAa;CAEpE,MAAM,wBAAwB,KAAK,eAC/B,6BAA6B,MAAM,KAAK,cAAc,OAAO,KAAK,aAAa,IAC/E,KAAA;CAEJ,MAAM,kBAAkB,qBAAqB;EAC3C;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;CACF,CAAC;CAED,OAAO;EACL,GAAG,KAAK;EACR;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,GAAI,iBAAiB,EAAE,eAAe,IAAI,CAAC;EAC3C,GAAI,wBAAwB,EAAE,sBAAsB,IAAI,CAAC;EACzD;CACF;AACF;AAEA,SAAS,wBAAwB,MAAmB,MAAgC;CAClF,MAAM,gBAAgB,KAAK,SAAS,QAAQ;EAC1C,MAAM,QAAQ,oBAAoB,GAAG;EACrC,OAAO,QAAQ,CAAC;GAAE;GAAO,SAAS,UAAU,KAAK,oBAAoB;EAAE,CAAC,IAAI,CAAC;CAC/E,CAAC;CACD,MAAM,iBAAiB,cAAc,SAAS,QAC5C,IAAI,YAAY,KAAA,IAAY,CAAC,IAAI,OAAO,IAAI,CAAC,CAC/C;CACA,MAAM,8BAAc,IAAI,IAAoB;CAC5C,IAAI,qBAAqB;CACzB,IAAI,uBAAuB;CAC3B,IAAI,qBAAqB;CACzB,IAAI,kBAAkB;CACtB,IAAI,yBAAyB;CAC7B,MAAM,mBAAuD;EAC3D,WAAW;EACX,QAAQ;EACR,WAAW;EACX,YAAY;EACZ,SAAS;CACX;CACA,MAAM,0BAAoF;EACxF,WAAW;GAAE,YAAY;GAAG,eAAe;GAAG,YAAY;EAAE;EAC5D,QAAQ;GAAE,YAAY;GAAG,eAAe;GAAG,YAAY;EAAE;EACzD,WAAW;GAAE,YAAY;GAAG,eAAe;GAAG,YAAY;EAAE;EAC5D,YAAY;GAAE,YAAY;GAAG,eAAe;GAAG,YAAY;EAAE;EAC7D,SAAS;GAAE,YAAY;GAAG,eAAe;GAAG,YAAY;EAAE;CAC5D;CACA,IAAI,gBAAgB;CACpB,IAAI,kBAAkB;CACtB,IAAI,yBAAyB;CAE7B,KAAK,MAAM,OAAO,MAAM;EACtB,YAAY,IAAI,IAAI,QAAQ,YAAY,IAAI,IAAI,KAAK,KAAK,KAAK,CAAC;EAChE,MAAM,kBAAkB,IAAI;EAC5B,iBAAiB,oBAAoB;EACrC,MAAM,aAAa,oBAAoB,KAAK,gBAAgB;EAC5D,IAAI,eAAe,KAAA,GAAW;GAC5B,mBAAmB;GACnB,0BAA0B;EAC5B;EACA,MAAM,QAAQ,IAAI;EAClB,KACG,cAAc,KAAK,KACpB,MAAM,QAAQ,KACd,MAAM,SAAS,MACd,MAAM,UAAU,KAAK,MACrB,MAAM,cAAc,KAAK,GAE1B,iBAAiB;EAEnB,MAAM,cAAc,6BAA6B,GAAG;EACpD,IAAI,gBAAgB,KAAA,GAAW;GAC7B,wBAAwB;GACxB,sBAAsB;GACtB,IAAI,cAAc,GAAG;IACnB,sBAAsB;IACtB,wBAAwB,gBAAgB,CAAC,cAAc;GACzD,OAAO,wBAAwB,gBAAgB,CAAC,iBAAiB;EACnE,OAAO,wBAAwB,gBAAgB,CAAC,cAAc;EAC9D,MAAM,qBAAqB,oBAAoB,KAAK,kBAAkB;EACtE,IAAI,uBAAuB,KAAA,GAAW;GACpC,mBAAmB;GACnB,0BAA0B;EAC5B;CACF;CAEA,OAAO;EACL,YAAY,eACV,KAAK,KAAK,QAAQ,IAAI,MAAM,GAC5B,IACF;EACA,SAAS,eACP,KAAK,QAAQ,QAAQ,IAAI,YAAY,KAAA,CAAS,CAAC,CAAC,KAAK,QAAQ,IAAI,OAAQ,GACzE,IACF;EACA,YAAY,oBACV,KAAK,KAAK,QAAQ,IAAI,UAAU,GAChC,IACF;EACA,gBAAgB;GACd,MAAM,cAAc;GACpB,YAAY,oBACV,cAAc,KAAK,QAAQ,IAAI,KAAK,GACpC,IACF;GACA,SAAS,eAAe,gBAAgB,IAAI;GAC5C,cAAc,eAAe,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC;EACxE;EACA,QAAQ,CAAC,GAAG,YAAY,QAAQ,CAAC,CAAC,CAC/B,KAAK,CAAC,OAAO,YAAY;GAAE;GAAO,MAAM;EAAM,EAAE,CAAC,CACjD,MAAM,MAAM,UAAU,MAAM,OAAO,KAAK,QAAQ,KAAK,MAAM,cAAc,MAAM,KAAK,CAAC;EACxF,YAAY;GACV,MAAM;GACN,QAAQ;GACR,eAAe;EACjB;EACA,iBAAiB;GACf,MAAM;GACN,UAAU,qBAAqB,IAAI,qBAAqB,qBAAqB;GAC7E,QAAQ;GACR,eAAe;GACf;GACA;GACA,mBAAmB;EACrB;EACA;CACF;AACF;AAEA,SAAS,6BAA6B,KAAoC;CACxE,OAAO,oBAAoB,KAAK,uBAAuB;AACzD;AAEA,SAAS,oBAAoB,KAAgB,KAAiC;CAC5E,MAAM,QAAQ,UAAU,KAAK,GAAG;CAChC,OAAO,UAAU,KAAA,KAAa,OAAO,UAAU,KAAK,KAAK,SAAS,IAAI,QAAQ,KAAA;AAChF;AAEA,SAAS,oBAAoB,QAAyB,MAAiC;CACrF,MAAM,YAAY,OAAO,SAAS,UAChC,MAAM,cAAc,KAAA,IAAY,CAAC,MAAM,SAAS,IAAI,CAAC,CACvD;CACA,MAAM,SAAS,OAAO,SAAS,UAAW,MAAM,WAAW,KAAA,IAAY,CAAC,MAAM,MAAM,IAAI,CAAC,CAAE;CAC3F,MAAM,aAAa,OAAO,SAAS,UACjC,MAAM,eAAe,KAAA,IAAY,CAAC,MAAM,UAAU,IAAI,CAAC,CACzD;CACA,OAAO;EACL,OAAO,eACL,OAAO,KAAK,UAAU,MAAM,KAAK,GACjC,IACF;EACA,QAAQ,eACN,OAAO,KAAK,UAAU,MAAM,MAAM,GAClC,IACF;EACA,WAAW,eAAe,WAAW,IAAI;EACzC,QAAQ,eAAe,QAAQ,IAAI;EACnC,YAAY,eAAe,YAAY,IAAI;EAC3C,QAAQ;GACN,OAAO,OAAO,QAAQ,OAAO,UAAU,QAAQ,MAAM,OAAO,CAAC;GAC7D,QAAQ,OAAO,QAAQ,OAAO,UAAU,QAAQ,MAAM,QAAQ,CAAC;GAC/D,WAAW,UAAU,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC;GAC9D,QAAQ,OAAO,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC;GACxD,YAAY,WAAW,QAAQ,OAAO,UAAU,QAAQ,OAAO,CAAC;EAClE;CACF;AACF;AAEA,SAAS,oBAAoB,KAA2C;CACtE,MAAM,QAAQ,UAAU,KAAK,yBAAyB;CACtD,MAAM,SAAS,UAAU,KAAK,6BAA6B;CAC3D,MAAM,YAAY,UAAU,KAAK,4BAA4B;CAC7D,MAAM,SAAS,UAAU,KAAK,yBAAyB;CACvD,MAAM,aAAa,UAAU,KAAK,8BAA8B;CAChE,IACE,UAAU,KAAA,KACV,WAAW,KAAA,KACX,cAAc,KAAA,KACd,WAAW,KAAA,KACX,eAAe,KAAA,GAEf,OAAO,KAAA;CACT,OAAO;EACL,OAAO,SAAS;EAChB,QAAQ,UAAU;EAClB,GAAI,cAAc,KAAA,IAAY,EAAE,UAAU,IAAI,CAAC;EAC/C,GAAI,WAAW,KAAA,IAAY,EAAE,OAAO,IAAI,CAAC;EACzC,GAAI,eAAe,KAAA,IAAY,EAAE,WAAW,IAAI,CAAC;CACnD;AACF;AAEA,SAAS,UAAU,KAAgB,KAAiC;CAClE,MAAM,QAAQ,IAAI,QAAQ,IAAI;CAC9B,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK,KAAK,SAAS,IAAI,QAAQ,KAAA;AACrF;AAEA,SAAS,wBAAwB,MAA0C;CACzE,MAAM,UAAiC;EACrC,UAAU;GAAE,GAAG;GAAG,UAAU;EAAE;EAC9B,WAAW;GAAE,GAAG;GAAG,UAAU;EAAE;EAC/B,YAAY,EAAE,GAAG,EAAE;EACnB,eAAe;CACjB;CACA,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,OAAO,IAAI;EACjB,IAAI,KAAK,SAAS,cAChB,QAAQ,WAAW,KAAK;OACnB;GACL,QAAQ,KAAK,KAAK,CAAC,KAAK;GACxB,QAAQ,KAAK,KAAK,CAAC,YAAY,KAAK;EACtC;CACF;CACA,MAAM,QAAQ,QAAQ,SAAS,IAAI,QAAQ,UAAU;CACrD,QAAQ,gBAAgB,KAAK,SAAS,IAAI,QAAQ,KAAK,SAAS;CAChE,OAAO;AACT;AAEA,SAAS,qBAAqB,MAAmB,YAA2C;CAC1F,MAAM,aAAa,WAAW,WAAW;CACzC,MAAM,QAAQ,WAAW,SAAS,IAAI,WAAW,UAAU;CAC3D,IAAI,eAAe,KAAK,QACtB,OAAO,+BAA+B,KAAK,OAAO;CAEpD,OAAO,2BAA2B,WAAW,GAAG,KAAK,OAAO,mDAAmD,MAAM,GAAG,KAAK,OAAO,aAAa,WAAW,SAAS,EAAE,aAAa,WAAW,UAAU,EAAE;AAC7M;;;;;;;AAQA,SAAS,sBACP,MACA,OACiC;CACjC,MAAM,yBAAS,IAAI,IAA0B;CAC7C,KAAK,MAAM,KAAK,MAAM;EACpB,IAAI,CAAC,cAAc,GAAG,KAAK,GAAG;EAC9B,MAAM,MACJ,EAAE,iBAAiB,KAAA,KAAa,EAAE,iBAAiB,YAAY,EAAE,eAAe;EAClF,OAAO,IAAI,MAAM,OAAO,IAAI,GAAG,KAAK,KAAK,CAAC;CAC5C;CACA,IAAI,OAAO,SAAS,GAAG,OAAO,KAAA;CAC9B,MAAM,IAAI,KAAK;CACf,OAAO,CAAC,GAAG,OAAO,QAAQ,CAAC,CAAC,CACzB,KAAK,CAAC,cAAc,YAAY;EAC/B;EACA;EACA,OAAO,IAAI,IAAI,QAAQ,IAAI;CAC7B,EAAE,CAAC,CACF,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,SAAS,EAAE,aAAa,cAAc,EAAE,YAAY,CAAC;AACrF;AAUA,SAAS,6BACP,SACA,UACA,OACA,aACmC;CACnC,IAAI,QAAQ,WAAW,KAAK,SAAS,WAAW,GAAG,OAAO,KAAA;CAE1D,MAAM,UAAuC,CAAC;CAC9C,MAAM,aAA8C,CAAC;CAErD,MAAM,mBAAmB,QACtB,KAAK,MAAM,YAAY,GAAG,KAAK,CAAC,CAAC,CACjC,OAAO,OAAO,QAAQ;CACzB,MAAM,oBAAoB,SACvB,KAAK,MAAM,YAAY,GAAG,KAAK,CAAC,CAAC,CACjC,OAAO,OAAO,QAAQ;CACzB,IAAI,iBAAiB,SAAS,KAAK,kBAAkB,SAAS,GAAG;EAC/D,QAAQ,YAAY,aAAa,mBAAmB,gBAAgB;EACpE,WAAW,YAAY;CACzB;CAEA,MAAM,cAAc,gBAAgB,OAAO;CAC3C,MAAM,eAAe,gBAAgB,QAAQ;CAC7C,IAAI,YAAY,SAAS,KAAK,aAAa,SAAS,GAAG;EACrD,QAAQ,OAAO,aAAa,cAAc,WAAW;EACrD,WAAW,OAAO;CACpB;CAEA,MAAM,aAAa,QAAQ,KAAK,MAAM,EAAE,MAAM,CAAC,CAAC,OAAO,OAAO,QAAQ;CACtE,MAAM,cAAc,SAAS,KAAK,MAAM,EAAE,MAAM,CAAC,CAAC,OAAO,OAAO,QAAQ;CACxE,IAAI,WAAW,SAAS,KAAK,YAAY,SAAS,GAAG;EACnD,QAAQ,WAAW,aAAa,aAAa,UAAU;EACvD,WAAW,WAAW;CACxB;CAEA,MAAM,aAAa,QAChB,KAAK,OAAO,EAAE,WAAW,SAAS,MAAM,EAAE,WAAW,UAAU,EAAE,CAAC,CAClE,OAAO,OAAO,QAAQ;CACzB,MAAM,cAAc,SACjB,KAAK,OAAO,EAAE,WAAW,SAAS,MAAM,EAAE,WAAW,UAAU,EAAE,CAAC,CAClE,OAAO,OAAO,QAAQ;CACzB,IAAI,WAAW,SAAS,KAAK,YAAY,SAAS,GAAG;EACnD,QAAQ,aAAa,aAAa,aAAa,UAAU;EACzD,WAAW,aAAa;CAC1B;CAKA,MAAM,cAAc,oBAAoB,OAAO;CAC/C,MAAM,eAAe,oBAAoB,QAAQ;CACjD,KAAK,MAAM,OAAO,OAAO,KAAK,WAAW,GAAG;EAC1C,MAAM,IAAI,aAAa;EACvB,MAAM,IAAI,YAAY;EACtB,IAAI,CAAC,KAAK,EAAE,WAAW,KAAK,CAAC,KAAK,EAAE,WAAW,GAAG;EAClD,QAAQ,OAAO,SAAS,aAAa,GAAG,CAAC;EACzC,WAAW,OAAO,SAAS;CAC7B;CAEA,MAAM,mBAA6B,CAAC;CACpC,MAAM,kBAA4B,CAAC;CACnC,MAAM,sBAAgC,CAAC;CACvC,KAAK,MAAM,CAAC,MAAM,UAAU,OAAO,QAAQ,OAAO,GAAG;EACnD,IAAI,MAAM,WAAW,MAAM;GACzB,oBAAoB,KAAK,IAAI;GAC7B;EACF;EACA,IAAI,CAAC,MAAM,aAAa;EAGxB,KAFY,WAAW,SAAS,wBACT,qBAAqB,MAAM,QAAQ,IAAI,MAAM,QAAQ,GAChE,gBAAgB,KAAK,IAAI;OAChC,iBAAiB,KAAK,IAAI;CACjC;CAEA,OAAO;EACL,WAAW,SAAS;EACpB,UAAU,QAAQ;EAClB,GAAI,cAAc,EAAE,YAAY,IAAI,CAAC;EACrC;EACA;EACA;EACA;CACF;AACF;AAEA,SAAS,gBAAgB,MAA6B;CACpD,OAAO,KACJ,QAAQ,QAAQ,IAAI,eAAe,SAAS,YAAY,CAAC,CACzD,KAAK,QAAQ,IAAI,OAAO,CAAC,CACzB,OAAO,cAAc;AAC1B;AAEA,SAAS,eAAe,OAAiC;CACvD,OAAO,OAAO,UAAU,YAAY,OAAO,SAAS,KAAK;AAC3D;;AAGA,SAAS,oBAAoB,MAA6C;CACxE,MAAM,MAAgC,CAAC;CACvC,KAAK,MAAM,KAAK,MAAM;EACpB,MAAM,SAAS,EAAE,QAAQ,aAAa;EACtC,IAAI,CAAC,QAAQ;EACb,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,MAAM,GAAG;GACjD,IAAI,CAAC,OAAO,SAAS,KAAK,GAAG;GAC7B,IAAI,CAAC,IAAI,MAAM,IAAI,OAAO,CAAC;GAC3B,IAAI,IAAI,CAAC,KAAK,KAAe;EAC/B;CACF;CACA,OAAO;AACT;;AAGA,SAAS,aAAa,UAAoB,SAAgC;CACxE,MAAM,SAAS,YAAY,UAAU,OAAO;CAC5C,MAAM,OAAO;EACX,SAAS,OAAO;EAChB,UAAU,OAAO;EACjB,OAAO,OAAO;EACd,WAAW,SAAS;EACpB,UAAU,QAAQ;CACpB;CACA,IAAI,OAAO,WAAW,MACpB,OAAO;EACL,GAAG;EACH,QAAQ,OAAO;EACf,MAAM;EACN,QAAQ;EACR,SAAS;EACT,aAAa;CACf;CAEF,OAAO;EACL,GAAG;EACH,QAAQ;EACR,MAAM,OAAO;EACb,QAAQ,OAAO;EACf,SAAS,OAAO;EAChB,aAAa,OAAO,IAAI,OAAQ,KAAK,IAAI,OAAO,OAAO,KAAK;CAC9D;AACF;AAIA,SAAS,aACP,MACA,MACsB;CACtB,IAAI,SAAS,QAAQ,OAAO;CAE5B,OADmB,KAAK,MAAM,MAAM,OAAO,SAAS,mBAAmB,GAAG,SAAS,CAAC,CACpE,IAAI,YAAY;AAClC;;;;;;;AAQA,SAAS,YAAY,KAAgB,OAAqC;CAIxE,MAAM,QAAQ,mBAAmB,KAAK,KAAK;CAC3C,OAAO,OAAO,SAAS,KAAK,IAAK,QAAmB;AACtD;AAIA,SAAS,eACP,QACA,MACA,SACoB;CACpB,IAAI,OAAO,WAAW,GACpB,OAAO;EACL,GAAG;EACH,MAAM;EACN,KAAK;EACL,KAAK;EACL,QAAQ;EACR,KAAK;EACL,KAAK;EACL,WAAW,CAAC;CACd;CAEF,MAAM,SAAS,CAAC,GAAG,MAAM,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC/C,MAAM,IAAI,OAAO;CACjB,MAAM,OAAO,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;CACjD,MAAM,WAAW,OAAO,QAAQ,GAAG,MAAM,KAAK,IAAI,SAAS,GAAG,CAAC,IAAI;CACnE,MAAM,SAAS,KAAK,KAAK,QAAQ;CACjC,MAAM,WAAW,UACb,CAAC,GAAG,OAAO,CAAC,CAAC,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK,CAAC,CAAC,MAAM,GAAG,KAAK,IAAI,GAAG,QAAQ,MAAM,CAAC,IACnF,KAAA;CACJ,OAAO;EACL;EACA;EACA,KAAK,WAAW,QAAQ,EAAG;EAC3B,KAAK,WAAW,QAAQ,GAAI;EAC5B;EACA,KAAK,OAAO;EACZ,KAAK,OAAO,IAAI;EAChB,WAAW,UAAU,QAAQ,IAAI;EACjC,GAAI,WAAW,EAAE,SAAS,IAAI,CAAC;CACjC;AACF;AAEA,SAAS,WAAW,QAAkB,GAAmB;CACvD,IAAI,OAAO,WAAW,GAAG,OAAO;CAChC,IAAI,OAAO,WAAW,GAAG,OAAO,OAAO;CACvC,MAAM,OAAO,OAAO,SAAS,KAAK;CAClC,MAAM,KAAK,KAAK,MAAM,GAAG;CACzB,MAAM,KAAK,KAAK,KAAK,GAAG;CACxB,IAAI,OAAO,IAAI,OAAO,OAAO;CAC7B,MAAM,IAAI,MAAM;CAChB,OAAO,OAAO,OAAQ,IAAI,KAAK,OAAO,MAAO;AAC/C;;;;AAKA,SAAS,UAAU,QAAkB,MAA+C;CAClF,IAAI,OAAO,WAAW,KAAK,OAAO,GAAG,OAAO,CAAC;CAC7C,MAAM,MAAM,OAAO;CACnB,MAAM,MAAM,OAAO,OAAO,SAAS;CACnC,IAAI,QAAQ,KAAK,OAAO,CAAC;EAAE,IAAI;EAAK,IAAI;EAAK,OAAO,OAAO;CAAO,CAAC;CACnE,MAAM,SAAS,MAAM,OAAO;CAC5B,MAAM,MAAuC,CAAC;CAC9C,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,KAAK;EAC7B,MAAM,KAAK,MAAM,IAAI;EACrB,MAAM,KAAK,MAAM,OAAO,IAAI,MAAM,KAAK;EACvC,IAAI,KAAK;GAAE;GAAI;GAAI,OAAO;EAAE,CAAC;CAC/B;CACA,KAAK,MAAM,KAAK,QAAQ;EACtB,MAAM,MAAM,KAAK,IAAI,OAAO,GAAG,KAAK,OAAO,IAAI,OAAO,KAAK,CAAC;EAC5D,IAAI,IAAI,CAAE;CACZ;CACA,OAAO;AACT;AAEA,SAAS,oBAAoB,MAAmB,MAAkD;CAKhG,MAAM,wBAAQ,IAAI,IAAsB;CACxC,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,SAAS,IAAI,QAAQ;EAC3B,IAAI,CAAC,QAAQ;EACb,KAAK,MAAM,CAAC,KAAK,UAAU,OAAO,QAAQ,OAAO,cAAc,CAAC,CAAC,GAAG;GAClE,IAAI,CAAC,OAAO,SAAS,KAAK,GAAG;GAC7B,MAAM,MAAM,MAAM,IAAI,GAAG,KAAK,CAAC;GAC/B,IAAI,KAAK,KAAK;GACd,MAAM,IAAI,KAAK,GAAG;EACpB;CACF;CACA,MAAM,MAA0C,CAAC;CACjD,KAAK,MAAM,CAAC,KAAK,WAAW,OAAO,IAAI,OAAO,eAAe,QAAQ,IAAI;CACzE,OAAO;AACT;AAIA,SAAS,qBAAqB,MAAiD;CAI7E,MAAM,MAAoC,CAAC;CAC3C,MAAM,0BAAU,IAAI,IAAsB;CAC1C,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,SAAS,IAAI,QAAQ;EAC3B,IAAI,CAAC,QAAQ,UAAU;EACvB,KAAK,MAAM,CAAC,SAAS,SAAS,OAAO,QAAQ,OAAO,QAAQ,GAAG;GAC7D,MAAM,YAAY,OAAO,OAAO,IAAI,CAAC,CAAC,OAAO,OAAO,QAAQ;GAC5D,IAAI,UAAU,WAAW,GAAG;GAC5B,MAAM,YAAY,UAAU,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,UAAU;GACnE,MAAM,MAAM,QAAQ,IAAI,OAAO,KAAK,CAAC;GACrC,IAAI,KAAK,SAAS;GAClB,QAAQ,IAAI,SAAS,GAAG;EAC1B;CACF;CACA,KAAK,MAAM,CAAC,SAAS,WAAW,SAC9B,IAAI,WAAW;EACb,GAAG,OAAO;EACV,WAAW,OAAO,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,OAAO;CACxD;CAEF,OAAO;AACT;AAIA,SAAS,kBACP,SAC+B;CAC/B,MAAM,wBAAQ,IAAI,IAAqD;CACvE,KAAK,MAAM,KAAK,SAAS;EACvB,IAAI,CAAC,OAAO,SAAS,EAAE,KAAK,GAAG;EAC/B,MAAM,OAAO,MAAM,IAAI,EAAE,KAAK,KAAK,CAAC;EACpC,KAAK,KAAK;GAAE,OAAO,EAAE;GAAO,OAAO,EAAE;EAAM,CAAC;EAC5C,MAAM,IAAI,EAAE,OAAO,IAAI;CACzB;CACA,MAAM,SAAS,IAAI,IAAI,QAAQ,KAAK,MAAM,EAAE,KAAK,CAAC;CAClD,MAAM,eAAyB,CAAC;CAChC,KAAK,MAAM,CAAC,OAAO,iBAAiB,OAAO;EACzC,MAAM,OAAO,IAAI,IAAI,aAAa,KAAK,MAAM,EAAE,KAAK,CAAC;EACrD,IAAI,MAAM;EACV,KAAK,MAAM,KAAK,QAAQ,IAAI,CAAC,KAAK,IAAI,CAAC,GAAG,MAAM;EAChD,IAAI,KAAK,aAAa,KAAK,KAAK;CAClC;CACA,IAAI,OAAO,OAAO,KAAK,aAAa,WAAW,GAAG,OAAO,KAAA;CAEzD,MAAM,YAAY,CAAC,GAAG,MAAM,CAAC,CAAC,KAAK;CACnC,MAAM,UAAkC,CAAC;CACzC,KAAK,IAAI,IAAI,GAAG,IAAI,UAAU,QAAQ,KACpC,KAAK,IAAI,IAAI,IAAI,GAAG,IAAI,UAAU,QAAQ,KAAK;EAC7C,MAAM,IAAI,UAAU;EACpB,MAAM,IAAI,UAAU;EACpB,MAAM,UAAoB,CAAC;EAC3B,MAAM,UAAoB,CAAC;EAC3B,KAAK,MAAM,SAAS,cAAc;GAChC,MAAM,eAAe,MAAM,IAAI,KAAK;GACpC,MAAM,KAAK,aAAa,MAAM,MAAM,EAAE,UAAU,CAAC,CAAC,EAAE;GACpD,MAAM,KAAK,aAAa,MAAM,MAAM,EAAE,UAAU,CAAC,CAAC,EAAE;GACpD,IAAI,OAAO,KAAA,KAAa,OAAO,KAAA,GAAW;IACxC,QAAQ,KAAK,EAAE;IACf,QAAQ,KAAK,EAAE;GACjB;EACF;EACA,MAAM,YAAY,oBAChB,QAAQ,KAAK,OAAO,UAAU,CAAC,OAAO,QAAQ,MAAO,CAAC,GACtD,EAAE,WAAW,EAAE,CACjB;EACA,QAAQ,GAAG,EAAE,IAAI,OAAO,UAAU;CACpC;CAMF,MAAM,YAAY,oBAJH,aAAa,KAAK,UAAU;EACzC,MAAM,iBAAiB,IAAI,IAAI,MAAM,IAAI,KAAK,CAAC,CAAE,KAAK,WAAW,CAAC,OAAO,OAAO,OAAO,KAAK,CAAC,CAAC;EAC9F,OAAO,UAAU,KAAK,UAAU,eAAe,IAAI,KAAK,CAAE;CAC5D,CAC2C,GAAG,EAAE,WAAW,EAAE,CAAC;CAE9D,MAAM,oBAAoB,aACvB,KAAK,UAAU;EACd,MAAM,eAAe,MAAM,IAAI,KAAK;EACpC,MAAM,SAAS,aAAa,KAAK,MAAM,EAAE,KAAK;EAE9C,OAAO;GAAE;GAAO,SAAS;GAAc,OADzB,KAAK,IAAI,GAAG,MAAM,IAAI,KAAK,IAAI,GAAG,MAAM;EACT;CAC/C,CAAC,CAAC,CACD,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK,CAAC,CACjC,MAAM,GAAG,EAAE;CAEd,OAAO;EACL,QAAQ,OAAO;EACf,cAAc,aAAa;EAC3B,OAAO,OAAO,SAAS,UAAU,aAAa,IAAI,UAAU,gBAAgB;EAC5E,KAAK,UAAU;EACf,SAAS,UAAU;EACnB,UAAU,UAAU;EACpB;EACA;CACF;AACF;AAIA,SAAS,YACP,MACA,YACA,aACA,OACyB;CACzB,IAAI,MAAM;CACV,IAAI,MAAM;CACV,IAAI,CAAC,OAAO,CAAC,KAAK;EAGhB,MAAM,MAAM,CAAC,GAAG,IAAI,IAAI,KAAK,KAAK,MAAM,EAAE,WAAW,CAAC,CAAC;EACvD,IAAI,IAAI,WAAW,GAAG,OAAO,KAAA;EAC7B,MAAM,CAAC,KAAK,OAAO;EACnB,MAAM,UAAU,sBACd,KAAK,QAAQ,QAAQ,IAAI,gBAAgB,GAAG,GAC5C,KACF;EACA,MAAM,UAAU,sBACd,KAAK,QAAQ,QAAQ,IAAI,gBAAgB,GAAG,GAC5C,KACF;EACA,IAAI,QAAQ,WAAW,KAAK,QAAQ,WAAW,GAAG,OAAO,KAAA;EACzD,MAAM,QAAQ,KAAK,OAAO;EAC1B,MAAM,QAAQ,KAAK,OAAO;EAC1B,MAAM,SAAS,QAAQ,MAAM;EAC7B,MAAM,SAAS,QAAQ,MAAM;CAC/B;CAEA,MAAM,WAAW,KAAK,QAAQ,MAAM,EAAE,gBAAgB,GAAG;CACzD,MAAM,YAAY,KAAK,QAAQ,MAAM,EAAE,gBAAgB,GAAG;CAC1D,IAAI,SAAS,WAAW,KAAK,UAAU,WAAW,GAAG,OAAO,KAAA;CAI5D,MAAM,UAAU,eAFO,SAAS,QAAQ,QAAQ,OAAO,SAAS,YAAY,KAAK,KAAK,CAAC,CAE3C,GADpB,UAAU,QAAQ,QAAQ,OAAO,SAAS,YAAY,KAAK,KAAK,CAAC,CAC5B,CAAC;CAC9D,MAAM,iBAAiB,QAAQ,MAAM,KAAK,SAAS,YAAY,KAAK,UAAU,KAAK,CAAC;CACpF,MAAM,kBAAkB,QAAQ,MAAM,KAAK,SAAS,YAAY,KAAK,WAAW,KAAK,CAAC;CACtF,IAAI,eAAe,WAAW,GAAG,OAAO,KAAA;CAExC,MAAM,eAAe,KAAK,cAAc;CACxC,MAAM,gBAAgB,KAAK,eAAe;CAC1C,MAAM,QAAQ,gBAAgB;CAE9B,MAAM,YAAY,gBAAgB,gBAAgB,iBAAiB;EACjE,YAAY;EACZ,WAAW;EACX,WAAW;CACb,CAAC;CACD,MAAM,QAAQ,YAAY,gBAAgB,eAAe;CACzD,MAAM,IAAI,eAAe,gBAAgB,eAAe;CACxD,MAAM,MAAM,UAAU;EAAE,SAAS,eAAe;EAAQ,OAAO;EAAK,OAAO;CAAK,CAAC;CACjF,MAAM,YACJ,MAAM,QAAQ,MAAM,IAChB,OACA,yBAAyB;EACvB,QAAQ,KAAK,IAAI,CAAC;EAClB,OAAO;EACP,OAAO;CACT,CAAC;CAEP,OAAO;EACL;EACA;EACA;EACA,MAAM,CAAC,UAAU,KAAK,UAAU,IAAI;EACpC,QAAQ,MAAM;EACd,GAAG,eAAe;EAClB,iBAAA;EACA,kBAAkB,UAAU;EAC5B,kBAAkB,QAAQ,iBAAiB;EAC3C,mBAAmB,QAAQ,kBAAkB;EAC7C,SAAS;EACT;EACA;CACF;AACF;AAEA,SAAS,KAAK,KAAuB;CACnC,OAAO,IAAI,WAAW,IAAI,IAAI,IAAI,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,IAAI;AACrE;AAIA,eAAe,uBACb,MACA,SACA,OAC4C;CAC5C,MAAM,SAAS,KAAK,QAAQ,QAAQ,cAAc,KAAK,KAAK,CAAC;CAC7D,IAAI,OAAO,WAAW,GAAG,OAAO;EAAE,UAAU,CAAC;EAAG,eAAe;CAAE;CAEjE,MAAM,2BAAW,IAAI,IAAoD;CACzE,KAAK,MAAM,OAAO,QAChB,IAAI;EAIF,MAAM,SAAS,MAAM,QAAQ,IAAI,IAAI,OAAO,EAAE,WAAW,IAAI,CAAC;EAC9D,KAAK,MAAM,WAAW,OAAO,UAA8B;GACzD,MAAM,MAAM,QAAQ,QAAQ,QAAQ,cAAc;GAClD,MAAM,IAAI,SAAS,IAAI,GAAG,KAAK;IAAE,WAAW,CAAC;IAAG,OAAO;GAAE;GACzD,IAAI,EAAE,UAAU,SAAS,GAAG,EAAE,UAAU,KAAK,IAAI,KAAK;GACtD,SAAS,IAAI,KAAK,CAAC;EACrB;CACF,QAAQ;EACN,MAAM,IAAI,SAAS,IAAI,eAAe,KAAK;GAAE,WAAW,CAAC;GAAG,OAAO;EAAE;EACrE,IAAI,EAAE,UAAU,SAAS,GAAG,EAAE,UAAU,KAAK,IAAI,KAAK;EACtD,SAAS,IAAI,iBAAiB,CAAC;CACjC;CAEF,MAAM,cAAc,CAAC,GAAG,SAAS,QAAQ,CAAC,CAAC,CAAC,KAAK,CAAC,IAAI,QAAQ;EAC5D;EACA,MAAM;EACN,OAAO,EAAE,UAAU,SAAS,OAAO;EACnC,WAAW,EAAE;CACf,EAAE;CACF,YAAY,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,KAAK;CAC5C,OAAO;EAAE,UAAU;EAAa,eAAe,OAAO;CAAO;AAC/D;AAEA,SAAS,sBAAsB,MAA4B,OAAuC;CAChG,OAAO,KAAK,KAAK,QAAQ,YAAY,KAAK,KAAK,CAAC,CAAC,CAAC,OAAO,OAAO,QAAQ;AAC1E;AAEA,SAAS,cAAc,KAAgB,OAAsC;CAC3E,IAAI,IAAI,iBAAiB,KAAA,KAAa,IAAI,iBAAiB,WAAW,OAAO;CAC7E,MAAM,QAAQ,YAAY,KAAK,KAAK;CACpC,OAAO,OAAO,SAAS,KAAK,KAAK,QAAQ;AAC3C;AAIA,SAAS,qBACP,MACA,UACgC;CAChC,IAAI,QAAQ;CACZ,MAAM,UAAqE,CAAC;CAC5E,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,SAAS,gBAAgB,GAAG;EAClC,IAAI,CAAC,QAAQ;EACb,MAAM,YAAY,cAAc,QAAQ,QAAQ;EAChD,KAAK,MAAM,QAAQ,WAAW;GAC5B;GACA,QAAQ,KAAK;IAAE,OAAO,IAAI;IAAO,QAAQ,KAAK;IAAQ,SAAS,KAAK;GAAS,CAAC;EAChF;CACF;CACA,OAAO;EAAE;EAAO,oBAAoB,UAAU;EAAG;CAAQ;AAC3D;AAEA,SAAS,gBAAgB,KAAoC;CAO3D,MAAM,WAAY,IAA0D;CAC5E,IAAI,OAAO,UAAU,WAAW,UAAU,OAAO,SAAS;CAC1D,IAAI,OAAO,UAAU,SAAS,UAAU,OAAO,SAAS;AAE1D;AAIA,SAAS,0BACP,MACA,SACA,OACuC;CACvC,MAAM,KAAe,CAAC;CACtB,MAAM,KAAe,CAAC;CACtB,KAAK,MAAM,OAAO,MAAM;EACtB,MAAM,IAAI,QAAQ,aAAa,IAAI;EACnC,IAAI,MAAM,KAAA,KAAa,CAAC,OAAO,SAAS,CAAC,GAAG;EAC5C,MAAM,IAAI,YAAY,KAAK,KAAK;EAChC,IAAI,CAAC,OAAO,SAAS,CAAC,GAAG;EACzB,GAAG,KAAK,CAAC;EACT,GAAG,KAAK,CAAC;CACX;CACA,IAAI,GAAG,SAAS,GAAG,OAAO,KAAA;CAE1B,MAAM,IAAI,SAAS,IAAI,EAAE;CACzB,MAAM,IAAI,UAAU,IAAI,EAAE;CAC1B,MAAM,QAAQ,KAAK,EAAE;CACrB,MAAM,QAAQ,KAAK,EAAE;CACrB,IAAI,MAAM;CACV,IAAI,QAAQ;CACZ,KAAK,IAAI,IAAI,GAAG,IAAI,GAAG,QAAQ,KAAK;EAClC,QAAQ,GAAG,KAAM,UAAU,GAAG,KAAM;EACpC,UAAU,GAAG,KAAM,UAAU;CAC/B;CACA,MAAM,QAAQ,UAAU,IAAI,IAAI,MAAM;CACtC,MAAM,YAAY,QAAQ,QAAQ;CAClC,MAAM,QAAQ,GAAG,QAAQ,GAAG,MAAM,KAAK,IAAI,UAAU,GAAG,CAAC;CACzD,MAAM,QAAQ,GAAG,QAAQ,GAAG,GAAG,MAAM,KAAK,KAAK,YAAY,QAAQ,GAAG,QAAS,GAAG,CAAC;CACnF,MAAM,KAAK,UAAU,IAAI,IAAI,IAAI,QAAQ;CAEzC,OAAO;EACL,QAAQ,QAAQ;EAChB,GAAG,GAAG;EACN,SAAS;EACT,UAAU;EACV,aAAa;GAAE;GAAW;GAAO;EAAG;CACtC;AACF;AAIA,SAAS,sBACP,WACA,MACA,eAC0B;CAO1B,MAAM,OAAyC,CAAC;CAChD,MAAM,WACJ,SAAS,KAAA,IACJ,kBACD,CAAC,KAAK,mBACH,kBACD,KAAK,KAAK,KAAK,KAAK,CAAC,UAAU,KAAK,IAAI,IACrC,SACD,KAAK,QAAQ,IACV,SACA;CACb,KAAK,KAAK;EACR,MAAM;EACN,QAAQ;EACR,QAAQ,OACJ,SAAS,KAAK,MAAM,QAAQ,CAAC,EAAE,UAAU,KAAK,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,IAAI,KAAK,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,OAAO,KAAK,IAAI,KAAK,mBAAmB,KAAK,uBAAuB,KAAK,gBAAgB,gBACtL;CACN,CAAC;CACD,MAAM,aACJ,kBAAkB,KAAA,IACb,kBACD,cAAc,UAAU,IACrB,SACA;CACT,KAAK,KAAK;EACR,MAAM;EACN,QAAQ;EACR,QAAQ,gBAAgB,GAAG,cAAc,MAAM,mBAAmB;CACpE,CAAC;CACD,KAAK,KACH,UAAU,MAAM,IACZ;EACE,MAAM;EACN,QAAQ;EACR,QAAQ;CACV,IACA;EACE,MAAM;EACN,QACE,UAAU,SAAS,QAAQ,UAAU,QAAQ,KACzC,SACA,UAAU,SAAS,QAAQ,UAAU,QAAQ,KAC3C,SACA;EACR,QACE,UAAU,SAAS,QAAQ,UAAU,QAAQ,QAAQ,UAAU,QAAQ,OACnE,uDACA,QAAQ,UAAU,KAAK,QAAQ,CAAC,EAAE,QAAQ,UAAU,IAAI,QAAQ,CAAC,EAAE,QAAQ,UAAU,IAAI,QAAQ,CAAC,EAAE,UAAU,UAAU;CAChI,CACN;CAMA,OAAO;EACL,QANa,KAAK,MAAM,MAAM,EAAE,WAAW,MAAM,IAC/C,SACA,KAAK,MAAM,MAAM,EAAE,WAAW,UAAU,EAAE,WAAW,eAAe,IAClE,SACA;EAGJ;EACA,QAAQ,CAAC;CACX;AACF;AAiBA,SAAS,qBAAqB,KAA8C;CAC1E,MAAM,MAAwB,CAAC;CAI/B,IAAI,IAAI,uBAAuB;EAC7B,MAAM,MAAM,IAAI;EAChB,MAAM,QAAQ,IAAI,eAAe;EACjC,KAAK,MAAM,QAAQ,IAAI,kBAAkB;GACvC,MAAM,IAAI,IAAI,QAAQ;GACtB,IAAI,GAAG,WAAW,MAAM;GACxB,IAAI,KAAK;IACP,UAAU;IACV,MAAM;IACN,OAAO,GAAG,KAAK,kBAAkB,EAAE,SAAS,QAAQ,CAAC,EAAE,KAAK,EAAE,QAAQ,QAAQ,CAAC,EAAE,MAAM;IACvF,QAAQ,iBAAiB,EAAE,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,IAAI,EAAE,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,OAAO,EAAE,OAAO,QAAQ,CAAC,EAAE,cAAc,EAAE,QAAQ,QAAQ,CAAC,EAAE,cAAc,EAAE,SAAS,eAAe,EAAE,UAAU;IACzL,cAAc,iCAAiC;GACjD,CAAC;EACH;EACA,KAAK,MAAM,QAAQ,IAAI,iBAAiB;GACtC,MAAM,IAAI,IAAI,QAAQ;GACtB,IAAI,GAAG,WAAW,MAAM;GACxB,IAAI,KAAK;IACP,UAAU;IACV,MAAM;IACN,OAAO,GAAG,KAAK,iBAAiB,EAAE,SAAS,QAAQ,CAAC,EAAE,KAAK,EAAE,QAAQ,QAAQ,CAAC,EAAE,MAAM;IACtF,QAAQ,iBAAiB,EAAE,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,IAAI,EAAE,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,OAAO,EAAE,OAAO,QAAQ,CAAC,EAAE,cAAc,EAAE,QAAQ,QAAQ,CAAC,EAAE,cAAc,EAAE,SAAS,eAAe,EAAE,UAAU;IACzL,cAAc,iCAAiC;GACjD,CAAC;EACH;EACA,KAAK,MAAM,QAAQ,IAAI,qBAAqB;GAC1C,MAAM,IAAI,IAAI,QAAQ;GACtB,IAAI,CAAC,KAAK,EAAE,WAAW,QAAQ,EAAE,UAAU,GAAG;GAC9C,MAAM,SACJ,EAAE,WAAW,kBACT,6CACA;GACN,IAAI,KAAK;IACP,UAAU;IACV,MAAM;IACN,OAAO,GAAG,KAAK,gBAAgB,EAAE,SAAS,QAAQ,CAAC,EAAE,KAAK,EAAE,QAAQ,QAAQ,CAAC,EAAE,MAAM,MAAM;IAC3F,QAAQ,kBAAkB,EAAE,MAAM,QAAQ,CAAC,EAAE,oBAAoB,EAAE,SAAS,kBAAkB,EAAE,UAAU,QAAQ,OAAO;IACzH,cAAc,iCAAiC;GACjD,CAAC;EACH;CACF;CAKA,IACE,IAAI,UAAU,IAAI,KAClB,IAAI,UAAU,SAAS,QACvB,IAAI,UAAU,QAAQ,QACtB,IAAI,UAAU,QAAQ,MAElB;MAAA,IAAI,UAAU,OAAO,IAAK;GAC5B,MAAM,OAAO,IAAI,UAAU,YAAY,CAAC;GACxC,MAAM,QAAQ,KACX,MAAM,GAAG,CAAC,CAAC,CACX,KAAK,MAAM,GAAG,EAAE,MAAM,GAAG,EAAE,MAAM,QAAQ,CAAC,GAAG,CAAC,CAC9C,KAAK,IAAI;GACZ,IAAI,KAAK;IACP,UAAU;IACV,MAAM;IACN,OAAO,kBAAkB,IAAI,UAAU,KAAK,QAAQ,CAAC,EAAE;IACvD,QACE,KAAK,SAAS,IACV,SAAS,KAAK,OAAO,MAAM,KAAK,WAAW,IAAI,KAAK,IAAI,qBAAqB,MAAM,kBAAkB,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE,QAAQ,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE,KACvK,iBAAiB,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE,QAAQ,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE;IACzF,cAAc;GAChB,CAAC;EACH,OAAO,IAAI,IAAI,UAAU,OAAO,IAAK;GACnC,MAAM,OAAO,IAAI,UAAU,YAAY,CAAC;GACxC,MAAM,QAAQ,KACX,MAAM,GAAG,CAAC,CAAC,CACX,KAAK,MAAM,GAAG,EAAE,MAAM,GAAG,EAAE,MAAM,QAAQ,CAAC,GAAG,CAAC,CAC9C,KAAK,IAAI;GACZ,IAAI,KAAK;IACP,UAAU;IACV,MAAM;IACN,OAAO,kBAAkB,IAAI,UAAU,KAAK,QAAQ,CAAC,EAAE;IACvD,QACE,KAAK,SAAS,IACV,SAAS,KAAK,OAAO,MAAM,KAAK,WAAW,IAAI,KAAK,IAAI,IAAI,MAAM,kBAAkB,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE,QAAQ,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE,KACtJ,iBAAiB,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE,QAAQ,IAAI,UAAU,IAAI,QAAQ,CAAC,EAAE;IACzF,cAAc;GAChB,CAAC;EACH;;CAKF,IAAI,IAAI,kBAAkB,IAAI,eAAe,SAAS,GAAG;EACvD,MAAM,MAAM,IAAI,eAAe;EAC/B,IAAI,IAAI,SAAS,KAAK,IAAI,SAAS,KACjC,IAAI,KAAK;GACP,UAAU,IAAI,SAAS,MAAO,SAAS;GACvC,MAAM;GACN,OAAO,IAAI,IAAI,aAAa,oCAAoC,IAAI,MAAM,UAAU,IAAI,QAAQ,IAAA,CAAK,QAAQ,CAAC,EAAE;GAChH,QAAQ,4FAA4F,IAAI,MAAM,MAAM,IAAI,UAAU,EAAE,qBAAqB,IAAI,aAAa,GAAG,IAAI,eAAe,SAAS,IAAI,YAAY,IAAI,eAAe,EAAE,CAAE,aAAa,KAAK,IAAI,eAAe,EAAE,CAAE,MAAM,KAAK,GAAG;GACvS,cAAc;EAChB,CAAC;CAEL;CAKA,IAAI,OAAO,KAAK,IAAI,MAAM,CAAC,CAAC,WAAW,KAAK,IAAI,UAAU,IAAI,GAC5D,IAAI,KAAK;EACP,UAAU;EACV,MAAM;EACN,OAAO;EACP,QACE;EACF,cAAc;CAChB,CAAC;CAGH,IAAI,IAAI,MACN,IAAI,CAAC,IAAI,KAAK,kBACZ,IAAI,KAAK;EACP,UAAU;EACV,MAAM;EACN,OAAO,kBAAkB,IAAI,KAAK,EAAE,gBAAgB,IAAI,KAAK,gBAAgB;EAC7E,QAAQ,+CAA+C,IAAI,KAAK,gBAAgB;EAChF,cAAc;CAChB,CAAC;MACI;EACL,MAAM,eACJ,IAAI,KAAK,YAAY,OAAO,oCAAoC,IAAI,KAAK,QAAQ,QAAQ,CAAC;EAC5F,MAAM,UACJ,IAAI,KAAK,WAAW,OAAO,oCAAoC,IAAI,KAAK,OAAO,QAAQ,CAAC;EAC1F,MAAM,eACJ,IAAI,KAAK,cAAc,OAAO,kBAAkB,IAAI,IAAI,KAAK,UAAU;EAMzE,MAAM,WAAW,CAAC,UAAU,IAAI,KAAK,IAAI,KAAK,IAAI,KAAK,KAAK,KAAK,IAAI;EACrE,MAAM,eAAe,IAAI,KAAK,KAAK,MAAM,IAAI,aAAa,IAAI,KAAK,KAAK,KAAK,IAAI;EACjF,IAAI,UACF,IAAI,KAAK;GACP,UAAU;GACV,MAAM;GACN,OAAO,eAAe,IAAI,KAAK,MAAM,QAAQ,CAAC,EAAE,WAAW,IAAI,KAAK,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,IAAI,IAAI,KAAK,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE;GACvH,QAAQ,kCAAkC,IAAI,UAAU,oCAAoC,IAAI,KAAK,EAAE,MAAM,QAAQ,aAAa,aAAa;GAC/I,cAAc;EAChB,CAAC;OACI,IAAI,cACT,IAAI,KAAK;GACP,UAAU;GACV,MAAM;GACN,OAAO,qCAAqC,aAAa,SAAS,IAAI,KAAK,EAAE;GAC7E,QAAQ,uDAAuD,IAAI,KAAK,IAAI,QAAQ,CAAC,EAAE,sBAAsB,IAAI,KAAK,MAAM,QAAQ,CAAC,EAAE;GACvI,cAAc;EAChB,CAAC;OAED,IAAI,KAAK;GACP,UAAU;GACV,MAAM;GACN,OAAO,8BAA8B,IAAI,KAAK,KAAK,EAAE,CAAC,QAAQ,CAAC,EAAE,4BAA4B,IAAI;GACjG,QAAQ;GACR,cAAc;EAChB,CAAC;CAEL;CAGF,IAAI,IAAI,iBAAiB,IAAI,cAAc,QAAQ,GACjD,IAAI,KAAK;EACP,UAAU;EACV,MAAM;EACN,OAAO,GAAG,IAAI,cAAc,MAAM,cAAc,IAAI,cAAc,UAAU,IAAI,KAAK,IAAI;EACzF,QAAQ;EACR,cAAc;CAChB,CAAC;CAGH,IAAI,IAAI,cAAc,IAAI,WAAW,QAAQ,IAC3C,IAAI,KAAK;EACP,UAAU;EACV,MAAM;EACN,OAAO,8BAA8B,IAAI,WAAW,MAAM,QAAQ,CAAC,EAAE;EACrE,QACE;EACF,cAAc;CAChB,CAAC;CAGH,IAAI,IAAI,mBAAmB,IAAI,gBAAgB,SAAS,SAAS,GAAG;EAClE,MAAM,MAAM,IAAI,gBAAgB,SAAS;EACzC,IAAI,KAAK;GACP,UAAU;GACV,MAAM;GACN,OAAO,wBAAwB,IAAI,KAAK,KAAK,IAAI,QAAQ,IAAA,CAAK,QAAQ,CAAC,EAAE;GACzE,QAAQ,GAAG,IAAI,gBAAgB,cAAc,2CAA2C,IAAI,UAAU,OAAO,oBAAoB,IAAI,KAAK;GAC1I,cAAc;EAChB,CAAC;CACH;CAEA,IAAI,IAAI,sBAAsB,KAAK,IAAI,IAAI,mBAAmB,QAAQ,IAAI,IACxE,IAAI,KAAK;EACP,UAAU;EACV,MAAM;EACN,OAAO,+BAA+B,IAAI,mBAAmB,OAAO,eAAe,IAAI,mBAAmB,SAAS,QAAQ,CAAC,EAAE;EAC9H,QAAQ,yFAAyF,IAAI,mBAAmB,OAAO,0CAA0C,IAAI,mBAAmB,OAAO;EACvM,cAAc;CAChB,CAAC;CAGH,OAAO;AACT"}
@@ -1,4 +1,4 @@
1
- import { B as studentTCdf, V as studentTQuantile, l as cohensD } from "./statistics-RwRNu2__.js";
1
+ import { K as studentTCdf, q as studentTQuantile, u as cohensD } from "./statistics-CnGCLLqc.js";
2
2
  //#region src/baseline.ts
3
3
  /**
4
4
  * Baseline regression detection.
@@ -146,4 +146,4 @@ function assertFiniteSample(name, sample) {
146
146
  //#endregion
147
147
  export { iqr as n, welchsTTest as r, compareToBaseline as t };
148
148
 
149
- //# sourceMappingURL=baseline-BaPxoROc.js.map
149
+ //# sourceMappingURL=baseline-BUeFcgrn.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"baseline-BaPxoROc.js","names":[],"sources":["../src/baseline.ts"],"sourcesContent":["/**\n * Baseline regression detection.\n *\n * Lifted from ADC baseline.ts. Every promotion-blocking signal boils down\n * to: \"is this run measurably worse than baseline?\" — with enough\n * statistical rigor to distinguish noise from drift.\n *\n * Uses:\n * - Welch's t-test (unequal variance) for per-metric mean comparison\n * - Cohen's d for effect size magnitude\n * - IQR for stability flag (unstable samples can't be trusted for comparisons)\n *\n * Returns a structured verdict: improved | regressed | stable | unstable.\n */\n\nimport { studentTCdf, studentTQuantile } from './math/student-t'\nimport { cohensD } from './statistics'\n\nexport interface MetricSamples {\n /** Stable metric key (e.g. \"overallScore\", \"firstTokenMs\"). */\n metric: string\n /** Whether higher values are better. */\n higherIsBetter: boolean\n baseline: number[]\n candidate: number[]\n}\n\nexport interface MetricVerdict {\n metric: string\n baselineMean: number\n candidateMean: number\n delta: number\n /** Null when both samples are constant with unequal means — the\n * standardized effect is unbounded there, not zero. */\n cohensD: number | null\n welchT: number\n welchDf: number\n welchP: number\n stable: boolean\n /** IQR of the combined samples — used as a rough stability indicator. */\n iqr: number\n verdict: 'improved' | 'regressed' | 'stable' | 'unstable'\n}\n\nexport interface BaselineReport {\n metrics: MetricVerdict[]\n /** True if any critical metric regressed. */\n hasRegression: boolean\n /** True if any metric is unstable (too noisy to judge). */\n hasUnstable: boolean\n}\n\nexport interface BaselineOptions {\n /** Effect size threshold for meaningful delta (default 0.5 — medium effect). */\n effectThreshold?: number\n /** p-value threshold for statistical significance (default 0.05). */\n alpha?: number\n /** IQR/mean ratio above which samples are flagged unstable (default 0.30). */\n unstableCvThreshold?: number\n}\n\nexport type WelchTestStatus = 'ok' | 'insufficient-sample' | 'zero-variance'\n\n/**\n * Full unequal-variance comparison for two independent samples.\n *\n * `meanB`, `delta`, `t`, and `cohensD` are oriented as `b - a`. Inferential\n * fields are `NaN` and `ci95` is null when the sample count or observed\n * variance cannot define a Student-t result. `status` makes that state\n * explicit so callers cannot mistake it for evidence of no difference.\n */\ninterface WelchTestResultBase {\n meanA: number\n meanB: number\n delta: number\n}\n\nexport type WelchTestResult =\n | (WelchTestResultBase & {\n status: 'ok'\n standardError: number\n t: number\n df: number\n p: number\n ci95: [number, number]\n cohensD: number\n })\n | (WelchTestResultBase & {\n status: 'insufficient-sample' | 'zero-variance'\n standardError: number\n t: number\n df: number\n p: number\n ci95: null\n cohensD: null\n })\n\n/**\n * Compare candidate samples against baseline per metric. Verdict logic:\n * - unstable: IQR/|mean| > threshold on either set — not enough signal\n * - improved: meaningful effect in the \"better\" direction AND p < alpha\n * - regressed: meaningful effect in the \"worse\" direction AND p < alpha\n * - stable: otherwise (no significant change)\n */\nexport function compareToBaseline(\n samples: MetricSamples[],\n options: BaselineOptions = {},\n): BaselineReport {\n const effectThreshold = options.effectThreshold ?? 0.5\n const alpha = options.alpha ?? 0.05\n const cvThreshold = options.unstableCvThreshold ?? 0.3\n\n const metrics: MetricVerdict[] = samples.map((s) => {\n const comparison = welchsTTest(s.baseline, s.candidate)\n if (comparison.status === 'insufficient-sample') {\n throw new Error(`compareToBaseline: need ≥2 samples per side for \"${s.metric}\"`)\n }\n const { meanA: bMean, meanB: cMean, delta, cohensD: d, t, df, p } = comparison\n // Stability is per-side: a comparison is trustworthy only when BOTH\n // samples are internally consistent. Combining the sides would flag\n // large-but-real deltas as \"unstable\" which is exactly what we want\n // to detect.\n const baselineIqr = iqr(s.baseline)\n const candidateIqr = iqr(s.candidate)\n const baselineStable = baselineIqr / Math.max(Math.abs(bMean), 1e-9) <= cvThreshold\n const candidateStable = candidateIqr / Math.max(Math.abs(cMean), 1e-9) <= cvThreshold\n const stable = baselineStable && candidateStable\n const reportedIqr = Math.max(baselineIqr, candidateIqr)\n\n // Both sides are guaranteed ≥ 2 samples above, so a null d means zero\n // pooled spread across a real mean gap: an unbounded standardized effect,\n // which clears any finite threshold.\n const effectClears = d === null || Math.abs(d) >= effectThreshold\n\n let verdict: MetricVerdict['verdict']\n if (!stable) {\n verdict = 'unstable'\n } else if (comparison.status === 'ok' && p < alpha && effectClears) {\n const candidateIsBetter = s.higherIsBetter ? delta > 0 : delta < 0\n verdict = candidateIsBetter ? 'improved' : 'regressed'\n } else {\n verdict = 'stable'\n }\n\n return {\n metric: s.metric,\n baselineMean: bMean,\n candidateMean: cMean,\n delta,\n cohensD: d,\n welchT: t,\n welchDf: df,\n welchP: p,\n stable,\n iqr: reportedIqr,\n verdict,\n }\n })\n\n return {\n metrics,\n hasRegression: metrics.some((m) => m.verdict === 'regressed'),\n hasUnstable: metrics.some((m) => m.verdict === 'unstable'),\n }\n}\n\nfunction mean(xs: number[]): number {\n return xs.reduce((a, b) => a + b, 0) / xs.length\n}\n\n/** Inter-quartile range; 0 when the sample has no spread. */\nexport function iqr(xs: number[]): number {\n if (xs.length === 0) return 0\n const sorted = [...xs].sort((a, b) => a - b)\n const q = (p: number) => {\n const idx = p * (sorted.length - 1)\n const lo = Math.floor(idx)\n const hi = Math.ceil(idx)\n return sorted[lo]! + (sorted[hi]! - sorted[lo]!) * (idx - lo)\n }\n return q(0.75) - q(0.25)\n}\n\n/**\n * Welch's t-test and 95% interval for two independent samples.\n *\n * Uses the Student-t distribution at every finite degree of freedom. Fewer\n * than two observations per side and zero observed standard error cannot\n * define a t reference distribution; those cases return an explicit status,\n * `NaN` inferential fields, and no interval rather than fabricated certainty.\n */\nexport function welchsTTest(a: number[], b: number[]): WelchTestResult {\n assertFiniteSample('a', a)\n assertFiniteSample('b', b)\n const mA = mean(a)\n const mB = mean(b)\n const delta = mB - mA\n if (a.length < 2 || b.length < 2) {\n return {\n status: 'insufficient-sample',\n meanA: mA,\n meanB: mB,\n delta,\n standardError: Number.NaN,\n t: Number.NaN,\n df: Number.NaN,\n p: Number.NaN,\n ci95: null,\n cohensD: null,\n }\n }\n\n const vA = variance(a, mA)\n const vB = variance(b, mB)\n const seSquared = vA / a.length + vB / b.length\n const d = cohensD(a, b)\n if (seSquared === 0) {\n return {\n status: 'zero-variance',\n meanA: mA,\n meanB: mB,\n delta,\n standardError: 0,\n t: delta === 0 ? 0 : Math.sign(delta) * Number.POSITIVE_INFINITY,\n df: Number.NaN,\n p: Number.NaN,\n ci95: null,\n cohensD: null,\n }\n }\n\n const standardError = Math.sqrt(seSquared)\n const t = delta / standardError\n const df =\n (seSquared * seSquared) /\n ((vA / a.length) ** 2 / (a.length - 1) + (vB / b.length) ** 2 / (b.length - 1))\n const p = 2 * (1 - studentTCdf(Math.abs(t), df))\n if (d === null) {\n throw new Error('welchsTTest: non-zero standard error produced no pooled effect size')\n }\n const halfWidth = studentTQuantile(0.975, df) * standardError\n return {\n status: 'ok',\n meanA: mA,\n meanB: mB,\n delta,\n standardError,\n t,\n df,\n p,\n ci95: [delta - halfWidth, delta + halfWidth],\n cohensD: d,\n }\n}\n\nfunction variance(xs: number[], m: number): number {\n return xs.reduce((acc, x) => acc + (x - m) ** 2, 0) / (xs.length - 1)\n}\n\nfunction assertFiniteSample(name: string, sample: readonly number[]): void {\n const invalid = sample.find((value) => !Number.isFinite(value))\n if (invalid !== undefined) {\n throw new RangeError(\n `welchsTTest: sample ${name} must contain only finite values, got ${invalid}`,\n )\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;AAwGA,SAAgB,kBACd,SACA,UAA2B,CAAC,GACZ;CAChB,MAAM,kBAAkB,QAAQ,mBAAmB;CACnD,MAAM,QAAQ,QAAQ,SAAS;CAC/B,MAAM,cAAc,QAAQ,uBAAuB;CAEnD,MAAM,UAA2B,QAAQ,KAAK,MAAM;EAClD,MAAM,aAAa,YAAY,EAAE,UAAU,EAAE,SAAS;EACtD,IAAI,WAAW,WAAW,uBACxB,MAAM,IAAI,MAAM,oDAAoD,EAAE,OAAO,EAAE;EAEjF,MAAM,EAAE,OAAO,OAAO,OAAO,OAAO,OAAO,SAAS,GAAG,GAAG,IAAI,MAAM;EAKpE,MAAM,cAAc,IAAI,EAAE,QAAQ;EAClC,MAAM,eAAe,IAAI,EAAE,SAAS;EACpC,MAAM,iBAAiB,cAAc,KAAK,IAAI,KAAK,IAAI,KAAK,GAAG,IAAI,KAAK;EACxE,MAAM,kBAAkB,eAAe,KAAK,IAAI,KAAK,IAAI,KAAK,GAAG,IAAI,KAAK;EAC1E,MAAM,SAAS,kBAAkB;EACjC,MAAM,cAAc,KAAK,IAAI,aAAa,YAAY;EAKtD,MAAM,eAAe,MAAM,QAAQ,KAAK,IAAI,CAAC,KAAK;EAElD,IAAI;EACJ,IAAI,CAAC,QACH,UAAU;OACL,IAAI,WAAW,WAAW,QAAQ,IAAI,SAAS,cAEpD,WAD0B,EAAE,iBAAiB,QAAQ,IAAI,QAAQ,KACnC,aAAa;OAE3C,UAAU;EAGZ,OAAO;GACL,QAAQ,EAAE;GACV,cAAc;GACd,eAAe;GACf;GACA,SAAS;GACT,QAAQ;GACR,SAAS;GACT,QAAQ;GACR;GACA,KAAK;GACL;EACF;CACF,CAAC;CAED,OAAO;EACL;EACA,eAAe,QAAQ,MAAM,MAAM,EAAE,YAAY,WAAW;EAC5D,aAAa,QAAQ,MAAM,MAAM,EAAE,YAAY,UAAU;CAC3D;AACF;AAEA,SAAS,KAAK,IAAsB;CAClC,OAAO,GAAG,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,GAAG;AAC5C;;AAGA,SAAgB,IAAI,IAAsB;CACxC,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,MAAM,SAAS,CAAC,GAAG,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3C,MAAM,KAAK,MAAc;EACvB,MAAM,MAAM,KAAK,OAAO,SAAS;EACjC,MAAM,KAAK,KAAK,MAAM,GAAG;EACzB,MAAM,KAAK,KAAK,KAAK,GAAG;EACxB,OAAO,OAAO,OAAQ,OAAO,MAAO,OAAO,QAAS,MAAM;CAC5D;CACA,OAAO,EAAE,GAAI,IAAI,EAAE,GAAI;AACzB;;;;;;;;;AAUA,SAAgB,YAAY,GAAa,GAA8B;CACrE,mBAAmB,KAAK,CAAC;CACzB,mBAAmB,KAAK,CAAC;CACzB,MAAM,KAAK,KAAK,CAAC;CACjB,MAAM,KAAK,KAAK,CAAC;CACjB,MAAM,QAAQ,KAAK;CACnB,IAAI,EAAE,SAAS,KAAK,EAAE,SAAS,GAC7B,OAAO;EACL,QAAQ;EACR,OAAO;EACP,OAAO;EACP;EACA,eAAe;EACf,GAAG;EACH,IAAI;EACJ,GAAG;EACH,MAAM;EACN,SAAS;CACX;CAGF,MAAM,KAAK,SAAS,GAAG,EAAE;CACzB,MAAM,KAAK,SAAS,GAAG,EAAE;CACzB,MAAM,YAAY,KAAK,EAAE,SAAS,KAAK,EAAE;CACzC,MAAM,IAAI,QAAQ,GAAG,CAAC;CACtB,IAAI,cAAc,GAChB,OAAO;EACL,QAAQ;EACR,OAAO;EACP,OAAO;EACP;EACA,eAAe;EACf,GAAG,UAAU,IAAI,IAAI,KAAK,KAAK,KAAK,IAAI,OAAO;EAC/C,IAAI;EACJ,GAAG;EACH,MAAM;EACN,SAAS;CACX;CAGF,MAAM,gBAAgB,KAAK,KAAK,SAAS;CACzC,MAAM,IAAI,QAAQ;CAClB,MAAM,KACH,YAAY,cACX,KAAK,EAAE,WAAW,KAAK,EAAE,SAAS,MAAM,KAAK,EAAE,WAAW,KAAK,EAAE,SAAS;CAC9E,MAAM,IAAI,KAAK,IAAI,YAAY,KAAK,IAAI,CAAC,GAAG,EAAE;CAC9C,IAAI,MAAM,MACR,MAAM,IAAI,MAAM,qEAAqE;CAEvF,MAAM,YAAY,iBAAiB,MAAO,EAAE,IAAI;CAChD,OAAO;EACL,QAAQ;EACR,OAAO;EACP,OAAO;EACP;EACA;EACA;EACA;EACA;EACA,MAAM,CAAC,QAAQ,WAAW,QAAQ,SAAS;EAC3C,SAAS;CACX;AACF;AAEA,SAAS,SAAS,IAAc,GAAmB;CACjD,OAAO,GAAG,QAAQ,KAAK,MAAM,OAAO,IAAI,MAAM,GAAG,CAAC,KAAK,GAAG,SAAS;AACrE;AAEA,SAAS,mBAAmB,MAAc,QAAiC;CACzE,MAAM,UAAU,OAAO,MAAM,UAAU,CAAC,OAAO,SAAS,KAAK,CAAC;CAC9D,IAAI,YAAY,KAAA,GACd,MAAM,IAAI,WACR,uBAAuB,KAAK,wCAAwC,SACtE;AAEJ"}
1
+ {"version":3,"file":"baseline-BUeFcgrn.js","names":[],"sources":["../src/baseline.ts"],"sourcesContent":["/**\n * Baseline regression detection.\n *\n * Lifted from ADC baseline.ts. Every promotion-blocking signal boils down\n * to: \"is this run measurably worse than baseline?\" — with enough\n * statistical rigor to distinguish noise from drift.\n *\n * Uses:\n * - Welch's t-test (unequal variance) for per-metric mean comparison\n * - Cohen's d for effect size magnitude\n * - IQR for stability flag (unstable samples can't be trusted for comparisons)\n *\n * Returns a structured verdict: improved | regressed | stable | unstable.\n */\n\nimport { studentTCdf, studentTQuantile } from './math/student-t'\nimport { cohensD } from './statistics'\n\nexport interface MetricSamples {\n /** Stable metric key (e.g. \"overallScore\", \"firstTokenMs\"). */\n metric: string\n /** Whether higher values are better. */\n higherIsBetter: boolean\n baseline: number[]\n candidate: number[]\n}\n\nexport interface MetricVerdict {\n metric: string\n baselineMean: number\n candidateMean: number\n delta: number\n /** Null when both samples are constant with unequal means — the\n * standardized effect is unbounded there, not zero. */\n cohensD: number | null\n welchT: number\n welchDf: number\n welchP: number\n stable: boolean\n /** IQR of the combined samples — used as a rough stability indicator. */\n iqr: number\n verdict: 'improved' | 'regressed' | 'stable' | 'unstable'\n}\n\nexport interface BaselineReport {\n metrics: MetricVerdict[]\n /** True if any critical metric regressed. */\n hasRegression: boolean\n /** True if any metric is unstable (too noisy to judge). */\n hasUnstable: boolean\n}\n\nexport interface BaselineOptions {\n /** Effect size threshold for meaningful delta (default 0.5 — medium effect). */\n effectThreshold?: number\n /** p-value threshold for statistical significance (default 0.05). */\n alpha?: number\n /** IQR/mean ratio above which samples are flagged unstable (default 0.30). */\n unstableCvThreshold?: number\n}\n\nexport type WelchTestStatus = 'ok' | 'insufficient-sample' | 'zero-variance'\n\n/**\n * Full unequal-variance comparison for two independent samples.\n *\n * `meanB`, `delta`, `t`, and `cohensD` are oriented as `b - a`. Inferential\n * fields are `NaN` and `ci95` is null when the sample count or observed\n * variance cannot define a Student-t result. `status` makes that state\n * explicit so callers cannot mistake it for evidence of no difference.\n */\ninterface WelchTestResultBase {\n meanA: number\n meanB: number\n delta: number\n}\n\nexport type WelchTestResult =\n | (WelchTestResultBase & {\n status: 'ok'\n standardError: number\n t: number\n df: number\n p: number\n ci95: [number, number]\n cohensD: number\n })\n | (WelchTestResultBase & {\n status: 'insufficient-sample' | 'zero-variance'\n standardError: number\n t: number\n df: number\n p: number\n ci95: null\n cohensD: null\n })\n\n/**\n * Compare candidate samples against baseline per metric. Verdict logic:\n * - unstable: IQR/|mean| > threshold on either set — not enough signal\n * - improved: meaningful effect in the \"better\" direction AND p < alpha\n * - regressed: meaningful effect in the \"worse\" direction AND p < alpha\n * - stable: otherwise (no significant change)\n */\nexport function compareToBaseline(\n samples: MetricSamples[],\n options: BaselineOptions = {},\n): BaselineReport {\n const effectThreshold = options.effectThreshold ?? 0.5\n const alpha = options.alpha ?? 0.05\n const cvThreshold = options.unstableCvThreshold ?? 0.3\n\n const metrics: MetricVerdict[] = samples.map((s) => {\n const comparison = welchsTTest(s.baseline, s.candidate)\n if (comparison.status === 'insufficient-sample') {\n throw new Error(`compareToBaseline: need ≥2 samples per side for \"${s.metric}\"`)\n }\n const { meanA: bMean, meanB: cMean, delta, cohensD: d, t, df, p } = comparison\n // Stability is per-side: a comparison is trustworthy only when BOTH\n // samples are internally consistent. Combining the sides would flag\n // large-but-real deltas as \"unstable\" which is exactly what we want\n // to detect.\n const baselineIqr = iqr(s.baseline)\n const candidateIqr = iqr(s.candidate)\n const baselineStable = baselineIqr / Math.max(Math.abs(bMean), 1e-9) <= cvThreshold\n const candidateStable = candidateIqr / Math.max(Math.abs(cMean), 1e-9) <= cvThreshold\n const stable = baselineStable && candidateStable\n const reportedIqr = Math.max(baselineIqr, candidateIqr)\n\n // Both sides are guaranteed ≥ 2 samples above, so a null d means zero\n // pooled spread across a real mean gap: an unbounded standardized effect,\n // which clears any finite threshold.\n const effectClears = d === null || Math.abs(d) >= effectThreshold\n\n let verdict: MetricVerdict['verdict']\n if (!stable) {\n verdict = 'unstable'\n } else if (comparison.status === 'ok' && p < alpha && effectClears) {\n const candidateIsBetter = s.higherIsBetter ? delta > 0 : delta < 0\n verdict = candidateIsBetter ? 'improved' : 'regressed'\n } else {\n verdict = 'stable'\n }\n\n return {\n metric: s.metric,\n baselineMean: bMean,\n candidateMean: cMean,\n delta,\n cohensD: d,\n welchT: t,\n welchDf: df,\n welchP: p,\n stable,\n iqr: reportedIqr,\n verdict,\n }\n })\n\n return {\n metrics,\n hasRegression: metrics.some((m) => m.verdict === 'regressed'),\n hasUnstable: metrics.some((m) => m.verdict === 'unstable'),\n }\n}\n\nfunction mean(xs: number[]): number {\n return xs.reduce((a, b) => a + b, 0) / xs.length\n}\n\n/** Inter-quartile range; 0 when the sample has no spread. */\nexport function iqr(xs: number[]): number {\n if (xs.length === 0) return 0\n const sorted = [...xs].sort((a, b) => a - b)\n const q = (p: number) => {\n const idx = p * (sorted.length - 1)\n const lo = Math.floor(idx)\n const hi = Math.ceil(idx)\n return sorted[lo]! + (sorted[hi]! - sorted[lo]!) * (idx - lo)\n }\n return q(0.75) - q(0.25)\n}\n\n/**\n * Welch's t-test and 95% interval for two independent samples.\n *\n * Uses the Student-t distribution at every finite degree of freedom. Fewer\n * than two observations per side and zero observed standard error cannot\n * define a t reference distribution; those cases return an explicit status,\n * `NaN` inferential fields, and no interval rather than fabricated certainty.\n */\nexport function welchsTTest(a: number[], b: number[]): WelchTestResult {\n assertFiniteSample('a', a)\n assertFiniteSample('b', b)\n const mA = mean(a)\n const mB = mean(b)\n const delta = mB - mA\n if (a.length < 2 || b.length < 2) {\n return {\n status: 'insufficient-sample',\n meanA: mA,\n meanB: mB,\n delta,\n standardError: Number.NaN,\n t: Number.NaN,\n df: Number.NaN,\n p: Number.NaN,\n ci95: null,\n cohensD: null,\n }\n }\n\n const vA = variance(a, mA)\n const vB = variance(b, mB)\n const seSquared = vA / a.length + vB / b.length\n const d = cohensD(a, b)\n if (seSquared === 0) {\n return {\n status: 'zero-variance',\n meanA: mA,\n meanB: mB,\n delta,\n standardError: 0,\n t: delta === 0 ? 0 : Math.sign(delta) * Number.POSITIVE_INFINITY,\n df: Number.NaN,\n p: Number.NaN,\n ci95: null,\n cohensD: null,\n }\n }\n\n const standardError = Math.sqrt(seSquared)\n const t = delta / standardError\n const df =\n (seSquared * seSquared) /\n ((vA / a.length) ** 2 / (a.length - 1) + (vB / b.length) ** 2 / (b.length - 1))\n const p = 2 * (1 - studentTCdf(Math.abs(t), df))\n if (d === null) {\n throw new Error('welchsTTest: non-zero standard error produced no pooled effect size')\n }\n const halfWidth = studentTQuantile(0.975, df) * standardError\n return {\n status: 'ok',\n meanA: mA,\n meanB: mB,\n delta,\n standardError,\n t,\n df,\n p,\n ci95: [delta - halfWidth, delta + halfWidth],\n cohensD: d,\n }\n}\n\nfunction variance(xs: number[], m: number): number {\n return xs.reduce((acc, x) => acc + (x - m) ** 2, 0) / (xs.length - 1)\n}\n\nfunction assertFiniteSample(name: string, sample: readonly number[]): void {\n const invalid = sample.find((value) => !Number.isFinite(value))\n if (invalid !== undefined) {\n throw new RangeError(\n `welchsTTest: sample ${name} must contain only finite values, got ${invalid}`,\n )\n }\n}\n"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;AAwGA,SAAgB,kBACd,SACA,UAA2B,CAAC,GACZ;CAChB,MAAM,kBAAkB,QAAQ,mBAAmB;CACnD,MAAM,QAAQ,QAAQ,SAAS;CAC/B,MAAM,cAAc,QAAQ,uBAAuB;CAEnD,MAAM,UAA2B,QAAQ,KAAK,MAAM;EAClD,MAAM,aAAa,YAAY,EAAE,UAAU,EAAE,SAAS;EACtD,IAAI,WAAW,WAAW,uBACxB,MAAM,IAAI,MAAM,oDAAoD,EAAE,OAAO,EAAE;EAEjF,MAAM,EAAE,OAAO,OAAO,OAAO,OAAO,OAAO,SAAS,GAAG,GAAG,IAAI,MAAM;EAKpE,MAAM,cAAc,IAAI,EAAE,QAAQ;EAClC,MAAM,eAAe,IAAI,EAAE,SAAS;EACpC,MAAM,iBAAiB,cAAc,KAAK,IAAI,KAAK,IAAI,KAAK,GAAG,IAAI,KAAK;EACxE,MAAM,kBAAkB,eAAe,KAAK,IAAI,KAAK,IAAI,KAAK,GAAG,IAAI,KAAK;EAC1E,MAAM,SAAS,kBAAkB;EACjC,MAAM,cAAc,KAAK,IAAI,aAAa,YAAY;EAKtD,MAAM,eAAe,MAAM,QAAQ,KAAK,IAAI,CAAC,KAAK;EAElD,IAAI;EACJ,IAAI,CAAC,QACH,UAAU;OACL,IAAI,WAAW,WAAW,QAAQ,IAAI,SAAS,cAEpD,WAD0B,EAAE,iBAAiB,QAAQ,IAAI,QAAQ,KACnC,aAAa;OAE3C,UAAU;EAGZ,OAAO;GACL,QAAQ,EAAE;GACV,cAAc;GACd,eAAe;GACf;GACA,SAAS;GACT,QAAQ;GACR,SAAS;GACT,QAAQ;GACR;GACA,KAAK;GACL;EACF;CACF,CAAC;CAED,OAAO;EACL;EACA,eAAe,QAAQ,MAAM,MAAM,EAAE,YAAY,WAAW;EAC5D,aAAa,QAAQ,MAAM,MAAM,EAAE,YAAY,UAAU;CAC3D;AACF;AAEA,SAAS,KAAK,IAAsB;CAClC,OAAO,GAAG,QAAQ,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI,GAAG;AAC5C;;AAGA,SAAgB,IAAI,IAAsB;CACxC,IAAI,GAAG,WAAW,GAAG,OAAO;CAC5B,MAAM,SAAS,CAAC,GAAG,EAAE,CAAC,CAAC,MAAM,GAAG,MAAM,IAAI,CAAC;CAC3C,MAAM,KAAK,MAAc;EACvB,MAAM,MAAM,KAAK,OAAO,SAAS;EACjC,MAAM,KAAK,KAAK,MAAM,GAAG;EACzB,MAAM,KAAK,KAAK,KAAK,GAAG;EACxB,OAAO,OAAO,OAAQ,OAAO,MAAO,OAAO,QAAS,MAAM;CAC5D;CACA,OAAO,EAAE,GAAI,IAAI,EAAE,GAAI;AACzB;;;;;;;;;AAUA,SAAgB,YAAY,GAAa,GAA8B;CACrE,mBAAmB,KAAK,CAAC;CACzB,mBAAmB,KAAK,CAAC;CACzB,MAAM,KAAK,KAAK,CAAC;CACjB,MAAM,KAAK,KAAK,CAAC;CACjB,MAAM,QAAQ,KAAK;CACnB,IAAI,EAAE,SAAS,KAAK,EAAE,SAAS,GAC7B,OAAO;EACL,QAAQ;EACR,OAAO;EACP,OAAO;EACP;EACA,eAAe;EACf,GAAG;EACH,IAAI;EACJ,GAAG;EACH,MAAM;EACN,SAAS;CACX;CAGF,MAAM,KAAK,SAAS,GAAG,EAAE;CACzB,MAAM,KAAK,SAAS,GAAG,EAAE;CACzB,MAAM,YAAY,KAAK,EAAE,SAAS,KAAK,EAAE;CACzC,MAAM,IAAI,QAAQ,GAAG,CAAC;CACtB,IAAI,cAAc,GAChB,OAAO;EACL,QAAQ;EACR,OAAO;EACP,OAAO;EACP;EACA,eAAe;EACf,GAAG,UAAU,IAAI,IAAI,KAAK,KAAK,KAAK,IAAI,OAAO;EAC/C,IAAI;EACJ,GAAG;EACH,MAAM;EACN,SAAS;CACX;CAGF,MAAM,gBAAgB,KAAK,KAAK,SAAS;CACzC,MAAM,IAAI,QAAQ;CAClB,MAAM,KACH,YAAY,cACX,KAAK,EAAE,WAAW,KAAK,EAAE,SAAS,MAAM,KAAK,EAAE,WAAW,KAAK,EAAE,SAAS;CAC9E,MAAM,IAAI,KAAK,IAAI,YAAY,KAAK,IAAI,CAAC,GAAG,EAAE;CAC9C,IAAI,MAAM,MACR,MAAM,IAAI,MAAM,qEAAqE;CAEvF,MAAM,YAAY,iBAAiB,MAAO,EAAE,IAAI;CAChD,OAAO;EACL,QAAQ;EACR,OAAO;EACP,OAAO;EACP;EACA;EACA;EACA;EACA;EACA,MAAM,CAAC,QAAQ,WAAW,QAAQ,SAAS;EAC3C,SAAS;CACX;AACF;AAEA,SAAS,SAAS,IAAc,GAAmB;CACjD,OAAO,GAAG,QAAQ,KAAK,MAAM,OAAO,IAAI,MAAM,GAAG,CAAC,KAAK,GAAG,SAAS;AACrE;AAEA,SAAS,mBAAmB,MAAc,QAAiC;CACzE,MAAM,UAAU,OAAO,MAAM,UAAU,CAAC,OAAO,SAAS,KAAK,CAAC;CAC9D,IAAI,YAAY,KAAA,GACd,MAAM,IAAI,WACR,uBAAuB,KAAK,wCAAwC,SACtE;AAEJ"}
@@ -1,2 +1,2 @@
1
- import { A as BenchmarkMetricCalibrationOptions, B as BenchmarkSource, C as BenchmarkRunOptions, D as runBenchmarkAdapter, E as renderBenchmarkReportMarkdown, F as BenchmarkDatasetItem, H as deterministicSplit, I as BenchmarkEvaluation, L as BenchmarkFamily, M as calibrateBenchmarkMetric, N as BENCHMARK_SPLIT_SEED, O as summarizeBenchmarkCampaign, P as BenchmarkAdapter, R as BenchmarkResponder, S as BenchmarkReport, T as BenchmarkSliceSummary, V as BenchmarkTaskKind, _ as parseJsonlRows, a as StandardRetrievalDocument, b as retrievalMetricsAtCutoff, c as StandardRetrievalQrel, d as buildStandardRetrievalItems, f as createRetrievalIdBenchmarkAdapter, g as parseBeirQueriesJsonl, h as parseBeirCorpusJsonl, i as StandardRetrievalArtifact, j as BenchmarkMetricCalibrationResult, k as index_d_exports, l as StandardRetrievalQuery, m as normalizeRetrievedDocumentIds, n as BuildStandardRetrievalItemsOptions, o as StandardRetrievalEvaluationOptions, p as evaluateStandardRetrieval, r as RetrievalIdAdapterOptions, s as StandardRetrievalPayload, u as StandardRetrievalResult, v as parseQrels, w as BenchmarkRunResult, x as BenchmarkDistribution, y as parseTsvRows, z as BenchmarkScenario } from "../index-C21xKtxu.js";
1
+ import { A as BenchmarkMetricCalibrationOptions, B as BenchmarkSource, C as BenchmarkRunOptions, D as runBenchmarkAdapter, E as renderBenchmarkReportMarkdown, F as BenchmarkDatasetItem, H as deterministicSplit, I as BenchmarkEvaluation, L as BenchmarkFamily, M as calibrateBenchmarkMetric, N as BENCHMARK_SPLIT_SEED, O as summarizeBenchmarkCampaign, P as BenchmarkAdapter, R as BenchmarkResponder, S as BenchmarkReport, T as BenchmarkSliceSummary, V as BenchmarkTaskKind, _ as parseJsonlRows, a as StandardRetrievalDocument, b as retrievalMetricsAtCutoff, c as StandardRetrievalQrel, d as buildStandardRetrievalItems, f as createRetrievalIdBenchmarkAdapter, g as parseBeirQueriesJsonl, h as parseBeirCorpusJsonl, i as StandardRetrievalArtifact, j as BenchmarkMetricCalibrationResult, k as index_d_exports, l as StandardRetrievalQuery, m as normalizeRetrievedDocumentIds, n as BuildStandardRetrievalItemsOptions, o as StandardRetrievalEvaluationOptions, p as evaluateStandardRetrieval, r as RetrievalIdAdapterOptions, s as StandardRetrievalPayload, u as StandardRetrievalResult, v as parseQrels, w as BenchmarkRunResult, x as BenchmarkDistribution, y as parseTsvRows, z as BenchmarkScenario } from "../index-CQsJcqch.js";
2
2
  export { BENCHMARK_SPLIT_SEED, type BenchmarkAdapter, type BenchmarkDatasetItem, type BenchmarkDistribution, type BenchmarkEvaluation, type BenchmarkFamily, type BenchmarkMetricCalibrationOptions, type BenchmarkMetricCalibrationResult, type BenchmarkReport, type BenchmarkResponder, type BenchmarkRunOptions, type BenchmarkRunResult, type BenchmarkScenario, type BenchmarkSliceSummary, type BenchmarkSource, type BenchmarkTaskKind, type BuildStandardRetrievalItemsOptions, type RetrievalIdAdapterOptions, type StandardRetrievalArtifact, type StandardRetrievalDocument, type StandardRetrievalEvaluationOptions, type StandardRetrievalPayload, type StandardRetrievalQrel, type StandardRetrievalQuery, type StandardRetrievalResult, buildStandardRetrievalItems, calibrateBenchmarkMetric, createRetrievalIdBenchmarkAdapter, deterministicSplit, evaluateStandardRetrieval, normalizeRetrievedDocumentIds, parseBeirCorpusJsonl, parseBeirQueriesJsonl, parseJsonlRows, parseQrels, parseTsvRows, renderBenchmarkReportMarkdown, retrievalMetricsAtCutoff, index_d_exports as routing, runBenchmarkAdapter, summarizeBenchmarkCampaign };
@@ -1,2 +1,2 @@
1
- import { _ as deterministicSplit, a as normalizeRetrievedDocumentIds, c as parseJsonlRows, d as retrievalMetricsAtCutoff, f as renderBenchmarkReportMarkdown, g as BENCHMARK_SPLIT_SEED, h as routing_exports, i as evaluateStandardRetrieval, l as parseQrels, m as summarizeBenchmarkCampaign, n as buildStandardRetrievalItems, o as parseBeirCorpusJsonl, p as runBenchmarkAdapter, r as createRetrievalIdBenchmarkAdapter, s as parseBeirQueriesJsonl, u as parseTsvRows, v as calibrateBenchmarkMetric } from "../benchmarks-Dw1Wv_JQ.js";
1
+ import { _ as deterministicSplit, a as normalizeRetrievedDocumentIds, c as parseJsonlRows, d as retrievalMetricsAtCutoff, f as renderBenchmarkReportMarkdown, g as BENCHMARK_SPLIT_SEED, h as routing_exports, i as evaluateStandardRetrieval, l as parseQrels, m as summarizeBenchmarkCampaign, n as buildStandardRetrievalItems, o as parseBeirCorpusJsonl, p as runBenchmarkAdapter, r as createRetrievalIdBenchmarkAdapter, s as parseBeirQueriesJsonl, u as parseTsvRows, v as calibrateBenchmarkMetric } from "../benchmarks-Mtu251Jz.js";
2
2
  export { BENCHMARK_SPLIT_SEED, buildStandardRetrievalItems, calibrateBenchmarkMetric, createRetrievalIdBenchmarkAdapter, deterministicSplit, evaluateStandardRetrieval, normalizeRetrievedDocumentIds, parseBeirCorpusJsonl, parseBeirQueriesJsonl, parseJsonlRows, parseQrels, parseTsvRows, renderBenchmarkReportMarkdown, retrievalMetricsAtCutoff, routing_exports as routing, runBenchmarkAdapter, summarizeBenchmarkCampaign };
@@ -1,6 +1,6 @@
1
1
  import { t as __exportAll } from "./rolldown-runtime-8H4AJuhK.js";
2
- import { Y as runCampaign, Z as fsCampaignStorage } from "./skillopt-optimization-method-Cl4XPkLC.js";
3
- import "./campaign-B1c1T0kv.js";
2
+ import { Y as runCampaign, Z as fsCampaignStorage } from "./skillopt-optimization-method-0UmPD6aP.js";
3
+ import "./campaign-RVIqtJh0.js";
4
4
  import { join } from "node:path";
5
5
  //#region src/benchmarks/calibration.ts
6
6
  async function calibrateBenchmarkMetric(options) {
@@ -751,4 +751,4 @@ var benchmarks_exports = /* @__PURE__ */ __exportAll({
751
751
  //#endregion
752
752
  export { deterministicSplit as _, normalizeRetrievedDocumentIds as a, parseJsonlRows as c, retrievalMetricsAtCutoff as d, renderBenchmarkReportMarkdown as f, BENCHMARK_SPLIT_SEED as g, routing_exports as h, evaluateStandardRetrieval as i, parseQrels as l, summarizeBenchmarkCampaign as m, buildStandardRetrievalItems as n, parseBeirCorpusJsonl as o, runBenchmarkAdapter as p, createRetrievalIdBenchmarkAdapter as r, parseBeirQueriesJsonl as s, benchmarks_exports as t, parseTsvRows as u, calibrateBenchmarkMetric as v };
753
753
 
754
- //# sourceMappingURL=benchmarks-Dw1Wv_JQ.js.map
754
+ //# sourceMappingURL=benchmarks-Mtu251Jz.js.map