jupytermind 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (193) hide show
  1. package/.github/skills/ai-chemistry-scientist/SKILL.md +97 -0
  2. package/.github/skills/ai-chemistry-scientist/manifest.json +156 -0
  3. package/.github/skills/ai-data-scientist/SKILL.md +330 -0
  4. package/.github/skills/ai-genomics-scientist/SKILL.md +98 -0
  5. package/.github/skills/ai-genomics-scientist/manifest.json +93 -0
  6. package/.github/skills/ai-materials-scientist/SKILL.md +51 -0
  7. package/.github/skills/ai-materials-scientist/manifest.json +58 -0
  8. package/.github/skills/ai-scientist/SKILL.md +69 -0
  9. package/.github/skills/ai-scientist/manifest.json +61 -0
  10. package/.github/skills/ai-structural-biology-scientist/SKILL.md +67 -0
  11. package/.github/skills/ai-structural-biology-scientist/manifest.json +72 -0
  12. package/.github/skills/japanese-prose/NOTICE.md +17 -0
  13. package/.github/skills/japanese-prose/SKILL.md +111 -0
  14. package/.github/skills/japanese-prose/references/review-workflow.md +50 -0
  15. package/.github/skills/japanese-prose/references/scoring.md +24 -0
  16. package/.github/skills/japanese-prose/references/writing-guidelines.md +60 -0
  17. package/.github/skills/japanese-prose/scripts/core.py +192 -0
  18. package/.github/skills/japanese-prose/scripts/fixtures/natural.md +5 -0
  19. package/.github/skills/japanese-prose/scripts/fixtures/unnatural.md +5 -0
  20. package/.github/skills/japanese-prose/scripts/lint.py +378 -0
  21. package/.github/skills/japanese-prose/scripts/outline.py +68 -0
  22. package/.github/skills/japanese-prose/scripts/terms.py +112 -0
  23. package/.github/skills/japanese-prose/scripts/test_engine.py +117 -0
  24. package/.github/skills/presentation-planner/SKILL.md +257 -0
  25. package/.github/skills/presentation-planner/assets/design-templates/data-report.yaml +97 -0
  26. package/.github/skills/presentation-planner/assets/design-templates/executive-proposal.yaml +92 -0
  27. package/.github/skills/presentation-planner/assets/design-templates/technical-briefing.yaml +96 -0
  28. package/.github/skills/presentation-planner/assets/scenario-templates/data-report.md +47 -0
  29. package/.github/skills/presentation-planner/assets/scenario-templates/executive-decision.md +43 -0
  30. package/.github/skills/presentation-planner/assets/scenario-templates/technical-briefing.md +45 -0
  31. package/.github/skills/presentation-planner/references/customizing-design-templates.md +160 -0
  32. package/.github/skills/presentation-planner/references/design-spec-schema.md +72 -0
  33. package/.github/skills/presentation-planner/references/handoff-contract.md +49 -0
  34. package/.github/skills/presentation-planner/references/responsibility-boundary.md +32 -0
  35. package/.github/skills/presentation-planner/references/scenario-templates.md +55 -0
  36. package/.github/skills/tech-writer/SKILL.md +434 -0
  37. package/.github/skills/tech-writer/assets/templates/blueprint.md +187 -0
  38. package/.github/skills/tech-writer/assets/templates/design-doc.md +29 -0
  39. package/.github/skills/tech-writer/assets/templates/migration-plan.md +173 -0
  40. package/.github/skills/tech-writer/assets/templates/operations-runbook.md +202 -0
  41. package/.github/skills/tech-writer/assets/templates/pr-description.md +23 -0
  42. package/.github/skills/tech-writer/assets/templates/qiita.md +44 -0
  43. package/.github/skills/tech-writer/assets/templates/readme.md +38 -0
  44. package/.github/skills/tech-writer/assets/templates/requirements-definition.md +170 -0
  45. package/.github/skills/tech-writer/assets/templates/rfi.md +113 -0
  46. package/.github/skills/tech-writer/assets/templates/rfp.md +180 -0
  47. package/.github/skills/tech-writer/assets/templates/security-design.md +167 -0
  48. package/.github/skills/tech-writer/assets/templates/system-design.md +220 -0
  49. package/.github/skills/tech-writer/assets/templates/technical-proposal.md +112 -0
  50. package/.github/skills/tech-writer/assets/templates/test-plan.md +153 -0
  51. package/.github/skills/tech-writer/assets/templates/user-manual.md +22 -0
  52. package/.github/skills/tech-writer/assets/templates/white-paper.md +192 -0
  53. package/.github/skills/tech-writer/references/doctypes/api-docs.md +33 -0
  54. package/.github/skills/tech-writer/references/doctypes/blueprint.md +81 -0
  55. package/.github/skills/tech-writer/references/doctypes/code-comments.md +39 -0
  56. package/.github/skills/tech-writer/references/doctypes/design-doc.md +42 -0
  57. package/.github/skills/tech-writer/references/doctypes/migration-plan.md +63 -0
  58. package/.github/skills/tech-writer/references/doctypes/operations-runbook.md +63 -0
  59. package/.github/skills/tech-writer/references/doctypes/pr-commit.md +82 -0
  60. package/.github/skills/tech-writer/references/doctypes/qiita.md +75 -0
  61. package/.github/skills/tech-writer/references/doctypes/readme.md +43 -0
  62. package/.github/skills/tech-writer/references/doctypes/release-notes.md +30 -0
  63. package/.github/skills/tech-writer/references/doctypes/requirements-definition.md +61 -0
  64. package/.github/skills/tech-writer/references/doctypes/rfi.md +43 -0
  65. package/.github/skills/tech-writer/references/doctypes/rfp.md +46 -0
  66. package/.github/skills/tech-writer/references/doctypes/security-design.md +71 -0
  67. package/.github/skills/tech-writer/references/doctypes/system-design.md +74 -0
  68. package/.github/skills/tech-writer/references/doctypes/technical-proposal.md +49 -0
  69. package/.github/skills/tech-writer/references/doctypes/test-plan.md +67 -0
  70. package/.github/skills/tech-writer/references/doctypes/user-manual.md +58 -0
  71. package/.github/skills/tech-writer/references/doctypes/white-paper.md +84 -0
  72. package/.github/skills/tech-writer/references/doctypes/zenn.md +66 -0
  73. package/.github/skills/tech-writer/references/japanese-prose-optimization.md +110 -0
  74. package/.github/skills/tech-writer/references/style-constitution.md +104 -0
  75. package/.github/skills/tech-writer/scripts/lint.py +412 -0
  76. package/LICENSE +21 -0
  77. package/README.md +92 -0
  78. package/bin/ai-data-scientist.js +123 -0
  79. package/package.json +41 -0
  80. package/pyproject.toml +45 -0
  81. package/src/ai_chemistry_scientist/__init__.py +0 -0
  82. package/src/ai_chemistry_scientist/admet_prediction.py +71 -0
  83. package/src/ai_chemistry_scientist/bioactivity_classification.py +73 -0
  84. package/src/ai_chemistry_scientist/data/sample_molecules.csv +21 -0
  85. package/src/ai_chemistry_scientist/dispatch.py +369 -0
  86. package/src/ai_chemistry_scientist/docking_score.py +97 -0
  87. package/src/ai_chemistry_scientist/drug_likeness_rules.py +84 -0
  88. package/src/ai_chemistry_scientist/evidence.py +41 -0
  89. package/src/ai_chemistry_scientist/molecular_descriptors.py +97 -0
  90. package/src/ai_chemistry_scientist/molecular_formula_mass.py +40 -0
  91. package/src/ai_chemistry_scientist/molecular_similarity.py +78 -0
  92. package/src/ai_chemistry_scientist/qsar_modeling.py +105 -0
  93. package/src/ai_chemistry_scientist/salt_standardization.py +81 -0
  94. package/src/ai_chemistry_scientist/structural_alerts.py +76 -0
  95. package/src/ai_chemistry_scientist/structure_format_conversion.py +84 -0
  96. package/src/ai_chemistry_scientist/validation.py +70 -0
  97. package/src/ai_data_scientist/__init__.py +0 -0
  98. package/src/ai_data_scientist/analysis_assumptions.py +121 -0
  99. package/src/ai_data_scientist/anomaly_detection.py +39 -0
  100. package/src/ai_data_scientist/automl.py +109 -0
  101. package/src/ai_data_scientist/cleaning.py +56 -0
  102. package/src/ai_data_scientist/cli.py +90 -0
  103. package/src/ai_data_scientist/clustering.py +54 -0
  104. package/src/ai_data_scientist/dashboard.py +33 -0
  105. package/src/ai_data_scientist/data_definition.py +100 -0
  106. package/src/ai_data_scientist/data_quality.py +164 -0
  107. package/src/ai_data_scientist/dataset_validation.py +135 -0
  108. package/src/ai_data_scientist/dependency_pins.py +60 -0
  109. package/src/ai_data_scientist/eda.py +82 -0
  110. package/src/ai_data_scientist/experiment_evaluation.py +635 -0
  111. package/src/ai_data_scientist/explainability.py +340 -0
  112. package/src/ai_data_scientist/feature_engineering.py +163 -0
  113. package/src/ai_data_scientist/gate_config.py +32 -0
  114. package/src/ai_data_scientist/ingestion.py +127 -0
  115. package/src/ai_data_scientist/insight_engine.py +180 -0
  116. package/src/ai_data_scientist/japanese_nlp.py +43 -0
  117. package/src/ai_data_scientist/jupyter_launcher.py +137 -0
  118. package/src/ai_data_scientist/jupyter_mcp_client.py +94 -0
  119. package/src/ai_data_scientist/language_router.py +28 -0
  120. package/src/ai_data_scientist/lifecycle.py +221 -0
  121. package/src/ai_data_scientist/mcp_gateway.py +113 -0
  122. package/src/ai_data_scientist/mcp_runtime.py +194 -0
  123. package/src/ai_data_scientist/mcp_transport.py +53 -0
  124. package/src/ai_data_scientist/ml_modeling.py +451 -0
  125. package/src/ai_data_scientist/model_tuning.py +104 -0
  126. package/src/ai_data_scientist/notebook_audit.py +574 -0
  127. package/src/ai_data_scientist/project_manager.py +243 -0
  128. package/src/ai_data_scientist/report_export.py +73 -0
  129. package/src/ai_data_scientist/sensitivity.py +445 -0
  130. package/src/ai_data_scientist/signal_analysis.py +201 -0
  131. package/src/ai_data_scientist/skill_packaging.py +40 -0
  132. package/src/ai_data_scientist/stats_analysis.py +88 -0
  133. package/src/ai_data_scientist/text_nlp.py +44 -0
  134. package/src/ai_data_scientist/timeseries.py +68 -0
  135. package/src/ai_data_scientist/visualization.py +708 -0
  136. package/src/ai_genomics_scientist/__init__.py +1 -0
  137. package/src/ai_genomics_scientist/differential_expression.py +147 -0
  138. package/src/ai_genomics_scientist/dispatch.py +267 -0
  139. package/src/ai_genomics_scientist/evidence.py +45 -0
  140. package/src/ai_genomics_scientist/gene_set_enrichment.py +76 -0
  141. package/src/ai_genomics_scientist/sequence_alignment.py +97 -0
  142. package/src/ai_genomics_scientist/sequence_features.py +111 -0
  143. package/src/ai_genomics_scientist/splice_site_scoring.py +66 -0
  144. package/src/ai_genomics_scientist/validation.py +83 -0
  145. package/src/ai_genomics_scientist/variant_effect.py +147 -0
  146. package/src/ai_genomics_scientist/variant_pathogenicity.py +125 -0
  147. package/src/ai_materials_scientist/__init__.py +0 -0
  148. package/src/ai_materials_scientist/calphad.py +117 -0
  149. package/src/ai_materials_scientist/classical_monte_carlo.py +165 -0
  150. package/src/ai_materials_scientist/crystal_plasticity.py +184 -0
  151. package/src/ai_materials_scientist/dispatch.py +100 -0
  152. package/src/ai_materials_scientist/evidence.py +84 -0
  153. package/src/ai_materials_scientist/fem.py +279 -0
  154. package/src/ai_materials_scientist/kinetic_monte_carlo.py +145 -0
  155. package/src/ai_materials_scientist/molecular_dynamics.py +240 -0
  156. package/src/ai_materials_scientist/phase_field.py +167 -0
  157. package/src/ai_materials_scientist/validation.py +70 -0
  158. package/src/ai_scientist/__init__.py +1 -0
  159. package/src/ai_scientist/completion_gate.py +15 -0
  160. package/src/ai_scientist/data_analysis.py +46 -0
  161. package/src/ai_scientist/evidence_registry.py +99 -0
  162. package/src/ai_scientist/experimental_design.py +20 -0
  163. package/src/ai_scientist/language.py +14 -0
  164. package/src/ai_scientist/latex_renderer.py +41 -0
  165. package/src/ai_scientist/literature_review.py +37 -0
  166. package/src/ai_scientist/manifest.py +87 -0
  167. package/src/ai_scientist/manuscript.py +94 -0
  168. package/src/ai_scientist/mcp_config.py +76 -0
  169. package/src/ai_scientist/mcp_external.py +42 -0
  170. package/src/ai_scientist/mcp_failures.py +23 -0
  171. package/src/ai_scientist/mcp_gateway.py +38 -0
  172. package/src/ai_scientist/mcp_managed.py +180 -0
  173. package/src/ai_scientist/npm_packaging.py +49 -0
  174. package/src/ai_scientist/orchestrator.py +133 -0
  175. package/src/ai_scientist/peer_review.py +60 -0
  176. package/src/ai_scientist/phase_gate.py +74 -0
  177. package/src/ai_scientist/phase_state.py +230 -0
  178. package/src/ai_scientist/presentation.py +56 -0
  179. package/src/ai_scientist/project_config.py +31 -0
  180. package/src/ai_scientist/project_handle.py +74 -0
  181. package/src/ai_scientist/reproducibility.py +20 -0
  182. package/src/ai_scientist/research_planning.py +20 -0
  183. package/src/ai_scientist/skill_invocation.py +21 -0
  184. package/src/ai_scientist/tdd_gate.py +99 -0
  185. package/src/ai_structural_biology_scientist/__init__.py +0 -0
  186. package/src/ai_structural_biology_scientist/contact_map.py +87 -0
  187. package/src/ai_structural_biology_scientist/dispatch.py +269 -0
  188. package/src/ai_structural_biology_scientist/evidence.py +43 -0
  189. package/src/ai_structural_biology_scientist/hydrophobicity.py +101 -0
  190. package/src/ai_structural_biology_scientist/protein_docking_score.py +104 -0
  191. package/src/ai_structural_biology_scientist/secondary_structure.py +95 -0
  192. package/src/ai_structural_biology_scientist/structural_similarity.py +74 -0
  193. package/src/ai_structural_biology_scientist/validation.py +100 -0
@@ -0,0 +1,635 @@
1
+ """A/B testing and experiment evaluation.
2
+
3
+ Implements DES-AIDS-020 (REQ-AIDS-022): computes the statistical
4
+ significance of the observed difference between two groups and reports
5
+ the result with a bilingual markdown interpretation.
6
+
7
+ CHANGE-010 (REQ-AIDS-079..081) adds paired_t/wilcoxon/paired_bootstrap
8
+ paths (DES-AIDS-067..069, CODE-AIDS-101..105).
9
+ CHANGE-016 (REQ-AIDS-090..092) adds repeated multi-seed comparison,
10
+ seed-variability adoption thresholds, and three-way holdout selection-bias
11
+ evaluation (DES-AIDS-090..092, CODE-AIDS-131..140).
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import math
17
+ from collections.abc import Callable, Sequence
18
+ from dataclasses import dataclass
19
+
20
+ import numpy as np
21
+ import pandas as pd
22
+ from scipy import stats as scipy_stats
23
+
24
+ MetricFn = Callable[[pd.Series, pd.Series], float]
25
+ SeedComparisonFn = Callable[[int, int | None], tuple[float, float]]
26
+ SelectionBiasFn = Callable[[pd.DataFrame, pd.DataFrame, pd.DataFrame, str, list[str]], dict]
27
+
28
+ _SUPPORTED_TESTS = ("ttest", "paired_t", "wilcoxon", "paired_bootstrap")
29
+
30
+
31
+ # @id CODE-AIDS-101
32
+ # @implements REQ-AIDS-081
33
+ # @design DES-AIDS-069
34
+ @dataclass(frozen=True)
35
+ class ExperimentResult:
36
+ statistic: float
37
+ p_value: float
38
+ interpretation: str
39
+ confidence_interval: tuple[float, float] | None = None
40
+
41
+
42
+ # @id CODE-AIDS-131
43
+ # @implements REQ-AIDS-090
44
+ # @design DES-AIDS-090
45
+ @dataclass(frozen=True)
46
+ class SeedComparisonResult:
47
+ split_seed: int
48
+ model_seed: int | None
49
+ control_metric: float
50
+ treatment_metric: float
51
+ improvement: float
52
+
53
+
54
+ # @id CODE-AIDS-132
55
+ # @implements REQ-AIDS-090
56
+ # @design DES-AIDS-090
57
+ @dataclass(frozen=True)
58
+ class SeedComparisonSummary:
59
+ results: tuple[SeedComparisonResult, ...]
60
+ mean_improvement: float
61
+ seed_variability: float
62
+ sign_counts: dict[str, int]
63
+
64
+
65
+ # @id CODE-AIDS-133
66
+ # @implements REQ-AIDS-091
67
+ # @design DES-AIDS-091
68
+ @dataclass(frozen=True)
69
+ class AdoptionDecision:
70
+ candidate_improvement: float
71
+ threshold: float
72
+ classification: str
73
+
74
+
75
+ # @id CODE-AIDS-134
76
+ # @implements REQ-AIDS-092
77
+ # @design DES-AIDS-092
78
+ @dataclass(frozen=True)
79
+ class SelectionBiasHoldoutResult:
80
+ train_index: tuple
81
+ selection_index: tuple
82
+ evaluation_index: tuple
83
+ selected_candidate: str
84
+ selection_improvement: float
85
+ evaluation_improvement: float
86
+ optimism: float
87
+ assumption_findings: tuple | None = None
88
+
89
+
90
+ def _interpret(p_value: float, language: str) -> str:
91
+ significant = p_value < 0.05
92
+ if language == "ja":
93
+ verdict = "統計的に有意な差があります" if significant else "統計的に有意な差は見られません"
94
+ return f"p値は {p_value:.4g} で、{verdict} (有意水準0.05)。"
95
+ verdict = (
96
+ "a statistically significant difference"
97
+ if significant
98
+ else "no statistically significant difference"
99
+ )
100
+ return f"The p-value is {p_value:.4g}, indicating {verdict} (alpha=0.05)."
101
+
102
+
103
+ def _interpret_bootstrap_interval(
104
+ statistic: float,
105
+ confidence_interval: tuple[float, float],
106
+ confidence_level: float,
107
+ language: str,
108
+ ) -> str:
109
+ percent = confidence_level * 100.0
110
+ if language == "ja":
111
+ return (
112
+ f"推定された treatment-control 差は {statistic:.4g} で、"
113
+ f"{percent:.1f}% ブートストラップ信頼区間は "
114
+ f"[{confidence_interval[0]:.4g}, {confidence_interval[1]:.4g}] です。"
115
+ "この経路は区間推定のみを返し、仮説検定のp値は返しません。"
116
+ )
117
+ return (
118
+ f"The estimated treatment-minus-control difference is {statistic:.4g}, "
119
+ f"with a {percent:.1f}% bootstrap confidence interval of "
120
+ f"[{confidence_interval[0]:.4g}, {confidence_interval[1]:.4g}]. "
121
+ "This path reports interval estimation only and does not provide a hypothesis-test "
122
+ "p-value."
123
+ )
124
+
125
+
126
+ def _as_numeric_series(values: pd.Series, *, name: str) -> pd.Series:
127
+ try:
128
+ numeric = pd.to_numeric(values, errors="raise")
129
+ except (TypeError, ValueError) as exc:
130
+ raise ValueError(f"{name} must contain numeric paired values.") from exc
131
+ if not np.isfinite(numeric.to_numpy(dtype=float, copy=False)).all():
132
+ raise ValueError(f"{name} must contain only finite paired values.")
133
+ return numeric
134
+
135
+
136
+ # @id CODE-AIDS-102
137
+ # @implements REQ-AIDS-079 REQ-AIDS-080 REQ-AIDS-081
138
+ # @design DES-AIDS-067 DES-AIDS-068
139
+ def _validate_paired_inputs(
140
+ control: pd.Series,
141
+ treatment: pd.Series,
142
+ *,
143
+ y_true: pd.Series | None = None,
144
+ metric_fn: MetricFn | None = None,
145
+ iterations: int = 1000,
146
+ confidence_level: float = 0.95,
147
+ ) -> tuple[pd.Series, pd.Series, pd.Series | None]:
148
+ if len(control) != len(treatment):
149
+ raise ValueError("Paired experiment tests require equal-length inputs.")
150
+ if not control.index.equals(treatment.index):
151
+ raise ValueError(
152
+ "Paired experiment tests require control and treatment to share identical indexes."
153
+ )
154
+
155
+ control_series = _as_numeric_series(control, name="control")
156
+ treatment_series = _as_numeric_series(treatment, name="treatment")
157
+
158
+ if (metric_fn is None) != (y_true is None):
159
+ raise ValueError("Paired bootstrap requires metric_fn and y_true together.")
160
+
161
+ truth_series = None
162
+ if y_true is not None:
163
+ if len(y_true) != len(control):
164
+ raise ValueError(
165
+ "Paired bootstrap requires y_true to align with control and treatment."
166
+ )
167
+ if not y_true.index.equals(control.index):
168
+ raise ValueError("Paired bootstrap requires y_true to share the paired input index.")
169
+ truth_series = _as_numeric_series(y_true, name="y_true")
170
+
171
+ if isinstance(iterations, bool) or int(iterations) != iterations or int(iterations) <= 0:
172
+ raise ValueError("Paired bootstrap requires iterations to be a positive integer.")
173
+ if not 0.0 < float(confidence_level) < 1.0:
174
+ raise ValueError("Paired bootstrap requires confidence_level to be between 0 and 1.")
175
+
176
+ return control_series, treatment_series, truth_series
177
+
178
+
179
+ # @id CODE-AIDS-022
180
+ # @implements REQ-AIDS-022
181
+ # @design DES-AIDS-020
182
+ def _run_independent_ttest(
183
+ control: pd.Series,
184
+ treatment: pd.Series,
185
+ *,
186
+ language: str,
187
+ ) -> ExperimentResult:
188
+ statistic, p_value = scipy_stats.ttest_ind(control, treatment)
189
+ return ExperimentResult(
190
+ statistic=float(statistic),
191
+ p_value=float(p_value),
192
+ interpretation=_interpret(float(p_value), language),
193
+ )
194
+
195
+
196
+ # @id CODE-AIDS-103
197
+ # @implements REQ-AIDS-079 REQ-AIDS-080
198
+ # @design DES-AIDS-067 DES-AIDS-069
199
+ def _run_paired_test(
200
+ control: pd.Series,
201
+ treatment: pd.Series,
202
+ *,
203
+ test: str,
204
+ language: str,
205
+ ) -> ExperimentResult:
206
+ control_series, treatment_series, _ = _validate_paired_inputs(control, treatment)
207
+ if test == "paired_t":
208
+ statistic, p_value = scipy_stats.ttest_rel(control_series, treatment_series)
209
+ else:
210
+ statistic, p_value = scipy_stats.wilcoxon(control_series, treatment_series)
211
+ return ExperimentResult(
212
+ statistic=float(statistic),
213
+ p_value=float(p_value),
214
+ interpretation=_interpret(float(p_value), language),
215
+ )
216
+
217
+
218
+ def _compute_difference(
219
+ control: pd.Series,
220
+ treatment: pd.Series,
221
+ *,
222
+ y_true: pd.Series | None = None,
223
+ metric_fn: MetricFn | None = None,
224
+ ) -> float:
225
+ if metric_fn is None:
226
+ return float((treatment - control).mean())
227
+ assert y_true is not None
228
+ return float(metric_fn(y_true, treatment) - metric_fn(y_true, control))
229
+
230
+
231
+ def _sample_paired_indices(
232
+ sample_size: int,
233
+ rng: np.random.Generator,
234
+ *,
235
+ y_true: pd.Series | None = None,
236
+ ) -> np.ndarray:
237
+ if y_true is None:
238
+ return rng.integers(0, sample_size, size=sample_size)
239
+
240
+ unique_counts = y_true.value_counts(sort=False)
241
+ if 1 < len(unique_counts) < sample_size:
242
+ return np.concatenate(
243
+ [
244
+ rng.choice(
245
+ np.flatnonzero(y_true.to_numpy() == label),
246
+ size=int(count),
247
+ replace=True,
248
+ )
249
+ for label, count in unique_counts.items()
250
+ ]
251
+ )
252
+ return rng.integers(0, sample_size, size=sample_size)
253
+
254
+
255
+ # @id CODE-AIDS-135
256
+ # @implements REQ-AIDS-090 REQ-AIDS-092
257
+ # @design DES-AIDS-090 DES-AIDS-092
258
+ def _as_float_pair(values: tuple[float, float]) -> tuple[float, float]:
259
+ if len(values) != 2:
260
+ raise ValueError("Comparison callbacks must return exactly two metric values.")
261
+ control_metric, treatment_metric = values
262
+ control_metric = float(control_metric)
263
+ treatment_metric = float(treatment_metric)
264
+ if not np.isfinite([control_metric, treatment_metric]).all():
265
+ raise ValueError("Comparison callbacks must return only finite metric values.")
266
+ return control_metric, treatment_metric
267
+
268
+
269
+ # @id CODE-AIDS-136
270
+ # @implements REQ-AIDS-090
271
+ # @design DES-AIDS-090
272
+ def summarize_seed_variability(
273
+ compare_fn: SeedComparisonFn,
274
+ *,
275
+ split_seeds: Sequence[int],
276
+ model_seeds: Sequence[int | None] | None = None,
277
+ ) -> SeedComparisonSummary:
278
+ """Repeat one comparison across seeds and summarize observed variability."""
279
+ effective_split_seeds = list(split_seeds)
280
+ if len(effective_split_seeds) == 0:
281
+ raise ValueError("split_seeds must contain at least one seed.")
282
+
283
+ effective_model_seeds = (
284
+ [None] * len(effective_split_seeds) if model_seeds is None else list(model_seeds)
285
+ )
286
+ if len(effective_model_seeds) != len(effective_split_seeds):
287
+ raise ValueError("model_seeds must be omitted or match split_seeds in length.")
288
+
289
+ results: list[SeedComparisonResult] = []
290
+ improvements: list[float] = []
291
+ sign_counts = {"positive": 0, "zero": 0, "negative": 0}
292
+
293
+ for split_seed, model_seed in zip(effective_split_seeds, effective_model_seeds, strict=True):
294
+ control_metric, treatment_metric = _as_float_pair(compare_fn(split_seed, model_seed))
295
+ improvement = float(treatment_metric - control_metric)
296
+ results.append(
297
+ SeedComparisonResult(
298
+ split_seed=int(split_seed),
299
+ model_seed=None if model_seed is None else int(model_seed),
300
+ control_metric=control_metric,
301
+ treatment_metric=treatment_metric,
302
+ improvement=improvement,
303
+ )
304
+ )
305
+ improvements.append(improvement)
306
+ if improvement > 0:
307
+ sign_counts["positive"] += 1
308
+ elif improvement < 0:
309
+ sign_counts["negative"] += 1
310
+ else:
311
+ sign_counts["zero"] += 1
312
+
313
+ return SeedComparisonSummary(
314
+ results=tuple(results),
315
+ mean_improvement=float(np.mean(improvements)),
316
+ seed_variability=float(max(improvements) - min(improvements)),
317
+ sign_counts=sign_counts,
318
+ )
319
+
320
+
321
+ # @id CODE-AIDS-137
322
+ # @implements REQ-AIDS-091
323
+ # @design DES-AIDS-091
324
+ def judge_improvement(
325
+ summary: SeedComparisonSummary,
326
+ *,
327
+ candidate_improvement: float | None = None,
328
+ threshold: float | None = None,
329
+ ) -> AdoptionDecision:
330
+ """Classify an improvement against the observed seed-variability threshold."""
331
+ effective_improvement = (
332
+ summary.mean_improvement if candidate_improvement is None else float(candidate_improvement)
333
+ )
334
+ effective_threshold = summary.seed_variability if threshold is None else float(threshold)
335
+ if not np.isfinite(effective_improvement):
336
+ raise ValueError("judge_improvement requires a finite candidate_improvement.")
337
+ if not np.isfinite(effective_threshold):
338
+ raise ValueError("judge_improvement requires a finite threshold.")
339
+ if effective_improvement < 0:
340
+ classification = "regression"
341
+ elif effective_improvement <= effective_threshold:
342
+ classification = "within_seed_variability"
343
+ else:
344
+ classification = "adopt"
345
+ return AdoptionDecision(
346
+ candidate_improvement=effective_improvement,
347
+ threshold=effective_threshold,
348
+ classification=classification,
349
+ )
350
+
351
+
352
+ # @id CODE-AIDS-138
353
+ # @implements REQ-AIDS-092
354
+ # @design DES-AIDS-092
355
+ def _split_holdout_partitions(
356
+ df: pd.DataFrame,
357
+ *,
358
+ split_seed: int,
359
+ train_fraction: float,
360
+ selection_fraction: float,
361
+ evaluation_fraction: float,
362
+ ) -> tuple[pd.DataFrame, pd.DataFrame, pd.DataFrame]:
363
+ fractions = np.array(
364
+ [float(train_fraction), float(selection_fraction), float(evaluation_fraction)], dtype=float
365
+ )
366
+ if np.any(fractions <= 0.0):
367
+ raise ValueError("train/selection/evaluation fractions must be positive.")
368
+ if not np.isclose(float(fractions.sum()), 1.0):
369
+ raise ValueError("train/selection/evaluation fractions must sum to 1.")
370
+
371
+ total_rows = len(df)
372
+ if total_rows < 3:
373
+ raise ValueError("Selection-bias holdout evaluation requires at least 3 rows.")
374
+
375
+ raw_counts = fractions * total_rows
376
+ counts = np.floor(raw_counts).astype(int)
377
+ remainder = total_rows - int(counts.sum())
378
+ if remainder > 0:
379
+ residual_order = np.argsort(-(raw_counts - counts))
380
+ for position in residual_order[:remainder]:
381
+ counts[position] += 1
382
+ while np.any(counts == 0):
383
+ zero_positions = np.flatnonzero(counts == 0)
384
+ donor_candidates = np.flatnonzero(counts > 1)
385
+ if len(donor_candidates) == 0:
386
+ raise ValueError(
387
+ "train/selection/evaluation fractions must yield non-empty partitions."
388
+ )
389
+ donor_position = int(donor_candidates[np.argmax(counts[donor_candidates])])
390
+ counts[donor_position] -= 1
391
+ counts[int(zero_positions[0])] += 1
392
+
393
+ rng = np.random.default_rng(split_seed)
394
+ shuffled_positions = rng.permutation(total_rows)
395
+ train_end = int(counts[0])
396
+ selection_end = int(counts[0] + counts[1])
397
+ train_positions = shuffled_positions[:train_end]
398
+ selection_positions = shuffled_positions[train_end:selection_end]
399
+ evaluation_positions = shuffled_positions[selection_end:]
400
+
401
+ return (
402
+ df.iloc[train_positions].copy(),
403
+ df.iloc[selection_positions].copy(),
404
+ df.iloc[evaluation_positions].copy(),
405
+ )
406
+
407
+
408
+ # @id CODE-AIDS-139
409
+ # @implements REQ-AIDS-092
410
+ # @design DES-AIDS-092
411
+ def _extract_selection_bias_metrics(
412
+ evaluation_payload: dict,
413
+ *,
414
+ baseline: str,
415
+ ) -> tuple[dict[str, float], dict[str, float]]:
416
+ try:
417
+ selection_metrics = evaluation_payload["selection_metrics"]
418
+ evaluation_metrics = evaluation_payload["evaluation_metrics"]
419
+ except KeyError as exc:
420
+ raise ValueError(
421
+ "evaluate_candidate_fn must return selection_metrics and evaluation_metrics."
422
+ ) from exc
423
+ normalized_selection = {str(name): float(value) for name, value in selection_metrics.items()}
424
+ normalized_evaluation = {str(name): float(value) for name, value in evaluation_metrics.items()}
425
+ for metric_name, metric_map in (
426
+ ("selection_metrics", normalized_selection),
427
+ ("evaluation_metrics", normalized_evaluation),
428
+ ):
429
+ if baseline not in metric_map:
430
+ raise ValueError(f"evaluate_candidate_fn must include baseline in {metric_name}.")
431
+ if not np.isfinite(list(metric_map.values())).all():
432
+ raise ValueError(
433
+ f"evaluate_candidate_fn must return only finite values in {metric_name}."
434
+ )
435
+ return normalized_selection, normalized_evaluation
436
+
437
+
438
+ def _choose_selection_winner(
439
+ selection_metrics: dict[str, float],
440
+ *,
441
+ baseline: str,
442
+ candidates: list[str],
443
+ ) -> str:
444
+ missing_candidates = [
445
+ candidate for candidate in candidates if candidate not in selection_metrics
446
+ ]
447
+ if missing_candidates:
448
+ raise ValueError(
449
+ "evaluate_candidate_fn must include every candidate in selection_metrics: "
450
+ + ", ".join(missing_candidates)
451
+ )
452
+ return max(
453
+ candidates,
454
+ key=lambda candidate: (selection_metrics[candidate], -candidates.index(candidate)),
455
+ )
456
+
457
+
458
+ # @id CODE-AIDS-140
459
+ # @implements REQ-AIDS-092
460
+ # @design DES-AIDS-092
461
+ def evaluate_selection_bias_holdout(
462
+ df: pd.DataFrame,
463
+ target: str,
464
+ *,
465
+ baseline: str,
466
+ candidates: list[str],
467
+ evaluate_candidate_fn: SelectionBiasFn,
468
+ split_seed: int = 42,
469
+ train_fraction: float = 0.6,
470
+ selection_fraction: float = 0.2,
471
+ evaluation_fraction: float = 0.2,
472
+ assumption_manifest: object | None = None,
473
+ ) -> SelectionBiasHoldoutResult:
474
+ """Evaluate selection optimism using disjoint train/selection/evaluation subsets."""
475
+ if target not in df.columns:
476
+ raise ValueError(f"target column {target!r} is not present in the dataframe.")
477
+ train_df, selection_df, evaluation_df = _split_holdout_partitions(
478
+ df,
479
+ split_seed=split_seed,
480
+ train_fraction=train_fraction,
481
+ selection_fraction=selection_fraction,
482
+ evaluation_fraction=evaluation_fraction,
483
+ )
484
+
485
+ evaluation_payload = evaluate_candidate_fn(
486
+ train_df,
487
+ selection_df,
488
+ evaluation_df,
489
+ baseline,
490
+ candidates,
491
+ )
492
+ selection_metrics, evaluation_metrics = _extract_selection_bias_metrics(
493
+ evaluation_payload,
494
+ baseline=baseline,
495
+ )
496
+ selected_candidate = _choose_selection_winner(
497
+ selection_metrics,
498
+ baseline=baseline,
499
+ candidates=candidates,
500
+ )
501
+ missing_evaluation_candidates = [
502
+ candidate for candidate in candidates if candidate not in evaluation_metrics
503
+ ]
504
+ if missing_evaluation_candidates:
505
+ raise ValueError(
506
+ "evaluate_candidate_fn must include every candidate in evaluation_metrics: "
507
+ + ", ".join(missing_evaluation_candidates)
508
+ )
509
+ selection_improvement = float(selection_metrics[selected_candidate]) - float(
510
+ selection_metrics[baseline]
511
+ )
512
+ evaluation_improvement = float(evaluation_metrics[selected_candidate]) - float(
513
+ evaluation_metrics[baseline]
514
+ )
515
+
516
+ assumption_findings = None
517
+ if assumption_manifest is not None:
518
+ from ai_data_scientist.analysis_assumptions import check_manifest
519
+
520
+ assumption_findings = check_manifest(assumption_manifest)
521
+
522
+ return SelectionBiasHoldoutResult(
523
+ train_index=tuple(train_df.index),
524
+ selection_index=tuple(selection_df.index),
525
+ evaluation_index=tuple(evaluation_df.index),
526
+ selected_candidate=selected_candidate,
527
+ selection_improvement=selection_improvement,
528
+ evaluation_improvement=evaluation_improvement,
529
+ optimism=float(selection_improvement - evaluation_improvement),
530
+ assumption_findings=assumption_findings,
531
+ )
532
+
533
+
534
+ # @id CODE-AIDS-104
535
+ # @implements REQ-AIDS-081
536
+ # @design DES-AIDS-068 DES-AIDS-069
537
+ def _run_paired_bootstrap(
538
+ control: pd.Series,
539
+ treatment: pd.Series,
540
+ *,
541
+ language: str,
542
+ y_true: pd.Series | None = None,
543
+ metric_fn: MetricFn | None = None,
544
+ iterations: int = 1000,
545
+ confidence_level: float = 0.95,
546
+ random_state: int | None = None,
547
+ ) -> ExperimentResult:
548
+ control_series, treatment_series, truth_series = _validate_paired_inputs(
549
+ control,
550
+ treatment,
551
+ y_true=y_true,
552
+ metric_fn=metric_fn,
553
+ iterations=iterations,
554
+ confidence_level=confidence_level,
555
+ )
556
+
557
+ observed_difference = _compute_difference(
558
+ control_series,
559
+ treatment_series,
560
+ y_true=truth_series,
561
+ metric_fn=metric_fn,
562
+ )
563
+
564
+ rng = np.random.default_rng(random_state)
565
+ sample_size = len(control_series)
566
+ bootstrap_differences = np.empty(int(iterations), dtype=float)
567
+
568
+ for index in range(int(iterations)):
569
+ sample_indices = _sample_paired_indices(sample_size, rng, y_true=truth_series)
570
+ control_sample = control_series.iloc[sample_indices].reset_index(drop=True)
571
+ treatment_sample = treatment_series.iloc[sample_indices].reset_index(drop=True)
572
+ truth_sample = (
573
+ None
574
+ if truth_series is None
575
+ else truth_series.iloc[sample_indices].reset_index(drop=True)
576
+ )
577
+ bootstrap_differences[index] = _compute_difference(
578
+ control_sample,
579
+ treatment_sample,
580
+ y_true=truth_sample,
581
+ metric_fn=metric_fn,
582
+ )
583
+
584
+ alpha = 1.0 - float(confidence_level)
585
+ confidence_interval = (
586
+ float(np.quantile(bootstrap_differences, alpha / 2.0)),
587
+ float(np.quantile(bootstrap_differences, 1.0 - (alpha / 2.0))),
588
+ )
589
+
590
+ return ExperimentResult(
591
+ statistic=observed_difference,
592
+ p_value=math.nan,
593
+ interpretation=_interpret_bootstrap_interval(
594
+ observed_difference,
595
+ confidence_interval,
596
+ float(confidence_level),
597
+ language,
598
+ ),
599
+ confidence_interval=confidence_interval,
600
+ )
601
+
602
+
603
+ # @id CODE-AIDS-105
604
+ # @implements REQ-AIDS-079 REQ-AIDS-080 REQ-AIDS-081
605
+ # @design DES-AIDS-067 DES-AIDS-068 DES-AIDS-069
606
+ def evaluate_experiment(
607
+ control: pd.Series,
608
+ treatment: pd.Series,
609
+ test: str = "ttest",
610
+ language: str = "en",
611
+ *,
612
+ y_true: pd.Series | None = None,
613
+ metric_fn: MetricFn | None = None,
614
+ iterations: int = 1000,
615
+ confidence_level: float = 0.95,
616
+ random_state: int | None = None,
617
+ ) -> ExperimentResult:
618
+ """Compute the significance of the difference between ``control`` and ``treatment``."""
619
+ if test not in _SUPPORTED_TESTS:
620
+ raise ValueError(f"Unsupported experiment test: {test!r}")
621
+
622
+ if test == "ttest":
623
+ return _run_independent_ttest(control, treatment, language=language)
624
+ if test in {"paired_t", "wilcoxon"}:
625
+ return _run_paired_test(control, treatment, test=test, language=language)
626
+ return _run_paired_bootstrap(
627
+ control,
628
+ treatment,
629
+ language=language,
630
+ y_true=y_true,
631
+ metric_fn=metric_fn,
632
+ iterations=iterations,
633
+ confidence_level=confidence_level,
634
+ random_state=random_state,
635
+ )