jupytermind 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/skills/ai-chemistry-scientist/SKILL.md +97 -0
- package/.github/skills/ai-chemistry-scientist/manifest.json +156 -0
- package/.github/skills/ai-data-scientist/SKILL.md +330 -0
- package/.github/skills/ai-genomics-scientist/SKILL.md +98 -0
- package/.github/skills/ai-genomics-scientist/manifest.json +93 -0
- package/.github/skills/ai-materials-scientist/SKILL.md +51 -0
- package/.github/skills/ai-materials-scientist/manifest.json +58 -0
- package/.github/skills/ai-scientist/SKILL.md +69 -0
- package/.github/skills/ai-scientist/manifest.json +61 -0
- package/.github/skills/ai-structural-biology-scientist/SKILL.md +67 -0
- package/.github/skills/ai-structural-biology-scientist/manifest.json +72 -0
- package/.github/skills/japanese-prose/NOTICE.md +17 -0
- package/.github/skills/japanese-prose/SKILL.md +111 -0
- package/.github/skills/japanese-prose/references/review-workflow.md +50 -0
- package/.github/skills/japanese-prose/references/scoring.md +24 -0
- package/.github/skills/japanese-prose/references/writing-guidelines.md +60 -0
- package/.github/skills/japanese-prose/scripts/core.py +192 -0
- package/.github/skills/japanese-prose/scripts/fixtures/natural.md +5 -0
- package/.github/skills/japanese-prose/scripts/fixtures/unnatural.md +5 -0
- package/.github/skills/japanese-prose/scripts/lint.py +378 -0
- package/.github/skills/japanese-prose/scripts/outline.py +68 -0
- package/.github/skills/japanese-prose/scripts/terms.py +112 -0
- package/.github/skills/japanese-prose/scripts/test_engine.py +117 -0
- package/.github/skills/presentation-planner/SKILL.md +257 -0
- package/.github/skills/presentation-planner/assets/design-templates/data-report.yaml +97 -0
- package/.github/skills/presentation-planner/assets/design-templates/executive-proposal.yaml +92 -0
- package/.github/skills/presentation-planner/assets/design-templates/technical-briefing.yaml +96 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/data-report.md +47 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/executive-decision.md +43 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/technical-briefing.md +45 -0
- package/.github/skills/presentation-planner/references/customizing-design-templates.md +160 -0
- package/.github/skills/presentation-planner/references/design-spec-schema.md +72 -0
- package/.github/skills/presentation-planner/references/handoff-contract.md +49 -0
- package/.github/skills/presentation-planner/references/responsibility-boundary.md +32 -0
- package/.github/skills/presentation-planner/references/scenario-templates.md +55 -0
- package/.github/skills/tech-writer/SKILL.md +434 -0
- package/.github/skills/tech-writer/assets/templates/blueprint.md +187 -0
- package/.github/skills/tech-writer/assets/templates/design-doc.md +29 -0
- package/.github/skills/tech-writer/assets/templates/migration-plan.md +173 -0
- package/.github/skills/tech-writer/assets/templates/operations-runbook.md +202 -0
- package/.github/skills/tech-writer/assets/templates/pr-description.md +23 -0
- package/.github/skills/tech-writer/assets/templates/qiita.md +44 -0
- package/.github/skills/tech-writer/assets/templates/readme.md +38 -0
- package/.github/skills/tech-writer/assets/templates/requirements-definition.md +170 -0
- package/.github/skills/tech-writer/assets/templates/rfi.md +113 -0
- package/.github/skills/tech-writer/assets/templates/rfp.md +180 -0
- package/.github/skills/tech-writer/assets/templates/security-design.md +167 -0
- package/.github/skills/tech-writer/assets/templates/system-design.md +220 -0
- package/.github/skills/tech-writer/assets/templates/technical-proposal.md +112 -0
- package/.github/skills/tech-writer/assets/templates/test-plan.md +153 -0
- package/.github/skills/tech-writer/assets/templates/user-manual.md +22 -0
- package/.github/skills/tech-writer/assets/templates/white-paper.md +192 -0
- package/.github/skills/tech-writer/references/doctypes/api-docs.md +33 -0
- package/.github/skills/tech-writer/references/doctypes/blueprint.md +81 -0
- package/.github/skills/tech-writer/references/doctypes/code-comments.md +39 -0
- package/.github/skills/tech-writer/references/doctypes/design-doc.md +42 -0
- package/.github/skills/tech-writer/references/doctypes/migration-plan.md +63 -0
- package/.github/skills/tech-writer/references/doctypes/operations-runbook.md +63 -0
- package/.github/skills/tech-writer/references/doctypes/pr-commit.md +82 -0
- package/.github/skills/tech-writer/references/doctypes/qiita.md +75 -0
- package/.github/skills/tech-writer/references/doctypes/readme.md +43 -0
- package/.github/skills/tech-writer/references/doctypes/release-notes.md +30 -0
- package/.github/skills/tech-writer/references/doctypes/requirements-definition.md +61 -0
- package/.github/skills/tech-writer/references/doctypes/rfi.md +43 -0
- package/.github/skills/tech-writer/references/doctypes/rfp.md +46 -0
- package/.github/skills/tech-writer/references/doctypes/security-design.md +71 -0
- package/.github/skills/tech-writer/references/doctypes/system-design.md +74 -0
- package/.github/skills/tech-writer/references/doctypes/technical-proposal.md +49 -0
- package/.github/skills/tech-writer/references/doctypes/test-plan.md +67 -0
- package/.github/skills/tech-writer/references/doctypes/user-manual.md +58 -0
- package/.github/skills/tech-writer/references/doctypes/white-paper.md +84 -0
- package/.github/skills/tech-writer/references/doctypes/zenn.md +66 -0
- package/.github/skills/tech-writer/references/japanese-prose-optimization.md +110 -0
- package/.github/skills/tech-writer/references/style-constitution.md +104 -0
- package/.github/skills/tech-writer/scripts/lint.py +412 -0
- package/LICENSE +21 -0
- package/README.md +92 -0
- package/bin/ai-data-scientist.js +123 -0
- package/package.json +41 -0
- package/pyproject.toml +45 -0
- package/src/ai_chemistry_scientist/__init__.py +0 -0
- package/src/ai_chemistry_scientist/admet_prediction.py +71 -0
- package/src/ai_chemistry_scientist/bioactivity_classification.py +73 -0
- package/src/ai_chemistry_scientist/data/sample_molecules.csv +21 -0
- package/src/ai_chemistry_scientist/dispatch.py +369 -0
- package/src/ai_chemistry_scientist/docking_score.py +97 -0
- package/src/ai_chemistry_scientist/drug_likeness_rules.py +84 -0
- package/src/ai_chemistry_scientist/evidence.py +41 -0
- package/src/ai_chemistry_scientist/molecular_descriptors.py +97 -0
- package/src/ai_chemistry_scientist/molecular_formula_mass.py +40 -0
- package/src/ai_chemistry_scientist/molecular_similarity.py +78 -0
- package/src/ai_chemistry_scientist/qsar_modeling.py +105 -0
- package/src/ai_chemistry_scientist/salt_standardization.py +81 -0
- package/src/ai_chemistry_scientist/structural_alerts.py +76 -0
- package/src/ai_chemistry_scientist/structure_format_conversion.py +84 -0
- package/src/ai_chemistry_scientist/validation.py +70 -0
- package/src/ai_data_scientist/__init__.py +0 -0
- package/src/ai_data_scientist/analysis_assumptions.py +121 -0
- package/src/ai_data_scientist/anomaly_detection.py +39 -0
- package/src/ai_data_scientist/automl.py +109 -0
- package/src/ai_data_scientist/cleaning.py +56 -0
- package/src/ai_data_scientist/cli.py +90 -0
- package/src/ai_data_scientist/clustering.py +54 -0
- package/src/ai_data_scientist/dashboard.py +33 -0
- package/src/ai_data_scientist/data_definition.py +100 -0
- package/src/ai_data_scientist/data_quality.py +164 -0
- package/src/ai_data_scientist/dataset_validation.py +135 -0
- package/src/ai_data_scientist/dependency_pins.py +60 -0
- package/src/ai_data_scientist/eda.py +82 -0
- package/src/ai_data_scientist/experiment_evaluation.py +635 -0
- package/src/ai_data_scientist/explainability.py +340 -0
- package/src/ai_data_scientist/feature_engineering.py +163 -0
- package/src/ai_data_scientist/gate_config.py +32 -0
- package/src/ai_data_scientist/ingestion.py +127 -0
- package/src/ai_data_scientist/insight_engine.py +180 -0
- package/src/ai_data_scientist/japanese_nlp.py +43 -0
- package/src/ai_data_scientist/jupyter_launcher.py +137 -0
- package/src/ai_data_scientist/jupyter_mcp_client.py +94 -0
- package/src/ai_data_scientist/language_router.py +28 -0
- package/src/ai_data_scientist/lifecycle.py +221 -0
- package/src/ai_data_scientist/mcp_gateway.py +113 -0
- package/src/ai_data_scientist/mcp_runtime.py +194 -0
- package/src/ai_data_scientist/mcp_transport.py +53 -0
- package/src/ai_data_scientist/ml_modeling.py +451 -0
- package/src/ai_data_scientist/model_tuning.py +104 -0
- package/src/ai_data_scientist/notebook_audit.py +574 -0
- package/src/ai_data_scientist/project_manager.py +243 -0
- package/src/ai_data_scientist/report_export.py +73 -0
- package/src/ai_data_scientist/sensitivity.py +445 -0
- package/src/ai_data_scientist/signal_analysis.py +201 -0
- package/src/ai_data_scientist/skill_packaging.py +40 -0
- package/src/ai_data_scientist/stats_analysis.py +88 -0
- package/src/ai_data_scientist/text_nlp.py +44 -0
- package/src/ai_data_scientist/timeseries.py +68 -0
- package/src/ai_data_scientist/visualization.py +708 -0
- package/src/ai_genomics_scientist/__init__.py +1 -0
- package/src/ai_genomics_scientist/differential_expression.py +147 -0
- package/src/ai_genomics_scientist/dispatch.py +267 -0
- package/src/ai_genomics_scientist/evidence.py +45 -0
- package/src/ai_genomics_scientist/gene_set_enrichment.py +76 -0
- package/src/ai_genomics_scientist/sequence_alignment.py +97 -0
- package/src/ai_genomics_scientist/sequence_features.py +111 -0
- package/src/ai_genomics_scientist/splice_site_scoring.py +66 -0
- package/src/ai_genomics_scientist/validation.py +83 -0
- package/src/ai_genomics_scientist/variant_effect.py +147 -0
- package/src/ai_genomics_scientist/variant_pathogenicity.py +125 -0
- package/src/ai_materials_scientist/__init__.py +0 -0
- package/src/ai_materials_scientist/calphad.py +117 -0
- package/src/ai_materials_scientist/classical_monte_carlo.py +165 -0
- package/src/ai_materials_scientist/crystal_plasticity.py +184 -0
- package/src/ai_materials_scientist/dispatch.py +100 -0
- package/src/ai_materials_scientist/evidence.py +84 -0
- package/src/ai_materials_scientist/fem.py +279 -0
- package/src/ai_materials_scientist/kinetic_monte_carlo.py +145 -0
- package/src/ai_materials_scientist/molecular_dynamics.py +240 -0
- package/src/ai_materials_scientist/phase_field.py +167 -0
- package/src/ai_materials_scientist/validation.py +70 -0
- package/src/ai_scientist/__init__.py +1 -0
- package/src/ai_scientist/completion_gate.py +15 -0
- package/src/ai_scientist/data_analysis.py +46 -0
- package/src/ai_scientist/evidence_registry.py +99 -0
- package/src/ai_scientist/experimental_design.py +20 -0
- package/src/ai_scientist/language.py +14 -0
- package/src/ai_scientist/latex_renderer.py +41 -0
- package/src/ai_scientist/literature_review.py +37 -0
- package/src/ai_scientist/manifest.py +87 -0
- package/src/ai_scientist/manuscript.py +94 -0
- package/src/ai_scientist/mcp_config.py +76 -0
- package/src/ai_scientist/mcp_external.py +42 -0
- package/src/ai_scientist/mcp_failures.py +23 -0
- package/src/ai_scientist/mcp_gateway.py +38 -0
- package/src/ai_scientist/mcp_managed.py +180 -0
- package/src/ai_scientist/npm_packaging.py +49 -0
- package/src/ai_scientist/orchestrator.py +133 -0
- package/src/ai_scientist/peer_review.py +60 -0
- package/src/ai_scientist/phase_gate.py +74 -0
- package/src/ai_scientist/phase_state.py +230 -0
- package/src/ai_scientist/presentation.py +56 -0
- package/src/ai_scientist/project_config.py +31 -0
- package/src/ai_scientist/project_handle.py +74 -0
- package/src/ai_scientist/reproducibility.py +20 -0
- package/src/ai_scientist/research_planning.py +20 -0
- package/src/ai_scientist/skill_invocation.py +21 -0
- package/src/ai_scientist/tdd_gate.py +99 -0
- package/src/ai_structural_biology_scientist/__init__.py +0 -0
- package/src/ai_structural_biology_scientist/contact_map.py +87 -0
- package/src/ai_structural_biology_scientist/dispatch.py +269 -0
- package/src/ai_structural_biology_scientist/evidence.py +43 -0
- package/src/ai_structural_biology_scientist/hydrophobicity.py +101 -0
- package/src/ai_structural_biology_scientist/protein_docking_score.py +104 -0
- package/src/ai_structural_biology_scientist/secondary_structure.py +95 -0
- package/src/ai_structural_biology_scientist/structural_similarity.py +74 -0
- package/src/ai_structural_biology_scientist/validation.py +100 -0
|
@@ -0,0 +1,635 @@
|
|
|
1
|
+
"""A/B testing and experiment evaluation.
|
|
2
|
+
|
|
3
|
+
Implements DES-AIDS-020 (REQ-AIDS-022): computes the statistical
|
|
4
|
+
significance of the observed difference between two groups and reports
|
|
5
|
+
the result with a bilingual markdown interpretation.
|
|
6
|
+
|
|
7
|
+
CHANGE-010 (REQ-AIDS-079..081) adds paired_t/wilcoxon/paired_bootstrap
|
|
8
|
+
paths (DES-AIDS-067..069, CODE-AIDS-101..105).
|
|
9
|
+
CHANGE-016 (REQ-AIDS-090..092) adds repeated multi-seed comparison,
|
|
10
|
+
seed-variability adoption thresholds, and three-way holdout selection-bias
|
|
11
|
+
evaluation (DES-AIDS-090..092, CODE-AIDS-131..140).
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import math
|
|
17
|
+
from collections.abc import Callable, Sequence
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
|
|
20
|
+
import numpy as np
|
|
21
|
+
import pandas as pd
|
|
22
|
+
from scipy import stats as scipy_stats
|
|
23
|
+
|
|
24
|
+
MetricFn = Callable[[pd.Series, pd.Series], float]
|
|
25
|
+
SeedComparisonFn = Callable[[int, int | None], tuple[float, float]]
|
|
26
|
+
SelectionBiasFn = Callable[[pd.DataFrame, pd.DataFrame, pd.DataFrame, str, list[str]], dict]
|
|
27
|
+
|
|
28
|
+
_SUPPORTED_TESTS = ("ttest", "paired_t", "wilcoxon", "paired_bootstrap")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
# @id CODE-AIDS-101
|
|
32
|
+
# @implements REQ-AIDS-081
|
|
33
|
+
# @design DES-AIDS-069
|
|
34
|
+
@dataclass(frozen=True)
|
|
35
|
+
class ExperimentResult:
|
|
36
|
+
statistic: float
|
|
37
|
+
p_value: float
|
|
38
|
+
interpretation: str
|
|
39
|
+
confidence_interval: tuple[float, float] | None = None
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
# @id CODE-AIDS-131
|
|
43
|
+
# @implements REQ-AIDS-090
|
|
44
|
+
# @design DES-AIDS-090
|
|
45
|
+
@dataclass(frozen=True)
|
|
46
|
+
class SeedComparisonResult:
|
|
47
|
+
split_seed: int
|
|
48
|
+
model_seed: int | None
|
|
49
|
+
control_metric: float
|
|
50
|
+
treatment_metric: float
|
|
51
|
+
improvement: float
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
# @id CODE-AIDS-132
|
|
55
|
+
# @implements REQ-AIDS-090
|
|
56
|
+
# @design DES-AIDS-090
|
|
57
|
+
@dataclass(frozen=True)
|
|
58
|
+
class SeedComparisonSummary:
|
|
59
|
+
results: tuple[SeedComparisonResult, ...]
|
|
60
|
+
mean_improvement: float
|
|
61
|
+
seed_variability: float
|
|
62
|
+
sign_counts: dict[str, int]
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# @id CODE-AIDS-133
|
|
66
|
+
# @implements REQ-AIDS-091
|
|
67
|
+
# @design DES-AIDS-091
|
|
68
|
+
@dataclass(frozen=True)
|
|
69
|
+
class AdoptionDecision:
|
|
70
|
+
candidate_improvement: float
|
|
71
|
+
threshold: float
|
|
72
|
+
classification: str
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
# @id CODE-AIDS-134
|
|
76
|
+
# @implements REQ-AIDS-092
|
|
77
|
+
# @design DES-AIDS-092
|
|
78
|
+
@dataclass(frozen=True)
|
|
79
|
+
class SelectionBiasHoldoutResult:
|
|
80
|
+
train_index: tuple
|
|
81
|
+
selection_index: tuple
|
|
82
|
+
evaluation_index: tuple
|
|
83
|
+
selected_candidate: str
|
|
84
|
+
selection_improvement: float
|
|
85
|
+
evaluation_improvement: float
|
|
86
|
+
optimism: float
|
|
87
|
+
assumption_findings: tuple | None = None
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _interpret(p_value: float, language: str) -> str:
|
|
91
|
+
significant = p_value < 0.05
|
|
92
|
+
if language == "ja":
|
|
93
|
+
verdict = "統計的に有意な差があります" if significant else "統計的に有意な差は見られません"
|
|
94
|
+
return f"p値は {p_value:.4g} で、{verdict} (有意水準0.05)。"
|
|
95
|
+
verdict = (
|
|
96
|
+
"a statistically significant difference"
|
|
97
|
+
if significant
|
|
98
|
+
else "no statistically significant difference"
|
|
99
|
+
)
|
|
100
|
+
return f"The p-value is {p_value:.4g}, indicating {verdict} (alpha=0.05)."
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _interpret_bootstrap_interval(
|
|
104
|
+
statistic: float,
|
|
105
|
+
confidence_interval: tuple[float, float],
|
|
106
|
+
confidence_level: float,
|
|
107
|
+
language: str,
|
|
108
|
+
) -> str:
|
|
109
|
+
percent = confidence_level * 100.0
|
|
110
|
+
if language == "ja":
|
|
111
|
+
return (
|
|
112
|
+
f"推定された treatment-control 差は {statistic:.4g} で、"
|
|
113
|
+
f"{percent:.1f}% ブートストラップ信頼区間は "
|
|
114
|
+
f"[{confidence_interval[0]:.4g}, {confidence_interval[1]:.4g}] です。"
|
|
115
|
+
"この経路は区間推定のみを返し、仮説検定のp値は返しません。"
|
|
116
|
+
)
|
|
117
|
+
return (
|
|
118
|
+
f"The estimated treatment-minus-control difference is {statistic:.4g}, "
|
|
119
|
+
f"with a {percent:.1f}% bootstrap confidence interval of "
|
|
120
|
+
f"[{confidence_interval[0]:.4g}, {confidence_interval[1]:.4g}]. "
|
|
121
|
+
"This path reports interval estimation only and does not provide a hypothesis-test "
|
|
122
|
+
"p-value."
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _as_numeric_series(values: pd.Series, *, name: str) -> pd.Series:
|
|
127
|
+
try:
|
|
128
|
+
numeric = pd.to_numeric(values, errors="raise")
|
|
129
|
+
except (TypeError, ValueError) as exc:
|
|
130
|
+
raise ValueError(f"{name} must contain numeric paired values.") from exc
|
|
131
|
+
if not np.isfinite(numeric.to_numpy(dtype=float, copy=False)).all():
|
|
132
|
+
raise ValueError(f"{name} must contain only finite paired values.")
|
|
133
|
+
return numeric
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
# @id CODE-AIDS-102
|
|
137
|
+
# @implements REQ-AIDS-079 REQ-AIDS-080 REQ-AIDS-081
|
|
138
|
+
# @design DES-AIDS-067 DES-AIDS-068
|
|
139
|
+
def _validate_paired_inputs(
|
|
140
|
+
control: pd.Series,
|
|
141
|
+
treatment: pd.Series,
|
|
142
|
+
*,
|
|
143
|
+
y_true: pd.Series | None = None,
|
|
144
|
+
metric_fn: MetricFn | None = None,
|
|
145
|
+
iterations: int = 1000,
|
|
146
|
+
confidence_level: float = 0.95,
|
|
147
|
+
) -> tuple[pd.Series, pd.Series, pd.Series | None]:
|
|
148
|
+
if len(control) != len(treatment):
|
|
149
|
+
raise ValueError("Paired experiment tests require equal-length inputs.")
|
|
150
|
+
if not control.index.equals(treatment.index):
|
|
151
|
+
raise ValueError(
|
|
152
|
+
"Paired experiment tests require control and treatment to share identical indexes."
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
control_series = _as_numeric_series(control, name="control")
|
|
156
|
+
treatment_series = _as_numeric_series(treatment, name="treatment")
|
|
157
|
+
|
|
158
|
+
if (metric_fn is None) != (y_true is None):
|
|
159
|
+
raise ValueError("Paired bootstrap requires metric_fn and y_true together.")
|
|
160
|
+
|
|
161
|
+
truth_series = None
|
|
162
|
+
if y_true is not None:
|
|
163
|
+
if len(y_true) != len(control):
|
|
164
|
+
raise ValueError(
|
|
165
|
+
"Paired bootstrap requires y_true to align with control and treatment."
|
|
166
|
+
)
|
|
167
|
+
if not y_true.index.equals(control.index):
|
|
168
|
+
raise ValueError("Paired bootstrap requires y_true to share the paired input index.")
|
|
169
|
+
truth_series = _as_numeric_series(y_true, name="y_true")
|
|
170
|
+
|
|
171
|
+
if isinstance(iterations, bool) or int(iterations) != iterations or int(iterations) <= 0:
|
|
172
|
+
raise ValueError("Paired bootstrap requires iterations to be a positive integer.")
|
|
173
|
+
if not 0.0 < float(confidence_level) < 1.0:
|
|
174
|
+
raise ValueError("Paired bootstrap requires confidence_level to be between 0 and 1.")
|
|
175
|
+
|
|
176
|
+
return control_series, treatment_series, truth_series
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
# @id CODE-AIDS-022
|
|
180
|
+
# @implements REQ-AIDS-022
|
|
181
|
+
# @design DES-AIDS-020
|
|
182
|
+
def _run_independent_ttest(
|
|
183
|
+
control: pd.Series,
|
|
184
|
+
treatment: pd.Series,
|
|
185
|
+
*,
|
|
186
|
+
language: str,
|
|
187
|
+
) -> ExperimentResult:
|
|
188
|
+
statistic, p_value = scipy_stats.ttest_ind(control, treatment)
|
|
189
|
+
return ExperimentResult(
|
|
190
|
+
statistic=float(statistic),
|
|
191
|
+
p_value=float(p_value),
|
|
192
|
+
interpretation=_interpret(float(p_value), language),
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
# @id CODE-AIDS-103
|
|
197
|
+
# @implements REQ-AIDS-079 REQ-AIDS-080
|
|
198
|
+
# @design DES-AIDS-067 DES-AIDS-069
|
|
199
|
+
def _run_paired_test(
|
|
200
|
+
control: pd.Series,
|
|
201
|
+
treatment: pd.Series,
|
|
202
|
+
*,
|
|
203
|
+
test: str,
|
|
204
|
+
language: str,
|
|
205
|
+
) -> ExperimentResult:
|
|
206
|
+
control_series, treatment_series, _ = _validate_paired_inputs(control, treatment)
|
|
207
|
+
if test == "paired_t":
|
|
208
|
+
statistic, p_value = scipy_stats.ttest_rel(control_series, treatment_series)
|
|
209
|
+
else:
|
|
210
|
+
statistic, p_value = scipy_stats.wilcoxon(control_series, treatment_series)
|
|
211
|
+
return ExperimentResult(
|
|
212
|
+
statistic=float(statistic),
|
|
213
|
+
p_value=float(p_value),
|
|
214
|
+
interpretation=_interpret(float(p_value), language),
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _compute_difference(
|
|
219
|
+
control: pd.Series,
|
|
220
|
+
treatment: pd.Series,
|
|
221
|
+
*,
|
|
222
|
+
y_true: pd.Series | None = None,
|
|
223
|
+
metric_fn: MetricFn | None = None,
|
|
224
|
+
) -> float:
|
|
225
|
+
if metric_fn is None:
|
|
226
|
+
return float((treatment - control).mean())
|
|
227
|
+
assert y_true is not None
|
|
228
|
+
return float(metric_fn(y_true, treatment) - metric_fn(y_true, control))
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _sample_paired_indices(
|
|
232
|
+
sample_size: int,
|
|
233
|
+
rng: np.random.Generator,
|
|
234
|
+
*,
|
|
235
|
+
y_true: pd.Series | None = None,
|
|
236
|
+
) -> np.ndarray:
|
|
237
|
+
if y_true is None:
|
|
238
|
+
return rng.integers(0, sample_size, size=sample_size)
|
|
239
|
+
|
|
240
|
+
unique_counts = y_true.value_counts(sort=False)
|
|
241
|
+
if 1 < len(unique_counts) < sample_size:
|
|
242
|
+
return np.concatenate(
|
|
243
|
+
[
|
|
244
|
+
rng.choice(
|
|
245
|
+
np.flatnonzero(y_true.to_numpy() == label),
|
|
246
|
+
size=int(count),
|
|
247
|
+
replace=True,
|
|
248
|
+
)
|
|
249
|
+
for label, count in unique_counts.items()
|
|
250
|
+
]
|
|
251
|
+
)
|
|
252
|
+
return rng.integers(0, sample_size, size=sample_size)
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
# @id CODE-AIDS-135
|
|
256
|
+
# @implements REQ-AIDS-090 REQ-AIDS-092
|
|
257
|
+
# @design DES-AIDS-090 DES-AIDS-092
|
|
258
|
+
def _as_float_pair(values: tuple[float, float]) -> tuple[float, float]:
|
|
259
|
+
if len(values) != 2:
|
|
260
|
+
raise ValueError("Comparison callbacks must return exactly two metric values.")
|
|
261
|
+
control_metric, treatment_metric = values
|
|
262
|
+
control_metric = float(control_metric)
|
|
263
|
+
treatment_metric = float(treatment_metric)
|
|
264
|
+
if not np.isfinite([control_metric, treatment_metric]).all():
|
|
265
|
+
raise ValueError("Comparison callbacks must return only finite metric values.")
|
|
266
|
+
return control_metric, treatment_metric
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
# @id CODE-AIDS-136
|
|
270
|
+
# @implements REQ-AIDS-090
|
|
271
|
+
# @design DES-AIDS-090
|
|
272
|
+
def summarize_seed_variability(
|
|
273
|
+
compare_fn: SeedComparisonFn,
|
|
274
|
+
*,
|
|
275
|
+
split_seeds: Sequence[int],
|
|
276
|
+
model_seeds: Sequence[int | None] | None = None,
|
|
277
|
+
) -> SeedComparisonSummary:
|
|
278
|
+
"""Repeat one comparison across seeds and summarize observed variability."""
|
|
279
|
+
effective_split_seeds = list(split_seeds)
|
|
280
|
+
if len(effective_split_seeds) == 0:
|
|
281
|
+
raise ValueError("split_seeds must contain at least one seed.")
|
|
282
|
+
|
|
283
|
+
effective_model_seeds = (
|
|
284
|
+
[None] * len(effective_split_seeds) if model_seeds is None else list(model_seeds)
|
|
285
|
+
)
|
|
286
|
+
if len(effective_model_seeds) != len(effective_split_seeds):
|
|
287
|
+
raise ValueError("model_seeds must be omitted or match split_seeds in length.")
|
|
288
|
+
|
|
289
|
+
results: list[SeedComparisonResult] = []
|
|
290
|
+
improvements: list[float] = []
|
|
291
|
+
sign_counts = {"positive": 0, "zero": 0, "negative": 0}
|
|
292
|
+
|
|
293
|
+
for split_seed, model_seed in zip(effective_split_seeds, effective_model_seeds, strict=True):
|
|
294
|
+
control_metric, treatment_metric = _as_float_pair(compare_fn(split_seed, model_seed))
|
|
295
|
+
improvement = float(treatment_metric - control_metric)
|
|
296
|
+
results.append(
|
|
297
|
+
SeedComparisonResult(
|
|
298
|
+
split_seed=int(split_seed),
|
|
299
|
+
model_seed=None if model_seed is None else int(model_seed),
|
|
300
|
+
control_metric=control_metric,
|
|
301
|
+
treatment_metric=treatment_metric,
|
|
302
|
+
improvement=improvement,
|
|
303
|
+
)
|
|
304
|
+
)
|
|
305
|
+
improvements.append(improvement)
|
|
306
|
+
if improvement > 0:
|
|
307
|
+
sign_counts["positive"] += 1
|
|
308
|
+
elif improvement < 0:
|
|
309
|
+
sign_counts["negative"] += 1
|
|
310
|
+
else:
|
|
311
|
+
sign_counts["zero"] += 1
|
|
312
|
+
|
|
313
|
+
return SeedComparisonSummary(
|
|
314
|
+
results=tuple(results),
|
|
315
|
+
mean_improvement=float(np.mean(improvements)),
|
|
316
|
+
seed_variability=float(max(improvements) - min(improvements)),
|
|
317
|
+
sign_counts=sign_counts,
|
|
318
|
+
)
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
# @id CODE-AIDS-137
|
|
322
|
+
# @implements REQ-AIDS-091
|
|
323
|
+
# @design DES-AIDS-091
|
|
324
|
+
def judge_improvement(
|
|
325
|
+
summary: SeedComparisonSummary,
|
|
326
|
+
*,
|
|
327
|
+
candidate_improvement: float | None = None,
|
|
328
|
+
threshold: float | None = None,
|
|
329
|
+
) -> AdoptionDecision:
|
|
330
|
+
"""Classify an improvement against the observed seed-variability threshold."""
|
|
331
|
+
effective_improvement = (
|
|
332
|
+
summary.mean_improvement if candidate_improvement is None else float(candidate_improvement)
|
|
333
|
+
)
|
|
334
|
+
effective_threshold = summary.seed_variability if threshold is None else float(threshold)
|
|
335
|
+
if not np.isfinite(effective_improvement):
|
|
336
|
+
raise ValueError("judge_improvement requires a finite candidate_improvement.")
|
|
337
|
+
if not np.isfinite(effective_threshold):
|
|
338
|
+
raise ValueError("judge_improvement requires a finite threshold.")
|
|
339
|
+
if effective_improvement < 0:
|
|
340
|
+
classification = "regression"
|
|
341
|
+
elif effective_improvement <= effective_threshold:
|
|
342
|
+
classification = "within_seed_variability"
|
|
343
|
+
else:
|
|
344
|
+
classification = "adopt"
|
|
345
|
+
return AdoptionDecision(
|
|
346
|
+
candidate_improvement=effective_improvement,
|
|
347
|
+
threshold=effective_threshold,
|
|
348
|
+
classification=classification,
|
|
349
|
+
)
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
# @id CODE-AIDS-138
|
|
353
|
+
# @implements REQ-AIDS-092
|
|
354
|
+
# @design DES-AIDS-092
|
|
355
|
+
def _split_holdout_partitions(
|
|
356
|
+
df: pd.DataFrame,
|
|
357
|
+
*,
|
|
358
|
+
split_seed: int,
|
|
359
|
+
train_fraction: float,
|
|
360
|
+
selection_fraction: float,
|
|
361
|
+
evaluation_fraction: float,
|
|
362
|
+
) -> tuple[pd.DataFrame, pd.DataFrame, pd.DataFrame]:
|
|
363
|
+
fractions = np.array(
|
|
364
|
+
[float(train_fraction), float(selection_fraction), float(evaluation_fraction)], dtype=float
|
|
365
|
+
)
|
|
366
|
+
if np.any(fractions <= 0.0):
|
|
367
|
+
raise ValueError("train/selection/evaluation fractions must be positive.")
|
|
368
|
+
if not np.isclose(float(fractions.sum()), 1.0):
|
|
369
|
+
raise ValueError("train/selection/evaluation fractions must sum to 1.")
|
|
370
|
+
|
|
371
|
+
total_rows = len(df)
|
|
372
|
+
if total_rows < 3:
|
|
373
|
+
raise ValueError("Selection-bias holdout evaluation requires at least 3 rows.")
|
|
374
|
+
|
|
375
|
+
raw_counts = fractions * total_rows
|
|
376
|
+
counts = np.floor(raw_counts).astype(int)
|
|
377
|
+
remainder = total_rows - int(counts.sum())
|
|
378
|
+
if remainder > 0:
|
|
379
|
+
residual_order = np.argsort(-(raw_counts - counts))
|
|
380
|
+
for position in residual_order[:remainder]:
|
|
381
|
+
counts[position] += 1
|
|
382
|
+
while np.any(counts == 0):
|
|
383
|
+
zero_positions = np.flatnonzero(counts == 0)
|
|
384
|
+
donor_candidates = np.flatnonzero(counts > 1)
|
|
385
|
+
if len(donor_candidates) == 0:
|
|
386
|
+
raise ValueError(
|
|
387
|
+
"train/selection/evaluation fractions must yield non-empty partitions."
|
|
388
|
+
)
|
|
389
|
+
donor_position = int(donor_candidates[np.argmax(counts[donor_candidates])])
|
|
390
|
+
counts[donor_position] -= 1
|
|
391
|
+
counts[int(zero_positions[0])] += 1
|
|
392
|
+
|
|
393
|
+
rng = np.random.default_rng(split_seed)
|
|
394
|
+
shuffled_positions = rng.permutation(total_rows)
|
|
395
|
+
train_end = int(counts[0])
|
|
396
|
+
selection_end = int(counts[0] + counts[1])
|
|
397
|
+
train_positions = shuffled_positions[:train_end]
|
|
398
|
+
selection_positions = shuffled_positions[train_end:selection_end]
|
|
399
|
+
evaluation_positions = shuffled_positions[selection_end:]
|
|
400
|
+
|
|
401
|
+
return (
|
|
402
|
+
df.iloc[train_positions].copy(),
|
|
403
|
+
df.iloc[selection_positions].copy(),
|
|
404
|
+
df.iloc[evaluation_positions].copy(),
|
|
405
|
+
)
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
# @id CODE-AIDS-139
|
|
409
|
+
# @implements REQ-AIDS-092
|
|
410
|
+
# @design DES-AIDS-092
|
|
411
|
+
def _extract_selection_bias_metrics(
|
|
412
|
+
evaluation_payload: dict,
|
|
413
|
+
*,
|
|
414
|
+
baseline: str,
|
|
415
|
+
) -> tuple[dict[str, float], dict[str, float]]:
|
|
416
|
+
try:
|
|
417
|
+
selection_metrics = evaluation_payload["selection_metrics"]
|
|
418
|
+
evaluation_metrics = evaluation_payload["evaluation_metrics"]
|
|
419
|
+
except KeyError as exc:
|
|
420
|
+
raise ValueError(
|
|
421
|
+
"evaluate_candidate_fn must return selection_metrics and evaluation_metrics."
|
|
422
|
+
) from exc
|
|
423
|
+
normalized_selection = {str(name): float(value) for name, value in selection_metrics.items()}
|
|
424
|
+
normalized_evaluation = {str(name): float(value) for name, value in evaluation_metrics.items()}
|
|
425
|
+
for metric_name, metric_map in (
|
|
426
|
+
("selection_metrics", normalized_selection),
|
|
427
|
+
("evaluation_metrics", normalized_evaluation),
|
|
428
|
+
):
|
|
429
|
+
if baseline not in metric_map:
|
|
430
|
+
raise ValueError(f"evaluate_candidate_fn must include baseline in {metric_name}.")
|
|
431
|
+
if not np.isfinite(list(metric_map.values())).all():
|
|
432
|
+
raise ValueError(
|
|
433
|
+
f"evaluate_candidate_fn must return only finite values in {metric_name}."
|
|
434
|
+
)
|
|
435
|
+
return normalized_selection, normalized_evaluation
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
def _choose_selection_winner(
|
|
439
|
+
selection_metrics: dict[str, float],
|
|
440
|
+
*,
|
|
441
|
+
baseline: str,
|
|
442
|
+
candidates: list[str],
|
|
443
|
+
) -> str:
|
|
444
|
+
missing_candidates = [
|
|
445
|
+
candidate for candidate in candidates if candidate not in selection_metrics
|
|
446
|
+
]
|
|
447
|
+
if missing_candidates:
|
|
448
|
+
raise ValueError(
|
|
449
|
+
"evaluate_candidate_fn must include every candidate in selection_metrics: "
|
|
450
|
+
+ ", ".join(missing_candidates)
|
|
451
|
+
)
|
|
452
|
+
return max(
|
|
453
|
+
candidates,
|
|
454
|
+
key=lambda candidate: (selection_metrics[candidate], -candidates.index(candidate)),
|
|
455
|
+
)
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
# @id CODE-AIDS-140
|
|
459
|
+
# @implements REQ-AIDS-092
|
|
460
|
+
# @design DES-AIDS-092
|
|
461
|
+
def evaluate_selection_bias_holdout(
|
|
462
|
+
df: pd.DataFrame,
|
|
463
|
+
target: str,
|
|
464
|
+
*,
|
|
465
|
+
baseline: str,
|
|
466
|
+
candidates: list[str],
|
|
467
|
+
evaluate_candidate_fn: SelectionBiasFn,
|
|
468
|
+
split_seed: int = 42,
|
|
469
|
+
train_fraction: float = 0.6,
|
|
470
|
+
selection_fraction: float = 0.2,
|
|
471
|
+
evaluation_fraction: float = 0.2,
|
|
472
|
+
assumption_manifest: object | None = None,
|
|
473
|
+
) -> SelectionBiasHoldoutResult:
|
|
474
|
+
"""Evaluate selection optimism using disjoint train/selection/evaluation subsets."""
|
|
475
|
+
if target not in df.columns:
|
|
476
|
+
raise ValueError(f"target column {target!r} is not present in the dataframe.")
|
|
477
|
+
train_df, selection_df, evaluation_df = _split_holdout_partitions(
|
|
478
|
+
df,
|
|
479
|
+
split_seed=split_seed,
|
|
480
|
+
train_fraction=train_fraction,
|
|
481
|
+
selection_fraction=selection_fraction,
|
|
482
|
+
evaluation_fraction=evaluation_fraction,
|
|
483
|
+
)
|
|
484
|
+
|
|
485
|
+
evaluation_payload = evaluate_candidate_fn(
|
|
486
|
+
train_df,
|
|
487
|
+
selection_df,
|
|
488
|
+
evaluation_df,
|
|
489
|
+
baseline,
|
|
490
|
+
candidates,
|
|
491
|
+
)
|
|
492
|
+
selection_metrics, evaluation_metrics = _extract_selection_bias_metrics(
|
|
493
|
+
evaluation_payload,
|
|
494
|
+
baseline=baseline,
|
|
495
|
+
)
|
|
496
|
+
selected_candidate = _choose_selection_winner(
|
|
497
|
+
selection_metrics,
|
|
498
|
+
baseline=baseline,
|
|
499
|
+
candidates=candidates,
|
|
500
|
+
)
|
|
501
|
+
missing_evaluation_candidates = [
|
|
502
|
+
candidate for candidate in candidates if candidate not in evaluation_metrics
|
|
503
|
+
]
|
|
504
|
+
if missing_evaluation_candidates:
|
|
505
|
+
raise ValueError(
|
|
506
|
+
"evaluate_candidate_fn must include every candidate in evaluation_metrics: "
|
|
507
|
+
+ ", ".join(missing_evaluation_candidates)
|
|
508
|
+
)
|
|
509
|
+
selection_improvement = float(selection_metrics[selected_candidate]) - float(
|
|
510
|
+
selection_metrics[baseline]
|
|
511
|
+
)
|
|
512
|
+
evaluation_improvement = float(evaluation_metrics[selected_candidate]) - float(
|
|
513
|
+
evaluation_metrics[baseline]
|
|
514
|
+
)
|
|
515
|
+
|
|
516
|
+
assumption_findings = None
|
|
517
|
+
if assumption_manifest is not None:
|
|
518
|
+
from ai_data_scientist.analysis_assumptions import check_manifest
|
|
519
|
+
|
|
520
|
+
assumption_findings = check_manifest(assumption_manifest)
|
|
521
|
+
|
|
522
|
+
return SelectionBiasHoldoutResult(
|
|
523
|
+
train_index=tuple(train_df.index),
|
|
524
|
+
selection_index=tuple(selection_df.index),
|
|
525
|
+
evaluation_index=tuple(evaluation_df.index),
|
|
526
|
+
selected_candidate=selected_candidate,
|
|
527
|
+
selection_improvement=selection_improvement,
|
|
528
|
+
evaluation_improvement=evaluation_improvement,
|
|
529
|
+
optimism=float(selection_improvement - evaluation_improvement),
|
|
530
|
+
assumption_findings=assumption_findings,
|
|
531
|
+
)
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
# @id CODE-AIDS-104
|
|
535
|
+
# @implements REQ-AIDS-081
|
|
536
|
+
# @design DES-AIDS-068 DES-AIDS-069
|
|
537
|
+
def _run_paired_bootstrap(
|
|
538
|
+
control: pd.Series,
|
|
539
|
+
treatment: pd.Series,
|
|
540
|
+
*,
|
|
541
|
+
language: str,
|
|
542
|
+
y_true: pd.Series | None = None,
|
|
543
|
+
metric_fn: MetricFn | None = None,
|
|
544
|
+
iterations: int = 1000,
|
|
545
|
+
confidence_level: float = 0.95,
|
|
546
|
+
random_state: int | None = None,
|
|
547
|
+
) -> ExperimentResult:
|
|
548
|
+
control_series, treatment_series, truth_series = _validate_paired_inputs(
|
|
549
|
+
control,
|
|
550
|
+
treatment,
|
|
551
|
+
y_true=y_true,
|
|
552
|
+
metric_fn=metric_fn,
|
|
553
|
+
iterations=iterations,
|
|
554
|
+
confidence_level=confidence_level,
|
|
555
|
+
)
|
|
556
|
+
|
|
557
|
+
observed_difference = _compute_difference(
|
|
558
|
+
control_series,
|
|
559
|
+
treatment_series,
|
|
560
|
+
y_true=truth_series,
|
|
561
|
+
metric_fn=metric_fn,
|
|
562
|
+
)
|
|
563
|
+
|
|
564
|
+
rng = np.random.default_rng(random_state)
|
|
565
|
+
sample_size = len(control_series)
|
|
566
|
+
bootstrap_differences = np.empty(int(iterations), dtype=float)
|
|
567
|
+
|
|
568
|
+
for index in range(int(iterations)):
|
|
569
|
+
sample_indices = _sample_paired_indices(sample_size, rng, y_true=truth_series)
|
|
570
|
+
control_sample = control_series.iloc[sample_indices].reset_index(drop=True)
|
|
571
|
+
treatment_sample = treatment_series.iloc[sample_indices].reset_index(drop=True)
|
|
572
|
+
truth_sample = (
|
|
573
|
+
None
|
|
574
|
+
if truth_series is None
|
|
575
|
+
else truth_series.iloc[sample_indices].reset_index(drop=True)
|
|
576
|
+
)
|
|
577
|
+
bootstrap_differences[index] = _compute_difference(
|
|
578
|
+
control_sample,
|
|
579
|
+
treatment_sample,
|
|
580
|
+
y_true=truth_sample,
|
|
581
|
+
metric_fn=metric_fn,
|
|
582
|
+
)
|
|
583
|
+
|
|
584
|
+
alpha = 1.0 - float(confidence_level)
|
|
585
|
+
confidence_interval = (
|
|
586
|
+
float(np.quantile(bootstrap_differences, alpha / 2.0)),
|
|
587
|
+
float(np.quantile(bootstrap_differences, 1.0 - (alpha / 2.0))),
|
|
588
|
+
)
|
|
589
|
+
|
|
590
|
+
return ExperimentResult(
|
|
591
|
+
statistic=observed_difference,
|
|
592
|
+
p_value=math.nan,
|
|
593
|
+
interpretation=_interpret_bootstrap_interval(
|
|
594
|
+
observed_difference,
|
|
595
|
+
confidence_interval,
|
|
596
|
+
float(confidence_level),
|
|
597
|
+
language,
|
|
598
|
+
),
|
|
599
|
+
confidence_interval=confidence_interval,
|
|
600
|
+
)
|
|
601
|
+
|
|
602
|
+
|
|
603
|
+
# @id CODE-AIDS-105
|
|
604
|
+
# @implements REQ-AIDS-079 REQ-AIDS-080 REQ-AIDS-081
|
|
605
|
+
# @design DES-AIDS-067 DES-AIDS-068 DES-AIDS-069
|
|
606
|
+
def evaluate_experiment(
|
|
607
|
+
control: pd.Series,
|
|
608
|
+
treatment: pd.Series,
|
|
609
|
+
test: str = "ttest",
|
|
610
|
+
language: str = "en",
|
|
611
|
+
*,
|
|
612
|
+
y_true: pd.Series | None = None,
|
|
613
|
+
metric_fn: MetricFn | None = None,
|
|
614
|
+
iterations: int = 1000,
|
|
615
|
+
confidence_level: float = 0.95,
|
|
616
|
+
random_state: int | None = None,
|
|
617
|
+
) -> ExperimentResult:
|
|
618
|
+
"""Compute the significance of the difference between ``control`` and ``treatment``."""
|
|
619
|
+
if test not in _SUPPORTED_TESTS:
|
|
620
|
+
raise ValueError(f"Unsupported experiment test: {test!r}")
|
|
621
|
+
|
|
622
|
+
if test == "ttest":
|
|
623
|
+
return _run_independent_ttest(control, treatment, language=language)
|
|
624
|
+
if test in {"paired_t", "wilcoxon"}:
|
|
625
|
+
return _run_paired_test(control, treatment, test=test, language=language)
|
|
626
|
+
return _run_paired_bootstrap(
|
|
627
|
+
control,
|
|
628
|
+
treatment,
|
|
629
|
+
language=language,
|
|
630
|
+
y_true=y_true,
|
|
631
|
+
metric_fn=metric_fn,
|
|
632
|
+
iterations=iterations,
|
|
633
|
+
confidence_level=confidence_level,
|
|
634
|
+
random_state=random_state,
|
|
635
|
+
)
|