jupytermind 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (193) hide show
  1. package/.github/skills/ai-chemistry-scientist/SKILL.md +97 -0
  2. package/.github/skills/ai-chemistry-scientist/manifest.json +156 -0
  3. package/.github/skills/ai-data-scientist/SKILL.md +330 -0
  4. package/.github/skills/ai-genomics-scientist/SKILL.md +98 -0
  5. package/.github/skills/ai-genomics-scientist/manifest.json +93 -0
  6. package/.github/skills/ai-materials-scientist/SKILL.md +51 -0
  7. package/.github/skills/ai-materials-scientist/manifest.json +58 -0
  8. package/.github/skills/ai-scientist/SKILL.md +69 -0
  9. package/.github/skills/ai-scientist/manifest.json +61 -0
  10. package/.github/skills/ai-structural-biology-scientist/SKILL.md +67 -0
  11. package/.github/skills/ai-structural-biology-scientist/manifest.json +72 -0
  12. package/.github/skills/japanese-prose/NOTICE.md +17 -0
  13. package/.github/skills/japanese-prose/SKILL.md +111 -0
  14. package/.github/skills/japanese-prose/references/review-workflow.md +50 -0
  15. package/.github/skills/japanese-prose/references/scoring.md +24 -0
  16. package/.github/skills/japanese-prose/references/writing-guidelines.md +60 -0
  17. package/.github/skills/japanese-prose/scripts/core.py +192 -0
  18. package/.github/skills/japanese-prose/scripts/fixtures/natural.md +5 -0
  19. package/.github/skills/japanese-prose/scripts/fixtures/unnatural.md +5 -0
  20. package/.github/skills/japanese-prose/scripts/lint.py +378 -0
  21. package/.github/skills/japanese-prose/scripts/outline.py +68 -0
  22. package/.github/skills/japanese-prose/scripts/terms.py +112 -0
  23. package/.github/skills/japanese-prose/scripts/test_engine.py +117 -0
  24. package/.github/skills/presentation-planner/SKILL.md +257 -0
  25. package/.github/skills/presentation-planner/assets/design-templates/data-report.yaml +97 -0
  26. package/.github/skills/presentation-planner/assets/design-templates/executive-proposal.yaml +92 -0
  27. package/.github/skills/presentation-planner/assets/design-templates/technical-briefing.yaml +96 -0
  28. package/.github/skills/presentation-planner/assets/scenario-templates/data-report.md +47 -0
  29. package/.github/skills/presentation-planner/assets/scenario-templates/executive-decision.md +43 -0
  30. package/.github/skills/presentation-planner/assets/scenario-templates/technical-briefing.md +45 -0
  31. package/.github/skills/presentation-planner/references/customizing-design-templates.md +160 -0
  32. package/.github/skills/presentation-planner/references/design-spec-schema.md +72 -0
  33. package/.github/skills/presentation-planner/references/handoff-contract.md +49 -0
  34. package/.github/skills/presentation-planner/references/responsibility-boundary.md +32 -0
  35. package/.github/skills/presentation-planner/references/scenario-templates.md +55 -0
  36. package/.github/skills/tech-writer/SKILL.md +434 -0
  37. package/.github/skills/tech-writer/assets/templates/blueprint.md +187 -0
  38. package/.github/skills/tech-writer/assets/templates/design-doc.md +29 -0
  39. package/.github/skills/tech-writer/assets/templates/migration-plan.md +173 -0
  40. package/.github/skills/tech-writer/assets/templates/operations-runbook.md +202 -0
  41. package/.github/skills/tech-writer/assets/templates/pr-description.md +23 -0
  42. package/.github/skills/tech-writer/assets/templates/qiita.md +44 -0
  43. package/.github/skills/tech-writer/assets/templates/readme.md +38 -0
  44. package/.github/skills/tech-writer/assets/templates/requirements-definition.md +170 -0
  45. package/.github/skills/tech-writer/assets/templates/rfi.md +113 -0
  46. package/.github/skills/tech-writer/assets/templates/rfp.md +180 -0
  47. package/.github/skills/tech-writer/assets/templates/security-design.md +167 -0
  48. package/.github/skills/tech-writer/assets/templates/system-design.md +220 -0
  49. package/.github/skills/tech-writer/assets/templates/technical-proposal.md +112 -0
  50. package/.github/skills/tech-writer/assets/templates/test-plan.md +153 -0
  51. package/.github/skills/tech-writer/assets/templates/user-manual.md +22 -0
  52. package/.github/skills/tech-writer/assets/templates/white-paper.md +192 -0
  53. package/.github/skills/tech-writer/references/doctypes/api-docs.md +33 -0
  54. package/.github/skills/tech-writer/references/doctypes/blueprint.md +81 -0
  55. package/.github/skills/tech-writer/references/doctypes/code-comments.md +39 -0
  56. package/.github/skills/tech-writer/references/doctypes/design-doc.md +42 -0
  57. package/.github/skills/tech-writer/references/doctypes/migration-plan.md +63 -0
  58. package/.github/skills/tech-writer/references/doctypes/operations-runbook.md +63 -0
  59. package/.github/skills/tech-writer/references/doctypes/pr-commit.md +82 -0
  60. package/.github/skills/tech-writer/references/doctypes/qiita.md +75 -0
  61. package/.github/skills/tech-writer/references/doctypes/readme.md +43 -0
  62. package/.github/skills/tech-writer/references/doctypes/release-notes.md +30 -0
  63. package/.github/skills/tech-writer/references/doctypes/requirements-definition.md +61 -0
  64. package/.github/skills/tech-writer/references/doctypes/rfi.md +43 -0
  65. package/.github/skills/tech-writer/references/doctypes/rfp.md +46 -0
  66. package/.github/skills/tech-writer/references/doctypes/security-design.md +71 -0
  67. package/.github/skills/tech-writer/references/doctypes/system-design.md +74 -0
  68. package/.github/skills/tech-writer/references/doctypes/technical-proposal.md +49 -0
  69. package/.github/skills/tech-writer/references/doctypes/test-plan.md +67 -0
  70. package/.github/skills/tech-writer/references/doctypes/user-manual.md +58 -0
  71. package/.github/skills/tech-writer/references/doctypes/white-paper.md +84 -0
  72. package/.github/skills/tech-writer/references/doctypes/zenn.md +66 -0
  73. package/.github/skills/tech-writer/references/japanese-prose-optimization.md +110 -0
  74. package/.github/skills/tech-writer/references/style-constitution.md +104 -0
  75. package/.github/skills/tech-writer/scripts/lint.py +412 -0
  76. package/LICENSE +21 -0
  77. package/README.md +92 -0
  78. package/bin/ai-data-scientist.js +123 -0
  79. package/package.json +41 -0
  80. package/pyproject.toml +45 -0
  81. package/src/ai_chemistry_scientist/__init__.py +0 -0
  82. package/src/ai_chemistry_scientist/admet_prediction.py +71 -0
  83. package/src/ai_chemistry_scientist/bioactivity_classification.py +73 -0
  84. package/src/ai_chemistry_scientist/data/sample_molecules.csv +21 -0
  85. package/src/ai_chemistry_scientist/dispatch.py +369 -0
  86. package/src/ai_chemistry_scientist/docking_score.py +97 -0
  87. package/src/ai_chemistry_scientist/drug_likeness_rules.py +84 -0
  88. package/src/ai_chemistry_scientist/evidence.py +41 -0
  89. package/src/ai_chemistry_scientist/molecular_descriptors.py +97 -0
  90. package/src/ai_chemistry_scientist/molecular_formula_mass.py +40 -0
  91. package/src/ai_chemistry_scientist/molecular_similarity.py +78 -0
  92. package/src/ai_chemistry_scientist/qsar_modeling.py +105 -0
  93. package/src/ai_chemistry_scientist/salt_standardization.py +81 -0
  94. package/src/ai_chemistry_scientist/structural_alerts.py +76 -0
  95. package/src/ai_chemistry_scientist/structure_format_conversion.py +84 -0
  96. package/src/ai_chemistry_scientist/validation.py +70 -0
  97. package/src/ai_data_scientist/__init__.py +0 -0
  98. package/src/ai_data_scientist/analysis_assumptions.py +121 -0
  99. package/src/ai_data_scientist/anomaly_detection.py +39 -0
  100. package/src/ai_data_scientist/automl.py +109 -0
  101. package/src/ai_data_scientist/cleaning.py +56 -0
  102. package/src/ai_data_scientist/cli.py +90 -0
  103. package/src/ai_data_scientist/clustering.py +54 -0
  104. package/src/ai_data_scientist/dashboard.py +33 -0
  105. package/src/ai_data_scientist/data_definition.py +100 -0
  106. package/src/ai_data_scientist/data_quality.py +164 -0
  107. package/src/ai_data_scientist/dataset_validation.py +135 -0
  108. package/src/ai_data_scientist/dependency_pins.py +60 -0
  109. package/src/ai_data_scientist/eda.py +82 -0
  110. package/src/ai_data_scientist/experiment_evaluation.py +635 -0
  111. package/src/ai_data_scientist/explainability.py +340 -0
  112. package/src/ai_data_scientist/feature_engineering.py +163 -0
  113. package/src/ai_data_scientist/gate_config.py +32 -0
  114. package/src/ai_data_scientist/ingestion.py +127 -0
  115. package/src/ai_data_scientist/insight_engine.py +180 -0
  116. package/src/ai_data_scientist/japanese_nlp.py +43 -0
  117. package/src/ai_data_scientist/jupyter_launcher.py +137 -0
  118. package/src/ai_data_scientist/jupyter_mcp_client.py +94 -0
  119. package/src/ai_data_scientist/language_router.py +28 -0
  120. package/src/ai_data_scientist/lifecycle.py +221 -0
  121. package/src/ai_data_scientist/mcp_gateway.py +113 -0
  122. package/src/ai_data_scientist/mcp_runtime.py +194 -0
  123. package/src/ai_data_scientist/mcp_transport.py +53 -0
  124. package/src/ai_data_scientist/ml_modeling.py +451 -0
  125. package/src/ai_data_scientist/model_tuning.py +104 -0
  126. package/src/ai_data_scientist/notebook_audit.py +574 -0
  127. package/src/ai_data_scientist/project_manager.py +243 -0
  128. package/src/ai_data_scientist/report_export.py +73 -0
  129. package/src/ai_data_scientist/sensitivity.py +445 -0
  130. package/src/ai_data_scientist/signal_analysis.py +201 -0
  131. package/src/ai_data_scientist/skill_packaging.py +40 -0
  132. package/src/ai_data_scientist/stats_analysis.py +88 -0
  133. package/src/ai_data_scientist/text_nlp.py +44 -0
  134. package/src/ai_data_scientist/timeseries.py +68 -0
  135. package/src/ai_data_scientist/visualization.py +708 -0
  136. package/src/ai_genomics_scientist/__init__.py +1 -0
  137. package/src/ai_genomics_scientist/differential_expression.py +147 -0
  138. package/src/ai_genomics_scientist/dispatch.py +267 -0
  139. package/src/ai_genomics_scientist/evidence.py +45 -0
  140. package/src/ai_genomics_scientist/gene_set_enrichment.py +76 -0
  141. package/src/ai_genomics_scientist/sequence_alignment.py +97 -0
  142. package/src/ai_genomics_scientist/sequence_features.py +111 -0
  143. package/src/ai_genomics_scientist/splice_site_scoring.py +66 -0
  144. package/src/ai_genomics_scientist/validation.py +83 -0
  145. package/src/ai_genomics_scientist/variant_effect.py +147 -0
  146. package/src/ai_genomics_scientist/variant_pathogenicity.py +125 -0
  147. package/src/ai_materials_scientist/__init__.py +0 -0
  148. package/src/ai_materials_scientist/calphad.py +117 -0
  149. package/src/ai_materials_scientist/classical_monte_carlo.py +165 -0
  150. package/src/ai_materials_scientist/crystal_plasticity.py +184 -0
  151. package/src/ai_materials_scientist/dispatch.py +100 -0
  152. package/src/ai_materials_scientist/evidence.py +84 -0
  153. package/src/ai_materials_scientist/fem.py +279 -0
  154. package/src/ai_materials_scientist/kinetic_monte_carlo.py +145 -0
  155. package/src/ai_materials_scientist/molecular_dynamics.py +240 -0
  156. package/src/ai_materials_scientist/phase_field.py +167 -0
  157. package/src/ai_materials_scientist/validation.py +70 -0
  158. package/src/ai_scientist/__init__.py +1 -0
  159. package/src/ai_scientist/completion_gate.py +15 -0
  160. package/src/ai_scientist/data_analysis.py +46 -0
  161. package/src/ai_scientist/evidence_registry.py +99 -0
  162. package/src/ai_scientist/experimental_design.py +20 -0
  163. package/src/ai_scientist/language.py +14 -0
  164. package/src/ai_scientist/latex_renderer.py +41 -0
  165. package/src/ai_scientist/literature_review.py +37 -0
  166. package/src/ai_scientist/manifest.py +87 -0
  167. package/src/ai_scientist/manuscript.py +94 -0
  168. package/src/ai_scientist/mcp_config.py +76 -0
  169. package/src/ai_scientist/mcp_external.py +42 -0
  170. package/src/ai_scientist/mcp_failures.py +23 -0
  171. package/src/ai_scientist/mcp_gateway.py +38 -0
  172. package/src/ai_scientist/mcp_managed.py +180 -0
  173. package/src/ai_scientist/npm_packaging.py +49 -0
  174. package/src/ai_scientist/orchestrator.py +133 -0
  175. package/src/ai_scientist/peer_review.py +60 -0
  176. package/src/ai_scientist/phase_gate.py +74 -0
  177. package/src/ai_scientist/phase_state.py +230 -0
  178. package/src/ai_scientist/presentation.py +56 -0
  179. package/src/ai_scientist/project_config.py +31 -0
  180. package/src/ai_scientist/project_handle.py +74 -0
  181. package/src/ai_scientist/reproducibility.py +20 -0
  182. package/src/ai_scientist/research_planning.py +20 -0
  183. package/src/ai_scientist/skill_invocation.py +21 -0
  184. package/src/ai_scientist/tdd_gate.py +99 -0
  185. package/src/ai_structural_biology_scientist/__init__.py +0 -0
  186. package/src/ai_structural_biology_scientist/contact_map.py +87 -0
  187. package/src/ai_structural_biology_scientist/dispatch.py +269 -0
  188. package/src/ai_structural_biology_scientist/evidence.py +43 -0
  189. package/src/ai_structural_biology_scientist/hydrophobicity.py +101 -0
  190. package/src/ai_structural_biology_scientist/protein_docking_score.py +104 -0
  191. package/src/ai_structural_biology_scientist/secondary_structure.py +95 -0
  192. package/src/ai_structural_biology_scientist/structural_similarity.py +74 -0
  193. package/src/ai_structural_biology_scientist/validation.py +100 -0
@@ -0,0 +1,340 @@
1
+ """Model explainability.
2
+
3
+ Implements DES-AIDS-019 plus CHANGE-011's DES-AIDS-070/071/072: preserves the
4
+ legacy feature-importance ranking by default, and adds opt-in signed local
5
+ contributions and permutation importance. CODE-AIDS-106 through CODE-AIDS-110
6
+ cover importance-kind labeling, the signed-contribution provider chain
7
+ (native pred_contrib / SHAP / linear fallback), and permutation importance.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from dataclasses import dataclass
13
+ from importlib import import_module
14
+ from typing import Any, Literal
15
+
16
+ import numpy as np
17
+ import pandas as pd
18
+ from sklearn.inspection import permutation_importance
19
+
20
+
21
+ @dataclass(frozen=True)
22
+ class ExplainabilityResult:
23
+ feature_importances: dict[str, float]
24
+ ranking: list[str]
25
+ importance_kind: str = "unknown"
26
+ contribution_kind: str | None = None
27
+ signed_contributions: list[dict[str, float]] | None = None
28
+ baseline_values: list[float] | None = None
29
+ raw_predictions: list[float] | None = None
30
+ additivity_check: dict[str, float | bool | None] | None = None
31
+ scoring: str | None = None
32
+
33
+
34
+ def _as_feature_frame(x: Any, feature_names: list[str]) -> pd.DataFrame:
35
+ if x is None:
36
+ raise ValueError("x is required for this explainability method.")
37
+ if isinstance(x, pd.DataFrame):
38
+ return x.loc[:, feature_names].copy()
39
+
40
+ values = np.asarray(x)
41
+ if values.ndim != 2 or values.shape[1] != len(feature_names):
42
+ raise ValueError("x must be a 2D array-like with one column per feature name.")
43
+ return pd.DataFrame(values, columns=feature_names)
44
+
45
+
46
+ def _build_ranking(feature_importances: dict[str, float]) -> list[str]:
47
+ return sorted(feature_importances, key=feature_importances.get, reverse=True)
48
+
49
+
50
+ def _build_feature_importance_map(
51
+ feature_names: list[str], raw_importances: np.ndarray
52
+ ) -> dict[str, float]:
53
+ return {name: float(value) for name, value in zip(feature_names, np.asarray(raw_importances))}
54
+
55
+
56
+ # @id CODE-AIDS-128
57
+ # @implements REQ-AIDS-082
58
+ # @design DES-AIDS-070
59
+ def _normalize_coefficient_importances(coef: Any) -> np.ndarray:
60
+ raw_importances = np.abs(np.asarray(coef, dtype=float))
61
+ if raw_importances.ndim == 2 and raw_importances.shape[0] > 1:
62
+ return raw_importances.mean(axis=0)
63
+ return raw_importances.reshape(-1)
64
+
65
+
66
+ # @id CODE-AIDS-107
67
+ # @implements REQ-AIDS-082
68
+ # @design DES-AIDS-070
69
+ def _compute_default_importance(
70
+ model: object, feature_names: list[str]
71
+ ) -> tuple[dict[str, float], list[str], str]:
72
+ """Return the legacy global-importance ranking and its explicit kind."""
73
+ if hasattr(model, "feature_importances_"):
74
+ raw_importances = np.asarray(model.feature_importances_, dtype=float)
75
+ importance_kind = "split"
76
+ elif hasattr(model, "coef_"):
77
+ raw_importances = _normalize_coefficient_importances(model.coef_)
78
+ importance_kind = "coefficient_magnitude"
79
+ else:
80
+ raise ValueError("Model exposes neither feature_importances_ nor coef_.")
81
+
82
+ feature_importances = _build_feature_importance_map(feature_names, raw_importances)
83
+ return feature_importances, _build_ranking(feature_importances), importance_kind
84
+
85
+
86
+ # @id CODE-AIDS-110
87
+ # @implements REQ-AIDS-083
88
+ # @design DES-AIDS-071
89
+ def _predict_raw_output(model: object, frame: pd.DataFrame) -> np.ndarray | None:
90
+ """Best-effort raw-output prediction for additivity checks."""
91
+ if hasattr(model, "decision_function"):
92
+ raw = np.asarray(model.decision_function(frame), dtype=float)
93
+ return raw.reshape(-1)
94
+
95
+ for kwargs in ({"raw_score": True}, {"output_margin": True}):
96
+ try:
97
+ raw = np.asarray(model.predict(frame, **kwargs), dtype=float)
98
+ except (TypeError, ValueError):
99
+ continue
100
+ if raw.ndim == 1:
101
+ return raw
102
+ if raw.ndim == 2 and raw.shape[1] == 1:
103
+ return raw.reshape(-1)
104
+
105
+ if hasattr(model, "predict") and not hasattr(model, "predict_proba"):
106
+ try:
107
+ raw = np.asarray(model.predict(frame), dtype=float)
108
+ except (TypeError, ValueError):
109
+ return None
110
+ if raw.ndim == 1:
111
+ return raw
112
+ if raw.ndim == 2 and raw.shape[1] == 1:
113
+ return raw.reshape(-1)
114
+
115
+ return None
116
+
117
+
118
+ def _finalize_signed_result(
119
+ feature_names: list[str],
120
+ contribution_kind: str,
121
+ contributions: np.ndarray,
122
+ baseline_values: np.ndarray,
123
+ raw_predictions: np.ndarray | None,
124
+ ) -> ExplainabilityResult:
125
+ contributions = np.asarray(contributions, dtype=float)
126
+ baseline_values = np.asarray(baseline_values, dtype=float).reshape(-1)
127
+ reconstructed = baseline_values + contributions.sum(axis=1)
128
+
129
+ if raw_predictions is None:
130
+ raw_prediction_values = None
131
+ checked_against_model_output = False
132
+ max_abs_error = None
133
+ passed = None
134
+ else:
135
+ raw_predictions = np.asarray(raw_predictions, dtype=float).reshape(-1)
136
+ raw_prediction_values = raw_predictions.astype(float).tolist()
137
+ checked_against_model_output = True
138
+ max_abs_error = float(np.max(np.abs(reconstructed - raw_predictions)))
139
+ passed = bool(max_abs_error <= 1e-6)
140
+ signed_contributions = [
141
+ {name: float(value) for name, value in zip(feature_names, row)} for row in contributions
142
+ ]
143
+ feature_importances = _build_feature_importance_map(
144
+ feature_names, np.mean(np.abs(contributions), axis=0)
145
+ )
146
+
147
+ return ExplainabilityResult(
148
+ feature_importances=feature_importances,
149
+ ranking=_build_ranking(feature_importances),
150
+ importance_kind="mean_absolute_signed_contribution",
151
+ contribution_kind=contribution_kind,
152
+ signed_contributions=signed_contributions,
153
+ baseline_values=baseline_values.astype(float).tolist(),
154
+ raw_predictions=raw_prediction_values,
155
+ additivity_check={
156
+ "passed": passed,
157
+ "max_abs_error": max_abs_error,
158
+ "checked_against_model_output": checked_against_model_output,
159
+ },
160
+ )
161
+
162
+
163
+ def _normalize_shap_output(
164
+ values: np.ndarray, base_values: np.ndarray, n_rows: int, n_features: int
165
+ ) -> tuple[np.ndarray, np.ndarray] | None:
166
+ values = np.asarray(values, dtype=float)
167
+ base_values = np.asarray(base_values, dtype=float)
168
+
169
+ if values.ndim == 3:
170
+ values = values[..., -1]
171
+ if values.ndim != 2 or values.shape != (n_rows, n_features):
172
+ return None
173
+
174
+ if base_values.ndim == 0:
175
+ baseline = np.full(n_rows, float(base_values))
176
+ elif base_values.ndim == 1:
177
+ baseline = base_values.reshape(-1)
178
+ elif base_values.ndim == 2:
179
+ baseline = base_values[:, -1].reshape(-1)
180
+ else:
181
+ return None
182
+
183
+ if baseline.shape[0] != n_rows:
184
+ return None
185
+ return values, baseline
186
+
187
+
188
+ # @id CODE-AIDS-109
189
+ # @implements REQ-AIDS-083
190
+ # @design DES-AIDS-071
191
+ def _compute_signed_contributions(
192
+ model: object, frame: pd.DataFrame, feature_names: list[str]
193
+ ) -> ExplainabilityResult:
194
+ """Return signed local contributions from the best available provider."""
195
+ try:
196
+ native = np.asarray(model.predict(frame, pred_contrib=True), dtype=float)
197
+ except (AttributeError, TypeError, ValueError):
198
+ native = None
199
+
200
+ if native is not None and native.ndim == 2 and native.shape[0] == len(frame):
201
+ if native.shape[1] == len(feature_names) + 1:
202
+ contributions = native[:, :-1]
203
+ baseline_values = native[:, -1]
204
+ elif native.shape[1] == len(feature_names):
205
+ contributions = native
206
+ baseline_values = np.zeros(native.shape[0], dtype=float)
207
+ else:
208
+ contributions = None
209
+ baseline_values = None
210
+
211
+ if contributions is not None and baseline_values is not None:
212
+ return _finalize_signed_result(
213
+ feature_names,
214
+ contribution_kind="pred_contrib",
215
+ contributions=contributions,
216
+ baseline_values=baseline_values,
217
+ raw_predictions=_predict_raw_output(model, frame),
218
+ )
219
+
220
+ try:
221
+ shap = import_module("shap")
222
+ except ModuleNotFoundError:
223
+ shap = None
224
+
225
+ if shap is not None:
226
+ try:
227
+ explanation = shap.Explainer(model, frame)(frame)
228
+ except (AttributeError, NotImplementedError, TypeError, ValueError):
229
+ explanation = None
230
+ if explanation is not None:
231
+ normalized = _normalize_shap_output(
232
+ explanation.values,
233
+ explanation.base_values,
234
+ n_rows=len(frame),
235
+ n_features=len(feature_names),
236
+ )
237
+ if normalized is not None:
238
+ values, baseline_values = normalized
239
+ return _finalize_signed_result(
240
+ feature_names,
241
+ contribution_kind="shap",
242
+ contributions=values,
243
+ baseline_values=baseline_values,
244
+ raw_predictions=_predict_raw_output(model, frame),
245
+ )
246
+
247
+ if hasattr(model, "coef_") and hasattr(model, "intercept_"):
248
+ coef = np.asarray(model.coef_, dtype=float).reshape(-1)
249
+ if coef.shape[0] != len(feature_names):
250
+ raise ValueError("Linear contribution path requires one coefficient per feature.")
251
+ baseline_values = np.full(len(frame), float(np.asarray(model.intercept_).reshape(-1)[0]))
252
+ contributions = frame.to_numpy(dtype=float) * coef
253
+ return _finalize_signed_result(
254
+ feature_names,
255
+ contribution_kind="linear",
256
+ contributions=contributions,
257
+ baseline_values=baseline_values,
258
+ raw_predictions=_predict_raw_output(model, frame),
259
+ )
260
+
261
+ raise ValueError(
262
+ "Signed contributions are unavailable for this model. Install optional "
263
+ "`shap`, use a model with native pred_contrib support, or request "
264
+ '`method="permutation"` instead.'
265
+ )
266
+
267
+
268
+ # @id CODE-AIDS-108
269
+ # @implements REQ-AIDS-084
270
+ # @design DES-AIDS-072
271
+ def _compute_permutation_importance(
272
+ model: object,
273
+ frame: pd.DataFrame,
274
+ y: Any,
275
+ feature_names: list[str],
276
+ scoring: str | None,
277
+ n_repeats: int,
278
+ random_state: int,
279
+ ) -> ExplainabilityResult:
280
+ """Return permutation importance with explicit scoring metadata."""
281
+ if y is None:
282
+ raise ValueError("y is required when method='permutation'.")
283
+
284
+ result = permutation_importance(
285
+ model,
286
+ frame,
287
+ np.asarray(y),
288
+ scoring=scoring,
289
+ n_repeats=n_repeats,
290
+ random_state=random_state,
291
+ )
292
+ feature_importances = _build_feature_importance_map(feature_names, result.importances_mean)
293
+ return ExplainabilityResult(
294
+ feature_importances=feature_importances,
295
+ ranking=_build_ranking(feature_importances),
296
+ importance_kind="permutation",
297
+ scoring=scoring,
298
+ )
299
+
300
+
301
+ # @id CODE-AIDS-106
302
+ # @implements REQ-AIDS-021 REQ-AIDS-082 REQ-AIDS-083 REQ-AIDS-084
303
+ # @design DES-AIDS-019 DES-AIDS-070 DES-AIDS-071 DES-AIDS-072
304
+ def explain_model(
305
+ model: object,
306
+ feature_names: list[str],
307
+ *,
308
+ method: Literal["default", "signed_contributions", "permutation"] = "default",
309
+ x: Any = None,
310
+ y: Any = None,
311
+ scoring: str | None = None,
312
+ n_repeats: int = 5,
313
+ random_state: int = 42,
314
+ ) -> ExplainabilityResult:
315
+ """Explain ``model`` with legacy ranking, signed contributions, or permutation importance."""
316
+ if method == "default":
317
+ feature_importances, ranking, importance_kind = _compute_default_importance(
318
+ model, feature_names
319
+ )
320
+ return ExplainabilityResult(
321
+ feature_importances=feature_importances,
322
+ ranking=ranking,
323
+ importance_kind=importance_kind,
324
+ )
325
+
326
+ frame = _as_feature_frame(x, feature_names)
327
+ if method == "signed_contributions":
328
+ return _compute_signed_contributions(model, frame, feature_names)
329
+ if method == "permutation":
330
+ return _compute_permutation_importance(
331
+ model=model,
332
+ frame=frame,
333
+ y=y,
334
+ feature_names=feature_names,
335
+ scoring=scoring,
336
+ n_repeats=n_repeats,
337
+ random_state=random_state,
338
+ )
339
+
340
+ raise ValueError(f"Unsupported explainability method: {method!r}")
@@ -0,0 +1,163 @@
1
+ """Feature engineering operations.
2
+
3
+ Implements DES-AIDS-013 (REQ-AIDS-015): applies the requested encoding,
4
+ scaling, group-wise aggregation, categorical interaction, missing-value
5
+ flagging, or explicit-edge binning transformation to a dataframe and reports
6
+ the resulting feature set.
7
+
8
+ Also implements DES-AIDS-061 (REQ-AIDS-073): a leakage-safe fit/transform
9
+ API that separates statistics estimation (``fit_features``) from
10
+ transformation application (``transform_features``), so a cross-validation
11
+ caller can fit on a training fold and transform a disjoint fold without ever
12
+ deriving statistics from the held-out rows.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from dataclasses import dataclass, field
18
+
19
+ import pandas as pd
20
+ from sklearn.preprocessing import StandardScaler
21
+
22
+ _SUPPORTED_OPERATIONS = (
23
+ "one_hot",
24
+ "scale",
25
+ "aggregate",
26
+ "interaction",
27
+ "missing_flag",
28
+ "bin",
29
+ )
30
+ _SUPPORTED_AGG_FUNCS = ("mean", "sum", "count_eq")
31
+
32
+
33
+ @dataclass(frozen=True)
34
+ class FeatureResult:
35
+ dataframe: pd.DataFrame
36
+ added_columns: list[str]
37
+ removed_columns: list[str]
38
+ definitions: dict = field(default_factory=dict)
39
+
40
+
41
+ # @id CODE-AIDS-093
42
+ # @implements REQ-AIDS-073
43
+ # @design DES-AIDS-061
44
+ @dataclass(frozen=True)
45
+ class FittedFeatureState:
46
+ operation: str
47
+ columns: tuple
48
+ scaler: StandardScaler
49
+
50
+
51
+ def _require_params(operation: str, params: dict, required: list) -> None:
52
+ missing = [name for name in required if name not in params or params[name] is None]
53
+ if missing:
54
+ raise ValueError(f"Missing required params for operation {operation!r}: {missing}")
55
+
56
+
57
+ # @id CODE-AIDS-015
58
+ # @implements REQ-AIDS-015
59
+ # @design DES-AIDS-013
60
+ def engineer_features(
61
+ df: pd.DataFrame, operation: str, columns: list[str] | None = None, **params
62
+ ) -> FeatureResult:
63
+ """Apply ``operation`` (e.g. one-hot encoding, scaling) to ``df``."""
64
+ if operation not in _SUPPORTED_OPERATIONS:
65
+ raise ValueError(f"Unsupported feature engineering operation: {operation!r}")
66
+
67
+ target_columns = columns or list(df.columns)
68
+ definitions: dict = {}
69
+
70
+ if operation == "one_hot":
71
+ result_df = pd.get_dummies(df, columns=target_columns)
72
+ added_columns = [c for c in result_df.columns if c not in df.columns]
73
+ removed_columns = [c for c in target_columns if c not in result_df.columns]
74
+ for col in added_columns:
75
+ definitions[col] = f"one_hot encoding of {', '.join(target_columns)}"
76
+ elif operation == "scale":
77
+ result_df = df.copy()
78
+ scaler = StandardScaler()
79
+ result_df[target_columns] = scaler.fit_transform(df[target_columns])
80
+ added_columns = []
81
+ removed_columns = []
82
+ elif operation == "aggregate":
83
+ _require_params(operation, params, ["group_col", "agg_func"])
84
+ group_col = params["group_col"]
85
+ agg_func = params["agg_func"]
86
+ if agg_func not in _SUPPORTED_AGG_FUNCS:
87
+ raise ValueError(f"Unsupported agg_func: {agg_func!r}")
88
+ if agg_func == "count_eq":
89
+ _require_params(operation, params, ["compare_value"])
90
+ result_df = df.copy()
91
+ added_columns = []
92
+ for source_col in target_columns:
93
+ new_col = f"{source_col}_{agg_func}_by_{group_col}"
94
+ if agg_func == "count_eq":
95
+ compare_value = params["compare_value"]
96
+ result_df[new_col] = (
97
+ df[source_col].eq(compare_value).groupby(df[group_col]).transform("sum")
98
+ )
99
+ else:
100
+ result_df[new_col] = df.groupby(group_col)[source_col].transform(agg_func)
101
+ definitions[new_col] = f"aggregate({agg_func}) of {source_col} grouped by {group_col}"
102
+ added_columns.append(new_col)
103
+ removed_columns = []
104
+ elif operation == "interaction":
105
+ _require_params(operation, params, ["col_a", "col_b"])
106
+ col_a = params["col_a"]
107
+ col_b = params["col_b"]
108
+ result_df = df.copy()
109
+ new_col = f"{col_a}__{col_b}_interaction"
110
+ result_df[new_col] = df[col_a].astype("string") + "__" + df[col_b].astype("string")
111
+ definitions[new_col] = f"interaction of {col_a} and {col_b}"
112
+ added_columns = [new_col]
113
+ removed_columns = []
114
+ elif operation == "missing_flag":
115
+ result_df = df.copy()
116
+ added_columns = []
117
+ for col in target_columns:
118
+ new_col = f"{col}_missing_flag"
119
+ result_df[new_col] = df[col].isna()
120
+ definitions[new_col] = f"missing_flag of {col}"
121
+ added_columns.append(new_col)
122
+ removed_columns = []
123
+ else: # bin
124
+ _require_params(operation, params, ["edges"])
125
+ edges = params["edges"]
126
+ result_df = df.copy()
127
+ added_columns = []
128
+ for col in target_columns:
129
+ new_col = f"{col}_bin"
130
+ result_df[new_col] = pd.cut(df[col], bins=edges, right=True, include_lowest=True)
131
+ definitions[new_col] = f"bin of {col} with edges {edges}"
132
+ added_columns.append(new_col)
133
+ removed_columns = []
134
+
135
+ return FeatureResult(
136
+ dataframe=result_df,
137
+ added_columns=added_columns,
138
+ removed_columns=removed_columns,
139
+ definitions=definitions,
140
+ )
141
+
142
+
143
+ # @id CODE-AIDS-094
144
+ # @implements REQ-AIDS-073
145
+ # @design DES-AIDS-061
146
+ def fit_features(df: pd.DataFrame, operation: str, columns: list[str]) -> FittedFeatureState:
147
+ """Estimate statistics for ``operation`` from ``df`` only (no leakage)."""
148
+ if operation != "scale":
149
+ raise ValueError(f"Unsupported fit_features operation: {operation!r}")
150
+ scaler = StandardScaler()
151
+ scaler.fit(df[columns])
152
+ return FittedFeatureState(operation=operation, columns=tuple(columns), scaler=scaler)
153
+
154
+
155
+ # @id CODE-AIDS-095
156
+ # @implements REQ-AIDS-073
157
+ # @design DES-AIDS-061
158
+ def transform_features(fitted_state: FittedFeatureState, df: pd.DataFrame) -> FeatureResult:
159
+ """Apply ``fitted_state``'s already-estimated statistics to ``df``."""
160
+ columns = list(fitted_state.columns)
161
+ result_df = df.copy()
162
+ result_df[columns] = fitted_state.scaler.transform(df[columns])
163
+ return FeatureResult(dataframe=result_df, added_columns=[], removed_columns=[], definitions={})
@@ -0,0 +1,32 @@
1
+ """TDD verification gate configuration.
2
+
3
+ Implements DES-AIDS-011 (REQ-AIDS-013): reads the musubix3 project
4
+ configuration to confirm the required pytest command the gate depends on
5
+ is correctly declared, so the gate can enforce a zero-failure pytest suite
6
+ before any implementation change is considered complete.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ from pathlib import Path
13
+
14
+
15
+ class TestCommandMissingError(ValueError):
16
+ """Raised when no required "test" command is declared in the config."""
17
+
18
+
19
+ # @id CODE-AIDS-013
20
+ # @implements REQ-AIDS-013
21
+ # @design DES-AIDS-011
22
+ def get_required_test_command(config_path: str = ".musubix/config.json") -> dict:
23
+ """Return the required "test" command entry from ``config_path``."""
24
+ config = json.loads(Path(config_path).read_text(encoding="utf-8"))
25
+ for command in config.get("commands", []):
26
+ if command.get("name") == "test":
27
+ if not command.get("required"):
28
+ raise TestCommandMissingError(
29
+ "The 'test' command is declared but not marked required."
30
+ )
31
+ return command
32
+ raise TestCommandMissingError("No required 'test' command declared in config.")
@@ -0,0 +1,127 @@
1
+ """Data source ingestion.
2
+
3
+ Implements DES-AIDS-005: loads CSV/Excel/database/API sources into an
4
+ in-memory dataframe. A network allowlist is enforced before any remote
5
+ call; a row-count limit is applied to the resulting dataframe after a
6
+ remote (database/API) fetch completes, bounding downstream processing —
7
+ not the fetch itself — and never applies to local CSV/Excel sources
8
+ (REQ-AIDS-014, REQ-AIDS-032; GitHub #57).
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import csv
14
+ from dataclasses import dataclass
15
+ from urllib.parse import urlparse
16
+
17
+ import pandas as pd
18
+
19
+ DEFAULT_ROW_LIMIT = 100_000
20
+ _REMOTE_KINDS = ("api", "database")
21
+
22
+
23
+ class NetworkAllowlistError(ValueError):
24
+ """Raised when a remote source host is not in the configured allowlist."""
25
+
26
+
27
+ @dataclass(frozen=True)
28
+ class SourceSpec:
29
+ """Describes a single ingestion request."""
30
+
31
+ kind: str # "csv" | "excel" | "api" | "database"
32
+ location: str
33
+
34
+
35
+ @dataclass(frozen=True)
36
+ class IngestionResult:
37
+ dataframe: pd.DataFrame
38
+ row_count: int
39
+ column_count: int
40
+ truncated: bool
41
+ warnings: tuple[str, ...] = ()
42
+
43
+
44
+ # @id CODE-AIDS-088
45
+ # @implements REQ-AIDS-068
46
+ # @design DES-AIDS-056
47
+ def _sniff_csv_delimiter(path: str) -> tuple[str | None, bool]:
48
+ """Return ``(delimiter, sniff_succeeded)`` for the CSV file at ``path``.
49
+
50
+ Reads a bounded text sample and attempts ``csv.Sniffer().sniff`` over
51
+ comma/tab/semicolon candidates (DES-AIDS-056); on ``csv.Error`` (an
52
+ ambiguous sample), returns ``(None, False)`` so the caller falls back to
53
+ the existing comma-default behavior.
54
+ """
55
+ with open(path, encoding="utf-8", errors="replace") as handle:
56
+ sample = handle.read(8192)
57
+ try:
58
+ dialect = csv.Sniffer().sniff(sample, delimiters=",\t;")
59
+ except csv.Error:
60
+ return None, False
61
+ return dialect.delimiter, True
62
+
63
+
64
+ # @id CODE-AIDS-014
65
+ # @implements REQ-AIDS-014
66
+ # @design DES-AIDS-005
67
+ # @id CODE-AIDS-032
68
+ # @implements REQ-AIDS-032
69
+ # @design DES-AIDS-005
70
+ def ingest(
71
+ source_spec: SourceSpec,
72
+ fetcher=None,
73
+ allowlist: tuple[str, ...] = (),
74
+ row_limit: int = DEFAULT_ROW_LIMIT,
75
+ ) -> IngestionResult:
76
+ """Load ``source_spec`` into a dataframe, applying ingestion safety rules.
77
+
78
+ For remote ``kind`` values ("api"/"database") the host is checked
79
+ against ``allowlist`` *before* ``fetcher`` is invoked, and the resulting
80
+ dataframe is truncated to ``row_limit`` rows if it exceeds that limit.
81
+ Local ``kind`` values ("csv"/"excel") are never truncated by
82
+ ``row_limit``: that limit is a remote-source safety control
83
+ (REQ-AIDS-032), not a general ingestion cap, so a local file is always
84
+ loaded in full (GitHub #57).
85
+ """
86
+ warnings: tuple[str, ...] = ()
87
+ if source_spec.kind == "csv":
88
+ delimiter, sniffed = _sniff_csv_delimiter(source_spec.location)
89
+ dataframe = pd.read_csv(source_spec.location, sep=delimiter if sniffed else ",")
90
+ if (
91
+ not sniffed
92
+ and len(dataframe.columns) == 1
93
+ and ("\t" in dataframe.columns[0] or ";" in dataframe.columns[0])
94
+ ):
95
+ warnings = (
96
+ f"CSV was parsed with the comma fallback as a single column named "
97
+ f"{dataframe.columns[0]!r}; the file may actually use a tab or "
98
+ "semicolon delimiter instead.",
99
+ )
100
+ elif source_spec.kind == "excel":
101
+ dataframe = pd.read_excel(source_spec.location)
102
+ elif source_spec.kind in _REMOTE_KINDS:
103
+ host = urlparse(source_spec.location).hostname
104
+ if host not in allowlist:
105
+ raise NetworkAllowlistError(
106
+ f"Host '{host}' is not in the configured allowlist "
107
+ f"(許可されていない接続先ホストです): {source_spec.location}"
108
+ )
109
+ if fetcher is None:
110
+ raise ValueError("A fetcher callable is required for remote ingestion sources.")
111
+ dataframe = fetcher(source_spec.location)
112
+ else:
113
+ raise ValueError(f"Unsupported ingestion source kind: {source_spec.kind!r}")
114
+
115
+ # GitHub #57: row_limit is a remote-source safety control (REQ-AIDS-032);
116
+ # local csv/excel files must never be silently truncated by it.
117
+ truncated = source_spec.kind in _REMOTE_KINDS and len(dataframe) > row_limit
118
+ if truncated:
119
+ dataframe = dataframe.iloc[:row_limit]
120
+
121
+ return IngestionResult(
122
+ dataframe=dataframe,
123
+ row_count=len(dataframe),
124
+ column_count=len(dataframe.columns),
125
+ truncated=truncated,
126
+ warnings=warnings,
127
+ )