jupytermind 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/skills/ai-chemistry-scientist/SKILL.md +97 -0
- package/.github/skills/ai-chemistry-scientist/manifest.json +156 -0
- package/.github/skills/ai-data-scientist/SKILL.md +330 -0
- package/.github/skills/ai-genomics-scientist/SKILL.md +98 -0
- package/.github/skills/ai-genomics-scientist/manifest.json +93 -0
- package/.github/skills/ai-materials-scientist/SKILL.md +51 -0
- package/.github/skills/ai-materials-scientist/manifest.json +58 -0
- package/.github/skills/ai-scientist/SKILL.md +69 -0
- package/.github/skills/ai-scientist/manifest.json +61 -0
- package/.github/skills/ai-structural-biology-scientist/SKILL.md +67 -0
- package/.github/skills/ai-structural-biology-scientist/manifest.json +72 -0
- package/.github/skills/japanese-prose/NOTICE.md +17 -0
- package/.github/skills/japanese-prose/SKILL.md +111 -0
- package/.github/skills/japanese-prose/references/review-workflow.md +50 -0
- package/.github/skills/japanese-prose/references/scoring.md +24 -0
- package/.github/skills/japanese-prose/references/writing-guidelines.md +60 -0
- package/.github/skills/japanese-prose/scripts/core.py +192 -0
- package/.github/skills/japanese-prose/scripts/fixtures/natural.md +5 -0
- package/.github/skills/japanese-prose/scripts/fixtures/unnatural.md +5 -0
- package/.github/skills/japanese-prose/scripts/lint.py +378 -0
- package/.github/skills/japanese-prose/scripts/outline.py +68 -0
- package/.github/skills/japanese-prose/scripts/terms.py +112 -0
- package/.github/skills/japanese-prose/scripts/test_engine.py +117 -0
- package/.github/skills/presentation-planner/SKILL.md +257 -0
- package/.github/skills/presentation-planner/assets/design-templates/data-report.yaml +97 -0
- package/.github/skills/presentation-planner/assets/design-templates/executive-proposal.yaml +92 -0
- package/.github/skills/presentation-planner/assets/design-templates/technical-briefing.yaml +96 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/data-report.md +47 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/executive-decision.md +43 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/technical-briefing.md +45 -0
- package/.github/skills/presentation-planner/references/customizing-design-templates.md +160 -0
- package/.github/skills/presentation-planner/references/design-spec-schema.md +72 -0
- package/.github/skills/presentation-planner/references/handoff-contract.md +49 -0
- package/.github/skills/presentation-planner/references/responsibility-boundary.md +32 -0
- package/.github/skills/presentation-planner/references/scenario-templates.md +55 -0
- package/.github/skills/tech-writer/SKILL.md +434 -0
- package/.github/skills/tech-writer/assets/templates/blueprint.md +187 -0
- package/.github/skills/tech-writer/assets/templates/design-doc.md +29 -0
- package/.github/skills/tech-writer/assets/templates/migration-plan.md +173 -0
- package/.github/skills/tech-writer/assets/templates/operations-runbook.md +202 -0
- package/.github/skills/tech-writer/assets/templates/pr-description.md +23 -0
- package/.github/skills/tech-writer/assets/templates/qiita.md +44 -0
- package/.github/skills/tech-writer/assets/templates/readme.md +38 -0
- package/.github/skills/tech-writer/assets/templates/requirements-definition.md +170 -0
- package/.github/skills/tech-writer/assets/templates/rfi.md +113 -0
- package/.github/skills/tech-writer/assets/templates/rfp.md +180 -0
- package/.github/skills/tech-writer/assets/templates/security-design.md +167 -0
- package/.github/skills/tech-writer/assets/templates/system-design.md +220 -0
- package/.github/skills/tech-writer/assets/templates/technical-proposal.md +112 -0
- package/.github/skills/tech-writer/assets/templates/test-plan.md +153 -0
- package/.github/skills/tech-writer/assets/templates/user-manual.md +22 -0
- package/.github/skills/tech-writer/assets/templates/white-paper.md +192 -0
- package/.github/skills/tech-writer/references/doctypes/api-docs.md +33 -0
- package/.github/skills/tech-writer/references/doctypes/blueprint.md +81 -0
- package/.github/skills/tech-writer/references/doctypes/code-comments.md +39 -0
- package/.github/skills/tech-writer/references/doctypes/design-doc.md +42 -0
- package/.github/skills/tech-writer/references/doctypes/migration-plan.md +63 -0
- package/.github/skills/tech-writer/references/doctypes/operations-runbook.md +63 -0
- package/.github/skills/tech-writer/references/doctypes/pr-commit.md +82 -0
- package/.github/skills/tech-writer/references/doctypes/qiita.md +75 -0
- package/.github/skills/tech-writer/references/doctypes/readme.md +43 -0
- package/.github/skills/tech-writer/references/doctypes/release-notes.md +30 -0
- package/.github/skills/tech-writer/references/doctypes/requirements-definition.md +61 -0
- package/.github/skills/tech-writer/references/doctypes/rfi.md +43 -0
- package/.github/skills/tech-writer/references/doctypes/rfp.md +46 -0
- package/.github/skills/tech-writer/references/doctypes/security-design.md +71 -0
- package/.github/skills/tech-writer/references/doctypes/system-design.md +74 -0
- package/.github/skills/tech-writer/references/doctypes/technical-proposal.md +49 -0
- package/.github/skills/tech-writer/references/doctypes/test-plan.md +67 -0
- package/.github/skills/tech-writer/references/doctypes/user-manual.md +58 -0
- package/.github/skills/tech-writer/references/doctypes/white-paper.md +84 -0
- package/.github/skills/tech-writer/references/doctypes/zenn.md +66 -0
- package/.github/skills/tech-writer/references/japanese-prose-optimization.md +110 -0
- package/.github/skills/tech-writer/references/style-constitution.md +104 -0
- package/.github/skills/tech-writer/scripts/lint.py +412 -0
- package/LICENSE +21 -0
- package/README.md +92 -0
- package/bin/ai-data-scientist.js +123 -0
- package/package.json +41 -0
- package/pyproject.toml +45 -0
- package/src/ai_chemistry_scientist/__init__.py +0 -0
- package/src/ai_chemistry_scientist/admet_prediction.py +71 -0
- package/src/ai_chemistry_scientist/bioactivity_classification.py +73 -0
- package/src/ai_chemistry_scientist/data/sample_molecules.csv +21 -0
- package/src/ai_chemistry_scientist/dispatch.py +369 -0
- package/src/ai_chemistry_scientist/docking_score.py +97 -0
- package/src/ai_chemistry_scientist/drug_likeness_rules.py +84 -0
- package/src/ai_chemistry_scientist/evidence.py +41 -0
- package/src/ai_chemistry_scientist/molecular_descriptors.py +97 -0
- package/src/ai_chemistry_scientist/molecular_formula_mass.py +40 -0
- package/src/ai_chemistry_scientist/molecular_similarity.py +78 -0
- package/src/ai_chemistry_scientist/qsar_modeling.py +105 -0
- package/src/ai_chemistry_scientist/salt_standardization.py +81 -0
- package/src/ai_chemistry_scientist/structural_alerts.py +76 -0
- package/src/ai_chemistry_scientist/structure_format_conversion.py +84 -0
- package/src/ai_chemistry_scientist/validation.py +70 -0
- package/src/ai_data_scientist/__init__.py +0 -0
- package/src/ai_data_scientist/analysis_assumptions.py +121 -0
- package/src/ai_data_scientist/anomaly_detection.py +39 -0
- package/src/ai_data_scientist/automl.py +109 -0
- package/src/ai_data_scientist/cleaning.py +56 -0
- package/src/ai_data_scientist/cli.py +90 -0
- package/src/ai_data_scientist/clustering.py +54 -0
- package/src/ai_data_scientist/dashboard.py +33 -0
- package/src/ai_data_scientist/data_definition.py +100 -0
- package/src/ai_data_scientist/data_quality.py +164 -0
- package/src/ai_data_scientist/dataset_validation.py +135 -0
- package/src/ai_data_scientist/dependency_pins.py +60 -0
- package/src/ai_data_scientist/eda.py +82 -0
- package/src/ai_data_scientist/experiment_evaluation.py +635 -0
- package/src/ai_data_scientist/explainability.py +340 -0
- package/src/ai_data_scientist/feature_engineering.py +163 -0
- package/src/ai_data_scientist/gate_config.py +32 -0
- package/src/ai_data_scientist/ingestion.py +127 -0
- package/src/ai_data_scientist/insight_engine.py +180 -0
- package/src/ai_data_scientist/japanese_nlp.py +43 -0
- package/src/ai_data_scientist/jupyter_launcher.py +137 -0
- package/src/ai_data_scientist/jupyter_mcp_client.py +94 -0
- package/src/ai_data_scientist/language_router.py +28 -0
- package/src/ai_data_scientist/lifecycle.py +221 -0
- package/src/ai_data_scientist/mcp_gateway.py +113 -0
- package/src/ai_data_scientist/mcp_runtime.py +194 -0
- package/src/ai_data_scientist/mcp_transport.py +53 -0
- package/src/ai_data_scientist/ml_modeling.py +451 -0
- package/src/ai_data_scientist/model_tuning.py +104 -0
- package/src/ai_data_scientist/notebook_audit.py +574 -0
- package/src/ai_data_scientist/project_manager.py +243 -0
- package/src/ai_data_scientist/report_export.py +73 -0
- package/src/ai_data_scientist/sensitivity.py +445 -0
- package/src/ai_data_scientist/signal_analysis.py +201 -0
- package/src/ai_data_scientist/skill_packaging.py +40 -0
- package/src/ai_data_scientist/stats_analysis.py +88 -0
- package/src/ai_data_scientist/text_nlp.py +44 -0
- package/src/ai_data_scientist/timeseries.py +68 -0
- package/src/ai_data_scientist/visualization.py +708 -0
- package/src/ai_genomics_scientist/__init__.py +1 -0
- package/src/ai_genomics_scientist/differential_expression.py +147 -0
- package/src/ai_genomics_scientist/dispatch.py +267 -0
- package/src/ai_genomics_scientist/evidence.py +45 -0
- package/src/ai_genomics_scientist/gene_set_enrichment.py +76 -0
- package/src/ai_genomics_scientist/sequence_alignment.py +97 -0
- package/src/ai_genomics_scientist/sequence_features.py +111 -0
- package/src/ai_genomics_scientist/splice_site_scoring.py +66 -0
- package/src/ai_genomics_scientist/validation.py +83 -0
- package/src/ai_genomics_scientist/variant_effect.py +147 -0
- package/src/ai_genomics_scientist/variant_pathogenicity.py +125 -0
- package/src/ai_materials_scientist/__init__.py +0 -0
- package/src/ai_materials_scientist/calphad.py +117 -0
- package/src/ai_materials_scientist/classical_monte_carlo.py +165 -0
- package/src/ai_materials_scientist/crystal_plasticity.py +184 -0
- package/src/ai_materials_scientist/dispatch.py +100 -0
- package/src/ai_materials_scientist/evidence.py +84 -0
- package/src/ai_materials_scientist/fem.py +279 -0
- package/src/ai_materials_scientist/kinetic_monte_carlo.py +145 -0
- package/src/ai_materials_scientist/molecular_dynamics.py +240 -0
- package/src/ai_materials_scientist/phase_field.py +167 -0
- package/src/ai_materials_scientist/validation.py +70 -0
- package/src/ai_scientist/__init__.py +1 -0
- package/src/ai_scientist/completion_gate.py +15 -0
- package/src/ai_scientist/data_analysis.py +46 -0
- package/src/ai_scientist/evidence_registry.py +99 -0
- package/src/ai_scientist/experimental_design.py +20 -0
- package/src/ai_scientist/language.py +14 -0
- package/src/ai_scientist/latex_renderer.py +41 -0
- package/src/ai_scientist/literature_review.py +37 -0
- package/src/ai_scientist/manifest.py +87 -0
- package/src/ai_scientist/manuscript.py +94 -0
- package/src/ai_scientist/mcp_config.py +76 -0
- package/src/ai_scientist/mcp_external.py +42 -0
- package/src/ai_scientist/mcp_failures.py +23 -0
- package/src/ai_scientist/mcp_gateway.py +38 -0
- package/src/ai_scientist/mcp_managed.py +180 -0
- package/src/ai_scientist/npm_packaging.py +49 -0
- package/src/ai_scientist/orchestrator.py +133 -0
- package/src/ai_scientist/peer_review.py +60 -0
- package/src/ai_scientist/phase_gate.py +74 -0
- package/src/ai_scientist/phase_state.py +230 -0
- package/src/ai_scientist/presentation.py +56 -0
- package/src/ai_scientist/project_config.py +31 -0
- package/src/ai_scientist/project_handle.py +74 -0
- package/src/ai_scientist/reproducibility.py +20 -0
- package/src/ai_scientist/research_planning.py +20 -0
- package/src/ai_scientist/skill_invocation.py +21 -0
- package/src/ai_scientist/tdd_gate.py +99 -0
- package/src/ai_structural_biology_scientist/__init__.py +0 -0
- package/src/ai_structural_biology_scientist/contact_map.py +87 -0
- package/src/ai_structural_biology_scientist/dispatch.py +269 -0
- package/src/ai_structural_biology_scientist/evidence.py +43 -0
- package/src/ai_structural_biology_scientist/hydrophobicity.py +101 -0
- package/src/ai_structural_biology_scientist/protein_docking_score.py +104 -0
- package/src/ai_structural_biology_scientist/secondary_structure.py +95 -0
- package/src/ai_structural_biology_scientist/structural_similarity.py +74 -0
- package/src/ai_structural_biology_scientist/validation.py +100 -0
|
@@ -0,0 +1,340 @@
|
|
|
1
|
+
"""Model explainability.
|
|
2
|
+
|
|
3
|
+
Implements DES-AIDS-019 plus CHANGE-011's DES-AIDS-070/071/072: preserves the
|
|
4
|
+
legacy feature-importance ranking by default, and adds opt-in signed local
|
|
5
|
+
contributions and permutation importance. CODE-AIDS-106 through CODE-AIDS-110
|
|
6
|
+
cover importance-kind labeling, the signed-contribution provider chain
|
|
7
|
+
(native pred_contrib / SHAP / linear fallback), and permutation importance.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from importlib import import_module
|
|
14
|
+
from typing import Any, Literal
|
|
15
|
+
|
|
16
|
+
import numpy as np
|
|
17
|
+
import pandas as pd
|
|
18
|
+
from sklearn.inspection import permutation_importance
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class ExplainabilityResult:
|
|
23
|
+
feature_importances: dict[str, float]
|
|
24
|
+
ranking: list[str]
|
|
25
|
+
importance_kind: str = "unknown"
|
|
26
|
+
contribution_kind: str | None = None
|
|
27
|
+
signed_contributions: list[dict[str, float]] | None = None
|
|
28
|
+
baseline_values: list[float] | None = None
|
|
29
|
+
raw_predictions: list[float] | None = None
|
|
30
|
+
additivity_check: dict[str, float | bool | None] | None = None
|
|
31
|
+
scoring: str | None = None
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _as_feature_frame(x: Any, feature_names: list[str]) -> pd.DataFrame:
|
|
35
|
+
if x is None:
|
|
36
|
+
raise ValueError("x is required for this explainability method.")
|
|
37
|
+
if isinstance(x, pd.DataFrame):
|
|
38
|
+
return x.loc[:, feature_names].copy()
|
|
39
|
+
|
|
40
|
+
values = np.asarray(x)
|
|
41
|
+
if values.ndim != 2 or values.shape[1] != len(feature_names):
|
|
42
|
+
raise ValueError("x must be a 2D array-like with one column per feature name.")
|
|
43
|
+
return pd.DataFrame(values, columns=feature_names)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _build_ranking(feature_importances: dict[str, float]) -> list[str]:
|
|
47
|
+
return sorted(feature_importances, key=feature_importances.get, reverse=True)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _build_feature_importance_map(
|
|
51
|
+
feature_names: list[str], raw_importances: np.ndarray
|
|
52
|
+
) -> dict[str, float]:
|
|
53
|
+
return {name: float(value) for name, value in zip(feature_names, np.asarray(raw_importances))}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
# @id CODE-AIDS-128
|
|
57
|
+
# @implements REQ-AIDS-082
|
|
58
|
+
# @design DES-AIDS-070
|
|
59
|
+
def _normalize_coefficient_importances(coef: Any) -> np.ndarray:
|
|
60
|
+
raw_importances = np.abs(np.asarray(coef, dtype=float))
|
|
61
|
+
if raw_importances.ndim == 2 and raw_importances.shape[0] > 1:
|
|
62
|
+
return raw_importances.mean(axis=0)
|
|
63
|
+
return raw_importances.reshape(-1)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
# @id CODE-AIDS-107
|
|
67
|
+
# @implements REQ-AIDS-082
|
|
68
|
+
# @design DES-AIDS-070
|
|
69
|
+
def _compute_default_importance(
|
|
70
|
+
model: object, feature_names: list[str]
|
|
71
|
+
) -> tuple[dict[str, float], list[str], str]:
|
|
72
|
+
"""Return the legacy global-importance ranking and its explicit kind."""
|
|
73
|
+
if hasattr(model, "feature_importances_"):
|
|
74
|
+
raw_importances = np.asarray(model.feature_importances_, dtype=float)
|
|
75
|
+
importance_kind = "split"
|
|
76
|
+
elif hasattr(model, "coef_"):
|
|
77
|
+
raw_importances = _normalize_coefficient_importances(model.coef_)
|
|
78
|
+
importance_kind = "coefficient_magnitude"
|
|
79
|
+
else:
|
|
80
|
+
raise ValueError("Model exposes neither feature_importances_ nor coef_.")
|
|
81
|
+
|
|
82
|
+
feature_importances = _build_feature_importance_map(feature_names, raw_importances)
|
|
83
|
+
return feature_importances, _build_ranking(feature_importances), importance_kind
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
# @id CODE-AIDS-110
|
|
87
|
+
# @implements REQ-AIDS-083
|
|
88
|
+
# @design DES-AIDS-071
|
|
89
|
+
def _predict_raw_output(model: object, frame: pd.DataFrame) -> np.ndarray | None:
|
|
90
|
+
"""Best-effort raw-output prediction for additivity checks."""
|
|
91
|
+
if hasattr(model, "decision_function"):
|
|
92
|
+
raw = np.asarray(model.decision_function(frame), dtype=float)
|
|
93
|
+
return raw.reshape(-1)
|
|
94
|
+
|
|
95
|
+
for kwargs in ({"raw_score": True}, {"output_margin": True}):
|
|
96
|
+
try:
|
|
97
|
+
raw = np.asarray(model.predict(frame, **kwargs), dtype=float)
|
|
98
|
+
except (TypeError, ValueError):
|
|
99
|
+
continue
|
|
100
|
+
if raw.ndim == 1:
|
|
101
|
+
return raw
|
|
102
|
+
if raw.ndim == 2 and raw.shape[1] == 1:
|
|
103
|
+
return raw.reshape(-1)
|
|
104
|
+
|
|
105
|
+
if hasattr(model, "predict") and not hasattr(model, "predict_proba"):
|
|
106
|
+
try:
|
|
107
|
+
raw = np.asarray(model.predict(frame), dtype=float)
|
|
108
|
+
except (TypeError, ValueError):
|
|
109
|
+
return None
|
|
110
|
+
if raw.ndim == 1:
|
|
111
|
+
return raw
|
|
112
|
+
if raw.ndim == 2 and raw.shape[1] == 1:
|
|
113
|
+
return raw.reshape(-1)
|
|
114
|
+
|
|
115
|
+
return None
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _finalize_signed_result(
|
|
119
|
+
feature_names: list[str],
|
|
120
|
+
contribution_kind: str,
|
|
121
|
+
contributions: np.ndarray,
|
|
122
|
+
baseline_values: np.ndarray,
|
|
123
|
+
raw_predictions: np.ndarray | None,
|
|
124
|
+
) -> ExplainabilityResult:
|
|
125
|
+
contributions = np.asarray(contributions, dtype=float)
|
|
126
|
+
baseline_values = np.asarray(baseline_values, dtype=float).reshape(-1)
|
|
127
|
+
reconstructed = baseline_values + contributions.sum(axis=1)
|
|
128
|
+
|
|
129
|
+
if raw_predictions is None:
|
|
130
|
+
raw_prediction_values = None
|
|
131
|
+
checked_against_model_output = False
|
|
132
|
+
max_abs_error = None
|
|
133
|
+
passed = None
|
|
134
|
+
else:
|
|
135
|
+
raw_predictions = np.asarray(raw_predictions, dtype=float).reshape(-1)
|
|
136
|
+
raw_prediction_values = raw_predictions.astype(float).tolist()
|
|
137
|
+
checked_against_model_output = True
|
|
138
|
+
max_abs_error = float(np.max(np.abs(reconstructed - raw_predictions)))
|
|
139
|
+
passed = bool(max_abs_error <= 1e-6)
|
|
140
|
+
signed_contributions = [
|
|
141
|
+
{name: float(value) for name, value in zip(feature_names, row)} for row in contributions
|
|
142
|
+
]
|
|
143
|
+
feature_importances = _build_feature_importance_map(
|
|
144
|
+
feature_names, np.mean(np.abs(contributions), axis=0)
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
return ExplainabilityResult(
|
|
148
|
+
feature_importances=feature_importances,
|
|
149
|
+
ranking=_build_ranking(feature_importances),
|
|
150
|
+
importance_kind="mean_absolute_signed_contribution",
|
|
151
|
+
contribution_kind=contribution_kind,
|
|
152
|
+
signed_contributions=signed_contributions,
|
|
153
|
+
baseline_values=baseline_values.astype(float).tolist(),
|
|
154
|
+
raw_predictions=raw_prediction_values,
|
|
155
|
+
additivity_check={
|
|
156
|
+
"passed": passed,
|
|
157
|
+
"max_abs_error": max_abs_error,
|
|
158
|
+
"checked_against_model_output": checked_against_model_output,
|
|
159
|
+
},
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _normalize_shap_output(
|
|
164
|
+
values: np.ndarray, base_values: np.ndarray, n_rows: int, n_features: int
|
|
165
|
+
) -> tuple[np.ndarray, np.ndarray] | None:
|
|
166
|
+
values = np.asarray(values, dtype=float)
|
|
167
|
+
base_values = np.asarray(base_values, dtype=float)
|
|
168
|
+
|
|
169
|
+
if values.ndim == 3:
|
|
170
|
+
values = values[..., -1]
|
|
171
|
+
if values.ndim != 2 or values.shape != (n_rows, n_features):
|
|
172
|
+
return None
|
|
173
|
+
|
|
174
|
+
if base_values.ndim == 0:
|
|
175
|
+
baseline = np.full(n_rows, float(base_values))
|
|
176
|
+
elif base_values.ndim == 1:
|
|
177
|
+
baseline = base_values.reshape(-1)
|
|
178
|
+
elif base_values.ndim == 2:
|
|
179
|
+
baseline = base_values[:, -1].reshape(-1)
|
|
180
|
+
else:
|
|
181
|
+
return None
|
|
182
|
+
|
|
183
|
+
if baseline.shape[0] != n_rows:
|
|
184
|
+
return None
|
|
185
|
+
return values, baseline
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
# @id CODE-AIDS-109
|
|
189
|
+
# @implements REQ-AIDS-083
|
|
190
|
+
# @design DES-AIDS-071
|
|
191
|
+
def _compute_signed_contributions(
|
|
192
|
+
model: object, frame: pd.DataFrame, feature_names: list[str]
|
|
193
|
+
) -> ExplainabilityResult:
|
|
194
|
+
"""Return signed local contributions from the best available provider."""
|
|
195
|
+
try:
|
|
196
|
+
native = np.asarray(model.predict(frame, pred_contrib=True), dtype=float)
|
|
197
|
+
except (AttributeError, TypeError, ValueError):
|
|
198
|
+
native = None
|
|
199
|
+
|
|
200
|
+
if native is not None and native.ndim == 2 and native.shape[0] == len(frame):
|
|
201
|
+
if native.shape[1] == len(feature_names) + 1:
|
|
202
|
+
contributions = native[:, :-1]
|
|
203
|
+
baseline_values = native[:, -1]
|
|
204
|
+
elif native.shape[1] == len(feature_names):
|
|
205
|
+
contributions = native
|
|
206
|
+
baseline_values = np.zeros(native.shape[0], dtype=float)
|
|
207
|
+
else:
|
|
208
|
+
contributions = None
|
|
209
|
+
baseline_values = None
|
|
210
|
+
|
|
211
|
+
if contributions is not None and baseline_values is not None:
|
|
212
|
+
return _finalize_signed_result(
|
|
213
|
+
feature_names,
|
|
214
|
+
contribution_kind="pred_contrib",
|
|
215
|
+
contributions=contributions,
|
|
216
|
+
baseline_values=baseline_values,
|
|
217
|
+
raw_predictions=_predict_raw_output(model, frame),
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
try:
|
|
221
|
+
shap = import_module("shap")
|
|
222
|
+
except ModuleNotFoundError:
|
|
223
|
+
shap = None
|
|
224
|
+
|
|
225
|
+
if shap is not None:
|
|
226
|
+
try:
|
|
227
|
+
explanation = shap.Explainer(model, frame)(frame)
|
|
228
|
+
except (AttributeError, NotImplementedError, TypeError, ValueError):
|
|
229
|
+
explanation = None
|
|
230
|
+
if explanation is not None:
|
|
231
|
+
normalized = _normalize_shap_output(
|
|
232
|
+
explanation.values,
|
|
233
|
+
explanation.base_values,
|
|
234
|
+
n_rows=len(frame),
|
|
235
|
+
n_features=len(feature_names),
|
|
236
|
+
)
|
|
237
|
+
if normalized is not None:
|
|
238
|
+
values, baseline_values = normalized
|
|
239
|
+
return _finalize_signed_result(
|
|
240
|
+
feature_names,
|
|
241
|
+
contribution_kind="shap",
|
|
242
|
+
contributions=values,
|
|
243
|
+
baseline_values=baseline_values,
|
|
244
|
+
raw_predictions=_predict_raw_output(model, frame),
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
if hasattr(model, "coef_") and hasattr(model, "intercept_"):
|
|
248
|
+
coef = np.asarray(model.coef_, dtype=float).reshape(-1)
|
|
249
|
+
if coef.shape[0] != len(feature_names):
|
|
250
|
+
raise ValueError("Linear contribution path requires one coefficient per feature.")
|
|
251
|
+
baseline_values = np.full(len(frame), float(np.asarray(model.intercept_).reshape(-1)[0]))
|
|
252
|
+
contributions = frame.to_numpy(dtype=float) * coef
|
|
253
|
+
return _finalize_signed_result(
|
|
254
|
+
feature_names,
|
|
255
|
+
contribution_kind="linear",
|
|
256
|
+
contributions=contributions,
|
|
257
|
+
baseline_values=baseline_values,
|
|
258
|
+
raw_predictions=_predict_raw_output(model, frame),
|
|
259
|
+
)
|
|
260
|
+
|
|
261
|
+
raise ValueError(
|
|
262
|
+
"Signed contributions are unavailable for this model. Install optional "
|
|
263
|
+
"`shap`, use a model with native pred_contrib support, or request "
|
|
264
|
+
'`method="permutation"` instead.'
|
|
265
|
+
)
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
# @id CODE-AIDS-108
|
|
269
|
+
# @implements REQ-AIDS-084
|
|
270
|
+
# @design DES-AIDS-072
|
|
271
|
+
def _compute_permutation_importance(
|
|
272
|
+
model: object,
|
|
273
|
+
frame: pd.DataFrame,
|
|
274
|
+
y: Any,
|
|
275
|
+
feature_names: list[str],
|
|
276
|
+
scoring: str | None,
|
|
277
|
+
n_repeats: int,
|
|
278
|
+
random_state: int,
|
|
279
|
+
) -> ExplainabilityResult:
|
|
280
|
+
"""Return permutation importance with explicit scoring metadata."""
|
|
281
|
+
if y is None:
|
|
282
|
+
raise ValueError("y is required when method='permutation'.")
|
|
283
|
+
|
|
284
|
+
result = permutation_importance(
|
|
285
|
+
model,
|
|
286
|
+
frame,
|
|
287
|
+
np.asarray(y),
|
|
288
|
+
scoring=scoring,
|
|
289
|
+
n_repeats=n_repeats,
|
|
290
|
+
random_state=random_state,
|
|
291
|
+
)
|
|
292
|
+
feature_importances = _build_feature_importance_map(feature_names, result.importances_mean)
|
|
293
|
+
return ExplainabilityResult(
|
|
294
|
+
feature_importances=feature_importances,
|
|
295
|
+
ranking=_build_ranking(feature_importances),
|
|
296
|
+
importance_kind="permutation",
|
|
297
|
+
scoring=scoring,
|
|
298
|
+
)
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
# @id CODE-AIDS-106
|
|
302
|
+
# @implements REQ-AIDS-021 REQ-AIDS-082 REQ-AIDS-083 REQ-AIDS-084
|
|
303
|
+
# @design DES-AIDS-019 DES-AIDS-070 DES-AIDS-071 DES-AIDS-072
|
|
304
|
+
def explain_model(
|
|
305
|
+
model: object,
|
|
306
|
+
feature_names: list[str],
|
|
307
|
+
*,
|
|
308
|
+
method: Literal["default", "signed_contributions", "permutation"] = "default",
|
|
309
|
+
x: Any = None,
|
|
310
|
+
y: Any = None,
|
|
311
|
+
scoring: str | None = None,
|
|
312
|
+
n_repeats: int = 5,
|
|
313
|
+
random_state: int = 42,
|
|
314
|
+
) -> ExplainabilityResult:
|
|
315
|
+
"""Explain ``model`` with legacy ranking, signed contributions, or permutation importance."""
|
|
316
|
+
if method == "default":
|
|
317
|
+
feature_importances, ranking, importance_kind = _compute_default_importance(
|
|
318
|
+
model, feature_names
|
|
319
|
+
)
|
|
320
|
+
return ExplainabilityResult(
|
|
321
|
+
feature_importances=feature_importances,
|
|
322
|
+
ranking=ranking,
|
|
323
|
+
importance_kind=importance_kind,
|
|
324
|
+
)
|
|
325
|
+
|
|
326
|
+
frame = _as_feature_frame(x, feature_names)
|
|
327
|
+
if method == "signed_contributions":
|
|
328
|
+
return _compute_signed_contributions(model, frame, feature_names)
|
|
329
|
+
if method == "permutation":
|
|
330
|
+
return _compute_permutation_importance(
|
|
331
|
+
model=model,
|
|
332
|
+
frame=frame,
|
|
333
|
+
y=y,
|
|
334
|
+
feature_names=feature_names,
|
|
335
|
+
scoring=scoring,
|
|
336
|
+
n_repeats=n_repeats,
|
|
337
|
+
random_state=random_state,
|
|
338
|
+
)
|
|
339
|
+
|
|
340
|
+
raise ValueError(f"Unsupported explainability method: {method!r}")
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
"""Feature engineering operations.
|
|
2
|
+
|
|
3
|
+
Implements DES-AIDS-013 (REQ-AIDS-015): applies the requested encoding,
|
|
4
|
+
scaling, group-wise aggregation, categorical interaction, missing-value
|
|
5
|
+
flagging, or explicit-edge binning transformation to a dataframe and reports
|
|
6
|
+
the resulting feature set.
|
|
7
|
+
|
|
8
|
+
Also implements DES-AIDS-061 (REQ-AIDS-073): a leakage-safe fit/transform
|
|
9
|
+
API that separates statistics estimation (``fit_features``) from
|
|
10
|
+
transformation application (``transform_features``), so a cross-validation
|
|
11
|
+
caller can fit on a training fold and transform a disjoint fold without ever
|
|
12
|
+
deriving statistics from the held-out rows.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from dataclasses import dataclass, field
|
|
18
|
+
|
|
19
|
+
import pandas as pd
|
|
20
|
+
from sklearn.preprocessing import StandardScaler
|
|
21
|
+
|
|
22
|
+
_SUPPORTED_OPERATIONS = (
|
|
23
|
+
"one_hot",
|
|
24
|
+
"scale",
|
|
25
|
+
"aggregate",
|
|
26
|
+
"interaction",
|
|
27
|
+
"missing_flag",
|
|
28
|
+
"bin",
|
|
29
|
+
)
|
|
30
|
+
_SUPPORTED_AGG_FUNCS = ("mean", "sum", "count_eq")
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(frozen=True)
|
|
34
|
+
class FeatureResult:
|
|
35
|
+
dataframe: pd.DataFrame
|
|
36
|
+
added_columns: list[str]
|
|
37
|
+
removed_columns: list[str]
|
|
38
|
+
definitions: dict = field(default_factory=dict)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
# @id CODE-AIDS-093
|
|
42
|
+
# @implements REQ-AIDS-073
|
|
43
|
+
# @design DES-AIDS-061
|
|
44
|
+
@dataclass(frozen=True)
|
|
45
|
+
class FittedFeatureState:
|
|
46
|
+
operation: str
|
|
47
|
+
columns: tuple
|
|
48
|
+
scaler: StandardScaler
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _require_params(operation: str, params: dict, required: list) -> None:
|
|
52
|
+
missing = [name for name in required if name not in params or params[name] is None]
|
|
53
|
+
if missing:
|
|
54
|
+
raise ValueError(f"Missing required params for operation {operation!r}: {missing}")
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
# @id CODE-AIDS-015
|
|
58
|
+
# @implements REQ-AIDS-015
|
|
59
|
+
# @design DES-AIDS-013
|
|
60
|
+
def engineer_features(
|
|
61
|
+
df: pd.DataFrame, operation: str, columns: list[str] | None = None, **params
|
|
62
|
+
) -> FeatureResult:
|
|
63
|
+
"""Apply ``operation`` (e.g. one-hot encoding, scaling) to ``df``."""
|
|
64
|
+
if operation not in _SUPPORTED_OPERATIONS:
|
|
65
|
+
raise ValueError(f"Unsupported feature engineering operation: {operation!r}")
|
|
66
|
+
|
|
67
|
+
target_columns = columns or list(df.columns)
|
|
68
|
+
definitions: dict = {}
|
|
69
|
+
|
|
70
|
+
if operation == "one_hot":
|
|
71
|
+
result_df = pd.get_dummies(df, columns=target_columns)
|
|
72
|
+
added_columns = [c for c in result_df.columns if c not in df.columns]
|
|
73
|
+
removed_columns = [c for c in target_columns if c not in result_df.columns]
|
|
74
|
+
for col in added_columns:
|
|
75
|
+
definitions[col] = f"one_hot encoding of {', '.join(target_columns)}"
|
|
76
|
+
elif operation == "scale":
|
|
77
|
+
result_df = df.copy()
|
|
78
|
+
scaler = StandardScaler()
|
|
79
|
+
result_df[target_columns] = scaler.fit_transform(df[target_columns])
|
|
80
|
+
added_columns = []
|
|
81
|
+
removed_columns = []
|
|
82
|
+
elif operation == "aggregate":
|
|
83
|
+
_require_params(operation, params, ["group_col", "agg_func"])
|
|
84
|
+
group_col = params["group_col"]
|
|
85
|
+
agg_func = params["agg_func"]
|
|
86
|
+
if agg_func not in _SUPPORTED_AGG_FUNCS:
|
|
87
|
+
raise ValueError(f"Unsupported agg_func: {agg_func!r}")
|
|
88
|
+
if agg_func == "count_eq":
|
|
89
|
+
_require_params(operation, params, ["compare_value"])
|
|
90
|
+
result_df = df.copy()
|
|
91
|
+
added_columns = []
|
|
92
|
+
for source_col in target_columns:
|
|
93
|
+
new_col = f"{source_col}_{agg_func}_by_{group_col}"
|
|
94
|
+
if agg_func == "count_eq":
|
|
95
|
+
compare_value = params["compare_value"]
|
|
96
|
+
result_df[new_col] = (
|
|
97
|
+
df[source_col].eq(compare_value).groupby(df[group_col]).transform("sum")
|
|
98
|
+
)
|
|
99
|
+
else:
|
|
100
|
+
result_df[new_col] = df.groupby(group_col)[source_col].transform(agg_func)
|
|
101
|
+
definitions[new_col] = f"aggregate({agg_func}) of {source_col} grouped by {group_col}"
|
|
102
|
+
added_columns.append(new_col)
|
|
103
|
+
removed_columns = []
|
|
104
|
+
elif operation == "interaction":
|
|
105
|
+
_require_params(operation, params, ["col_a", "col_b"])
|
|
106
|
+
col_a = params["col_a"]
|
|
107
|
+
col_b = params["col_b"]
|
|
108
|
+
result_df = df.copy()
|
|
109
|
+
new_col = f"{col_a}__{col_b}_interaction"
|
|
110
|
+
result_df[new_col] = df[col_a].astype("string") + "__" + df[col_b].astype("string")
|
|
111
|
+
definitions[new_col] = f"interaction of {col_a} and {col_b}"
|
|
112
|
+
added_columns = [new_col]
|
|
113
|
+
removed_columns = []
|
|
114
|
+
elif operation == "missing_flag":
|
|
115
|
+
result_df = df.copy()
|
|
116
|
+
added_columns = []
|
|
117
|
+
for col in target_columns:
|
|
118
|
+
new_col = f"{col}_missing_flag"
|
|
119
|
+
result_df[new_col] = df[col].isna()
|
|
120
|
+
definitions[new_col] = f"missing_flag of {col}"
|
|
121
|
+
added_columns.append(new_col)
|
|
122
|
+
removed_columns = []
|
|
123
|
+
else: # bin
|
|
124
|
+
_require_params(operation, params, ["edges"])
|
|
125
|
+
edges = params["edges"]
|
|
126
|
+
result_df = df.copy()
|
|
127
|
+
added_columns = []
|
|
128
|
+
for col in target_columns:
|
|
129
|
+
new_col = f"{col}_bin"
|
|
130
|
+
result_df[new_col] = pd.cut(df[col], bins=edges, right=True, include_lowest=True)
|
|
131
|
+
definitions[new_col] = f"bin of {col} with edges {edges}"
|
|
132
|
+
added_columns.append(new_col)
|
|
133
|
+
removed_columns = []
|
|
134
|
+
|
|
135
|
+
return FeatureResult(
|
|
136
|
+
dataframe=result_df,
|
|
137
|
+
added_columns=added_columns,
|
|
138
|
+
removed_columns=removed_columns,
|
|
139
|
+
definitions=definitions,
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
# @id CODE-AIDS-094
|
|
144
|
+
# @implements REQ-AIDS-073
|
|
145
|
+
# @design DES-AIDS-061
|
|
146
|
+
def fit_features(df: pd.DataFrame, operation: str, columns: list[str]) -> FittedFeatureState:
|
|
147
|
+
"""Estimate statistics for ``operation`` from ``df`` only (no leakage)."""
|
|
148
|
+
if operation != "scale":
|
|
149
|
+
raise ValueError(f"Unsupported fit_features operation: {operation!r}")
|
|
150
|
+
scaler = StandardScaler()
|
|
151
|
+
scaler.fit(df[columns])
|
|
152
|
+
return FittedFeatureState(operation=operation, columns=tuple(columns), scaler=scaler)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
# @id CODE-AIDS-095
|
|
156
|
+
# @implements REQ-AIDS-073
|
|
157
|
+
# @design DES-AIDS-061
|
|
158
|
+
def transform_features(fitted_state: FittedFeatureState, df: pd.DataFrame) -> FeatureResult:
|
|
159
|
+
"""Apply ``fitted_state``'s already-estimated statistics to ``df``."""
|
|
160
|
+
columns = list(fitted_state.columns)
|
|
161
|
+
result_df = df.copy()
|
|
162
|
+
result_df[columns] = fitted_state.scaler.transform(df[columns])
|
|
163
|
+
return FeatureResult(dataframe=result_df, added_columns=[], removed_columns=[], definitions={})
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""TDD verification gate configuration.
|
|
2
|
+
|
|
3
|
+
Implements DES-AIDS-011 (REQ-AIDS-013): reads the musubix3 project
|
|
4
|
+
configuration to confirm the required pytest command the gate depends on
|
|
5
|
+
is correctly declared, so the gate can enforce a zero-failure pytest suite
|
|
6
|
+
before any implementation change is considered complete.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class TestCommandMissingError(ValueError):
|
|
16
|
+
"""Raised when no required "test" command is declared in the config."""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
# @id CODE-AIDS-013
|
|
20
|
+
# @implements REQ-AIDS-013
|
|
21
|
+
# @design DES-AIDS-011
|
|
22
|
+
def get_required_test_command(config_path: str = ".musubix/config.json") -> dict:
|
|
23
|
+
"""Return the required "test" command entry from ``config_path``."""
|
|
24
|
+
config = json.loads(Path(config_path).read_text(encoding="utf-8"))
|
|
25
|
+
for command in config.get("commands", []):
|
|
26
|
+
if command.get("name") == "test":
|
|
27
|
+
if not command.get("required"):
|
|
28
|
+
raise TestCommandMissingError(
|
|
29
|
+
"The 'test' command is declared but not marked required."
|
|
30
|
+
)
|
|
31
|
+
return command
|
|
32
|
+
raise TestCommandMissingError("No required 'test' command declared in config.")
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
"""Data source ingestion.
|
|
2
|
+
|
|
3
|
+
Implements DES-AIDS-005: loads CSV/Excel/database/API sources into an
|
|
4
|
+
in-memory dataframe. A network allowlist is enforced before any remote
|
|
5
|
+
call; a row-count limit is applied to the resulting dataframe after a
|
|
6
|
+
remote (database/API) fetch completes, bounding downstream processing —
|
|
7
|
+
not the fetch itself — and never applies to local CSV/Excel sources
|
|
8
|
+
(REQ-AIDS-014, REQ-AIDS-032; GitHub #57).
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import csv
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from urllib.parse import urlparse
|
|
16
|
+
|
|
17
|
+
import pandas as pd
|
|
18
|
+
|
|
19
|
+
DEFAULT_ROW_LIMIT = 100_000
|
|
20
|
+
_REMOTE_KINDS = ("api", "database")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class NetworkAllowlistError(ValueError):
|
|
24
|
+
"""Raised when a remote source host is not in the configured allowlist."""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True)
|
|
28
|
+
class SourceSpec:
|
|
29
|
+
"""Describes a single ingestion request."""
|
|
30
|
+
|
|
31
|
+
kind: str # "csv" | "excel" | "api" | "database"
|
|
32
|
+
location: str
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True)
|
|
36
|
+
class IngestionResult:
|
|
37
|
+
dataframe: pd.DataFrame
|
|
38
|
+
row_count: int
|
|
39
|
+
column_count: int
|
|
40
|
+
truncated: bool
|
|
41
|
+
warnings: tuple[str, ...] = ()
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
# @id CODE-AIDS-088
|
|
45
|
+
# @implements REQ-AIDS-068
|
|
46
|
+
# @design DES-AIDS-056
|
|
47
|
+
def _sniff_csv_delimiter(path: str) -> tuple[str | None, bool]:
|
|
48
|
+
"""Return ``(delimiter, sniff_succeeded)`` for the CSV file at ``path``.
|
|
49
|
+
|
|
50
|
+
Reads a bounded text sample and attempts ``csv.Sniffer().sniff`` over
|
|
51
|
+
comma/tab/semicolon candidates (DES-AIDS-056); on ``csv.Error`` (an
|
|
52
|
+
ambiguous sample), returns ``(None, False)`` so the caller falls back to
|
|
53
|
+
the existing comma-default behavior.
|
|
54
|
+
"""
|
|
55
|
+
with open(path, encoding="utf-8", errors="replace") as handle:
|
|
56
|
+
sample = handle.read(8192)
|
|
57
|
+
try:
|
|
58
|
+
dialect = csv.Sniffer().sniff(sample, delimiters=",\t;")
|
|
59
|
+
except csv.Error:
|
|
60
|
+
return None, False
|
|
61
|
+
return dialect.delimiter, True
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
# @id CODE-AIDS-014
|
|
65
|
+
# @implements REQ-AIDS-014
|
|
66
|
+
# @design DES-AIDS-005
|
|
67
|
+
# @id CODE-AIDS-032
|
|
68
|
+
# @implements REQ-AIDS-032
|
|
69
|
+
# @design DES-AIDS-005
|
|
70
|
+
def ingest(
|
|
71
|
+
source_spec: SourceSpec,
|
|
72
|
+
fetcher=None,
|
|
73
|
+
allowlist: tuple[str, ...] = (),
|
|
74
|
+
row_limit: int = DEFAULT_ROW_LIMIT,
|
|
75
|
+
) -> IngestionResult:
|
|
76
|
+
"""Load ``source_spec`` into a dataframe, applying ingestion safety rules.
|
|
77
|
+
|
|
78
|
+
For remote ``kind`` values ("api"/"database") the host is checked
|
|
79
|
+
against ``allowlist`` *before* ``fetcher`` is invoked, and the resulting
|
|
80
|
+
dataframe is truncated to ``row_limit`` rows if it exceeds that limit.
|
|
81
|
+
Local ``kind`` values ("csv"/"excel") are never truncated by
|
|
82
|
+
``row_limit``: that limit is a remote-source safety control
|
|
83
|
+
(REQ-AIDS-032), not a general ingestion cap, so a local file is always
|
|
84
|
+
loaded in full (GitHub #57).
|
|
85
|
+
"""
|
|
86
|
+
warnings: tuple[str, ...] = ()
|
|
87
|
+
if source_spec.kind == "csv":
|
|
88
|
+
delimiter, sniffed = _sniff_csv_delimiter(source_spec.location)
|
|
89
|
+
dataframe = pd.read_csv(source_spec.location, sep=delimiter if sniffed else ",")
|
|
90
|
+
if (
|
|
91
|
+
not sniffed
|
|
92
|
+
and len(dataframe.columns) == 1
|
|
93
|
+
and ("\t" in dataframe.columns[0] or ";" in dataframe.columns[0])
|
|
94
|
+
):
|
|
95
|
+
warnings = (
|
|
96
|
+
f"CSV was parsed with the comma fallback as a single column named "
|
|
97
|
+
f"{dataframe.columns[0]!r}; the file may actually use a tab or "
|
|
98
|
+
"semicolon delimiter instead.",
|
|
99
|
+
)
|
|
100
|
+
elif source_spec.kind == "excel":
|
|
101
|
+
dataframe = pd.read_excel(source_spec.location)
|
|
102
|
+
elif source_spec.kind in _REMOTE_KINDS:
|
|
103
|
+
host = urlparse(source_spec.location).hostname
|
|
104
|
+
if host not in allowlist:
|
|
105
|
+
raise NetworkAllowlistError(
|
|
106
|
+
f"Host '{host}' is not in the configured allowlist "
|
|
107
|
+
f"(許可されていない接続先ホストです): {source_spec.location}"
|
|
108
|
+
)
|
|
109
|
+
if fetcher is None:
|
|
110
|
+
raise ValueError("A fetcher callable is required for remote ingestion sources.")
|
|
111
|
+
dataframe = fetcher(source_spec.location)
|
|
112
|
+
else:
|
|
113
|
+
raise ValueError(f"Unsupported ingestion source kind: {source_spec.kind!r}")
|
|
114
|
+
|
|
115
|
+
# GitHub #57: row_limit is a remote-source safety control (REQ-AIDS-032);
|
|
116
|
+
# local csv/excel files must never be silently truncated by it.
|
|
117
|
+
truncated = source_spec.kind in _REMOTE_KINDS and len(dataframe) > row_limit
|
|
118
|
+
if truncated:
|
|
119
|
+
dataframe = dataframe.iloc[:row_limit]
|
|
120
|
+
|
|
121
|
+
return IngestionResult(
|
|
122
|
+
dataframe=dataframe,
|
|
123
|
+
row_count=len(dataframe),
|
|
124
|
+
column_count=len(dataframe.columns),
|
|
125
|
+
truncated=truncated,
|
|
126
|
+
warnings=warnings,
|
|
127
|
+
)
|