jupytermind 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (193) hide show
  1. package/.github/skills/ai-chemistry-scientist/SKILL.md +97 -0
  2. package/.github/skills/ai-chemistry-scientist/manifest.json +156 -0
  3. package/.github/skills/ai-data-scientist/SKILL.md +330 -0
  4. package/.github/skills/ai-genomics-scientist/SKILL.md +98 -0
  5. package/.github/skills/ai-genomics-scientist/manifest.json +93 -0
  6. package/.github/skills/ai-materials-scientist/SKILL.md +51 -0
  7. package/.github/skills/ai-materials-scientist/manifest.json +58 -0
  8. package/.github/skills/ai-scientist/SKILL.md +69 -0
  9. package/.github/skills/ai-scientist/manifest.json +61 -0
  10. package/.github/skills/ai-structural-biology-scientist/SKILL.md +67 -0
  11. package/.github/skills/ai-structural-biology-scientist/manifest.json +72 -0
  12. package/.github/skills/japanese-prose/NOTICE.md +17 -0
  13. package/.github/skills/japanese-prose/SKILL.md +111 -0
  14. package/.github/skills/japanese-prose/references/review-workflow.md +50 -0
  15. package/.github/skills/japanese-prose/references/scoring.md +24 -0
  16. package/.github/skills/japanese-prose/references/writing-guidelines.md +60 -0
  17. package/.github/skills/japanese-prose/scripts/core.py +192 -0
  18. package/.github/skills/japanese-prose/scripts/fixtures/natural.md +5 -0
  19. package/.github/skills/japanese-prose/scripts/fixtures/unnatural.md +5 -0
  20. package/.github/skills/japanese-prose/scripts/lint.py +378 -0
  21. package/.github/skills/japanese-prose/scripts/outline.py +68 -0
  22. package/.github/skills/japanese-prose/scripts/terms.py +112 -0
  23. package/.github/skills/japanese-prose/scripts/test_engine.py +117 -0
  24. package/.github/skills/presentation-planner/SKILL.md +257 -0
  25. package/.github/skills/presentation-planner/assets/design-templates/data-report.yaml +97 -0
  26. package/.github/skills/presentation-planner/assets/design-templates/executive-proposal.yaml +92 -0
  27. package/.github/skills/presentation-planner/assets/design-templates/technical-briefing.yaml +96 -0
  28. package/.github/skills/presentation-planner/assets/scenario-templates/data-report.md +47 -0
  29. package/.github/skills/presentation-planner/assets/scenario-templates/executive-decision.md +43 -0
  30. package/.github/skills/presentation-planner/assets/scenario-templates/technical-briefing.md +45 -0
  31. package/.github/skills/presentation-planner/references/customizing-design-templates.md +160 -0
  32. package/.github/skills/presentation-planner/references/design-spec-schema.md +72 -0
  33. package/.github/skills/presentation-planner/references/handoff-contract.md +49 -0
  34. package/.github/skills/presentation-planner/references/responsibility-boundary.md +32 -0
  35. package/.github/skills/presentation-planner/references/scenario-templates.md +55 -0
  36. package/.github/skills/tech-writer/SKILL.md +434 -0
  37. package/.github/skills/tech-writer/assets/templates/blueprint.md +187 -0
  38. package/.github/skills/tech-writer/assets/templates/design-doc.md +29 -0
  39. package/.github/skills/tech-writer/assets/templates/migration-plan.md +173 -0
  40. package/.github/skills/tech-writer/assets/templates/operations-runbook.md +202 -0
  41. package/.github/skills/tech-writer/assets/templates/pr-description.md +23 -0
  42. package/.github/skills/tech-writer/assets/templates/qiita.md +44 -0
  43. package/.github/skills/tech-writer/assets/templates/readme.md +38 -0
  44. package/.github/skills/tech-writer/assets/templates/requirements-definition.md +170 -0
  45. package/.github/skills/tech-writer/assets/templates/rfi.md +113 -0
  46. package/.github/skills/tech-writer/assets/templates/rfp.md +180 -0
  47. package/.github/skills/tech-writer/assets/templates/security-design.md +167 -0
  48. package/.github/skills/tech-writer/assets/templates/system-design.md +220 -0
  49. package/.github/skills/tech-writer/assets/templates/technical-proposal.md +112 -0
  50. package/.github/skills/tech-writer/assets/templates/test-plan.md +153 -0
  51. package/.github/skills/tech-writer/assets/templates/user-manual.md +22 -0
  52. package/.github/skills/tech-writer/assets/templates/white-paper.md +192 -0
  53. package/.github/skills/tech-writer/references/doctypes/api-docs.md +33 -0
  54. package/.github/skills/tech-writer/references/doctypes/blueprint.md +81 -0
  55. package/.github/skills/tech-writer/references/doctypes/code-comments.md +39 -0
  56. package/.github/skills/tech-writer/references/doctypes/design-doc.md +42 -0
  57. package/.github/skills/tech-writer/references/doctypes/migration-plan.md +63 -0
  58. package/.github/skills/tech-writer/references/doctypes/operations-runbook.md +63 -0
  59. package/.github/skills/tech-writer/references/doctypes/pr-commit.md +82 -0
  60. package/.github/skills/tech-writer/references/doctypes/qiita.md +75 -0
  61. package/.github/skills/tech-writer/references/doctypes/readme.md +43 -0
  62. package/.github/skills/tech-writer/references/doctypes/release-notes.md +30 -0
  63. package/.github/skills/tech-writer/references/doctypes/requirements-definition.md +61 -0
  64. package/.github/skills/tech-writer/references/doctypes/rfi.md +43 -0
  65. package/.github/skills/tech-writer/references/doctypes/rfp.md +46 -0
  66. package/.github/skills/tech-writer/references/doctypes/security-design.md +71 -0
  67. package/.github/skills/tech-writer/references/doctypes/system-design.md +74 -0
  68. package/.github/skills/tech-writer/references/doctypes/technical-proposal.md +49 -0
  69. package/.github/skills/tech-writer/references/doctypes/test-plan.md +67 -0
  70. package/.github/skills/tech-writer/references/doctypes/user-manual.md +58 -0
  71. package/.github/skills/tech-writer/references/doctypes/white-paper.md +84 -0
  72. package/.github/skills/tech-writer/references/doctypes/zenn.md +66 -0
  73. package/.github/skills/tech-writer/references/japanese-prose-optimization.md +110 -0
  74. package/.github/skills/tech-writer/references/style-constitution.md +104 -0
  75. package/.github/skills/tech-writer/scripts/lint.py +412 -0
  76. package/LICENSE +21 -0
  77. package/README.md +92 -0
  78. package/bin/ai-data-scientist.js +123 -0
  79. package/package.json +41 -0
  80. package/pyproject.toml +45 -0
  81. package/src/ai_chemistry_scientist/__init__.py +0 -0
  82. package/src/ai_chemistry_scientist/admet_prediction.py +71 -0
  83. package/src/ai_chemistry_scientist/bioactivity_classification.py +73 -0
  84. package/src/ai_chemistry_scientist/data/sample_molecules.csv +21 -0
  85. package/src/ai_chemistry_scientist/dispatch.py +369 -0
  86. package/src/ai_chemistry_scientist/docking_score.py +97 -0
  87. package/src/ai_chemistry_scientist/drug_likeness_rules.py +84 -0
  88. package/src/ai_chemistry_scientist/evidence.py +41 -0
  89. package/src/ai_chemistry_scientist/molecular_descriptors.py +97 -0
  90. package/src/ai_chemistry_scientist/molecular_formula_mass.py +40 -0
  91. package/src/ai_chemistry_scientist/molecular_similarity.py +78 -0
  92. package/src/ai_chemistry_scientist/qsar_modeling.py +105 -0
  93. package/src/ai_chemistry_scientist/salt_standardization.py +81 -0
  94. package/src/ai_chemistry_scientist/structural_alerts.py +76 -0
  95. package/src/ai_chemistry_scientist/structure_format_conversion.py +84 -0
  96. package/src/ai_chemistry_scientist/validation.py +70 -0
  97. package/src/ai_data_scientist/__init__.py +0 -0
  98. package/src/ai_data_scientist/analysis_assumptions.py +121 -0
  99. package/src/ai_data_scientist/anomaly_detection.py +39 -0
  100. package/src/ai_data_scientist/automl.py +109 -0
  101. package/src/ai_data_scientist/cleaning.py +56 -0
  102. package/src/ai_data_scientist/cli.py +90 -0
  103. package/src/ai_data_scientist/clustering.py +54 -0
  104. package/src/ai_data_scientist/dashboard.py +33 -0
  105. package/src/ai_data_scientist/data_definition.py +100 -0
  106. package/src/ai_data_scientist/data_quality.py +164 -0
  107. package/src/ai_data_scientist/dataset_validation.py +135 -0
  108. package/src/ai_data_scientist/dependency_pins.py +60 -0
  109. package/src/ai_data_scientist/eda.py +82 -0
  110. package/src/ai_data_scientist/experiment_evaluation.py +635 -0
  111. package/src/ai_data_scientist/explainability.py +340 -0
  112. package/src/ai_data_scientist/feature_engineering.py +163 -0
  113. package/src/ai_data_scientist/gate_config.py +32 -0
  114. package/src/ai_data_scientist/ingestion.py +127 -0
  115. package/src/ai_data_scientist/insight_engine.py +180 -0
  116. package/src/ai_data_scientist/japanese_nlp.py +43 -0
  117. package/src/ai_data_scientist/jupyter_launcher.py +137 -0
  118. package/src/ai_data_scientist/jupyter_mcp_client.py +94 -0
  119. package/src/ai_data_scientist/language_router.py +28 -0
  120. package/src/ai_data_scientist/lifecycle.py +221 -0
  121. package/src/ai_data_scientist/mcp_gateway.py +113 -0
  122. package/src/ai_data_scientist/mcp_runtime.py +194 -0
  123. package/src/ai_data_scientist/mcp_transport.py +53 -0
  124. package/src/ai_data_scientist/ml_modeling.py +451 -0
  125. package/src/ai_data_scientist/model_tuning.py +104 -0
  126. package/src/ai_data_scientist/notebook_audit.py +574 -0
  127. package/src/ai_data_scientist/project_manager.py +243 -0
  128. package/src/ai_data_scientist/report_export.py +73 -0
  129. package/src/ai_data_scientist/sensitivity.py +445 -0
  130. package/src/ai_data_scientist/signal_analysis.py +201 -0
  131. package/src/ai_data_scientist/skill_packaging.py +40 -0
  132. package/src/ai_data_scientist/stats_analysis.py +88 -0
  133. package/src/ai_data_scientist/text_nlp.py +44 -0
  134. package/src/ai_data_scientist/timeseries.py +68 -0
  135. package/src/ai_data_scientist/visualization.py +708 -0
  136. package/src/ai_genomics_scientist/__init__.py +1 -0
  137. package/src/ai_genomics_scientist/differential_expression.py +147 -0
  138. package/src/ai_genomics_scientist/dispatch.py +267 -0
  139. package/src/ai_genomics_scientist/evidence.py +45 -0
  140. package/src/ai_genomics_scientist/gene_set_enrichment.py +76 -0
  141. package/src/ai_genomics_scientist/sequence_alignment.py +97 -0
  142. package/src/ai_genomics_scientist/sequence_features.py +111 -0
  143. package/src/ai_genomics_scientist/splice_site_scoring.py +66 -0
  144. package/src/ai_genomics_scientist/validation.py +83 -0
  145. package/src/ai_genomics_scientist/variant_effect.py +147 -0
  146. package/src/ai_genomics_scientist/variant_pathogenicity.py +125 -0
  147. package/src/ai_materials_scientist/__init__.py +0 -0
  148. package/src/ai_materials_scientist/calphad.py +117 -0
  149. package/src/ai_materials_scientist/classical_monte_carlo.py +165 -0
  150. package/src/ai_materials_scientist/crystal_plasticity.py +184 -0
  151. package/src/ai_materials_scientist/dispatch.py +100 -0
  152. package/src/ai_materials_scientist/evidence.py +84 -0
  153. package/src/ai_materials_scientist/fem.py +279 -0
  154. package/src/ai_materials_scientist/kinetic_monte_carlo.py +145 -0
  155. package/src/ai_materials_scientist/molecular_dynamics.py +240 -0
  156. package/src/ai_materials_scientist/phase_field.py +167 -0
  157. package/src/ai_materials_scientist/validation.py +70 -0
  158. package/src/ai_scientist/__init__.py +1 -0
  159. package/src/ai_scientist/completion_gate.py +15 -0
  160. package/src/ai_scientist/data_analysis.py +46 -0
  161. package/src/ai_scientist/evidence_registry.py +99 -0
  162. package/src/ai_scientist/experimental_design.py +20 -0
  163. package/src/ai_scientist/language.py +14 -0
  164. package/src/ai_scientist/latex_renderer.py +41 -0
  165. package/src/ai_scientist/literature_review.py +37 -0
  166. package/src/ai_scientist/manifest.py +87 -0
  167. package/src/ai_scientist/manuscript.py +94 -0
  168. package/src/ai_scientist/mcp_config.py +76 -0
  169. package/src/ai_scientist/mcp_external.py +42 -0
  170. package/src/ai_scientist/mcp_failures.py +23 -0
  171. package/src/ai_scientist/mcp_gateway.py +38 -0
  172. package/src/ai_scientist/mcp_managed.py +180 -0
  173. package/src/ai_scientist/npm_packaging.py +49 -0
  174. package/src/ai_scientist/orchestrator.py +133 -0
  175. package/src/ai_scientist/peer_review.py +60 -0
  176. package/src/ai_scientist/phase_gate.py +74 -0
  177. package/src/ai_scientist/phase_state.py +230 -0
  178. package/src/ai_scientist/presentation.py +56 -0
  179. package/src/ai_scientist/project_config.py +31 -0
  180. package/src/ai_scientist/project_handle.py +74 -0
  181. package/src/ai_scientist/reproducibility.py +20 -0
  182. package/src/ai_scientist/research_planning.py +20 -0
  183. package/src/ai_scientist/skill_invocation.py +21 -0
  184. package/src/ai_scientist/tdd_gate.py +99 -0
  185. package/src/ai_structural_biology_scientist/__init__.py +0 -0
  186. package/src/ai_structural_biology_scientist/contact_map.py +87 -0
  187. package/src/ai_structural_biology_scientist/dispatch.py +269 -0
  188. package/src/ai_structural_biology_scientist/evidence.py +43 -0
  189. package/src/ai_structural_biology_scientist/hydrophobicity.py +101 -0
  190. package/src/ai_structural_biology_scientist/protein_docking_score.py +104 -0
  191. package/src/ai_structural_biology_scientist/secondary_structure.py +95 -0
  192. package/src/ai_structural_biology_scientist/structural_similarity.py +74 -0
  193. package/src/ai_structural_biology_scientist/validation.py +100 -0
@@ -0,0 +1,84 @@
1
+ """Chemical structure format conversion module (DES-ACHEM-110 / REQ-ACHEM-110)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from rdkit import Chem
6
+ from rdkit.Chem import inchi as rdkit_inchi
7
+
8
+ from ai_chemistry_scientist.validation import fail, ok, register_validator
9
+
10
+ _MODULE_NAME = "structure-format-conversion"
11
+
12
+ INPUT_FORMATS = ("smiles", "inchi", "molblock")
13
+ OUTPUT_FORMATS = ("smiles", "inchi", "inchikey", "molblock")
14
+
15
+ _PARSERS = {
16
+ "smiles": Chem.MolFromSmiles,
17
+ "inchi": rdkit_inchi.MolFromInchi,
18
+ "molblock": Chem.MolFromMolBlock,
19
+ }
20
+ _WRITERS = {
21
+ "smiles": Chem.MolToSmiles,
22
+ "inchi": rdkit_inchi.MolToInchi,
23
+ "inchikey": rdkit_inchi.MolToInchiKey,
24
+ "molblock": Chem.MolToMolBlock,
25
+ }
26
+
27
+
28
+ def _parse_structure(input_format: str, input_value):
29
+ """Parse ``input_value`` with the ``input_format``-matching RDKit parser.
30
+
31
+ Rejects a non-str, empty, or unparseable value; a value that parses to a
32
+ zero-atom molecule (e.g. an empty SMILES or a ``0 0`` atom/bond-count
33
+ Molblock, which RDKit parses without error); and (same chemical-validity
34
+ domain as REQ-ACHEM-100 and every other module in this skill) any value
35
+ that parses but contains a dummy/query atom (RDKit atomic number 0).
36
+ """
37
+ if not isinstance(input_value, str) or not input_value:
38
+ return None
39
+ mol = _PARSERS[input_format](input_value)
40
+ if mol is None or mol.GetNumAtoms() == 0:
41
+ return None
42
+ if any(atom.GetAtomicNum() == 0 for atom in mol.GetAtoms()):
43
+ return None
44
+ return mol
45
+
46
+
47
+ # @id CODE-ACHEM-922
48
+ # @implements REQ-ACHEM-003 REQ-ACHEM-110
49
+ # @design DES-ACHEM-002
50
+ def _structure_format_conversion_validator(params: dict) -> dict:
51
+ for name in ("input_format", "output_format", "input_value"):
52
+ if name not in params:
53
+ return fail(name, "is required")
54
+ input_format = params["input_format"]
55
+ output_format = params["output_format"]
56
+ if input_format not in INPUT_FORMATS:
57
+ return fail("input_format", "must be one of the supported formats")
58
+ if output_format not in OUTPUT_FORMATS:
59
+ return fail("output_format", "must be one of the supported formats")
60
+ if _parse_structure(input_format, params["input_value"]) is None:
61
+ return fail(
62
+ "input_value",
63
+ f"must parse with the {input_format}-matching RDKit parser",
64
+ )
65
+ return ok()
66
+
67
+
68
+ register_validator(_MODULE_NAME, _structure_format_conversion_validator)
69
+
70
+
71
+ # @id CODE-ACHEM-110
72
+ # @implements REQ-ACHEM-110
73
+ # @design DES-ACHEM-110
74
+ def run_structure_conversion(input_format: str, input_value: str, output_format: str) -> dict:
75
+ """Parse ``input_value`` per ``input_format`` and render per ``output_format``."""
76
+ mol = _parse_structure(input_format, input_value)
77
+ if mol is None or mol.GetNumAtoms() == 0:
78
+ raise ValueError("input_value must already be validated by the handler wrapper")
79
+
80
+ output_value = _WRITERS[output_format](mol)
81
+ return {
82
+ "output_format": output_format,
83
+ "output_value": output_value,
84
+ }
@@ -0,0 +1,70 @@
1
+ """Shared parameter & chemical-validity validator (DES-ACHEM-002 / REQ-ACHEM-003)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable
6
+ from typing import Any
7
+
8
+ ValidationResult = dict
9
+ ValidatorFn = Callable[[dict[str, Any]], ValidationResult]
10
+ BatchItemValidatorFn = Callable[[dict[str, Any]], ValidationResult]
11
+
12
+ _REGISTRY: dict[str, ValidatorFn] = {}
13
+ _BATCH_ITEM_REGISTRY: dict[str, BatchItemValidatorFn] = {}
14
+
15
+
16
+ def ok() -> ValidationResult:
17
+ return {"ok": True}
18
+
19
+
20
+ def fail(parameter: str, constraint: str) -> ValidationResult:
21
+ return {"ok": False, "parameter": parameter, "constraint": constraint}
22
+
23
+
24
+ # @id CODE-ACHEM-002
25
+ # @implements REQ-ACHEM-003
26
+ # @design DES-ACHEM-002
27
+ def register_validator(module_name: str, validator: ValidatorFn) -> None:
28
+ """Register ``module_name``'s own documented atomic-validity validator."""
29
+ _REGISTRY[module_name] = validator
30
+
31
+
32
+ # @id CODE-ACHEM-913
33
+ # @implements REQ-ACHEM-003
34
+ # @design DES-ACHEM-002
35
+ def register_batch_item_validator(module_name: str, validator: BatchItemValidatorFn) -> None:
36
+ """Register ``module_name``'s own documented per-item validator."""
37
+ _BATCH_ITEM_REGISTRY[module_name] = validator
38
+
39
+
40
+ # @id CODE-ACHEM-914
41
+ # @implements REQ-ACHEM-003
42
+ # @design DES-ACHEM-002
43
+ def validate_parameters(module_name: str, params: dict[str, Any]) -> ValidationResult:
44
+ """Dispatch to ``module_name``'s registered atomic validator.
45
+
46
+ Must run to completion before any descriptor computation, model fit, or
47
+ similarity/score calculation for the module's whole run (REQ-ACHEM-003).
48
+ """
49
+ validator = _REGISTRY.get(module_name)
50
+ if validator is None:
51
+ return fail("module", f"no validator registered for module '{module_name}'")
52
+ if not isinstance(params, dict):
53
+ return fail("params", "must be a dict")
54
+ return validator(params)
55
+
56
+
57
+ # @id CODE-ACHEM-915
58
+ # @implements REQ-ACHEM-003
59
+ # @design DES-ACHEM-002
60
+ def validate_batch_item(module_name: str, item_params: dict[str, Any]) -> ValidationResult:
61
+ """Dispatch to ``module_name``'s registered per-item validator.
62
+
63
+ Invoked once per batch item (REQ-ACHEM-010's per-item granularity); an
64
+ invalid item is rejected without aborting computation of the rest of the
65
+ batch.
66
+ """
67
+ validator = _BATCH_ITEM_REGISTRY.get(module_name)
68
+ if validator is None:
69
+ return fail("module", f"no batch-item validator registered for module '{module_name}'")
70
+ return validator(item_params)
File without changes
@@ -0,0 +1,121 @@
1
+ """Analysis-assumption and applicability manifest.
2
+
3
+ Implements DES-AIDS-042 (REQ-AIDS-054): records conclusion-critical
4
+ analytical choices (preprocessing, sampling, causal scope) with an
5
+ explicit status, so a written caveat is never mistaken for a verified
6
+ check, and surfaces unresolved risk before a conclusion is finalized.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from dataclasses import dataclass, field
12
+
13
+ _VALID_ASSUMPTION_STATUSES = frozenset({"verified", "tested", "assumed", "rejected"})
14
+ _VALID_CAUSAL_SCOPES = frozenset({"descriptive", "associational", "causal"})
15
+ _TESTED_OR_VERIFIED = frozenset({"tested", "verified"})
16
+
17
+
18
+ # @id CODE-AIDS-074
19
+ # @implements REQ-AIDS-054
20
+ # @design DES-AIDS-042
21
+ @dataclass(frozen=True)
22
+ class Assumption:
23
+ """A single analytical assumption and its verification status."""
24
+
25
+ id: str
26
+ statement: str
27
+ status: str
28
+ evidence_cell: int | None = None
29
+ impact_if_false: str | None = None
30
+ conclusion_critical: bool = False
31
+
32
+ def __post_init__(self) -> None:
33
+ if self.status not in _VALID_ASSUMPTION_STATUSES:
34
+ raise ValueError(
35
+ f"status must be one of {sorted(_VALID_ASSUMPTION_STATUSES)}, got {self.status!r}."
36
+ )
37
+
38
+
39
+ @dataclass(frozen=True)
40
+ class AssumptionFinding:
41
+ """A single applicability-check observation."""
42
+
43
+ code: str
44
+ severity: str # "error" | "warning"
45
+ message: str
46
+ assumption_id: str | None = None
47
+
48
+
49
+ @dataclass(frozen=True)
50
+ class AnalysisAssumptionManifest:
51
+ """Scope, assumptions, and causal classification for one analysis."""
52
+
53
+ analysis_scope: dict
54
+ assumptions: tuple[Assumption, ...] = field(default_factory=tuple)
55
+ causal_scope: str = "descriptive"
56
+ sampling: dict | None = None
57
+
58
+ def __post_init__(self) -> None:
59
+ if self.causal_scope not in _VALID_CAUSAL_SCOPES:
60
+ raise ValueError(
61
+ f"causal_scope must be one of {sorted(_VALID_CAUSAL_SCOPES)}, "
62
+ f"got {self.causal_scope!r}."
63
+ )
64
+
65
+ def unresolved_risks(self) -> tuple[Assumption, ...]:
66
+ """Conclusion-critical assumptions whose status is assumed or rejected."""
67
+ return tuple(
68
+ assumption
69
+ for assumption in self.assumptions
70
+ if assumption.conclusion_critical and assumption.status in ("assumed", "rejected")
71
+ )
72
+
73
+
74
+ # @id CODE-AIDS-075
75
+ # @implements REQ-AIDS-054
76
+ # @design DES-AIDS-042
77
+ def check_manifest(manifest: AnalysisAssumptionManifest) -> tuple[AssumptionFinding, ...]:
78
+ """Validate ``manifest``, returning one finding per detected gap."""
79
+ findings: list[AssumptionFinding] = []
80
+
81
+ if manifest.causal_scope == "causal":
82
+ has_identification = any(
83
+ assumption.status in _TESTED_OR_VERIFIED for assumption in manifest.assumptions
84
+ )
85
+ if not has_identification:
86
+ findings.append(
87
+ AssumptionFinding(
88
+ code="missing_causal_identification",
89
+ severity="error",
90
+ message=(
91
+ "causal_scope is 'causal' but no assumption has status "
92
+ "'tested' or 'verified' to support identification."
93
+ ),
94
+ )
95
+ )
96
+
97
+ for assumption in manifest.unresolved_risks():
98
+ findings.append(
99
+ AssumptionFinding(
100
+ code="unresolved_conclusion_critical_assumption",
101
+ severity="warning",
102
+ message=(
103
+ f"Conclusion-critical assumption {assumption.id!r} has status "
104
+ f"{assumption.status!r}, not verified/tested."
105
+ ),
106
+ assumption_id=assumption.id,
107
+ )
108
+ )
109
+
110
+ if manifest.sampling is not None:
111
+ missing = {"n", "seed"} - manifest.sampling.keys()
112
+ if missing:
113
+ findings.append(
114
+ AssumptionFinding(
115
+ code="incomplete_sampling_record",
116
+ severity="error",
117
+ message=f"sampling is missing required keys: {sorted(missing)}.",
118
+ )
119
+ )
120
+
121
+ return tuple(findings)
@@ -0,0 +1,39 @@
1
+ """Anomaly / outlier detection.
2
+
3
+ Implements DES-AIDS-015 (REQ-AIDS-017): flags anomalous records in a
4
+ dataframe using the requested detection method and reports the count of
5
+ flagged records.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass
11
+
12
+ import pandas as pd
13
+
14
+ _SUPPORTED_METHODS = ("zscore",)
15
+
16
+
17
+ @dataclass(frozen=True)
18
+ class AnomalyResult:
19
+ flagged_indices: list
20
+ method: str
21
+
22
+
23
+ # @id CODE-AIDS-017
24
+ # @implements REQ-AIDS-017
25
+ # @design DES-AIDS-015
26
+ def detect_anomalies(
27
+ df: pd.DataFrame, column: str, method: str = "zscore", params: dict | None = None
28
+ ) -> AnomalyResult:
29
+ """Flag anomalous rows of ``df[column]`` using ``method``."""
30
+ if method not in _SUPPORTED_METHODS:
31
+ raise ValueError(f"Unsupported anomaly detection method: {method!r}")
32
+ params = params or {}
33
+ threshold = params.get("threshold", 3.0)
34
+
35
+ series = df[column]
36
+ z_scores = (series - series.mean()) / series.std(ddof=0)
37
+ flagged = series.index[z_scores.abs() > threshold].tolist()
38
+
39
+ return AnomalyResult(flagged_indices=flagged, method=method)
@@ -0,0 +1,109 @@
1
+ """Automated model selection (AutoML).
2
+
3
+ Implements DES-AIDS-018 (REQ-AIDS-020): trains multiple candidate model
4
+ types via the supervised modeling and tuning interfaces and reports a
5
+ ranked comparison of their evaluation metrics.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from collections.abc import Callable, Sequence
11
+ from dataclasses import dataclass
12
+ from typing import Any
13
+
14
+ import pandas as pd
15
+
16
+ from ai_data_scientist.ml_modeling import (
17
+ _DEFAULT_SCORING,
18
+ MODEL_BUILDERS,
19
+ _metric_direction,
20
+ build_cv_splits,
21
+ train_model,
22
+ )
23
+
24
+
25
+ @dataclass(frozen=True)
26
+ class AutoMLResult:
27
+ ranked_candidates: list
28
+ scoring: str | None = None
29
+ cv_splits: list[tuple[list, list]] | None = None
30
+
31
+
32
+ # @id CODE-AIDS-124
33
+ # @implements REQ-AIDS-078
34
+ # @design DES-AIDS-065 DES-AIDS-066
35
+ def build_candidate_estimators(
36
+ model_type: str, candidate_estimators: dict[str, object | Callable[..., object]] | None = None
37
+ ) -> dict[str, object | Callable[..., object] | None]:
38
+ """Build the AutoML candidate registry."""
39
+ candidates: dict[str, object | Callable[..., object] | None] = {
40
+ name: None for name in MODEL_BUILDERS[model_type]
41
+ }
42
+ if candidate_estimators:
43
+ candidates.update(candidate_estimators)
44
+ return candidates
45
+
46
+
47
+ # @id CODE-AIDS-020
48
+ # @implements REQ-AIDS-020
49
+ # @design DES-AIDS-018
50
+ def run_automl(
51
+ df: pd.DataFrame,
52
+ target: str,
53
+ model_type: str = "classification",
54
+ scoring: str | None = None,
55
+ cv_strategy: str | None = None,
56
+ n_splits: int = 5,
57
+ random_state: int = 42,
58
+ groups: str | Sequence[Any] | pd.Series | None = None,
59
+ cv_splits: Sequence[tuple[Sequence[Any], Sequence[Any]]] | None = None,
60
+ candidate_estimators: dict[str, object | Callable[..., object]] | None = None,
61
+ ) -> AutoMLResult:
62
+ """Train several candidate model types and rank them by metric."""
63
+ metric_name = scoring or _DEFAULT_SCORING[model_type]
64
+
65
+ # @id CODE-AIDS-100
66
+ # @implements REQ-AIDS-077 REQ-AIDS-078
67
+ # @design DES-AIDS-065
68
+ shared_cv_splits = build_cv_splits(
69
+ df=df,
70
+ target=target,
71
+ model_type=model_type,
72
+ cv_strategy=cv_strategy,
73
+ n_splits=n_splits,
74
+ random_state=random_state,
75
+ groups=groups,
76
+ cv_splits=cv_splits,
77
+ )
78
+ candidates = []
79
+ for model_name, estimator in build_candidate_estimators(
80
+ model_type, candidate_estimators
81
+ ).items():
82
+ result = train_model(
83
+ df,
84
+ target=target,
85
+ model_type=model_type,
86
+ model_name=model_name if estimator is None else next(iter(MODEL_BUILDERS[model_type])),
87
+ scoring=scoring,
88
+ cv_strategy=cv_strategy,
89
+ n_splits=n_splits,
90
+ random_state=random_state,
91
+ groups=groups,
92
+ cv_splits=shared_cv_splits,
93
+ estimator=estimator,
94
+ )
95
+ candidates.append(
96
+ {
97
+ "model_name": model_name,
98
+ "metric": result.metrics[metric_name],
99
+ "fold_scores": result.fold_scores,
100
+ "result": result,
101
+ }
102
+ )
103
+
104
+ ranked = sorted(
105
+ candidates,
106
+ key=lambda candidate: candidate["metric"],
107
+ reverse=_metric_direction(model_type, scoring),
108
+ )
109
+ return AutoMLResult(ranked_candidates=ranked, scoring=metric_name, cv_splits=shared_cv_splits)
@@ -0,0 +1,56 @@
1
+ """Data cleaning operations.
2
+
3
+ Implements DES-AIDS-006 (REQ-AIDS-004): performs a requested cleaning
4
+ operation and reports the row/column impact.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from dataclasses import dataclass
10
+
11
+ import pandas as pd
12
+
13
+ _SUPPORTED_OPERATIONS = ("drop_duplicates", "drop_na", "fillna")
14
+
15
+
16
+ @dataclass(frozen=True)
17
+ class CleaningReport:
18
+ dataframe: pd.DataFrame
19
+ rows_before: int
20
+ rows_after: int
21
+ rows_removed: int
22
+ columns_affected: list
23
+
24
+
25
+ # @id CODE-AIDS-004
26
+ # @implements REQ-AIDS-004
27
+ # @design DES-AIDS-006
28
+ def clean_dataset(
29
+ df: pd.DataFrame, operation: str, columns: list | None = None, fill_value=None
30
+ ) -> CleaningReport:
31
+ """Apply ``operation`` to ``df`` and report its row/column impact."""
32
+ if operation not in _SUPPORTED_OPERATIONS:
33
+ raise ValueError(f"Unsupported cleaning operation: {operation!r}")
34
+
35
+ rows_before = len(df)
36
+ target_columns = columns or list(df.columns)
37
+
38
+ if operation == "drop_duplicates":
39
+ cleaned = df.drop_duplicates()
40
+ columns_affected = list(df.columns)
41
+ elif operation == "drop_na":
42
+ cleaned = df.dropna(subset=target_columns)
43
+ columns_affected = target_columns
44
+ else: # fillna
45
+ cleaned = df.copy()
46
+ cleaned[target_columns] = cleaned[target_columns].fillna(fill_value)
47
+ columns_affected = target_columns
48
+
49
+ rows_after = len(cleaned)
50
+ return CleaningReport(
51
+ dataframe=cleaned,
52
+ rows_before=rows_before,
53
+ rows_after=rows_after,
54
+ rows_removed=rows_before - rows_after,
55
+ columns_affected=columns_affected,
56
+ )
@@ -0,0 +1,90 @@
1
+ """Command-line entrypoint for the ai-data-scientist skill.
2
+
3
+ This is a thin diagnostic/bootstrap CLI invoked via the npm wrapper
4
+ (`bin/ai-data-scientist.js`); it is not itself part of the SDD requirement
5
+ set. Copilot invokes the skill's Python modules directly per
6
+ `.github/skills/ai-data-scientist/SKILL.md`, not through this CLI.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import argparse
12
+ import importlib
13
+
14
+ _MODULES = [
15
+ "ai_data_scientist.language_router",
16
+ "ai_data_scientist.project_manager",
17
+ "ai_data_scientist.mcp_gateway",
18
+ "ai_data_scientist.ingestion",
19
+ "ai_data_scientist.cleaning",
20
+ "ai_data_scientist.eda",
21
+ "ai_data_scientist.stats_analysis",
22
+ "ai_data_scientist.visualization",
23
+ "ai_data_scientist.insight_engine",
24
+ "ai_data_scientist.gate_config",
25
+ "ai_data_scientist.skill_packaging",
26
+ ]
27
+
28
+
29
+ def doctor() -> int:
30
+ """Import every skill module to confirm the environment is ready."""
31
+ failures = []
32
+ for module_name in _MODULES:
33
+ try:
34
+ importlib.import_module(module_name)
35
+ except Exception as exc: # noqa: BLE001 - report every import failure
36
+ failures.append(f"{module_name}: {exc}")
37
+
38
+ if failures:
39
+ print("NG: 以下のモジュールを読み込めませんでした / failed to import:")
40
+ for failure in failures:
41
+ print(f" - {failure}")
42
+ return 1
43
+
44
+ print(f"OK: {len(_MODULES)} モジュールを正常に読み込みました / modules import cleanly.")
45
+ return 0
46
+
47
+
48
+ def validate_notebook(path: str) -> int:
49
+ """Audit a notebook (read-only) and print findings; exit 1 if any error finding."""
50
+ from ai_data_scientist.notebook_audit import audit_notebook
51
+
52
+ report = audit_notebook(path)
53
+ print(f"Notebook: {report.path}")
54
+ print(f"nbformat_valid={report.nbformat_valid} ok={report.ok}")
55
+ print(
56
+ f"code_cells={report.code_cell_count} executed={report.executed_code_cell_count} "
57
+ f"insight_cells={report.insight_cell_count}"
58
+ )
59
+ if not report.findings:
60
+ print("OK: 問題は見つかりませんでした / no issues found.")
61
+ return 0
62
+
63
+ print("findings:")
64
+ for finding in report.findings:
65
+ location = f"cell[{finding.cell_index}]" if finding.cell_index is not None else "-"
66
+ print(f" - [{finding.severity}] {location}: {finding.message}")
67
+ return 0 if report.ok else 1
68
+
69
+
70
+ def main(argv: list[str] | None = None) -> int:
71
+ parser = argparse.ArgumentParser(prog="ai-data-scientist")
72
+ subparsers = parser.add_subparsers(dest="command")
73
+ subparsers.add_parser("doctor", help="Verify the Python environment can import every module.")
74
+ validate_parser = subparsers.add_parser(
75
+ "validate-notebook", help="Audit a notebook's execution/evidence completeness (read-only)."
76
+ )
77
+ validate_parser.add_argument("path", help="Path to the .ipynb file to audit.")
78
+ args = parser.parse_args(argv)
79
+
80
+ if args.command in (None, "doctor"):
81
+ return doctor()
82
+ if args.command == "validate-notebook":
83
+ return validate_notebook(args.path)
84
+
85
+ parser.error(f"Unknown command: {args.command}")
86
+ return 2
87
+
88
+
89
+ if __name__ == "__main__":
90
+ raise SystemExit(main())
@@ -0,0 +1,54 @@
1
+ """Clustering and dimensionality reduction.
2
+
3
+ Implements DES-AIDS-014 (REQ-AIDS-016): fits the requested unsupervised
4
+ model (clustering or dimensionality reduction) and reports cluster
5
+ assignments or reduced component values.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass
11
+
12
+ import pandas as pd
13
+ from sklearn.cluster import KMeans
14
+ from sklearn.decomposition import PCA
15
+
16
+ _SUPPORTED_METHODS = ("kmeans", "pca")
17
+
18
+
19
+ @dataclass(frozen=True)
20
+ class UnsupervisedResult:
21
+ labels_or_components: object
22
+ method: str
23
+ params: dict
24
+
25
+
26
+ # @id CODE-AIDS-016
27
+ # @implements REQ-AIDS-016
28
+ # @design DES-AIDS-014
29
+ def cluster_or_reduce(
30
+ df: pd.DataFrame, method: str = "kmeans", params: dict | None = None
31
+ ) -> UnsupervisedResult:
32
+ """Fit ``method`` (e.g. kmeans, pca) on ``df`` and report the result."""
33
+ if method not in _SUPPORTED_METHODS:
34
+ raise ValueError(f"Unsupported unsupervised method: {method!r}")
35
+ params = dict(params or {})
36
+
37
+ if method == "kmeans":
38
+ params.setdefault("n_clusters", 2)
39
+ params.setdefault("n_init", 10)
40
+ if params["n_clusters"] > len(df):
41
+ raise ValueError(
42
+ f"n_clusters ({params['n_clusters']}) must not exceed the "
43
+ f"number of rows ({len(df)})"
44
+ )
45
+ model = KMeans(**params)
46
+ labels_or_components = model.fit_predict(df).tolist()
47
+ else: # pca
48
+ params.setdefault("n_components", min(2, df.shape[1]))
49
+ model = PCA(**params)
50
+ labels_or_components = model.fit_transform(df).tolist()
51
+
52
+ return UnsupervisedResult(
53
+ labels_or_components=labels_or_components, method=method, params=params
54
+ )
@@ -0,0 +1,33 @@
1
+ """Interactive dashboard rendering.
2
+
3
+ Implements DES-AIDS-023 (REQ-AIDS-024): renders an interactive widget or
4
+ chart embedded in the notebook output, extending the MVP's static chart
5
+ rendering with an interactive HTML MIME bundle.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import nbformat
11
+ import pandas as pd
12
+ import plotly.express as px
13
+
14
+ _SUPPORTED_KINDS = ("scatter", "line", "bar")
15
+
16
+
17
+ # @id CODE-AIDS-024
18
+ # @implements REQ-AIDS-024
19
+ # @design DES-AIDS-023
20
+ def render_dashboard(df: pd.DataFrame, spec: dict) -> nbformat.NotebookNode:
21
+ """Render ``df`` per ``spec`` as an interactive HTML output bundle."""
22
+ kind = spec.get("kind", "scatter")
23
+ if kind not in _SUPPORTED_KINDS:
24
+ raise ValueError(f"Unsupported dashboard chart kind: {kind!r}")
25
+
26
+ plot_fn = {"scatter": px.scatter, "line": px.line, "bar": px.bar}[kind]
27
+ figure = plot_fn(df, x=spec.get("x"), y=spec.get("y"))
28
+ html = figure.to_html(include_plotlyjs="cdn", full_html=False)
29
+
30
+ return nbformat.v4.new_output(
31
+ "display_data",
32
+ data={"text/html": html, "text/plain": "<interactive plotly dashboard>"},
33
+ )