jupytermind 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (193) hide show
  1. package/.github/skills/ai-chemistry-scientist/SKILL.md +97 -0
  2. package/.github/skills/ai-chemistry-scientist/manifest.json +156 -0
  3. package/.github/skills/ai-data-scientist/SKILL.md +330 -0
  4. package/.github/skills/ai-genomics-scientist/SKILL.md +98 -0
  5. package/.github/skills/ai-genomics-scientist/manifest.json +93 -0
  6. package/.github/skills/ai-materials-scientist/SKILL.md +51 -0
  7. package/.github/skills/ai-materials-scientist/manifest.json +58 -0
  8. package/.github/skills/ai-scientist/SKILL.md +69 -0
  9. package/.github/skills/ai-scientist/manifest.json +61 -0
  10. package/.github/skills/ai-structural-biology-scientist/SKILL.md +67 -0
  11. package/.github/skills/ai-structural-biology-scientist/manifest.json +72 -0
  12. package/.github/skills/japanese-prose/NOTICE.md +17 -0
  13. package/.github/skills/japanese-prose/SKILL.md +111 -0
  14. package/.github/skills/japanese-prose/references/review-workflow.md +50 -0
  15. package/.github/skills/japanese-prose/references/scoring.md +24 -0
  16. package/.github/skills/japanese-prose/references/writing-guidelines.md +60 -0
  17. package/.github/skills/japanese-prose/scripts/core.py +192 -0
  18. package/.github/skills/japanese-prose/scripts/fixtures/natural.md +5 -0
  19. package/.github/skills/japanese-prose/scripts/fixtures/unnatural.md +5 -0
  20. package/.github/skills/japanese-prose/scripts/lint.py +378 -0
  21. package/.github/skills/japanese-prose/scripts/outline.py +68 -0
  22. package/.github/skills/japanese-prose/scripts/terms.py +112 -0
  23. package/.github/skills/japanese-prose/scripts/test_engine.py +117 -0
  24. package/.github/skills/presentation-planner/SKILL.md +257 -0
  25. package/.github/skills/presentation-planner/assets/design-templates/data-report.yaml +97 -0
  26. package/.github/skills/presentation-planner/assets/design-templates/executive-proposal.yaml +92 -0
  27. package/.github/skills/presentation-planner/assets/design-templates/technical-briefing.yaml +96 -0
  28. package/.github/skills/presentation-planner/assets/scenario-templates/data-report.md +47 -0
  29. package/.github/skills/presentation-planner/assets/scenario-templates/executive-decision.md +43 -0
  30. package/.github/skills/presentation-planner/assets/scenario-templates/technical-briefing.md +45 -0
  31. package/.github/skills/presentation-planner/references/customizing-design-templates.md +160 -0
  32. package/.github/skills/presentation-planner/references/design-spec-schema.md +72 -0
  33. package/.github/skills/presentation-planner/references/handoff-contract.md +49 -0
  34. package/.github/skills/presentation-planner/references/responsibility-boundary.md +32 -0
  35. package/.github/skills/presentation-planner/references/scenario-templates.md +55 -0
  36. package/.github/skills/tech-writer/SKILL.md +434 -0
  37. package/.github/skills/tech-writer/assets/templates/blueprint.md +187 -0
  38. package/.github/skills/tech-writer/assets/templates/design-doc.md +29 -0
  39. package/.github/skills/tech-writer/assets/templates/migration-plan.md +173 -0
  40. package/.github/skills/tech-writer/assets/templates/operations-runbook.md +202 -0
  41. package/.github/skills/tech-writer/assets/templates/pr-description.md +23 -0
  42. package/.github/skills/tech-writer/assets/templates/qiita.md +44 -0
  43. package/.github/skills/tech-writer/assets/templates/readme.md +38 -0
  44. package/.github/skills/tech-writer/assets/templates/requirements-definition.md +170 -0
  45. package/.github/skills/tech-writer/assets/templates/rfi.md +113 -0
  46. package/.github/skills/tech-writer/assets/templates/rfp.md +180 -0
  47. package/.github/skills/tech-writer/assets/templates/security-design.md +167 -0
  48. package/.github/skills/tech-writer/assets/templates/system-design.md +220 -0
  49. package/.github/skills/tech-writer/assets/templates/technical-proposal.md +112 -0
  50. package/.github/skills/tech-writer/assets/templates/test-plan.md +153 -0
  51. package/.github/skills/tech-writer/assets/templates/user-manual.md +22 -0
  52. package/.github/skills/tech-writer/assets/templates/white-paper.md +192 -0
  53. package/.github/skills/tech-writer/references/doctypes/api-docs.md +33 -0
  54. package/.github/skills/tech-writer/references/doctypes/blueprint.md +81 -0
  55. package/.github/skills/tech-writer/references/doctypes/code-comments.md +39 -0
  56. package/.github/skills/tech-writer/references/doctypes/design-doc.md +42 -0
  57. package/.github/skills/tech-writer/references/doctypes/migration-plan.md +63 -0
  58. package/.github/skills/tech-writer/references/doctypes/operations-runbook.md +63 -0
  59. package/.github/skills/tech-writer/references/doctypes/pr-commit.md +82 -0
  60. package/.github/skills/tech-writer/references/doctypes/qiita.md +75 -0
  61. package/.github/skills/tech-writer/references/doctypes/readme.md +43 -0
  62. package/.github/skills/tech-writer/references/doctypes/release-notes.md +30 -0
  63. package/.github/skills/tech-writer/references/doctypes/requirements-definition.md +61 -0
  64. package/.github/skills/tech-writer/references/doctypes/rfi.md +43 -0
  65. package/.github/skills/tech-writer/references/doctypes/rfp.md +46 -0
  66. package/.github/skills/tech-writer/references/doctypes/security-design.md +71 -0
  67. package/.github/skills/tech-writer/references/doctypes/system-design.md +74 -0
  68. package/.github/skills/tech-writer/references/doctypes/technical-proposal.md +49 -0
  69. package/.github/skills/tech-writer/references/doctypes/test-plan.md +67 -0
  70. package/.github/skills/tech-writer/references/doctypes/user-manual.md +58 -0
  71. package/.github/skills/tech-writer/references/doctypes/white-paper.md +84 -0
  72. package/.github/skills/tech-writer/references/doctypes/zenn.md +66 -0
  73. package/.github/skills/tech-writer/references/japanese-prose-optimization.md +110 -0
  74. package/.github/skills/tech-writer/references/style-constitution.md +104 -0
  75. package/.github/skills/tech-writer/scripts/lint.py +412 -0
  76. package/LICENSE +21 -0
  77. package/README.md +92 -0
  78. package/bin/ai-data-scientist.js +123 -0
  79. package/package.json +41 -0
  80. package/pyproject.toml +45 -0
  81. package/src/ai_chemistry_scientist/__init__.py +0 -0
  82. package/src/ai_chemistry_scientist/admet_prediction.py +71 -0
  83. package/src/ai_chemistry_scientist/bioactivity_classification.py +73 -0
  84. package/src/ai_chemistry_scientist/data/sample_molecules.csv +21 -0
  85. package/src/ai_chemistry_scientist/dispatch.py +369 -0
  86. package/src/ai_chemistry_scientist/docking_score.py +97 -0
  87. package/src/ai_chemistry_scientist/drug_likeness_rules.py +84 -0
  88. package/src/ai_chemistry_scientist/evidence.py +41 -0
  89. package/src/ai_chemistry_scientist/molecular_descriptors.py +97 -0
  90. package/src/ai_chemistry_scientist/molecular_formula_mass.py +40 -0
  91. package/src/ai_chemistry_scientist/molecular_similarity.py +78 -0
  92. package/src/ai_chemistry_scientist/qsar_modeling.py +105 -0
  93. package/src/ai_chemistry_scientist/salt_standardization.py +81 -0
  94. package/src/ai_chemistry_scientist/structural_alerts.py +76 -0
  95. package/src/ai_chemistry_scientist/structure_format_conversion.py +84 -0
  96. package/src/ai_chemistry_scientist/validation.py +70 -0
  97. package/src/ai_data_scientist/__init__.py +0 -0
  98. package/src/ai_data_scientist/analysis_assumptions.py +121 -0
  99. package/src/ai_data_scientist/anomaly_detection.py +39 -0
  100. package/src/ai_data_scientist/automl.py +109 -0
  101. package/src/ai_data_scientist/cleaning.py +56 -0
  102. package/src/ai_data_scientist/cli.py +90 -0
  103. package/src/ai_data_scientist/clustering.py +54 -0
  104. package/src/ai_data_scientist/dashboard.py +33 -0
  105. package/src/ai_data_scientist/data_definition.py +100 -0
  106. package/src/ai_data_scientist/data_quality.py +164 -0
  107. package/src/ai_data_scientist/dataset_validation.py +135 -0
  108. package/src/ai_data_scientist/dependency_pins.py +60 -0
  109. package/src/ai_data_scientist/eda.py +82 -0
  110. package/src/ai_data_scientist/experiment_evaluation.py +635 -0
  111. package/src/ai_data_scientist/explainability.py +340 -0
  112. package/src/ai_data_scientist/feature_engineering.py +163 -0
  113. package/src/ai_data_scientist/gate_config.py +32 -0
  114. package/src/ai_data_scientist/ingestion.py +127 -0
  115. package/src/ai_data_scientist/insight_engine.py +180 -0
  116. package/src/ai_data_scientist/japanese_nlp.py +43 -0
  117. package/src/ai_data_scientist/jupyter_launcher.py +137 -0
  118. package/src/ai_data_scientist/jupyter_mcp_client.py +94 -0
  119. package/src/ai_data_scientist/language_router.py +28 -0
  120. package/src/ai_data_scientist/lifecycle.py +221 -0
  121. package/src/ai_data_scientist/mcp_gateway.py +113 -0
  122. package/src/ai_data_scientist/mcp_runtime.py +194 -0
  123. package/src/ai_data_scientist/mcp_transport.py +53 -0
  124. package/src/ai_data_scientist/ml_modeling.py +451 -0
  125. package/src/ai_data_scientist/model_tuning.py +104 -0
  126. package/src/ai_data_scientist/notebook_audit.py +574 -0
  127. package/src/ai_data_scientist/project_manager.py +243 -0
  128. package/src/ai_data_scientist/report_export.py +73 -0
  129. package/src/ai_data_scientist/sensitivity.py +445 -0
  130. package/src/ai_data_scientist/signal_analysis.py +201 -0
  131. package/src/ai_data_scientist/skill_packaging.py +40 -0
  132. package/src/ai_data_scientist/stats_analysis.py +88 -0
  133. package/src/ai_data_scientist/text_nlp.py +44 -0
  134. package/src/ai_data_scientist/timeseries.py +68 -0
  135. package/src/ai_data_scientist/visualization.py +708 -0
  136. package/src/ai_genomics_scientist/__init__.py +1 -0
  137. package/src/ai_genomics_scientist/differential_expression.py +147 -0
  138. package/src/ai_genomics_scientist/dispatch.py +267 -0
  139. package/src/ai_genomics_scientist/evidence.py +45 -0
  140. package/src/ai_genomics_scientist/gene_set_enrichment.py +76 -0
  141. package/src/ai_genomics_scientist/sequence_alignment.py +97 -0
  142. package/src/ai_genomics_scientist/sequence_features.py +111 -0
  143. package/src/ai_genomics_scientist/splice_site_scoring.py +66 -0
  144. package/src/ai_genomics_scientist/validation.py +83 -0
  145. package/src/ai_genomics_scientist/variant_effect.py +147 -0
  146. package/src/ai_genomics_scientist/variant_pathogenicity.py +125 -0
  147. package/src/ai_materials_scientist/__init__.py +0 -0
  148. package/src/ai_materials_scientist/calphad.py +117 -0
  149. package/src/ai_materials_scientist/classical_monte_carlo.py +165 -0
  150. package/src/ai_materials_scientist/crystal_plasticity.py +184 -0
  151. package/src/ai_materials_scientist/dispatch.py +100 -0
  152. package/src/ai_materials_scientist/evidence.py +84 -0
  153. package/src/ai_materials_scientist/fem.py +279 -0
  154. package/src/ai_materials_scientist/kinetic_monte_carlo.py +145 -0
  155. package/src/ai_materials_scientist/molecular_dynamics.py +240 -0
  156. package/src/ai_materials_scientist/phase_field.py +167 -0
  157. package/src/ai_materials_scientist/validation.py +70 -0
  158. package/src/ai_scientist/__init__.py +1 -0
  159. package/src/ai_scientist/completion_gate.py +15 -0
  160. package/src/ai_scientist/data_analysis.py +46 -0
  161. package/src/ai_scientist/evidence_registry.py +99 -0
  162. package/src/ai_scientist/experimental_design.py +20 -0
  163. package/src/ai_scientist/language.py +14 -0
  164. package/src/ai_scientist/latex_renderer.py +41 -0
  165. package/src/ai_scientist/literature_review.py +37 -0
  166. package/src/ai_scientist/manifest.py +87 -0
  167. package/src/ai_scientist/manuscript.py +94 -0
  168. package/src/ai_scientist/mcp_config.py +76 -0
  169. package/src/ai_scientist/mcp_external.py +42 -0
  170. package/src/ai_scientist/mcp_failures.py +23 -0
  171. package/src/ai_scientist/mcp_gateway.py +38 -0
  172. package/src/ai_scientist/mcp_managed.py +180 -0
  173. package/src/ai_scientist/npm_packaging.py +49 -0
  174. package/src/ai_scientist/orchestrator.py +133 -0
  175. package/src/ai_scientist/peer_review.py +60 -0
  176. package/src/ai_scientist/phase_gate.py +74 -0
  177. package/src/ai_scientist/phase_state.py +230 -0
  178. package/src/ai_scientist/presentation.py +56 -0
  179. package/src/ai_scientist/project_config.py +31 -0
  180. package/src/ai_scientist/project_handle.py +74 -0
  181. package/src/ai_scientist/reproducibility.py +20 -0
  182. package/src/ai_scientist/research_planning.py +20 -0
  183. package/src/ai_scientist/skill_invocation.py +21 -0
  184. package/src/ai_scientist/tdd_gate.py +99 -0
  185. package/src/ai_structural_biology_scientist/__init__.py +0 -0
  186. package/src/ai_structural_biology_scientist/contact_map.py +87 -0
  187. package/src/ai_structural_biology_scientist/dispatch.py +269 -0
  188. package/src/ai_structural_biology_scientist/evidence.py +43 -0
  189. package/src/ai_structural_biology_scientist/hydrophobicity.py +101 -0
  190. package/src/ai_structural_biology_scientist/protein_docking_score.py +104 -0
  191. package/src/ai_structural_biology_scientist/secondary_structure.py +95 -0
  192. package/src/ai_structural_biology_scientist/structural_similarity.py +74 -0
  193. package/src/ai_structural_biology_scientist/validation.py +100 -0
package/pyproject.toml ADDED
@@ -0,0 +1,45 @@
1
+ [project]
2
+ name = "ai-data-scientist"
3
+ version = "0.2.0"
4
+ description = "GitHub Copilot Agent Skill: AI Data Scientist (MVP) over Jupyter MCP"
5
+ license = { text = "MIT" }
6
+ requires-python = ">=3.10"
7
+ dependencies = [
8
+ "nbformat>=5.9",
9
+ "pandas>=2.0",
10
+ "scipy>=1.11",
11
+ "matplotlib>=3.8",
12
+ "openpyxl>=3.1",
13
+ "scikit-learn>=1.5",
14
+ "statsmodels>=0.14",
15
+ "nbconvert>=7.16",
16
+ "spacy>=3.7",
17
+ "ginza>=5.2",
18
+ "ja-ginza>=5.2",
19
+ "vaderSentiment>=3.3",
20
+ "plotly>=5.20",
21
+ "httpx>=0.27",
22
+ "mcp>=2.2,<2.3",
23
+ "jupyterlab>=4.2",
24
+ "jupyter-mcp-server>=2.2,<2.3",
25
+ "japanize-matplotlib>=1.1",
26
+ "packaging>=23",
27
+ "rdkit>=2024.3",
28
+ ]
29
+
30
+ [project.optional-dependencies]
31
+ dev = ["pytest>=8", "pytest-json-report>=1.5", "ruff>=0.6", "pyyaml>=6.0"]
32
+
33
+ [tool.pytest.ini_options]
34
+ testpaths = ["tests"]
35
+
36
+ [tool.ruff]
37
+ line-length = 100
38
+ src = ["src", "tests"]
39
+
40
+ [build-system]
41
+ requires = ["setuptools>=68", "wheel"]
42
+ build-backend = "setuptools.build_meta"
43
+
44
+ [tool.setuptools.packages.find]
45
+ where = ["src"]
File without changes
@@ -0,0 +1,71 @@
1
+ """ADMET heuristic screening module (DES-ACHEM-020 / REQ-ACHEM-020)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from ai_chemistry_scientist.molecular_descriptors import compute_descriptors, parse_smiles
6
+ from ai_chemistry_scientist.validation import fail, ok, register_validator
7
+
8
+ _MODULE_NAME = "admet-prediction"
9
+
10
+ #: Fixed-order Lipinski Rule-of-Five criteria (REQ-ACHEM-020 acceptance).
11
+ _LIPINSKI_CRITERIA = (
12
+ ("MolWt", lambda d: d["mol_wt"] <= 500),
13
+ ("MolLogP", lambda d: d["mol_logp"] <= 5),
14
+ ("NumHDonors", lambda d: d["num_h_donors"] <= 5),
15
+ ("NumHAcceptors", lambda d: d["num_h_acceptors"] <= 10),
16
+ )
17
+ #: Fixed-order Veber rule criteria (REQ-ACHEM-020 acceptance).
18
+ _VEBER_CRITERIA = (
19
+ ("NumRotatableBonds", lambda d: d["num_rotatable_bonds"] <= 10),
20
+ ("TPSA", lambda d: d["tpsa"] <= 140),
21
+ )
22
+
23
+ #: The handler wrapper of DES-ACHEM-001 substitutes this key for the
24
+ #: `language`-specific text before `record_run` (REQ-ACHEM-020 Constraints).
25
+ LIMITATION_LABEL_KEY = "admet_heuristic_limitation"
26
+ LIMITATION_LABEL_TEXT = {
27
+ "en": "Heuristic only: not a physically or clinically validated ADMET prediction.",
28
+ "ja": ("ヒューリスティックのみ: 物理的または臨床的に検証されたADMET予測ではありません。"),
29
+ }
30
+
31
+
32
+ def _admet_prediction_validator(params: dict) -> dict:
33
+ """DES-ACHEM-002 registered atomic validator for this module."""
34
+ if "smiles" not in params:
35
+ return fail("smiles", "is required")
36
+ if parse_smiles(params["smiles"]) is None:
37
+ return fail("smiles", "must parse to a valid RDKit molecule")
38
+ return ok()
39
+
40
+
41
+ register_validator(_MODULE_NAME, _admet_prediction_validator)
42
+
43
+
44
+ # @id CODE-ACHEM-020
45
+ # @implements REQ-ACHEM-020
46
+ # @design DES-ACHEM-020
47
+ def run_admet_prediction(smiles: str) -> dict:
48
+ """Compute descriptors and the Lipinski/Veber heuristic flags.
49
+
50
+ Receives ``smiles`` already validated atomically by its handler wrapper
51
+ (DES-ACHEM-001); performs no revalidation of its own.
52
+ """
53
+ mol = parse_smiles(smiles)
54
+ if mol is None:
55
+ raise ValueError("smiles must already be validated by the handler wrapper")
56
+ descriptors = compute_descriptors(mol)
57
+
58
+ lipinski_violations = [name for name, check in _LIPINSKI_CRITERIA if not check(descriptors)]
59
+ lipinski_pass = len(lipinski_violations) <= 1
60
+
61
+ veber_violations = [name for name, check in _VEBER_CRITERIA if not check(descriptors)]
62
+ veber_pass = len(veber_violations) == 0
63
+
64
+ return {
65
+ "descriptors": descriptors,
66
+ "lipinski_violations": lipinski_violations,
67
+ "lipinski_pass": lipinski_pass,
68
+ "veber_violations": veber_violations,
69
+ "veber_pass": veber_pass,
70
+ "limitation_label_key": LIMITATION_LABEL_KEY,
71
+ }
@@ -0,0 +1,73 @@
1
+ """Bioactivity classification module (DES-ACHEM-090 / REQ-ACHEM-090)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from rdkit.Chem import rdMolDescriptors
6
+
7
+ from ai_chemistry_scientist.molecular_descriptors import compute_descriptors, parse_smiles
8
+ from ai_chemistry_scientist.validation import fail, ok, register_validator
9
+
10
+ _MODULE_NAME = "bioactivity-classification"
11
+ # Named decision-threshold constants (DES-ACHEM-090) so the fixed,
12
+ # order-sensitive CNS-like/kinase-inhibitor-like boundaries are documented in
13
+ # one place instead of as inline magic numbers.
14
+ CNS_LIKE_TPSA_MAX = 90
15
+ CNS_LIKE_MOLLOGP_MIN, CNS_LIKE_MOLLOGP_MAX = 2.0, 5.0
16
+ KINASE_INHIBITOR_LIKE_MOLWT_MIN = 400
17
+ KINASE_INHIBITOR_LIKE_AROMATIC_RING_COUNT_MIN = 3
18
+ LIMITATION_LABEL_KEY = "bioactivity_classifier_heuristic_limitation"
19
+ LIMITATION_LABEL_TEXT = {
20
+ "en": "Heuristic only: not a ChEMBL-trained or experimentally validated bioactivity classifier.",
21
+ "ja": (
22
+ "ヒューリスティックのみ: ChEMBLで学習済みでも実験的に検証済みでもない"
23
+ "生物活性分類器ではない。"
24
+ ),
25
+ }
26
+
27
+
28
+ # @id CODE-ACHEM-920
29
+ # @implements REQ-ACHEM-003 REQ-ACHEM-090
30
+ # @design DES-ACHEM-002
31
+ def _bioactivity_classification_validator(params: dict) -> dict:
32
+ if "smiles" not in params:
33
+ return fail("smiles", "is required")
34
+ if parse_smiles(params["smiles"]) is None:
35
+ return fail("smiles", "must parse to a valid RDKit molecule")
36
+ return ok()
37
+
38
+
39
+ register_validator(_MODULE_NAME, _bioactivity_classification_validator)
40
+
41
+
42
+ # @id CODE-ACHEM-090
43
+ # @implements REQ-ACHEM-090
44
+ # @design DES-ACHEM-090
45
+ def run_bioactivity_classification(smiles: str) -> dict:
46
+ """Assign the fixed-order heuristic label for ``smiles``."""
47
+ mol = parse_smiles(smiles)
48
+ if mol is None:
49
+ raise ValueError("smiles must already be validated by the handler wrapper")
50
+ descriptors = compute_descriptors(mol)
51
+ aromatic_ring_count = int(rdMolDescriptors.CalcNumAromaticRings(mol))
52
+
53
+ if (
54
+ descriptors["tpsa"] < CNS_LIKE_TPSA_MAX
55
+ and CNS_LIKE_MOLLOGP_MIN <= descriptors["mol_logp"] <= CNS_LIKE_MOLLOGP_MAX
56
+ ):
57
+ label = "CNS_like"
58
+ elif (
59
+ descriptors["mol_wt"] > KINASE_INHIBITOR_LIKE_MOLWT_MIN
60
+ and aromatic_ring_count >= KINASE_INHIBITOR_LIKE_AROMATIC_RING_COUNT_MIN
61
+ ):
62
+ label = "kinase_inhibitor_like"
63
+ else:
64
+ label = "other"
65
+
66
+ return {
67
+ "label": label,
68
+ "tpsa": descriptors["tpsa"],
69
+ "mol_logp": descriptors["mol_logp"],
70
+ "mol_wt": descriptors["mol_wt"],
71
+ "aromatic_ring_count": aromatic_ring_count,
72
+ "limitation_label_key": LIMITATION_LABEL_KEY,
73
+ }
@@ -0,0 +1,21 @@
1
+ name,smiles
2
+ aspirin,CC(=O)OC1=CC=CC=C1C(=O)O
3
+ ibuprofen,CC(C)CC1=CC=C(C=C1)C(C)C(=O)O
4
+ caffeine,CN1C=NC2=C1C(=O)N(C(=O)N2C)C
5
+ paracetamol,CC(=O)NC1=CC=C(C=C1)O
6
+ naproxen,COC1=CC2=CC(=CC=C2C=C1)C(C)C(=O)O
7
+ diclofenac,OC(=O)Cc1ccccc1Nc1c(Cl)cccc1Cl
8
+ warfarin,CC(=O)CC(c1ccccc1)c1c(O)c2ccccc2oc1=O
9
+ metformin,CN(C)C(=N)NC(=N)N
10
+ atorvastatin,CC(C)c1c(C(=O)Nc2ccccc2)c(-c2ccc(F)cc2)c(-c2ccc(O)cc2)n1CCC(O)CC(O)CC(=O)O
11
+ omeprazole,CC1=CN=C(C(=C1OC)C)CS(=O)c1nc2ccc(OC)cc2[nH]1
12
+ loratadine,CCOC(=O)N1CCC(=C2c3ccc(Cl)cc3CCc3cccnc23)CC1
13
+ cetirizine,OC(=O)COCCN1CCN(CC1)C(c1ccccc1)c1ccc(Cl)cc1
14
+ simvastatin,CCC(C)(C)C(=O)OC1CC(C)C=C2C=CC(C)C(CCC3CC(O)CC(=O)O3)C12
15
+ lisinopril,CCCCC(N(CCCC(N)C(=O)O)C(=O)C(CCc1ccccc1)N)C(=O)N1CCCC1C(=O)O
16
+ amlodipine,CCOC(=O)C1=C(COCCN)NC(C)=C(C(=O)OC)C1c1ccccc1Cl
17
+ losartan,CCCCc1nc(Cl)c(CO)n1Cc1ccc(-c2ccccc2-c2nnn[nH]2)cc1
18
+ metoprolol,COCCc1ccc(OCC(O)CNC(C)C)cc1
19
+ furosemide,NS(=O)(=O)c1cc(C(=O)O)c(NCc2ccco2)cc1Cl
20
+ ranitidine,CNC(=C[N+](=O)[O-])NCCSCc1ccc(CN(C)C)o1
21
+ sertraline,CNC1CCC(c2ccc(Cl)c(Cl)c2)c2ccccc21
@@ -0,0 +1,369 @@
1
+ """Method manifest & request dispatcher (DES-ACHEM-001 / REQ-ACHEM-001/002)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import importlib
6
+ import json
7
+ from pathlib import Path
8
+
9
+ import rdkit
10
+
11
+ # Imported eagerly (not lazily inside `_resolve_run_function`) so each
12
+ # module's `register_validator`/`register_batch_item_validator` call runs
13
+ # before any `validate_parameters` lookup below.
14
+ import ai_chemistry_scientist.admet_prediction as _admet_prediction # noqa: F401
15
+ import ai_chemistry_scientist.bioactivity_classification as _bioactivity_classification # noqa: F401
16
+ import ai_chemistry_scientist.docking_score as _docking_score # noqa: F401
17
+ import ai_chemistry_scientist.drug_likeness_rules as _drug_likeness_rules # noqa: F401
18
+ import ai_chemistry_scientist.molecular_descriptors as _molecular_descriptors # noqa: F401
19
+ import ai_chemistry_scientist.molecular_formula_mass as _molecular_formula_mass # noqa: F401
20
+ import ai_chemistry_scientist.molecular_similarity as _molecular_similarity # noqa: F401
21
+ import ai_chemistry_scientist.qsar_modeling as _qsar_modeling # noqa: F401
22
+ import ai_chemistry_scientist.salt_standardization as _salt_standardization # noqa: F401
23
+ import ai_chemistry_scientist.structural_alerts as _structural_alerts # noqa: F401
24
+ import ai_chemistry_scientist.structure_format_conversion as _structure_format_conversion # noqa: F401
25
+ from ai_chemistry_scientist.evidence import record_run
26
+ from ai_chemistry_scientist.validation import validate_parameters
27
+ from ai_data_scientist.language_router import detect_language as _detect_language
28
+
29
+ REPO_ROOT = Path(__file__).resolve().parents[2]
30
+ DEFAULT_MANIFEST_PATH = (
31
+ REPO_ROOT / ".github" / "skills" / "ai-chemistry-scientist" / "manifest.json"
32
+ )
33
+
34
+ #: The one module whose handler wrapper skips the separate upfront
35
+ #: `validate_parameters` call, because its own `run_*` function interleaves
36
+ #: per-item validation with per-item computation (ADR-0026, DES-ACHEM-001).
37
+ _PER_ITEM_VALIDATED_MODULES = frozenset({"molecular-descriptors"})
38
+
39
+ #: modules whose raw result's `limitation_label_key` is substituted with the
40
+ #: matching `language`-specific text before `record_run` (DES-ACHEM-001).
41
+ _LIMITATION_LABEL_MODULES = frozenset(
42
+ {
43
+ "admet-prediction",
44
+ "docking-score",
45
+ "structural-alerts",
46
+ "bioactivity-classification",
47
+ "salt-removal",
48
+ }
49
+ )
50
+
51
+ _RUN_MODULE_PATHS = {
52
+ "molecular-descriptors": "ai_chemistry_scientist.molecular_descriptors",
53
+ "admet-prediction": "ai_chemistry_scientist.admet_prediction",
54
+ "qsar-modeling": "ai_chemistry_scientist.qsar_modeling",
55
+ "molecular-similarity": "ai_chemistry_scientist.molecular_similarity",
56
+ "docking-score": "ai_chemistry_scientist.docking_score",
57
+ "drug-likeness-rules": "ai_chemistry_scientist.drug_likeness_rules",
58
+ "structural-alerts": "ai_chemistry_scientist.structural_alerts",
59
+ "molecular-formula-mass": "ai_chemistry_scientist.molecular_formula_mass",
60
+ "bioactivity-classification": "ai_chemistry_scientist.bioactivity_classification",
61
+ "salt-removal": "ai_chemistry_scientist.salt_standardization",
62
+ "structure-format-conversion": "ai_chemistry_scientist.structure_format_conversion",
63
+ }
64
+ _RUN_FUNCTION_NAMES = {
65
+ "molecular-descriptors": "run_molecular_descriptors",
66
+ "admet-prediction": "run_admet_prediction",
67
+ "qsar-modeling": "run_qsar_modeling",
68
+ "molecular-similarity": "run_molecular_similarity",
69
+ "docking-score": "run_docking_score",
70
+ "drug-likeness-rules": "run_drug_likeness_rules",
71
+ "structural-alerts": "run_structural_alerts",
72
+ "molecular-formula-mass": "run_molecular_formula_mass",
73
+ "bioactivity-classification": "run_bioactivity_classification",
74
+ "salt-removal": "run_salt_removal",
75
+ "structure-format-conversion": "run_structure_conversion",
76
+ }
77
+
78
+
79
+ def load_manifest(manifest_path: Path | None = None) -> dict:
80
+ """Load the static method-name-to-module manifest (DES-ACHEM-001)."""
81
+ path = manifest_path or DEFAULT_MANIFEST_PATH
82
+ return json.loads(path.read_text(encoding="utf-8"))
83
+
84
+
85
+ def _matched_methods(request_text: str, manifest: dict) -> list[str]:
86
+ # Collapse runs of whitespace so extra spacing between words in a
87
+ # request (e.g. multi-space or tab-separated phrasing) still matches an
88
+ # English candidate phrase (REQ-ACHEM-002); Japanese candidates have no
89
+ # internal whitespace, so matching against the raw text is unaffected.
90
+ lowered = " ".join(request_text.lower().split())
91
+ matched = []
92
+ for method, entry in manifest.items():
93
+ names = entry.get("names", {})
94
+ candidates = list(names.get("en", [])) + list(names.get("ja", []))
95
+ if any(
96
+ (candidate.lower() in lowered) if candidate.isascii() else (candidate in request_text)
97
+ for candidate in candidates
98
+ ):
99
+ matched.append(method)
100
+ return matched
101
+
102
+
103
+ # @id CODE-ACHEM-916
104
+ # @implements REQ-ACHEM-002
105
+ # @design DES-ACHEM-001
106
+ def extract_params(request_text: str) -> dict | None:
107
+ """Extract a module's structured ``params`` embedded in ``request_text``.
108
+
109
+ Each handler wrapper's own documented extraction responsibility
110
+ (DES-ACHEM-001): a calling context supplies structured parameters by
111
+ embedding exactly one JSON object literal anywhere in ``request_text``
112
+ (e.g. a natural-language instruction followed by
113
+ ``{"smiles": "CC(=O)OC1=CC=CC=C1C(=O)O"}``). Returns ``None`` when no
114
+ balanced top-level JSON object is present or it fails to parse.
115
+ """
116
+ start = request_text.find("{")
117
+ if start == -1:
118
+ return None
119
+ depth = 0
120
+ for index in range(start, len(request_text)):
121
+ char = request_text[index]
122
+ if char == "{":
123
+ depth += 1
124
+ elif char == "}":
125
+ depth -= 1
126
+ if depth == 0:
127
+ candidate = request_text[start : index + 1]
128
+ try:
129
+ parsed = json.loads(candidate)
130
+ except json.JSONDecodeError:
131
+ return None
132
+ return parsed if isinstance(parsed, dict) else None
133
+ return None
134
+
135
+
136
+ def _resolve_run_function(method: str):
137
+ """Resolve each module's raw ``run_*`` compute function by method name."""
138
+ module = importlib.import_module(_RUN_MODULE_PATHS[method])
139
+ return getattr(module, _RUN_FUNCTION_NAMES[method])
140
+
141
+
142
+ def _localize_limitation_label(method: str, result: dict, language: str) -> dict:
143
+ """Substitute `limitation_label_key` with its `language` text (DES-ACHEM-001)."""
144
+ if method not in _LIMITATION_LABEL_MODULES:
145
+ return result
146
+ module_path = {
147
+ "admet-prediction": "ai_chemistry_scientist.admet_prediction",
148
+ "docking-score": "ai_chemistry_scientist.docking_score",
149
+ "structural-alerts": "ai_chemistry_scientist.structural_alerts",
150
+ "bioactivity-classification": "ai_chemistry_scientist.bioactivity_classification",
151
+ "salt-removal": "ai_chemistry_scientist.salt_standardization",
152
+ }[method]
153
+ module = importlib.import_module(module_path)
154
+ key = result["limitation_label_key"]
155
+ assert key == module.LIMITATION_LABEL_KEY
156
+ text = module.LIMITATION_LABEL_TEXT[language]
157
+ localized = dict(result)
158
+ del localized["limitation_label_key"]
159
+ localized["limitation_label"] = text
160
+ return localized
161
+
162
+
163
+ def _no_params_outcome(language: str) -> dict:
164
+ return {
165
+ "ok": False,
166
+ "parameter": "params",
167
+ "constraint": "must be extractable as a JSON object embedded in request_text",
168
+ "language": language,
169
+ }
170
+
171
+
172
+ def _handle_module(
173
+ method: str, request_text: str, language: str, params: dict | None = None
174
+ ) -> dict:
175
+ """Shared handler-wrapper body for ``method`` (DES-ACHEM-001's `handle_<method>`).
176
+
177
+ Extracts params from ``request_text``, validates (except for the
178
+ per-item-validated module), computes via the raw `run_*` function,
179
+ localizes limitation labels, and wraps the result into a RunRecord
180
+ (DES-ACHEM-003) — a `ModuleOutcome`.
181
+ """
182
+ resolved_params = params if params is not None else extract_params(request_text)
183
+ if resolved_params is None:
184
+ return _no_params_outcome(language)
185
+
186
+ if method not in _PER_ITEM_VALIDATED_MODULES:
187
+ validation = validate_parameters(method, resolved_params)
188
+ if not validation["ok"]:
189
+ return {
190
+ "ok": False,
191
+ "parameter": validation["parameter"],
192
+ "constraint": validation["constraint"],
193
+ "language": language,
194
+ }
195
+
196
+ run_function = _resolve_run_function(method)
197
+ result = run_function(**resolved_params)
198
+ result = _localize_limitation_label(method, result, language)
199
+
200
+ extra_kwargs = {}
201
+ if method == "qsar-modeling":
202
+ import sklearn
203
+
204
+ extra_kwargs["scikit_learn_version"] = sklearn.__version__
205
+
206
+ run_record = record_run(
207
+ module_name=method,
208
+ params=resolved_params,
209
+ result=result,
210
+ rdkit_version=rdkit.__version__,
211
+ **extra_kwargs,
212
+ )
213
+ return {"ok": True, "run_record": run_record}
214
+
215
+
216
+ # @id CODE-ACHEM-011
217
+ # @implements REQ-ACHEM-002 REQ-ACHEM-003
218
+ # @design DES-ACHEM-001
219
+ def handle_molecular_descriptors(request_text: str, language: str, **params) -> dict:
220
+ """Handler wrapper for the molecular-descriptors module (DES-ACHEM-010)."""
221
+ return _handle_module("molecular-descriptors", request_text, language, params or None)
222
+
223
+
224
+ # @id CODE-ACHEM-021
225
+ # @implements REQ-ACHEM-002 REQ-ACHEM-003
226
+ # @design DES-ACHEM-001
227
+ def handle_admet_prediction(request_text: str, language: str, **params) -> dict:
228
+ """Handler wrapper for the admet-prediction module (DES-ACHEM-020)."""
229
+ return _handle_module("admet-prediction", request_text, language, params or None)
230
+
231
+
232
+ # @id CODE-ACHEM-031
233
+ # @implements REQ-ACHEM-002 REQ-ACHEM-003
234
+ # @design DES-ACHEM-001
235
+ def handle_qsar_modeling(request_text: str, language: str, **params) -> dict:
236
+ """Handler wrapper for the qsar-modeling module (DES-ACHEM-030)."""
237
+ return _handle_module("qsar-modeling", request_text, language, params or None)
238
+
239
+
240
+ # @id CODE-ACHEM-041
241
+ # @implements REQ-ACHEM-002 REQ-ACHEM-003
242
+ # @design DES-ACHEM-001
243
+ def handle_molecular_similarity(request_text: str, language: str, **params) -> dict:
244
+ """Handler wrapper for the molecular-similarity module (DES-ACHEM-040)."""
245
+ return _handle_module("molecular-similarity", request_text, language, params or None)
246
+
247
+
248
+ # @id CODE-ACHEM-051
249
+ # @implements REQ-ACHEM-002 REQ-ACHEM-003
250
+ # @design DES-ACHEM-001
251
+ def handle_docking_score(request_text: str, language: str, **params) -> dict:
252
+ """Handler wrapper for the docking-score module (DES-ACHEM-050)."""
253
+ return _handle_module("docking-score", request_text, language, params or None)
254
+
255
+
256
+ # @id CODE-ACHEM-061
257
+ # @implements REQ-ACHEM-002 REQ-ACHEM-003
258
+ # @design DES-ACHEM-001
259
+ def handle_drug_likeness_rules(request_text: str, language: str, **params) -> dict:
260
+ """Handler wrapper for the drug-likeness-rules module (DES-ACHEM-060)."""
261
+ return _handle_module("drug-likeness-rules", request_text, language, params or None)
262
+
263
+
264
+ # @id CODE-ACHEM-071
265
+ # @implements REQ-ACHEM-002 REQ-ACHEM-003
266
+ # @design DES-ACHEM-001
267
+ def handle_structural_alerts(request_text: str, language: str, **params) -> dict:
268
+ """Handler wrapper for the structural-alerts module (DES-ACHEM-070)."""
269
+ return _handle_module("structural-alerts", request_text, language, params or None)
270
+
271
+
272
+ # @id CODE-ACHEM-081
273
+ # @implements REQ-ACHEM-002 REQ-ACHEM-003
274
+ # @design DES-ACHEM-001
275
+ def handle_molecular_formula_mass(request_text: str, language: str, **params) -> dict:
276
+ """Handler wrapper for the molecular-formula-mass module (DES-ACHEM-080)."""
277
+ return _handle_module("molecular-formula-mass", request_text, language, params or None)
278
+
279
+
280
+ # @id CODE-ACHEM-091
281
+ # @implements REQ-ACHEM-002 REQ-ACHEM-003
282
+ # @design DES-ACHEM-001
283
+ def handle_bioactivity_classification(request_text: str, language: str, **params) -> dict:
284
+ """Handler wrapper for the bioactivity-classification module (DES-ACHEM-090)."""
285
+ return _handle_module("bioactivity-classification", request_text, language, params or None)
286
+
287
+
288
+ # @id CODE-ACHEM-101
289
+ # @implements REQ-ACHEM-002 REQ-ACHEM-003
290
+ # @design DES-ACHEM-001
291
+ def handle_salt_removal(request_text: str, language: str, **params) -> dict:
292
+ """Handler wrapper for the salt-removal module (DES-ACHEM-100)."""
293
+ return _handle_module("salt-removal", request_text, language, params or None)
294
+
295
+
296
+ # @id CODE-ACHEM-111
297
+ # @implements REQ-ACHEM-002 REQ-ACHEM-003
298
+ # @design DES-ACHEM-001
299
+ def handle_structure_format_conversion(request_text: str, language: str, **params) -> dict:
300
+ """Handler wrapper for the structure-format-conversion module (DES-ACHEM-110)."""
301
+ return _handle_module("structure-format-conversion", request_text, language, params or None)
302
+
303
+
304
+ # @id CODE-ACHEM-001
305
+ # @implements REQ-ACHEM-001 REQ-ACHEM-002
306
+ # @design DES-ACHEM-001
307
+ def dispatch(
308
+ request_text: str,
309
+ language: str | None = None,
310
+ manifest_path: Path | None = None,
311
+ ) -> dict:
312
+ """Classify ``request_text`` and dispatch to exactly one matched module.
313
+
314
+ ``language`` may be supplied explicitly; otherwise it is detected from
315
+ ``request_text`` (REQ-ACHEM-001) and propagated into the result so every
316
+ downstream (module handler, clarification, rejection) path can render
317
+ its user-facing text in that language.
318
+ """
319
+ if not isinstance(request_text, str):
320
+ raise ValueError("request_text: must be a str") # noqa: TRY004
321
+ if language is not None and language not in ("en", "ja"):
322
+ raise ValueError("language: must be 'en' or 'ja' when explicitly supplied")
323
+ detected_language = language or _detect_language(request_text)
324
+ manifest = load_manifest(manifest_path)
325
+ matched = _matched_methods(request_text, manifest)
326
+
327
+ if len(matched) == 1:
328
+ method = matched[0]
329
+ entry = manifest[method]
330
+ handler_module = importlib.import_module(entry["modulePath"])
331
+ handler = getattr(handler_module, entry["functionName"])
332
+ handler_result = handler(request_text, detected_language)
333
+ return {
334
+ "outcome": "dispatch",
335
+ "module": method,
336
+ "language": detected_language,
337
+ "handler_result": handler_result,
338
+ }
339
+ if len(matched) > 1:
340
+ return {
341
+ "outcome": "clarification",
342
+ "candidates": matched,
343
+ "language": detected_language,
344
+ "clarification_question": _render_clarification(matched, detected_language),
345
+ }
346
+ return {
347
+ "outcome": "rejected",
348
+ "language": detected_language,
349
+ "rejected_method": _render_rejection(detected_language),
350
+ }
351
+
352
+
353
+ def _render_clarification(candidates: list[str], language: str) -> str:
354
+ """Render a single-sentence clarification question in ``language``.
355
+
356
+ Method names (e.g. ``molecular-descriptors``) are the only permitted
357
+ non-``language`` tokens, per REQ-ACHEM-001's acceptance.
358
+ """
359
+ names = "、".join(candidates) if language == "ja" else ", ".join(candidates)
360
+ if language == "ja":
361
+ return f"{names} のどちらを意図していますか。明確にしてください。"
362
+ return f"Did you mean {names}? Please clarify which method you want."
363
+
364
+
365
+ def _render_rejection(language: str) -> str:
366
+ """Render a single-sentence rejection message in ``language``."""
367
+ if language == "ja":
368
+ return "対応する手法が要求から認識されませんでした。"
369
+ return "No supported method was recognized in your request."
@@ -0,0 +1,97 @@
1
+ """Simplified docking-score heuristic module (DES-ACHEM-050 / REQ-ACHEM-050)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+
7
+ from ai_chemistry_scientist.molecular_descriptors import parse_smiles
8
+ from ai_chemistry_scientist.validation import fail, ok, register_validator
9
+
10
+ _MODULE_NAME = "docking-score"
11
+ _VOLUME_PER_HEAVY_ATOM_A3 = 15.0
12
+
13
+ #: The handler wrapper of DES-ACHEM-001 substitutes this key for the
14
+ #: `language`-specific text before `record_run` (REQ-ACHEM-050 Constraints).
15
+ LIMITATION_LABEL_KEY = "docking_heuristic_limitation"
16
+ LIMITATION_LABEL_TEXT = {
17
+ "en": (
18
+ "Heuristic only: not a physically accurate docking simulation (no 3D "
19
+ "conformer generation, no energy function)."
20
+ ),
21
+ "ja": (
22
+ "ヒューリスティックのみ: 物理的に正確なドッキングシミュレーションでは"
23
+ "ありません(3D配座生成・エネルギー関数なし)。"
24
+ ),
25
+ }
26
+
27
+
28
+ def _is_finite_non_negative_int(value) -> bool:
29
+ return isinstance(value, int) and not isinstance(value, bool) and value >= 0
30
+
31
+
32
+ def _docking_score_validator(params: dict) -> dict:
33
+ """DES-ACHEM-002 registered atomic validator for this module."""
34
+ if "ligand_smiles" not in params:
35
+ return fail("ligand_smiles", "is required")
36
+ if "pocket_spec" not in params:
37
+ return fail("pocket_spec", "is required")
38
+ if parse_smiles(params["ligand_smiles"]) is None:
39
+ return fail("smiles", "must parse to a valid RDKit molecule")
40
+ pocket_spec = params["pocket_spec"]
41
+ if not isinstance(pocket_spec, dict):
42
+ return fail("pocket_spec", "must be a dict")
43
+ for key in ("pocket_volume_A3", "pocket_hba_sites", "pocket_hbd_sites"):
44
+ if key not in pocket_spec:
45
+ return fail(key, "is required")
46
+ pocket_volume_a3 = pocket_spec["pocket_volume_A3"]
47
+ if (
48
+ not isinstance(pocket_volume_a3, (int, float))
49
+ or isinstance(pocket_volume_a3, bool)
50
+ or not math.isfinite(pocket_volume_a3)
51
+ or pocket_volume_a3 <= 0
52
+ ):
53
+ return fail("pocket_volume_A3", "must be a finite number > 0")
54
+ if not _is_finite_non_negative_int(pocket_spec["pocket_hba_sites"]):
55
+ return fail("pocket_hba_sites", "must be a finite non-negative integer")
56
+ if not _is_finite_non_negative_int(pocket_spec["pocket_hbd_sites"]):
57
+ return fail("pocket_hbd_sites", "must be a finite non-negative integer")
58
+ return ok()
59
+
60
+
61
+ register_validator(_MODULE_NAME, _docking_score_validator)
62
+
63
+
64
+ # @id CODE-ACHEM-050
65
+ # @implements REQ-ACHEM-050
66
+ # @design DES-ACHEM-050
67
+ def run_docking_score(ligand_smiles: str, pocket_spec: dict) -> dict:
68
+ """Compute the fixed-formula heuristic docking score for the ligand-pocket pair.
69
+
70
+ Receives ``ligand_smiles``/``pocket_spec`` already validated atomically
71
+ by its handler wrapper; performs no revalidation of its own.
72
+ """
73
+ from rdkit.Chem import Descriptors
74
+
75
+ mol = parse_smiles(ligand_smiles)
76
+ if mol is None:
77
+ raise ValueError("ligand_smiles must already be validated by the handler wrapper")
78
+ heavy_atom_count = mol.GetNumHeavyAtoms()
79
+ ligand_hbd = int(Descriptors.NumHDonors(mol))
80
+ ligand_hba = int(Descriptors.NumHAcceptors(mol))
81
+
82
+ pocket_volume_a3 = pocket_spec["pocket_volume_A3"]
83
+ pocket_hba_sites = pocket_spec["pocket_hba_sites"]
84
+ pocket_hbd_sites = pocket_spec["pocket_hbd_sites"]
85
+
86
+ ligand_volume = heavy_atom_count * _VOLUME_PER_HEAVY_ATOM_A3
87
+ size_fit = min(1.0, max(0.0, 1 - abs(ligand_volume - pocket_volume_a3) / pocket_volume_a3))
88
+ matched_pairs = min(ligand_hbd, pocket_hba_sites) + min(ligand_hba, pocket_hbd_sites)
89
+ hbond_fit = matched_pairs / max(1, ligand_hbd + ligand_hba)
90
+ score = 0.5 * size_fit + 0.5 * hbond_fit
91
+
92
+ return {
93
+ "score": score,
94
+ "size_fit": size_fit,
95
+ "hbond_fit": hbond_fit,
96
+ "limitation_label_key": LIMITATION_LABEL_KEY,
97
+ }