jupytermind 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/skills/ai-chemistry-scientist/SKILL.md +97 -0
- package/.github/skills/ai-chemistry-scientist/manifest.json +156 -0
- package/.github/skills/ai-data-scientist/SKILL.md +330 -0
- package/.github/skills/ai-genomics-scientist/SKILL.md +98 -0
- package/.github/skills/ai-genomics-scientist/manifest.json +93 -0
- package/.github/skills/ai-materials-scientist/SKILL.md +51 -0
- package/.github/skills/ai-materials-scientist/manifest.json +58 -0
- package/.github/skills/ai-scientist/SKILL.md +69 -0
- package/.github/skills/ai-scientist/manifest.json +61 -0
- package/.github/skills/ai-structural-biology-scientist/SKILL.md +67 -0
- package/.github/skills/ai-structural-biology-scientist/manifest.json +72 -0
- package/.github/skills/japanese-prose/NOTICE.md +17 -0
- package/.github/skills/japanese-prose/SKILL.md +111 -0
- package/.github/skills/japanese-prose/references/review-workflow.md +50 -0
- package/.github/skills/japanese-prose/references/scoring.md +24 -0
- package/.github/skills/japanese-prose/references/writing-guidelines.md +60 -0
- package/.github/skills/japanese-prose/scripts/core.py +192 -0
- package/.github/skills/japanese-prose/scripts/fixtures/natural.md +5 -0
- package/.github/skills/japanese-prose/scripts/fixtures/unnatural.md +5 -0
- package/.github/skills/japanese-prose/scripts/lint.py +378 -0
- package/.github/skills/japanese-prose/scripts/outline.py +68 -0
- package/.github/skills/japanese-prose/scripts/terms.py +112 -0
- package/.github/skills/japanese-prose/scripts/test_engine.py +117 -0
- package/.github/skills/presentation-planner/SKILL.md +257 -0
- package/.github/skills/presentation-planner/assets/design-templates/data-report.yaml +97 -0
- package/.github/skills/presentation-planner/assets/design-templates/executive-proposal.yaml +92 -0
- package/.github/skills/presentation-planner/assets/design-templates/technical-briefing.yaml +96 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/data-report.md +47 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/executive-decision.md +43 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/technical-briefing.md +45 -0
- package/.github/skills/presentation-planner/references/customizing-design-templates.md +160 -0
- package/.github/skills/presentation-planner/references/design-spec-schema.md +72 -0
- package/.github/skills/presentation-planner/references/handoff-contract.md +49 -0
- package/.github/skills/presentation-planner/references/responsibility-boundary.md +32 -0
- package/.github/skills/presentation-planner/references/scenario-templates.md +55 -0
- package/.github/skills/tech-writer/SKILL.md +434 -0
- package/.github/skills/tech-writer/assets/templates/blueprint.md +187 -0
- package/.github/skills/tech-writer/assets/templates/design-doc.md +29 -0
- package/.github/skills/tech-writer/assets/templates/migration-plan.md +173 -0
- package/.github/skills/tech-writer/assets/templates/operations-runbook.md +202 -0
- package/.github/skills/tech-writer/assets/templates/pr-description.md +23 -0
- package/.github/skills/tech-writer/assets/templates/qiita.md +44 -0
- package/.github/skills/tech-writer/assets/templates/readme.md +38 -0
- package/.github/skills/tech-writer/assets/templates/requirements-definition.md +170 -0
- package/.github/skills/tech-writer/assets/templates/rfi.md +113 -0
- package/.github/skills/tech-writer/assets/templates/rfp.md +180 -0
- package/.github/skills/tech-writer/assets/templates/security-design.md +167 -0
- package/.github/skills/tech-writer/assets/templates/system-design.md +220 -0
- package/.github/skills/tech-writer/assets/templates/technical-proposal.md +112 -0
- package/.github/skills/tech-writer/assets/templates/test-plan.md +153 -0
- package/.github/skills/tech-writer/assets/templates/user-manual.md +22 -0
- package/.github/skills/tech-writer/assets/templates/white-paper.md +192 -0
- package/.github/skills/tech-writer/references/doctypes/api-docs.md +33 -0
- package/.github/skills/tech-writer/references/doctypes/blueprint.md +81 -0
- package/.github/skills/tech-writer/references/doctypes/code-comments.md +39 -0
- package/.github/skills/tech-writer/references/doctypes/design-doc.md +42 -0
- package/.github/skills/tech-writer/references/doctypes/migration-plan.md +63 -0
- package/.github/skills/tech-writer/references/doctypes/operations-runbook.md +63 -0
- package/.github/skills/tech-writer/references/doctypes/pr-commit.md +82 -0
- package/.github/skills/tech-writer/references/doctypes/qiita.md +75 -0
- package/.github/skills/tech-writer/references/doctypes/readme.md +43 -0
- package/.github/skills/tech-writer/references/doctypes/release-notes.md +30 -0
- package/.github/skills/tech-writer/references/doctypes/requirements-definition.md +61 -0
- package/.github/skills/tech-writer/references/doctypes/rfi.md +43 -0
- package/.github/skills/tech-writer/references/doctypes/rfp.md +46 -0
- package/.github/skills/tech-writer/references/doctypes/security-design.md +71 -0
- package/.github/skills/tech-writer/references/doctypes/system-design.md +74 -0
- package/.github/skills/tech-writer/references/doctypes/technical-proposal.md +49 -0
- package/.github/skills/tech-writer/references/doctypes/test-plan.md +67 -0
- package/.github/skills/tech-writer/references/doctypes/user-manual.md +58 -0
- package/.github/skills/tech-writer/references/doctypes/white-paper.md +84 -0
- package/.github/skills/tech-writer/references/doctypes/zenn.md +66 -0
- package/.github/skills/tech-writer/references/japanese-prose-optimization.md +110 -0
- package/.github/skills/tech-writer/references/style-constitution.md +104 -0
- package/.github/skills/tech-writer/scripts/lint.py +412 -0
- package/LICENSE +21 -0
- package/README.md +92 -0
- package/bin/ai-data-scientist.js +123 -0
- package/package.json +41 -0
- package/pyproject.toml +45 -0
- package/src/ai_chemistry_scientist/__init__.py +0 -0
- package/src/ai_chemistry_scientist/admet_prediction.py +71 -0
- package/src/ai_chemistry_scientist/bioactivity_classification.py +73 -0
- package/src/ai_chemistry_scientist/data/sample_molecules.csv +21 -0
- package/src/ai_chemistry_scientist/dispatch.py +369 -0
- package/src/ai_chemistry_scientist/docking_score.py +97 -0
- package/src/ai_chemistry_scientist/drug_likeness_rules.py +84 -0
- package/src/ai_chemistry_scientist/evidence.py +41 -0
- package/src/ai_chemistry_scientist/molecular_descriptors.py +97 -0
- package/src/ai_chemistry_scientist/molecular_formula_mass.py +40 -0
- package/src/ai_chemistry_scientist/molecular_similarity.py +78 -0
- package/src/ai_chemistry_scientist/qsar_modeling.py +105 -0
- package/src/ai_chemistry_scientist/salt_standardization.py +81 -0
- package/src/ai_chemistry_scientist/structural_alerts.py +76 -0
- package/src/ai_chemistry_scientist/structure_format_conversion.py +84 -0
- package/src/ai_chemistry_scientist/validation.py +70 -0
- package/src/ai_data_scientist/__init__.py +0 -0
- package/src/ai_data_scientist/analysis_assumptions.py +121 -0
- package/src/ai_data_scientist/anomaly_detection.py +39 -0
- package/src/ai_data_scientist/automl.py +109 -0
- package/src/ai_data_scientist/cleaning.py +56 -0
- package/src/ai_data_scientist/cli.py +90 -0
- package/src/ai_data_scientist/clustering.py +54 -0
- package/src/ai_data_scientist/dashboard.py +33 -0
- package/src/ai_data_scientist/data_definition.py +100 -0
- package/src/ai_data_scientist/data_quality.py +164 -0
- package/src/ai_data_scientist/dataset_validation.py +135 -0
- package/src/ai_data_scientist/dependency_pins.py +60 -0
- package/src/ai_data_scientist/eda.py +82 -0
- package/src/ai_data_scientist/experiment_evaluation.py +635 -0
- package/src/ai_data_scientist/explainability.py +340 -0
- package/src/ai_data_scientist/feature_engineering.py +163 -0
- package/src/ai_data_scientist/gate_config.py +32 -0
- package/src/ai_data_scientist/ingestion.py +127 -0
- package/src/ai_data_scientist/insight_engine.py +180 -0
- package/src/ai_data_scientist/japanese_nlp.py +43 -0
- package/src/ai_data_scientist/jupyter_launcher.py +137 -0
- package/src/ai_data_scientist/jupyter_mcp_client.py +94 -0
- package/src/ai_data_scientist/language_router.py +28 -0
- package/src/ai_data_scientist/lifecycle.py +221 -0
- package/src/ai_data_scientist/mcp_gateway.py +113 -0
- package/src/ai_data_scientist/mcp_runtime.py +194 -0
- package/src/ai_data_scientist/mcp_transport.py +53 -0
- package/src/ai_data_scientist/ml_modeling.py +451 -0
- package/src/ai_data_scientist/model_tuning.py +104 -0
- package/src/ai_data_scientist/notebook_audit.py +574 -0
- package/src/ai_data_scientist/project_manager.py +243 -0
- package/src/ai_data_scientist/report_export.py +73 -0
- package/src/ai_data_scientist/sensitivity.py +445 -0
- package/src/ai_data_scientist/signal_analysis.py +201 -0
- package/src/ai_data_scientist/skill_packaging.py +40 -0
- package/src/ai_data_scientist/stats_analysis.py +88 -0
- package/src/ai_data_scientist/text_nlp.py +44 -0
- package/src/ai_data_scientist/timeseries.py +68 -0
- package/src/ai_data_scientist/visualization.py +708 -0
- package/src/ai_genomics_scientist/__init__.py +1 -0
- package/src/ai_genomics_scientist/differential_expression.py +147 -0
- package/src/ai_genomics_scientist/dispatch.py +267 -0
- package/src/ai_genomics_scientist/evidence.py +45 -0
- package/src/ai_genomics_scientist/gene_set_enrichment.py +76 -0
- package/src/ai_genomics_scientist/sequence_alignment.py +97 -0
- package/src/ai_genomics_scientist/sequence_features.py +111 -0
- package/src/ai_genomics_scientist/splice_site_scoring.py +66 -0
- package/src/ai_genomics_scientist/validation.py +83 -0
- package/src/ai_genomics_scientist/variant_effect.py +147 -0
- package/src/ai_genomics_scientist/variant_pathogenicity.py +125 -0
- package/src/ai_materials_scientist/__init__.py +0 -0
- package/src/ai_materials_scientist/calphad.py +117 -0
- package/src/ai_materials_scientist/classical_monte_carlo.py +165 -0
- package/src/ai_materials_scientist/crystal_plasticity.py +184 -0
- package/src/ai_materials_scientist/dispatch.py +100 -0
- package/src/ai_materials_scientist/evidence.py +84 -0
- package/src/ai_materials_scientist/fem.py +279 -0
- package/src/ai_materials_scientist/kinetic_monte_carlo.py +145 -0
- package/src/ai_materials_scientist/molecular_dynamics.py +240 -0
- package/src/ai_materials_scientist/phase_field.py +167 -0
- package/src/ai_materials_scientist/validation.py +70 -0
- package/src/ai_scientist/__init__.py +1 -0
- package/src/ai_scientist/completion_gate.py +15 -0
- package/src/ai_scientist/data_analysis.py +46 -0
- package/src/ai_scientist/evidence_registry.py +99 -0
- package/src/ai_scientist/experimental_design.py +20 -0
- package/src/ai_scientist/language.py +14 -0
- package/src/ai_scientist/latex_renderer.py +41 -0
- package/src/ai_scientist/literature_review.py +37 -0
- package/src/ai_scientist/manifest.py +87 -0
- package/src/ai_scientist/manuscript.py +94 -0
- package/src/ai_scientist/mcp_config.py +76 -0
- package/src/ai_scientist/mcp_external.py +42 -0
- package/src/ai_scientist/mcp_failures.py +23 -0
- package/src/ai_scientist/mcp_gateway.py +38 -0
- package/src/ai_scientist/mcp_managed.py +180 -0
- package/src/ai_scientist/npm_packaging.py +49 -0
- package/src/ai_scientist/orchestrator.py +133 -0
- package/src/ai_scientist/peer_review.py +60 -0
- package/src/ai_scientist/phase_gate.py +74 -0
- package/src/ai_scientist/phase_state.py +230 -0
- package/src/ai_scientist/presentation.py +56 -0
- package/src/ai_scientist/project_config.py +31 -0
- package/src/ai_scientist/project_handle.py +74 -0
- package/src/ai_scientist/reproducibility.py +20 -0
- package/src/ai_scientist/research_planning.py +20 -0
- package/src/ai_scientist/skill_invocation.py +21 -0
- package/src/ai_scientist/tdd_gate.py +99 -0
- package/src/ai_structural_biology_scientist/__init__.py +0 -0
- package/src/ai_structural_biology_scientist/contact_map.py +87 -0
- package/src/ai_structural_biology_scientist/dispatch.py +269 -0
- package/src/ai_structural_biology_scientist/evidence.py +43 -0
- package/src/ai_structural_biology_scientist/hydrophobicity.py +101 -0
- package/src/ai_structural_biology_scientist/protein_docking_score.py +104 -0
- package/src/ai_structural_biology_scientist/secondary_structure.py +95 -0
- package/src/ai_structural_biology_scientist/structural_similarity.py +74 -0
- package/src/ai_structural_biology_scientist/validation.py +100 -0
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""Drug-likeness rule screening module (DES-ACHEM-060 / REQ-ACHEM-060)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from rdkit.Chem import Descriptors, rdMolDescriptors
|
|
6
|
+
|
|
7
|
+
from ai_chemistry_scientist.molecular_descriptors import compute_descriptors, parse_smiles
|
|
8
|
+
from ai_chemistry_scientist.validation import fail, ok, register_validator
|
|
9
|
+
|
|
10
|
+
_MODULE_NAME = "drug-likeness-rules"
|
|
11
|
+
|
|
12
|
+
# Named Ghose/Egan threshold constants (DES-ACHEM-060) so the fixed rule
|
|
13
|
+
# boundaries are documented in one place instead of as inline magic numbers.
|
|
14
|
+
GHOSE_MOLWT_MIN, GHOSE_MOLWT_MAX = 160, 480
|
|
15
|
+
GHOSE_MOLLOGP_MIN, GHOSE_MOLLOGP_MAX = -0.4, 5.6
|
|
16
|
+
GHOSE_MOLMR_MIN, GHOSE_MOLMR_MAX = 40, 130
|
|
17
|
+
GHOSE_HEAVY_ATOM_COUNT_MIN, GHOSE_HEAVY_ATOM_COUNT_MAX = 20, 70
|
|
18
|
+
EGAN_TPSA_MAX = 131.6
|
|
19
|
+
EGAN_MOLLOGP_MAX = 5.88
|
|
20
|
+
|
|
21
|
+
_GHOSE_CRITERIA = (
|
|
22
|
+
(
|
|
23
|
+
"MolWt",
|
|
24
|
+
lambda d, heavy_atom_count, mol_mr: GHOSE_MOLWT_MIN <= d["mol_wt"] <= GHOSE_MOLWT_MAX,
|
|
25
|
+
),
|
|
26
|
+
(
|
|
27
|
+
"MolLogP",
|
|
28
|
+
lambda d, heavy_atom_count, mol_mr: GHOSE_MOLLOGP_MIN <= d["mol_logp"] <= GHOSE_MOLLOGP_MAX,
|
|
29
|
+
),
|
|
30
|
+
("MolMR", lambda d, heavy_atom_count, mol_mr: GHOSE_MOLMR_MIN <= mol_mr <= GHOSE_MOLMR_MAX),
|
|
31
|
+
(
|
|
32
|
+
"heavy_atom_count",
|
|
33
|
+
lambda d, heavy_atom_count, mol_mr: (
|
|
34
|
+
GHOSE_HEAVY_ATOM_COUNT_MIN <= heavy_atom_count <= GHOSE_HEAVY_ATOM_COUNT_MAX
|
|
35
|
+
),
|
|
36
|
+
),
|
|
37
|
+
)
|
|
38
|
+
_EGAN_CRITERIA = (
|
|
39
|
+
("TPSA", lambda d: d["tpsa"] <= EGAN_TPSA_MAX),
|
|
40
|
+
("MolLogP", lambda d: d["mol_logp"] <= EGAN_MOLLOGP_MAX),
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
# @id CODE-ACHEM-917
|
|
45
|
+
# @implements REQ-ACHEM-003 REQ-ACHEM-060
|
|
46
|
+
# @design DES-ACHEM-002
|
|
47
|
+
def _drug_likeness_rules_validator(params: dict) -> dict:
|
|
48
|
+
if "smiles" not in params:
|
|
49
|
+
return fail("smiles", "is required")
|
|
50
|
+
if parse_smiles(params["smiles"]) is None:
|
|
51
|
+
return fail("smiles", "must parse to a valid RDKit molecule")
|
|
52
|
+
return ok()
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
register_validator(_MODULE_NAME, _drug_likeness_rules_validator)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
# @id CODE-ACHEM-060
|
|
59
|
+
# @implements REQ-ACHEM-060
|
|
60
|
+
# @design DES-ACHEM-060
|
|
61
|
+
def run_drug_likeness_rules(smiles: str) -> dict:
|
|
62
|
+
"""Compute fixed-threshold Ghose and Egan rule outcomes for ``smiles``."""
|
|
63
|
+
mol = parse_smiles(smiles)
|
|
64
|
+
if mol is None:
|
|
65
|
+
raise ValueError("smiles must already be validated by the handler wrapper")
|
|
66
|
+
descriptors = compute_descriptors(mol)
|
|
67
|
+
aromatic_ring_count = int(rdMolDescriptors.CalcNumAromaticRings(mol))
|
|
68
|
+
mol_mr = float(Descriptors.MolMR(mol))
|
|
69
|
+
heavy_atom_count = int(mol.GetNumHeavyAtoms())
|
|
70
|
+
|
|
71
|
+
ghose_violations = [
|
|
72
|
+
name for name, check in _GHOSE_CRITERIA if not check(descriptors, heavy_atom_count, mol_mr)
|
|
73
|
+
]
|
|
74
|
+
egan_violations = [name for name, check in _EGAN_CRITERIA if not check(descriptors)]
|
|
75
|
+
|
|
76
|
+
return {
|
|
77
|
+
"aromatic_ring_count": aromatic_ring_count,
|
|
78
|
+
"mol_mr": mol_mr,
|
|
79
|
+
"heavy_atom_count": heavy_atom_count,
|
|
80
|
+
"ghose_violations": ghose_violations,
|
|
81
|
+
"ghose_pass": len(ghose_violations) == 0,
|
|
82
|
+
"egan_violations": egan_violations,
|
|
83
|
+
"egan_pass": len(egan_violations) == 0,
|
|
84
|
+
}
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Run evidence recorder (DES-ACHEM-003 / REQ-ACHEM-004)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import copy
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
SCHEMA_VERSION = 1
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
# @id CODE-ACHEM-003
|
|
12
|
+
# @implements REQ-ACHEM-004
|
|
13
|
+
# @design DES-ACHEM-003
|
|
14
|
+
def record_run(
|
|
15
|
+
module_name: str,
|
|
16
|
+
params: dict[str, Any],
|
|
17
|
+
result: Any,
|
|
18
|
+
*,
|
|
19
|
+
rdkit_version: str,
|
|
20
|
+
scikit_learn_version: str | None = None,
|
|
21
|
+
) -> dict:
|
|
22
|
+
"""Build a RunRecord with exactly `metadata`, `parameters`, `result`.
|
|
23
|
+
|
|
24
|
+
Every module's ``result`` is already JSON-safe (scalars, strings, lists,
|
|
25
|
+
and nested dicts of these), so no ndarray codec is needed here (unlike
|
|
26
|
+
`ai_materials_scientist.evidence`, ADR-0027). ``scikit_learn_version`` is
|
|
27
|
+
included in `metadata` only when supplied (the QSAR module always
|
|
28
|
+
supplies it; the other 4 modules omit it).
|
|
29
|
+
"""
|
|
30
|
+
metadata: dict[str, Any] = {
|
|
31
|
+
"module": module_name,
|
|
32
|
+
"schema_version": SCHEMA_VERSION,
|
|
33
|
+
"rdkit_version": rdkit_version,
|
|
34
|
+
}
|
|
35
|
+
if scikit_learn_version is not None:
|
|
36
|
+
metadata["scikit_learn_version"] = scikit_learn_version
|
|
37
|
+
return {
|
|
38
|
+
"metadata": metadata,
|
|
39
|
+
"parameters": dict(params),
|
|
40
|
+
"result": copy.deepcopy(result),
|
|
41
|
+
}
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Molecular descriptor calculation module (DES-ACHEM-010 / REQ-ACHEM-010)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from rdkit import Chem
|
|
6
|
+
from rdkit.Chem import Descriptors, rdMolDescriptors
|
|
7
|
+
|
|
8
|
+
from ai_chemistry_scientist.validation import (
|
|
9
|
+
fail,
|
|
10
|
+
ok,
|
|
11
|
+
register_batch_item_validator,
|
|
12
|
+
validate_batch_item,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
_MODULE_NAME = "molecular-descriptors"
|
|
16
|
+
|
|
17
|
+
#: The 7 descriptors every molecule in this skill is characterized by
|
|
18
|
+
#: (REQ-ACHEM-010). Reused by the descriptor needs of DES-ACHEM-020/030.
|
|
19
|
+
DESCRIPTOR_NAMES = (
|
|
20
|
+
"mol_wt",
|
|
21
|
+
"mol_logp",
|
|
22
|
+
"tpsa",
|
|
23
|
+
"num_h_donors",
|
|
24
|
+
"num_h_acceptors",
|
|
25
|
+
"num_rotatable_bonds",
|
|
26
|
+
"num_rings",
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def parse_smiles(smiles: str):
|
|
31
|
+
"""Parse ``smiles`` with RDKit, returning ``None`` on failure."""
|
|
32
|
+
if not isinstance(smiles, str) or not smiles:
|
|
33
|
+
return None
|
|
34
|
+
mol = Chem.MolFromSmiles(smiles)
|
|
35
|
+
if mol is None:
|
|
36
|
+
return None
|
|
37
|
+
if any(atom.GetAtomicNum() == 0 for atom in mol.GetAtoms()):
|
|
38
|
+
return None
|
|
39
|
+
return mol
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def compute_descriptors(mol) -> dict:
|
|
43
|
+
"""Compute the 7 documented descriptors for an already-parsed ``mol``."""
|
|
44
|
+
return {
|
|
45
|
+
"mol_wt": float(Descriptors.MolWt(mol)),
|
|
46
|
+
"mol_logp": float(Descriptors.MolLogP(mol)),
|
|
47
|
+
"tpsa": float(Descriptors.TPSA(mol)),
|
|
48
|
+
"num_h_donors": int(Descriptors.NumHDonors(mol)),
|
|
49
|
+
"num_h_acceptors": int(Descriptors.NumHAcceptors(mol)),
|
|
50
|
+
"num_rotatable_bonds": int(Descriptors.NumRotatableBonds(mol)),
|
|
51
|
+
"num_rings": int(rdMolDescriptors.CalcNumRings(mol)),
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _molecular_descriptors_batch_item_validator(item_params: dict) -> dict:
|
|
56
|
+
"""DES-ACHEM-002 registered per-item validator for this module."""
|
|
57
|
+
if "smiles" not in item_params:
|
|
58
|
+
return fail("smiles", "is required")
|
|
59
|
+
smiles = item_params["smiles"]
|
|
60
|
+
if parse_smiles(smiles) is None:
|
|
61
|
+
return fail("smiles", "must parse to a valid RDKit molecule")
|
|
62
|
+
return ok()
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
register_batch_item_validator(_MODULE_NAME, _molecular_descriptors_batch_item_validator)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
# @id CODE-ACHEM-010
|
|
69
|
+
# @implements REQ-ACHEM-010 REQ-ACHEM-003
|
|
70
|
+
# @design DES-ACHEM-010
|
|
71
|
+
def run_molecular_descriptors(smiles_list: list[str]) -> list[dict]:
|
|
72
|
+
"""Parse and compute descriptors for each SMILES, in input order.
|
|
73
|
+
|
|
74
|
+
An invalid item is reported as a per-item rejection (naming the
|
|
75
|
+
``smiles`` parameter and the violated constraint) without aborting
|
|
76
|
+
computation of the rest of the batch (REQ-ACHEM-003's per-item
|
|
77
|
+
granularity for this module).
|
|
78
|
+
"""
|
|
79
|
+
if not smiles_list:
|
|
80
|
+
return []
|
|
81
|
+
results = []
|
|
82
|
+
for smiles in smiles_list:
|
|
83
|
+
validation = validate_batch_item(_MODULE_NAME, {"smiles": smiles})
|
|
84
|
+
if not validation["ok"]:
|
|
85
|
+
results.append(
|
|
86
|
+
{
|
|
87
|
+
"smiles": smiles,
|
|
88
|
+
"ok": False,
|
|
89
|
+
"parameter": validation["parameter"],
|
|
90
|
+
"constraint": validation["constraint"],
|
|
91
|
+
}
|
|
92
|
+
)
|
|
93
|
+
continue
|
|
94
|
+
mol = parse_smiles(smiles)
|
|
95
|
+
descriptors = compute_descriptors(mol)
|
|
96
|
+
results.append({"smiles": smiles, **descriptors})
|
|
97
|
+
return results
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Molecular formula and exact mass module (DES-ACHEM-080 / REQ-ACHEM-080)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from rdkit.Chem import Descriptors, rdMolDescriptors
|
|
6
|
+
|
|
7
|
+
from ai_chemistry_scientist.molecular_descriptors import parse_smiles
|
|
8
|
+
from ai_chemistry_scientist.validation import fail, ok, register_validator
|
|
9
|
+
|
|
10
|
+
_MODULE_NAME = "molecular-formula-mass"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
# @id CODE-ACHEM-919
|
|
14
|
+
# @implements REQ-ACHEM-003 REQ-ACHEM-080
|
|
15
|
+
# @design DES-ACHEM-002
|
|
16
|
+
def _molecular_formula_mass_validator(params: dict) -> dict:
|
|
17
|
+
if "smiles" not in params:
|
|
18
|
+
return fail("smiles", "is required")
|
|
19
|
+
if parse_smiles(params["smiles"]) is None:
|
|
20
|
+
return fail("smiles", "must parse to a valid RDKit molecule")
|
|
21
|
+
return ok()
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
register_validator(_MODULE_NAME, _molecular_formula_mass_validator)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
# @id CODE-ACHEM-080
|
|
28
|
+
# @implements REQ-ACHEM-080
|
|
29
|
+
# @design DES-ACHEM-080
|
|
30
|
+
def run_molecular_formula_mass(smiles: str) -> dict:
|
|
31
|
+
"""Compute the molecular formula and exact mass for ``smiles``."""
|
|
32
|
+
mol = parse_smiles(smiles)
|
|
33
|
+
if mol is None:
|
|
34
|
+
raise ValueError("smiles must already be validated by the handler wrapper")
|
|
35
|
+
molecular_formula = rdMolDescriptors.CalcMolFormula(mol)
|
|
36
|
+
exact_mass = float(Descriptors.ExactMolWt(mol))
|
|
37
|
+
return {
|
|
38
|
+
"molecular_formula": molecular_formula,
|
|
39
|
+
"exact_mass": exact_mass,
|
|
40
|
+
}
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Molecular similarity search module (DES-ACHEM-040 / REQ-ACHEM-040)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from rdkit import DataStructs
|
|
9
|
+
from rdkit.Chem import rdFingerprintGenerator
|
|
10
|
+
|
|
11
|
+
from ai_chemistry_scientist.molecular_descriptors import parse_smiles
|
|
12
|
+
from ai_chemistry_scientist.validation import fail, ok, register_validator
|
|
13
|
+
|
|
14
|
+
_MODULE_NAME = "molecular-similarity"
|
|
15
|
+
_DATA_PATH = Path(__file__).resolve().parent / "data" / "sample_molecules.csv"
|
|
16
|
+
_MIN_K = 1
|
|
17
|
+
_MAX_K = 20
|
|
18
|
+
_MORGAN_GENERATOR = rdFingerprintGenerator.GetMorganGenerator(radius=2, fpSize=2048)
|
|
19
|
+
|
|
20
|
+
_dataset_cache: list[dict] | None = None
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _load_dataset() -> list[dict]:
|
|
24
|
+
"""Load and cache the bundled 20-row dataset (module-level, read-only)."""
|
|
25
|
+
global _dataset_cache
|
|
26
|
+
if _dataset_cache is None:
|
|
27
|
+
with _DATA_PATH.open(encoding="utf-8", newline="") as handle:
|
|
28
|
+
rows = list(csv.DictReader(handle))
|
|
29
|
+
_dataset_cache = [
|
|
30
|
+
{
|
|
31
|
+
"name": row["name"],
|
|
32
|
+
"smiles": row["smiles"],
|
|
33
|
+
"fingerprint": _MORGAN_GENERATOR.GetFingerprint(parse_smiles(row["smiles"])),
|
|
34
|
+
}
|
|
35
|
+
for row in rows
|
|
36
|
+
]
|
|
37
|
+
return _dataset_cache
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _molecular_similarity_validator(params: dict) -> dict:
|
|
41
|
+
"""DES-ACHEM-002 registered atomic validator for this module."""
|
|
42
|
+
if "query_smiles" not in params:
|
|
43
|
+
return fail("query_smiles", "is required")
|
|
44
|
+
if "k" not in params:
|
|
45
|
+
return fail("k", "is required")
|
|
46
|
+
if parse_smiles(params["query_smiles"]) is None:
|
|
47
|
+
return fail("smiles", "must parse to a valid RDKit molecule")
|
|
48
|
+
k = params["k"]
|
|
49
|
+
if not isinstance(k, int) or isinstance(k, bool) or not (_MIN_K <= k <= _MAX_K):
|
|
50
|
+
return fail("k", "must be an integer in [1, 20]")
|
|
51
|
+
return ok()
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
register_validator(_MODULE_NAME, _molecular_similarity_validator)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
# @id CODE-ACHEM-040
|
|
58
|
+
# @implements REQ-ACHEM-040
|
|
59
|
+
# @design DES-ACHEM-040
|
|
60
|
+
def run_molecular_similarity(query_smiles: str, k: int = 5) -> dict:
|
|
61
|
+
"""Return the top-``k`` dataset entries by descending Tanimoto similarity.
|
|
62
|
+
|
|
63
|
+
Receives ``query_smiles``/``k`` already validated atomically by its
|
|
64
|
+
handler wrapper; performs no revalidation of its own.
|
|
65
|
+
"""
|
|
66
|
+
query_mol = parse_smiles(query_smiles)
|
|
67
|
+
query_fp = _MORGAN_GENERATOR.GetFingerprint(query_mol)
|
|
68
|
+
|
|
69
|
+
scored = [
|
|
70
|
+
{
|
|
71
|
+
"name": entry["name"],
|
|
72
|
+
"similarity": DataStructs.TanimotoSimilarity(query_fp, entry["fingerprint"]),
|
|
73
|
+
}
|
|
74
|
+
for entry in _load_dataset()
|
|
75
|
+
]
|
|
76
|
+
scored.sort(key=lambda entry: (-entry["similarity"], entry["name"]))
|
|
77
|
+
|
|
78
|
+
return {"query_smiles": query_smiles, "results": scored[:k]}
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""QSAR linear-regression module (DES-ACHEM-030 / REQ-ACHEM-030)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
from sklearn.linear_model import LinearRegression
|
|
7
|
+
|
|
8
|
+
from ai_chemistry_scientist.molecular_descriptors import compute_descriptors, parse_smiles
|
|
9
|
+
from ai_chemistry_scientist.validation import fail, ok, register_validator
|
|
10
|
+
|
|
11
|
+
_MODULE_NAME = "qsar-modeling"
|
|
12
|
+
_MIN_TRAINING_COMPOUNDS = 5
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _design_row(descriptors: dict) -> list[float]:
|
|
16
|
+
return [1.0, descriptors["mol_wt"], descriptors["mol_logp"], descriptors["tpsa"]]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _qsar_modeling_validator(params: dict) -> dict:
|
|
20
|
+
"""DES-ACHEM-002 registered atomic validator for this module.
|
|
21
|
+
|
|
22
|
+
Validates every training/query SMILES parses, the training set has at
|
|
23
|
+
least 5 compounds, and its augmented design matrix is full column rank
|
|
24
|
+
4. This descriptor extraction is part of validation itself, distinct
|
|
25
|
+
from the governed "model fit" computation (REQ-ACHEM-003, ADR-0030).
|
|
26
|
+
"""
|
|
27
|
+
training_set = params.get("training_set")
|
|
28
|
+
query_smiles_list = params.get("query_smiles_list")
|
|
29
|
+
|
|
30
|
+
if training_set is None:
|
|
31
|
+
return fail("training_set", "is required")
|
|
32
|
+
if query_smiles_list is None:
|
|
33
|
+
return fail("query_smiles_list", "is required")
|
|
34
|
+
|
|
35
|
+
if len(training_set) < _MIN_TRAINING_COMPOUNDS:
|
|
36
|
+
return fail(
|
|
37
|
+
"training_set",
|
|
38
|
+
"must contain at least 5 compounds with a full-rank descriptor matrix",
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
design_matrix = []
|
|
42
|
+
for compound in training_set:
|
|
43
|
+
mol = parse_smiles(compound["smiles"])
|
|
44
|
+
if mol is None:
|
|
45
|
+
return fail("training_set", "must parse to a valid RDKit molecule")
|
|
46
|
+
design_matrix.append(_design_row(compute_descriptors(mol)))
|
|
47
|
+
|
|
48
|
+
rank = np.linalg.matrix_rank(np.array(design_matrix, dtype=np.float64))
|
|
49
|
+
if rank < 4:
|
|
50
|
+
return fail(
|
|
51
|
+
"training_set",
|
|
52
|
+
"must contain at least 5 compounds with a full-rank descriptor matrix",
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
for query_smiles in query_smiles_list:
|
|
56
|
+
if parse_smiles(query_smiles) is None:
|
|
57
|
+
return fail("query_smiles_list", "must parse to a valid RDKit molecule")
|
|
58
|
+
|
|
59
|
+
return ok()
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
register_validator(_MODULE_NAME, _qsar_modeling_validator)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# @id CODE-ACHEM-030
|
|
66
|
+
# @implements REQ-ACHEM-030
|
|
67
|
+
# @design DES-ACHEM-030
|
|
68
|
+
def run_qsar_modeling(training_set: list[dict], query_smiles_list: list[str]) -> dict:
|
|
69
|
+
"""Fit an OLS linear regression and predict each query's activity.
|
|
70
|
+
|
|
71
|
+
Receives ``training_set``/``query_smiles_list`` already validated
|
|
72
|
+
atomically by its handler wrapper; independently (re)computes each
|
|
73
|
+
molecule's descriptors here for the actual fit/prediction, since
|
|
74
|
+
`validate_parameters`'s boolean-shaped return carries no descriptor
|
|
75
|
+
payload to reuse (ADR-0030) — this duplication is intentional and
|
|
76
|
+
always agrees, since REQ-ACHEM-004 requires every computation to be a
|
|
77
|
+
deterministic pure function of its inputs.
|
|
78
|
+
"""
|
|
79
|
+
features = []
|
|
80
|
+
activities = []
|
|
81
|
+
for compound in training_set:
|
|
82
|
+
mol = parse_smiles(compound["smiles"])
|
|
83
|
+
descriptors = compute_descriptors(mol)
|
|
84
|
+
features.append([descriptors["mol_wt"], descriptors["mol_logp"], descriptors["tpsa"]])
|
|
85
|
+
activities.append(compound["activity"])
|
|
86
|
+
|
|
87
|
+
model = LinearRegression()
|
|
88
|
+
model.fit(np.array(features, dtype=np.float64), np.array(activities, dtype=np.float64))
|
|
89
|
+
|
|
90
|
+
predictions = []
|
|
91
|
+
for query_smiles in query_smiles_list:
|
|
92
|
+
mol = parse_smiles(query_smiles)
|
|
93
|
+
descriptors = compute_descriptors(mol)
|
|
94
|
+
feature_row = np.array(
|
|
95
|
+
[[descriptors["mol_wt"], descriptors["mol_logp"], descriptors["tpsa"]]],
|
|
96
|
+
dtype=np.float64,
|
|
97
|
+
)
|
|
98
|
+
predicted_activity = float(model.predict(feature_row)[0])
|
|
99
|
+
predictions.append({"smiles": query_smiles, "predicted_activity": predicted_activity})
|
|
100
|
+
|
|
101
|
+
return {
|
|
102
|
+
"coefficients": [float(c) for c in model.coef_],
|
|
103
|
+
"intercept": float(model.intercept_),
|
|
104
|
+
"predictions": predictions,
|
|
105
|
+
}
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"""SMILES salt removal / structure standardization module (DES-ACHEM-100 / REQ-ACHEM-100)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from rdkit import Chem
|
|
6
|
+
|
|
7
|
+
from ai_chemistry_scientist.molecular_descriptors import parse_smiles
|
|
8
|
+
from ai_chemistry_scientist.validation import fail, ok, register_validator
|
|
9
|
+
|
|
10
|
+
_MODULE_NAME = "salt-removal"
|
|
11
|
+
|
|
12
|
+
LIMITATION_LABEL_KEY = "salt_removal_heuristic_limitation"
|
|
13
|
+
LIMITATION_LABEL_TEXT = {
|
|
14
|
+
"en": (
|
|
15
|
+
"Heuristic only: treats every disconnected fragment except the one "
|
|
16
|
+
"with the greatest heavy-atom count as removable salt/solvent; not "
|
|
17
|
+
"always chemically correct (e.g. for a genuine covalent "
|
|
18
|
+
"multi-component cocrystal)."
|
|
19
|
+
),
|
|
20
|
+
"ja": (
|
|
21
|
+
"ヒューリスティックのみ: 最大重原子数を持つフラグメント以外のすべての"
|
|
22
|
+
"分離フラグメントを除去可能な塩・溶媒として扱うが、常に化学的に"
|
|
23
|
+
"正しいとは限らない(例: 真の共有結合性多成分共結晶の場合)。"
|
|
24
|
+
),
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
# @id CODE-ACHEM-921
|
|
29
|
+
# @implements REQ-ACHEM-003 REQ-ACHEM-100
|
|
30
|
+
# @design DES-ACHEM-002
|
|
31
|
+
def _salt_removal_validator(params: dict) -> dict:
|
|
32
|
+
"""Reject a missing/unparseable/dummy-atom ``smiles`` via `parse_smiles`.
|
|
33
|
+
|
|
34
|
+
Unlike `structure_format_conversion._parse_structure`'s Molblock parser,
|
|
35
|
+
`parse_smiles` never needs an explicit zero-atom-molecule guard: RDKit's
|
|
36
|
+
SMILES parser rejects the only syntactically valid empty input (the
|
|
37
|
+
empty string) before reaching `Chem.MolFromSmiles` (the leading
|
|
38
|
+
not-`smiles` check), so every string it accepts yields at least one atom.
|
|
39
|
+
|
|
40
|
+
The required-key check runs first so a missing ``smiles`` parameter is
|
|
41
|
+
reported as a missing-parameter error rather than a parse failure.
|
|
42
|
+
"""
|
|
43
|
+
if "smiles" not in params:
|
|
44
|
+
return fail("smiles", "is required")
|
|
45
|
+
if parse_smiles(params["smiles"]) is None:
|
|
46
|
+
return fail("smiles", "must parse to a valid RDKit molecule")
|
|
47
|
+
return ok()
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
register_validator(_MODULE_NAME, _salt_removal_validator)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _fragment_sort_key(fragment) -> tuple[int, str]:
|
|
54
|
+
"""`(-heavy_atom_count, canonical_smiles)` ascending (REQ-ACHEM-100/ADR-0105)."""
|
|
55
|
+
return (-fragment.GetNumHeavyAtoms(), Chem.MolToSmiles(fragment))
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
# @id CODE-ACHEM-100
|
|
59
|
+
# @implements REQ-ACHEM-100
|
|
60
|
+
# @design DES-ACHEM-100
|
|
61
|
+
def run_salt_removal(smiles: str) -> dict:
|
|
62
|
+
"""Select the dominant fragment of ``smiles`` and report the rest as removed.
|
|
63
|
+
|
|
64
|
+
Fragment selection follows the fixed sort key
|
|
65
|
+
``(-heavy_atom_count, canonical_smiles)`` (ADR-0105), not RDKit's
|
|
66
|
+
built-in ``rdMolStandardize.LargestFragmentChooser`` heuristic.
|
|
67
|
+
"""
|
|
68
|
+
mol = parse_smiles(smiles)
|
|
69
|
+
if mol is None:
|
|
70
|
+
raise ValueError("smiles must already be validated by the handler wrapper")
|
|
71
|
+
|
|
72
|
+
fragments = Chem.GetMolFrags(mol, asMols=True, sanitizeFrags=False)
|
|
73
|
+
ranked = sorted(fragments, key=_fragment_sort_key)
|
|
74
|
+
ranked_smiles = [Chem.MolToSmiles(fragment) for fragment in ranked]
|
|
75
|
+
|
|
76
|
+
return {
|
|
77
|
+
"standardized_smiles": ranked_smiles[0],
|
|
78
|
+
"removed_fragments": ranked_smiles[1:],
|
|
79
|
+
"fragments_removed": len(ranked_smiles) > 1,
|
|
80
|
+
"limitation_label_key": LIMITATION_LABEL_KEY,
|
|
81
|
+
}
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Structural alert screening module (DES-ACHEM-070 / REQ-ACHEM-070)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from rdkit import Chem
|
|
6
|
+
|
|
7
|
+
from ai_chemistry_scientist.molecular_descriptors import parse_smiles
|
|
8
|
+
from ai_chemistry_scientist.validation import fail, ok, register_validator
|
|
9
|
+
|
|
10
|
+
_MODULE_NAME = "structural-alerts"
|
|
11
|
+
ALERT_SMARTS = (
|
|
12
|
+
("nitro_group", "[NX3](=O)=O"),
|
|
13
|
+
("aldehyde", "[CX3H1](=O)"),
|
|
14
|
+
("michael_acceptor_enone", "C=CC(=O)"),
|
|
15
|
+
("epoxide", "C1OC1"),
|
|
16
|
+
("free_thiol", "[SX2H]"),
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _compile_alert_queries():
|
|
21
|
+
compiled_queries = []
|
|
22
|
+
for name, smarts in ALERT_SMARTS:
|
|
23
|
+
query = Chem.MolFromSmarts(smarts)
|
|
24
|
+
if query is None:
|
|
25
|
+
raise ValueError(f"structural alert SMARTS for {name!r} failed to compile: {smarts}")
|
|
26
|
+
compiled_queries.append((name, query))
|
|
27
|
+
return tuple(compiled_queries)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# Pre-compiled once at import time (the DES-ACHEM-070 fixed alert list never
|
|
31
|
+
# changes at runtime), rather than re-parsing every SMARTS pattern string on
|
|
32
|
+
# every call.
|
|
33
|
+
_ALERT_QUERIES = _compile_alert_queries()
|
|
34
|
+
LIMITATION_LABEL_KEY = "structural_alerts_heuristic_limitation"
|
|
35
|
+
LIMITATION_LABEL_TEXT = {
|
|
36
|
+
"en": "Heuristic only: a small fixed illustrative SMARTS alert list, not the validated PAINS/Brenk filter catalog.",
|
|
37
|
+
"ja": (
|
|
38
|
+
"ヒューリスティックのみ: 固定の小規模な例示用SMARTSアラート一覧であり、"
|
|
39
|
+
"検証済みのPAINS/Brenkフィルタ・カタログではない。"
|
|
40
|
+
),
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
# @id CODE-ACHEM-918
|
|
45
|
+
# @implements REQ-ACHEM-003 REQ-ACHEM-070
|
|
46
|
+
# @design DES-ACHEM-002
|
|
47
|
+
def _structural_alerts_validator(params: dict) -> dict:
|
|
48
|
+
if "smiles" not in params:
|
|
49
|
+
return fail("smiles", "is required")
|
|
50
|
+
if parse_smiles(params["smiles"]) is None:
|
|
51
|
+
return fail("smiles", "must parse to a valid RDKit molecule")
|
|
52
|
+
return ok()
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
register_validator(_MODULE_NAME, _structural_alerts_validator)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
# @id CODE-ACHEM-070
|
|
59
|
+
# @implements REQ-ACHEM-070
|
|
60
|
+
# @design DES-ACHEM-070
|
|
61
|
+
def run_structural_alerts(smiles: str) -> dict:
|
|
62
|
+
"""Match the fixed SMARTS alert list against ``smiles``.
|
|
63
|
+
|
|
64
|
+
``_ALERT_QUERIES`` is guaranteed non-``None`` for every entry by
|
|
65
|
+
``_compile_alert_queries()`` at import time (CHANGE-013 hardening), so
|
|
66
|
+
this function never needs to re-check for a malformed SMARTS query.
|
|
67
|
+
"""
|
|
68
|
+
mol = parse_smiles(smiles)
|
|
69
|
+
if mol is None:
|
|
70
|
+
raise ValueError("smiles must already be validated by the handler wrapper")
|
|
71
|
+
alerts_matched = [name for name, query in _ALERT_QUERIES if mol.HasSubstructMatch(query)]
|
|
72
|
+
return {
|
|
73
|
+
"alerts_matched": alerts_matched,
|
|
74
|
+
"alert_count": len(alerts_matched),
|
|
75
|
+
"limitation_label_key": LIMITATION_LABEL_KEY,
|
|
76
|
+
}
|