jupytermind 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (193) hide show
  1. package/.github/skills/ai-chemistry-scientist/SKILL.md +97 -0
  2. package/.github/skills/ai-chemistry-scientist/manifest.json +156 -0
  3. package/.github/skills/ai-data-scientist/SKILL.md +330 -0
  4. package/.github/skills/ai-genomics-scientist/SKILL.md +98 -0
  5. package/.github/skills/ai-genomics-scientist/manifest.json +93 -0
  6. package/.github/skills/ai-materials-scientist/SKILL.md +51 -0
  7. package/.github/skills/ai-materials-scientist/manifest.json +58 -0
  8. package/.github/skills/ai-scientist/SKILL.md +69 -0
  9. package/.github/skills/ai-scientist/manifest.json +61 -0
  10. package/.github/skills/ai-structural-biology-scientist/SKILL.md +67 -0
  11. package/.github/skills/ai-structural-biology-scientist/manifest.json +72 -0
  12. package/.github/skills/japanese-prose/NOTICE.md +17 -0
  13. package/.github/skills/japanese-prose/SKILL.md +111 -0
  14. package/.github/skills/japanese-prose/references/review-workflow.md +50 -0
  15. package/.github/skills/japanese-prose/references/scoring.md +24 -0
  16. package/.github/skills/japanese-prose/references/writing-guidelines.md +60 -0
  17. package/.github/skills/japanese-prose/scripts/core.py +192 -0
  18. package/.github/skills/japanese-prose/scripts/fixtures/natural.md +5 -0
  19. package/.github/skills/japanese-prose/scripts/fixtures/unnatural.md +5 -0
  20. package/.github/skills/japanese-prose/scripts/lint.py +378 -0
  21. package/.github/skills/japanese-prose/scripts/outline.py +68 -0
  22. package/.github/skills/japanese-prose/scripts/terms.py +112 -0
  23. package/.github/skills/japanese-prose/scripts/test_engine.py +117 -0
  24. package/.github/skills/presentation-planner/SKILL.md +257 -0
  25. package/.github/skills/presentation-planner/assets/design-templates/data-report.yaml +97 -0
  26. package/.github/skills/presentation-planner/assets/design-templates/executive-proposal.yaml +92 -0
  27. package/.github/skills/presentation-planner/assets/design-templates/technical-briefing.yaml +96 -0
  28. package/.github/skills/presentation-planner/assets/scenario-templates/data-report.md +47 -0
  29. package/.github/skills/presentation-planner/assets/scenario-templates/executive-decision.md +43 -0
  30. package/.github/skills/presentation-planner/assets/scenario-templates/technical-briefing.md +45 -0
  31. package/.github/skills/presentation-planner/references/customizing-design-templates.md +160 -0
  32. package/.github/skills/presentation-planner/references/design-spec-schema.md +72 -0
  33. package/.github/skills/presentation-planner/references/handoff-contract.md +49 -0
  34. package/.github/skills/presentation-planner/references/responsibility-boundary.md +32 -0
  35. package/.github/skills/presentation-planner/references/scenario-templates.md +55 -0
  36. package/.github/skills/tech-writer/SKILL.md +434 -0
  37. package/.github/skills/tech-writer/assets/templates/blueprint.md +187 -0
  38. package/.github/skills/tech-writer/assets/templates/design-doc.md +29 -0
  39. package/.github/skills/tech-writer/assets/templates/migration-plan.md +173 -0
  40. package/.github/skills/tech-writer/assets/templates/operations-runbook.md +202 -0
  41. package/.github/skills/tech-writer/assets/templates/pr-description.md +23 -0
  42. package/.github/skills/tech-writer/assets/templates/qiita.md +44 -0
  43. package/.github/skills/tech-writer/assets/templates/readme.md +38 -0
  44. package/.github/skills/tech-writer/assets/templates/requirements-definition.md +170 -0
  45. package/.github/skills/tech-writer/assets/templates/rfi.md +113 -0
  46. package/.github/skills/tech-writer/assets/templates/rfp.md +180 -0
  47. package/.github/skills/tech-writer/assets/templates/security-design.md +167 -0
  48. package/.github/skills/tech-writer/assets/templates/system-design.md +220 -0
  49. package/.github/skills/tech-writer/assets/templates/technical-proposal.md +112 -0
  50. package/.github/skills/tech-writer/assets/templates/test-plan.md +153 -0
  51. package/.github/skills/tech-writer/assets/templates/user-manual.md +22 -0
  52. package/.github/skills/tech-writer/assets/templates/white-paper.md +192 -0
  53. package/.github/skills/tech-writer/references/doctypes/api-docs.md +33 -0
  54. package/.github/skills/tech-writer/references/doctypes/blueprint.md +81 -0
  55. package/.github/skills/tech-writer/references/doctypes/code-comments.md +39 -0
  56. package/.github/skills/tech-writer/references/doctypes/design-doc.md +42 -0
  57. package/.github/skills/tech-writer/references/doctypes/migration-plan.md +63 -0
  58. package/.github/skills/tech-writer/references/doctypes/operations-runbook.md +63 -0
  59. package/.github/skills/tech-writer/references/doctypes/pr-commit.md +82 -0
  60. package/.github/skills/tech-writer/references/doctypes/qiita.md +75 -0
  61. package/.github/skills/tech-writer/references/doctypes/readme.md +43 -0
  62. package/.github/skills/tech-writer/references/doctypes/release-notes.md +30 -0
  63. package/.github/skills/tech-writer/references/doctypes/requirements-definition.md +61 -0
  64. package/.github/skills/tech-writer/references/doctypes/rfi.md +43 -0
  65. package/.github/skills/tech-writer/references/doctypes/rfp.md +46 -0
  66. package/.github/skills/tech-writer/references/doctypes/security-design.md +71 -0
  67. package/.github/skills/tech-writer/references/doctypes/system-design.md +74 -0
  68. package/.github/skills/tech-writer/references/doctypes/technical-proposal.md +49 -0
  69. package/.github/skills/tech-writer/references/doctypes/test-plan.md +67 -0
  70. package/.github/skills/tech-writer/references/doctypes/user-manual.md +58 -0
  71. package/.github/skills/tech-writer/references/doctypes/white-paper.md +84 -0
  72. package/.github/skills/tech-writer/references/doctypes/zenn.md +66 -0
  73. package/.github/skills/tech-writer/references/japanese-prose-optimization.md +110 -0
  74. package/.github/skills/tech-writer/references/style-constitution.md +104 -0
  75. package/.github/skills/tech-writer/scripts/lint.py +412 -0
  76. package/LICENSE +21 -0
  77. package/README.md +92 -0
  78. package/bin/ai-data-scientist.js +123 -0
  79. package/package.json +41 -0
  80. package/pyproject.toml +45 -0
  81. package/src/ai_chemistry_scientist/__init__.py +0 -0
  82. package/src/ai_chemistry_scientist/admet_prediction.py +71 -0
  83. package/src/ai_chemistry_scientist/bioactivity_classification.py +73 -0
  84. package/src/ai_chemistry_scientist/data/sample_molecules.csv +21 -0
  85. package/src/ai_chemistry_scientist/dispatch.py +369 -0
  86. package/src/ai_chemistry_scientist/docking_score.py +97 -0
  87. package/src/ai_chemistry_scientist/drug_likeness_rules.py +84 -0
  88. package/src/ai_chemistry_scientist/evidence.py +41 -0
  89. package/src/ai_chemistry_scientist/molecular_descriptors.py +97 -0
  90. package/src/ai_chemistry_scientist/molecular_formula_mass.py +40 -0
  91. package/src/ai_chemistry_scientist/molecular_similarity.py +78 -0
  92. package/src/ai_chemistry_scientist/qsar_modeling.py +105 -0
  93. package/src/ai_chemistry_scientist/salt_standardization.py +81 -0
  94. package/src/ai_chemistry_scientist/structural_alerts.py +76 -0
  95. package/src/ai_chemistry_scientist/structure_format_conversion.py +84 -0
  96. package/src/ai_chemistry_scientist/validation.py +70 -0
  97. package/src/ai_data_scientist/__init__.py +0 -0
  98. package/src/ai_data_scientist/analysis_assumptions.py +121 -0
  99. package/src/ai_data_scientist/anomaly_detection.py +39 -0
  100. package/src/ai_data_scientist/automl.py +109 -0
  101. package/src/ai_data_scientist/cleaning.py +56 -0
  102. package/src/ai_data_scientist/cli.py +90 -0
  103. package/src/ai_data_scientist/clustering.py +54 -0
  104. package/src/ai_data_scientist/dashboard.py +33 -0
  105. package/src/ai_data_scientist/data_definition.py +100 -0
  106. package/src/ai_data_scientist/data_quality.py +164 -0
  107. package/src/ai_data_scientist/dataset_validation.py +135 -0
  108. package/src/ai_data_scientist/dependency_pins.py +60 -0
  109. package/src/ai_data_scientist/eda.py +82 -0
  110. package/src/ai_data_scientist/experiment_evaluation.py +635 -0
  111. package/src/ai_data_scientist/explainability.py +340 -0
  112. package/src/ai_data_scientist/feature_engineering.py +163 -0
  113. package/src/ai_data_scientist/gate_config.py +32 -0
  114. package/src/ai_data_scientist/ingestion.py +127 -0
  115. package/src/ai_data_scientist/insight_engine.py +180 -0
  116. package/src/ai_data_scientist/japanese_nlp.py +43 -0
  117. package/src/ai_data_scientist/jupyter_launcher.py +137 -0
  118. package/src/ai_data_scientist/jupyter_mcp_client.py +94 -0
  119. package/src/ai_data_scientist/language_router.py +28 -0
  120. package/src/ai_data_scientist/lifecycle.py +221 -0
  121. package/src/ai_data_scientist/mcp_gateway.py +113 -0
  122. package/src/ai_data_scientist/mcp_runtime.py +194 -0
  123. package/src/ai_data_scientist/mcp_transport.py +53 -0
  124. package/src/ai_data_scientist/ml_modeling.py +451 -0
  125. package/src/ai_data_scientist/model_tuning.py +104 -0
  126. package/src/ai_data_scientist/notebook_audit.py +574 -0
  127. package/src/ai_data_scientist/project_manager.py +243 -0
  128. package/src/ai_data_scientist/report_export.py +73 -0
  129. package/src/ai_data_scientist/sensitivity.py +445 -0
  130. package/src/ai_data_scientist/signal_analysis.py +201 -0
  131. package/src/ai_data_scientist/skill_packaging.py +40 -0
  132. package/src/ai_data_scientist/stats_analysis.py +88 -0
  133. package/src/ai_data_scientist/text_nlp.py +44 -0
  134. package/src/ai_data_scientist/timeseries.py +68 -0
  135. package/src/ai_data_scientist/visualization.py +708 -0
  136. package/src/ai_genomics_scientist/__init__.py +1 -0
  137. package/src/ai_genomics_scientist/differential_expression.py +147 -0
  138. package/src/ai_genomics_scientist/dispatch.py +267 -0
  139. package/src/ai_genomics_scientist/evidence.py +45 -0
  140. package/src/ai_genomics_scientist/gene_set_enrichment.py +76 -0
  141. package/src/ai_genomics_scientist/sequence_alignment.py +97 -0
  142. package/src/ai_genomics_scientist/sequence_features.py +111 -0
  143. package/src/ai_genomics_scientist/splice_site_scoring.py +66 -0
  144. package/src/ai_genomics_scientist/validation.py +83 -0
  145. package/src/ai_genomics_scientist/variant_effect.py +147 -0
  146. package/src/ai_genomics_scientist/variant_pathogenicity.py +125 -0
  147. package/src/ai_materials_scientist/__init__.py +0 -0
  148. package/src/ai_materials_scientist/calphad.py +117 -0
  149. package/src/ai_materials_scientist/classical_monte_carlo.py +165 -0
  150. package/src/ai_materials_scientist/crystal_plasticity.py +184 -0
  151. package/src/ai_materials_scientist/dispatch.py +100 -0
  152. package/src/ai_materials_scientist/evidence.py +84 -0
  153. package/src/ai_materials_scientist/fem.py +279 -0
  154. package/src/ai_materials_scientist/kinetic_monte_carlo.py +145 -0
  155. package/src/ai_materials_scientist/molecular_dynamics.py +240 -0
  156. package/src/ai_materials_scientist/phase_field.py +167 -0
  157. package/src/ai_materials_scientist/validation.py +70 -0
  158. package/src/ai_scientist/__init__.py +1 -0
  159. package/src/ai_scientist/completion_gate.py +15 -0
  160. package/src/ai_scientist/data_analysis.py +46 -0
  161. package/src/ai_scientist/evidence_registry.py +99 -0
  162. package/src/ai_scientist/experimental_design.py +20 -0
  163. package/src/ai_scientist/language.py +14 -0
  164. package/src/ai_scientist/latex_renderer.py +41 -0
  165. package/src/ai_scientist/literature_review.py +37 -0
  166. package/src/ai_scientist/manifest.py +87 -0
  167. package/src/ai_scientist/manuscript.py +94 -0
  168. package/src/ai_scientist/mcp_config.py +76 -0
  169. package/src/ai_scientist/mcp_external.py +42 -0
  170. package/src/ai_scientist/mcp_failures.py +23 -0
  171. package/src/ai_scientist/mcp_gateway.py +38 -0
  172. package/src/ai_scientist/mcp_managed.py +180 -0
  173. package/src/ai_scientist/npm_packaging.py +49 -0
  174. package/src/ai_scientist/orchestrator.py +133 -0
  175. package/src/ai_scientist/peer_review.py +60 -0
  176. package/src/ai_scientist/phase_gate.py +74 -0
  177. package/src/ai_scientist/phase_state.py +230 -0
  178. package/src/ai_scientist/presentation.py +56 -0
  179. package/src/ai_scientist/project_config.py +31 -0
  180. package/src/ai_scientist/project_handle.py +74 -0
  181. package/src/ai_scientist/reproducibility.py +20 -0
  182. package/src/ai_scientist/research_planning.py +20 -0
  183. package/src/ai_scientist/skill_invocation.py +21 -0
  184. package/src/ai_scientist/tdd_gate.py +99 -0
  185. package/src/ai_structural_biology_scientist/__init__.py +0 -0
  186. package/src/ai_structural_biology_scientist/contact_map.py +87 -0
  187. package/src/ai_structural_biology_scientist/dispatch.py +269 -0
  188. package/src/ai_structural_biology_scientist/evidence.py +43 -0
  189. package/src/ai_structural_biology_scientist/hydrophobicity.py +101 -0
  190. package/src/ai_structural_biology_scientist/protein_docking_score.py +104 -0
  191. package/src/ai_structural_biology_scientist/secondary_structure.py +95 -0
  192. package/src/ai_structural_biology_scientist/structural_similarity.py +74 -0
  193. package/src/ai_structural_biology_scientist/validation.py +100 -0
@@ -0,0 +1,111 @@
1
+ """Sequence feature analysis module (DES-AGENOM-010 / REQ-AGENOM-010)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from ai_genomics_scientist.validation import (
6
+ fail,
7
+ ok,
8
+ register_batch_item_validator,
9
+ validate_batch_item,
10
+ )
11
+
12
+ _MODULE_NAME = "sequence-features"
13
+ _DNA_BASES = frozenset({"A", "C", "G", "T"})
14
+ _STOP_CODONS = frozenset({"TAA", "TAG", "TGA"})
15
+ _SEQUENCE_CONSTRAINT = "must be a non-empty uppercase DNA string over {A,C,G,T} with length >= 3"
16
+
17
+
18
+ def _is_valid_dna_sequence(sequence: str, *, min_length: int) -> bool:
19
+ return (
20
+ isinstance(sequence, str)
21
+ and len(sequence) >= min_length
22
+ and sequence.isupper()
23
+ and set(sequence).issubset(_DNA_BASES)
24
+ )
25
+
26
+
27
+ def _sequence_features_batch_item_validator(item_params: dict) -> dict:
28
+ """DES-AGENOM-002 registered per-item validator for this module."""
29
+ sequence = item_params.get("sequence")
30
+ if not _is_valid_dna_sequence(sequence, min_length=3):
31
+ return fail("sequence", _SEQUENCE_CONSTRAINT)
32
+ return ok()
33
+
34
+
35
+ register_batch_item_validator(_MODULE_NAME, _sequence_features_batch_item_validator)
36
+
37
+
38
+ def _find_longest_orf(sequence: str) -> dict | None:
39
+ best_orf: dict | None = None
40
+ best_length = -1
41
+
42
+ for frame in range(3):
43
+ for start_index in range(frame, len(sequence) - 2, 3):
44
+ if sequence[start_index : start_index + 3] != "ATG":
45
+ continue
46
+ for stop_index in range(start_index + 3, len(sequence) - 2, 3):
47
+ codon = sequence[stop_index : stop_index + 3]
48
+ if codon not in _STOP_CODONS:
49
+ continue
50
+ orf_length = stop_index + 3 - start_index
51
+ candidate = {
52
+ "frame": frame,
53
+ "start_index": start_index,
54
+ "length": orf_length,
55
+ }
56
+ if orf_length > best_length or (
57
+ orf_length == best_length
58
+ and best_orf is not None
59
+ and (frame, start_index) < (best_orf["frame"], best_orf["start_index"])
60
+ ):
61
+ best_orf = candidate
62
+ best_length = orf_length
63
+ break
64
+
65
+ return best_orf
66
+
67
+
68
+ def _count_codons(sequence: str, longest_orf: dict | None) -> dict[str, int]:
69
+ if longest_orf is None:
70
+ return {}
71
+
72
+ start = longest_orf["start_index"]
73
+ stop = start + longest_orf["length"]
74
+ counts: dict[str, int] = {}
75
+ for index in range(start, stop, 3):
76
+ codon = sequence[index : index + 3]
77
+ counts[codon] = counts.get(codon, 0) + 1
78
+ return counts
79
+
80
+
81
+ # @id CODE-AGENOM-010
82
+ # @implements REQ-AGENOM-010
83
+ # @design DES-AGENOM-010
84
+ def run_sequence_features(sequences: list[str]) -> list[dict]:
85
+ """Compute fixed sequence features for each valid DNA string in ``sequences``."""
86
+ results: list[dict] = []
87
+
88
+ for sequence in sequences:
89
+ validation = validate_batch_item(_MODULE_NAME, {"sequence": sequence})
90
+ if not validation["ok"]:
91
+ results.append(
92
+ {
93
+ "sequence": sequence,
94
+ "ok": False,
95
+ "parameter": validation["parameter"],
96
+ "constraint": validation["constraint"],
97
+ }
98
+ )
99
+ continue
100
+
101
+ longest_orf = _find_longest_orf(sequence)
102
+ results.append(
103
+ {
104
+ "length": len(sequence),
105
+ "gc_content": (sequence.count("G") + sequence.count("C")) / len(sequence),
106
+ "longest_orf": longest_orf,
107
+ "codon_usage": _count_codons(sequence, longest_orf),
108
+ }
109
+ )
110
+
111
+ return results
@@ -0,0 +1,66 @@
1
+ """Splice-site strength heuristic module (DES-AGENOM-030 / REQ-AGENOM-030)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import math
6
+
7
+ from ai_genomics_scientist.validation import fail, ok, register_validator
8
+
9
+ _MODULE_NAME = "splice-site-strength"
10
+ _DNA_BASES = frozenset({"A", "C", "G", "T"})
11
+
12
+ LIMITATION_LABEL_TEXT = {
13
+ "en": (
14
+ "Heuristic only: a fixed illustrative position-weight scoring scheme, not a "
15
+ "validated splice-site predictor (not SpliceAI, not based on real "
16
+ "splice-site frequency data)."
17
+ ),
18
+ "ja": (
19
+ "ヒューリスティックのみ:これは固定された例示用の"
20
+ " position-weight スコアリング方式であり、検証済みのスプライス"
21
+ "部位予測器ではない(SpliceAI ではなく、実際のスプライス部位"
22
+ "頻度データにも基づかない)。"
23
+ ),
24
+ }
25
+
26
+ _PFM = (
27
+ {"A": 0.30, "C": 0.20, "G": 0.25, "T": 0.25},
28
+ {"A": 0.60, "C": 0.15, "G": 0.15, "T": 0.10},
29
+ {"A": 0.15, "C": 0.15, "G": 0.60, "T": 0.10},
30
+ {"A": 0.00, "C": 0.00, "G": 1.00, "T": 0.00},
31
+ {"A": 0.00, "C": 0.00, "G": 0.00, "T": 1.00},
32
+ {"A": 0.55, "C": 0.05, "G": 0.35, "T": 0.05},
33
+ {"A": 0.70, "C": 0.10, "G": 0.10, "T": 0.10},
34
+ {"A": 0.08, "C": 0.05, "G": 0.80, "T": 0.07},
35
+ {"A": 0.15, "C": 0.20, "G": 0.20, "T": 0.45},
36
+ )
37
+
38
+
39
+ def _is_dna_string(value: object) -> bool:
40
+ return isinstance(value, str) and value.isupper() and value and set(value).issubset(_DNA_BASES)
41
+
42
+
43
+ def _splice_site_validator(params: dict) -> dict:
44
+ """DES-AGENOM-002 registered atomic validator for this module."""
45
+ window = params.get("window")
46
+ if not _is_dna_string(window):
47
+ return fail("window", "must be a non-empty uppercase DNA string over {A,C,G,T}")
48
+ if len(window) != 9:
49
+ return fail("window", "must be exactly 9 characters")
50
+ if window[3:5] != "GT":
51
+ return fail("window", "position 0,+1 must be the canonical GT dinucleotide")
52
+ return ok()
53
+
54
+
55
+ register_validator(_MODULE_NAME, _splice_site_validator)
56
+
57
+
58
+ # @id CODE-AGENOM-030
59
+ # @implements REQ-AGENOM-030
60
+ # @design DES-AGENOM-030
61
+ def run_splice_site_scoring(window: str) -> dict:
62
+ """Score a validated 9-base donor-site window with the fixed illustrative PFM."""
63
+ score_bits = 0.0
64
+ for index, base in enumerate(window):
65
+ score_bits += math.log2(_PFM[index][base] / 0.25)
66
+ return {"window": window, "score_bits": score_bits, "canonical_site": True}
@@ -0,0 +1,83 @@
1
+ """Shared parameter & validity validator (DES-AGENOM-002 / REQ-AGENOM-003)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import importlib
6
+ from collections.abc import Callable, Mapping
7
+ from typing import Any
8
+
9
+ ValidationResult = dict[str, Any]
10
+ ValidatorFn = Callable[[dict[str, Any]], ValidationResult]
11
+ BatchItemValidatorFn = Callable[[dict[str, Any]], ValidationResult]
12
+
13
+ _REGISTRY: dict[str, ValidatorFn] = {}
14
+ _BATCH_ITEM_REGISTRY: dict[str, BatchItemValidatorFn] = {}
15
+ _VALIDATOR_MODULES = {
16
+ "sequence-features": "ai_genomics_scientist.sequence_features",
17
+ "variant-effect-annotation": "ai_genomics_scientist.variant_effect",
18
+ "splice-site-strength": "ai_genomics_scientist.splice_site_scoring",
19
+ "gene-set-enrichment": "ai_genomics_scientist.gene_set_enrichment",
20
+ "pairwise-sequence-alignment": "ai_genomics_scientist.sequence_alignment",
21
+ "differential-expression": "ai_genomics_scientist.differential_expression",
22
+ "variant-pathogenicity": "ai_genomics_scientist.variant_pathogenicity",
23
+ }
24
+
25
+
26
+ def ok() -> ValidationResult:
27
+ return {"ok": True}
28
+
29
+
30
+ def fail(parameter: str, constraint: str) -> ValidationResult:
31
+ return {"ok": False, "parameter": parameter, "constraint": constraint}
32
+
33
+
34
+ def _ensure_validator_registered(module_name: str) -> None:
35
+ if module_name in _REGISTRY and module_name in _BATCH_ITEM_REGISTRY:
36
+ return
37
+ module_path = _VALIDATOR_MODULES.get(module_name)
38
+ if module_path is not None:
39
+ importlib.import_module(module_path)
40
+
41
+
42
+ # @id CODE-AGENOM-002
43
+ # @implements REQ-AGENOM-003
44
+ # @design DES-AGENOM-002
45
+ def register_validator(module_name: str, validator: ValidatorFn) -> None:
46
+ """Register ``module_name``'s atomic validator."""
47
+ _REGISTRY[module_name] = validator
48
+
49
+
50
+ # @id CODE-AGENOM-003
51
+ # @implements REQ-AGENOM-003
52
+ # @design DES-AGENOM-002
53
+ def validate_parameters(module_name: str, params: dict[str, Any]) -> ValidationResult:
54
+ """Dispatch to ``module_name``'s registered atomic validator."""
55
+ _ensure_validator_registered(module_name)
56
+ validator = _REGISTRY.get(module_name)
57
+ if validator is None:
58
+ return fail("module", f"no validator registered for module '{module_name}'")
59
+ if not isinstance(params, Mapping):
60
+ return fail("params", "must be a dict")
61
+ return validator(dict(params))
62
+
63
+
64
+ # @id CODE-AGENOM-004
65
+ # @implements REQ-AGENOM-003
66
+ # @design DES-AGENOM-002
67
+ def register_batch_item_validator(module_name: str, validator: BatchItemValidatorFn) -> None:
68
+ """Register ``module_name``'s per-item validator."""
69
+ _BATCH_ITEM_REGISTRY[module_name] = validator
70
+
71
+
72
+ # @id CODE-AGENOM-005
73
+ # @implements REQ-AGENOM-003
74
+ # @design DES-AGENOM-002
75
+ def validate_batch_item(module_name: str, item_params: dict[str, Any]) -> ValidationResult:
76
+ """Dispatch to ``module_name``'s registered per-item validator."""
77
+ _ensure_validator_registered(module_name)
78
+ validator = _BATCH_ITEM_REGISTRY.get(module_name)
79
+ if validator is None:
80
+ return fail("module", f"no batch-item validator registered for module '{module_name}'")
81
+ if not isinstance(item_params, Mapping):
82
+ return fail("params", "must be a dict")
83
+ return validator(dict(item_params))
@@ -0,0 +1,147 @@
1
+ """Variant effect heuristic annotation module (DES-AGENOM-020 / REQ-AGENOM-020)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from numbers import Integral
6
+
7
+ from ai_genomics_scientist.validation import fail, ok, register_validator
8
+
9
+ _MODULE_NAME = "variant-effect-annotation"
10
+ _DNA_BASES = frozenset({"A", "C", "G", "T"})
11
+ _STOP_REFERENCE_CONSTRAINT = "stop-reference codons must mutate to a non-stop codon"
12
+
13
+ _STANDARD_GENETIC_CODE = {
14
+ "TTT": "F",
15
+ "TTC": "F",
16
+ "TTA": "L",
17
+ "TTG": "L",
18
+ "TCT": "S",
19
+ "TCC": "S",
20
+ "TCA": "S",
21
+ "TCG": "S",
22
+ "TAT": "Y",
23
+ "TAC": "Y",
24
+ "TAA": "*",
25
+ "TAG": "*",
26
+ "TGT": "C",
27
+ "TGC": "C",
28
+ "TGA": "*",
29
+ "TGG": "W",
30
+ "CTT": "L",
31
+ "CTC": "L",
32
+ "CTA": "L",
33
+ "CTG": "L",
34
+ "CCT": "P",
35
+ "CCC": "P",
36
+ "CCA": "P",
37
+ "CCG": "P",
38
+ "CAT": "H",
39
+ "CAC": "H",
40
+ "CAA": "Q",
41
+ "CAG": "Q",
42
+ "CGT": "R",
43
+ "CGC": "R",
44
+ "CGA": "R",
45
+ "CGG": "R",
46
+ "ATT": "I",
47
+ "ATC": "I",
48
+ "ATA": "I",
49
+ "ATG": "M",
50
+ "ACT": "T",
51
+ "ACC": "T",
52
+ "ACA": "T",
53
+ "ACG": "T",
54
+ "AAT": "N",
55
+ "AAC": "N",
56
+ "AAA": "K",
57
+ "AAG": "K",
58
+ "AGT": "S",
59
+ "AGC": "S",
60
+ "AGA": "R",
61
+ "AGG": "R",
62
+ "GTT": "V",
63
+ "GTC": "V",
64
+ "GTA": "V",
65
+ "GTG": "V",
66
+ "GCT": "A",
67
+ "GCC": "A",
68
+ "GCA": "A",
69
+ "GCG": "A",
70
+ "GAT": "D",
71
+ "GAC": "D",
72
+ "GAA": "E",
73
+ "GAG": "E",
74
+ "GGT": "G",
75
+ "GGC": "G",
76
+ "GGA": "G",
77
+ "GGG": "G",
78
+ }
79
+
80
+
81
+ def _is_dna_string(value: object, *, length: int) -> bool:
82
+ return (
83
+ isinstance(value, str)
84
+ and len(value) == length
85
+ and value.isupper()
86
+ and set(value).issubset(_DNA_BASES)
87
+ )
88
+
89
+
90
+ def _translate_codon(codon: str) -> str:
91
+ return _STANDARD_GENETIC_CODE[codon]
92
+
93
+
94
+ def _variant_effect_validator(params: dict) -> dict:
95
+ """DES-AGENOM-002 registered atomic validator for this module."""
96
+ ref_codon = params.get("ref_codon")
97
+ position = params.get("position")
98
+ alt_base = params.get("alt_base")
99
+
100
+ if not _is_dna_string(ref_codon, length=3):
101
+ return fail("ref_codon", "must be exactly 3 uppercase DNA bases over {A,C,G,T}")
102
+ if (
103
+ not isinstance(position, Integral)
104
+ or isinstance(position, bool)
105
+ or position not in {0, 1, 2}
106
+ ):
107
+ return fail("position", "must be an integer in {0,1,2}")
108
+ if not _is_dna_string(alt_base, length=1):
109
+ return fail("alt_base", "must be exactly 1 uppercase DNA base over {A,C,G,T}")
110
+ if alt_base == ref_codon[position]:
111
+ return fail("alt_base", "alt_base must differ from the reference base at position")
112
+
113
+ alt_codon = ref_codon[:position] + alt_base + ref_codon[position + 1 :]
114
+ if _translate_codon(ref_codon) == "*" and _translate_codon(alt_codon) == "*":
115
+ return fail("ref_codon", _STOP_REFERENCE_CONSTRAINT)
116
+
117
+ return ok()
118
+
119
+
120
+ register_validator(_MODULE_NAME, _variant_effect_validator)
121
+
122
+
123
+ def _classify_effect(ref_amino_acid: str, alt_amino_acid: str) -> str:
124
+ if ref_amino_acid == "*" and alt_amino_acid != "*":
125
+ return "readthrough"
126
+ if ref_amino_acid != "*" and alt_amino_acid == "*":
127
+ return "nonsense"
128
+ if ref_amino_acid == alt_amino_acid:
129
+ return "synonymous"
130
+ return "missense"
131
+
132
+
133
+ # @id CODE-AGENOM-020
134
+ # @implements REQ-AGENOM-020
135
+ # @design DES-AGENOM-020
136
+ def run_variant_effect(ref_codon: str, position: int, alt_base: str) -> dict:
137
+ """Annotate a single-codon substitution under the standard genetic code."""
138
+ alt_codon = ref_codon[:position] + alt_base + ref_codon[position + 1 :]
139
+ ref_amino_acid = _translate_codon(ref_codon)
140
+ alt_amino_acid = _translate_codon(alt_codon)
141
+ return {
142
+ "ref_codon": ref_codon,
143
+ "alt_codon": alt_codon,
144
+ "ref_amino_acid": ref_amino_acid,
145
+ "alt_amino_acid": alt_amino_acid,
146
+ "effect": _classify_effect(ref_amino_acid, alt_amino_acid),
147
+ }
@@ -0,0 +1,125 @@
1
+ """Variant pathogenicity heuristic module (DES-AGENOM-070 / REQ-AGENOM-070)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from ai_genomics_scientist.validation import fail, ok, register_validator
6
+
7
+ _MODULE_NAME = "variant-pathogenicity"
8
+
9
+ _AMINO_ACID_ORDER = [
10
+ "A",
11
+ "R",
12
+ "N",
13
+ "D",
14
+ "C",
15
+ "Q",
16
+ "E",
17
+ "G",
18
+ "H",
19
+ "I",
20
+ "L",
21
+ "K",
22
+ "M",
23
+ "F",
24
+ "P",
25
+ "S",
26
+ "T",
27
+ "W",
28
+ "Y",
29
+ "V",
30
+ ]
31
+ _STANDARD_AMINO_ACIDS = frozenset(_AMINO_ACID_ORDER)
32
+
33
+ # Standard published BLOSUM62 substitution matrix (symmetric, 20x20),
34
+ # row/column order matching _AMINO_ACID_ORDER exactly, per ADR-0108.
35
+ _BLOSUM62_ROWS = [
36
+ [4, -1, -2, -2, 0, -1, -1, 0, -2, -1, -1, -1, -1, -2, -1, 1, 0, -3, -2, 0],
37
+ [-1, 5, 0, -2, -3, 1, 0, -2, 0, -3, -2, 2, -1, -3, -2, -1, -1, -3, -2, -3],
38
+ [-2, 0, 6, 1, -3, 0, 0, 0, 1, -3, -3, 0, -2, -3, -2, 1, 0, -4, -2, -3],
39
+ [-2, -2, 1, 6, -3, 0, 2, -1, -1, -3, -4, -1, -3, -3, -1, 0, -1, -4, -3, -3],
40
+ [0, -3, -3, -3, 9, -3, -4, -3, -3, -1, -1, -3, -1, -2, -3, -1, -1, -2, -2, -1],
41
+ [-1, 1, 0, 0, -3, 5, 2, -2, 0, -3, -2, 1, 0, -3, -1, 0, -1, -2, -1, -2],
42
+ [-1, 0, 0, 2, -4, 2, 5, -2, 0, -3, -3, 1, -2, -3, -1, 0, -1, -3, -2, -2],
43
+ [0, -2, 0, -1, -3, -2, -2, 6, -2, -4, -4, -2, -3, -3, -2, 0, -2, -2, -3, -3],
44
+ [-2, 0, 1, -1, -3, 0, 0, -2, 8, -3, -3, -1, -2, -1, -2, -1, -2, -2, 2, -3],
45
+ [-1, -3, -3, -3, -1, -3, -3, -4, -3, 4, 2, -3, 1, 0, -3, -2, -1, -3, -1, 3],
46
+ [-1, -2, -3, -4, -1, -2, -3, -4, -3, 2, 4, -2, 2, 0, -3, -2, -1, -2, -1, 1],
47
+ [-1, 2, 0, -1, -3, 1, 1, -2, -1, -3, -2, 5, -1, -3, -1, 0, -1, -3, -2, -2],
48
+ [-1, -1, -2, -3, -1, 0, -2, -3, -2, 1, 2, -1, 5, 0, -2, -1, -1, -1, -1, 1],
49
+ [-2, -3, -3, -3, -2, -3, -3, -3, -1, 0, 0, -3, 0, 6, -4, -2, -2, 1, 3, -1],
50
+ [-1, -2, -2, -1, -3, -1, -1, -2, -2, -3, -3, -1, -2, -4, 7, -1, -1, -4, -3, -2],
51
+ [1, -1, 1, 0, -1, 0, 0, 0, -1, -2, -2, 0, -1, -2, -1, 4, 1, -3, -2, -2],
52
+ [0, -1, 0, -1, -1, -1, -1, -2, -2, -1, -1, -1, -1, -2, -1, 1, 5, -2, -2, 0],
53
+ [-3, -3, -4, -4, -2, -2, -3, -2, -2, -3, -2, -3, -1, 1, -4, -3, -2, 11, 2, -3],
54
+ [-2, -2, -2, -3, -2, -1, -2, -3, 2, -1, -1, -2, -1, 3, -3, -2, -2, 2, 7, -1],
55
+ [0, -3, -3, -3, -1, -2, -2, -3, -3, 3, 1, -2, 1, -1, -2, -2, 0, -3, -1, 4],
56
+ ]
57
+ _BLOSUM62 = {
58
+ (row_aa, col_aa): _BLOSUM62_ROWS[row_index][col_index]
59
+ for row_index, row_aa in enumerate(_AMINO_ACID_ORDER)
60
+ for col_index, col_aa in enumerate(_AMINO_ACID_ORDER)
61
+ }
62
+
63
+ _TIER_THRESHOLDS = [
64
+ (0.3, "benign"),
65
+ (0.5, "likely_benign"),
66
+ (0.7, "uncertain_significance"),
67
+ (0.85, "likely_pathogenic"),
68
+ ]
69
+
70
+
71
+ def _variant_pathogenicity_validator(params: dict) -> dict:
72
+ """DES-AGENOM-002 registered atomic validator for this module."""
73
+ ref_aa = params.get("ref_aa")
74
+ alt_aa = params.get("alt_aa")
75
+ conservation_score = params.get("conservation_score")
76
+ in_functional_domain = params.get("in_functional_domain")
77
+
78
+ if not isinstance(ref_aa, str) or ref_aa not in _STANDARD_AMINO_ACIDS:
79
+ return fail("ref_aa", "must be one of the 20 standard single-letter amino acid codes")
80
+ if not isinstance(alt_aa, str) or alt_aa not in _STANDARD_AMINO_ACIDS:
81
+ return fail("alt_aa", "must be one of the 20 standard single-letter amino acid codes")
82
+ if alt_aa == ref_aa:
83
+ return fail("alt_aa", "alt_aa must differ from ref_aa")
84
+ if (
85
+ not isinstance(conservation_score, float)
86
+ or isinstance(conservation_score, bool)
87
+ or not (0 <= conservation_score <= 1)
88
+ ):
89
+ return fail("conservation_score", "must be a float in the closed interval [0, 1]")
90
+ if not isinstance(in_functional_domain, bool):
91
+ return fail("in_functional_domain", "must be a boolean")
92
+
93
+ return ok()
94
+
95
+
96
+ register_validator(_MODULE_NAME, _variant_pathogenicity_validator)
97
+
98
+
99
+ def _classify(score: float) -> str:
100
+ for threshold, tier in _TIER_THRESHOLDS:
101
+ if score < threshold:
102
+ return tier
103
+ return "pathogenic"
104
+
105
+
106
+ # @id CODE-AGENOM-070
107
+ # @implements REQ-AGENOM-070
108
+ # @design DES-AGENOM-070
109
+ def run_variant_pathogenicity(
110
+ ref_aa: str, alt_aa: str, conservation_score: float, in_functional_domain: bool
111
+ ) -> dict:
112
+ """Compute the fixed BLOSUM62 + weighted-sum pathogenicity heuristic score."""
113
+ blosum_score = _BLOSUM62[(ref_aa, alt_aa)]
114
+ dissimilarity = min(max((3 - blosum_score) / 7, 0.0), 1.0)
115
+ domain_bonus = 0.15 if in_functional_domain else 0.0
116
+ pathogenicity_score = min(
117
+ max(0.5 * dissimilarity + 0.35 * conservation_score + domain_bonus, 0.0), 1.0
118
+ )
119
+ return {
120
+ "ref_aa": ref_aa,
121
+ "alt_aa": alt_aa,
122
+ "blosum_score": blosum_score,
123
+ "pathogenicity_score": pathogenicity_score,
124
+ "classification": _classify(pathogenicity_score),
125
+ }
File without changes
@@ -0,0 +1,117 @@
1
+ """Simplified binary CALPHAD phase diagram module (DES-AIMS-070 / REQ-AIMS-070)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import numpy as np
6
+ from scipy.optimize import brentq
7
+
8
+ from ai_materials_scientist.evidence import record_run
9
+ from ai_materials_scientist.validation import register_validator, validate_parameters
10
+
11
+ _MODULE_NAME = "calphad"
12
+ _R = 8.314
13
+ _ROOT_BRACKET_EPSILON = 1e-9
14
+
15
+
16
+ def _free_energy_of_mixing(x: np.ndarray, omega: float, temperature: float) -> np.ndarray:
17
+ return _R * temperature * (x * np.log(x) + (1 - x) * np.log(1 - x)) + omega * x * (1 - x)
18
+
19
+
20
+ def _coexistence_residual(x: float, omega: float, temperature: float) -> float:
21
+ return np.log(x / (1 - x)) + (omega / (_R * temperature)) * (1 - 2 * x)
22
+
23
+
24
+ def _calphad_validator(params: dict) -> dict:
25
+ """DES-AIMS-002 registered validator for module_name='calphad'."""
26
+ omega = params["omega"]
27
+ temperature = params["temperature"]
28
+ composition_grid = params["composition_grid"]
29
+
30
+ if not np.isfinite(omega) or omega <= 0:
31
+ return {"ok": False, "parameter": "omega", "constraint": "omega > 0"}
32
+ if not np.isfinite(temperature) or temperature <= 0:
33
+ return {"ok": False, "parameter": "temperature", "constraint": "temperature > 0"}
34
+
35
+ t_c = omega / (2 * _R)
36
+ if temperature >= t_c:
37
+ return {
38
+ "ok": False,
39
+ "parameter": "temperature",
40
+ "constraint": "temperature < consolute_temperature (omega / (2 * R))",
41
+ }
42
+
43
+ composition_grid = np.asarray(composition_grid, dtype=np.float64)
44
+ if not np.all(np.isfinite(composition_grid)):
45
+ return {
46
+ "ok": False,
47
+ "parameter": "composition_grid",
48
+ "constraint": "composition_grid entries must be finite",
49
+ }
50
+ if np.any(composition_grid <= 0) or np.any(composition_grid >= 1):
51
+ return {
52
+ "ok": False,
53
+ "parameter": "composition_grid",
54
+ "constraint": "composition_grid entries must lie in the open interval (0, 1)",
55
+ }
56
+ return {"ok": True}
57
+
58
+
59
+ register_validator(_MODULE_NAME, _calphad_validator)
60
+
61
+
62
+ def _validate(omega: float, temperature: float, composition_grid: np.ndarray) -> None:
63
+ result = validate_parameters(
64
+ _MODULE_NAME,
65
+ {"omega": omega, "temperature": temperature, "composition_grid": composition_grid},
66
+ )
67
+ if not result["ok"]:
68
+ raise ValueError(f"{result['parameter']}: {result['constraint']}")
69
+
70
+
71
+ # @id CODE-AIMS-070
72
+ # @implements REQ-AIMS-070 REQ-AIMS-003
73
+ # @design DES-AIMS-070
74
+ def run_calphad(omega: float, temperature: float, composition_grid: np.ndarray) -> dict:
75
+ """Compute the regular-solution G_mix curve and the two binodal compositions."""
76
+ composition_grid = np.asarray(composition_grid, dtype=np.float64)
77
+ _validate(omega, temperature, composition_grid)
78
+
79
+ free_energy_curve = _free_energy_of_mixing(composition_grid, omega, temperature)
80
+
81
+ x_alpha = brentq(
82
+ _coexistence_residual,
83
+ _ROOT_BRACKET_EPSILON,
84
+ 0.5 - _ROOT_BRACKET_EPSILON,
85
+ args=(omega, temperature),
86
+ )
87
+ x_beta = brentq(
88
+ _coexistence_residual,
89
+ 0.5 + _ROOT_BRACKET_EPSILON,
90
+ 1.0 - _ROOT_BRACKET_EPSILON,
91
+ args=(omega, temperature),
92
+ )
93
+
94
+ return {
95
+ "x_alpha": x_alpha,
96
+ "x_beta": x_beta,
97
+ "free_energy_curve": free_energy_curve,
98
+ }
99
+
100
+
101
+ # @id CODE-AIMS-900
102
+ # @implements REQ-AIMS-070 REQ-AIMS-004 REQ-AIMS-005
103
+ # @design DES-AIMS-070
104
+ def run_calphad_with_evidence(**kwargs) -> dict:
105
+ """Run the CALPHAD module and wrap the result as a reproducible RunRecord."""
106
+ result = run_calphad(**kwargs)
107
+ return record_run(
108
+ module_name=_MODULE_NAME,
109
+ unit_system="J-per-mol",
110
+ params={k: v for k, v in kwargs.items() if k != "composition_grid"},
111
+ arrays={
112
+ "x_alpha": np.array([result["x_alpha"]], dtype=np.float64),
113
+ "x_beta": np.array([result["x_beta"]], dtype=np.float64),
114
+ "free_energy_curve": np.asarray(result["free_energy_curve"], dtype=np.float64),
115
+ },
116
+ seed=None,
117
+ )