jupytermind 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/skills/ai-chemistry-scientist/SKILL.md +97 -0
- package/.github/skills/ai-chemistry-scientist/manifest.json +156 -0
- package/.github/skills/ai-data-scientist/SKILL.md +330 -0
- package/.github/skills/ai-genomics-scientist/SKILL.md +98 -0
- package/.github/skills/ai-genomics-scientist/manifest.json +93 -0
- package/.github/skills/ai-materials-scientist/SKILL.md +51 -0
- package/.github/skills/ai-materials-scientist/manifest.json +58 -0
- package/.github/skills/ai-scientist/SKILL.md +69 -0
- package/.github/skills/ai-scientist/manifest.json +61 -0
- package/.github/skills/ai-structural-biology-scientist/SKILL.md +67 -0
- package/.github/skills/ai-structural-biology-scientist/manifest.json +72 -0
- package/.github/skills/japanese-prose/NOTICE.md +17 -0
- package/.github/skills/japanese-prose/SKILL.md +111 -0
- package/.github/skills/japanese-prose/references/review-workflow.md +50 -0
- package/.github/skills/japanese-prose/references/scoring.md +24 -0
- package/.github/skills/japanese-prose/references/writing-guidelines.md +60 -0
- package/.github/skills/japanese-prose/scripts/core.py +192 -0
- package/.github/skills/japanese-prose/scripts/fixtures/natural.md +5 -0
- package/.github/skills/japanese-prose/scripts/fixtures/unnatural.md +5 -0
- package/.github/skills/japanese-prose/scripts/lint.py +378 -0
- package/.github/skills/japanese-prose/scripts/outline.py +68 -0
- package/.github/skills/japanese-prose/scripts/terms.py +112 -0
- package/.github/skills/japanese-prose/scripts/test_engine.py +117 -0
- package/.github/skills/presentation-planner/SKILL.md +257 -0
- package/.github/skills/presentation-planner/assets/design-templates/data-report.yaml +97 -0
- package/.github/skills/presentation-planner/assets/design-templates/executive-proposal.yaml +92 -0
- package/.github/skills/presentation-planner/assets/design-templates/technical-briefing.yaml +96 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/data-report.md +47 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/executive-decision.md +43 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/technical-briefing.md +45 -0
- package/.github/skills/presentation-planner/references/customizing-design-templates.md +160 -0
- package/.github/skills/presentation-planner/references/design-spec-schema.md +72 -0
- package/.github/skills/presentation-planner/references/handoff-contract.md +49 -0
- package/.github/skills/presentation-planner/references/responsibility-boundary.md +32 -0
- package/.github/skills/presentation-planner/references/scenario-templates.md +55 -0
- package/.github/skills/tech-writer/SKILL.md +434 -0
- package/.github/skills/tech-writer/assets/templates/blueprint.md +187 -0
- package/.github/skills/tech-writer/assets/templates/design-doc.md +29 -0
- package/.github/skills/tech-writer/assets/templates/migration-plan.md +173 -0
- package/.github/skills/tech-writer/assets/templates/operations-runbook.md +202 -0
- package/.github/skills/tech-writer/assets/templates/pr-description.md +23 -0
- package/.github/skills/tech-writer/assets/templates/qiita.md +44 -0
- package/.github/skills/tech-writer/assets/templates/readme.md +38 -0
- package/.github/skills/tech-writer/assets/templates/requirements-definition.md +170 -0
- package/.github/skills/tech-writer/assets/templates/rfi.md +113 -0
- package/.github/skills/tech-writer/assets/templates/rfp.md +180 -0
- package/.github/skills/tech-writer/assets/templates/security-design.md +167 -0
- package/.github/skills/tech-writer/assets/templates/system-design.md +220 -0
- package/.github/skills/tech-writer/assets/templates/technical-proposal.md +112 -0
- package/.github/skills/tech-writer/assets/templates/test-plan.md +153 -0
- package/.github/skills/tech-writer/assets/templates/user-manual.md +22 -0
- package/.github/skills/tech-writer/assets/templates/white-paper.md +192 -0
- package/.github/skills/tech-writer/references/doctypes/api-docs.md +33 -0
- package/.github/skills/tech-writer/references/doctypes/blueprint.md +81 -0
- package/.github/skills/tech-writer/references/doctypes/code-comments.md +39 -0
- package/.github/skills/tech-writer/references/doctypes/design-doc.md +42 -0
- package/.github/skills/tech-writer/references/doctypes/migration-plan.md +63 -0
- package/.github/skills/tech-writer/references/doctypes/operations-runbook.md +63 -0
- package/.github/skills/tech-writer/references/doctypes/pr-commit.md +82 -0
- package/.github/skills/tech-writer/references/doctypes/qiita.md +75 -0
- package/.github/skills/tech-writer/references/doctypes/readme.md +43 -0
- package/.github/skills/tech-writer/references/doctypes/release-notes.md +30 -0
- package/.github/skills/tech-writer/references/doctypes/requirements-definition.md +61 -0
- package/.github/skills/tech-writer/references/doctypes/rfi.md +43 -0
- package/.github/skills/tech-writer/references/doctypes/rfp.md +46 -0
- package/.github/skills/tech-writer/references/doctypes/security-design.md +71 -0
- package/.github/skills/tech-writer/references/doctypes/system-design.md +74 -0
- package/.github/skills/tech-writer/references/doctypes/technical-proposal.md +49 -0
- package/.github/skills/tech-writer/references/doctypes/test-plan.md +67 -0
- package/.github/skills/tech-writer/references/doctypes/user-manual.md +58 -0
- package/.github/skills/tech-writer/references/doctypes/white-paper.md +84 -0
- package/.github/skills/tech-writer/references/doctypes/zenn.md +66 -0
- package/.github/skills/tech-writer/references/japanese-prose-optimization.md +110 -0
- package/.github/skills/tech-writer/references/style-constitution.md +104 -0
- package/.github/skills/tech-writer/scripts/lint.py +412 -0
- package/LICENSE +21 -0
- package/README.md +92 -0
- package/bin/ai-data-scientist.js +123 -0
- package/package.json +41 -0
- package/pyproject.toml +45 -0
- package/src/ai_chemistry_scientist/__init__.py +0 -0
- package/src/ai_chemistry_scientist/admet_prediction.py +71 -0
- package/src/ai_chemistry_scientist/bioactivity_classification.py +73 -0
- package/src/ai_chemistry_scientist/data/sample_molecules.csv +21 -0
- package/src/ai_chemistry_scientist/dispatch.py +369 -0
- package/src/ai_chemistry_scientist/docking_score.py +97 -0
- package/src/ai_chemistry_scientist/drug_likeness_rules.py +84 -0
- package/src/ai_chemistry_scientist/evidence.py +41 -0
- package/src/ai_chemistry_scientist/molecular_descriptors.py +97 -0
- package/src/ai_chemistry_scientist/molecular_formula_mass.py +40 -0
- package/src/ai_chemistry_scientist/molecular_similarity.py +78 -0
- package/src/ai_chemistry_scientist/qsar_modeling.py +105 -0
- package/src/ai_chemistry_scientist/salt_standardization.py +81 -0
- package/src/ai_chemistry_scientist/structural_alerts.py +76 -0
- package/src/ai_chemistry_scientist/structure_format_conversion.py +84 -0
- package/src/ai_chemistry_scientist/validation.py +70 -0
- package/src/ai_data_scientist/__init__.py +0 -0
- package/src/ai_data_scientist/analysis_assumptions.py +121 -0
- package/src/ai_data_scientist/anomaly_detection.py +39 -0
- package/src/ai_data_scientist/automl.py +109 -0
- package/src/ai_data_scientist/cleaning.py +56 -0
- package/src/ai_data_scientist/cli.py +90 -0
- package/src/ai_data_scientist/clustering.py +54 -0
- package/src/ai_data_scientist/dashboard.py +33 -0
- package/src/ai_data_scientist/data_definition.py +100 -0
- package/src/ai_data_scientist/data_quality.py +164 -0
- package/src/ai_data_scientist/dataset_validation.py +135 -0
- package/src/ai_data_scientist/dependency_pins.py +60 -0
- package/src/ai_data_scientist/eda.py +82 -0
- package/src/ai_data_scientist/experiment_evaluation.py +635 -0
- package/src/ai_data_scientist/explainability.py +340 -0
- package/src/ai_data_scientist/feature_engineering.py +163 -0
- package/src/ai_data_scientist/gate_config.py +32 -0
- package/src/ai_data_scientist/ingestion.py +127 -0
- package/src/ai_data_scientist/insight_engine.py +180 -0
- package/src/ai_data_scientist/japanese_nlp.py +43 -0
- package/src/ai_data_scientist/jupyter_launcher.py +137 -0
- package/src/ai_data_scientist/jupyter_mcp_client.py +94 -0
- package/src/ai_data_scientist/language_router.py +28 -0
- package/src/ai_data_scientist/lifecycle.py +221 -0
- package/src/ai_data_scientist/mcp_gateway.py +113 -0
- package/src/ai_data_scientist/mcp_runtime.py +194 -0
- package/src/ai_data_scientist/mcp_transport.py +53 -0
- package/src/ai_data_scientist/ml_modeling.py +451 -0
- package/src/ai_data_scientist/model_tuning.py +104 -0
- package/src/ai_data_scientist/notebook_audit.py +574 -0
- package/src/ai_data_scientist/project_manager.py +243 -0
- package/src/ai_data_scientist/report_export.py +73 -0
- package/src/ai_data_scientist/sensitivity.py +445 -0
- package/src/ai_data_scientist/signal_analysis.py +201 -0
- package/src/ai_data_scientist/skill_packaging.py +40 -0
- package/src/ai_data_scientist/stats_analysis.py +88 -0
- package/src/ai_data_scientist/text_nlp.py +44 -0
- package/src/ai_data_scientist/timeseries.py +68 -0
- package/src/ai_data_scientist/visualization.py +708 -0
- package/src/ai_genomics_scientist/__init__.py +1 -0
- package/src/ai_genomics_scientist/differential_expression.py +147 -0
- package/src/ai_genomics_scientist/dispatch.py +267 -0
- package/src/ai_genomics_scientist/evidence.py +45 -0
- package/src/ai_genomics_scientist/gene_set_enrichment.py +76 -0
- package/src/ai_genomics_scientist/sequence_alignment.py +97 -0
- package/src/ai_genomics_scientist/sequence_features.py +111 -0
- package/src/ai_genomics_scientist/splice_site_scoring.py +66 -0
- package/src/ai_genomics_scientist/validation.py +83 -0
- package/src/ai_genomics_scientist/variant_effect.py +147 -0
- package/src/ai_genomics_scientist/variant_pathogenicity.py +125 -0
- package/src/ai_materials_scientist/__init__.py +0 -0
- package/src/ai_materials_scientist/calphad.py +117 -0
- package/src/ai_materials_scientist/classical_monte_carlo.py +165 -0
- package/src/ai_materials_scientist/crystal_plasticity.py +184 -0
- package/src/ai_materials_scientist/dispatch.py +100 -0
- package/src/ai_materials_scientist/evidence.py +84 -0
- package/src/ai_materials_scientist/fem.py +279 -0
- package/src/ai_materials_scientist/kinetic_monte_carlo.py +145 -0
- package/src/ai_materials_scientist/molecular_dynamics.py +240 -0
- package/src/ai_materials_scientist/phase_field.py +167 -0
- package/src/ai_materials_scientist/validation.py +70 -0
- package/src/ai_scientist/__init__.py +1 -0
- package/src/ai_scientist/completion_gate.py +15 -0
- package/src/ai_scientist/data_analysis.py +46 -0
- package/src/ai_scientist/evidence_registry.py +99 -0
- package/src/ai_scientist/experimental_design.py +20 -0
- package/src/ai_scientist/language.py +14 -0
- package/src/ai_scientist/latex_renderer.py +41 -0
- package/src/ai_scientist/literature_review.py +37 -0
- package/src/ai_scientist/manifest.py +87 -0
- package/src/ai_scientist/manuscript.py +94 -0
- package/src/ai_scientist/mcp_config.py +76 -0
- package/src/ai_scientist/mcp_external.py +42 -0
- package/src/ai_scientist/mcp_failures.py +23 -0
- package/src/ai_scientist/mcp_gateway.py +38 -0
- package/src/ai_scientist/mcp_managed.py +180 -0
- package/src/ai_scientist/npm_packaging.py +49 -0
- package/src/ai_scientist/orchestrator.py +133 -0
- package/src/ai_scientist/peer_review.py +60 -0
- package/src/ai_scientist/phase_gate.py +74 -0
- package/src/ai_scientist/phase_state.py +230 -0
- package/src/ai_scientist/presentation.py +56 -0
- package/src/ai_scientist/project_config.py +31 -0
- package/src/ai_scientist/project_handle.py +74 -0
- package/src/ai_scientist/reproducibility.py +20 -0
- package/src/ai_scientist/research_planning.py +20 -0
- package/src/ai_scientist/skill_invocation.py +21 -0
- package/src/ai_scientist/tdd_gate.py +99 -0
- package/src/ai_structural_biology_scientist/__init__.py +0 -0
- package/src/ai_structural_biology_scientist/contact_map.py +87 -0
- package/src/ai_structural_biology_scientist/dispatch.py +269 -0
- package/src/ai_structural_biology_scientist/evidence.py +43 -0
- package/src/ai_structural_biology_scientist/hydrophobicity.py +101 -0
- package/src/ai_structural_biology_scientist/protein_docking_score.py +104 -0
- package/src/ai_structural_biology_scientist/secondary_structure.py +95 -0
- package/src/ai_structural_biology_scientist/structural_similarity.py +74 -0
- package/src/ai_structural_biology_scientist/validation.py +100 -0
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
"""Differential expression heuristic module (DES-AGENOM-060 / REQ-AGENOM-060)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from numbers import Integral
|
|
6
|
+
|
|
7
|
+
import numpy
|
|
8
|
+
from scipy import stats
|
|
9
|
+
|
|
10
|
+
from ai_genomics_scientist.validation import fail, ok, register_validator
|
|
11
|
+
|
|
12
|
+
_MODULE_NAME = "differential-expression"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _is_non_negative_integer(value: object) -> bool:
|
|
16
|
+
return isinstance(value, Integral) and not isinstance(value, bool) and value >= 0
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _differential_expression_validator(params: dict) -> dict:
|
|
20
|
+
"""DES-AGENOM-002 registered atomic validator for this module."""
|
|
21
|
+
counts = params.get("counts")
|
|
22
|
+
sample_groups = params.get("sample_groups")
|
|
23
|
+
|
|
24
|
+
if not isinstance(counts, dict) or not counts:
|
|
25
|
+
return fail("counts", "must be a non-empty dict of gene IDs to count lists")
|
|
26
|
+
if not isinstance(sample_groups, list) or not sample_groups:
|
|
27
|
+
return fail("sample_groups", "must be a non-empty list of group labels")
|
|
28
|
+
if not all(isinstance(gene, str) for gene in counts):
|
|
29
|
+
return fail("counts", "must have gene-ID strings as keys")
|
|
30
|
+
if not all(isinstance(label, str) for label in sample_groups):
|
|
31
|
+
return fail("sample_groups", "must contain only string group labels")
|
|
32
|
+
|
|
33
|
+
distinct_labels = sorted(set(sample_groups))
|
|
34
|
+
if len(distinct_labels) != 2:
|
|
35
|
+
return fail("sample_groups", "must contain exactly 2 distinct group labels")
|
|
36
|
+
if any(sample_groups.count(label) < 2 for label in distinct_labels):
|
|
37
|
+
return fail(
|
|
38
|
+
"sample_groups",
|
|
39
|
+
"each of the exactly 2 groups must have at least 2 replicate samples",
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
num_samples = len(sample_groups)
|
|
43
|
+
for values in counts.values():
|
|
44
|
+
if not isinstance(values, list) or len(values) != num_samples:
|
|
45
|
+
return fail("counts", "each gene's count list length must equal len(sample_groups)")
|
|
46
|
+
|
|
47
|
+
invalid_genes = [
|
|
48
|
+
gene
|
|
49
|
+
for gene, values in counts.items()
|
|
50
|
+
if not all(_is_non_negative_integer(value) for value in values)
|
|
51
|
+
]
|
|
52
|
+
if invalid_genes:
|
|
53
|
+
listed = ", ".join(f'"{gene}"' for gene in invalid_genes)
|
|
54
|
+
return fail(
|
|
55
|
+
"counts", f"must contain only non-negative integers (invalid gene(s): {listed})"
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
has_reference_gene = any(all(value > 0 for value in values) for values in counts.values())
|
|
59
|
+
if not has_reference_gene:
|
|
60
|
+
return fail(
|
|
61
|
+
"counts",
|
|
62
|
+
"must contain at least one gene with positive counts in every sample to compute size factors",
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
return ok()
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
register_validator(_MODULE_NAME, _differential_expression_validator)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _median_of_ratios_size_factors(
|
|
72
|
+
gene_ids: list[str], counts: dict, num_samples: int
|
|
73
|
+
) -> numpy.ndarray:
|
|
74
|
+
"""Classic median-of-ratios size factors, excluding any gene with a zero count."""
|
|
75
|
+
reference_rows = [
|
|
76
|
+
counts[gene_id] for gene_id in gene_ids if all(value > 0 for value in counts[gene_id])
|
|
77
|
+
]
|
|
78
|
+
reference_matrix = numpy.array(reference_rows, dtype=float)
|
|
79
|
+
geometric_means = numpy.exp(numpy.mean(numpy.log(reference_matrix), axis=1))
|
|
80
|
+
ratios = reference_matrix / geometric_means[:, None]
|
|
81
|
+
return numpy.median(ratios, axis=0)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _benjamini_hochberg(p_values: numpy.ndarray) -> numpy.ndarray:
|
|
85
|
+
"""Pure-numpy Benjamini-Hochberg step-up correction."""
|
|
86
|
+
num_tests = len(p_values)
|
|
87
|
+
order = numpy.argsort(p_values)
|
|
88
|
+
ranked = p_values[order]
|
|
89
|
+
raw = ranked * num_tests / (numpy.arange(num_tests) + 1)
|
|
90
|
+
monotone = numpy.minimum.accumulate(raw[::-1])[::-1]
|
|
91
|
+
adjusted = numpy.empty(num_tests)
|
|
92
|
+
adjusted[order] = numpy.clip(monotone, 0.0, 1.0)
|
|
93
|
+
return adjusted
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
# @id CODE-AGENOM-060
|
|
97
|
+
# @implements REQ-AGENOM-060
|
|
98
|
+
# @design DES-AGENOM-060
|
|
99
|
+
def run_differential_expression(counts: dict, sample_groups: list) -> list[dict]:
|
|
100
|
+
"""Compute median-of-ratios normalized differential expression results."""
|
|
101
|
+
gene_ids = list(counts.keys())
|
|
102
|
+
num_samples = len(sample_groups)
|
|
103
|
+
size_factors = _median_of_ratios_size_factors(gene_ids, counts, num_samples)
|
|
104
|
+
|
|
105
|
+
group1_label, group2_label = sorted(set(sample_groups))
|
|
106
|
+
group1_idx = [i for i, label in enumerate(sample_groups) if label == group1_label]
|
|
107
|
+
group2_idx = [i for i, label in enumerate(sample_groups) if label == group2_label]
|
|
108
|
+
|
|
109
|
+
results = []
|
|
110
|
+
p_values = []
|
|
111
|
+
base_means = []
|
|
112
|
+
log2_fold_changes = []
|
|
113
|
+
for gene_id in gene_ids:
|
|
114
|
+
raw = numpy.array(counts[gene_id], dtype=float)
|
|
115
|
+
normalized = raw / size_factors
|
|
116
|
+
base_mean = float(normalized.mean())
|
|
117
|
+
mean_group1 = normalized[group1_idx].mean()
|
|
118
|
+
mean_group2 = normalized[group2_idx].mean()
|
|
119
|
+
log2_fold_change = float(numpy.log2((mean_group2 + 1) / (mean_group1 + 1)))
|
|
120
|
+
|
|
121
|
+
log_normalized = numpy.log2(normalized + 1)
|
|
122
|
+
log_group1 = log_normalized[group1_idx]
|
|
123
|
+
log_group2 = log_normalized[group2_idx]
|
|
124
|
+
if numpy.var(log_group1) == 0 and numpy.var(log_group2) == 0:
|
|
125
|
+
p_value = 1.0
|
|
126
|
+
else:
|
|
127
|
+
_, p_value = stats.ttest_ind(log_group1, log_group2, equal_var=False)
|
|
128
|
+
p_value = float(p_value)
|
|
129
|
+
|
|
130
|
+
base_means.append(base_mean)
|
|
131
|
+
log2_fold_changes.append(log2_fold_change)
|
|
132
|
+
p_values.append(p_value)
|
|
133
|
+
|
|
134
|
+
padj = _benjamini_hochberg(numpy.array(p_values))
|
|
135
|
+
|
|
136
|
+
for index, gene_id in enumerate(gene_ids):
|
|
137
|
+
results.append(
|
|
138
|
+
{
|
|
139
|
+
"gene_id": gene_id,
|
|
140
|
+
"base_mean": base_means[index],
|
|
141
|
+
"log2_fold_change": log2_fold_changes[index],
|
|
142
|
+
"p_value": p_values[index],
|
|
143
|
+
"padj": float(padj[index]),
|
|
144
|
+
}
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
return sorted(results, key=lambda item: (item["padj"], item["gene_id"]))
|
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
"""Method manifest & request dispatcher (DES-AGENOM-001 / REQ-AGENOM-001/002)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import importlib
|
|
6
|
+
import json
|
|
7
|
+
import unicodedata
|
|
8
|
+
from collections.abc import Mapping
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
import numpy
|
|
13
|
+
import scipy
|
|
14
|
+
|
|
15
|
+
import ai_genomics_scientist.gene_set_enrichment as _gene_set_enrichment # noqa: F401
|
|
16
|
+
import ai_genomics_scientist.sequence_alignment as _sequence_alignment # noqa: F401
|
|
17
|
+
import ai_genomics_scientist.sequence_features as _sequence_features # noqa: F401
|
|
18
|
+
import ai_genomics_scientist.splice_site_scoring as _splice_site_scoring # noqa: F401
|
|
19
|
+
import ai_genomics_scientist.variant_effect as _variant_effect # noqa: F401
|
|
20
|
+
import ai_genomics_scientist.differential_expression as _differential_expression # noqa: F401
|
|
21
|
+
import ai_genomics_scientist.variant_pathogenicity as _variant_pathogenicity # noqa: F401
|
|
22
|
+
from ai_data_scientist.language_router import detect_language as _detect_language
|
|
23
|
+
from ai_genomics_scientist.evidence import record_run
|
|
24
|
+
from ai_genomics_scientist.validation import validate_parameters
|
|
25
|
+
|
|
26
|
+
REPO_ROOT = Path(__file__).resolve().parents[2]
|
|
27
|
+
DEFAULT_MANIFEST_PATH = REPO_ROOT / ".github" / "skills" / "ai-genomics-scientist" / "manifest.json"
|
|
28
|
+
_PER_ITEM_VALIDATED_MODULES = frozenset({"sequence-features"})
|
|
29
|
+
_PER_ITEM_BATCH_PARAM_NAMES = {"sequence-features": "sequences"}
|
|
30
|
+
_RUN_MODULE_PATHS = {
|
|
31
|
+
"sequence-features": "ai_genomics_scientist.sequence_features",
|
|
32
|
+
"variant-effect-annotation": "ai_genomics_scientist.variant_effect",
|
|
33
|
+
"splice-site-strength": "ai_genomics_scientist.splice_site_scoring",
|
|
34
|
+
"gene-set-enrichment": "ai_genomics_scientist.gene_set_enrichment",
|
|
35
|
+
"pairwise-sequence-alignment": "ai_genomics_scientist.sequence_alignment",
|
|
36
|
+
"differential-expression": "ai_genomics_scientist.differential_expression",
|
|
37
|
+
"variant-pathogenicity": "ai_genomics_scientist.variant_pathogenicity",
|
|
38
|
+
}
|
|
39
|
+
_RUN_FUNCTION_NAMES = {
|
|
40
|
+
"sequence-features": "run_sequence_features",
|
|
41
|
+
"variant-effect-annotation": "run_variant_effect",
|
|
42
|
+
"splice-site-strength": "run_splice_site_scoring",
|
|
43
|
+
"gene-set-enrichment": "run_gene_set_enrichment",
|
|
44
|
+
"pairwise-sequence-alignment": "run_sequence_alignment",
|
|
45
|
+
"differential-expression": "run_differential_expression",
|
|
46
|
+
"variant-pathogenicity": "run_variant_pathogenicity",
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def load_manifest(manifest_path: Path | None = None) -> dict:
|
|
51
|
+
"""Load the static method-name-to-module manifest."""
|
|
52
|
+
path = manifest_path or DEFAULT_MANIFEST_PATH
|
|
53
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _normalize_text(text: str) -> str:
|
|
57
|
+
return unicodedata.normalize("NFKC", text).casefold()
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _matched_methods(request_text: str, manifest: dict) -> list[str]:
|
|
61
|
+
normalized_request = _normalize_text(request_text)
|
|
62
|
+
matched: list[str] = []
|
|
63
|
+
for method, entry in manifest.items():
|
|
64
|
+
names = entry.get("names", {})
|
|
65
|
+
candidates = list(names.get("en", [])) + list(names.get("ja", []))
|
|
66
|
+
if any(_normalize_text(candidate) in normalized_request for candidate in candidates):
|
|
67
|
+
matched.append(method)
|
|
68
|
+
return matched
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def extract_params(request_text: str | Mapping[str, Any]) -> dict | None:
|
|
72
|
+
"""Extract exactly one balanced top-level JSON object from ``request_text``."""
|
|
73
|
+
if isinstance(request_text, Mapping):
|
|
74
|
+
return dict(request_text)
|
|
75
|
+
start = request_text.find("{")
|
|
76
|
+
if start == -1:
|
|
77
|
+
return None
|
|
78
|
+
depth = 0
|
|
79
|
+
for index in range(start, len(request_text)):
|
|
80
|
+
char = request_text[index]
|
|
81
|
+
if char == "{":
|
|
82
|
+
depth += 1
|
|
83
|
+
elif char == "}":
|
|
84
|
+
depth -= 1
|
|
85
|
+
if depth == 0:
|
|
86
|
+
candidate = request_text[start : index + 1]
|
|
87
|
+
try:
|
|
88
|
+
parsed = json.loads(candidate)
|
|
89
|
+
except json.JSONDecodeError:
|
|
90
|
+
return None
|
|
91
|
+
return parsed if isinstance(parsed, dict) else None
|
|
92
|
+
return None
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _resolve_run_function(method: str):
|
|
96
|
+
module = importlib.import_module(_RUN_MODULE_PATHS[method])
|
|
97
|
+
return getattr(module, _RUN_FUNCTION_NAMES[method])
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _no_params_outcome(language: str) -> dict:
|
|
101
|
+
constraint = (
|
|
102
|
+
"`request_text` に埋め込まれた `JSON` オブジェクトとして抽出可能でなければなりません"
|
|
103
|
+
if language == "ja"
|
|
104
|
+
else "must be extractable as a JSON object embedded in request_text"
|
|
105
|
+
)
|
|
106
|
+
return {
|
|
107
|
+
"ok": False,
|
|
108
|
+
"parameter": "params",
|
|
109
|
+
"constraint": constraint,
|
|
110
|
+
"language": language,
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _handle_module(
|
|
115
|
+
method: str, request_text: str, language: str, params: dict[str, Any] | None = None
|
|
116
|
+
) -> dict:
|
|
117
|
+
resolved_params = params if params is not None else extract_params(request_text)
|
|
118
|
+
if resolved_params is None:
|
|
119
|
+
return _no_params_outcome(language)
|
|
120
|
+
|
|
121
|
+
if method not in _PER_ITEM_VALIDATED_MODULES:
|
|
122
|
+
validation = validate_parameters(method, resolved_params)
|
|
123
|
+
if not validation["ok"]:
|
|
124
|
+
return {
|
|
125
|
+
"ok": False,
|
|
126
|
+
"parameter": validation["parameter"],
|
|
127
|
+
"constraint": validation["constraint"],
|
|
128
|
+
"language": language,
|
|
129
|
+
}
|
|
130
|
+
else:
|
|
131
|
+
# The per-item-validated module still needs its own batch container
|
|
132
|
+
# (``sequences``) checked for basic shape before iterating it: a
|
|
133
|
+
# missing key or a non-list value (e.g. a bare string, which Python
|
|
134
|
+
# would otherwise iterate character-by-character) must be rejected
|
|
135
|
+
# the same way an atomic module rejects a malformed parameter,
|
|
136
|
+
# rather than crashing or silently producing bogus per-item results.
|
|
137
|
+
batch_param_name = _PER_ITEM_BATCH_PARAM_NAMES[method]
|
|
138
|
+
batch_value = resolved_params.get(batch_param_name)
|
|
139
|
+
if not isinstance(batch_value, list):
|
|
140
|
+
return {
|
|
141
|
+
"ok": False,
|
|
142
|
+
"parameter": batch_param_name,
|
|
143
|
+
"constraint": "must be a list",
|
|
144
|
+
"language": language,
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
result = _resolve_run_function(method)(**resolved_params)
|
|
148
|
+
run_record = record_run(
|
|
149
|
+
module_name=method,
|
|
150
|
+
params=resolved_params,
|
|
151
|
+
result=result,
|
|
152
|
+
numpy_version=numpy.__version__,
|
|
153
|
+
scipy_version=scipy.__version__,
|
|
154
|
+
)
|
|
155
|
+
return {"ok": True, "run_record": run_record}
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
# @id CODE-AGENOM-011
|
|
159
|
+
# @implements REQ-AGENOM-002 REQ-AGENOM-004
|
|
160
|
+
# @design DES-AGENOM-001
|
|
161
|
+
def handle_sequence_features(request_text: str, language: str, **params) -> dict:
|
|
162
|
+
"""Handler wrapper for the sequence-features module."""
|
|
163
|
+
return _handle_module("sequence-features", request_text, language, params or None)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
# @id CODE-AGENOM-021
|
|
167
|
+
# @implements REQ-AGENOM-002 REQ-AGENOM-004
|
|
168
|
+
# @design DES-AGENOM-001
|
|
169
|
+
def handle_variant_effect_annotation(request_text: str, language: str, **params) -> dict:
|
|
170
|
+
"""Handler wrapper for the variant-effect-annotation module."""
|
|
171
|
+
return _handle_module("variant-effect-annotation", request_text, language, params or None)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
# @id CODE-AGENOM-031
|
|
175
|
+
# @implements REQ-AGENOM-002 REQ-AGENOM-004
|
|
176
|
+
# @design DES-AGENOM-001
|
|
177
|
+
def handle_splice_site_strength(request_text: str, language: str, **params) -> dict:
|
|
178
|
+
"""Handler wrapper for the splice-site-strength module."""
|
|
179
|
+
return _handle_module("splice-site-strength", request_text, language, params or None)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
# @id CODE-AGENOM-041
|
|
183
|
+
# @implements REQ-AGENOM-002 REQ-AGENOM-004
|
|
184
|
+
# @design DES-AGENOM-001
|
|
185
|
+
def handle_gene_set_enrichment(request_text: str, language: str, **params) -> dict:
|
|
186
|
+
"""Handler wrapper for the gene-set-enrichment module."""
|
|
187
|
+
return _handle_module("gene-set-enrichment", request_text, language, params or None)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
# @id CODE-AGENOM-051
|
|
191
|
+
# @implements REQ-AGENOM-002 REQ-AGENOM-004
|
|
192
|
+
# @design DES-AGENOM-001
|
|
193
|
+
def handle_pairwise_sequence_alignment(request_text: str, language: str, **params) -> dict:
|
|
194
|
+
"""Handler wrapper for the pairwise-sequence-alignment module."""
|
|
195
|
+
return _handle_module("pairwise-sequence-alignment", request_text, language, params or None)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
# @id CODE-AGENOM-061
|
|
199
|
+
# @implements REQ-AGENOM-002 REQ-AGENOM-004
|
|
200
|
+
# @design DES-AGENOM-001
|
|
201
|
+
def handle_differential_expression(request_text: str, language: str, **params) -> dict:
|
|
202
|
+
"""Handler wrapper for the differential-expression module."""
|
|
203
|
+
return _handle_module("differential-expression", request_text, language, params or None)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
# @id CODE-AGENOM-071
|
|
207
|
+
# @implements REQ-AGENOM-002 REQ-AGENOM-004
|
|
208
|
+
# @design DES-AGENOM-001
|
|
209
|
+
def handle_variant_pathogenicity(request_text: str, language: str, **params) -> dict:
|
|
210
|
+
"""Handler wrapper for the variant-pathogenicity module."""
|
|
211
|
+
return _handle_module("variant-pathogenicity", request_text, language, params or None)
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _render_clarification(candidates: list[str], language: str) -> str:
|
|
215
|
+
names = "、".join(candidates) if language == "ja" else ", ".join(candidates)
|
|
216
|
+
if language == "ja":
|
|
217
|
+
return f"{names} のどちらを意図していますか。明確にしてください。"
|
|
218
|
+
return f"Did you mean {names}? Please clarify which method you want."
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _render_rejection(language: str) -> str:
|
|
222
|
+
if language == "ja":
|
|
223
|
+
return "対応する手法が要求から認識されませんでした。"
|
|
224
|
+
return "No supported method was recognized in your request."
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
# @id CODE-AGENOM-006
|
|
228
|
+
# @implements REQ-AGENOM-001 REQ-AGENOM-002
|
|
229
|
+
# @design DES-AGENOM-001
|
|
230
|
+
def dispatch(
|
|
231
|
+
request_text: str,
|
|
232
|
+
language: str | None = None,
|
|
233
|
+
manifest_path: Path | None = None,
|
|
234
|
+
) -> dict:
|
|
235
|
+
"""Classify ``request_text`` and dispatch to exactly one matched module."""
|
|
236
|
+
if not isinstance(request_text, str):
|
|
237
|
+
raise ValueError("request_text: must be a str")
|
|
238
|
+
if language is not None and language not in {"en", "ja"}:
|
|
239
|
+
raise ValueError("language: must be 'en' or 'ja' when explicitly supplied")
|
|
240
|
+
|
|
241
|
+
detected_language = language or _detect_language(request_text)
|
|
242
|
+
manifest = load_manifest(manifest_path)
|
|
243
|
+
matched = _matched_methods(request_text, manifest)
|
|
244
|
+
|
|
245
|
+
if len(matched) == 1:
|
|
246
|
+
method = matched[0]
|
|
247
|
+
entry = manifest[method]
|
|
248
|
+
handler_module = importlib.import_module(entry["modulePath"])
|
|
249
|
+
handler = getattr(handler_module, entry["functionName"])
|
|
250
|
+
return {
|
|
251
|
+
"outcome": "dispatch",
|
|
252
|
+
"module": method,
|
|
253
|
+
"language": detected_language,
|
|
254
|
+
"handler_result": handler(request_text, detected_language),
|
|
255
|
+
}
|
|
256
|
+
if len(matched) > 1:
|
|
257
|
+
return {
|
|
258
|
+
"outcome": "clarification",
|
|
259
|
+
"candidates": matched,
|
|
260
|
+
"language": detected_language,
|
|
261
|
+
"clarification_question": _render_clarification(matched, detected_language),
|
|
262
|
+
}
|
|
263
|
+
return {
|
|
264
|
+
"outcome": "rejected",
|
|
265
|
+
"language": detected_language,
|
|
266
|
+
"rejected_method": _render_rejection(detected_language),
|
|
267
|
+
}
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Run evidence recorder (DES-AGENOM-003 / REQ-AGENOM-004)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
import numpy
|
|
8
|
+
|
|
9
|
+
SCHEMA_VERSION = 1
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _json_safe_value(value: Any) -> Any:
|
|
13
|
+
if value is None or isinstance(value, (str, bool, int, float)):
|
|
14
|
+
return value
|
|
15
|
+
if isinstance(value, numpy.generic):
|
|
16
|
+
return _json_safe_value(value.item())
|
|
17
|
+
if isinstance(value, dict):
|
|
18
|
+
return {key: _json_safe_value(nested) for key, nested in value.items()}
|
|
19
|
+
if isinstance(value, (list, tuple)):
|
|
20
|
+
return [_json_safe_value(item) for item in value]
|
|
21
|
+
raise TypeError(f"unsupported non-JSON-safe value: {type(value).__name__}")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
# @id CODE-AGENOM-001
|
|
25
|
+
# @implements REQ-AGENOM-004
|
|
26
|
+
# @design DES-AGENOM-003
|
|
27
|
+
def record_run(
|
|
28
|
+
module_name: str,
|
|
29
|
+
params: dict[str, Any],
|
|
30
|
+
result: Any,
|
|
31
|
+
*,
|
|
32
|
+
numpy_version: str,
|
|
33
|
+
scipy_version: str,
|
|
34
|
+
) -> dict:
|
|
35
|
+
"""Build a RunRecord with exactly metadata, parameters, and result."""
|
|
36
|
+
return {
|
|
37
|
+
"metadata": {
|
|
38
|
+
"module": module_name,
|
|
39
|
+
"schema_version": SCHEMA_VERSION,
|
|
40
|
+
"numpy_version": numpy_version,
|
|
41
|
+
"scipy_version": scipy_version,
|
|
42
|
+
},
|
|
43
|
+
"parameters": _json_safe_value(dict(params)),
|
|
44
|
+
"result": _json_safe_value(result),
|
|
45
|
+
}
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Gene-set enrichment heuristic module (DES-AGENOM-040 / REQ-AGENOM-040)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from scipy.stats import hypergeom
|
|
9
|
+
|
|
10
|
+
from ai_genomics_scientist.validation import fail, ok, register_validator
|
|
11
|
+
|
|
12
|
+
_MODULE_NAME = "gene-set-enrichment"
|
|
13
|
+
_REPO_DATA_PATH = Path(__file__).resolve().parent / "data" / "sample_gene_sets.csv"
|
|
14
|
+
_BACKGROUND_SIZE = 50
|
|
15
|
+
_GENE_UNIVERSE = {f"GENE{index:02d}" for index in range(1, _BACKGROUND_SIZE + 1)}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _load_gene_sets() -> list[dict[str, object]]:
|
|
19
|
+
with _REPO_DATA_PATH.open(encoding="utf-8", newline="") as handle:
|
|
20
|
+
reader = csv.DictReader(handle)
|
|
21
|
+
return [
|
|
22
|
+
{"pathway": row["pathway"], "genes": row["genes"].split(";")}
|
|
23
|
+
for row in reader
|
|
24
|
+
if row["pathway"] and row["genes"]
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _gene_set_enrichment_validator(params: dict) -> dict:
|
|
29
|
+
"""DES-AGENOM-002 registered atomic validator for this module."""
|
|
30
|
+
query_genes = params.get("query_genes")
|
|
31
|
+
if not isinstance(query_genes, list):
|
|
32
|
+
return fail("query_genes", "must be a list")
|
|
33
|
+
if not query_genes:
|
|
34
|
+
return fail("query_genes", "must be non-empty")
|
|
35
|
+
invalid_genes = [gene for gene in query_genes if not isinstance(gene, str)]
|
|
36
|
+
if invalid_genes:
|
|
37
|
+
listed = ", ".join(f'"{gene}"' for gene in invalid_genes)
|
|
38
|
+
return fail("query_genes", f"contains invalid gene(s): {listed}")
|
|
39
|
+
if len(set(query_genes)) != len(query_genes):
|
|
40
|
+
return fail("query_genes", "must be deduplicated")
|
|
41
|
+
|
|
42
|
+
invalid_genes = [gene for gene in query_genes if gene not in _GENE_UNIVERSE]
|
|
43
|
+
if invalid_genes:
|
|
44
|
+
listed = ", ".join(f'"{gene}"' for gene in invalid_genes)
|
|
45
|
+
return fail("query_genes", f"contains invalid gene(s): {listed}")
|
|
46
|
+
return ok()
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
register_validator(_MODULE_NAME, _gene_set_enrichment_validator)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
# @id CODE-AGENOM-040
|
|
53
|
+
# @implements REQ-AGENOM-040
|
|
54
|
+
# @design DES-AGENOM-040
|
|
55
|
+
def run_gene_set_enrichment(query_genes: list[str]) -> list[dict]:
|
|
56
|
+
"""Compute hypergeometric toy-pathway enrichment for validated ``query_genes``."""
|
|
57
|
+
query_gene_set = set(query_genes)
|
|
58
|
+
query_size = len(query_genes)
|
|
59
|
+
results = []
|
|
60
|
+
|
|
61
|
+
for entry in _load_gene_sets():
|
|
62
|
+
pathway = entry["pathway"]
|
|
63
|
+
pathway_genes = set(entry["genes"])
|
|
64
|
+
overlap = len(query_gene_set & pathway_genes)
|
|
65
|
+
pathway_size = len(pathway_genes)
|
|
66
|
+
p_value = float(hypergeom.sf(overlap - 1, _BACKGROUND_SIZE, pathway_size, query_size))
|
|
67
|
+
results.append(
|
|
68
|
+
{
|
|
69
|
+
"pathway": pathway,
|
|
70
|
+
"overlap": overlap,
|
|
71
|
+
"pathway_size": pathway_size,
|
|
72
|
+
"p_value": p_value,
|
|
73
|
+
}
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
return sorted(results, key=lambda item: (item["p_value"], item["pathway"]))
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Pairwise sequence alignment module (DES-AGENOM-050 / REQ-AGENOM-050)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from ai_genomics_scientist.validation import fail, ok, register_validator
|
|
6
|
+
|
|
7
|
+
_MODULE_NAME = "pairwise-sequence-alignment"
|
|
8
|
+
_DNA_BASES = frozenset({"A", "C", "G", "T"})
|
|
9
|
+
_MATCH_SCORE = 1
|
|
10
|
+
_MISMATCH_SCORE = -1
|
|
11
|
+
_GAP_SCORE = -2
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _is_dna_string(value: object) -> bool:
|
|
15
|
+
return isinstance(value, str) and value.isupper() and value and set(value).issubset(_DNA_BASES)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _sequence_alignment_validator(params: dict) -> dict:
|
|
19
|
+
"""DES-AGENOM-002 registered atomic validator for this module."""
|
|
20
|
+
seq1 = params.get("seq1")
|
|
21
|
+
seq2 = params.get("seq2")
|
|
22
|
+
if not _is_dna_string(seq1):
|
|
23
|
+
return fail("seq1", "must be a non-empty uppercase DNA string over {A,C,G,T}")
|
|
24
|
+
if not _is_dna_string(seq2):
|
|
25
|
+
return fail("seq2", "must be a non-empty uppercase DNA string over {A,C,G,T}")
|
|
26
|
+
return ok()
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
register_validator(_MODULE_NAME, _sequence_alignment_validator)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
# @id CODE-AGENOM-050
|
|
33
|
+
# @implements REQ-AGENOM-050
|
|
34
|
+
# @design DES-AGENOM-050
|
|
35
|
+
def run_sequence_alignment(seq1: str, seq2: str) -> dict:
|
|
36
|
+
"""Compute the fixed-score Needleman-Wunsch global alignment."""
|
|
37
|
+
rows = len(seq1) + 1
|
|
38
|
+
cols = len(seq2) + 1
|
|
39
|
+
scores = [[0] * cols for _ in range(rows)]
|
|
40
|
+
traceback = [[""] * cols for _ in range(rows)]
|
|
41
|
+
|
|
42
|
+
for row in range(1, rows):
|
|
43
|
+
scores[row][0] = row * _GAP_SCORE
|
|
44
|
+
traceback[row][0] = "up"
|
|
45
|
+
for col in range(1, cols):
|
|
46
|
+
scores[0][col] = col * _GAP_SCORE
|
|
47
|
+
traceback[0][col] = "left"
|
|
48
|
+
|
|
49
|
+
for row in range(1, rows):
|
|
50
|
+
for col in range(1, cols):
|
|
51
|
+
match_score = _MATCH_SCORE if seq1[row - 1] == seq2[col - 1] else _MISMATCH_SCORE
|
|
52
|
+
diagonal = scores[row - 1][col - 1] + match_score
|
|
53
|
+
up = scores[row - 1][col] + _GAP_SCORE
|
|
54
|
+
left = scores[row][col - 1] + _GAP_SCORE
|
|
55
|
+
best = max(diagonal, up, left)
|
|
56
|
+
scores[row][col] = best
|
|
57
|
+
if diagonal == best:
|
|
58
|
+
traceback[row][col] = "diag"
|
|
59
|
+
elif up == best:
|
|
60
|
+
traceback[row][col] = "up"
|
|
61
|
+
else:
|
|
62
|
+
traceback[row][col] = "left"
|
|
63
|
+
|
|
64
|
+
aligned_seq1: list[str] = []
|
|
65
|
+
aligned_seq2: list[str] = []
|
|
66
|
+
row = len(seq1)
|
|
67
|
+
col = len(seq2)
|
|
68
|
+
while row > 0 or col > 0:
|
|
69
|
+
move = traceback[row][col]
|
|
70
|
+
if move == "diag":
|
|
71
|
+
aligned_seq1.append(seq1[row - 1])
|
|
72
|
+
aligned_seq2.append(seq2[col - 1])
|
|
73
|
+
row -= 1
|
|
74
|
+
col -= 1
|
|
75
|
+
elif move == "up":
|
|
76
|
+
aligned_seq1.append(seq1[row - 1])
|
|
77
|
+
aligned_seq2.append("-")
|
|
78
|
+
row -= 1
|
|
79
|
+
else:
|
|
80
|
+
aligned_seq1.append("-")
|
|
81
|
+
aligned_seq2.append(seq2[col - 1])
|
|
82
|
+
col -= 1
|
|
83
|
+
|
|
84
|
+
aligned1 = "".join(reversed(aligned_seq1))
|
|
85
|
+
aligned2 = "".join(reversed(aligned_seq2))
|
|
86
|
+
matches = sum(
|
|
87
|
+
1
|
|
88
|
+
for base1, base2 in zip(aligned1, aligned2, strict=True)
|
|
89
|
+
if base1 == base2 and base1 != "-" and base2 != "-"
|
|
90
|
+
)
|
|
91
|
+
identity = matches / len(aligned1) if aligned1 else 0.0
|
|
92
|
+
return {
|
|
93
|
+
"aligned_seq1": aligned1,
|
|
94
|
+
"aligned_seq2": aligned2,
|
|
95
|
+
"score": scores[-1][-1],
|
|
96
|
+
"identity": identity,
|
|
97
|
+
}
|