jupytermind 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/skills/ai-chemistry-scientist/SKILL.md +97 -0
- package/.github/skills/ai-chemistry-scientist/manifest.json +156 -0
- package/.github/skills/ai-data-scientist/SKILL.md +330 -0
- package/.github/skills/ai-genomics-scientist/SKILL.md +98 -0
- package/.github/skills/ai-genomics-scientist/manifest.json +93 -0
- package/.github/skills/ai-materials-scientist/SKILL.md +51 -0
- package/.github/skills/ai-materials-scientist/manifest.json +58 -0
- package/.github/skills/ai-scientist/SKILL.md +69 -0
- package/.github/skills/ai-scientist/manifest.json +61 -0
- package/.github/skills/ai-structural-biology-scientist/SKILL.md +67 -0
- package/.github/skills/ai-structural-biology-scientist/manifest.json +72 -0
- package/.github/skills/japanese-prose/NOTICE.md +17 -0
- package/.github/skills/japanese-prose/SKILL.md +111 -0
- package/.github/skills/japanese-prose/references/review-workflow.md +50 -0
- package/.github/skills/japanese-prose/references/scoring.md +24 -0
- package/.github/skills/japanese-prose/references/writing-guidelines.md +60 -0
- package/.github/skills/japanese-prose/scripts/core.py +192 -0
- package/.github/skills/japanese-prose/scripts/fixtures/natural.md +5 -0
- package/.github/skills/japanese-prose/scripts/fixtures/unnatural.md +5 -0
- package/.github/skills/japanese-prose/scripts/lint.py +378 -0
- package/.github/skills/japanese-prose/scripts/outline.py +68 -0
- package/.github/skills/japanese-prose/scripts/terms.py +112 -0
- package/.github/skills/japanese-prose/scripts/test_engine.py +117 -0
- package/.github/skills/presentation-planner/SKILL.md +257 -0
- package/.github/skills/presentation-planner/assets/design-templates/data-report.yaml +97 -0
- package/.github/skills/presentation-planner/assets/design-templates/executive-proposal.yaml +92 -0
- package/.github/skills/presentation-planner/assets/design-templates/technical-briefing.yaml +96 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/data-report.md +47 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/executive-decision.md +43 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/technical-briefing.md +45 -0
- package/.github/skills/presentation-planner/references/customizing-design-templates.md +160 -0
- package/.github/skills/presentation-planner/references/design-spec-schema.md +72 -0
- package/.github/skills/presentation-planner/references/handoff-contract.md +49 -0
- package/.github/skills/presentation-planner/references/responsibility-boundary.md +32 -0
- package/.github/skills/presentation-planner/references/scenario-templates.md +55 -0
- package/.github/skills/tech-writer/SKILL.md +434 -0
- package/.github/skills/tech-writer/assets/templates/blueprint.md +187 -0
- package/.github/skills/tech-writer/assets/templates/design-doc.md +29 -0
- package/.github/skills/tech-writer/assets/templates/migration-plan.md +173 -0
- package/.github/skills/tech-writer/assets/templates/operations-runbook.md +202 -0
- package/.github/skills/tech-writer/assets/templates/pr-description.md +23 -0
- package/.github/skills/tech-writer/assets/templates/qiita.md +44 -0
- package/.github/skills/tech-writer/assets/templates/readme.md +38 -0
- package/.github/skills/tech-writer/assets/templates/requirements-definition.md +170 -0
- package/.github/skills/tech-writer/assets/templates/rfi.md +113 -0
- package/.github/skills/tech-writer/assets/templates/rfp.md +180 -0
- package/.github/skills/tech-writer/assets/templates/security-design.md +167 -0
- package/.github/skills/tech-writer/assets/templates/system-design.md +220 -0
- package/.github/skills/tech-writer/assets/templates/technical-proposal.md +112 -0
- package/.github/skills/tech-writer/assets/templates/test-plan.md +153 -0
- package/.github/skills/tech-writer/assets/templates/user-manual.md +22 -0
- package/.github/skills/tech-writer/assets/templates/white-paper.md +192 -0
- package/.github/skills/tech-writer/references/doctypes/api-docs.md +33 -0
- package/.github/skills/tech-writer/references/doctypes/blueprint.md +81 -0
- package/.github/skills/tech-writer/references/doctypes/code-comments.md +39 -0
- package/.github/skills/tech-writer/references/doctypes/design-doc.md +42 -0
- package/.github/skills/tech-writer/references/doctypes/migration-plan.md +63 -0
- package/.github/skills/tech-writer/references/doctypes/operations-runbook.md +63 -0
- package/.github/skills/tech-writer/references/doctypes/pr-commit.md +82 -0
- package/.github/skills/tech-writer/references/doctypes/qiita.md +75 -0
- package/.github/skills/tech-writer/references/doctypes/readme.md +43 -0
- package/.github/skills/tech-writer/references/doctypes/release-notes.md +30 -0
- package/.github/skills/tech-writer/references/doctypes/requirements-definition.md +61 -0
- package/.github/skills/tech-writer/references/doctypes/rfi.md +43 -0
- package/.github/skills/tech-writer/references/doctypes/rfp.md +46 -0
- package/.github/skills/tech-writer/references/doctypes/security-design.md +71 -0
- package/.github/skills/tech-writer/references/doctypes/system-design.md +74 -0
- package/.github/skills/tech-writer/references/doctypes/technical-proposal.md +49 -0
- package/.github/skills/tech-writer/references/doctypes/test-plan.md +67 -0
- package/.github/skills/tech-writer/references/doctypes/user-manual.md +58 -0
- package/.github/skills/tech-writer/references/doctypes/white-paper.md +84 -0
- package/.github/skills/tech-writer/references/doctypes/zenn.md +66 -0
- package/.github/skills/tech-writer/references/japanese-prose-optimization.md +110 -0
- package/.github/skills/tech-writer/references/style-constitution.md +104 -0
- package/.github/skills/tech-writer/scripts/lint.py +412 -0
- package/LICENSE +21 -0
- package/README.md +92 -0
- package/bin/ai-data-scientist.js +123 -0
- package/package.json +41 -0
- package/pyproject.toml +45 -0
- package/src/ai_chemistry_scientist/__init__.py +0 -0
- package/src/ai_chemistry_scientist/admet_prediction.py +71 -0
- package/src/ai_chemistry_scientist/bioactivity_classification.py +73 -0
- package/src/ai_chemistry_scientist/data/sample_molecules.csv +21 -0
- package/src/ai_chemistry_scientist/dispatch.py +369 -0
- package/src/ai_chemistry_scientist/docking_score.py +97 -0
- package/src/ai_chemistry_scientist/drug_likeness_rules.py +84 -0
- package/src/ai_chemistry_scientist/evidence.py +41 -0
- package/src/ai_chemistry_scientist/molecular_descriptors.py +97 -0
- package/src/ai_chemistry_scientist/molecular_formula_mass.py +40 -0
- package/src/ai_chemistry_scientist/molecular_similarity.py +78 -0
- package/src/ai_chemistry_scientist/qsar_modeling.py +105 -0
- package/src/ai_chemistry_scientist/salt_standardization.py +81 -0
- package/src/ai_chemistry_scientist/structural_alerts.py +76 -0
- package/src/ai_chemistry_scientist/structure_format_conversion.py +84 -0
- package/src/ai_chemistry_scientist/validation.py +70 -0
- package/src/ai_data_scientist/__init__.py +0 -0
- package/src/ai_data_scientist/analysis_assumptions.py +121 -0
- package/src/ai_data_scientist/anomaly_detection.py +39 -0
- package/src/ai_data_scientist/automl.py +109 -0
- package/src/ai_data_scientist/cleaning.py +56 -0
- package/src/ai_data_scientist/cli.py +90 -0
- package/src/ai_data_scientist/clustering.py +54 -0
- package/src/ai_data_scientist/dashboard.py +33 -0
- package/src/ai_data_scientist/data_definition.py +100 -0
- package/src/ai_data_scientist/data_quality.py +164 -0
- package/src/ai_data_scientist/dataset_validation.py +135 -0
- package/src/ai_data_scientist/dependency_pins.py +60 -0
- package/src/ai_data_scientist/eda.py +82 -0
- package/src/ai_data_scientist/experiment_evaluation.py +635 -0
- package/src/ai_data_scientist/explainability.py +340 -0
- package/src/ai_data_scientist/feature_engineering.py +163 -0
- package/src/ai_data_scientist/gate_config.py +32 -0
- package/src/ai_data_scientist/ingestion.py +127 -0
- package/src/ai_data_scientist/insight_engine.py +180 -0
- package/src/ai_data_scientist/japanese_nlp.py +43 -0
- package/src/ai_data_scientist/jupyter_launcher.py +137 -0
- package/src/ai_data_scientist/jupyter_mcp_client.py +94 -0
- package/src/ai_data_scientist/language_router.py +28 -0
- package/src/ai_data_scientist/lifecycle.py +221 -0
- package/src/ai_data_scientist/mcp_gateway.py +113 -0
- package/src/ai_data_scientist/mcp_runtime.py +194 -0
- package/src/ai_data_scientist/mcp_transport.py +53 -0
- package/src/ai_data_scientist/ml_modeling.py +451 -0
- package/src/ai_data_scientist/model_tuning.py +104 -0
- package/src/ai_data_scientist/notebook_audit.py +574 -0
- package/src/ai_data_scientist/project_manager.py +243 -0
- package/src/ai_data_scientist/report_export.py +73 -0
- package/src/ai_data_scientist/sensitivity.py +445 -0
- package/src/ai_data_scientist/signal_analysis.py +201 -0
- package/src/ai_data_scientist/skill_packaging.py +40 -0
- package/src/ai_data_scientist/stats_analysis.py +88 -0
- package/src/ai_data_scientist/text_nlp.py +44 -0
- package/src/ai_data_scientist/timeseries.py +68 -0
- package/src/ai_data_scientist/visualization.py +708 -0
- package/src/ai_genomics_scientist/__init__.py +1 -0
- package/src/ai_genomics_scientist/differential_expression.py +147 -0
- package/src/ai_genomics_scientist/dispatch.py +267 -0
- package/src/ai_genomics_scientist/evidence.py +45 -0
- package/src/ai_genomics_scientist/gene_set_enrichment.py +76 -0
- package/src/ai_genomics_scientist/sequence_alignment.py +97 -0
- package/src/ai_genomics_scientist/sequence_features.py +111 -0
- package/src/ai_genomics_scientist/splice_site_scoring.py +66 -0
- package/src/ai_genomics_scientist/validation.py +83 -0
- package/src/ai_genomics_scientist/variant_effect.py +147 -0
- package/src/ai_genomics_scientist/variant_pathogenicity.py +125 -0
- package/src/ai_materials_scientist/__init__.py +0 -0
- package/src/ai_materials_scientist/calphad.py +117 -0
- package/src/ai_materials_scientist/classical_monte_carlo.py +165 -0
- package/src/ai_materials_scientist/crystal_plasticity.py +184 -0
- package/src/ai_materials_scientist/dispatch.py +100 -0
- package/src/ai_materials_scientist/evidence.py +84 -0
- package/src/ai_materials_scientist/fem.py +279 -0
- package/src/ai_materials_scientist/kinetic_monte_carlo.py +145 -0
- package/src/ai_materials_scientist/molecular_dynamics.py +240 -0
- package/src/ai_materials_scientist/phase_field.py +167 -0
- package/src/ai_materials_scientist/validation.py +70 -0
- package/src/ai_scientist/__init__.py +1 -0
- package/src/ai_scientist/completion_gate.py +15 -0
- package/src/ai_scientist/data_analysis.py +46 -0
- package/src/ai_scientist/evidence_registry.py +99 -0
- package/src/ai_scientist/experimental_design.py +20 -0
- package/src/ai_scientist/language.py +14 -0
- package/src/ai_scientist/latex_renderer.py +41 -0
- package/src/ai_scientist/literature_review.py +37 -0
- package/src/ai_scientist/manifest.py +87 -0
- package/src/ai_scientist/manuscript.py +94 -0
- package/src/ai_scientist/mcp_config.py +76 -0
- package/src/ai_scientist/mcp_external.py +42 -0
- package/src/ai_scientist/mcp_failures.py +23 -0
- package/src/ai_scientist/mcp_gateway.py +38 -0
- package/src/ai_scientist/mcp_managed.py +180 -0
- package/src/ai_scientist/npm_packaging.py +49 -0
- package/src/ai_scientist/orchestrator.py +133 -0
- package/src/ai_scientist/peer_review.py +60 -0
- package/src/ai_scientist/phase_gate.py +74 -0
- package/src/ai_scientist/phase_state.py +230 -0
- package/src/ai_scientist/presentation.py +56 -0
- package/src/ai_scientist/project_config.py +31 -0
- package/src/ai_scientist/project_handle.py +74 -0
- package/src/ai_scientist/reproducibility.py +20 -0
- package/src/ai_scientist/research_planning.py +20 -0
- package/src/ai_scientist/skill_invocation.py +21 -0
- package/src/ai_scientist/tdd_gate.py +99 -0
- package/src/ai_structural_biology_scientist/__init__.py +0 -0
- package/src/ai_structural_biology_scientist/contact_map.py +87 -0
- package/src/ai_structural_biology_scientist/dispatch.py +269 -0
- package/src/ai_structural_biology_scientist/evidence.py +43 -0
- package/src/ai_structural_biology_scientist/hydrophobicity.py +101 -0
- package/src/ai_structural_biology_scientist/protein_docking_score.py +104 -0
- package/src/ai_structural_biology_scientist/secondary_structure.py +95 -0
- package/src/ai_structural_biology_scientist/structural_similarity.py +74 -0
- package/src/ai_structural_biology_scientist/validation.py +100 -0
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""Data-definition and provenance manifest with per-field confidence status.
|
|
2
|
+
|
|
3
|
+
Implements DES-AIDS-040 (REQ-AIDS-052): structurally distinguishes
|
|
4
|
+
reproducible file identity (owner/slug, SHA-256, retrieval time) from
|
|
5
|
+
verified semantic metadata (units, definitions, measurement pathway),
|
|
6
|
+
making data-definition uncertainty visible and queryable instead of
|
|
7
|
+
silently promoting an inferred value to "verified".
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from dataclasses import dataclass, field
|
|
13
|
+
|
|
14
|
+
_VALID_STATUSES = frozenset({"verified", "inferred", "reported", "unknown"})
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
# @id CODE-AIDS-071
|
|
18
|
+
# @implements REQ-AIDS-052
|
|
19
|
+
# @design DES-AIDS-040
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class FieldValue:
|
|
22
|
+
"""A single semantic field paired with its confidence status.
|
|
23
|
+
|
|
24
|
+
Frozen so a constructed instance's status cannot be mutated in place;
|
|
25
|
+
changing a field's confidence always requires building a brand-new
|
|
26
|
+
``FieldValue``, which is always an explicit caller action rather than an
|
|
27
|
+
automatic promotion from "inferred"/"unknown" to "verified".
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
value: object
|
|
31
|
+
status: str
|
|
32
|
+
source: str | None = None
|
|
33
|
+
|
|
34
|
+
def __post_init__(self) -> None:
|
|
35
|
+
if self.status not in _VALID_STATUSES:
|
|
36
|
+
raise ValueError(
|
|
37
|
+
f"status must be one of {sorted(_VALID_STATUSES)}, got {self.status!r}."
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
# @id CODE-AIDS-083
|
|
42
|
+
# @implements REQ-AIDS-052
|
|
43
|
+
# @design DES-AIDS-040
|
|
44
|
+
@dataclass(frozen=True)
|
|
45
|
+
class DataDefinitionManifest:
|
|
46
|
+
"""Aggregated data-definition manifest for one ingested dataset."""
|
|
47
|
+
|
|
48
|
+
source: dict
|
|
49
|
+
dataset_scope: dict
|
|
50
|
+
variables: dict[str, dict[str, FieldValue]]
|
|
51
|
+
transformations: tuple = field(default_factory=tuple)
|
|
52
|
+
|
|
53
|
+
def unresolved_fields(self) -> list[tuple[str, FieldValue]]:
|
|
54
|
+
"""Return every ``FieldValue`` across the manifest whose status is "unknown".
|
|
55
|
+
|
|
56
|
+
Each entry's first element is a dotted path identifying its location
|
|
57
|
+
(e.g. ``"source.license"`` or ``"variables.value.unit"``).
|
|
58
|
+
"""
|
|
59
|
+
return self._fields_with_status("unknown")
|
|
60
|
+
|
|
61
|
+
def inferred_fields(self) -> list[tuple[str, FieldValue]]:
|
|
62
|
+
"""Return every ``FieldValue`` across the manifest whose status is "inferred".
|
|
63
|
+
|
|
64
|
+
GitHub #33: surfaces unconfirmed-but-assumed fields as their own
|
|
65
|
+
actionable, listed item, separate from (and never overlapping
|
|
66
|
+
with) :meth:`unresolved_fields`'s "unknown" list.
|
|
67
|
+
"""
|
|
68
|
+
return self._fields_with_status("inferred")
|
|
69
|
+
|
|
70
|
+
def _fields_with_status(self, status: str) -> list[tuple[str, FieldValue]]:
|
|
71
|
+
matches: list[tuple[str, FieldValue]] = []
|
|
72
|
+
for key, field_value in self.source.items():
|
|
73
|
+
if isinstance(field_value, FieldValue) and field_value.status == status:
|
|
74
|
+
matches.append((f"source.{key}", field_value))
|
|
75
|
+
for key, field_value in self.dataset_scope.items():
|
|
76
|
+
if isinstance(field_value, FieldValue) and field_value.status == status:
|
|
77
|
+
matches.append((f"dataset_scope.{key}", field_value))
|
|
78
|
+
for variable_name, fields in self.variables.items():
|
|
79
|
+
for field_name, field_value in fields.items():
|
|
80
|
+
if isinstance(field_value, FieldValue) and field_value.status == status:
|
|
81
|
+
matches.append((f"variables.{variable_name}.{field_name}", field_value))
|
|
82
|
+
return matches
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
# @id CODE-AIDS-072
|
|
86
|
+
# @implements REQ-AIDS-052
|
|
87
|
+
# @design DES-AIDS-040
|
|
88
|
+
def build_manifest(
|
|
89
|
+
source: dict,
|
|
90
|
+
dataset_scope: dict,
|
|
91
|
+
variables: dict[str, dict[str, FieldValue]],
|
|
92
|
+
transformations: tuple = (),
|
|
93
|
+
) -> DataDefinitionManifest:
|
|
94
|
+
"""Construct a ``DataDefinitionManifest`` from its component dicts."""
|
|
95
|
+
return DataDefinitionManifest(
|
|
96
|
+
source=source,
|
|
97
|
+
dataset_scope=dataset_scope,
|
|
98
|
+
variables=variables,
|
|
99
|
+
transformations=tuple(transformations),
|
|
100
|
+
)
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""Semantic data-quality checks: schema-driven anomaly detection and
|
|
2
|
+
independent-dataset overlap validation.
|
|
3
|
+
|
|
4
|
+
Implements DES-AIDS-043 (REQ-AIDS-055). This module is deliberately
|
|
5
|
+
separate from ``anomaly_detection`` (which performs statistical
|
|
6
|
+
z-score outlier detection on a single numeric column); here, anomalies
|
|
7
|
+
are semantic constraint violations defined by a declarative schema
|
|
8
|
+
(ranges, allowed categories, non-null, uniqueness), and overlap
|
|
9
|
+
validation cross-checks summary statistics between a primary dataset
|
|
10
|
+
and an independent reference.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
|
|
17
|
+
import pandas as pd
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
# @id CODE-AIDS-076
|
|
21
|
+
# @implements REQ-AIDS-055
|
|
22
|
+
# @design DES-AIDS-043
|
|
23
|
+
@dataclass(frozen=True)
|
|
24
|
+
class AnomalyRecord:
|
|
25
|
+
"""A single schema-constraint violation."""
|
|
26
|
+
|
|
27
|
+
column: str
|
|
28
|
+
rule: str
|
|
29
|
+
row_count: int
|
|
30
|
+
message: str
|
|
31
|
+
row_indices: tuple[int, ...] = ()
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
# @id CODE-AIDS-077
|
|
35
|
+
# @implements REQ-AIDS-055
|
|
36
|
+
# @design DES-AIDS-043
|
|
37
|
+
def detect_anomalies(df: pd.DataFrame, schema: dict) -> tuple[AnomalyRecord, ...]:
|
|
38
|
+
"""Detect semantic constraint violations declared by ``schema``.
|
|
39
|
+
|
|
40
|
+
``schema`` maps column name -> a dict of constraints, any of:
|
|
41
|
+
- ``"min"`` / ``"max"``: numeric range bounds (inclusive).
|
|
42
|
+
- ``"allowed"``: iterable of permitted categorical values.
|
|
43
|
+
- ``"not_null"``: bool, require no missing values.
|
|
44
|
+
- ``"unique"``: bool, require no duplicate values.
|
|
45
|
+
Unknown columns in ``schema`` that are absent from ``df`` are skipped.
|
|
46
|
+
"""
|
|
47
|
+
records: list[AnomalyRecord] = []
|
|
48
|
+
for column, rules in schema.items():
|
|
49
|
+
if column not in df.columns:
|
|
50
|
+
continue
|
|
51
|
+
series = df[column]
|
|
52
|
+
|
|
53
|
+
if rules.get("not_null"):
|
|
54
|
+
missing = series.isna()
|
|
55
|
+
if missing.any():
|
|
56
|
+
records.append(
|
|
57
|
+
AnomalyRecord(
|
|
58
|
+
column=column,
|
|
59
|
+
rule="not_null",
|
|
60
|
+
row_count=int(missing.sum()),
|
|
61
|
+
message=f"Column {column!r} has {int(missing.sum())} null value(s).",
|
|
62
|
+
row_indices=tuple(series.index[missing]),
|
|
63
|
+
)
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
if "min" in rules or "max" in rules:
|
|
67
|
+
numeric = pd.to_numeric(series, errors="coerce")
|
|
68
|
+
below = (
|
|
69
|
+
numeric < rules["min"] if "min" in rules else pd.Series(False, index=series.index)
|
|
70
|
+
)
|
|
71
|
+
above = (
|
|
72
|
+
numeric > rules["max"] if "max" in rules else pd.Series(False, index=series.index)
|
|
73
|
+
)
|
|
74
|
+
out_of_range = (below | above) & numeric.notna()
|
|
75
|
+
if out_of_range.any():
|
|
76
|
+
records.append(
|
|
77
|
+
AnomalyRecord(
|
|
78
|
+
column=column,
|
|
79
|
+
rule="range",
|
|
80
|
+
row_count=int(out_of_range.sum()),
|
|
81
|
+
message=(
|
|
82
|
+
f"Column {column!r} has {int(out_of_range.sum())} value(s) "
|
|
83
|
+
f"outside [{rules.get('min')}, {rules.get('max')}]."
|
|
84
|
+
),
|
|
85
|
+
row_indices=tuple(series.index[out_of_range]),
|
|
86
|
+
)
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
if "allowed" in rules:
|
|
90
|
+
allowed = set(rules["allowed"])
|
|
91
|
+
disallowed = ~series.isin(allowed) & series.notna()
|
|
92
|
+
if disallowed.any():
|
|
93
|
+
records.append(
|
|
94
|
+
AnomalyRecord(
|
|
95
|
+
column=column,
|
|
96
|
+
rule="allowed_values",
|
|
97
|
+
row_count=int(disallowed.sum()),
|
|
98
|
+
message=(
|
|
99
|
+
f"Column {column!r} has {int(disallowed.sum())} value(s) "
|
|
100
|
+
f"not in the allowed set."
|
|
101
|
+
),
|
|
102
|
+
row_indices=tuple(series.index[disallowed]),
|
|
103
|
+
)
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
if rules.get("unique"):
|
|
107
|
+
duplicated = series.duplicated(keep=False) & series.notna()
|
|
108
|
+
if duplicated.any():
|
|
109
|
+
records.append(
|
|
110
|
+
AnomalyRecord(
|
|
111
|
+
column=column,
|
|
112
|
+
rule="unique",
|
|
113
|
+
row_count=int(duplicated.sum()),
|
|
114
|
+
message=f"Column {column!r} has {int(duplicated.sum())} duplicate value(s).",
|
|
115
|
+
row_indices=tuple(series.index[duplicated]),
|
|
116
|
+
)
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
return tuple(records)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
# @id CODE-AIDS-078
|
|
123
|
+
# @implements REQ-AIDS-055
|
|
124
|
+
# @design DES-AIDS-043
|
|
125
|
+
@dataclass(frozen=True)
|
|
126
|
+
class OverlapMismatch:
|
|
127
|
+
"""A single column whose summary statistic disagrees between datasets."""
|
|
128
|
+
|
|
129
|
+
column: str
|
|
130
|
+
primary_value: float
|
|
131
|
+
reference_value: float
|
|
132
|
+
relative_difference: float
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def validate_anomalies(
|
|
136
|
+
primary: pd.DataFrame,
|
|
137
|
+
reference: pd.DataFrame,
|
|
138
|
+
columns: list[str],
|
|
139
|
+
tolerance: float = 0.1,
|
|
140
|
+
) -> tuple[OverlapMismatch, ...]:
|
|
141
|
+
"""Compare per-column means between an independent ``reference`` dataset
|
|
142
|
+
and ``primary``, flagging columns whose relative difference exceeds
|
|
143
|
+
``tolerance``. Used to validate that detected anomalies are not an
|
|
144
|
+
artifact of a single dataset.
|
|
145
|
+
"""
|
|
146
|
+
mismatches: list[OverlapMismatch] = []
|
|
147
|
+
for column in columns:
|
|
148
|
+
if column not in primary.columns or column not in reference.columns:
|
|
149
|
+
continue
|
|
150
|
+
primary_mean = float(pd.to_numeric(primary[column], errors="coerce").mean())
|
|
151
|
+
reference_mean = float(pd.to_numeric(reference[column], errors="coerce").mean())
|
|
152
|
+
if reference_mean == 0:
|
|
153
|
+
continue
|
|
154
|
+
relative_difference = abs(primary_mean - reference_mean) / abs(reference_mean)
|
|
155
|
+
if relative_difference > tolerance:
|
|
156
|
+
mismatches.append(
|
|
157
|
+
OverlapMismatch(
|
|
158
|
+
column=column,
|
|
159
|
+
primary_value=primary_mean,
|
|
160
|
+
reference_value=reference_mean,
|
|
161
|
+
relative_difference=relative_difference,
|
|
162
|
+
)
|
|
163
|
+
)
|
|
164
|
+
return tuple(mismatches)
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""Independent-dataset overlap comparison (narrowed scope).
|
|
2
|
+
|
|
3
|
+
Implements DES-AIDS-045 (REQ-AIDS-057). Per the approved design, this
|
|
4
|
+
module operates only on already-loaded dataframes: ``compare_datasets``
|
|
5
|
+
checks key overlap and per-column value agreement between a primary
|
|
6
|
+
dataset and a candidate independent dataset. Automated dataset
|
|
7
|
+
*discovery* (e.g. searching Kaggle or other catalogs) is explicitly
|
|
8
|
+
out of scope for this repository and is not implemented here; callers
|
|
9
|
+
are expected to load the candidate dataset themselves before calling
|
|
10
|
+
this function.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
|
|
17
|
+
import pandas as pd
|
|
18
|
+
|
|
19
|
+
_VALID_RELATIONSHIPS = frozenset({"unknown", "independent", "derived", "overlapping"})
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
# @id CODE-AIDS-081
|
|
23
|
+
# @implements REQ-AIDS-057
|
|
24
|
+
# @design DES-AIDS-045
|
|
25
|
+
@dataclass(frozen=True)
|
|
26
|
+
class ColumnComparison:
|
|
27
|
+
"""Agreement statistics for one mapped column pair."""
|
|
28
|
+
|
|
29
|
+
primary_column: str
|
|
30
|
+
candidate_column: str
|
|
31
|
+
matched_rows: int
|
|
32
|
+
mismatched_rows: int
|
|
33
|
+
agreement_rate: float | None
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass(frozen=True)
|
|
37
|
+
class DatasetComparisonReport:
|
|
38
|
+
"""Outcome of comparing a primary dataset against a candidate dataset."""
|
|
39
|
+
|
|
40
|
+
candidate_relationship: str
|
|
41
|
+
matched_keys: int
|
|
42
|
+
primary_only_keys: int
|
|
43
|
+
candidate_only_keys: int
|
|
44
|
+
column_comparisons: tuple[ColumnComparison, ...] = field(default_factory=tuple)
|
|
45
|
+
|
|
46
|
+
def __post_init__(self) -> None:
|
|
47
|
+
if self.candidate_relationship not in _VALID_RELATIONSHIPS:
|
|
48
|
+
raise ValueError(
|
|
49
|
+
f"candidate_relationship must be one of {sorted(_VALID_RELATIONSHIPS)}, "
|
|
50
|
+
f"got {self.candidate_relationship!r}."
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
# @id CODE-AIDS-082
|
|
55
|
+
# @implements REQ-AIDS-057 REQ-AIDS-062
|
|
56
|
+
# @design DES-AIDS-045 DES-AIDS-050
|
|
57
|
+
def compare_datasets(
|
|
58
|
+
primary: pd.DataFrame,
|
|
59
|
+
candidate: pd.DataFrame,
|
|
60
|
+
key_mapping: dict[str, str],
|
|
61
|
+
value_mapping: dict[str, str],
|
|
62
|
+
candidate_relationship: str = "unknown",
|
|
63
|
+
) -> DatasetComparisonReport:
|
|
64
|
+
"""Compare ``primary`` against an independent ``candidate`` dataset.
|
|
65
|
+
|
|
66
|
+
``key_mapping`` maps primary key column name(s) -> candidate column
|
|
67
|
+
name(s), used to join the two datasets. ``value_mapping`` maps
|
|
68
|
+
primary value column name -> candidate value column name for
|
|
69
|
+
per-column agreement checks on the joined rows.
|
|
70
|
+
|
|
71
|
+
Rows whose key is null on either side are excluded from
|
|
72
|
+
``matched_keys``/``primary_only_keys``/``candidate_only_keys`` and from
|
|
73
|
+
the per-column agreement computation (REQ-AIDS-062): otherwise two
|
|
74
|
+
null keys (e.g. both ``None``) would incorrectly count as a matched key
|
|
75
|
+
pair under Python tuple-set equality.
|
|
76
|
+
"""
|
|
77
|
+
primary_keys = list(key_mapping.keys())
|
|
78
|
+
candidate_keys = list(key_mapping.values())
|
|
79
|
+
|
|
80
|
+
eligible_primary = primary.dropna(subset=primary_keys)
|
|
81
|
+
eligible_candidate = candidate.dropna(subset=candidate_keys)
|
|
82
|
+
|
|
83
|
+
primary_key_values = set(
|
|
84
|
+
map(tuple, eligible_primary[primary_keys].itertuples(index=False, name=None))
|
|
85
|
+
)
|
|
86
|
+
candidate_key_values = set(
|
|
87
|
+
map(tuple, eligible_candidate[candidate_keys].itertuples(index=False, name=None))
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
matched_keys = primary_key_values & candidate_key_values
|
|
91
|
+
primary_only_keys = primary_key_values - candidate_key_values
|
|
92
|
+
candidate_only_keys = candidate_key_values - primary_key_values
|
|
93
|
+
|
|
94
|
+
merged = eligible_primary.merge(
|
|
95
|
+
eligible_candidate,
|
|
96
|
+
left_on=primary_keys,
|
|
97
|
+
right_on=candidate_keys,
|
|
98
|
+
how="inner",
|
|
99
|
+
suffixes=("_primary", "_candidate"),
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
column_comparisons: list[ColumnComparison] = []
|
|
103
|
+
for primary_column, candidate_column in value_mapping.items():
|
|
104
|
+
primary_col_name = (
|
|
105
|
+
f"{primary_column}_primary" if primary_column in candidate.columns else primary_column
|
|
106
|
+
)
|
|
107
|
+
candidate_col_name = (
|
|
108
|
+
f"{candidate_column}_candidate"
|
|
109
|
+
if candidate_column in primary.columns
|
|
110
|
+
else candidate_column
|
|
111
|
+
)
|
|
112
|
+
if primary_col_name not in merged.columns or candidate_col_name not in merged.columns:
|
|
113
|
+
continue
|
|
114
|
+
matches = merged[primary_col_name] == merged[candidate_col_name]
|
|
115
|
+
matched_rows = int(matches.sum())
|
|
116
|
+
mismatched_rows = int((~matches).sum())
|
|
117
|
+
total = matched_rows + mismatched_rows
|
|
118
|
+
agreement_rate = (matched_rows / total) if total else None
|
|
119
|
+
column_comparisons.append(
|
|
120
|
+
ColumnComparison(
|
|
121
|
+
primary_column=primary_column,
|
|
122
|
+
candidate_column=candidate_column,
|
|
123
|
+
matched_rows=matched_rows,
|
|
124
|
+
mismatched_rows=mismatched_rows,
|
|
125
|
+
agreement_rate=agreement_rate,
|
|
126
|
+
)
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
return DatasetComparisonReport(
|
|
130
|
+
candidate_relationship=candidate_relationship,
|
|
131
|
+
matched_keys=len(matched_keys),
|
|
132
|
+
primary_only_keys=len(primary_only_keys),
|
|
133
|
+
candidate_only_keys=len(candidate_only_keys),
|
|
134
|
+
column_comparisons=tuple(column_comparisons),
|
|
135
|
+
)
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""Read declared dependency version specifiers from pyproject.toml.
|
|
2
|
+
|
|
3
|
+
Implements DES-AIDS-038 (REQ-AIDS-050): a minimal, dependency-free reader of
|
|
4
|
+
pyproject.toml's [project].dependencies array, used to assert the
|
|
5
|
+
jupyter-mcp-server / mcp version pins stay explicitly bounded on both sides
|
|
6
|
+
(no silent drift to an untested release whose negotiated MCP protocol
|
|
7
|
+
version is incompatible with the rest of the pinned stack).
|
|
8
|
+
|
|
9
|
+
Deliberately avoids a TOML-parsing library: stdlib ``tomllib`` only ships
|
|
10
|
+
from Python 3.11, and this project still supports 3.10, so the
|
|
11
|
+
``dependencies = [...]`` array is scanned as plain text instead.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import re
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
_PYPROJECT_PATH = Path(__file__).resolve().parents[2] / "pyproject.toml"
|
|
20
|
+
_DEPENDENCY_LINE_PATTERN = re.compile(r'^\s*"([A-Za-z0-9_.-]+)([^"]*)"\s*,?\s*$')
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _normalize(name: str) -> str:
|
|
24
|
+
"""Normalize a package name per PEP 503 (case/separator-insensitive)."""
|
|
25
|
+
return re.sub(r"[-_.]+", "-", name).lower()
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
# @id CODE-AIDS-059
|
|
29
|
+
# @implements REQ-AIDS-050
|
|
30
|
+
# @design DES-AIDS-038
|
|
31
|
+
def get_dependency_specifier(name: str, pyproject_path: Path = _PYPROJECT_PATH) -> str:
|
|
32
|
+
"""Return the raw dependency requirement string declared for ``name``.
|
|
33
|
+
|
|
34
|
+
Scans ``pyproject_path``'s ``[project].dependencies`` array for the
|
|
35
|
+
single entry whose package name matches ``name`` (case/separator
|
|
36
|
+
insensitive), returning it verbatim including its version specifier
|
|
37
|
+
(e.g. ``"jupyter-mcp-server>=2.2,<2.3"``).
|
|
38
|
+
|
|
39
|
+
Raises ``KeyError`` if no dependency named ``name`` is declared.
|
|
40
|
+
"""
|
|
41
|
+
target = _normalize(name)
|
|
42
|
+
in_dependencies = False
|
|
43
|
+
for line in pyproject_path.read_text(encoding="utf-8").splitlines():
|
|
44
|
+
stripped = line.strip()
|
|
45
|
+
if not in_dependencies:
|
|
46
|
+
if stripped.startswith("dependencies"):
|
|
47
|
+
in_dependencies = True
|
|
48
|
+
continue
|
|
49
|
+
if stripped.startswith("]"):
|
|
50
|
+
break
|
|
51
|
+
match = _DEPENDENCY_LINE_PATTERN.match(line)
|
|
52
|
+
if not match:
|
|
53
|
+
continue
|
|
54
|
+
raw_name = match.group(1)
|
|
55
|
+
if _normalize(raw_name) == target:
|
|
56
|
+
return stripped.rstrip(",").strip('"')
|
|
57
|
+
|
|
58
|
+
raise KeyError(
|
|
59
|
+
f"No dependency named {name!r} declared in {pyproject_path}'s [project.dependencies]."
|
|
60
|
+
)
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""Exploratory data analysis.
|
|
2
|
+
|
|
3
|
+
Implements DES-AIDS-007 (REQ-AIDS-005): summary statistics, dtypes, and
|
|
4
|
+
missing-value counts matching pandas describe()/info() reference values.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
|
|
11
|
+
import pandas as pd
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class EDAReport:
|
|
16
|
+
describe: dict
|
|
17
|
+
dtypes: dict
|
|
18
|
+
non_null_counts: dict
|
|
19
|
+
missing_summary: dict
|
|
20
|
+
categorical_summary: dict
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
DEFAULT_TOP_N = 10
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
# @id CODE-AIDS-005
|
|
27
|
+
# @implements REQ-AIDS-005
|
|
28
|
+
# @design DES-AIDS-007
|
|
29
|
+
# @id CODE-AIDS-051
|
|
30
|
+
# @implements REQ-AIDS-043
|
|
31
|
+
# @design DES-AIDS-031
|
|
32
|
+
def explore(df: pd.DataFrame, top_n: int = DEFAULT_TOP_N) -> EDAReport:
|
|
33
|
+
"""Compute summary statistics, dtypes and non-null counts for ``df``.
|
|
34
|
+
|
|
35
|
+
Also reports, per column, a missing-value count/ratio, and, for
|
|
36
|
+
categorical (object/category/bool) columns, a unique-value count and
|
|
37
|
+
the ``top_n`` most frequent values with their counts and ratios
|
|
38
|
+
(``truncated`` is set when more unique values exist than ``top_n``).
|
|
39
|
+
The existing ``describe``/``dtypes``/``non_null_counts`` fields are
|
|
40
|
+
unchanged by this extension.
|
|
41
|
+
"""
|
|
42
|
+
describe = df.describe().to_dict() if len(df.columns) > 0 else {}
|
|
43
|
+
dtypes = {column: str(dtype) for column, dtype in df.dtypes.items()}
|
|
44
|
+
non_null_counts = df.count().to_dict()
|
|
45
|
+
|
|
46
|
+
row_count = len(df)
|
|
47
|
+
missing_summary = {}
|
|
48
|
+
for column in df.columns:
|
|
49
|
+
missing_count = int(df[column].isna().sum())
|
|
50
|
+
missing_ratio = (missing_count / row_count) if row_count else 0.0
|
|
51
|
+
missing_summary[column] = {
|
|
52
|
+
"missing_count": missing_count,
|
|
53
|
+
"missing_ratio": missing_ratio,
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
categorical_summary = {}
|
|
57
|
+
categorical_columns = df.select_dtypes(include=["object", "str", "category", "bool"]).columns
|
|
58
|
+
for column in categorical_columns:
|
|
59
|
+
value_counts = df[column].value_counts(dropna=True)
|
|
60
|
+
unique_count = int(value_counts.shape[0])
|
|
61
|
+
non_null_total = int(value_counts.sum())
|
|
62
|
+
top_values = [
|
|
63
|
+
{
|
|
64
|
+
"value": value,
|
|
65
|
+
"count": int(count),
|
|
66
|
+
"ratio": (count / non_null_total) if non_null_total else 0.0,
|
|
67
|
+
}
|
|
68
|
+
for value, count in value_counts.head(top_n).items()
|
|
69
|
+
]
|
|
70
|
+
categorical_summary[column] = {
|
|
71
|
+
"unique_count": unique_count,
|
|
72
|
+
"top_values": top_values,
|
|
73
|
+
"truncated": unique_count > top_n,
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
return EDAReport(
|
|
77
|
+
describe=describe,
|
|
78
|
+
dtypes=dtypes,
|
|
79
|
+
non_null_counts=non_null_counts,
|
|
80
|
+
missing_summary=missing_summary,
|
|
81
|
+
categorical_summary=categorical_summary,
|
|
82
|
+
)
|