jupytermind 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (193) hide show
  1. package/.github/skills/ai-chemistry-scientist/SKILL.md +97 -0
  2. package/.github/skills/ai-chemistry-scientist/manifest.json +156 -0
  3. package/.github/skills/ai-data-scientist/SKILL.md +330 -0
  4. package/.github/skills/ai-genomics-scientist/SKILL.md +98 -0
  5. package/.github/skills/ai-genomics-scientist/manifest.json +93 -0
  6. package/.github/skills/ai-materials-scientist/SKILL.md +51 -0
  7. package/.github/skills/ai-materials-scientist/manifest.json +58 -0
  8. package/.github/skills/ai-scientist/SKILL.md +69 -0
  9. package/.github/skills/ai-scientist/manifest.json +61 -0
  10. package/.github/skills/ai-structural-biology-scientist/SKILL.md +67 -0
  11. package/.github/skills/ai-structural-biology-scientist/manifest.json +72 -0
  12. package/.github/skills/japanese-prose/NOTICE.md +17 -0
  13. package/.github/skills/japanese-prose/SKILL.md +111 -0
  14. package/.github/skills/japanese-prose/references/review-workflow.md +50 -0
  15. package/.github/skills/japanese-prose/references/scoring.md +24 -0
  16. package/.github/skills/japanese-prose/references/writing-guidelines.md +60 -0
  17. package/.github/skills/japanese-prose/scripts/core.py +192 -0
  18. package/.github/skills/japanese-prose/scripts/fixtures/natural.md +5 -0
  19. package/.github/skills/japanese-prose/scripts/fixtures/unnatural.md +5 -0
  20. package/.github/skills/japanese-prose/scripts/lint.py +378 -0
  21. package/.github/skills/japanese-prose/scripts/outline.py +68 -0
  22. package/.github/skills/japanese-prose/scripts/terms.py +112 -0
  23. package/.github/skills/japanese-prose/scripts/test_engine.py +117 -0
  24. package/.github/skills/presentation-planner/SKILL.md +257 -0
  25. package/.github/skills/presentation-planner/assets/design-templates/data-report.yaml +97 -0
  26. package/.github/skills/presentation-planner/assets/design-templates/executive-proposal.yaml +92 -0
  27. package/.github/skills/presentation-planner/assets/design-templates/technical-briefing.yaml +96 -0
  28. package/.github/skills/presentation-planner/assets/scenario-templates/data-report.md +47 -0
  29. package/.github/skills/presentation-planner/assets/scenario-templates/executive-decision.md +43 -0
  30. package/.github/skills/presentation-planner/assets/scenario-templates/technical-briefing.md +45 -0
  31. package/.github/skills/presentation-planner/references/customizing-design-templates.md +160 -0
  32. package/.github/skills/presentation-planner/references/design-spec-schema.md +72 -0
  33. package/.github/skills/presentation-planner/references/handoff-contract.md +49 -0
  34. package/.github/skills/presentation-planner/references/responsibility-boundary.md +32 -0
  35. package/.github/skills/presentation-planner/references/scenario-templates.md +55 -0
  36. package/.github/skills/tech-writer/SKILL.md +434 -0
  37. package/.github/skills/tech-writer/assets/templates/blueprint.md +187 -0
  38. package/.github/skills/tech-writer/assets/templates/design-doc.md +29 -0
  39. package/.github/skills/tech-writer/assets/templates/migration-plan.md +173 -0
  40. package/.github/skills/tech-writer/assets/templates/operations-runbook.md +202 -0
  41. package/.github/skills/tech-writer/assets/templates/pr-description.md +23 -0
  42. package/.github/skills/tech-writer/assets/templates/qiita.md +44 -0
  43. package/.github/skills/tech-writer/assets/templates/readme.md +38 -0
  44. package/.github/skills/tech-writer/assets/templates/requirements-definition.md +170 -0
  45. package/.github/skills/tech-writer/assets/templates/rfi.md +113 -0
  46. package/.github/skills/tech-writer/assets/templates/rfp.md +180 -0
  47. package/.github/skills/tech-writer/assets/templates/security-design.md +167 -0
  48. package/.github/skills/tech-writer/assets/templates/system-design.md +220 -0
  49. package/.github/skills/tech-writer/assets/templates/technical-proposal.md +112 -0
  50. package/.github/skills/tech-writer/assets/templates/test-plan.md +153 -0
  51. package/.github/skills/tech-writer/assets/templates/user-manual.md +22 -0
  52. package/.github/skills/tech-writer/assets/templates/white-paper.md +192 -0
  53. package/.github/skills/tech-writer/references/doctypes/api-docs.md +33 -0
  54. package/.github/skills/tech-writer/references/doctypes/blueprint.md +81 -0
  55. package/.github/skills/tech-writer/references/doctypes/code-comments.md +39 -0
  56. package/.github/skills/tech-writer/references/doctypes/design-doc.md +42 -0
  57. package/.github/skills/tech-writer/references/doctypes/migration-plan.md +63 -0
  58. package/.github/skills/tech-writer/references/doctypes/operations-runbook.md +63 -0
  59. package/.github/skills/tech-writer/references/doctypes/pr-commit.md +82 -0
  60. package/.github/skills/tech-writer/references/doctypes/qiita.md +75 -0
  61. package/.github/skills/tech-writer/references/doctypes/readme.md +43 -0
  62. package/.github/skills/tech-writer/references/doctypes/release-notes.md +30 -0
  63. package/.github/skills/tech-writer/references/doctypes/requirements-definition.md +61 -0
  64. package/.github/skills/tech-writer/references/doctypes/rfi.md +43 -0
  65. package/.github/skills/tech-writer/references/doctypes/rfp.md +46 -0
  66. package/.github/skills/tech-writer/references/doctypes/security-design.md +71 -0
  67. package/.github/skills/tech-writer/references/doctypes/system-design.md +74 -0
  68. package/.github/skills/tech-writer/references/doctypes/technical-proposal.md +49 -0
  69. package/.github/skills/tech-writer/references/doctypes/test-plan.md +67 -0
  70. package/.github/skills/tech-writer/references/doctypes/user-manual.md +58 -0
  71. package/.github/skills/tech-writer/references/doctypes/white-paper.md +84 -0
  72. package/.github/skills/tech-writer/references/doctypes/zenn.md +66 -0
  73. package/.github/skills/tech-writer/references/japanese-prose-optimization.md +110 -0
  74. package/.github/skills/tech-writer/references/style-constitution.md +104 -0
  75. package/.github/skills/tech-writer/scripts/lint.py +412 -0
  76. package/LICENSE +21 -0
  77. package/README.md +92 -0
  78. package/bin/ai-data-scientist.js +123 -0
  79. package/package.json +41 -0
  80. package/pyproject.toml +45 -0
  81. package/src/ai_chemistry_scientist/__init__.py +0 -0
  82. package/src/ai_chemistry_scientist/admet_prediction.py +71 -0
  83. package/src/ai_chemistry_scientist/bioactivity_classification.py +73 -0
  84. package/src/ai_chemistry_scientist/data/sample_molecules.csv +21 -0
  85. package/src/ai_chemistry_scientist/dispatch.py +369 -0
  86. package/src/ai_chemistry_scientist/docking_score.py +97 -0
  87. package/src/ai_chemistry_scientist/drug_likeness_rules.py +84 -0
  88. package/src/ai_chemistry_scientist/evidence.py +41 -0
  89. package/src/ai_chemistry_scientist/molecular_descriptors.py +97 -0
  90. package/src/ai_chemistry_scientist/molecular_formula_mass.py +40 -0
  91. package/src/ai_chemistry_scientist/molecular_similarity.py +78 -0
  92. package/src/ai_chemistry_scientist/qsar_modeling.py +105 -0
  93. package/src/ai_chemistry_scientist/salt_standardization.py +81 -0
  94. package/src/ai_chemistry_scientist/structural_alerts.py +76 -0
  95. package/src/ai_chemistry_scientist/structure_format_conversion.py +84 -0
  96. package/src/ai_chemistry_scientist/validation.py +70 -0
  97. package/src/ai_data_scientist/__init__.py +0 -0
  98. package/src/ai_data_scientist/analysis_assumptions.py +121 -0
  99. package/src/ai_data_scientist/anomaly_detection.py +39 -0
  100. package/src/ai_data_scientist/automl.py +109 -0
  101. package/src/ai_data_scientist/cleaning.py +56 -0
  102. package/src/ai_data_scientist/cli.py +90 -0
  103. package/src/ai_data_scientist/clustering.py +54 -0
  104. package/src/ai_data_scientist/dashboard.py +33 -0
  105. package/src/ai_data_scientist/data_definition.py +100 -0
  106. package/src/ai_data_scientist/data_quality.py +164 -0
  107. package/src/ai_data_scientist/dataset_validation.py +135 -0
  108. package/src/ai_data_scientist/dependency_pins.py +60 -0
  109. package/src/ai_data_scientist/eda.py +82 -0
  110. package/src/ai_data_scientist/experiment_evaluation.py +635 -0
  111. package/src/ai_data_scientist/explainability.py +340 -0
  112. package/src/ai_data_scientist/feature_engineering.py +163 -0
  113. package/src/ai_data_scientist/gate_config.py +32 -0
  114. package/src/ai_data_scientist/ingestion.py +127 -0
  115. package/src/ai_data_scientist/insight_engine.py +180 -0
  116. package/src/ai_data_scientist/japanese_nlp.py +43 -0
  117. package/src/ai_data_scientist/jupyter_launcher.py +137 -0
  118. package/src/ai_data_scientist/jupyter_mcp_client.py +94 -0
  119. package/src/ai_data_scientist/language_router.py +28 -0
  120. package/src/ai_data_scientist/lifecycle.py +221 -0
  121. package/src/ai_data_scientist/mcp_gateway.py +113 -0
  122. package/src/ai_data_scientist/mcp_runtime.py +194 -0
  123. package/src/ai_data_scientist/mcp_transport.py +53 -0
  124. package/src/ai_data_scientist/ml_modeling.py +451 -0
  125. package/src/ai_data_scientist/model_tuning.py +104 -0
  126. package/src/ai_data_scientist/notebook_audit.py +574 -0
  127. package/src/ai_data_scientist/project_manager.py +243 -0
  128. package/src/ai_data_scientist/report_export.py +73 -0
  129. package/src/ai_data_scientist/sensitivity.py +445 -0
  130. package/src/ai_data_scientist/signal_analysis.py +201 -0
  131. package/src/ai_data_scientist/skill_packaging.py +40 -0
  132. package/src/ai_data_scientist/stats_analysis.py +88 -0
  133. package/src/ai_data_scientist/text_nlp.py +44 -0
  134. package/src/ai_data_scientist/timeseries.py +68 -0
  135. package/src/ai_data_scientist/visualization.py +708 -0
  136. package/src/ai_genomics_scientist/__init__.py +1 -0
  137. package/src/ai_genomics_scientist/differential_expression.py +147 -0
  138. package/src/ai_genomics_scientist/dispatch.py +267 -0
  139. package/src/ai_genomics_scientist/evidence.py +45 -0
  140. package/src/ai_genomics_scientist/gene_set_enrichment.py +76 -0
  141. package/src/ai_genomics_scientist/sequence_alignment.py +97 -0
  142. package/src/ai_genomics_scientist/sequence_features.py +111 -0
  143. package/src/ai_genomics_scientist/splice_site_scoring.py +66 -0
  144. package/src/ai_genomics_scientist/validation.py +83 -0
  145. package/src/ai_genomics_scientist/variant_effect.py +147 -0
  146. package/src/ai_genomics_scientist/variant_pathogenicity.py +125 -0
  147. package/src/ai_materials_scientist/__init__.py +0 -0
  148. package/src/ai_materials_scientist/calphad.py +117 -0
  149. package/src/ai_materials_scientist/classical_monte_carlo.py +165 -0
  150. package/src/ai_materials_scientist/crystal_plasticity.py +184 -0
  151. package/src/ai_materials_scientist/dispatch.py +100 -0
  152. package/src/ai_materials_scientist/evidence.py +84 -0
  153. package/src/ai_materials_scientist/fem.py +279 -0
  154. package/src/ai_materials_scientist/kinetic_monte_carlo.py +145 -0
  155. package/src/ai_materials_scientist/molecular_dynamics.py +240 -0
  156. package/src/ai_materials_scientist/phase_field.py +167 -0
  157. package/src/ai_materials_scientist/validation.py +70 -0
  158. package/src/ai_scientist/__init__.py +1 -0
  159. package/src/ai_scientist/completion_gate.py +15 -0
  160. package/src/ai_scientist/data_analysis.py +46 -0
  161. package/src/ai_scientist/evidence_registry.py +99 -0
  162. package/src/ai_scientist/experimental_design.py +20 -0
  163. package/src/ai_scientist/language.py +14 -0
  164. package/src/ai_scientist/latex_renderer.py +41 -0
  165. package/src/ai_scientist/literature_review.py +37 -0
  166. package/src/ai_scientist/manifest.py +87 -0
  167. package/src/ai_scientist/manuscript.py +94 -0
  168. package/src/ai_scientist/mcp_config.py +76 -0
  169. package/src/ai_scientist/mcp_external.py +42 -0
  170. package/src/ai_scientist/mcp_failures.py +23 -0
  171. package/src/ai_scientist/mcp_gateway.py +38 -0
  172. package/src/ai_scientist/mcp_managed.py +180 -0
  173. package/src/ai_scientist/npm_packaging.py +49 -0
  174. package/src/ai_scientist/orchestrator.py +133 -0
  175. package/src/ai_scientist/peer_review.py +60 -0
  176. package/src/ai_scientist/phase_gate.py +74 -0
  177. package/src/ai_scientist/phase_state.py +230 -0
  178. package/src/ai_scientist/presentation.py +56 -0
  179. package/src/ai_scientist/project_config.py +31 -0
  180. package/src/ai_scientist/project_handle.py +74 -0
  181. package/src/ai_scientist/reproducibility.py +20 -0
  182. package/src/ai_scientist/research_planning.py +20 -0
  183. package/src/ai_scientist/skill_invocation.py +21 -0
  184. package/src/ai_scientist/tdd_gate.py +99 -0
  185. package/src/ai_structural_biology_scientist/__init__.py +0 -0
  186. package/src/ai_structural_biology_scientist/contact_map.py +87 -0
  187. package/src/ai_structural_biology_scientist/dispatch.py +269 -0
  188. package/src/ai_structural_biology_scientist/evidence.py +43 -0
  189. package/src/ai_structural_biology_scientist/hydrophobicity.py +101 -0
  190. package/src/ai_structural_biology_scientist/protein_docking_score.py +104 -0
  191. package/src/ai_structural_biology_scientist/secondary_structure.py +95 -0
  192. package/src/ai_structural_biology_scientist/structural_similarity.py +74 -0
  193. package/src/ai_structural_biology_scientist/validation.py +100 -0
@@ -0,0 +1,100 @@
1
+ """Data-definition and provenance manifest with per-field confidence status.
2
+
3
+ Implements DES-AIDS-040 (REQ-AIDS-052): structurally distinguishes
4
+ reproducible file identity (owner/slug, SHA-256, retrieval time) from
5
+ verified semantic metadata (units, definitions, measurement pathway),
6
+ making data-definition uncertainty visible and queryable instead of
7
+ silently promoting an inferred value to "verified".
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from dataclasses import dataclass, field
13
+
14
+ _VALID_STATUSES = frozenset({"verified", "inferred", "reported", "unknown"})
15
+
16
+
17
+ # @id CODE-AIDS-071
18
+ # @implements REQ-AIDS-052
19
+ # @design DES-AIDS-040
20
+ @dataclass(frozen=True)
21
+ class FieldValue:
22
+ """A single semantic field paired with its confidence status.
23
+
24
+ Frozen so a constructed instance's status cannot be mutated in place;
25
+ changing a field's confidence always requires building a brand-new
26
+ ``FieldValue``, which is always an explicit caller action rather than an
27
+ automatic promotion from "inferred"/"unknown" to "verified".
28
+ """
29
+
30
+ value: object
31
+ status: str
32
+ source: str | None = None
33
+
34
+ def __post_init__(self) -> None:
35
+ if self.status not in _VALID_STATUSES:
36
+ raise ValueError(
37
+ f"status must be one of {sorted(_VALID_STATUSES)}, got {self.status!r}."
38
+ )
39
+
40
+
41
+ # @id CODE-AIDS-083
42
+ # @implements REQ-AIDS-052
43
+ # @design DES-AIDS-040
44
+ @dataclass(frozen=True)
45
+ class DataDefinitionManifest:
46
+ """Aggregated data-definition manifest for one ingested dataset."""
47
+
48
+ source: dict
49
+ dataset_scope: dict
50
+ variables: dict[str, dict[str, FieldValue]]
51
+ transformations: tuple = field(default_factory=tuple)
52
+
53
+ def unresolved_fields(self) -> list[tuple[str, FieldValue]]:
54
+ """Return every ``FieldValue`` across the manifest whose status is "unknown".
55
+
56
+ Each entry's first element is a dotted path identifying its location
57
+ (e.g. ``"source.license"`` or ``"variables.value.unit"``).
58
+ """
59
+ return self._fields_with_status("unknown")
60
+
61
+ def inferred_fields(self) -> list[tuple[str, FieldValue]]:
62
+ """Return every ``FieldValue`` across the manifest whose status is "inferred".
63
+
64
+ GitHub #33: surfaces unconfirmed-but-assumed fields as their own
65
+ actionable, listed item, separate from (and never overlapping
66
+ with) :meth:`unresolved_fields`'s "unknown" list.
67
+ """
68
+ return self._fields_with_status("inferred")
69
+
70
+ def _fields_with_status(self, status: str) -> list[tuple[str, FieldValue]]:
71
+ matches: list[tuple[str, FieldValue]] = []
72
+ for key, field_value in self.source.items():
73
+ if isinstance(field_value, FieldValue) and field_value.status == status:
74
+ matches.append((f"source.{key}", field_value))
75
+ for key, field_value in self.dataset_scope.items():
76
+ if isinstance(field_value, FieldValue) and field_value.status == status:
77
+ matches.append((f"dataset_scope.{key}", field_value))
78
+ for variable_name, fields in self.variables.items():
79
+ for field_name, field_value in fields.items():
80
+ if isinstance(field_value, FieldValue) and field_value.status == status:
81
+ matches.append((f"variables.{variable_name}.{field_name}", field_value))
82
+ return matches
83
+
84
+
85
+ # @id CODE-AIDS-072
86
+ # @implements REQ-AIDS-052
87
+ # @design DES-AIDS-040
88
+ def build_manifest(
89
+ source: dict,
90
+ dataset_scope: dict,
91
+ variables: dict[str, dict[str, FieldValue]],
92
+ transformations: tuple = (),
93
+ ) -> DataDefinitionManifest:
94
+ """Construct a ``DataDefinitionManifest`` from its component dicts."""
95
+ return DataDefinitionManifest(
96
+ source=source,
97
+ dataset_scope=dataset_scope,
98
+ variables=variables,
99
+ transformations=tuple(transformations),
100
+ )
@@ -0,0 +1,164 @@
1
+ """Semantic data-quality checks: schema-driven anomaly detection and
2
+ independent-dataset overlap validation.
3
+
4
+ Implements DES-AIDS-043 (REQ-AIDS-055). This module is deliberately
5
+ separate from ``anomaly_detection`` (which performs statistical
6
+ z-score outlier detection on a single numeric column); here, anomalies
7
+ are semantic constraint violations defined by a declarative schema
8
+ (ranges, allowed categories, non-null, uniqueness), and overlap
9
+ validation cross-checks summary statistics between a primary dataset
10
+ and an independent reference.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from dataclasses import dataclass
16
+
17
+ import pandas as pd
18
+
19
+
20
+ # @id CODE-AIDS-076
21
+ # @implements REQ-AIDS-055
22
+ # @design DES-AIDS-043
23
+ @dataclass(frozen=True)
24
+ class AnomalyRecord:
25
+ """A single schema-constraint violation."""
26
+
27
+ column: str
28
+ rule: str
29
+ row_count: int
30
+ message: str
31
+ row_indices: tuple[int, ...] = ()
32
+
33
+
34
+ # @id CODE-AIDS-077
35
+ # @implements REQ-AIDS-055
36
+ # @design DES-AIDS-043
37
+ def detect_anomalies(df: pd.DataFrame, schema: dict) -> tuple[AnomalyRecord, ...]:
38
+ """Detect semantic constraint violations declared by ``schema``.
39
+
40
+ ``schema`` maps column name -> a dict of constraints, any of:
41
+ - ``"min"`` / ``"max"``: numeric range bounds (inclusive).
42
+ - ``"allowed"``: iterable of permitted categorical values.
43
+ - ``"not_null"``: bool, require no missing values.
44
+ - ``"unique"``: bool, require no duplicate values.
45
+ Unknown columns in ``schema`` that are absent from ``df`` are skipped.
46
+ """
47
+ records: list[AnomalyRecord] = []
48
+ for column, rules in schema.items():
49
+ if column not in df.columns:
50
+ continue
51
+ series = df[column]
52
+
53
+ if rules.get("not_null"):
54
+ missing = series.isna()
55
+ if missing.any():
56
+ records.append(
57
+ AnomalyRecord(
58
+ column=column,
59
+ rule="not_null",
60
+ row_count=int(missing.sum()),
61
+ message=f"Column {column!r} has {int(missing.sum())} null value(s).",
62
+ row_indices=tuple(series.index[missing]),
63
+ )
64
+ )
65
+
66
+ if "min" in rules or "max" in rules:
67
+ numeric = pd.to_numeric(series, errors="coerce")
68
+ below = (
69
+ numeric < rules["min"] if "min" in rules else pd.Series(False, index=series.index)
70
+ )
71
+ above = (
72
+ numeric > rules["max"] if "max" in rules else pd.Series(False, index=series.index)
73
+ )
74
+ out_of_range = (below | above) & numeric.notna()
75
+ if out_of_range.any():
76
+ records.append(
77
+ AnomalyRecord(
78
+ column=column,
79
+ rule="range",
80
+ row_count=int(out_of_range.sum()),
81
+ message=(
82
+ f"Column {column!r} has {int(out_of_range.sum())} value(s) "
83
+ f"outside [{rules.get('min')}, {rules.get('max')}]."
84
+ ),
85
+ row_indices=tuple(series.index[out_of_range]),
86
+ )
87
+ )
88
+
89
+ if "allowed" in rules:
90
+ allowed = set(rules["allowed"])
91
+ disallowed = ~series.isin(allowed) & series.notna()
92
+ if disallowed.any():
93
+ records.append(
94
+ AnomalyRecord(
95
+ column=column,
96
+ rule="allowed_values",
97
+ row_count=int(disallowed.sum()),
98
+ message=(
99
+ f"Column {column!r} has {int(disallowed.sum())} value(s) "
100
+ f"not in the allowed set."
101
+ ),
102
+ row_indices=tuple(series.index[disallowed]),
103
+ )
104
+ )
105
+
106
+ if rules.get("unique"):
107
+ duplicated = series.duplicated(keep=False) & series.notna()
108
+ if duplicated.any():
109
+ records.append(
110
+ AnomalyRecord(
111
+ column=column,
112
+ rule="unique",
113
+ row_count=int(duplicated.sum()),
114
+ message=f"Column {column!r} has {int(duplicated.sum())} duplicate value(s).",
115
+ row_indices=tuple(series.index[duplicated]),
116
+ )
117
+ )
118
+
119
+ return tuple(records)
120
+
121
+
122
+ # @id CODE-AIDS-078
123
+ # @implements REQ-AIDS-055
124
+ # @design DES-AIDS-043
125
+ @dataclass(frozen=True)
126
+ class OverlapMismatch:
127
+ """A single column whose summary statistic disagrees between datasets."""
128
+
129
+ column: str
130
+ primary_value: float
131
+ reference_value: float
132
+ relative_difference: float
133
+
134
+
135
+ def validate_anomalies(
136
+ primary: pd.DataFrame,
137
+ reference: pd.DataFrame,
138
+ columns: list[str],
139
+ tolerance: float = 0.1,
140
+ ) -> tuple[OverlapMismatch, ...]:
141
+ """Compare per-column means between an independent ``reference`` dataset
142
+ and ``primary``, flagging columns whose relative difference exceeds
143
+ ``tolerance``. Used to validate that detected anomalies are not an
144
+ artifact of a single dataset.
145
+ """
146
+ mismatches: list[OverlapMismatch] = []
147
+ for column in columns:
148
+ if column not in primary.columns or column not in reference.columns:
149
+ continue
150
+ primary_mean = float(pd.to_numeric(primary[column], errors="coerce").mean())
151
+ reference_mean = float(pd.to_numeric(reference[column], errors="coerce").mean())
152
+ if reference_mean == 0:
153
+ continue
154
+ relative_difference = abs(primary_mean - reference_mean) / abs(reference_mean)
155
+ if relative_difference > tolerance:
156
+ mismatches.append(
157
+ OverlapMismatch(
158
+ column=column,
159
+ primary_value=primary_mean,
160
+ reference_value=reference_mean,
161
+ relative_difference=relative_difference,
162
+ )
163
+ )
164
+ return tuple(mismatches)
@@ -0,0 +1,135 @@
1
+ """Independent-dataset overlap comparison (narrowed scope).
2
+
3
+ Implements DES-AIDS-045 (REQ-AIDS-057). Per the approved design, this
4
+ module operates only on already-loaded dataframes: ``compare_datasets``
5
+ checks key overlap and per-column value agreement between a primary
6
+ dataset and a candidate independent dataset. Automated dataset
7
+ *discovery* (e.g. searching Kaggle or other catalogs) is explicitly
8
+ out of scope for this repository and is not implemented here; callers
9
+ are expected to load the candidate dataset themselves before calling
10
+ this function.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from dataclasses import dataclass, field
16
+
17
+ import pandas as pd
18
+
19
+ _VALID_RELATIONSHIPS = frozenset({"unknown", "independent", "derived", "overlapping"})
20
+
21
+
22
+ # @id CODE-AIDS-081
23
+ # @implements REQ-AIDS-057
24
+ # @design DES-AIDS-045
25
+ @dataclass(frozen=True)
26
+ class ColumnComparison:
27
+ """Agreement statistics for one mapped column pair."""
28
+
29
+ primary_column: str
30
+ candidate_column: str
31
+ matched_rows: int
32
+ mismatched_rows: int
33
+ agreement_rate: float | None
34
+
35
+
36
+ @dataclass(frozen=True)
37
+ class DatasetComparisonReport:
38
+ """Outcome of comparing a primary dataset against a candidate dataset."""
39
+
40
+ candidate_relationship: str
41
+ matched_keys: int
42
+ primary_only_keys: int
43
+ candidate_only_keys: int
44
+ column_comparisons: tuple[ColumnComparison, ...] = field(default_factory=tuple)
45
+
46
+ def __post_init__(self) -> None:
47
+ if self.candidate_relationship not in _VALID_RELATIONSHIPS:
48
+ raise ValueError(
49
+ f"candidate_relationship must be one of {sorted(_VALID_RELATIONSHIPS)}, "
50
+ f"got {self.candidate_relationship!r}."
51
+ )
52
+
53
+
54
+ # @id CODE-AIDS-082
55
+ # @implements REQ-AIDS-057 REQ-AIDS-062
56
+ # @design DES-AIDS-045 DES-AIDS-050
57
+ def compare_datasets(
58
+ primary: pd.DataFrame,
59
+ candidate: pd.DataFrame,
60
+ key_mapping: dict[str, str],
61
+ value_mapping: dict[str, str],
62
+ candidate_relationship: str = "unknown",
63
+ ) -> DatasetComparisonReport:
64
+ """Compare ``primary`` against an independent ``candidate`` dataset.
65
+
66
+ ``key_mapping`` maps primary key column name(s) -> candidate column
67
+ name(s), used to join the two datasets. ``value_mapping`` maps
68
+ primary value column name -> candidate value column name for
69
+ per-column agreement checks on the joined rows.
70
+
71
+ Rows whose key is null on either side are excluded from
72
+ ``matched_keys``/``primary_only_keys``/``candidate_only_keys`` and from
73
+ the per-column agreement computation (REQ-AIDS-062): otherwise two
74
+ null keys (e.g. both ``None``) would incorrectly count as a matched key
75
+ pair under Python tuple-set equality.
76
+ """
77
+ primary_keys = list(key_mapping.keys())
78
+ candidate_keys = list(key_mapping.values())
79
+
80
+ eligible_primary = primary.dropna(subset=primary_keys)
81
+ eligible_candidate = candidate.dropna(subset=candidate_keys)
82
+
83
+ primary_key_values = set(
84
+ map(tuple, eligible_primary[primary_keys].itertuples(index=False, name=None))
85
+ )
86
+ candidate_key_values = set(
87
+ map(tuple, eligible_candidate[candidate_keys].itertuples(index=False, name=None))
88
+ )
89
+
90
+ matched_keys = primary_key_values & candidate_key_values
91
+ primary_only_keys = primary_key_values - candidate_key_values
92
+ candidate_only_keys = candidate_key_values - primary_key_values
93
+
94
+ merged = eligible_primary.merge(
95
+ eligible_candidate,
96
+ left_on=primary_keys,
97
+ right_on=candidate_keys,
98
+ how="inner",
99
+ suffixes=("_primary", "_candidate"),
100
+ )
101
+
102
+ column_comparisons: list[ColumnComparison] = []
103
+ for primary_column, candidate_column in value_mapping.items():
104
+ primary_col_name = (
105
+ f"{primary_column}_primary" if primary_column in candidate.columns else primary_column
106
+ )
107
+ candidate_col_name = (
108
+ f"{candidate_column}_candidate"
109
+ if candidate_column in primary.columns
110
+ else candidate_column
111
+ )
112
+ if primary_col_name not in merged.columns or candidate_col_name not in merged.columns:
113
+ continue
114
+ matches = merged[primary_col_name] == merged[candidate_col_name]
115
+ matched_rows = int(matches.sum())
116
+ mismatched_rows = int((~matches).sum())
117
+ total = matched_rows + mismatched_rows
118
+ agreement_rate = (matched_rows / total) if total else None
119
+ column_comparisons.append(
120
+ ColumnComparison(
121
+ primary_column=primary_column,
122
+ candidate_column=candidate_column,
123
+ matched_rows=matched_rows,
124
+ mismatched_rows=mismatched_rows,
125
+ agreement_rate=agreement_rate,
126
+ )
127
+ )
128
+
129
+ return DatasetComparisonReport(
130
+ candidate_relationship=candidate_relationship,
131
+ matched_keys=len(matched_keys),
132
+ primary_only_keys=len(primary_only_keys),
133
+ candidate_only_keys=len(candidate_only_keys),
134
+ column_comparisons=tuple(column_comparisons),
135
+ )
@@ -0,0 +1,60 @@
1
+ """Read declared dependency version specifiers from pyproject.toml.
2
+
3
+ Implements DES-AIDS-038 (REQ-AIDS-050): a minimal, dependency-free reader of
4
+ pyproject.toml's [project].dependencies array, used to assert the
5
+ jupyter-mcp-server / mcp version pins stay explicitly bounded on both sides
6
+ (no silent drift to an untested release whose negotiated MCP protocol
7
+ version is incompatible with the rest of the pinned stack).
8
+
9
+ Deliberately avoids a TOML-parsing library: stdlib ``tomllib`` only ships
10
+ from Python 3.11, and this project still supports 3.10, so the
11
+ ``dependencies = [...]`` array is scanned as plain text instead.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import re
17
+ from pathlib import Path
18
+
19
+ _PYPROJECT_PATH = Path(__file__).resolve().parents[2] / "pyproject.toml"
20
+ _DEPENDENCY_LINE_PATTERN = re.compile(r'^\s*"([A-Za-z0-9_.-]+)([^"]*)"\s*,?\s*$')
21
+
22
+
23
+ def _normalize(name: str) -> str:
24
+ """Normalize a package name per PEP 503 (case/separator-insensitive)."""
25
+ return re.sub(r"[-_.]+", "-", name).lower()
26
+
27
+
28
+ # @id CODE-AIDS-059
29
+ # @implements REQ-AIDS-050
30
+ # @design DES-AIDS-038
31
+ def get_dependency_specifier(name: str, pyproject_path: Path = _PYPROJECT_PATH) -> str:
32
+ """Return the raw dependency requirement string declared for ``name``.
33
+
34
+ Scans ``pyproject_path``'s ``[project].dependencies`` array for the
35
+ single entry whose package name matches ``name`` (case/separator
36
+ insensitive), returning it verbatim including its version specifier
37
+ (e.g. ``"jupyter-mcp-server>=2.2,<2.3"``).
38
+
39
+ Raises ``KeyError`` if no dependency named ``name`` is declared.
40
+ """
41
+ target = _normalize(name)
42
+ in_dependencies = False
43
+ for line in pyproject_path.read_text(encoding="utf-8").splitlines():
44
+ stripped = line.strip()
45
+ if not in_dependencies:
46
+ if stripped.startswith("dependencies"):
47
+ in_dependencies = True
48
+ continue
49
+ if stripped.startswith("]"):
50
+ break
51
+ match = _DEPENDENCY_LINE_PATTERN.match(line)
52
+ if not match:
53
+ continue
54
+ raw_name = match.group(1)
55
+ if _normalize(raw_name) == target:
56
+ return stripped.rstrip(",").strip('"')
57
+
58
+ raise KeyError(
59
+ f"No dependency named {name!r} declared in {pyproject_path}'s [project.dependencies]."
60
+ )
@@ -0,0 +1,82 @@
1
+ """Exploratory data analysis.
2
+
3
+ Implements DES-AIDS-007 (REQ-AIDS-005): summary statistics, dtypes, and
4
+ missing-value counts matching pandas describe()/info() reference values.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from dataclasses import dataclass
10
+
11
+ import pandas as pd
12
+
13
+
14
+ @dataclass(frozen=True)
15
+ class EDAReport:
16
+ describe: dict
17
+ dtypes: dict
18
+ non_null_counts: dict
19
+ missing_summary: dict
20
+ categorical_summary: dict
21
+
22
+
23
+ DEFAULT_TOP_N = 10
24
+
25
+
26
+ # @id CODE-AIDS-005
27
+ # @implements REQ-AIDS-005
28
+ # @design DES-AIDS-007
29
+ # @id CODE-AIDS-051
30
+ # @implements REQ-AIDS-043
31
+ # @design DES-AIDS-031
32
+ def explore(df: pd.DataFrame, top_n: int = DEFAULT_TOP_N) -> EDAReport:
33
+ """Compute summary statistics, dtypes and non-null counts for ``df``.
34
+
35
+ Also reports, per column, a missing-value count/ratio, and, for
36
+ categorical (object/category/bool) columns, a unique-value count and
37
+ the ``top_n`` most frequent values with their counts and ratios
38
+ (``truncated`` is set when more unique values exist than ``top_n``).
39
+ The existing ``describe``/``dtypes``/``non_null_counts`` fields are
40
+ unchanged by this extension.
41
+ """
42
+ describe = df.describe().to_dict() if len(df.columns) > 0 else {}
43
+ dtypes = {column: str(dtype) for column, dtype in df.dtypes.items()}
44
+ non_null_counts = df.count().to_dict()
45
+
46
+ row_count = len(df)
47
+ missing_summary = {}
48
+ for column in df.columns:
49
+ missing_count = int(df[column].isna().sum())
50
+ missing_ratio = (missing_count / row_count) if row_count else 0.0
51
+ missing_summary[column] = {
52
+ "missing_count": missing_count,
53
+ "missing_ratio": missing_ratio,
54
+ }
55
+
56
+ categorical_summary = {}
57
+ categorical_columns = df.select_dtypes(include=["object", "str", "category", "bool"]).columns
58
+ for column in categorical_columns:
59
+ value_counts = df[column].value_counts(dropna=True)
60
+ unique_count = int(value_counts.shape[0])
61
+ non_null_total = int(value_counts.sum())
62
+ top_values = [
63
+ {
64
+ "value": value,
65
+ "count": int(count),
66
+ "ratio": (count / non_null_total) if non_null_total else 0.0,
67
+ }
68
+ for value, count in value_counts.head(top_n).items()
69
+ ]
70
+ categorical_summary[column] = {
71
+ "unique_count": unique_count,
72
+ "top_values": top_values,
73
+ "truncated": unique_count > top_n,
74
+ }
75
+
76
+ return EDAReport(
77
+ describe=describe,
78
+ dtypes=dtypes,
79
+ non_null_counts=non_null_counts,
80
+ missing_summary=missing_summary,
81
+ categorical_summary=categorical_summary,
82
+ )