jupytermind 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/skills/ai-chemistry-scientist/SKILL.md +97 -0
- package/.github/skills/ai-chemistry-scientist/manifest.json +156 -0
- package/.github/skills/ai-data-scientist/SKILL.md +330 -0
- package/.github/skills/ai-genomics-scientist/SKILL.md +98 -0
- package/.github/skills/ai-genomics-scientist/manifest.json +93 -0
- package/.github/skills/ai-materials-scientist/SKILL.md +51 -0
- package/.github/skills/ai-materials-scientist/manifest.json +58 -0
- package/.github/skills/ai-scientist/SKILL.md +69 -0
- package/.github/skills/ai-scientist/manifest.json +61 -0
- package/.github/skills/ai-structural-biology-scientist/SKILL.md +67 -0
- package/.github/skills/ai-structural-biology-scientist/manifest.json +72 -0
- package/.github/skills/japanese-prose/NOTICE.md +17 -0
- package/.github/skills/japanese-prose/SKILL.md +111 -0
- package/.github/skills/japanese-prose/references/review-workflow.md +50 -0
- package/.github/skills/japanese-prose/references/scoring.md +24 -0
- package/.github/skills/japanese-prose/references/writing-guidelines.md +60 -0
- package/.github/skills/japanese-prose/scripts/core.py +192 -0
- package/.github/skills/japanese-prose/scripts/fixtures/natural.md +5 -0
- package/.github/skills/japanese-prose/scripts/fixtures/unnatural.md +5 -0
- package/.github/skills/japanese-prose/scripts/lint.py +378 -0
- package/.github/skills/japanese-prose/scripts/outline.py +68 -0
- package/.github/skills/japanese-prose/scripts/terms.py +112 -0
- package/.github/skills/japanese-prose/scripts/test_engine.py +117 -0
- package/.github/skills/presentation-planner/SKILL.md +257 -0
- package/.github/skills/presentation-planner/assets/design-templates/data-report.yaml +97 -0
- package/.github/skills/presentation-planner/assets/design-templates/executive-proposal.yaml +92 -0
- package/.github/skills/presentation-planner/assets/design-templates/technical-briefing.yaml +96 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/data-report.md +47 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/executive-decision.md +43 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/technical-briefing.md +45 -0
- package/.github/skills/presentation-planner/references/customizing-design-templates.md +160 -0
- package/.github/skills/presentation-planner/references/design-spec-schema.md +72 -0
- package/.github/skills/presentation-planner/references/handoff-contract.md +49 -0
- package/.github/skills/presentation-planner/references/responsibility-boundary.md +32 -0
- package/.github/skills/presentation-planner/references/scenario-templates.md +55 -0
- package/.github/skills/tech-writer/SKILL.md +434 -0
- package/.github/skills/tech-writer/assets/templates/blueprint.md +187 -0
- package/.github/skills/tech-writer/assets/templates/design-doc.md +29 -0
- package/.github/skills/tech-writer/assets/templates/migration-plan.md +173 -0
- package/.github/skills/tech-writer/assets/templates/operations-runbook.md +202 -0
- package/.github/skills/tech-writer/assets/templates/pr-description.md +23 -0
- package/.github/skills/tech-writer/assets/templates/qiita.md +44 -0
- package/.github/skills/tech-writer/assets/templates/readme.md +38 -0
- package/.github/skills/tech-writer/assets/templates/requirements-definition.md +170 -0
- package/.github/skills/tech-writer/assets/templates/rfi.md +113 -0
- package/.github/skills/tech-writer/assets/templates/rfp.md +180 -0
- package/.github/skills/tech-writer/assets/templates/security-design.md +167 -0
- package/.github/skills/tech-writer/assets/templates/system-design.md +220 -0
- package/.github/skills/tech-writer/assets/templates/technical-proposal.md +112 -0
- package/.github/skills/tech-writer/assets/templates/test-plan.md +153 -0
- package/.github/skills/tech-writer/assets/templates/user-manual.md +22 -0
- package/.github/skills/tech-writer/assets/templates/white-paper.md +192 -0
- package/.github/skills/tech-writer/references/doctypes/api-docs.md +33 -0
- package/.github/skills/tech-writer/references/doctypes/blueprint.md +81 -0
- package/.github/skills/tech-writer/references/doctypes/code-comments.md +39 -0
- package/.github/skills/tech-writer/references/doctypes/design-doc.md +42 -0
- package/.github/skills/tech-writer/references/doctypes/migration-plan.md +63 -0
- package/.github/skills/tech-writer/references/doctypes/operations-runbook.md +63 -0
- package/.github/skills/tech-writer/references/doctypes/pr-commit.md +82 -0
- package/.github/skills/tech-writer/references/doctypes/qiita.md +75 -0
- package/.github/skills/tech-writer/references/doctypes/readme.md +43 -0
- package/.github/skills/tech-writer/references/doctypes/release-notes.md +30 -0
- package/.github/skills/tech-writer/references/doctypes/requirements-definition.md +61 -0
- package/.github/skills/tech-writer/references/doctypes/rfi.md +43 -0
- package/.github/skills/tech-writer/references/doctypes/rfp.md +46 -0
- package/.github/skills/tech-writer/references/doctypes/security-design.md +71 -0
- package/.github/skills/tech-writer/references/doctypes/system-design.md +74 -0
- package/.github/skills/tech-writer/references/doctypes/technical-proposal.md +49 -0
- package/.github/skills/tech-writer/references/doctypes/test-plan.md +67 -0
- package/.github/skills/tech-writer/references/doctypes/user-manual.md +58 -0
- package/.github/skills/tech-writer/references/doctypes/white-paper.md +84 -0
- package/.github/skills/tech-writer/references/doctypes/zenn.md +66 -0
- package/.github/skills/tech-writer/references/japanese-prose-optimization.md +110 -0
- package/.github/skills/tech-writer/references/style-constitution.md +104 -0
- package/.github/skills/tech-writer/scripts/lint.py +412 -0
- package/LICENSE +21 -0
- package/README.md +92 -0
- package/bin/ai-data-scientist.js +123 -0
- package/package.json +41 -0
- package/pyproject.toml +45 -0
- package/src/ai_chemistry_scientist/__init__.py +0 -0
- package/src/ai_chemistry_scientist/admet_prediction.py +71 -0
- package/src/ai_chemistry_scientist/bioactivity_classification.py +73 -0
- package/src/ai_chemistry_scientist/data/sample_molecules.csv +21 -0
- package/src/ai_chemistry_scientist/dispatch.py +369 -0
- package/src/ai_chemistry_scientist/docking_score.py +97 -0
- package/src/ai_chemistry_scientist/drug_likeness_rules.py +84 -0
- package/src/ai_chemistry_scientist/evidence.py +41 -0
- package/src/ai_chemistry_scientist/molecular_descriptors.py +97 -0
- package/src/ai_chemistry_scientist/molecular_formula_mass.py +40 -0
- package/src/ai_chemistry_scientist/molecular_similarity.py +78 -0
- package/src/ai_chemistry_scientist/qsar_modeling.py +105 -0
- package/src/ai_chemistry_scientist/salt_standardization.py +81 -0
- package/src/ai_chemistry_scientist/structural_alerts.py +76 -0
- package/src/ai_chemistry_scientist/structure_format_conversion.py +84 -0
- package/src/ai_chemistry_scientist/validation.py +70 -0
- package/src/ai_data_scientist/__init__.py +0 -0
- package/src/ai_data_scientist/analysis_assumptions.py +121 -0
- package/src/ai_data_scientist/anomaly_detection.py +39 -0
- package/src/ai_data_scientist/automl.py +109 -0
- package/src/ai_data_scientist/cleaning.py +56 -0
- package/src/ai_data_scientist/cli.py +90 -0
- package/src/ai_data_scientist/clustering.py +54 -0
- package/src/ai_data_scientist/dashboard.py +33 -0
- package/src/ai_data_scientist/data_definition.py +100 -0
- package/src/ai_data_scientist/data_quality.py +164 -0
- package/src/ai_data_scientist/dataset_validation.py +135 -0
- package/src/ai_data_scientist/dependency_pins.py +60 -0
- package/src/ai_data_scientist/eda.py +82 -0
- package/src/ai_data_scientist/experiment_evaluation.py +635 -0
- package/src/ai_data_scientist/explainability.py +340 -0
- package/src/ai_data_scientist/feature_engineering.py +163 -0
- package/src/ai_data_scientist/gate_config.py +32 -0
- package/src/ai_data_scientist/ingestion.py +127 -0
- package/src/ai_data_scientist/insight_engine.py +180 -0
- package/src/ai_data_scientist/japanese_nlp.py +43 -0
- package/src/ai_data_scientist/jupyter_launcher.py +137 -0
- package/src/ai_data_scientist/jupyter_mcp_client.py +94 -0
- package/src/ai_data_scientist/language_router.py +28 -0
- package/src/ai_data_scientist/lifecycle.py +221 -0
- package/src/ai_data_scientist/mcp_gateway.py +113 -0
- package/src/ai_data_scientist/mcp_runtime.py +194 -0
- package/src/ai_data_scientist/mcp_transport.py +53 -0
- package/src/ai_data_scientist/ml_modeling.py +451 -0
- package/src/ai_data_scientist/model_tuning.py +104 -0
- package/src/ai_data_scientist/notebook_audit.py +574 -0
- package/src/ai_data_scientist/project_manager.py +243 -0
- package/src/ai_data_scientist/report_export.py +73 -0
- package/src/ai_data_scientist/sensitivity.py +445 -0
- package/src/ai_data_scientist/signal_analysis.py +201 -0
- package/src/ai_data_scientist/skill_packaging.py +40 -0
- package/src/ai_data_scientist/stats_analysis.py +88 -0
- package/src/ai_data_scientist/text_nlp.py +44 -0
- package/src/ai_data_scientist/timeseries.py +68 -0
- package/src/ai_data_scientist/visualization.py +708 -0
- package/src/ai_genomics_scientist/__init__.py +1 -0
- package/src/ai_genomics_scientist/differential_expression.py +147 -0
- package/src/ai_genomics_scientist/dispatch.py +267 -0
- package/src/ai_genomics_scientist/evidence.py +45 -0
- package/src/ai_genomics_scientist/gene_set_enrichment.py +76 -0
- package/src/ai_genomics_scientist/sequence_alignment.py +97 -0
- package/src/ai_genomics_scientist/sequence_features.py +111 -0
- package/src/ai_genomics_scientist/splice_site_scoring.py +66 -0
- package/src/ai_genomics_scientist/validation.py +83 -0
- package/src/ai_genomics_scientist/variant_effect.py +147 -0
- package/src/ai_genomics_scientist/variant_pathogenicity.py +125 -0
- package/src/ai_materials_scientist/__init__.py +0 -0
- package/src/ai_materials_scientist/calphad.py +117 -0
- package/src/ai_materials_scientist/classical_monte_carlo.py +165 -0
- package/src/ai_materials_scientist/crystal_plasticity.py +184 -0
- package/src/ai_materials_scientist/dispatch.py +100 -0
- package/src/ai_materials_scientist/evidence.py +84 -0
- package/src/ai_materials_scientist/fem.py +279 -0
- package/src/ai_materials_scientist/kinetic_monte_carlo.py +145 -0
- package/src/ai_materials_scientist/molecular_dynamics.py +240 -0
- package/src/ai_materials_scientist/phase_field.py +167 -0
- package/src/ai_materials_scientist/validation.py +70 -0
- package/src/ai_scientist/__init__.py +1 -0
- package/src/ai_scientist/completion_gate.py +15 -0
- package/src/ai_scientist/data_analysis.py +46 -0
- package/src/ai_scientist/evidence_registry.py +99 -0
- package/src/ai_scientist/experimental_design.py +20 -0
- package/src/ai_scientist/language.py +14 -0
- package/src/ai_scientist/latex_renderer.py +41 -0
- package/src/ai_scientist/literature_review.py +37 -0
- package/src/ai_scientist/manifest.py +87 -0
- package/src/ai_scientist/manuscript.py +94 -0
- package/src/ai_scientist/mcp_config.py +76 -0
- package/src/ai_scientist/mcp_external.py +42 -0
- package/src/ai_scientist/mcp_failures.py +23 -0
- package/src/ai_scientist/mcp_gateway.py +38 -0
- package/src/ai_scientist/mcp_managed.py +180 -0
- package/src/ai_scientist/npm_packaging.py +49 -0
- package/src/ai_scientist/orchestrator.py +133 -0
- package/src/ai_scientist/peer_review.py +60 -0
- package/src/ai_scientist/phase_gate.py +74 -0
- package/src/ai_scientist/phase_state.py +230 -0
- package/src/ai_scientist/presentation.py +56 -0
- package/src/ai_scientist/project_config.py +31 -0
- package/src/ai_scientist/project_handle.py +74 -0
- package/src/ai_scientist/reproducibility.py +20 -0
- package/src/ai_scientist/research_planning.py +20 -0
- package/src/ai_scientist/skill_invocation.py +21 -0
- package/src/ai_scientist/tdd_gate.py +99 -0
- package/src/ai_structural_biology_scientist/__init__.py +0 -0
- package/src/ai_structural_biology_scientist/contact_map.py +87 -0
- package/src/ai_structural_biology_scientist/dispatch.py +269 -0
- package/src/ai_structural_biology_scientist/evidence.py +43 -0
- package/src/ai_structural_biology_scientist/hydrophobicity.py +101 -0
- package/src/ai_structural_biology_scientist/protein_docking_score.py +104 -0
- package/src/ai_structural_biology_scientist/secondary_structure.py +95 -0
- package/src/ai_structural_biology_scientist/structural_similarity.py +74 -0
- package/src/ai_structural_biology_scientist/validation.py +100 -0
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
"""Insight & evidence engine.
|
|
2
|
+
|
|
3
|
+
Implements DES-AIDS-010 (REQ-AIDS-009/010/027, ADR-0003): validates that a
|
|
4
|
+
candidate insight is backed by an actually-executed notebook cell whose
|
|
5
|
+
output contains the cited value, embeds a structured evidence manifest in
|
|
6
|
+
the insight markdown cell, and refuses to write anything when evidence is
|
|
7
|
+
missing.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import re
|
|
14
|
+
|
|
15
|
+
import nbformat
|
|
16
|
+
|
|
17
|
+
from ai_data_scientist.project_manager import ProjectHandle, enqueue_write
|
|
18
|
+
|
|
19
|
+
_MANIFEST_FENCE = "```evidence\n{payload}\n```"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class EvidenceMissingError(ValueError):
|
|
23
|
+
"""Raised when no executed cell backs a candidate insight's evidence."""
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class CitedValueNotFoundError(ValueError):
|
|
27
|
+
"""Raised when a cited-value pattern has no match in a result's output."""
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# @id CODE-AIDS-125
|
|
31
|
+
# @implements REQ-AIDS-010
|
|
32
|
+
# @design DES-AIDS-010
|
|
33
|
+
class AmbiguousEvidenceError(EvidenceMissingError):
|
|
34
|
+
"""Raised when more than one executed cell matches the same evidence.
|
|
35
|
+
|
|
36
|
+
GitHub #54: duplicate ``execution_count`` values (common after a kernel
|
|
37
|
+
restart or appending to a notebook in a new session) can make more than
|
|
38
|
+
one code cell match an insight's ``execution_count``/``cited_value``
|
|
39
|
+
pair. Silently resolving to whichever cell is encountered first risks
|
|
40
|
+
attributing an insight to the wrong evidence, so this is raised instead
|
|
41
|
+
(a subclass of ``EvidenceMissingError`` so callers that already treat
|
|
42
|
+
missing evidence as "withhold the insight" handle this the same way).
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
# @id CODE-AIDS-047
|
|
47
|
+
# @implements REQ-AIDS-039
|
|
48
|
+
# @design DES-AIDS-027
|
|
49
|
+
def extract_cited_value(result: dict, pattern: str) -> str:
|
|
50
|
+
"""Extract the exact substring a caller should pass as ``cited_value``.
|
|
51
|
+
|
|
52
|
+
Searches ``result["output"]`` (the dict returned by
|
|
53
|
+
``mcp_gateway.run_and_record``/``execute_cell``) for ``pattern`` and
|
|
54
|
+
returns its first capture group verbatim, or the whole match when
|
|
55
|
+
``pattern`` defines no group. Raises ``CitedValueNotFoundError`` on no
|
|
56
|
+
match instead of returning a guessed or empty value, so callers never
|
|
57
|
+
hand-transcribe (and risk rounding/mistyping) a value for
|
|
58
|
+
``record_insight``.
|
|
59
|
+
"""
|
|
60
|
+
output = str(result.get("output", ""))
|
|
61
|
+
match = re.search(pattern, output)
|
|
62
|
+
if match is None:
|
|
63
|
+
raise CitedValueNotFoundError(
|
|
64
|
+
f"Pattern {pattern!r} did not match the result output; "
|
|
65
|
+
"no cited value could be extracted."
|
|
66
|
+
)
|
|
67
|
+
return match.group(1) if match.lastindex else match.group(0)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _cell_output_contains(cell, cited_value: str) -> bool:
|
|
71
|
+
for output in cell.get("outputs", []):
|
|
72
|
+
for value in output.get("data", {}).values():
|
|
73
|
+
if cited_value in str(value):
|
|
74
|
+
return True
|
|
75
|
+
# GitHub #29: execute_result/display_data outputs store their
|
|
76
|
+
# payload under "data", but print()-produced stream output stores
|
|
77
|
+
# it under "text" instead; a value genuinely printed by the
|
|
78
|
+
# executed cell is equally valid evidence.
|
|
79
|
+
if output.get("output_type") == "stream" and cited_value in str(output.get("text", "")):
|
|
80
|
+
return True
|
|
81
|
+
return False
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _matching_evidence_cells(notebook, execution_count: int, cited_value: str) -> list:
|
|
85
|
+
"""Return every code cell whose execution_count/output matches evidence."""
|
|
86
|
+
return [
|
|
87
|
+
cell
|
|
88
|
+
for cell in notebook.cells
|
|
89
|
+
if cell.get("cell_type") == "code"
|
|
90
|
+
and cell.get("execution_count") == execution_count
|
|
91
|
+
and _cell_output_contains(cell, cited_value)
|
|
92
|
+
]
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _find_evidence_cell(notebook, execution_count: int, cited_value: str):
|
|
96
|
+
"""Return the single code cell matching this evidence, or ``None``.
|
|
97
|
+
|
|
98
|
+
GitHub #54 (CODE-AIDS-125, REQ-AIDS-010): raises
|
|
99
|
+
``AmbiguousEvidenceError`` instead of silently returning the first
|
|
100
|
+
match when more than one code cell shares the same ``execution_count``
|
|
101
|
+
and both produced ``cited_value`` in their output (e.g. after a kernel
|
|
102
|
+
restart or appending to a notebook in a new session re-uses an
|
|
103
|
+
``execution_count``); callers must not guess which cell is the real
|
|
104
|
+
evidentiary basis.
|
|
105
|
+
"""
|
|
106
|
+
matches = _matching_evidence_cells(notebook, execution_count, cited_value)
|
|
107
|
+
if len(matches) > 1:
|
|
108
|
+
raise AmbiguousEvidenceError(
|
|
109
|
+
f"{len(matches)} executed cells share execution_count="
|
|
110
|
+
f"{execution_count!r} and an output containing {cited_value!r}; "
|
|
111
|
+
"the evidentiary cell is ambiguous."
|
|
112
|
+
)
|
|
113
|
+
return matches[0] if matches else None
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
# @id CODE-AIDS-009
|
|
117
|
+
# @implements REQ-AIDS-009
|
|
118
|
+
# @design DES-AIDS-010
|
|
119
|
+
# @id CODE-AIDS-010
|
|
120
|
+
# @implements REQ-AIDS-010
|
|
121
|
+
# @design DES-AIDS-010
|
|
122
|
+
# @id CODE-AIDS-027
|
|
123
|
+
# @implements REQ-AIDS-027
|
|
124
|
+
# @design DES-AIDS-010
|
|
125
|
+
def record_insight(
|
|
126
|
+
handle: ProjectHandle,
|
|
127
|
+
insight_text: str,
|
|
128
|
+
evidence_execution_count: int,
|
|
129
|
+
cited_value: str,
|
|
130
|
+
claim_type: str,
|
|
131
|
+
language: str = "en",
|
|
132
|
+
) -> None:
|
|
133
|
+
"""Append an evidence-backed insight markdown cell, or refuse to.
|
|
134
|
+
|
|
135
|
+
Verifies the executed evidentiary cell exists and actually produced
|
|
136
|
+
``cited_value`` before writing anything; raises ``EvidenceMissingError``
|
|
137
|
+
(REQ-AIDS-010) without touching the notebook otherwise. Raises
|
|
138
|
+
``AmbiguousEvidenceError`` (a subclass of ``EvidenceMissingError``)
|
|
139
|
+
instead of guessing when more than one executed cell matches the same
|
|
140
|
+
``execution_count``/``cited_value`` pair (GitHub #54).
|
|
141
|
+
"""
|
|
142
|
+
notebook = nbformat.read(handle.notebook_path, as_version=4)
|
|
143
|
+
try:
|
|
144
|
+
evidence_cell = _find_evidence_cell(notebook, evidence_execution_count, cited_value)
|
|
145
|
+
except AmbiguousEvidenceError as exc:
|
|
146
|
+
message = (
|
|
147
|
+
f"Insight '{insight_text}' の根拠セルが一意に決まらないため記録を保留しました"
|
|
148
|
+
f"(execution_count={evidence_execution_count}の実行済みセルが複数あり、"
|
|
149
|
+
"いずれも該当する出力を含みます)。"
|
|
150
|
+
if language == "ja"
|
|
151
|
+
else (f"Withheld insight '{insight_text}': evidence is ambiguous ({exc}).")
|
|
152
|
+
)
|
|
153
|
+
raise AmbiguousEvidenceError(message) from exc
|
|
154
|
+
if evidence_cell is None:
|
|
155
|
+
message = (
|
|
156
|
+
f"Insight '{insight_text}' に根拠となる実行済みセル(execution_count="
|
|
157
|
+
f"{evidence_execution_count})が見つからないため記録を保留しました。"
|
|
158
|
+
if language == "ja"
|
|
159
|
+
else (
|
|
160
|
+
f"Could not establish supporting evidence for insight "
|
|
161
|
+
f"'{insight_text}' (no executed cell with execution_count="
|
|
162
|
+
f"{evidence_execution_count} producing '{cited_value}'); withheld."
|
|
163
|
+
)
|
|
164
|
+
)
|
|
165
|
+
raise EvidenceMissingError(message)
|
|
166
|
+
|
|
167
|
+
manifest = json.dumps(
|
|
168
|
+
{
|
|
169
|
+
"execution_count": evidence_execution_count,
|
|
170
|
+
"cited_value": cited_value,
|
|
171
|
+
"claim_type": claim_type,
|
|
172
|
+
},
|
|
173
|
+
separators=(",", ":"),
|
|
174
|
+
)
|
|
175
|
+
source = f"{insight_text}\n\n{_MANIFEST_FENCE.format(payload=manifest)}"
|
|
176
|
+
|
|
177
|
+
def add_cell(nb):
|
|
178
|
+
nb.cells.append(nbformat.v4.new_markdown_cell(source))
|
|
179
|
+
|
|
180
|
+
enqueue_write(handle, add_cell)
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""Japanese NLP via the pinned GiNZA (`ja_ginza`) pipeline.
|
|
2
|
+
|
|
3
|
+
Implements DES-AIDS-021 (REQ-AIDS-023, ADR-0006): runs the GiNZA Japanese
|
|
4
|
+
NLP pipeline on Japanese text to perform the requested operation and
|
|
5
|
+
reports the result.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
|
|
12
|
+
import spacy
|
|
13
|
+
|
|
14
|
+
_SUPPORTED_OPERATIONS = ("tokenize",)
|
|
15
|
+
_nlp = None # module-level lazily-loaded ja_ginza pipeline singleton
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _get_pipeline():
|
|
19
|
+
global _nlp
|
|
20
|
+
if _nlp is None:
|
|
21
|
+
_nlp = spacy.load("ja_ginza")
|
|
22
|
+
return _nlp
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass(frozen=True)
|
|
26
|
+
class JapaneseNLPResult:
|
|
27
|
+
tokens: list
|
|
28
|
+
pos_tags: list
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
# @id CODE-AIDS-023
|
|
32
|
+
# @implements REQ-AIDS-023
|
|
33
|
+
# @design DES-AIDS-021
|
|
34
|
+
def analyze_japanese_text(text: str, operation: str = "tokenize") -> JapaneseNLPResult:
|
|
35
|
+
"""Tokenize and POS-tag ``text`` using the pinned ja_ginza pipeline."""
|
|
36
|
+
if operation not in _SUPPORTED_OPERATIONS:
|
|
37
|
+
raise ValueError(f"Unsupported Japanese NLP operation: {operation!r}")
|
|
38
|
+
|
|
39
|
+
doc = _get_pipeline()(text)
|
|
40
|
+
tokens = [token.text for token in doc]
|
|
41
|
+
pos_tags = [token.pos_ for token in doc]
|
|
42
|
+
|
|
43
|
+
return JapaneseNLPResult(tokens=tokens, pos_tags=pos_tags)
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""Real Jupyter MCP runtime launcher.
|
|
2
|
+
|
|
3
|
+
Implements the production-grade RuntimeLauncher (DES-AIDS-025) backing
|
|
4
|
+
mcp_runtime.ensure_runtime: starts real JupyterLab and jupyter-mcp-server
|
|
5
|
+
processes inside the project's managed Python environment, bound to
|
|
6
|
+
127.0.0.1 with auto-selected free ports and random tokens, and health-checks
|
|
7
|
+
them over real HTTP. Verified interactively against a live jupyter-mcp-server
|
|
8
|
+
2.2.3 instance: JupyterLab requires ``--IdentityProvider.token``, and the
|
|
9
|
+
streamable-http transport requires its own ``--mcp-token`` (distinct from
|
|
10
|
+
JUPYTER_TOKEN) or it refuses to start.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import os
|
|
16
|
+
import secrets
|
|
17
|
+
import signal
|
|
18
|
+
import socket
|
|
19
|
+
import subprocess
|
|
20
|
+
import sys
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
import httpx
|
|
24
|
+
|
|
25
|
+
from ai_data_scientist.mcp_runtime import RuntimeInfo
|
|
26
|
+
|
|
27
|
+
_HEALTH_CHECK_TIMEOUT_S = 2.0
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _pick_free_port() -> int:
|
|
31
|
+
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock:
|
|
32
|
+
sock.bind(("127.0.0.1", 0))
|
|
33
|
+
return sock.getsockname()[1]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _venv_executable(name: str) -> str:
|
|
37
|
+
"""Resolve ``name`` next to the current Python interpreter's venv bin dir."""
|
|
38
|
+
candidate = Path(sys.executable).parent / name
|
|
39
|
+
return str(candidate) if candidate.exists() else name
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
# @id CODE-AIDS-044
|
|
43
|
+
# @implements REQ-AIDS-034, REQ-AIDS-038
|
|
44
|
+
# @design DES-AIDS-025
|
|
45
|
+
class JupyterLabMCPServerLauncher:
|
|
46
|
+
"""Starts real JupyterLab + jupyter-mcp-server subprocesses.
|
|
47
|
+
|
|
48
|
+
Satisfies the mcp_runtime.RuntimeLauncher Protocol with the real process
|
|
49
|
+
and HTTP transport details confirmed by interactive verification against
|
|
50
|
+
jupyter-mcp-server 2.2.3's streamable-http transport.
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
def __init__(self, working_dir: Path | None = None) -> None:
|
|
54
|
+
self._working_dir = working_dir or Path.cwd()
|
|
55
|
+
|
|
56
|
+
def start_jupyter(self) -> tuple[int, int, str]:
|
|
57
|
+
port = _pick_free_port()
|
|
58
|
+
token = secrets.token_urlsafe(32)
|
|
59
|
+
process = subprocess.Popen(
|
|
60
|
+
[
|
|
61
|
+
_venv_executable("jupyter-lab"),
|
|
62
|
+
f"--port={port}",
|
|
63
|
+
f"--IdentityProvider.token={token}",
|
|
64
|
+
"--ip=127.0.0.1",
|
|
65
|
+
"--no-browser",
|
|
66
|
+
f"--ServerApp.root_dir={self._working_dir}",
|
|
67
|
+
],
|
|
68
|
+
stdout=subprocess.DEVNULL,
|
|
69
|
+
stderr=subprocess.DEVNULL,
|
|
70
|
+
)
|
|
71
|
+
return process.pid, port, token
|
|
72
|
+
|
|
73
|
+
def start_mcp_server(self, jupyter_port: int, jupyter_token: str) -> tuple[int, int, str]:
|
|
74
|
+
port = _pick_free_port()
|
|
75
|
+
token = secrets.token_urlsafe(32)
|
|
76
|
+
env = {
|
|
77
|
+
"JUPYTER_URL": f"http://127.0.0.1:{jupyter_port}",
|
|
78
|
+
"JUPYTER_TOKEN": jupyter_token,
|
|
79
|
+
}
|
|
80
|
+
process = subprocess.Popen(
|
|
81
|
+
[
|
|
82
|
+
_venv_executable("jupyter-mcp-server"),
|
|
83
|
+
"start",
|
|
84
|
+
"--transport=streamable-http",
|
|
85
|
+
f"--port={port}",
|
|
86
|
+
"--host=127.0.0.1",
|
|
87
|
+
f"--mcp-token={token}",
|
|
88
|
+
],
|
|
89
|
+
env={**os.environ, **env},
|
|
90
|
+
stdout=subprocess.DEVNULL,
|
|
91
|
+
stderr=subprocess.DEVNULL,
|
|
92
|
+
)
|
|
93
|
+
return process.pid, port, token
|
|
94
|
+
|
|
95
|
+
def is_healthy(self, info: RuntimeInfo) -> bool:
|
|
96
|
+
try:
|
|
97
|
+
response = httpx.post(
|
|
98
|
+
f"http://127.0.0.1:{info.mcp_port}/mcp",
|
|
99
|
+
headers={
|
|
100
|
+
"Authorization": f"Bearer {info.mcp_token}",
|
|
101
|
+
"Accept": "application/json, text/event-stream",
|
|
102
|
+
"Content-Type": "application/json",
|
|
103
|
+
},
|
|
104
|
+
json={
|
|
105
|
+
"jsonrpc": "2.0",
|
|
106
|
+
"id": 0,
|
|
107
|
+
"method": "initialize",
|
|
108
|
+
"params": {
|
|
109
|
+
"protocolVersion": "2024-11-05",
|
|
110
|
+
"capabilities": {},
|
|
111
|
+
"clientInfo": {"name": "ai-data-scientist", "version": "0.1"},
|
|
112
|
+
},
|
|
113
|
+
},
|
|
114
|
+
timeout=_HEALTH_CHECK_TIMEOUT_S,
|
|
115
|
+
)
|
|
116
|
+
return response.status_code == 200
|
|
117
|
+
except httpx.HTTPError:
|
|
118
|
+
return False
|
|
119
|
+
|
|
120
|
+
def terminate(self, pid: int) -> None:
|
|
121
|
+
try:
|
|
122
|
+
os.kill(pid, signal.SIGTERM)
|
|
123
|
+
except ProcessLookupError:
|
|
124
|
+
pass
|
|
125
|
+
|
|
126
|
+
# @id CODE-AIDS-049
|
|
127
|
+
# @implements REQ-AIDS-041
|
|
128
|
+
# @design DES-AIDS-029
|
|
129
|
+
def is_process_alive(self, pid: int) -> bool:
|
|
130
|
+
"""Probe whether ``pid`` is still alive without sending a real signal."""
|
|
131
|
+
try:
|
|
132
|
+
os.kill(pid, 0)
|
|
133
|
+
except ProcessLookupError:
|
|
134
|
+
return False
|
|
135
|
+
except PermissionError:
|
|
136
|
+
return True
|
|
137
|
+
return True
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""Concrete Jupyter MCP client implementation.
|
|
2
|
+
|
|
3
|
+
Implements DES-AIDS-026: satisfies the mcp_gateway.MCPClient contract by
|
|
4
|
+
communicating with the jupyter-mcp-server runtime started by mcp_runtime, so
|
|
5
|
+
run_and_record can execute real code against a live Jupyter kernel without
|
|
6
|
+
any caller-supplied client (REQ-AIDS-038).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Protocol
|
|
13
|
+
|
|
14
|
+
from ai_data_scientist.mcp_gateway import MCPUnavailableError
|
|
15
|
+
from ai_data_scientist.mcp_runtime import (
|
|
16
|
+
DEFAULT_STARTUP_TIMEOUT_MS,
|
|
17
|
+
DEFAULT_STATE_PATH,
|
|
18
|
+
RuntimeInfo,
|
|
19
|
+
ensure_runtime,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class _Transport(Protocol):
|
|
24
|
+
def __call__(self, port: int, token: str, code: str) -> dict: ...
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
# @id CODE-AIDS-042
|
|
28
|
+
# @implements REQ-AIDS-038
|
|
29
|
+
# @design DES-AIDS-026
|
|
30
|
+
class JupyterMCPClient:
|
|
31
|
+
"""MCPClient implementation backed by a running jupyter-mcp-server.
|
|
32
|
+
|
|
33
|
+
``transport`` is injected (rather than hard-coding an HTTP/stdio library
|
|
34
|
+
call) so this class is unit-testable without a real jupyter-mcp-server,
|
|
35
|
+
mirroring the Protocol-based testability used elsewhere in this codebase.
|
|
36
|
+
Transport-level connection failures are reclassified as the same
|
|
37
|
+
MCPUnavailableError already defined by mcp_gateway (DES-AIDS-004), so
|
|
38
|
+
callers never see transport-specific exceptions.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
def __init__(self, runtime_info: RuntimeInfo, transport: _Transport) -> None:
|
|
42
|
+
self._runtime_info = runtime_info
|
|
43
|
+
self._transport = transport
|
|
44
|
+
|
|
45
|
+
def execute(self, code: str) -> dict:
|
|
46
|
+
try:
|
|
47
|
+
return self._transport(self._runtime_info.mcp_port, self._runtime_info.mcp_token, code)
|
|
48
|
+
except OSError as exc:
|
|
49
|
+
raise MCPUnavailableError(
|
|
50
|
+
"Jupyter MCP server connection failed (Jupyter MCPサーバーへの接続に失敗しました)."
|
|
51
|
+
) from exc
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
# @id CODE-AIDS-043
|
|
55
|
+
# @implements REQ-AIDS-038
|
|
56
|
+
# @design DES-AIDS-026
|
|
57
|
+
def default_client(
|
|
58
|
+
launcher,
|
|
59
|
+
transport: _Transport,
|
|
60
|
+
timeout_ms: int = DEFAULT_STARTUP_TIMEOUT_MS,
|
|
61
|
+
state_path: Path = DEFAULT_STATE_PATH,
|
|
62
|
+
) -> JupyterMCPClient:
|
|
63
|
+
"""Ensure a runtime is running and return a client wired to it.
|
|
64
|
+
|
|
65
|
+
This is the factory mcp_gateway.run_and_record uses when no caller
|
|
66
|
+
supplies an explicit MCPClient, fulfilling REQ-AIDS-038's acceptance
|
|
67
|
+
criterion end-to-end.
|
|
68
|
+
"""
|
|
69
|
+
runtime_info = ensure_runtime(launcher, state_path=state_path, timeout_ms=timeout_ms)
|
|
70
|
+
return JupyterMCPClient(runtime_info, transport)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
# @id CODE-AIDS-046
|
|
74
|
+
# @implements REQ-AIDS-038
|
|
75
|
+
# @design DES-AIDS-026
|
|
76
|
+
def real_client(
|
|
77
|
+
timeout_ms: int = DEFAULT_STARTUP_TIMEOUT_MS,
|
|
78
|
+
state_path: Path = DEFAULT_STATE_PATH,
|
|
79
|
+
) -> JupyterMCPClient:
|
|
80
|
+
"""Convenience factory wiring the real JupyterLab/jupyter-mcp-server stack.
|
|
81
|
+
|
|
82
|
+
This is what production callers use in place of ``default_client`` when
|
|
83
|
+
they want the genuine subprocess launcher and streamable-http transport
|
|
84
|
+
rather than a test double.
|
|
85
|
+
"""
|
|
86
|
+
from ai_data_scientist.jupyter_launcher import JupyterLabMCPServerLauncher
|
|
87
|
+
from ai_data_scientist.mcp_transport import execute_code
|
|
88
|
+
|
|
89
|
+
return default_client(
|
|
90
|
+
JupyterLabMCPServerLauncher(),
|
|
91
|
+
execute_code,
|
|
92
|
+
timeout_ms=timeout_ms,
|
|
93
|
+
state_path=state_path,
|
|
94
|
+
)
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""Bilingual (Japanese/English) instruction language detection."""
|
|
2
|
+
|
|
3
|
+
_JAPANESE_RANGES = (
|
|
4
|
+
(0x3040, 0x309F), # Hiragana
|
|
5
|
+
(0x30A0, 0x30FF), # Katakana
|
|
6
|
+
(0x4E00, 0x9FFF), # CJK Unified Ideographs
|
|
7
|
+
(0x3400, 0x4DBF), # CJK Extension A
|
|
8
|
+
(0xFF66, 0xFF9F), # Halfwidth Katakana
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _is_japanese_char(ch: str) -> bool:
|
|
13
|
+
code = ord(ch)
|
|
14
|
+
return any(low <= code <= high for low, high in _JAPANESE_RANGES)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
# @id CODE-AIDS-001
|
|
18
|
+
# @implements REQ-AIDS-001
|
|
19
|
+
# @design DES-AIDS-002
|
|
20
|
+
def detect_language(text: str) -> str:
|
|
21
|
+
"""Detect whether ``text`` is Japanese ("ja") or English ("en").
|
|
22
|
+
|
|
23
|
+
Deterministic, offline heuristic: any Japanese-script character present
|
|
24
|
+
marks the instruction as Japanese; otherwise it is treated as English.
|
|
25
|
+
"""
|
|
26
|
+
if any(_is_japanese_char(ch) for ch in text):
|
|
27
|
+
return "ja"
|
|
28
|
+
return "en"
|