jupytermind 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (193) hide show
  1. package/.github/skills/ai-chemistry-scientist/SKILL.md +97 -0
  2. package/.github/skills/ai-chemistry-scientist/manifest.json +156 -0
  3. package/.github/skills/ai-data-scientist/SKILL.md +330 -0
  4. package/.github/skills/ai-genomics-scientist/SKILL.md +98 -0
  5. package/.github/skills/ai-genomics-scientist/manifest.json +93 -0
  6. package/.github/skills/ai-materials-scientist/SKILL.md +51 -0
  7. package/.github/skills/ai-materials-scientist/manifest.json +58 -0
  8. package/.github/skills/ai-scientist/SKILL.md +69 -0
  9. package/.github/skills/ai-scientist/manifest.json +61 -0
  10. package/.github/skills/ai-structural-biology-scientist/SKILL.md +67 -0
  11. package/.github/skills/ai-structural-biology-scientist/manifest.json +72 -0
  12. package/.github/skills/japanese-prose/NOTICE.md +17 -0
  13. package/.github/skills/japanese-prose/SKILL.md +111 -0
  14. package/.github/skills/japanese-prose/references/review-workflow.md +50 -0
  15. package/.github/skills/japanese-prose/references/scoring.md +24 -0
  16. package/.github/skills/japanese-prose/references/writing-guidelines.md +60 -0
  17. package/.github/skills/japanese-prose/scripts/core.py +192 -0
  18. package/.github/skills/japanese-prose/scripts/fixtures/natural.md +5 -0
  19. package/.github/skills/japanese-prose/scripts/fixtures/unnatural.md +5 -0
  20. package/.github/skills/japanese-prose/scripts/lint.py +378 -0
  21. package/.github/skills/japanese-prose/scripts/outline.py +68 -0
  22. package/.github/skills/japanese-prose/scripts/terms.py +112 -0
  23. package/.github/skills/japanese-prose/scripts/test_engine.py +117 -0
  24. package/.github/skills/presentation-planner/SKILL.md +257 -0
  25. package/.github/skills/presentation-planner/assets/design-templates/data-report.yaml +97 -0
  26. package/.github/skills/presentation-planner/assets/design-templates/executive-proposal.yaml +92 -0
  27. package/.github/skills/presentation-planner/assets/design-templates/technical-briefing.yaml +96 -0
  28. package/.github/skills/presentation-planner/assets/scenario-templates/data-report.md +47 -0
  29. package/.github/skills/presentation-planner/assets/scenario-templates/executive-decision.md +43 -0
  30. package/.github/skills/presentation-planner/assets/scenario-templates/technical-briefing.md +45 -0
  31. package/.github/skills/presentation-planner/references/customizing-design-templates.md +160 -0
  32. package/.github/skills/presentation-planner/references/design-spec-schema.md +72 -0
  33. package/.github/skills/presentation-planner/references/handoff-contract.md +49 -0
  34. package/.github/skills/presentation-planner/references/responsibility-boundary.md +32 -0
  35. package/.github/skills/presentation-planner/references/scenario-templates.md +55 -0
  36. package/.github/skills/tech-writer/SKILL.md +434 -0
  37. package/.github/skills/tech-writer/assets/templates/blueprint.md +187 -0
  38. package/.github/skills/tech-writer/assets/templates/design-doc.md +29 -0
  39. package/.github/skills/tech-writer/assets/templates/migration-plan.md +173 -0
  40. package/.github/skills/tech-writer/assets/templates/operations-runbook.md +202 -0
  41. package/.github/skills/tech-writer/assets/templates/pr-description.md +23 -0
  42. package/.github/skills/tech-writer/assets/templates/qiita.md +44 -0
  43. package/.github/skills/tech-writer/assets/templates/readme.md +38 -0
  44. package/.github/skills/tech-writer/assets/templates/requirements-definition.md +170 -0
  45. package/.github/skills/tech-writer/assets/templates/rfi.md +113 -0
  46. package/.github/skills/tech-writer/assets/templates/rfp.md +180 -0
  47. package/.github/skills/tech-writer/assets/templates/security-design.md +167 -0
  48. package/.github/skills/tech-writer/assets/templates/system-design.md +220 -0
  49. package/.github/skills/tech-writer/assets/templates/technical-proposal.md +112 -0
  50. package/.github/skills/tech-writer/assets/templates/test-plan.md +153 -0
  51. package/.github/skills/tech-writer/assets/templates/user-manual.md +22 -0
  52. package/.github/skills/tech-writer/assets/templates/white-paper.md +192 -0
  53. package/.github/skills/tech-writer/references/doctypes/api-docs.md +33 -0
  54. package/.github/skills/tech-writer/references/doctypes/blueprint.md +81 -0
  55. package/.github/skills/tech-writer/references/doctypes/code-comments.md +39 -0
  56. package/.github/skills/tech-writer/references/doctypes/design-doc.md +42 -0
  57. package/.github/skills/tech-writer/references/doctypes/migration-plan.md +63 -0
  58. package/.github/skills/tech-writer/references/doctypes/operations-runbook.md +63 -0
  59. package/.github/skills/tech-writer/references/doctypes/pr-commit.md +82 -0
  60. package/.github/skills/tech-writer/references/doctypes/qiita.md +75 -0
  61. package/.github/skills/tech-writer/references/doctypes/readme.md +43 -0
  62. package/.github/skills/tech-writer/references/doctypes/release-notes.md +30 -0
  63. package/.github/skills/tech-writer/references/doctypes/requirements-definition.md +61 -0
  64. package/.github/skills/tech-writer/references/doctypes/rfi.md +43 -0
  65. package/.github/skills/tech-writer/references/doctypes/rfp.md +46 -0
  66. package/.github/skills/tech-writer/references/doctypes/security-design.md +71 -0
  67. package/.github/skills/tech-writer/references/doctypes/system-design.md +74 -0
  68. package/.github/skills/tech-writer/references/doctypes/technical-proposal.md +49 -0
  69. package/.github/skills/tech-writer/references/doctypes/test-plan.md +67 -0
  70. package/.github/skills/tech-writer/references/doctypes/user-manual.md +58 -0
  71. package/.github/skills/tech-writer/references/doctypes/white-paper.md +84 -0
  72. package/.github/skills/tech-writer/references/doctypes/zenn.md +66 -0
  73. package/.github/skills/tech-writer/references/japanese-prose-optimization.md +110 -0
  74. package/.github/skills/tech-writer/references/style-constitution.md +104 -0
  75. package/.github/skills/tech-writer/scripts/lint.py +412 -0
  76. package/LICENSE +21 -0
  77. package/README.md +92 -0
  78. package/bin/ai-data-scientist.js +123 -0
  79. package/package.json +41 -0
  80. package/pyproject.toml +45 -0
  81. package/src/ai_chemistry_scientist/__init__.py +0 -0
  82. package/src/ai_chemistry_scientist/admet_prediction.py +71 -0
  83. package/src/ai_chemistry_scientist/bioactivity_classification.py +73 -0
  84. package/src/ai_chemistry_scientist/data/sample_molecules.csv +21 -0
  85. package/src/ai_chemistry_scientist/dispatch.py +369 -0
  86. package/src/ai_chemistry_scientist/docking_score.py +97 -0
  87. package/src/ai_chemistry_scientist/drug_likeness_rules.py +84 -0
  88. package/src/ai_chemistry_scientist/evidence.py +41 -0
  89. package/src/ai_chemistry_scientist/molecular_descriptors.py +97 -0
  90. package/src/ai_chemistry_scientist/molecular_formula_mass.py +40 -0
  91. package/src/ai_chemistry_scientist/molecular_similarity.py +78 -0
  92. package/src/ai_chemistry_scientist/qsar_modeling.py +105 -0
  93. package/src/ai_chemistry_scientist/salt_standardization.py +81 -0
  94. package/src/ai_chemistry_scientist/structural_alerts.py +76 -0
  95. package/src/ai_chemistry_scientist/structure_format_conversion.py +84 -0
  96. package/src/ai_chemistry_scientist/validation.py +70 -0
  97. package/src/ai_data_scientist/__init__.py +0 -0
  98. package/src/ai_data_scientist/analysis_assumptions.py +121 -0
  99. package/src/ai_data_scientist/anomaly_detection.py +39 -0
  100. package/src/ai_data_scientist/automl.py +109 -0
  101. package/src/ai_data_scientist/cleaning.py +56 -0
  102. package/src/ai_data_scientist/cli.py +90 -0
  103. package/src/ai_data_scientist/clustering.py +54 -0
  104. package/src/ai_data_scientist/dashboard.py +33 -0
  105. package/src/ai_data_scientist/data_definition.py +100 -0
  106. package/src/ai_data_scientist/data_quality.py +164 -0
  107. package/src/ai_data_scientist/dataset_validation.py +135 -0
  108. package/src/ai_data_scientist/dependency_pins.py +60 -0
  109. package/src/ai_data_scientist/eda.py +82 -0
  110. package/src/ai_data_scientist/experiment_evaluation.py +635 -0
  111. package/src/ai_data_scientist/explainability.py +340 -0
  112. package/src/ai_data_scientist/feature_engineering.py +163 -0
  113. package/src/ai_data_scientist/gate_config.py +32 -0
  114. package/src/ai_data_scientist/ingestion.py +127 -0
  115. package/src/ai_data_scientist/insight_engine.py +180 -0
  116. package/src/ai_data_scientist/japanese_nlp.py +43 -0
  117. package/src/ai_data_scientist/jupyter_launcher.py +137 -0
  118. package/src/ai_data_scientist/jupyter_mcp_client.py +94 -0
  119. package/src/ai_data_scientist/language_router.py +28 -0
  120. package/src/ai_data_scientist/lifecycle.py +221 -0
  121. package/src/ai_data_scientist/mcp_gateway.py +113 -0
  122. package/src/ai_data_scientist/mcp_runtime.py +194 -0
  123. package/src/ai_data_scientist/mcp_transport.py +53 -0
  124. package/src/ai_data_scientist/ml_modeling.py +451 -0
  125. package/src/ai_data_scientist/model_tuning.py +104 -0
  126. package/src/ai_data_scientist/notebook_audit.py +574 -0
  127. package/src/ai_data_scientist/project_manager.py +243 -0
  128. package/src/ai_data_scientist/report_export.py +73 -0
  129. package/src/ai_data_scientist/sensitivity.py +445 -0
  130. package/src/ai_data_scientist/signal_analysis.py +201 -0
  131. package/src/ai_data_scientist/skill_packaging.py +40 -0
  132. package/src/ai_data_scientist/stats_analysis.py +88 -0
  133. package/src/ai_data_scientist/text_nlp.py +44 -0
  134. package/src/ai_data_scientist/timeseries.py +68 -0
  135. package/src/ai_data_scientist/visualization.py +708 -0
  136. package/src/ai_genomics_scientist/__init__.py +1 -0
  137. package/src/ai_genomics_scientist/differential_expression.py +147 -0
  138. package/src/ai_genomics_scientist/dispatch.py +267 -0
  139. package/src/ai_genomics_scientist/evidence.py +45 -0
  140. package/src/ai_genomics_scientist/gene_set_enrichment.py +76 -0
  141. package/src/ai_genomics_scientist/sequence_alignment.py +97 -0
  142. package/src/ai_genomics_scientist/sequence_features.py +111 -0
  143. package/src/ai_genomics_scientist/splice_site_scoring.py +66 -0
  144. package/src/ai_genomics_scientist/validation.py +83 -0
  145. package/src/ai_genomics_scientist/variant_effect.py +147 -0
  146. package/src/ai_genomics_scientist/variant_pathogenicity.py +125 -0
  147. package/src/ai_materials_scientist/__init__.py +0 -0
  148. package/src/ai_materials_scientist/calphad.py +117 -0
  149. package/src/ai_materials_scientist/classical_monte_carlo.py +165 -0
  150. package/src/ai_materials_scientist/crystal_plasticity.py +184 -0
  151. package/src/ai_materials_scientist/dispatch.py +100 -0
  152. package/src/ai_materials_scientist/evidence.py +84 -0
  153. package/src/ai_materials_scientist/fem.py +279 -0
  154. package/src/ai_materials_scientist/kinetic_monte_carlo.py +145 -0
  155. package/src/ai_materials_scientist/molecular_dynamics.py +240 -0
  156. package/src/ai_materials_scientist/phase_field.py +167 -0
  157. package/src/ai_materials_scientist/validation.py +70 -0
  158. package/src/ai_scientist/__init__.py +1 -0
  159. package/src/ai_scientist/completion_gate.py +15 -0
  160. package/src/ai_scientist/data_analysis.py +46 -0
  161. package/src/ai_scientist/evidence_registry.py +99 -0
  162. package/src/ai_scientist/experimental_design.py +20 -0
  163. package/src/ai_scientist/language.py +14 -0
  164. package/src/ai_scientist/latex_renderer.py +41 -0
  165. package/src/ai_scientist/literature_review.py +37 -0
  166. package/src/ai_scientist/manifest.py +87 -0
  167. package/src/ai_scientist/manuscript.py +94 -0
  168. package/src/ai_scientist/mcp_config.py +76 -0
  169. package/src/ai_scientist/mcp_external.py +42 -0
  170. package/src/ai_scientist/mcp_failures.py +23 -0
  171. package/src/ai_scientist/mcp_gateway.py +38 -0
  172. package/src/ai_scientist/mcp_managed.py +180 -0
  173. package/src/ai_scientist/npm_packaging.py +49 -0
  174. package/src/ai_scientist/orchestrator.py +133 -0
  175. package/src/ai_scientist/peer_review.py +60 -0
  176. package/src/ai_scientist/phase_gate.py +74 -0
  177. package/src/ai_scientist/phase_state.py +230 -0
  178. package/src/ai_scientist/presentation.py +56 -0
  179. package/src/ai_scientist/project_config.py +31 -0
  180. package/src/ai_scientist/project_handle.py +74 -0
  181. package/src/ai_scientist/reproducibility.py +20 -0
  182. package/src/ai_scientist/research_planning.py +20 -0
  183. package/src/ai_scientist/skill_invocation.py +21 -0
  184. package/src/ai_scientist/tdd_gate.py +99 -0
  185. package/src/ai_structural_biology_scientist/__init__.py +0 -0
  186. package/src/ai_structural_biology_scientist/contact_map.py +87 -0
  187. package/src/ai_structural_biology_scientist/dispatch.py +269 -0
  188. package/src/ai_structural_biology_scientist/evidence.py +43 -0
  189. package/src/ai_structural_biology_scientist/hydrophobicity.py +101 -0
  190. package/src/ai_structural_biology_scientist/protein_docking_score.py +104 -0
  191. package/src/ai_structural_biology_scientist/secondary_structure.py +95 -0
  192. package/src/ai_structural_biology_scientist/structural_similarity.py +74 -0
  193. package/src/ai_structural_biology_scientist/validation.py +100 -0
@@ -0,0 +1,111 @@
1
+ ---
2
+ name: japanese-prose
3
+ description: >-
4
+ Improves Japanese prose in technical and business documents using
5
+ kotonoha's original GiNZA-based diagnostics. Use when writing, rewriting,
6
+ reviewing, or scoring Japanese prose; when the user asks for more natural,
7
+ readable, concise, or less AI-like Japanese; or when tech-writer delegates
8
+ its sentence-level quality pass. Owns wording, sentence rhythm, reading
9
+ load, terminology review, and formulaic-expression detection. Does not own
10
+ technical-document structure or presentation storylines.
11
+ license: MIT
12
+ argument-hint: "[write|review|score] [quick|full] <target file or request>"
13
+ ---
14
+
15
+ # japanese-prose
16
+
17
+ Improves Japanese wording without changing approved facts, obligations,
18
+ identifiers, evidence, code, or document structure. This is an original
19
+ kotonoha implementation. It uses
20
+ [GiNZA](https://github.com/megagonlabs/ginza) for tokenization, part-of-speech
21
+ tagging, dependency parsing, lemmatization, and named-entity recognition.
22
+
23
+ ## Responsibility boundary
24
+
25
+ - `tech-writer` owns doctype selection, section structure, completeness,
26
+ traceability, and Markdown organization.
27
+ - `presentation-planner` owns audience strategy, scenarios, slide order, and
28
+ design specifications.
29
+ - `japanese-prose` owns sentence-level Japanese: clarity, rhythm, reading
30
+ load, terminology, repeated patterns, and contextual rewriting.
31
+
32
+ Never move, add, or remove sections unless the calling skill explicitly
33
+ authorizes it. Never alter requirement IDs, risk IDs, control IDs, numbers,
34
+ units, dates, proper nouns, citations, URLs, commands, tables, schemas,
35
+ acceptance criteria, approval states, or normative force.
36
+
37
+ ## Modes
38
+
39
+ - `write`: write or rewrite Japanese prose. Use `quick` unless the document
40
+ is external-facing, high-risk, or longer than roughly 10,000 Japanese
41
+ characters.
42
+ - `review`: report problems and proposed corrections without editing.
43
+ - `score`: run diagnostics and report the 0–100 score without rewriting.
44
+ - `quick`: run prose lint once, review findings in context, edit, and rerun
45
+ until no new actionable findings appear.
46
+ - `full`: run prose lint, reading-load lint, outline extraction, and
47
+ terminology extraction; then review the whole document against
48
+ `references/review-workflow.md`.
49
+
50
+ ## Workflow
51
+
52
+ 1. Read the target, audience, intended outcome, and calling skill's frozen
53
+ invariants.
54
+ 2. Read `references/writing-guidelines.md` before generating or rewriting
55
+ prose.
56
+ 3. Run the GiNZA prose diagnostic:
57
+
58
+ ```bash
59
+ uv run scripts/lint.py <target-file> --genre tech --json > <workdir>/prose-baseline.json
60
+ ```
61
+
62
+ 4. For `full` mode, also run:
63
+
64
+ ```bash
65
+ uv run scripts/lint.py <target-file> --genre tech --reading-load --json
66
+ uv run scripts/outline.py <target-file>
67
+ uv run scripts/terms.py <target-file> --json
68
+ ```
69
+
70
+ 5. Classify each finding as `fix` or `keep`. A detector identifies a review
71
+ target; it does not authorize blind replacement.
72
+ 6. Apply only fixes that improve the intended reader's understanding while
73
+ preserving the frozen invariants.
74
+ 7. Rerun the prose diagnostic with
75
+ `--baseline <workdir>/prose-baseline.json` to classify new, persisting,
76
+ and resolved findings. Replace the baseline only after recording the
77
+ decisions for the current round.
78
+ 8. Stop when every finding has a decision, no new actionable finding appears,
79
+ and the document passes the checks in `references/review-workflow.md`.
80
+
81
+ Allow at most three edit-and-diagnose rounds per invocation. If actionable
82
+ findings or invariant violations remain, report that the optimization did not
83
+ converge and list the unresolved items.
84
+
85
+ ## Markdown emphasis
86
+
87
+ When strong emphasis touches surrounding prose, put half-width spaces outside
88
+ the delimiters: `これは **重要** です`, not `これは**重要**です`. Spaces are
89
+ unnecessary at line boundaries or next to punctuation, and must not be placed
90
+ inside `**`. Use ASCII spaces (`U+0020`), never full-width spaces (`U+3000`),
91
+ tabs, or non-breaking spaces, immediately before and after emphasis embedded
92
+ in prose. Write `これは **「重要」** と説明する`, not
93
+ `これは **「重要」** と説明する`.
94
+
95
+ ## Diagnostic interpretation
96
+
97
+ The score is a triage aid, not a quality certificate. Read
98
+ `references/scoring.md` before presenting it. GiNZA provides linguistic
99
+ observations; the agent remains responsible for deciding whether a change is
100
+ correct in context.
101
+
102
+ ## Completion report
103
+
104
+ Report:
105
+
106
+ - mode and diagnostics executed
107
+ - initial and final scores
108
+ - findings fixed and findings deliberately kept
109
+ - any skipped diagnostic and its reason
110
+ - invariant verification result
111
+ - `completed` or `did not converge`
@@ -0,0 +1,50 @@
1
+ # Japanese prose review workflow
2
+
3
+ ## Review findings in context
4
+
5
+ For every diagnostic finding, record one decision:
6
+
7
+ - `fix`: the expression increases ambiguity, reading load, or mechanical
8
+ repetition for the intended reader
9
+ - `keep`: the expression is required by accuracy, domain convention, quoted
10
+ material, or deliberate rhythm
11
+
12
+ Do not count a finding as resolved merely because the triggering text
13
+ disappeared. Confirm that the replacement preserves meaning.
14
+
15
+ ## Check sentence structure
16
+
17
+ - The subject and predicate can be identified without rereading.
18
+ - Modifiers sit close to the words they modify.
19
+ - A long sentence contains one main relationship.
20
+ - Negation does not require the reader to reverse the meaning twice.
21
+ - Noun chains expose ownership, purpose, and target relationships.
22
+
23
+ ## Check document rhythm
24
+
25
+ - Consecutive sentences do not begin with the same two lemmas without reason.
26
+ - Paragraphs do not all begin with the same connective.
27
+ - Sentence lengths vary with information weight.
28
+ - Nominal endings are not used as the default ending for every sentence.
29
+
30
+ ## Check terminology
31
+
32
+ - Product names, identifiers, values, and citations are unchanged.
33
+ - An unfamiliar term is explained near its first use.
34
+ - One concept uses one preferred term unless a distinction is intentional.
35
+ - An acronym is expanded when the audience cannot be expected to know it.
36
+
37
+ ## Check the whole document
38
+
39
+ Read only headings and paragraph openings. The argument must still be
40
+ predictable. Then read the finished prose continuously and verify that local
41
+ rewrites did not damage transitions or create repeated explanations.
42
+
43
+ ## Convergence
44
+
45
+ The review is complete when:
46
+
47
+ - every finding has a `fix` or `keep` decision
48
+ - rerunning diagnostics produces no new actionable finding
49
+ - frozen invariants match the pre-edit document
50
+ - the reader can reach the requested outcome without an unstated inference
@@ -0,0 +1,24 @@
1
+ # Diagnostic scoring
2
+
3
+ The prose lint starts at 100 and subtracts:
4
+
5
+ - 10 points for each `critical` finding
6
+ - 4 points for each `warning` finding
7
+ - 1 point for each `info` finding
8
+
9
+ The minimum is 0. Scores compare repeated runs of the same document and
10
+ configuration; they do not compare authors, genres, or unrelated documents.
11
+
12
+ Use these bands only as triage:
13
+
14
+ | Score | Interpretation |
15
+ |---|---|
16
+ | 90–100 | Few mechanically detectable review targets |
17
+ | 75–89 | Several passages need contextual review |
18
+ | 50–74 | Reading load or repeated patterns are widespread |
19
+ | 0–49 | Review the document section by section before publication |
20
+
21
+ A high score does not prove that facts are correct, the argument is complete,
22
+ or the prose is natural. A low score does not require every finding to be
23
+ changed. Always report the finding categories and contextual decisions beside
24
+ the score.
@@ -0,0 +1,60 @@
1
+ # Japanese prose writing guidelines
2
+
3
+ Use these rules while drafting or rewriting. They are constraints for
4
+ judgment, not search-and-replace recipes.
5
+
6
+ ## Lead with the reader's required conclusion
7
+
8
+ State the decision, action, result, or answer before the background that
9
+ supports it. An introductory sentence must earn its place by changing what
10
+ the reader understands or does.
11
+
12
+ ## Keep one main relationship per sentence
13
+
14
+ A sentence may contain qualifications, but its subject, predicate, and main
15
+ object must remain visible. Split a sentence when the reader must retain one
16
+ unfinished relationship while parsing another.
17
+
18
+ ## Prefer explicit relationships over compressed noun chains
19
+
20
+ Long noun sequences hide whether terms express ownership, purpose, target, or
21
+ sequence. Add particles or a predicate when GiNZA identifies five or more
22
+ consecutive nouns.
23
+
24
+ ## Use concrete actors and actions
25
+
26
+ Name who checks, changes, approves, records, or observes something. Avoid
27
+ turning actions into abstract nouns when a direct verb is clearer.
28
+
29
+ ## Treat formulaic expressions as review targets
30
+
31
+ Expressions such as 「と言えるでしょう」 or 「以下のとおりです」 are not
32
+ automatically wrong. Keep them only when their rhetorical function is
33
+ necessary. Otherwise state the evidence, conclusion, or preview directly.
34
+
35
+ ## Vary rhythm according to information weight
36
+
37
+ Do not force every sentence or paragraph to the same length. Use a short
38
+ sentence for a decisive result. Use a longer sentence only when its
39
+ relationships remain clear.
40
+
41
+ ## Explain terminology at first use
42
+
43
+ Introduce an unfamiliar term by function before or beside its name. Preserve
44
+ established product names, protocol names, identifiers, and domain terms.
45
+
46
+ ## Preserve uncertainty accurately
47
+
48
+ Keep distinctions among confirmed facts, estimates, assumptions, and
49
+ recommendations. Do not make prose smoother by increasing certainty.
50
+
51
+ ## Use emphasis sparingly and safely
52
+
53
+ Emphasize only the phrase that changes the reader's decision. In Markdown,
54
+ separate `**strong emphasis**` from adjacent prose with half-width spaces.
55
+
56
+ ## End with the consequence
57
+
58
+ Close a section with what the evidence means for the reader's next decision
59
+ or action. Do not append a generic summary when the consequence is already
60
+ clear.
@@ -0,0 +1,192 @@
1
+ """Shared Markdown and GiNZA utilities for kotonoha's Japanese prose tools."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import re
7
+ from dataclasses import asdict, dataclass
8
+ from functools import lru_cache
9
+ from pathlib import Path
10
+ from typing import Iterable
11
+
12
+
13
+ FENCE_RE = re.compile(r"^\s{0,3}(`{3,}|~{3,})")
14
+ HEADING_RE = re.compile(r"^\s{0,3}#{1,6}\s+")
15
+ LIST_RE = re.compile(r"^\s*(?:[-+*]|\d+[.)])\s+")
16
+ BLOCKQUOTE_RE = re.compile(r"^\s*>\s?")
17
+ INLINE_CODE_RE = re.compile(r"`[^`\n]*`")
18
+ LINK_RE = re.compile(r"!?\[([^\]]*)\]\([^)]+\)")
19
+ URL_RE = re.compile(r"https?://\S+")
20
+ HTML_TAG_RE = re.compile(r"<[^>]+>")
21
+ SENTENCE_END_RE = re.compile(r"[。!?!?]+(?:[」』)】〉》]*)")
22
+ JAPANESE_RE = re.compile(r"[ぁ-んァ-ヶ一-龠々]")
23
+
24
+
25
+ @dataclass
26
+ class Finding:
27
+ line: int
28
+ category: str
29
+ severity: str
30
+ message: str
31
+ excerpt: str
32
+ evidence: dict | None = None
33
+ status: str | None = None
34
+
35
+ def to_dict(self) -> dict:
36
+ value = asdict(self)
37
+ return {key: item for key, item in value.items() if item is not None}
38
+
39
+
40
+ @dataclass
41
+ class ProseBlock:
42
+ text: str
43
+ start_line: int
44
+
45
+
46
+ def read_text(path: Path) -> str:
47
+ if not path.is_file():
48
+ raise FileNotFoundError(f"file not found: {path}")
49
+ return path.read_text(encoding="utf-8")
50
+
51
+
52
+ def strip_markdown_inline(text: str) -> str:
53
+ text = INLINE_CODE_RE.sub(" コード ", text)
54
+ text = LINK_RE.sub(lambda match: match.group(1), text)
55
+ text = URL_RE.sub(" ", text)
56
+ text = HTML_TAG_RE.sub(" ", text)
57
+ text = text.replace("**", "").replace("__", "")
58
+ text = text.replace("~~", "")
59
+ return re.sub(r"[ \t]+", " ", text).strip()
60
+
61
+
62
+ def iter_content_lines(text: str) -> Iterable[tuple[int, str]]:
63
+ """Yield line-numbered Markdown outside metadata, code, and comments."""
64
+ lines = text.splitlines()
65
+ fence_marker = ""
66
+ in_frontmatter = bool(lines and lines[0].strip() == "---")
67
+ in_comment = False
68
+
69
+ for index, raw in enumerate(lines, start=1):
70
+ stripped = raw.strip()
71
+ if in_frontmatter:
72
+ if index > 1 and stripped == "---":
73
+ in_frontmatter = False
74
+ yield index, ""
75
+ continue
76
+
77
+ fence = FENCE_RE.match(raw)
78
+ if fence:
79
+ marker = fence.group(1)[0]
80
+ if not fence_marker:
81
+ fence_marker = marker
82
+ elif marker == fence_marker:
83
+ fence_marker = ""
84
+ yield index, ""
85
+ continue
86
+ if fence_marker:
87
+ yield index, ""
88
+ continue
89
+
90
+ if "<!--" in raw:
91
+ in_comment = True
92
+ if in_comment:
93
+ if "-->" in raw:
94
+ in_comment = False
95
+ yield index, ""
96
+ continue
97
+ yield index, raw
98
+
99
+
100
+ def iter_prose_blocks(text: str) -> list[ProseBlock]:
101
+ """Return Markdown prose blocks while excluding metadata and code."""
102
+ blocks: list[ProseBlock] = []
103
+ buffer: list[str] = []
104
+ buffer_line = 0
105
+
106
+ def flush() -> None:
107
+ nonlocal buffer, buffer_line
108
+ if buffer:
109
+ cleaned = strip_markdown_inline("\n".join(buffer))
110
+ if cleaned and JAPANESE_RE.search(cleaned):
111
+ blocks.append(ProseBlock(cleaned, buffer_line))
112
+ buffer = []
113
+ buffer_line = 0
114
+
115
+ for index, raw in iter_content_lines(text):
116
+ stripped = raw.strip()
117
+ if not stripped or stripped.startswith("|") or HEADING_RE.match(raw):
118
+ flush()
119
+ continue
120
+
121
+ list_item = LIST_RE.match(raw)
122
+ if list_item:
123
+ flush()
124
+ line = LIST_RE.sub("", raw)
125
+ line = BLOCKQUOTE_RE.sub("", line)
126
+ if not buffer:
127
+ buffer_line = index
128
+ buffer.append(line)
129
+
130
+ flush()
131
+ return blocks
132
+
133
+
134
+ def sentence_line(block: ProseBlock, start_char: int) -> int:
135
+ return block.start_line + block.text[:start_char].count("\n")
136
+
137
+
138
+ def split_sentences_fallback(text: str) -> Iterable[tuple[str, int]]:
139
+ start = 0
140
+ for match in SENTENCE_END_RE.finditer(text):
141
+ end = match.end()
142
+ sentence = text[start:end].strip()
143
+ if sentence:
144
+ yield sentence, start
145
+ start = end
146
+ tail = text[start:].strip()
147
+ if tail:
148
+ yield tail, start
149
+
150
+
151
+ @lru_cache(maxsize=2)
152
+ def load_ginza(enable_ner: bool = False):
153
+ import spacy
154
+
155
+ disabled = [] if enable_ner else ["ner"]
156
+ return spacy.load("ja_ginza", disable=disabled)
157
+
158
+
159
+ def content_tokens(span) -> list:
160
+ return [
161
+ token
162
+ for token in span
163
+ if not token.is_space and not token.is_punct and token.pos_ != "SYM"
164
+ ]
165
+
166
+
167
+ def dependency_depth(token) -> int:
168
+ depth = 0
169
+ current = token
170
+ seen: set[int] = set()
171
+ while current.head.i != current.i and current.i not in seen and depth < 40:
172
+ seen.add(current.i)
173
+ current = current.head
174
+ depth += 1
175
+ return depth
176
+
177
+
178
+ def excerpt(text: str, limit: int = 100) -> str:
179
+ value = re.sub(r"\s+", " ", text).strip()
180
+ return value if len(value) <= limit else f"{value[:limit - 1]}…"
181
+
182
+
183
+ def finding_key(finding: Finding) -> str:
184
+ normalized = re.sub(r"\d+", "#", finding.excerpt.lower())
185
+ raw = f"{finding.category}\0{normalized}".encode()
186
+ return hashlib.sha256(raw).hexdigest()[:20]
187
+
188
+
189
+ def score_findings(findings: list[Finding]) -> int:
190
+ weights = {"info": 1, "warning": 4, "critical": 10}
191
+ deduction = sum(weights.get(item.severity, 4) for item in findings)
192
+ return max(0, 100 - deduction)
@@ -0,0 +1,5 @@
1
+ # 日本語診断の確認
2
+
3
+ この機能は、設定ミスを公開前に見つけます。管理者が確認画面を開き、対象ユーザーの権限と申請内容を照合してください。
4
+
5
+ 権限が一致すれば承認できます。一致しない場合は、申請者へ差し戻します。
@@ -0,0 +1,5 @@
1
+ # 日本語診断の確認
2
+
3
+ 結論として、この機能を利用することが可能です。また、この機能は非常に重要なポイントです。また、設定を確認することが可能です。また、結果を確認することが可能です。
4
+
5
+ 利用者権限設定変更承認処理実行結果確認画面を利用する場合においては、設定が正しくないことがないとは言えないため、管理者が確認しなくてはならないということが可能であると言えるでしょう。