jupytermind 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/skills/ai-chemistry-scientist/SKILL.md +97 -0
- package/.github/skills/ai-chemistry-scientist/manifest.json +156 -0
- package/.github/skills/ai-data-scientist/SKILL.md +330 -0
- package/.github/skills/ai-genomics-scientist/SKILL.md +98 -0
- package/.github/skills/ai-genomics-scientist/manifest.json +93 -0
- package/.github/skills/ai-materials-scientist/SKILL.md +51 -0
- package/.github/skills/ai-materials-scientist/manifest.json +58 -0
- package/.github/skills/ai-scientist/SKILL.md +69 -0
- package/.github/skills/ai-scientist/manifest.json +61 -0
- package/.github/skills/ai-structural-biology-scientist/SKILL.md +67 -0
- package/.github/skills/ai-structural-biology-scientist/manifest.json +72 -0
- package/.github/skills/japanese-prose/NOTICE.md +17 -0
- package/.github/skills/japanese-prose/SKILL.md +111 -0
- package/.github/skills/japanese-prose/references/review-workflow.md +50 -0
- package/.github/skills/japanese-prose/references/scoring.md +24 -0
- package/.github/skills/japanese-prose/references/writing-guidelines.md +60 -0
- package/.github/skills/japanese-prose/scripts/core.py +192 -0
- package/.github/skills/japanese-prose/scripts/fixtures/natural.md +5 -0
- package/.github/skills/japanese-prose/scripts/fixtures/unnatural.md +5 -0
- package/.github/skills/japanese-prose/scripts/lint.py +378 -0
- package/.github/skills/japanese-prose/scripts/outline.py +68 -0
- package/.github/skills/japanese-prose/scripts/terms.py +112 -0
- package/.github/skills/japanese-prose/scripts/test_engine.py +117 -0
- package/.github/skills/presentation-planner/SKILL.md +257 -0
- package/.github/skills/presentation-planner/assets/design-templates/data-report.yaml +97 -0
- package/.github/skills/presentation-planner/assets/design-templates/executive-proposal.yaml +92 -0
- package/.github/skills/presentation-planner/assets/design-templates/technical-briefing.yaml +96 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/data-report.md +47 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/executive-decision.md +43 -0
- package/.github/skills/presentation-planner/assets/scenario-templates/technical-briefing.md +45 -0
- package/.github/skills/presentation-planner/references/customizing-design-templates.md +160 -0
- package/.github/skills/presentation-planner/references/design-spec-schema.md +72 -0
- package/.github/skills/presentation-planner/references/handoff-contract.md +49 -0
- package/.github/skills/presentation-planner/references/responsibility-boundary.md +32 -0
- package/.github/skills/presentation-planner/references/scenario-templates.md +55 -0
- package/.github/skills/tech-writer/SKILL.md +434 -0
- package/.github/skills/tech-writer/assets/templates/blueprint.md +187 -0
- package/.github/skills/tech-writer/assets/templates/design-doc.md +29 -0
- package/.github/skills/tech-writer/assets/templates/migration-plan.md +173 -0
- package/.github/skills/tech-writer/assets/templates/operations-runbook.md +202 -0
- package/.github/skills/tech-writer/assets/templates/pr-description.md +23 -0
- package/.github/skills/tech-writer/assets/templates/qiita.md +44 -0
- package/.github/skills/tech-writer/assets/templates/readme.md +38 -0
- package/.github/skills/tech-writer/assets/templates/requirements-definition.md +170 -0
- package/.github/skills/tech-writer/assets/templates/rfi.md +113 -0
- package/.github/skills/tech-writer/assets/templates/rfp.md +180 -0
- package/.github/skills/tech-writer/assets/templates/security-design.md +167 -0
- package/.github/skills/tech-writer/assets/templates/system-design.md +220 -0
- package/.github/skills/tech-writer/assets/templates/technical-proposal.md +112 -0
- package/.github/skills/tech-writer/assets/templates/test-plan.md +153 -0
- package/.github/skills/tech-writer/assets/templates/user-manual.md +22 -0
- package/.github/skills/tech-writer/assets/templates/white-paper.md +192 -0
- package/.github/skills/tech-writer/references/doctypes/api-docs.md +33 -0
- package/.github/skills/tech-writer/references/doctypes/blueprint.md +81 -0
- package/.github/skills/tech-writer/references/doctypes/code-comments.md +39 -0
- package/.github/skills/tech-writer/references/doctypes/design-doc.md +42 -0
- package/.github/skills/tech-writer/references/doctypes/migration-plan.md +63 -0
- package/.github/skills/tech-writer/references/doctypes/operations-runbook.md +63 -0
- package/.github/skills/tech-writer/references/doctypes/pr-commit.md +82 -0
- package/.github/skills/tech-writer/references/doctypes/qiita.md +75 -0
- package/.github/skills/tech-writer/references/doctypes/readme.md +43 -0
- package/.github/skills/tech-writer/references/doctypes/release-notes.md +30 -0
- package/.github/skills/tech-writer/references/doctypes/requirements-definition.md +61 -0
- package/.github/skills/tech-writer/references/doctypes/rfi.md +43 -0
- package/.github/skills/tech-writer/references/doctypes/rfp.md +46 -0
- package/.github/skills/tech-writer/references/doctypes/security-design.md +71 -0
- package/.github/skills/tech-writer/references/doctypes/system-design.md +74 -0
- package/.github/skills/tech-writer/references/doctypes/technical-proposal.md +49 -0
- package/.github/skills/tech-writer/references/doctypes/test-plan.md +67 -0
- package/.github/skills/tech-writer/references/doctypes/user-manual.md +58 -0
- package/.github/skills/tech-writer/references/doctypes/white-paper.md +84 -0
- package/.github/skills/tech-writer/references/doctypes/zenn.md +66 -0
- package/.github/skills/tech-writer/references/japanese-prose-optimization.md +110 -0
- package/.github/skills/tech-writer/references/style-constitution.md +104 -0
- package/.github/skills/tech-writer/scripts/lint.py +412 -0
- package/LICENSE +21 -0
- package/README.md +92 -0
- package/bin/ai-data-scientist.js +123 -0
- package/package.json +41 -0
- package/pyproject.toml +45 -0
- package/src/ai_chemistry_scientist/__init__.py +0 -0
- package/src/ai_chemistry_scientist/admet_prediction.py +71 -0
- package/src/ai_chemistry_scientist/bioactivity_classification.py +73 -0
- package/src/ai_chemistry_scientist/data/sample_molecules.csv +21 -0
- package/src/ai_chemistry_scientist/dispatch.py +369 -0
- package/src/ai_chemistry_scientist/docking_score.py +97 -0
- package/src/ai_chemistry_scientist/drug_likeness_rules.py +84 -0
- package/src/ai_chemistry_scientist/evidence.py +41 -0
- package/src/ai_chemistry_scientist/molecular_descriptors.py +97 -0
- package/src/ai_chemistry_scientist/molecular_formula_mass.py +40 -0
- package/src/ai_chemistry_scientist/molecular_similarity.py +78 -0
- package/src/ai_chemistry_scientist/qsar_modeling.py +105 -0
- package/src/ai_chemistry_scientist/salt_standardization.py +81 -0
- package/src/ai_chemistry_scientist/structural_alerts.py +76 -0
- package/src/ai_chemistry_scientist/structure_format_conversion.py +84 -0
- package/src/ai_chemistry_scientist/validation.py +70 -0
- package/src/ai_data_scientist/__init__.py +0 -0
- package/src/ai_data_scientist/analysis_assumptions.py +121 -0
- package/src/ai_data_scientist/anomaly_detection.py +39 -0
- package/src/ai_data_scientist/automl.py +109 -0
- package/src/ai_data_scientist/cleaning.py +56 -0
- package/src/ai_data_scientist/cli.py +90 -0
- package/src/ai_data_scientist/clustering.py +54 -0
- package/src/ai_data_scientist/dashboard.py +33 -0
- package/src/ai_data_scientist/data_definition.py +100 -0
- package/src/ai_data_scientist/data_quality.py +164 -0
- package/src/ai_data_scientist/dataset_validation.py +135 -0
- package/src/ai_data_scientist/dependency_pins.py +60 -0
- package/src/ai_data_scientist/eda.py +82 -0
- package/src/ai_data_scientist/experiment_evaluation.py +635 -0
- package/src/ai_data_scientist/explainability.py +340 -0
- package/src/ai_data_scientist/feature_engineering.py +163 -0
- package/src/ai_data_scientist/gate_config.py +32 -0
- package/src/ai_data_scientist/ingestion.py +127 -0
- package/src/ai_data_scientist/insight_engine.py +180 -0
- package/src/ai_data_scientist/japanese_nlp.py +43 -0
- package/src/ai_data_scientist/jupyter_launcher.py +137 -0
- package/src/ai_data_scientist/jupyter_mcp_client.py +94 -0
- package/src/ai_data_scientist/language_router.py +28 -0
- package/src/ai_data_scientist/lifecycle.py +221 -0
- package/src/ai_data_scientist/mcp_gateway.py +113 -0
- package/src/ai_data_scientist/mcp_runtime.py +194 -0
- package/src/ai_data_scientist/mcp_transport.py +53 -0
- package/src/ai_data_scientist/ml_modeling.py +451 -0
- package/src/ai_data_scientist/model_tuning.py +104 -0
- package/src/ai_data_scientist/notebook_audit.py +574 -0
- package/src/ai_data_scientist/project_manager.py +243 -0
- package/src/ai_data_scientist/report_export.py +73 -0
- package/src/ai_data_scientist/sensitivity.py +445 -0
- package/src/ai_data_scientist/signal_analysis.py +201 -0
- package/src/ai_data_scientist/skill_packaging.py +40 -0
- package/src/ai_data_scientist/stats_analysis.py +88 -0
- package/src/ai_data_scientist/text_nlp.py +44 -0
- package/src/ai_data_scientist/timeseries.py +68 -0
- package/src/ai_data_scientist/visualization.py +708 -0
- package/src/ai_genomics_scientist/__init__.py +1 -0
- package/src/ai_genomics_scientist/differential_expression.py +147 -0
- package/src/ai_genomics_scientist/dispatch.py +267 -0
- package/src/ai_genomics_scientist/evidence.py +45 -0
- package/src/ai_genomics_scientist/gene_set_enrichment.py +76 -0
- package/src/ai_genomics_scientist/sequence_alignment.py +97 -0
- package/src/ai_genomics_scientist/sequence_features.py +111 -0
- package/src/ai_genomics_scientist/splice_site_scoring.py +66 -0
- package/src/ai_genomics_scientist/validation.py +83 -0
- package/src/ai_genomics_scientist/variant_effect.py +147 -0
- package/src/ai_genomics_scientist/variant_pathogenicity.py +125 -0
- package/src/ai_materials_scientist/__init__.py +0 -0
- package/src/ai_materials_scientist/calphad.py +117 -0
- package/src/ai_materials_scientist/classical_monte_carlo.py +165 -0
- package/src/ai_materials_scientist/crystal_plasticity.py +184 -0
- package/src/ai_materials_scientist/dispatch.py +100 -0
- package/src/ai_materials_scientist/evidence.py +84 -0
- package/src/ai_materials_scientist/fem.py +279 -0
- package/src/ai_materials_scientist/kinetic_monte_carlo.py +145 -0
- package/src/ai_materials_scientist/molecular_dynamics.py +240 -0
- package/src/ai_materials_scientist/phase_field.py +167 -0
- package/src/ai_materials_scientist/validation.py +70 -0
- package/src/ai_scientist/__init__.py +1 -0
- package/src/ai_scientist/completion_gate.py +15 -0
- package/src/ai_scientist/data_analysis.py +46 -0
- package/src/ai_scientist/evidence_registry.py +99 -0
- package/src/ai_scientist/experimental_design.py +20 -0
- package/src/ai_scientist/language.py +14 -0
- package/src/ai_scientist/latex_renderer.py +41 -0
- package/src/ai_scientist/literature_review.py +37 -0
- package/src/ai_scientist/manifest.py +87 -0
- package/src/ai_scientist/manuscript.py +94 -0
- package/src/ai_scientist/mcp_config.py +76 -0
- package/src/ai_scientist/mcp_external.py +42 -0
- package/src/ai_scientist/mcp_failures.py +23 -0
- package/src/ai_scientist/mcp_gateway.py +38 -0
- package/src/ai_scientist/mcp_managed.py +180 -0
- package/src/ai_scientist/npm_packaging.py +49 -0
- package/src/ai_scientist/orchestrator.py +133 -0
- package/src/ai_scientist/peer_review.py +60 -0
- package/src/ai_scientist/phase_gate.py +74 -0
- package/src/ai_scientist/phase_state.py +230 -0
- package/src/ai_scientist/presentation.py +56 -0
- package/src/ai_scientist/project_config.py +31 -0
- package/src/ai_scientist/project_handle.py +74 -0
- package/src/ai_scientist/reproducibility.py +20 -0
- package/src/ai_scientist/research_planning.py +20 -0
- package/src/ai_scientist/skill_invocation.py +21 -0
- package/src/ai_scientist/tdd_gate.py +99 -0
- package/src/ai_structural_biology_scientist/__init__.py +0 -0
- package/src/ai_structural_biology_scientist/contact_map.py +87 -0
- package/src/ai_structural_biology_scientist/dispatch.py +269 -0
- package/src/ai_structural_biology_scientist/evidence.py +43 -0
- package/src/ai_structural_biology_scientist/hydrophobicity.py +101 -0
- package/src/ai_structural_biology_scientist/protein_docking_score.py +104 -0
- package/src/ai_structural_biology_scientist/secondary_structure.py +95 -0
- package/src/ai_structural_biology_scientist/structural_similarity.py +74 -0
- package/src/ai_structural_biology_scientist/validation.py +100 -0
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: japanese-prose
|
|
3
|
+
description: >-
|
|
4
|
+
Improves Japanese prose in technical and business documents using
|
|
5
|
+
kotonoha's original GiNZA-based diagnostics. Use when writing, rewriting,
|
|
6
|
+
reviewing, or scoring Japanese prose; when the user asks for more natural,
|
|
7
|
+
readable, concise, or less AI-like Japanese; or when tech-writer delegates
|
|
8
|
+
its sentence-level quality pass. Owns wording, sentence rhythm, reading
|
|
9
|
+
load, terminology review, and formulaic-expression detection. Does not own
|
|
10
|
+
technical-document structure or presentation storylines.
|
|
11
|
+
license: MIT
|
|
12
|
+
argument-hint: "[write|review|score] [quick|full] <target file or request>"
|
|
13
|
+
---
|
|
14
|
+
|
|
15
|
+
# japanese-prose
|
|
16
|
+
|
|
17
|
+
Improves Japanese wording without changing approved facts, obligations,
|
|
18
|
+
identifiers, evidence, code, or document structure. This is an original
|
|
19
|
+
kotonoha implementation. It uses
|
|
20
|
+
[GiNZA](https://github.com/megagonlabs/ginza) for tokenization, part-of-speech
|
|
21
|
+
tagging, dependency parsing, lemmatization, and named-entity recognition.
|
|
22
|
+
|
|
23
|
+
## Responsibility boundary
|
|
24
|
+
|
|
25
|
+
- `tech-writer` owns doctype selection, section structure, completeness,
|
|
26
|
+
traceability, and Markdown organization.
|
|
27
|
+
- `presentation-planner` owns audience strategy, scenarios, slide order, and
|
|
28
|
+
design specifications.
|
|
29
|
+
- `japanese-prose` owns sentence-level Japanese: clarity, rhythm, reading
|
|
30
|
+
load, terminology, repeated patterns, and contextual rewriting.
|
|
31
|
+
|
|
32
|
+
Never move, add, or remove sections unless the calling skill explicitly
|
|
33
|
+
authorizes it. Never alter requirement IDs, risk IDs, control IDs, numbers,
|
|
34
|
+
units, dates, proper nouns, citations, URLs, commands, tables, schemas,
|
|
35
|
+
acceptance criteria, approval states, or normative force.
|
|
36
|
+
|
|
37
|
+
## Modes
|
|
38
|
+
|
|
39
|
+
- `write`: write or rewrite Japanese prose. Use `quick` unless the document
|
|
40
|
+
is external-facing, high-risk, or longer than roughly 10,000 Japanese
|
|
41
|
+
characters.
|
|
42
|
+
- `review`: report problems and proposed corrections without editing.
|
|
43
|
+
- `score`: run diagnostics and report the 0–100 score without rewriting.
|
|
44
|
+
- `quick`: run prose lint once, review findings in context, edit, and rerun
|
|
45
|
+
until no new actionable findings appear.
|
|
46
|
+
- `full`: run prose lint, reading-load lint, outline extraction, and
|
|
47
|
+
terminology extraction; then review the whole document against
|
|
48
|
+
`references/review-workflow.md`.
|
|
49
|
+
|
|
50
|
+
## Workflow
|
|
51
|
+
|
|
52
|
+
1. Read the target, audience, intended outcome, and calling skill's frozen
|
|
53
|
+
invariants.
|
|
54
|
+
2. Read `references/writing-guidelines.md` before generating or rewriting
|
|
55
|
+
prose.
|
|
56
|
+
3. Run the GiNZA prose diagnostic:
|
|
57
|
+
|
|
58
|
+
```bash
|
|
59
|
+
uv run scripts/lint.py <target-file> --genre tech --json > <workdir>/prose-baseline.json
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
4. For `full` mode, also run:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
uv run scripts/lint.py <target-file> --genre tech --reading-load --json
|
|
66
|
+
uv run scripts/outline.py <target-file>
|
|
67
|
+
uv run scripts/terms.py <target-file> --json
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
5. Classify each finding as `fix` or `keep`. A detector identifies a review
|
|
71
|
+
target; it does not authorize blind replacement.
|
|
72
|
+
6. Apply only fixes that improve the intended reader's understanding while
|
|
73
|
+
preserving the frozen invariants.
|
|
74
|
+
7. Rerun the prose diagnostic with
|
|
75
|
+
`--baseline <workdir>/prose-baseline.json` to classify new, persisting,
|
|
76
|
+
and resolved findings. Replace the baseline only after recording the
|
|
77
|
+
decisions for the current round.
|
|
78
|
+
8. Stop when every finding has a decision, no new actionable finding appears,
|
|
79
|
+
and the document passes the checks in `references/review-workflow.md`.
|
|
80
|
+
|
|
81
|
+
Allow at most three edit-and-diagnose rounds per invocation. If actionable
|
|
82
|
+
findings or invariant violations remain, report that the optimization did not
|
|
83
|
+
converge and list the unresolved items.
|
|
84
|
+
|
|
85
|
+
## Markdown emphasis
|
|
86
|
+
|
|
87
|
+
When strong emphasis touches surrounding prose, put half-width spaces outside
|
|
88
|
+
the delimiters: `これは **重要** です`, not `これは**重要**です`. Spaces are
|
|
89
|
+
unnecessary at line boundaries or next to punctuation, and must not be placed
|
|
90
|
+
inside `**`. Use ASCII spaces (`U+0020`), never full-width spaces (`U+3000`),
|
|
91
|
+
tabs, or non-breaking spaces, immediately before and after emphasis embedded
|
|
92
|
+
in prose. Write `これは **「重要」** と説明する`, not
|
|
93
|
+
`これは **「重要」** と説明する`.
|
|
94
|
+
|
|
95
|
+
## Diagnostic interpretation
|
|
96
|
+
|
|
97
|
+
The score is a triage aid, not a quality certificate. Read
|
|
98
|
+
`references/scoring.md` before presenting it. GiNZA provides linguistic
|
|
99
|
+
observations; the agent remains responsible for deciding whether a change is
|
|
100
|
+
correct in context.
|
|
101
|
+
|
|
102
|
+
## Completion report
|
|
103
|
+
|
|
104
|
+
Report:
|
|
105
|
+
|
|
106
|
+
- mode and diagnostics executed
|
|
107
|
+
- initial and final scores
|
|
108
|
+
- findings fixed and findings deliberately kept
|
|
109
|
+
- any skipped diagnostic and its reason
|
|
110
|
+
- invariant verification result
|
|
111
|
+
- `completed` or `did not converge`
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
# Japanese prose review workflow
|
|
2
|
+
|
|
3
|
+
## Review findings in context
|
|
4
|
+
|
|
5
|
+
For every diagnostic finding, record one decision:
|
|
6
|
+
|
|
7
|
+
- `fix`: the expression increases ambiguity, reading load, or mechanical
|
|
8
|
+
repetition for the intended reader
|
|
9
|
+
- `keep`: the expression is required by accuracy, domain convention, quoted
|
|
10
|
+
material, or deliberate rhythm
|
|
11
|
+
|
|
12
|
+
Do not count a finding as resolved merely because the triggering text
|
|
13
|
+
disappeared. Confirm that the replacement preserves meaning.
|
|
14
|
+
|
|
15
|
+
## Check sentence structure
|
|
16
|
+
|
|
17
|
+
- The subject and predicate can be identified without rereading.
|
|
18
|
+
- Modifiers sit close to the words they modify.
|
|
19
|
+
- A long sentence contains one main relationship.
|
|
20
|
+
- Negation does not require the reader to reverse the meaning twice.
|
|
21
|
+
- Noun chains expose ownership, purpose, and target relationships.
|
|
22
|
+
|
|
23
|
+
## Check document rhythm
|
|
24
|
+
|
|
25
|
+
- Consecutive sentences do not begin with the same two lemmas without reason.
|
|
26
|
+
- Paragraphs do not all begin with the same connective.
|
|
27
|
+
- Sentence lengths vary with information weight.
|
|
28
|
+
- Nominal endings are not used as the default ending for every sentence.
|
|
29
|
+
|
|
30
|
+
## Check terminology
|
|
31
|
+
|
|
32
|
+
- Product names, identifiers, values, and citations are unchanged.
|
|
33
|
+
- An unfamiliar term is explained near its first use.
|
|
34
|
+
- One concept uses one preferred term unless a distinction is intentional.
|
|
35
|
+
- An acronym is expanded when the audience cannot be expected to know it.
|
|
36
|
+
|
|
37
|
+
## Check the whole document
|
|
38
|
+
|
|
39
|
+
Read only headings and paragraph openings. The argument must still be
|
|
40
|
+
predictable. Then read the finished prose continuously and verify that local
|
|
41
|
+
rewrites did not damage transitions or create repeated explanations.
|
|
42
|
+
|
|
43
|
+
## Convergence
|
|
44
|
+
|
|
45
|
+
The review is complete when:
|
|
46
|
+
|
|
47
|
+
- every finding has a `fix` or `keep` decision
|
|
48
|
+
- rerunning diagnostics produces no new actionable finding
|
|
49
|
+
- frozen invariants match the pre-edit document
|
|
50
|
+
- the reader can reach the requested outcome without an unstated inference
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# Diagnostic scoring
|
|
2
|
+
|
|
3
|
+
The prose lint starts at 100 and subtracts:
|
|
4
|
+
|
|
5
|
+
- 10 points for each `critical` finding
|
|
6
|
+
- 4 points for each `warning` finding
|
|
7
|
+
- 1 point for each `info` finding
|
|
8
|
+
|
|
9
|
+
The minimum is 0. Scores compare repeated runs of the same document and
|
|
10
|
+
configuration; they do not compare authors, genres, or unrelated documents.
|
|
11
|
+
|
|
12
|
+
Use these bands only as triage:
|
|
13
|
+
|
|
14
|
+
| Score | Interpretation |
|
|
15
|
+
|---|---|
|
|
16
|
+
| 90–100 | Few mechanically detectable review targets |
|
|
17
|
+
| 75–89 | Several passages need contextual review |
|
|
18
|
+
| 50–74 | Reading load or repeated patterns are widespread |
|
|
19
|
+
| 0–49 | Review the document section by section before publication |
|
|
20
|
+
|
|
21
|
+
A high score does not prove that facts are correct, the argument is complete,
|
|
22
|
+
or the prose is natural. A low score does not require every finding to be
|
|
23
|
+
changed. Always report the finding categories and contextual decisions beside
|
|
24
|
+
the score.
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
# Japanese prose writing guidelines
|
|
2
|
+
|
|
3
|
+
Use these rules while drafting or rewriting. They are constraints for
|
|
4
|
+
judgment, not search-and-replace recipes.
|
|
5
|
+
|
|
6
|
+
## Lead with the reader's required conclusion
|
|
7
|
+
|
|
8
|
+
State the decision, action, result, or answer before the background that
|
|
9
|
+
supports it. An introductory sentence must earn its place by changing what
|
|
10
|
+
the reader understands or does.
|
|
11
|
+
|
|
12
|
+
## Keep one main relationship per sentence
|
|
13
|
+
|
|
14
|
+
A sentence may contain qualifications, but its subject, predicate, and main
|
|
15
|
+
object must remain visible. Split a sentence when the reader must retain one
|
|
16
|
+
unfinished relationship while parsing another.
|
|
17
|
+
|
|
18
|
+
## Prefer explicit relationships over compressed noun chains
|
|
19
|
+
|
|
20
|
+
Long noun sequences hide whether terms express ownership, purpose, target, or
|
|
21
|
+
sequence. Add particles or a predicate when GiNZA identifies five or more
|
|
22
|
+
consecutive nouns.
|
|
23
|
+
|
|
24
|
+
## Use concrete actors and actions
|
|
25
|
+
|
|
26
|
+
Name who checks, changes, approves, records, or observes something. Avoid
|
|
27
|
+
turning actions into abstract nouns when a direct verb is clearer.
|
|
28
|
+
|
|
29
|
+
## Treat formulaic expressions as review targets
|
|
30
|
+
|
|
31
|
+
Expressions such as 「と言えるでしょう」 or 「以下のとおりです」 are not
|
|
32
|
+
automatically wrong. Keep them only when their rhetorical function is
|
|
33
|
+
necessary. Otherwise state the evidence, conclusion, or preview directly.
|
|
34
|
+
|
|
35
|
+
## Vary rhythm according to information weight
|
|
36
|
+
|
|
37
|
+
Do not force every sentence or paragraph to the same length. Use a short
|
|
38
|
+
sentence for a decisive result. Use a longer sentence only when its
|
|
39
|
+
relationships remain clear.
|
|
40
|
+
|
|
41
|
+
## Explain terminology at first use
|
|
42
|
+
|
|
43
|
+
Introduce an unfamiliar term by function before or beside its name. Preserve
|
|
44
|
+
established product names, protocol names, identifiers, and domain terms.
|
|
45
|
+
|
|
46
|
+
## Preserve uncertainty accurately
|
|
47
|
+
|
|
48
|
+
Keep distinctions among confirmed facts, estimates, assumptions, and
|
|
49
|
+
recommendations. Do not make prose smoother by increasing certainty.
|
|
50
|
+
|
|
51
|
+
## Use emphasis sparingly and safely
|
|
52
|
+
|
|
53
|
+
Emphasize only the phrase that changes the reader's decision. In Markdown,
|
|
54
|
+
separate `**strong emphasis**` from adjacent prose with half-width spaces.
|
|
55
|
+
|
|
56
|
+
## End with the consequence
|
|
57
|
+
|
|
58
|
+
Close a section with what the evidence means for the reader's next decision
|
|
59
|
+
or action. Do not append a generic summary when the consequence is already
|
|
60
|
+
clear.
|
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
"""Shared Markdown and GiNZA utilities for kotonoha's Japanese prose tools."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import re
|
|
7
|
+
from dataclasses import asdict, dataclass
|
|
8
|
+
from functools import lru_cache
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Iterable
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
FENCE_RE = re.compile(r"^\s{0,3}(`{3,}|~{3,})")
|
|
14
|
+
HEADING_RE = re.compile(r"^\s{0,3}#{1,6}\s+")
|
|
15
|
+
LIST_RE = re.compile(r"^\s*(?:[-+*]|\d+[.)])\s+")
|
|
16
|
+
BLOCKQUOTE_RE = re.compile(r"^\s*>\s?")
|
|
17
|
+
INLINE_CODE_RE = re.compile(r"`[^`\n]*`")
|
|
18
|
+
LINK_RE = re.compile(r"!?\[([^\]]*)\]\([^)]+\)")
|
|
19
|
+
URL_RE = re.compile(r"https?://\S+")
|
|
20
|
+
HTML_TAG_RE = re.compile(r"<[^>]+>")
|
|
21
|
+
SENTENCE_END_RE = re.compile(r"[。!?!?]+(?:[」』)】〉》]*)")
|
|
22
|
+
JAPANESE_RE = re.compile(r"[ぁ-んァ-ヶ一-龠々]")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class Finding:
|
|
27
|
+
line: int
|
|
28
|
+
category: str
|
|
29
|
+
severity: str
|
|
30
|
+
message: str
|
|
31
|
+
excerpt: str
|
|
32
|
+
evidence: dict | None = None
|
|
33
|
+
status: str | None = None
|
|
34
|
+
|
|
35
|
+
def to_dict(self) -> dict:
|
|
36
|
+
value = asdict(self)
|
|
37
|
+
return {key: item for key, item in value.items() if item is not None}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass
|
|
41
|
+
class ProseBlock:
|
|
42
|
+
text: str
|
|
43
|
+
start_line: int
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def read_text(path: Path) -> str:
|
|
47
|
+
if not path.is_file():
|
|
48
|
+
raise FileNotFoundError(f"file not found: {path}")
|
|
49
|
+
return path.read_text(encoding="utf-8")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def strip_markdown_inline(text: str) -> str:
|
|
53
|
+
text = INLINE_CODE_RE.sub(" コード ", text)
|
|
54
|
+
text = LINK_RE.sub(lambda match: match.group(1), text)
|
|
55
|
+
text = URL_RE.sub(" ", text)
|
|
56
|
+
text = HTML_TAG_RE.sub(" ", text)
|
|
57
|
+
text = text.replace("**", "").replace("__", "")
|
|
58
|
+
text = text.replace("~~", "")
|
|
59
|
+
return re.sub(r"[ \t]+", " ", text).strip()
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def iter_content_lines(text: str) -> Iterable[tuple[int, str]]:
|
|
63
|
+
"""Yield line-numbered Markdown outside metadata, code, and comments."""
|
|
64
|
+
lines = text.splitlines()
|
|
65
|
+
fence_marker = ""
|
|
66
|
+
in_frontmatter = bool(lines and lines[0].strip() == "---")
|
|
67
|
+
in_comment = False
|
|
68
|
+
|
|
69
|
+
for index, raw in enumerate(lines, start=1):
|
|
70
|
+
stripped = raw.strip()
|
|
71
|
+
if in_frontmatter:
|
|
72
|
+
if index > 1 and stripped == "---":
|
|
73
|
+
in_frontmatter = False
|
|
74
|
+
yield index, ""
|
|
75
|
+
continue
|
|
76
|
+
|
|
77
|
+
fence = FENCE_RE.match(raw)
|
|
78
|
+
if fence:
|
|
79
|
+
marker = fence.group(1)[0]
|
|
80
|
+
if not fence_marker:
|
|
81
|
+
fence_marker = marker
|
|
82
|
+
elif marker == fence_marker:
|
|
83
|
+
fence_marker = ""
|
|
84
|
+
yield index, ""
|
|
85
|
+
continue
|
|
86
|
+
if fence_marker:
|
|
87
|
+
yield index, ""
|
|
88
|
+
continue
|
|
89
|
+
|
|
90
|
+
if "<!--" in raw:
|
|
91
|
+
in_comment = True
|
|
92
|
+
if in_comment:
|
|
93
|
+
if "-->" in raw:
|
|
94
|
+
in_comment = False
|
|
95
|
+
yield index, ""
|
|
96
|
+
continue
|
|
97
|
+
yield index, raw
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def iter_prose_blocks(text: str) -> list[ProseBlock]:
|
|
101
|
+
"""Return Markdown prose blocks while excluding metadata and code."""
|
|
102
|
+
blocks: list[ProseBlock] = []
|
|
103
|
+
buffer: list[str] = []
|
|
104
|
+
buffer_line = 0
|
|
105
|
+
|
|
106
|
+
def flush() -> None:
|
|
107
|
+
nonlocal buffer, buffer_line
|
|
108
|
+
if buffer:
|
|
109
|
+
cleaned = strip_markdown_inline("\n".join(buffer))
|
|
110
|
+
if cleaned and JAPANESE_RE.search(cleaned):
|
|
111
|
+
blocks.append(ProseBlock(cleaned, buffer_line))
|
|
112
|
+
buffer = []
|
|
113
|
+
buffer_line = 0
|
|
114
|
+
|
|
115
|
+
for index, raw in iter_content_lines(text):
|
|
116
|
+
stripped = raw.strip()
|
|
117
|
+
if not stripped or stripped.startswith("|") or HEADING_RE.match(raw):
|
|
118
|
+
flush()
|
|
119
|
+
continue
|
|
120
|
+
|
|
121
|
+
list_item = LIST_RE.match(raw)
|
|
122
|
+
if list_item:
|
|
123
|
+
flush()
|
|
124
|
+
line = LIST_RE.sub("", raw)
|
|
125
|
+
line = BLOCKQUOTE_RE.sub("", line)
|
|
126
|
+
if not buffer:
|
|
127
|
+
buffer_line = index
|
|
128
|
+
buffer.append(line)
|
|
129
|
+
|
|
130
|
+
flush()
|
|
131
|
+
return blocks
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def sentence_line(block: ProseBlock, start_char: int) -> int:
|
|
135
|
+
return block.start_line + block.text[:start_char].count("\n")
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def split_sentences_fallback(text: str) -> Iterable[tuple[str, int]]:
|
|
139
|
+
start = 0
|
|
140
|
+
for match in SENTENCE_END_RE.finditer(text):
|
|
141
|
+
end = match.end()
|
|
142
|
+
sentence = text[start:end].strip()
|
|
143
|
+
if sentence:
|
|
144
|
+
yield sentence, start
|
|
145
|
+
start = end
|
|
146
|
+
tail = text[start:].strip()
|
|
147
|
+
if tail:
|
|
148
|
+
yield tail, start
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
@lru_cache(maxsize=2)
|
|
152
|
+
def load_ginza(enable_ner: bool = False):
|
|
153
|
+
import spacy
|
|
154
|
+
|
|
155
|
+
disabled = [] if enable_ner else ["ner"]
|
|
156
|
+
return spacy.load("ja_ginza", disable=disabled)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def content_tokens(span) -> list:
|
|
160
|
+
return [
|
|
161
|
+
token
|
|
162
|
+
for token in span
|
|
163
|
+
if not token.is_space and not token.is_punct and token.pos_ != "SYM"
|
|
164
|
+
]
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def dependency_depth(token) -> int:
|
|
168
|
+
depth = 0
|
|
169
|
+
current = token
|
|
170
|
+
seen: set[int] = set()
|
|
171
|
+
while current.head.i != current.i and current.i not in seen and depth < 40:
|
|
172
|
+
seen.add(current.i)
|
|
173
|
+
current = current.head
|
|
174
|
+
depth += 1
|
|
175
|
+
return depth
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def excerpt(text: str, limit: int = 100) -> str:
|
|
179
|
+
value = re.sub(r"\s+", " ", text).strip()
|
|
180
|
+
return value if len(value) <= limit else f"{value[:limit - 1]}…"
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def finding_key(finding: Finding) -> str:
|
|
184
|
+
normalized = re.sub(r"\d+", "#", finding.excerpt.lower())
|
|
185
|
+
raw = f"{finding.category}\0{normalized}".encode()
|
|
186
|
+
return hashlib.sha256(raw).hexdigest()[:20]
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def score_findings(findings: list[Finding]) -> int:
|
|
190
|
+
weights = {"info": 1, "warning": 4, "critical": 10}
|
|
191
|
+
deduction = sum(weights.get(item.severity, 4) for item in findings)
|
|
192
|
+
return max(0, 100 - deduction)
|