libkuraji 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,29 @@
1
+ BSD 3-Clause License
2
+
3
+ Copyright (c) 2019,2023 Takuya Nishimoto
4
+ All rights reserved.
5
+
6
+ Redistribution and use in source and binary forms, with or without
7
+ modification, are permitted provided that the following conditions are met:
8
+
9
+ * Redistributions of source code must retain the above copyright notice, this
10
+ list of conditions and the following disclaimer.
11
+
12
+ * Redistributions in binary form must reproduce the above copyright notice,
13
+ this list of conditions and the following disclaimer in the documentation
14
+ and/or other materials provided with the distribution.
15
+
16
+ * Neither the name of the copyright holder nor the names of its
17
+ contributors may be used to endorse or promote products derived from
18
+ this software without specific prior written permission.
19
+
20
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
21
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
22
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
23
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
24
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
25
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
26
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
27
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
28
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
29
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,248 @@
1
+ Metadata-Version: 2.4
2
+ Name: libkuraji
3
+ Version: 1.0.0
4
+ Summary: Japanese Braille translator originally developed for NVDAJP
5
+ Author: Takuya Nishimoto
6
+ License-Expression: BSD-3-Clause
7
+ Project-URL: Homepage, https://github.com/nishimotz/libkuraji
8
+ Project-URL: Repository, https://github.com/nishimotz/libkuraji
9
+ Project-URL: Issues, https://github.com/nishimotz/libkuraji/issues
10
+ Keywords: braille,japanese,accessibility,nvda,mecab
11
+ Classifier: Development Status :: 5 - Production/Stable
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Operating System :: OS Independent
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.10
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Topic :: Adaptive Technologies
20
+ Classifier: Topic :: Text Processing :: Linguistic
21
+ Requires-Python: >=3.10
22
+ Description-Content-Type: text/markdown
23
+ License-File: LICENSE
24
+ Provides-Extra: dev
25
+ Requires-Dist: pytest; extra == "dev"
26
+ Provides-Extra: integration
27
+ Requires-Dist: fugashi; extra == "integration"
28
+ Dynamic: license-file
29
+
30
+ # libkuraji
31
+
32
+ [日本語](README-ja.md)
33
+
34
+ Japanese Braille translator originally developed for NVDAJP.
35
+
36
+ Translates mixed Kanji/Kana Japanese text into Japanese 6-dot braille (Unicode braille patterns), with morphological word segmentation (masuake / spacing). Returns a position map from each braille cell back to the source text.
37
+
38
+ ## Requirements
39
+
40
+ - Python 3.10 or later
41
+ - `pip install libkuraji[integration]` (`fugashi`)
42
+ - Environment variable `LIBKURAJI_INTEGRATION=1` (enables the JTalk extended dictionary)
43
+
44
+ ## Installation
45
+
46
+ ```console
47
+ pip install 'libkuraji[integration]'
48
+ export LIBKURAJI_INTEGRATION=1 # Windows PowerShell: $env:LIBKURAJI_INTEGRATION=1
49
+ ```
50
+
51
+ On macOS and Homebrew Python, system-wide installs may be restricted. Use a virtual environment in that case.
52
+
53
+ ```console
54
+ python3 -m venv .venv
55
+ source .venv/bin/activate # Windows: .venv\Scripts\activate
56
+ pip install 'libkuraji[integration]'
57
+ export LIBKURAJI_INTEGRATION=1
58
+ ```
59
+
60
+ ### For development (from the repository)
61
+
62
+ ```console
63
+ git clone https://github.com/nishimotz/libkuraji.git
64
+ cd libkuraji
65
+ python3 -m venv .venv
66
+ source .venv/bin/activate
67
+ pip install -e '.[dev,integration]'
68
+ export LIBKURAJI_INTEGRATION=1
69
+ ```
70
+
71
+ In zsh, quote the extras: `'.[dev,integration]'`.
72
+
73
+ ## Quick start
74
+
75
+ On first run, dictionary binaries from [libkuraji-jtalk-dic](https://github.com/nishimotz/libkuraji-jtalk-dic) are downloaded automatically from GitHub Releases.
76
+
77
+ ```console
78
+ kuraji "私は点字を読みます。"
79
+ # => ⠄⠕⠳⠄ ⠟⠴⠐⠳⠔ ⠜⠷⠵⠹⠲
80
+ ```
81
+
82
+ ```python
83
+ import libkuraji
84
+
85
+ cells, inpos, outpos, cursor = libkuraji.translate_kanji(
86
+ "私は点字を読みます。",
87
+ unicodeIO=True,
88
+ )
89
+ # => '⠄⠕⠳⠄ ⠟⠴⠐⠳⠔ ⠜⠷⠵⠹⠲'
90
+ ```
91
+
92
+ Pass `unicodeIO=True` to get Unicode braille (the same format as the CLI).
93
+
94
+ Input is limited to 65,536 characters by default (`InputTooLongError` when exceeded). Override with `LIBKURAJI_MAX_INPUT_CHARS`; set `0` to disable the limit.
95
+
96
+ ## Usage
97
+
98
+ ### Python API
99
+
100
+ ```python
101
+ import libkuraji
102
+
103
+ cells, inpos, outpos, cursor = libkuraji.translate_kanji(
104
+ "私は点字を読みます。",
105
+ unicodeIO=True,
106
+ )
107
+ # inpos[i] is the input index for cells[i]
108
+ # outpos[j] is the braille index for input position j
109
+ ```
110
+
111
+ To use a custom morphological analyzer, inject it with `initialize`. The analyzer must implement `analyze(text, logwrite)` and `is_ready()`.
112
+
113
+ ```python
114
+ libkuraji.initialize(analyzer=my_analyzer)
115
+ cells, inpos, outpos, cursor = libkuraji.translate_kanji(
116
+ "私は点字を読みます。",
117
+ unicodeIO=True,
118
+ )
119
+ ```
120
+
121
+ For NABCC (computer braille) mode: `translate_kanji(text, nabcc=True, unicodeIO=True)`.
122
+
123
+ #### Foreign quotation marks and information processing braille
124
+
125
+ libkuraji automatically detects words containing Latin letters and symbols, wrapping them in one of two braille indicators.
126
+
127
+ | Mode | Indicator | Target | Grade 2 English |
128
+ |------|-----------|--------|-----------------|
129
+ | Foreign quotation marks | `⠦...⠴` | Natural language foreign text (with spaces/apostrophes) | Only when liblouis is injected |
130
+ | Information processing braille | `⠠⠦...⠠⠴` | URLs, email addresses, file paths, etc. | Not applicable |
131
+
132
+ #### Grade 2 English braille inside foreign quotation marks
133
+
134
+ Pass a `louisTranslate` function and `louisTableList` (e.g. `["en-ueb-g2.ctb"]`) to `translate_kanji` to apply Grade 2 English braille translation to the inner text of foreign quotation marks `⠦...⠴`. libkuraji does not provide a `louis` translate function — the caller must inject it.
135
+
136
+ ```python
137
+ import louis
138
+
139
+ def my_louis_translate(table_list, text, cursorPos=0, mode=0):
140
+ return louis.translate(
141
+ table_list, text,
142
+ cursorPos=cursorPos, mode=mode,
143
+ )
144
+
145
+ cells, inpos, outpos, cursor = libkuraji.translate_kanji(
146
+ "これは ⠦English text⠴ です。",
147
+ unicodeIO=True,
148
+ louisTranslate=my_louis_translate,
149
+ louisTableList=["en-ueb-g2.ctb"],
150
+ )
151
+ ```
152
+
153
+ See [docs/encoding.md](docs/encoding.md) for output encoding details (Unicode braille vs liblouis dotsIO, and what `unicodeIO` means).
154
+
155
+ #### MeCab setup
156
+
157
+ | Platform | Notes |
158
+ |----------|-------|
159
+ | Windows | The `fugashi` wheel bundles `libmecab.dll`, so `pip` alone is sufficient |
160
+ | macOS / Linux | Verified in this repository with `fugashi` only. Some environments may require a system [MeCab](https://taku910.github.io/mecab/) install |
161
+
162
+ Override the dictionary release tag with `LIBKURAJI_JTALK_DIC_TAG` (default: `DEFAULT_DIC_TAG` in `src/libkuraji/jtalk_dic.py`). To skip download, point `LIBKURAJI_JTALK_DIC_DIR` at an already extracted dictionary directory.
163
+
164
+ ### CLI
165
+
166
+ ```console
167
+ kuraji "私は点字を読みます。"
168
+ ⠄⠕⠳⠄ ⠟⠴⠐⠳⠔ ⠜⠷⠵⠹⠲
169
+
170
+ kuraji "私は点字を読みます。" --positions
171
+ {"text": "私は点字を読みます。", "braille": "⠄⠕⠳⠄ ⠟⠴⠐⠳⠔ ⠜⠷⠵⠹⠲", ...}
172
+ ```
173
+
174
+ Main options:
175
+
176
+ | Option | Description |
177
+ |--------|-------------|
178
+ | `-p` / `--positions` | Output position map as JSON |
179
+ | `--nabcc` | NABCC (computer braille) mode |
180
+ | `-j` / `--kanji` | Force mixed Kanji/Kana mode |
181
+ | `-k` / `--kana` | Kana-only mode (see below) |
182
+
183
+ ### Kana input only (supplement)
184
+
185
+ If the input is already in katakana, use `translate` without MeCab.
186
+
187
+ ```python
188
+ from libkuraji import translate
189
+
190
+ translate("ワタシワ テンジヲ ヨミマス。")
191
+ ```
192
+
193
+ ```console
194
+ kuraji -k "ワタシワ テンジヲ ヨミマス。"
195
+ ```
196
+
197
+ ## Testing
198
+
199
+ ### Default tests (same as CI)
200
+
201
+ Runs without MeCab. Replays recorded MeCab output from `tests/mecabFixture.json` to verify translator2.
202
+
203
+ ```console
204
+ pip install -e '.[dev]'
205
+ pytest
206
+ ```
207
+
208
+ Test data in `tests/harness.json` and related files follow rules from the Japanese braille guide (*Ten'yaku no Tebiki*). To re-record `tests/mecabFixture.json`, use `miscDepsJp/jptools/recordMecabFixture.py` in the nvdajp repository.
209
+
210
+ ### Integration tests with the real dictionary (optional)
211
+
212
+ Opt-in tests that verify translation against the live JTalk extended dictionary and MeCab.
213
+
214
+ Prerequisites:
215
+
216
+ - `pip install -e '.[dev,integration]'`
217
+ - `gh` CLI (falls back to the GitHub REST API if unavailable)
218
+
219
+ Run:
220
+
221
+ ```console
222
+ export LIBKURAJI_INTEGRATION=1
223
+ pytest tests/test_integration.py -q
224
+
225
+ # Also run exact MeCab output parity checks
226
+ export LIBKURAJI_PARITY_CHECK=1
227
+ pytest tests/test_integration.py -q
228
+ ```
229
+
230
+ On Windows PowerShell:
231
+
232
+ ```powershell
233
+ $env:LIBKURAJI_INTEGRATION=1
234
+ pytest tests/test_integration.py -q
235
+ ```
236
+
237
+ Some integration tests also run in CI (GitHub Actions) on Windows and Linux. `test_mecab_fixture_parity` (internal MeCab output comparison across versions) is skipped by default and intended for dictionary update verification.
238
+
239
+ Integration tests also reproduce MeCab output correction via `mecab_correct.py` (ported from nvdajp's `Mecab_correctFeatures` under BSD relicensing). The `libmecab.dll` bundled with `fugashi` and the nvdajp MeCab used when recording fixtures may differ on a few symbol-only inputs (about 98% match overall). Final braille output consistency is preserved and CI passes.
240
+
241
+ ## Relationship to NVDA Japanese
242
+
243
+ - This library separates and standalone-izes the braille engine (translator1/translator2) from [nvdajp](https://github.com/nvdajp/nvdajp). The translator1 equivalent (`kana` module) is a clean-room rewrite driven by tests; translator2 was ported under BSD relicensing by the copyright holder.
244
+ - Decoupling plan: `projectDocs/jp/braille-engine-decoupling-plan.md` in the nvdajp repository.
245
+
246
+ ## License
247
+
248
+ BSD 3-Clause License. See [LICENSE](LICENSE).
@@ -0,0 +1,219 @@
1
+ # libkuraji
2
+
3
+ [日本語](README-ja.md)
4
+
5
+ Japanese Braille translator originally developed for NVDAJP.
6
+
7
+ Translates mixed Kanji/Kana Japanese text into Japanese 6-dot braille (Unicode braille patterns), with morphological word segmentation (masuake / spacing). Returns a position map from each braille cell back to the source text.
8
+
9
+ ## Requirements
10
+
11
+ - Python 3.10 or later
12
+ - `pip install libkuraji[integration]` (`fugashi`)
13
+ - Environment variable `LIBKURAJI_INTEGRATION=1` (enables the JTalk extended dictionary)
14
+
15
+ ## Installation
16
+
17
+ ```console
18
+ pip install 'libkuraji[integration]'
19
+ export LIBKURAJI_INTEGRATION=1 # Windows PowerShell: $env:LIBKURAJI_INTEGRATION=1
20
+ ```
21
+
22
+ On macOS and Homebrew Python, system-wide installs may be restricted. Use a virtual environment in that case.
23
+
24
+ ```console
25
+ python3 -m venv .venv
26
+ source .venv/bin/activate # Windows: .venv\Scripts\activate
27
+ pip install 'libkuraji[integration]'
28
+ export LIBKURAJI_INTEGRATION=1
29
+ ```
30
+
31
+ ### For development (from the repository)
32
+
33
+ ```console
34
+ git clone https://github.com/nishimotz/libkuraji.git
35
+ cd libkuraji
36
+ python3 -m venv .venv
37
+ source .venv/bin/activate
38
+ pip install -e '.[dev,integration]'
39
+ export LIBKURAJI_INTEGRATION=1
40
+ ```
41
+
42
+ In zsh, quote the extras: `'.[dev,integration]'`.
43
+
44
+ ## Quick start
45
+
46
+ On first run, dictionary binaries from [libkuraji-jtalk-dic](https://github.com/nishimotz/libkuraji-jtalk-dic) are downloaded automatically from GitHub Releases.
47
+
48
+ ```console
49
+ kuraji "私は点字を読みます。"
50
+ # => ⠄⠕⠳⠄ ⠟⠴⠐⠳⠔ ⠜⠷⠵⠹⠲
51
+ ```
52
+
53
+ ```python
54
+ import libkuraji
55
+
56
+ cells, inpos, outpos, cursor = libkuraji.translate_kanji(
57
+ "私は点字を読みます。",
58
+ unicodeIO=True,
59
+ )
60
+ # => '⠄⠕⠳⠄ ⠟⠴⠐⠳⠔ ⠜⠷⠵⠹⠲'
61
+ ```
62
+
63
+ Pass `unicodeIO=True` to get Unicode braille (the same format as the CLI).
64
+
65
+ Input is limited to 65,536 characters by default (`InputTooLongError` when exceeded). Override with `LIBKURAJI_MAX_INPUT_CHARS`; set `0` to disable the limit.
66
+
67
+ ## Usage
68
+
69
+ ### Python API
70
+
71
+ ```python
72
+ import libkuraji
73
+
74
+ cells, inpos, outpos, cursor = libkuraji.translate_kanji(
75
+ "私は点字を読みます。",
76
+ unicodeIO=True,
77
+ )
78
+ # inpos[i] is the input index for cells[i]
79
+ # outpos[j] is the braille index for input position j
80
+ ```
81
+
82
+ To use a custom morphological analyzer, inject it with `initialize`. The analyzer must implement `analyze(text, logwrite)` and `is_ready()`.
83
+
84
+ ```python
85
+ libkuraji.initialize(analyzer=my_analyzer)
86
+ cells, inpos, outpos, cursor = libkuraji.translate_kanji(
87
+ "私は点字を読みます。",
88
+ unicodeIO=True,
89
+ )
90
+ ```
91
+
92
+ For NABCC (computer braille) mode: `translate_kanji(text, nabcc=True, unicodeIO=True)`.
93
+
94
+ #### Foreign quotation marks and information processing braille
95
+
96
+ libkuraji automatically detects words containing Latin letters and symbols, wrapping them in one of two braille indicators.
97
+
98
+ | Mode | Indicator | Target | Grade 2 English |
99
+ |------|-----------|--------|-----------------|
100
+ | Foreign quotation marks | `⠦...⠴` | Natural language foreign text (with spaces/apostrophes) | Only when liblouis is injected |
101
+ | Information processing braille | `⠠⠦...⠠⠴` | URLs, email addresses, file paths, etc. | Not applicable |
102
+
103
+ #### Grade 2 English braille inside foreign quotation marks
104
+
105
+ Pass a `louisTranslate` function and `louisTableList` (e.g. `["en-ueb-g2.ctb"]`) to `translate_kanji` to apply Grade 2 English braille translation to the inner text of foreign quotation marks `⠦...⠴`. libkuraji does not provide a `louis` translate function — the caller must inject it.
106
+
107
+ ```python
108
+ import louis
109
+
110
+ def my_louis_translate(table_list, text, cursorPos=0, mode=0):
111
+ return louis.translate(
112
+ table_list, text,
113
+ cursorPos=cursorPos, mode=mode,
114
+ )
115
+
116
+ cells, inpos, outpos, cursor = libkuraji.translate_kanji(
117
+ "これは ⠦English text⠴ です。",
118
+ unicodeIO=True,
119
+ louisTranslate=my_louis_translate,
120
+ louisTableList=["en-ueb-g2.ctb"],
121
+ )
122
+ ```
123
+
124
+ See [docs/encoding.md](docs/encoding.md) for output encoding details (Unicode braille vs liblouis dotsIO, and what `unicodeIO` means).
125
+
126
+ #### MeCab setup
127
+
128
+ | Platform | Notes |
129
+ |----------|-------|
130
+ | Windows | The `fugashi` wheel bundles `libmecab.dll`, so `pip` alone is sufficient |
131
+ | macOS / Linux | Verified in this repository with `fugashi` only. Some environments may require a system [MeCab](https://taku910.github.io/mecab/) install |
132
+
133
+ Override the dictionary release tag with `LIBKURAJI_JTALK_DIC_TAG` (default: `DEFAULT_DIC_TAG` in `src/libkuraji/jtalk_dic.py`). To skip download, point `LIBKURAJI_JTALK_DIC_DIR` at an already extracted dictionary directory.
134
+
135
+ ### CLI
136
+
137
+ ```console
138
+ kuraji "私は点字を読みます。"
139
+ ⠄⠕⠳⠄ ⠟⠴⠐⠳⠔ ⠜⠷⠵⠹⠲
140
+
141
+ kuraji "私は点字を読みます。" --positions
142
+ {"text": "私は点字を読みます。", "braille": "⠄⠕⠳⠄ ⠟⠴⠐⠳⠔ ⠜⠷⠵⠹⠲", ...}
143
+ ```
144
+
145
+ Main options:
146
+
147
+ | Option | Description |
148
+ |--------|-------------|
149
+ | `-p` / `--positions` | Output position map as JSON |
150
+ | `--nabcc` | NABCC (computer braille) mode |
151
+ | `-j` / `--kanji` | Force mixed Kanji/Kana mode |
152
+ | `-k` / `--kana` | Kana-only mode (see below) |
153
+
154
+ ### Kana input only (supplement)
155
+
156
+ If the input is already in katakana, use `translate` without MeCab.
157
+
158
+ ```python
159
+ from libkuraji import translate
160
+
161
+ translate("ワタシワ テンジヲ ヨミマス。")
162
+ ```
163
+
164
+ ```console
165
+ kuraji -k "ワタシワ テンジヲ ヨミマス。"
166
+ ```
167
+
168
+ ## Testing
169
+
170
+ ### Default tests (same as CI)
171
+
172
+ Runs without MeCab. Replays recorded MeCab output from `tests/mecabFixture.json` to verify translator2.
173
+
174
+ ```console
175
+ pip install -e '.[dev]'
176
+ pytest
177
+ ```
178
+
179
+ Test data in `tests/harness.json` and related files follow rules from the Japanese braille guide (*Ten'yaku no Tebiki*). To re-record `tests/mecabFixture.json`, use `miscDepsJp/jptools/recordMecabFixture.py` in the nvdajp repository.
180
+
181
+ ### Integration tests with the real dictionary (optional)
182
+
183
+ Opt-in tests that verify translation against the live JTalk extended dictionary and MeCab.
184
+
185
+ Prerequisites:
186
+
187
+ - `pip install -e '.[dev,integration]'`
188
+ - `gh` CLI (falls back to the GitHub REST API if unavailable)
189
+
190
+ Run:
191
+
192
+ ```console
193
+ export LIBKURAJI_INTEGRATION=1
194
+ pytest tests/test_integration.py -q
195
+
196
+ # Also run exact MeCab output parity checks
197
+ export LIBKURAJI_PARITY_CHECK=1
198
+ pytest tests/test_integration.py -q
199
+ ```
200
+
201
+ On Windows PowerShell:
202
+
203
+ ```powershell
204
+ $env:LIBKURAJI_INTEGRATION=1
205
+ pytest tests/test_integration.py -q
206
+ ```
207
+
208
+ Some integration tests also run in CI (GitHub Actions) on Windows and Linux. `test_mecab_fixture_parity` (internal MeCab output comparison across versions) is skipped by default and intended for dictionary update verification.
209
+
210
+ Integration tests also reproduce MeCab output correction via `mecab_correct.py` (ported from nvdajp's `Mecab_correctFeatures` under BSD relicensing). The `libmecab.dll` bundled with `fugashi` and the nvdajp MeCab used when recording fixtures may differ on a few symbol-only inputs (about 98% match overall). Final braille output consistency is preserved and CI passes.
211
+
212
+ ## Relationship to NVDA Japanese
213
+
214
+ - This library separates and standalone-izes the braille engine (translator1/translator2) from [nvdajp](https://github.com/nvdajp/nvdajp). The translator1 equivalent (`kana` module) is a clean-room rewrite driven by tests; translator2 was ported under BSD relicensing by the copyright holder.
215
+ - Decoupling plan: `projectDocs/jp/braille-engine-decoupling-plan.md` in the nvdajp repository.
216
+
217
+ ## License
218
+
219
+ BSD 3-Clause License. See [LICENSE](LICENSE).
@@ -0,0 +1,43 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "libkuraji"
7
+ version = "1.0.0"
8
+ description = "Japanese Braille translator originally developed for NVDAJP"
9
+ readme = "README.md"
10
+ license = "BSD-3-Clause"
11
+ authors = [{ name = "Takuya Nishimoto" }]
12
+ requires-python = ">=3.10"
13
+ keywords = ["braille", "japanese", "accessibility", "nvda", "mecab"]
14
+ classifiers = [
15
+ "Development Status :: 5 - Production/Stable",
16
+ "Intended Audience :: Developers",
17
+ "Operating System :: OS Independent",
18
+ "Programming Language :: Python :: 3",
19
+ "Programming Language :: Python :: 3.10",
20
+ "Programming Language :: Python :: 3.11",
21
+ "Programming Language :: Python :: 3.12",
22
+ "Programming Language :: Python :: 3.13",
23
+ "Topic :: Adaptive Technologies",
24
+ "Topic :: Text Processing :: Linguistic",
25
+ ]
26
+
27
+ [project.urls]
28
+ Homepage = "https://github.com/nishimotz/libkuraji"
29
+ Repository = "https://github.com/nishimotz/libkuraji"
30
+ Issues = "https://github.com/nishimotz/libkuraji/issues"
31
+
32
+ [project.scripts]
33
+ kuraji = "libkuraji.cli:main"
34
+
35
+ [project.optional-dependencies]
36
+ dev = ["pytest"]
37
+ integration = ["fugashi"]
38
+
39
+ [tool.setuptools.packages.find]
40
+ where = ["src"]
41
+
42
+ [tool.pytest.ini_options]
43
+ testpaths = ["tests"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,69 @@
1
+ # libkuraji
2
+ # Copyright (C) 2019-2026 Takuya Nishimoto
3
+ # License: BSD 3-Clause. See LICENSE.
4
+
5
+ from .kana import translate_with_pos
6
+ from .limits import InputTooLongError
7
+
8
+
9
+ def translate(text: str, nabcc: bool = False) -> str:
10
+ """Translate kana text (translator2 output) into braille cells."""
11
+ return translate_with_pos(text, nabcc=nabcc)[0]
12
+
13
+
14
+ def initialize(analyzer=None, logwrite=None):
15
+ """Initialize the morphological analyzer for mixed-text (Kanji/Kana) translation.
16
+
17
+ If analyzer is None, a default JTalkDicAnalyzer (which uses fugashi and
18
+ automatically downloads the JTalk dictionary if needed) will be used.
19
+ """
20
+ if analyzer is None:
21
+ try:
22
+ from .jtalk_dic import make_analyzer
23
+ analyzer = make_analyzer()
24
+ except ImportError as e:
25
+ raise ImportError(
26
+ "The default analyzer requires 'fugashi' and dictionary dependencies. "
27
+ "Please run 'pip install libkuraji[integration]' to install them, "
28
+ "or pass a custom analyzer instance."
29
+ ) from e
30
+ from . import translator2
31
+ translator2.initialize(analyzer=analyzer, logwrite=logwrite)
32
+
33
+
34
+ def translate_kanji(
35
+ text: str, cursorPos: int = 0, nabcc: bool = False, **kwargs
36
+ ) -> tuple[str, list[int], list[int], int]:
37
+ """Translate mixed Kanji/Kana Japanese text, returning braille and position maps.
38
+
39
+ If the analyzer has not been initialized, it will be automatically initialized
40
+ with the default JTalkDicAnalyzer.
41
+
42
+ Keyword arguments are forwarded to ``translator2.translate``. Notable options:
43
+
44
+ * ``unicodeIO`` (bool, default ``False``): If ``True``, return Unicode braille
45
+ (U+2800..U+28FF, blanks as U+0020), matching the CLI and liblouis
46
+ ``dotsIO | ucBrl``. If ``False``, return liblouis ``dotsIO`` cells
47
+ (U+8000..U+80FF, blanks as U+2800), the inherited nvdajp translator2
48
+ default. See ``docs/encoding.md``.
49
+ """
50
+ from . import translator2
51
+ if not translator2.mecab_initialized:
52
+ initialize()
53
+ return translator2.translate(text, cursorPos=cursorPos, nabcc=nabcc, **kwargs)
54
+
55
+
56
+ def terminate():
57
+ """Terminate the morphological analyzer and clean up resources."""
58
+ from . import translator2
59
+ translator2.terminate()
60
+
61
+
62
+ __all__ = [
63
+ "translate",
64
+ "translate_with_pos",
65
+ "initialize",
66
+ "translate_kanji",
67
+ "terminate",
68
+ "InputTooLongError",
69
+ ]