noentenc 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- noentenc-0.1.0/PKG-INFO +199 -0
- noentenc-0.1.0/README.md +173 -0
- noentenc-0.1.0/pyproject.toml +115 -0
- noentenc-0.1.0/pyproject.toml.orig +111 -0
- noentenc-0.1.0/src/noentenc/__init__.py +13 -0
- noentenc-0.1.0/src/noentenc/_dataframe.py +41 -0
- noentenc-0.1.0/src/noentenc/_onnx.py +25 -0
- noentenc-0.1.0/src/noentenc/language_detection/__init__.py +23 -0
- noentenc-0.1.0/src/noentenc/language_detection/_download.py +72 -0
- noentenc-0.1.0/src/noentenc/language_detection/_iso639.py +854 -0
- noentenc-0.1.0/src/noentenc/language_detection/_optional.py +13 -0
- noentenc-0.1.0/src/noentenc/language_detection/base.py +141 -0
- noentenc-0.1.0/src/noentenc/language_detection/labels.py +145 -0
- noentenc-0.1.0/src/noentenc/language_detection/models/__init__.py +23 -0
- noentenc-0.1.0/src/noentenc/language_detection/models/base.py +128 -0
- noentenc-0.1.0/src/noentenc/language_detection/models/cld3_model.py +64 -0
- noentenc-0.1.0/src/noentenc/language_detection/models/fasttext/__init__.py +7 -0
- noentenc-0.1.0/src/noentenc/language_detection/models/fasttext/format.py +219 -0
- noentenc-0.1.0/src/noentenc/language_detection/models/fasttext/model.py +266 -0
- noentenc-0.1.0/src/noentenc/language_detection/models/fasttext/tokenizer.py +167 -0
- noentenc-0.1.0/src/noentenc/language_detection/models/heliport_model.py +70 -0
- noentenc-0.1.0/src/noentenc/language_detection/models/langid_model.py +178 -0
- noentenc-0.1.0/src/noentenc/language_detection/models/lingua_model.py +86 -0
- noentenc-0.1.0/src/noentenc/language_detection/models/onnx_classifier.py +150 -0
- noentenc-0.1.0/src/noentenc/languages.py +274 -0
- noentenc-0.1.0/src/noentenc/translation/__init__.py +16 -0
- noentenc-0.1.0/src/noentenc/translation/base.py +123 -0
- noentenc-0.1.0/src/noentenc/translation/models/__init__.py +0 -0
- noentenc-0.1.0/src/noentenc/translation/models/_engine.py +172 -0
- noentenc-0.1.0/src/noentenc/translation/models/_hub.py +35 -0
- noentenc-0.1.0/src/noentenc/translation/models/_seq2seq.py +151 -0
- noentenc-0.1.0/src/noentenc/translation/models/base.py +30 -0
- noentenc-0.1.0/src/noentenc/translation/models/m2m100.py +70 -0
- noentenc-0.1.0/src/noentenc/translation/models/nllb.py +258 -0
- noentenc-0.1.0/src/noentenc/translation/models/opus_mt.py +222 -0
- noentenc-0.1.0/src/noentenc/translation/models/small100.py +30 -0
noentenc-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: noentenc
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Lightweight, CPU-only language detection and translation for Python. Extra detection backends are optional extras.
|
|
5
|
+
Author: marti-jorda-roca
|
|
6
|
+
Author-email: marti-jorda-roca <mjorda98@gmail.com>
|
|
7
|
+
Requires-Dist: huggingface-hub>=1.33.0
|
|
8
|
+
Requires-Dist: numpy>=2.0
|
|
9
|
+
Requires-Dist: onnxruntime>=1.30.0
|
|
10
|
+
Requires-Dist: tokenizers>=0.23.2
|
|
11
|
+
Requires-Dist: tqdm>=4.70.1
|
|
12
|
+
Requires-Dist: noentenc[pandas,polars,lingua,cld3,heliport] ; extra == 'all'
|
|
13
|
+
Requires-Dist: cld3-py>=3.1 ; extra == 'cld3'
|
|
14
|
+
Requires-Dist: heliport>=1.0.1 ; sys_platform != 'win32' and extra == 'heliport'
|
|
15
|
+
Requires-Dist: lingua-language-detector>=2.2 ; extra == 'lingua'
|
|
16
|
+
Requires-Dist: pandas>=3.0.6 ; extra == 'pandas'
|
|
17
|
+
Requires-Dist: polars>=1.44.2 ; extra == 'polars'
|
|
18
|
+
Requires-Python: >=3.14
|
|
19
|
+
Provides-Extra: all
|
|
20
|
+
Provides-Extra: cld3
|
|
21
|
+
Provides-Extra: heliport
|
|
22
|
+
Provides-Extra: lingua
|
|
23
|
+
Provides-Extra: pandas
|
|
24
|
+
Provides-Extra: polars
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
|
|
27
|
+
<div align="center">
|
|
28
|
+
<picture>
|
|
29
|
+
<source media="(prefers-color-scheme: dark)" srcset="docs/assets/logo-dark.svg">
|
|
30
|
+
<source media="(prefers-color-scheme: light)" srcset="docs/assets/logo-light.svg">
|
|
31
|
+
<img src="docs/assets/logo-light.svg" alt="noentenc" width="520">
|
|
32
|
+
</picture>
|
|
33
|
+
<p>Language detection and machine translation for Python, on CPU, without torch.</p>
|
|
34
|
+
</div>
|
|
35
|
+
|
|
36
|
+
```python
|
|
37
|
+
from noentenc import Language
|
|
38
|
+
from noentenc.language_detection import LanguageDetector
|
|
39
|
+
from noentenc.translation import Translator
|
|
40
|
+
|
|
41
|
+
LanguageDetector().detect("Bon dia! Com estàs?")
|
|
42
|
+
# 'cat'
|
|
43
|
+
|
|
44
|
+
Translator().translate("The weather is nice today.", Language.SPANISH, Language.ENGLISH)
|
|
45
|
+
# 'El tiempo es bueno hoy.'
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
- **Fast.** The default detector labels about 350,000 texts per second on a laptop CPU, and a direct Opus-MT translation takes about 60 ms per sentence. See [benchmarks](docs/benchmarks.md).
|
|
49
|
+
|
|
50
|
+
| Task | Model | Single input | Batched |
|
|
51
|
+
|---|---|---:|---:|
|
|
52
|
+
| Detection | heliport | 4 µs | 872k texts/s |
|
|
53
|
+
| Detection | **lid176 (default)** | 17 µs | 356k texts/s |
|
|
54
|
+
| Detection | cld3 | 16 µs | 66k texts/s |
|
|
55
|
+
| Detection | lingua | 0.44 ms | 9k texts/s |
|
|
56
|
+
| Detection | bert-openlid | 0.35 ms | 3.5k texts/s |
|
|
57
|
+
| Translation (en→es) | **Opus-MT (default for direct pairs)** | 63 ms | 100 sent/s |
|
|
58
|
+
| Translation (en→es) | **SMaLL-100 (default otherwise)** | 52 ms | 75 sent/s |
|
|
59
|
+
| Translation (en→es) | M2M100 418M | 231 ms | 15 sent/s |
|
|
60
|
+
| Translation (en→es) | NLLB-200 600M (int8) | 770 ms | 10 sent/s |
|
|
61
|
+
|
|
62
|
+
Apple M3, batches of 256 texts for detection and 32 sentences for translation.
|
|
63
|
+
|
|
64
|
+
- **Small.** The default detection model is 0.9 MB. noentenc has no torch, transformers or GPU dependency, only numpy, onnxruntime, tokenizers, huggingface-hub and tqdm.
|
|
65
|
+
- **One API, many models.** 12 detection models and 4 translation model families sit behind the same two classes. Swapping one is a one-line change, and every detector returns the same ISO 639-3 labels.
|
|
66
|
+
- **Built for datasets.** You can pass a single string, a list or a pandas or polars column.
|
|
67
|
+
|
|
68
|
+
## Install
|
|
69
|
+
|
|
70
|
+
noentenc uses [uv](https://docs.astral.sh/uv/) to manage dependencies.
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
uv add noentenc # fastText, ONNX and langid detection + every translation model
|
|
74
|
+
uv add 'noentenc[lingua,cld3]' # extra detection backends, see the table below
|
|
75
|
+
uv add 'noentenc[all]'
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Weights download on first use to `~/.cache/noentenc` (detection) and the Hugging Face cache (translation). Nothing is bundled in the wheel.
|
|
79
|
+
|
|
80
|
+
## Detect a language
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
from noentenc.language_detection import LanguageDetector
|
|
84
|
+
|
|
85
|
+
detector = LanguageDetector() # fastText lid.176: 176 languages, 0.9 MB
|
|
86
|
+
|
|
87
|
+
detector.detect("Bon dia! Com estàs?")
|
|
88
|
+
# 'cat'
|
|
89
|
+
detector.detect("Bon dia! Com estàs?", with_score=True, top_k=2)
|
|
90
|
+
# {'cat': 0.88, 'por': 0.11}
|
|
91
|
+
detector.detect_batch(["Hello there", "Hola, ¿qué tal?", "你好", ""])
|
|
92
|
+
# ['eng', 'spa', 'zho', 'und']
|
|
93
|
+
|
|
94
|
+
# Add a "lang" column to a pandas or polars DataFrame.
|
|
95
|
+
detector.detect_dataset(df, "text", "lang")
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
## Translate
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
from noentenc import Language
|
|
102
|
+
from noentenc.translation import Translator
|
|
103
|
+
|
|
104
|
+
translator = Translator()
|
|
105
|
+
|
|
106
|
+
translator.translate_batch(
|
|
107
|
+
["Where is the station?", "I love this city."], Language.SPANISH, Language.ENGLISH
|
|
108
|
+
)
|
|
109
|
+
# ['¿Dónde está la estación?', 'Me encanta esta ciudad.']
|
|
110
|
+
|
|
111
|
+
# The source language is optional.
|
|
112
|
+
translator.translate("Bon dia a tothom!", Language.ENGLISH)
|
|
113
|
+
# 'Good day to everyone!'
|
|
114
|
+
|
|
115
|
+
translator.translate_dataset(df, "review", "review_en", Language.ENGLISH)
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
`Translator()` picks the lightest model for each pair. It uses a dedicated Opus-MT model when one exists for the direction (66 directions, about 75M parameters each). For any other pair it uses SMaLL-100, which covers 100 languages. Languages are always `Language` enum members, so a typo fails at the call site, and a model that can't handle a pair raises `UnsupportedLanguageError`.
|
|
119
|
+
|
|
120
|
+
## Examples
|
|
121
|
+
|
|
122
|
+
| Example | Shows how to |
|
|
123
|
+
|---|---|
|
|
124
|
+
| [detect_dataset.py](docs/examples/detect_dataset.py) | Tag a DataFrame column and keep only confident English rows. |
|
|
125
|
+
| [translate_to_english.py](docs/examples/translate_to_english.py) | Detect each message's language, then batch-translate everything into English. |
|
|
126
|
+
| [choose_detection_backend.py](docs/examples/choose_detection_backend.py) | Swap backends, restrict candidate languages, collapse macrolanguages. |
|
|
127
|
+
| [choose_translation_model.py](docs/examples/choose_translation_model.py) | Pick a model, precision and thread count, and run offline. |
|
|
128
|
+
| [custom_models.py](docs/examples/custom_models.py) | Plug your own detector and translator into the same API. |
|
|
129
|
+
|
|
130
|
+
## Models
|
|
131
|
+
|
|
132
|
+
### Language detection
|
|
133
|
+
|
|
134
|
+
| Backend | `model=` | Install | Languages | Size | Licence (weights) |
|
|
135
|
+
|---|---|---|---|---|---|
|
|
136
|
+
| `FastTextModel` | `lid176` (default) | core | 176 | 0.9 MB | CC-BY-SA-3.0 |
|
|
137
|
+
| | `lid176-bin` | core | 176 | 126 MB | CC-BY-SA-3.0 |
|
|
138
|
+
| | `openlid-v2` | core | 200 | 1.2 GB | GPL-3.0 |
|
|
139
|
+
| | `openlid-v3` | core | 195 | 1.2 GB | GPL-3.0 |
|
|
140
|
+
| | `glotlid` | core | 2102 | 1.7 GB | Apache-2.0 |
|
|
141
|
+
| | `nllb-lid218e` | core | 218 | 1.2 GB | CC-BY-NC-4.0 |
|
|
142
|
+
| | path to a `.bin`/`.ftz` | core | | | |
|
|
143
|
+
| `OnnxClassifierModel` | `bert-openlid` (int8) | core | 201 | 25 MB | MIT |
|
|
144
|
+
| | `xlm-roberta-lid` (int8) | core | 20 | 279 MB | MIT |
|
|
145
|
+
| `LangidModel` | | core | 97 | 1.9 MB | BSD-2-Clause |
|
|
146
|
+
| `LinguaModel` | | `noentenc[lingua]` | 75 | ~300 MB wheel | Apache-2.0 |
|
|
147
|
+
| `Cld3Model` | | `noentenc[cld3]` | 107 | 1 MB | Apache-2.0 |
|
|
148
|
+
| `HeliportModel` | | `noentenc[heliport]` (no Windows) | 220 | ~130 MB wheel | GPL-3.0 |
|
|
149
|
+
|
|
150
|
+
```python
|
|
151
|
+
from noentenc.language_detection import FastTextModel, LanguageDetector, LinguaModel
|
|
152
|
+
|
|
153
|
+
LanguageDetector(FastTextModel("glotlid")) # 2102 languages
|
|
154
|
+
LanguageDetector(LinguaModel(languages=["cat", "spa", "eng"])) # only these candidates
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
- Labels are ISO 639-3 codes (`eng`, `cat`, `zho`). Empty and whitespace-only texts return `und`.
|
|
158
|
+
- `normalize_labels=False` returns each model's native codes instead.
|
|
159
|
+
- `collapse_macrolanguages=True` folds individual languages into their macrolanguage (`arb` → `ara`, `cmn` → `zho`), so results from different models line up.
|
|
160
|
+
- `FastTextModel` is our own numpy implementation of fastText inference. It needs neither the `fasttext` package nor onnxruntime, and it matches `fasttext`'s output to within 1e-6. Large `.bin` models are memory-mapped, so they open instantly.
|
|
161
|
+
- `LangidModel` is our own numpy implementation of [langid.py](https://github.com/saffsd/langid.py). It reads the weights from the langid 1.1.6 source release on PyPI, without installing or importing the `langid` package, and matches its probabilities to within 1e-9.
|
|
162
|
+
|
|
163
|
+
### Translation
|
|
164
|
+
|
|
165
|
+
| Model | Languages | Download (default precision) | Licence (weights) |
|
|
166
|
+
|---|---|---|---|
|
|
167
|
+
| `OpusMTModel.from_pair(src, tgt)` | 1 direction each, 66 available (`OPUS_MT_PAIRS`) | 287 MB (q4), 107 MB (int8) | CC-BY-4.0 or Apache-2.0, per pair |
|
|
168
|
+
| `SMaLL100Model` | any source → 100 targets | 595 MB (int8) | MIT |
|
|
169
|
+
| `M2M100Model` | 100 ↔ 100 | 1.2 GB (q4), 603 MB (int8) | MIT |
|
|
170
|
+
| `NLLBModel` | 196 ↔ 196 | 860 MB (int8) | CC-BY-NC-4.0, never picked by default |
|
|
171
|
+
|
|
172
|
+
Every translation model takes `precision=` (`fp32`, `int8`, `q4`, where the export has it), `num_threads=` and a Hugging Face repo id or local directory as `model=`.
|
|
173
|
+
|
|
174
|
+
### Weights and licences
|
|
175
|
+
|
|
176
|
+
Weights are pinned to a revision and, where we download them ourselves, checked against a sha256. Set `NOENTENC_CACHE` to change the detection cache location. With `only_local_files=True`, a model that isn't cached raises instead of downloading, which is what you want on an air-gapped server.
|
|
177
|
+
|
|
178
|
+
The weights' licences apply to your use of the weights. They don't affect this package's licence.
|
|
179
|
+
|
|
180
|
+
## Bring your own model
|
|
181
|
+
|
|
182
|
+
Load your own fastText, ONNX classifier or seq2seq weights into an existing backend, or wrap any detector or translator by subclassing a base class with a few methods. See [docs/custom-models.md](docs/custom-models.md).
|
|
183
|
+
|
|
184
|
+
```python
|
|
185
|
+
from noentenc.language_detection import FastTextModel, LanguageDetector
|
|
186
|
+
|
|
187
|
+
LanguageDetector(FastTextModel("models/my-domain-lid.ftz"))
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
## Development
|
|
191
|
+
|
|
192
|
+
```bash
|
|
193
|
+
make install # uv sync with every extra + dev tools
|
|
194
|
+
make lint
|
|
195
|
+
make unit-tests
|
|
196
|
+
make integration-tests # tests/integrations; downloads real weights
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
`scripts/benchmark_lid.py` and `scripts/benchmark_translation.py` reproduce the speed numbers. `scripts/make_fasttext_fixtures.py` regenerates the fastText parity fixtures with the reference `fasttext` package. That package needs Python 3.12, and the script's docstring has the command. `scripts/make_langid_fixtures.py` does the same for langid with the reference `langid` package.
|
noentenc-0.1.0/README.md
ADDED
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
<div align="center">
|
|
2
|
+
<picture>
|
|
3
|
+
<source media="(prefers-color-scheme: dark)" srcset="docs/assets/logo-dark.svg">
|
|
4
|
+
<source media="(prefers-color-scheme: light)" srcset="docs/assets/logo-light.svg">
|
|
5
|
+
<img src="docs/assets/logo-light.svg" alt="noentenc" width="520">
|
|
6
|
+
</picture>
|
|
7
|
+
<p>Language detection and machine translation for Python, on CPU, without torch.</p>
|
|
8
|
+
</div>
|
|
9
|
+
|
|
10
|
+
```python
|
|
11
|
+
from noentenc import Language
|
|
12
|
+
from noentenc.language_detection import LanguageDetector
|
|
13
|
+
from noentenc.translation import Translator
|
|
14
|
+
|
|
15
|
+
LanguageDetector().detect("Bon dia! Com estàs?")
|
|
16
|
+
# 'cat'
|
|
17
|
+
|
|
18
|
+
Translator().translate("The weather is nice today.", Language.SPANISH, Language.ENGLISH)
|
|
19
|
+
# 'El tiempo es bueno hoy.'
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
- **Fast.** The default detector labels about 350,000 texts per second on a laptop CPU, and a direct Opus-MT translation takes about 60 ms per sentence. See [benchmarks](docs/benchmarks.md).
|
|
23
|
+
|
|
24
|
+
| Task | Model | Single input | Batched |
|
|
25
|
+
|---|---|---:|---:|
|
|
26
|
+
| Detection | heliport | 4 µs | 872k texts/s |
|
|
27
|
+
| Detection | **lid176 (default)** | 17 µs | 356k texts/s |
|
|
28
|
+
| Detection | cld3 | 16 µs | 66k texts/s |
|
|
29
|
+
| Detection | lingua | 0.44 ms | 9k texts/s |
|
|
30
|
+
| Detection | bert-openlid | 0.35 ms | 3.5k texts/s |
|
|
31
|
+
| Translation (en→es) | **Opus-MT (default for direct pairs)** | 63 ms | 100 sent/s |
|
|
32
|
+
| Translation (en→es) | **SMaLL-100 (default otherwise)** | 52 ms | 75 sent/s |
|
|
33
|
+
| Translation (en→es) | M2M100 418M | 231 ms | 15 sent/s |
|
|
34
|
+
| Translation (en→es) | NLLB-200 600M (int8) | 770 ms | 10 sent/s |
|
|
35
|
+
|
|
36
|
+
Apple M3, batches of 256 texts for detection and 32 sentences for translation.
|
|
37
|
+
|
|
38
|
+
- **Small.** The default detection model is 0.9 MB. noentenc has no torch, transformers or GPU dependency, only numpy, onnxruntime, tokenizers, huggingface-hub and tqdm.
|
|
39
|
+
- **One API, many models.** 12 detection models and 4 translation model families sit behind the same two classes. Swapping one is a one-line change, and every detector returns the same ISO 639-3 labels.
|
|
40
|
+
- **Built for datasets.** You can pass a single string, a list or a pandas or polars column.
|
|
41
|
+
|
|
42
|
+
## Install
|
|
43
|
+
|
|
44
|
+
noentenc uses [uv](https://docs.astral.sh/uv/) to manage dependencies.
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
uv add noentenc # fastText, ONNX and langid detection + every translation model
|
|
48
|
+
uv add 'noentenc[lingua,cld3]' # extra detection backends, see the table below
|
|
49
|
+
uv add 'noentenc[all]'
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Weights download on first use to `~/.cache/noentenc` (detection) and the Hugging Face cache (translation). Nothing is bundled in the wheel.
|
|
53
|
+
|
|
54
|
+
## Detect a language
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
from noentenc.language_detection import LanguageDetector
|
|
58
|
+
|
|
59
|
+
detector = LanguageDetector() # fastText lid.176: 176 languages, 0.9 MB
|
|
60
|
+
|
|
61
|
+
detector.detect("Bon dia! Com estàs?")
|
|
62
|
+
# 'cat'
|
|
63
|
+
detector.detect("Bon dia! Com estàs?", with_score=True, top_k=2)
|
|
64
|
+
# {'cat': 0.88, 'por': 0.11}
|
|
65
|
+
detector.detect_batch(["Hello there", "Hola, ¿qué tal?", "你好", ""])
|
|
66
|
+
# ['eng', 'spa', 'zho', 'und']
|
|
67
|
+
|
|
68
|
+
# Add a "lang" column to a pandas or polars DataFrame.
|
|
69
|
+
detector.detect_dataset(df, "text", "lang")
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## Translate
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
from noentenc import Language
|
|
76
|
+
from noentenc.translation import Translator
|
|
77
|
+
|
|
78
|
+
translator = Translator()
|
|
79
|
+
|
|
80
|
+
translator.translate_batch(
|
|
81
|
+
["Where is the station?", "I love this city."], Language.SPANISH, Language.ENGLISH
|
|
82
|
+
)
|
|
83
|
+
# ['¿Dónde está la estación?', 'Me encanta esta ciudad.']
|
|
84
|
+
|
|
85
|
+
# The source language is optional.
|
|
86
|
+
translator.translate("Bon dia a tothom!", Language.ENGLISH)
|
|
87
|
+
# 'Good day to everyone!'
|
|
88
|
+
|
|
89
|
+
translator.translate_dataset(df, "review", "review_en", Language.ENGLISH)
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
`Translator()` picks the lightest model for each pair. It uses a dedicated Opus-MT model when one exists for the direction (66 directions, about 75M parameters each). For any other pair it uses SMaLL-100, which covers 100 languages. Languages are always `Language` enum members, so a typo fails at the call site, and a model that can't handle a pair raises `UnsupportedLanguageError`.
|
|
93
|
+
|
|
94
|
+
## Examples
|
|
95
|
+
|
|
96
|
+
| Example | Shows how to |
|
|
97
|
+
|---|---|
|
|
98
|
+
| [detect_dataset.py](docs/examples/detect_dataset.py) | Tag a DataFrame column and keep only confident English rows. |
|
|
99
|
+
| [translate_to_english.py](docs/examples/translate_to_english.py) | Detect each message's language, then batch-translate everything into English. |
|
|
100
|
+
| [choose_detection_backend.py](docs/examples/choose_detection_backend.py) | Swap backends, restrict candidate languages, collapse macrolanguages. |
|
|
101
|
+
| [choose_translation_model.py](docs/examples/choose_translation_model.py) | Pick a model, precision and thread count, and run offline. |
|
|
102
|
+
| [custom_models.py](docs/examples/custom_models.py) | Plug your own detector and translator into the same API. |
|
|
103
|
+
|
|
104
|
+
## Models
|
|
105
|
+
|
|
106
|
+
### Language detection
|
|
107
|
+
|
|
108
|
+
| Backend | `model=` | Install | Languages | Size | Licence (weights) |
|
|
109
|
+
|---|---|---|---|---|---|
|
|
110
|
+
| `FastTextModel` | `lid176` (default) | core | 176 | 0.9 MB | CC-BY-SA-3.0 |
|
|
111
|
+
| | `lid176-bin` | core | 176 | 126 MB | CC-BY-SA-3.0 |
|
|
112
|
+
| | `openlid-v2` | core | 200 | 1.2 GB | GPL-3.0 |
|
|
113
|
+
| | `openlid-v3` | core | 195 | 1.2 GB | GPL-3.0 |
|
|
114
|
+
| | `glotlid` | core | 2102 | 1.7 GB | Apache-2.0 |
|
|
115
|
+
| | `nllb-lid218e` | core | 218 | 1.2 GB | CC-BY-NC-4.0 |
|
|
116
|
+
| | path to a `.bin`/`.ftz` | core | | | |
|
|
117
|
+
| `OnnxClassifierModel` | `bert-openlid` (int8) | core | 201 | 25 MB | MIT |
|
|
118
|
+
| | `xlm-roberta-lid` (int8) | core | 20 | 279 MB | MIT |
|
|
119
|
+
| `LangidModel` | | core | 97 | 1.9 MB | BSD-2-Clause |
|
|
120
|
+
| `LinguaModel` | | `noentenc[lingua]` | 75 | ~300 MB wheel | Apache-2.0 |
|
|
121
|
+
| `Cld3Model` | | `noentenc[cld3]` | 107 | 1 MB | Apache-2.0 |
|
|
122
|
+
| `HeliportModel` | | `noentenc[heliport]` (no Windows) | 220 | ~130 MB wheel | GPL-3.0 |
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
from noentenc.language_detection import FastTextModel, LanguageDetector, LinguaModel
|
|
126
|
+
|
|
127
|
+
LanguageDetector(FastTextModel("glotlid")) # 2102 languages
|
|
128
|
+
LanguageDetector(LinguaModel(languages=["cat", "spa", "eng"])) # only these candidates
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
- Labels are ISO 639-3 codes (`eng`, `cat`, `zho`). Empty and whitespace-only texts return `und`.
|
|
132
|
+
- `normalize_labels=False` returns each model's native codes instead.
|
|
133
|
+
- `collapse_macrolanguages=True` folds individual languages into their macrolanguage (`arb` → `ara`, `cmn` → `zho`), so results from different models line up.
|
|
134
|
+
- `FastTextModel` is our own numpy implementation of fastText inference. It needs neither the `fasttext` package nor onnxruntime, and it matches `fasttext`'s output to within 1e-6. Large `.bin` models are memory-mapped, so they open instantly.
|
|
135
|
+
- `LangidModel` is our own numpy implementation of [langid.py](https://github.com/saffsd/langid.py). It reads the weights from the langid 1.1.6 source release on PyPI, without installing or importing the `langid` package, and matches its probabilities to within 1e-9.
|
|
136
|
+
|
|
137
|
+
### Translation
|
|
138
|
+
|
|
139
|
+
| Model | Languages | Download (default precision) | Licence (weights) |
|
|
140
|
+
|---|---|---|---|
|
|
141
|
+
| `OpusMTModel.from_pair(src, tgt)` | 1 direction each, 66 available (`OPUS_MT_PAIRS`) | 287 MB (q4), 107 MB (int8) | CC-BY-4.0 or Apache-2.0, per pair |
|
|
142
|
+
| `SMaLL100Model` | any source → 100 targets | 595 MB (int8) | MIT |
|
|
143
|
+
| `M2M100Model` | 100 ↔ 100 | 1.2 GB (q4), 603 MB (int8) | MIT |
|
|
144
|
+
| `NLLBModel` | 196 ↔ 196 | 860 MB (int8) | CC-BY-NC-4.0, never picked by default |
|
|
145
|
+
|
|
146
|
+
Every translation model takes `precision=` (`fp32`, `int8`, `q4`, where the export has it), `num_threads=` and a Hugging Face repo id or local directory as `model=`.
|
|
147
|
+
|
|
148
|
+
### Weights and licences
|
|
149
|
+
|
|
150
|
+
Weights are pinned to a revision and, where we download them ourselves, checked against a sha256. Set `NOENTENC_CACHE` to change the detection cache location. With `only_local_files=True`, a model that isn't cached raises instead of downloading, which is what you want on an air-gapped server.
|
|
151
|
+
|
|
152
|
+
The weights' licences apply to your use of the weights. They don't affect this package's licence.
|
|
153
|
+
|
|
154
|
+
## Bring your own model
|
|
155
|
+
|
|
156
|
+
Load your own fastText, ONNX classifier or seq2seq weights into an existing backend, or wrap any detector or translator by subclassing a base class with a few methods. See [docs/custom-models.md](docs/custom-models.md).
|
|
157
|
+
|
|
158
|
+
```python
|
|
159
|
+
from noentenc.language_detection import FastTextModel, LanguageDetector
|
|
160
|
+
|
|
161
|
+
LanguageDetector(FastTextModel("models/my-domain-lid.ftz"))
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
## Development
|
|
165
|
+
|
|
166
|
+
```bash
|
|
167
|
+
make install # uv sync with every extra + dev tools
|
|
168
|
+
make lint
|
|
169
|
+
make unit-tests
|
|
170
|
+
make integration-tests # tests/integrations; downloads real weights
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
`scripts/benchmark_lid.py` and `scripts/benchmark_translation.py` reproduce the speed numbers. `scripts/make_fasttext_fixtures.py` regenerates the fastText parity fixtures with the reference `fasttext` package. That package needs Python 3.12, and the script's docstring has the command. `scripts/make_langid_fixtures.py` does the same for langid with the reference `langid` package.
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "noentenc"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Lightweight, CPU-only language detection and translation for Python. Extra detection backends are optional extras."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.14"
|
|
7
|
+
dependencies = [
|
|
8
|
+
"huggingface-hub>=1.33.0",
|
|
9
|
+
"numpy>=2.0",
|
|
10
|
+
"onnxruntime>=1.30.0",
|
|
11
|
+
"tokenizers>=0.23.2",
|
|
12
|
+
"tqdm>=4.70.1",
|
|
13
|
+
]
|
|
14
|
+
|
|
15
|
+
[[project.authors]]
|
|
16
|
+
name = "marti-jorda-roca"
|
|
17
|
+
email = "mjorda98@gmail.com"
|
|
18
|
+
|
|
19
|
+
[project.optional-dependencies]
|
|
20
|
+
pandas = ["pandas>=3.0.6"]
|
|
21
|
+
polars = ["polars>=1.44.2"]
|
|
22
|
+
lingua = ["lingua-language-detector>=2.2"]
|
|
23
|
+
cld3 = ["cld3-py>=3.1"]
|
|
24
|
+
heliport = ["heliport>=1.0.1; sys_platform != 'win32'"]
|
|
25
|
+
all = ["noentenc[pandas,polars,lingua,cld3,heliport]"]
|
|
26
|
+
|
|
27
|
+
[build-system]
|
|
28
|
+
requires = ["uv_build>=0.12.22,<0.13.0"]
|
|
29
|
+
build-backend = "uv_build"
|
|
30
|
+
|
|
31
|
+
[dependency-groups]
|
|
32
|
+
dev = [
|
|
33
|
+
"bandit[toml]>=1.9.4",
|
|
34
|
+
"coverage>=7.16.2",
|
|
35
|
+
"library-skills>=0.0.19",
|
|
36
|
+
"pandas>=3.0.6",
|
|
37
|
+
"polars>=1.44.2",
|
|
38
|
+
"prek>=0.5.4",
|
|
39
|
+
"pytest>=9.1.1",
|
|
40
|
+
"ruff>=0.16.10",
|
|
41
|
+
"ty>=0.0.84",
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
[tool.ruff]
|
|
45
|
+
target-version = "py314"
|
|
46
|
+
|
|
47
|
+
[tool.ruff.lint]
|
|
48
|
+
select = [
|
|
49
|
+
"E",
|
|
50
|
+
"W",
|
|
51
|
+
"F",
|
|
52
|
+
"I",
|
|
53
|
+
"B",
|
|
54
|
+
"C4",
|
|
55
|
+
"UP",
|
|
56
|
+
"ARG001",
|
|
57
|
+
"T201",
|
|
58
|
+
"C90",
|
|
59
|
+
"PLR",
|
|
60
|
+
"ERA",
|
|
61
|
+
"SIM",
|
|
62
|
+
"RET",
|
|
63
|
+
"PIE",
|
|
64
|
+
"B",
|
|
65
|
+
"ANN",
|
|
66
|
+
]
|
|
67
|
+
ignore = [
|
|
68
|
+
"E501",
|
|
69
|
+
"B008",
|
|
70
|
+
"W191",
|
|
71
|
+
"B904",
|
|
72
|
+
"PLR0913",
|
|
73
|
+
]
|
|
74
|
+
|
|
75
|
+
[tool.ruff.lint.per-file-ignores]
|
|
76
|
+
"docs/examples/**/*.py" = [
|
|
77
|
+
"PLR2004",
|
|
78
|
+
"T201",
|
|
79
|
+
"ERA001",
|
|
80
|
+
]
|
|
81
|
+
"scripts/**/*.py" = [
|
|
82
|
+
"PLR2004",
|
|
83
|
+
"T201",
|
|
84
|
+
]
|
|
85
|
+
"tests/**/*.py" = ["PLR2004"]
|
|
86
|
+
|
|
87
|
+
[tool.ruff.lint.pydocstyle]
|
|
88
|
+
convention = "google"
|
|
89
|
+
|
|
90
|
+
[tool.ruff.lint.mccabe]
|
|
91
|
+
max-complexity = 10
|
|
92
|
+
|
|
93
|
+
[tool.ruff.lint.pylint]
|
|
94
|
+
max-branches = 12
|
|
95
|
+
max-returns = 6
|
|
96
|
+
max-statements = 50
|
|
97
|
+
max-args = 8
|
|
98
|
+
max-nested-blocks = 5
|
|
99
|
+
|
|
100
|
+
[tool.ty.terminal]
|
|
101
|
+
error-on-warning = true
|
|
102
|
+
|
|
103
|
+
[tool.ty.src]
|
|
104
|
+
include = [
|
|
105
|
+
"src",
|
|
106
|
+
"tests",
|
|
107
|
+
"docs",
|
|
108
|
+
]
|
|
109
|
+
|
|
110
|
+
[tool.pytest.ini_options]
|
|
111
|
+
testpaths = ["tests"]
|
|
112
|
+
pythonpath = ["."]
|
|
113
|
+
addopts = "--import-mode=importlib"
|
|
114
|
+
consider_namespace_packages = true
|
|
115
|
+
markers = ["integration: needs real model weights (downloads them); run with NOENTENC_RUN_INTEGRATION=1"]
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "noentenc"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Lightweight, CPU-only language detection and translation for Python. Extra detection backends are optional extras."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
authors = [
|
|
7
|
+
{ name = "marti-jorda-roca", email = "mjorda98@gmail.com" }
|
|
8
|
+
]
|
|
9
|
+
requires-python = ">=3.14"
|
|
10
|
+
dependencies = [
|
|
11
|
+
"huggingface-hub>=1.33.0",
|
|
12
|
+
"numpy>=2.0",
|
|
13
|
+
"onnxruntime>=1.30.0",
|
|
14
|
+
"tokenizers>=0.23.2",
|
|
15
|
+
"tqdm>=4.70.1",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
[project.optional-dependencies]
|
|
19
|
+
pandas = ["pandas>=3.0.6"]
|
|
20
|
+
polars = ["polars>=1.44.2"]
|
|
21
|
+
lingua = ["lingua-language-detector>=2.2"]
|
|
22
|
+
cld3 = ["cld3-py>=3.1"]
|
|
23
|
+
heliport = ["heliport>=1.0.1; sys_platform != 'win32'"]
|
|
24
|
+
all = ["noentenc[pandas,polars,lingua,cld3,heliport]"]
|
|
25
|
+
|
|
26
|
+
[build-system]
|
|
27
|
+
requires = ["uv_build>=0.12.22,<0.13.0"]
|
|
28
|
+
build-backend = "uv_build"
|
|
29
|
+
|
|
30
|
+
[dependency-groups]
|
|
31
|
+
dev = [
|
|
32
|
+
"bandit[toml]>=1.9.4",
|
|
33
|
+
"coverage>=7.16.2",
|
|
34
|
+
"library-skills>=0.0.19",
|
|
35
|
+
"pandas>=3.0.6",
|
|
36
|
+
"polars>=1.44.2",
|
|
37
|
+
"prek>=0.5.4",
|
|
38
|
+
"pytest>=9.1.1",
|
|
39
|
+
"ruff>=0.16.10",
|
|
40
|
+
"ty>=0.0.84",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
[tool.ruff]
|
|
45
|
+
target-version = "py314"
|
|
46
|
+
|
|
47
|
+
[tool.ruff.lint]
|
|
48
|
+
select = [
|
|
49
|
+
"E", # pycodestyle errors
|
|
50
|
+
"W", # pycodestyle warnings
|
|
51
|
+
"F", # pyflakes
|
|
52
|
+
"I", # isort
|
|
53
|
+
"B", # flake8-bugbear
|
|
54
|
+
"C4", # flake8-comprehensions
|
|
55
|
+
"UP", # pyupgrade
|
|
56
|
+
"ARG001", # unused arguments in functions
|
|
57
|
+
"T201", # print statements are not allowed,
|
|
58
|
+
"C90", # cyclomatic complexity
|
|
59
|
+
"PLR", # pylint-style size limits
|
|
60
|
+
"ERA", # Commented-out code
|
|
61
|
+
"SIM", # Simplifiable/redundant constructs
|
|
62
|
+
"RET", # Redundant return patterns
|
|
63
|
+
"PIE", # Miscellaneous unnecessary code
|
|
64
|
+
"B", # Bugbear, including duplicate/redundant constructs
|
|
65
|
+
"ANN", # Missing annotations and explicit Any checks
|
|
66
|
+
]
|
|
67
|
+
ignore = [
|
|
68
|
+
"E501", # line too long, handled by black
|
|
69
|
+
"B008", # do not perform function calls in argument defaults
|
|
70
|
+
"W191", # indentation contains tabs
|
|
71
|
+
"B904", # Allow raising exceptions without from e
|
|
72
|
+
"PLR0913", # Too many arguments
|
|
73
|
+
]
|
|
74
|
+
|
|
75
|
+
[tool.ruff.lint.per-file-ignores]
|
|
76
|
+
# Magic values are the point in tests and one-off scripts; only library code must name them.
|
|
77
|
+
# Examples print results and show expected output in comments.
|
|
78
|
+
"docs/examples/**/*.py" = ["PLR2004", "T201", "ERA001"]
|
|
79
|
+
"scripts/**/*.py" = ["PLR2004", "T201"]
|
|
80
|
+
"tests/**/*.py" = ["PLR2004"]
|
|
81
|
+
|
|
82
|
+
[tool.ruff.lint.pydocstyle]
|
|
83
|
+
convention = "google"
|
|
84
|
+
|
|
85
|
+
[tool.ruff.lint.mccabe]
|
|
86
|
+
max-complexity = 10
|
|
87
|
+
|
|
88
|
+
[tool.ruff.lint.pylint]
|
|
89
|
+
max-branches = 12
|
|
90
|
+
max-returns = 6
|
|
91
|
+
max-statements = 50
|
|
92
|
+
max-args = 8
|
|
93
|
+
max-nested-blocks = 5
|
|
94
|
+
|
|
95
|
+
[tool.ty.terminal]
|
|
96
|
+
error-on-warning = true
|
|
97
|
+
|
|
98
|
+
[tool.ty.src]
|
|
99
|
+
include = ["src", "tests", "docs"]
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
[tool.pytest.ini_options]
|
|
103
|
+
testpaths = ["tests"]
|
|
104
|
+
pythonpath = ["."]
|
|
105
|
+
# Projects share test module basenames (e.g. test_train_model.py); importlib mode
|
|
106
|
+
# gives each file a unique module name instead of colliding on sys.modules.
|
|
107
|
+
addopts = "--import-mode=importlib"
|
|
108
|
+
consider_namespace_packages = true
|
|
109
|
+
markers = [
|
|
110
|
+
"integration: needs real model weights (downloads them); run with NOENTENC_RUN_INTEGRATION=1",
|
|
111
|
+
]
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
from typing import TYPE_CHECKING, Any, cast
|
|
2
|
+
|
|
3
|
+
if TYPE_CHECKING:
|
|
4
|
+
import pandas as pd
|
|
5
|
+
import polars as pl
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def column_values(dataset: pl.DataFrame | pd.DataFrame, column: str) -> list[Any]:
|
|
9
|
+
_frame_module(dataset)
|
|
10
|
+
return dataset[column].to_list()
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def with_column(
|
|
14
|
+
dataset: pl.DataFrame | pd.DataFrame,
|
|
15
|
+
column: str,
|
|
16
|
+
values: list[Any],
|
|
17
|
+
*,
|
|
18
|
+
strings: bool = False,
|
|
19
|
+
) -> pl.DataFrame | pd.DataFrame:
|
|
20
|
+
"""Return `dataset` with `column` set to `values`.
|
|
21
|
+
|
|
22
|
+
`strings` types a polars column as String even when every value is null.
|
|
23
|
+
"""
|
|
24
|
+
if _frame_module(dataset) == "polars":
|
|
25
|
+
# The caller passed a polars frame, so polars is installed.
|
|
26
|
+
import polars
|
|
27
|
+
|
|
28
|
+
frame = cast("pl.DataFrame", dataset)
|
|
29
|
+
return frame.with_columns(
|
|
30
|
+
polars.Series(column, values, polars.String if strings else None)
|
|
31
|
+
)
|
|
32
|
+
return cast("pd.DataFrame", dataset).assign(**{column: values})
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _frame_module(dataset: object) -> str:
|
|
36
|
+
module = type(dataset).__module__.split(".", 1)[0]
|
|
37
|
+
if module not in {"polars", "pandas"}:
|
|
38
|
+
raise TypeError(
|
|
39
|
+
f"expected a polars or pandas DataFrame, got {type(dataset).__name__}"
|
|
40
|
+
)
|
|
41
|
+
return module
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import onnxruntime as ort
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def create_session(path: Path, num_threads: int | None) -> ort.InferenceSession:
|
|
8
|
+
options = ort.SessionOptions()
|
|
9
|
+
options.graph_optimization_level = ort.GraphOptimizationLevel.ORT_ENABLE_ALL
|
|
10
|
+
options.execution_mode = ort.ExecutionMode.ORT_SEQUENTIAL
|
|
11
|
+
options.inter_op_num_threads = 1
|
|
12
|
+
if num_threads is not None:
|
|
13
|
+
options.intra_op_num_threads = num_threads
|
|
14
|
+
return ort.InferenceSession(str(path), options, providers=["CPUExecutionProvider"])
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def pad(sequences: list[list[int]], pad_id: int) -> tuple[np.ndarray, np.ndarray]:
|
|
18
|
+
"""Right-pad token id sequences into `(input_ids, attention_mask)`."""
|
|
19
|
+
width = max(len(s) for s in sequences)
|
|
20
|
+
input_ids = np.full((len(sequences), width), pad_id, dtype=np.int64)
|
|
21
|
+
attention_mask = np.zeros((len(sequences), width), dtype=np.int64)
|
|
22
|
+
for row, sequence in enumerate(sequences):
|
|
23
|
+
input_ids[row, : len(sequence)] = sequence
|
|
24
|
+
attention_mask[row, : len(sequence)] = 1
|
|
25
|
+
return input_ids, attention_mask
|