laghu 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
laghu-0.1.0/.gitignore ADDED
@@ -0,0 +1,158 @@
1
+ _docs/
2
+ _proc/
3
+
4
+ *.bak
5
+ .gitattributes
6
+ .last_checked
7
+ .gitconfig
8
+ *.bak
9
+ *.log
10
+ *~
11
+ ~*
12
+ _tmp*
13
+ tmp*
14
+ tags
15
+ *.pkg
16
+
17
+ # Byte-compiled / optimized / DLL files
18
+ __pycache__/
19
+ *.py[cod]
20
+ *$py.class
21
+
22
+ # C extensions
23
+ *.so
24
+
25
+ # Distribution / packaging
26
+ .Python
27
+ env/
28
+ build/
29
+ conda/
30
+ develop-eggs/
31
+ dist/
32
+ downloads/
33
+ eggs/
34
+ .eggs/
35
+ lib/
36
+ lib64/
37
+ parts/
38
+ sdist/
39
+ var/
40
+ wheels/
41
+ *.egg-info/
42
+ .installed.cfg
43
+ *.egg
44
+
45
+ # PyInstaller
46
+ # Usually these files are written by a python script from a template
47
+ # before PyInstaller builds the exe, so as to inject date/other infos into it.
48
+ *.manifest
49
+ *.spec
50
+
51
+ # Installer logs
52
+ pip-log.txt
53
+ pip-delete-this-directory.txt
54
+
55
+ # Unit test / coverage reports
56
+ htmlcov/
57
+ .tox/
58
+ .coverage
59
+ .coverage.*
60
+ .cache
61
+ nosetests.xml
62
+ coverage.xml
63
+ *.cover
64
+ .hypothesis/
65
+
66
+ # Translations
67
+ *.mo
68
+ *.pot
69
+
70
+ # Django stuff:
71
+ *.log
72
+ local_settings.py
73
+
74
+ # Flask stuff:
75
+ instance/
76
+ .webassets-cache
77
+
78
+ # Scrapy stuff:
79
+ .scrapy
80
+
81
+ # Sphinx documentation
82
+ docs/_build/
83
+
84
+ # PyBuilder
85
+ target/
86
+
87
+ # Jupyter Notebook
88
+ .ipynb_checkpoints
89
+
90
+ # pyenv
91
+ .python-version
92
+
93
+ # celery beat schedule file
94
+ celerybeat-schedule
95
+
96
+ # SageMath parsed files
97
+ *.sage.py
98
+
99
+ # dotenv
100
+ .env
101
+
102
+ # virtualenv
103
+ .venv
104
+ venv/
105
+ ENV/
106
+
107
+ # Spyder project settings
108
+ .spyderproject
109
+ .spyproject
110
+
111
+ # Rope project settings
112
+ .ropeproject
113
+
114
+ # mkdocs documentation
115
+ /site
116
+
117
+ # mypy
118
+ .mypy_cache/
119
+
120
+ .vscode
121
+ *.swp
122
+
123
+ # osx generated files
124
+ .DS_Store
125
+ .DS_Store?
126
+ .Trashes
127
+ ehthumbs.db
128
+ Thumbs.db
129
+ .idea
130
+
131
+ # pytest
132
+ .pytest_cache
133
+
134
+ # tools/trust-doc-nbs
135
+ docs_src/.last_checked
136
+
137
+ # symlinks to fastai
138
+ docs_src/fastai
139
+ tools/fastai
140
+
141
+ # link checker
142
+ checklink/cookies.txt
143
+
144
+ # .gitconfig is now autogenerated
145
+ .gitconfig
146
+
147
+ # Quarto installer
148
+ .deb
149
+ .pkg
150
+
151
+ # Quarto
152
+ .quarto
153
+
154
+ .venv/
155
+ .kosha/
156
+ _docs/
157
+ _proc/
158
+ nbs/_fixtures/
laghu-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Karthik
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
laghu-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,142 @@
1
+ Metadata-Version: 2.5
2
+ Name: laghu
3
+ Version: 0.1.0
4
+ Summary: model2vec static models in a fraction of the memory, with the same vectors
5
+ Project-URL: Repository, https://github.com/vedicreader/laghu
6
+ Project-URL: Documentation, https://vedicreader.github.io/laghu/
7
+ Author-email: Karthik <karthik.rajgopal@hotmail.com>
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: embeddings,mmap,model2vec,nbdev,static-embeddings,tokenizers,unigram
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3 :: Only
15
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
16
+ Requires-Python: >=3.10
17
+ Requires-Dist: fastcore>=1.8
18
+ Requires-Dist: huggingface-hub>=0.24
19
+ Requires-Dist: numpy>=1.24
20
+ Requires-Dist: tokenizers>=0.20
21
+ Provides-Extra: fallback
22
+ Requires-Dist: model2vec>=0.9; extra == 'fallback'
23
+ Description-Content-Type: text/markdown
24
+
25
+ # laghu
26
+
27
+
28
+ <!-- WARNING: THIS FILE WAS AUTOGENERATED! DO NOT EDIT! -->
29
+
30
+ laghu (Sanskrit for light) loads a [model2vec](https://github.com/MinishLab/model2vec) static embedding model with `encode`, `encode_as_sequence` and `tokenize`. The output matches model2vec’s bits, apart from the empty-input difference documented below.
31
+
32
+ model2vec reads the embedding table into each process. For potion-multilingual-128M, the table takes 512 MB. Its Unigram tokenizer adds a character trie over 500,353 pieces, about another 610 MB. After encoding, the process uses about 1.3 GB of private memory.
33
+
34
+ laghu maps the table read-only from `model.safetensors`. Processes share the mapped pages through the operating system’s page cache. The operating system can reclaim clean pages under memory pressure. For supported Unigram models, laghu replaces the trie with Python Viterbi over a dict of pieces. The normalizer and pre-tokenizer come from `tokenizers`, built from the same tokenizer.json with the vocabulary removed. The same model then takes about 135 MB of private memory per process.
35
+
36
+ WordPiece, BPE and word-level models still use `tokenizers`. laghu maps their embedding tables but does not replace their tokenizers. The memory saving depends on the table’s size relative to the tokenizer.
37
+
38
+ ## Install
39
+
40
+ ``` sh
41
+ uv add laghu # or: pip install laghu
42
+ ```
43
+
44
+ model2vec itself is not a dependency. For the fallback described below, and for [`same_as_model2vec`](https://vedicreader.github.io/laghu/encoder.html#same_as_model2vec), install the extra:
45
+ `uv add 'laghu[fallback]'`.
46
+
47
+ ## Use
48
+
49
+ ``` python
50
+ from laghu import load
51
+
52
+ m = load('minishlab/potion-multilingual-128M') # a folder, or a Hugging Face repo id (the local cache is tried first)
53
+ v = m.encode(['The quick brown fox.', 'कर्मण्येवाधिकारस्ते']) # float32, (2, 256), unit rows: the same array model2vec returns
54
+ m.tokenize(['hello world'])
55
+ m.encode_as_sequence('hello world').shape # (n_tokens, 256)
56
+ ```
57
+
58
+ (2, 256)
59
+
60
+ [`load`](https://vedicreader.github.io/laghu/encoder.html#load) returns a [`LeanModel`](https://vedicreader.github.io/laghu/encoder.html#leanmodel). If loading fails, it logs the cause and tries model2vec’s `StaticModel`. This fallback needs `laghu[fallback]`. If the fallback also fails, [`load`](https://vedicreader.github.io/laghu/encoder.html#load) raises the original laghu error. `load(..., fallback=False)` raises without trying model2vec. Both loaders refuse bfloat16 and float8 tables because numpy cannot hold those dtypes.
61
+
62
+ `same_as_model2vec(path)` loads a model both ways and compares them on the built-in corpus, or on texts of your own:
63
+
64
+ ``` python
65
+ from laghu import same_as_model2vec
66
+ same_as_model2vec('minishlab/potion-base-8M', ['my', 'own texts'])
67
+ ```
68
+
69
+ /Users/71293/code/personal/orgs/laghu/.venv/lib/python3.13/site-packages/tqdm/auto.py:21: TqdmWarning: IProgress not found. Please update jupyter and ipywidgets. See https://ipywidgets.readthedocs.io/en/stable/user_install.html
70
+ from .autonotebook import tqdm as notebook_tqdm
71
+
72
+ {'lean': False,
73
+ 'n': 2,
74
+ 'ids': True,
75
+ 'dtype': True,
76
+ 'f32': True,
77
+ 'f16': True,
78
+ 'exact': True,
79
+ 'seq': True,
80
+ 'median': True,
81
+ 'unk': True,
82
+ 'tokens': True}
83
+
84
+ It checks the token ids, the vectors cast to float32 and to float16, the exact dtype and bits, `encode_as_sequence`,
85
+ `median_token_length`, the unknown token id and the token list.
86
+
87
+ ## What matches, and what it costs
88
+
89
+ Every model below matches model2vec’s ids and vectors, bit for bit, on 4,726 texts. The corpus includes prose in 36 languages and random strings from 30 Unicode blocks. It also tests control characters, special tokens, emoji sequences, combining marks, bidi controls and a 200,000-character string. Each memory measurement uses a fresh process. These macOS measurements use `phys_footprint` after encoding and `lifetime_max_phys_footprint` for peak. On Linux, the helper reports `RssAnon` and peak resident memory (`VmHWM`) instead. Speed is `encode` over 3,226 texts on an M3 Max.
90
+
91
+ | model | tokenizer | laghu tokenizes with | vocab × dims | dtype | norm | identical | private MB, model2vec → laghu | peak MB, model2vec → laghu | texts/s, model2vec → laghu |
92
+ |----|----|----|----|----|----|----|----|----|----|
93
+ | potion-base-2M | WordPiece | tokenizers | 29,528 × 64 | float32 | on | yes | 81 → 58 | 82 → 58 | 84,510 → 92,982 |
94
+ | potion-base-4M | WordPiece | tokenizers | 29,528 × 128 | float32 | on | yes | 94 → 62 | 95 → 62 | 87,248 → 90,242 |
95
+ | potion-base-8M | WordPiece | tokenizers | 29,528 × 256 | float32 | on | yes | 114 → 70 | 115 → 70 | 79,415 → 81,810 |
96
+ | potion-base-32M | WordPiece | tokenizers | 63,091 × 512 | float32 | on | yes | 244 → 97 | 244 → 97 | 75,009 → 81,379 |
97
+ | potion-retrieval-32M | WordPiece | tokenizers | 63,091 × 512 | float32 | on | yes | 244 → 97 | 244 → 97 | 70,799 → 79,717 |
98
+ | potion-science-8M | WordPiece | tokenizers | 29,528 × 256 | float32 | on | yes | 113 → 70 | 114 → 70 | 81,008 → 84,532 |
99
+ | potion-code-16M | WordPiece | tokenizers | 61,826 × 256 | float32 +map +weights | on | yes | 147 → 110 | 211 → 117 | 10,674 → 10,425 |
100
+ | potion-code-16M-v2 | WordPiece | tokenizers | 63,457 × 256 | float16 | on | yes | 108 → 101 | 186 → 119 | 21,819 → 22,164 |
101
+ | potion-multilingual-128M | Unigram | Python Viterbi | 500,353 × 256 | float32 | on | yes | 1,348 → 135 | 1,348 → 135 | 39,105 → 12,795 |
102
+ | M2V_base_output | WordPiece | tokenizers | 29,528 × 256 | float32 | off | yes | 112 → 69 | 113 → 69 | 65,460 → 87,100 |
103
+ | M2V_multilingual_output | WordPiece | tokenizers | 501,054 × 256 | float32 | off | yes | 834 → 198 | 842 → 295 | 42,565 → 90,283 |
104
+ | M2V_base_glove | WordLevel | tokenizers | 400,002 × 256 | float32 | off | yes | 656 → 186 | 656 → 194 | 67,028 → 85,058 |
105
+ | M2V_base_glove_subword | WordPiece | tokenizers | 400,779 × 256 | float32 | off | yes | 647 → 178 | 649 → 195 | 66,256 → 85,430 |
106
+ | jina-embeddings-v3-separation-distilled | Unigram | Python Viterbi | 250,002 × 256 | float32 | off | yes | 722 → 88 | 722 → 88 | 51,836 → 20,840 |
107
+ | multilingual-e5-large-m2v | Unigram | Python Viterbi | 250,002 × 512 | float32 | on | yes | 981 → 109 | 981 → 109 | 24,139 → 19,670 |
108
+ | m2v-gemma-embedding-300m | Unigram | Python Viterbi | 255,732 × 256 | float16 +weights | on | yes | 542 → 93 | 542 → 93 | 20,319 → 12,134 |
109
+ | potion-multilingual-128M-i8-tokenade | Unigram | Python Viterbi | 500,353 × 256 | int8 | on | yes | 985 → 138 | 985 → 138 | 27,676 → 12,090 |
110
+ | m2v-qwen3-embedding-0.6b-1024d | Unigram | Python Viterbi | 151,644 × 1024 | float16 +weights | on | yes | 602 → 132 | 602 → 132 | 16,792 → 7,168 |
111
+ | ovos-m2v-intents-es-ES-bne-v5 | Unigram | Python Viterbi | 50,259 × 64 | float32 | on | yes | 169 → 52 | 169 → 52 | 11,786 → 11,265 |
112
+ | m2v-gte-256-edu | Unigram | tokenizers | 313,845 × 256 | float32 | on | yes | 1,028 → 663 | 1,028 → 675 | 25,363 → 24,981 |
113
+ | bge-m3-m2v-256 | Unigram | Python Viterbi | 250,002 × 256 | float32 | off | yes | 725 → 89 | 725 → 89 | 34,518 → 20,670 |
114
+ | potion-mxbai-micro | WordPiece | tokenizers | 2,000 × 256 | int8 +map +weights | on | yes | 90 → 72 | 91 → 72 | 68,385 → 70,998 |
115
+ | NIFE-mxbai-embed-large-v1_model2vec | WordPiece | tokenizers | 107,962 × 1024 | float32 | off | yes | 558 → 106 | 572 → 116 | 41,042 → 26,756 |
116
+ | ovos-m2v-intents-pt-PT-albertina-v5 | BPE | tokenizers | 50,262 × 256 | float32 | on | yes | 162 → 90 | 162 → 90 | 46,343 → 22,813 |
117
+ | F2LLM-v2-330M-model2vec | BPE | tokenizers | 151,644 × 512 | float16 | on | yes | 356 → 171 | 356 → 171 | 28,795 → 40,132 |
118
+ | static-retrieval-multilingual-69m-v1 | BPE | tokenizers | 179,936 × 384 | float32 | off | yes | 537 → 229 | 541 → 263 | 32,235 → 31,848 |
119
+
120
+ Unigram throughput ranges from about a third of model2vec’s rate to about the same. `tokenizers` parallelizes batches across cores. Python Viterbi runs on one core. These batch measurements do not establish latency for one query.
121
+
122
+ The word cache helps when the pre-tokenizer splits words, as in jina-v3, bge-m3 and multilingual-e5. Natural text runs 1.6 to 1.9 times faster on a first pass and 3 to 5 times faster on repeated text. The `bench` notebook records these measurements and the reasons for rejecting tokie and gigatoken.
123
+
124
+ Other models use the same `tokenizers` code in both loaders. The timing samples last about 0.1 seconds and vary by up to half on repeated runs. They do not establish a speed advantage.
125
+
126
+ ## What is not supported
127
+
128
+ - Quantizing or truncating on load (`quantize_to`, `dimensionality`, `vocabulary_quantization`): each would copy the table, which is
129
+ what laghu exists to avoid. Load with model2vec for those.
130
+ - `single_word` added tokens on a Unigram model fall back to `tokenizers` for tokenizing (the table is still mapped). Rust decides a
131
+ word boundary by Unicode’s Alphabetic property, which Python’s `str.isalpha` does not match.
132
+ - `use_multiprocessing` is accepted and ignored; the result is the same.
133
+ - `encode([])` returns an empty `(0, dim)` array. model2vec raises.
134
+
135
+ ## Developing
136
+
137
+ nbdev 3 on hatchling: the notebooks in `nbs/` are the source. `uv run nbdev-prepare` exports, tests and rebuilds this README;
138
+ `uv run nbdev-test --flags slow` also runs the bench, which downloads about 5 GB of models.
139
+
140
+ ## Licence
141
+
142
+ MIT.
laghu-0.1.0/README.md ADDED
@@ -0,0 +1,118 @@
1
+ # laghu
2
+
3
+
4
+ <!-- WARNING: THIS FILE WAS AUTOGENERATED! DO NOT EDIT! -->
5
+
6
+ laghu (Sanskrit for light) loads a [model2vec](https://github.com/MinishLab/model2vec) static embedding model with `encode`, `encode_as_sequence` and `tokenize`. The output matches model2vec’s bits, apart from the empty-input difference documented below.
7
+
8
+ model2vec reads the embedding table into each process. For potion-multilingual-128M, the table takes 512 MB. Its Unigram tokenizer adds a character trie over 500,353 pieces, about another 610 MB. After encoding, the process uses about 1.3 GB of private memory.
9
+
10
+ laghu maps the table read-only from `model.safetensors`. Processes share the mapped pages through the operating system’s page cache. The operating system can reclaim clean pages under memory pressure. For supported Unigram models, laghu replaces the trie with Python Viterbi over a dict of pieces. The normalizer and pre-tokenizer come from `tokenizers`, built from the same tokenizer.json with the vocabulary removed. The same model then takes about 135 MB of private memory per process.
11
+
12
+ WordPiece, BPE and word-level models still use `tokenizers`. laghu maps their embedding tables but does not replace their tokenizers. The memory saving depends on the table’s size relative to the tokenizer.
13
+
14
+ ## Install
15
+
16
+ ``` sh
17
+ uv add laghu # or: pip install laghu
18
+ ```
19
+
20
+ model2vec itself is not a dependency. For the fallback described below, and for [`same_as_model2vec`](https://vedicreader.github.io/laghu/encoder.html#same_as_model2vec), install the extra:
21
+ `uv add 'laghu[fallback]'`.
22
+
23
+ ## Use
24
+
25
+ ``` python
26
+ from laghu import load
27
+
28
+ m = load('minishlab/potion-multilingual-128M') # a folder, or a Hugging Face repo id (the local cache is tried first)
29
+ v = m.encode(['The quick brown fox.', 'कर्मण्येवाधिकारस्ते']) # float32, (2, 256), unit rows: the same array model2vec returns
30
+ m.tokenize(['hello world'])
31
+ m.encode_as_sequence('hello world').shape # (n_tokens, 256)
32
+ ```
33
+
34
+ (2, 256)
35
+
36
+ [`load`](https://vedicreader.github.io/laghu/encoder.html#load) returns a [`LeanModel`](https://vedicreader.github.io/laghu/encoder.html#leanmodel). If loading fails, it logs the cause and tries model2vec’s `StaticModel`. This fallback needs `laghu[fallback]`. If the fallback also fails, [`load`](https://vedicreader.github.io/laghu/encoder.html#load) raises the original laghu error. `load(..., fallback=False)` raises without trying model2vec. Both loaders refuse bfloat16 and float8 tables because numpy cannot hold those dtypes.
37
+
38
+ `same_as_model2vec(path)` loads a model both ways and compares them on the built-in corpus, or on texts of your own:
39
+
40
+ ``` python
41
+ from laghu import same_as_model2vec
42
+ same_as_model2vec('minishlab/potion-base-8M', ['my', 'own texts'])
43
+ ```
44
+
45
+ /Users/71293/code/personal/orgs/laghu/.venv/lib/python3.13/site-packages/tqdm/auto.py:21: TqdmWarning: IProgress not found. Please update jupyter and ipywidgets. See https://ipywidgets.readthedocs.io/en/stable/user_install.html
46
+ from .autonotebook import tqdm as notebook_tqdm
47
+
48
+ {'lean': False,
49
+ 'n': 2,
50
+ 'ids': True,
51
+ 'dtype': True,
52
+ 'f32': True,
53
+ 'f16': True,
54
+ 'exact': True,
55
+ 'seq': True,
56
+ 'median': True,
57
+ 'unk': True,
58
+ 'tokens': True}
59
+
60
+ It checks the token ids, the vectors cast to float32 and to float16, the exact dtype and bits, `encode_as_sequence`,
61
+ `median_token_length`, the unknown token id and the token list.
62
+
63
+ ## What matches, and what it costs
64
+
65
+ Every model below matches model2vec’s ids and vectors, bit for bit, on 4,726 texts. The corpus includes prose in 36 languages and random strings from 30 Unicode blocks. It also tests control characters, special tokens, emoji sequences, combining marks, bidi controls and a 200,000-character string. Each memory measurement uses a fresh process. These macOS measurements use `phys_footprint` after encoding and `lifetime_max_phys_footprint` for peak. On Linux, the helper reports `RssAnon` and peak resident memory (`VmHWM`) instead. Speed is `encode` over 3,226 texts on an M3 Max.
66
+
67
+ | model | tokenizer | laghu tokenizes with | vocab × dims | dtype | norm | identical | private MB, model2vec → laghu | peak MB, model2vec → laghu | texts/s, model2vec → laghu |
68
+ |----|----|----|----|----|----|----|----|----|----|
69
+ | potion-base-2M | WordPiece | tokenizers | 29,528 × 64 | float32 | on | yes | 81 → 58 | 82 → 58 | 84,510 → 92,982 |
70
+ | potion-base-4M | WordPiece | tokenizers | 29,528 × 128 | float32 | on | yes | 94 → 62 | 95 → 62 | 87,248 → 90,242 |
71
+ | potion-base-8M | WordPiece | tokenizers | 29,528 × 256 | float32 | on | yes | 114 → 70 | 115 → 70 | 79,415 → 81,810 |
72
+ | potion-base-32M | WordPiece | tokenizers | 63,091 × 512 | float32 | on | yes | 244 → 97 | 244 → 97 | 75,009 → 81,379 |
73
+ | potion-retrieval-32M | WordPiece | tokenizers | 63,091 × 512 | float32 | on | yes | 244 → 97 | 244 → 97 | 70,799 → 79,717 |
74
+ | potion-science-8M | WordPiece | tokenizers | 29,528 × 256 | float32 | on | yes | 113 → 70 | 114 → 70 | 81,008 → 84,532 |
75
+ | potion-code-16M | WordPiece | tokenizers | 61,826 × 256 | float32 +map +weights | on | yes | 147 → 110 | 211 → 117 | 10,674 → 10,425 |
76
+ | potion-code-16M-v2 | WordPiece | tokenizers | 63,457 × 256 | float16 | on | yes | 108 → 101 | 186 → 119 | 21,819 → 22,164 |
77
+ | potion-multilingual-128M | Unigram | Python Viterbi | 500,353 × 256 | float32 | on | yes | 1,348 → 135 | 1,348 → 135 | 39,105 → 12,795 |
78
+ | M2V_base_output | WordPiece | tokenizers | 29,528 × 256 | float32 | off | yes | 112 → 69 | 113 → 69 | 65,460 → 87,100 |
79
+ | M2V_multilingual_output | WordPiece | tokenizers | 501,054 × 256 | float32 | off | yes | 834 → 198 | 842 → 295 | 42,565 → 90,283 |
80
+ | M2V_base_glove | WordLevel | tokenizers | 400,002 × 256 | float32 | off | yes | 656 → 186 | 656 → 194 | 67,028 → 85,058 |
81
+ | M2V_base_glove_subword | WordPiece | tokenizers | 400,779 × 256 | float32 | off | yes | 647 → 178 | 649 → 195 | 66,256 → 85,430 |
82
+ | jina-embeddings-v3-separation-distilled | Unigram | Python Viterbi | 250,002 × 256 | float32 | off | yes | 722 → 88 | 722 → 88 | 51,836 → 20,840 |
83
+ | multilingual-e5-large-m2v | Unigram | Python Viterbi | 250,002 × 512 | float32 | on | yes | 981 → 109 | 981 → 109 | 24,139 → 19,670 |
84
+ | m2v-gemma-embedding-300m | Unigram | Python Viterbi | 255,732 × 256 | float16 +weights | on | yes | 542 → 93 | 542 → 93 | 20,319 → 12,134 |
85
+ | potion-multilingual-128M-i8-tokenade | Unigram | Python Viterbi | 500,353 × 256 | int8 | on | yes | 985 → 138 | 985 → 138 | 27,676 → 12,090 |
86
+ | m2v-qwen3-embedding-0.6b-1024d | Unigram | Python Viterbi | 151,644 × 1024 | float16 +weights | on | yes | 602 → 132 | 602 → 132 | 16,792 → 7,168 |
87
+ | ovos-m2v-intents-es-ES-bne-v5 | Unigram | Python Viterbi | 50,259 × 64 | float32 | on | yes | 169 → 52 | 169 → 52 | 11,786 → 11,265 |
88
+ | m2v-gte-256-edu | Unigram | tokenizers | 313,845 × 256 | float32 | on | yes | 1,028 → 663 | 1,028 → 675 | 25,363 → 24,981 |
89
+ | bge-m3-m2v-256 | Unigram | Python Viterbi | 250,002 × 256 | float32 | off | yes | 725 → 89 | 725 → 89 | 34,518 → 20,670 |
90
+ | potion-mxbai-micro | WordPiece | tokenizers | 2,000 × 256 | int8 +map +weights | on | yes | 90 → 72 | 91 → 72 | 68,385 → 70,998 |
91
+ | NIFE-mxbai-embed-large-v1_model2vec | WordPiece | tokenizers | 107,962 × 1024 | float32 | off | yes | 558 → 106 | 572 → 116 | 41,042 → 26,756 |
92
+ | ovos-m2v-intents-pt-PT-albertina-v5 | BPE | tokenizers | 50,262 × 256 | float32 | on | yes | 162 → 90 | 162 → 90 | 46,343 → 22,813 |
93
+ | F2LLM-v2-330M-model2vec | BPE | tokenizers | 151,644 × 512 | float16 | on | yes | 356 → 171 | 356 → 171 | 28,795 → 40,132 |
94
+ | static-retrieval-multilingual-69m-v1 | BPE | tokenizers | 179,936 × 384 | float32 | off | yes | 537 → 229 | 541 → 263 | 32,235 → 31,848 |
95
+
96
+ Unigram throughput ranges from about a third of model2vec’s rate to about the same. `tokenizers` parallelizes batches across cores. Python Viterbi runs on one core. These batch measurements do not establish latency for one query.
97
+
98
+ The word cache helps when the pre-tokenizer splits words, as in jina-v3, bge-m3 and multilingual-e5. Natural text runs 1.6 to 1.9 times faster on a first pass and 3 to 5 times faster on repeated text. The `bench` notebook records these measurements and the reasons for rejecting tokie and gigatoken.
99
+
100
+ Other models use the same `tokenizers` code in both loaders. The timing samples last about 0.1 seconds and vary by up to half on repeated runs. They do not establish a speed advantage.
101
+
102
+ ## What is not supported
103
+
104
+ - Quantizing or truncating on load (`quantize_to`, `dimensionality`, `vocabulary_quantization`): each would copy the table, which is
105
+ what laghu exists to avoid. Load with model2vec for those.
106
+ - `single_word` added tokens on a Unigram model fall back to `tokenizers` for tokenizing (the table is still mapped). Rust decides a
107
+ word boundary by Unicode’s Alphabetic property, which Python’s `str.isalpha` does not match.
108
+ - `use_multiprocessing` is accepted and ignored; the result is the same.
109
+ - `encode([])` returns an empty `(0, dim)` array. model2vec raises.
110
+
111
+ ## Developing
112
+
113
+ nbdev 3 on hatchling: the notebooks in `nbs/` are the source. `uv run nbdev-prepare` exports, tests and rebuilds this README;
114
+ `uv run nbdev-test --flags slow` also runs the bench, which downloads about 5 GB of models.
115
+
116
+ ## Licence
117
+
118
+ MIT.
@@ -0,0 +1,3 @@
1
+ __version__ = "0.1.0"
2
+ from .core import *
3
+ from .encoder import LeanModel, load, same_as_model2vec
@@ -0,0 +1,68 @@
1
+ # Autogenerated by nbdev
2
+
3
+ d = { 'settings': { 'branch': 'main',
4
+ 'doc_baseurl': '/laghu',
5
+ 'doc_host': 'https://vedicreader.github.io',
6
+ 'git_url': 'https://github.com/vedicreader/laghu',
7
+ 'lib_path': 'laghu'},
8
+ 'syms': { 'laghu.bench': { 'laghu.bench._run': ('bench.html#_run', 'laghu/bench.py'),
9
+ 'laghu.bench.check': ('bench.html#check', 'laghu/bench.py'),
10
+ 'laghu.bench.footprint': ('bench.html#footprint', 'laghu/bench.py'),
11
+ 'laghu.bench.measure': ('bench.html#measure', 'laghu/bench.py'),
12
+ 'laghu.bench.row': ('bench.html#row', 'laghu/bench.py'),
13
+ 'laghu.bench.same': ('bench.html#same', 'laghu/bench.py'),
14
+ 'laghu.bench.table': ('bench.html#table', 'laghu/bench.py')},
15
+ 'laghu.core': { 'laghu.core.UnsupportedDtype': ('core.html#unsupporteddtype', 'laghu/core.py'),
16
+ 'laghu.core.cached': ('core.html#cached', 'laghu/core.py'),
17
+ 'laghu.core.corpus': ('core.html#corpus', 'laghu/core.py'),
18
+ 'laghu.core.find_vocab': ('core.html#find_vocab', 'laghu/core.py'),
19
+ 'laghu.core.is_unigram': ('core.html#is_unigram', 'laghu/core.py'),
20
+ 'laghu.core.layout': ('core.html#layout', 'laghu/core.py'),
21
+ 'laghu.core.model_files': ('core.html#model_files', 'laghu/core.py'),
22
+ 'laghu.core.noise': ('core.html#noise', 'laghu/core.py'),
23
+ 'laghu.core.resolve': ('core.html#resolve', 'laghu/core.py'),
24
+ 'laghu.core.st_header': ('core.html#st_header', 'laghu/core.py'),
25
+ 'laghu.core.st_map': ('core.html#st_map', 'laghu/core.py'),
26
+ 'laghu.core.st_write': ('core.html#st_write', 'laghu/core.py'),
27
+ 'laghu.core.try_resolve': ('core.html#try_resolve', 'laghu/core.py')},
28
+ 'laghu.encoder': { 'laghu.encoder.LeanModel': ('encoder.html#leanmodel', 'laghu/encoder.py'),
29
+ 'laghu.encoder.LeanModel.__init__': ('encoder.html#leanmodel.__init__', 'laghu/encoder.py'),
30
+ 'laghu.encoder.LeanModel.__repr__': ('encoder.html#leanmodel.__repr__', 'laghu/encoder.py'),
31
+ 'laghu.encoder.LeanModel._encode_batch': ('encoder.html#leanmodel._encode_batch', 'laghu/encoder.py'),
32
+ 'laghu.encoder.LeanModel._rows': ('encoder.html#leanmodel._rows', 'laghu/encoder.py'),
33
+ 'laghu.encoder.LeanModel.dim': ('encoder.html#leanmodel.dim', 'laghu/encoder.py'),
34
+ 'laghu.encoder.LeanModel.embedding_dtype': ('encoder.html#leanmodel.embedding_dtype', 'laghu/encoder.py'),
35
+ 'laghu.encoder.LeanModel.encode': ('encoder.html#leanmodel.encode', 'laghu/encoder.py'),
36
+ 'laghu.encoder.LeanModel.encode_as_sequence': ( 'encoder.html#leanmodel.encode_as_sequence',
37
+ 'laghu/encoder.py'),
38
+ 'laghu.encoder.LeanModel.from_pretrained': ('encoder.html#leanmodel.from_pretrained', 'laghu/encoder.py'),
39
+ 'laghu.encoder.LeanModel.lean': ('encoder.html#leanmodel.lean', 'laghu/encoder.py'),
40
+ 'laghu.encoder.LeanModel.tokenize': ('encoder.html#leanmodel.tokenize', 'laghu/encoder.py'),
41
+ 'laghu.encoder.LeanModel.tokens': ('encoder.html#leanmodel.tokens', 'laghu/encoder.py'),
42
+ 'laghu.encoder._batches': ('encoder.html#_batches', 'laghu/encoder.py'),
43
+ 'laghu.encoder.load': ('encoder.html#load', 'laghu/encoder.py'),
44
+ 'laghu.encoder.same_as_model2vec': ('encoder.html#same_as_model2vec', 'laghu/encoder.py')},
45
+ 'laghu.unigram': { 'laghu.unigram.Added': ('unigram.html#added', 'laghu/unigram.py'),
46
+ 'laghu.unigram.Added.__init__': ('unigram.html#added.__init__', 'laghu/unigram.py'),
47
+ 'laghu.unigram.Added.split': ('unigram.html#added.split', 'laghu/unigram.py'),
48
+ 'laghu.unigram.HFTok': ('unigram.html#hftok', 'laghu/unigram.py'),
49
+ 'laghu.unigram.HFTok.__call__': ('unigram.html#hftok.__call__', 'laghu/unigram.py'),
50
+ 'laghu.unigram.HFTok.__init__': ('unigram.html#hftok.__init__', 'laghu/unigram.py'),
51
+ 'laghu.unigram.HFTok.vocab': ('unigram.html#hftok.vocab', 'laghu/unigram.py'),
52
+ 'laghu.unigram.Unigram': ('unigram.html#unigram', 'laghu/unigram.py'),
53
+ 'laghu.unigram.Unigram.__call__': ('unigram.html#unigram.__call__', 'laghu/unigram.py'),
54
+ 'laghu.unigram.Unigram.__init__': ('unigram.html#unigram.__init__', 'laghu/unigram.py'),
55
+ 'laghu.unigram.Unigram._first': ('unigram.html#unigram._first', 'laghu/unigram.py'),
56
+ 'laghu.unigram.Unigram._pieces': ('unigram.html#unigram._pieces', 'laghu/unigram.py'),
57
+ 'laghu.unigram.Unigram._seg': ('unigram.html#unigram._seg', 'laghu/unigram.py'),
58
+ 'laghu.unigram.Unigram._viterbi': ('unigram.html#unigram._viterbi', 'laghu/unigram.py'),
59
+ 'laghu.unigram.Unigram.encode': ('unigram.html#unigram.encode', 'laghu/unigram.py'),
60
+ 'laghu.unigram.Unigram.unk_token_id': ('unigram.html#unigram.unk_token_id', 'laghu/unigram.py'),
61
+ 'laghu.unigram.Unigram.vocab': ('unigram.html#unigram.vocab', 'laghu/unigram.py'),
62
+ 'laghu.unigram.Unigram.word': ('unigram.html#unigram.word', 'laghu/unigram.py'),
63
+ 'laghu.unigram._metaspace': ('unigram.html#_metaspace', 'laghu/unigram.py'),
64
+ 'laghu.unigram._pipeline': ('unigram.html#_pipeline', 'laghu/unigram.py'),
65
+ 'laghu.unigram._unk': ('unigram.html#_unk', 'laghu/unigram.py'),
66
+ 'laghu.unigram.read_vocab': ('unigram.html#read_vocab', 'laghu/unigram.py'),
67
+ 'laghu.unigram.tokenizer': ('unigram.html#tokenizer', 'laghu/unigram.py'),
68
+ 'laghu.unigram.viterbi': ('unigram.html#viterbi', 'laghu/unigram.py')}}}
@@ -0,0 +1,80 @@
1
+ # AUTOGENERATED! DO NOT EDIT! File to edit: ../nbs/03_bench.ipynb.
2
+
3
+ # %% auto #0
4
+ __all__ = ['footprint', 'measure', 'check', 'row', 'same', 'table']
5
+
6
+ # %% ../nbs/03_bench.ipynb #630431fe
7
+ import json, sys, os, ctypes, platform, subprocess
8
+ from fastcore.all import AttrDict
9
+
10
+ # %% ../nbs/03_bench.ipynb #ad5d5182
11
+ def footprint():
12
+ "This process's resident, private and peak memory in MB."
13
+ if platform.system() == 'Darwin':
14
+ buf = (ctypes.c_uint64 * 64)() # rusage_info_v4 as uint64s: a 16-byte uuid, then resident_size at 8, phys_footprint 9, lifetime_max_phys_footprint 30
15
+ ctypes.CDLL(None).proc_pid_rusage(ctypes.c_int(os.getpid()), 4, buf)
16
+ return AttrDict(rss=buf[8] / 2**20, private=buf[9] / 2**20, peak=buf[30] / 2**20)
17
+ kv = dict(l.split(':', 1) for l in open('/proc/self/status'))
18
+ mb = lambda k: int(kv[k].split()[0]) / 1024
19
+ return AttrDict(rss=mb('VmRSS'), private=mb('RssAnon'), peak=mb('VmHWM'))
20
+
21
+ # %% ../nbs/03_bench.ipynb #b4cc8766
22
+ _PROBE = '''
23
+ import json, time
24
+ from laghu.bench import footprint
25
+ from laghu.core import corpus, resolve
26
+ t = time.time()
27
+ if {how!r} == 'model2vec':
28
+ from model2vec import StaticModel
29
+ m = StaticModel.from_pretrained(str(resolve({path!r})), force_download=False)
30
+ else:
31
+ from laghu import load
32
+ m = load({path!r}, fallback=False)
33
+ tl = time.time() - t; f1 = footprint()
34
+ txts = corpus({n}); t = time.time(); m.encode(txts); te = time.time() - t; f2 = footprint()
35
+ print(json.dumps(dict(how={how!r}, load_s=tl, encode_s=te, texts=len(txts), private_load=f1.private, private=f2.private, rss=f2.rss, peak=f2.peak)))
36
+ '''
37
+
38
+ _CHECK = '''
39
+ import json
40
+ from laghu import load, same_as_model2vec
41
+ from laghu.core import corpus, model_files, resolve, find_vocab
42
+ m = load({path!r}, fallback=False); f = model_files(m.folder)
43
+ kind = json.loads(f.tok.read_text())['model'].get('type') or 'Unigram'
44
+ r = same_as_model2vec({path!r}, corpus({n}))
45
+ print(json.dumps(dict(r, kind=kind, vocab=len(m.embedding), dim=m.dim, dtype=m.embedding_dtype, normalize=bool(m.normalize),
46
+ quantized=m.token_mapping is not None, weighted=m.weights is not None)))
47
+ '''
48
+
49
+ def _run(code):
50
+ r = subprocess.run([sys.executable, '-c', code], capture_output=True, text=True)
51
+ if r.returncode: raise RuntimeError(r.stderr[-3000:])
52
+ return AttrDict(json.loads(r.stdout.strip().splitlines()[-1]))
53
+
54
+ def measure(path, how='laghu', n=2000):
55
+ "Memory (MB) and time (s) of loading `path` with `how` ('laghu' or 'model2vec') and encoding `corpus(n)`, in a new process."
56
+ return _run(_PROBE.format(path=str(path), how=how, n=n))
57
+
58
+ def check(path, n=3000):
59
+ "`same_as_model2vec(path, corpus(n))` and the model's shape, in a new process."
60
+ return _run(_CHECK.format(path=str(path), n=n))
61
+
62
+ def row(path, n_check=3000, n_speed=2000):
63
+ "One model's line: what it is, whether it matches, and both ways' memory and speed."
64
+ r = check(path, n_check); r.model = str(path)
65
+ r.m2v, r.lg = measure(path, 'model2vec', n_speed), measure(path, 'laghu', n_speed)
66
+ return r
67
+
68
+ # %% ../nbs/03_bench.ipynb #33c0bdb7
69
+ def same(r): return all(r[k] for k in ('ids', 'dtype', 'f32', 'f16', 'exact', 'seq', 'median', 'unk', 'tokens'))
70
+
71
+ def table(rows):
72
+ "Rows of `row` as a markdown table."
73
+ hd = ('| model | tokenizer | laghu tokenizes with | vocab × dims | dtype | norm | identical | private MB, model2vec → laghu | peak MB, model2vec → laghu | texts/s, model2vec → laghu |\n'
74
+ '|---|---|---|---|---|---|---|---|---|---|\n')
75
+ def ln(r):
76
+ extra = ' +map' * r.quantized + ' +weights' * r.weighted
77
+ return (f"| {r.model.split('/')[-1]} | {r.kind} | {'Python Viterbi' if r.lean else 'tokenizers'} | {r.vocab:,} × {r.dim} | {r.dtype}{extra} | "
78
+ f"{'on' if r.normalize else 'off'} | {'yes' if same(r) else 'NO'} | {r.m2v.private:,.0f} → {r.lg.private:,.0f} | {r.m2v.peak:,.0f} → {r.lg.peak:,.0f} | "
79
+ f"{r.m2v.texts / r.m2v.encode_s:,.0f} → {r.lg.texts / r.lg.encode_s:,.0f} |")
80
+ return hd + '\n'.join(map(ln, rows))
@@ -0,0 +1,147 @@
1
+ # AUTOGENERATED! DO NOT EDIT! File to edit: ../nbs/00_core.ipynb.
2
+
3
+ # %% auto #0
4
+ __all__ = ['log', 'LAYOUTS', 'ST_DTYPES', 'SENTENCES', 'ODD', 'cached', 'resolve', 'layout', 'model_files', 'UnsupportedDtype',
5
+ 'st_header', 'st_map', 'st_write', 'find_vocab', 'is_unigram', 'try_resolve', 'noise', 'corpus']
6
+
7
+ # %% ../nbs/00_core.ipynb #fd476c3c
8
+ import json, re, mmap, logging, numpy as np
9
+ from pathlib import Path
10
+ from fastcore.all import AttrDict, ifnone
11
+
12
+ log = logging.getLogger('laghu')
13
+
14
+ # %% ../nbs/00_core.ipynb #9073650b
15
+ def cached(repo):
16
+ "The newest cached snapshot of Hugging Face repo `repo`, or None."
17
+ if repo.count('/') != 1: return None
18
+ from huggingface_hub.constants import HF_HUB_CACHE
19
+ d = Path(HF_HUB_CACHE)/f"models--{repo.replace('/', '--')}"/'snapshots'
20
+ snaps = [o for o in d.iterdir() if o.is_dir()] if d.exists() else []
21
+ return max(snaps, key=lambda o: o.stat().st_mtime) if snaps else None
22
+
23
+ def resolve(path, token=None, subfolder=None, force_download=False):
24
+ "The local folder for `path`: a folder as given, else a complete snapshot of repo id `path` from the cache, else downloaded."
25
+ p = Path(path)
26
+ if p.exists(): return p/subfolder if subfolder else p
27
+ p = None if force_download else cached(str(path))
28
+ if p is None or layout(p/subfolder if subfolder else p) is None: # a snapshot holding only some files (one hf_hub_download) is not the model
29
+ from huggingface_hub import snapshot_download
30
+ p = Path(snapshot_download(Path(path).as_posix(), repo_type='model', token=token))
31
+ return p/subfolder if subfolder else p
32
+
33
+ # %% ../nbs/00_core.ipynb #b57979e0
34
+ def layout(folder):
35
+ "The config, table and tokenizer files of the first layout `folder` has, and the table's tensor name, or None."
36
+ folder = Path(folder)
37
+ for c, e, t, nm in LAYOUTS:
38
+ if all((folder/o).exists() for o in (c, e, t)): return AttrDict(cfg=folder/c, emb=folder/e, tok=folder/t, name=nm)
39
+
40
+ LAYOUTS = (('config.json', 'model.safetensors', 'tokenizer.json', 'embeddings'),
41
+ ('config_sentence_transformers.json', 'model.safetensors', 'tokenizer.json', 'embedding.weight'),
42
+ ('config_sentence_transformers.json', '0_StaticEmbedding/model.safetensors', '0_StaticEmbedding/tokenizer.json', 'embedding.weight'))
43
+
44
+ def model_files(folder):
45
+ "`layout(folder)`, raising when there is none."
46
+ if f := layout(folder): return f
47
+ raise FileNotFoundError(f'{folder}: no model2vec, sentence-transformers or 0_StaticEmbedding layout')
48
+
49
+ # %% ../nbs/00_core.ipynb #86a0c03a
50
+ ST_DTYPES = dict(F64=np.float64, F32=np.float32, F16=np.float16, I64=np.int64, I32=np.int32, I16=np.int16, I8=np.int8, U8=np.uint8)
51
+
52
+ class UnsupportedDtype(ValueError): "A tensor in a dtype numpy cannot hold."
53
+
54
+ def st_header(fn):
55
+ "Safetensors file `fn`'s header and the offset its data starts at."
56
+ with open(fn, 'rb') as f: n = int.from_bytes(f.read(8), 'little'); return json.loads(f.read(n)), 8 + n
57
+
58
+ def st_map(fn, names):
59
+ "Tensors `names` of safetensors file `fn`, memory-mapped read-only; a name the file lacks is None."
60
+ hdr, at = st_header(fn)
61
+ def _m(k):
62
+ if k not in hdr: return None
63
+ t = hdr[k]; s, e = t['data_offsets']; dt = ST_DTYPES.get(t['dtype']); shape = tuple(t['shape'])
64
+ if dt is None: raise UnsupportedDtype(f"{fn}: tensor {k!r} is {t['dtype']}; laghu reads {', '.join(ST_DTYPES)}")
65
+ if e == s: return np.zeros(shape, dt) # np.memmap refuses an empty map
66
+ m = np.memmap(fn, dtype=dt, mode='r', offset=at + s, shape=shape)
67
+ if m.nbytes != e - s: raise ValueError(f'{fn}: tensor {k!r} has {e - s} bytes, its shape and dtype make {m.nbytes}')
68
+ return m
69
+ return [_m(k) for k in names]
70
+
71
+ # %% ../nbs/00_core.ipynb #4ea0cd84
72
+ _NP2ST = {np.dtype(v): k for k, v in ST_DTYPES.items()}
73
+
74
+ def st_write(fn, tensors):
75
+ "Write dict `tensors` of numpy arrays to safetensors file `fn`."
76
+ hdr, at, bufs = {}, 0, []
77
+ for k, a in tensors.items():
78
+ b = np.ascontiguousarray(a).astype(a.dtype.newbyteorder('<'), copy=False).tobytes()
79
+ hdr[k] = dict(dtype=_NP2ST[a.dtype], shape=list(a.shape), data_offsets=[at, at + len(b)]); at += len(b); bufs.append(b)
80
+ h = json.dumps(hdr, separators=(',', ':')).encode(); h += b' ' * (-len(h) % 8)
81
+ Path(fn).parent.mkdir(parents=True, exist_ok=True)
82
+ with open(fn, 'wb') as f: f.write(len(h).to_bytes(8, 'little')); f.write(h); [f.write(b) for b in bufs]
83
+
84
+ # %% ../nbs/00_core.ipynb #ad1cbbef
85
+ _MODEL = re.compile(rb'"model"\s*:\s*\{')
86
+ _VOCAB = re.compile(rb'"vocab"\s*:\s*([\[{])')
87
+
88
+ def find_vocab(t):
89
+ "Where the model's vocabulary opens in tokenizer.json bytes `t`, and its bracket."
90
+ ms = list(_MODEL.finditer(t))
91
+ if not ms: return None, None
92
+ v = _VOCAB.search(t, ms[-1].end())
93
+ return (v.end(), v.group(1)) if v else (None, None)
94
+
95
+ def is_unigram(fn):
96
+ "Whether tokenizer.json `fn` holds a Unigram model."
97
+ with open(fn, 'rb') as f, mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ) as t: return find_vocab(t)[1] == b'['
98
+
99
+ # %% ../nbs/00_core.ipynb #13167816
100
+ def try_resolve(repo, **kw):
101
+ "`resolve(repo)`, or None (logged) when it cannot be had, e.g. offline."
102
+ try: return resolve(repo, **kw)
103
+ except Exception as e: log.warning(f'laghu: {repo} unavailable: {e!r}'); return None
104
+
105
+ # %% ../nbs/00_core.ipynb #b4cc1c2d
106
+ SENTENCES = [
107
+ 'The quick brown fox jumps over the lazy dog.', "It's 3:45 p.m. on 2026-10-05; the price is $1,299.99 (incl. 18% GST).",
108
+ 'Les élèves ont étudié l’histoire de la Révolution française.', 'Größere Äpfel schmecken süßer als kleinere, sagt Jürgen.',
109
+ 'El niño comió paella en la playa de Valencia.', 'Il gatto dorme sul divano mentre piove.', 'O pássaro voou sobre a floresta amazônica.',
110
+ 'Москва — столица России, крупнейший город Европы.', 'Η Αθήνα είναι η πρωτεύουσα της Ελλάδας.', 'Zażółć gęślą jaźń.',
111
+ 'Příliš žluťoučký kůň úpěl ďábelské ódy.', 'Árvíztűrő tükörfúrógép.', 'İstanbul’da çay içip simit yedik.', 'Hà Nội là thủ đô của Việt Nam.',
112
+ 'مرحبا بالعالم، كيف حالك اليوم؟', 'שלום עולם, מה שלומך?', 'سلام دنیا، امروز هوا خوب است.', 'नमस्ते दुनिया, आज मौसम अच्छा है।',
113
+ 'कर्मण्येवाधिकारस्ते मा फलेषु कदाचन ।', 'অামি বাংলায় গান গাই।', 'வணக்கம் உலகம், இன்று வானிலை நன்றாக உள்ளது.',
114
+ 'నమస్కారం ప్రపంచం.', 'ನಮಸ್ಕಾರ ಜಗತ್ತು.', 'നമസ്കാരം ലോകം.', 'ਸਤਿ ਸ੍ਰੀ ਅਕਾਲ ਦੁਨੀਆ।', 'નમસ્તે દુનિયા.', 'සුභ උදෑසනක්.',
115
+ 'สวัสดีชาวโลก วันนี้อากาศดี', 'ສະບາຍດີ', 'မင်္ဂလာပါ', 'ជំរាបសួរ', '你好,世界!今天天气很好。', '日本語の文章を正しく分割できますか?',
116
+ '안녕하세요, 세계! 오늘 날씨가 좋네요.', 'გამარჯობა მსოფლიო', 'Բարեւ աշխարհ', 'ሰላም ልዑል', 'Habari ya dunia, leo ni siku nzuri.',
117
+ 'def f(x): return x**2 # squares\nprint(f(3))', 'SELECT * FROM users WHERE id = 42;', 'https://example.com/a/b?c=d&e=f#g',
118
+ 'user@example.org, +1 (555) 010-9999', 'E = mc², ∑ᵢ xᵢ² ≤ ∞, ∀x∈ℝ', '¯\\_(ツ)_/¯ :-) <3 ^_^']
119
+
120
+ ODD = ['', ' ', ' ', '\t\n', '\n\n\n', '\x00', '\x00\x01\x1f\x7f', '\x85

', 'tab\tsep\x0bvt\x0cff', 'CRLF\r\nline\rcr',
121
+ '[UNK]', '[PAD]', '[CLS] [SEP] [MASK]', 'a[PAD]b [UNK] c', '[PAD][PAD]', '[UNK]x[UNK]', ' [PAD]', '<unk>', '<pad></s>', '<s>x</s>',
122
+ '<extra_id_0>x<extra_id_99>', '<mask>', '▁', '▁▁x', ' leading', 'trailing ', 'a b c', ' nbsp ideographic thin',
123
+ '​‌‍', 'bom', '\U0001F600 emoji', '👩‍👩‍👧‍👦 👍🏽 🇮🇳 🏳️‍🌈', 'fi fl Ⅻ ① ㎏ ㍿', 'é é ñ ñ', 'à́̂̃̄',
124
+ 'z̷̢̛a̴l̸g̵o̶', 'FULLWIDTH カタカナ', '𝔘𝔫𝔦𝔠𝔬𝔡𝔢 𝟙𝟚𝟛', '𑌗𑍍𑌰𑌨𑍍𑌥', ' private', 'RTL ‮evil‬ ‏',
125
+ '\U000E0041\U000E007F tags', '\U0010FFFF', 'x' * 5000, 'x ' * 3000, ' '.join(['word'] * 600), 'ab' * 100_000, 'नम' * 100_000]
126
+
127
+ _BLOCKS = [(0x20, 0x7e), (0xa0, 0xff), (0x100, 0x24f), (0x300, 0x36f), (0x370, 0x3ff), (0x400, 0x4ff), (0x530, 0x58f), (0x590, 0x5ff),
128
+ (0x600, 0x6ff), (0x900, 0x97f), (0x980, 0x9ff), (0xb80, 0xbff), (0xc00, 0xc7f), (0xd00, 0xd7f), (0xe00, 0xe7f), (0x10a0, 0x10ff),
129
+ (0x1100, 0x11ff), (0x2000, 0x206f), (0x2100, 0x214f), (0x2190, 0x22ff), (0x3000, 0x303f), (0x3040, 0x30ff), (0x4e00, 0x4fff),
130
+ (0xac00, 0xadff), (0xfb00, 0xfb4f), (0xff00, 0xffef), (0x1d400, 0x1d4ff), (0x1f300, 0x1f6ff), (0x0, 0x1f), (0x11300, 0x1137f)]
131
+
132
+ def noise(n=2000, seed=0):
133
+ "`n` seeded random strings, each from one to three Unicode blocks, with spaces between runs."
134
+ import random
135
+ r = random.Random(seed); out = []
136
+ for _ in range(n):
137
+ bs = r.sample(_BLOCKS, r.randint(1, 3))
138
+ out.append(''.join(chr(r.randint(*r.choice(bs))) if r.random() > 0.15 else ' ' for _ in range(r.randint(0, 60))))
139
+ return out
140
+
141
+ def corpus(n=2000, seed=0):
142
+ "SENTENCES, ODD, `noise(n, seed)` and the sentences recombined: about `2.4 * n` texts."
143
+ import random
144
+ r = random.Random(seed); ws = ' '.join(SENTENCES).split()
145
+ mix = [' '.join(r.sample(ws, r.randint(1, 20))) for _ in range(n // 2)]
146
+ mix += [o.upper() for o in SENTENCES] + [o.replace(' ', '') for o in SENTENCES] + [s[:r.randint(0, len(s))] for s in SENTENCES]
147
+ return SENTENCES + ODD + noise(n, seed) + mix + ['\n'.join(SENTENCES), ' '.join(SENTENCES) * 20]
@@ -0,0 +1,112 @@
1
+ # AUTOGENERATED! DO NOT EDIT! File to edit: ../nbs/02_encoder.ipynb.
2
+
3
+ # %% auto #0
4
+ __all__ = ['LeanModel', 'load', 'same_as_model2vec']
5
+
6
+ # %% ../nbs/02_encoder.ipynb #07e60f33
7
+ import json, numpy as np
8
+ from pathlib import Path
9
+ from fastcore.all import ifnone, patch
10
+ from .core import log, resolve, model_files, st_map
11
+ from .unigram import tokenizer, Unigram
12
+
13
+ # %% ../nbs/02_encoder.ipynb #bf662691
14
+ class LeanModel:
15
+ "model2vec's `StaticModel` for reading a model, with its table mapped from the file and, for Unigram, no trie."
16
+ def __init__(self, folder, normalize=None, lean=True):
17
+ f = model_files(folder); self.folder = Path(folder)
18
+ self.config = json.loads(f.cfg.read_text())
19
+ self.embedding, self.weights, self.token_mapping = st_map(f.emb, [f.name, 'weights', 'mapping'])
20
+ if self.embedding is None: raise KeyError(f'{f.emb}: no tensor {f.name!r}')
21
+ self.tokenizer = tokenizer(f.tok, lean)
22
+ if self.token_mapping is None and self.tokenizer.n != len(self.embedding):
23
+ raise ValueError(f'Number of tokens ({self.tokenizer.n}) does not match number of vectors ({len(self.embedding)}).')
24
+ self.median_token_length, self.unk_token_id = self.tokenizer.median, self.tokenizer.unk_token_id
25
+ self.normalize = ifnone(normalize, self.config.get('normalize', False))
26
+ self.config['normalize'] = self.normalize
27
+
28
+ @classmethod
29
+ def from_pretrained(cls, path, token=None, normalize=None, subfolder=None, force_download=False, lean=True):
30
+ "A `LeanModel` for a folder or Hugging Face repo id `path`; a cached repo loads without the network."
31
+ return cls(resolve(path, token=token, subfolder=subfolder, force_download=force_download), normalize=normalize, lean=lean)
32
+
33
+ @property
34
+ def dim(self): return self.embedding.shape[1]
35
+ @property
36
+ def embedding_dtype(self): return np.dtype(self.embedding.dtype).name
37
+ @property
38
+ def lean(self): return isinstance(self.tokenizer, Unigram)
39
+ @property
40
+ def tokens(self):
41
+ "Every token, in id order, as `StaticModel.tokens`."
42
+ if not hasattr(self, '_tokens'): v = self.tokenizer.vocab(); self._tokens = tuple(sorted(v, key=v.get))
43
+ return self._tokens
44
+
45
+ def __repr__(self):
46
+ return f"{type(self).__name__}({str(self.folder)!r}, {len(self.embedding)}×{self.dim} {self.embedding_dtype}, {'unigram' if self.lean else 'tokenizers'})"
47
+
48
+ # %% ../nbs/02_encoder.ipynb #6fb7680d
49
+ def _batches(xs, bs): return (xs[i:i + bs] for i in range(0, len(xs), bs))
50
+
51
+ @patch
52
+ def tokenize(self:LeanModel, sentences, max_length=None):
53
+ "The ids of each of `sentences`, as `StaticModel.tokenize`."
54
+ if max_length is not None: m = max_length * self.median_token_length; sentences = [s[:m] for s in sentences]
55
+ ids = self.tokenizer(sentences)
56
+ if self.unk_token_id is not None: ids = [[k for k in o if k != self.unk_token_id] for o in ids]
57
+ return [o[:max_length] for o in ids] if max_length is not None else ids
58
+
59
+ @patch
60
+ def _rows(self:LeanModel, ids):
61
+ "The vectors of `ids`, remapped and weighted when the model is vocabulary-quantized."
62
+ emb = self.embedding[ids if self.token_mapping is None else self.token_mapping[ids]]
63
+ return emb * self.weights[ids][:, None] if self.weights is not None else emb
64
+
65
+ @patch
66
+ def _encode_batch(self:LeanModel, sentences, max_length):
67
+ "One batch: model2vec stacks the means, so an empty text's float64 zero row makes its whole batch float64."
68
+ out = np.stack([self._rows(o).mean(axis=0) if o else np.zeros(self.dim) for o in self.tokenize(sentences, max_length)])
69
+ return out / (np.linalg.norm(out, axis=1, keepdims=True) + 1e-32) if self.normalize else out
70
+
71
+ @patch
72
+ def encode(self:LeanModel, sentences, show_progress_bar=False, max_length=512, batch_size=1024, use_multiprocessing=True, multiprocessing_threshold=10_000, **kwargs):
73
+ "Each sentence's mean token vector (unit length when `normalize`), as `StaticModel.encode`; one string gives one vector."
74
+ one = isinstance(sentences, str)
75
+ xs = [sentences] if one else list(sentences)
76
+ if not xs: return np.zeros((0, self.dim), self.embedding.dtype)
77
+ out = np.concatenate([self._encode_batch(b, max_length) for b in _batches(xs, batch_size)], axis=0)
78
+ return out[0] if one else out
79
+
80
+ @patch
81
+ def encode_as_sequence(self:LeanModel, sentences, max_length=None, batch_size=1024, show_progress_bar=False, use_multiprocessing=True, multiprocessing_threshold=10_000):
82
+ "Each sentence's token vectors, `(n_tokens, dim)`, as `StaticModel.encode_as_sequence`."
83
+ one = isinstance(sentences, str)
84
+ xs = [sentences] if one else list(sentences)
85
+ out = [self._rows(o) if o else np.zeros((0, self.dim)) for b in _batches(xs, batch_size) for o in self.tokenize(b, max_length)]
86
+ return out[0] if one else out
87
+
88
+ # %% ../nbs/02_encoder.ipynb #e91068cb
89
+ def load(path, normalize=None, token=None, subfolder=None, force_download=False, lean=True, fallback=True):
90
+ "A `LeanModel` for `path`, else model2vec's `StaticModel` with a logged warning; `fallback=False` raises instead."
91
+ try: return LeanModel.from_pretrained(path, token=token, normalize=normalize, subfolder=subfolder, force_download=force_download, lean=lean)
92
+ except Exception as e:
93
+ if not fallback: raise
94
+ log.warning(f'laghu: cannot read {path} ({e!r}); loading it with model2vec instead')
95
+ try:
96
+ from model2vec import StaticModel
97
+ return StaticModel.from_pretrained(path, token=token, normalize=normalize, subfolder=subfolder, force_download=force_download)
98
+ except Exception as e2: raise e from e2
99
+
100
+ # %% ../nbs/02_encoder.ipynb #c807791f
101
+ def same_as_model2vec(path, txts=None, max_length=512, seq=200):
102
+ "Whether `load(path)` matches model2vec on `txts` (default `corpus()`): a dict of the checks."
103
+ from model2vec import StaticModel
104
+ from laghu.core import corpus
105
+ txts = ifnone(txts, corpus())
106
+ lm, sm = load(path, fallback=False), StaticModel.from_pretrained(str(resolve(path)), force_download=False)
107
+ a, b = lm.encode(txts, max_length=max_length), sm.encode(txts, max_length=max_length)
108
+ sa, sb = lm.encode_as_sequence(txts[:seq]), sm.encode_as_sequence(txts[:seq])
109
+ return dict(lean=lm.lean, n=len(txts), ids=lm.tokenize(txts, max_length) == sm.tokenize(txts, max_length), dtype=a.dtype == b.dtype,
110
+ f32=np.array_equal(a.astype(np.float32), b.astype(np.float32)), f16=np.array_equal(a.astype(np.float16), b.astype(np.float16)),
111
+ exact=a.dtype == b.dtype and np.array_equal(a, b), seq=all(x.dtype == y.dtype and np.array_equal(x, y) for x, y in zip(sa, sb)),
112
+ median=lm.median_token_length == sm.median_token_length, unk=lm.unk_token_id == sm.unk_token_id, tokens=lm.tokens == tuple(sm.tokens))
@@ -0,0 +1,214 @@
1
+ # AUTOGENERATED! DO NOT EDIT! File to edit: ../nbs/01_unigram.ipynb.
2
+
3
+ # %% auto #0
4
+ __all__ = ['UNK_PENALTY', 'WORD_CACHE', 'WORD_LEN', 'WS', 'read_vocab', 'viterbi', 'Added', 'Unigram', 'HFTok', 'tokenizer']
5
+
6
+ # %% ../nbs/01_unigram.ipynb #d9193c51
7
+ import json, re, mmap, numpy as np
8
+ from array import array
9
+ from functools import lru_cache
10
+ from tokenizers import Tokenizer, PreTokenizedString
11
+ from fastcore.all import ifnone
12
+ from .core import log, find_vocab, is_unigram
13
+
14
+ # %% ../nbs/01_unigram.ipynb #60cb7473
15
+ _ENTRY = re.compile(rb'\s*\[\s*"((?:[^"\\]|\\.)*)"\s*,\s*([^\],\s]+)\s*\]\s*(,?)')
16
+ _CLOSE = re.compile(rb'\s*\]')
17
+
18
+ def read_vocab(fn):
19
+ "Unigram tokenizer.json `fn` as (piece → id, scores, the JSON with the vocabulary emptied); a repeated piece keeps its last id, as in `tokenizers`."
20
+ with open(fn, 'rb') as f, mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ) as t:
21
+ v0, br = find_vocab(t)
22
+ if br != b'[': raise NotImplementedError(f'{fn}: not a Unigram vocabulary')
23
+ ids, scores, at, i = {}, array('d'), v0, 0
24
+ while m := _ENTRY.match(t, at):
25
+ p = m.group(1); ids[json.loads(b'"' + p + b'"') if b'\\' in p else p.decode()] = i; scores.append(float(m.group(2))); i += 1; at = m.end()
26
+ if not m.group(3): break
27
+ if not (c := _CLOSE.match(t, at)): raise ValueError(f'{fn}: vocabulary not read to its end (stopped at byte {at})')
28
+ return ids, scores, json.loads(t[:v0] + t[c.start():])
29
+
30
+ # %% ../nbs/01_unigram.ipynb #e3c950b2
31
+ UNK_PENALTY = 10.0
32
+ WORD_CACHE, WORD_LEN = 2**16, 32 # pre-tokens cached per Unigram, and the longest one cached
33
+
34
+ def _unk(p, ids, unk_id, byte_fallback):
35
+ "The ids of an unknown run `p`: a piece if it spells one, else its UTF-8 bytes as `<0xXX>` pieces under byte fallback, else the unknown id."
36
+ if (k := ids.get(p)) is not None: return [k]
37
+ if byte_fallback and None not in (bs := [ids.get(f'<0x{b:02X}>') for b in p.encode()]): return bs
38
+ if unk_id < 0: raise ValueError(f'Encountered an unknown token but `unk_id` is missing: {p!r}') # tokenizers' error, and model2vec's
39
+ return [unk_id]
40
+
41
+ def viterbi(s, ids, scores, reach, unk_id, unk_score, byte_fallback=False):
42
+ "The ids of the best segmentation of `s` into pieces of `ids`, unknown runs fused."
43
+ n = len(s)
44
+ best, start, tid = [0.0] + [None] * n, [0] * (n + 1), [0] * (n + 1)
45
+ for i in range(n):
46
+ b, single = best[i], False
47
+ for j in range(i + 1, min(n, i + reach.get(s[i:i + 2], 1)) + 1):
48
+ k = ids.get(s[i:j])
49
+ if k is None: continue
50
+ c = scores[k] + b
51
+ if best[j] is None or c > best[j]: best[j], start[j], tid[j] = c, i, k
52
+ if j == i + 1: single = True
53
+ if not single:
54
+ c = unk_score + b
55
+ if best[i + 1] is None or c > best[i + 1]: best[i + 1], start[i + 1], tid[i + 1] = c, i, unk_id
56
+ out, e, run = [], n, None
57
+ while e > 0:
58
+ if tid[e] == unk_id: run = run or e
59
+ else:
60
+ if run: out += _unk(s[e:run], ids, unk_id, byte_fallback)[::-1]; run = None
61
+ out.append(tid[e])
62
+ e = start[e]
63
+ if run and e == 0: out += _unk(s[:run], ids, unk_id, byte_fallback)[::-1]
64
+ return out[::-1]
65
+
66
+ # %% ../nbs/01_unigram.ipynb #a88855f9
67
+ WS = frozenset('\t\n\x0b\x0c\r \x85\xa0\u1680\u2000\u2001\u2002\u2003\u2004\u2005\u2006\u2007\u2008\u2009\u200a\u2028\u2029\u202f\u205f\u3000') # Rust's char::is_whitespace
68
+
69
+ def _metaspace(pt, scheme):
70
+ "Pre-tokenizer JSON `pt` with every Metaspace's `prepend_scheme: first` set to `scheme`."
71
+ if isinstance(pt, list): return [_metaspace(o, scheme) for o in pt]
72
+ if not isinstance(pt, dict): return pt
73
+ pt = {k: _metaspace(v, scheme) for k, v in pt.items()}
74
+ if pt.get('type') == 'Metaspace' and pt.get('prepend_scheme') == 'first': pt['prepend_scheme'] = scheme
75
+ return pt
76
+
77
+ def _pipeline(d, pre):
78
+ "The normalizer and pre-tokenizer of tokenizer JSON `d`, with pre-tokenizer `pre`, and nothing else of it."
79
+ d = dict(d, pre_tokenizer=pre, added_tokens=[], post_processor=None, decoder=None, truncation=None, padding=None)
80
+ d['model'] = dict(d['model'], vocab=[['<unk>', 0.0]], unk_id=0, byte_fallback=False)
81
+ t = Tokenizer.from_str(json.dumps(d))
82
+ return t.normalizer, t.pre_tokenizer, t
83
+
84
+ class Added:
85
+ "A set of added tokens found in a string as `tokenizers`' `AddedVocabulary.find_matches` finds them."
86
+ def __init__(self, toks, norm=None):
87
+ "`toks` are matched as written, or as `norm` normalizes them when it is given (`normalized` tokens)."
88
+ self.toks = {(norm.normalize_str(o['content']) if norm else o['content']): (o['id'], o.get('lstrip'), o.get('rstrip')) for o in toks}
89
+ self.toks.pop('', None)
90
+ self.rx = re.compile('|'.join(map(re.escape, sorted(self.toks, key=len, reverse=True)))) if self.toks else None
91
+
92
+ def split(self, s):
93
+ "`s` cut into `(text, None)` and `(content, id)` parts, in order, empty text dropped."
94
+ if not self.rx: return [(s, None)] if s else []
95
+ out, prev = [], 0
96
+ for m in self.rx.finditer(s):
97
+ st, sp = m.span(); k, ls, rs = self.toks[m.group()]
98
+ if ls:
99
+ while st > prev and s[st - 1] in WS: st -= 1
100
+ if rs:
101
+ while sp < len(s) and s[sp] in WS: sp += 1
102
+ if st > prev: out.append((s[prev:st], None))
103
+ out.append((s[st:sp], k)); prev = sp
104
+ if prev < len(s): out.append((s[prev:], None))
105
+ return out
106
+
107
+ class Unigram:
108
+ "A Unigram tokenizer.json's `encode_batch(texts, add_special_tokens=False)` ids, with no trie."
109
+ def __init__(self, fn):
110
+ self.ids, self.scores, d = read_vocab(fn)
111
+ mdl = d['model']
112
+ if mdl.get('type', 'Unigram') != 'Unigram': raise NotImplementedError(f"{fn}: model is {mdl.get('type')}")
113
+ added = d.get('added_tokens', [])
114
+ if any(o.get('single_word') for o in added): raise NotImplementedError(f'{fn}: single_word added tokens')
115
+ self.byte_fallback, self.unk_id = bool(mdl.get('byte_fallback')), ifnone(mdl.get('unk_id'), -1) # -1: none, an error if met
116
+ self.trunc, self.pad = d.get('truncation'), d.get('padding')
117
+ self.unk_score = min(self.scores) - UNK_PENALTY
118
+ self.reach = {} # a piece's first two letters → the longest piece that starts so: no longer substring is tried
119
+ for w in self.ids:
120
+ if len(w) > 1 and len(w) > self.reach.get(w[:2], 0): self.reach[w[:2]] = len(w)
121
+ self.added = {o['content']: o['id'] for o in added}
122
+ self.norm, self.pre, t = _pipeline(d, d.get('pre_tokenizer'))
123
+ self.raw, self.nrm = Added([o for o in added if not o.get('normalized')]), Added([o for o in added if o.get('normalized')], self.norm)
124
+ later = _metaspace(d.get('pre_tokenizer'), 'never')
125
+ self.pre_later = self.pre if later == d.get('pre_tokenizer') else _pipeline(d, later)[1]
126
+ if self.nrm.rx and self.pre_later is not self.pre: raise NotImplementedError(f'{fn}: normalized added tokens under prepend_scheme first')
127
+ self.model_unk = getattr(t.model, 'unk_token', None) # model2vec drops this token's id; a Unigram model has none
128
+ extra = [k for k in self.added if k not in self.ids]
129
+ self.n = len(self.ids) + len(extra) # the size of `Tokenizer.get_vocab()`
130
+ lens = np.fromiter((len(k) for k in [*self.ids, *extra]), np.int32, self.n)
131
+ self.median = int(np.median(lens)) if self.n else 0
132
+ self._cached = lru_cache(WORD_CACHE)(lambda q: tuple(self._viterbi(q)))
133
+
134
+ def _viterbi(self, q): return viterbi(q, self.ids, self.scores, self.reach, self.unk_id, self.unk_score, self.byte_fallback)
135
+
136
+ def word(self, q):
137
+ "The ids of pre-token `q`; one of at most `WORD_LEN` characters, as natural text repeats them, from a cache of `WORD_CACHE`."
138
+ return self._cached(q) if len(q) <= WORD_LEN else self._viterbi(q)
139
+
140
+ def _first(self, s):
141
+ "The pre-tokens of the stretch opening the text, offsets kept, so `first` prepends only where the normalized text starts at 0."
142
+ pts = PreTokenizedString(s)
143
+ if self.norm: pts.normalize(self.norm.normalize)
144
+ pts.split(lambda i, ns: [ns[0:len(ns.normalized)]] if ns.normalized else []) # as tokenizers re-slices after its added-token pass
145
+ self.pre.pre_tokenize(pts)
146
+ return [p for p, *_ in pts.get_splits()]
147
+
148
+ def _pieces(self, nz, pt):
149
+ out = []
150
+ for p, k in self.nrm.split(nz):
151
+ if k is not None: out.append(k); continue
152
+ for q in ([q for q, _ in pt.pre_tokenize_str(p)] if pt else [p]):
153
+ out += self.word(q)
154
+ return out
155
+
156
+ def _seg(self, s, first):
157
+ if first and self.pre_later is not self.pre:
158
+ return [k for p in self._first(s) for k in self.word(p)]
159
+ nz = self.norm.normalize_str(s) if self.norm else s
160
+ return self._pieces(nz, self.pre if first else self.pre_later) if nz else []
161
+
162
+ def encode(self, s):
163
+ "The ids of text `s`, before truncation and padding."
164
+ out, at = [], 0
165
+ for p, k in self.raw.split(s):
166
+ out += self._seg(p, at == 0) if k is None else [k]; at += len(p)
167
+ return out
168
+
169
+ def __call__(self, txts):
170
+ "The ids of each of `txts`, as one batch: truncated and padded when tokenizer.json says so."
171
+ out = [self.encode(t) for t in txts]
172
+ if t := self.trunc:
173
+ L, left = t['max_length'], t.get('direction') == 'Left'
174
+ out = [(o[-L:] if L else []) if left else o[:L] for o in out]
175
+ if (p := self.pad) and out:
176
+ st = p.get('strategy')
177
+ n = max(map(len, out)) if st == 'BatchLongest' else (st or {}).get('Fixed', 0)
178
+ if (m := p.get('pad_to_multiple_of')) and n % m: n += m - n % m
179
+ pid, left = p.get('pad_id', 0), p.get('direction') == 'Left'
180
+ out = [([pid] * (n - len(o)) + o if left else o + [pid] * (n - len(o))) if len(o) < n else o for o in out]
181
+ return out
182
+
183
+ @property
184
+ def unk_token_id(self):
185
+ "The id model2vec drops from every encoding, or None."
186
+ return None if self.model_unk is None else self.vocab()[self.model_unk]
187
+
188
+ def vocab(self):
189
+ "`Tokenizer.get_vocab()`: piece → id, the added tokens over the model's."
190
+ return {**self.ids, **self.added}
191
+
192
+ # %% ../nbs/01_unigram.ipynb #4946b03a
193
+ class HFTok:
194
+ "`tokenizers`' own encoding, as model2vec calls it."
195
+ def __init__(self, fn):
196
+ self.tok = Tokenizer.from_file(str(fn))
197
+ v = self.tok.get_vocab()
198
+ self.n, self.median = len(v), int(np.median([len(o) for o in v])) if v else 0
199
+ unk = getattr(self.tok.model, 'unk_token', None)
200
+ self.unk_token_id = v[unk] if unk is not None else None
201
+ self._enc = getattr(self.tok, 'encode_batch_fast', self.tok.encode_batch)
202
+
203
+ def __call__(self, txts):
204
+ "The ids of each of `txts`, one batch as `tokenizers` encodes it (padding, if the file asks for it, is per batch)."
205
+ return [e.ids for e in self._enc(list(txts), add_special_tokens=False)]
206
+
207
+ def vocab(self): return self.tok.get_vocab()
208
+
209
+ def tokenizer(fn, lean=True):
210
+ "A `Unigram` for a Unigram tokenizer.json `fn` when `lean` and it can be done, else an `HFTok`."
211
+ if lean and is_unigram(fn):
212
+ try: return Unigram(fn)
213
+ except NotImplementedError as e: log.info(f'laghu: {e}; tokenizing with tokenizers')
214
+ return HFTok(fn)
@@ -0,0 +1,57 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "laghu"
7
+ dynamic = ["version"]
8
+ description = "model2vec static models in a fraction of the memory, with the same vectors"
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{name = "Karthik", email = "karthik.rajgopal@hotmail.com"}]
14
+ keywords = ['nbdev', 'model2vec', 'embeddings', 'static-embeddings', 'mmap', 'tokenizers', 'unigram']
15
+ classifiers = [
16
+ "Development Status :: 3 - Alpha",
17
+ "Intended Audience :: Developers",
18
+ "Programming Language :: Python :: 3",
19
+ "Programming Language :: Python :: 3 :: Only",
20
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
21
+ ]
22
+ dependencies = [
23
+ "fastcore>=1.8",
24
+ "numpy>=1.24",
25
+ "tokenizers>=0.20",
26
+ "huggingface-hub>=0.24",
27
+ ]
28
+
29
+ [project.optional-dependencies]
30
+ fallback = ["model2vec>=0.9"]
31
+
32
+ [project.urls]
33
+ Repository = "https://github.com/vedicreader/laghu"
34
+ Documentation = "https://vedicreader.github.io/laghu/"
35
+
36
+ [project.entry-points.nbdev]
37
+ laghu = "laghu._modidx:d"
38
+
39
+ [tool.nbdev]
40
+ tst_flags = "slow"
41
+
42
+ [tool.hatch.build.targets.wheel]
43
+ packages = ["laghu"]
44
+
45
+ [tool.hatch.build.targets.sdist]
46
+ include = ["/laghu", "/README.md", "/pyproject.toml"]
47
+
48
+ [tool.hatch.version]
49
+ path = "laghu/__init__.py"
50
+
51
+ [dependency-groups]
52
+ dev = [
53
+ "ipykernel>=7.3.0",
54
+ "model2vec>=0.9",
55
+ "nbdev>=3.3.14",
56
+ "notebook>=7.6.3",
57
+ ]