laghu 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- laghu-0.1.0/.gitignore +158 -0
- laghu-0.1.0/LICENSE +21 -0
- laghu-0.1.0/PKG-INFO +142 -0
- laghu-0.1.0/README.md +118 -0
- laghu-0.1.0/laghu/__init__.py +3 -0
- laghu-0.1.0/laghu/_modidx.py +68 -0
- laghu-0.1.0/laghu/bench.py +80 -0
- laghu-0.1.0/laghu/core.py +147 -0
- laghu-0.1.0/laghu/encoder.py +112 -0
- laghu-0.1.0/laghu/unigram.py +214 -0
- laghu-0.1.0/pyproject.toml +57 -0
laghu-0.1.0/.gitignore
ADDED
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
_docs/
|
|
2
|
+
_proc/
|
|
3
|
+
|
|
4
|
+
*.bak
|
|
5
|
+
.gitattributes
|
|
6
|
+
.last_checked
|
|
7
|
+
.gitconfig
|
|
8
|
+
*.bak
|
|
9
|
+
*.log
|
|
10
|
+
*~
|
|
11
|
+
~*
|
|
12
|
+
_tmp*
|
|
13
|
+
tmp*
|
|
14
|
+
tags
|
|
15
|
+
*.pkg
|
|
16
|
+
|
|
17
|
+
# Byte-compiled / optimized / DLL files
|
|
18
|
+
__pycache__/
|
|
19
|
+
*.py[cod]
|
|
20
|
+
*$py.class
|
|
21
|
+
|
|
22
|
+
# C extensions
|
|
23
|
+
*.so
|
|
24
|
+
|
|
25
|
+
# Distribution / packaging
|
|
26
|
+
.Python
|
|
27
|
+
env/
|
|
28
|
+
build/
|
|
29
|
+
conda/
|
|
30
|
+
develop-eggs/
|
|
31
|
+
dist/
|
|
32
|
+
downloads/
|
|
33
|
+
eggs/
|
|
34
|
+
.eggs/
|
|
35
|
+
lib/
|
|
36
|
+
lib64/
|
|
37
|
+
parts/
|
|
38
|
+
sdist/
|
|
39
|
+
var/
|
|
40
|
+
wheels/
|
|
41
|
+
*.egg-info/
|
|
42
|
+
.installed.cfg
|
|
43
|
+
*.egg
|
|
44
|
+
|
|
45
|
+
# PyInstaller
|
|
46
|
+
# Usually these files are written by a python script from a template
|
|
47
|
+
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
|
48
|
+
*.manifest
|
|
49
|
+
*.spec
|
|
50
|
+
|
|
51
|
+
# Installer logs
|
|
52
|
+
pip-log.txt
|
|
53
|
+
pip-delete-this-directory.txt
|
|
54
|
+
|
|
55
|
+
# Unit test / coverage reports
|
|
56
|
+
htmlcov/
|
|
57
|
+
.tox/
|
|
58
|
+
.coverage
|
|
59
|
+
.coverage.*
|
|
60
|
+
.cache
|
|
61
|
+
nosetests.xml
|
|
62
|
+
coverage.xml
|
|
63
|
+
*.cover
|
|
64
|
+
.hypothesis/
|
|
65
|
+
|
|
66
|
+
# Translations
|
|
67
|
+
*.mo
|
|
68
|
+
*.pot
|
|
69
|
+
|
|
70
|
+
# Django stuff:
|
|
71
|
+
*.log
|
|
72
|
+
local_settings.py
|
|
73
|
+
|
|
74
|
+
# Flask stuff:
|
|
75
|
+
instance/
|
|
76
|
+
.webassets-cache
|
|
77
|
+
|
|
78
|
+
# Scrapy stuff:
|
|
79
|
+
.scrapy
|
|
80
|
+
|
|
81
|
+
# Sphinx documentation
|
|
82
|
+
docs/_build/
|
|
83
|
+
|
|
84
|
+
# PyBuilder
|
|
85
|
+
target/
|
|
86
|
+
|
|
87
|
+
# Jupyter Notebook
|
|
88
|
+
.ipynb_checkpoints
|
|
89
|
+
|
|
90
|
+
# pyenv
|
|
91
|
+
.python-version
|
|
92
|
+
|
|
93
|
+
# celery beat schedule file
|
|
94
|
+
celerybeat-schedule
|
|
95
|
+
|
|
96
|
+
# SageMath parsed files
|
|
97
|
+
*.sage.py
|
|
98
|
+
|
|
99
|
+
# dotenv
|
|
100
|
+
.env
|
|
101
|
+
|
|
102
|
+
# virtualenv
|
|
103
|
+
.venv
|
|
104
|
+
venv/
|
|
105
|
+
ENV/
|
|
106
|
+
|
|
107
|
+
# Spyder project settings
|
|
108
|
+
.spyderproject
|
|
109
|
+
.spyproject
|
|
110
|
+
|
|
111
|
+
# Rope project settings
|
|
112
|
+
.ropeproject
|
|
113
|
+
|
|
114
|
+
# mkdocs documentation
|
|
115
|
+
/site
|
|
116
|
+
|
|
117
|
+
# mypy
|
|
118
|
+
.mypy_cache/
|
|
119
|
+
|
|
120
|
+
.vscode
|
|
121
|
+
*.swp
|
|
122
|
+
|
|
123
|
+
# osx generated files
|
|
124
|
+
.DS_Store
|
|
125
|
+
.DS_Store?
|
|
126
|
+
.Trashes
|
|
127
|
+
ehthumbs.db
|
|
128
|
+
Thumbs.db
|
|
129
|
+
.idea
|
|
130
|
+
|
|
131
|
+
# pytest
|
|
132
|
+
.pytest_cache
|
|
133
|
+
|
|
134
|
+
# tools/trust-doc-nbs
|
|
135
|
+
docs_src/.last_checked
|
|
136
|
+
|
|
137
|
+
# symlinks to fastai
|
|
138
|
+
docs_src/fastai
|
|
139
|
+
tools/fastai
|
|
140
|
+
|
|
141
|
+
# link checker
|
|
142
|
+
checklink/cookies.txt
|
|
143
|
+
|
|
144
|
+
# .gitconfig is now autogenerated
|
|
145
|
+
.gitconfig
|
|
146
|
+
|
|
147
|
+
# Quarto installer
|
|
148
|
+
.deb
|
|
149
|
+
.pkg
|
|
150
|
+
|
|
151
|
+
# Quarto
|
|
152
|
+
.quarto
|
|
153
|
+
|
|
154
|
+
.venv/
|
|
155
|
+
.kosha/
|
|
156
|
+
_docs/
|
|
157
|
+
_proc/
|
|
158
|
+
nbs/_fixtures/
|
laghu-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Karthik
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
laghu-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: laghu
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: model2vec static models in a fraction of the memory, with the same vectors
|
|
5
|
+
Project-URL: Repository, https://github.com/vedicreader/laghu
|
|
6
|
+
Project-URL: Documentation, https://vedicreader.github.io/laghu/
|
|
7
|
+
Author-email: Karthik <karthik.rajgopal@hotmail.com>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: embeddings,mmap,model2vec,nbdev,static-embeddings,tokenizers,unigram
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
15
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
16
|
+
Requires-Python: >=3.10
|
|
17
|
+
Requires-Dist: fastcore>=1.8
|
|
18
|
+
Requires-Dist: huggingface-hub>=0.24
|
|
19
|
+
Requires-Dist: numpy>=1.24
|
|
20
|
+
Requires-Dist: tokenizers>=0.20
|
|
21
|
+
Provides-Extra: fallback
|
|
22
|
+
Requires-Dist: model2vec>=0.9; extra == 'fallback'
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
|
|
25
|
+
# laghu
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
<!-- WARNING: THIS FILE WAS AUTOGENERATED! DO NOT EDIT! -->
|
|
29
|
+
|
|
30
|
+
laghu (Sanskrit for light) loads a [model2vec](https://github.com/MinishLab/model2vec) static embedding model with `encode`, `encode_as_sequence` and `tokenize`. The output matches model2vec’s bits, apart from the empty-input difference documented below.
|
|
31
|
+
|
|
32
|
+
model2vec reads the embedding table into each process. For potion-multilingual-128M, the table takes 512 MB. Its Unigram tokenizer adds a character trie over 500,353 pieces, about another 610 MB. After encoding, the process uses about 1.3 GB of private memory.
|
|
33
|
+
|
|
34
|
+
laghu maps the table read-only from `model.safetensors`. Processes share the mapped pages through the operating system’s page cache. The operating system can reclaim clean pages under memory pressure. For supported Unigram models, laghu replaces the trie with Python Viterbi over a dict of pieces. The normalizer and pre-tokenizer come from `tokenizers`, built from the same tokenizer.json with the vocabulary removed. The same model then takes about 135 MB of private memory per process.
|
|
35
|
+
|
|
36
|
+
WordPiece, BPE and word-level models still use `tokenizers`. laghu maps their embedding tables but does not replace their tokenizers. The memory saving depends on the table’s size relative to the tokenizer.
|
|
37
|
+
|
|
38
|
+
## Install
|
|
39
|
+
|
|
40
|
+
``` sh
|
|
41
|
+
uv add laghu # or: pip install laghu
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
model2vec itself is not a dependency. For the fallback described below, and for [`same_as_model2vec`](https://vedicreader.github.io/laghu/encoder.html#same_as_model2vec), install the extra:
|
|
45
|
+
`uv add 'laghu[fallback]'`.
|
|
46
|
+
|
|
47
|
+
## Use
|
|
48
|
+
|
|
49
|
+
``` python
|
|
50
|
+
from laghu import load
|
|
51
|
+
|
|
52
|
+
m = load('minishlab/potion-multilingual-128M') # a folder, or a Hugging Face repo id (the local cache is tried first)
|
|
53
|
+
v = m.encode(['The quick brown fox.', 'कर्मण्येवाधिकारस्ते']) # float32, (2, 256), unit rows: the same array model2vec returns
|
|
54
|
+
m.tokenize(['hello world'])
|
|
55
|
+
m.encode_as_sequence('hello world').shape # (n_tokens, 256)
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
(2, 256)
|
|
59
|
+
|
|
60
|
+
[`load`](https://vedicreader.github.io/laghu/encoder.html#load) returns a [`LeanModel`](https://vedicreader.github.io/laghu/encoder.html#leanmodel). If loading fails, it logs the cause and tries model2vec’s `StaticModel`. This fallback needs `laghu[fallback]`. If the fallback also fails, [`load`](https://vedicreader.github.io/laghu/encoder.html#load) raises the original laghu error. `load(..., fallback=False)` raises without trying model2vec. Both loaders refuse bfloat16 and float8 tables because numpy cannot hold those dtypes.
|
|
61
|
+
|
|
62
|
+
`same_as_model2vec(path)` loads a model both ways and compares them on the built-in corpus, or on texts of your own:
|
|
63
|
+
|
|
64
|
+
``` python
|
|
65
|
+
from laghu import same_as_model2vec
|
|
66
|
+
same_as_model2vec('minishlab/potion-base-8M', ['my', 'own texts'])
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
/Users/71293/code/personal/orgs/laghu/.venv/lib/python3.13/site-packages/tqdm/auto.py:21: TqdmWarning: IProgress not found. Please update jupyter and ipywidgets. See https://ipywidgets.readthedocs.io/en/stable/user_install.html
|
|
70
|
+
from .autonotebook import tqdm as notebook_tqdm
|
|
71
|
+
|
|
72
|
+
{'lean': False,
|
|
73
|
+
'n': 2,
|
|
74
|
+
'ids': True,
|
|
75
|
+
'dtype': True,
|
|
76
|
+
'f32': True,
|
|
77
|
+
'f16': True,
|
|
78
|
+
'exact': True,
|
|
79
|
+
'seq': True,
|
|
80
|
+
'median': True,
|
|
81
|
+
'unk': True,
|
|
82
|
+
'tokens': True}
|
|
83
|
+
|
|
84
|
+
It checks the token ids, the vectors cast to float32 and to float16, the exact dtype and bits, `encode_as_sequence`,
|
|
85
|
+
`median_token_length`, the unknown token id and the token list.
|
|
86
|
+
|
|
87
|
+
## What matches, and what it costs
|
|
88
|
+
|
|
89
|
+
Every model below matches model2vec’s ids and vectors, bit for bit, on 4,726 texts. The corpus includes prose in 36 languages and random strings from 30 Unicode blocks. It also tests control characters, special tokens, emoji sequences, combining marks, bidi controls and a 200,000-character string. Each memory measurement uses a fresh process. These macOS measurements use `phys_footprint` after encoding and `lifetime_max_phys_footprint` for peak. On Linux, the helper reports `RssAnon` and peak resident memory (`VmHWM`) instead. Speed is `encode` over 3,226 texts on an M3 Max.
|
|
90
|
+
|
|
91
|
+
| model | tokenizer | laghu tokenizes with | vocab × dims | dtype | norm | identical | private MB, model2vec → laghu | peak MB, model2vec → laghu | texts/s, model2vec → laghu |
|
|
92
|
+
|----|----|----|----|----|----|----|----|----|----|
|
|
93
|
+
| potion-base-2M | WordPiece | tokenizers | 29,528 × 64 | float32 | on | yes | 81 → 58 | 82 → 58 | 84,510 → 92,982 |
|
|
94
|
+
| potion-base-4M | WordPiece | tokenizers | 29,528 × 128 | float32 | on | yes | 94 → 62 | 95 → 62 | 87,248 → 90,242 |
|
|
95
|
+
| potion-base-8M | WordPiece | tokenizers | 29,528 × 256 | float32 | on | yes | 114 → 70 | 115 → 70 | 79,415 → 81,810 |
|
|
96
|
+
| potion-base-32M | WordPiece | tokenizers | 63,091 × 512 | float32 | on | yes | 244 → 97 | 244 → 97 | 75,009 → 81,379 |
|
|
97
|
+
| potion-retrieval-32M | WordPiece | tokenizers | 63,091 × 512 | float32 | on | yes | 244 → 97 | 244 → 97 | 70,799 → 79,717 |
|
|
98
|
+
| potion-science-8M | WordPiece | tokenizers | 29,528 × 256 | float32 | on | yes | 113 → 70 | 114 → 70 | 81,008 → 84,532 |
|
|
99
|
+
| potion-code-16M | WordPiece | tokenizers | 61,826 × 256 | float32 +map +weights | on | yes | 147 → 110 | 211 → 117 | 10,674 → 10,425 |
|
|
100
|
+
| potion-code-16M-v2 | WordPiece | tokenizers | 63,457 × 256 | float16 | on | yes | 108 → 101 | 186 → 119 | 21,819 → 22,164 |
|
|
101
|
+
| potion-multilingual-128M | Unigram | Python Viterbi | 500,353 × 256 | float32 | on | yes | 1,348 → 135 | 1,348 → 135 | 39,105 → 12,795 |
|
|
102
|
+
| M2V_base_output | WordPiece | tokenizers | 29,528 × 256 | float32 | off | yes | 112 → 69 | 113 → 69 | 65,460 → 87,100 |
|
|
103
|
+
| M2V_multilingual_output | WordPiece | tokenizers | 501,054 × 256 | float32 | off | yes | 834 → 198 | 842 → 295 | 42,565 → 90,283 |
|
|
104
|
+
| M2V_base_glove | WordLevel | tokenizers | 400,002 × 256 | float32 | off | yes | 656 → 186 | 656 → 194 | 67,028 → 85,058 |
|
|
105
|
+
| M2V_base_glove_subword | WordPiece | tokenizers | 400,779 × 256 | float32 | off | yes | 647 → 178 | 649 → 195 | 66,256 → 85,430 |
|
|
106
|
+
| jina-embeddings-v3-separation-distilled | Unigram | Python Viterbi | 250,002 × 256 | float32 | off | yes | 722 → 88 | 722 → 88 | 51,836 → 20,840 |
|
|
107
|
+
| multilingual-e5-large-m2v | Unigram | Python Viterbi | 250,002 × 512 | float32 | on | yes | 981 → 109 | 981 → 109 | 24,139 → 19,670 |
|
|
108
|
+
| m2v-gemma-embedding-300m | Unigram | Python Viterbi | 255,732 × 256 | float16 +weights | on | yes | 542 → 93 | 542 → 93 | 20,319 → 12,134 |
|
|
109
|
+
| potion-multilingual-128M-i8-tokenade | Unigram | Python Viterbi | 500,353 × 256 | int8 | on | yes | 985 → 138 | 985 → 138 | 27,676 → 12,090 |
|
|
110
|
+
| m2v-qwen3-embedding-0.6b-1024d | Unigram | Python Viterbi | 151,644 × 1024 | float16 +weights | on | yes | 602 → 132 | 602 → 132 | 16,792 → 7,168 |
|
|
111
|
+
| ovos-m2v-intents-es-ES-bne-v5 | Unigram | Python Viterbi | 50,259 × 64 | float32 | on | yes | 169 → 52 | 169 → 52 | 11,786 → 11,265 |
|
|
112
|
+
| m2v-gte-256-edu | Unigram | tokenizers | 313,845 × 256 | float32 | on | yes | 1,028 → 663 | 1,028 → 675 | 25,363 → 24,981 |
|
|
113
|
+
| bge-m3-m2v-256 | Unigram | Python Viterbi | 250,002 × 256 | float32 | off | yes | 725 → 89 | 725 → 89 | 34,518 → 20,670 |
|
|
114
|
+
| potion-mxbai-micro | WordPiece | tokenizers | 2,000 × 256 | int8 +map +weights | on | yes | 90 → 72 | 91 → 72 | 68,385 → 70,998 |
|
|
115
|
+
| NIFE-mxbai-embed-large-v1_model2vec | WordPiece | tokenizers | 107,962 × 1024 | float32 | off | yes | 558 → 106 | 572 → 116 | 41,042 → 26,756 |
|
|
116
|
+
| ovos-m2v-intents-pt-PT-albertina-v5 | BPE | tokenizers | 50,262 × 256 | float32 | on | yes | 162 → 90 | 162 → 90 | 46,343 → 22,813 |
|
|
117
|
+
| F2LLM-v2-330M-model2vec | BPE | tokenizers | 151,644 × 512 | float16 | on | yes | 356 → 171 | 356 → 171 | 28,795 → 40,132 |
|
|
118
|
+
| static-retrieval-multilingual-69m-v1 | BPE | tokenizers | 179,936 × 384 | float32 | off | yes | 537 → 229 | 541 → 263 | 32,235 → 31,848 |
|
|
119
|
+
|
|
120
|
+
Unigram throughput ranges from about a third of model2vec’s rate to about the same. `tokenizers` parallelizes batches across cores. Python Viterbi runs on one core. These batch measurements do not establish latency for one query.
|
|
121
|
+
|
|
122
|
+
The word cache helps when the pre-tokenizer splits words, as in jina-v3, bge-m3 and multilingual-e5. Natural text runs 1.6 to 1.9 times faster on a first pass and 3 to 5 times faster on repeated text. The `bench` notebook records these measurements and the reasons for rejecting tokie and gigatoken.
|
|
123
|
+
|
|
124
|
+
Other models use the same `tokenizers` code in both loaders. The timing samples last about 0.1 seconds and vary by up to half on repeated runs. They do not establish a speed advantage.
|
|
125
|
+
|
|
126
|
+
## What is not supported
|
|
127
|
+
|
|
128
|
+
- Quantizing or truncating on load (`quantize_to`, `dimensionality`, `vocabulary_quantization`): each would copy the table, which is
|
|
129
|
+
what laghu exists to avoid. Load with model2vec for those.
|
|
130
|
+
- `single_word` added tokens on a Unigram model fall back to `tokenizers` for tokenizing (the table is still mapped). Rust decides a
|
|
131
|
+
word boundary by Unicode’s Alphabetic property, which Python’s `str.isalpha` does not match.
|
|
132
|
+
- `use_multiprocessing` is accepted and ignored; the result is the same.
|
|
133
|
+
- `encode([])` returns an empty `(0, dim)` array. model2vec raises.
|
|
134
|
+
|
|
135
|
+
## Developing
|
|
136
|
+
|
|
137
|
+
nbdev 3 on hatchling: the notebooks in `nbs/` are the source. `uv run nbdev-prepare` exports, tests and rebuilds this README;
|
|
138
|
+
`uv run nbdev-test --flags slow` also runs the bench, which downloads about 5 GB of models.
|
|
139
|
+
|
|
140
|
+
## Licence
|
|
141
|
+
|
|
142
|
+
MIT.
|
laghu-0.1.0/README.md
ADDED
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
# laghu
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
<!-- WARNING: THIS FILE WAS AUTOGENERATED! DO NOT EDIT! -->
|
|
5
|
+
|
|
6
|
+
laghu (Sanskrit for light) loads a [model2vec](https://github.com/MinishLab/model2vec) static embedding model with `encode`, `encode_as_sequence` and `tokenize`. The output matches model2vec’s bits, apart from the empty-input difference documented below.
|
|
7
|
+
|
|
8
|
+
model2vec reads the embedding table into each process. For potion-multilingual-128M, the table takes 512 MB. Its Unigram tokenizer adds a character trie over 500,353 pieces, about another 610 MB. After encoding, the process uses about 1.3 GB of private memory.
|
|
9
|
+
|
|
10
|
+
laghu maps the table read-only from `model.safetensors`. Processes share the mapped pages through the operating system’s page cache. The operating system can reclaim clean pages under memory pressure. For supported Unigram models, laghu replaces the trie with Python Viterbi over a dict of pieces. The normalizer and pre-tokenizer come from `tokenizers`, built from the same tokenizer.json with the vocabulary removed. The same model then takes about 135 MB of private memory per process.
|
|
11
|
+
|
|
12
|
+
WordPiece, BPE and word-level models still use `tokenizers`. laghu maps their embedding tables but does not replace their tokenizers. The memory saving depends on the table’s size relative to the tokenizer.
|
|
13
|
+
|
|
14
|
+
## Install
|
|
15
|
+
|
|
16
|
+
``` sh
|
|
17
|
+
uv add laghu # or: pip install laghu
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
model2vec itself is not a dependency. For the fallback described below, and for [`same_as_model2vec`](https://vedicreader.github.io/laghu/encoder.html#same_as_model2vec), install the extra:
|
|
21
|
+
`uv add 'laghu[fallback]'`.
|
|
22
|
+
|
|
23
|
+
## Use
|
|
24
|
+
|
|
25
|
+
``` python
|
|
26
|
+
from laghu import load
|
|
27
|
+
|
|
28
|
+
m = load('minishlab/potion-multilingual-128M') # a folder, or a Hugging Face repo id (the local cache is tried first)
|
|
29
|
+
v = m.encode(['The quick brown fox.', 'कर्मण्येवाधिकारस्ते']) # float32, (2, 256), unit rows: the same array model2vec returns
|
|
30
|
+
m.tokenize(['hello world'])
|
|
31
|
+
m.encode_as_sequence('hello world').shape # (n_tokens, 256)
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
(2, 256)
|
|
35
|
+
|
|
36
|
+
[`load`](https://vedicreader.github.io/laghu/encoder.html#load) returns a [`LeanModel`](https://vedicreader.github.io/laghu/encoder.html#leanmodel). If loading fails, it logs the cause and tries model2vec’s `StaticModel`. This fallback needs `laghu[fallback]`. If the fallback also fails, [`load`](https://vedicreader.github.io/laghu/encoder.html#load) raises the original laghu error. `load(..., fallback=False)` raises without trying model2vec. Both loaders refuse bfloat16 and float8 tables because numpy cannot hold those dtypes.
|
|
37
|
+
|
|
38
|
+
`same_as_model2vec(path)` loads a model both ways and compares them on the built-in corpus, or on texts of your own:
|
|
39
|
+
|
|
40
|
+
``` python
|
|
41
|
+
from laghu import same_as_model2vec
|
|
42
|
+
same_as_model2vec('minishlab/potion-base-8M', ['my', 'own texts'])
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
/Users/71293/code/personal/orgs/laghu/.venv/lib/python3.13/site-packages/tqdm/auto.py:21: TqdmWarning: IProgress not found. Please update jupyter and ipywidgets. See https://ipywidgets.readthedocs.io/en/stable/user_install.html
|
|
46
|
+
from .autonotebook import tqdm as notebook_tqdm
|
|
47
|
+
|
|
48
|
+
{'lean': False,
|
|
49
|
+
'n': 2,
|
|
50
|
+
'ids': True,
|
|
51
|
+
'dtype': True,
|
|
52
|
+
'f32': True,
|
|
53
|
+
'f16': True,
|
|
54
|
+
'exact': True,
|
|
55
|
+
'seq': True,
|
|
56
|
+
'median': True,
|
|
57
|
+
'unk': True,
|
|
58
|
+
'tokens': True}
|
|
59
|
+
|
|
60
|
+
It checks the token ids, the vectors cast to float32 and to float16, the exact dtype and bits, `encode_as_sequence`,
|
|
61
|
+
`median_token_length`, the unknown token id and the token list.
|
|
62
|
+
|
|
63
|
+
## What matches, and what it costs
|
|
64
|
+
|
|
65
|
+
Every model below matches model2vec’s ids and vectors, bit for bit, on 4,726 texts. The corpus includes prose in 36 languages and random strings from 30 Unicode blocks. It also tests control characters, special tokens, emoji sequences, combining marks, bidi controls and a 200,000-character string. Each memory measurement uses a fresh process. These macOS measurements use `phys_footprint` after encoding and `lifetime_max_phys_footprint` for peak. On Linux, the helper reports `RssAnon` and peak resident memory (`VmHWM`) instead. Speed is `encode` over 3,226 texts on an M3 Max.
|
|
66
|
+
|
|
67
|
+
| model | tokenizer | laghu tokenizes with | vocab × dims | dtype | norm | identical | private MB, model2vec → laghu | peak MB, model2vec → laghu | texts/s, model2vec → laghu |
|
|
68
|
+
|----|----|----|----|----|----|----|----|----|----|
|
|
69
|
+
| potion-base-2M | WordPiece | tokenizers | 29,528 × 64 | float32 | on | yes | 81 → 58 | 82 → 58 | 84,510 → 92,982 |
|
|
70
|
+
| potion-base-4M | WordPiece | tokenizers | 29,528 × 128 | float32 | on | yes | 94 → 62 | 95 → 62 | 87,248 → 90,242 |
|
|
71
|
+
| potion-base-8M | WordPiece | tokenizers | 29,528 × 256 | float32 | on | yes | 114 → 70 | 115 → 70 | 79,415 → 81,810 |
|
|
72
|
+
| potion-base-32M | WordPiece | tokenizers | 63,091 × 512 | float32 | on | yes | 244 → 97 | 244 → 97 | 75,009 → 81,379 |
|
|
73
|
+
| potion-retrieval-32M | WordPiece | tokenizers | 63,091 × 512 | float32 | on | yes | 244 → 97 | 244 → 97 | 70,799 → 79,717 |
|
|
74
|
+
| potion-science-8M | WordPiece | tokenizers | 29,528 × 256 | float32 | on | yes | 113 → 70 | 114 → 70 | 81,008 → 84,532 |
|
|
75
|
+
| potion-code-16M | WordPiece | tokenizers | 61,826 × 256 | float32 +map +weights | on | yes | 147 → 110 | 211 → 117 | 10,674 → 10,425 |
|
|
76
|
+
| potion-code-16M-v2 | WordPiece | tokenizers | 63,457 × 256 | float16 | on | yes | 108 → 101 | 186 → 119 | 21,819 → 22,164 |
|
|
77
|
+
| potion-multilingual-128M | Unigram | Python Viterbi | 500,353 × 256 | float32 | on | yes | 1,348 → 135 | 1,348 → 135 | 39,105 → 12,795 |
|
|
78
|
+
| M2V_base_output | WordPiece | tokenizers | 29,528 × 256 | float32 | off | yes | 112 → 69 | 113 → 69 | 65,460 → 87,100 |
|
|
79
|
+
| M2V_multilingual_output | WordPiece | tokenizers | 501,054 × 256 | float32 | off | yes | 834 → 198 | 842 → 295 | 42,565 → 90,283 |
|
|
80
|
+
| M2V_base_glove | WordLevel | tokenizers | 400,002 × 256 | float32 | off | yes | 656 → 186 | 656 → 194 | 67,028 → 85,058 |
|
|
81
|
+
| M2V_base_glove_subword | WordPiece | tokenizers | 400,779 × 256 | float32 | off | yes | 647 → 178 | 649 → 195 | 66,256 → 85,430 |
|
|
82
|
+
| jina-embeddings-v3-separation-distilled | Unigram | Python Viterbi | 250,002 × 256 | float32 | off | yes | 722 → 88 | 722 → 88 | 51,836 → 20,840 |
|
|
83
|
+
| multilingual-e5-large-m2v | Unigram | Python Viterbi | 250,002 × 512 | float32 | on | yes | 981 → 109 | 981 → 109 | 24,139 → 19,670 |
|
|
84
|
+
| m2v-gemma-embedding-300m | Unigram | Python Viterbi | 255,732 × 256 | float16 +weights | on | yes | 542 → 93 | 542 → 93 | 20,319 → 12,134 |
|
|
85
|
+
| potion-multilingual-128M-i8-tokenade | Unigram | Python Viterbi | 500,353 × 256 | int8 | on | yes | 985 → 138 | 985 → 138 | 27,676 → 12,090 |
|
|
86
|
+
| m2v-qwen3-embedding-0.6b-1024d | Unigram | Python Viterbi | 151,644 × 1024 | float16 +weights | on | yes | 602 → 132 | 602 → 132 | 16,792 → 7,168 |
|
|
87
|
+
| ovos-m2v-intents-es-ES-bne-v5 | Unigram | Python Viterbi | 50,259 × 64 | float32 | on | yes | 169 → 52 | 169 → 52 | 11,786 → 11,265 |
|
|
88
|
+
| m2v-gte-256-edu | Unigram | tokenizers | 313,845 × 256 | float32 | on | yes | 1,028 → 663 | 1,028 → 675 | 25,363 → 24,981 |
|
|
89
|
+
| bge-m3-m2v-256 | Unigram | Python Viterbi | 250,002 × 256 | float32 | off | yes | 725 → 89 | 725 → 89 | 34,518 → 20,670 |
|
|
90
|
+
| potion-mxbai-micro | WordPiece | tokenizers | 2,000 × 256 | int8 +map +weights | on | yes | 90 → 72 | 91 → 72 | 68,385 → 70,998 |
|
|
91
|
+
| NIFE-mxbai-embed-large-v1_model2vec | WordPiece | tokenizers | 107,962 × 1024 | float32 | off | yes | 558 → 106 | 572 → 116 | 41,042 → 26,756 |
|
|
92
|
+
| ovos-m2v-intents-pt-PT-albertina-v5 | BPE | tokenizers | 50,262 × 256 | float32 | on | yes | 162 → 90 | 162 → 90 | 46,343 → 22,813 |
|
|
93
|
+
| F2LLM-v2-330M-model2vec | BPE | tokenizers | 151,644 × 512 | float16 | on | yes | 356 → 171 | 356 → 171 | 28,795 → 40,132 |
|
|
94
|
+
| static-retrieval-multilingual-69m-v1 | BPE | tokenizers | 179,936 × 384 | float32 | off | yes | 537 → 229 | 541 → 263 | 32,235 → 31,848 |
|
|
95
|
+
|
|
96
|
+
Unigram throughput ranges from about a third of model2vec’s rate to about the same. `tokenizers` parallelizes batches across cores. Python Viterbi runs on one core. These batch measurements do not establish latency for one query.
|
|
97
|
+
|
|
98
|
+
The word cache helps when the pre-tokenizer splits words, as in jina-v3, bge-m3 and multilingual-e5. Natural text runs 1.6 to 1.9 times faster on a first pass and 3 to 5 times faster on repeated text. The `bench` notebook records these measurements and the reasons for rejecting tokie and gigatoken.
|
|
99
|
+
|
|
100
|
+
Other models use the same `tokenizers` code in both loaders. The timing samples last about 0.1 seconds and vary by up to half on repeated runs. They do not establish a speed advantage.
|
|
101
|
+
|
|
102
|
+
## What is not supported
|
|
103
|
+
|
|
104
|
+
- Quantizing or truncating on load (`quantize_to`, `dimensionality`, `vocabulary_quantization`): each would copy the table, which is
|
|
105
|
+
what laghu exists to avoid. Load with model2vec for those.
|
|
106
|
+
- `single_word` added tokens on a Unigram model fall back to `tokenizers` for tokenizing (the table is still mapped). Rust decides a
|
|
107
|
+
word boundary by Unicode’s Alphabetic property, which Python’s `str.isalpha` does not match.
|
|
108
|
+
- `use_multiprocessing` is accepted and ignored; the result is the same.
|
|
109
|
+
- `encode([])` returns an empty `(0, dim)` array. model2vec raises.
|
|
110
|
+
|
|
111
|
+
## Developing
|
|
112
|
+
|
|
113
|
+
nbdev 3 on hatchling: the notebooks in `nbs/` are the source. `uv run nbdev-prepare` exports, tests and rebuilds this README;
|
|
114
|
+
`uv run nbdev-test --flags slow` also runs the bench, which downloads about 5 GB of models.
|
|
115
|
+
|
|
116
|
+
## Licence
|
|
117
|
+
|
|
118
|
+
MIT.
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
# Autogenerated by nbdev
|
|
2
|
+
|
|
3
|
+
d = { 'settings': { 'branch': 'main',
|
|
4
|
+
'doc_baseurl': '/laghu',
|
|
5
|
+
'doc_host': 'https://vedicreader.github.io',
|
|
6
|
+
'git_url': 'https://github.com/vedicreader/laghu',
|
|
7
|
+
'lib_path': 'laghu'},
|
|
8
|
+
'syms': { 'laghu.bench': { 'laghu.bench._run': ('bench.html#_run', 'laghu/bench.py'),
|
|
9
|
+
'laghu.bench.check': ('bench.html#check', 'laghu/bench.py'),
|
|
10
|
+
'laghu.bench.footprint': ('bench.html#footprint', 'laghu/bench.py'),
|
|
11
|
+
'laghu.bench.measure': ('bench.html#measure', 'laghu/bench.py'),
|
|
12
|
+
'laghu.bench.row': ('bench.html#row', 'laghu/bench.py'),
|
|
13
|
+
'laghu.bench.same': ('bench.html#same', 'laghu/bench.py'),
|
|
14
|
+
'laghu.bench.table': ('bench.html#table', 'laghu/bench.py')},
|
|
15
|
+
'laghu.core': { 'laghu.core.UnsupportedDtype': ('core.html#unsupporteddtype', 'laghu/core.py'),
|
|
16
|
+
'laghu.core.cached': ('core.html#cached', 'laghu/core.py'),
|
|
17
|
+
'laghu.core.corpus': ('core.html#corpus', 'laghu/core.py'),
|
|
18
|
+
'laghu.core.find_vocab': ('core.html#find_vocab', 'laghu/core.py'),
|
|
19
|
+
'laghu.core.is_unigram': ('core.html#is_unigram', 'laghu/core.py'),
|
|
20
|
+
'laghu.core.layout': ('core.html#layout', 'laghu/core.py'),
|
|
21
|
+
'laghu.core.model_files': ('core.html#model_files', 'laghu/core.py'),
|
|
22
|
+
'laghu.core.noise': ('core.html#noise', 'laghu/core.py'),
|
|
23
|
+
'laghu.core.resolve': ('core.html#resolve', 'laghu/core.py'),
|
|
24
|
+
'laghu.core.st_header': ('core.html#st_header', 'laghu/core.py'),
|
|
25
|
+
'laghu.core.st_map': ('core.html#st_map', 'laghu/core.py'),
|
|
26
|
+
'laghu.core.st_write': ('core.html#st_write', 'laghu/core.py'),
|
|
27
|
+
'laghu.core.try_resolve': ('core.html#try_resolve', 'laghu/core.py')},
|
|
28
|
+
'laghu.encoder': { 'laghu.encoder.LeanModel': ('encoder.html#leanmodel', 'laghu/encoder.py'),
|
|
29
|
+
'laghu.encoder.LeanModel.__init__': ('encoder.html#leanmodel.__init__', 'laghu/encoder.py'),
|
|
30
|
+
'laghu.encoder.LeanModel.__repr__': ('encoder.html#leanmodel.__repr__', 'laghu/encoder.py'),
|
|
31
|
+
'laghu.encoder.LeanModel._encode_batch': ('encoder.html#leanmodel._encode_batch', 'laghu/encoder.py'),
|
|
32
|
+
'laghu.encoder.LeanModel._rows': ('encoder.html#leanmodel._rows', 'laghu/encoder.py'),
|
|
33
|
+
'laghu.encoder.LeanModel.dim': ('encoder.html#leanmodel.dim', 'laghu/encoder.py'),
|
|
34
|
+
'laghu.encoder.LeanModel.embedding_dtype': ('encoder.html#leanmodel.embedding_dtype', 'laghu/encoder.py'),
|
|
35
|
+
'laghu.encoder.LeanModel.encode': ('encoder.html#leanmodel.encode', 'laghu/encoder.py'),
|
|
36
|
+
'laghu.encoder.LeanModel.encode_as_sequence': ( 'encoder.html#leanmodel.encode_as_sequence',
|
|
37
|
+
'laghu/encoder.py'),
|
|
38
|
+
'laghu.encoder.LeanModel.from_pretrained': ('encoder.html#leanmodel.from_pretrained', 'laghu/encoder.py'),
|
|
39
|
+
'laghu.encoder.LeanModel.lean': ('encoder.html#leanmodel.lean', 'laghu/encoder.py'),
|
|
40
|
+
'laghu.encoder.LeanModel.tokenize': ('encoder.html#leanmodel.tokenize', 'laghu/encoder.py'),
|
|
41
|
+
'laghu.encoder.LeanModel.tokens': ('encoder.html#leanmodel.tokens', 'laghu/encoder.py'),
|
|
42
|
+
'laghu.encoder._batches': ('encoder.html#_batches', 'laghu/encoder.py'),
|
|
43
|
+
'laghu.encoder.load': ('encoder.html#load', 'laghu/encoder.py'),
|
|
44
|
+
'laghu.encoder.same_as_model2vec': ('encoder.html#same_as_model2vec', 'laghu/encoder.py')},
|
|
45
|
+
'laghu.unigram': { 'laghu.unigram.Added': ('unigram.html#added', 'laghu/unigram.py'),
|
|
46
|
+
'laghu.unigram.Added.__init__': ('unigram.html#added.__init__', 'laghu/unigram.py'),
|
|
47
|
+
'laghu.unigram.Added.split': ('unigram.html#added.split', 'laghu/unigram.py'),
|
|
48
|
+
'laghu.unigram.HFTok': ('unigram.html#hftok', 'laghu/unigram.py'),
|
|
49
|
+
'laghu.unigram.HFTok.__call__': ('unigram.html#hftok.__call__', 'laghu/unigram.py'),
|
|
50
|
+
'laghu.unigram.HFTok.__init__': ('unigram.html#hftok.__init__', 'laghu/unigram.py'),
|
|
51
|
+
'laghu.unigram.HFTok.vocab': ('unigram.html#hftok.vocab', 'laghu/unigram.py'),
|
|
52
|
+
'laghu.unigram.Unigram': ('unigram.html#unigram', 'laghu/unigram.py'),
|
|
53
|
+
'laghu.unigram.Unigram.__call__': ('unigram.html#unigram.__call__', 'laghu/unigram.py'),
|
|
54
|
+
'laghu.unigram.Unigram.__init__': ('unigram.html#unigram.__init__', 'laghu/unigram.py'),
|
|
55
|
+
'laghu.unigram.Unigram._first': ('unigram.html#unigram._first', 'laghu/unigram.py'),
|
|
56
|
+
'laghu.unigram.Unigram._pieces': ('unigram.html#unigram._pieces', 'laghu/unigram.py'),
|
|
57
|
+
'laghu.unigram.Unigram._seg': ('unigram.html#unigram._seg', 'laghu/unigram.py'),
|
|
58
|
+
'laghu.unigram.Unigram._viterbi': ('unigram.html#unigram._viterbi', 'laghu/unigram.py'),
|
|
59
|
+
'laghu.unigram.Unigram.encode': ('unigram.html#unigram.encode', 'laghu/unigram.py'),
|
|
60
|
+
'laghu.unigram.Unigram.unk_token_id': ('unigram.html#unigram.unk_token_id', 'laghu/unigram.py'),
|
|
61
|
+
'laghu.unigram.Unigram.vocab': ('unigram.html#unigram.vocab', 'laghu/unigram.py'),
|
|
62
|
+
'laghu.unigram.Unigram.word': ('unigram.html#unigram.word', 'laghu/unigram.py'),
|
|
63
|
+
'laghu.unigram._metaspace': ('unigram.html#_metaspace', 'laghu/unigram.py'),
|
|
64
|
+
'laghu.unigram._pipeline': ('unigram.html#_pipeline', 'laghu/unigram.py'),
|
|
65
|
+
'laghu.unigram._unk': ('unigram.html#_unk', 'laghu/unigram.py'),
|
|
66
|
+
'laghu.unigram.read_vocab': ('unigram.html#read_vocab', 'laghu/unigram.py'),
|
|
67
|
+
'laghu.unigram.tokenizer': ('unigram.html#tokenizer', 'laghu/unigram.py'),
|
|
68
|
+
'laghu.unigram.viterbi': ('unigram.html#viterbi', 'laghu/unigram.py')}}}
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
# AUTOGENERATED! DO NOT EDIT! File to edit: ../nbs/03_bench.ipynb.
|
|
2
|
+
|
|
3
|
+
# %% auto #0
|
|
4
|
+
__all__ = ['footprint', 'measure', 'check', 'row', 'same', 'table']
|
|
5
|
+
|
|
6
|
+
# %% ../nbs/03_bench.ipynb #630431fe
|
|
7
|
+
import json, sys, os, ctypes, platform, subprocess
|
|
8
|
+
from fastcore.all import AttrDict
|
|
9
|
+
|
|
10
|
+
# %% ../nbs/03_bench.ipynb #ad5d5182
|
|
11
|
+
def footprint():
|
|
12
|
+
"This process's resident, private and peak memory in MB."
|
|
13
|
+
if platform.system() == 'Darwin':
|
|
14
|
+
buf = (ctypes.c_uint64 * 64)() # rusage_info_v4 as uint64s: a 16-byte uuid, then resident_size at 8, phys_footprint 9, lifetime_max_phys_footprint 30
|
|
15
|
+
ctypes.CDLL(None).proc_pid_rusage(ctypes.c_int(os.getpid()), 4, buf)
|
|
16
|
+
return AttrDict(rss=buf[8] / 2**20, private=buf[9] / 2**20, peak=buf[30] / 2**20)
|
|
17
|
+
kv = dict(l.split(':', 1) for l in open('/proc/self/status'))
|
|
18
|
+
mb = lambda k: int(kv[k].split()[0]) / 1024
|
|
19
|
+
return AttrDict(rss=mb('VmRSS'), private=mb('RssAnon'), peak=mb('VmHWM'))
|
|
20
|
+
|
|
21
|
+
# %% ../nbs/03_bench.ipynb #b4cc8766
|
|
22
|
+
_PROBE = '''
|
|
23
|
+
import json, time
|
|
24
|
+
from laghu.bench import footprint
|
|
25
|
+
from laghu.core import corpus, resolve
|
|
26
|
+
t = time.time()
|
|
27
|
+
if {how!r} == 'model2vec':
|
|
28
|
+
from model2vec import StaticModel
|
|
29
|
+
m = StaticModel.from_pretrained(str(resolve({path!r})), force_download=False)
|
|
30
|
+
else:
|
|
31
|
+
from laghu import load
|
|
32
|
+
m = load({path!r}, fallback=False)
|
|
33
|
+
tl = time.time() - t; f1 = footprint()
|
|
34
|
+
txts = corpus({n}); t = time.time(); m.encode(txts); te = time.time() - t; f2 = footprint()
|
|
35
|
+
print(json.dumps(dict(how={how!r}, load_s=tl, encode_s=te, texts=len(txts), private_load=f1.private, private=f2.private, rss=f2.rss, peak=f2.peak)))
|
|
36
|
+
'''
|
|
37
|
+
|
|
38
|
+
_CHECK = '''
|
|
39
|
+
import json
|
|
40
|
+
from laghu import load, same_as_model2vec
|
|
41
|
+
from laghu.core import corpus, model_files, resolve, find_vocab
|
|
42
|
+
m = load({path!r}, fallback=False); f = model_files(m.folder)
|
|
43
|
+
kind = json.loads(f.tok.read_text())['model'].get('type') or 'Unigram'
|
|
44
|
+
r = same_as_model2vec({path!r}, corpus({n}))
|
|
45
|
+
print(json.dumps(dict(r, kind=kind, vocab=len(m.embedding), dim=m.dim, dtype=m.embedding_dtype, normalize=bool(m.normalize),
|
|
46
|
+
quantized=m.token_mapping is not None, weighted=m.weights is not None)))
|
|
47
|
+
'''
|
|
48
|
+
|
|
49
|
+
def _run(code):
|
|
50
|
+
r = subprocess.run([sys.executable, '-c', code], capture_output=True, text=True)
|
|
51
|
+
if r.returncode: raise RuntimeError(r.stderr[-3000:])
|
|
52
|
+
return AttrDict(json.loads(r.stdout.strip().splitlines()[-1]))
|
|
53
|
+
|
|
54
|
+
def measure(path, how='laghu', n=2000):
|
|
55
|
+
"Memory (MB) and time (s) of loading `path` with `how` ('laghu' or 'model2vec') and encoding `corpus(n)`, in a new process."
|
|
56
|
+
return _run(_PROBE.format(path=str(path), how=how, n=n))
|
|
57
|
+
|
|
58
|
+
def check(path, n=3000):
|
|
59
|
+
"`same_as_model2vec(path, corpus(n))` and the model's shape, in a new process."
|
|
60
|
+
return _run(_CHECK.format(path=str(path), n=n))
|
|
61
|
+
|
|
62
|
+
def row(path, n_check=3000, n_speed=2000):
|
|
63
|
+
"One model's line: what it is, whether it matches, and both ways' memory and speed."
|
|
64
|
+
r = check(path, n_check); r.model = str(path)
|
|
65
|
+
r.m2v, r.lg = measure(path, 'model2vec', n_speed), measure(path, 'laghu', n_speed)
|
|
66
|
+
return r
|
|
67
|
+
|
|
68
|
+
# %% ../nbs/03_bench.ipynb #33c0bdb7
|
|
69
|
+
def same(r): return all(r[k] for k in ('ids', 'dtype', 'f32', 'f16', 'exact', 'seq', 'median', 'unk', 'tokens'))
|
|
70
|
+
|
|
71
|
+
def table(rows):
|
|
72
|
+
"Rows of `row` as a markdown table."
|
|
73
|
+
hd = ('| model | tokenizer | laghu tokenizes with | vocab × dims | dtype | norm | identical | private MB, model2vec → laghu | peak MB, model2vec → laghu | texts/s, model2vec → laghu |\n'
|
|
74
|
+
'|---|---|---|---|---|---|---|---|---|---|\n')
|
|
75
|
+
def ln(r):
|
|
76
|
+
extra = ' +map' * r.quantized + ' +weights' * r.weighted
|
|
77
|
+
return (f"| {r.model.split('/')[-1]} | {r.kind} | {'Python Viterbi' if r.lean else 'tokenizers'} | {r.vocab:,} × {r.dim} | {r.dtype}{extra} | "
|
|
78
|
+
f"{'on' if r.normalize else 'off'} | {'yes' if same(r) else 'NO'} | {r.m2v.private:,.0f} → {r.lg.private:,.0f} | {r.m2v.peak:,.0f} → {r.lg.peak:,.0f} | "
|
|
79
|
+
f"{r.m2v.texts / r.m2v.encode_s:,.0f} → {r.lg.texts / r.lg.encode_s:,.0f} |")
|
|
80
|
+
return hd + '\n'.join(map(ln, rows))
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
# AUTOGENERATED! DO NOT EDIT! File to edit: ../nbs/00_core.ipynb.
|
|
2
|
+
|
|
3
|
+
# %% auto #0
|
|
4
|
+
__all__ = ['log', 'LAYOUTS', 'ST_DTYPES', 'SENTENCES', 'ODD', 'cached', 'resolve', 'layout', 'model_files', 'UnsupportedDtype',
|
|
5
|
+
'st_header', 'st_map', 'st_write', 'find_vocab', 'is_unigram', 'try_resolve', 'noise', 'corpus']
|
|
6
|
+
|
|
7
|
+
# %% ../nbs/00_core.ipynb #fd476c3c
|
|
8
|
+
import json, re, mmap, logging, numpy as np
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from fastcore.all import AttrDict, ifnone
|
|
11
|
+
|
|
12
|
+
log = logging.getLogger('laghu')
|
|
13
|
+
|
|
14
|
+
# %% ../nbs/00_core.ipynb #9073650b
|
|
15
|
+
def cached(repo):
|
|
16
|
+
"The newest cached snapshot of Hugging Face repo `repo`, or None."
|
|
17
|
+
if repo.count('/') != 1: return None
|
|
18
|
+
from huggingface_hub.constants import HF_HUB_CACHE
|
|
19
|
+
d = Path(HF_HUB_CACHE)/f"models--{repo.replace('/', '--')}"/'snapshots'
|
|
20
|
+
snaps = [o for o in d.iterdir() if o.is_dir()] if d.exists() else []
|
|
21
|
+
return max(snaps, key=lambda o: o.stat().st_mtime) if snaps else None
|
|
22
|
+
|
|
23
|
+
def resolve(path, token=None, subfolder=None, force_download=False):
|
|
24
|
+
"The local folder for `path`: a folder as given, else a complete snapshot of repo id `path` from the cache, else downloaded."
|
|
25
|
+
p = Path(path)
|
|
26
|
+
if p.exists(): return p/subfolder if subfolder else p
|
|
27
|
+
p = None if force_download else cached(str(path))
|
|
28
|
+
if p is None or layout(p/subfolder if subfolder else p) is None: # a snapshot holding only some files (one hf_hub_download) is not the model
|
|
29
|
+
from huggingface_hub import snapshot_download
|
|
30
|
+
p = Path(snapshot_download(Path(path).as_posix(), repo_type='model', token=token))
|
|
31
|
+
return p/subfolder if subfolder else p
|
|
32
|
+
|
|
33
|
+
# %% ../nbs/00_core.ipynb #b57979e0
|
|
34
|
+
def layout(folder):
|
|
35
|
+
"The config, table and tokenizer files of the first layout `folder` has, and the table's tensor name, or None."
|
|
36
|
+
folder = Path(folder)
|
|
37
|
+
for c, e, t, nm in LAYOUTS:
|
|
38
|
+
if all((folder/o).exists() for o in (c, e, t)): return AttrDict(cfg=folder/c, emb=folder/e, tok=folder/t, name=nm)
|
|
39
|
+
|
|
40
|
+
LAYOUTS = (('config.json', 'model.safetensors', 'tokenizer.json', 'embeddings'),
|
|
41
|
+
('config_sentence_transformers.json', 'model.safetensors', 'tokenizer.json', 'embedding.weight'),
|
|
42
|
+
('config_sentence_transformers.json', '0_StaticEmbedding/model.safetensors', '0_StaticEmbedding/tokenizer.json', 'embedding.weight'))
|
|
43
|
+
|
|
44
|
+
def model_files(folder):
|
|
45
|
+
"`layout(folder)`, raising when there is none."
|
|
46
|
+
if f := layout(folder): return f
|
|
47
|
+
raise FileNotFoundError(f'{folder}: no model2vec, sentence-transformers or 0_StaticEmbedding layout')
|
|
48
|
+
|
|
49
|
+
# %% ../nbs/00_core.ipynb #86a0c03a
|
|
50
|
+
ST_DTYPES = dict(F64=np.float64, F32=np.float32, F16=np.float16, I64=np.int64, I32=np.int32, I16=np.int16, I8=np.int8, U8=np.uint8)
|
|
51
|
+
|
|
52
|
+
class UnsupportedDtype(ValueError): "A tensor in a dtype numpy cannot hold."
|
|
53
|
+
|
|
54
|
+
def st_header(fn):
|
|
55
|
+
"Safetensors file `fn`'s header and the offset its data starts at."
|
|
56
|
+
with open(fn, 'rb') as f: n = int.from_bytes(f.read(8), 'little'); return json.loads(f.read(n)), 8 + n
|
|
57
|
+
|
|
58
|
+
def st_map(fn, names):
|
|
59
|
+
"Tensors `names` of safetensors file `fn`, memory-mapped read-only; a name the file lacks is None."
|
|
60
|
+
hdr, at = st_header(fn)
|
|
61
|
+
def _m(k):
|
|
62
|
+
if k not in hdr: return None
|
|
63
|
+
t = hdr[k]; s, e = t['data_offsets']; dt = ST_DTYPES.get(t['dtype']); shape = tuple(t['shape'])
|
|
64
|
+
if dt is None: raise UnsupportedDtype(f"{fn}: tensor {k!r} is {t['dtype']}; laghu reads {', '.join(ST_DTYPES)}")
|
|
65
|
+
if e == s: return np.zeros(shape, dt) # np.memmap refuses an empty map
|
|
66
|
+
m = np.memmap(fn, dtype=dt, mode='r', offset=at + s, shape=shape)
|
|
67
|
+
if m.nbytes != e - s: raise ValueError(f'{fn}: tensor {k!r} has {e - s} bytes, its shape and dtype make {m.nbytes}')
|
|
68
|
+
return m
|
|
69
|
+
return [_m(k) for k in names]
|
|
70
|
+
|
|
71
|
+
# %% ../nbs/00_core.ipynb #4ea0cd84
|
|
72
|
+
_NP2ST = {np.dtype(v): k for k, v in ST_DTYPES.items()}
|
|
73
|
+
|
|
74
|
+
def st_write(fn, tensors):
|
|
75
|
+
"Write dict `tensors` of numpy arrays to safetensors file `fn`."
|
|
76
|
+
hdr, at, bufs = {}, 0, []
|
|
77
|
+
for k, a in tensors.items():
|
|
78
|
+
b = np.ascontiguousarray(a).astype(a.dtype.newbyteorder('<'), copy=False).tobytes()
|
|
79
|
+
hdr[k] = dict(dtype=_NP2ST[a.dtype], shape=list(a.shape), data_offsets=[at, at + len(b)]); at += len(b); bufs.append(b)
|
|
80
|
+
h = json.dumps(hdr, separators=(',', ':')).encode(); h += b' ' * (-len(h) % 8)
|
|
81
|
+
Path(fn).parent.mkdir(parents=True, exist_ok=True)
|
|
82
|
+
with open(fn, 'wb') as f: f.write(len(h).to_bytes(8, 'little')); f.write(h); [f.write(b) for b in bufs]
|
|
83
|
+
|
|
84
|
+
# %% ../nbs/00_core.ipynb #ad1cbbef
|
|
85
|
+
_MODEL = re.compile(rb'"model"\s*:\s*\{')
|
|
86
|
+
_VOCAB = re.compile(rb'"vocab"\s*:\s*([\[{])')
|
|
87
|
+
|
|
88
|
+
def find_vocab(t):
|
|
89
|
+
"Where the model's vocabulary opens in tokenizer.json bytes `t`, and its bracket."
|
|
90
|
+
ms = list(_MODEL.finditer(t))
|
|
91
|
+
if not ms: return None, None
|
|
92
|
+
v = _VOCAB.search(t, ms[-1].end())
|
|
93
|
+
return (v.end(), v.group(1)) if v else (None, None)
|
|
94
|
+
|
|
95
|
+
def is_unigram(fn):
|
|
96
|
+
"Whether tokenizer.json `fn` holds a Unigram model."
|
|
97
|
+
with open(fn, 'rb') as f, mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ) as t: return find_vocab(t)[1] == b'['
|
|
98
|
+
|
|
99
|
+
# %% ../nbs/00_core.ipynb #13167816
|
|
100
|
+
def try_resolve(repo, **kw):
|
|
101
|
+
"`resolve(repo)`, or None (logged) when it cannot be had, e.g. offline."
|
|
102
|
+
try: return resolve(repo, **kw)
|
|
103
|
+
except Exception as e: log.warning(f'laghu: {repo} unavailable: {e!r}'); return None
|
|
104
|
+
|
|
105
|
+
# %% ../nbs/00_core.ipynb #b4cc1c2d
|
|
106
|
+
SENTENCES = [
|
|
107
|
+
'The quick brown fox jumps over the lazy dog.', "It's 3:45 p.m. on 2026-10-05; the price is $1,299.99 (incl. 18% GST).",
|
|
108
|
+
'Les élèves ont étudié l’histoire de la Révolution française.', 'Größere Äpfel schmecken süßer als kleinere, sagt Jürgen.',
|
|
109
|
+
'El niño comió paella en la playa de Valencia.', 'Il gatto dorme sul divano mentre piove.', 'O pássaro voou sobre a floresta amazônica.',
|
|
110
|
+
'Москва — столица России, крупнейший город Европы.', 'Η Αθήνα είναι η πρωτεύουσα της Ελλάδας.', 'Zażółć gęślą jaźń.',
|
|
111
|
+
'Příliš žluťoučký kůň úpěl ďábelské ódy.', 'Árvíztűrő tükörfúrógép.', 'İstanbul’da çay içip simit yedik.', 'Hà Nội là thủ đô của Việt Nam.',
|
|
112
|
+
'مرحبا بالعالم، كيف حالك اليوم؟', 'שלום עולם, מה שלומך?', 'سلام دنیا، امروز هوا خوب است.', 'नमस्ते दुनिया, आज मौसम अच्छा है।',
|
|
113
|
+
'कर्मण्येवाधिकारस्ते मा फलेषु कदाचन ।', 'অামি বাংলায় গান গাই।', 'வணக்கம் உலகம், இன்று வானிலை நன்றாக உள்ளது.',
|
|
114
|
+
'నమస్కారం ప్రపంచం.', 'ನಮಸ್ಕಾರ ಜಗತ್ತು.', 'നമസ്കാരം ലോകം.', 'ਸਤਿ ਸ੍ਰੀ ਅਕਾਲ ਦੁਨੀਆ।', 'નમસ્તે દુનિયા.', 'සුභ උදෑසනක්.',
|
|
115
|
+
'สวัสดีชาวโลก วันนี้อากาศดี', 'ສະບາຍດີ', 'မင်္ဂလာပါ', 'ជំរាបសួរ', '你好,世界!今天天气很好。', '日本語の文章を正しく分割できますか?',
|
|
116
|
+
'안녕하세요, 세계! 오늘 날씨가 좋네요.', 'გამარჯობა მსოფლიო', 'Բարեւ աշխարհ', 'ሰላም ልዑል', 'Habari ya dunia, leo ni siku nzuri.',
|
|
117
|
+
'def f(x): return x**2 # squares\nprint(f(3))', 'SELECT * FROM users WHERE id = 42;', 'https://example.com/a/b?c=d&e=f#g',
|
|
118
|
+
'user@example.org, +1 (555) 010-9999', 'E = mc², ∑ᵢ xᵢ² ≤ ∞, ∀x∈ℝ', '¯\\_(ツ)_/¯ :-) <3 ^_^']
|
|
119
|
+
|
|
120
|
+
ODD = ['', ' ', ' ', '\t\n', '\n\n\n', '\x00', '\x00\x01\x1f\x7f', '\x85
', 'tab\tsep\x0bvt\x0cff', 'CRLF\r\nline\rcr',
|
|
121
|
+
'[UNK]', '[PAD]', '[CLS] [SEP] [MASK]', 'a[PAD]b [UNK] c', '[PAD][PAD]', '[UNK]x[UNK]', ' [PAD]', '<unk>', '<pad></s>', '<s>x</s>',
|
|
122
|
+
'<extra_id_0>x<extra_id_99>', '<mask>', '▁', '▁▁x', ' leading', 'trailing ', 'a b c', ' nbsp ideographic thin',
|
|
123
|
+
'', 'bom', '\U0001F600 emoji', '👩👩👧👦 👍🏽 🇮🇳 🏳️🌈', 'fi fl Ⅻ ① ㎏ ㍿', 'é é ñ ñ', 'à́̂̃̄',
|
|
124
|
+
'z̷̢̛a̴l̸g̵o̶', 'FULLWIDTH カタカナ', '𝔘𝔫𝔦𝔠𝔬𝔡𝔢 𝟙𝟚𝟛', '𑌗𑍍𑌰𑌨𑍍𑌥', ' private', 'RTL evil ',
|
|
125
|
+
'\U000E0041\U000E007F tags', '\U0010FFFF', 'x' * 5000, 'x ' * 3000, ' '.join(['word'] * 600), 'ab' * 100_000, 'नम' * 100_000]
|
|
126
|
+
|
|
127
|
+
_BLOCKS = [(0x20, 0x7e), (0xa0, 0xff), (0x100, 0x24f), (0x300, 0x36f), (0x370, 0x3ff), (0x400, 0x4ff), (0x530, 0x58f), (0x590, 0x5ff),
|
|
128
|
+
(0x600, 0x6ff), (0x900, 0x97f), (0x980, 0x9ff), (0xb80, 0xbff), (0xc00, 0xc7f), (0xd00, 0xd7f), (0xe00, 0xe7f), (0x10a0, 0x10ff),
|
|
129
|
+
(0x1100, 0x11ff), (0x2000, 0x206f), (0x2100, 0x214f), (0x2190, 0x22ff), (0x3000, 0x303f), (0x3040, 0x30ff), (0x4e00, 0x4fff),
|
|
130
|
+
(0xac00, 0xadff), (0xfb00, 0xfb4f), (0xff00, 0xffef), (0x1d400, 0x1d4ff), (0x1f300, 0x1f6ff), (0x0, 0x1f), (0x11300, 0x1137f)]
|
|
131
|
+
|
|
132
|
+
def noise(n=2000, seed=0):
|
|
133
|
+
"`n` seeded random strings, each from one to three Unicode blocks, with spaces between runs."
|
|
134
|
+
import random
|
|
135
|
+
r = random.Random(seed); out = []
|
|
136
|
+
for _ in range(n):
|
|
137
|
+
bs = r.sample(_BLOCKS, r.randint(1, 3))
|
|
138
|
+
out.append(''.join(chr(r.randint(*r.choice(bs))) if r.random() > 0.15 else ' ' for _ in range(r.randint(0, 60))))
|
|
139
|
+
return out
|
|
140
|
+
|
|
141
|
+
def corpus(n=2000, seed=0):
|
|
142
|
+
"SENTENCES, ODD, `noise(n, seed)` and the sentences recombined: about `2.4 * n` texts."
|
|
143
|
+
import random
|
|
144
|
+
r = random.Random(seed); ws = ' '.join(SENTENCES).split()
|
|
145
|
+
mix = [' '.join(r.sample(ws, r.randint(1, 20))) for _ in range(n // 2)]
|
|
146
|
+
mix += [o.upper() for o in SENTENCES] + [o.replace(' ', '') for o in SENTENCES] + [s[:r.randint(0, len(s))] for s in SENTENCES]
|
|
147
|
+
return SENTENCES + ODD + noise(n, seed) + mix + ['\n'.join(SENTENCES), ' '.join(SENTENCES) * 20]
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
# AUTOGENERATED! DO NOT EDIT! File to edit: ../nbs/02_encoder.ipynb.
|
|
2
|
+
|
|
3
|
+
# %% auto #0
|
|
4
|
+
__all__ = ['LeanModel', 'load', 'same_as_model2vec']
|
|
5
|
+
|
|
6
|
+
# %% ../nbs/02_encoder.ipynb #07e60f33
|
|
7
|
+
import json, numpy as np
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from fastcore.all import ifnone, patch
|
|
10
|
+
from .core import log, resolve, model_files, st_map
|
|
11
|
+
from .unigram import tokenizer, Unigram
|
|
12
|
+
|
|
13
|
+
# %% ../nbs/02_encoder.ipynb #bf662691
|
|
14
|
+
class LeanModel:
|
|
15
|
+
"model2vec's `StaticModel` for reading a model, with its table mapped from the file and, for Unigram, no trie."
|
|
16
|
+
def __init__(self, folder, normalize=None, lean=True):
|
|
17
|
+
f = model_files(folder); self.folder = Path(folder)
|
|
18
|
+
self.config = json.loads(f.cfg.read_text())
|
|
19
|
+
self.embedding, self.weights, self.token_mapping = st_map(f.emb, [f.name, 'weights', 'mapping'])
|
|
20
|
+
if self.embedding is None: raise KeyError(f'{f.emb}: no tensor {f.name!r}')
|
|
21
|
+
self.tokenizer = tokenizer(f.tok, lean)
|
|
22
|
+
if self.token_mapping is None and self.tokenizer.n != len(self.embedding):
|
|
23
|
+
raise ValueError(f'Number of tokens ({self.tokenizer.n}) does not match number of vectors ({len(self.embedding)}).')
|
|
24
|
+
self.median_token_length, self.unk_token_id = self.tokenizer.median, self.tokenizer.unk_token_id
|
|
25
|
+
self.normalize = ifnone(normalize, self.config.get('normalize', False))
|
|
26
|
+
self.config['normalize'] = self.normalize
|
|
27
|
+
|
|
28
|
+
@classmethod
|
|
29
|
+
def from_pretrained(cls, path, token=None, normalize=None, subfolder=None, force_download=False, lean=True):
|
|
30
|
+
"A `LeanModel` for a folder or Hugging Face repo id `path`; a cached repo loads without the network."
|
|
31
|
+
return cls(resolve(path, token=token, subfolder=subfolder, force_download=force_download), normalize=normalize, lean=lean)
|
|
32
|
+
|
|
33
|
+
@property
|
|
34
|
+
def dim(self): return self.embedding.shape[1]
|
|
35
|
+
@property
|
|
36
|
+
def embedding_dtype(self): return np.dtype(self.embedding.dtype).name
|
|
37
|
+
@property
|
|
38
|
+
def lean(self): return isinstance(self.tokenizer, Unigram)
|
|
39
|
+
@property
|
|
40
|
+
def tokens(self):
|
|
41
|
+
"Every token, in id order, as `StaticModel.tokens`."
|
|
42
|
+
if not hasattr(self, '_tokens'): v = self.tokenizer.vocab(); self._tokens = tuple(sorted(v, key=v.get))
|
|
43
|
+
return self._tokens
|
|
44
|
+
|
|
45
|
+
def __repr__(self):
|
|
46
|
+
return f"{type(self).__name__}({str(self.folder)!r}, {len(self.embedding)}×{self.dim} {self.embedding_dtype}, {'unigram' if self.lean else 'tokenizers'})"
|
|
47
|
+
|
|
48
|
+
# %% ../nbs/02_encoder.ipynb #6fb7680d
|
|
49
|
+
def _batches(xs, bs): return (xs[i:i + bs] for i in range(0, len(xs), bs))
|
|
50
|
+
|
|
51
|
+
@patch
|
|
52
|
+
def tokenize(self:LeanModel, sentences, max_length=None):
|
|
53
|
+
"The ids of each of `sentences`, as `StaticModel.tokenize`."
|
|
54
|
+
if max_length is not None: m = max_length * self.median_token_length; sentences = [s[:m] for s in sentences]
|
|
55
|
+
ids = self.tokenizer(sentences)
|
|
56
|
+
if self.unk_token_id is not None: ids = [[k for k in o if k != self.unk_token_id] for o in ids]
|
|
57
|
+
return [o[:max_length] for o in ids] if max_length is not None else ids
|
|
58
|
+
|
|
59
|
+
@patch
|
|
60
|
+
def _rows(self:LeanModel, ids):
|
|
61
|
+
"The vectors of `ids`, remapped and weighted when the model is vocabulary-quantized."
|
|
62
|
+
emb = self.embedding[ids if self.token_mapping is None else self.token_mapping[ids]]
|
|
63
|
+
return emb * self.weights[ids][:, None] if self.weights is not None else emb
|
|
64
|
+
|
|
65
|
+
@patch
|
|
66
|
+
def _encode_batch(self:LeanModel, sentences, max_length):
|
|
67
|
+
"One batch: model2vec stacks the means, so an empty text's float64 zero row makes its whole batch float64."
|
|
68
|
+
out = np.stack([self._rows(o).mean(axis=0) if o else np.zeros(self.dim) for o in self.tokenize(sentences, max_length)])
|
|
69
|
+
return out / (np.linalg.norm(out, axis=1, keepdims=True) + 1e-32) if self.normalize else out
|
|
70
|
+
|
|
71
|
+
@patch
|
|
72
|
+
def encode(self:LeanModel, sentences, show_progress_bar=False, max_length=512, batch_size=1024, use_multiprocessing=True, multiprocessing_threshold=10_000, **kwargs):
|
|
73
|
+
"Each sentence's mean token vector (unit length when `normalize`), as `StaticModel.encode`; one string gives one vector."
|
|
74
|
+
one = isinstance(sentences, str)
|
|
75
|
+
xs = [sentences] if one else list(sentences)
|
|
76
|
+
if not xs: return np.zeros((0, self.dim), self.embedding.dtype)
|
|
77
|
+
out = np.concatenate([self._encode_batch(b, max_length) for b in _batches(xs, batch_size)], axis=0)
|
|
78
|
+
return out[0] if one else out
|
|
79
|
+
|
|
80
|
+
@patch
|
|
81
|
+
def encode_as_sequence(self:LeanModel, sentences, max_length=None, batch_size=1024, show_progress_bar=False, use_multiprocessing=True, multiprocessing_threshold=10_000):
|
|
82
|
+
"Each sentence's token vectors, `(n_tokens, dim)`, as `StaticModel.encode_as_sequence`."
|
|
83
|
+
one = isinstance(sentences, str)
|
|
84
|
+
xs = [sentences] if one else list(sentences)
|
|
85
|
+
out = [self._rows(o) if o else np.zeros((0, self.dim)) for b in _batches(xs, batch_size) for o in self.tokenize(b, max_length)]
|
|
86
|
+
return out[0] if one else out
|
|
87
|
+
|
|
88
|
+
# %% ../nbs/02_encoder.ipynb #e91068cb
|
|
89
|
+
def load(path, normalize=None, token=None, subfolder=None, force_download=False, lean=True, fallback=True):
|
|
90
|
+
"A `LeanModel` for `path`, else model2vec's `StaticModel` with a logged warning; `fallback=False` raises instead."
|
|
91
|
+
try: return LeanModel.from_pretrained(path, token=token, normalize=normalize, subfolder=subfolder, force_download=force_download, lean=lean)
|
|
92
|
+
except Exception as e:
|
|
93
|
+
if not fallback: raise
|
|
94
|
+
log.warning(f'laghu: cannot read {path} ({e!r}); loading it with model2vec instead')
|
|
95
|
+
try:
|
|
96
|
+
from model2vec import StaticModel
|
|
97
|
+
return StaticModel.from_pretrained(path, token=token, normalize=normalize, subfolder=subfolder, force_download=force_download)
|
|
98
|
+
except Exception as e2: raise e from e2
|
|
99
|
+
|
|
100
|
+
# %% ../nbs/02_encoder.ipynb #c807791f
|
|
101
|
+
def same_as_model2vec(path, txts=None, max_length=512, seq=200):
|
|
102
|
+
"Whether `load(path)` matches model2vec on `txts` (default `corpus()`): a dict of the checks."
|
|
103
|
+
from model2vec import StaticModel
|
|
104
|
+
from laghu.core import corpus
|
|
105
|
+
txts = ifnone(txts, corpus())
|
|
106
|
+
lm, sm = load(path, fallback=False), StaticModel.from_pretrained(str(resolve(path)), force_download=False)
|
|
107
|
+
a, b = lm.encode(txts, max_length=max_length), sm.encode(txts, max_length=max_length)
|
|
108
|
+
sa, sb = lm.encode_as_sequence(txts[:seq]), sm.encode_as_sequence(txts[:seq])
|
|
109
|
+
return dict(lean=lm.lean, n=len(txts), ids=lm.tokenize(txts, max_length) == sm.tokenize(txts, max_length), dtype=a.dtype == b.dtype,
|
|
110
|
+
f32=np.array_equal(a.astype(np.float32), b.astype(np.float32)), f16=np.array_equal(a.astype(np.float16), b.astype(np.float16)),
|
|
111
|
+
exact=a.dtype == b.dtype and np.array_equal(a, b), seq=all(x.dtype == y.dtype and np.array_equal(x, y) for x, y in zip(sa, sb)),
|
|
112
|
+
median=lm.median_token_length == sm.median_token_length, unk=lm.unk_token_id == sm.unk_token_id, tokens=lm.tokens == tuple(sm.tokens))
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
# AUTOGENERATED! DO NOT EDIT! File to edit: ../nbs/01_unigram.ipynb.
|
|
2
|
+
|
|
3
|
+
# %% auto #0
|
|
4
|
+
__all__ = ['UNK_PENALTY', 'WORD_CACHE', 'WORD_LEN', 'WS', 'read_vocab', 'viterbi', 'Added', 'Unigram', 'HFTok', 'tokenizer']
|
|
5
|
+
|
|
6
|
+
# %% ../nbs/01_unigram.ipynb #d9193c51
|
|
7
|
+
import json, re, mmap, numpy as np
|
|
8
|
+
from array import array
|
|
9
|
+
from functools import lru_cache
|
|
10
|
+
from tokenizers import Tokenizer, PreTokenizedString
|
|
11
|
+
from fastcore.all import ifnone
|
|
12
|
+
from .core import log, find_vocab, is_unigram
|
|
13
|
+
|
|
14
|
+
# %% ../nbs/01_unigram.ipynb #60cb7473
|
|
15
|
+
_ENTRY = re.compile(rb'\s*\[\s*"((?:[^"\\]|\\.)*)"\s*,\s*([^\],\s]+)\s*\]\s*(,?)')
|
|
16
|
+
_CLOSE = re.compile(rb'\s*\]')
|
|
17
|
+
|
|
18
|
+
def read_vocab(fn):
|
|
19
|
+
"Unigram tokenizer.json `fn` as (piece → id, scores, the JSON with the vocabulary emptied); a repeated piece keeps its last id, as in `tokenizers`."
|
|
20
|
+
with open(fn, 'rb') as f, mmap.mmap(f.fileno(), 0, access=mmap.ACCESS_READ) as t:
|
|
21
|
+
v0, br = find_vocab(t)
|
|
22
|
+
if br != b'[': raise NotImplementedError(f'{fn}: not a Unigram vocabulary')
|
|
23
|
+
ids, scores, at, i = {}, array('d'), v0, 0
|
|
24
|
+
while m := _ENTRY.match(t, at):
|
|
25
|
+
p = m.group(1); ids[json.loads(b'"' + p + b'"') if b'\\' in p else p.decode()] = i; scores.append(float(m.group(2))); i += 1; at = m.end()
|
|
26
|
+
if not m.group(3): break
|
|
27
|
+
if not (c := _CLOSE.match(t, at)): raise ValueError(f'{fn}: vocabulary not read to its end (stopped at byte {at})')
|
|
28
|
+
return ids, scores, json.loads(t[:v0] + t[c.start():])
|
|
29
|
+
|
|
30
|
+
# %% ../nbs/01_unigram.ipynb #e3c950b2
|
|
31
|
+
UNK_PENALTY = 10.0
|
|
32
|
+
WORD_CACHE, WORD_LEN = 2**16, 32 # pre-tokens cached per Unigram, and the longest one cached
|
|
33
|
+
|
|
34
|
+
def _unk(p, ids, unk_id, byte_fallback):
|
|
35
|
+
"The ids of an unknown run `p`: a piece if it spells one, else its UTF-8 bytes as `<0xXX>` pieces under byte fallback, else the unknown id."
|
|
36
|
+
if (k := ids.get(p)) is not None: return [k]
|
|
37
|
+
if byte_fallback and None not in (bs := [ids.get(f'<0x{b:02X}>') for b in p.encode()]): return bs
|
|
38
|
+
if unk_id < 0: raise ValueError(f'Encountered an unknown token but `unk_id` is missing: {p!r}') # tokenizers' error, and model2vec's
|
|
39
|
+
return [unk_id]
|
|
40
|
+
|
|
41
|
+
def viterbi(s, ids, scores, reach, unk_id, unk_score, byte_fallback=False):
|
|
42
|
+
"The ids of the best segmentation of `s` into pieces of `ids`, unknown runs fused."
|
|
43
|
+
n = len(s)
|
|
44
|
+
best, start, tid = [0.0] + [None] * n, [0] * (n + 1), [0] * (n + 1)
|
|
45
|
+
for i in range(n):
|
|
46
|
+
b, single = best[i], False
|
|
47
|
+
for j in range(i + 1, min(n, i + reach.get(s[i:i + 2], 1)) + 1):
|
|
48
|
+
k = ids.get(s[i:j])
|
|
49
|
+
if k is None: continue
|
|
50
|
+
c = scores[k] + b
|
|
51
|
+
if best[j] is None or c > best[j]: best[j], start[j], tid[j] = c, i, k
|
|
52
|
+
if j == i + 1: single = True
|
|
53
|
+
if not single:
|
|
54
|
+
c = unk_score + b
|
|
55
|
+
if best[i + 1] is None or c > best[i + 1]: best[i + 1], start[i + 1], tid[i + 1] = c, i, unk_id
|
|
56
|
+
out, e, run = [], n, None
|
|
57
|
+
while e > 0:
|
|
58
|
+
if tid[e] == unk_id: run = run or e
|
|
59
|
+
else:
|
|
60
|
+
if run: out += _unk(s[e:run], ids, unk_id, byte_fallback)[::-1]; run = None
|
|
61
|
+
out.append(tid[e])
|
|
62
|
+
e = start[e]
|
|
63
|
+
if run and e == 0: out += _unk(s[:run], ids, unk_id, byte_fallback)[::-1]
|
|
64
|
+
return out[::-1]
|
|
65
|
+
|
|
66
|
+
# %% ../nbs/01_unigram.ipynb #a88855f9
|
|
67
|
+
WS = frozenset('\t\n\x0b\x0c\r \x85\xa0\u1680\u2000\u2001\u2002\u2003\u2004\u2005\u2006\u2007\u2008\u2009\u200a\u2028\u2029\u202f\u205f\u3000') # Rust's char::is_whitespace
|
|
68
|
+
|
|
69
|
+
def _metaspace(pt, scheme):
|
|
70
|
+
"Pre-tokenizer JSON `pt` with every Metaspace's `prepend_scheme: first` set to `scheme`."
|
|
71
|
+
if isinstance(pt, list): return [_metaspace(o, scheme) for o in pt]
|
|
72
|
+
if not isinstance(pt, dict): return pt
|
|
73
|
+
pt = {k: _metaspace(v, scheme) for k, v in pt.items()}
|
|
74
|
+
if pt.get('type') == 'Metaspace' and pt.get('prepend_scheme') == 'first': pt['prepend_scheme'] = scheme
|
|
75
|
+
return pt
|
|
76
|
+
|
|
77
|
+
def _pipeline(d, pre):
|
|
78
|
+
"The normalizer and pre-tokenizer of tokenizer JSON `d`, with pre-tokenizer `pre`, and nothing else of it."
|
|
79
|
+
d = dict(d, pre_tokenizer=pre, added_tokens=[], post_processor=None, decoder=None, truncation=None, padding=None)
|
|
80
|
+
d['model'] = dict(d['model'], vocab=[['<unk>', 0.0]], unk_id=0, byte_fallback=False)
|
|
81
|
+
t = Tokenizer.from_str(json.dumps(d))
|
|
82
|
+
return t.normalizer, t.pre_tokenizer, t
|
|
83
|
+
|
|
84
|
+
class Added:
|
|
85
|
+
"A set of added tokens found in a string as `tokenizers`' `AddedVocabulary.find_matches` finds them."
|
|
86
|
+
def __init__(self, toks, norm=None):
|
|
87
|
+
"`toks` are matched as written, or as `norm` normalizes them when it is given (`normalized` tokens)."
|
|
88
|
+
self.toks = {(norm.normalize_str(o['content']) if norm else o['content']): (o['id'], o.get('lstrip'), o.get('rstrip')) for o in toks}
|
|
89
|
+
self.toks.pop('', None)
|
|
90
|
+
self.rx = re.compile('|'.join(map(re.escape, sorted(self.toks, key=len, reverse=True)))) if self.toks else None
|
|
91
|
+
|
|
92
|
+
def split(self, s):
|
|
93
|
+
"`s` cut into `(text, None)` and `(content, id)` parts, in order, empty text dropped."
|
|
94
|
+
if not self.rx: return [(s, None)] if s else []
|
|
95
|
+
out, prev = [], 0
|
|
96
|
+
for m in self.rx.finditer(s):
|
|
97
|
+
st, sp = m.span(); k, ls, rs = self.toks[m.group()]
|
|
98
|
+
if ls:
|
|
99
|
+
while st > prev and s[st - 1] in WS: st -= 1
|
|
100
|
+
if rs:
|
|
101
|
+
while sp < len(s) and s[sp] in WS: sp += 1
|
|
102
|
+
if st > prev: out.append((s[prev:st], None))
|
|
103
|
+
out.append((s[st:sp], k)); prev = sp
|
|
104
|
+
if prev < len(s): out.append((s[prev:], None))
|
|
105
|
+
return out
|
|
106
|
+
|
|
107
|
+
class Unigram:
|
|
108
|
+
"A Unigram tokenizer.json's `encode_batch(texts, add_special_tokens=False)` ids, with no trie."
|
|
109
|
+
def __init__(self, fn):
|
|
110
|
+
self.ids, self.scores, d = read_vocab(fn)
|
|
111
|
+
mdl = d['model']
|
|
112
|
+
if mdl.get('type', 'Unigram') != 'Unigram': raise NotImplementedError(f"{fn}: model is {mdl.get('type')}")
|
|
113
|
+
added = d.get('added_tokens', [])
|
|
114
|
+
if any(o.get('single_word') for o in added): raise NotImplementedError(f'{fn}: single_word added tokens')
|
|
115
|
+
self.byte_fallback, self.unk_id = bool(mdl.get('byte_fallback')), ifnone(mdl.get('unk_id'), -1) # -1: none, an error if met
|
|
116
|
+
self.trunc, self.pad = d.get('truncation'), d.get('padding')
|
|
117
|
+
self.unk_score = min(self.scores) - UNK_PENALTY
|
|
118
|
+
self.reach = {} # a piece's first two letters → the longest piece that starts so: no longer substring is tried
|
|
119
|
+
for w in self.ids:
|
|
120
|
+
if len(w) > 1 and len(w) > self.reach.get(w[:2], 0): self.reach[w[:2]] = len(w)
|
|
121
|
+
self.added = {o['content']: o['id'] for o in added}
|
|
122
|
+
self.norm, self.pre, t = _pipeline(d, d.get('pre_tokenizer'))
|
|
123
|
+
self.raw, self.nrm = Added([o for o in added if not o.get('normalized')]), Added([o for o in added if o.get('normalized')], self.norm)
|
|
124
|
+
later = _metaspace(d.get('pre_tokenizer'), 'never')
|
|
125
|
+
self.pre_later = self.pre if later == d.get('pre_tokenizer') else _pipeline(d, later)[1]
|
|
126
|
+
if self.nrm.rx and self.pre_later is not self.pre: raise NotImplementedError(f'{fn}: normalized added tokens under prepend_scheme first')
|
|
127
|
+
self.model_unk = getattr(t.model, 'unk_token', None) # model2vec drops this token's id; a Unigram model has none
|
|
128
|
+
extra = [k for k in self.added if k not in self.ids]
|
|
129
|
+
self.n = len(self.ids) + len(extra) # the size of `Tokenizer.get_vocab()`
|
|
130
|
+
lens = np.fromiter((len(k) for k in [*self.ids, *extra]), np.int32, self.n)
|
|
131
|
+
self.median = int(np.median(lens)) if self.n else 0
|
|
132
|
+
self._cached = lru_cache(WORD_CACHE)(lambda q: tuple(self._viterbi(q)))
|
|
133
|
+
|
|
134
|
+
def _viterbi(self, q): return viterbi(q, self.ids, self.scores, self.reach, self.unk_id, self.unk_score, self.byte_fallback)
|
|
135
|
+
|
|
136
|
+
def word(self, q):
|
|
137
|
+
"The ids of pre-token `q`; one of at most `WORD_LEN` characters, as natural text repeats them, from a cache of `WORD_CACHE`."
|
|
138
|
+
return self._cached(q) if len(q) <= WORD_LEN else self._viterbi(q)
|
|
139
|
+
|
|
140
|
+
def _first(self, s):
|
|
141
|
+
"The pre-tokens of the stretch opening the text, offsets kept, so `first` prepends only where the normalized text starts at 0."
|
|
142
|
+
pts = PreTokenizedString(s)
|
|
143
|
+
if self.norm: pts.normalize(self.norm.normalize)
|
|
144
|
+
pts.split(lambda i, ns: [ns[0:len(ns.normalized)]] if ns.normalized else []) # as tokenizers re-slices after its added-token pass
|
|
145
|
+
self.pre.pre_tokenize(pts)
|
|
146
|
+
return [p for p, *_ in pts.get_splits()]
|
|
147
|
+
|
|
148
|
+
def _pieces(self, nz, pt):
|
|
149
|
+
out = []
|
|
150
|
+
for p, k in self.nrm.split(nz):
|
|
151
|
+
if k is not None: out.append(k); continue
|
|
152
|
+
for q in ([q for q, _ in pt.pre_tokenize_str(p)] if pt else [p]):
|
|
153
|
+
out += self.word(q)
|
|
154
|
+
return out
|
|
155
|
+
|
|
156
|
+
def _seg(self, s, first):
|
|
157
|
+
if first and self.pre_later is not self.pre:
|
|
158
|
+
return [k for p in self._first(s) for k in self.word(p)]
|
|
159
|
+
nz = self.norm.normalize_str(s) if self.norm else s
|
|
160
|
+
return self._pieces(nz, self.pre if first else self.pre_later) if nz else []
|
|
161
|
+
|
|
162
|
+
def encode(self, s):
|
|
163
|
+
"The ids of text `s`, before truncation and padding."
|
|
164
|
+
out, at = [], 0
|
|
165
|
+
for p, k in self.raw.split(s):
|
|
166
|
+
out += self._seg(p, at == 0) if k is None else [k]; at += len(p)
|
|
167
|
+
return out
|
|
168
|
+
|
|
169
|
+
def __call__(self, txts):
|
|
170
|
+
"The ids of each of `txts`, as one batch: truncated and padded when tokenizer.json says so."
|
|
171
|
+
out = [self.encode(t) for t in txts]
|
|
172
|
+
if t := self.trunc:
|
|
173
|
+
L, left = t['max_length'], t.get('direction') == 'Left'
|
|
174
|
+
out = [(o[-L:] if L else []) if left else o[:L] for o in out]
|
|
175
|
+
if (p := self.pad) and out:
|
|
176
|
+
st = p.get('strategy')
|
|
177
|
+
n = max(map(len, out)) if st == 'BatchLongest' else (st or {}).get('Fixed', 0)
|
|
178
|
+
if (m := p.get('pad_to_multiple_of')) and n % m: n += m - n % m
|
|
179
|
+
pid, left = p.get('pad_id', 0), p.get('direction') == 'Left'
|
|
180
|
+
out = [([pid] * (n - len(o)) + o if left else o + [pid] * (n - len(o))) if len(o) < n else o for o in out]
|
|
181
|
+
return out
|
|
182
|
+
|
|
183
|
+
@property
|
|
184
|
+
def unk_token_id(self):
|
|
185
|
+
"The id model2vec drops from every encoding, or None."
|
|
186
|
+
return None if self.model_unk is None else self.vocab()[self.model_unk]
|
|
187
|
+
|
|
188
|
+
def vocab(self):
|
|
189
|
+
"`Tokenizer.get_vocab()`: piece → id, the added tokens over the model's."
|
|
190
|
+
return {**self.ids, **self.added}
|
|
191
|
+
|
|
192
|
+
# %% ../nbs/01_unigram.ipynb #4946b03a
|
|
193
|
+
class HFTok:
|
|
194
|
+
"`tokenizers`' own encoding, as model2vec calls it."
|
|
195
|
+
def __init__(self, fn):
|
|
196
|
+
self.tok = Tokenizer.from_file(str(fn))
|
|
197
|
+
v = self.tok.get_vocab()
|
|
198
|
+
self.n, self.median = len(v), int(np.median([len(o) for o in v])) if v else 0
|
|
199
|
+
unk = getattr(self.tok.model, 'unk_token', None)
|
|
200
|
+
self.unk_token_id = v[unk] if unk is not None else None
|
|
201
|
+
self._enc = getattr(self.tok, 'encode_batch_fast', self.tok.encode_batch)
|
|
202
|
+
|
|
203
|
+
def __call__(self, txts):
|
|
204
|
+
"The ids of each of `txts`, one batch as `tokenizers` encodes it (padding, if the file asks for it, is per batch)."
|
|
205
|
+
return [e.ids for e in self._enc(list(txts), add_special_tokens=False)]
|
|
206
|
+
|
|
207
|
+
def vocab(self): return self.tok.get_vocab()
|
|
208
|
+
|
|
209
|
+
def tokenizer(fn, lean=True):
|
|
210
|
+
"A `Unigram` for a Unigram tokenizer.json `fn` when `lean` and it can be done, else an `HFTok`."
|
|
211
|
+
if lean and is_unigram(fn):
|
|
212
|
+
try: return Unigram(fn)
|
|
213
|
+
except NotImplementedError as e: log.info(f'laghu: {e}; tokenizing with tokenizers')
|
|
214
|
+
return HFTok(fn)
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "laghu"
|
|
7
|
+
dynamic = ["version"]
|
|
8
|
+
description = "model2vec static models in a fraction of the memory, with the same vectors"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{name = "Karthik", email = "karthik.rajgopal@hotmail.com"}]
|
|
14
|
+
keywords = ['nbdev', 'model2vec', 'embeddings', 'static-embeddings', 'mmap', 'tokenizers', 'unigram']
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 3 - Alpha",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
20
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
21
|
+
]
|
|
22
|
+
dependencies = [
|
|
23
|
+
"fastcore>=1.8",
|
|
24
|
+
"numpy>=1.24",
|
|
25
|
+
"tokenizers>=0.20",
|
|
26
|
+
"huggingface-hub>=0.24",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
[project.optional-dependencies]
|
|
30
|
+
fallback = ["model2vec>=0.9"]
|
|
31
|
+
|
|
32
|
+
[project.urls]
|
|
33
|
+
Repository = "https://github.com/vedicreader/laghu"
|
|
34
|
+
Documentation = "https://vedicreader.github.io/laghu/"
|
|
35
|
+
|
|
36
|
+
[project.entry-points.nbdev]
|
|
37
|
+
laghu = "laghu._modidx:d"
|
|
38
|
+
|
|
39
|
+
[tool.nbdev]
|
|
40
|
+
tst_flags = "slow"
|
|
41
|
+
|
|
42
|
+
[tool.hatch.build.targets.wheel]
|
|
43
|
+
packages = ["laghu"]
|
|
44
|
+
|
|
45
|
+
[tool.hatch.build.targets.sdist]
|
|
46
|
+
include = ["/laghu", "/README.md", "/pyproject.toml"]
|
|
47
|
+
|
|
48
|
+
[tool.hatch.version]
|
|
49
|
+
path = "laghu/__init__.py"
|
|
50
|
+
|
|
51
|
+
[dependency-groups]
|
|
52
|
+
dev = [
|
|
53
|
+
"ipykernel>=7.3.0",
|
|
54
|
+
"model2vec>=0.9",
|
|
55
|
+
"nbdev>=3.3.14",
|
|
56
|
+
"notebook>=7.6.3",
|
|
57
|
+
]
|