remex 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- remex-0.5.0/LICENSE +21 -0
- remex-0.5.0/PKG-INFO +314 -0
- remex-0.5.0/README.md +290 -0
- remex-0.5.0/pyproject.toml +29 -0
- remex-0.5.0/remex/__init__.py +40 -0
- remex-0.5.0/remex/codebook.py +113 -0
- remex-0.5.0/remex/core.py +898 -0
- remex-0.5.0/remex/gpu.py +503 -0
- remex-0.5.0/remex/packing.py +200 -0
- remex-0.5.0/remex/rotation.py +27 -0
- remex-0.5.0/remex.egg-info/PKG-INFO +314 -0
- remex-0.5.0/remex.egg-info/SOURCES.txt +19 -0
- remex-0.5.0/remex.egg-info/dependency_links.txt +1 -0
- remex-0.5.0/remex.egg-info/requires.txt +13 -0
- remex-0.5.0/remex.egg-info/top_level.txt +1 -0
- remex-0.5.0/setup.cfg +4 -0
- remex-0.5.0/tests/test_adc_gpu.py +232 -0
- remex-0.5.0/tests/test_coverage_gaps.py +207 -0
- remex-0.5.0/tests/test_matryoshka.py +155 -0
- remex-0.5.0/tests/test_packed_vectors.py +445 -0
- remex-0.5.0/tests/test_polar_embed.py +418 -0
remex-0.5.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Oskar Austegard
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
remex-0.5.0/PKG-INFO
ADDED
|
@@ -0,0 +1,314 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: remex
|
|
3
|
+
Version: 0.5.0
|
|
4
|
+
Summary: Retrieval-validated embedding compression. 4-8x smaller vectors, proven recall.
|
|
5
|
+
Author: Oskar Austegard
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/oaustegard/remex
|
|
8
|
+
Project-URL: Repository, https://github.com/oaustegard/remex
|
|
9
|
+
Project-URL: Issues, https://github.com/oaustegard/remex/issues
|
|
10
|
+
Requires-Python: >=3.9
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Requires-Dist: numpy>=1.24
|
|
14
|
+
Requires-Dist: scipy>=1.10
|
|
15
|
+
Provides-Extra: dev
|
|
16
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
17
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
18
|
+
Provides-Extra: arrow
|
|
19
|
+
Requires-Dist: pyarrow>=12.0; extra == "arrow"
|
|
20
|
+
Provides-Extra: bench
|
|
21
|
+
Requires-Dist: faiss-cpu; extra == "bench"
|
|
22
|
+
Requires-Dist: sentence-transformers; extra == "bench"
|
|
23
|
+
Dynamic: license-file
|
|
24
|
+
|
|
25
|
+
# remex
|
|
26
|
+
|
|
27
|
+
Retrieval-validated embedding compression. 2-16x smaller vectors with measured recall.
|
|
28
|
+
|
|
29
|
+
> Formerly known as **polar-embed**.
|
|
30
|
+
|
|
31
|
+
Based on the rotation + Lloyd-Max scalar quantization insight from [TurboQuant](https://arxiv.org/abs/2504.19874) (Zandieh et al., ICLR 2026), focused on the use case that matters most: **embedding storage and retrieval for RAG systems**.
|
|
32
|
+
|
|
33
|
+
## Quick start
|
|
34
|
+
|
|
35
|
+
```python
|
|
36
|
+
from remex import Quantizer
|
|
37
|
+
|
|
38
|
+
# Compress embeddings — no training data needed
|
|
39
|
+
pq = Quantizer(d=384, bits=4) # d = your embedding dimension
|
|
40
|
+
compressed = pq.encode(embeddings) # (n, 384) float32 → compressed
|
|
41
|
+
indices, scores = pq.search(compressed, query, k=10)
|
|
42
|
+
|
|
43
|
+
# Save/load (bit-packed on disk)
|
|
44
|
+
compressed.save("index.npz")
|
|
45
|
+
from remex import CompressedVectors
|
|
46
|
+
loaded = CompressedVectors.load("index.npz")
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
The quantizer is fully determined by `(d, bits, seed)` — no training, no fitting, no index to ship.
|
|
50
|
+
|
|
51
|
+
## How it works
|
|
52
|
+
|
|
53
|
+
Three steps, each with a clear purpose:
|
|
54
|
+
|
|
55
|
+
1. **Random rotation** — A fixed orthogonal matrix (Haar-distributed via QR decomposition) transforms any embedding distribution so that coordinates become approximately i.i.d. N(0, 1/d). This is the key insight from TurboQuant: it makes quantization **data-oblivious**, meaning no training data is required.
|
|
56
|
+
|
|
57
|
+
2. **Lloyd-Max scalar quantization** — Each coordinate is independently quantized using optimal boundaries for the N(0, 1/d) distribution. The codebook is computed from the theoretical Gaussian CDF, not from data. This produces the minimum mean-squared-error scalar quantizer for Gaussian inputs.
|
|
58
|
+
|
|
59
|
+
3. **Bit-packing** — Indices are stored at their actual bit width (not wasteful uint8), giving honest compression ratios. A 4-bit codebook uses 4 bits per coordinate on disk.
|
|
60
|
+
|
|
61
|
+
Norms are stored separately as float32, preserving inner-product ranking up to quantization error.
|
|
62
|
+
|
|
63
|
+
**Why not QJL?** TurboQuant includes a QJL (quantized Johnson-Lindenstrauss) residual correction stage for unbiased inner product estimation. We omit it because QJL adds variance that hurts retrieval — when only ranking order matters (not absolute scores), the MSE-optimal rotation + Lloyd-Max stage empirically dominates.
|
|
64
|
+
|
|
65
|
+
## Matryoshka bit precision
|
|
66
|
+
|
|
67
|
+
An n-bit quantized index's top k bits are a valid k-bit code. remex exploits this: **encode once at full bit-width, search at any lower precision** by right-shifting indices. Centroid tables are precomputed for all bit levels.
|
|
68
|
+
|
|
69
|
+
This enables two-stage coarse-to-fine retrieval from a single encoded representation:
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
pq = Quantizer(d=384, bits=8)
|
|
73
|
+
compressed = pq.encode(corpus)
|
|
74
|
+
|
|
75
|
+
# Two-stage: coarse ADC scan at reduced bits, then full-precision rerank
|
|
76
|
+
indices, scores = pq.search_twostage(
|
|
77
|
+
compressed, query, k=10,
|
|
78
|
+
candidates=200, # coarse pass returns 200 candidates
|
|
79
|
+
coarse_precision=4, # coarse scan at 4-bit (default: bits-2)
|
|
80
|
+
)
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
The nesting incurs a small penalty vs independently optimized codebooks: ~1.2% at 4-bit, up to ~10% at 2-bit. In practice this matters little for the coarse stage, which only needs to identify the right neighborhood.
|
|
84
|
+
|
|
85
|
+
## Benchmarks
|
|
86
|
+
|
|
87
|
+
### Recall vs bit level (synthetic, d=384, 10k corpus, 200 queries)
|
|
88
|
+
|
|
89
|
+
| Method | Compression | MSE | R@10 | R@100 |
|
|
90
|
+
|--------|------------|-----|------|-------|
|
|
91
|
+
| remex 8-bit | 4.0x | 0.0000 | 0.987 | 0.991 |
|
|
92
|
+
| remex 4-bit | 7.8x | 0.0094 | 0.850 | 0.895 |
|
|
93
|
+
| remex 3-bit | 10.4x | 0.0343 | 0.719 | 0.800 |
|
|
94
|
+
| remex 2-bit | 15.4x | 0.1171 | 0.538 | 0.634 |
|
|
95
|
+
|
|
96
|
+
### Real embeddings (all-MiniLM-L6-v2, d=384, 10k corpus, 500 queries)
|
|
97
|
+
|
|
98
|
+
| Method | Compression | MSE | R@10 | R@100 |
|
|
99
|
+
|--------|------------|-----|------|-------|
|
|
100
|
+
| remex 8-bit | 2.0x | 0.0000 | 0.974 | 0.995 |
|
|
101
|
+
| remex 4-bit | 7.8x | 0.0093 | 0.707 | 0.932 |
|
|
102
|
+
| remex 3-bit | 10.4x | 0.0341 | 0.599 | 0.897 |
|
|
103
|
+
| remex 2-bit | 16x | 0.1164 | 0.517 | 0.860 |
|
|
104
|
+
| FAISS PQ (m=96, trained) | 16x | 0.0341 | 0.816 | 0.946 |
|
|
105
|
+
| FAISS PQ (m=48, trained) | 32x | 0.0636 | 0.618 | 0.877 |
|
|
106
|
+
|
|
107
|
+
### Scaling with corpus size (synthetic, 4-bit)
|
|
108
|
+
|
|
109
|
+
| Corpus | R@10 | R@100 | Encode (ms) | Search (ms) |
|
|
110
|
+
|--------|------|-------|-------------|-------------|
|
|
111
|
+
| 1k | 0.880 | 0.930 | 12 | 4 |
|
|
112
|
+
| 5k | 0.862 | 0.905 | 63 | 13 |
|
|
113
|
+
| 10k | 0.850 | 0.895 | 134 | 21 |
|
|
114
|
+
| 50k | 0.839 | 0.872 | 689 | 140 |
|
|
115
|
+
|
|
116
|
+
Full benchmark details and distribution sensitivity analysis in [`bench/RESULTS.md`](bench/RESULTS.md).
|
|
117
|
+
|
|
118
|
+
## When to use remex / when not to
|
|
119
|
+
|
|
120
|
+
### Use remex when
|
|
121
|
+
|
|
122
|
+
- **You want zero training.** The quantizer is deterministic and portable — just `(d, bits, seed)`. No codebook to train, no index to ship, no retraining when your corpus changes.
|
|
123
|
+
- **You need fast encode.** Encoding is ~20μs/vector (rotation + searchsorted). Adding new vectors never requires retraining.
|
|
124
|
+
- **8-bit caching is enough.** At 8-bit (4x compression), R@10 = 0.974 on real embeddings. Near-lossless and much cheaper than float32.
|
|
125
|
+
- **You want coarse retrieval + reranking.** 4-bit R@10=0.707 is enough for a first pass if you rerank the top candidates with a cross-encoder or full-precision search.
|
|
126
|
+
|
|
127
|
+
### Do not use remex when
|
|
128
|
+
|
|
129
|
+
- **You need high recall at aggressive compression on real data.** At 4-bit, FAISS PQ (m=96) achieves R@10=0.816 vs remex's 0.707 on real embeddings. Data-adaptive methods exploit structure that data-oblivious methods cannot.
|
|
130
|
+
- **Your embeddings form very tight clusters.** When cluster spread σ < 0.05, 4-bit R@10 drops to 0.53 (from 0.85 at normal spread). Quantization errors flip rankings among near-identical vectors. 8-bit is much more robust (R@10 stays above 0.95).
|
|
131
|
+
- **You need sublinear search.** remex is brute-force only. For >100k vectors, consider FAISS IVF, HNSW, or similar ANN indices. remex's compact encoding can feed into an external ANN index.
|
|
132
|
+
|
|
133
|
+
### Distribution sensitivity (10k corpus, 4-bit, varying cluster tightness)
|
|
134
|
+
|
|
135
|
+
| Cluster spread (σ) | 2-bit R@10 | 4-bit R@10 | 8-bit R@10 |
|
|
136
|
+
|-------------------|-----------|-----------|-----------|
|
|
137
|
+
| 0.01 (very tight) | 0.163 | 0.533 | 0.954 |
|
|
138
|
+
| 0.05 | 0.478 | 0.831 | 0.984 |
|
|
139
|
+
| 0.10 | 0.532 | 0.846 | 0.987 |
|
|
140
|
+
| 0.30 (typical) | 0.538 | 0.850 | 0.987 |
|
|
141
|
+
| 1.00 (diffuse) | 0.525 | 0.848 | 0.984 |
|
|
142
|
+
|
|
143
|
+
**Detection**: If your 4-bit R@10 is significantly below 0.80 on a held-out set, your embeddings likely have tight clusters. Use 8-bit, or switch to a data-adaptive method.
|
|
144
|
+
|
|
145
|
+
## Compression ratios
|
|
146
|
+
|
|
147
|
+
Honest packed sizes (bit-packed on disk, d=384):
|
|
148
|
+
|
|
149
|
+
| Bits | Bytes per vector | vs float32 | File size per 10k vectors |
|
|
150
|
+
|------|-----------------|------------|--------------------------|
|
|
151
|
+
| 2 | 100 | **15.4x** | 0.93 MB |
|
|
152
|
+
| 3 | 148 | **10.4x** | 1.42 MB |
|
|
153
|
+
| 4 | 196 | **7.8x** | 1.83 MB |
|
|
154
|
+
| 8 | 388 | **4.0x** | 3.61 MB |
|
|
155
|
+
|
|
156
|
+
Float32 baseline: 1,536 bytes/vector (15.36 MB per 10k vectors).
|
|
157
|
+
|
|
158
|
+
In-memory, indices are stored as uint8 for fast search. The `PackedVectors` class keeps them bit-packed in memory too, using 2-4x less RAM for sub-byte widths.
|
|
159
|
+
|
|
160
|
+
## API reference
|
|
161
|
+
|
|
162
|
+
### `Quantizer(d, bits=4, seed=42)`
|
|
163
|
+
|
|
164
|
+
Main quantizer class (formerly `PolarQuantizer`, which remains available as a deprecated alias).
|
|
165
|
+
|
|
166
|
+
- **`d`** — Vector dimension (must match your embeddings).
|
|
167
|
+
- **`bits`** — Bits per coordinate: 1-4 or 8. Sweet spot is 3-4. Use 8 for near-lossless.
|
|
168
|
+
- **`seed`** — Random seed for the rotation matrix. Same seed = same quantizer.
|
|
169
|
+
|
|
170
|
+
#### Methods
|
|
171
|
+
|
|
172
|
+
**`encode(X)`** — Quantize `(n, d)` float32 array. Returns `CompressedVectors`.
|
|
173
|
+
|
|
174
|
+
**`decode(compressed, precision=None)`** — Reconstruct `(n, d)` float32 from compressed. Optional `precision` (1 to bits) for Matryoshka decode.
|
|
175
|
+
|
|
176
|
+
**`search(compressed, query, k=10, precision=None)`** — Find k nearest neighbors by approximate inner product. Caches a dequantized float32 matrix for fast repeated queries. Returns `(indices, scores)`.
|
|
177
|
+
|
|
178
|
+
**`search_batch(compressed, queries, k=10, precision=None)`** — Batch version of `search()` using matrix multiplication for better throughput. Returns `(indices, scores)` where both are `(n_queries, k)`.
|
|
179
|
+
|
|
180
|
+
**`search_adc(compressed, query, k=10, precision=None, chunk_size=4096)`** — Memory-efficient search via lookup-table scoring. No float32 cache — peak memory is `chunk_size * d * 4` bytes (~6 MB). Slower per-query but uses ~5x less RAM. Returns `(indices, scores)`.
|
|
181
|
+
|
|
182
|
+
**`search_twostage(compressed, query, k=10, candidates=500, coarse_precision=None)`** — Two-stage Matryoshka retrieval: ADC coarse scan (no cache) then full-precision rerank on candidates only. Memory-efficient: only the small candidate set is dequantized. Returns `(indices, scores)`.
|
|
183
|
+
|
|
184
|
+
**`mse(X, precision=None)`** — Mean per-vector reconstruction error (L2 squared).
|
|
185
|
+
|
|
186
|
+
### `CompressedVectors`
|
|
187
|
+
|
|
188
|
+
Container for quantized data. Created by `Quantizer.encode()`. Stores indices as uint8 in memory for fast search/decode.
|
|
189
|
+
|
|
190
|
+
#### Properties
|
|
191
|
+
|
|
192
|
+
- **`n`** — Number of vectors.
|
|
193
|
+
- **`nbytes`** — Bit-packed size in bytes (honest compression).
|
|
194
|
+
- **`nbytes_unpacked`** — In-memory size (uint8 indices + float32 norms).
|
|
195
|
+
- **`compression_ratio`** — `(n * d * 4) / nbytes`.
|
|
196
|
+
- **`resident_bytes`** — Actual RAM including any active caches.
|
|
197
|
+
|
|
198
|
+
#### Methods
|
|
199
|
+
|
|
200
|
+
- **`save(path)`** / **`load(path)`** — Save/load to `.npz` with bit-packed indices.
|
|
201
|
+
- **`save_arrow(path)`** / **`load_arrow(path)`** — Save/load to Arrow IPC (Feather v2) format. Requires `pyarrow`.
|
|
202
|
+
- **`subset(idx)`** — Return a new `CompressedVectors` with only the given row indices.
|
|
203
|
+
- **`drop_cache()`** — Free the dequantized float32 cache to reclaim memory.
|
|
204
|
+
|
|
205
|
+
### `PackedVectors`
|
|
206
|
+
|
|
207
|
+
Memory-efficient packed storage. Keeps indices bit-packed in memory, unpacking on demand. Uses 2-4x less RAM than `CompressedVectors` for sub-byte widths.
|
|
208
|
+
|
|
209
|
+
```python
|
|
210
|
+
from remex import PackedVectors
|
|
211
|
+
|
|
212
|
+
packed = PackedVectors.from_compressed(compressed) # pack in memory
|
|
213
|
+
packed = PackedVectors.from_rows(rows, norms, d=384, bits=4) # from DB rows
|
|
214
|
+
|
|
215
|
+
# ADC and two-stage search work directly on PackedVectors
|
|
216
|
+
indices, scores = pq.search_adc(packed, query, k=10)
|
|
217
|
+
indices, scores = pq.search_twostage(packed, query, k=10)
|
|
218
|
+
|
|
219
|
+
# Matryoshka precision reduction
|
|
220
|
+
packed_2bit = packed.at_precision(2)
|
|
221
|
+
|
|
222
|
+
# Convert back if needed
|
|
223
|
+
compressed = packed.to_compressed()
|
|
224
|
+
```
|
|
225
|
+
|
|
226
|
+
Cached `search()` is not supported on `PackedVectors` — use `search_adc()` or `search_twostage()`, or convert with `to_compressed()`.
|
|
227
|
+
|
|
228
|
+
### `GPUSearcher` (optional)
|
|
229
|
+
|
|
230
|
+
GPU-accelerated search wrapper. Requires CuPy or PyTorch with CUDA. Falls back to NumPy.
|
|
231
|
+
|
|
232
|
+
```python
|
|
233
|
+
from remex.gpu import GPUSearcher
|
|
234
|
+
|
|
235
|
+
searcher = GPUSearcher(pq, compressed)
|
|
236
|
+
indices, scores = searcher.search(query, k=10)
|
|
237
|
+
indices, scores = searcher.search_adc(query, k=10)
|
|
238
|
+
indices, scores = searcher.search_twostage(query, k=10, candidates=200)
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
### Memory profiles (100k vectors, d=384, 8-bit)
|
|
242
|
+
|
|
243
|
+
| Strategy | Resident RAM | ms/query |
|
|
244
|
+
|----------|-------------|----------|
|
|
245
|
+
| `search()` (cached) | 192 MB | 3.9 |
|
|
246
|
+
| `search()` (cold) | 39 MB | 137 |
|
|
247
|
+
| `search_adc()` (no cache) | 39 MB | 152 |
|
|
248
|
+
| `search_twostage()` (no cache) | 39 MB | 152 |
|
|
249
|
+
|
|
250
|
+
Choose `search()` when latency matters and RAM is available. Choose `search_adc()` or `search_twostage()` when memory is constrained (serverless, edge, or very large corpora).
|
|
251
|
+
|
|
252
|
+
### Low-level utilities
|
|
253
|
+
|
|
254
|
+
```python
|
|
255
|
+
from remex import pack, unpack, packed_nbytes
|
|
256
|
+
from remex import lloyd_max_codebook, nested_codebooks
|
|
257
|
+
```
|
|
258
|
+
|
|
259
|
+
- **`pack(indices, bits)`** / **`unpack(packed, bits, n_values)`** — Bit-pack/unpack uint8 arrays.
|
|
260
|
+
- **`packed_nbytes(n_values, d, bits)`** — Compute packed byte count.
|
|
261
|
+
- **`lloyd_max_codebook(d, bits)`** — Generate optimal boundaries and centroids for N(0, 1/d).
|
|
262
|
+
- **`nested_codebooks(d, max_bits)`** — Build Matryoshka centroid tables for all bit levels 1..max_bits.
|
|
263
|
+
|
|
264
|
+
## vs TurboQuant
|
|
265
|
+
|
|
266
|
+
TurboQuant (Zandieh et al., ICLR 2026) adds QJL (quantized Johnson-Lindenstrauss) residual correction for unbiased inner product estimates. This is important for KV cache attention, where unbiased estimation matters. For **retrieval** (ranking by approximate inner product), the QJL variance hurts more than the debiasing helps. remex implements only the MSE-optimal rotation + Lloyd-Max stage, which empirically dominates for nearest-neighbor search.
|
|
267
|
+
|
|
268
|
+
## vs FAISS Product Quantization
|
|
269
|
+
|
|
270
|
+
| | remex | FAISS PQ |
|
|
271
|
+
|---|---|---|
|
|
272
|
+
| Training | None | Required (trains on corpus) |
|
|
273
|
+
| Recall at matched compression | Lower on real data | Higher (learns structure) |
|
|
274
|
+
| Encode speed | ~20μs/vec | ~200μs+/vec |
|
|
275
|
+
| Corpus updates | Re-encode only new vectors | Retrain or accept stale codebook |
|
|
276
|
+
| Index portability | Quantizer is `(d, bits, seed)` | Must ship trained index |
|
|
277
|
+
| Sublinear search | No (brute-force) | Yes (IVF, HNSW) |
|
|
278
|
+
| GPU support | NumPy/CuPy/PyTorch fallback | Native CUDA |
|
|
279
|
+
|
|
280
|
+
**Use FAISS when**: You have a stable, large corpus, need sublinear search, and can afford training time.
|
|
281
|
+
|
|
282
|
+
**Use remex when**: You want zero training, fast encode, frequently changing corpora, or near-lossless 8-bit caching (R@10=0.974 at 4x compression).
|
|
283
|
+
|
|
284
|
+
### vs scalar quantization (naive rounding)
|
|
285
|
+
|
|
286
|
+
Without the rotation step, scalar quantization on raw embeddings is catastrophically bad — embeddings are highly anisotropic (variance ratios of 10^7x across dimensions). The random rotation spreads information uniformly across coordinates, making scalar quantization viable.
|
|
287
|
+
|
|
288
|
+
At 3-bit, remex achieves 72-80% R@10 vs ~40% for naive scalar quantization on the same data.
|
|
289
|
+
|
|
290
|
+
## Installation
|
|
291
|
+
|
|
292
|
+
```bash
|
|
293
|
+
pip install remex # from PyPI (when published)
|
|
294
|
+
pip install -e ".[dev]" # development: + pytest, pytest-cov
|
|
295
|
+
pip install -e ".[bench]" # benchmarking: + faiss-cpu, sentence-transformers
|
|
296
|
+
```
|
|
297
|
+
|
|
298
|
+
## Testing
|
|
299
|
+
|
|
300
|
+
```bash
|
|
301
|
+
pytest # 126 tests (~6 min)
|
|
302
|
+
pytest tests/test_polar_embed.py -v # core tests
|
|
303
|
+
pytest tests/test_matryoshka.py -v # Matryoshka/nested codebook tests
|
|
304
|
+
pytest tests/test_adc_gpu.py -v # ADC and GPU searcher tests
|
|
305
|
+
pytest tests/test_packed_vectors.py -v # PackedVectors tests
|
|
306
|
+
```
|
|
307
|
+
|
|
308
|
+
## References
|
|
309
|
+
|
|
310
|
+
- Zandieh et al. (2025). *TurboQuant: Online Vector Quantization with Near-optimal Distortion Rate.* ICLR 2026. [arXiv:2504.19874](https://arxiv.org/abs/2504.19874)
|
|
311
|
+
|
|
312
|
+
## License
|
|
313
|
+
|
|
314
|
+
MIT
|
remex-0.5.0/README.md
ADDED
|
@@ -0,0 +1,290 @@
|
|
|
1
|
+
# remex
|
|
2
|
+
|
|
3
|
+
Retrieval-validated embedding compression. 2-16x smaller vectors with measured recall.
|
|
4
|
+
|
|
5
|
+
> Formerly known as **polar-embed**.
|
|
6
|
+
|
|
7
|
+
Based on the rotation + Lloyd-Max scalar quantization insight from [TurboQuant](https://arxiv.org/abs/2504.19874) (Zandieh et al., ICLR 2026), focused on the use case that matters most: **embedding storage and retrieval for RAG systems**.
|
|
8
|
+
|
|
9
|
+
## Quick start
|
|
10
|
+
|
|
11
|
+
```python
|
|
12
|
+
from remex import Quantizer
|
|
13
|
+
|
|
14
|
+
# Compress embeddings — no training data needed
|
|
15
|
+
pq = Quantizer(d=384, bits=4) # d = your embedding dimension
|
|
16
|
+
compressed = pq.encode(embeddings) # (n, 384) float32 → compressed
|
|
17
|
+
indices, scores = pq.search(compressed, query, k=10)
|
|
18
|
+
|
|
19
|
+
# Save/load (bit-packed on disk)
|
|
20
|
+
compressed.save("index.npz")
|
|
21
|
+
from remex import CompressedVectors
|
|
22
|
+
loaded = CompressedVectors.load("index.npz")
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
The quantizer is fully determined by `(d, bits, seed)` — no training, no fitting, no index to ship.
|
|
26
|
+
|
|
27
|
+
## How it works
|
|
28
|
+
|
|
29
|
+
Three steps, each with a clear purpose:
|
|
30
|
+
|
|
31
|
+
1. **Random rotation** — A fixed orthogonal matrix (Haar-distributed via QR decomposition) transforms any embedding distribution so that coordinates become approximately i.i.d. N(0, 1/d). This is the key insight from TurboQuant: it makes quantization **data-oblivious**, meaning no training data is required.
|
|
32
|
+
|
|
33
|
+
2. **Lloyd-Max scalar quantization** — Each coordinate is independently quantized using optimal boundaries for the N(0, 1/d) distribution. The codebook is computed from the theoretical Gaussian CDF, not from data. This produces the minimum mean-squared-error scalar quantizer for Gaussian inputs.
|
|
34
|
+
|
|
35
|
+
3. **Bit-packing** — Indices are stored at their actual bit width (not wasteful uint8), giving honest compression ratios. A 4-bit codebook uses 4 bits per coordinate on disk.
|
|
36
|
+
|
|
37
|
+
Norms are stored separately as float32, preserving inner-product ranking up to quantization error.
|
|
38
|
+
|
|
39
|
+
**Why not QJL?** TurboQuant includes a QJL (quantized Johnson-Lindenstrauss) residual correction stage for unbiased inner product estimation. We omit it because QJL adds variance that hurts retrieval — when only ranking order matters (not absolute scores), the MSE-optimal rotation + Lloyd-Max stage empirically dominates.
|
|
40
|
+
|
|
41
|
+
## Matryoshka bit precision
|
|
42
|
+
|
|
43
|
+
An n-bit quantized index's top k bits are a valid k-bit code. remex exploits this: **encode once at full bit-width, search at any lower precision** by right-shifting indices. Centroid tables are precomputed for all bit levels.
|
|
44
|
+
|
|
45
|
+
This enables two-stage coarse-to-fine retrieval from a single encoded representation:
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
pq = Quantizer(d=384, bits=8)
|
|
49
|
+
compressed = pq.encode(corpus)
|
|
50
|
+
|
|
51
|
+
# Two-stage: coarse ADC scan at reduced bits, then full-precision rerank
|
|
52
|
+
indices, scores = pq.search_twostage(
|
|
53
|
+
compressed, query, k=10,
|
|
54
|
+
candidates=200, # coarse pass returns 200 candidates
|
|
55
|
+
coarse_precision=4, # coarse scan at 4-bit (default: bits-2)
|
|
56
|
+
)
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
The nesting incurs a small penalty vs independently optimized codebooks: ~1.2% at 4-bit, up to ~10% at 2-bit. In practice this matters little for the coarse stage, which only needs to identify the right neighborhood.
|
|
60
|
+
|
|
61
|
+
## Benchmarks
|
|
62
|
+
|
|
63
|
+
### Recall vs bit level (synthetic, d=384, 10k corpus, 200 queries)
|
|
64
|
+
|
|
65
|
+
| Method | Compression | MSE | R@10 | R@100 |
|
|
66
|
+
|--------|------------|-----|------|-------|
|
|
67
|
+
| remex 8-bit | 4.0x | 0.0000 | 0.987 | 0.991 |
|
|
68
|
+
| remex 4-bit | 7.8x | 0.0094 | 0.850 | 0.895 |
|
|
69
|
+
| remex 3-bit | 10.4x | 0.0343 | 0.719 | 0.800 |
|
|
70
|
+
| remex 2-bit | 15.4x | 0.1171 | 0.538 | 0.634 |
|
|
71
|
+
|
|
72
|
+
### Real embeddings (all-MiniLM-L6-v2, d=384, 10k corpus, 500 queries)
|
|
73
|
+
|
|
74
|
+
| Method | Compression | MSE | R@10 | R@100 |
|
|
75
|
+
|--------|------------|-----|------|-------|
|
|
76
|
+
| remex 8-bit | 2.0x | 0.0000 | 0.974 | 0.995 |
|
|
77
|
+
| remex 4-bit | 7.8x | 0.0093 | 0.707 | 0.932 |
|
|
78
|
+
| remex 3-bit | 10.4x | 0.0341 | 0.599 | 0.897 |
|
|
79
|
+
| remex 2-bit | 16x | 0.1164 | 0.517 | 0.860 |
|
|
80
|
+
| FAISS PQ (m=96, trained) | 16x | 0.0341 | 0.816 | 0.946 |
|
|
81
|
+
| FAISS PQ (m=48, trained) | 32x | 0.0636 | 0.618 | 0.877 |
|
|
82
|
+
|
|
83
|
+
### Scaling with corpus size (synthetic, 4-bit)
|
|
84
|
+
|
|
85
|
+
| Corpus | R@10 | R@100 | Encode (ms) | Search (ms) |
|
|
86
|
+
|--------|------|-------|-------------|-------------|
|
|
87
|
+
| 1k | 0.880 | 0.930 | 12 | 4 |
|
|
88
|
+
| 5k | 0.862 | 0.905 | 63 | 13 |
|
|
89
|
+
| 10k | 0.850 | 0.895 | 134 | 21 |
|
|
90
|
+
| 50k | 0.839 | 0.872 | 689 | 140 |
|
|
91
|
+
|
|
92
|
+
Full benchmark details and distribution sensitivity analysis in [`bench/RESULTS.md`](bench/RESULTS.md).
|
|
93
|
+
|
|
94
|
+
## When to use remex / when not to
|
|
95
|
+
|
|
96
|
+
### Use remex when
|
|
97
|
+
|
|
98
|
+
- **You want zero training.** The quantizer is deterministic and portable — just `(d, bits, seed)`. No codebook to train, no index to ship, no retraining when your corpus changes.
|
|
99
|
+
- **You need fast encode.** Encoding is ~20μs/vector (rotation + searchsorted). Adding new vectors never requires retraining.
|
|
100
|
+
- **8-bit caching is enough.** At 8-bit (4x compression), R@10 = 0.974 on real embeddings. Near-lossless and much cheaper than float32.
|
|
101
|
+
- **You want coarse retrieval + reranking.** 4-bit R@10=0.707 is enough for a first pass if you rerank the top candidates with a cross-encoder or full-precision search.
|
|
102
|
+
|
|
103
|
+
### Do not use remex when
|
|
104
|
+
|
|
105
|
+
- **You need high recall at aggressive compression on real data.** At 4-bit, FAISS PQ (m=96) achieves R@10=0.816 vs remex's 0.707 on real embeddings. Data-adaptive methods exploit structure that data-oblivious methods cannot.
|
|
106
|
+
- **Your embeddings form very tight clusters.** When cluster spread σ < 0.05, 4-bit R@10 drops to 0.53 (from 0.85 at normal spread). Quantization errors flip rankings among near-identical vectors. 8-bit is much more robust (R@10 stays above 0.95).
|
|
107
|
+
- **You need sublinear search.** remex is brute-force only. For >100k vectors, consider FAISS IVF, HNSW, or similar ANN indices. remex's compact encoding can feed into an external ANN index.
|
|
108
|
+
|
|
109
|
+
### Distribution sensitivity (10k corpus, 4-bit, varying cluster tightness)
|
|
110
|
+
|
|
111
|
+
| Cluster spread (σ) | 2-bit R@10 | 4-bit R@10 | 8-bit R@10 |
|
|
112
|
+
|-------------------|-----------|-----------|-----------|
|
|
113
|
+
| 0.01 (very tight) | 0.163 | 0.533 | 0.954 |
|
|
114
|
+
| 0.05 | 0.478 | 0.831 | 0.984 |
|
|
115
|
+
| 0.10 | 0.532 | 0.846 | 0.987 |
|
|
116
|
+
| 0.30 (typical) | 0.538 | 0.850 | 0.987 |
|
|
117
|
+
| 1.00 (diffuse) | 0.525 | 0.848 | 0.984 |
|
|
118
|
+
|
|
119
|
+
**Detection**: If your 4-bit R@10 is significantly below 0.80 on a held-out set, your embeddings likely have tight clusters. Use 8-bit, or switch to a data-adaptive method.
|
|
120
|
+
|
|
121
|
+
## Compression ratios
|
|
122
|
+
|
|
123
|
+
Honest packed sizes (bit-packed on disk, d=384):
|
|
124
|
+
|
|
125
|
+
| Bits | Bytes per vector | vs float32 | File size per 10k vectors |
|
|
126
|
+
|------|-----------------|------------|--------------------------|
|
|
127
|
+
| 2 | 100 | **15.4x** | 0.93 MB |
|
|
128
|
+
| 3 | 148 | **10.4x** | 1.42 MB |
|
|
129
|
+
| 4 | 196 | **7.8x** | 1.83 MB |
|
|
130
|
+
| 8 | 388 | **4.0x** | 3.61 MB |
|
|
131
|
+
|
|
132
|
+
Float32 baseline: 1,536 bytes/vector (15.36 MB per 10k vectors).
|
|
133
|
+
|
|
134
|
+
In-memory, indices are stored as uint8 for fast search. The `PackedVectors` class keeps them bit-packed in memory too, using 2-4x less RAM for sub-byte widths.
|
|
135
|
+
|
|
136
|
+
## API reference
|
|
137
|
+
|
|
138
|
+
### `Quantizer(d, bits=4, seed=42)`
|
|
139
|
+
|
|
140
|
+
Main quantizer class (formerly `PolarQuantizer`, which remains available as a deprecated alias).
|
|
141
|
+
|
|
142
|
+
- **`d`** — Vector dimension (must match your embeddings).
|
|
143
|
+
- **`bits`** — Bits per coordinate: 1-4 or 8. Sweet spot is 3-4. Use 8 for near-lossless.
|
|
144
|
+
- **`seed`** — Random seed for the rotation matrix. Same seed = same quantizer.
|
|
145
|
+
|
|
146
|
+
#### Methods
|
|
147
|
+
|
|
148
|
+
**`encode(X)`** — Quantize `(n, d)` float32 array. Returns `CompressedVectors`.
|
|
149
|
+
|
|
150
|
+
**`decode(compressed, precision=None)`** — Reconstruct `(n, d)` float32 from compressed. Optional `precision` (1 to bits) for Matryoshka decode.
|
|
151
|
+
|
|
152
|
+
**`search(compressed, query, k=10, precision=None)`** — Find k nearest neighbors by approximate inner product. Caches a dequantized float32 matrix for fast repeated queries. Returns `(indices, scores)`.
|
|
153
|
+
|
|
154
|
+
**`search_batch(compressed, queries, k=10, precision=None)`** — Batch version of `search()` using matrix multiplication for better throughput. Returns `(indices, scores)` where both are `(n_queries, k)`.
|
|
155
|
+
|
|
156
|
+
**`search_adc(compressed, query, k=10, precision=None, chunk_size=4096)`** — Memory-efficient search via lookup-table scoring. No float32 cache — peak memory is `chunk_size * d * 4` bytes (~6 MB). Slower per-query but uses ~5x less RAM. Returns `(indices, scores)`.
|
|
157
|
+
|
|
158
|
+
**`search_twostage(compressed, query, k=10, candidates=500, coarse_precision=None)`** — Two-stage Matryoshka retrieval: ADC coarse scan (no cache) then full-precision rerank on candidates only. Memory-efficient: only the small candidate set is dequantized. Returns `(indices, scores)`.
|
|
159
|
+
|
|
160
|
+
**`mse(X, precision=None)`** — Mean per-vector reconstruction error (L2 squared).
|
|
161
|
+
|
|
162
|
+
### `CompressedVectors`
|
|
163
|
+
|
|
164
|
+
Container for quantized data. Created by `Quantizer.encode()`. Stores indices as uint8 in memory for fast search/decode.
|
|
165
|
+
|
|
166
|
+
#### Properties
|
|
167
|
+
|
|
168
|
+
- **`n`** — Number of vectors.
|
|
169
|
+
- **`nbytes`** — Bit-packed size in bytes (honest compression).
|
|
170
|
+
- **`nbytes_unpacked`** — In-memory size (uint8 indices + float32 norms).
|
|
171
|
+
- **`compression_ratio`** — `(n * d * 4) / nbytes`.
|
|
172
|
+
- **`resident_bytes`** — Actual RAM including any active caches.
|
|
173
|
+
|
|
174
|
+
#### Methods
|
|
175
|
+
|
|
176
|
+
- **`save(path)`** / **`load(path)`** — Save/load to `.npz` with bit-packed indices.
|
|
177
|
+
- **`save_arrow(path)`** / **`load_arrow(path)`** — Save/load to Arrow IPC (Feather v2) format. Requires `pyarrow`.
|
|
178
|
+
- **`subset(idx)`** — Return a new `CompressedVectors` with only the given row indices.
|
|
179
|
+
- **`drop_cache()`** — Free the dequantized float32 cache to reclaim memory.
|
|
180
|
+
|
|
181
|
+
### `PackedVectors`
|
|
182
|
+
|
|
183
|
+
Memory-efficient packed storage. Keeps indices bit-packed in memory, unpacking on demand. Uses 2-4x less RAM than `CompressedVectors` for sub-byte widths.
|
|
184
|
+
|
|
185
|
+
```python
|
|
186
|
+
from remex import PackedVectors
|
|
187
|
+
|
|
188
|
+
packed = PackedVectors.from_compressed(compressed) # pack in memory
|
|
189
|
+
packed = PackedVectors.from_rows(rows, norms, d=384, bits=4) # from DB rows
|
|
190
|
+
|
|
191
|
+
# ADC and two-stage search work directly on PackedVectors
|
|
192
|
+
indices, scores = pq.search_adc(packed, query, k=10)
|
|
193
|
+
indices, scores = pq.search_twostage(packed, query, k=10)
|
|
194
|
+
|
|
195
|
+
# Matryoshka precision reduction
|
|
196
|
+
packed_2bit = packed.at_precision(2)
|
|
197
|
+
|
|
198
|
+
# Convert back if needed
|
|
199
|
+
compressed = packed.to_compressed()
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
Cached `search()` is not supported on `PackedVectors` — use `search_adc()` or `search_twostage()`, or convert with `to_compressed()`.
|
|
203
|
+
|
|
204
|
+
### `GPUSearcher` (optional)
|
|
205
|
+
|
|
206
|
+
GPU-accelerated search wrapper. Requires CuPy or PyTorch with CUDA. Falls back to NumPy.
|
|
207
|
+
|
|
208
|
+
```python
|
|
209
|
+
from remex.gpu import GPUSearcher
|
|
210
|
+
|
|
211
|
+
searcher = GPUSearcher(pq, compressed)
|
|
212
|
+
indices, scores = searcher.search(query, k=10)
|
|
213
|
+
indices, scores = searcher.search_adc(query, k=10)
|
|
214
|
+
indices, scores = searcher.search_twostage(query, k=10, candidates=200)
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
### Memory profiles (100k vectors, d=384, 8-bit)
|
|
218
|
+
|
|
219
|
+
| Strategy | Resident RAM | ms/query |
|
|
220
|
+
|----------|-------------|----------|
|
|
221
|
+
| `search()` (cached) | 192 MB | 3.9 |
|
|
222
|
+
| `search()` (cold) | 39 MB | 137 |
|
|
223
|
+
| `search_adc()` (no cache) | 39 MB | 152 |
|
|
224
|
+
| `search_twostage()` (no cache) | 39 MB | 152 |
|
|
225
|
+
|
|
226
|
+
Choose `search()` when latency matters and RAM is available. Choose `search_adc()` or `search_twostage()` when memory is constrained (serverless, edge, or very large corpora).
|
|
227
|
+
|
|
228
|
+
### Low-level utilities
|
|
229
|
+
|
|
230
|
+
```python
|
|
231
|
+
from remex import pack, unpack, packed_nbytes
|
|
232
|
+
from remex import lloyd_max_codebook, nested_codebooks
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
- **`pack(indices, bits)`** / **`unpack(packed, bits, n_values)`** — Bit-pack/unpack uint8 arrays.
|
|
236
|
+
- **`packed_nbytes(n_values, d, bits)`** — Compute packed byte count.
|
|
237
|
+
- **`lloyd_max_codebook(d, bits)`** — Generate optimal boundaries and centroids for N(0, 1/d).
|
|
238
|
+
- **`nested_codebooks(d, max_bits)`** — Build Matryoshka centroid tables for all bit levels 1..max_bits.
|
|
239
|
+
|
|
240
|
+
## vs TurboQuant
|
|
241
|
+
|
|
242
|
+
TurboQuant (Zandieh et al., ICLR 2026) adds QJL (quantized Johnson-Lindenstrauss) residual correction for unbiased inner product estimates. This is important for KV cache attention, where unbiased estimation matters. For **retrieval** (ranking by approximate inner product), the QJL variance hurts more than the debiasing helps. remex implements only the MSE-optimal rotation + Lloyd-Max stage, which empirically dominates for nearest-neighbor search.
|
|
243
|
+
|
|
244
|
+
## vs FAISS Product Quantization
|
|
245
|
+
|
|
246
|
+
| | remex | FAISS PQ |
|
|
247
|
+
|---|---|---|
|
|
248
|
+
| Training | None | Required (trains on corpus) |
|
|
249
|
+
| Recall at matched compression | Lower on real data | Higher (learns structure) |
|
|
250
|
+
| Encode speed | ~20μs/vec | ~200μs+/vec |
|
|
251
|
+
| Corpus updates | Re-encode only new vectors | Retrain or accept stale codebook |
|
|
252
|
+
| Index portability | Quantizer is `(d, bits, seed)` | Must ship trained index |
|
|
253
|
+
| Sublinear search | No (brute-force) | Yes (IVF, HNSW) |
|
|
254
|
+
| GPU support | NumPy/CuPy/PyTorch fallback | Native CUDA |
|
|
255
|
+
|
|
256
|
+
**Use FAISS when**: You have a stable, large corpus, need sublinear search, and can afford training time.
|
|
257
|
+
|
|
258
|
+
**Use remex when**: You want zero training, fast encode, frequently changing corpora, or near-lossless 8-bit caching (R@10=0.974 at 4x compression).
|
|
259
|
+
|
|
260
|
+
### vs scalar quantization (naive rounding)
|
|
261
|
+
|
|
262
|
+
Without the rotation step, scalar quantization on raw embeddings is catastrophically bad — embeddings are highly anisotropic (variance ratios of 10^7x across dimensions). The random rotation spreads information uniformly across coordinates, making scalar quantization viable.
|
|
263
|
+
|
|
264
|
+
At 3-bit, remex achieves 72-80% R@10 vs ~40% for naive scalar quantization on the same data.
|
|
265
|
+
|
|
266
|
+
## Installation
|
|
267
|
+
|
|
268
|
+
```bash
|
|
269
|
+
pip install remex # from PyPI (when published)
|
|
270
|
+
pip install -e ".[dev]" # development: + pytest, pytest-cov
|
|
271
|
+
pip install -e ".[bench]" # benchmarking: + faiss-cpu, sentence-transformers
|
|
272
|
+
```
|
|
273
|
+
|
|
274
|
+
## Testing
|
|
275
|
+
|
|
276
|
+
```bash
|
|
277
|
+
pytest # 126 tests (~6 min)
|
|
278
|
+
pytest tests/test_polar_embed.py -v # core tests
|
|
279
|
+
pytest tests/test_matryoshka.py -v # Matryoshka/nested codebook tests
|
|
280
|
+
pytest tests/test_adc_gpu.py -v # ADC and GPU searcher tests
|
|
281
|
+
pytest tests/test_packed_vectors.py -v # PackedVectors tests
|
|
282
|
+
```
|
|
283
|
+
|
|
284
|
+
## References
|
|
285
|
+
|
|
286
|
+
- Zandieh et al. (2025). *TurboQuant: Online Vector Quantization with Near-optimal Distortion Rate.* ICLR 2026. [arXiv:2504.19874](https://arxiv.org/abs/2504.19874)
|
|
287
|
+
|
|
288
|
+
## License
|
|
289
|
+
|
|
290
|
+
MIT
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "remex"
|
|
7
|
+
version = "0.5.0"
|
|
8
|
+
description = "Retrieval-validated embedding compression. 4-8x smaller vectors, proven recall."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = {text = "MIT"}
|
|
11
|
+
requires-python = ">=3.9"
|
|
12
|
+
authors = [{name = "Oskar Austegard"}]
|
|
13
|
+
dependencies = ["numpy>=1.24", "scipy>=1.10"]
|
|
14
|
+
|
|
15
|
+
[project.optional-dependencies]
|
|
16
|
+
dev = ["pytest>=7.0", "pytest-cov"]
|
|
17
|
+
arrow = ["pyarrow>=12.0"]
|
|
18
|
+
bench = ["faiss-cpu", "sentence-transformers"]
|
|
19
|
+
|
|
20
|
+
[tool.setuptools.packages.find]
|
|
21
|
+
include = ["remex*"]
|
|
22
|
+
|
|
23
|
+
[tool.pytest.ini_options]
|
|
24
|
+
testpaths = ["tests"]
|
|
25
|
+
|
|
26
|
+
[project.urls]
|
|
27
|
+
Homepage = "https://github.com/oaustegard/remex"
|
|
28
|
+
Repository = "https://github.com/oaustegard/remex"
|
|
29
|
+
Issues = "https://github.com/oaustegard/remex/issues"
|