quantize-py 0.3.2__cp314-cp314t-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quantize/__init__.py +41 -0
- quantize/_native.cp314t-win_amd64.pyd +0 -0
- quantize/adaptive.py +3 -0
- quantize/asymmetric.py +4 -0
- quantize/learned.py +3 -0
- quantize/symmetric.py +3 -0
- quantize_py-0.3.2.dist-info/METADATA +108 -0
- quantize_py-0.3.2.dist-info/RECORD +11 -0
- quantize_py-0.3.2.dist-info/WHEEL +4 -0
- quantize_py-0.3.2.dist-info/licenses/LICENSE +21 -0
- quantize_py-0.3.2.dist-info/sboms/quantize-py.cyclonedx.json +1430 -0
quantize/__init__.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Python bindings for quantize."""
|
|
2
|
+
|
|
3
|
+
from quantize._native import (
|
|
4
|
+
InvalidBitsError,
|
|
5
|
+
InvalidBlockError,
|
|
6
|
+
InvalidToleranceError,
|
|
7
|
+
LengthMismatchError,
|
|
8
|
+
NotAMatrixError,
|
|
9
|
+
QuantizeError,
|
|
10
|
+
Quantized,
|
|
11
|
+
Scale,
|
|
12
|
+
ScaleOutOfRangeError,
|
|
13
|
+
Scheme,
|
|
14
|
+
ShapeMismatchError,
|
|
15
|
+
ToleranceTooTightError,
|
|
16
|
+
__version__,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
from . import adaptive, asymmetric, learned, symmetric
|
|
20
|
+
from .symmetric import quantize, quantize_tensor
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
"Scale",
|
|
24
|
+
"Scheme",
|
|
25
|
+
"Quantized",
|
|
26
|
+
"QuantizeError",
|
|
27
|
+
"InvalidBitsError",
|
|
28
|
+
"InvalidBlockError",
|
|
29
|
+
"InvalidToleranceError",
|
|
30
|
+
"ToleranceTooTightError",
|
|
31
|
+
"ScaleOutOfRangeError",
|
|
32
|
+
"LengthMismatchError",
|
|
33
|
+
"ShapeMismatchError",
|
|
34
|
+
"NotAMatrixError",
|
|
35
|
+
"quantize",
|
|
36
|
+
"quantize_tensor",
|
|
37
|
+
"symmetric",
|
|
38
|
+
"asymmetric",
|
|
39
|
+
"adaptive",
|
|
40
|
+
"learned",
|
|
41
|
+
]
|
|
Binary file
|
quantize/adaptive.py
ADDED
quantize/asymmetric.py
ADDED
quantize/learned.py
ADDED
quantize/symmetric.py
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: quantize-py
|
|
3
|
+
Version: 0.3.2
|
|
4
|
+
Classifier: Programming Language :: Python :: 3
|
|
5
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
6
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
7
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
8
|
+
Classifier: Programming Language :: Rust
|
|
9
|
+
Classifier: Topic :: Scientific/Engineering
|
|
10
|
+
Requires-Dist: numpy>=1.26
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Summary: Python bindings for the quantize crate.
|
|
13
|
+
Keywords: quantization,machine-learning,numpy
|
|
14
|
+
Author: Akshey Deokule
|
|
15
|
+
License-Expression: MIT
|
|
16
|
+
Requires-Python: >=3.12
|
|
17
|
+
Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
|
|
18
|
+
Project-URL: Documentation, https://github.com/aksheyd/quantize/tree/main/python
|
|
19
|
+
Project-URL: Homepage, https://github.com/aksheyd/quantize
|
|
20
|
+
Project-URL: Repository, https://github.com/aksheyd/quantize
|
|
21
|
+
|
|
22
|
+
# quantize-py
|
|
23
|
+
|
|
24
|
+
python bindings for [quantize](https://github.com/aksheyd/quantize), a simple, fast quantization library written in rust. quantization stores numbers in fewer bits, trading a little accuracy for a lot less memory.
|
|
25
|
+
|
|
26
|
+
```
|
|
27
|
+
pip install quantize-py
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
requires python 3.12 or newer, or 3.14 or newer for free-threaded python. numpy is installed with it. where no wheel fits, like alpine linux, pip builds it from source, which needs rust 1.88 or newer.
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
from quantize import Scale, quantize
|
|
34
|
+
|
|
35
|
+
weights = [0.42, -0.10, 0.70, -0.50]
|
|
36
|
+
|
|
37
|
+
q = quantize(weights, bits=8, block=32, scale=Scale.F16)
|
|
38
|
+
back = q.dequantize() # [0.421, -0.098, 0.700, -0.498]
|
|
39
|
+
dot = q.dot(weights) # 0.926
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
`bits` is the width of each code, from 2 to 16. `block` is how many values share one scale, and `scale` is how that scale is stored: `Scale.F32` (the default), `Scale.F16`, or `Scale.BF16`. values can be a list, a numpy array, or anything else `np.asarray` reads, like a pytorch tensor. a 2-d array keeps its shape, so `q.dequantize()` gives back a matrix and `q.matmul(x)` computes `x @ W.T`, like a linear layer. both can write into a float32 numpy array or pytorch tensor you pass, like `q.matmul(x, out=y)`, so a loop can reuse it. `q.shape` gives its rows and columns, and `len(q)` the number of values, as in rust.
|
|
43
|
+
|
|
44
|
+
the scales count toward the size: 4-bit codes with one f16 scale per 32 values cost 4.5 bits per value, or 5 with the default f32 scale. `q.bits_per_element` reports it. above about 10 bits, use f32 scales with `asymmetric.quantize`, since f16 and bf16 zero-points cap its accuracy.
|
|
45
|
+
|
|
46
|
+
the other schemes return the same `Quantized` type:
|
|
47
|
+
|
|
48
|
+
- `asymmetric.quantize(weights, bits=8, block=32)` adds a zero-point per block, for values that aren't centered on zero
|
|
49
|
+
- `adaptive.quantize(weights, tolerance=0.1 * weights.std())` gives each block the fewest bits, from 2 to 8, that round every weight within `tolerance`, in the weights' own units. a tenth of their standard deviation gives about 5 bits a block. for a list, use `np.std(weights)`
|
|
50
|
+
- `learned.refine(q, weights)` refits each block's scale, and its zero-point if it has one, to lower the mean squared error. it changes `q` in place, so call `q.copy()` first to keep the original. it lets other threads run while it works, so threads can refit several layers at once
|
|
51
|
+
- `learned.alternate(q, weights)` refits too, then rounds each value to the nearest code on its block's new line, and repeats until no code moves. it also changes `q` in place and lets other threads run. both can raise the worst error past an adaptive tensor's tolerance, and lowering the error of the weights doesn't always lower the error of a model's outputs, so check those too
|
|
52
|
+
- `Scheme.Q4_32.quantize(weights)` picks a scheme at run time, and `Scheme("symmetric(bits=4)")` reads one from text, like a config value, which `str(scheme)` writes
|
|
53
|
+
|
|
54
|
+
a block with outliers can need more than 8 bits, which raises `ToleranceTooTightError`. retrying with its `smallest_tolerance` works, but loosens every block, not just that one:
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
import numpy as np
|
|
58
|
+
from quantize import ToleranceTooTightError, adaptive
|
|
59
|
+
|
|
60
|
+
weights = np.array(weights) # a list has no .std(). skip this line for a pytorch tensor
|
|
61
|
+
try:
|
|
62
|
+
q = adaptive.quantize(weights, tolerance=0.1 * weights.std())
|
|
63
|
+
except ToleranceTooTightError as error:
|
|
64
|
+
q = adaptive.quantize(weights, tolerance=error.smallest_tolerance)
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
quantized values can be pickled, and compared with `==`. `q.to_bytes()` saves one as bytes, in the same format as the rust crate, and `Quantized.from_bytes(data)` loads it back. to keep it in an `np.savez` or safetensors file, store `np.frombuffer(q.to_bytes(), np.uint8)`.
|
|
68
|
+
|
|
69
|
+
to save its parts as plain arrays instead, like with `np.savez`, pass them back by name to `Quantized.from_parts`. leave out the one that's `None`: `bits` for an adaptive tensor, or `block_bits` for the others:
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
parts = dict(kind=q.kind, shape=q.shape, block=q.block, bits=q.bits, block_bits=q.block_bits,
|
|
73
|
+
codes=q.codes, scales=q.scales, zero_points=q.zero_points, scale=q.scale.name)
|
|
74
|
+
np.savez("layer.npz", **{name: part for name, part in parts.items() if part is not None})
|
|
75
|
+
q = Quantized.from_parts(**np.load("layer.npz"))
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
to keep quantized values in a `torch.save` checkpoint, store `torch.frombuffer(bytearray(q.to_bytes()), dtype=torch.uint8)`, and load each back with `Quantized(t)`. the checkpoint is then as small as the bytes, and `torch.load` reads it without `add_safe_globals`. pickling `q` itself makes the checkpoint about 1.5 times larger, since `torch.save` stores bytes as text, and needs `torch.serialization.add_safe_globals([Quantized])` before `torch.load`.
|
|
79
|
+
|
|
80
|
+
`q.matmul` runs on one core, but it lets other threads run while it multiplies, so threads can share out a batch. on an 8-core intel xeon, this multiplies a batch of 512 by a 4-bit 1536 × 576 matrix in 6 ms instead of 33 ms, with the same result, bit for bit:
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
import os
|
|
84
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
85
|
+
|
|
86
|
+
pool = ThreadPoolExecutor() # make it once, and reuse it for every layer
|
|
87
|
+
|
|
88
|
+
def linear(q, x): # x has shape (batch, columns)
|
|
89
|
+
# a piece per core, or pieces of 64 rows for a big batch, which stay in a core's cache
|
|
90
|
+
pieces = np.array_split(x, max(os.cpu_count(), len(x) // 64))
|
|
91
|
+
return np.concatenate(list(pool.map(q.matmul, pieces)))
|
|
92
|
+
|
|
93
|
+
out = linear(q, x)
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
with a gil, while another thread keeps running python, each piece can wait up to `sys.getswitchinterval()`, 5 ms by default, to get the gil back, so splitting pays off only while your other threads are idle or in native code, or on free-threaded python. `dequantize`, `dot`, and `matmul` on 65,536 values or fewer, like one vector times a 256 × 256 matrix, keep the gil, so they don't wait. quantizing and refitting keep it only up to 4,096 values, so threads can quantize a model's layers in parallel.
|
|
97
|
+
|
|
98
|
+
each value decodes as `code * scale`, or `(code - zero_point) * scale` with zero-points, using the scale and zero-point of its block. codes are signed and `bits` wide, and `q.codes` packs them low bits first. scales can be negative, since a symmetric block puts its value farthest from zero on the most negative code. `help(Quantized)` has the details.
|
|
99
|
+
|
|
100
|
+
to build and test from a clone of the repo, with rust 1.88 or newer and [just](https://github.com/casey/just):
|
|
101
|
+
|
|
102
|
+
```
|
|
103
|
+
just setup
|
|
104
|
+
just python
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
on debian or ubuntu, run `sudo apt install python3-venv` first.
|
|
108
|
+
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
quantize/__init__.py,sha256=KlP4rPPdqK2gOUWvc7vUvigj7nAeg63fLVvfk5D_LMA,883
|
|
2
|
+
quantize/_native.cp314t-win_amd64.pyd,sha256=5qB4daeLSNzB4hrhm3XRyXoXe20izArFMIBirUaJ_KA,771584
|
|
3
|
+
quantize/adaptive.py,sha256=B_VjnT5qedjJPOgTBNb21PG0py6gv1CmEwJzQ8ZR9R4,92
|
|
4
|
+
quantize/asymmetric.py,sha256=Wv25ZTOXFiSEwgiwuQwgtrv5Zz3nImDNBlt3YM5LUFw,172
|
|
5
|
+
quantize/learned.py,sha256=9CYixc3Beo6_w3so-j4g-FexdO6oKuBQbisX-osJQ7I,113
|
|
6
|
+
quantize/symmetric.py,sha256=qioEFBtQtaSHJSWzfL-0ZWGETnhazUvuCh-JLPK8qHs,89
|
|
7
|
+
quantize_py-0.3.2.dist-info/METADATA,sha256=b93PcIsSAILUu_uXo23Ub8fBmTdK4uo6pWxzfGwrLOI,7389
|
|
8
|
+
quantize_py-0.3.2.dist-info/WHEEL,sha256=kZsxXhvuxY5k9M6Ytk-lN_-XjHM99LBuWSI_qRbPktY,98
|
|
9
|
+
quantize_py-0.3.2.dist-info/licenses/LICENSE,sha256=4fE7Sr6YkE7fFrmdBRIK8N1sqjCdBnOuWlxxzEIWYh8,1092
|
|
10
|
+
quantize_py-0.3.2.dist-info/sboms/quantize-py.cyclonedx.json,sha256=9ztFhY2nHcO6LrIt6GSjSzzjUy9O9_uDpa8HlSJYqqI,44736
|
|
11
|
+
quantize_py-0.3.2.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025 Akshey Deokule
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|