quantize-py 0.3.2__cp314-cp314t-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
quantize/__init__.py ADDED
@@ -0,0 +1,41 @@
1
+ """Python bindings for quantize."""
2
+
3
+ from quantize._native import (
4
+ InvalidBitsError,
5
+ InvalidBlockError,
6
+ InvalidToleranceError,
7
+ LengthMismatchError,
8
+ NotAMatrixError,
9
+ QuantizeError,
10
+ Quantized,
11
+ Scale,
12
+ ScaleOutOfRangeError,
13
+ Scheme,
14
+ ShapeMismatchError,
15
+ ToleranceTooTightError,
16
+ __version__,
17
+ )
18
+
19
+ from . import adaptive, asymmetric, learned, symmetric
20
+ from .symmetric import quantize, quantize_tensor
21
+
22
+ __all__ = [
23
+ "Scale",
24
+ "Scheme",
25
+ "Quantized",
26
+ "QuantizeError",
27
+ "InvalidBitsError",
28
+ "InvalidBlockError",
29
+ "InvalidToleranceError",
30
+ "ToleranceTooTightError",
31
+ "ScaleOutOfRangeError",
32
+ "LengthMismatchError",
33
+ "ShapeMismatchError",
34
+ "NotAMatrixError",
35
+ "quantize",
36
+ "quantize_tensor",
37
+ "symmetric",
38
+ "asymmetric",
39
+ "adaptive",
40
+ "learned",
41
+ ]
Binary file
quantize/adaptive.py ADDED
@@ -0,0 +1,3 @@
1
+ """Adaptive quantization."""
2
+
3
+ from quantize._native import adaptive_quantize as quantize
quantize/asymmetric.py ADDED
@@ -0,0 +1,4 @@
1
+ """Asymmetric quantization."""
2
+
3
+ from quantize._native import asymmetric_quantize as quantize
4
+ from quantize._native import asymmetric_quantize_tensor as quantize_tensor
quantize/learned.py ADDED
@@ -0,0 +1,3 @@
1
+ """Learned quantization helpers."""
2
+
3
+ from quantize._native import alternate, fit_scale_and_zero_point, refine
quantize/symmetric.py ADDED
@@ -0,0 +1,3 @@
1
+ """Symmetric quantization."""
2
+
3
+ from quantize._native import quantize, quantize_tensor
@@ -0,0 +1,108 @@
1
+ Metadata-Version: 2.4
2
+ Name: quantize-py
3
+ Version: 0.3.2
4
+ Classifier: Programming Language :: Python :: 3
5
+ Classifier: Programming Language :: Python :: 3.12
6
+ Classifier: Programming Language :: Python :: 3.13
7
+ Classifier: Programming Language :: Python :: 3.14
8
+ Classifier: Programming Language :: Rust
9
+ Classifier: Topic :: Scientific/Engineering
10
+ Requires-Dist: numpy>=1.26
11
+ License-File: LICENSE
12
+ Summary: Python bindings for the quantize crate.
13
+ Keywords: quantization,machine-learning,numpy
14
+ Author: Akshey Deokule
15
+ License-Expression: MIT
16
+ Requires-Python: >=3.12
17
+ Description-Content-Type: text/markdown; charset=UTF-8; variant=GFM
18
+ Project-URL: Documentation, https://github.com/aksheyd/quantize/tree/main/python
19
+ Project-URL: Homepage, https://github.com/aksheyd/quantize
20
+ Project-URL: Repository, https://github.com/aksheyd/quantize
21
+
22
+ # quantize-py
23
+
24
+ python bindings for [quantize](https://github.com/aksheyd/quantize), a simple, fast quantization library written in rust. quantization stores numbers in fewer bits, trading a little accuracy for a lot less memory.
25
+
26
+ ```
27
+ pip install quantize-py
28
+ ```
29
+
30
+ requires python 3.12 or newer, or 3.14 or newer for free-threaded python. numpy is installed with it. where no wheel fits, like alpine linux, pip builds it from source, which needs rust 1.88 or newer.
31
+
32
+ ```python
33
+ from quantize import Scale, quantize
34
+
35
+ weights = [0.42, -0.10, 0.70, -0.50]
36
+
37
+ q = quantize(weights, bits=8, block=32, scale=Scale.F16)
38
+ back = q.dequantize() # [0.421, -0.098, 0.700, -0.498]
39
+ dot = q.dot(weights) # 0.926
40
+ ```
41
+
42
+ `bits` is the width of each code, from 2 to 16. `block` is how many values share one scale, and `scale` is how that scale is stored: `Scale.F32` (the default), `Scale.F16`, or `Scale.BF16`. values can be a list, a numpy array, or anything else `np.asarray` reads, like a pytorch tensor. a 2-d array keeps its shape, so `q.dequantize()` gives back a matrix and `q.matmul(x)` computes `x @ W.T`, like a linear layer. both can write into a float32 numpy array or pytorch tensor you pass, like `q.matmul(x, out=y)`, so a loop can reuse it. `q.shape` gives its rows and columns, and `len(q)` the number of values, as in rust.
43
+
44
+ the scales count toward the size: 4-bit codes with one f16 scale per 32 values cost 4.5 bits per value, or 5 with the default f32 scale. `q.bits_per_element` reports it. above about 10 bits, use f32 scales with `asymmetric.quantize`, since f16 and bf16 zero-points cap its accuracy.
45
+
46
+ the other schemes return the same `Quantized` type:
47
+
48
+ - `asymmetric.quantize(weights, bits=8, block=32)` adds a zero-point per block, for values that aren't centered on zero
49
+ - `adaptive.quantize(weights, tolerance=0.1 * weights.std())` gives each block the fewest bits, from 2 to 8, that round every weight within `tolerance`, in the weights' own units. a tenth of their standard deviation gives about 5 bits a block. for a list, use `np.std(weights)`
50
+ - `learned.refine(q, weights)` refits each block's scale, and its zero-point if it has one, to lower the mean squared error. it changes `q` in place, so call `q.copy()` first to keep the original. it lets other threads run while it works, so threads can refit several layers at once
51
+ - `learned.alternate(q, weights)` refits too, then rounds each value to the nearest code on its block's new line, and repeats until no code moves. it also changes `q` in place and lets other threads run. both can raise the worst error past an adaptive tensor's tolerance, and lowering the error of the weights doesn't always lower the error of a model's outputs, so check those too
52
+ - `Scheme.Q4_32.quantize(weights)` picks a scheme at run time, and `Scheme("symmetric(bits=4)")` reads one from text, like a config value, which `str(scheme)` writes
53
+
54
+ a block with outliers can need more than 8 bits, which raises `ToleranceTooTightError`. retrying with its `smallest_tolerance` works, but loosens every block, not just that one:
55
+
56
+ ```python
57
+ import numpy as np
58
+ from quantize import ToleranceTooTightError, adaptive
59
+
60
+ weights = np.array(weights) # a list has no .std(). skip this line for a pytorch tensor
61
+ try:
62
+ q = adaptive.quantize(weights, tolerance=0.1 * weights.std())
63
+ except ToleranceTooTightError as error:
64
+ q = adaptive.quantize(weights, tolerance=error.smallest_tolerance)
65
+ ```
66
+
67
+ quantized values can be pickled, and compared with `==`. `q.to_bytes()` saves one as bytes, in the same format as the rust crate, and `Quantized.from_bytes(data)` loads it back. to keep it in an `np.savez` or safetensors file, store `np.frombuffer(q.to_bytes(), np.uint8)`.
68
+
69
+ to save its parts as plain arrays instead, like with `np.savez`, pass them back by name to `Quantized.from_parts`. leave out the one that's `None`: `bits` for an adaptive tensor, or `block_bits` for the others:
70
+
71
+ ```python
72
+ parts = dict(kind=q.kind, shape=q.shape, block=q.block, bits=q.bits, block_bits=q.block_bits,
73
+ codes=q.codes, scales=q.scales, zero_points=q.zero_points, scale=q.scale.name)
74
+ np.savez("layer.npz", **{name: part for name, part in parts.items() if part is not None})
75
+ q = Quantized.from_parts(**np.load("layer.npz"))
76
+ ```
77
+
78
+ to keep quantized values in a `torch.save` checkpoint, store `torch.frombuffer(bytearray(q.to_bytes()), dtype=torch.uint8)`, and load each back with `Quantized(t)`. the checkpoint is then as small as the bytes, and `torch.load` reads it without `add_safe_globals`. pickling `q` itself makes the checkpoint about 1.5 times larger, since `torch.save` stores bytes as text, and needs `torch.serialization.add_safe_globals([Quantized])` before `torch.load`.
79
+
80
+ `q.matmul` runs on one core, but it lets other threads run while it multiplies, so threads can share out a batch. on an 8-core intel xeon, this multiplies a batch of 512 by a 4-bit 1536 × 576 matrix in 6 ms instead of 33 ms, with the same result, bit for bit:
81
+
82
+ ```python
83
+ import os
84
+ from concurrent.futures import ThreadPoolExecutor
85
+
86
+ pool = ThreadPoolExecutor() # make it once, and reuse it for every layer
87
+
88
+ def linear(q, x): # x has shape (batch, columns)
89
+ # a piece per core, or pieces of 64 rows for a big batch, which stay in a core's cache
90
+ pieces = np.array_split(x, max(os.cpu_count(), len(x) // 64))
91
+ return np.concatenate(list(pool.map(q.matmul, pieces)))
92
+
93
+ out = linear(q, x)
94
+ ```
95
+
96
+ with a gil, while another thread keeps running python, each piece can wait up to `sys.getswitchinterval()`, 5 ms by default, to get the gil back, so splitting pays off only while your other threads are idle or in native code, or on free-threaded python. `dequantize`, `dot`, and `matmul` on 65,536 values or fewer, like one vector times a 256 × 256 matrix, keep the gil, so they don't wait. quantizing and refitting keep it only up to 4,096 values, so threads can quantize a model's layers in parallel.
97
+
98
+ each value decodes as `code * scale`, or `(code - zero_point) * scale` with zero-points, using the scale and zero-point of its block. codes are signed and `bits` wide, and `q.codes` packs them low bits first. scales can be negative, since a symmetric block puts its value farthest from zero on the most negative code. `help(Quantized)` has the details.
99
+
100
+ to build and test from a clone of the repo, with rust 1.88 or newer and [just](https://github.com/casey/just):
101
+
102
+ ```
103
+ just setup
104
+ just python
105
+ ```
106
+
107
+ on debian or ubuntu, run `sudo apt install python3-venv` first.
108
+
@@ -0,0 +1,11 @@
1
+ quantize/__init__.py,sha256=KlP4rPPdqK2gOUWvc7vUvigj7nAeg63fLVvfk5D_LMA,883
2
+ quantize/_native.cp314t-win_amd64.pyd,sha256=5qB4daeLSNzB4hrhm3XRyXoXe20izArFMIBirUaJ_KA,771584
3
+ quantize/adaptive.py,sha256=B_VjnT5qedjJPOgTBNb21PG0py6gv1CmEwJzQ8ZR9R4,92
4
+ quantize/asymmetric.py,sha256=Wv25ZTOXFiSEwgiwuQwgtrv5Zz3nImDNBlt3YM5LUFw,172
5
+ quantize/learned.py,sha256=9CYixc3Beo6_w3so-j4g-FexdO6oKuBQbisX-osJQ7I,113
6
+ quantize/symmetric.py,sha256=qioEFBtQtaSHJSWzfL-0ZWGETnhazUvuCh-JLPK8qHs,89
7
+ quantize_py-0.3.2.dist-info/METADATA,sha256=b93PcIsSAILUu_uXo23Ub8fBmTdK4uo6pWxzfGwrLOI,7389
8
+ quantize_py-0.3.2.dist-info/WHEEL,sha256=kZsxXhvuxY5k9M6Ytk-lN_-XjHM99LBuWSI_qRbPktY,98
9
+ quantize_py-0.3.2.dist-info/licenses/LICENSE,sha256=4fE7Sr6YkE7fFrmdBRIK8N1sqjCdBnOuWlxxzEIWYh8,1092
10
+ quantize_py-0.3.2.dist-info/sboms/quantize-py.cyclonedx.json,sha256=9ztFhY2nHcO6LrIt6GSjSzzjUy9O9_uDpa8HlSJYqqI,44736
11
+ quantize_py-0.3.2.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: maturin (1.15.0)
3
+ Root-Is-Purelib: false
4
+ Tag: cp314-cp314t-win_amd64
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025 Akshey Deokule
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.