quantdiff 0.1.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quantdiff/__init__.py +53 -0
- quantdiff/__main__.py +5 -0
- quantdiff/_http.py +151 -0
- quantdiff/_text.py +13 -0
- quantdiff/_version.py +1 -0
- quantdiff/api.py +340 -0
- quantdiff/backends/__init__.py +28 -0
- quantdiff/backends/_common.py +342 -0
- quantdiff/backends/base.py +91 -0
- quantdiff/backends/llamacpp.py +428 -0
- quantdiff/backends/ollama.py +359 -0
- quantdiff/backends/openai_compat.py +338 -0
- quantdiff/cache.py +240 -0
- quantdiff/card.py +1664 -0
- quantdiff/cli.py +377 -0
- quantdiff/discover.py +488 -0
- quantdiff/errors.py +45 -0
- quantdiff/metrics/__init__.py +36 -0
- quantdiff/metrics/codeexec.py +428 -0
- quantdiff/metrics/jsonschema.py +610 -0
- quantdiff/metrics/logit.py +214 -0
- quantdiff/metrics/tasks.py +114 -0
- quantdiff/metrics/textsim.py +66 -0
- quantdiff/metrics/toolcheck.py +99 -0
- quantdiff/png.py +360 -0
- quantdiff/preflight.py +365 -0
- quantdiff/progress.py +283 -0
- quantdiff/py.typed +0 -0
- quantdiff/report.py +780 -0
- quantdiff/runner.py +492 -0
- quantdiff/spec.py +154 -0
- quantdiff/stats.py +226 -0
- quantdiff/suites/__init__.py +462 -0
- quantdiff/suites/data/chat.jsonl +22 -0
- quantdiff/suites/data/code.jsonl +32 -0
- quantdiff/suites/data/json.jsonl +34 -0
- quantdiff/suites/data/scoring.jsonl +41 -0
- quantdiff/suites/data/tools.jsonl +32 -0
- quantdiff/types.py +322 -0
- quantdiff/verdict.py +1513 -0
- quantdiff-0.1.0rc1.dist-info/METADATA +514 -0
- quantdiff-0.1.0rc1.dist-info/RECORD +45 -0
- quantdiff-0.1.0rc1.dist-info/WHEEL +4 -0
- quantdiff-0.1.0rc1.dist-info/entry_points.txt +2 -0
- quantdiff-0.1.0rc1.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,514 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: quantdiff
|
|
3
|
+
Version: 0.1.0rc1
|
|
4
|
+
Summary: Find out which download of a local model gives the best answers on your own prompts.
|
|
5
|
+
Project-URL: Homepage, https://github.com/MadhankumarAI/Quantdiff
|
|
6
|
+
Project-URL: Issues, https://github.com/MadhankumarAI/Quantdiff/issues
|
|
7
|
+
Author: quantdiff contributors
|
|
8
|
+
License-Expression: Apache-2.0
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: evaluation,gguf,kl-divergence,llama.cpp,llm,ollama,quantization
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
23
|
+
Classifier: Typing :: Typed
|
|
24
|
+
Requires-Python: >=3.10
|
|
25
|
+
Provides-Extra: dev
|
|
26
|
+
Requires-Dist: mypy==2.4.0; extra == 'dev'
|
|
27
|
+
Requires-Dist: pytest==8.4.2; extra == 'dev'
|
|
28
|
+
Requires-Dist: ruff==0.16.10; extra == 'dev'
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
|
|
31
|
+
# quantdiff
|
|
32
|
+
|
|
33
|
+
**Find out which download of a local model to actually run, judged on your own prompts.**
|
|
34
|
+
|
|
35
|
+
quantdiff compares quantized variants of the same model (an Ollama tag, an Unsloth GGUF, a
|
|
36
|
+
bartowski GGUF in llama-server, the same model in LM Studio or vLLM) against a higher precision
|
|
37
|
+
reference and prints a scorecard you can paste into a terminal, a Reddit post, or a browser.
|
|
38
|
+
|
|
39
|
+
llmfit tells you what fits. quantdiff tells you which version is best and whether it is being
|
|
40
|
+
served correctly.
|
|
41
|
+
|
|
42
|
+
- Zero runtime dependencies. Standard library only.
|
|
43
|
+
- Works with Ollama, llama-server (llama.cpp), and any OpenAI-compatible server (LM Studio, vLLM,
|
|
44
|
+
mlx_lm.server, and others).
|
|
45
|
+
- Outputs a shareable `card.png`, a Reddit-ready `card.md`, a self-contained `card.html`, and
|
|
46
|
+
the raw `report.json`.
|
|
47
|
+
|
|
48
|
+
## Why
|
|
49
|
+
|
|
50
|
+
Every time a model ships, several uploaders race to publish quants within hours. Chat templates
|
|
51
|
+
get patched after release. Some uploads use an imatrix, some do not, some get re-uploaded twice in
|
|
52
|
+
the first week. Most people pick a quant by habit, by file size, or by someone else's wikitext
|
|
53
|
+
perplexity numbers measured on a different build.
|
|
54
|
+
|
|
55
|
+
None of that tells you how a given download behaves on the prompts you care about, or whether the
|
|
56
|
+
server you run it in is applying the right template and the full context window. quantdiff
|
|
57
|
+
measures that directly: same prompts, same greedy decoding, every variant checked against one
|
|
58
|
+
reference, with the setup problems caught before scoring starts.
|
|
59
|
+
|
|
60
|
+
## Install
|
|
61
|
+
|
|
62
|
+
```
|
|
63
|
+
pip install quantdiff
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
or run it without installing:
|
|
67
|
+
|
|
68
|
+
```
|
|
69
|
+
uvx quantdiff --version
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Python 3.10 or newer. Nothing else is pulled in.
|
|
73
|
+
|
|
74
|
+
## Quickstart: Ollama (60 seconds)
|
|
75
|
+
|
|
76
|
+
Pull two quantizations of a small model, then let quantdiff find them:
|
|
77
|
+
|
|
78
|
+
```
|
|
79
|
+
ollama pull qwen2.5:0.5b-instruct-q8_0
|
|
80
|
+
ollama pull qwen2.5:0.5b-instruct-q4_K_M
|
|
81
|
+
quantdiff discover
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
`discover` groups the downloads of each model, picks the most precise one as the reference, and
|
|
85
|
+
prints the command to run:
|
|
86
|
+
|
|
87
|
+
```
|
|
88
|
+
qwen2.5:0.5b-instruct
|
|
89
|
+
Q8_0 531 MB qwen2.5:0.5b-instruct-q8_0 (reference)
|
|
90
|
+
Q4_K_M 398 MB qwen2.5:0.5b-instruct-q4_K_M
|
|
91
|
+
Compare them:
|
|
92
|
+
quantdiff run --ref ollama:qwen2.5:0.5b-instruct-q8_0 --cand ollama:qwen2.5:0.5b-instruct-q4_K_M
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
Paste it. quantdiff checks that every server and model is reachable before doing any work, shows
|
|
96
|
+
a progress bar with an ETA, prints the scorecard, and writes `runs/<UTC timestamp>/` with
|
|
97
|
+
`card.png`, `card.md`, `card.html` and `report.json`. Add `--open` to open the card when it is
|
|
98
|
+
done.
|
|
99
|
+
|
|
100
|
+
quantdiff uses `OLLAMA_HOST` if set, otherwise `http://127.0.0.1:11434`.
|
|
101
|
+
|
|
102
|
+
## Quickstart: llama-server
|
|
103
|
+
|
|
104
|
+
Start one llama-server per variant on different ports. quantdiff does not launch servers for you.
|
|
105
|
+
|
|
106
|
+
```
|
|
107
|
+
llama-server -hf bartowski/Qwen2.5-1.5B-Instruct-GGUF:Q8_0 --port 8080
|
|
108
|
+
llama-server -hf bartowski/Qwen2.5-1.5B-Instruct-GGUF:Q4_K_M --port 8081
|
|
109
|
+
llama-server -hf unsloth/Qwen2.5-1.5B-Instruct-GGUF:Q4_K_M --port 8082
|
|
110
|
+
|
|
111
|
+
quantdiff run \
|
|
112
|
+
--ref q8=llamacpp:http://127.0.0.1:8080 \
|
|
113
|
+
--cand bartowski-q4km=llamacpp:http://127.0.0.1:8081 \
|
|
114
|
+
--cand unsloth-q4km=llamacpp:http://127.0.0.1:8082 \
|
|
115
|
+
--hf-repo Qwen/Qwen2.5-1.5B-Instruct
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
llama-server is the most precise backend: when the reference is also llama-server, quantdiff
|
|
119
|
+
feeds exact token ids for the logit metrics, and `--hf-repo` lets it compare each server's embedded
|
|
120
|
+
chat template with the upstream `tokenizer_config.json` on Hugging Face.
|
|
121
|
+
|
|
122
|
+
You can mix backends. A common question is "is the Ollama tag as good as the GGUF I would run in
|
|
123
|
+
llama-server?" With a text-only reference such as Ollama, every candidate is teacher-forced by
|
|
124
|
+
text, and the card footnotes that:
|
|
125
|
+
|
|
126
|
+
```
|
|
127
|
+
quantdiff run --ref ollama:qwen2.5:1.5b-instruct-q8_0 \
|
|
128
|
+
--cand ollama=ollama:qwen2.5:1.5b-instruct-q4_K_M \
|
|
129
|
+
--cand gguf=llamacpp:http://127.0.0.1:8081
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
## LM Studio, vLLM, MLX, and other OpenAI-compatible servers
|
|
133
|
+
|
|
134
|
+
Use `openai:<base_url>#<model>`, where `base_url` is the same URL you would give an OpenAI client.
|
|
135
|
+
|
|
136
|
+
```
|
|
137
|
+
# LM Studio (Developer tab, start server)
|
|
138
|
+
quantdiff run --ref ollama:qwen2.5:1.5b-instruct-q8_0 \
|
|
139
|
+
--cand lmstudio=openai:http://127.0.0.1:1234/v1#qwen2.5-1.5b-instruct
|
|
140
|
+
|
|
141
|
+
# vLLM
|
|
142
|
+
vllm serve Qwen/Qwen2.5-1.5B-Instruct-AWQ --port 8000
|
|
143
|
+
quantdiff run --ref ollama:qwen2.5:1.5b-instruct-q8_0 \
|
|
144
|
+
--cand awq=openai:http://127.0.0.1:8000/v1#Qwen/Qwen2.5-1.5B-Instruct-AWQ
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
If the server needs an API key, put it in an environment variable and name the variable in the
|
|
148
|
+
spec. The key itself never appears on the command line or in the report.
|
|
149
|
+
|
|
150
|
+
```
|
|
151
|
+
export MY_SERVER_KEY=...
|
|
152
|
+
quantdiff run ... --cand remote=openai:https://gpu-box.lan/v1#my-model@env:MY_SERVER_KEY
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
## Spec grammar
|
|
156
|
+
|
|
157
|
+
```
|
|
158
|
+
[label=]ollama:<tag> e.g. ollama:qwen2.5:7b-instruct-q4_K_M
|
|
159
|
+
[label=]llamacpp:<base_url> e.g. llamacpp:http://127.0.0.1:8080
|
|
160
|
+
[label=]openai:<base_url>#<model>[@env:VAR] e.g. openai:http://127.0.0.1:1234/v1#qwen2.5-7b-instruct
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
The optional `label=` prefix sets the name shown on the scorecard. Without it quantdiff derives
|
|
164
|
+
one from the spec.
|
|
165
|
+
|
|
166
|
+
## CLI reference
|
|
167
|
+
|
|
168
|
+
```
|
|
169
|
+
quantdiff discover [--host URL] # list local Ollama models
|
|
170
|
+
quantdiff run --ref SPEC --cand SPEC [--cand SPEC ...] [options]
|
|
171
|
+
quantdiff card PATH [--format txt|md|html|png] [-o FILE] # PATH = run dir or report.json
|
|
172
|
+
quantdiff suites # list built-in suites
|
|
173
|
+
quantdiff --version
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
`-v` / `--verbose` (before or after the command) prints debug logs to stderr.
|
|
177
|
+
|
|
178
|
+
| `run` option | Meaning |
|
|
179
|
+
| --- | --- |
|
|
180
|
+
| `--ref` | The reference model. Use the highest precision you can serve. |
|
|
181
|
+
| `--cand` | A candidate. Repeat for each variant. |
|
|
182
|
+
| `--suite` | Comma separated task suites. Default `json,tools,chat`, plus `code` with `--allow-code-exec`. |
|
|
183
|
+
| `--prompts` | Your own chat prompts as JSONL (see below). Without `--suite`, only these run. |
|
|
184
|
+
| `--scoring-prompts` | Your own raw text prompts for the logit metrics, as JSONL. |
|
|
185
|
+
| `--max-cases N` | Cap each suite, your prompts file and the scoring prompts at N. For quick runs. |
|
|
186
|
+
| `--allow-code-exec` | Run the `code` suite's hidden tests against model output. Off by default. |
|
|
187
|
+
| `--top-k` | Top tokens requested per position for the logit metrics (1 to 20, default 10). |
|
|
188
|
+
| `--score-tokens` | Reference tokens scored per scoring prompt (default 32; 0 disables Tier 1). |
|
|
189
|
+
| `--seed` | Seed sent to every server (default 0). |
|
|
190
|
+
| `--hf-repo` | Upstream Hugging Face repo used for the chat template check. |
|
|
191
|
+
| `--offline` | Never contact huggingface.co. |
|
|
192
|
+
| `--no-preflight` | Skip the pre-flight checks. Not recommended. |
|
|
193
|
+
| `--context-probe-tokens` | Length of the truncation probe (default 6000, minimum 1000, 0 disables). |
|
|
194
|
+
| `--title` | Title printed on the scorecard. |
|
|
195
|
+
| `--out` | Parent directory for runs. Default `runs`. |
|
|
196
|
+
| `--no-png` | Skip `card.png`. |
|
|
197
|
+
| `--open` | Open the HTML card in your browser when the run finishes. |
|
|
198
|
+
| `--no-cache` | Rerun the reference even if its outputs are cached. |
|
|
199
|
+
| `-q`, `--quiet` | No progress output. |
|
|
200
|
+
| `--max-size SIZE` | Largest download you can run, such as `6GB`, `6.5G` or `800MB` (decimal units). The pick is the smallest close download that fits; with none, the best USABLE one that fits. |
|
|
201
|
+
| `--fail-on LEVEL` | `avoid`: exit 3 on any AVOID or FAILED candidate, or when nothing is RUN and nothing USABLE fits. `inconclusive`: also on any UNSURE or USABLE candidate, or when nothing is RUN. Default `never`. |
|
|
202
|
+
|
|
203
|
+
`quantdiff card` re-renders a scorecard from a finished run without touching any server.
|
|
204
|
+
`--format png` writes a 2x image for Reddit or X.
|
|
205
|
+
|
|
206
|
+
## What the scorecard shows
|
|
207
|
+
|
|
208
|
+
The card answers the question first: one headline sentence ("Run q4_K_M ..."), then a row per
|
|
209
|
+
model with a status chip (REF, RUN, OK, AVOID, UNSURE, FAILED) and a short reason. The metrics
|
|
210
|
+
behind it:
|
|
211
|
+
|
|
212
|
+
| Tier | Metric | What it measures | Backends |
|
|
213
|
+
| --- | --- | --- | --- |
|
|
214
|
+
| 1 (logit) | top-1 agreement | Fraction of positions where the candidate's most likely next token equals the reference's greedy token, with the reference's text fed in. Higher is better. | Any backend that returns logprobs |
|
|
215
|
+
| 1 (logit) | KLD mean / p99 | KL divergence from the reference's next-token distribution. Lower is better. A lower bound on full-vocabulary KL. p99 is shown only with 1000+ positions. | Same |
|
|
216
|
+
| 2 (task) | json | Output parses and validates against the case's JSON schema. | All |
|
|
217
|
+
| 2 (task) | tools | Calls the expected tool, with valid arguments, matching the expected values. | All |
|
|
218
|
+
| 2 (task) | code | Generated function passes hidden asserts. Only with `--allow-code-exec`. | All |
|
|
219
|
+
| 2 (task) | agree | Text similarity of chat answers to the reference model's answers. | All |
|
|
220
|
+
| size | Size | Size of the download on disk, and the change against the reference. | Ollama, llama-server |
|
|
221
|
+
| perf | tok/s | Decode speed as measured by the server (labelled "wall" when only wall-clock time is available). | All |
|
|
222
|
+
|
|
223
|
+
KLD is put in plain bands: near-lossless (under 0.01), small (under 0.04), moderate (under 0.10)
|
|
224
|
+
and large, at the default `--top-k 10`. The bars are calibrated against llama.cpp's
|
|
225
|
+
full-vocabulary `llama-perplexity --kl-divergence` on the same weights (quantdiff's top-10 bound
|
|
226
|
+
reads about two thirds of it), so 0.04 here is about 0.06 there, a typical Q4_K_M of a 7 to 8B
|
|
227
|
+
model. See [docs/calibration.md](docs/calibration.md).
|
|
228
|
+
|
|
229
|
+
The reference appears as its own REF row with its task pass rates. Candidate pass rates carry a
|
|
230
|
+
paired delta against it, for example `-30*`; a `*` marks a significant difference, and the HTML
|
|
231
|
+
card adds the 95% interval. A gain that is within noise shows as `=`, never as a win. Suites
|
|
232
|
+
where the reference itself passes under half the cases are greyed out and not used to judge.
|
|
233
|
+
Columns for suites that did not run are hidden. When every model shares a name prefix, the card
|
|
234
|
+
states it once and labels rows by their quant (`q4_K_M`, `q2_K`).
|
|
235
|
+
|
|
236
|
+
Logit metrics are exact when the reference and candidate are both llama-server. Everywhere else
|
|
237
|
+
teacher forcing goes through text, and the card notes it (see FAQ).
|
|
238
|
+
|
|
239
|
+
Pre-flight checks run on every server before it is scored. They are listed once under
|
|
240
|
+
**Server checks**, each with a sentence on whether it could have changed these scores:
|
|
241
|
+
|
|
242
|
+
- **Context truncation.** A needle placed at the start of a long prompt (default 6000 tokens),
|
|
243
|
+
plus a short control prompt. If the model finds the needle in the control but not in the long
|
|
244
|
+
prompt, the server is silently truncating, which is common with Ollama's default `num_ctx`.
|
|
245
|
+
- **Chat template.** llama-server only: the template embedded in the GGUF versus the upstream
|
|
246
|
+
`tokenizer_config.json` from `--hf-repo`. Catches quants uploaded before a template fix.
|
|
247
|
+
- **Tokenizer match.** Candidate and reference must tokenize the same text identically, otherwise
|
|
248
|
+
logit metrics are not comparable and are withheld.
|
|
249
|
+
- **Logprob availability.** If a server does not return logprobs, Tier 1 is skipped for it and
|
|
250
|
+
the card says so.
|
|
251
|
+
|
|
252
|
+
## Example scorecard
|
|
253
|
+
|
|
254
|
+
A real run with default settings, exactly the command `quantdiff discover` printed (Qwen2.5 0.5B
|
|
255
|
+
Instruct from Ollama on a GTX 1650, 2 minutes 49 seconds):
|
|
256
|
+
|
|
257
|
+
```
|
|
258
|
+
quantdiff run --ref ollama:qwen2.5:0.5b-instruct-q8_0 --cand ollama:qwen2.5:0.5b-instruct-q4_K_M --cand ollama:qwen2.5:0.5b-instruct-q2_K
|
|
259
|
+
```
|
|
260
|
+
|
|
261
|
+
This is `card.png` exactly as quantdiff wrote it:
|
|
262
|
+
|
|
263
|
+

|
|
264
|
+
|
|
265
|
+
Q4_K_M is the pick: its KLD interval (0.028 to 0.036) lies entirely under the 0.04 closeness bar,
|
|
266
|
+
and its task scores are within 10 points of Q8_0 on 66 paired cases. The card flags it as near
|
|
267
|
+
the bar, because the interval's upper end is within 10% of it. Q2_K is marked AVOID because its
|
|
268
|
+
whole KLD interval (0.104 to 0.134) is in the large band, about Q2_K territory on llama.cpp's
|
|
269
|
+
full-vocabulary scale; its 19-point drop on tool calls is shown as an unresolved caveat, since 32
|
|
270
|
+
cases cannot settle it. The context-length note is muted because it cannot have affected these
|
|
271
|
+
short prompts.
|
|
272
|
+
|
|
273
|
+
Running only `--ref ...q8_0 --cand ...q4_K_M` answers the most common question, "can I drop from
|
|
274
|
+
Q8 to Q4?", with the same headline in about a minute and a half, since the reference's outputs
|
|
275
|
+
come from the cache.
|
|
276
|
+
|
|
277
|
+
## Reading the verdict
|
|
278
|
+
|
|
279
|
+
The card opens with one sentence that answers the question you ran it for, such as:
|
|
280
|
+
|
|
281
|
+
```
|
|
282
|
+
Run q4_K_M: 25% smaller than q8_0, close on logits (KLD 0.03, CI up to 0.036) on 41 prompts.
|
|
283
|
+
Best that fits 6 GB: q4_K_M, moderate loss (KLD 0.044).
|
|
284
|
+
No download is close to q8_0; q4_K_M has the smallest loss among the smaller downloads (moderate, KLD 0.044).
|
|
285
|
+
Keep q8_0 for now: no candidate is shown to be close on 12 prompts and 24 cases.
|
|
286
|
+
Keep q8_0: q2_K shows a measured loss.
|
|
287
|
+
```
|
|
288
|
+
|
|
289
|
+
The sentences under it give the numbers, for example "q4_K_M: KLD 0.029 (95% CI 0.024 to
|
|
290
|
+
0.035) and no significant task loss vs q8_0 on 24 cases (95% CI -21 to +12)." When nothing is
|
|
291
|
+
recommended, a separate next step says what to do, such as "Rerun with --max-cases 40 to
|
|
292
|
+
decide." or "Pass --max-size with the memory you can spare".
|
|
293
|
+
|
|
294
|
+
Every candidate gets a status: **RUN** (the one to use), **OK** (also close to the reference,
|
|
295
|
+
but larger than the pick or over your `--max-size`), **USABLE** (a measured, moderate loss; a
|
|
296
|
+
sound choice when nothing closer fits), **AVOID** (a large loss or measured task breakage),
|
|
297
|
+
**UNSURE** (not enough evidence either way) or **FAILED**.
|
|
298
|
+
|
|
299
|
+
Each candidate is judged against the reference on its own, so adding or removing a candidate
|
|
300
|
+
never changes another one's status. A candidate earns RUN or OK only with positive evidence
|
|
301
|
+
that it is close:
|
|
302
|
+
|
|
303
|
+
- **With logit metrics** (any server that returns logprobs), closeness rests on KLD: at least 8
|
|
304
|
+
scoring prompts, and the 95% interval of the mean KLD ends below 0.04 (at the default
|
|
305
|
+
`--top-k 10`; the bar scales with top-k). A KLD shown that small also bounds how far the two
|
|
306
|
+
models' outputs can differ. Task suites of a few dozen cases cannot bound small differences,
|
|
307
|
+
so here they work as a breakage detector: any significant loss is AVOID, and a wide interval
|
|
308
|
+
is shown honestly ("no significant task loss vs q8_0 on 24 cases"), never claimed as proof.
|
|
309
|
+
- **Without logit metrics**, tasks must prove closeness on their own: the suites the reference
|
|
310
|
+
passes at least half of are pooled, and with at least 20 paired cases the 95% interval of the
|
|
311
|
+
pass-rate difference must not reach more than 10 points below zero. That usually takes more
|
|
312
|
+
cases than the built-in suites hold, so expect UNSURE and add your own prompts.
|
|
313
|
+
|
|
314
|
+
The closeness bar sits about where a Q4_K_M of a 7 to 8B model lands, so a popular quant can
|
|
315
|
+
fall just above it. With at least 8 prompts, a KLD interval that starts above 0.04 with a mean
|
|
316
|
+
under 0.10 is USABLE, not AVOID. A candidate is AVOID when a task suite, or the suites pooled
|
|
317
|
+
together, drop significantly against the reference (paired McNemar test), or when at least 8
|
|
318
|
+
prompts put both its mean KLD and the low end of its interval at 0.10 or more. A large-looking
|
|
319
|
+
KLD on fewer prompts is UNSURE ("looks large on 3 prompts"). A candidate never "beats" the
|
|
320
|
+
reference: a higher score that is within noise is shown as equal, and a significantly higher
|
|
321
|
+
one is flagged as a sign the suite is too small or the reference is itself quantized.
|
|
322
|
+
|
|
323
|
+
**Pick by size.** Pass `--max-size` with the largest download you can run (decimal units:
|
|
324
|
+
`6GB`, `6.5G`, `800MB`). The pick is then the smallest close candidate that fits. When no close
|
|
325
|
+
candidate fits, the headline names the USABLE candidate with the lowest KLD that fits ("Best
|
|
326
|
+
that fits 6 GB: ...") and the details name the close download that does not ("q5_K_M is close
|
|
327
|
+
but needs 7.1 GB."). A candidate whose size the server does not report is never assumed to fit.
|
|
328
|
+
|
|
329
|
+
**UNSURE does not mean safe.** It means this run could not decide. The card names the candidate
|
|
330
|
+
that looks closest, says what is missing, and estimates how much more would decide, for example
|
|
331
|
+
"Rerun with --max-cases 40 to decide.", or how many of your own prompts to add when the
|
|
332
|
+
built-in suites are too small.
|
|
333
|
+
|
|
334
|
+
Some calls come with a caveat that does not change the status but is worth reading:
|
|
335
|
+
|
|
336
|
+
- "near the closeness bar; a rerun could change this": the KLD interval straddles a bar or ends
|
|
337
|
+
within 10% of it.
|
|
338
|
+
- "tools -17 unresolved (95% CI -42 to +6); rerun with --max-cases 60": a suite scored more than
|
|
339
|
+
10 points lower without the drop being significant. The first one is shown right under the
|
|
340
|
+
headline.
|
|
341
|
+
- On Ollama and other text-forced servers, when most scoring prompts are not Latin script, the
|
|
342
|
+
KLD may read high; the card says to confirm with llama-server (exact token ids) before ruling
|
|
343
|
+
out a download.
|
|
344
|
+
|
|
345
|
+
The full rules are in [docs/methodology.md](docs/methodology.md#how-the-verdict-is-decided).
|
|
346
|
+
|
|
347
|
+
Things to check before trusting a verdict:
|
|
348
|
+
|
|
349
|
+
- **Server checks** marked as affecting the scores. A truncated context or a stale template
|
|
350
|
+
usually explains a bad score better than the quantization does. Checks that cannot have changed
|
|
351
|
+
these numbers (for example a 4096-token context when every prompt is shorter) are shown muted.
|
|
352
|
+
- How many cases ran. `--max-cases 5` is for smoke tests, not conclusions.
|
|
353
|
+
- Whether the reference is BF16/F16 or a Q8_0 proxy. A Q8_0 reference makes every candidate look
|
|
354
|
+
slightly closer than it is.
|
|
355
|
+
|
|
356
|
+
To use quantdiff in CI, pass `--fail-on avoid`: the run exits with code 3 when any candidate is
|
|
357
|
+
AVOID or FAILED, or when no candidate is RUN and no USABLE candidate fits your `--max-size`
|
|
358
|
+
(any USABLE one counts when there is no budget), so a run with too little evidence never
|
|
359
|
+
passes. `--fail-on inconclusive` also fails on any UNSURE or USABLE candidate, and whenever no
|
|
360
|
+
candidate is RUN.
|
|
361
|
+
|
|
362
|
+
## Your own prompts
|
|
363
|
+
|
|
364
|
+
The built-in suites are a starting point. The prompts that matter are yours.
|
|
365
|
+
|
|
366
|
+
`--prompts` takes JSONL. The shorthand form is one prompt per line:
|
|
367
|
+
|
|
368
|
+
```
|
|
369
|
+
{"prompt": "Summarize this incident report in three bullet points: ..."}
|
|
370
|
+
{"prompt": "Write a SQL query that returns the top 5 customers by revenue in 2025."}
|
|
371
|
+
```
|
|
372
|
+
|
|
373
|
+
Shorthand prompts are scored as `chat` cases: by agreement with the reference model's answer.
|
|
374
|
+
|
|
375
|
+
For pass/fail checks, write full task cases. Field names match `quantdiff.types.TaskCase`:
|
|
376
|
+
|
|
377
|
+
```
|
|
378
|
+
{"id": "invoice-1", "kind": "json", "messages": [{"role": "user", "content": "Extract vendor and total from: ..."}], "json_schema": {"type": "object", "required": ["vendor", "total"], "properties": {"vendor": {"type": "string"}, "total": {"type": "number"}}}}
|
|
379
|
+
{"id": "weather-1", "kind": "tools", "messages": [{"role": "user", "content": "What is the weather in Oslo?"}], "tools": [{"name": "get_weather", "description": "Current weather for a city", "parameters": {"type": "object", "properties": {"city": {"type": "string"}}, "required": ["city"]}}], "expected_tool": "get_weather", "expected_arguments": {"city": "Oslo"}}
|
|
380
|
+
```
|
|
381
|
+
|
|
382
|
+
`expected_arguments` is a subset match: every key listed must be present with an equal value.
|
|
383
|
+
|
|
384
|
+
`--scoring-prompts` takes raw text prompts (no chat template) for the Tier 1 metrics, one
|
|
385
|
+
`{"id": "...", "text": "..."}` object per line. Use text that looks like your workload: code,
|
|
386
|
+
legal prose, a language other than English.
|
|
387
|
+
|
|
388
|
+
## Python API
|
|
389
|
+
|
|
390
|
+
```python
|
|
391
|
+
import quantdiff
|
|
392
|
+
|
|
393
|
+
report = quantdiff.compare(
|
|
394
|
+
ref="ollama:qwen2.5:7b-instruct-q8_0",
|
|
395
|
+
candidates=["ollama:qwen2.5:7b-instruct-q4_K_M", "llamacpp:http://127.0.0.1:8080"],
|
|
396
|
+
suites=("json", "tools"),
|
|
397
|
+
prompts=None,
|
|
398
|
+
)
|
|
399
|
+
print(quantdiff.verdict(report))
|
|
400
|
+
with open("card.html", "w", encoding="utf-8") as f:
|
|
401
|
+
f.write(quantdiff.render_html(report))
|
|
402
|
+
```
|
|
403
|
+
|
|
404
|
+
## How it works
|
|
405
|
+
|
|
406
|
+
1. Pre-flight checks on every server (context, template, tokenizer, logprobs).
|
|
407
|
+
2. Tier 1: the reference greedily continues each raw scoring prompt and records its top-k
|
|
408
|
+
distribution at each step. Each candidate is then teacher-forced on the same text and its top-k
|
|
409
|
+
distributions are compared position by position.
|
|
410
|
+
3. Tier 2: each task case is sent through each server's own chat template with greedy decoding,
|
|
411
|
+
and the answer is checked (schema, tool call, hidden tests, or agreement with the reference).
|
|
412
|
+
4. Results are aggregated, tested against noise, and rendered.
|
|
413
|
+
|
|
414
|
+
The reference model's outputs are cached, so the second run against the same reference only runs
|
|
415
|
+
the candidates. The cache key covers the server URL, the weights fingerprint (the Ollama digest or
|
|
416
|
+
the GGUF size), the chat template, the suites and the settings, so re-pulling a fixed upload
|
|
417
|
+
invalidates it. The cache lives in `%LOCALAPPDATA%\quantdiff\cache` on Windows and in
|
|
418
|
+
`$XDG_CACHE_HOME/quantdiff` (or `~/.cache/quantdiff`) elsewhere. Set `QUANTDIFF_CACHE_DIR` to move
|
|
419
|
+
it, or pass `--no-cache` to skip it.
|
|
420
|
+
|
|
421
|
+
The details, including the KL lower bound derivation and the limitations, are in
|
|
422
|
+
[docs/methodology.md](docs/methodology.md).
|
|
423
|
+
|
|
424
|
+
## Security model
|
|
425
|
+
|
|
426
|
+
quantdiff talks to model servers and executes nothing from them by default. In short:
|
|
427
|
+
|
|
428
|
+
- **Zero runtime dependencies.** After the March 2026 LiteLLM PyPI compromise, we decided the
|
|
429
|
+
smallest useful supply chain is none. Everything is standard library.
|
|
430
|
+
- **No code execution** unless you pass `--allow-code-exec`. With it, generated code for the
|
|
431
|
+
`code` suite runs in a separate subprocess with a scrubbed environment, a timeout, and POSIX
|
|
432
|
+
resource limits. That is isolation, not a security sandbox: only enable it for models you would
|
|
433
|
+
run code from anyway.
|
|
434
|
+
- **Network.** quantdiff connects only to the servers you name, plus `huggingface.co` for the chat
|
|
435
|
+
template check. `--offline` disables the Hugging Face request. The HTTP client ignores
|
|
436
|
+
`HTTP(S)_PROXY`, refuses redirects, non-http(s) schemes, and credentials embedded in URLs, and
|
|
437
|
+
caps response size. Server text is sanitized before it reaches your terminal.
|
|
438
|
+
- **PNG cards** are rendered by a Chromium-based browser already on your machine (Chrome, Edge,
|
|
439
|
+
Brave, or Chromium; set `QUANTDIFF_BROWSER` to pick one), headless, with a throwaway profile.
|
|
440
|
+
The card has no scripts or external resources, so the browser makes no network requests.
|
|
441
|
+
- **Secrets.** API keys are read from environment variables named in the spec, never passed as
|
|
442
|
+
arguments and never written to reports.
|
|
443
|
+
- **Output.** The HTML card escapes all model text and contains no JavaScript. The reference cache
|
|
444
|
+
is plain JSON, never pickle.
|
|
445
|
+
|
|
446
|
+
See [SECURITY.md](SECURITY.md) for the threat model and how to report a vulnerability.
|
|
447
|
+
|
|
448
|
+
## FAQ
|
|
449
|
+
|
|
450
|
+
**Why not just run perplexity on wikitext?**
|
|
451
|
+
Perplexity on wikitext measures how well a model predicts Wikipedia, averaged over every token. It
|
|
452
|
+
is a fine sanity check, but it does not tell you whether tool calls still parse, whether JSON still
|
|
453
|
+
validates, or whether the model you are serving has the right template. Two quants can have nearly
|
|
454
|
+
identical perplexity and behave differently on structured output. quantdiff measures distance from
|
|
455
|
+
the reference on your text and pass rates on tasks that break first.
|
|
456
|
+
|
|
457
|
+
**Why is KLD a lower bound?**
|
|
458
|
+
Servers return at most 20 logprobs per position, so the full distributions are not available.
|
|
459
|
+
quantdiff computes KL on a coarser partition: the tokens present in both top-k lists, the
|
|
460
|
+
reference's tokens missing from the candidate's list, and everything else. Merging outcomes can
|
|
461
|
+
only reduce KL divergence (the data processing inequality). The one mass the server does not
|
|
462
|
+
report is bounded by the candidate's smallest listed probability and set to the value that
|
|
463
|
+
minimizes the result. So the reported number is never larger than the true full-vocabulary KL,
|
|
464
|
+
yet a candidate that drops the reference's likely tokens from its top-k is still penalized. For
|
|
465
|
+
full-vocabulary KLD on llama.cpp, use `llama-perplexity --kl-divergence`. The methodology doc
|
|
466
|
+
shows how to cross-check.
|
|
467
|
+
|
|
468
|
+
**Why are Ollama numbers footnoted?**
|
|
469
|
+
Teacher forcing needs the candidate to see exactly the reference's tokens. llama-server accepts
|
|
470
|
+
token ids, so quantdiff feeds them directly. Ollama and OpenAI-compatible servers only accept text,
|
|
471
|
+
so quantdiff concatenates the reference's tokens as text and the server re-tokenizes it. Usually
|
|
472
|
+
that produces the same tokens; sometimes it merges or splits differently, which shifts positions.
|
|
473
|
+
The card footnotes every candidate scored this way. Compare those numbers with each other, not
|
|
474
|
+
with exact ones.
|
|
475
|
+
|
|
476
|
+
**Does it work with MLX?**
|
|
477
|
+
Yes, through an OpenAI-compatible server such as `mlx_lm.server`. Use
|
|
478
|
+
`openai:http://127.0.0.1:8080/v1#<model>`. Tier 1 needs the server to return logprobs.
|
|
479
|
+
|
|
480
|
+
**Can I compare different models, not quants of one model?**
|
|
481
|
+
Tier 2 will run, but quantdiff is built for variants of one model against its own reference. The
|
|
482
|
+
tokenizer check will withhold logit metrics when tokenizers differ.
|
|
483
|
+
|
|
484
|
+
**What should the reference be?**
|
|
485
|
+
BF16 or F16 if you can serve it. If memory is tight, Q8_0 is a close proxy. Note that this makes
|
|
486
|
+
every candidate look slightly better than it would against BF16; the methodology doc explains.
|
|
487
|
+
|
|
488
|
+
**Does quantdiff download models or start servers?**
|
|
489
|
+
Not in v0.1. You start Ollama, llama-server, LM Studio, or vLLM yourself. `quantdiff discover`
|
|
490
|
+
finds what Ollama already has.
|
|
491
|
+
|
|
492
|
+
**My PNG is missing.**
|
|
493
|
+
`card.png` needs a Chromium-based browser. Without one the run still writes `card.md` and
|
|
494
|
+
`card.html` and says why the PNG was skipped. Install Chrome, Edge, Brave or Chromium, or point
|
|
495
|
+
`QUANTDIFF_BROWSER` at one.
|
|
496
|
+
|
|
497
|
+
## Roadmap
|
|
498
|
+
|
|
499
|
+
Not in v0.1, planned later:
|
|
500
|
+
|
|
501
|
+
- Download and launch servers for you.
|
|
502
|
+
- AWQ and GPTQ specific handling.
|
|
503
|
+
- KV cache quantization comparisons.
|
|
504
|
+
- Vision models.
|
|
505
|
+
- A hosted leaderboard of submitted scorecards.
|
|
506
|
+
|
|
507
|
+
## Contributing
|
|
508
|
+
|
|
509
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md). Bug reports with a `report.json` attached are the most
|
|
510
|
+
useful kind.
|
|
511
|
+
|
|
512
|
+
## License
|
|
513
|
+
|
|
514
|
+
Apache License 2.0. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
quantdiff/__init__.py,sha256=KYDRY328jx-ecB5h2Vj__jlvnZAufYCaV2edBDitSs4,1285
|
|
2
|
+
quantdiff/__main__.py,sha256=uys9kEiw4ObDgSxHUTjzaw8nqdakAUt6GJAMHO3m4Ck,93
|
|
3
|
+
quantdiff/_http.py,sha256=vB_eMWk-FiyljrLcmIvbXuQLT6gVgDGlWkQkMHpdMOs,5417
|
|
4
|
+
quantdiff/_text.py,sha256=MEt3FvtkxmBSUgvEEkHp8RwQGM1MrbUxpe-cY9u_JVY,521
|
|
5
|
+
quantdiff/_version.py,sha256=9nixZb9lFjFQvbG9Q11wvWamgwFTwVrV0QxTzWAolNc,25
|
|
6
|
+
quantdiff/api.py,sha256=CiFKUMbbF8YkZnLeyoo5ZznVcXYj_6Si8_A3_3deB_Q,12137
|
|
7
|
+
quantdiff/cache.py,sha256=I__n3sU5yBrs_YlhNQSdKT21o4IJ7FqiOx1mkAg8O2c,8305
|
|
8
|
+
quantdiff/card.py,sha256=AeSXrCbargQYfJd4jrvyX1xLmnnACF71LTSvLWBnIdE,60126
|
|
9
|
+
quantdiff/cli.py,sha256=cmSfpelSfrru94zHSAXZLwvov-zVRKKT2PiH83jymfM,15564
|
|
10
|
+
quantdiff/discover.py,sha256=Sk6Co3M1C5FRZmp_IXEfisp7WEzBoTb7a8YJLcGYdD4,18769
|
|
11
|
+
quantdiff/errors.py,sha256=O2Nd3gSx2zshi_NZ6bBOhe5i8gjiHQwVdsWIvBn90I0,1490
|
|
12
|
+
quantdiff/png.py,sha256=0dm-SJ0WRwBZq89YWLLszkaTSg_8ZDESfJcRPCBnvbU,13002
|
|
13
|
+
quantdiff/preflight.py,sha256=IYs9gz-fWoyDYSfEw2lZDnKhu3ZGs3khWUHoweF3StI,13364
|
|
14
|
+
quantdiff/progress.py,sha256=8Cp5uQA7MvMCEgLhD5uo_3nSEeRf8jOJA187cwOlFXM,10858
|
|
15
|
+
quantdiff/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
16
|
+
quantdiff/report.py,sha256=-S7o-CxH49RiCa0S3aXAiGoMiHICndf-hfHMrpXySpI,29632
|
|
17
|
+
quantdiff/runner.py,sha256=JlAjf7vXv3vnCO4E7g6P_FsQp-tV11MOLRyyKnf2_Jk,17160
|
|
18
|
+
quantdiff/spec.py,sha256=xv-6JP_JklzDgqeLQMcGDYluINER7BCRh1lWbuN3wqk,5518
|
|
19
|
+
quantdiff/stats.py,sha256=1f-KYSCbMmTjVzOxCSSdjJWx6CxP91XANU3Kk_WLmno,8857
|
|
20
|
+
quantdiff/types.py,sha256=9UpZ_LJQ9VxYDED19Oa6hhCTLTRc3FjraCN7eagkEjA,10003
|
|
21
|
+
quantdiff/verdict.py,sha256=P0ML4gDu5nrDyYjONbPk18OQQh731j_N5OxjWG_pR6E,60538
|
|
22
|
+
quantdiff/backends/__init__.py,sha256=-c0FZa4vqT3dxRvGvstkr6cwnqFpufUOQzTMHgxvD3o,983
|
|
23
|
+
quantdiff/backends/_common.py,sha256=oGBy6O1o4f9hp8MQjYQLDiNczvZFjqjiSRml71QSTPY,12491
|
|
24
|
+
quantdiff/backends/base.py,sha256=f63RjUN6ifaEiB_OFmFWjrPjyz3lhG2slNKuWh5-_Vk,3085
|
|
25
|
+
quantdiff/backends/llamacpp.py,sha256=jtvbeK2PsrwtB4JQeuNFNZYVjjzgqNNZB0FV9EkRXb0,18397
|
|
26
|
+
quantdiff/backends/ollama.py,sha256=VUOYdo-WUEACfPO1mMcs6Y3EvtNUMcqOcxoghI5e6NM,14354
|
|
27
|
+
quantdiff/backends/openai_compat.py,sha256=7lW3jJwA3ubSel3ksm097ddidhPLbeespUMrR4YLMYU,13401
|
|
28
|
+
quantdiff/metrics/__init__.py,sha256=1s-5Q5wQILw7Fagg96yWCzwHywxc3wSFufnMYKoD73M,1041
|
|
29
|
+
quantdiff/metrics/codeexec.py,sha256=8diIbkP7ALLgSz2THIzIhkC1TYVGLZJyvw4TrR1ATuM,16742
|
|
30
|
+
quantdiff/metrics/jsonschema.py,sha256=qmx7nKbfYoBuTKoHjgcddPq1Vgo2ROaZBkPsQEUBH7w,23433
|
|
31
|
+
quantdiff/metrics/logit.py,sha256=VzpMpJ-8rg-aNdrI1FzJ5hZyVXi2BTQTgi-l5Me36Og,8596
|
|
32
|
+
quantdiff/metrics/tasks.py,sha256=SqH8533623U8U5k2md1PgS6eL78DitoRBaGeIYtdeR8,4882
|
|
33
|
+
quantdiff/metrics/textsim.py,sha256=BEk-l91CDIOsP3sXnhKMr0q4HyG-kaw-FcQbA06CfRQ,2486
|
|
34
|
+
quantdiff/metrics/toolcheck.py,sha256=TdP3yeF8ycg_w796L9WhTucCNkef9AbSNxBnbZuWcd4,4020
|
|
35
|
+
quantdiff/suites/__init__.py,sha256=eoE_PpiIM6CP3mLSWJORp69JAmaUD4nznfkI7dbiTuI,17581
|
|
36
|
+
quantdiff/suites/data/chat.jsonl,sha256=_6Y29umJQb-0mDyTCk8m6X_uf0UQlsE_UbttcBx6lwY,6326
|
|
37
|
+
quantdiff/suites/data/code.jsonl,sha256=EhDp6vk_N1KobeddEA6axSm1v6lokpFiQ88Wss5fcP4,28580
|
|
38
|
+
quantdiff/suites/data/json.jsonl,sha256=8SZq4oORnuvFiip4GUhrMFq1r56SjxUqIuma6_Is54E,36155
|
|
39
|
+
quantdiff/suites/data/scoring.jsonl,sha256=CbZUGMVoDAbSYsJH6IV71KXLYj-4JvRGz1OyzeqcGNE,18015
|
|
40
|
+
quantdiff/suites/data/tools.jsonl,sha256=nuQToSsRA6vvsv9PM8mBL9r9MdofEbhpvarSWO_E1lw,43098
|
|
41
|
+
quantdiff-0.1.0rc1.dist-info/METADATA,sha256=LEEtRAvpGtpgvVI6QEJ2PlQMscO1o5X07bAv7TZ_tZQ,27106
|
|
42
|
+
quantdiff-0.1.0rc1.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
43
|
+
quantdiff-0.1.0rc1.dist-info/entry_points.txt,sha256=9AGK7-YxutKIBfDeo5Cu7_WEAdaHLq4yiUzQY9Sx-eM,49
|
|
44
|
+
quantdiff-0.1.0rc1.dist-info/licenses/LICENSE,sha256=z8d0m5b2O9McPEK1xHG_dWgUBT6EfBDz6wA0F7xSPTA,11358
|
|
45
|
+
quantdiff-0.1.0rc1.dist-info/RECORD,,
|