quantdiff 0.1.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quantdiff/__init__.py +53 -0
- quantdiff/__main__.py +5 -0
- quantdiff/_http.py +151 -0
- quantdiff/_text.py +13 -0
- quantdiff/_version.py +1 -0
- quantdiff/api.py +340 -0
- quantdiff/backends/__init__.py +28 -0
- quantdiff/backends/_common.py +342 -0
- quantdiff/backends/base.py +91 -0
- quantdiff/backends/llamacpp.py +428 -0
- quantdiff/backends/ollama.py +359 -0
- quantdiff/backends/openai_compat.py +338 -0
- quantdiff/cache.py +240 -0
- quantdiff/card.py +1664 -0
- quantdiff/cli.py +377 -0
- quantdiff/discover.py +488 -0
- quantdiff/errors.py +45 -0
- quantdiff/metrics/__init__.py +36 -0
- quantdiff/metrics/codeexec.py +428 -0
- quantdiff/metrics/jsonschema.py +610 -0
- quantdiff/metrics/logit.py +214 -0
- quantdiff/metrics/tasks.py +114 -0
- quantdiff/metrics/textsim.py +66 -0
- quantdiff/metrics/toolcheck.py +99 -0
- quantdiff/png.py +360 -0
- quantdiff/preflight.py +365 -0
- quantdiff/progress.py +283 -0
- quantdiff/py.typed +0 -0
- quantdiff/report.py +780 -0
- quantdiff/runner.py +492 -0
- quantdiff/spec.py +154 -0
- quantdiff/stats.py +226 -0
- quantdiff/suites/__init__.py +462 -0
- quantdiff/suites/data/chat.jsonl +22 -0
- quantdiff/suites/data/code.jsonl +32 -0
- quantdiff/suites/data/json.jsonl +34 -0
- quantdiff/suites/data/scoring.jsonl +41 -0
- quantdiff/suites/data/tools.jsonl +32 -0
- quantdiff/types.py +322 -0
- quantdiff/verdict.py +1513 -0
- quantdiff-0.1.0rc1.dist-info/METADATA +514 -0
- quantdiff-0.1.0rc1.dist-info/RECORD +45 -0
- quantdiff-0.1.0rc1.dist-info/WHEEL +4 -0
- quantdiff-0.1.0rc1.dist-info/entry_points.txt +2 -0
- quantdiff-0.1.0rc1.dist-info/licenses/LICENSE +202 -0
quantdiff/discover.py
ADDED
|
@@ -0,0 +1,488 @@
|
|
|
1
|
+
"""Find the models installed in Ollama and suggest a ready-to-run comparison.
|
|
2
|
+
|
|
3
|
+
Downloads are grouped into families of the same model in two steps.
|
|
4
|
+
|
|
5
|
+
1. Every tag gets a name key: the model name and tag with the quantization suffix removed,
|
|
6
|
+
split into lowercase word and number tokens so "-", "_", "." and case do not matter.
|
|
7
|
+
For `hf.co/<uploader>/<repo>:<quant>` the uploader is dropped, and so are a trailing
|
|
8
|
+
`-GGUF`, an imatrix `-i1` and a leading `<org>_` prefix (bartowski names repos
|
|
9
|
+
`Qwen_Qwen3-8B-GGUF`). `qwen3:8b`, `hf.co/unsloth/Qwen3-8B-GGUF:UD-Q4_K_XL` and
|
|
10
|
+
`hf.co/bartowski/Qwen_Qwen3-8B-GGUF:Q4_K_M` all get the tokens qwen, 3, 8b.
|
|
11
|
+
2. Ollama's /api/show reports the GGUF metadata of each file. When it has an architecture,
|
|
12
|
+
basename and size label, and the name key contains a number and only tokens that the
|
|
13
|
+
metadata (basename, size label, fine-tune) also has, the download is keyed by that
|
|
14
|
+
metadata. So `qwen2.5:3b`, whose metadata says Qwen2.5 3B Instruct, joins
|
|
15
|
+
`hf.co/x/Qwen2.5-3B-Instruct-GGUF:Q8_0`, while a custom `mychat:latest` built on the
|
|
16
|
+
same weights stays on its own. Without usable metadata a download is keyed by its name
|
|
17
|
+
key alone, and Ollama library tags never share a name key with Hugging Face downloads:
|
|
18
|
+
library tags often leave out "instruct", so `qwen2.5:3b` and a base `Qwen2.5-3B` repo
|
|
19
|
+
have the same name.
|
|
20
|
+
|
|
21
|
+
Within a family the most precise download is the suggested reference. A family with
|
|
22
|
+
downloads from more than one source (the Ollama library and Hugging Face uploaders) gets a
|
|
23
|
+
command with one labelled candidate per source, so the uploads can be compared directly.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import difflib
|
|
29
|
+
import functools
|
|
30
|
+
import logging
|
|
31
|
+
import os
|
|
32
|
+
import re
|
|
33
|
+
import shlex
|
|
34
|
+
from collections import Counter
|
|
35
|
+
from collections.abc import Iterable, Sequence
|
|
36
|
+
from dataclasses import dataclass
|
|
37
|
+
from typing import Final, NamedTuple
|
|
38
|
+
|
|
39
|
+
from quantdiff._http import get_json, post_json, validate_base_url
|
|
40
|
+
from quantdiff._text import printable
|
|
41
|
+
from quantdiff.backends._common import (
|
|
42
|
+
expect_dict,
|
|
43
|
+
expect_list,
|
|
44
|
+
expect_str,
|
|
45
|
+
explained_failures,
|
|
46
|
+
optional_int,
|
|
47
|
+
optional_str,
|
|
48
|
+
)
|
|
49
|
+
from quantdiff.backends.ollama import explain_ollama_failure
|
|
50
|
+
from quantdiff.errors import BackendError
|
|
51
|
+
from quantdiff.spec import ollama_base_url
|
|
52
|
+
from quantdiff.types import JSONValue
|
|
53
|
+
|
|
54
|
+
logger = logging.getLogger(__name__)
|
|
55
|
+
|
|
56
|
+
_QUANT_SUFFIX: Final = re.compile(
|
|
57
|
+
r"(?:^|[-_.])(?P<quant>(?:ud-)?i?q\d+(?:_[0-9a-z]+)*|bf16|fp16|f16|fp32|f32)$",
|
|
58
|
+
re.IGNORECASE,
|
|
59
|
+
)
|
|
60
|
+
_BITS: Final = re.compile(r"^(?:ud-)?(?:i?q|mxfp|b?f|fp)(?P<bits>\d+)", re.IGNORECASE)
|
|
61
|
+
_TOKEN: Final = re.compile(r"[a-z]+|\d+(?:\.\d+)*(?:[a-z](?![a-z]))?")
|
|
62
|
+
_HF_HOSTS: Final = ("hf.co/", "huggingface.co/")
|
|
63
|
+
_REPO_SUFFIXES: Final = (
|
|
64
|
+
re.compile(r"[-_.]gguf$", re.IGNORECASE),
|
|
65
|
+
re.compile(r"-i1$", re.IGNORECASE),
|
|
66
|
+
)
|
|
67
|
+
_ORG_PREFIX: Final = re.compile(r"^[A-Za-z0-9][A-Za-z0-9.-]*_(?=[A-Za-z])")
|
|
68
|
+
_IGNORED_TOKENS: Final = frozenset({"latest"})
|
|
69
|
+
_REFERENCE_MIN_BITS: Final = 8
|
|
70
|
+
"""Below this, a download is a poor stand-in for the original weights."""
|
|
71
|
+
_DECIMAL_GB: Final = 1000**3
|
|
72
|
+
_DECIMAL_MB: Final = 1000**2
|
|
73
|
+
_SUGGESTION_CUTOFF: Final = 0.6
|
|
74
|
+
UNKNOWN_QUANTIZATION: Final = "unknown"
|
|
75
|
+
LIBRARY_SOURCE: Final = "ollama"
|
|
76
|
+
"""Source name for tags from the Ollama library, as opposed to a Hugging Face uploader."""
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class Member(NamedTuple):
|
|
80
|
+
"""One download in a family."""
|
|
81
|
+
|
|
82
|
+
tag: str
|
|
83
|
+
quantization: str
|
|
84
|
+
"""Such as Q4_K_M or UD-Q4_K_XL."""
|
|
85
|
+
size: int
|
|
86
|
+
"""Bytes."""
|
|
87
|
+
|
|
88
|
+
@property
|
|
89
|
+
def source(self) -> str:
|
|
90
|
+
return source_of(self.tag)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@dataclass(frozen=True, slots=True)
|
|
94
|
+
class ModelFamily:
|
|
95
|
+
"""Downloads of one model that differ in quantization or uploader.
|
|
96
|
+
|
|
97
|
+
`members` is sorted from most to least precise, larger files first on ties.
|
|
98
|
+
"""
|
|
99
|
+
|
|
100
|
+
name: str
|
|
101
|
+
members: tuple[Member, ...]
|
|
102
|
+
|
|
103
|
+
@property
|
|
104
|
+
def suggested_reference(self) -> str:
|
|
105
|
+
"""The most precise download, the best stand-in for the original weights."""
|
|
106
|
+
return self.members[0].tag
|
|
107
|
+
|
|
108
|
+
@property
|
|
109
|
+
def suggested_candidates(self) -> tuple[str, ...]:
|
|
110
|
+
return tuple(member.tag for member in self.members[1:])
|
|
111
|
+
|
|
112
|
+
@property
|
|
113
|
+
def sources(self) -> tuple[str, ...]:
|
|
114
|
+
"""Where the downloads came from, in member order, each once."""
|
|
115
|
+
return tuple(dict.fromkeys(member.source for member in self.members))
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
@dataclass(frozen=True, slots=True)
|
|
119
|
+
class _Download:
|
|
120
|
+
tag: str
|
|
121
|
+
display: str
|
|
122
|
+
"""The family name this tag alone suggests, such as `qwen2.5:0.5b-instruct`."""
|
|
123
|
+
name_tokens: tuple[str, ...]
|
|
124
|
+
quantization: str
|
|
125
|
+
size: int
|
|
126
|
+
digest: str
|
|
127
|
+
explicit_suffix: bool
|
|
128
|
+
"""True when the tag itself names the quantization, as in `-q4_K_M`."""
|
|
129
|
+
|
|
130
|
+
@property
|
|
131
|
+
def member(self) -> Member:
|
|
132
|
+
return Member(self.tag, self.quantization, self.size)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
@dataclass(frozen=True, slots=True)
|
|
136
|
+
class _SplitTag:
|
|
137
|
+
display: str
|
|
138
|
+
stem: str
|
|
139
|
+
"""The tag up to and including the separator before the quantization suffix."""
|
|
140
|
+
suffix: str | None
|
|
141
|
+
name_tokens: tuple[str, ...]
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def discover_ollama(base_url: str | None = None, *, timeout: float = 5.0) -> list[ModelFamily]:
|
|
145
|
+
"""List installed Ollama models grouped into families, sorted by family name.
|
|
146
|
+
|
|
147
|
+
`base_url` defaults to OLLAMA_HOST, resolved the way the Ollama CLI does. Sends one
|
|
148
|
+
/api/show request per distinct file; a failed one only means that file is grouped by
|
|
149
|
+
its name.
|
|
150
|
+
"""
|
|
151
|
+
url = _resolve(base_url)
|
|
152
|
+
downloads = [_read_download(entry) for entry in _listing(url, timeout)]
|
|
153
|
+
identities: dict[str, frozenset[str] | None] = {}
|
|
154
|
+
for download in downloads:
|
|
155
|
+
if download.digest not in identities:
|
|
156
|
+
identities[download.digest] = _show_identity(url, download.tag, timeout)
|
|
157
|
+
return _group(downloads, identities)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def installed_tags(base_url: str | None = None, *, timeout: float = 5.0) -> list[str]:
|
|
161
|
+
"""Every tag Ollama has installed, in its listing order."""
|
|
162
|
+
url = _resolve(base_url)
|
|
163
|
+
return [_tag_name(entry) for entry in _listing(url, timeout)]
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def closest_tags(wanted: str, installed: Iterable[str], *, limit: int = 3) -> list[str]:
|
|
167
|
+
"""Up to `limit` installed tags that `wanted` was probably meant to be.
|
|
168
|
+
|
|
169
|
+
Tags that extend `wanted` or that `wanted` extends (`qwen2.5` and `qwen2.5:3b-instrct`
|
|
170
|
+
both suggest `qwen2.5:3b`) come first, then near spellings.
|
|
171
|
+
"""
|
|
172
|
+
tags = list(dict.fromkeys(installed))
|
|
173
|
+
folded = wanted.casefold()
|
|
174
|
+
|
|
175
|
+
def similarity(tag: str) -> float:
|
|
176
|
+
return difflib.SequenceMatcher(None, folded, tag.casefold()).ratio()
|
|
177
|
+
|
|
178
|
+
def related(tag: str) -> bool:
|
|
179
|
+
other = tag.casefold()
|
|
180
|
+
return other.startswith(folded) or folded.startswith(other)
|
|
181
|
+
|
|
182
|
+
extending = sorted(
|
|
183
|
+
(tag for tag in tags if related(tag)), key=lambda tag: (-similarity(tag), tag)
|
|
184
|
+
)
|
|
185
|
+
by_folded = {tag.casefold(): tag for tag in tags}
|
|
186
|
+
similar = [
|
|
187
|
+
by_folded[match]
|
|
188
|
+
for match in difflib.get_close_matches(
|
|
189
|
+
folded, by_folded, n=limit, cutoff=_SUGGESTION_CUTOFF
|
|
190
|
+
)
|
|
191
|
+
]
|
|
192
|
+
return list(dict.fromkeys(tag for tag in [*extending, *similar] if tag != wanted))[:limit]
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def source_of(tag: str) -> str:
|
|
196
|
+
"""The Hugging Face uploader or Ollama namespace a tag comes from, else `ollama`."""
|
|
197
|
+
name = _name_part(tag)
|
|
198
|
+
hf = _hf_repo(name)
|
|
199
|
+
if hf is not None:
|
|
200
|
+
return hf[0]
|
|
201
|
+
namespace, separator, _ = name.rpartition("/")
|
|
202
|
+
owner = namespace.rpartition("/")[2]
|
|
203
|
+
if not separator or owner == "library":
|
|
204
|
+
return LIBRARY_SOURCE
|
|
205
|
+
return owner
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def suggest_command(family: ModelFamily) -> str | None:
|
|
209
|
+
"""A `quantdiff run` command comparing every download in the family, if it has two.
|
|
210
|
+
|
|
211
|
+
Candidates are labelled by source when the family has more than one.
|
|
212
|
+
"""
|
|
213
|
+
if len(family.members) < 2:
|
|
214
|
+
return None
|
|
215
|
+
labels = _candidate_labels(family)
|
|
216
|
+
parts = ["quantdiff run", f"--ref ollama:{shlex.quote(family.suggested_reference)}"]
|
|
217
|
+
for member in family.members[1:]:
|
|
218
|
+
label = labels.get(member.tag)
|
|
219
|
+
spec = f"ollama:{member.tag}" if label is None else f"{label}=ollama:{member.tag}"
|
|
220
|
+
parts.append(f"--cand {shlex.quote(spec)}")
|
|
221
|
+
return " ".join(parts)
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def format_discovery(families: Iterable[ModelFamily]) -> str:
|
|
225
|
+
"""Describe the families for a terminal, with a next step for each."""
|
|
226
|
+
ordered = sorted(families, key=lambda family: (len(family.members) < 2, family.name))
|
|
227
|
+
if not ordered:
|
|
228
|
+
return (
|
|
229
|
+
"No models found in Ollama. Pull two downloads of the same model to compare,"
|
|
230
|
+
" for example:\n"
|
|
231
|
+
" ollama pull qwen2.5:0.5b-instruct-q8_0\n"
|
|
232
|
+
" ollama pull qwen2.5:0.5b-instruct-q4_K_M\n"
|
|
233
|
+
"then run quantdiff discover again."
|
|
234
|
+
)
|
|
235
|
+
downloads = sum(len(family.members) for family in ordered)
|
|
236
|
+
header = f"Found {downloads} downloads of {len(ordered)} models in Ollama."
|
|
237
|
+
return "\n\n".join([header, *(_format_family(family) for family in ordered)])
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def precision_bits(quantization: str) -> int:
|
|
241
|
+
"""Bits per weight named by a quantization such as Q4_K_M or BF16; 0 when unknown."""
|
|
242
|
+
match = _BITS.match(quantization)
|
|
243
|
+
return 0 if match is None else int(match["bits"])
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def format_size(size: int) -> str:
|
|
247
|
+
"""Decimal units, matching `ollama list`."""
|
|
248
|
+
if size >= _DECIMAL_GB:
|
|
249
|
+
return f"{size / _DECIMAL_GB:.1f} GB"
|
|
250
|
+
return f"{round(size / _DECIMAL_MB)} MB"
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
# Reading Ollama ---------------------------------------------------------------------------
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def _resolve(base_url: str | None) -> str:
|
|
257
|
+
return validate_base_url(base_url or ollama_base_url(os.environ.get("OLLAMA_HOST")))
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _listing(url: str, timeout: float) -> list[JSONValue]:
|
|
261
|
+
with explained_failures(functools.partial(explain_ollama_failure, base_url=url)):
|
|
262
|
+
body = get_json(f"{url}/api/tags", timeout=timeout)
|
|
263
|
+
listing = expect_dict(body, "Ollama /api/tags response")
|
|
264
|
+
return expect_list(listing.get("models") or [], "Ollama /api/tags models")
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _tag_name(entry: JSONValue) -> str:
|
|
268
|
+
model = expect_dict(entry, "Ollama /api/tags entry")
|
|
269
|
+
return printable(expect_str(model.get("name") or model.get("model"), "Ollama model name"))
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def _read_download(entry: JSONValue) -> _Download:
|
|
273
|
+
model = expect_dict(entry, "Ollama /api/tags entry")
|
|
274
|
+
tag = _tag_name(entry)
|
|
275
|
+
details = model.get("details")
|
|
276
|
+
listed = details.get("quantization_level") if isinstance(details, dict) else None
|
|
277
|
+
split = _split_tag(tag)
|
|
278
|
+
# The tag is more specific than the file type Ollama lists: Unsloth's UD-Q4_K_XL is
|
|
279
|
+
# listed as Q4_K_M.
|
|
280
|
+
named = None if split.suffix is None else split.suffix.upper()
|
|
281
|
+
return _Download(
|
|
282
|
+
tag=tag,
|
|
283
|
+
display=split.display,
|
|
284
|
+
name_tokens=split.name_tokens,
|
|
285
|
+
quantization=printable(named or optional_str(listed) or UNKNOWN_QUANTIZATION),
|
|
286
|
+
size=optional_int(model.get("size")) or 0,
|
|
287
|
+
digest=optional_str(model.get("digest")) or tag,
|
|
288
|
+
explicit_suffix=split.suffix is not None,
|
|
289
|
+
)
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def _show_identity(url: str, tag: str, timeout: float) -> frozenset[str] | None:
|
|
293
|
+
"""Tokens naming the model in a file's GGUF metadata, or None if Ollama has too little."""
|
|
294
|
+
try:
|
|
295
|
+
body = expect_dict(
|
|
296
|
+
post_json(f"{url}/api/show", {"model": tag}, timeout=timeout), "Ollama /api/show"
|
|
297
|
+
)
|
|
298
|
+
except BackendError as exc:
|
|
299
|
+
logger.debug("no metadata for %s: %s", tag, exc)
|
|
300
|
+
return None
|
|
301
|
+
info = body.get("model_info")
|
|
302
|
+
if not isinstance(info, dict):
|
|
303
|
+
return None
|
|
304
|
+
architecture = optional_str(info.get("general.architecture"))
|
|
305
|
+
basename = optional_str(info.get("general.basename"))
|
|
306
|
+
size_label = optional_str(info.get("general.size_label"))
|
|
307
|
+
if not (architecture and basename and size_label):
|
|
308
|
+
return None
|
|
309
|
+
finetune = optional_str(info.get("general.finetune")) or ""
|
|
310
|
+
return frozenset(
|
|
311
|
+
{f"arch={architecture.casefold()}", *_tokens(f"{basename} {size_label} {finetune}")}
|
|
312
|
+
)
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
# Grouping ---------------------------------------------------------------------------------
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _split_tag(tag: str) -> _SplitTag:
|
|
319
|
+
name, separator, version = tag.rpartition(":")
|
|
320
|
+
if not separator or "/" in version:
|
|
321
|
+
name, version = tag, "latest"
|
|
322
|
+
match = _QUANT_SUFFIX.search(version)
|
|
323
|
+
base = version if match is None else version[: match.start()]
|
|
324
|
+
suffix = None if match is None else match["quant"]
|
|
325
|
+
stem = "" if suffix is None else tag[: len(tag) - len(suffix)]
|
|
326
|
+
return _SplitTag(
|
|
327
|
+
display=f"{name}:{base}" if base else name,
|
|
328
|
+
stem=stem,
|
|
329
|
+
suffix=suffix,
|
|
330
|
+
name_tokens=_name_tokens(name, base),
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def _name_part(tag: str) -> str:
|
|
335
|
+
name, separator, version = tag.rpartition(":")
|
|
336
|
+
return tag if not separator or "/" in version else name
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def _hf_repo(name: str) -> tuple[str, str] | None:
|
|
340
|
+
"""(uploader, repo) for `hf.co/<uploader>/<repo>`, else None."""
|
|
341
|
+
folded = name.casefold()
|
|
342
|
+
for host in _HF_HOSTS:
|
|
343
|
+
if folded.startswith(host):
|
|
344
|
+
uploader, separator, repo = name[len(host) :].partition("/")
|
|
345
|
+
if separator and uploader and repo:
|
|
346
|
+
return uploader, repo
|
|
347
|
+
return None
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def _model_name(name: str) -> str:
|
|
351
|
+
"""The model's own name: the Hugging Face repo without packaging marks, or the last
|
|
352
|
+
path segment of an Ollama name."""
|
|
353
|
+
hf = _hf_repo(name)
|
|
354
|
+
if hf is None:
|
|
355
|
+
return name.rpartition("/")[2]
|
|
356
|
+
repo = hf[1]
|
|
357
|
+
for suffix in _REPO_SUFFIXES:
|
|
358
|
+
repo = suffix.sub("", repo)
|
|
359
|
+
return _ORG_PREFIX.sub("", repo)
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def _name_tokens(name: str, base: str) -> tuple[str, ...]:
|
|
363
|
+
tokens = _tokens(f"{_model_name(name)} {base}")
|
|
364
|
+
return tuple(token for token in tokens if token not in _IGNORED_TOKENS)
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def _tokens(text: str) -> list[str]:
|
|
368
|
+
return _TOKEN.findall(text.casefold())
|
|
369
|
+
|
|
370
|
+
|
|
371
|
+
def _family_key(download: _Download, identity: frozenset[str] | None) -> tuple[str, ...]:
|
|
372
|
+
tokens = download.name_tokens
|
|
373
|
+
numbered = any(char.isdigit() for token in tokens for char in token)
|
|
374
|
+
if identity is not None and numbered and identity.issuperset(tokens):
|
|
375
|
+
return ("metadata", *sorted(identity))
|
|
376
|
+
hub = "hf" if _hf_repo(_name_part(download.tag)) is not None else "ollama"
|
|
377
|
+
return ("name", hub, *tokens)
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def _group(
|
|
381
|
+
downloads: Sequence[_Download], identities: dict[str, frozenset[str] | None]
|
|
382
|
+
) -> list[ModelFamily]:
|
|
383
|
+
by_family: dict[tuple[str, ...], dict[str, _Download]] = {}
|
|
384
|
+
for download in downloads:
|
|
385
|
+
key = _family_key(download, identities.get(download.digest))
|
|
386
|
+
# Tags that share a digest are the same file; keep the most descriptive tag.
|
|
387
|
+
same_family = by_family.setdefault(key, {})
|
|
388
|
+
kept = same_family.get(download.digest)
|
|
389
|
+
if kept is None or _tag_preference(download) > _tag_preference(kept):
|
|
390
|
+
same_family[download.digest] = download
|
|
391
|
+
families = [
|
|
392
|
+
ModelFamily(name=_family_name(list(kept.values())), members=_by_precision(kept.values()))
|
|
393
|
+
for kept in by_family.values()
|
|
394
|
+
]
|
|
395
|
+
return sorted(families, key=lambda family: (family.name, family.members))
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def _tag_preference(download: _Download) -> tuple[bool, bool]:
|
|
399
|
+
return download.explicit_suffix, not download.tag.endswith(":latest")
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
def _family_name(downloads: list[_Download]) -> str:
|
|
403
|
+
sources = {source_of(item.tag) for item in downloads}
|
|
404
|
+
if len(sources) > 1:
|
|
405
|
+
library = [item for item in downloads if source_of(item.tag) == LIBRARY_SOURCE]
|
|
406
|
+
if not library:
|
|
407
|
+
return _most_common(_model_name(_name_part(item.tag)) for item in downloads)
|
|
408
|
+
downloads = library
|
|
409
|
+
return _most_common(item.display for item in downloads)
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def _most_common(names: Iterable[str]) -> str:
|
|
413
|
+
counts = Counter(names)
|
|
414
|
+
return min(counts, key=lambda name: (-counts[name], name))
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def _by_precision(downloads: Iterable[_Download]) -> tuple[Member, ...]:
|
|
418
|
+
ranked = sorted(
|
|
419
|
+
downloads,
|
|
420
|
+
key=lambda item: (precision_bits(item.quantization), item.size),
|
|
421
|
+
reverse=True,
|
|
422
|
+
)
|
|
423
|
+
return tuple(item.member for item in ranked)
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def _candidate_labels(family: ModelFamily) -> dict[str, str]:
|
|
427
|
+
"""Labels for candidates of a family with several sources: the source, plus the
|
|
428
|
+
quantization when a source has more than one candidate."""
|
|
429
|
+
if len(family.sources) < 2:
|
|
430
|
+
return {}
|
|
431
|
+
candidates = family.members[1:]
|
|
432
|
+
per_source = Counter(member.source for member in candidates)
|
|
433
|
+
return {
|
|
434
|
+
member.tag: member.source
|
|
435
|
+
if per_source[member.source] == 1
|
|
436
|
+
else f"{member.source}-{member.quantization}"
|
|
437
|
+
for member in candidates
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
# Output -----------------------------------------------------------------------------------
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def _format_family(family: ModelFamily) -> str:
|
|
445
|
+
quant_width = max(len(member.quantization) for member in family.members)
|
|
446
|
+
several_sources = len(family.sources) > 1
|
|
447
|
+
source_width = max(len(source) for source in family.sources)
|
|
448
|
+
lines = [
|
|
449
|
+
f"{family.name} (from {', '.join(family.sources)})" if several_sources else family.name
|
|
450
|
+
]
|
|
451
|
+
for index, member in enumerate(family.members):
|
|
452
|
+
marker = " (reference)" if index == 0 and len(family.members) > 1 else ""
|
|
453
|
+
source = f"{member.source:<{source_width}} " if several_sources else ""
|
|
454
|
+
lines.append(
|
|
455
|
+
f" {source}{member.quantization:<{quant_width}} {format_size(member.size):>7}"
|
|
456
|
+
f" {member.tag}{marker}"
|
|
457
|
+
)
|
|
458
|
+
command = suggest_command(family)
|
|
459
|
+
if command is None:
|
|
460
|
+
lines.extend(_single_download_hint(family.members[0]))
|
|
461
|
+
else:
|
|
462
|
+
lines.extend([" Compare them:", f" {command}"])
|
|
463
|
+
lines.extend(_weak_reference_note(family.members[0].quantization))
|
|
464
|
+
return "\n".join(lines)
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
def _weak_reference_note(quantization: str) -> list[str]:
|
|
468
|
+
if precision_bits(quantization) >= _REFERENCE_MIN_BITS:
|
|
469
|
+
return []
|
|
470
|
+
return [
|
|
471
|
+
f" The most precise download here is {quantization}; pull a q8_0 or fp16 download",
|
|
472
|
+
" of this model for a reference closer to the original weights.",
|
|
473
|
+
]
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
def _single_download_hint(member: Member) -> list[str]:
|
|
477
|
+
split = _split_tag(member.tag)
|
|
478
|
+
if split.suffix is None:
|
|
479
|
+
return [
|
|
480
|
+
" Only one download. Pull a q8_0 or fp16 download of this model from its tag",
|
|
481
|
+
" list to use as a reference, then run quantdiff discover again.",
|
|
482
|
+
]
|
|
483
|
+
upper = split.suffix[0].isupper()
|
|
484
|
+
if precision_bits(member.quantization) >= _REFERENCE_MIN_BITS:
|
|
485
|
+
pull, purpose = ("Q4_K_M" if upper else "q4_K_M"), "a smaller download to compare"
|
|
486
|
+
else:
|
|
487
|
+
pull, purpose = ("Q8_0" if upper else "q8_0"), "a reference to compare against"
|
|
488
|
+
return [f" Only one download. To get {purpose}:", f" ollama pull {split.stem}{pull}"]
|
quantdiff/errors.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Exception hierarchy. Every error quantdiff raises on purpose derives from QuantdiffError."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class QuantdiffError(Exception):
|
|
7
|
+
"""Base class for all quantdiff errors."""
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class SpecError(QuantdiffError, ValueError):
|
|
11
|
+
"""A candidate spec string or option could not be parsed."""
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class BackendError(QuantdiffError):
|
|
15
|
+
"""A model server returned an error, an unexpected payload, or could not be reached."""
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class RequestError(BackendError):
|
|
19
|
+
"""An HTTP request failed: the server was unreachable or answered with an error status.
|
|
20
|
+
|
|
21
|
+
`status` is None for transport failures (refused, timed out); `detail` is the server's
|
|
22
|
+
error body preview or the transport reason, already sanitized for terminals.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
def __init__(self, message: str, *, url: str, status: int | None, detail: str) -> None:
|
|
26
|
+
super().__init__(message)
|
|
27
|
+
self.url = url
|
|
28
|
+
self.status = status
|
|
29
|
+
self.detail = detail
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class CapabilityError(BackendError):
|
|
33
|
+
"""The backend cannot perform the requested operation, such as returning logprobs."""
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class SuiteError(QuantdiffError, ValueError):
|
|
37
|
+
"""A prompt suite or user prompts file is missing or malformed."""
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class ReportError(QuantdiffError, ValueError):
|
|
41
|
+
"""A report file is missing, malformed, or from an unsupported schema version."""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class RenderError(QuantdiffError):
|
|
45
|
+
"""A scorecard could not be rendered, for example no browser is available for PNG output."""
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Scoring functions: logit divergence, task checks, answer agreement and speed."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from quantdiff.metrics.codeexec import extract_code, run_code_case
|
|
6
|
+
from quantdiff.metrics.jsonschema import (
|
|
7
|
+
SUPPORTED_KEYWORDS,
|
|
8
|
+
extract_json,
|
|
9
|
+
unsupported_keywords,
|
|
10
|
+
validate,
|
|
11
|
+
)
|
|
12
|
+
from quantdiff.metrics.logit import logit_metrics, partition_kld, top1_match
|
|
13
|
+
from quantdiff.metrics.tasks import evaluate_case, perf_metrics, summarize_tasks
|
|
14
|
+
from quantdiff.metrics.textsim import agreement_metrics, normalize, similarity, strip_reasoning
|
|
15
|
+
from quantdiff.metrics.toolcheck import arguments_match, check_tool_case
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"SUPPORTED_KEYWORDS",
|
|
19
|
+
"agreement_metrics",
|
|
20
|
+
"arguments_match",
|
|
21
|
+
"check_tool_case",
|
|
22
|
+
"evaluate_case",
|
|
23
|
+
"extract_code",
|
|
24
|
+
"extract_json",
|
|
25
|
+
"logit_metrics",
|
|
26
|
+
"normalize",
|
|
27
|
+
"partition_kld",
|
|
28
|
+
"perf_metrics",
|
|
29
|
+
"run_code_case",
|
|
30
|
+
"similarity",
|
|
31
|
+
"strip_reasoning",
|
|
32
|
+
"summarize_tasks",
|
|
33
|
+
"top1_match",
|
|
34
|
+
"unsupported_keywords",
|
|
35
|
+
"validate",
|
|
36
|
+
]
|