quantdiff 0.1.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. quantdiff/__init__.py +53 -0
  2. quantdiff/__main__.py +5 -0
  3. quantdiff/_http.py +151 -0
  4. quantdiff/_text.py +13 -0
  5. quantdiff/_version.py +1 -0
  6. quantdiff/api.py +340 -0
  7. quantdiff/backends/__init__.py +28 -0
  8. quantdiff/backends/_common.py +342 -0
  9. quantdiff/backends/base.py +91 -0
  10. quantdiff/backends/llamacpp.py +428 -0
  11. quantdiff/backends/ollama.py +359 -0
  12. quantdiff/backends/openai_compat.py +338 -0
  13. quantdiff/cache.py +240 -0
  14. quantdiff/card.py +1664 -0
  15. quantdiff/cli.py +377 -0
  16. quantdiff/discover.py +488 -0
  17. quantdiff/errors.py +45 -0
  18. quantdiff/metrics/__init__.py +36 -0
  19. quantdiff/metrics/codeexec.py +428 -0
  20. quantdiff/metrics/jsonschema.py +610 -0
  21. quantdiff/metrics/logit.py +214 -0
  22. quantdiff/metrics/tasks.py +114 -0
  23. quantdiff/metrics/textsim.py +66 -0
  24. quantdiff/metrics/toolcheck.py +99 -0
  25. quantdiff/png.py +360 -0
  26. quantdiff/preflight.py +365 -0
  27. quantdiff/progress.py +283 -0
  28. quantdiff/py.typed +0 -0
  29. quantdiff/report.py +780 -0
  30. quantdiff/runner.py +492 -0
  31. quantdiff/spec.py +154 -0
  32. quantdiff/stats.py +226 -0
  33. quantdiff/suites/__init__.py +462 -0
  34. quantdiff/suites/data/chat.jsonl +22 -0
  35. quantdiff/suites/data/code.jsonl +32 -0
  36. quantdiff/suites/data/json.jsonl +34 -0
  37. quantdiff/suites/data/scoring.jsonl +41 -0
  38. quantdiff/suites/data/tools.jsonl +32 -0
  39. quantdiff/types.py +322 -0
  40. quantdiff/verdict.py +1513 -0
  41. quantdiff-0.1.0rc1.dist-info/METADATA +514 -0
  42. quantdiff-0.1.0rc1.dist-info/RECORD +45 -0
  43. quantdiff-0.1.0rc1.dist-info/WHEEL +4 -0
  44. quantdiff-0.1.0rc1.dist-info/entry_points.txt +2 -0
  45. quantdiff-0.1.0rc1.dist-info/licenses/LICENSE +202 -0
quantdiff/cli.py ADDED
@@ -0,0 +1,377 @@
1
+ """Command line interface: `quantdiff run | card | discover | suites`."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import logging
7
+ import os
8
+ import sys
9
+ import webbrowser
10
+ from collections.abc import Sequence
11
+ from pathlib import Path
12
+ from typing import TextIO
13
+
14
+ from quantdiff._text import printable_lines
15
+ from quantdiff._version import __version__
16
+ from quantdiff.api import RunFiles, build_plan, write_run
17
+ from quantdiff.cache import ReferenceCache
18
+ from quantdiff.card import render_html, render_markdown, render_terminal
19
+ from quantdiff.discover import closest_tags, discover_ollama, format_discovery, installed_tags
20
+ from quantdiff.errors import BackendError, QuantdiffError, SpecError
21
+ from quantdiff.png import render_png
22
+ from quantdiff.preflight import DEFAULT_CONTEXT_PROBE_TOKENS
23
+ from quantdiff.progress import make_progress
24
+ from quantdiff.report import load_report
25
+ from quantdiff.runner import execute
26
+ from quantdiff.spec import parse_spec
27
+ from quantdiff.suites import BUILTIN_SUITES, load_builtin
28
+ from quantdiff.verdict import Status, Verdict, judge
29
+
30
+ EXIT_OK = 0
31
+ EXIT_ERROR = 1
32
+ EXIT_REGRESSION = 3
33
+ EXIT_INTERRUPTED = 130
34
+
35
+ _FAILING_STATUSES: dict[str, frozenset[Status]] = {
36
+ "never": frozenset(),
37
+ "avoid": frozenset({"avoid", "failed"}),
38
+ "inconclusive": frozenset({"avoid", "failed", "inconclusive", "usable"}),
39
+ }
40
+ """Candidate statuses that make `run --fail-on LEVEL` exit with EXIT_REGRESSION. Any level
41
+ but never also fails when no candidate is recommended, so thin evidence never passes CI;
42
+ avoid lets that through only when a usable candidate fits the --max-size budget."""
43
+
44
+ _DISCOVER_HINT = "`quantdiff discover` lists the models your local Ollama has"
45
+ _SUGGESTION_TIMEOUT_SECONDS = 2.0
46
+ _MISSING_MODEL_WORDS = ("not available", "not found", "does not exist")
47
+
48
+ _SUITE_SUMMARIES = {
49
+ "json": "structured output that must match a JSON Schema",
50
+ "tools": "picking the right tool with valid, correct arguments (or none)",
51
+ "code": "Python functions checked by hidden asserts (needs --allow-code-exec)",
52
+ "chat": "short answers scored by agreement with the reference",
53
+ }
54
+
55
+ _EPILOG = """\
56
+ quick start:
57
+ 1. quantdiff discover
58
+ lists your Ollama models and prints a ready-to-run command for each model
59
+ 2. quantdiff run --ref ollama:qwen2.5:7b-instruct-q8_0 --cand ollama:qwen2.5:7b-instruct-q4_K_M
60
+ compares each candidate with the reference and ends with the verdict
61
+ 3. share runs/<timestamp>/card.png
62
+ the scorecard image; card.md in the same folder pastes into Reddit or GitHub
63
+
64
+ more examples:
65
+ quantdiff run --ref ollama:qwen2.5:7b-instruct-q8_0 \\
66
+ --cand ollama:qwen2.5:7b-instruct-q4_K_M --cand ollama:qwen2.5:7b-instruct-q3_K_M
67
+ quantdiff run --ref ollama:qwen3:8b-q8_0 --cand ollama=ollama:qwen3:8b \\
68
+ --cand unsloth=ollama:hf.co/unsloth/Qwen3-8B-GGUF:Q4_K_M
69
+ quantdiff run --ref ollama:qwen2.5:7b-instruct-q8_0 --cand gguf=llamacpp:http://127.0.0.1:8080
70
+ quantdiff run ... --max-size 6GB pick the closest download that fits in 6 GB
71
+ quantdiff run ... --fail-on avoid for CI: exit 3 on any avoid, or when nothing is
72
+ recommended and nothing usable fits
73
+ quantdiff card runs/20261003T090504Z --format png -o card.png
74
+
75
+ exit codes:
76
+ 0 ok
77
+ 1 error
78
+ 3 regression found or nothing recommended (run --fail-on)
79
+ 130 interrupted
80
+ """
81
+
82
+
83
+ def main(argv: Sequence[str] | None = None) -> int:
84
+ parser = _build_parser()
85
+ args = parser.parse_args(argv)
86
+ logging.basicConfig(
87
+ level=logging.DEBUG if args.verbose else logging.WARNING,
88
+ format="%(levelname)s %(name)s: %(message)s",
89
+ )
90
+ try:
91
+ return int(args.handler(args))
92
+ except BackendError as exc:
93
+ _error(str(exc), hints=[*_did_you_mean(str(exc), args), _DISCOVER_HINT])
94
+ return EXIT_ERROR
95
+ except QuantdiffError as exc:
96
+ _error(str(exc))
97
+ return EXIT_ERROR
98
+ except KeyboardInterrupt:
99
+ print("\nquantdiff: interrupted", file=sys.stderr)
100
+ return EXIT_INTERRUPTED
101
+
102
+
103
+ def _build_parser() -> argparse.ArgumentParser:
104
+ parser = argparse.ArgumentParser(
105
+ prog="quantdiff",
106
+ description="Find out which download of a model gives the best answers on your prompts.",
107
+ epilog=_EPILOG,
108
+ formatter_class=argparse.RawDescriptionHelpFormatter,
109
+ )
110
+ parser.add_argument("--version", action="version", version=f"quantdiff {__version__}")
111
+ parser.add_argument("-v", "--verbose", action="store_true", help="debug logging to stderr")
112
+ common = argparse.ArgumentParser(add_help=False)
113
+ # SUPPRESS keeps a root-level -v from being reset to False by the subcommand parser.
114
+ common.add_argument(
115
+ "-v", "--verbose", action="store_true", default=argparse.SUPPRESS, help="debug logging"
116
+ )
117
+ commands = parser.add_subparsers(dest="command", required=True, metavar="COMMAND")
118
+ _add_run(commands, common)
119
+ _add_card(commands, common)
120
+
121
+ discover = commands.add_parser(
122
+ "discover", parents=[common], help="list local Ollama models and suggest a comparison"
123
+ )
124
+ discover.add_argument("--host", metavar="URL", help="Ollama URL (default: OLLAMA_HOST)")
125
+ discover.set_defaults(handler=_cmd_discover)
126
+
127
+ suites = commands.add_parser("suites", parents=[common], help="list built-in suites")
128
+ suites.set_defaults(handler=_cmd_suites)
129
+ return parser
130
+
131
+
132
+ def _add_run(
133
+ commands: argparse._SubParsersAction[argparse.ArgumentParser], common: argparse.ArgumentParser
134
+ ) -> None:
135
+ run = commands.add_parser(
136
+ "run",
137
+ parents=[common],
138
+ help="compare candidates against a reference",
139
+ description="Compare candidate downloads of a model against a reference download.",
140
+ epilog=_EPILOG,
141
+ formatter_class=argparse.RawDescriptionHelpFormatter,
142
+ )
143
+ models = run.add_argument_group("models")
144
+ models.add_argument("--ref", required=True, metavar="SPEC", help="reference model spec")
145
+ models.add_argument(
146
+ "--cand", required=True, action="append", metavar="SPEC", help="candidate spec, repeatable"
147
+ )
148
+ work = run.add_argument_group("what to run")
149
+ work.add_argument("--suite", type=_csv, metavar="NAMES", help="comma list of built-in suites")
150
+ work.add_argument("--prompts", metavar="FILE", help="your own prompts as JSONL")
151
+ work.add_argument("--scoring-prompts", metavar="FILE", help="raw prompts for logit metrics")
152
+ work.add_argument(
153
+ "--max-cases",
154
+ type=int,
155
+ metavar="N",
156
+ help="cap each suite, your prompts file and the scoring prompts at N",
157
+ )
158
+ work.add_argument(
159
+ "--allow-code-exec", action="store_true", help="run model-written code for the code suite"
160
+ )
161
+ tuning = run.add_argument_group("tuning")
162
+ tuning.add_argument("--top-k", type=int, default=10, help="logprobs per position, 1-20")
163
+ tuning.add_argument(
164
+ "--score-tokens", type=int, default=32, help="tokens per scoring prompt; 0 disables"
165
+ )
166
+ tuning.add_argument("--seed", type=int, default=0, help="sampling seed sent to servers")
167
+ checks = run.add_argument_group("pre-flight checks")
168
+ checks.add_argument("--hf-repo", metavar="OWNER/NAME", help="upstream repo for template check")
169
+ checks.add_argument("--offline", action="store_true", help="never contact huggingface.co")
170
+ checks.add_argument("--no-preflight", action="store_true", help="skip pre-flight checks")
171
+ checks.add_argument(
172
+ "--context-probe-tokens",
173
+ type=int,
174
+ default=DEFAULT_CONTEXT_PROBE_TOKENS,
175
+ metavar="N",
176
+ help="length of the truncation probe; 0 disables",
177
+ )
178
+ output = run.add_argument_group("output")
179
+ output.add_argument("--title", help="scorecard title")
180
+ output.add_argument("--out", default="runs", metavar="DIR", help="output directory")
181
+ output.add_argument("--no-png", action="store_true", help="skip card.png")
182
+ output.add_argument("--open", action="store_true", help="open the scorecard when done")
183
+ output.add_argument("--no-cache", action="store_true", help="always rerun the reference")
184
+ output.add_argument("-q", "--quiet", action="store_true", help="no progress output")
185
+ verdict = run.add_argument_group("verdict")
186
+ verdict.add_argument(
187
+ "--max-size",
188
+ metavar="SIZE",
189
+ help="largest download you can run, e.g. 6GB, 6.5G or 800MB (decimal units); the"
190
+ " pick is the closest download that fits",
191
+ )
192
+ verdict.add_argument(
193
+ "--fail-on",
194
+ choices=tuple(_FAILING_STATUSES),
195
+ default="never",
196
+ help="avoid: exit 3 when any candidate is avoid or failed, or nothing is recommended"
197
+ " and nothing usable fits; inconclusive: also when any candidate is inconclusive or"
198
+ " usable, or nothing is recommended; default: never",
199
+ )
200
+ run.set_defaults(handler=_cmd_run)
201
+
202
+
203
+ def _add_card(
204
+ commands: argparse._SubParsersAction[argparse.ArgumentParser], common: argparse.ArgumentParser
205
+ ) -> None:
206
+ card = commands.add_parser("card", parents=[common], help="render a saved report")
207
+ card.add_argument("path", metavar="PATH", help="run directory or report.json")
208
+ card.add_argument("--format", choices=("txt", "md", "html", "png"), default="txt")
209
+ card.add_argument("-o", "--output", metavar="FILE", help="write to FILE instead of stdout")
210
+ card.set_defaults(handler=_cmd_card)
211
+
212
+
213
+ def _cmd_run(args: argparse.Namespace) -> int:
214
+ plan = build_plan(
215
+ args.ref,
216
+ args.cand,
217
+ suites=args.suite,
218
+ prompts=args.prompts,
219
+ scoring_prompts=args.scoring_prompts,
220
+ top_k=args.top_k,
221
+ score_tokens=args.score_tokens,
222
+ max_cases=args.max_cases,
223
+ allow_code_exec=args.allow_code_exec,
224
+ seed=args.seed,
225
+ hf_repo=args.hf_repo,
226
+ offline=args.offline,
227
+ preflight=not args.no_preflight,
228
+ context_probe_tokens=args.context_probe_tokens,
229
+ title=args.title,
230
+ max_size=args.max_size,
231
+ )
232
+ progress = None if args.quiet else make_progress(sys.stderr)
233
+ cache = None if args.no_cache else ReferenceCache()
234
+ try:
235
+ report = execute(plan, cache=cache, progress=progress)
236
+ finally:
237
+ if progress is not None:
238
+ progress.close()
239
+ verdict = judge(report, scoring=plan.scoring)
240
+ files = write_run(report, args.out, png=not args.no_png, verdict=verdict)
241
+ print(render_terminal(report, color=_use_color(sys.stdout), verdict=verdict), flush=True)
242
+ _print_saved(files, sys.stderr)
243
+ # Repeated as the last line of stdout because people skim the end of the output.
244
+ print(f"\n{verdict.headline}", flush=True)
245
+ if args.open:
246
+ webbrowser.open(files.html.resolve().as_uri())
247
+ regression = _regression(verdict, args.fail_on, budget=report.settings.max_size_bytes)
248
+ if regression is None:
249
+ return EXIT_OK
250
+ print(f"quantdiff: {regression}", file=sys.stderr)
251
+ return EXIT_REGRESSION
252
+
253
+
254
+ def _cmd_card(args: argparse.Namespace) -> int:
255
+ path = Path(args.path)
256
+ report = load_report(path / "report.json" if path.is_dir() else path)
257
+ if args.format == "png":
258
+ target = Path(args.output or "card.png")
259
+ render_png(render_html(report), target)
260
+ print(f"Saved {target}", file=sys.stderr)
261
+ return EXIT_OK
262
+ if args.format == "html":
263
+ text = render_html(report)
264
+ elif args.format == "md":
265
+ text = render_markdown(report)
266
+ else:
267
+ text = render_terminal(report, color=args.output is None and _use_color(sys.stdout))
268
+ if args.output is None:
269
+ print(text)
270
+ else:
271
+ Path(args.output).write_text(text, encoding="utf-8")
272
+ return EXIT_OK
273
+
274
+
275
+ def _cmd_discover(args: argparse.Namespace) -> int:
276
+ print(format_discovery(discover_ollama(args.host)))
277
+ return EXIT_OK
278
+
279
+
280
+ def _cmd_suites(_: argparse.Namespace) -> int:
281
+ for name in BUILTIN_SUITES:
282
+ print(f"{name:<6} {len(load_builtin(name)):>3} cases {_SUITE_SUMMARIES[name]}")
283
+ return EXIT_OK
284
+
285
+
286
+ def _regression(verdict: Verdict, level: str, *, budget: int | None) -> str | None:
287
+ """Why `--fail-on level` fails the run, or None when it passes."""
288
+ failing = _FAILING_STATUSES[level]
289
+ found = [
290
+ f"{item.label} is {item.status}" for item in verdict.candidates if item.status in failing
291
+ ]
292
+ if failing and not any(item.status == "recommended" for item in verdict.candidates):
293
+ usable_fits = any(
294
+ item.status == "usable" and (budget is None or item.fits_budget is True)
295
+ for item in verdict.candidates
296
+ )
297
+ if "usable" in failing:
298
+ found.append("no candidate is recommended")
299
+ elif not usable_fits:
300
+ within = "" if budget is None else " within --max-size"
301
+ found.append(f"no candidate is recommended or usable{within}")
302
+ if not found:
303
+ return None
304
+ return f"--fail-on {level}: {', '.join(found)}; exiting with code {EXIT_REGRESSION}"
305
+
306
+
307
+ def _did_you_mean(message: str, args: argparse.Namespace) -> list[str]:
308
+ """Installed Ollama tags close to each tag that `message` says is missing.
309
+
310
+ Best effort: when Ollama cannot be listed there are simply no suggestions.
311
+ """
312
+ if args.command != "run":
313
+ return []
314
+ lines = message.lower().splitlines()
315
+ listings: dict[str, list[str]] = {}
316
+ hints = []
317
+ for text in dict.fromkeys([args.ref, *args.cand]):
318
+ try:
319
+ spec = parse_spec(text)
320
+ except SpecError:
321
+ continue
322
+ if spec.kind != "ollama" or not _says_missing(lines, spec.model):
323
+ continue
324
+ if spec.base_url not in listings:
325
+ try:
326
+ listings[spec.base_url] = installed_tags(
327
+ spec.base_url, timeout=_SUGGESTION_TIMEOUT_SECONDS
328
+ )
329
+ except QuantdiffError:
330
+ listings[spec.base_url] = []
331
+ close = closest_tags(spec.model, listings[spec.base_url])
332
+ if close:
333
+ hints.append(f"{spec.model} is not installed; did you mean: {', '.join(close)}")
334
+ return hints
335
+
336
+
337
+ def _says_missing(lines: Sequence[str], model: str) -> bool:
338
+ quoted = repr(model).lower()
339
+ return any(
340
+ quoted in line and any(words in line for words in _MISSING_MODEL_WORDS) for line in lines
341
+ )
342
+
343
+
344
+ def _print_saved(files: RunFiles, stream: TextIO) -> None:
345
+ lines = []
346
+ if files.png is not None:
347
+ lines.append(f"\nScorecard image to share: {files.png}")
348
+ lines.append(f"\nSaved to {files.directory}")
349
+ if files.png is not None:
350
+ lines.append(" card.png share on Reddit or X")
351
+ lines.append(" card.md paste into a Reddit post or GitHub issue")
352
+ lines.append(" card.html open in a browser")
353
+ lines.append(" report.json raw results; re-render with `quantdiff card`")
354
+ if files.png_error is not None:
355
+ lines.append(f"No card.png: {files.png_error}")
356
+ print("\n".join(lines), file=stream)
357
+
358
+
359
+ def _error(message: str, *, hints: Sequence[str] = ()) -> None:
360
+ print(f"quantdiff: error: {printable_lines(message)}", file=sys.stderr)
361
+ for hint in hints:
362
+ print(f"hint: {printable_lines(hint)}", file=sys.stderr)
363
+
364
+
365
+ def _csv(value: str) -> list[str]:
366
+ names = [part.strip() for part in value.split(",") if part.strip()]
367
+ if not names:
368
+ raise argparse.ArgumentTypeError("expected a comma separated list")
369
+ return names
370
+
371
+
372
+ def _use_color(stream: TextIO) -> bool:
373
+ return stream.isatty() and "NO_COLOR" not in os.environ and os.environ.get("TERM") != "dumb"
374
+
375
+
376
+ if __name__ == "__main__":
377
+ raise SystemExit(main())