quantdiff 0.1.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. quantdiff/__init__.py +53 -0
  2. quantdiff/__main__.py +5 -0
  3. quantdiff/_http.py +151 -0
  4. quantdiff/_text.py +13 -0
  5. quantdiff/_version.py +1 -0
  6. quantdiff/api.py +340 -0
  7. quantdiff/backends/__init__.py +28 -0
  8. quantdiff/backends/_common.py +342 -0
  9. quantdiff/backends/base.py +91 -0
  10. quantdiff/backends/llamacpp.py +428 -0
  11. quantdiff/backends/ollama.py +359 -0
  12. quantdiff/backends/openai_compat.py +338 -0
  13. quantdiff/cache.py +240 -0
  14. quantdiff/card.py +1664 -0
  15. quantdiff/cli.py +377 -0
  16. quantdiff/discover.py +488 -0
  17. quantdiff/errors.py +45 -0
  18. quantdiff/metrics/__init__.py +36 -0
  19. quantdiff/metrics/codeexec.py +428 -0
  20. quantdiff/metrics/jsonschema.py +610 -0
  21. quantdiff/metrics/logit.py +214 -0
  22. quantdiff/metrics/tasks.py +114 -0
  23. quantdiff/metrics/textsim.py +66 -0
  24. quantdiff/metrics/toolcheck.py +99 -0
  25. quantdiff/png.py +360 -0
  26. quantdiff/preflight.py +365 -0
  27. quantdiff/progress.py +283 -0
  28. quantdiff/py.typed +0 -0
  29. quantdiff/report.py +780 -0
  30. quantdiff/runner.py +492 -0
  31. quantdiff/spec.py +154 -0
  32. quantdiff/stats.py +226 -0
  33. quantdiff/suites/__init__.py +462 -0
  34. quantdiff/suites/data/chat.jsonl +22 -0
  35. quantdiff/suites/data/code.jsonl +32 -0
  36. quantdiff/suites/data/json.jsonl +34 -0
  37. quantdiff/suites/data/scoring.jsonl +41 -0
  38. quantdiff/suites/data/tools.jsonl +32 -0
  39. quantdiff/types.py +322 -0
  40. quantdiff/verdict.py +1513 -0
  41. quantdiff-0.1.0rc1.dist-info/METADATA +514 -0
  42. quantdiff-0.1.0rc1.dist-info/RECORD +45 -0
  43. quantdiff-0.1.0rc1.dist-info/WHEEL +4 -0
  44. quantdiff-0.1.0rc1.dist-info/entry_points.txt +2 -0
  45. quantdiff-0.1.0rc1.dist-info/licenses/LICENSE +202 -0
quantdiff/types.py ADDED
@@ -0,0 +1,322 @@
1
+ """Plain data types shared by every quantdiff module.
2
+
3
+ All types are frozen dataclasses so results can be cached and serialized without
4
+ defensive copies. Nothing here performs I/O.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from dataclasses import dataclass, field
10
+ from typing import Any, Literal
11
+
12
+ JSONValue = Any
13
+ """A value produced by json.loads. Kept as Any because JSON is recursive."""
14
+
15
+ BackendKind = Literal["ollama", "llamacpp", "openai"]
16
+ TaskKind = Literal["json", "tools", "code", "chat"]
17
+ Role = Literal["system", "user", "assistant", "tool"]
18
+ Severity = Literal["ok", "warn", "fail", "skip"]
19
+ TemplateDialect = Literal["jinja", "go", "unknown"]
20
+ PerfSource = Literal["server", "wall_clock", "cached"]
21
+ ProgressPhase = Literal["connect", "reference", "preflight", "cases", "scoring", "done"]
22
+
23
+
24
+ # Model servers ---------------------------------------------------------------------------
25
+
26
+
27
+ @dataclass(frozen=True, slots=True)
28
+ class CandidateSpec:
29
+ """Where a model is served and how to label it on the scorecard.
30
+
31
+ `model` is the Ollama tag or the served model name. It is empty for llama-server,
32
+ which serves exactly one model. `api_key_env` names an environment variable that
33
+ holds the key; the key itself never lives in a spec, a report, or a log line.
34
+ """
35
+
36
+ kind: BackendKind
37
+ base_url: str
38
+ model: str
39
+ label: str
40
+ api_key_env: str | None = None
41
+
42
+
43
+ @dataclass(frozen=True, slots=True)
44
+ class ServerInfo:
45
+ """What a backend reports about the model it serves."""
46
+
47
+ backend: BackendKind
48
+ model: str
49
+ context_length: int | None
50
+ chat_template: str | None
51
+ template_dialect: TemplateDialect
52
+ supports_logprobs: bool
53
+ exact_token_ids: bool
54
+ """True when the backend scores by token id, so teacher forcing never retokenizes."""
55
+ details: tuple[tuple[str, str], ...] = ()
56
+ """Extra key/value facts worth showing, such as quantization type or file size."""
57
+ size_bytes: int | None = None
58
+ """Size of the served weights on disk, when the server reports it."""
59
+ weights_id: str | None = None
60
+ """Stable identity of the served weights (Ollama digest, GGUF size and params), used to
61
+ spot a candidate that is the same file as the reference."""
62
+
63
+
64
+ # Tokens and distributions ----------------------------------------------------------------
65
+
66
+
67
+ @dataclass(frozen=True, slots=True)
68
+ class TokenProb:
69
+ """One token and its natural-log probability."""
70
+
71
+ token: str
72
+ logprob: float
73
+ token_id: int | None = None
74
+ token_bytes: bytes | None = None
75
+ """The token's raw bytes when the server reports them. A token can hold part of a
76
+ multi-byte character, so text rebuilt from `token` strings can be corrupted while text
77
+ rebuilt from bytes is exact."""
78
+
79
+
80
+ @dataclass(frozen=True, slots=True)
81
+ class TokenStep:
82
+ """One generated position: the token chosen greedily and the top-k alternatives.
83
+
84
+ `top` is sorted by descending logprob and includes the chosen token when it is in
85
+ the top k.
86
+ """
87
+
88
+ chosen: TokenProb
89
+ top: tuple[TokenProb, ...]
90
+
91
+
92
+ TopK = tuple[TokenProb, ...]
93
+ """A next-token distribution truncated to the k most likely tokens, sorted descending."""
94
+
95
+
96
+ @dataclass(frozen=True, slots=True)
97
+ class ScoringPrompt:
98
+ """A raw-completion prompt used for logit metrics. No chat template is applied."""
99
+
100
+ id: str
101
+ text: str
102
+
103
+
104
+ @dataclass(frozen=True, slots=True)
105
+ class ReferenceTrace:
106
+ """The reference model's greedy continuation of one scoring prompt."""
107
+
108
+ prompt_id: str
109
+ prompt_token_ids: tuple[int, ...] | None
110
+ steps: tuple[TokenStep, ...]
111
+
112
+
113
+ # Chat and task cases ---------------------------------------------------------------------
114
+
115
+
116
+ @dataclass(frozen=True, slots=True)
117
+ class Message:
118
+ role: Role
119
+ content: str
120
+
121
+
122
+ @dataclass(frozen=True, slots=True)
123
+ class ToolSpec:
124
+ """A function tool in OpenAI format. `parameters` is a JSON Schema object."""
125
+
126
+ name: str
127
+ description: str
128
+ parameters: dict[str, JSONValue]
129
+
130
+
131
+ @dataclass(frozen=True, slots=True)
132
+ class ToolCall:
133
+ name: str
134
+ arguments: dict[str, JSONValue] | None
135
+ """Parsed arguments, or None when `raw_arguments` was not a JSON object."""
136
+ raw_arguments: str
137
+
138
+
139
+ @dataclass(frozen=True, slots=True)
140
+ class ChatResult:
141
+ text: str
142
+ tool_calls: tuple[ToolCall, ...]
143
+ finish_reason: str | None
144
+ prompt_tokens: int | None
145
+ completion_tokens: int | None
146
+ seconds: float
147
+ decode_tokens_per_second: float | None = None
148
+ """Generation speed as measured by the server itself (excludes load and prompt time)."""
149
+
150
+
151
+ @dataclass(frozen=True, slots=True)
152
+ class TaskCase:
153
+ """One chat prompt plus the checks that decide whether an answer passes.
154
+
155
+ Which optional fields are required depends on `kind`:
156
+ json needs `json_schema`; tools needs `tools` and `expected_tool`;
157
+ code needs `entry_point` and `tests`; chat needs nothing and is scored by
158
+ agreement with the reference answer only.
159
+ """
160
+
161
+ id: str
162
+ kind: TaskKind
163
+ messages: tuple[Message, ...]
164
+ max_tokens: int = 512
165
+ json_schema: dict[str, JSONValue] | None = None
166
+ tools: tuple[ToolSpec, ...] = ()
167
+ expected_tool: str | None = None
168
+ expected_arguments: dict[str, JSONValue] | None = None
169
+ """Subset match: every key here must be present in the call with an equal value."""
170
+ entry_point: str | None = None
171
+ tests: str | None = None
172
+ """Python source with plain assert statements that exercise `entry_point`."""
173
+
174
+
175
+ @dataclass(frozen=True, slots=True)
176
+ class CaseOutcome:
177
+ case_id: str
178
+ kind: TaskKind
179
+ passed: bool | None
180
+ """None means the case was skipped, for example code cases without --allow-code-exec."""
181
+ reason: str = ""
182
+
183
+
184
+ # Metrics and report ----------------------------------------------------------------------
185
+
186
+
187
+ @dataclass(frozen=True, slots=True)
188
+ class LogitMetrics:
189
+ """Teacher-forced comparison against the reference distribution.
190
+
191
+ `kld_*` values are lower bounds on the true KL(reference || candidate): they are
192
+ computed on the partition {top-k tokens, everything else}, which can only
193
+ shrink KL divergence. See docs/methodology.md.
194
+ """
195
+
196
+ prompts: int
197
+ positions: int
198
+ top1_agreement: float
199
+ kld_mean: float
200
+ kld_p99: float
201
+ kld_max: float
202
+ exact_token_ids: bool
203
+ per_prompt: tuple[PromptLogit, ...] = ()
204
+ """Per scoring prompt results, in trace order, so comparisons can be paired by prompt."""
205
+
206
+
207
+ @dataclass(frozen=True, slots=True)
208
+ class PromptLogit:
209
+ """Logit metrics for one scoring prompt. `kld_mean` is None when no position had a
210
+ candidate distribution (the candidate stopped immediately)."""
211
+
212
+ prompt_id: str
213
+ positions: int
214
+ top1_matches: int
215
+ kld_mean: float | None
216
+
217
+
218
+ @dataclass(frozen=True, slots=True)
219
+ class TaskMetrics:
220
+ kind: TaskKind
221
+ total: int
222
+ passed: int
223
+ skipped: int
224
+ failures: tuple[CaseOutcome, ...] = ()
225
+
226
+ @property
227
+ def rate(self) -> float | None:
228
+ scored = self.total - self.skipped
229
+ return None if scored == 0 else self.passed / scored
230
+
231
+
232
+ @dataclass(frozen=True, slots=True)
233
+ class AgreementMetrics:
234
+ """How often the candidate's chat answers match the reference's answers."""
235
+
236
+ cases: int
237
+ exact_match_rate: float
238
+ mean_similarity: float
239
+ per_case: tuple[tuple[str, float], ...] = ()
240
+ """(case id, similarity) pairs, so agreement can be compared case by case."""
241
+
242
+
243
+ @dataclass(frozen=True, slots=True)
244
+ class PerfMetrics:
245
+ tokens_per_second: float | None
246
+ mean_latency_seconds: float | None
247
+ source: PerfSource = "wall_clock"
248
+ """Where tokens_per_second came from: the server's own decode timing (comparable across
249
+ models), wall-clock request time (includes prompt processing), or a cached earlier run
250
+ (not comparable with fresh numbers)."""
251
+
252
+
253
+ @dataclass(frozen=True, slots=True)
254
+ class PreflightFinding:
255
+ check: str
256
+ severity: Severity
257
+ message: str
258
+ fix: str | None = None
259
+
260
+
261
+ @dataclass(frozen=True, slots=True)
262
+ class CandidateResult:
263
+ spec: CandidateSpec
264
+ info: ServerInfo | None
265
+ logit: LogitMetrics | None
266
+ tasks: tuple[TaskMetrics, ...]
267
+ agreement: AgreementMetrics | None
268
+ perf: PerfMetrics | None
269
+ preflight: tuple[PreflightFinding, ...]
270
+ errors: tuple[str, ...] = ()
271
+ outcomes: tuple[CaseOutcome, ...] = ()
272
+ """Every scored case outcome (json, tools, code), so candidates can be compared to the
273
+ reference case by case."""
274
+
275
+
276
+ @dataclass(frozen=True, slots=True)
277
+ class RunSettings:
278
+ suites: tuple[str, ...]
279
+ top_k: int
280
+ score_tokens: int
281
+ allow_code_exec: bool
282
+ seed: int
283
+ prompts_file: str | None = None
284
+ """Base name of the user's prompts file. Never a full path: cards are shared publicly."""
285
+ max_size_bytes: int | None = None
286
+ """The user's --max-size budget: the pick must be a download at most this large."""
287
+ longest_prompt_tokens: int | None = None
288
+ """Estimated length of the longest task or scoring prompt, to judge whether a context
289
+ limit found by pre-flight could have affected these scores."""
290
+
291
+
292
+ @dataclass(frozen=True, slots=True)
293
+ class Report:
294
+ quantdiff_version: str
295
+ created_at: str
296
+ """UTC timestamp in ISO 8601 format."""
297
+ title: str
298
+ settings: RunSettings
299
+ reference: CandidateResult
300
+ candidates: tuple[CandidateResult, ...]
301
+ schema_version: int = 2
302
+ notes: tuple[str, ...] = field(default_factory=tuple)
303
+
304
+
305
+ # Progress ---------------------------------------------------------------------------------
306
+
307
+
308
+ @dataclass(frozen=True, slots=True)
309
+ class ProgressEvent:
310
+ """One step of a run, for progress displays.
311
+
312
+ `completed` and `total` count weighted work units across the whole run, roughly one per
313
+ server request, so a display can show an overall bar and an ETA. `total` is fixed for
314
+ the run; `completed` only grows and reaches `total` on the final "done" event.
315
+ """
316
+
317
+ phase: ProgressPhase
318
+ model: str
319
+ """Label of the model being worked on; empty for run-level events."""
320
+ completed: int
321
+ total: int
322
+ detail: str = ""