quantdiff 0.1.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quantdiff/__init__.py +53 -0
- quantdiff/__main__.py +5 -0
- quantdiff/_http.py +151 -0
- quantdiff/_text.py +13 -0
- quantdiff/_version.py +1 -0
- quantdiff/api.py +340 -0
- quantdiff/backends/__init__.py +28 -0
- quantdiff/backends/_common.py +342 -0
- quantdiff/backends/base.py +91 -0
- quantdiff/backends/llamacpp.py +428 -0
- quantdiff/backends/ollama.py +359 -0
- quantdiff/backends/openai_compat.py +338 -0
- quantdiff/cache.py +240 -0
- quantdiff/card.py +1664 -0
- quantdiff/cli.py +377 -0
- quantdiff/discover.py +488 -0
- quantdiff/errors.py +45 -0
- quantdiff/metrics/__init__.py +36 -0
- quantdiff/metrics/codeexec.py +428 -0
- quantdiff/metrics/jsonschema.py +610 -0
- quantdiff/metrics/logit.py +214 -0
- quantdiff/metrics/tasks.py +114 -0
- quantdiff/metrics/textsim.py +66 -0
- quantdiff/metrics/toolcheck.py +99 -0
- quantdiff/png.py +360 -0
- quantdiff/preflight.py +365 -0
- quantdiff/progress.py +283 -0
- quantdiff/py.typed +0 -0
- quantdiff/report.py +780 -0
- quantdiff/runner.py +492 -0
- quantdiff/spec.py +154 -0
- quantdiff/stats.py +226 -0
- quantdiff/suites/__init__.py +462 -0
- quantdiff/suites/data/chat.jsonl +22 -0
- quantdiff/suites/data/code.jsonl +32 -0
- quantdiff/suites/data/json.jsonl +34 -0
- quantdiff/suites/data/scoring.jsonl +41 -0
- quantdiff/suites/data/tools.jsonl +32 -0
- quantdiff/types.py +322 -0
- quantdiff/verdict.py +1513 -0
- quantdiff-0.1.0rc1.dist-info/METADATA +514 -0
- quantdiff-0.1.0rc1.dist-info/RECORD +45 -0
- quantdiff-0.1.0rc1.dist-info/WHEEL +4 -0
- quantdiff-0.1.0rc1.dist-info/entry_points.txt +2 -0
- quantdiff-0.1.0rc1.dist-info/licenses/LICENSE +202 -0
quantdiff/types.py
ADDED
|
@@ -0,0 +1,322 @@
|
|
|
1
|
+
"""Plain data types shared by every quantdiff module.
|
|
2
|
+
|
|
3
|
+
All types are frozen dataclasses so results can be cached and serialized without
|
|
4
|
+
defensive copies. Nothing here performs I/O.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from typing import Any, Literal
|
|
11
|
+
|
|
12
|
+
JSONValue = Any
|
|
13
|
+
"""A value produced by json.loads. Kept as Any because JSON is recursive."""
|
|
14
|
+
|
|
15
|
+
BackendKind = Literal["ollama", "llamacpp", "openai"]
|
|
16
|
+
TaskKind = Literal["json", "tools", "code", "chat"]
|
|
17
|
+
Role = Literal["system", "user", "assistant", "tool"]
|
|
18
|
+
Severity = Literal["ok", "warn", "fail", "skip"]
|
|
19
|
+
TemplateDialect = Literal["jinja", "go", "unknown"]
|
|
20
|
+
PerfSource = Literal["server", "wall_clock", "cached"]
|
|
21
|
+
ProgressPhase = Literal["connect", "reference", "preflight", "cases", "scoring", "done"]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
# Model servers ---------------------------------------------------------------------------
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True, slots=True)
|
|
28
|
+
class CandidateSpec:
|
|
29
|
+
"""Where a model is served and how to label it on the scorecard.
|
|
30
|
+
|
|
31
|
+
`model` is the Ollama tag or the served model name. It is empty for llama-server,
|
|
32
|
+
which serves exactly one model. `api_key_env` names an environment variable that
|
|
33
|
+
holds the key; the key itself never lives in a spec, a report, or a log line.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
kind: BackendKind
|
|
37
|
+
base_url: str
|
|
38
|
+
model: str
|
|
39
|
+
label: str
|
|
40
|
+
api_key_env: str | None = None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass(frozen=True, slots=True)
|
|
44
|
+
class ServerInfo:
|
|
45
|
+
"""What a backend reports about the model it serves."""
|
|
46
|
+
|
|
47
|
+
backend: BackendKind
|
|
48
|
+
model: str
|
|
49
|
+
context_length: int | None
|
|
50
|
+
chat_template: str | None
|
|
51
|
+
template_dialect: TemplateDialect
|
|
52
|
+
supports_logprobs: bool
|
|
53
|
+
exact_token_ids: bool
|
|
54
|
+
"""True when the backend scores by token id, so teacher forcing never retokenizes."""
|
|
55
|
+
details: tuple[tuple[str, str], ...] = ()
|
|
56
|
+
"""Extra key/value facts worth showing, such as quantization type or file size."""
|
|
57
|
+
size_bytes: int | None = None
|
|
58
|
+
"""Size of the served weights on disk, when the server reports it."""
|
|
59
|
+
weights_id: str | None = None
|
|
60
|
+
"""Stable identity of the served weights (Ollama digest, GGUF size and params), used to
|
|
61
|
+
spot a candidate that is the same file as the reference."""
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
# Tokens and distributions ----------------------------------------------------------------
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass(frozen=True, slots=True)
|
|
68
|
+
class TokenProb:
|
|
69
|
+
"""One token and its natural-log probability."""
|
|
70
|
+
|
|
71
|
+
token: str
|
|
72
|
+
logprob: float
|
|
73
|
+
token_id: int | None = None
|
|
74
|
+
token_bytes: bytes | None = None
|
|
75
|
+
"""The token's raw bytes when the server reports them. A token can hold part of a
|
|
76
|
+
multi-byte character, so text rebuilt from `token` strings can be corrupted while text
|
|
77
|
+
rebuilt from bytes is exact."""
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass(frozen=True, slots=True)
|
|
81
|
+
class TokenStep:
|
|
82
|
+
"""One generated position: the token chosen greedily and the top-k alternatives.
|
|
83
|
+
|
|
84
|
+
`top` is sorted by descending logprob and includes the chosen token when it is in
|
|
85
|
+
the top k.
|
|
86
|
+
"""
|
|
87
|
+
|
|
88
|
+
chosen: TokenProb
|
|
89
|
+
top: tuple[TokenProb, ...]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
TopK = tuple[TokenProb, ...]
|
|
93
|
+
"""A next-token distribution truncated to the k most likely tokens, sorted descending."""
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
@dataclass(frozen=True, slots=True)
|
|
97
|
+
class ScoringPrompt:
|
|
98
|
+
"""A raw-completion prompt used for logit metrics. No chat template is applied."""
|
|
99
|
+
|
|
100
|
+
id: str
|
|
101
|
+
text: str
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
@dataclass(frozen=True, slots=True)
|
|
105
|
+
class ReferenceTrace:
|
|
106
|
+
"""The reference model's greedy continuation of one scoring prompt."""
|
|
107
|
+
|
|
108
|
+
prompt_id: str
|
|
109
|
+
prompt_token_ids: tuple[int, ...] | None
|
|
110
|
+
steps: tuple[TokenStep, ...]
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
# Chat and task cases ---------------------------------------------------------------------
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
@dataclass(frozen=True, slots=True)
|
|
117
|
+
class Message:
|
|
118
|
+
role: Role
|
|
119
|
+
content: str
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
@dataclass(frozen=True, slots=True)
|
|
123
|
+
class ToolSpec:
|
|
124
|
+
"""A function tool in OpenAI format. `parameters` is a JSON Schema object."""
|
|
125
|
+
|
|
126
|
+
name: str
|
|
127
|
+
description: str
|
|
128
|
+
parameters: dict[str, JSONValue]
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
@dataclass(frozen=True, slots=True)
|
|
132
|
+
class ToolCall:
|
|
133
|
+
name: str
|
|
134
|
+
arguments: dict[str, JSONValue] | None
|
|
135
|
+
"""Parsed arguments, or None when `raw_arguments` was not a JSON object."""
|
|
136
|
+
raw_arguments: str
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
@dataclass(frozen=True, slots=True)
|
|
140
|
+
class ChatResult:
|
|
141
|
+
text: str
|
|
142
|
+
tool_calls: tuple[ToolCall, ...]
|
|
143
|
+
finish_reason: str | None
|
|
144
|
+
prompt_tokens: int | None
|
|
145
|
+
completion_tokens: int | None
|
|
146
|
+
seconds: float
|
|
147
|
+
decode_tokens_per_second: float | None = None
|
|
148
|
+
"""Generation speed as measured by the server itself (excludes load and prompt time)."""
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
@dataclass(frozen=True, slots=True)
|
|
152
|
+
class TaskCase:
|
|
153
|
+
"""One chat prompt plus the checks that decide whether an answer passes.
|
|
154
|
+
|
|
155
|
+
Which optional fields are required depends on `kind`:
|
|
156
|
+
json needs `json_schema`; tools needs `tools` and `expected_tool`;
|
|
157
|
+
code needs `entry_point` and `tests`; chat needs nothing and is scored by
|
|
158
|
+
agreement with the reference answer only.
|
|
159
|
+
"""
|
|
160
|
+
|
|
161
|
+
id: str
|
|
162
|
+
kind: TaskKind
|
|
163
|
+
messages: tuple[Message, ...]
|
|
164
|
+
max_tokens: int = 512
|
|
165
|
+
json_schema: dict[str, JSONValue] | None = None
|
|
166
|
+
tools: tuple[ToolSpec, ...] = ()
|
|
167
|
+
expected_tool: str | None = None
|
|
168
|
+
expected_arguments: dict[str, JSONValue] | None = None
|
|
169
|
+
"""Subset match: every key here must be present in the call with an equal value."""
|
|
170
|
+
entry_point: str | None = None
|
|
171
|
+
tests: str | None = None
|
|
172
|
+
"""Python source with plain assert statements that exercise `entry_point`."""
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
@dataclass(frozen=True, slots=True)
|
|
176
|
+
class CaseOutcome:
|
|
177
|
+
case_id: str
|
|
178
|
+
kind: TaskKind
|
|
179
|
+
passed: bool | None
|
|
180
|
+
"""None means the case was skipped, for example code cases without --allow-code-exec."""
|
|
181
|
+
reason: str = ""
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
# Metrics and report ----------------------------------------------------------------------
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
@dataclass(frozen=True, slots=True)
|
|
188
|
+
class LogitMetrics:
|
|
189
|
+
"""Teacher-forced comparison against the reference distribution.
|
|
190
|
+
|
|
191
|
+
`kld_*` values are lower bounds on the true KL(reference || candidate): they are
|
|
192
|
+
computed on the partition {top-k tokens, everything else}, which can only
|
|
193
|
+
shrink KL divergence. See docs/methodology.md.
|
|
194
|
+
"""
|
|
195
|
+
|
|
196
|
+
prompts: int
|
|
197
|
+
positions: int
|
|
198
|
+
top1_agreement: float
|
|
199
|
+
kld_mean: float
|
|
200
|
+
kld_p99: float
|
|
201
|
+
kld_max: float
|
|
202
|
+
exact_token_ids: bool
|
|
203
|
+
per_prompt: tuple[PromptLogit, ...] = ()
|
|
204
|
+
"""Per scoring prompt results, in trace order, so comparisons can be paired by prompt."""
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
@dataclass(frozen=True, slots=True)
|
|
208
|
+
class PromptLogit:
|
|
209
|
+
"""Logit metrics for one scoring prompt. `kld_mean` is None when no position had a
|
|
210
|
+
candidate distribution (the candidate stopped immediately)."""
|
|
211
|
+
|
|
212
|
+
prompt_id: str
|
|
213
|
+
positions: int
|
|
214
|
+
top1_matches: int
|
|
215
|
+
kld_mean: float | None
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
@dataclass(frozen=True, slots=True)
|
|
219
|
+
class TaskMetrics:
|
|
220
|
+
kind: TaskKind
|
|
221
|
+
total: int
|
|
222
|
+
passed: int
|
|
223
|
+
skipped: int
|
|
224
|
+
failures: tuple[CaseOutcome, ...] = ()
|
|
225
|
+
|
|
226
|
+
@property
|
|
227
|
+
def rate(self) -> float | None:
|
|
228
|
+
scored = self.total - self.skipped
|
|
229
|
+
return None if scored == 0 else self.passed / scored
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
@dataclass(frozen=True, slots=True)
|
|
233
|
+
class AgreementMetrics:
|
|
234
|
+
"""How often the candidate's chat answers match the reference's answers."""
|
|
235
|
+
|
|
236
|
+
cases: int
|
|
237
|
+
exact_match_rate: float
|
|
238
|
+
mean_similarity: float
|
|
239
|
+
per_case: tuple[tuple[str, float], ...] = ()
|
|
240
|
+
"""(case id, similarity) pairs, so agreement can be compared case by case."""
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
@dataclass(frozen=True, slots=True)
|
|
244
|
+
class PerfMetrics:
|
|
245
|
+
tokens_per_second: float | None
|
|
246
|
+
mean_latency_seconds: float | None
|
|
247
|
+
source: PerfSource = "wall_clock"
|
|
248
|
+
"""Where tokens_per_second came from: the server's own decode timing (comparable across
|
|
249
|
+
models), wall-clock request time (includes prompt processing), or a cached earlier run
|
|
250
|
+
(not comparable with fresh numbers)."""
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
@dataclass(frozen=True, slots=True)
|
|
254
|
+
class PreflightFinding:
|
|
255
|
+
check: str
|
|
256
|
+
severity: Severity
|
|
257
|
+
message: str
|
|
258
|
+
fix: str | None = None
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
@dataclass(frozen=True, slots=True)
|
|
262
|
+
class CandidateResult:
|
|
263
|
+
spec: CandidateSpec
|
|
264
|
+
info: ServerInfo | None
|
|
265
|
+
logit: LogitMetrics | None
|
|
266
|
+
tasks: tuple[TaskMetrics, ...]
|
|
267
|
+
agreement: AgreementMetrics | None
|
|
268
|
+
perf: PerfMetrics | None
|
|
269
|
+
preflight: tuple[PreflightFinding, ...]
|
|
270
|
+
errors: tuple[str, ...] = ()
|
|
271
|
+
outcomes: tuple[CaseOutcome, ...] = ()
|
|
272
|
+
"""Every scored case outcome (json, tools, code), so candidates can be compared to the
|
|
273
|
+
reference case by case."""
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
@dataclass(frozen=True, slots=True)
|
|
277
|
+
class RunSettings:
|
|
278
|
+
suites: tuple[str, ...]
|
|
279
|
+
top_k: int
|
|
280
|
+
score_tokens: int
|
|
281
|
+
allow_code_exec: bool
|
|
282
|
+
seed: int
|
|
283
|
+
prompts_file: str | None = None
|
|
284
|
+
"""Base name of the user's prompts file. Never a full path: cards are shared publicly."""
|
|
285
|
+
max_size_bytes: int | None = None
|
|
286
|
+
"""The user's --max-size budget: the pick must be a download at most this large."""
|
|
287
|
+
longest_prompt_tokens: int | None = None
|
|
288
|
+
"""Estimated length of the longest task or scoring prompt, to judge whether a context
|
|
289
|
+
limit found by pre-flight could have affected these scores."""
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
@dataclass(frozen=True, slots=True)
|
|
293
|
+
class Report:
|
|
294
|
+
quantdiff_version: str
|
|
295
|
+
created_at: str
|
|
296
|
+
"""UTC timestamp in ISO 8601 format."""
|
|
297
|
+
title: str
|
|
298
|
+
settings: RunSettings
|
|
299
|
+
reference: CandidateResult
|
|
300
|
+
candidates: tuple[CandidateResult, ...]
|
|
301
|
+
schema_version: int = 2
|
|
302
|
+
notes: tuple[str, ...] = field(default_factory=tuple)
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
# Progress ---------------------------------------------------------------------------------
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
@dataclass(frozen=True, slots=True)
|
|
309
|
+
class ProgressEvent:
|
|
310
|
+
"""One step of a run, for progress displays.
|
|
311
|
+
|
|
312
|
+
`completed` and `total` count weighted work units across the whole run, roughly one per
|
|
313
|
+
server request, so a display can show an overall bar and an ETA. `total` is fixed for
|
|
314
|
+
the run; `completed` only grows and reaches `total` on the final "done" event.
|
|
315
|
+
"""
|
|
316
|
+
|
|
317
|
+
phase: ProgressPhase
|
|
318
|
+
model: str
|
|
319
|
+
"""Label of the model being worked on; empty for run-level events."""
|
|
320
|
+
completed: int
|
|
321
|
+
total: int
|
|
322
|
+
detail: str = ""
|