zeroquantz 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- zeroquantz/__init__.py +14 -0
- zeroquantz/__main__.py +8 -0
- zeroquantz/agent/__init__.py +16 -0
- zeroquantz/agent/dispatcher.py +520 -0
- zeroquantz/agent/intents.py +46 -0
- zeroquantz/agent/parser.py +255 -0
- zeroquantz/benchmark/__init__.py +7 -0
- zeroquantz/benchmark/latency.py +66 -0
- zeroquantz/benchmark/memory.py +41 -0
- zeroquantz/benchmark/quality.py +38 -0
- zeroquantz/benchmark/runner.py +151 -0
- zeroquantz/cli/__init__.py +7 -0
- zeroquantz/cli/app.py +98 -0
- zeroquantz/cli/commands.py +459 -0
- zeroquantz/cli/interactive.py +56 -0
- zeroquantz/core/__init__.py +7 -0
- zeroquantz/core/artifacts.py +179 -0
- zeroquantz/core/context.py +127 -0
- zeroquantz/core/events.py +30 -0
- zeroquantz/core/exceptions.py +105 -0
- zeroquantz/core/session.py +202 -0
- zeroquantz/core/subenv.py +202 -0
- zeroquantz/deploy/__init__.py +25 -0
- zeroquantz/deploy/assets.py +161 -0
- zeroquantz/deploy/launcher.py +80 -0
- zeroquantz/deploy/runtime_env.py +66 -0
- zeroquantz/deploy/targets.py +154 -0
- zeroquantz/export/__init__.py +8 -0
- zeroquantz/export/exporter.py +68 -0
- zeroquantz/export/report.py +203 -0
- zeroquantz/hardware/__init__.py +15 -0
- zeroquantz/hardware/capabilities.py +152 -0
- zeroquantz/hardware/detector.py +200 -0
- zeroquantz/hardware/gpu.py +31 -0
- zeroquantz/models/__init__.py +8 -0
- zeroquantz/models/architecture.py +168 -0
- zeroquantz/models/downloader.py +161 -0
- zeroquantz/models/hf_auth.py +105 -0
- zeroquantz/models/inspector.py +249 -0
- zeroquantz/models/metadata.py +108 -0
- zeroquantz/models/search.py +71 -0
- zeroquantz/optimization/__init__.py +22 -0
- zeroquantz/optimization/candidate.py +272 -0
- zeroquantz/optimization/constraints.py +70 -0
- zeroquantz/optimization/fit.py +203 -0
- zeroquantz/optimization/pareto.py +66 -0
- zeroquantz/optimization/planner.py +297 -0
- zeroquantz/optimization/recommender.py +149 -0
- zeroquantz/profiling/__init__.py +18 -0
- zeroquantz/profiling/calibration.py +74 -0
- zeroquantz/profiling/sensitivity.py +234 -0
- zeroquantz/quantization/__init__.py +17 -0
- zeroquantz/quantization/backends/__init__.py +8 -0
- zeroquantz/quantization/backends/bitsandbytes.py +210 -0
- zeroquantz/quantization/backends/torchao.py +198 -0
- zeroquantz/quantization/base.py +136 -0
- zeroquantz/quantization/catalog.py +321 -0
- zeroquantz/quantization/config.py +106 -0
- zeroquantz/quantization/gguf_pipeline.py +210 -0
- zeroquantz/quantization/isolated.py +248 -0
- zeroquantz/quantization/memory.py +133 -0
- zeroquantz/quantization/native.py +91 -0
- zeroquantz/quantization/registry.py +101 -0
- zeroquantz/render.py +341 -0
- zeroquantz/runtimes/__init__.py +18 -0
- zeroquantz/runtimes/base.py +64 -0
- zeroquantz/runtimes/compatibility.py +91 -0
- zeroquantz/runtimes/registry.py +70 -0
- zeroquantz/runtimes/transformers.py +53 -0
- zeroquantz/runtimes/vllm.py +83 -0
- zeroquantz/tui/__init__.py +13 -0
- zeroquantz/tui/app.py +77 -0
- zeroquantz/tui/banner.py +47 -0
- zeroquantz/tui/screens/__init__.py +25 -0
- zeroquantz/tui/screens/confirm.py +41 -0
- zeroquantz/tui/screens/execute.py +194 -0
- zeroquantz/tui/screens/model_select.py +206 -0
- zeroquantz/tui/screens/plan.py +177 -0
- zeroquantz/tui/screens/quantize_select.py +272 -0
- zeroquantz/tui/screens/settings.py +219 -0
- zeroquantz/tui/screens/token.py +94 -0
- zeroquantz/tui/screens/welcome.py +128 -0
- zeroquantz/tui/screens/workspace.py +175 -0
- zeroquantz/tui/styles/app.tcss +424 -0
- zeroquantz/tui/widgets/__init__.py +9 -0
- zeroquantz/tui/widgets/chip.py +36 -0
- zeroquantz/tui/widgets/sidebar.py +107 -0
- zeroquantz/tui/widgets/status_bar.py +43 -0
- zeroquantz/utils/__init__.py +8 -0
- zeroquantz/utils/config.py +46 -0
- zeroquantz/utils/env.py +78 -0
- zeroquantz/utils/logging.py +73 -0
- zeroquantz/utils/metrics.py +98 -0
- zeroquantz/utils/paths.py +57 -0
- zeroquantz/utils/units.py +134 -0
- zeroquantz/verification/__init__.py +17 -0
- zeroquantz/verification/logits.py +55 -0
- zeroquantz/verification/report.py +186 -0
- zeroquantz/verification/weights.py +44 -0
- zeroquantz/version.py +8 -0
- zeroquantz-0.1.0.dist-info/METADATA +72 -0
- zeroquantz-0.1.0.dist-info/RECORD +105 -0
- zeroquantz-0.1.0.dist-info/WHEEL +4 -0
- zeroquantz-0.1.0.dist-info/entry_points.txt +2 -0
- zeroquantz-0.1.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,297 @@
|
|
|
1
|
+
"""Deterministic mixed-precision planning.
|
|
2
|
+
|
|
3
|
+
Given a memory budget, per-layer sensitivity, and the precisions the hardware
|
|
4
|
+
supports, assign a precision to each module. The v0.1 algorithm is an explicit,
|
|
5
|
+
replaceable greedy strategy (spec §12):
|
|
6
|
+
|
|
7
|
+
1. Start every transformer sub-module at the lowest requested precision.
|
|
8
|
+
2. Keep critical modules (embeddings, LM head) at base precision.
|
|
9
|
+
3. Rank sub-modules by sensitivity (from a profiler, or a sensible prior).
|
|
10
|
+
4. Promote the most sensitive sub-modules to the higher precision while the
|
|
11
|
+
memory budget still allows.
|
|
12
|
+
|
|
13
|
+
The result is both a human-readable plan and a :class:`QuantizationConfig` the
|
|
14
|
+
mixed backend can consume.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from typing import TYPE_CHECKING
|
|
20
|
+
|
|
21
|
+
from pydantic import BaseModel, Field
|
|
22
|
+
|
|
23
|
+
from zeroquantz.models import architecture as arch
|
|
24
|
+
from zeroquantz.profiling.sensitivity import layer_sensitivity_prior
|
|
25
|
+
from zeroquantz.quantization import memory
|
|
26
|
+
from zeroquantz.quantization.config import QuantizationConfig
|
|
27
|
+
from zeroquantz.utils import units
|
|
28
|
+
|
|
29
|
+
if TYPE_CHECKING:
|
|
30
|
+
from zeroquantz.hardware.capabilities import HardwareProfile
|
|
31
|
+
from zeroquantz.models.metadata import ModelProfile
|
|
32
|
+
from zeroquantz.optimization.constraints import OptimizationGoal
|
|
33
|
+
from zeroquantz.profiling.sensitivity import SensitivityProfile
|
|
34
|
+
|
|
35
|
+
_PRECISION_BITS = {"bf16": 16, "fp16": 16, "int8": 8, "int4": 4, "nf4": 4}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class PrecisionAssignment(BaseModel):
|
|
39
|
+
module: str
|
|
40
|
+
precision: str
|
|
41
|
+
params: int = 0
|
|
42
|
+
sensitivity: float | None = None
|
|
43
|
+
critical: bool = False
|
|
44
|
+
|
|
45
|
+
@property
|
|
46
|
+
def weight_bytes(self) -> float:
|
|
47
|
+
return units.param_bytes(self.params, self.precision)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class MixedPrecisionPlan(BaseModel):
|
|
51
|
+
"""A per-module precision assignment plus its estimated cost."""
|
|
52
|
+
|
|
53
|
+
model_id: str
|
|
54
|
+
base_precision: str
|
|
55
|
+
low_precision: str
|
|
56
|
+
high_precision: str
|
|
57
|
+
assignments: list[PrecisionAssignment]
|
|
58
|
+
estimated_size_gb: float
|
|
59
|
+
estimated_vram_gb: float
|
|
60
|
+
budget_gb: float | None = None
|
|
61
|
+
fits_budget: bool = True
|
|
62
|
+
notes: list[str] = Field(default_factory=list)
|
|
63
|
+
|
|
64
|
+
def to_config(self, backend: str = "mixed") -> QuantizationConfig:
|
|
65
|
+
overrides = {
|
|
66
|
+
a.module: _PRECISION_BITS.get(a.precision, 4)
|
|
67
|
+
for a in self.assignments
|
|
68
|
+
if not a.critical
|
|
69
|
+
}
|
|
70
|
+
return QuantizationConfig(
|
|
71
|
+
backend=backend,
|
|
72
|
+
method="mixed",
|
|
73
|
+
weight_bits=_PRECISION_BITS.get(self.low_precision, 4),
|
|
74
|
+
activation_bits=16,
|
|
75
|
+
compute_dtype="bfloat16" if self.base_precision == "bf16" else "float16",
|
|
76
|
+
skip_modules=[a.module for a in self.assignments if a.critical],
|
|
77
|
+
module_overrides=overrides,
|
|
78
|
+
extra={
|
|
79
|
+
"low_precision": self.low_precision,
|
|
80
|
+
"high_precision": self.high_precision,
|
|
81
|
+
"high_fraction": self._high_fraction(),
|
|
82
|
+
},
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
def _high_fraction(self) -> float:
|
|
86
|
+
quant = [a for a in self.assignments if not a.critical]
|
|
87
|
+
if not quant:
|
|
88
|
+
return 0.0
|
|
89
|
+
promoted = sum(a.params for a in quant if a.precision == self.high_precision)
|
|
90
|
+
total = sum(a.params for a in quant) or 1
|
|
91
|
+
return round(promoted / total, 3)
|
|
92
|
+
|
|
93
|
+
def grouped_rows(self) -> list[tuple[str, str]]:
|
|
94
|
+
"""Collapse consecutive same-precision layers into ranges for display,
|
|
95
|
+
e.g. ``("layers.1-33", "INT4")`` (spec §12 style)."""
|
|
96
|
+
rows: list[tuple[str, str]] = []
|
|
97
|
+
run_start: int | None = None
|
|
98
|
+
run_prec: str | None = None
|
|
99
|
+
run_end = -1
|
|
100
|
+
|
|
101
|
+
def flush() -> None:
|
|
102
|
+
if run_start is None or run_prec is None:
|
|
103
|
+
return
|
|
104
|
+
name = f"layers.{run_start}" if run_start == run_end else f"layers.{run_start}-{run_end}"
|
|
105
|
+
rows.append((name, run_prec.upper()))
|
|
106
|
+
|
|
107
|
+
layer_assigns = _per_layer_precision(self.assignments)
|
|
108
|
+
for idx, prec in layer_assigns:
|
|
109
|
+
if prec == run_prec:
|
|
110
|
+
run_end = idx
|
|
111
|
+
else:
|
|
112
|
+
flush()
|
|
113
|
+
run_start = run_end = idx
|
|
114
|
+
run_prec = prec
|
|
115
|
+
flush()
|
|
116
|
+
|
|
117
|
+
# Prepend/append critical modules.
|
|
118
|
+
critical = [(a.module, a.precision.upper()) for a in self.assignments if a.critical]
|
|
119
|
+
return critical[:1] + rows + critical[1:]
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
class MixedPrecisionPlanner:
|
|
123
|
+
"""Greedy budget-aware precision assignment (spec §12)."""
|
|
124
|
+
|
|
125
|
+
@staticmethod
|
|
126
|
+
def plan(
|
|
127
|
+
model: ModelProfile,
|
|
128
|
+
hardware: HardwareProfile | None = None,
|
|
129
|
+
goal: OptimizationGoal | None = None,
|
|
130
|
+
sensitivity: SensitivityProfile | None = None,
|
|
131
|
+
*,
|
|
132
|
+
low_precision: str = "int4",
|
|
133
|
+
high_precision: str = "int8",
|
|
134
|
+
base_precision: str | None = None,
|
|
135
|
+
) -> MixedPrecisionPlan:
|
|
136
|
+
base_precision = base_precision or (
|
|
137
|
+
"bf16" if (hardware and hardware.supports_bf16) else "fp16"
|
|
138
|
+
)
|
|
139
|
+
assignments = MixedPrecisionPlanner._initial_assignments(
|
|
140
|
+
model, low_precision, base_precision, sensitivity
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
budget_gb = MixedPrecisionPlanner._resolve_budget(goal, hardware)
|
|
144
|
+
budget_bytes = units.gb_to_bytes(budget_gb) if budget_gb is not None else None
|
|
145
|
+
overhead = memory.estimate_runtime_overhead_bytes(
|
|
146
|
+
model, compute_dtype=base_precision if base_precision != "bf16" else "bfloat16"
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
promotable = sorted(
|
|
150
|
+
(a for a in assignments if not a.critical and a.params > 0),
|
|
151
|
+
key=lambda a: (a.sensitivity or 0.0),
|
|
152
|
+
reverse=True,
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
current = _weight_bytes(assignments)
|
|
156
|
+
notes: list[str] = []
|
|
157
|
+
|
|
158
|
+
if budget_bytes is not None and current + overhead > budget_bytes:
|
|
159
|
+
notes.append(
|
|
160
|
+
"Budget cannot be met even with every layer at the lowest precision; "
|
|
161
|
+
"consider a smaller model or quantizing embeddings/LM head."
|
|
162
|
+
)
|
|
163
|
+
else:
|
|
164
|
+
for assign in promotable:
|
|
165
|
+
delta = units.param_bytes(assign.params, high_precision) - units.param_bytes(
|
|
166
|
+
assign.params, low_precision
|
|
167
|
+
)
|
|
168
|
+
if budget_bytes is None:
|
|
169
|
+
# No hard budget: promote the most sensitive ~25% by param mass.
|
|
170
|
+
promoted_mass = sum(
|
|
171
|
+
a.params for a in assignments if a.precision == high_precision
|
|
172
|
+
)
|
|
173
|
+
quant_mass = sum(a.params for a in assignments if not a.critical) or 1
|
|
174
|
+
if promoted_mass / quant_mass >= 0.25:
|
|
175
|
+
break
|
|
176
|
+
assign.precision = high_precision
|
|
177
|
+
current += delta
|
|
178
|
+
elif current + overhead + delta <= budget_bytes:
|
|
179
|
+
assign.precision = high_precision
|
|
180
|
+
current += delta
|
|
181
|
+
|
|
182
|
+
size_gb = round(units.bytes_to_gb(current), 2)
|
|
183
|
+
vram_gb = round(units.bytes_to_gb(current + overhead), 2)
|
|
184
|
+
fits = budget_gb is None or vram_gb <= budget_gb
|
|
185
|
+
|
|
186
|
+
return MixedPrecisionPlan(
|
|
187
|
+
model_id=model.model_id,
|
|
188
|
+
base_precision=base_precision,
|
|
189
|
+
low_precision=low_precision,
|
|
190
|
+
high_precision=high_precision,
|
|
191
|
+
assignments=assignments,
|
|
192
|
+
estimated_size_gb=size_gb,
|
|
193
|
+
estimated_vram_gb=vram_gb,
|
|
194
|
+
budget_gb=budget_gb,
|
|
195
|
+
fits_budget=fits,
|
|
196
|
+
notes=notes,
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
# ---- internals ----------------------------------------------------------
|
|
200
|
+
|
|
201
|
+
@staticmethod
|
|
202
|
+
def _resolve_budget(
|
|
203
|
+
goal: OptimizationGoal | None, hardware: HardwareProfile | None
|
|
204
|
+
) -> float | None:
|
|
205
|
+
if goal and goal.max_vram_gb is not None:
|
|
206
|
+
return goal.max_vram_gb
|
|
207
|
+
if hardware and hardware.total_vram_gb > 0:
|
|
208
|
+
return round(hardware.total_vram_gb * 0.9, 2)
|
|
209
|
+
return None
|
|
210
|
+
|
|
211
|
+
@staticmethod
|
|
212
|
+
def _initial_assignments(
|
|
213
|
+
model: ModelProfile,
|
|
214
|
+
low_precision: str,
|
|
215
|
+
base_precision: str,
|
|
216
|
+
sensitivity: SensitivityProfile | None,
|
|
217
|
+
) -> list[PrecisionAssignment]:
|
|
218
|
+
layers = model.num_layers or 0
|
|
219
|
+
hidden = model.hidden_size or 0
|
|
220
|
+
heads = model.num_attention_heads or 0
|
|
221
|
+
kv_heads = model.num_key_value_heads or heads
|
|
222
|
+
head_dim = model.head_dim or (hidden // heads if heads else 0)
|
|
223
|
+
inter = model.intermediate_size or (4 * hidden)
|
|
224
|
+
|
|
225
|
+
attn_params = (
|
|
226
|
+
hidden * (heads * head_dim)
|
|
227
|
+
+ (heads * head_dim) * hidden
|
|
228
|
+
+ 2 * hidden * (kv_heads * head_dim)
|
|
229
|
+
)
|
|
230
|
+
gated = arch.uses_gated_mlp(model.model_type)
|
|
231
|
+
mlp_params = (3 if gated else 2) * hidden * inter
|
|
232
|
+
|
|
233
|
+
sens_lookup = sensitivity.as_lookup() if sensitivity else {}
|
|
234
|
+
|
|
235
|
+
assignments: list[PrecisionAssignment] = [
|
|
236
|
+
PrecisionAssignment(
|
|
237
|
+
module="model.embed_tokens",
|
|
238
|
+
precision=base_precision,
|
|
239
|
+
params=model.embedding_params,
|
|
240
|
+
critical=True,
|
|
241
|
+
)
|
|
242
|
+
]
|
|
243
|
+
for i in range(layers):
|
|
244
|
+
attn_mod = f"model.layers.{i}.self_attn"
|
|
245
|
+
mlp_mod = f"model.layers.{i}.mlp"
|
|
246
|
+
assignments.append(
|
|
247
|
+
PrecisionAssignment(
|
|
248
|
+
module=attn_mod,
|
|
249
|
+
precision=low_precision,
|
|
250
|
+
params=attn_params,
|
|
251
|
+
sensitivity=sens_lookup.get(attn_mod, layer_sensitivity_prior(i, layers, "attn")),
|
|
252
|
+
)
|
|
253
|
+
)
|
|
254
|
+
assignments.append(
|
|
255
|
+
PrecisionAssignment(
|
|
256
|
+
module=mlp_mod,
|
|
257
|
+
precision=low_precision,
|
|
258
|
+
params=mlp_params,
|
|
259
|
+
sensitivity=sens_lookup.get(mlp_mod, layer_sensitivity_prior(i, layers, "mlp")),
|
|
260
|
+
)
|
|
261
|
+
)
|
|
262
|
+
assignments.append(
|
|
263
|
+
PrecisionAssignment(
|
|
264
|
+
module="lm_head",
|
|
265
|
+
precision=base_precision,
|
|
266
|
+
params=model.lm_head_params,
|
|
267
|
+
critical=True,
|
|
268
|
+
)
|
|
269
|
+
)
|
|
270
|
+
return assignments
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _weight_bytes(assignments: list[PrecisionAssignment]) -> float:
|
|
274
|
+
return sum(a.weight_bytes for a in assignments)
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def _per_layer_precision(
|
|
278
|
+
assignments: list[PrecisionAssignment],
|
|
279
|
+
) -> list[tuple[int, str]]:
|
|
280
|
+
"""Reduce attn/mlp assignments to one precision per layer index (the lower of
|
|
281
|
+
the two, since that dominates the visual grouping)."""
|
|
282
|
+
per_layer: dict[int, list[str]] = {}
|
|
283
|
+
for a in assignments:
|
|
284
|
+
if ".layers." not in a.module:
|
|
285
|
+
continue
|
|
286
|
+
try:
|
|
287
|
+
idx = int(a.module.split(".layers.")[1].split(".")[0])
|
|
288
|
+
except (IndexError, ValueError):
|
|
289
|
+
continue
|
|
290
|
+
per_layer.setdefault(idx, []).append(a.precision)
|
|
291
|
+
result: list[tuple[int, str]] = []
|
|
292
|
+
for idx in sorted(per_layer):
|
|
293
|
+
precs = per_layer[idx]
|
|
294
|
+
# If mixed within a layer, show the lower precision as the layer summary.
|
|
295
|
+
chosen = min(precs, key=lambda p: _PRECISION_BITS.get(p, 4))
|
|
296
|
+
result.append((idx, chosen))
|
|
297
|
+
return result
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
"""Rank quantization candidates against an :class:`OptimizationGoal`.
|
|
2
|
+
|
|
3
|
+
The ranking is a normalized multi-objective score whose weights depend on the
|
|
4
|
+
goal's objective, plus penalties for infeasibility (over budget, quality bound
|
|
5
|
+
exceeded, poor runtime fit, not-yet-implemented). Different objectives genuinely
|
|
6
|
+
reorder the candidates, which is the whole point of the recommender.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from pydantic import BaseModel
|
|
12
|
+
|
|
13
|
+
from zeroquantz.optimization.candidate import QuantizationCandidate
|
|
14
|
+
from zeroquantz.optimization.constraints import Objective, OptimizationGoal
|
|
15
|
+
from zeroquantz.runtimes.compatibility import RuntimeCompat
|
|
16
|
+
|
|
17
|
+
# objective -> (quality_weight, memory_weight, speed_weight)
|
|
18
|
+
_WEIGHTS: dict[Objective, tuple[float, float, float]] = {
|
|
19
|
+
Objective.QUALITY: (0.60, 0.25, 0.15),
|
|
20
|
+
Objective.MEMORY: (0.25, 0.60, 0.15),
|
|
21
|
+
Objective.SPEED: (0.25, 0.20, 0.55),
|
|
22
|
+
Objective.BALANCED: (0.34, 0.33, 0.33),
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class Recommendation(BaseModel):
|
|
27
|
+
"""A scored candidate with a human-readable justification."""
|
|
28
|
+
|
|
29
|
+
candidate: QuantizationCandidate
|
|
30
|
+
score: float
|
|
31
|
+
reason: str
|
|
32
|
+
rank: int = 0
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _normalize(value: float, lo: float, hi: float, direction: str) -> float:
|
|
36
|
+
if hi <= lo:
|
|
37
|
+
return 0.5
|
|
38
|
+
t = (value - lo) / (hi - lo)
|
|
39
|
+
return (1.0 - t) if direction == "min" else t
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class Recommender:
|
|
43
|
+
"""Score and order candidates for a goal."""
|
|
44
|
+
|
|
45
|
+
@staticmethod
|
|
46
|
+
def rank(
|
|
47
|
+
candidates: list[QuantizationCandidate],
|
|
48
|
+
goal: OptimizationGoal | None = None,
|
|
49
|
+
) -> list[Recommendation]:
|
|
50
|
+
goal = goal or OptimizationGoal()
|
|
51
|
+
if not candidates:
|
|
52
|
+
return []
|
|
53
|
+
|
|
54
|
+
vrams = [c.estimated_vram_gb for c in candidates]
|
|
55
|
+
risks = [c.estimated_quality_risk for c in candidates]
|
|
56
|
+
speeds = [c.estimated_speedup for c in candidates]
|
|
57
|
+
v_lo, v_hi = min(vrams), max(vrams)
|
|
58
|
+
r_lo, r_hi = min(risks), max(risks)
|
|
59
|
+
s_lo, s_hi = min(speeds), max(speeds)
|
|
60
|
+
|
|
61
|
+
qw, mw, sw = _WEIGHTS[goal.objective]
|
|
62
|
+
|
|
63
|
+
scored: list[Recommendation] = []
|
|
64
|
+
for cand in candidates:
|
|
65
|
+
quality = _normalize(cand.estimated_quality_risk, r_lo, r_hi, "min")
|
|
66
|
+
mem = _normalize(cand.estimated_vram_gb, v_lo, v_hi, "min")
|
|
67
|
+
speed = _normalize(cand.estimated_speedup, s_lo, s_hi, "max")
|
|
68
|
+
score = qw * quality + mw * mem + sw * speed
|
|
69
|
+
score *= Recommender._penalty(cand)
|
|
70
|
+
scored.append(
|
|
71
|
+
Recommendation(
|
|
72
|
+
candidate=cand,
|
|
73
|
+
score=round(score, 4),
|
|
74
|
+
reason=Recommender._reason(cand, goal),
|
|
75
|
+
)
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
scored.sort(
|
|
79
|
+
key=lambda r: (r.candidate.feasible, r.score),
|
|
80
|
+
reverse=True,
|
|
81
|
+
)
|
|
82
|
+
for i, rec in enumerate(scored, start=1):
|
|
83
|
+
rec.rank = i
|
|
84
|
+
return scored
|
|
85
|
+
|
|
86
|
+
@staticmethod
|
|
87
|
+
def recommend(
|
|
88
|
+
candidates: list[QuantizationCandidate],
|
|
89
|
+
goal: OptimizationGoal | None = None,
|
|
90
|
+
) -> Recommendation | None:
|
|
91
|
+
"""The single best pick — preferring a feasible, runnable candidate."""
|
|
92
|
+
ranked = Recommender.rank(candidates, goal)
|
|
93
|
+
if not ranked:
|
|
94
|
+
return None
|
|
95
|
+
for rec in ranked:
|
|
96
|
+
if rec.candidate.feasible:
|
|
97
|
+
return rec
|
|
98
|
+
return ranked[0]
|
|
99
|
+
|
|
100
|
+
# ---- internals ----------------------------------------------------------
|
|
101
|
+
|
|
102
|
+
@staticmethod
|
|
103
|
+
def _penalty(cand: QuantizationCandidate) -> float:
|
|
104
|
+
penalty = 1.0
|
|
105
|
+
if not cand.fits_memory:
|
|
106
|
+
penalty *= 0.15
|
|
107
|
+
if not cand.meets_quality:
|
|
108
|
+
penalty *= 0.30
|
|
109
|
+
if not cand.implemented:
|
|
110
|
+
penalty *= 0.50
|
|
111
|
+
elif not cand.backend_available:
|
|
112
|
+
penalty *= 0.85
|
|
113
|
+
compat_mult = {
|
|
114
|
+
RuntimeCompat.COMPATIBLE: 1.0,
|
|
115
|
+
RuntimeCompat.UNKNOWN: 1.0,
|
|
116
|
+
RuntimeCompat.LIMITED: 0.9,
|
|
117
|
+
RuntimeCompat.NOT_RECOMMENDED: 0.6,
|
|
118
|
+
RuntimeCompat.UNSUPPORTED: 0.1,
|
|
119
|
+
}[cand.runtime_compat]
|
|
120
|
+
return penalty * compat_mult
|
|
121
|
+
|
|
122
|
+
@staticmethod
|
|
123
|
+
def _reason(cand: QuantizationCandidate, goal: OptimizationGoal) -> str:
|
|
124
|
+
parts: list[str] = []
|
|
125
|
+
|
|
126
|
+
if cand.method == "mixed":
|
|
127
|
+
parts.append(
|
|
128
|
+
"Keeps sensitive layers at higher precision while pushing the rest to 4-bit"
|
|
129
|
+
)
|
|
130
|
+
elif cand.weight_bits <= 4:
|
|
131
|
+
parts.append(f"4-bit weights give the smallest footprint (~{cand.estimated_size_gb:.1f} GB)")
|
|
132
|
+
elif cand.weight_bits == 8:
|
|
133
|
+
parts.append(f"8-bit weights preserve quality at ~{cand.estimated_size_gb:.1f} GB")
|
|
134
|
+
|
|
135
|
+
if goal.max_vram_gb is not None and cand.fits_memory:
|
|
136
|
+
parts.append(f"fits the {goal.max_vram_gb:g} GB budget (est. {cand.estimated_vram_gb:.1f} GB)")
|
|
137
|
+
elif goal.max_vram_gb is not None and not cand.fits_memory:
|
|
138
|
+
parts.append(f"exceeds the {goal.max_vram_gb:g} GB budget (est. {cand.estimated_vram_gb:.1f} GB)")
|
|
139
|
+
|
|
140
|
+
parts.append(f"quality risk {cand.quality_risk_label}")
|
|
141
|
+
|
|
142
|
+
if goal.runtime:
|
|
143
|
+
parts.append(f"{cand.runtime_compat.label} with {goal.runtime}")
|
|
144
|
+
|
|
145
|
+
if not cand.implemented:
|
|
146
|
+
parts.append("planned for a future release")
|
|
147
|
+
|
|
148
|
+
reason = "; ".join(parts)
|
|
149
|
+
return reason[0].upper() + reason[1:] + "." if reason else ""
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""Layer-sensitivity profiling and calibration data."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from zeroquantz.profiling.calibration import CalibrationDataset, default_calibration
|
|
6
|
+
from zeroquantz.profiling.sensitivity import (
|
|
7
|
+
LayerSensitivity,
|
|
8
|
+
SensitivityProfile,
|
|
9
|
+
SensitivityProfiler,
|
|
10
|
+
)
|
|
11
|
+
|
|
12
|
+
__all__ = [
|
|
13
|
+
"CalibrationDataset",
|
|
14
|
+
"LayerSensitivity",
|
|
15
|
+
"SensitivityProfile",
|
|
16
|
+
"SensitivityProfiler",
|
|
17
|
+
"default_calibration",
|
|
18
|
+
]
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Calibration data for sensitivity profiling and calibrated quantization.
|
|
2
|
+
|
|
3
|
+
Ships a tiny built-in prompt set so profiling works with zero external data, and
|
|
4
|
+
can load user prompts from ``.jsonl`` (one ``{"text": ...}`` or raw string per
|
|
5
|
+
line) or plain ``.txt`` (one prompt per line).
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from pydantic import BaseModel, Field
|
|
14
|
+
|
|
15
|
+
# A small, domain-diverse default calibration set (kept intentionally short).
|
|
16
|
+
_DEFAULT_PROMPTS: tuple[str, ...] = (
|
|
17
|
+
"The theory of relativity fundamentally changed our understanding of space and time.",
|
|
18
|
+
"def fibonacci(n):\n return n if n < 2 else fibonacci(n - 1) + fibonacci(n - 2)",
|
|
19
|
+
"In 1969, the Apollo 11 mission successfully landed the first humans on the Moon.",
|
|
20
|
+
"The mitochondria is often described as the powerhouse of the cell.",
|
|
21
|
+
"To be, or not to be, that is the question posed in Shakespeare's Hamlet.",
|
|
22
|
+
"A balanced diet includes proteins, carbohydrates, fats, vitamins, and minerals.",
|
|
23
|
+
"Machine learning models learn patterns from data to make predictions.",
|
|
24
|
+
"The stock market reacted sharply to the central bank's interest rate decision.",
|
|
25
|
+
"Photosynthesis converts sunlight, water, and carbon dioxide into glucose and oxygen.",
|
|
26
|
+
"She walked along the quiet beach as the sun dipped below the horizon.",
|
|
27
|
+
"The French Revolution began in 1789 and reshaped modern European politics.",
|
|
28
|
+
"Quantum computers exploit superposition and entanglement to process information.",
|
|
29
|
+
"A good API is easy to use correctly and hard to use incorrectly.",
|
|
30
|
+
"The recipe calls for two cups of flour, one egg, and a pinch of salt.",
|
|
31
|
+
"Climate change is driven largely by greenhouse gas emissions from human activity.",
|
|
32
|
+
"The ancient library of Alexandria was one of the largest of the ancient world.",
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class CalibrationDataset(BaseModel):
|
|
37
|
+
"""A named set of calibration prompts."""
|
|
38
|
+
|
|
39
|
+
name: str = "default"
|
|
40
|
+
prompts: list[str] = Field(default_factory=list)
|
|
41
|
+
|
|
42
|
+
def __len__(self) -> int:
|
|
43
|
+
return len(self.prompts)
|
|
44
|
+
|
|
45
|
+
def sample(self, n: int) -> list[str]:
|
|
46
|
+
"""First ``n`` prompts (deterministic — reproducible profiling)."""
|
|
47
|
+
return self.prompts[:n]
|
|
48
|
+
|
|
49
|
+
@classmethod
|
|
50
|
+
def from_file(cls, path: str | Path) -> CalibrationDataset:
|
|
51
|
+
p = Path(path)
|
|
52
|
+
if not p.exists():
|
|
53
|
+
raise FileNotFoundError(f"calibration file not found: {p}")
|
|
54
|
+
prompts: list[str] = []
|
|
55
|
+
if p.suffix == ".jsonl":
|
|
56
|
+
for line in p.read_text(encoding="utf-8").splitlines():
|
|
57
|
+
line = line.strip()
|
|
58
|
+
if not line:
|
|
59
|
+
continue
|
|
60
|
+
obj = json.loads(line)
|
|
61
|
+
if isinstance(obj, str):
|
|
62
|
+
prompts.append(obj)
|
|
63
|
+
elif isinstance(obj, dict):
|
|
64
|
+
prompts.append(obj.get("text") or obj.get("prompt") or "")
|
|
65
|
+
else:
|
|
66
|
+
prompts = [ln for ln in p.read_text(encoding="utf-8").splitlines() if ln.strip()]
|
|
67
|
+
return cls(name=p.stem, prompts=[pr for pr in prompts if pr])
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def default_calibration(n: int | None = None) -> CalibrationDataset:
|
|
71
|
+
prompts = list(_DEFAULT_PROMPTS)
|
|
72
|
+
if n is not None:
|
|
73
|
+
prompts = prompts[:n]
|
|
74
|
+
return CalibrationDataset(name="default", prompts=prompts)
|