zeroquantz 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- zeroquantz/__init__.py +14 -0
- zeroquantz/__main__.py +8 -0
- zeroquantz/agent/__init__.py +16 -0
- zeroquantz/agent/dispatcher.py +520 -0
- zeroquantz/agent/intents.py +46 -0
- zeroquantz/agent/parser.py +255 -0
- zeroquantz/benchmark/__init__.py +7 -0
- zeroquantz/benchmark/latency.py +66 -0
- zeroquantz/benchmark/memory.py +41 -0
- zeroquantz/benchmark/quality.py +38 -0
- zeroquantz/benchmark/runner.py +151 -0
- zeroquantz/cli/__init__.py +7 -0
- zeroquantz/cli/app.py +98 -0
- zeroquantz/cli/commands.py +459 -0
- zeroquantz/cli/interactive.py +56 -0
- zeroquantz/core/__init__.py +7 -0
- zeroquantz/core/artifacts.py +179 -0
- zeroquantz/core/context.py +127 -0
- zeroquantz/core/events.py +30 -0
- zeroquantz/core/exceptions.py +105 -0
- zeroquantz/core/session.py +202 -0
- zeroquantz/core/subenv.py +202 -0
- zeroquantz/deploy/__init__.py +25 -0
- zeroquantz/deploy/assets.py +161 -0
- zeroquantz/deploy/launcher.py +80 -0
- zeroquantz/deploy/runtime_env.py +66 -0
- zeroquantz/deploy/targets.py +154 -0
- zeroquantz/export/__init__.py +8 -0
- zeroquantz/export/exporter.py +68 -0
- zeroquantz/export/report.py +203 -0
- zeroquantz/hardware/__init__.py +15 -0
- zeroquantz/hardware/capabilities.py +152 -0
- zeroquantz/hardware/detector.py +200 -0
- zeroquantz/hardware/gpu.py +31 -0
- zeroquantz/models/__init__.py +8 -0
- zeroquantz/models/architecture.py +168 -0
- zeroquantz/models/downloader.py +161 -0
- zeroquantz/models/hf_auth.py +105 -0
- zeroquantz/models/inspector.py +249 -0
- zeroquantz/models/metadata.py +108 -0
- zeroquantz/models/search.py +71 -0
- zeroquantz/optimization/__init__.py +22 -0
- zeroquantz/optimization/candidate.py +272 -0
- zeroquantz/optimization/constraints.py +70 -0
- zeroquantz/optimization/fit.py +203 -0
- zeroquantz/optimization/pareto.py +66 -0
- zeroquantz/optimization/planner.py +297 -0
- zeroquantz/optimization/recommender.py +149 -0
- zeroquantz/profiling/__init__.py +18 -0
- zeroquantz/profiling/calibration.py +74 -0
- zeroquantz/profiling/sensitivity.py +234 -0
- zeroquantz/quantization/__init__.py +17 -0
- zeroquantz/quantization/backends/__init__.py +8 -0
- zeroquantz/quantization/backends/bitsandbytes.py +210 -0
- zeroquantz/quantization/backends/torchao.py +198 -0
- zeroquantz/quantization/base.py +136 -0
- zeroquantz/quantization/catalog.py +321 -0
- zeroquantz/quantization/config.py +106 -0
- zeroquantz/quantization/gguf_pipeline.py +210 -0
- zeroquantz/quantization/isolated.py +248 -0
- zeroquantz/quantization/memory.py +133 -0
- zeroquantz/quantization/native.py +91 -0
- zeroquantz/quantization/registry.py +101 -0
- zeroquantz/render.py +341 -0
- zeroquantz/runtimes/__init__.py +18 -0
- zeroquantz/runtimes/base.py +64 -0
- zeroquantz/runtimes/compatibility.py +91 -0
- zeroquantz/runtimes/registry.py +70 -0
- zeroquantz/runtimes/transformers.py +53 -0
- zeroquantz/runtimes/vllm.py +83 -0
- zeroquantz/tui/__init__.py +13 -0
- zeroquantz/tui/app.py +77 -0
- zeroquantz/tui/banner.py +47 -0
- zeroquantz/tui/screens/__init__.py +25 -0
- zeroquantz/tui/screens/confirm.py +41 -0
- zeroquantz/tui/screens/execute.py +194 -0
- zeroquantz/tui/screens/model_select.py +206 -0
- zeroquantz/tui/screens/plan.py +177 -0
- zeroquantz/tui/screens/quantize_select.py +272 -0
- zeroquantz/tui/screens/settings.py +219 -0
- zeroquantz/tui/screens/token.py +94 -0
- zeroquantz/tui/screens/welcome.py +128 -0
- zeroquantz/tui/screens/workspace.py +175 -0
- zeroquantz/tui/styles/app.tcss +424 -0
- zeroquantz/tui/widgets/__init__.py +9 -0
- zeroquantz/tui/widgets/chip.py +36 -0
- zeroquantz/tui/widgets/sidebar.py +107 -0
- zeroquantz/tui/widgets/status_bar.py +43 -0
- zeroquantz/utils/__init__.py +8 -0
- zeroquantz/utils/config.py +46 -0
- zeroquantz/utils/env.py +78 -0
- zeroquantz/utils/logging.py +73 -0
- zeroquantz/utils/metrics.py +98 -0
- zeroquantz/utils/paths.py +57 -0
- zeroquantz/utils/units.py +134 -0
- zeroquantz/verification/__init__.py +17 -0
- zeroquantz/verification/logits.py +55 -0
- zeroquantz/verification/report.py +186 -0
- zeroquantz/verification/weights.py +44 -0
- zeroquantz/version.py +8 -0
- zeroquantz-0.1.0.dist-info/METADATA +72 -0
- zeroquantz-0.1.0.dist-info/RECORD +105 -0
- zeroquantz-0.1.0.dist-info/WHEEL +4 -0
- zeroquantz-0.1.0.dist-info/entry_points.txt +2 -0
- zeroquantz-0.1.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
"""Production HF → quantized-GGUF pipeline — pip-only, no C/C++ toolchain.
|
|
2
|
+
|
|
3
|
+
Every stage runs in an isolated ``tools/gguf`` env (built once) and hard-fails on
|
|
4
|
+
error, so the output is a *proven* llama.cpp-loadable GGUF:
|
|
5
|
+
|
|
6
|
+
convert HF safetensors -> f16 GGUF via llama.cpp's convert_hf_to_gguf.py
|
|
7
|
+
(fetched as source, no build; all architectures + tokenizers)
|
|
8
|
+
quantize f16 GGUF -> quantized GGUF via llama-cpp-python (llama_model_quantize)
|
|
9
|
+
verify load-test via llama-cpp-python (Llama load) — the gate
|
|
10
|
+
|
|
11
|
+
llama-cpp-python ships prebuilt CPU wheels, so the whole pipeline installs with pip
|
|
12
|
+
and needs no compiler. (GPU serving can use its CUDA wheel index; quantization is
|
|
13
|
+
CPU work regardless.) ``plan()`` previews the commands without running.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import contextlib
|
|
19
|
+
from dataclasses import dataclass
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import TYPE_CHECKING
|
|
22
|
+
|
|
23
|
+
from zeroquantz.core.exceptions import QuantizationError
|
|
24
|
+
from zeroquantz.core.subenv import EnvSpec, SubEnvManager
|
|
25
|
+
from zeroquantz.quantization.native import ensure_llamacpp_src, llamacpp_src
|
|
26
|
+
from zeroquantz.utils.logging import get_logger
|
|
27
|
+
|
|
28
|
+
if TYPE_CHECKING:
|
|
29
|
+
from collections.abc import Callable
|
|
30
|
+
|
|
31
|
+
log = get_logger(__name__)
|
|
32
|
+
|
|
33
|
+
# Quantize a GGUF via llama-cpp-python's binding (handles top-level or low-level layout).
|
|
34
|
+
_QUANTIZE_SCRIPT = '''
|
|
35
|
+
import ctypes, sys
|
|
36
|
+
import llama_cpp
|
|
37
|
+
low = getattr(llama_cpp, "llama_cpp", llama_cpp)
|
|
38
|
+
inp, out, ftype = sys.argv[1], sys.argv[2], sys.argv[3]
|
|
39
|
+
const = "LLAMA_FTYPE_MOSTLY_" + ftype
|
|
40
|
+
if not hasattr(low, const):
|
|
41
|
+
sys.stderr.write("unsupported type: %s\\n" % ftype); sys.exit(2)
|
|
42
|
+
params = low.llama_model_quantize_default_params()
|
|
43
|
+
params.ftype = getattr(low, const)
|
|
44
|
+
rc = low.llama_model_quantize(inp.encode("utf-8"), out.encode("utf-8"), ctypes.byref(params))
|
|
45
|
+
sys.exit(0 if rc == 0 else 1)
|
|
46
|
+
'''
|
|
47
|
+
|
|
48
|
+
# Load-test gate: if llama.cpp can construct the model, it is compatible.
|
|
49
|
+
_VERIFY_SCRIPT = '''
|
|
50
|
+
import sys
|
|
51
|
+
from llama_cpp import Llama
|
|
52
|
+
try:
|
|
53
|
+
Llama(model_path=sys.argv[1], n_ctx=64, n_gpu_layers=0, verbose=False)
|
|
54
|
+
print("OK: loaded", sys.argv[1]); sys.exit(0)
|
|
55
|
+
except Exception as e:
|
|
56
|
+
sys.stderr.write("FAIL: %s\\n" % e); sys.exit(1)
|
|
57
|
+
'''
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass
|
|
61
|
+
class PipelineStage:
|
|
62
|
+
name: str
|
|
63
|
+
ready: bool
|
|
64
|
+
command: list[str]
|
|
65
|
+
hint: str = ""
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@dataclass
|
|
69
|
+
class GgufArtifacts:
|
|
70
|
+
out_dir: Path
|
|
71
|
+
f16_gguf: Path
|
|
72
|
+
quantized_gguf: Path
|
|
73
|
+
quant_type: str
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
class GgufPipeline:
|
|
77
|
+
"""Orchestrates convert → quantize → verify into a proven GGUF (pip-only)."""
|
|
78
|
+
|
|
79
|
+
# One isolated env holds the converter deps + llama-cpp-python. Pinned to 3.12
|
|
80
|
+
# so prebuilt wheels are available (no compiler). Built on first use.
|
|
81
|
+
TOOLS_ENV = EnvSpec(
|
|
82
|
+
kind="tools",
|
|
83
|
+
name="gguf",
|
|
84
|
+
pip=("llama-cpp-python", "gguf", "sentencepiece", "protobuf",
|
|
85
|
+
"transformers", "safetensors", "numpy", "torch"),
|
|
86
|
+
python_version="3.12",
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
def __init__(self, subenv: SubEnvManager | None = None) -> None:
|
|
90
|
+
self._envs = subenv or SubEnvManager()
|
|
91
|
+
|
|
92
|
+
# ---- paths / setup ------------------------------------------------------
|
|
93
|
+
|
|
94
|
+
def _paths(self, quant: str, out_dir: str | Path) -> tuple[Path, Path, Path]:
|
|
95
|
+
out = Path(out_dir)
|
|
96
|
+
return out, out / "model.f16.gguf", out / f"model.{quant}.gguf"
|
|
97
|
+
|
|
98
|
+
def _env_dir(self) -> Path:
|
|
99
|
+
return self._envs.env_dir(self.TOOLS_ENV)
|
|
100
|
+
|
|
101
|
+
def _tools_python(self) -> Path:
|
|
102
|
+
return self._envs.venv_python(self._env_dir())
|
|
103
|
+
|
|
104
|
+
def _write_scripts(self) -> tuple[Path, Path]:
|
|
105
|
+
env_dir = self._env_dir()
|
|
106
|
+
q = env_dir / "zq_quantize.py"
|
|
107
|
+
v = env_dir / "zq_verify.py"
|
|
108
|
+
q.write_text(_QUANTIZE_SCRIPT, encoding="utf-8")
|
|
109
|
+
v.write_text(_VERIFY_SCRIPT, encoding="utf-8")
|
|
110
|
+
return q, v
|
|
111
|
+
|
|
112
|
+
# ---- preview ------------------------------------------------------------
|
|
113
|
+
|
|
114
|
+
def plan(self, model_dir: str, quant: str, out_dir: str | Path) -> list[PipelineStage]:
|
|
115
|
+
_out, f16, quantized = self._paths(quant, out_dir)
|
|
116
|
+
src = llamacpp_src()
|
|
117
|
+
py = self._tools_python()
|
|
118
|
+
env_ready = self._envs.is_ready(self.TOOLS_ENV)
|
|
119
|
+
script = str((src / "convert_hf_to_gguf.py") if src else "convert_hf_to_gguf.py")
|
|
120
|
+
env_dir = self._env_dir()
|
|
121
|
+
return [
|
|
122
|
+
PipelineStage(
|
|
123
|
+
"convert",
|
|
124
|
+
ready=bool(src) and env_ready,
|
|
125
|
+
command=[str(py), script, model_dir, "--outfile", str(f16), "--outtype", "f16"],
|
|
126
|
+
hint="fetched via shallow git clone; tools/gguf env built on first run",
|
|
127
|
+
),
|
|
128
|
+
PipelineStage(
|
|
129
|
+
"quantize",
|
|
130
|
+
ready=env_ready,
|
|
131
|
+
command=[str(py), str(env_dir / "zq_quantize.py"), str(f16), str(quantized), quant],
|
|
132
|
+
hint="llama-cpp-python (prebuilt CPU wheel) in the tools/gguf env",
|
|
133
|
+
),
|
|
134
|
+
PipelineStage(
|
|
135
|
+
"verify",
|
|
136
|
+
ready=env_ready,
|
|
137
|
+
command=[str(py), str(env_dir / "zq_verify.py"), str(quantized)],
|
|
138
|
+
hint="llama-cpp-python load-test",
|
|
139
|
+
),
|
|
140
|
+
]
|
|
141
|
+
|
|
142
|
+
# ---- building blocks (also usable standalone) ---------------------------
|
|
143
|
+
|
|
144
|
+
def ensure_env(self, *, progress: Callable[[str, float], None] | None = None) -> Path:
|
|
145
|
+
py = self._envs.ensure(self.TOOLS_ENV, progress=progress)
|
|
146
|
+
self._write_scripts()
|
|
147
|
+
return py
|
|
148
|
+
|
|
149
|
+
def quantize(
|
|
150
|
+
self, in_gguf: str, out_gguf: str, quant: str,
|
|
151
|
+
*, progress: Callable[[str, float], None] | None = None,
|
|
152
|
+
) -> Path:
|
|
153
|
+
py = self.ensure_env(progress=progress)
|
|
154
|
+
script = self._env_dir() / "zq_quantize.py"
|
|
155
|
+
if progress:
|
|
156
|
+
progress(f"quantize -> {quant}", 0.6)
|
|
157
|
+
self._envs.run([str(py), str(script), str(in_gguf), str(out_gguf), quant])
|
|
158
|
+
out = Path(out_gguf)
|
|
159
|
+
if not out.exists():
|
|
160
|
+
raise QuantizationError("quantize produced no output GGUF.", detail=str(out))
|
|
161
|
+
return out
|
|
162
|
+
|
|
163
|
+
def verify(self, gguf: str, *, progress: Callable[[str, float], None] | None = None) -> None:
|
|
164
|
+
py = self.ensure_env(progress=progress)
|
|
165
|
+
script = self._env_dir() / "zq_verify.py"
|
|
166
|
+
if progress:
|
|
167
|
+
progress("verify (llama.cpp load-test)", 0.9)
|
|
168
|
+
self._envs.run([str(py), str(script), str(gguf)]) # raises on nonzero exit
|
|
169
|
+
|
|
170
|
+
def serve_command(self, gguf: str, port: int) -> list[str]:
|
|
171
|
+
return [str(self._tools_python()), "-m", "llama_cpp.server",
|
|
172
|
+
"--model", str(gguf), "--host", "0.0.0.0", "--port", str(port)]
|
|
173
|
+
|
|
174
|
+
# ---- full pipeline ------------------------------------------------------
|
|
175
|
+
|
|
176
|
+
def run(
|
|
177
|
+
self,
|
|
178
|
+
model_dir: str,
|
|
179
|
+
quant: str,
|
|
180
|
+
out_dir: str | Path,
|
|
181
|
+
*,
|
|
182
|
+
progress: Callable[[str, float], None] | None = None,
|
|
183
|
+
keep_f16: bool = False,
|
|
184
|
+
) -> GgufArtifacts:
|
|
185
|
+
out, f16, quantized = self._paths(quant, out_dir)
|
|
186
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
187
|
+
|
|
188
|
+
def step(msg: str, frac: float) -> None:
|
|
189
|
+
log.info("gguf-pipeline: %s", msg)
|
|
190
|
+
if progress:
|
|
191
|
+
progress(msg, frac)
|
|
192
|
+
|
|
193
|
+
step("preparing tools/gguf env (first run installs the pip deps)", 0.05)
|
|
194
|
+
py = self.ensure_env(progress=progress)
|
|
195
|
+
src = ensure_llamacpp_src(progress=progress)
|
|
196
|
+
|
|
197
|
+
step("convert: HF -> f16 GGUF", 0.2)
|
|
198
|
+
self._envs.run([str(py), str(src / "convert_hf_to_gguf.py"), model_dir,
|
|
199
|
+
"--outfile", str(f16), "--outtype", "f16"])
|
|
200
|
+
if not f16.exists():
|
|
201
|
+
raise QuantizationError("convert produced no f16 GGUF.", detail=str(f16))
|
|
202
|
+
|
|
203
|
+
self.quantize(str(f16), str(quantized), quant, progress=progress)
|
|
204
|
+
self.verify(str(quantized), progress=progress)
|
|
205
|
+
|
|
206
|
+
if not keep_f16:
|
|
207
|
+
with contextlib.suppress(OSError):
|
|
208
|
+
f16.unlink()
|
|
209
|
+
step("done", 1.0)
|
|
210
|
+
return GgufArtifacts(out_dir=out, f16_gguf=f16, quantized_gguf=quantized, quant_type=quant)
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
"""Run quantization toolchains in isolated sub-environments.
|
|
2
|
+
|
|
3
|
+
Many quantizers pin mutually-incompatible torch/transformers versions (e.g.
|
|
4
|
+
GPTQModel wants torch≥2.8 while the archived AutoGPTQ froze at 2.2.1), so they
|
|
5
|
+
cannot share one environment. Each toolchain gets a dedicated venv (built and
|
|
6
|
+
cached by :class:`~zeroquantz.core.subenv.SubEnvManager` under
|
|
7
|
+
``~/.zeroquantz/envs/quant/<family>/``); this module installs just that toolchain
|
|
8
|
+
and runs a generated quantization script as a subprocess — passing config via a
|
|
9
|
+
JSON file and reading results back from a ``manifest.json`` (rather than parsing
|
|
10
|
+
tqdm-polluted stdout).
|
|
11
|
+
|
|
12
|
+
The heavy work (multi-GB installs + quantization) happens in the child process.
|
|
13
|
+
``plan_only=True`` returns the exact env + commands + script without executing,
|
|
14
|
+
so a run can be inspected first.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import json
|
|
20
|
+
from dataclasses import dataclass, field
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import TYPE_CHECKING
|
|
23
|
+
|
|
24
|
+
from zeroquantz.core.exceptions import QuantizationError
|
|
25
|
+
from zeroquantz.core.subenv import EnvSpec, SubEnvManager, torch_index_for
|
|
26
|
+
from zeroquantz.utils.logging import get_logger
|
|
27
|
+
|
|
28
|
+
if TYPE_CHECKING:
|
|
29
|
+
from collections.abc import Callable
|
|
30
|
+
|
|
31
|
+
from zeroquantz.quantization.catalog import QuantFormat
|
|
32
|
+
|
|
33
|
+
log = get_logger(__name__)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
# ---- per-toolchain quantization scripts ------------------------------------
|
|
37
|
+
# Each reads {"model", "output", "bits", "group_size", "calibration"} from a JSON
|
|
38
|
+
# config (argv --config) and writes {"output_model_dir": ...} to --manifest.
|
|
39
|
+
|
|
40
|
+
_MANIFEST_TAIL = '''
|
|
41
|
+
import json, sys
|
|
42
|
+
from pathlib import Path
|
|
43
|
+
_manifest = sys.argv[sys.argv.index("--manifest") + 1]
|
|
44
|
+
Path(_manifest).write_text(json.dumps({"output_model_dir": cfg["output"], "status": "ok"}))
|
|
45
|
+
print("ZEROQUANTZ_DONE", cfg["output"])
|
|
46
|
+
'''
|
|
47
|
+
|
|
48
|
+
# NOTE: recipes target each library's current documented API. They are best-effort
|
|
49
|
+
# and may need updates as libraries evolve; run with plan_only to inspect first.
|
|
50
|
+
_RECIPES: dict[str, str] = {
|
|
51
|
+
"awq": '''
|
|
52
|
+
import json, sys
|
|
53
|
+
from pathlib import Path
|
|
54
|
+
cfg = json.loads(Path(sys.argv[sys.argv.index("--config") + 1]).read_text())
|
|
55
|
+
from awq import AutoAWQForCausalLM
|
|
56
|
+
from transformers import AutoTokenizer
|
|
57
|
+
model = AutoAWQForCausalLM.from_pretrained(cfg["model"], safetensors=True)
|
|
58
|
+
tok = AutoTokenizer.from_pretrained(cfg["model"], trust_remote_code=True)
|
|
59
|
+
model.quantize(tok, quant_config={
|
|
60
|
+
"w_bit": cfg.get("bits", 4), "q_group_size": cfg.get("group_size", 128),
|
|
61
|
+
"zero_point": True, "version": "GEMM",
|
|
62
|
+
})
|
|
63
|
+
model.save_quantized(cfg["output"]); tok.save_pretrained(cfg["output"])
|
|
64
|
+
''' + _MANIFEST_TAIL,
|
|
65
|
+
|
|
66
|
+
"gptq": '''
|
|
67
|
+
import json, sys
|
|
68
|
+
from pathlib import Path
|
|
69
|
+
cfg = json.loads(Path(sys.argv[sys.argv.index("--config") + 1]).read_text())
|
|
70
|
+
from gptqmodel import GPTQModel, QuantizeConfig
|
|
71
|
+
qc = QuantizeConfig(bits=cfg.get("bits", 4), group_size=cfg.get("group_size", 128))
|
|
72
|
+
model = GPTQModel.load(cfg["model"], qc)
|
|
73
|
+
model.quantize(cfg["calibration"], batch_size=1)
|
|
74
|
+
model.save(cfg["output"])
|
|
75
|
+
''' + _MANIFEST_TAIL,
|
|
76
|
+
|
|
77
|
+
"autoround": '''
|
|
78
|
+
import json, sys
|
|
79
|
+
from pathlib import Path
|
|
80
|
+
cfg = json.loads(Path(sys.argv[sys.argv.index("--config") + 1]).read_text())
|
|
81
|
+
from transformers import AutoModelForCausalLM, AutoTokenizer
|
|
82
|
+
from auto_round import AutoRound
|
|
83
|
+
model = AutoModelForCausalLM.from_pretrained(cfg["model"], torch_dtype="auto", device_map="auto")
|
|
84
|
+
tok = AutoTokenizer.from_pretrained(cfg["model"])
|
|
85
|
+
ar = AutoRound(model, tok, bits=cfg.get("bits", 4), group_size=cfg.get("group_size", 128), nsamples=128)
|
|
86
|
+
ar.quantize_and_save(cfg["output"], format="auto_gptq")
|
|
87
|
+
''' + _MANIFEST_TAIL,
|
|
88
|
+
|
|
89
|
+
"hqq": '''
|
|
90
|
+
import json, sys
|
|
91
|
+
from pathlib import Path
|
|
92
|
+
cfg = json.loads(Path(sys.argv[sys.argv.index("--config") + 1]).read_text())
|
|
93
|
+
import torch
|
|
94
|
+
from transformers import AutoModelForCausalLM, AutoTokenizer
|
|
95
|
+
from hqq.models.hf.base import AutoHQQHFModel
|
|
96
|
+
from hqq.core.quantize import BaseQuantizeConfig
|
|
97
|
+
model = AutoModelForCausalLM.from_pretrained(cfg["model"], torch_dtype=torch.float16)
|
|
98
|
+
tok = AutoTokenizer.from_pretrained(cfg["model"])
|
|
99
|
+
qcfg = BaseQuantizeConfig(nbits=cfg.get("bits", 4), group_size=cfg.get("group_size", 64))
|
|
100
|
+
AutoHQQHFModel.quantize_model(model, quant_config=qcfg, compute_dtype=torch.float16, device="cuda")
|
|
101
|
+
AutoHQQHFModel.save_quantized(model, cfg["output"]); tok.save_pretrained(cfg["output"])
|
|
102
|
+
''' + _MANIFEST_TAIL,
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
@dataclass
|
|
107
|
+
class IsolatedPlan:
|
|
108
|
+
"""A previewable plan for an isolated quantization run."""
|
|
109
|
+
|
|
110
|
+
family: str
|
|
111
|
+
env_dir: Path
|
|
112
|
+
python: Path
|
|
113
|
+
pip_specs: tuple[str, ...]
|
|
114
|
+
torch_index: str | None
|
|
115
|
+
script_path: Path
|
|
116
|
+
commands: list[str] = field(default_factory=list)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
@dataclass
|
|
120
|
+
class IsolatedResult:
|
|
121
|
+
family: str
|
|
122
|
+
output_dir: str
|
|
123
|
+
env_dir: Path
|
|
124
|
+
elapsed_s: float = 0.0
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
class IsolatedEnvRunner:
|
|
128
|
+
"""Create and drive per-toolchain isolated environments (kind='quant')."""
|
|
129
|
+
|
|
130
|
+
def __init__(self, root: Path | None = None) -> None:
|
|
131
|
+
self._envs = SubEnvManager(root=root)
|
|
132
|
+
|
|
133
|
+
# ---- spec / helpers -----------------------------------------------------
|
|
134
|
+
|
|
135
|
+
def _spec(self, fmt: QuantFormat, cuda_version: str | None) -> EnvSpec:
|
|
136
|
+
return EnvSpec(
|
|
137
|
+
kind="quant", name=fmt.family, pip=tuple(fmt.pip),
|
|
138
|
+
torch_index=torch_index_for(cuda_version),
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
def supports(self, family: str) -> bool:
|
|
142
|
+
return family in _RECIPES
|
|
143
|
+
|
|
144
|
+
@staticmethod
|
|
145
|
+
def venv_python(env_dir: Path) -> Path:
|
|
146
|
+
return SubEnvManager.venv_python(env_dir)
|
|
147
|
+
|
|
148
|
+
# ---- planning -----------------------------------------------------------
|
|
149
|
+
|
|
150
|
+
def plan(self, fmt: QuantFormat, *, cuda_version: str | None = None) -> IsolatedPlan:
|
|
151
|
+
if not self.supports(fmt.family):
|
|
152
|
+
raise QuantizationError(
|
|
153
|
+
f"No isolated quantization recipe for '{fmt.family}'.",
|
|
154
|
+
detail="ZeroQuantz can plan/estimate this format but not run it yet.",
|
|
155
|
+
suggestions=["/recommend", "pick an in-env or supported isolated format"],
|
|
156
|
+
)
|
|
157
|
+
spec = self._spec(fmt, cuda_version)
|
|
158
|
+
base = self._envs.plan(spec)
|
|
159
|
+
commands = [*base.commands,
|
|
160
|
+
f"{base.python} quantize.py --config run_cfg.json --manifest manifest.json"]
|
|
161
|
+
return IsolatedPlan(
|
|
162
|
+
family=fmt.family, env_dir=base.env_dir, python=base.python, pip_specs=fmt.pip,
|
|
163
|
+
torch_index=spec.torch_index, script_path=base.env_dir / "quantize.py",
|
|
164
|
+
commands=commands,
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
# ---- environment build --------------------------------------------------
|
|
168
|
+
|
|
169
|
+
def ensure_env(
|
|
170
|
+
self,
|
|
171
|
+
fmt: QuantFormat,
|
|
172
|
+
*,
|
|
173
|
+
cuda_version: str | None = None,
|
|
174
|
+
progress: Callable[[str, float], None] | None = None,
|
|
175
|
+
) -> Path:
|
|
176
|
+
return self._envs.ensure(self._spec(fmt, cuda_version), progress=progress)
|
|
177
|
+
|
|
178
|
+
# ---- execution ----------------------------------------------------------
|
|
179
|
+
|
|
180
|
+
def quantize(
|
|
181
|
+
self,
|
|
182
|
+
fmt: QuantFormat,
|
|
183
|
+
model_ref: str,
|
|
184
|
+
output_dir: str,
|
|
185
|
+
*,
|
|
186
|
+
bits: int | None = None,
|
|
187
|
+
group_size: int = 128,
|
|
188
|
+
calibration: list[str] | None = None,
|
|
189
|
+
cuda_version: str | None = None,
|
|
190
|
+
plan_only: bool = False,
|
|
191
|
+
progress: Callable[[str, float], None] | None = None,
|
|
192
|
+
) -> IsolatedResult | IsolatedPlan:
|
|
193
|
+
if not self.supports(fmt.family):
|
|
194
|
+
raise QuantizationError(
|
|
195
|
+
f"No isolated quantization recipe for '{fmt.family}'.",
|
|
196
|
+
detail=f"'{fmt.label}' is produced by {fmt.produced_by}; "
|
|
197
|
+
"ZeroQuantz catalogs/estimates it but does not run it automatically yet.",
|
|
198
|
+
)
|
|
199
|
+
if plan_only:
|
|
200
|
+
return self.plan(fmt, cuda_version=cuda_version)
|
|
201
|
+
|
|
202
|
+
import time
|
|
203
|
+
|
|
204
|
+
spec = self._spec(fmt, cuda_version)
|
|
205
|
+
py = self.ensure_env(fmt, cuda_version=cuda_version, progress=progress)
|
|
206
|
+
env_dir = self._envs.env_dir(spec)
|
|
207
|
+
script_path = env_dir / "quantize.py"
|
|
208
|
+
script_path.write_text(_RECIPES[fmt.family], encoding="utf-8")
|
|
209
|
+
config = {
|
|
210
|
+
"model": model_ref,
|
|
211
|
+
"output": str(Path(output_dir).resolve()),
|
|
212
|
+
"bits": bits or max(2, round(fmt.bits_per_weight)),
|
|
213
|
+
"group_size": group_size,
|
|
214
|
+
"calibration": calibration or _default_calibration(),
|
|
215
|
+
}
|
|
216
|
+
cfg_path = env_dir / "run_cfg.json"
|
|
217
|
+
cfg_path.write_text(json.dumps(config), encoding="utf-8")
|
|
218
|
+
manifest_path = env_dir / "manifest.json"
|
|
219
|
+
manifest_path.unlink(missing_ok=True)
|
|
220
|
+
|
|
221
|
+
if progress:
|
|
222
|
+
progress(f"quantizing with {fmt.family}", 0.6)
|
|
223
|
+
started = time.perf_counter()
|
|
224
|
+
self._envs.run(
|
|
225
|
+
[str(py), "-u", str(script_path), "--config", str(cfg_path),
|
|
226
|
+
"--manifest", str(manifest_path)],
|
|
227
|
+
progress=progress,
|
|
228
|
+
)
|
|
229
|
+
elapsed = time.perf_counter() - started
|
|
230
|
+
|
|
231
|
+
if not manifest_path.exists():
|
|
232
|
+
raise QuantizationError(
|
|
233
|
+
f"Isolated {fmt.family} run finished without producing a manifest.",
|
|
234
|
+
detail="The quantization script did not complete; check the logs.",
|
|
235
|
+
)
|
|
236
|
+
manifest = json.loads(manifest_path.read_text())
|
|
237
|
+
if progress:
|
|
238
|
+
progress("done", 1.0)
|
|
239
|
+
return IsolatedResult(
|
|
240
|
+
family=fmt.family, output_dir=manifest["output_model_dir"], env_dir=env_dir,
|
|
241
|
+
elapsed_s=elapsed,
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _default_calibration() -> list[str]:
|
|
246
|
+
from zeroquantz.profiling.calibration import default_calibration
|
|
247
|
+
|
|
248
|
+
return default_calibration().prompts
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"""Heuristic memory estimation shared by backends and the candidate generator.
|
|
2
|
+
|
|
3
|
+
Everything here is an *estimate* derived from parameter counts and architecture
|
|
4
|
+
dimensions — never a measurement. The candidate generator labels it as such, and
|
|
5
|
+
the benchmark subsystem is what produces measured numbers.
|
|
6
|
+
|
|
7
|
+
Model of VRAM at inference time:
|
|
8
|
+
|
|
9
|
+
vram = quantized_weights + cuda_context + kv_cache + activations
|
|
10
|
+
|
|
11
|
+
* ``quantized_weights`` — a two-bucket model: parameters a weight-only backend
|
|
12
|
+
actually quantizes (linear layers) at the target precision, plus the
|
|
13
|
+
parameters it keeps at compute precision (embeddings, and the untied LM head).
|
|
14
|
+
* ``cuda_context`` — a fixed allocator/context overhead.
|
|
15
|
+
* ``kv_cache`` — grows with sequence length and (grouped) KV heads.
|
|
16
|
+
* ``activations`` — a modest term proportional to hidden size and sequence.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from typing import TYPE_CHECKING
|
|
22
|
+
|
|
23
|
+
from zeroquantz.utils import units
|
|
24
|
+
|
|
25
|
+
if TYPE_CHECKING:
|
|
26
|
+
from zeroquantz.models.metadata import ModelProfile
|
|
27
|
+
from zeroquantz.quantization.config import QuantizationConfig
|
|
28
|
+
|
|
29
|
+
# Fixed CUDA context / allocator overhead, in GiB. Realistic for a single device.
|
|
30
|
+
CUDA_CONTEXT_GIB = 0.6
|
|
31
|
+
# Activation memory is roughly proportional to hidden*seq*batch; this multiplies it.
|
|
32
|
+
ACTIVATION_FACTOR = 2.0
|
|
33
|
+
# Default profiling shape used for the estimate.
|
|
34
|
+
DEFAULT_SEQ_LEN = 2048
|
|
35
|
+
DEFAULT_BATCH = 1
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def estimate_weight_bytes(model: ModelProfile, config: QuantizationConfig) -> float:
|
|
39
|
+
"""Estimated bytes of *weights* after applying ``config`` to ``model``."""
|
|
40
|
+
base_bytes = units.bytes_per_param(config.compute_dtype)
|
|
41
|
+
non_quant = model.non_quantizable_params * base_bytes
|
|
42
|
+
q_params = model.quantizable_params
|
|
43
|
+
|
|
44
|
+
if config.method == "mixed":
|
|
45
|
+
low = config.extra.get("low_precision", "int4")
|
|
46
|
+
high = config.extra.get("high_precision", "int8")
|
|
47
|
+
high_fraction = float(config.extra.get("high_fraction", 0.25))
|
|
48
|
+
per_param = high_fraction * units.bytes_per_param(high) + (
|
|
49
|
+
1 - high_fraction
|
|
50
|
+
) * units.bytes_per_param(low)
|
|
51
|
+
quant = q_params * per_param
|
|
52
|
+
else:
|
|
53
|
+
quant = q_params * units.bytes_per_param(config.weight_precision)
|
|
54
|
+
|
|
55
|
+
return non_quant + quant
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def estimate_kv_cache_bytes(
|
|
59
|
+
model: ModelProfile,
|
|
60
|
+
*,
|
|
61
|
+
seq_len: int = DEFAULT_SEQ_LEN,
|
|
62
|
+
batch: int = DEFAULT_BATCH,
|
|
63
|
+
cache_dtype: str = "float16",
|
|
64
|
+
) -> float:
|
|
65
|
+
"""KV-cache bytes for a given sequence length and batch."""
|
|
66
|
+
layers = model.num_layers
|
|
67
|
+
kv_heads = model.num_key_value_heads or model.num_attention_heads
|
|
68
|
+
head_dim = model.head_dim
|
|
69
|
+
if not (layers and kv_heads and head_dim):
|
|
70
|
+
return 0.0
|
|
71
|
+
per_token = 2 * layers * kv_heads * head_dim * units.bytes_per_param(cache_dtype)
|
|
72
|
+
return per_token * seq_len * batch
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def estimate_activation_bytes(
|
|
76
|
+
model: ModelProfile,
|
|
77
|
+
*,
|
|
78
|
+
seq_len: int = DEFAULT_SEQ_LEN,
|
|
79
|
+
batch: int = DEFAULT_BATCH,
|
|
80
|
+
compute_dtype: str = "bfloat16",
|
|
81
|
+
) -> float:
|
|
82
|
+
"""A coarse activation-memory estimate."""
|
|
83
|
+
hidden = model.hidden_size
|
|
84
|
+
if not hidden:
|
|
85
|
+
return 0.0
|
|
86
|
+
return hidden * seq_len * batch * units.bytes_per_param(compute_dtype) * ACTIVATION_FACTOR
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def estimate_runtime_overhead_bytes(
|
|
90
|
+
model: ModelProfile,
|
|
91
|
+
*,
|
|
92
|
+
seq_len: int = DEFAULT_SEQ_LEN,
|
|
93
|
+
batch: int = DEFAULT_BATCH,
|
|
94
|
+
compute_dtype: str = "bfloat16",
|
|
95
|
+
) -> float:
|
|
96
|
+
context = units.gb_to_bytes(CUDA_CONTEXT_GIB)
|
|
97
|
+
kv = estimate_kv_cache_bytes(model, seq_len=seq_len, batch=batch)
|
|
98
|
+
act = estimate_activation_bytes(
|
|
99
|
+
model, seq_len=seq_len, batch=batch, compute_dtype=compute_dtype
|
|
100
|
+
)
|
|
101
|
+
return context + kv + act
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def estimate_vram_bytes(
|
|
105
|
+
model: ModelProfile,
|
|
106
|
+
config: QuantizationConfig,
|
|
107
|
+
*,
|
|
108
|
+
seq_len: int = DEFAULT_SEQ_LEN,
|
|
109
|
+
batch: int = DEFAULT_BATCH,
|
|
110
|
+
) -> float:
|
|
111
|
+
"""Total estimated VRAM to *run* ``model`` under ``config``."""
|
|
112
|
+
weights = estimate_weight_bytes(model, config)
|
|
113
|
+
overhead = estimate_runtime_overhead_bytes(
|
|
114
|
+
model, seq_len=seq_len, batch=batch, compute_dtype=config.compute_dtype
|
|
115
|
+
)
|
|
116
|
+
return weights + overhead
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def estimate_weight_gb(model: ModelProfile, config: QuantizationConfig) -> float:
|
|
120
|
+
return round(units.bytes_to_gb(estimate_weight_bytes(model, config)), 2)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def estimate_vram_gb(
|
|
124
|
+
model: ModelProfile,
|
|
125
|
+
config: QuantizationConfig,
|
|
126
|
+
*,
|
|
127
|
+
seq_len: int = DEFAULT_SEQ_LEN,
|
|
128
|
+
batch: int = DEFAULT_BATCH,
|
|
129
|
+
) -> float:
|
|
130
|
+
return round(
|
|
131
|
+
units.bytes_to_gb(estimate_vram_bytes(model, config, seq_len=seq_len, batch=batch)),
|
|
132
|
+
2,
|
|
133
|
+
)
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""Locate ZeroQuantz's optional Rust binary and the llama.cpp converter source.
|
|
2
|
+
|
|
3
|
+
The GGUF pipeline is otherwise pip-only (llama-cpp-python); the only native binary
|
|
4
|
+
is the experimental Rust converter, and llama.cpp is needed only as *source* for its
|
|
5
|
+
``convert_hf_to_gguf.py`` script (no compilation).
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import shutil
|
|
11
|
+
import subprocess
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import TYPE_CHECKING
|
|
14
|
+
|
|
15
|
+
from zeroquantz.core.exceptions import ZeroQuantzError
|
|
16
|
+
from zeroquantz.utils.paths import paths
|
|
17
|
+
|
|
18
|
+
if TYPE_CHECKING:
|
|
19
|
+
from collections.abc import Callable
|
|
20
|
+
|
|
21
|
+
_LLAMACPP_REPO = "https://github.com/ggml-org/llama.cpp.git"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def repo_root() -> Path:
|
|
25
|
+
"""The repository root (parent of the installed ``zeroquantz`` package)."""
|
|
26
|
+
import zeroquantz
|
|
27
|
+
|
|
28
|
+
return Path(zeroquantz.__file__).resolve().parents[1]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def find_native_binary(name: str) -> Path | None:
|
|
32
|
+
"""Return the path to a built Rust binary (on PATH or ./target/release), or None."""
|
|
33
|
+
on_path = shutil.which(name)
|
|
34
|
+
if on_path:
|
|
35
|
+
return Path(on_path)
|
|
36
|
+
base = repo_root() / "target" / "release"
|
|
37
|
+
for exe in (f"{name}.exe", name):
|
|
38
|
+
if (base / exe).exists():
|
|
39
|
+
return base / exe
|
|
40
|
+
return None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def llamacpp_src() -> Path | None:
|
|
44
|
+
"""Locate a llama.cpp source tree containing ``convert_hf_to_gguf.py``.
|
|
45
|
+
|
|
46
|
+
Prefers ``$ZEROQUANTZ_LLAMACPP_SRC``, then the shallow clone ZeroQuantz manages at
|
|
47
|
+
``~/.zeroquantz/llama.cpp``.
|
|
48
|
+
"""
|
|
49
|
+
import os
|
|
50
|
+
|
|
51
|
+
env = os.environ.get("ZEROQUANTZ_LLAMACPP_SRC")
|
|
52
|
+
if env and (Path(env) / "convert_hf_to_gguf.py").exists():
|
|
53
|
+
return Path(env)
|
|
54
|
+
managed = paths().home / "llama.cpp"
|
|
55
|
+
if (managed / "convert_hf_to_gguf.py").exists():
|
|
56
|
+
return managed
|
|
57
|
+
return None
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def ensure_llamacpp_src(progress: Callable[[str, float], None] | None = None) -> Path:
|
|
61
|
+
"""Return a llama.cpp source tree, shallow-cloning it if needed (no build).
|
|
62
|
+
|
|
63
|
+
Only the Python converter (``convert_hf_to_gguf.py``) is used from it, so this is
|
|
64
|
+
a source checkout, not a compile — needs ``git`` but no C/C++ toolchain.
|
|
65
|
+
"""
|
|
66
|
+
existing = llamacpp_src()
|
|
67
|
+
if existing is not None:
|
|
68
|
+
return existing
|
|
69
|
+
if shutil.which("git") is None:
|
|
70
|
+
raise ZeroQuantzError(
|
|
71
|
+
"git is required to fetch llama.cpp's converter script.",
|
|
72
|
+
suggestions=[
|
|
73
|
+
"install git, or",
|
|
74
|
+
"set ZEROQUANTZ_LLAMACPP_SRC=/path/to/a/llama.cpp/checkout",
|
|
75
|
+
],
|
|
76
|
+
)
|
|
77
|
+
dest = paths().ensure().home / "llama.cpp"
|
|
78
|
+
if progress:
|
|
79
|
+
progress("fetching llama.cpp converter (shallow clone, no build)", 0.02)
|
|
80
|
+
try:
|
|
81
|
+
subprocess.run(
|
|
82
|
+
["git", "clone", "--depth", "1", _LLAMACPP_REPO, str(dest)],
|
|
83
|
+
check=True, capture_output=True, text=True,
|
|
84
|
+
)
|
|
85
|
+
except subprocess.CalledProcessError as exc:
|
|
86
|
+
raise ZeroQuantzError(
|
|
87
|
+
"Failed to clone llama.cpp.", detail=(exc.stderr or "").strip()[:500]
|
|
88
|
+
) from exc
|
|
89
|
+
if not (dest / "convert_hf_to_gguf.py").exists():
|
|
90
|
+
raise ZeroQuantzError(f"cloned llama.cpp but convert_hf_to_gguf.py is missing in {dest}.")
|
|
91
|
+
return dest
|