zeroquantz 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. zeroquantz/__init__.py +14 -0
  2. zeroquantz/__main__.py +8 -0
  3. zeroquantz/agent/__init__.py +16 -0
  4. zeroquantz/agent/dispatcher.py +520 -0
  5. zeroquantz/agent/intents.py +46 -0
  6. zeroquantz/agent/parser.py +255 -0
  7. zeroquantz/benchmark/__init__.py +7 -0
  8. zeroquantz/benchmark/latency.py +66 -0
  9. zeroquantz/benchmark/memory.py +41 -0
  10. zeroquantz/benchmark/quality.py +38 -0
  11. zeroquantz/benchmark/runner.py +151 -0
  12. zeroquantz/cli/__init__.py +7 -0
  13. zeroquantz/cli/app.py +98 -0
  14. zeroquantz/cli/commands.py +459 -0
  15. zeroquantz/cli/interactive.py +56 -0
  16. zeroquantz/core/__init__.py +7 -0
  17. zeroquantz/core/artifacts.py +179 -0
  18. zeroquantz/core/context.py +127 -0
  19. zeroquantz/core/events.py +30 -0
  20. zeroquantz/core/exceptions.py +105 -0
  21. zeroquantz/core/session.py +202 -0
  22. zeroquantz/core/subenv.py +202 -0
  23. zeroquantz/deploy/__init__.py +25 -0
  24. zeroquantz/deploy/assets.py +161 -0
  25. zeroquantz/deploy/launcher.py +80 -0
  26. zeroquantz/deploy/runtime_env.py +66 -0
  27. zeroquantz/deploy/targets.py +154 -0
  28. zeroquantz/export/__init__.py +8 -0
  29. zeroquantz/export/exporter.py +68 -0
  30. zeroquantz/export/report.py +203 -0
  31. zeroquantz/hardware/__init__.py +15 -0
  32. zeroquantz/hardware/capabilities.py +152 -0
  33. zeroquantz/hardware/detector.py +200 -0
  34. zeroquantz/hardware/gpu.py +31 -0
  35. zeroquantz/models/__init__.py +8 -0
  36. zeroquantz/models/architecture.py +168 -0
  37. zeroquantz/models/downloader.py +161 -0
  38. zeroquantz/models/hf_auth.py +105 -0
  39. zeroquantz/models/inspector.py +249 -0
  40. zeroquantz/models/metadata.py +108 -0
  41. zeroquantz/models/search.py +71 -0
  42. zeroquantz/optimization/__init__.py +22 -0
  43. zeroquantz/optimization/candidate.py +272 -0
  44. zeroquantz/optimization/constraints.py +70 -0
  45. zeroquantz/optimization/fit.py +203 -0
  46. zeroquantz/optimization/pareto.py +66 -0
  47. zeroquantz/optimization/planner.py +297 -0
  48. zeroquantz/optimization/recommender.py +149 -0
  49. zeroquantz/profiling/__init__.py +18 -0
  50. zeroquantz/profiling/calibration.py +74 -0
  51. zeroquantz/profiling/sensitivity.py +234 -0
  52. zeroquantz/quantization/__init__.py +17 -0
  53. zeroquantz/quantization/backends/__init__.py +8 -0
  54. zeroquantz/quantization/backends/bitsandbytes.py +210 -0
  55. zeroquantz/quantization/backends/torchao.py +198 -0
  56. zeroquantz/quantization/base.py +136 -0
  57. zeroquantz/quantization/catalog.py +321 -0
  58. zeroquantz/quantization/config.py +106 -0
  59. zeroquantz/quantization/gguf_pipeline.py +210 -0
  60. zeroquantz/quantization/isolated.py +248 -0
  61. zeroquantz/quantization/memory.py +133 -0
  62. zeroquantz/quantization/native.py +91 -0
  63. zeroquantz/quantization/registry.py +101 -0
  64. zeroquantz/render.py +341 -0
  65. zeroquantz/runtimes/__init__.py +18 -0
  66. zeroquantz/runtimes/base.py +64 -0
  67. zeroquantz/runtimes/compatibility.py +91 -0
  68. zeroquantz/runtimes/registry.py +70 -0
  69. zeroquantz/runtimes/transformers.py +53 -0
  70. zeroquantz/runtimes/vllm.py +83 -0
  71. zeroquantz/tui/__init__.py +13 -0
  72. zeroquantz/tui/app.py +77 -0
  73. zeroquantz/tui/banner.py +47 -0
  74. zeroquantz/tui/screens/__init__.py +25 -0
  75. zeroquantz/tui/screens/confirm.py +41 -0
  76. zeroquantz/tui/screens/execute.py +194 -0
  77. zeroquantz/tui/screens/model_select.py +206 -0
  78. zeroquantz/tui/screens/plan.py +177 -0
  79. zeroquantz/tui/screens/quantize_select.py +272 -0
  80. zeroquantz/tui/screens/settings.py +219 -0
  81. zeroquantz/tui/screens/token.py +94 -0
  82. zeroquantz/tui/screens/welcome.py +128 -0
  83. zeroquantz/tui/screens/workspace.py +175 -0
  84. zeroquantz/tui/styles/app.tcss +424 -0
  85. zeroquantz/tui/widgets/__init__.py +9 -0
  86. zeroquantz/tui/widgets/chip.py +36 -0
  87. zeroquantz/tui/widgets/sidebar.py +107 -0
  88. zeroquantz/tui/widgets/status_bar.py +43 -0
  89. zeroquantz/utils/__init__.py +8 -0
  90. zeroquantz/utils/config.py +46 -0
  91. zeroquantz/utils/env.py +78 -0
  92. zeroquantz/utils/logging.py +73 -0
  93. zeroquantz/utils/metrics.py +98 -0
  94. zeroquantz/utils/paths.py +57 -0
  95. zeroquantz/utils/units.py +134 -0
  96. zeroquantz/verification/__init__.py +17 -0
  97. zeroquantz/verification/logits.py +55 -0
  98. zeroquantz/verification/report.py +186 -0
  99. zeroquantz/verification/weights.py +44 -0
  100. zeroquantz/version.py +8 -0
  101. zeroquantz-0.1.0.dist-info/METADATA +72 -0
  102. zeroquantz-0.1.0.dist-info/RECORD +105 -0
  103. zeroquantz-0.1.0.dist-info/WHEEL +4 -0
  104. zeroquantz-0.1.0.dist-info/entry_points.txt +2 -0
  105. zeroquantz-0.1.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,210 @@
1
+ """Production HF → quantized-GGUF pipeline — pip-only, no C/C++ toolchain.
2
+
3
+ Every stage runs in an isolated ``tools/gguf`` env (built once) and hard-fails on
4
+ error, so the output is a *proven* llama.cpp-loadable GGUF:
5
+
6
+ convert HF safetensors -> f16 GGUF via llama.cpp's convert_hf_to_gguf.py
7
+ (fetched as source, no build; all architectures + tokenizers)
8
+ quantize f16 GGUF -> quantized GGUF via llama-cpp-python (llama_model_quantize)
9
+ verify load-test via llama-cpp-python (Llama load) — the gate
10
+
11
+ llama-cpp-python ships prebuilt CPU wheels, so the whole pipeline installs with pip
12
+ and needs no compiler. (GPU serving can use its CUDA wheel index; quantization is
13
+ CPU work regardless.) ``plan()`` previews the commands without running.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import contextlib
19
+ from dataclasses import dataclass
20
+ from pathlib import Path
21
+ from typing import TYPE_CHECKING
22
+
23
+ from zeroquantz.core.exceptions import QuantizationError
24
+ from zeroquantz.core.subenv import EnvSpec, SubEnvManager
25
+ from zeroquantz.quantization.native import ensure_llamacpp_src, llamacpp_src
26
+ from zeroquantz.utils.logging import get_logger
27
+
28
+ if TYPE_CHECKING:
29
+ from collections.abc import Callable
30
+
31
+ log = get_logger(__name__)
32
+
33
+ # Quantize a GGUF via llama-cpp-python's binding (handles top-level or low-level layout).
34
+ _QUANTIZE_SCRIPT = '''
35
+ import ctypes, sys
36
+ import llama_cpp
37
+ low = getattr(llama_cpp, "llama_cpp", llama_cpp)
38
+ inp, out, ftype = sys.argv[1], sys.argv[2], sys.argv[3]
39
+ const = "LLAMA_FTYPE_MOSTLY_" + ftype
40
+ if not hasattr(low, const):
41
+ sys.stderr.write("unsupported type: %s\\n" % ftype); sys.exit(2)
42
+ params = low.llama_model_quantize_default_params()
43
+ params.ftype = getattr(low, const)
44
+ rc = low.llama_model_quantize(inp.encode("utf-8"), out.encode("utf-8"), ctypes.byref(params))
45
+ sys.exit(0 if rc == 0 else 1)
46
+ '''
47
+
48
+ # Load-test gate: if llama.cpp can construct the model, it is compatible.
49
+ _VERIFY_SCRIPT = '''
50
+ import sys
51
+ from llama_cpp import Llama
52
+ try:
53
+ Llama(model_path=sys.argv[1], n_ctx=64, n_gpu_layers=0, verbose=False)
54
+ print("OK: loaded", sys.argv[1]); sys.exit(0)
55
+ except Exception as e:
56
+ sys.stderr.write("FAIL: %s\\n" % e); sys.exit(1)
57
+ '''
58
+
59
+
60
+ @dataclass
61
+ class PipelineStage:
62
+ name: str
63
+ ready: bool
64
+ command: list[str]
65
+ hint: str = ""
66
+
67
+
68
+ @dataclass
69
+ class GgufArtifacts:
70
+ out_dir: Path
71
+ f16_gguf: Path
72
+ quantized_gguf: Path
73
+ quant_type: str
74
+
75
+
76
+ class GgufPipeline:
77
+ """Orchestrates convert → quantize → verify into a proven GGUF (pip-only)."""
78
+
79
+ # One isolated env holds the converter deps + llama-cpp-python. Pinned to 3.12
80
+ # so prebuilt wheels are available (no compiler). Built on first use.
81
+ TOOLS_ENV = EnvSpec(
82
+ kind="tools",
83
+ name="gguf",
84
+ pip=("llama-cpp-python", "gguf", "sentencepiece", "protobuf",
85
+ "transformers", "safetensors", "numpy", "torch"),
86
+ python_version="3.12",
87
+ )
88
+
89
+ def __init__(self, subenv: SubEnvManager | None = None) -> None:
90
+ self._envs = subenv or SubEnvManager()
91
+
92
+ # ---- paths / setup ------------------------------------------------------
93
+
94
+ def _paths(self, quant: str, out_dir: str | Path) -> tuple[Path, Path, Path]:
95
+ out = Path(out_dir)
96
+ return out, out / "model.f16.gguf", out / f"model.{quant}.gguf"
97
+
98
+ def _env_dir(self) -> Path:
99
+ return self._envs.env_dir(self.TOOLS_ENV)
100
+
101
+ def _tools_python(self) -> Path:
102
+ return self._envs.venv_python(self._env_dir())
103
+
104
+ def _write_scripts(self) -> tuple[Path, Path]:
105
+ env_dir = self._env_dir()
106
+ q = env_dir / "zq_quantize.py"
107
+ v = env_dir / "zq_verify.py"
108
+ q.write_text(_QUANTIZE_SCRIPT, encoding="utf-8")
109
+ v.write_text(_VERIFY_SCRIPT, encoding="utf-8")
110
+ return q, v
111
+
112
+ # ---- preview ------------------------------------------------------------
113
+
114
+ def plan(self, model_dir: str, quant: str, out_dir: str | Path) -> list[PipelineStage]:
115
+ _out, f16, quantized = self._paths(quant, out_dir)
116
+ src = llamacpp_src()
117
+ py = self._tools_python()
118
+ env_ready = self._envs.is_ready(self.TOOLS_ENV)
119
+ script = str((src / "convert_hf_to_gguf.py") if src else "convert_hf_to_gguf.py")
120
+ env_dir = self._env_dir()
121
+ return [
122
+ PipelineStage(
123
+ "convert",
124
+ ready=bool(src) and env_ready,
125
+ command=[str(py), script, model_dir, "--outfile", str(f16), "--outtype", "f16"],
126
+ hint="fetched via shallow git clone; tools/gguf env built on first run",
127
+ ),
128
+ PipelineStage(
129
+ "quantize",
130
+ ready=env_ready,
131
+ command=[str(py), str(env_dir / "zq_quantize.py"), str(f16), str(quantized), quant],
132
+ hint="llama-cpp-python (prebuilt CPU wheel) in the tools/gguf env",
133
+ ),
134
+ PipelineStage(
135
+ "verify",
136
+ ready=env_ready,
137
+ command=[str(py), str(env_dir / "zq_verify.py"), str(quantized)],
138
+ hint="llama-cpp-python load-test",
139
+ ),
140
+ ]
141
+
142
+ # ---- building blocks (also usable standalone) ---------------------------
143
+
144
+ def ensure_env(self, *, progress: Callable[[str, float], None] | None = None) -> Path:
145
+ py = self._envs.ensure(self.TOOLS_ENV, progress=progress)
146
+ self._write_scripts()
147
+ return py
148
+
149
+ def quantize(
150
+ self, in_gguf: str, out_gguf: str, quant: str,
151
+ *, progress: Callable[[str, float], None] | None = None,
152
+ ) -> Path:
153
+ py = self.ensure_env(progress=progress)
154
+ script = self._env_dir() / "zq_quantize.py"
155
+ if progress:
156
+ progress(f"quantize -> {quant}", 0.6)
157
+ self._envs.run([str(py), str(script), str(in_gguf), str(out_gguf), quant])
158
+ out = Path(out_gguf)
159
+ if not out.exists():
160
+ raise QuantizationError("quantize produced no output GGUF.", detail=str(out))
161
+ return out
162
+
163
+ def verify(self, gguf: str, *, progress: Callable[[str, float], None] | None = None) -> None:
164
+ py = self.ensure_env(progress=progress)
165
+ script = self._env_dir() / "zq_verify.py"
166
+ if progress:
167
+ progress("verify (llama.cpp load-test)", 0.9)
168
+ self._envs.run([str(py), str(script), str(gguf)]) # raises on nonzero exit
169
+
170
+ def serve_command(self, gguf: str, port: int) -> list[str]:
171
+ return [str(self._tools_python()), "-m", "llama_cpp.server",
172
+ "--model", str(gguf), "--host", "0.0.0.0", "--port", str(port)]
173
+
174
+ # ---- full pipeline ------------------------------------------------------
175
+
176
+ def run(
177
+ self,
178
+ model_dir: str,
179
+ quant: str,
180
+ out_dir: str | Path,
181
+ *,
182
+ progress: Callable[[str, float], None] | None = None,
183
+ keep_f16: bool = False,
184
+ ) -> GgufArtifacts:
185
+ out, f16, quantized = self._paths(quant, out_dir)
186
+ out.mkdir(parents=True, exist_ok=True)
187
+
188
+ def step(msg: str, frac: float) -> None:
189
+ log.info("gguf-pipeline: %s", msg)
190
+ if progress:
191
+ progress(msg, frac)
192
+
193
+ step("preparing tools/gguf env (first run installs the pip deps)", 0.05)
194
+ py = self.ensure_env(progress=progress)
195
+ src = ensure_llamacpp_src(progress=progress)
196
+
197
+ step("convert: HF -> f16 GGUF", 0.2)
198
+ self._envs.run([str(py), str(src / "convert_hf_to_gguf.py"), model_dir,
199
+ "--outfile", str(f16), "--outtype", "f16"])
200
+ if not f16.exists():
201
+ raise QuantizationError("convert produced no f16 GGUF.", detail=str(f16))
202
+
203
+ self.quantize(str(f16), str(quantized), quant, progress=progress)
204
+ self.verify(str(quantized), progress=progress)
205
+
206
+ if not keep_f16:
207
+ with contextlib.suppress(OSError):
208
+ f16.unlink()
209
+ step("done", 1.0)
210
+ return GgufArtifacts(out_dir=out, f16_gguf=f16, quantized_gguf=quantized, quant_type=quant)
@@ -0,0 +1,248 @@
1
+ """Run quantization toolchains in isolated sub-environments.
2
+
3
+ Many quantizers pin mutually-incompatible torch/transformers versions (e.g.
4
+ GPTQModel wants torch≥2.8 while the archived AutoGPTQ froze at 2.2.1), so they
5
+ cannot share one environment. Each toolchain gets a dedicated venv (built and
6
+ cached by :class:`~zeroquantz.core.subenv.SubEnvManager` under
7
+ ``~/.zeroquantz/envs/quant/<family>/``); this module installs just that toolchain
8
+ and runs a generated quantization script as a subprocess — passing config via a
9
+ JSON file and reading results back from a ``manifest.json`` (rather than parsing
10
+ tqdm-polluted stdout).
11
+
12
+ The heavy work (multi-GB installs + quantization) happens in the child process.
13
+ ``plan_only=True`` returns the exact env + commands + script without executing,
14
+ so a run can be inspected first.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import json
20
+ from dataclasses import dataclass, field
21
+ from pathlib import Path
22
+ from typing import TYPE_CHECKING
23
+
24
+ from zeroquantz.core.exceptions import QuantizationError
25
+ from zeroquantz.core.subenv import EnvSpec, SubEnvManager, torch_index_for
26
+ from zeroquantz.utils.logging import get_logger
27
+
28
+ if TYPE_CHECKING:
29
+ from collections.abc import Callable
30
+
31
+ from zeroquantz.quantization.catalog import QuantFormat
32
+
33
+ log = get_logger(__name__)
34
+
35
+
36
+ # ---- per-toolchain quantization scripts ------------------------------------
37
+ # Each reads {"model", "output", "bits", "group_size", "calibration"} from a JSON
38
+ # config (argv --config) and writes {"output_model_dir": ...} to --manifest.
39
+
40
+ _MANIFEST_TAIL = '''
41
+ import json, sys
42
+ from pathlib import Path
43
+ _manifest = sys.argv[sys.argv.index("--manifest") + 1]
44
+ Path(_manifest).write_text(json.dumps({"output_model_dir": cfg["output"], "status": "ok"}))
45
+ print("ZEROQUANTZ_DONE", cfg["output"])
46
+ '''
47
+
48
+ # NOTE: recipes target each library's current documented API. They are best-effort
49
+ # and may need updates as libraries evolve; run with plan_only to inspect first.
50
+ _RECIPES: dict[str, str] = {
51
+ "awq": '''
52
+ import json, sys
53
+ from pathlib import Path
54
+ cfg = json.loads(Path(sys.argv[sys.argv.index("--config") + 1]).read_text())
55
+ from awq import AutoAWQForCausalLM
56
+ from transformers import AutoTokenizer
57
+ model = AutoAWQForCausalLM.from_pretrained(cfg["model"], safetensors=True)
58
+ tok = AutoTokenizer.from_pretrained(cfg["model"], trust_remote_code=True)
59
+ model.quantize(tok, quant_config={
60
+ "w_bit": cfg.get("bits", 4), "q_group_size": cfg.get("group_size", 128),
61
+ "zero_point": True, "version": "GEMM",
62
+ })
63
+ model.save_quantized(cfg["output"]); tok.save_pretrained(cfg["output"])
64
+ ''' + _MANIFEST_TAIL,
65
+
66
+ "gptq": '''
67
+ import json, sys
68
+ from pathlib import Path
69
+ cfg = json.loads(Path(sys.argv[sys.argv.index("--config") + 1]).read_text())
70
+ from gptqmodel import GPTQModel, QuantizeConfig
71
+ qc = QuantizeConfig(bits=cfg.get("bits", 4), group_size=cfg.get("group_size", 128))
72
+ model = GPTQModel.load(cfg["model"], qc)
73
+ model.quantize(cfg["calibration"], batch_size=1)
74
+ model.save(cfg["output"])
75
+ ''' + _MANIFEST_TAIL,
76
+
77
+ "autoround": '''
78
+ import json, sys
79
+ from pathlib import Path
80
+ cfg = json.loads(Path(sys.argv[sys.argv.index("--config") + 1]).read_text())
81
+ from transformers import AutoModelForCausalLM, AutoTokenizer
82
+ from auto_round import AutoRound
83
+ model = AutoModelForCausalLM.from_pretrained(cfg["model"], torch_dtype="auto", device_map="auto")
84
+ tok = AutoTokenizer.from_pretrained(cfg["model"])
85
+ ar = AutoRound(model, tok, bits=cfg.get("bits", 4), group_size=cfg.get("group_size", 128), nsamples=128)
86
+ ar.quantize_and_save(cfg["output"], format="auto_gptq")
87
+ ''' + _MANIFEST_TAIL,
88
+
89
+ "hqq": '''
90
+ import json, sys
91
+ from pathlib import Path
92
+ cfg = json.loads(Path(sys.argv[sys.argv.index("--config") + 1]).read_text())
93
+ import torch
94
+ from transformers import AutoModelForCausalLM, AutoTokenizer
95
+ from hqq.models.hf.base import AutoHQQHFModel
96
+ from hqq.core.quantize import BaseQuantizeConfig
97
+ model = AutoModelForCausalLM.from_pretrained(cfg["model"], torch_dtype=torch.float16)
98
+ tok = AutoTokenizer.from_pretrained(cfg["model"])
99
+ qcfg = BaseQuantizeConfig(nbits=cfg.get("bits", 4), group_size=cfg.get("group_size", 64))
100
+ AutoHQQHFModel.quantize_model(model, quant_config=qcfg, compute_dtype=torch.float16, device="cuda")
101
+ AutoHQQHFModel.save_quantized(model, cfg["output"]); tok.save_pretrained(cfg["output"])
102
+ ''' + _MANIFEST_TAIL,
103
+ }
104
+
105
+
106
+ @dataclass
107
+ class IsolatedPlan:
108
+ """A previewable plan for an isolated quantization run."""
109
+
110
+ family: str
111
+ env_dir: Path
112
+ python: Path
113
+ pip_specs: tuple[str, ...]
114
+ torch_index: str | None
115
+ script_path: Path
116
+ commands: list[str] = field(default_factory=list)
117
+
118
+
119
+ @dataclass
120
+ class IsolatedResult:
121
+ family: str
122
+ output_dir: str
123
+ env_dir: Path
124
+ elapsed_s: float = 0.0
125
+
126
+
127
+ class IsolatedEnvRunner:
128
+ """Create and drive per-toolchain isolated environments (kind='quant')."""
129
+
130
+ def __init__(self, root: Path | None = None) -> None:
131
+ self._envs = SubEnvManager(root=root)
132
+
133
+ # ---- spec / helpers -----------------------------------------------------
134
+
135
+ def _spec(self, fmt: QuantFormat, cuda_version: str | None) -> EnvSpec:
136
+ return EnvSpec(
137
+ kind="quant", name=fmt.family, pip=tuple(fmt.pip),
138
+ torch_index=torch_index_for(cuda_version),
139
+ )
140
+
141
+ def supports(self, family: str) -> bool:
142
+ return family in _RECIPES
143
+
144
+ @staticmethod
145
+ def venv_python(env_dir: Path) -> Path:
146
+ return SubEnvManager.venv_python(env_dir)
147
+
148
+ # ---- planning -----------------------------------------------------------
149
+
150
+ def plan(self, fmt: QuantFormat, *, cuda_version: str | None = None) -> IsolatedPlan:
151
+ if not self.supports(fmt.family):
152
+ raise QuantizationError(
153
+ f"No isolated quantization recipe for '{fmt.family}'.",
154
+ detail="ZeroQuantz can plan/estimate this format but not run it yet.",
155
+ suggestions=["/recommend", "pick an in-env or supported isolated format"],
156
+ )
157
+ spec = self._spec(fmt, cuda_version)
158
+ base = self._envs.plan(spec)
159
+ commands = [*base.commands,
160
+ f"{base.python} quantize.py --config run_cfg.json --manifest manifest.json"]
161
+ return IsolatedPlan(
162
+ family=fmt.family, env_dir=base.env_dir, python=base.python, pip_specs=fmt.pip,
163
+ torch_index=spec.torch_index, script_path=base.env_dir / "quantize.py",
164
+ commands=commands,
165
+ )
166
+
167
+ # ---- environment build --------------------------------------------------
168
+
169
+ def ensure_env(
170
+ self,
171
+ fmt: QuantFormat,
172
+ *,
173
+ cuda_version: str | None = None,
174
+ progress: Callable[[str, float], None] | None = None,
175
+ ) -> Path:
176
+ return self._envs.ensure(self._spec(fmt, cuda_version), progress=progress)
177
+
178
+ # ---- execution ----------------------------------------------------------
179
+
180
+ def quantize(
181
+ self,
182
+ fmt: QuantFormat,
183
+ model_ref: str,
184
+ output_dir: str,
185
+ *,
186
+ bits: int | None = None,
187
+ group_size: int = 128,
188
+ calibration: list[str] | None = None,
189
+ cuda_version: str | None = None,
190
+ plan_only: bool = False,
191
+ progress: Callable[[str, float], None] | None = None,
192
+ ) -> IsolatedResult | IsolatedPlan:
193
+ if not self.supports(fmt.family):
194
+ raise QuantizationError(
195
+ f"No isolated quantization recipe for '{fmt.family}'.",
196
+ detail=f"'{fmt.label}' is produced by {fmt.produced_by}; "
197
+ "ZeroQuantz catalogs/estimates it but does not run it automatically yet.",
198
+ )
199
+ if plan_only:
200
+ return self.plan(fmt, cuda_version=cuda_version)
201
+
202
+ import time
203
+
204
+ spec = self._spec(fmt, cuda_version)
205
+ py = self.ensure_env(fmt, cuda_version=cuda_version, progress=progress)
206
+ env_dir = self._envs.env_dir(spec)
207
+ script_path = env_dir / "quantize.py"
208
+ script_path.write_text(_RECIPES[fmt.family], encoding="utf-8")
209
+ config = {
210
+ "model": model_ref,
211
+ "output": str(Path(output_dir).resolve()),
212
+ "bits": bits or max(2, round(fmt.bits_per_weight)),
213
+ "group_size": group_size,
214
+ "calibration": calibration or _default_calibration(),
215
+ }
216
+ cfg_path = env_dir / "run_cfg.json"
217
+ cfg_path.write_text(json.dumps(config), encoding="utf-8")
218
+ manifest_path = env_dir / "manifest.json"
219
+ manifest_path.unlink(missing_ok=True)
220
+
221
+ if progress:
222
+ progress(f"quantizing with {fmt.family}", 0.6)
223
+ started = time.perf_counter()
224
+ self._envs.run(
225
+ [str(py), "-u", str(script_path), "--config", str(cfg_path),
226
+ "--manifest", str(manifest_path)],
227
+ progress=progress,
228
+ )
229
+ elapsed = time.perf_counter() - started
230
+
231
+ if not manifest_path.exists():
232
+ raise QuantizationError(
233
+ f"Isolated {fmt.family} run finished without producing a manifest.",
234
+ detail="The quantization script did not complete; check the logs.",
235
+ )
236
+ manifest = json.loads(manifest_path.read_text())
237
+ if progress:
238
+ progress("done", 1.0)
239
+ return IsolatedResult(
240
+ family=fmt.family, output_dir=manifest["output_model_dir"], env_dir=env_dir,
241
+ elapsed_s=elapsed,
242
+ )
243
+
244
+
245
+ def _default_calibration() -> list[str]:
246
+ from zeroquantz.profiling.calibration import default_calibration
247
+
248
+ return default_calibration().prompts
@@ -0,0 +1,133 @@
1
+ """Heuristic memory estimation shared by backends and the candidate generator.
2
+
3
+ Everything here is an *estimate* derived from parameter counts and architecture
4
+ dimensions — never a measurement. The candidate generator labels it as such, and
5
+ the benchmark subsystem is what produces measured numbers.
6
+
7
+ Model of VRAM at inference time:
8
+
9
+ vram = quantized_weights + cuda_context + kv_cache + activations
10
+
11
+ * ``quantized_weights`` — a two-bucket model: parameters a weight-only backend
12
+ actually quantizes (linear layers) at the target precision, plus the
13
+ parameters it keeps at compute precision (embeddings, and the untied LM head).
14
+ * ``cuda_context`` — a fixed allocator/context overhead.
15
+ * ``kv_cache`` — grows with sequence length and (grouped) KV heads.
16
+ * ``activations`` — a modest term proportional to hidden size and sequence.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ from typing import TYPE_CHECKING
22
+
23
+ from zeroquantz.utils import units
24
+
25
+ if TYPE_CHECKING:
26
+ from zeroquantz.models.metadata import ModelProfile
27
+ from zeroquantz.quantization.config import QuantizationConfig
28
+
29
+ # Fixed CUDA context / allocator overhead, in GiB. Realistic for a single device.
30
+ CUDA_CONTEXT_GIB = 0.6
31
+ # Activation memory is roughly proportional to hidden*seq*batch; this multiplies it.
32
+ ACTIVATION_FACTOR = 2.0
33
+ # Default profiling shape used for the estimate.
34
+ DEFAULT_SEQ_LEN = 2048
35
+ DEFAULT_BATCH = 1
36
+
37
+
38
+ def estimate_weight_bytes(model: ModelProfile, config: QuantizationConfig) -> float:
39
+ """Estimated bytes of *weights* after applying ``config`` to ``model``."""
40
+ base_bytes = units.bytes_per_param(config.compute_dtype)
41
+ non_quant = model.non_quantizable_params * base_bytes
42
+ q_params = model.quantizable_params
43
+
44
+ if config.method == "mixed":
45
+ low = config.extra.get("low_precision", "int4")
46
+ high = config.extra.get("high_precision", "int8")
47
+ high_fraction = float(config.extra.get("high_fraction", 0.25))
48
+ per_param = high_fraction * units.bytes_per_param(high) + (
49
+ 1 - high_fraction
50
+ ) * units.bytes_per_param(low)
51
+ quant = q_params * per_param
52
+ else:
53
+ quant = q_params * units.bytes_per_param(config.weight_precision)
54
+
55
+ return non_quant + quant
56
+
57
+
58
+ def estimate_kv_cache_bytes(
59
+ model: ModelProfile,
60
+ *,
61
+ seq_len: int = DEFAULT_SEQ_LEN,
62
+ batch: int = DEFAULT_BATCH,
63
+ cache_dtype: str = "float16",
64
+ ) -> float:
65
+ """KV-cache bytes for a given sequence length and batch."""
66
+ layers = model.num_layers
67
+ kv_heads = model.num_key_value_heads or model.num_attention_heads
68
+ head_dim = model.head_dim
69
+ if not (layers and kv_heads and head_dim):
70
+ return 0.0
71
+ per_token = 2 * layers * kv_heads * head_dim * units.bytes_per_param(cache_dtype)
72
+ return per_token * seq_len * batch
73
+
74
+
75
+ def estimate_activation_bytes(
76
+ model: ModelProfile,
77
+ *,
78
+ seq_len: int = DEFAULT_SEQ_LEN,
79
+ batch: int = DEFAULT_BATCH,
80
+ compute_dtype: str = "bfloat16",
81
+ ) -> float:
82
+ """A coarse activation-memory estimate."""
83
+ hidden = model.hidden_size
84
+ if not hidden:
85
+ return 0.0
86
+ return hidden * seq_len * batch * units.bytes_per_param(compute_dtype) * ACTIVATION_FACTOR
87
+
88
+
89
+ def estimate_runtime_overhead_bytes(
90
+ model: ModelProfile,
91
+ *,
92
+ seq_len: int = DEFAULT_SEQ_LEN,
93
+ batch: int = DEFAULT_BATCH,
94
+ compute_dtype: str = "bfloat16",
95
+ ) -> float:
96
+ context = units.gb_to_bytes(CUDA_CONTEXT_GIB)
97
+ kv = estimate_kv_cache_bytes(model, seq_len=seq_len, batch=batch)
98
+ act = estimate_activation_bytes(
99
+ model, seq_len=seq_len, batch=batch, compute_dtype=compute_dtype
100
+ )
101
+ return context + kv + act
102
+
103
+
104
+ def estimate_vram_bytes(
105
+ model: ModelProfile,
106
+ config: QuantizationConfig,
107
+ *,
108
+ seq_len: int = DEFAULT_SEQ_LEN,
109
+ batch: int = DEFAULT_BATCH,
110
+ ) -> float:
111
+ """Total estimated VRAM to *run* ``model`` under ``config``."""
112
+ weights = estimate_weight_bytes(model, config)
113
+ overhead = estimate_runtime_overhead_bytes(
114
+ model, seq_len=seq_len, batch=batch, compute_dtype=config.compute_dtype
115
+ )
116
+ return weights + overhead
117
+
118
+
119
+ def estimate_weight_gb(model: ModelProfile, config: QuantizationConfig) -> float:
120
+ return round(units.bytes_to_gb(estimate_weight_bytes(model, config)), 2)
121
+
122
+
123
+ def estimate_vram_gb(
124
+ model: ModelProfile,
125
+ config: QuantizationConfig,
126
+ *,
127
+ seq_len: int = DEFAULT_SEQ_LEN,
128
+ batch: int = DEFAULT_BATCH,
129
+ ) -> float:
130
+ return round(
131
+ units.bytes_to_gb(estimate_vram_bytes(model, config, seq_len=seq_len, batch=batch)),
132
+ 2,
133
+ )
@@ -0,0 +1,91 @@
1
+ """Locate ZeroQuantz's optional Rust binary and the llama.cpp converter source.
2
+
3
+ The GGUF pipeline is otherwise pip-only (llama-cpp-python); the only native binary
4
+ is the experimental Rust converter, and llama.cpp is needed only as *source* for its
5
+ ``convert_hf_to_gguf.py`` script (no compilation).
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import shutil
11
+ import subprocess
12
+ from pathlib import Path
13
+ from typing import TYPE_CHECKING
14
+
15
+ from zeroquantz.core.exceptions import ZeroQuantzError
16
+ from zeroquantz.utils.paths import paths
17
+
18
+ if TYPE_CHECKING:
19
+ from collections.abc import Callable
20
+
21
+ _LLAMACPP_REPO = "https://github.com/ggml-org/llama.cpp.git"
22
+
23
+
24
+ def repo_root() -> Path:
25
+ """The repository root (parent of the installed ``zeroquantz`` package)."""
26
+ import zeroquantz
27
+
28
+ return Path(zeroquantz.__file__).resolve().parents[1]
29
+
30
+
31
+ def find_native_binary(name: str) -> Path | None:
32
+ """Return the path to a built Rust binary (on PATH or ./target/release), or None."""
33
+ on_path = shutil.which(name)
34
+ if on_path:
35
+ return Path(on_path)
36
+ base = repo_root() / "target" / "release"
37
+ for exe in (f"{name}.exe", name):
38
+ if (base / exe).exists():
39
+ return base / exe
40
+ return None
41
+
42
+
43
+ def llamacpp_src() -> Path | None:
44
+ """Locate a llama.cpp source tree containing ``convert_hf_to_gguf.py``.
45
+
46
+ Prefers ``$ZEROQUANTZ_LLAMACPP_SRC``, then the shallow clone ZeroQuantz manages at
47
+ ``~/.zeroquantz/llama.cpp``.
48
+ """
49
+ import os
50
+
51
+ env = os.environ.get("ZEROQUANTZ_LLAMACPP_SRC")
52
+ if env and (Path(env) / "convert_hf_to_gguf.py").exists():
53
+ return Path(env)
54
+ managed = paths().home / "llama.cpp"
55
+ if (managed / "convert_hf_to_gguf.py").exists():
56
+ return managed
57
+ return None
58
+
59
+
60
+ def ensure_llamacpp_src(progress: Callable[[str, float], None] | None = None) -> Path:
61
+ """Return a llama.cpp source tree, shallow-cloning it if needed (no build).
62
+
63
+ Only the Python converter (``convert_hf_to_gguf.py``) is used from it, so this is
64
+ a source checkout, not a compile — needs ``git`` but no C/C++ toolchain.
65
+ """
66
+ existing = llamacpp_src()
67
+ if existing is not None:
68
+ return existing
69
+ if shutil.which("git") is None:
70
+ raise ZeroQuantzError(
71
+ "git is required to fetch llama.cpp's converter script.",
72
+ suggestions=[
73
+ "install git, or",
74
+ "set ZEROQUANTZ_LLAMACPP_SRC=/path/to/a/llama.cpp/checkout",
75
+ ],
76
+ )
77
+ dest = paths().ensure().home / "llama.cpp"
78
+ if progress:
79
+ progress("fetching llama.cpp converter (shallow clone, no build)", 0.02)
80
+ try:
81
+ subprocess.run(
82
+ ["git", "clone", "--depth", "1", _LLAMACPP_REPO, str(dest)],
83
+ check=True, capture_output=True, text=True,
84
+ )
85
+ except subprocess.CalledProcessError as exc:
86
+ raise ZeroQuantzError(
87
+ "Failed to clone llama.cpp.", detail=(exc.stderr or "").strip()[:500]
88
+ ) from exc
89
+ if not (dest / "convert_hf_to_gguf.py").exists():
90
+ raise ZeroQuantzError(f"cloned llama.cpp but convert_hf_to_gguf.py is missing in {dest}.")
91
+ return dest