modelspec-dev 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- api/__init__.py +0 -0
- api/class_fit.py +334 -0
- api/classes.py +557 -0
- api/ranking/__init__.py +12 -0
- api/ranking/engine.py +1943 -0
- cli/__init__.py +0 -0
- cli/modelspec/__init__.py +0 -0
- cli/modelspec/cli.py +1819 -0
- cli/modelspec/commands/__init__.py +0 -0
- cli/modelspec/decide_cmd.py +333 -0
- cli/modelspec/offline.py +623 -0
- cli/modelspec/snapshot.py +698 -0
- cli/modelspec/snapshot_build_cmd.py +49 -0
- cli/modelspec/verify_cmd.py +125 -0
- cli/modelspec/vocab_cmd.py +204 -0
- cli/modelspec/vocabulary_cache.py +54 -0
- decision/__init__.py +13 -0
- decision/capability.py +872 -0
- decision/computed.py +125 -0
- decision/contract.py +1575 -0
- decision/engine.py +238 -0
- decision/excluded.py +34 -0
- decision/explain.py +908 -0
- decision/filter.py +796 -0
- decision/model.py +438 -0
- decision/normalise.py +604 -0
- decision/optimise.py +320 -0
- decision/registry.py +717 -0
- decision/relax.py +132 -0
- decision/resolve.py +111 -0
- decision/schema.py +21 -0
- decision/snapshot.py +1483 -0
- decision/sources.py +544 -0
- decision/templates.py +134 -0
- decision/verify.py +1745 -0
- decision/vocabulary.py +433 -0
- modelspec_dev-0.1.0.dist-info/METADATA +101 -0
- modelspec_dev-0.1.0.dist-info/RECORD +63 -0
- modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
- modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
- modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
- pipeline/__init__.py +0 -0
- pipeline/class_export.py +172 -0
- pipeline/hardware.py +434 -0
- pipeline/hosts.py +247 -0
- pipeline/load.py +224 -0
- pipeline/ranking.py +551 -0
- registry/domains.yaml +130 -0
- registry/facets.yaml +888 -0
- registry/harnesses.yaml +79 -0
- registry/providers.yaml +354 -0
- registry/sources.yaml +3059 -0
- registry/templates.yaml +166 -0
- schema/__init__.py +0 -0
- schema/applicability.py +147 -0
- schema/benchmark.py +175 -0
- schema/benchmark_eligibility.py +304 -0
- schema/card.py +1463 -0
- schema/enrichment.py +162 -0
- schema/enums.py +327 -0
- schema/graph.py +406 -0
- schema/suppliers.py +72 -0
pipeline/hardware.py
ADDED
|
@@ -0,0 +1,434 @@
|
|
|
1
|
+
"""Compute which models fit on which devices, and how fast they would decode.
|
|
2
|
+
|
|
3
|
+
Everything here is *computed*, never measured, and is labelled as such all the
|
|
4
|
+
way to the page. A predicted figure presented as a measurement is a lie a reader
|
|
5
|
+
will plan around.
|
|
6
|
+
|
|
7
|
+
Decode speed for a mixture-of-experts model depends on *active* parameters, not
|
|
8
|
+
total. With no active-parameter data the prediction uses total, which understates
|
|
9
|
+
MoE speed — sometimes by a large factor. Every prediction says so.
|
|
10
|
+
|
|
11
|
+
Max context is leftover memory after weights and the working allowance, divided
|
|
12
|
+
by the KV-cache bytes per token. That needs layer and head geometry
|
|
13
|
+
(`num_layers`, `num_kv_heads` or `num_attention_heads`, `hidden_size`). The
|
|
14
|
+
schema has no `head_dim`; it is derived as `hidden_size / num_attention_heads`.
|
|
15
|
+
When any of that is missing, `max_context_at_quant` is null — never zero, never
|
|
16
|
+
a guess — and `max_context_missing_geometry` is true so a consumer can tell
|
|
17
|
+
"we do not know" from "it does not fit".
|
|
18
|
+
|
|
19
|
+
`FITS_ON` is only computed for open-weights models. Asking whether a
|
|
20
|
+
closed-weights model "fits" on your GPU is meaningless — you cannot obtain it.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
from dataclasses import dataclass
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
from typing import Any
|
|
28
|
+
|
|
29
|
+
import yaml
|
|
30
|
+
|
|
31
|
+
from schema.enums import DeviceClass, ModelType
|
|
32
|
+
from schema.graph import CollectingSink
|
|
33
|
+
|
|
34
|
+
#: model_type values that autoregressively decode tokens, so a tok/s figure is
|
|
35
|
+
#: a meaningful prediction. Everything else (embeddings, rerankers, safety
|
|
36
|
+
#: classifiers, generation of another modality, OCR, reward models, and the
|
|
37
|
+
#: perception/encoder/time-series types) still gets a capacity FITS_ON answer
|
|
38
|
+
#: — the weights either fit in memory or they don't — but never a decode
|
|
39
|
+
#: speed, because there is no token being decoded. A card with no model_type
|
|
40
|
+
#: is treated the same as a non-token type: absence is not a guess that it
|
|
41
|
+
#: chats. Single source of truth for this distinction — do not scatter
|
|
42
|
+
#: per-type checks elsewhere.
|
|
43
|
+
TOKEN_GENERATING_MODEL_TYPES: frozenset[ModelType] = frozenset({
|
|
44
|
+
ModelType.LLM_CHAT, ModelType.LLM_REASONING, ModelType.LLM_CODE, ModelType.LLM_BASE,
|
|
45
|
+
ModelType.VLM, ModelType.MEDICAL, ModelType.LEGAL, ModelType.FINANCIAL,
|
|
46
|
+
ModelType.AGENT_MODEL, ModelType.ROUTER,
|
|
47
|
+
ModelType.ADAPTER, ModelType.QUANTIZED_VARIANT, ModelType.DISTILLED, ModelType.MERGED,
|
|
48
|
+
})
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def is_token_generating(model_type: Any) -> bool:
|
|
52
|
+
"""Whether `model_type` decodes tokens and so gets a decode-speed prediction.
|
|
53
|
+
|
|
54
|
+
Accepts the raw `Identity.model_type` (a `ModelType | None`, possibly
|
|
55
|
+
already a plain string in code that reads snapshots rather than cards).
|
|
56
|
+
"""
|
|
57
|
+
if model_type is None:
|
|
58
|
+
return False
|
|
59
|
+
if isinstance(model_type, ModelType):
|
|
60
|
+
return model_type in TOKEN_GENERATING_MODEL_TYPES
|
|
61
|
+
return str(model_type) in {t.value for t in TOKEN_GENERATING_MODEL_TYPES}
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
#: Bytes per parameter at each quantisation, best case. Real files carry
|
|
65
|
+
#: metadata and some layers stay at higher precision, which the working
|
|
66
|
+
#: allowance below absorbs.
|
|
67
|
+
QUANT_BYTES: dict[str, float] = {
|
|
68
|
+
"fp16": 2.0, "bf16": 2.0, "fp8": 1.0, "int8": 1.0,
|
|
69
|
+
"q6": 0.75, "q5": 0.625, "int4": 0.5, "q4": 0.5,
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
#: Preference order when choosing the best quantisation that fits. Higher
|
|
73
|
+
#: precision first: we want the best quality that fits, not the smallest.
|
|
74
|
+
QUANT_PREFERENCE = ("bf16", "fp16", "fp8", "int8", "q6", "q5", "q4")
|
|
75
|
+
|
|
76
|
+
#: Headroom for activations, the framework and the OS, as a fraction of device
|
|
77
|
+
#: memory. KV-cache size is computed from layer geometry when the card has it;
|
|
78
|
+
#: this allowance is the rest of the working set, not a stand-in for the cache.
|
|
79
|
+
WORKING_ALLOWANCE = 0.25
|
|
80
|
+
|
|
81
|
+
#: Bytes per KV-cache element. The cache is commonly kept at fp16 even when the
|
|
82
|
+
#: weights are quantised; using QUANT_BYTES for the weight quant would understate
|
|
83
|
+
#: the cache (and overstate how much context fits) at q4/int8.
|
|
84
|
+
KV_BYTES_PER_ELEMENT = 2.0
|
|
85
|
+
|
|
86
|
+
#: Real bandwidth utilisation. No decoder achieves the theoretical roofline;
|
|
87
|
+
#: measured llama.cpp and vLLM figures typically land in the 60-80% band.
|
|
88
|
+
BANDWIDTH_EFFICIENCY = 0.70
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@dataclass(frozen=True)
|
|
92
|
+
class Device:
|
|
93
|
+
id: str
|
|
94
|
+
display_name: str
|
|
95
|
+
vendor: str
|
|
96
|
+
device_class: str
|
|
97
|
+
bandwidth_gb_s: float
|
|
98
|
+
capacity_options_gb: tuple[float, ...]
|
|
99
|
+
precisions_native: tuple[str, ...]
|
|
100
|
+
unified: bool
|
|
101
|
+
single_device_fit: bool = True
|
|
102
|
+
single_device_fit_reason: str | None = None
|
|
103
|
+
|
|
104
|
+
@property
|
|
105
|
+
def max_capacity_gb(self) -> float:
|
|
106
|
+
return max(self.capacity_options_gb)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _single_device_fit(raw: dict[str, Any], path: Path) -> tuple[bool, str | None]:
|
|
110
|
+
"""DATA flag: false means this part must not answer single-device FITS_ON.
|
|
111
|
+
|
|
112
|
+
Default is true (omit the field). false requires a non-empty reason so a
|
|
113
|
+
reader can see why the layer refused rather than guessing from the device id.
|
|
114
|
+
"""
|
|
115
|
+
if "single_device_fit" not in raw:
|
|
116
|
+
return True, None
|
|
117
|
+
flag = raw["single_device_fit"]
|
|
118
|
+
if flag is True:
|
|
119
|
+
reason = raw.get("single_device_fit_reason")
|
|
120
|
+
if reason is None:
|
|
121
|
+
return True, None
|
|
122
|
+
if not isinstance(reason, str) or not reason.strip():
|
|
123
|
+
raise ValueError(
|
|
124
|
+
f"{path.name}: single_device_fit_reason must be a non-empty string when present"
|
|
125
|
+
)
|
|
126
|
+
return True, reason.strip()
|
|
127
|
+
if flag is False:
|
|
128
|
+
reason = raw.get("single_device_fit_reason")
|
|
129
|
+
if not isinstance(reason, str) or not reason.strip():
|
|
130
|
+
raise ValueError(
|
|
131
|
+
f"{path.name}: single_device_fit is false but single_device_fit_reason "
|
|
132
|
+
"is missing. A part that cannot answer single-device fit must say why."
|
|
133
|
+
)
|
|
134
|
+
return False, reason.strip()
|
|
135
|
+
raise ValueError(
|
|
136
|
+
f"{path.name}: single_device_fit must be true or false, not {flag!r}"
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _device_class(raw: dict[str, Any], path: Path) -> str:
|
|
141
|
+
"""The record's `device_class`, checked against the vocabulary (MODEL-76).
|
|
142
|
+
|
|
143
|
+
The graph groups Hardware nodes by this value, so `datacenter` or `Consumer`
|
|
144
|
+
would not be a near miss — it would be a sixth class that every query for
|
|
145
|
+
the real one silently skips. Rejected at load, where the file name is still
|
|
146
|
+
in hand to name in the error.
|
|
147
|
+
"""
|
|
148
|
+
value = raw.get("device_class")
|
|
149
|
+
try:
|
|
150
|
+
return DeviceClass(value).value
|
|
151
|
+
except ValueError:
|
|
152
|
+
allowed = ", ".join(c.value for c in DeviceClass)
|
|
153
|
+
raise ValueError(
|
|
154
|
+
f"{path.name}: device_class {value!r} is not one of {allowed}"
|
|
155
|
+
) from None
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
@dataclass(frozen=True)
|
|
159
|
+
class MaxContext:
|
|
160
|
+
"""Predicted context length at one (device, quant) pair.
|
|
161
|
+
|
|
162
|
+
`tokens` is null only when geometry is missing. Zero means the weights fit
|
|
163
|
+
but leftover memory cannot hold a single token of KV — that is not the
|
|
164
|
+
same as "we do not know".
|
|
165
|
+
"""
|
|
166
|
+
tokens: int | None
|
|
167
|
+
bound_by: str | None
|
|
168
|
+
missing_geometry: bool
|
|
169
|
+
kv_heads_from_attention: bool
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def load_devices(root: Path) -> list[Device]:
|
|
173
|
+
"""Read hardware/*.yaml. A device without bandwidth is rejected, not defaulted."""
|
|
174
|
+
out: list[Device] = []
|
|
175
|
+
for path in sorted((root / "hardware").glob("*.yaml")):
|
|
176
|
+
if path.name.startswith("_"):
|
|
177
|
+
continue
|
|
178
|
+
raw = yaml.safe_load(path.read_text(encoding="utf-8"))
|
|
179
|
+
memory = raw["memory"]
|
|
180
|
+
bandwidth = memory.get("bandwidth_gb_s")
|
|
181
|
+
if not bandwidth:
|
|
182
|
+
raise ValueError(
|
|
183
|
+
f"{path.name}: no memory.bandwidth_gb_s. A device without bandwidth "
|
|
184
|
+
"cannot answer how fast a model will run, which is the question this "
|
|
185
|
+
"layer exists to answer. Fix the definition or remove it."
|
|
186
|
+
)
|
|
187
|
+
options = memory.get("capacity_options_gb") or [memory["capacity_gb"]]
|
|
188
|
+
fit, fit_reason = _single_device_fit(raw, path)
|
|
189
|
+
out.append(Device(
|
|
190
|
+
id=raw["id"], display_name=raw["display_name"], vendor=raw["vendor"],
|
|
191
|
+
device_class=_device_class(raw, path), bandwidth_gb_s=float(bandwidth),
|
|
192
|
+
capacity_options_gb=tuple(float(c) for c in options),
|
|
193
|
+
precisions_native=tuple(raw.get("precisions_native") or []),
|
|
194
|
+
unified=bool(memory.get("unified_with_host")),
|
|
195
|
+
single_device_fit=fit,
|
|
196
|
+
single_device_fit_reason=fit_reason,
|
|
197
|
+
))
|
|
198
|
+
return out
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def device_classes(devices: list[Device]) -> dict[str, str]:
|
|
202
|
+
"""Device id → `DeviceClass` value, for the graph derivation (MODEL-76).
|
|
203
|
+
|
|
204
|
+
`schema.graph` sits below this loader and cannot read `hardware/*.yaml`
|
|
205
|
+
itself, so it takes this mapping and stamps the class onto the Hardware
|
|
206
|
+
nodes a card's deployment profiles produce. One source of truth for the
|
|
207
|
+
class, whether the graph is being exported as JSON or ingested into
|
|
208
|
+
FalkorDB.
|
|
209
|
+
"""
|
|
210
|
+
return {device.id: device.device_class for device in devices}
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def weights_gb(params: float, quant: str) -> float:
|
|
214
|
+
return params * QUANT_BYTES[quant] / 1e9
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def fitting_quants(params: float, capacity_gb: float) -> list[str]:
|
|
218
|
+
"""Every quantisation whose weights fit in the usable memory, best quality first.
|
|
219
|
+
|
|
220
|
+
A quantisation the silicon does not accelerate natively is still allowed — it
|
|
221
|
+
runs, just without the speed benefit — so this gates on memory, not on
|
|
222
|
+
`precisions_native`. That field informs the prediction, not the fit.
|
|
223
|
+
"""
|
|
224
|
+
usable = capacity_gb * (1.0 - WORKING_ALLOWANCE)
|
|
225
|
+
return [q for q in QUANT_PREFERENCE if weights_gb(params, q) <= usable]
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def best_quant(params: float, capacity_gb: float, native: tuple[str, ...] = ()) -> str | None:
|
|
229
|
+
"""The highest-quality quantisation that fits, or None."""
|
|
230
|
+
quants = fitting_quants(params, capacity_gb)
|
|
231
|
+
return quants[0] if quants else None
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def predicted_decode_tps(bandwidth_gb_s: float, params: float, quant: str) -> float:
|
|
235
|
+
"""Roofline estimate: bandwidth divided by the bytes read per token.
|
|
236
|
+
|
|
237
|
+
`params` must be the parameters actually *read* to emit a token, which for a
|
|
238
|
+
mixture-of-experts model is its active count, not its total. The difference
|
|
239
|
+
is not marginal: qwen3-coder-next holds 480B and activates 3.2B, a factor of
|
|
240
|
+
148. Using total parameters there understates its speed by that factor and
|
|
241
|
+
makes the architecture built for speed look like the worst local choice.
|
|
242
|
+
"""
|
|
243
|
+
per_token_gb = weights_gb(params, quant)
|
|
244
|
+
if per_token_gb <= 0:
|
|
245
|
+
return 0.0
|
|
246
|
+
return round(bandwidth_gb_s * BANDWIDTH_EFFICIENCY / per_token_gb, 1)
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _kv_geometry(card: Any) -> tuple[float, bool] | None:
|
|
250
|
+
"""(bytes_per_token, kv_heads_taken_from_attention) or None.
|
|
251
|
+
|
|
252
|
+
Never infers missing layer counts or hidden size. Absent `num_kv_heads` is
|
|
253
|
+
MHA, so it falls back to `num_attention_heads` and reports that it did.
|
|
254
|
+
"""
|
|
255
|
+
arch = card.architecture
|
|
256
|
+
layers = arch.num_layers
|
|
257
|
+
attn = arch.num_attention_heads
|
|
258
|
+
kv = arch.num_kv_heads
|
|
259
|
+
hidden = arch.hidden_size
|
|
260
|
+
|
|
261
|
+
if not layers or layers <= 0:
|
|
262
|
+
return None
|
|
263
|
+
|
|
264
|
+
from_attention = kv is None
|
|
265
|
+
heads = attn if from_attention else kv
|
|
266
|
+
if not heads or heads <= 0:
|
|
267
|
+
return None
|
|
268
|
+
|
|
269
|
+
# Schema has no head_dim. For MHA/GQA it is hidden_size / num_attention_heads.
|
|
270
|
+
if not attn or attn <= 0 or not hidden or hidden <= 0:
|
|
271
|
+
return None
|
|
272
|
+
if hidden % attn != 0:
|
|
273
|
+
return None
|
|
274
|
+
head_dim = hidden // attn
|
|
275
|
+
if head_dim <= 0:
|
|
276
|
+
return None
|
|
277
|
+
|
|
278
|
+
bytes_per_token = 2 * layers * heads * head_dim * KV_BYTES_PER_ELEMENT
|
|
279
|
+
return float(bytes_per_token), from_attention
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def kv_bytes_per_token(card: Any) -> float | None:
|
|
283
|
+
"""KV-cache bytes per token at KV_BYTES_PER_ELEMENT, or None if geometry is missing.
|
|
284
|
+
|
|
285
|
+
Independent of weight quantisation: the cache stays at fp16 even when the
|
|
286
|
+
weights are q4. GQA/MQA use `num_kv_heads`, not `num_attention_heads`.
|
|
287
|
+
"""
|
|
288
|
+
geo = _kv_geometry(card)
|
|
289
|
+
return None if geo is None else geo[0]
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def predicted_max_context(card: Any, capacity_gb: float, quant: str) -> MaxContext:
|
|
293
|
+
"""How many tokens of context the leftover memory can hold at `quant`.
|
|
294
|
+
|
|
295
|
+
`(device_memory - weights - working_allowance) / kv_bytes_per_token`, then
|
|
296
|
+
clamped to the model's own `context_window`. The clamp, not the raw device
|
|
297
|
+
figure, is what a reader can actually use.
|
|
298
|
+
"""
|
|
299
|
+
geo = _kv_geometry(card)
|
|
300
|
+
if geo is None:
|
|
301
|
+
return MaxContext(
|
|
302
|
+
tokens=None,
|
|
303
|
+
bound_by=None,
|
|
304
|
+
missing_geometry=True,
|
|
305
|
+
kv_heads_from_attention=False,
|
|
306
|
+
)
|
|
307
|
+
kv_bpt, from_attn = geo
|
|
308
|
+
params = card.architecture.total_parameters
|
|
309
|
+
device_memory_bytes = capacity_gb * 1e9
|
|
310
|
+
weight_bytes = weights_gb(params, quant) * 1e9
|
|
311
|
+
working_allowance_bytes = device_memory_bytes * WORKING_ALLOWANCE
|
|
312
|
+
available = device_memory_bytes - weight_bytes - working_allowance_bytes
|
|
313
|
+
device_tokens = 0 if available <= 0 else int(available / kv_bpt)
|
|
314
|
+
window = card.modalities.text.context_window
|
|
315
|
+
if window and window > 0 and device_tokens >= window:
|
|
316
|
+
return MaxContext(
|
|
317
|
+
tokens=int(window),
|
|
318
|
+
bound_by="model",
|
|
319
|
+
missing_geometry=False,
|
|
320
|
+
kv_heads_from_attention=from_attn,
|
|
321
|
+
)
|
|
322
|
+
return MaxContext(
|
|
323
|
+
tokens=device_tokens,
|
|
324
|
+
bound_by="device",
|
|
325
|
+
missing_geometry=False,
|
|
326
|
+
kv_heads_from_attention=from_attn,
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def compute(sink: CollectingSink, cards: list[Any], devices: list[Device]) -> dict[str, Any]:
|
|
331
|
+
"""Add Hardware nodes and FITS_ON edges. Returns counts for the summary."""
|
|
332
|
+
for device in devices:
|
|
333
|
+
sink.node("Hardware", "id", device.id, {
|
|
334
|
+
"id": device.id,
|
|
335
|
+
"display_name": device.display_name,
|
|
336
|
+
"vendor": device.vendor,
|
|
337
|
+
"device_class": device.device_class,
|
|
338
|
+
"memory_bandwidth_gb_s": device.bandwidth_gb_s,
|
|
339
|
+
"memory_gb": device.max_capacity_gb,
|
|
340
|
+
"memory_options_gb": ",".join(str(c) for c in device.capacity_options_gb),
|
|
341
|
+
"unified_memory": device.unified,
|
|
342
|
+
})
|
|
343
|
+
|
|
344
|
+
considered = fitted = skipped_closed = skipped_no_params = 0
|
|
345
|
+
edges = edges_with_max_context = 0
|
|
346
|
+
decode_predictions_omitted = 0
|
|
347
|
+
|
|
348
|
+
used_active = 0
|
|
349
|
+
for card in cards:
|
|
350
|
+
params = card.architecture.total_parameters
|
|
351
|
+
# Capacity and speed read different numbers. Every weight must be
|
|
352
|
+
# resident, so `fits` uses the total; only the active experts are read
|
|
353
|
+
# per token, so the decode prediction uses the active count.
|
|
354
|
+
active = card.architecture.active_parameters or params
|
|
355
|
+
if not card.licensing.open_weights:
|
|
356
|
+
skipped_closed += 1
|
|
357
|
+
continue
|
|
358
|
+
if not params:
|
|
359
|
+
skipped_no_params += 1
|
|
360
|
+
continue
|
|
361
|
+
considered += 1
|
|
362
|
+
is_moe = bool(card.architecture.num_experts)
|
|
363
|
+
has_active = bool(card.architecture.active_parameters)
|
|
364
|
+
if has_active:
|
|
365
|
+
used_active += 1
|
|
366
|
+
fitted_any = False
|
|
367
|
+
# Whether the weights fit is meaningful for every open-weights model.
|
|
368
|
+
# Decode tok/s is only meaningful for a model that decodes tokens.
|
|
369
|
+
decode_eligible = is_token_generating(card.identity.model_type)
|
|
370
|
+
|
|
371
|
+
for device in devices:
|
|
372
|
+
if not device.single_device_fit:
|
|
373
|
+
continue
|
|
374
|
+
capacity = device.max_capacity_gb
|
|
375
|
+
quants = fitting_quants(params, capacity)
|
|
376
|
+
if not quants:
|
|
377
|
+
continue
|
|
378
|
+
fitted_any = True
|
|
379
|
+
edges += 1
|
|
380
|
+
# Report the range, not a single "best". Highest quality that fits and
|
|
381
|
+
# smallest that fits answer different questions, and quoting only the
|
|
382
|
+
# first makes a large slow machine look worse than a small fast one
|
|
383
|
+
# purely because it chose a heavier quantisation.
|
|
384
|
+
best, smallest = quants[0], quants[-1]
|
|
385
|
+
ctx = predicted_max_context(card, capacity, best)
|
|
386
|
+
if ctx.tokens is not None:
|
|
387
|
+
edges_with_max_context += 1
|
|
388
|
+
if not decode_eligible:
|
|
389
|
+
decode_predictions_omitted += 1
|
|
390
|
+
# Store unrounded GB. Rounding to 2 decimals collapsed 616k-param
|
|
391
|
+
# bf16 weights to 0.0, which the page then printed as "0.0 GB".
|
|
392
|
+
sink.edge("Model", card.identity.model_id, "FITS_ON", "Hardware", device.id, {
|
|
393
|
+
"quantization": best,
|
|
394
|
+
"weights_gb": weights_gb(params, best),
|
|
395
|
+
"device_memory_gb": capacity,
|
|
396
|
+
"predicted_decode_tps": predicted_decode_tps(
|
|
397
|
+
device.bandwidth_gb_s, active, best) if decode_eligible else None,
|
|
398
|
+
"decode_reads_params": active if decode_eligible else None,
|
|
399
|
+
"fastest_quantization": smallest,
|
|
400
|
+
"fastest_weights_gb": weights_gb(params, smallest),
|
|
401
|
+
"fastest_predicted_decode_tps": predicted_decode_tps(
|
|
402
|
+
device.bandwidth_gb_s, active, smallest) if decode_eligible else None,
|
|
403
|
+
"quantizations_that_fit": ",".join(quants),
|
|
404
|
+
"max_context_at_quant": ctx.tokens,
|
|
405
|
+
"max_context_bound_by": ctx.bound_by,
|
|
406
|
+
"max_context_missing_geometry": ctx.missing_geometry,
|
|
407
|
+
"kv_heads_from_attention_heads": ctx.kv_heads_from_attention,
|
|
408
|
+
# Everything above is arithmetic, not observation. The page and
|
|
409
|
+
# the export must never render it as a measurement.
|
|
410
|
+
"basis": "computed",
|
|
411
|
+
"assumes_working_allowance": WORKING_ALLOWANCE,
|
|
412
|
+
"assumes_bandwidth_efficiency": BANDWIDTH_EFFICIENCY,
|
|
413
|
+
# No card carries active_parameters, so an MoE prediction uses
|
|
414
|
+
# total parameters and understates its real speed.
|
|
415
|
+
# Only conservative where the active count is still unknown.
|
|
416
|
+
"moe_prediction_is_conservative": decode_eligible and is_moe and not has_active,
|
|
417
|
+
})
|
|
418
|
+
fitted += 1 if fitted_any else 0
|
|
419
|
+
|
|
420
|
+
return {
|
|
421
|
+
"devices": len(devices),
|
|
422
|
+
"models_considered": considered,
|
|
423
|
+
"models_fitting_somewhere": fitted,
|
|
424
|
+
"skipped_closed_weights": skipped_closed,
|
|
425
|
+
"skipped_no_parameter_count": skipped_no_params,
|
|
426
|
+
"skipped_single_device_fit": sum(
|
|
427
|
+
1 for device in devices if not device.single_device_fit),
|
|
428
|
+
"edges": edges,
|
|
429
|
+
"edges_with_max_context": edges_with_max_context,
|
|
430
|
+
"decode_predictions_omitted": decode_predictions_omitted,
|
|
431
|
+
"working_allowance": WORKING_ALLOWANCE,
|
|
432
|
+
"bandwidth_efficiency": BANDWIDTH_EFFICIENCY,
|
|
433
|
+
"used_active_parameters": used_active,
|
|
434
|
+
}
|
pipeline/hosts.py
ADDED
|
@@ -0,0 +1,247 @@
|
|
|
1
|
+
"""The host layer (MODEL-26 phase B): offload-aware fit for a (model, device, host).
|
|
2
|
+
|
|
3
|
+
A host is the machine around an accelerator (`hosts/*.yaml`, see
|
|
4
|
+
docs/host-layer.md). For a discrete host, weights that do not fit in
|
|
5
|
+
accelerator memory can spill to system RAM. Decode is then bounded by the
|
|
6
|
+
slower pool, so the answer is "fits, slowly", not "does not fit".
|
|
7
|
+
|
|
8
|
+
Three fit states:
|
|
9
|
+
|
|
10
|
+
* `accelerator`: `W + A <= C_acc`, today's `fits: true`.
|
|
11
|
+
* `offload`: not `accelerator`, `W + A <= C_acc + C_host`, and the host is not
|
|
12
|
+
unified. It carries `offload_fraction = (W + A - C_acc) / W`.
|
|
13
|
+
* `does_not_fit`: neither. A unified host has `C_host = 0` by definition, so
|
|
14
|
+
its fit is two-state.
|
|
15
|
+
|
|
16
|
+
`W` is weights at the quant, `A = WORKING_ALLOWANCE * C_acc`, and
|
|
17
|
+
`C_host = host RAM - OS_RESERVE_GB`. Host RAM is the profile's
|
|
18
|
+
`capacity_max_gb` unless the caller states what this machine has
|
|
19
|
+
(`--host-ram`).
|
|
20
|
+
|
|
21
|
+
Everything here is computed, never measured. The offload formula assumes
|
|
22
|
+
layer-split offload with the CPU not compute-bound; see "Known wrong" in
|
|
23
|
+
docs/host-layer.md.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import json
|
|
29
|
+
from dataclasses import dataclass
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
from typing import Any
|
|
32
|
+
|
|
33
|
+
import yaml
|
|
34
|
+
|
|
35
|
+
from pipeline.hardware import (
|
|
36
|
+
BANDWIDTH_EFFICIENCY,
|
|
37
|
+
QUANT_PREFERENCE,
|
|
38
|
+
WORKING_ALLOWANCE,
|
|
39
|
+
is_token_generating,
|
|
40
|
+
predicted_decode_tps,
|
|
41
|
+
weights_gb,
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
#: RAM held back for the OS and everything that is not model weights. One
|
|
45
|
+
#: constant, not a per-OS table (Jamie, 2026-09-15). `--host-ram` overrides the
|
|
46
|
+
#: RAM figure the reserve is taken from, not the reserve itself.
|
|
47
|
+
OS_RESERVE_GB = 8.0
|
|
48
|
+
|
|
49
|
+
FIT_ACCELERATOR = "accelerator"
|
|
50
|
+
FIT_OFFLOAD = "offload"
|
|
51
|
+
FIT_NONE = "does_not_fit"
|
|
52
|
+
|
|
53
|
+
BASIS_ACCELERATOR = "accelerator-roofline"
|
|
54
|
+
BASIS_OFFLOAD = "offload-roofline"
|
|
55
|
+
|
|
56
|
+
_SECTIONS = {
|
|
57
|
+
"cpu": {"model", "cores", "threads"},
|
|
58
|
+
"system_memory": {"type", "channels", "max_speed_mt_s", "capacity_max_gb",
|
|
59
|
+
"capacity_options_gb", "bandwidth_gb_s", "bandwidth_derivation"},
|
|
60
|
+
"pcie": {"cpu_gen", "cpu_lanes_usable", "accelerator_link_gen", "accelerator_link_width"},
|
|
61
|
+
"storage": {"class", "capacity_options_gb"},
|
|
62
|
+
}
|
|
63
|
+
_TOP = {"id", "display_name", "kind", "unified", "hardware_ref", *_SECTIONS,
|
|
64
|
+
"field_sources", "figures_are", "notes"}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass(frozen=True)
|
|
68
|
+
class Host:
|
|
69
|
+
id: str
|
|
70
|
+
display_name: str
|
|
71
|
+
kind: str
|
|
72
|
+
unified: bool
|
|
73
|
+
hardware_ref: str | None
|
|
74
|
+
capacity_max_gb: float | None
|
|
75
|
+
bandwidth_gb_s: float | None
|
|
76
|
+
raw: dict[str, Any]
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _numeric(v: Any) -> bool:
|
|
80
|
+
if isinstance(v, bool):
|
|
81
|
+
return False
|
|
82
|
+
if isinstance(v, (int, float)):
|
|
83
|
+
return True
|
|
84
|
+
return isinstance(v, list) and bool(v) and all(_numeric(x) for x in v)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def validate(raw: Any, name: str) -> None:
|
|
88
|
+
"""Raise ValueError when a profile breaks hosts/_schema.yaml."""
|
|
89
|
+
if not isinstance(raw, dict):
|
|
90
|
+
raise ValueError(f"{name}: a host profile must be a mapping")
|
|
91
|
+
if set(raw) != _TOP:
|
|
92
|
+
raise ValueError(f"{name}: top-level keys differ from the schema: {sorted(set(raw) ^ _TOP)}")
|
|
93
|
+
if raw["kind"] not in {"platform", "system"}:
|
|
94
|
+
raise ValueError(f"{name}: kind must be platform or system")
|
|
95
|
+
if not isinstance(raw["unified"], bool):
|
|
96
|
+
raise ValueError(f"{name}: unified is required and must be a boolean")
|
|
97
|
+
for section, keys in _SECTIONS.items():
|
|
98
|
+
if not isinstance(raw[section], dict) or set(raw[section]) != keys:
|
|
99
|
+
raise ValueError(f"{name}: {section} keys differ from the schema")
|
|
100
|
+
if raw["unified"] and (raw["system_memory"]["type"] != "unified"
|
|
101
|
+
or raw["pcie"]["accelerator_link_gen"] is not None):
|
|
102
|
+
raise ValueError(f"{name}: a unified host has memory type unified and no accelerator link")
|
|
103
|
+
sources = raw["field_sources"] or {}
|
|
104
|
+
numeric = {f"{s}.{k}" for s in _SECTIONS for k, v in raw[s].items() if _numeric(v)}
|
|
105
|
+
missing = numeric - set(sources)
|
|
106
|
+
if missing:
|
|
107
|
+
raise ValueError(f"{name}: unsourced numeric fields {sorted(missing)}")
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def load_hosts(root: Path) -> list[Host]:
|
|
111
|
+
"""Read and validate hosts/*.yaml. An invalid profile is rejected, not defaulted."""
|
|
112
|
+
out: list[Host] = []
|
|
113
|
+
for path in sorted((root / "hosts").glob("*.yaml")):
|
|
114
|
+
if path.name.startswith("_"):
|
|
115
|
+
continue
|
|
116
|
+
raw = yaml.safe_load(path.read_text(encoding="utf-8"))
|
|
117
|
+
validate(raw, path.name)
|
|
118
|
+
if raw["id"] != path.stem:
|
|
119
|
+
raise ValueError(f"{path.name}: id {raw['id']!r} does not match the file name")
|
|
120
|
+
out.append(host_from_raw(raw))
|
|
121
|
+
return out
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def host_from_raw(raw: dict[str, Any]) -> Host:
|
|
125
|
+
mem = raw["system_memory"]
|
|
126
|
+
return Host(
|
|
127
|
+
id=raw["id"], display_name=raw["display_name"], kind=raw["kind"],
|
|
128
|
+
unified=bool(raw["unified"]), hardware_ref=raw.get("hardware_ref"),
|
|
129
|
+
capacity_max_gb=float(mem["capacity_max_gb"]) if mem.get("capacity_max_gb") else None,
|
|
130
|
+
bandwidth_gb_s=float(mem["bandwidth_gb_s"]) if mem.get("bandwidth_gb_s") else None,
|
|
131
|
+
raw=raw,
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def host_capacity_gb(host: Host, host_ram_gb: float | None = None) -> float:
|
|
136
|
+
"""`C_host`: host memory available for spilled weights. 0 for a unified host."""
|
|
137
|
+
if host.unified:
|
|
138
|
+
return 0.0
|
|
139
|
+
ram = host_ram_gb if host_ram_gb is not None else host.capacity_max_gb
|
|
140
|
+
if ram is None:
|
|
141
|
+
return 0.0
|
|
142
|
+
return max(0.0, float(ram) - OS_RESERVE_GB)
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def fit_state(weights: float, accelerator_gb: float, host_gb: float, unified: bool) -> str:
|
|
146
|
+
need = weights + WORKING_ALLOWANCE * accelerator_gb
|
|
147
|
+
if need <= accelerator_gb:
|
|
148
|
+
return FIT_ACCELERATOR
|
|
149
|
+
if not unified and need <= accelerator_gb + host_gb:
|
|
150
|
+
return FIT_OFFLOAD
|
|
151
|
+
return FIT_NONE
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def offload_fraction(weights: float, accelerator_gb: float) -> float:
|
|
155
|
+
"""`f = (W + A - C_acc) / W`, clamped to [0, 1]. 0 when everything is resident."""
|
|
156
|
+
if weights <= 0:
|
|
157
|
+
return 0.0
|
|
158
|
+
f = (weights + WORKING_ALLOWANCE * accelerator_gb - accelerator_gb) / weights
|
|
159
|
+
return min(1.0, max(0.0, f))
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def offload_decode_tps(accelerator_bandwidth: float, host_bandwidth: float,
|
|
163
|
+
params: float, quant: str, fraction: float) -> float:
|
|
164
|
+
"""`BANDWIDTH_EFFICIENCY / ((1-f)*P/B_acc + f*P/B_host)`, rounded like the roofline.
|
|
165
|
+
|
|
166
|
+
At f = 0 this is exactly `predicted_decode_tps`: the same function answers,
|
|
167
|
+
so no floating-point path can make the two disagree.
|
|
168
|
+
"""
|
|
169
|
+
if fraction <= 0:
|
|
170
|
+
return predicted_decode_tps(accelerator_bandwidth, params, quant)
|
|
171
|
+
per_token = weights_gb(params, quant)
|
|
172
|
+
if per_token <= 0:
|
|
173
|
+
return 0.0
|
|
174
|
+
seconds = (1 - fraction) * per_token / accelerator_bandwidth + fraction * per_token / host_bandwidth
|
|
175
|
+
return round(BANDWIDTH_EFFICIENCY / seconds, 1)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def assess(*, total_params: float, active_params: float | None, model_type: Any,
|
|
179
|
+
accelerator_gb: float, accelerator_bandwidth: float,
|
|
180
|
+
host: Host, host_ram_gb: float | None = None) -> dict[str, Any]:
|
|
181
|
+
"""Fit state and decode prediction for one (model, device, host) triple.
|
|
182
|
+
|
|
183
|
+
The quant chosen is the smallest that reaches the best state: for
|
|
184
|
+
`accelerator` it matches `fastest_quantization`, the figure `fits` carries;
|
|
185
|
+
for `offload` it spills the least, which is the fastest offload.
|
|
186
|
+
Decode tps is null for a model that does not decode tokens, and for an
|
|
187
|
+
offload row whose host has no published bandwidth.
|
|
188
|
+
"""
|
|
189
|
+
host_gb = host_capacity_gb(host, host_ram_gb)
|
|
190
|
+
reads = active_params or total_params
|
|
191
|
+
best: tuple[str, str] | None = None
|
|
192
|
+
for state in (FIT_ACCELERATOR, FIT_OFFLOAD):
|
|
193
|
+
ok = [q for q in QUANT_PREFERENCE
|
|
194
|
+
if fit_state(weights_gb(total_params, q), accelerator_gb, host_gb, host.unified) == state]
|
|
195
|
+
if ok:
|
|
196
|
+
best = (state, ok[-1])
|
|
197
|
+
break
|
|
198
|
+
if best is None:
|
|
199
|
+
return {"fit_state": FIT_NONE, "quantization": None, "offload_fraction": None,
|
|
200
|
+
"predicted_decode_tps": None, "predicted_decode_tps_basis": None}
|
|
201
|
+
state, quant = best
|
|
202
|
+
tokens = is_token_generating(model_type)
|
|
203
|
+
if state == FIT_ACCELERATOR:
|
|
204
|
+
f = 0.0
|
|
205
|
+
tps = predicted_decode_tps(accelerator_bandwidth, reads, quant) if tokens else None
|
|
206
|
+
basis = BASIS_ACCELERATOR
|
|
207
|
+
else:
|
|
208
|
+
f = round(offload_fraction(weights_gb(total_params, quant), accelerator_gb), 4)
|
|
209
|
+
tps = (offload_decode_tps(accelerator_bandwidth, host.bandwidth_gb_s, reads, quant, f)
|
|
210
|
+
if tokens and host.bandwidth_gb_s else None)
|
|
211
|
+
basis = BASIS_OFFLOAD
|
|
212
|
+
return {"fit_state": state, "quantization": quant, "offload_fraction": f,
|
|
213
|
+
"predicted_decode_tps": tps,
|
|
214
|
+
"predicted_decode_tps_basis": basis if tps is not None else None}
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def write_export(api_dir: Path, hosts: list[Host], build_json: dict[str, Any]) -> dict[str, Any]:
|
|
218
|
+
"""Emit `api/hosts.json`: every profile as authored, plus the reserve used."""
|
|
219
|
+
api_dir.mkdir(parents=True, exist_ok=True)
|
|
220
|
+
payload = {
|
|
221
|
+
"build": build_json,
|
|
222
|
+
"count": len(hosts),
|
|
223
|
+
"os_reserve_gb": OS_RESERVE_GB,
|
|
224
|
+
"hosts": [h.raw for h in hosts],
|
|
225
|
+
}
|
|
226
|
+
(api_dir / "hosts.json").write_text(
|
|
227
|
+
json.dumps(payload, indent=1, sort_keys=True, default=str), encoding="utf-8")
|
|
228
|
+
return {"hosts": len(hosts)}
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def add_to_graph(sink: Any, hosts: list[Host]) -> int:
|
|
232
|
+
"""`:Host` nodes and `(:Host)-[:HOSTS]->(:Hardware)` for unified hosts only.
|
|
233
|
+
|
|
234
|
+
A discrete pairing is a query-time input, not a stored record, so discrete
|
|
235
|
+
hosts get no node. Not wired into the published graph yet (see
|
|
236
|
+
docs/graph-ontology.md). Returns the number of HOSTS edges.
|
|
237
|
+
"""
|
|
238
|
+
edges = 0
|
|
239
|
+
for host in hosts:
|
|
240
|
+
if not host.unified:
|
|
241
|
+
continue
|
|
242
|
+
sink.node("Host", "id", host.id, {"id": host.id, "display_name": host.display_name,
|
|
243
|
+
"unified": True, "kind": host.kind})
|
|
244
|
+
if host.hardware_ref:
|
|
245
|
+
sink.edge("Host", host.id, "HOSTS", "Hardware", host.hardware_ref, {})
|
|
246
|
+
edges += 1
|
|
247
|
+
return edges
|