modelspec-dev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. api/__init__.py +0 -0
  2. api/class_fit.py +334 -0
  3. api/classes.py +557 -0
  4. api/ranking/__init__.py +12 -0
  5. api/ranking/engine.py +1943 -0
  6. cli/__init__.py +0 -0
  7. cli/modelspec/__init__.py +0 -0
  8. cli/modelspec/cli.py +1819 -0
  9. cli/modelspec/commands/__init__.py +0 -0
  10. cli/modelspec/decide_cmd.py +333 -0
  11. cli/modelspec/offline.py +623 -0
  12. cli/modelspec/snapshot.py +698 -0
  13. cli/modelspec/snapshot_build_cmd.py +49 -0
  14. cli/modelspec/verify_cmd.py +125 -0
  15. cli/modelspec/vocab_cmd.py +204 -0
  16. cli/modelspec/vocabulary_cache.py +54 -0
  17. decision/__init__.py +13 -0
  18. decision/capability.py +872 -0
  19. decision/computed.py +125 -0
  20. decision/contract.py +1575 -0
  21. decision/engine.py +238 -0
  22. decision/excluded.py +34 -0
  23. decision/explain.py +908 -0
  24. decision/filter.py +796 -0
  25. decision/model.py +438 -0
  26. decision/normalise.py +604 -0
  27. decision/optimise.py +320 -0
  28. decision/registry.py +717 -0
  29. decision/relax.py +132 -0
  30. decision/resolve.py +111 -0
  31. decision/schema.py +21 -0
  32. decision/snapshot.py +1483 -0
  33. decision/sources.py +544 -0
  34. decision/templates.py +134 -0
  35. decision/verify.py +1745 -0
  36. decision/vocabulary.py +433 -0
  37. modelspec_dev-0.1.0.dist-info/METADATA +101 -0
  38. modelspec_dev-0.1.0.dist-info/RECORD +63 -0
  39. modelspec_dev-0.1.0.dist-info/WHEEL +4 -0
  40. modelspec_dev-0.1.0.dist-info/entry_points.txt +2 -0
  41. modelspec_dev-0.1.0.dist-info/licenses/LICENSE +43 -0
  42. modelspec_dev-0.1.0.dist-info/licenses/LICENSE-DATA +428 -0
  43. pipeline/__init__.py +0 -0
  44. pipeline/class_export.py +172 -0
  45. pipeline/hardware.py +434 -0
  46. pipeline/hosts.py +247 -0
  47. pipeline/load.py +224 -0
  48. pipeline/ranking.py +551 -0
  49. registry/domains.yaml +130 -0
  50. registry/facets.yaml +888 -0
  51. registry/harnesses.yaml +79 -0
  52. registry/providers.yaml +354 -0
  53. registry/sources.yaml +3059 -0
  54. registry/templates.yaml +166 -0
  55. schema/__init__.py +0 -0
  56. schema/applicability.py +147 -0
  57. schema/benchmark.py +175 -0
  58. schema/benchmark_eligibility.py +304 -0
  59. schema/card.py +1463 -0
  60. schema/enrichment.py +162 -0
  61. schema/enums.py +327 -0
  62. schema/graph.py +406 -0
  63. schema/suppliers.py +72 -0
pipeline/hardware.py ADDED
@@ -0,0 +1,434 @@
1
+ """Compute which models fit on which devices, and how fast they would decode.
2
+
3
+ Everything here is *computed*, never measured, and is labelled as such all the
4
+ way to the page. A predicted figure presented as a measurement is a lie a reader
5
+ will plan around.
6
+
7
+ Decode speed for a mixture-of-experts model depends on *active* parameters, not
8
+ total. With no active-parameter data the prediction uses total, which understates
9
+ MoE speed — sometimes by a large factor. Every prediction says so.
10
+
11
+ Max context is leftover memory after weights and the working allowance, divided
12
+ by the KV-cache bytes per token. That needs layer and head geometry
13
+ (`num_layers`, `num_kv_heads` or `num_attention_heads`, `hidden_size`). The
14
+ schema has no `head_dim`; it is derived as `hidden_size / num_attention_heads`.
15
+ When any of that is missing, `max_context_at_quant` is null — never zero, never
16
+ a guess — and `max_context_missing_geometry` is true so a consumer can tell
17
+ "we do not know" from "it does not fit".
18
+
19
+ `FITS_ON` is only computed for open-weights models. Asking whether a
20
+ closed-weights model "fits" on your GPU is meaningless — you cannot obtain it.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ from dataclasses import dataclass
26
+ from pathlib import Path
27
+ from typing import Any
28
+
29
+ import yaml
30
+
31
+ from schema.enums import DeviceClass, ModelType
32
+ from schema.graph import CollectingSink
33
+
34
+ #: model_type values that autoregressively decode tokens, so a tok/s figure is
35
+ #: a meaningful prediction. Everything else (embeddings, rerankers, safety
36
+ #: classifiers, generation of another modality, OCR, reward models, and the
37
+ #: perception/encoder/time-series types) still gets a capacity FITS_ON answer
38
+ #: — the weights either fit in memory or they don't — but never a decode
39
+ #: speed, because there is no token being decoded. A card with no model_type
40
+ #: is treated the same as a non-token type: absence is not a guess that it
41
+ #: chats. Single source of truth for this distinction — do not scatter
42
+ #: per-type checks elsewhere.
43
+ TOKEN_GENERATING_MODEL_TYPES: frozenset[ModelType] = frozenset({
44
+ ModelType.LLM_CHAT, ModelType.LLM_REASONING, ModelType.LLM_CODE, ModelType.LLM_BASE,
45
+ ModelType.VLM, ModelType.MEDICAL, ModelType.LEGAL, ModelType.FINANCIAL,
46
+ ModelType.AGENT_MODEL, ModelType.ROUTER,
47
+ ModelType.ADAPTER, ModelType.QUANTIZED_VARIANT, ModelType.DISTILLED, ModelType.MERGED,
48
+ })
49
+
50
+
51
+ def is_token_generating(model_type: Any) -> bool:
52
+ """Whether `model_type` decodes tokens and so gets a decode-speed prediction.
53
+
54
+ Accepts the raw `Identity.model_type` (a `ModelType | None`, possibly
55
+ already a plain string in code that reads snapshots rather than cards).
56
+ """
57
+ if model_type is None:
58
+ return False
59
+ if isinstance(model_type, ModelType):
60
+ return model_type in TOKEN_GENERATING_MODEL_TYPES
61
+ return str(model_type) in {t.value for t in TOKEN_GENERATING_MODEL_TYPES}
62
+
63
+
64
+ #: Bytes per parameter at each quantisation, best case. Real files carry
65
+ #: metadata and some layers stay at higher precision, which the working
66
+ #: allowance below absorbs.
67
+ QUANT_BYTES: dict[str, float] = {
68
+ "fp16": 2.0, "bf16": 2.0, "fp8": 1.0, "int8": 1.0,
69
+ "q6": 0.75, "q5": 0.625, "int4": 0.5, "q4": 0.5,
70
+ }
71
+
72
+ #: Preference order when choosing the best quantisation that fits. Higher
73
+ #: precision first: we want the best quality that fits, not the smallest.
74
+ QUANT_PREFERENCE = ("bf16", "fp16", "fp8", "int8", "q6", "q5", "q4")
75
+
76
+ #: Headroom for activations, the framework and the OS, as a fraction of device
77
+ #: memory. KV-cache size is computed from layer geometry when the card has it;
78
+ #: this allowance is the rest of the working set, not a stand-in for the cache.
79
+ WORKING_ALLOWANCE = 0.25
80
+
81
+ #: Bytes per KV-cache element. The cache is commonly kept at fp16 even when the
82
+ #: weights are quantised; using QUANT_BYTES for the weight quant would understate
83
+ #: the cache (and overstate how much context fits) at q4/int8.
84
+ KV_BYTES_PER_ELEMENT = 2.0
85
+
86
+ #: Real bandwidth utilisation. No decoder achieves the theoretical roofline;
87
+ #: measured llama.cpp and vLLM figures typically land in the 60-80% band.
88
+ BANDWIDTH_EFFICIENCY = 0.70
89
+
90
+
91
+ @dataclass(frozen=True)
92
+ class Device:
93
+ id: str
94
+ display_name: str
95
+ vendor: str
96
+ device_class: str
97
+ bandwidth_gb_s: float
98
+ capacity_options_gb: tuple[float, ...]
99
+ precisions_native: tuple[str, ...]
100
+ unified: bool
101
+ single_device_fit: bool = True
102
+ single_device_fit_reason: str | None = None
103
+
104
+ @property
105
+ def max_capacity_gb(self) -> float:
106
+ return max(self.capacity_options_gb)
107
+
108
+
109
+ def _single_device_fit(raw: dict[str, Any], path: Path) -> tuple[bool, str | None]:
110
+ """DATA flag: false means this part must not answer single-device FITS_ON.
111
+
112
+ Default is true (omit the field). false requires a non-empty reason so a
113
+ reader can see why the layer refused rather than guessing from the device id.
114
+ """
115
+ if "single_device_fit" not in raw:
116
+ return True, None
117
+ flag = raw["single_device_fit"]
118
+ if flag is True:
119
+ reason = raw.get("single_device_fit_reason")
120
+ if reason is None:
121
+ return True, None
122
+ if not isinstance(reason, str) or not reason.strip():
123
+ raise ValueError(
124
+ f"{path.name}: single_device_fit_reason must be a non-empty string when present"
125
+ )
126
+ return True, reason.strip()
127
+ if flag is False:
128
+ reason = raw.get("single_device_fit_reason")
129
+ if not isinstance(reason, str) or not reason.strip():
130
+ raise ValueError(
131
+ f"{path.name}: single_device_fit is false but single_device_fit_reason "
132
+ "is missing. A part that cannot answer single-device fit must say why."
133
+ )
134
+ return False, reason.strip()
135
+ raise ValueError(
136
+ f"{path.name}: single_device_fit must be true or false, not {flag!r}"
137
+ )
138
+
139
+
140
+ def _device_class(raw: dict[str, Any], path: Path) -> str:
141
+ """The record's `device_class`, checked against the vocabulary (MODEL-76).
142
+
143
+ The graph groups Hardware nodes by this value, so `datacenter` or `Consumer`
144
+ would not be a near miss — it would be a sixth class that every query for
145
+ the real one silently skips. Rejected at load, where the file name is still
146
+ in hand to name in the error.
147
+ """
148
+ value = raw.get("device_class")
149
+ try:
150
+ return DeviceClass(value).value
151
+ except ValueError:
152
+ allowed = ", ".join(c.value for c in DeviceClass)
153
+ raise ValueError(
154
+ f"{path.name}: device_class {value!r} is not one of {allowed}"
155
+ ) from None
156
+
157
+
158
+ @dataclass(frozen=True)
159
+ class MaxContext:
160
+ """Predicted context length at one (device, quant) pair.
161
+
162
+ `tokens` is null only when geometry is missing. Zero means the weights fit
163
+ but leftover memory cannot hold a single token of KV — that is not the
164
+ same as "we do not know".
165
+ """
166
+ tokens: int | None
167
+ bound_by: str | None
168
+ missing_geometry: bool
169
+ kv_heads_from_attention: bool
170
+
171
+
172
+ def load_devices(root: Path) -> list[Device]:
173
+ """Read hardware/*.yaml. A device without bandwidth is rejected, not defaulted."""
174
+ out: list[Device] = []
175
+ for path in sorted((root / "hardware").glob("*.yaml")):
176
+ if path.name.startswith("_"):
177
+ continue
178
+ raw = yaml.safe_load(path.read_text(encoding="utf-8"))
179
+ memory = raw["memory"]
180
+ bandwidth = memory.get("bandwidth_gb_s")
181
+ if not bandwidth:
182
+ raise ValueError(
183
+ f"{path.name}: no memory.bandwidth_gb_s. A device without bandwidth "
184
+ "cannot answer how fast a model will run, which is the question this "
185
+ "layer exists to answer. Fix the definition or remove it."
186
+ )
187
+ options = memory.get("capacity_options_gb") or [memory["capacity_gb"]]
188
+ fit, fit_reason = _single_device_fit(raw, path)
189
+ out.append(Device(
190
+ id=raw["id"], display_name=raw["display_name"], vendor=raw["vendor"],
191
+ device_class=_device_class(raw, path), bandwidth_gb_s=float(bandwidth),
192
+ capacity_options_gb=tuple(float(c) for c in options),
193
+ precisions_native=tuple(raw.get("precisions_native") or []),
194
+ unified=bool(memory.get("unified_with_host")),
195
+ single_device_fit=fit,
196
+ single_device_fit_reason=fit_reason,
197
+ ))
198
+ return out
199
+
200
+
201
+ def device_classes(devices: list[Device]) -> dict[str, str]:
202
+ """Device id → `DeviceClass` value, for the graph derivation (MODEL-76).
203
+
204
+ `schema.graph` sits below this loader and cannot read `hardware/*.yaml`
205
+ itself, so it takes this mapping and stamps the class onto the Hardware
206
+ nodes a card's deployment profiles produce. One source of truth for the
207
+ class, whether the graph is being exported as JSON or ingested into
208
+ FalkorDB.
209
+ """
210
+ return {device.id: device.device_class for device in devices}
211
+
212
+
213
+ def weights_gb(params: float, quant: str) -> float:
214
+ return params * QUANT_BYTES[quant] / 1e9
215
+
216
+
217
+ def fitting_quants(params: float, capacity_gb: float) -> list[str]:
218
+ """Every quantisation whose weights fit in the usable memory, best quality first.
219
+
220
+ A quantisation the silicon does not accelerate natively is still allowed — it
221
+ runs, just without the speed benefit — so this gates on memory, not on
222
+ `precisions_native`. That field informs the prediction, not the fit.
223
+ """
224
+ usable = capacity_gb * (1.0 - WORKING_ALLOWANCE)
225
+ return [q for q in QUANT_PREFERENCE if weights_gb(params, q) <= usable]
226
+
227
+
228
+ def best_quant(params: float, capacity_gb: float, native: tuple[str, ...] = ()) -> str | None:
229
+ """The highest-quality quantisation that fits, or None."""
230
+ quants = fitting_quants(params, capacity_gb)
231
+ return quants[0] if quants else None
232
+
233
+
234
+ def predicted_decode_tps(bandwidth_gb_s: float, params: float, quant: str) -> float:
235
+ """Roofline estimate: bandwidth divided by the bytes read per token.
236
+
237
+ `params` must be the parameters actually *read* to emit a token, which for a
238
+ mixture-of-experts model is its active count, not its total. The difference
239
+ is not marginal: qwen3-coder-next holds 480B and activates 3.2B, a factor of
240
+ 148. Using total parameters there understates its speed by that factor and
241
+ makes the architecture built for speed look like the worst local choice.
242
+ """
243
+ per_token_gb = weights_gb(params, quant)
244
+ if per_token_gb <= 0:
245
+ return 0.0
246
+ return round(bandwidth_gb_s * BANDWIDTH_EFFICIENCY / per_token_gb, 1)
247
+
248
+
249
+ def _kv_geometry(card: Any) -> tuple[float, bool] | None:
250
+ """(bytes_per_token, kv_heads_taken_from_attention) or None.
251
+
252
+ Never infers missing layer counts or hidden size. Absent `num_kv_heads` is
253
+ MHA, so it falls back to `num_attention_heads` and reports that it did.
254
+ """
255
+ arch = card.architecture
256
+ layers = arch.num_layers
257
+ attn = arch.num_attention_heads
258
+ kv = arch.num_kv_heads
259
+ hidden = arch.hidden_size
260
+
261
+ if not layers or layers <= 0:
262
+ return None
263
+
264
+ from_attention = kv is None
265
+ heads = attn if from_attention else kv
266
+ if not heads or heads <= 0:
267
+ return None
268
+
269
+ # Schema has no head_dim. For MHA/GQA it is hidden_size / num_attention_heads.
270
+ if not attn or attn <= 0 or not hidden or hidden <= 0:
271
+ return None
272
+ if hidden % attn != 0:
273
+ return None
274
+ head_dim = hidden // attn
275
+ if head_dim <= 0:
276
+ return None
277
+
278
+ bytes_per_token = 2 * layers * heads * head_dim * KV_BYTES_PER_ELEMENT
279
+ return float(bytes_per_token), from_attention
280
+
281
+
282
+ def kv_bytes_per_token(card: Any) -> float | None:
283
+ """KV-cache bytes per token at KV_BYTES_PER_ELEMENT, or None if geometry is missing.
284
+
285
+ Independent of weight quantisation: the cache stays at fp16 even when the
286
+ weights are q4. GQA/MQA use `num_kv_heads`, not `num_attention_heads`.
287
+ """
288
+ geo = _kv_geometry(card)
289
+ return None if geo is None else geo[0]
290
+
291
+
292
+ def predicted_max_context(card: Any, capacity_gb: float, quant: str) -> MaxContext:
293
+ """How many tokens of context the leftover memory can hold at `quant`.
294
+
295
+ `(device_memory - weights - working_allowance) / kv_bytes_per_token`, then
296
+ clamped to the model's own `context_window`. The clamp, not the raw device
297
+ figure, is what a reader can actually use.
298
+ """
299
+ geo = _kv_geometry(card)
300
+ if geo is None:
301
+ return MaxContext(
302
+ tokens=None,
303
+ bound_by=None,
304
+ missing_geometry=True,
305
+ kv_heads_from_attention=False,
306
+ )
307
+ kv_bpt, from_attn = geo
308
+ params = card.architecture.total_parameters
309
+ device_memory_bytes = capacity_gb * 1e9
310
+ weight_bytes = weights_gb(params, quant) * 1e9
311
+ working_allowance_bytes = device_memory_bytes * WORKING_ALLOWANCE
312
+ available = device_memory_bytes - weight_bytes - working_allowance_bytes
313
+ device_tokens = 0 if available <= 0 else int(available / kv_bpt)
314
+ window = card.modalities.text.context_window
315
+ if window and window > 0 and device_tokens >= window:
316
+ return MaxContext(
317
+ tokens=int(window),
318
+ bound_by="model",
319
+ missing_geometry=False,
320
+ kv_heads_from_attention=from_attn,
321
+ )
322
+ return MaxContext(
323
+ tokens=device_tokens,
324
+ bound_by="device",
325
+ missing_geometry=False,
326
+ kv_heads_from_attention=from_attn,
327
+ )
328
+
329
+
330
+ def compute(sink: CollectingSink, cards: list[Any], devices: list[Device]) -> dict[str, Any]:
331
+ """Add Hardware nodes and FITS_ON edges. Returns counts for the summary."""
332
+ for device in devices:
333
+ sink.node("Hardware", "id", device.id, {
334
+ "id": device.id,
335
+ "display_name": device.display_name,
336
+ "vendor": device.vendor,
337
+ "device_class": device.device_class,
338
+ "memory_bandwidth_gb_s": device.bandwidth_gb_s,
339
+ "memory_gb": device.max_capacity_gb,
340
+ "memory_options_gb": ",".join(str(c) for c in device.capacity_options_gb),
341
+ "unified_memory": device.unified,
342
+ })
343
+
344
+ considered = fitted = skipped_closed = skipped_no_params = 0
345
+ edges = edges_with_max_context = 0
346
+ decode_predictions_omitted = 0
347
+
348
+ used_active = 0
349
+ for card in cards:
350
+ params = card.architecture.total_parameters
351
+ # Capacity and speed read different numbers. Every weight must be
352
+ # resident, so `fits` uses the total; only the active experts are read
353
+ # per token, so the decode prediction uses the active count.
354
+ active = card.architecture.active_parameters or params
355
+ if not card.licensing.open_weights:
356
+ skipped_closed += 1
357
+ continue
358
+ if not params:
359
+ skipped_no_params += 1
360
+ continue
361
+ considered += 1
362
+ is_moe = bool(card.architecture.num_experts)
363
+ has_active = bool(card.architecture.active_parameters)
364
+ if has_active:
365
+ used_active += 1
366
+ fitted_any = False
367
+ # Whether the weights fit is meaningful for every open-weights model.
368
+ # Decode tok/s is only meaningful for a model that decodes tokens.
369
+ decode_eligible = is_token_generating(card.identity.model_type)
370
+
371
+ for device in devices:
372
+ if not device.single_device_fit:
373
+ continue
374
+ capacity = device.max_capacity_gb
375
+ quants = fitting_quants(params, capacity)
376
+ if not quants:
377
+ continue
378
+ fitted_any = True
379
+ edges += 1
380
+ # Report the range, not a single "best". Highest quality that fits and
381
+ # smallest that fits answer different questions, and quoting only the
382
+ # first makes a large slow machine look worse than a small fast one
383
+ # purely because it chose a heavier quantisation.
384
+ best, smallest = quants[0], quants[-1]
385
+ ctx = predicted_max_context(card, capacity, best)
386
+ if ctx.tokens is not None:
387
+ edges_with_max_context += 1
388
+ if not decode_eligible:
389
+ decode_predictions_omitted += 1
390
+ # Store unrounded GB. Rounding to 2 decimals collapsed 616k-param
391
+ # bf16 weights to 0.0, which the page then printed as "0.0 GB".
392
+ sink.edge("Model", card.identity.model_id, "FITS_ON", "Hardware", device.id, {
393
+ "quantization": best,
394
+ "weights_gb": weights_gb(params, best),
395
+ "device_memory_gb": capacity,
396
+ "predicted_decode_tps": predicted_decode_tps(
397
+ device.bandwidth_gb_s, active, best) if decode_eligible else None,
398
+ "decode_reads_params": active if decode_eligible else None,
399
+ "fastest_quantization": smallest,
400
+ "fastest_weights_gb": weights_gb(params, smallest),
401
+ "fastest_predicted_decode_tps": predicted_decode_tps(
402
+ device.bandwidth_gb_s, active, smallest) if decode_eligible else None,
403
+ "quantizations_that_fit": ",".join(quants),
404
+ "max_context_at_quant": ctx.tokens,
405
+ "max_context_bound_by": ctx.bound_by,
406
+ "max_context_missing_geometry": ctx.missing_geometry,
407
+ "kv_heads_from_attention_heads": ctx.kv_heads_from_attention,
408
+ # Everything above is arithmetic, not observation. The page and
409
+ # the export must never render it as a measurement.
410
+ "basis": "computed",
411
+ "assumes_working_allowance": WORKING_ALLOWANCE,
412
+ "assumes_bandwidth_efficiency": BANDWIDTH_EFFICIENCY,
413
+ # No card carries active_parameters, so an MoE prediction uses
414
+ # total parameters and understates its real speed.
415
+ # Only conservative where the active count is still unknown.
416
+ "moe_prediction_is_conservative": decode_eligible and is_moe and not has_active,
417
+ })
418
+ fitted += 1 if fitted_any else 0
419
+
420
+ return {
421
+ "devices": len(devices),
422
+ "models_considered": considered,
423
+ "models_fitting_somewhere": fitted,
424
+ "skipped_closed_weights": skipped_closed,
425
+ "skipped_no_parameter_count": skipped_no_params,
426
+ "skipped_single_device_fit": sum(
427
+ 1 for device in devices if not device.single_device_fit),
428
+ "edges": edges,
429
+ "edges_with_max_context": edges_with_max_context,
430
+ "decode_predictions_omitted": decode_predictions_omitted,
431
+ "working_allowance": WORKING_ALLOWANCE,
432
+ "bandwidth_efficiency": BANDWIDTH_EFFICIENCY,
433
+ "used_active_parameters": used_active,
434
+ }
pipeline/hosts.py ADDED
@@ -0,0 +1,247 @@
1
+ """The host layer (MODEL-26 phase B): offload-aware fit for a (model, device, host).
2
+
3
+ A host is the machine around an accelerator (`hosts/*.yaml`, see
4
+ docs/host-layer.md). For a discrete host, weights that do not fit in
5
+ accelerator memory can spill to system RAM. Decode is then bounded by the
6
+ slower pool, so the answer is "fits, slowly", not "does not fit".
7
+
8
+ Three fit states:
9
+
10
+ * `accelerator`: `W + A <= C_acc`, today's `fits: true`.
11
+ * `offload`: not `accelerator`, `W + A <= C_acc + C_host`, and the host is not
12
+ unified. It carries `offload_fraction = (W + A - C_acc) / W`.
13
+ * `does_not_fit`: neither. A unified host has `C_host = 0` by definition, so
14
+ its fit is two-state.
15
+
16
+ `W` is weights at the quant, `A = WORKING_ALLOWANCE * C_acc`, and
17
+ `C_host = host RAM - OS_RESERVE_GB`. Host RAM is the profile's
18
+ `capacity_max_gb` unless the caller states what this machine has
19
+ (`--host-ram`).
20
+
21
+ Everything here is computed, never measured. The offload formula assumes
22
+ layer-split offload with the CPU not compute-bound; see "Known wrong" in
23
+ docs/host-layer.md.
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ import json
29
+ from dataclasses import dataclass
30
+ from pathlib import Path
31
+ from typing import Any
32
+
33
+ import yaml
34
+
35
+ from pipeline.hardware import (
36
+ BANDWIDTH_EFFICIENCY,
37
+ QUANT_PREFERENCE,
38
+ WORKING_ALLOWANCE,
39
+ is_token_generating,
40
+ predicted_decode_tps,
41
+ weights_gb,
42
+ )
43
+
44
+ #: RAM held back for the OS and everything that is not model weights. One
45
+ #: constant, not a per-OS table (Jamie, 2026-09-15). `--host-ram` overrides the
46
+ #: RAM figure the reserve is taken from, not the reserve itself.
47
+ OS_RESERVE_GB = 8.0
48
+
49
+ FIT_ACCELERATOR = "accelerator"
50
+ FIT_OFFLOAD = "offload"
51
+ FIT_NONE = "does_not_fit"
52
+
53
+ BASIS_ACCELERATOR = "accelerator-roofline"
54
+ BASIS_OFFLOAD = "offload-roofline"
55
+
56
+ _SECTIONS = {
57
+ "cpu": {"model", "cores", "threads"},
58
+ "system_memory": {"type", "channels", "max_speed_mt_s", "capacity_max_gb",
59
+ "capacity_options_gb", "bandwidth_gb_s", "bandwidth_derivation"},
60
+ "pcie": {"cpu_gen", "cpu_lanes_usable", "accelerator_link_gen", "accelerator_link_width"},
61
+ "storage": {"class", "capacity_options_gb"},
62
+ }
63
+ _TOP = {"id", "display_name", "kind", "unified", "hardware_ref", *_SECTIONS,
64
+ "field_sources", "figures_are", "notes"}
65
+
66
+
67
+ @dataclass(frozen=True)
68
+ class Host:
69
+ id: str
70
+ display_name: str
71
+ kind: str
72
+ unified: bool
73
+ hardware_ref: str | None
74
+ capacity_max_gb: float | None
75
+ bandwidth_gb_s: float | None
76
+ raw: dict[str, Any]
77
+
78
+
79
+ def _numeric(v: Any) -> bool:
80
+ if isinstance(v, bool):
81
+ return False
82
+ if isinstance(v, (int, float)):
83
+ return True
84
+ return isinstance(v, list) and bool(v) and all(_numeric(x) for x in v)
85
+
86
+
87
+ def validate(raw: Any, name: str) -> None:
88
+ """Raise ValueError when a profile breaks hosts/_schema.yaml."""
89
+ if not isinstance(raw, dict):
90
+ raise ValueError(f"{name}: a host profile must be a mapping")
91
+ if set(raw) != _TOP:
92
+ raise ValueError(f"{name}: top-level keys differ from the schema: {sorted(set(raw) ^ _TOP)}")
93
+ if raw["kind"] not in {"platform", "system"}:
94
+ raise ValueError(f"{name}: kind must be platform or system")
95
+ if not isinstance(raw["unified"], bool):
96
+ raise ValueError(f"{name}: unified is required and must be a boolean")
97
+ for section, keys in _SECTIONS.items():
98
+ if not isinstance(raw[section], dict) or set(raw[section]) != keys:
99
+ raise ValueError(f"{name}: {section} keys differ from the schema")
100
+ if raw["unified"] and (raw["system_memory"]["type"] != "unified"
101
+ or raw["pcie"]["accelerator_link_gen"] is not None):
102
+ raise ValueError(f"{name}: a unified host has memory type unified and no accelerator link")
103
+ sources = raw["field_sources"] or {}
104
+ numeric = {f"{s}.{k}" for s in _SECTIONS for k, v in raw[s].items() if _numeric(v)}
105
+ missing = numeric - set(sources)
106
+ if missing:
107
+ raise ValueError(f"{name}: unsourced numeric fields {sorted(missing)}")
108
+
109
+
110
+ def load_hosts(root: Path) -> list[Host]:
111
+ """Read and validate hosts/*.yaml. An invalid profile is rejected, not defaulted."""
112
+ out: list[Host] = []
113
+ for path in sorted((root / "hosts").glob("*.yaml")):
114
+ if path.name.startswith("_"):
115
+ continue
116
+ raw = yaml.safe_load(path.read_text(encoding="utf-8"))
117
+ validate(raw, path.name)
118
+ if raw["id"] != path.stem:
119
+ raise ValueError(f"{path.name}: id {raw['id']!r} does not match the file name")
120
+ out.append(host_from_raw(raw))
121
+ return out
122
+
123
+
124
+ def host_from_raw(raw: dict[str, Any]) -> Host:
125
+ mem = raw["system_memory"]
126
+ return Host(
127
+ id=raw["id"], display_name=raw["display_name"], kind=raw["kind"],
128
+ unified=bool(raw["unified"]), hardware_ref=raw.get("hardware_ref"),
129
+ capacity_max_gb=float(mem["capacity_max_gb"]) if mem.get("capacity_max_gb") else None,
130
+ bandwidth_gb_s=float(mem["bandwidth_gb_s"]) if mem.get("bandwidth_gb_s") else None,
131
+ raw=raw,
132
+ )
133
+
134
+
135
+ def host_capacity_gb(host: Host, host_ram_gb: float | None = None) -> float:
136
+ """`C_host`: host memory available for spilled weights. 0 for a unified host."""
137
+ if host.unified:
138
+ return 0.0
139
+ ram = host_ram_gb if host_ram_gb is not None else host.capacity_max_gb
140
+ if ram is None:
141
+ return 0.0
142
+ return max(0.0, float(ram) - OS_RESERVE_GB)
143
+
144
+
145
+ def fit_state(weights: float, accelerator_gb: float, host_gb: float, unified: bool) -> str:
146
+ need = weights + WORKING_ALLOWANCE * accelerator_gb
147
+ if need <= accelerator_gb:
148
+ return FIT_ACCELERATOR
149
+ if not unified and need <= accelerator_gb + host_gb:
150
+ return FIT_OFFLOAD
151
+ return FIT_NONE
152
+
153
+
154
+ def offload_fraction(weights: float, accelerator_gb: float) -> float:
155
+ """`f = (W + A - C_acc) / W`, clamped to [0, 1]. 0 when everything is resident."""
156
+ if weights <= 0:
157
+ return 0.0
158
+ f = (weights + WORKING_ALLOWANCE * accelerator_gb - accelerator_gb) / weights
159
+ return min(1.0, max(0.0, f))
160
+
161
+
162
+ def offload_decode_tps(accelerator_bandwidth: float, host_bandwidth: float,
163
+ params: float, quant: str, fraction: float) -> float:
164
+ """`BANDWIDTH_EFFICIENCY / ((1-f)*P/B_acc + f*P/B_host)`, rounded like the roofline.
165
+
166
+ At f = 0 this is exactly `predicted_decode_tps`: the same function answers,
167
+ so no floating-point path can make the two disagree.
168
+ """
169
+ if fraction <= 0:
170
+ return predicted_decode_tps(accelerator_bandwidth, params, quant)
171
+ per_token = weights_gb(params, quant)
172
+ if per_token <= 0:
173
+ return 0.0
174
+ seconds = (1 - fraction) * per_token / accelerator_bandwidth + fraction * per_token / host_bandwidth
175
+ return round(BANDWIDTH_EFFICIENCY / seconds, 1)
176
+
177
+
178
+ def assess(*, total_params: float, active_params: float | None, model_type: Any,
179
+ accelerator_gb: float, accelerator_bandwidth: float,
180
+ host: Host, host_ram_gb: float | None = None) -> dict[str, Any]:
181
+ """Fit state and decode prediction for one (model, device, host) triple.
182
+
183
+ The quant chosen is the smallest that reaches the best state: for
184
+ `accelerator` it matches `fastest_quantization`, the figure `fits` carries;
185
+ for `offload` it spills the least, which is the fastest offload.
186
+ Decode tps is null for a model that does not decode tokens, and for an
187
+ offload row whose host has no published bandwidth.
188
+ """
189
+ host_gb = host_capacity_gb(host, host_ram_gb)
190
+ reads = active_params or total_params
191
+ best: tuple[str, str] | None = None
192
+ for state in (FIT_ACCELERATOR, FIT_OFFLOAD):
193
+ ok = [q for q in QUANT_PREFERENCE
194
+ if fit_state(weights_gb(total_params, q), accelerator_gb, host_gb, host.unified) == state]
195
+ if ok:
196
+ best = (state, ok[-1])
197
+ break
198
+ if best is None:
199
+ return {"fit_state": FIT_NONE, "quantization": None, "offload_fraction": None,
200
+ "predicted_decode_tps": None, "predicted_decode_tps_basis": None}
201
+ state, quant = best
202
+ tokens = is_token_generating(model_type)
203
+ if state == FIT_ACCELERATOR:
204
+ f = 0.0
205
+ tps = predicted_decode_tps(accelerator_bandwidth, reads, quant) if tokens else None
206
+ basis = BASIS_ACCELERATOR
207
+ else:
208
+ f = round(offload_fraction(weights_gb(total_params, quant), accelerator_gb), 4)
209
+ tps = (offload_decode_tps(accelerator_bandwidth, host.bandwidth_gb_s, reads, quant, f)
210
+ if tokens and host.bandwidth_gb_s else None)
211
+ basis = BASIS_OFFLOAD
212
+ return {"fit_state": state, "quantization": quant, "offload_fraction": f,
213
+ "predicted_decode_tps": tps,
214
+ "predicted_decode_tps_basis": basis if tps is not None else None}
215
+
216
+
217
+ def write_export(api_dir: Path, hosts: list[Host], build_json: dict[str, Any]) -> dict[str, Any]:
218
+ """Emit `api/hosts.json`: every profile as authored, plus the reserve used."""
219
+ api_dir.mkdir(parents=True, exist_ok=True)
220
+ payload = {
221
+ "build": build_json,
222
+ "count": len(hosts),
223
+ "os_reserve_gb": OS_RESERVE_GB,
224
+ "hosts": [h.raw for h in hosts],
225
+ }
226
+ (api_dir / "hosts.json").write_text(
227
+ json.dumps(payload, indent=1, sort_keys=True, default=str), encoding="utf-8")
228
+ return {"hosts": len(hosts)}
229
+
230
+
231
+ def add_to_graph(sink: Any, hosts: list[Host]) -> int:
232
+ """`:Host` nodes and `(:Host)-[:HOSTS]->(:Hardware)` for unified hosts only.
233
+
234
+ A discrete pairing is a query-time input, not a stored record, so discrete
235
+ hosts get no node. Not wired into the published graph yet (see
236
+ docs/graph-ontology.md). Returns the number of HOSTS edges.
237
+ """
238
+ edges = 0
239
+ for host in hosts:
240
+ if not host.unified:
241
+ continue
242
+ sink.node("Host", "id", host.id, {"id": host.id, "display_name": host.display_name,
243
+ "unified": True, "kind": host.kind})
244
+ if host.hardware_ref:
245
+ sink.edge("Host", host.id, "HOSTS", "Hardware", host.hardware_ref, {})
246
+ edges += 1
247
+ return edges