prefer-inference-core 0.0.0-g0be6ef9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +166 -0
  3. package/dist/catalog.d.ts +27 -0
  4. package/dist/catalog.d.ts.map +1 -0
  5. package/dist/catalog.js +547 -0
  6. package/dist/catalog.js.map +1 -0
  7. package/dist/cli.d.ts +3 -0
  8. package/dist/cli.d.ts.map +1 -0
  9. package/dist/cli.js +412 -0
  10. package/dist/cli.js.map +1 -0
  11. package/dist/huggingface.d.ts +18 -0
  12. package/dist/huggingface.d.ts.map +1 -0
  13. package/dist/huggingface.js +296 -0
  14. package/dist/huggingface.js.map +1 -0
  15. package/dist/index.d.ts +8 -0
  16. package/dist/index.d.ts.map +1 -0
  17. package/dist/index.js +8 -0
  18. package/dist/index.js.map +1 -0
  19. package/dist/planning.d.ts +11 -0
  20. package/dist/planning.d.ts.map +1 -0
  21. package/dist/planning.js +610 -0
  22. package/dist/planning.js.map +1 -0
  23. package/dist/prefer.mjs +9493 -0
  24. package/dist/prefer.mjs.map +7 -0
  25. package/dist/releases.d.ts +33 -0
  26. package/dist/releases.d.ts.map +1 -0
  27. package/dist/releases.js +288 -0
  28. package/dist/releases.js.map +1 -0
  29. package/dist/resources.d.ts +10 -0
  30. package/dist/resources.d.ts.map +1 -0
  31. package/dist/resources.js +250 -0
  32. package/dist/resources.js.map +1 -0
  33. package/dist/types.d.ts +409 -0
  34. package/dist/types.d.ts.map +1 -0
  35. package/dist/types.js +2 -0
  36. package/dist/types.js.map +1 -0
  37. package/dist/utils.d.ts +20 -0
  38. package/dist/utils.d.ts.map +1 -0
  39. package/dist/utils.js +112 -0
  40. package/dist/utils.js.map +1 -0
  41. package/package.json +58 -0
  42. package/schemas/prefer-model-catalog-extension.schema.json +25 -0
  43. package/schemas/prefer-model-catalog.schema.json +54 -0
  44. package/schemas/prefer-model-plan.schema.json +115 -0
  45. package/schemas/prefer-resource-profile.schema.json +70 -0
@@ -0,0 +1,610 @@
1
+ import { listModelVariants, resolveCatalogModelId, resolveModelVariant } from "./catalog.js";
2
+ import { isObject } from "./utils.js";
3
+ const GIB = 1024 ** 3;
4
+ const DEFAULT_HEADROOM_FRACTION = 0.04;
5
+ const DEFAULT_MINIMUM_HEADROOM_BYTES = 1.5 * GIB;
6
+ const DEFAULT_MAXIMUM_HEADROOM_BYTES = 4 * GIB;
7
+ const QUALITY_SCORES = {
8
+ reference: 100,
9
+ "near-lossless": 90,
10
+ high: 80,
11
+ deployment: 60,
12
+ compromise: 45,
13
+ "fit-floor": 25,
14
+ extreme: 10,
15
+ unknown: 0
16
+ };
17
+ export function quantQuality(quant) {
18
+ const normalized = quant.toLowerCase().replaceAll("_", "-");
19
+ const scores = [];
20
+ if (/(?:^|-)(?:f32|fp32|f16|fp16|bf16)(?:-|$)/u.test(normalized))
21
+ scores.push(100);
22
+ if (/(?:^|-)(?:fp8|q8|int8)(?:-|$)/u.test(normalized))
23
+ scores.push(90);
24
+ if (/(?:^|-)(?:q6|iq6)(?:-|$)/u.test(normalized))
25
+ scores.push(80);
26
+ if (/(?:^|-)(?:q5|iq5)(?:-|$)/u.test(normalized))
27
+ scores.push(70);
28
+ if (/(?:^|-)(?:nvfp4|mxfp4|w4a16|q4|iq4|int4)(?:-|$)/u.test(normalized))
29
+ scores.push(60);
30
+ if (/(?:^|-)(?:q3|iq3)(?:-|$)/u.test(normalized))
31
+ scores.push(45);
32
+ if (/(?:^|-)(?:q2|iq2)(?:-|$)/u.test(normalized))
33
+ scores.push(25);
34
+ if (/(?:^|-)(?:q1|iq1)(?:-|$)/u.test(normalized))
35
+ scores.push(10);
36
+ if (!scores.length)
37
+ return { tier: "unknown", score: 0 };
38
+ let score = Math.min(...scores);
39
+ if (normalized.includes("xl"))
40
+ score += 3;
41
+ else if (normalized.includes("k-m"))
42
+ score += 2;
43
+ else if (normalized.includes("k-s"))
44
+ score += 1;
45
+ if (normalized.includes("iq") && score >= 45)
46
+ score -= 1;
47
+ return { tier: tierForScore(score), score };
48
+ }
49
+ export function estimateVariantFit(variant, resources, options = {}) {
50
+ const capacity = acceleratorCapacity(resources, options.allow_multi_gpu ?? true);
51
+ const weightBytes = options.device_weight_bytes ?? variant.artifact_bytes;
52
+ const runtimeOverhead = options.runtime_overhead_bytes ?? Math.max(512 * 1024 ** 2, Math.ceil(weightBytes * 0.02));
53
+ const workloadBytes = workloadMemory(options);
54
+ const required = weightBytes + runtimeOverhead + workloadBytes;
55
+ const reasons = [];
56
+ const confidence = options.workload?.source === "measured"
57
+ ? "measured"
58
+ : options.workload?.source === "architecture"
59
+ ? "architecture"
60
+ : capacity === undefined
61
+ ? "unknown"
62
+ : "artifact-only";
63
+ if (capacity === undefined) {
64
+ return {
65
+ status: "unknown",
66
+ confidence,
67
+ weight_bytes: weightBytes,
68
+ runtime_overhead_bytes: runtimeOverhead,
69
+ workload_bytes: workloadBytes,
70
+ required_device_bytes: required,
71
+ reasons: ["No usable accelerator or unified-memory capacity was supplied."]
72
+ };
73
+ }
74
+ const headroom = reservedHeadroom(capacity, options);
75
+ const usable = Math.max(0, capacity - headroom);
76
+ const remaining = capacity - required;
77
+ if (required <= usable) {
78
+ reasons.push(`Estimated device use leaves ${formatGiB(remaining)} GiB before the reserved headroom boundary.`);
79
+ return fitResult("fits", confidence, capacity, usable, weightBytes, runtimeOverhead, workloadBytes, required, remaining, reasons);
80
+ }
81
+ const deficit = Math.max(0, required - usable);
82
+ const hostAvailable = hostCapacity(resources);
83
+ const maximumOffload = options.maximum_host_offload_bytes ?? hostAvailable;
84
+ if (options.allow_host_offload && resources.memory_topology === "discrete" && deficit > 0) {
85
+ if (hostAvailable !== undefined &&
86
+ maximumOffload !== undefined &&
87
+ deficit <= Math.min(hostAvailable, maximumOffload)) {
88
+ reasons.push(`The route requires approximately ${formatGiB(deficit)} GiB of host offload.`);
89
+ return {
90
+ ...fitResult("host-offload", confidence, capacity, usable, weightBytes, runtimeOverhead, workloadBytes, required, remaining, reasons),
91
+ host_offload_bytes: deficit
92
+ };
93
+ }
94
+ if (hostAvailable === undefined
95
+ && options.allow_unknown_host_offload
96
+ && (maximumOffload === undefined || deficit <= maximumOffload)) {
97
+ reasons.push(`The configured offload route needs approximately ${formatGiB(deficit)} GiB of host memory; runtime host capacity has not been observed.`);
98
+ return {
99
+ ...fitResult("unknown", "unknown", capacity, usable, weightBytes, runtimeOverhead, workloadBytes, required, remaining, reasons),
100
+ host_offload_bytes: deficit
101
+ };
102
+ }
103
+ }
104
+ if (required <= capacity) {
105
+ reasons.push(`The variant fits nominal capacity but leaves less than the ${formatGiB(headroom)} GiB reserve.`);
106
+ return fitResult("tight", confidence, capacity, usable, weightBytes, runtimeOverhead, workloadBytes, required, remaining, reasons);
107
+ }
108
+ reasons.push(`Estimated device use exceeds the usable device pool by ${formatGiB(deficit)} GiB.`);
109
+ return fitResult("does-not-fit", confidence, capacity, usable, weightBytes, runtimeOverhead, workloadBytes, required, remaining, reasons);
110
+ }
111
+ export function planModelSet(catalog, options) {
112
+ const selected = [];
113
+ const skipped = [];
114
+ const floor = typeof options.quality_floor === "number"
115
+ ? options.quality_floor
116
+ : QUALITY_SCORES[options.quality_floor ?? "deployment"];
117
+ const maximumModels = options.max_models
118
+ ?? (options.engine === "sglang" || options.engine === "vllm" ? 1 : Number.POSITIVE_INFINITY);
119
+ const storageCapacity = options.max_staged_bytes ?? options.resources.storage?.available_bytes ?? options.resources.storage?.total_bytes ?? Number.POSITIVE_INFINITY;
120
+ let stagedBytes = 0;
121
+ const hints = normalizeHints(options.hints);
122
+ if (options.workload)
123
+ validateWorkload(options.workload);
124
+ const ordered = options.models.map((request, index) => ({ request, index })).sort((left, right) => requestUtility(catalog, right.request, hints) - requestUtility(catalog, left.request, hints) || left.index - right.index);
125
+ for (const { request } of ordered) {
126
+ if (request.workload)
127
+ validateWorkload(request.workload);
128
+ if (selected.length >= maximumModels) {
129
+ skipped.push(skip(request, "model-limit", [`The deployment is limited to ${maximumModels} selected models.`]));
130
+ continue;
131
+ }
132
+ let candidates;
133
+ try {
134
+ candidates = variantCandidates(catalog, request, options, floor);
135
+ }
136
+ catch (error) {
137
+ skipped.push(skip(request, "no-compatible-variant", [message(error)]));
138
+ continue;
139
+ }
140
+ if (!candidates.length) {
141
+ skipped.push(skip(request, "no-compatible-variant", [`No ${options.engine} variant satisfies the quantization floor.`]));
142
+ continue;
143
+ }
144
+ const capabilityCandidates = candidates.filter((variant) => !missingModelCapabilities(variant, hints).length);
145
+ if (!capabilityCandidates.length) {
146
+ skipped.push(skip(request, "missing-capability", [
147
+ `The model does not provide required capabilities: ${hints.required_capabilities?.join(", ") ?? "unknown"}.`
148
+ ]));
149
+ continue;
150
+ }
151
+ const attempted = [];
152
+ const workloadRequest = request.workload ?? options.workload;
153
+ const baseFit = mergeFit(options.fit, request.fit);
154
+ const desiredFit = workloadRequest ? fitForWorkload(baseFit, workloadRequest) : baseFit;
155
+ let choice;
156
+ for (const variant of capabilityCandidates) {
157
+ const missing = missingCapabilities(variant, options.resources);
158
+ if (missing.length) {
159
+ attempted.push(`${variant.quant}: missing ${missing.join(", ")}`);
160
+ continue;
161
+ }
162
+ const fit = estimateVariantFit(variant, options.resources, desiredFit);
163
+ attempted.push(`${variant.quant}: ${fit.status}`);
164
+ if (acceptableFit(fit, desiredFit, options.accept_tight)) {
165
+ choice = { variant, fit };
166
+ break;
167
+ }
168
+ }
169
+ if (!choice && workloadRequest) {
170
+ for (const variant of capabilityCandidates) {
171
+ if (missingCapabilities(variant, options.resources).length)
172
+ continue;
173
+ const workload = tuneWorkloadToFit(variant, options.resources, workloadRequest, omitWorkload(baseFit));
174
+ if (!workload.fits) {
175
+ attempted.push(`${variant.quant}: minimum workload does not fit`);
176
+ continue;
177
+ }
178
+ const fit = estimateVariantFit(variant, options.resources, fitForWorkload(baseFit, {
179
+ ...workloadRequest,
180
+ desired_context_tokens: workload.context_tokens,
181
+ desired_concurrency: workload.concurrency
182
+ }));
183
+ if (acceptableFit(fit, baseFit, options.accept_tight)) {
184
+ choice = { variant, fit, workload };
185
+ attempted.push(`${variant.quant}: ${workload.reason}`);
186
+ break;
187
+ }
188
+ }
189
+ }
190
+ if (!choice) {
191
+ skipped.push(skip(request, "insufficient-memory", attempted));
192
+ continue;
193
+ }
194
+ if (stagedBytes + choice.variant.artifact_bytes > storageCapacity) {
195
+ skipped.push(skip(request, "storage-budget", [
196
+ `${choice.variant.quant} needs ${choice.variant.artifact_bytes} artifact bytes; ${Math.max(0, storageCapacity - stagedBytes)} remain.`
197
+ ]));
198
+ continue;
199
+ }
200
+ const preferredQuant = preferredVariant(catalog, request, options).quant;
201
+ const quantChanged = choice.variant.quant !== preferredQuant;
202
+ selected.push({
203
+ model_id: choice.variant.model_id,
204
+ preferred_quant: preferredQuant,
205
+ ...(request.quant ? { requested_quant: request.quant } : {}),
206
+ selected: choice.variant,
207
+ fit: choice.fit,
208
+ quant_changed: quantChanged,
209
+ ...(workloadRequest ? { workload_request: structuredClone(workloadRequest) } : {}),
210
+ ...(choice.workload ? { workload: choice.workload } : {}),
211
+ reason: quantChanged
212
+ ? `${preferredQuant} did not meet the resource policy; selected ${choice.variant.quant}.`
213
+ : `${choice.variant.quant} meets the resource policy.`
214
+ });
215
+ stagedBytes += choice.variant.artifact_bytes;
216
+ }
217
+ const requiredSkipped = skipped.some((entry) => entry.required);
218
+ return {
219
+ schema_version: "prefer.model-plan.v1",
220
+ engine: options.engine,
221
+ resources: structuredClone(options.resources),
222
+ selected,
223
+ skipped,
224
+ capabilities: [...new Set(selected.flatMap((entry) => entry.selected.capabilities))].sort(),
225
+ staged_artifact_bytes: stagedBytes,
226
+ complete: !requiredSkipped,
227
+ ...(Object.keys(hints).length ? { hints } : {})
228
+ };
229
+ }
230
+ /** Convert a generated engine deployment's existing model or bundle members
231
+ * into planner requests. Bundle members are optional by default; single-model
232
+ * deployments remain required. A deployment quant is a starting point, not an
233
+ * exact lock, so constrained hardware may choose a smaller published lane. */
234
+ export function modelRequestsFromDeployment(value) {
235
+ if (!isObject(value) || !Array.isArray(value.models))
236
+ return [];
237
+ const bundle = value.kind === "bundle"
238
+ || String(value.id ?? "").endsWith("/general")
239
+ || (value.models.length > 1 && value.kind !== "single-model");
240
+ const result = new Map();
241
+ const total = value.models.length;
242
+ for (const [index, raw] of value.models.entries()) {
243
+ if (!isObject(raw))
244
+ continue;
245
+ const modelId = firstString(raw.request_model_id, raw.model_slug, raw.profile_id);
246
+ if (!modelId || result.has(modelId))
247
+ continue;
248
+ const preferredQuant = firstString(raw.quant_slug, raw.quant);
249
+ const configuredOffload = hasConfiguredHostOffload(raw) || hasConfiguredHostOffload(value);
250
+ result.set(modelId, {
251
+ model_id: modelId,
252
+ ...(preferredQuant ? { preferred_quant: preferredQuant } : {}),
253
+ required: !bundle,
254
+ priority: total - index,
255
+ ...(configuredOffload ? {
256
+ fit: {
257
+ allow_host_offload: true,
258
+ allow_unknown_host_offload: true
259
+ }
260
+ } : {})
261
+ });
262
+ }
263
+ return [...result.values()];
264
+ }
265
+ export function tuneWorkloadToFit(variant, resources, request, options = {}) {
266
+ if (!Number.isFinite(request.bytes_per_token) || request.bytes_per_token <= 0) {
267
+ throw new Error("bytes_per_token must be greater than zero");
268
+ }
269
+ const desiredContext = positive(request.desired_context_tokens, "desired_context_tokens");
270
+ const desiredConcurrency = positive(request.desired_concurrency, "desired_concurrency");
271
+ const minimumContext = positive(request.minimum_context_tokens ?? 1, "minimum_context_tokens");
272
+ const minimumConcurrency = positive(request.minimum_concurrency ?? 1, "minimum_concurrency");
273
+ const capacity = acceleratorCapacity(resources, options.allow_multi_gpu ?? true);
274
+ if (capacity === undefined)
275
+ return {
276
+ fits: false,
277
+ context_tokens: 0,
278
+ concurrency: 0,
279
+ aggregate_token_capacity: 0,
280
+ changed: true,
281
+ reason: "No usable device capacity is known."
282
+ };
283
+ const weightBytes = options.device_weight_bytes ?? variant.artifact_bytes;
284
+ const runtimeOverhead = options.runtime_overhead_bytes ?? Math.max(512 * 1024 ** 2, Math.ceil(weightBytes * 0.02));
285
+ const headroom = reservedHeadroom(capacity, options);
286
+ const tokenBudgetBytes = capacity - headroom - weightBytes - runtimeOverhead - (request.fixed_device_bytes ?? 0);
287
+ const aggregateCapacity = Math.max(0, Math.floor(tokenBudgetBytes / request.bytes_per_token));
288
+ const desiredTotal = desiredContext * desiredConcurrency;
289
+ if (desiredTotal <= aggregateCapacity)
290
+ return {
291
+ fits: true,
292
+ context_tokens: desiredContext,
293
+ concurrency: desiredConcurrency,
294
+ aggregate_token_capacity: aggregateCapacity,
295
+ changed: false,
296
+ reason: "The requested context and concurrency fit the estimated token pool."
297
+ };
298
+ let context = desiredContext;
299
+ let concurrency = desiredConcurrency;
300
+ if ((request.priority ?? "context") === "context") {
301
+ concurrency = Math.min(desiredConcurrency, Math.floor(aggregateCapacity / desiredContext));
302
+ concurrency = Math.max(minimumConcurrency, concurrency);
303
+ context = Math.min(desiredContext, Math.floor(aggregateCapacity / concurrency));
304
+ }
305
+ else {
306
+ context = Math.min(desiredContext, Math.floor(aggregateCapacity / desiredConcurrency));
307
+ context = Math.max(minimumContext, context);
308
+ concurrency = Math.min(desiredConcurrency, Math.floor(aggregateCapacity / context));
309
+ }
310
+ const fits = context >= minimumContext && concurrency >= minimumConcurrency && context * concurrency <= aggregateCapacity;
311
+ return {
312
+ fits,
313
+ context_tokens: fits ? context : 0,
314
+ concurrency: fits ? concurrency : 0,
315
+ aggregate_token_capacity: aggregateCapacity,
316
+ changed: true,
317
+ reason: fits
318
+ ? `Adjusted to ${concurrency} request slot(s) at ${context} tokens each.`
319
+ : "The minimum context and concurrency do not fit the estimated token pool."
320
+ };
321
+ }
322
+ function variantCandidates(catalog, request, options, floor) {
323
+ const preferred = preferredVariant(catalog, request, options);
324
+ if (request.quant)
325
+ return [preferred];
326
+ const bias = options.hints?.quant_bias ?? "balanced";
327
+ const preferredScore = quantQuality(preferred.quant).score;
328
+ const variants = listModelVariants(catalog, request.model_id, options.engine)
329
+ .filter((variant) => request.repository
330
+ ? variant.repository === request.repository
331
+ : bias === "balanced"
332
+ ? variant.repository === preferred.repository || variant.artifact_bytes < preferred.artifact_bytes
333
+ : true)
334
+ .filter((variant) => options.use_nvfp4 || variant.quant === preferred.quant || !variant.quant.toLowerCase().includes("nvfp4"))
335
+ .filter((variant) => {
336
+ const score = quantQuality(variant.quant).score;
337
+ return variant.quant === preferred.quant || (score >= floor && (bias !== "balanced" || preferredScore === 0 || score <= preferredScore));
338
+ })
339
+ .sort((left, right) => {
340
+ if (bias === "capacity") {
341
+ return left.artifact_bytes - right.artifact_bytes || quantQuality(right.quant).score - quantQuality(left.quant).score;
342
+ }
343
+ if (bias === "quality") {
344
+ return quantQuality(right.quant).score - quantQuality(left.quant).score || left.artifact_bytes - right.artifact_bytes;
345
+ }
346
+ if (left.repository === preferred.repository && right.repository !== preferred.repository)
347
+ return -1;
348
+ if (right.repository === preferred.repository && left.repository !== preferred.repository)
349
+ return 1;
350
+ return quantQuality(right.quant).score - quantQuality(left.quant).score || right.artifact_bytes - left.artifact_bytes;
351
+ });
352
+ const deduped = new Map();
353
+ if (bias === "balanced")
354
+ deduped.set(`${preferred.repository}@${preferred.quant}`, preferred);
355
+ for (const variant of variants)
356
+ deduped.set(`${variant.repository}@${variant.quant}`, variant);
357
+ if (!deduped.has(`${preferred.repository}@${preferred.quant}`)) {
358
+ deduped.set(`${preferred.repository}@${preferred.quant}`, preferred);
359
+ }
360
+ return [...deduped.values()];
361
+ }
362
+ function preferredVariant(catalog, request, options) {
363
+ return resolveModelVariant(catalog, request.model_id, {
364
+ engine: options.engine,
365
+ ...(request.repository ? { repository: request.repository } : {}),
366
+ ...(request.quant ?? request.preferred_quant ? { quant: (request.quant ?? request.preferred_quant) } : {}),
367
+ useNvfp4: options.use_nvfp4,
368
+ overrides: request.overrides
369
+ });
370
+ }
371
+ function missingCapabilities(variant, resources) {
372
+ const normalized = variant.quant.toLowerCase().replaceAll("_", "-");
373
+ const required = [];
374
+ if (normalized.includes("nvfp4"))
375
+ required.push("nvfp4");
376
+ if (normalized === "fp8" && (variant.engine === "sglang" || variant.engine === "vllm"))
377
+ required.push("fp8");
378
+ return required.filter((capability) => !resources.capabilities.includes(capability));
379
+ }
380
+ function missingModelCapabilities(variant, hints) {
381
+ return (hints.required_capabilities ?? []).filter((capability) => !variant.capabilities.includes(capability));
382
+ }
383
+ function requestUtility(catalog, request, hints) {
384
+ let modelId;
385
+ try {
386
+ modelId = resolveCatalogModelId(catalog, request.model_id);
387
+ }
388
+ catch {
389
+ return request.priority ?? 0;
390
+ }
391
+ const model = catalog.models[modelId];
392
+ const capabilities = new Set(model.capabilities ?? []);
393
+ const preferredRoles = new Set(model.profile?.roles?.preferred ?? []);
394
+ const capableRoles = new Set(model.profile?.roles?.capable ?? []);
395
+ const avoidRoles = new Set(model.profile?.roles?.avoid ?? []);
396
+ let score = request.priority ?? 0;
397
+ for (const capability of hints.preferred_capabilities ?? [])
398
+ if (capabilities.has(capability))
399
+ score += 100;
400
+ for (const role of hints.preferred_roles ?? []) {
401
+ if (preferredRoles.has(role))
402
+ score += 80;
403
+ else if (capableRoles.has(role))
404
+ score += 40;
405
+ if (avoidRoles.has(role))
406
+ score -= 80;
407
+ }
408
+ const speedImportance = hints.speed_importance ?? 0;
409
+ if (speedImportance > 0) {
410
+ if (request.speed_score !== undefined && (!Number.isFinite(request.speed_score) || request.speed_score < 0 || request.speed_score > 100)) {
411
+ throw new Error(`${request.model_id} speed_score must be between 0 and 100`);
412
+ }
413
+ const activeParameters = model.profile?.architecture?.active_parameters_b;
414
+ const inferredSpeed = typeof activeParameters === "number" && activeParameters > 0
415
+ ? 100 / (1 + Math.log2(1 + activeParameters))
416
+ : 0;
417
+ score += speedImportance * (request.speed_score ?? inferredSpeed);
418
+ }
419
+ const qualityImportance = hints.quality_importance ?? 0;
420
+ if (qualityImportance > 0 && request.quality_score !== undefined) {
421
+ if (!Number.isFinite(request.quality_score) || request.quality_score < 0 || request.quality_score > 100) {
422
+ throw new Error(`${request.model_id} quality_score must be between 0 and 100`);
423
+ }
424
+ score += qualityImportance * request.quality_score;
425
+ }
426
+ return score;
427
+ }
428
+ function normalizeHints(value) {
429
+ if (!value)
430
+ return {};
431
+ if (value.speed_importance !== undefined && (!Number.isFinite(value.speed_importance) || value.speed_importance < 0 || value.speed_importance > 1))
432
+ throw new Error("speed_importance must be between 0 and 1");
433
+ if (value.quality_importance !== undefined && (!Number.isFinite(value.quality_importance) || value.quality_importance < 0 || value.quality_importance > 1))
434
+ throw new Error("quality_importance must be between 0 and 1");
435
+ if (value.quant_bias && !["quality", "balanced", "capacity"].includes(value.quant_bias)) {
436
+ throw new Error("quant_bias must be quality, balanced, or capacity");
437
+ }
438
+ return {
439
+ ...(value.required_capabilities?.length ? { required_capabilities: uniqueStrings(value.required_capabilities) } : {}),
440
+ ...(value.preferred_capabilities?.length ? { preferred_capabilities: uniqueStrings(value.preferred_capabilities) } : {}),
441
+ ...(value.preferred_roles?.length ? { preferred_roles: uniqueStrings(value.preferred_roles) } : {}),
442
+ ...(value.speed_importance !== undefined ? { speed_importance: value.speed_importance } : {}),
443
+ ...(value.quality_importance !== undefined ? { quality_importance: value.quality_importance } : {}),
444
+ ...(value.quant_bias ? { quant_bias: value.quant_bias } : {})
445
+ };
446
+ }
447
+ function uniqueStrings(values) {
448
+ return [...new Set(values.map((value) => value.trim()).filter(Boolean))];
449
+ }
450
+ function validateWorkload(value) {
451
+ positive(value.desired_context_tokens, "desired_context_tokens");
452
+ positive(value.desired_concurrency, "desired_concurrency");
453
+ positive(value.minimum_context_tokens ?? 1, "minimum_context_tokens");
454
+ positive(value.minimum_concurrency ?? 1, "minimum_concurrency");
455
+ if (!Number.isFinite(value.bytes_per_token) || value.bytes_per_token <= 0) {
456
+ throw new Error("bytes_per_token must be greater than zero");
457
+ }
458
+ if (value.fixed_device_bytes !== undefined && (!Number.isFinite(value.fixed_device_bytes) || value.fixed_device_bytes < 0)) {
459
+ throw new Error("fixed_device_bytes must be non-negative");
460
+ }
461
+ if ((value.minimum_context_tokens ?? 1) > value.desired_context_tokens) {
462
+ throw new Error("minimum_context_tokens may not exceed desired_context_tokens");
463
+ }
464
+ if ((value.minimum_concurrency ?? 1) > value.desired_concurrency) {
465
+ throw new Error("minimum_concurrency may not exceed desired_concurrency");
466
+ }
467
+ }
468
+ function acceleratorCapacity(resources, allowMultiGpu) {
469
+ if (resources.memory_topology === "unified") {
470
+ const unified = resources.unified_memory?.available_bytes ?? resources.unified_memory?.total_bytes;
471
+ const host = resources.host_memory?.available_bytes ?? resources.host_memory?.total_bytes;
472
+ const accelerator = resources.accelerators
473
+ .map((entry) => entry.available_bytes ?? entry.total_bytes)
474
+ .filter((entry) => entry !== undefined)
475
+ .reduce((minimum, entry) => minimum === undefined ? entry : Math.min(minimum, entry), undefined);
476
+ const observations = [unified, host, accelerator].filter((entry) => entry !== undefined);
477
+ return observations.length ? Math.min(...observations) : undefined;
478
+ }
479
+ if (resources.memory_topology === "host" && !resources.accelerators.length) {
480
+ return resources.host_memory?.available_bytes ?? resources.host_memory?.total_bytes;
481
+ }
482
+ const deviceCapacities = resources.accelerators
483
+ .map((entry) => entry.available_bytes ?? entry.total_bytes)
484
+ .filter((entry) => entry !== undefined);
485
+ if (!deviceCapacities.length)
486
+ return undefined;
487
+ return allowMultiGpu ? deviceCapacities.reduce((total, entry) => total + entry, 0) : Math.max(...deviceCapacities);
488
+ }
489
+ function hostCapacity(resources) {
490
+ return resources.host_memory?.available_bytes ?? resources.host_memory?.total_bytes;
491
+ }
492
+ function workloadMemory(options) {
493
+ const workload = options.workload;
494
+ if (!workload)
495
+ return 0;
496
+ const fixed = workload.fixed_device_bytes ?? 0;
497
+ const perToken = workload.bytes_per_token ?? 0;
498
+ const context = workload.context_tokens ?? 0;
499
+ const concurrency = workload.concurrency ?? 1;
500
+ return fixed + perToken * context * concurrency;
501
+ }
502
+ function reservedHeadroom(capacity, options) {
503
+ const fraction = options.headroom_fraction ?? DEFAULT_HEADROOM_FRACTION;
504
+ const minimum = options.minimum_headroom_bytes ?? DEFAULT_MINIMUM_HEADROOM_BYTES;
505
+ const maximum = options.maximum_headroom_bytes ?? DEFAULT_MAXIMUM_HEADROOM_BYTES;
506
+ if (!Number.isFinite(fraction) || fraction < 0 || fraction > 1) {
507
+ throw new Error("headroom_fraction must be between 0 and 1");
508
+ }
509
+ for (const [label, value] of [["minimum_headroom_bytes", minimum], ["maximum_headroom_bytes", maximum]]) {
510
+ if (!Number.isFinite(value) || value < 0)
511
+ throw new Error(`${label} must be non-negative`);
512
+ }
513
+ if (minimum > maximum)
514
+ throw new Error("minimum_headroom_bytes may not exceed maximum_headroom_bytes");
515
+ return Math.min(capacity, Math.max(minimum, Math.min(maximum, Math.floor(capacity * fraction))));
516
+ }
517
+ function acceptableFit(fit, options, acceptTight = false) {
518
+ return fit.status === "fits"
519
+ || fit.status === "host-offload"
520
+ || (acceptTight && fit.status === "tight")
521
+ || (fit.status === "unknown"
522
+ && fit.host_offload_bytes !== undefined
523
+ && options.allow_unknown_host_offload === true);
524
+ }
525
+ function hasConfiguredHostOffload(value) {
526
+ const settings = isObject(value.settings) ? value.settings : undefined;
527
+ if (settings && Number(settings["n-cpu-moe"] ?? 0) > 0)
528
+ return true;
529
+ const server = isObject(value.server) ? value.server : undefined;
530
+ if (server && Number(server.cpu_offload_gb ?? 0) > 0)
531
+ return true;
532
+ const residency = isObject(value.residency) ? value.residency : undefined;
533
+ const offload = residency && isObject(residency.offload) ? residency.offload : undefined;
534
+ if (!offload)
535
+ return false;
536
+ if (offload.enabled === true)
537
+ return true;
538
+ return typeof offload.components === "string"
539
+ ? offload.components.trim().length > 0
540
+ : Array.isArray(offload.components) && offload.components.length > 0;
541
+ }
542
+ function fitResult(status, confidence, capacity, usable, weight, runtime, workload, required, remaining, reasons) {
543
+ return {
544
+ status,
545
+ confidence,
546
+ capacity_bytes: capacity,
547
+ usable_capacity_bytes: usable,
548
+ weight_bytes: weight,
549
+ runtime_overhead_bytes: runtime,
550
+ workload_bytes: workload,
551
+ required_device_bytes: required,
552
+ remaining_device_bytes: remaining,
553
+ reasons
554
+ };
555
+ }
556
+ function mergeFit(base, exact) {
557
+ return {
558
+ ...base,
559
+ ...exact,
560
+ ...(base?.workload || exact?.workload ? { workload: { ...base?.workload, ...exact?.workload } } : {})
561
+ };
562
+ }
563
+ function fitForWorkload(base, workload) {
564
+ const fixed = workload.fixed_device_bytes ?? base.workload?.fixed_device_bytes;
565
+ const source = workload.source ?? base.workload?.source;
566
+ return {
567
+ ...base,
568
+ workload: {
569
+ ...(base.workload ?? {}),
570
+ ...(fixed !== undefined ? { fixed_device_bytes: fixed } : {}),
571
+ bytes_per_token: workload.bytes_per_token,
572
+ context_tokens: workload.desired_context_tokens,
573
+ concurrency: workload.desired_concurrency,
574
+ ...(source ? { source } : {})
575
+ }
576
+ };
577
+ }
578
+ function omitWorkload(value) {
579
+ const { workload: _workload, ...result } = value;
580
+ return result;
581
+ }
582
+ function skip(request, reason, details) {
583
+ return { model_id: request.model_id, required: request.required ?? false, reason, details };
584
+ }
585
+ function tierForScore(score) {
586
+ if (score >= 100)
587
+ return "reference";
588
+ if (score >= 90)
589
+ return "near-lossless";
590
+ if (score >= 75)
591
+ return "high";
592
+ if (score >= 55)
593
+ return "deployment";
594
+ if (score >= 35)
595
+ return "compromise";
596
+ if (score >= 20)
597
+ return "fit-floor";
598
+ return "extreme";
599
+ }
600
+ function positive(value, label) {
601
+ if (!Number.isInteger(value) || value <= 0)
602
+ throw new Error(`${label} must be a positive integer`);
603
+ return value;
604
+ }
605
+ function formatGiB(bytes) { return (bytes / GIB).toFixed(2); }
606
+ function message(error) { return error instanceof Error ? error.message : String(error); }
607
+ function firstString(...values) {
608
+ return values.find((value) => typeof value === "string" && value.length > 0);
609
+ }
610
+ //# sourceMappingURL=planning.js.map