prefer-inference-core 0.0.0-g0be6ef9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +166 -0
- package/dist/catalog.d.ts +27 -0
- package/dist/catalog.d.ts.map +1 -0
- package/dist/catalog.js +547 -0
- package/dist/catalog.js.map +1 -0
- package/dist/cli.d.ts +3 -0
- package/dist/cli.d.ts.map +1 -0
- package/dist/cli.js +412 -0
- package/dist/cli.js.map +1 -0
- package/dist/huggingface.d.ts +18 -0
- package/dist/huggingface.d.ts.map +1 -0
- package/dist/huggingface.js +296 -0
- package/dist/huggingface.js.map +1 -0
- package/dist/index.d.ts +8 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +8 -0
- package/dist/index.js.map +1 -0
- package/dist/planning.d.ts +11 -0
- package/dist/planning.d.ts.map +1 -0
- package/dist/planning.js +610 -0
- package/dist/planning.js.map +1 -0
- package/dist/prefer.mjs +9493 -0
- package/dist/prefer.mjs.map +7 -0
- package/dist/releases.d.ts +33 -0
- package/dist/releases.d.ts.map +1 -0
- package/dist/releases.js +288 -0
- package/dist/releases.js.map +1 -0
- package/dist/resources.d.ts +10 -0
- package/dist/resources.d.ts.map +1 -0
- package/dist/resources.js +250 -0
- package/dist/resources.js.map +1 -0
- package/dist/types.d.ts +409 -0
- package/dist/types.d.ts.map +1 -0
- package/dist/types.js +2 -0
- package/dist/types.js.map +1 -0
- package/dist/utils.d.ts +20 -0
- package/dist/utils.d.ts.map +1 -0
- package/dist/utils.js +112 -0
- package/dist/utils.js.map +1 -0
- package/package.json +58 -0
- package/schemas/prefer-model-catalog-extension.schema.json +25 -0
- package/schemas/prefer-model-catalog.schema.json +54 -0
- package/schemas/prefer-model-plan.schema.json +115 -0
- package/schemas/prefer-resource-profile.schema.json +70 -0
package/dist/planning.js
ADDED
|
@@ -0,0 +1,610 @@
|
|
|
1
|
+
import { listModelVariants, resolveCatalogModelId, resolveModelVariant } from "./catalog.js";
|
|
2
|
+
import { isObject } from "./utils.js";
|
|
3
|
+
const GIB = 1024 ** 3;
|
|
4
|
+
const DEFAULT_HEADROOM_FRACTION = 0.04;
|
|
5
|
+
const DEFAULT_MINIMUM_HEADROOM_BYTES = 1.5 * GIB;
|
|
6
|
+
const DEFAULT_MAXIMUM_HEADROOM_BYTES = 4 * GIB;
|
|
7
|
+
const QUALITY_SCORES = {
|
|
8
|
+
reference: 100,
|
|
9
|
+
"near-lossless": 90,
|
|
10
|
+
high: 80,
|
|
11
|
+
deployment: 60,
|
|
12
|
+
compromise: 45,
|
|
13
|
+
"fit-floor": 25,
|
|
14
|
+
extreme: 10,
|
|
15
|
+
unknown: 0
|
|
16
|
+
};
|
|
17
|
+
export function quantQuality(quant) {
|
|
18
|
+
const normalized = quant.toLowerCase().replaceAll("_", "-");
|
|
19
|
+
const scores = [];
|
|
20
|
+
if (/(?:^|-)(?:f32|fp32|f16|fp16|bf16)(?:-|$)/u.test(normalized))
|
|
21
|
+
scores.push(100);
|
|
22
|
+
if (/(?:^|-)(?:fp8|q8|int8)(?:-|$)/u.test(normalized))
|
|
23
|
+
scores.push(90);
|
|
24
|
+
if (/(?:^|-)(?:q6|iq6)(?:-|$)/u.test(normalized))
|
|
25
|
+
scores.push(80);
|
|
26
|
+
if (/(?:^|-)(?:q5|iq5)(?:-|$)/u.test(normalized))
|
|
27
|
+
scores.push(70);
|
|
28
|
+
if (/(?:^|-)(?:nvfp4|mxfp4|w4a16|q4|iq4|int4)(?:-|$)/u.test(normalized))
|
|
29
|
+
scores.push(60);
|
|
30
|
+
if (/(?:^|-)(?:q3|iq3)(?:-|$)/u.test(normalized))
|
|
31
|
+
scores.push(45);
|
|
32
|
+
if (/(?:^|-)(?:q2|iq2)(?:-|$)/u.test(normalized))
|
|
33
|
+
scores.push(25);
|
|
34
|
+
if (/(?:^|-)(?:q1|iq1)(?:-|$)/u.test(normalized))
|
|
35
|
+
scores.push(10);
|
|
36
|
+
if (!scores.length)
|
|
37
|
+
return { tier: "unknown", score: 0 };
|
|
38
|
+
let score = Math.min(...scores);
|
|
39
|
+
if (normalized.includes("xl"))
|
|
40
|
+
score += 3;
|
|
41
|
+
else if (normalized.includes("k-m"))
|
|
42
|
+
score += 2;
|
|
43
|
+
else if (normalized.includes("k-s"))
|
|
44
|
+
score += 1;
|
|
45
|
+
if (normalized.includes("iq") && score >= 45)
|
|
46
|
+
score -= 1;
|
|
47
|
+
return { tier: tierForScore(score), score };
|
|
48
|
+
}
|
|
49
|
+
export function estimateVariantFit(variant, resources, options = {}) {
|
|
50
|
+
const capacity = acceleratorCapacity(resources, options.allow_multi_gpu ?? true);
|
|
51
|
+
const weightBytes = options.device_weight_bytes ?? variant.artifact_bytes;
|
|
52
|
+
const runtimeOverhead = options.runtime_overhead_bytes ?? Math.max(512 * 1024 ** 2, Math.ceil(weightBytes * 0.02));
|
|
53
|
+
const workloadBytes = workloadMemory(options);
|
|
54
|
+
const required = weightBytes + runtimeOverhead + workloadBytes;
|
|
55
|
+
const reasons = [];
|
|
56
|
+
const confidence = options.workload?.source === "measured"
|
|
57
|
+
? "measured"
|
|
58
|
+
: options.workload?.source === "architecture"
|
|
59
|
+
? "architecture"
|
|
60
|
+
: capacity === undefined
|
|
61
|
+
? "unknown"
|
|
62
|
+
: "artifact-only";
|
|
63
|
+
if (capacity === undefined) {
|
|
64
|
+
return {
|
|
65
|
+
status: "unknown",
|
|
66
|
+
confidence,
|
|
67
|
+
weight_bytes: weightBytes,
|
|
68
|
+
runtime_overhead_bytes: runtimeOverhead,
|
|
69
|
+
workload_bytes: workloadBytes,
|
|
70
|
+
required_device_bytes: required,
|
|
71
|
+
reasons: ["No usable accelerator or unified-memory capacity was supplied."]
|
|
72
|
+
};
|
|
73
|
+
}
|
|
74
|
+
const headroom = reservedHeadroom(capacity, options);
|
|
75
|
+
const usable = Math.max(0, capacity - headroom);
|
|
76
|
+
const remaining = capacity - required;
|
|
77
|
+
if (required <= usable) {
|
|
78
|
+
reasons.push(`Estimated device use leaves ${formatGiB(remaining)} GiB before the reserved headroom boundary.`);
|
|
79
|
+
return fitResult("fits", confidence, capacity, usable, weightBytes, runtimeOverhead, workloadBytes, required, remaining, reasons);
|
|
80
|
+
}
|
|
81
|
+
const deficit = Math.max(0, required - usable);
|
|
82
|
+
const hostAvailable = hostCapacity(resources);
|
|
83
|
+
const maximumOffload = options.maximum_host_offload_bytes ?? hostAvailable;
|
|
84
|
+
if (options.allow_host_offload && resources.memory_topology === "discrete" && deficit > 0) {
|
|
85
|
+
if (hostAvailable !== undefined &&
|
|
86
|
+
maximumOffload !== undefined &&
|
|
87
|
+
deficit <= Math.min(hostAvailable, maximumOffload)) {
|
|
88
|
+
reasons.push(`The route requires approximately ${formatGiB(deficit)} GiB of host offload.`);
|
|
89
|
+
return {
|
|
90
|
+
...fitResult("host-offload", confidence, capacity, usable, weightBytes, runtimeOverhead, workloadBytes, required, remaining, reasons),
|
|
91
|
+
host_offload_bytes: deficit
|
|
92
|
+
};
|
|
93
|
+
}
|
|
94
|
+
if (hostAvailable === undefined
|
|
95
|
+
&& options.allow_unknown_host_offload
|
|
96
|
+
&& (maximumOffload === undefined || deficit <= maximumOffload)) {
|
|
97
|
+
reasons.push(`The configured offload route needs approximately ${formatGiB(deficit)} GiB of host memory; runtime host capacity has not been observed.`);
|
|
98
|
+
return {
|
|
99
|
+
...fitResult("unknown", "unknown", capacity, usable, weightBytes, runtimeOverhead, workloadBytes, required, remaining, reasons),
|
|
100
|
+
host_offload_bytes: deficit
|
|
101
|
+
};
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
if (required <= capacity) {
|
|
105
|
+
reasons.push(`The variant fits nominal capacity but leaves less than the ${formatGiB(headroom)} GiB reserve.`);
|
|
106
|
+
return fitResult("tight", confidence, capacity, usable, weightBytes, runtimeOverhead, workloadBytes, required, remaining, reasons);
|
|
107
|
+
}
|
|
108
|
+
reasons.push(`Estimated device use exceeds the usable device pool by ${formatGiB(deficit)} GiB.`);
|
|
109
|
+
return fitResult("does-not-fit", confidence, capacity, usable, weightBytes, runtimeOverhead, workloadBytes, required, remaining, reasons);
|
|
110
|
+
}
|
|
111
|
+
export function planModelSet(catalog, options) {
|
|
112
|
+
const selected = [];
|
|
113
|
+
const skipped = [];
|
|
114
|
+
const floor = typeof options.quality_floor === "number"
|
|
115
|
+
? options.quality_floor
|
|
116
|
+
: QUALITY_SCORES[options.quality_floor ?? "deployment"];
|
|
117
|
+
const maximumModels = options.max_models
|
|
118
|
+
?? (options.engine === "sglang" || options.engine === "vllm" ? 1 : Number.POSITIVE_INFINITY);
|
|
119
|
+
const storageCapacity = options.max_staged_bytes ?? options.resources.storage?.available_bytes ?? options.resources.storage?.total_bytes ?? Number.POSITIVE_INFINITY;
|
|
120
|
+
let stagedBytes = 0;
|
|
121
|
+
const hints = normalizeHints(options.hints);
|
|
122
|
+
if (options.workload)
|
|
123
|
+
validateWorkload(options.workload);
|
|
124
|
+
const ordered = options.models.map((request, index) => ({ request, index })).sort((left, right) => requestUtility(catalog, right.request, hints) - requestUtility(catalog, left.request, hints) || left.index - right.index);
|
|
125
|
+
for (const { request } of ordered) {
|
|
126
|
+
if (request.workload)
|
|
127
|
+
validateWorkload(request.workload);
|
|
128
|
+
if (selected.length >= maximumModels) {
|
|
129
|
+
skipped.push(skip(request, "model-limit", [`The deployment is limited to ${maximumModels} selected models.`]));
|
|
130
|
+
continue;
|
|
131
|
+
}
|
|
132
|
+
let candidates;
|
|
133
|
+
try {
|
|
134
|
+
candidates = variantCandidates(catalog, request, options, floor);
|
|
135
|
+
}
|
|
136
|
+
catch (error) {
|
|
137
|
+
skipped.push(skip(request, "no-compatible-variant", [message(error)]));
|
|
138
|
+
continue;
|
|
139
|
+
}
|
|
140
|
+
if (!candidates.length) {
|
|
141
|
+
skipped.push(skip(request, "no-compatible-variant", [`No ${options.engine} variant satisfies the quantization floor.`]));
|
|
142
|
+
continue;
|
|
143
|
+
}
|
|
144
|
+
const capabilityCandidates = candidates.filter((variant) => !missingModelCapabilities(variant, hints).length);
|
|
145
|
+
if (!capabilityCandidates.length) {
|
|
146
|
+
skipped.push(skip(request, "missing-capability", [
|
|
147
|
+
`The model does not provide required capabilities: ${hints.required_capabilities?.join(", ") ?? "unknown"}.`
|
|
148
|
+
]));
|
|
149
|
+
continue;
|
|
150
|
+
}
|
|
151
|
+
const attempted = [];
|
|
152
|
+
const workloadRequest = request.workload ?? options.workload;
|
|
153
|
+
const baseFit = mergeFit(options.fit, request.fit);
|
|
154
|
+
const desiredFit = workloadRequest ? fitForWorkload(baseFit, workloadRequest) : baseFit;
|
|
155
|
+
let choice;
|
|
156
|
+
for (const variant of capabilityCandidates) {
|
|
157
|
+
const missing = missingCapabilities(variant, options.resources);
|
|
158
|
+
if (missing.length) {
|
|
159
|
+
attempted.push(`${variant.quant}: missing ${missing.join(", ")}`);
|
|
160
|
+
continue;
|
|
161
|
+
}
|
|
162
|
+
const fit = estimateVariantFit(variant, options.resources, desiredFit);
|
|
163
|
+
attempted.push(`${variant.quant}: ${fit.status}`);
|
|
164
|
+
if (acceptableFit(fit, desiredFit, options.accept_tight)) {
|
|
165
|
+
choice = { variant, fit };
|
|
166
|
+
break;
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
if (!choice && workloadRequest) {
|
|
170
|
+
for (const variant of capabilityCandidates) {
|
|
171
|
+
if (missingCapabilities(variant, options.resources).length)
|
|
172
|
+
continue;
|
|
173
|
+
const workload = tuneWorkloadToFit(variant, options.resources, workloadRequest, omitWorkload(baseFit));
|
|
174
|
+
if (!workload.fits) {
|
|
175
|
+
attempted.push(`${variant.quant}: minimum workload does not fit`);
|
|
176
|
+
continue;
|
|
177
|
+
}
|
|
178
|
+
const fit = estimateVariantFit(variant, options.resources, fitForWorkload(baseFit, {
|
|
179
|
+
...workloadRequest,
|
|
180
|
+
desired_context_tokens: workload.context_tokens,
|
|
181
|
+
desired_concurrency: workload.concurrency
|
|
182
|
+
}));
|
|
183
|
+
if (acceptableFit(fit, baseFit, options.accept_tight)) {
|
|
184
|
+
choice = { variant, fit, workload };
|
|
185
|
+
attempted.push(`${variant.quant}: ${workload.reason}`);
|
|
186
|
+
break;
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
if (!choice) {
|
|
191
|
+
skipped.push(skip(request, "insufficient-memory", attempted));
|
|
192
|
+
continue;
|
|
193
|
+
}
|
|
194
|
+
if (stagedBytes + choice.variant.artifact_bytes > storageCapacity) {
|
|
195
|
+
skipped.push(skip(request, "storage-budget", [
|
|
196
|
+
`${choice.variant.quant} needs ${choice.variant.artifact_bytes} artifact bytes; ${Math.max(0, storageCapacity - stagedBytes)} remain.`
|
|
197
|
+
]));
|
|
198
|
+
continue;
|
|
199
|
+
}
|
|
200
|
+
const preferredQuant = preferredVariant(catalog, request, options).quant;
|
|
201
|
+
const quantChanged = choice.variant.quant !== preferredQuant;
|
|
202
|
+
selected.push({
|
|
203
|
+
model_id: choice.variant.model_id,
|
|
204
|
+
preferred_quant: preferredQuant,
|
|
205
|
+
...(request.quant ? { requested_quant: request.quant } : {}),
|
|
206
|
+
selected: choice.variant,
|
|
207
|
+
fit: choice.fit,
|
|
208
|
+
quant_changed: quantChanged,
|
|
209
|
+
...(workloadRequest ? { workload_request: structuredClone(workloadRequest) } : {}),
|
|
210
|
+
...(choice.workload ? { workload: choice.workload } : {}),
|
|
211
|
+
reason: quantChanged
|
|
212
|
+
? `${preferredQuant} did not meet the resource policy; selected ${choice.variant.quant}.`
|
|
213
|
+
: `${choice.variant.quant} meets the resource policy.`
|
|
214
|
+
});
|
|
215
|
+
stagedBytes += choice.variant.artifact_bytes;
|
|
216
|
+
}
|
|
217
|
+
const requiredSkipped = skipped.some((entry) => entry.required);
|
|
218
|
+
return {
|
|
219
|
+
schema_version: "prefer.model-plan.v1",
|
|
220
|
+
engine: options.engine,
|
|
221
|
+
resources: structuredClone(options.resources),
|
|
222
|
+
selected,
|
|
223
|
+
skipped,
|
|
224
|
+
capabilities: [...new Set(selected.flatMap((entry) => entry.selected.capabilities))].sort(),
|
|
225
|
+
staged_artifact_bytes: stagedBytes,
|
|
226
|
+
complete: !requiredSkipped,
|
|
227
|
+
...(Object.keys(hints).length ? { hints } : {})
|
|
228
|
+
};
|
|
229
|
+
}
|
|
230
|
+
/** Convert a generated engine deployment's existing model or bundle members
|
|
231
|
+
* into planner requests. Bundle members are optional by default; single-model
|
|
232
|
+
* deployments remain required. A deployment quant is a starting point, not an
|
|
233
|
+
* exact lock, so constrained hardware may choose a smaller published lane. */
|
|
234
|
+
export function modelRequestsFromDeployment(value) {
|
|
235
|
+
if (!isObject(value) || !Array.isArray(value.models))
|
|
236
|
+
return [];
|
|
237
|
+
const bundle = value.kind === "bundle"
|
|
238
|
+
|| String(value.id ?? "").endsWith("/general")
|
|
239
|
+
|| (value.models.length > 1 && value.kind !== "single-model");
|
|
240
|
+
const result = new Map();
|
|
241
|
+
const total = value.models.length;
|
|
242
|
+
for (const [index, raw] of value.models.entries()) {
|
|
243
|
+
if (!isObject(raw))
|
|
244
|
+
continue;
|
|
245
|
+
const modelId = firstString(raw.request_model_id, raw.model_slug, raw.profile_id);
|
|
246
|
+
if (!modelId || result.has(modelId))
|
|
247
|
+
continue;
|
|
248
|
+
const preferredQuant = firstString(raw.quant_slug, raw.quant);
|
|
249
|
+
const configuredOffload = hasConfiguredHostOffload(raw) || hasConfiguredHostOffload(value);
|
|
250
|
+
result.set(modelId, {
|
|
251
|
+
model_id: modelId,
|
|
252
|
+
...(preferredQuant ? { preferred_quant: preferredQuant } : {}),
|
|
253
|
+
required: !bundle,
|
|
254
|
+
priority: total - index,
|
|
255
|
+
...(configuredOffload ? {
|
|
256
|
+
fit: {
|
|
257
|
+
allow_host_offload: true,
|
|
258
|
+
allow_unknown_host_offload: true
|
|
259
|
+
}
|
|
260
|
+
} : {})
|
|
261
|
+
});
|
|
262
|
+
}
|
|
263
|
+
return [...result.values()];
|
|
264
|
+
}
|
|
265
|
+
export function tuneWorkloadToFit(variant, resources, request, options = {}) {
|
|
266
|
+
if (!Number.isFinite(request.bytes_per_token) || request.bytes_per_token <= 0) {
|
|
267
|
+
throw new Error("bytes_per_token must be greater than zero");
|
|
268
|
+
}
|
|
269
|
+
const desiredContext = positive(request.desired_context_tokens, "desired_context_tokens");
|
|
270
|
+
const desiredConcurrency = positive(request.desired_concurrency, "desired_concurrency");
|
|
271
|
+
const minimumContext = positive(request.minimum_context_tokens ?? 1, "minimum_context_tokens");
|
|
272
|
+
const minimumConcurrency = positive(request.minimum_concurrency ?? 1, "minimum_concurrency");
|
|
273
|
+
const capacity = acceleratorCapacity(resources, options.allow_multi_gpu ?? true);
|
|
274
|
+
if (capacity === undefined)
|
|
275
|
+
return {
|
|
276
|
+
fits: false,
|
|
277
|
+
context_tokens: 0,
|
|
278
|
+
concurrency: 0,
|
|
279
|
+
aggregate_token_capacity: 0,
|
|
280
|
+
changed: true,
|
|
281
|
+
reason: "No usable device capacity is known."
|
|
282
|
+
};
|
|
283
|
+
const weightBytes = options.device_weight_bytes ?? variant.artifact_bytes;
|
|
284
|
+
const runtimeOverhead = options.runtime_overhead_bytes ?? Math.max(512 * 1024 ** 2, Math.ceil(weightBytes * 0.02));
|
|
285
|
+
const headroom = reservedHeadroom(capacity, options);
|
|
286
|
+
const tokenBudgetBytes = capacity - headroom - weightBytes - runtimeOverhead - (request.fixed_device_bytes ?? 0);
|
|
287
|
+
const aggregateCapacity = Math.max(0, Math.floor(tokenBudgetBytes / request.bytes_per_token));
|
|
288
|
+
const desiredTotal = desiredContext * desiredConcurrency;
|
|
289
|
+
if (desiredTotal <= aggregateCapacity)
|
|
290
|
+
return {
|
|
291
|
+
fits: true,
|
|
292
|
+
context_tokens: desiredContext,
|
|
293
|
+
concurrency: desiredConcurrency,
|
|
294
|
+
aggregate_token_capacity: aggregateCapacity,
|
|
295
|
+
changed: false,
|
|
296
|
+
reason: "The requested context and concurrency fit the estimated token pool."
|
|
297
|
+
};
|
|
298
|
+
let context = desiredContext;
|
|
299
|
+
let concurrency = desiredConcurrency;
|
|
300
|
+
if ((request.priority ?? "context") === "context") {
|
|
301
|
+
concurrency = Math.min(desiredConcurrency, Math.floor(aggregateCapacity / desiredContext));
|
|
302
|
+
concurrency = Math.max(minimumConcurrency, concurrency);
|
|
303
|
+
context = Math.min(desiredContext, Math.floor(aggregateCapacity / concurrency));
|
|
304
|
+
}
|
|
305
|
+
else {
|
|
306
|
+
context = Math.min(desiredContext, Math.floor(aggregateCapacity / desiredConcurrency));
|
|
307
|
+
context = Math.max(minimumContext, context);
|
|
308
|
+
concurrency = Math.min(desiredConcurrency, Math.floor(aggregateCapacity / context));
|
|
309
|
+
}
|
|
310
|
+
const fits = context >= minimumContext && concurrency >= minimumConcurrency && context * concurrency <= aggregateCapacity;
|
|
311
|
+
return {
|
|
312
|
+
fits,
|
|
313
|
+
context_tokens: fits ? context : 0,
|
|
314
|
+
concurrency: fits ? concurrency : 0,
|
|
315
|
+
aggregate_token_capacity: aggregateCapacity,
|
|
316
|
+
changed: true,
|
|
317
|
+
reason: fits
|
|
318
|
+
? `Adjusted to ${concurrency} request slot(s) at ${context} tokens each.`
|
|
319
|
+
: "The minimum context and concurrency do not fit the estimated token pool."
|
|
320
|
+
};
|
|
321
|
+
}
|
|
322
|
+
function variantCandidates(catalog, request, options, floor) {
|
|
323
|
+
const preferred = preferredVariant(catalog, request, options);
|
|
324
|
+
if (request.quant)
|
|
325
|
+
return [preferred];
|
|
326
|
+
const bias = options.hints?.quant_bias ?? "balanced";
|
|
327
|
+
const preferredScore = quantQuality(preferred.quant).score;
|
|
328
|
+
const variants = listModelVariants(catalog, request.model_id, options.engine)
|
|
329
|
+
.filter((variant) => request.repository
|
|
330
|
+
? variant.repository === request.repository
|
|
331
|
+
: bias === "balanced"
|
|
332
|
+
? variant.repository === preferred.repository || variant.artifact_bytes < preferred.artifact_bytes
|
|
333
|
+
: true)
|
|
334
|
+
.filter((variant) => options.use_nvfp4 || variant.quant === preferred.quant || !variant.quant.toLowerCase().includes("nvfp4"))
|
|
335
|
+
.filter((variant) => {
|
|
336
|
+
const score = quantQuality(variant.quant).score;
|
|
337
|
+
return variant.quant === preferred.quant || (score >= floor && (bias !== "balanced" || preferredScore === 0 || score <= preferredScore));
|
|
338
|
+
})
|
|
339
|
+
.sort((left, right) => {
|
|
340
|
+
if (bias === "capacity") {
|
|
341
|
+
return left.artifact_bytes - right.artifact_bytes || quantQuality(right.quant).score - quantQuality(left.quant).score;
|
|
342
|
+
}
|
|
343
|
+
if (bias === "quality") {
|
|
344
|
+
return quantQuality(right.quant).score - quantQuality(left.quant).score || left.artifact_bytes - right.artifact_bytes;
|
|
345
|
+
}
|
|
346
|
+
if (left.repository === preferred.repository && right.repository !== preferred.repository)
|
|
347
|
+
return -1;
|
|
348
|
+
if (right.repository === preferred.repository && left.repository !== preferred.repository)
|
|
349
|
+
return 1;
|
|
350
|
+
return quantQuality(right.quant).score - quantQuality(left.quant).score || right.artifact_bytes - left.artifact_bytes;
|
|
351
|
+
});
|
|
352
|
+
const deduped = new Map();
|
|
353
|
+
if (bias === "balanced")
|
|
354
|
+
deduped.set(`${preferred.repository}@${preferred.quant}`, preferred);
|
|
355
|
+
for (const variant of variants)
|
|
356
|
+
deduped.set(`${variant.repository}@${variant.quant}`, variant);
|
|
357
|
+
if (!deduped.has(`${preferred.repository}@${preferred.quant}`)) {
|
|
358
|
+
deduped.set(`${preferred.repository}@${preferred.quant}`, preferred);
|
|
359
|
+
}
|
|
360
|
+
return [...deduped.values()];
|
|
361
|
+
}
|
|
362
|
+
function preferredVariant(catalog, request, options) {
|
|
363
|
+
return resolveModelVariant(catalog, request.model_id, {
|
|
364
|
+
engine: options.engine,
|
|
365
|
+
...(request.repository ? { repository: request.repository } : {}),
|
|
366
|
+
...(request.quant ?? request.preferred_quant ? { quant: (request.quant ?? request.preferred_quant) } : {}),
|
|
367
|
+
useNvfp4: options.use_nvfp4,
|
|
368
|
+
overrides: request.overrides
|
|
369
|
+
});
|
|
370
|
+
}
|
|
371
|
+
function missingCapabilities(variant, resources) {
|
|
372
|
+
const normalized = variant.quant.toLowerCase().replaceAll("_", "-");
|
|
373
|
+
const required = [];
|
|
374
|
+
if (normalized.includes("nvfp4"))
|
|
375
|
+
required.push("nvfp4");
|
|
376
|
+
if (normalized === "fp8" && (variant.engine === "sglang" || variant.engine === "vllm"))
|
|
377
|
+
required.push("fp8");
|
|
378
|
+
return required.filter((capability) => !resources.capabilities.includes(capability));
|
|
379
|
+
}
|
|
380
|
+
function missingModelCapabilities(variant, hints) {
|
|
381
|
+
return (hints.required_capabilities ?? []).filter((capability) => !variant.capabilities.includes(capability));
|
|
382
|
+
}
|
|
383
|
+
function requestUtility(catalog, request, hints) {
|
|
384
|
+
let modelId;
|
|
385
|
+
try {
|
|
386
|
+
modelId = resolveCatalogModelId(catalog, request.model_id);
|
|
387
|
+
}
|
|
388
|
+
catch {
|
|
389
|
+
return request.priority ?? 0;
|
|
390
|
+
}
|
|
391
|
+
const model = catalog.models[modelId];
|
|
392
|
+
const capabilities = new Set(model.capabilities ?? []);
|
|
393
|
+
const preferredRoles = new Set(model.profile?.roles?.preferred ?? []);
|
|
394
|
+
const capableRoles = new Set(model.profile?.roles?.capable ?? []);
|
|
395
|
+
const avoidRoles = new Set(model.profile?.roles?.avoid ?? []);
|
|
396
|
+
let score = request.priority ?? 0;
|
|
397
|
+
for (const capability of hints.preferred_capabilities ?? [])
|
|
398
|
+
if (capabilities.has(capability))
|
|
399
|
+
score += 100;
|
|
400
|
+
for (const role of hints.preferred_roles ?? []) {
|
|
401
|
+
if (preferredRoles.has(role))
|
|
402
|
+
score += 80;
|
|
403
|
+
else if (capableRoles.has(role))
|
|
404
|
+
score += 40;
|
|
405
|
+
if (avoidRoles.has(role))
|
|
406
|
+
score -= 80;
|
|
407
|
+
}
|
|
408
|
+
const speedImportance = hints.speed_importance ?? 0;
|
|
409
|
+
if (speedImportance > 0) {
|
|
410
|
+
if (request.speed_score !== undefined && (!Number.isFinite(request.speed_score) || request.speed_score < 0 || request.speed_score > 100)) {
|
|
411
|
+
throw new Error(`${request.model_id} speed_score must be between 0 and 100`);
|
|
412
|
+
}
|
|
413
|
+
const activeParameters = model.profile?.architecture?.active_parameters_b;
|
|
414
|
+
const inferredSpeed = typeof activeParameters === "number" && activeParameters > 0
|
|
415
|
+
? 100 / (1 + Math.log2(1 + activeParameters))
|
|
416
|
+
: 0;
|
|
417
|
+
score += speedImportance * (request.speed_score ?? inferredSpeed);
|
|
418
|
+
}
|
|
419
|
+
const qualityImportance = hints.quality_importance ?? 0;
|
|
420
|
+
if (qualityImportance > 0 && request.quality_score !== undefined) {
|
|
421
|
+
if (!Number.isFinite(request.quality_score) || request.quality_score < 0 || request.quality_score > 100) {
|
|
422
|
+
throw new Error(`${request.model_id} quality_score must be between 0 and 100`);
|
|
423
|
+
}
|
|
424
|
+
score += qualityImportance * request.quality_score;
|
|
425
|
+
}
|
|
426
|
+
return score;
|
|
427
|
+
}
|
|
428
|
+
function normalizeHints(value) {
|
|
429
|
+
if (!value)
|
|
430
|
+
return {};
|
|
431
|
+
if (value.speed_importance !== undefined && (!Number.isFinite(value.speed_importance) || value.speed_importance < 0 || value.speed_importance > 1))
|
|
432
|
+
throw new Error("speed_importance must be between 0 and 1");
|
|
433
|
+
if (value.quality_importance !== undefined && (!Number.isFinite(value.quality_importance) || value.quality_importance < 0 || value.quality_importance > 1))
|
|
434
|
+
throw new Error("quality_importance must be between 0 and 1");
|
|
435
|
+
if (value.quant_bias && !["quality", "balanced", "capacity"].includes(value.quant_bias)) {
|
|
436
|
+
throw new Error("quant_bias must be quality, balanced, or capacity");
|
|
437
|
+
}
|
|
438
|
+
return {
|
|
439
|
+
...(value.required_capabilities?.length ? { required_capabilities: uniqueStrings(value.required_capabilities) } : {}),
|
|
440
|
+
...(value.preferred_capabilities?.length ? { preferred_capabilities: uniqueStrings(value.preferred_capabilities) } : {}),
|
|
441
|
+
...(value.preferred_roles?.length ? { preferred_roles: uniqueStrings(value.preferred_roles) } : {}),
|
|
442
|
+
...(value.speed_importance !== undefined ? { speed_importance: value.speed_importance } : {}),
|
|
443
|
+
...(value.quality_importance !== undefined ? { quality_importance: value.quality_importance } : {}),
|
|
444
|
+
...(value.quant_bias ? { quant_bias: value.quant_bias } : {})
|
|
445
|
+
};
|
|
446
|
+
}
|
|
447
|
+
function uniqueStrings(values) {
|
|
448
|
+
return [...new Set(values.map((value) => value.trim()).filter(Boolean))];
|
|
449
|
+
}
|
|
450
|
+
function validateWorkload(value) {
|
|
451
|
+
positive(value.desired_context_tokens, "desired_context_tokens");
|
|
452
|
+
positive(value.desired_concurrency, "desired_concurrency");
|
|
453
|
+
positive(value.minimum_context_tokens ?? 1, "minimum_context_tokens");
|
|
454
|
+
positive(value.minimum_concurrency ?? 1, "minimum_concurrency");
|
|
455
|
+
if (!Number.isFinite(value.bytes_per_token) || value.bytes_per_token <= 0) {
|
|
456
|
+
throw new Error("bytes_per_token must be greater than zero");
|
|
457
|
+
}
|
|
458
|
+
if (value.fixed_device_bytes !== undefined && (!Number.isFinite(value.fixed_device_bytes) || value.fixed_device_bytes < 0)) {
|
|
459
|
+
throw new Error("fixed_device_bytes must be non-negative");
|
|
460
|
+
}
|
|
461
|
+
if ((value.minimum_context_tokens ?? 1) > value.desired_context_tokens) {
|
|
462
|
+
throw new Error("minimum_context_tokens may not exceed desired_context_tokens");
|
|
463
|
+
}
|
|
464
|
+
if ((value.minimum_concurrency ?? 1) > value.desired_concurrency) {
|
|
465
|
+
throw new Error("minimum_concurrency may not exceed desired_concurrency");
|
|
466
|
+
}
|
|
467
|
+
}
|
|
468
|
+
function acceleratorCapacity(resources, allowMultiGpu) {
|
|
469
|
+
if (resources.memory_topology === "unified") {
|
|
470
|
+
const unified = resources.unified_memory?.available_bytes ?? resources.unified_memory?.total_bytes;
|
|
471
|
+
const host = resources.host_memory?.available_bytes ?? resources.host_memory?.total_bytes;
|
|
472
|
+
const accelerator = resources.accelerators
|
|
473
|
+
.map((entry) => entry.available_bytes ?? entry.total_bytes)
|
|
474
|
+
.filter((entry) => entry !== undefined)
|
|
475
|
+
.reduce((minimum, entry) => minimum === undefined ? entry : Math.min(minimum, entry), undefined);
|
|
476
|
+
const observations = [unified, host, accelerator].filter((entry) => entry !== undefined);
|
|
477
|
+
return observations.length ? Math.min(...observations) : undefined;
|
|
478
|
+
}
|
|
479
|
+
if (resources.memory_topology === "host" && !resources.accelerators.length) {
|
|
480
|
+
return resources.host_memory?.available_bytes ?? resources.host_memory?.total_bytes;
|
|
481
|
+
}
|
|
482
|
+
const deviceCapacities = resources.accelerators
|
|
483
|
+
.map((entry) => entry.available_bytes ?? entry.total_bytes)
|
|
484
|
+
.filter((entry) => entry !== undefined);
|
|
485
|
+
if (!deviceCapacities.length)
|
|
486
|
+
return undefined;
|
|
487
|
+
return allowMultiGpu ? deviceCapacities.reduce((total, entry) => total + entry, 0) : Math.max(...deviceCapacities);
|
|
488
|
+
}
|
|
489
|
+
function hostCapacity(resources) {
|
|
490
|
+
return resources.host_memory?.available_bytes ?? resources.host_memory?.total_bytes;
|
|
491
|
+
}
|
|
492
|
+
function workloadMemory(options) {
|
|
493
|
+
const workload = options.workload;
|
|
494
|
+
if (!workload)
|
|
495
|
+
return 0;
|
|
496
|
+
const fixed = workload.fixed_device_bytes ?? 0;
|
|
497
|
+
const perToken = workload.bytes_per_token ?? 0;
|
|
498
|
+
const context = workload.context_tokens ?? 0;
|
|
499
|
+
const concurrency = workload.concurrency ?? 1;
|
|
500
|
+
return fixed + perToken * context * concurrency;
|
|
501
|
+
}
|
|
502
|
+
function reservedHeadroom(capacity, options) {
|
|
503
|
+
const fraction = options.headroom_fraction ?? DEFAULT_HEADROOM_FRACTION;
|
|
504
|
+
const minimum = options.minimum_headroom_bytes ?? DEFAULT_MINIMUM_HEADROOM_BYTES;
|
|
505
|
+
const maximum = options.maximum_headroom_bytes ?? DEFAULT_MAXIMUM_HEADROOM_BYTES;
|
|
506
|
+
if (!Number.isFinite(fraction) || fraction < 0 || fraction > 1) {
|
|
507
|
+
throw new Error("headroom_fraction must be between 0 and 1");
|
|
508
|
+
}
|
|
509
|
+
for (const [label, value] of [["minimum_headroom_bytes", minimum], ["maximum_headroom_bytes", maximum]]) {
|
|
510
|
+
if (!Number.isFinite(value) || value < 0)
|
|
511
|
+
throw new Error(`${label} must be non-negative`);
|
|
512
|
+
}
|
|
513
|
+
if (minimum > maximum)
|
|
514
|
+
throw new Error("minimum_headroom_bytes may not exceed maximum_headroom_bytes");
|
|
515
|
+
return Math.min(capacity, Math.max(minimum, Math.min(maximum, Math.floor(capacity * fraction))));
|
|
516
|
+
}
|
|
517
|
+
function acceptableFit(fit, options, acceptTight = false) {
|
|
518
|
+
return fit.status === "fits"
|
|
519
|
+
|| fit.status === "host-offload"
|
|
520
|
+
|| (acceptTight && fit.status === "tight")
|
|
521
|
+
|| (fit.status === "unknown"
|
|
522
|
+
&& fit.host_offload_bytes !== undefined
|
|
523
|
+
&& options.allow_unknown_host_offload === true);
|
|
524
|
+
}
|
|
525
|
+
function hasConfiguredHostOffload(value) {
|
|
526
|
+
const settings = isObject(value.settings) ? value.settings : undefined;
|
|
527
|
+
if (settings && Number(settings["n-cpu-moe"] ?? 0) > 0)
|
|
528
|
+
return true;
|
|
529
|
+
const server = isObject(value.server) ? value.server : undefined;
|
|
530
|
+
if (server && Number(server.cpu_offload_gb ?? 0) > 0)
|
|
531
|
+
return true;
|
|
532
|
+
const residency = isObject(value.residency) ? value.residency : undefined;
|
|
533
|
+
const offload = residency && isObject(residency.offload) ? residency.offload : undefined;
|
|
534
|
+
if (!offload)
|
|
535
|
+
return false;
|
|
536
|
+
if (offload.enabled === true)
|
|
537
|
+
return true;
|
|
538
|
+
return typeof offload.components === "string"
|
|
539
|
+
? offload.components.trim().length > 0
|
|
540
|
+
: Array.isArray(offload.components) && offload.components.length > 0;
|
|
541
|
+
}
|
|
542
|
+
function fitResult(status, confidence, capacity, usable, weight, runtime, workload, required, remaining, reasons) {
|
|
543
|
+
return {
|
|
544
|
+
status,
|
|
545
|
+
confidence,
|
|
546
|
+
capacity_bytes: capacity,
|
|
547
|
+
usable_capacity_bytes: usable,
|
|
548
|
+
weight_bytes: weight,
|
|
549
|
+
runtime_overhead_bytes: runtime,
|
|
550
|
+
workload_bytes: workload,
|
|
551
|
+
required_device_bytes: required,
|
|
552
|
+
remaining_device_bytes: remaining,
|
|
553
|
+
reasons
|
|
554
|
+
};
|
|
555
|
+
}
|
|
556
|
+
function mergeFit(base, exact) {
|
|
557
|
+
return {
|
|
558
|
+
...base,
|
|
559
|
+
...exact,
|
|
560
|
+
...(base?.workload || exact?.workload ? { workload: { ...base?.workload, ...exact?.workload } } : {})
|
|
561
|
+
};
|
|
562
|
+
}
|
|
563
|
+
function fitForWorkload(base, workload) {
|
|
564
|
+
const fixed = workload.fixed_device_bytes ?? base.workload?.fixed_device_bytes;
|
|
565
|
+
const source = workload.source ?? base.workload?.source;
|
|
566
|
+
return {
|
|
567
|
+
...base,
|
|
568
|
+
workload: {
|
|
569
|
+
...(base.workload ?? {}),
|
|
570
|
+
...(fixed !== undefined ? { fixed_device_bytes: fixed } : {}),
|
|
571
|
+
bytes_per_token: workload.bytes_per_token,
|
|
572
|
+
context_tokens: workload.desired_context_tokens,
|
|
573
|
+
concurrency: workload.desired_concurrency,
|
|
574
|
+
...(source ? { source } : {})
|
|
575
|
+
}
|
|
576
|
+
};
|
|
577
|
+
}
|
|
578
|
+
function omitWorkload(value) {
|
|
579
|
+
const { workload: _workload, ...result } = value;
|
|
580
|
+
return result;
|
|
581
|
+
}
|
|
582
|
+
function skip(request, reason, details) {
|
|
583
|
+
return { model_id: request.model_id, required: request.required ?? false, reason, details };
|
|
584
|
+
}
|
|
585
|
+
function tierForScore(score) {
|
|
586
|
+
if (score >= 100)
|
|
587
|
+
return "reference";
|
|
588
|
+
if (score >= 90)
|
|
589
|
+
return "near-lossless";
|
|
590
|
+
if (score >= 75)
|
|
591
|
+
return "high";
|
|
592
|
+
if (score >= 55)
|
|
593
|
+
return "deployment";
|
|
594
|
+
if (score >= 35)
|
|
595
|
+
return "compromise";
|
|
596
|
+
if (score >= 20)
|
|
597
|
+
return "fit-floor";
|
|
598
|
+
return "extreme";
|
|
599
|
+
}
|
|
600
|
+
function positive(value, label) {
|
|
601
|
+
if (!Number.isInteger(value) || value <= 0)
|
|
602
|
+
throw new Error(`${label} must be a positive integer`);
|
|
603
|
+
return value;
|
|
604
|
+
}
|
|
605
|
+
function formatGiB(bytes) { return (bytes / GIB).toFixed(2); }
|
|
606
|
+
function message(error) { return error instanceof Error ? error.message : String(error); }
|
|
607
|
+
function firstString(...values) {
|
|
608
|
+
return values.find((value) => typeof value === "string" && value.length > 0);
|
|
609
|
+
}
|
|
610
|
+
//# sourceMappingURL=planning.js.map
|