@norskvideo/norsk-auto-manager 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +344 -0
- package/lib/src/automanager.d.ts +1070 -0
- package/lib/src/automanager.js +2746 -0
- package/lib/src/clock.d.ts +14 -0
- package/lib/src/clock.js +21 -0
- package/lib/src/conversions.d.ts +59 -0
- package/lib/src/conversions.js +416 -0
- package/lib/src/index.d.ts +9 -0
- package/lib/src/index.js +37 -0
- package/lib/src/inventory.d.ts +34 -0
- package/lib/src/inventory.js +64 -0
- package/lib/src/knownCapabilities.d.ts +18 -0
- package/lib/src/knownCapabilities.js +22 -0
- package/lib/src/placement.d.ts +268 -0
- package/lib/src/placement.js +468 -0
- package/lib/src/settingsValidation.d.ts +34 -0
- package/lib/src/settingsValidation.js +232 -0
- package/lib/src/shared/utils.d.ts +17 -0
- package/lib/src/shared/utils.js +220 -0
- package/lib/src/types.d.ts +295 -0
- package/lib/src/types.js +20 -0
- package/lib/src/validation.d.ts +68 -0
- package/lib/src/validation.js +198 -0
- package/package.json +66 -0
- package/src/automanager.ts +3570 -0
- package/src/clock.ts +29 -0
- package/src/conversions.ts +458 -0
- package/src/index.ts +27 -0
- package/src/inventory.ts +64 -0
- package/src/knownCapabilities.ts +23 -0
- package/src/placement.ts +804 -0
- package/src/settingsValidation.ts +386 -0
- package/src/types.ts +346 -0
- package/src/validation.ts +297 -0
- package/tsconfig.json +25 -0
package/src/placement.ts
ADDED
|
@@ -0,0 +1,804 @@
|
|
|
1
|
+
// Pure-function placement engine. Given a job-to-place, the bundle it
|
|
2
|
+
// belongs to, the current inventory snapshot, and the pool configuration,
|
|
3
|
+
// returns a placement decision: place on a specific node, provision a new
|
|
4
|
+
// node in an elastic pool, or fail.
|
|
5
|
+
//
|
|
6
|
+
// Per §6 of the design doc: filter by hard constraints, score by packing
|
|
7
|
+
// strategy, tiebreak by nodeId for determinism. Per-replica pool
|
|
8
|
+
// preferences (replicaOverrides) take priority over the bundle default.
|
|
9
|
+
//
|
|
10
|
+
// Soft constraints (same-AZ preference, hot-spare scoring bonus) are
|
|
11
|
+
// not yet implemented in this slice — they'll layer on top of the score
|
|
12
|
+
// function once the hard-constraint path is settled and tested.
|
|
13
|
+
|
|
14
|
+
import {
|
|
15
|
+
Bundle,
|
|
16
|
+
BundleId,
|
|
17
|
+
BundleJobSpec,
|
|
18
|
+
Capability,
|
|
19
|
+
CapabilityRequirement,
|
|
20
|
+
GpuResource,
|
|
21
|
+
JobRequirements,
|
|
22
|
+
NodeId,
|
|
23
|
+
NodeInventory,
|
|
24
|
+
ResiliencePolicy,
|
|
25
|
+
} from "./types";
|
|
26
|
+
import {
|
|
27
|
+
freeCapability,
|
|
28
|
+
freeCapacity,
|
|
29
|
+
freeCores,
|
|
30
|
+
usedCapacityFraction,
|
|
31
|
+
} from "./inventory";
|
|
32
|
+
|
|
33
|
+
/** @public */
|
|
34
|
+
export interface NodeView {
|
|
35
|
+
nodeId: NodeId;
|
|
36
|
+
poolName: string;
|
|
37
|
+
/**
|
|
38
|
+
* Which tier within the pool this node belongs to. Set from
|
|
39
|
+
* `tags["tier"]` at provision time for elastic tiers. Pre-registered
|
|
40
|
+
* cluster nodes carry no tier tag (undefined) and are treated as
|
|
41
|
+
* belonging to their pool's `fixed` tier — see `nodeMatchesTier`.
|
|
42
|
+
*/
|
|
43
|
+
tierName?: string;
|
|
44
|
+
inventory: NodeInventory;
|
|
45
|
+
/** Jobs currently running on this node (or pending placement onto it). */
|
|
46
|
+
runningJobs: RunningJob[];
|
|
47
|
+
/** Topology key for the resilience sameAz rule. Cluster pools can leave undefined. */
|
|
48
|
+
az?: string;
|
|
49
|
+
/**
|
|
50
|
+
* Cloud / failure-domain key for the resilience sameCloud rule — the
|
|
51
|
+
* node's provider (e.g. "aws" / "oci" / "cluster"). Undefined when
|
|
52
|
+
* unknown; sameCloud then can't be evaluated against this node.
|
|
53
|
+
*/
|
|
54
|
+
cloud?: string;
|
|
55
|
+
/** Hot spare flag (Phase C). Currently unused by hard-constraint filters. */
|
|
56
|
+
isHotSpare: boolean;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/** @public */
|
|
60
|
+
export interface RunningJob {
|
|
61
|
+
bundleId: BundleId;
|
|
62
|
+
replicaIndex: number;
|
|
63
|
+
jobName: string;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* A pool is an ordered list of capacity tiers. Placement walks the tiers
|
|
68
|
+
* in order (cheapest first) and takes the first that can satisfy the job
|
|
69
|
+
* — the tier order *is* the cost cascade (no cost weight in v1). The
|
|
70
|
+
* homogeneous single-cloud pool of before is just a one-tier pool.
|
|
71
|
+
*
|
|
72
|
+
* @public
|
|
73
|
+
*/
|
|
74
|
+
export interface PlacementPool {
|
|
75
|
+
name: string;
|
|
76
|
+
/** Ordered cheapest → most expensive. Placement tries them in order. */
|
|
77
|
+
tiers: PlacementTier[];
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* One homogeneous capacity segment within a pool: a single cloud/cluster
|
|
82
|
+
* kind + region with its own scale-out behaviour and candidate shapes.
|
|
83
|
+
*
|
|
84
|
+
* @public
|
|
85
|
+
*/
|
|
86
|
+
export interface PlacementTier {
|
|
87
|
+
/** Unique within the pool. Tagged onto provisioned nodes as `tags["tier"]`. */
|
|
88
|
+
name: string;
|
|
89
|
+
/**
|
|
90
|
+
* Cloud / local-servers kind. Used by AutoManager to decide which gRPC
|
|
91
|
+
* call to make when provisioning (createAwsNode vs createOciNode vs
|
|
92
|
+
* local-servers-tier placement, which uses startJob against an existing
|
|
93
|
+
* registered server).
|
|
94
|
+
*/
|
|
95
|
+
kind: "aws" | "oci" | "local-servers";
|
|
96
|
+
/**
|
|
97
|
+
* For aws/oci elastic tiers: which region/availability-domain to
|
|
98
|
+
* provision into. Cluster tiers leave undefined.
|
|
99
|
+
*/
|
|
100
|
+
region?: string;
|
|
101
|
+
packingStrategy: "binpack" | "spread";
|
|
102
|
+
scaleOut: "elastic" | "fixed";
|
|
103
|
+
/**
|
|
104
|
+
* For elastic tiers: shapes that can be provisioned on demand. For
|
|
105
|
+
* fixed tiers: typically empty (the tier consists of registered
|
|
106
|
+
* cluster nodes already in `inventory`).
|
|
107
|
+
*/
|
|
108
|
+
candidateInstanceTypes: InstanceTypeOption[];
|
|
109
|
+
/**
|
|
110
|
+
* How long ahead of a scheduled job's `startDateTime` AutoManager
|
|
111
|
+
* should kick off placement, so the worker is ready to run by the
|
|
112
|
+
* requested time. Cluster tiers place against an existing running
|
|
113
|
+
* node (~instant) so the default is 0. Cloud tiers have to boot a
|
|
114
|
+
* fresh instance — defaults to 5 minutes; operators set per-tier
|
|
115
|
+
* when their boot time differs. AutoManager picks the maximum across
|
|
116
|
+
* the bundle pool's tiers (worst case — guarantees readiness
|
|
117
|
+
* regardless of which tier the placement engine settles on).
|
|
118
|
+
*/
|
|
119
|
+
placementLeadMs?: number;
|
|
120
|
+
/**
|
|
121
|
+
* How nodes in this tier are purchased. `"spot"` requests interruptible
|
|
122
|
+
* spot-market capacity; `"reserved"` is an on-demand launch the operator
|
|
123
|
+
* knows is covered by a standing reservation (same AWS call as on-demand,
|
|
124
|
+
* but capped — see `maxNodes`). Defaults to `"on-demand"`.
|
|
125
|
+
*/
|
|
126
|
+
launchMode?: LaunchMode;
|
|
127
|
+
/**
|
|
128
|
+
* Failure-domain reliability of this tier's capacity. Defaults to
|
|
129
|
+
* `"interruptible"` for spot, `"durable"` otherwise. Read by the
|
|
130
|
+
* resilience engine — an interruptible-tier primary with no durable
|
|
131
|
+
* backup is flagged degraded.
|
|
132
|
+
*/
|
|
133
|
+
reliability?: Reliability;
|
|
134
|
+
/**
|
|
135
|
+
* Cap on the number of nodes this tier may hold (tier-wide, across
|
|
136
|
+
* bundles). The cost cascade spills to the next tier once the cap is
|
|
137
|
+
* reached. Used mainly for reserved tiers — exceeding the reservation
|
|
138
|
+
* count just bills as on-demand, defeating the point. Unset ⇒ no cap.
|
|
139
|
+
*/
|
|
140
|
+
maxNodes?: number;
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/** @public */
|
|
144
|
+
export type LaunchMode = "on-demand" | "spot" | "reserved";
|
|
145
|
+
|
|
146
|
+
/** @public */
|
|
147
|
+
export type Reliability = "durable" | "interruptible";
|
|
148
|
+
|
|
149
|
+
/** The tier's launch mode, defaulting to on-demand. @public */
|
|
150
|
+
export function tierLaunchMode(tier: PlacementTier): LaunchMode {
|
|
151
|
+
return tier.launchMode ?? "on-demand";
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/** The tier's reliability — explicit, else interruptible for spot. @public */
|
|
155
|
+
export function tierReliability(tier: PlacementTier): Reliability {
|
|
156
|
+
return tier.reliability ?? (tierLaunchMode(tier) === "spot" ? "interruptible" : "durable");
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/** Default placement lead time by tier kind. Cluster tiers place
|
|
160
|
+
* against existing running nodes so we don't need lead. Cloud tiers
|
|
161
|
+
* default to 5 minutes — typical EC2/OCI instance boot + image pull
|
|
162
|
+
* budget; operators can override per-tier via PlacementTier. */
|
|
163
|
+
export function defaultPlacementLeadMs(kind: PlacementTier["kind"]): number {
|
|
164
|
+
return kind === "local-servers" ? 0 : 5 * 60 * 1000;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* Whether a node belongs to (pool, tier). Elastic nodes carry an explicit
|
|
169
|
+
* `tierName` (set from `tags["tier"]` at provision). Pre-registered cluster
|
|
170
|
+
* nodes carry no tier tag and are taken to belong to the pool's `fixed`
|
|
171
|
+
* tier — so a pool must have at most one fixed tier (enforced in
|
|
172
|
+
* settingsValidation) for this to be unambiguous.
|
|
173
|
+
*
|
|
174
|
+
* @public
|
|
175
|
+
*/
|
|
176
|
+
export function nodeMatchesTier(
|
|
177
|
+
n: NodeView,
|
|
178
|
+
pool: PlacementPool,
|
|
179
|
+
tier: PlacementTier
|
|
180
|
+
): boolean {
|
|
181
|
+
if (n.poolName !== pool.name) return false;
|
|
182
|
+
if (n.tierName !== undefined) return n.tierName === tier.name;
|
|
183
|
+
return tier.scaleOut === "fixed";
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
/** @public */
|
|
187
|
+
export interface InstanceTypeOption {
|
|
188
|
+
instanceType: string;
|
|
189
|
+
totalCapacity: number;
|
|
190
|
+
totalCores: number;
|
|
191
|
+
capabilities: Capability[];
|
|
192
|
+
/** GPUs we could provision a node with — for feasibility check. */
|
|
193
|
+
gpus?: { totalCapacity: number; model?: string }[];
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
/** @public */
|
|
197
|
+
export interface RecoveryContext {
|
|
198
|
+
/** NodeIds the placement engine must skip (e.g., recently failed). */
|
|
199
|
+
excludeNodes: Set<NodeId>;
|
|
200
|
+
/** Reserved for Phase C — biases scoring toward hot spares. */
|
|
201
|
+
preferHotSpares: boolean;
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/** @public */
|
|
205
|
+
export interface PlacementInput {
|
|
206
|
+
bundle: Bundle;
|
|
207
|
+
job: BundleJobSpec;
|
|
208
|
+
replicaIndex: number;
|
|
209
|
+
inventory: NodeView[];
|
|
210
|
+
pools: Map<string, PlacementPool>;
|
|
211
|
+
recoveryContext?: RecoveryContext;
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
/** @public */
|
|
215
|
+
/** Which anti-affinity flag(s) a backup placement couldn't honour. */
|
|
216
|
+
export type ResilienceFlag = "sameNode" | "sameAz" | "sameCloud";
|
|
217
|
+
|
|
218
|
+
/**
|
|
219
|
+
* Set on a backup placement that landed inside a failure domain its
|
|
220
|
+
* resilience policy asked to avoid (best-effort: no compliant node had
|
|
221
|
+
* capacity). The engine places it anyway and surfaces this so the
|
|
222
|
+
* operator knows the bundle's DR posture is currently reduced.
|
|
223
|
+
*
|
|
224
|
+
* @public
|
|
225
|
+
*/
|
|
226
|
+
export interface ResilienceDegradation {
|
|
227
|
+
violated: ResilienceFlag[];
|
|
228
|
+
/**
|
|
229
|
+
* Set when the primary landed on interruptible (spot) capacity with no
|
|
230
|
+
* durable backup — a single interruption takes the workload down. Surfaced
|
|
231
|
+
* so the operator can add a durable backup.
|
|
232
|
+
*/
|
|
233
|
+
interruptiblePrimary?: boolean;
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
/** @public */
|
|
237
|
+
export type PlacementResult =
|
|
238
|
+
| {
|
|
239
|
+
kind: "place";
|
|
240
|
+
nodeId: NodeId;
|
|
241
|
+
gpuIndex?: number;
|
|
242
|
+
pool: string;
|
|
243
|
+
tier: string;
|
|
244
|
+
degraded?: ResilienceDegradation;
|
|
245
|
+
}
|
|
246
|
+
| {
|
|
247
|
+
kind: "provision";
|
|
248
|
+
pool: string;
|
|
249
|
+
tier: string;
|
|
250
|
+
instanceType: string;
|
|
251
|
+
launchMode: LaunchMode;
|
|
252
|
+
degraded?: ResilienceDegradation;
|
|
253
|
+
}
|
|
254
|
+
| { kind: "failure"; reason: PlacementFailureReason };
|
|
255
|
+
|
|
256
|
+
/** @public */
|
|
257
|
+
export interface PlacementFailureReason {
|
|
258
|
+
code: "noCapacityInAnyPool" | "noConfiguredPools";
|
|
259
|
+
triedPools: string[];
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
/**
|
|
263
|
+
* Structured trace of a placement decision — one entry per pool
|
|
264
|
+
* attempted, with per-node filter outcomes and resulting scores. Used
|
|
265
|
+
* for "why did my job land here?" debugging.
|
|
266
|
+
*
|
|
267
|
+
* @public
|
|
268
|
+
*/
|
|
269
|
+
export interface PlacementTrace {
|
|
270
|
+
attempts: PoolAttempt[];
|
|
271
|
+
/** Final decision kind, mirroring PlacementResult.kind. */
|
|
272
|
+
decision: "place" | "provision" | "failure";
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
/** @public */
|
|
276
|
+
export interface PoolAttempt {
|
|
277
|
+
pool: string;
|
|
278
|
+
/** Which tier of the pool this attempt covers. */
|
|
279
|
+
tier: string;
|
|
280
|
+
/** Total nodes considered in this pool before any filtering. */
|
|
281
|
+
initialCandidates: number;
|
|
282
|
+
/** Why each non-matching node was filtered out. Indexed by nodeId. */
|
|
283
|
+
filtered: { nodeId: NodeId; reason: FilterReason }[];
|
|
284
|
+
/** Nodes that survived all filters, with their assigned scores. */
|
|
285
|
+
scored: { nodeId: NodeId; score: number }[];
|
|
286
|
+
/** What this pool attempt produced. */
|
|
287
|
+
outcome:
|
|
288
|
+
| { kind: "placed"; nodeId: NodeId; gpuIndex?: number }
|
|
289
|
+
| { kind: "provisioned"; instanceType: string }
|
|
290
|
+
| { kind: "exhausted" }
|
|
291
|
+
| { kind: "no-instance-type" }
|
|
292
|
+
| { kind: "missing-pool" };
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
/** @public */
|
|
296
|
+
export type FilterReason =
|
|
297
|
+
| "wrong-pool"
|
|
298
|
+
| "wrong-tier"
|
|
299
|
+
| "excluded-recovery"
|
|
300
|
+
| "unreachable"
|
|
301
|
+
| "cordoned"
|
|
302
|
+
| "missing-capability"
|
|
303
|
+
| "insufficient-capacity"
|
|
304
|
+
| "insufficient-cores"
|
|
305
|
+
| "no-fitting-gpu"
|
|
306
|
+
| "fails-intra-replica"
|
|
307
|
+
| "fails-intra-az";
|
|
308
|
+
|
|
309
|
+
/**
|
|
310
|
+
* Core placement function. Pure — no side effects, no I/O.
|
|
311
|
+
*
|
|
312
|
+
* @public
|
|
313
|
+
*/
|
|
314
|
+
export function place(input: PlacementInput): PlacementResult {
|
|
315
|
+
return placeWithTrace(input).result;
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
/**
|
|
319
|
+
* Placement with a structured trace of every pool attempted and every
|
|
320
|
+
* node's filter outcome. AutoManager logs the trace via debuglog when
|
|
321
|
+
* placing; tests use it to assert on the precise reason a candidate
|
|
322
|
+
* was rejected.
|
|
323
|
+
*
|
|
324
|
+
* @public
|
|
325
|
+
*/
|
|
326
|
+
export function placeWithTrace(
|
|
327
|
+
input: PlacementInput
|
|
328
|
+
): { result: PlacementResult; trace: PlacementTrace } {
|
|
329
|
+
const { bundle, job } = input;
|
|
330
|
+
const tried: string[] = [];
|
|
331
|
+
const attempts: PoolAttempt[] = [];
|
|
332
|
+
|
|
333
|
+
const pool = input.pools.get(bundle.pool);
|
|
334
|
+
if (!pool || pool.tiers.length === 0) {
|
|
335
|
+
return {
|
|
336
|
+
result: { kind: "failure", reason: { code: "noConfiguredPools", triedPools: [] } },
|
|
337
|
+
trace: { attempts: [], decision: "failure" },
|
|
338
|
+
};
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
// Walk the tiers cheapest-first; first tier that can place or provision
|
|
342
|
+
// wins. This positional order is the cost cascade (no cost weight in v1).
|
|
343
|
+
for (const tier of pool.tiers) {
|
|
344
|
+
tried.push(tier.name);
|
|
345
|
+
|
|
346
|
+
const { matched, filtered } = filterCandidatesTraced(input, pool, tier);
|
|
347
|
+
const initialCandidates = input.inventory.length;
|
|
348
|
+
|
|
349
|
+
if (matched.length > 0) {
|
|
350
|
+
const scored = matched.map((n) => ({
|
|
351
|
+
node: n,
|
|
352
|
+
score: scoreNode(n, tier.packingStrategy, input),
|
|
353
|
+
}));
|
|
354
|
+
// Sort descending by score, lexicographic nodeId tiebreak.
|
|
355
|
+
scored.sort((a, b) => {
|
|
356
|
+
if (a.score !== b.score) return b.score - a.score;
|
|
357
|
+
return a.node.nodeId.localeCompare(b.node.nodeId);
|
|
358
|
+
});
|
|
359
|
+
const winner = scored[0].node;
|
|
360
|
+
const gpuIndex = pickGpuIndex(winner, job.requirements, tier.packingStrategy);
|
|
361
|
+
const degraded = placeDegradation(winner, input) ?? primaryInterruptibleDegradation(tier, input);
|
|
362
|
+
attempts.push({
|
|
363
|
+
pool: pool.name,
|
|
364
|
+
tier: tier.name,
|
|
365
|
+
initialCandidates,
|
|
366
|
+
filtered,
|
|
367
|
+
scored: scored.map((s) => ({ nodeId: s.node.nodeId, score: s.score })),
|
|
368
|
+
outcome: { kind: "placed", nodeId: winner.nodeId, gpuIndex },
|
|
369
|
+
});
|
|
370
|
+
return {
|
|
371
|
+
result: {
|
|
372
|
+
kind: "place",
|
|
373
|
+
nodeId: winner.nodeId,
|
|
374
|
+
gpuIndex,
|
|
375
|
+
pool: pool.name,
|
|
376
|
+
tier: tier.name,
|
|
377
|
+
...(degraded ? { degraded } : {}),
|
|
378
|
+
},
|
|
379
|
+
trace: { attempts, decision: "place" },
|
|
380
|
+
};
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
// A tier at its node cap can't grow — the cost cascade spills to the
|
|
384
|
+
// next tier. (Placing onto an existing node above doesn't add a node,
|
|
385
|
+
// so the cap only gates provisioning.)
|
|
386
|
+
const atCap = tier.maxNodes !== undefined && tierNodeCount(input.inventory, pool, tier) >= tier.maxNodes;
|
|
387
|
+
|
|
388
|
+
if (tier.scaleOut === "elastic" && !atCap) {
|
|
389
|
+
const instanceType = findProvisionableInstanceType(tier, job.requirements);
|
|
390
|
+
if (instanceType) {
|
|
391
|
+
const degraded = provisionDegradation(tier, input) ?? primaryInterruptibleDegradation(tier, input);
|
|
392
|
+
attempts.push({
|
|
393
|
+
pool: pool.name,
|
|
394
|
+
tier: tier.name,
|
|
395
|
+
initialCandidates,
|
|
396
|
+
filtered,
|
|
397
|
+
scored: [],
|
|
398
|
+
outcome: { kind: "provisioned", instanceType },
|
|
399
|
+
});
|
|
400
|
+
return {
|
|
401
|
+
result: {
|
|
402
|
+
kind: "provision",
|
|
403
|
+
pool: pool.name,
|
|
404
|
+
tier: tier.name,
|
|
405
|
+
instanceType,
|
|
406
|
+
launchMode: tierLaunchMode(tier),
|
|
407
|
+
...(degraded ? { degraded } : {}),
|
|
408
|
+
},
|
|
409
|
+
trace: { attempts, decision: "provision" },
|
|
410
|
+
};
|
|
411
|
+
}
|
|
412
|
+
attempts.push({
|
|
413
|
+
pool: pool.name,
|
|
414
|
+
tier: tier.name,
|
|
415
|
+
initialCandidates,
|
|
416
|
+
filtered,
|
|
417
|
+
scored: [],
|
|
418
|
+
outcome: { kind: "no-instance-type" },
|
|
419
|
+
});
|
|
420
|
+
continue;
|
|
421
|
+
}
|
|
422
|
+
attempts.push({
|
|
423
|
+
pool: pool.name,
|
|
424
|
+
tier: tier.name,
|
|
425
|
+
initialCandidates,
|
|
426
|
+
filtered,
|
|
427
|
+
scored: [],
|
|
428
|
+
outcome: { kind: "exhausted" },
|
|
429
|
+
});
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
return {
|
|
433
|
+
result: { kind: "failure", reason: { code: "noCapacityInAnyPool", triedPools: tried } },
|
|
434
|
+
trace: { attempts, decision: "failure" },
|
|
435
|
+
};
|
|
436
|
+
}
|
|
437
|
+
|
|
438
|
+
// ---------- filter ----------
|
|
439
|
+
|
|
440
|
+
function filterCandidatesTraced(
|
|
441
|
+
input: PlacementInput,
|
|
442
|
+
pool: PlacementPool,
|
|
443
|
+
tier: PlacementTier
|
|
444
|
+
): { matched: NodeView[]; filtered: { nodeId: NodeId; reason: FilterReason }[] } {
|
|
445
|
+
const { bundle, job, replicaIndex, inventory, recoveryContext } = input;
|
|
446
|
+
const matched: NodeView[] = [];
|
|
447
|
+
const filtered: { nodeId: NodeId; reason: FilterReason }[] = [];
|
|
448
|
+
|
|
449
|
+
for (const n of inventory) {
|
|
450
|
+
const reason = firstFilterMiss(n, input, pool, tier, recoveryContext);
|
|
451
|
+
if (reason === undefined) {
|
|
452
|
+
matched.push(n);
|
|
453
|
+
} else {
|
|
454
|
+
// Skip "wrong-pool"/"wrong-tier" entries from the trace by default —
|
|
455
|
+
// for a large cross-pool inventory the "this isn't in my pool/tier"
|
|
456
|
+
// entries dominate the trace and aren't useful. Keep all other reasons.
|
|
457
|
+
if (reason !== "wrong-pool" && reason !== "wrong-tier")
|
|
458
|
+
filtered.push({ nodeId: n.nodeId, reason });
|
|
459
|
+
}
|
|
460
|
+
}
|
|
461
|
+
// Suppressed unused-import lint when no consumer references these
|
|
462
|
+
// utilities directly any more.
|
|
463
|
+
void bundle;
|
|
464
|
+
void job;
|
|
465
|
+
void replicaIndex;
|
|
466
|
+
|
|
467
|
+
return { matched, filtered };
|
|
468
|
+
}
|
|
469
|
+
|
|
470
|
+
function firstFilterMiss(
|
|
471
|
+
n: NodeView,
|
|
472
|
+
input: PlacementInput,
|
|
473
|
+
pool: PlacementPool,
|
|
474
|
+
tier: PlacementTier,
|
|
475
|
+
recoveryContext: RecoveryContext | undefined
|
|
476
|
+
): FilterReason | undefined {
|
|
477
|
+
const { bundle, job, replicaIndex, inventory } = input;
|
|
478
|
+
if (n.poolName !== pool.name) return "wrong-pool";
|
|
479
|
+
if (!nodeMatchesTier(n, pool, tier)) return "wrong-tier";
|
|
480
|
+
if (recoveryContext?.excludeNodes.has(n.nodeId)) return "excluded-recovery";
|
|
481
|
+
if (!n.inventory.reachable) return "unreachable";
|
|
482
|
+
if (n.inventory.cordoned) return "cordoned";
|
|
483
|
+
if (!matchesAllCapabilities(n.inventory, job.requirements.requiredCapabilities))
|
|
484
|
+
return "missing-capability";
|
|
485
|
+
if (freeCapacity(n.inventory) < job.requirements.requiredCapacity)
|
|
486
|
+
return "insufficient-capacity";
|
|
487
|
+
if (freeCores(n.inventory) < (job.requirements.requiredCores ?? 0))
|
|
488
|
+
return "insufficient-cores";
|
|
489
|
+
if (!matchesGpu(n.inventory.gpus, job.requirements)) return "no-fitting-gpu";
|
|
490
|
+
if (!satisfiesIntraReplica(n, bundle, job, replicaIndex, inventory))
|
|
491
|
+
return "fails-intra-replica";
|
|
492
|
+
if (!satisfiesIntraReplicaAz(n, bundle, replicaIndex, inventory))
|
|
493
|
+
return "fails-intra-az";
|
|
494
|
+
// Resilience anti-affinity (backup vs primary) is NOT a hard filter — it
|
|
495
|
+
// is best-effort, applied as a scoring penalty so the backup still
|
|
496
|
+
// places when no compliant node has capacity (then flagged degraded).
|
|
497
|
+
return undefined;
|
|
498
|
+
}
|
|
499
|
+
|
|
500
|
+
function matchesAllCapabilities(
|
|
501
|
+
inv: NodeInventory,
|
|
502
|
+
reqs: CapabilityRequirement[]
|
|
503
|
+
): boolean {
|
|
504
|
+
for (const req of reqs) {
|
|
505
|
+
if (freeCapability(inv, req.name) < req.count) return false;
|
|
506
|
+
if (req.attributeMatches) {
|
|
507
|
+
const cap = inv.capabilities.find((c) => c.name === req.name);
|
|
508
|
+
if (!cap) return false;
|
|
509
|
+
for (const [k, v] of Object.entries(req.attributeMatches)) {
|
|
510
|
+
if (cap.attributes?.[k] !== v) return false;
|
|
511
|
+
}
|
|
512
|
+
}
|
|
513
|
+
}
|
|
514
|
+
return true;
|
|
515
|
+
}
|
|
516
|
+
|
|
517
|
+
function matchesGpu(gpus: GpuResource[], req: JobRequirements): boolean {
|
|
518
|
+
if (req.requiredGpuCapacity === undefined || req.requiredGpuCapacity <= 0) {
|
|
519
|
+
return true;
|
|
520
|
+
}
|
|
521
|
+
return gpus.some((g) => gpuFits(g, req));
|
|
522
|
+
}
|
|
523
|
+
|
|
524
|
+
function gpuFits(g: GpuResource, req: JobRequirements): boolean {
|
|
525
|
+
if (g.exclusivelyReserved) return false;
|
|
526
|
+
if (req.requiredGpuModel !== undefined && g.model !== req.requiredGpuModel)
|
|
527
|
+
return false;
|
|
528
|
+
const free = g.totalCapacity - g.reservedCapacity;
|
|
529
|
+
return free >= (req.requiredGpuCapacity ?? 0);
|
|
530
|
+
}
|
|
531
|
+
|
|
532
|
+
function satisfiesIntraReplica(
|
|
533
|
+
n: NodeView,
|
|
534
|
+
bundle: Bundle,
|
|
535
|
+
job: BundleJobSpec,
|
|
536
|
+
replicaIndex: number,
|
|
537
|
+
inventory: NodeView[]
|
|
538
|
+
): boolean {
|
|
539
|
+
// coLocateWith: every named job in this same replica must be running on this node.
|
|
540
|
+
if (job.coLocateWith && job.coLocateWith.length > 0) {
|
|
541
|
+
for (const peer of job.coLocateWith) {
|
|
542
|
+
const peerNode = findRunningJob(inventory, bundle.bundleId, replicaIndex, peer);
|
|
543
|
+
// If peer not yet placed, no constraint to check yet.
|
|
544
|
+
if (peerNode && peerNode.nodeId !== n.nodeId) return false;
|
|
545
|
+
}
|
|
546
|
+
}
|
|
547
|
+
// separateFrom: every named peer must be on a different node.
|
|
548
|
+
if (job.separateFrom && job.separateFrom.length > 0) {
|
|
549
|
+
for (const peer of job.separateFrom) {
|
|
550
|
+
const peerNode = findRunningJob(inventory, bundle.bundleId, replicaIndex, peer);
|
|
551
|
+
if (peerNode && peerNode.nodeId === n.nodeId) return false;
|
|
552
|
+
}
|
|
553
|
+
}
|
|
554
|
+
return true;
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
function satisfiesIntraReplicaAz(
|
|
558
|
+
n: NodeView,
|
|
559
|
+
bundle: Bundle,
|
|
560
|
+
replicaIndex: number,
|
|
561
|
+
inventory: NodeView[]
|
|
562
|
+
): boolean {
|
|
563
|
+
const sameAz = bundle.intraReplicaPlacement?.sameAz ?? "soft";
|
|
564
|
+
if (sameAz !== "hard") return true;
|
|
565
|
+
const peerAzs = peerAzsInSameReplica(inventory, bundle.bundleId, replicaIndex);
|
|
566
|
+
if (peerAzs.length === 0) return true;
|
|
567
|
+
// Hard sameAz with peers in known AZs: this node's AZ must match one of theirs.
|
|
568
|
+
return n.az !== undefined && peerAzs.includes(n.az);
|
|
569
|
+
}
|
|
570
|
+
|
|
571
|
+
function peerAzsInSameReplica(
|
|
572
|
+
inventory: NodeView[],
|
|
573
|
+
bundleId: BundleId,
|
|
574
|
+
replicaIndex: number
|
|
575
|
+
): string[] {
|
|
576
|
+
const azs = new Set<string>();
|
|
577
|
+
for (const n of inventory) {
|
|
578
|
+
if (
|
|
579
|
+
n.az !== undefined &&
|
|
580
|
+
n.runningJobs.some(
|
|
581
|
+
(rj) => rj.bundleId === bundleId && rj.replicaIndex === replicaIndex
|
|
582
|
+
)
|
|
583
|
+
) {
|
|
584
|
+
azs.add(n.az);
|
|
585
|
+
}
|
|
586
|
+
}
|
|
587
|
+
return [...azs];
|
|
588
|
+
}
|
|
589
|
+
|
|
590
|
+
// ---------- resilience anti-affinity (backup vs primary) ----------
|
|
591
|
+
|
|
592
|
+
/** True when this placement is a backup whose policy carries anti-affinity. */
|
|
593
|
+
function backupPolicy(input: PlacementInput): ResiliencePolicy | undefined {
|
|
594
|
+
return input.replicaIndex > 0 ? input.bundle.resiliencePolicy : undefined;
|
|
595
|
+
}
|
|
596
|
+
|
|
597
|
+
interface PrimaryDomains {
|
|
598
|
+
nodeIds: Set<NodeId>;
|
|
599
|
+
azs: Set<string>;
|
|
600
|
+
clouds: Set<string>;
|
|
601
|
+
}
|
|
602
|
+
|
|
603
|
+
/** Failure domains the primary (replica 0) currently occupies. */
|
|
604
|
+
function primaryDomains(inventory: NodeView[], bundleId: BundleId): PrimaryDomains {
|
|
605
|
+
const nodeIds = new Set<NodeId>();
|
|
606
|
+
const azs = new Set<string>();
|
|
607
|
+
const clouds = new Set<string>();
|
|
608
|
+
for (const n of inventory) {
|
|
609
|
+
if (n.runningJobs.some((rj) => rj.bundleId === bundleId && rj.replicaIndex === 0)) {
|
|
610
|
+
nodeIds.add(n.nodeId);
|
|
611
|
+
if (n.az !== undefined) azs.add(n.az);
|
|
612
|
+
if (n.cloud !== undefined) clouds.add(n.cloud);
|
|
613
|
+
}
|
|
614
|
+
}
|
|
615
|
+
return { nodeIds, azs, clouds };
|
|
616
|
+
}
|
|
617
|
+
|
|
618
|
+
/** Which `forbid` flags placing the backup on `n` would violate. */
|
|
619
|
+
function resilienceViolations(
|
|
620
|
+
n: NodeView,
|
|
621
|
+
policy: ResiliencePolicy,
|
|
622
|
+
primary: PrimaryDomains
|
|
623
|
+
): ResilienceFlag[] {
|
|
624
|
+
const violated: ResilienceFlag[] = [];
|
|
625
|
+
if (policy.sameNode === "forbid" && primary.nodeIds.has(n.nodeId)) violated.push("sameNode");
|
|
626
|
+
if (policy.sameAz === "forbid" && n.az !== undefined && primary.azs.has(n.az)) violated.push("sameAz");
|
|
627
|
+
if (policy.sameCloud === "forbid" && n.cloud !== undefined && primary.clouds.has(n.cloud))
|
|
628
|
+
violated.push("sameCloud");
|
|
629
|
+
return violated;
|
|
630
|
+
}
|
|
631
|
+
|
|
632
|
+
/** Degradation for placing a backup on an existing node `n`, if any. */
|
|
633
|
+
function placeDegradation(n: NodeView, input: PlacementInput): ResilienceDegradation | undefined {
|
|
634
|
+
const policy = backupPolicy(input);
|
|
635
|
+
if (!policy) return undefined;
|
|
636
|
+
const violated = resilienceViolations(n, policy, primaryDomains(input.inventory, input.bundle.bundleId));
|
|
637
|
+
return violated.length > 0 ? { violated } : undefined;
|
|
638
|
+
}
|
|
639
|
+
|
|
640
|
+
/**
|
|
641
|
+
* Degradation for provisioning a backup into `tier`. Only the cloud domain
|
|
642
|
+
* is knowable pre-boot (it is the tier's provider); node/AZ are decided
|
|
643
|
+
* once the instance starts, so they aren't evaluated here.
|
|
644
|
+
*/
|
|
645
|
+
function provisionDegradation(tier: PlacementTier, input: PlacementInput): ResilienceDegradation | undefined {
|
|
646
|
+
const policy = backupPolicy(input);
|
|
647
|
+
if (!policy || policy.sameCloud !== "forbid") return undefined;
|
|
648
|
+
const primary = primaryDomains(input.inventory, input.bundle.bundleId);
|
|
649
|
+
return primary.clouds.has(tier.kind) ? { violated: ["sameCloud"] } : undefined;
|
|
650
|
+
}
|
|
651
|
+
|
|
652
|
+
/** Degradation for placing the *primary* (replica 0) onto interruptible
|
|
653
|
+
* capacity with no durable backup configured — one interruption and the
|
|
654
|
+
* workload is gone. (Backups have their own anti-affinity degradation.) */
|
|
655
|
+
function primaryInterruptibleDegradation(
|
|
656
|
+
tier: PlacementTier,
|
|
657
|
+
input: PlacementInput
|
|
658
|
+
): ResilienceDegradation | undefined {
|
|
659
|
+
if (input.replicaIndex !== 0 || input.bundle.resiliencePolicy) return undefined;
|
|
660
|
+
if (tierReliability(tier) !== "interruptible") return undefined;
|
|
661
|
+
return { violated: [], interruptiblePrimary: true };
|
|
662
|
+
}
|
|
663
|
+
|
|
664
|
+
/** Count of nodes currently in (pool, tier) — used for the tier node cap. */
|
|
665
|
+
function tierNodeCount(inventory: NodeView[], pool: PlacementPool, tier: PlacementTier): number {
|
|
666
|
+
let n = 0;
|
|
667
|
+
for (const node of inventory) if (nodeMatchesTier(node, pool, tier)) n++;
|
|
668
|
+
return n;
|
|
669
|
+
}
|
|
670
|
+
|
|
671
|
+
function findRunningJob(
|
|
672
|
+
inventory: NodeView[],
|
|
673
|
+
bundleId: BundleId,
|
|
674
|
+
replicaIndex: number,
|
|
675
|
+
jobName: string
|
|
676
|
+
): NodeView | undefined {
|
|
677
|
+
return inventory.find((n) =>
|
|
678
|
+
n.runningJobs.some(
|
|
679
|
+
(rj) =>
|
|
680
|
+
rj.bundleId === bundleId &&
|
|
681
|
+
rj.replicaIndex === replicaIndex &&
|
|
682
|
+
rj.jobName === jobName
|
|
683
|
+
)
|
|
684
|
+
);
|
|
685
|
+
}
|
|
686
|
+
|
|
687
|
+
// ---------- score ----------
|
|
688
|
+
//
|
|
689
|
+
// Score = base (pack/spread on capacity) + soft-constraint bonuses.
|
|
690
|
+
// Bonuses are deliberately small relative to a fully-laden vs empty
|
|
691
|
+
// difference (which is 1.0 for binpack), but large enough to break
|
|
692
|
+
// ties between similar candidates. Hot-spare bonus dominates because
|
|
693
|
+
// recovery placements should aggressively prefer prewarmed slots.
|
|
694
|
+
|
|
695
|
+
const HOT_SPARE_BONUS = 1.0;
|
|
696
|
+
const INTRA_SAME_AZ_BONUS = 0.1;
|
|
697
|
+
// A resilience `forbid` violation is best-effort, not a hard reject: a
|
|
698
|
+
// large per-violation penalty so the backup lands outside the primary's
|
|
699
|
+
// failure domains whenever a node there has capacity, but still places
|
|
700
|
+
// (and is flagged degraded) when none does. Dwarfs the [-1,1] capacity
|
|
701
|
+
// score and the small AZ/hot-spare bonuses so a compliant node always wins.
|
|
702
|
+
const RESILIENCE_VIOLATION_PENALTY = 100;
|
|
703
|
+
|
|
704
|
+
function scoreNode(
|
|
705
|
+
n: NodeView,
|
|
706
|
+
strategy: "binpack" | "spread",
|
|
707
|
+
input: PlacementInput
|
|
708
|
+
): number {
|
|
709
|
+
const used = usedCapacityFraction(n.inventory);
|
|
710
|
+
let score = strategy === "binpack" ? used : -used;
|
|
711
|
+
score += softBonus(n, input);
|
|
712
|
+
return score;
|
|
713
|
+
}
|
|
714
|
+
|
|
715
|
+
function softBonus(n: NodeView, input: PlacementInput): number {
|
|
716
|
+
const { bundle, replicaIndex, inventory, recoveryContext } = input;
|
|
717
|
+
let bonus = 0;
|
|
718
|
+
|
|
719
|
+
if (recoveryContext?.preferHotSpares && n.isHotSpare) {
|
|
720
|
+
bonus += HOT_SPARE_BONUS;
|
|
721
|
+
}
|
|
722
|
+
|
|
723
|
+
// Soft sameAz (intra-replica): bonus if node's AZ matches a peer in the same replica.
|
|
724
|
+
const intraAz = bundle.intraReplicaPlacement?.sameAz ?? "soft";
|
|
725
|
+
if (intraAz === "soft" && n.az !== undefined) {
|
|
726
|
+
const peerAzs = peerAzsInSameReplica(inventory, bundle.bundleId, replicaIndex);
|
|
727
|
+
if (peerAzs.includes(n.az)) bonus += INTRA_SAME_AZ_BONUS;
|
|
728
|
+
}
|
|
729
|
+
|
|
730
|
+
// Resilience anti-affinity (backup vs primary): strong best-effort penalty
|
|
731
|
+
// for each `forbid` flag this node would violate, so the backup prefers a
|
|
732
|
+
// node outside the primary's failure domains.
|
|
733
|
+
const policy = backupPolicy(input);
|
|
734
|
+
if (policy) {
|
|
735
|
+
const primary = primaryDomains(inventory, bundle.bundleId);
|
|
736
|
+
bonus -= resilienceViolations(n, policy, primary).length * RESILIENCE_VIOLATION_PENALTY;
|
|
737
|
+
}
|
|
738
|
+
|
|
739
|
+
return bonus;
|
|
740
|
+
}
|
|
741
|
+
|
|
742
|
+
// ---------- GPU pick ----------
|
|
743
|
+
|
|
744
|
+
function pickGpuIndex(
|
|
745
|
+
n: NodeView,
|
|
746
|
+
req: JobRequirements,
|
|
747
|
+
strategy: "binpack" | "spread"
|
|
748
|
+
): number | undefined {
|
|
749
|
+
if (req.requiredGpuCapacity === undefined || req.requiredGpuCapacity <= 0) {
|
|
750
|
+
return undefined;
|
|
751
|
+
}
|
|
752
|
+
const eligible = n.inventory.gpus.filter((g) => gpuFits(g, req));
|
|
753
|
+
if (eligible.length === 0) return undefined;
|
|
754
|
+
// Pack: pick the GPU with least free capacity that still fits. Spread:
|
|
755
|
+
// pick the most free. Same packing intent as node-level scoring.
|
|
756
|
+
const sorted = eligible.slice().sort((a, b) => {
|
|
757
|
+
const freeA = a.totalCapacity - a.reservedCapacity;
|
|
758
|
+
const freeB = b.totalCapacity - b.reservedCapacity;
|
|
759
|
+
if (freeA !== freeB) {
|
|
760
|
+
return strategy === "binpack" ? freeA - freeB : freeB - freeA;
|
|
761
|
+
}
|
|
762
|
+
return a.index - b.index;
|
|
763
|
+
});
|
|
764
|
+
return sorted[0].index;
|
|
765
|
+
}
|
|
766
|
+
|
|
767
|
+
// ---------- elastic provision ----------
|
|
768
|
+
|
|
769
|
+
function findProvisionableInstanceType(
|
|
770
|
+
tier: PlacementTier,
|
|
771
|
+
req: JobRequirements
|
|
772
|
+
): string | undefined {
|
|
773
|
+
for (const it of tier.candidateInstanceTypes) {
|
|
774
|
+
if (instanceTypeSatisfies(it, req)) return it.instanceType;
|
|
775
|
+
}
|
|
776
|
+
return undefined;
|
|
777
|
+
}
|
|
778
|
+
|
|
779
|
+
function instanceTypeSatisfies(
|
|
780
|
+
it: InstanceTypeOption,
|
|
781
|
+
req: JobRequirements
|
|
782
|
+
): boolean {
|
|
783
|
+
if (it.totalCapacity < req.requiredCapacity) return false;
|
|
784
|
+
if (it.totalCores < (req.requiredCores ?? 0)) return false;
|
|
785
|
+
for (const cap of req.requiredCapabilities) {
|
|
786
|
+
const found = it.capabilities.find((c) => c.name === cap.name);
|
|
787
|
+
if (!found || found.count < cap.count) return false;
|
|
788
|
+
if (cap.attributeMatches) {
|
|
789
|
+
for (const [k, v] of Object.entries(cap.attributeMatches)) {
|
|
790
|
+
if (found.attributes?.[k] !== v) return false;
|
|
791
|
+
}
|
|
792
|
+
}
|
|
793
|
+
}
|
|
794
|
+
if (req.requiredGpuCapacity !== undefined && req.requiredGpuCapacity > 0) {
|
|
795
|
+
const gpus = it.gpus ?? [];
|
|
796
|
+
const fits = gpus.some(
|
|
797
|
+
(g) =>
|
|
798
|
+
(req.requiredGpuModel === undefined || g.model === req.requiredGpuModel) &&
|
|
799
|
+
g.totalCapacity >= (req.requiredGpuCapacity ?? 0)
|
|
800
|
+
);
|
|
801
|
+
if (!fits) return false;
|
|
802
|
+
}
|
|
803
|
+
return true;
|
|
804
|
+
}
|