@norskvideo/norsk-auto-manager 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,804 @@
1
+ // Pure-function placement engine. Given a job-to-place, the bundle it
2
+ // belongs to, the current inventory snapshot, and the pool configuration,
3
+ // returns a placement decision: place on a specific node, provision a new
4
+ // node in an elastic pool, or fail.
5
+ //
6
+ // Per §6 of the design doc: filter by hard constraints, score by packing
7
+ // strategy, tiebreak by nodeId for determinism. Per-replica pool
8
+ // preferences (replicaOverrides) take priority over the bundle default.
9
+ //
10
+ // Soft constraints (same-AZ preference, hot-spare scoring bonus) are
11
+ // not yet implemented in this slice — they'll layer on top of the score
12
+ // function once the hard-constraint path is settled and tested.
13
+
14
+ import {
15
+ Bundle,
16
+ BundleId,
17
+ BundleJobSpec,
18
+ Capability,
19
+ CapabilityRequirement,
20
+ GpuResource,
21
+ JobRequirements,
22
+ NodeId,
23
+ NodeInventory,
24
+ ResiliencePolicy,
25
+ } from "./types";
26
+ import {
27
+ freeCapability,
28
+ freeCapacity,
29
+ freeCores,
30
+ usedCapacityFraction,
31
+ } from "./inventory";
32
+
33
+ /** @public */
34
+ export interface NodeView {
35
+ nodeId: NodeId;
36
+ poolName: string;
37
+ /**
38
+ * Which tier within the pool this node belongs to. Set from
39
+ * `tags["tier"]` at provision time for elastic tiers. Pre-registered
40
+ * cluster nodes carry no tier tag (undefined) and are treated as
41
+ * belonging to their pool's `fixed` tier — see `nodeMatchesTier`.
42
+ */
43
+ tierName?: string;
44
+ inventory: NodeInventory;
45
+ /** Jobs currently running on this node (or pending placement onto it). */
46
+ runningJobs: RunningJob[];
47
+ /** Topology key for the resilience sameAz rule. Cluster pools can leave undefined. */
48
+ az?: string;
49
+ /**
50
+ * Cloud / failure-domain key for the resilience sameCloud rule — the
51
+ * node's provider (e.g. "aws" / "oci" / "cluster"). Undefined when
52
+ * unknown; sameCloud then can't be evaluated against this node.
53
+ */
54
+ cloud?: string;
55
+ /** Hot spare flag (Phase C). Currently unused by hard-constraint filters. */
56
+ isHotSpare: boolean;
57
+ }
58
+
59
+ /** @public */
60
+ export interface RunningJob {
61
+ bundleId: BundleId;
62
+ replicaIndex: number;
63
+ jobName: string;
64
+ }
65
+
66
+ /**
67
+ * A pool is an ordered list of capacity tiers. Placement walks the tiers
68
+ * in order (cheapest first) and takes the first that can satisfy the job
69
+ * — the tier order *is* the cost cascade (no cost weight in v1). The
70
+ * homogeneous single-cloud pool of before is just a one-tier pool.
71
+ *
72
+ * @public
73
+ */
74
+ export interface PlacementPool {
75
+ name: string;
76
+ /** Ordered cheapest → most expensive. Placement tries them in order. */
77
+ tiers: PlacementTier[];
78
+ }
79
+
80
+ /**
81
+ * One homogeneous capacity segment within a pool: a single cloud/cluster
82
+ * kind + region with its own scale-out behaviour and candidate shapes.
83
+ *
84
+ * @public
85
+ */
86
+ export interface PlacementTier {
87
+ /** Unique within the pool. Tagged onto provisioned nodes as `tags["tier"]`. */
88
+ name: string;
89
+ /**
90
+ * Cloud / local-servers kind. Used by AutoManager to decide which gRPC
91
+ * call to make when provisioning (createAwsNode vs createOciNode vs
92
+ * local-servers-tier placement, which uses startJob against an existing
93
+ * registered server).
94
+ */
95
+ kind: "aws" | "oci" | "local-servers";
96
+ /**
97
+ * For aws/oci elastic tiers: which region/availability-domain to
98
+ * provision into. Cluster tiers leave undefined.
99
+ */
100
+ region?: string;
101
+ packingStrategy: "binpack" | "spread";
102
+ scaleOut: "elastic" | "fixed";
103
+ /**
104
+ * For elastic tiers: shapes that can be provisioned on demand. For
105
+ * fixed tiers: typically empty (the tier consists of registered
106
+ * cluster nodes already in `inventory`).
107
+ */
108
+ candidateInstanceTypes: InstanceTypeOption[];
109
+ /**
110
+ * How long ahead of a scheduled job's `startDateTime` AutoManager
111
+ * should kick off placement, so the worker is ready to run by the
112
+ * requested time. Cluster tiers place against an existing running
113
+ * node (~instant) so the default is 0. Cloud tiers have to boot a
114
+ * fresh instance — defaults to 5 minutes; operators set per-tier
115
+ * when their boot time differs. AutoManager picks the maximum across
116
+ * the bundle pool's tiers (worst case — guarantees readiness
117
+ * regardless of which tier the placement engine settles on).
118
+ */
119
+ placementLeadMs?: number;
120
+ /**
121
+ * How nodes in this tier are purchased. `"spot"` requests interruptible
122
+ * spot-market capacity; `"reserved"` is an on-demand launch the operator
123
+ * knows is covered by a standing reservation (same AWS call as on-demand,
124
+ * but capped — see `maxNodes`). Defaults to `"on-demand"`.
125
+ */
126
+ launchMode?: LaunchMode;
127
+ /**
128
+ * Failure-domain reliability of this tier's capacity. Defaults to
129
+ * `"interruptible"` for spot, `"durable"` otherwise. Read by the
130
+ * resilience engine — an interruptible-tier primary with no durable
131
+ * backup is flagged degraded.
132
+ */
133
+ reliability?: Reliability;
134
+ /**
135
+ * Cap on the number of nodes this tier may hold (tier-wide, across
136
+ * bundles). The cost cascade spills to the next tier once the cap is
137
+ * reached. Used mainly for reserved tiers — exceeding the reservation
138
+ * count just bills as on-demand, defeating the point. Unset ⇒ no cap.
139
+ */
140
+ maxNodes?: number;
141
+ }
142
+
143
+ /** @public */
144
+ export type LaunchMode = "on-demand" | "spot" | "reserved";
145
+
146
+ /** @public */
147
+ export type Reliability = "durable" | "interruptible";
148
+
149
+ /** The tier's launch mode, defaulting to on-demand. @public */
150
+ export function tierLaunchMode(tier: PlacementTier): LaunchMode {
151
+ return tier.launchMode ?? "on-demand";
152
+ }
153
+
154
+ /** The tier's reliability — explicit, else interruptible for spot. @public */
155
+ export function tierReliability(tier: PlacementTier): Reliability {
156
+ return tier.reliability ?? (tierLaunchMode(tier) === "spot" ? "interruptible" : "durable");
157
+ }
158
+
159
+ /** Default placement lead time by tier kind. Cluster tiers place
160
+ * against existing running nodes so we don't need lead. Cloud tiers
161
+ * default to 5 minutes — typical EC2/OCI instance boot + image pull
162
+ * budget; operators can override per-tier via PlacementTier. */
163
+ export function defaultPlacementLeadMs(kind: PlacementTier["kind"]): number {
164
+ return kind === "local-servers" ? 0 : 5 * 60 * 1000;
165
+ }
166
+
167
+ /**
168
+ * Whether a node belongs to (pool, tier). Elastic nodes carry an explicit
169
+ * `tierName` (set from `tags["tier"]` at provision). Pre-registered cluster
170
+ * nodes carry no tier tag and are taken to belong to the pool's `fixed`
171
+ * tier — so a pool must have at most one fixed tier (enforced in
172
+ * settingsValidation) for this to be unambiguous.
173
+ *
174
+ * @public
175
+ */
176
+ export function nodeMatchesTier(
177
+ n: NodeView,
178
+ pool: PlacementPool,
179
+ tier: PlacementTier
180
+ ): boolean {
181
+ if (n.poolName !== pool.name) return false;
182
+ if (n.tierName !== undefined) return n.tierName === tier.name;
183
+ return tier.scaleOut === "fixed";
184
+ }
185
+
186
+ /** @public */
187
+ export interface InstanceTypeOption {
188
+ instanceType: string;
189
+ totalCapacity: number;
190
+ totalCores: number;
191
+ capabilities: Capability[];
192
+ /** GPUs we could provision a node with — for feasibility check. */
193
+ gpus?: { totalCapacity: number; model?: string }[];
194
+ }
195
+
196
+ /** @public */
197
+ export interface RecoveryContext {
198
+ /** NodeIds the placement engine must skip (e.g., recently failed). */
199
+ excludeNodes: Set<NodeId>;
200
+ /** Reserved for Phase C — biases scoring toward hot spares. */
201
+ preferHotSpares: boolean;
202
+ }
203
+
204
+ /** @public */
205
+ export interface PlacementInput {
206
+ bundle: Bundle;
207
+ job: BundleJobSpec;
208
+ replicaIndex: number;
209
+ inventory: NodeView[];
210
+ pools: Map<string, PlacementPool>;
211
+ recoveryContext?: RecoveryContext;
212
+ }
213
+
214
+ /** @public */
215
+ /** Which anti-affinity flag(s) a backup placement couldn't honour. */
216
+ export type ResilienceFlag = "sameNode" | "sameAz" | "sameCloud";
217
+
218
+ /**
219
+ * Set on a backup placement that landed inside a failure domain its
220
+ * resilience policy asked to avoid (best-effort: no compliant node had
221
+ * capacity). The engine places it anyway and surfaces this so the
222
+ * operator knows the bundle's DR posture is currently reduced.
223
+ *
224
+ * @public
225
+ */
226
+ export interface ResilienceDegradation {
227
+ violated: ResilienceFlag[];
228
+ /**
229
+ * Set when the primary landed on interruptible (spot) capacity with no
230
+ * durable backup — a single interruption takes the workload down. Surfaced
231
+ * so the operator can add a durable backup.
232
+ */
233
+ interruptiblePrimary?: boolean;
234
+ }
235
+
236
+ /** @public */
237
+ export type PlacementResult =
238
+ | {
239
+ kind: "place";
240
+ nodeId: NodeId;
241
+ gpuIndex?: number;
242
+ pool: string;
243
+ tier: string;
244
+ degraded?: ResilienceDegradation;
245
+ }
246
+ | {
247
+ kind: "provision";
248
+ pool: string;
249
+ tier: string;
250
+ instanceType: string;
251
+ launchMode: LaunchMode;
252
+ degraded?: ResilienceDegradation;
253
+ }
254
+ | { kind: "failure"; reason: PlacementFailureReason };
255
+
256
+ /** @public */
257
+ export interface PlacementFailureReason {
258
+ code: "noCapacityInAnyPool" | "noConfiguredPools";
259
+ triedPools: string[];
260
+ }
261
+
262
+ /**
263
+ * Structured trace of a placement decision — one entry per pool
264
+ * attempted, with per-node filter outcomes and resulting scores. Used
265
+ * for "why did my job land here?" debugging.
266
+ *
267
+ * @public
268
+ */
269
+ export interface PlacementTrace {
270
+ attempts: PoolAttempt[];
271
+ /** Final decision kind, mirroring PlacementResult.kind. */
272
+ decision: "place" | "provision" | "failure";
273
+ }
274
+
275
+ /** @public */
276
+ export interface PoolAttempt {
277
+ pool: string;
278
+ /** Which tier of the pool this attempt covers. */
279
+ tier: string;
280
+ /** Total nodes considered in this pool before any filtering. */
281
+ initialCandidates: number;
282
+ /** Why each non-matching node was filtered out. Indexed by nodeId. */
283
+ filtered: { nodeId: NodeId; reason: FilterReason }[];
284
+ /** Nodes that survived all filters, with their assigned scores. */
285
+ scored: { nodeId: NodeId; score: number }[];
286
+ /** What this pool attempt produced. */
287
+ outcome:
288
+ | { kind: "placed"; nodeId: NodeId; gpuIndex?: number }
289
+ | { kind: "provisioned"; instanceType: string }
290
+ | { kind: "exhausted" }
291
+ | { kind: "no-instance-type" }
292
+ | { kind: "missing-pool" };
293
+ }
294
+
295
+ /** @public */
296
+ export type FilterReason =
297
+ | "wrong-pool"
298
+ | "wrong-tier"
299
+ | "excluded-recovery"
300
+ | "unreachable"
301
+ | "cordoned"
302
+ | "missing-capability"
303
+ | "insufficient-capacity"
304
+ | "insufficient-cores"
305
+ | "no-fitting-gpu"
306
+ | "fails-intra-replica"
307
+ | "fails-intra-az";
308
+
309
+ /**
310
+ * Core placement function. Pure — no side effects, no I/O.
311
+ *
312
+ * @public
313
+ */
314
+ export function place(input: PlacementInput): PlacementResult {
315
+ return placeWithTrace(input).result;
316
+ }
317
+
318
+ /**
319
+ * Placement with a structured trace of every pool attempted and every
320
+ * node's filter outcome. AutoManager logs the trace via debuglog when
321
+ * placing; tests use it to assert on the precise reason a candidate
322
+ * was rejected.
323
+ *
324
+ * @public
325
+ */
326
+ export function placeWithTrace(
327
+ input: PlacementInput
328
+ ): { result: PlacementResult; trace: PlacementTrace } {
329
+ const { bundle, job } = input;
330
+ const tried: string[] = [];
331
+ const attempts: PoolAttempt[] = [];
332
+
333
+ const pool = input.pools.get(bundle.pool);
334
+ if (!pool || pool.tiers.length === 0) {
335
+ return {
336
+ result: { kind: "failure", reason: { code: "noConfiguredPools", triedPools: [] } },
337
+ trace: { attempts: [], decision: "failure" },
338
+ };
339
+ }
340
+
341
+ // Walk the tiers cheapest-first; first tier that can place or provision
342
+ // wins. This positional order is the cost cascade (no cost weight in v1).
343
+ for (const tier of pool.tiers) {
344
+ tried.push(tier.name);
345
+
346
+ const { matched, filtered } = filterCandidatesTraced(input, pool, tier);
347
+ const initialCandidates = input.inventory.length;
348
+
349
+ if (matched.length > 0) {
350
+ const scored = matched.map((n) => ({
351
+ node: n,
352
+ score: scoreNode(n, tier.packingStrategy, input),
353
+ }));
354
+ // Sort descending by score, lexicographic nodeId tiebreak.
355
+ scored.sort((a, b) => {
356
+ if (a.score !== b.score) return b.score - a.score;
357
+ return a.node.nodeId.localeCompare(b.node.nodeId);
358
+ });
359
+ const winner = scored[0].node;
360
+ const gpuIndex = pickGpuIndex(winner, job.requirements, tier.packingStrategy);
361
+ const degraded = placeDegradation(winner, input) ?? primaryInterruptibleDegradation(tier, input);
362
+ attempts.push({
363
+ pool: pool.name,
364
+ tier: tier.name,
365
+ initialCandidates,
366
+ filtered,
367
+ scored: scored.map((s) => ({ nodeId: s.node.nodeId, score: s.score })),
368
+ outcome: { kind: "placed", nodeId: winner.nodeId, gpuIndex },
369
+ });
370
+ return {
371
+ result: {
372
+ kind: "place",
373
+ nodeId: winner.nodeId,
374
+ gpuIndex,
375
+ pool: pool.name,
376
+ tier: tier.name,
377
+ ...(degraded ? { degraded } : {}),
378
+ },
379
+ trace: { attempts, decision: "place" },
380
+ };
381
+ }
382
+
383
+ // A tier at its node cap can't grow — the cost cascade spills to the
384
+ // next tier. (Placing onto an existing node above doesn't add a node,
385
+ // so the cap only gates provisioning.)
386
+ const atCap = tier.maxNodes !== undefined && tierNodeCount(input.inventory, pool, tier) >= tier.maxNodes;
387
+
388
+ if (tier.scaleOut === "elastic" && !atCap) {
389
+ const instanceType = findProvisionableInstanceType(tier, job.requirements);
390
+ if (instanceType) {
391
+ const degraded = provisionDegradation(tier, input) ?? primaryInterruptibleDegradation(tier, input);
392
+ attempts.push({
393
+ pool: pool.name,
394
+ tier: tier.name,
395
+ initialCandidates,
396
+ filtered,
397
+ scored: [],
398
+ outcome: { kind: "provisioned", instanceType },
399
+ });
400
+ return {
401
+ result: {
402
+ kind: "provision",
403
+ pool: pool.name,
404
+ tier: tier.name,
405
+ instanceType,
406
+ launchMode: tierLaunchMode(tier),
407
+ ...(degraded ? { degraded } : {}),
408
+ },
409
+ trace: { attempts, decision: "provision" },
410
+ };
411
+ }
412
+ attempts.push({
413
+ pool: pool.name,
414
+ tier: tier.name,
415
+ initialCandidates,
416
+ filtered,
417
+ scored: [],
418
+ outcome: { kind: "no-instance-type" },
419
+ });
420
+ continue;
421
+ }
422
+ attempts.push({
423
+ pool: pool.name,
424
+ tier: tier.name,
425
+ initialCandidates,
426
+ filtered,
427
+ scored: [],
428
+ outcome: { kind: "exhausted" },
429
+ });
430
+ }
431
+
432
+ return {
433
+ result: { kind: "failure", reason: { code: "noCapacityInAnyPool", triedPools: tried } },
434
+ trace: { attempts, decision: "failure" },
435
+ };
436
+ }
437
+
438
+ // ---------- filter ----------
439
+
440
+ function filterCandidatesTraced(
441
+ input: PlacementInput,
442
+ pool: PlacementPool,
443
+ tier: PlacementTier
444
+ ): { matched: NodeView[]; filtered: { nodeId: NodeId; reason: FilterReason }[] } {
445
+ const { bundle, job, replicaIndex, inventory, recoveryContext } = input;
446
+ const matched: NodeView[] = [];
447
+ const filtered: { nodeId: NodeId; reason: FilterReason }[] = [];
448
+
449
+ for (const n of inventory) {
450
+ const reason = firstFilterMiss(n, input, pool, tier, recoveryContext);
451
+ if (reason === undefined) {
452
+ matched.push(n);
453
+ } else {
454
+ // Skip "wrong-pool"/"wrong-tier" entries from the trace by default —
455
+ // for a large cross-pool inventory the "this isn't in my pool/tier"
456
+ // entries dominate the trace and aren't useful. Keep all other reasons.
457
+ if (reason !== "wrong-pool" && reason !== "wrong-tier")
458
+ filtered.push({ nodeId: n.nodeId, reason });
459
+ }
460
+ }
461
+ // Suppressed unused-import lint when no consumer references these
462
+ // utilities directly any more.
463
+ void bundle;
464
+ void job;
465
+ void replicaIndex;
466
+
467
+ return { matched, filtered };
468
+ }
469
+
470
+ function firstFilterMiss(
471
+ n: NodeView,
472
+ input: PlacementInput,
473
+ pool: PlacementPool,
474
+ tier: PlacementTier,
475
+ recoveryContext: RecoveryContext | undefined
476
+ ): FilterReason | undefined {
477
+ const { bundle, job, replicaIndex, inventory } = input;
478
+ if (n.poolName !== pool.name) return "wrong-pool";
479
+ if (!nodeMatchesTier(n, pool, tier)) return "wrong-tier";
480
+ if (recoveryContext?.excludeNodes.has(n.nodeId)) return "excluded-recovery";
481
+ if (!n.inventory.reachable) return "unreachable";
482
+ if (n.inventory.cordoned) return "cordoned";
483
+ if (!matchesAllCapabilities(n.inventory, job.requirements.requiredCapabilities))
484
+ return "missing-capability";
485
+ if (freeCapacity(n.inventory) < job.requirements.requiredCapacity)
486
+ return "insufficient-capacity";
487
+ if (freeCores(n.inventory) < (job.requirements.requiredCores ?? 0))
488
+ return "insufficient-cores";
489
+ if (!matchesGpu(n.inventory.gpus, job.requirements)) return "no-fitting-gpu";
490
+ if (!satisfiesIntraReplica(n, bundle, job, replicaIndex, inventory))
491
+ return "fails-intra-replica";
492
+ if (!satisfiesIntraReplicaAz(n, bundle, replicaIndex, inventory))
493
+ return "fails-intra-az";
494
+ // Resilience anti-affinity (backup vs primary) is NOT a hard filter — it
495
+ // is best-effort, applied as a scoring penalty so the backup still
496
+ // places when no compliant node has capacity (then flagged degraded).
497
+ return undefined;
498
+ }
499
+
500
+ function matchesAllCapabilities(
501
+ inv: NodeInventory,
502
+ reqs: CapabilityRequirement[]
503
+ ): boolean {
504
+ for (const req of reqs) {
505
+ if (freeCapability(inv, req.name) < req.count) return false;
506
+ if (req.attributeMatches) {
507
+ const cap = inv.capabilities.find((c) => c.name === req.name);
508
+ if (!cap) return false;
509
+ for (const [k, v] of Object.entries(req.attributeMatches)) {
510
+ if (cap.attributes?.[k] !== v) return false;
511
+ }
512
+ }
513
+ }
514
+ return true;
515
+ }
516
+
517
+ function matchesGpu(gpus: GpuResource[], req: JobRequirements): boolean {
518
+ if (req.requiredGpuCapacity === undefined || req.requiredGpuCapacity <= 0) {
519
+ return true;
520
+ }
521
+ return gpus.some((g) => gpuFits(g, req));
522
+ }
523
+
524
+ function gpuFits(g: GpuResource, req: JobRequirements): boolean {
525
+ if (g.exclusivelyReserved) return false;
526
+ if (req.requiredGpuModel !== undefined && g.model !== req.requiredGpuModel)
527
+ return false;
528
+ const free = g.totalCapacity - g.reservedCapacity;
529
+ return free >= (req.requiredGpuCapacity ?? 0);
530
+ }
531
+
532
+ function satisfiesIntraReplica(
533
+ n: NodeView,
534
+ bundle: Bundle,
535
+ job: BundleJobSpec,
536
+ replicaIndex: number,
537
+ inventory: NodeView[]
538
+ ): boolean {
539
+ // coLocateWith: every named job in this same replica must be running on this node.
540
+ if (job.coLocateWith && job.coLocateWith.length > 0) {
541
+ for (const peer of job.coLocateWith) {
542
+ const peerNode = findRunningJob(inventory, bundle.bundleId, replicaIndex, peer);
543
+ // If peer not yet placed, no constraint to check yet.
544
+ if (peerNode && peerNode.nodeId !== n.nodeId) return false;
545
+ }
546
+ }
547
+ // separateFrom: every named peer must be on a different node.
548
+ if (job.separateFrom && job.separateFrom.length > 0) {
549
+ for (const peer of job.separateFrom) {
550
+ const peerNode = findRunningJob(inventory, bundle.bundleId, replicaIndex, peer);
551
+ if (peerNode && peerNode.nodeId === n.nodeId) return false;
552
+ }
553
+ }
554
+ return true;
555
+ }
556
+
557
+ function satisfiesIntraReplicaAz(
558
+ n: NodeView,
559
+ bundle: Bundle,
560
+ replicaIndex: number,
561
+ inventory: NodeView[]
562
+ ): boolean {
563
+ const sameAz = bundle.intraReplicaPlacement?.sameAz ?? "soft";
564
+ if (sameAz !== "hard") return true;
565
+ const peerAzs = peerAzsInSameReplica(inventory, bundle.bundleId, replicaIndex);
566
+ if (peerAzs.length === 0) return true;
567
+ // Hard sameAz with peers in known AZs: this node's AZ must match one of theirs.
568
+ return n.az !== undefined && peerAzs.includes(n.az);
569
+ }
570
+
571
+ function peerAzsInSameReplica(
572
+ inventory: NodeView[],
573
+ bundleId: BundleId,
574
+ replicaIndex: number
575
+ ): string[] {
576
+ const azs = new Set<string>();
577
+ for (const n of inventory) {
578
+ if (
579
+ n.az !== undefined &&
580
+ n.runningJobs.some(
581
+ (rj) => rj.bundleId === bundleId && rj.replicaIndex === replicaIndex
582
+ )
583
+ ) {
584
+ azs.add(n.az);
585
+ }
586
+ }
587
+ return [...azs];
588
+ }
589
+
590
+ // ---------- resilience anti-affinity (backup vs primary) ----------
591
+
592
+ /** True when this placement is a backup whose policy carries anti-affinity. */
593
+ function backupPolicy(input: PlacementInput): ResiliencePolicy | undefined {
594
+ return input.replicaIndex > 0 ? input.bundle.resiliencePolicy : undefined;
595
+ }
596
+
597
+ interface PrimaryDomains {
598
+ nodeIds: Set<NodeId>;
599
+ azs: Set<string>;
600
+ clouds: Set<string>;
601
+ }
602
+
603
+ /** Failure domains the primary (replica 0) currently occupies. */
604
+ function primaryDomains(inventory: NodeView[], bundleId: BundleId): PrimaryDomains {
605
+ const nodeIds = new Set<NodeId>();
606
+ const azs = new Set<string>();
607
+ const clouds = new Set<string>();
608
+ for (const n of inventory) {
609
+ if (n.runningJobs.some((rj) => rj.bundleId === bundleId && rj.replicaIndex === 0)) {
610
+ nodeIds.add(n.nodeId);
611
+ if (n.az !== undefined) azs.add(n.az);
612
+ if (n.cloud !== undefined) clouds.add(n.cloud);
613
+ }
614
+ }
615
+ return { nodeIds, azs, clouds };
616
+ }
617
+
618
+ /** Which `forbid` flags placing the backup on `n` would violate. */
619
+ function resilienceViolations(
620
+ n: NodeView,
621
+ policy: ResiliencePolicy,
622
+ primary: PrimaryDomains
623
+ ): ResilienceFlag[] {
624
+ const violated: ResilienceFlag[] = [];
625
+ if (policy.sameNode === "forbid" && primary.nodeIds.has(n.nodeId)) violated.push("sameNode");
626
+ if (policy.sameAz === "forbid" && n.az !== undefined && primary.azs.has(n.az)) violated.push("sameAz");
627
+ if (policy.sameCloud === "forbid" && n.cloud !== undefined && primary.clouds.has(n.cloud))
628
+ violated.push("sameCloud");
629
+ return violated;
630
+ }
631
+
632
+ /** Degradation for placing a backup on an existing node `n`, if any. */
633
+ function placeDegradation(n: NodeView, input: PlacementInput): ResilienceDegradation | undefined {
634
+ const policy = backupPolicy(input);
635
+ if (!policy) return undefined;
636
+ const violated = resilienceViolations(n, policy, primaryDomains(input.inventory, input.bundle.bundleId));
637
+ return violated.length > 0 ? { violated } : undefined;
638
+ }
639
+
640
+ /**
641
+ * Degradation for provisioning a backup into `tier`. Only the cloud domain
642
+ * is knowable pre-boot (it is the tier's provider); node/AZ are decided
643
+ * once the instance starts, so they aren't evaluated here.
644
+ */
645
+ function provisionDegradation(tier: PlacementTier, input: PlacementInput): ResilienceDegradation | undefined {
646
+ const policy = backupPolicy(input);
647
+ if (!policy || policy.sameCloud !== "forbid") return undefined;
648
+ const primary = primaryDomains(input.inventory, input.bundle.bundleId);
649
+ return primary.clouds.has(tier.kind) ? { violated: ["sameCloud"] } : undefined;
650
+ }
651
+
652
+ /** Degradation for placing the *primary* (replica 0) onto interruptible
653
+ * capacity with no durable backup configured — one interruption and the
654
+ * workload is gone. (Backups have their own anti-affinity degradation.) */
655
+ function primaryInterruptibleDegradation(
656
+ tier: PlacementTier,
657
+ input: PlacementInput
658
+ ): ResilienceDegradation | undefined {
659
+ if (input.replicaIndex !== 0 || input.bundle.resiliencePolicy) return undefined;
660
+ if (tierReliability(tier) !== "interruptible") return undefined;
661
+ return { violated: [], interruptiblePrimary: true };
662
+ }
663
+
664
+ /** Count of nodes currently in (pool, tier) — used for the tier node cap. */
665
+ function tierNodeCount(inventory: NodeView[], pool: PlacementPool, tier: PlacementTier): number {
666
+ let n = 0;
667
+ for (const node of inventory) if (nodeMatchesTier(node, pool, tier)) n++;
668
+ return n;
669
+ }
670
+
671
+ function findRunningJob(
672
+ inventory: NodeView[],
673
+ bundleId: BundleId,
674
+ replicaIndex: number,
675
+ jobName: string
676
+ ): NodeView | undefined {
677
+ return inventory.find((n) =>
678
+ n.runningJobs.some(
679
+ (rj) =>
680
+ rj.bundleId === bundleId &&
681
+ rj.replicaIndex === replicaIndex &&
682
+ rj.jobName === jobName
683
+ )
684
+ );
685
+ }
686
+
687
+ // ---------- score ----------
688
+ //
689
+ // Score = base (pack/spread on capacity) + soft-constraint bonuses.
690
+ // Bonuses are deliberately small relative to a fully-laden vs empty
691
+ // difference (which is 1.0 for binpack), but large enough to break
692
+ // ties between similar candidates. Hot-spare bonus dominates because
693
+ // recovery placements should aggressively prefer prewarmed slots.
694
+
695
+ const HOT_SPARE_BONUS = 1.0;
696
+ const INTRA_SAME_AZ_BONUS = 0.1;
697
+ // A resilience `forbid` violation is best-effort, not a hard reject: a
698
+ // large per-violation penalty so the backup lands outside the primary's
699
+ // failure domains whenever a node there has capacity, but still places
700
+ // (and is flagged degraded) when none does. Dwarfs the [-1,1] capacity
701
+ // score and the small AZ/hot-spare bonuses so a compliant node always wins.
702
+ const RESILIENCE_VIOLATION_PENALTY = 100;
703
+
704
+ function scoreNode(
705
+ n: NodeView,
706
+ strategy: "binpack" | "spread",
707
+ input: PlacementInput
708
+ ): number {
709
+ const used = usedCapacityFraction(n.inventory);
710
+ let score = strategy === "binpack" ? used : -used;
711
+ score += softBonus(n, input);
712
+ return score;
713
+ }
714
+
715
+ function softBonus(n: NodeView, input: PlacementInput): number {
716
+ const { bundle, replicaIndex, inventory, recoveryContext } = input;
717
+ let bonus = 0;
718
+
719
+ if (recoveryContext?.preferHotSpares && n.isHotSpare) {
720
+ bonus += HOT_SPARE_BONUS;
721
+ }
722
+
723
+ // Soft sameAz (intra-replica): bonus if node's AZ matches a peer in the same replica.
724
+ const intraAz = bundle.intraReplicaPlacement?.sameAz ?? "soft";
725
+ if (intraAz === "soft" && n.az !== undefined) {
726
+ const peerAzs = peerAzsInSameReplica(inventory, bundle.bundleId, replicaIndex);
727
+ if (peerAzs.includes(n.az)) bonus += INTRA_SAME_AZ_BONUS;
728
+ }
729
+
730
+ // Resilience anti-affinity (backup vs primary): strong best-effort penalty
731
+ // for each `forbid` flag this node would violate, so the backup prefers a
732
+ // node outside the primary's failure domains.
733
+ const policy = backupPolicy(input);
734
+ if (policy) {
735
+ const primary = primaryDomains(inventory, bundle.bundleId);
736
+ bonus -= resilienceViolations(n, policy, primary).length * RESILIENCE_VIOLATION_PENALTY;
737
+ }
738
+
739
+ return bonus;
740
+ }
741
+
742
+ // ---------- GPU pick ----------
743
+
744
+ function pickGpuIndex(
745
+ n: NodeView,
746
+ req: JobRequirements,
747
+ strategy: "binpack" | "spread"
748
+ ): number | undefined {
749
+ if (req.requiredGpuCapacity === undefined || req.requiredGpuCapacity <= 0) {
750
+ return undefined;
751
+ }
752
+ const eligible = n.inventory.gpus.filter((g) => gpuFits(g, req));
753
+ if (eligible.length === 0) return undefined;
754
+ // Pack: pick the GPU with least free capacity that still fits. Spread:
755
+ // pick the most free. Same packing intent as node-level scoring.
756
+ const sorted = eligible.slice().sort((a, b) => {
757
+ const freeA = a.totalCapacity - a.reservedCapacity;
758
+ const freeB = b.totalCapacity - b.reservedCapacity;
759
+ if (freeA !== freeB) {
760
+ return strategy === "binpack" ? freeA - freeB : freeB - freeA;
761
+ }
762
+ return a.index - b.index;
763
+ });
764
+ return sorted[0].index;
765
+ }
766
+
767
+ // ---------- elastic provision ----------
768
+
769
+ function findProvisionableInstanceType(
770
+ tier: PlacementTier,
771
+ req: JobRequirements
772
+ ): string | undefined {
773
+ for (const it of tier.candidateInstanceTypes) {
774
+ if (instanceTypeSatisfies(it, req)) return it.instanceType;
775
+ }
776
+ return undefined;
777
+ }
778
+
779
+ function instanceTypeSatisfies(
780
+ it: InstanceTypeOption,
781
+ req: JobRequirements
782
+ ): boolean {
783
+ if (it.totalCapacity < req.requiredCapacity) return false;
784
+ if (it.totalCores < (req.requiredCores ?? 0)) return false;
785
+ for (const cap of req.requiredCapabilities) {
786
+ const found = it.capabilities.find((c) => c.name === cap.name);
787
+ if (!found || found.count < cap.count) return false;
788
+ if (cap.attributeMatches) {
789
+ for (const [k, v] of Object.entries(cap.attributeMatches)) {
790
+ if (found.attributes?.[k] !== v) return false;
791
+ }
792
+ }
793
+ }
794
+ if (req.requiredGpuCapacity !== undefined && req.requiredGpuCapacity > 0) {
795
+ const gpus = it.gpus ?? [];
796
+ const fits = gpus.some(
797
+ (g) =>
798
+ (req.requiredGpuModel === undefined || g.model === req.requiredGpuModel) &&
799
+ g.totalCapacity >= (req.requiredGpuCapacity ?? 0)
800
+ );
801
+ if (!fits) return false;
802
+ }
803
+ return true;
804
+ }