@kici-dev/engine 0.12.0 → 0.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -131,6 +131,15 @@ export declare const peerHeartbeatSchema: z.ZodObject<{
131
131
  runningAsUid: z.ZodOptional<z.ZodNullable<z.ZodNumber>>;
132
132
  version: z.ZodOptional<z.ZodString>;
133
133
  }, z.core.$strip>;
134
+ /**
135
+ * The spawn-retry budget a worker applies to one rerouted job: how many agent
136
+ * spawns it attempts, and how long it waits after a failed one. The sending
137
+ * coordinator resolves it per org and enforces the same budget on its side.
138
+ */
139
+ export declare const rerouteSpawnRetrySchema: z.ZodObject<{
140
+ maxAttempts: z.ZodNumber;
141
+ backoffMs: z.ZodNumber;
142
+ }, z.core.$strip>;
134
143
  /** Request to reroute a job to another orchestrator (no local agent can handle it). */
135
144
  export declare const jobRerouteSchema: z.ZodObject<{
136
145
  type: z.ZodLiteral<"job.reroute">;
@@ -164,6 +173,10 @@ export declare const jobRerouteSchema: z.ZodObject<{
164
173
  }, z.core.$strip>], "kind">>>;
165
174
  triedConnections: z.ZodArray<z.ZodString>;
166
175
  maxHops: z.ZodNumber;
176
+ spawnRetry: z.ZodOptional<z.ZodObject<{
177
+ maxAttempts: z.ZodNumber;
178
+ backoffMs: z.ZodNumber;
179
+ }, z.core.$strip>>;
167
180
  coordinatorId: z.ZodString;
168
181
  requestId: z.ZodOptional<z.ZodString>;
169
182
  traceId: z.ZodOptional<z.ZodString>;
@@ -351,8 +364,177 @@ export declare const peerConfigReloadResponseSchema: z.ZodObject<{
351
364
  restartRequired: z.ZodOptional<z.ZodArray<z.ZodString>>;
352
365
  fieldsChanged: z.ZodOptional<z.ZodArray<z.ZodString>>;
353
366
  }, z.core.$strip>;
367
+ /** How a live Firecracker VM relates to the orchestrator on its host. */
368
+ export declare const ScalerVmStatus: z.ZodEnum<{
369
+ orphaned: "orphaned";
370
+ tracked: "tracked";
371
+ unverified: "unverified";
372
+ }>;
373
+ export type ScalerVmStatus = z.infer<typeof ScalerVmStatus>;
374
+ /** What tracks a live VM. `bound-job` is answered from a coordinator's dispatch queue. */
375
+ export declare const ScalerVmTracker: z.ZodEnum<{
376
+ backend: "backend";
377
+ "bound-job": "bound-job";
378
+ registered: "registered";
379
+ spawning: "spawning";
380
+ }>;
381
+ export type ScalerVmTracker = z.infer<typeof ScalerVmTracker>;
382
+ /** What a stop did to one VM. Only `stopped` signalled anything. */
383
+ export declare const ScalerVmStopOutcome: z.ZodEnum<{
384
+ error: "error";
385
+ "not-found": "not-found";
386
+ "not-live": "not-live";
387
+ stopped: "stopped";
388
+ tracked: "tracked";
389
+ unverified: "unverified";
390
+ }>;
391
+ export type ScalerVmStopOutcome = z.infer<typeof ScalerVmStopOutcome>;
392
+ /** What a scaler orphan request asks the target node to do. */
393
+ export declare const ScalerOrphansAction: z.ZodEnum<{
394
+ list: "list";
395
+ stop: "stop";
396
+ }>;
397
+ export type ScalerOrphansAction = z.infer<typeof ScalerOrphansAction>;
398
+ /** One live Firecracker VM on a node, as its orchestrator sees it. */
399
+ export declare const scalerLiveVmSchema: z.ZodObject<{
400
+ vmId: z.ZodString;
401
+ scaler: z.ZodString;
402
+ pid: z.ZodNumber;
403
+ startedAt: z.ZodString;
404
+ ageSeconds: z.ZodNumber;
405
+ chrootDir: z.ZodString;
406
+ status: z.ZodEnum<{
407
+ orphaned: "orphaned";
408
+ tracked: "tracked";
409
+ unverified: "unverified";
410
+ }>;
411
+ trackedBy: z.ZodArray<z.ZodEnum<{
412
+ backend: "backend";
413
+ "bound-job": "bound-job";
414
+ registered: "registered";
415
+ spawning: "spawning";
416
+ }>>;
417
+ reason: z.ZodString;
418
+ }, z.core.$strip>;
419
+ export type ScalerLiveVm = z.infer<typeof scalerLiveVmSchema>;
420
+ /** The result of stopping one VM. `pid` is the process that was (or would be) signalled. */
421
+ export declare const scalerVmStopResultSchema: z.ZodObject<{
422
+ vmId: z.ZodString;
423
+ outcome: z.ZodEnum<{
424
+ error: "error";
425
+ "not-found": "not-found";
426
+ "not-live": "not-live";
427
+ stopped: "stopped";
428
+ tracked: "tracked";
429
+ unverified: "unverified";
430
+ }>;
431
+ pid: z.ZodOptional<z.ZodNumber>;
432
+ detail: z.ZodString;
433
+ }, z.core.$strip>;
434
+ export type ScalerVmStopResult = z.infer<typeof scalerVmStopResultSchema>;
435
+ /**
436
+ * Scaler orphan request: forwarded by a coordinator to the node an operator
437
+ * targeted with `kici-admin scaler orphans --target`. The node answers from
438
+ * its own host and its own tracking with peer.scaler.orphans.response.
439
+ */
440
+ export declare const peerScalerOrphansRequestSchema: z.ZodObject<{
441
+ type: z.ZodLiteral<"peer.scaler.orphans.request">;
442
+ messageId: z.ZodString;
443
+ action: z.ZodEnum<{
444
+ list: "list";
445
+ stop: "stop";
446
+ }>;
447
+ vmIds: z.ZodOptional<z.ZodArray<z.ZodString>>;
448
+ }, z.core.$strip>;
449
+ /**
450
+ * Scaler orphan response. `vms` answers a `list`, `results` answers a `stop`;
451
+ * `ok: false` carries the node's `error` instead.
452
+ */
453
+ export declare const peerScalerOrphansResponseSchema: z.ZodObject<{
454
+ type: z.ZodLiteral<"peer.scaler.orphans.response">;
455
+ messageId: z.ZodString;
456
+ ok: z.ZodBoolean;
457
+ error: z.ZodOptional<z.ZodString>;
458
+ firecrackerScalers: z.ZodOptional<z.ZodArray<z.ZodString>>;
459
+ vms: z.ZodOptional<z.ZodArray<z.ZodObject<{
460
+ vmId: z.ZodString;
461
+ scaler: z.ZodString;
462
+ pid: z.ZodNumber;
463
+ startedAt: z.ZodString;
464
+ ageSeconds: z.ZodNumber;
465
+ chrootDir: z.ZodString;
466
+ status: z.ZodEnum<{
467
+ orphaned: "orphaned";
468
+ tracked: "tracked";
469
+ unverified: "unverified";
470
+ }>;
471
+ trackedBy: z.ZodArray<z.ZodEnum<{
472
+ backend: "backend";
473
+ "bound-job": "bound-job";
474
+ registered: "registered";
475
+ spawning: "spawning";
476
+ }>>;
477
+ reason: z.ZodString;
478
+ }, z.core.$strip>>>;
479
+ results: z.ZodOptional<z.ZodArray<z.ZodObject<{
480
+ vmId: z.ZodString;
481
+ outcome: z.ZodEnum<{
482
+ error: "error";
483
+ "not-found": "not-found";
484
+ "not-live": "not-live";
485
+ stopped: "stopped";
486
+ tracked: "tracked";
487
+ unverified: "unverified";
488
+ }>;
489
+ pid: z.ZodOptional<z.ZodNumber>;
490
+ detail: z.ZodString;
491
+ }, z.core.$strip>>>;
492
+ }, z.core.$strip>;
493
+ /**
494
+ * What forgetting a departed peer did on one coordinator. `recent`: the peer
495
+ * was heard from inside the window this coordinator still treats it as alive
496
+ * for (the longer of the stale window and the reroute flap grace, or the
497
+ * backstop grace of a peer that adopted provisions), so it may be partitioned
498
+ * rather than gone. `acknowledgement-required`: forgetting it would switch this
499
+ * coordinator's event-provision backstop back on, and the request did not
500
+ * acknowledge that. Every outcome but `forgotten` keeps the peer.
501
+ */
502
+ export declare const PeerForgetOutcome: z.ZodEnum<{
503
+ "acknowledgement-required": "acknowledgement-required";
504
+ connected: "connected";
505
+ error: "error";
506
+ forgotten: "forgotten";
507
+ "not-found": "not-found";
508
+ recent: "recent";
509
+ }>;
510
+ export type PeerForgetOutcome = z.infer<typeof PeerForgetOutcome>;
511
+ /**
512
+ * Forget request: a coordinator that forgot a departed peer (`kici-admin peer
513
+ * forget`) asks each connected sibling coordinator to drop it from its own
514
+ * live peer registry too.
515
+ */
516
+ export declare const peerForgetRequestSchema: z.ZodObject<{
517
+ type: z.ZodLiteral<"peer.forget.request">;
518
+ messageId: z.ZodString;
519
+ instanceId: z.ZodString;
520
+ acknowledgeBackstop: z.ZodOptional<z.ZodBoolean>;
521
+ }, z.core.$strip>;
522
+ /** Forget response: what the sibling did with the request. */
523
+ export declare const peerForgetResponseSchema: z.ZodObject<{
524
+ type: z.ZodLiteral<"peer.forget.response">;
525
+ messageId: z.ZodString;
526
+ outcome: z.ZodEnum<{
527
+ "acknowledgement-required": "acknowledgement-required";
528
+ connected: "connected";
529
+ error: "error";
530
+ forgotten: "forgotten";
531
+ "not-found": "not-found";
532
+ recent: "recent";
533
+ }>;
534
+ detail: z.ZodString;
535
+ }, z.core.$strip>;
354
536
  /**
355
- * Worker-relevant cluster settings a DB-less worker pulls from the leader.
537
+ * Worker-relevant cluster settings a DB-less worker pulls from a coordinator.
356
538
  *
357
539
  * A typed, concrete snapshot (not an open key/value bus): each worker-consumed
358
540
  * cluster knob is a field here. Adding the next worker-relevant knob is one
@@ -360,11 +542,12 @@ export declare const peerConfigReloadResponseSchema: z.ZodObject<{
360
542
  */
361
543
  export declare const workerClusterSettingsSchema: z.ZodObject<{
362
544
  agentTokenTtlMs: z.ZodNumber;
545
+ firecrackerApiSocketWaitMs: z.ZodOptional<z.ZodNumber>;
363
546
  }, z.core.$strip>;
364
547
  /**
365
- * Cluster-settings pull request: a DB-less worker asks the leader for the
366
- * current worker-settings snapshot after observing the leader's advertised
367
- * clusterSettingsVersion ahead of its own. The leader replies with
548
+ * Cluster-settings pull request: a DB-less worker asks a connected coordinator
549
+ * for the current worker-settings snapshot after a coordinator advertised a
550
+ * clusterSettingsVersion ahead of its own. The coordinator replies with
368
551
  * peer.clusterSettings.response.
369
552
  */
370
553
  export declare const peerClusterSettingsRequestSchema: z.ZodObject<{
@@ -372,7 +555,7 @@ export declare const peerClusterSettingsRequestSchema: z.ZodObject<{
372
555
  messageId: z.ZodString;
373
556
  }, z.core.$strip>;
374
557
  /**
375
- * Cluster-settings pull response: the leader resolves the snapshot from
558
+ * Cluster-settings pull response: the coordinator resolves the snapshot from
376
559
  * cluster_settings and replies with it plus the current version.
377
560
  */
378
561
  export declare const peerClusterSettingsResponseSchema: z.ZodObject<{
@@ -381,6 +564,7 @@ export declare const peerClusterSettingsResponseSchema: z.ZodObject<{
381
564
  version: z.ZodNumber;
382
565
  settings: z.ZodObject<{
383
566
  agentTokenTtlMs: z.ZodNumber;
567
+ firecrackerApiSocketWaitMs: z.ZodOptional<z.ZodNumber>;
384
568
  }, z.core.$strip>;
385
569
  }, z.core.$strip>;
386
570
  /** Which of THIS peer's downstream nodes to gather. all=true ignores the id lists. */
@@ -455,6 +639,7 @@ export declare const peerScalerEventSchema: z.ZodObject<{
455
639
  }>;
456
640
  detail: z.ZodString;
457
641
  timestampMs: z.ZodNumber;
642
+ final: z.ZodOptional<z.ZodBoolean>;
458
643
  }, z.core.$strip>;
459
644
  /** All peer-to-peer messages (outbound from this node). */
460
645
  export declare const peerToPeerMessageSchema: z.ZodDiscriminatedUnion<[z.ZodObject<{
@@ -588,6 +773,10 @@ export declare const peerToPeerMessageSchema: z.ZodDiscriminatedUnion<[z.ZodObje
588
773
  }, z.core.$strip>], "kind">>>;
589
774
  triedConnections: z.ZodArray<z.ZodString>;
590
775
  maxHops: z.ZodNumber;
776
+ spawnRetry: z.ZodOptional<z.ZodObject<{
777
+ maxAttempts: z.ZodNumber;
778
+ backoffMs: z.ZodNumber;
779
+ }, z.core.$strip>>;
591
780
  coordinatorId: z.ZodString;
592
781
  requestId: z.ZodOptional<z.ZodString>;
593
782
  traceId: z.ZodOptional<z.ZodString>;
@@ -718,6 +907,70 @@ export declare const peerToPeerMessageSchema: z.ZodDiscriminatedUnion<[z.ZodObje
718
907
  errors: z.ZodOptional<z.ZodArray<z.ZodString>>;
719
908
  restartRequired: z.ZodOptional<z.ZodArray<z.ZodString>>;
720
909
  fieldsChanged: z.ZodOptional<z.ZodArray<z.ZodString>>;
910
+ }, z.core.$strip>, z.ZodObject<{
911
+ type: z.ZodLiteral<"peer.scaler.orphans.request">;
912
+ messageId: z.ZodString;
913
+ action: z.ZodEnum<{
914
+ list: "list";
915
+ stop: "stop";
916
+ }>;
917
+ vmIds: z.ZodOptional<z.ZodArray<z.ZodString>>;
918
+ }, z.core.$strip>, z.ZodObject<{
919
+ type: z.ZodLiteral<"peer.scaler.orphans.response">;
920
+ messageId: z.ZodString;
921
+ ok: z.ZodBoolean;
922
+ error: z.ZodOptional<z.ZodString>;
923
+ firecrackerScalers: z.ZodOptional<z.ZodArray<z.ZodString>>;
924
+ vms: z.ZodOptional<z.ZodArray<z.ZodObject<{
925
+ vmId: z.ZodString;
926
+ scaler: z.ZodString;
927
+ pid: z.ZodNumber;
928
+ startedAt: z.ZodString;
929
+ ageSeconds: z.ZodNumber;
930
+ chrootDir: z.ZodString;
931
+ status: z.ZodEnum<{
932
+ orphaned: "orphaned";
933
+ tracked: "tracked";
934
+ unverified: "unverified";
935
+ }>;
936
+ trackedBy: z.ZodArray<z.ZodEnum<{
937
+ backend: "backend";
938
+ "bound-job": "bound-job";
939
+ registered: "registered";
940
+ spawning: "spawning";
941
+ }>>;
942
+ reason: z.ZodString;
943
+ }, z.core.$strip>>>;
944
+ results: z.ZodOptional<z.ZodArray<z.ZodObject<{
945
+ vmId: z.ZodString;
946
+ outcome: z.ZodEnum<{
947
+ error: "error";
948
+ "not-found": "not-found";
949
+ "not-live": "not-live";
950
+ stopped: "stopped";
951
+ tracked: "tracked";
952
+ unverified: "unverified";
953
+ }>;
954
+ pid: z.ZodOptional<z.ZodNumber>;
955
+ detail: z.ZodString;
956
+ }, z.core.$strip>>>;
957
+ }, z.core.$strip>, z.ZodObject<{
958
+ type: z.ZodLiteral<"peer.forget.request">;
959
+ messageId: z.ZodString;
960
+ instanceId: z.ZodString;
961
+ acknowledgeBackstop: z.ZodOptional<z.ZodBoolean>;
962
+ }, z.core.$strip>, z.ZodObject<{
963
+ type: z.ZodLiteral<"peer.forget.response">;
964
+ messageId: z.ZodString;
965
+ outcome: z.ZodEnum<{
966
+ "acknowledgement-required": "acknowledgement-required";
967
+ connected: "connected";
968
+ error: "error";
969
+ forgotten: "forgotten";
970
+ "not-found": "not-found";
971
+ recent: "recent";
972
+ }>;
973
+ detail: z.ZodString;
721
974
  }, z.core.$strip>, z.ZodObject<{
722
975
  type: z.ZodLiteral<"peer.clusterSettings.request">;
723
976
  messageId: z.ZodString;
@@ -727,6 +980,7 @@ export declare const peerToPeerMessageSchema: z.ZodDiscriminatedUnion<[z.ZodObje
727
980
  version: z.ZodNumber;
728
981
  settings: z.ZodObject<{
729
982
  agentTokenTtlMs: z.ZodNumber;
983
+ firecrackerApiSocketWaitMs: z.ZodOptional<z.ZodNumber>;
730
984
  }, z.core.$strip>;
731
985
  }, z.core.$strip>, z.ZodObject<{
732
986
  type: z.ZodLiteral<"peer.logs.collect.request">;
@@ -770,6 +1024,7 @@ export declare const peerToPeerMessageSchema: z.ZodDiscriminatedUnion<[z.ZodObje
770
1024
  }>;
771
1025
  detail: z.ZodString;
772
1026
  timestampMs: z.ZodNumber;
1027
+ final: z.ZodOptional<z.ZodBoolean>;
773
1028
  }, z.core.$strip>], "type">;
774
1029
  /** All peer-to-peer messages (inbound to this node). */
775
1030
  export declare const peerFromPeerMessageSchema: z.ZodDiscriminatedUnion<[z.ZodObject<{
@@ -903,6 +1158,10 @@ export declare const peerFromPeerMessageSchema: z.ZodDiscriminatedUnion<[z.ZodOb
903
1158
  }, z.core.$strip>], "kind">>>;
904
1159
  triedConnections: z.ZodArray<z.ZodString>;
905
1160
  maxHops: z.ZodNumber;
1161
+ spawnRetry: z.ZodOptional<z.ZodObject<{
1162
+ maxAttempts: z.ZodNumber;
1163
+ backoffMs: z.ZodNumber;
1164
+ }, z.core.$strip>>;
906
1165
  coordinatorId: z.ZodString;
907
1166
  requestId: z.ZodOptional<z.ZodString>;
908
1167
  traceId: z.ZodOptional<z.ZodString>;
@@ -1033,6 +1292,70 @@ export declare const peerFromPeerMessageSchema: z.ZodDiscriminatedUnion<[z.ZodOb
1033
1292
  errors: z.ZodOptional<z.ZodArray<z.ZodString>>;
1034
1293
  restartRequired: z.ZodOptional<z.ZodArray<z.ZodString>>;
1035
1294
  fieldsChanged: z.ZodOptional<z.ZodArray<z.ZodString>>;
1295
+ }, z.core.$strip>, z.ZodObject<{
1296
+ type: z.ZodLiteral<"peer.scaler.orphans.request">;
1297
+ messageId: z.ZodString;
1298
+ action: z.ZodEnum<{
1299
+ list: "list";
1300
+ stop: "stop";
1301
+ }>;
1302
+ vmIds: z.ZodOptional<z.ZodArray<z.ZodString>>;
1303
+ }, z.core.$strip>, z.ZodObject<{
1304
+ type: z.ZodLiteral<"peer.scaler.orphans.response">;
1305
+ messageId: z.ZodString;
1306
+ ok: z.ZodBoolean;
1307
+ error: z.ZodOptional<z.ZodString>;
1308
+ firecrackerScalers: z.ZodOptional<z.ZodArray<z.ZodString>>;
1309
+ vms: z.ZodOptional<z.ZodArray<z.ZodObject<{
1310
+ vmId: z.ZodString;
1311
+ scaler: z.ZodString;
1312
+ pid: z.ZodNumber;
1313
+ startedAt: z.ZodString;
1314
+ ageSeconds: z.ZodNumber;
1315
+ chrootDir: z.ZodString;
1316
+ status: z.ZodEnum<{
1317
+ orphaned: "orphaned";
1318
+ tracked: "tracked";
1319
+ unverified: "unverified";
1320
+ }>;
1321
+ trackedBy: z.ZodArray<z.ZodEnum<{
1322
+ backend: "backend";
1323
+ "bound-job": "bound-job";
1324
+ registered: "registered";
1325
+ spawning: "spawning";
1326
+ }>>;
1327
+ reason: z.ZodString;
1328
+ }, z.core.$strip>>>;
1329
+ results: z.ZodOptional<z.ZodArray<z.ZodObject<{
1330
+ vmId: z.ZodString;
1331
+ outcome: z.ZodEnum<{
1332
+ error: "error";
1333
+ "not-found": "not-found";
1334
+ "not-live": "not-live";
1335
+ stopped: "stopped";
1336
+ tracked: "tracked";
1337
+ unverified: "unverified";
1338
+ }>;
1339
+ pid: z.ZodOptional<z.ZodNumber>;
1340
+ detail: z.ZodString;
1341
+ }, z.core.$strip>>>;
1342
+ }, z.core.$strip>, z.ZodObject<{
1343
+ type: z.ZodLiteral<"peer.forget.request">;
1344
+ messageId: z.ZodString;
1345
+ instanceId: z.ZodString;
1346
+ acknowledgeBackstop: z.ZodOptional<z.ZodBoolean>;
1347
+ }, z.core.$strip>, z.ZodObject<{
1348
+ type: z.ZodLiteral<"peer.forget.response">;
1349
+ messageId: z.ZodString;
1350
+ outcome: z.ZodEnum<{
1351
+ "acknowledgement-required": "acknowledgement-required";
1352
+ connected: "connected";
1353
+ error: "error";
1354
+ forgotten: "forgotten";
1355
+ "not-found": "not-found";
1356
+ recent: "recent";
1357
+ }>;
1358
+ detail: z.ZodString;
1036
1359
  }, z.core.$strip>, z.ZodObject<{
1037
1360
  type: z.ZodLiteral<"peer.clusterSettings.request">;
1038
1361
  messageId: z.ZodString;
@@ -1042,6 +1365,7 @@ export declare const peerFromPeerMessageSchema: z.ZodDiscriminatedUnion<[z.ZodOb
1042
1365
  version: z.ZodNumber;
1043
1366
  settings: z.ZodObject<{
1044
1367
  agentTokenTtlMs: z.ZodNumber;
1368
+ firecrackerApiSocketWaitMs: z.ZodOptional<z.ZodNumber>;
1045
1369
  }, z.core.$strip>;
1046
1370
  }, z.core.$strip>, z.ZodObject<{
1047
1371
  type: z.ZodLiteral<"peer.logs.collect.request">;
@@ -1085,11 +1409,13 @@ export declare const peerFromPeerMessageSchema: z.ZodDiscriminatedUnion<[z.ZodOb
1085
1409
  }>;
1086
1410
  detail: z.ZodString;
1087
1411
  timestampMs: z.ZodNumber;
1412
+ final: z.ZodOptional<z.ZodBoolean>;
1088
1413
  }, z.core.$strip>], "type">;
1089
1414
  export type PeerCapabilities = z.infer<typeof peerCapabilitiesSchema>;
1090
1415
  export type ScalerCapacitySummary = z.infer<typeof scalerCapacitySummarySchema>;
1091
1416
  export type PeerHeartbeat = z.infer<typeof peerHeartbeatSchema>;
1092
1417
  export type JobReroute = z.infer<typeof jobRerouteSchema>;
1418
+ export type RerouteSpawnRetry = z.infer<typeof rerouteSpawnRetrySchema>;
1093
1419
  export type JobProgress = z.infer<typeof jobProgressSchema>;
1094
1420
  export type JobProgressAck = z.infer<typeof jobProgressAckSchema>;
1095
1421
  export type PeerScalerEvent = z.infer<typeof peerScalerEventSchema>;
@@ -1102,6 +1428,10 @@ export type PeerCacheUploadRequest = z.infer<typeof peerCacheUploadRequestSchema
1102
1428
  export type PeerCacheUploadResponse = z.infer<typeof peerCacheUploadResponseSchema>;
1103
1429
  export type PeerConfigReload = z.infer<typeof peerConfigReloadSchema>;
1104
1430
  export type PeerConfigReloadResponse = z.infer<typeof peerConfigReloadResponseSchema>;
1431
+ export type PeerScalerOrphansRequest = z.infer<typeof peerScalerOrphansRequestSchema>;
1432
+ export type PeerScalerOrphansResponse = z.infer<typeof peerScalerOrphansResponseSchema>;
1433
+ export type PeerForgetRequest = z.infer<typeof peerForgetRequestSchema>;
1434
+ export type PeerForgetResponse = z.infer<typeof peerForgetResponseSchema>;
1105
1435
  export type WorkerClusterSettings = z.infer<typeof workerClusterSettingsSchema>;
1106
1436
  export type PeerClusterSettingsRequest = z.infer<typeof peerClusterSettingsRequestSchema>;
1107
1437
  export type PeerClusterSettingsResponse = z.infer<typeof peerClusterSettingsResponseSchema>;
@@ -126,7 +126,7 @@ const peerHeartbeatSchema = z.object({
126
126
  configVersion: z.number().optional(),
127
127
  /** Registry version for cross-orchestrator registration sync (backward compatible). */
128
128
  registryVersion: z.number().optional(),
129
- /** Shared cluster-settings version for leader→worker settings pull (backward compatible). */
129
+ /** Cluster-settings version for the coordinator→worker settings pull (backward compatible). */
130
130
  clusterSettingsVersion: z.number().optional(),
131
131
  timestamp: z.number(),
132
132
  hostname: z.string().optional(),
@@ -141,6 +141,15 @@ const peerHeartbeatSchema = z.object({
141
141
  runningAsUid: z.number().nullable().optional(),
142
142
  version: z.string().optional()
143
143
  });
144
+ /**
145
+ * The spawn-retry budget a worker applies to one rerouted job: how many agent
146
+ * spawns it attempts, and how long it waits after a failed one. The sending
147
+ * coordinator resolves it per org and enforces the same budget on its side.
148
+ */
149
+ const rerouteSpawnRetrySchema = z.object({
150
+ maxAttempts: z.number().int().min(1),
151
+ backoffMs: z.number().int().min(0)
152
+ });
144
153
  /** Request to reroute a job to another orchestrator (no local agent can handle it). */
145
154
  const jobRerouteSchema = z.object({
146
155
  type: z.literal("job.reroute"),
@@ -178,6 +187,11 @@ const jobRerouteSchema = z.object({
178
187
  excludePatterns: z.array(LabelMatcher).optional(),
179
188
  triedConnections: z.array(z.string()),
180
189
  maxHops: z.number(),
190
+ /**
191
+ * The worker's spawn-retry budget for this job. Absent from an older sender:
192
+ * the worker applies its own configured default.
193
+ */
194
+ spawnRetry: rerouteSpawnRetrySchema.optional(),
181
195
  coordinatorId: z.string(),
182
196
  requestId: z.string().optional(),
183
197
  traceId: z.string().optional(),
@@ -346,18 +360,135 @@ const peerConfigReloadResponseSchema = z.object({
346
360
  restartRequired: z.array(z.string()).optional(),
347
361
  fieldsChanged: z.array(z.string()).optional()
348
362
  });
363
+ /** How a live Firecracker VM relates to the orchestrator on its host. */
364
+ const ScalerVmStatus = z.enum([
365
+ "orphaned",
366
+ "unverified",
367
+ "tracked"
368
+ ]);
369
+ /** What tracks a live VM. `bound-job` is answered from a coordinator's dispatch queue. */
370
+ const ScalerVmTracker = z.enum([
371
+ "backend",
372
+ "spawning",
373
+ "registered",
374
+ "bound-job"
375
+ ]);
376
+ /** What a stop did to one VM. Only `stopped` signalled anything. */
377
+ const ScalerVmStopOutcome = z.enum([
378
+ "stopped",
379
+ "tracked",
380
+ "unverified",
381
+ "not-live",
382
+ "not-found",
383
+ "error"
384
+ ]);
385
+ /** What a scaler orphan request asks the target node to do. */
386
+ const ScalerOrphansAction = z.enum(["list", "stop"]);
387
+ /** One live Firecracker VM on a node, as its orchestrator sees it. */
388
+ const scalerLiveVmSchema = z.object({
389
+ vmId: z.string(),
390
+ scaler: z.string(),
391
+ pid: z.number().int().positive(),
392
+ startedAt: z.string(),
393
+ ageSeconds: z.number().nonnegative(),
394
+ chrootDir: z.string(),
395
+ status: ScalerVmStatus,
396
+ trackedBy: z.array(ScalerVmTracker),
397
+ reason: z.string()
398
+ });
399
+ /** The result of stopping one VM. `pid` is the process that was (or would be) signalled. */
400
+ const scalerVmStopResultSchema = z.object({
401
+ vmId: z.string(),
402
+ outcome: ScalerVmStopOutcome,
403
+ pid: z.number().int().positive().optional(),
404
+ detail: z.string()
405
+ });
406
+ /**
407
+ * Scaler orphan request: forwarded by a coordinator to the node an operator
408
+ * targeted with `kici-admin scaler orphans --target`. The node answers from
409
+ * its own host and its own tracking with peer.scaler.orphans.response.
410
+ */
411
+ const peerScalerOrphansRequestSchema = z.object({
412
+ type: z.literal("peer.scaler.orphans.request"),
413
+ messageId: z.string(),
414
+ action: ScalerOrphansAction,
415
+ /** `stop` only: the VM ids the operator approved. */
416
+ vmIds: z.array(z.string()).optional()
417
+ });
418
+ /**
419
+ * Scaler orphan response. `vms` answers a `list`, `results` answers a `stop`;
420
+ * `ok: false` carries the node's `error` instead.
421
+ */
422
+ const peerScalerOrphansResponseSchema = z.object({
423
+ type: z.literal("peer.scaler.orphans.response"),
424
+ messageId: z.string(),
425
+ ok: z.boolean(),
426
+ error: z.string().optional(),
427
+ /** The node's Firecracker scaler names; empty when it runs none. */
428
+ firecrackerScalers: z.array(z.string()).optional(),
429
+ vms: z.array(scalerLiveVmSchema).optional(),
430
+ results: z.array(scalerVmStopResultSchema).optional()
431
+ });
349
432
  /**
350
- * Worker-relevant cluster settings a DB-less worker pulls from the leader.
433
+ * What forgetting a departed peer did on one coordinator. `recent`: the peer
434
+ * was heard from inside the window this coordinator still treats it as alive
435
+ * for (the longer of the stale window and the reroute flap grace, or the
436
+ * backstop grace of a peer that adopted provisions), so it may be partitioned
437
+ * rather than gone. `acknowledgement-required`: forgetting it would switch this
438
+ * coordinator's event-provision backstop back on, and the request did not
439
+ * acknowledge that. Every outcome but `forgotten` keeps the peer.
440
+ */
441
+ const PeerForgetOutcome = z.enum([
442
+ "forgotten",
443
+ "not-found",
444
+ "connected",
445
+ "recent",
446
+ "acknowledgement-required",
447
+ "error"
448
+ ]);
449
+ /**
450
+ * Forget request: a coordinator that forgot a departed peer (`kici-admin peer
451
+ * forget`) asks each connected sibling coordinator to drop it from its own
452
+ * live peer registry too.
453
+ */
454
+ const peerForgetRequestSchema = z.object({
455
+ type: z.literal("peer.forget.request"),
456
+ messageId: z.string(),
457
+ /** The departed peer's instance id. */
458
+ instanceId: z.string(),
459
+ /**
460
+ * The operator acknowledged that forgetting the peer may switch the
461
+ * event-provision backstop back on. Absent means not acknowledged.
462
+ */
463
+ acknowledgeBackstop: z.boolean().optional()
464
+ });
465
+ /** Forget response: what the sibling did with the request. */
466
+ const peerForgetResponseSchema = z.object({
467
+ type: z.literal("peer.forget.response"),
468
+ messageId: z.string(),
469
+ outcome: PeerForgetOutcome,
470
+ detail: z.string()
471
+ });
472
+ /**
473
+ * Worker-relevant cluster settings a DB-less worker pulls from a coordinator.
351
474
  *
352
475
  * A typed, concrete snapshot (not an open key/value bus): each worker-consumed
353
476
  * cluster knob is a field here. Adding the next worker-relevant knob is one
354
477
  * field on this schema, not a new protocol message.
355
478
  */
356
- const workerClusterSettingsSchema = z.object({ agentTokenTtlMs: z.number() });
479
+ const workerClusterSettingsSchema = z.object({
480
+ agentTokenTtlMs: z.number(),
481
+ /**
482
+ * How long a Firecracker spawn waits for the VM's API socket. Optional: a
483
+ * leader that predates the knob omits it, and the worker keeps its own
484
+ * configured default.
485
+ */
486
+ firecrackerApiSocketWaitMs: z.number().optional()
487
+ });
357
488
  /**
358
- * Cluster-settings pull request: a DB-less worker asks the leader for the
359
- * current worker-settings snapshot after observing the leader's advertised
360
- * clusterSettingsVersion ahead of its own. The leader replies with
489
+ * Cluster-settings pull request: a DB-less worker asks a connected coordinator
490
+ * for the current worker-settings snapshot after a coordinator advertised a
491
+ * clusterSettingsVersion ahead of its own. The coordinator replies with
361
492
  * peer.clusterSettings.response.
362
493
  */
363
494
  const peerClusterSettingsRequestSchema = z.object({
@@ -365,7 +496,7 @@ const peerClusterSettingsRequestSchema = z.object({
365
496
  messageId: z.string()
366
497
  });
367
498
  /**
368
- * Cluster-settings pull response: the leader resolves the snapshot from
499
+ * Cluster-settings pull response: the coordinator resolves the snapshot from
369
500
  * cluster_settings and replies with it plus the current version.
370
501
  */
371
502
  const peerClusterSettingsResponseSchema = z.object({
@@ -443,7 +574,14 @@ const peerScalerEventSchema = z.object({
443
574
  /** Human-readable detail, including any captured spawn stderr tail. */
444
575
  detail: z.string(),
445
576
  /** Event timestamp in epoch milliseconds. */
446
- timestampMs: z.number()
577
+ timestampMs: z.number(),
578
+ /**
579
+ * The worker's retry verdict on a `scaler.failed` for a job under its spawn-retry
580
+ * budget: false while attempts remain, true on the last one. Absent on a repeated
581
+ * report of one failed spawn, on every other relay, and from an older worker; the
582
+ * coordinator then keeps its spawn window running unchanged.
583
+ */
584
+ final: z.boolean().optional()
447
585
  });
448
586
  /** All peer-to-peer messages (outbound from this node). */
449
587
  const peerToPeerMessageSchema = z.discriminatedUnion("type", [
@@ -465,6 +603,10 @@ const peerToPeerMessageSchema = z.discriminatedUnion("type", [
465
603
  peerCacheUploadResponseSchema,
466
604
  peerConfigReloadSchema,
467
605
  peerConfigReloadResponseSchema,
606
+ peerScalerOrphansRequestSchema,
607
+ peerScalerOrphansResponseSchema,
608
+ peerForgetRequestSchema,
609
+ peerForgetResponseSchema,
468
610
  peerClusterSettingsRequestSchema,
469
611
  peerClusterSettingsResponseSchema,
470
612
  peerLogsCollectRequestSchema,
@@ -494,6 +636,10 @@ const peerFromPeerMessageSchema = z.discriminatedUnion("type", [
494
636
  peerCacheUploadResponseSchema,
495
637
  peerConfigReloadSchema,
496
638
  peerConfigReloadResponseSchema,
639
+ peerScalerOrphansRequestSchema,
640
+ peerScalerOrphansResponseSchema,
641
+ peerForgetRequestSchema,
642
+ peerForgetResponseSchema,
497
643
  peerClusterSettingsRequestSchema,
498
644
  peerClusterSettingsResponseSchema,
499
645
  peerLogsCollectRequestSchema,
@@ -504,6 +650,6 @@ const peerFromPeerMessageSchema = z.discriminatedUnion("type", [
504
650
  peerScalerEventSchema
505
651
  ]);
506
652
  //#endregion
507
- export { fleetSelectionSchema, jobProgressAckSchema, jobProgressSchema, jobRerouteAckSchema, jobRerouteSchema, peerAgentTokenRevokeSchema, peerAuthRequestSchema, peerAuthResponseSchema, peerCapabilitiesSchema, peerClusterSettingsRequestSchema, peerClusterSettingsResponseSchema, peerConfigReloadResponseSchema, peerConfigReloadSchema, peerFromPeerMessageSchema, peerHeartbeatSchema, peerHelloResponseSchema, peerHelloSchema, peerJobCancelSchema, peerLeavingSchema, peerLogsCollectChunkSchema, peerLogsCollectErrorSchema, peerLogsCollectRequestSchema, peerScalerEventSchema, peerToPeerMessageSchema, raftAppendEntriesSchema, raftVoteRequestSchema, raftVoteResponseSchema, workerClusterSettingsSchema };
653
+ export { PeerForgetOutcome, ScalerOrphansAction, ScalerVmStatus, ScalerVmStopOutcome, ScalerVmTracker, fleetSelectionSchema, jobProgressAckSchema, jobProgressSchema, jobRerouteAckSchema, jobRerouteSchema, peerAgentTokenRevokeSchema, peerAuthRequestSchema, peerAuthResponseSchema, peerCapabilitiesSchema, peerClusterSettingsRequestSchema, peerClusterSettingsResponseSchema, peerConfigReloadResponseSchema, peerConfigReloadSchema, peerForgetRequestSchema, peerForgetResponseSchema, peerFromPeerMessageSchema, peerHeartbeatSchema, peerHelloResponseSchema, peerHelloSchema, peerJobCancelSchema, peerLeavingSchema, peerLogsCollectChunkSchema, peerLogsCollectErrorSchema, peerLogsCollectRequestSchema, peerScalerEventSchema, peerScalerOrphansRequestSchema, peerScalerOrphansResponseSchema, peerToPeerMessageSchema, raftAppendEntriesSchema, raftVoteRequestSchema, raftVoteResponseSchema, rerouteSpawnRetrySchema, scalerLiveVmSchema, scalerVmStopResultSchema, workerClusterSettingsSchema };
508
654
 
509
655
  //# sourceMappingURL=peer.js.map