@cotal-ai/delivery 0.56.1 → 0.58.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/delivery.js CHANGED
@@ -1,10 +1,11 @@
1
1
  import { existsSync, readFileSync, rmSync, writeFileSync } from "node:fs";
2
2
  import { basename, dirname, join, resolve } from "node:path";
3
- import { CotalEndpoint, DEFAULT_SERVER, LEASE_TTL_MS, accountFromCreds, credsClaims, deliveryBucket, dialerFor, idFromCreds, defaultProbeTimeoutMs, isCasLoss, isPermissionDenied, isReachable, leaseKey, mintCreds, newIdentity, formatSecretStoreIdentity, parseSecretStoreIdentity, sameSecretStoreIdentity, standaloneConnectOpts, startTimerWriter, } from "@cotal-ai/core";
3
+ import { CotalEndpoint, DEFAULT_SERVER, LEASE_TTL_MS, accountFromCreds, credsClaims, deliveryBucket, dialerFor, idFromCreds, isCasLoss, isPermissionDenied, isReachable, leaseKey, mintCreds, newIdentity, formatSecretStoreIdentity, parseSecretStoreIdentity, sameSecretStoreIdentity, standaloneConnectOpts, startTimerWriter, } from "@cotal-ai/core";
4
4
  import { PermissionViolationError } from "@nats-io/transport-node";
5
- import { DELIVERY_CREDS_KIND, DELIVERY_PIDFILE, FsSecretStore, authDir, canonicalLocalProcessPath, canonicalRoot, deliveryCredsKey, findCotalRoot, isWorkspaceTargetError, loadSpaceAuth, reclaimDeadPreUpgradeRecord, removeIdentityPin, resolveMeshTarget, segmentedKey, soleSpaceOf, spaceSegment, workspaceSecretStore, writeIdentityPin } from "@cotal-ai/workspace";
5
+ import { DELIVERY_CREDS_KIND, DELIVERY_PIDFILE, FsSecretStore, authDir, canonicalLocalProcessPath, canonicalRoot, deliveryCredsKey, findCotalRoot, isWorkspaceTargetError, loadSpaceAuth, reclaimDeadPreUpgradeRecord, removeIdentityPin, resolveMeshTarget, segmentedKey, soleSpaceOf, spaceSegment, workspaceSecretStore, writePidPair } from "@cotal-ai/workspace";
6
6
  import { startMembership } from "./membership.js";
7
- import { mayServeOn, brokerGoneVerdict, classifyProbe, DescheduleSampler, leaseAction, LoopLagMeter, PROBE_INTERVAL_MS, PROBE_LATE_FACTOR } from "./watchdog.js";
7
+ import { mayServeOn, leaseAction } from "./watchdog.js";
8
+ import { DeliveryTransportHealth } from "./transport-health.js";
8
9
  import { executeEviction, executePlaneLiveness, executePrincipalLiveness, validateScanTargetAdmission } from "./evict-exec.js";
9
10
  /** Re-exported for hosted compositions: the daemon cred's KIND, and the builder that turns it into
10
11
  * the {@link SecretStore} key for a space. Defined once in workspace (the layout is the workspace's)
@@ -258,9 +259,8 @@ export function recordDeliveryPid(root, space) {
258
259
  reclaimDeadPreUpgradeRecord(DELIVERY_PIDFILE, ctx);
259
260
  const pidPath = canonicalLocalProcessPath(DELIVERY_PIDFILE, ctx);
260
261
  const mine = String(process.pid);
261
- writeFileSync(pidPath, mine);
262
- // #969: pin the pid to its process start so a later teardown can refuse a reused pid.
263
- writeIdentityPin(pidPath, process.pid);
262
+ // #969/#1238: publish the pair by rename so a later teardown never sees a torn pairing.
263
+ writePidPair(pidPath, process.pid);
264
264
  return () => {
265
265
  try {
266
266
  if (readFileSync(pidPath, "utf8").trim() !== mine)
@@ -321,7 +321,26 @@ export async function runDelivery(args, store) {
321
321
  throw e;
322
322
  }
323
323
  }
324
- async function runStartedDelivery(args, store, publishReleaser) {
324
+ /** Start one independently owned delivery context. No cwd, process signal, pidfile, or exit
325
+ * policy is borrowed from the CLI runner. The caller owns its explicit state path and store. */
326
+ export async function startDeliveryService(inputs) {
327
+ if (!inputs.context.accountPublicKey || !inputs.context.lifecycleUid || !inputs.space || !inputs.servers || !inputs.stateDir)
328
+ throw new Error("delivery: hosted context needs an account, lifecycle, space, server and stateDir");
329
+ if (inputs.store.identity === undefined)
330
+ throw new Error("delivery: hosted SecretStore must declare a stable identity");
331
+ const actual = parseSecretStoreIdentity(inputs.store.identity);
332
+ if (!sameSecretStoreIdentity(actual, inputs.storeIdentity) || actual.kind !== "injected")
333
+ throw new Error("delivery: hosted SecretStore identity does not match the assigned store identity");
334
+ let releaseOnStartupFailure;
335
+ try {
336
+ return await runStartedDelivery({ values: { space: inputs.space, server: inputs.servers }, positionals: [], raw: [] }, inputs.store, (release) => { releaseOnStartupFailure = release; }, inputs);
337
+ }
338
+ catch (error) {
339
+ await releaseOnStartupFailure?.();
340
+ throw error;
341
+ }
342
+ }
343
+ async function runStartedDelivery(args, store, publishReleaser, hosted) {
325
344
  const v = args.values;
326
345
  const shard = v.shard ? Number(v.shard) : 0;
327
346
  const shards = v.shards ? Number(v.shards) : 1;
@@ -341,6 +360,8 @@ async function runStartedDelivery(args, store, publishReleaser) {
341
360
  if (v.creds !== undefined)
342
361
  assertUninjectedCredsSharesCwdRoot({ injected: credsSrc.injected, credsPath: resolve(v.creds) });
343
362
  const creds = await loadDeliveryCreds(credsSrc, v); // pre-minted scoped cred; NO signer/loadSpaceAuth in this path
363
+ if (hosted !== undefined && accountFromCreds(creds.initial) !== hosted.context.accountPublicKey)
364
+ throw new Error("delivery: scoped credential account does not match the assigned hosted context");
344
365
  let latestCreds = creds.initial; // freshest renewal — the broker-reachability poll below presents it
345
366
  // WHICH BROKER, WORKSTATION COMPOSITION (#756, the deliver half). `cotal ps`/`spawn`/`attach`/
346
367
  // `status`/`supervise` all resolve the registered mesh for the space/root and dial its broker;
@@ -407,9 +428,13 @@ async function runStartedDelivery(args, store, publishReleaser) {
407
428
  console.error(registered.origin === "manual" || registered.origin === "catalog"
408
429
  ? `✗ delivery: no broker answered at ${server} - "${space}" is registered here but its mesh is not up; start it where it runs, or \`cotal meshes rm ${space}\` to unregister it`
409
430
  : `✗ delivery: no mesh running at ${server} - mesh "${space}" is recorded at ${registered.root} but not running; run \`cotal up\` there to restart`);
431
+ if (hosted !== undefined)
432
+ throw new Error(`delivery: no broker answered at ${server}`);
410
433
  process.exit(1);
411
434
  }
412
435
  console.error(`✗ delivery: can't reach NATS at ${server}. Run: cotal up`);
436
+ if (hosted !== undefined)
437
+ throw new Error(`delivery: can't reach NATS at ${server}`);
413
438
  process.exit(1);
414
439
  }
415
440
  // PIN THE SCAN TARGET ONCE, HERE. The $SYS sweeps below resolve an ACCOUNT, and a complete sweep
@@ -424,7 +449,7 @@ async function runStartedDelivery(args, store, publishReleaser) {
424
449
  // is known for certain, rather than inferred later by probing the store or sniffing the
425
450
  // filesystem, both of which report "workstation" for a hosted daemon and would emit a CLI repair
426
451
  // the host cannot run. It selects the REPAIR IDIOM only; failure semantics never fork on it.
427
- const scanRoot = findCotalRoot();
452
+ const scanRoot = hosted?.stateDir ?? findCotalRoot();
428
453
  const scanTarget = {
429
454
  root: scanRoot,
430
455
  expectedAccount: accountFromCreds(creds.initial),
@@ -456,6 +481,8 @@ async function runStartedDelivery(args, store, publishReleaser) {
456
481
  watchPresence: true, // read the roster for @mention resolution …
457
482
  registerPresence: false, // … but NEVER publish the daemon onto the roster (it's infra, not a peer)
458
483
  card: { id: ownId, name: "delivery", role: "delivery", kind: "endpoint" },
484
+ transportPingIntervalMs: Number(process.env.COTAL_DELIVERY_PING_INTERVAL_MS) || 2500,
485
+ transportMaxPingOut: Number(process.env.COTAL_DELIVERY_MAX_PING_OUT) || 2,
459
486
  });
460
487
  // Both channels: raw connection errors ride `error`, while every condition the endpoint is
461
488
  // already surviving — a failed 75% renewal, the passive backstop's "still holds the previous
@@ -464,7 +491,49 @@ async function runStartedDelivery(args, store, publishReleaser) {
464
491
  const say = (e) => console.error(`! delivery endpoint: ${e.message}`);
465
492
  ep.on("error", say);
466
493
  ep.on("warning", say);
467
- await ep.start();
494
+ try {
495
+ await ep.start();
496
+ }
497
+ catch (error) {
498
+ await ep.stop();
499
+ throw error;
500
+ }
501
+ const BROKER_GONE_MS = Number(process.env.COTAL_DELIVERY_BROKER_GONE_MS) || 15_000;
502
+ const BROKER_GONE_BACKSTOP_MS = Math.max(BROKER_GONE_MS, Number(process.env.COTAL_DELIVERY_BROKER_GONE_BACKSTOP_MS) || BROKER_GONE_MS * 4);
503
+ let stopHandler;
504
+ let hostedStartupFailure;
505
+ let stopHosted;
506
+ const health = new DeliveryTransportHealth((reason) => {
507
+ console.error(reason === "credential-expired"
508
+ ? "✗ delivery: credential expired without renewal"
509
+ : reason === "backstop"
510
+ ? "✗ delivery: broker connection unavailable past backstop (broker unreachable), exiting (coupled to the broker)"
511
+ : "✗ delivery: broker connection unavailable (broker unreachable), exiting (coupled to the broker)");
512
+ const cause = reason === "credential-expired" ? "credential expired without renewal" : `broker connection unavailable (${reason})`;
513
+ if (hosted !== undefined) {
514
+ hostedStartupFailure ??= new Error(`delivery context stopped during startup: ${cause}`);
515
+ if (stopHosted !== undefined)
516
+ stopHosted(cause);
517
+ else
518
+ void ep.stop(); // before lease acquisition, no context-owned lease can be released
519
+ return;
520
+ }
521
+ if (stopHandler !== undefined)
522
+ stopHandler(1, cause);
523
+ else
524
+ earlyStop(1);
525
+ }, () => console.error("! delivery: credential expired, awaiting proved renewal"), BROKER_GONE_MS, BROKER_GONE_BACKSTOP_MS);
526
+ ep.on("transport", ({ connected }) => {
527
+ health.transport(connected);
528
+ });
529
+ ep.on("error", (e) => {
530
+ if (e.name === "UserAuthenticationExpiredError" || e.cause?.name === "UserAuthenticationExpiredError") {
531
+ health.credentialExpired();
532
+ }
533
+ });
534
+ ep.on("creds-adopted", () => {
535
+ health.adopted();
536
+ });
468
537
  // Acquire the single-flight lease BEFORE binding the loops: a loud refusal-to-bind if another daemon
469
538
  // already holds this shard (two clients binding the same durable name SPLIT delivery). The bucket TTL
470
539
  // frees a crashed holder's lease so a fresh daemon re-acquires.
@@ -498,6 +567,8 @@ async function runStartedDelivery(args, store, publishReleaser) {
498
567
  console.error(`✗ delivery: acquiring the lease for shard ${shard} failed: ${e.message}. Not binding.`);
499
568
  }
500
569
  await ep.stop();
570
+ if (hosted !== undefined)
571
+ throw e;
501
572
  process.exit(1);
502
573
  return;
503
574
  }
@@ -521,6 +592,13 @@ async function runStartedDelivery(args, store, publishReleaser) {
521
592
  // it is provably ours, then exit. Once the real `shutdown` is installed it REPLACES this, and
522
593
  // `stopping` makes the pair idempotent, so a signal arriving mid-swap cannot run both.
523
594
  let stopping = false;
595
+ let unavailable;
596
+ let closePromise;
597
+ let membership;
598
+ let timerWriter;
599
+ let timerWriterAttempt;
600
+ let stopLeaseWatch;
601
+ let renew;
524
602
  /** Removes THIS process's liveness record, once it has one. Declared BEFORE the handlers that
525
603
  * call it and assigned below, so a signal landing between the two is a no-op rather than a
526
604
  * temporal-dead-zone throw: at that instant nothing has been written, so there is nothing to
@@ -531,6 +609,7 @@ async function runStartedDelivery(args, store, publishReleaser) {
531
609
  if (stopping)
532
610
  return;
533
611
  stopping = true;
612
+ health.stop();
534
613
  setTimeout(() => process.exit(code), 2000);
535
614
  unrecordPid?.();
536
615
  void (async () => {
@@ -547,10 +626,13 @@ async function runStartedDelivery(args, store, publishReleaser) {
547
626
  process.exit(code);
548
627
  })();
549
628
  };
629
+ stopHandler = earlyStop;
550
630
  const earlySigint = () => earlyStop(0);
551
631
  const earlySigterm = () => earlyStop(0);
552
- process.on("SIGINT", earlySigint);
553
- process.on("SIGTERM", earlySigterm);
632
+ if (hosted === undefined) {
633
+ process.on("SIGINT", earlySigint);
634
+ process.on("SIGTERM", earlySigterm);
635
+ }
554
636
  // THE LIVENESS RECORD GOES HERE, AFTER THE ACQUIRE AND NOT BEFORE IT (#1528).
555
637
  //
556
638
  // The record answers "which process is this space's delivery daemon", and until this line this
@@ -566,7 +648,8 @@ async function runStartedDelivery(args, store, publishReleaser) {
566
648
  // own `ready` flag already carries it; the pidfile only ever claimed a process.
567
649
  //
568
650
  // Placed one statement after the signal handlers, so every exit path from here on can remove it.
569
- unrecordPid = recordDeliveryPid(findCotalRoot(), space);
651
+ if (hosted === undefined)
652
+ unrecordPid = recordDeliveryPid(findCotalRoot(), space);
570
653
  // AND THE SAME RELEASE, REACHABLE BY AN ORDINARY `catch`. The two process-level guards below
571
654
  // only see a fault that reaches the RUNTIME. A start-up rejection on the public CLI path does
572
655
  // not: `runCli` awaits this function inside its own try/catch (cli/src/command.ts), so the
@@ -577,22 +660,58 @@ async function runStartedDelivery(args, store, publishReleaser) {
577
660
  // exists". So the release is published HERE, to a scope that a plain `catch` around the rest of
578
661
  // start-up can reach, and the process guards stay as the backstop for faults that never become
579
662
  // a rejection this function can see (a synchronous throw in a timer, say).
580
- publishReleaser(async () => {
581
- if (stopping)
582
- return;
663
+ const close = () => {
664
+ if (closePromise !== undefined)
665
+ return closePromise;
583
666
  stopping = true;
584
- unrecordPid?.(); // the record dies with the daemon it describes, including on a start-up failure
585
- try {
586
- const own = await ep.readDeliveryLeaseEntry(shard);
587
- if (own !== undefined && ep.ownsDeliveryLease(own.info))
588
- await ep.releaseDeliveryLease(shard, own.revision);
589
- }
590
- catch { /* broker may be gone - the bucket TTL is the crash-safe release authority */ }
591
- try {
592
- await ep.stop();
593
- }
594
- catch { /* broker may be gone */ }
595
- });
667
+ if (renew !== undefined)
668
+ clearInterval(renew);
669
+ health.stop();
670
+ stopLeaseWatch?.();
671
+ unrecordPid?.();
672
+ closePromise = (async () => {
673
+ try {
674
+ await ep.quiescePlane3();
675
+ }
676
+ catch { /* stop still releases the connection */ }
677
+ try {
678
+ const own = await ep.readDeliveryLeaseEntry(shard);
679
+ if (own !== undefined && ep.ownsDeliveryLease(own.info))
680
+ await ep.releaseDeliveryLease(shard, own.revision);
681
+ }
682
+ catch { /* broker may be gone - the bucket TTL is the crash-safe release authority */ }
683
+ try {
684
+ await membership?.stop();
685
+ }
686
+ catch { /* broker may be gone */ }
687
+ try {
688
+ await timerWriter?.handle.stop();
689
+ }
690
+ catch { /* broker may be gone */ }
691
+ try {
692
+ await Promise.race([timerWriter?.nc.drain(), new Promise((r) => setTimeout(r, 1000))]);
693
+ }
694
+ catch { /* broker may be gone */ }
695
+ try {
696
+ await timerWriterAttempt;
697
+ }
698
+ catch { /* startup or broker fault */ }
699
+ try {
700
+ await ep.stop();
701
+ }
702
+ catch { /* broker may be gone */ }
703
+ })();
704
+ return closePromise;
705
+ };
706
+ publishReleaser(close);
707
+ stopHosted = (cause) => {
708
+ unavailable = `delivery context stopped (code 1): ${cause}`;
709
+ void close();
710
+ };
711
+ if (hostedStartupFailure !== undefined) {
712
+ await close();
713
+ throw hostedStartupFailure;
714
+ }
596
715
  // A START-UP FAILURE AFTER THE ACQUIRE MUST ALSO GIVE THE SHARD BACK. Between this point and the
597
716
  // handler swap far below, a throw would otherwise propagate out of `runDelivery` with the row
598
717
  // still claiming the shard, stranding it for the bucket TTL exactly as an unhandled signal did -
@@ -618,11 +737,12 @@ async function runStartedDelivery(args, store, publishReleaser) {
618
737
  };
619
738
  const earlyUncaught = (e) => earlyFault("faulted", e);
620
739
  const earlyRejection = (e) => earlyFault("rejected a promise", e);
621
- process.on("uncaughtException", earlyUncaught);
622
- process.on("unhandledRejection", earlyRejection);
740
+ if (hosted === undefined) {
741
+ process.on("uncaughtException", earlyUncaught);
742
+ process.on("unhandledRejection", earlyRejection);
743
+ }
623
744
  // Broker-sourced graph membership handle — declared BEFORE Plane-3 so the delivery-admin reload
624
745
  // hook below can close over it (it starts further down; the closure reads it live).
625
- let membership;
626
746
  // WHY the feed is down, carried from `startMembership` so the adoption refusal below names the real
627
747
  // fault instead of its symptom. Without it, an expired $SYS observer cred surfaces to the operator
628
748
  // only as "membership feed is not running" (#338).
@@ -655,6 +775,7 @@ async function runStartedDelivery(args, store, publishReleaser) {
655
775
  // killing it to discover it was alive (any refusal/unknown blocks the repair, fail-closed).
656
776
  principalLiveness: (principal) => executePrincipalLiveness(server, scanTarget, principal),
657
777
  reloadStoreIdentity: () => reloadStoreIdentity,
778
+ onDeliveryCredsAdopted: () => health.adopted(),
658
779
  });
659
780
  // Flip the lease to READY only now — after the loops + ctl.delivery responder are bound — so readiness
660
781
  // waiters (ensureDelivery) and the cotal_channels health surface see "ready" iff the responder is up,
@@ -662,7 +783,15 @@ async function runStartedDelivery(args, store, publishReleaser) {
662
783
  try {
663
784
  revision = await ep.markDeliveryLeaseReady(shard, revision);
664
785
  }
665
- catch { /* lost the lease between acquire and ready — the renew loop's CAS failure will exit us */ }
786
+ catch (error) {
787
+ if (hosted !== undefined)
788
+ throw error;
789
+ /* CLI: the renew loop's CAS failure will exit us if the lease was lost. */
790
+ }
791
+ if (hostedStartupFailure !== undefined) {
792
+ await close();
793
+ throw hostedStartupFailure;
794
+ }
666
795
  console.log(`✓ delivery daemon up (space ${space}${shards > 1 ? `, shard ${shard}/${shards}` : ""}) — stop with: cotal down`);
667
796
  // Broker-sourced graph membership: a SEPARATE module on its OWN connections (system-account CONNZ
668
797
  // reader + data-account feed writer), isolated from Plane-3. Fail-soft — a missing cred / start error
@@ -681,6 +810,17 @@ async function runStartedDelivery(args, store, publishReleaser) {
681
810
  membershipDown = e.message;
682
811
  console.error(`! membership: failed to start (${membershipDown}); graph membership degraded, delivery unaffected`);
683
812
  }
813
+ // A store read can finish after a startup health fault closed the endpoint. Dispose the feed
814
+ // it just returned before rejecting; it did not exist when the original close ran.
815
+ if (hostedStartupFailure !== undefined) {
816
+ try {
817
+ await membership?.stop();
818
+ }
819
+ finally {
820
+ await close();
821
+ }
822
+ throw hostedStartupFailure;
823
+ }
684
824
  // The TIMER WRITER (SPEC 13.2): the pump that turns workflow `.schedule` requests into armed
685
825
  // broker schedules. Hosted here because this daemon is the space's standing server-side process;
686
826
  // without a running writer no pause on the space ever expires. Its OWN connection under the same
@@ -688,8 +828,7 @@ async function runStartedDelivery(args, store, publishReleaser) {
688
828
  // module-per-connection isolation), re-dialed with the FRESHEST cred by a supervised restart
689
829
  // loop: a fault never kills delivery, is never silent (each attempt logs why the space cannot
690
830
  // expire pauses right now), and a cred that gains rows at renewal is picked up on the next dial.
691
- let timerWriter;
692
- void (async () => {
831
+ timerWriterAttempt = (async () => {
693
832
  let backoffMs = 5_000;
694
833
  for (;;) {
695
834
  if (stopping)
@@ -702,7 +841,16 @@ async function runStartedDelivery(args, store, publishReleaser) {
702
841
  name: "cotal-delivery-timer-writer",
703
842
  maxReconnectAttempts: -1,
704
843
  });
844
+ if (stopping) {
845
+ await nc.close();
846
+ return;
847
+ }
705
848
  const handle = await startTimerWriter(nc, space);
849
+ if (stopping) {
850
+ await handle.stop();
851
+ await nc.close();
852
+ return;
853
+ }
706
854
  timerWriter = { handle, nc };
707
855
  backoffMs = 5_000;
708
856
  console.log(`✓ timer writer up (space ${space}) — workflow pauses on this space expire`);
@@ -710,8 +858,13 @@ async function runStartedDelivery(args, store, publishReleaser) {
710
858
  return;
711
859
  }
712
860
  catch (e) {
713
- if (stopping)
861
+ if (stopping) {
862
+ try {
863
+ await nc?.close();
864
+ }
865
+ catch { /* already closed */ }
714
866
  return;
867
+ }
715
868
  console.error(`! timer writer: down (${e.message}) — retrying in ${Math.round(backoffMs / 1000)}s; pauses on this space do not expire until it is back`);
716
869
  try {
717
870
  await nc?.drain();
@@ -723,13 +876,19 @@ async function runStartedDelivery(args, store, publishReleaser) {
723
876
  }
724
877
  }
725
878
  })();
726
- const shutdown = (code) => {
879
+ const shutdown = (code, cause) => {
880
+ if (hosted !== undefined) {
881
+ unavailable = cause ? `delivery context stopped (code ${code}): ${cause}` : `delivery context stopped (code ${code})`;
882
+ void close();
883
+ return;
884
+ }
727
885
  if (stopping)
728
886
  return;
729
887
  stopping = true;
730
- clearInterval(renew);
731
- clearInterval(brokerWatch);
732
- stopLeaseWatch();
888
+ if (renew !== undefined)
889
+ clearInterval(renew);
890
+ health.stop();
891
+ stopLeaseWatch?.();
733
892
  // THE RECORD GOES FIRST, AND SYNCHRONOUSLY. Everything below this line talks to a broker that
734
893
  // may be dead and is bounded only by the 2s hard exit; a record left behind because a drain
735
894
  // hung is a record that outlives its process, which is this issue's defect re-entering through
@@ -797,6 +956,7 @@ async function runStartedDelivery(args, store, publishReleaser) {
797
956
  process.exit(code);
798
957
  })();
799
958
  };
959
+ stopHandler = shutdown;
800
960
  // SIGNAL HANDLERS GO UP THE MOMENT `shutdown` EXISTS, NOT AFTER STARTUP FINISHES.
801
961
  //
802
962
  // The lease is acquired long before this point, and registration used to sit at the very end of
@@ -818,16 +978,18 @@ async function runStartedDelivery(args, store, publishReleaser) {
818
978
  // knows nothing of the renew timer, the membership feed or the timer writer - would run first and
819
979
  // set `stopping`, making the full teardown a no-op. `stopping` still guards the swap itself, so a
820
980
  // signal delivered between these two statements is handled exactly once.
821
- process.off("SIGINT", earlySigint);
822
- process.off("SIGTERM", earlySigterm);
823
- // The start-up fault guards go too, and BY REFERENCE: `removeAllListeners` here would strip
824
- // listeners this daemon does not own. They exist to cover the window where no `shutdown` exists;
825
- // past this line a fault should surface normally rather than become a quiet exit that skips the
826
- // full teardown.
827
- process.off("uncaughtException", earlyUncaught);
828
- process.off("unhandledRejection", earlyRejection);
829
- process.on("SIGINT", () => shutdown(0));
830
- process.on("SIGTERM", () => shutdown(0));
981
+ if (hosted === undefined) {
982
+ process.off("SIGINT", earlySigint);
983
+ process.off("SIGTERM", earlySigterm);
984
+ // The start-up fault guards go too, and BY REFERENCE: `removeAllListeners` here would strip
985
+ // listeners this daemon does not own. They exist to cover the window where no `shutdown` exists;
986
+ // past this line a fault should surface normally rather than become a quiet exit that skips the
987
+ // full teardown.
988
+ process.off("uncaughtException", earlyUncaught);
989
+ process.off("unhandledRejection", earlyRejection);
990
+ process.on("SIGINT", () => shutdown(0));
991
+ process.on("SIGTERM", () => shutdown(0));
992
+ }
831
993
  /** What the broker says about THIS shard's lease key right now, the verdict a failed renew does
832
994
  * NOT have. `unknown` never collapses into `gone`: not being able to look is not the same fact as
833
995
  * looking and finding nothing (#1318).
@@ -896,6 +1058,8 @@ async function runStartedDelivery(args, store, publishReleaser) {
896
1058
  * caller has just proven ownership; `why` is what proved it, since "it started serving again" is
897
1059
  * only auditable next to the evidence that permitted it. */
898
1060
  const resumeServing = async (why) => {
1061
+ if (stopping)
1062
+ return;
899
1063
  try {
900
1064
  await ep.rearmPlane3();
901
1065
  }
@@ -950,7 +1114,7 @@ async function runStartedDelivery(args, store, publishReleaser) {
950
1114
  // establish it refuses to start, loudly (the await below throws out of start-up and the
951
1115
  // releaser gives the shard back), rather than degrade into the polled window this issue is
952
1116
  // about. The stop handle is cleared on every exit path the renew interval is: `shutdown`.
953
- const stopLeaseWatch = await ep.watchDeliveryLease(shard, (info) => {
1117
+ stopLeaseWatch = await ep.watchDeliveryLease(shard, (info) => {
954
1118
  if (stopping)
955
1119
  return;
956
1120
  // The row as this event states it: an event for a row we own answers ITSELF (no state
@@ -1009,6 +1173,11 @@ async function runStartedDelivery(args, store, publishReleaser) {
1009
1173
  }
1010
1174
  })().catch((e) => console.error(`! delivery: the lease watch trigger faulted (${e.message}); the renew interval remains the arbiter`));
1011
1175
  });
1176
+ if (hosted !== undefined && stopping) {
1177
+ stopLeaseWatch?.();
1178
+ await close();
1179
+ throw hostedStartupFailure ?? new Error(unavailable ?? "delivery context stopped during startup");
1180
+ }
1012
1181
  // Renew the lease at ~half the TTL so a healthy holder never self-evicts.
1013
1182
  //
1014
1183
  // A FAILED RENEW IS A QUESTION, NOT A VERDICT (#1318). The shipped code exited on ANY renew
@@ -1021,7 +1190,7 @@ async function runStartedDelivery(args, store, publishReleaser) {
1021
1190
  // second refused over a sequence the first legitimately moved, a conflict this daemon manufactures
1022
1191
  // itself and then reads as someone else's takeover.
1023
1192
  let renewInFlight = false;
1024
- const renew = setInterval(() => {
1193
+ renew = setInterval(() => {
1025
1194
  if (stopping || renewInFlight)
1026
1195
  return;
1027
1196
  renewInFlight = true;
@@ -1126,187 +1295,19 @@ async function runStartedDelivery(args, store, publishReleaser) {
1126
1295
  }
1127
1296
  })();
1128
1297
  }, Math.max(1000, Math.floor(LEASE_TTL_MS / 2)));
1129
- // Coupled to the broker: POLL its reachability. Survive brief blips (the endpoint reconnects on its
1130
- // own), but EXIT if the broker is GONE, the endpoint would otherwise retry reconnect forever (its
1131
- // terminal-close never fires), so this is what stops the daemon outliving the server it serves.
1132
- // (`cotal up`/`down` teardown stops it too.) The window is env-overridable for tests.
1133
- //
1134
- // WHAT "GONE" MEANS IS NOW DECIDED FROM EVIDENCE, NOT FROM A CLOCK (#1318). Three conditions used
1135
- // to produce one signal and only one of them was a dead server: the broker being down, this
1136
- // process being descheduled so the interval never fired, and a probe that could not complete a
1137
- // handshake because the local process could not get scheduled to finish it. A wall-clock
1138
- // `Date.now() - lastReachable` cannot tell them apart, and widening it only moves the threshold.
1139
- // Three signals separate them, and all three are things this process can actually observe:
1140
- //
1141
- // • MEASURED LOOP LAG. The gap between consecutive firings of a timer we own, minus its nominal
1142
- // period, is local starvation by direct measurement. It is credited back, so time this process
1143
- // spent off the runqueue is never counted against the broker.
1144
- // • COMPLETED NEGATIVE PROBES. Only a probe that RAN TO COMPLETION and returned false is
1145
- // evidence about the server. A probe that never ran contributes nothing (that was the whole
1146
- // defect: the window aged with no probe having failed), and one that REJECTED is an
1147
- // unanswered question, it now resets the counter and logs, where it used to be swallowed by
1148
- // `.catch(() => {})` and silently age the window.
1149
- // • TRANSPORT-LEVEL LIVENESS. A `transport: connected` edge from the endpoint's own connection
1150
- // cannot happen without a server on the other end, so it is positive evidence obtained for
1151
- // free, on a path that does not need this process to schedule a probe at all.
1152
- //
1153
- // A starvation diagnosis therefore reports DEGRADED and keeps serving; a genuinely dead broker
1154
- // still exits, on the same window, as fast as completed probes can say so.
1155
- const BROKER_GONE_MS = Number(process.env.COTAL_DELIVERY_BROKER_GONE_MS) || 15_000;
1156
- // How many completed negatives make elapsed time believable. Two is the floor: one completed
1157
- // negative is a single refused connect, which a loopback under momentary pressure can produce.
1158
- // The default scales with the window so a test that shortens the window does not thereby demand
1159
- // more evidence than the window has room for.
1160
- const BROKER_GONE_PROBES = Math.max(2, Number(process.env.COTAL_DELIVERY_BROKER_GONE_PROBES) || Math.ceil(BROKER_GONE_MS / PROBE_INTERVAL_MS / 2));
1161
- // The hard backstop: past this, no amount of measured lag or transport optimism keeps the daemon
1162
- // alive. A daemon that OUTLIVES a dead broker is worse than one that restarts unnecessarily, so
1163
- // the repair is bounded and fails toward exiting.
1164
- const BROKER_GONE_BACKSTOP_MS = Math.max(BROKER_GONE_MS, Number(process.env.COTAL_DELIVERY_BROKER_GONE_BACKSTOP_MS) || BROKER_GONE_MS * 4);
1165
- let lastReachable = Date.now();
1166
- let completedNegatives = 0;
1167
- let degraded = false;
1168
- // What the endpoint's OWN connection reports about its socket to this same broker. Seeded true
1169
- // because `ep.start()` above completed, which it cannot do without a server having answered.
1170
- let transportConnected = true;
1171
- const lag = new LoopLagMeter(PROBE_INTERVAL_MS);
1172
- /** Positive evidence, from wherever it came: restart the window and its lag budget together. */
1173
- const sawBroker = () => {
1174
- lastReachable = Date.now();
1175
- completedNegatives = 0;
1176
- lag.reset();
1177
- if (degraded) {
1178
- degraded = false;
1179
- console.error(`✓ delivery: the broker is answering again, Plane-3 is serving normally (space ${space})`);
1180
- }
1181
- };
1182
- // The endpoint's OWN transport edge. This is evidence the daemon gets without being scheduled to
1183
- // probe for it, and it is the signal that distinguishes "the connection object reports a
1184
- // transport-level close" from "silence": a live connection to that address proves a server, while
1185
- // a disconnect is the endpoint's own business (nats.js reconnects through blips of its own
1186
- // accord) and merely stops excusing a probe that will not complete.
1187
- ep.on("transport", (t) => {
1188
- transportConnected = t.connected;
1189
- if (t.connected)
1190
- sawBroker();
1191
- });
1192
- // THE BUDGET THE PROBE IS ACTUALLY GIVEN, asked of the one function that decides it. A ws(s)
1193
- // broker rides an HTTPS edge and gets 5s, not the 1s a loopback TCP broker gets; judging a ws
1194
- // probe against 1000 would read every honest refusal on such a broker as this process's own
1195
- // starvation, so the completed-negative count could never rise and a genuinely dead ws broker
1196
- // would be ended only by the backstop, with the wrong reason in its log. The budget and the
1197
- // judgment have to come from the same place.
1198
- const probeBudgetMs = defaultProbeTimeoutMs(server);
1199
- const brokerWatch = setInterval(() => {
1200
- if (stopping)
1201
- return;
1202
- // Measure FIRST, before any await: this is the gap since the previous firing, and it is the
1203
- // only place the daemon can learn that it was not scheduled.
1204
- lag.tick(Date.now());
1205
- // THE SAME TRANSPORT AS EVERY OTHER DIAL IN THIS PROCESS. This poll carries `latestCreds` — a
1206
- // standing credential — and `isReachable` performs a real authenticated connect whenever creds
1207
- // are supplied, every 2 seconds, for the life of the daemon. It is the most repeated credential
1208
- // presentation in the system, and it was the one dial here that did not name its transport.
1209
- //
1210
- // The poll predates TLS and is identical on `main`, where nothing is encrypted and it is at
1211
- // least consistent. What is new is the ASYMMETRY: with the two dials above upgraded, an operator
1212
- // who enables TLS would get a protected main path and an unprotected watchdog. An inconsistent
1213
- // guarantee is worse than a uniformly absent one, because the operator now believes something.
1214
- const probeStarted = Date.now();
1215
- // Watch this process's own scheduling FOR THE DURATION OF THE PROBE. A refusal is only evidence
1216
- // about the server if the server was actually given the time the deadline promised it, and
1217
- // under short CPU slices most of a probe's wall-clock can be time this process was not running.
1218
- const sampler = new DescheduleSampler();
1219
- sampler.start(probeStarted);
1220
- void isReachable(server, { creds: latestCreds, ...(tls ? { tls: true } : {}) })
1221
- .then((ok) => classifyProbe(ok, Date.now() - probeStarted, probeBudgetMs, PROBE_LATE_FACTOR, sampler.stop()),
1222
- // A REJECTED probe is an unanswered question, not a negative answer. It used to be swallowed
1223
- // whole by `.catch(() => {})`, so it neither refreshed the window nor evaluated anything and
1224
- // silently aged the daemon toward an exit it had gathered no evidence for.
1225
- (e) => {
1226
- sampler.stop();
1227
- console.error(`! delivery: the broker probe did not complete (${e.message}), no verdict from it; serving, retrying`);
1228
- return classifyProbe(undefined, Date.now() - probeStarted);
1229
- })
1230
- .then((probe) => {
1231
- if (stopping)
1232
- return;
1233
- if (probe.counts === "positive") {
1234
- sawBroker();
1235
- return;
1236
- }
1237
- if (probe.counts === "incomplete") {
1238
- // The run of credible refusals is broken by a probe that did not complete.
1239
- completedNegatives = 0;
1240
- return;
1241
- }
1242
- if (probe.counts === "starved") {
1243
- // The probe RAN and said no, but its answer arrived so far past its own deadline that the
1244
- // deadline was enforced against this process rather than against the server. That is the
1245
- // starved-client case, and it is the one an elapsed-time predicate cannot see at all: the
1246
- // timer is firing, the probes are completing, and every one of them is `false`.
1247
- // Dated with the instant the answer ARRIVED, so the meter can tell whether this stall is
1248
- // the same wall-clock interval the tick above already charged (union, counted once) or an
1249
- // adjacent one (disjoint, both counted). The overlap question is settled by timestamps
1250
- // rather than by comparing magnitudes, which cannot distinguish the two.
1251
- lag.credit(probe.lateBy, Date.now());
1252
- completedNegatives = 0;
1253
- }
1254
- else {
1255
- // A refusal that arrived on time. This is the only thing that may accrue against the broker.
1256
- completedNegatives += 1;
1257
- }
1258
- // ONE SPAN, MEASURED ONCE, USED FOR BOTH. Reading the clock twice would let the elapsed span
1259
- // and the lag clamp disagree by the cost of the call itself.
1260
- const sinceReachable = Date.now() - lastReachable;
1261
- const verdict = brokerGoneVerdict({
1262
- msSinceLastReachable: sinceReachable,
1263
- // CLAMPED TO THE SPAN IT IS SUBTRACTED FROM. The interval gap and a late probe answer can
1264
- // testify to the same stall, and the meter cannot tell that from two adjacent stalls, so it
1265
- // keeps both charges and the bound is applied here, where the span is known. Without it a
1266
- // 10s stall read 17s and drove unstarved time negative, which cannot be cleared by any
1267
- // amount of real outage: a dead broker then survives the evidence clause and exits only on
1268
- // the backstop, 44s -> 62s measured. Time not had cannot exceed time passed.
1269
- starvedMs: lag.starvedMsWithin(sinceReachable),
1270
- completedNegatives,
1271
- transportConnected,
1272
- windowMs: BROKER_GONE_MS,
1273
- requiredNegatives: BROKER_GONE_PROBES,
1274
- backstopMs: BROKER_GONE_BACKSTOP_MS,
1275
- });
1276
- if (verdict.exit) {
1277
- // SAY WHICH EXIT THIS IS. The single message here previously claimed completed probes over
1278
- // ">15s of unstarved time" on BOTH paths, and on the backstop path that sentence is simply
1279
- // false: the backstop fires precisely when the evidence clauses did NOT conclude, often
1280
- // with zero completed refusals. An operator debugging a starved host was handed a
1281
- // confident evidentiary claim the daemon had not established.
1282
- console.error(verdict.reason === "backstop"
1283
- ? `✗ delivery: giving up on elapsed time alone, ${Math.round(BROKER_GONE_BACKSTOP_MS / 1000)}s since the last ` +
1284
- `confirmed reachability with no sufficient evidence either way (${completedNegatives} completed probes refused, ` +
1285
- `${Math.round(lag.starvedMsWithin(sinceReachable) / 1000)}s of local scheduler lag credited). This is a BOUND, not a diagnosis: the ` +
1286
- `broker may be gone or this process may have been starved past the bound, exiting (coupled to the broker)`
1287
- : `✗ delivery: broker unreachable, ${completedNegatives} completed probes refused within their deadline over ` +
1288
- `>${BROKER_GONE_MS / 1000}s of unstarved time (${Math.round(lag.starvedMsWithin(sinceReachable) / 1000)}s of local scheduler lag credited), exiting (coupled to the broker)`);
1289
- shutdown(1);
1290
- return;
1291
- }
1292
- // NOT an exit. Say so once per episode, naming WHICH condition it is, so an operator reading
1293
- // the log during a load incident sees "this host starved me" rather than a daemon that
1294
- // silently vanished.
1295
- if (!degraded && verdict.reason !== "reachable") {
1296
- degraded = true;
1297
- console.error(verdict.reason === "starved"
1298
- ? `! delivery: DEGRADED, cannot reach the broker, but ${Math.round(lag.starvedMsWithin(sinceReachable) / 1000)}s of that window was local scheduler lag (this host is starving this process, not the broker); serving, retrying`
1299
- : verdict.reason === "transport-live"
1300
- ? `! delivery: DEGRADED, a fresh probe cannot complete, but this daemon's own connection to ${server} is still open, so the broker is there and this process cannot ask; serving, retrying`
1301
- : `! delivery: DEGRADED, the broker has not answered for >${BROKER_GONE_MS / 1000}s but only ${completedNegatives} of ${BROKER_GONE_PROBES} probes have refused within their deadline; serving, retrying`);
1302
- }
1303
- })
1304
- .catch((e) => {
1305
- // The verdict path itself faulted. Never silent, and never an exit: a bug in the detector
1306
- // must not end the daemon it is meant to keep alive.
1307
- console.error(`! delivery: the broker watch tick faulted (${e.message}); serving, retrying`);
1308
- });
1309
- }, PROBE_INTERVAL_MS);
1310
- await new Promise(() => { }); // run until signalled
1298
+ // Native transport health is driven by DeliveryTransportHealth above (resident transport events).
1299
+ if (hosted !== undefined) {
1300
+ const context = hosted.context;
1301
+ return {
1302
+ readiness() {
1303
+ return unavailable !== undefined
1304
+ ? { state: "unavailable", context, cause: unavailable }
1305
+ : { state: stopping ? "draining" : "ready", context };
1306
+ },
1307
+ async drain() { await close(); },
1308
+ close,
1309
+ };
1310
+ }
1311
+ await new Promise(() => { }); // CLI runs until signalled
1311
1312
  }
1312
1313
  //# sourceMappingURL=delivery.js.map