@cotal-ai/delivery 0.49.0 → 0.50.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/delivery.js CHANGED
@@ -1,8 +1,9 @@
1
- import { existsSync } from "node:fs";
1
+ import { existsSync, readFileSync, rmSync, writeFileSync } from "node:fs";
2
2
  import { basename, dirname, join, resolve } from "node:path";
3
- import { CotalEndpoint, DEFAULT_SERVER, LEASE_TTL_MS, accountFromCreds, credsClaims, dialerFor, idFromCreds, isReachable, mintCreds, newIdentity, formatSecretStoreIdentity, parseSecretStoreIdentity, sameSecretStoreIdentity, standaloneConnectOpts, startTimerWriter, } from "@cotal-ai/core";
4
- import { DELIVERY_CREDS_KIND, FsSecretStore, authDir, deliveryCredsKey, findCotalRoot, loadSpaceAuth, segmentedKey, soleSpaceOf, spaceSegment, workspaceSecretStore } from "@cotal-ai/workspace";
3
+ import { CotalEndpoint, DEFAULT_SERVER, LEASE_TTL_MS, accountFromCreds, credsClaims, dialerFor, idFromCreds, defaultProbeTimeoutMs, isReachable, mintCreds, newIdentity, formatSecretStoreIdentity, parseSecretStoreIdentity, sameSecretStoreIdentity, standaloneConnectOpts, startTimerWriter, } from "@cotal-ai/core";
4
+ import { DELIVERY_CREDS_KIND, DELIVERY_PIDFILE, FsSecretStore, authDir, canonicalLocalProcessPath, deliveryCredsKey, findCotalRoot, loadSpaceAuth, reclaimDeadPreUpgradeRecord, removeIdentityPin, segmentedKey, soleSpaceOf, spaceSegment, workspaceSecretStore, writeIdentityPin } from "@cotal-ai/workspace";
5
5
  import { startMembership } from "./membership.js";
6
+ import { mayServeOn, brokerGoneVerdict, classifyProbe, DescheduleSampler, leaseAction, LoopLagMeter, PROBE_INTERVAL_MS, PROBE_LATE_FACTOR } from "./watchdog.js";
6
7
  import { executeEviction, executePlaneLiveness, executePrincipalLiveness, validateScanTargetAdmission } from "./evict-exec.js";
7
8
  /** Re-exported for hosted compositions: the daemon cred's KIND, and the builder that turns it into
8
9
  * the {@link SecretStore} key for a space. Defined once in workspace (the layout is the workspace's)
@@ -224,6 +225,53 @@ async function loadDeliveryCreds(src, v) {
224
225
  throw new Error(`delivery: no scoped creds at ${where}. Launch via \`cotal setup\`/\`cotal go\` (the setup helper mints + writes it), or pass --creds <file>; for a standalone dev run use --dev-mint.`);
225
226
  }
226
227
  // Parsing lives in the dispatcher now, driven by the `deliver` command's declared flags.
228
+ /**
229
+ * RECORD THIS PROCESS as the space's delivery daemon, and un-record it on a clean exit (#1528).
230
+ *
231
+ * The daemon writes its OWN record because it is the only participant that always knows it is
232
+ * running. Until now `delivery.<space>.pid` was written ONLY by the CLI launcher
233
+ * (`startDeliveryDetached`), so every other route to a live daemon — a container entrypoint,
234
+ * systemd, an operator typing `cotal deliver --space …`, a hosted composition calling
235
+ * {@link runDelivery} — left whatever was on disk untouched and readers believed it. A record naming
236
+ * a pid that died days ago does not merely under-report: `down`'s `mayBeRunning` guard exists to
237
+ * fail CLOSED so `cotal down nats` cannot pull the broker out from under a live dependant, and a
238
+ * stale record supplies exactly the proof-of-death that guard requires.
239
+ *
240
+ * The launcher's shape is followed rather than re-invented: the CANONICAL path (never a pre-upgrade
241
+ * name), a provably dead pre-upgrade record reclaimed first so an upgraded root never holds both
242
+ * spellings, then the #969 identity pin beside it. The pin goes with the record on the way out.
243
+ *
244
+ * The removal is pid-CHECKED, like the manager's: a daemon that exits must not delete a record that
245
+ * already belongs to its successor (a fast restart writes the new pid before the old process has
246
+ * finished unwinding), so the record goes only while it still names this process.
247
+ *
248
+ * A root with no `.cotal/` is not a workstation and gets no record — a hosted composition has none,
249
+ * and creating one here would plant a stray workspace root wherever the daemon happened to run.
250
+ */
251
+ export function recordDeliveryPid(root, space) {
252
+ if (!existsSync(join(root, ".cotal"))) {
253
+ console.error(`• delivery: no ${join(root, ".cotal")} here, so this daemon is not recorded in a pidfile (nothing local will find it by pid)`);
254
+ return () => { };
255
+ }
256
+ const ctx = { root, space };
257
+ reclaimDeadPreUpgradeRecord(DELIVERY_PIDFILE, ctx);
258
+ const pidPath = canonicalLocalProcessPath(DELIVERY_PIDFILE, ctx);
259
+ const mine = String(process.pid);
260
+ writeFileSync(pidPath, mine);
261
+ // #969: pin the pid to its process start so a later teardown can refuse a reused pid.
262
+ writeIdentityPin(pidPath, process.pid);
263
+ return () => {
264
+ try {
265
+ if (readFileSync(pidPath, "utf8").trim() !== mine)
266
+ return; // a successor's record: not ours to remove
267
+ removeIdentityPin(pidPath);
268
+ rmSync(pidPath, { force: true });
269
+ }
270
+ catch {
271
+ /* already gone, or unreadable: leaving a record we cannot prove is ours is the safe error */
272
+ }
273
+ };
274
+ }
227
275
  /**
228
276
  * Run the delivery daemon: the server-side Plane-3 durable backstop. A thin composition root that
229
277
  * builds a scoped `delivery` endpoint, acquires the single-flight lease, and runs the existing
@@ -244,6 +292,35 @@ async function loadDeliveryCreds(src, v) {
244
292
  * the manager and this daemon are handed the SAME store.
245
293
  */
246
294
  export async function runDelivery(args, store) {
295
+ // A START-UP FAILURE AFTER THE ACQUIRE MUST GIVE THE SHARD BACK, INCLUDING WHEN THE CALLER
296
+ // CATCHES IT. `runDelivery` is awaited inside the CLI dispatcher's own try/catch, so a rejection
297
+ // out of start-up is an ordinary handled error: the process prints one line and exits 1 while the
298
+ // lease row still claims the shard for the rest of the 30s bucket TTL, and the next `cotal up` is
299
+ // refused. That is this issue's outage reached without any signal, starvation or broker fault.
300
+ //
301
+ // The release therefore hangs off the REJECTION rather than off a process event. `runStartedDelivery`
302
+ // publishes its own releaser once the shard is actually ours, so before the acquire there is nothing
303
+ // to give back and this catch re-throws untouched.
304
+ //
305
+ // THE FAULT STAYS A FAULT. This never swallows: the original error is re-thrown after the release,
306
+ // so the CLI prints the same line and exits with the same non-zero code it would have without any
307
+ // of this. Cleaning up a lease must not turn a start-up failure into a quiet one.
308
+ let releaseOnStartupFailure;
309
+ try {
310
+ await runStartedDelivery(args, store, (r) => { releaseOnStartupFailure = r; });
311
+ }
312
+ catch (e) {
313
+ if (releaseOnStartupFailure !== undefined) {
314
+ console.error(`\u2717 delivery: start-up failed - releasing the shard before exiting`);
315
+ try {
316
+ await releaseOnStartupFailure();
317
+ }
318
+ catch { /* the throw below is the report */ }
319
+ }
320
+ throw e;
321
+ }
322
+ }
323
+ async function runStartedDelivery(args, store, publishReleaser) {
247
324
  const v = args.values;
248
325
  const shard = v.shard ? Number(v.shard) : 0;
249
326
  const shards = v.shards ? Number(v.shards) : 1;
@@ -302,6 +379,11 @@ export async function runDelivery(args, store) {
302
379
  await validateScanTargetAdmission(scanTarget);
303
380
  console.error(`• delivery: $SYS sweeps bound to ${join(scanTarget.root, ".cotal")} (account ${scanTarget.expectedAccount})`);
304
381
  const reloadStoreIdentity = reloadStoreIdentityOf(credsSrc);
382
+ // THIS daemon's connection nkey, pinned once. It is what the endpoint AUTHENTICATES as, and it is
383
+ // what `card.id` must carry at construction (a creds SOURCE has no cred to derive it from yet).
384
+ // It is NOT what the lease record carries as `holder`, see readOwnLease, which compares against
385
+ // `ep.card.id`, the value the endpoint actually stamps.
386
+ const ownId = idFromCreds(creds.initial);
305
387
  const ep = new CotalEndpoint({
306
388
  space,
307
389
  servers: server,
@@ -314,7 +396,7 @@ export async function runDelivery(args, store) {
314
396
  consume: false, // it pulls the Plane-3 consumers itself; no agent live-tail
315
397
  watchPresence: true, // read the roster for @mention resolution …
316
398
  registerPresence: false, // … but NEVER publish the daemon onto the roster (it's infra, not a peer)
317
- card: { id: idFromCreds(creds.initial), name: "delivery", role: "delivery", kind: "endpoint" },
399
+ card: { id: ownId, name: "delivery", role: "delivery", kind: "endpoint" },
318
400
  });
319
401
  // Both channels: raw connection errors ride `error`, while every condition the endpoint is
320
402
  // already surviving — a failed 75% renewal, the passive backstop's "still holds the previous
@@ -327,6 +409,11 @@ export async function runDelivery(args, store) {
327
409
  // Acquire the single-flight lease BEFORE binding the loops: a loud refusal-to-bind if another daemon
328
410
  // already holds this shard (two clients binding the same durable name SPLIT delivery). The bucket TTL
329
411
  // frees a crashed holder's lease so a fresh daemon re-acquires.
412
+ // THE REVISION THIS PROCESS OWNS, or `undefined` once it knows it does not own the row any more.
413
+ // `releaseDeliveryLease` takes it as a compare-and-swap, so a shutdown that no longer holds the
414
+ // shard releases NOTHING rather than deleting whatever row is there. The exits where that matters
415
+ // are exactly the takeover exits below: without it, the departing daemon removes the REPLACEMENT's
416
+ // lease on its way out and leaves the shard with no holder at all.
330
417
  let revision;
331
418
  try {
332
419
  revision = await ep.acquireDeliveryLease(shard);
@@ -337,6 +424,125 @@ export async function runDelivery(args, store) {
337
424
  process.exit(1);
338
425
  return;
339
426
  }
427
+ // SIGNAL HANDLING IS ARMED HERE, THE STATEMENT AFTER THE SHARD BECOMES OURS, not at the end of
428
+ // start-up, and not merely before the readiness flip. From this line on there is a row on the
429
+ // broker with this daemon's name on it, and every instant until a handler exists is an instant in
430
+ // which SIGTERM takes Node's DEFAULT action: immediate death, no release, the shard claimed by a
431
+ // process that no longer exists for the rest of the 30s bucket TTL. The next `cotal up` is then
432
+ // refused outright with "a live lease already exists".
433
+ //
434
+ // THE WINDOW WAS REACHABLE FROM THE PUBLIC CLI, and reviewers reproduced it deterministically:
435
+ // SIGTERM at the instant the lease row turns ready killed the daemon 6/6 times with the row left
436
+ // behind and a real replacement refused, while the same signal 1500ms later released cleanly 2/2.
437
+ // Registration used to sit ~460 lines below, past the readiness flip and the awaited membership
438
+ // and timer-writer starts, so `cotal up` could return on a ready row while the daemon was still
439
+ // defenceless. Moving it merely below `shutdown`'s own definition is NOT enough: that still sits
440
+ // after markReady and after `startMembership`.
441
+ //
442
+ // The handler cannot call `shutdown` (defined far below, over state that does not exist yet), so
443
+ // it does the one thing that is always correct here and never wrong later: give the lease back if
444
+ // it is provably ours, then exit. Once the real `shutdown` is installed it REPLACES this, and
445
+ // `stopping` makes the pair idempotent, so a signal arriving mid-swap cannot run both.
446
+ let stopping = false;
447
+ /** Removes THIS process's liveness record, once it has one. Declared BEFORE the handlers that
448
+ * call it and assigned below, so a signal landing between the two is a no-op rather than a
449
+ * temporal-dead-zone throw: at that instant nothing has been written, so there is nothing to
450
+ * remove. Every exit path from the acquire onward goes through one of the three closures that
451
+ * call it, which is what keeps the record's lifetime equal to this daemon's. */
452
+ let unrecordPid;
453
+ const earlyStop = (code) => {
454
+ if (stopping)
455
+ return;
456
+ stopping = true;
457
+ setTimeout(() => process.exit(code), 2000);
458
+ unrecordPid?.();
459
+ void (async () => {
460
+ try {
461
+ const own = await ep.readDeliveryLeaseEntry(shard);
462
+ if (own !== undefined && ep.ownsDeliveryLease(own.info))
463
+ await ep.releaseDeliveryLease(shard, own.revision);
464
+ }
465
+ catch { /* broker may be gone, the bucket TTL is the crash-safe release authority */ }
466
+ try {
467
+ await ep.stop();
468
+ }
469
+ catch { /* broker may be gone */ }
470
+ process.exit(code);
471
+ })();
472
+ };
473
+ const earlySigint = () => earlyStop(0);
474
+ const earlySigterm = () => earlyStop(0);
475
+ process.on("SIGINT", earlySigint);
476
+ process.on("SIGTERM", earlySigterm);
477
+ // THE LIVENESS RECORD GOES HERE, AFTER THE ACQUIRE AND NOT BEFORE IT (#1528).
478
+ //
479
+ // The record answers "which process is this space's delivery daemon", and until this line this
480
+ // process cannot answer it: the lease is the single-flight admission, and a daemon that loses the
481
+ // CAS above REFUSES TO BIND and exits. Writing on entry would let such a loser overwrite the live
482
+ // holder's record on its way out the door — a fresh record naming a process that is about to be
483
+ // gone, which is the same lie this issue is about with a newer timestamp on it. The acquire is
484
+ // the first instant the answer is this process, so it is the first instant the record may say so.
485
+ //
486
+ // It is also written BEFORE `startPlane3`, deliberately: from the acquire onward there is a row on
487
+ // the broker with this daemon's name on it, so an operator's `cotal down` must be able to FIND
488
+ // this process even if the bind below hangs or fails. Readiness is a separate fact and the lease's
489
+ // own `ready` flag already carries it; the pidfile only ever claimed a process.
490
+ //
491
+ // Placed one statement after the signal handlers, so every exit path from here on can remove it.
492
+ unrecordPid = recordDeliveryPid(findCotalRoot(), space);
493
+ // AND THE SAME RELEASE, REACHABLE BY AN ORDINARY `catch`. The two process-level guards below
494
+ // only see a fault that reaches the RUNTIME. A start-up rejection on the public CLI path does
495
+ // not: `runCli` awaits this function inside its own try/catch (cli/src/command.ts), so the
496
+ // rejection is HANDLED, `unhandledRejection` never fires, and the process exits 1 through the
497
+ // CLI's own error line with the shard still claimed. A reviewer reproduced exactly that against
498
+ // a real broker by pre-creating a conflicting fan-out durable so `startPlane3` rejects after the
499
+ // acquire: exit 1, row still on the broker, replacement refused with "a live lease already
500
+ // exists". So the release is published HERE, to a scope that a plain `catch` around the rest of
501
+ // start-up can reach, and the process guards stay as the backstop for faults that never become
502
+ // a rejection this function can see (a synchronous throw in a timer, say).
503
+ publishReleaser(async () => {
504
+ if (stopping)
505
+ return;
506
+ stopping = true;
507
+ unrecordPid?.(); // the record dies with the daemon it describes, including on a start-up failure
508
+ try {
509
+ const own = await ep.readDeliveryLeaseEntry(shard);
510
+ if (own !== undefined && ep.ownsDeliveryLease(own.info))
511
+ await ep.releaseDeliveryLease(shard, own.revision);
512
+ }
513
+ catch { /* broker may be gone - the bucket TTL is the crash-safe release authority */ }
514
+ try {
515
+ await ep.stop();
516
+ }
517
+ catch { /* broker may be gone */ }
518
+ });
519
+ // A START-UP FAILURE AFTER THE ACQUIRE MUST ALSO GIVE THE SHARD BACK. Between this point and the
520
+ // handler swap far below, a throw would otherwise propagate out of `runDelivery` with the row
521
+ // still claiming the shard, stranding it for the bucket TTL exactly as an unhandled signal did -
522
+ // a different door into the same outage. So the release happens HERE, where the failure is,
523
+ // rather than being trusted to a caller that may not have one.
524
+ //
525
+ // THE FAULT STAYS LOUD, AND STAYS A FAULT. These guards change WHEN the process dies, never
526
+ // whether it reports why: the original error is printed with its stack, as Node's default handler
527
+ // would, and the exit code is non-zero. Cleaning up a lease must not turn a crash into a quiet
528
+ // stop - that would trade this issue's outage for a silent one.
529
+ //
530
+ // DISARMED BEFORE THEY CAN RECURSE. `earlyStop`'s own async body can itself reject (the broker may
531
+ // be mid-teardown), which would re-enter these listeners while cleanup is in flight. They are
532
+ // removed on first use, so a cleanup fault falls through to the default handler and the 2s
533
+ // hard-exit timer remains the backstop. `stopping` already makes a second entry a no-op; this
534
+ // makes the recursion impossible rather than merely harmless.
535
+ const earlyFault = (what, err) => {
536
+ process.off("uncaughtException", earlyUncaught);
537
+ process.off("unhandledRejection", earlyRejection);
538
+ console.error(`\u2717 delivery: start-up ${what} - releasing shard ${shard} before exiting`);
539
+ console.error(err instanceof Error ? (err.stack ?? err.message) : String(err));
540
+ earlyStop(1);
541
+ };
542
+ const earlyUncaught = (e) => earlyFault("faulted", e);
543
+ const earlyRejection = (e) => earlyFault("rejected a promise", e);
544
+ process.on("uncaughtException", earlyUncaught);
545
+ process.on("unhandledRejection", earlyRejection);
340
546
  // Broker-sourced graph membership handle — declared BEFORE Plane-3 so the delivery-admin reload
341
547
  // hook below can close over it (it starts further down; the closure reads it live).
342
548
  let membership;
@@ -398,7 +604,6 @@ export async function runDelivery(args, store) {
398
604
  membershipDown = e.message;
399
605
  console.error(`! membership: failed to start (${membershipDown}); graph membership degraded, delivery unaffected`);
400
606
  }
401
- let stopping = false;
402
607
  // The TIMER WRITER (SPEC 13.2): the pump that turns workflow `.schedule` requests into armed
403
608
  // broker schedules. Hosted here because this daemon is the space's standing server-side process;
404
609
  // without a running writer no pause on the space ever expires. Its OWN connection under the same
@@ -447,10 +652,54 @@ export async function runDelivery(args, store) {
447
652
  stopping = true;
448
653
  clearInterval(renew);
449
654
  clearInterval(brokerWatch);
655
+ // THE RECORD GOES FIRST, AND SYNCHRONOUSLY. Everything below this line talks to a broker that
656
+ // may be dead and is bounded only by the 2s hard exit; a record left behind because a drain
657
+ // hung is a record that outlives its process, which is this issue's defect re-entering through
658
+ // the exit path. `rmSync` needs no broker and cannot hang, so it runs before any of it.
659
+ unrecordPid?.();
450
660
  // Hard-exit fallback: a graceful release/stop talks to the broker, which may be DEAD (the broker-gone
451
661
  // exit path) — don't let that hang the process. Force exit if the graceful path doesn't finish quickly.
452
662
  setTimeout(() => process.exit(code), 2000);
453
663
  void (async () => {
664
+ // THE LEASE GOES FIRST. Everything else here is this process's own teardown, which nobody is
665
+ // waiting on; the lease row is the one piece of SHARED state, and while it survives no
666
+ // replacement daemon can acquire the shard at all (the acquire is an atomic create, so it is
667
+ // refused outright and the row blocks the slot for the rest of the 30s bucket TTL). Behind
668
+ // three awaits it was reachable only if all three finished inside the 2s hard-exit above, and
669
+ // membership.stop() and the timer writer's stop+drain each talk to the broker on their own
670
+ // connections.
671
+ //
672
+ // RELEASE AGAINST THE BROKER'S REVISION, NOT THE ONE THIS PROCESS HAPPENS TO HOLD.
673
+ //
674
+ // The release is a compare-and-swap, so a token one step behind frees NOTHING, and it fails
675
+ // SILENTLY: `releaseDeliveryLease` swallows every error by design, because at shutdown the
676
+ // broker may already be gone. A daemon can therefore stop cleanly, believe it released, and
677
+ // leave its row claiming the shard for the rest of the 30s bucket TTL - after which the next
678
+ // `cotal up` is refused with "a live lease already exists" and the shard is unservable by
679
+ // anyone. That is this issue's outage reached without any starvation or broker fault at all.
680
+ //
681
+ // THE CACHED TOKEN CAN LAG THE BROKER, and it does not take a fault to get there. The claim is
682
+ // deliberately that general, because more than one ordinary interleaving produces it and the
683
+ // trace does not distinguish them: `markDeliveryLeaseReady` MOVES the row, so between the
684
+ // broker applying that write and the daemon assigning the returned value, `revision` still
685
+ // holds the acquire token - and a shutdown running in that interval argues it. A markReady
686
+ // that rejects AFTER its write landed leaves the same lag permanently, since the failure is
687
+ // caught (correctly: the renew loop repairs it). The launcher can already have observed the
688
+ // row ready in either case, so from outside this is a started daemon. Measured, with both
689
+ // controls in one run: release at the acquire revision left the row in place, release at the
690
+ // current revision removed it, and a replacement endpoint could then acquire.
691
+ //
692
+ // So: re-read, and release only what is PROVABLY ours. `ownsDeliveryLease` is the same test
693
+ // the renew path uses, so a row belonging to a successor is left alone - which is the property
694
+ // the CAS was protecting in the first place, now argued from evidence rather than from a
695
+ // token that may be out of date. No row, or not ours: nothing to release, and the TTL remains
696
+ // the crash-safe authority for anything this path cannot reach.
697
+ try {
698
+ const own = await ep.readDeliveryLeaseEntry(shard);
699
+ if (own !== undefined && ep.ownsDeliveryLease(own.info))
700
+ await ep.releaseDeliveryLease(shard, own.revision);
701
+ }
702
+ catch { /* broker may be gone - the bucket TTL is the crash-safe release authority */ }
454
703
  try {
455
704
  await membership?.stop();
456
705
  }
@@ -463,10 +712,6 @@ export async function runDelivery(args, store) {
463
712
  await timerWriter?.nc.drain();
464
713
  }
465
714
  catch { /* broker may be gone */ }
466
- try {
467
- await ep.releaseDeliveryLease(shard);
468
- }
469
- catch { /* broker may be gone */ }
470
715
  try {
471
716
  await ep.stop();
472
717
  }
@@ -474,25 +719,325 @@ export async function runDelivery(args, store) {
474
719
  process.exit(code);
475
720
  })();
476
721
  };
477
- // Renew the lease at ~half the TTL so a healthy holder never self-evicts; losing the CAS means
478
- // another daemon took over (we exit rather than double-deliver).
722
+ // SIGNAL HANDLERS GO UP THE MOMENT `shutdown` EXISTS, NOT AFTER STARTUP FINISHES.
723
+ //
724
+ // The lease is acquired long before this point, and registration used to sit at the very end of
725
+ // start-up, past 22 further `await`s (Plane-3 bind, membership feed, timer writer, broker watch).
726
+ // A SIGTERM landing in that window hit Node's DEFAULT handler and killed the process outright:
727
+ // no release, no CAS, the row left behind for the rest of the 30s bucket TTL. The daemon had
728
+ // already announced itself up, so from outside it was a fully started daemon that died silently
729
+ // holding the shard, and the next `cotal up` was refused with "a live lease already exists", the
730
+ // shard unservable for 30s after a perfectly ordinary stop-and-restart.
731
+ //
732
+ // Measured, not reasoned: with a diagnostic on the handler, the failing run's daemon log shows
733
+ // the up line and NO `SIGTERM received`, while the two roots that passed in the same run show it.
734
+ // That is what reddened `delivery-refresh-keeps-tls`, which stops the daemon and relaunches at
735
+ // once. Registering here closes the window to the span before the lease exists, where there is
736
+ // nothing to release.
737
+ //
738
+ // HAND OVER FROM `earlyStop`. The early handlers armed at the lease acquire are REMOVED rather
739
+ // than left alongside these: `process.on` appends, so both would fire and the early one - which
740
+ // knows nothing of the renew timer, the membership feed or the timer writer - would run first and
741
+ // set `stopping`, making the full teardown a no-op. `stopping` still guards the swap itself, so a
742
+ // signal delivered between these two statements is handled exactly once.
743
+ process.off("SIGINT", earlySigint);
744
+ process.off("SIGTERM", earlySigterm);
745
+ // The start-up fault guards go too, and BY REFERENCE: `removeAllListeners` here would strip
746
+ // listeners this daemon does not own. They exist to cover the window where no `shutdown` exists;
747
+ // past this line a fault should surface normally rather than become a quiet exit that skips the
748
+ // full teardown.
749
+ process.off("uncaughtException", earlyUncaught);
750
+ process.off("unhandledRejection", earlyRejection);
751
+ process.on("SIGINT", () => shutdown(0));
752
+ process.on("SIGTERM", () => shutdown(0));
753
+ /** What the broker says about THIS shard's lease key right now, the verdict a failed renew does
754
+ * NOT have. `unknown` never collapses into `gone`: not being able to look is not the same fact as
755
+ * looking and finding nothing (#1318).
756
+ *
757
+ * The ownership test is `ep.ownsDeliveryLease`, which compares the row against BOTH this
758
+ * endpoint's wire identity and its per-run incarnation. Neither half is optional, and the two
759
+ * failures they rule out are different:
760
+ *
761
+ * 1. Comparing against `ownId` (the bare connection nkey `U…`) is comparing across NAMESPACES:
762
+ * the endpoint rewrites `card.id` in its constructor to the wire PRINCIPAL dot-form
763
+ * `${owner}.${actor}` (`local.U…`), and that is what `encodeLease` stamps. That mismatch made
764
+ * `held` UNREACHABLE, the daemon read its own row, failed to recognise itself, took the
765
+ * `taken` branch and exited naming ITSELF as the thief, leaving a not-ready row and no
766
+ * process. `held` is the one survivable reading ("your renew failed but the shard is still
767
+ * yours"), so making it unreachable turned every recoverable renew failure into a permanent
768
+ * withdrawal: #1318's own outage re-entering through the path built to prevent it.
769
+ *
770
+ * 2. Comparing on the principal ALONE is not sufficient either, and this is the one that looks
771
+ * correct: the daemon's cred is a FILE that every restart re-reads, so a REPLACEMENT daemon
772
+ * authenticates as the same nkey and writes the same `holder`. A displaced daemon would then
773
+ * read its successor's row, conclude the shard was still its own, carry on serving a shard it
774
+ * had lost (two daemons on one durable, which is the split this lease exists to prevent) and
775
+ * CAS-release the live holder's row on the way out. Cells F and G stage exactly that, with two
776
+ * daemons sharing one creds file, because that is what the product does. */
777
+ const readOwnLease = async () => {
778
+ let current;
779
+ try {
780
+ current = await ep.readDeliveryLeaseEntry(shard);
781
+ }
782
+ catch (e) {
783
+ return { kind: "unknown", why: e.message };
784
+ }
785
+ if (current === undefined)
786
+ return { kind: "gone" };
787
+ if (!ep.ownsDeliveryLease(current.info))
788
+ return { kind: "taken", by: current.info.holder };
789
+ // Held by us, AND at the broker's own revision rather than the one this process last cached.
790
+ // That distinction is load-bearing: the renew that just failed may have been applied before its
791
+ // reply was lost, in which case the cached revision is permanently one behind and every later
792
+ // CAS is refused over a sequence this daemon moved itself. Adopting the read revision is what
793
+ // lets a survivable renew failure actually be survived. (An earlier comment here claimed the
794
+ // record carries no revision; it does, the KV entry's own `revision`, and that claim was
795
+ // wrong, which is why the stale token went unnoticed.)
796
+ return { kind: "held", revision: current.revision };
797
+ };
798
+ /** Print a lease state only when it CHANGES. The renew ticks every few seconds, so a broker
799
+ * outage would otherwise write the same line thousands of times into the daemon's log; an
800
+ * operator needs the line going in and the line coming out. */
801
+ let leaseState = "held";
802
+ const noteLease = (state, what) => {
803
+ if (state === leaseState)
804
+ return;
805
+ leaseState = state;
806
+ console.error(`${state === "held" ? "✓" : "!"} delivery: ${what} (space ${space}, shard ${shard})`);
807
+ };
808
+ /** Announce going quiet, once per quiesced episode. Same rule as {@link noteLease}: the renew ticks
809
+ * forever, and an operator needs the edge, not a log line per tick. */
810
+ let quiesced = false;
811
+ const noteQuiesce = () => {
812
+ if (quiesced)
813
+ return;
814
+ quiesced = true;
815
+ console.error(`! delivery: stopped serving shard ${shard} (fan-out, reader and control unbound) while it re-checks who owns the lease (space ${space})`);
816
+ };
817
+ /** Resume serving, and CLOSE the quiesced episode so a later one announces itself again. Every
818
+ * caller has just proven ownership; `why` is what proved it, since "it started serving again" is
819
+ * only auditable next to the evidence that permitted it. */
820
+ const resumeServing = async (why) => {
821
+ try {
822
+ await ep.rearmPlane3();
823
+ }
824
+ catch (e) {
825
+ console.error(`! delivery: ${why} but could not resume Plane-3 (${e.message})`);
826
+ return;
827
+ }
828
+ // `ready` MEANS "THE RESPONDER IS UP", and it is what `ensureDelivery` waits on and what the
829
+ // `cotal_channels` health surface reports. Startup flips it only after binding, for exactly that
830
+ // reason, and a re-acquire creates the row afresh, which `acquireDeliveryLease` deliberately
831
+ // makes NOT-ready. Without this flip a daemon that recovered would serve correctly while every
832
+ // readiness waiter in the space timed out against a permanently not-ready lease: the outage
833
+ // #1318 is about, surviving the repair by hiding in the readiness flag instead of the exit path.
834
+ // Ordered AFTER the re-arm so the flag never claims more than has actually been bound.
835
+ if (revision !== undefined) {
836
+ try {
837
+ revision = await ep.markDeliveryLeaseReady(shard, revision);
838
+ }
839
+ catch (e) {
840
+ console.error(`! delivery: resumed serving but could not mark the lease ready (${e.message}); the next renew re-synchronises`);
841
+ }
842
+ }
843
+ if (!quiesced)
844
+ return;
845
+ quiesced = false;
846
+ console.error(`\u2713 delivery: serving shard ${shard} again, ${why} (space ${space})`);
847
+ };
848
+ // Renew the lease at ~half the TTL so a healthy holder never self-evicts.
849
+ //
850
+ // A FAILED RENEW IS A QUESTION, NOT A VERDICT (#1318). The shipped code exited on ANY renew
851
+ // error, so a key that had EXPIRED under local CPU starvation, with nobody else holding it,
852
+ // reported as `wrong last sequence: 0`, was indistinguishable from a genuine takeover, and the
853
+ // only holder there was ended itself. The verdict comes from RE-READING the key: `taken` exits so
854
+ // the holder stays single, `gone` is repaired by an ATOMIC create (which arbitrates: if a
855
+ // replacement got there first the create fails and THAT is the genuine loss), and `held`/`unknown`
856
+ // keep serving. Overlap-guarded: two CAS attempts against the same cached revision would have the
857
+ // second refused over a sequence the first legitimately moved, a conflict this daemon manufactures
858
+ // itself and then reads as someone else's takeover.
859
+ let renewInFlight = false;
479
860
  const renew = setInterval(() => {
480
- ep.renewDeliveryLease(shard, revision)
481
- .then((r) => (revision = r))
482
- .catch((e) => {
483
- console.error(`✗ delivery: lost the lease (${e.message}) exiting so the holder is single`);
484
- shutdown(1);
485
- });
861
+ if (stopping || renewInFlight)
862
+ return;
863
+ renewInFlight = true;
864
+ void (async () => {
865
+ try {
866
+ // A renew with no owned revision is not a renew: this process already established that the
867
+ // shard is someone else's and is on its way out. Attempting one would either fail noisily or,
868
+ // worse, be graded as a lease question when ownership is already settled.
869
+ if (revision === undefined)
870
+ return;
871
+ revision = await ep.renewDeliveryLease(shard, revision);
872
+ // A SUCCESSFUL CAS RENEW IS PROOF OF OWNERSHIP, so it is also a re-arm point. Without this a
873
+ // daemon that quiesced on `unknown` (the broker could not answer who owns the shard) would
874
+ // stay silent forever once the broker came back: the `held` re-arm below only runs on the
875
+ // failure path, and the failure path stops running as soon as renews succeed again. Quiesce
876
+ // must be recoverable by the same evidence that makes it unnecessary.
877
+ await resumeServing("it renewed its lease");
878
+ noteLease("held", "renews its delivery lease again");
879
+ return;
880
+ }
881
+ catch (renewError) {
882
+ const why = renewError.message;
883
+ // GO QUIET BEFORE ASKING. From here this process does not know whether the shard is still
884
+ // its own, and a CAS on the lease row keeps one ROW, not one SERVER: across the re-read and
885
+ // the re-acquire below, an un-quiesced daemon would still be consuming the fan-out durable,
886
+ // still running the reader, and still answering ctl.delivery, so a replacement that won the
887
+ // shard in that window would SPLIT the durable with it. A review finding, and the pre-fix
888
+ // code did not have this window because it simply exited on the first renew failure.
889
+ try {
890
+ await ep.quiescePlane3();
891
+ // WITHDRAW THE READINESS CLAIM TOO. `ready` asserts the responder is up; it is now down, so
892
+ // leaving it set would tell every `ensureDelivery` waiter in the space to keep waiting on a
893
+ // daemon that is deliberately not answering. Best-effort: the renew that just failed means
894
+ // the CAS may fail too, and staying quiet matters more than the flag being tidy. The row
895
+ // itself is kept, the shard is still claimed, only the answering claim is withdrawn.
896
+ if (revision !== undefined) {
897
+ try {
898
+ revision = await ep.markDeliveryLeaseNotReady(shard, revision);
899
+ }
900
+ catch { /* the row may have moved on; the ownership read below is what decides */ }
901
+ }
902
+ // ANNOUNCED, because an operator watching a stall needs to know the daemon stopped serving
903
+ // on purpose rather than silently wedged, and because the live cell anchors on this line
904
+ // to know when to start demanding that this process holds no Plane-3 bindings.
905
+ noteQuiesce();
906
+ }
907
+ catch (e) {
908
+ console.error(`! delivery: could not quiesce Plane-3 while checking the lease (${e.message})`);
909
+ }
910
+ const reading = await readOwnLease();
911
+ switch (leaseAction(reading)) {
912
+ case "keep-serving":
913
+ // Still ours, or unanswerable. Only a PROVEN `held` re-arms: `unknown` stays quiet,
914
+ // because not being able to ask is not permission to keep acting on the shard.
915
+ if (mayServeOn(reading)) {
916
+ // ADOPT THE BROKER'S REVISION before serving again. The failed renew may have landed
917
+ // with its reply lost, so the cached token can be stale; re-arming on it would leave
918
+ // every later CAS refused over this daemon's own write and manufacture the takeover
919
+ // it is trying to rule out. The read just told us the sequence, use it.
920
+ // Narrowed by the guard above: `mayServeOn` admits only `held`, which is the one
921
+ // reading that carries a revision. Stated as an assert rather than a cast so a future
922
+ // widening of `mayServeOn` fails here loudly instead of serving without a CAS token.
923
+ if (reading.kind !== "held")
924
+ throw new Error(`delivery: mayServeOn admitted a "${reading.kind}" reading, which carries no revision to serve on`);
925
+ revision = reading.revision;
926
+ await resumeServing("re-reading the key showed the lease is still its own");
927
+ }
928
+ noteLease(reading.kind === "held" ? "held-unrenewed" : "unknown", reading.kind === "held"
929
+ ? `could not renew its lease (${why}) but the key is still its own; serving, retrying`
930
+ // NOT "serving": the quiesce above is still in force on this branch, and `mayServeOn`
931
+ // refuses `unknown`, so nothing re-armed. Saying "serving" here would describe the
932
+ // pre-fix behaviour and hide the very state this repair introduced.
933
+ : `could not renew its lease (${why}) or re-read it (${reading.kind === "unknown" ? reading.why : ""}); staying quiet, retrying until the broker answers`);
934
+ return;
935
+ case "reacquire":
936
+ try {
937
+ revision = await ep.acquireDeliveryLease(shard);
938
+ // Won the atomic create, so the shard is provably ours again: resume serving.
939
+ await resumeServing(`it won the atomic create at revision ${revision}`);
940
+ noteLease("held", `found its lease key gone (renew: ${why}) and re-acquired it at revision ${revision}`);
941
+ }
942
+ catch (e) {
943
+ // Refused: a live lease exists that is not ours, so a replacement daemon holds this
944
+ // shard. THIS is the genuine loss, and it exits, WITHOUT a revision to release, or
945
+ // the exit would delete the replacement's row.
946
+ revision = undefined;
947
+ console.error(`✗ delivery: lost the lease (${why}) and another daemon has taken shard ${shard} (${e.message}), exiting so the holder is single`);
948
+ shutdown(1);
949
+ }
950
+ return;
951
+ case "exit":
952
+ // Another daemon's row. Forget our revision first: releasing on the way out would delete
953
+ // the holder's lease and turn a clean handover into an unheld shard.
954
+ revision = undefined;
955
+ console.error(`✗ delivery: the lease for shard ${shard} is held by ${reading.kind === "taken" ? reading.by : "another daemon"}, not by this process (renew: ${why}), exiting so the holder is single`);
956
+ shutdown(1);
957
+ return;
958
+ }
959
+ }
960
+ finally {
961
+ renewInFlight = false;
962
+ }
963
+ })();
486
964
  }, Math.max(1000, Math.floor(LEASE_TTL_MS / 2)));
487
965
  // Coupled to the broker: POLL its reachability. Survive brief blips (the endpoint reconnects on its
488
- // own), but EXIT if the broker has been gone for BROKER_GONE_MS — the endpoint would otherwise retry
489
- // reconnect forever (its terminal-close never fires), so this is what stops the daemon outliving the
490
- // server it serves. (`cotal up`/`down` teardown stops it too.) The window is env-overridable for tests.
966
+ // own), but EXIT if the broker is GONE, the endpoint would otherwise retry reconnect forever (its
967
+ // terminal-close never fires), so this is what stops the daemon outliving the server it serves.
968
+ // (`cotal up`/`down` teardown stops it too.) The window is env-overridable for tests.
969
+ //
970
+ // WHAT "GONE" MEANS IS NOW DECIDED FROM EVIDENCE, NOT FROM A CLOCK (#1318). Three conditions used
971
+ // to produce one signal and only one of them was a dead server: the broker being down, this
972
+ // process being descheduled so the interval never fired, and a probe that could not complete a
973
+ // handshake because the local process could not get scheduled to finish it. A wall-clock
974
+ // `Date.now() - lastReachable` cannot tell them apart, and widening it only moves the threshold.
975
+ // Three signals separate them, and all three are things this process can actually observe:
976
+ //
977
+ // • MEASURED LOOP LAG. The gap between consecutive firings of a timer we own, minus its nominal
978
+ // period, is local starvation by direct measurement. It is credited back, so time this process
979
+ // spent off the runqueue is never counted against the broker.
980
+ // • COMPLETED NEGATIVE PROBES. Only a probe that RAN TO COMPLETION and returned false is
981
+ // evidence about the server. A probe that never ran contributes nothing (that was the whole
982
+ // defect: the window aged with no probe having failed), and one that REJECTED is an
983
+ // unanswered question, it now resets the counter and logs, where it used to be swallowed by
984
+ // `.catch(() => {})` and silently age the window.
985
+ // • TRANSPORT-LEVEL LIVENESS. A `transport: connected` edge from the endpoint's own connection
986
+ // cannot happen without a server on the other end, so it is positive evidence obtained for
987
+ // free, on a path that does not need this process to schedule a probe at all.
988
+ //
989
+ // A starvation diagnosis therefore reports DEGRADED and keeps serving; a genuinely dead broker
990
+ // still exits, on the same window, as fast as completed probes can say so.
491
991
  const BROKER_GONE_MS = Number(process.env.COTAL_DELIVERY_BROKER_GONE_MS) || 15_000;
992
+ // How many completed negatives make elapsed time believable. Two is the floor: one completed
993
+ // negative is a single refused connect, which a loopback under momentary pressure can produce.
994
+ // The default scales with the window so a test that shortens the window does not thereby demand
995
+ // more evidence than the window has room for.
996
+ const BROKER_GONE_PROBES = Math.max(2, Number(process.env.COTAL_DELIVERY_BROKER_GONE_PROBES) || Math.ceil(BROKER_GONE_MS / PROBE_INTERVAL_MS / 2));
997
+ // The hard backstop: past this, no amount of measured lag or transport optimism keeps the daemon
998
+ // alive. A daemon that OUTLIVES a dead broker is worse than one that restarts unnecessarily, so
999
+ // the repair is bounded and fails toward exiting.
1000
+ const BROKER_GONE_BACKSTOP_MS = Math.max(BROKER_GONE_MS, Number(process.env.COTAL_DELIVERY_BROKER_GONE_BACKSTOP_MS) || BROKER_GONE_MS * 4);
492
1001
  let lastReachable = Date.now();
1002
+ let completedNegatives = 0;
1003
+ let degraded = false;
1004
+ // What the endpoint's OWN connection reports about its socket to this same broker. Seeded true
1005
+ // because `ep.start()` above completed, which it cannot do without a server having answered.
1006
+ let transportConnected = true;
1007
+ const lag = new LoopLagMeter(PROBE_INTERVAL_MS);
1008
+ /** Positive evidence, from wherever it came: restart the window and its lag budget together. */
1009
+ const sawBroker = () => {
1010
+ lastReachable = Date.now();
1011
+ completedNegatives = 0;
1012
+ lag.reset();
1013
+ if (degraded) {
1014
+ degraded = false;
1015
+ console.error(`✓ delivery: the broker is answering again, Plane-3 is serving normally (space ${space})`);
1016
+ }
1017
+ };
1018
+ // The endpoint's OWN transport edge. This is evidence the daemon gets without being scheduled to
1019
+ // probe for it, and it is the signal that distinguishes "the connection object reports a
1020
+ // transport-level close" from "silence": a live connection to that address proves a server, while
1021
+ // a disconnect is the endpoint's own business (nats.js reconnects through blips of its own
1022
+ // accord) and merely stops excusing a probe that will not complete.
1023
+ ep.on("transport", (t) => {
1024
+ transportConnected = t.connected;
1025
+ if (t.connected)
1026
+ sawBroker();
1027
+ });
1028
+ // THE BUDGET THE PROBE IS ACTUALLY GIVEN, asked of the one function that decides it. A ws(s)
1029
+ // broker rides an HTTPS edge and gets 5s, not the 1s a loopback TCP broker gets; judging a ws
1030
+ // probe against 1000 would read every honest refusal on such a broker as this process's own
1031
+ // starvation, so the completed-negative count could never rise and a genuinely dead ws broker
1032
+ // would be ended only by the backstop, with the wrong reason in its log. The budget and the
1033
+ // judgment have to come from the same place.
1034
+ const probeBudgetMs = defaultProbeTimeoutMs(server);
493
1035
  const brokerWatch = setInterval(() => {
494
1036
  if (stopping)
495
1037
  return;
1038
+ // Measure FIRST, before any await: this is the gap since the previous firing, and it is the
1039
+ // only place the daemon can learn that it was not scheduled.
1040
+ lag.tick(Date.now());
496
1041
  // THE SAME TRANSPORT AS EVERY OTHER DIAL IN THIS PROCESS. This poll carries `latestCreds` — a
497
1042
  // standing credential — and `isReachable` performs a real authenticated connect whenever creds
498
1043
  // are supplied, every 2 seconds, for the life of the daemon. It is the most repeated credential
@@ -502,21 +1047,102 @@ export async function runDelivery(args, store) {
502
1047
  // least consistent. What is new is the ASYMMETRY: with the two dials above upgraded, an operator
503
1048
  // who enables TLS would get a protected main path and an unprotected watchdog. An inconsistent
504
1049
  // guarantee is worse than a uniformly absent one, because the operator now believes something.
1050
+ const probeStarted = Date.now();
1051
+ // Watch this process's own scheduling FOR THE DURATION OF THE PROBE. A refusal is only evidence
1052
+ // about the server if the server was actually given the time the deadline promised it, and
1053
+ // under short CPU slices most of a probe's wall-clock can be time this process was not running.
1054
+ const sampler = new DescheduleSampler();
1055
+ sampler.start(probeStarted);
505
1056
  void isReachable(server, { creds: latestCreds, ...(tls ? { tls: true } : {}) })
506
- .then((ok) => {
507
- if (ok) {
508
- lastReachable = Date.now();
1057
+ .then((ok) => classifyProbe(ok, Date.now() - probeStarted, probeBudgetMs, PROBE_LATE_FACTOR, sampler.stop()),
1058
+ // A REJECTED probe is an unanswered question, not a negative answer. It used to be swallowed
1059
+ // whole by `.catch(() => {})`, so it neither refreshed the window nor evaluated anything and
1060
+ // silently aged the daemon toward an exit it had gathered no evidence for.
1061
+ (e) => {
1062
+ sampler.stop();
1063
+ console.error(`! delivery: the broker probe did not complete (${e.message}), no verdict from it; serving, retrying`);
1064
+ return classifyProbe(undefined, Date.now() - probeStarted);
1065
+ })
1066
+ .then((probe) => {
1067
+ if (stopping)
1068
+ return;
1069
+ if (probe.counts === "positive") {
1070
+ sawBroker();
1071
+ return;
1072
+ }
1073
+ if (probe.counts === "incomplete") {
1074
+ // The run of credible refusals is broken by a probe that did not complete.
1075
+ completedNegatives = 0;
509
1076
  return;
510
1077
  }
511
- if (Date.now() - lastReachable > BROKER_GONE_MS) {
512
- console.error(`✗ delivery: broker unreachable for >${BROKER_GONE_MS / 1000}s exiting (coupled to the broker)`);
1078
+ if (probe.counts === "starved") {
1079
+ // The probe RAN and said no, but its answer arrived so far past its own deadline that the
1080
+ // deadline was enforced against this process rather than against the server. That is the
1081
+ // starved-client case, and it is the one an elapsed-time predicate cannot see at all: the
1082
+ // timer is firing, the probes are completing, and every one of them is `false`.
1083
+ // Dated with the instant the answer ARRIVED, so the meter can tell whether this stall is
1084
+ // the same wall-clock interval the tick above already charged (union, counted once) or an
1085
+ // adjacent one (disjoint, both counted). The overlap question is settled by timestamps
1086
+ // rather than by comparing magnitudes, which cannot distinguish the two.
1087
+ lag.credit(probe.lateBy, Date.now());
1088
+ completedNegatives = 0;
1089
+ }
1090
+ else {
1091
+ // A refusal that arrived on time. This is the only thing that may accrue against the broker.
1092
+ completedNegatives += 1;
1093
+ }
1094
+ // ONE SPAN, MEASURED ONCE, USED FOR BOTH. Reading the clock twice would let the elapsed span
1095
+ // and the lag clamp disagree by the cost of the call itself.
1096
+ const sinceReachable = Date.now() - lastReachable;
1097
+ const verdict = brokerGoneVerdict({
1098
+ msSinceLastReachable: sinceReachable,
1099
+ // CLAMPED TO THE SPAN IT IS SUBTRACTED FROM. The interval gap and a late probe answer can
1100
+ // testify to the same stall, and the meter cannot tell that from two adjacent stalls, so it
1101
+ // keeps both charges and the bound is applied here, where the span is known. Without it a
1102
+ // 10s stall read 17s and drove unstarved time negative, which cannot be cleared by any
1103
+ // amount of real outage: a dead broker then survives the evidence clause and exits only on
1104
+ // the backstop, 44s -> 62s measured. Time not had cannot exceed time passed.
1105
+ starvedMs: lag.starvedMsWithin(sinceReachable),
1106
+ completedNegatives,
1107
+ transportConnected,
1108
+ windowMs: BROKER_GONE_MS,
1109
+ requiredNegatives: BROKER_GONE_PROBES,
1110
+ backstopMs: BROKER_GONE_BACKSTOP_MS,
1111
+ });
1112
+ if (verdict.exit) {
1113
+ // SAY WHICH EXIT THIS IS. The single message here previously claimed completed probes over
1114
+ // ">15s of unstarved time" on BOTH paths, and on the backstop path that sentence is simply
1115
+ // false: the backstop fires precisely when the evidence clauses did NOT conclude, often
1116
+ // with zero completed refusals. An operator debugging a starved host was handed a
1117
+ // confident evidentiary claim the daemon had not established.
1118
+ console.error(verdict.reason === "backstop"
1119
+ ? `✗ delivery: giving up on elapsed time alone, ${Math.round(BROKER_GONE_BACKSTOP_MS / 1000)}s since the last ` +
1120
+ `confirmed reachability with no sufficient evidence either way (${completedNegatives} completed probes refused, ` +
1121
+ `${Math.round(lag.starvedMsWithin(sinceReachable) / 1000)}s of local scheduler lag credited). This is a BOUND, not a diagnosis: the ` +
1122
+ `broker may be gone or this process may have been starved past the bound, exiting (coupled to the broker)`
1123
+ : `✗ delivery: broker unreachable, ${completedNegatives} completed probes refused within their deadline over ` +
1124
+ `>${BROKER_GONE_MS / 1000}s of unstarved time (${Math.round(lag.starvedMsWithin(sinceReachable) / 1000)}s of local scheduler lag credited), exiting (coupled to the broker)`);
513
1125
  shutdown(1);
1126
+ return;
1127
+ }
1128
+ // NOT an exit. Say so once per episode, naming WHICH condition it is, so an operator reading
1129
+ // the log during a load incident sees "this host starved me" rather than a daemon that
1130
+ // silently vanished.
1131
+ if (!degraded && verdict.reason !== "reachable") {
1132
+ degraded = true;
1133
+ console.error(verdict.reason === "starved"
1134
+ ? `! delivery: DEGRADED, cannot reach the broker, but ${Math.round(lag.starvedMsWithin(sinceReachable) / 1000)}s of that window was local scheduler lag (this host is starving this process, not the broker); serving, retrying`
1135
+ : verdict.reason === "transport-live"
1136
+ ? `! delivery: DEGRADED, a fresh probe cannot complete, but this daemon's own connection to ${server} is still open, so the broker is there and this process cannot ask; serving, retrying`
1137
+ : `! delivery: DEGRADED, the broker has not answered for >${BROKER_GONE_MS / 1000}s but only ${completedNegatives} of ${BROKER_GONE_PROBES} probes have refused within their deadline; serving, retrying`);
514
1138
  }
515
1139
  })
516
- .catch(() => { });
517
- }, 2000);
518
- process.on("SIGINT", () => shutdown(0));
519
- process.on("SIGTERM", () => shutdown(0));
1140
+ .catch((e) => {
1141
+ // The verdict path itself faulted. Never silent, and never an exit: a bug in the detector
1142
+ // must not end the daemon it is meant to keep alive.
1143
+ console.error(`! delivery: the broker watch tick faulted (${e.message}); serving, retrying`);
1144
+ });
1145
+ }, PROBE_INTERVAL_MS);
520
1146
  await new Promise(() => { }); // run until signalled
521
1147
  }
522
1148
  //# sourceMappingURL=delivery.js.map