@cotal-ai/delivery 0.57.0 → 0.59.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/delivery.js CHANGED
@@ -1,11 +1,12 @@
1
- import { existsSync, readFileSync, rmSync, writeFileSync } from "node:fs";
1
+ import { existsSync, writeFileSync } from "node:fs";
2
2
  import { basename, dirname, join, resolve } from "node:path";
3
- import { CotalEndpoint, DEFAULT_SERVER, LEASE_TTL_MS, accountFromCreds, credsClaims, deliveryBucket, dialerFor, idFromCreds, defaultProbeTimeoutMs, isCasLoss, isPermissionDenied, isReachable, leaseKey, mintCreds, newIdentity, formatSecretStoreIdentity, parseSecretStoreIdentity, sameSecretStoreIdentity, standaloneConnectOpts, startTimerWriter, } from "@cotal-ai/core";
3
+ import { CotalEndpoint, DEFAULT_SERVER, LEASE_TTL_MS, accountFromCreds, credsClaims, deliveryBucket, dialerFor, idFromCreds, isCasLoss, isPermissionDenied, isReachable, leaseKey, mintCreds, newIdentity, formatSecretStoreIdentity, parseSecretStoreIdentity, sameSecretStoreIdentity, standaloneConnectOpts, startTimerWriter, } from "@cotal-ai/core";
4
4
  import { PermissionViolationError } from "@nats-io/transport-node";
5
- import { DELIVERY_CREDS_KIND, DELIVERY_PIDFILE, FsSecretStore, authDir, canonicalLocalProcessPath, canonicalRoot, deliveryCredsKey, findCotalRoot, isWorkspaceTargetError, loadSpaceAuth, reclaimDeadPreUpgradeRecord, removeIdentityPin, resolveMeshTarget, segmentedKey, soleSpaceOf, spaceSegment, workspaceSecretStore, writeIdentityPin } from "@cotal-ai/workspace";
5
+ import { DELIVERY_CREDS_KIND, DELIVERY_PIDFILE, FsSecretStore, authDir, canonicalLocalProcessPath, canonicalRoot, deliveryCredsKey, findCotalRoot, isWorkspaceTargetError, loadSpaceAuth, reclaimDeadPreUpgradeRecord, removePidPair, requireCotalRoot, resolveMeshTarget, segmentedKey, soleSpaceOf, spaceSegment, workspaceSecretStore, writePidPair } from "@cotal-ai/workspace";
6
6
  import { startMembership } from "./membership.js";
7
- import { mayServeOn, brokerGoneVerdict, classifyProbe, DescheduleSampler, leaseAction, LoopLagMeter, PROBE_INTERVAL_MS, PROBE_LATE_FACTOR } from "./watchdog.js";
8
- import { executeEviction, executePlaneLiveness, executePrincipalLiveness, validateScanTargetAdmission } from "./evict-exec.js";
7
+ import { mayServeOn, leaseAction } from "./watchdog.js";
8
+ import { DeliveryTransportHealth } from "./transport-health.js";
9
+ import { executeEviction, executeEvictions, executePlaneLiveness, executePrincipalLiveness, validateScanTargetAdmission } from "./evict-exec.js";
9
10
  /** Re-exported for hosted compositions: the daemon cred's KIND, and the builder that turns it into
10
11
  * the {@link SecretStore} key for a space. Defined once in workspace (the layout is the workspace's)
11
12
  * so the writer, the renewal owner, and this reader can never drift apart. As of P7 the kind is NOT
@@ -106,7 +107,7 @@ export function assertUninjectedCredsSharesCwdRoot(opts) {
106
107
  const credsIdentity = { kind: "fs", root: credsWorkspace };
107
108
  if (sameSecretStoreIdentity(credsIdentity, cwdIdentity))
108
109
  return;
109
- throw new Error(`delivery: --creds names workstation ${formatSecretStoreIdentity(credsIdentity)} while membership-rw resolves under ${formatSecretStoreIdentity(cwdIdentity)} (process cwd). Pass both the same workstation root, or inject one SecretStore.`);
110
+ throw new Error(`delivery: --creds names workstation ${formatSecretStoreIdentity(credsIdentity)} while membership-rw resolves under ${formatSecretStoreIdentity(cwdIdentity)} (${opts.via ?? "process cwd"}). Pass both the same workstation root, or inject one SecretStore.`);
110
111
  }
111
112
  /**
112
113
  * THE ORDERING-CRITICAL HALF of the cred-source decision, split out so it cannot drift below the
@@ -124,6 +125,28 @@ function assertNoLocalCredSourceFlags(v, injected) {
124
125
  if (local.length)
125
126
  throw new Error(`delivery: ${local.map((f) => `--${f}`).join(" and ")} cannot be combined with an injected secret store — the store is the cred's only source`);
126
127
  }
128
+ /** The workspace root a workstation daemon serves, chosen ONCE at start (#723). `--root` names it;
129
+ * otherwise it is the nearest `.cotal/` above the working directory. With neither, the daemon refuses
130
+ * here and names where it searched: the start directory is not a root anyone set up, and reading the
131
+ * cred, the registry and the $SYS pair from it only failed later, after the dial, on a symptom. An
132
+ * injected store is the composition's only credential source, so there the root holds no trust
133
+ * material and a missing `.cotal/` stays allowed, as {@link recordDeliveryPid} documents. */
134
+ function daemonRoot(v, injected) {
135
+ if (v.root !== undefined) {
136
+ const root = resolve(v.root);
137
+ if (!existsSync(join(root, ".cotal")))
138
+ throw new Error(`delivery: --root ${v.root} holds no .cotal/ - name the directory that contains the workspace's .cotal/`);
139
+ return root;
140
+ }
141
+ if (injected)
142
+ return findCotalRoot();
143
+ try {
144
+ return requireCotalRoot();
145
+ }
146
+ catch (e) {
147
+ throw new Error(`delivery: ${e.message} - run the daemon from inside its workspace, or name the root with --root <dir>`);
148
+ }
149
+ }
127
150
  /** Where the daemon's pre-minted cred lives — exactly ONE source: an injected {@link SecretStore}
128
151
  * (a hosted composition), an explicit `--creds <file>` as an FS store over that exact file
129
152
  * (uninjected `--creds` that names one real workstation while process cwd
@@ -139,7 +162,7 @@ function assertNoLocalCredSourceFlags(v, injected) {
139
162
  * The injected arm builds the key without touching a filesystem it does not have. `--creds` names
140
163
  * ONE file and is per-space by the operator's own choice of path, so it is neither segmented nor
141
164
  * migrated — the store there is rooted at that file's own directory. */
142
- function resolveCredsStore(v, space, injected) {
165
+ function resolveCredsStore(v, space, root, injected) {
143
166
  if (injected) {
144
167
  const key = segmentedKey(DELIVERY_CREDS_KIND, space);
145
168
  return {
@@ -162,7 +185,6 @@ function resolveCredsStore(v, space, injected) {
162
185
  identity: reloadStoreIdentityFromCredsPath(p, space),
163
186
  };
164
187
  }
165
- const root = findCotalRoot();
166
188
  const key = deliveryCredsKey(space, { injected: false, root });
167
189
  return {
168
190
  store: workspaceSecretStore(root),
@@ -183,7 +205,7 @@ function resolveCredsStore(v, space, injected) {
183
205
  * dev run with no stored cred can opt into `--dev-mint`, which loads the local signer and self-remints
184
206
  * a scoped `delivery` cred (one stable identity) — LOUDLY flagged as dev-only, never the production
185
207
  * contract. */
186
- async function loadDeliveryCreds(src, v) {
208
+ async function loadDeliveryCreds(src, v, root) {
187
209
  const { store, key, where } = src;
188
210
  const initial = await store.get(key);
189
211
  if (initial !== undefined) {
@@ -213,7 +235,7 @@ async function loadDeliveryCreds(src, v) {
213
235
  throw new Error(`delivery: no cred in the injected secret store under key "${key}" — the hosted composition must put it before starting the daemon (local sources are never consulted when a store is injected). The key is PER-SPACE as of P7; a cred put under the bare kind "${DELIVERY_CREDS_KIND}" is not read here.`);
214
236
  if (v["dev-mint"] !== undefined) {
215
237
  // Space-blind dev path: the root must name exactly one space (soleSpaceOf fails loud on several).
216
- const devRoot = authDir(findCotalRoot());
238
+ const devRoot = authDir(root);
217
239
  const devSpace = soleSpaceOf(devRoot);
218
240
  const auth = devSpace ? loadSpaceAuth(devRoot, devSpace) : undefined;
219
241
  if (!auth)
@@ -258,15 +280,11 @@ export function recordDeliveryPid(root, space) {
258
280
  reclaimDeadPreUpgradeRecord(DELIVERY_PIDFILE, ctx);
259
281
  const pidPath = canonicalLocalProcessPath(DELIVERY_PIDFILE, ctx);
260
282
  const mine = String(process.pid);
261
- writeFileSync(pidPath, mine);
262
- // #969: pin the pid to its process start so a later teardown can refuse a reused pid.
263
- writeIdentityPin(pidPath, process.pid);
283
+ // #969/#1238: publish the pair by rename so a later teardown never sees a torn pairing.
284
+ writePidPair(pidPath, process.pid);
264
285
  return () => {
265
286
  try {
266
- if (readFileSync(pidPath, "utf8").trim() !== mine)
267
- return; // a successor's record: not ours to remove
268
- removeIdentityPin(pidPath);
269
- rmSync(pidPath, { force: true });
287
+ removePidPair(pidPath, mine); // a successor's record is not ours to remove
270
288
  }
271
289
  catch {
272
290
  /* already gone, or unreadable: leaving a record we cannot prove is ours is the safe error */
@@ -321,7 +339,26 @@ export async function runDelivery(args, store) {
321
339
  throw e;
322
340
  }
323
341
  }
324
- async function runStartedDelivery(args, store, publishReleaser) {
342
+ /** Start one independently owned delivery context. No cwd, process signal, pidfile, or exit
343
+ * policy is borrowed from the CLI runner. The caller owns its explicit state path and store. */
344
+ export async function startDeliveryService(inputs) {
345
+ if (!inputs.context.accountPublicKey || !inputs.context.lifecycleUid || !inputs.space || !inputs.servers || !inputs.stateDir)
346
+ throw new Error("delivery: hosted context needs an account, lifecycle, space, server and stateDir");
347
+ if (inputs.store.identity === undefined)
348
+ throw new Error("delivery: hosted SecretStore must declare a stable identity");
349
+ const actual = parseSecretStoreIdentity(inputs.store.identity);
350
+ if (!sameSecretStoreIdentity(actual, inputs.storeIdentity) || actual.kind !== "injected")
351
+ throw new Error("delivery: hosted SecretStore identity does not match the assigned store identity");
352
+ let releaseOnStartupFailure;
353
+ try {
354
+ return await runStartedDelivery({ values: { space: inputs.space, server: inputs.servers }, positionals: [], raw: [] }, inputs.store, (release) => { releaseOnStartupFailure = release; }, inputs);
355
+ }
356
+ catch (error) {
357
+ await releaseOnStartupFailure?.();
358
+ throw error;
359
+ }
360
+ }
361
+ async function runStartedDelivery(args, store, publishReleaser, hosted) {
325
362
  const v = args.values;
326
363
  const shard = v.shard ? Number(v.shard) : 0;
327
364
  const shards = v.shards ? Number(v.shards) : 1;
@@ -333,14 +370,19 @@ async function runStartedDelivery(args, store, publishReleaser) {
333
370
  // the source now needs the space that this check must precede; the check is therefore split out
334
371
  // and stays here, ahead of everything ambient.
335
372
  assertNoLocalCredSourceFlags(v, store);
373
+ // Every workstation read below (cred, registry, $SYS pair, membership feed, pidfile) uses this one
374
+ // root. A hosted context names its own state directory and borrows no cwd.
375
+ const root = hosted?.stateDir ?? daemonRoot(v, store);
336
376
  // Space comes from --space (the CLI passes it). Only --dev-mint may derive it from the local signer.
337
- const space = v.space ?? (v["dev-mint"] !== undefined ? soleSpaceOf(authDir(findCotalRoot())) : undefined);
377
+ const space = v.space ?? (v["dev-mint"] !== undefined ? soleSpaceOf(authDir(root)) : undefined);
338
378
  if (!space)
339
379
  throw new Error("delivery: --space is required (the scoped creds file does not encode it)");
340
- const credsSrc = resolveCredsStore(v, space, store);
380
+ const credsSrc = resolveCredsStore(v, space, root, store);
341
381
  if (v.creds !== undefined)
342
- assertUninjectedCredsSharesCwdRoot({ injected: credsSrc.injected, credsPath: resolve(v.creds) });
343
- const creds = await loadDeliveryCreds(credsSrc, v); // pre-minted scoped cred; NO signer/loadSpaceAuth in this path
382
+ assertUninjectedCredsSharesCwdRoot({ injected: credsSrc.injected, credsPath: resolve(v.creds), cwdRoot: root, via: v.root !== undefined ? "--root" : "process cwd" });
383
+ const creds = await loadDeliveryCreds(credsSrc, v, root); // pre-minted scoped cred; NO signer/loadSpaceAuth in this path
384
+ if (hosted !== undefined && accountFromCreds(creds.initial) !== hosted.context.accountPublicKey)
385
+ throw new Error("delivery: scoped credential account does not match the assigned hosted context");
344
386
  let latestCreds = creds.initial; // freshest renewal — the broker-reachability poll below presents it
345
387
  // WHICH BROKER, WORKSTATION COMPOSITION (#756, the deliver half). `cotal ps`/`spawn`/`attach`/
346
388
  // `status`/`supervise` all resolve the registered mesh for the space/root and dial its broker;
@@ -358,7 +400,6 @@ async function runStartedDelivery(args, store, publishReleaser) {
358
400
  let server = v.server ?? DEFAULT_SERVER;
359
401
  let registered;
360
402
  if (store === undefined) {
361
- const root = findCotalRoot();
362
403
  try {
363
404
  registered = resolveMeshTarget(root, { space });
364
405
  }
@@ -407,9 +448,13 @@ async function runStartedDelivery(args, store, publishReleaser) {
407
448
  console.error(registered.origin === "manual" || registered.origin === "catalog"
408
449
  ? `✗ delivery: no broker answered at ${server} - "${space}" is registered here but its mesh is not up; start it where it runs, or \`cotal meshes rm ${space}\` to unregister it`
409
450
  : `✗ delivery: no mesh running at ${server} - mesh "${space}" is recorded at ${registered.root} but not running; run \`cotal up\` there to restart`);
451
+ if (hosted !== undefined)
452
+ throw new Error(`delivery: no broker answered at ${server}`);
410
453
  process.exit(1);
411
454
  }
412
455
  console.error(`✗ delivery: can't reach NATS at ${server}. Run: cotal up`);
456
+ if (hosted !== undefined)
457
+ throw new Error(`delivery: can't reach NATS at ${server}`);
413
458
  process.exit(1);
414
459
  }
415
460
  // PIN THE SCAN TARGET ONCE, HERE. The $SYS sweeps below resolve an ACCOUNT, and a complete sweep
@@ -424,12 +469,12 @@ async function runStartedDelivery(args, store, publishReleaser) {
424
469
  // is known for certain, rather than inferred later by probing the store or sniffing the
425
470
  // filesystem, both of which report "workstation" for a hosted daemon and would emit a CLI repair
426
471
  // the host cannot run. It selects the REPAIR IDIOM only; failure semantics never fork on it.
427
- const scanRoot = findCotalRoot();
428
472
  const scanTarget = {
429
- root: scanRoot,
473
+ root,
474
+ named: v.root !== undefined,
430
475
  expectedAccount: accountFromCreds(creds.initial),
431
476
  source: store === undefined
432
- ? { secrets: workspaceSecretStore(scanRoot), space, injected: false, root: scanRoot }
477
+ ? { secrets: workspaceSecretStore(root), space, injected: false, root }
433
478
  : { secrets: store, space, injected: true },
434
479
  };
435
480
  // The singleton lease is admission to serve this space, so the daemon must prove its scan tenancy
@@ -456,6 +501,8 @@ async function runStartedDelivery(args, store, publishReleaser) {
456
501
  watchPresence: true, // read the roster for @mention resolution …
457
502
  registerPresence: false, // … but NEVER publish the daemon onto the roster (it's infra, not a peer)
458
503
  card: { id: ownId, name: "delivery", role: "delivery", kind: "endpoint" },
504
+ transportPingIntervalMs: Number(process.env.COTAL_DELIVERY_PING_INTERVAL_MS) || 2500,
505
+ transportMaxPingOut: Number(process.env.COTAL_DELIVERY_MAX_PING_OUT) || 2,
459
506
  });
460
507
  // Both channels: raw connection errors ride `error`, while every condition the endpoint is
461
508
  // already surviving — a failed 75% renewal, the passive backstop's "still holds the previous
@@ -464,7 +511,49 @@ async function runStartedDelivery(args, store, publishReleaser) {
464
511
  const say = (e) => console.error(`! delivery endpoint: ${e.message}`);
465
512
  ep.on("error", say);
466
513
  ep.on("warning", say);
467
- await ep.start();
514
+ try {
515
+ await ep.start();
516
+ }
517
+ catch (error) {
518
+ await ep.stop();
519
+ throw error;
520
+ }
521
+ const BROKER_GONE_MS = Number(process.env.COTAL_DELIVERY_BROKER_GONE_MS) || 15_000;
522
+ const BROKER_GONE_BACKSTOP_MS = Math.max(BROKER_GONE_MS, Number(process.env.COTAL_DELIVERY_BROKER_GONE_BACKSTOP_MS) || BROKER_GONE_MS * 4);
523
+ let stopHandler;
524
+ let hostedStartupFailure;
525
+ let stopHosted;
526
+ const health = new DeliveryTransportHealth((reason) => {
527
+ console.error(reason === "credential-expired"
528
+ ? "✗ delivery: credential expired without renewal"
529
+ : reason === "backstop"
530
+ ? "✗ delivery: broker connection unavailable past backstop (broker unreachable), exiting (coupled to the broker)"
531
+ : "✗ delivery: broker connection unavailable (broker unreachable), exiting (coupled to the broker)");
532
+ const cause = reason === "credential-expired" ? "credential expired without renewal" : `broker connection unavailable (${reason})`;
533
+ if (hosted !== undefined) {
534
+ hostedStartupFailure ??= new Error(`delivery context stopped during startup: ${cause}`);
535
+ if (stopHosted !== undefined)
536
+ stopHosted(cause);
537
+ else
538
+ void ep.stop(); // before lease acquisition, no context-owned lease can be released
539
+ return;
540
+ }
541
+ if (stopHandler !== undefined)
542
+ stopHandler(1, cause);
543
+ else
544
+ earlyStop(1);
545
+ }, () => console.error("! delivery: credential expired, awaiting proved renewal"), BROKER_GONE_MS, BROKER_GONE_BACKSTOP_MS);
546
+ ep.on("transport", ({ connected }) => {
547
+ health.transport(connected);
548
+ });
549
+ ep.on("error", (e) => {
550
+ if (e.name === "UserAuthenticationExpiredError" || e.cause?.name === "UserAuthenticationExpiredError") {
551
+ health.credentialExpired();
552
+ }
553
+ });
554
+ ep.on("creds-adopted", () => {
555
+ health.adopted();
556
+ });
468
557
  // Acquire the single-flight lease BEFORE binding the loops: a loud refusal-to-bind if another daemon
469
558
  // already holds this shard (two clients binding the same durable name SPLIT delivery). The bucket TTL
470
559
  // frees a crashed holder's lease so a fresh daemon re-acquires.
@@ -498,6 +587,8 @@ async function runStartedDelivery(args, store, publishReleaser) {
498
587
  console.error(`✗ delivery: acquiring the lease for shard ${shard} failed: ${e.message}. Not binding.`);
499
588
  }
500
589
  await ep.stop();
590
+ if (hosted !== undefined)
591
+ throw e;
501
592
  process.exit(1);
502
593
  return;
503
594
  }
@@ -521,6 +612,13 @@ async function runStartedDelivery(args, store, publishReleaser) {
521
612
  // it is provably ours, then exit. Once the real `shutdown` is installed it REPLACES this, and
522
613
  // `stopping` makes the pair idempotent, so a signal arriving mid-swap cannot run both.
523
614
  let stopping = false;
615
+ let unavailable;
616
+ let closePromise;
617
+ let membership;
618
+ let timerWriter;
619
+ let timerWriterAttempt;
620
+ let stopLeaseWatch;
621
+ let renew;
524
622
  /** Removes THIS process's liveness record, once it has one. Declared BEFORE the handlers that
525
623
  * call it and assigned below, so a signal landing between the two is a no-op rather than a
526
624
  * temporal-dead-zone throw: at that instant nothing has been written, so there is nothing to
@@ -531,6 +629,7 @@ async function runStartedDelivery(args, store, publishReleaser) {
531
629
  if (stopping)
532
630
  return;
533
631
  stopping = true;
632
+ health.stop();
534
633
  setTimeout(() => process.exit(code), 2000);
535
634
  unrecordPid?.();
536
635
  void (async () => {
@@ -547,10 +646,20 @@ async function runStartedDelivery(args, store, publishReleaser) {
547
646
  process.exit(code);
548
647
  })();
549
648
  };
550
- const earlySigint = () => earlyStop(0);
551
- const earlySigterm = () => earlyStop(0);
552
- process.on("SIGINT", earlySigint);
553
- process.on("SIGTERM", earlySigterm);
649
+ stopHandler = earlyStop;
650
+ // A SIGNALLED STOP NAMES ITSELF BEFORE TEARDOWN (#1443). Every other deliberate exit writes its
651
+ // reason first; without this line a `cotal down`, a systemd stop or Ctrl-C left the log ending on
652
+ // routine work, which an operator cannot tell from a silent death.
653
+ const onSignal = (signal, stop) => () => {
654
+ console.error(`• delivery: received ${signal}, exiting (space ${space}, shard ${shard})`);
655
+ stop(0);
656
+ };
657
+ const earlySigint = onSignal("SIGINT", earlyStop);
658
+ const earlySigterm = onSignal("SIGTERM", earlyStop);
659
+ if (hosted === undefined) {
660
+ process.on("SIGINT", earlySigint);
661
+ process.on("SIGTERM", earlySigterm);
662
+ }
554
663
  // THE LIVENESS RECORD GOES HERE, AFTER THE ACQUIRE AND NOT BEFORE IT (#1528).
555
664
  //
556
665
  // The record answers "which process is this space's delivery daemon", and until this line this
@@ -566,7 +675,8 @@ async function runStartedDelivery(args, store, publishReleaser) {
566
675
  // own `ready` flag already carries it; the pidfile only ever claimed a process.
567
676
  //
568
677
  // Placed one statement after the signal handlers, so every exit path from here on can remove it.
569
- unrecordPid = recordDeliveryPid(findCotalRoot(), space);
678
+ if (hosted === undefined)
679
+ unrecordPid = recordDeliveryPid(root, space);
570
680
  // AND THE SAME RELEASE, REACHABLE BY AN ORDINARY `catch`. The two process-level guards below
571
681
  // only see a fault that reaches the RUNTIME. A start-up rejection on the public CLI path does
572
682
  // not: `runCli` awaits this function inside its own try/catch (cli/src/command.ts), so the
@@ -577,22 +687,58 @@ async function runStartedDelivery(args, store, publishReleaser) {
577
687
  // exists". So the release is published HERE, to a scope that a plain `catch` around the rest of
578
688
  // start-up can reach, and the process guards stay as the backstop for faults that never become
579
689
  // a rejection this function can see (a synchronous throw in a timer, say).
580
- publishReleaser(async () => {
581
- if (stopping)
582
- return;
690
+ const close = () => {
691
+ if (closePromise !== undefined)
692
+ return closePromise;
583
693
  stopping = true;
584
- unrecordPid?.(); // the record dies with the daemon it describes, including on a start-up failure
585
- try {
586
- const own = await ep.readDeliveryLeaseEntry(shard);
587
- if (own !== undefined && ep.ownsDeliveryLease(own.info))
588
- await ep.releaseDeliveryLease(shard, own.revision);
589
- }
590
- catch { /* broker may be gone - the bucket TTL is the crash-safe release authority */ }
591
- try {
592
- await ep.stop();
593
- }
594
- catch { /* broker may be gone */ }
595
- });
694
+ if (renew !== undefined)
695
+ clearInterval(renew);
696
+ health.stop();
697
+ stopLeaseWatch?.();
698
+ unrecordPid?.();
699
+ closePromise = (async () => {
700
+ try {
701
+ await ep.quiescePlane3();
702
+ }
703
+ catch { /* stop still releases the connection */ }
704
+ try {
705
+ const own = await ep.readDeliveryLeaseEntry(shard);
706
+ if (own !== undefined && ep.ownsDeliveryLease(own.info))
707
+ await ep.releaseDeliveryLease(shard, own.revision);
708
+ }
709
+ catch { /* broker may be gone - the bucket TTL is the crash-safe release authority */ }
710
+ try {
711
+ await membership?.stop();
712
+ }
713
+ catch { /* broker may be gone */ }
714
+ try {
715
+ await timerWriter?.handle.stop();
716
+ }
717
+ catch { /* broker may be gone */ }
718
+ try {
719
+ await Promise.race([timerWriter?.nc.drain(), new Promise((r) => setTimeout(r, 1000))]);
720
+ }
721
+ catch { /* broker may be gone */ }
722
+ try {
723
+ await timerWriterAttempt;
724
+ }
725
+ catch { /* startup or broker fault */ }
726
+ try {
727
+ await ep.stop();
728
+ }
729
+ catch { /* broker may be gone */ }
730
+ })();
731
+ return closePromise;
732
+ };
733
+ publishReleaser(close);
734
+ stopHosted = (cause) => {
735
+ unavailable = `delivery context stopped (code 1): ${cause}`;
736
+ void close();
737
+ };
738
+ if (hostedStartupFailure !== undefined) {
739
+ await close();
740
+ throw hostedStartupFailure;
741
+ }
596
742
  // A START-UP FAILURE AFTER THE ACQUIRE MUST ALSO GIVE THE SHARD BACK. Between this point and the
597
743
  // handler swap far below, a throw would otherwise propagate out of `runDelivery` with the row
598
744
  // still claiming the shard, stranding it for the bucket TTL exactly as an unhandled signal did -
@@ -618,15 +764,21 @@ async function runStartedDelivery(args, store, publishReleaser) {
618
764
  };
619
765
  const earlyUncaught = (e) => earlyFault("faulted", e);
620
766
  const earlyRejection = (e) => earlyFault("rejected a promise", e);
621
- process.on("uncaughtException", earlyUncaught);
622
- process.on("unhandledRejection", earlyRejection);
767
+ if (hosted === undefined) {
768
+ process.on("uncaughtException", earlyUncaught);
769
+ process.on("unhandledRejection", earlyRejection);
770
+ }
623
771
  // Broker-sourced graph membership handle — declared BEFORE Plane-3 so the delivery-admin reload
624
772
  // hook below can close over it (it starts further down; the closure reads it live).
625
- let membership;
626
773
  // WHY the feed is down, carried from `startMembership` so the adoption refusal below names the real
627
774
  // fault instead of its symptom. Without it, an expired $SYS observer cred surfaces to the operator
628
775
  // only as "membership feed is not running" (#338).
629
776
  let membershipDown;
777
+ // START-UP IS NOT DONE UNTIL THE LEASE WATCH IS BOUND (#2304). The ready flip below releases
778
+ // `ensureDelivery`, and so the manager, whose boot renewal pass sends `reloadCreds` at once. The
779
+ // endpoint refuses that adoption while this is false, rather than reconnecting this connection
780
+ // underneath the membership start and the lease watch that are still pending on it.
781
+ let startedUp = false;
630
782
  // Host Plane-3 (fan-out writer + trusted reader) AND serve the ctl.delivery runtime durable ops. The
631
783
  // reader re-authorizes each entry against the durable ACL registry, read FRESH per entry. The
632
784
  // delivery-admin rail's `reloadCreds` (explicit class-2 adoption) also reloads the membership feed's
@@ -647,6 +799,8 @@ async function runStartedDelivery(args, store, publishReleaser) {
647
799
  // Live-eviction executor (D5 slice 6): per-call $SYS observer/evictor connections; refuses
648
800
  // loudly on a pre-evictor space. Rare repair/flip step — never a standing $SYS conn here.
649
801
  evictPrincipal: (principal) => executeEviction(server, scanTarget, principal),
802
+ // The same executor over a set: one shared sweep for a large credential family's holders.
803
+ evictPrincipals: (principals) => executeEvictions(server, scanTarget, principals),
650
804
  // Plane-claim liveness oracle (#29 HIGH 3): read-only $SYS CONNZ per call; the auth plane's
651
805
  // stale-claim reclaim gates on this verdict (any refusal/unknown blocks takeover, fail-closed).
652
806
  planeConnLiveness: (query) => executePlaneLiveness(server, scanTarget, query),
@@ -655,14 +809,30 @@ async function runStartedDelivery(args, store, publishReleaser) {
655
809
  // killing it to discover it was alive (any refusal/unknown blocks the repair, fail-closed).
656
810
  principalLiveness: (principal) => executePrincipalLiveness(server, scanTarget, principal),
657
811
  reloadStoreIdentity: () => reloadStoreIdentity,
812
+ onDeliveryCredsAdopted: () => health.adopted(),
813
+ startupComplete: () => startedUp,
658
814
  });
815
+ // The peer-readable liveness responder (#1577), bound BEFORE the ready flip below and
816
+ // deliberately so. It answers from the lease at PROBE time, so during the window between binding
817
+ // the loops and flipping ready it answers `unbound` — which is the truth, and is the whole point:
818
+ // a surface that only came up once everything was healthy could never report the unhealthy state
819
+ // it exists to report. Presence only; the lease row never crosses the wire.
820
+ ep.serveDeliveryLiveness(shard);
659
821
  // Flip the lease to READY only now — after the loops + ctl.delivery responder are bound — so readiness
660
822
  // waiters (ensureDelivery) and the cotal_channels health surface see "ready" iff the responder is up,
661
823
  // not merely that the single-flight slot was claimed.
662
824
  try {
663
825
  revision = await ep.markDeliveryLeaseReady(shard, revision);
664
826
  }
665
- catch { /* lost the lease between acquire and ready — the renew loop's CAS failure will exit us */ }
827
+ catch (error) {
828
+ if (hosted !== undefined)
829
+ throw error;
830
+ /* CLI: the renew loop's CAS failure will exit us if the lease was lost. */
831
+ }
832
+ if (hostedStartupFailure !== undefined) {
833
+ await close();
834
+ throw hostedStartupFailure;
835
+ }
666
836
  console.log(`✓ delivery daemon up (space ${space}${shards > 1 ? `, shard ${shard}/${shards}` : ""}) — stop with: cotal down`);
667
837
  // Broker-sourced graph membership: a SEPARATE module on its OWN connections (system-account CONNZ
668
838
  // reader + data-account feed writer), isolated from Plane-3. Fail-soft — a missing cred / start error
@@ -675,12 +845,23 @@ async function runStartedDelivery(args, store, publishReleaser) {
675
845
  // `accountId` is the account THIS daemon's own cred authenticates as (pinned above), not a
676
846
  // `.cotal/membership.json` read: the file was never an independent source (same directory as the
677
847
  // creds, so a wrong root was wrong for both) and a hosted composition has none.
678
- ({ handle: membership, down: membershipDown } = await startMembership({ space, server, accountId: scanTarget.expectedAccount }, store));
848
+ ({ handle: membership, down: membershipDown } = await startMembership({ space, server, accountId: scanTarget.expectedAccount, root }, store));
679
849
  }
680
850
  catch (e) {
681
851
  membershipDown = e.message;
682
852
  console.error(`! membership: failed to start (${membershipDown}); graph membership degraded, delivery unaffected`);
683
853
  }
854
+ // A store read can finish after a startup health fault closed the endpoint. Dispose the feed
855
+ // it just returned before rejecting; it did not exist when the original close ran.
856
+ if (hostedStartupFailure !== undefined) {
857
+ try {
858
+ await membership?.stop();
859
+ }
860
+ finally {
861
+ await close();
862
+ }
863
+ throw hostedStartupFailure;
864
+ }
684
865
  // The TIMER WRITER (SPEC 13.2): the pump that turns workflow `.schedule` requests into armed
685
866
  // broker schedules. Hosted here because this daemon is the space's standing server-side process;
686
867
  // without a running writer no pause on the space ever expires. Its OWN connection under the same
@@ -688,8 +869,7 @@ async function runStartedDelivery(args, store, publishReleaser) {
688
869
  // module-per-connection isolation), re-dialed with the FRESHEST cred by a supervised restart
689
870
  // loop: a fault never kills delivery, is never silent (each attempt logs why the space cannot
690
871
  // expire pauses right now), and a cred that gains rows at renewal is picked up on the next dial.
691
- let timerWriter;
692
- void (async () => {
872
+ timerWriterAttempt = (async () => {
693
873
  let backoffMs = 5_000;
694
874
  for (;;) {
695
875
  if (stopping)
@@ -702,7 +882,16 @@ async function runStartedDelivery(args, store, publishReleaser) {
702
882
  name: "cotal-delivery-timer-writer",
703
883
  maxReconnectAttempts: -1,
704
884
  });
885
+ if (stopping) {
886
+ await nc.close();
887
+ return;
888
+ }
705
889
  const handle = await startTimerWriter(nc, space);
890
+ if (stopping) {
891
+ await handle.stop();
892
+ await nc.close();
893
+ return;
894
+ }
706
895
  timerWriter = { handle, nc };
707
896
  backoffMs = 5_000;
708
897
  console.log(`✓ timer writer up (space ${space}) — workflow pauses on this space expire`);
@@ -710,8 +899,13 @@ async function runStartedDelivery(args, store, publishReleaser) {
710
899
  return;
711
900
  }
712
901
  catch (e) {
713
- if (stopping)
902
+ if (stopping) {
903
+ try {
904
+ await nc?.close();
905
+ }
906
+ catch { /* already closed */ }
714
907
  return;
908
+ }
715
909
  console.error(`! timer writer: down (${e.message}) — retrying in ${Math.round(backoffMs / 1000)}s; pauses on this space do not expire until it is back`);
716
910
  try {
717
911
  await nc?.drain();
@@ -723,17 +917,24 @@ async function runStartedDelivery(args, store, publishReleaser) {
723
917
  }
724
918
  }
725
919
  })();
726
- const shutdown = (code) => {
920
+ const shutdown = (code, cause) => {
921
+ if (hosted !== undefined) {
922
+ unavailable = cause ? `delivery context stopped (code ${code}): ${cause}` : `delivery context stopped (code ${code})`;
923
+ void close();
924
+ return;
925
+ }
727
926
  if (stopping)
728
927
  return;
729
928
  stopping = true;
730
- clearInterval(renew);
731
- clearInterval(brokerWatch);
732
- stopLeaseWatch();
929
+ if (renew !== undefined)
930
+ clearInterval(renew);
931
+ health.stop();
932
+ stopLeaseWatch?.();
733
933
  // THE RECORD GOES FIRST, AND SYNCHRONOUSLY. Everything below this line talks to a broker that
734
934
  // may be dead and is bounded only by the 2s hard exit; a record left behind because a drain
735
935
  // hung is a record that outlives its process, which is this issue's defect re-entering through
736
- // the exit path. `rmSync` needs no broker and cannot hang, so it runs before any of it.
936
+ // the exit path. Removing it needs no broker, and waits only on a publish of the same record
937
+ // that is in flight (#1238), so it runs before any of it.
737
938
  unrecordPid?.();
738
939
  // Hard-exit fallback: a graceful release/stop talks to the broker, which may be DEAD (the broker-gone
739
940
  // exit path) — don't let that hang the process. Force exit if the graceful path doesn't finish quickly.
@@ -797,6 +998,7 @@ async function runStartedDelivery(args, store, publishReleaser) {
797
998
  process.exit(code);
798
999
  })();
799
1000
  };
1001
+ stopHandler = shutdown;
800
1002
  // SIGNAL HANDLERS GO UP THE MOMENT `shutdown` EXISTS, NOT AFTER STARTUP FINISHES.
801
1003
  //
802
1004
  // The lease is acquired long before this point, and registration used to sit at the very end of
@@ -818,16 +1020,18 @@ async function runStartedDelivery(args, store, publishReleaser) {
818
1020
  // knows nothing of the renew timer, the membership feed or the timer writer - would run first and
819
1021
  // set `stopping`, making the full teardown a no-op. `stopping` still guards the swap itself, so a
820
1022
  // signal delivered between these two statements is handled exactly once.
821
- process.off("SIGINT", earlySigint);
822
- process.off("SIGTERM", earlySigterm);
823
- // The start-up fault guards go too, and BY REFERENCE: `removeAllListeners` here would strip
824
- // listeners this daemon does not own. They exist to cover the window where no `shutdown` exists;
825
- // past this line a fault should surface normally rather than become a quiet exit that skips the
826
- // full teardown.
827
- process.off("uncaughtException", earlyUncaught);
828
- process.off("unhandledRejection", earlyRejection);
829
- process.on("SIGINT", () => shutdown(0));
830
- process.on("SIGTERM", () => shutdown(0));
1023
+ if (hosted === undefined) {
1024
+ process.off("SIGINT", earlySigint);
1025
+ process.off("SIGTERM", earlySigterm);
1026
+ // The start-up fault guards go too, and BY REFERENCE: `removeAllListeners` here would strip
1027
+ // listeners this daemon does not own. They exist to cover the window where no `shutdown` exists;
1028
+ // past this line a fault should surface normally rather than become a quiet exit that skips the
1029
+ // full teardown.
1030
+ process.off("uncaughtException", earlyUncaught);
1031
+ process.off("unhandledRejection", earlyRejection);
1032
+ process.on("SIGINT", onSignal("SIGINT", shutdown));
1033
+ process.on("SIGTERM", onSignal("SIGTERM", shutdown));
1034
+ }
831
1035
  /** What the broker says about THIS shard's lease key right now, the verdict a failed renew does
832
1036
  * NOT have. `unknown` never collapses into `gone`: not being able to look is not the same fact as
833
1037
  * looking and finding nothing (#1318).
@@ -896,6 +1100,8 @@ async function runStartedDelivery(args, store, publishReleaser) {
896
1100
  * caller has just proven ownership; `why` is what proved it, since "it started serving again" is
897
1101
  * only auditable next to the evidence that permitted it. */
898
1102
  const resumeServing = async (why) => {
1103
+ if (stopping)
1104
+ return;
899
1105
  try {
900
1106
  await ep.rearmPlane3();
901
1107
  }
@@ -950,7 +1156,7 @@ async function runStartedDelivery(args, store, publishReleaser) {
950
1156
  // establish it refuses to start, loudly (the await below throws out of start-up and the
951
1157
  // releaser gives the shard back), rather than degrade into the polled window this issue is
952
1158
  // about. The stop handle is cleared on every exit path the renew interval is: `shutdown`.
953
- const stopLeaseWatch = await ep.watchDeliveryLease(shard, (info) => {
1159
+ stopLeaseWatch = await ep.watchDeliveryLease(shard, (info) => {
954
1160
  if (stopping)
955
1161
  return;
956
1162
  // The row as this event states it: an event for a row we own answers ITSELF (no state
@@ -1009,6 +1215,12 @@ async function runStartedDelivery(args, store, publishReleaser) {
1009
1215
  }
1010
1216
  })().catch((e) => console.error(`! delivery: the lease watch trigger faulted (${e.message}); the renew interval remains the arbiter`));
1011
1217
  });
1218
+ if (hosted !== undefined && stopping) {
1219
+ stopLeaseWatch?.();
1220
+ await close();
1221
+ throw hostedStartupFailure ?? new Error(unavailable ?? "delivery context stopped during startup");
1222
+ }
1223
+ startedUp = true;
1012
1224
  // Renew the lease at ~half the TTL so a healthy holder never self-evicts.
1013
1225
  //
1014
1226
  // A FAILED RENEW IS A QUESTION, NOT A VERDICT (#1318). The shipped code exited on ANY renew
@@ -1021,7 +1233,7 @@ async function runStartedDelivery(args, store, publishReleaser) {
1021
1233
  // second refused over a sequence the first legitimately moved, a conflict this daemon manufactures
1022
1234
  // itself and then reads as someone else's takeover.
1023
1235
  let renewInFlight = false;
1024
- const renew = setInterval(() => {
1236
+ renew = setInterval(() => {
1025
1237
  if (stopping || renewInFlight)
1026
1238
  return;
1027
1239
  renewInFlight = true;
@@ -1126,187 +1338,19 @@ async function runStartedDelivery(args, store, publishReleaser) {
1126
1338
  }
1127
1339
  })();
1128
1340
  }, Math.max(1000, Math.floor(LEASE_TTL_MS / 2)));
1129
- // Coupled to the broker: POLL its reachability. Survive brief blips (the endpoint reconnects on its
1130
- // own), but EXIT if the broker is GONE, the endpoint would otherwise retry reconnect forever (its
1131
- // terminal-close never fires), so this is what stops the daemon outliving the server it serves.
1132
- // (`cotal up`/`down` teardown stops it too.) The window is env-overridable for tests.
1133
- //
1134
- // WHAT "GONE" MEANS IS NOW DECIDED FROM EVIDENCE, NOT FROM A CLOCK (#1318). Three conditions used
1135
- // to produce one signal and only one of them was a dead server: the broker being down, this
1136
- // process being descheduled so the interval never fired, and a probe that could not complete a
1137
- // handshake because the local process could not get scheduled to finish it. A wall-clock
1138
- // `Date.now() - lastReachable` cannot tell them apart, and widening it only moves the threshold.
1139
- // Three signals separate them, and all three are things this process can actually observe:
1140
- //
1141
- // • MEASURED LOOP LAG. The gap between consecutive firings of a timer we own, minus its nominal
1142
- // period, is local starvation by direct measurement. It is credited back, so time this process
1143
- // spent off the runqueue is never counted against the broker.
1144
- // • COMPLETED NEGATIVE PROBES. Only a probe that RAN TO COMPLETION and returned false is
1145
- // evidence about the server. A probe that never ran contributes nothing (that was the whole
1146
- // defect: the window aged with no probe having failed), and one that REJECTED is an
1147
- // unanswered question, it now resets the counter and logs, where it used to be swallowed by
1148
- // `.catch(() => {})` and silently age the window.
1149
- // • TRANSPORT-LEVEL LIVENESS. A `transport: connected` edge from the endpoint's own connection
1150
- // cannot happen without a server on the other end, so it is positive evidence obtained for
1151
- // free, on a path that does not need this process to schedule a probe at all.
1152
- //
1153
- // A starvation diagnosis therefore reports DEGRADED and keeps serving; a genuinely dead broker
1154
- // still exits, on the same window, as fast as completed probes can say so.
1155
- const BROKER_GONE_MS = Number(process.env.COTAL_DELIVERY_BROKER_GONE_MS) || 15_000;
1156
- // How many completed negatives make elapsed time believable. Two is the floor: one completed
1157
- // negative is a single refused connect, which a loopback under momentary pressure can produce.
1158
- // The default scales with the window so a test that shortens the window does not thereby demand
1159
- // more evidence than the window has room for.
1160
- const BROKER_GONE_PROBES = Math.max(2, Number(process.env.COTAL_DELIVERY_BROKER_GONE_PROBES) || Math.ceil(BROKER_GONE_MS / PROBE_INTERVAL_MS / 2));
1161
- // The hard backstop: past this, no amount of measured lag or transport optimism keeps the daemon
1162
- // alive. A daemon that OUTLIVES a dead broker is worse than one that restarts unnecessarily, so
1163
- // the repair is bounded and fails toward exiting.
1164
- const BROKER_GONE_BACKSTOP_MS = Math.max(BROKER_GONE_MS, Number(process.env.COTAL_DELIVERY_BROKER_GONE_BACKSTOP_MS) || BROKER_GONE_MS * 4);
1165
- let lastReachable = Date.now();
1166
- let completedNegatives = 0;
1167
- let degraded = false;
1168
- // What the endpoint's OWN connection reports about its socket to this same broker. Seeded true
1169
- // because `ep.start()` above completed, which it cannot do without a server having answered.
1170
- let transportConnected = true;
1171
- const lag = new LoopLagMeter(PROBE_INTERVAL_MS);
1172
- /** Positive evidence, from wherever it came: restart the window and its lag budget together. */
1173
- const sawBroker = () => {
1174
- lastReachable = Date.now();
1175
- completedNegatives = 0;
1176
- lag.reset();
1177
- if (degraded) {
1178
- degraded = false;
1179
- console.error(`✓ delivery: the broker is answering again, Plane-3 is serving normally (space ${space})`);
1180
- }
1181
- };
1182
- // The endpoint's OWN transport edge. This is evidence the daemon gets without being scheduled to
1183
- // probe for it, and it is the signal that distinguishes "the connection object reports a
1184
- // transport-level close" from "silence": a live connection to that address proves a server, while
1185
- // a disconnect is the endpoint's own business (nats.js reconnects through blips of its own
1186
- // accord) and merely stops excusing a probe that will not complete.
1187
- ep.on("transport", (t) => {
1188
- transportConnected = t.connected;
1189
- if (t.connected)
1190
- sawBroker();
1191
- });
1192
- // THE BUDGET THE PROBE IS ACTUALLY GIVEN, asked of the one function that decides it. A ws(s)
1193
- // broker rides an HTTPS edge and gets 5s, not the 1s a loopback TCP broker gets; judging a ws
1194
- // probe against 1000 would read every honest refusal on such a broker as this process's own
1195
- // starvation, so the completed-negative count could never rise and a genuinely dead ws broker
1196
- // would be ended only by the backstop, with the wrong reason in its log. The budget and the
1197
- // judgment have to come from the same place.
1198
- const probeBudgetMs = defaultProbeTimeoutMs(server);
1199
- const brokerWatch = setInterval(() => {
1200
- if (stopping)
1201
- return;
1202
- // Measure FIRST, before any await: this is the gap since the previous firing, and it is the
1203
- // only place the daemon can learn that it was not scheduled.
1204
- lag.tick(Date.now());
1205
- // THE SAME TRANSPORT AS EVERY OTHER DIAL IN THIS PROCESS. This poll carries `latestCreds` — a
1206
- // standing credential — and `isReachable` performs a real authenticated connect whenever creds
1207
- // are supplied, every 2 seconds, for the life of the daemon. It is the most repeated credential
1208
- // presentation in the system, and it was the one dial here that did not name its transport.
1209
- //
1210
- // The poll predates TLS and is identical on `main`, where nothing is encrypted and it is at
1211
- // least consistent. What is new is the ASYMMETRY: with the two dials above upgraded, an operator
1212
- // who enables TLS would get a protected main path and an unprotected watchdog. An inconsistent
1213
- // guarantee is worse than a uniformly absent one, because the operator now believes something.
1214
- const probeStarted = Date.now();
1215
- // Watch this process's own scheduling FOR THE DURATION OF THE PROBE. A refusal is only evidence
1216
- // about the server if the server was actually given the time the deadline promised it, and
1217
- // under short CPU slices most of a probe's wall-clock can be time this process was not running.
1218
- const sampler = new DescheduleSampler();
1219
- sampler.start(probeStarted);
1220
- void isReachable(server, { creds: latestCreds, ...(tls ? { tls: true } : {}) })
1221
- .then((ok) => classifyProbe(ok, Date.now() - probeStarted, probeBudgetMs, PROBE_LATE_FACTOR, sampler.stop()),
1222
- // A REJECTED probe is an unanswered question, not a negative answer. It used to be swallowed
1223
- // whole by `.catch(() => {})`, so it neither refreshed the window nor evaluated anything and
1224
- // silently aged the daemon toward an exit it had gathered no evidence for.
1225
- (e) => {
1226
- sampler.stop();
1227
- console.error(`! delivery: the broker probe did not complete (${e.message}), no verdict from it; serving, retrying`);
1228
- return classifyProbe(undefined, Date.now() - probeStarted);
1229
- })
1230
- .then((probe) => {
1231
- if (stopping)
1232
- return;
1233
- if (probe.counts === "positive") {
1234
- sawBroker();
1235
- return;
1236
- }
1237
- if (probe.counts === "incomplete") {
1238
- // The run of credible refusals is broken by a probe that did not complete.
1239
- completedNegatives = 0;
1240
- return;
1241
- }
1242
- if (probe.counts === "starved") {
1243
- // The probe RAN and said no, but its answer arrived so far past its own deadline that the
1244
- // deadline was enforced against this process rather than against the server. That is the
1245
- // starved-client case, and it is the one an elapsed-time predicate cannot see at all: the
1246
- // timer is firing, the probes are completing, and every one of them is `false`.
1247
- // Dated with the instant the answer ARRIVED, so the meter can tell whether this stall is
1248
- // the same wall-clock interval the tick above already charged (union, counted once) or an
1249
- // adjacent one (disjoint, both counted). The overlap question is settled by timestamps
1250
- // rather than by comparing magnitudes, which cannot distinguish the two.
1251
- lag.credit(probe.lateBy, Date.now());
1252
- completedNegatives = 0;
1253
- }
1254
- else {
1255
- // A refusal that arrived on time. This is the only thing that may accrue against the broker.
1256
- completedNegatives += 1;
1257
- }
1258
- // ONE SPAN, MEASURED ONCE, USED FOR BOTH. Reading the clock twice would let the elapsed span
1259
- // and the lag clamp disagree by the cost of the call itself.
1260
- const sinceReachable = Date.now() - lastReachable;
1261
- const verdict = brokerGoneVerdict({
1262
- msSinceLastReachable: sinceReachable,
1263
- // CLAMPED TO THE SPAN IT IS SUBTRACTED FROM. The interval gap and a late probe answer can
1264
- // testify to the same stall, and the meter cannot tell that from two adjacent stalls, so it
1265
- // keeps both charges and the bound is applied here, where the span is known. Without it a
1266
- // 10s stall read 17s and drove unstarved time negative, which cannot be cleared by any
1267
- // amount of real outage: a dead broker then survives the evidence clause and exits only on
1268
- // the backstop, 44s -> 62s measured. Time not had cannot exceed time passed.
1269
- starvedMs: lag.starvedMsWithin(sinceReachable),
1270
- completedNegatives,
1271
- transportConnected,
1272
- windowMs: BROKER_GONE_MS,
1273
- requiredNegatives: BROKER_GONE_PROBES,
1274
- backstopMs: BROKER_GONE_BACKSTOP_MS,
1275
- });
1276
- if (verdict.exit) {
1277
- // SAY WHICH EXIT THIS IS. The single message here previously claimed completed probes over
1278
- // ">15s of unstarved time" on BOTH paths, and on the backstop path that sentence is simply
1279
- // false: the backstop fires precisely when the evidence clauses did NOT conclude, often
1280
- // with zero completed refusals. An operator debugging a starved host was handed a
1281
- // confident evidentiary claim the daemon had not established.
1282
- console.error(verdict.reason === "backstop"
1283
- ? `✗ delivery: giving up on elapsed time alone, ${Math.round(BROKER_GONE_BACKSTOP_MS / 1000)}s since the last ` +
1284
- `confirmed reachability with no sufficient evidence either way (${completedNegatives} completed probes refused, ` +
1285
- `${Math.round(lag.starvedMsWithin(sinceReachable) / 1000)}s of local scheduler lag credited). This is a BOUND, not a diagnosis: the ` +
1286
- `broker may be gone or this process may have been starved past the bound, exiting (coupled to the broker)`
1287
- : `✗ delivery: broker unreachable, ${completedNegatives} completed probes refused within their deadline over ` +
1288
- `>${BROKER_GONE_MS / 1000}s of unstarved time (${Math.round(lag.starvedMsWithin(sinceReachable) / 1000)}s of local scheduler lag credited), exiting (coupled to the broker)`);
1289
- shutdown(1);
1290
- return;
1291
- }
1292
- // NOT an exit. Say so once per episode, naming WHICH condition it is, so an operator reading
1293
- // the log during a load incident sees "this host starved me" rather than a daemon that
1294
- // silently vanished.
1295
- if (!degraded && verdict.reason !== "reachable") {
1296
- degraded = true;
1297
- console.error(verdict.reason === "starved"
1298
- ? `! delivery: DEGRADED, cannot reach the broker, but ${Math.round(lag.starvedMsWithin(sinceReachable) / 1000)}s of that window was local scheduler lag (this host is starving this process, not the broker); serving, retrying`
1299
- : verdict.reason === "transport-live"
1300
- ? `! delivery: DEGRADED, a fresh probe cannot complete, but this daemon's own connection to ${server} is still open, so the broker is there and this process cannot ask; serving, retrying`
1301
- : `! delivery: DEGRADED, the broker has not answered for >${BROKER_GONE_MS / 1000}s but only ${completedNegatives} of ${BROKER_GONE_PROBES} probes have refused within their deadline; serving, retrying`);
1302
- }
1303
- })
1304
- .catch((e) => {
1305
- // The verdict path itself faulted. Never silent, and never an exit: a bug in the detector
1306
- // must not end the daemon it is meant to keep alive.
1307
- console.error(`! delivery: the broker watch tick faulted (${e.message}); serving, retrying`);
1308
- });
1309
- }, PROBE_INTERVAL_MS);
1310
- await new Promise(() => { }); // run until signalled
1341
+ // Native transport health is driven by DeliveryTransportHealth above (resident transport events).
1342
+ if (hosted !== undefined) {
1343
+ const context = hosted.context;
1344
+ return {
1345
+ readiness() {
1346
+ return unavailable !== undefined
1347
+ ? { state: "unavailable", context, cause: unavailable }
1348
+ : { state: stopping ? "draining" : "ready", context };
1349
+ },
1350
+ async drain() { await close(); },
1351
+ close,
1352
+ };
1353
+ }
1354
+ await new Promise(() => { }); // CLI runs until signalled
1311
1355
  }
1312
1356
  //# sourceMappingURL=delivery.js.map