@cello-protocol/daemon 0.0.208 → 0.0.210
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-handlers.d.ts.map +1 -1
- package/dist/agent-handlers.js +2 -1
- package/dist/agent-handlers.js.map +1 -1
- package/dist/attendance-wiring.js +2 -2
- package/dist/attendance-wiring.js.map +1 -1
- package/dist/bin/cello-daemon.js +11 -1
- package/dist/bin/cello-daemon.js.map +1 -1
- package/dist/boot-agents.d.ts.map +1 -1
- package/dist/boot-agents.js +5 -5
- package/dist/boot-agents.js.map +1 -1
- package/dist/close-session-handler.d.ts.map +1 -1
- package/dist/close-session-handler.js +26 -3
- package/dist/close-session-handler.js.map +1 -1
- package/dist/connect-or-start.js +8 -2
- package/dist/connect-or-start.js.map +1 -1
- package/dist/content-park.d.ts.map +1 -1
- package/dist/content-park.js +5 -3
- package/dist/content-park.js.map +1 -1
- package/dist/log-collapse.d.ts +66 -0
- package/dist/log-collapse.d.ts.map +1 -0
- package/dist/log-collapse.js +244 -0
- package/dist/log-collapse.js.map +1 -0
- package/dist/log-rotate.d.ts +67 -0
- package/dist/log-rotate.d.ts.map +1 -0
- package/dist/log-rotate.js +134 -0
- package/dist/log-rotate.js.map +1 -0
- package/dist/outbound-sessions.d.ts.map +1 -1
- package/dist/outbound-sessions.js +28 -0
- package/dist/outbound-sessions.js.map +1 -1
- package/dist/park-recovery.d.ts +3 -2
- package/dist/park-recovery.d.ts.map +1 -1
- package/dist/park-recovery.js +3 -2
- package/dist/park-recovery.js.map +1 -1
- package/dist/refusal-reasons.d.ts +46 -0
- package/dist/refusal-reasons.d.ts.map +1 -1
- package/dist/refusal-reasons.js +44 -0
- package/dist/refusal-reasons.js.map +1 -1
- package/dist/relay-only.d.ts +9 -2
- package/dist/relay-only.d.ts.map +1 -1
- package/dist/relay-only.js +9 -2
- package/dist/relay-only.js.map +1 -1
- package/dist/restart-seal-resolver.d.ts.map +1 -1
- package/dist/restart-seal-resolver.js +13 -0
- package/dist/restart-seal-resolver.js.map +1 -1
- package/dist/session-ceremony.d.ts +17 -0
- package/dist/session-ceremony.d.ts.map +1 -1
- package/dist/session-ceremony.js +77 -1
- package/dist/session-ceremony.js.map +1 -1
- package/dist/session-content-send.js +12 -11
- package/dist/session-content-send.js.map +1 -1
- package/dist/session-lifecycle.d.ts.map +1 -1
- package/dist/session-lifecycle.js +36 -0
- package/dist/session-lifecycle.js.map +1 -1
- package/dist/session-node-factory.js +4 -4
- package/dist/session-node-factory.js.map +1 -1
- package/dist/session-node-manager.d.ts +7 -2
- package/dist/session-node-manager.d.ts.map +1 -1
- package/dist/session-node-manager.js +22 -6
- package/dist/session-node-manager.js.map +1 -1
- package/dist/session-node-types.d.ts +40 -13
- package/dist/session-node-types.d.ts.map +1 -1
- package/dist/session-node-types.js +18 -0
- package/dist/session-node-types.js.map +1 -1
- package/dist/session-relay-client.d.ts +21 -0
- package/dist/session-relay-client.d.ts.map +1 -1
- package/dist/session-relay-client.js +219 -3
- package/dist/session-relay-client.js.map +1 -1
- package/dist/session-relay.d.ts +15 -1
- package/dist/session-relay.d.ts.map +1 -1
- package/dist/session-relay.js +290 -161
- package/dist/session-relay.js.map +1 -1
- package/dist/signaling-connect.d.ts.map +1 -1
- package/dist/signaling-connect.js +23 -4
- package/dist/signaling-connect.js.map +1 -1
- package/dist/signaling-wiring.js +7 -7
- package/dist/signaling-wiring.js.map +1 -1
- package/dist/standing-receivers.d.ts +53 -14
- package/dist/standing-receivers.d.ts.map +1 -1
- package/dist/standing-receivers.js +551 -544
- package/dist/standing-receivers.js.map +1 -1
- package/dist/trust-signal-sweep-tick.d.ts +15 -0
- package/dist/trust-signal-sweep-tick.d.ts.map +1 -1
- package/dist/trust-signal-sweep-tick.js +19 -2
- package/dist/trust-signal-sweep-tick.js.map +1 -1
- package/dist/trust-signal-sweep.d.ts +16 -1
- package/dist/trust-signal-sweep.d.ts.map +1 -1
- package/dist/trust-signal-sweep.js +12 -10
- package/dist/trust-signal-sweep.js.map +1 -1
- package/dist/types.d.ts +18 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js.map +1 -1
- package/package.json +5 -5
package/dist/session-relay.js
CHANGED
|
@@ -130,8 +130,15 @@ export class SessionRelay {
|
|
|
130
130
|
* what looks like a fleet-wide outage.
|
|
131
131
|
*/
|
|
132
132
|
if (refusal?.tryAnotherRelay && !this.#ctx.shuttingDown) {
|
|
133
|
+
/**
|
|
134
|
+
* ⚠️ **THE QUARANTINE ACTS; THE REBUILD USED TO AND NO LONGER CAN — 056-SLOTDEAD.**
|
|
135
|
+
*
|
|
136
|
+
* This used to `rebuildStandingReceiver` so the new node would reserve against the rest of
|
|
137
|
+
* the pool. A rebuilt receiver reserves NOTHING now (055-ONDEMAND), so the rebuild bought
|
|
138
|
+
* nothing and cost the agent its transport identity. The quarantine is the half that still
|
|
139
|
+
* does the work: the next offer reserves against the pool minus this relay.
|
|
140
|
+
*/
|
|
133
141
|
this.#quarantineRelay(agentName, relayPeerId, refusal.reason);
|
|
134
|
-
void this.#ctx.receivers.rebuildStandingReceiver(agentName);
|
|
135
142
|
}
|
|
136
143
|
}
|
|
137
144
|
else {
|
|
@@ -742,18 +749,20 @@ export class SessionRelay {
|
|
|
742
749
|
this.#ctx.srReservationRetry.delete(agentName);
|
|
743
750
|
this.#ctx.srLastRejectionReason.delete(agentName);
|
|
744
751
|
}
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
752
|
+
/**
|
|
753
|
+
* ⚠️ **ENDPOINTS ARRIVING NO LONGER REBUILD THE RECEIVER — 055-ONDEMAND.**
|
|
754
|
+
*
|
|
755
|
+
* This used to rebuild so the new node could reserve with the relays that had just been
|
|
756
|
+
* announced, because reservations were fixed at node creation and acquired at login. Neither is
|
|
757
|
+
* true now: nothing is reserved at login, and a running node can take one on demand.
|
|
758
|
+
*
|
|
759
|
+
* So a rebuild here would buy nothing and cost something real. The receiver it replaces may be
|
|
760
|
+
* holding the circuit a LIVE SESSION's counterparty was told to dial — throwing that away in
|
|
761
|
+
* response to a routine directory announcement would drop the route silently, at both ends.
|
|
762
|
+
*
|
|
763
|
+
* The endpoints are still recorded above, which is all they were ever needed for: they are the
|
|
764
|
+
* candidate list an offer reserves against.
|
|
765
|
+
*/
|
|
757
766
|
}
|
|
758
767
|
/**
|
|
759
768
|
* Is this agent currently skipping this relay? The observable half of the failover decision — a
|
|
@@ -880,35 +889,45 @@ export class SessionRelay {
|
|
|
880
889
|
* That is precisely the silent-loss-of-inbound failure this whole story exists to
|
|
881
890
|
* kill, so it cannot be left to chance: we watch for it and re-pick a relay.
|
|
882
891
|
*
|
|
883
|
-
* Only receivers that HAD a reservation are watched.
|
|
884
|
-
* already degraded and already loud (reservation.none /
|
|
885
|
-
* rebuilding it on a timer would just thrash against relays we know are
|
|
892
|
+
* ⚠️ 056-SLOTDEAD: this paragraph used to end "Only receivers that HAD a reservation are watched.
|
|
893
|
+
* One that never got one is already degraded and already loud (reservation.none /
|
|
894
|
+
* reservation.timeout); rebuilding it on a timer would just thrash against relays we know are
|
|
895
|
+
* refusing." Every clause of that is now false — an idle receiver holds nothing BY DESIGN so
|
|
896
|
+
* "degraded" is the normal state, `reservation.none` is not emitted at build any more, and
|
|
897
|
+
* nothing rebuilds. What is watched is a receiver whose recorded set has shrunk, and what
|
|
898
|
+
* happens is a re-take in place for a LIVE session, never a rebuild.
|
|
886
899
|
*/
|
|
887
900
|
/**
|
|
888
|
-
* DOD-M12B-RESERVATION-RETRY-1
|
|
901
|
+
* DOD-M12B-RESERVATION-RETRY-1 / 055-ONDEMAND — is a re-attempt due, and claim it if so.
|
|
902
|
+
*
|
|
903
|
+
* ⚠️ **TWO DOC BLOCKS DESCRIBING `#retryReservationIfDue` USED TO SIT HERE — 056-SLOTDEAD, review
|
|
904
|
+
* F7.** That function is deleted. One of them said "the rebuild is the re-attempt … which is why
|
|
905
|
+
* it stays", so a reader following the comment landed on this function and concluded a rebuild
|
|
906
|
+
* ladder was still running. There is no rebuild anywhere in this file.
|
|
889
907
|
*
|
|
890
|
-
*
|
|
891
|
-
*
|
|
892
|
-
*
|
|
908
|
+
* What this does: a live session that lost its circuit re-takes it IN PLACE, on the node that
|
|
909
|
+
* already has the session, and this owns whether it is allowed to try yet. ONE ladder — a
|
|
910
|
+
* reservation is scarce, the relay holds it for its full TTL even after the client disconnects,
|
|
911
|
+
* and this file's own warning is that churning attempts across a fleet is how a relay is
|
|
912
|
+
* exhausted.
|
|
893
913
|
*/
|
|
894
|
-
#
|
|
914
|
+
#retryDue(agentName) {
|
|
895
915
|
const now = Date.now();
|
|
896
916
|
const state = this.#ctx.srReservationRetry.get(agentName)
|
|
897
917
|
?? { attempts: 0, nextAt: now + this.#ctx.srReservationRetryMs, correlationId: randomUUID() };
|
|
898
|
-
// The reason the LAST attempt was refused, captured where it is actually known.
|
|
899
918
|
const lastReason = this.#ctx.srLastRejectionReason.get(agentName);
|
|
900
919
|
if (lastReason !== undefined)
|
|
901
920
|
state.lastReason = lastReason;
|
|
902
|
-
if (
|
|
903
|
-
// First sighting — schedule, do not fire.
|
|
921
|
+
if (!this.#ctx.srReservationRetry.has(agentName)) {
|
|
922
|
+
// First sighting — schedule, do not fire. Something just tried.
|
|
904
923
|
this.#ctx.srReservationRetry.set(agentName, state);
|
|
905
|
-
return;
|
|
924
|
+
return false;
|
|
906
925
|
}
|
|
907
926
|
if (now < state.nextAt)
|
|
908
|
-
return;
|
|
927
|
+
return false;
|
|
909
928
|
if (state.attempts >= SR_RESERVATION_MAX_RETRIES) {
|
|
910
929
|
if (state.attempts === SR_RESERVATION_MAX_RETRIES) {
|
|
911
|
-
state.attempts += 1; //
|
|
930
|
+
state.attempts += 1; // report once
|
|
912
931
|
this.#ctx.srReservationRetry.set(agentName, state);
|
|
913
932
|
this.#ctx.logger.error("session.standing_receiver.reservation.gave_up", {
|
|
914
933
|
agentName,
|
|
@@ -917,51 +936,43 @@ export class SessionRelay {
|
|
|
917
936
|
// WHY, not just the consequence. Three different problems reach this one message and they
|
|
918
937
|
// need three different responses: `relay_granted_no_reservation` is relay CAPACITY (and a
|
|
919
938
|
// trustless-cello problem), `relay_unreachable` is the NETWORK, and
|
|
920
|
-
// `reservation_did_not_complete_in_time` is LATENCY
|
|
921
|
-
// can pin a slot it never uses, so its appearance is also the signal that this retry
|
|
922
|
-
// budget needs tightening.
|
|
939
|
+
// `reservation_did_not_complete_in_time` is LATENCY.
|
|
923
940
|
...(state.lastReason !== undefined ? { lastRejectionReason: state.lastReason } : {}),
|
|
924
|
-
|
|
925
|
-
|
|
926
|
-
|
|
927
|
-
|
|
928
|
-
|
|
929
|
-
|
|
930
|
-
|
|
931
|
-
|
|
932
|
-
|
|
933
|
-
|
|
941
|
+
/**
|
|
942
|
+
* ⚠️ **THESE TWO CAME BACK — 056-SLOTDEAD, review F9.** The deleted emitter of this SAME
|
|
943
|
+
* event name carried them; the surviving one did not, so the fields quietly vanished from
|
|
944
|
+
* an operator's view while the event kept appearing.
|
|
945
|
+
*
|
|
946
|
+
* "No relay would grant" and "there was no relay to ask" are different facts that lead to
|
|
947
|
+
* different places — the first at relay capacity, the second at this agent's directory
|
|
948
|
+
* connection. Without them they are the same sentence.
|
|
949
|
+
*
|
|
950
|
+
* `hadRelayToAsk` reads the directory pool alone; `relaysOffered` is the merged,
|
|
951
|
+
* quarantine-filtered candidate list the re-take actually walks. Two populations, so two
|
|
952
|
+
* names — the mis-naming that made `reservationsRequested` unreadable is the reason this
|
|
953
|
+
* note exists.
|
|
954
|
+
*/
|
|
934
955
|
hadRelayToAsk: (this.#ctx.directoryRelayEndpoints.get(agentName)?.length ?? 0) > 0,
|
|
935
|
-
// …and HOW MANY the walk actually asks, so this event stands on its own instead of
|
|
936
|
-
// needing the last reachability line to be read beside it. Same population and same
|
|
937
|
-
// meaning as `relaysOffered` everywhere else: the merged, quarantine-filtered candidate
|
|
938
|
-
// list.
|
|
939
956
|
relaysOffered: this.reservationCircuitAddrs(agentName).addrs.length,
|
|
940
|
-
impact: "
|
|
957
|
+
impact: "a live session lost the circuit its counterparty dials, and no relay would give " +
|
|
958
|
+
"it back inside the retry budget. Messages still reach this agent through the relay's " +
|
|
959
|
+
"store-and-forward; a direct dial to it will not connect until the session is rebuilt.",
|
|
941
960
|
});
|
|
942
961
|
}
|
|
943
|
-
return;
|
|
962
|
+
return false;
|
|
944
963
|
}
|
|
945
964
|
state.attempts += 1;
|
|
946
|
-
//
|
|
947
|
-
//
|
|
948
|
-
//
|
|
949
|
-
|
|
950
|
-
state.nextAt = now + (state.attempts >= SR_RESERVATION_MAX_RETRIES
|
|
951
|
-
? this.#ctx.srReservationRetryMs
|
|
952
|
-
: this.#ctx.srReservationRetryMs * 2 ** (state.attempts - 1));
|
|
965
|
+
// Doubling, floored at the configured interval. 056-SLOTDEAD: this said "the same shape the
|
|
966
|
+
// rebuild ladder uses" — there is no rebuild ladder any more, this IS the ladder. The reason is
|
|
967
|
+
// unchanged: a fixed short interval is what exhausts a relay.
|
|
968
|
+
state.nextAt = now + this.#ctx.srReservationRetryMs * Math.pow(2, state.attempts - 1);
|
|
953
969
|
this.#ctx.srReservationRetry.set(agentName, state);
|
|
954
|
-
this.#ctx.logger.
|
|
955
|
-
agentName,
|
|
956
|
-
attempt: state.attempts,
|
|
957
|
-
maxAttempts: SR_RESERVATION_MAX_RETRIES,
|
|
958
|
-
correlationId: state.correlationId,
|
|
959
|
-
...(state.lastReason !== undefined ? { lastRejectionReason: state.lastReason } : {}),
|
|
960
|
-
impact: "this agent currently holds no circuit reservation, so a NAT'd peer cannot dial it",
|
|
970
|
+
this.#ctx.logger.info("session.standing_receiver.reservation.retry", {
|
|
971
|
+
agentName, attempts: state.attempts, correlationId: state.correlationId,
|
|
961
972
|
});
|
|
962
|
-
|
|
973
|
+
return true;
|
|
963
974
|
}
|
|
964
|
-
#reservationWatchdogTick() {
|
|
975
|
+
async #reservationWatchdogTick() {
|
|
965
976
|
if (this.#ctx.shuttingDown)
|
|
966
977
|
return;
|
|
967
978
|
for (const [agentName, sr] of this.#ctx.standingReceivers) {
|
|
@@ -982,6 +993,99 @@ export class SessionRelay {
|
|
|
982
993
|
// full TTL even after the client disconnects, and churning attempts across a fleet is how a
|
|
983
994
|
// relay is exhausted (`#startReceiverNode` records that hazard).
|
|
984
995
|
if (sr.relayPeerIds.length === 0) {
|
|
996
|
+
/**
|
|
997
|
+
* ⚠️ **AN IDLE AGENT HOLDING NOTHING IS THE DESIGN NOW, NOT A DEGRADATION — 055-ONDEMAND.**
|
|
998
|
+
*
|
|
999
|
+
* Everything above this line was written when a receiver reserved at login and holding zero
|
|
1000
|
+
* meant an agent dialable by nobody for its whole life. A reservation is taken at OFFER time
|
|
1001
|
+
* now and given back at the seal, so zero is the correct steady state for an agent nobody is
|
|
1002
|
+
* calling — and retrying here would have every idle agent in the fleet asking relays for
|
|
1003
|
+
* slots the design says it must not hold. That is the exact churn the note below warns
|
|
1004
|
+
* about, pointed at the whole fleet instead of one receiver.
|
|
1005
|
+
*
|
|
1006
|
+
* The retry still matters for an agent that HAS a live session and lost the circuit that
|
|
1007
|
+
* session depends on, which is what the condition now says.
|
|
1008
|
+
*/
|
|
1009
|
+
const liveSessions = [...this.#ctx.activeNodes.values()].filter((e) => e.agentName === agentName);
|
|
1010
|
+
if (liveSessions.length === 0)
|
|
1011
|
+
continue;
|
|
1012
|
+
/**
|
|
1013
|
+
* ⚠️ **RE-TAKE THE SESSION'S CIRCUIT — REBUILDING THE RECEIVER NO LONGER DOES IT.**
|
|
1014
|
+
*
|
|
1015
|
+
* `#retryReservationIfDue` rebuilds the receiver, and a rebuilt receiver reserves NOTHING
|
|
1016
|
+
* (055-ONDEMAND). So on its own the retry ladder became a no-op by construction: a relay
|
|
1017
|
+
* that flapped mid-session would leave that session permanently un-dialable while the
|
|
1018
|
+
* ladder churned receivers to no effect.
|
|
1019
|
+
*
|
|
1020
|
+
* The circuit a live session needs is the one its ASSIGNMENT named, which is persisted with
|
|
1021
|
+
* the session — so ask for that one back, rather than rebuilding and hoping.
|
|
1022
|
+
*/
|
|
1023
|
+
/**
|
|
1024
|
+
* ⚠️ **ON THE SAME BUDGET AND BACKOFF AS THE LADDER IT REPLACES.** A reservation is scarce:
|
|
1025
|
+
* the relay holds it for its full TTL even after the client disconnects, and this file's own
|
|
1026
|
+
* warning is that churning attempts across a fleet is how a relay is exhausted. Asking on
|
|
1027
|
+
* every 30-second tick would be exactly that churn, wearing a new name — measured here as 52
|
|
1028
|
+
* asks where the budget allows 37.
|
|
1029
|
+
*/
|
|
1030
|
+
/**
|
|
1031
|
+
* ⚠️ **WHAT ACTUALLY NEEDS A CIRCUIT, DECIDED BEFORE THE BUDGET IS TOUCHED —
|
|
1032
|
+
* `DOD-M15-IDLERETRY-1`.**
|
|
1033
|
+
*
|
|
1034
|
+
* `#retryDue` was called FIRST, and it does not merely answer a question: on an agent it has
|
|
1035
|
+
* not seen before it WRITES a retry entry and returns false. So simply having a live session
|
|
1036
|
+
* while holding no circuit was enough to open a retry ladder — and for the INITIATOR of a
|
|
1037
|
+
* call that state is permanent and wrong, because an initiator DIALS OUT and never needs to
|
|
1038
|
+
* be dialable (055-ONDEMAND narrowed the reservation to the party being dialed, deliberately).
|
|
1039
|
+
*
|
|
1040
|
+
* Two things followed, and both were measured live on 2026-09-11 rather than reasoned:
|
|
1041
|
+
* - `cello_status` reported `retrying` for a healthy agent from its first call onward,
|
|
1042
|
+
* cleared only by a daemon restart — the same "a broken agent looks like a fine one"
|
|
1043
|
+
* defect 060-READY had just fixed, in a case it did not cover;
|
|
1044
|
+
* - worse, it is not cosmetic. Held past the retry interval, both agents emitted a real
|
|
1045
|
+
* `session.standing_receiver.reservation.retry` — the daemon asking relays for slots on
|
|
1046
|
+
* behalf of sessions that need none. That is churn against the scarce resource this
|
|
1047
|
+
* whole story exists to conserve.
|
|
1048
|
+
*
|
|
1049
|
+
* So the question "is there anything to re-take?" is answered from the SESSION NODES first.
|
|
1050
|
+
* A session that already holds its circuit, or has no relay endpoint recorded, needs
|
|
1051
|
+
* nothing. If none needs anything, this tick is a no-op and must leave no trace.
|
|
1052
|
+
*/
|
|
1053
|
+
const needRetake = liveSessions.filter((entry) => {
|
|
1054
|
+
const ep = this.#ctx.queries.getPersistedRelayEndpoint(agentName, entry.sessionId);
|
|
1055
|
+
if (!ep || ep.relayAddrs.length === 0)
|
|
1056
|
+
return false;
|
|
1057
|
+
/**
|
|
1058
|
+
* ⚠️ `heldRelayIdsOf`, NOT a substring test — the same distinction 056-SLOTDEAD had to
|
|
1059
|
+
* make one layer up. A circuit address that does not NAME its relay cannot be dialled
|
|
1060
|
+
* through, so a session announcing one holds nothing usable and DOES need a re-take.
|
|
1061
|
+
* A substring test counts it as held and leaves that session silently unreachable.
|
|
1062
|
+
*/
|
|
1063
|
+
return heldRelayIdsOf(entry.node).length === 0;
|
|
1064
|
+
});
|
|
1065
|
+
if (needRetake.length === 0)
|
|
1066
|
+
continue;
|
|
1067
|
+
if (!this.#retryDue(agentName))
|
|
1068
|
+
continue;
|
|
1069
|
+
/**
|
|
1070
|
+
* ⚠️ **ON THE SESSION'S OWN NODE, NOT THE IDLE RECEIVER — and the first version got this
|
|
1071
|
+
* wrong in a way that was worse than doing nothing.**
|
|
1072
|
+
*
|
|
1073
|
+
* It called the take path, which reserves on `standingReceivers.get(agentName).node`. That
|
|
1074
|
+
* is the fresh receiver built after the promotion — a DIFFERENT peer id from the one the
|
|
1075
|
+
* counterparty was told to dial. So the circuit it obtained helped no one, and it made an
|
|
1076
|
+
* IDLE receiver hold a relay slot, which is the exact thing this unit exists to stop. It
|
|
1077
|
+
* also set `relayPeerIds` non-empty, hiding the real loss from every later tick.
|
|
1078
|
+
*/
|
|
1079
|
+
let retook = false;
|
|
1080
|
+
for (const entry of needRetake) {
|
|
1081
|
+
const ep = this.#ctx.queries.getPersistedRelayEndpoint(agentName, entry.sessionId);
|
|
1082
|
+
const base = ep.relayAddrs[0];
|
|
1083
|
+
const circuitAddr = base.includes(`/p2p/${ep.relayPeerId}`) ? `${base}/p2p-circuit` : `${base}/p2p/${ep.relayPeerId}/p2p-circuit`;
|
|
1084
|
+
if (await this.#ctx.retakeReservationOn(agentName, entry.node, circuitAddr, entry.correlationId))
|
|
1085
|
+
retook = true;
|
|
1086
|
+
}
|
|
1087
|
+
if (retook)
|
|
1088
|
+
continue;
|
|
985
1089
|
// …unless one has arrived since. Review F4, same class as the recompute below: the
|
|
986
1090
|
// slow-start path installs a receiver before every circuit has bound, so "held nothing at
|
|
987
1091
|
// install" is not the same fact as "holds nothing now". Adopting it here is what stops the
|
|
@@ -989,7 +1093,19 @@ export class SessionRelay {
|
|
|
989
1093
|
const arrived = heldRelayIdsOf(sr.node)
|
|
990
1094
|
.filter((id) => sr.node.getConnections().some((c) => c.peerId === id && c.status === "open"));
|
|
991
1095
|
if (arrived.length === 0) {
|
|
992
|
-
|
|
1096
|
+
/**
|
|
1097
|
+
* ⚠️ **`#retryReservationIfDue` WAS CALLED HERE AND COULD NEVER DO ANYTHING — 056-SLOTDEAD.**
|
|
1098
|
+
*
|
|
1099
|
+
* It guards on `now < state.nextAt`, and `#retryDue` above advances that same state on
|
|
1100
|
+
* this very tick, so the second call always returned immediately. Reachable, referenced,
|
|
1101
|
+
* and inert — the shape a reference scan cannot find.
|
|
1102
|
+
*
|
|
1103
|
+
* *(The first version of this order claimed it DOUBLE-ADVANCED the budget and triggered a
|
|
1104
|
+
* useless rebuild. That was wrong, and it was wrong because it was reasoned rather than
|
|
1105
|
+
* run. It was a no-op.)*
|
|
1106
|
+
*
|
|
1107
|
+
* Nothing replaces it: `#retryDue` owns the budget and the re-take above owns the work.
|
|
1108
|
+
*/
|
|
993
1109
|
continue;
|
|
994
1110
|
}
|
|
995
1111
|
sr.relayPeerIds = arrived;
|
|
@@ -1069,117 +1185,75 @@ export class SessionRelay {
|
|
|
1069
1185
|
// client for this (agent, relay) pair kept the error that ended its reader; that is the
|
|
1070
1186
|
// nearest thing to an upstream cause available here, and its absence is how 2,061 of these
|
|
1071
1187
|
// went untraced.
|
|
1072
|
-
|
|
1188
|
+
/**
|
|
1189
|
+
* ⚠️ **A DIAGNOSTIC MUST NOT BE ABLE TO KILL THE WATCHDOG TICK.** This is optional-chained on
|
|
1190
|
+
* the map lookup but the METHOD was called unguarded, so a client without it threw — an
|
|
1191
|
+
* unhandled rejection inside the tick, which takes the rest of the sweep with it. Surfaced
|
|
1192
|
+
* when 055-ONDEMAND made this branch reachable in more cases. The cause line is worth
|
|
1193
|
+
* having; it is not worth the loss detection it rides on.
|
|
1194
|
+
*/
|
|
1195
|
+
const client = this.#ctx.relayClients.get(`${agentName}::${relayPeerId}`);
|
|
1196
|
+
const upstreamReason = typeof client?.getLastReaderError === "function" ? client.getLastReaderError() : null;
|
|
1073
1197
|
this.#ctx.logger.warn("session.standing_receiver.reservation.lost", {
|
|
1074
1198
|
agentName,
|
|
1075
1199
|
relayPeerId,
|
|
1076
1200
|
reason: open.includes(relayPeerId) ? "circuit_address_vanished" : "relay_connection_gone",
|
|
1077
1201
|
...(upstreamReason ? { upstreamReason } : {}),
|
|
1078
1202
|
reservationsHeld: stillHeld.length,
|
|
1079
|
-
|
|
1203
|
+
/**
|
|
1204
|
+
* The line an operator reads, and the two cases are not the same event at all.
|
|
1205
|
+
*
|
|
1206
|
+
* ⚠️ **THE ZERO CASE PROMISED A REPAIR THAT NO LONGER HAPPENS — 056-SLOTDEAD, review F3.**
|
|
1207
|
+
* It ended "The receiver is being rebuilt against the rest of the pool", which was true
|
|
1208
|
+
* when this branch rebuilt. Nothing rebuilds now. An operator reading the old line would
|
|
1209
|
+
* wait for a recovery that was never coming, and the only thing that WOULD have told them
|
|
1210
|
+
* otherwise is a `zero_held` line at debug level they are not reading. Both branches now
|
|
1211
|
+
* say what actually happens next and who owns it.
|
|
1212
|
+
*/
|
|
1080
1213
|
impact: stillHeld.length > 0
|
|
1081
1214
|
? "this agent still holds " + stillHeld.length + " other circuit reservation(s), so it "
|
|
1082
|
-
+ "stays dialable from behind NAT
|
|
1083
|
-
|
|
1084
|
-
|
|
1085
|
-
+ "
|
|
1215
|
+
+ "stays dialable from behind NAT. Losing one relay costs this agent nothing it can feel."
|
|
1216
|
+
: "this agent now holds NO circuit reservation. If it has no live session that is the "
|
|
1217
|
+
+ "normal idle state and costs nothing — a slot is taken when someone calls. If it "
|
|
1218
|
+
+ "DOES have a live session, that session's counterparty can no longer dial it, and "
|
|
1219
|
+
+ "the re-take above is what gets it back, on a bounded budget. Nothing rebuilds the "
|
|
1220
|
+
+ "receiver any more, so do not wait for one.",
|
|
1086
1221
|
});
|
|
1087
1222
|
}
|
|
1223
|
+
/**
|
|
1224
|
+
* ⚠️ **DRAIN ON THE LOSS ITSELF — 056-SLOTDEAD.** This used to happen one step downstream, as
|
|
1225
|
+
* a side effect of the rebuild below: lose every reservation → rebuild → the new receiver
|
|
1226
|
+
* reports ready → drain. Deleting the rebuild would have deleted the drain with it, leaving
|
|
1227
|
+
* content the counterparty parked to sit until the periodic backstop. The loss is the cause
|
|
1228
|
+
* and always was; the rebuild was only where it happened to be noticed.
|
|
1229
|
+
*/
|
|
1230
|
+
this.#ctx.park.fireParkedDrain(agentName, "reservation_lost");
|
|
1088
1231
|
if (stillHeld.length === 0) {
|
|
1089
|
-
|
|
1090
|
-
|
|
1091
|
-
|
|
1092
|
-
|
|
1232
|
+
/**
|
|
1233
|
+
* ⚠️ **THIS REBUILT THE RECEIVER ON A JUSTIFICATION THAT STOPPED BEING TRUE IN UNIT 1, AND
|
|
1234
|
+
* UNDER ON-DEMAND IT DESTROYS AN IN-FLIGHT OFFER'S RESERVATION — 056-SLOTDEAD.**
|
|
1235
|
+
*
|
|
1236
|
+
* It read: *"only a new node can take a new reservation, because a circuit listener is fixed
|
|
1237
|
+
* at node creation."* `listenOnCircuit` removed that constraint (`DOD-M15-RELAYPROVE-ORDER-1`),
|
|
1238
|
+
* and a rebuilt receiver now reserves nothing at all.
|
|
1239
|
+
*
|
|
1240
|
+
* The state it fires in is the dangerous one: between an offer taking a reservation and the
|
|
1241
|
+
* session being created, the receiver holds exactly one circuit. Lose the connection in that
|
|
1242
|
+
* window and this rebuilt the node — discarding the slot the offer just took, after the
|
|
1243
|
+
* accept may already have advertised it. **Observed**, not reasoned: a fixture that reported
|
|
1244
|
+
* no relay connection drove this path and the test caught the rebuild.
|
|
1245
|
+
*
|
|
1246
|
+
* Nothing replaces it here. An idle agent holding zero is the design; a LIVE session that
|
|
1247
|
+
* lost its circuit is re-taken on its own node by the branch above, on a bounded budget.
|
|
1248
|
+
*/
|
|
1249
|
+
this.#ctx.logger.debug("session.standing_receiver.reservation.zero_held", {
|
|
1250
|
+
agentName,
|
|
1251
|
+
impact: "this agent holds no circuit. If it has no live session that is the idle steady " +
|
|
1252
|
+
"state; if it does, the re-take above owns getting it back.",
|
|
1253
|
+
});
|
|
1093
1254
|
continue;
|
|
1094
1255
|
}
|
|
1095
|
-
/**
|
|
1096
|
-
* STILL REACHABLE, SO THE RECEIVER STANDS, AND NOTHING ELSE HAPPENS HERE. That second half is
|
|
1097
|
-
* the part worth reading, because the obvious next line is wrong twice over.
|
|
1098
|
-
*
|
|
1099
|
-
* **A LOST CONFIGURED CIRCUIT CANNOT BE RETAKEN BY THIS NODE.** Read out of
|
|
1100
|
-
* `@libp2p/circuit-relay-v2@4.2.5`, not assumed: for an explicit relay address
|
|
1101
|
-
* `transport/listener.js#listen()` is a ONE-SHOT — it reserves once and nothing calls it
|
|
1102
|
-
* again; `reservation-store.js#removeReservation()` clears the refresh timeout and deletes
|
|
1103
|
-
* the entry; and the listener's `_onAddRelayPeer` returns early for `type === 'configured'`,
|
|
1104
|
-
* so even a later reservation would not be announced. A circuit listener is fixed at node
|
|
1105
|
-
* creation, and the only thing that takes a new one is a NEW NODE — which is exactly the
|
|
1106
|
-
* rebuild this branch exists to refuse.
|
|
1107
|
-
*
|
|
1108
|
-
* **AND RE-PROVING TO THE LOST RELAY WOULD REBUILD THE RECEIVER ANYWAY.** Review F3: an
|
|
1109
|
-
* earlier version called `authenticateStandingReceiver` here to "remove the relay-side
|
|
1110
|
-
* reason for the revocation". That function ends with `if (refusal?.tryAnotherRelay) { …
|
|
1111
|
-
* void this.#ctx.receivers.rebuildStandingReceiver(agentName); }` — and a dead or misconfigured relay is
|
|
1112
|
-
* precisely the one that answers that way. So the common case was: lose relay A while
|
|
1113
|
-
* holding B, decline to rebuild, prove to A, A refuses, rebuild the whole receiver and throw
|
|
1114
|
-
* B's healthy reservation away. The churn engine, re-entered through the back door.
|
|
1115
|
-
*
|
|
1116
|
-
* **THE BOUND, STATED PLAINLY BECAUSE IT IS A REAL SHORTFALL AGAINST THE DoD:** a lost
|
|
1117
|
-
* circuit is gone until the receiver is next rebuilt for another reason. What the agent buys
|
|
1118
|
-
* is that it never STOPS BEING REACHABLE while that is true — the surviving relays carry it,
|
|
1119
|
-
* the loss is named in the log with its cause, and the lost relay's inbound carve-out is
|
|
1120
|
-
* revoked above. That is availability, not restoration in place.
|
|
1121
|
-
*
|
|
1122
|
-
* WHICH LEAVES A RATCHET, and `#respreadIfDecayed` below is what stops it: relays are only
|
|
1123
|
-
* ever lost between rebuilds, never regained, so an agent nobody talks to walks itself back
|
|
1124
|
-
* down to one relay — the exact state this unit exists to get it out of.
|
|
1125
|
-
*/
|
|
1126
|
-
}
|
|
1127
|
-
for (const agentName of this.#ctx.standingReceivers.keys()) {
|
|
1128
|
-
if (this.#ctx.agentsWantingReceiver.has(agentName))
|
|
1129
|
-
this.#respreadIfDecayed(agentName);
|
|
1130
|
-
}
|
|
1131
|
-
}
|
|
1132
|
-
/**
|
|
1133
|
-
* 032-RELAYSPREAD — **AN IDLE AGENT MUST NOT RATCHET ITSELF BACK DOWN TO ONE RELAY.**
|
|
1134
|
-
*
|
|
1135
|
-
* Spreading happens when a receiver is BUILT, and between builds the count only falls: a lost
|
|
1136
|
-
* circuit is not retaken from here (`listenOnCircuit` could; nothing does), and a relay the
|
|
1137
|
-
* directory announces later is skipped while any circuit is held. An agent in
|
|
1138
|
-
* conversation re-spreads constantly — the receiver is handed into each session and a fresh one
|
|
1139
|
-
* is built behind it — so this is about the agent nobody has talked to for a day. It loses relays
|
|
1140
|
-
* one at a time, nothing pulls it back up, and it ends up exactly where this unit found it:
|
|
1141
|
-
* reachable through one relay, one relay away from being reachable through none.
|
|
1142
|
-
*
|
|
1143
|
-
* **THE COST OF FIXING IT IS A NEW PEER ID**, which is why it is fenced three ways rather than
|
|
1144
|
-
* simply rebuilding on sight:
|
|
1145
|
-
* - **ONLY WHEN IDLE.** A rebuild replaces the receiver's transport identity, and a counterparty
|
|
1146
|
-
* may be holding the old one from a `session_offer_accept`. With a live session for this agent
|
|
1147
|
-
* we leave it alone — a degraded spread costs redundancy, a changed peer id mid-conversation
|
|
1148
|
-
* costs the conversation.
|
|
1149
|
-
* - **ONLY WHEN THERE IS SOMETHING TO GAIN.** Holding every relay that was offered is not decay.
|
|
1150
|
-
* - **ON ITS OWN SLOW CLOCK**, never the watchdog's 30-second grid. A reservation is scarce —
|
|
1151
|
-
* the relay holds it for its full TTL even after we disconnect — so this reuses the
|
|
1152
|
-
* reservation retry interval rather than inventing a faster one.
|
|
1153
|
-
*/
|
|
1154
|
-
#respreadIfDecayed(agentName) {
|
|
1155
|
-
if (this.#ctx.shuttingDown)
|
|
1156
|
-
return;
|
|
1157
|
-
const sr = this.#ctx.standingReceivers.get(agentName);
|
|
1158
|
-
if (!sr || sr.relayPeerIds.length === 0)
|
|
1159
|
-
return; // zero held is the loud path
|
|
1160
|
-
for (const entry of this.#ctx.activeNodes.values()) {
|
|
1161
|
-
if (entry.agentName === agentName)
|
|
1162
|
-
return; // in conversation — hands off
|
|
1163
1256
|
}
|
|
1164
|
-
const offered = this.reservationCircuitAddrs(agentName).addrs.length;
|
|
1165
|
-
if (sr.relayPeerIds.length >= offered)
|
|
1166
|
-
return; // nothing to gain
|
|
1167
|
-
const now = Date.now();
|
|
1168
|
-
const last = this.#ctx.srLastRespreadAt.get(agentName) ?? 0;
|
|
1169
|
-
if (now - last < this.#ctx.srReservationRetryMs)
|
|
1170
|
-
return;
|
|
1171
|
-
this.#ctx.srLastRespreadAt.set(agentName, now);
|
|
1172
|
-
this.#ctx.logger.info("session.standing_receiver.respread", {
|
|
1173
|
-
agentName,
|
|
1174
|
-
reservationsHeld: sr.relayPeerIds.length,
|
|
1175
|
-
relaysOffered: offered,
|
|
1176
|
-
impact: "this agent is idle and holds fewer relay reservations than it was offered, so its " +
|
|
1177
|
-
"receiver is being rebuilt to take the rest. Without this it can only lose relays between " +
|
|
1178
|
-
"rebuilds, and an agent nobody talks to drifts back down to a single relay — one relay " +
|
|
1179
|
-
"away from being unreachable behind NAT, which is the state this whole mechanism exists " +
|
|
1180
|
-
"to keep it out of.",
|
|
1181
|
-
});
|
|
1182
|
-
void this.#ctx.receivers.rebuildStandingReceiver(agentName);
|
|
1183
1257
|
}
|
|
1184
1258
|
/** Start the reservation watchdog (idempotent). Stopped by gracefulShutdown. */
|
|
1185
1259
|
startReservationWatchdog() {
|
|
@@ -1190,7 +1264,7 @@ export class SessionRelay {
|
|
|
1190
1264
|
this.#ctx.park.armBackstopClock(Date.now());
|
|
1191
1265
|
this.#ctx.reservationWatchdog = setInterval(() => {
|
|
1192
1266
|
try {
|
|
1193
|
-
this.#reservationWatchdogTick();
|
|
1267
|
+
void this.#reservationWatchdogTick();
|
|
1194
1268
|
this.#ctx.park.parkedDrainBackstopTick(Date.now());
|
|
1195
1269
|
}
|
|
1196
1270
|
catch (err) {
|
|
@@ -1249,6 +1323,61 @@ export class SessionRelay {
|
|
|
1249
1323
|
* `tryAnotherRelay: false` precisely so the client STOPS walking the fleet; without the verdict,
|
|
1250
1324
|
* the loop walked it anyway, turning one client-side fault into what reads as a fleet outage.
|
|
1251
1325
|
*/
|
|
1326
|
+
/**
|
|
1327
|
+
* 055-ONDEMAND — **tell the relay this agent has finished with its slot.**
|
|
1328
|
+
*
|
|
1329
|
+
* The only thing that actually frees one: closing a circuit listener sends the relay nothing, and
|
|
1330
|
+
* the relay reclaims on its own only at the reservation TTL (two hours) or under reaper pressure.
|
|
1331
|
+
* Best-effort by design — an undelivered release costs a slot until that TTL, and must never fail
|
|
1332
|
+
* the seal that triggered it.
|
|
1333
|
+
*/
|
|
1334
|
+
async tellRelayReleased(agentName, relayPeerId, node, correlationId) {
|
|
1335
|
+
/**
|
|
1336
|
+
* ⚠️ **THE NODE IS THE SOURCE OF TRUTH FOR WHERE THIS RELAY IS — review MEDIUM-6.**
|
|
1337
|
+
*
|
|
1338
|
+
* The first version looked the relay up in `directoryRelayEndpoints`. But the reservation was
|
|
1339
|
+
* taken on the relay the OFFER named, whose addresses came off that frame; if the directory's
|
|
1340
|
+
* last announcement does not contain it — a pool relay dropped from the roster, a stale
|
|
1341
|
+
* announcement — the release warns `no_endpoint` and the slot is held to its TTL. The node
|
|
1342
|
+
* announces the circuit it actually holds, so derive the address from that and fall back to the
|
|
1343
|
+
* directory list only when there is nothing to derive from.
|
|
1344
|
+
*/
|
|
1345
|
+
const fromNode = node.listenAddresses()
|
|
1346
|
+
.filter((a) => a.split("/").includes("p2p-circuit") && a.includes(`/p2p/${relayPeerId}/`))
|
|
1347
|
+
.map((a) => a.split("/p2p-circuit")[0])
|
|
1348
|
+
.filter((a) => a.length > 0);
|
|
1349
|
+
const ep = this.#ctx.directoryRelayEndpoints.get(agentName)?.find((e) => e.relayPeerId === relayPeerId);
|
|
1350
|
+
const relayAddrs = fromNode.length > 0 ? [...new Set(fromNode)] : (ep ? [...ep.relayAddrs] : []);
|
|
1351
|
+
if (relayAddrs.length === 0) {
|
|
1352
|
+
this.#ctx.logger.warn("session.relay.reservation_release.no_endpoint", {
|
|
1353
|
+
agentName, relayPeerId, correlationId,
|
|
1354
|
+
impact: "no address is known for this relay, so it cannot be told the slot is free and " +
|
|
1355
|
+
"holds it until the reservation TTL expires.",
|
|
1356
|
+
});
|
|
1357
|
+
return;
|
|
1358
|
+
}
|
|
1359
|
+
let client;
|
|
1360
|
+
try {
|
|
1361
|
+
client = this.#ctx.detachedRelayClientBuilder?.(agentName, relayPeerId, relayAddrs, {
|
|
1362
|
+
receiptStore: this.#ctx.relayReceiptStore ?? undefined,
|
|
1363
|
+
sealLeafStore: this.#ctx.sealLeafStore ?? undefined,
|
|
1364
|
+
onlineToken: () => this.#ctx.getDirectoryOnlineToken(agentName),
|
|
1365
|
+
});
|
|
1366
|
+
if (!client)
|
|
1367
|
+
return;
|
|
1368
|
+
await client.releaseReservation(node);
|
|
1369
|
+
}
|
|
1370
|
+
catch (err) {
|
|
1371
|
+
this.#ctx.logger.warn("session.relay.reservation_release.failed", {
|
|
1372
|
+
agentName, relayPeerId, correlationId,
|
|
1373
|
+
error: extractErrorMessage(err),
|
|
1374
|
+
impact: "the relay holds this slot until its TTL expires.",
|
|
1375
|
+
});
|
|
1376
|
+
}
|
|
1377
|
+
finally {
|
|
1378
|
+
client?.close();
|
|
1379
|
+
}
|
|
1380
|
+
}
|
|
1252
1381
|
async proveToRelay(agentName, circuitAddr, node, correlationId,
|
|
1253
1382
|
/**
|
|
1254
1383
|
* Whether this proof is the STANDING RECEIVER's, and may therefore write the surface
|