@rebasepro/server-postgres 0.9.1-canary.ff338b5 → 0.10.1-canary.14e53ae

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/README.md +21 -0
  2. package/dist/PostgresBackendDriver.d.ts +18 -0
  3. package/dist/PostgresBootstrapper.d.ts +11 -1
  4. package/dist/auth/services.d.ts +93 -54
  5. package/dist/backup/backup-logic.d.ts +23 -0
  6. package/dist/backup/backup-service.d.ts +44 -2
  7. package/dist/backup/pg-tools.d.ts +41 -1
  8. package/dist/chunk-DSJWtz9O.js +40 -0
  9. package/dist/cli-helpers.d.ts +33 -1
  10. package/dist/ensure-collection-tables-CNlIONzj.js +304 -0
  11. package/dist/ensure-collection-tables-CNlIONzj.js.map +1 -0
  12. package/dist/index.d.ts +1 -0
  13. package/dist/index.es.js +2090 -4741
  14. package/dist/index.es.js.map +1 -1
  15. package/dist/schema/auth-bootstrap-sql.d.ts +1 -1
  16. package/dist/schema/auth-schema.d.ts +194 -24
  17. package/dist/schema/destructive-sql.d.ts +49 -0
  18. package/dist/schema/ensure-collection-tables.d.ts +79 -0
  19. package/dist/schema/generate-postgres-ddl-logic.d.ts +4 -1
  20. package/dist/schema/introspect-db-logic.d.ts +0 -5
  21. package/dist/schema/introspect-db-naming.d.ts +10 -0
  22. package/dist/security/policy-drift.d.ts +24 -0
  23. package/dist/security/rls-enforcement.d.ts +2 -2
  24. package/dist/services/cdc/CdcListener.d.ts +7 -14
  25. package/dist/services/channel-bus/ChannelBus.d.ts +29 -0
  26. package/dist/services/channel-bus/PostgresChannelBus.d.ts +111 -0
  27. package/dist/services/channel-bus/index.d.ts +55 -0
  28. package/dist/services/channel-history.d.ts +129 -0
  29. package/dist/services/channel-presence.d.ts +66 -0
  30. package/dist/services/pg-notify-listener.d.ts +47 -0
  31. package/dist/services/realtimeService.d.ts +183 -8
  32. package/dist/src-B0v4IKaI.js +329 -0
  33. package/dist/src-B0v4IKaI.js.map +1 -0
  34. package/dist/src-DmsRg8MR.js +4056 -0
  35. package/dist/src-DmsRg8MR.js.map +1 -0
  36. package/package.json +7 -31
  37. package/src/PostgresBackendDriver.ts +56 -5
  38. package/src/PostgresBootstrapper.ts +87 -1
  39. package/src/auth/ensure-tables.ts +187 -19
  40. package/src/auth/services.ts +309 -170
  41. package/src/backup/backup-cli.ts +60 -1
  42. package/src/backup/backup-cron.ts +24 -1
  43. package/src/backup/backup-logic.ts +62 -0
  44. package/src/backup/backup-service.ts +132 -13
  45. package/src/backup/pg-tools.ts +70 -2
  46. package/src/cli-helpers.ts +82 -27
  47. package/src/cli.ts +152 -6
  48. package/src/index.ts +4 -0
  49. package/src/schema/auth-bootstrap-sql.ts +7 -1
  50. package/src/schema/auth-schema.ts +53 -15
  51. package/src/schema/destructive-sql.ts +94 -0
  52. package/src/schema/ensure-collection-tables.test.ts +156 -0
  53. package/src/schema/ensure-collection-tables.ts +297 -0
  54. package/src/schema/generate-postgres-ddl-logic.ts +3 -3
  55. package/src/schema/introspect-db-inference.ts +1 -1
  56. package/src/schema/introspect-db-logic.ts +1 -10
  57. package/src/schema/introspect-db-naming.ts +15 -0
  58. package/src/schema/introspect-runtime.ts +1 -1
  59. package/src/security/policy-drift.test.ts +46 -0
  60. package/src/security/policy-drift.ts +70 -4
  61. package/src/security/rls-enforcement.ts +11 -5
  62. package/src/services/cdc/CdcListener.ts +27 -91
  63. package/src/services/channel-bus/ChannelBus.ts +44 -0
  64. package/src/services/channel-bus/PostgresChannelBus.ts +299 -0
  65. package/src/services/channel-bus/index.ts +123 -0
  66. package/src/services/channel-history.ts +378 -0
  67. package/src/services/channel-presence.ts +148 -0
  68. package/src/services/pg-notify-listener.ts +137 -0
  69. package/src/services/realtimeService.ts +581 -24
  70. package/src/websocket.ts +30 -11
@@ -0,0 +1,111 @@
1
+ /**
2
+ * Channel bus over Postgres LISTEN/NOTIFY.
3
+ *
4
+ * Chosen because it needs nothing that a Rebase deployment does not already
5
+ * have — the same database, the same direct URL the CDC listener uses. Three
6
+ * properties of `NOTIFY` shape everything below:
7
+ *
8
+ * - **8000 bytes per payload.** Presence and cursors fit with room to spare; a
9
+ * scene snapshot does not. Rather than truncate or drop, an oversized frame
10
+ * on a *retained* channel is published as a pointer — the body is already in
11
+ * `rebase.channel_messages` with a sequence number, so the receiver reads it
12
+ * back. That is the same trick the entity path uses (notify an address,
13
+ * refetch the row), applied to a different table. On an ephemeral channel
14
+ * there is nothing to point at, so the publish is refused loudly instead of
15
+ * reaching some instances and not others.
16
+ *
17
+ * - **A notify is a query on the primary database.** Not a slow one, but it
18
+ * competes with the application's real queries, and that — not throughput —
19
+ * is what actually limits this transport. Measured, it carried ~10k
20
+ * cross-instance messages/second and stayed flat out to eight instances; what
21
+ * it should not do is spend 10k queries/second of the database's budget on
22
+ * cursor movement. Hence the batching below.
23
+ *
24
+ * - **Delivery is best-effort.** Retained channels repair themselves through
25
+ * the client's history replay, so a lost frame costs a live update rather
26
+ * than correctness. That is what makes coalescing safe.
27
+ */
28
+ import { NodePgDatabase } from "drizzle-orm/node-postgres";
29
+ import { ChannelBus, ChannelBusFrame, ChannelBusHandler } from "./ChannelBus";
30
+ /** NOTIFY channel carrying channel-bus frames. */
31
+ export declare const CHANNEL_BUS_NOTIFY_CHANNEL = "rebase_channel_bus";
32
+ /**
33
+ * Postgres refuses a NOTIFY payload of 8000 bytes or more. The margin below it
34
+ * is for nothing in particular — it is there so that a payload which passes this
35
+ * check cannot fail at the server for being a few bytes over.
36
+ */
37
+ export declare const PG_NOTIFY_MAX_PAYLOAD_BYTES = 7500;
38
+ /**
39
+ * How long a batching window stays open.
40
+ *
41
+ * Ten milliseconds is below the threshold where a human notices a cursor lag,
42
+ * and it is the difference between one query per message and one query per
43
+ * window under load. Set to 0 to disable coalescing entirely.
44
+ */
45
+ export declare const DEFAULT_BATCH_WINDOW_MS = 10;
46
+ export declare class PostgresChannelBus implements ChannelBus {
47
+ private readonly db;
48
+ private readonly connectionString;
49
+ readonly kind: "postgres";
50
+ readonly maxFrameBytes = 7500;
51
+ private listener?;
52
+ private readonly batchWindowMs;
53
+ /**
54
+ * Frames waiting for the current window to close.
55
+ *
56
+ * The window is opened by a publish that found none open, and that publish
57
+ * is sent *immediately* rather than joining a batch — see {@link publish}.
58
+ */
59
+ private pending;
60
+ private pendingBytes;
61
+ private windowTimer?;
62
+ private stopped;
63
+ constructor(db: NodePgDatabase<Record<string, unknown>>, connectionString: string, options?: {
64
+ batchWindowMs?: number;
65
+ });
66
+ start(handler: ChannelBusHandler): Promise<void>;
67
+ /**
68
+ * Publish, coalescing under load.
69
+ *
70
+ * The window is *leading edge*: a publish arriving when no window is open is
71
+ * sent straight away and opens one, so an idle channel pays no added latency
72
+ * at all. Frames arriving while it is open are collected and leave together
73
+ * when it closes. The effect is that cost tracks elapsed time rather than
74
+ * message count — one query per window instead of one per message — which is
75
+ * the same shape as the retention pruning throttle, for the same reason.
76
+ *
77
+ * The returned promise settles when the frame has actually left, not when it
78
+ * was queued, so the contract ("reaches the other instances, or rejects")
79
+ * still holds.
80
+ */
81
+ publish(frame: ChannelBusFrame): Promise<void>;
82
+ stop(): Promise<void>;
83
+ private openWindow;
84
+ /** Send everything queued and settle the promises waiting on it. */
85
+ private flush;
86
+ /**
87
+ * One NOTIFY.
88
+ *
89
+ * A single frame goes out in the plain, unwrapped shape. That is not just
90
+ * economy: during a rolling deploy an instance running the previous build
91
+ * understands only that shape, and low-rate traffic — presence, the tail of
92
+ * a session — is exactly what is flowing while pods restart. Batching only
93
+ * appears under load, which shrinks the mixed-version window to almost
94
+ * nothing.
95
+ */
96
+ private send;
97
+ }
98
+ /**
99
+ * Parse a bus payload into the frames it carries.
100
+ *
101
+ * Accepts both wire shapes — a bare frame and a `{ batch: [...] }` envelope —
102
+ * so an instance on the new build understands one on the old. Returns an empty
103
+ * array for anything unrecognisable: a malformed or future-versioned message
104
+ * must never take the listener down.
105
+ */
106
+ export declare function parseChannelBusPayload(payload: string): ChannelBusFrame[];
107
+ /**
108
+ * Parse a single bus frame, returning null for anything that is not a frame we
109
+ * understand.
110
+ */
111
+ export declare function parseChannelBusFrame(payload: string): ChannelBusFrame | null;
@@ -0,0 +1,55 @@
1
+ /**
2
+ * Resolution of the channel bus from config, environment, or a supplied instance.
3
+ *
4
+ * Opt-in, like every other cross-cutting realtime switch here: with nothing
5
+ * configured a deployment gets the memory bus and behaves exactly as it did
6
+ * before this existed. Unlike `REALTIME_CDC=auto`, there is no "try it and see"
7
+ * default — a bus changes where messages go, and quietly turning on a Postgres
8
+ * NOTIFY per broadcast because a direct URL happened to be set is not a
9
+ * decision to make on the user's behalf.
10
+ *
11
+ * Two transports ship, and neither adds a service to a deployment. A third is
12
+ * not a code change here: `realtime.bus` also accepts an already-constructed
13
+ * {@link ChannelBus}, so a transport published as its own package plugs in
14
+ * without this file learning about it. See `@rebasepro/types` →
15
+ * `types/channel_bus.ts` for the contract such a package implements.
16
+ */
17
+ import { NodePgDatabase } from "drizzle-orm/node-postgres";
18
+ import { type ChannelBus, type ChannelBusConfig, type ChannelBusSetting } from "@rebasepro/types";
19
+ export * from "./ChannelBus";
20
+ export { PostgresChannelBus, CHANNEL_BUS_NOTIFY_CHANNEL, PG_NOTIFY_MAX_PAYLOAD_BYTES, DEFAULT_BATCH_WINDOW_MS, parseChannelBusFrame, parseChannelBusPayload } from "./PostgresChannelBus";
21
+ export interface ChannelBusDeps {
22
+ db: NodePgDatabase<Record<string, unknown>>;
23
+ /**
24
+ * Direct (non-pooled) Postgres URL for the LISTEN client. `LISTEN` is
25
+ * session state, so behind PgBouncer in transaction mode this must be the
26
+ * database itself and not the pooler.
27
+ */
28
+ directUrl?: string;
29
+ }
30
+ /**
31
+ * Merge `REALTIME_CHANNEL_BUS` into the configured bus.
32
+ *
33
+ * The environment wins over a *named* built-in, so the transport can be changed
34
+ * per deployment without a rebuild — the same reason `REALTIME_CDC` is an env
35
+ * var. It does **not** win over a supplied instance: the env var can only name
36
+ * transports this package knows how to construct, so honouring it there would
37
+ * mean silently discarding the object the application handed us.
38
+ */
39
+ export declare function resolveChannelBusSetting(configured?: ChannelBusSetting): ChannelBusSetting;
40
+ /**
41
+ * @deprecated Use {@link resolveChannelBusSetting}, which also accepts a
42
+ * supplied {@link ChannelBus} instance. Kept as a narrow alias so existing
43
+ * config-only callers keep their exact types.
44
+ */
45
+ export declare function resolveChannelBusConfig(configured?: ChannelBusConfig): ChannelBusConfig;
46
+ /**
47
+ * Produce the bus a setting asks for.
48
+ *
49
+ * An instance is handed straight back — constructing it was the application's
50
+ * job, and this function has nothing to add. A named built-in that turns out to
51
+ * be unusable degrades to the memory bus, with the reason logged, rather than
52
+ * throwing: a misconfigured bus should cost a deployment its cross-instance
53
+ * fan-out, not its ability to boot.
54
+ */
55
+ export declare function createChannelBus(setting: ChannelBusSetting, deps: ChannelBusDeps): ChannelBus;
@@ -0,0 +1,129 @@
1
+ /**
2
+ * Ordered, replayable per-channel message history.
3
+ *
4
+ * Broadcast on its own is fire-and-forget to whoever is connected at the
5
+ * instant it is sent: fine for presence and for "someone saved" notifications,
6
+ * not enough for op-based collaborative editing, where a client that blinks
7
+ * out for two seconds has to resync a whole document rather than catch up on
8
+ * the four operations it missed. This adds the missing half — every retained
9
+ * broadcast gets a per-channel sequence number, and a client can ask for
10
+ * everything after the last one it saw.
11
+ *
12
+ * Three decisions worth stating, because each rules out a simpler-looking one:
13
+ *
14
+ * - **Retention is server-side and opt-in.** A channel is created by whoever
15
+ * names it, so a client-supplied history depth would let any visitor commit
16
+ * the backend to unbounded storage. And presence channels — the common case
17
+ * — must not pay for this: with no rules configured nothing is written, no
18
+ * table is created, and `broadcast` runs exactly the code it ran before.
19
+ *
20
+ * - **Sequence numbers come from the database, not from a counter in this
21
+ * process.** They have to survive a restart and be shared across instances;
22
+ * an in-memory counter would restart at 1 after a deploy and hand a
23
+ * reconnecting client a replay from the wrong era, silently.
24
+ *
25
+ * - **The cursor row outlives the messages it numbered.** Pruning is what
26
+ * makes retention affordable, but pruning the cursor along with the messages
27
+ * would restart the sequence and make `sinceSeq` mean something different
28
+ * before and after — the worst kind of bug, because replay would still
29
+ * return rows and they would look plausible. Cursors are tiny and are kept
30
+ * forever; see {@link prune}, which touches only `channel_messages`.
31
+ */
32
+ import { NodePgDatabase } from "drizzle-orm/node-postgres";
33
+ import type { ChannelHistoryEntry, ChannelRetentionRule } from "@rebasepro/types";
34
+ /**
35
+ * Parse a retention TTL into milliseconds.
36
+ *
37
+ * Accepts a raw millisecond count or a short duration string (`"30s"`, `"15m"`,
38
+ * `"24h"`, `"7d"`). Returns undefined for anything unparseable, which the
39
+ * caller treats as "no TTL" — a misspelt duration must not silently become an
40
+ * aggressive one.
41
+ */
42
+ export declare function parseTtlMs(ttl: number | string | undefined): number | undefined;
43
+ /**
44
+ * Whether `channel` is covered by `rule`.
45
+ *
46
+ * Exact match, or a trailing `*` acting as a prefix. Not a general glob: this
47
+ * decides what reaches disk, and a pattern language whose reach is not obvious
48
+ * at a glance is the wrong tool for that job.
49
+ */
50
+ export declare function channelMatchesRule(channel: string, rule: ChannelRetentionRule): boolean;
51
+ /** A rule with its TTL already resolved to milliseconds. */
52
+ export interface ResolvedRetention {
53
+ limit?: number;
54
+ ttlMs?: number;
55
+ }
56
+ /**
57
+ * Persistence and replay for retained channels.
58
+ *
59
+ * Inert unless constructed with at least one rule: {@link enabled} is false,
60
+ * {@link ensureTables} does nothing, and {@link retentionFor} answers undefined
61
+ * for every channel, so the realtime service never reaches the SQL below.
62
+ */
63
+ export declare class ChannelHistoryStore {
64
+ private db;
65
+ private rules;
66
+ /** Resolved rule per channel name, so the match runs once per channel. */
67
+ private resolved;
68
+ /** Channel → timestamp of its last prune, for {@link PRUNE_THROTTLE_MS}. */
69
+ private lastPruned;
70
+ private tablesReady;
71
+ constructor(db: NodePgDatabase<Record<string, unknown>>, rules?: ChannelRetentionRule[]);
72
+ /** Whether any channel retains anything at all. */
73
+ get enabled(): boolean;
74
+ /**
75
+ * The retention that applies to `channel`, or undefined when none does.
76
+ *
77
+ * First matching rule wins, so callers order them most-specific first.
78
+ */
79
+ retentionFor(channel: string): ResolvedRetention | undefined;
80
+ /**
81
+ * Create the history tables. Idempotent, and a no-op when no rule is set —
82
+ * a deployment that never retains anything gets no schema for it.
83
+ */
84
+ ensureTables(): Promise<void>;
85
+ /**
86
+ * Append a broadcast and return the sequence number it was given.
87
+ *
88
+ * The sequence is allocated by the same statement that stores the message,
89
+ * so a crash between the two is not a possibility. `ON CONFLICT DO UPDATE`
90
+ * takes a row lock on the channel's cursor, which is what makes concurrent
91
+ * broadcasts to one channel line up in a single order — and what keeps
92
+ * different channels from contending with each other at all.
93
+ */
94
+ append(channel: string, event: string, payload: unknown, senderId?: string): Promise<{
95
+ seq: number;
96
+ at: string;
97
+ }>;
98
+ /**
99
+ * Everything retained for `channel` after `sinceSeq`, oldest first.
100
+ *
101
+ * `latestSeq` is reported whether or not the messages were capped, so a
102
+ * client that is further behind than one page can tell.
103
+ */
104
+ replay(channel: string, sinceSeq?: number, limit?: number): Promise<{
105
+ messages: ChannelHistoryEntry[];
106
+ latestSeq: number;
107
+ }>;
108
+ /**
109
+ * One retained message by its address.
110
+ *
111
+ * This is what makes the cross-instance pointer path work: a broadcast too
112
+ * large to travel inside a `pg_notify` payload is already stored here, so
113
+ * the notification carries `(channel, seq)` and each receiving instance
114
+ * reads the body back. Returns null when the message has since been pruned
115
+ * — a receiver that is that far behind has nothing useful to deliver, and
116
+ * the client's own `channel_history` replay is the repair path.
117
+ */
118
+ getBySeq(channel: string, seq: number): Promise<ChannelHistoryEntry | null>;
119
+ /**
120
+ * Enforce a channel's retention bounds.
121
+ *
122
+ * Throttled per channel, so a burst of operations prunes once rather than
123
+ * once per message — the cost then tracks elapsed time instead of write
124
+ * volume, which is what makes retention affordable on a hot channel.
125
+ */
126
+ prune(channel: string, retention: ResolvedRetention): Promise<number>;
127
+ /** Forget throttle and match caches. Called on shutdown. */
128
+ clear(): void;
129
+ }
@@ -0,0 +1,66 @@
1
+ /**
2
+ * The shared presence roster.
3
+ *
4
+ * Broadcast only ever needed *fan-out* to work across instances — a frame goes
5
+ * out, whoever is connected receives it. Presence needs more than that, because
6
+ * `presence_state` is a question ("who is in this document?") and a per-process
7
+ * `Map` can only answer for the clients that happen to share a replica with the
8
+ * asker. Two people editing the same scene through different pods would each
9
+ * see an empty room while broadcasting cursors at each other perfectly.
10
+ *
11
+ * So presence gets one row per tracked client, in Postgres, readable by every
12
+ * instance. Three consequences worth stating:
13
+ *
14
+ * - **The table is the roster; the in-process map is a cache of our own
15
+ * clients.** Reads answer from the table when this store is active, so the
16
+ * answer is the same whichever instance is asked.
17
+ *
18
+ * - **`last_seen` is the liveness signal, and it is already there.** The client
19
+ * heartbeats presence every ~20 s against a 30 s window; the sweep that has
20
+ * always reaped local stale entries now also reaps rows belonging to
21
+ * instances that stopped writing — which is exactly what a crashed pod looks
22
+ * like. Crash recovery is a property of the TTL, not a separate mechanism.
23
+ *
24
+ * - **The sweep deletes with `RETURNING`.** Whichever instance wins the delete
25
+ * is the one that announces the departures, so a stale client produces one
26
+ * `presence_diff` for the cluster rather than one per replica.
27
+ */
28
+ import { NodePgDatabase } from "drizzle-orm/node-postgres";
29
+ /** A tracked client, as any instance sees it. */
30
+ export interface PresenceRow {
31
+ channel: string;
32
+ clientId: string;
33
+ state: Record<string, unknown>;
34
+ }
35
+ export declare class ChannelPresenceStore {
36
+ private readonly db;
37
+ private readonly instanceId;
38
+ private tablesReady;
39
+ constructor(db: NodePgDatabase<Record<string, unknown>>, instanceId: string);
40
+ /** Create the roster table. Idempotent. */
41
+ ensureTables(): Promise<void>;
42
+ /** Record (or refresh) a client's presence. */
43
+ track(channel: string, clientId: string, state: Record<string, unknown>): Promise<void>;
44
+ /** Drop one client's presence in one channel. */
45
+ remove(channel: string, clientId: string): Promise<void>;
46
+ /** Drop a client from every channel — used when its socket closes. */
47
+ removeClient(clientId: string): Promise<void>;
48
+ /** The global roster for a channel. */
49
+ roster(channel: string): Promise<Record<string, Record<string, unknown>>>;
50
+ /**
51
+ * Reap rows this instance is not responsible for and that have gone quiet.
52
+ *
53
+ * Own rows are excluded because the in-process sweep already handles them —
54
+ * and handles them better, since it can tell "the socket is gone" from "the
55
+ * heartbeat is late". What is left is precisely the interesting case: rows
56
+ * written by an instance that is no longer writing.
57
+ *
58
+ * Returns what was removed, so the caller can announce it.
59
+ */
60
+ sweepStale(ttlMs: number): Promise<PresenceRow[]>;
61
+ /**
62
+ * Remove every row this instance owns. Called on graceful shutdown so a
63
+ * rolling deploy does not leave a TTL window of ghosts in every roster.
64
+ */
65
+ removeInstance(): Promise<void>;
66
+ }
@@ -0,0 +1,47 @@
1
+ /**
2
+ * A dedicated, self-healing Postgres `LISTEN` connection.
3
+ *
4
+ * Every cross-instance feature in the backend needs the same thing: one
5
+ * connection *outside* the Drizzle pool that stays open, holds a `LISTEN`, and
6
+ * comes back on its own after the database or the network drops it. CDC needed
7
+ * it first; the channel bus needs it too. This is that connection, with the one
8
+ * behaviour that matters to callers preserved: the **first** connect is
9
+ * validated and rethrown, so a caller can fall back to a different strategy,
10
+ * while every later drop is repaired quietly in the background.
11
+ *
12
+ * `LISTEN` is session state, so this connection must not go through a
13
+ * transaction-mode pooler (PgBouncer): give it the direct database URL.
14
+ */
15
+ export interface PgNotifyListenerOptions {
16
+ /** Direct Postgres connection string (must bypass a transaction-mode pooler). */
17
+ connectionString: string;
18
+ /** NOTIFY channel to LISTEN on. Must be a plain identifier — it is interpolated. */
19
+ channel: string;
20
+ /** Called for every notification payload received. */
21
+ onPayload: (payload: string) => void | Promise<void>;
22
+ /** Prefix for log lines, e.g. `"[CDC]"`. */
23
+ logLabel: string;
24
+ /** Delay before a reconnect attempt. */
25
+ reconnectDelayMs?: number;
26
+ }
27
+ export declare class PgNotifyListener {
28
+ private readonly options;
29
+ private client?;
30
+ private running;
31
+ private reconnectTimer?;
32
+ constructor(options: PgNotifyListenerOptions);
33
+ /** Whether the listener is meant to be connected right now. */
34
+ get active(): boolean;
35
+ /**
36
+ * Connect and begin listening. Idempotent.
37
+ *
38
+ * Rejects if the *initial* connection or `LISTEN` fails, leaving the
39
+ * listener stopped — callers use that to degrade deliberately instead of
40
+ * running blind against a channel nothing is delivering.
41
+ */
42
+ start(): Promise<void>;
43
+ /** Stop listening and release the connection. Idempotent. */
44
+ stop(): Promise<void>;
45
+ private connect;
46
+ private scheduleReconnect;
47
+ }
@@ -4,12 +4,14 @@ import { DataDriver, WebSocketMessage } from "@rebasepro/types";
4
4
  import { NodePgDatabase } from "drizzle-orm/node-postgres";
5
5
  import { RealtimeProvider, CollectionSubscriptionConfig, SingleSubscriptionConfig } from "../interfaces";
6
6
  import { PostgresCollectionRegistry } from "../collections/PostgresCollectionRegistry";
7
+ import { ChannelBus } from "./channel-bus";
8
+ import type { ChannelRetentionRule } from "@rebasepro/types";
7
9
  /**
8
10
  * Auth context stored per-subscription so real-time refetches respect RLS.
9
11
  * Mirrors the session variables set by PostgresBackendDriver.withAuth().
10
12
  */
11
13
  export interface SubscriptionAuthContext {
12
- userId: string;
14
+ uid: string;
13
15
  roles: string[];
14
16
  }
15
17
  /**
@@ -24,8 +26,53 @@ export declare class RealtimeService extends EventEmitter implements RealtimePro
24
26
  private clients;
25
27
  private channels;
26
28
  private presence;
29
+ /**
30
+ * Ordered, replayable history for channels that opt into it.
31
+ *
32
+ * Undefined until {@link configureChannelHistory} is called, and inert even
33
+ * then unless retention rules were supplied — so presence and ephemeral
34
+ * notification channels never touch it. See `channel-history.ts`.
35
+ */
36
+ private channelHistory?;
37
+ /**
38
+ * One promise chain per retained channel, so that assigning a sequence
39
+ * number and fanning the message out happen in the same order for every
40
+ * message on that channel.
41
+ *
42
+ * Without it, two concurrent broadcasts can be numbered 4 and 5 by the
43
+ * database and still reach subscribers as 5 then 4 — live order and replay
44
+ * order would disagree, which is exactly the divergence sequence numbers
45
+ * are supposed to rule out. Keyed by channel, so unrelated channels never
46
+ * wait on each other.
47
+ */
48
+ private channelSendQueues;
49
+ /**
50
+ * Cross-instance transport for channel frames and presence.
51
+ *
52
+ * Defaults to the memory bus, which publishes nowhere — so a single-instance
53
+ * deployment runs the same fan-out it always did, with one resolved promise
54
+ * per broadcast for company. See `channel-bus/ChannelBus.ts`.
55
+ */
56
+ private bus;
57
+ /**
58
+ * The shared presence roster, present only when a real bus is active.
59
+ *
60
+ * Fan-out alone is not enough for presence: `presence_state` has to answer
61
+ * with everyone in the channel, and per-process maps can only answer for
62
+ * this replica's clients. See `channel-presence.ts`.
63
+ */
64
+ private presenceStore?;
65
+ /** Sweeps roster rows left behind by instances that stopped heartbeating. */
66
+ private presenceSweepInterval?;
67
+ /**
68
+ * Channels whose oversized ephemeral broadcasts have already been reported,
69
+ * so a hot channel logs the problem once rather than once per message.
70
+ */
71
+ private oversizedBroadcastWarned;
27
72
  private presenceInterval?;
28
73
  private static readonly PRESENCE_TIMEOUT_MS;
74
+ /** How often stale roster rows from other instances are reaped. */
75
+ private static readonly PRESENCE_SWEEP_INTERVAL_MS;
29
76
  private dataService;
30
77
  private _subscriptions;
31
78
  private subscriptionCallbacks;
@@ -197,18 +244,146 @@ export declare class RealtimeService extends EventEmitter implements RealtimePro
197
244
  joinChannel(clientId: string, channel: string): void;
198
245
  /** Leave a broadcast channel */
199
246
  leaveChannel(clientId: string, channel: string): void;
200
- /** Broadcast a message to all clients in a channel except sender */
247
+ /**
248
+ * Broadcast a message to all clients in a channel except the sender.
249
+ *
250
+ * On a channel with no retention rule this is what it always was: a
251
+ * synchronous fan-out to whoever is connected, with no sequence number, no
252
+ * SQL and no await — the body below runs to completion before returning.
253
+ *
254
+ * On a retained channel the message is durably numbered first and only then
255
+ * delivered, through a per-channel queue so that delivery order matches
256
+ * sequence order. That ordering is the whole point: a client that catches up
257
+ * with `sinceSeq` has to arrive at the same state as one that never
258
+ * disconnected.
259
+ */
201
260
  broadcastToChannel(clientId: string, channel: string, event: string, payload: unknown): void;
202
- /** Track presence in a channel */
261
+ /**
262
+ * Number a broadcast, store it, then deliver it.
263
+ *
264
+ * A message that cannot be stored is **not** delivered. Delivering it would
265
+ * put it in front of live subscribers while leaving it absent from every
266
+ * future replay — the two views of the channel would disagree permanently,
267
+ * and no later message could repair the gap. Failing loudly to the sender
268
+ * instead lets it retry, which for an operation stream is the only outcome
269
+ * that keeps clients convergent.
270
+ */
271
+ private persistAndFanOut;
272
+ /** Deliver a broadcast frame to every member of a channel but the sender. */
273
+ private fanOutBroadcast;
274
+ /**
275
+ * Install the transport that carries channel frames between instances.
276
+ *
277
+ * Called once at boot. A bus that cannot start is reported and replaced with
278
+ * the memory bus: losing cross-instance fan-out degrades collaboration to
279
+ * what it was before this existed, whereas refusing to boot takes the whole
280
+ * backend down for it.
281
+ */
282
+ configureChannelBus(bus: ChannelBus): Promise<void>;
283
+ /** Which transport is in use — `"memory"` means per-instance only. */
284
+ getChannelBusKind(): ChannelBus["kind"];
285
+ /**
286
+ * Send a broadcast to the other instances.
287
+ *
288
+ * Fire-and-forget by design: the clients on this instance have already been
289
+ * served, and a bus that is briefly unreachable must not turn a broadcast
290
+ * into an error for the sender.
291
+ */
292
+ private publishBroadcast;
293
+ private publishFrame;
294
+ /**
295
+ * Tell the sender that a message was delivered locally but nowhere else.
296
+ *
297
+ * Staying quiet here would be the worst option available: on one instance
298
+ * the app works, on two it works for half the users, and nothing in the
299
+ * logs connects the two. The fix is a one-liner in config — give the
300
+ * channel a retention rule and the message travels as a pointer instead —
301
+ * so the message says exactly that.
302
+ */
303
+ private reportOversizedBroadcast;
304
+ /**
305
+ * Deliver a frame published by another instance to this one's clients.
306
+ *
307
+ * Frames we published ourselves are dropped on arrival — the local fan-out
308
+ * happened before the publish — exactly as the entity-change handler skips
309
+ * its own `sid`.
310
+ */
311
+ private handleBusFrame;
312
+ /**
313
+ * Install retention rules and create the tables they need.
314
+ *
315
+ * Safe to call with no rules (and safe not to call at all): the store stays
316
+ * inert, no schema is created, and broadcast keeps its original
317
+ * fire-and-forget path.
318
+ */
319
+ configureChannelHistory(rules: ChannelRetentionRule[] | undefined): Promise<void>;
320
+ /** Whether any channel is configured to retain messages. */
321
+ isChannelHistoryEnabled(): boolean;
322
+ /**
323
+ * Answer a client's catch-up request.
324
+ *
325
+ * A channel with no retention rule is answered with `retained: false`
326
+ * rather than an empty list, so the client can tell "you missed nothing"
327
+ * apart from "this channel never keeps anything" — the second means its
328
+ * reconnect strategy has to be a full resync, and silence would leave it
329
+ * guessing.
330
+ */
331
+ private handleChannelHistoryRequest;
332
+ private sendChannelHistory;
333
+ /**
334
+ * Track presence in a channel.
335
+ *
336
+ * The client re-sends this every ~20s as a heartbeat against the 30s
337
+ * timeout, so most calls carry the state that is already recorded. Those
338
+ * refresh `last_seen` and stop there: re-announcing an unchanged state to
339
+ * every instance would put a bus message per client per heartbeat on the
340
+ * wire to tell everyone nothing happened.
341
+ */
203
342
  trackPresence(clientId: string, channel: string, state: Record<string, unknown>): void;
204
- /** Remove presence from a channel */
205
- removePresence(clientId: string, channel: string): void;
206
- /** Send full presence state to a specific client */
343
+ /**
344
+ * Remove presence from a channel.
345
+ *
346
+ * `skipStore` is for the socket-close path, which clears every channel at
347
+ * once and then deletes the client's rows in a single statement instead of
348
+ * one per channel.
349
+ */
350
+ removePresence(clientId: string, channel: string, options?: {
351
+ skipStore?: boolean;
352
+ }): void;
353
+ /**
354
+ * Send the full roster for a channel to one client.
355
+ *
356
+ * Answered from the shared table when there is one, because "who is in this
357
+ * document?" has a single answer that must not depend on which replica the
358
+ * asker happens to be connected to. Without a bus there is nothing to share
359
+ * and the local map *is* the roster — that path stays synchronous, which is
360
+ * what it always was.
361
+ */
207
362
  sendPresenceState(clientId: string, channel: string): void;
208
- /** Broadcast presence diff (joins/leaves) to channel */
209
- private broadcastPresenceDiff;
363
+ /** Presence of the clients connected to this instance. */
364
+ private localPresences;
365
+ private sendPresenceStateMessage;
366
+ /** Deliver a presence diff to this instance's members of the channel. */
367
+ private deliverPresenceDiff;
368
+ /** Tell the other instances about a presence change. */
369
+ private publishPresenceDiff;
370
+ /** Run a roster write when there is a roster, and never let it throw. */
371
+ private presenceStoreOp;
210
372
  /** Periodic cleanup for stale presences */
211
373
  private ensurePresenceCleanup;
374
+ /**
375
+ * Reap roster rows whose owning instance stopped heartbeating.
376
+ *
377
+ * This is the cross-instance half of the sweep above, and it doubles as
378
+ * crash recovery: a pod that dies takes its clients with it but leaves
379
+ * their rows behind, and after one TTL window they look exactly like any
380
+ * other client that went quiet. The delete returns what it removed, so
381
+ * whichever instance wins the race is the one that announces the
382
+ * departures — once for the cluster, not once per replica.
383
+ */
384
+ private ensurePresenceSweep;
385
+ /** One pass of the stale-roster sweep. See {@link ensurePresenceSweep}. */
386
+ private sweepStalePresence;
212
387
  /**
213
388
  * Gracefully tear down all realtime resources.
214
389
  *