@rebasepro/server-postgres 0.9.1-canary.ff338b5 → 0.10.1-canary.14e53ae
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +21 -0
- package/dist/PostgresBackendDriver.d.ts +18 -0
- package/dist/PostgresBootstrapper.d.ts +11 -1
- package/dist/auth/services.d.ts +93 -54
- package/dist/backup/backup-logic.d.ts +23 -0
- package/dist/backup/backup-service.d.ts +44 -2
- package/dist/backup/pg-tools.d.ts +41 -1
- package/dist/chunk-DSJWtz9O.js +40 -0
- package/dist/cli-helpers.d.ts +33 -1
- package/dist/ensure-collection-tables-CNlIONzj.js +304 -0
- package/dist/ensure-collection-tables-CNlIONzj.js.map +1 -0
- package/dist/index.d.ts +1 -0
- package/dist/index.es.js +2090 -4741
- package/dist/index.es.js.map +1 -1
- package/dist/schema/auth-bootstrap-sql.d.ts +1 -1
- package/dist/schema/auth-schema.d.ts +194 -24
- package/dist/schema/destructive-sql.d.ts +49 -0
- package/dist/schema/ensure-collection-tables.d.ts +79 -0
- package/dist/schema/generate-postgres-ddl-logic.d.ts +4 -1
- package/dist/schema/introspect-db-logic.d.ts +0 -5
- package/dist/schema/introspect-db-naming.d.ts +10 -0
- package/dist/security/policy-drift.d.ts +24 -0
- package/dist/security/rls-enforcement.d.ts +2 -2
- package/dist/services/cdc/CdcListener.d.ts +7 -14
- package/dist/services/channel-bus/ChannelBus.d.ts +29 -0
- package/dist/services/channel-bus/PostgresChannelBus.d.ts +111 -0
- package/dist/services/channel-bus/index.d.ts +55 -0
- package/dist/services/channel-history.d.ts +129 -0
- package/dist/services/channel-presence.d.ts +66 -0
- package/dist/services/pg-notify-listener.d.ts +47 -0
- package/dist/services/realtimeService.d.ts +183 -8
- package/dist/src-B0v4IKaI.js +329 -0
- package/dist/src-B0v4IKaI.js.map +1 -0
- package/dist/src-DmsRg8MR.js +4056 -0
- package/dist/src-DmsRg8MR.js.map +1 -0
- package/package.json +7 -31
- package/src/PostgresBackendDriver.ts +56 -5
- package/src/PostgresBootstrapper.ts +87 -1
- package/src/auth/ensure-tables.ts +187 -19
- package/src/auth/services.ts +309 -170
- package/src/backup/backup-cli.ts +60 -1
- package/src/backup/backup-cron.ts +24 -1
- package/src/backup/backup-logic.ts +62 -0
- package/src/backup/backup-service.ts +132 -13
- package/src/backup/pg-tools.ts +70 -2
- package/src/cli-helpers.ts +82 -27
- package/src/cli.ts +152 -6
- package/src/index.ts +4 -0
- package/src/schema/auth-bootstrap-sql.ts +7 -1
- package/src/schema/auth-schema.ts +53 -15
- package/src/schema/destructive-sql.ts +94 -0
- package/src/schema/ensure-collection-tables.test.ts +156 -0
- package/src/schema/ensure-collection-tables.ts +297 -0
- package/src/schema/generate-postgres-ddl-logic.ts +3 -3
- package/src/schema/introspect-db-inference.ts +1 -1
- package/src/schema/introspect-db-logic.ts +1 -10
- package/src/schema/introspect-db-naming.ts +15 -0
- package/src/schema/introspect-runtime.ts +1 -1
- package/src/security/policy-drift.test.ts +46 -0
- package/src/security/policy-drift.ts +70 -4
- package/src/security/rls-enforcement.ts +11 -5
- package/src/services/cdc/CdcListener.ts +27 -91
- package/src/services/channel-bus/ChannelBus.ts +44 -0
- package/src/services/channel-bus/PostgresChannelBus.ts +299 -0
- package/src/services/channel-bus/index.ts +123 -0
- package/src/services/channel-history.ts +378 -0
- package/src/services/channel-presence.ts +148 -0
- package/src/services/pg-notify-listener.ts +137 -0
- package/src/services/realtimeService.ts +581 -24
- package/src/websocket.ts +30 -11
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Channel bus over Postgres LISTEN/NOTIFY.
|
|
3
|
+
*
|
|
4
|
+
* Chosen because it needs nothing that a Rebase deployment does not already
|
|
5
|
+
* have — the same database, the same direct URL the CDC listener uses. Three
|
|
6
|
+
* properties of `NOTIFY` shape everything below:
|
|
7
|
+
*
|
|
8
|
+
* - **8000 bytes per payload.** Presence and cursors fit with room to spare; a
|
|
9
|
+
* scene snapshot does not. Rather than truncate or drop, an oversized frame
|
|
10
|
+
* on a *retained* channel is published as a pointer — the body is already in
|
|
11
|
+
* `rebase.channel_messages` with a sequence number, so the receiver reads it
|
|
12
|
+
* back. That is the same trick the entity path uses (notify an address,
|
|
13
|
+
* refetch the row), applied to a different table. On an ephemeral channel
|
|
14
|
+
* there is nothing to point at, so the publish is refused loudly instead of
|
|
15
|
+
* reaching some instances and not others.
|
|
16
|
+
*
|
|
17
|
+
* - **A notify is a query on the primary database.** Not a slow one, but it
|
|
18
|
+
* competes with the application's real queries, and that — not throughput —
|
|
19
|
+
* is what actually limits this transport. Measured, it carried ~10k
|
|
20
|
+
* cross-instance messages/second and stayed flat out to eight instances; what
|
|
21
|
+
* it should not do is spend 10k queries/second of the database's budget on
|
|
22
|
+
* cursor movement. Hence the batching below.
|
|
23
|
+
*
|
|
24
|
+
* - **Delivery is best-effort.** Retained channels repair themselves through
|
|
25
|
+
* the client's history replay, so a lost frame costs a live update rather
|
|
26
|
+
* than correctness. That is what makes coalescing safe.
|
|
27
|
+
*/
|
|
28
|
+
import { NodePgDatabase } from "drizzle-orm/node-postgres";
|
|
29
|
+
import { ChannelBus, ChannelBusFrame, ChannelBusHandler } from "./ChannelBus";
|
|
30
|
+
/** NOTIFY channel carrying channel-bus frames. */
|
|
31
|
+
export declare const CHANNEL_BUS_NOTIFY_CHANNEL = "rebase_channel_bus";
|
|
32
|
+
/**
|
|
33
|
+
* Postgres refuses a NOTIFY payload of 8000 bytes or more. The margin below it
|
|
34
|
+
* is for nothing in particular — it is there so that a payload which passes this
|
|
35
|
+
* check cannot fail at the server for being a few bytes over.
|
|
36
|
+
*/
|
|
37
|
+
export declare const PG_NOTIFY_MAX_PAYLOAD_BYTES = 7500;
|
|
38
|
+
/**
|
|
39
|
+
* How long a batching window stays open.
|
|
40
|
+
*
|
|
41
|
+
* Ten milliseconds is below the threshold where a human notices a cursor lag,
|
|
42
|
+
* and it is the difference between one query per message and one query per
|
|
43
|
+
* window under load. Set to 0 to disable coalescing entirely.
|
|
44
|
+
*/
|
|
45
|
+
export declare const DEFAULT_BATCH_WINDOW_MS = 10;
|
|
46
|
+
export declare class PostgresChannelBus implements ChannelBus {
|
|
47
|
+
private readonly db;
|
|
48
|
+
private readonly connectionString;
|
|
49
|
+
readonly kind: "postgres";
|
|
50
|
+
readonly maxFrameBytes = 7500;
|
|
51
|
+
private listener?;
|
|
52
|
+
private readonly batchWindowMs;
|
|
53
|
+
/**
|
|
54
|
+
* Frames waiting for the current window to close.
|
|
55
|
+
*
|
|
56
|
+
* The window is opened by a publish that found none open, and that publish
|
|
57
|
+
* is sent *immediately* rather than joining a batch — see {@link publish}.
|
|
58
|
+
*/
|
|
59
|
+
private pending;
|
|
60
|
+
private pendingBytes;
|
|
61
|
+
private windowTimer?;
|
|
62
|
+
private stopped;
|
|
63
|
+
constructor(db: NodePgDatabase<Record<string, unknown>>, connectionString: string, options?: {
|
|
64
|
+
batchWindowMs?: number;
|
|
65
|
+
});
|
|
66
|
+
start(handler: ChannelBusHandler): Promise<void>;
|
|
67
|
+
/**
|
|
68
|
+
* Publish, coalescing under load.
|
|
69
|
+
*
|
|
70
|
+
* The window is *leading edge*: a publish arriving when no window is open is
|
|
71
|
+
* sent straight away and opens one, so an idle channel pays no added latency
|
|
72
|
+
* at all. Frames arriving while it is open are collected and leave together
|
|
73
|
+
* when it closes. The effect is that cost tracks elapsed time rather than
|
|
74
|
+
* message count — one query per window instead of one per message — which is
|
|
75
|
+
* the same shape as the retention pruning throttle, for the same reason.
|
|
76
|
+
*
|
|
77
|
+
* The returned promise settles when the frame has actually left, not when it
|
|
78
|
+
* was queued, so the contract ("reaches the other instances, or rejects")
|
|
79
|
+
* still holds.
|
|
80
|
+
*/
|
|
81
|
+
publish(frame: ChannelBusFrame): Promise<void>;
|
|
82
|
+
stop(): Promise<void>;
|
|
83
|
+
private openWindow;
|
|
84
|
+
/** Send everything queued and settle the promises waiting on it. */
|
|
85
|
+
private flush;
|
|
86
|
+
/**
|
|
87
|
+
* One NOTIFY.
|
|
88
|
+
*
|
|
89
|
+
* A single frame goes out in the plain, unwrapped shape. That is not just
|
|
90
|
+
* economy: during a rolling deploy an instance running the previous build
|
|
91
|
+
* understands only that shape, and low-rate traffic — presence, the tail of
|
|
92
|
+
* a session — is exactly what is flowing while pods restart. Batching only
|
|
93
|
+
* appears under load, which shrinks the mixed-version window to almost
|
|
94
|
+
* nothing.
|
|
95
|
+
*/
|
|
96
|
+
private send;
|
|
97
|
+
}
|
|
98
|
+
/**
|
|
99
|
+
* Parse a bus payload into the frames it carries.
|
|
100
|
+
*
|
|
101
|
+
* Accepts both wire shapes — a bare frame and a `{ batch: [...] }` envelope —
|
|
102
|
+
* so an instance on the new build understands one on the old. Returns an empty
|
|
103
|
+
* array for anything unrecognisable: a malformed or future-versioned message
|
|
104
|
+
* must never take the listener down.
|
|
105
|
+
*/
|
|
106
|
+
export declare function parseChannelBusPayload(payload: string): ChannelBusFrame[];
|
|
107
|
+
/**
|
|
108
|
+
* Parse a single bus frame, returning null for anything that is not a frame we
|
|
109
|
+
* understand.
|
|
110
|
+
*/
|
|
111
|
+
export declare function parseChannelBusFrame(payload: string): ChannelBusFrame | null;
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Resolution of the channel bus from config, environment, or a supplied instance.
|
|
3
|
+
*
|
|
4
|
+
* Opt-in, like every other cross-cutting realtime switch here: with nothing
|
|
5
|
+
* configured a deployment gets the memory bus and behaves exactly as it did
|
|
6
|
+
* before this existed. Unlike `REALTIME_CDC=auto`, there is no "try it and see"
|
|
7
|
+
* default — a bus changes where messages go, and quietly turning on a Postgres
|
|
8
|
+
* NOTIFY per broadcast because a direct URL happened to be set is not a
|
|
9
|
+
* decision to make on the user's behalf.
|
|
10
|
+
*
|
|
11
|
+
* Two transports ship, and neither adds a service to a deployment. A third is
|
|
12
|
+
* not a code change here: `realtime.bus` also accepts an already-constructed
|
|
13
|
+
* {@link ChannelBus}, so a transport published as its own package plugs in
|
|
14
|
+
* without this file learning about it. See `@rebasepro/types` →
|
|
15
|
+
* `types/channel_bus.ts` for the contract such a package implements.
|
|
16
|
+
*/
|
|
17
|
+
import { NodePgDatabase } from "drizzle-orm/node-postgres";
|
|
18
|
+
import { type ChannelBus, type ChannelBusConfig, type ChannelBusSetting } from "@rebasepro/types";
|
|
19
|
+
export * from "./ChannelBus";
|
|
20
|
+
export { PostgresChannelBus, CHANNEL_BUS_NOTIFY_CHANNEL, PG_NOTIFY_MAX_PAYLOAD_BYTES, DEFAULT_BATCH_WINDOW_MS, parseChannelBusFrame, parseChannelBusPayload } from "./PostgresChannelBus";
|
|
21
|
+
export interface ChannelBusDeps {
|
|
22
|
+
db: NodePgDatabase<Record<string, unknown>>;
|
|
23
|
+
/**
|
|
24
|
+
* Direct (non-pooled) Postgres URL for the LISTEN client. `LISTEN` is
|
|
25
|
+
* session state, so behind PgBouncer in transaction mode this must be the
|
|
26
|
+
* database itself and not the pooler.
|
|
27
|
+
*/
|
|
28
|
+
directUrl?: string;
|
|
29
|
+
}
|
|
30
|
+
/**
|
|
31
|
+
* Merge `REALTIME_CHANNEL_BUS` into the configured bus.
|
|
32
|
+
*
|
|
33
|
+
* The environment wins over a *named* built-in, so the transport can be changed
|
|
34
|
+
* per deployment without a rebuild — the same reason `REALTIME_CDC` is an env
|
|
35
|
+
* var. It does **not** win over a supplied instance: the env var can only name
|
|
36
|
+
* transports this package knows how to construct, so honouring it there would
|
|
37
|
+
* mean silently discarding the object the application handed us.
|
|
38
|
+
*/
|
|
39
|
+
export declare function resolveChannelBusSetting(configured?: ChannelBusSetting): ChannelBusSetting;
|
|
40
|
+
/**
|
|
41
|
+
* @deprecated Use {@link resolveChannelBusSetting}, which also accepts a
|
|
42
|
+
* supplied {@link ChannelBus} instance. Kept as a narrow alias so existing
|
|
43
|
+
* config-only callers keep their exact types.
|
|
44
|
+
*/
|
|
45
|
+
export declare function resolveChannelBusConfig(configured?: ChannelBusConfig): ChannelBusConfig;
|
|
46
|
+
/**
|
|
47
|
+
* Produce the bus a setting asks for.
|
|
48
|
+
*
|
|
49
|
+
* An instance is handed straight back — constructing it was the application's
|
|
50
|
+
* job, and this function has nothing to add. A named built-in that turns out to
|
|
51
|
+
* be unusable degrades to the memory bus, with the reason logged, rather than
|
|
52
|
+
* throwing: a misconfigured bus should cost a deployment its cross-instance
|
|
53
|
+
* fan-out, not its ability to boot.
|
|
54
|
+
*/
|
|
55
|
+
export declare function createChannelBus(setting: ChannelBusSetting, deps: ChannelBusDeps): ChannelBus;
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Ordered, replayable per-channel message history.
|
|
3
|
+
*
|
|
4
|
+
* Broadcast on its own is fire-and-forget to whoever is connected at the
|
|
5
|
+
* instant it is sent: fine for presence and for "someone saved" notifications,
|
|
6
|
+
* not enough for op-based collaborative editing, where a client that blinks
|
|
7
|
+
* out for two seconds has to resync a whole document rather than catch up on
|
|
8
|
+
* the four operations it missed. This adds the missing half — every retained
|
|
9
|
+
* broadcast gets a per-channel sequence number, and a client can ask for
|
|
10
|
+
* everything after the last one it saw.
|
|
11
|
+
*
|
|
12
|
+
* Three decisions worth stating, because each rules out a simpler-looking one:
|
|
13
|
+
*
|
|
14
|
+
* - **Retention is server-side and opt-in.** A channel is created by whoever
|
|
15
|
+
* names it, so a client-supplied history depth would let any visitor commit
|
|
16
|
+
* the backend to unbounded storage. And presence channels — the common case
|
|
17
|
+
* — must not pay for this: with no rules configured nothing is written, no
|
|
18
|
+
* table is created, and `broadcast` runs exactly the code it ran before.
|
|
19
|
+
*
|
|
20
|
+
* - **Sequence numbers come from the database, not from a counter in this
|
|
21
|
+
* process.** They have to survive a restart and be shared across instances;
|
|
22
|
+
* an in-memory counter would restart at 1 after a deploy and hand a
|
|
23
|
+
* reconnecting client a replay from the wrong era, silently.
|
|
24
|
+
*
|
|
25
|
+
* - **The cursor row outlives the messages it numbered.** Pruning is what
|
|
26
|
+
* makes retention affordable, but pruning the cursor along with the messages
|
|
27
|
+
* would restart the sequence and make `sinceSeq` mean something different
|
|
28
|
+
* before and after — the worst kind of bug, because replay would still
|
|
29
|
+
* return rows and they would look plausible. Cursors are tiny and are kept
|
|
30
|
+
* forever; see {@link prune}, which touches only `channel_messages`.
|
|
31
|
+
*/
|
|
32
|
+
import { NodePgDatabase } from "drizzle-orm/node-postgres";
|
|
33
|
+
import type { ChannelHistoryEntry, ChannelRetentionRule } from "@rebasepro/types";
|
|
34
|
+
/**
|
|
35
|
+
* Parse a retention TTL into milliseconds.
|
|
36
|
+
*
|
|
37
|
+
* Accepts a raw millisecond count or a short duration string (`"30s"`, `"15m"`,
|
|
38
|
+
* `"24h"`, `"7d"`). Returns undefined for anything unparseable, which the
|
|
39
|
+
* caller treats as "no TTL" — a misspelt duration must not silently become an
|
|
40
|
+
* aggressive one.
|
|
41
|
+
*/
|
|
42
|
+
export declare function parseTtlMs(ttl: number | string | undefined): number | undefined;
|
|
43
|
+
/**
|
|
44
|
+
* Whether `channel` is covered by `rule`.
|
|
45
|
+
*
|
|
46
|
+
* Exact match, or a trailing `*` acting as a prefix. Not a general glob: this
|
|
47
|
+
* decides what reaches disk, and a pattern language whose reach is not obvious
|
|
48
|
+
* at a glance is the wrong tool for that job.
|
|
49
|
+
*/
|
|
50
|
+
export declare function channelMatchesRule(channel: string, rule: ChannelRetentionRule): boolean;
|
|
51
|
+
/** A rule with its TTL already resolved to milliseconds. */
|
|
52
|
+
export interface ResolvedRetention {
|
|
53
|
+
limit?: number;
|
|
54
|
+
ttlMs?: number;
|
|
55
|
+
}
|
|
56
|
+
/**
|
|
57
|
+
* Persistence and replay for retained channels.
|
|
58
|
+
*
|
|
59
|
+
* Inert unless constructed with at least one rule: {@link enabled} is false,
|
|
60
|
+
* {@link ensureTables} does nothing, and {@link retentionFor} answers undefined
|
|
61
|
+
* for every channel, so the realtime service never reaches the SQL below.
|
|
62
|
+
*/
|
|
63
|
+
export declare class ChannelHistoryStore {
|
|
64
|
+
private db;
|
|
65
|
+
private rules;
|
|
66
|
+
/** Resolved rule per channel name, so the match runs once per channel. */
|
|
67
|
+
private resolved;
|
|
68
|
+
/** Channel → timestamp of its last prune, for {@link PRUNE_THROTTLE_MS}. */
|
|
69
|
+
private lastPruned;
|
|
70
|
+
private tablesReady;
|
|
71
|
+
constructor(db: NodePgDatabase<Record<string, unknown>>, rules?: ChannelRetentionRule[]);
|
|
72
|
+
/** Whether any channel retains anything at all. */
|
|
73
|
+
get enabled(): boolean;
|
|
74
|
+
/**
|
|
75
|
+
* The retention that applies to `channel`, or undefined when none does.
|
|
76
|
+
*
|
|
77
|
+
* First matching rule wins, so callers order them most-specific first.
|
|
78
|
+
*/
|
|
79
|
+
retentionFor(channel: string): ResolvedRetention | undefined;
|
|
80
|
+
/**
|
|
81
|
+
* Create the history tables. Idempotent, and a no-op when no rule is set —
|
|
82
|
+
* a deployment that never retains anything gets no schema for it.
|
|
83
|
+
*/
|
|
84
|
+
ensureTables(): Promise<void>;
|
|
85
|
+
/**
|
|
86
|
+
* Append a broadcast and return the sequence number it was given.
|
|
87
|
+
*
|
|
88
|
+
* The sequence is allocated by the same statement that stores the message,
|
|
89
|
+
* so a crash between the two is not a possibility. `ON CONFLICT DO UPDATE`
|
|
90
|
+
* takes a row lock on the channel's cursor, which is what makes concurrent
|
|
91
|
+
* broadcasts to one channel line up in a single order — and what keeps
|
|
92
|
+
* different channels from contending with each other at all.
|
|
93
|
+
*/
|
|
94
|
+
append(channel: string, event: string, payload: unknown, senderId?: string): Promise<{
|
|
95
|
+
seq: number;
|
|
96
|
+
at: string;
|
|
97
|
+
}>;
|
|
98
|
+
/**
|
|
99
|
+
* Everything retained for `channel` after `sinceSeq`, oldest first.
|
|
100
|
+
*
|
|
101
|
+
* `latestSeq` is reported whether or not the messages were capped, so a
|
|
102
|
+
* client that is further behind than one page can tell.
|
|
103
|
+
*/
|
|
104
|
+
replay(channel: string, sinceSeq?: number, limit?: number): Promise<{
|
|
105
|
+
messages: ChannelHistoryEntry[];
|
|
106
|
+
latestSeq: number;
|
|
107
|
+
}>;
|
|
108
|
+
/**
|
|
109
|
+
* One retained message by its address.
|
|
110
|
+
*
|
|
111
|
+
* This is what makes the cross-instance pointer path work: a broadcast too
|
|
112
|
+
* large to travel inside a `pg_notify` payload is already stored here, so
|
|
113
|
+
* the notification carries `(channel, seq)` and each receiving instance
|
|
114
|
+
* reads the body back. Returns null when the message has since been pruned
|
|
115
|
+
* — a receiver that is that far behind has nothing useful to deliver, and
|
|
116
|
+
* the client's own `channel_history` replay is the repair path.
|
|
117
|
+
*/
|
|
118
|
+
getBySeq(channel: string, seq: number): Promise<ChannelHistoryEntry | null>;
|
|
119
|
+
/**
|
|
120
|
+
* Enforce a channel's retention bounds.
|
|
121
|
+
*
|
|
122
|
+
* Throttled per channel, so a burst of operations prunes once rather than
|
|
123
|
+
* once per message — the cost then tracks elapsed time instead of write
|
|
124
|
+
* volume, which is what makes retention affordable on a hot channel.
|
|
125
|
+
*/
|
|
126
|
+
prune(channel: string, retention: ResolvedRetention): Promise<number>;
|
|
127
|
+
/** Forget throttle and match caches. Called on shutdown. */
|
|
128
|
+
clear(): void;
|
|
129
|
+
}
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The shared presence roster.
|
|
3
|
+
*
|
|
4
|
+
* Broadcast only ever needed *fan-out* to work across instances — a frame goes
|
|
5
|
+
* out, whoever is connected receives it. Presence needs more than that, because
|
|
6
|
+
* `presence_state` is a question ("who is in this document?") and a per-process
|
|
7
|
+
* `Map` can only answer for the clients that happen to share a replica with the
|
|
8
|
+
* asker. Two people editing the same scene through different pods would each
|
|
9
|
+
* see an empty room while broadcasting cursors at each other perfectly.
|
|
10
|
+
*
|
|
11
|
+
* So presence gets one row per tracked client, in Postgres, readable by every
|
|
12
|
+
* instance. Three consequences worth stating:
|
|
13
|
+
*
|
|
14
|
+
* - **The table is the roster; the in-process map is a cache of our own
|
|
15
|
+
* clients.** Reads answer from the table when this store is active, so the
|
|
16
|
+
* answer is the same whichever instance is asked.
|
|
17
|
+
*
|
|
18
|
+
* - **`last_seen` is the liveness signal, and it is already there.** The client
|
|
19
|
+
* heartbeats presence every ~20 s against a 30 s window; the sweep that has
|
|
20
|
+
* always reaped local stale entries now also reaps rows belonging to
|
|
21
|
+
* instances that stopped writing — which is exactly what a crashed pod looks
|
|
22
|
+
* like. Crash recovery is a property of the TTL, not a separate mechanism.
|
|
23
|
+
*
|
|
24
|
+
* - **The sweep deletes with `RETURNING`.** Whichever instance wins the delete
|
|
25
|
+
* is the one that announces the departures, so a stale client produces one
|
|
26
|
+
* `presence_diff` for the cluster rather than one per replica.
|
|
27
|
+
*/
|
|
28
|
+
import { NodePgDatabase } from "drizzle-orm/node-postgres";
|
|
29
|
+
/** A tracked client, as any instance sees it. */
|
|
30
|
+
export interface PresenceRow {
|
|
31
|
+
channel: string;
|
|
32
|
+
clientId: string;
|
|
33
|
+
state: Record<string, unknown>;
|
|
34
|
+
}
|
|
35
|
+
export declare class ChannelPresenceStore {
|
|
36
|
+
private readonly db;
|
|
37
|
+
private readonly instanceId;
|
|
38
|
+
private tablesReady;
|
|
39
|
+
constructor(db: NodePgDatabase<Record<string, unknown>>, instanceId: string);
|
|
40
|
+
/** Create the roster table. Idempotent. */
|
|
41
|
+
ensureTables(): Promise<void>;
|
|
42
|
+
/** Record (or refresh) a client's presence. */
|
|
43
|
+
track(channel: string, clientId: string, state: Record<string, unknown>): Promise<void>;
|
|
44
|
+
/** Drop one client's presence in one channel. */
|
|
45
|
+
remove(channel: string, clientId: string): Promise<void>;
|
|
46
|
+
/** Drop a client from every channel — used when its socket closes. */
|
|
47
|
+
removeClient(clientId: string): Promise<void>;
|
|
48
|
+
/** The global roster for a channel. */
|
|
49
|
+
roster(channel: string): Promise<Record<string, Record<string, unknown>>>;
|
|
50
|
+
/**
|
|
51
|
+
* Reap rows this instance is not responsible for and that have gone quiet.
|
|
52
|
+
*
|
|
53
|
+
* Own rows are excluded because the in-process sweep already handles them —
|
|
54
|
+
* and handles them better, since it can tell "the socket is gone" from "the
|
|
55
|
+
* heartbeat is late". What is left is precisely the interesting case: rows
|
|
56
|
+
* written by an instance that is no longer writing.
|
|
57
|
+
*
|
|
58
|
+
* Returns what was removed, so the caller can announce it.
|
|
59
|
+
*/
|
|
60
|
+
sweepStale(ttlMs: number): Promise<PresenceRow[]>;
|
|
61
|
+
/**
|
|
62
|
+
* Remove every row this instance owns. Called on graceful shutdown so a
|
|
63
|
+
* rolling deploy does not leave a TTL window of ghosts in every roster.
|
|
64
|
+
*/
|
|
65
|
+
removeInstance(): Promise<void>;
|
|
66
|
+
}
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A dedicated, self-healing Postgres `LISTEN` connection.
|
|
3
|
+
*
|
|
4
|
+
* Every cross-instance feature in the backend needs the same thing: one
|
|
5
|
+
* connection *outside* the Drizzle pool that stays open, holds a `LISTEN`, and
|
|
6
|
+
* comes back on its own after the database or the network drops it. CDC needed
|
|
7
|
+
* it first; the channel bus needs it too. This is that connection, with the one
|
|
8
|
+
* behaviour that matters to callers preserved: the **first** connect is
|
|
9
|
+
* validated and rethrown, so a caller can fall back to a different strategy,
|
|
10
|
+
* while every later drop is repaired quietly in the background.
|
|
11
|
+
*
|
|
12
|
+
* `LISTEN` is session state, so this connection must not go through a
|
|
13
|
+
* transaction-mode pooler (PgBouncer): give it the direct database URL.
|
|
14
|
+
*/
|
|
15
|
+
export interface PgNotifyListenerOptions {
|
|
16
|
+
/** Direct Postgres connection string (must bypass a transaction-mode pooler). */
|
|
17
|
+
connectionString: string;
|
|
18
|
+
/** NOTIFY channel to LISTEN on. Must be a plain identifier — it is interpolated. */
|
|
19
|
+
channel: string;
|
|
20
|
+
/** Called for every notification payload received. */
|
|
21
|
+
onPayload: (payload: string) => void | Promise<void>;
|
|
22
|
+
/** Prefix for log lines, e.g. `"[CDC]"`. */
|
|
23
|
+
logLabel: string;
|
|
24
|
+
/** Delay before a reconnect attempt. */
|
|
25
|
+
reconnectDelayMs?: number;
|
|
26
|
+
}
|
|
27
|
+
export declare class PgNotifyListener {
|
|
28
|
+
private readonly options;
|
|
29
|
+
private client?;
|
|
30
|
+
private running;
|
|
31
|
+
private reconnectTimer?;
|
|
32
|
+
constructor(options: PgNotifyListenerOptions);
|
|
33
|
+
/** Whether the listener is meant to be connected right now. */
|
|
34
|
+
get active(): boolean;
|
|
35
|
+
/**
|
|
36
|
+
* Connect and begin listening. Idempotent.
|
|
37
|
+
*
|
|
38
|
+
* Rejects if the *initial* connection or `LISTEN` fails, leaving the
|
|
39
|
+
* listener stopped — callers use that to degrade deliberately instead of
|
|
40
|
+
* running blind against a channel nothing is delivering.
|
|
41
|
+
*/
|
|
42
|
+
start(): Promise<void>;
|
|
43
|
+
/** Stop listening and release the connection. Idempotent. */
|
|
44
|
+
stop(): Promise<void>;
|
|
45
|
+
private connect;
|
|
46
|
+
private scheduleReconnect;
|
|
47
|
+
}
|
|
@@ -4,12 +4,14 @@ import { DataDriver, WebSocketMessage } from "@rebasepro/types";
|
|
|
4
4
|
import { NodePgDatabase } from "drizzle-orm/node-postgres";
|
|
5
5
|
import { RealtimeProvider, CollectionSubscriptionConfig, SingleSubscriptionConfig } from "../interfaces";
|
|
6
6
|
import { PostgresCollectionRegistry } from "../collections/PostgresCollectionRegistry";
|
|
7
|
+
import { ChannelBus } from "./channel-bus";
|
|
8
|
+
import type { ChannelRetentionRule } from "@rebasepro/types";
|
|
7
9
|
/**
|
|
8
10
|
* Auth context stored per-subscription so real-time refetches respect RLS.
|
|
9
11
|
* Mirrors the session variables set by PostgresBackendDriver.withAuth().
|
|
10
12
|
*/
|
|
11
13
|
export interface SubscriptionAuthContext {
|
|
12
|
-
|
|
14
|
+
uid: string;
|
|
13
15
|
roles: string[];
|
|
14
16
|
}
|
|
15
17
|
/**
|
|
@@ -24,8 +26,53 @@ export declare class RealtimeService extends EventEmitter implements RealtimePro
|
|
|
24
26
|
private clients;
|
|
25
27
|
private channels;
|
|
26
28
|
private presence;
|
|
29
|
+
/**
|
|
30
|
+
* Ordered, replayable history for channels that opt into it.
|
|
31
|
+
*
|
|
32
|
+
* Undefined until {@link configureChannelHistory} is called, and inert even
|
|
33
|
+
* then unless retention rules were supplied — so presence and ephemeral
|
|
34
|
+
* notification channels never touch it. See `channel-history.ts`.
|
|
35
|
+
*/
|
|
36
|
+
private channelHistory?;
|
|
37
|
+
/**
|
|
38
|
+
* One promise chain per retained channel, so that assigning a sequence
|
|
39
|
+
* number and fanning the message out happen in the same order for every
|
|
40
|
+
* message on that channel.
|
|
41
|
+
*
|
|
42
|
+
* Without it, two concurrent broadcasts can be numbered 4 and 5 by the
|
|
43
|
+
* database and still reach subscribers as 5 then 4 — live order and replay
|
|
44
|
+
* order would disagree, which is exactly the divergence sequence numbers
|
|
45
|
+
* are supposed to rule out. Keyed by channel, so unrelated channels never
|
|
46
|
+
* wait on each other.
|
|
47
|
+
*/
|
|
48
|
+
private channelSendQueues;
|
|
49
|
+
/**
|
|
50
|
+
* Cross-instance transport for channel frames and presence.
|
|
51
|
+
*
|
|
52
|
+
* Defaults to the memory bus, which publishes nowhere — so a single-instance
|
|
53
|
+
* deployment runs the same fan-out it always did, with one resolved promise
|
|
54
|
+
* per broadcast for company. See `channel-bus/ChannelBus.ts`.
|
|
55
|
+
*/
|
|
56
|
+
private bus;
|
|
57
|
+
/**
|
|
58
|
+
* The shared presence roster, present only when a real bus is active.
|
|
59
|
+
*
|
|
60
|
+
* Fan-out alone is not enough for presence: `presence_state` has to answer
|
|
61
|
+
* with everyone in the channel, and per-process maps can only answer for
|
|
62
|
+
* this replica's clients. See `channel-presence.ts`.
|
|
63
|
+
*/
|
|
64
|
+
private presenceStore?;
|
|
65
|
+
/** Sweeps roster rows left behind by instances that stopped heartbeating. */
|
|
66
|
+
private presenceSweepInterval?;
|
|
67
|
+
/**
|
|
68
|
+
* Channels whose oversized ephemeral broadcasts have already been reported,
|
|
69
|
+
* so a hot channel logs the problem once rather than once per message.
|
|
70
|
+
*/
|
|
71
|
+
private oversizedBroadcastWarned;
|
|
27
72
|
private presenceInterval?;
|
|
28
73
|
private static readonly PRESENCE_TIMEOUT_MS;
|
|
74
|
+
/** How often stale roster rows from other instances are reaped. */
|
|
75
|
+
private static readonly PRESENCE_SWEEP_INTERVAL_MS;
|
|
29
76
|
private dataService;
|
|
30
77
|
private _subscriptions;
|
|
31
78
|
private subscriptionCallbacks;
|
|
@@ -197,18 +244,146 @@ export declare class RealtimeService extends EventEmitter implements RealtimePro
|
|
|
197
244
|
joinChannel(clientId: string, channel: string): void;
|
|
198
245
|
/** Leave a broadcast channel */
|
|
199
246
|
leaveChannel(clientId: string, channel: string): void;
|
|
200
|
-
/**
|
|
247
|
+
/**
|
|
248
|
+
* Broadcast a message to all clients in a channel except the sender.
|
|
249
|
+
*
|
|
250
|
+
* On a channel with no retention rule this is what it always was: a
|
|
251
|
+
* synchronous fan-out to whoever is connected, with no sequence number, no
|
|
252
|
+
* SQL and no await — the body below runs to completion before returning.
|
|
253
|
+
*
|
|
254
|
+
* On a retained channel the message is durably numbered first and only then
|
|
255
|
+
* delivered, through a per-channel queue so that delivery order matches
|
|
256
|
+
* sequence order. That ordering is the whole point: a client that catches up
|
|
257
|
+
* with `sinceSeq` has to arrive at the same state as one that never
|
|
258
|
+
* disconnected.
|
|
259
|
+
*/
|
|
201
260
|
broadcastToChannel(clientId: string, channel: string, event: string, payload: unknown): void;
|
|
202
|
-
/**
|
|
261
|
+
/**
|
|
262
|
+
* Number a broadcast, store it, then deliver it.
|
|
263
|
+
*
|
|
264
|
+
* A message that cannot be stored is **not** delivered. Delivering it would
|
|
265
|
+
* put it in front of live subscribers while leaving it absent from every
|
|
266
|
+
* future replay — the two views of the channel would disagree permanently,
|
|
267
|
+
* and no later message could repair the gap. Failing loudly to the sender
|
|
268
|
+
* instead lets it retry, which for an operation stream is the only outcome
|
|
269
|
+
* that keeps clients convergent.
|
|
270
|
+
*/
|
|
271
|
+
private persistAndFanOut;
|
|
272
|
+
/** Deliver a broadcast frame to every member of a channel but the sender. */
|
|
273
|
+
private fanOutBroadcast;
|
|
274
|
+
/**
|
|
275
|
+
* Install the transport that carries channel frames between instances.
|
|
276
|
+
*
|
|
277
|
+
* Called once at boot. A bus that cannot start is reported and replaced with
|
|
278
|
+
* the memory bus: losing cross-instance fan-out degrades collaboration to
|
|
279
|
+
* what it was before this existed, whereas refusing to boot takes the whole
|
|
280
|
+
* backend down for it.
|
|
281
|
+
*/
|
|
282
|
+
configureChannelBus(bus: ChannelBus): Promise<void>;
|
|
283
|
+
/** Which transport is in use — `"memory"` means per-instance only. */
|
|
284
|
+
getChannelBusKind(): ChannelBus["kind"];
|
|
285
|
+
/**
|
|
286
|
+
* Send a broadcast to the other instances.
|
|
287
|
+
*
|
|
288
|
+
* Fire-and-forget by design: the clients on this instance have already been
|
|
289
|
+
* served, and a bus that is briefly unreachable must not turn a broadcast
|
|
290
|
+
* into an error for the sender.
|
|
291
|
+
*/
|
|
292
|
+
private publishBroadcast;
|
|
293
|
+
private publishFrame;
|
|
294
|
+
/**
|
|
295
|
+
* Tell the sender that a message was delivered locally but nowhere else.
|
|
296
|
+
*
|
|
297
|
+
* Staying quiet here would be the worst option available: on one instance
|
|
298
|
+
* the app works, on two it works for half the users, and nothing in the
|
|
299
|
+
* logs connects the two. The fix is a one-liner in config — give the
|
|
300
|
+
* channel a retention rule and the message travels as a pointer instead —
|
|
301
|
+
* so the message says exactly that.
|
|
302
|
+
*/
|
|
303
|
+
private reportOversizedBroadcast;
|
|
304
|
+
/**
|
|
305
|
+
* Deliver a frame published by another instance to this one's clients.
|
|
306
|
+
*
|
|
307
|
+
* Frames we published ourselves are dropped on arrival — the local fan-out
|
|
308
|
+
* happened before the publish — exactly as the entity-change handler skips
|
|
309
|
+
* its own `sid`.
|
|
310
|
+
*/
|
|
311
|
+
private handleBusFrame;
|
|
312
|
+
/**
|
|
313
|
+
* Install retention rules and create the tables they need.
|
|
314
|
+
*
|
|
315
|
+
* Safe to call with no rules (and safe not to call at all): the store stays
|
|
316
|
+
* inert, no schema is created, and broadcast keeps its original
|
|
317
|
+
* fire-and-forget path.
|
|
318
|
+
*/
|
|
319
|
+
configureChannelHistory(rules: ChannelRetentionRule[] | undefined): Promise<void>;
|
|
320
|
+
/** Whether any channel is configured to retain messages. */
|
|
321
|
+
isChannelHistoryEnabled(): boolean;
|
|
322
|
+
/**
|
|
323
|
+
* Answer a client's catch-up request.
|
|
324
|
+
*
|
|
325
|
+
* A channel with no retention rule is answered with `retained: false`
|
|
326
|
+
* rather than an empty list, so the client can tell "you missed nothing"
|
|
327
|
+
* apart from "this channel never keeps anything" — the second means its
|
|
328
|
+
* reconnect strategy has to be a full resync, and silence would leave it
|
|
329
|
+
* guessing.
|
|
330
|
+
*/
|
|
331
|
+
private handleChannelHistoryRequest;
|
|
332
|
+
private sendChannelHistory;
|
|
333
|
+
/**
|
|
334
|
+
* Track presence in a channel.
|
|
335
|
+
*
|
|
336
|
+
* The client re-sends this every ~20s as a heartbeat against the 30s
|
|
337
|
+
* timeout, so most calls carry the state that is already recorded. Those
|
|
338
|
+
* refresh `last_seen` and stop there: re-announcing an unchanged state to
|
|
339
|
+
* every instance would put a bus message per client per heartbeat on the
|
|
340
|
+
* wire to tell everyone nothing happened.
|
|
341
|
+
*/
|
|
203
342
|
trackPresence(clientId: string, channel: string, state: Record<string, unknown>): void;
|
|
204
|
-
/**
|
|
205
|
-
|
|
206
|
-
|
|
343
|
+
/**
|
|
344
|
+
* Remove presence from a channel.
|
|
345
|
+
*
|
|
346
|
+
* `skipStore` is for the socket-close path, which clears every channel at
|
|
347
|
+
* once and then deletes the client's rows in a single statement instead of
|
|
348
|
+
* one per channel.
|
|
349
|
+
*/
|
|
350
|
+
removePresence(clientId: string, channel: string, options?: {
|
|
351
|
+
skipStore?: boolean;
|
|
352
|
+
}): void;
|
|
353
|
+
/**
|
|
354
|
+
* Send the full roster for a channel to one client.
|
|
355
|
+
*
|
|
356
|
+
* Answered from the shared table when there is one, because "who is in this
|
|
357
|
+
* document?" has a single answer that must not depend on which replica the
|
|
358
|
+
* asker happens to be connected to. Without a bus there is nothing to share
|
|
359
|
+
* and the local map *is* the roster — that path stays synchronous, which is
|
|
360
|
+
* what it always was.
|
|
361
|
+
*/
|
|
207
362
|
sendPresenceState(clientId: string, channel: string): void;
|
|
208
|
-
/**
|
|
209
|
-
private
|
|
363
|
+
/** Presence of the clients connected to this instance. */
|
|
364
|
+
private localPresences;
|
|
365
|
+
private sendPresenceStateMessage;
|
|
366
|
+
/** Deliver a presence diff to this instance's members of the channel. */
|
|
367
|
+
private deliverPresenceDiff;
|
|
368
|
+
/** Tell the other instances about a presence change. */
|
|
369
|
+
private publishPresenceDiff;
|
|
370
|
+
/** Run a roster write when there is a roster, and never let it throw. */
|
|
371
|
+
private presenceStoreOp;
|
|
210
372
|
/** Periodic cleanup for stale presences */
|
|
211
373
|
private ensurePresenceCleanup;
|
|
374
|
+
/**
|
|
375
|
+
* Reap roster rows whose owning instance stopped heartbeating.
|
|
376
|
+
*
|
|
377
|
+
* This is the cross-instance half of the sweep above, and it doubles as
|
|
378
|
+
* crash recovery: a pod that dies takes its clients with it but leaves
|
|
379
|
+
* their rows behind, and after one TTL window they look exactly like any
|
|
380
|
+
* other client that went quiet. The delete returns what it removed, so
|
|
381
|
+
* whichever instance wins the race is the one that announces the
|
|
382
|
+
* departures — once for the cluster, not once per replica.
|
|
383
|
+
*/
|
|
384
|
+
private ensurePresenceSweep;
|
|
385
|
+
/** One pass of the stale-roster sweep. See {@link ensurePresenceSweep}. */
|
|
386
|
+
private sweepStalePresence;
|
|
212
387
|
/**
|
|
213
388
|
* Gracefully tear down all realtime resources.
|
|
214
389
|
*
|