@mcpwarp/ws-mixer 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +357 -0
- package/LICENSE +202 -0
- package/README.md +431 -0
- package/dist/client.d.ts +453 -0
- package/dist/conn.d.ts +328 -0
- package/dist/control.d.ts +74 -0
- package/dist/errors.d.ts +96 -0
- package/dist/frame.d.ts +52 -0
- package/dist/index.cjs +2866 -0
- package/dist/index.cjs.map +1 -0
- package/dist/index.d.ts +13 -0
- package/dist/index.js +2806 -0
- package/dist/index.js.map +1 -0
- package/dist/stream.d.ts +124 -0
- package/dist/util.d.ts +3 -0
- package/package.json +59 -0
package/README.md
ADDED
|
@@ -0,0 +1,431 @@
|
|
|
1
|
+
# @mcpwarp/ws-mixer
|
|
2
|
+
|
|
3
|
+
JS/TypeScript client SDK for `ws-mixer.v1`: N independent byte streams multiplexed over one
|
|
4
|
+
WebSocket connection. See [`ws-mixer-spec`'s `docs/WIRE.md`](https://github.com/mcpwarp/ws-mixer-spec/blob/main/docs/WIRE.md)
|
|
5
|
+
for the normative wire spec; this package is always the **answering peer** (client) -- it never
|
|
6
|
+
opens streams, only receives `OPEN` from the server and reads/writes/closes/resets what arrives.
|
|
7
|
+
|
|
8
|
+
See [`docs/DESIGN.md`](./docs/DESIGN.md) for this SDK's original design (package/library choice, Node
|
|
9
|
+
`Duplex` stream exposure, the event surface) — this README documents the SDK as actually built, which has
|
|
10
|
+
grown a fuller reconnect/disconnect-reason surface than the original design sketch; where they differ,
|
|
11
|
+
this README and the code win.
|
|
12
|
+
|
|
13
|
+
## Install
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
npm install @mcpwarp/ws-mixer
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
Requires Node >= 20. Zero runtime dependencies besides [`ws`](https://github.com/websockets/ws).
|
|
20
|
+
|
|
21
|
+
## Usage
|
|
22
|
+
|
|
23
|
+
```ts
|
|
24
|
+
import { connect } from "@mcpwarp/ws-mixer";
|
|
25
|
+
import { pipeline } from "node:stream/promises";
|
|
26
|
+
|
|
27
|
+
const client = await connect("wss://edge.mcpwarp.io/v1/tunnel", {
|
|
28
|
+
token: process.env.MCPWARP_TOKEN!, // string | (() => Promise<string> | string) -- see "Authentication" below
|
|
29
|
+
meta: { mcpwarp: { v: 1, services: [{ id: "anki", name: "Anki MCP" }] } }, // -> hello.meta
|
|
30
|
+
|
|
31
|
+
// A stream arrived. `stream` is a Node Duplex.
|
|
32
|
+
onStream: async (stream) => {
|
|
33
|
+
const chunks: Buffer[] = [];
|
|
34
|
+
for await (const chunk of stream) chunks.push(chunk as Buffer); // reads until EOF (peer's CLOSE)
|
|
35
|
+
try {
|
|
36
|
+
const res = await callLocalMcpServer(Buffer.concat(chunks));
|
|
37
|
+
stream.end(res); // CLOSE
|
|
38
|
+
} catch (e) {
|
|
39
|
+
stream.reset(2 /* INTERNAL_ERROR */, String(e));
|
|
40
|
+
}
|
|
41
|
+
},
|
|
42
|
+
|
|
43
|
+
onApp: (body) => mcpwarp.handleApp(client, body),
|
|
44
|
+
onDrain: ({ reason, deadlineMs, message }) => log.info({ reason, deadlineMs }, message ?? "server draining"),
|
|
45
|
+
|
|
46
|
+
reconnect: { base: 1000, cap: 60_000, connectTimeout: 10_000 },
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
client.on("welcome", (w) => log.info({ session: w.session }, "connected"));
|
|
50
|
+
client.on("fatal", (e) => { log.error(e); process.exit(1); }); // see "Reconnect semantics" below for the full fatal set
|
|
51
|
+
|
|
52
|
+
await client.sendApp({ mcpwarp: { v: 1, op: "unregister", id: "anki" } }); // resolves once written (see caveat below)
|
|
53
|
+
await client.close(); // drain{client_requested}, 5s grace, close 1000
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
### Handler delivery
|
|
57
|
+
|
|
58
|
+
`onStream`/`onApp`/`onDrain` are delivered in wire order from **one** delivery loop per connection,
|
|
59
|
+
one at a time -- never concurrently with each other or with themselves; a slow handler holds up the
|
|
60
|
+
next queued event, exactly like a blocking handler would in any single-threaded event loop.
|
|
61
|
+
|
|
62
|
+
An event already received when the connection ends is still owed to its handler: `error{}` is
|
|
63
|
+
always the *last* message on the wire, so a stream `OPEN`/`app`/`drain` queued ahead of it arrived
|
|
64
|
+
before the connection ended and is still delivered, even after `close()`/the underlying `MixerConn`
|
|
65
|
+
has already torn down. Concretely: **a handler MAY still fire shortly after `client.close()` (or
|
|
66
|
+
`close({message})`)'s own promise has already resolved -- or after `onDisconnect`/the `'close'` event
|
|
67
|
+
already reported that disconnect**, for a message that arrived before it. A `drain` delivered this
|
|
68
|
+
way never starts a reconnect for a connection that is no longer the live one: on a client that is
|
|
69
|
+
already closing/closed it never does (an app-initiated shutdown); on a conn that instead died for
|
|
70
|
+
an *unrelated* reason while the `drain` was still queued behind a blocked handler, its own `'close'`
|
|
71
|
+
handler already ran and already scheduled the real reconnect before the flushed `drain` is ever
|
|
72
|
+
delivered, so it's skipped there too -- the `drain` is still delivered to `onDrain`, only the
|
|
73
|
+
reconnect it would otherwise trigger is suppressed. A stream `OPEN` delivered this way hands the app
|
|
74
|
+
a stream that is already dead -- its read/write fails promptly with an error rather than hanging
|
|
75
|
+
(see "In-flight streams are lost on reconnect" below). A handler that throws or rejects mid-flush
|
|
76
|
+
doesn't stop the rest of the backlog from being delivered, and never produces an unhandled rejection
|
|
77
|
+
(`onDisconnect`'s "loud" guarantee holds for a handler failure too -- see
|
|
78
|
+
`client.stats().handlerErrors`, "Counters" below).
|
|
79
|
+
|
|
80
|
+
## Authentication
|
|
81
|
+
|
|
82
|
+
`token` is either a static string, or a provider callback (`() => string | Promise<string>`)
|
|
83
|
+
called **fresh on every dial** -- initial connect and every reconnect, never cached. This is the
|
|
84
|
+
shape to reach for whenever the token is short-lived (a Keycloak/OIDC access token, for example):
|
|
85
|
+
|
|
86
|
+
```ts
|
|
87
|
+
import { connect } from "@mcpwarp/ws-mixer";
|
|
88
|
+
|
|
89
|
+
let cached: { token: string; expiresAt: number } | null = null;
|
|
90
|
+
|
|
91
|
+
async function getToken(): Promise<string> {
|
|
92
|
+
if (cached && cached.expiresAt > Date.now() + 5000) return cached.token;
|
|
93
|
+
const res = await fetch(`${process.env.KEYCLOAK_URL}/realms/mcpwarp/protocol/openid-connect/token`, {
|
|
94
|
+
method: "POST",
|
|
95
|
+
headers: { "Content-Type": "application/x-www-form-urlencoded" },
|
|
96
|
+
body: new URLSearchParams({
|
|
97
|
+
grant_type: "client_credentials",
|
|
98
|
+
client_id: process.env.KEYCLOAK_CLIENT_ID!,
|
|
99
|
+
client_secret: process.env.KEYCLOAK_CLIENT_SECRET!,
|
|
100
|
+
}),
|
|
101
|
+
});
|
|
102
|
+
if (!res.ok) throw new Error(`token refresh failed: ${res.status}`);
|
|
103
|
+
const body = (await res.json()) as { access_token: string; expires_in: number };
|
|
104
|
+
cached = { token: body.access_token, expiresAt: Date.now() + body.expires_in * 1000 };
|
|
105
|
+
return cached.token;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
const client = await connect("wss://edge.mcpwarp.io/v1/tunnel", { token: getToken, /* ... */ });
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
A provider that throws or rejects is **fatal** by default: no retry, and the thrown/rejected
|
|
112
|
+
error is surfaced verbatim as `DisconnectReason.cause` (`message` is copied from `error.message`).
|
|
113
|
+
|
|
114
|
+
If the provider merely *couldn't obtain* a token for a temporary reason -- the network isn't back
|
|
115
|
+
yet after a laptop wakes, or the auth server is briefly unreachable while refreshing an expired
|
|
116
|
+
token -- throw or reject with a `TokenUnavailableError` instead (directly, or wrapped via `cause`)
|
|
117
|
+
to opt that one failure into the same treatment as a failed dial (a non-fatal report, normal
|
|
118
|
+
backoff) rather than going fatal:
|
|
119
|
+
|
|
120
|
+
```ts
|
|
121
|
+
import { TokenUnavailableError } from "@mcpwarp/ws-mixer";
|
|
122
|
+
|
|
123
|
+
async function getToken(): Promise<string> {
|
|
124
|
+
const res = await fetch(`${process.env.KEYCLOAK_URL}/realms/mcpwarp/protocol/openid-connect/token`, {
|
|
125
|
+
method: "POST",
|
|
126
|
+
/* ... */
|
|
127
|
+
}).catch((e) => {
|
|
128
|
+
throw new TokenUnavailableError("token refresh unreachable", { cause: e });
|
|
129
|
+
});
|
|
130
|
+
if (!res.ok) throw new Error(`token refresh failed: ${res.status}`); // an HTTP 4xx/5xx is NOT retried
|
|
131
|
+
return (await res.json()).access_token;
|
|
132
|
+
}
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
Detection is deliberately `instanceof TokenUnavailableError` only (following `cause` up to 10 links
|
|
136
|
+
deep, never a duck-typed property or method): an accidental match on some unrelated library's error
|
|
137
|
+
must never turn a genuinely fatal provider failure into an endless retry loop. Because detection is
|
|
138
|
+
`instanceof`, import `TokenUnavailableError` from the same installed copy of `@mcpwarp/ws-mixer` the
|
|
139
|
+
client itself comes from -- a duplicated install (a dedupe miss, or a bundle containing both the ESM
|
|
140
|
+
and CJS builds) makes a marked failure look unmarked, i.e. fatal.
|
|
141
|
+
|
|
142
|
+
**A pre-`welcome` token rejection gets exactly one immediate refresh-retry -- but only when `token`
|
|
143
|
+
is a provider, and it's ONE budget shared across both rejection shapes:**
|
|
144
|
+
|
|
145
|
+
- An HTTP 401 on the upgrade (the **dial** phase).
|
|
146
|
+
- A handshake-phase close carrying `UNAUTHORIZED` (`4011`) -- whether that's the normative
|
|
147
|
+
`error{code:11}` + close `4011` rejection, or a bare `4011` close with no `error{}` at all.
|
|
148
|
+
|
|
149
|
+
Either shape, on the first rejection, makes the SDK call the provider again and redial immediately
|
|
150
|
+
(no backoff). This is deliberately **not** gated by `reconnect.maxAttempts`/reconnect being disabled
|
|
151
|
+
(the rejected-token rule applies to the very first connect too, not only reconnects), and it does
|
|
152
|
+
**not** itself increment the attempt counter -- it's an immediate redial within the same connect
|
|
153
|
+
attempt, not a new reconnect cycle. A **second** rejection -- in *either* shape, so `401` then `4011`,
|
|
154
|
+
or `4011` then `401`, counts just the same as two of the same shape -- is fatal, and nothing is
|
|
155
|
+
reported for the first rejection (only the final outcome is). If the retry's own dial instead fails
|
|
156
|
+
for an unrelated, non-auth reason (a network error, an HTTP 5xx, a transport death before `welcome`),
|
|
157
|
+
that is *not* "a second rejection": it takes the ordinary recoverable path (a non-fatal report,
|
|
158
|
+
normal backoff) -- but the refresh-retry budget stays spent (see below), so a genuine token rejection
|
|
159
|
+
on some later cycle, before the client ever goes stable, is fatal with no further refresh.
|
|
160
|
+
A **static string** `token` skips the retry entirely and is fatal on the very first rejection, of any
|
|
161
|
+
of the three forms above, since there is nothing to refresh. HTTP `403`/`404` are always fatal, with
|
|
162
|
+
no retry, provider or not.
|
|
163
|
+
|
|
164
|
+
This retry budget is a client-level flag, not reset on every dial: it re-arms only once the
|
|
165
|
+
connection reaches **stability** (`reconnect.stableAfter` past `welcome` -- see "Reconnect semantics"
|
|
166
|
+
below), the same as the `4013` KEEPALIVE_TIMEOUT one-shot retry. Otherwise a server that welcomes and
|
|
167
|
+
then closes shortly after could make the client hit the token endpoint on every single reconnect
|
|
168
|
+
cycle forever. After a long, healthy (stable) session, the budget is available again -- an ordinary
|
|
169
|
+
token expiry on some later reconnect still gets its one retry.
|
|
170
|
+
|
|
171
|
+
## Reconnect semantics
|
|
172
|
+
|
|
173
|
+
`connect()` resolves once the first `welcome` completes (and **rejects** if that never happens --
|
|
174
|
+
a fatal close/HTTP status, or `maxAttempts` exhausted before any connection ever succeeded). After
|
|
175
|
+
that, `MixerClient` owns a persistent reconnect loop, a single state machine
|
|
176
|
+
(`idle -> dialing -> connected -> backoff -> ...`, terminating in `closed`) driven entirely by
|
|
177
|
+
`src/client.ts` and WIRE.md section 2.9.
|
|
178
|
+
|
|
179
|
+
Three phases, referenced throughout this section and in `DisconnectReason.phase`:
|
|
180
|
+
|
|
181
|
+
- **`dial`**: resolving the token and opening the WebSocket, up to and including the HTTP upgrade
|
|
182
|
+
response (the 101, or a rejecting status like `401`/`403`/`404`/`429`).
|
|
183
|
+
- **`handshake`**: after the 101, sending `hello` and waiting for `welcome` (or a rejection --
|
|
184
|
+
`error{}` and/or a WS close -- instead).
|
|
185
|
+
- **`connected`**: after `welcome` has been received and validated.
|
|
186
|
+
|
|
187
|
+
| Trigger | Policy |
|
|
188
|
+
|---|---|
|
|
189
|
+
| Normal disconnect (1006, 5xx, timeouts, SDK-side protocol errors 4001/4003/4004, ...) | AWS "full jitter": `delay = random(0, min(cap, base * 2^attempt))`, base 1000ms, cap 60000ms. The attempt counter (and every once-only retry budget it gates, e.g. the `4013`/auth-rejection retries below) resets only once a connection has stayed up `reconnect.stableAfter` ms past `welcome` -- **never** on `welcome` itself, and never on a bare TCP connect/101. See `stableAfter` below. |
|
|
190
|
+
| `drain` received | New connection **immediately and in parallel** (jitter `random(0, 2000)ms` only), before the old socket closes. The old connection is torn down as soon as the replacement's `welcome` lands. Stability for the replacement is measured from *its own* `welcome`, independent of the connection it replaced. If `reconnect.maxAttempts` is `0` (reconnecting disabled), no parallel connection is started: the conn is left alone, in-flight streams finish normally, and once the server's own deadline closes it with `4012` that close is reported once as a fatal disconnect (`"drained; reconnect disabled"`). |
|
|
191
|
+
| Close `4012` (`GOING_AWAY`) or `1001`, with no preceding `drain` | Same immediate-reconnect treatment as `drain` (unless it *is* the eventual close of a `drain` received with `reconnect.maxAttempts: 0`, per the row above). |
|
|
192
|
+
| Close `4013` (`KEEPALIVE_TIMEOUT`) | **One** immediate retry (no delay); if that retry itself fails to connect, falls back to normal backoff. This is a once-only budget, re-armed only at stability (see the first row). |
|
|
193
|
+
| Close `4009` (`ENHANCE_YOUR_CALM`) | Backoff starts **at the cap**, not at `base` -- an explicit "the server refused this on purpose, back off hard immediately" signal, not a workaround for anything about the attempt counter's own reset timing. |
|
|
194
|
+
| HTTP `429` during the dial, with a `Retry-After` header | Honoured verbatim (seconds or an HTTP-date) instead of the usual jitter. |
|
|
195
|
+
| A pre-`welcome` token rejection (HTTP `401` on the upgrade, or handshake-phase `4011`), with a token provider present, and the once-only refresh-retry budget not yet spent | One immediate refresh-retry: the provider is called again and the dial redialed right away (no backoff). Unlike every other row here, this is **not** gated by `reconnect.maxAttempts`/reconnect being disabled (it applies to the very first connect too, and matches `ws-mixer-go`'s own `dialAndHandshake`), and it does **not** itself increment the attempt counter -- it's an immediate redial within the same connect attempt, not a new reconnect cycle. See "Authentication" above -- this is ONE budget shared by both rejection shapes. |
|
|
196
|
+
| Close `4010` (`UNSUPPORTED`) in any phase; close `4011` (`UNAUTHORIZED`) *after* `welcome`; a pre-`welcome` token rejection (`401`/`4011`) on its second occurrence with a token provider, or on its very first occurrence with a static string; HTTP `403`/`404`; missing/mismatched subprotocol echo; a token provider that throws/rejects an UNMARKED error | **Fatal.** Surfaced on the `fatal` event and via `onDisconnect({ ..., fatal: true })`; `close()` is issued; **never retried**. `connect()` rejects if this happens before any `welcome`. |
|
|
197
|
+
| A token provider that throws/rejects a `TokenUnavailableError` (directly, or reachable via `cause`) | Treated exactly like a failed dial: phase `"dial"`, `fatal: false`, normal full-jitter backoff via the ordinary path -- it counts as a failed attempt (`reconnect.maxAttempts` applies, the stability rule is unaffected). Not gated by whether `token` is a provider or static string. See "Authentication" above. |
|
|
198
|
+
| Close `4014` (`APPLICATION_CLOSE`), connected phase | Same "start at the cap" treatment as `4009` above (WIRE.md section 2.9): it is by nature sent *after* `welcome` (the app accepted, then refused -- e.g. a per-account connection cap) -- another "refused on purpose" signal, handled the same way regardless of how recently the attempt counter last reset. Applies to both the `error{14}+close` and bare-`4014` shapes. Never emitted by ws-mixer itself -- reserved for the application above to close a connection for its own reason (the reason text is in `error.message`/the WS close reason). See `close({ message })` below. A **handshake-phase** `4014` (before `welcome`) is not special-cased: it follows the ordinary handshake-failure path, where the attempt counter climbs normally. |
|
|
199
|
+
| `reconnect.maxAttempts` exhausted | Stops reconnecting. `connect()` rejects if it never connected once. `maxAttempts` counts *consecutive reconnect attempts without a stable connection in between* -- since the attempt counter itself only resets at stability (see below), a server that always welcomes and then disconnects before a connection ever proves stable exhausts this ceiling and goes fatal exactly like a server that never welcomes at all. |
|
|
200
|
+
| `close()` called while a dial is in flight | The in-flight socket is closed as soon as the dial resolves; nothing reconnects afterward. |
|
|
201
|
+
|
|
202
|
+
### Stability and `reconnect.stableAfter`
|
|
203
|
+
|
|
204
|
+
`reconnect.stableAfter` (default `10000`ms) is how long a connection must stay up past `welcome`
|
|
205
|
+
before it's considered **stable** (WIRE.md section 2.9's `stable`). Only once a
|
|
206
|
+
connection reaches stability does the SDK reset:
|
|
207
|
+
|
|
208
|
+
- The backoff attempt counter (`attempt`), which drives both the full-jitter delay's ceiling and
|
|
209
|
+
`reconnect.maxAttempts`'s exhaustion check.
|
|
210
|
+
- The `4013` KEEPALIVE_TIMEOUT one-shot immediate-retry budget.
|
|
211
|
+
- The pre-`welcome` token-rejection one-shot refresh-retry budget (see "Authentication" above).
|
|
212
|
+
|
|
213
|
+
None of these reset on `welcome` itself, or on a bare TCP connect/101 -- a server that welcomes and
|
|
214
|
+
then disconnects shortly after (whether that's a flaky server, a load balancer resetting connections,
|
|
215
|
+
or a token-endpoint-hostile actor exploiting the refresh-retry) would otherwise reset them every
|
|
216
|
+
single cycle, turning backoff into a redial-roughly-once-a-second loop forever instead of actually
|
|
217
|
+
backing off, and could repeatedly burn the auth refresh-retry against a token service that's never
|
|
218
|
+
going to accept the same rejected credential anyway. Stability is tracked **per connection**: a
|
|
219
|
+
`drain` hand-over's retired (old) connection ending, however long after its replacement's own
|
|
220
|
+
`welcome`, never resets or clears the replacement's own stability timer -- only that specific
|
|
221
|
+
connection ending (while it's still the active one) does.
|
|
222
|
+
|
|
223
|
+
Must be a finite number `>= 0` -- the constructor throws a `RangeError` synchronously otherwise
|
|
224
|
+
(the same validation style as `close()`'s `code`), since a non-finite or negative value would
|
|
225
|
+
silently disable this entirely (Node just clamps `setTimeout`'s delay to ~1ms rather than rejecting
|
|
226
|
+
it). `0` itself stays legal and well-defined: it reproduces the pre-0.4 "reset on `welcome`"
|
|
227
|
+
behaviour, since the stability timer then fires on the very next tick after `welcome` regardless of
|
|
228
|
+
whether the connection is still up.
|
|
229
|
+
|
|
230
|
+
**Differs from `ws-mixer-go`:** here, `stableAfter: 0` means "no stability window" (see above). In
|
|
231
|
+
the Go SDK, `ReconnectOptions.StableAfter: 0` instead means "unset" and falls back to the 10s
|
|
232
|
+
default, the same way `Base`/`Cap`/`ConnectTimeout: 0` do there. A `0` config value is **not**
|
|
233
|
+
portable between the two SDKs; to genuinely disable the stability window in Go, nothing short of a
|
|
234
|
+
very small positive duration achieves it.
|
|
235
|
+
|
|
236
|
+
```ts
|
|
237
|
+
reconnect: { base: 1000, cap: 60_000, connectTimeout: 10_000, stableAfter: 10_000 }
|
|
238
|
+
```
|
|
239
|
+
|
|
240
|
+
Also true regardless of trigger:
|
|
241
|
+
|
|
242
|
+
- On every replacement, the connection being retired (a `drain`-superseded connection, or one torn
|
|
243
|
+
down by `close()` mid-handshake) is fully closed, and its timers and listeners go with it. Its
|
|
244
|
+
retirement is not itself reported as a disconnect (the client stays connected throughout, via the
|
|
245
|
+
replacement); every in-flight stream on it gets a stream-scoped `StreamError(CANCEL, "connection
|
|
246
|
+
drained")`, not a connection-level error.
|
|
247
|
+
- **In-flight streams are lost on reconnect.** There is no resumption: stream ids, buffers and
|
|
248
|
+
credit all restart from zero on the new connection. A handler must be able to tell "the
|
|
249
|
+
response ended" (`end`/EOF) from "the tunnel died" -- this SDK does not paper over the
|
|
250
|
+
difference; retry is the application's job.
|
|
251
|
+
|
|
252
|
+
Whether a stream still open when the connection ends ends cleanly or with an error depends on
|
|
253
|
+
whether the peer's own `CLOSE` for **that stream** had already arrived, per WIRE.md's "`CLOSE`
|
|
254
|
+
preserves buffered data; `RESET` discards it" rule:
|
|
255
|
+
- The peer's `CLOSE` had already arrived (its read side had already legitimately finished on the
|
|
256
|
+
wire, independent of the connection dying): the response DID complete. Whatever was buffered is
|
|
257
|
+
still delivered, `'end'` still fires once it's drained, `stream.errored` stays `null`, and no
|
|
258
|
+
`'error'` event fires -- exactly as if the connection hadn't died. Only the **write** side fails:
|
|
259
|
+
a write already pending, or started afterward, fails promptly (its callback receives the
|
|
260
|
+
connection's error) instead of hanging or silently succeeding, since the connection is gone even
|
|
261
|
+
though this stream's response was already complete.
|
|
262
|
+
- Otherwise (a peer `RESET`, or the connection ending abnormally -- `1006`, or any other way other
|
|
263
|
+
than that stream's own clean `CLOSE` -- with this stream's read side never having legitimately
|
|
264
|
+
finished): the stream always ends with an error, never a false clean end. `stream.errored`
|
|
265
|
+
carries it and `'close'` fires, whether or not the consumer attached an `'error'` listener (one
|
|
266
|
+
with only `'data'`/`'end'`/`'close'` listeners never sees `'end'` for this case, only `'close'`);
|
|
267
|
+
a read or write already pending, or started afterward, fails promptly instead of hanging. A peer
|
|
268
|
+
`RESET` always discards whatever was buffered, even if this stream's `CLOSE` had *also* already
|
|
269
|
+
arrived (WIRE.md: `RESET` always wins).
|
|
270
|
+
|
|
271
|
+
### Disconnect reason shape
|
|
272
|
+
|
|
273
|
+
Every disconnect, recoverable or fatal, is reported the same way -- to `onDisconnect(reason)` and
|
|
274
|
+
the `'close'` event -- as **exactly one** `DisconnectReason` object. In particular, when a failure is
|
|
275
|
+
also the one that exhausts `reconnect.maxAttempts`, that's still a single report: `fatal: true`, the
|
|
276
|
+
exhaustion message, but carrying the underlying failure's `wsCode`/`errorCode`/`httpStatus`/`cause` --
|
|
277
|
+
never a first report for the failure followed by a second one for giving up.
|
|
278
|
+
|
|
279
|
+
| Field | Meaning |
|
|
280
|
+
|---|---|
|
|
281
|
+
| `phase` | `"dial"` (opening the socket / the auth handshake), `"handshake"` (`hello`/`welcome` after the socket opened), or `"connected"` (after `welcome`). Always present. |
|
|
282
|
+
| `wsCode` | The WebSocket close code, when a WS close occurred (including the SDK's own locally-generated code for a close it initiated itself, e.g. `4001` for a hello/welcome timeout). When derived from a ws-mixer error (this side's own error, or a peer's `error{code}`), this is always the semantic `4000+error_code` -- for a `code > 999` that differs from the `4002` actually sent on the wire (illegal WS close codes are clamped; see `errorCode`, which always keeps the real, unclamped value). When instead observed directly from a bare close frame with no preceding `error{}`, `wsCode` is exactly what was on the wire. |
|
|
283
|
+
| `errorCode` | The ws-mixer error code (WIRE.md section 2.8), present under exactly three conditions (D-2026-09-20-09): a ws-mixer `error{}` preceded the close; the close carries a **bare** ws-mixer close code in `4001`-`4999` (derived mechanically as `wsCode - 4000`); or the SDK itself raised a ws-mixer error locally, with no close frame involved yet (a missing/mismatched subprotocol echo -> `UNSUPPORTED`, the welcome timeout -> `PROTOCOL_ERROR`, and the SDK's own other protocol-violation failures). **Never** synthesised for anything else: an HTTP upgrade rejection (401/403/404/429) is `httpStatus` alone, and an abnormal closure (`1006`) or other non-ws-mixer close code (`1000`, `1001`, `1009`, `1011`, ...) carries neither. Also absent for a token-provider throw/reject: that's an application error, not a wire error. |
|
|
284
|
+
| `errorName` | That error code's wire name (e.g. `"KEEPALIVE_TIMEOUT"`). |
|
|
285
|
+
| `httpStatus` | The HTTP status of the upgrade response, when the dial failed at the HTTP layer (401/403/404/429). |
|
|
286
|
+
| `fatal` | Whether the SDK will never reconnect after this (includes `reconnect.maxAttempts` exhaustion). |
|
|
287
|
+
| `message` | Human-readable description; never empty -- a close observed with no reason at all (from the peer, or `ws` itself) still falls back to a description like `"socket closed with code 4014"`. |
|
|
288
|
+
| `closeReason` | The reason field of the close frame **received from the peer**, verbatim -- never this side's own outgoing reason. Absent or empty whenever no reason was received from the peer: an abnormal closure (no close frame at all), this side having initiated the close itself (a peer's echo carries no information and RFC 6455 doesn't require it to copy the reason), or the SDK closing on a peer's `error{}` without reading whatever close frame follows it (WIRE.md section 2.7 allows "logs, surfaces and closes"). The human-readable text is in `message` for all of those cases instead -- consumers SHOULD prefer `closeReason` and fall back to `message`. |
|
|
289
|
+
| `cause` | The token provider's thrown/rejected error, when that's why the dial failed (including a `TokenUnavailableError`-marked, non-fatal one). |
|
|
290
|
+
|
|
291
|
+
`sendApp()` returns a `Promise<void>` that resolves once the frame is actually written to the
|
|
292
|
+
socket (`docs/DESIGN.md`), or rejects with the connection's terminal error if the connection
|
|
293
|
+
fails before it gets there -- `conn.ts`'s control queue carries a resolver per queued frame the same
|
|
294
|
+
way `sendData()`'s per-stream outbox already does for stream bytes.
|
|
295
|
+
|
|
296
|
+
### Application-initiated close
|
|
297
|
+
|
|
298
|
+
`close()` normally performs the default graceful shutdown shown above (`drain{client_requested}`,
|
|
299
|
+
a grace period, then close `1000`). Pass `{ message }` instead to close the connection with
|
|
300
|
+
`APPLICATION_CLOSE` (`0x0e`/WS close `4014` -- the only code an application may close a *connection*
|
|
301
|
+
with, WIRE.md section 2.8; there is no caller-chosen code, D-2026-09-25-01): `error{code: 14,
|
|
302
|
+
message}` on stream 0, then WS close `4014` with `message` truncated to 123 UTF-8 bytes on a
|
|
303
|
+
character boundary, then the socket -- no `drain`, no grace period, and no reconnect is scheduled
|
|
304
|
+
(matches the default `close()`'s one-report-then-done shape).
|
|
305
|
+
|
|
306
|
+
```ts
|
|
307
|
+
await client.close({ message: "operator requested shutdown" });
|
|
308
|
+
```
|
|
309
|
+
|
|
310
|
+
### Errors
|
|
311
|
+
|
|
312
|
+
`WsMixerError` (and its `ConnError`/`StreamError` subclasses) is the single error type this SDK
|
|
313
|
+
throws or emits (`'error'`/`'fatal'`, and every `Promise` rejection). Its `wsCode`/`closeReason`
|
|
314
|
+
fields exist for the same reason as the identically-named `DisconnectReason` fields above -- set
|
|
315
|
+
only when this particular error was built from an actually-observed close frame (as opposed to a
|
|
316
|
+
locally-raised protocol violation), and under the same "never this side's own outgoing reason" rule
|
|
317
|
+
as `closeReason`.
|
|
318
|
+
|
|
319
|
+
The `'fatal'` event's `WsMixerError.code` follows the same "present only when a ws-mixer error code
|
|
320
|
+
actually applies" rule as `DisconnectReason.errorCode` above (D-2026-09-20-09) -- it is `INTERNAL_ERROR`
|
|
321
|
+
whenever no ws-mixer code exists for the underlying failure (an HTTP `401`/`403` upgrade rejection,
|
|
322
|
+
for instance, used to surface `UNAUTHORIZED` here and no longer does), and otherwise the same derived
|
|
323
|
+
code the disconnect reason carries (e.g. a bare `4011` close's `code` is `UNAUTHORIZED`, matching
|
|
324
|
+
`errorCode`, not `INTERNAL_ERROR`). Consumers that need to distinguish *why* a fatal happened should
|
|
325
|
+
branch on the disconnect reason's `httpStatus`/`wsCode`/`errorCode`, not on the fatal error's `code`
|
|
326
|
+
alone.
|
|
327
|
+
|
|
328
|
+
## Tests
|
|
329
|
+
|
|
330
|
+
`npm test` (vitest) runs:
|
|
331
|
+
|
|
332
|
+
- `test/frame.test.ts`, `test/control.test.ts` -- every fixture under `{frames,control}` in a
|
|
333
|
+
`ws-mixer-spec` checkout (see "Spec fixtures" below) is decoded/validated by this SDK's own codec
|
|
334
|
+
and hand-written validator and checked against the fixture's expected outcome (`valid`/`wire_valid`),
|
|
335
|
+
per `ws-mixer-spec`'s `spec/README.md`'s "SDK runtime validators assert `wire_valid`" rule. Skips
|
|
336
|
+
with a reason if no spec checkout can be found.
|
|
337
|
+
- `test/sequence.test.ts` -- every `sequences/*.json` transcript with `role: "client"` is replayed
|
|
338
|
+
against a real `MixerConn` over a deterministic fake transport (`test/helpers/fake-ws.ts`); the
|
|
339
|
+
`role: "server"` transcripts are scripted from the Go server's perspective and are `it.skip`'d with
|
|
340
|
+
that reason, since this SDK is client-only. Same spec-checkout resolution and skip behavior as above.
|
|
341
|
+
- `test/stream.test.ts`, `test/conn.test.ts`, `test/keepalive.test.ts`, `test/reconnect.test.ts` --
|
|
342
|
+
unit coverage for the stream state machine, credit/WINDOW accounting, half-close, the
|
|
343
|
+
control-priority + round-robin write scheduler (exact interleaving asserted against the fake
|
|
344
|
+
transport), keepalive/dead-peer detection, and the reconnect backoff math.
|
|
345
|
+
- `test/interop.test.ts` -- `go build`s `cmd/testserver` from a `ws-mixer-go` checkout and drives it
|
|
346
|
+
as a plain child process (not `go run .`, so killing the child actually kills the server, not a
|
|
347
|
+
`go run` wrapper around it), connects this SDK as a real client over a real socket, and exercises a
|
|
348
|
+
multi-stream request/response round trip, an `app` round trip, and an immediate reconnect after
|
|
349
|
+
`Drain`.
|
|
350
|
+
|
|
351
|
+
**Spec fixtures and the Go interop server now live in their own repos** (`ws-mixer-spec`,
|
|
352
|
+
`ws-mixer-go`), not under this one. Both are resolved the same way: an env var override, else a
|
|
353
|
+
fetched checkout at the pinned tag, else a sibling directory next to this repo.
|
|
354
|
+
|
|
355
|
+
| | env var | fetched checkout | sibling fallback | pin file |
|
|
356
|
+
|---|---|---|---|---|
|
|
357
|
+
| spec fixtures | `WSMIXER_SPEC_DIR` | `.spec/spec` (`npm run fetch-spec`) | `../ws-mixer-spec/spec` | `spec.pin` |
|
|
358
|
+
| Go interop server | `WSMIXER_GO_DIR` | `.goserver` (`npm run fetch-goserver`) | `../ws-mixer-go` | `goserver.pin` |
|
|
359
|
+
|
|
360
|
+
`WSMIXER_SPEC_DIR` accepts either directory shape: this repo's own convention (the spec subdir
|
|
361
|
+
itself, i.e. `$WSMIXER_SPEC_DIR/fixtures` exists directly) or `ws-mixer-go`'s convention
|
|
362
|
+
(repo-root, i.e. `$WSMIXER_SPEC_DIR/spec/fixtures`) -- `test/helpers/spec-dir.ts` tries the former
|
|
363
|
+
first, then falls back to `$WSMIXER_SPEC_DIR/spec`.
|
|
364
|
+
|
|
365
|
+
Every one of `test/frame.test.ts`, `test/control.test.ts`, `test/sequence.test.ts` and
|
|
366
|
+
`test/interop.test.ts` skips its suite with the resolution failure as the reason -- never fails or
|
|
367
|
+
silently passes -- when none of the three resolve.
|
|
368
|
+
|
|
369
|
+
Toolchain resolution for the Go build: `$GO` (a path to a `go` binary) is preferred if set, else
|
|
370
|
+
whatever `go` is on `PATH`. Either way, the build runs with `GOTOOLCHAIN=auto`, so a resolved `go`
|
|
371
|
+
older than `cmd/testserver`'s `go` directive transparently downloads a matching toolchain instead of
|
|
372
|
+
failing -- `$GO` does not need to point at an exact-version match.
|
|
373
|
+
|
|
374
|
+
Run just this one (needs a working `go` on `PATH`, or set `GO=/path/to/go`, and a resolvable
|
|
375
|
+
`ws-mixer-go` checkout):
|
|
376
|
+
|
|
377
|
+
```bash
|
|
378
|
+
npx vitest run test/interop.test.ts
|
|
379
|
+
```
|
|
380
|
+
|
|
381
|
+
## Counters
|
|
382
|
+
|
|
383
|
+
`client.stats()` (and `conn.stats()`) return "ignore and count" counters -- unknown frame types,
|
|
384
|
+
stale frames, duplicate pongs, refused opens -- for the sites that already tolerate and discard
|
|
385
|
+
those per the wire spec. Unlike the Go server, this SDK does not escalate a sustained run of any of
|
|
386
|
+
them (e.g. repeated `STREAM_LIMIT` refusals) into a connection error; that's left for later.
|
|
387
|
+
|
|
388
|
+
`bytesIn`/`bytesOut` count raw WS message bytes, header included, for every message sent or
|
|
389
|
+
received -- not just DATA payload bytes. This is not the same measurement as Go's
|
|
390
|
+
`BytesTransferred` metric, which counts payload only; don't compare the two directly.
|
|
391
|
+
|
|
392
|
+
## Build
|
|
393
|
+
|
|
394
|
+
```bash
|
|
395
|
+
npm run build # tsup (dist/{index.js,index.cjs}) + build:types (dist/*.d.ts)
|
|
396
|
+
npm run build:types # plain `tsc --emitDeclarationOnly`, stripping @internal members
|
|
397
|
+
npm run typecheck # tsc --noEmit, strict
|
|
398
|
+
npm test # vitest run
|
|
399
|
+
```
|
|
400
|
+
|
|
401
|
+
Declarations are generated by a plain `tsc --emitDeclarationOnly` pass (per-module `dist/*.d.ts`,
|
|
402
|
+
not a single bundled file), not by tsup's own `dts` option: tsup 8.x's bundled-declaration step does
|
|
403
|
+
not honour `stripInternal`, so it was leaking every `@internal`-tagged member straight into the
|
|
404
|
+
public API surface. `tsconfig.json`'s `stripInternal: true` is the setting that actually matters;
|
|
405
|
+
`build:types` just repeats its flags on the CLI since `tsc -p` cannot mix a project file with
|
|
406
|
+
explicit entry points, and `src/index.ts` (not `test/**`) is the only entry point declarations are
|
|
407
|
+
needed for.
|
|
408
|
+
|
|
409
|
+
## Before this is published
|
|
410
|
+
|
|
411
|
+
While `@mcpwarp/ws-mixer` is unpublished (publishing is guarded by `scripts/check-registry.mjs`'s
|
|
412
|
+
`prepublishOnly` check, which refuses to publish anywhere but the public npm registry, not a
|
|
413
|
+
`"private"` field), a consumer (e.g. the mcpwarp
|
|
414
|
+
tunnel client) links against a built copy directly:
|
|
415
|
+
|
|
416
|
+
```bash
|
|
417
|
+
cd ws-mixer-js && npm install && npm run build # produces dist/{index.js,index.cjs,index.d.ts,...}
|
|
418
|
+
```
|
|
419
|
+
|
|
420
|
+
```json
|
|
421
|
+
{
|
|
422
|
+
"dependencies": {
|
|
423
|
+
"@mcpwarp/ws-mixer": "file:../ws-mixer-js"
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
```
|
|
427
|
+
|
|
428
|
+
then `npm install` in the consumer. `file:` dependencies are copied (not symlinked) by npm on
|
|
429
|
+
install, so **re-run `npm run build` here and `npm install` there** after every change -- there is
|
|
430
|
+
no live-reload across the `file:` link. Once this package is ready to publish for real, pick a
|
|
431
|
+
version and `npm publish` as usual; nothing else in this setup changes.
|