@bevel-software/platform-mcp-core 0.14.0 → 0.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,378 @@
1
+ import type { CodeModeUtcpClient } from '@utcp/code-mode';
2
+ import type { CallTemplate } from '@utcp/sdk';
3
+ import { registerManual } from './dispatch.js';
4
+
5
+ /**
6
+ * Client-side recovery from MCP session loss: when a remote MCP server has
7
+ * forgotten the session our manual holds — almost always because it restarted —
8
+ * re-register that manual once and retry the call once.
9
+ *
10
+ * WHY IT LIVES ON THE CLIENT OBJECT. Two paths reach a tool call in our stack:
11
+ * `dispatchToolCall` (the MCP surface of the hosted proxy and of the local
12
+ * server) goes through `callToolStreaming`, and a code-mode chain
13
+ * (`call_tool_chain`, gate probes) goes through `callTool` — `callToolChain`
14
+ * bridges every in-isolate tool function to `this.callTool`. Wrapping those two
15
+ * methods ON THE CLIENT INSTANCE is the one seam both paths cross, so the
16
+ * policy exists exactly once instead of being copied into each caller.
17
+ *
18
+ * WHY ONLY SESSION LOSS. A retry is only safe when the first attempt provably
19
+ * did nothing. Session loss is decided in the server's ROUTING layer, before
20
+ * the request is dispatched to a tool, so a mutating tool cannot have run.
21
+ * Every other failure — a tool error, an auth refusal, a timeout, a reset
22
+ * connection — could have executed the tool, and is surfaced unchanged. That is
23
+ * the whole safety argument: it rests on the trigger class, so
24
+ * {@link isSessionLoss} is deliberately narrow and separately testable.
25
+ */
26
+
27
+ /** The JSON-RPC code the Streamable HTTP transport reserves for a missing session. */
28
+ const SESSION_NOT_FOUND_CODE = -32001;
29
+
30
+ /** How far up a `cause` chain to look before giving up. */
31
+ const MAX_CAUSE_DEPTH = 8;
32
+
33
+ /** Each error in `err`'s `cause` chain, nearest first, bounded and cycle-safe. */
34
+ function* errorChain(err: unknown): Generator<Record<string, unknown>> {
35
+ const seen = new Set<unknown>();
36
+ let current = err;
37
+ for (let depth = 0; depth < MAX_CAUSE_DEPTH; depth += 1) {
38
+ if (current === null || typeof current !== 'object' || seen.has(current)) return;
39
+ seen.add(current);
40
+ yield current as Record<string, unknown>;
41
+ current = (current as { cause?: unknown }).cause;
42
+ }
43
+ }
44
+
45
+ /**
46
+ * The HTTP status an error carries, or undefined when it carries none.
47
+ *
48
+ * The MCP SDK's `StreamableHTTPError` puts the HTTP status in `code`, which is
49
+ * also where `McpError` puts its JSON-RPC code — and those two namespaces
50
+ * OVERLAP on the very number this module cares about: `-32001` is "Session not
51
+ * found" on the wire but `ErrorCode.RequestTimeout` in the SDK's own enum. The
52
+ * range test is what keeps them apart: a JSON-RPC code is negative and can
53
+ * never be read as a status, so a local request timeout never looks like a
54
+ * session miss (and is never retried).
55
+ */
56
+ function httpStatusOf(err: Record<string, unknown>): number | undefined {
57
+ for (const key of ['code', 'status', 'statusCode'] as const) {
58
+ const value = err[key];
59
+ if (typeof value === 'number' && value >= 100 && value <= 599) return value;
60
+ }
61
+ return undefined;
62
+ }
63
+
64
+ /**
65
+ * Does this error's text carry the session-not-found signal?
66
+ *
67
+ * The transport surfaces a failed POST as `Error POSTing to endpoint: <body>`,
68
+ * so the server's JSON-RPC body rides along in the message and is the only
69
+ * place the `-32001` is visible. Both halves of the spec's signal are accepted
70
+ * — the code (structural) and the reserved message — because a server may send
71
+ * either; requiring the 404 alongside is what keeps this from over-matching.
72
+ */
73
+ function saysSessionNotFound(message: string): boolean {
74
+ return (
75
+ new RegExp(`"code"\\s*:\\s*${SESSION_NOT_FOUND_CODE}\\b`).test(message) ||
76
+ /\bsession not found\b/i.test(message)
77
+ );
78
+ }
79
+
80
+ /**
81
+ * Is `err` unambiguously "the server has forgotten this session"?
82
+ *
83
+ * The signal is HTTP 404 carrying JSON-RPC `-32001` / "Session not found", which
84
+ * the MCP spec reserves for exactly one meaning: the request presented an
85
+ * `Mcp-Session-Id` the server no longer holds, and the client should
86
+ * re-initialize. The presented-a-session-id half is not observable from here,
87
+ * and does not need to be: a request with NO session id is answered 400 /
88
+ * `-32000` (a client mistake with nothing to recover), so a 404 in this shape
89
+ * implies a session id was sent.
90
+ *
91
+ * Everything else is false — including a 404 whose body is an ordinary
92
+ * not-found, an auth refusal, a timeout (see {@link httpStatusOf}), and a
93
+ * refused connection. Pure and exported so this boundary can be pinned by test
94
+ * rather than inferred from the recovery path around it.
95
+ */
96
+ export function isSessionLoss(err: unknown): boolean {
97
+ for (const candidate of errorChain(err)) {
98
+ if (httpStatusOf(candidate) !== 404) continue;
99
+ const message = candidate.message;
100
+ if (typeof message === 'string' && saysSessionNotFound(message)) return true;
101
+ }
102
+ return false;
103
+ }
104
+
105
+ export interface SessionRecoveryOptions {
106
+ /**
107
+ * The template to re-register `manualName` with, or undefined when this
108
+ * surface holds none — an unknown manual is not recovered, and its failure
109
+ * surfaces unchanged.
110
+ *
111
+ * Resolved at RECOVERY time, not at install time, so a manual whose template
112
+ * carries a credential that can be rotated (the local server renews its
113
+ * connection key) re-registers with the current one. May be async, which is
114
+ * also the hook a surface uses to wait out a re-registration of its own that
115
+ * is already in flight.
116
+ */
117
+ manualTemplate: (manualName: string) => CallTemplate | undefined | Promise<CallTemplate | undefined>;
118
+ /**
119
+ * Ran after a successful re-registration. Re-registration REDISCOVERS the
120
+ * manual, so a surface that prunes something from the registry at first
121
+ * registration (the local server drops the deployment's copies of the
122
+ * code-mode meta-tools) has to prune it again here.
123
+ */
124
+ afterReregister?: (manualName: string) => Promise<void> | void;
125
+ /**
126
+ * Runs one whole re-registration — resolve the template, deregister,
127
+ * register, clean up — for a surface that has a re-registration path of its
128
+ * own. `hexis-mcp` renews its connection key by re-registering the remote
129
+ * manual and holds arriving calls while it does; handing that same gate in
130
+ * here makes the two ONE serialized operation instead of two that can
131
+ * interleave (the window between a deregister and its register finds no
132
+ * manual in the repository) or, worse, wait on each other. Whatever `run`
133
+ * settles to is what recovery uses; a hook that throws counts as a failed
134
+ * recovery, never as a failed call. Defaults to running `run` directly.
135
+ */
136
+ withReregister?: <T>(manualName: string, run: () => Promise<T>) => Promise<T>;
137
+ /** Where the one-line-per-recovery log goes. Defaults to stderr. */
138
+ log?: (message: string) => void;
139
+ }
140
+
141
+ /** What one re-registration produced: may we retry, and anything odd worth saying. */
142
+ interface ReregisterOutcome {
143
+ /** True only when the manual now holds a fresh session. */
144
+ ok: boolean;
145
+ /**
146
+ * True when the fresh session was somebody else's work — this call waited
147
+ * for a re-registration already under way and inherited its result rather
148
+ * than dialing a third session.
149
+ */
150
+ reused?: boolean;
151
+ /**
152
+ * Abnormalities met on the way. Folded into the ONE line the recovered call
153
+ * logs rather than printed as they happen: a recovery is a single event, and
154
+ * two lines for one read as two recoveries — which is how a retry loop that
155
+ * never existed gets diagnosed.
156
+ */
157
+ notes: string[];
158
+ }
159
+
160
+ const messageOf = (err: unknown): string => (err instanceof Error ? err.message : String(err));
161
+
162
+ /**
163
+ * Per-client re-registration counters, keyed by manual — the same maps the
164
+ * installed recovery reads, reachable from {@link noteManualReregistered} so a
165
+ * surface can report a re-registration of its own.
166
+ */
167
+ const GENERATIONS = new WeakMap<object, Map<string, number>>();
168
+
169
+ /**
170
+ * Tell recovery that `manualName`'s session was replaced by something OTHER
171
+ * than recovery. `hexis-mcp` re-registers the remote manual whenever a renewed
172
+ * connection key arrives; a call that lost its session around such a swap must
173
+ * then retry against THAT session instead of deregistering it and dialing a
174
+ * third one. Without this the two paths each replace a session the other just
175
+ * made — correct, but a round trip and a discovery pass wasted on every
176
+ * overlap. No-op when recovery is not installed on `client`.
177
+ */
178
+ export function noteManualReregistered(client: CodeModeUtcpClient, manualName: string): void {
179
+ const generations = GENERATIONS.get(client as unknown as object);
180
+ if (!generations) return;
181
+ generations.set(manualName, (generations.get(manualName) ?? 0) + 1);
182
+ }
183
+
184
+ /**
185
+ * Wrap `client`'s tool-call entry points with session recovery, in place.
186
+ *
187
+ * Returns the same client so it can be used as an expression at the
188
+ * construction site. Installing twice is a no-op: the second call would stack a
189
+ * second retry on the first, which is precisely the "exactly once" guarantee
190
+ * this module exists to make.
191
+ */
192
+ const INSTALLED = Symbol.for('@bevel-software/platform-mcp-core.sessionRecovery');
193
+
194
+ export function installSessionRecovery(
195
+ client: CodeModeUtcpClient,
196
+ options: SessionRecoveryOptions,
197
+ ): CodeModeUtcpClient {
198
+ const marked = client as unknown as Record<symbol, unknown>;
199
+ if (marked[INSTALLED]) return client;
200
+ marked[INSTALLED] = true;
201
+
202
+ const log = options.log ?? ((message: string) => console.error(message));
203
+ const callTool = client.callTool.bind(client);
204
+ const callToolStreaming = client.callToolStreaming.bind(client);
205
+
206
+ /**
207
+ * How many times each manual has been re-registered. A call reads this
208
+ * BEFORE its attempt; if the number moved while the attempt was in flight,
209
+ * some other call already replaced the session this one was using and the
210
+ * retry can go straight through — no second re-registration, and no window
211
+ * in which a burst of failures that arrive slightly apart each starts its
212
+ * own. Together with `inflight` below, this is what makes the coalescing
213
+ * hold for concurrent calls whether they fail together or in sequence.
214
+ */
215
+ const generations = new Map<string, number>();
216
+ GENERATIONS.set(client as unknown as object, generations);
217
+ const inflight = new Map<string, Promise<ReregisterOutcome>>();
218
+
219
+ const generationOf = (manualName: string): number => generations.get(manualName) ?? 0;
220
+
221
+ /**
222
+ * Deregister + register once. Reports what happened; the caller does the
223
+ * logging. `generation` is what the failing call saw before its attempt: if
224
+ * the counter has moved, the session has ALREADY been replaced — by the
225
+ * surface's own credential swap, or by another call — and replacing it again
226
+ * would throw away a live session to dial an identical one.
227
+ *
228
+ * Tested TWICE, because this function has two places it can wait and either
229
+ * is long enough for a swap to land: the surface's gate (before the call
230
+ * arrives here) and `manualTemplate`, which a surface may deliberately park
231
+ * in. The check that matters is the one immediately before the deregister —
232
+ * the first is only there to skip work nobody needs.
233
+ */
234
+ async function reregisterNow(manualName: string, generation: number): Promise<ReregisterOutcome> {
235
+ const notes: string[] = [];
236
+ const reused = { ok: true, reused: true, notes } as const;
237
+ if (generationOf(manualName) !== generation) return reused;
238
+ const template = await options.manualTemplate(manualName);
239
+ // Not ours to re-register — or not an MCP manual at all, in which case it
240
+ // holds no session and the 404 came from somewhere we must not second-guess.
241
+ // Silent on purpose: nothing happened, so there is nothing to report.
242
+ if (!template || template.call_template_type !== 'mcp') return { ok: false, notes };
243
+ // Resolving the template is itself a place a surface waits.
244
+ if (generationOf(manualName) !== generation) return reused;
245
+ try {
246
+ // Deregistering is what closes the manual's (now dead) session, so the
247
+ // registration below dials a fresh one instead of reusing the cached
248
+ // transport. Best effort: a manual already gone is not a failure here.
249
+ await client.deregisterManual(manualName);
250
+ } catch (err) {
251
+ notes.push(`deregistering first failed: ${messageOf(err)}`);
252
+ }
253
+ const result = await registerManual(client, template);
254
+ if (!result.ok) {
255
+ // The server is very likely still coming back up. The call fails as it
256
+ // would have anyway; the NEXT one recovers once the server answers.
257
+ notes.push(`re-registration failed: ${result.error}`);
258
+ return { ok: false, notes };
259
+ }
260
+ generations.set(manualName, generationOf(manualName) + 1);
261
+ if (options.afterReregister) {
262
+ try {
263
+ await options.afterReregister(manualName);
264
+ } catch (err) {
265
+ notes.push(`post-re-registration cleanup failed: ${messageOf(err)}`);
266
+ }
267
+ }
268
+ return { ok: true, notes };
269
+ }
270
+
271
+ /**
272
+ * {@link reregisterNow} under the surface's own gate, if it has one, and with
273
+ * every escape route closed. NEVER throws — and that is load-bearing: a
274
+ * throwing template resolver (or gate) must not replace the caller's real
275
+ * session-loss error with a recovery-internal one. The answer here is only
276
+ * ever "may we retry?".
277
+ */
278
+ async function reregister(manualName: string, generation: number): Promise<ReregisterOutcome> {
279
+ try {
280
+ return options.withReregister
281
+ ? await options.withReregister(manualName, () => reregisterNow(manualName, generation))
282
+ : await reregisterNow(manualName, generation);
283
+ } catch (err) {
284
+ return { ok: false, notes: [`re-registration threw: ${messageOf(err)}`] };
285
+ }
286
+ }
287
+
288
+ /**
289
+ * Single-flight {@link reregister}: concurrent losers share one attempt.
290
+ *
291
+ * The generation belongs to whoever started the attempt, and that is right
292
+ * for the joiners too — {@link recover} sends nobody here whose generation
293
+ * differs from the current one, so every sharer of this promise failed
294
+ * against the same session.
295
+ */
296
+ function reregisterOnce(manualName: string, generation: number): Promise<ReregisterOutcome> {
297
+ const existing = inflight.get(manualName);
298
+ if (existing) return existing;
299
+ const tracked = reregister(manualName, generation).finally(() => {
300
+ if (inflight.get(manualName) === tracked) inflight.delete(manualName);
301
+ });
302
+ inflight.set(manualName, tracked);
303
+ return tracked;
304
+ }
305
+
306
+ /**
307
+ * May this failed call be retried? True only for session loss on a manual we
308
+ * hold a template for, and only after the session has actually been replaced.
309
+ */
310
+ async function recover(toolName: string, generation: number, err: unknown): Promise<boolean> {
311
+ if (!isSessionLoss(err)) return false;
312
+ const manualName = toolName.split('.')[0];
313
+ if (!manualName) return false;
314
+ /** The session was replaced by someone else — a concurrent call, or the surface. */
315
+ const alreadyReplaced = (): boolean => {
316
+ log(`[mcp] session lost on '${manualName}' — re-registered by a concurrent call; retrying.`);
317
+ return true;
318
+ };
319
+ // Someone else re-registered while this call was in flight: the session it
320
+ // failed against is already gone, so retry against the new one directly.
321
+ if (generationOf(manualName) !== generation) return alreadyReplaced();
322
+ const outcome = await reregisterOnce(manualName, generation);
323
+ // The same finding, made too late to skip the queue: the re-registration
324
+ // landed while this call waited for the surface's gate.
325
+ if (outcome.reused) return alreadyReplaced();
326
+ // One recovered call, one line — whatever went sideways on the way rides
327
+ // along in it rather than arriving as a line of its own.
328
+ const notes = outcome.notes.length > 0 ? ` (${outcome.notes.join('; ')})` : '';
329
+ if (!outcome.ok) {
330
+ if (notes) {
331
+ log(
332
+ `[mcp] session lost on '${manualName}' — not recovered${notes}; the original failure stands.`,
333
+ );
334
+ }
335
+ return false;
336
+ }
337
+ log(`[mcp] session lost on '${manualName}' — re-registered and retried.${notes}`);
338
+ return true;
339
+ }
340
+
341
+ client.callTool = async function recoveringCallTool(
342
+ toolName: string,
343
+ toolArgs: Record<string, unknown>,
344
+ ): Promise<unknown> {
345
+ const generation = generationOf(toolName.split('.')[0] ?? '');
346
+ try {
347
+ return await callTool(toolName, toolArgs);
348
+ } catch (err) {
349
+ if (!(await recover(toolName, generation, err))) throw err;
350
+ // Exactly one retry: whatever this produces is what the caller sees,
351
+ // including a second session loss.
352
+ return await callTool(toolName, toolArgs);
353
+ }
354
+ };
355
+
356
+ client.callToolStreaming = async function* recoveringCallToolStreaming(
357
+ toolName: string,
358
+ toolArgs: Record<string, unknown>,
359
+ ): AsyncGenerator<unknown, void, unknown> {
360
+ const generation = generationOf(toolName.split('.')[0] ?? '');
361
+ let yielded = false;
362
+ try {
363
+ for await (const chunk of callToolStreaming(toolName, toolArgs)) {
364
+ yielded = true;
365
+ yield chunk;
366
+ }
367
+ return;
368
+ } catch (err) {
369
+ // A stream that already produced output is past the point where session
370
+ // loss can happen (the session is validated before the first byte), and
371
+ // replaying it would duplicate the chunks the caller already saw.
372
+ if (yielded || !(await recover(toolName, generation, err))) throw err;
373
+ }
374
+ yield* callToolStreaming(toolName, toolArgs);
375
+ };
376
+
377
+ return client;
378
+ }