@proof-holdings/mcp-server 1.0.0 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +1 -1
- package/README.md +220 -217
- package/dist/authBoundary.d.ts +193 -0
- package/dist/authBoundary.d.ts.map +1 -0
- package/dist/authBoundary.js +341 -0
- package/dist/authBoundary.js.map +1 -0
- package/dist/factory.d.ts +35 -0
- package/dist/factory.d.ts.map +1 -0
- package/dist/factory.js +130 -0
- package/dist/factory.js.map +1 -0
- package/dist/http.d.ts +78 -3
- package/dist/http.d.ts.map +1 -1
- package/dist/http.js +251 -6
- package/dist/http.js.map +1 -1
- package/dist/remote.d.ts +111 -0
- package/dist/remote.d.ts.map +1 -0
- package/dist/remote.js +789 -0
- package/dist/remote.js.map +1 -0
- package/dist/server.js +11 -57
- package/dist/server.js.map +1 -1
- package/dist/toolAnnotations.d.ts +27 -0
- package/dist/toolAnnotations.d.ts.map +1 -0
- package/dist/toolAnnotations.js +158 -0
- package/dist/toolAnnotations.js.map +1 -0
- package/dist/tools/{projects.d.ts → accounts.d.ts} +1 -1
- package/dist/tools/accounts.d.ts.map +1 -0
- package/dist/tools/accounts.js +70 -0
- package/dist/tools/accounts.js.map +1 -0
- package/dist/tools/api-keys.d.ts.map +1 -1
- package/dist/tools/api-keys.js +14 -5
- package/dist/tools/api-keys.js.map +1 -1
- package/dist/tools/auth-flows.d.ts +4 -0
- package/dist/tools/auth-flows.d.ts.map +1 -0
- package/dist/tools/auth-flows.js +34 -0
- package/dist/tools/auth-flows.js.map +1 -0
- package/dist/tools/auth.js +1 -1
- package/dist/tools/auth.js.map +1 -1
- package/dist/tools/authorizations.d.ts +4 -0
- package/dist/tools/authorizations.d.ts.map +1 -0
- package/dist/tools/authorizations.js +111 -0
- package/dist/tools/authorizations.js.map +1 -0
- package/dist/tools/circles.d.ts +4 -0
- package/dist/tools/circles.d.ts.map +1 -0
- package/dist/tools/circles.js +215 -0
- package/dist/tools/circles.js.map +1 -0
- package/dist/tools/confirmations.d.ts +4 -0
- package/dist/tools/confirmations.d.ts.map +1 -0
- package/dist/tools/confirmations.js +86 -0
- package/dist/tools/confirmations.js.map +1 -0
- package/dist/tools/delegation-verify-outcomes.d.ts +23 -0
- package/dist/tools/delegation-verify-outcomes.d.ts.map +1 -0
- package/dist/tools/delegation-verify-outcomes.js +51 -0
- package/dist/tools/delegation-verify-outcomes.js.map +1 -0
- package/dist/tools/delegation-verify.d.ts +24 -0
- package/dist/tools/delegation-verify.d.ts.map +1 -0
- package/dist/tools/delegation-verify.js +192 -0
- package/dist/tools/delegation-verify.js.map +1 -0
- package/dist/tools/delegations.d.ts +4 -0
- package/dist/tools/delegations.d.ts.map +1 -0
- package/dist/tools/delegations.js +84 -0
- package/dist/tools/delegations.js.map +1 -0
- package/dist/tools/domains.d.ts.map +1 -1
- package/dist/tools/domains.js +1 -2
- package/dist/tools/domains.js.map +1 -1
- package/dist/tools/hitl-keys.d.ts +4 -0
- package/dist/tools/hitl-keys.d.ts.map +1 -0
- package/dist/tools/hitl-keys.js +52 -0
- package/dist/tools/hitl-keys.js.map +1 -0
- package/dist/tools/hitl.d.ts +4 -0
- package/dist/tools/hitl.d.ts.map +1 -0
- package/dist/tools/hitl.js +151 -0
- package/dist/tools/hitl.js.map +1 -0
- package/dist/tools/phones.js +1 -1
- package/dist/tools/phones.js.map +1 -1
- package/dist/tools/profiles.d.ts.map +1 -1
- package/dist/tools/profiles.js +73 -0
- package/dist/tools/profiles.js.map +1 -1
- package/dist/tools/proof-me.d.ts +4 -0
- package/dist/tools/proof-me.d.ts.map +1 -0
- package/dist/tools/proof-me.js +36 -0
- package/dist/tools/proof-me.js.map +1 -0
- package/dist/tools/proofs.d.ts.map +1 -1
- package/dist/tools/proofs.js +9 -6
- package/dist/tools/proofs.js.map +1 -1
- package/dist/tools/render-auth-link.d.ts +3 -0
- package/dist/tools/render-auth-link.d.ts.map +1 -0
- package/dist/tools/render-auth-link.js +30 -0
- package/dist/tools/render-auth-link.js.map +1 -0
- package/dist/tools/sessions.js +5 -5
- package/dist/tools/sessions.js.map +1 -1
- package/dist/tools/settings.d.ts.map +1 -1
- package/dist/tools/settings.js +77 -2
- package/dist/tools/settings.js.map +1 -1
- package/dist/tools/twofa.d.ts.map +1 -1
- package/dist/tools/twofa.js +16 -3
- package/dist/tools/twofa.js.map +1 -1
- package/dist/tools/user-requests.d.ts.map +1 -1
- package/dist/tools/user-requests.js +1 -2
- package/dist/tools/user-requests.js.map +1 -1
- package/dist/tools/verification-requests.d.ts.map +1 -1
- package/dist/tools/verification-requests.js +40 -13
- package/dist/tools/verification-requests.js.map +1 -1
- package/dist/tools/verifications.d.ts.map +1 -1
- package/dist/tools/verifications.js +59 -12
- package/dist/tools/verifications.js.map +1 -1
- package/dist/types.d.ts +18 -0
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +114 -5
- package/dist/types.js.map +1 -1
- package/package.json +11 -5
- package/dist/tools/projects.d.ts.map +0 -1
- package/dist/tools/projects.js +0 -159
- package/dist/tools/projects.js.map +0 -1
package/dist/remote.js
ADDED
|
@@ -0,0 +1,789 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* Remote entrypoint: the Proof MCP server over Streamable HTTP (`h-mcp-remote`, SC-3 + SC-4).
|
|
4
|
+
*
|
|
5
|
+
* Deployed as its own service, NOT as a route inside the existing backend. The reason is
|
|
6
|
+
* structural rather than stylistic: `src/cluster.ts` forks one worker per CPU and round-robins
|
|
7
|
+
* between them, while a Streamable HTTP session lives in the memory of ONE process. Mounting this
|
|
8
|
+
* on the clustered backend would lose sessions by default, not in an edge case. The deployment
|
|
9
|
+
* therefore pins requests to a replica by the REPLICA PREFIX this server puts in every session id
|
|
10
|
+
* it mints (`<replica>.<uuid>`); see `replicaId` below for why hashing the id itself cannot work,
|
|
11
|
+
* and `k8s/mcp-remote/nginx-mcp.conf` for the routing that consumes the prefix. That routing is
|
|
12
|
+
* deliberately NOT in the base nginx configmap — see that directory's README for the ingress it
|
|
13
|
+
* would otherwise take down, and `src/__tests__/drift/mcp-remote-deployment.test.ts`, which asserts
|
|
14
|
+
* the base stays free of it.
|
|
15
|
+
*
|
|
16
|
+
* ANONYMOUS FIRST, KEYED ON REQUEST (`h-mcp-oauth-remote-wiring`). Every connection starts with the
|
|
17
|
+
* keyless surface and the ability to log in through the normal flows; a client that has been
|
|
18
|
+
* through the OAuth ceremony (`h-mcp-oauth-server`) presents the API key it was granted, and this
|
|
19
|
+
* server reads it from the `Authorization` header ONLY. A `?api_key=` query parameter is never
|
|
20
|
+
* read: a token in a URL is copied into proxy logs and `Referer`, and the standard clients all send
|
|
21
|
+
* a header. The token is resolved ONCE, at open, by self-inspection (`GET /api/v1/me/api-key`) —
|
|
22
|
+
* purely so the refusal is legible; nothing is cached from it, and the key travels to the backend
|
|
23
|
+
* on every call, which is what makes a revocation take effect immediately rather than at session
|
|
24
|
+
* TTL. Where the 401 lands is a measurement, not a preference: on the TOOL CALL that needs an
|
|
25
|
+
* account, because a 401 on `initialize` reached the human as a connection timeout and cost the
|
|
26
|
+
* anonymous connection entirely (`docs/mcp-oauth-client-probe.md`).
|
|
27
|
+
*
|
|
28
|
+
* Each connection needs its OWN `HttpClient` for two reasons now: `captureSetCookie` writes a login
|
|
29
|
+
* session onto the instance, and the API key of one connection must never be reachable from
|
|
30
|
+
* another. The boundary between keyless and keyed tools lives in `./authBoundary.js`.
|
|
31
|
+
*/
|
|
32
|
+
import { createServer } from 'node:http';
|
|
33
|
+
import { createHash, randomUUID } from 'node:crypto';
|
|
34
|
+
import os from 'node:os';
|
|
35
|
+
import path from 'node:path';
|
|
36
|
+
import { fileURLToPath } from 'node:url';
|
|
37
|
+
import { StreamableHTTPServerTransport } from '@modelcontextprotocol/sdk/server/streamableHttp.js';
|
|
38
|
+
import { authorizationChallenge, authorizationRequiredBody, keyedToolsIn, malformedToolCallBody, sessionCredentialMismatchBody, } from './authBoundary.js';
|
|
39
|
+
import { createProofServer } from './factory.js';
|
|
40
|
+
import { ApiError, HttpClient } from './http.js';
|
|
41
|
+
/**
|
|
42
|
+
* Measured, not guessed (the task called this out as an empirical question). Opening sessions
|
|
43
|
+
* against a built `remote.js` and reading `process.memoryUsage()` after forced GC:
|
|
44
|
+
*
|
|
45
|
+
* 0 sessions heap 17.3MB rss 89.9MB
|
|
46
|
+
* 10 sessions heap 38.4MB rss 148.1MB ~2.11MB/session
|
|
47
|
+
* 50 sessions heap 110.9MB rss 301.6MB ~1.87MB/session
|
|
48
|
+
* 100 sessions heap 201.4MB rss 406.5MB ~1.84MB/session
|
|
49
|
+
* 200 sessions heap 382.1MB rss 578.4MB ~1.82MB/session
|
|
50
|
+
*
|
|
51
|
+
* Linear at ~1.84MB of heap per session — the full tool set is registered per connection, and that
|
|
52
|
+
* is what it costs. 100 sessions therefore sits near 400MB RSS, which is what the container's
|
|
53
|
+
* memory limit is sized against; `src/__tests__/drift/mcp-remote-deployment.test.ts` keeps the two
|
|
54
|
+
* numbers in agreement, because raising this cap without raising the limit buys an OOMKill.
|
|
55
|
+
*/
|
|
56
|
+
const DEFAULT_MAX_SESSIONS = 100;
|
|
57
|
+
const DEFAULT_SESSION_TTL_MS = 30 * 60_000;
|
|
58
|
+
const DEFAULT_SWEEP_INTERVAL_MS = 60_000;
|
|
59
|
+
/**
|
|
60
|
+
* Largest request body accepted. JSON-RPC envelopes are small; this exists so that one anonymous
|
|
61
|
+
* request cannot buffer the container out of memory (measured before the bound: a single 256MB body
|
|
62
|
+
* took the process to 859MB RSS against a 768Mi limit).
|
|
63
|
+
*/
|
|
64
|
+
const MAX_BODY_BYTES = 4 * 1024 * 1024;
|
|
65
|
+
const json = (res, status, body) => {
|
|
66
|
+
res.writeHead(status, { 'content-type': 'application/json' }).end(JSON.stringify(body));
|
|
67
|
+
};
|
|
68
|
+
/**
|
|
69
|
+
* The credential this request presents, read from the `Authorization` header and NOWHERE else.
|
|
70
|
+
*
|
|
71
|
+
* The absence of a query-parameter branch is the point, and it is guarded BEHAVIOURALLY:
|
|
72
|
+
* `__tests__/remote.test.ts`'s SC-8 block measures that `?api_key=` opens an ANONYMOUS session — the
|
|
73
|
+
* keyed tool is still refused — and that the value in the query never appears in an outbound
|
|
74
|
+
* request. Teaching this function to read it turns that block red (verified by mutation).
|
|
75
|
+
*
|
|
76
|
+
* A source scan for the string `api_key` was considered and DELIBERATELY NOT written: the only
|
|
77
|
+
* occurrences left in this file are the sentences explaining the rule, so such a scan would force a
|
|
78
|
+
* choice between a rule that may not be explained and a guard loosened until it catches nothing.
|
|
79
|
+
* The drift suite therefore pins reachability only (that this file still reads the header and still
|
|
80
|
+
* gets its refusals from `./authBoundary.js`), never an absence.
|
|
81
|
+
*/
|
|
82
|
+
const bearerOf = (req) => {
|
|
83
|
+
const header = req.headers.authorization ?? '';
|
|
84
|
+
if (!header.toLowerCase().startsWith('bearer '))
|
|
85
|
+
return undefined;
|
|
86
|
+
const value = header.slice(7).trim();
|
|
87
|
+
return value || undefined;
|
|
88
|
+
};
|
|
89
|
+
/**
|
|
90
|
+
* Identity of a credential WITHOUT keeping a second copy of it in the open.
|
|
91
|
+
*
|
|
92
|
+
* The key itself already lives inside the session's `HttpClient`; what the session map needs is
|
|
93
|
+
* only the ability to answer "is this the same caller". A plain comparison of two SHA-256 digests
|
|
94
|
+
* is enough for that — the digest is not a secret whose leak buys anything, since the key cannot be
|
|
95
|
+
* recovered from it.
|
|
96
|
+
*/
|
|
97
|
+
const fingerprintOf = (credential) => credential ? createHash('sha256').update(credential).digest('hex') : undefined;
|
|
98
|
+
/**
|
|
99
|
+
* 429 for a caller that must simply wait. `Retry-After` travels only when the number is OURS —
|
|
100
|
+
* i.e. when our own budget refused — because a header we cannot compute is worse absent than wrong.
|
|
101
|
+
*/
|
|
102
|
+
const rateLimited = (res, description, retryAfterSeconds) => {
|
|
103
|
+
res
|
|
104
|
+
.writeHead(429, {
|
|
105
|
+
'content-type': 'application/json',
|
|
106
|
+
...(retryAfterSeconds ? { 'retry-after': String(retryAfterSeconds) } : {}),
|
|
107
|
+
})
|
|
108
|
+
.end(JSON.stringify({ error: 'rate_limited', error_description: description }));
|
|
109
|
+
};
|
|
110
|
+
/**
|
|
111
|
+
* The body for a refused tool call. A call that named no tool is refused for a DIFFERENT reason
|
|
112
|
+
* than a call that needs an account, and saying so is the difference between an instruction the
|
|
113
|
+
* caller can act on and one that sends them to sign in over a typo.
|
|
114
|
+
*/
|
|
115
|
+
const refusalFor = (baseUrl, call) => call.malformed ? malformedToolCallBody() : authorizationRequiredBody(baseUrl, call.name);
|
|
116
|
+
/** 401 + the challenge that sends a client to the authorization server (RFC 9728). */
|
|
117
|
+
const challenge = (res, baseUrl, body) => {
|
|
118
|
+
res
|
|
119
|
+
.writeHead(401, {
|
|
120
|
+
'content-type': 'application/json',
|
|
121
|
+
'www-authenticate': authorizationChallenge(baseUrl),
|
|
122
|
+
})
|
|
123
|
+
.end(JSON.stringify(body));
|
|
124
|
+
};
|
|
125
|
+
/** Thrown when a request body exceeds `MAX_BODY_BYTES`; answered as 413 rather than buffered. */
|
|
126
|
+
class BodyTooLargeError extends Error {
|
|
127
|
+
}
|
|
128
|
+
/**
|
|
129
|
+
* Reads and parses a request body, bounded in size and settled on every outcome.
|
|
130
|
+
*
|
|
131
|
+
* All three properties are load-bearing on a public endpoint:
|
|
132
|
+
* - BOUNDED: an unbounded read let one anonymous request take the process to 859MB RSS against a
|
|
133
|
+
* 768Mi container limit (measured). MCP payloads are JSON-RPC envelopes; 4MB is generous.
|
|
134
|
+
* - `error`/`aborted` HANDLED: a client that disconnects mid-body never emits `end`, so a promise
|
|
135
|
+
* awaiting only `end` never settles — which silently leaks anything the caller accounts for
|
|
136
|
+
* after the await (the session reservation below is exactly that).
|
|
137
|
+
* - `resolve(undefined)` on unparseable JSON is deliberate: the transport answers such a request
|
|
138
|
+
* itself, and distinguishing "no body" from "bad body" is not this function's job.
|
|
139
|
+
*/
|
|
140
|
+
const readBody = (req) => new Promise((resolve, reject) => {
|
|
141
|
+
const chunks = [];
|
|
142
|
+
let size = 0;
|
|
143
|
+
const fail = (error) => {
|
|
144
|
+
req.removeAllListeners('data');
|
|
145
|
+
reject(error);
|
|
146
|
+
};
|
|
147
|
+
req.on('data', (chunk) => {
|
|
148
|
+
size += chunk.length;
|
|
149
|
+
if (size > MAX_BODY_BYTES) {
|
|
150
|
+
// Stop accumulating, but do NOT destroy the socket here: the caller still has to send 413,
|
|
151
|
+
// and tearing the connection down first makes the client see a write error instead of the
|
|
152
|
+
// answer (EPIPE, observed).
|
|
153
|
+
fail(new BodyTooLargeError(`request body exceeds ${MAX_BODY_BYTES} bytes`));
|
|
154
|
+
return;
|
|
155
|
+
}
|
|
156
|
+
chunks.push(chunk);
|
|
157
|
+
});
|
|
158
|
+
req.on('error', fail);
|
|
159
|
+
req.on('aborted', () => fail(new Error('request aborted')));
|
|
160
|
+
req.on('end', () => {
|
|
161
|
+
const raw = Buffer.concat(chunks).toString('utf8');
|
|
162
|
+
if (!raw)
|
|
163
|
+
return resolve(undefined);
|
|
164
|
+
try {
|
|
165
|
+
resolve(JSON.parse(raw));
|
|
166
|
+
}
|
|
167
|
+
catch {
|
|
168
|
+
resolve(undefined);
|
|
169
|
+
}
|
|
170
|
+
});
|
|
171
|
+
});
|
|
172
|
+
/**
|
|
173
|
+
* Answers a body-read failure and only THEN tears the connection down — the client still owes us
|
|
174
|
+
* the rest of an oversized upload, and closing first would replace the answer with a write error.
|
|
175
|
+
*/
|
|
176
|
+
const rejectBody = (req, res, error) => {
|
|
177
|
+
const tooLarge = error instanceof BodyTooLargeError;
|
|
178
|
+
// Drain FIRST. The client is still uploading, and any arrangement that closes the connection
|
|
179
|
+
// while it writes reaches it as a write error instead of the answer — measured twice: destroying
|
|
180
|
+
// the socket outright gave EPIPE every time, and answering with `connection: close` gave it
|
|
181
|
+
// intermittently (1 run in 4), which is a flaky test standing on a real race. Draining costs no
|
|
182
|
+
// memory (bytes are counted and dropped, never buffered) and lets the response arrive intact.
|
|
183
|
+
req.resume();
|
|
184
|
+
if (!res.headersSent) {
|
|
185
|
+
res.writeHead(tooLarge ? 413 : 400, { 'content-type': 'application/json' });
|
|
186
|
+
res.end(JSON.stringify({ error: tooLarge ? 'body_too_large' : 'bad_request' }));
|
|
187
|
+
}
|
|
188
|
+
else if (!res.writableEnded) {
|
|
189
|
+
// Headers already went out. Nothing can be said about the status any more, but the response
|
|
190
|
+
// must still be TERMINATED — an unfinished one hangs the socket until the client gives up, and
|
|
191
|
+
// through the proxy that is a 3600s hold on a connection and a worker slot.
|
|
192
|
+
//
|
|
193
|
+
// Unreachable with SDK 1.27.1, stated rather than implied: both callers reach `rejectBody` only
|
|
194
|
+
// out of `readBody`, which throws before any header is written, and a throw inside
|
|
195
|
+
// `transport.handleRequest` is caught by Hono's request listener, which terminates the response
|
|
196
|
+
// itself. The arm is kept because it costs nothing and is correct if that changes — not
|
|
197
|
+
// because a test covers it.
|
|
198
|
+
res.end();
|
|
199
|
+
}
|
|
200
|
+
};
|
|
201
|
+
/**
|
|
202
|
+
* How long the self-inspection may take, and how many times it may be tried.
|
|
203
|
+
*
|
|
204
|
+
* NOT the client's ordinary settings, and the difference is a capacity bound rather than a
|
|
205
|
+
* preference. This call runs INSIDE the reservation taken below, so its duration is time a session
|
|
206
|
+
* slot is held by a request that has not proved anything yet. The ordinary client retries twice and
|
|
207
|
+
* honours `Retry-After`, and the backend answers `Retry-After: 60` from ONE per-pod IP bucket
|
|
208
|
+
* shared by every tenant — so with those settings a single credential-less request holds a slot for
|
|
209
|
+
* two minutes, and a hundred of them close the anonymous surface this service exists to keep open
|
|
210
|
+
* (measured: 120s hold, 3 backend calls, a legitimate anonymous open answered 503).
|
|
211
|
+
*/
|
|
212
|
+
const RESOLVE_TIMEOUT_MS = 5_000;
|
|
213
|
+
/**
|
|
214
|
+
* How fast this process may send self-inspections, sustained.
|
|
215
|
+
*
|
|
216
|
+
* A LOCAL bound on OUR outbound calls, and it exists because the refusal path is cheap: a bearer
|
|
217
|
+
* that resolves to nothing releases its reservation immediately, so nothing else in this process
|
|
218
|
+
* limits how fast such opens arrive. The proxy allows 30 r/s per client IP on `/mcp`
|
|
219
|
+
* (`k8s/mcp-remote/nginx-mcp.conf`), while the backend's `/me` surface is one 300/60s bucket keyed
|
|
220
|
+
* on the POD's egress IP (`src/routes/me.ts`) — about 5 r/s for the whole fleet, and that same
|
|
221
|
+
* bucket also carries every ordinary `/me/*` tool call from every live session. Without this bound
|
|
222
|
+
* one source staying comfortably inside the proxy's limit drains it in seconds, and every
|
|
223
|
+
* LEGITIMATE keyed connection is answered 429 while the anonymous surface keeps working and the
|
|
224
|
+
* service looks healthy.
|
|
225
|
+
*
|
|
226
|
+
* THE ARITHMETIC, so the next reader can check it rather than trust it: 1/s per process × the
|
|
227
|
+
* replica count in `k8s/mcp-remote/statefulset.yaml` — the drift case named below multiplies the
|
|
228
|
+
* two and compares the product with the backend's own literal, so neither figure is restated here
|
|
229
|
+
* — leaving the remainder of that bucket for
|
|
230
|
+
* the ordinary `/me` traffic that shares the bucket. A self-inspection happens once per CONNECTION,
|
|
231
|
+
* not once per call, so 1/s is ~86k sign-ins a day per replica. It caches nothing about any token —
|
|
232
|
+
* a refusal here is "ask again in a moment", never a verdict — so the rule that liveness is decided
|
|
233
|
+
* only by the backend is untouched.
|
|
234
|
+
*
|
|
235
|
+
* STATED BOUND, and it is the residual of this cure rather than its elimination: the bucket is
|
|
236
|
+
* keyed on NOTHING — one counter for the process. A single source sending garbage bearers at just
|
|
237
|
+
* over the refill rate keeps it empty, and new keyed connections are then refused 429 while the
|
|
238
|
+
* anonymous surface and every already-open session keep working. That is strictly better than the
|
|
239
|
+
* shape it replaced (the same traffic used to spend the SHARED backend bucket, taking every
|
|
240
|
+
* tenant's ordinary `/me` calls down with it), but it is not nothing. Keying per client IP is the
|
|
241
|
+
* fix, and it needs a decision about trusting `X-Forwarded-For` that this task did not take.
|
|
242
|
+
*
|
|
243
|
+
* The other tail of the same residual is LOCAL rather than remote: the burst equals the whole
|
|
244
|
+
* session cap, and each self-inspection sits inside a reservation for up to `RESOLVE_TIMEOUT_MS`.
|
|
245
|
+
* The measured refusals (401, 429) release it immediately, so this only bites if the API HANGS —
|
|
246
|
+
* then one source can hold a replica's reservations for that window before the budget throttles it
|
|
247
|
+
* to the sustained rate. Bounded by the 5s timeout rather than by anything cleverer.
|
|
248
|
+
*/
|
|
249
|
+
const RESOLVE_RATE_PER_SECOND = 1;
|
|
250
|
+
/**
|
|
251
|
+
* Ceiling on the burst, expressed FLEET-WIDE because the bucket it spends is fleet-wide.
|
|
252
|
+
*
|
|
253
|
+
* The burst below is sized off `maxSessions`, which is per replica — and the bucket it lands in is
|
|
254
|
+
* `ipRateLimit(300, 60, 'me')` keyed on the pod's egress IP, i.e. shared by every replica. So the
|
|
255
|
+
* figure that matters is `replicas × burst`, and stating the burst per replica beside a rate stated
|
|
256
|
+
* fleet-wide is how those two drift apart: at `replicas: 3` and the default cap, an unbounded burst
|
|
257
|
+
* is 300 — the entire minute's bucket, spent at exactly the moment those sessions also begin their
|
|
258
|
+
* ordinary `/me/*` calls through it. `src/__tests__/drift/mcp-remote-deployment.test.ts` derives
|
|
259
|
+
* `replicas × min(MCP_MAX_SESSIONS, RESOLVE_BURST_CEILING)` — the EFFECTIVE burst, not this
|
|
260
|
+
* constant alone — against the backend's own literal, so SCALING THE FLEET reddens it instead of
|
|
261
|
+
* surfacing as 429s in production. Raising THIS constant is caught elsewhere and deliberately so:
|
|
262
|
+
* while the manifest cap sits at or below the ceiling, the effective burst does not move, and the
|
|
263
|
+
* unit case in `mcp/__tests__/remote.test.ts` that runs a cap ABOVE the ceiling is what holds it.
|
|
264
|
+
*/
|
|
265
|
+
export const RESOLVE_BURST_CEILING = 100;
|
|
266
|
+
/**
|
|
267
|
+
* How many may burst — sized off the capacity it can never exceed rather than picked, then clamped.
|
|
268
|
+
*
|
|
269
|
+
* A self-inspection happens once per connection, and a replica holds at most `maxSessions` of them,
|
|
270
|
+
* so a reconnect herd after a restart or a rolling deploy is bounded by that number. A burst
|
|
271
|
+
* smaller than the cap turns every such herd into refusals for everyone past the burst: the SDK
|
|
272
|
+
* client does not retry a 429, so each one reaches the human as a failed connection. The ceiling is
|
|
273
|
+
* what keeps that generosity inside the shared bucket above.
|
|
274
|
+
*/
|
|
275
|
+
function resolveBurstFor(maxSessions) {
|
|
276
|
+
return Math.min(maxSessions, RESOLVE_BURST_CEILING);
|
|
277
|
+
}
|
|
278
|
+
/** What a rate-limited caller is told to wait: long enough for one token, not for a whole burst. */
|
|
279
|
+
const RESOLVE_RETRY_AFTER_SECONDS = Math.ceil(1 / RESOLVE_RATE_PER_SECOND);
|
|
280
|
+
/**
|
|
281
|
+
* A token bucket over the self-inspection above. Process-local on purpose: what it protects is this
|
|
282
|
+
* process's own outbound channel, and a shared counter would need a store this service does not
|
|
283
|
+
* have and must not grow for a defensive bound.
|
|
284
|
+
*
|
|
285
|
+
* Exported for its own test. The arithmetic below has a failure mode no HTTP-level case can reach
|
|
286
|
+
* cheaply — a clock that steps BACKWARDS — and the previous version's clock seam was a default
|
|
287
|
+
* parameter on an unexported function, i.e. unreachable from a test even in principle. That is the
|
|
288
|
+
* same shape `positiveIntEnv` above was fixed for once already.
|
|
289
|
+
*/
|
|
290
|
+
export function createResolveBudget(maxSessions, now = () => performance.now()) {
|
|
291
|
+
const burst = resolveBurstFor(maxSessions);
|
|
292
|
+
let tokens = burst;
|
|
293
|
+
let last = now();
|
|
294
|
+
return {
|
|
295
|
+
/** True when a self-inspection may be sent; consumes one token. */
|
|
296
|
+
take() {
|
|
297
|
+
const at = now();
|
|
298
|
+
// Elapsed is clamped at BOTH ends. `performance.now()` is monotonic, so the floor is belt and
|
|
299
|
+
// braces there — but the seam accepts any clock, and a wall clock stepped backwards by an
|
|
300
|
+
// hour would otherwise drive the bucket to -7190 tokens and refuse every keyed connection for
|
|
301
|
+
// about an hour, while `/healthz` and the anonymous surface looked perfect (measured).
|
|
302
|
+
const elapsed = Math.max(0, at - last);
|
|
303
|
+
tokens = Math.min(burst, tokens + (elapsed / 1000) * RESOLVE_RATE_PER_SECOND);
|
|
304
|
+
last = at;
|
|
305
|
+
if (tokens < 1)
|
|
306
|
+
return false;
|
|
307
|
+
tokens -= 1;
|
|
308
|
+
return true;
|
|
309
|
+
},
|
|
310
|
+
};
|
|
311
|
+
}
|
|
312
|
+
/**
|
|
313
|
+
* Confirms the presented token names a live key, and answers the request itself if it does not.
|
|
314
|
+
*
|
|
315
|
+
* Returns true when the session may be opened. Every refusal arm exists because the alternative is
|
|
316
|
+
* a session that LOOKS open and cannot work:
|
|
317
|
+
* - 401: the token is dead. Answered with the challenge rather than by opening an anonymous
|
|
318
|
+
* session, because the challenge is what makes a client refresh — measured on Inspector, which
|
|
319
|
+
* rotated its token and replayed the call inside a second (`docs/mcp-oauth-client-probe.md`).
|
|
320
|
+
* - 403: the key exists but cannot inspect itself, i.e. it lacks `settings:read`. Named out loud,
|
|
321
|
+
* since the alternative is an opaque `invalid_token` on some later call that has nothing to do
|
|
322
|
+
* with scopes. The ceremony always grants that scope (`MCP_OAUTH_MANDATORY_SCOPE`), so this arm
|
|
323
|
+
* is reachable mainly by a hand-made key.
|
|
324
|
+
* - 429: rate-limited, which is neither a verdict on the token nor an outage — either by our own
|
|
325
|
+
* budget below or by the API upstream. Answered 429 so a client waits instead of concluding its
|
|
326
|
+
* credential died or the service is down. `Retry-After` is sent ONLY for our own budget, where
|
|
327
|
+
* the number is ours to know; the upstream's own header does not survive `ApiError` (which
|
|
328
|
+
* carries `body.error.details`, and the backend puts its `retryAfter` beside that, not in it),
|
|
329
|
+
* so claiming to pass it through would be a sentence with nothing behind it.
|
|
330
|
+
* - 400: the credential is not an API key at all (the endpoint answers `api_key_required` to a
|
|
331
|
+
* JWT). A live credential of the wrong KIND, reached successfully — so it is named as such
|
|
332
|
+
* rather than reported as an outage.
|
|
333
|
+
* - anything else (timeout, network, 5xx): 503. Opening the session would hand the caller a
|
|
334
|
+
* connection whose identity nothing has established.
|
|
335
|
+
*
|
|
336
|
+
* It builds its OWN client rather than taking the session's: the session's is configured for real
|
|
337
|
+
* work (retries, 30s timeout), and those settings are exactly what turn a rate-limited upstream
|
|
338
|
+
* into a parked capacity slot.
|
|
339
|
+
*/
|
|
340
|
+
async function resolvePresentedToken(presented, baseUrl, res, budget) {
|
|
341
|
+
if (!budget.take()) {
|
|
342
|
+
rateLimited(res, 'This server is checking too many credentials right now. Try again in a moment.', RESOLVE_RETRY_AFTER_SECONDS);
|
|
343
|
+
return false;
|
|
344
|
+
}
|
|
345
|
+
try {
|
|
346
|
+
// Inside the `try` so the docstring's list of outcomes stays exhaustive: a throw from the
|
|
347
|
+
// constructor would otherwise leave through the opener's catch and be answered `bad_request`,
|
|
348
|
+
// which is neither true nor one of the arms below. Unreachable today (the constructor only
|
|
349
|
+
// normalises a string), and cheaper to keep correct than to explain.
|
|
350
|
+
const probe = new HttpClient({ apiKey: presented, baseUrl, maxRetries: 0, timeout: RESOLVE_TIMEOUT_MS });
|
|
351
|
+
await probe.get('/api/v1/me/api-key');
|
|
352
|
+
return true;
|
|
353
|
+
}
|
|
354
|
+
catch (error) {
|
|
355
|
+
const status = error instanceof ApiError ? error.statusCode : 0;
|
|
356
|
+
const code = error instanceof ApiError ? error.code : '';
|
|
357
|
+
// The CODE as well as the status, because a 400 produced in FRONT of the API (a proxy refusing
|
|
358
|
+
// an oversized header, say) carries `http_400` and has nothing to say about the credential —
|
|
359
|
+
// answering it "that is not an API key" would be a confident wrong diagnosis. Everything but
|
|
360
|
+
// the endpoint's own `api_key_required` keeps falling into the 503 arm it had before.
|
|
361
|
+
if (status === 400 && code === 'api_key_required') {
|
|
362
|
+
// `GET /me/api-key` answers 400 `api_key_required` to a credential that is not an API key —
|
|
363
|
+
// a JWT, say. That is a live credential of the WRONG KIND reached successfully, so reporting
|
|
364
|
+
// it as "could not reach the API" would send the one person likely to hit it (somebody
|
|
365
|
+
// pasting a token by hand) looking for an outage.
|
|
366
|
+
json(res, 400, {
|
|
367
|
+
error: 'unsupported_credential',
|
|
368
|
+
error_description: 'That credential is not a Proof API key. This server accepts the API key an OAuth ' +
|
|
369
|
+
'connection grants; sign in through your client to get one.',
|
|
370
|
+
});
|
|
371
|
+
}
|
|
372
|
+
else if (status === 401) {
|
|
373
|
+
challenge(res, baseUrl, {
|
|
374
|
+
error: 'invalid_token',
|
|
375
|
+
error_description: 'The presented token is not a live Proof API key. Sign in again to get a new one.',
|
|
376
|
+
});
|
|
377
|
+
}
|
|
378
|
+
else if (status === 403) {
|
|
379
|
+
json(res, 403, {
|
|
380
|
+
error: 'insufficient_scope',
|
|
381
|
+
error_description: 'This key cannot inspect itself, so this server cannot tell whose connection it is: the ' +
|
|
382
|
+
'settings:read scope is required. Reconnect and approve it, or use a key that carries it.',
|
|
383
|
+
required_scope: 'settings:read',
|
|
384
|
+
});
|
|
385
|
+
}
|
|
386
|
+
else if (status === 429) {
|
|
387
|
+
// No `Retry-After` here, deliberately: the upstream's own header is not reachable from
|
|
388
|
+
// `ApiError` (see the docstring), and inventing a number would be worse than sending none.
|
|
389
|
+
rateLimited(res, 'The Proof API is rate-limiting this server right now. Try again shortly.');
|
|
390
|
+
}
|
|
391
|
+
else {
|
|
392
|
+
json(res, 503, {
|
|
393
|
+
error: 'upstream_unavailable',
|
|
394
|
+
error_description: 'Could not reach the Proof API to check the presented token. Try again shortly.',
|
|
395
|
+
});
|
|
396
|
+
}
|
|
397
|
+
return false;
|
|
398
|
+
}
|
|
399
|
+
}
|
|
400
|
+
/** True for the one JSON-RPC method allowed to open a session. */
|
|
401
|
+
const isInitialize = (body) => {
|
|
402
|
+
if (Array.isArray(body))
|
|
403
|
+
return body.some((entry) => entry?.method === 'initialize');
|
|
404
|
+
return body?.method === 'initialize';
|
|
405
|
+
};
|
|
406
|
+
/**
|
|
407
|
+
* The replica's own index: `MCP_REPLICA_ID` if set, otherwise the ordinal a StatefulSet puts at the
|
|
408
|
+
* end of the pod hostname (`mcp-remote-1` → `1`). Falls back to `0` for a single process, which is
|
|
409
|
+
* what the stdio-style local run and the tests get.
|
|
410
|
+
*/
|
|
411
|
+
/**
|
|
412
|
+
* Reads a positive-integer setting, falling back to the built-in default on anything else.
|
|
413
|
+
*
|
|
414
|
+
* A typo must not disable the bound it configures. `Number('abc')` is NaN, and both
|
|
415
|
+
* `size >= NaN` and `lastSeen < NaN` are false — so an unparseable value would fail OPEN on exactly
|
|
416
|
+
* the two limits this server relies on, with the pod still passing every probe. INTEGER rather than
|
|
417
|
+
* merely finite-and-positive, because `MCP_SESSION_TTL_MS=0.5` would sweep every session on the
|
|
418
|
+
* first tick and `MCP_MAX_SESSIONS=0.5` would silently mean one.
|
|
419
|
+
*
|
|
420
|
+
* Exported and at module scope so it can be tested — the previous version was a closure inside the
|
|
421
|
+
* entrypoint guard, unreachable from a test even in principle, which left the fix one edit wide.
|
|
422
|
+
*/
|
|
423
|
+
export function positiveIntEnv(name, env = process.env) {
|
|
424
|
+
const raw = env[name];
|
|
425
|
+
if (!raw)
|
|
426
|
+
return undefined;
|
|
427
|
+
// Digits only, which says the contract exactly once instead of spreading it across a numeric
|
|
428
|
+
// check and a textual exclusion: the earlier form tested `raw` for exponents but `parsed` for
|
|
429
|
+
// integrality, so `0xD` was accepted as 13 while `0xE` was refused, and `1.8e6` — a plausible
|
|
430
|
+
// shorthand for 30 minutes in ms — silently became the default.
|
|
431
|
+
const parsed = /^\d+$/.test(raw.trim()) ? Number(raw.trim()) : NaN;
|
|
432
|
+
// The upper bound is what a long decimal literal needs: 20 nines parses to 1e20, which is
|
|
433
|
+
// effectively unbounded, and no exponent appears in the string to catch it.
|
|
434
|
+
if (parsed > 0 && parsed <= Number.MAX_SAFE_INTEGER)
|
|
435
|
+
return parsed;
|
|
436
|
+
console.error(`Ignoring ${name}=${raw}: not a positive integer. Falling back to the built-in default.`);
|
|
437
|
+
return undefined;
|
|
438
|
+
}
|
|
439
|
+
export function resolveReplicaId(explicit, hostname = os.hostname()) {
|
|
440
|
+
if (explicit)
|
|
441
|
+
return explicit;
|
|
442
|
+
const ordinal = hostname.match(/-(\d+)$/);
|
|
443
|
+
return ordinal ? ordinal[1] : '0';
|
|
444
|
+
}
|
|
445
|
+
export function createRemoteServer(options) {
|
|
446
|
+
const maxSessions = options.maxSessions ?? DEFAULT_MAX_SESSIONS;
|
|
447
|
+
const ttlMs = options.sessionTtlMs ?? DEFAULT_SESSION_TTL_MS;
|
|
448
|
+
const sweepMs = options.sweepIntervalMs ?? DEFAULT_SWEEP_INTERVAL_MS;
|
|
449
|
+
const replicaId = resolveReplicaId(options.replicaId);
|
|
450
|
+
const sessions = new Map();
|
|
451
|
+
/** Opens that passed the cap check but have not reached `onsessioninitialized` yet. */
|
|
452
|
+
let pendingOpens = 0;
|
|
453
|
+
/** Bounds how fast this process may ask the API to resolve presented tokens. */
|
|
454
|
+
const resolveBudget = createResolveBudget(maxSessions);
|
|
455
|
+
const dropSession = async (sessionId) => {
|
|
456
|
+
const session = sessions.get(sessionId);
|
|
457
|
+
if (!session)
|
|
458
|
+
return;
|
|
459
|
+
sessions.delete(sessionId);
|
|
460
|
+
await session.close().catch(() => undefined);
|
|
461
|
+
};
|
|
462
|
+
const sweep = setInterval(() => {
|
|
463
|
+
const deadline = Date.now() - ttlMs;
|
|
464
|
+
for (const [sessionId, session] of sessions) {
|
|
465
|
+
if (session.inFlight > 0)
|
|
466
|
+
continue;
|
|
467
|
+
if (session.lastSeen < deadline)
|
|
468
|
+
void dropSession(sessionId);
|
|
469
|
+
}
|
|
470
|
+
}, sweepMs);
|
|
471
|
+
// Never hold the process open for the sake of the sweeper alone.
|
|
472
|
+
sweep.unref();
|
|
473
|
+
const httpServer = createServer(async (req, res) => {
|
|
474
|
+
const requestPath = (req.url ?? '').split('?')[0];
|
|
475
|
+
if (requestPath === '/healthz') {
|
|
476
|
+
json(res, 200, { status: 'ok', sessions: sessions.size });
|
|
477
|
+
return;
|
|
478
|
+
}
|
|
479
|
+
if (requestPath !== '/mcp') {
|
|
480
|
+
// Everything else is a plain 404 — including the OAuth metadata paths. This server has no
|
|
481
|
+
// authorization server, and answering there would advertise one that does not exist.
|
|
482
|
+
json(res, 404, { error: 'not_found' });
|
|
483
|
+
return;
|
|
484
|
+
}
|
|
485
|
+
const sessionId = req.headers['mcp-session-id'];
|
|
486
|
+
const existing = typeof sessionId === 'string' ? sessions.get(sessionId) : undefined;
|
|
487
|
+
if (existing) {
|
|
488
|
+
// BEFORE the counter, because a refusal here takes no slot: the credential presented now must
|
|
489
|
+
// be the one that opened this session. Both directions are refused — a foreign token on
|
|
490
|
+
// somebody else's session id, and a request that drops the token the session was opened with
|
|
491
|
+
// — so neither a leaked id nor a stripped header reaches another tenant's key.
|
|
492
|
+
//
|
|
493
|
+
// STATED BOUND: this runs before any method check, so it applies to the standalone `GET /mcp`
|
|
494
|
+
// event stream and to the session-terminating `DELETE` as well as to tool calls. A keyed
|
|
495
|
+
// client that omits the header on those gets a challenge rather than its stream or its
|
|
496
|
+
// teardown (the session then lingers to TTL). Standards-compliant clients attach the token to
|
|
497
|
+
// every request to the resource — the live run shows Claude Code doing exactly that — so this
|
|
498
|
+
// is a bound rather than a defect, and refusing is the safe direction: the alternative is a
|
|
499
|
+
// credential-less request acting on a keyed session.
|
|
500
|
+
if (fingerprintOf(bearerOf(req)) !== existing.keyFingerprint) {
|
|
501
|
+
challenge(res, options.baseUrl, sessionCredentialMismatchBody());
|
|
502
|
+
return;
|
|
503
|
+
}
|
|
504
|
+
// Counted from BEFORE the body is read: a slow upload is time the session is in use, and the
|
|
505
|
+
// sweeper must not reclaim it midway.
|
|
506
|
+
//
|
|
507
|
+
// POST ONLY, and that is the whole point of the distinction. `handleRequest` for a standalone
|
|
508
|
+
// `GET /mcp` SSE stream does not resolve until the stream ends, so counting it would keep
|
|
509
|
+
// `inFlight` at 1 for the stream's entire life — and the MCP SDK client opens that stream
|
|
510
|
+
// right after `initialize`. Counting it made the documented 30-minute idle TTL unreachable
|
|
511
|
+
// for every real client (measured: a session with an open GET survived 6× its TTL), leaving
|
|
512
|
+
// nginx's 3600s read timeout as the only bound. A stream is a connection, not work.
|
|
513
|
+
//
|
|
514
|
+
// The cost of that choice, measured rather than assumed: the official client does NOT
|
|
515
|
+
// re-initialize after its session is swept — the stream's reconnect gives up after two
|
|
516
|
+
// attempts and the next call fails `unknown_session` until the host reconnects. So the TTL
|
|
517
|
+
// is a memory bound paid for in reconnects. Counting the stream instead would trade that for
|
|
518
|
+
// sessions that are never reclaimed at all, which is the worse of the two.
|
|
519
|
+
const counted = req.method === 'POST';
|
|
520
|
+
// THE RULE, in the words the code can actually perform: the idle clock moves for a POST the
|
|
521
|
+
// transport ACCEPTED — one it answered below 400. `inFlight` is what protects a request that
|
|
522
|
+
// is still running; `lastSeen` is set in the `finally` below, the only place that can see the
|
|
523
|
+
// answer.
|
|
524
|
+
//
|
|
525
|
+
// The word "work" is retired here on purpose. It was false five times running, each time on a
|
|
526
|
+
// path the previous fix had not looked at:
|
|
527
|
+
// - an ARRIVAL stamp refreshed the clock for requests that never reached the transport
|
|
528
|
+
// (measured: five oversized bodies, each answered 413, held a session across twice its TTL);
|
|
529
|
+
// - a stamp on every COMPLETED request refreshed it for the four shapes the TRANSPORT itself
|
|
530
|
+
// refuses — unparseable body, empty body, unsupported `mcp-protocol-version`, a second
|
|
531
|
+
// `initialize` — each answered 400 from inside the SDK before `onmessage` runs, so 16 bytes
|
|
532
|
+
// of garbage postponed the TTL while 5MB of it did not;
|
|
533
|
+
// - and "accepted, therefore worked" is false too. `POST []` runs the SDK's message loop zero
|
|
534
|
+
// times and is still answered 202, so nothing is dispatched and the clock moves anyway
|
|
535
|
+
// (the original measurement: six of them over 360ms held a session whose TTL was 150ms —
|
|
536
|
+
// the case guarding it now uses a shorter TTL and more requests, which is margin, not a
|
|
537
|
+
// new fact).
|
|
538
|
+
// ACCEPTED is what
|
|
539
|
+
// this code can observe, so ACCEPTED is what it claims — and the empty batch has a test of
|
|
540
|
+
// its own, so the day someone refuses it, the sentence gets rewritten rather than quietly
|
|
541
|
+
// becoming a different claim.
|
|
542
|
+
// Worth naming rather than leaving to be rediscovered: this makes the TTL no kind of abuse
|
|
543
|
+
// bound — a single `notifications/initialized` sent just inside the TTL holds a slot just as
|
|
544
|
+
// cheaply, and it IS dispatched. Capacity is bounded by `MCP_MAX_SESSIONS` and the proxy's rate limit instead.
|
|
545
|
+
//
|
|
546
|
+
// The rule ships in two artifacts — here and `README.md`'s Transport section — and the two are
|
|
547
|
+
// tied together by `src/__tests__/drift/mcp-remote-deployment.test.ts`, which requires each to
|
|
548
|
+
// state it exactly once. (The cases in `__tests__/remote.test.ts` restate it too, in prose that
|
|
549
|
+
// explains what they measure; those are not shipped, and the guard does not read them.)
|
|
550
|
+
if (counted)
|
|
551
|
+
existing.inFlight += 1;
|
|
552
|
+
let body;
|
|
553
|
+
try {
|
|
554
|
+
body = req.method === 'POST' ? await readBody(req) : undefined;
|
|
555
|
+
}
|
|
556
|
+
catch (error) {
|
|
557
|
+
// Same bound on the established-session path: a live session is not a licence to buffer.
|
|
558
|
+
// The release is symmetric with the increment above — unconditional here while the
|
|
559
|
+
// increment was conditional would drive the counter negative on a path that never took a
|
|
560
|
+
// slot, and a negative count disables the sweeper's skip for that session entirely.
|
|
561
|
+
if (counted)
|
|
562
|
+
existing.inFlight -= 1;
|
|
563
|
+
rejectBody(req, res, error);
|
|
564
|
+
return;
|
|
565
|
+
}
|
|
566
|
+
// THE GATE, and the whole reason it sits here rather than in a tool handler: a 401 with
|
|
567
|
+
// `WWW-Authenticate` is an HTTP answer, and once the message reaches the transport the reply
|
|
568
|
+
// is a JSON-RPC result no client's OAuth stack will read. A session with NO account behind it
|
|
569
|
+
// may run the keyless surface; a registered tool outside it is refused with the challenge that
|
|
570
|
+
// starts the ceremony, and a name the session's server does not register goes on to the SDK,
|
|
571
|
+
// which answers it as an unknown tool (`keyedToolsIn` with `existing.toolNames`). For a session
|
|
572
|
+
// with no account behind it `initialize` and `tools/list` are never refused — the control measurement for that arrangement was a connection timeout, not a login
|
|
573
|
+
// prompt. THREE things do answer 401 at the handshake, and the third is on the anonymous
|
|
574
|
+
// road, so "never refused" needs all of them: a PRESENTED token that fails to resolve
|
|
575
|
+
// (`resolvePresentedToken`), a credential that does not match this session (above), and a
|
|
576
|
+
// batch that smuggles a keyed `tools/call` alongside `initialize` — refused by this same rule
|
|
577
|
+
// applied on the opening path, because the tool call is what asks for an account.
|
|
578
|
+
//
|
|
579
|
+
// TWO ways to have an account, not one, and reading only the first was a real defect: besides
|
|
580
|
+
// the API key an OAuth client presents, a caller can SIGN IN from inside the session through
|
|
581
|
+
// `start_login` / `wait_for_login`, after which `captureSetCookie` holds a live JWT on this
|
|
582
|
+
// connection's client and the `SESSION_AUTH_PREFIXES` tools travel with it. That is the road
|
|
583
|
+
// the refusal text itself advertises to clients that cannot speak OAuth, and gating on the
|
|
584
|
+
// key alone closed it — the user signed in and was told to sign in. A tool outside that
|
|
585
|
+
// surface still answers `api_key_required` from the client, exactly as it did before this
|
|
586
|
+
// task; what the gate must not do is refuse the ones login actually unlocks.
|
|
587
|
+
//
|
|
588
|
+
// STATED BOUND, and harmless because the backend stays authoritative on every call: the
|
|
589
|
+
// client clears its session only on `token_expired` (see the refresh path in `./http.js`), so
|
|
590
|
+
// a JWT REVOKED server-side leaves this condition true. Such a caller keeps being forwarded
|
|
591
|
+
// and keeps being refused one call at a time by the backend, rather than getting the
|
|
592
|
+
// challenge that would restart the ceremony. It grants nothing — it is a worse error message,
|
|
593
|
+
// not a hole.
|
|
594
|
+
if (!existing.keyFingerprint && !existing.http.hasSessionToken()) {
|
|
595
|
+
const keyed = keyedToolsIn(body, existing.toolNames);
|
|
596
|
+
if (keyed.length > 0) {
|
|
597
|
+
// Symmetric with `rejectBody` above: the counter was taken before the body was read, and
|
|
598
|
+
// an unbalanced exit would leave this session permanently "busy" for the sweeper.
|
|
599
|
+
if (counted)
|
|
600
|
+
existing.inFlight -= 1;
|
|
601
|
+
// `lastSeen` deliberately untouched — the file's rule is that the idle clock moves for an
|
|
602
|
+
// answer BELOW 400, and this is a refusal.
|
|
603
|
+
challenge(res, options.baseUrl, refusalFor(options.baseUrl, keyed[0]));
|
|
604
|
+
return;
|
|
605
|
+
}
|
|
606
|
+
}
|
|
607
|
+
try {
|
|
608
|
+
// No `catch` here, unlike the new-session path below, and the asymmetry is deliberate:
|
|
609
|
+
// there the `catch` exists to release the RESERVATION, which this path never takes. A
|
|
610
|
+
// throw out of `handleRequest` is caught inside the SDK's own listener; if that ever
|
|
611
|
+
// changed, the process-level handler keeps the server serving and this socket would be
|
|
612
|
+
// held until the proxy's read timeout.
|
|
613
|
+
await existing.transport.handleRequest(req, res, body);
|
|
614
|
+
}
|
|
615
|
+
finally {
|
|
616
|
+
if (counted) {
|
|
617
|
+
existing.inFlight -= 1;
|
|
618
|
+
// Accepted: the idle clock restarts when the answer went out, not when the request began.
|
|
619
|
+
//
|
|
620
|
+
// `headersSent` is DEFENSIVE and currently unreachable — stated rather than implied, the
|
|
621
|
+
// same posture as `rejectBody`'s `res.end()` arm. `res.statusCode` reads 200 until
|
|
622
|
+
// something writes it, so without the conjunct a request that died before answering would
|
|
623
|
+
// stamp as a successful one; but every route out of `handleRequest` writes a response
|
|
624
|
+
// before resolving, and the one that does not (a client abort, which is swallowed as
|
|
625
|
+
// `ERR_STREAM_PREMATURE_CLOSE`) has already dispatched its message, so the stamp would have
|
|
626
|
+
// been right anyway. No test covers it, and none is claimed to.
|
|
627
|
+
//
|
|
628
|
+
// Name the right dependency, though: the guarantee belongs to `@hono/node-server` (1.19.9
|
|
629
|
+
// here), which the SDK uses to bridge to `node:http`. Stated as the invariant rather than
|
|
630
|
+
// as a mechanism — headers are SENT by the time it resolves — because not every branch
|
|
631
|
+
// sends them itself: two hand off to something that already did. It is a TRANSITIVE
|
|
632
|
+
// dependency, so a hono bump can move it while the SDK version pinned elsewhere in this
|
|
633
|
+
// file stays put.
|
|
634
|
+
if (res.headersSent && res.statusCode < 400)
|
|
635
|
+
existing.lastSeen = Date.now();
|
|
636
|
+
}
|
|
637
|
+
}
|
|
638
|
+
return;
|
|
639
|
+
}
|
|
640
|
+
if (typeof sessionId === 'string') {
|
|
641
|
+
// A session id this process does not know. Refused rather than treated as a fresh session:
|
|
642
|
+
// silently starting a new one would make a misrouted request LOOK like a success, and the
|
|
643
|
+
// absence of sticky routing would surface much later as inexplicably lost state.
|
|
644
|
+
json(res, 404, { error: 'unknown_session' });
|
|
645
|
+
return;
|
|
646
|
+
}
|
|
647
|
+
if (req.method !== 'POST') {
|
|
648
|
+
json(res, 400, { error: 'session_required' });
|
|
649
|
+
return;
|
|
650
|
+
}
|
|
651
|
+
// RESERVE the slot before the first await, and count reservations alongside live sessions.
|
|
652
|
+
// Checking `sessions.size` alone is not a bound: the map is written in `onsessioninitialized`,
|
|
653
|
+
// which is several awaits away, so every concurrent opener reads the pre-request count.
|
|
654
|
+
// Measured on the unreserved version: `maxSessions: 2` admitted 25 concurrent opens.
|
|
655
|
+
//
|
|
656
|
+
// The deliberate trade: a slow uploader holds its reservation while it dribbles, so parked
|
|
657
|
+
// openers can occupy the cap without creating a session. Node's `requestTimeout` (300s) bounds
|
|
658
|
+
// it, and the documented deployment puts nginx in front with request buffering on, so the
|
|
659
|
+
// shape is not reachable there — a capacity pause is accepted in exchange for the memory bound.
|
|
660
|
+
if (sessions.size + pendingOpens >= maxSessions) {
|
|
661
|
+
json(res, 503, { error: 'session_capacity_reached' });
|
|
662
|
+
return;
|
|
663
|
+
}
|
|
664
|
+
pendingOpens += 1;
|
|
665
|
+
try {
|
|
666
|
+
const body = await readBody(req);
|
|
667
|
+
// Build nothing until the body says it is an opening handshake. Previously any malformed POST
|
|
668
|
+
// constructed a full server + transport (measured 1.35ms in the factory, plus the whole tool set) only for the SDK to
|
|
669
|
+
// answer 400 — an uncounted allocation amplifier on the cheapest possible request.
|
|
670
|
+
if (!isInitialize(body)) {
|
|
671
|
+
json(res, 400, { error: 'initialize_expected' });
|
|
672
|
+
return;
|
|
673
|
+
}
|
|
674
|
+
const presented = bearerOf(req);
|
|
675
|
+
// The gate again, on the OPENING request. `isInitialize` accepts a batch that merely
|
|
676
|
+
// CONTAINS an initialize, so in principle `[initialize, tools/call …]` reaches the transport
|
|
677
|
+
// without ever passing the check above. SDK 1.27.1 refuses such a batch itself (400, "Only
|
|
678
|
+
// one initialization request is allowed") and an anonymous session has no credential to
|
|
679
|
+
// spend — but both of those are somebody else's properties, and the boundary should not rest
|
|
680
|
+
// on a dependency's version. Two lines make it self-standing.
|
|
681
|
+
const smuggled = keyedToolsIn(body);
|
|
682
|
+
if (!presented && smuggled.length > 0) {
|
|
683
|
+
challenge(res, options.baseUrl, refusalFor(options.baseUrl, smuggled[0]));
|
|
684
|
+
return;
|
|
685
|
+
}
|
|
686
|
+
// RESOLVE the presented token BEFORE anything is built for it, by the SAME self-inspection an
|
|
687
|
+
// agent uses (`GET /api/v1/me/api-key`, `settings:read`). No new authorization branch is
|
|
688
|
+
// invented here and nothing is remembered: every scope decision stays with
|
|
689
|
+
// `requireScopeIfApiKey` on the backend, on every call. What this buys is a legible refusal —
|
|
690
|
+
// without it a dead token opens a session that answers `authorization_required` to
|
|
691
|
+
// everything, and a token missing `settings:read` fails later as an opaque `invalid_token`.
|
|
692
|
+
//
|
|
693
|
+
// Ordered before `createProofServer` deliberately, and it is the same rule the `isInitialize`
|
|
694
|
+
// check above states: build nothing for a request that is about to be refused. A full server
|
|
695
|
+
// is the whole tool set, ~1.84MB of heap by the table above, and a garbage bearer is the
|
|
696
|
+
// cheapest request an attacker can send.
|
|
697
|
+
if (presented && !(await resolvePresentedToken(presented, options.baseUrl, res, resolveBudget)))
|
|
698
|
+
return;
|
|
699
|
+
const { server, http, toolNames } = createProofServer({ baseUrl: options.baseUrl, apiKey: presented, remote: true });
|
|
700
|
+
const transport = new StreamableHTTPServerTransport({
|
|
701
|
+
// `<replica>.<uuid>` — the prefix is what the proxy routes on. See `replicaId` above.
|
|
702
|
+
sessionIdGenerator: () => `${replicaId}.${randomUUID()}`,
|
|
703
|
+
onsessioninitialized: (id) => {
|
|
704
|
+
sessions.set(id, {
|
|
705
|
+
transport,
|
|
706
|
+
http,
|
|
707
|
+
toolNames,
|
|
708
|
+
close: async () => {
|
|
709
|
+
await transport.close().catch(() => undefined);
|
|
710
|
+
await server.close().catch(() => undefined);
|
|
711
|
+
},
|
|
712
|
+
// The one arrival-shaped stamp left, and deliberately so: the opening POST creates the
|
|
713
|
+
// entry, so there is nothing to stamp on its completion. It is safe for a reason
|
|
714
|
+
// outside our code, and measured rather than assumed: SDK 1.27.1 answers any batch
|
|
715
|
+
// carrying `initialize` alongside anything else with 400 "Only one initialization
|
|
716
|
+
// request is allowed", so this request cannot also be carrying work whose completion
|
|
717
|
+
// would be the honest moment to stamp.
|
|
718
|
+
lastSeen: Date.now(),
|
|
719
|
+
inFlight: 0,
|
|
720
|
+
keyFingerprint: fingerprintOf(presented),
|
|
721
|
+
});
|
|
722
|
+
},
|
|
723
|
+
});
|
|
724
|
+
transport.onclose = () => {
|
|
725
|
+
if (transport.sessionId)
|
|
726
|
+
sessions.delete(transport.sessionId);
|
|
727
|
+
};
|
|
728
|
+
await server.connect(transport);
|
|
729
|
+
await transport.handleRequest(req, res, body);
|
|
730
|
+
}
|
|
731
|
+
catch (error) {
|
|
732
|
+
// `readBody` rejects on an oversized body and on a client that disappears mid-upload; the
|
|
733
|
+
// reservation must come back in both cases, which is what the `finally` below is for.
|
|
734
|
+
rejectBody(req, res, error);
|
|
735
|
+
}
|
|
736
|
+
finally {
|
|
737
|
+
pendingOpens -= 1;
|
|
738
|
+
}
|
|
739
|
+
});
|
|
740
|
+
return {
|
|
741
|
+
httpServer,
|
|
742
|
+
sessionCount: () => sessions.size,
|
|
743
|
+
debugClients: () => [...sessions.values()].map((session) => session.http),
|
|
744
|
+
close: async () => {
|
|
745
|
+
clearInterval(sweep);
|
|
746
|
+
await Promise.all([...sessions.keys()].map((sessionId) => dropSession(sessionId)));
|
|
747
|
+
await new Promise((resolve) => httpServer.close(() => resolve()));
|
|
748
|
+
},
|
|
749
|
+
};
|
|
750
|
+
}
|
|
751
|
+
const isEntrypoint = () => {
|
|
752
|
+
const invoked = process.argv[1];
|
|
753
|
+
return !!invoked && path.resolve(invoked) === fileURLToPath(import.meta.url);
|
|
754
|
+
};
|
|
755
|
+
if (isEntrypoint()) {
|
|
756
|
+
// Bare `Number()` on purpose, unlike the two settings below: `listen(NaN)` throws
|
|
757
|
+
// ERR_SOCKET_BAD_PORT synchronously, so a typo here is fail-LOUD and cannot silently disable
|
|
758
|
+
// anything. The pod carries no `PORT` env at all (pinned by the deployment drift suite).
|
|
759
|
+
const port = Number(process.env.PORT ?? 3100);
|
|
760
|
+
const remote = createRemoteServer({
|
|
761
|
+
baseUrl: process.env.PROOF_BASE_URL ?? 'https://api.proof.holdings',
|
|
762
|
+
replicaId: process.env.MCP_REPLICA_ID,
|
|
763
|
+
maxSessions: positiveIntEnv('MCP_MAX_SESSIONS'),
|
|
764
|
+
sessionTtlMs: positiveIntEnv('MCP_SESSION_TTL_MS'),
|
|
765
|
+
});
|
|
766
|
+
remote.httpServer.listen(port, () => {
|
|
767
|
+
console.error(`Proof MCP remote server listening on ${port} ` +
|
|
768
|
+
'(anonymous connections allowed; an API key is read from the Authorization header only)');
|
|
769
|
+
});
|
|
770
|
+
// The stdio entrypoint has carried these since it shipped; this one serves MANY users, so an
|
|
771
|
+
// uncaught error here kills every live session rather than one. Log and keep serving: on Node 22
|
|
772
|
+
// an unhandled rejection would otherwise terminate the process by default.
|
|
773
|
+
process.on('uncaughtException', (error) => {
|
|
774
|
+
console.error('Uncaught exception:', error instanceof Error ? error.message : error);
|
|
775
|
+
});
|
|
776
|
+
process.on('unhandledRejection', (error) => {
|
|
777
|
+
console.error('Unhandled rejection:', error instanceof Error ? error.message : error);
|
|
778
|
+
});
|
|
779
|
+
const shutdown = () => {
|
|
780
|
+
// Bounded: `httpServer.close()` waits on keep-alive connections, and a Streamable HTTP client
|
|
781
|
+
// holds them open by design, so an unbounded wait would sit here until SIGKILL.
|
|
782
|
+
const forced = setTimeout(() => process.exit(0), 15_000);
|
|
783
|
+
forced.unref();
|
|
784
|
+
void remote.close().then(() => process.exit(0));
|
|
785
|
+
};
|
|
786
|
+
process.on('SIGTERM', shutdown);
|
|
787
|
+
process.on('SIGINT', shutdown);
|
|
788
|
+
}
|
|
789
|
+
//# sourceMappingURL=remote.js.map
|