mcp-tool-bridge 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Julien
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,492 @@
1
+ # mcp-tool-bridge
2
+
3
+ A generic [Model Context Protocol](https://modelcontextprotocol.io) server that exposes a
4
+ catalog of **declared** tools to a model, filters them by role, validates every argument
5
+ and holds sensitive calls behind a confirmation the model cannot give itself.
6
+
7
+ You declare what a tool is (its arguments, how much harm it can do, whether it can be
8
+ undone, who may use it); the bridge enforces it on every call, whatever the model says.
9
+
10
+ > **Status: 1.1.** Tool declarations, the registry, role-based access, argument
11
+ > validation, the bridge, the confirmation guard with lifetimes per class of tool, an
12
+ > audit log that keeps no content by default, the stdio server and three adapters
13
+ > (`envelope`, `adapt`, `importDefinitions`), with a runnable example in
14
+ > [`examples/minimal`](./examples/minimal). The HTTP transport comes next.
15
+
16
+ ## Requirements
17
+
18
+ Node.js 20 or later. The package ships as both ES modules and CommonJS. Zod is optional:
19
+ install `zod@^4` only if you use `mcp-tool-bridge/zod`.
20
+
21
+ ## A first look
22
+
23
+ ```ts
24
+ import { defineTool, jsonSchema, ToolRegistry, text } from 'mcp-tool-bridge'
25
+
26
+ const sendInvoice = defineTool({
27
+ name: 'send_invoice',
28
+ description: 'Emails an existing invoice to the customer it belongs to.',
29
+ args: jsonSchema({
30
+ type: 'object',
31
+ properties: { invoiceId: { type: 'string', pattern: '^INV-[0-9]+$' } },
32
+ required: ['invoiceId'],
33
+ additionalProperties: false,
34
+ }),
35
+ sensitivity: 'high', // none | low | medium | high | critical
36
+ reversible: false, // an email cannot be unsent
37
+ roles: ['billing'],
38
+ summarize: ({ invoiceId }) => `Email invoice ${invoiceId} to its customer`,
39
+ handler: async ({ invoiceId }) => {
40
+ // `invoiceId` is a string here: the type comes from the schema above.
41
+ return text(`Invoice ${invoiceId} sent.`)
42
+ },
43
+ })
44
+
45
+ const registry = new ToolRegistry().register(sendInvoice)
46
+ ```
47
+
48
+ With Zod:
49
+
50
+ ```ts
51
+ import * as z from 'zod'
52
+ import { zodSchema } from 'mcp-tool-bridge/zod'
53
+
54
+ const args = zodSchema(z.object({ invoiceId: z.string().regex(/^INV-[0-9]+$/) }))
55
+ ```
56
+
57
+ The bridge is the only way to run a tool. It works without any transport, which is how
58
+ a host calls it directly and how the tests exercise it:
59
+
60
+ ```ts
61
+ import { createBridge } from 'mcp-tool-bridge'
62
+
63
+ const bridge = createBridge({
64
+ registry,
65
+ context: (principal) => ({ db, tenantId: principal.id }), // what handlers get as call.context
66
+ })
67
+
68
+ const alice = { id: 'alice', roles: ['billing'] }
69
+ bridge.listTools(alice) // what the model may see
70
+
71
+ const outcome = await bridge.callTool(alice, {
72
+ name: 'send_invoice',
73
+ arguments: { invoiceId: 'INV-12' },
74
+ })
75
+ // send_invoice is high-sensitivity and irreversible: nothing ran.
76
+ // outcome.status === 'confirmation_required'
77
+ // outcome.confirmation = { token, expiresAt, tool, summary: 'Email invoice INV-12 to its customer' }
78
+
79
+ // Later, once a human has said yes in the host's own interface:
80
+ await bridge.executeConfirmed(alice, outcome.confirmation.token) // { status: 'ok', … }
81
+ // …or no:
82
+ await bridge.revokeConfirmation(alice, outcome.confirmation.token)
83
+ ```
84
+
85
+ Every outcome is a value, never an exception: `ok`, `tool_error`, `invalid_arguments`,
86
+ `confirmation_required` or `rejected`, each with the `callId` found in the audit log.
87
+
88
+ ## Serving over MCP
89
+
90
+ `serveStdio` serves the bridge over stdin/stdout, for one principal fixed when the
91
+ process starts:
92
+
93
+ ```ts
94
+ import { parsePrincipal, serveStdio } from 'mcp-tool-bridge'
95
+
96
+ await serveStdio(bridge, {
97
+ info: { name: 'billing-tools', version: '1.0.0' },
98
+ principal: parsePrincipal({ id: 'alice', roles: ['billing'] }),
99
+ })
100
+ ```
101
+
102
+ For another transport, `createMcpServer(bridge, options)` returns the SDK's `Server`,
103
+ ready to connect.
104
+
105
+ On the wire:
106
+
107
+ - `tools/list` returns the principal's tools only. The server announces
108
+ `tools.listChanged` and notifies the client whenever the registry changes.
109
+ - An unknown tool, or a tool the principal may not use, is a JSON-RPC error
110
+ (`-32602 Unknown tool: …`), the same for both. Invalid arguments and handler failures
111
+ are results with `isError: true`, which the model can read and act on. A refused
112
+ confirmation token reads the same whatever the reason: "This confirmation cannot be
113
+ used. Nothing was done."
114
+ - A call that needs confirmation is resolved in one of two ways:
115
+ - **The client supports elicitation:** the server asks the user through the client
116
+ and runs the call on a yes. A no, or a dismissed question, withdraws it.
117
+ - **It does not, or elicitation is turned off** (`elicitConfirmations: false`): the
118
+ result says that confirmation is pending, and the token travels in the result's
119
+ `_meta` under `mcp-tool-bridge/confirmation`. To redeem it, the host repeats the same
120
+ call with `{ token }` under the same key in the request's `_meta`, after a human said
121
+ yes. The model has no way to do this: it writes arguments, not `_meta`, and a token
122
+ placed in the arguments is ignored.
123
+
124
+ Install it in an MCP client such as Claude Code with
125
+ `claude mcp add billing -- npx tsx path/to/server.ts`.
126
+
127
+ ## Plugging in existing code
128
+
129
+ Most tools already exist somewhere, with their own input shape and their own way of
130
+ reporting failure. Three adapters connect them without rewriting them:
131
+
132
+ - `envelope()` reads a result envelope such as `{ success, data, error }` or
133
+ `{ ok, detail }`. A failed envelope becomes a `ToolError`, so the model reads its
134
+ message; a successful one becomes the tool output.
135
+ - `adapt()` wraps an existing function. The bridge validates the arguments, `input`
136
+ maps them to what the function expects, and `output` (often an `envelope`) maps the
137
+ result back.
138
+
139
+ ```ts
140
+ handler: adapt((action: LegacyAction) => legacy.execute(action), {
141
+ input: (args, call) => ({ userId: call.principal.id, data: JSON.stringify(args) }),
142
+ output: fromLegacy,
143
+ }),
144
+ ```
145
+
146
+ - `importDefinitions()` turns tool definitions written for an LLM API (Anthropic
147
+ `input_schema`, MCP `inputSchema`, OpenAI `parameters`) and one dispatcher into
148
+ tools. What those formats do not say has to be declared, for every tool:
149
+
150
+ ```ts
151
+ const tools = importDefinitions(definitions, (name, args, call) => run(name, args, call), {
152
+ search_orders: { sensitivity: 'none', reversible: true, roles: ['support'] },
153
+ refund_order: { sensitivity: 'high', reversible: false, roles: ['billing'] },
154
+ })
155
+ ```
156
+
157
+ A definition without governance, or governance for a name that has no definition,
158
+ fails at startup with every name listed. When one tool is unclassified, none starts.
159
+ Imported schemas are compiled in the same strict mode as `jsonSchema()`.
160
+
161
+ ## What the audit log keeps
162
+
163
+ An audit log that records arguments and results in full is a second copy of every
164
+ email body, every value written to a spreadsheet, every document read. It is also the
165
+ place nobody thinks of when data is deleted. So the default is the opposite: **the
166
+ audit log keeps no content at all.**
167
+
168
+ Every event carries metadata only:
169
+
170
+ | Field | What it is |
171
+ | ------------------------------------------------ | --------------------------------------------------------------------------------------------------------------------------------- |
172
+ | `tool`, `principal`, `callId`, `transport`, `at` | Who called what, when, and through which channel. |
173
+ | `argsDigest` | SHA-256 of the canonical arguments. It tells two calls apart and links a call to its confirmation, without keeping what was sent. |
174
+ | `type`, `reason`, `detail` | The verdict: started, succeeded, failed, rejected, and why. |
175
+ | `durationMs`, `result.isError` | How long it took, and whether the result was an error. |
176
+
177
+ A tool that needs to keep some content says so, field by field, with JSON Pointers
178
+ into its arguments and into its structured result:
179
+
180
+ ```ts
181
+ defineTool({
182
+ name: 'read_email',
183
+ // …
184
+ audit: { args: ['/messageId'] }, // which message was read; never its content
185
+ })
186
+
187
+ defineTool({
188
+ name: 'search_email',
189
+ // …
190
+ audit: { result: ['/messages/*/id'] }, // which messages came back
191
+ })
192
+ ```
193
+
194
+ Kept fields appear in the event keyed by their pointer: `"args": { "/messageId":
195
+ "m-1" }`. A `*` segment collects every match. Keeping content is a written decision,
196
+ in the declaration, visible in review. This is the same rule as governance: the default
197
+ path is the safe one, and a forgotten line costs a missing detail in the log, not a
198
+ leaked email.
199
+
200
+ Some guarantees hold whatever a tool declares:
201
+
202
+ - **Reduced on the way in.** Arguments are reduced to their digest and their declared
203
+ fields when the call's audit scope is built, before any event exists. The full value
204
+ never lives in an audit event object, so it cannot resurface in an exception trace or
205
+ a debug dump. Serialization only ever sees the reduced value.
206
+ - **Second line of defence.** Inside a kept value, keys that look like secrets
207
+ (`password`, `token`, `apiKey`, `authorization`, `secret`, `cookie`…) are masked.
208
+ Kept strings are cut at 500 characters, and at most 100 matches are kept per pointer.
209
+ - **No summary, no text.** The confirmation summary is written from the arguments ("Send
210
+ 'Invoice 12' to ada@…"), so it goes to the host, never to the audit log. Result text
211
+ and binary content are never kept, only declared fields of the structured result.
212
+ - **Every refused token is recorded with its reason**: `unknown`, `consumed` (a
213
+ replay), `expired`, `principal_mismatch`, `tool_mismatch`, `arguments_mismatch`. These
214
+ are the lines that show whether something is trying to force its way. The model gets
215
+ one generic refusal and never learns which check stopped it.
216
+
217
+ ## Design decisions
218
+
219
+ ### At a glance
220
+
221
+ **Tools are frozen descriptors that only the bridge can run.** `defineTool` returns an
222
+ immutable description that holds no reference to the handler. The handler is kept
223
+ privately and only the bridge can reach it, so every execution goes through the same
224
+ checks: role, arguments, confirmation, audit. No code in the host can call a handler
225
+ by mistake, and no one can widen a tool's roles after its declaration has been checked.
226
+
227
+ **`roles` is required and cannot be empty.** A tool without roles would be open to
228
+ everyone or to no one. Either way, that is a decision someone must make, and make
229
+ visibly. Failing at startup turns a forgotten line into an error the developer sees
230
+ immediately. Otherwise it would surface in production as a tool silently open to all,
231
+ or silently dead.
232
+
233
+ **The same function decides access for the list and for the call.** The access rule
234
+ lives in one place, `canAccess`. The list hides what a principal cannot use. The call
235
+ checks again, because a client can send any tool name and roles can change in between.
236
+ Two implementations would end up disagreeing. Either the model is shown tools it cannot
237
+ call, or, worse, it can call tools it is never shown.
238
+
239
+ ### Why a declarative registry
240
+
241
+ A tool is two things: code that does something, and facts about that code. How sensitive
242
+ is it? Can it be undone? Who may use it? What arguments does it accept? In most agent
243
+ code these facts live in the head of whoever wrote the handler. At best they are
244
+ scattered: an `if` in the handler, a sentence in the system prompt, a comment.
245
+
246
+ Here they are fields of the declaration, and the governance fields (`sensitivity`,
247
+ `reversible`, `roles`) have **no default**. A tool nobody classified does not start. It
248
+ does not silently become "low sensitivity, everyone". Because the facts are data, the
249
+ bridge can apply one policy to every tool. It can also hand the facts to a reviewer, or
250
+ to an audit, as a table rather than as code to read.
251
+
252
+ The registry only accepts descriptors created by `defineTool`, which checks the whole
253
+ declaration at startup. A descriptor carries no handle on the handler: the only way to
254
+ run a tool is through the bridge and its checks. A copy made with `{ ...tool }`
255
+ type-checks, but registering it is refused.
256
+
257
+ ### Why filter twice
258
+
259
+ The access rule is one function, `canAccess(tool, principal)`: the principal must hold
260
+ at least one of the tool's roles. Names are compared exactly. There is no wildcard and
261
+ no hierarchy, so anyone can read who has access to what from the declarations alone.
262
+
263
+ The bridge applies this rule twice: when it lists tools for a principal, and again when
264
+ it executes a call. Two reasons:
265
+
266
+ - **The list is a convenience, not a barrier.** Hiding a tool keeps it out of the
267
+ model's context. It saves tokens and avoids tempting the model with a tool it cannot
268
+ use. But nothing forces a client to call only the tools it was shown. A model can
269
+ hallucinate a name, a prompt injection can supply one, a client can be buggy or
270
+ hostile. The check that matters is the one made when the call is executed.
271
+ - **Time passes between the two.** Roles can be revoked and tools unregistered while a
272
+ session is open. The decision is taken again, at the moment the effect would happen.
273
+
274
+ A call refused for lack of a role gets the same answer as a call to a tool that does not
275
+ exist (`unknown_tool`), so probing names reveals nothing. The audit log still records
276
+ the real reason.
277
+
278
+ ### Why validate on the server
279
+
280
+ The argument schema is declared once and serves two purposes: the JSON Schema shown to
281
+ the model, and the check enforced before the handler runs. Both adapters (`jsonSchema`
282
+ and `zodSchema`) derive the two from the same declaration, so they cannot drift apart.
283
+ The handler's argument type is inferred from the same source.
284
+
285
+ Validation is strict on purpose:
286
+
287
+ - **No coercion.** `"3"` is not a number. A model that gets a type wrong should be told
288
+ so, not second-guessed.
289
+ - **Defaults are not applied.** In a JSON Schema, `default` tells the model what happens
290
+ if it leaves a field out. The handler decides what that means. The inferred type keeps
291
+ defaulted properties optional, so the handler cannot assume they are present.
292
+ - **Only plain JSON gets in.** Arguments are deep-copied, and anything that is not JSON
293
+ (dates, functions, class instances, cycles, `NaN`) is rejected. The handler, the audit
294
+ log and the confirmation guard each see a value that nobody else can mutate.
295
+ - **Bad schemas fail at startup.** JSON Schemas are compiled with Ajv in strict mode.
296
+ Unknown keywords, unknown formats and `required` properties that are never declared
297
+ are errors. A Zod type with no JSON equivalent (`z.date()`, `z.bigint()`) is refused
298
+ too, because the model could never send it.
299
+
300
+ Rejected arguments come back to the model as a tool result, with one JSON Pointer and
301
+ one message per problem (`/to: must match format "email"`). The model can then correct
302
+ itself, and the handler never sees the bad input.
303
+
304
+ ### Why the confirmation lives on the server, not in the prompt
305
+
306
+ "Ask the user before sending anything" in a system prompt is a request made to the
307
+ model, and the model is the very component the guard protects against. It can lose the
308
+ instruction in a long context. A prompt injection can override it. It can judge that
309
+ this case does not count. And nothing in the code enforces it.
310
+
311
+ The guard is code on the execution path. A tool whose declaration says
312
+ `sensitivity: 'high'` (or `'medium'` and irreversible) returns `confirmation_required`
313
+ instead of running, and the handler cannot be reached without a valid token. The
314
+ decision comes from the declaration, not from the model's reading of the situation: the
315
+ same tool is always confirmed, or never.
316
+
317
+ The token is built so that approving one thing cannot authorise another:
318
+
319
+ - **Bound to the call.** A token belongs to one principal, one tool and the exact
320
+ arguments. Key order does not matter, values do. Approving "email invoice INV-12"
321
+ cannot email INV-13, and a token presented by anyone else is refused.
322
+ - **Single-use.** Redeeming takes the record out of the store in one atomic step, and
323
+ that same step checks the expiry. Two concurrent redemptions cannot both run, and a
324
+ second presentation is recorded as a replay (`consumed`).
325
+ - **Short-lived, by class.** The lifetime can be a table per sensitivity and
326
+ reversibility: an irreversible critical action deserves a shorter window than a
327
+ reversible medium one, because past it the context of the decision is gone. The
328
+ expiry is computed when the confirmation is issued and stored with it; changing the
329
+ configuration later does not move confirmations already issued. Five minutes by
330
+ default.
331
+ - **Never stored.** The store keeps a SHA-256 hash. Whoever can read the store cannot
332
+ redeem anything.
333
+ - **Checked again.** At redemption the role is checked again and the arguments are
334
+ revalidated against the tool registered at that moment. The arguments that run are
335
+ the ones that were confirmed, frozen when the confirmation was issued.
336
+ - **Not for the model.** The token is meant for the host. The MCP server carries it in
337
+ the result's `_meta`, which MCP clients are not expected to pass to the model,
338
+ while the model only reads that confirmation is pending. The host redeems it after a
339
+ human decision: by repeating the call with the token, by calling `executeConfirmed()`
340
+ later, or through MCP elicitation when the client supports it. There is deliberately
341
+ no `confirm_action` tool. A model under prompt injection would simply call it, and the
342
+ guard would be reduced to a delay.
343
+
344
+ ### What the guard does not cover
345
+
346
+ The guard stops the model from running a sensitive call on its own. It does not make
347
+ the system safe by itself. Each limit below comes with what you should put in place
348
+ around it.
349
+
350
+ - **A compromised host or client.** Whoever controls the MCP client can attach a token
351
+ it was given and replay a decision: the guard protects against the model, not against
352
+ the host.
353
+ _Put in place:_ run the client in a component you control, authenticate it (the
354
+ `authenticate` hook of the upcoming HTTP transport), and keep the tokens it receives
355
+ in memory, out of logs and transcripts.
356
+ - **A host that shows the token to the model.** If a host copies `_meta`, or the whole
357
+ outcome, into the conversation, the model can confirm its own calls, and the guard is
358
+ gone.
359
+ _Put in place:_ strip `_meta` and confirmation outcomes before anything reaches the
360
+ model's context, and add a test that fails if `mtb_` (the token prefix) ever appears
361
+ in a transcript.
362
+ - **Misclassified tools.** Below the threshold, tools run directly. A tool declared
363
+ `low` that actually deletes data is not caught.
364
+ _Put in place:_ review declarations like permissions: print `registry.list()` as a
365
+ table (name, sensitivity, reversible, roles) in code review, and require a second
366
+ reviewer for any new or reclassified tool.
367
+ - **What the handler really does.** The guard confirms a call, not the behaviour of the
368
+ code behind it.
369
+ _Put in place:_ give each handler credentials scoped to the one effect its declaration
370
+ describes, so that it cannot do more even by mistake.
371
+ - **Information leaving through reads.** `sensitivity: 'none'` tools run freely. A
372
+ model can read data and repeat it elsewhere. Confirmation governs actions, not
373
+ information flow.
374
+ _Put in place:_ restrict reads of personal or confidential data by role, return only
375
+ the fields the task needs, and watch the audit log for unusual read volumes.
376
+ - **The quality of the human decision.** The guard makes sure someone confirmed. It
377
+ cannot make sure they read what they confirmed.
378
+ _Put in place:_ write `summarize` for the person who confirms (what happens, to whom,
379
+ with what consequence) and show it next to the arguments, never as a bare "Confirm?".
380
+ - **A world that changed.** Roles and arguments are checked again at redemption, but
381
+ the invoice may have been paid in the meantime.
382
+ _Put in place:_ have handlers check their preconditions when they run (invoice still
383
+ unpaid, slot still free) and throw a `ToolError` otherwise, and shorten `ttlMs` where
384
+ the context moves fast.
385
+ - **Timeouts.** A handler that ignores its abort signal keeps running after the bridge
386
+ has given up. Its result is discarded, but its effects are not undone.
387
+ _Put in place:_ pass `call.signal` to every I/O the handler starts (`fetch` and most
388
+ database drivers accept one) and make side effects idempotent, so that a cancelled
389
+ call either stops or can be retried safely.
390
+ - **Several processes.** The default store lives in one process. A token issued by one
391
+ instance cannot be redeemed on another.
392
+ _Put in place:_ before running a second instance, implement `ConfirmationStore` over
393
+ shared storage with an atomic `take` (Redis `GETDEL`, SQL `DELETE … RETURNING`).
394
+
395
+ ### Why the audit log records before running
396
+
397
+ Each call that runs leaves two events under one `callId`: `call.started`, then
398
+ `call.succeeded` or `call.failed`. A confirmed call is preceded by `confirmation.issued`
399
+ and possibly `confirmation.declined`, under the same id. Rejections are recorded with
400
+ their real reason (`forbidden`, `invalid_arguments`, an invalid or expired confirmation),
401
+ even when the model is told `unknown_tool`.
402
+
403
+ `call.started` is written before the handler runs, so the log knows about the call even
404
+ if the process dies during it. With `auditFailure: 'block'`, a call or a confirmation
405
+ that could not be recorded does not happen. The default, `continue`, warns on stderr and
406
+ proceeds. Events that follow an effect (`call.succeeded`, `call.failed`) can only be
407
+ reported: the effect has already happened.
408
+
409
+ What these events contain is the subject of [What the audit log keeps](#what-the-audit-log-keeps):
410
+ by default, metadata only. Messages of unexpected exceptions go to the log only, cut
411
+ to 500 characters; the model gets a generic failure. The default sink writes JSON
412
+ lines to stderr, because under stdio, stdout is the protocol channel.
413
+
414
+ ## Declaring tools: reference
415
+
416
+ | Field | Required | Meaning |
417
+ | ------------- | -------- | ------------------------------------------------------------------------------------------------------------------------------------------------- |
418
+ | `name` | yes | Unique in a registry. MCP allows `A-Z a-z 0-9 _ - .`, 1–128 characters. Some LLM APIs are stricter (no dots, 64 characters): prefer `snake_case`. |
419
+ | `description` | yes | What the model reads to decide whether and how to call the tool. |
420
+ | `args` | yes | `jsonSchema({...})` or `zodSchema(z.object({...}))`. The root must be an object. |
421
+ | `sensitivity` | yes | `none` (no side effect at all), `low`, `medium`, `high`, `critical`. |
422
+ | `reversible` | yes | Can the effect be undone? |
423
+ | `roles` | yes | Non-empty. A principal needs one of them. |
424
+ | `confirm` | no | `auto` (default: the bridge's policy decides) or `always`. There is no opt-out. |
425
+ | `summarize` | no | One sentence describing a specific call, shown to whoever confirms it. |
426
+ | `audit` | no | What the audit log may keep: `{ args?, result? }`, JSON Pointers (`*` allowed). Nothing but metadata by default. |
427
+ | `timeoutMs` | no | Per-call time limit. |
428
+ | `title` | no | Human-readable name. |
429
+
430
+ Behaviour hints sent to MCP clients are derived from the declaration. `readOnlyHint`
431
+ is true for `sensitivity: "none"`. `destructiveHint` is true for irreversible tools that
432
+ have an effect. Roles and sensitivity levels are never sent to the client.
433
+
434
+ Handlers throw `ToolError` for failures the model may read ("no invoice INV-12"). Any
435
+ other exception reaches the model as a generic failure; its details go to the audit log.
436
+
437
+ ## Roadmap
438
+
439
+ **v1.0**: declarations and registry, role-based access, validation (JSON Schema and
440
+ Zod), the confirmation guard with single-use tokens and MCP elicitation, the audit log,
441
+ the `envelope` adapter, the stdio transport, a three-tool example.
442
+
443
+ **v1.1**: `adapt()` and `importDefinitions()` (see
444
+ [Plugging in existing code](#plugging-in-existing-code)); confirmation lifetimes per
445
+ sensitivity and reversibility, an atomic take that reports why a token was refused;
446
+ an audit log that keeps no content unless a tool declares it (see
447
+ [What the audit log keeps](#what-the-audit-log-keeps)).
448
+
449
+ **Next**:
450
+
451
+ - HTTP transport: Streamable HTTP, with a per-request `authenticate` hook,
452
+ Origin/Host checks and sessions bound to the principal who opened them. The deprecated
453
+ HTTP+SSE transport will be available behind `legacySse: true`.
454
+
455
+ Out of scope for now: an OAuth authorization server, MCP resources and prompts, retries
456
+ and rate limiting.
457
+
458
+ ## Development
459
+
460
+ ```sh
461
+ npm ci
462
+ npm run lint && npm run format:check && npm run typecheck
463
+ npm test
464
+ npm run build && npm run check:dist
465
+ npm run check:mutations
466
+ ```
467
+
468
+ The tests make no network calls.
469
+
470
+ The safety guards are checked by **targeted mutation testing**: `npm run
471
+ check:mutations` removes each guard in turn and fails if the test suite still passes.
472
+ Eighteen guards are covered, among them:
473
+
474
+ - the role is checked again at call time;
475
+ - a confirmation token is single-use;
476
+ - a token only runs the exact arguments it was issued for;
477
+ - the role is checked again when a confirmation is redeemed;
478
+ - the token never appears in what the model reads;
479
+ - invalid arguments never reach a handler;
480
+ - an imported tool without governance never starts;
481
+ - the store checks the expiry in the same step as the take, and the bridge checks it
482
+ again from the expiry stored at issue time;
483
+ - the audit keeps no argument or result content unless the tool declares it, and no
484
+ confirmation summary;
485
+ - every refused token leaves its reason in the audit, and the model never learns it.
486
+
487
+ CI runs this on every push. See [CONTRIBUTING](./CONTRIBUTING.md#mutation-checks) for
488
+ the mutants and the tests that catch them.
489
+
490
+ ## License
491
+
492
+ [MIT](./LICENSE)
@@ -0,0 +1,111 @@
1
+ "use strict";Object.defineProperty(exports, "__esModule", {value: true}); function _optionalChain(ops) { let lastAccessLHS = undefined; let value = ops[0]; let i = 1; while (i < ops.length) { const op = ops[i]; const fn = ops[i + 1]; i += 2; if ((op === 'optionalAccess' || op === 'optionalCall') && value == null) { return undefined; } if (op === 'access' || op === 'optionalAccess') { lastAccessLHS = value; value = fn(value); } else if (op === 'call' || op === 'optionalCall') { value = fn((...args) => value.call(lastAccessLHS, ...args)); lastAccessLHS = undefined; } } return value; } var _class; var _class2;// src/errors.ts
2
+ var ToolDefinitionError = (_class = class extends Error {
3
+ __init() {this.name = "ToolDefinitionError"}
4
+
5
+ /** The offending tool, when its name is known. */
6
+
7
+ constructor(code, message, options) {
8
+ super(message, _optionalChain([options, 'optionalAccess', _ => _.cause]) === void 0 ? void 0 : { cause: options.cause });_class.prototype.__init.call(this);;
9
+ this.code = code;
10
+ this.tool = _optionalChain([options, 'optionalAccess', _2 => _2.tool]);
11
+ }
12
+ }, _class);
13
+ var ToolError = (_class2 = class extends Error {constructor(...args) { super(...args); _class2.prototype.__init2.call(this); }
14
+ __init2() {this.name = "ToolError"}
15
+ }, _class2);
16
+
17
+ // src/json.ts
18
+ var NOT_JSON = /* @__PURE__ */ Symbol("not-json");
19
+ function cloneJson(value) {
20
+ const copy = cloneValue(value, /* @__PURE__ */ new Set());
21
+ return copy === NOT_JSON ? { ok: false } : { ok: true, value: copy };
22
+ }
23
+ function cloneValue(value, ancestors) {
24
+ if (value === null || typeof value === "string" || typeof value === "boolean") return value;
25
+ if (typeof value === "number") return Number.isFinite(value) ? value : NOT_JSON;
26
+ if (typeof value !== "object") return NOT_JSON;
27
+ if (ancestors.has(value)) return NOT_JSON;
28
+ if (Array.isArray(value)) {
29
+ ancestors.add(value);
30
+ const items = [];
31
+ for (const item of value) {
32
+ const copy = cloneValue(item, ancestors);
33
+ if (copy === NOT_JSON) return NOT_JSON;
34
+ items.push(copy);
35
+ }
36
+ ancestors.delete(value);
37
+ return items;
38
+ }
39
+ const proto = Object.getPrototypeOf(value);
40
+ if (proto !== Object.prototype && proto !== null) return NOT_JSON;
41
+ ancestors.add(value);
42
+ const out = {};
43
+ for (const [key, item] of Object.entries(value)) {
44
+ if (item === void 0) continue;
45
+ const copy = cloneValue(item, ancestors);
46
+ if (copy === NOT_JSON) return NOT_JSON;
47
+ out[key] = copy;
48
+ }
49
+ ancestors.delete(value);
50
+ return out;
51
+ }
52
+ function isJsonArray(value) {
53
+ return Array.isArray(value);
54
+ }
55
+ function isJsonObject(value) {
56
+ return typeof value === "object" && value !== null && !Array.isArray(value);
57
+ }
58
+ function deepFreeze(value) {
59
+ if (typeof value === "object" && value !== null) {
60
+ for (const item of Object.values(value)) deepFreeze(item);
61
+ Object.freeze(value);
62
+ }
63
+ return value;
64
+ }
65
+ function pointerSegment(segment) {
66
+ return segment.replaceAll("~", "~0").replaceAll("/", "~1");
67
+ }
68
+ function readProperty(source, key) {
69
+ return Reflect.get(source, key);
70
+ }
71
+
72
+ // src/schema/types.ts
73
+ var MAX_ISSUES = 20;
74
+ var NOT_JSON_ISSUE = Object.freeze({
75
+ path: "",
76
+ message: "must be plain JSON data"
77
+ });
78
+ function freezeRootSchema(schema, tool) {
79
+ const copy = cloneJson(schema);
80
+ if (!copy.ok || !isJsonObject(copy.value)) {
81
+ throw new ToolDefinitionError("invalid_schema", "an argument schema must be plain JSON data", {
82
+ tool
83
+ });
84
+ }
85
+ const root = copy.value;
86
+ if (!isRootObjectSchema(root)) {
87
+ throw new ToolDefinitionError(
88
+ "invalid_schema",
89
+ 'MCP requires the root of an argument schema to be `type: "object"`',
90
+ { tool }
91
+ );
92
+ }
93
+ return deepFreeze(root);
94
+ }
95
+ function isRootObjectSchema(schema) {
96
+ return schema.type === "object";
97
+ }
98
+
99
+
100
+
101
+
102
+
103
+
104
+
105
+
106
+
107
+
108
+
109
+
110
+
111
+ exports.ToolDefinitionError = ToolDefinitionError; exports.ToolError = ToolError; exports.cloneJson = cloneJson; exports.isJsonArray = isJsonArray; exports.isJsonObject = isJsonObject; exports.deepFreeze = deepFreeze; exports.pointerSegment = pointerSegment; exports.readProperty = readProperty; exports.MAX_ISSUES = MAX_ISSUES; exports.NOT_JSON_ISSUE = NOT_JSON_ISSUE; exports.freezeRootSchema = freezeRootSchema;