@nexusbloom/mcp-server 2.0.2 → 2.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -42,6 +42,7 @@ All variables are optional.
42
42
  | `NEXUSBLOOM_MCP_TIMEOUT_MS` | `15000` | Per-request timeout. |
43
43
  | `NEXUSBLOOM_MCP_CACHE_TTL_MS` | `60000` | Tool-list cache lifetime. `0` disables caching. |
44
44
  | `NEXUSBLOOM_MCP_DEBUG` | *(off)* | `1` logs requests to stderr. |
45
+ | `NEXUSBLOOM_MCP_EXECUTION` | `remote` | `local` or `auto` to execute tool source on this machine. **Off by default on purpose** — see below. |
45
46
 
46
47
  Diagnostics always go to **stderr**; stdout is the MCP transport and writing
47
48
  anything else there corrupts the protocol.
@@ -51,6 +52,16 @@ anything else there corrupts the protocol.
51
52
  Every published tool is advertised as an MCP tool **named by its slug**, with its
52
53
  real JSON Schema — so it can be called directly:
53
54
 
55
+ > **One caveat on slugs.** A slug has to match MCP's tool-name grammar
56
+ > (`[A-Za-z0-9_-]{1,64}`) to be advertised. A catalogue slug that does not — a
57
+ > space, a slash, non-ASCII, or more than 64 characters — is *withheld from the
58
+ > tool list* and reported once on stderr, because a strict host validates every
59
+ > name in that array and rejects the whole response over one bad entry. That
60
+ > would cost an agent every tool rather than one. A withheld tool is still
61
+ > reachable through `{"command":"run"}` and `nexusbloom://catalogue`; it is only
62
+ > not directly callable. `npm run catalogue-check` verifies a catalogue against
63
+ > the grammar.
64
+
54
65
  ```jsonc
55
66
  { "name": "env-validator", "arguments": { "env_content": "DEBUG=true" } }
56
67
  ```
@@ -64,6 +75,164 @@ which tool it wants:
64
75
  | `{"command":"list"}` | Every published tool, one line each. |
65
76
  | `{"command":"schema","slug":"…"}` | Exact parameters, plus a ready-to-send example invocation. |
66
77
  | `{"command":"run","slug":"…","params":{…}}` | Execute a tool. |
78
+ | `{"command":"batch","runs":[{"slug":"…","params":{…}},…]}` | Execute up to 10 tools in one call. |
79
+ | `{"command":"history"}` | What this **server process** has run, newest first. |
80
+ | `{"command":"history","show":"<id>"}` | One run in full: input, result, timing. |
81
+ | `{"command":"diff","from":"<id>","to":"<id>"}` | Compare two runs field by field. |
82
+
83
+ ### Batching
84
+
85
+ Several tools belonging to one task go in a single turn:
86
+
87
+ ```jsonc
88
+ { "command": "batch", "runs": [
89
+ { "slug": "env-validator", "params": { "env_content": "DEBUG=true" } },
90
+ { "slug": "data-generator", "params": { "count": 2 } }
91
+ ] }
92
+ ```
93
+
94
+ Two behaviours make it safe rather than merely fast:
95
+
96
+ - **Validate everything first.** Every slug is resolved and every schema checked
97
+ before the first run. A typo in item four costs zero quota and rejects the
98
+ whole batch — a batch is one intent, so half of it is not a useful answer.
99
+ - **Then isolate failures.** Past that point each run is independent. One tool
100
+ erroring does not discard the seven that worked; the response reports the
101
+ tally, lists failures first, and the structured block keeps every payload.
102
+
103
+ Capped at 10 runs, because anonymous execution allows 30 requests/minute and one
104
+ turn should not spend the budget the next ten calls need.
105
+
106
+ ### Progress notifications
107
+
108
+ If the host sends a `progressToken` in `_meta`, the server emits
109
+ `notifications/progress` while a batch runs — one tick per entry, plus an opening
110
+ tick so the host is never silent. Progress is monotonic, and a failed
111
+ notification never fails the call. A host that sends no token receives none:
112
+ unsolicited progress is a protocol violation, not a nicety.
113
+
114
+ ### Resources — read the catalogue without spending a call
115
+
116
+ Hosts that support MCP resources can pull the catalogue directly, which is how
117
+ an agent holds the whole surface in context instead of discovering it one slug
118
+ at a time:
119
+
120
+ | URI | Contents |
121
+ |---|---|
122
+ | `nexusbloom://guide` | This usage guide as markdown: meta commands, recovery paths, limits. |
123
+ | `nexusbloom://catalogue` | Every tool as JSON — slug, description, category, tags, parameter counts, `has_schema`, and the URI of its full manifest. |
124
+ | `nexusbloom://tools/{slug}` | One tool's full manifest: input **and** output JSON Schema, plus `call: { tool, arguments }` — a ready-to-send example. |
125
+
126
+ The catalogue deliberately omits full schemas; that is what the per-tool URI is
127
+ for. A 31-manifest inline blob would exhaust the context budget before the agent
128
+ had chosen anything.
129
+
130
+ Slugs in resource URIs resolve by exact match or unambiguous abbreviation only,
131
+ identically to tool calls, and an unknown slug produces a protocol error naming
132
+ the close matches rather than an empty body an agent might quote as fact.
133
+
134
+ ### Prompts — one-click starts a host can offer
135
+
136
+ Four prompts, each assembled against the live catalogue rather than canned, so
137
+ the text a model receives names tools that actually exist and carries calls that
138
+ already validate:
139
+
140
+ | Prompt | Arguments | Gives the model |
141
+ |---|---|---|
142
+ | `find-tool` | `goal` | Tools matching the goal, ranked, with each one's parameter cost and the manifest URI. |
143
+ | `use-tool` | `slug` | One tool's exact parameters, its `output_schema`, and a valid call to copy. |
144
+ | `plan-batch` | `task` | Likely participants, the batch call shape, the 10-run cap, and the whole-batch validation rule. |
145
+ | `recover` | `error`, `slug` | What each error code means, what to do about it, and the failed tool's real call shape. |
146
+
147
+ `recover` exists because the common failure is not "the tool broke" — it is an
148
+ agent retrying a validation failure unchanged, or guessing slugs. The prompt
149
+ states the codes, marks which are retryable, and embeds the required fields of the
150
+ tool that just failed.
151
+
152
+ `use-tool` resolves slugs **exactly**, unlike tool calls. A prompt naming a tool
153
+ is a deliberate user choice; silently running a different tool because a prefix
154
+ happened to match would be the wrong answer to a request the user made explicitly.
155
+
156
+
157
+ ### Local execution (opt-in, and read this first)
158
+
159
+ `NEXUSBLOOM_MCP_EXECUTION` chooses where tool code runs:
160
+
161
+ | Mode | Behaviour |
162
+ |---|---|
163
+ | `remote` *(default)* | Always call the API. No tool source is ever fetched. |
164
+ | `local` | Always execute the published source in a child process. A tool with no source is an error. |
165
+ | `auto` | Execute locally when source exists, otherwise fall back to the API. |
166
+
167
+ An unrecognised value falls back to `remote`, so a typo cannot silently switch on
168
+ code execution.
169
+
170
+ **The threat model.** Executing a tool locally means executing code that arrived
171
+ over the network. The sandbox from `@nexusbloom/core` gives the tool its own
172
+ process, a SIGKILL deadline, a heap cap, and no access to this process's stdout.
173
+ It is **isolation, not a security sandbox**: that code still runs with your user
174
+ permissions and can read files, open sockets, and spawn processes.
175
+
176
+ That is why it is off by default. An MCP server is driven by whatever the model
177
+ decides to ask for, so the default posture has to be the one where a surprising
178
+ tool call cannot become arbitrary code execution on a developer's machine. Turn it
179
+ on when you trust the catalogue — a self-hosted deployment publishing only your
180
+ own tools — or to keep working while the API is unreachable (`auto`).
181
+
182
+ Enabling it prints the same warning to stderr at startup, once, because the
183
+ threat model belongs on the record at the moment it is switched on.
184
+
185
+ **Credentials are stripped from the child.** The sandbox forks the tool with the
186
+ parent's environment minus anything whose name looks like a credential —
187
+ `NEXUSBLOOM_API_KEY`, `STRIPE_SECRET_KEY`, `AWS_SESSION_TOKEN`, `*_PASSWORD`,
188
+ `*_AUTH`, and so on, matched by substring. Tool code has no legitimate reason to
189
+ read your API key, and without this a published tool could simply read it and
190
+ exfiltrate it — which turns "runs with your permissions" into "runs with your
191
+ account". Ordinary variables (`PATH`, project config, anything you have set) pass
192
+ through untouched, so tools that read their own configuration still work.
193
+
194
+ ### Run history and diff
195
+
196
+ Every run this server performs is recorded, so an agent can see what already
197
+ happened instead of guessing:
198
+
199
+ - `{"command":"history"}` — an index: run id, tool, status, duration, newest first.
200
+ - `{"command":"history","show":"r3"}` — one run with its input and its result.
201
+ - `{"command":"diff","from":"-2","to":"-1"}` — two runs compared field by field. `-1`
202
+ is the most recent run, `-2` the one before it, so comparing consecutive runs
203
+ needs no lookup.
204
+
205
+ A diff leads with the verdict and flags changed inputs, because a result that
206
+ moved because its *input* moved is not a regression — conflating the two is how a
207
+ diff gets dismissed as noise and then ignored when it mattered.
208
+
209
+ Three deliberate constraints:
210
+
211
+ - **In memory, never on disk.** Run payloads are user data and frequently
212
+ secrets. A log on disk would be an unencrypted store nobody asked for, and
213
+ losing it on restart costs nothing the tool call did not.
214
+ - **Scoped to the process, not the conversation.** Most hosts keep the stdio
215
+ process alive across turns, so history spans every conversation turn the
216
+ process has served — it is not reset between them. Every history response says
217
+ so, and carries the process start time, because an agent that finds a run it
218
+ cannot account for needs somewhere to stop guessing. Observed in a real
219
+ session: an agent correctly reported that an unexplained earlier run had
220
+ different input, then invented a story about who had changed it and when,
221
+ and escalated it into a security incident. The facts were right; the cause
222
+ was fiction. Naming the boundary is what makes "this predates me" a
223
+ checkable statement.
224
+ - **Bounded.** 50 runs, each payload capped at 8,000 characters. Truncation is
225
+ recorded, and a diff says so — a truncated payload reporting "no differences"
226
+ would be the worst possible failure mode here.
227
+ - **Inputs redacted.** Any key matching `key`, `secret`, `token`, `password`,
228
+ `auth`, `cookie` or `session` — plus `passwd`, `passphrase` and `credential` —
229
+ is masked before storage, at any depth up to 6 levels, without preserving
230
+ length. Beyond that depth the value is replaced with `[deep]` rather than
231
+ inspected, so a deeply nested payload cannot smuggle a credential past the
232
+ walk.
233
+
234
+ The index is also readable as `nexusbloom://history`, for hosts that would rather
235
+ pull it than call for it.
67
236
 
68
237
  ### Errors are actionable
69
238
 
package/index.js CHANGED
@@ -24,6 +24,7 @@ import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js"
24
24
  import { ApiClient } from "./src/client.js";
25
25
  import { candidateRoots, loadConfig } from "./src/config.js";
26
26
  import { checkConnectivity } from "./src/handlers.js";
27
+ import { localExecutionNotice } from "./src/local.js";
27
28
  import { ManifestCache } from "./src/manifests.js";
28
29
  import { createServer } from "./src/server.js";
29
30
 
@@ -50,6 +51,12 @@ export async function main(env = process.env, deps = {}) {
50
51
  const connectivity = await checkConnectivity(app.client, candidateRoots(env), config, deps);
51
52
  process.stderr.write(`${connectivity.report}\n`);
52
53
 
54
+ // Printed once, before serving, and only when the operator opted in. The
55
+ // threat model belongs on the record at the moment it is enabled, not in a
56
+ // README nobody re-reads when a host later behaves strangely.
57
+ const notice = localExecutionNotice(config.execution);
58
+ if (notice) process.stderr.write(notice);
59
+
53
60
  const transport = new StdioServerTransport();
54
61
 
55
62
  // Close on the signals a host actually sends. Without this the process
@@ -66,7 +73,12 @@ export async function main(env = process.env, deps = {}) {
66
73
  process.once("SIGTERM", shutdown);
67
74
 
68
75
  await app.server.connect(transport);
69
- process.stderr.write("NexusBloom MCP server running on stdio\n");
76
+ // Name the capabilities explicitly: a user debugging a host that shows no
77
+ // tools is usually looking at whether the host negotiated them, and a
78
+ // capability list on stderr answers that in one line.
79
+ process.stderr.write(
80
+ "NexusBloom MCP server running on stdio (capabilities: tools, resources, prompts)\n",
81
+ );
70
82
 
71
83
  return { ...app, transport, close: shutdown };
72
84
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@nexusbloom/mcp-server",
3
- "version": "2.0.2",
3
+ "version": "2.1.1",
4
4
  "description": "MCP server for NexusBloom — agents discover tools by intent, read exact schemas, and execute them. Built on @nexusbloom/core.",
5
5
  "type": "module",
6
6
  "main": "index.js",
@@ -34,10 +34,13 @@
34
34
  "test": "node --test --import ./test/setup.mjs test/*.test.js",
35
35
  "test:coverage": "node --test --experimental-test-coverage --import ./test/setup.mjs test/*.test.js",
36
36
  "test:watch": "node --test --watch --import ./test/setup.mjs test/*.test.js",
37
- "test:src-only": "node --test --import ./test/setup.mjs test/config.test.js test/errors.test.js test/manifests.test.js test/discovery.test.js test/client.test.js test/validate.test.js test/render.test.js test/cache.test.js test/handlers.test.js",
37
+ "test:src-only": "node --test --import ./test/setup.mjs test/config.test.js test/errors.test.js test/manifests.test.js test/discovery.test.js test/client.test.js test/validate.test.js test/render.test.js test/cache.test.js test/handlers.test.js test/resources.test.js test/batch.test.js test/progress.test.js test/prompts.test.js test/history.test.js test/local.test.js",
38
38
  "shell": "node scripts/mcp-shell.mjs",
39
39
  "shell:mock": "node scripts/mcp-shell.mjs --mock",
40
40
  "mock-api": "node scripts/mock-api.mjs",
41
+ "agent-demo": "node scripts/agent-demo.mjs",
42
+ "agent-demo:mock": "node scripts/agent-demo.mjs --url http://127.0.0.1:8787/api",
43
+ "catalogue-check": "node scripts/full-catalogue-check.mjs",
41
44
  "lint": "node --check index.js && for f in src/*.js; do node --check \"$f\" || exit 1; done"
42
45
  }
43
46
  }
package/src/config.js CHANGED
@@ -1,3 +1,5 @@
1
+ import { resolveExecutionMode } from "./local.js";
2
+
1
3
  /**
2
4
  * Configuration — resolved once, injected everywhere.
3
5
  *
@@ -108,6 +110,9 @@ export function loadConfig(env = process.env) {
108
110
  return {
109
111
  apiKey,
110
112
  apiBase: normaliseApiBase(env.NEXUSBLOOM_API_URL),
113
+ // Where tool code runs: "remote" (default), "local", or "auto". See local.js
114
+ // for why this is not a preference but a trust decision.
115
+ execution: resolveExecutionMode(env.NEXUSBLOOM_MCP_EXECUTION),
111
116
  timeoutMs,
112
117
  cacheTtlMs,
113
118
  debug,