@nexusbloom/mcp-server 2.0.0 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -42,6 +42,7 @@ All variables are optional.
42
42
  | `NEXUSBLOOM_MCP_TIMEOUT_MS` | `15000` | Per-request timeout. |
43
43
  | `NEXUSBLOOM_MCP_CACHE_TTL_MS` | `60000` | Tool-list cache lifetime. `0` disables caching. |
44
44
  | `NEXUSBLOOM_MCP_DEBUG` | *(off)* | `1` logs requests to stderr. |
45
+ | `NEXUSBLOOM_MCP_EXECUTION` | `remote` | `local` or `auto` to execute tool source on this machine. **Off by default on purpose** — see below. |
45
46
 
46
47
  Diagnostics always go to **stderr**; stdout is the MCP transport and writing
47
48
  anything else there corrupts the protocol.
@@ -64,6 +65,154 @@ which tool it wants:
64
65
  | `{"command":"list"}` | Every published tool, one line each. |
65
66
  | `{"command":"schema","slug":"…"}` | Exact parameters, plus a ready-to-send example invocation. |
66
67
  | `{"command":"run","slug":"…","params":{…}}` | Execute a tool. |
68
+ | `{"command":"batch","runs":[{"slug":"…","params":{…}},…]}` | Execute up to 10 tools in one call. |
69
+ | `{"command":"history"}` | What this session has run, newest first. |
70
+ | `{"command":"history","show":"<id>"}` | One run in full: input, result, timing. |
71
+ | `{"command":"diff","from":"<id>","to":"<id>"}` | Compare two runs field by field. |
72
+
73
+ ### Batching
74
+
75
+ Several tools belonging to one task go in a single turn:
76
+
77
+ ```jsonc
78
+ { "command": "batch", "runs": [
79
+ { "slug": "env-validator", "params": { "env_content": "DEBUG=true" } },
80
+ { "slug": "data-generator", "params": { "count": 2 } }
81
+ ] }
82
+ ```
83
+
84
+ Two behaviours make it safe rather than merely fast:
85
+
86
+ - **Validate everything first.** Every slug is resolved and every schema checked
87
+ before the first run. A typo in item four costs zero quota and rejects the
88
+ whole batch — a batch is one intent, so half of it is not a useful answer.
89
+ - **Then isolate failures.** Past that point each run is independent. One tool
90
+ erroring does not discard the seven that worked; the response reports the
91
+ tally, lists failures first, and the structured block keeps every payload.
92
+
93
+ Capped at 10 runs, because anonymous execution allows 30 requests/minute and one
94
+ turn should not spend the budget the next ten calls need.
95
+
96
+ ### Progress notifications
97
+
98
+ If the host sends a `progressToken` in `_meta`, the server emits
99
+ `notifications/progress` while a batch runs — one tick per entry, plus an opening
100
+ tick so the host is never silent. Progress is monotonic, and a failed
101
+ notification never fails the call. A host that sends no token receives none:
102
+ unsolicited progress is a protocol violation, not a nicety.
103
+
104
+ ### Resources — read the catalogue without spending a call
105
+
106
+ Hosts that support MCP resources can pull the catalogue directly, which is how
107
+ an agent holds the whole surface in context instead of discovering it one slug
108
+ at a time:
109
+
110
+ | URI | Contents |
111
+ |---|---|
112
+ | `nexusbloom://guide` | This usage guide as markdown: meta commands, recovery paths, limits. |
113
+ | `nexusbloom://catalogue` | Every tool as JSON — slug, description, category, tags, parameter counts, `has_schema`, and the URI of its full manifest. |
114
+ | `nexusbloom://tools/{slug}` | One tool's full manifest: input **and** output JSON Schema, plus `call: { tool, arguments }` — a ready-to-send example. |
115
+
116
+ The catalogue deliberately omits full schemas; that is what the per-tool URI is
117
+ for. A 31-manifest inline blob would exhaust the context budget before the agent
118
+ had chosen anything.
119
+
120
+ Slugs in resource URIs resolve by exact match or unambiguous abbreviation only,
121
+ identically to tool calls, and an unknown slug produces a protocol error naming
122
+ the close matches rather than an empty body an agent might quote as fact.
123
+
124
+ ### Prompts — one-click starts a host can offer
125
+
126
+ Four prompts, each assembled against the live catalogue rather than canned, so
127
+ the text a model receives names tools that actually exist and carries calls that
128
+ already validate:
129
+
130
+ | Prompt | Arguments | Gives the model |
131
+ |---|---|---|
132
+ | `find-tool` | `goal` | Tools matching the goal, ranked, with each one's parameter cost and the manifest URI. |
133
+ | `use-tool` | `slug` | One tool's exact parameters, its `output_schema`, and a valid call to copy. |
134
+ | `plan-batch` | `task` | Likely participants, the batch call shape, the 10-run cap, and the whole-batch validation rule. |
135
+ | `recover` | `error`, `slug` | What each error code means, what to do about it, and the failed tool's real call shape. |
136
+
137
+ `recover` exists because the common failure is not "the tool broke" — it is an
138
+ agent retrying a validation failure unchanged, or guessing slugs. The prompt
139
+ states the codes, marks which are retryable, and embeds the required fields of the
140
+ tool that just failed.
141
+
142
+ `use-tool` resolves slugs **exactly**, unlike tool calls. A prompt naming a tool
143
+ is a deliberate user choice; silently running a different tool because a prefix
144
+ happened to match would be the wrong answer to a request the user made explicitly.
145
+
146
+
147
+ ### Local execution (opt-in, and read this first)
148
+
149
+ `NEXUSBLOOM_MCP_EXECUTION` chooses where tool code runs:
150
+
151
+ | Mode | Behaviour |
152
+ |---|---|
153
+ | `remote` *(default)* | Always call the API. No tool source is ever fetched. |
154
+ | `local` | Always execute the published source in a child process. A tool with no source is an error. |
155
+ | `auto` | Execute locally when source exists, otherwise fall back to the API. |
156
+
157
+ An unrecognised value falls back to `remote`, so a typo cannot silently switch on
158
+ code execution.
159
+
160
+ **The threat model.** Executing a tool locally means executing code that arrived
161
+ over the network. The sandbox from `@nexusbloom/core` gives the tool its own
162
+ process, a SIGKILL deadline, a heap cap, and no access to this process's stdout.
163
+ It is **isolation, not a security sandbox**: that code still runs with your user
164
+ permissions and can read files, open sockets, and spawn processes.
165
+
166
+ That is why it is off by default. An MCP server is driven by whatever the model
167
+ decides to ask for, so the default posture has to be the one where a surprising
168
+ tool call cannot become arbitrary code execution on a developer's machine. Turn it
169
+ on when you trust the catalogue — a self-hosted deployment publishing only your
170
+ own tools — or to keep working while the API is unreachable (`auto`).
171
+
172
+ Enabling it prints the same warning to stderr at startup, once, because the
173
+ threat model belongs on the record at the moment it is switched on.
174
+
175
+ **Credentials are stripped from the child.** The sandbox forks the tool with the
176
+ parent's environment minus anything whose name looks like a credential —
177
+ `NEXUSBLOOM_API_KEY`, `STRIPE_SECRET_KEY`, `AWS_SESSION_TOKEN`, `*_PASSWORD`,
178
+ `*_AUTH`, and so on, matched by substring. Tool code has no legitimate reason to
179
+ read your API key, and without this a published tool could simply read it and
180
+ exfiltrate it — which turns "runs with your permissions" into "runs with your
181
+ account". Ordinary variables (`PATH`, project config, anything you have set) pass
182
+ through untouched, so tools that read their own configuration still work.
183
+
184
+ ### Run history and diff
185
+
186
+ Every run this server performs is recorded, so an agent can see what already
187
+ happened instead of guessing:
188
+
189
+ - `{"command":"history"}` — an index: run id, tool, status, duration, newest first.
190
+ - `{"command":"history","show":"r3"}` — one run with its input and its result.
191
+ - `{"command":"diff","from":"-2","to":"-1"}` — two runs compared field by field. `-1`
192
+ is the most recent run, `-2` the one before it, so comparing consecutive runs
193
+ needs no lookup.
194
+
195
+ A diff leads with the verdict and flags changed inputs, because a result that
196
+ moved because its *input* moved is not a regression — conflating the two is how a
197
+ diff gets dismissed as noise and then ignored when it mattered.
198
+
199
+ Three deliberate constraints:
200
+
201
+ - **In memory, never on disk.** Run payloads are user data and frequently
202
+ secrets. A log on disk would be an unencrypted store nobody asked for, and
203
+ losing it on restart costs nothing the tool call did not.
204
+ - **Bounded.** 50 runs, each payload capped at 8,000 characters. Truncation is
205
+ recorded, and a diff says so — a truncated payload reporting "no differences"
206
+ would be the worst possible failure mode here.
207
+ - **Inputs redacted.** Any key matching `key`, `secret`, `token`, `password`,
208
+ `auth`, `cookie` or `session` — plus `passwd`, `passphrase` and `credential` —
209
+ is masked before storage, at any depth up to 6 levels, without preserving
210
+ length. Beyond that depth the value is replaced with `[deep]` rather than
211
+ inspected, so a deeply nested payload cannot smuggle a credential past the
212
+ walk.
213
+
214
+ The index is also readable as `nexusbloom://history`, for hosts that would rather
215
+ pull it than call for it.
67
216
 
68
217
  ### Errors are actionable
69
218
 
@@ -107,19 +256,4 @@ server that explains itself on each request, not a process that exits or a host
107
256
  that concludes the server is broken. A failed catalogue refresh keeps serving the
108
257
  last good list rather than emptying it.
109
258
 
110
- ## Development
111
-
112
- ```bash
113
- npm test # unit + integration, no network
114
- npm run test:coverage # with coverage
115
- npm run lint
116
- ```
117
-
118
- The suite has **417 tests** at **100% line coverage** of `src/`. It includes
119
- end-to-end tests that drive the real server as a child process over stdio, using
120
- scripted miniature agents that discover, choose, read a schema and execute — the
121
- same loop a model runs, asserted on whether the *task* succeeded.
122
259
 
123
- No test touches the network: `NEXUSBLOOM_API_URL` is pointed at an unroutable
124
- `.invalid` host and every HTTP client is injected, so the suite cannot pass
125
- while production is broken, and cannot fail because production is down.
package/index.js CHANGED
@@ -24,6 +24,7 @@ import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js"
24
24
  import { ApiClient } from "./src/client.js";
25
25
  import { candidateRoots, loadConfig } from "./src/config.js";
26
26
  import { checkConnectivity } from "./src/handlers.js";
27
+ import { localExecutionNotice } from "./src/local.js";
27
28
  import { ManifestCache } from "./src/manifests.js";
28
29
  import { createServer } from "./src/server.js";
29
30
 
@@ -50,6 +51,12 @@ export async function main(env = process.env, deps = {}) {
50
51
  const connectivity = await checkConnectivity(app.client, candidateRoots(env), config, deps);
51
52
  process.stderr.write(`${connectivity.report}\n`);
52
53
 
54
+ // Printed once, before serving, and only when the operator opted in. The
55
+ // threat model belongs on the record at the moment it is enabled, not in a
56
+ // README nobody re-reads when a host later behaves strangely.
57
+ const notice = localExecutionNotice(config.execution);
58
+ if (notice) process.stderr.write(notice);
59
+
53
60
  const transport = new StdioServerTransport();
54
61
 
55
62
  // Close on the signals a host actually sends. Without this the process
@@ -66,7 +73,12 @@ export async function main(env = process.env, deps = {}) {
66
73
  process.once("SIGTERM", shutdown);
67
74
 
68
75
  await app.server.connect(transport);
69
- process.stderr.write("NexusBloom MCP server running on stdio\n");
76
+ // Name the capabilities explicitly: a user debugging a host that shows no
77
+ // tools is usually looking at whether the host negotiated them, and a
78
+ // capability list on stderr answers that in one line.
79
+ process.stderr.write(
80
+ "NexusBloom MCP server running on stdio (capabilities: tools, resources, prompts)\n",
81
+ );
70
82
 
71
83
  return { ...app, transport, close: shutdown };
72
84
  }
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@nexusbloom/mcp-server",
3
- "version": "2.0.0",
4
- "description": "MCP server for NexusBloom — agents discover tools by intent, read exact schemas, and execute them. Built on @nexusbloom/core.",
3
+ "version": "2.1.0",
4
+ "description": "MCP server for NexusBloom \u2014 agents discover tools by intent, read exact schemas, and execute them. Built on @nexusbloom/core.",
5
5
  "type": "module",
6
6
  "main": "index.js",
7
7
  "bin": {
@@ -34,7 +34,12 @@
34
34
  "test": "node --test --import ./test/setup.mjs test/*.test.js",
35
35
  "test:coverage": "node --test --experimental-test-coverage --import ./test/setup.mjs test/*.test.js",
36
36
  "test:watch": "node --test --watch --import ./test/setup.mjs test/*.test.js",
37
- "test:src-only": "node --test --import ./test/setup.mjs test/config.test.js test/errors.test.js test/manifests.test.js test/discovery.test.js test/client.test.js test/validate.test.js test/render.test.js test/cache.test.js test/handlers.test.js",
37
+ "test:src-only": "node --test --import ./test/setup.mjs test/config.test.js test/errors.test.js test/manifests.test.js test/discovery.test.js test/client.test.js test/validate.test.js test/render.test.js test/cache.test.js test/handlers.test.js test/resources.test.js test/batch.test.js test/progress.test.js test/prompts.test.js test/history.test.js test/local.test.js",
38
+ "shell": "node scripts/mcp-shell.mjs",
39
+ "shell:mock": "node scripts/mcp-shell.mjs --mock",
40
+ "mock-api": "node scripts/mock-api.mjs",
41
+ "agent-demo": "node scripts/agent-demo.mjs",
42
+ "agent-demo:mock": "node scripts/agent-demo.mjs --url http://127.0.0.1:8787/api",
38
43
  "lint": "node --check index.js && for f in src/*.js; do node --check \"$f\" || exit 1; done"
39
44
  }
40
45
  }
package/src/client.js CHANGED
@@ -41,6 +41,15 @@ export class ApiClient {
41
41
  // not fire (some fetch polyfills). Whichever rejects first wins; the loser
42
42
  // is an unhandled rejection we must not let crash the process.
43
43
  this._inFlight = new Set();
44
+
45
+ /**
46
+ * Most recent rate-limit state reported by the API, as
47
+ * `{remaining, limit, reset}`. Anonymous execution is capped per minute, so
48
+ * an agent benefits from seeing its own budget rather than discovering it
49
+ * as a 429.
50
+ * @type {{remaining: number|null, limit: number|null, reset: string|null}|null}
51
+ */
52
+ this.rateLimit = null;
44
53
  }
45
54
 
46
55
  /**
@@ -61,12 +70,13 @@ export class ApiClient {
61
70
  const headers = { Accept: "application/json" };
62
71
  if (body !== undefined) headers["Content-Type"] = "application/json";
63
72
  if (auth && this.config.apiKey) headers["Authorization"] = `Bearer ${this.config.apiKey}`;
73
+ Object.assign(headers, opts.headers || {});
64
74
 
65
75
  // Own controller so an external signal can cancel too, and so the timer is
66
76
  // always cleared — a leaked timer keeps the event loop alive and the MCP
67
77
  // server never exits cleanly.
68
78
  const controller = new AbortController();
69
- const timer = setTimeout(() => controller.abort(), this.config.timeoutMs);
79
+ const timer = setTimeout(() => controller.abort(), opts.timeoutMs ?? this.config.timeoutMs);
70
80
  this._inFlight.add(controller);
71
81
 
72
82
  const onExternalAbort = () => controller.abort();
@@ -106,6 +116,23 @@ export class ApiClient {
106
116
  async _decode(res, ctx) {
107
117
  const text = await res.text().catch(() => "");
108
118
 
119
+ // Capture quota state before any early return. The API advertises these for
120
+ // CORS exposure but nothing consumed them, so agents hit 429 blind.
121
+ // `Number(null)` is 0, so an absent header must be rejected explicitly —
122
+ // otherwise a client with full quota would be told it has none left.
123
+ const numHeader = (name) => {
124
+ const raw = res.headers?.get?.(name);
125
+ if (raw === null || raw === undefined || raw === "") return null;
126
+ const n = Number(raw);
127
+ return Number.isFinite(n) ? n : null;
128
+ };
129
+
130
+ this.rateLimit = {
131
+ remaining: numHeader("x-ratelimit-remaining"),
132
+ limit: numHeader("x-ratelimit-limit"),
133
+ reset: res.headers?.get?.("x-ratelimit-reset") ?? null,
134
+ };
135
+
109
136
  let parsed = null;
110
137
  if (text) {
111
138
  try {
package/src/config.js CHANGED
@@ -1,3 +1,5 @@
1
+ import { resolveExecutionMode } from "./local.js";
2
+
1
3
  /**
2
4
  * Configuration — resolved once, injected everywhere.
3
5
  *
@@ -108,6 +110,9 @@ export function loadConfig(env = process.env) {
108
110
  return {
109
111
  apiKey,
110
112
  apiBase: normaliseApiBase(env.NEXUSBLOOM_API_URL),
113
+ // Where tool code runs: "remote" (default), "local", or "auto". See local.js
114
+ // for why this is not a preference but a trust decision.
115
+ execution: resolveExecutionMode(env.NEXUSBLOOM_MCP_EXECUTION),
111
116
  timeoutMs,
112
117
  cacheTtlMs,
113
118
  debug,