@nexusbloom/mcp-server 2.0.2 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +149 -0
- package/index.js +13 -1
- package/package.json +5 -3
- package/src/config.js +5 -0
- package/src/handlers.js +343 -13
- package/src/history.js +285 -0
- package/src/local.js +164 -0
- package/src/progress.js +88 -0
- package/src/prompts.js +282 -0
- package/src/render.js +198 -9
- package/src/resources.js +318 -0
- package/src/server.js +106 -7
- package/src/validate.js +14 -1
package/README.md
CHANGED
|
@@ -42,6 +42,7 @@ All variables are optional.
|
|
|
42
42
|
| `NEXUSBLOOM_MCP_TIMEOUT_MS` | `15000` | Per-request timeout. |
|
|
43
43
|
| `NEXUSBLOOM_MCP_CACHE_TTL_MS` | `60000` | Tool-list cache lifetime. `0` disables caching. |
|
|
44
44
|
| `NEXUSBLOOM_MCP_DEBUG` | *(off)* | `1` logs requests to stderr. |
|
|
45
|
+
| `NEXUSBLOOM_MCP_EXECUTION` | `remote` | `local` or `auto` to execute tool source on this machine. **Off by default on purpose** — see below. |
|
|
45
46
|
|
|
46
47
|
Diagnostics always go to **stderr**; stdout is the MCP transport and writing
|
|
47
48
|
anything else there corrupts the protocol.
|
|
@@ -64,6 +65,154 @@ which tool it wants:
|
|
|
64
65
|
| `{"command":"list"}` | Every published tool, one line each. |
|
|
65
66
|
| `{"command":"schema","slug":"…"}` | Exact parameters, plus a ready-to-send example invocation. |
|
|
66
67
|
| `{"command":"run","slug":"…","params":{…}}` | Execute a tool. |
|
|
68
|
+
| `{"command":"batch","runs":[{"slug":"…","params":{…}},…]}` | Execute up to 10 tools in one call. |
|
|
69
|
+
| `{"command":"history"}` | What this session has run, newest first. |
|
|
70
|
+
| `{"command":"history","show":"<id>"}` | One run in full: input, result, timing. |
|
|
71
|
+
| `{"command":"diff","from":"<id>","to":"<id>"}` | Compare two runs field by field. |
|
|
72
|
+
|
|
73
|
+
### Batching
|
|
74
|
+
|
|
75
|
+
Several tools belonging to one task go in a single turn:
|
|
76
|
+
|
|
77
|
+
```jsonc
|
|
78
|
+
{ "command": "batch", "runs": [
|
|
79
|
+
{ "slug": "env-validator", "params": { "env_content": "DEBUG=true" } },
|
|
80
|
+
{ "slug": "data-generator", "params": { "count": 2 } }
|
|
81
|
+
] }
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
Two behaviours make it safe rather than merely fast:
|
|
85
|
+
|
|
86
|
+
- **Validate everything first.** Every slug is resolved and every schema checked
|
|
87
|
+
before the first run. A typo in item four costs zero quota and rejects the
|
|
88
|
+
whole batch — a batch is one intent, so half of it is not a useful answer.
|
|
89
|
+
- **Then isolate failures.** Past that point each run is independent. One tool
|
|
90
|
+
erroring does not discard the seven that worked; the response reports the
|
|
91
|
+
tally, lists failures first, and the structured block keeps every payload.
|
|
92
|
+
|
|
93
|
+
Capped at 10 runs, because anonymous execution allows 30 requests/minute and one
|
|
94
|
+
turn should not spend the budget the next ten calls need.
|
|
95
|
+
|
|
96
|
+
### Progress notifications
|
|
97
|
+
|
|
98
|
+
If the host sends a `progressToken` in `_meta`, the server emits
|
|
99
|
+
`notifications/progress` while a batch runs — one tick per entry, plus an opening
|
|
100
|
+
tick so the host is never silent. Progress is monotonic, and a failed
|
|
101
|
+
notification never fails the call. A host that sends no token receives none:
|
|
102
|
+
unsolicited progress is a protocol violation, not a nicety.
|
|
103
|
+
|
|
104
|
+
### Resources — read the catalogue without spending a call
|
|
105
|
+
|
|
106
|
+
Hosts that support MCP resources can pull the catalogue directly, which is how
|
|
107
|
+
an agent holds the whole surface in context instead of discovering it one slug
|
|
108
|
+
at a time:
|
|
109
|
+
|
|
110
|
+
| URI | Contents |
|
|
111
|
+
|---|---|
|
|
112
|
+
| `nexusbloom://guide` | This usage guide as markdown: meta commands, recovery paths, limits. |
|
|
113
|
+
| `nexusbloom://catalogue` | Every tool as JSON — slug, description, category, tags, parameter counts, `has_schema`, and the URI of its full manifest. |
|
|
114
|
+
| `nexusbloom://tools/{slug}` | One tool's full manifest: input **and** output JSON Schema, plus `call: { tool, arguments }` — a ready-to-send example. |
|
|
115
|
+
|
|
116
|
+
The catalogue deliberately omits full schemas; that is what the per-tool URI is
|
|
117
|
+
for. A 31-manifest inline blob would exhaust the context budget before the agent
|
|
118
|
+
had chosen anything.
|
|
119
|
+
|
|
120
|
+
Slugs in resource URIs resolve by exact match or unambiguous abbreviation only,
|
|
121
|
+
identically to tool calls, and an unknown slug produces a protocol error naming
|
|
122
|
+
the close matches rather than an empty body an agent might quote as fact.
|
|
123
|
+
|
|
124
|
+
### Prompts — one-click starts a host can offer
|
|
125
|
+
|
|
126
|
+
Four prompts, each assembled against the live catalogue rather than canned, so
|
|
127
|
+
the text a model receives names tools that actually exist and carries calls that
|
|
128
|
+
already validate:
|
|
129
|
+
|
|
130
|
+
| Prompt | Arguments | Gives the model |
|
|
131
|
+
|---|---|---|
|
|
132
|
+
| `find-tool` | `goal` | Tools matching the goal, ranked, with each one's parameter cost and the manifest URI. |
|
|
133
|
+
| `use-tool` | `slug` | One tool's exact parameters, its `output_schema`, and a valid call to copy. |
|
|
134
|
+
| `plan-batch` | `task` | Likely participants, the batch call shape, the 10-run cap, and the whole-batch validation rule. |
|
|
135
|
+
| `recover` | `error`, `slug` | What each error code means, what to do about it, and the failed tool's real call shape. |
|
|
136
|
+
|
|
137
|
+
`recover` exists because the common failure is not "the tool broke" — it is an
|
|
138
|
+
agent retrying a validation failure unchanged, or guessing slugs. The prompt
|
|
139
|
+
states the codes, marks which are retryable, and embeds the required fields of the
|
|
140
|
+
tool that just failed.
|
|
141
|
+
|
|
142
|
+
`use-tool` resolves slugs **exactly**, unlike tool calls. A prompt naming a tool
|
|
143
|
+
is a deliberate user choice; silently running a different tool because a prefix
|
|
144
|
+
happened to match would be the wrong answer to a request the user made explicitly.
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
### Local execution (opt-in, and read this first)
|
|
148
|
+
|
|
149
|
+
`NEXUSBLOOM_MCP_EXECUTION` chooses where tool code runs:
|
|
150
|
+
|
|
151
|
+
| Mode | Behaviour |
|
|
152
|
+
|---|---|
|
|
153
|
+
| `remote` *(default)* | Always call the API. No tool source is ever fetched. |
|
|
154
|
+
| `local` | Always execute the published source in a child process. A tool with no source is an error. |
|
|
155
|
+
| `auto` | Execute locally when source exists, otherwise fall back to the API. |
|
|
156
|
+
|
|
157
|
+
An unrecognised value falls back to `remote`, so a typo cannot silently switch on
|
|
158
|
+
code execution.
|
|
159
|
+
|
|
160
|
+
**The threat model.** Executing a tool locally means executing code that arrived
|
|
161
|
+
over the network. The sandbox from `@nexusbloom/core` gives the tool its own
|
|
162
|
+
process, a SIGKILL deadline, a heap cap, and no access to this process's stdout.
|
|
163
|
+
It is **isolation, not a security sandbox**: that code still runs with your user
|
|
164
|
+
permissions and can read files, open sockets, and spawn processes.
|
|
165
|
+
|
|
166
|
+
That is why it is off by default. An MCP server is driven by whatever the model
|
|
167
|
+
decides to ask for, so the default posture has to be the one where a surprising
|
|
168
|
+
tool call cannot become arbitrary code execution on a developer's machine. Turn it
|
|
169
|
+
on when you trust the catalogue — a self-hosted deployment publishing only your
|
|
170
|
+
own tools — or to keep working while the API is unreachable (`auto`).
|
|
171
|
+
|
|
172
|
+
Enabling it prints the same warning to stderr at startup, once, because the
|
|
173
|
+
threat model belongs on the record at the moment it is switched on.
|
|
174
|
+
|
|
175
|
+
**Credentials are stripped from the child.** The sandbox forks the tool with the
|
|
176
|
+
parent's environment minus anything whose name looks like a credential —
|
|
177
|
+
`NEXUSBLOOM_API_KEY`, `STRIPE_SECRET_KEY`, `AWS_SESSION_TOKEN`, `*_PASSWORD`,
|
|
178
|
+
`*_AUTH`, and so on, matched by substring. Tool code has no legitimate reason to
|
|
179
|
+
read your API key, and without this a published tool could simply read it and
|
|
180
|
+
exfiltrate it — which turns "runs with your permissions" into "runs with your
|
|
181
|
+
account". Ordinary variables (`PATH`, project config, anything you have set) pass
|
|
182
|
+
through untouched, so tools that read their own configuration still work.
|
|
183
|
+
|
|
184
|
+
### Run history and diff
|
|
185
|
+
|
|
186
|
+
Every run this server performs is recorded, so an agent can see what already
|
|
187
|
+
happened instead of guessing:
|
|
188
|
+
|
|
189
|
+
- `{"command":"history"}` — an index: run id, tool, status, duration, newest first.
|
|
190
|
+
- `{"command":"history","show":"r3"}` — one run with its input and its result.
|
|
191
|
+
- `{"command":"diff","from":"-2","to":"-1"}` — two runs compared field by field. `-1`
|
|
192
|
+
is the most recent run, `-2` the one before it, so comparing consecutive runs
|
|
193
|
+
needs no lookup.
|
|
194
|
+
|
|
195
|
+
A diff leads with the verdict and flags changed inputs, because a result that
|
|
196
|
+
moved because its *input* moved is not a regression — conflating the two is how a
|
|
197
|
+
diff gets dismissed as noise and then ignored when it mattered.
|
|
198
|
+
|
|
199
|
+
Three deliberate constraints:
|
|
200
|
+
|
|
201
|
+
- **In memory, never on disk.** Run payloads are user data and frequently
|
|
202
|
+
secrets. A log on disk would be an unencrypted store nobody asked for, and
|
|
203
|
+
losing it on restart costs nothing the tool call did not.
|
|
204
|
+
- **Bounded.** 50 runs, each payload capped at 8,000 characters. Truncation is
|
|
205
|
+
recorded, and a diff says so — a truncated payload reporting "no differences"
|
|
206
|
+
would be the worst possible failure mode here.
|
|
207
|
+
- **Inputs redacted.** Any key matching `key`, `secret`, `token`, `password`,
|
|
208
|
+
`auth`, `cookie` or `session` — plus `passwd`, `passphrase` and `credential` —
|
|
209
|
+
is masked before storage, at any depth up to 6 levels, without preserving
|
|
210
|
+
length. Beyond that depth the value is replaced with `[deep]` rather than
|
|
211
|
+
inspected, so a deeply nested payload cannot smuggle a credential past the
|
|
212
|
+
walk.
|
|
213
|
+
|
|
214
|
+
The index is also readable as `nexusbloom://history`, for hosts that would rather
|
|
215
|
+
pull it than call for it.
|
|
67
216
|
|
|
68
217
|
### Errors are actionable
|
|
69
218
|
|
package/index.js
CHANGED
|
@@ -24,6 +24,7 @@ import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js"
|
|
|
24
24
|
import { ApiClient } from "./src/client.js";
|
|
25
25
|
import { candidateRoots, loadConfig } from "./src/config.js";
|
|
26
26
|
import { checkConnectivity } from "./src/handlers.js";
|
|
27
|
+
import { localExecutionNotice } from "./src/local.js";
|
|
27
28
|
import { ManifestCache } from "./src/manifests.js";
|
|
28
29
|
import { createServer } from "./src/server.js";
|
|
29
30
|
|
|
@@ -50,6 +51,12 @@ export async function main(env = process.env, deps = {}) {
|
|
|
50
51
|
const connectivity = await checkConnectivity(app.client, candidateRoots(env), config, deps);
|
|
51
52
|
process.stderr.write(`${connectivity.report}\n`);
|
|
52
53
|
|
|
54
|
+
// Printed once, before serving, and only when the operator opted in. The
|
|
55
|
+
// threat model belongs on the record at the moment it is enabled, not in a
|
|
56
|
+
// README nobody re-reads when a host later behaves strangely.
|
|
57
|
+
const notice = localExecutionNotice(config.execution);
|
|
58
|
+
if (notice) process.stderr.write(notice);
|
|
59
|
+
|
|
53
60
|
const transport = new StdioServerTransport();
|
|
54
61
|
|
|
55
62
|
// Close on the signals a host actually sends. Without this the process
|
|
@@ -66,7 +73,12 @@ export async function main(env = process.env, deps = {}) {
|
|
|
66
73
|
process.once("SIGTERM", shutdown);
|
|
67
74
|
|
|
68
75
|
await app.server.connect(transport);
|
|
69
|
-
|
|
76
|
+
// Name the capabilities explicitly: a user debugging a host that shows no
|
|
77
|
+
// tools is usually looking at whether the host negotiated them, and a
|
|
78
|
+
// capability list on stderr answers that in one line.
|
|
79
|
+
process.stderr.write(
|
|
80
|
+
"NexusBloom MCP server running on stdio (capabilities: tools, resources, prompts)\n",
|
|
81
|
+
);
|
|
70
82
|
|
|
71
83
|
return { ...app, transport, close: shutdown };
|
|
72
84
|
}
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@nexusbloom/mcp-server",
|
|
3
|
-
"version": "2.0
|
|
4
|
-
"description": "MCP server for NexusBloom
|
|
3
|
+
"version": "2.1.0",
|
|
4
|
+
"description": "MCP server for NexusBloom \u2014 agents discover tools by intent, read exact schemas, and execute them. Built on @nexusbloom/core.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "index.js",
|
|
7
7
|
"bin": {
|
|
@@ -34,10 +34,12 @@
|
|
|
34
34
|
"test": "node --test --import ./test/setup.mjs test/*.test.js",
|
|
35
35
|
"test:coverage": "node --test --experimental-test-coverage --import ./test/setup.mjs test/*.test.js",
|
|
36
36
|
"test:watch": "node --test --watch --import ./test/setup.mjs test/*.test.js",
|
|
37
|
-
"test:src-only": "node --test --import ./test/setup.mjs test/config.test.js test/errors.test.js test/manifests.test.js test/discovery.test.js test/client.test.js test/validate.test.js test/render.test.js test/cache.test.js test/handlers.test.js",
|
|
37
|
+
"test:src-only": "node --test --import ./test/setup.mjs test/config.test.js test/errors.test.js test/manifests.test.js test/discovery.test.js test/client.test.js test/validate.test.js test/render.test.js test/cache.test.js test/handlers.test.js test/resources.test.js test/batch.test.js test/progress.test.js test/prompts.test.js test/history.test.js test/local.test.js",
|
|
38
38
|
"shell": "node scripts/mcp-shell.mjs",
|
|
39
39
|
"shell:mock": "node scripts/mcp-shell.mjs --mock",
|
|
40
40
|
"mock-api": "node scripts/mock-api.mjs",
|
|
41
|
+
"agent-demo": "node scripts/agent-demo.mjs",
|
|
42
|
+
"agent-demo:mock": "node scripts/agent-demo.mjs --url http://127.0.0.1:8787/api",
|
|
41
43
|
"lint": "node --check index.js && for f in src/*.js; do node --check \"$f\" || exit 1; done"
|
|
42
44
|
}
|
|
43
45
|
}
|
package/src/config.js
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { resolveExecutionMode } from "./local.js";
|
|
2
|
+
|
|
1
3
|
/**
|
|
2
4
|
* Configuration — resolved once, injected everywhere.
|
|
3
5
|
*
|
|
@@ -108,6 +110,9 @@ export function loadConfig(env = process.env) {
|
|
|
108
110
|
return {
|
|
109
111
|
apiKey,
|
|
110
112
|
apiBase: normaliseApiBase(env.NEXUSBLOOM_API_URL),
|
|
113
|
+
// Where tool code runs: "remote" (default), "local", or "auto". See local.js
|
|
114
|
+
// for why this is not a preference but a trust decision.
|
|
115
|
+
execution: resolveExecutionMode(env.NEXUSBLOOM_MCP_EXECUTION),
|
|
111
116
|
timeoutMs,
|
|
112
117
|
cacheTtlMs,
|
|
113
118
|
debug,
|