@nexusbloom/mcp-server 2.1.0 → 2.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -52,6 +52,16 @@ anything else there corrupts the protocol.
52
52
  Every published tool is advertised as an MCP tool **named by its slug**, with its
53
53
  real JSON Schema — so it can be called directly:
54
54
 
55
+ > **One caveat on slugs.** A slug has to match MCP's tool-name grammar
56
+ > (`[A-Za-z0-9_-]{1,64}`) to be advertised. A catalogue slug that does not — a
57
+ > space, a slash, non-ASCII, or more than 64 characters — is *withheld from the
58
+ > tool list* and reported once on stderr, because a strict host validates every
59
+ > name in that array and rejects the whole response over one bad entry. That
60
+ > would cost an agent every tool rather than one. A withheld tool is still
61
+ > reachable through `{"command":"run"}` and `nexusbloom://catalogue`; it is only
62
+ > not directly callable. `npm run catalogue-check` verifies a catalogue against
63
+ > the grammar.
64
+
55
65
  ```jsonc
56
66
  { "name": "env-validator", "arguments": { "env_content": "DEBUG=true" } }
57
67
  ```
@@ -66,7 +76,7 @@ which tool it wants:
66
76
  | `{"command":"schema","slug":"…"}` | Exact parameters, plus a ready-to-send example invocation. |
67
77
  | `{"command":"run","slug":"…","params":{…}}` | Execute a tool. |
68
78
  | `{"command":"batch","runs":[{"slug":"…","params":{…}},…]}` | Execute up to 10 tools in one call. |
69
- | `{"command":"history"}` | What this session has run, newest first. |
79
+ | `{"command":"history"}` | What this **server process** has run, newest first. |
70
80
  | `{"command":"history","show":"<id>"}` | One run in full: input, result, timing. |
71
81
  | `{"command":"diff","from":"<id>","to":"<id>"}` | Compare two runs field by field. |
72
82
 
@@ -201,6 +211,16 @@ Three deliberate constraints:
201
211
  - **In memory, never on disk.** Run payloads are user data and frequently
202
212
  secrets. A log on disk would be an unencrypted store nobody asked for, and
203
213
  losing it on restart costs nothing the tool call did not.
214
+ - **Scoped to the process, not the conversation.** Most hosts keep the stdio
215
+ process alive across turns, so history spans every conversation turn the
216
+ process has served — it is not reset between them. Every history response says
217
+ so, and carries the process start time, because an agent that finds a run it
218
+ cannot account for needs somewhere to stop guessing. Observed in a real
219
+ session: an agent correctly reported that an unexplained earlier run had
220
+ different input, then invented a story about who had changed it and when,
221
+ and escalated it into a security incident. The facts were right; the cause
222
+ was fiction. Naming the boundary is what makes "this predates me" a
223
+ checkable statement.
204
224
  - **Bounded.** 50 runs, each payload capped at 8,000 characters. Truncation is
205
225
  recorded, and a diff says so — a truncated payload reporting "no differences"
206
226
  would be the worst possible failure mode here.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@nexusbloom/mcp-server",
3
- "version": "2.1.0",
4
- "description": "MCP server for NexusBloom \u2014 agents discover tools by intent, read exact schemas, and execute them. Built on @nexusbloom/core.",
3
+ "version": "2.1.1",
4
+ "description": "MCP server for NexusBloom — agents discover tools by intent, read exact schemas, and execute them. Built on @nexusbloom/core.",
5
5
  "type": "module",
6
6
  "main": "index.js",
7
7
  "bin": {
@@ -40,6 +40,7 @@
40
40
  "mock-api": "node scripts/mock-api.mjs",
41
41
  "agent-demo": "node scripts/agent-demo.mjs",
42
42
  "agent-demo:mock": "node scripts/agent-demo.mjs --url http://127.0.0.1:8787/api",
43
+ "catalogue-check": "node scripts/full-catalogue-check.mjs",
43
44
  "lint": "node --check index.js && for f in src/*.js; do node --check \"$f\" || exit 1; done"
44
45
  }
45
46
  }
package/src/handlers.js CHANGED
@@ -18,7 +18,7 @@ import {
18
18
  suggest,
19
19
  } from "./discovery.js";
20
20
  import { asNexusBloomError, ErrorCode, NexusBloomError } from "./errors.js";
21
- import { ManifestCache, normaliseTool } from "./manifests.js";
21
+ import { ManifestCache, normaliseTool, partitionAdvertisable } from "./manifests.js";
22
22
  import {
23
23
  renderBatch,
24
24
  renderConnectivityReport,
@@ -298,7 +298,7 @@ export function createHandlers({ client, cache, config, history: providedHistory
298
298
  }
299
299
  const limit = Number.isInteger(args?.limit) && args.limit > 0 ? Math.min(args.limit, 50) : 10;
300
300
  const all = history.list();
301
- return respond(renderHistory(history.recent(limit), { total: all.length }));
301
+ return respond(renderHistory(history.recent(limit), { total: all.length, scope: history.scope() }));
302
302
  }
303
303
 
304
304
  if (command === "diff") {
@@ -449,7 +449,19 @@ export function createHandlers({ client, cache, config, history: providedHistory
449
449
  }
450
450
 
451
451
  return {
452
- listTools: buildToolList(cache),
452
+ // Withheld tools are reported once, on stderr, and not behind a debug flag:
453
+ // a slug the catalogue cannot express as an MCP tool name is a data problem
454
+ // somebody has to fix, and silence leaves the tool quietly unreachable.
455
+ listTools: buildToolList(cache, {
456
+ onWithheld(withheld) {
457
+ for (const w of withheld) {
458
+ process.stderr.write(
459
+ `[catalogue] "${w.slug}" is not advertised as an MCP tool — ${w.reason}. ` +
460
+ 'It is still reachable via {"command":"run"} and nexusbloom://catalogue.\n',
461
+ );
462
+ }
463
+ },
464
+ }),
453
465
  meta,
454
466
  batch: runBatch,
455
467
  history,
@@ -569,7 +581,7 @@ function enrichNotFound(err, slug, tools) {
569
581
  * meta-tool is advertised too — it is how an agent recovers when a direct call
570
582
  * is not what it wanted.
571
583
  */
572
- export function buildToolList(cache) {
584
+ export function buildToolList(cache, { onWithheld } = {}) {
573
585
  return async function listTools() {
574
586
  // A catalogue failure must not fail the whole request. If the API is
575
587
  // unreachable, a host that receives a JSON-RPC error considers the *server*
@@ -582,7 +594,16 @@ export function buildToolList(cache) {
582
594
  tools = [];
583
595
  }
584
596
 
585
- const entries = tools.map((t) => ({
597
+ // A slug that is not a legal MCP tool name is held back rather than
598
+ // emitted. A strict host validates every name in this array and rejects the
599
+ // entire response over one bad entry, so a single odd slug in the catalogue
600
+ // would cost the agent every tool — not just that one. The meta-tool still
601
+ // reaches the withheld tools, and the reason is surfaced rather than
602
+ // swallowed, so the catalogue can be fixed at source.
603
+ const { advertisable, withheld } = partitionAdvertisable(tools);
604
+ if (withheld.length) onWithheld?.(withheld);
605
+
606
+ const entries = advertisable.map((t) => ({
586
607
  name: t.slug,
587
608
  description: buildToolDescription(t),
588
609
  inputSchema: t.input_schema,
@@ -597,7 +618,7 @@ export function buildToolList(cache) {
597
618
  '- {"command":"list"} — every published tool, one line each\n' +
598
619
  '- {"command":"schema","slug":"<slug>"} — exact parameters plus a ready-to-send example\n' +
599
620
  '- {"command":"batch","runs":[{"slug":"<slug>","params":{…}},…]} — run up to 10 tools in one call\n' +
600
- '- {"command":"history"} / {"command":"history","show":"<id>"} — what this session has run\n' +
621
+ '- {"command":"history"} / {"command":"history","show":"<id>"} — what this server process has run\n' +
601
622
  '- {"command":"diff","from":"<id>","to":"<id>"} — compare two runs field by field\n\n' +
602
623
  "Any published tool is also callable directly by its slug, with its own " +
603
624
  "parameters as arguments — this tool is for when you do not yet know which one to use.\n\n" +
@@ -613,7 +634,7 @@ export function buildToolList(cache) {
613
634
  description:
614
635
  "list: all tools. search: find by intent. schema: a tool's parameters. " +
615
636
  "run: execute one tool. batch: execute up to 10 tools in one call. " +
616
- "history: what this session has run. diff: compare two recorded runs.",
637
+ "history: what this server process has run, and since when. diff: compare two recorded runs.",
617
638
  },
618
639
  slug: {
619
640
  type: "string",
package/src/history.js CHANGED
@@ -61,6 +61,37 @@ export class RunHistory {
61
61
  this.now = now;
62
62
  this.runs = [];
63
63
  this._nextId = 1;
64
+ // When this process started holding history. Present so an agent can tell
65
+ // "a run this process saw" from "a run from before I was talking" — see
66
+ // the note on scope below.
67
+ this.startedAt = new Date(this.now()).toISOString();
68
+ }
69
+
70
+ /**
71
+ * What "history" actually covers, for callers to surface verbatim.
72
+ *
73
+ * This is deliberately not called a session. A host that keeps the stdio
74
+ * process alive across turns — which is most of them — will hand the same
75
+ * process several separate conversations, and history spans all of them.
76
+ *
77
+ * Observed in a real agent session: the agent found an `r1` it could not
78
+ * account for, correctly reported that its input differed from its own, and
79
+ * then invented a causal story about who changed what and when — including
80
+ * an urgent security conclusion — because nothing in the output said where
81
+ * the boundary was. The facts were right and the cause was fiction.
82
+ *
83
+ * Naming the scope is what stops that: an agent can say "this run predates my
84
+ * involvement in this process", which is checkable, instead of inventing why.
85
+ */
86
+ scope() {
87
+ return {
88
+ scope: "server-process",
89
+ startedAt: this.startedAt,
90
+ note:
91
+ "In-memory only. Covers every run since this server process started, " +
92
+ "which may span multiple conversation turns if your host keeps the " +
93
+ "process alive. Never written to disk.",
94
+ };
64
95
  }
65
96
 
66
97
  /**
package/src/manifests.js CHANGED
@@ -37,6 +37,50 @@ export function normaliseTags(tags) {
37
37
  * cannot be called, and letting it through would surface as a tool the agent
38
38
  * sees in the list but can never invoke.
39
39
  */
40
+ /**
41
+ * The MCP tool-name grammar.
42
+ *
43
+ * A published slug is whatever the catalogue happens to contain, and nothing
44
+ * upstream constrains it to this. Emitting an illegal name in `tools/list` is
45
+ * not a cosmetic problem: a strict host validates every name in the array and
46
+ * rejects the *whole response* over one bad entry, which leaves the agent with
47
+ * no tools at all rather than one fewer. So illegal slugs are held back from
48
+ * the tool list and reported instead — see `partitionAdvertisable`.
49
+ */
50
+ export const MCP_TOOL_NAME = /^[a-zA-Z0-9_-]{1,64}$/;
51
+
52
+ /**
53
+ * Split a catalogue into tools that can be advertised as MCP tools and tools
54
+ * whose slug cannot be.
55
+ *
56
+ * The held-back tools stay fully usable through the meta-tool and the catalogue
57
+ * resource — they are only prevented from poisoning the tool list for
58
+ * everything else. That is the difference between "this tool is awkward to
59
+ * address" and "this tool does not exist".
60
+ *
61
+ * @param {object[]} tools normalised tools
62
+ * @returns {{advertisable: object[], withheld: {slug: string, reason: string}[]}}
63
+ */
64
+ export function partitionAdvertisable(tools) {
65
+ const advertisable = [];
66
+ const withheld = [];
67
+
68
+ for (const tool of tools) {
69
+ const slug = tool?.slug ?? "";
70
+ if (MCP_TOOL_NAME.test(slug)) {
71
+ advertisable.push(tool);
72
+ continue;
73
+ }
74
+ let reason;
75
+ if (!slug) reason = "slug is empty";
76
+ else if (slug.length > 64) reason = `slug is ${slug.length} characters, the MCP maximum is 64`;
77
+ else reason = `slug contains characters MCP does not allow in a tool name: ${JSON.stringify(slug)}`;
78
+ withheld.push({ slug: String(slug), reason });
79
+ }
80
+
81
+ return { advertisable, withheld };
82
+ }
83
+
40
84
  export function normaliseTool(raw) {
41
85
  if (!raw || typeof raw !== "object") return null;
42
86
  const slug = typeof raw.slug === "string" ? raw.slug.trim() : "";
package/src/render.js CHANGED
@@ -340,13 +340,22 @@ export function renderResult(slug, data, { durationMs, execution = "remote", rat
340
340
  * compare, and inlining fifty payloads would cost more context than the diff the
341
341
  * reader actually came for. `history show` carries the payload for one run.
342
342
  */
343
- export function renderHistory(runs, { total } = {}) {
343
+ export function renderHistory(runs, { total, scope } = {}) {
344
+ // The scope line is load-bearing, not decoration. Without it an agent that
345
+ // finds a run it cannot account for has no way to tell "this process saw it"
346
+ // from "it happened before I was involved", and will fill the gap with a
347
+ // guess. Naming the boundary lets it say "this predates me" and stop there.
348
+ const scopeLine = scope
349
+ ? `\nScope: ${scope.scope}, started ${scope.startedAt}. ${scope.note}\n`
350
+ : "";
351
+
344
352
  if (!runs.length) {
345
353
  return {
346
354
  text:
347
- `# Run history\n\nNo runs recorded in this session.\n\n` +
348
- `History lives in memory only: it starts empty on every server start and is never written to disk.`,
349
- data: { runs: [], count: 0 },
355
+ `# Run history\n\nNo runs recorded since this server started.` +
356
+ scopeLine +
357
+ `\nHistory is never written to disk.`,
358
+ data: { runs: [], count: 0, ...(scope ? { scope } : {}) },
350
359
  };
351
360
  }
352
361
 
@@ -358,13 +367,14 @@ export function renderHistory(runs, { total } = {}) {
358
367
 
359
368
  const text =
360
369
  `# Run history\n\n` +
361
- `${runs.length} of ${total ?? runs.length} recorded run${runs.length === 1 ? "" : "s"}, newest first.\n\n` +
362
- `${lines.join("\n")}\n\n` +
370
+ `${runs.length} of ${total ?? runs.length} recorded run${runs.length === 1 ? "" : "s"}, newest first.` +
371
+ scopeLine +
372
+ `\n${lines.join("\n")}\n\n` +
363
373
  `---\n` +
364
374
  `Compare two runs with \`{"command":"diff","from":"<id>","to":"<id>"}\`. ` +
365
375
  `\`-1\` is the most recent run, \`-2\` the one before it.`;
366
376
 
367
- return { text, data: { runs, count: runs.length } };
377
+ return { text, data: { runs, count: runs.length, ...(scope ? { scope } : {}) } };
368
378
  }
369
379
 
370
380
  /** One run in full, with its result. */
package/src/resources.js CHANGED
@@ -62,7 +62,8 @@ export function createResources({ cache, config, history } = {}) {
62
62
  uri: HISTORY_URI,
63
63
  name: "Run history",
64
64
  description:
65
- "What this session has run, newest first: run id, tool, status, duration. " +
65
+ "What this server process has run, newest first: run id, tool, status, duration, " +
66
+ "plus the process start time so the boundary is explicit. " +
66
67
  "In memory only — never written to disk.",
67
68
  mimeType: JSON_MIME,
68
69
  },
@@ -137,7 +138,10 @@ export function createResources({ cache, config, history } = {}) {
137
138
  function readHistory(log) {
138
139
  if (!log) return { runs: [], count: 0, note: "No run log is attached to this resource set." };
139
140
  const runs = log.recent(50).map((r) => summariseRun(r));
140
- return { runs, count: runs.length };
141
+ // Same scope block as `{"command":"history"}` — a host pulling this as a
142
+ // resource has no more idea when the process started than one calling the
143
+ // command does.
144
+ return { runs, count: runs.length, scope: log.scope() };
141
145
  }
142
146
 
143
147
  /**
@@ -265,7 +269,7 @@ Meta commands:
265
269
  - \`{"command":"schema","slug":"<slug>"}\` — parameters plus a ready-to-send example
266
270
  - \`{"command":"run","slug":"<slug>","params":{…}}\` — execute
267
271
  - \`{"command":"batch","runs":[…]}\` — up to 10 runs in one call
268
- - \`{"command":"history"}\` / \`{"command":"history","show":"<id>"}\` — what this session ran
272
+ - \`{"command":"history"}\` / \`{"command":"history","show":"<id>"}\` — what this server process ran
269
273
  - \`{"command":"diff","from":"<id>","to":"<id>"}\` — compare two runs, field by field
270
274
 
271
275
  And when several tools belong to one task, run them in a single turn: