@nexusbloom/mcp-server 2.0.0 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/history.js ADDED
@@ -0,0 +1,285 @@
1
+ /**
2
+ * Run history — a bounded, in-process record of what this server ran.
3
+ *
4
+ * Two things it exists for:
5
+ *
6
+ * 1. **`diff`.** Comparing what a tool produced before and after a change is the
7
+ * fastest way to see whether an edit actually changed behaviour, and it is
8
+ * impossible without both sides captured.
9
+ * 2. **Self-correction.** An agent that has just seen three failed calls in a row
10
+ * is guessing; being able to read the last few runs, with their inputs, turns
11
+ * a guess into a comparison.
12
+ *
13
+ * Three constraints are deliberate:
14
+ *
15
+ * - **In memory, never on disk.** Run payloads are user data and frequently
16
+ * secrets. Persisting them would create an unencrypted store nobody asked for;
17
+ * losing history on restart costs nothing that the tool call itself did not.
18
+ * - **Bounded.** A ring buffer, so a long-lived server in a chat window cannot
19
+ * grow without limit.
20
+ * - **Redacted inputs.** Anything whose key looks like a credential is masked
21
+ * before it is stored, because history is readable by anything that can read
22
+ * the conversation that produced it.
23
+ */
24
+
25
+ import { NexusBloomError, ErrorCode } from "./errors.js";
26
+
27
+ /** How many runs to keep. Oldest are dropped first. */
28
+ export const MAX_HISTORY = 50;
29
+
30
+ /**
31
+ * How much of a result to keep, in characters.
32
+ *
33
+ * Generous for the JSON a tool returns, small enough that fifty of them cannot
34
+ * exhaust a host's memory. Truncation is recorded rather than silent: a diff over
35
+ * a truncated payload says so, instead of reporting "no differences" because the
36
+ * tail is missing.
37
+ */
38
+ export const MAX_PAYLOAD_CHARS = 8_000;
39
+
40
+ /** Keys whose values are masked in stored inputs. */
41
+ const SECRET_KEY = /(key|secret|token|password|passwd|passphrase|credential|auth|cookie|session)/i;
42
+
43
+ /** Deepest level the differ walks. Beyond this, values compare as opaque. */
44
+ const MAX_DIFF_DEPTH = 6;
45
+
46
+ /** Maximum number of differences reported, so a whole-object change is not a wall. */
47
+ const MAX_DIFF_ENTRIES = 200;
48
+
49
+ /**
50
+ * A bounded run log.
51
+ *
52
+ * @param {object} [opts]
53
+ * @param {number} [opts.limit] How many runs to keep.
54
+ * @param {number} [opts.maxPayload] Character cap per stored payload.
55
+ * @param {() => number} [opts.now] Clock, injected so tests are deterministic.
56
+ */
57
+ export class RunHistory {
58
+ constructor({ limit = MAX_HISTORY, maxPayload = MAX_PAYLOAD_CHARS, now = () => Date.now() } = {}) {
59
+ this.limit = limit;
60
+ this.maxPayload = maxPayload;
61
+ this.now = now;
62
+ this.runs = [];
63
+ this._nextId = 1;
64
+ }
65
+
66
+ /**
67
+ * Append one run and return its record.
68
+ *
69
+ * The id is per-process and monotonic rather than a timestamp, because ids
70
+ * have to be short, ordered and unguessable-by-accident when they are quoted
71
+ * back in a `diff`.
72
+ */
73
+ record({ tool, params, data, ok, durationMs, error, code, retryable, batch }) {
74
+ const payload = this._cap(data);
75
+ const entry = {
76
+ id: `r${this._nextId++}`,
77
+ tool,
78
+ ok: Boolean(ok),
79
+ at: new Date(this.now()).toISOString(),
80
+ durationMs: typeof durationMs === "number" ? durationMs : null,
81
+ batch: Boolean(batch),
82
+ params: redactParams(params),
83
+ result: payload.value,
84
+ truncated: payload.truncated,
85
+ ...(ok ? {} : { error: error || "Unknown error.", code: code || null, retryable: Boolean(retryable) }),
86
+ };
87
+
88
+ this.runs.push(entry);
89
+ if (this.runs.length > this.limit) this.runs.splice(0, this.runs.length - this.limit);
90
+ return entry;
91
+ }
92
+
93
+ /** Every record, oldest first. */
94
+ list() {
95
+ return this.runs.slice();
96
+ }
97
+
98
+ /** The most recent record, or null. */
99
+ latest() {
100
+ return this.runs.at(-1) || null;
101
+ }
102
+
103
+ /**
104
+ * Find a record by id, or by a fractional reference.
105
+ *
106
+ * "the last one" is a thing people and models both say, so `-1` means the most
107
+ * recent and `-2` the one before it. Without that, every diff of consecutive
108
+ * runs needs a lookup first, which is a turn wasted on arithmetic.
109
+ */
110
+ find(reference) {
111
+ if (reference === undefined || reference === null || reference === "") return null;
112
+ const ref = String(reference).trim();
113
+
114
+ const byId = this.runs.find((r) => r.id === ref);
115
+ if (byId) return byId;
116
+
117
+ if (/^-?\d+$/.test(ref)) {
118
+ const n = Number(ref);
119
+ if (n < 0) return this.runs.at(n) || null;
120
+ // A positive integer is also accepted as an id, since ids are numeric: only
121
+ // fall through to ordinal lookup when nothing matched as an id.
122
+ return null;
123
+ }
124
+ return null;
125
+ }
126
+
127
+ /** Most recent records, newest first — the order a reader wants them in. */
128
+ recent(n = 10) {
129
+ return this.runs.slice(-n).reverse();
130
+ }
131
+
132
+ /**
133
+ * Compare two stored runs.
134
+ *
135
+ * Compares the *results*, because that is what a change is supposed to alter.
136
+ * Inputs are reported alongside when they differ, since a result that changed
137
+ * because its input changed is not a regression at all — and conflating the two
138
+ * is how a diff gets dismissed as noise.
139
+ */
140
+ diff(fromRef, toRef) {
141
+ const from = this.find(fromRef);
142
+ const to = this.find(toRef);
143
+ const missing = [];
144
+ if (!from) missing.push(fromRef ?? "(none)");
145
+ if (!to) missing.push(toRef ?? "(none)");
146
+
147
+ if (missing.length) {
148
+ throw new NexusBloomError(
149
+ `No run recorded for ${missing.map((m) => `"${m}"`).join(" or ")}. ` +
150
+ `Known run ids: ${this.runs.map((r) => r.id).join(", ") || "(none yet)"}. ` +
151
+ `Pass an id like "${this.runs.at(-1)?.id ?? "r1"}", or -1 for the most recent.`,
152
+ ErrorCode.NOT_FOUND,
153
+ );
154
+ }
155
+
156
+ const changes = diffValues(from.result, to.result, "", 0);
157
+ return {
158
+ from: summariseRun(from),
159
+ to: summariseRun(to),
160
+ identical: changes.length === 0 && from.ok === to.ok,
161
+ inputsChanged: JSON.stringify(from.params) !== JSON.stringify(to.params),
162
+ changes,
163
+ ...(from.truncated || to.truncated ? { truncated: true } : {}),
164
+ };
165
+ }
166
+
167
+ /** Drop everything; used by tests and by a `history clear` command. */
168
+ clear() {
169
+ this.runs = [];
170
+ }
171
+
172
+ _cap(value) {
173
+ if (value === undefined) return { value: null, truncated: false };
174
+ let text;
175
+ try {
176
+ text = JSON.stringify(value) ?? "null";
177
+ } catch {
178
+ text = String(value);
179
+ }
180
+ if (text.length <= this.maxPayload) return { value, truncated: false };
181
+
182
+ // Truncate the serialised form rather than the object: a half-object would
183
+ // break the differ, and a marked string is honest about what is missing.
184
+ return { value: `${text.slice(0, this.maxPayload)}…[truncated]`, truncated: true };
185
+ }
186
+ }
187
+
188
+ /** Mask credential-shaped values, at any depth. */
189
+ export function redactParams(params, depth = 0) {
190
+ if (depth > MAX_DIFF_DEPTH) return "[deep]";
191
+ if (Array.isArray(params)) return params.map((p) => redactParams(p, depth + 1));
192
+ if (params && typeof params === "object") {
193
+ const out = {};
194
+ for (const [key, value] of Object.entries(params)) {
195
+ out[key] = SECRET_KEY.test(key) && typeof value === "string" ? maskValue(value) : redactParams(value, depth + 1);
196
+ }
197
+ return out;
198
+ }
199
+ return params;
200
+ }
201
+
202
+ /**
203
+ * Mask a secret without leaking its length.
204
+ *
205
+ * Keeping the length would let someone confirm a guessed password, so a
206
+ * non-empty value collapses to a fixed marker.
207
+ */
208
+ function maskValue(value) {
209
+ return value ? "[redacted]" : "";
210
+ }
211
+
212
+ /**
213
+ * Structural diff of two JSON-ish values.
214
+ *
215
+ * Reports a flat list of `{path, type, from, to}` rather than a nested tree: an
216
+ * agent reads a flat list correctly, and a tree invites it to summarise a
217
+ * subtree it never looked at.
218
+ */
219
+ export function diffValues(a, b, path = "", depth = 0, out = []) {
220
+ if (out.length >= MAX_DIFF_ENTRIES) return out;
221
+ if (depth > MAX_DIFF_DEPTH) {
222
+ if (!Object.is(a, b)) out.push({ path: path || "(root)", type: "changed", from: a, to: b });
223
+ return out;
224
+ }
225
+
226
+ if (Object.is(a, b)) return out;
227
+
228
+ const bothObjects = isPlainObject(a) && isPlainObject(b);
229
+ const bothArrays = Array.isArray(a) && Array.isArray(b);
230
+
231
+ if (bothObjects) {
232
+ const keys = [...new Set([...Object.keys(a), ...Object.keys(b)])].sort();
233
+ for (const key of keys) {
234
+ const childPath = path ? `${path}.${key}` : key;
235
+ const inA = key in a;
236
+ const inB = key in b;
237
+ if (inA && !inB) out.push({ path: childPath, type: "removed", from: a[key] });
238
+ else if (!inA && inB) out.push({ path: childPath, type: "added", to: b[key] });
239
+ else diffValues(a[key], b[key], childPath, depth + 1, out);
240
+ if (out.length >= MAX_DIFF_ENTRIES) return out;
241
+ }
242
+ return out;
243
+ }
244
+
245
+ if (bothArrays) {
246
+ const length = Math.max(a.length, b.length);
247
+ for (let i = 0; i < length; i++) {
248
+ const childPath = `${path}[${i}]`;
249
+ if (i >= a.length) out.push({ path: childPath, type: "added", to: b[i] });
250
+ else if (i >= b.length) out.push({ path: childPath, type: "removed", from: a[i] });
251
+ else diffValues(a[i], b[i], childPath, depth + 1, out);
252
+ if (out.length >= MAX_DIFF_ENTRIES) return out;
253
+ }
254
+ return out;
255
+ }
256
+
257
+ if (a === undefined) return out.push({ path: path || "(root)", type: "added", to: b });
258
+ if (b === undefined) return out.push({ path: path || "(root)", type: "removed", from: a });
259
+
260
+ // Different types is a replacement, not a field-level change: reporting it as
261
+ // one change per field would imply a shape the data never had.
262
+ if (typeof a !== typeof b || (Array.isArray(a) !== Array.isArray(b))) {
263
+ out.push({ path: path || "(root)", type: "changed", from: a, to: b });
264
+ return out;
265
+ }
266
+
267
+ out.push({ path: path || "(root)", type: "changed", from: a, to: b });
268
+ return out;
269
+ }
270
+
271
+ function isPlainObject(value) {
272
+ return Boolean(value) && typeof value === "object" && !Array.isArray(value);
273
+ }
274
+
275
+ /** The one-line description of a run, used in diffs and listings. */
276
+ export function summariseRun(run) {
277
+ return {
278
+ id: run.id,
279
+ tool: run.tool,
280
+ ok: run.ok,
281
+ at: run.at,
282
+ durationMs: run.durationMs,
283
+ ...(run.ok ? {} : { error: run.error, code: run.code }),
284
+ };
285
+ }
package/src/local.js ADDED
@@ -0,0 +1,164 @@
1
+ /**
2
+ * Local execution — run a tool's published source instead of calling the API.
3
+ *
4
+ * **The threat model, stated plainly.** Executing a tool locally means executing
5
+ * code that arrived over the network. `@nexusbloom/core`'s sandbox is *isolation*,
6
+ * not a security sandbox: it gives the tool its own process, a hard kill on
7
+ * timeout, a heap cap, and no access to this process's stdout. It does **not**
8
+ * stop that code reading files, opening sockets, or spawning processes as the
9
+ * user running the host.
10
+ *
11
+ * So local execution is **off by default** and requires an explicit opt-in. That
12
+ * is not timidity: an MCP server is driven by whatever the model decides to ask
13
+ * for, so the default posture has to be the one where a surprising tool call
14
+ * cannot become arbitrary code execution on a developer's machine. An operator who
15
+ * turns it on is choosing that trade deliberately — typically for a self-hosted
16
+ * deployment publishing only their own tools, or to keep working while the API is
17
+ * unreachable.
18
+ *
19
+ * Three modes:
20
+ * - `remote` (default) — always call the API. Nothing local is fetched.
21
+ * - `local` — always execute locally; a tool with no published source is an error.
22
+ * - `auto` — run locally when source is available, otherwise fall back to the API.
23
+ *
24
+ * `auto` is the useful middle: it degrades to remote for the handful of tools
25
+ * whose source is not published, instead of failing a call the API could serve.
26
+ */
27
+
28
+ import { execInSandbox } from "@nexusbloom/core";
29
+
30
+ import { NexusBloomError, ErrorCode } from "./errors.js";
31
+
32
+ /**
33
+ * Largest source the server will hand to the sandbox.
34
+ *
35
+ * 2 MB is far above any real tool and well below anything that would stall the
36
+ * compiler. A cap is the cheapest defence against a hostile manifest trying to
37
+ * exhaust memory before execution even starts.
38
+ */
39
+ export const MAX_SOURCE_BYTES = 2 * 1024 * 1024;
40
+
41
+ /** Heap ceiling for the tool process. Mirrors the CLI's default. */
42
+ export const DEFAULT_MEMORY_MB = 256;
43
+
44
+ /**
45
+ * Local-execution modes, as resolved from configuration.
46
+ *
47
+ * An unrecognised value falls back to `remote` rather than being honoured:
48
+ * a typo in a config variable should not silently switch on code execution.
49
+ */
50
+ export function resolveExecutionMode(input) {
51
+ const value = (input || "").trim().toLowerCase();
52
+ return ["local", "auto", "remote"].includes(value) ? value : "remote";
53
+ }
54
+
55
+ /**
56
+ * Fetch a tool's runnable source.
57
+ *
58
+ * The API exposes it as `GET /run/<slug>?source=true`. Returns null when the tool
59
+ * has no published source, which is a normal state — not an error to report as
60
+ * one, because `auto` and `local` handle it differently.
61
+ */
62
+ export async function fetchSource(client, slug) {
63
+ let body;
64
+ try {
65
+ body = await client.request(`/run/${encodeURIComponent(slug)}?source=true`);
66
+ } catch (err) {
67
+ // A failed source fetch is not a failed tool call: `auto` must still be able
68
+ // to fall back to the API rather than surfacing a transport error.
69
+ return null;
70
+ }
71
+
72
+ const source =
73
+ body?.data?.v2_source ??
74
+ body?.v2_source ??
75
+ body?.data?.source ??
76
+ body?.source ??
77
+ null;
78
+
79
+ if (typeof source !== "string" || !source.trim()) return null;
80
+ if (source.length > MAX_SOURCE_BYTES) return null;
81
+ return source;
82
+ }
83
+
84
+ /**
85
+ * Execute a tool locally and return the same shape `execute()` expects.
86
+ *
87
+ * @param {object} opts
88
+ * @param {import("./client.js").ApiClient} opts.client
89
+ * @param {string} opts.slug
90
+ * @param {object} opts.params
91
+ * @param {string} opts.mode One of local | auto | remote.
92
+ * @param {object} opts.config
93
+ * @param {object} [opts.deps] Injected for tests: { execInSandbox }.
94
+ * @returns {Promise<{executed: boolean, data?: any, error?: NexusBloomError, source?: string}>}
95
+ */
96
+ export async function runLocal({ client, slug, params, mode = "remote", config, deps = {} }) {
97
+ if (mode === "remote") return { executed: false };
98
+
99
+ const exec = deps.execInSandbox ?? execInSandbox;
100
+ const source = await fetchSource(client, slug);
101
+
102
+ if (!source) {
103
+ if (mode === "local") {
104
+ throw new NexusBloomError(
105
+ `No runnable source is published for "${slug}", so it cannot run locally. ` +
106
+ `Set NEXUSBLOOM_MCP_EXECUTION=auto to fall back to the API for tools like this, ` +
107
+ `or remote to always call the API.`,
108
+ ErrorCode.NOT_FOUND,
109
+ );
110
+ }
111
+ return { executed: false };
112
+ }
113
+
114
+ // The tool's own timeout is the ceiling when it declares one: a tool that says
115
+ // it needs 30s should not be killed at the API's default, and a tool that says
116
+ // nothing gets the configured budget.
117
+ //
118
+ // `> 0` is load-bearing. config.js accepts 0 as "use the default", so a
119
+ // configured 0 can reach here; without this guard `Number.isFinite(0)` passes,
120
+ // the sandbox gets a zero-millisecond budget, and every local run is SIGKILLed
121
+ // before it produces a result — reported to the agent as a tool timeout rather
122
+ // than the misconfiguration it is.
123
+ const timeoutMs =
124
+ Number.isFinite(config?.timeoutMs) && config.timeoutMs > 0 ? config.timeoutMs : 15_000;
125
+
126
+ const result = await exec({
127
+ source,
128
+ input: params ?? {},
129
+ timeoutMs,
130
+ memoryMb: deps.memoryMb ?? DEFAULT_MEMORY_MB,
131
+ });
132
+
133
+ if (!result?.ok) {
134
+ throw new NexusBloomError(
135
+ result?.timedOut
136
+ ? `Local execution of "${slug}" timed out after ${timeoutMs}ms and was killed.`
137
+ : result?.error || `Local execution of "${slug}" failed.`,
138
+ // A timeout is worth retrying; a tool error is the tool's own answer and is
139
+ // not — retrying it unchanged is exactly what the error guidance says not to do.
140
+ result?.timedOut ? ErrorCode.TIMEOUT : ErrorCode.API_ERROR,
141
+ { retryable: Boolean(result?.timedOut) },
142
+ );
143
+ }
144
+
145
+ return { executed: true, data: result.result, source };
146
+ }
147
+
148
+ /**
149
+ * The startup warning, written to stderr when local execution is enabled.
150
+ *
151
+ * Printed once at startup rather than per call: an operator who has opted in does
152
+ * not need the warning on every run, but the person debugging a host that hangs
153
+ * later absolutely needs to know it was on.
154
+ */
155
+ export function localExecutionNotice(mode) {
156
+ if (mode === "remote") return "";
157
+ return (
158
+ `\nNexusBloom MCP: local execution is ENABLED (NEXUSBLOOM_MCP_EXECUTION=${mode}).\n` +
159
+ `Tool source is downloaded and executed on this machine in a child process. That isolates\n` +
160
+ `crashes, hangs and memory, but it does NOT sandbox the code: it runs with your user\n` +
161
+ `permissions and can read files and open network connections. Use remote execution for\n` +
162
+ `tools you do not trust.\n`
163
+ );
164
+ }