@nexusbloom/mcp-server 1.0.3 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +125 -0
- package/index.js +59 -397
- package/package.json +27 -3
- package/src/client.js +156 -0
- package/src/config.js +117 -0
- package/src/discovery.js +326 -0
- package/src/errors.js +107 -0
- package/src/handlers.js +426 -0
- package/src/manifests.js +247 -0
- package/src/render.js +348 -0
- package/src/server.js +83 -0
- package/src/validate.js +104 -0
- package/test-mcp-client.mjs +0 -104
package/src/client.js
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* HTTP client — the single egress point.
|
|
3
|
+
*
|
|
4
|
+
* Every request goes through here so that timeouts, auth headers, error
|
|
5
|
+
* decoding and diagnostics are applied uniformly. v1 had three separate `fetch`
|
|
6
|
+
* call sites and one of them (`main()`'s connectivity probe) sent no auth
|
|
7
|
+
* header and no timeout, so an unreachable API hung startup indefinitely
|
|
8
|
+
* instead of falling through to degraded mode.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { asNexusBloomError, codeForStatus, ErrorCode, NexusBloomError } from "./errors.js";
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* JSON logging to stderr.
|
|
15
|
+
*
|
|
16
|
+
* stdout is the MCP stdio transport: any byte written there that is not a
|
|
17
|
+
* JSON-RPC frame desynchronises the protocol and the client disconnects. This
|
|
18
|
+
* is the only sanctioned way to emit diagnostics.
|
|
19
|
+
*/
|
|
20
|
+
export function makeLogger(debug) {
|
|
21
|
+
return (...args) => {
|
|
22
|
+
if (!debug) return;
|
|
23
|
+
process.stderr.write(`[nexusbloom-mcp] ${args.map(String).join(" ")}\n`);
|
|
24
|
+
};
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
export class ApiClient {
|
|
28
|
+
/**
|
|
29
|
+
* @param {object} config From loadConfig().
|
|
30
|
+
* @param {object} [deps]
|
|
31
|
+
* @param {typeof fetch} [deps.fetchImpl] Injected in tests; never real network.
|
|
32
|
+
* @param {Function} [deps.now] Clock, injected so TTL is testable.
|
|
33
|
+
*/
|
|
34
|
+
constructor(config, deps = {}) {
|
|
35
|
+
this.config = config;
|
|
36
|
+
this.log = makeLogger(config.debug);
|
|
37
|
+
this.fetchImpl = deps.fetchImpl || globalThis.fetch;
|
|
38
|
+
this.now = deps.now || Date.now;
|
|
39
|
+
|
|
40
|
+
// Guards a hung request in environments where the AbortSignal timeout does
|
|
41
|
+
// not fire (some fetch polyfills). Whichever rejects first wins; the loser
|
|
42
|
+
// is an unhandled rejection we must not let crash the process.
|
|
43
|
+
this._inFlight = new Set();
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* Perform one request and decode JSON.
|
|
48
|
+
*
|
|
49
|
+
* @param {string} path Appended to the API base, e.g. "/tools".
|
|
50
|
+
* @param {object} [opts]
|
|
51
|
+
* @param {string} [opts.method]
|
|
52
|
+
* @param {*} [opts.body] Serialised as JSON when present.
|
|
53
|
+
* @param {boolean}[opts.auth] Send the API key. Default true.
|
|
54
|
+
* @returns {Promise<*>} Decoded body.
|
|
55
|
+
* @throws {NexusBloomError} Always, for any failure mode.
|
|
56
|
+
*/
|
|
57
|
+
async request(path, opts = {}) {
|
|
58
|
+
const { method = "GET", body, auth = true } = opts;
|
|
59
|
+
const url = `${this.config.apiBase}${path}`;
|
|
60
|
+
|
|
61
|
+
const headers = { Accept: "application/json" };
|
|
62
|
+
if (body !== undefined) headers["Content-Type"] = "application/json";
|
|
63
|
+
if (auth && this.config.apiKey) headers["Authorization"] = `Bearer ${this.config.apiKey}`;
|
|
64
|
+
|
|
65
|
+
// Own controller so an external signal can cancel too, and so the timer is
|
|
66
|
+
// always cleared — a leaked timer keeps the event loop alive and the MCP
|
|
67
|
+
// server never exits cleanly.
|
|
68
|
+
const controller = new AbortController();
|
|
69
|
+
const timer = setTimeout(() => controller.abort(), this.config.timeoutMs);
|
|
70
|
+
this._inFlight.add(controller);
|
|
71
|
+
|
|
72
|
+
const onExternalAbort = () => controller.abort();
|
|
73
|
+
if (opts.signal) {
|
|
74
|
+
if (opts.signal.aborted) controller.abort();
|
|
75
|
+
else opts.signal.addEventListener("abort", onExternalAbort, { once: true });
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
this.log(`${method} ${url}`);
|
|
79
|
+
|
|
80
|
+
let res;
|
|
81
|
+
try {
|
|
82
|
+
res = await this.fetchImpl(url, {
|
|
83
|
+
method,
|
|
84
|
+
headers,
|
|
85
|
+
body: body === undefined ? undefined : JSON.stringify(body),
|
|
86
|
+
signal: controller.signal,
|
|
87
|
+
});
|
|
88
|
+
} catch (err) {
|
|
89
|
+
throw asNexusBloomError(err);
|
|
90
|
+
} finally {
|
|
91
|
+
clearTimeout(timer);
|
|
92
|
+
this._inFlight.delete(controller);
|
|
93
|
+
if (opts.signal) opts.signal.removeEventListener?.("abort", onExternalAbort);
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
return this._decode(res, { url, method });
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* Turn a Response into a decoded body or a typed error.
|
|
101
|
+
*
|
|
102
|
+
* The API's own `code` wins over the status-derived guess — it distinguishes
|
|
103
|
+
* a 400 caused by a bad slug from one caused by a schema violation, and that
|
|
104
|
+
* is exactly the distinction an agent needs to pick its next move.
|
|
105
|
+
*/
|
|
106
|
+
async _decode(res, ctx) {
|
|
107
|
+
const text = await res.text().catch(() => "");
|
|
108
|
+
|
|
109
|
+
let parsed = null;
|
|
110
|
+
if (text) {
|
|
111
|
+
try {
|
|
112
|
+
parsed = JSON.parse(text);
|
|
113
|
+
} catch {
|
|
114
|
+
parsed = null;
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
if (!res.ok) {
|
|
119
|
+
const apiMessage =
|
|
120
|
+
parsed?.error || parsed?.message || parsed?.data?.error || text?.slice(0, 300);
|
|
121
|
+
const code = parsed?.code || codeForStatus(res.status);
|
|
122
|
+
|
|
123
|
+
const message = apiMessage?.trim()
|
|
124
|
+
? `${apiMessage}`
|
|
125
|
+
: `API returned ${res.status} for ${ctx.method} ${ctx.url}`;
|
|
126
|
+
|
|
127
|
+
throw new NexusBloomError(message, code, {
|
|
128
|
+
status: res.status,
|
|
129
|
+
// The API returns `fields` on schema failures; keep it so an agent can
|
|
130
|
+
// repair exactly the bad keys instead of re-reading the whole schema.
|
|
131
|
+
details: parsed?.details ?? parsed?.fields ?? null,
|
|
132
|
+
});
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
if (parsed === null) {
|
|
136
|
+
throw new NexusBloomError(
|
|
137
|
+
`API returned a non-JSON response (${res.status}) for ${ctx.method} ${ctx.url}`,
|
|
138
|
+
ErrorCode.API_ERROR,
|
|
139
|
+
{ status: res.status },
|
|
140
|
+
);
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
return parsed;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* Abort every in-flight request; used on shutdown.
|
|
148
|
+
*
|
|
149
|
+
* `AbortController.abort()` does not throw, and a controller whose request
|
|
150
|
+
* already settled is a no-op, so there is nothing to guard against here.
|
|
151
|
+
*/
|
|
152
|
+
abortAll() {
|
|
153
|
+
for (const controller of this._inFlight) controller.abort();
|
|
154
|
+
this._inFlight.clear();
|
|
155
|
+
}
|
|
156
|
+
}
|
package/src/config.js
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Configuration — resolved once, injected everywhere.
|
|
3
|
+
*
|
|
4
|
+
* The v1 server read `process.env` at module top level, which meant the tests
|
|
5
|
+
* could not exercise any branch: importing the module froze the configuration
|
|
6
|
+
* before a test had a chance to change it. Everything here is a function of an
|
|
7
|
+
* explicit env object so tests can construct any configuration they want.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* The API base differs from the CLI's on purpose, and historically by accident.
|
|
12
|
+
*
|
|
13
|
+
* The CLI stores `https://www.nexusbloom.dev` and appends `/api/...`. v1 stored
|
|
14
|
+
* `https://nexusbloom.dev/api` and appended `/tools`. Both forms are accepted
|
|
15
|
+
* here: `apiRoot()` normalises whatever the operator supplied into a base that
|
|
16
|
+
* already ends in `/api`, so every call site can just say `${root}/tools`.
|
|
17
|
+
*
|
|
18
|
+
* Non-www is the default because it is what the platform serves, but the `www`
|
|
19
|
+
* host is tried as a fallback by `candidateRoots()` — the two redirect
|
|
20
|
+
* differently in some networks and neither is reliably better.
|
|
21
|
+
*/
|
|
22
|
+
export const DEFAULT_API_HOST = "https://nexusbloom.dev";
|
|
23
|
+
|
|
24
|
+
/** Requests that outlive this are reported as a timeout, not a hang. */
|
|
25
|
+
export const DEFAULT_TIMEOUT_MS = 15_000;
|
|
26
|
+
|
|
27
|
+
/** Long enough to absorb a burst, short enough that a stale list self-heals. */
|
|
28
|
+
export const DEFAULT_CACHE_TTL_MS = 60_000;
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Normalise an operator-supplied base into something ending in `/api`.
|
|
32
|
+
*
|
|
33
|
+
* Accepts, and returns for each:
|
|
34
|
+
* (nothing) → https://nexusbloom.dev/api
|
|
35
|
+
* https://nexusbloom.dev → https://nexusbloom.dev/api
|
|
36
|
+
* https://nexusbloom.dev/ → https://nexusbloom.dev/api
|
|
37
|
+
* https://nexusbloom.dev/api → https://nexusbloom.dev/api
|
|
38
|
+
* https://nexusbloom.dev/api/ → https://nexusbloom.dev/api
|
|
39
|
+
* http://localhost:3000 → http://localhost:3000/api
|
|
40
|
+
*/
|
|
41
|
+
export function normaliseApiBase(input) {
|
|
42
|
+
const raw = (input || "").trim();
|
|
43
|
+
if (!raw) return `${DEFAULT_API_HOST}/api`;
|
|
44
|
+
|
|
45
|
+
let base = raw.replace(/\/+$/, "");
|
|
46
|
+
if (!base) return `${DEFAULT_API_HOST}/api`;
|
|
47
|
+
if (!/^https?:\/\//i.test(base)) base = `https://${base}`;
|
|
48
|
+
|
|
49
|
+
// A base that already carries the prefix is left alone, so both the host form
|
|
50
|
+
// and the full form converge on one string and callers never double it up.
|
|
51
|
+
if (!/\/api$/i.test(base)) base = `${base}/api`;
|
|
52
|
+
return base;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Every base worth trying, in preference order.
|
|
57
|
+
*
|
|
58
|
+
* Used only by the startup connectivity probe. Retrying the alternate host is
|
|
59
|
+
* what makes the server survive a `www`/non-`www` split in the operator's DNS.
|
|
60
|
+
*/
|
|
61
|
+
export function candidateRoots(env = {}) {
|
|
62
|
+
const primary = normaliseApiBase(env.NEXUSBLOOM_API_URL);
|
|
63
|
+
|
|
64
|
+
// Swap only the host's www prefix. Anything else (a port, a non-public host)
|
|
65
|
+
// has no meaningful alternate, so it is left alone rather than producing a
|
|
66
|
+
// second nonsense candidate.
|
|
67
|
+
const withoutApi = primary.replace(/\/api$/, "");
|
|
68
|
+
const bare = withoutApi.replace(/^https?:\/\//i, "");
|
|
69
|
+
const labels = bare.split(".");
|
|
70
|
+
|
|
71
|
+
// Only an apex domain (name.tld, optionally www-prefixed) has a meaningful
|
|
72
|
+
// www twin. A localhost, a port, or a subdomain such as api.example.com does
|
|
73
|
+
// not — prepending www to those invents a host that cannot resolve.
|
|
74
|
+
const isWww = labels[0] === "www";
|
|
75
|
+
const apexLabels = isWww ? labels.slice(1) : labels;
|
|
76
|
+
if (apexLabels.length !== 2 || /:\d+$/.test(bare) || bare === "localhost") {
|
|
77
|
+
return [primary];
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
const alternateHost = isWww ? apexLabels.join(".") : `www.${apexLabels.join(".")}`;
|
|
81
|
+
const scheme = withoutApi.match(/^(https?):\/\//i)?.[1] || "https";
|
|
82
|
+
const alternate = `${scheme}://${alternateHost}/api`;
|
|
83
|
+
|
|
84
|
+
return alternate === primary ? [primary] : [primary, alternate];
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* Build the immutable config object.
|
|
89
|
+
*
|
|
90
|
+
* Returns a fresh object every call — tests depend on there being no shared
|
|
91
|
+
* mutable module state to reset between cases.
|
|
92
|
+
*/
|
|
93
|
+
export function loadConfig(env = process.env) {
|
|
94
|
+
const apiKey = (env.NEXUSBLOOM_API_KEY || "").trim();
|
|
95
|
+
|
|
96
|
+
// Opt-in diagnostic logging. Always on stderr: stdout is the MCP transport and
|
|
97
|
+
// a single stray byte there corrupts the JSON-RPC framing.
|
|
98
|
+
const debug = /^(1|true|yes)$/i.test(env.NEXUSBLOOM_MCP_DEBUG || "");
|
|
99
|
+
|
|
100
|
+
const timeoutRaw = Number.parseInt(env.NEXUSBLOOM_MCP_TIMEOUT_MS || "", 10);
|
|
101
|
+
const timeoutMs =
|
|
102
|
+
Number.isFinite(timeoutRaw) && timeoutRaw > 0 ? timeoutRaw : DEFAULT_TIMEOUT_MS;
|
|
103
|
+
|
|
104
|
+
const ttlRaw = Number.parseInt(env.NEXUSBLOOM_MCP_CACHE_TTL_MS || "", 10);
|
|
105
|
+
const cacheTtlMs =
|
|
106
|
+
Number.isFinite(ttlRaw) && ttlRaw >= 0 ? ttlRaw : DEFAULT_CACHE_TTL_MS;
|
|
107
|
+
|
|
108
|
+
return {
|
|
109
|
+
apiKey,
|
|
110
|
+
apiBase: normaliseApiBase(env.NEXUSBLOOM_API_URL),
|
|
111
|
+
timeoutMs,
|
|
112
|
+
cacheTtlMs,
|
|
113
|
+
debug,
|
|
114
|
+
/** True when no key was supplied, which changes how limits are described. */
|
|
115
|
+
anonymous: apiKey === "",
|
|
116
|
+
};
|
|
117
|
+
}
|
package/src/discovery.js
ADDED
|
@@ -0,0 +1,326 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tool discovery — making 30 tools findable by an agent that only has a prose
|
|
3
|
+
* goal like "validate my environment file".
|
|
4
|
+
*
|
|
5
|
+
* An agent's problem is not that the catalogue is large; it is that slugs are
|
|
6
|
+
* opaque. `env-validator` tells an agent nothing about what to pass it. Ranking
|
|
7
|
+
* surfaces likely candidates first, and `suggest()` gives a wrong-slug error a
|
|
8
|
+
* recovery path instead of a dead end.
|
|
9
|
+
*
|
|
10
|
+
* Scoring is deliberately lexical, not embedding-based: it must run in a stdio
|
|
11
|
+
* process with no network, no model and no vector store, be deterministic, and
|
|
12
|
+
* return identical output for identical input so tests can assert on it. The
|
|
13
|
+
* fields below are chosen because they are what the manifest actually carries.
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
const STOP_WORDS = new Set([
|
|
17
|
+
"a", "an", "the", "my", "our", "your", "to", "for", "of", "in", "on", "with",
|
|
18
|
+
"and", "or", "is", "are", "be", "it", "that", "this", "from", "by", "at",
|
|
19
|
+
"me", "i", "we", "you", "can", "do", "does", "need", "want", "please", "help",
|
|
20
|
+
]);
|
|
21
|
+
|
|
22
|
+
/** Tokenise a phrase into comparable terms. */
|
|
23
|
+
export function tokenise(text) {
|
|
24
|
+
if (typeof text !== "string") return [];
|
|
25
|
+
return text
|
|
26
|
+
.toLowerCase()
|
|
27
|
+
.split(/[^a-z0-9]+/i)
|
|
28
|
+
.filter((t) => t.length > 1 && !STOP_WORDS.has(t));
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* The search text for one tool: everything a user might plausibly type.
|
|
33
|
+
*
|
|
34
|
+
* Input field names are included because agents search by what they want to
|
|
35
|
+
* *supply* ("the tool that takes a regex pattern"), not only by what it does.
|
|
36
|
+
*/
|
|
37
|
+
function haystack(tool) {
|
|
38
|
+
return [
|
|
39
|
+
tool.slug,
|
|
40
|
+
tool.slug.replace(/-/g, " "),
|
|
41
|
+
tool.name,
|
|
42
|
+
tool.short_description,
|
|
43
|
+
tool.category,
|
|
44
|
+
...(tool.tags || []),
|
|
45
|
+
...Object.keys(tool.input_schema?.properties || {}),
|
|
46
|
+
]
|
|
47
|
+
.filter(Boolean)
|
|
48
|
+
.join(" ");
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Score one tool against search terms. Higher is better; 0 means no match.
|
|
53
|
+
*
|
|
54
|
+
* Weights encode where a match is more trustworthy: the slug is authored, the
|
|
55
|
+
* description is authored, the tags are authored, and input field names are
|
|
56
|
+
* structural. A slug match therefore outranks a description match of the same
|
|
57
|
+
* term.
|
|
58
|
+
*/
|
|
59
|
+
export function scoreTool(tool, terms) {
|
|
60
|
+
if (!terms.length) return 0;
|
|
61
|
+
|
|
62
|
+
const slug = (tool.slug || "").toLowerCase();
|
|
63
|
+
const name = (tool.name || "").toLowerCase();
|
|
64
|
+
const desc = (tool.short_description || "").toLowerCase();
|
|
65
|
+
const tags = (tool.tags || []).map((t) => t.toLowerCase());
|
|
66
|
+
const fields = Object.keys(tool.input_schema?.properties || {}).map((f) => f.toLowerCase());
|
|
67
|
+
const all = haystack(tool).toLowerCase();
|
|
68
|
+
|
|
69
|
+
let score = 0;
|
|
70
|
+
for (const term of terms) {
|
|
71
|
+
let best = 0;
|
|
72
|
+
|
|
73
|
+
// Exact slug token, e.g. "cron" in "cron-expression-builder".
|
|
74
|
+
if (slug === term) best = Math.max(best, 100);
|
|
75
|
+
else if (slug.split("-").includes(term)) best = Math.max(best, 45);
|
|
76
|
+
else if (slug.includes(term)) best = Math.max(best, 30);
|
|
77
|
+
|
|
78
|
+
if (name.split(/\s+/).includes(term)) best = Math.max(best, 35);
|
|
79
|
+
else if (name.includes(term)) best = Math.max(best, 20);
|
|
80
|
+
|
|
81
|
+
if (tags.includes(term)) best = Math.max(best, 30);
|
|
82
|
+
else if (tags.some((t) => t.includes(term))) best = Math.max(best, 15);
|
|
83
|
+
|
|
84
|
+
if (fields.includes(term)) best = Math.max(best, 25);
|
|
85
|
+
else if (fields.some((f) => f.includes(term))) best = Math.max(best, 12);
|
|
86
|
+
|
|
87
|
+
if (desc.includes(term)) best = Math.max(best, 18);
|
|
88
|
+
if (all.includes(term)) best = Math.max(best, 8);
|
|
89
|
+
|
|
90
|
+
// Morphological variants. Slugs are terse by convention, so an agent
|
|
91
|
+
// searching "environment" should still reach "env-validator", and
|
|
92
|
+
// "validation" should reach "validate". Weighted below a real match so an
|
|
93
|
+
// exact hit always outranks a stemmed one.
|
|
94
|
+
if (best === 0 && sharesStem(term, slug, name, tags, desc)) {
|
|
95
|
+
best = Math.max(best, 14);
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
// Every term must land somewhere: an AND of ORs. Without this a tool
|
|
99
|
+
// matching one common word outranks one matching the whole query.
|
|
100
|
+
if (best === 0) return 0;
|
|
101
|
+
score += best;
|
|
102
|
+
}
|
|
103
|
+
return score;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/** Minimum shared prefix before a stem comparison counts as evidence. */
|
|
107
|
+
const STEM_MIN = 4;
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Do a term and any vocabulary word share a meaningful prefix?
|
|
111
|
+
*
|
|
112
|
+
* "environment" and "env" share three characters, which is below the threshold,
|
|
113
|
+
* so the match comes from `all.includes` via the longer direction — this checks
|
|
114
|
+
* the other way too, so "env" finds "environment". A three-character floor keeps
|
|
115
|
+
* "e" from matching everything.
|
|
116
|
+
*/
|
|
117
|
+
function sharesStem(term, slug, name, tags, desc) {
|
|
118
|
+
const words = `${slug} ${name} ${tags.join(" ")} ${desc}`.split(/[^a-z0-9]+/i).filter(Boolean);
|
|
119
|
+
const t = term.toLowerCase();
|
|
120
|
+
for (const w of words) {
|
|
121
|
+
if (w.length < 3 || t.length < 3) continue;
|
|
122
|
+
const shared = w.startsWith(t) || t.startsWith(w) ? Math.min(w.length, t.length) : 0;
|
|
123
|
+
if (shared >= STEM_MIN) return true;
|
|
124
|
+
// Slug segments like "env" inside "env-validator" are whole words, and a
|
|
125
|
+
// longer term containing them is a morphological variant.
|
|
126
|
+
if (t.length >= STEM_MIN && w.length >= 3 && (t.includes(w) || w.includes(t))) return true;
|
|
127
|
+
}
|
|
128
|
+
return false;
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* Rank tools against a query.
|
|
133
|
+
*
|
|
134
|
+
* @param {object[]} tools
|
|
135
|
+
* @param {string} query
|
|
136
|
+
* @param {object} [opts]
|
|
137
|
+
* @param {number} [opts.limit] Max results. Default 10.
|
|
138
|
+
* @param {boolean} [opts.includeAll] Return everything, unscored, when query is empty.
|
|
139
|
+
*/
|
|
140
|
+
export function searchTools(tools, query, opts = {}) {
|
|
141
|
+
const { limit = 10, includeAll = true } = opts;
|
|
142
|
+
const terms = tokenise(query);
|
|
143
|
+
|
|
144
|
+
if (terms.length === 0) {
|
|
145
|
+
return includeAll ? tools.slice(0, limit) : [];
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
return tools
|
|
149
|
+
.map((tool) => ({ tool, score: scoreTool(tool, terms) }))
|
|
150
|
+
.filter((r) => r.score > 0)
|
|
151
|
+
// Ties broken by slug so ordering is stable across runs and processes.
|
|
152
|
+
.sort((a, b) => b.score - a.score || a.tool.slug.localeCompare(b.tool.slug))
|
|
153
|
+
.slice(0, limit)
|
|
154
|
+
.map((r) => r.tool);
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* Resolve a slug for **execution**.
|
|
159
|
+
*
|
|
160
|
+
* Deliberately stricter than `resolveSlug`: only an exact slug or an
|
|
161
|
+
* unambiguous prefix is accepted. Fuzzy recovery — edit distance, scored search —
|
|
162
|
+
* is available when *describing* a tool, where guessing costs nothing, but
|
|
163
|
+
* running `env-validatr` and silently executing `env-validator` could hand an
|
|
164
|
+
* agent a result from a tool it never chose. Here a near miss becomes an error
|
|
165
|
+
* carrying the suggestion instead.
|
|
166
|
+
*/
|
|
167
|
+
export function resolveSlugStrict(tools, query) {
|
|
168
|
+
const q = (query || "").trim().toLowerCase();
|
|
169
|
+
if (!q) return { tool: null, ambiguous: [] };
|
|
170
|
+
|
|
171
|
+
const exact = tools.find((t) => t.slug.toLowerCase() === q);
|
|
172
|
+
if (exact) return { tool: exact, ambiguous: [] };
|
|
173
|
+
|
|
174
|
+
const prefix = tools.filter((t) => t.slug.toLowerCase().startsWith(q));
|
|
175
|
+
if (prefix.length === 1) return { tool: prefix[0], ambiguous: [] };
|
|
176
|
+
if (prefix.length > 1) return { tool: null, ambiguous: prefix.slice(0, 5) };
|
|
177
|
+
|
|
178
|
+
// Abbreviation that is not a slug prefix but a unique token match, e.g. "env"
|
|
179
|
+
// for "env-validator" is a prefix, but "validator" is not. Accept only when
|
|
180
|
+
// exactly one tool claims it.
|
|
181
|
+
const searched = searchTools(tools, q, { limit: 2, includeAll: false });
|
|
182
|
+
if (searched.length === 1) return { tool: searched[0], ambiguous: [] };
|
|
183
|
+
|
|
184
|
+
return { tool: null, ambiguous: searched.slice(0, 5) };
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
/**
|
|
188
|
+
* Resolve a possibly-abbreviated or partial slug to a real tool.
|
|
189
|
+
*
|
|
190
|
+
* Exact slug wins. Otherwise the shortest tool whose slug starts with the
|
|
191
|
+
* fragment is returned, which is what makes `nxb`'s abbreviations work here
|
|
192
|
+
* too — an agent that read one slug in a previous turn should be able to
|
|
193
|
+
* abbreviate it in the next.
|
|
194
|
+
*
|
|
195
|
+
* @returns {{tool: object|null, ambiguous: object[]}}
|
|
196
|
+
* `ambiguous` is non-empty when the fragment matches several tools closely;
|
|
197
|
+
* the caller should ask rather than guess.
|
|
198
|
+
*/
|
|
199
|
+
export function resolveSlug(tools, query) {
|
|
200
|
+
const q = (query || "").trim().toLowerCase();
|
|
201
|
+
if (!q) return { tool: null, ambiguous: [] };
|
|
202
|
+
|
|
203
|
+
const exact = tools.find((t) => t.slug.toLowerCase() === q);
|
|
204
|
+
if (exact) return { tool: exact, ambiguous: [] };
|
|
205
|
+
|
|
206
|
+
// Prefix match. A truncated prefix that fits several tools is genuinely
|
|
207
|
+
// ambiguous, and guessing would silently run the wrong tool — so it is
|
|
208
|
+
// reported instead. ("cron-v" resolves; "cron-" does not.)
|
|
209
|
+
const prefix = tools.filter((t) => t.slug.toLowerCase().startsWith(q));
|
|
210
|
+
if (prefix.length === 1) return { tool: prefix[0], ambiguous: [] };
|
|
211
|
+
if (prefix.length > 1) return { tool: null, ambiguous: prefix.slice(0, 5) };
|
|
212
|
+
|
|
213
|
+
// Scored search, so a prose reference or a morphological variant lands.
|
|
214
|
+
const searched = searchTools(tools, q, { limit: 5, includeAll: false });
|
|
215
|
+
if (searched.length === 1) return { tool: searched[0], ambiguous: [] };
|
|
216
|
+
|
|
217
|
+
// Typo recovery. `env-validatr` has no matching term at all, so the search
|
|
218
|
+
// above finds nothing; edit distance is the only signal that can help.
|
|
219
|
+
// A strictly closer candidate is treated as the answer — "env-validatr" is
|
|
220
|
+
// one edit from "env-validator" and five from "cron-validator", so the
|
|
221
|
+
// runner-up is noise rather than a genuine ambiguity.
|
|
222
|
+
const ranked = rankByDistance(tools, q, 5);
|
|
223
|
+
if (ranked.length > 0 && (ranked.length === 1 || ranked[0].distance < ranked[1].distance)) {
|
|
224
|
+
return { tool: ranked[0].tool, ambiguous: [] };
|
|
225
|
+
}
|
|
226
|
+
if (ranked.length > 1) {
|
|
227
|
+
return { tool: null, ambiguous: ranked.slice(0, 5).map((r) => r.tool) };
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
return { tool: null, ambiguous: searched.slice(0, 5) };
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
/**
|
|
234
|
+
* Tools ordered by edit distance to a query, nearest first.
|
|
235
|
+
*
|
|
236
|
+
* Exported separately from `suggest` because resolution needs the distances to
|
|
237
|
+
* reason about ties, while an error message only needs the names.
|
|
238
|
+
*/
|
|
239
|
+
export function rankByDistance(tools, query, limit = 3) {
|
|
240
|
+
const q = (query || "").trim().toLowerCase();
|
|
241
|
+
if (!q) return [];
|
|
242
|
+
|
|
243
|
+
return tools
|
|
244
|
+
.map((tool) => ({ tool, distance: levenshtein(q, tool.slug.toLowerCase()) }))
|
|
245
|
+
.filter((r) => r.distance <= Math.max(2, Math.ceil(r.tool.slug.length / 3)))
|
|
246
|
+
.sort((a, b) => a.distance - b.distance || a.tool.slug.localeCompare(b.tool.slug))
|
|
247
|
+
.slice(0, limit);
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
/**
|
|
251
|
+
* Near-miss suggestions for an error message.
|
|
252
|
+
*
|
|
253
|
+
* Ordered by edit distance against the slug, so the closest few are offered
|
|
254
|
+
* rather than an arbitrary slice of the catalogue.
|
|
255
|
+
*/
|
|
256
|
+
export function suggest(tools, query, limit = 3) {
|
|
257
|
+
return rankByDistance(tools, query, limit).map((r) => r.tool);
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
/** Levenshtein distance, two-row variant. */
|
|
261
|
+
export function levenshtein(a, b) {
|
|
262
|
+
if (a === b) return 0;
|
|
263
|
+
if (!a.length) return b.length;
|
|
264
|
+
if (!b.length) return a.length;
|
|
265
|
+
|
|
266
|
+
let prev = Array.from({ length: b.length + 1 }, (_, i) => i);
|
|
267
|
+
let curr = new Array(b.length + 1);
|
|
268
|
+
|
|
269
|
+
for (let i = 1; i <= a.length; i++) {
|
|
270
|
+
curr[0] = i;
|
|
271
|
+
for (let j = 1; j <= b.length; j++) {
|
|
272
|
+
const cost = a[i - 1] === b[j - 1] ? 0 : 1;
|
|
273
|
+
curr[j] = Math.min(curr[j - 1] + 1, prev[j] + 1, prev[j - 1] + cost);
|
|
274
|
+
}
|
|
275
|
+
[prev, curr] = [curr, prev];
|
|
276
|
+
}
|
|
277
|
+
return prev[b.length];
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
/**
|
|
281
|
+
* Group tools by category, for a browsable overview.
|
|
282
|
+
*
|
|
283
|
+
* Sorted by group size descending so the largest, most useful categories lead.
|
|
284
|
+
*/
|
|
285
|
+
export function groupByCategory(tools) {
|
|
286
|
+
const map = new Map();
|
|
287
|
+
for (const tool of tools) {
|
|
288
|
+
const key = tool.category || "uncategorized";
|
|
289
|
+
if (!map.has(key)) map.set(key, []);
|
|
290
|
+
map.get(key).push(tool);
|
|
291
|
+
}
|
|
292
|
+
return [...map.entries()]
|
|
293
|
+
.map(([category, items]) => ({ category, tools: items }))
|
|
294
|
+
.sort((a, b) => b.tools.length - a.tools.length || a.category.localeCompare(b.category));
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
/**
|
|
298
|
+
* Summarise a tool's input contract in one line each, for a schema response.
|
|
299
|
+
*
|
|
300
|
+
* Reuses core's `describeSchema` so the wording matches the CLI and VS Code
|
|
301
|
+
* extension exactly — one rule, three surfaces.
|
|
302
|
+
*/
|
|
303
|
+
export function describeParameters(inputSchema) {
|
|
304
|
+
const properties = inputSchema?.properties || {};
|
|
305
|
+
const required = new Set(inputSchema?.required || []);
|
|
306
|
+
|
|
307
|
+
return Object.entries(properties).map(([key, prop]) => {
|
|
308
|
+
const bits = [];
|
|
309
|
+
bits.push(typeWord(prop));
|
|
310
|
+
if (required.has(key)) bits.push("required");
|
|
311
|
+
if (prop?.enum) bits.push(`one of: ${prop.enum.map((v) => JSON.stringify(v)).join(", ")}`);
|
|
312
|
+
if (prop?.default !== undefined) bits.push(`default: ${JSON.stringify(prop.default)}`);
|
|
313
|
+
if (typeof prop?.minLength === "number") bits.push(`minLength: ${prop.minLength}`);
|
|
314
|
+
if (typeof prop?.maxLength === "number") bits.push(`maxLength: ${prop.maxLength}`);
|
|
315
|
+
if (typeof prop?.minimum === "number") bits.push(`minimum: ${prop.minimum}`);
|
|
316
|
+
if (typeof prop?.maximum === "number") bits.push(`maximum: ${prop.maximum}`);
|
|
317
|
+
if (typeof prop?.pattern === "string") bits.push(`pattern: ${prop.pattern}`);
|
|
318
|
+
return { name: key, summary: bits.join(", "), description: prop?.description || "", required: required.has(key) };
|
|
319
|
+
});
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
function typeWord(prop) {
|
|
323
|
+
if (!prop?.type) return "any";
|
|
324
|
+
if (Array.isArray(prop.type)) return prop.type.join(" | ");
|
|
325
|
+
return prop.type;
|
|
326
|
+
}
|