@nexusbloom/mcp-server 2.0.0 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/prompts.js ADDED
@@ -0,0 +1,282 @@
1
+ /**
2
+ * MCP Prompts — workflow recipes a host can offer the user as one-click starts.
3
+ *
4
+ * A prompt here is not a canned instruction. Each one is assembled against the
5
+ * live catalogue, so `use-tool` carries *this* deployment's parameter names and a
6
+ * call that is already valid, and `find-tool` carries the tools that actually
7
+ * exist rather than a description of tools that might. That is the difference
8
+ * between a prompt that saves the user a turn and one that just adds words to
9
+ * their context.
10
+ *
11
+ * Kept SDK-free, like handlers.js and resources.js: `prompts/list` and
12
+ * `prompts/get` are plain async functions, and server.js is the only place the
13
+ * protocol lives.
14
+ */
15
+
16
+ import { searchTools, suggest } from "./discovery.js";
17
+ import { ErrorCode, NexusBloomError } from "./errors.js";
18
+ import { buildExampleArgs, formatParameterList } from "./render.js";
19
+ import { CATALOGUE_URI, toolUri } from "./resources.js";
20
+
21
+ /** How many candidates a prompt embeds. Past this it is a catalogue dump. */
22
+ const MAX_CANDIDATES = 8;
23
+
24
+ /**
25
+ * The advertised prompts.
26
+ *
27
+ * Four, in the order a user meets them: find the tool, use the tool, plan a
28
+ * multi-step task, recover from a failure.
29
+ */
30
+ export const PROMPT_DEFINITIONS = [
31
+ {
32
+ name: "find-tool",
33
+ title: "Find a NexusBloom tool",
34
+ description: "Locate the tools that can accomplish a stated goal, ranked by intent.",
35
+ arguments: [{ name: "goal", description: "What you want to accomplish, in plain language.", required: true }],
36
+ },
37
+ {
38
+ name: "use-tool",
39
+ title: "Use a specific NexusBloom tool",
40
+ description: "Get the exact parameters, a valid example call, and the recovery paths for one tool.",
41
+ arguments: [{ name: "slug", description: "Tool slug, e.g. env-validator.", required: true }],
42
+ },
43
+ {
44
+ name: "plan-batch",
45
+ title: "Plan a multi-tool task",
46
+ description: "Break a task into NexusBloom runs and shape them into one batched call.",
47
+ arguments: [{ name: "task", description: "The task to accomplish, in plain language.", required: true }],
48
+ },
49
+ {
50
+ name: "recover",
51
+ title: "Recover from a NexusBloom failure",
52
+ description: "Explain what a failure means and what to do next.",
53
+ arguments: [
54
+ { name: "error", description: "The error message or code you received.", required: false },
55
+ { name: "slug", description: "The tool you were calling, if any.", required: false },
56
+ ],
57
+ },
58
+ ];
59
+
60
+ /**
61
+ * Assemble the prompt handlers over a manifest cache.
62
+ *
63
+ * @param {object} deps
64
+ * @param {import("./manifests.js").ManifestCache} deps.cache
65
+ * @returns {{listPrompts: Function, getPrompt: Function, names: string[]}}
66
+ */
67
+ export function createPrompts({ cache }) {
68
+ async function listPrompts() {
69
+ return { prompts: PROMPT_DEFINITIONS };
70
+ }
71
+
72
+ async function getPrompt(params) {
73
+ const name = params?.name;
74
+ const args = params?.arguments ?? {};
75
+
76
+ switch (name) {
77
+ case "find-tool":
78
+ return text("Tools matching your goal, with a valid call for the best one", await findToolPrompt(requireArg(args, "goal", name), cache));
79
+ case "use-tool":
80
+ return text("One tool's exact parameters and a valid call", await useToolPrompt(requireArg(args, "slug", name), cache));
81
+ case "plan-batch":
82
+ return text("A batched plan for a multi-tool task", await planBatchPrompt(requireArg(args, "task", name), cache));
83
+ case "recover":
84
+ return text("What a failure means and what to do next", await recoverPrompt(args, cache));
85
+ default:
86
+ throw new NexusBloomError(
87
+ `Unknown prompt "${name ?? ""}". Available: ${PROMPT_DEFINITIONS.map((p) => p.name).join(", ")}.`,
88
+ ErrorCode.NOT_FOUND,
89
+ );
90
+ }
91
+ }
92
+
93
+ return { listPrompts, getPrompt, names: PROMPT_DEFINITIONS.map((p) => p.name) };
94
+ }
95
+
96
+ /**
97
+ * Wrap body text as a single user message.
98
+ *
99
+ * `user`, not `assistant`: a prompt is something the host hands to the model as
100
+ * the human's request, and getting the role wrong makes a host either suppress it
101
+ * or render it as the model's own prior output.
102
+ */
103
+ function text(description, body) {
104
+ return { description, messages: [{ role: "user", content: { type: "text", text: body } }] };
105
+ }
106
+
107
+ /** A required argument, with the prompt name in the error so the host can show it. */
108
+ function requireArg(args, key, prompt) {
109
+ const value = (args?.[key] ?? "").trim();
110
+ if (!value) {
111
+ throw new NexusBloomError(
112
+ `The "${prompt}" prompt needs a "${key}" argument.`,
113
+ ErrorCode.INVALID_ARGS,
114
+ );
115
+ }
116
+ return value;
117
+ }
118
+
119
+ /** One line per candidate, reusing the description an agent would see in a list. */
120
+ function candidateLines(tools) {
121
+ return tools
122
+ .map((t) => {
123
+ const required = t.input_schema?.required || [];
124
+ const shape = required.length
125
+ ? `${required.length} required of ${Object.keys(t.input_schema?.properties || {}).length} params`
126
+ : "no required params";
127
+ return `- \`${t.slug}\` — ${t.short_description || "No description"} (${shape})`;
128
+ })
129
+ .join("\n");
130
+ }
131
+
132
+ async function findToolPrompt(goal, cache) {
133
+ const tools = await cache.tools();
134
+ const ranked = searchTools(tools, goal, { limit: MAX_CANDIDATES, includeAll: false });
135
+
136
+ const body = ranked.length
137
+ ? `Ranked by how well they match "${goal}":\n\n${candidateLines(ranked)}`
138
+ : `Nothing in the catalogue scored against "${goal}". The catalogue holds ${tools.length} tools; ` +
139
+ `try broader wording (drop the tool-specific noun), or read ${CATALOGUE_URI} and pick from the list.`;
140
+
141
+ return `Find the NexusBloom tool that accomplishes this goal, then use it.
142
+
143
+ **Goal:** ${goal}
144
+
145
+ ${body}
146
+
147
+ Do this:
148
+ 1. Pick the best match above. If two are plausible, read both manifests before choosing.
149
+ 2. Read the winner's full manifest at \`${toolUri("<slug>")}\` — it has the exact parameter names and a valid example.
150
+ 3. Call the tool directly by slug with your own values.
151
+ 4. If a candidate looks wrong, do not guess a new slug. Run \`{"command":"search","query":"…"}\` with different wording.
152
+
153
+ Catalogue: ${CATALOGUE_URI}`;
154
+ }
155
+
156
+ async function useToolPrompt(slug, cache) {
157
+ const tools = await cache.tools();
158
+ const match = exactMatch(tools, slug);
159
+
160
+ if (!match) {
161
+ // Same recovery the tool path gives, so a user who picked the wrong prompt
162
+ // lands on the same next step.
163
+ const near = suggest(tools, slug).map((t) => t.slug);
164
+ return `No tool named "${slug}" is published.${near.length ? ` Did you mean: ${near.join(", ")}?` : ""}
165
+
166
+ The catalogue holds ${tools.length} tools. Read ${CATALOGUE_URI} for the full list, or call the
167
+ \`nexusbloom\` meta-tool with \`{"command":"search","query":"<what you need>"}\` to rank by intent.
168
+
169
+ Then call the tool you choose directly by slug.`;
170
+ }
171
+
172
+ const required = match.input_schema?.required || [];
173
+ const example = buildExampleArgs(match);
174
+
175
+ return `Use the \`${match.slug}\` tool.
176
+
177
+ **${match.name}** — ${match.short_description || "No description"}
178
+ ${match.category && match.category !== "uncategorized" ? `Category: ${match.category}\n` : ""}${match.tags?.length ? `Tags: ${match.tags.join(", ")}\n` : ""}
179
+ ${required.length === 0 ? "Takes no required parameters." : `Required: ${required.join(", ")}`}
180
+
181
+ Parameters:
182
+ ${formatParameterList(match.input_schema)}
183
+
184
+ Valid call — replace the example values, keep the shape:
185
+ \`\`\`json
186
+ { "name": "${match.slug}", "arguments": ${JSON.stringify(example, null, 2)} }
187
+ \`\`\`
188
+
189
+ ${match.output_schema ? `Expected output shape:\n\`\`\`json\n${JSON.stringify(match.output_schema, null, 2)}\n\`\`\`\n` : ""}If a call fails, read the error rather than retrying it: a validation failure names the
190
+ offending field, an unknown slug suggests the right one, and a timeout is marked
191
+ \`retryable\`. Full manifest: ${toolUri(match.slug)}`;
192
+ }
193
+
194
+ async function planBatchPrompt(task, cache) {
195
+ const tools = await cache.tools();
196
+ const candidates = searchTools(tools, task, { limit: MAX_CANDIDATES, includeAll: false });
197
+ const relevant = candidates.length ? candidates : tools.slice(0, MAX_CANDIDATES);
198
+
199
+ return `Plan and run this task with NexusBloom, batching what belongs together.
200
+
201
+ **Task:** ${task}
202
+
203
+ Likely participants:
204
+ ${candidateLines(relevant)}
205
+
206
+ How to plan it:
207
+ 1. Decide what the task actually requires. A task that one tool covers should be a direct call — batching one run only adds framing.
208
+ 2. Read each chosen tool's manifest at \`${toolUri("<slug>")}\` and collect its required parameters *before* running anything.
209
+ 3. Put the runs in one call, in dependency order, when two or more tools contribute:
210
+
211
+ \`\`\`json
212
+ { "command": "batch", "runs": [
213
+ { "slug": "<slug>", "params": { … } },
214
+ { "slug": "<slug>", "params": { … } }
215
+ ] }
216
+ \`\`\`
217
+
218
+ 4. Read the tally first. Failures are listed above successes, and each is independent — fix those inputs and re-run only those slugs.
219
+
220
+ Two rules that are not obvious:
221
+ - A batch is validated as a whole before anything runs, so an unresolvable slug or a missing required
222
+ field costs zero quota and rejects the batch. Get the arguments right first.
223
+ - Maximum 10 runs per batch. More than that and anonymous callers hit the 30 requests/minute cap.
224
+
225
+ Catalogue: ${CATALOGUE_URI}`;
226
+ }
227
+
228
+ async function recoverPrompt({ error, slug }, cache) {
229
+ const parts = [`A NexusBloom call failed. Work out why before retrying — most failures need a different call, not the same one again.`];
230
+
231
+ if (error) parts.push(`**Reported:** ${error}`);
232
+
233
+ if (slug) {
234
+ const tools = await cache.tools();
235
+ const match = exactMatch(tools, slug);
236
+ if (match) {
237
+ const example = buildExampleArgs(match);
238
+ const required = match.input_schema?.required || [];
239
+ parts.push(
240
+ `**Tool:** \`${match.slug}\`\n\nIts exact parameters:\n${formatParameterList(match.input_schema)}\n\n` +
241
+ `Valid call shape:\n\`\`\`json\n{ "name": "${match.slug}", "arguments": ${JSON.stringify(example, null, 2)} }\n\`\`\`` +
242
+ (required.length
243
+ ? `\n\nEvery one of these is required: ${required.join(", ")}.`
244
+ : "\n\nThis tool has no required parameters, so a failure here is about the value, not a missing field."),
245
+ );
246
+ } else {
247
+ const near = suggest(tools, slug).map((t) => t.slug);
248
+ parts.push(
249
+ `**Tool:** \`${slug}\` is not in the catalogue.` +
250
+ (near.length ? ` Closest: ${near.join(", ")}.` : "") +
251
+ ` Read ${CATALOGUE_URI} or search by intent.`,
252
+ );
253
+ }
254
+ }
255
+
256
+ parts.push(
257
+ `## What each code means
258
+
259
+ | Code | Meaning | What to do |
260
+ |---|---|---|
261
+ | \`TOOL_NOT_FOUND\` | The slug is not published, or an abbreviation matched several tools. | Use the candidate the error named, or \`{"command":"search","query":"…"}\`. |
262
+ | \`VALIDATION_FAILED\` | Input did not match the tool's schema. | Fix the named field. The message lists every missing one. Do not retry unchanged. |
263
+ | \`TIMEOUT\` | The API did not answer in time. | \`retryable\`. Retry once; if it times out again, the payload is probably too large. |
264
+ | \`RATE_LIMITED\` | 30 requests/minute, anonymous only. | Pace down. Fewer, larger calls: batch rather than looping. |
265
+ | \`UNAUTHORIZED\` | A key is set but invalid. | Check \`NEXUSBLOOM_API_KEY\`. |
266
+ | \`API_ERROR\` | The API failed. | \`retryable\`. If it persists, the deployment is down — say so rather than looping. |
267
+
268
+ ## The rules that save the most time
269
+
270
+ - Never guess a slug. Search by intent, or read ${CATALOGUE_URI}.
271
+ - A tool-reported \`{"success":false}\` is the tool's own answer, not a transport failure — the call worked.
272
+ - Read \`{"command":"schema","slug":"…"}\` before retrying a validation failure; do not re-derive the schema from memory.`,
273
+ );
274
+
275
+ return parts.join("\n\n");
276
+ }
277
+
278
+ /** Exact slug match only. A prompt is a deliberate choice; do not auto-resolve it. */
279
+ function exactMatch(tools, slug) {
280
+ const wanted = (slug || "").trim().toLowerCase();
281
+ return tools.find((t) => t.slug.toLowerCase() === wanted) || null;
282
+ }
package/src/render.js CHANGED
@@ -17,6 +17,7 @@
17
17
  import { describeParameters, groupByCategory } from "./discovery.js";
18
18
  import { ErrorCode } from "./errors.js";
19
19
  import { normaliseTool } from "./manifests.js";
20
+ import { previewResult } from "./preview.js";
20
21
 
21
22
  /**
22
23
  * Build an MCP response.
@@ -26,7 +27,9 @@ import { normaliseTool } from "./manifests.js";
26
27
  * over prose gets an exact payload instead of parsing markdown.
27
28
  */
28
29
  export function respond(payload) {
29
- const content = [{ type: "text", text: payload.text }];
30
+ // A renderer may supply its own block list (tier-1 preview). When it does we
31
+ // still append the JSON resource so the model always has the raw payload.
32
+ const content = Array.isArray(payload.content) ? [...payload.content] : [{ type: "text", text: payload.text }];
30
33
  if (payload.data !== undefined) {
31
34
  content.push({
32
35
  type: "resource",
@@ -116,9 +119,10 @@ export function renderToolList(tools, { total } = {}) {
116
119
  */
117
120
  export function renderSearchResults(query, tools, catalogueSize, categories = []) {
118
121
  if (tools.length === 0) {
119
- const categoryHint = categories.length
120
- ? `\n• A category name: ${categories.slice(0, 8).join(", ")}`
121
- : "";
122
+ // "uncategorized" is a placeholder, not a category an agent can search by,
123
+ // so offering it as a hint wastes one of the three suggestions.
124
+ const useful = [...new Set(categories.filter((c) => c && c !== "uncategorized"))];
125
+ const categoryHint = useful.length ? `\n• A category name: ${useful.slice(0, 8).join(", ")}` : "";
122
126
  return {
123
127
  text:
124
128
  `# No tools match "${query}"\n\n` +
@@ -155,18 +159,10 @@ export function renderSchema(rawSchema, tools = []) {
155
159
  // literal string "undefined".
156
160
  const schema = normaliseTool({ slug: "tool", ...rawSchema }) || normaliseTool({ slug: "tool" });
157
161
 
162
+ const paramLines = formatParameterList(schema.input_schema);
158
163
  const params = describeParameters(schema.input_schema);
159
164
  const required = params.filter((p) => p.required);
160
165
 
161
- const paramLines = params.length
162
- ? params
163
- .map((p) => {
164
- const detail = p.description ? ` — ${p.description}` : "";
165
- return `- \`${p.name}\` (${p.summary || "any"})${detail}`;
166
- })
167
- .join("\n")
168
- : "_This tool takes no parameters._";
169
-
170
166
  const exampleArgs = buildExampleArgs(schema);
171
167
 
172
168
  let example;
@@ -212,6 +208,24 @@ export function renderSchema(rawSchema, tools = []) {
212
208
  return { text, data: schema };
213
209
  }
214
210
 
211
+ /**
212
+ * Parameter bullet lines for an input schema.
213
+ *
214
+ * Exported because the `schema` response and the `use-tool` prompt both have to
215
+ * describe a tool's parameters, and two formatters would drift — a prompt that
216
+ * omits a `minLength` the tool actually enforces is worse than no prompt.
217
+ */
218
+ export function formatParameterList(inputSchema) {
219
+ const params = describeParameters(inputSchema);
220
+ if (!params.length) return "_This tool takes no parameters._";
221
+ return params
222
+ .map((p) => {
223
+ const detail = p.description ? ` — ${p.description}` : "";
224
+ return `- \`${p.name}\` (${p.summary || "any"})${detail}`;
225
+ })
226
+ .join("\n");
227
+ }
228
+
215
229
  /** Describe an output schema's top-level shape in one clause. */
216
230
  function describeOutputShape(schema) {
217
231
  const props = schema?.properties || {};
@@ -282,14 +296,145 @@ function placeholderFor(prop) {
282
296
  * re-read it, and deeply indented JSON costs tokens on every turn. Kept at 2
283
297
  * spaces because it is far easier to diff and reason about.
284
298
  */
285
- export function renderResult(slug, data, { durationMs, execution = "remote" } = {}) {
299
+ export function renderResult(slug, data, { durationMs, execution = "remote", rateLimit } = {}) {
286
300
  const json = safeStringify(data);
287
301
  const meta = [`Tool: \`${slug}\``, `Execution: ${execution}`];
288
302
  if (typeof durationMs === "number") meta.push(`Duration: ${durationMs}ms`);
289
303
 
290
- return {
304
+ // Anonymous execution is capped at 30 requests/minute. Showing the remaining
305
+ // budget lets an agent pace itself instead of discovering the cap as a 429.
306
+ if (rateLimit && Number.isFinite(rateLimit.remaining)) {
307
+ const limit = Number.isFinite(rateLimit.limit) ? `/${rateLimit.limit}` : "";
308
+ const warns = rateLimit.remaining <= 3 ? " — **pace down, further calls will 429**" : "";
309
+ meta.push(`Quota: ${rateLimit.remaining}${limit} left${warns}`);
310
+ }
311
+
312
+ // Tier-1 preview. Value-shape inference only — never the slug — so tools
313
+ // that do not exist yet still render. Returns null when nothing matches, in
314
+ // which case the response is exactly what it was before.
315
+ const preview = previewResult(data, { slug });
316
+
317
+ // The visual is addressed to the human; the JSON to the model. That split is
318
+ // what makes rich output affordable: an image in context costs roughly 1.5k
319
+ // tokens, so showing one unconditionally would make the agent worse.
320
+ const text = {
321
+ type: "text",
291
322
  text: `# Result: ${slug}\n\n${meta.join(" · ")}\n\n\`\`\`json\n${json}\n\`\`\``,
323
+ annotations: { audience: preview?.block ? ["assistant"] : ["user", "assistant"] },
324
+ };
325
+
326
+ const content = preview?.block ? [text, preview.block] : [text];
327
+
328
+ return {
329
+ text: text.text,
292
330
  data,
331
+ content,
332
+ previewKind: preview?.kind ?? null,
333
+ };
334
+ }
335
+
336
+ /**
337
+ * The recorded run log.
338
+ *
339
+ * Results are omitted here: this is an index for deciding which two runs to
340
+ * compare, and inlining fifty payloads would cost more context than the diff the
341
+ * reader actually came for. `history show` carries the payload for one run.
342
+ */
343
+ export function renderHistory(runs, { total } = {}) {
344
+ if (!runs.length) {
345
+ return {
346
+ text:
347
+ `# Run history\n\nNo runs recorded in this session.\n\n` +
348
+ `History lives in memory only: it starts empty on every server start and is never written to disk.`,
349
+ data: { runs: [], count: 0 },
350
+ };
351
+ }
352
+
353
+ const lines = runs.map((r) => {
354
+ const state = r.ok ? "ok" : `FAILED (${r.code || "error"})`;
355
+ const ms = typeof r.durationMs === "number" ? ` · ${r.durationMs}ms` : "";
356
+ return `- \`${r.id}\` ${r.tool} · ${state}${ms}${r.truncated ? " · result truncated" : ""}`;
357
+ });
358
+
359
+ const text =
360
+ `# Run history\n\n` +
361
+ `${runs.length} of ${total ?? runs.length} recorded run${runs.length === 1 ? "" : "s"}, newest first.\n\n` +
362
+ `${lines.join("\n")}\n\n` +
363
+ `---\n` +
364
+ `Compare two runs with \`{"command":"diff","from":"<id>","to":"<id>"}\`. ` +
365
+ `\`-1\` is the most recent run, \`-2\` the one before it.`;
366
+
367
+ return { text, data: { runs, count: runs.length } };
368
+ }
369
+
370
+ /** One run in full, with its result. */
371
+ export function renderRun(run) {
372
+ const head =
373
+ `# Run ${run.id} — ${run.tool}\n\n` +
374
+ [
375
+ `Status: ${run.ok ? "ok" : `FAILED (${run.code || "error"})`}`,
376
+ `At: ${run.at}`,
377
+ run.durationMs !== null ? `Duration: ${run.durationMs}ms` : null,
378
+ run.batch ? "Part of a batch" : null,
379
+ ]
380
+ .filter(Boolean)
381
+ .join(" · ");
382
+
383
+ const body = run.ok
384
+ ? "```json\n" + safeStringify(run.result) + "\n```"
385
+ : `${run.error}${run.retryable ? "\n\nThis failure is marked retryable." : "\n\nThis failure is not retryable — change the call before trying again."}`;
386
+
387
+ const input = `Input (credential-shaped values are redacted):\n\`\`\`json\n${safeStringify(run.params)}\n\`\`\``;
388
+
389
+ return {
390
+ text: `${head}\n\n${input}\n\n${body}`,
391
+ data: run,
392
+ };
393
+ }
394
+
395
+ /**
396
+ * A comparison of two runs.
397
+ *
398
+ * Leads with the verdict, then the inputs, then the field-level changes: the
399
+ * first question is "did anything change at all", and an agent that has to read
400
+ * forty diff lines to answer that will skim them and get it wrong.
401
+ */
402
+ export function renderDiff({ from, to, identical, inputsChanged, changes, truncated }) {
403
+ const verdict = identical
404
+ ? "Identical — the two runs produced the same result."
405
+ : `${changes.length} difference${changes.length === 1 ? "" : "s"}.`;
406
+
407
+ const lines = changes.map((c) => {
408
+ if (c.type === "added") return `- \`${c.path}\` — added: ${safeStringify(c.to)}`;
409
+ if (c.type === "removed") return `- \`${c.path}\` — removed: ${safeStringify(c.from)}`;
410
+ return `- \`${c.path}\` — changed: ${safeStringify(c.from)} → ${safeStringify(c.to)}`;
411
+ });
412
+
413
+ const sections = [
414
+ `# Diff: ${from.id} → ${to.id}\n\n${verdict}`,
415
+ `**From:** \`${from.id}\` ${from.tool} · ${from.ok ? "ok" : "failed"}${from.durationMs !== null ? ` · ${from.durationMs}ms` : ""}\n` +
416
+ `**To:** \`${to.id}\` ${to.tool} · ${to.ok ? "ok" : "failed"}${to.durationMs !== null ? ` · ${to.durationMs}ms` : ""}`,
417
+ ];
418
+
419
+ if (inputsChanged) {
420
+ sections.push(
421
+ `**Inputs differ.** A result that changed because its input changed is not a regression — check the inputs before concluding the tool changed behaviour.`,
422
+ );
423
+ }
424
+
425
+ if (truncated) {
426
+ sections.push(
427
+ `> One payload was truncated at the ${8_000}-character cap, so differences beyond that point are not reported.`,
428
+ );
429
+ }
430
+
431
+ sections.push(
432
+ identical ? "No field-level differences." : `## Changes\n\n${lines.join("\n")}`,
433
+ );
434
+
435
+ return {
436
+ text: sections.join("\n\n"),
437
+ data: { from, to, identical, inputsChanged, changes, ...(truncated ? { truncated: true } : {}) },
293
438
  };
294
439
  }
295
440
 
@@ -306,6 +451,80 @@ export function renderToolReportedError(slug, data) {
306
451
  };
307
452
  }
308
453
 
454
+ /**
455
+ * A batch of runs.
456
+ *
457
+ * Laid out so the failures come first: in a batch, the interesting output is
458
+ * whichever item broke, and burying that under five successful JSON blobs is how
459
+ * a bug gets missed. Each result keeps its own payload, so a caller that needs
460
+ * everything still has it — the ordering is for the reader, the data block is for
461
+ * the machine.
462
+ */
463
+ export function renderBatch({ runs, results, durationMs } = {}) {
464
+ const items = results || [];
465
+ const ok = items.filter((r) => r.success);
466
+ const failed = items.filter((r) => !r.success);
467
+
468
+ const header = [
469
+ `# Batch: ${ok.length}/${items.length} succeeded`,
470
+ `Runs: ${items.length} · Failed: ${failed.length}` +
471
+ (typeof durationMs === "number" ? ` · Duration: ${durationMs}ms` : ""),
472
+ ].join("\n\n");
473
+
474
+ const sections = [];
475
+
476
+ if (failed.length) {
477
+ sections.push(
478
+ `## Failed\n\n` +
479
+ failed
480
+ .map(
481
+ (r) =>
482
+ `- **${r.slug}** — ${r.error}` +
483
+ (r.code ? ` (\`${r.code}\`)` : "") +
484
+ (r.retryable ? " · retryable" : ""),
485
+ )
486
+ .join("\n"),
487
+ );
488
+ }
489
+
490
+ if (ok.length) {
491
+ sections.push(
492
+ `## Succeeded\n\n` +
493
+ ok
494
+ .map(
495
+ (r) =>
496
+ `### ${r.slug}\n\n` +
497
+ (typeof r.durationMs === "number" ? `\`${r.durationMs}ms\`\n\n` : "") +
498
+ "```json\n" +
499
+ safeStringify(r.data) +
500
+ "\n```",
501
+ )
502
+ .join("\n\n"),
503
+ );
504
+ }
505
+
506
+ if (!items.length) {
507
+ return {
508
+ text: `${header}\n\nNothing ran.`,
509
+ data: { results: [], succeeded: 0, failed: 0 },
510
+ };
511
+ }
512
+
513
+ const next = failed.length
514
+ ? `\n\n---\nEach failure above is independent: the other runs completed. Fix the listed inputs and re-run only those slugs.`
515
+ : "";
516
+
517
+ return {
518
+ text: `${header}\n\n${sections.join("\n\n")}${next}`,
519
+ data: {
520
+ success: failed.length === 0,
521
+ succeeded: ok.length,
522
+ failed: failed.length,
523
+ results: items,
524
+ },
525
+ };
526
+ }
527
+
309
528
  function safeStringify(value) {
310
529
  try {
311
530
  return JSON.stringify(value, null, 2);