runbios-mcp 0.2.1-dev.100

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. package/README.md +363 -0
  2. package/dist/api-client.d.ts +147 -0
  3. package/dist/api-client.d.ts.map +1 -0
  4. package/dist/api-client.js +806 -0
  5. package/dist/api-client.js.map +1 -0
  6. package/dist/auth.d.ts +3 -0
  7. package/dist/auth.d.ts.map +1 -0
  8. package/dist/auth.js +11 -0
  9. package/dist/auth.js.map +1 -0
  10. package/dist/config.d.ts +30 -0
  11. package/dist/config.d.ts.map +1 -0
  12. package/dist/config.js +16 -0
  13. package/dist/config.js.map +1 -0
  14. package/dist/deployment-contract.d.ts +78 -0
  15. package/dist/deployment-contract.d.ts.map +1 -0
  16. package/dist/deployment-contract.js +155 -0
  17. package/dist/deployment-contract.js.map +1 -0
  18. package/dist/gpu-priorities.d.ts +25 -0
  19. package/dist/gpu-priorities.d.ts.map +1 -0
  20. package/dist/gpu-priorities.js +61 -0
  21. package/dist/gpu-priorities.js.map +1 -0
  22. package/dist/http/app.d.ts +14 -0
  23. package/dist/http/app.d.ts.map +1 -0
  24. package/dist/http/app.js +140 -0
  25. package/dist/http/app.js.map +1 -0
  26. package/dist/http/audit.d.ts +23 -0
  27. package/dist/http/audit.d.ts.map +1 -0
  28. package/dist/http/audit.js +41 -0
  29. package/dist/http/audit.js.map +1 -0
  30. package/dist/http/config.d.ts +23 -0
  31. package/dist/http/config.d.ts.map +1 -0
  32. package/dist/http/config.js +70 -0
  33. package/dist/http/config.js.map +1 -0
  34. package/dist/http/consent.d.ts +18 -0
  35. package/dist/http/consent.d.ts.map +1 -0
  36. package/dist/http/consent.js +148 -0
  37. package/dist/http/consent.js.map +1 -0
  38. package/dist/http/internal-auth.d.ts +22 -0
  39. package/dist/http/internal-auth.d.ts.map +1 -0
  40. package/dist/http/internal-auth.js +73 -0
  41. package/dist/http/internal-auth.js.map +1 -0
  42. package/dist/http/main.d.ts +3 -0
  43. package/dist/http/main.d.ts.map +1 -0
  44. package/dist/http/main.js +65 -0
  45. package/dist/http/main.js.map +1 -0
  46. package/dist/http/mcp-handler.d.ts +12 -0
  47. package/dist/http/mcp-handler.d.ts.map +1 -0
  48. package/dist/http/mcp-handler.js +103 -0
  49. package/dist/http/mcp-handler.js.map +1 -0
  50. package/dist/http/metadata.d.ts +6 -0
  51. package/dist/http/metadata.d.ts.map +1 -0
  52. package/dist/http/metadata.js +32 -0
  53. package/dist/http/metadata.js.map +1 -0
  54. package/dist/http/oauth.d.ts +56 -0
  55. package/dist/http/oauth.d.ts.map +1 -0
  56. package/dist/http/oauth.js +491 -0
  57. package/dist/http/oauth.js.map +1 -0
  58. package/dist/http/serviceHostGuard.d.ts +31 -0
  59. package/dist/http/serviceHostGuard.d.ts.map +1 -0
  60. package/dist/http/serviceHostGuard.js +68 -0
  61. package/dist/http/serviceHostGuard.js.map +1 -0
  62. package/dist/http/store.d.ts +150 -0
  63. package/dist/http/store.d.ts.map +1 -0
  64. package/dist/http/store.js +366 -0
  65. package/dist/http/store.js.map +1 -0
  66. package/dist/index.d.ts +3 -0
  67. package/dist/index.d.ts.map +1 -0
  68. package/dist/index.js +79 -0
  69. package/dist/index.js.map +1 -0
  70. package/dist/inference-contract.d.ts +29 -0
  71. package/dist/inference-contract.d.ts.map +1 -0
  72. package/dist/inference-contract.js +112 -0
  73. package/dist/inference-contract.js.map +1 -0
  74. package/dist/redaction.d.ts +74 -0
  75. package/dist/redaction.d.ts.map +1 -0
  76. package/dist/redaction.js +316 -0
  77. package/dist/redaction.js.map +1 -0
  78. package/dist/server.d.ts +63 -0
  79. package/dist/server.d.ts.map +1 -0
  80. package/dist/server.js +2239 -0
  81. package/dist/server.js.map +1 -0
  82. package/dist/training-contract.d.ts +108 -0
  83. package/dist/training-contract.d.ts.map +1 -0
  84. package/dist/training-contract.js +119 -0
  85. package/dist/training-contract.js.map +1 -0
  86. package/dist/version.d.ts +2 -0
  87. package/dist/version.d.ts.map +1 -0
  88. package/dist/version.js +2 -0
  89. package/dist/version.js.map +1 -0
  90. package/package.json +48 -0
package/dist/server.js ADDED
@@ -0,0 +1,2239 @@
1
+ import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
2
+ import { z } from "zod";
3
+ import { readFile, stat } from "node:fs/promises";
4
+ import { basename, resolve } from "node:path";
5
+ import { randomUUID } from "node:crypto";
6
+ import { buildGPUOptionsParams, buildTrainingCreateBody, TRAINING_ADAPTERS, TRAINING_METHODS, trimTrainingLogs, } from "./training-contract.js";
7
+ import { buildDeploymentBody, buildDeploymentGPUOptionsParams, buildDeploymentPolicyBody, DEPLOYMENT_GPU_TIERS, DEPLOYMENT_SOURCE_TYPES, } from "./deployment-contract.js";
8
+ import { buildInferenceBody } from "./inference-contract.js";
9
+ import { redactBuildIdentity, redactBuildIdentityDeep, redactFreeTextFields, sanitizeArchitectureRegistry, stripBuildIdentity, } from "./redaction.js";
10
+ import { gpuRejectionCodeFor, gpuRejectionInstruction, gpuRejectionStatusFor, isPermanentGpuCode, } from "./api-client.js";
11
+ /**
12
+ * Job/deployment fields whose upstream value is RAW text rather than a curated
13
+ * sentence. `stop_last_error` is the proven one: the training stop saga stores
14
+ * the transport error from its last call to the machine verbatim, and that text
15
+ * is the request URL — a hostname that names the capacity supplier. The others
16
+ * are the same class (free text written from an upstream failure) and are
17
+ * covered so a future writer that stops curating cannot reopen the hole.
18
+ */
19
+ const RAW_FAILURE_TEXT_FIELDS = [
20
+ "stop_last_error",
21
+ "error_message",
22
+ "last_error",
23
+ "stage_detail",
24
+ ];
25
+ import { resolveLaunchGates } from "./config.js";
26
+ import { VERSION } from "./version.js";
27
+ export { VERSION };
28
+ /**
29
+ * Wrap a tool handler so hooks.onToolCall observes the tool name, success
30
+ * flag, elapsed milliseconds, and on failure the error message string only
31
+ * (never tool arguments). The wrapper does not change what the handler
32
+ * returns or throws; without a hook the handler is registered untouched.
33
+ */
34
+ function wrapToolHandler(name, handler, onToolCall) {
35
+ if (!onToolCall)
36
+ return handler;
37
+ const fn = handler;
38
+ const wrapped = async (...args) => {
39
+ const start = Date.now();
40
+ try {
41
+ const out = await fn(...args);
42
+ onToolCall({ tool: name, ok: true, ms: Date.now() - start });
43
+ return out;
44
+ }
45
+ catch (error) {
46
+ onToolCall({
47
+ tool: name,
48
+ ok: false,
49
+ ms: Date.now() - start,
50
+ error: error instanceof Error ? error.message : String(error),
51
+ });
52
+ throw error;
53
+ }
54
+ };
55
+ return wrapped;
56
+ }
57
+ const SAFE_ID_RE = /^[a-zA-Z0-9_-]+$/;
58
+ function validateId(value, label) {
59
+ if (!SAFE_ID_RE.test(value)) {
60
+ throw new Error(`Invalid ${label}: must contain only alphanumeric characters, hyphens, and underscores`);
61
+ }
62
+ }
63
+ function result(text) {
64
+ return { content: [{ type: "text", text }] };
65
+ }
66
+ function json(data) {
67
+ // A tool result of literally `null` is unreadable: an agent cannot tell an
68
+ // empty answer from a failure, and "null" is the one string that invites a
69
+ // wrong conclusion either way. The client already normalizes empty bodies;
70
+ // this is the last-resort guard for any other path.
71
+ if (data === null || data === undefined) {
72
+ return result(JSON.stringify({
73
+ code: "EMPTY_RESPONSE",
74
+ message: "The Run BiOS API answered this read with no JSON body. This is not an error verdict and nothing was changed.",
75
+ instruction: "Retry once; if it repeats, confirm the resource with the matching list/status tool before drawing a conclusion.",
76
+ }, null, 2));
77
+ }
78
+ return result(JSON.stringify(data, null, 2));
79
+ }
80
+ /* ─── Honest, actionable recovery text ───
81
+ * An MCP client ACTS on this text, so recovery advice must not dead-end on a
82
+ * single tool: list_models and search_models read the same registry through the
83
+ * same dependency, so one outage takes both down. Name every route the agent
84
+ * has, and tell it what to do when none of them answer, instead of pointing at
85
+ * one tool that may be unavailable. */
86
+ const CATALOG_RECOVERY = "Pick an id from the Run BiOS catalog: call list_models (everything hosted) or search_models (filter by name/provider) — "
87
+ + "either one answers from the same registry. If BOTH are unavailable, do NOT guess another id: any id already returned "
88
+ + "by list_inferences or list_training_jobs is known-hosted, the same catalog is on the Models page of the Run BiOS console, "
89
+ + "and get_platform_guide (topic \"models\") states the rule. When nothing can confirm an id, report the catalog as "
90
+ + "unavailable instead of retrying with an unverified one.";
91
+ /* ─── Context-length policy (owner-mandated, enforced server-side) ───
92
+ * The window is DERIVED unless the caller overrides it: the default is
93
+ * min(model native max, 262144), a model whose native window is unknown falls
94
+ * back to 32768, a genuinely small-context model keeps its own smaller window,
95
+ * and the hard ceiling is always the model's OWN native max — the server
96
+ * rejects anything above it. No static JSON Schema can express a per-model
97
+ * ceiling, so the bounds below are an honest sanity range and the description
98
+ * names the real authority (model.native_max_context from preflight). */
99
+ const CONTEXT_DEFAULT_CEILING = 262_144;
100
+ const CONTEXT_UNKNOWN_FALLBACK = 32_768;
101
+ const CONTEXT_TOOL_MIN = 1_024;
102
+ const CONTEXT_TOOL_MAX = 1_048_576;
103
+ const CONTEXT_POLICY_TEXT = `Omit it to accept the platform default: min(model native max, ${CONTEXT_DEFAULT_CEILING}) tokens, `
104
+ + `falling back to ${CONTEXT_UNKNOWN_FALLBACK} when the model's native window is unknown (a model whose own window is `
105
+ + `smaller keeps its own). The hard ceiling is the model's OWN native window — a larger value is rejected. `
106
+ + `Read model.native_max_context from preflight_inference before proposing a value; `
107
+ + `the ${CONTEXT_TOOL_MIN}-${CONTEXT_TOOL_MAX} range accepted here is only a sanity bound, never a per-model ceiling.`;
108
+ /**
109
+ * Context-length input with truthful bounds and self-contained rejection text.
110
+ * The MCP SDK renders a zod failure as its raw issue list, so the message on
111
+ * each bound has to carry the whole answer by itself — otherwise the caller
112
+ * gets a schema dump with no instruction.
113
+ */
114
+ /* ─── Precision ───
115
+ * Precision is a sizing input, so it is a PRICE input: it sets the VRAM
116
+ * estimate, which sets the minimum GPU count, which sets the hourly total. An
117
+ * agent that writes "FP8" is asking for fp8, and must not be quoted the bf16
118
+ * price for it. Canonicalize case and whitespace before validating, then refuse
119
+ * anything outside the vocabulary BY NAME. bf16/fp16 are accepted spellings of
120
+ * "serve the checkpoint at its own precision", which is what none means.
121
+ */
122
+ const QUANT_FORMATS = ["fp8", "awq", "gptq", "int4"];
123
+ const NATIVE_PRECISION_ALIASES = ["none", "bf16", "bfloat16", "fp16", "float16"];
124
+ function quantSchema(purpose) {
125
+ return z
126
+ .string()
127
+ .transform((v) => v.trim().toLowerCase())
128
+ .refine((v) => NATIVE_PRECISION_ALIASES.includes(v)
129
+ || QUANT_FORMATS.includes(v), {
130
+ error: (issue) => `${String(issue.input)} is not a precision this platform can serve. `
131
+ + `Use one of: ${QUANT_FORMATS.join(", ")}, or none.`,
132
+ })
133
+ .transform((v) => (NATIVE_PRECISION_ALIASES.includes(v) ? "none" : v))
134
+ .optional()
135
+ .describe(`${purpose} One of ${QUANT_FORMATS.join(", ")} or none. Case does not matter, `
136
+ + "and bf16/fp16 mean the same as none. Anything else is refused by name rather "
137
+ + "than sized at another precision.");
138
+ }
139
+ function contextLengthSchema(purpose) {
140
+ return z
141
+ .number()
142
+ .int({ error: "context_length must be a whole number of tokens." })
143
+ .min(CONTEXT_TOOL_MIN, {
144
+ error: (issue) => `context_length ${String(issue.input)} is below ${CONTEXT_TOOL_MIN} tokens, which cannot serve a request. `
145
+ + CONTEXT_POLICY_TEXT,
146
+ })
147
+ .max(CONTEXT_TOOL_MAX, {
148
+ error: (issue) => `context_length ${String(issue.input)} is above the ${CONTEXT_TOOL_MAX}-token sanity bound of this tool. `
149
+ + CONTEXT_POLICY_TEXT,
150
+ })
151
+ .optional()
152
+ .describe(`${purpose} ${CONTEXT_POLICY_TEXT}`);
153
+ }
154
+ /* ─── Spend-consent bounds (the one field that authorizes money) ───
155
+ * max_price_hour_cents is CONSENT, not a filter: it is the maximum total hourly
156
+ * price for the complete GPU count that the platform may charge without asking
157
+ * the user again. The server has no ceiling of its own — it only refuses a cap
158
+ * BELOW the primary placement's verified price (deployment-service
159
+ * validateInferenceGPUPriceContract, training-service validateTrainingRequest)
160
+ * — so `z.number().int()` alone advertised MAX_SAFE_INTEGER, i.e. told an agent
161
+ * that consenting to nine quadrillion cents an hour is a legal request. The
162
+ * ceiling below comes from what the platform can actually book: the most
163
+ * expensive placement in the GPU catalog is 8 x B300 at 1035 cents per GPU-hour
164
+ * = 8280 cents/hour total, and the beta serving cap (models up to ~130B total
165
+ * parameters) keeps a deployment inside that ladder. 100000 cents ($1,000/hour)
166
+ * is roughly twelve times the dearest bookable configuration: high enough that
167
+ * it can never reject a real cap, low enough that it is a limit rather than a
168
+ * blank cheque. It is a sanity bound, not a price quote — the authority is
169
+ * always the preflight response. */
170
+ const PRICE_CAP_TOOL_MAX = 100_000;
171
+ const priceCapPolicyText = (floor) => "This is SPEND CONSENT, not a filter: the maximum accepted TOTAL hourly price for the complete GPU count, in cents. "
172
+ + "Copy the cap the user approved from the preflight response (billing.max_price_hour_cents, or the chosen placement's "
173
+ + "total_price_hour_cents) — never invent one. The server rejects a cap BELOW the primary placement's current price, and "
174
+ + `omitting it accepts the verified price of the placements already approved. The ${floor}-${PRICE_CAP_TOOL_MAX} range here `
175
+ + "is a sanity bound on consent (the dearest configuration the GPU catalog can book is well under it), not a price quote and "
176
+ + "not a per-model ceiling; get_gpu_pricing and preflight give the real numbers.";
177
+ /**
178
+ * Price-cap input with a defensible ceiling and self-contained rejection text.
179
+ * `allowZero` covers the training tools, where 0 is the documented "no explicit
180
+ * cap — accept the verified placement price" value; the deployment tools require
181
+ * a positive cap (deployment-contract.ts rejects <= 0).
182
+ */
183
+ function priceCapSchema(allowZero) {
184
+ const floor = allowZero ? 0 : 1;
185
+ const policy = priceCapPolicyText(floor);
186
+ return z
187
+ .number()
188
+ .int({ error: "max_price_hour_cents must be a whole number of cents." })
189
+ .min(floor, {
190
+ error: (issue) => `max_price_hour_cents ${String(issue.input)} is below ${floor}. `
191
+ + (allowZero
192
+ ? "A cap cannot be negative; pass 0 (or omit it) to accept the verified placement price. "
193
+ : "Omit it instead of passing 0 or a negative number to accept the verified placement price. ")
194
+ + policy,
195
+ })
196
+ .max(PRICE_CAP_TOOL_MAX, {
197
+ error: (issue) => `max_price_hour_cents ${String(issue.input)} is above the ${PRICE_CAP_TOOL_MAX}-cent/hour ($1,000/hour) sanity bound `
198
+ + "of this tool. A cap that large is not a limit — it authorizes any price the platform could ever charge. "
199
+ + policy,
200
+ })
201
+ .optional()
202
+ .describe(policy);
203
+ }
204
+ /* ─── Training context window (max_seq_length) ───
205
+ * Same defect as context_length before #789: `max(10_000_000)` is not a policy,
206
+ * it is a number nothing enforces. The authoritative contract is
207
+ * get_training_capabilities, whose `max_length` field states the platform's
208
+ * default (2048) and its floor (128); the real ceiling is the base model's own
209
+ * position-embedding window, which no static JSON Schema can express. So the
210
+ * schema advertises the contract's floor and the same sanity ceiling serving
211
+ * uses, and the description names the real authority. */
212
+ const TRAINING_SEQ_TOOL_MIN = 128;
213
+ const TRAINING_SEQ_DEFAULT = 2_048;
214
+ const TRAINING_SEQ_POLICY_TEXT = `Training sequence length in tokens (sent as config.max_length). Omit it to accept the platform default of ${TRAINING_SEQ_DEFAULT}. `
215
+ + "The hard ceiling is the BASE MODEL's own context window — a longer value wastes GPU memory on padding and can fail the fit "
216
+ + `check — and the authoritative field bounds live in get_training_capabilities (max_length). The ${TRAINING_SEQ_TOOL_MIN}-`
217
+ + `${CONTEXT_TOOL_MAX} range accepted here is only a sanity bound, never a per-model ceiling: read the model's window from `
218
+ + "get_model_config or preflight_training_job before proposing a value. Longer sequences raise memory use roughly linearly, "
219
+ + "so raise it deliberately, not by default.";
220
+ const MAX_TOKENS_POLICY_TEXT = "Maximum tokens to GENERATE for this completion. Omit it to let the endpoint choose. The real ceiling is the deployment's "
221
+ + `served context window MINUS the prompt (read context_length from get_inference_status or the create response); the 1-`
222
+ + `${CONTEXT_TOOL_MAX} range accepted here is only a sanity bound, never the window of the model being called. Tokens are `
223
+ + "billed, so ask for what the answer needs rather than the maximum.";
224
+ function trainingSeqLengthSchema() {
225
+ return z
226
+ .number()
227
+ .int({ error: "max_seq_length must be a whole number of tokens." })
228
+ .min(TRAINING_SEQ_TOOL_MIN, {
229
+ error: (issue) => `max_seq_length ${String(issue.input)} is below the ${TRAINING_SEQ_TOOL_MIN}-token floor of the training contract. `
230
+ + TRAINING_SEQ_POLICY_TEXT,
231
+ })
232
+ .max(CONTEXT_TOOL_MAX, {
233
+ error: (issue) => `max_seq_length ${String(issue.input)} is above the ${CONTEXT_TOOL_MAX}-token sanity bound of this tool. `
234
+ + TRAINING_SEQ_POLICY_TEXT,
235
+ })
236
+ .optional()
237
+ .describe(TRAINING_SEQ_POLICY_TEXT);
238
+ }
239
+ // Reuse this exact schema for preflight and create. The helper beneath it
240
+ // performs the source-dependent checks that a flat MCP input schema cannot.
241
+ const deploymentCreateToolSchema = {
242
+ name: z.string().min(1).max(255).describe("Deployment display name."),
243
+ source_type: z
244
+ .enum(DEPLOYMENT_SOURCE_TYPES)
245
+ .describe("Deploy a verified training checkpoint or a base model from the catalog."),
246
+ source_job_id: z
247
+ .string()
248
+ .optional()
249
+ .describe("Owning training job ID; required with source_type=checkpoint."),
250
+ source_checkpoint_id: z
251
+ .string()
252
+ .optional()
253
+ .describe("Verified checkpoint ID; required with source_type=checkpoint."),
254
+ hf_model_id: z
255
+ .string()
256
+ .optional()
257
+ .describe("Catalog model id (see list_models); required when deploying a hosted base model (source_type=hf_model, the legacy wire name). Must be hosted on Run BiOS."),
258
+ hf_model_revision: z
259
+ .string()
260
+ .regex(/^[0-9a-f]{40}$/)
261
+ .optional()
262
+ .describe("Exact model commit from preflight_inference canonical_request. Copy it unchanged into create_inference."),
263
+ hf_integration_id: z
264
+ .string()
265
+ .optional()
266
+ .describe("Legacy field kept for compatibility — catalog models are pre-mirrored and never gated, so no integration is needed. Raw tokens are not accepted by MCP."),
267
+ base_model_id: z
268
+ .string()
269
+ .optional()
270
+ .describe("Base model for an adapter checkpoint; normally resolved from training lineage."),
271
+ base_model_revision: z
272
+ .string()
273
+ .regex(/^[0-9a-f]{40}$/)
274
+ .optional()
275
+ .describe("Exact base-model commit from preflight_inference canonical_request. Copy it unchanged into create_inference."),
276
+ // serving_mode, model_task and supports_images are DELIBERATELY ABSENT.
277
+ // All three are derived server-side and immutable: the serving mode comes
278
+ // from the verified checkpoint's lineage, the OpenAI task from the resolved
279
+ // model (chat, or completion for a base model with no chat template, or
280
+ // embedding/rerank for those architectures), and image support from the
281
+ // model's own config. Advertising them as inputs invited exactly the 400s the
282
+ // create gate now returns ("configured automatically from the model itself
283
+ // ... and cannot be changed"). They are reported back on the
284
+ // preflight/create/status response, which is where a caller reads them.
285
+ gpu_type: z
286
+ .string()
287
+ .min(1)
288
+ .describe("Exact GPU SKU approved after reviewing authoritative deployment options."),
289
+ gpu_count: z
290
+ .number()
291
+ .int()
292
+ .min(1)
293
+ .max(8)
294
+ .describe("Tensor-parallel GPU count accepted by the model-fit option."),
295
+ gpu_priorities: z
296
+ .array(z.object({
297
+ gpu_type: z.string().min(1), gpu_count: z.number().int().min(1).max(8),
298
+ provider: z.string().min(1).max(30).optional(), region: z.string().min(1).max(64).optional(),
299
+ tier: z.literal("secure").optional(),
300
+ }))
301
+ .min(1)
302
+ .max(5)
303
+ .optional()
304
+ .describe("Ranked GPU placement choices. Copy these unchanged from the preflight response's accepted choices; the platform fills in and manages infrastructure placement automatically. Queueing requires 3-5 distinct entries; entry one must match gpu_type/gpu_count."),
305
+ gpu_tier: z
306
+ .enum(DEPLOYMENT_GPU_TIERS)
307
+ .optional()
308
+ .describe("Deployment serving supports secure capacity only; defaults to secure."),
309
+ allow_capacity_queue: z
310
+ .boolean()
311
+ .optional()
312
+ .describe("Explicit consent to wait up to seven days if this exact SKU is unavailable; defaults to false."),
313
+ max_price_hour_cents: priceCapSchema(false),
314
+ storage_gb: z.number().int().min(1).max(10000).optional(),
315
+ context_length: contextLengthSchema("Serving context window in tokens."),
316
+ quant: quantSchema("Runtime precision. Pre-quantized checkpoint precision may be locked."),
317
+ serving_config: z
318
+ .record(z.string(), z.unknown())
319
+ .optional()
320
+ .describe("Advanced serving arguments; the server applies its strict allowlist."),
321
+ };
322
+ const deploymentCreateMutationToolSchema = {
323
+ ...deploymentCreateToolSchema,
324
+ idempotency_key: z
325
+ .string()
326
+ .regex(/^[A-Za-z0-9._:-]{8,128}$/)
327
+ .describe("Stable retry key. Reuse the same value with the unchanged payload after timeouts to recover the same deployment and one-time inference key."),
328
+ };
329
+ /* ─── Response shaping that keeps provenance honest ─── */
330
+ function asRecord(value) {
331
+ return value && typeof value === "object" && !Array.isArray(value)
332
+ ? value
333
+ : undefined;
334
+ }
335
+ function nonEmptyString(value) {
336
+ return typeof value === "string" && value.trim() ? value.trim() : undefined;
337
+ }
338
+ /* ─── Internal build references never leave the platform ─── */
339
+ /**
340
+ * Recursively redact every string in a JSON-shaped value, counting the fields
341
+ * that carried something.
342
+ *
343
+ * Redacting by KEY was defeated the moment the same value appeared under a
344
+ * different key: dropping `image_tag` left the identical
345
+ * `<registry>/<engine>@sha256:<digest>` in the free-text `notes` of 194
346
+ * training-scope rows, and the inference scope has the same shape waiting in
347
+ * the "dropped by engine <tag>" note its auto-disable pass writes. So the rule
348
+ * is content, not field name, and it runs over every string in the payload.
349
+ *
350
+ * The matching itself lives in redaction.ts and is NOT repeated here. An
351
+ * earlier revision of this module carried its own pattern list anchored on a
352
+ * plaintext array of the engine repositories Run BiOS builds — and `npm publish`
353
+ * ships dist/, so that array handed every customer the exact roster the
354
+ * redactor exists to protect (test/build-identity-leak.test.ts pins this, and
355
+ * the console hit the identical defect in its own leak guard). redaction.ts
356
+ * matches the same references by SHAPE plus salted hashes of the names, so it
357
+ * is strictly stronger AND publishes nothing.
358
+ */
359
+ function redactDeep(value, counter) {
360
+ if (typeof value === "string") {
361
+ const out = redactBuildIdentity(value);
362
+ if (out !== value)
363
+ counter.hits += 1;
364
+ return out;
365
+ }
366
+ if (Array.isArray(value))
367
+ return value.map((entry) => redactDeep(entry, counter));
368
+ const record = asRecord(value);
369
+ if (!record)
370
+ return value;
371
+ const out = {};
372
+ for (const [key, entry] of Object.entries(record))
373
+ out[key] = redactDeep(entry, counter);
374
+ return out;
375
+ }
376
+ /**
377
+ * Label the architecture registry's provenance instead of shipping it bare.
378
+ *
379
+ * Each row in the registry TABLE records image_tag/bios_version/manifest_sha256
380
+ * from the serving image whose capability manifest last published it. That is
381
+ * REGISTRY PROVENANCE, not the image the platform is running now, and the two
382
+ * drift apart whenever a newer image has not re-published the manifest. An
383
+ * agent that reads a build reference next to a support list concludes "the
384
+ * platform runs this build" and tells the user a stale version — a wrong action
385
+ * caused purely by presentation. The support decision itself IS authoritative
386
+ * (the deploy gate reads these same rows), so the rows are preserved; the
387
+ * internal image reference is removed by CONTENT from every field of every
388
+ * scope (see redactBuildReferences) and replaced with an explicit provenance
389
+ * block. Fixing the underlying manifest staleness is a control-plane concern
390
+ * and is deliberately NOT duplicated here.
391
+ *
392
+ * Those three columns no longer REACH this function: the public endpoint
393
+ * projects them away before the response leaves the control plane. What arrives
394
+ * here is the customer projection, and this block must describe that.
395
+ *
396
+ * The block itself only asserts what the rows actually carry, and since the
397
+ * control plane started projecting this endpoint that is LESS than it once was:
398
+ * image_tag, bios_version and manifest_sha256 are dropped at the boundary now
399
+ * (publicArchRow in deployment-service), so no response can still see them.
400
+ * Counting them here would therefore report 0 forever and drive the block to
401
+ * announce "no row records a build version" about a registry whose rows do —
402
+ * narrating a field that is empty on every row is the same defect one level up,
403
+ * and asserting a fact about rows this layer can no longer inspect is worse.
404
+ * `source` is the discriminator that survives the projection, so it is the one
405
+ * this block counts on.
406
+ */
407
+ export function annotateArchitectureProvenance(data) {
408
+ const payload = asRecord(data);
409
+ if (!payload || !Array.isArray(payload.architectures))
410
+ return data;
411
+ const counter = { hits: 0 };
412
+ let manifestRows = 0;
413
+ let lastPublished = "";
414
+ const architectures = payload.architectures.map((entry) => {
415
+ const row = asRecord(entry);
416
+ if (!row)
417
+ return redactDeep(entry, counter);
418
+ // Belt and braces: the boundary already omits image_tag, so this destructure
419
+ // only matters if the projection ever regresses. It is cheap and it is the
420
+ // reason this file carries a REVIEWED entry in the leak-gate allowlist —
421
+ // deleting it means deleting that entry too, or the gate fails on a stale one.
422
+ const { image_tag: _internalImageReference, ...rest } = row;
423
+ if (nonEmptyString(row.source) === "manifest")
424
+ manifestRows += 1;
425
+ const updated = nonEmptyString(row.updated_at);
426
+ if (updated && updated > lastPublished)
427
+ lastPublished = updated;
428
+ return redactDeep(rest, counter);
429
+ });
430
+ // Every OTHER field of the envelope goes through the same content rule: the
431
+ // whole point is that no field name is trusted, including one the control
432
+ // plane has not added yet.
433
+ const { architectures: _rows, ...envelope } = payload;
434
+ return {
435
+ ...redactDeep(envelope, counter),
436
+ architectures,
437
+ manifest_provenance: {
438
+ note: "Architecture support and the enabled flags are authoritative — the deployment gate reads these same rows. "
439
+ + "Some rows were published by an internal build's capability manifest, and the count below says how many. WHICH "
440
+ + "build is not part of this response at all: the platform does not publish it on this endpoint, so nothing here "
441
+ + "names or numbers an engine build. Do not report a platform version from this response, and do not infer from "
442
+ + "the absence that the registry is stale or current — this response cannot tell you either way. "
443
+ + "Build identity is removed by key AND by content from every field, so nothing here names the engine the platform runs.",
444
+ // The COUNTS are the honest signal; the VALUES are not. An earlier revision
445
+ // of this block also listed `recorded_bios_versions` — the distinct engine
446
+ // build versions read off the rows — which handed the caller the exact
447
+ // build identity every other rule here removes. Counting says as much about
448
+ // staleness and names nothing. Its siblings that counted build versions and
449
+ // manifest digests are gone for the opposite reason: after the boundary
450
+ // projection they could only ever have counted 0.
451
+ rows: architectures.length,
452
+ rows_from_a_published_manifest: manifestRows,
453
+ last_row_update: lastPublished || null,
454
+ // Counts FIELDS that carried a reference, not individual references: the
455
+ // matcher rewrites a whole string in one pass, so a per-occurrence tally
456
+ // would be a guess. Named for what it actually measures.
457
+ fields_with_build_references_redacted: counter.hits,
458
+ },
459
+ };
460
+ }
461
+ /** Say what the reported GPU means in the deployment's current lifecycle state. */
462
+ function gpuState(status) {
463
+ switch (status) {
464
+ case "queued_capacity":
465
+ return "not held yet — waiting for stock, nothing charged";
466
+ case "running":
467
+ case "degraded":
468
+ return "held and serving";
469
+ case "provisioning":
470
+ case "downloading_weights":
471
+ case "loading_model":
472
+ return "being booked or booting — not serving yet";
473
+ case "stopped":
474
+ case "paused_insufficient_funds":
475
+ case "failed":
476
+ case "crash_loop":
477
+ case "deleting":
478
+ case "deleted":
479
+ return "not held — this is the SKU the deployment would resume on";
480
+ default:
481
+ return "unknown";
482
+ }
483
+ }
484
+ /**
485
+ * Summarize what a deployment detail payload ACTUALLY proves about the GPU and
486
+ * the lifecycle phase.
487
+ *
488
+ * The detail response is wide and reports several things as null until a pod
489
+ * callback lands, which reads to an agent as "no GPU" and "no phase" on a
490
+ * deployment that is demonstrably provisioning on a known SKU. Everything here
491
+ * is derived from fields the same payload returned — nothing is invented, and a
492
+ * value that cannot be derived stays "unknown" instead of being guessed.
493
+ */
494
+ export function summarizeInferenceStatus(data) {
495
+ const payload = asRecord(data);
496
+ if (!payload)
497
+ return data;
498
+ const status = nonEmptyString(payload.status) ?? "unknown";
499
+ const gpuType = nonEmptyString(payload.gpu_type);
500
+ const gpuCount = typeof payload.gpu_count === "number" ? payload.gpu_count : undefined;
501
+ const gpu = gpuType ? `${gpuCount && gpuCount > 0 ? gpuCount : 1}x ${gpuType}` : "unknown";
502
+ const download = asRecord(payload.download_progress);
503
+ const boot = asRecord(payload.boot_progress);
504
+ // Phase precedence: the agent's own weight/load report is the most advanced
505
+ // signal, then the pre-agent boot report, then the durable lifecycle status.
506
+ const reportedPhase = nonEmptyString(download?.phase) ?? nonEmptyString(boot?.phase);
507
+ const phaseSource = nonEmptyString(download?.phase)
508
+ ? "agent_progress_report"
509
+ : nonEmptyString(boot?.phase)
510
+ ? "pre_agent_boot_report"
511
+ : "deployment_status";
512
+ const percent = typeof download?.percent === "number"
513
+ ? download.percent
514
+ : typeof boot?.percent === "number" ? boot.percent : null;
515
+ const endpoint = nonEmptyString(payload.endpoint_url);
516
+ const summary = {
517
+ status,
518
+ phase: reportedPhase ?? status,
519
+ phase_source: phaseSource,
520
+ phase_percent: percent,
521
+ gpu,
522
+ gpu_tier: nonEmptyString(payload.gpu_tier) ?? "unknown",
523
+ // What the GPU above MEANS right now: a queued or provisioning deployment
524
+ // does not hold the SKU yet, it is the one being waited/booked for.
525
+ gpu_state: gpuState(status),
526
+ endpoint_usable: status === "running" && Boolean(endpoint),
527
+ desired_state: nonEmptyString(payload.desired_state) ?? "unknown",
528
+ error_code: nonEmptyString(payload.error_code) ?? nonEmptyString(payload.status_reason) ?? null,
529
+ note: "Derived by the MCP server from the fields in this same response; the raw fields above are unchanged. "
530
+ + "Report the endpoint as live only when endpoint_usable is true.",
531
+ };
532
+ const shaped = { ...payload, status_summary: summary };
533
+ // A present-but-null `gpu`/`phase` reads as "there is none" even when the row
534
+ // proves otherwise. Fill from the derived value and say so, rather than
535
+ // passing a null the agent will misread.
536
+ for (const key of ["gpu", "phase"]) {
537
+ if (key in shaped && shaped[key] === null && summary[key] !== "unknown") {
538
+ shaped[key] = summary[key];
539
+ summary.filled_null_fields = [...(summary.filled_null_fields ?? []), key];
540
+ }
541
+ }
542
+ return shaped;
543
+ }
544
+ /* ─── Server setup ─── */
545
+ export function createBiosMcpServer(client, hooks, deploymentCaps,
546
+ // Pre-launch surface gates (src/config.ts). While a gate is on, its creation
547
+ // tools are NOT REGISTERED at all — hidden from tools/list so an agent never
548
+ // learns they exist — and the platform guide stops teaching their workflows.
549
+ // Defaults read the environment so a deployment with nothing set fails CLOSED
550
+ // (gated) while pre-launch; tests that exercise creation pass
551
+ // { training: false, datasets: false }.
552
+ launchGates = resolveLaunchGates()) {
553
+ // `instructions` reaches the client at initialize, BEFORE any tool description
554
+ // is read, which is the only place to state the rules a tool schema cannot: an
555
+ // agent that learns the derived-settings and catalog rules up front never
556
+ // proposes a value the control plane must reject.
557
+ // While a creation surface is gated, the instructions must not present it as
558
+ // available: an agent reads these BEFORE any tool list, and a capability named
559
+ // here is a capability it will try. The gated wording names what is live and
560
+ // states — once, without naming tool names — that creation is not offered.
561
+ const gatedSurfaces = [
562
+ ...(launchGates.training ? ["fine-tuning"] : []),
563
+ ...(launchGates.datasets ? ["dataset upload/import"] : []),
564
+ ];
565
+ const prelaunchNote = gatedSurfaces.length
566
+ ? `${gatedSurfaces.join(" and ")} ${gatedSurfaces.length > 1 ? "are" : "is"} not available in this deployment yet — `
567
+ + "no creation tools for them are offered, so do not attempt to start one or look for a workaround. "
568
+ + "Existing jobs and datasets stay readable.\n"
569
+ : "";
570
+ const mcp = new McpServer({
571
+ name: "Run BiOS",
572
+ version: VERSION,
573
+ }, {
574
+ instructions: (launchGates.training
575
+ ? "Run BiOS serves models on its own GPU control plane. Call get_platform_guide first.\n"
576
+ : "Run BiOS trains and serves models on its own GPU control plane. Call get_platform_guide first.\n")
577
+ + prelaunchNote
578
+ + "Rules that no tool schema can express:\n"
579
+ + (launchGates.training
580
+ ? "1. Models: only ids in the Run BiOS catalog can be served (list_models / search_models). "
581
+ : "1. Models: only ids in the Run BiOS catalog can be trained or served (list_models / search_models). ")
582
+ + "This is not a Hugging Face lookup; an id outside the catalog is rejected, never mirrored on demand.\n"
583
+ + "2. Serving is platform-derived: serving mode (full/adapter/merged), the OpenAI task "
584
+ + "(chat, or completion for a base model with no chat template, or embedding/rerank), and image support are "
585
+ + "derived from the model and checkpoint lineage. They are not tool inputs — read them from the "
586
+ + "preflight/create/status response.\n"
587
+ + `3. Context window: the default is min(model native max, ${CONTEXT_DEFAULT_CEILING}) tokens, falling back to `
588
+ + `${CONTEXT_UNKNOWN_FALLBACK} when the native window is unknown, and the hard ceiling is the model's own native max. `
589
+ + "Read model.native_max_context from preflight_inference before proposing a value.\n"
590
+ + "4. GPUs and money: never substitute a GPU or exceed an approved price cap without asking. A capacity "
591
+ + "rejection arrives as compact JSON with bookable alternatives; a 503 is a transient outage, never an out-of-stock verdict.\n"
592
+ + "5. Report an endpoint as live only when the status says running.",
593
+ });
594
+ // The registrations below are moved verbatim from the original single-file
595
+ // entry and intentionally keep their original indentation. `server.tool(...)`
596
+ // is a thin delegate that adds the onToolCall hook around each handler.
597
+ const server = {
598
+ tool(name, description, paramsSchema, cb) {
599
+ mcp.tool(name, description, paramsSchema, wrapToolHandler(name, cb, hooks?.onToolCall));
600
+ },
601
+ };
602
+ /* ══════════════════════════════════════════════════════════════════════════ */
603
+ /* TOOL: get_platform_guide */
604
+ /* ══════════════════════════════════════════════════════════════════════════ */
605
+ server.tool("get_platform_guide", launchGates.training || launchGates.datasets
606
+ ? "Get a comprehensive guide to the Run BiOS platform. Call this FIRST to understand the available capabilities, the correct workflow order, supported models, GPU tiers, and best practices for serving models."
607
+ : "Get a comprehensive guide explaining how to use the Run BiOS fine-tuning platform. Call this FIRST to understand the available capabilities, the correct workflow order, supported models, training methods, GPU tiers, dataset formats, and best practices. This gives you all the context needed to help users fine-tune models effectively.", {
608
+ topic: z
609
+ .enum([
610
+ "overview",
611
+ "quick_start",
612
+ "models",
613
+ "datasets",
614
+ "training_methods",
615
+ "gpu_selection",
616
+ "inference",
617
+ "hyperparameters",
618
+ "monitoring",
619
+ "cost_optimization",
620
+ "capacity_errors",
621
+ ])
622
+ .optional()
623
+ .describe("Specific topic to learn about. Omit for the full platform overview."),
624
+ }, async ({ topic }) => {
625
+ /* PRE-LAUNCH GATES (src/config.ts): while a creation surface is gated, the
626
+ guide must not teach its workflow — an agent that reads "call
627
+ upload_dataset" will try a tool that is not registered. Gated topics get
628
+ a stand-in that says what launches soon and names only tools that ARE
629
+ registered (reads/lifecycle stay live). At launch the gates open and the
630
+ real topics return verbatim. */
631
+ // 42 tools with every creation tool registered; the gates hide three
632
+ // (create_training_job, upload_dataset, import_huggingface_dataset). The
633
+ // coming-soon gate test pins this claim to the wire count, so drift fails.
634
+ const liveToolCount = 42
635
+ - (launchGates.training ? 1 : 0)
636
+ - (launchGates.datasets ? 2 : 0);
637
+ const comingSoonNote = `# Coming soon
638
+
639
+ ${gatedSurfaces.join(" and ")} ${gatedSurfaces.length > 1 ? "are" : "is"} not available in this build yet — the creation tools are not registered, so no tool call (and no workaround) can start one today. Do not retry or search for one.
640
+
641
+ What still works: every read and lifecycle tool for existing datasets and training jobs (list_datasets, preview_dataset, delete_dataset, list_training_jobs, get_training_status, get_training_metrics, get_training_logs, get_training_evals, stop_training_job, resume_training_job, get_training_checkpoints, get_checkpoint_download_manifest, delete_training_checkpoint), and the full inference surface (preflight_inference, create_inference, get_inference_status, chat_with_inference).`;
642
+ const gatedOverview = `# Run BiOS Platform
643
+
644
+ Run BiOS serves models on its own GPU control plane: a curated catalog of verified models — mirrored in its own storage, never fetched live from Hugging Face — deployed to dedicated OpenAI-compatible endpoints.
645
+
646
+ ## Core workflow (serving, fully live)
647
+ 1. **Find a model** — search_models / list_models read the hosted catalog; these are the only models that can be deployed
648
+ 2. **Check GPU fit and price** — get_inference_gpu_options joins model-fit counts to live stock and total hourly prices
649
+ 3. **Preflight** — preflight_inference validates the exact request with no wallet, queue, database, or GPU side effects
650
+ 4. **Create the deployment** — create_inference books capacity book-before-reveal; a definitive miss returns CAPACITY_UNAVAILABLE and nothing exists
651
+ 5. **Run** — poll get_inference_status until status is running; call the endpoint with chat_with_inference
652
+ 6. **Control spend** — stop_inference / resume_inference / restart_inference / delete_inference
653
+
654
+ ## Launching soon — not offered in this build
655
+ ${gatedSurfaces.join(" and ")} ${gatedSurfaces.length > 1 ? "launch" : "launches"} soon: the creation tools are not registered, so no tool call — and no workaround — can start one today. Existing datasets and training jobs are unaffected: every read and lifecycle tool below still works.
656
+
657
+ ## What the PLATFORM decides for you (do not try to set these)
658
+ Serving settings are derived from the model itself and are immutable. They are
659
+ NOT tool inputs, and attempting to choose one is rejected:
660
+ - **Serving mode** (full / adapter / merged) comes from the verified checkpoint's
661
+ lineage; a catalog base model is always served whole.
662
+ - **OpenAI task** comes from the resolved model: an instruct/chat model and a
663
+ vision-language model serve as chat, a base model with no chat template serves
664
+ as text completion, and an embedding/reranker architecture serves as
665
+ embeddings/reranking. Every model in the current catalog resolves to chat.
666
+ - **Image input** comes from the model's own config, never from a caller claim.
667
+ - **Context window**: the default is min(model native max, 262144) tokens; a
668
+ model whose native window is unknown falls back to 32768, a model with a
669
+ smaller window keeps its own, and the hard ceiling is always the model's OWN
670
+ native max. You may pass a smaller context_length; a larger one is rejected.
671
+ Read model.native_max_context from preflight_inference before proposing a value.
672
+ Read the actual values back from preflight_inference, create_inference, and
673
+ get_inference_status instead of asserting them.
674
+
675
+ ## Billing
676
+ - Per-second GPU billing (NOT hourly) — you only pay for actual compute time
677
+ - Wallet-based prepaid system with auto top-up option
678
+ - New accounts get a $1 welcome credit — enough to try inference, not to fund a deployment
679
+ - First top-up earns a deposit match: 50% within 24h of signup, 30% to 48h, 10% to 72h (deposits of $10+, bonus capped at $100)
680
+
681
+ ## Authentication
682
+ - API Key (recommended): Use \`X-API-Key: bios-...\` header — org and workspace are resolved automatically
683
+ - JWT token: Use \`Authorization: Bearer <token>\` header with \`X-Org-ID\` and \`X-Workspace-ID\` headers
684
+ - Call introspect_api_key to see your permissions, scopes, and which tools you can use
685
+
686
+ ## Available MCP Tools (${liveToolCount} in this build, listed in recommended order)
687
+ 1. get_platform_guide - You're reading this! Context for all capabilities
688
+ 2. introspect_api_key - Discover your API key's permissions, scopes, and bound workspace
689
+ 3. get_wallet_balance - Check your credits before creating a deployment
690
+ 4. search_models / list_models - Find a model from the hosted catalog
691
+ 5. get_model_config - Recommended settings for a chosen model
692
+ 6. list_supported_architectures - Which architectures can be served
693
+ 7. get_gpu_pricing / get_recommended_gpu - GPU options and pricing
694
+ 8. list_integrations - Connected Hugging Face accounts
695
+ 9. list_datasets / preview_dataset / delete_dataset - Read and clean up existing datasets
696
+ 10. get_training_capabilities - Read the authoritative training contract
697
+ 11. preflight_training_job - Validate a proposed job payload (no side effects)
698
+ 12. list_training_jobs / get_training_status - Monitor existing jobs
699
+ 13. get_training_metrics / get_training_evals / get_training_logs - Loss curves, benchmarks, logs
700
+ 14. stop_training_job / resume_training_job - Control existing jobs
701
+ 15. get_training_checkpoints / get_checkpoint_download_manifest / delete_training_checkpoint - Checkpoint access
702
+
703
+ ## Inference MCP Tools
704
+ 1. get_inference_gpu_options - Read model-fit choices joined to authoritative deployment stock and prices
705
+ 2. preflight_inference - Validate the exact request without wallet, queue, database, or GPU side effects
706
+ 3. create_inference - Create only after reviewing preflight alternatives, queue consent, and the price cap
707
+ 4. get_inference_booking - Poll a book-before-reveal booking handle from a long-running create_inference
708
+ 5. list_inferences / get_inference_status - Read durable lifecycle, queue expiry reason, and wallet-authorization state
709
+ 6. get_inference_metrics - Read request, token, latency, throughput, and GPU utilization metrics plus a recent time series
710
+ 7. get_inference_notifications - Inspect durable email delivery, retries, and dead letters
711
+ 8. update_inference_policy - Change explicit queue consent or the maximum accepted hourly price
712
+ 9. stop_inference / resume_inference / restart_inference - Control serving lifecycle
713
+ 10. chat_with_inference - Call a serverless catalog model by id or a dedicated deployment via the unified /v1 endpoint (optional streaming)
714
+ 11. delete_inference - Permanently remove a deployment after verified infrastructure teardown`;
715
+ const guides = {
716
+ overview: `# Run BiOS Fine-Tuning Platform
717
+
718
+ Run BiOS is a cloud fine-tuning platform for large language models. It hosts a curated catalog of verified base models — mirrored in its own storage, never fetched live from Hugging Face — trained natively by the BIOS training engine (supervised fine-tuning and continued pre-training; vision-language models train through SFT).
719
+
720
+ ## Core Workflow
721
+ 1. **Upload a dataset** — JSONL, Parquet, or CSV. The platform auto-validates format, detects columns, and maps fields.
722
+ 2. **Choose a base model** — Search the hosted Run BiOS catalog (search_models / list_models); every listed model is verified end-to-end and these are the only models that can be trained or deployed
723
+ 3. **Pick a training method** — SFT for instruction tuning (including vision-language models), CPT for continued pre-training on raw domain text
724
+ 4. **Select adapter** — LoRA (fast, efficient), QLoRA (lower VRAM), or Full fine-tune (max quality)
725
+ 5. **Configure hyperparameters** — learning rate, batch size, epochs, LoRA rank, etc. Smart defaults provided.
726
+ 6. **Choose a GPU** — A100 80GB, H100, A6000 etc. Platform recommends based on model size.
727
+ 7. **Launch and monitor** — Real-time loss curves, eval metrics, training logs, checkpoint saving
728
+ 8. **Download or deploy** — Get your fine-tuned model weights or merged checkpoints
729
+
730
+ ## What the PLATFORM decides for you (do not try to set these)
731
+ Serving settings are derived from the model itself and are immutable. They are
732
+ NOT tool inputs, and attempting to choose one is rejected:
733
+ - **Serving mode** (full / adapter / merged) comes from the verified checkpoint's
734
+ lineage; a catalog base model is always served whole.
735
+ - **OpenAI task** comes from the resolved model: an instruct/chat model and a
736
+ vision-language model serve as chat, a base model with no chat template serves
737
+ as text completion, and an embedding/reranker architecture serves as
738
+ embeddings/reranking. Every model in the current catalog resolves to chat.
739
+ - **Image input** comes from the model's own config, never from a caller claim.
740
+ - **Context window**: the default is min(model native max, 262144) tokens; a
741
+ model whose native window is unknown falls back to 32768, a model with a
742
+ smaller window keeps its own, and the hard ceiling is always the model's OWN
743
+ native max. You may pass a smaller context_length; a larger one is rejected.
744
+ Read model.native_max_context from preflight_inference before proposing a value.
745
+ Read the actual values back from preflight_inference, create_inference, and
746
+ get_inference_status instead of asserting them.
747
+
748
+ ## Billing
749
+ - Per-second GPU billing (NOT hourly) — you only pay for actual compute time
750
+ - Wallet-based prepaid system with auto top-up option
751
+ - New accounts get a $1 welcome credit — enough to try inference, not to fund a deployment
752
+ - First top-up earns a deposit match: 50% within 24h of signup, 30% to 48h, 10% to 72h (deposits of $10+, bonus capped at $100)
753
+
754
+ ## Authentication
755
+ - API Key (recommended): Use \`X-API-Key: bios-...\` header — org and workspace are resolved automatically
756
+ - JWT token: Use \`Authorization: Bearer <token>\` header with \`X-Org-ID\` and \`X-Workspace-ID\` headers
757
+ - Call introspect_api_key to see your permissions, scopes, and which tools you can use
758
+
759
+ ## Available MCP Tools (42 in this build, listed in recommended order)
760
+ 1. get_platform_guide - You're reading this! Context for all capabilities
761
+ 2. introspect_api_key - Discover your API key's permissions, scopes, and bound workspace
762
+ 3. get_wallet_balance - Check if you have enough credits before training
763
+ 4. search_models / list_models - Find the right base model from the hosted catalog
764
+ 5. get_model_config - Get recommended settings for your chosen model
765
+ 6. list_supported_architectures - Check which architectures can be trained or served
766
+ 7. get_training_capabilities - Read the authoritative training contract; call this before building automated payloads
767
+ 8. get_gpu_pricing - See GPU options and pricing
768
+ 9. get_recommended_gpu - Get the best GPU for your model + adapter combo
769
+ 10. list_integrations - See the Hugging Face accounts connected to your workspace
770
+ 11. list_datasets / upload_dataset / import_huggingface_dataset - Prepare training data
771
+ 12. preview_dataset - Verify dataset format before training
772
+ 13. delete_dataset - Remove a dataset you no longer need
773
+ 14. preflight_training_job - Validate the full job before creating it
774
+ 15. create_training_job - Launch the fine-tuning job
775
+ 16. list_training_jobs / get_training_status - Monitor progress
776
+ 17. get_training_metrics - View loss curves and eval results
777
+ 18. get_training_evals - Review benchmark scores and quality assessments
778
+ 19. get_training_logs - Read training output logs
779
+ 20. stop_training_job / resume_training_job - Control running jobs
780
+ 21. get_training_checkpoints - Access saved model checkpoints
781
+ 22. get_checkpoint_download_manifest - Per-file download URLs for a checkpoint (only where model weight downloads are enabled for the workspace)
782
+ 23. delete_training_checkpoint - Remove a checkpoint you no longer need
783
+
784
+ ## Inference MCP Tools
785
+ 1. get_inference_gpu_options - Read model-fit choices joined to authoritative deployment stock and prices
786
+ 2. preflight_inference - Validate the exact request without wallet, queue, database, or GPU side effects
787
+ 3. create_inference - Create only after reviewing preflight alternatives, queue consent, and the price cap
788
+ 4. get_inference_booking - Poll a book-before-reveal booking handle from a long-running create_inference
789
+ 5. list_inferences / get_inference_status - Read durable lifecycle, queue expiry reason, and wallet-authorization state
790
+ 6. get_inference_metrics - Read request, token, latency, throughput, and GPU utilization metrics plus a recent time series
791
+ 7. get_inference_notifications - Inspect durable email delivery, retries, and dead letters
792
+ 8. update_inference_policy - Change explicit queue consent or the maximum accepted hourly price
793
+ 9. stop_inference / resume_inference / restart_inference - Control serving lifecycle
794
+ 10. chat_with_inference - Call a serverless catalog model by id or a dedicated deployment via the unified /v1 endpoint (optional streaming)
795
+ 11. delete_inference - Permanently remove a deployment after verified infrastructure teardown`,
796
+ quick_start: `# Quick Start Guide
797
+
798
+ ## Fastest path to a fine-tuned model:
799
+
800
+ ### Step 1: Check your balance
801
+ Call get_wallet_balance to confirm you have credits available.
802
+
803
+ ### Step 2: Search for a model
804
+ Call search_models with a query like "llama 8B" or "mistral 7B".
805
+ For beginners: start with meta-llama/Llama-3.1-8B-Instruct (good balance of quality and speed).
806
+
807
+ ### Step 3: Prepare your dataset
808
+ Option A: Upload your own file — call upload_dataset with the file path
809
+ Option B: Import from HuggingFace — call import_huggingface_dataset with a repo ID
810
+
811
+ Dataset must be JSONL: {"messages": [...]} per line for SFT, or {"text": "..."} per line for CPT. Other layouts are refused with conversion guidance.
812
+
813
+ ### Step 4: Get recommended config
814
+ Call get_model_config with your model name to see recommended GPU, adapter, and hyperparameters.
815
+
816
+ ### Step 5: Create the training job
817
+ Call create_training_job with:
818
+ - model: the model name from step 2
819
+ - dataset_id: the ID from step 3
820
+ - method: "sft" for instruction tuning (most common)
821
+ - adapter: "lora" (fastest, recommended for first jobs)
822
+ - Leave other params as defaults unless you have specific needs
823
+
824
+ ### Step 6: Monitor progress
825
+ create_training_job books a GPU before returning (~40s). A 'booked' status means the GPU is secured (booked==secured) and training is starting; 'securing' means it is still booking — poll get_training_status until 'booked'/'running'. A booking-time capacity miss returns CAPACITY_UNAVAILABLE and no job is created.
826
+ Call get_training_status periodically. Training typically takes 1-4 hours for small models.
827
+ If a started job loses its pod it rests at 'interrupted' (billing already stopped, not 'failed') — call resume_training_job to continue from the last checkpoint.
828
+ Call get_training_metrics to see loss curves — loss should decrease steadily.
829
+
830
+ ### Tips
831
+ - Start with LoRA, not full fine-tune — it's 10x faster and 5x cheaper
832
+ - 3 epochs is usually enough — more can overfit
833
+ - If loss plateaus early, try increasing learning rate slightly
834
+ - Use early stopping if loss increases for several steps`,
835
+ models: `# Model Catalog
836
+
837
+ Run BiOS serves a curated catalog of models hosted in its own storage — every
838
+ listed model is verified end-to-end for training and serving, and these are
839
+ the ONLY models that can be trained or deployed. The catalog is the source of
840
+ truth: call list_models for everything currently hosted, or search_models to
841
+ filter by name, provider, or size. A model that is not in the catalog is not
842
+ hosted yet — ask an admin to mirror it.
843
+
844
+ list_models and search_models read the same registry, so either one answers.
845
+ If BOTH are unavailable, do not guess an id: any id already returned by
846
+ list_inferences or list_training_jobs is known-hosted, the same catalog is on
847
+ the Models page of the console, and reporting the catalog as unavailable is the
848
+ correct answer. An unverified id fails the create gate anyway.
849
+
850
+ Hosted families typically include Llama, Mistral, Qwen, Gemma, and Phi across
851
+ small (1B-4B), medium (7B-13B), and large (30B-70B+) sizes, plus vision-
852
+ language models — but the live catalog always wins over any list.
853
+
854
+ ## Choosing a Model
855
+ - For general tasks: a 7B-8B instruct model (good balance of quality and cost)
856
+ - For coding: a code-specialized instruct model from the catalog
857
+ - For multilingual: Qwen models
858
+ - For low-resource: a 1B-4B mini model
859
+ - For maximum quality: a 70B-class model (requires multi-GPU)
860
+
861
+ Use search_models to find hosted models by name or provider.
862
+ Use get_model_config to see what training methods and GPUs work with a specific model.`,
863
+ datasets: `# Dataset Format Guide
864
+
865
+ ## One file type, one shape per training method
866
+ Run BiOS accepts exactly one container -- **JSONL** (.jsonl, one JSON object per line) -- and exactly
867
+ one shape per training method -- the industry-standard chat-transcript convention. Files in any
868
+ other layout or container are refused at upload with a message naming what was detected and the exact
869
+ shape to convert to. There is no column mapping: convert before uploading.
870
+
871
+ ## SFT (Supervised Fine-Tuning)
872
+ Each line: {"messages": [...]} -- an ordered list of {role, content} turns.
873
+ Roles: "system" (optional, first), "user", "assistant", "tool". The last assistant turn is what the
874
+ model learns to produce. Assistant turns may include "tool_calls". Vision models use the same shape
875
+ with image content parts.
876
+
877
+ Example JSONL:
878
+ {"messages": [{"role": "system", "content": "You are a helpful assistant."}, {"role": "user", "content": "What is the capital of France?"}, {"role": "assistant", "content": "Paris."}]}
879
+ {"messages": [{"role": "user", "content": "Translate to French: Hello world"}, {"role": "assistant", "content": "Bonjour le monde"}]}
880
+
881
+ ## CPT (Continued Pre-Training)
882
+ Each line: {"text": "..."} -- one document of raw text per line. Documents are packed to the training
883
+ sequence length.
884
+
885
+ Example JSONL:
886
+ {"text": "One document of plain text per line."}
887
+
888
+ ## Converting other layouts
889
+ - instruction/input/output -> one user turn (instruction + input) and one assistant turn (output)
890
+ - prompt/completion, question/answer -> user turn and assistant turn
891
+ - ShareGPT conversations (from/value) -> role/content with human -> user, gpt -> assistant
892
+ - Preference pairs (chosen/rejected) and prompt-only RL formats are not accepted for training.
893
+
894
+ ## Best Practices
895
+ - Minimum 100 rows recommended (1,000+ for best results)
896
+ - Maximum 500MB file size through this tool
897
+ - Consistent formatting across all rows; remove duplicates and low-quality examples
898
+ - Pick max_seq_length above your longest sample: samples longer than it are dropped and reported`,
899
+ training_methods: `# Training Methods
900
+
901
+ ## SFT — Supervised Fine-Tuning ⭐ Most Common
902
+ Best for: Teaching a model to follow specific instructions or generate in a particular style.
903
+ Input: JSONL of {"messages": [...]} chat transcripts.
904
+ When to use: First fine-tuning job, domain-specific tasks, style transfer, format compliance.
905
+ Vision-language models: the same messages shape with image content parts.
906
+
907
+ ## CPT — Continued Pre-Training
908
+ Best for: Injecting domain knowledge into the base model before fine-tuning.
909
+ Input: Raw text documents.
910
+ When to use: Before SFT, when your domain has specialized vocabulary/knowledge.
911
+
912
+ ## Recommended Pipeline
913
+ 1. CPT (if domain-specific knowledge needed) → 2. SFT
914
+ For most users: SFT alone is sufficient.`,
915
+ gpu_selection: `# GPU Selection Guide
916
+
917
+ ## Available GPUs
918
+
919
+ | GPU | VRAM | Best For | Per-Second Cost |
920
+ |-----|------|----------|-----------------|
921
+ | A6000 | 48 GB | Models up to 13B with LoRA | ~$0.0007/sec |
922
+ | A100 80GB | 80 GB | Models up to 70B with QLoRA, up to 13B full | ~$0.0007/sec |
923
+ | H100 | 80 GB | Large models, fastest training | ~$0.0011/sec |
924
+
925
+ ## Model Size → GPU Recommendations
926
+
927
+ ### LoRA Fine-Tuning (most common)
928
+ - 1B-3B models: A6000 (48GB) — plenty of headroom
929
+ - 7B-13B models: A100 80GB ⭐ sweet spot
930
+ - 30B-70B models: A100 80GB or H100 (may need multi-GPU)
931
+
932
+ ### QLoRA Fine-Tuning (4-bit quantized, uses less VRAM)
933
+ - 7B-13B models: A6000 (48GB) — enough with quantization
934
+ - 30B-70B models: A100 80GB
935
+ - 70B+ models: H100 or multi-GPU A100
936
+
937
+ ### Full Fine-Tuning (no adapter, maximum quality)
938
+ - 1B-3B models: A100 80GB
939
+ - 7B models: 2-4x A100 80GB
940
+ - 13B+ models: 4-8x A100 or H100
941
+
942
+ ## Cost Optimization Tips
943
+ - Run BiOS bills per-SECOND, not per-hour — stop jobs early to save money
944
+ - LoRA training is 5-10x cheaper than full fine-tune for comparable quality
945
+ - QLoRA adds ~10% training time but halves VRAM needs
946
+ - Start with fewer epochs (1-2) to validate, then train full if results look good
947
+ - Use the get_recommended_gpu tool to get the optimal GPU for your specific setup
948
+
949
+ ## Multi-GPU
950
+ For models that don't fit in a single GPU's VRAM, the platform automatically shards across GPUs.
951
+ Specify gpu_count in create_training_job (default: 1).
952
+
953
+ ## Infrastructure and Price Caps
954
+ The platform selects and manages the underlying infrastructure automatically. You choose only the GPU type, the GPU count, and a maximum total hourly price cap. Preflight returns ranked alternatives and the exact price that will be honored. A price above the approved cap is rejected rather than silently accepted.`,
955
+ inference: `# Model Inference Guide
956
+
957
+ Deployments serve either a verified fine-tuning checkpoint or a base model from the catalog through an OpenAI-compatible endpoint.
958
+
959
+ ## Settings the platform derives (not tool inputs)
960
+ serving_mode, model_task and supports_images are SERVER-DERIVED and immutable,
961
+ so preflight_inference/create_inference do not accept them:
962
+ - serving_mode (full / adapter / merged) follows the verified checkpoint's
963
+ lineage; a catalog base model is served whole.
964
+ - model_task follows the resolved model — chat for an instruct or
965
+ vision-language model, completion for a base model with no chat template,
966
+ embedding/rerank for those architectures. Every model in the current catalog
967
+ resolves to chat, so there is nothing to choose.
968
+ - supports_images follows the model's own config.
969
+ Read all three from the preflight/create response and from get_inference_status.
970
+ If you copy canonical_request from preflight into create, these fields are
971
+ ignored rather than honored — the server derives them again either way.
972
+
973
+ ## Context window
974
+ - Default: min(model native max, 262144) tokens. A model whose native window is
975
+ unknown falls back to 32768; a model with a smaller native window keeps it.
976
+ - Hard ceiling: the model's OWN native max. A larger context_length is rejected
977
+ with the model's maximum in the message; there is no platform-wide maximum you
978
+ can rely on, so read model.native_max_context from preflight_inference.
979
+ - Raising the window enlarges the KV cache and can raise min_gpus, so re-read
980
+ get_inference_gpu_options after changing it.
981
+
982
+ ## Safe automation workflow
983
+ 1. Call get_inference_gpu_options with model facts to obtain model-fit counts plus authoritative deployment stock and total hourly prices.
984
+ 2. Choose a concrete GPU type/count. Never infer availability from fit alone and never substitute a different SKU without user approval.
985
+ 3. Call preflight_inference. Review selected_gpu, alternatives, queue_required, billing.authorization_amount_cents, and billing.max_price_hour_cents.
986
+ 4. If the chosen SKU is unavailable, ask the user to choose an available alternative or explicitly consent to allow_capacity_queue. The queue can wait up to seven days.
987
+ 5. Copy canonical_request from preflight—including hf_model_revision/base_model_revision—into create_inference with the approved maximum total hourly price. Store the one-time inference_key securely.
988
+ 6. Poll get_inference_status for durable queue/provisioning/loading/running state. Do not report the endpoint as live until status is running.
989
+ 7. Stop or delete when no longer needed. Resume reauthorizes capacity under the saved price cap.
990
+
991
+ ## Billing and capacity semantics
992
+ - Preflight is side-effect free.
993
+ - Create authorizes two hours at the accepted maximum price without immediately reducing wallet balance.
994
+ - One hour at the final confirmed price is captured only after capacity is accepted, and a price above the cap is rejected.
995
+ - Deployment serving supports the secure inventory tier only. Unknown inventory is reported as unknown, never as zero or available.
996
+ - Training and deployment use the shared capacity broker. It evaluates priority first, then rotates fairly across users at equal priority while preserving FIFO inside each user's own requests. This prevents a high-volume user from starving others; do not describe the queue as strict global FIFO.`,
997
+ hyperparameters: `# Hyperparameter Guide
998
+
999
+ ## Key Hyperparameters
1000
+
1001
+ ### Learning Rate
1002
+ - Default: 2e-4 (0.0002)
1003
+ - Range: 1e-5 to 5e-4
1004
+ - Higher = faster learning but risk of instability
1005
+ - Lower = more stable but slower convergence
1006
+ - Tip: Start with 2e-4 for LoRA, 2e-5 for full fine-tune
1007
+
1008
+ ### Batch Size (per_device_train_batch_size)
1009
+ - Default: 4
1010
+ - Range: 1-32 (limited by GPU VRAM)
1011
+ - Larger = smoother gradients, more stable training
1012
+ - Smaller = faster per-step but noisier
1013
+ - Use gradient_accumulation_steps to simulate larger batches
1014
+
1015
+ ### Epochs (num_train_epochs)
1016
+ - Default: 3
1017
+ - Range: 1-10
1018
+ - More epochs = more passes through data
1019
+ - Watch for overfitting: if eval loss starts increasing, you've trained too long
1020
+ - For small datasets (<1000 rows): 3-5 epochs
1021
+ - For large datasets (>10000 rows): 1-3 epochs
1022
+
1023
+ ### LoRA Rank (lora_r)
1024
+ - Default: 16
1025
+ - Range: 4-256
1026
+ - Higher = more trainable parameters, better quality, but slower
1027
+ - 8-16 is sufficient for most tasks
1028
+ - 32-64 for complex tasks requiring more model capacity
1029
+ - 128+ for near-full-fine-tune quality
1030
+
1031
+ ### LoRA Alpha (lora_alpha)
1032
+ - Default: 32 (typically 2x lora_rank)
1033
+ - Controls the scaling of LoRA updates
1034
+ - Rule of thumb: set to 2x your lora_rank
1035
+
1036
+ ### Max Sequence Length
1037
+ - Default: 2048
1038
+ - Range: 512-8192
1039
+ - Must cover your longest training example
1040
+ - Longer = more VRAM usage
1041
+ - Set to slightly above your longest example
1042
+
1043
+ ### LR Scheduler
1044
+ - cosine (default) — gradually decreases, good for most tasks
1045
+ - linear — constant decrease, predictable
1046
+ - constant_with_warmup — stays flat after warmup, good for short training
1047
+
1048
+ ### Warmup Ratio
1049
+ - Default: 0.03 (3% of total steps)
1050
+ - Prevents early instability by slowly ramping up learning rate
1051
+
1052
+ ## Recommended Defaults by Use Case
1053
+ | Parameter | Quick Test | Standard | High Quality |
1054
+ |-----------|-----------|----------|--------------|
1055
+ | Adapter | LoRA | LoRA | Full |
1056
+ | Learning Rate | 3e-4 | 2e-4 | 2e-5 |
1057
+ | Epochs | 1 | 3 | 5 |
1058
+ | Batch Size | 4 | 4 | 8 |
1059
+ | LoRA Rank | 8 | 16 | 64 |
1060
+ | Scheduler | cosine | cosine | cosine |`,
1061
+ monitoring: `# Training Monitoring Guide
1062
+
1063
+ ## Key Metrics to Watch
1064
+
1065
+ ### Training Loss
1066
+ - Should decrease steadily during training
1067
+ - Sharp drops early are normal (learning basic patterns)
1068
+ - Gradual decrease later shows fine-grained learning
1069
+ - If loss plateaus: try higher learning rate or more LoRA rank
1070
+ - If loss spikes: learning rate is too high, reduce it
1071
+
1072
+ ### Eval Loss (if eval dataset provided)
1073
+ - Should track training loss but may be slightly higher
1074
+ - If eval loss increases while training loss decreases = OVERFITTING
1075
+ - Early sign of overfitting: gap between train and eval loss growing
1076
+ - Action: Stop training and use the last checkpoint before divergence
1077
+
1078
+ ### Learning Rate
1079
+ - Should follow your chosen schedule (cosine, linear, etc.)
1080
+ - Warmup period: LR starts low and ramps up
1081
+ - Main phase: LR follows the schedule downward
1082
+
1083
+ ## Monitoring Tools
1084
+ 1. get_training_status — Progress %, current step, elapsed time, ETA
1085
+ 2. get_training_metrics — Full loss curve data for analysis
1086
+ 3. get_training_logs — Raw training output for debugging
1087
+ 4. get_training_checkpoints — Saved model snapshots at intervals
1088
+
1089
+ ## When to Stop Early
1090
+ - Loss has plateaued for many steps (no improvement)
1091
+ - Eval loss is increasing (overfitting)
1092
+ - You've reached satisfactory quality and want to save GPU cost
1093
+ - The model is producing good outputs in eval samples
1094
+
1095
+ ## Checkpoints
1096
+ - Saved automatically at regular intervals during training
1097
+ - Each checkpoint is a complete model snapshot you can use
1098
+ - If training is interrupted, resume from the last checkpoint
1099
+ - Best practice: compare outputs from different checkpoints to pick the best one`,
1100
+ cost_optimization: `# Cost Optimization Guide
1101
+
1102
+ ## Per-Second Billing
1103
+ Run BiOS charges per-SECOND of GPU usage, not per hour. This means:
1104
+ - You only pay for actual compute time
1105
+ - Stopping a job immediately stops billing
1106
+ - No wasted time rounding up to the next hour
1107
+
1108
+ ## Cost Estimation
1109
+ Before launching a job, estimate cost:
1110
+ 1. Call get_recommended_gpu for your model — it includes a cost estimate
1111
+ 2. Rough formula: (dataset_rows / batch_size) × epochs × seconds_per_step × gpu_price_per_second
1112
+
1113
+ ## Money-Saving Strategies
1114
+
1115
+ ### 1. Start Small
1116
+ - Train for 1 epoch first to validate your setup
1117
+ - Check if the loss is decreasing and outputs look reasonable
1118
+ - Only then train for the full 3-5 epochs
1119
+
1120
+ ### 2. Use LoRA
1121
+ - LoRA is 5-10x cheaper than full fine-tune
1122
+ - Quality is comparable for most tasks
1123
+ - QLoRA is even cheaper (uses 4-bit quantization)
1124
+
1125
+ ### 3. Optimize Batch Size
1126
+ - Larger batch sizes = fewer total steps = less time
1127
+ - Use gradient accumulation to simulate large batches on limited VRAM
1128
+
1129
+ ### 4. Choose the Right GPU
1130
+ - Don't pick H100 for a 7B LoRA job — A100 is plenty and cheaper
1131
+ - Use get_recommended_gpu to find the cost-optimal choice
1132
+
1133
+ ### 5. Monitor and Stop Early
1134
+ - Watch the loss curve via get_training_metrics
1135
+ - If loss has plateaued, stop early — more epochs won't help
1136
+ - Check wallet balance with get_wallet_balance before long jobs
1137
+
1138
+ ### 6. Check Your Balance First
1139
+ - Call get_wallet_balance before creating a job
1140
+ - Ensure you have enough credits for the estimated training time
1141
+ - Set up auto top-up to avoid interruptions mid-training`,
1142
+ capacity_errors: `# Standard GPU Rejection Errors
1143
+
1144
+ Training and inference speak ONE GPU rejection contract: one body shape, and a
1145
+ code that tells you whether waiting can ever help. This MCP server hands you
1146
+ that body as compact JSON in the tool-error text.
1147
+
1148
+ ## Two classes, and the difference matters
1149
+ - CAPACITY_UNAVAILABLE (HTTP 409, reason insufficient_stock) is the ONLY
1150
+ transient one: the GPU is simply not free right now. Retrying later or
1151
+ joining the queue can succeed. Training also echoes the deprecated alias
1152
+ SELECTED_GPU_UNAVAILABLE as legacy_code for one release.
1153
+ - GPU_TYPE_TOO_SMALL (model_too_large), GPU_COUNT_BELOW_MINIMUM
1154
+ (below_model_minimum), GPU_COUNT_INVALID (invalid_gpu_count) and
1155
+ GPU_TYPE_UNSUPPORTED (gpu_unsupported) are HTTP 400 and PERMANENT: the
1156
+ request as submitted can never run on that GPU, whatever frees up. Change
1157
+ gpu_type or gpu_count. Do NOT wait and do NOT queue.
1158
+
1159
+ Read retryable_as_submitted (false = permanent) or queue_offered instead of
1160
+ parsing the message.
1161
+
1162
+ ## Body fields
1163
+ - reason: insufficient_stock | model_too_large | below_model_minimum | invalid_gpu_count | gpu_unsupported
1164
+ - selected: the gpu_type/gpu_count/tier that was refused, with its live availability_status/available_count
1165
+ - minimum_requirement: selected_gpu_min + selected_valid_counts for the chosen type, plus the per_type table (min_gpus + valid_counts for every type that fits the model). NEVER submit below min_gpus or outside valid_counts.
1166
+ - available_gpus: ALL currently bookable alternatives, cheapest first, each at its minimum count with valid_counts, available_count and price_hour_cents (available_alternatives is the deprecated alias)
1167
+ - queue_offered: true only for a stock miss. When it is true, you may resubmit with allow_capacity_queue=true to wait for stock, and nothing is charged while waiting. When it is false the queue cannot help this request at all.
1168
+ - queue_eligible: whether THIS request already asked to queue (never true when queue_offered is false)
1169
+ - gpu_priorities_entry: 1-based rank of the gpu_priorities entry that was refused, when the rejection belongs to one
1170
+ - checked_at: when the market snapshot behind the verdict was read
1171
+
1172
+ ## How to recover
1173
+ 1. Pick an entry from available_gpus and retry the SAME call with its gpu_type and a gpu_count >= its min_gpus (the listed gpu_count is that minimum). With zero alternatives, read minimum_requirement.per_type for the types that fit.
1174
+ 2. Only when queue_offered is true: resubmit with allow_capacity_queue=true (explicit user consent required) to auto-deploy when stock frees.
1175
+ 3. Never invent a GPU/count; never treat unknown availability as out-of-stock; never re-send an unchanged request that was refused as permanent.
1176
+
1177
+ ## Book-before-reveal (inference creates)
1178
+ A non-queued create_inference answers 202 with a booking handle while a
1179
+ GPU is secured on real capacity (30-40s typical). The deployment id and the
1180
+ ONE-TIME inference_key exist only after the booking is confirmed. A definitive
1181
+ miss returns this standard error with FRESH alternatives and NO deployment
1182
+ exists (nothing charged, nothing to clean up). Poll get_inference_booking for
1183
+ long bookings.
1184
+
1185
+ ## Transient outages are NOT capacity answers
1186
+ 503 ADMISSION_UNAVAILABLE (market/broker unreadable) means retry shortly.
1187
+ Never conclude out-of-stock from it.
1188
+
1189
+ ## After start
1190
+ Once a deployment has ever been booked, losing its GPU NEVER ends in a
1191
+ capacity failure: the platform replaces it across the ranked backup ladder and,
1192
+ when no stock exists anywhere, parks it as queued_capacity (waiting, zero
1193
+ charge) until stock returns. gpu_priorities backups may be ranked on ANY
1194
+ create (1-5 choices), and the queue itself stays opt-in.`,
1195
+ };
1196
+ // The overview and quick_start teach both creation workflows, so either
1197
+ // gate swaps them out; the single-surface topics follow their own gate.
1198
+ if (launchGates.training || launchGates.datasets) {
1199
+ guides.overview = gatedOverview;
1200
+ guides.quick_start = comingSoonNote;
1201
+ }
1202
+ if (launchGates.training) {
1203
+ guides.training_methods = comingSoonNote;
1204
+ guides.gpu_selection = comingSoonNote;
1205
+ guides.hyperparameters = comingSoonNote;
1206
+ guides.cost_optimization = comingSoonNote;
1207
+ }
1208
+ if (launchGates.datasets) {
1209
+ guides.datasets = comingSoonNote;
1210
+ }
1211
+ if (topic && guides[topic]) {
1212
+ return result(guides[topic]);
1213
+ }
1214
+ return result(guides.overview);
1215
+ });
1216
+ /* ══════════════════════════════════════════════════════════════════════════ */
1217
+ /* TOOL: introspect_api_key */
1218
+ /* ══════════════════════════════════════════════════════════════════════════ */
1219
+ server.tool("introspect_api_key", "Discover what this API key can do. Returns the key's permissions (scopes), the org and workspace it's bound to, allowed MCP tools and SDK methods, rate limits, and expiration. Call this first to understand your access level and available capabilities.", {}, async () => {
1220
+ const data = await client.api("/api/api-keys/introspect");
1221
+ return json(data);
1222
+ });
1223
+ /* ══════════════════════════════════════════════════════════════════════════ */
1224
+ /* TOOL: get_wallet_balance */
1225
+ /* ══════════════════════════════════════════════════════════════════════════ */
1226
+ server.tool("get_wallet_balance", "Check your current wallet balance, available credits, and pending charges. Call this before creating a training job to make sure you have enough credits. Returns available balance in dollars, pending charges from running jobs, and auto top-up status.", {}, async () => {
1227
+ const data = await client.api("/api/billing/wallet");
1228
+ return json(data);
1229
+ });
1230
+ /* ══════════════════════════════════════════════════════════════════════════ */
1231
+ /* TOOL: search_models */
1232
+ /* ══════════════════════════════════════════════════════════════════════════ */
1233
+ server.tool("search_models", launchGates.training
1234
+ ? "Search the Run BiOS model catalog — the platform's own hosted, verified models. These are the ONLY models that can be deployed on Run BiOS; a model that is not listed is not hosted yet (ask an admin to mirror it). Filter by name, provider, or parameter count. Returns model id, author, parameter counts, architecture, and downloads/likes for each result."
1235
+ : "Search the Run BiOS model catalog — the platform's own hosted, verified models. These are the ONLY models that can be fine-tuned or deployed on Run BiOS; a model that is not listed is not hosted yet (ask an admin to mirror it). Filter by name, provider, or parameter count. Returns model id, author, parameter counts, architecture, and downloads/likes for each result.", {
1236
+ query: z.string().optional().describe("Search by model name (e.g., 'llama', 'mistral', 'phi', 'qwen')"),
1237
+ provider: z.string().optional().describe("Filter by provider/author (e.g., 'meta-llama', 'mistralai', 'microsoft', 'Qwen')"),
1238
+ max_params: z.string().optional().describe("Maximum parameter count (e.g., '7B', '13B', '70B')"),
1239
+ }, async ({ query, provider, max_params }) => {
1240
+ // The registry matches query tokens against normalized repo ids
1241
+ // ("author/name"), so folding the provider into q turns it into a real
1242
+ // author filter instead of a silently ignored parameter.
1243
+ const q = [provider, query].filter(Boolean).join(" ").trim() || undefined;
1244
+ const data = await client.api("/api/public/model-search", {
1245
+ params: { q, max_params },
1246
+ });
1247
+ return json(data);
1248
+ });
1249
+ /* ══════════════════════════════════════════════════════════════════════════ */
1250
+ /* TOOL: list_models */
1251
+ /* ══════════════════════════════════════════════════════════════════════════ */
1252
+ server.tool("list_models", launchGates.training
1253
+ ? "List every model hosted on Run BiOS — the platform's own verified registry, mirrored in Run BiOS storage (never a live Hugging Face listing). These are the only models that can be deployed. Call this before create_inference to pick a valid model id; use search_models to filter the same catalog by text."
1254
+ : "List every model hosted on Run BiOS — the platform's own verified registry, mirrored in Run BiOS storage (never a live Hugging Face listing). These are the only models that can be fine-tuned or deployed. Call this before create_training_job or create_inference to pick a valid model id; use search_models to filter the same catalog by text.", {
1255
+ type: z.enum(["all", "llm", "vlm"]).optional().describe("Filter by model surface (default: all)."),
1256
+ sort: z.enum(["downloads", "likes", "trending", "name"]).optional().describe("Sort order (default: downloads)."),
1257
+ limit: z.number().int().min(1).max(60).optional().describe("Page size (default 24, max 60)."),
1258
+ offset: z.number().int().min(0).optional().describe("Pagination offset (default 0)."),
1259
+ }, async ({ type, sort, limit, offset }) => {
1260
+ const data = await client.api("/api/public/model-search", {
1261
+ params: {
1262
+ type: type && type !== "all" ? type : undefined,
1263
+ sort,
1264
+ limit: limit !== undefined ? String(limit) : undefined,
1265
+ offset: offset !== undefined ? String(offset) : undefined,
1266
+ },
1267
+ });
1268
+ return json(data);
1269
+ });
1270
+ /**
1271
+ * Registry gate for model selection: Run BiOS can only train and serve models it
1272
+ * hosts in its own registry. A definitive registry miss gets a clear,
1273
+ * actionable error pointing at the catalog; any failure to ANSWER (registry
1274
+ * unreachable) stays advisory — the server-side gate re-validates
1275
+ * authoritatively and unknown never fails closed.
1276
+ */
1277
+ async function assertModelHosted(modelId, tool) {
1278
+ const status = await client.modelRegistryStatus(modelId);
1279
+ if (status === "not_hosted") {
1280
+ // Same compact-JSON convention as the capacity errors: an MCP client is
1281
+ // an AI agent and can act on the instruction directly. The instruction
1282
+ // names every route to a valid id (see CATALOG_RECOVERY) rather than a
1283
+ // single tool, so the advice still works when that tool is degraded.
1284
+ throw new Error(JSON.stringify({
1285
+ code: "MODEL_NOT_HOSTED",
1286
+ message: `"${modelId}" is not in the Run BiOS model catalog, so it cannot be trained or served. Run BiOS only runs models it hosts and mirrors itself; ask an admin to mirror this one if you need it.`,
1287
+ recoverable: true,
1288
+ instruction: `${CATALOG_RECOVERY} Then retry ${tool} with that id.`,
1289
+ }));
1290
+ }
1291
+ }
1292
+ /**
1293
+ * Rewrite a model-resolution failure into the answer that is both TRUE and
1294
+ * actionable. The control plane surfaces its mirror's transport error verbatim
1295
+ * ("Could not resolve the model for sizing: Hugging Face request returned
1296
+ * 404"), which names an upstream vendor the caller cannot act on and never
1297
+ * states the real answer: that id is not a Run BiOS catalog model. Callers here
1298
+ * have already passed the registry gate, so a definitive catalog miss is
1299
+ * impossible at this point — the remaining causes are a bad revision or a
1300
+ * catalog entry whose mirrored facts cannot be read right now. Anything that is
1301
+ * not a resolution failure (capacity JSON, auth, validation) is returned
1302
+ * untouched.
1303
+ */
1304
+ function modelResolveFailure(error, modelId, tool) {
1305
+ const text = error instanceof Error ? error.message : String(error);
1306
+ const isResolveFailure = /hugging\s*face/i.test(text)
1307
+ || /could not resolve (the model|an immutable)/i.test(text)
1308
+ || /could not establish .*(immutable|commit)/i.test(text);
1309
+ if (!isResolveFailure)
1310
+ return error instanceof Error ? error : new Error(text);
1311
+ const subject = modelId ? `"${modelId}"` : "the requested model";
1312
+ if (/revision|commit/i.test(text)) {
1313
+ return new Error(JSON.stringify({
1314
+ code: "MODEL_REVISION_UNRESOLVED",
1315
+ message: `Run BiOS could not pin an exact, immutable commit for ${subject}, so it refused to size or deploy it. The revision you supplied is not one it can verify.`,
1316
+ recoverable: true,
1317
+ instruction: `Omit the revision to let the platform pin the catalog's current commit, or copy hf_model_revision/base_model_revision unchanged from preflight_inference's canonical_request. Then retry ${tool}.`,
1318
+ model: modelId,
1319
+ }));
1320
+ }
1321
+ return new Error(JSON.stringify({
1322
+ code: "MODEL_NOT_RESOLVED",
1323
+ message: `Run BiOS could not resolve ${subject} from its own model catalog, so it cannot be sized or deployed. A model that is not in the catalog can never resolve; a catalog model can also fail this way while its mirrored facts are being re-synced.`,
1324
+ recoverable: true,
1325
+ instruction: `${CATALOG_RECOVERY} Then retry ${tool}. If the id IS listed in the catalog, report a catalog sync problem to the user or an admin instead of retrying in a loop.`,
1326
+ model: modelId,
1327
+ }));
1328
+ }
1329
+ /* ══════════════════════════════════════════════════════════════════════════ */
1330
+ /* TOOL: get_model_config */
1331
+ /* ══════════════════════════════════════════════════════════════════════════ */
1332
+ server.tool("get_model_config", launchGates.training
1333
+ ? "Get configuration recommendations for a specific model hosted on Run BiOS (see list_models). Returns supported training methods, recommended GPU type and count, default hyperparameters, minimum VRAM required, and estimated cost per hour."
1334
+ : "Get training configuration recommendations for a specific model hosted on Run BiOS (see list_models). Returns supported training methods, recommended GPU type and count, default hyperparameters, minimum VRAM required, and estimated cost per hour. Use this after search_models to plan your training job.", {
1335
+ model: z.string().describe("Full model id from the Run BiOS catalog (e.g., 'meta-llama/Llama-3.1-8B-Instruct')"),
1336
+ model_revision: z
1337
+ .string()
1338
+ .max(256)
1339
+ .optional()
1340
+ .describe("Model branch, tag, or commit. The response returns the exact immutable commit used for sizing."),
1341
+ integration_id: z
1342
+ .string()
1343
+ .optional()
1344
+ .describe("Legacy no-op for catalog models — hosted models are never gated. Integrations are used for Hugging Face dataset imports."),
1345
+ }, async ({ model }) => {
1346
+ const data = await client.api("/api/public/model-config", { params: { id: model } });
1347
+ return json(data);
1348
+ });
1349
+ /* ══════════════════════════════════════════════════════════════════════════ */
1350
+ /* TOOL: list_supported_architectures */
1351
+ /* ══════════════════════════════════════════════════════════════════════════ */
1352
+ server.tool("list_supported_architectures", "List the architectures Run BiOS supports, so you know what can be trained or served before you try. scope='inference' returns the model architecture classes (e.g. LlamaForCausalLM) that can be DEPLOYED for serving; scope='training' returns the architecture keys the fine-tuning gate accepts. If count is 0 the registry is empty and nothing is explicitly restricted (every architecture Run BiOS supports is allowed). Support here is authoritative: the deployment and training gates read these same rows. The response carries NO build or version identity of any kind — read manifest_provenance, which says how many rows came from a published manifest without naming any build. Never report a platform version from this tool. No authentication required.", {
1353
+ scope: z.enum(["inference", "training"]).optional().describe("Which registry to read. Defaults to 'inference' (servable/deployable architectures)."),
1354
+ }, async ({ scope }) => {
1355
+ const data = await client.api("/api/public/serving-architectures", { params: { scope: scope || "inference" } });
1356
+ // The registry TABLE holds the build identity of the engine image whose
1357
+ // capability manifest published each row (image_tag/bios_version/
1358
+ // manifest_sha256, and the same tag again inside free-text `notes`). None
1359
+ // of it is a customer fact and it must never leave the platform, so the
1360
+ // endpoint now projects those columns away before answering.
1361
+ //
1362
+ // Both passes still run, in this order, and neither is redundant — they
1363
+ // defend against the projection regressing, which is the exact way this
1364
+ // leak reached production the first time:
1365
+ // 1. annotateArchitectureProvenance COUNTS how many rows came from a
1366
+ // manifest (from `source`, which survives the projection) and states
1367
+ // what the caller may conclude, so an agent cannot read provenance as
1368
+ // "the version the platform runs now". It goes FIRST because it needs
1369
+ // the rows before step 2 rewrites them.
1370
+ // 2. sanitizeArchitectureRegistry then DROPS every build-identity key at
1371
+ // any depth and content-redacts what is left, so anything the
1372
+ // projection ever fails to strip still never ships.
1373
+ return json(sanitizeArchitectureRegistry(annotateArchitectureProvenance(data)));
1374
+ });
1375
+ /* ══════════════════════════════════════════════════════════════════════════ */
1376
+ /* TOOL: get_gpu_pricing */
1377
+ /* ══════════════════════════════════════════════════════════════════════════ */
1378
+ server.tool("get_gpu_pricing", "Get current GPU pricing and availability. Returns all available GPU types with display name, VRAM in GB, per-second cost, availability status, and supported model sizes. No authentication required.", {}, async () => {
1379
+ const data = await client.api("/api/public/gpu-pricing");
1380
+ return json(data);
1381
+ });
1382
+ /* ══════════════════════════════════════════════════════════════════════════ */
1383
+ /* TOOL: get_recommended_gpu */
1384
+ /* ══════════════════════════════════════════════════════════════════════════ */
1385
+ server.tool("get_recommended_gpu", "Get authoritative, model-aware GPU options from the training control plane. Returns required counts, live availability, total hourly prices, the cheapest currently bookable recommendation, and compatible alternatives when no GPU is available. It does not invent a duration or total-cost estimate.", {
1386
+ model: z.string().describe("Full model name (e.g., 'meta-llama/Llama-3.1-8B-Instruct')"),
1387
+ adapter: z.enum(TRAINING_ADAPTERS).optional().describe("Adapter type (default: 'lora')."),
1388
+ method: z.enum(["sft", "pt"]).optional().describe("Canonical training method."),
1389
+ model_params_b: z.number().positive().optional().describe("Optional total-parameter hint in billions."),
1390
+ model_active_params_b: z
1391
+ .number()
1392
+ .positive()
1393
+ .optional()
1394
+ .describe("Optional active-parameter hint for MoE models in billions."),
1395
+ }, async (params) => {
1396
+ const data = await client.api("/api/training/gpu-options", {
1397
+ params: buildGPUOptionsParams(params),
1398
+ });
1399
+ return json(data);
1400
+ });
1401
+ /* ══════════════════════════════════════════════════════════════════════════ */
1402
+ /* TOOL: get_inference_gpu_options */
1403
+ /* ══════════════════════════════════════════════════════════════════════════ */
1404
+ server.tool("get_inference_gpu_options", "Get inference model-fit GPU choices joined to the authoritative deployment inventory and pricing snapshot. PREFER model-addressed sizing: pass model=<model_id> (optionally revision / hf_integration_id) and the SERVER resolves the facts (vision-aware) and returns computed min_gpus, valid_counts, and bookable_counts — the exact minimums the create gate enforces. Returns valid tensor-parallel counts, exact total hourly prices, availability as available/out_of_stock/unknown, and ranked alternatives. Never interpret unknown inventory as available, never offer a count below min_gpus or outside bookable_counts, and never substitute a SKU without user approval.", {
1405
+ model: z.string().optional().describe("Model id for SERVER-resolved sizing (recommended); replaces every client-fact field below."),
1406
+ revision: z.string().optional().describe("Exact model commit for model-addressed sizing."),
1407
+ inference_id: z.string().optional().describe("Size from an existing deployment's pinned facts."),
1408
+ hf_integration_id: z.string().optional().describe("Legacy field kept for compatibility — catalog models are pre-mirrored and never gated, so no integration is needed."),
1409
+ params_b: z.number().positive().optional().describe("DEPRECATED client-fact path: total model parameters in billions; required only when model/inference_id are absent."),
1410
+ active_params_b: z.number().positive().optional().describe("Active MoE parameters in billions, used only for throughput hints."),
1411
+ is_moe: z.boolean().optional(),
1412
+ is_multimodal: z.boolean().optional().describe("Vision tower present (client-fact path only): reserves extra VRAM and can raise minimums. Model-addressed sizing detects this automatically."),
1413
+ quant: quantSchema("Precision to size the weights and KV cache for."),
1414
+ context_length: contextLengthSchema("Context window to size the KV cache for."),
1415
+ kv_heads: z.number().int().positive().optional(),
1416
+ num_layers: z.number().int().positive().optional(),
1417
+ kv_layers: z.number().int().positive().optional(),
1418
+ head_dim: z.number().int().positive().optional(),
1419
+ attention: z.enum(["gqa", "mla"]).optional(),
1420
+ num_attention_heads: z.number().int().positive().optional(),
1421
+ kv_lora_rank: z.number().int().positive().optional(),
1422
+ qk_rope_head_dim: z.number().int().positive().optional(),
1423
+ gpu_tier: z.enum(DEPLOYMENT_GPU_TIERS).optional(),
1424
+ }, async (params) => {
1425
+ // Gate the model against the Run BiOS catalog FIRST. Without it the control
1426
+ // plane answers a sizing miss with its upstream mirror's 404 text, which
1427
+ // names a vendor and hides the real answer ("that id is not in the Run BiOS
1428
+ // catalog"). A registry that cannot answer stays advisory, exactly as on
1429
+ // the create path. This endpoint also accepts the "repo@revision" form, so
1430
+ // probe the repo id alone — the revision is not part of catalog identity.
1431
+ const catalogId = params.model?.split("@")[0]?.trim();
1432
+ if (catalogId)
1433
+ await assertModelHosted(catalogId, "get_inference_gpu_options");
1434
+ try {
1435
+ const data = await client.api("/api/inference/gpu-options", {
1436
+ params: buildDeploymentGPUOptionsParams(params),
1437
+ });
1438
+ return json(data);
1439
+ }
1440
+ catch (error) {
1441
+ throw modelResolveFailure(error, params.model, "get_inference_gpu_options");
1442
+ }
1443
+ });
1444
+ /* ══════════════════════════════════════════════════════════════════════════ */
1445
+ /* TOOL: preflight_inference */
1446
+ /* ══════════════════════════════════════════════════════════════════════════ */
1447
+ server.tool("preflight_inference", "Validate and canonicalize a deployment without creating a database row, authorizing funds, entering a queue, or allocating a GPU. Returns verified model facts, selected price/stock, compatible alternatives, queue requirements, a request hash, and two-hour wallet-authorization terms. The response is also where you READ the derived serving settings: serving_mode, model_task, supports_images and the resolved context_length are chosen by the platform from the model and checkpoint lineage and cannot be set as inputs. Review this result with the user before create_inference.", deploymentCreateToolSchema, async (params) => {
1448
+ if (params.source_type === "hf_model" && params.hf_model_id) {
1449
+ await assertModelHosted(params.hf_model_id, "preflight_inference");
1450
+ }
1451
+ try {
1452
+ const data = await client.api("/api/inference/preflight", {
1453
+ method: "POST",
1454
+ body: buildDeploymentBody(params),
1455
+ });
1456
+ return json(data);
1457
+ }
1458
+ catch (error) {
1459
+ throw modelResolveFailure(error, params.hf_model_id ?? params.base_model_id, "preflight_inference");
1460
+ }
1461
+ });
1462
+ /* ══════════════════════════════════════════════════════════════════════════ */
1463
+ /* TOOL: create_inference */
1464
+ /* ══════════════════════════════════════════════════════════════════════════ */
1465
+ /**
1466
+ * Render a GPU type at a count readably: the type name already says GPU, so
1467
+ * only the plural marker is added ("1 H100_80GB" vs "2 H100_80GB GPUs"). Keeps
1468
+ * the synthesized copy identical to the server's.
1469
+ */
1470
+ const gpusOfType = (gpuType, count) => (count === 1 ? gpuType : `${gpuType} GPUs`);
1471
+ /**
1472
+ * Self-preflight (book-first §2): validate the chosen gpu_type/gpu_count
1473
+ * against the SERVER's model-addressed sizing (computed min_gpus/valid_counts)
1474
+ * BEFORE any create POST, and synthesize the standard structured GPU rejection
1475
+ * when the selection can never be booked. Every rejection this raises is
1476
+ * permanent (GPU_TYPE_TOO_SMALL / GPU_COUNT_BELOW_MINIMUM / GPU_COUNT_INVALID),
1477
+ * never CAPACITY_UNAVAILABLE: the selection is unbookable whatever the stock is.
1478
+ * Advisory only: a failure to ANSWER (endpoint unreachable) never blocks, the
1479
+ * create gate re-validates authoritatively and unknown never fails closed.
1480
+ */
1481
+ async function assertInferenceGpuSelectionBookable(body) {
1482
+ const model = (body.hf_model_id || body.base_model_id);
1483
+ const gpuType = body.gpu_type;
1484
+ const gpuCount = body.gpu_count;
1485
+ if (!model || !gpuType || !Number.isInteger(gpuCount))
1486
+ return;
1487
+ let options;
1488
+ try {
1489
+ options = await client.api("/api/inference/gpu-options", {
1490
+ params: {
1491
+ model: String(model),
1492
+ revision: (body.hf_model_revision || body.base_model_revision),
1493
+ quant: body.quant,
1494
+ context_length: body.context_length !== undefined ? String(body.context_length) : undefined,
1495
+ hf_integration_id: body.hf_integration_id,
1496
+ },
1497
+ });
1498
+ }
1499
+ catch {
1500
+ return; // advisory only — the create gate is the enforcement floor
1501
+ }
1502
+ const chosen = options?.options?.find((option) => option?.gpu_type === gpuType);
1503
+ if (!chosen)
1504
+ return;
1505
+ const minGpus = chosen.min_gpus || 0;
1506
+ const validCounts = (chosen.valid_counts || []).filter((count) => Number.isInteger(count));
1507
+ let reason;
1508
+ let message = "";
1509
+ if (chosen.selectable === false) {
1510
+ reason = "model_too_large";
1511
+ message = `This model does not fit on ${gpuType}, at any GPU count.`;
1512
+ }
1513
+ else if (minGpus > 0 && gpuCount < minGpus) {
1514
+ reason = "below_model_minimum";
1515
+ message = `This model needs at least ${minGpus} ${gpusOfType(gpuType, minGpus)} to run, and this request asked for ${gpuCount}.`;
1516
+ }
1517
+ else if (validCounts.length > 0 && !validCounts.includes(gpuCount)) {
1518
+ reason = "invalid_gpu_count";
1519
+ message = `This model cannot be split across ${gpuCount} ${gpusOfType(gpuType, gpuCount)}. Use one of these GPU counts instead: ${validCounts.join(", ")}.`;
1520
+ }
1521
+ if (!reason)
1522
+ return;
1523
+ const selectable = (options.options || []).filter((option) => option && option.selectable !== false);
1524
+ const alternatives = selectable
1525
+ .filter((option) => option.market?.availability_status === "available")
1526
+ .map((option) => ({
1527
+ gpu_type: option.gpu_type,
1528
+ gpu_count: option.min_gpus,
1529
+ min_gpus: option.min_gpus,
1530
+ valid_counts: option.valid_counts,
1531
+ tier: "secure",
1532
+ available_count: option.market?.available_count ?? undefined,
1533
+ price_hour_cents: (option.market?.price_per_gpu_hour_cents || 0) * (option.min_gpus || 1),
1534
+ }));
1535
+ // Same structured shape, codes and instruction text the transport serializes
1536
+ // for real API GPU rejections, so agent clients handle exactly one contract.
1537
+ // Every reason this pre-check can raise is a permanent fact about the request
1538
+ // (the model does not fit, the count is below the model's minimum, the count
1539
+ // cannot be used), so the code is never CAPACITY_UNAVAILABLE and the queue is
1540
+ // never offered: waiting for stock could not book any of them.
1541
+ const permanent = isPermanentGpuCode(gpuRejectionCodeFor(reason));
1542
+ throw new Error(JSON.stringify({
1543
+ code: gpuRejectionCodeFor(reason),
1544
+ status: gpuRejectionStatusFor(reason),
1545
+ reason,
1546
+ message,
1547
+ recoverable: true,
1548
+ retryable_as_submitted: !permanent,
1549
+ instruction: gpuRejectionInstruction({
1550
+ permanent,
1551
+ hasAlternatives: alternatives.length > 0,
1552
+ call: "create_inference",
1553
+ }),
1554
+ available_gpus: alternatives,
1555
+ minimum_requirement: {
1556
+ selected_gpu_min: minGpus,
1557
+ selected_valid_counts: validCounts,
1558
+ per_type: selectable
1559
+ .filter((option) => (option.min_gpus || 0) >= 1)
1560
+ .map((option) => ({
1561
+ gpu_type: option.gpu_type,
1562
+ min_gpus: option.min_gpus,
1563
+ valid_counts: option.valid_counts || [],
1564
+ })),
1565
+ },
1566
+ selected: { gpu_type: gpuType, gpu_count: gpuCount, tier: body.gpu_tier || "secure", availability_status: "unknown" },
1567
+ // A permanent rejection must not advertise the queue, and must not report
1568
+ // itself eligible for it either: a caller branching on queue_eligible would
1569
+ // park a request that can never book.
1570
+ queue_offered: !permanent,
1571
+ queue_eligible: body.allow_capacity_queue === true && !permanent,
1572
+ checked_at: options.availability_checked_at,
1573
+ }));
1574
+ }
1575
+ /**
1576
+ * Book-before-reveal (book-first §1): poll a 202 booking handle to its
1577
+ * terminal outcome. Pending keeps polling; a transient 503/market outage
1578
+ * keeps polling (never a capacity verdict); the definitive miss throws the
1579
+ * structured CAPACITY_UNAVAILABLE the transport already serialized; on
1580
+ * timeout the raw booking body is returned with an instruction to keep
1581
+ * polling get_inference_booking.
1582
+ */
1583
+ async function pollInferenceBookingToOutcome(handle, timeoutMs) {
1584
+ const deadline = Date.now() + timeoutMs;
1585
+ for (;;) {
1586
+ if (Date.now() > deadline) {
1587
+ return {
1588
+ booking: { handle, status: "booking" },
1589
+ instruction: `The GPU booking is still concluding. Nothing is charged and no deployment exists until a GPU is confirmed. Call get_inference_booking with handle "${handle}" to continue polling.`,
1590
+ };
1591
+ }
1592
+ await new Promise((resolve) => setTimeout(resolve, 2_000));
1593
+ let outcome;
1594
+ try {
1595
+ outcome = await client.api(`/api/inference/bookings/${encodeURIComponent(handle)}`);
1596
+ }
1597
+ catch (error) {
1598
+ const text = error instanceof Error ? error.message : String(error);
1599
+ // ADMISSION_UNAVAILABLE / 503: transient market outage — keep polling,
1600
+ // never fail-closed. Every other error (including the structured
1601
+ // CAPACITY_UNAVAILABLE miss) is terminal.
1602
+ if (text.includes("ADMISSION_UNAVAILABLE") || text.includes("error 503"))
1603
+ continue;
1604
+ throw error;
1605
+ }
1606
+ if (outcome?.booking?.status === "booking")
1607
+ continue;
1608
+ return outcome;
1609
+ }
1610
+ }
1611
+ server.tool("create_inference", "Create an OpenAI-compatible serving deployment using the exact canonical_request approved through preflight_inference, including hf_model_revision/base_model_revision. Serving mode, OpenAI task and image support are derived by the platform (not inputs) and come back on the response. BOOK-BEFORE-REVEAL: a non-queued create answers 202 with a booking handle while the GPU is secured on real capacity (30-40s typical); this tool polls it and returns the full payload (one-time inference_key) only after the GPU is confirmed — a definitive miss returns the structured CAPACITY_UNAVAILABLE with fresh available_gpus + minimum_requirement and NO deployment exists. Queueing occurs only with explicit allow_capacity_queue=true, max_price_hour_cents prevents a later price increase. Reuse idempotency_key after a timeout to recover the same booking/deployment.", deploymentCreateMutationToolSchema, async ({ idempotency_key, ...params }) => {
1612
+ if (params.source_type === "hf_model" && params.hf_model_id) {
1613
+ await assertModelHosted(params.hf_model_id, "create_inference");
1614
+ }
1615
+ const body = buildDeploymentBody(params);
1616
+ await assertInferenceGpuSelectionBookable(body);
1617
+ try {
1618
+ const data = await client.api("/api/inference", {
1619
+ method: "POST",
1620
+ body,
1621
+ headers: { "Idempotency-Key": idempotency_key },
1622
+ });
1623
+ if (data?.booking?.handle) {
1624
+ return json(await pollInferenceBookingToOutcome(String(data.booking.handle), 120_000));
1625
+ }
1626
+ return json(data);
1627
+ }
1628
+ catch (error) {
1629
+ throw modelResolveFailure(error, params.hf_model_id ?? params.base_model_id, "create_inference");
1630
+ }
1631
+ });
1632
+ /* ══════════════════════════════════════════════════════════════════════════ */
1633
+ /* TOOL: get_inference_booking */
1634
+ /* ══════════════════════════════════════════════════════════════════════════ */
1635
+ server.tool("get_inference_booking", "Poll a book-before-reveal booking handle from create_inference. Returns {booking:{status:'booking'}} while the GPU is being confirmed, the full create payload (with the ONE-TIME inference_key, exactly once) once the GPU is secured, the structured CAPACITY_UNAVAILABLE error on a definitive miss (no deployment exists, nothing charged), or a retryable ADMISSION_UNAVAILABLE 503 during a transient market outage (never treat it as out-of-stock).", { handle: z.string().describe("Booking handle from create_inference's 202 response.") }, async ({ handle }) => {
1636
+ const data = await client.api(`/api/inference/bookings/${encodeURIComponent(handle)}`);
1637
+ return json(data);
1638
+ });
1639
+ /* ══════════════════════════════════════════════════════════════════════════ */
1640
+ /* TOOL: list_inferences */
1641
+ /* ══════════════════════════════════════════════════════════════════════════ */
1642
+ server.tool("list_inferences", "List one bounded, newest-first page of serving deployments in the API key's workspace. Reuse next_cursor with the same status/search filters to continue; start without a cursor after filters change. The returned lifecycle values are durable control-plane state, not a promise that an endpoint is live unless status is running.", {
1643
+ limit: z.number().int().min(1).max(200).optional().describe("Page size (default 50, maximum 200)."),
1644
+ cursor: z.string().max(1024).optional().describe("Opaque next_cursor from the preceding page."),
1645
+ status: z.enum([
1646
+ "all", "running", "provisioning", "stopped", "failed", "queued_capacity", "degraded",
1647
+ "paused_insufficient_funds", "crash_loop", "downloading_weights", "loading_model",
1648
+ ]).optional(),
1649
+ search: z.string().max(120).optional().describe("Case-insensitive deployment-name prefix."),
1650
+ }, async ({ limit, cursor, status, search }) => {
1651
+ const params = new URLSearchParams();
1652
+ if (limit !== undefined)
1653
+ params.set("limit", String(limit));
1654
+ if (cursor)
1655
+ params.set("cursor", cursor);
1656
+ if (status && status !== "all")
1657
+ params.set("status", status);
1658
+ if (search?.trim())
1659
+ params.set("search", search.trim());
1660
+ const query = params.toString();
1661
+ const data = await client.api(`/api/inference${query ? `?${query}` : ""}`);
1662
+ return json(data);
1663
+ });
1664
+ /* ══════════════════════════════════════════════════════════════════════════ */
1665
+ /* TOOL: get_inference_status */
1666
+ /* ══════════════════════════════════════════════════════════════════════════ */
1667
+ server.tool("get_inference_status", "Read durable deployment status, desired state, loading progress, endpoint, errors, selected GPU/price cap, queue entry/expiry, notification summary, and wallet-authorization state. Also returns status_summary: the lifecycle phase, the GPU (count x SKU), whether the endpoint may be used, and the queue/GPU state, derived from this same response so a field that is null before the pod reports is not mistaken for 'no GPU' or 'no phase'. A queue deadline keeps status=failed for filter compatibility and sets status_reason/error_code=queue_expired. Treat queued_capacity as waiting and report the endpoint live only when status_summary.endpoint_usable is true. The derived serving settings (serving_mode, model_task, supports_images, context_length) are reported here too — they are platform-chosen, never caller-chosen.", { deployment_id: z.string().describe("Deployment ID.") }, async ({ deployment_id }) => {
1668
+ validateId(deployment_id, "deployment_id");
1669
+ const data = await client.api(`/api/inference/${deployment_id}`);
1670
+ // Same raw-text class as the training job detail (top-level last_error and
1671
+ // the nested queue.last_error). The control plane curates these today; this
1672
+ // is the boundary backstop, and it touches no other field.
1673
+ //
1674
+ // Redaction runs OUTSIDE the summariser deliberately: summarizeInferenceStatus
1675
+ // derives status_summary from these same raw fields, so redacting first would
1676
+ // let it summarise text the backstop had already cleaned — and redacting last
1677
+ // covers anything it copied forward.
1678
+ return json(redactFreeTextFields(summarizeInferenceStatus(data), RAW_FAILURE_TEXT_FIELDS));
1679
+ });
1680
+ /* ══════════════════════════════════════════════════════════════════════════ */
1681
+ /* TOOL: get_inference_metrics */
1682
+ /* ══════════════════════════════════════════════════════════════════════════ */
1683
+ server.tool("get_inference_metrics", "Read serving metrics for a deployment: request counts, prompt/generation token totals, cache hit rate, time to first token, inter-token latency, throughput, GPU utilization and memory, KV cache usage, and a recent time series. Counters are lifetime totals; the time series reflects the selected window. Responses are served from a short cache; zeros are real measurements, not missing data.", {
1684
+ deployment_id: z.string().describe("Deployment ID."),
1685
+ window: z
1686
+ .enum(["1h", "24h", "7d", "30d"])
1687
+ .optional()
1688
+ .describe("Time-series window (default 24h). Sets the series range and bucket size: 1h/1min, 24h/5min, 7d/30min, 30d/1h."),
1689
+ }, async ({ deployment_id, window }) => {
1690
+ validateId(deployment_id, "deployment_id");
1691
+ const params = new URLSearchParams();
1692
+ if (window)
1693
+ params.set("window", window);
1694
+ const query = params.toString();
1695
+ const data = await client.api(`/api/inference/${deployment_id}/metrics${query ? `?${query}` : ""}`);
1696
+ return json(data);
1697
+ });
1698
+ /* ══════════════════════════════════════════════════════════════════════════ */
1699
+ /* TOOL: get_inference_notifications */
1700
+ /* ══════════════════════════════════════════════════════════════════════════ */
1701
+ server.tool("get_inference_notifications", "Read the durable lifecycle-email history for a deployment. States are pending, sending, delivered, or dead_letter. Dead letters retain the bounded retry count and last error; deployment status remains authoritative even when email delivery fails.", {
1702
+ deployment_id: z.string().describe("Deployment ID."),
1703
+ limit: z.number().int().min(1).max(100).optional().describe("Newest rows to return (default 50, maximum 100)."),
1704
+ }, async ({ deployment_id, limit }) => {
1705
+ validateId(deployment_id, "deployment_id");
1706
+ const params = new URLSearchParams();
1707
+ if (limit !== undefined)
1708
+ params.set("limit", String(limit));
1709
+ const query = params.toString();
1710
+ const data = await client.api(`/api/inference/${deployment_id}/notifications${query ? `?${query}` : ""}`);
1711
+ return json(data);
1712
+ });
1713
+ /* ══════════════════════════════════════════════════════════════════════════ */
1714
+ /* TOOL: update_inference_policy */
1715
+ /* ══════════════════════════════════════════════════════════════════════════ */
1716
+ server.tool("update_inference_policy", "Update explicit capacity-queue consent or the maximum accepted total hourly GPU price. A cap below the current verified price is rejected. This call does not silently choose another GPU.", {
1717
+ deployment_id: z.string().describe("Deployment ID."),
1718
+ allow_capacity_queue: z.boolean().optional(),
1719
+ max_price_hour_cents: priceCapSchema(false),
1720
+ }, async ({ deployment_id, allow_capacity_queue, max_price_hour_cents }) => {
1721
+ validateId(deployment_id, "deployment_id");
1722
+ const data = await client.api(`/api/inference/${deployment_id}`, {
1723
+ method: "PATCH",
1724
+ body: buildDeploymentPolicyBody({ allow_capacity_queue, max_price_hour_cents }),
1725
+ });
1726
+ return json(data);
1727
+ });
1728
+ /* ══════════════════════════════════════════════════════════════════════════ */
1729
+ /* TOOL: stop_inference */
1730
+ /* ══════════════════════════════════════════════════════════════════════════ */
1731
+ server.tool("stop_inference", "Stop a deployment and release GPU capacity plus its wallet authorization. The acknowledgement means the durable stop workflow was accepted; poll get_inference_status until stopped or failed.", { deployment_id: z.string().describe("Deployment ID.") }, async ({ deployment_id }) => {
1732
+ validateId(deployment_id, "deployment_id");
1733
+ const data = await client.api(`/api/inference/${deployment_id}/stop`, { method: "POST" });
1734
+ return json(data);
1735
+ });
1736
+ /* ══════════════════════════════════════════════════════════════════════════ */
1737
+ /* TOOL: resume_inference */
1738
+ /* ══════════════════════════════════════════════════════════════════════════ */
1739
+ server.tool("resume_inference", "Atomically resume a stopped deployment. Exactly one concurrent caller can claim the lifecycle generation. The control plane revalidates the saved GPU price cap and wallet authorization before provisioning; capacity may queue only when the saved queue consent allows it.", { deployment_id: z.string().describe("Deployment ID.") }, async ({ deployment_id }) => {
1740
+ validateId(deployment_id, "deployment_id");
1741
+ const data = await client.api(`/api/inference/${deployment_id}/resume`, { method: "POST" });
1742
+ return json(data);
1743
+ });
1744
+ /* ══════════════════════════════════════════════════════════════════════════ */
1745
+ /* TOOL: restart_inference */
1746
+ /* ══════════════════════════════════════════════════════════════════════════ */
1747
+ server.tool("restart_inference", "Claim one fenced restart generation. A same-config running deployment restarts in place and never silently releases its GPU for a replacement when the platform cannot confirm that restart. Pending configuration changes and failed/crash-loop recovery explicitly provision a replacement. Poll status because the acknowledgement is not endpoint readiness.", { deployment_id: z.string().describe("Deployment ID.") }, async ({ deployment_id }) => {
1748
+ validateId(deployment_id, "deployment_id");
1749
+ const data = await client.api(`/api/inference/${deployment_id}/restart`, { method: "POST" });
1750
+ return json(data);
1751
+ });
1752
+ /* ══════════════════════════════════════════════════════════════════════════ */
1753
+ /* TOOL: delete_inference */
1754
+ /* ══════════════════════════════════════════════════════════════════════════ */
1755
+ server.tool("delete_inference", "Permanently delete a serving deployment. The control plane first verifies infrastructure teardown and records retryable orphan cleanup if needed, then releases/refunds the generation-scoped wallet authorization. This cannot be undone.", { deployment_id: z.string().describe("Deployment ID.") }, async ({ deployment_id }) => {
1756
+ validateId(deployment_id, "deployment_id");
1757
+ const data = await client.api(`/api/inference/${deployment_id}`, { method: "DELETE" });
1758
+ return json(data);
1759
+ });
1760
+ /* ══════════════════════════════════════════════════════════════════════════ */
1761
+ /* TOOL: list_datasets */
1762
+ /* ══════════════════════════════════════════════════════════════════════════ */
1763
+ server.tool("list_datasets", "List all datasets in your workspace. Returns dataset ID, name, format (JSONL/Parquet/CSV), row count, file size, column names, and creation date for each dataset.", {}, async () => {
1764
+ const data = await client.api("/api/datasets");
1765
+ return json(data);
1766
+ });
1767
+ /* ══════════════════════════════════════════════════════════════════════════ */
1768
+ /* TOOL: upload_dataset */
1769
+ /* ══════════════════════════════════════════════════════════════════════════ */
1770
+ // PRE-LAUNCH GATE (launchGates.datasets — src/config.ts): while datasets are
1771
+ // coming soon this tool is NOT REGISTERED — hidden from tools/list so an agent
1772
+ // never learns the capability exists. The service enforces the same gate
1773
+ // server-side (403 DATASETS_COMING_SOON) as the authority. At launch the flag
1774
+ // opens and the registration returns verbatim.
1775
+ if (!launchGates.datasets) {
1776
+ server.tool("upload_dataset", "Upload a dataset file for fine-tuning. Supports JSONL, Parquet, and CSV formats up to 500MB. The file is automatically validated: format detection, row counting, column identification, and schema validation. Returns the dataset ID needed for create_training_job.", {
1777
+ file_path: z.string().describe("Absolute path to the dataset file on disk"),
1778
+ name: z.string().optional().describe("Display name for the dataset (defaults to filename)"),
1779
+ }, async ({ file_path, name }) => {
1780
+ const absPath = resolve(file_path);
1781
+ const allowedExtensions = [".jsonl"];
1782
+ const ext = absPath.toLowerCase().slice(absPath.lastIndexOf("."));
1783
+ if (!allowedExtensions.includes(ext)) {
1784
+ throw new Error(`Unsupported file type: ${ext}. Allowed: ${allowedExtensions.join(", ")}`);
1785
+ }
1786
+ const normalizedPath = absPath.replace(/\\/g, "/");
1787
+ const sensitivePatterns = ["/etc/", "/.ssh/", "/.env", "/.git/", "/passwd", "/shadow", "/.aws/"];
1788
+ for (const pat of sensitivePatterns) {
1789
+ if (normalizedPath.includes(pat)) {
1790
+ throw new Error(`Access denied: cannot read files from restricted paths`);
1791
+ }
1792
+ }
1793
+ let fileStat;
1794
+ try {
1795
+ fileStat = await stat(absPath);
1796
+ }
1797
+ catch {
1798
+ throw new Error(`File not found: ${absPath}`);
1799
+ }
1800
+ if (fileStat.size > 500 * 1024 * 1024) {
1801
+ throw new Error(`File too large (${(fileStat.size / 1024 / 1024).toFixed(1)}MB). Maximum is 500MB.`);
1802
+ }
1803
+ const fileBuffer = await readFile(absPath);
1804
+ const fileName = name ?? basename(absPath);
1805
+ const blob = new Blob([fileBuffer]);
1806
+ const formData = new FormData();
1807
+ formData.append("file", blob, basename(absPath));
1808
+ formData.append("name", fileName);
1809
+ formData.append("size", fileStat.size.toString());
1810
+ const data = await client.api("/api/datasets/upload", {
1811
+ method: "POST",
1812
+ formData,
1813
+ });
1814
+ return json(data);
1815
+ });
1816
+ }
1817
+ /* ══════════════════════════════════════════════════════════════════════════ */
1818
+ /* TOOL: preview_dataset */
1819
+ /* ══════════════════════════════════════════════════════════════════════════ */
1820
+ server.tool("preview_dataset", "Preview the first rows of a dataset to inspect its structure, column names, and content before training. Useful for verifying the dataset format is correct.", {
1821
+ dataset_id: z.string().describe("Dataset ID (e.g., 'ds_abc123')"),
1822
+ rows: z.number().optional().describe("Number of rows to preview (default: 10, max: 50)"),
1823
+ }, async ({ dataset_id, rows }) => {
1824
+ validateId(dataset_id, "dataset_id");
1825
+ const data = await client.api(`/api/datasets/${dataset_id}/preview`, {
1826
+ params: { page: "1", page_size: (rows ?? 10).toString() },
1827
+ });
1828
+ return json(data);
1829
+ });
1830
+ /* ══════════════════════════════════════════════════════════════════════════ */
1831
+ /* TOOL: delete_dataset */
1832
+ /* ══════════════════════════════════════════════════════════════════════════ */
1833
+ server.tool("delete_dataset", "Permanently delete a dataset from your workspace. This cannot be undone. Active training jobs using this dataset are not affected.", {
1834
+ dataset_id: z.string().describe("Dataset ID to delete"),
1835
+ }, async ({ dataset_id }) => {
1836
+ validateId(dataset_id, "dataset_id");
1837
+ await client.api(`/api/datasets/${dataset_id}`, { method: "DELETE" });
1838
+ return result(`Dataset ${dataset_id} deleted successfully.`);
1839
+ });
1840
+ /* ══════════════════════════════════════════════════════════════════════════ */
1841
+ /* TOOL: list_integrations */
1842
+ /* ══════════════════════════════════════════════════════════════════════════ */
1843
+ server.tool("list_integrations", "List the integrations connected to your workspace, such as Hugging Face accounts. Returns each integration's ID, label, provider type, and status. Use an integration ID wherever a tool accepts integration_id: importing HuggingFace datasets. This is how you pick WHICH connected account an import should use when several are connected. This read is workspace-scoped: without workspace context it fails with WORKSPACE_CONTEXT_REQUIRED, which is a configuration fix (BIOS_WORKSPACE_ID or a workspace-bound API key), not something to retry.", {}, async () => {
1844
+ const data = await client.api("/api/datasets/integrations");
1845
+ return json(data);
1846
+ });
1847
+ /* ══════════════════════════════════════════════════════════════════════════ */
1848
+ /* TOOL: import_huggingface_dataset */
1849
+ /* ══════════════════════════════════════════════════════════════════════════ */
1850
+ // PRE-LAUNCH GATE (launchGates.datasets — src/config.ts): not registered while
1851
+ // gated, exactly like upload_dataset above; the service's 403
1852
+ // DATASETS_COMING_SOON stays the authority.
1853
+ if (!launchGates.datasets) {
1854
+ server.tool("import_huggingface_dataset", "Import a dataset from HuggingFace Hub into your workspace. Works with public datasets and private datasets if your HuggingFace account is connected via Integrations. The dataset is downloaded, validated, and stored in your workspace. Returns the dataset ID for use with create_training_job.", {
1855
+ repo_id: z
1856
+ .string()
1857
+ .describe("HuggingFace dataset repo whose rows already carry a messages array (SFT) or a text field (CPT)"),
1858
+ integration_id: z
1859
+ .string()
1860
+ .describe("HuggingFace integration ID from your connected integrations"),
1861
+ split: z.string().optional().describe("Dataset split (default: 'train')"),
1862
+ max_samples: z.number().optional().describe("Max rows to import (imports all if omitted)"),
1863
+ name: z.string().optional().describe("Display name for the imported dataset"),
1864
+ }, async ({ repo_id, integration_id, split, max_samples, name }) => {
1865
+ const data = await client.api(`/api/datasets/integrations/${encodeURIComponent(integration_id)}/import`, {
1866
+ method: "POST",
1867
+ body: {
1868
+ dataset_id: repo_id,
1869
+ split: split ?? "train",
1870
+ max_samples,
1871
+ name: name ?? repo_id.split("/").pop(),
1872
+ },
1873
+ });
1874
+ return json(data);
1875
+ });
1876
+ }
1877
+ /* ══════════════════════════════════════════════════════════════════════════ */
1878
+ /* TOOL: get_training_capabilities */
1879
+ /* ══════════════════════════════════════════════════════════════════════════ */
1880
+ server.tool("get_training_capabilities", "Return the authoritative training contract Run BiOS enforces today: enabled and disabled methods, algorithms and adapters; exact hyperparameter names, aliases, types, defaults, enums and ranges; dependencies and incompatibilities. contract_version/schema_version identify the contract you are reading. Call this before constructing an automated training payload instead of relying on a hard-coded client list.", {}, async () => {
1881
+ const data = await client.api("/api/training/capabilities");
1882
+ // The contract used to carry image_compatibility — the training-engine image
1883
+ // VERSION range it was written against. training-service no longer emits it;
1884
+ // this strip stays as a backstop because a customer cannot select a build and
1885
+ // must never learn one, and contract_version/schema_version already version
1886
+ // the contract for clients.
1887
+ return json(stripBuildIdentity(data));
1888
+ });
1889
+ /* ══════════════════════════════════════════════════════════════════════════ */
1890
+ /* TOOL: preflight_training_job */
1891
+ /* ══════════════════════════════════════════════════════════════════════════ */
1892
+ server.tool("preflight_training_job", "Validate and canonicalize a proposed training job without creating a job, charging the wallet, entering the queue, or allocating a GPU. Returns dataset compatibility, model facts, authoritative GPU choices/prices, queue eligibility, alternatives, warnings, and a request hash.", {
1893
+ model: z.string().describe("Base model id from the Run BiOS catalog (e.g., 'meta-llama/Llama-3.1-8B-Instruct'). Must be hosted on Run BiOS — pick from list_models/search_models. Training always starts from a catalog base model; training FROM a fine-tuned model (adapter stacking) is not supported."),
1894
+ model_revision: z.string().max(256).optional().describe("Requested model branch, tag, or commit. Copy the exact 40-hex value returned in canonical_request into create."),
1895
+ dataset_id: z.string().optional(),
1896
+ dataset_ids: z.array(z.string()).min(1).optional(),
1897
+ method: z.enum(TRAINING_METHODS),
1898
+ adapter: z.enum(TRAINING_ADAPTERS).optional(),
1899
+ gpu_type: z.string().optional(),
1900
+ gpu_count: z.number().int().min(1).max(8).optional(),
1901
+ gpu_priorities: z
1902
+ .array(z.object({
1903
+ gpu_type: z.string(), gpu_count: z.number().int().min(1).max(8),
1904
+ provider: z.string().min(1).max(30).optional(), region: z.string().min(1).max(64).optional(),
1905
+ tier: z.literal("secure").optional(),
1906
+ }))
1907
+ .min(1)
1908
+ .max(5)
1909
+ .optional()
1910
+ .describe("Ranked GPU placement choices. Copy these unchanged from the preflight response's accepted choices; the platform fills in and manages infrastructure placement automatically. Queueing requires 3-5 distinct entries; entry one must match gpu_type/gpu_count."),
1911
+ queue_if_unavailable: z.boolean().optional().describe("Explicit consent to wait in the capacity queue when none of the ranked GPU choices are free; requires 3 to 5 gpu_priorities."),
1912
+ queue_deadline: z.string().min(1).optional().describe("Optional RFC 3339 deadline from one minute through seven days in the future; requires queue_if_unavailable=true."),
1913
+ max_price_hour_cents: priceCapSchema(true),
1914
+ storage_gb: z.number().int().min(1).max(10000).optional(),
1915
+ num_checkpoints: z.number().int().min(1).max(100).optional(),
1916
+ integration_id: z.string().optional(),
1917
+ network_volume_id: z.string().optional(),
1918
+ cache_dataset: z.boolean().optional(),
1919
+ dataset_mixing: z.enum(["shuffle", "sequential", "interleave"]).optional(),
1920
+ config: z.record(z.string(), z.unknown()).optional(),
1921
+ }, async (params) => {
1922
+ await assertModelHosted(params.model, "preflight_training_job");
1923
+ const body = buildTrainingCreateBody(params);
1924
+ const data = await client.api("/api/training/preflight", { method: "POST", body });
1925
+ return json(data);
1926
+ });
1927
+ /* ══════════════════════════════════════════════════════════════════════════ */
1928
+ /* TOOL: create_training_job */
1929
+ /* ══════════════════════════════════════════════════════════════════════════ */
1930
+ // PRE-LAUNCH GATE (launchGates.training — src/config.ts): while fine-tuning is
1931
+ // coming soon this tool is NOT REGISTERED — hidden from tools/list so an agent
1932
+ // never learns the capability exists. The service enforces the same gate
1933
+ // server-side (403 TRAINING_COMING_SOON before any billing or booking) as the
1934
+ // authority. At launch the flag opens and the registration returns verbatim.
1935
+ if (!launchGates.training) {
1936
+ server.tool("create_training_job", "Create a fine-tuning job using the same contract as the web UI, following BOOK-BEFORE-REVEAL: the call blocks while the ranked GPU ladder is booked (~40s). A job id and the started email exist only once a real pod is secured, so status is 'booked' (booked==secured; training then provisions/downloads/runs on its own) or 'securing' (still booking at the deadline — poll get_training_status until booked/running). With explicit queue consent it may return 'queued'. A definitive booking-time miss returns the structured CAPACITY_UNAVAILABLE (409) with fresh available_gpus and NO job exists (nothing charged); never auto-substitute a GPU. Call preflight_training_job first, preserve the accepted ranked GPU choices, queue deadline, and maximum total hourly price, then reuse one idempotency key after timeouts.", {
1937
+ model: z.string().describe("Base model id from the Run BiOS catalog (e.g., 'meta-llama/Llama-3.1-8B-Instruct'). Must be hosted on Run BiOS — pick from list_models/search_models. Training always starts from a catalog base model; training FROM a fine-tuned model (adapter stacking) is not supported."),
1938
+ model_revision: z.string().max(256).optional().describe("Exact model_revision from preflight canonical_request; a branch or tag is accepted but resolved again before any paid mutation."),
1939
+ idempotency_key: z
1940
+ .string()
1941
+ .min(8)
1942
+ .max(128)
1943
+ .regex(/^[A-Za-z0-9._:-]+$/)
1944
+ .optional()
1945
+ .describe("Stable retry key. Reuse it after a timeout; omit to generate one for this tool call."),
1946
+ dataset_id: z.string().optional().describe("Compatibility field for one dataset ID."),
1947
+ dataset_ids: z.array(z.string()).min(1).optional().describe("One or more dataset IDs to compose for training."),
1948
+ method: z.enum(TRAINING_METHODS).describe("Training method: sft, or pt/cpt for continued pre-training."),
1949
+ adapter: z.enum(TRAINING_ADAPTERS).optional().describe("Any adapter supported by the Run BiOS contract."),
1950
+ gpu_type: z.string().optional().describe("GPU type (e.g., 'A100_80GB'). Use get_recommended_gpu to find the best option. Auto-selected if omitted."),
1951
+ gpu_count: z.number().int().min(1).max(8).optional().describe("Number of GPUs for the primary choice."),
1952
+ gpu_priorities: z
1953
+ .array(z.object({
1954
+ gpu_type: z.string(), gpu_count: z.number().int().min(1).max(8),
1955
+ provider: z.string().min(1).max(30).optional(), region: z.string().min(1).max(64).optional(),
1956
+ tier: z.literal("secure").optional(),
1957
+ }))
1958
+ .min(1)
1959
+ .max(5)
1960
+ .optional()
1961
+ .describe("Ranked GPU placement choices. Copy these unchanged from the preflight response's accepted choices; the platform fills in and manages infrastructure placement automatically. Queueing requires 3-5 distinct entries; entry one must match gpu_type/gpu_count."),
1962
+ queue_if_unavailable: z.boolean().optional().describe("Explicit consent to wait in the capacity queue when none of the ranked GPU choices are free; requires 3 to 5 gpu_priorities."),
1963
+ queue_deadline: z.string().min(1).optional().describe("Optional RFC 3339 deadline from one minute through seven days in the future; requires queue_if_unavailable=true."),
1964
+ max_price_hour_cents: priceCapSchema(true),
1965
+ storage_gb: z.number().int().min(1).max(10000).optional(),
1966
+ num_checkpoints: z.number().int().min(1).max(100).optional(),
1967
+ epochs: z.number().positive().max(1000).optional().describe("Number of training epochs."),
1968
+ learning_rate: z.number().positive().max(1).optional().describe("Learning rate."),
1969
+ batch_size: z.number().int().min(1).max(65536).optional().describe("Per-device train batch size."),
1970
+ gradient_accumulation_steps: z.number().int().min(1).max(65536).optional(),
1971
+ max_seq_length: trainingSeqLengthSchema(),
1972
+ lora_rank: z.number().int().min(1).max(4096).optional(),
1973
+ lora_alpha: z.number().positive().max(1_000_000).optional(),
1974
+ warmup_ratio: z.number().min(0).max(1).optional(),
1975
+ weight_decay: z.number().min(0).max(10).optional(),
1976
+ scheduler: z
1977
+ .enum(["linear", "cosine", "cosine_with_restarts", "polynomial", "constant", "constant_with_warmup"])
1978
+ .optional()
1979
+ .describe("Learning-rate scheduler."),
1980
+ job_name: z.string().max(255).optional(),
1981
+ eval_split_ratio: z.number().min(0).max(0.99).optional(),
1982
+ early_stopping_patience: z.number().int().min(0).max(20).optional()
1983
+ .describe("Stop after N evaluations without a better validation loss (0 = off; default 3 when eval_split_ratio is set). The best checkpoint is always saved."),
1984
+ overlong_policy: z.enum(["truncate", "drop"]).optional()
1985
+ .describe("Samples longer than max_seq_length: 'truncate' keeps the last tokens so the answer is trained (default); 'drop' skips them."),
1986
+ integration_id: z.string().optional(),
1987
+ network_volume_id: z.string().optional(),
1988
+ cache_dataset: z.boolean().optional(),
1989
+ dataset_mixing: z.enum(["shuffle", "sequential", "interleave"]).optional(),
1990
+ config: z.record(z.string(), z.unknown()).optional().describe("Additional Run BiOS training configuration fields."),
1991
+ }, async (params) => {
1992
+ await assertModelHosted(params.model, "create_training_job");
1993
+ const body = buildTrainingCreateBody(params);
1994
+ const data = await client.api("/api/training/jobs", {
1995
+ method: "POST",
1996
+ body,
1997
+ headers: { "Idempotency-Key": params.idempotency_key ?? randomUUID() },
1998
+ });
1999
+ return json(data);
2000
+ });
2001
+ }
2002
+ /* ══════════════════════════════════════════════════════════════════════════ */
2003
+ /* TOOL: list_training_jobs */
2004
+ /* ══════════════════════════════════════════════════════════════════════════ */
2005
+ server.tool("list_training_jobs", "List all training jobs with their status, progress percentage, base model, training method, GPU type, cost so far, and timestamps. Filter by status to find running, completed, or stopped jobs.", {
2006
+ status: z
2007
+ .enum(["running", "completed", "stopped", "queued", "securing", "booked", "interrupted", "failed"])
2008
+ .optional()
2009
+ .describe("Filter by job status"),
2010
+ }, async ({ status }) => {
2011
+ const data = await client.api("/api/training/jobs", { params: { status } });
2012
+ return json(data);
2013
+ });
2014
+ /* ══════════════════════════════════════════════════════════════════════════ */
2015
+ /* TOOL: get_training_status */
2016
+ /* ══════════════════════════════════════════════════════════════════════════ */
2017
+ server.tool("get_training_status", "Get detailed status for a specific training job: progress percentage, current step/total steps, current loss, learning rate, elapsed time, estimated time remaining, GPU utilization, and job configuration. Status 'securing' means a create is still booking a pod (poll until 'booked'); 'booked' means the GPU is secured and the job is starting. 'interrupted' is a self-heal rest state after a started job loses its pod — billing is already stopped and it is distinct from 'failed'; call resume_training_job to continue from the last checkpoint.", {
2018
+ job_id: z.string().describe("Training job ID (e.g., 'job_abc123')"),
2019
+ }, async ({ job_id }) => {
2020
+ validateId(job_id, "job_id");
2021
+ const data = await client.api(`/api/training/jobs/${job_id}`);
2022
+ // A stopped job's stop_last_error is the raw transport error from the last
2023
+ // call to the machine, so it carries the request URL and with it the
2024
+ // supplier's hostname. Redact the raw-text fields only; model ids, config
2025
+ // and every other value pass through untouched.
2026
+ return json(redactFreeTextFields(data, RAW_FAILURE_TEXT_FIELDS));
2027
+ });
2028
+ /* ══════════════════════════════════════════════════════════════════════════ */
2029
+ /* TOOL: get_training_metrics */
2030
+ /* ══════════════════════════════════════════════════════════════════════════ */
2031
+ server.tool("get_training_metrics", "Get training metrics history for a job: training loss over time, eval loss, learning rate schedule, and evaluation results. Use this to check if training is progressing well (loss should decrease) or if the model is overfitting (eval loss increasing while train loss decreases).", {
2032
+ job_id: z.string().describe("Training job ID"),
2033
+ }, async ({ job_id }) => {
2034
+ validateId(job_id, "job_id");
2035
+ const data = await client.api(`/api/training/jobs/${job_id}/metrics`);
2036
+ return json(data);
2037
+ });
2038
+ /* ══════════════════════════════════════════════════════════════════════════ */
2039
+ /* TOOL: get_training_logs */
2040
+ /* ══════════════════════════════════════════════════════════════════════════ */
2041
+ server.tool("get_training_logs", "Get structured recent training log entries (level, message, timestamp). Results are newest first and bounded by tail.", {
2042
+ job_id: z.string().describe("Training job ID"),
2043
+ tail: z
2044
+ .number()
2045
+ .int()
2046
+ .min(1)
2047
+ .max(1000)
2048
+ .optional()
2049
+ .describe("Number of most recent log entries to return (default: 200, max: 1000)"),
2050
+ }, async ({ job_id, tail }) => {
2051
+ validateId(job_id, "job_id");
2052
+ const data = await client.api(`/api/training/jobs/${job_id}/logs`, {
2053
+ params: { tail: tail?.toString() },
2054
+ });
2055
+ // Log lines are written by the training engine, so a boot/pull line can
2056
+ // quote the image reference it started from. Redact by content: the
2057
+ // customer keeps every line, minus any build identity inside it.
2058
+ return json(redactBuildIdentityDeep(trimTrainingLogs(data, tail)));
2059
+ });
2060
+ /* ══════════════════════════════════════════════════════════════════════════ */
2061
+ /* TOOL: stop_training_job */
2062
+ /* ══════════════════════════════════════════════════════════════════════════ */
2063
+ server.tool("stop_training_job", "Request a durable stop for an active training job. The acknowledgement means the stop workflow was accepted; checkpoint preservation and resumability are reported by later job/checkpoint state, not promised by this call.", {
2064
+ job_id: z.string().describe("Training job ID to stop"),
2065
+ keep_data: z.boolean().optional().describe("Attempt to preserve checkpoint/output data (default: true)."),
2066
+ }, async ({ job_id, keep_data }) => {
2067
+ validateId(job_id, "job_id");
2068
+ const data = await client.api(`/api/training/jobs/${job_id}/stop`, {
2069
+ method: "POST",
2070
+ body: { keep_data: keep_data ?? true },
2071
+ });
2072
+ return json(data);
2073
+ });
2074
+ /* ══════════════════════════════════════════════════════════════════════════ */
2075
+ /* TOOL: resume_training_job */
2076
+ /* ══════════════════════════════════════════════════════════════════════════ */
2077
+ server.tool("resume_training_job", "Resume a stopped or 'interrupted' training job from its latest checkpoint when the server has verified it as resumable. 'interrupted' is the self-heal rest state a started job enters after losing its pod (billing already stopped, distinct from 'failed'). Resume re-books on secured capacity before reporting resumed (book-before-reveal still applies). The response identifies the job; exact optimizer/epoch continuation depends on the checkpoint manifest and is not assumed by this tool.", {
2078
+ job_id: z.string().describe("Training job ID to resume"),
2079
+ idempotency_key: z
2080
+ .string()
2081
+ .min(8)
2082
+ .max(128)
2083
+ .regex(/^[A-Za-z0-9._:-]+$/)
2084
+ .optional()
2085
+ .describe("Stable retry key. Reuse it after a timeout; omit to generate one for this tool call."),
2086
+ }, async ({ job_id, idempotency_key }) => {
2087
+ validateId(job_id, "job_id");
2088
+ const data = await client.api(`/api/training/jobs/${job_id}/resume`, {
2089
+ method: "POST",
2090
+ headers: { "Idempotency-Key": idempotency_key ?? randomUUID() },
2091
+ });
2092
+ return json(data);
2093
+ });
2094
+ /* ══════════════════════════════════════════════════════════════════════════ */
2095
+ /* TOOL: get_training_checkpoints */
2096
+ /* ══════════════════════════════════════════════════════════════════════════ */
2097
+ server.tool("get_training_checkpoints", "List recorded checkpoints for a training job. Returns checkpoint identity, step/epoch, size and final/best flags. Use get_checkpoint_download_manifest for file URLs; this list does not fabricate loss, eval metrics, or download URLs.", {
2098
+ job_id: z.string().describe("Training job ID"),
2099
+ }, async ({ job_id }) => {
2100
+ validateId(job_id, "job_id");
2101
+ const data = await client.api(`/api/training/jobs/${job_id}/checkpoints`);
2102
+ return json(data);
2103
+ });
2104
+ /* ══════════════════════════════════════════════════════════════════════════ */
2105
+ /* TOOL: get_checkpoint_download_manifest */
2106
+ /* ══════════════════════════════════════════════════════════════════════════ */
2107
+ server.tool("get_checkpoint_download_manifest", "Per-file download manifest for a checkpoint, available only in workspaces where model weight downloads are enabled; otherwise the checkpoint is reported as not found. URLs are short-lived.", {
2108
+ job_id: z.string().describe("Training job ID"),
2109
+ checkpoint_id: z.string().describe("Checkpoint ID from get_training_checkpoints"),
2110
+ }, async ({ job_id, checkpoint_id }) => {
2111
+ validateId(job_id, "job_id");
2112
+ validateId(checkpoint_id, "checkpoint_id");
2113
+ const data = await client.api(`/api/training/jobs/${job_id}/checkpoints/${checkpoint_id}/download`);
2114
+ return json(data);
2115
+ });
2116
+ /* ══════════════════════════════════════════════════════════════════════════ */
2117
+ /* TOOL: delete_training_checkpoint */
2118
+ /* ══════════════════════════════════════════════════════════════════════════ */
2119
+ server.tool("delete_training_checkpoint", "Delete a checkpoint after the API verifies ownership. This is destructive and can remove the ability to deploy or resume from that checkpoint.", {
2120
+ job_id: z.string().describe("Training job ID"),
2121
+ checkpoint_id: z.string().describe("Checkpoint ID"),
2122
+ }, async ({ job_id, checkpoint_id }) => {
2123
+ validateId(job_id, "job_id");
2124
+ validateId(checkpoint_id, "checkpoint_id");
2125
+ const data = await client.api(`/api/training/jobs/${job_id}/checkpoints/${checkpoint_id}`, {
2126
+ method: "DELETE",
2127
+ });
2128
+ return json(data);
2129
+ });
2130
+ /* ══════════════════════════════════════════════════════════════════════════ */
2131
+ /* TOOL: get_training_evals */
2132
+ /* ══════════════════════════════════════════════════════════════════════════ */
2133
+ server.tool("get_training_evals", "Get evaluation results for a training job. Returns benchmark scores, eval metrics, and model quality assessments computed during or after training.", {
2134
+ job_id: z.string().describe("Training job ID"),
2135
+ }, async ({ job_id }) => {
2136
+ validateId(job_id, "job_id");
2137
+ const data = await client.api(`/api/training/jobs/${job_id}/evals`);
2138
+ return json(data);
2139
+ });
2140
+ /* ══════════════════════════════════════════════════════════════════════════ */
2141
+ /* TOOL: chat_with_inference */
2142
+ /* ══════════════════════════════════════════════════════════════════════════ */
2143
+ // The schema is capability-aware per deployment (honest advertisement only —
2144
+ // the runtime is gated server-side). When the inference key's deployment is
2145
+ // known to be text-only (no tool parser), the tools/tool_choice/
2146
+ // parallel_tool_calls params are omitted so a caller never offers tools a
2147
+ // text-only model can't call. When it is known to accept no image input, the
2148
+ // messages description says so. Unknown capabilities (undefined) advertise the
2149
+ // full schema (fail-open) — the /v1 endpoint still enforces.
2150
+ const inferenceSupportsTools = deploymentCaps?.supportsTools !== false;
2151
+ const inferenceMessagesDesc = deploymentCaps?.supportsImages === false
2152
+ ? "OpenAI chat messages (text-only — this deployment does not accept image input), including assistant tool_calls and matching role=tool responses."
2153
+ : "OpenAI chat messages, including assistant tool_calls and matching role=tool responses.";
2154
+ const chatWithInferenceDescription = "Call a model through the unified OpenAI-compatible /v1 endpoint. Pass a serverless catalog model id (author/name) to call it by name using a workspace platform key (RUNBIOS_API_KEY with serverless scope), or set RUNBIOS_INFERENCE_KEY to target a dedicated deployment (with or without a model override). Supports complete tool definitions, assistant tool calls, and tool response messages. Set stream=true to consume SSE and receive one aggregated completion with reasoning surfaced as produced. Calls are never retried automatically.";
2155
+ // Shared, always-present params. The tools params are added only for a
2156
+ // deployment that can call tools — kept as two concrete literals (not a Record)
2157
+ // so each registration's `input` stays precisely typed for buildInferenceBody.
2158
+ const chatBaseSchema = {
2159
+ messages: z
2160
+ .array(z.record(z.string(), z.unknown()))
2161
+ .min(1)
2162
+ .describe(inferenceMessagesDesc),
2163
+ model: z
2164
+ .string()
2165
+ .optional()
2166
+ .describe("Model to call on the unified /v1 endpoint. For a serverless catalog model pass its id "
2167
+ + "(author/name) and the gateway routes it; for a dedicated deployment pass its name/id, or "
2168
+ + "omit when a dedicated inference key uniquely selects the deployment."),
2169
+ // Completion length had the same defect the context bounds had: `int()` alone
2170
+ // advertised MAX_SAFE_INTEGER, so the schema told an agent that asking for
2171
+ // nine quadrillion tokens is a legal request. The real ceiling is the SERVED
2172
+ // window minus the prompt, which no static schema can express, so the bound
2173
+ // is the same sanity ceiling serving uses and the description names the
2174
+ // authority.
2175
+ max_tokens: z
2176
+ .number()
2177
+ .int({ error: "max_tokens must be a whole number of tokens." })
2178
+ .min(1, { error: "max_tokens must be at least 1 — omit it to let the endpoint choose." })
2179
+ .max(CONTEXT_TOOL_MAX, {
2180
+ error: (issue) => `max_tokens ${String(issue.input)} is above the ${CONTEXT_TOOL_MAX}-token sanity bound of this tool. `
2181
+ + MAX_TOKENS_POLICY_TEXT,
2182
+ })
2183
+ .optional()
2184
+ .describe(MAX_TOKENS_POLICY_TEXT),
2185
+ temperature: z.number().min(0).max(2).optional(),
2186
+ top_p: z.number().min(0).max(1).optional(),
2187
+ response_format: z.record(z.string(), z.unknown()).optional(),
2188
+ reasoning_effort: z
2189
+ .enum(["none", "minimal", "low", "medium", "high", "xhigh", "max"])
2190
+ .optional()
2191
+ .describe("Standardized reasoning effort. Use 'none' to disable reasoning where the model allows it."),
2192
+ stream: z
2193
+ .boolean()
2194
+ .optional()
2195
+ .describe("Request server-sent streaming. The tool consumes the SSE and returns one aggregated "
2196
+ + "completion (content + reasoning_content surfaced as produced, tool calls reassembled). "
2197
+ + "Defaults to false."),
2198
+ idempotency_key: z
2199
+ .string()
2200
+ .min(1)
2201
+ .max(255)
2202
+ .optional()
2203
+ .describe("Stable caller-generated key to reuse only when explicitly retrying the same POST."),
2204
+ request_id: z.string().min(1).max(255).optional(),
2205
+ };
2206
+ if (inferenceSupportsTools) {
2207
+ server.tool("chat_with_inference", chatWithInferenceDescription, {
2208
+ ...chatBaseSchema,
2209
+ tools: z
2210
+ .array(z.record(z.string(), z.unknown()))
2211
+ .optional()
2212
+ .describe("OpenAI function-tool definitions with JSON Schema object parameters."),
2213
+ tool_choice: z
2214
+ .union([
2215
+ z.enum(["none", "auto", "required"]),
2216
+ z.object({
2217
+ type: z.literal("function"),
2218
+ function: z.object({ name: z.string().min(1).max(64) }),
2219
+ }),
2220
+ ])
2221
+ .optional(),
2222
+ parallel_tool_calls: z.boolean().optional(),
2223
+ }, async (input, extra) => {
2224
+ const body = buildInferenceBody(input);
2225
+ const data = await client.inferenceApi(body, input.idempotency_key, input.request_id, extra.signal);
2226
+ return json(data);
2227
+ });
2228
+ }
2229
+ else {
2230
+ // Text-only deployment: no tools/tool_choice params advertised.
2231
+ server.tool("chat_with_inference", chatWithInferenceDescription, chatBaseSchema, async (input, extra) => {
2232
+ const body = buildInferenceBody(input);
2233
+ const data = await client.inferenceApi(body, input.idempotency_key, input.request_id, extra.signal);
2234
+ return json(data);
2235
+ });
2236
+ }
2237
+ return mcp;
2238
+ }
2239
+ //# sourceMappingURL=server.js.map