corent-mcp 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/server.js CHANGED
@@ -1,5 +1,5 @@
1
1
  /**
2
- * Shared Corent MCP server definition — the 10 tools, used by both the
2
+ * Shared Corent MCP server definition — the tools and prompts, used by both the
3
3
  * stdio entrypoint (index.ts, for local `npx` use) and the hosted HTTP
4
4
  * entrypoint (http.ts). Keeping the tools in one place means the two
5
5
  * transports can never drift apart.
@@ -16,8 +16,15 @@
16
16
  import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
17
17
  import { z } from "zod";
18
18
  import { MEDIA_WIDGET_HTML } from "./widget-html.js";
19
+ import { registerPrompts } from "./prompts.js";
19
20
  export const DEFAULT_API_URL = "https://api.corent.tech";
20
- export const SERVER_VERSION = "0.6.0";
21
+ export const SERVER_VERSION = "0.8.0";
22
+ // Transport resilience, matching the two SDKs: a 429 or 5xx is retried with
23
+ // backoff, but ONLY on calls that carry an Idempotency-Key (or are GETs), so a
24
+ // retry can never buy a second generation.
25
+ const MAX_RETRIES = 3;
26
+ const RETRY_BASE_MS = 500;
27
+ const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
21
28
  // --- MCP Apps (SEP-1865): the media tools render an inline widget in hosts
22
29
  // that support it (claude.ai web/desktop, others). Hosts without the extension
23
30
  // ignore the _meta and fall back to plain text results, so this is additive.
@@ -29,6 +36,19 @@ const WIDGET_TOOL_META = {
29
36
  ui: { resourceUri: WIDGET_URI },
30
37
  "ui/resourceUri": WIDGET_URI,
31
38
  };
39
+ /**
40
+ * Per-tool _meta. ChatGPT's Apps SDK reads two plain status strings from a
41
+ * tool's _meta and shows them while the call runs and once it lands
42
+ * ("openai/toolInvocation/invoking" / "invoked"); every other host ignores
43
+ * unknown _meta keys. Media tools also carry the MCP Apps widget pointer.
44
+ */
45
+ function toolMeta(invoking, invoked, widget = false) {
46
+ return {
47
+ "openai/toolInvocation/invoking": invoking,
48
+ "openai/toolInvocation/invoked": invoked,
49
+ ...(widget ? WIDGET_TOOL_META : {}),
50
+ };
51
+ }
32
52
  // Generated media lives on Supabase storage; the logo and any brand assets on
33
53
  // corent.tech. Everything else stays blocked by the host's CSP.
34
54
  const WIDGET_CSP = {
@@ -80,6 +100,16 @@ function errorCode(err) {
80
100
  export function createCorentServer(config = {}) {
81
101
  const apiUrl = config.apiUrl ?? DEFAULT_API_URL;
82
102
  const apiKey = config.apiKey;
103
+ /**
104
+ * A fresh Idempotency-Key for one money-spending call. The MCP used to send
105
+ * none at all, so any retry -- the agent's, the host's, or a user clicking
106
+ * again -- bought a second generation at full price (parity audit
107
+ * 2026-08-30). The API keys replays off (account, key), so the same key
108
+ * returns the ORIGINAL job instead of generating twice.
109
+ */
110
+ function idempotencyKey() {
111
+ return globalThis.crypto?.randomUUID?.() ?? `mcp-${Date.now()}-${Math.random().toString(36).slice(2)}`;
112
+ }
83
113
  async function corent(path, init) {
84
114
  // Key is enforced here (at call time) rather than at startup, so the server
85
115
  // can advertise its tools to catalogs/agents without a key configured.
@@ -88,19 +118,40 @@ export function createCorentServer(config = {}) {
88
118
  message: "No Corent credentials. On claude.ai or other remote MCP clients, reconnect the Corent connector and approve access (OAuth). For local/stdio use, set CORENT_API_KEY in your MCP client config. Keys: https://corent.tech/dashboard/api-keys",
89
119
  });
90
120
  }
91
- const res = await fetch(`${apiUrl}${path}`, {
92
- ...init,
93
- headers: {
94
- Authorization: `Bearer ${apiKey}`,
95
- "Content-Type": "application/json",
96
- ...init?.headers,
97
- },
98
- });
99
- const body = await res.json().catch(() => ({}));
100
- if (!res.ok) {
101
- throw new CorentApiError(res.status, body.detail ?? body);
121
+ const headers = {
122
+ Authorization: `Bearer ${apiKey}`,
123
+ "Content-Type": "application/json",
124
+ ...(init?.headers ?? {}),
125
+ };
126
+ // Retrying is only safe when the call cannot be charged twice: a GET, or a
127
+ // POST carrying an Idempotency-Key (the API replays the original job for a
128
+ // repeated key). A POST without one is sent exactly once, forever.
129
+ const method = (init?.method ?? "GET").toUpperCase();
130
+ const replaySafe = method === "GET" || Boolean(headers["Idempotency-Key"]);
131
+ const maxAttempts = replaySafe ? MAX_RETRIES : 1;
132
+ let lastErr;
133
+ for (let attempt = 0; attempt < maxAttempts; attempt++) {
134
+ if (attempt > 0)
135
+ await sleep(RETRY_BASE_MS * 2 ** (attempt - 1));
136
+ let res;
137
+ try {
138
+ res = await fetch(`${apiUrl}${path}`, { ...init, headers });
139
+ }
140
+ catch (err) {
141
+ // Transport failure: on a replay-safe call the same key returns the
142
+ // original job, so retrying cannot double-charge.
143
+ lastErr = err;
144
+ continue;
145
+ }
146
+ const body = await res.json().catch(() => ({}));
147
+ if (res.ok)
148
+ return body;
149
+ const retriable = res.status === 429 || (res.status >= 500 && res.status !== 501);
150
+ lastErr = new CorentApiError(res.status, body.detail ?? body);
151
+ if (!retriable)
152
+ throw lastErr;
102
153
  }
103
- return body;
154
+ throw lastErr;
104
155
  }
105
156
  /** Success: typed structuredContent plus a text fallback for older clients. */
106
157
  function ok(data) {
@@ -159,7 +210,13 @@ export function createCorentServer(config = {}) {
159
210
  // yet) -- must be nullable, not just optional, or the SDK's output validator
160
211
  // throws on the normal poll-loop response.
161
212
  const mediaMeta = z
162
- .object({ model: z.string().nullable().optional(), cost_cents: z.number().nullable().optional() })
213
+ .object({
214
+ model: z.string().nullable().optional(),
215
+ cost_cents: z.number().nullable().optional(),
216
+ seed: z.number().nullable().optional(),
217
+ unsupported_options: z.array(z.string()).nullable().optional(),
218
+ has_audio: z.boolean().nullable().optional(),
219
+ })
163
220
  .passthrough();
164
221
  const imageOutput = {
165
222
  id: z.string().optional(),
@@ -167,50 +224,128 @@ export function createCorentServer(config = {}) {
167
224
  images: z.array(z.object({ url: z.string() }).passthrough()).optional(),
168
225
  meta: mediaMeta.optional(),
169
226
  };
227
+ // meta carries two honesty fields worth surfacing to an agent: the seed
228
+ // that was actually used (so a render can be repeated) and the settings the
229
+ // chosen model could NOT honour (so an ignored knob never reads as applied).
170
230
  const jobOutput = {
171
231
  id: z.string().optional(),
172
232
  status: z.string().optional(),
173
233
  images: z.array(z.object({ url: z.string() }).passthrough()).optional(),
174
234
  videos: z.array(z.object({ url: z.string() }).passthrough()).optional(),
235
+ // Speech jobs answer with `audio` (routers/jobs.py). It was missing from
236
+ // this schema, so a polled speech job validated as a result with no
237
+ // deliverable in it (parity audit 2026-08-30).
238
+ audio: z.array(z.object({ url: z.string() }).passthrough()).optional(),
239
+ progress_percent: z.number().nullable().optional(),
175
240
  meta: mediaMeta.optional(),
176
241
  error: z.string().optional(),
177
242
  };
243
+ const batchOutput = {
244
+ batch_id: z.string().optional(),
245
+ status: z.string().optional(),
246
+ jobs: z.array(z.object({ id: z.string().optional() }).passthrough()).optional(),
247
+ counts: z.object({}).passthrough().optional(),
248
+ error: z.string().optional(),
249
+ };
250
+ // Shared with generate_image and the image batch items, so the two can never
251
+ // drift apart on which shapes and styles are reachable.
252
+ const IMAGE_ASPECTS = ["1:1", "16:9", "9:16", "4:3", "3:4", "3:2", "2:3"];
253
+ const IMAGE_STYLES = ["photorealistic", "artistic", "anime", "logo", "text_focused", "cinematic"];
254
+ const TIERS = ["air", "lite", "premium", "pro", "max_pro"];
178
255
  server.registerTool("generate_image", {
179
256
  title: "Generate image",
180
- _meta: WIDGET_TOOL_META,
181
- description: "Generate an image from a text prompt. Returns a permanent public URL. Synchronous: typically completes in 2-20 seconds. Costs a few cents, billed to the Corent account. Failed generations are never billed.",
257
+ _meta: toolMeta("Generating image", "Image ready", true),
258
+ description: "Generate an image from a text prompt. Returns a permanent public URL. Synchronous, usually 2-20 seconds. Costs a few cents (premium ~7c); failed generations are never billed. Takes reference_image_urls to keep a face, character or product consistent (tier premium or up).",
182
259
  inputSchema: {
183
260
  prompt: z.string().describe("What to generate, in plain language"),
184
261
  tier: z
185
- .enum(["air", "lite", "premium", "pro", "max_pro"])
262
+ .enum(TIERS)
186
263
  .optional()
187
264
  .describe("Quality tier: air=cheapest (best value, not the quickest), lite=everyday, premium=production, pro=advanced, max_pro=maximum fidelity (flagship models). Omit for a sensible default."),
188
265
  model: z
189
266
  .string()
190
267
  .optional()
191
- .describe("Direct model access: pin an exact model by name (list_models shows the menu). Bypasses Corent's routing -- never substituted, flat cost-plus price. Mutually exclusive with tier; prefer tier unless the user named a specific model."),
268
+ .describe('Direct model access: pin an exact model by name, e.g. "corent-flux-schnell" (list_models shows the menu; every name there is spelled corent-*). Bypasses Corent\'s routing -- never substituted, flat cost-plus price. Mutually exclusive with tier; prefer tier unless the user named a specific model.'),
192
269
  style: z
193
- .enum(["photorealistic", "artistic", "anime", "logo", "text_focused"])
270
+ .enum(IMAGE_STYLES)
194
271
  .optional()
195
272
  .describe("Optional content style hint"),
196
- aspect_ratio: z.enum(["1:1", "16:9", "9:16", "4:3", "3:4"]).default("1:1"),
273
+ aspect_ratio: z.enum(IMAGE_ASPECTS).default("1:1"),
274
+ n: z
275
+ .number()
276
+ .int()
277
+ .min(1)
278
+ .max(10)
279
+ .optional()
280
+ .describe("How many versions to render, 1 to 10. Each is a REAL render at the normal price, so n=4 " +
281
+ "costs four images; confirm with the user before spending on more than a couple. They come " +
282
+ "back together and meta.cost_cents is the total."),
283
+ seed: z
284
+ .number()
285
+ .int()
286
+ .min(0)
287
+ .optional()
288
+ .describe("Reproducibility. The same seed with the same prompt and model gives the same image, so use " +
289
+ "it to re-render a picture the user liked and change one thing. The seed that was used " +
290
+ "comes back in meta.seed even when you did not pass one."),
291
+ negative_prompt: z.string().optional().describe('What must NOT appear, e.g. "text, watermark".'),
292
+ source_image_url: z
293
+ .string()
294
+ .url()
295
+ .optional()
296
+ .describe("Image-to-image: start from this picture rather than from scratch. Pair with `strength`. " +
297
+ "Different from reference_image_urls, which pins an identity across scenes."),
298
+ strength: z
299
+ .number()
300
+ .min(0)
301
+ .max(1)
302
+ .optional()
303
+ .describe("How far to travel from source_image_url. 0 keeps it, 1 ignores it."),
304
+ output_format: z.enum(["png", "jpeg", "webp"]).optional().describe("Delivered file type."),
305
+ transparent: z
306
+ .boolean()
307
+ .optional()
308
+ .describe("Transparent background, for logos and product cutouts. Forces png."),
309
+ width: z.number().int().min(256).max(4096).optional().describe("Explicit pixel width; pass with height."),
310
+ height: z.number().int().min(256).max(4096).optional().describe("Explicit pixel height; pass with width."),
311
+ enhance_prompt: z
312
+ .boolean()
313
+ .optional()
314
+ .describe("Corent rewrites the prompt before dispatch to get a better render. Pass false when the user " +
315
+ "has carefully worded their own prompt and wants it sent verbatim."),
316
+ reference_image_urls: z
317
+ .array(z.string().url())
318
+ .min(1)
319
+ .max(4)
320
+ .optional()
321
+ .describe("1 to 4 public https image URLs that keep a CHARACTER, FACE, or PRODUCT consistent across " +
322
+ "generations. The prompt is applied as an EDIT of these references, so use it whenever the " +
323
+ "user wants the same person/object again in a new scene, a product placed somewhere, or a " +
324
+ "series that has to match. Served only by edit-capable models, which sit at premium and up: " +
325
+ "pass tier premium/pro/max_pro (or omit tier), never air or lite. Omit entirely for plain " +
326
+ "text-to-image."),
197
327
  },
198
328
  outputSchema: imageOutput,
199
329
  annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: true },
200
- }, wrap(async ({ prompt, tier, model, style, aspect_ratio }) => corent("/v1/images/generate", {
330
+ }, wrap(async (args) => corent("/v1/images/generate", {
201
331
  method: "POST",
202
- body: JSON.stringify({ prompt, tier, model, style, aspect_ratio }),
332
+ headers: { "Idempotency-Key": idempotencyKey() },
333
+ body: JSON.stringify(args),
203
334
  })));
204
335
  server.registerTool("generate_video", {
205
336
  title: "Generate video",
206
- _meta: WIDGET_TOOL_META,
207
- description: "Start generating a video from a text prompt (optionally animating a source image). Asynchronous: returns a job id immediately; poll get_job until status is 'completed'. Video costs more than images (tens of cents to a few dollars depending on tier). Failed generations are never billed.",
337
+ _meta: toolMeta("Starting video", "Video job started", true),
338
+ description: "Start a video from a text prompt, or animate a source image. Asynchronous: returns a job id at once; poll get_job until status is completed. Costs tens of cents to a few dollars per clip (premium ~52c, pro ~84c). Failed generations are never billed.",
208
339
  inputSchema: {
209
340
  prompt: z.string().describe("What to generate, in plain language"),
210
341
  tier: z
211
- .enum(["air", "lite", "premium", "pro", "max_pro"])
342
+ .enum(TIERS)
212
343
  .optional()
213
344
  .describe("Quality tier: air=cheapest (best value, not the quickest), lite=everyday, premium=production, pro=advanced, max_pro=maximum fidelity (flagship models). Omit for a sensible default."),
345
+ style: z
346
+ .enum(IMAGE_STYLES)
347
+ .optional()
348
+ .describe("Optional content style hint, same vocabulary as generate_image"),
214
349
  aspect_ratio: z
215
350
  .enum(["16:9", "9:16", "1:1"])
216
351
  .optional()
@@ -221,20 +356,46 @@ export function createCorentServer(config = {}) {
221
356
  .optional()
222
357
  .describe("Pixel resolution (default 720p). Tier-capped: pro allows up to 1080p, max_pro up to 4k; above the tier's cap it is clamped down, never rejected. Higher resolutions cost more (1080p ~2.5x, 4k ~5.5x). GET /v1/tiers lists each tier's menu with prices."),
223
358
  image_url: z.string().url().optional().describe("If set, animates this image instead of pure text-to-video"),
359
+ audio: z
360
+ .boolean()
361
+ .optional()
362
+ .describe("Ask for native SOUND. Some models render audio and some are silent. true routes only to " +
363
+ "models that actually deliver sound, so a silent model can never quietly serve the request; " +
364
+ "false prefers a silent one; omit to let Corent choose. list_tiers reports supports_audio " +
365
+ "per tier. Use true whenever the user asks for a clip with sound, music, or speech."),
366
+ end_image_url: z
367
+ .string()
368
+ .url()
369
+ .optional()
370
+ .describe("The frame the clip should END on. Together with image_url this is the go-from-A-to-B / morph " +
371
+ "effect; alone it is a target to move toward."),
372
+ camera: z
373
+ .enum([
374
+ "static", "pan_left", "pan_right", "zoom_in", "zoom_out",
375
+ "orbit_left", "orbit_right", "tilt_up", "tilt_down",
376
+ ])
377
+ .optional()
378
+ .describe("Camera move."),
379
+ negative_prompt: z.string().optional().describe("What must NOT appear in the clip."),
380
+ seed: z.number().int().min(0).optional().describe("Reproducibility, same contract as generate_image."),
381
+ fps: z.number().int().min(8).max(60).optional().describe("Frames per second, where the model offers a choice."),
382
+ enhance_prompt: z.boolean().optional().describe("false sends the prompt exactly as written."),
224
383
  model: z
225
384
  .string()
226
385
  .optional()
227
- .describe("Direct model access: pin an exact model by name (list_models shows the menu). Bypasses routing -- never substituted; duration and resolution snap to THAT model's own menu rather than a tier cap. Mutually exclusive with tier."),
386
+ .describe('Direct model access: pin an exact model by name, e.g. "corent-seedance-2.0" (list_models shows the menu; every name there is spelled corent-*). Bypasses routing -- never substituted; duration and resolution snap to THAT model\'s own menu rather than a tier cap. Mutually exclusive with tier.'),
228
387
  },
229
388
  outputSchema: jobOutput,
230
389
  annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: true },
231
- }, wrap(async ({ prompt, tier, model, aspect_ratio, duration_s, resolution, image_url }) => corent("/v1/videos/generate", {
390
+ }, wrap(async (args) => corent("/v1/videos/generate", {
232
391
  method: "POST",
233
- body: JSON.stringify({ prompt, tier, model, aspect_ratio, duration_s, resolution, image_url }),
392
+ headers: { "Idempotency-Key": idempotencyKey() },
393
+ body: JSON.stringify(args),
234
394
  })));
235
395
  server.registerTool("generate_speech", {
236
396
  title: "Generate speech (voice)",
237
- description: "Generate spoken audio (text to speech) from text. Synchronous: returns the finished audio URL and the exact charge. Billed in started blocks of 1,000 input characters, so even a one-sentence request bills the full first block (roughly 25-30 cents); longer scripts amortize better. Failed generations are never billed.",
397
+ _meta: toolMeta("Generating speech", "Speech ready"),
398
+ description: "Text to speech. Synchronous: returns the finished audio URL and the exact charge. Billed in started blocks of 1,000 characters (roughly 25-30c for the first block), so a short line costs the same as a paragraph. Call list_voices first to pick a voice_id. Failed generations are never billed.",
238
399
  inputSchema: {
239
400
  text: z.string().min(1).max(5000).describe("The text to speak, max 5000 characters"),
240
401
  voice_id: z
@@ -245,7 +406,18 @@ export function createCorentServer(config = {}) {
245
406
  model: z
246
407
  .string()
247
408
  .optional()
248
- .describe("Direct model access: pin an exact speech model by name (list_models shows the menu)."),
409
+ .describe('Direct model access: pin an exact speech model by name, e.g. "corent-eleven-multilingual-v2" (list_models shows the menu with kind="speech"; every name there is spelled corent-*).'),
410
+ stability: z
411
+ .number()
412
+ .min(0)
413
+ .max(1)
414
+ .optional()
415
+ .describe("0 to 1. Low is more expressive and varies more take to take; high is steadier."),
416
+ similarity: z.number().min(0).max(1).optional().describe("0 to 1. How closely to hold the voice's character."),
417
+ style: z.number().min(0).max(1).optional().describe("0 to 1. Extra expressiveness."),
418
+ speed: z.number().min(0.5).max(2).optional().describe("Playback speed; 1.0 is natural pace."),
419
+ language: z.string().optional().describe('ISO code for the multilingual models, e.g. "en" or "pt-BR".'),
420
+ output_format: z.string().optional().describe('Audio format, e.g. "mp3" or "wav".'),
249
421
  },
250
422
  outputSchema: {
251
423
  id: z.string().optional(),
@@ -255,24 +427,42 @@ export function createCorentServer(config = {}) {
255
427
  error: z.string().optional(),
256
428
  },
257
429
  annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: true },
258
- }, wrap(async ({ text, voice_id, model }) => corent("/v1/audio/speech", {
430
+ }, wrap(async (args) => corent("/v1/audio/speech", {
259
431
  method: "POST",
260
- body: JSON.stringify({ text, voice_id, model }),
432
+ headers: { "Idempotency-Key": idempotencyKey() },
433
+ body: JSON.stringify(args),
261
434
  })));
262
435
  server.registerTool("generate_text", {
263
436
  title: "Generate text (language model)",
264
- description: "Run a prompt through a frontier language model on the user's Corent account. Synchronous: returns the finished text, the token usage and the exact charge. Billed per token (a short answer is a fraction of a cent). Use this when the user wants a SPECIFIC model's answer, a second opinion from another lab, or work billed to their Corent balance -- not for your own reasoning, which costs them nothing. Failed generations are never billed.",
437
+ _meta: toolMeta("Asking the model", "Answer ready"),
438
+ description: "Run a prompt or conversation through a frontier language model on the user's Corent account. Synchronous: returns the text, token usage and exact charge (fractions of a cent). Use it when the user wants a specific model's answer or work billed to Corent, not for your own reasoning. Supports tool calling and JSON mode.",
265
439
  inputSchema: {
266
- prompt: z.string().min(1).describe("What to ask, in plain language"),
440
+ prompt: z.string().min(1).optional().describe("What to ask, in plain language. Use this for a single question; use `messages` instead to continue a conversation."),
267
441
  system: z.string().optional().describe("Optional system instruction that frames the request"),
442
+ messages: z
443
+ .array(z
444
+ .object({
445
+ role: z.enum(["system", "user", "assistant", "tool"]),
446
+ content: z.unknown().optional(),
447
+ name: z.string().optional(),
448
+ tool_call_id: z.string().optional(),
449
+ tool_calls: z.array(z.object({}).passthrough()).optional(),
450
+ })
451
+ .passthrough())
452
+ .min(1)
453
+ .max(200)
454
+ .optional()
455
+ .describe("Full OpenAI-shaped conversation, for multi-turn work: prior assistant turns, and the tool " +
456
+ "results you are feeding back after a tool call. Takes precedence over prompt/system. " +
457
+ "Without this the model sees only one question and has no memory of the exchange."),
268
458
  tier: z
269
- .enum(["air", "lite", "premium", "pro", "max_pro"])
459
+ .enum(TIERS)
270
460
  .optional()
271
461
  .describe("Quality tier: air=cheapest, lite=everyday, premium=production, pro=advanced, max_pro=frontier flagships. Omit for a sensible default."),
272
462
  model: z
273
463
  .string()
274
464
  .optional()
275
- .describe("Direct model access: pin an exact text model by name (list_models shows the menu, kind='text'). Bypasses routing -- never substituted, flat cost-plus price. Mutually exclusive with tier."),
465
+ .describe('Direct model access: pin an exact text model by name, e.g. "corent-claude-opus-5" (list_models shows the menu with kind="text"; every name there is spelled corent-*). Bypasses routing -- never substituted, flat cost-plus price. Mutually exclusive with tier.'),
276
466
  max_tokens: z
277
467
  .number()
278
468
  .int()
@@ -281,9 +471,25 @@ export function createCorentServer(config = {}) {
281
471
  .optional()
282
472
  .describe("Cap on the reply length in tokens (default 1024). The balance gate reserves against this, so keep it realistic."),
283
473
  temperature: z.number().min(0).max(2).optional().describe("Sampling temperature; omit for the model's default"),
474
+ tools: z
475
+ .array(z.object({}).passthrough())
476
+ .optional()
477
+ .describe("OpenAI-shaped function definitions the model may call. When the model answers with tool " +
478
+ "calls, they come back in `tool_calls` and the reply text is empty: run them, then call " +
479
+ "again with `messages` carrying the assistant turn and your tool results."),
480
+ tool_choice: z
481
+ .unknown()
482
+ .optional()
483
+ .describe('How the model may use tools: "auto", "none", "required", or {type:"function",function:{name}}.'),
484
+ response_format: z
485
+ .object({})
486
+ .passthrough()
487
+ .optional()
488
+ .describe('Structured output, e.g. {"type":"json_object"} for JSON mode, or a json_schema block.'),
284
489
  },
285
490
  outputSchema: {
286
491
  text: z.string().optional(),
492
+ tool_calls: z.array(z.object({}).passthrough()).nullable().optional(),
287
493
  model: z.string().optional(),
288
494
  finish_reason: z.string().optional(),
289
495
  usage: z
@@ -298,13 +504,21 @@ export function createCorentServer(config = {}) {
298
504
  error: z.string().optional(),
299
505
  },
300
506
  annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: true },
301
- }, wrap(async ({ prompt, system, tier, model, max_tokens, temperature }) => {
302
- const messages = system
303
- ? [{ role: "system", content: system }, { role: "user", content: prompt }]
304
- : [{ role: "user", content: prompt }];
507
+ }, wrap(async ({ prompt, system, messages: history, tier, model, max_tokens, temperature, tools, tool_choice, response_format }) => {
508
+ if (!history && !prompt) {
509
+ throw new CorentApiError(422, { message: "Pass either `prompt` (single question) or `messages` (conversation)." });
510
+ }
511
+ // An explicit conversation wins: it carries the turns and tool results a
512
+ // single prompt cannot express.
513
+ const messages = history ??
514
+ (system
515
+ ? [{ role: "system", content: system }, { role: "user", content: prompt }]
516
+ : [{ role: "user", content: prompt }]);
305
517
  // The API's `model` field is required and carries both modes: a pinned
306
- // catalog name, or "corent/text-<tier>" for the brain-routed lane. lite
307
- // matches the router's own default for an unrecognised tier.
518
+ // model name ("corent-claude-opus-5"), or "corent/text-<tier>" for the
519
+ // brain-routed lane -- note the slash, a separate namespace from the
520
+ // dash-prefixed model names. lite matches the router's own default for an
521
+ // unrecognised tier.
308
522
  const r = await corent("/v1/chat/completions", {
309
523
  method: "POST",
310
524
  body: JSON.stringify({
@@ -312,11 +526,18 @@ export function createCorentServer(config = {}) {
312
526
  messages,
313
527
  max_tokens,
314
528
  temperature,
529
+ tools,
530
+ tool_choice,
531
+ response_format,
315
532
  }),
316
533
  });
317
534
  const choice = (r.choices ?? [{}])[0] ?? {};
318
535
  return {
319
536
  text: choice.message?.content ?? "",
537
+ // When the model answers with tool calls the content is empty and THIS
538
+ // is the answer. Returning only `text` made a tool-calling reply look
539
+ // like an empty response (parity audit 2026-08-30).
540
+ tool_calls: choice.message?.tool_calls ?? null,
320
541
  model: r.model,
321
542
  finish_reason: choice.finish_reason,
322
543
  usage: r.usage,
@@ -326,7 +547,8 @@ export function createCorentServer(config = {}) {
326
547
  }));
327
548
  server.registerTool("list_models", {
328
549
  title: "List available models",
329
- description: "The direct-access menu: every model that can be pinned by name on generate_image / generate_video / generate_speech / generate_text, with its kind (image, video, speech, text), quality score and live status. Carries NO price -- do not promise the user a per-model rate from this; billing is flat cost-plus and the exact charge is returned on each generation. Read-only and free. Use this only when the user wants a SPECIFIC model -- otherwise omit `model` and let Corent route to the best one for the prompt.",
550
+ _meta: toolMeta("Listing models", "Models listed"),
551
+ description: "The direct-access menu: every model that can be pinned by name (spelled corent-*) on the generate tools, with kind, quality score and live status. Carries no price; billing is flat cost-plus and each generation returns its exact charge. Read-only and free. Only needed when the user names a specific model.",
330
552
  inputSchema: {},
331
553
  outputSchema: {
332
554
  models: z
@@ -339,24 +561,145 @@ export function createCorentServer(config = {}) {
339
561
  },
340
562
  annotations: { readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: true },
341
563
  }, wrap(async () => corent("/v1/models")));
564
+ server.registerTool("list_voices", {
565
+ title: "List voices",
566
+ _meta: toolMeta("Listing voices", "Voices listed"),
567
+ description: "The voices generate_speech accepts as voice_id, with display name, preview clip and descriptors (gender, age, accent, use case). Read-only and free. Call it before generate_speech whenever the user wants a particular narrator; a voice_id cannot be guessed.",
568
+ inputSchema: {},
569
+ outputSchema: {
570
+ voices: z
571
+ .array(z.object({ voice_id: z.string() }).passthrough())
572
+ .optional(),
573
+ count: z.number().optional(),
574
+ default_voice_id: z.string().nullable().optional(),
575
+ error: z.string().optional(),
576
+ },
577
+ annotations: { readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: true },
578
+ }, wrap(async () => corent("/v1/voices")));
579
+ server.registerTool("cancel_job", {
580
+ title: "Cancel a running job",
581
+ _meta: toolMeta("Cancelling job", "Job cancelled"),
582
+ description: "Stop a job that has not finished and release the money held for it. Charges nothing. Use it as soon as the user says they did not mean to start something, especially a video. A job that already finished stays completed and billed for what was delivered.",
583
+ inputSchema: { job_id: z.string().describe("The job id to cancel") },
584
+ outputSchema: {
585
+ id: z.string().optional(),
586
+ status: z.string().optional(),
587
+ cancelled: z.boolean().optional(),
588
+ detail: z.string().optional(),
589
+ error: z.string().optional(),
590
+ },
591
+ annotations: { readOnlyHint: false, destructiveHint: true, idempotentHint: true, openWorldHint: true },
592
+ }, wrap(async ({ job_id }) => corent(`/v1/jobs/${job_id}/cancel`, { method: "POST" })));
593
+ server.registerTool("list_tiers", {
594
+ title: "List quality tiers and prices",
595
+ _meta: toolMeta("Listing tiers", "Tiers listed"),
596
+ description: "The live tier menu for image and video: each tier's price estimate, shapes, video durations, per-resolution pricing, and whether it accepts reference images (image) or a source image (video). Read-only and free. Check it before quoting a price. Tier-level only; never names a model.",
597
+ inputSchema: {},
598
+ outputSchema: {
599
+ tiers: z
600
+ .array(z
601
+ .object({
602
+ name: z.string().optional(),
603
+ estimated_cost_cents: z.number().nullable().optional(),
604
+ capabilities: z.object({}).passthrough().optional(),
605
+ })
606
+ .passthrough())
607
+ .optional(),
608
+ pricing_note: z.string().optional(),
609
+ error: z.string().optional(),
610
+ },
611
+ annotations: { readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: true },
612
+ }, wrap(async () => corent("/v1/tiers")));
613
+ // --- Batch tools: one call, up to 50 renders. The single-item tools are the
614
+ // right default; these exist for the "make me 30 variations" ask, where 30
615
+ // separate tool calls would be slower and far more expensive in context. ---
616
+ const imageBatchItem = z
617
+ .object({
618
+ prompt: z.string(),
619
+ tier: z.enum(TIERS).optional(),
620
+ style: z.enum(IMAGE_STYLES).optional(),
621
+ aspect_ratio: z.enum(IMAGE_ASPECTS).optional(),
622
+ })
623
+ .describe("One image in the batch. Batch items are tier-routed: to pin an exact model, use generate_image.");
624
+ server.registerTool("generate_image_batch", {
625
+ title: "Generate many images (batch)",
626
+ _meta: toolMeta("Submitting image batch", "Image batch submitted"),
627
+ description: "Submit up to 50 image generations in one call. Asynchronous: returns a batch_id and a job id per item; poll get_batch, then get_job per image. Every item bills at the normal rate, so check get_balance first. Items are tier-routed; pin a model with generate_image instead. No reference images here.",
628
+ inputSchema: {
629
+ items: z.array(imageBatchItem).min(1).max(50).describe("1 to 50 image requests"),
630
+ },
631
+ outputSchema: batchOutput,
632
+ annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: true },
633
+ }, wrap(async ({ items }) => corent("/v1/images/generate/batch", {
634
+ method: "POST",
635
+ headers: { "Idempotency-Key": idempotencyKey() },
636
+ body: JSON.stringify({ items }),
637
+ })));
638
+ server.registerTool("generate_video_batch", {
639
+ title: "Generate many videos (batch)",
640
+ _meta: toolMeta("Submitting video batch", "Video batch submitted"),
641
+ description: "Submit up to 50 video generations in one call. Asynchronous: returns a batch_id and a job id per item; poll get_batch, then get_job per clip. Video is the expensive lane: a 10-item batch can cost several dollars, so confirm with the user and check get_balance first. Items are tier-routed.",
642
+ inputSchema: {
643
+ items: z
644
+ .array(z.object({
645
+ prompt: z.string(),
646
+ tier: z.enum(TIERS).optional(),
647
+ style: z.enum(IMAGE_STYLES).optional(),
648
+ aspect_ratio: z.enum(["16:9", "9:16", "1:1"]).optional(),
649
+ duration_s: z.number().int().min(1).max(30).optional(),
650
+ resolution: z.enum(["720p", "1080p", "4k"]).optional(),
651
+ }))
652
+ .min(1)
653
+ .max(50)
654
+ .describe("1 to 50 video requests"),
655
+ },
656
+ outputSchema: batchOutput,
657
+ annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: true },
658
+ }, wrap(async ({ items }) => corent("/v1/videos/generate/batch", {
659
+ method: "POST",
660
+ headers: { "Idempotency-Key": idempotencyKey() },
661
+ body: JSON.stringify({ items }),
662
+ })));
663
+ server.registerTool("get_batch", {
664
+ title: "Check batch progress",
665
+ _meta: toolMeta("Checking batch", "Batch checked"),
666
+ description: "Progress of a batch from generate_image_batch or generate_video_batch: completed, failed and pending counts plus each item's job id. Read-only and free. Fetch a finished item's media with get_job.",
667
+ inputSchema: { batch_id: z.string().describe("The batch_id returned by a batch tool") },
668
+ outputSchema: {
669
+ batch_id: z.string().optional(),
670
+ total: z.number().optional(),
671
+ completed: z.number().optional(),
672
+ failed: z.number().optional(),
673
+ pending: z.number().optional(),
674
+ jobs: z.array(z.object({ id: z.string(), status: z.string() }).passthrough()).optional(),
675
+ error: z.string().optional(),
676
+ },
677
+ annotations: { readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: true },
678
+ }, wrap(async ({ batch_id }) => corent(`/v1/batches/${batch_id}`)));
342
679
  server.registerTool("get_job", {
343
680
  title: "Check job status",
344
- _meta: WIDGET_TOOL_META,
345
- description: "Check the status of a generation job (mainly videos). When completed, the response includes the permanent media URL and the cost. Read-only and free.",
681
+ _meta: toolMeta("Checking job", "Job checked", true),
682
+ description: "Check a generation job (mainly video). When completed, the response carries the permanent media URL and the cost. Read-only and free.",
346
683
  inputSchema: { job_id: z.string().describe("The job id returned by generate_video or generate_image") },
347
684
  outputSchema: jobOutput,
348
685
  annotations: { readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: true },
349
686
  }, wrap(async ({ job_id }) => corent(`/v1/jobs/${job_id}`)));
350
687
  server.registerTool("get_balance", {
351
688
  title: "Check account balance",
352
- description: "Get the Corent account's remaining balance in cents. Useful before starting expensive video generations. Read-only and free.",
689
+ _meta: toolMeta("Checking balance", "Balance checked"),
690
+ description: "The account balance in cents. balance_cents is the total, held_cents is reserved by running generations, available_cents is what a new request can actually spend; check that one before an expensive video. Read-only and free.",
353
691
  inputSchema: {},
354
- outputSchema: { balance_cents: z.number().optional() },
692
+ outputSchema: {
693
+ balance_cents: z.number().optional(),
694
+ held_cents: z.number().optional(),
695
+ available_cents: z.number().optional(),
696
+ },
355
697
  annotations: { readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: true },
356
698
  }, wrap(async () => corent("/v1/account/balance")));
357
699
  server.registerTool("get_status", {
358
700
  title: "Check service status",
359
- description: "Get live operational status of Corent's generation tiers (operational/degraded). Read-only and free; useful to pick a healthy tier.",
701
+ _meta: toolMeta("Checking status", "Status checked"),
702
+ description: "Live operational status of Corent's generation tiers (operational or degraded). Read-only and free; useful to pick a healthy tier.",
360
703
  inputSchema: {},
361
704
  // /v1/status returns tiers as an ARRAY of {name, status} — declaring it
362
705
  // as a record made every call fail output validation (audit 2026-07-20).
@@ -371,7 +714,8 @@ export function createCorentServer(config = {}) {
371
714
  // prompt. The agent never picks a model or tier. ---
372
715
  server.registerTool("plan", {
373
716
  title: "Plan (cost preview, no generation)",
374
- description: "Preview how Corent would handle a plain-language request WITHOUT generating anything (costs a fraction of a cent). Corent decides whether it's an image or video, which tier, aspect ratio, and duration, and returns the plan plus an estimated cost. Use this to decide or confirm cost before spending. The planner covers image and video; for spoken audio call generate_speech directly. If the request is something Corent can't generate (text documents, 3D, editing, real-world actions), can_fulfill is false with a reason.",
717
+ _meta: toolMeta("Planning", "Plan ready"),
718
+ description: "Preview how Corent would handle a plain-language request without generating anything (a fraction of a cent). Returns image vs video, tier, aspect ratio, duration and an estimated cost. Covers image and video only; for speech call generate_speech. Impossible asks come back with can_fulfill false and a reason.",
375
719
  inputSchema: {
376
720
  intent: z
377
721
  .string()
@@ -385,8 +729,8 @@ export function createCorentServer(config = {}) {
385
729
  }, wrap(async ({ intent }) => corent("/v1/intent", { method: "POST", body: JSON.stringify({ intent }) })));
386
730
  server.registerTool("create", {
387
731
  title: "Create (plan + generate, budget-capped)",
388
- _meta: WIDGET_TOOL_META,
389
- description: "Describe what you want in plain language and Corent plans AND generates it — choosing image vs video, tier, aspect ratio, and duration for you (the 'zero decisions' path). You MUST pass max_cost_cents as a spend ceiling; if the estimated cost exceeds it, nothing is generated and you're told the estimate so you can raise the ceiling. Image results come back with a URL; video results come back as a job id to poll with get_job. Failed generations are never billed.",
732
+ _meta: toolMeta("Creating", "Created", true),
733
+ description: "Describe what you want in plain language; Corent plans and generates it, choosing image vs video, tier, shape and duration. max_cost_cents is a required spend ceiling: above it nothing is generated and you are told the estimate. Images return a URL; videos return a job id for get_job. Failed generations are never billed.",
390
734
  inputSchema: {
391
735
  intent: z.string().describe("What you want, in natural language"),
392
736
  max_cost_cents: z
@@ -416,5 +760,151 @@ export function createCorentServer(config = {}) {
416
760
  }
417
761
  return res;
418
762
  }));
763
+ // --- Image tools (v0.8.0): edit what already exists instead of generating
764
+ // anew. All three are synchronous, flat-priced, and replay-safe under an
765
+ // Idempotency-Key exactly like generate_image. ---
766
+ server.registerTool("upscale_image", {
767
+ title: "Upscale image",
768
+ _meta: toolMeta("Upscaling image", "Upscaled image ready", true),
769
+ description: "Make an existing image bigger and sharper from its public https URL. Synchronous, returns a new permanent URL. Flat price, no tier; failed runs are never billed. Optional prompt guides the added detail; omit it to enlarge faithfully. Upload a local file with upload_image first.",
770
+ inputSchema: {
771
+ image_url: z.string().url().describe("Public https URL of the image to enlarge and sharpen"),
772
+ prompt: z.string().max(2000).optional().describe("Optional guidance for the added detail; omit to enlarge faithfully"),
773
+ },
774
+ outputSchema: imageOutput,
775
+ annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: true },
776
+ }, wrap(async (args) => corent("/v1/images/upscale", {
777
+ method: "POST",
778
+ headers: { "Idempotency-Key": idempotencyKey() },
779
+ body: JSON.stringify(args),
780
+ })));
781
+ server.registerTool("remove_background", {
782
+ title: "Remove background",
783
+ _meta: toolMeta("Removing background", "Cutout ready", true),
784
+ description: "Cut the subject out of an image and return a PNG with a real transparent background, for logos, product shots and compositing. Synchronous, flat price, no tier; failed runs are never billed. Takes a public https URL; upload a local file with upload_image first.",
785
+ inputSchema: {
786
+ image_url: z.string().url().describe("Public https URL of the image to cut out"),
787
+ },
788
+ outputSchema: imageOutput,
789
+ annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: true },
790
+ }, wrap(async (args) => corent("/v1/images/remove-background", {
791
+ method: "POST",
792
+ headers: { "Idempotency-Key": idempotencyKey() },
793
+ body: JSON.stringify(args),
794
+ })));
795
+ server.registerTool("inpaint_image", {
796
+ title: "Inpaint image (masked edit)",
797
+ _meta: toolMeta("Inpainting image", "Edited image ready", true),
798
+ description: "Regenerate ONLY the white region of a mask and leave every pixel outside it byte-for-byte unchanged, so a product or logo stays exactly your file. Takes image_url, a same-size black-and-white mask_url (white = redo, black = keep) and a prompt for the masked area. Synchronous; failed runs are never billed.",
799
+ inputSchema: {
800
+ image_url: z.string().url().describe("Public https URL of the image to edit"),
801
+ mask_url: z
802
+ .string()
803
+ .url()
804
+ .describe("Public https URL of a black-and-white mask the same size as the image: WHITE is regenerated, BLACK is preserved byte for byte"),
805
+ prompt: z.string().min(1).max(2000).describe("What to generate inside the white (masked) region"),
806
+ tier: z.enum(TIERS).optional().describe("Quality tier for the regenerated region; omit for the default"),
807
+ },
808
+ outputSchema: imageOutput,
809
+ annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: true },
810
+ }, wrap(async (args) => corent("/v1/images/edit", {
811
+ method: "POST",
812
+ headers: { "Idempotency-Key": idempotencyKey() },
813
+ body: JSON.stringify(args),
814
+ })));
815
+ // --- Memory: what the account already made, so an assistant can find "the
816
+ // beach one from yesterday" and reuse the URL instead of paying again. ---
817
+ const MEDIA_TYPES = ["image", "video", "speech", "text"];
818
+ const PROMPT_PREVIEW_CHARS = 120;
819
+ server.registerTool("list_media", {
820
+ title: "List recent media",
821
+ _meta: toolMeta("Searching your media", "Media listed"),
822
+ description: "Find media the account already made, so a result can be reused instead of paid for again. Reads the last 100 jobs, filtered by type and a prompt keyword (e.g. 'beach'), newest first. Returns id, type, status, date, tier, cost and URL per job. Read-only and free.",
823
+ inputSchema: {
824
+ type: z.enum(MEDIA_TYPES).optional().describe("Only this lane: image, video, speech or text"),
825
+ limit: z.number().int().min(1).max(100).default(20).describe("How many rows to return, 1-100 (default 20)"),
826
+ query: z.string().optional().describe("Case-insensitive substring to match against each job's prompt"),
827
+ },
828
+ outputSchema: {
829
+ items: z
830
+ .array(z
831
+ .object({
832
+ id: z.string(),
833
+ type: z.string().nullable().optional(),
834
+ status: z.string().nullable().optional(),
835
+ created_at: z.string().nullable().optional(),
836
+ tier: z.string().nullable().optional(),
837
+ cost_cents: z.number().nullable().optional(),
838
+ prompt: z.string().nullable().optional(),
839
+ url: z.string().nullable().optional(),
840
+ })
841
+ .passthrough())
842
+ .optional(),
843
+ count: z.number().optional(),
844
+ scanned: z.number().optional(),
845
+ error: z.string().optional(),
846
+ },
847
+ annotations: { readOnlyHint: true, destructiveHint: false, idempotentHint: true, openWorldHint: true },
848
+ }, wrap(async ({ type, limit, query }) => {
849
+ const usage = await corent("/v1/account/usage");
850
+ const recent = Array.isArray(usage?.recent_jobs) ? usage.recent_jobs : [];
851
+ const needle = query?.trim().toLowerCase();
852
+ const items = recent
853
+ .filter((j) => !type || j.type === type)
854
+ .filter((j) => !needle || String(j.prompt ?? "").toLowerCase().includes(needle))
855
+ .slice(0, limit ?? 20)
856
+ .map((j) => {
857
+ const prompt = typeof j.prompt === "string" ? j.prompt : null;
858
+ return {
859
+ id: String(j.id),
860
+ type: j.type ?? null,
861
+ status: j.status ?? null,
862
+ created_at: j.created_at ?? null,
863
+ tier: j.model ?? null,
864
+ cost_cents: j.charge_cents ?? null,
865
+ prompt: prompt && prompt.length > PROMPT_PREVIEW_CHARS ? `${prompt.slice(0, PROMPT_PREVIEW_CHARS - 1)}…` : prompt,
866
+ url: j.output_url ?? null,
867
+ };
868
+ });
869
+ return { items, count: items.length, scanned: recent.length };
870
+ }));
871
+ // --- Upload: turn a local file into a URL the other tools accept. ---
872
+ const UPLOAD_MAX_BYTES = 25 * 1024 * 1024;
873
+ server.registerTool("upload_image", {
874
+ title: "Upload image",
875
+ _meta: toolMeta("Uploading image", "Upload complete"),
876
+ description: "Upload a local image (base64) to Corent and get back a public https URL. Use that URL as image_url, mask_url, source_image_url or a reference in the other tools. Max 25 MB. Free; the file is stored on the account.",
877
+ inputSchema: {
878
+ data_base64: z.string().min(1).describe("The file bytes, base64 encoded (no data: prefix)"),
879
+ content_type: z
880
+ .string()
881
+ .regex(/^image\/[a-z0-9.+-]+$/i)
882
+ .describe('MIME type, e.g. "image/png" or "image/jpeg"'),
883
+ filename: z.string().max(255).optional().describe("Optional file name, e.g. product.png"),
884
+ },
885
+ outputSchema: {
886
+ url: z.string().optional(),
887
+ id: z.string().optional(),
888
+ content_type: z.string().optional(),
889
+ bytes: z.number().optional(),
890
+ error: z.string().optional(),
891
+ },
892
+ annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: true, openWorldHint: true },
893
+ }, wrap(async ({ data_base64, content_type, filename }) => {
894
+ // Strip an accidental data: URL prefix so a copy-pasted value still works.
895
+ const raw = data_base64.replace(/^data:[^;]+;base64,/, "").trim();
896
+ // Decoded size, computed without decoding: 3 bytes per 4 characters,
897
+ // minus padding. Refusing here saves shipping 30 MB to the API for a 413.
898
+ const padding = raw.endsWith("==") ? 2 : raw.endsWith("=") ? 1 : 0;
899
+ const bytes = Math.floor((raw.length * 3) / 4) - padding;
900
+ if (bytes > UPLOAD_MAX_BYTES) {
901
+ throw new CorentApiError(413, { message: `File is ${bytes} bytes; the upload limit is 25 MB.` });
902
+ }
903
+ return corent("/v1/uploads", {
904
+ method: "POST",
905
+ body: JSON.stringify({ data_base64: raw, content_type, filename }),
906
+ });
907
+ }));
908
+ registerPrompts(server);
419
909
  return server;
420
910
  }