ottoport 1.5.2 → 1.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "ottoport",
3
3
  "description": "One API for every model. Call chat, image, video, speech and music models through the OttoPort gateway — as MCP tools, slash commands, or the bundled CLI.",
4
- "version": "1.5.2",
4
+ "version": "1.6.0",
5
5
  "author": {
6
6
  "name": "LITBOX LLC",
7
7
  "email": "support@ottoport.ai"
package/README.md CHANGED
@@ -26,6 +26,8 @@ ottoport image --model qwen-image-layered --image <url> [--layers 4]
26
26
  ottoport video "<prompt>" [--model kling-3.0] [--duration 5] [--resolution 1080p]
27
27
  [--aspect-ratio 16:9] [--image <url>] [--last-frame <url>]
28
28
  [--ref <url>[,<url>…]] [--out clip.mp4]
29
+ [--webhook-url https://…] [--no-wait]
30
+ ottoport webhook status|retry <job-id> # delivery attempts / redeliver now
29
31
  ottoport speech "<text>" [--model gpt-4o-mini-tts] [--voice alloy] [--out speech.mp3]
30
32
  ottoport music "<prompt>" [--model suno-v5] [--duration 15] [--out song.mp3]
31
33
 
@@ -75,7 +77,8 @@ forwarded from the environment rather than written into any config file.
75
77
  ## MCP tools
76
78
 
77
79
  `ottoport_list_models`, `ottoport_chat`, `ottoport_generate_image`,
78
- `ottoport_generate_video`, `ottoport_generate_speech`, `ottoport_generate_music`.
80
+ `ottoport_generate_video`, `ottoport_video_webhook`, `ottoport_generate_speech`,
81
+ `ottoport_generate_music`.
79
82
  The server speaks stdio JSON-RPC and is launched by the host as a subprocess, so
80
83
  it needs Node 20+ on the PATH of whatever starts that host.
81
84
 
@@ -99,6 +102,17 @@ reads, pass `model` to that tool or run `ottoport models <model-id>`. Asking
99
102
  for something a model does not offer is refused with its actual list, before a
100
103
  generation is spent.
101
104
 
105
+ ### Webhooks
106
+
107
+ Pass `webhook_url` (MCP) or `--webhook-url` (CLI) with a video to have OttoPort
108
+ POST the finished job to that HTTPS endpoint as `video.completed` or
109
+ `video.failed`. Deliveries are signed:
110
+ `x-ottoport-signature: t=<unix>,v1=<hex HMAC-SHA256(secret, "<t>.<raw body>")>`.
111
+ Failed deliveries are retried after 1, 5, 30 and 120 minutes, 5 attempts in
112
+ all (redirects are not followed); `ottoport webhook retry <job-id>` or `ottoport_video_webhook` sends
113
+ one again at any time. Verification examples are at
114
+ https://ottoport.ai/docs#webhooks.
115
+
102
116
  ## Configuration
103
117
 
104
118
  - `OTTOPORT_API_KEY` — your `op-…` key. Export it from your shell profile: the
@@ -108,13 +122,17 @@ generation is spent.
108
122
  only for local development or a self-hosted gateway.
109
123
  - `OTTOPORT_BASE_URL` configures the MCP server only; it does not silently
110
124
  redirect CLI traffic.
125
+ - `OTTOPORT_WEBHOOK_SECRET` — the secret video webhooks are signed with, used
126
+ when a webhook URL is given without an explicit secret. Keep it in the
127
+ environment rather than on the command line.
111
128
 
112
129
  ## Notes
113
130
 
114
131
  - Image, video, and audio calls return **URLs**, and provider URLs expire.
115
132
  Download anything worth keeping.
116
133
  - Video generation takes a minute or more and the call blocks until the job is
117
- terminal. A retry is a second billable generation, not a resumption.
134
+ terminal, unless a webhook is given or `--no-wait` is passed. A retry is a
135
+ second billable generation, not a resumption.
118
136
  - `resolution` is a price multiplier, not a formatting hint. Leave it unset to
119
137
  bill at the model's base tier.
120
138
  - Requests are billed against the prepaid balance on your OttoPort account.
package/cli/ottoport.mjs CHANGED
@@ -257,11 +257,13 @@ async function cmdImage(prompt, flags) {
257
257
  }
258
258
 
259
259
  async function cmdVideo(prompt, flags) {
260
- if (!prompt) die("video requires a prompt: ottoport video \"a drone shot\"");
260
+ // A depth model needs only --video; the gateway refuses a prompt there.
261
+ if (!prompt && !last(flags.video)) die("video requires a prompt: ottoport video \"a drone shot\" (or --video <url> on a depth model)");
261
262
  const { baseUrl, apiKey } = config(flags);
262
263
  const body = {
263
264
  model: last(flags.model) || "kling-3.0",
264
- prompt,
265
+ ...(prompt ? { prompt } : {}),
266
+ ...(last(flags.video) ? { video_url: last(flags.video) } : {}),
265
267
  ...(last(flags.duration) ? { duration: Number(last(flags.duration)) } : {}),
266
268
  ...(last(flags.resolution) ? { resolution: last(flags.resolution) } : {}),
267
269
  ...(last(flags["aspect-ratio"]) ? { aspect_ratio: flags["aspect-ratio"] } : {}),
@@ -269,6 +271,7 @@ async function cmdVideo(prompt, flags) {
269
271
  ...(last(flags.image) ? { image_url: flags.image } : {}),
270
272
  ...(last(flags["last-frame"]) ? { last_frame_url: flags["last-frame"] } : {}),
271
273
  ...(references(flags) ? { reference_image_urls: references(flags) } : {}),
274
+ ...webhook(flags),
272
275
  };
273
276
  console.error("submitting video job (this can take a minute)…");
274
277
  const res = await fetch(`${baseUrl}/api/v1/videos/generations`, {
@@ -279,7 +282,8 @@ async function cmdVideo(prompt, flags) {
279
282
  if (!res.ok) die(await readError(res));
280
283
  let job = await res.json();
281
284
  if (job.status === "failed") die(job.error || "video generation failed", 2);
282
- if (!flags["no-wait"]) {
285
+ if (body.webhook_url) process.stderr.write(`webhook: ${body.webhook_url} will be notified when the job finishes\n`);
286
+ if (flags.wait !== false) {
283
287
  process.stderr.write(`job ${job.id} accepted; waiting for completion…\n`);
284
288
  while (job.status === "queued" || job.status === "processing") {
285
289
  await new Promise((resolve) => setTimeout(resolve, 3_000));
@@ -298,6 +302,42 @@ async function cmdVideo(prompt, flags) {
298
302
  }
299
303
  }
300
304
 
305
+ /**
306
+ * `--webhook-url` registers the job's webhook. The secret comes from
307
+ * `--webhook-secret` or OTTOPORT_WEBHOOK_SECRET; flags are visible in shell
308
+ * history and `ps`, so the environment is the better home for it.
309
+ */
310
+ function webhook(flags) {
311
+ const url = last(flags["webhook-url"]);
312
+ if (!url) return {};
313
+ const secret = last(flags["webhook-secret"]) || process.env.OTTOPORT_WEBHOOK_SECRET;
314
+ if (!secret) die("--webhook-url needs a signing secret: set OTTOPORT_WEBHOOK_SECRET or pass --webhook-secret");
315
+ return { webhook_url: url, webhook_secret: secret };
316
+ }
317
+
318
+ function describeDelivery(d) {
319
+ const response = d.response_status ? `HTTP ${d.response_status}` : d.attempts ? "no response" : "not sent";
320
+ return `${d.event} ${d.status} ${d.attempts}/${d.max_attempts} attempts ${response}${d.response_body ? ` ${d.response_body.slice(0, 120)}` : ""}`;
321
+ }
322
+
323
+ async function cmdWebhook(action, jobId, flags) {
324
+ if (!["status", "retry"].includes(action) || !jobId) die("usage: ottoport webhook status|retry <job-id> [--json]");
325
+ const { baseUrl, apiKey } = config(flags);
326
+ const path = `${baseUrl}/api/v1/videos/generations/${encodeURIComponent(jobId)}`;
327
+ const res = action === "retry"
328
+ ? await fetch(`${path}/webhook`, { method: "POST", headers: headers(apiKey, false) })
329
+ : await fetch(path, { headers: headers(apiKey, false) });
330
+ if (!res.ok) die(await readError(res));
331
+ const body = await res.json();
332
+ if (flags.json) return void console.log(JSON.stringify(body, null, 2));
333
+ const deliveries = action === "retry" ? body.deliveries : body.webhook?.deliveries;
334
+ if (action === "status" && !body.webhook) return void console.log(`job ${jobId} (${body.status}) was submitted without a webhook`);
335
+ if (action === "status") console.log(`job ${jobId} (${body.status}) → ${body.webhook.url}`);
336
+ if (!deliveries?.length) return void console.log("no deliveries yet — one is sent when the job finishes");
337
+ for (const d of deliveries) console.log(describeDelivery(d));
338
+ if (action === "retry" && deliveries[0].status !== "delivered") process.exitCode = 2;
339
+ }
340
+
301
341
  async function cmdAudio(kind, prompt, flags) {
302
342
  if (!prompt) die(`${kind} requires a prompt: ottoport ${kind} "hello"`);
303
343
  const { baseUrl, apiKey } = config(flags);
@@ -345,7 +385,12 @@ Usage:
345
385
  ottoport video "<prompt>" [--model kling-3.0] [--duration 5] [--resolution 1080p]
346
386
  [--aspect-ratio 16:9] [--seed 7] [--image <url>]
347
387
  [--last-frame <url>] [--ref <url>[,<url>…]]
388
+ [--webhook-url https://…] [--webhook-secret <s>]
348
389
  [--no-wait] [--out clip.mp4]
390
+ ottoport video --model video-depth-anything --video <url> [--out depth.mp4]
391
+ turn a video (≤60s) into a grayscale depth video
392
+ ottoport webhook status <job-id> [--json] delivery attempts and responses
393
+ ottoport webhook retry <job-id> [--json] redeliver a finished job's webhook now
349
394
  ottoport speech "<text>" [--model gpt-4o-mini-tts] [--voice alloy] [--format mp3] [--out speech.mp3]
350
395
  ottoport music "<prompt>" [--model suno-v5] [--duration 15] [--format mp3] [--out song.mp3]
351
396
  ottoport mcp
@@ -356,6 +401,7 @@ Usage:
356
401
  Global flags:
357
402
  --url <base> override https://ottoport.ai (local/self-hosted development)
358
403
  --key <key> OttoPort API key (env OTTOPORT_API_KEY)
404
+ --webhook-secret <s> signs webhook deliveries (env OTTOPORT_WEBHOOK_SECRET)
359
405
 
360
406
  Examples:
361
407
  export OTTOPORT_API_KEY=op-...
@@ -365,6 +411,8 @@ Examples:
365
411
  ottoport image "isometric city at dusk" --model gpt-image-2 --out city.png
366
412
  ottoport video "cinematic product reveal" --model seedance-2.0-fast --resolution 1080p
367
413
  ottoport video "the logo unfolds" --image start.png --last-frame end.png
414
+ ottoport video "drone over fjords" --webhook-url https://example.com/hooks/ottoport --no-wait
415
+ ottoport webhook retry 7c1e9a52-…
368
416
  ottoport image "her, on a beach" --ref face.png,style.png
369
417
  ottoport image --model seedream-5.0-layers --image https://…/poster.png
370
418
  ottoport speech "Welcome to OttoPort" --voice alloy --out welcome.mp3
@@ -384,6 +432,7 @@ async function main() {
384
432
  case "chat": return await cmdChat(prompt, flags);
385
433
  case "image": return await cmdImage(prompt, flags);
386
434
  case "video": return await cmdVideo(prompt, flags);
435
+ case "webhook": return await cmdWebhook(rest[0], rest[1], flags);
387
436
  case "speech": return await cmdAudio("speech", prompt, flags);
388
437
  case "music": return await cmdAudio("music", prompt, flags);
389
438
  case "mcp": return await import(new URL("../mcp/server.mjs", import.meta.url));
package/mcp/server.mjs CHANGED
@@ -177,17 +177,25 @@ async function generateImage({ prompt, model, ...rest }) {
177
177
  return lines.length ? lines.join("\n") : "(no image returned)";
178
178
  }
179
179
 
180
- async function generateVideo({ prompt, model, ...rest }) {
181
- if (!prompt) throw new Error("`prompt` is required");
180
+ async function generateVideo({ prompt, model, webhook_url, webhook_secret, ...rest }) {
181
+ // A depth model takes only `video_url`; the gateway says which models need a prompt.
182
+ if (!prompt && !rest.video_url) throw new Error("`prompt` is required");
183
+ // The secret is best left in the server's environment, so an agent never has
184
+ // to carry it through a conversation.
185
+ const secret = webhook_secret || process.env.OTTOPORT_WEBHOOK_SECRET;
186
+ if (webhook_url && !secret) throw new Error("`webhook_url` needs a signing secret: set OTTOPORT_WEBHOOK_SECRET for the MCP server or pass `webhook_secret`");
182
187
  const res = await fetch(`${BASE}/api/v1/videos/generations`, {
183
188
  method: "POST",
184
189
  headers: { ...headers(), "idempotency-key": crypto.randomUUID() },
185
190
  // Passed through; see `generateImage` for why this layer names nothing.
186
- body: JSON.stringify({ model: model || "kling-3.0", prompt, ...rest }),
191
+ body: JSON.stringify({ model: model || "kling-3.0", ...(prompt ? { prompt } : {}), ...rest, ...(webhook_url ? { webhook_url, webhook_secret: secret } : {}) }),
187
192
  });
188
193
  if (!res.ok) throw new Error(await gwError(res));
189
194
  let job = await res.json();
190
195
  if (job.status === "failed") throw new Error(job.error || "video generation failed");
196
+ // With a webhook the caller has said where the result should go; holding the
197
+ // tool call open for minutes as well would only block the agent.
198
+ if (webhook_url) return `Video job ${job.id} is ${job.status}. ${webhook_url} will receive video.completed or video.failed when it finishes; check deliveries with ottoport_video_webhook.`;
191
199
  // The public video endpoint is deliberately asynchronous. MCP tools should
192
200
  // still fulfil their promise of returning usable output, so wait for the
193
201
  // job rather than returning a queued-job JSON blob to the agent.
@@ -204,6 +212,18 @@ async function generateVideo({ prompt, model, ...rest }) {
204
212
  return url || JSON.stringify(job);
205
213
  }
206
214
 
215
+ async function videoWebhook({ job_id, action = "status" }) {
216
+ if (!job_id) throw new Error("`job_id` is required");
217
+ const path = `${BASE}/api/v1/videos/generations/${encodeURIComponent(job_id)}`;
218
+ const res = action === "retry"
219
+ ? await fetch(`${path}/webhook`, { method: "POST", headers: headers(false) })
220
+ : await fetch(path, { headers: headers(false) });
221
+ if (!res.ok) throw new Error(await gwError(res));
222
+ const body = await res.json();
223
+ if (action === "retry") return JSON.stringify(body.deliveries ?? [], null, 2);
224
+ return JSON.stringify({ id: body.id, status: body.status, webhook: body.webhook ?? null }, null, 2);
225
+ }
226
+
207
227
  async function generateAudio(kind, { prompt, model, voice, duration, format }) {
208
228
  if (!prompt) throw new Error("`prompt` is required");
209
229
  const res = await fetch(`${BASE}/api/v1/audio/${kind === "tts" ? "speech" : "music"}`, {
@@ -275,18 +295,21 @@ const TOOLS = [
275
295
  image_url: { type: "string", description: "One reference image, for editing and image-to-image — or, on a layer-decomposition model, the image to split." },
276
296
  reference_image_urls: { type: "array", items: { type: "string" }, description: "Several references — a character sheet, a product plus a scene — up to the model's maxInputImages. Call ottoport_list_models to see it." },
277
297
  num_layers: { type: "number", description: "Layer models that let you choose (qwen-image-layered, 2–10): how many RGBA layers to split into. Each bills separately. seedream-5.0-layers decides for itself and refuses this." },
298
+ horizontal_angle: { type: "number", description: "Camera models only (qwen-image-angles): orbit in degrees, 0–360. Needs image_url; prompt optional." },
299
+ vertical_angle: { type: "number", description: "Camera models only: elevation in degrees, -30 (below) to 90 (overhead)." },
300
+ zoom: { type: "number", description: "Camera models only: 0 (wide) to 10 (close)." },
278
301
  },
279
302
  },
280
303
  handler: generateImage,
281
304
  },
282
305
  {
283
306
  name: "ottoport_generate_video",
284
- description: "Generate a video through an OttoPort video model (Veo, Kling, Seedance, …). Four modes: text-to-video, image-to-video (`image_url`), first-and-last-frame (plus `last_frame_url`), and reference-to-video (`reference_image_urls`). Which a model accepts is in its capabilities — call ottoport_list_models. Blocks on the queue and returns a video URL.",
307
+ description: "Generate a video through an OttoPort video model (Veo, Kling, Seedance, …). Modes: text-to-video, image-to-video (`image_url`), first-and-last-frame (plus `last_frame_url`), reference-to-video (`reference_image_urls`), and on some models video edit/extend (`video_url`). video-depth-anything instead turns `video_url` into a grayscale depth video and takes no prompt. Which a model accepts is in its capabilities — call ottoport_list_models. Blocks on the queue and returns a video URL, unless `webhook_url` is given.",
285
308
  inputSchema: {
286
309
  type: "object",
287
310
  properties: {
288
- prompt: { type: "string" },
289
- model: { type: "string", description: "Model id, e.g. veo-3.1, kling-3.0, seedance-2.0-fast. Default kling-3.0." },
311
+ prompt: { type: "string", description: "Required, except on video-depth-anything, which takes none." },
312
+ model: { type: "string", description: "Model id, e.g. veo-3.1, kling-3.0, seedance-2.0-fast, video-depth-anything. Default kling-3.0." },
290
313
  duration: { type: "number", description: "Seconds. The main lever on what this costs." },
291
314
  // Per-model, not universal — see the note on the image tool. kling-3.0
292
315
  // has no 480p and no 4k; gemini-omni-flash has only 720p.
@@ -296,11 +319,29 @@ const TOOLS = [
296
319
  image_url: { type: "string", description: "First frame, for image-to-video." },
297
320
  last_frame_url: { type: "string", description: "Final frame. With `image_url`, this is first-and-last-frame generation." },
298
321
  reference_image_urls: { type: "array", items: { type: "string" }, description: "Subjects the shot carries through — a character, a product, a style — without being a frame of it." },
322
+ keyframes: { type: "array", items: { type: "object", properties: { image_url: { type: "string" }, time: { type: "number", description: "Seconds from the start; omit to spread evenly." } }, required: ["image_url"] }, description: "Images pinned in time, for keyframes-to-video." },
323
+ video_url: { type: "string", description: "A video to edit or extend, on models listing video-edit / video-extend — or the only input to video-depth-anything. Edits and depth bill per second of this video (at most 60s)." },
324
+ video_task: { type: "string", enum: ["edit", "extend"], description: "edit (default) restyles video_url; extend continues it for `duration` more seconds." },
325
+ webhook_url: { type: "string", description: "Public HTTPS endpoint to POST the finished job to (events video.completed / video.failed, signed with HMAC-SHA256). With it the tool returns the job id at once instead of waiting." },
326
+ webhook_secret: { type: "string", description: "Signing secret for webhook_url. Omit to use the server's OTTOPORT_WEBHOOK_SECRET." },
299
327
  },
300
- required: ["prompt"],
328
+ required: [],
301
329
  },
302
330
  handler: generateVideo,
303
331
  },
332
+ {
333
+ name: "ottoport_video_webhook",
334
+ description: "Inspect or redeliver a video job's webhook. `status` returns the job and every delivery attempt with the endpoint's response; `retry` sends a finished job's webhook again now (resetting its 5 automatic attempts) and returns the result.",
335
+ inputSchema: {
336
+ type: "object",
337
+ properties: {
338
+ job_id: { type: "string", description: "The video job id returned by ottoport_generate_video." },
339
+ action: { type: "string", enum: ["status", "retry"], description: "Default status." },
340
+ },
341
+ required: ["job_id"],
342
+ },
343
+ handler: videoWebhook,
344
+ },
304
345
  {
305
346
  name: "ottoport_generate_speech",
306
347
  description: "Convert text to speech through an OttoPort TTS model. Returns an audio URL or data URL.",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "ottoport",
3
- "version": "1.5.2",
3
+ "version": "1.6.0",
4
4
  "description": "Claude Code plugin, CLI and MCP server for OttoPort — one OpenAI-compatible API for every LLM, image, video, and speech model.",
5
5
  "homepage": "https://ottoport.ai",
6
6
  "repository": {
@@ -17,7 +17,8 @@ and do not call the HTTP API, when a tool covers the job:
17
17
  | `ottoport_list_models` | The catalog: what each model is for, and its rate in credits and USD. Filter with `modality`, or pass `model` for ONE model in full — its modes, its resolution tiers with prices, and every parameter it reads. |
18
18
  | `ottoport_chat` | `prompt`, plus optional `model`, `system`, `temperature`, `max_tokens`. |
19
19
  | `ottoport_generate_image` | `prompt`, plus optional `model`, `size` or `resolution`, `aspect_ratio`, `n`, `seed`, `image_url` (one reference), `reference_image_urls` (several). Layer decomposition too: `model` `qwen-image-layered` or `seedream-5.0-layers` with `image_url` alone (prompt optional, `num_layers` on Qwen) returns one URL per RGBA layer. |
20
- | `ottoport_generate_video` | `prompt`, plus optional `model`, `duration`, `resolution`, `aspect_ratio`, `seed`, `image_url` (first frame), `last_frame_url`, `reference_image_urls`. |
20
+ | `ottoport_generate_video` | `prompt`, plus optional `model`, `duration`, `resolution`, `aspect_ratio`, `seed`, `image_url` (first frame), `last_frame_url`, `reference_image_urls`, `webhook_url` (returns the job id at once and POSTs the result there; the secret comes from `webhook_secret` or the server's `OTTOPORT_WEBHOOK_SECRET`). |
21
+ | `ottoport_video_webhook` | `job_id`, plus `action` `status` (every delivery attempt and the endpoint's response) or `retry` (redeliver a finished job's webhook now). |
21
22
  | `ottoport_generate_speech` | `prompt` (the text), plus optional `model`, `voice`, `format`. |
22
23
  | `ottoport_generate_music` | `prompt`, plus optional `model`, `duration`, `format`. |
23
24