ottoport 1.8.0 → 1.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/cli/ottoport.mjs +6 -3
- package/mcp/server.mjs +57 -18
- package/package.json +1 -1
- package/skills/ottoport/SKILL.md +2 -2
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "ottoport",
|
|
3
3
|
"description": "One API for every model. Call chat, image, video, speech and music models through the OttoPort gateway — as MCP tools, slash commands, or the bundled CLI.",
|
|
4
|
-
"version": "1.
|
|
4
|
+
"version": "1.10.0",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "LITBOX LLC",
|
|
7
7
|
"email": "support@ottoport.ai"
|
package/cli/ottoport.mjs
CHANGED
|
@@ -253,8 +253,8 @@ async function download(url, out) {
|
|
|
253
253
|
* what a shell user reaches for anyway, so both spellings are accepted:
|
|
254
254
|
* `--ref a.png --ref b.png` and `--ref a.png,b.png`.
|
|
255
255
|
*/
|
|
256
|
-
function references(flags) {
|
|
257
|
-
const raw = flags.ref ?? flags.reference;
|
|
256
|
+
function references(flags, name = "ref") {
|
|
257
|
+
const raw = name === "ref" ? flags.ref ?? flags.reference : flags[name];
|
|
258
258
|
const list = (Array.isArray(raw) ? raw : [raw])
|
|
259
259
|
.filter(Boolean)
|
|
260
260
|
.flatMap((value) => String(value).split(",").map((one) => one.trim()))
|
|
@@ -308,6 +308,8 @@ async function cmdVideo(prompt, flags) {
|
|
|
308
308
|
...(last(flags.image) ? { image_url: flags.image } : {}),
|
|
309
309
|
...(last(flags["last-frame"]) ? { last_frame_url: flags["last-frame"] } : {}),
|
|
310
310
|
...(references(flags) ? { reference_image_urls: references(flags) } : {}),
|
|
311
|
+
...(references(flags, "ref-video") ? { reference_video_urls: references(flags, "ref-video") } : {}),
|
|
312
|
+
...(references(flags, "ref-audio") ? { reference_audio_urls: references(flags, "ref-audio") } : {}),
|
|
311
313
|
...webhook(flags),
|
|
312
314
|
};
|
|
313
315
|
console.error("submitting video job (this can take a minute)…");
|
|
@@ -410,7 +412,7 @@ async function cmdAudio(kind, prompt, flags) {
|
|
|
410
412
|
const HELP = `OttoPort — one command for every model.
|
|
411
413
|
|
|
412
414
|
Usage:
|
|
413
|
-
ottoport models [--modality chat|image|video|tts|music] [--json]
|
|
415
|
+
ottoport models [--modality chat|image|video|tts|stt|music|embedding] [--json]
|
|
414
416
|
ottoport models <model-id> [--json] what that one model accepts
|
|
415
417
|
ottoport chat "<prompt>" [--model claude-sonnet-5] [--system "..."] [--no-stream]
|
|
416
418
|
[--video <url|file>[,<url|file>…]] ask about a video
|
|
@@ -423,6 +425,7 @@ Usage:
|
|
|
423
425
|
ottoport video "<prompt>" [--model kling-3.0] [--duration 5] [--resolution 1080p]
|
|
424
426
|
[--aspect-ratio 16:9] [--seed 7] [--image <url>]
|
|
425
427
|
[--last-frame <url>] [--ref <url>[,<url>…]]
|
|
428
|
+
[--ref-video <url>[,<url>…]] [--ref-audio <url>[,<url>…]]
|
|
426
429
|
[--webhook-url https://…] [--webhook-secret <s>]
|
|
427
430
|
[--no-wait] [--out clip.mp4]
|
|
428
431
|
ottoport video --model video-depth-anything --video <url> [--out depth.mp4]
|
package/mcp/server.mjs
CHANGED
|
@@ -151,6 +151,34 @@ async function chat({ prompt, model, system, temperature, max_tokens }) {
|
|
|
151
151
|
return json?.choices?.[0]?.message?.content ?? "";
|
|
152
152
|
}
|
|
153
153
|
|
|
154
|
+
/**
|
|
155
|
+
* Poll a job until it finishes, or until `timeoutMs` runs out.
|
|
156
|
+
*
|
|
157
|
+
* `statusPath` is the job's status route with `{id}` for the id — the same
|
|
158
|
+
* template the gateway's quote reports as `job_path` — and it is always read
|
|
159
|
+
* from BASE: every request this server makes goes to OTTOPORT_BASE_URL, which
|
|
160
|
+
* is what lets a reseller put its own gateway in front without a code change.
|
|
161
|
+
*
|
|
162
|
+
* Returns the finished job, or the last one seen with `timedOut` set so the
|
|
163
|
+
* caller can say where to pick it up. A failed job throws.
|
|
164
|
+
*/
|
|
165
|
+
async function waitForJob(statusPath, job, { timeoutMs, label }) {
|
|
166
|
+
const path = statusPath.replace("{id}", encodeURIComponent(job.id));
|
|
167
|
+
const deadline = Date.now() + timeoutMs;
|
|
168
|
+
while (job.status === "queued" || job.status === "processing") {
|
|
169
|
+
if (Date.now() >= deadline) return { job, timedOut: true, path };
|
|
170
|
+
await new Promise((resolve) => setTimeout(resolve, 3_000));
|
|
171
|
+
const status = await fetch(`${BASE}${path}`, { headers: headers(false) });
|
|
172
|
+
if (!status.ok) throw new Error(await gwError(status));
|
|
173
|
+
job = await status.json();
|
|
174
|
+
}
|
|
175
|
+
if (job.status === "failed") throw new Error(job.error || `${label} generation failed`);
|
|
176
|
+
return { job, timedOut: false, path };
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
const IMAGE_JOB_PATH = "/api/v1/images/generations/{id}";
|
|
180
|
+
const VIDEO_JOB_PATH = "/api/v1/videos/generations/{id}";
|
|
181
|
+
|
|
154
182
|
async function generateImage({ prompt, model, ...rest }) {
|
|
155
183
|
// A layer-decomposition model splits `image_url` and takes the prompt as an
|
|
156
184
|
// optional caption; every other image model needs the prompt. The gateway
|
|
@@ -165,10 +193,21 @@ async function generateImage({ prompt, model, ...rest }) {
|
|
|
165
193
|
// live on the gateway and unreachable from any MCP client, silently. The
|
|
166
194
|
// gateway validates against the model's capabilities and says what it
|
|
167
195
|
// does not accept; this layer has no business having an opinion.
|
|
168
|
-
|
|
196
|
+
//
|
|
197
|
+
// Queued as a job and polled: a 4K or layered image can take longer than
|
|
198
|
+
// a proxy in front of the gateway holds a request open, and the hold is
|
|
199
|
+
// then settled on what actually came back.
|
|
200
|
+
body: JSON.stringify({ model: model || "gpt-image-2", ...(prompt ? { prompt } : {}), ...rest, async: true }),
|
|
169
201
|
});
|
|
170
202
|
if (!res.ok) throw new Error(await gwError(res));
|
|
171
|
-
|
|
203
|
+
let json = await res.json();
|
|
204
|
+
// A 202 is a job. A 200 with `data` is the image itself — the dev key, or a
|
|
205
|
+
// gateway that predates image jobs — and is read exactly as before.
|
|
206
|
+
if (res.status === 202) {
|
|
207
|
+
const { job, timedOut, path } = await waitForJob(IMAGE_JOB_PATH, json, { timeoutMs: 5 * 60_000, label: "image" });
|
|
208
|
+
if (timedOut) return `Image job ${job.id} is still processing. Poll ${path} for its result.`;
|
|
209
|
+
json = job;
|
|
210
|
+
}
|
|
172
211
|
// A layer stack prints one line per layer with what the model called it, so
|
|
173
212
|
// the agent can pick "headline" out of eight URLs without opening them.
|
|
174
213
|
const lines = (json.data ?? []).filter((d) => d.url).map((d) =>
|
|
@@ -179,7 +218,8 @@ async function generateImage({ prompt, model, ...rest }) {
|
|
|
179
218
|
|
|
180
219
|
async function generateVideo({ prompt, model, webhook_url, webhook_secret, ...rest }) {
|
|
181
220
|
// A depth model takes only `video_url`; the gateway says which models need a prompt.
|
|
182
|
-
|
|
221
|
+
// A draft completion takes everything from its draft, the prompt included.
|
|
222
|
+
if (!prompt && !rest.video_url && !rest.draft_id) throw new Error("`prompt` is required");
|
|
183
223
|
// The secret is best left in the server's environment, so an agent never has
|
|
184
224
|
// to carry it through a conversation.
|
|
185
225
|
const secret = webhook_secret || process.env.OTTOPORT_WEBHOOK_SECRET;
|
|
@@ -191,30 +231,25 @@ async function generateVideo({ prompt, model, webhook_url, webhook_secret, ...re
|
|
|
191
231
|
body: JSON.stringify({ model: model || "kling-3.0", ...(prompt ? { prompt } : {}), ...rest, ...(webhook_url ? { webhook_url, webhook_secret: secret } : {}) }),
|
|
192
232
|
});
|
|
193
233
|
if (!res.ok) throw new Error(await gwError(res));
|
|
194
|
-
|
|
195
|
-
if (
|
|
234
|
+
const submitted = await res.json();
|
|
235
|
+
if (submitted.status === "failed") throw new Error(submitted.error || "video generation failed");
|
|
196
236
|
// With a webhook the caller has said where the result should go; holding the
|
|
197
237
|
// tool call open for minutes as well would only block the agent.
|
|
198
|
-
if (webhook_url) return `Video job ${
|
|
238
|
+
if (webhook_url) return `Video job ${submitted.id} is ${submitted.status}. ${webhook_url} will receive video.completed or video.failed when it finishes; check deliveries with ottoport_video_webhook.`;
|
|
199
239
|
// The public video endpoint is deliberately asynchronous. MCP tools should
|
|
200
240
|
// still fulfil their promise of returning usable output, so wait for the
|
|
201
241
|
// job rather than returning a queued-job JSON blob to the agent.
|
|
202
|
-
const
|
|
203
|
-
|
|
204
|
-
if (Date.now() >= deadline) return `Video job ${job.id} is still processing. Poll /api/v1/videos/generations/${job.id} for its result.`;
|
|
205
|
-
await new Promise((resolve) => setTimeout(resolve, 3_000));
|
|
206
|
-
const status = await fetch(`${BASE}/api/v1/videos/generations/${encodeURIComponent(job.id)}`, { headers: headers(false) });
|
|
207
|
-
if (!status.ok) throw new Error(await gwError(status));
|
|
208
|
-
job = await status.json();
|
|
209
|
-
}
|
|
210
|
-
if (job.status === "failed") throw new Error(job.error || "video generation failed");
|
|
242
|
+
const { job, timedOut, path } = await waitForJob(VIDEO_JOB_PATH, submitted, { timeoutMs: 10 * 60_000, label: "video" });
|
|
243
|
+
if (timedOut) return `Video job ${job.id} is still processing. Poll ${path} for its result.`;
|
|
211
244
|
const url = job.data?.[0]?.url;
|
|
245
|
+
// The draft's job id is what completes it, so it is worth saying.
|
|
246
|
+
if (url && rest.draft) return `${url}\nDraft ${job.id} — complete it at 1080p with draft_id "${job.id}" within seven days.`;
|
|
212
247
|
return url || JSON.stringify(job);
|
|
213
248
|
}
|
|
214
249
|
|
|
215
250
|
async function videoWebhook({ job_id, action = "status" }) {
|
|
216
251
|
if (!job_id) throw new Error("`job_id` is required");
|
|
217
|
-
const path = `${BASE}
|
|
252
|
+
const path = `${BASE}${VIDEO_JOB_PATH.replace("{id}", encodeURIComponent(job_id))}`;
|
|
218
253
|
const res = action === "retry"
|
|
219
254
|
? await fetch(`${path}/webhook`, { method: "POST", headers: headers(false) })
|
|
220
255
|
: await fetch(path, { headers: headers(false) });
|
|
@@ -252,7 +287,7 @@ const TOOLS = [
|
|
|
252
287
|
inputSchema: {
|
|
253
288
|
type: "object",
|
|
254
289
|
properties: {
|
|
255
|
-
modality: { type: "string", enum: ["chat", "image", "video", "tts", "music"], description: "Filter to one modality." },
|
|
290
|
+
modality: { type: "string", enum: ["chat", "image", "video", "tts", "stt", "music", "embedding"], description: "Filter to one modality." },
|
|
256
291
|
model: { type: "string", description: "A model id, e.g. veo-3.1. Returns that one model's full parameter list instead of the catalog." },
|
|
257
292
|
},
|
|
258
293
|
},
|
|
@@ -305,7 +340,7 @@ const TOOLS = [
|
|
|
305
340
|
},
|
|
306
341
|
{
|
|
307
342
|
name: "ottoport_generate_video",
|
|
308
|
-
description: "Generate a video through an OttoPort video model (Veo, Kling, Seedance, …). Modes: text-to-video, image-to-video (`image_url`), first-and-last-frame (plus `last_frame_url`), reference-to-video (`reference_image_urls`), and on some models video edit/extend (`video_url`). video-depth-anything instead turns `video_url` into a grayscale depth video and takes no prompt. Which a model accepts is in its capabilities — call ottoport_list_models. Blocks on the queue and returns a video URL, unless `webhook_url` is given.",
|
|
343
|
+
description: "Generate a video through an OttoPort video model (Veo, Kling, Seedance, …). Modes: text-to-video, image-to-video (`image_url`), first-and-last-frame (plus `last_frame_url`), reference-to-video (`reference_image_urls`, plus `reference_video_urls` / `reference_audio_urls` on models whose capabilities list `referenceMedia`), and on some models video edit/extend (`video_url`). Seedance 2.5 also drafts: `draft: true` renders a cheap 480p preview, and `draft_id` set to that job's id renders the same shot at 1080p. video-depth-anything instead turns `video_url` into a grayscale depth video and takes no prompt. Which a model accepts is in its capabilities — call ottoport_list_models. Blocks on the queue and returns a video URL, unless `webhook_url` is given.",
|
|
309
344
|
inputSchema: {
|
|
310
345
|
type: "object",
|
|
311
346
|
properties: {
|
|
@@ -320,9 +355,13 @@ const TOOLS = [
|
|
|
320
355
|
image_url: { type: "string", description: "First frame, for image-to-video." },
|
|
321
356
|
last_frame_url: { type: "string", description: "Final frame. With `image_url`, this is first-and-last-frame generation." },
|
|
322
357
|
reference_image_urls: { type: "array", items: { type: "string" }, description: "Subjects the shot carries through — a character, a product, a style — without being a frame of it." },
|
|
358
|
+
reference_video_urls: { type: "array", items: { type: "string" }, description: "Reference videos — motion, a performance, a scene — on models listing `referenceMedia` (Seedance 2.x, MiniMax H3, Wan 3.0). Billed per second of reference video where the upstream charges for it. To change one clip rather than draw on it, send it as `video_url` on a model listing video-edit." },
|
|
359
|
+
reference_audio_urls: { type: "array", items: { type: "string" }, description: "Reference audio — a voice, a beat — beside reference images or videos, on the same models. Not billed." },
|
|
323
360
|
keyframes: { type: "array", items: { type: "object", properties: { image_url: { type: "string" }, time: { type: "number", description: "Seconds from the start; omit to spread evenly." } }, required: ["image_url"] }, description: "Images pinned in time, for keyframes-to-video." },
|
|
324
361
|
video_url: { type: "string", description: "A video to edit or extend, on models listing video-edit / video-extend — or the only input to video-depth-anything. Edits and depth bill per second of this video (at most 60s)." },
|
|
325
362
|
video_task: { type: "string", enum: ["edit", "extend"], description: "edit (default) restyles video_url; extend continues it for `duration` more seconds." },
|
|
363
|
+
draft: { type: "boolean", description: "Seedance 2.5 only: render a 480p preview at about a quarter of the 1080p price, to check the shot before paying for it. Text, image and reference-image requests; no `resolution`." },
|
|
364
|
+
draft_id: { type: "string", description: "Seedance 2.5 only: the job id of a finished draft, to render that same shot at 1080p. Send `model` and this alone — prompt, inputs and duration come from the draft. Within seven days of the draft." },
|
|
326
365
|
webhook_url: { type: "string", description: "Public HTTPS endpoint to POST the finished job to (events video.completed / video.failed, signed with HMAC-SHA256). With it the tool returns the job id at once instead of waiting." },
|
|
327
366
|
webhook_secret: { type: "string", description: "Signing secret for webhook_url. Omit to use the server's OTTOPORT_WEBHOOK_SECRET." },
|
|
328
367
|
},
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "ottoport",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.10.0",
|
|
4
4
|
"description": "Claude Code plugin, CLI and MCP server for OttoPort — one OpenAI-compatible API for every LLM, image, video, and speech model.",
|
|
5
5
|
"homepage": "https://ottoport.ai",
|
|
6
6
|
"repository": {
|
package/skills/ottoport/SKILL.md
CHANGED
|
@@ -17,7 +17,7 @@ and do not call the HTTP API, when a tool covers the job:
|
|
|
17
17
|
| `ottoport_list_models` | The catalog: what each model is for, and its rate in credits and USD. Filter with `modality`, or pass `model` for ONE model in full — its modes, its resolution tiers with prices, and every parameter it reads. |
|
|
18
18
|
| `ottoport_chat` | `prompt`, plus optional `model`, `system`, `temperature`, `max_tokens`. |
|
|
19
19
|
| `ottoport_generate_image` | `prompt`, plus optional `model`, `size` or `resolution`, `quality` (GPT Image only), `aspect_ratio`, `n`, `seed`, `image_url` (one reference), `reference_image_urls` (several). Layer decomposition too: `model` `qwen-image-layered` or `seedream-5.0-layers` with `image_url` alone (prompt optional, `num_layers` on Qwen) returns one URL per RGBA layer. |
|
|
20
|
-
| `ottoport_generate_video` | `prompt`, plus optional `model`, `duration`, `resolution`, `aspect_ratio`, `seed`, `image_url` (first frame), `last_frame_url`, `reference_image_urls`, `webhook_url` (returns the job id at once and POSTs the result there; the secret comes from `webhook_secret` or the server's `OTTOPORT_WEBHOOK_SECRET`). |
|
|
20
|
+
| `ottoport_generate_video` | `prompt`, plus optional `model`, `duration`, `resolution`, `aspect_ratio`, `seed`, `image_url` (first frame), `last_frame_url`, `reference_image_urls`, `reference_video_urls` / `reference_audio_urls` (models listing `referenceMedia`), `video_url` (edit / extend), `draft` / `draft_id` (Seedance 2.5: a 480p preview, then the same shot at 1080p from the preview's job id), `webhook_url` (returns the job id at once and POSTs the result there; the secret comes from `webhook_secret` or the server's `OTTOPORT_WEBHOOK_SECRET`). |
|
|
21
21
|
| `ottoport_video_webhook` | `job_id`, plus `action` `status` (every delivery attempt and the endpoint's response) or `retry` (redeliver a finished job's webhook now). |
|
|
22
22
|
| `ottoport_generate_speech` | `prompt` (the text), plus optional `model`, `voice`, `format`. |
|
|
23
23
|
| `ottoport_generate_music` | `prompt`, plus optional `model`, `duration`, `format`. |
|
|
@@ -36,7 +36,7 @@ prints each one's menu beside its price:
|
|
|
36
36
|
| text-to-video | `prompt` alone |
|
|
37
37
|
| image-to-video | `image_url` — the first frame |
|
|
38
38
|
| first-and-last-frame | `image_url` **and** `last_frame_url` |
|
|
39
|
-
| reference-to-video | `reference_image_urls` — a character, a product, a style the shot carries through without either being a frame of it |
|
|
39
|
+
| reference-to-video | `reference_image_urls` — a character, a product, a style the shot carries through without either being a frame of it; plus `reference_video_urls` and `reference_audio_urls` (motion, a performance, a voice) on models whose capabilities list `referenceMedia` — Seedance 2.x, MiniMax H3, Wan 3.0 |
|
|
40
40
|
|
|
41
41
|
Images are the same idea with one axis: `image_url` for a single edit, or
|
|
42
42
|
`reference_image_urls` for several at once, up to that model's limit.
|