videodraft 0.25.2 → 0.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -24,8 +24,8 @@ function readVersionFromDisk() {
24
24
  }
25
25
  }
26
26
  function resolveVersion() {
27
- if ("0.25.2") {
28
- return "0.25.2";
27
+ if ("0.26.0") {
28
+ return "0.26.0";
29
29
  }
30
30
  return readVersionFromDisk();
31
31
  }
@@ -1756,507 +1756,6 @@ ${fmt.bold(ctxOut, "Open this URL to log in:")}
1756
1756
  });
1757
1757
  }
1758
1758
 
1759
- // src/commands/account.ts
1760
- function registerAccountCommands(program) {
1761
- program.command("credits").description("Show your credit balance").action(async function() {
1762
- const ctx = buildContext(this);
1763
- const balance = await ctx.client.callTool("get_credits_balance");
1764
- emit(ctx.out, balance, (o) => {
1765
- kv(o, [
1766
- ["Plan", balance?.planId],
1767
- ["Available credits", balance?.availableCredits],
1768
- ["Monthly allowance", balance?.totalCreditsMonthly],
1769
- ["Used this month", balance?.monthlyCreditsUsed],
1770
- ["Bonus credits", balance?.bonusCredits],
1771
- ["Bonus expiry", balance?.bonusCreditsExpiry],
1772
- ["Last monthly reset", balance?.lastMonthlyReset],
1773
- ["Next monthly reset", balance?.nextMonthlyReset]
1774
- ]);
1775
- });
1776
- });
1777
- program.command("costs [model]").description(
1778
- "Show credit costs. Pass a model id, or an image display name, plus settings for an exact estimate"
1779
- ).option("--type <type>", "image | video | audio").option("--duration <seconds>", "video/audio duration in seconds").option(
1780
- "--length <seconds>",
1781
- "ElevenLabs Music output length in seconds (for a composition plan, the sum of its sections)"
1782
- ).option(
1783
- "--chars <n>",
1784
- 'character count (ElevenLabs Dialogue, or voiceover TTS via model id "voiceover")'
1785
- ).option("--resolution <res>", 'e.g. "720p", "1080p", "1K", "2K"').option("--quality <tier>", 'e.g. "standard", "pro", "fast"').option("--ar <ratio>", "image aspect ratio for accurate credit quotes").option(
1786
- "--rendering-speed <tier>",
1787
- 'image speed/cost tier, e.g. Ideogram V4 "Turbo"/"Balanced"/"Quality"'
1788
- ).option("--audio", "include native model audio in the estimate").option("--no-audio", "exclude native model audio").option(
1789
- "--voice-control",
1790
- "Kling element voice_id pricing (V3 Standard/Pro only; O3/4K are unavailable)"
1791
- ).option(
1792
- "--allow-real-people",
1793
- "Seedance 2.x: estimate the higher tier-specific Fal rate (the default, matching AI Studio)"
1794
- ).option(
1795
- "--no-allow-real-people",
1796
- "Seedance 2.x: estimate the lower Byteplus-only rate"
1797
- ).option(
1798
- "--ref-images <n>",
1799
- "input/reference image count for MiniMax H3, MiniMax H3 Max, or Grok 1.5"
1800
- ).option(
1801
- "--ref-video-seconds <seconds>",
1802
- "MiniMax H3 / H3 Max combined reference-video duration"
1803
- ).option("--num <n>", "image batch size").option(
1804
- "--width <px>",
1805
- "Topaz: source width (with --height gives an exact quote)"
1806
- ).option("--height <px>", "Topaz: source height").option(
1807
- "--source-fps <n>",
1808
- "Topaz video upscale: source frame rate (>30 bills the 60fps rate)"
1809
- ).option("--scale <factor>", 'Topaz: "1x" | "2x" | "4x" or a factor 1-4').option("--fps <n>", "Topaz: delivered / target frame rate").option(
1810
- "--mode <mode>",
1811
- "Topaz upscale mode: generative | precision | creative"
1812
- ).option(
1813
- "--topaz-model <name>",
1814
- 'Topaz model inside the mode, or "Apollo" | "Chronos" | "Aion" for interpolation'
1815
- ).option("--slowdown <1-8>", "Topaz interpolate: slow-motion factor").action(async function(model) {
1816
- const ctx = buildContext(this);
1817
- const opts = this.opts();
1818
- const result = await ctx.client.callTool(
1819
- "get_model_costs",
1820
- compact({
1821
- model_id: model,
1822
- type: opts.type,
1823
- duration_seconds: opts.duration ? Number(opts.duration) : void 0,
1824
- length_seconds: opts.length ? Number(opts.length) : void 0,
1825
- characters: opts.chars ? Number(opts.chars) : void 0,
1826
- resolution: opts.resolution,
1827
- quality: opts.quality,
1828
- aspect_ratio: opts.ar,
1829
- rendering_speed: opts.renderingSpeed,
1830
- generate_audio: opts.audio,
1831
- voice_control: opts.voiceControl,
1832
- allow_real_people: opts.allowRealPeople,
1833
- reference_image_count: opts.refImages ? Number(opts.refImages) : void 0,
1834
- reference_video_duration_seconds: opts.refVideoSeconds ? Number(opts.refVideoSeconds) : void 0,
1835
- num_images: opts.num ? Number(opts.num) : void 0,
1836
- source_width: opts.width ? Number(opts.width) : void 0,
1837
- source_height: opts.height ? Number(opts.height) : void 0,
1838
- source_fps: opts.sourceFps ? Number(opts.sourceFps) : void 0,
1839
- scale: opts.scale,
1840
- target_resolution: opts.resolution && /^(720p|1080p|4k)$/i.test(opts.resolution) ? opts.resolution.toLowerCase() : void 0,
1841
- target_fps: opts.fps ? Number(opts.fps) : void 0,
1842
- mode: opts.mode,
1843
- topaz_model: opts.topazModel,
1844
- slowdown_factor: opts.slowdown ? Number(opts.slowdown) : void 0
1845
- })
1846
- );
1847
- emit(ctx.out, result);
1848
- });
1849
- program.command("models [kind]").description(
1850
- "List available models: image | video | audio | 3d | voices | styles (default: image + video + audio)"
1851
- ).option(
1852
- "--category <name>",
1853
- "video only: generation | video_edit | motion_control | avatar_lipsync | upscale"
1854
- ).action(async function(kind) {
1855
- const ctx = buildContext(this);
1856
- const opts = this.opts();
1857
- const wanted = kind ?? "all";
1858
- if (!["all", "image", "video", "audio", "3d", "voices", "styles"].includes(
1859
- wanted
1860
- )) {
1861
- throw new UsageError(
1862
- `Unknown model kind "${wanted}". Use image, video, audio, 3d, voices, or styles.`
1863
- );
1864
- }
1865
- const videoCategories = /* @__PURE__ */ new Set([
1866
- "generation",
1867
- "video_edit",
1868
- "motion_control",
1869
- "avatar_lipsync",
1870
- "upscale"
1871
- ]);
1872
- if (opts.category && !videoCategories.has(opts.category)) {
1873
- throw new UsageError(
1874
- `Unknown video category "${opts.category}". Use generation, video_edit, motion_control, avatar_lipsync, or upscale.`
1875
- );
1876
- }
1877
- if (opts.category && wanted !== "video" && wanted !== "all") {
1878
- throw new UsageError(
1879
- "--category is only valid for the video model catalog."
1880
- );
1881
- }
1882
- const result = {};
1883
- if (wanted === "image" || wanted === "all") {
1884
- result.image = await ctx.client.callTool("list_available_image_models");
1885
- }
1886
- if (wanted === "video" || wanted === "all") {
1887
- const video = await ctx.client.callTool(
1888
- "list_available_video_models"
1889
- );
1890
- result.video = opts.category ? {
1891
- ...video,
1892
- // A card may belong to more than one category — gemini-omni-1.1-flash
1893
- // is both the generation default and the preferred video_edit
1894
- // model — and declares that in `categories`. Match either field so
1895
- // dual-category cards surface under both.
1896
- models: (video?.models ?? []).filter(
1897
- (model) => Array.isArray(model.categories) ? model.categories.includes(opts.category) : model.category === opts.category
1898
- ),
1899
- selected_category: opts.category
1900
- } : video;
1901
- }
1902
- if (wanted === "audio" || wanted === "all") {
1903
- result.audio = await ctx.client.callTool("list_available_audio_models");
1904
- }
1905
- if (wanted === "3d") {
1906
- result["3d"] = await ctx.client.callTool("get_3d_models");
1907
- }
1908
- if (wanted === "voices") {
1909
- result.voices = await ctx.client.callTool("list_available_voices");
1910
- }
1911
- if (wanted === "styles") {
1912
- result.styles = await ctx.client.callTool("list_available_styles");
1913
- }
1914
- emit(ctx.out, result, (o) => {
1915
- for (const [section, payload] of Object.entries(result)) {
1916
- const models = Array.isArray(payload) ? payload : payload?.models ?? payload?.voices ?? payload?.styles ?? [];
1917
- process.stdout.write(`
1918
- ${section.toUpperCase()}
1919
- `);
1920
- if (!Array.isArray(models) || models.length === 0) {
1921
- process.stdout.write(JSON.stringify(payload, null, 2) + "\n");
1922
- continue;
1923
- }
1924
- if (section === "3d") {
1925
- table(
1926
- o,
1927
- ["model", "input", "default credits", "Fal endpoint"],
1928
- models.flatMap(
1929
- (model) => (model.inputs ?? []).map((input) => [
1930
- String(model.id ?? ""),
1931
- String(input.input_mode ?? ""),
1932
- String(input.default_credits ?? ""),
1933
- String(input.endpoint_id ?? "")
1934
- ])
1935
- )
1936
- );
1937
- const rig = payload?.rigging;
1938
- if (rig)
1939
- note(
1940
- o,
1941
- `Humanoid rigging: ${rig.credits} credits; preset animation adds ${rig.animation_addon_credits} credits.`
1942
- );
1943
- note(
1944
- o,
1945
- "Use --json for exact options, defaults, input limits, and pricing notes. Use generate 3d --estimate for your selected recipe and BYOK status."
1946
- );
1947
- continue;
1948
- }
1949
- const rows = models.map((m) => [
1950
- String(m.id ?? m.model_id ?? m.voice_id ?? ""),
1951
- String(m.name ?? "").slice(0, 40),
1952
- String(m.category ?? ""),
1953
- String(m.tool ?? ""),
1954
- String(m.credit_cost ?? m.cost ?? m.pricing?.summary ?? "")
1955
- ]);
1956
- table(
1957
- o,
1958
- section === "video" ? ["id", "name", "category", "tool", "cost"] : ["id", "name", "cost"],
1959
- section === "video" ? rows : rows.map((row) => [row[0] ?? "", row[1] ?? "", row[4] ?? ""])
1960
- );
1961
- }
1962
- });
1963
- });
1964
- program.command("workspaces").description("List your workspaces (the active one is bound to your token)").action(async function() {
1965
- const ctx = buildContext(this);
1966
- const result = await ctx.client.callTool("list_workspaces");
1967
- const workspaces = result?.workspaces ?? result ?? [];
1968
- emit(ctx.out, result, (o) => {
1969
- table(
1970
- o,
1971
- ["id", "name", "role", "active"],
1972
- workspaces.map((w) => [
1973
- String(w.id ?? ""),
1974
- String(w.name ?? ""),
1975
- String(w.role ?? ""),
1976
- w.is_active || w.active ? "\u2713" : ""
1977
- ])
1978
- );
1979
- });
1980
- });
1981
- const sessions = program.command("sessions").description("AI Studio sessions");
1982
- sessions.command("list", { isDefault: true }).description("List AI Studio sessions (owned + shared)").option("--project <id>", "filter to a project's sessions").option("--name <fragment>", "filter by name (case-insensitive)").option("--limit <n>", "max rows (default 50)").option("--offset <n>", "pagination offset (default 0)").action(async function() {
1983
- const ctx = buildContext(this);
1984
- const opts = this.opts();
1985
- const result = await ctx.client.callTool(
1986
- "list_ai_studio_sessions",
1987
- compact({
1988
- project_id: opts.project,
1989
- name: opts.name,
1990
- limit: opts.limit ? Number(opts.limit) : void 0,
1991
- offset: opts.offset ? Number(opts.offset) : void 0
1992
- })
1993
- );
1994
- const rows = result?.sessions ?? result ?? [];
1995
- emit(ctx.out, result, (o) => {
1996
- if (!Array.isArray(rows)) return;
1997
- table(
1998
- o,
1999
- ["id", "name", "role", "items", "created"],
2000
- rows.map((s) => [
2001
- String(s.id ?? s.session_id ?? ""),
2002
- String(s.name ?? "").slice(0, 40),
2003
- String(s.role ?? (s.is_owner === false ? "member" : "owner")),
2004
- `${s.image_count ?? 0}i/${s.video_count ?? 0}v/${s.sound_count ?? 0}s`,
2005
- String(s.created_at ?? s.createdAt ?? "").slice(0, 19)
2006
- ])
2007
- );
2008
- });
2009
- });
2010
- sessions.command("current").description("Show the current scope's automatic AI Studio session").action(async function() {
2011
- const ctx = buildContext(this);
2012
- const knownIds = /* @__PURE__ */ new Set();
2013
- let listed = false;
2014
- if (ctx.session) {
2015
- try {
2016
- for (let offset = 0; offset < 2e3; ) {
2017
- const page = await ctx.client.callTool(
2018
- "list_ai_studio_sessions",
2019
- { limit: 200, offset }
2020
- );
2021
- const pageRows = page?.sessions ?? [];
2022
- for (const row of pageRows) {
2023
- const id = row?.id ?? row?.session_id;
2024
- if (typeof id === "string") knownIds.add(id);
2025
- }
2026
- if (pageRows.length < 200) break;
2027
- offset += pageRows.length;
2028
- }
2029
- listed = true;
2030
- } catch {
2031
- listed = false;
2032
- }
2033
- }
2034
- const info = ctx.session?.describe();
2035
- const record = info?.record ?? null;
2036
- const active = Boolean(record) && !info?.expired;
2037
- const sessionId = active ? info?.sessionId ?? null : null;
2038
- const created = Boolean(sessionId) && listed && knownIds.has(sessionId);
2039
- const result = {
2040
- enabled: Boolean(ctx.session),
2041
- scope: info?.scope ?? null,
2042
- active,
2043
- expired: Boolean(info?.expired),
2044
- session_id: sessionId,
2045
- /** False until naming or the first generation creates the row. */
2046
- session_created: created,
2047
- url: sessionId && created ? `${ctx.baseUrl}/ai-studio?session=${sessionId}` : null,
2048
- created_at: active ? record?.createdAt ?? null : null,
2049
- last_used_at: active ? record?.lastUsedAt ?? null : null,
2050
- file: info?.file ?? null,
2051
- note: ctx.session ? active ? created ? "Project-less generations in this scope share this AI Studio session. `videodraft sessions reset` starts a new one; --session <id> pins a specific session." : "Session id reserved for this scope; name it with `videodraft sessions name <title>` or generate the first asset to create it. `videodraft sessions reset` starts a new one; --session <id> pins a specific session." : info?.expired ? "The previous session idled out (12h); the next command starts a new one. --session <id> pins a specific session instead." : "No session yet; the next command creates one. --session <id> pins a specific session instead." : 'Disabled via VIDEODRAFT_NO_SESSION; generations fall back to the shared "Agent (MCP)" session unless --session is passed.'
2052
- };
2053
- emit(ctx.out, result, () => {
2054
- const state = !result.enabled ? "disabled" : result.active ? `${result.session_created ? "active" : "reserved"} session=${result.session_id ?? "?"}` : result.expired ? "expired" : "none yet";
2055
- process.stdout.write(
2056
- `${state} scope=${result.scope ?? "-"}
2057
- ${result.url ? `${result.url}
2058
- ` : ""}${result.note}
2059
- `
2060
- );
2061
- });
2062
- });
2063
- sessions.command("name <name>").description("Name the automatic AI Studio session before first use").action(async function(name) {
2064
- if (process.env.VIDEODRAFT_SESSION?.trim()) {
2065
- throw new UsageError(
2066
- "Cannot name the automatic session while VIDEODRAFT_SESSION is set. Unset it first, or rename the pinned session in AI Studio."
2067
- );
2068
- }
2069
- const ctx = buildContext(this);
2070
- const result = await ctx.client.callTool(
2071
- "name_current_ai_studio_session",
2072
- { name }
2073
- );
2074
- emit(ctx.out, result, () => {
2075
- const session = result?.session ?? result;
2076
- process.stdout.write(
2077
- `${String(session?.name ?? name)} ${String(session?.id ?? result?.session_id ?? "")}
2078
- `
2079
- );
2080
- });
2081
- });
2082
- sessions.command("reset").description(
2083
- "Forget this connection session so the next generation starts a fresh AI Studio session (--all: every scope)"
2084
- ).option("--all", "reset every stored connection session").action(async function() {
2085
- const ctx = buildContext(this);
2086
- const opts = this.opts();
2087
- let removed = 0;
2088
- let declined = false;
2089
- if (opts.all) {
2090
- removed = resetAllConnectionSessions();
2091
- } else if (ctx.session) {
2092
- const had = Boolean(ctx.session.describe().record);
2093
- const ran = ctx.session.reset();
2094
- removed = had && ran ? 1 : 0;
2095
- declined = had && !ran;
2096
- }
2097
- const result = {
2098
- reset: removed,
2099
- declined,
2100
- scope: opts.all ? "all" : ctx.session?.describe().scope ?? null
2101
- };
2102
- emit(ctx.out, result, () => {
2103
- process.stdout.write(
2104
- removed > 0 ? `Reset ${removed} connection session${removed === 1 ? "" : "s"}; the next generation starts a new AI Studio session.
2105
- ` : declined ? "Could not reset: the session store is locked or unwritable; the session may still be in use.\n" : "Nothing to reset.\n"
2106
- );
2107
- });
2108
- });
2109
- sessions.command("create <name>").description(
2110
- "Create an AI Studio session (reuse its id across standalone generations)"
2111
- ).option("--project <id>", "attach the session to a project").action(async function(name) {
2112
- const ctx = buildContext(this);
2113
- const result = await ctx.client.callTool(
2114
- "create_ai_studio_session",
2115
- compact({ name, project_id: this.opts().project })
2116
- );
2117
- emit(ctx.out, result, (o) => {
2118
- const id = result?.session?.id ?? result?.session_id ?? result?.id;
2119
- process.stdout.write(`${id ?? JSON.stringify(result)}
2120
- `);
2121
- });
2122
- });
2123
- }
2124
-
2125
- // src/commands/projects.ts
2126
- import { spawn as spawn2 } from "child_process";
2127
- function registerProjectCommands(program) {
2128
- const projects = program.command("projects").description("List and manage projects");
2129
- projects.command("list", { isDefault: true }).description("List projects").option("--limit <n>", "max projects (default 50)").option("--offset <n>", "pagination offset").option("--favorites", "only favorited projects").action(async function() {
2130
- const ctx = buildContext(this);
2131
- const opts = this.opts();
2132
- const result = await ctx.client.callTool(
2133
- "list_projects",
2134
- compact({
2135
- limit: opts.limit ? Number(opts.limit) : void 0,
2136
- offset: opts.offset ? Number(opts.offset) : void 0,
2137
- favorites_only: opts.favorites || void 0
2138
- })
2139
- );
2140
- const rows = result?.projects ?? [];
2141
- emit(ctx.out, result, (o) => {
2142
- table(
2143
- o,
2144
- ["id", "title", "status", "modified"],
2145
- rows.map((p) => [
2146
- String(p.id ?? ""),
2147
- String(p.title ?? "Untitled").slice(0, 44),
2148
- String(p.status ?? ""),
2149
- String(p.lastModified ?? "").slice(0, 19)
2150
- ])
2151
- );
2152
- });
2153
- });
2154
- projects.command("get <project_id>").description("Fetch a project (summary view; --raw for the editable blob)").option("--raw", "return the raw editable JSON blob").action(async function(projectId) {
2155
- const ctx = buildContext(this);
2156
- const result = await ctx.client.callTool("get_project", {
2157
- project_id: projectId,
2158
- view: this.opts().raw ? "raw" : "summary"
2159
- });
2160
- emit(ctx.out, result);
2161
- });
2162
- projects.command("delete <project_id>").description("Permanently delete a project (cannot be undone)").option("--yes", "skip the confirmation").action(async function(projectId) {
2163
- const ctx = buildContext(this);
2164
- if (!this.opts().yes) {
2165
- throw new CliError(
2166
- "Refusing to delete without --yes. Deletion is permanent for every collaborator.",
2167
- EXIT.USAGE
2168
- );
2169
- }
2170
- const result = await ctx.client.callTool("delete_project", {
2171
- project_id: projectId
2172
- });
2173
- emit(
2174
- ctx.out,
2175
- result,
2176
- (o) => note(o, fmt.green(o, `Deleted project ${projectId}.`))
2177
- );
2178
- });
2179
- projects.command("favorite <project_id>").description("Star or unstar a project").option("--remove", "unstar instead of star").action(async function(projectId) {
2180
- const ctx = buildContext(this);
2181
- const favorite = !this.opts().remove;
2182
- const result = await ctx.client.callTool("set_project_favorite", {
2183
- project_id: projectId,
2184
- favorite
2185
- });
2186
- emit(
2187
- ctx.out,
2188
- result,
2189
- (o) => note(o, favorite ? "Starred." : "Unstarred.")
2190
- );
2191
- });
2192
- projects.command("open <project_id>").description("Open the project in your browser").action(async function(projectId) {
2193
- const ctx = buildContext(this);
2194
- const result = await ctx.client.callTool("get_project", {
2195
- project_id: projectId,
2196
- view: "summary"
2197
- });
2198
- const url = result?.urls?.storyboard ?? result?.urls?.script ?? `${ctx.baseUrl}/projects/${projectId}`;
2199
- const cmd = process.platform === "darwin" ? "open" : process.platform === "win32" ? "start" : "xdg-open";
2200
- try {
2201
- const child = spawn2(cmd, [url], {
2202
- stdio: "ignore",
2203
- detached: true,
2204
- shell: process.platform === "win32"
2205
- });
2206
- child.on("error", () => {
2207
- });
2208
- child.unref();
2209
- } catch {
2210
- }
2211
- emit(ctx.out, { url }, (o) => note(o, url));
2212
- });
2213
- const checkpoint = program.command("checkpoint").description("Project version checkpoints");
2214
- checkpoint.command("create <project_id>").description("Snapshot the project as a new version").option("--name <name>", "version name").option("--description <text>", "version description").action(async function(projectId) {
2215
- const ctx = buildContext(this);
2216
- const opts = this.opts();
2217
- const result = await ctx.client.callTool(
2218
- "create_project_checkpoint",
2219
- compact({
2220
- project_id: projectId,
2221
- name: opts.name,
2222
- description: opts.description
2223
- })
2224
- );
2225
- emit(ctx.out, result);
2226
- });
2227
- checkpoint.command("list <project_id>").description("List a project's checkpoints").action(async function(projectId) {
2228
- const ctx = buildContext(this);
2229
- const result = await ctx.client.callTool("list_project_checkpoints", {
2230
- project_id: projectId
2231
- });
2232
- emit(ctx.out, result);
2233
- });
2234
- checkpoint.command("restore <project_id> [version_number]").description(
2235
- "Restore a project to a saved version (current state is snapshotted first)"
2236
- ).option(
2237
- "--checkpoint-id <id>",
2238
- "restore by checkpoint UUID instead of version number"
2239
- ).action(async function(projectId, versionNumber) {
2240
- const ctx = buildContext(this);
2241
- const opts = this.opts();
2242
- if (!versionNumber && !opts.checkpointId) {
2243
- throw new CliError(
2244
- "Pass a version_number or --checkpoint-id.",
2245
- EXIT.USAGE
2246
- );
2247
- }
2248
- const result = await ctx.client.callTool(
2249
- "restore_project_checkpoint",
2250
- compact({
2251
- project_id: projectId,
2252
- version_number: versionNumber ? Number(versionNumber) : void 0,
2253
- checkpoint_id: opts.checkpointId
2254
- })
2255
- );
2256
- emit(ctx.out, result);
2257
- });
2258
- }
2259
-
2260
1759
  // src/commands/generate.ts
2261
1760
  import fs9 from "fs";
2262
1761
  import path8 from "path";
@@ -3950,12 +3449,23 @@ var ELEVENLABS_MUSIC_MODELS = [
3950
3449
  ];
3951
3450
  var ELEVENLABS_MUSIC_V1 = "elevenlabs-music-v1";
3952
3451
  var MUSIC_REFERENCE_STRENGTHS = ["low", "medium", "high", "xhigh"];
3452
+ var VOICEOVER_MODES = ["standard", "turbo"];
3953
3453
  var UUID_RE3 = /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i;
3954
3454
  function normalizeGeminiOmniResolutionOption(value) {
3955
3455
  if (typeof value !== "string") return void 0;
3956
3456
  const normalized = value.trim().toLowerCase();
3957
3457
  return normalized === "360p" || normalized === "720p" || normalized === "1080p" || normalized === "4k" ? normalized : void 0;
3958
3458
  }
3459
+ function parseVoiceoverMode(value, flag) {
3460
+ if (value === void 0) return void 0;
3461
+ if (!VOICEOVER_MODES.includes(value)) {
3462
+ throw new CliError(
3463
+ `${flag} must be one of: ${VOICEOVER_MODES.join(", ")}.`,
3464
+ EXIT.USAGE
3465
+ );
3466
+ }
3467
+ return value;
3468
+ }
3959
3469
  function inferDubMediaType(source, explicit) {
3960
3470
  if (explicit) {
3961
3471
  if (explicit === "audio" || explicit === "video") return explicit;
@@ -4428,10 +3938,23 @@ async function handleAsyncJob(ctx, submitted, options) {
4428
3938
  throw err;
4429
3939
  }
4430
3940
  }
3941
+ function parseImageColor(value) {
3942
+ if (!/^#?[0-9a-f]{6}$/i.test(value))
3943
+ throw new CliError(
3944
+ "Colors must be six hexadecimal digits, e.g. #7f3f98.",
3945
+ EXIT.USAGE
3946
+ );
3947
+ const hex = value.replace(/^#/, "");
3948
+ return {
3949
+ r: parseInt(hex.slice(0, 2), 16),
3950
+ g: parseInt(hex.slice(2, 4), 16),
3951
+ b: parseInt(hex.slice(4, 6), 16)
3952
+ };
3953
+ }
4431
3954
  function registerGenerateCommands(program) {
4432
3955
  const generate = program.command("generate").description("Generate images, video, audio and 3D assets");
4433
3956
  register3DGenerationCommand(generate);
4434
- generate.command("image <prompt...>").description("Generate an image (async; waits by default)").option(
3957
+ generate.command("image [prompt...]").description("Generate an image (async; waits by default)").option(
4435
3958
  "--model <id|name>",
4436
3959
  "image model id or display name (default nano-banana-2); run `videodraft models image`"
4437
3960
  ).option("--ar <ratio>", 'aspect ratio, e.g. "16:9"').option("--resolution <res>", 'e.g. "1K", "2K", "4K"').option(
@@ -4440,6 +3963,17 @@ function registerGenerateCommands(program) {
4440
3963
  ).option(
4441
3964
  "--rendering-speed <tier>",
4442
3965
  'Ideogram speed/cost tier, e.g. V4 "Turbo"/"Balanced"/"Quality"'
3966
+ ).option("--temperature <n>", "Nano Banana Pro/2 creativity (0-2)").option(
3967
+ "--google-search-grounding <true|false>",
3968
+ "Nano Banana Pro/2 Google Search grounding"
3969
+ ).option("--horizontal-angle <degrees>", "Qwen horizontal rotation (0-360)").option("--vertical-angle <degrees>", "Qwen elevation (-30 to 90)").option("--zoom <n>", "Qwen zoom (0-10)").option(
3970
+ "--recraft-color <hex>",
3971
+ "Recraft RGB palette color, e.g. #7f3f98 (repeat up to five)",
3972
+ collect,
3973
+ []
3974
+ ).option(
3975
+ "--recraft-background <hex>",
3976
+ "Recraft background color, e.g. #ffffff"
4443
3977
  ).option("--num <n>", "variations of this prompt in one call (1-4)").option(
4444
3978
  "--seed <n>",
4445
3979
  "seed (supported models only, e.g. Flux, Ideogram V4)"
@@ -4465,6 +3999,34 @@ function registerGenerateCommands(program) {
4465
3999
  const ctx = buildContext(this);
4466
4000
  const opts = this.opts();
4467
4001
  const prompt = promptWords.join(" ");
4002
+ const imageOptions = {
4003
+ temperature: optionalRangedNumber(
4004
+ opts.temperature,
4005
+ "--temperature",
4006
+ 0,
4007
+ 2
4008
+ ),
4009
+ google_search_grounding: optionalBooleanChoice(
4010
+ opts.googleSearchGrounding,
4011
+ "--google-search-grounding"
4012
+ ),
4013
+ horizontal_angle: optionalRangedNumber(
4014
+ opts.horizontalAngle,
4015
+ "--horizontal-angle",
4016
+ 0,
4017
+ 360
4018
+ ),
4019
+ vertical_angle: optionalRangedNumber(
4020
+ opts.verticalAngle,
4021
+ "--vertical-angle",
4022
+ -30,
4023
+ 90
4024
+ ),
4025
+ zoom: optionalRangedNumber(opts.zoom, "--zoom", 0, 10),
4026
+ recraft_colors: opts.recraftColor?.length ? opts.recraftColor.map(parseImageColor) : void 0,
4027
+ recraft_background_color: opts.recraftBackground ? parseImageColor(opts.recraftBackground) : void 0
4028
+ };
4029
+ const imageSeed = optionalSeed(opts.seed);
4468
4030
  if (opts.estimate) {
4469
4031
  await printEstimate(ctx, {
4470
4032
  // Cost lookup needs a concrete model. Keep runtime generation
@@ -4497,6 +4059,7 @@ function registerGenerateCommands(program) {
4497
4059
  const submitted = await ctx.client.callTool(
4498
4060
  "generate_image",
4499
4061
  compact({
4062
+ ...imageOptions,
4500
4063
  prompt,
4501
4064
  model: opts.model,
4502
4065
  aspect_ratio: opts.ar,
@@ -4504,7 +4067,7 @@ function registerGenerateCommands(program) {
4504
4067
  quality: opts.quality,
4505
4068
  rendering_speed: opts.renderingSpeed,
4506
4069
  num_images: opts.num ? Number(opts.num) : void 0,
4507
- seed: opts.seed ? Number(opts.seed) : void 0,
4070
+ seed: imageSeed,
4508
4071
  reference_images: refs.length > 0 ? refs : void 0,
4509
4072
  video_url: videoRef,
4510
4073
  style: opts.style,
@@ -4527,7 +4090,7 @@ function registerGenerateCommands(program) {
4527
4090
  "video model id (task-aware when omitted; Grok 1.5 supports text, first frame, or 1-7 image refs)"
4528
4091
  ).option("--ar <ratio>", 'aspect ratio, e.g. "16:9", "9:16"').option("--duration <seconds>", "clip duration in seconds").option(
4529
4092
  "--auto-duration",
4530
- "Wan 3.0 only: provider selects 2-30s; reserves 30s and reconciles unused credits"
4093
+ "Wan 3.0, Seedance 2/2.5, FLUX 3 or Gemini Omni: automatic duration, reserves the model ceiling and reconciles unused credits"
4531
4094
  ).option("--resolution <res>", 'e.g. "360p", "720p", "1080p", "2K", "4K"').option(
4532
4095
  "--quality <tier>",
4533
4096
  'e.g. "mini", "fast", "standard", "quality", "pro"'
@@ -4581,7 +4144,7 @@ function registerGenerateCommands(program) {
4581
4144
  "multi-prompt segment (repeatable; Kling 3.0 / 3.0 Turbo / O3)",
4582
4145
  collect,
4583
4146
  []
4584
- ).option("--negative <text>", "negative prompt (Kling/Luma; not Wan 3.0)").option("--camera-fixed", "Seedance 1.5 Pro: lock camera motion").option(
4147
+ ).option("--cfg-scale <n>", "Kling 3.0/2.5 Pro prompt adherence (0-1)").option("--negative <text>", "negative prompt (Kling/Luma; not Wan 3.0)").option("--camera-fixed", "Seedance 1.5 Pro: lock camera motion").option(
4585
4148
  "--prompt-expansion <true|false>",
4586
4149
  "Wan 3.0 only: enable or disable prompt expansion (default true)"
4587
4150
  ).option(
@@ -4589,7 +4152,7 @@ function registerGenerateCommands(program) {
4589
4152
  "MiniMax H3 Max only: disabled, balanced (default), or quality"
4590
4153
  ).option(
4591
4154
  "--safety-checker <true|false>",
4592
- "MiniMax H3 Max only: opt into provider safety checking (off by default)"
4155
+ "MiniMax H3 Max / Happy Horse: opt into provider safety checking (off by default)"
4593
4156
  ).option(
4594
4157
  "--thinking",
4595
4158
  "Wan 3.0 only: enable provider thinking; required with --file-url/--web-url"
@@ -4665,6 +4228,24 @@ function registerGenerateCommands(program) {
4665
4228
  );
4666
4229
  }
4667
4230
  const seed = optionalSeed(opts.seed);
4231
+ const cfgScale = optionalRangedNumber(opts.cfgScale, "--cfg-scale", 0, 1);
4232
+ if (cfgScale !== void 0 && !["kling-3.0", "kling-2.5-pro"].includes(opts.model))
4233
+ throw new CliError(
4234
+ "--cfg-scale requires --model kling-3.0 or kling-2.5-pro.",
4235
+ EXIT.USAGE
4236
+ );
4237
+ if (opts.autoDuration && duration !== void 0)
4238
+ throw new CliError(
4239
+ "--auto-duration and --duration are mutually exclusive.",
4240
+ EXIT.USAGE
4241
+ );
4242
+ const isFlux3Request = ["flux-3", "flux3", "flux_3"].includes(opts.model);
4243
+ const flux3FixedFrameMode = isFlux3Request && (Boolean(opts.startImage && opts.endImage) || (opts.keyframe?.length ?? 0) > 0);
4244
+ if (isFlux3Request && opts.autoDuration && (opts.endImage || (opts.keyframe?.length ?? 0) > 0))
4245
+ throw new CliError(
4246
+ "FLUX 3 --auto-duration supports only text and first-frame generation.",
4247
+ EXIT.USAGE
4248
+ );
4668
4249
  const rawElements = parseKlingElements(opts.element ?? []);
4669
4250
  const voiceIds = opts.voiceId ?? [];
4670
4251
  const segments = parseSegments(opts.segment ?? []);
@@ -4704,7 +4285,16 @@ function registerGenerateCommands(program) {
4704
4285
  EXIT.USAGE
4705
4286
  );
4706
4287
  }
4707
- const hasWan3OnlyControls = opts.autoDuration === true || promptExpansion !== void 0 || opts.thinking === true || Boolean(opts.fileUrl) || Boolean(opts.webUrl);
4288
+ const hasWan3OnlyControls = opts.autoDuration === true && ![
4289
+ "seedance-2",
4290
+ "seedance2",
4291
+ "seedance-2.5",
4292
+ "seedance25",
4293
+ "flux-3",
4294
+ "flux3",
4295
+ "flux_3",
4296
+ "gemini-omni-1.1-flash"
4297
+ ].includes(opts.model) || promptExpansion !== void 0 || opts.thinking === true || Boolean(opts.fileUrl) || Boolean(opts.webUrl);
4708
4298
  const hasH3MaxOnlyControls = promptExpansionMode !== void 0 || safetyChecker !== void 0;
4709
4299
  const hasGeminiOmniOnlyControls = Boolean(opts.sourceVideo) || Boolean(opts.previousInteractionId) || Boolean(opts.videoTask) || opts.extend === true || referenceVideoDurations.length > 0;
4710
4300
  if (!opts.model && hasGeminiOmniOnlyControls) {
@@ -5068,9 +4658,9 @@ function registerGenerateCommands(program) {
5068
4658
  EXIT.USAGE
5069
4659
  );
5070
4660
  }
5071
- if (opts.model !== "minimax-h3-max" && hasH3MaxOnlyControls) {
4661
+ if (opts.model !== "minimax-h3-max" && (promptExpansionMode !== void 0 || safetyChecker !== void 0 && opts.model !== "happy-horse")) {
5072
4662
  throw new CliError(
5073
- "--prompt-expansion-mode and --safety-checker are supported only by --model minimax-h3-max.",
4663
+ "--prompt-expansion-mode requires --model minimax-h3-max; --safety-checker supports minimax-h3-max and happy-horse.",
5074
4664
  EXIT.USAGE
5075
4665
  );
5076
4666
  }
@@ -5256,334 +4846,819 @@ function registerGenerateCommands(program) {
5256
4846
  );
5257
4847
  }
5258
4848
  }
5259
- if (opts.estimate) {
5260
- const estimateModel = estimateVideoModel(opts, duration);
5261
- const refVideoWindow = refVideoSecondsWindow(estimateModel);
5262
- if (refVideoSeconds !== void 0 && refVideoWindow !== null && refVideoSeconds > refVideoWindow) {
5263
- throw new CliError(
5264
- `--ref-video-seconds must be from 0 to ${refVideoWindow} for ${estimateModel}.`,
5265
- EXIT.USAGE
5266
- );
5267
- }
5268
- const estimateReferenceImageCount = estimateModel === "minimax-h3" ? Array.isArray(opts.ref) ? opts.ref.length : 0 : (
5269
- // H3 Max prices reference images as pooled tokens, so send the
5270
- // count only when there is one; a plain text quote stays clean.
5271
- estimateModel === "minimax-h3-max" ? Array.isArray(opts.ref) && opts.ref.length > 0 ? opts.ref.length : void 0 : estimateModel === "grok-imagine-video-1.5" ? Array.isArray(opts.ref) && opts.ref.length > 0 ? opts.ref.length : opts.startImage ? 1 : 0 : void 0
4849
+ if (opts.estimate) {
4850
+ const estimateModel = estimateVideoModel(opts, duration);
4851
+ const refVideoWindow = refVideoSecondsWindow(estimateModel);
4852
+ if (refVideoSeconds !== void 0 && refVideoWindow !== null && refVideoSeconds > refVideoWindow) {
4853
+ throw new CliError(
4854
+ `--ref-video-seconds must be from 0 to ${refVideoWindow} for ${estimateModel}.`,
4855
+ EXIT.USAGE
4856
+ );
4857
+ }
4858
+ const estimateReferenceImageCount = estimateModel === "minimax-h3" ? Array.isArray(opts.ref) ? opts.ref.length : 0 : (
4859
+ // H3 Max prices reference images as pooled tokens, so send the
4860
+ // count only when there is one; a plain text quote stays clean.
4861
+ estimateModel === "minimax-h3-max" ? Array.isArray(opts.ref) && opts.ref.length > 0 ? opts.ref.length : void 0 : estimateModel === "grok-imagine-video-1.5" ? Array.isArray(opts.ref) && opts.ref.length > 0 ? opts.ref.length : opts.startImage ? 1 : 0 : void 0
4862
+ );
4863
+ const hasRefVideos = Array.isArray(opts.refVideo) && opts.refVideo.length > 0;
4864
+ const estimateReferenceVideoDuration = estimateModel === "minimax-h3" ? hasRefVideos ? refVideoSeconds : 0 : refVideoWindow !== null && hasRefVideos ? refVideoSeconds : void 0;
4865
+ const h3MaxMissingDurationFlags = estimateModel === "minimax-h3-max" ? [
4866
+ hasRefVideos && refVideoSeconds === void 0 ? "--ref-video-seconds" : null,
4867
+ Array.isArray(opts.refAudio) && opts.refAudio.length > 0 && refAudioSeconds === void 0 ? "--ref-audio-seconds" : null
4868
+ ].filter((flag) => flag !== null) : [];
4869
+ const h3MaxHasImageRefs = estimateModel === "minimax-h3-max" && Array.isArray(opts.ref) && opts.ref.length > 0;
4870
+ const h3MaxLowerBoundReason = h3MaxMissingDurationFlags.length > 0 || h3MaxHasImageRefs ? [
4871
+ "minimax-h3-max bills reference media as pooled tokens.",
4872
+ h3MaxMissingDurationFlags.length > 0 ? `Pass ${h3MaxMissingDurationFlags.join(" and ")} for an exact quote.` : null,
4873
+ h3MaxHasImageRefs ? "Reference images are priced on their measured pixel area, so this quote assumes 1024x1024 and the charge may be higher." : null
4874
+ ].filter(Boolean).join(" ") : void 0;
4875
+ const estimateDuration = (segments.length > 0 ? segmentDuration : duration) ?? // The generic FLUX quote assumes text/first-frame auto generation.
4876
+ // Fixed-frame modes instead default to five seconds in the MCP
4877
+ // request builder, and their mode is not sent to get_model_costs.
4878
+ (flux3FixedFrameMode ? 5 : estimateModel === "grok-imagine-video-1.5" ? Array.isArray(opts.ref) && opts.ref.length > 0 ? 8 : 6 : !opts.model && estimateModel === "google-veo3.1" ? Array.isArray(opts.ref) && opts.ref.length > 0 ? 8 : 6 : void 0);
4879
+ await printEstimate(ctx, {
4880
+ model: estimateModel,
4881
+ type: "video",
4882
+ duration: estimateDuration,
4883
+ autoDuration: opts.autoDuration === true,
4884
+ resolution: opts.resolution,
4885
+ quality: opts.quality,
4886
+ audio: opts.audio,
4887
+ referenceImageCount: estimateReferenceImageCount,
4888
+ referenceVideoDurationSeconds: estimateReferenceVideoDuration,
4889
+ referenceAudioDurationSeconds: estimateModel === "minimax-h3-max" && Array.isArray(opts.refAudio) && opts.refAudio.length > 0 ? refAudioSeconds : void 0,
4890
+ // H3 Max bills reference video and audio as pooled tokens, so a
4891
+ // quote that cannot see their durations omits a real surcharge.
4892
+ lowerBoundReason: h3MaxLowerBoundReason,
4893
+ voiceControl: voiceIds.length > 0 || rawElements.some((element) => element.voice_id),
4894
+ allowRealPeople: opts.allowRealPeople
4895
+ });
4896
+ return;
4897
+ }
4898
+ const [
4899
+ refs,
4900
+ refVideos,
4901
+ refAudios,
4902
+ startImage,
4903
+ endImage,
4904
+ sourceVideo,
4905
+ elements
4906
+ ] = await Promise.all([
4907
+ resolveRefs(ctx, opts.ref ?? []),
4908
+ resolveRefs(ctx, opts.refVideo ?? []),
4909
+ resolveRefs(ctx, opts.refAudio ?? []),
4910
+ opts.startImage ? resolveRefs(ctx, [opts.startImage]).then((r) => r[0]) : void 0,
4911
+ opts.endImage ? resolveRefs(ctx, [opts.endImage]).then((r) => r[0]) : void 0,
4912
+ opts.sourceVideo ? resolveRefs(ctx, [opts.sourceVideo]).then((r) => r[0]) : void 0,
4913
+ resolveKlingElements(ctx, rawElements)
4914
+ ]);
4915
+ const parsedKeyframes = parseKeyframes(opts.keyframe ?? []);
4916
+ const keyframes = parsedKeyframes.length > 0 ? await resolveRefs(
4917
+ ctx,
4918
+ parsedKeyframes.map((keyframe) => keyframe.source)
4919
+ ).then(
4920
+ (urls) => parsedKeyframes.map((keyframe, index) => ({
4921
+ image_url: urls[index],
4922
+ time_seconds: keyframe.time_seconds
4923
+ }))
4924
+ ) : [];
4925
+ if (!prompt && segments.length === 0 && !startImage && keyframes.length === 0 && refs.length === 0 && refVideos.length === 0 && refAudios.length === 0 && !sourceVideo && !opts.fileUrl && !opts.webUrl && !opts.previousInteractionId) {
4926
+ throw new CliError(
4927
+ "Provide a prompt, --segment, a frame, or a reference source.",
4928
+ EXIT.USAGE
4929
+ );
4930
+ }
4931
+ capture("cli_generate", {
4932
+ kind: "video",
4933
+ model: opts.model ?? "default",
4934
+ wait: opts.wait !== false
4935
+ });
4936
+ const videoEditModels = /* @__PURE__ */ new Set([
4937
+ "happy-horse-video-edit",
4938
+ "grok-imagine-video-edit"
4939
+ ]);
4940
+ const motionControlModels = /* @__PURE__ */ new Set([
4941
+ "kling-v3-motion-control",
4942
+ "kling-2.6-motion-control"
4943
+ ]);
4944
+ let toolName = "generate_video";
4945
+ let toolArgs;
4946
+ if (videoEditModels.has(opts.model)) {
4947
+ if (refVideos.length !== 1) {
4948
+ throw new CliError(
4949
+ `${opts.model} requires exactly one --ref-video source. You can also use videodraft edit video.`,
4950
+ EXIT.USAGE
4951
+ );
4952
+ }
4953
+ if (startImage || endImage || refAudios.length > 0 || segments.length > 0 || opts.ar || opts.negative || opts.cameraFixed || opts.seed || elements.length > 0 || voiceIds.length > 0) {
4954
+ throw new CliError(
4955
+ `${opts.model} does not support --start-image, --end-image, --ref-audio, --element, --voice-id, --segment, --ar, --negative, --camera-fixed, or --seed in video-edit mode. Use --ref for supported reference images.`,
4956
+ EXIT.USAGE
4957
+ );
4958
+ }
4959
+ toolName = "edit_video";
4960
+ toolArgs = {
4961
+ model: opts.model,
4962
+ prompt: prompt || void 0,
4963
+ video_url: refVideos[0],
4964
+ reference_images: refs.length > 0 ? refs : void 0,
4965
+ resolution: opts.resolution,
4966
+ quality: opts.quality,
4967
+ duration_seconds: duration,
4968
+ preserve_audio: opts.audio,
4969
+ project_id: opts.project,
4970
+ session_id: sessionArg(this, opts),
4971
+ scene_index: opts.scene !== void 0 ? Number(opts.scene) : void 0,
4972
+ shot_index: opts.shot !== void 0 ? Number(opts.shot) : void 0
4973
+ };
4974
+ } else if (motionControlModels.has(opts.model)) {
4975
+ if (!startImage || refVideos.length !== 1) {
4976
+ throw new CliError(
4977
+ `${opts.model} requires --start-image plus exactly one --ref-video motion source. You can also use videodraft edit motion.`,
4978
+ EXIT.USAGE
4979
+ );
4980
+ }
4981
+ if (endImage || refs.length > 0 || refAudios.length > 0 || segments.length > 0 || opts.ar || opts.negative || opts.cameraFixed || opts.seed || opts.resolution || voiceIds.length > 0) {
4982
+ throw new CliError(
4983
+ `${opts.model} does not support --end-image, --ref, --ref-audio, --voice-id, --segment, --ar, --negative, --camera-fixed, --seed, or --resolution in motion-control mode.`,
4984
+ EXIT.USAGE
4985
+ );
4986
+ }
4987
+ if (duration !== void 0) {
4988
+ throw new CliError(
4989
+ `${opts.model} follows the motion video duration and orientation cap; use videodraft edit motion --estimate for a duration-based estimate.`,
4990
+ EXIT.USAGE
4991
+ );
4992
+ }
4993
+ toolName = "generate_motion_control_video";
4994
+ const resolvedElement = elements[0];
4995
+ toolArgs = {
4996
+ model: opts.model,
4997
+ prompt: prompt || void 0,
4998
+ image_url: startImage,
4999
+ motion_video_url: refVideos[0],
5000
+ quality: opts.quality,
5001
+ keep_original_sound: opts.audio,
5002
+ character_orientation: elements.length > 0 ? "video" : void 0,
5003
+ duration_seconds: duration,
5004
+ project_id: opts.project,
5005
+ session_id: sessionArg(this, opts),
5006
+ scene_index: opts.scene !== void 0 ? Number(opts.scene) : void 0,
5007
+ shot_index: opts.shot !== void 0 ? Number(opts.shot) : void 0,
5008
+ element: elements.length === 1 && resolvedElement ? {
5009
+ frontal_image_url: resolvedElement.frontal_image_url,
5010
+ reference_image_urls: resolvedElement.reference_image_urls
5011
+ } : void 0
5012
+ };
5013
+ } else {
5014
+ toolArgs = {
5015
+ prompt: prompt || void 0,
5016
+ model: opts.model,
5017
+ aspect_ratio: opts.ar,
5018
+ duration_seconds: duration,
5019
+ auto_duration: opts.autoDuration ? true : void 0,
5020
+ resolution: opts.resolution,
5021
+ quality: opts.quality,
5022
+ generate_audio: opts.audio,
5023
+ allow_real_people: opts.allowRealPeople,
5024
+ start_image_url: startImage,
5025
+ end_image_url: endImage,
5026
+ reference_images: refs.length > 0 ? refs : void 0,
5027
+ video_url: sourceVideo,
5028
+ reference_videos: refVideos.length > 0 ? refVideos : void 0,
5029
+ reference_video_durations: referenceVideoDurations.length > 0 ? referenceVideoDurations : void 0,
5030
+ reference_audio: refAudios.length > 0 ? refAudios : void 0,
5031
+ file_url: opts.fileUrl,
5032
+ web_url: opts.webUrl,
5033
+ previous_interaction_id: opts.previousInteractionId,
5034
+ video_task: geminiVideoTask,
5035
+ enable_prompt_expansion: promptExpansion,
5036
+ prompt_expansion_mode: promptExpansionMode,
5037
+ enable_safety_checker: safetyChecker,
5038
+ enable_thinking: opts.thinking ? true : void 0,
5039
+ elements: elements.length > 0 ? elements : void 0,
5040
+ voice_ids: voiceIds.length > 0 ? voiceIds.map((voiceId) => voiceId.trim()) : void 0,
5041
+ multi_prompt: segments.length > 0 ? segments : void 0,
5042
+ keyframes: keyframes.length > 0 ? keyframes : void 0,
5043
+ negative_prompt: opts.negative,
5044
+ cfg_scale: cfgScale,
5045
+ camera_fixed: opts.cameraFixed ? true : void 0,
5046
+ seed,
5047
+ project_id: opts.project,
5048
+ session_id: sessionArg(this, opts),
5049
+ scene_index: opts.scene !== void 0 ? Number(opts.scene) : void 0,
5050
+ shot_index: opts.shot !== void 0 ? Number(opts.shot) : void 0
5051
+ };
5052
+ }
5053
+ const submitted = await ctx.client.callTool(toolName, compact(toolArgs));
5054
+ await handleAsyncJob(ctx, submitted, {
5055
+ wait: opts.wait !== false,
5056
+ download: opts.download,
5057
+ label: "Generating video"
5058
+ });
5059
+ });
5060
+ generate.command("audio <prompt...>").description(
5061
+ "Generate or edit audio with ByteDance Seed Audio 1.0 (synchronous)"
5062
+ ).option("--voice <id>", "preset Seed Audio voice or custom cloned voice id").option(
5063
+ "--ref-audio <url|file>",
5064
+ "reference audio for @Audio1..@Audio3 (repeatable; local files uploaded)",
5065
+ collect,
5066
+ []
5067
+ ).option(
5068
+ "--image <url|file>",
5069
+ "reference image (cannot be combined with --ref-audio)"
5070
+ ).option("--format <wav|mp3|pcm|ogg_opus>", "output format (default mp3)").option(
5071
+ "--sample-rate <hz>",
5072
+ "8000 | 16000 | 24000 | 32000 | 44100 | 48000 (default 24000)"
5073
+ ).option("--speed <0.5-2>", "playback speed multiplier (default 1)").option("--volume <0.5-2>", "volume multiplier (default 1)").option("--pitch <-12-12>", "pitch in whole semitones (default 0)").option("--project <id>", "link to a project's AI Studio session").option(
5074
+ "--session <id>",
5075
+ "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
5076
+ process.env.VIDEODRAFT_SESSION
5077
+ ).option(
5078
+ "--idempotency-key <uuid>",
5079
+ "set a stable UUID for recovery after a process interruption"
5080
+ ).option("--download <path>", "download the generated audio file").option(
5081
+ "--estimate",
5082
+ "show 19 credits/minute pricing and the 38-credit maximum reservation"
5083
+ ).action(async function(promptWords) {
5084
+ const ctx = buildContext(this);
5085
+ const opts = this.opts();
5086
+ const prompt = promptWords.join(" ").trim();
5087
+ if (!prompt) {
5088
+ throw new CliError("A Seed Audio prompt is required.", EXIT.USAGE);
5089
+ }
5090
+ if (prompt.length > SEED_AUDIO_PROMPT_MAX_CHARS) {
5091
+ throw new CliError(
5092
+ `Seed Audio prompts must be ${SEED_AUDIO_PROMPT_MAX_CHARS} characters or fewer.`,
5093
+ EXIT.USAGE
5272
5094
  );
5273
- const hasRefVideos = Array.isArray(opts.refVideo) && opts.refVideo.length > 0;
5274
- const estimateReferenceVideoDuration = estimateModel === "minimax-h3" ? hasRefVideos ? refVideoSeconds : 0 : refVideoWindow !== null && hasRefVideos ? refVideoSeconds : void 0;
5275
- const h3MaxMissingDurationFlags = estimateModel === "minimax-h3-max" ? [
5276
- hasRefVideos && refVideoSeconds === void 0 ? "--ref-video-seconds" : null,
5277
- Array.isArray(opts.refAudio) && opts.refAudio.length > 0 && refAudioSeconds === void 0 ? "--ref-audio-seconds" : null
5278
- ].filter((flag) => flag !== null) : [];
5279
- const h3MaxHasImageRefs = estimateModel === "minimax-h3-max" && Array.isArray(opts.ref) && opts.ref.length > 0;
5280
- const h3MaxLowerBoundReason = h3MaxMissingDurationFlags.length > 0 || h3MaxHasImageRefs ? [
5281
- "minimax-h3-max bills reference media as pooled tokens.",
5282
- h3MaxMissingDurationFlags.length > 0 ? `Pass ${h3MaxMissingDurationFlags.join(" and ")} for an exact quote.` : null,
5283
- h3MaxHasImageRefs ? "Reference images are priced on their measured pixel area, so this quote assumes 1024x1024 and the charge may be higher." : null
5284
- ].filter(Boolean).join(" ") : void 0;
5285
- const estimateDuration = (segments.length > 0 ? segmentDuration : duration) ?? (estimateModel === "grok-imagine-video-1.5" ? Array.isArray(opts.ref) && opts.ref.length > 0 ? 8 : 6 : !opts.model && estimateModel === "google-veo3.1" ? Array.isArray(opts.ref) && opts.ref.length > 0 ? 8 : 6 : void 0);
5095
+ }
5096
+ const rawAudioRefs = opts.refAudio ?? [];
5097
+ if (rawAudioRefs.length > SEED_AUDIO_MAX_REFERENCES) {
5098
+ throw new CliError(
5099
+ `Seed Audio accepts at most ${SEED_AUDIO_MAX_REFERENCES} --ref-audio values.`,
5100
+ EXIT.USAGE
5101
+ );
5102
+ }
5103
+ if (opts.image && rawAudioRefs.length > 0) {
5104
+ throw new CliError(
5105
+ "--image and --ref-audio cannot be used together.",
5106
+ EXIT.USAGE
5107
+ );
5108
+ }
5109
+ const outputFormat = opts.format ?? "mp3";
5110
+ if (!SEED_AUDIO_FORMATS.includes(outputFormat)) {
5111
+ throw new CliError(
5112
+ `--format must be one of: ${SEED_AUDIO_FORMATS.join(", ")}.`,
5113
+ EXIT.USAGE
5114
+ );
5115
+ }
5116
+ const sampleRate = opts.sampleRate ? Number(opts.sampleRate) : 24e3;
5117
+ if (!SEED_AUDIO_SAMPLE_RATES.includes(sampleRate)) {
5118
+ throw new CliError(
5119
+ `--sample-rate must be one of: ${SEED_AUDIO_SAMPLE_RATES.join(", ")}.`,
5120
+ EXIT.USAGE
5121
+ );
5122
+ }
5123
+ const speed = optionalRangedNumber(opts.speed, "--speed", 0.5, 2);
5124
+ const volume = optionalRangedNumber(opts.volume, "--volume", 0.5, 2);
5125
+ const pitch = optionalRangedNumber(opts.pitch, "--pitch", -12, 12, true);
5126
+ const idempotencyKey = opts.idempotencyKey ?? randomUUID4();
5127
+ if (!UUID_RE3.test(idempotencyKey)) {
5128
+ throw new CliError(
5129
+ "--idempotency-key must be a valid UUID.",
5130
+ EXIT.USAGE
5131
+ );
5132
+ }
5133
+ if (opts.estimate) {
5286
5134
  await printEstimate(ctx, {
5287
- model: estimateModel,
5288
- type: "video",
5289
- duration: estimateDuration,
5290
- autoDuration: opts.autoDuration === true,
5291
- resolution: opts.resolution,
5292
- quality: opts.quality,
5293
- audio: opts.audio,
5294
- referenceImageCount: estimateReferenceImageCount,
5295
- referenceVideoDurationSeconds: estimateReferenceVideoDuration,
5296
- referenceAudioDurationSeconds: estimateModel === "minimax-h3-max" && Array.isArray(opts.refAudio) && opts.refAudio.length > 0 ? refAudioSeconds : void 0,
5297
- // H3 Max bills reference video and audio as pooled tokens, so a
5298
- // quote that cannot see their durations omits a real surcharge.
5299
- lowerBoundReason: h3MaxLowerBoundReason,
5300
- voiceControl: voiceIds.length > 0 || rawElements.some((element) => element.voice_id),
5301
- allowRealPeople: opts.allowRealPeople
5135
+ model: "seed-audio-1.0",
5136
+ type: "audio"
5302
5137
  });
5303
5138
  return;
5304
5139
  }
5305
- const [
5306
- refs,
5307
- refVideos,
5308
- refAudios,
5309
- startImage,
5310
- endImage,
5311
- sourceVideo,
5312
- elements
5313
- ] = await Promise.all([
5314
- resolveRefs(ctx, opts.ref ?? []),
5315
- resolveRefs(ctx, opts.refVideo ?? []),
5316
- resolveRefs(ctx, opts.refAudio ?? []),
5317
- opts.startImage ? resolveRefs(ctx, [opts.startImage]).then((r) => r[0]) : void 0,
5318
- opts.endImage ? resolveRefs(ctx, [opts.endImage]).then((r) => r[0]) : void 0,
5319
- opts.sourceVideo ? resolveRefs(ctx, [opts.sourceVideo]).then((r) => r[0]) : void 0,
5320
- resolveKlingElements(ctx, rawElements)
5140
+ const [audioUrls, imageUrl] = await Promise.all([
5141
+ resolveRefs(ctx, rawAudioRefs),
5142
+ opts.image ? resolveRefs(ctx, [opts.image]).then((urls2) => urls2[0]) : void 0
5321
5143
  ]);
5322
- const parsedKeyframes = parseKeyframes(opts.keyframe ?? []);
5323
- const keyframes = parsedKeyframes.length > 0 ? await resolveRefs(
5324
- ctx,
5325
- parsedKeyframes.map((keyframe) => keyframe.source)
5326
- ).then(
5327
- (urls) => parsedKeyframes.map((keyframe, index) => ({
5328
- image_url: urls[index],
5329
- time_seconds: keyframe.time_seconds
5330
- }))
5331
- ) : [];
5332
- if (!prompt && segments.length === 0 && !startImage && keyframes.length === 0 && refs.length === 0 && refVideos.length === 0 && refAudios.length === 0 && !sourceVideo && !opts.fileUrl && !opts.webUrl && !opts.previousInteractionId) {
5144
+ capture("cli_generate", { kind: "audio", model: "seed-audio-1.0" });
5145
+ const toolArgs = compact({
5146
+ prompt,
5147
+ voice: opts.voice,
5148
+ audio_urls: audioUrls.length > 0 ? audioUrls : void 0,
5149
+ image_url: imageUrl,
5150
+ output_format: outputFormat,
5151
+ sample_rate: sampleRate,
5152
+ speed,
5153
+ volume,
5154
+ pitch,
5155
+ project_id: opts.project,
5156
+ session_id: sessionArg(this, opts),
5157
+ idempotency_key: idempotencyKey
5158
+ });
5159
+ let result;
5160
+ try {
5161
+ result = await callAudioWithRetry(
5162
+ () => ctx.client.callTool("generate_audio", toolArgs)
5163
+ );
5164
+ } catch (error) {
5165
+ const hint = isRetryableAudioError(error) ? `Retry this exact request with --idempotency-key ${idempotencyKey}` : void 0;
5166
+ if (error instanceof CliError) {
5167
+ if (hint) error.hint = hint;
5168
+ throw error;
5169
+ }
5333
5170
  throw new CliError(
5334
- "Provide a prompt, --segment, a frame, or a reference source.",
5335
- EXIT.USAGE
5171
+ error instanceof Error ? error.message : "Audio generation failed",
5172
+ EXIT.ERROR,
5173
+ hint
5336
5174
  );
5337
5175
  }
5338
- capture("cli_generate", {
5339
- kind: "video",
5340
- model: opts.model ?? "default",
5341
- wait: opts.wait !== false
5342
- });
5343
- const videoEditModels = /* @__PURE__ */ new Set([
5344
- "happy-horse-video-edit",
5345
- "grok-imagine-video-edit"
5346
- ]);
5347
- const motionControlModels = /* @__PURE__ */ new Set([
5348
- "kling-v3-motion-control",
5349
- "kling-2.6-motion-control"
5350
- ]);
5351
- let toolName = "generate_video";
5352
- let toolArgs;
5353
- if (videoEditModels.has(opts.model)) {
5354
- if (refVideos.length !== 1) {
5176
+ const urls = extractOutputUrls(result);
5177
+ let downloaded;
5178
+ if (opts.download && urls.length > 0) {
5179
+ downloaded = await downloadOutputs(urls, opts.download, {
5180
+ name: "seed-audio"
5181
+ });
5182
+ }
5183
+ const media = buildMediaDescriptors(urls, "audio");
5184
+ emit(
5185
+ ctx.out,
5186
+ { ...result, downloaded_files: downloaded, output_media: media },
5187
+ (o) => {
5188
+ for (const url of urls) process.stdout.write(`${url}
5189
+ `);
5190
+ for (const file of downloaded ?? []) {
5191
+ note(o, fmt.dim(o, savedLine(file)));
5192
+ }
5193
+ }
5194
+ );
5195
+ });
5196
+ generate.command("voiceover <text...>").description(
5197
+ "Generate TTS audio (synchronous, returns an audio URL). ElevenLabs voices run on Eleven v4"
5198
+ ).option(
5199
+ "--voice <id>",
5200
+ "TTS voice ID; accepts raw ElevenLabs IDs or elevenlabs-<id>, including voices outside `videodraft models voices`; must be accessible to the provider account"
5201
+ ).option(
5202
+ "--mode <standard|turbo>",
5203
+ "ElevenLabs voices only: standard (default; Eleven v4, best quality, 10 credits per 1000 chars) | turbo (Eleven v4 Turbo, faster, 5 credits per 1000 chars). Google, OpenAI and cloned custom-* voices ignore it"
5204
+ ).option("--language <bcp47>", 'target language, default "en"').option("--project <id>", "attach to a project").option(
5205
+ "--scene <n>",
5206
+ "0-based scene index; wires the audio onto that scene"
5207
+ ).option(
5208
+ "--session <id>",
5209
+ "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
5210
+ process.env.VIDEODRAFT_SESSION
5211
+ ).option("--download <path>", "download the audio file").action(async function(textWords) {
5212
+ const ctx = buildContext(this);
5213
+ const opts = this.opts();
5214
+ const mode = parseVoiceoverMode(opts.mode, "--mode");
5215
+ capture("cli_generate", { kind: "voiceover" });
5216
+ const result = await ctx.client.callTool(
5217
+ "generate_voiceover",
5218
+ compact({
5219
+ text: textWords.join(" "),
5220
+ voice_id: opts.voice,
5221
+ mode,
5222
+ target_language: opts.language,
5223
+ project_id: opts.project,
5224
+ session_id: sessionArg(this, opts),
5225
+ scene_index: opts.scene !== void 0 ? Number(opts.scene) : void 0
5226
+ })
5227
+ );
5228
+ const urls = extractOutputUrls(result);
5229
+ let downloaded;
5230
+ if (opts.download && urls.length > 0) {
5231
+ downloaded = await downloadOutputs(urls, opts.download, {
5232
+ name: "voiceover"
5233
+ });
5234
+ }
5235
+ const media = buildMediaDescriptors(urls, "audio");
5236
+ emit(
5237
+ ctx.out,
5238
+ { ...result, downloaded_files: downloaded, output_media: media },
5239
+ (o) => {
5240
+ for (const url of urls) process.stdout.write(`${url}
5241
+ `);
5242
+ for (const f of downloaded ?? []) note(o, fmt.dim(o, savedLine(f)));
5243
+ }
5244
+ );
5245
+ });
5246
+ generate.command("music [prompt...]").description(
5247
+ "Generate music (Lyria 3.5 for any length, vocals or instrumental; ElevenLabs for exact timing and composition plans)"
5248
+ ).option(
5249
+ "--model <id>",
5250
+ "omit for the server default (lyria-3.5 where the backend supports it) | lyria-3.5 (short or long, 10 credits) | lyria-3-clip-preview (fixed 30s, 4 credits) | lyria-3-pro-preview (legacy) | elevenlabs-music-v2.5 | elevenlabs-music-v1 (elevenlabs-music means v2.5)"
5251
+ ).option(
5252
+ "--length <seconds>",
5253
+ "ElevenLabs prompt mode: length 3-300s (default 30)"
5254
+ ).option(
5255
+ "--instrumental",
5256
+ "ElevenLabs prompt mode: force instrumental (no vocals)"
5257
+ ).option(
5258
+ "--plan <file|json>",
5259
+ 'ElevenLabs v2.5: composition plan JSON ({"chunks":[...]}) as a file or inline; replaces the prompt. Local audio_reference paths are uploaded. Picks elevenlabs-music-v2.5 when --model is omitted'
5260
+ ).option(
5261
+ "--section <spec>",
5262
+ 'ElevenLabs v2.5: add a plan chunk "<seconds>|<style, style>|<text>" (repeatable; \\n for line breaks; empty seconds = 20, empty text = instrumental), e.g. "20|synthwave,female vocals|[Verse 1]\\nNeon rain". Picks elevenlabs-music-v2.5 when --model is omitted',
5263
+ collect,
5264
+ []
5265
+ ).option(
5266
+ "--ref-audio <url|file>",
5267
+ "ElevenLabs v2.5 plan: style reference audio for the first chunk (local files uploaded)"
5268
+ ).option("--ref-start <ms>", "reference window start in ms (default 0)").option(
5269
+ "--ref-end <ms>",
5270
+ "reference window end in ms (window up to 30000 ms)"
5271
+ ).option(
5272
+ "--ref-strength <level>",
5273
+ "how closely to follow the reference: low | medium | high | xhigh"
5274
+ ).option("--seed <n>", "ElevenLabs v2.5 plan: random seed (0-4294967295)").option(
5275
+ "--format <output_format>",
5276
+ "ElevenLabs output format, e.g. mp3_48000_192 (v2.5 default), mp3_44100_128 (v1 default), opus_48000_128, pcm_44100 (WAV)"
5277
+ ).option(
5278
+ "--idempotency-key <uuid>",
5279
+ "ElevenLabs: stable UUID for recovery after an interruption"
5280
+ ).option(
5281
+ "--ref <url|file>",
5282
+ "reference image to inspire the music (Lyria only; max 10 on Google, 1 with Fal BYOK)",
5283
+ collect,
5284
+ []
5285
+ ).option(
5286
+ "--project <id>",
5287
+ "link the generation to a project's AI Studio session"
5288
+ ).option(
5289
+ "--attach <project_id>",
5290
+ "also set the track as that project's background music"
5291
+ ).option("--volume <n>", "0-100 BGM volume when attaching (default 30)").option(
5292
+ "--bgm-disabled",
5293
+ "when attaching, store the BGM as disabled (enabled:false)"
5294
+ ).option(
5295
+ "--session <id>",
5296
+ "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
5297
+ process.env.VIDEODRAFT_SESSION
5298
+ ).option("--download <path>", "download the audio file").option("--estimate", "show the credit quote and exit without generating").action(async function(promptWords) {
5299
+ const ctx = buildContext(this);
5300
+ const opts = this.opts();
5301
+ const sections = opts.section ?? [];
5302
+ const usesPlan = Boolean(opts.plan) || sections.length > 0;
5303
+ const musicModel = opts.model ?? (usesPlan ? "elevenlabs-music-v2.5" : "lyria-3.5");
5304
+ const usesServerDefault = !opts.model && !usesPlan;
5305
+ const isElevenMusic = ELEVENLABS_MUSIC_MODELS.includes(musicModel);
5306
+ const prompt = (promptWords ?? []).join(" ").trim();
5307
+ const elevenOnlyFlags = [
5308
+ ["--plan", opts.plan],
5309
+ ["--section", sections.length > 0 ? true : void 0],
5310
+ ["--ref-audio", opts.refAudio],
5311
+ ["--seed", opts.seed],
5312
+ ["--format", opts.format],
5313
+ ["--idempotency-key", opts.idempotencyKey]
5314
+ ].filter(([, value]) => value !== void 0 && value !== false);
5315
+ if (!isElevenMusic) {
5316
+ if (!LYRIA_MUSIC_MODELS.includes(musicModel)) {
5355
5317
  throw new CliError(
5356
- `${opts.model} requires exactly one --ref-video source. You can also use videodraft edit video.`,
5318
+ `Unknown music model "${musicModel}". Use ${[...LYRIA_MUSIC_MODELS, ...ELEVENLABS_MUSIC_MODELS].join(", ")}.`,
5357
5319
  EXIT.USAGE
5358
5320
  );
5359
5321
  }
5360
- if (startImage || endImage || refAudios.length > 0 || segments.length > 0 || opts.ar || opts.negative || opts.cameraFixed || opts.seed || elements.length > 0 || voiceIds.length > 0) {
5322
+ if (elevenOnlyFlags.length > 0) {
5361
5323
  throw new CliError(
5362
- `${opts.model} does not support --start-image, --end-image, --ref-audio, --element, --voice-id, --segment, --ar, --negative, --camera-fixed, or --seed in video-edit mode. Use --ref for supported reference images.`,
5324
+ `${elevenOnlyFlags.map(([flag]) => flag).join(", ")} only work with ElevenLabs Music. Add --model elevenlabs-music-v2.5.`,
5363
5325
  EXIT.USAGE
5364
5326
  );
5365
5327
  }
5366
- toolName = "edit_video";
5367
- toolArgs = {
5368
- model: opts.model,
5369
- prompt: prompt || void 0,
5370
- video_url: refVideos[0],
5371
- reference_images: refs.length > 0 ? refs : void 0,
5372
- resolution: opts.resolution,
5373
- quality: opts.quality,
5374
- duration_seconds: duration,
5375
- preserve_audio: opts.audio,
5376
- project_id: opts.project,
5377
- session_id: sessionArg(this, opts),
5378
- scene_index: opts.scene !== void 0 ? Number(opts.scene) : void 0,
5379
- shot_index: opts.shot !== void 0 ? Number(opts.shot) : void 0
5380
- };
5381
- } else if (motionControlModels.has(opts.model)) {
5382
- if (!startImage || refVideos.length !== 1) {
5328
+ if ((opts.ref ?? []).length > 10) {
5383
5329
  throw new CliError(
5384
- `${opts.model} requires --start-image plus exactly one --ref-video motion source. You can also use videodraft edit motion.`,
5330
+ "Lyria accepts at most 10 --ref images on Google, or 1 with Fal BYOK.",
5385
5331
  EXIT.USAGE
5386
5332
  );
5387
5333
  }
5388
- if (endImage || refs.length > 0 || refAudios.length > 0 || segments.length > 0 || opts.ar || opts.negative || opts.cameraFixed || opts.seed || opts.resolution || voiceIds.length > 0) {
5334
+ if (!prompt && (opts.ref ?? []).length === 0) {
5389
5335
  throw new CliError(
5390
- `${opts.model} does not support --end-image, --ref, --ref-audio, --voice-id, --segment, --ar, --negative, --camera-fixed, --seed, or --resolution in motion-control mode.`,
5336
+ "A prompt or at least one --ref image is required.",
5391
5337
  EXIT.USAGE
5392
5338
  );
5393
5339
  }
5394
- if (duration !== void 0) {
5340
+ if (opts.length !== void 0 || opts.instrumental) {
5341
+ note(
5342
+ ctx.out,
5343
+ fmt.dim(
5344
+ ctx.out,
5345
+ "--length and --instrumental only apply to ElevenLabs Music. For Lyria, put duration and instrumental/no-vocals instructions in the prompt; timing is approximate."
5346
+ )
5347
+ );
5348
+ }
5349
+ } else {
5350
+ if (usesPlan && musicModel === ELEVENLABS_MUSIC_V1) {
5395
5351
  throw new CliError(
5396
- `${opts.model} follows the motion video duration and orientation cap; use videodraft edit motion --estimate for a duration-based estimate.`,
5352
+ "Composition plans need elevenlabs-music-v2.5; v1 only takes a prompt.",
5353
+ EXIT.USAGE
5354
+ );
5355
+ }
5356
+ if (opts.plan && sections.length > 0) {
5357
+ throw new CliError("Use --plan or --section, not both.", EXIT.USAGE);
5358
+ }
5359
+ if (usesPlan && (prompt || opts.length || opts.instrumental)) {
5360
+ throw new CliError(
5361
+ "A composition plan replaces the prompt, --length and --instrumental; its length is the sum of its sections.",
5362
+ EXIT.USAGE
5363
+ );
5364
+ }
5365
+ if (!usesPlan && !prompt) {
5366
+ throw new CliError(
5367
+ "A prompt is required (or pass --plan / --section).",
5368
+ EXIT.USAGE
5369
+ );
5370
+ }
5371
+ if (!usesPlan && (opts.seed !== void 0 || opts.refAudio)) {
5372
+ throw new CliError(
5373
+ "--seed and --ref-audio need a composition plan (--plan or --section).",
5374
+ EXIT.USAGE
5375
+ );
5376
+ }
5377
+ if ((opts.ref ?? []).length > 0) {
5378
+ throw new CliError(
5379
+ "--ref images only work with Lyria. For ElevenLabs, use --ref-audio with a composition plan.",
5397
5380
  EXIT.USAGE
5398
5381
  );
5399
5382
  }
5400
- toolName = "generate_motion_control_video";
5401
- const resolvedElement = elements[0];
5402
- toolArgs = {
5403
- model: opts.model,
5404
- prompt: prompt || void 0,
5405
- image_url: startImage,
5406
- motion_video_url: refVideos[0],
5407
- quality: opts.quality,
5408
- keep_original_sound: opts.audio,
5409
- character_orientation: elements.length > 0 ? "video" : void 0,
5410
- duration_seconds: duration,
5411
- project_id: opts.project,
5412
- session_id: sessionArg(this, opts),
5413
- scene_index: opts.scene !== void 0 ? Number(opts.scene) : void 0,
5414
- shot_index: opts.shot !== void 0 ? Number(opts.shot) : void 0,
5415
- element: elements.length === 1 && resolvedElement ? {
5416
- frontal_image_url: resolvedElement.frontal_image_url,
5417
- reference_image_urls: resolvedElement.reference_image_urls
5418
- } : void 0
5419
- };
5420
- } else {
5421
- toolArgs = {
5422
- prompt: prompt || void 0,
5423
- model: opts.model,
5424
- aspect_ratio: opts.ar,
5425
- duration_seconds: duration,
5426
- auto_duration: opts.autoDuration ? true : void 0,
5427
- resolution: opts.resolution,
5428
- quality: opts.quality,
5429
- generate_audio: opts.audio,
5430
- allow_real_people: opts.allowRealPeople,
5431
- start_image_url: startImage,
5432
- end_image_url: endImage,
5433
- reference_images: refs.length > 0 ? refs : void 0,
5434
- video_url: sourceVideo,
5435
- reference_videos: refVideos.length > 0 ? refVideos : void 0,
5436
- reference_video_durations: referenceVideoDurations.length > 0 ? referenceVideoDurations : void 0,
5437
- reference_audio: refAudios.length > 0 ? refAudios : void 0,
5438
- file_url: opts.fileUrl,
5439
- web_url: opts.webUrl,
5440
- previous_interaction_id: opts.previousInteractionId,
5441
- video_task: geminiVideoTask,
5442
- enable_prompt_expansion: promptExpansion,
5443
- prompt_expansion_mode: promptExpansionMode,
5444
- enable_safety_checker: safetyChecker,
5445
- enable_thinking: opts.thinking ? true : void 0,
5446
- elements: elements.length > 0 ? elements : void 0,
5447
- voice_ids: voiceIds.length > 0 ? voiceIds.map((voiceId) => voiceId.trim()) : void 0,
5448
- multi_prompt: segments.length > 0 ? segments : void 0,
5449
- keyframes: keyframes.length > 0 ? keyframes : void 0,
5450
- negative_prompt: opts.negative,
5451
- camera_fixed: opts.cameraFixed ? true : void 0,
5452
- seed,
5453
- project_id: opts.project,
5454
- session_id: sessionArg(this, opts),
5455
- scene_index: opts.scene !== void 0 ? Number(opts.scene) : void 0,
5456
- shot_index: opts.shot !== void 0 ? Number(opts.shot) : void 0
5457
- };
5458
- }
5459
- const submitted = await ctx.client.callTool(toolName, compact(toolArgs));
5460
- await handleAsyncJob(ctx, submitted, {
5461
- wait: opts.wait !== false,
5462
- download: opts.download,
5463
- label: "Generating video"
5464
- });
5465
- });
5466
- generate.command("audio <prompt...>").description(
5467
- "Generate or edit audio with ByteDance Seed Audio 1.0 (synchronous)"
5468
- ).option("--voice <id>", "preset Seed Audio voice or custom cloned voice id").option(
5469
- "--ref-audio <url|file>",
5470
- "reference audio for @Audio1..@Audio3 (repeatable; local files uploaded)",
5471
- collect,
5472
- []
5473
- ).option(
5474
- "--image <url|file>",
5475
- "reference image (cannot be combined with --ref-audio)"
5476
- ).option("--format <wav|mp3|pcm|ogg_opus>", "output format (default mp3)").option(
5477
- "--sample-rate <hz>",
5478
- "8000 | 16000 | 24000 | 32000 | 44100 | 48000 (default 24000)"
5479
- ).option("--speed <0.5-2>", "playback speed multiplier (default 1)").option("--volume <0.5-2>", "volume multiplier (default 1)").option("--pitch <-12-12>", "pitch in whole semitones (default 0)").option("--project <id>", "link to a project's AI Studio session").option(
5480
- "--session <id>",
5481
- "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
5482
- process.env.VIDEODRAFT_SESSION
5483
- ).option(
5484
- "--idempotency-key <uuid>",
5485
- "set a stable UUID for recovery after a process interruption"
5486
- ).option("--download <path>", "download the generated audio file").option(
5487
- "--estimate",
5488
- "show 19 credits/minute pricing and the 38-credit maximum reservation"
5489
- ).action(async function(promptWords) {
5490
- const ctx = buildContext(this);
5491
- const opts = this.opts();
5492
- const prompt = promptWords.join(" ").trim();
5493
- if (!prompt) {
5494
- throw new CliError("A Seed Audio prompt is required.", EXIT.USAGE);
5495
- }
5496
- if (prompt.length > SEED_AUDIO_PROMPT_MAX_CHARS) {
5497
- throw new CliError(
5498
- `Seed Audio prompts must be ${SEED_AUDIO_PROMPT_MAX_CHARS} characters or fewer.`,
5499
- EXIT.USAGE
5500
- );
5501
5383
  }
5502
- const rawAudioRefs = opts.refAudio ?? [];
5503
- if (rawAudioRefs.length > SEED_AUDIO_MAX_REFERENCES) {
5384
+ if (!opts.refAudio && (opts.refStart !== void 0 || opts.refEnd !== void 0 || opts.refStrength !== void 0)) {
5504
5385
  throw new CliError(
5505
- `Seed Audio accepts at most ${SEED_AUDIO_MAX_REFERENCES} --ref-audio values.`,
5386
+ "--ref-start, --ref-end and --ref-strength need --ref-audio.",
5506
5387
  EXIT.USAGE
5507
5388
  );
5508
5389
  }
5509
- if (opts.image && rawAudioRefs.length > 0) {
5390
+ if (opts.refStrength !== void 0 && !MUSIC_REFERENCE_STRENGTHS.includes(opts.refStrength)) {
5510
5391
  throw new CliError(
5511
- "--image and --ref-audio cannot be used together.",
5392
+ `--ref-strength must be one of: ${MUSIC_REFERENCE_STRENGTHS.join(", ")}.`,
5512
5393
  EXIT.USAGE
5513
5394
  );
5514
5395
  }
5515
- const outputFormat = opts.format ?? "mp3";
5516
- if (!SEED_AUDIO_FORMATS.includes(outputFormat)) {
5396
+ const lengthSeconds = isElevenMusic ? optionalRangedNumber(opts.length, "--length", 3, 300) : void 0;
5397
+ const seed = optionalRangedNumber(
5398
+ opts.seed,
5399
+ "--seed",
5400
+ 0,
5401
+ 4294967295,
5402
+ true
5403
+ );
5404
+ const refStart = optionalRangedNumber(
5405
+ opts.refStart,
5406
+ "--ref-start",
5407
+ 0,
5408
+ Number.MAX_SAFE_INTEGER,
5409
+ true
5410
+ );
5411
+ const refEnd = optionalRangedNumber(
5412
+ opts.refEnd,
5413
+ "--ref-end",
5414
+ 1,
5415
+ Number.MAX_SAFE_INTEGER,
5416
+ true
5417
+ );
5418
+ const idempotencyKey = isElevenMusic ? opts.idempotencyKey ?? randomUUID4() : void 0;
5419
+ if (idempotencyKey && !UUID_RE3.test(idempotencyKey)) {
5517
5420
  throw new CliError(
5518
- `--format must be one of: ${SEED_AUDIO_FORMATS.join(", ")}.`,
5421
+ "--idempotency-key must be a valid UUID.",
5519
5422
  EXIT.USAGE
5520
5423
  );
5521
5424
  }
5522
- const sampleRate = opts.sampleRate ? Number(opts.sampleRate) : 24e3;
5523
- if (!SEED_AUDIO_SAMPLE_RATES.includes(sampleRate)) {
5524
- throw new CliError(
5525
- `--sample-rate must be one of: ${SEED_AUDIO_SAMPLE_RATES.join(", ")}.`,
5526
- EXIT.USAGE
5527
- );
5425
+ let planChunks;
5426
+ let planBaseDir = process.cwd();
5427
+ if (opts.plan) {
5428
+ const loaded = loadMusicPlan(opts.plan);
5429
+ planChunks = loaded.plan.chunks;
5430
+ planBaseDir = loaded.baseDir;
5431
+ } else if (sections.length > 0) {
5432
+ planChunks = parseMusicSections(sections);
5528
5433
  }
5529
- const speed = optionalRangedNumber(opts.speed, "--speed", 0.5, 2);
5530
- const volume = optionalRangedNumber(opts.volume, "--volume", 0.5, 2);
5531
- const pitch = optionalRangedNumber(opts.pitch, "--pitch", -12, 12, true);
5532
- const idempotencyKey = opts.idempotencyKey ?? randomUUID4();
5533
- if (!UUID_RE3.test(idempotencyKey)) {
5534
- throw new CliError(
5535
- "--idempotency-key must be a valid UUID.",
5536
- EXIT.USAGE
5537
- );
5434
+ if (planChunks && opts.refAudio) {
5435
+ const firstChunk = planChunks[0];
5436
+ if (firstChunk.audio_reference) {
5437
+ throw new CliError(
5438
+ "The plan's first chunk already has an audio_reference; drop --ref-audio or edit the plan.",
5439
+ EXIT.USAGE
5440
+ );
5441
+ }
5442
+ planChunks = [
5443
+ {
5444
+ ...firstChunk,
5445
+ audio_reference: compact({
5446
+ // A --ref-audio path is relative to where the command runs, not
5447
+ // to the plan file, so pin it before the plan's base dir applies.
5448
+ audio_url: URI_SCHEME.test(opts.refAudio) ? opts.refAudio : path8.resolve(opts.refAudio),
5449
+ start_ms: refStart,
5450
+ end_ms: refEnd,
5451
+ strength: opts.refStrength
5452
+ })
5453
+ },
5454
+ ...planChunks.slice(1)
5455
+ ];
5538
5456
  }
5539
5457
  if (opts.estimate) {
5458
+ const planSeconds = planChunks ? musicPlanSeconds(planChunks) : void 0;
5540
5459
  await printEstimate(ctx, {
5541
- model: "seed-audio-1.0",
5542
- type: "audio"
5460
+ // Quote the model the server will actually default to.
5461
+ model: usesServerDefault ? await serverDefaultMusicModel(ctx) : musicModel,
5462
+ type: "audio",
5463
+ duration: isElevenMusic ? planSeconds ?? lengthSeconds : void 0
5543
5464
  });
5544
5465
  return;
5545
5466
  }
5546
- const [audioUrls, imageUrl] = await Promise.all([
5547
- resolveRefs(ctx, rawAudioRefs),
5548
- opts.image ? resolveRefs(ctx, [opts.image]).then((urls2) => urls2[0]) : void 0
5549
- ]);
5550
- capture("cli_generate", { kind: "audio", model: "seed-audio-1.0" });
5467
+ const refs = isElevenMusic ? [] : await resolveRefs(ctx, opts.ref ?? []);
5468
+ const compositionPlan = planChunks ? {
5469
+ chunks: await resolveMusicPlanReferences(
5470
+ ctx,
5471
+ planChunks,
5472
+ planBaseDir
5473
+ )
5474
+ } : void 0;
5475
+ capture("cli_generate", {
5476
+ kind: "music",
5477
+ model: usesServerDefault ? "server-default" : musicModel,
5478
+ mode: compositionPlan ? "composition_plan" : "prompt"
5479
+ });
5551
5480
  const toolArgs = compact({
5552
- prompt,
5553
- voice: opts.voice,
5554
- audio_urls: audioUrls.length > 0 ? audioUrls : void 0,
5555
- image_url: imageUrl,
5556
- output_format: outputFormat,
5557
- sample_rate: sampleRate,
5558
- speed,
5559
- volume,
5560
- pitch,
5481
+ prompt: prompt || void 0,
5482
+ model: usesServerDefault ? void 0 : musicModel,
5483
+ length_seconds: lengthSeconds,
5484
+ force_instrumental: isElevenMusic && opts.instrumental ? true : void 0,
5485
+ composition_plan: compositionPlan,
5486
+ seed,
5487
+ output_format: opts.format,
5488
+ idempotency_key: idempotencyKey,
5489
+ image_urls: refs.length > 0 ? refs : void 0,
5561
5490
  project_id: opts.project,
5562
- session_id: sessionArg(this, opts),
5563
- idempotency_key: idempotencyKey
5491
+ attach_to_project_id: opts.attach,
5492
+ volume: opts.volume ? Number(opts.volume) : void 0,
5493
+ enabled: opts.bgmDisabled ? false : void 0,
5494
+ session_id: sessionArg(this, opts)
5564
5495
  });
5565
5496
  let result;
5566
- try {
5567
- result = await callAudioWithRetry(
5568
- () => ctx.client.callTool("generate_audio", toolArgs)
5569
- );
5570
- } catch (error) {
5571
- const hint = isRetryableAudioError(error) ? `Retry this exact request with --idempotency-key ${idempotencyKey}` : void 0;
5572
- if (error instanceof CliError) {
5573
- if (hint) error.hint = hint;
5574
- throw error;
5497
+ if (isElevenMusic) {
5498
+ try {
5499
+ result = await callAudioWithRetry(
5500
+ () => ctx.client.callTool("generate_music", toolArgs)
5501
+ );
5502
+ } catch (error) {
5503
+ const hint = isRetryableAudioError(error) ? `Retry this exact request with --idempotency-key ${idempotencyKey}` : void 0;
5504
+ if (error instanceof CliError) {
5505
+ if (hint) error.hint = hint;
5506
+ throw error;
5507
+ }
5508
+ throw new CliError(
5509
+ error instanceof Error ? error.message : "Music generation failed",
5510
+ EXIT.ERROR,
5511
+ hint
5512
+ );
5575
5513
  }
5576
- throw new CliError(
5577
- error instanceof Error ? error.message : "Audio generation failed",
5578
- EXIT.ERROR,
5579
- hint
5514
+ } else {
5515
+ result = await ctx.client.callTool("generate_music", toolArgs);
5516
+ }
5517
+ const urls = extractOutputUrls(result);
5518
+ let downloaded;
5519
+ if (opts.download && urls.length > 0) {
5520
+ downloaded = await downloadOutputs(urls, opts.download, {
5521
+ name: "music"
5522
+ });
5523
+ }
5524
+ const media = buildMediaDescriptors(urls, "music");
5525
+ emit(
5526
+ ctx.out,
5527
+ { ...result, downloaded_files: downloaded, output_media: media },
5528
+ (o) => {
5529
+ for (const url of urls) process.stdout.write(`${url}
5530
+ `);
5531
+ for (const f of downloaded ?? []) note(o, fmt.dim(o, savedLine(f)));
5532
+ }
5533
+ );
5534
+ });
5535
+ generate.command("sound-effect <prompt...>").description("Generate a sound effect (ElevenLabs Sound Effects)").option("--duration <seconds>", "length 0.5\u201322s (default 5)").option("--influence <0-1>", "prompt influence (default 0.3)").option("--project <id>", "link to a project's AI Studio session").option(
5536
+ "--session <id>",
5537
+ "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
5538
+ process.env.VIDEODRAFT_SESSION
5539
+ ).option("--download <path>", "download the audio file").action(async function(promptWords) {
5540
+ const ctx = buildContext(this);
5541
+ const opts = this.opts();
5542
+ capture("cli_generate", { kind: "sound_effect" });
5543
+ const result = await ctx.client.callTool(
5544
+ "generate_sound_effect",
5545
+ compact({
5546
+ prompt: promptWords.join(" "),
5547
+ duration_seconds: opts.duration ? Number(opts.duration) : void 0,
5548
+ prompt_influence: opts.influence ? Number(opts.influence) : void 0,
5549
+ project_id: opts.project,
5550
+ session_id: sessionArg(this, opts)
5551
+ })
5552
+ );
5553
+ const urls = extractOutputUrls(result);
5554
+ let downloaded;
5555
+ if (opts.download && urls.length > 0) {
5556
+ downloaded = await downloadOutputs(urls, opts.download, {
5557
+ name: "sound-effect"
5558
+ });
5559
+ }
5560
+ const media = buildMediaDescriptors(urls, "audio");
5561
+ emit(
5562
+ ctx.out,
5563
+ { ...result, downloaded_files: downloaded, output_media: media },
5564
+ (o) => {
5565
+ for (const url of urls) process.stdout.write(`${url}
5566
+ `);
5567
+ for (const f of downloaded ?? []) note(o, fmt.dim(o, savedLine(f)));
5568
+ }
5569
+ );
5570
+ });
5571
+ generate.command("dialogue").description(
5572
+ "Generate multi-speaker dialogue (ElevenLabs Text-to-Dialogue). Repeat --line."
5573
+ ).option(
5574
+ "--line <voiceId:text>",
5575
+ 'a dialogue line as "voiceId:text" (repeatable); accepts raw ElevenLabs IDs or elevenlabs-<id>, including voices outside the catalog; must be accessible to the provider account',
5576
+ collect,
5577
+ []
5578
+ ).option("--stability <0|0.5|1>", "voice stability").option("--language <iso>", "ISO 639-1 language code").option("--project <id>", "link to a project's AI Studio session").option(
5579
+ "--session <id>",
5580
+ "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
5581
+ process.env.VIDEODRAFT_SESSION
5582
+ ).option("--download <path>", "download the audio file").action(async function() {
5583
+ const ctx = buildContext(this);
5584
+ const opts = this.opts();
5585
+ const lines = opts.line.map((raw) => {
5586
+ const i = raw.indexOf(":");
5587
+ if (i < 0) {
5588
+ throw new Error(`--line must be "voiceId:text" (got "${raw}")`);
5589
+ }
5590
+ return {
5591
+ voice_id: raw.slice(0, i).trim(),
5592
+ text: raw.slice(i + 1).trim()
5593
+ };
5594
+ });
5595
+ if (lines.length === 0)
5596
+ throw new Error("at least one --line is required");
5597
+ capture("cli_generate", { kind: "dialogue" });
5598
+ const result = await ctx.client.callTool(
5599
+ "generate_dialogue",
5600
+ compact({
5601
+ lines,
5602
+ stability: opts.stability !== void 0 ? Number(opts.stability) : void 0,
5603
+ language_code: opts.language,
5604
+ project_id: opts.project,
5605
+ session_id: sessionArg(this, opts)
5606
+ })
5607
+ );
5608
+ const urls = extractOutputUrls(result);
5609
+ let downloaded;
5610
+ if (opts.download && urls.length > 0) {
5611
+ downloaded = await downloadOutputs(urls, opts.download, {
5612
+ name: "dialogue"
5613
+ });
5614
+ }
5615
+ const media = buildMediaDescriptors(urls, "audio");
5616
+ emit(
5617
+ ctx.out,
5618
+ { ...result, downloaded_files: downloaded, output_media: media },
5619
+ (o) => {
5620
+ for (const url of urls) process.stdout.write(`${url}
5621
+ `);
5622
+ for (const f of downloaded ?? []) note(o, fmt.dim(o, savedLine(f)));
5623
+ }
5624
+ );
5625
+ });
5626
+ generate.command("voice-changer <audio>").description("Restyle speech into another ElevenLabs voice (Voice Changer)").option(
5627
+ "--voice <id>",
5628
+ "target ElevenLabs voice ID, raw or elevenlabs-<id>; catalog membership is not required, but provider account access is (default Brittney, or an account voice with ElevenLabs BYOK)"
5629
+ ).option(
5630
+ "--duration <seconds>",
5631
+ "length of the source audio in seconds (required, max 300)"
5632
+ ).option("--remove-noise", "remove background noise from the input").option("--project <id>", "link to a project's AI Studio session").option(
5633
+ "--session <id>",
5634
+ "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
5635
+ process.env.VIDEODRAFT_SESSION
5636
+ ).option("--download <path>", "download the audio file").action(async function(source) {
5637
+ const ctx = buildContext(this);
5638
+ const opts = this.opts();
5639
+ if (!opts.duration) {
5640
+ throw new Error(
5641
+ "--duration <seconds> is required (length of the source audio)"
5580
5642
  );
5581
5643
  }
5644
+ const [audioUrl] = await resolveRefs(ctx, [source]);
5645
+ capture("cli_generate", { kind: "voice_changer" });
5646
+ const result = await ctx.client.callTool(
5647
+ "change_voice",
5648
+ compact({
5649
+ audio_url: audioUrl,
5650
+ voice_id: opts.voice,
5651
+ duration_seconds: Number(opts.duration),
5652
+ remove_background_noise: opts.removeNoise ? true : void 0,
5653
+ project_id: opts.project,
5654
+ session_id: sessionArg(this, opts)
5655
+ })
5656
+ );
5582
5657
  const urls = extractOutputUrls(result);
5583
5658
  let downloaded;
5584
5659
  if (opts.download && urls.length > 0) {
5585
5660
  downloaded = await downloadOutputs(urls, opts.download, {
5586
- name: "seed-audio"
5661
+ name: "voice-changed"
5587
5662
  });
5588
5663
  }
5589
5664
  const media = buildMediaDescriptors(urls, "audio");
@@ -5593,45 +5668,58 @@ function registerGenerateCommands(program) {
5593
5668
  (o) => {
5594
5669
  for (const url of urls) process.stdout.write(`${url}
5595
5670
  `);
5596
- for (const file of downloaded ?? []) {
5597
- note(o, fmt.dim(o, savedLine(file)));
5598
- }
5671
+ for (const f of downloaded ?? []) note(o, fmt.dim(o, savedLine(f)));
5599
5672
  }
5600
5673
  );
5601
5674
  });
5602
- generate.command("voiceover <text...>").description("Generate TTS audio (synchronous \u2014 returns an audio URL)").option(
5603
- "--voice <id>",
5604
- "TTS voice ID; accepts raw ElevenLabs IDs or elevenlabs-<id>, including voices outside `videodraft models voices`; must be accessible to the provider account"
5605
- ).option("--language <bcp47>", 'target language, default "en"').option("--project <id>", "attach to a project").option(
5606
- "--scene <n>",
5607
- "0-based scene index; wires the audio onto that scene"
5675
+ generate.command("dub <media>").description(
5676
+ "Dub a video/audio file into another language (ElevenLabs Dubbing)"
5677
+ ).option(
5678
+ "--to <iso>",
5679
+ "target language ISO 639-1 code, e.g. es or te (required)"
5608
5680
  ).option(
5681
+ "--from <iso>",
5682
+ "source language ISO 639-1 code (auto-detected if omitted)"
5683
+ ).option("--type <audio|video>", "source media type override").option(
5684
+ "--duration <seconds>",
5685
+ "length of the source media in seconds (required, max 300)"
5686
+ ).option("--speakers <n>", "number of speakers (auto-detected if omitted)").option("--project <id>", "link to a project's AI Studio session").option(
5609
5687
  "--session <id>",
5610
5688
  "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
5611
5689
  process.env.VIDEODRAFT_SESSION
5612
- ).option("--download <path>", "download the audio file").action(async function(textWords) {
5690
+ ).option("--download <path>", "download the dubbed file").action(async function(source) {
5613
5691
  const ctx = buildContext(this);
5614
5692
  const opts = this.opts();
5615
- capture("cli_generate", { kind: "voiceover" });
5693
+ if (!opts.to) throw new Error("--to <iso> (target language) is required");
5694
+ if (!opts.duration) {
5695
+ throw new Error(
5696
+ "--duration <seconds> is required (length of the source media)"
5697
+ );
5698
+ }
5699
+ const mediaType = inferDubMediaType(source, opts.type);
5700
+ const [mediaUrl] = await resolveRefs(ctx, [source]);
5701
+ capture("cli_generate", { kind: "dub" });
5616
5702
  const result = await ctx.client.callTool(
5617
- "generate_voiceover",
5703
+ "dub_media",
5618
5704
  compact({
5619
- text: textWords.join(" "),
5620
- voice_id: opts.voice,
5621
- target_language: opts.language,
5705
+ video_url: mediaType === "video" ? mediaUrl : void 0,
5706
+ audio_url: mediaType === "audio" ? mediaUrl : void 0,
5707
+ target_lang: opts.to,
5708
+ source_lang: opts.from,
5709
+ num_speakers: opts.speakers ? Number(opts.speakers) : void 0,
5710
+ duration_seconds: Number(opts.duration),
5622
5711
  project_id: opts.project,
5623
- session_id: sessionArg(this, opts),
5624
- scene_index: opts.scene !== void 0 ? Number(opts.scene) : void 0
5712
+ session_id: sessionArg(this, opts)
5625
5713
  })
5626
5714
  );
5627
5715
  const urls = extractOutputUrls(result);
5628
5716
  let downloaded;
5629
5717
  if (opts.download && urls.length > 0) {
5630
5718
  downloaded = await downloadOutputs(urls, opts.download, {
5631
- name: "voiceover"
5719
+ name: "dubbed"
5632
5720
  });
5633
5721
  }
5634
- const media = buildMediaDescriptors(urls, "audio");
5722
+ const media = buildMediaDescriptors(urls, mediaType);
5635
5723
  emit(
5636
5724
  ctx.out,
5637
5725
  { ...result, downloaded_files: downloaded, output_media: media },
@@ -5642,681 +5730,737 @@ function registerGenerateCommands(program) {
5642
5730
  }
5643
5731
  );
5644
5732
  });
5645
- generate.command("music [prompt...]").description(
5646
- "Generate music (Lyria 3.5 for any length, vocals or instrumental; ElevenLabs for exact timing and composition plans)"
5647
- ).option(
5648
- "--model <id>",
5649
- "omit for the server default (lyria-3.5 where the backend supports it) | lyria-3.5 (short or long, 10 credits) | lyria-3-clip-preview (fixed 30s, 4 credits) | lyria-3-pro-preview (legacy) | elevenlabs-music-v2.5 | elevenlabs-music-v1 (elevenlabs-music means v2.5)"
5650
- ).option(
5651
- "--length <seconds>",
5652
- "ElevenLabs prompt mode: length 3-300s (default 30)"
5653
- ).option(
5654
- "--instrumental",
5655
- "ElevenLabs prompt mode: force instrumental (no vocals)"
5656
- ).option(
5657
- "--plan <file|json>",
5658
- 'ElevenLabs v2.5: composition plan JSON ({"chunks":[...]}) as a file or inline; replaces the prompt. Local audio_reference paths are uploaded. Picks elevenlabs-music-v2.5 when --model is omitted'
5659
- ).option(
5660
- "--section <spec>",
5661
- 'ElevenLabs v2.5: add a plan chunk "<seconds>|<style, style>|<text>" (repeatable; \\n for line breaks; empty seconds = 20, empty text = instrumental), e.g. "20|synthwave,female vocals|[Verse 1]\\nNeon rain". Picks elevenlabs-music-v2.5 when --model is omitted',
5662
- collect,
5663
- []
5664
- ).option(
5665
- "--ref-audio <url|file>",
5666
- "ElevenLabs v2.5 plan: style reference audio for the first chunk (local files uploaded)"
5667
- ).option("--ref-start <ms>", "reference window start in ms (default 0)").option(
5668
- "--ref-end <ms>",
5669
- "reference window end in ms (window up to 30000 ms)"
5670
- ).option(
5671
- "--ref-strength <level>",
5672
- "how closely to follow the reference: low | medium | high | xhigh"
5673
- ).option("--seed <n>", "ElevenLabs v2.5 plan: random seed (0-4294967295)").option(
5674
- "--format <output_format>",
5675
- "ElevenLabs output format, e.g. mp3_48000_192 (v2.5 default), mp3_44100_128 (v1 default), opus_48000_128, pcm_44100 (WAV)"
5676
- ).option(
5677
- "--idempotency-key <uuid>",
5678
- "ElevenLabs: stable UUID for recovery after an interruption"
5733
+ const upscale = program.command("upscale").description("Upscale / enhance images and videos (Topaz)");
5734
+ const numOpt = (v) => v === void 0 || v === null || v === "" ? void 0 : Number(v);
5735
+ upscale.command("image <url|file>").description(
5736
+ "Enhance or upscale an existing image with Topaz (waits by default)"
5737
+ ).option("--scale <factor>", '"1x" | "2x" | "4x" (default 2x)').option(
5738
+ "--mode <mode>",
5739
+ "generative (default; Wonder 3.5, best for AI images) | precision (faithful, cheapest; real photos) | creative (Bloom, artistic)"
5679
5740
  ).option(
5680
- "--ref <url|file>",
5681
- "reference image to inspire the music (Lyria only; max 10 on Google, 1 with Fal BYOK)",
5682
- collect,
5683
- []
5741
+ "--model <name>",
5742
+ 'Topaz model inside the mode, e.g. "High Fidelity V3", "Wonder 3.5", "Redefine", "Bloom 2" (see: videodraft models image --json)'
5743
+ ).option("--format <jpeg|png>", "output format (default jpeg)").option("--no-face-enhance", "disable Topaz face recovery").option("--face-strength <0-1>", "face recovery strength (default 0.8)").option("--sharpen <0-1>", "extra sharpening (precision/generative)").option("--denoise <0-1>", "noise reduction (precision/generative)").option(
5744
+ "--fix-compression <0-1>",
5745
+ "compression-artifact repair (precision)"
5684
5746
  ).option(
5685
- "--project <id>",
5686
- "link the generation to a project's AI Studio session"
5747
+ "--prompt <text>",
5748
+ "guide detail (generative Redefine / creative Bloom)"
5749
+ ).option("--creativity <n>", "generative 1-6 / creative 1-9").option("--texture <1-5>", "texture amount (generative Redefine)").option(
5750
+ "--width <px>",
5751
+ "source width hint (server must verify dimensions; max source 50MB)"
5687
5752
  ).option(
5688
- "--attach <project_id>",
5689
- "also set the track as that project's background music"
5690
- ).option("--volume <n>", "0-100 BGM volume when attaching (default 30)").option(
5691
- "--bgm-disabled",
5692
- "when attaching, store the BGM as disabled (enabled:false)"
5753
+ "--height <px>",
5754
+ "source height hint (server must verify dimensions; max source 50MB)"
5693
5755
  ).option(
5694
5756
  "--session <id>",
5695
5757
  "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
5696
5758
  process.env.VIDEODRAFT_SESSION
5697
- ).option("--download <path>", "download the audio file").option("--estimate", "show the credit quote and exit without generating").action(async function(promptWords) {
5759
+ ).option("--download <path>", "download the result").option(
5760
+ "--no-wait",
5761
+ "return a job ID immediately when the server queues the upscale"
5762
+ ).action(async function(source) {
5698
5763
  const ctx = buildContext(this);
5699
5764
  const opts = this.opts();
5700
- const sections = opts.section ?? [];
5701
- const usesPlan = Boolean(opts.plan) || sections.length > 0;
5702
- const musicModel = opts.model ?? (usesPlan ? "elevenlabs-music-v2.5" : "lyria-3.5");
5703
- const usesServerDefault = !opts.model && !usesPlan;
5704
- const isElevenMusic = ELEVENLABS_MUSIC_MODELS.includes(musicModel);
5705
- const prompt = (promptWords ?? []).join(" ").trim();
5706
- const elevenOnlyFlags = [
5707
- ["--plan", opts.plan],
5708
- ["--section", sections.length > 0 ? true : void 0],
5709
- ["--ref-audio", opts.refAudio],
5710
- ["--seed", opts.seed],
5711
- ["--format", opts.format],
5712
- ["--idempotency-key", opts.idempotencyKey]
5713
- ].filter(([, value]) => value !== void 0 && value !== false);
5714
- if (!isElevenMusic) {
5715
- if (!LYRIA_MUSIC_MODELS.includes(musicModel)) {
5716
- throw new CliError(
5717
- `Unknown music model "${musicModel}". Use ${[...LYRIA_MUSIC_MODELS, ...ELEVENLABS_MUSIC_MODELS].join(", ")}.`,
5718
- EXIT.USAGE
5719
- );
5720
- }
5721
- if (elevenOnlyFlags.length > 0) {
5722
- throw new CliError(
5723
- `${elevenOnlyFlags.map(([flag]) => flag).join(", ")} only work with ElevenLabs Music. Add --model elevenlabs-music-v2.5.`,
5724
- EXIT.USAGE
5725
- );
5726
- }
5727
- if ((opts.ref ?? []).length > 10) {
5728
- throw new CliError(
5729
- "Lyria accepts at most 10 --ref images on Google, or 1 with Fal BYOK.",
5730
- EXIT.USAGE
5731
- );
5732
- }
5733
- if (!prompt && (opts.ref ?? []).length === 0) {
5734
- throw new CliError(
5735
- "A prompt or at least one --ref image is required.",
5736
- EXIT.USAGE
5737
- );
5738
- }
5739
- if (opts.length !== void 0 || opts.instrumental) {
5740
- note(
5741
- ctx.out,
5742
- fmt.dim(
5743
- ctx.out,
5744
- "--length and --instrumental only apply to ElevenLabs Music. For Lyria, put duration and instrumental/no-vocals instructions in the prompt; timing is approximate."
5745
- )
5746
- );
5747
- }
5748
- } else {
5749
- if (usesPlan && musicModel === ELEVENLABS_MUSIC_V1) {
5750
- throw new CliError(
5751
- "Composition plans need elevenlabs-music-v2.5; v1 only takes a prompt.",
5752
- EXIT.USAGE
5753
- );
5754
- }
5755
- if (opts.plan && sections.length > 0) {
5756
- throw new CliError("Use --plan or --section, not both.", EXIT.USAGE);
5757
- }
5758
- if (usesPlan && (prompt || opts.length || opts.instrumental)) {
5759
- throw new CliError(
5760
- "A composition plan replaces the prompt, --length and --instrumental; its length is the sum of its sections.",
5761
- EXIT.USAGE
5762
- );
5763
- }
5764
- if (!usesPlan && !prompt) {
5765
- throw new CliError(
5766
- "A prompt is required (or pass --plan / --section).",
5767
- EXIT.USAGE
5768
- );
5769
- }
5770
- if (!usesPlan && (opts.seed !== void 0 || opts.refAudio)) {
5771
- throw new CliError(
5772
- "--seed and --ref-audio need a composition plan (--plan or --section).",
5773
- EXIT.USAGE
5774
- );
5775
- }
5776
- if ((opts.ref ?? []).length > 0) {
5777
- throw new CliError(
5778
- "--ref images only work with Lyria. For ElevenLabs, use --ref-audio with a composition plan.",
5779
- EXIT.USAGE
5780
- );
5781
- }
5782
- }
5783
- if (!opts.refAudio && (opts.refStart !== void 0 || opts.refEnd !== void 0 || opts.refStrength !== void 0)) {
5784
- throw new CliError(
5785
- "--ref-start, --ref-end and --ref-strength need --ref-audio.",
5786
- EXIT.USAGE
5787
- );
5788
- }
5789
- if (opts.refStrength !== void 0 && !MUSIC_REFERENCE_STRENGTHS.includes(opts.refStrength)) {
5790
- throw new CliError(
5791
- `--ref-strength must be one of: ${MUSIC_REFERENCE_STRENGTHS.join(", ")}.`,
5792
- EXIT.USAGE
5793
- );
5794
- }
5795
- const lengthSeconds = isElevenMusic ? optionalRangedNumber(opts.length, "--length", 3, 300) : void 0;
5796
- const seed = optionalRangedNumber(
5797
- opts.seed,
5798
- "--seed",
5799
- 0,
5800
- 4294967295,
5801
- true
5802
- );
5803
- const refStart = optionalRangedNumber(
5804
- opts.refStart,
5805
- "--ref-start",
5806
- 0,
5807
- Number.MAX_SAFE_INTEGER,
5808
- true
5809
- );
5810
- const refEnd = optionalRangedNumber(
5811
- opts.refEnd,
5812
- "--ref-end",
5813
- 1,
5814
- Number.MAX_SAFE_INTEGER,
5815
- true
5765
+ const [url] = await resolveRefs(ctx, [source]);
5766
+ capture("cli_upscale", {
5767
+ kind: "image",
5768
+ mode: opts.mode ?? "generative"
5769
+ });
5770
+ const result = await ctx.client.callTool(
5771
+ "upscale_image",
5772
+ compact({
5773
+ image_url: url,
5774
+ scale: opts.scale,
5775
+ mode: opts.mode,
5776
+ model: opts.model,
5777
+ output_format: opts.format,
5778
+ face_enhancement: opts.faceEnhance === false ? false : void 0,
5779
+ face_enhancement_strength: numOpt(opts.faceStrength),
5780
+ sharpen: numOpt(opts.sharpen),
5781
+ denoise: numOpt(opts.denoise),
5782
+ fix_compression: numOpt(opts.fixCompression),
5783
+ prompt: opts.prompt,
5784
+ creativity: numOpt(opts.creativity),
5785
+ texture: numOpt(opts.texture),
5786
+ image_width: numOpt(opts.width),
5787
+ image_height: numOpt(opts.height),
5788
+ session_id: sessionArg(this, opts)
5789
+ })
5816
5790
  );
5817
- const idempotencyKey = isElevenMusic ? opts.idempotencyKey ?? randomUUID4() : void 0;
5818
- if (idempotencyKey && !UUID_RE3.test(idempotencyKey)) {
5819
- throw new CliError(
5820
- "--idempotency-key must be a valid UUID.",
5821
- EXIT.USAGE
5822
- );
5823
- }
5824
- let planChunks;
5825
- let planBaseDir = process.cwd();
5826
- if (opts.plan) {
5827
- const loaded = loadMusicPlan(opts.plan);
5828
- planChunks = loaded.plan.chunks;
5829
- planBaseDir = loaded.baseDir;
5830
- } else if (sections.length > 0) {
5831
- planChunks = parseMusicSections(sections);
5832
- }
5833
- if (planChunks && opts.refAudio) {
5834
- const firstChunk = planChunks[0];
5835
- if (firstChunk.audio_reference) {
5836
- throw new CliError(
5837
- "The plan's first chunk already has an audio_reference; drop --ref-audio or edit the plan.",
5838
- EXIT.USAGE
5839
- );
5840
- }
5841
- planChunks = [
5842
- {
5843
- ...firstChunk,
5844
- audio_reference: compact({
5845
- // A --ref-audio path is relative to where the command runs, not
5846
- // to the plan file, so pin it before the plan's base dir applies.
5847
- audio_url: URI_SCHEME.test(opts.refAudio) ? opts.refAudio : path8.resolve(opts.refAudio),
5848
- start_ms: refStart,
5849
- end_ms: refEnd,
5850
- strength: opts.refStrength
5851
- })
5852
- },
5853
- ...planChunks.slice(1)
5854
- ];
5855
- }
5856
- if (opts.estimate) {
5857
- const planSeconds = planChunks ? musicPlanSeconds(planChunks) : void 0;
5858
- await printEstimate(ctx, {
5859
- // Quote the model the server will actually default to.
5860
- model: usesServerDefault ? await serverDefaultMusicModel(ctx) : musicModel,
5861
- type: "audio",
5862
- duration: isElevenMusic ? planSeconds ?? lengthSeconds : void 0
5791
+ if (result?.job_id || result?.jobId) {
5792
+ await handleAsyncJob(ctx, result, {
5793
+ wait: opts.wait !== false,
5794
+ download: opts.download,
5795
+ label: "Upscaling image"
5863
5796
  });
5864
5797
  return;
5865
5798
  }
5866
- const refs = isElevenMusic ? [] : await resolveRefs(ctx, opts.ref ?? []);
5867
- const compositionPlan = planChunks ? {
5868
- chunks: await resolveMusicPlanReferences(
5869
- ctx,
5870
- planChunks,
5871
- planBaseDir
5872
- )
5873
- } : void 0;
5874
- capture("cli_generate", {
5875
- kind: "music",
5876
- model: usesServerDefault ? "server-default" : musicModel,
5877
- mode: compositionPlan ? "composition_plan" : "prompt"
5878
- });
5879
- const toolArgs = compact({
5880
- prompt: prompt || void 0,
5881
- model: usesServerDefault ? void 0 : musicModel,
5882
- length_seconds: lengthSeconds,
5883
- force_instrumental: isElevenMusic && opts.instrumental ? true : void 0,
5884
- composition_plan: compositionPlan,
5885
- seed,
5886
- output_format: opts.format,
5887
- idempotency_key: idempotencyKey,
5888
- image_urls: refs.length > 0 ? refs : void 0,
5889
- project_id: opts.project,
5890
- attach_to_project_id: opts.attach,
5891
- volume: opts.volume ? Number(opts.volume) : void 0,
5892
- enabled: opts.bgmDisabled ? false : void 0,
5893
- session_id: sessionArg(this, opts)
5894
- });
5895
- let result;
5896
- if (isElevenMusic) {
5897
- try {
5898
- result = await callAudioWithRetry(
5899
- () => ctx.client.callTool("generate_music", toolArgs)
5900
- );
5901
- } catch (error) {
5902
- const hint = isRetryableAudioError(error) ? `Retry this exact request with --idempotency-key ${idempotencyKey}` : void 0;
5903
- if (error instanceof CliError) {
5904
- if (hint) error.hint = hint;
5905
- throw error;
5906
- }
5907
- throw new CliError(
5908
- error instanceof Error ? error.message : "Music generation failed",
5909
- EXIT.ERROR,
5910
- hint
5911
- );
5912
- }
5913
- } else {
5914
- result = await ctx.client.callTool("generate_music", toolArgs);
5915
- }
5916
5799
  const urls = extractOutputUrls(result);
5917
5800
  let downloaded;
5918
5801
  if (opts.download && urls.length > 0) {
5919
5802
  downloaded = await downloadOutputs(urls, opts.download, {
5920
- name: "music"
5803
+ name: "upscaled"
5921
5804
  });
5922
5805
  }
5923
- const media = buildMediaDescriptors(urls, "music");
5806
+ const media = buildMediaDescriptors(urls, "image");
5924
5807
  emit(
5925
5808
  ctx.out,
5926
5809
  { ...result, downloaded_files: downloaded, output_media: media },
5927
5810
  (o) => {
5928
- for (const url of urls) process.stdout.write(`${url}
5811
+ for (const u of urls) process.stdout.write(`${u}
5929
5812
  `);
5930
5813
  for (const f of downloaded ?? []) note(o, fmt.dim(o, savedLine(f)));
5931
5814
  }
5932
5815
  );
5933
5816
  });
5934
- generate.command("sound-effect <prompt...>").description("Generate a sound effect (ElevenLabs Sound Effects)").option("--duration <seconds>", "length 0.5\u201322s (default 5)").option("--influence <0-1>", "prompt influence (default 0.3)").option("--project <id>", "link to a project's AI Studio session").option(
5935
- "--session <id>",
5936
- "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
5937
- process.env.VIDEODRAFT_SESSION
5938
- ).option("--download <path>", "download the audio file").action(async function(promptWords) {
5817
+ upscale.command("video <url|file>").description(
5818
+ "Enhance or upscale an existing video with Topaz (async; waits by default)"
5819
+ ).option(
5820
+ "--resolution <720p|1080p|4k>",
5821
+ "output preset by short edge (preferred over --scale)"
5822
+ ).option(
5823
+ "--scale <factor>",
5824
+ '"1x" | "2x" | "4x" (default 2x when no --resolution)'
5825
+ ).option(
5826
+ "--mode <mode>",
5827
+ "generative (default; Starlight Precise 2.6, best for AI clips) | precision (Proteus, 6x cheaper; real footage) | creative (Astra 2)"
5828
+ ).option(
5829
+ "--model <name>",
5830
+ 'Topaz model inside the mode, e.g. "Proteus", "Gaia 2", "Starlight Precise 2.6", "Starlight Fast 2" (see: videodraft models video --category upscale --json)'
5831
+ ).option("--fps <n>", "deliver at this frame rate (24-120; 60 doubles cost)").option("--prompt <text>", "creative (Astra 2) only").option("--creativity <0-1>", "creative (Astra 2) only").option("--realism <0-1>", "creative (Astra 2) only").option("--sharp <0-1>", "creative (Astra 2) only").option("--softness <1-5>", "generative (Starlight Precise 2.6) only").option("--compression <0-1>", "precision only").option("--noise <0-1>", "precision only").option("--halo <0-1>", "precision only").option("--grain <0-0.1>", "precision only").option("--recover-detail <0-1>", "precision only").option(
5832
+ "--session <id>",
5833
+ "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
5834
+ process.env.VIDEODRAFT_SESSION
5835
+ ).option(
5836
+ "--duration <seconds>",
5837
+ "source duration hint (compatibility only; server must probe MP4/MOV up to 100MB)"
5838
+ ).option(
5839
+ "--width <px>",
5840
+ "source width hint (server measurement is required)"
5841
+ ).option(
5842
+ "--height <px>",
5843
+ "source height hint (server measurement is required)"
5844
+ ).option(
5845
+ "--source-fps <n>",
5846
+ "source frame rate hint (compatibility only; server measurement controls billing)"
5847
+ ).option("--download <path>", "download the result").option("--no-wait", "submit and return the job id immediately").action(async function(source) {
5848
+ const ctx = buildContext(this);
5849
+ const opts = this.opts();
5850
+ const [url] = await resolveRefs(ctx, [source]);
5851
+ capture("cli_upscale", {
5852
+ kind: "video",
5853
+ mode: opts.mode ?? "generative"
5854
+ });
5855
+ const submitted = await ctx.client.callTool(
5856
+ "upscale_video",
5857
+ compact({
5858
+ video_url: url,
5859
+ scale: opts.scale,
5860
+ target_resolution: opts.resolution?.toLowerCase(),
5861
+ mode: opts.mode,
5862
+ model: opts.model,
5863
+ target_fps: numOpt(opts.fps),
5864
+ prompt: opts.prompt,
5865
+ creativity: numOpt(opts.creativity),
5866
+ realism: numOpt(opts.realism),
5867
+ sharp: numOpt(opts.sharp),
5868
+ softness: numOpt(opts.softness),
5869
+ compression: numOpt(opts.compression),
5870
+ noise: numOpt(opts.noise),
5871
+ halo: numOpt(opts.halo),
5872
+ grain: numOpt(opts.grain),
5873
+ recover_detail: numOpt(opts.recoverDetail),
5874
+ session_id: sessionArg(this, opts),
5875
+ duration_seconds: numOpt(opts.duration),
5876
+ video_width: numOpt(opts.width),
5877
+ video_height: numOpt(opts.height),
5878
+ source_fps: numOpt(opts.sourceFps)
5879
+ })
5880
+ );
5881
+ await handleAsyncJob(ctx, submitted, {
5882
+ wait: opts.wait !== false,
5883
+ download: opts.download,
5884
+ label: "Upscaling video"
5885
+ });
5886
+ });
5887
+ program.command("interpolate <url|file>").description(
5888
+ "Raise a video's frame rate or make slow motion with Topaz (async; waits by default)"
5889
+ ).option(
5890
+ "--model <Apollo|Chronos|Aion>",
5891
+ "interpolation model (default Apollo)"
5892
+ ).option("--fps <n>", "target frame rate 24-120 (default 60)").option("--slowdown <1-8>", "slow-motion factor (default 1)").option(
5893
+ "--session <id>",
5894
+ "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
5895
+ process.env.VIDEODRAFT_SESSION
5896
+ ).option(
5897
+ "--duration <seconds>",
5898
+ "source duration hint (compatibility only; server must probe MP4/MOV up to 100MB)"
5899
+ ).option(
5900
+ "--width <px>",
5901
+ "source width hint (server measurement is required)"
5902
+ ).option(
5903
+ "--height <px>",
5904
+ "source height hint (server measurement is required)"
5905
+ ).option(
5906
+ "--source-fps <n>",
5907
+ "source frame rate hint (compatibility only; server measurement controls billing)"
5908
+ ).option("--download <path>", "download the result").option("--no-wait", "submit and return the job id immediately").action(async function(source) {
5909
+ const ctx = buildContext(this);
5910
+ const opts = this.opts();
5911
+ const [url] = await resolveRefs(ctx, [source]);
5912
+ capture("cli_interpolate", { model: opts.model ?? "Apollo" });
5913
+ const submitted = await ctx.client.callTool(
5914
+ "interpolate_video",
5915
+ compact({
5916
+ video_url: url,
5917
+ model: opts.model,
5918
+ target_fps: numOpt(opts.fps),
5919
+ slowdown_factor: numOpt(opts.slowdown),
5920
+ session_id: sessionArg(this, opts),
5921
+ duration_seconds: numOpt(opts.duration),
5922
+ video_width: numOpt(opts.width),
5923
+ video_height: numOpt(opts.height),
5924
+ source_fps: numOpt(opts.sourceFps)
5925
+ })
5926
+ );
5927
+ await handleAsyncJob(ctx, submitted, {
5928
+ wait: opts.wait !== false,
5929
+ download: opts.download,
5930
+ label: "Interpolating video"
5931
+ });
5932
+ });
5933
+ }
5934
+
5935
+ // src/commands/account.ts
5936
+ var VOICEOVER_COST_MODELS = ["voiceover", "tts", "speech", "text-to-speech"];
5937
+ var CLONED_VOICEOVER_COST_MODELS = [
5938
+ "voiceover-cloned",
5939
+ "cloned-voice",
5940
+ "custom-voice"
5941
+ ];
5942
+ var TURBO_VOICEOVER_COST_MODELS = ["voiceover-turbo", "tts-turbo"];
5943
+ function voiceoverCostModelId(model, mode) {
5944
+ if (!model) return void 0;
5945
+ const cloned = CLONED_VOICEOVER_COST_MODELS.includes(model);
5946
+ const turbo = TURBO_VOICEOVER_COST_MODELS.includes(model);
5947
+ if (!cloned && !turbo && !VOICEOVER_COST_MODELS.includes(model)) {
5948
+ return void 0;
5949
+ }
5950
+ const voiceMode = parseVoiceoverMode(mode, "--mode for a voiceover estimate");
5951
+ if (cloned && voiceMode === "turbo") {
5952
+ throw new UsageError(
5953
+ `Cloned custom-* voices have no Turbo mode, so ${model} cannot take --mode turbo.`,
5954
+ 'Drop --mode for the cloned rate, or run "costs voiceover --mode turbo" for ElevenLabs Turbo.'
5955
+ );
5956
+ }
5957
+ if (turbo && voiceMode === "standard") {
5958
+ throw new UsageError(
5959
+ `${model} already quotes Turbo, so it cannot take --mode standard.`,
5960
+ 'Run "costs voiceover" for the standard rate.'
5961
+ );
5962
+ }
5963
+ return voiceMode === "turbo" && !turbo ? "voiceover-turbo" : model;
5964
+ }
5965
+ function registerAccountCommands(program) {
5966
+ program.command("credits").description("Show your credit balance").action(async function() {
5967
+ const ctx = buildContext(this);
5968
+ const balance = await ctx.client.callTool("get_credits_balance");
5969
+ emit(ctx.out, balance, (o) => {
5970
+ kv(o, [
5971
+ ["Plan", balance?.planId],
5972
+ ["Available credits", balance?.availableCredits],
5973
+ ["Monthly allowance", balance?.totalCreditsMonthly],
5974
+ ["Used this month", balance?.monthlyCreditsUsed],
5975
+ ["Bonus credits", balance?.bonusCredits],
5976
+ ["Bonus expiry", balance?.bonusCreditsExpiry],
5977
+ ["Last monthly reset", balance?.lastMonthlyReset],
5978
+ ["Next monthly reset", balance?.nextMonthlyReset]
5979
+ ]);
5980
+ });
5981
+ });
5982
+ program.command("costs [model]").description(
5983
+ "Show credit costs. Pass a model id, or an image display name, plus settings for an exact estimate"
5984
+ ).option("--type <type>", "image | video | audio").option("--duration <seconds>", "video/audio duration in seconds").option(
5985
+ "--length <seconds>",
5986
+ "ElevenLabs Music output length in seconds (for a composition plan, the sum of its sections)"
5987
+ ).option(
5988
+ "--chars <n>",
5989
+ 'character count (ElevenLabs Dialogue, or voiceover TTS via model id "voiceover", "voiceover-cloned" or "voiceover-turbo")'
5990
+ ).option("--resolution <res>", 'e.g. "720p", "1080p", "1K", "2K"').option("--quality <tier>", 'e.g. "standard", "pro", "fast"').option("--ar <ratio>", "image aspect ratio for accurate credit quotes").option(
5991
+ "--rendering-speed <tier>",
5992
+ 'image speed/cost tier, e.g. Ideogram V4 "Turbo"/"Balanced"/"Quality"'
5993
+ ).option("--audio", "include native model audio in the estimate").option("--no-audio", "exclude native model audio").option(
5994
+ "--voice-control",
5995
+ "Kling element voice_id pricing (V3 Standard/Pro only; O3/4K are unavailable)"
5996
+ ).option(
5997
+ "--allow-real-people",
5998
+ "Seedance 2.x: estimate the higher tier-specific Fal rate (the default, matching AI Studio)"
5999
+ ).option(
6000
+ "--no-allow-real-people",
6001
+ "Seedance 2.x: estimate the lower Byteplus-only rate"
6002
+ ).option(
6003
+ "--ref-images <n>",
6004
+ "input/reference image count for MiniMax H3, MiniMax H3 Max, or Grok 1.5"
6005
+ ).option(
6006
+ "--ref-video-seconds <seconds>",
6007
+ "MiniMax H3 / H3 Max combined reference-video duration"
6008
+ ).option("--num <n>", "image batch size").option(
6009
+ "--width <px>",
6010
+ "Topaz: source width (with --height gives an exact quote)"
6011
+ ).option("--height <px>", "Topaz: source height").option(
6012
+ "--source-fps <n>",
6013
+ "Topaz video upscale: source frame rate (>30 bills the 60fps rate)"
6014
+ ).option("--scale <factor>", 'Topaz: "1x" | "2x" | "4x" or a factor 1-4').option("--fps <n>", "Topaz: delivered / target frame rate").option(
6015
+ "--mode <mode>",
6016
+ "Topaz upscale mode: generative | precision | creative. For voiceover, the ElevenLabs TTS mode: standard (Eleven v4, 10 credits per 1000 chars) | turbo (Eleven v4 Turbo, 5 credits per 1000 chars)"
6017
+ ).option(
6018
+ "--topaz-model <name>",
6019
+ 'Topaz model inside the mode, or "Apollo" | "Chronos" | "Aion" for interpolation'
6020
+ ).option("--slowdown <1-8>", "Topaz interpolate: slow-motion factor").action(async function(model) {
5939
6021
  const ctx = buildContext(this);
5940
6022
  const opts = this.opts();
5941
- capture("cli_generate", { kind: "sound_effect" });
6023
+ const voiceoverModelId = voiceoverCostModelId(model, opts.mode);
5942
6024
  const result = await ctx.client.callTool(
5943
- "generate_sound_effect",
6025
+ "get_model_costs",
5944
6026
  compact({
5945
- prompt: promptWords.join(" "),
6027
+ model_id: voiceoverModelId ?? model,
6028
+ type: opts.type,
5946
6029
  duration_seconds: opts.duration ? Number(opts.duration) : void 0,
5947
- prompt_influence: opts.influence ? Number(opts.influence) : void 0,
5948
- project_id: opts.project,
5949
- session_id: sessionArg(this, opts)
6030
+ length_seconds: opts.length ? Number(opts.length) : void 0,
6031
+ characters: opts.chars ? Number(opts.chars) : void 0,
6032
+ resolution: opts.resolution,
6033
+ quality: opts.quality,
6034
+ aspect_ratio: opts.ar,
6035
+ rendering_speed: opts.renderingSpeed,
6036
+ generate_audio: opts.audio,
6037
+ voice_control: opts.voiceControl,
6038
+ allow_real_people: opts.allowRealPeople,
6039
+ reference_image_count: opts.refImages ? Number(opts.refImages) : void 0,
6040
+ reference_video_duration_seconds: opts.refVideoSeconds ? Number(opts.refVideoSeconds) : void 0,
6041
+ num_images: opts.num ? Number(opts.num) : void 0,
6042
+ source_width: opts.width ? Number(opts.width) : void 0,
6043
+ source_height: opts.height ? Number(opts.height) : void 0,
6044
+ source_fps: opts.sourceFps ? Number(opts.sourceFps) : void 0,
6045
+ scale: opts.scale,
6046
+ target_resolution: opts.resolution && /^(720p|1080p|4k)$/i.test(opts.resolution) ? opts.resolution.toLowerCase() : void 0,
6047
+ target_fps: opts.fps ? Number(opts.fps) : void 0,
6048
+ // A voiceover's --mode is already folded into its model id.
6049
+ mode: voiceoverModelId ? void 0 : opts.mode,
6050
+ topaz_model: opts.topazModel,
6051
+ slowdown_factor: opts.slowdown ? Number(opts.slowdown) : void 0
5950
6052
  })
5951
6053
  );
5952
- const urls = extractOutputUrls(result);
5953
- let downloaded;
5954
- if (opts.download && urls.length > 0) {
5955
- downloaded = await downloadOutputs(urls, opts.download, {
5956
- name: "sound-effect"
5957
- });
5958
- }
5959
- const media = buildMediaDescriptors(urls, "audio");
5960
- emit(
5961
- ctx.out,
5962
- { ...result, downloaded_files: downloaded, output_media: media },
5963
- (o) => {
5964
- for (const url of urls) process.stdout.write(`${url}
5965
- `);
5966
- for (const f of downloaded ?? []) note(o, fmt.dim(o, savedLine(f)));
5967
- }
5968
- );
6054
+ emit(ctx.out, result);
5969
6055
  });
5970
- generate.command("dialogue").description(
5971
- "Generate multi-speaker dialogue (ElevenLabs Text-to-Dialogue). Repeat --line."
6056
+ program.command("models [kind]").description(
6057
+ "List available models: image | video | audio | 3d | voices | styles (default: image + video + audio)"
5972
6058
  ).option(
5973
- "--line <voiceId:text>",
5974
- 'a dialogue line as "voiceId:text" (repeatable); accepts raw ElevenLabs IDs or elevenlabs-<id>, including voices outside the catalog; must be accessible to the provider account',
5975
- collect,
5976
- []
5977
- ).option("--stability <0|0.5|1>", "voice stability").option("--language <iso>", "ISO 639-1 language code").option("--project <id>", "link to a project's AI Studio session").option(
5978
- "--session <id>",
5979
- "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
5980
- process.env.VIDEODRAFT_SESSION
5981
- ).option("--download <path>", "download the audio file").action(async function() {
6059
+ "--category <name>",
6060
+ "video only: generation | video_edit | motion_control | avatar_lipsync | upscale"
6061
+ ).action(async function(kind) {
5982
6062
  const ctx = buildContext(this);
5983
6063
  const opts = this.opts();
5984
- const lines = opts.line.map((raw) => {
5985
- const i = raw.indexOf(":");
5986
- if (i < 0) {
5987
- throw new Error(`--line must be "voiceId:text" (got "${raw}")`);
5988
- }
5989
- return {
5990
- voice_id: raw.slice(0, i).trim(),
5991
- text: raw.slice(i + 1).trim()
5992
- };
5993
- });
5994
- if (lines.length === 0)
5995
- throw new Error("at least one --line is required");
5996
- capture("cli_generate", { kind: "dialogue" });
5997
- const result = await ctx.client.callTool(
5998
- "generate_dialogue",
5999
- compact({
6000
- lines,
6001
- stability: opts.stability !== void 0 ? Number(opts.stability) : void 0,
6002
- language_code: opts.language,
6003
- project_id: opts.project,
6004
- session_id: sessionArg(this, opts)
6005
- })
6006
- );
6007
- const urls = extractOutputUrls(result);
6008
- let downloaded;
6009
- if (opts.download && urls.length > 0) {
6010
- downloaded = await downloadOutputs(urls, opts.download, {
6011
- name: "dialogue"
6012
- });
6064
+ const wanted = kind ?? "all";
6065
+ if (!["all", "image", "video", "audio", "3d", "voices", "styles"].includes(
6066
+ wanted
6067
+ )) {
6068
+ throw new UsageError(
6069
+ `Unknown model kind "${wanted}". Use image, video, audio, 3d, voices, or styles.`
6070
+ );
6013
6071
  }
6014
- const media = buildMediaDescriptors(urls, "audio");
6015
- emit(
6016
- ctx.out,
6017
- { ...result, downloaded_files: downloaded, output_media: media },
6018
- (o) => {
6019
- for (const url of urls) process.stdout.write(`${url}
6072
+ const videoCategories = /* @__PURE__ */ new Set([
6073
+ "generation",
6074
+ "video_edit",
6075
+ "motion_control",
6076
+ "avatar_lipsync",
6077
+ "upscale"
6078
+ ]);
6079
+ if (opts.category && !videoCategories.has(opts.category)) {
6080
+ throw new UsageError(
6081
+ `Unknown video category "${opts.category}". Use generation, video_edit, motion_control, avatar_lipsync, or upscale.`
6082
+ );
6083
+ }
6084
+ if (opts.category && wanted !== "video" && wanted !== "all") {
6085
+ throw new UsageError(
6086
+ "--category is only valid for the video model catalog."
6087
+ );
6088
+ }
6089
+ const result = {};
6090
+ if (wanted === "image" || wanted === "all") {
6091
+ result.image = await ctx.client.callTool("list_available_image_models");
6092
+ }
6093
+ if (wanted === "video" || wanted === "all") {
6094
+ const video = await ctx.client.callTool(
6095
+ "list_available_video_models"
6096
+ );
6097
+ result.video = opts.category ? {
6098
+ ...video,
6099
+ // A card may belong to more than one category — gemini-omni-1.1-flash
6100
+ // is both the generation default and the preferred video_edit
6101
+ // model — and declares that in `categories`. Match either field so
6102
+ // dual-category cards surface under both.
6103
+ models: (video?.models ?? []).filter(
6104
+ (model) => Array.isArray(model.categories) ? model.categories.includes(opts.category) : model.category === opts.category
6105
+ ),
6106
+ selected_category: opts.category
6107
+ } : video;
6108
+ }
6109
+ if (wanted === "audio" || wanted === "all") {
6110
+ result.audio = await ctx.client.callTool("list_available_audio_models");
6111
+ }
6112
+ if (wanted === "3d") {
6113
+ result["3d"] = await ctx.client.callTool("get_3d_models");
6114
+ }
6115
+ if (wanted === "voices") {
6116
+ result.voices = await ctx.client.callTool("list_available_voices");
6117
+ }
6118
+ if (wanted === "styles") {
6119
+ result.styles = await ctx.client.callTool("list_available_styles");
6120
+ }
6121
+ emit(ctx.out, result, (o) => {
6122
+ for (const [section, payload] of Object.entries(result)) {
6123
+ const models = Array.isArray(payload) ? payload : payload?.models ?? payload?.voices ?? payload?.styles ?? [];
6124
+ process.stdout.write(`
6125
+ ${section.toUpperCase()}
6020
6126
  `);
6021
- for (const f of downloaded ?? []) note(o, fmt.dim(o, savedLine(f)));
6127
+ if (!Array.isArray(models) || models.length === 0) {
6128
+ process.stdout.write(JSON.stringify(payload, null, 2) + "\n");
6129
+ continue;
6130
+ }
6131
+ if (section === "3d") {
6132
+ table(
6133
+ o,
6134
+ ["model", "input", "default credits", "Fal endpoint"],
6135
+ models.flatMap(
6136
+ (model) => (model.inputs ?? []).map((input) => [
6137
+ String(model.id ?? ""),
6138
+ String(input.input_mode ?? ""),
6139
+ String(input.default_credits ?? ""),
6140
+ String(input.endpoint_id ?? "")
6141
+ ])
6142
+ )
6143
+ );
6144
+ const rig = payload?.rigging;
6145
+ if (rig)
6146
+ note(
6147
+ o,
6148
+ `Humanoid rigging: ${rig.credits} credits; preset animation adds ${rig.animation_addon_credits} credits.`
6149
+ );
6150
+ note(
6151
+ o,
6152
+ "Use --json for exact options, defaults, input limits, and pricing notes. Use generate 3d --estimate for your selected recipe and BYOK status."
6153
+ );
6154
+ continue;
6155
+ }
6156
+ const rows = models.map((m) => [
6157
+ String(m.id ?? m.model_id ?? m.voice_id ?? ""),
6158
+ String(m.name ?? "").slice(0, 40),
6159
+ String(m.category ?? ""),
6160
+ String(m.tool ?? ""),
6161
+ String(m.credit_cost ?? m.cost ?? m.pricing?.summary ?? "")
6162
+ ]);
6163
+ table(
6164
+ o,
6165
+ section === "video" ? ["id", "name", "category", "tool", "cost"] : ["id", "name", "cost"],
6166
+ section === "video" ? rows : rows.map((row) => [row[0] ?? "", row[1] ?? "", row[4] ?? ""])
6167
+ );
6022
6168
  }
6023
- );
6169
+ });
6170
+ });
6171
+ program.command("workspaces").description("List your workspaces (the active one is bound to your token)").action(async function() {
6172
+ const ctx = buildContext(this);
6173
+ const result = await ctx.client.callTool("list_workspaces");
6174
+ const workspaces = result?.workspaces ?? result ?? [];
6175
+ emit(ctx.out, result, (o) => {
6176
+ table(
6177
+ o,
6178
+ ["id", "name", "role", "active"],
6179
+ workspaces.map((w) => [
6180
+ String(w.id ?? ""),
6181
+ String(w.name ?? ""),
6182
+ String(w.role ?? ""),
6183
+ w.is_active || w.active ? "\u2713" : ""
6184
+ ])
6185
+ );
6186
+ });
6024
6187
  });
6025
- generate.command("voice-changer <audio>").description("Restyle speech into another ElevenLabs voice (Voice Changer)").option(
6026
- "--voice <id>",
6027
- "target ElevenLabs voice ID, raw or elevenlabs-<id>; catalog membership is not required, but provider account access is (default Brittney, or an account voice with ElevenLabs BYOK)"
6028
- ).option(
6029
- "--duration <seconds>",
6030
- "length of the source audio in seconds (required, max 300)"
6031
- ).option("--remove-noise", "remove background noise from the input").option("--project <id>", "link to a project's AI Studio session").option(
6032
- "--session <id>",
6033
- "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
6034
- process.env.VIDEODRAFT_SESSION
6035
- ).option("--download <path>", "download the audio file").action(async function(source) {
6188
+ const sessions = program.command("sessions").description("AI Studio sessions");
6189
+ sessions.command("list", { isDefault: true }).description("List AI Studio sessions (owned + shared)").option("--project <id>", "filter to a project's sessions").option("--name <fragment>", "filter by name (case-insensitive)").option("--limit <n>", "max rows (default 50)").option("--offset <n>", "pagination offset (default 0)").action(async function() {
6036
6190
  const ctx = buildContext(this);
6037
6191
  const opts = this.opts();
6038
- if (!opts.duration) {
6039
- throw new Error(
6040
- "--duration <seconds> is required (length of the source audio)"
6041
- );
6042
- }
6043
- const [audioUrl] = await resolveRefs(ctx, [source]);
6044
- capture("cli_generate", { kind: "voice_changer" });
6045
6192
  const result = await ctx.client.callTool(
6046
- "change_voice",
6193
+ "list_ai_studio_sessions",
6047
6194
  compact({
6048
- audio_url: audioUrl,
6049
- voice_id: opts.voice,
6050
- duration_seconds: Number(opts.duration),
6051
- remove_background_noise: opts.removeNoise ? true : void 0,
6052
6195
  project_id: opts.project,
6053
- session_id: sessionArg(this, opts)
6196
+ name: opts.name,
6197
+ limit: opts.limit ? Number(opts.limit) : void 0,
6198
+ offset: opts.offset ? Number(opts.offset) : void 0
6054
6199
  })
6055
6200
  );
6056
- const urls = extractOutputUrls(result);
6057
- let downloaded;
6058
- if (opts.download && urls.length > 0) {
6059
- downloaded = await downloadOutputs(urls, opts.download, {
6060
- name: "voice-changed"
6061
- });
6062
- }
6063
- const media = buildMediaDescriptors(urls, "audio");
6064
- emit(
6065
- ctx.out,
6066
- { ...result, downloaded_files: downloaded, output_media: media },
6067
- (o) => {
6068
- for (const url of urls) process.stdout.write(`${url}
6069
- `);
6070
- for (const f of downloaded ?? []) note(o, fmt.dim(o, savedLine(f)));
6201
+ const rows = result?.sessions ?? result ?? [];
6202
+ emit(ctx.out, result, (o) => {
6203
+ if (!Array.isArray(rows)) return;
6204
+ table(
6205
+ o,
6206
+ ["id", "name", "role", "items", "created"],
6207
+ rows.map((s) => [
6208
+ String(s.id ?? s.session_id ?? ""),
6209
+ String(s.name ?? "").slice(0, 40),
6210
+ String(s.role ?? (s.is_owner === false ? "member" : "owner")),
6211
+ `${s.image_count ?? 0}i/${s.video_count ?? 0}v/${s.sound_count ?? 0}s`,
6212
+ String(s.created_at ?? s.createdAt ?? "").slice(0, 19)
6213
+ ])
6214
+ );
6215
+ });
6216
+ });
6217
+ sessions.command("current").description("Show the current scope's automatic AI Studio session").action(async function() {
6218
+ const ctx = buildContext(this);
6219
+ const knownIds = /* @__PURE__ */ new Set();
6220
+ let listed = false;
6221
+ if (ctx.session) {
6222
+ try {
6223
+ for (let offset = 0; offset < 2e3; ) {
6224
+ const page = await ctx.client.callTool(
6225
+ "list_ai_studio_sessions",
6226
+ { limit: 200, offset }
6227
+ );
6228
+ const pageRows = page?.sessions ?? [];
6229
+ for (const row of pageRows) {
6230
+ const id = row?.id ?? row?.session_id;
6231
+ if (typeof id === "string") knownIds.add(id);
6232
+ }
6233
+ if (pageRows.length < 200) break;
6234
+ offset += pageRows.length;
6235
+ }
6236
+ listed = true;
6237
+ } catch {
6238
+ listed = false;
6071
6239
  }
6240
+ }
6241
+ const info = ctx.session?.describe();
6242
+ const record = info?.record ?? null;
6243
+ const active = Boolean(record) && !info?.expired;
6244
+ const sessionId = active ? info?.sessionId ?? null : null;
6245
+ const created = Boolean(sessionId) && listed && knownIds.has(sessionId);
6246
+ const result = {
6247
+ enabled: Boolean(ctx.session),
6248
+ scope: info?.scope ?? null,
6249
+ active,
6250
+ expired: Boolean(info?.expired),
6251
+ session_id: sessionId,
6252
+ /** False until naming or the first generation creates the row. */
6253
+ session_created: created,
6254
+ url: sessionId && created ? `${ctx.baseUrl}/ai-studio?session=${sessionId}` : null,
6255
+ created_at: active ? record?.createdAt ?? null : null,
6256
+ last_used_at: active ? record?.lastUsedAt ?? null : null,
6257
+ file: info?.file ?? null,
6258
+ note: ctx.session ? active ? created ? "Project-less generations in this scope share this AI Studio session. `videodraft sessions reset` starts a new one; --session <id> pins a specific session." : "Session id reserved for this scope; name it with `videodraft sessions name <title>` or generate the first asset to create it. `videodraft sessions reset` starts a new one; --session <id> pins a specific session." : info?.expired ? "The previous session idled out (12h); the next command starts a new one. --session <id> pins a specific session instead." : "No session yet; the next command creates one. --session <id> pins a specific session instead." : 'Disabled via VIDEODRAFT_NO_SESSION; generations fall back to the shared "Agent (MCP)" session unless --session is passed.'
6259
+ };
6260
+ emit(ctx.out, result, () => {
6261
+ const state = !result.enabled ? "disabled" : result.active ? `${result.session_created ? "active" : "reserved"} session=${result.session_id ?? "?"}` : result.expired ? "expired" : "none yet";
6262
+ process.stdout.write(
6263
+ `${state} scope=${result.scope ?? "-"}
6264
+ ${result.url ? `${result.url}
6265
+ ` : ""}${result.note}
6266
+ `
6267
+ );
6268
+ });
6269
+ });
6270
+ sessions.command("name <name>").description("Name the automatic AI Studio session before first use").action(async function(name) {
6271
+ if (process.env.VIDEODRAFT_SESSION?.trim()) {
6272
+ throw new UsageError(
6273
+ "Cannot name the automatic session while VIDEODRAFT_SESSION is set. Unset it first, or rename the pinned session in AI Studio."
6274
+ );
6275
+ }
6276
+ const ctx = buildContext(this);
6277
+ const result = await ctx.client.callTool(
6278
+ "name_current_ai_studio_session",
6279
+ { name }
6072
6280
  );
6281
+ emit(ctx.out, result, () => {
6282
+ const session = result?.session ?? result;
6283
+ process.stdout.write(
6284
+ `${String(session?.name ?? name)} ${String(session?.id ?? result?.session_id ?? "")}
6285
+ `
6286
+ );
6287
+ });
6073
6288
  });
6074
- generate.command("dub <media>").description(
6075
- "Dub a video/audio file into another language (ElevenLabs Dubbing)"
6076
- ).option(
6077
- "--to <iso>",
6078
- "target language ISO 639-1 code, e.g. es or te (required)"
6079
- ).option(
6080
- "--from <iso>",
6081
- "source language ISO 639-1 code (auto-detected if omitted)"
6082
- ).option("--type <audio|video>", "source media type override").option(
6083
- "--duration <seconds>",
6084
- "length of the source media in seconds (required, max 300)"
6085
- ).option("--speakers <n>", "number of speakers (auto-detected if omitted)").option("--project <id>", "link to a project's AI Studio session").option(
6086
- "--session <id>",
6087
- "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
6088
- process.env.VIDEODRAFT_SESSION
6089
- ).option("--download <path>", "download the dubbed file").action(async function(source) {
6289
+ sessions.command("reset").description(
6290
+ "Forget this connection session so the next generation starts a fresh AI Studio session (--all: every scope)"
6291
+ ).option("--all", "reset every stored connection session").action(async function() {
6090
6292
  const ctx = buildContext(this);
6091
6293
  const opts = this.opts();
6092
- if (!opts.to) throw new Error("--to <iso> (target language) is required");
6093
- if (!opts.duration) {
6094
- throw new Error(
6095
- "--duration <seconds> is required (length of the source media)"
6096
- );
6294
+ let removed = 0;
6295
+ let declined = false;
6296
+ if (opts.all) {
6297
+ removed = resetAllConnectionSessions();
6298
+ } else if (ctx.session) {
6299
+ const had = Boolean(ctx.session.describe().record);
6300
+ const ran = ctx.session.reset();
6301
+ removed = had && ran ? 1 : 0;
6302
+ declined = had && !ran;
6097
6303
  }
6098
- const mediaType = inferDubMediaType(source, opts.type);
6099
- const [mediaUrl] = await resolveRefs(ctx, [source]);
6100
- capture("cli_generate", { kind: "dub" });
6304
+ const result = {
6305
+ reset: removed,
6306
+ declined,
6307
+ scope: opts.all ? "all" : ctx.session?.describe().scope ?? null
6308
+ };
6309
+ emit(ctx.out, result, () => {
6310
+ process.stdout.write(
6311
+ removed > 0 ? `Reset ${removed} connection session${removed === 1 ? "" : "s"}; the next generation starts a new AI Studio session.
6312
+ ` : declined ? "Could not reset: the session store is locked or unwritable; the session may still be in use.\n" : "Nothing to reset.\n"
6313
+ );
6314
+ });
6315
+ });
6316
+ sessions.command("create <name>").description(
6317
+ "Create an AI Studio session (reuse its id across standalone generations)"
6318
+ ).option("--project <id>", "attach the session to a project").action(async function(name) {
6319
+ const ctx = buildContext(this);
6101
6320
  const result = await ctx.client.callTool(
6102
- "dub_media",
6103
- compact({
6104
- video_url: mediaType === "video" ? mediaUrl : void 0,
6105
- audio_url: mediaType === "audio" ? mediaUrl : void 0,
6106
- target_lang: opts.to,
6107
- source_lang: opts.from,
6108
- num_speakers: opts.speakers ? Number(opts.speakers) : void 0,
6109
- duration_seconds: Number(opts.duration),
6110
- project_id: opts.project,
6111
- session_id: sessionArg(this, opts)
6112
- })
6321
+ "create_ai_studio_session",
6322
+ compact({ name, project_id: this.opts().project })
6113
6323
  );
6114
- const urls = extractOutputUrls(result);
6115
- let downloaded;
6116
- if (opts.download && urls.length > 0) {
6117
- downloaded = await downloadOutputs(urls, opts.download, {
6118
- name: "dubbed"
6119
- });
6120
- }
6121
- const media = buildMediaDescriptors(urls, mediaType);
6122
- emit(
6123
- ctx.out,
6124
- { ...result, downloaded_files: downloaded, output_media: media },
6125
- (o) => {
6126
- for (const url of urls) process.stdout.write(`${url}
6324
+ emit(ctx.out, result, (o) => {
6325
+ const id = result?.session?.id ?? result?.session_id ?? result?.id;
6326
+ process.stdout.write(`${id ?? JSON.stringify(result)}
6127
6327
  `);
6128
- for (const f of downloaded ?? []) note(o, fmt.dim(o, savedLine(f)));
6129
- }
6130
- );
6328
+ });
6131
6329
  });
6132
- const upscale = program.command("upscale").description("Upscale / enhance images and videos (Topaz)");
6133
- const numOpt = (v) => v === void 0 || v === null || v === "" ? void 0 : Number(v);
6134
- upscale.command("image <url|file>").description(
6135
- "Enhance or upscale an existing image with Topaz (synchronous)"
6136
- ).option("--scale <factor>", '"1x" | "2x" | "4x" (default 2x)').option(
6137
- "--mode <mode>",
6138
- "generative (default; Wonder 3.5, best for AI images) | precision (faithful, cheapest; real photos) | creative (Bloom, artistic)"
6139
- ).option(
6140
- "--model <name>",
6141
- 'Topaz model inside the mode, e.g. "High Fidelity V3", "Wonder 3.5", "Redefine", "Bloom 2" (see: videodraft models image --json)'
6142
- ).option("--format <jpeg|png>", "output format (default jpeg)").option("--no-face-enhance", "disable Topaz face recovery").option("--face-strength <0-1>", "face recovery strength (default 0.8)").option("--sharpen <0-1>", "extra sharpening (precision/generative)").option("--denoise <0-1>", "noise reduction (precision/generative)").option(
6143
- "--fix-compression <0-1>",
6144
- "compression-artifact repair (precision)"
6145
- ).option(
6146
- "--prompt <text>",
6147
- "guide detail (generative Redefine / creative Bloom)"
6148
- ).option("--creativity <n>", "generative 1-6 / creative 1-9").option("--texture <1-5>", "texture amount (generative Redefine)").option(
6149
- "--width <px>",
6150
- "source width hint (server must verify dimensions; max source 50MB)"
6151
- ).option(
6152
- "--height <px>",
6153
- "source height hint (server must verify dimensions; max source 50MB)"
6154
- ).option(
6155
- "--session <id>",
6156
- "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
6157
- process.env.VIDEODRAFT_SESSION
6158
- ).option("--download <path>", "download the result").action(async function(source) {
6330
+ }
6331
+
6332
+ // src/commands/projects.ts
6333
+ import { spawn as spawn2 } from "child_process";
6334
+ function registerProjectCommands(program) {
6335
+ const projects = program.command("projects").description("List and manage projects");
6336
+ projects.command("list", { isDefault: true }).description("List projects").option("--limit <n>", "max projects (default 50)").option("--offset <n>", "pagination offset").option("--favorites", "only favorited projects").action(async function() {
6159
6337
  const ctx = buildContext(this);
6160
6338
  const opts = this.opts();
6161
- const [url] = await resolveRefs(ctx, [source]);
6162
- capture("cli_upscale", {
6163
- kind: "image",
6164
- mode: opts.mode ?? "generative"
6165
- });
6166
6339
  const result = await ctx.client.callTool(
6167
- "upscale_image",
6340
+ "list_projects",
6168
6341
  compact({
6169
- image_url: url,
6170
- scale: opts.scale,
6171
- mode: opts.mode,
6172
- model: opts.model,
6173
- output_format: opts.format,
6174
- face_enhancement: opts.faceEnhance === false ? false : void 0,
6175
- face_enhancement_strength: numOpt(opts.faceStrength),
6176
- sharpen: numOpt(opts.sharpen),
6177
- denoise: numOpt(opts.denoise),
6178
- fix_compression: numOpt(opts.fixCompression),
6179
- prompt: opts.prompt,
6180
- creativity: numOpt(opts.creativity),
6181
- texture: numOpt(opts.texture),
6182
- image_width: numOpt(opts.width),
6183
- image_height: numOpt(opts.height),
6184
- session_id: sessionArg(this, opts)
6342
+ limit: opts.limit ? Number(opts.limit) : void 0,
6343
+ offset: opts.offset ? Number(opts.offset) : void 0,
6344
+ favorites_only: opts.favorites || void 0
6185
6345
  })
6186
6346
  );
6187
- const urls = extractOutputUrls(result);
6188
- let downloaded;
6189
- if (opts.download && urls.length > 0) {
6190
- downloaded = await downloadOutputs(urls, opts.download, {
6191
- name: "upscaled"
6192
- });
6347
+ const rows = result?.projects ?? [];
6348
+ emit(ctx.out, result, (o) => {
6349
+ table(
6350
+ o,
6351
+ ["id", "title", "status", "modified"],
6352
+ rows.map((p) => [
6353
+ String(p.id ?? ""),
6354
+ String(p.title ?? "Untitled").slice(0, 44),
6355
+ String(p.status ?? ""),
6356
+ String(p.lastModified ?? "").slice(0, 19)
6357
+ ])
6358
+ );
6359
+ });
6360
+ });
6361
+ projects.command("get <project_id>").description("Fetch a project (summary view; --raw for the editable blob)").option("--raw", "return the raw editable JSON blob").action(async function(projectId) {
6362
+ const ctx = buildContext(this);
6363
+ const result = await ctx.client.callTool("get_project", {
6364
+ project_id: projectId,
6365
+ view: this.opts().raw ? "raw" : "summary"
6366
+ });
6367
+ emit(ctx.out, result);
6368
+ });
6369
+ projects.command("delete <project_id>").description("Permanently delete a project (cannot be undone)").option("--yes", "skip the confirmation").action(async function(projectId) {
6370
+ const ctx = buildContext(this);
6371
+ if (!this.opts().yes) {
6372
+ throw new CliError(
6373
+ "Refusing to delete without --yes. Deletion is permanent for every collaborator.",
6374
+ EXIT.USAGE
6375
+ );
6193
6376
  }
6194
- const media = buildMediaDescriptors(urls, "image");
6377
+ const result = await ctx.client.callTool("delete_project", {
6378
+ project_id: projectId
6379
+ });
6195
6380
  emit(
6196
6381
  ctx.out,
6197
- { ...result, downloaded_files: downloaded, output_media: media },
6198
- (o) => {
6199
- for (const u of urls) process.stdout.write(`${u}
6200
- `);
6201
- for (const f of downloaded ?? []) note(o, fmt.dim(o, savedLine(f)));
6202
- }
6382
+ result,
6383
+ (o) => note(o, fmt.green(o, `Deleted project ${projectId}.`))
6203
6384
  );
6204
6385
  });
6205
- upscale.command("video <url|file>").description(
6206
- "Enhance or upscale an existing video with Topaz (async; waits by default)"
6207
- ).option(
6208
- "--resolution <720p|1080p|4k>",
6209
- "output preset by short edge (preferred over --scale)"
6210
- ).option(
6211
- "--scale <factor>",
6212
- '"1x" | "2x" | "4x" (default 2x when no --resolution)'
6213
- ).option(
6214
- "--mode <mode>",
6215
- "generative (default; Starlight Precise 2.6, best for AI clips) | precision (Proteus, 6x cheaper; real footage) | creative (Astra 2)"
6216
- ).option(
6217
- "--model <name>",
6218
- 'Topaz model inside the mode, e.g. "Proteus", "Gaia 2", "Starlight Precise 2.6", "Starlight Fast 2" (see: videodraft models video --category upscale --json)'
6219
- ).option("--fps <n>", "deliver at this frame rate (24-120; 60 doubles cost)").option("--prompt <text>", "creative (Astra 2) only").option("--creativity <0-1>", "creative (Astra 2) only").option("--realism <0-1>", "creative (Astra 2) only").option("--sharp <0-1>", "creative (Astra 2) only").option("--softness <1-5>", "generative (Starlight Precise 2.6) only").option("--compression <0-1>", "precision only").option("--noise <0-1>", "precision only").option("--halo <0-1>", "precision only").option("--grain <0-0.1>", "precision only").option("--recover-detail <0-1>", "precision only").option(
6220
- "--session <id>",
6221
- "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
6222
- process.env.VIDEODRAFT_SESSION
6223
- ).option(
6224
- "--duration <seconds>",
6225
- "source duration hint (compatibility only; server must probe MP4/MOV up to 100MB)"
6226
- ).option(
6227
- "--width <px>",
6228
- "source width hint (server measurement is required)"
6229
- ).option(
6230
- "--height <px>",
6231
- "source height hint (server measurement is required)"
6232
- ).option(
6233
- "--source-fps <n>",
6234
- "source frame rate hint (compatibility only; server measurement controls billing)"
6235
- ).option("--download <path>", "download the result").option("--no-wait", "submit and return the job id immediately").action(async function(source) {
6386
+ projects.command("favorite <project_id>").description("Star or unstar a project").option("--remove", "unstar instead of star").action(async function(projectId) {
6236
6387
  const ctx = buildContext(this);
6237
- const opts = this.opts();
6238
- const [url] = await resolveRefs(ctx, [source]);
6239
- capture("cli_upscale", {
6240
- kind: "video",
6241
- mode: opts.mode ?? "generative"
6388
+ const favorite = !this.opts().remove;
6389
+ const result = await ctx.client.callTool("set_project_favorite", {
6390
+ project_id: projectId,
6391
+ favorite
6242
6392
  });
6243
- const submitted = await ctx.client.callTool(
6244
- "upscale_video",
6393
+ emit(
6394
+ ctx.out,
6395
+ result,
6396
+ (o) => note(o, favorite ? "Starred." : "Unstarred.")
6397
+ );
6398
+ });
6399
+ projects.command("open <project_id>").description("Open the project in your browser").action(async function(projectId) {
6400
+ const ctx = buildContext(this);
6401
+ const result = await ctx.client.callTool("get_project", {
6402
+ project_id: projectId,
6403
+ view: "summary"
6404
+ });
6405
+ const url = result?.urls?.storyboard ?? result?.urls?.script ?? `${ctx.baseUrl}/projects/${projectId}`;
6406
+ const cmd = process.platform === "darwin" ? "open" : process.platform === "win32" ? "start" : "xdg-open";
6407
+ try {
6408
+ const child = spawn2(cmd, [url], {
6409
+ stdio: "ignore",
6410
+ detached: true,
6411
+ shell: process.platform === "win32"
6412
+ });
6413
+ child.on("error", () => {
6414
+ });
6415
+ child.unref();
6416
+ } catch {
6417
+ }
6418
+ emit(ctx.out, { url }, (o) => note(o, url));
6419
+ });
6420
+ const checkpoint = program.command("checkpoint").description("Project version checkpoints");
6421
+ checkpoint.command("create <project_id>").description("Snapshot the project as a new version").option("--name <name>", "version name").option("--description <text>", "version description").action(async function(projectId) {
6422
+ const ctx = buildContext(this);
6423
+ const opts = this.opts();
6424
+ const result = await ctx.client.callTool(
6425
+ "create_project_checkpoint",
6245
6426
  compact({
6246
- video_url: url,
6247
- scale: opts.scale,
6248
- target_resolution: opts.resolution?.toLowerCase(),
6249
- mode: opts.mode,
6250
- model: opts.model,
6251
- target_fps: numOpt(opts.fps),
6252
- prompt: opts.prompt,
6253
- creativity: numOpt(opts.creativity),
6254
- realism: numOpt(opts.realism),
6255
- sharp: numOpt(opts.sharp),
6256
- softness: numOpt(opts.softness),
6257
- compression: numOpt(opts.compression),
6258
- noise: numOpt(opts.noise),
6259
- halo: numOpt(opts.halo),
6260
- grain: numOpt(opts.grain),
6261
- recover_detail: numOpt(opts.recoverDetail),
6262
- session_id: sessionArg(this, opts),
6263
- duration_seconds: numOpt(opts.duration),
6264
- video_width: numOpt(opts.width),
6265
- video_height: numOpt(opts.height),
6266
- source_fps: numOpt(opts.sourceFps)
6427
+ project_id: projectId,
6428
+ name: opts.name,
6429
+ description: opts.description
6267
6430
  })
6268
6431
  );
6269
- await handleAsyncJob(ctx, submitted, {
6270
- wait: opts.wait !== false,
6271
- download: opts.download,
6272
- label: "Upscaling video"
6432
+ emit(ctx.out, result);
6433
+ });
6434
+ checkpoint.command("list <project_id>").description("List a project's checkpoints").action(async function(projectId) {
6435
+ const ctx = buildContext(this);
6436
+ const result = await ctx.client.callTool("list_project_checkpoints", {
6437
+ project_id: projectId
6273
6438
  });
6439
+ emit(ctx.out, result);
6274
6440
  });
6275
- program.command("interpolate <url|file>").description(
6276
- "Raise a video's frame rate or make slow motion with Topaz (async; waits by default)"
6277
- ).option(
6278
- "--model <Apollo|Chronos|Aion>",
6279
- "interpolation model (default Apollo)"
6280
- ).option("--fps <n>", "target frame rate 24-120 (default 60)").option("--slowdown <1-8>", "slow-motion factor (default 1)").option(
6281
- "--session <id>",
6282
- "pin an AI Studio session id (default: the current connection scope; env VIDEODRAFT_SESSION)",
6283
- process.env.VIDEODRAFT_SESSION
6284
- ).option(
6285
- "--duration <seconds>",
6286
- "source duration hint (compatibility only; server must probe MP4/MOV up to 100MB)"
6287
- ).option(
6288
- "--width <px>",
6289
- "source width hint (server measurement is required)"
6290
- ).option(
6291
- "--height <px>",
6292
- "source height hint (server measurement is required)"
6441
+ checkpoint.command("restore <project_id> [version_number]").description(
6442
+ "Restore a project to a saved version (current state is snapshotted first)"
6293
6443
  ).option(
6294
- "--source-fps <n>",
6295
- "source frame rate hint (compatibility only; server measurement controls billing)"
6296
- ).option("--download <path>", "download the result").option("--no-wait", "submit and return the job id immediately").action(async function(source) {
6444
+ "--checkpoint-id <id>",
6445
+ "restore by checkpoint UUID instead of version number"
6446
+ ).action(async function(projectId, versionNumber) {
6297
6447
  const ctx = buildContext(this);
6298
6448
  const opts = this.opts();
6299
- const [url] = await resolveRefs(ctx, [source]);
6300
- capture("cli_interpolate", { model: opts.model ?? "Apollo" });
6301
- const submitted = await ctx.client.callTool(
6302
- "interpolate_video",
6449
+ if (!versionNumber && !opts.checkpointId) {
6450
+ throw new CliError(
6451
+ "Pass a version_number or --checkpoint-id.",
6452
+ EXIT.USAGE
6453
+ );
6454
+ }
6455
+ const result = await ctx.client.callTool(
6456
+ "restore_project_checkpoint",
6303
6457
  compact({
6304
- video_url: url,
6305
- model: opts.model,
6306
- target_fps: numOpt(opts.fps),
6307
- slowdown_factor: numOpt(opts.slowdown),
6308
- session_id: sessionArg(this, opts),
6309
- duration_seconds: numOpt(opts.duration),
6310
- video_width: numOpt(opts.width),
6311
- video_height: numOpt(opts.height),
6312
- source_fps: numOpt(opts.sourceFps)
6458
+ project_id: projectId,
6459
+ version_number: versionNumber ? Number(versionNumber) : void 0,
6460
+ checkpoint_id: opts.checkpointId
6313
6461
  })
6314
6462
  );
6315
- await handleAsyncJob(ctx, submitted, {
6316
- wait: opts.wait !== false,
6317
- download: opts.download,
6318
- label: "Interpolating video"
6319
- });
6463
+ emit(ctx.out, result);
6320
6464
  });
6321
6465
  }
6322
6466
 
@@ -6524,9 +6668,13 @@ function registerPipelineCommands(program) {
6524
6668
  ).option("--no-voiceover", "skip per-scene voiceovers + captions").option("--no-video-prompts", "skip advisory per-shot motion prompts").option(
6525
6669
  "--shot-duration <seconds>",
6526
6670
  "per-shot clip length for silent/no-voiceover scenes (default 3)"
6527
- ).option("--voice <id>", "TTS voice id for voiceovers").option("--language <bcp47>", "voiceover + caption language").option("--captions", "force burn captions").option("--no-captions", "force no captions (default follows voiceover)").action(async function(projectId) {
6671
+ ).option("--voice <id>", "TTS voice id for voiceovers").option(
6672
+ "--voice-mode <standard|turbo>",
6673
+ "ElevenLabs narration voices only: standard (default; Eleven v4, 10 credits per 1000 chars) | turbo (Eleven v4 Turbo, faster, 5 credits per 1000 chars)"
6674
+ ).option("--language <bcp47>", "voiceover + caption language").option("--captions", "force burn captions").option("--no-captions", "force no captions (default follows voiceover)").action(async function(projectId) {
6528
6675
  const ctx = buildContext(this);
6529
6676
  const opts = this.opts();
6677
+ const voiceMode = parseVoiceoverMode(opts.voiceMode, "--voice-mode");
6530
6678
  const captionsSrc = this.getOptionValueSource("captions");
6531
6679
  capture("cli_produce", { mode: opts.mode ?? "animatic" });
6532
6680
  const spin = spinner(
@@ -6547,6 +6695,7 @@ function registerPipelineCommands(program) {
6547
6695
  generate_video_prompts: opts.videoPrompts === false ? false : void 0,
6548
6696
  shot_duration: opts.shotDuration ? Number(opts.shotDuration) : void 0,
6549
6697
  voice_id: opts.voice,
6698
+ voice_mode: voiceMode,
6550
6699
  language: opts.language,
6551
6700
  show_captions: captionsSrc === "cli" ? opts.captions : void 0
6552
6701
  })
@@ -7607,8 +7756,8 @@ function bundledSkillDir() {
7607
7756
  );
7608
7757
  }
7609
7758
  function bundledSkillFiles() {
7610
- if ('{"SKILL.md":"---\\nname: videodraft\\ndescription: Create and edit AI videos, images, 3D meshes, Seed Audio, voiceovers, music, sound effects, dialogue, dubbing, storyboards, avatar videos, media upscales, and product/ad videos with VideoDraft. Use whenever the user mentions VideoDraft; asks to generate a video, image, 3D asset, audio asset, ad, explainer, storyboard, avatar, upscale, or batch/CI workflow; wants to download a video from YouTube, Instagram, TikTok, X or another site; or wants to assemble, cut, caption, mix, lay out, inspect, or export a native VideoDraft Editor timeline. Covers the cloud `videodraft` CLI/MCP and local headless `videodraft_editor` MCP. When the editor MCP is exposed, prefer it for production, timeline assembly, and export; use cloud production/export only when explicitly requested or the editor is unavailable.\\n---\\n\\n# VideoDraft\\n\\nVideoDraft is an AI video creation platform where asset generation is the priority lane:\\n\\n- **Asset generation**: standalone images, video clips, Seed Audio, voiceovers, music, sound effects, dialogue, voice-changed audio, dubbed media, upscales, and image descriptions. This is the fastest and most important lane. Treat these as complete deliverables when the user asks for assets.\\n- **Asset I/O**: upload local files, download outputs, auto-upload local references, and save generated media where the user can see it.\\n- **3D assets**: standalone Meshy 7 and Tripo H3.1 meshes, multi-view generation, complete artifact downloads, and Meshy humanoid rigging through MCP/CLI. Read [references/3d.md](references/3d.md), or `videodraft skills show 3d`. These assets have their own saved library and do not require an AI Studio session or web viewer.\\n- **Native editing**: local `.vdproject` timelines, cuts, layouts, captions, effects, audio, and exports through the headless VideoDraft Editor. Inside VideoDraft ADE, this is the default production and export lane whenever `videodraft_editor` is available.\\n- **Hosted project production**: idea \u2192 script \u2192 storyboard \u2192 hosted production timeline \u2192 exported MP4. Use the early stages for scripts, storyboards, and generated assets when useful. Treat hosted production and export as a fallback when the native editor is unavailable, or as an explicit destination when the user asks for an editable web project or hosted workflow.\\n\\n## How to connect\\n\\nCloud generation has two equivalent surfaces (same backend, credits, and hosted projects). Native timeline editing is a separate local surface:\\n\\n1. **CLI** (preferred when you have a shell): run `videodraft` if it\'s on PATH; otherwise `npx -y videodraft@latest` runs it with no install (needs Node \u226520; the `-y` skips npx\'s install prompt so it runs non-interactively; the package is fetched on first use and cached). For heavy use, `npm install -g videodraft`. If there\'s no Node/shell here but the MCP connector below is available, use that instead; if neither works, tell the user how to install (https://videodraft.ai/cli).\\n - Auth \u2014 pick by context, don\'t guess:\\n \u2022 INTERACTIVE (a human is in the session, e.g. Claude Code / Codex): on exit code 3 (\\"not authenticated\\"), tell the user to run `videodraft login` in their terminal \u2014 it opens their browser for a one-click VideoDraft sign-in (OAuth), no key to copy. Wait for them to confirm it succeeded, then retry the command. This is the preferred path when the user is present.\\n \u2022 HEADLESS / CI (no browser): set `VIDEODRAFT_API_KEY=vd_mcp_...` (a token the user mints at https://app.videodraft.ai/mcp-keys).\\n \u2022 SECURITY: never ask the user to paste a `vd_mcp_...` token into the chat \u2014 use browser `login` or the env var so the token never lands in the transcript.\\n - Every command accepts `--json` (parse this, don\'t scrape text). Exit codes: 0 ok, 1 error, 2 usage, 3 auth (see Auth above), 4 insufficient credits (\u2192 tell the user, don\'t retry).\\n - Tool discovery: start with `videodraft tools list` for the grouped catalog, then narrow with `videodraft tools list --lane assets`, `--lane asset_io`, `--lane project_data`, or `--lane production`.\\n - Asset lane: `videodraft generate ...`, `videodraft edit video|motion`, `videodraft avatar ...`, `videodraft upscale ...`, `videodraft interpolate ...`, `videodraft upload`, and `videodraft download`.\\n - Full API access: `videodraft tools schema <name>`, `videodraft call <tool> --args \'<json>\'`.\\n2. **MCP connector**: if VideoDraft MCP tools (e.g. `generate_storyboard_from_idea`) are available, call them directly \u2014 the CLI\'s curated commands map 1:1 onto these tools.\\n3. **Native editor MCP** (`videodraft_editor`): prefer this for project production, timeline assembly, cutting, layouts, transitions, captions, audio placement, and final export. Inside VideoDraft ADE on a supported Mac, Claude, Codex, OpenCode and Grok receive it automatically in both Code and VideoDraft modes. It runs headlessly, so an Open Editor click is not required. Start with `project_manage` (`operation: \\"project\\"`, action `list`, `open`, or `create`); standalone asset generation remains in the cloud CLI or MCP.\\n\\nSend native editor edits serially, batching everything a request needs into one `edit_apply` recipe. Pass the latest `context` when a person may be editing at the same time, so a stale write is refused. See [references/editor.md](references/editor.md) for project selection, media import, timing units, edit recipes, verification, export, and the `videodraft-editor` terminal bridge.\\n\\nIf you are reading this skill through `videodraft skills show skill`, run `videodraft skills show editor` before native editor work to load that reference.\\n\\n**VideoDraft ADE routing rule:** the presence of `videodraft_editor` means the native editor is ready, even when no editor window is visible. Use cloud tools to generate or source assets and, when helpful, scripts or storyboards. Do not call hosted `produce_project` / `videodraft produce` or `export_video` / `videodraft export` by default. Import the assets into the native project, assemble there, and export with native `delivery_manage` `submit`. Use hosted production/export only when the user explicitly asks for the web workflow or the native editor tools are unavailable. Do not silently fall back to hosted production after a native tool error.\\n\\n## First decision: asset, hosted project, or native edit?\\n\\n- **One standalone asset** (image, clip, voiceover, music track, sound effect, dialogue track, voice-changed file, dubbed media file, upscale, or description): generate it directly. Do NOT create a project.\\n - `videodraft generate image \\"a red fox in snow, cinematic\\" --ar 16:9 --download ./out/`\\n - `videodraft generate video \\"slow dolly over a misty lake\\" --model gemini-omni-1.1-flash --duration 6 --download ./out/`\\n- **Any final video, production timeline, existing footage, local `.vdproject`, or hands-on edit**: use `videodraft_editor` when available. List or open the intended local project, or create a native project for a new production. The editor can work without showing its UI.\\n- **A small set of related assets**: still stay in the asset lane. They are grouped automatically (see [AI Studio sessions](#ai-studio-sessions)); give the current group one useful task-specific name with `videodraft sessions name \\"<name>\\"`. Switch to a project only when the deliverable matches the project criteria below or the user asks to attach the assets to one.\\n- **A generated multi-scene video / ad / explainer**: when the editor is available, use hosted tools only for any needed script, storyboard, shot planning, or generated assets; stop before hosted production, import the assets, and build/export the native timeline. A hosted project is optional unless the user wants the web project or its storyboard workflow.\\n- **A hosted web project or hosted export**: use the hosted pipeline only when the user explicitly asks for it or the native editor is unavailable.\\n- **Just a script** (no video asked for): A script-only request creates a script-stage project but stops at the script. Use `videodraft create \\"...\\" --script-only`; do not build a storyboard the user didn\'t ask for.\\n- **Iterating on existing work**: identify the surface first. Use `project_manage` `project` with `action: \\"list\\"` for native projects and `videodraft projects list` only for hosted work. Never create a replacement project just to change an existing one.\\n\\n## Choose the model from the task\\n\\nIf the user names a model, use it when compatible. If it cannot handle the request, explain why and recommend alternatives instead of silently switching. Otherwise inspect the inputs, duration, audio, quality, speed, and cost, check the live catalog, and pass an explicit model.\\n\\n### Seedance 2.x real-person rule\\n\\nSeedance 2.0 and 2.5 allow real people by default, the same as AI Studio:\\n\\n- The default is true. Generation tries Byteplus first and can fall back to Fal, which allows real-person likenesses, so a person in the prompt, a start frame, end frame, reference image, or reference video does not hard-fail. The request is charged at Fal\'s higher tier-specific rate even if Byteplus serves it.\\n- Turn it off for the cheapest run, only when the user wants that and nothing in the job is a real identifiable person (text-only requests, non-people, anime, clearly synthetic or stylized characters). For MCP use `allow_real_people: false`. For the CLI use `--no-allow-real-people`. That pins the job to the lower-priced Byteplus path, and a Byteplus likeness-policy refusal does not fall back to Fal. Pass the same value to `get_model_costs` or `videodraft costs` so the estimate matches the charge.\\n- If a request made with the option off fails with code `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, estimate the higher rate, follow the user\'s spend-confirmation preference, and retry exactly once with the option on. The structured recovery fields are `retryable: true`, `retry_with: { allow_real_people: true }`, `cli_flag: \\"--allow-real-people\\"`, and `retry_policy: \\"once\\"`. CLI `--json` submit errors expose them under `details`; `status` and `wait` include them on the failed job result. Do not treat an unrelated moderation or provider error as that signal.\\n- Do not loop when the option was on. Byteplus can accept a task and reject its generated output later. VideoDraft refunds that failed generation, but the late asynchronous failure cannot be rerouted to Fal. Rephrase the prompt or use different references before trying again.\\n- For hosted AI Production, the same default applies to every Seedance scene segment: `produce_project` with `mode: \\"full_video\\"` and `videodraft produce <project> --mode full_video` allow real people unless you pass `allow_real_people: false` or `--no-allow-real-people`. If an earlier run with the option off partially submitted and returns the opt-in code, rerun that same project once with an explicit `allow_real_people: true` / `--allow-real-people`. The server reconciles asynchronous results first, preserves running/completed jobs, and resubmits only failed scene-video placeholders carrying the exact opt-in signal. Keep the native-first VideoDraft ADE routing rule above: hosted full-video production is still explicit/fallback-only when the local editor is available.\\n\\n**How to pick a model. Follow this order every time:**\\n\\n1. The user named a model: use it, never substitute.\\n2. Otherwise pick a **Tier 1** model.\\n3. Use a **Tier 2** model only when no Tier 1 model supports the request, and say why. Never pick Tier 2 on your own for price, speed, novelty or variety.\\n\\nTier 1: images `nano-banana-2`, `nano-banana-pro`, `nano-banana-2-lite`, `gpt-image-2.5-flare`, `gpt-image-2.5-sunburst`, and `recraft-v4` for vector/SVG only; videos `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0`, `kling-o3`, `minimax-h3-max`; video edits `gemini-omni-1.1-flash`; talking-head portraits `veed-fabric`, `minimax-h3-max-lipsync`; motion transfer `kling-v3-motion-control`; audio ElevenLabs for speech tools, Lyria 3.5 for music at any length.\\n\\nTier 2 is every other catalog model (`wan-3.0`, `flux-3`, `minimax-h3`, `kling-v3-turbo`, `kling-2.6-pro`, `grok-imagine-video-1.5`, `happy-horse`, Veo 3.1, Sora, Seedream, Grok Imagine images, FLUX images, `sync-lipsync-2` and the rest). All of them stay available by name. Real Tier 2-only jobs: document or web-page references (`wan-3.0`), keyframes pinned to timestamps (`flux-3`), re-syncing existing footage (`sync-lipsync-2`).\\n\\nEvery `videodraft models image|video|audio --json` entry carries `tier` (1 or 2), and the top-level `recommended` array is Tier 1, best first. That catalog beats this page when they disagree.\\n\\n**Speech in video:** write the dialogue in the video prompt. `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` and `kling-o3` speak it natively. For a specific voice add a Seedance reference audio clip (`--ref-audio`) or a Kling bound voice. Do not plan silent video + TTS + lip-sync. Lip-sync tools are only for a portrait presenter talking to camera, or for re-syncing footage that already exists. Off-screen narration and voiceover are unaffected: keep using `generate voiceover` for those.\\n\\n**Images:**\\n\\n- `nano-banana-2`: general default, editing, consistency, and references.\\n- `nano-banana-pro`: maximum quality. `nano-banana-2-lite`: fast, inexpensive drafts.\\n- `gpt-image-2.5-flare`: replaces the GPT Image 2 recommendation for posters, logos, signs, title cards, readable text and composition. Use `gpt-image-2.5-sunburst` for extra precision/detail. Both use the same basic options as GPT Image 2: references (up to 16), aspect ratio, 1K/2K/4K resolution, image count (1-4), and quality. Quality adds xhigh/max; auto is charged at the max tier. Other model recommendations are unchanged; explicit `gpt-image-2` requests still use GPT Image 2.\\n\\n**Videos:**\\n\\n- `gemini-omni-1.1-flash`: general default for 3-10s generation, image-to-video, uploaded source edits up to 10s, and silent or voiceover-backed shots (audio is always on, so describe quiet ambience with no speech and mute or mix it on the timeline; only a file with no audio track at all needs `kling-3.0`, `kling-o3` or Seedance with `--no-audio`). It supports first and last frames, up to 10 total image inputs, up to 3 creative reference videos of at most 3 seconds each on `--video-task generate` ONLY, uploaded-video extension, and continuation of an earlier generation through `--previous-interaction-id`. Output is 360p/720p/1080p/4K at 3/10/15/30 cr/s with audio always on. Use `--source-video` for the uploaded edit/extension source; extension sources must be 1-30s. Edit and extend accept EXACTLY ONE input video, so `--ref-video` cannot accompany `--source-video` or `--previous-interaction-id`; only `--ref` images can. One `--ref-video` with no separate source remains a legacy source edit. A previous interaction defaults to conversational edit; add `--video-task extend` or `--extend` with an explicit 3-10s duration to append at the end. A continuation resolves to the prior turn\'s output and is submitted as an ordinary source, so the same limits apply to it: at most 30s to extend, at most 10s to edit. 40 seconds total is reachable, but only by extending from a source of 30s or less, so a ladder dead-ends once it passes 30s. New dialogue can only be added when the SOURCE video is silent; adding speech on top of a source that already contains speech is refused with \\"the model is currently unable to process speech edits\\". The server safely measures creative-reference durations, or you can repeat `--ref-video-duration` when a host blocks metadata probing. Fal BYOK supports its currently callable v1.1 generation and basic-edit endpoints at zero VideoDraft credits, but Fal does not expose continuation/extension or mixed source-edit references as callable endpoints and VideoDraft must never fall back to paid Google.\\n- `grok-imagine-video-1.5` (Tier 2): 1-15s text, first-frame, or 1-7 reference-image generation with native audio. Text/first-frame modes support 480p, 720p, and 1080p; reference mode supports 480p/720p. Cite references as `<IMAGE_0>` through `<IMAGE_6>`. It has no last frame, seed, negative prompt, quality tier, reference video, or reference audio.\\n- `minimax-h3` (Tier 2): 480p/768p/2K/4K (5/6/13/16 cr/s, 768p default), native stereo audio, and 5-15s text, first/last-frame, or mixed-reference generation. Reference mode accepts up to 9 images, 3 videos, and 3 audio clips, with at most 12 files total. Cite them as `Image 1`, `Video 1`, and `Audio 1` in array order.\\n- `minimax-h3-max`: 480p/768p pricing (5/8 cr/s, 768p default), native audio, and 5-15s text, first/last-frame, or mixed-reference generation. Reference mode accepts up to 9 images, 3 videos, and 3 audio clips, with at most 12 files total, cited as `Image 1`, `Video 1`, and `Audio 1` in array order. It supports a reproducibility seed and `disabled` / `balanced` / `quality` prompt expansion.\\n- `wan-3.0` (Tier 2): unified 2-30s text, first/last-frame, or ordered mixed-reference generation at 480p/720p/1080p (7/14/28 cr/s), with optional native audio. Reference mode accepts up to 10 images, 5 videos, and 5 audio clips, at most 20 media files total; video and audio each total at most 15 seconds. `--auto-duration` reserves 30 seconds and reconciles to the provider-reported output length. Document/web references use `--file-url` or `--web-url` and require `--thinking`.\\n- `flux-3` (Tier 2): Black Forest Labs FLUX 3. 5-20s at 720p/1080p with 24fps native audio, from a prompt, a first frame, first + last frames, or up to 10 keyframes pinned to specific moments (`--keyframe shot.png@2.5`, repeatable). `--quality draft` renders the same shot at 720p for roughly a third of the cost. Auto duration is text/first-frame only.\\n- `seedance-2`: 11-15s, video/audio/mixed references, wider ratios, selectable audio, or first/last frames. `mini` is the economical default; move to `standard` when the user asks for high quality, when a Mini result is not good enough, or for 1080p/4K; `fast` for speed.\\n- `seedance-2.5`: 4-30s single takes and up to 50 references (30 image, 10 video, 10 audio). Same modes as 2.0, one quality tier, 480p/720p/1080p. Reach for it when a shot must run past 15s or carry more references than 2.0 allows.\\n- `kling-o3`: reference images plus structured image/video elements, first/last frames, multi-prompt, audio control, or 4K. O3 allows 7 combined image references and image-backed elements, reduced to 4 combined items when a video-backed element is present. `kling-3.0`: image-to-video can use structured image/video elements and bind a custom Kling voice ID to either element form. Tier 2, by name: `kling-v3-turbo` (fast polished 3-15s with first frame, multi-prompt, and audio, but no elements) and Kling 2.6 Pro (top-level voice IDs cited as `<<<voice_1>>>` and `<<<voice_2>>>`).\\n- Existing-video edits use `videodraft edit video`, not generic generation. `gemini-omni-1.1-flash` is the preferred edit model and is chosen automatically when you omit `--model`: source up to 10s, up to 10 reference images, 360p/720p/1080p/4K with audio. Creative `--ref-video` inputs are NOT accepted on an edit, because an edit takes exactly one input video; use `generate video --video-task generate` to guide a new clip with video references instead. Omitting `--model` on a longer or unmeasurable source spends nothing and prints a priced menu (what each model edits, what it drops, what it costs) so you can put the choice to the user. **Truncation:** only Gemini refuses an over-length source. Happy Horse silently edits just the first 15s, Kling O3 the first 10s, and Grok the first 8s. The command warns you when that will happen; always relay it to the user. Gemini regenerates the audio track, so use Happy Horse or Kling O3 with `--preserve-audio` when the source audio must survive. Choose Grok for the cheapest prompt-only edit, Happy Horse for up to 5 image references, or Kling O3 for controlled reference-image edits.\\n- To edit a source longer than 10s without losing its tail, cut it into <=10s pieces in the native editor, edit each with Gemini, and reassemble. VideoDraft has no server-side split/concat, so this path needs `videodraft_editor`.\\n- Kling O3 also has a reference-generation mode. Use `videodraft generate video --model kling-o3-video-ref-edit` with exactly one `--ref-video` to generate a new guided clip; use `videodraft edit video` when changing the source itself. For new mixed-reference generation use Seedance 2 / 2.5.\\n- Motion transfer uses `videodraft edit motion` with Kling V3 by default, or Kling 2.6 when explicitly requested or lower cost matters. It requires a subject image and a motion-reference video.\\n- Veo 3.1, Sora, Grok, Happy Horse and the other Tier 2 video models: only when explicitly requested or when no Tier 1 model fits.\\n\\n**Audio and utilities:**\\n\\n- Use Seed Audio 1.0 for open-ended text-to-audio, speech/music/sound synthesis, voice conditioning, or prompt-driven editing with up to three audio references or one image. Use `videodraft generate audio`. Reference clips are `@Audio1`, `@Audio2`, and `@Audio3` in array order. There is no duration input. Output is up to two minutes and settles at 19 credits per actual minute, with up to 38 credits reserved during generation. The CLI automatically retries transient responses with one operation key. To recover after the CLI process itself is interrupted, set `--idempotency-key <uuid>` on the original command and reuse it.\\n- Prefer ElevenLabs for voiceover, dialogue, voice changing, dubbing, and sound effects. Honor an explicitly selected supported TTS voice/provider. Use `lyria-3.5` for all music, short or long (up to ~3 minutes, 10 credits flat), with vocals/lyrics or instrumental arrangements. When `--model` is omitted the CLI sends no model and the server default applies, which is `lyria-3.5` wherever the backend supports it. If the server answers that `lyria-3.5` is an unknown model, that backend predates it: rerun without `--model` (the server default then applies), or use `lyria-3-pro-preview` for a track longer than 30 seconds. Ask for the length in the prompt. Keep `lyria-3-clip-preview` (fixed ~30s, 4 credits) and `lyria-3-pro-preview` (8 credits) for explicit requests. For Lyria, put desired length, lyrics and \\"instrumental only, no vocals\\" in the prompt; `--length` and `--instrumental` are ElevenLabs-only. Use ElevenLabs Music for exact timing, composition plans, or a style reference track. Lyria allows 10 reference images on Google, 1 on Fal BYOK; Fal 3.5 prompts are limited to 5000 characters. BYOK uses zero VideoDraft credits. ElevenLabs Music means `elevenlabs-music-v2.5` (`elevenlabs-music` is an alias for it); use `elevenlabs-music-v1` only when the user asks for v1 by name.\\n- For ElevenLabs voiceover, dialogue, and voice changing, accept a supplied raw voice ID (16-64 alphanumeric characters) or `elevenlabs-<id>`. The voice catalog is for discovery, not an allowlist. Do not reject or substitute a supplied ID because it is absent from `videodraft models voices` / `list_available_voices`. Use `--voice <id>` for voiceover and voice changing, or repeat `--line \\"<id>:Text\\"` for dialogue; for example, `--voice kPzsL2i3teMYv0FxEYQ6` and `--line \\"elevenlabs-kPzsL2i3teMYv0FxEYQ6:Hello.\\"` use the same voice. The voice must be accessible to the provider account used for generation. Private or cloned voices may require the user\'s connected ElevenLabs key. Kling video-control IDs and MiniMax `custom-*` IDs are separate voice systems.\\n- A character who needs to TALK:\\n - **Speaking inside a scene**, or any ordinary clip with dialogue: `generate video` with the line in the prompt, in quotes, with who says it and how. `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` and `kling-o3` all voice it natively. For a specific voice use a Seedance `--ref-audio` clip or a Kling voice bound per element. This is the default path; do not generate speech separately and lip-sync it on.\\n - **Talking to camera from a portrait** (presenter, spokesperson, explainer): the avatar lane, and VEED Fabric is preferred. Use managed `avatar create` then `avatar render` for a reusable avatar record with bundled speech, `avatar fabric` for a one-off portrait plus text or existing audio, and `avatar h3-lipsync` for a short (5-14.8s) portrait clip from existing audio at up to 2K. Not for in-scene shots: Fabric animates a portrait facing the lens, so a cinematic request comes back as a head-on talking headshot.\\n - **Footage that already exists plus replacement audio** (re-dub, translation): `avatar lipsync`. This is the only job for Sync Labs unless the user names it.\\n- Enhancement: use Topaz image/video upscaling only when the content is already correct. Use image 1x for cleanup, 2x by default, 4x when justified; use video 2x by default. Edit or regenerate creative errors.\\n\\nSee [references/models.md](references/models.md) for the detailed routing table and exact capability limits.\\n\\n## Prefer references when continuity matters\\n\\nPure text-to-image or text-to-video is fine for a generic one-off asset. When a specific character, product, location, style, composition, or brand identity must survive generation, use references instead of hoping the prompt recreates it.\\n\\n- If the user supplies reference media, preserve and pass it. Never reduce the request to text alone.\\n- When continuity matters, generate/select a strong still first with the selected image model (`nano-banana-2` by default), wait for its URL, then animate it as a start frame/reference. Confirm the combined image and video cost.\\n- When using a hosted storyboard stage for multiple shots, use `videodraft shots <project_id> --model <selected-image-model> --grid`, then animate the decoded shots. Preserve explicit models. In VideoDraft ADE, import the resulting assets into the native editor instead of continuing into hosted production. A requested non-Seedance video model must use manual per-shot generation instead of Seedance full-video mode.\\n\\n## Cost and credits\\n\\nDo not call `videodraft credits` before routine generations. Paid endpoints validate and deduct atomically; if the balance is insufficient, the request is rejected before the provider job starts (CLI exit code 4). Check the balance only when the user asks, gives a credit budget, or a large workflow needs budget planning.\\n\\nFor expensive work, estimate with `--estimate` or `videodraft costs`, state the selected model/settings/cost, and get a go-ahead. This matters most for shot-image batches, long or high-resolution video, AI Production, and paid audio batches. Honor the user\'s confirmation preference for the session.\\n\\nKling voice creation costs 1 VideoDraft credit on the platform Fal account and 0 credits with Fal BYOK. Preview it with `videodraft kling-voices create <sample> --name <name> --estimate`; the estimate does not create a voice or require consent confirmation. Actual creation requires `--confirm-consent`.\\n\\nMiniMax H3 costs 5 / 6 / 13 / 16 credits per output second at 480p / 768p / 2K / 4K (768p by default). In reference mode the first 5 images are included and each additional image costs 8 credits. Reference video and reference audio are NOT billed.\\n\\nMiniMax H3 Max costs 5 credits per output second at 480p and 8 credits per second at 768p (default), plus pooled reference tokens in reference mode: the first 4,096 are free and every 1,000 after that costs 2 credits, where an image is `(width x height) / 1024` tokens, a reference video is 2,886 tokens per second at 480p or 7,459 at 768p, and reference audio is about 2,121 per second. Use `--prompt-expansion-mode disabled|balanced|quality`; balanced is the default. Fal\'s provider safety checker is OFF by default and is not exposed in the app \u2014 only pass `--safety-checker true` if the user explicitly asks for it.\\n\\nWan 3.0 costs 7 / 14 / 28 credits per output second at 480p / 720p / 1080p. Auto duration reserves the 30-second maximum and refunds the unused reserve after the provider reports the actual whole-second output length. Fal BYOK runs on the connected user key and charges zero VideoDraft credits.\\n\\nGrok Imagine Video 1.5 costs 8 credits per output second at 480p, 14 at 720p, or 25 at 1080p, plus 1 credit for each first-frame or reference image. Native audio is always generated.\\n\\n`videodraft models image|video` lists the live image and video catalogs with supported inputs. Video entries are grouped as `generation`, `video_edit`, `motion_control`, `avatar_lipsync`, and `upscale`, and each reports the exact tool. Use `videodraft models video --category video_edit` to narrow the list. `videodraft models audio` lists Seed Audio, Google Lyria, and ElevenLabs audio/media tools, while `videodraft models voices` helps discover TTS voices. Consult them for capabilities; a supplied ElevenLabs voice ID does not need to appear in the voice catalog.\\n\\n## Async jobs\\n\\nImage/video generation is asynchronous: commands submit a job and **wait by default**, printing output URLs (and saving files with `--download`). Large downloaded images also get a downscaled copy in `previews/` next to them (the `preview` field / \\"inspect via preview\\" line in the output) \u2014 **look at the preview, deliver the original**; viewing full-resolution images bloats the chat permanently. In scripts/CI prefer explicit control:\\n\\n```bash\\nJOB=$(videodraft generate image \\"...\\" --no-wait --json | jq -r .job_id)\\nvideodraft wait \\"$JOB\\" --download \\"./outputs/{job_id}_{index}.{ext}\\" --json\\n```\\n\\nFor MANY jobs: submit each with `--no-wait`, collect ALL with one command \u2014 `videodraft wait <id1> <id2> ...` polls every job from one process with one batched request per tick. Do NOT spawn parallel `wait`/`generate --wait` processes for a batch.\\n\\nIf a wait times out, the job is still running server-side \u2014 `videodraft status <job_id>` later. Never re-submit just because a wait timed out (that double-spends credits).\\n\\nFor completed Wan 3.0 jobs, MCP `check_generation_status` and CLI `status`/`wait --json` include `outputMetadata` with Fal\'s returned `seed`, `duration`, and `actual_prompt` when present.\\n\\n## AI Studio sessions\\n\\nStandalone image/video/audio generations are filed into an AI Studio session in the web app. 3D assets use `assets 3d list/get` separately. You do not have to create an AI Studio session:\\n\\n- **MCP hosts** (Claude Code, claude.ai, Codex, VideoDraft ADE): on 2025-era MCP the server mints an `Mcp-Session-Id` on `initialize`; your host echoes it, and this conversation\'s generations land in their own session. On stateless MCP (2026-07-28) hosts that send a conversation id (ChatGPT) get the same. Otherwise `name_current_ai_studio_session` returns `pass_session_id: true` with a `session_id`: pass that `session_id` to every later standalone generation in the conversation. Tool results echo the session as `ai_studio_session_id`.\\n- **CLI**: the same handshake runs once per (profile, server, working directory) and is cached for 12 idle hours, so everything generated from one directory shares one session. `videodraft sessions current` shows it; `videodraft sessions reset` starts a new one.\\n- Project generations (`--project <id>` / `project_id`) always go to that project\'s session.\\n\\nOnce you understand the creative task, name the current automatic session before the first standalone generation:\\n\\n```bash\\nvideodraft sessions name \\"Purple Seal Rescue Short\\"\\n```\\n\\nChoose a concise, specific 3-6 word title for the intended work. Do not copy the client name, date, or exact chat title. Name it once: the operation creates the session with that title. If generation, a user, or an earlier agent created the session first, its existing name is preserved.\\n\\nApart from the `pass_session_id: true` case above, pass `--session <id>` / `session_id` only to **continue earlier work** or create an explicit separate group:\\n\\n```bash\\nSESSION=$(videodraft sessions create \\"Fox brand explorations\\" --json | jq -r \'.session.id\')\\nvideodraft generate image \\"a red fox in snow, cinematic\\" --session \\"$SESSION\\"\\nvideodraft generations --session \\"$SESSION\\" # what is in it\\nvideodraft sessions list --name fox # find it again later\\n```\\n\\n`VIDEODRAFT_SESSION=<id>` sets the default for every command (ignored when `--project` is given). While it is set, `sessions name` refuses to run because that command names the current automatic connection session, not the pinned override; unset it first or rename the pinned session in AI Studio. `VIDEODRAFT_SESSION_SCOPE=<label>` groups several directories into one connection session; `VIDEODRAFT_CLIENT_NAME=<host>` labels the fallback placeholder used when generation creates the session before `sessions name`; `VIDEODRAFT_NO_SESSION=1` disables the handshake. Generations that reach the server with no session at all fall back to the account-wide \\"Agent (MCP)\\" session; if you see work landing there, pass `--session` explicitly.\\n\\n## Generation history\\n\\nPast work is queryable \u2014 reuse a previous setup instead of guessing. `videodraft generations` lists recent generations; scope with `--session <id>` or `--project <id>` (includes collaborators\' rows in shared scopes; pass one, project wins), and filter with `--type`, `--model`, `--favorites`. `--full --json` returns each row\'s exact parameters (aspect ratio, resolution, duration, references) \u2014 the human table stays compact, so pair `--full` with `--json`. `videodraft generation <id>` prints one generation\'s complete recipe (prompt, input image, parameters, outputs); `--favorite` / `--unfavorite` stars it. `videodraft sessions list` shows AI Studio sessions (owned + shared) with `--name` search \u2014 take a session id from there to read its history.\\n\\n## Free stock footage and photos\\n\\nPexels and Pixabay are wired in through `search_stock_media` / `import_stock_media` (`videodraft stock search \\"<query>\\"`, `videodraft stock import <ref>`). Both libraries are free, watermark-free and cost **zero credits**, so use stock for b-roll, establishing shots, backgrounds and ordinary real-world footage, and spend credits on the shots that must be specific: the product, the character, the scripted action.\\n\\n```bash\\nvideodraft stock search \\"city skyline night\\" --min-duration 5 --orientation landscape\\nURL=$(videodraft stock import pexels:video:35379336 --quality hd --json | jq -r .url)\\n```\\n\\n- Always two steps. A search result\'s `preview_url` is a thumbnail for judging the shot; never place it on a timeline, attach it to a shot or send it to a model. `import_stock_media` copies the file onto the VideoDraft CDN and returns the URL every other surface accepts, including the native editor\'s `library_manage` import.\\n- `--quality hd` (default) caps video at 1080p; `4k` caps at 2160p, `best` takes the largest the provider has, `sd` suits rough cuts. Stills ignore quality and import at full size. Use `--orientation portrait` for 9:16, and `--min-resolution 1920` to drop anything below HD on its long edge.\\n- Photos come from Pexels. Pixabay contributes video only, because its full-size image host refuses server-side downloads.\\n- Credit the creator and link the provider page when you show results or deliver the finished work; both come back on every result. Skip clips that imply a person or brand endorses the product, and avoid recognisable logos in ads.\\n- Search and import are rate limited per user (30 searches and 15 imports a minute) and searches are cached for 24h, so a burst of searches while planning a montage is fine. Bulk downloading a stock library is prohibited by both providers. A note saying Pixabay was skipped means its provider budget for that minute is spent; the Pexels results still stand.\\n\\n## Downloading from YouTube, Instagram, TikTok and other sites\\n\\nVideoDraft ships no downloader. When the user asks to pull a video from YouTube, Instagram, TikTok, X, LinkedIn or anywhere else, `yt-dlp` on the user\'s own machine does the work. Check for it with `command -v yt-dlp` before promising anything.\\n\\n**Present:** pin the format. On its defaults yt-dlp takes the best stream, usually VP9 or AV1 in WebM, which the native editor\'s import rejects (mp4, mov and m4v only). This returns one pre-muxed H.264 + AAC MP4 and needs no ffmpeg:\\n\\n```bash\\nyt-dlp -f \\"b[ext=mp4]\\" -o \\"media/%(title)s.%(ext)s\\" \\"<url>\\"\\n```\\n\\n**Missing:** say so in one line, then offer `brew install yt-dlp` only where `brew` exists, and only after asking. With no Homebrew, say the tool is unavailable and stop.\\n\\n- Never install Homebrew, never `sudo`, never write to `/usr/local/bin` or `/opt`, never pipe a script into a shell.\\n- `ffmpeg` is optional, matters only above 720p, and the selector above never needs it.\\n- Auth-gated content needs `--cookies-from-browser`; raise it only after a download fails on auth.\\n- Fetch only what was asked for, never a channel, playlist or back catalogue.\\n\\n## Local files and reference images\\n\\nReference inputs must be public URLs. The CLI uploads local files automatically wherever a URL is expected (`--ref photo.jpg`, `--start-image frame.png`), or explicitly:\\n\\n```bash\\nURL=$(videodraft upload ./product.png --json | jq -r .url)\\n```\\n\\nNever silently drop a reference you couldn\'t upload \u2014 stop and tell the user. Never upload a user\'s file to a third-party host.\\n\\nWhen the user attaches media for a native production, import actual footage into the editor by default. For hosted generation/storyboarding, classify each item before acting: a recurring **visual asset** (character/product/location/style), actual **footage to place as shots**, or **inspiration only**. See [references/pipeline.md](references/pipeline.md) for the hosted role mapping.\\n\\n## Showing media to the user\\n\\nGenerated media is **not** displayed in the chat automatically \u2014 you decide what to show. To preview an asset inline, save it locally (use `--download` so it lands under `media/`) and reference its **local path** as a Markdown link with a **leading `./`**:\\n\\n```\\n[ferrari shot](./media/ferrari_01.png) \u2190 image card\\n[the clip](./media/clip.mp4) \u2190 video player\\n[voiceover](./media/vo.mp3) \u2190 audio player\\n```\\n\\nPut the Markdown link **in your message text** \u2014 video and audio embed exactly like images. Do **not** use `SendUserFile` (or other file-send tools) to display media: that renders inside a collapsible tool card and gets buried in the tool list. The Markdown link in your prose is what produces the inline card.\\n\\nUse the path you saved to: a **workspace-relative** path (`./media/clip.mp4`, or `./<any-folder>/clip.mp4` \u2014 any folder in the workspace works), or the **absolute** path for a file outside the workspace (e.g. `/Users/you/Desktop/clip.mp4` or another workspace\'s path). Both render. Show the finished results worth showing (and only those \u2014 not every intermediate job). A bare CDN URL or a JSON dump of output URLs does **not** render; the local-path Markdown link is what produces an inline card.\\n\\n## Native-first VideoDraft ADE pipeline (idea \u2192 MP4)\\n\\nWhen `videodraft_editor` is present:\\n\\n1. Generate or source the script, storyboard, shot images, clips, voiceovers, music, and other assets through the cloud CLI/MCP as needed.\\n2. Call native `project_manage` (`project`, action `open` or `create`) to open or create the `.vdproject`.\\n3. Call native `library_manage` `import`, wait for imports to become ready, then assemble and refine the timeline with `edit_apply` recipes.\\n4. Call native `delivery_manage` `submit` and use its `jobs` operation for progress and results.\\n\\nDo not run the hosted production or export steps in this path unless the user explicitly asks for a web production.\\n\\n## Hosted fallback pipeline (idea \u2192 MP4)\\n\\nUse this only when there is NO native editor at all, or the user explicitly requests the hosted web\\nworkflow. The native surface is not only the injected `videodraft_editor` MCP: a `videodraft-editor`\\nexecutable on PATH is the same editor reached through its terminal bridge, and\\n[references/editor.md](references/editor.md) covers driving it that way. Treating a missing MCP as\\n\\"no editor\\" sends sessions that have the binary into hosted production for no reason.\\n\\n```bash\\nvideodraft create \\"<idea>\\" --ar 9:16 # project: script \u2192 visual assets \u2192 storyboard\\nvideodraft shots <project_id> --grid --estimate # cost preview, confirm with user\\nvideodraft shots <project_id> --grid # batch shot images (waits, writes onto shot cards)\\nvideodraft produce <project_id> # voiceovers + captions + production timeline\\nvideodraft export <project_id> --download final.mp4\\n```\\n\\nOptional between produce and export: per-shot motion clips (`videodraft generate video ... --project <id>` then place it with `videodraft attach <project> --scene N --shot M --media <url|file> --type video --duration <s>`), music (`videodraft generate music \\"...\\" --attach <project_id>`), and standalone audio assets (`generate audio`, `generate sound-effect`, `generate dialogue`, `generate voice-changer`, `generate dub`). Details, per-step tools and editing rules: [references/pipeline.md](references/pipeline.md).\\n\\n## Avatar and talking-head videos (both surfaces)\\n\\nAvatar generation is cloud-only \u2014 the native editor has no avatar or lipsync tools \u2014 so this applies whether or not `videodraft_editor` is present. Generate the avatar in the cloud; in VideoDraft ADE, import the rendered clip and cut it on the native timeline like any other footage.\\n\\nAvatar/talking-head videos use dedicated commands. For a reusable managed avatar, obtain or generate a clear portrait \u2192 `videodraft avatar script` when needed \u2192 `videodraft avatar create` \u2192 `videodraft avatar render --resolution 720p`. For a one-off portrait, use `videodraft avatar fabric <portrait> --text \\"...\\"` or `--audio <file>`, or `videodraft avatar h3-lipsync <portrait> --audio <file>` for a short high-resolution clip from existing audio. For an existing video plus replacement audio, use `videodraft avatar lipsync <video> --audio <file>`. Managed script/creation is bundled/free; direct Fabric, H3 Max Lip Sync, Sync, the managed Fabric render, and optional portrait generation/upscaling are paid. Confirm expensive steps first.\\n\\n## Working with hosted project data\\n\\nA hosted project is one JSON blob (script, storyboard scenes, shot cards, visual assets, production timeline). To inspect: `videodraft projects get <id>`. To edit: fetch `--raw`, modify, then `videodraft call update_project` \u2014 objects deep-merge, **arrays replace wholesale** (send the complete `storyboard.scenes` array to change one scene). Snapshot first with `videodraft checkpoint create <id>` before risky edits. Schema reference: `videodraft call get_project_schema`. This does not replace native editor tools when `videodraft_editor` is available for the production itself.\\n\\n## More\\n\\n- [references/pipeline.md](references/pipeline.md) \u2014 hosted fallback data model and production workflow\\n- [references/editor.md](references/editor.md) \u2014 native headless editor routing, project selection, import, timeline edits, verification, and export\\n- [references/models.md](references/models.md) \u2014 choosing image/video models, pricing patterns, voices and styles\\n- [references/examples.md](references/examples.md) \u2014 recipes: batch product videos from a CSV, talking-head from a script, changelog video in CI\\n","references/3d.md":"# Standalone 3D assets through MCP and CLI\\n\\nUse this lane for downloadable meshes, textured 3D models, and humanoid rigging. These are independent of AI Studio sessions and do not require a storyboard or web viewer. Never create a project just to generate a mesh. Native editor media import does not establish support for GLB/FBX assets; use external 3D software unless the live editor schema explicitly supports them.\\n\\n## Discover and quote\\n\\n```bash\\nvideodraft models 3d --json\\nvideodraft generate 3d \\"a worn brass telescope\\" --model meshy-7 --estimate\\nvideodraft generate 3d --model tripo-h3.1 --ref ./reference.png --options @options.json --estimate\\nvideodraft rig 3d --estimate\\n```\\n\\nMCP tools: `get_3d_models`, `estimate_3d`, `generate_3d`, `rig_3d`, `check_generation_status`, `list_3d_assets`, `get_3d_asset`, `create_3d_upload`, and `finalize_3d_upload`.\\n\\nThe generator catalog is deliberately limited to `meshy-7` and `tripo-h3.1`. Both expose text/image/multi-image modes; use the live schemas for option availability, view order, limits, topology, texture/PBR, and pose controls. Do not invent common provider option names. Server defaults prioritize quality. Rigging is a separate Meshy operation for compatible textured humanoid GLBs; it is not an arbitrary creature or facial rig service.\\n\\nEvery option is accessible through `--options` JSON (`@path.json` supported), plus repeatable `--option key=value` overrides. Values parse as JSON when valid. Options and explicit input mode affect pricing, so obtain a fresh quote for the exact recipe. Estimates perform no file upload or provider generation. VideoDraft bills whole credits at 100 credits per dollar using the server\'s rounding rules. BYOK costs zero VideoDraft credits and runs on the connected user Fal key. Do not infer the user\'s Fal-account charges from a zero VideoDraft-credit quote.\\n\\n## Generate and retrieve\\n\\n```bash\\nvideodraft generate 3d \\"a worn brass telescope\\" --model meshy-7 --download --json\\nvideodraft generate 3d --model meshy-7 --ref ./hero.png --no-wait --json\\nvideodraft generate 3d --model tripo-h3.1 --input-mode multi_image --ref ./front.png --ref ./left.png --ref ./back.png --ref ./right.png --no-wait --json\\nvideodraft status JOB_ID --json\\nvideodraft wait JOB_ID --download --json\\nvideodraft assets 3d list --status completed --json\\nvideodraft assets 3d get ASSET_ID --download ./media/3d --json\\nvideodraft rig 3d --asset ASSET_ID --download --json\\nvideodraft rig 3d ./textured-humanoid.glb --download --json\\n```\\n\\nLocal image references upload automatically. Local rigging inputs must be self-contained `.glb` files and upload via the dedicated 3D route. Input mode is inferred from reference count unless `--input-mode text|image|multi_image` is supplied. A text prompt requires no images; image mode requires exactly one and no geometry prompt. Meshy multi-image accepts 1-4 views, and Tripo accepts 2-4 in front, left, back, right order. Use Meshy\'s `texture_prompt` option when texturing guidance is needed with image inputs. Do not mix unrelated images as if they were views of the same object.\\n\\nGeneration and rigging wait by default. `--no-wait` returns a persisted job that continues server-side. For many jobs use one `wait ID1 ID2 ... --download`; do not run parallel polling processes. Recover timeout with the same job ID, not another paid generation.\\n\\nSubmission returns a `request_id` UUID. A same-request retry must use `--request-id UUID` and the original arguments. The CLI journals resolved uploads in its private config directory so retries reuse the same remote files. Changed inputs under that UUID are rejected. Submission errors include the UUID in JSON `details.request_id` where available. Never retry a failed request with a new UUID until the prior job status is known.\\n\\n## Complete artifact packages\\n\\n`--download` defaults to `media/3d/<asset_id>/`. A user directory is respected and receives a per-asset folder. `{job_id}` and `{asset_id}` directory templates are accepted. Do not pass a `.glb` filename or a media `{index}.{ext}` template: a 3D result can contain multiple formats and texture/material dependencies.\\n\\nDownloads preserve every returned artifact. The manifest records provider URLs, source filenames, local paths, role, content type, and actual downloaded sizes. glTF/OBJ/MTL references are rewritten to saved dependency paths; binary GLB/FBX and provider ZIPs are retained as returned. Supplied dependency aliases become local copies at the paths needed by original FBX files. Unsafe or colliding aliases produce warnings. Existing packages are never overwritten. A successful package has `manifest.json`; read its `complete` field and `warnings`, plus CLI `package_warnings`. Do not claim a package with unresolved dependencies is ready for offline use.\\n\\nMachine results use `type: \\"model3d\\"` with `output_files` for files and `output_media` only for ordinary preview media. Texture maps and models are not image/video gallery entries. Show a supplied preview via a local image Markdown link and deliver the original mesh/package files. Do not imply an interactive 3D viewer exists in AI Studio or VideoDraft ADE.\\n","references/editor.md":"# Native VideoDraft Editor reference\\n\\nUse this reference when the user wants to assemble, cut, caption, mix, lay out, inspect, or export a local VideoDraft Editor project. The native editor is deterministic and local. Cloud generation remains in the `videodraft` CLI or hosted MCP.\\n\\n## VideoDraft ADE preference rule\\n\\nWhen `videodraft_editor` tools are exposed, treat the native editor as available and make it the default surface for production, timeline assembly, and final export. It is headless by design, so a hidden window or an untouched Open Editor button does not justify using hosted production instead.\\n\\nUse cloud tools for asset generation and optional script/storyboard work, then import the results. Do not call hosted `produce_project` / `videodraft produce` or `export_video` / `videodraft export` unless the user explicitly requests an editable web production or the native editor tools are unavailable. If a native tool call fails after the editor was available, report or recover that native failure rather than silently switching surfaces.\\n\\n## Choose the correct surface\\n\\n- `videodraft` and the hosted VideoDraft MCP generate assets and can manage hosted web projects. They use the user\'s VideoDraft account and credits. In VideoDraft ADE, use them mainly as the source of generated media and optional storyboards for the native production.\\n- `videodraft_editor` edits local `.vdproject` packages. It has no generation, account, model, or credit tools.\\n- Inside VideoDraft ADE on a supported Mac, the editor MCP is injected automatically for Claude, Codex, OpenCode and Grok in both Code and VideoDraft modes. It starts headlessly before the chat opens. The user does not need to click Open Editor, and closing or hiding the editor window does not stop headless editing.\\n- Outside that environment, use the editor only if `videodraft_editor` MCP tools are already exposed or the `videodraft-editor` executable is on PATH. Do not confuse the public `videodraft` cloud CLI with the separate native editor executable.\\n\\nPrefer the direct MCP tools when they are available. The terminal bridge is useful for scripts, diagnostics, or an agent session where the MCP was not injected.\\n\\n## Which editor tools you have\\n\\nCurrent editors expose eight workflow tools: `project_manage`, `edit_snapshot`, `edit_apply`, `edit_undo`, `library_manage`, `inspect`, `speech_apply`, and `delivery_manage`. Use them whenever `edit_apply` is available. Everything below describes them.\\n\\nEarlier editors expose a different set of tools. If `edit_apply` is not in your catalog, follow the live tool descriptions; the working rules below still apply.\\n\\nThe live tool descriptions and schemas are authoritative. When they differ from this reference, follow them.\\n\\nService tools (`project_manage`, `library_manage`, `inspect`, `speech_apply`, `delivery_manage`) take `operation` and `parameters`, plus an optional `context` (see [Keep a reliable editing model](#keep-a-reliable-editing-model)):\\n\\n```json\\n{\\"operation\\": \\"import\\", \\"parameters\\": {\\"source\\": {\\"path\\": \\"/Users/me/clips\\"}}}\\n```\\n\\n### Parameters at a glance\\n\\nEach operation\'s schema has the full list. These are the fields that matter most, and the ones most often confused:\\n\\n| Tool and operation | Key `parameters` |\\n| --- | --- |\\n| `project_manage` `project` | `action` (`list`, `open`, `create`, `close`); `id`, `name` or `path` to open; `name`, `fps`, `aspectRatio`, `quality` to create |\\n| `edit_snapshot` (no `parameters`) | `includeLibrary`, a `startFrame`/`endFrame` window, `offset`/`limit` paging, `context` for later pages |\\n| `edit_apply` (no `parameters`) | `actions`; optional `context`, `requestId`, `previewOnly` |\\n| `library_manage` `import` | `source` with one of `path`, `url`, `bytes` + `mimeType`, or `matte`; optional `name`, `folder`. A `.srt` or `.vtt` file becomes captions |\\n| `library_manage` `list` | `ids` to poll imports, `pending`, `folder` |\\n| `inspect` `timeline` | a `startFrame`/`endFrame` window; there is no clip filter |\\n| `inspect` `frame` | `startFrame` for one frame; add `endFrame` and `maxFrames` to sample a range |\\n| `inspect` `media` | `mediaRef` (a library asset, not a file path), `startSeconds`/`endSeconds` in source seconds, `wordTimestamps`, `overview` |\\n| `inspect` `transcript` | a `startFrame`/`endFrame` window, `granularity` (`words` or `segments`), `clipId` |\\n| `inspect` `color` | `clipId` with `atFrame`, or `mediaRef`; optional `reference` |\\n| `speech_apply` `words` | `words`: transcript word indices, each one index or a `[first, last]` pair; or `matches`: exact words to remove everywhere; `pacing` |\\n| `speech_apply` `silence` | none |\\n| `speech_apply` `captions` | caption `style`, `position` and look fields; it finds the speech itself. `style` `plain` (sentences without a preset) takes `positionY` or `transform.centerY`, not `position` or `punctuation` |\\n| `delivery_manage` `submit` | `mode`, `codec`, `resolution`, `outputPath` (its folder must already exist); `captionGroupId` and `wordTiming` for `srt` and `vtt`; `check: false` skips a video\'s export check |\\n| `delivery_manage` `jobs` | `action` `list` (every job with its progress, warnings, output path and check findings; with a `jobId`, that job with all its findings), `cancel` with a `jobId`, or `dismiss` with a `jobId` and optional `findingIds` |\\n\\n`previewOnly` exists only on `edit_apply`. Service operations apply immediately.\\n\\n### `edit_apply` actions at a glance\\n\\nEvery action puts its fields beside `kind`, for example `{\\"kind\\": \\"adjust\\", \\"clipId\\": \\"c1\\", \\"speed\\": 2}`. `remove`, `adjust`, `replace`, `transition`, `mask`, `text`, `grade` and `effects` take `clipId` for one clip or `clipIds` for several; `animate`, `extract`, and the rows of `move` and `split` take a single `clipId`. Advanced actions cannot take `as`, but their clip ids accept `@name`. The one exception is `transition`, whose own `kind` field names the style, so its fields go inside `parameters`: `{\\"kind\\": \\"transition\\", \\"parameters\\": {\\"clipId\\": \\"c2\\", \\"kind\\": \\"dipToBlack\\"}}`. These `actions` place a clip, warm it and add a vignette:\\n\\n```json\\n[{\\"kind\\": \\"place\\", \\"assetId\\": \\"\u2026\\", \\"atFrame\\": 0, \\"as\\": \\"shot\\"},\\n {\\"kind\\": \\"grade\\", \\"clipId\\": \\"@shot\\", \\"adjustments\\": {\\"temperature\\": 7500}},\\n {\\"kind\\": \\"effects\\", \\"clipId\\": \\"@shot\\", \\"effects\\": [{\\"type\\": \\"finish.vignette\\", \\"params\\": {\\"strength\\": 35}}]}]\\n```\\n\\n| Advanced `kind` | Key fields |\\n| --- | --- |\\n| `place_batch` | `entries: [{mediaRef, startFrame, endFrame or source, trackIndex}]`, not the concise `assetId`, `atFrame`, `durationFrames`; leave `trackIndex` off every entry for a new track |\\n| `insert_batch` | `trackIndex`, `atFrame`, `entries: [{mediaRef, durationFrames or source}]`; later clips move right |\\n| `move` | `moves: [{clipId, toFrame, toTrack}]` |\\n| `remove` | `clipId` or `clipIds`; leaves a gap |\\n| `split` | `splits: [{clipId, atFrame}]`, or `trackIndex` with `frames` |\\n| `extract` | `trackIndex` with `ranges: [[start, end]]` in frames, or `clipId` with `ranges` and `units` (`frames`, or source `seconds`); closes the gap |\\n| `adjust` | `clipId` or `clipIds` plus any of `durationFrames`, `trimStartFrame`, `trimEndFrame`, `speed`, `speedCurve` (`{preset}`, or `{points: [{t, rate}]}` with `t` from 0 to 1 along the source), `preservesPitch`, `volume`, `opacity`, `transform` (`centerX`, `centerY`, `width`, `height`, `flipHorizontal`, `flipVertical`), `blendMode` |\\n| `animate` | `clipId`, `property` (`volume`, `opacity`, `rotation`, `position`, `scale`, `crop`), `keyframes: [[frame, ...values]]` with frames counted from the clip\'s start; `position` is the top-left corner |\\n| `arrange` | `layout`, `slots: [{slot, clipIds or mediaRef, anchor}]`, `fit` (`fill` or `fit`); `mediaRef` slots also need `endFrame` |\\n| `transition` (fields inside `parameters`) | `clipId` or `clipIds` (the clip after each cut), `kind` (`crossDissolve`, `dipToBlack`, `dipToWhite`, `blurDissolve`, `push`, `linearWipe`, `whipPan`, `crossZoom`, or `none` to remove), `durationFrames`, `params` (`direction` 0 to 3 for left, right, up, down; `feather`; `blur`; `intensity`) |\\n| `mask` | `clipId` or `clipIds`, `shape` (`rectangle`, `ellipse`, `none`), `centerX`, `centerY`, `width`, `height` (0 to 1 of the clip\'s own frame), `feather`, `strength`, `inverted` |\\n| `replace` | `clipId` or `clipIds`, `mediaRef` (a library asset of the same kind), `trim` (`keep`, the default, or `reset`), `linkedAudio` (`follow`, the default, or `keep`) |\\n| `tracks` | `reorder: [{trackId, to}]`, `set: [{trackId, muted, hidden, syncLocked}]`, `remove: [{trackId}]`; there is no add, since placing without a track makes one |\\n| `titles` | `entries: [{startFrame, endFrame, content}]`, optionally with `trackIndex` (on every entry or none), `style`, `animation`, `transform` and typography (`fontName`, `fontSize`, `color`) |\\n| `text` | `clipId`, `clipIds` or `captionGroupId`, with `content`, `style`, `position`, `animation` or typography |\\n| `grade` | `clipId` or `clipIds`, `adjustments`, `wheels`, `curves`, `hueCurves`, `lut`, `reset` |\\n| `effects` | `clipId` or `clipIds`, `effects: [{type, params, enabled}]`, `remove: [type]` |\\n\\n- `grade`: `adjustments` holds `exposure` (EV, -4 to 4), `temperature` (kelvin, 1800 to 15000, 6500 neutral, higher is warmer), `tint` (-150 magenta to 150 green); `contrast`, `highlights`, `shadows`, `whites`, `blacks`, `vibrance` and `saturation` are signed percents (-100 to 100, 0 neutral), not factors like 1.2. `wheels`: `shadows`, `midtones`, `highlights`, each `{hue, strength, brightness}`. `curves`: `luma`, `red`, `green`, `blue` as `[x, y]` points from 0 to 1. `hueCurves`: `targets: [{hue, rotate, saturation, lightness}]`. `lut`: `{path, mix}`, with the `storedPath` from `prepare_look`. Out-of-range values are refused.\\n- Effect `type` ids and their knobs, which go inside `params` and run 0 to 100 unless noted. Defaults leave the picture unchanged, so send the knob you want to see:\\n - `finish.vignette`: `strength` and `curvature` (-100 to 100; positive `strength` darkens the edges), `size`, `falloff`\\n - `finish.grain`: `strength`, `grainSize` (0.5 to 6 px); `finish.glow`: `strength`, `haloRadius` (px), `cutoff`, `halation`\\n - `defocus.gaussian`: `blurRadius` (px); `defocus.motion`: `streakLength` (px), `streakAngle` (-180 to 180 degrees)\\n - `texture.clarity`: `localContrast`, `hazeRemoval` (-100 to 100); `texture.sharpen`: `strength` (0 to 200); `texture.denoise`: `strength`\\n - `matte.chroma`: `screenHue` (0 to 360 degrees, 120 is green), `range`, `edgeSoftness`, `spillSuppression`\\n- `arrange` layouts and their slots (fill every slot): `fullscreen` (`stage`); `split` (`left`, `right`); `stack` (`top`, `bottom`); `corner_top_left`, `corner_top_right`, `corner_bottom_left`, `corner_bottom_right` (`stage`, `corner`); `quad` (`top_left`, `top_right`, `bottom_left`, `bottom_right`); `side_panel` (`stage`, `panel`); `thirds` (`left`, `middle`, `right`). The layout picks the corner; `anchor` only biases the crop.\\n- `trackIndex` and `toTrack` are an existing track\'s `order` from `edit_snapshot` (0 draws on top), counted at that point in the recipe: a track an earlier action adds shifts them. `tracks` takes `trackId` handles instead.\\n\\n## Start with the intended project\\n\\nAn MCP session can begin without a project selected. Project selection belongs to the session, not to whichever editor window happens to be frontmost.\\n\\n1. If the user named an existing project but its identity is unclear, call `project_manage` with `operation: \\"project\\"` and `parameters.action: \\"list\\"`.\\n2. Open the exact project with `action: \\"open\\"` and the returned `id`, an unambiguous `name`, or the `.vdproject` `path`.\\n3. Create only when the user wants a new local edit. `action: \\"create\\"` accepts optional `name`, `fps`, `aspectRatio`, and `quality`.\\n4. Treat `isActive` as this session\'s target and `isVisible` as the project shown in the UI. Headless editing only needs the session target.\\n5. Use `action: \\"close\\"` only when closing is part of the task. It saves first and never deletes the project. Afterwards other calls answer `no_project` until you open or create a project again.\\n\\nOther `project_manage` operations: `create_timeline` and `select_timeline` for additional timelines, and `configure` for project settings (a frame-rate change applies to every timeline).\\n\\nDo not substitute a hosted project ID for a native project. A hosted project can supply scripts, storyboards, and generated media, but the native edit is a separate `.vdproject` package.\\n\\n## Keep a reliable editing model\\n\\n- Opening or creating a project, and creating or selecting a timeline, returns `data.snapshot` (the `edit_snapshot` view with the library: format, tracks, clips and assets) and a `context`, so you can edit right away. Call `edit_snapshot` only after an out-of-band user edit, to page a long timeline with `offset` and `limit`, or for handles you lack.\\n- `context` (`epoch`, `timelineId`, `revision`) is optional on every write except `edit_undo` and `project_manage` `project`, which take none. Without it, a write applies to the current state. With it, the editor refuses the write if the project changed since that context, which is what you want when a person may be editing at the same time. Every reply returns a fresh `context`. After `context_expired` (a reopen or restart), take a new snapshot.\\n- Timeline positions and durations are frames at `format.fps`. `sourceSeconds` values are seconds in the source file. Pass values as returned; do not multiply or divide by fps yourself.\\n- IDs are short handles. Pass them back exactly as returned; do not derive them from UUIDs. Tracks keep stable handles; indexes can change.\\n- When the user says \\"this clip\\", \\"these captions\\", or \\"here\\", take a fresh `edit_snapshot` (a one-frame window is enough). Its `selection` names what they selected in the project\'s window, in the snapshot\'s handles: `clipIds`, `captionGroupIds`, a `gap`, the marked `range`, and library `mediaIds`; no `selection` means nothing is selected. `currentFrame` is the playhead. `visible: false` means the user is not looking at this project.\\n- Send writes serially. Parallel writes against one project race each other\'s context.\\n- A refused call answers `status: \\"rejected\\"`: nothing changed, so fix what the message names and send it again. A service write that answers `failed` may have applied before a save or connection failed; inspect before repeating it.\\n- Use `inspect` for detail and verification: `timeline` for exact clip and track properties, `frame` for rendered frames of the composited result, `media` before describing source content, `color` for scopes, and `transcript` to locate spoken words.\\n- Volume values, including volume keyframes, are linear from `0` to `1`.\\n\\n## Edit with recipes\\n\\n`edit_apply` takes an ordered list of `actions`, plus an optional `context` and `requestId`. The editor validates the whole recipe before changing anything, then applies it as one undo step. If any action fails, nothing changes and the error names the action. An applied reply lists the resulting `clips` (position, length, source window, text); use it to confirm the edit instead of re-reading.\\n\\n- Concise actions: `place` (an `assetId` at `atFrame`, optionally on a `trackId`, with `durationFrames` or `sourceSeconds`, `mode` `overwrite` or `insert`), `trim` (a `clipId` to a `sourceSeconds` window), and `title` (`text` at `atFrame` for `durationFrames`, optional `look` and `motion`). Name an action with `as` and address its clip in later actions as `@name`.\\n- Advanced actions put their fields beside `kind` too: `place_batch`, `insert_batch`, `move`, `remove`, `split`, `extract`, `adjust`, `animate`, `arrange`, `transition` (fields inside `parameters`), `mask`, `replace`, `tracks`, `titles`, `text`, `grade`, and `effects`. Their field help lists every supported effect, grading control, mask, layout, speed curve, and caption control.\\n- Use `arrange` for split screens, picture-in-picture, grids, and canvas placement, and `tracks` to fix stacking. Do not synthesize layouts from generic transforms or keyframes.\\n- Use `replace` to swap a clip\'s media and keep its timing, effects, keyframes, and links, for example a re-render of the same shot. `trim: \\"keep\\"` holds the clip\'s source offsets; `trim: \\"reset\\"` plays the new media from its first frame and keeps linked audio in sync.\\n- Put every edit a request needs into one recipe: placing, trimming, titling, grading and effects together are one call and one undo step. Use `previewOnly: true` to validate a large or risky recipe without editing.\\n- Send a `requestId` when a retry must not edit twice: repeating the identical request with the same `requestId` returns the original receipt, even after a dropped connection. Without one, a repeated request applies again. Use a new `requestId` for a different edit.\\n- `applied_unsaved` means the edit applied but saving failed. If you sent your own `requestId`, repeat the identical request to retry the save, not the edit. Without one, do not resend: a repeat would apply the edit again.\\n- `edit_undo` reverses this connection\'s latest edit while it is still the top undo step. It never undoes the user\'s own edits. Take a new snapshot afterward.\\n\\nEdits are undoable. Do not ask for confirmation before each ordinary edit. Ask one focused question only when the user\'s creative direction is materially ambiguous.\\n\\n## Bring generated or local media into the editor\\n\\nFor b-roll and establishing shots, search free stock first (`videodraft stock search \\"<query>\\"`), import the pick with `videodraft stock import <ref>`, and feed the returned CDN URL to `library_manage` `import` as `source.url`. It costs no credits.\\n\\nOtherwise use cloud generation for new assets, save or download the outputs, then call `library_manage` with `operation: \\"import\\"`:\\n\\n- `source.path`: absolute local file or directory. A directory imports recursively and preserves its folder structure.\\n- `source.url`: HTTPS asset URL. Set `mimeType` when a signed URL has no usable extension.\\n- `source.bytes`: small base64 media with a required `mimeType`.\\n- `source.matte`: generated solid-color image.\\n\\nReadiness differs by source, and so does the poll that detects it:\\n\\n- **URL and single-file path** imports return `status: \\"downloading\\"` with one `mediaRef`. Poll `library_manage` `list` with `parameters.ids: [mediaRef]` until `generationStatus` is absent.\\n- **Directory** imports also return `status: \\"downloading\\"` with one placeholder `mediaRef` for the batch. Poll it the same way; the folder\'s assets appear when `generationStatus` clears. (`pending: true` remains a fallback that lists every unresolved import.)\\n- **Inline bytes and matte** imports finish inline and come back `status: \\"ready\\"`; no polling is needed.\\n\\nAn import\'s `mediaRef` is the `assetId` that recipe actions take. Never place a pending asset on the timeline. `generationStatus` is the signal: `preparing`, `generating`, `downloading`, and `rendering` mean keep polling, absent means usable, and **`failed` is terminal**. Report it or retry the import explicitly; never poll on. Do not treat \\"not downloading\\" as ready.\\n\\nA `.srt` or `.vtt` caption file (by `path`, `url`, or `bytes` with `mimeType` `application/x-subrip`, `text/srt`, or `text/vtt`) is not added to the library. Its captions go onto a new caption track at the times the file gives, as one undo step, and the reply names the new `captionGroupId` to restyle with a `text` action.\\n\\nFor a batch of local outputs, download them into one workspace directory and import that directory once when practical. This is safer and faster than racing many import calls.\\n\\nTo apply a LUT, first store it with `library_manage` `prepare_look` (`parameters.path` to a `.cube` file), then use the returned `storedPath` in a `grade` action.\\n\\n## Speech edits\\n\\n`speech_apply` runs speech work as separate operations, not recipe actions: `words` removes transcript words, `silence` trims dead air, and `captions` generates styled captions. Read a fresh transcript with `inspect` `transcript` after any speech edit, because word positions change.\\n\\n## Service operations and timeouts\\n\\n`project_manage`, `library_manage`, `speech_apply`, and `delivery_manage` writes have no retry receipts. After a timeout, inspect the outcome (a snapshot, a library `list`, or delivery `jobs`) before repeating a write. A failed service write may have applied an edit before saving failed, so read its `data` and the current state.\\n\\n## Edit and verify\\n\\nA dependable sequence:\\n\\n1. `project_manage` to select or create the local project; its reply carries the snapshot.\\n2. `inspect` `media` when content selection matters.\\n3. One `edit_apply` recipe with every clip, track, layout, text, audio, color, and effect change the request needs; `speech_apply` for word cuts, silence, and captions.\\n4. Check the reply\'s `clips`; use `inspect` `frame` only when visual composition or layer order matters.\\n5. `edit_undo` if the result is wrong and the next recipe would not cleanly correct it.\\n\\n## Export\\n\\nCall `delivery_manage` with `operation: \\"submit\\"`. Submission is not completion: it returns a job, destination, and `started` or `queued` status.\\n\\n- Use `mode: \\"video\\"` for H.264, H.265, or ProRes.\\n- Use `mode: \\"xml\\"` (XMEML) for Premiere Pro **and DaVinci Resolve**. Resolve reads XMEML natively. Use `fcpxml` only for Final Cut Pro; sending Resolve an FCPXML produces a package it cannot open cleanly.\\n- Use `mode: \\"videodraft\\"` for a self-contained project package.\\n- Use `mode: \\"srt\\"` or `mode: \\"vtt\\"` for a subtitle file of the captions: words and timing only, without styling. `captionGroupId` picks a caption group; `wordTiming: true` adds each word\'s start to a `vtt`.\\n- Omit `outputPath` unless the user named a destination; the default is `~/Downloads`. The editor does not create folders, so a named destination\'s folder must already exist.\\n- Use `operation: \\"jobs\\"` with `parameters.action` `list` to follow progress, warnings, and results. Cancel only when the user asks or the just-queued settings were wrong. Do not infer that an export is stuck from elapsed time alone.\\n- A finished video export is then checked in the background for black picture, gaps, sound that drops out or clips, and missing media. `list` shows each job\'s `findingCount` and its first three findings; `list` with that `jobId` returns every finding. Findings are warnings on a completed export: tell the user what was found and where, rather than treating the export as failed.\\n\\n## Terminal bridge\\n\\nVideoDraft desktop terminals expose `videodraft-editor`, which controls the same process and tools:\\n\\n```bash\\nvideodraft-editor status\\nvideodraft-editor list-tools\\nvideodraft-editor tool project_manage --json \'{\\"operation\\":\\"project\\",\\"parameters\\":{\\"action\\":\\"list\\"}}\'\\nvideodraft-editor tool edit_snapshot --project \\"/path/to/My Video.vdproject\\" --json \'{\\"includeLibrary\\":true}\'\\nvideodraft-editor show\\nvideodraft-editor hide\\n```\\n\\nEach `videodraft-editor` invocation is a separate connection, so a project opened by one call is NOT remembered by the next. Pass `--project <path>` on every `tool` call that operates on a project; without it, a follow-up call answers `no_project` (no project is open for this session) even though the open succeeded. For the same reason, `edit_undo` has nothing to undo from the terminal: it only reverses an edit made on its own connection.\\n\\nControl and tool commands auto-start a headless editor if none is running. `show` only reveals the already-running UI. Use `videodraft-editor tool <name> --json -` to read a JSON object from stdin when shell quoting would be fragile. Never read, copy, or expose the editor\'s rotating local authentication secret.\\n","references/examples.md":"# Recipes\\n\\nWorking patterns for common asks. All assume auth (`videodraft login` once, or `VIDEODRAFT_API_KEY` in the environment) and use `--json` for parsing.\\n\\nInside VideoDraft ADE, if a native editor is available, use these recipes for asset generation and\\noptional script/storyboard stages. Hand the results to the editor for production and export ONLY\\nwhen the deliverable the user asked for is a composed production. A standalone output \u2014 the batch\\nproduct clips in recipe 1, the upscale in recipe 7 \u2014 is finished when it is generated; importing it\\ninto a project and exporting a timeline builds an edit nobody asked for. Recipes below that call `videodraft produce` or `videodraft export` are hosted fallbacks only. Do not choose them over the available native editor unless the user explicitly asks for the hosted web workflow.\\n\\n## 1. Batch product videos from a CSV\\n\\nOne 9:16 product clip per row of `products.csv` (`name,image_url,tagline`):\\n\\n```bash\\n#!/usr/bin/env bash\\nset -euo pipefail\\nmkdir -p outputs\\n\\nwhile IFS=, read -r name image tagline; do\\n job=$(videodraft generate video \\\\\\n \\"Premium product shot of ${name}: ${tagline}. Slow orbit, studio lighting.\\" \\\\\\n --model gemini-omni-1.1-flash --ar 9:16 --duration 6 \\\\\\n --start-image \\"$image\\" \\\\\\n --no-wait --json | jq -r .job_id)\\n echo \\"$name,$job\\" >> outputs/jobs.csv\\ndone < <(tail -n +2 products.csv)\\n\\n# Collect ALL results with ONE process (batched polling \u2014 one request per tick)\\nvideodraft wait $(cut -d, -f2 outputs/jobs.csv) \\\\\\n --download \\"outputs/{job_id}_{index}.{ext}\\" --json > outputs/results.json\\n# map job ids back to product names via outputs/jobs.csv\\n```\\n\\nSubmit-then-collect parallelizes server-side generation; the single multi-id `wait` keeps it to one local process and one batched poll request per tick no matter how many jobs. Gemini Omni 1.1 Flash is selected because these are six-second first-frame product clips. Estimate first: `videodraft costs gemini-omni-1.1-flash --type video --duration 6 --resolution 720p --audio` \xD7 rows, and confirm with the user.\\n\\nExtend an uploaded clip or conversationally edit an official Google interaction with Gemini Omni 1.1 Flash:\\n\\nEach extension appends 3-10 seconds at the end. Edit and extend accept EXACTLY ONE input video, so `--ref-video` cannot accompany `--source-video` or `--previous-interaction-id`; only `--ref` images can. New dialogue is allowed only when the source video is silent, so a chain of spoken beats must carry its speech in the first turn. A continuation is submitted as the prior turn\'s output, so 40 seconds total is reachable only by extending from a source of 30s or less, and a ladder dead-ends past 30s.\\n\\n```bash\\n# Uploaded-video extension. Reference IMAGES are fine here; reference videos are not.\\nvideodraft generate video --model gemini-omni-1.1-flash \\\\\\n --source-video ./ending.mp4 --ref ./wardrobe.png \\\\\\n --extend --duration 6 --resolution 1080p \\\\\\n --download ./media/extended.mp4\\n\\n# Conversational edit from the interaction_id of an earlier generation. Add --extend to lengthen instead.\\nvideodraft generate video --model gemini-omni-1.1-flash \\\\\\n --previous-interaction-id \\"$INTERACTION_ID\\" \\\\\\n --resolution 720p \\\\\\n --download ./media/continued.mp4\\n```\\n\\nFal BYOK supports the currently callable Gemini Omni 1.1 generation and basic source-edit endpoints at zero VideoDraft credits. Its published callable v1.1 endpoints do not expose continuation, extension, or separate creative references on a source edit. Do not infer a Fal route or retry on paid Google while Fal BYOK is active.\\n\\n## 2. Hosted full marketing video from one idea (fallback)\\n\\n```bash\\nvideodraft create \\"30-second launch video for Solace, a sleep-tracking ring. Calm, premium, dark palette.\\" \\\\\\n --ar 9:16 --style cinematic --json > project.json\\nPROJECT=$(jq -r .project_id project.json)\\n\\nvideodraft shots \\"$PROJECT\\" --grid --estimate # show the user the cost; get a go-ahead\\nvideodraft shots \\"$PROJECT\\" --grid\\nvideodraft produce \\"$PROJECT\\"\\nvideodraft generate music \\"minimal ambient, warm pads, 60 BPM\\" --attach \\"$PROJECT\\"\\nvideodraft generate audio \\"Extend @Audio1 into a 20-second transition\\" --ref-audio ./intro.wav --format wav --download ./transition.wav\\nvideodraft export \\"$PROJECT\\" --download solace-launch.mp4\\n```\\n\\nUse this complete hosted path only when the user requested a web project or the native editor is unavailable. Otherwise stop after the storyboard/assets, import them into the native `.vdproject`, and export with `delivery_manage`. The hosted project stays editable at the URL in `project.json` (`.urls`).\\n\\n## 3. Talking-head (avatar) video\\n\\nWhen the user has no portrait, generate a clear front-facing avatar image first. Skip this step when they supplied one or an existing character should be reused.\\n\\n```bash\\nvideodraft generate image \\\\\\n \\"Front-facing head-and-shoulders portrait of a friendly coffee expert, direct eye contact, natural expression, clean studio background\\" \\\\\\n --model nano-banana-2 --ar 9:16 --download ./media/avatar.png\\n\\nSCRIPT=$(videodraft avatar script \\"why our espresso subscription saves you money\\" --style ad-style --json | jq -r .script)\\nAVATAR=$(videodraft avatar create ./media/avatar.png --script \\"$SCRIPT\\" --voice elevenlabs-kPzsL2i3teMYv0FxEYQ6 --ar 9:16 --json | jq -r .avatar_video_id)\\nvideodraft avatar render \\"$AVATAR\\" --resolution 720p # VEED Fabric paid step; confirm cost first (~20 credits/sec)\\n```\\n\\n`avatar script` and `avatar create` (including speech) are bundled/free. In this example only the optional portrait generation and Fabric render spend credits.\\n\\nIf the portrait is low resolution, enhance it before `avatar create`:\\n\\n```bash\\nvideodraft upscale image ./founder-small.jpg --scale 2x --download ./media/founder-upscaled.png\\n```\\n\\nFor a one-off portrait animation without creating a managed avatar record:\\n\\n```bash\\nvideodraft avatar fabric ./founder.jpg \\\\\\n --text \\"Welcome to the weekly product update.\\" \\\\\\n --voice-description \\"warm, confident American presenter\\" \\\\\\n --resolution 720p --download ./media/presenter.mp4\\n```\\n\\nFor a short, sharp clip from speech or a song the user already has (5-14.8s of audio is used):\\n\\n```bash\\nvideodraft avatar h3-lipsync ./singer.png --audio ./chorus.mp3 \\\\\\n --resolution 1080P --audio-duration 12 --estimate\\nvideodraft avatar h3-lipsync ./singer.png --audio ./chorus.mp3 \\\\\\n --resolution 1080P --download ./media/singer-lipsync.mp4\\n```\\n\\nWhen the user already has both the video and replacement speech:\\n\\n```bash\\nvideodraft avatar lipsync ./presenter.mp4 \\\\\\n --audio ./localized-voiceover.mp3 \\\\\\n --sync-mode loop --download ./media/presenter-localized.mp4\\n```\\n\\nEdit an existing video with a dedicated edit model:\\n\\n```bash\\nvideodraft models video --category video_edit\\nvideodraft edit video ./product-demo.mp4 \\\\\\n \\"Preserve the product but follow the reference camera rhythm\\" \\\\\\n --model gemini-omni-1.1-flash --ref-video ./camera-rhythm.mp4 \\\\\\n --ref-video-duration 2.5 --resolution 1080p \\\\\\n --download ./media/product-demo-reframed.mp4\\n\\nvideodraft edit video ./product-demo.mp4 \\\\\\n \\"Turn the room into a warm evening scene while preserving the product and camera motion\\" \\\\\\n --model kling-o3-video-ref-edit --ref ./evening-style.jpg \\\\\\n --preserve-audio --download ./media/product-demo-evening.mp4\\n```\\n\\nTransfer motion from a reference clip onto a character image:\\n\\n```bash\\nvideodraft edit motion ./character.png \\\\\\n \\"Apply the dancer\'s movement to this character while preserving identity\\" \\\\\\n --motion-video ./dance-reference.mp4 \\\\\\n --model kling-v3-motion-control --quality pro \\\\\\n --download ./media/character-dance.mp4\\n```\\n\\n## 4. Hosted changelog video in CI\\n\\nIn a GitHub Action with `VIDEODRAFT_API_KEY` set as a secret:\\n\\n```bash\\nNOTES=$(git log --oneline v1.2.0..HEAD | head -20)\\nvideodraft create \\"Weekly product update video. Energetic, 20 seconds. Changes: ${NOTES}\\" --ar 16:9 --json > p.json\\nPROJECT=$(jq -r .project_id p.json)\\nvideodraft shots \\"$PROJECT\\" && videodraft produce \\"$PROJECT\\"\\nvideodraft export \\"$PROJECT\\" --download changelog.mp4 --wait-timeout 30m\\n```\\n\\n## 5. Variations and picking a winner\\n\\n```bash\\nvideodraft generate image \\"logo concept: minimalist fox, geometric\\" --num 4 --download \\"./concepts/{job_id}_{index}.{ext}\\" --json\\n# Show all 4 to the user; regenerate the chosen one at higher res:\\nvideodraft generate image \\"<same prompt>\\" --model nano-banana-pro --resolution 4K\\n```\\n\\n## 6. Reaching tools without a curated command\\n\\n```bash\\nvideodraft tools list --json | jq -r \'.[].name\'\\nvideodraft tools schema attach_media_to_shot --json\\nvideodraft call attach_media_to_shot --args \'{\\"project_id\\":\\"...\\",\\"scene_index\\":0,\\"shot_index\\":1,\\"media_url\\":\\"https://...\\",\\"media_type\\":\\"video\\",\\"duration_seconds\\":6}\'\\n```\\n\\nAnything the hosted VideoDraft MCP exposes, including character studio, product studio, and hosted project data, is reachable this way even before it gets a curated command. Native `.vdproject` editing uses the separate `videodraft_editor` MCP described in SKILL.md.\\n\\n## 7. Enhance an existing asset without changing it\\n\\n```bash\\n# Light image cleanup, no enlargement\\nvideodraft upscale image ./poster.png --scale 1x --download ./media/poster-enhanced.png\\n\\n# General image and video enlargement\\nvideodraft upscale image ./frame.png --scale 2x --download ./media/frame-2x.png\\nvideodraft upscale video ./clip.mp4 --resolution 1080p --download ./media/clip-1080p.mp4\\n\\n# Rebuild detail in a tiny/blurry source (generative), or reimagine it (creative)\\nvideodraft upscale image ./tiny.jpg --scale 4x --mode generative --model \\"Wonder 3.5\\" --download ./media/tiny-4x.png\\nvideodraft upscale video ./ai-clip.mp4 --resolution 4k --mode generative --download ./media/ai-clip-4k.mp4\\n\\n# Smoother motion or slow motion (resolution unchanged)\\nvideodraft interpolate ./clip.mp4 --fps 60 --download ./media/clip-60fps.mp4\\nvideodraft interpolate ./clip.mp4 --model Chronos --slowdown 4 --download ./media/clip-slowmo.mp4\\n```\\n\\nUse these when the content is correct and only quality or resolution needs improvement. If the poster text, composition, subject, or motion is wrong, edit or regenerate instead.\\n","references/models.md":"# Choosing models (and predicting cost)\\n\\nAlways consult the live catalog instead of memorizing this page \u2014 models change weekly:\\n\\n```bash\\nvideodraft models image --json # every image model + inputs (aspect ratios, resolutions, max refs)\\nvideodraft models video --json # every video model + inputs + per-second pricing metadata\\nvideodraft models audio --json # standalone audio/media models + pricing inputs\\nvideodraft models voices --json # TTS voices\\nvideodraft models styles --json # visual style presets\\n```\\n\\n## Task-based model selection\\n\\nFollow this order every time:\\n\\n1. The user named a model: use it, never substitute.\\n2. Otherwise pick a **Tier 1** model from the task\'s inputs, duration, audio, quality, speed, and cost, and pass it explicitly instead of relying on a blind platform fallback.\\n3. Use a **Tier 2** model only when no Tier 1 model supports the request, and say why. Never pick Tier 2 on your own for price, speed, novelty or variety.\\n\\nTier 1 is the rows marked **T1** below. Every other catalog model is Tier 2 and stays available by name. The catalogs carry this themselves: every `videodraft models image|video|audio --json` entry has `tier` (1 or 2) plus `recommended` / `recommended_for`, and the top-level `recommended` array is Tier 1, best first. Trust the catalog over this page when they disagree.\\n\\n### Images\\n\\n| Need | Choose | Why |\\n| ------------------------------------------------------------------------------------------ | ------------------------ | ----------------------------------------------------------------- |\\n| T1: Most generation, editing, character consistency, or reference work | `nano-banana-2` | Best general default; 1K/2K/4K and up to 14 reference images |\\n| T1: Highest-quality complex generation or reasoning | `nano-banana-pro` | Premium Nano Banana quality and reasoning |\\n| T1: Fast, inexpensive drafts and iteration | `nano-banana-2-lite` | Fastest/cheapest Nano Banana option; 1K only, up to 14 references |\\n| T1: Posters, title cards, signs, logos, or any image with important readable text | `gpt-image-2.5-flare` | Fast OpenAI option; 16 references, 1K/2K/4K and PNG output |\\n| T1: Complex multi-image composition, precise editing, or a strong alternate interpretation | `gpt-image-2.5-sunburst` | More fidelity/detail; 16 references, quality through max |\\n| T1: Vector / SVG output (logos, icons, illustrations that must scale) | `recraft-v4` | Recraft V4.1 text-to-vector; SVG only, no reference images |\\n| T2: By name, or an explicitly requested xAI look | `grok-imagine` | 2 cr flat (3 with a reference); 1 reference image |\\n| T2: xAI at 2K or with a quality tier, up to 3 edit references | `grok-imagine-2.0` | 1K 4 (low) / 6 (medium), 2K 6 / 8, +1 cr per reference image |\\n\\nGPT Image 2.5 has two IDs: `gpt-image-2.5-flare` replaces the former GPT Image 2 recommendation; `gpt-image-2.5-sunburst` is the precision/detail alternative. Other model recommendations are unchanged. Explicit `gpt-image-2` selections continue to use the previous model.\\n\\nUse the same basic image controls as GPT Image 2: `--ref`, `--ar`, `--resolution 1K|2K|4K`, `--quality`, and `--num 1..4`. GPT Image 2.5 quality supports auto/low/medium/high/xhigh/max. Auto is priced at Max. Generation runs directly on OpenAI, or exclusively on the user\'s Fal key when active.\\n\\n```bash\\nvideodraft generate image \'A blue glass bird\' --model gpt-image-2.5-flare --ar 1:1 --resolution 2K --quality high --estimate\\nvideodraft generate image \'A blue glass bird\' --model gpt-image-2.5-flare --ar 1:1 --resolution 2K --quality high --num 2\\n```\\n\\n`videodraft shots` forwards resolution and quality for both normal and grid generation. Estimates use the project\'s model and aspect when omitted; grid canvas costs may differ from ordinary shot costs. Use `videodraft costs --model <id> --ar <ratio> --resolution <tier> --quality <tier>` for a specific per-image quote.\\n\\n### Videos\\n\\n| Need | Choose | Important limits |\\n| -------------------------------------------------------------------------------------------------------------- | ------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |\\n| T1: Most generation, first/last-frame, mixed-reference, source-edit, or extension requests | `gemini-omni-1.1-flash` | 3-10s output; 360p/720p/1080p/4K at 3/10/15/30 cr/s; audio always; up to 10 image inputs, and 3 reference videos <=3s on generate only; edit source <=10s; continuation or 3-10s extension of a 1-30s source, 40s total reachable only by extending from <=30s |\\n| T2: Grok 1.5 text, first-frame, or 1-7 image-reference clips with native audio and optional 1080p | `grok-imagine-video-1.5` | 1-15s; 480p/720p/1080p for text/first-frame; references are 480p/720p only; no last frame |\\n| T2: Document or webpage reference generation; by name otherwise | `wan-3.0` | 2-30s or auto; 480p/720p/1080p at 7/14/28 cr/s; 10 image, 5 video, 5 audio refs, 20 media files total; document/web refs require thinking |\\n| T2: 480p/768p/2K/4K with native stereo audio, first/last frames, or mixed image/video/audio references | `minimax-h3` | 5-15s; 5/6/13/16 cr/s by resolution; up to 9 image, 3 video, 3 audio refs, 12 files total; first 5 reference images free then 8 cr each; reference video/audio each total <=15s |\\n| T1: Text, first/last-frame, or mixed-reference video with native audio and stronger prompt adherence | `minimax-h3-max` | 5-15s; 5/8 cr/s at 480p/768p plus pooled reference tokens; up to 9 image, 3 video, 3 audio refs, 12 files total; seed and disabled/balanced/quality prompt expansion; provider safety checker off by default |\\n| T2: Images pinned to specific moments (keyframes); by name otherwise | `flux-3` | 5-20s (auto for text/first-frame only); 720p/1080p; up to 10 keyframes; `--quality draft` is 720p-only at ~1/3 the cost |\\n| T1: Video/audio references, mixed reference media, broad aspect ratios, frame-mode first+last frame, or 11-15s | `seedance-2` | 4-15s or auto; up to 9 image, 3 video, and 3 audio refs; audio toggle; Mini/Fast are 480p/720p only |\\n| T1: Single takes past 15s, or more references than Seedance 2.0 allows | `seedance-2.5` | 4-30s or auto; up to 30 image, 10 video, and 10 audio refs (50 files total); one quality tier; 480p/720p/1080p |\\n| T2: Fast polished 3-15s video with first frame, multi-prompt, and native audio | `kling-v3-turbo` | Audio always on; Pro default; no end frame, reference-media mode, or elements |\\n| T1: Cinematic 3-15s with reference images, image/video elements, first+last frame, audio, or 4K | `kling-o3` | 7 combined image refs/elements, reduced to 4 with a video element; Standard/Pro/4K |\\n| T1: Kling 3-15s image-to-video with first+last frame, image/video elements, multi-prompt, audio, or 4K | `kling-3.0` | Elements require a start image; image or video elements can bind voice_id; Standard/Pro/4K |\\n| T2: User explicitly requests Veo, or the selected workflow specifically needs Veo | `google-veo3.1` | Good fallback, but not the preferred general model |\\n\\nRouting rules:\\n\\n- Inexpensive pass or draft at 480p/768p with native audio: use MiniMax H3 Max (5/8 cr/s plus reference tokens).\\n- Grok 1.5 reference mode accepts 1-7 images only. Address them in array order as `<IMAGE_0>` through `<IMAGE_6>`. Do not combine reference images with `--start-image`, `--end-image`, `--ref-video`, or `--ref-audio`.\\n- Grok 1.5 first-frame mode accepts one `--start-image`, derives the output aspect ratio from that image, and does not support `--end-image`. Text and first-frame modes support 480p, 720p, or 1080p. Reference mode supports 480p or 720p.\\n- Grok 1.5 always generates native audio. Do not pass `--no-audio`, `--seed`, `--negative`, or `--quality`.\\n- Around 11-15 seconds with native audio: use Seedance, Kling 3.0 / O3, or MiniMax H3 Max, not Gemini. Past 15 seconds: Seedance 2.5.\\n- Dialogue: write the line in the prompt, in quotes, with who says it and how. `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` and `kling-o3` speak it natively; add `--ref-audio` on Seedance or a bound Kling voice for a specific voice. Do not generate speech separately and lip-sync it on.\\n- One existing source video that should be edited: use `videodraft edit video`, which auto-selects Gemini Omni 1.1 Flash for a source up to 10s. Generic `generate video --ref-video` retains source-edit behavior when `--video-task` is omitted, but the dedicated edit command is clearer.\\n- Video supplied as a creative reference: Gemini Omni 1.1 Flash accepts up to 3 videos of at most 3 seconds each, including mixed image and video input. Creative video references are accepted on `--video-task generate` ONLY: edit and extend take exactly one input video, so `--ref-video` cannot accompany `--source-video` or `--previous-interaction-id`. Use Seedance 2.5 for more or longer video/audio references, or Seedance 2.0 when quality-tier control matters.\\n- First and last frame control: every Tier 1 video model supports it (Gemini Omni 1.1 Flash, Seedance, Kling O3, Kling 3.0, MiniMax H3 Max). For Gemini, every start/end frame counts toward the 10-image total, leaving up to 9 references with a start frame or 8 with both frames.\\n- Extension with a 1-30s uploaded source uses Gemini Omni 1.1 Flash plus `--source-video` and `--video-task extend` (or `--extend`). Pass an explicit 3-10s output duration; the model appends it at the end, up to 40 seconds total. Creative `--ref-video` clips CANNOT accompany the source: edit and extend take exactly one input video. New dialogue is allowed only when the source video is silent; adding speech on top of a source that already has speech is refused with \\"the model is currently unable to process speech edits\\". Continuation uses `--previous-interaction-id <id>` and defaults to a conversational edit; `--video-task extend` lengthens it. A continuation is submitted as the prior turn\'s output video, so it obeys the same source limits (<=30s to extend, <=10s to edit) and the same silent-source dialogue rule. 40s total is reachable only by extending from a source of 30s or less, so a ladder dead-ends past 30s. Fal returns interaction IDs, but its currently published callable v1.1 endpoints do not expose continuation/extension or mixed source-edit references; never infer a route or silently fall back to paid Google.\\n- Wan 3.0 reference mode and frame mode are separate. It accepts 10 images, 5 videos, and 5 audio clips, at most 20 media files total. Video and audio each total at most 15 seconds. `--file-url` and `--web-url` require `--thinking`. `--auto-duration` cannot be combined with `--duration`.\\n- MiniMax H3 and H3 Max reference mode and first-plus-last-frame mode are separate. Audio cannot be the only reference. Address references as `Image 1`, `Video 1`, and `Audio 1` in array order.\\n- Seedance reference mode and first-plus-last-frame mode are separate. Do not promise reference video/audio plus a last frame in one generation.\\n- Multi-prompt sequencing: use Kling 3.0 Turbo, Kling O3, or Kling 3.0.\\n- Kling 3.0 image-to-video and Kling O3 reference-to-video accept repeatable structured `--element` JSON. Each element is image-backed (`frontal_image_url` plus 1-3 `reference_image_urls`) or video-backed (`video_url`). Either form may include `voice_id`. Keep the voice ID inside the same element so the association is preserved. Reference them as `@Element1`, `@Element2`. Kling V3 Turbo rejects elements.\\n- Kling 2.6 Pro image-to-video accepts one or two repeatable `--voice-id` values. Cite them as `<<<voice_1>>>` and `<<<voice_2>>>` in the prompt. Voice control costs 17 cr/s. Use `videodraft kling-voices list|create|delete` to manage the separate Kling video-control voice library. Creating a voice requires a clean 5-30 second, up-to-50MB single-speaker `.mp3`, `.wav`, `.mp4`, or `.mov` sample plus `--confirm-consent`. Creation costs 1 VideoDraft credit on the platform Fal account and 0 credits with Fal BYOK; use the create command\'s `--estimate` flag to check without creating.\\n- Kling V3 voice control costs 16 cr/s for Standard and 20 cr/s for Pro.\\n- Seedance quality: `mini` for the lowest cost, `fast` for speed, `standard` for maximum quality and for 1080p/4K. Seedance 2.5 has a single tier and ignores `--quality`.\\n- Longer than 15 seconds, or more than 9 image / 3 video / 3 audio references: use Seedance 2.5. It reaches 30s and 30/10/10 references (50 files total) at 480p/720p/1080p, but has no 4K.\\n- Seedance 2.x allows real people by default, the same as AI Studio: Byteplus first, a submit-time Fal fallback, and Fal\'s higher tier-specific rate. Turn it off with MCP `allow_real_people: false` or CLI `--no-allow-real-people` for the lower Byteplus-only rate, only when the user wants the cheapest run and nothing in the job is a real identifiable person (text-only requests, non-people, anime, clearly synthetic or stylized characters).\\n- If a Seedance request made with the option off fails with `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, estimate the higher rate, follow the user\'s spend policy, and retry once with the option on. The response also carries `retry_with: { allow_real_people: true }`, `cli_flag: \\"--allow-real-people\\"`, and `retry_policy: \\"once\\"`. If the option was already on, do not repeat the same request. Byteplus may accept a task and reject the output later. VideoDraft refunds that failed generation, but it cannot reroute the asynchronous failure to Fal. Rephrase or change the references instead.\\n\\n### Video edit and motion-control categories\\n\\nUse `videodraft models video --category video_edit` for existing-video transforms and `--category motion_control` for motion transfer.\\n\\n| Need | Command/model | Important limits |\\n| ------------------------------------------ | ---------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------- |\\n| Most edits of a source up to 10s (DEFAULT) | `videodraft edit video <video> \\"...\\"` (auto-selects `gemini-omni-1.1-flash`) | Source <=10s or it REFUSES; up to 10 image refs and 3 creative video refs <=3s each; 360p/720p/1080p/4K; audio always regenerated (no `--preserve-audio`) |\\n| Cheapest simple prompt edit | `videodraft edit video <video> \\"...\\" --model grok-imagine-video-edit` | No image refs; source silently truncated to 8s; auto/480p/720p |\\n| Edit with several image references | `--model happy-horse-video-edit --ref ...` | Up to 5 refs; 720p/1080p; source silently truncated to 15s; highest rate (28-56 cr/s) |\\n| Controlled Kling edit | `--model kling-o3-video-ref-edit --ref ...` | Up to 4 refs; Standard/Pro; source silently clamped to 3-10s |\\n| Transfer reference motion to an image | `videodraft edit motion <image> [direction] --motion-video <video>` | Prompt/direction is optional; Kling V3 default; optional one image-only `--element`; element requires video orientation |\\n\\nIf the user explicitly names one of these models, preserve it. The CLI uploads local source videos and reference images automatically. Editing returns an async job and waits by default.\\n\\n**Omit `--model` and the SERVER chooses**, because only it can measure the source. It picks `gemini-omni-1.1-flash` for any source up to 10s. When that is not safe (source longer than 10s, unmeasurable duration, `--preserve-audio`, more than 10 refs, or Fal BYOK with refs) it spends nothing and returns a priced menu: per model, the seconds it would edit, the seconds it would drop, and the credit cost. The CLI prints that table and exits 2. Show it to the user, then re-run with `--model`.\\n\\n**Truncation is the trap here.** Only Gemini refuses a source it cannot fully consume. Every other edit model accepts a 30s clip and returns an edit of its first 8-15 seconds with no error. `videodraft edit video` now warns when this will happen and the tool response carries `source.truncated` / `source.dropped_seconds`; relay it to the user rather than letting them discover it in the output.\\n\\nTo edit a longer source without losing its tail, cut it into <=10s pieces with `videodraft_editor`, edit each with Gemini, then reassemble and export there. VideoDraft has no server-side split/concat, so this needs the native editor.\\n\\nKling O3 is also exposed for reference generation. `videodraft generate video --model kling-o3-video-ref-edit` requires exactly one `--ref-video` and generates a new reference-guided clip. Wan 3.0 handles new text, frame, and mixed-reference generation, but is not an existing-source edit model.\\n\\n### Reference-first video workflow\\n\\n- Prefer a start frame or reference image whenever a specific character, product, location, style, composition, or brand identity must stay recognizable.\\n- If the user gives a reference, pass it. Never silently replace it with a text description.\\n- If no reference exists and continuity matters, generate a still first with the user\'s explicitly requested compatible image model, otherwise use Nano Banana 2. Wait for the image URL, then animate it with the selected video model. Confirm the combined image plus video cost before starting.\\n- For multi-shot scenes, generate shot images with `videodraft shots <project_id> --model <selected-image-model> --grid`. Preserve an explicitly requested compatible image model; otherwise use `nano-banana-2`. The grid establishes the scene and characters together, then decodes into individual shot images.\\n- Animate the decoded shot images as per-shot start frames or references. Do not independently text-generate each video clip when the shots need to match.\\n- Pure text-to-video remains appropriate for generic one-off footage where no subject, composition, or continuity needs to be preserved.\\n\\n### Audio\\n\\n- **Seed Audio 1.0**: use `videodraft generate audio` for open-ended speech, sound, music, or prompt-driven audio editing. It accepts up to three audio references or one image. Address audio references as `@Audio1`, `@Audio2`, and `@Audio3`. Preset and custom cloned voice IDs are supported. Output is up to 120 seconds. There is no requested-duration input. The CLI automatically retries transient responses with one idempotency key. To recover after the CLI process itself is interrupted, set `--idempotency-key <uuid>` on the original command and reuse it.\\n- **Voiceover/TTS**: prefer ElevenLabs. Brittney is the platform default voice; under ElevenLabs BYOK, use a compatible voice from the user\'s account. Honor another supported voice/provider when the user explicitly selects it.\\n- **Dialogue, voice changing, and dubbing**: ElevenLabs only.\\n- **Sound effects**: ElevenLabs Sound Effects only.\\n- **Music**: use `lyria-3.5` for all music, short or long (up to ~3 minutes), with vocals/lyrics or instrumental music. When `--model` is omitted the CLI sends no model and the server default applies, which is `lyria-3.5` wherever the backend supports it. If the server answers that `lyria-3.5` is an unknown model, that backend predates it: rerun without `--model` (the server default then applies), or use `lyria-3-pro-preview` for a track longer than 30 seconds. `lyria-3-clip-preview` (fixed ~30s) and `lyria-3-pro-preview` remain available for explicit requests. Lyria length and structure are prompt-guided, not exact; for an instrumental, include \\"instrumental only, no vocals\\" in the prompt. `--length` and `--instrumental` only apply to ElevenLabs. Use `elevenlabs-music-v2.5` when a specified 3-300 second length, composition plan, or style reference track matters. Lyria references: up to 10 images on Google, 1 on Fal BYOK; 3.5 Fal prompts are 1-5000 characters (an image-only request gets a neutral prompt). `elevenlabs-music` is an alias for v2.5. Use `elevenlabs-music-v1` only when the user asks for v1 by name; it takes a prompt, `--length` and `--instrumental` only.\\n- **ElevenLabs Music v2.5 inputs**: either a prompt (`--length`, `--instrumental`) or a composition plan, never both. A plan is 1-30 sections of 3-120 seconds each, up to 300 seconds in total, and `--plan`/`--section` pick v2.5 when `--model` is omitted. Empty section parts default to 20 seconds and an instrumental part. Build one with repeatable `--section \\"<seconds>|<style, style>|<text>\\"` (use `\\\\n` for line breaks) or pass `--plan plan.json` with `{\\"chunks\\":[{\\"text\\",\\"duration_ms\\",\\"positive_styles\\",\\"negative_styles\\",\\"context_adherence\\",\\"audio_reference\\"}]}`. Section text is an optional `[Section name]`, lyric lines, and `{inline directions}`. Put 6-7 specific English styles on the first section; it sets the genre. `--ref-audio <url|file>` adds a style reference clip to the first section (window up to 30 seconds via `--ref-start`/`--ref-end` in ms, `--ref-strength low|medium|high|xhigh`); local files are uploaded, including relative `audio_reference.audio_url` paths in a plan file. `--seed` works only with a plan. `--format` picks the output (`mp3_48000_192` default on v2.5; `pcm_*`, `ulaw_8000` and `alaw_8000` arrive as stereo WAV files). Style references don\'t run on the user\'s own ElevenLabs key; with that key connected, drop the reference or ask the user to turn the key off. The CLI retries transient responses with one idempotency key; set `--idempotency-key <uuid>` to recover after an interruption.\\n- Voice Changer and Dubbing require the source media duration and currently accept source media up to 300 seconds.\\n\\n**User-supplied ElevenLabs voice IDs:** voiceover, dialogue, and voice changing accept raw IDs (16-64 alphanumeric characters) or `elevenlabs-<id>`. `videodraft models voices` / MCP `list_available_voices` is for discovery, not an allowlist. Pass a supplied ID directly even when it is absent from the catalog; do not reject it or substitute a catalog voice. The voice must be accessible to the provider account used for generation. Private or cloned voices may require the user\'s connected ElevenLabs key. A failed voice listing does not prove that a supplied ID cannot generate; a connected key with generation access may still work. If the provider rejects generation access, report that error rather than silently changing the voice.\\n\\nUse `--voice <id>` for voiceover and voice changing, and repeat `--line \\"<id>:Text\\"` for dialogue. These examples show both ID forms; replace the sample IDs with the user\'s supplied IDs:\\n\\n```bash\\nvideodraft generate voiceover \\"Hello there.\\" --voice kPzsL2i3teMYv0FxEYQ6 --download ./media/voiceover.mp3\\nvideodraft generate dialogue --line \\"kPzsL2i3teMYv0FxEYQ6:Hello.\\" --line \\"elevenlabs-kmSVBPu7loj4ayNinwWM:Welcome back.\\" --download ./media/dialogue.mp3\\nvideodraft generate voice-changer ./media/source.wav --voice elevenlabs-kPzsL2i3teMYv0FxEYQ6 --duration 12 --download ./media/changed-voice.mp3\\n```\\n\\nFor MCP, use `generate_voiceover.voice_id`, `generate_dialogue.lines[].voice_id`, or `change_voice.voice_id` with either form. ElevenLabs IDs are separate from Kling video-control voice IDs and MiniMax `custom-*` cloned-voice IDs. Do not convert IDs from those systems into ElevenLabs IDs.\\n\\n### Avatar / talking head\\n\\n**First decide the framing, not just \\"someone talks\\".** This lane animates a PORTRAIT facing the lens. A character delivering a line inside a real scene, with blocking, framing or camera movement, belongs to `generate video` instead: write the dialogue in the prompt and `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` or `kling-o3` voices it natively (add a Seedance `--ref-audio` clip or a Kling voice bound per element for a specific voice). That is the default for any talking character; never plan silent video + TTS + lip-sync. Sending a cinematic shot to Fabric returns a head-on talking headshot, not the shot that was asked for. `videodraft models video --json` lists these under `recommended.in_scene_dialogue`.\\n\\nFor a presenter, spokesperson or explainer speaking to camera, VEED Fabric is the preferred model. Choose the dedicated path from the media the user already has:\\n\\n| Starting media | Command | Use |\\n| ------------------------------------------ | ------------------------------------------------------------------------- | ---------------------------------------------------------- |\\n| Portrait + script, reusable avatar record | `videodraft avatar create <portrait> --script \\"...\\"` then `avatar render` | Managed avatar flow with bundled speech preparation |\\n| Portrait + text | `videodraft avatar fabric <portrait> --text \\"...\\"` | One-off direct VEED Fabric text mode |\\n| Portrait + existing audio | `videodraft avatar fabric <portrait> --audio <audio>` | One-off direct VEED Fabric audio lip sync |\\n| Portrait + existing audio, short and sharp | `videodraft avatar h3-lipsync <portrait> --audio <audio>` | MiniMax H3 Max Lip Sync, 5-14.8s at up to 2K |\\n| Existing video + existing audio | `videodraft avatar lipsync <video> --audio <audio>` | Sync Labs Lipsync 2 (Tier 2): re-dub existing footage only |\\n\\nThe managed renderer is VEED Fabric Fast (`veed/fabric-1.0/fast`). Direct Fabric, MiniMax H3 Max Lip Sync, and Sync Labs are paid AI Studio generations and return async job IDs.\\n\\n1. Obtain the avatar image. Prefer the user\'s supplied portrait or an existing character. If none exists, use the user\'s explicitly requested compatible image model, otherwise generate a front-facing head-and-shoulders portrait with `nano-banana-2`, direct eye contact, a natural expression, and a clean background. Match the intended video aspect ratio when practical.\\n2. If the portrait is visibly soft or too small, run Topaz image enhancement/upscaling before animation.\\n3. Generate a script only if needed: `videodraft avatar script \\"<idea>\\"`.\\n4. Create the avatar record and speech: `videodraft avatar create <portrait-url-or-file> --script \\"...\\" --voice <id> --ar 9:16`. Prefer ElevenLabs when unspecified, but honor another explicitly selected supported voice/provider.\\n5. Render with VEED Fabric: `videodraft avatar render <avatar_video_id> --resolution 720p`.\\n\\nThe portrait is passed as the avatar\'s character image, not as a generic video\'s start frame. Prefer rendering directly at 720p. Use 480p only when the user prioritizes lower cost. Avatar script generation and `avatar create` (including speech) are bundled/free. Confirm the Fabric render cost, plus portrait generation or upscaling when needed.\\n\\nDirect Fabric text/audio, H3 Max Lip Sync, and Sync Labs do not use the managed avatar record. The CLI uploads local portrait, video, and audio files automatically. `avatar fabric --speed fast` applies only to audio mode. `avatar h3-lipsync` takes no prompt: pick `--resolution 480P|768P|1080P|2K` (default 768P), an optional `--seed`, `--no-transcription` to sync without transcribing the audio (transcription is on by default), and `--safety-checker` only when the user asks for Fal\'s checker. The portrait\'s width / height must be 0.4-2.5, and audio shorter than 5 seconds is refused before any charge. Sync costs 5 credits per verified audio second; under Fal BYOK, `sync_mode` remains available but `temperature` and `active_speaker` are ignored by the provider.\\n\\n### Upscaling / enhancement\\n\\n- **Images**: Topaz via `videodraft upscale image <url-or-file> --scale 1x|2x|4x [--mode precision|generative|creative] [--model <name>]`. Modes: `generative` (default; Wonder 3.5, Topaz\'s recommended model for AI-generated images; `Redefine` takes `--prompt`, `--creativity 1-6`, `--texture 1-5`), `precision` (faithful and cheapest; Standard V2, High Fidelity V3, Low Resolution V2, CGI, Text Refine; use for clean real photos), `creative` (Bloom 2; artistic, `--prompt`, `--creativity 1-9`). Extra knobs: `--no-face-enhance`, `--face-strength`, `--sharpen`, `--denoise`, `--fix-compression`, `--format png`. Use 1x for light enhancement without enlargement, 2x as the general default, and 4x only when the source quality and target size justify it. The server must verify dimensions from a readable image of at most 50 MB; `--width` and `--height` are compatibility hints and cannot bypass a failed probe. The result is synchronous. Cost: 8 credits per started 24 MP of output in precision, per 8 MP (Wonder 3/3.5) or 4 MP in generative, per 2 MP in creative.\\n- **Videos**: Topaz via `videodraft upscale video <url-or-file> --resolution 720p|1080p|4k` (preferred) or `--scale 2x`, plus `--mode precision|generative|creative` and `--model <name>`. `generative` (default; Starlight Precise 2.6, Topaz\'s recommended model for AI-generated footage; Starlight Fast 2 at half price) costs 12 credits/s up to 1080p and 26 at 4K. `precision` (Proteus, Proteus Natural, Iris, Gaia 2 for animation, Rhea, Theia, Artemis, Dione) is 6x cheaper at 1 / 2 / 6 credits per second for 720p / 1080p / 4K output; use it for real footage or when cost matters. `creative` (Astra 2, `--prompt`, `--creativity`, `--realism`, `--sharp`) always renders 4K at 50 credits/s. `--fps 60` delivers 60fps on the same pass; any output above 30fps (including a 50/60fps source) doubles every rate; the server must verify duration, dimensions, and frame rate from an MP4/MOV source of at most 100 MB. `--duration`, `--width`, `--height`, and `--source-fps` are compatibility hints and cannot override billing or bypass a failed probe. The same source-verification requirement applies to frame interpolation. `--scale` also accepts intermediate factors such as `1.5x`. Max source length 5 minutes. The job is asynchronous; the CLI waits by default, while MCP callers poll `check_generation_status`. MCP video input must be VideoDraft-hosted, so upload local or external sources first.\\n- **Frame interpolation / slow motion**: `videodraft interpolate <url-or-file> --fps 60 [--model Apollo|Chronos|Aion] [--slowdown 1-8]` (MCP `interpolate_video`). Resolution is unchanged. Apollo (default) for smooth general conversion, Chronos for natural slow motion, Aion for extreme slow motion. Apollo/Chronos cost 3 credits per output second up to 1080p (6 at 4K); Aion 5 / 17. These rates cover targets up to 60fps; above 60fps multiply by target FPS / 60 (120fps doubles the rate). Output seconds = source seconds \xD7 slowdown. The final charge rounds up to a whole credit.\\n- Use upscaling to preserve the image/video while improving detail, resolution, or cleanup. It cannot fix the wrong subject, misspelled text, bad framing, unwanted objects, broken continuity, or incorrect motion. Use an edit or regeneration for those problems. Keep `generative` for AI-generated sources; switch to `precision` for clean real photos/footage or a cheap pass, and `creative` only when the user wants an artistic reinterpretation.\\n- For a new Fabric avatar, render directly at 720p instead of rendering at 480p and then upscaling. Upscale the source portrait first only when the portrait itself is low quality.\\n\\n## Capability gotchas\\n\\n- Each model\'s `inputs` block is authoritative: supported `aspect_ratios`, `resolutions`, `quality_options`, `start_frame`/`end_frame`, `max_reference_images/videos/audio`, `multi_prompt`, `audio_toggle`. Passing an unsupported input fails with a clear error \u2014 check first, don\'t trial-and-error paid calls.\\n- Most video models support only 16:9 / 9:16 / 1:1. A 3:4 request hard-fails on most.\\n- `--seed` reproduces a specific output on models that support it (e.g. Flux, Ideogram V4). GPT Image 2.5 rejects any explicit seed; older models may ignore unsupported seeds. You do not need a seed for variation \u2014 `--num` already varies.\\n- `--rendering-speed` applies to Ideogram (V3: `Default`/`Turbo`/`Quality`; V4: `Turbo`/`Balanced`/`Quality`) and affects image cost \u2014 pass it to `videodraft costs ... --rendering-speed <tier>` for an accurate estimate. Always trust `videodraft models image --json` over this list; new models and tiers appear there the moment the platform ships them, with no CLI update.\\n- `seedream-v5-pro` supports unified text-to-image and reference-image editing with up to 10 image references. Use `--resolution 1K` for 7 credits/image or `--resolution 2K` for 14 credits/image.\\n- Reference inputs: `--ref <img>` (images, including up to 10 total for Gemini Omni 1.1 Flash and Wan 3.0, and 7 for Grok 1.5), `--source-video <v>` (Gemini uploaded edit/extension source), `--ref-video <v>` (up to 3 creative videos <=3s each for Gemini Omni 1.1 Flash; also Wan 3.0, MiniMax H3, MiniMax H3 Max, and Seedance 2), `--ref-audio <a>` (Wan 3.0, MiniMax H3, MiniMax H3 Max, Seedance 2), and `--element \'<json>\'` or `--element @elements.json` for Kling V3/O3. For an exact Seedance 2.x reference-video `--estimate`, add `--ref-video-seconds <combined-seconds>`; Seedance bills input seconds alongside output. Wan 3.0 and MiniMax H3 do not bill input reference seconds; MiniMax H3 Max bills them as pooled reference tokens. The CLI uploads local media references and every structured element without flattening the source/reference roles. `--segment \\"<prompt>:<seconds>\\"` (repeatable) drives Kling 3.0, Kling 3.0 Turbo, and O3 multi-prompt generation. Use 1-6 segments of 1-15 whole seconds each, with 3-15 seconds total. `generate image --video-ref` is the nano-banana-2 video reference.\\n- The top-level prompt is OPTIONAL for Gemini Omni 1.1 Flash media-input or continuation calls, Wan 3.0 frame/reference modes, `generate video` with multi-prompt models, and Kling 3.0 Turbo (`--model kling-v3-turbo`) image-to-video. Text-only Gemini and Wan calls still require a prompt unless their live schema says otherwise.\\n- Hosted AI Production fallback: `videodraft produce <project> --mode full_video` generates one Seedance 2 video per scene, with real people allowed by default at the higher Fal-tier rate on every submitted scene segment; add `--no-allow-real-people` for the lower Byteplus-only rate when no scene shows a real identifiable person. The MCP equivalent is `produce_project` with `mode: \\"full_video\\"` (and `allow_real_people: false` to opt out). If a partial opted-out run returns `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, rerun the same project once with an explicit `--allow-real-people` after cost confirmation. The server reconciles asynchronous results first, preserves running/completed jobs, and retries only failed placeholders carrying that exact code. Poll with `videodraft generations`, then `videodraft finalize <project>` swaps them into the hosted timeline before `export`. In VideoDraft ADE, do not choose this path while `videodraft_editor` is available unless the user explicitly requests hosted production. Generate or download the scene assets, import them, and assemble/export with the native editor instead. If the user explicitly requests another compatible video model for a hosted production, do not use this fixed Seedance path; generate the project shots manually with the requested model and attach them to the hosted timeline.\\n\\n## Cost model\\n\\n- Images: per image (\xD7 `--num`). Matrix-priced models (GPT-Image, Nano Banana Pro, Seedream v5 Pro) vary by resolution/quality.\\n- Video: usually credits/second \xD7 duration; rate depends on model + resolution + quality + native audio on/off.\\n- Gemini Omni 1.1 Flash: 3 / 10 / 15 / 30 credits per output second at 360p / 720p / 1080p / 4K (720p default). Extend appends an explicit 3-10 seconds to a 1-30s source, whether uploaded or resolved from a prior interaction; 40 seconds total is reachable only while the source stays at or under 30s. Fal BYOK generation and basic source edits cost zero VideoDraft credits; continuation, extension, and separate creative references on a source edit are unavailable under Fal BYOK.\\n- MiniMax H3: 5 / 6 / 13 / 16 credits per output second at 480p / 768p / 2K / 4K (768p default). The first 5 reference images are included, then 8 credits for each additional image. Reference video and reference audio are NOT billed.\\n- MiniMax H3 Max: 5 credits per output second at 480p or 8 credits per second at 768p (default), plus pooled reference tokens. The first 4,096 reference tokens are free and each additional 1,000 costs 2 credits. An image is `(width x height) / 1024` tokens (a 1024x1024 reference is 1,024, so four of them are free); a reference video is 2,886 tokens per second at 480p or 7,459 at 768p; reference audio is about 2,121 per second. Pass `--ref-audio-seconds` for an exact estimate with audio references. The server measures each reference image, so a quote that assumes 1024x1024 is a floor.\\n- Wan 3.0: 7 / 14 / 28 credits per output second at 480p / 720p / 1080p. Auto duration reserves 30 seconds and reconciles unused credits from the provider-reported output duration. Fal BYOK charges zero VideoDraft credits.\\n- Grok Imagine Video 1.5: 8 credits/output second at 480p, 14 at 720p, or 25 at 1080p, plus 1 credit for each first-frame or reference image. Native generated audio is part of every output.\\n- Kling voice control: 17 credits/output second for Kling 2.6 Pro, 16 at Kling V3 Standard, and 20 at Kling V3 Pro.\\n- Shot-image batches: one image per shot (+1 grid image per scene in `--grid` mode) \u2014 the largest single spend in the pipeline.\\n- VEED Fabric avatar renders: ~10 credits/sec at 480p, ~20/sec at 720p. Avatar creation and its speech are bundled/free; only optional portrait generation/upscaling adds cost before the render.\\n- Direct VEED Fabric: text or normal audio is 8 credits/sec at 480p and 15/sec at 720p; fast audio is 10/sec at 480p and 20/sec at 720p.\\n- MiniMax H3 Max Lip Sync: 5 / 8 / 16 / 32 credits per output second at 480P / 768P (default) / 1080P / 2K. The video runs as long as the audio (at least 5s; only the first 14.8s is used), billed on the server-measured length rounded up, so at most 15 seconds. Use MP3, WAV, M4A (AAC), or AAC audio; other formats run only on the user\'s own Fal key.\\n- Sync Labs Lipsync 2: 5 credits per verified audio second.\\n- Voiceover TTS: 10 credits per 1000 characters for standard voices, 30 per 1000 for cloned `custom-*` voices (min 1, pro-rated); applies to standalone voiceovers AND per-scene narration during `produce`. Silent tracks are free. Voice cloning itself is a flat 150 credits per clone.\\n- Lyria music: flat per track, 4 credits (Clip) / 10 credits (3.5) / 8 credits (legacy Pro). Fal BYOK uses zero VideoDraft credits.\\n- Seed Audio 1.0: 19 credits per actual output minute, prorated and rounded up to a whole credit. VideoDraft reserves the 120-second maximum of 38 credits and refunds the unused portion after generation. Fal BYOK is free.\\n- ElevenLabs audio: sound effects are per second, dialogue is per character, music/voice-changer/dubbing are per started minute. ElevenLabs Music v2.5 and v1 both cost 60 credits per started output minute; estimate a composition plan with its total length. Voice changer and dubbing reject source media above 300s in the current synchronous flow.\\n- Seedance 2.0 / 2.5 real people: allowed by default, which keeps Byteplus first, permits a submit-time Fal fallback, and prices at Fal\'s rate for that tier: 2.0 Mini 8/16 cr/s, Fast 11/25, Standard 14/31/69/156, 2.5 23/48/114 for 480p/720p/1080p. `--no-allow-real-people` uses the Byteplus-priced path instead (2.0 Mini 4/8, Fast 6/13, Standard 7/16/38/78, 2.5 11/24/57); Byteplus refuses real-person likenesses, so a likeness-policy failure then does not fall back. The gap is roughly 2x but not exactly: the Seedance 2.0 1080p pair is 38/69, or about 1.82x. If Byteplus accepts the task and later rejects the generated output, VideoDraft refunds the failure but does not resubmit it to Fal. Turn the option off only for the cheapest run when nothing in the job is a real identifiable person.\\n- Grok Imagine images: `grok-imagine` is a flat 2 cr (3 with a reference). `grok-imagine-2.0` is a separate, newer model priced by resolution and quality: 1K 4 (low) / 6 (medium), 2K 6 / 8, plus 1 cr per reference image (up to 3). v1 is NOT superseded \u2014 pick it when cost matters more than 2K.\\n- xAI bills refused requests, so failed Grok generations are not refunded.\\n- Upscales: priced by output size, mode, and (video) duration/fps; see the Upscaling section above.\\n\\nQuote before spending:\\n\\n```bash\\nvideodraft costs gemini-omni-1.1-flash --type video --duration 8 --resolution 720p --audio\\nvideodraft costs topaz-upscale-video --type video --duration 10 --width 1920 --height 1080 --resolution 4k --mode generative\\nvideodraft costs topaz-upscale --type image --width 2048 --height 2048 --scale 2x\\nvideodraft costs topaz-interpolate-video --type video --duration 10 --width 1920 --height 1080 --fps 60 --topaz-model Apollo\\nvideodraft costs minimax-h3 --type video --duration 10 --resolution 2K --ref-images 7\\nvideodraft costs minimax-h3-max --type video --duration 10 --resolution 768p --ref-images 2\\nvideodraft costs grok-imagine-video-1.5 --type video --duration 8 --resolution 720p --ref-images 4\\nvideodraft costs seedance-2 --type video --duration 15 --resolution 720p --quality standard --audio\\nvideodraft costs grok-imagine-2.0 --type image --resolution 2K --quality medium --num 2\\nvideodraft costs seedance-2 --type video --duration 15 --resolution 720p --quality standard --audio --no-allow-real-people # lower Byteplus-only rate\\nvideodraft costs elevenlabs-dubbing --type audio --duration 60\\nvideodraft costs seed-audio-1.0 --type audio --duration 60 # scenario only; model controls actual length\\nvideodraft costs elevenlabs-dialogue --type audio --chars 350\\nvideodraft costs voiceover --type audio --chars 800 # TTS: 10 cr / 1000 chars\\nvideodraft generate video \\"...\\" --model gemini-omni-1.1-flash --estimate # same quote, inline\\nvideodraft generate video \\"...\\" --model minimax-h3-max --duration 8 --resolution 768p --prompt-expansion-mode balanced\\n```\\n","references/pipeline.md":"# VideoDraft pipeline reference\\n\\nEverything here describes the hosted fallback pipeline through the CLI (`videodraft <command>` / `videodraft call <tool>`) or hosted MCP connector (tool names in backticks). When the local `videodraft_editor` MCP is available, do not use hosted production or export by default. Use hosted tools only for asset generation and optional script/storyboard stages, then import the results and finish with the native editor reference linked from SKILL.md. Continue through `produce_project` and `export_video` only when the user explicitly requests a hosted web production or the native editor is unavailable.\\n\\nUse direct asset tools for standalone images, clips, audio, upscales, and descriptions. Use a hosted project when the user explicitly wants the editable web project, when a hosted storyboard stage is useful, or when the native editor is unavailable. Script-only uses a script-stage project and stops at the script. In VideoDraft ADE with editor tools present, stop before hosted production, import the generated assets, and build/export the native project.\\n\\n## Stages and their tools\\n\\n| Stage | CLI | Underlying tool |\\n| ---------------------------------------- | --------------------------------------------------------------------------- | ------------------------------------------- |\\n| Idea \u2192 full storyboard project | `videodraft create \\"<idea>\\"` | `generate_storyboard_from_idea` |\\n| Idea \u2192 script only (stop there) | `videodraft create \\"<idea>\\" --script-only` | `generate_script_from_idea` |\\n| Footage IS the video | `videodraft call generate_storyboard_from_media` | `generate_storyboard_from_media` |\\n| Batch shot images | `videodraft shots <project>` | `generate_shot_images` |\\n| One shot image | `videodraft generate image --project <id> --scene N --shot M` | `generate_image` |\\n| Produce (voiceover, captions, timeline) | `videodraft produce <project>` | `produce_project` |\\n| Seedance full-video production | `videodraft produce <project> --mode full_video` | `produce_project` with `mode: \\"full_video\\"` |\\n| Per-shot motion prompts | `videodraft video-prompts <project>` | `generate_video_prompts` |\\n| Motion clip for a shot | `videodraft generate video --project <id>` | `generate_video` |\\n| Attach a finished clip to the timeline | `videodraft attach <project> --scene N --shot M --media <url> --type video` | `attach_media_to_shot` |\\n| Find free b-roll (no credits) | `videodraft stock search \\"<query>\\"` | `search_stock_media` |\\n| Copy stock media onto the CDN | `videodraft stock import <ref>` | `import_stock_media` |\\n| Background music | `videodraft generate music --attach <project>` | `generate_music` / `set_background_music` |\\n| Song with vocals, lyrics or sections | `videodraft generate music --model elevenlabs-music-v2.5 --section \\"...\\"` | `generate_music` with `composition_plan` |\\n| General or reference-driven audio | `videodraft generate audio \\"...\\"` | `generate_audio` |\\n| Sound effect | `videodraft generate sound-effect \\"...\\"` | `generate_sound_effect` |\\n| Dialogue audio | `videodraft generate dialogue --line \\"voice:text\\"` | `generate_dialogue` |\\n| Voice changer | `videodraft generate voice-changer <audio>` | `change_voice` |\\n| Dubbing | `videodraft generate dub <audio_or_video>` | `dub_media` |\\n| Scene voiceover | `videodraft generate voiceover --project <id> --scene N` | `generate_voiceover` |\\n| Avatar script | `videodraft avatar script \\"<idea>\\"` | `generate_avatar_script` |\\n| Avatar + speech | `videodraft avatar create <portrait> --script \\"...\\"` | `create_avatar_video` |\\n| Talking-head render | `videodraft avatar render <avatar_video_id>` | `render_avatar_video` + `get_avatar_video` |\\n| Direct portrait + text/audio | `videodraft avatar fabric <portrait> --text \\"...\\"` or `--audio <file>` | `generate_veed_fabric_video` |\\n| Short portrait clip from audio, up to 2K | `videodraft avatar h3-lipsync <portrait> --audio <file>` | `generate_minimax_h3_lipsync_video` |\\n| Existing video + replacement audio | `videodraft avatar lipsync <video> --audio <file>` | `generate_sync_lipsync_video` |\\n| Existing-video AI edit | `videodraft edit video <video> \\"<change>\\" --model <video-edit-model>` | `edit_video` |\\n| Motion transfer | `videodraft edit motion <image> [direction] --motion-video <video>` | `generate_motion_control_video` |\\n| Image enhancement/upscale | `videodraft upscale image <image>` | `upscale_image` |\\n| Video enhancement/upscale | `videodraft upscale video <video>` | `upscale_video` |\\n| Frame rate / slow motion | `videodraft interpolate <video> --fps 60 [--slowdown 4]` | `interpolate_video` |\\n| Final MP4 | `videodraft export <project>` | `export_video` + `check_export_status` |\\n\\n## Rules that prevent broken results\\n\\n- **The storyboard is generated FROM the script**, never from the raw idea. `videodraft create` runs the whole chain correctly. Don\'t call `generate_storyboard_scenes` with a raw idea as the \\"script\\".\\n- **Visual consistency**: never generate a storyboard shot in isolation. Shot prompts carry `[[asset:Name]]` / `[[shot:X-Y]]` tags that `generate_shot_images` resolves against the project\'s visual assets and prior shots. When generating a single shot whose prompt has no tags, pass `--ref` images yourself (the project\'s visual assets and/or the previous shot\'s image; `projects get` exposes both). For scenes with multiple shots or recurring characters, prefer `videodraft shots <project> --model <selected-image-model> --grid`: preserve an explicitly requested compatible image model, otherwise use `nano-banana-2`. It creates one coherent scene grid, then decodes it into individual shot images.\\n- **Reference-first video**: when identity, styling, or composition matters, do not generate each motion clip from text alone. Generate or select the shot still first, then pass the decoded shot image as `--start-image` or `--ref` to the selected video model. AI Production already composes scene grids and sends them to Seedance as references. If the user explicitly requests another compatible video model, bypass fixed Seedance full-video mode and generate the per-shot clips with the requested model, using the individual decoded shot images as anchors.\\n- **Seedance full-video real people**: hosted `full_video` allows real people by default, which applies Fal-tier pricing to every submitted segment and permits the Byteplus-to-Fal fallback. For the lower Byteplus-only rate when no scene grid shows a real identifiable person (non-people, anime, clearly synthetic or stylized characters), use `videodraft produce <project> --mode full_video --no-allow-real-people`, or MCP `produce_project` with `mode: \\"full_video\\", allow_real_people: false`. If a partial opted-out run returns `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, re-estimate, follow the user\'s spend-confirmation preference, and rerun the same project once with an explicit `--allow-real-people` / `allow_real_people: true`. The server reconciles asynchronous results first, preserves running/completed jobs, and retries only failed placeholders carrying that exact code. Do not loop when it was already on. VideoDraft refunds a Byteplus task rejected after asynchronous acceptance, but cannot reroute it; rephrase or change the scene references instead.\\n- **Hold off generating shot images while the user is still iterating** on storyboard structure.\\n- **produce \u2192 export ordering**: `export` requires a produced project where every production scene has timeline media. If `produce` returns `generating_shot_images`, poll the job ids it returns, then re-run produce.\\n- **Do not attach motion clips before production exists**: run `produce` successfully first, then attach finished motion clips to the production timeline. Attaching before `production_data` exists cannot place them in the final timeline.\\n- **Generated motion clips do not auto-attach**: after `generate video` completes, attach the clip with `attach_media_to_shot` (`media_type:\\"video\\"`, include `duration_seconds`) \u2014 it replaces the production timeline clip while keeping the storyboard still.\\n- **Talking heads use dedicated avatar tools**: do not use `generate video`. Use managed `avatar create` and `avatar render` for reusable avatars, direct `avatar fabric` for a portrait plus text/audio, `avatar h3-lipsync` for a short (5-14.8s) portrait clip from existing audio at up to 2K, and `avatar lipsync` for an existing video plus replacement audio. Reuse a supplied person image or generate a clear front-facing portrait with the explicitly requested compatible image model, otherwise Nano Banana 2. Managed avatar creation and speech are bundled/free; direct Fabric, H3 Max Lip Sync, Sync, and render are paid.\\n- **Existing-video edits use their own category**: call `edit_video` or `videodraft edit video` with a `video_edit` model when transforming the source itself. Kling O3 also has a reference-generation mode that creates a new guided clip. Wan 3.0 is a text/frame/reference generation model, not a source-video editor. Motion transfer similarly uses `generate_motion_control_video` or `videodraft edit motion` with a `motion_control` model.\\n- **Upscaling preserves rather than redesigns**: use Topaz when resolution, detail, or cleanup is the problem (generative mode by default: Wonder 3.5 / Starlight Precise 2.6, Topaz\'s pick for AI sources; precision for real photos/footage or the cheapest pass; creative only for an artistic reinterpretation; `interpolate` for frame rate or slow motion). Regenerate or edit when the subject, text, framing, continuity, or motion is wrong. Upscale a low-quality avatar portrait before Fabric; do not render a new avatar at 480p just to upscale the result.\\n- **Timeouts on the one-shot create**: if `create` times out at the transport layer, the project was still created server-side \u2014 `videodraft projects list`, take the most recent, and resume with its id. Don\'t start a duplicate.\\n\\n## User-attached media: classify roles first\\n\\nFor EACH attached file decide:\\n\\n- **visual_asset** \u2014 recurring reference (character / product / location / style). Upload, then pass in `visual_assets` of `generate_storyboard_from_idea` (via `videodraft call`), or add to an existing project with `add_visual_assets`. Type must be one of `character | object | location | style | custom` with a short name + concrete description.\\n- **shot** \u2014 the media IS footage for the video. Whole video = footage \u2192 `generate_storyboard_from_media`. Idea + footage \u2192 `generate_storyboard_from_idea` with `shot_media`. Existing storyboard \u2192 `attach_media_to_shots`.\\n- **reference** \u2014 inspiration only \u2192 fold a description into the idea/instructions; don\'t place it as a shot or asset.\\n\\nAmbiguous (e.g. a person holding a product)? Ask the user.\\n\\nUploads persist in the media library \u2014 recall later with `videodraft media list`.\\n\\n## Editing project data safely\\n\\n1. `videodraft call get_project_schema` \u2014 read the structure once per session.\\n2. `videodraft projects get <id> --raw` \u2014 the exact editable blob.\\n3. Modify; then `videodraft call update_project --stdin` with `{\\"project_id\\": \\"...\\", \\"data\\": {...}}`.\\n - Objects deep-merge key-by-key; **arrays replace wholesale** \u2014 send the complete array you\'re changing (e.g. all of `storyboard.scenes`).\\n - Scene shot arrays (`image_prompt` / `shot_types` / `shot_actions` / `search_prompt` / `preview_media`) are auto-aligned; fix-ups come back as warnings.\\n4. Snapshot before risky edits: `videodraft checkpoint create <id> --name \\"before re-script\\"`. Restore with `videodraft checkpoint restore <id> <version>`.\\n\\n## AI Studio sessions (standalone generations)\\n\\nProject generations group automatically, and standalone work is grouped per conversation (MCP) or per working directory (CLI) by the connection session \u2014 see the \\"AI Studio sessions\\" section of SKILL.md. Once the task is clear, give that automatic session a concise 3-6 word title before the first generation:\\n\\n```bash\\nvideodraft sessions name \\"Fox Brand Explorations\\"\\nvideodraft generate image \\"...\\"\\n```\\n\\nNaming creates the session with that title. If generation, a user, or an earlier agent created it first, the existing name is preserved. The command refuses to run while `VIDEODRAFT_SESSION` is set, because that override would send later generations to a different session. Use `sessions create` plus `--session` only to continue or deliberately create a separate group.\\n"}') {
7611
- return JSON.parse('{"SKILL.md":"---\\nname: videodraft\\ndescription: Create and edit AI videos, images, 3D meshes, Seed Audio, voiceovers, music, sound effects, dialogue, dubbing, storyboards, avatar videos, media upscales, and product/ad videos with VideoDraft. Use whenever the user mentions VideoDraft; asks to generate a video, image, 3D asset, audio asset, ad, explainer, storyboard, avatar, upscale, or batch/CI workflow; wants to download a video from YouTube, Instagram, TikTok, X or another site; or wants to assemble, cut, caption, mix, lay out, inspect, or export a native VideoDraft Editor timeline. Covers the cloud `videodraft` CLI/MCP and local headless `videodraft_editor` MCP. When the editor MCP is exposed, prefer it for production, timeline assembly, and export; use cloud production/export only when explicitly requested or the editor is unavailable.\\n---\\n\\n# VideoDraft\\n\\nVideoDraft is an AI video creation platform where asset generation is the priority lane:\\n\\n- **Asset generation**: standalone images, video clips, Seed Audio, voiceovers, music, sound effects, dialogue, voice-changed audio, dubbed media, upscales, and image descriptions. This is the fastest and most important lane. Treat these as complete deliverables when the user asks for assets.\\n- **Asset I/O**: upload local files, download outputs, auto-upload local references, and save generated media where the user can see it.\\n- **3D assets**: standalone Meshy 7 and Tripo H3.1 meshes, multi-view generation, complete artifact downloads, and Meshy humanoid rigging through MCP/CLI. Read [references/3d.md](references/3d.md), or `videodraft skills show 3d`. These assets have their own saved library and do not require an AI Studio session or web viewer.\\n- **Native editing**: local `.vdproject` timelines, cuts, layouts, captions, effects, audio, and exports through the headless VideoDraft Editor. Inside VideoDraft ADE, this is the default production and export lane whenever `videodraft_editor` is available.\\n- **Hosted project production**: idea \u2192 script \u2192 storyboard \u2192 hosted production timeline \u2192 exported MP4. Use the early stages for scripts, storyboards, and generated assets when useful. Treat hosted production and export as a fallback when the native editor is unavailable, or as an explicit destination when the user asks for an editable web project or hosted workflow.\\n\\n## How to connect\\n\\nCloud generation has two equivalent surfaces (same backend, credits, and hosted projects). Native timeline editing is a separate local surface:\\n\\n1. **CLI** (preferred when you have a shell): run `videodraft` if it\'s on PATH; otherwise `npx -y videodraft@latest` runs it with no install (needs Node \u226520; the `-y` skips npx\'s install prompt so it runs non-interactively; the package is fetched on first use and cached). For heavy use, `npm install -g videodraft`. If there\'s no Node/shell here but the MCP connector below is available, use that instead; if neither works, tell the user how to install (https://videodraft.ai/cli).\\n - Auth \u2014 pick by context, don\'t guess:\\n \u2022 INTERACTIVE (a human is in the session, e.g. Claude Code / Codex): on exit code 3 (\\"not authenticated\\"), tell the user to run `videodraft login` in their terminal \u2014 it opens their browser for a one-click VideoDraft sign-in (OAuth), no key to copy. Wait for them to confirm it succeeded, then retry the command. This is the preferred path when the user is present.\\n \u2022 HEADLESS / CI (no browser): set `VIDEODRAFT_API_KEY=vd_mcp_...` (a token the user mints at https://app.videodraft.ai/mcp-keys).\\n \u2022 SECURITY: never ask the user to paste a `vd_mcp_...` token into the chat \u2014 use browser `login` or the env var so the token never lands in the transcript.\\n - Every command accepts `--json` (parse this, don\'t scrape text). Exit codes: 0 ok, 1 error, 2 usage, 3 auth (see Auth above), 4 insufficient credits (\u2192 tell the user, don\'t retry).\\n - Tool discovery: start with `videodraft tools list` for the grouped catalog, then narrow with `videodraft tools list --lane assets`, `--lane asset_io`, `--lane project_data`, or `--lane production`.\\n - Asset lane: `videodraft generate ...`, `videodraft edit video|motion`, `videodraft avatar ...`, `videodraft upscale ...`, `videodraft interpolate ...`, `videodraft upload`, and `videodraft download`.\\n - Full API access: `videodraft tools schema <name>`, `videodraft call <tool> --args \'<json>\'`.\\n2. **MCP connector**: if VideoDraft MCP tools (e.g. `generate_storyboard_from_idea`) are available, call them directly \u2014 the CLI\'s curated commands map 1:1 onto these tools.\\n3. **Native editor MCP** (`videodraft_editor`): prefer this for project production, timeline assembly, cutting, layouts, transitions, captions, audio placement, and final export. Inside VideoDraft ADE on a supported Mac, Claude, Codex, OpenCode and Grok receive it automatically in both Code and VideoDraft modes. It runs headlessly, so an Open Editor click is not required. Start with `project_manage` (`operation: \\"project\\"`, action `list`, `open`, or `create`); standalone asset generation remains in the cloud CLI or MCP.\\n\\nSend native editor edits serially, batching everything a request needs into one `edit_apply` recipe. Pass the latest `context` when a person may be editing at the same time, so a stale write is refused. See [references/editor.md](references/editor.md) for project selection, media import, timing units, edit recipes, verification, export, and the `videodraft-editor` terminal bridge.\\n\\nIf you are reading this skill through `videodraft skills show skill`, run `videodraft skills show editor` before native editor work to load that reference.\\n\\n**VideoDraft ADE routing rule:** the presence of `videodraft_editor` means the native editor is ready, even when no editor window is visible. Use cloud tools to generate or source assets and, when helpful, scripts or storyboards. Do not call hosted `produce_project` / `videodraft produce` or `export_video` / `videodraft export` by default. Import the assets into the native project, assemble there, and export with native `delivery_manage` `submit`. Use hosted production/export only when the user explicitly asks for the web workflow or the native editor tools are unavailable. Do not silently fall back to hosted production after a native tool error.\\n\\n## First decision: asset, hosted project, or native edit?\\n\\n- **One standalone asset** (image, clip, voiceover, music track, sound effect, dialogue track, voice-changed file, dubbed media file, upscale, or description): generate it directly. Do NOT create a project.\\n - `videodraft generate image \\"a red fox in snow, cinematic\\" --ar 16:9 --download ./out/`\\n - `videodraft generate video \\"slow dolly over a misty lake\\" --model gemini-omni-1.1-flash --duration 6 --download ./out/`\\n- **Any final video, production timeline, existing footage, local `.vdproject`, or hands-on edit**: use `videodraft_editor` when available. List or open the intended local project, or create a native project for a new production. The editor can work without showing its UI.\\n- **A small set of related assets**: still stay in the asset lane. They are grouped automatically (see [AI Studio sessions](#ai-studio-sessions)); give the current group one useful task-specific name with `videodraft sessions name \\"<name>\\"`. Switch to a project only when the deliverable matches the project criteria below or the user asks to attach the assets to one.\\n- **A generated multi-scene video / ad / explainer**: when the editor is available, use hosted tools only for any needed script, storyboard, shot planning, or generated assets; stop before hosted production, import the assets, and build/export the native timeline. A hosted project is optional unless the user wants the web project or its storyboard workflow.\\n- **A hosted web project or hosted export**: use the hosted pipeline only when the user explicitly asks for it or the native editor is unavailable.\\n- **Just a script** (no video asked for): A script-only request creates a script-stage project but stops at the script. Use `videodraft create \\"...\\" --script-only`; do not build a storyboard the user didn\'t ask for.\\n- **Iterating on existing work**: identify the surface first. Use `project_manage` `project` with `action: \\"list\\"` for native projects and `videodraft projects list` only for hosted work. Never create a replacement project just to change an existing one.\\n\\n## Choose the model from the task\\n\\nIf the user names a model, use it when compatible. If it cannot handle the request, explain why and recommend alternatives instead of silently switching. Otherwise inspect the inputs, duration, audio, quality, speed, and cost, check the live catalog, and pass an explicit model.\\n\\n### Seedance 2.x real-person rule\\n\\nSeedance 2.0 and 2.5 allow real people by default, the same as AI Studio:\\n\\n- The default is true. Generation tries Byteplus first and can fall back to Fal, which allows real-person likenesses, so a person in the prompt, a start frame, end frame, reference image, or reference video does not hard-fail. The request is charged at Fal\'s higher tier-specific rate even if Byteplus serves it.\\n- Turn it off for the cheapest run, only when the user wants that and nothing in the job is a real identifiable person (text-only requests, non-people, anime, clearly synthetic or stylized characters). For MCP use `allow_real_people: false`. For the CLI use `--no-allow-real-people`. That pins the job to the lower-priced Byteplus path, and a Byteplus likeness-policy refusal does not fall back to Fal. Pass the same value to `get_model_costs` or `videodraft costs` so the estimate matches the charge.\\n- If a request made with the option off fails with code `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, estimate the higher rate, follow the user\'s spend-confirmation preference, and retry exactly once with the option on. The structured recovery fields are `retryable: true`, `retry_with: { allow_real_people: true }`, `cli_flag: \\"--allow-real-people\\"`, and `retry_policy: \\"once\\"`. CLI `--json` submit errors expose them under `details`; `status` and `wait` include them on the failed job result. Do not treat an unrelated moderation or provider error as that signal.\\n- Do not loop when the option was on. Byteplus can accept a task and reject its generated output later. VideoDraft refunds that failed generation, but the late asynchronous failure cannot be rerouted to Fal. Rephrase the prompt or use different references before trying again.\\n- For hosted AI Production, the same default applies to every Seedance scene segment: `produce_project` with `mode: \\"full_video\\"` and `videodraft produce <project> --mode full_video` allow real people unless you pass `allow_real_people: false` or `--no-allow-real-people`. If an earlier run with the option off partially submitted and returns the opt-in code, rerun that same project once with an explicit `allow_real_people: true` / `--allow-real-people`. The server reconciles asynchronous results first, preserves running/completed jobs, and resubmits only failed scene-video placeholders carrying the exact opt-in signal. Keep the native-first VideoDraft ADE routing rule above: hosted full-video production is still explicit/fallback-only when the local editor is available.\\n\\n**How to pick a model. Follow this order every time:**\\n\\n1. The user named a model: use it, never substitute.\\n2. Otherwise pick a **Tier 1** model.\\n3. Use a **Tier 2** model only when no Tier 1 model supports the request, and say why. Never pick Tier 2 on your own for price, speed, novelty or variety.\\n\\nTier 1: images `nano-banana-2`, `nano-banana-pro`, `nano-banana-2-lite`, `gpt-image-2.5-flare`, `gpt-image-2.5-sunburst`, and `recraft-v4` for vector/SVG only; videos `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0`, `kling-o3`, `minimax-h3-max`; video edits `gemini-omni-1.1-flash`; talking-head portraits `veed-fabric`, `minimax-h3-max-lipsync`; motion transfer `kling-v3-motion-control`; audio ElevenLabs for speech tools, Lyria 3.5 for music at any length.\\n\\nTier 2 is every other catalog model (`wan-3.0`, `flux-3`, `minimax-h3`, `kling-v3-turbo`, `kling-2.6-pro`, `grok-imagine-video-1.5`, `happy-horse`, Veo 3.1, Sora, Seedream, Grok Imagine images, FLUX images, `sync-lipsync-2` and the rest). All of them stay available by name. Real Tier 2-only jobs: document or web-page references (`wan-3.0`), keyframes pinned to timestamps (`flux-3`), re-syncing existing footage (`sync-lipsync-2`).\\n\\nEvery `videodraft models image|video|audio --json` entry carries `tier` (1 or 2), and the top-level `recommended` array is Tier 1, best first. That catalog beats this page when they disagree.\\n\\n**Speech in video:** write the dialogue in the video prompt. `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` and `kling-o3` speak it natively. For a specific voice add a Seedance reference audio clip (`--ref-audio`) or a Kling bound voice. Do not plan silent video + TTS + lip-sync. Lip-sync tools are only for a portrait presenter talking to camera, or for re-syncing footage that already exists. Off-screen narration and voiceover are unaffected: keep using `generate voiceover` for those.\\n\\n**Images:**\\n\\n- `nano-banana-2`: general default, editing, consistency, and references.\\n- `nano-banana-pro`: maximum quality. `nano-banana-2-lite`: fast, inexpensive drafts.\\n- `gpt-image-2.5-flare`: replaces the GPT Image 2 recommendation for posters, logos, signs, title cards, readable text and composition. Use `gpt-image-2.5-sunburst` for extra precision/detail. Both use the same basic options as GPT Image 2: references (up to 16), aspect ratio, 1K/2K/4K resolution, image count (1-4), and quality. Quality adds xhigh/max; auto is charged at the max tier. Other model recommendations are unchanged; explicit `gpt-image-2` requests still use GPT Image 2.\\n\\n**Videos:**\\n\\n- `gemini-omni-1.1-flash`: general default for 3-10s generation, image-to-video, uploaded source edits up to 10s, and silent or voiceover-backed shots (audio is always on, so describe quiet ambience with no speech and mute or mix it on the timeline; only a file with no audio track at all needs `kling-3.0`, `kling-o3` or Seedance with `--no-audio`). It supports first and last frames, up to 10 total image inputs, up to 3 creative reference videos of at most 3 seconds each on `--video-task generate` ONLY, uploaded-video extension, and continuation of an earlier generation through `--previous-interaction-id`. Output is 360p/720p/1080p/4K at 3/10/15/30 cr/s with audio always on. Use `--source-video` for the uploaded edit/extension source; extension sources must be 1-30s. Edit and extend accept EXACTLY ONE input video, so `--ref-video` cannot accompany `--source-video` or `--previous-interaction-id`; only `--ref` images can. One `--ref-video` with no separate source remains a legacy source edit. A previous interaction defaults to conversational edit; add `--video-task extend` or `--extend` with an explicit 3-10s duration to append at the end. A continuation resolves to the prior turn\'s output and is submitted as an ordinary source, so the same limits apply to it: at most 30s to extend, at most 10s to edit. 40 seconds total is reachable, but only by extending from a source of 30s or less, so a ladder dead-ends once it passes 30s. New dialogue can only be added when the SOURCE video is silent; adding speech on top of a source that already contains speech is refused with \\"the model is currently unable to process speech edits\\". The server safely measures creative-reference durations, or you can repeat `--ref-video-duration` when a host blocks metadata probing. Fal BYOK supports its currently callable v1.1 generation and basic-edit endpoints at zero VideoDraft credits, but Fal does not expose continuation/extension or mixed source-edit references as callable endpoints and VideoDraft must never fall back to paid Google.\\n- `grok-imagine-video-1.5` (Tier 2): 1-15s text, first-frame, or 1-7 reference-image generation with native audio. Text/first-frame modes support 480p, 720p, and 1080p; reference mode supports 480p/720p. Cite references as `<IMAGE_0>` through `<IMAGE_6>`. It has no last frame, seed, negative prompt, quality tier, reference video, or reference audio.\\n- `minimax-h3` (Tier 2): 480p/768p/2K/4K (5/6/13/16 cr/s, 768p default), native stereo audio, and 5-15s text, first/last-frame, or mixed-reference generation. Reference mode accepts up to 9 images, 3 videos, and 3 audio clips, with at most 12 files total. Cite them as `Image 1`, `Video 1`, and `Audio 1` in array order.\\n- `minimax-h3-max`: 480p/768p pricing (5/8 cr/s, 768p default), native audio, and 5-15s text, first/last-frame, or mixed-reference generation. Reference mode accepts up to 9 images, 3 videos, and 3 audio clips, with at most 12 files total, cited as `Image 1`, `Video 1`, and `Audio 1` in array order. It supports a reproducibility seed and `disabled` / `balanced` / `quality` prompt expansion.\\n- `wan-3.0` (Tier 2): unified 2-30s text, first/last-frame, or ordered mixed-reference generation at 480p/720p/1080p (7/14/28 cr/s), with optional native audio. Reference mode accepts up to 10 images, 5 videos, and 5 audio clips, at most 20 media files total; video and audio each total at most 15 seconds. `--auto-duration` reserves 30 seconds and reconciles to the provider-reported output length. Document/web references use `--file-url` or `--web-url` and require `--thinking`.\\n- `flux-3` (Tier 2): Black Forest Labs FLUX 3. 5-20s at 720p/1080p with 24fps native audio, from a prompt, a first frame, first + last frames, or up to 10 keyframes pinned to specific moments (`--keyframe shot.png@2.5`, repeatable). `--quality draft` renders the same shot at 720p for roughly a third of the cost. Auto duration is text/first-frame only.\\n- `seedance-2`: 11-15s, video/audio/mixed references, wider ratios, selectable audio, or first/last frames. `mini` is the economical default; move to `standard` when the user asks for high quality, when a Mini result is not good enough, or for 1080p/4K; `fast` for speed.\\n- `seedance-2.5`: 4-30s single takes and up to 50 references (30 image, 10 video, 10 audio). Same modes as 2.0, one quality tier, 480p/720p/1080p. Reach for it when a shot must run past 15s or carry more references than 2.0 allows.\\n- `kling-o3`: reference images plus structured image/video elements, first/last frames, multi-prompt, audio control, or 4K. O3 allows 7 combined image references and image-backed elements, reduced to 4 combined items when a video-backed element is present. `kling-3.0`: image-to-video can use structured image/video elements and bind a custom Kling voice ID to either element form. Tier 2, by name: `kling-v3-turbo` (fast polished 3-15s with first frame, multi-prompt, and audio, but no elements) and Kling 2.6 Pro (top-level voice IDs cited as `<<<voice_1>>>` and `<<<voice_2>>>`).\\n- Existing-video edits use `videodraft edit video`, not generic generation. `gemini-omni-1.1-flash` is the preferred edit model and is chosen automatically when you omit `--model`: source up to 10s, up to 10 reference images, 360p/720p/1080p/4K with audio. Creative `--ref-video` inputs are NOT accepted on an edit, because an edit takes exactly one input video; use `generate video --video-task generate` to guide a new clip with video references instead. Omitting `--model` on a longer or unmeasurable source spends nothing and prints a priced menu (what each model edits, what it drops, what it costs) so you can put the choice to the user. **Truncation:** only Gemini refuses an over-length source. Happy Horse silently edits just the first 15s, Kling O3 the first 10s, and Grok the first 8s. The command warns you when that will happen; always relay it to the user. Gemini regenerates the audio track, so use Happy Horse or Kling O3 with `--preserve-audio` when the source audio must survive. Choose Grok for the cheapest prompt-only edit, Happy Horse for up to 5 image references, or Kling O3 for controlled reference-image edits.\\n- To edit a source longer than 10s without losing its tail, cut it into <=10s pieces in the native editor, edit each with Gemini, and reassemble. VideoDraft has no server-side split/concat, so this path needs `videodraft_editor`.\\n- Kling O3 also has a reference-generation mode. Use `videodraft generate video --model kling-o3-video-ref-edit` with exactly one `--ref-video` to generate a new guided clip; use `videodraft edit video` when changing the source itself. For new mixed-reference generation use Seedance 2 / 2.5.\\n- Motion transfer uses `videodraft edit motion` with Kling V3 by default, or Kling 2.6 when explicitly requested or lower cost matters. It requires a subject image and a motion-reference video.\\n- Veo 3.1, Sora, Grok, Happy Horse and the other Tier 2 video models: only when explicitly requested or when no Tier 1 model fits.\\n\\n**Audio and utilities:**\\n\\n- Use Seed Audio 1.0 for open-ended text-to-audio, speech/music/sound synthesis, voice conditioning, or prompt-driven editing with up to three audio references or one image. Use `videodraft generate audio`. Reference clips are `@Audio1`, `@Audio2`, and `@Audio3` in array order. There is no duration input. Output is up to two minutes and settles at 19 credits per actual minute, with up to 38 credits reserved during generation. The CLI automatically retries transient responses with one operation key. To recover after the CLI process itself is interrupted, set `--idempotency-key <uuid>` on the original command and reuse it.\\n- Prefer ElevenLabs for voiceover, dialogue, voice changing, dubbing, and sound effects. Honor an explicitly selected supported TTS voice/provider. Use `lyria-3.5` for all music, short or long (up to ~3 minutes, 10 credits flat), with vocals/lyrics or instrumental arrangements. When `--model` is omitted the CLI sends no model and the server default applies, which is `lyria-3.5` wherever the backend supports it. If the server answers that `lyria-3.5` is an unknown model, that backend predates it: rerun without `--model` (the server default then applies), or use `lyria-3-pro-preview` for a track longer than 30 seconds. Ask for the length in the prompt. Keep `lyria-3-clip-preview` (fixed ~30s, 4 credits) and `lyria-3-pro-preview` (8 credits) for explicit requests. For Lyria, put desired length, lyrics and \\"instrumental only, no vocals\\" in the prompt; `--length` and `--instrumental` are ElevenLabs-only. Use ElevenLabs Music for exact timing, composition plans, or a style reference track. Lyria allows 10 reference images on Google, 1 on Fal BYOK; Fal 3.5 prompts are limited to 5000 characters. BYOK uses zero VideoDraft credits. ElevenLabs Music means `elevenlabs-music-v2.5` (`elevenlabs-music` is an alias for it); use `elevenlabs-music-v1` only when the user asks for v1 by name.\\n- For ElevenLabs voiceover, dialogue, and voice changing, accept a supplied raw voice ID (16-64 alphanumeric characters) or `elevenlabs-<id>`. The voice catalog is for discovery, not an allowlist. Do not reject or substitute a supplied ID because it is absent from `videodraft models voices` / `list_available_voices`. Use `--voice <id>` for voiceover and voice changing, or repeat `--line \\"<id>:Text\\"` for dialogue; for example, `--voice kPzsL2i3teMYv0FxEYQ6` and `--line \\"elevenlabs-kPzsL2i3teMYv0FxEYQ6:Hello.\\"` use the same voice. The voice must be accessible to the provider account used for generation. Private or cloned voices may require the user\'s connected ElevenLabs key. Kling video-control IDs and MiniMax `custom-*` IDs are separate voice systems.\\n- A character who needs to TALK:\\n - **Speaking inside a scene**, or any ordinary clip with dialogue: `generate video` with the line in the prompt, in quotes, with who says it and how. `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` and `kling-o3` all voice it natively. For a specific voice use a Seedance `--ref-audio` clip or a Kling voice bound per element. This is the default path; do not generate speech separately and lip-sync it on.\\n - **Talking to camera from a portrait** (presenter, spokesperson, explainer): the avatar lane, and VEED Fabric is preferred. Use managed `avatar create` then `avatar render` for a reusable avatar record with bundled speech, `avatar fabric` for a one-off portrait plus text or existing audio, and `avatar h3-lipsync` for a short (5-14.8s) portrait clip from existing audio at up to 2K. Not for in-scene shots: Fabric animates a portrait facing the lens, so a cinematic request comes back as a head-on talking headshot.\\n - **Footage that already exists plus replacement audio** (re-dub, translation): `avatar lipsync`. This is the only job for Sync Labs unless the user names it.\\n- Enhancement: use Topaz image/video upscaling only when the content is already correct. Use image 1x for cleanup, 2x by default, 4x when justified; use video 2x by default. Edit or regenerate creative errors.\\n\\nSee [references/models.md](references/models.md) for the detailed routing table and exact capability limits.\\n\\n## Prefer references when continuity matters\\n\\nPure text-to-image or text-to-video is fine for a generic one-off asset. When a specific character, product, location, style, composition, or brand identity must survive generation, use references instead of hoping the prompt recreates it.\\n\\n- If the user supplies reference media, preserve and pass it. Never reduce the request to text alone.\\n- When continuity matters, generate/select a strong still first with the selected image model (`nano-banana-2` by default), wait for its URL, then animate it as a start frame/reference. Confirm the combined image and video cost.\\n- When using a hosted storyboard stage for multiple shots, use `videodraft shots <project_id> --model <selected-image-model> --grid`, then animate the decoded shots. Preserve explicit models. In VideoDraft ADE, import the resulting assets into the native editor instead of continuing into hosted production. A requested non-Seedance video model must use manual per-shot generation instead of Seedance full-video mode.\\n\\n## Cost and credits\\n\\nDo not call `videodraft credits` before routine generations. Paid endpoints validate and deduct atomically; if the balance is insufficient, the request is rejected before the provider job starts (CLI exit code 4). Check the balance only when the user asks, gives a credit budget, or a large workflow needs budget planning.\\n\\nFor expensive work, estimate with `--estimate` or `videodraft costs`, state the selected model/settings/cost, and get a go-ahead. This matters most for shot-image batches, long or high-resolution video, AI Production, and paid audio batches. Honor the user\'s confirmation preference for the session.\\n\\nKling voice creation costs 1 VideoDraft credit on the platform Fal account and 0 credits with Fal BYOK. Preview it with `videodraft kling-voices create <sample> --name <name> --estimate`; the estimate does not create a voice or require consent confirmation. Actual creation requires `--confirm-consent`.\\n\\nMiniMax H3 costs 5 / 6 / 13 / 16 credits per output second at 480p / 768p / 2K / 4K (768p by default). In reference mode the first 5 images are included and each additional image costs 8 credits. Reference video and reference audio are NOT billed.\\n\\nMiniMax H3 Max costs 5 credits per output second at 480p and 8 credits per second at 768p (default), plus pooled reference tokens in reference mode: the first 4,096 are free and every 1,000 after that costs 2 credits, where an image is `(width x height) / 1024` tokens, a reference video is 2,886 tokens per second at 480p or 7,459 at 768p, and reference audio is about 2,121 per second. Use `--prompt-expansion-mode disabled|balanced|quality`; balanced is the default. Fal\'s provider safety checker is OFF by default and is not exposed in the app \u2014 only pass `--safety-checker true` if the user explicitly asks for it.\\n\\nWan 3.0 costs 7 / 14 / 28 credits per output second at 480p / 720p / 1080p. Auto duration reserves the 30-second maximum and refunds the unused reserve after the provider reports the actual whole-second output length. Fal BYOK runs on the connected user key and charges zero VideoDraft credits.\\n\\nGrok Imagine Video 1.5 costs 8 credits per output second at 480p, 14 at 720p, or 25 at 1080p, plus 1 credit for each first-frame or reference image. Native audio is always generated.\\n\\n`videodraft models image|video` lists the live image and video catalogs with supported inputs. Video entries are grouped as `generation`, `video_edit`, `motion_control`, `avatar_lipsync`, and `upscale`, and each reports the exact tool. Use `videodraft models video --category video_edit` to narrow the list. `videodraft models audio` lists Seed Audio, Google Lyria, and ElevenLabs audio/media tools, while `videodraft models voices` helps discover TTS voices. Consult them for capabilities; a supplied ElevenLabs voice ID does not need to appear in the voice catalog.\\n\\n## Async jobs\\n\\nImage/video generation is asynchronous: commands submit a job and **wait by default**, printing output URLs (and saving files with `--download`). Large downloaded images also get a downscaled copy in `previews/` next to them (the `preview` field / \\"inspect via preview\\" line in the output) \u2014 **look at the preview, deliver the original**; viewing full-resolution images bloats the chat permanently. In scripts/CI prefer explicit control:\\n\\n```bash\\nJOB=$(videodraft generate image \\"...\\" --no-wait --json | jq -r .job_id)\\nvideodraft wait \\"$JOB\\" --download \\"./outputs/{job_id}_{index}.{ext}\\" --json\\n```\\n\\nFor MANY jobs: submit each with `--no-wait`, collect ALL with one command \u2014 `videodraft wait <id1> <id2> ...` polls every job from one process with one batched request per tick. Do NOT spawn parallel `wait`/`generate --wait` processes for a batch.\\n\\nIf a wait times out, the job is still running server-side \u2014 `videodraft status <job_id>` later. Never re-submit just because a wait timed out (that double-spends credits).\\n\\nFor completed Wan 3.0 jobs, MCP `check_generation_status` and CLI `status`/`wait --json` include `outputMetadata` with Fal\'s returned `seed`, `duration`, and `actual_prompt` when present.\\n\\n## AI Studio sessions\\n\\nStandalone image/video/audio generations are filed into an AI Studio session in the web app. 3D assets use `assets 3d list/get` separately. You do not have to create an AI Studio session:\\n\\n- **MCP hosts** (Claude Code, claude.ai, Codex, VideoDraft ADE): on 2025-era MCP the server mints an `Mcp-Session-Id` on `initialize`; your host echoes it, and this conversation\'s generations land in their own session. On stateless MCP (2026-07-28) hosts that send a conversation id (ChatGPT) get the same. Otherwise `name_current_ai_studio_session` returns `pass_session_id: true` with a `session_id`: pass that `session_id` to every later standalone generation in the conversation. Tool results echo the session as `ai_studio_session_id`.\\n- **CLI**: the same handshake runs once per (profile, server, working directory) and is cached for 12 idle hours, so everything generated from one directory shares one session. `videodraft sessions current` shows it; `videodraft sessions reset` starts a new one.\\n- Project generations (`--project <id>` / `project_id`) always go to that project\'s session.\\n\\nOnce you understand the creative task, name the current automatic session before the first standalone generation:\\n\\n```bash\\nvideodraft sessions name \\"Purple Seal Rescue Short\\"\\n```\\n\\nChoose a concise, specific 3-6 word title for the intended work. Do not copy the client name, date, or exact chat title. Name it once: the operation creates the session with that title. If generation, a user, or an earlier agent created the session first, its existing name is preserved.\\n\\nApart from the `pass_session_id: true` case above, pass `--session <id>` / `session_id` only to **continue earlier work** or create an explicit separate group:\\n\\n```bash\\nSESSION=$(videodraft sessions create \\"Fox brand explorations\\" --json | jq -r \'.session.id\')\\nvideodraft generate image \\"a red fox in snow, cinematic\\" --session \\"$SESSION\\"\\nvideodraft generations --session \\"$SESSION\\" # what is in it\\nvideodraft sessions list --name fox # find it again later\\n```\\n\\n`VIDEODRAFT_SESSION=<id>` sets the default for every command (ignored when `--project` is given). While it is set, `sessions name` refuses to run because that command names the current automatic connection session, not the pinned override; unset it first or rename the pinned session in AI Studio. `VIDEODRAFT_SESSION_SCOPE=<label>` groups several directories into one connection session; `VIDEODRAFT_CLIENT_NAME=<host>` labels the fallback placeholder used when generation creates the session before `sessions name`; `VIDEODRAFT_NO_SESSION=1` disables the handshake. Generations that reach the server with no session at all fall back to the account-wide \\"Agent (MCP)\\" session; if you see work landing there, pass `--session` explicitly.\\n\\n## Generation history\\n\\nPast work is queryable \u2014 reuse a previous setup instead of guessing. `videodraft generations` lists recent generations; scope with `--session <id>` or `--project <id>` (includes collaborators\' rows in shared scopes; pass one, project wins), and filter with `--type`, `--model`, `--favorites`. `--full --json` returns each row\'s exact parameters (aspect ratio, resolution, duration, references) \u2014 the human table stays compact, so pair `--full` with `--json`. `videodraft generation <id>` prints one generation\'s complete recipe (prompt, input image, parameters, outputs); `--favorite` / `--unfavorite` stars it. `videodraft sessions list` shows AI Studio sessions (owned + shared) with `--name` search \u2014 take a session id from there to read its history.\\n\\n## Free stock footage and photos\\n\\nPexels and Pixabay are wired in through `search_stock_media` / `import_stock_media` (`videodraft stock search \\"<query>\\"`, `videodraft stock import <ref>`). Both libraries are free, watermark-free and cost **zero credits**, so use stock for b-roll, establishing shots, backgrounds and ordinary real-world footage, and spend credits on the shots that must be specific: the product, the character, the scripted action.\\n\\n```bash\\nvideodraft stock search \\"city skyline night\\" --min-duration 5 --orientation landscape\\nURL=$(videodraft stock import pexels:video:35379336 --quality hd --json | jq -r .url)\\n```\\n\\n- Always two steps. A search result\'s `preview_url` is a thumbnail for judging the shot; never place it on a timeline, attach it to a shot or send it to a model. `import_stock_media` copies the file onto the VideoDraft CDN and returns the URL every other surface accepts, including the native editor\'s `library_manage` import.\\n- `--quality hd` (default) caps video at 1080p; `4k` caps at 2160p, `best` takes the largest the provider has, `sd` suits rough cuts. Stills ignore quality and import at full size. Use `--orientation portrait` for 9:16, and `--min-resolution 1920` to drop anything below HD on its long edge.\\n- Photos come from Pexels. Pixabay contributes video only, because its full-size image host refuses server-side downloads.\\n- Credit the creator and link the provider page when you show results or deliver the finished work; both come back on every result. Skip clips that imply a person or brand endorses the product, and avoid recognisable logos in ads.\\n- Search and import are rate limited per user (30 searches and 15 imports a minute) and searches are cached for 24h, so a burst of searches while planning a montage is fine. Bulk downloading a stock library is prohibited by both providers. A note saying Pixabay was skipped means its provider budget for that minute is spent; the Pexels results still stand.\\n\\n## Downloading from YouTube, Instagram, TikTok and other sites\\n\\nVideoDraft ships no downloader. When the user asks to pull a video from YouTube, Instagram, TikTok, X, LinkedIn or anywhere else, `yt-dlp` on the user\'s own machine does the work. Check for it with `command -v yt-dlp` before promising anything.\\n\\n**Present:** pin the format. On its defaults yt-dlp takes the best stream, usually VP9 or AV1 in WebM, which the native editor\'s import rejects (mp4, mov and m4v only). This returns one pre-muxed H.264 + AAC MP4 and needs no ffmpeg:\\n\\n```bash\\nyt-dlp -f \\"b[ext=mp4]\\" -o \\"media/%(title)s.%(ext)s\\" \\"<url>\\"\\n```\\n\\n**Missing:** say so in one line, then offer `brew install yt-dlp` only where `brew` exists, and only after asking. With no Homebrew, say the tool is unavailable and stop.\\n\\n- Never install Homebrew, never `sudo`, never write to `/usr/local/bin` or `/opt`, never pipe a script into a shell.\\n- `ffmpeg` is optional, matters only above 720p, and the selector above never needs it.\\n- Auth-gated content needs `--cookies-from-browser`; raise it only after a download fails on auth.\\n- Fetch only what was asked for, never a channel, playlist or back catalogue.\\n\\n## Local files and reference images\\n\\nReference inputs must be public URLs. The CLI uploads local files automatically wherever a URL is expected (`--ref photo.jpg`, `--start-image frame.png`), or explicitly:\\n\\n```bash\\nURL=$(videodraft upload ./product.png --json | jq -r .url)\\n```\\n\\nNever silently drop a reference you couldn\'t upload \u2014 stop and tell the user. Never upload a user\'s file to a third-party host.\\n\\nWhen the user attaches media for a native production, import actual footage into the editor by default. For hosted generation/storyboarding, classify each item before acting: a recurring **visual asset** (character/product/location/style), actual **footage to place as shots**, or **inspiration only**. See [references/pipeline.md](references/pipeline.md) for the hosted role mapping.\\n\\n## Showing media to the user\\n\\nGenerated media is **not** displayed in the chat automatically \u2014 you decide what to show. To preview an asset inline, save it locally (use `--download` so it lands under `media/`) and reference its **local path** as a Markdown link with a **leading `./`**:\\n\\n```\\n[ferrari shot](./media/ferrari_01.png) \u2190 image card\\n[the clip](./media/clip.mp4) \u2190 video player\\n[voiceover](./media/vo.mp3) \u2190 audio player\\n```\\n\\nPut the Markdown link **in your message text** \u2014 video and audio embed exactly like images. Do **not** use `SendUserFile` (or other file-send tools) to display media: that renders inside a collapsible tool card and gets buried in the tool list. The Markdown link in your prose is what produces the inline card.\\n\\nUse the path you saved to: a **workspace-relative** path (`./media/clip.mp4`, or `./<any-folder>/clip.mp4` \u2014 any folder in the workspace works), or the **absolute** path for a file outside the workspace (e.g. `/Users/you/Desktop/clip.mp4` or another workspace\'s path). Both render. Show the finished results worth showing (and only those \u2014 not every intermediate job). A bare CDN URL or a JSON dump of output URLs does **not** render; the local-path Markdown link is what produces an inline card.\\n\\n## Native-first VideoDraft ADE pipeline (idea \u2192 MP4)\\n\\nWhen `videodraft_editor` is present:\\n\\n1. Generate or source the script, storyboard, shot images, clips, voiceovers, music, and other assets through the cloud CLI/MCP as needed.\\n2. Call native `project_manage` (`project`, action `open` or `create`) to open or create the `.vdproject`.\\n3. Call native `library_manage` `import`, wait for imports to become ready, then assemble and refine the timeline with `edit_apply` recipes.\\n4. Call native `delivery_manage` `submit` and use its `jobs` operation for progress and results.\\n\\nDo not run the hosted production or export steps in this path unless the user explicitly asks for a web production.\\n\\n## Hosted fallback pipeline (idea \u2192 MP4)\\n\\nUse this only when there is NO native editor at all, or the user explicitly requests the hosted web\\nworkflow. The native surface is not only the injected `videodraft_editor` MCP: a `videodraft-editor`\\nexecutable on PATH is the same editor reached through its terminal bridge, and\\n[references/editor.md](references/editor.md) covers driving it that way. Treating a missing MCP as\\n\\"no editor\\" sends sessions that have the binary into hosted production for no reason.\\n\\n```bash\\nvideodraft create \\"<idea>\\" --ar 9:16 # project: script \u2192 visual assets \u2192 storyboard\\nvideodraft shots <project_id> --grid --estimate # cost preview, confirm with user\\nvideodraft shots <project_id> --grid # batch shot images (waits, writes onto shot cards)\\nvideodraft produce <project_id> # voiceovers + captions + production timeline\\nvideodraft export <project_id> --download final.mp4\\n```\\n\\nOptional between produce and export: per-shot motion clips (`videodraft generate video ... --project <id>` then place it with `videodraft attach <project> --scene N --shot M --media <url|file> --type video --duration <s>`), music (`videodraft generate music \\"...\\" --attach <project_id>`), and standalone audio assets (`generate audio`, `generate sound-effect`, `generate dialogue`, `generate voice-changer`, `generate dub`). Details, per-step tools and editing rules: [references/pipeline.md](references/pipeline.md).\\n\\n## Avatar and talking-head videos (both surfaces)\\n\\nAvatar generation is cloud-only \u2014 the native editor has no avatar or lipsync tools \u2014 so this applies whether or not `videodraft_editor` is present. Generate the avatar in the cloud; in VideoDraft ADE, import the rendered clip and cut it on the native timeline like any other footage.\\n\\nAvatar/talking-head videos use dedicated commands. For a reusable managed avatar, obtain or generate a clear portrait \u2192 `videodraft avatar script` when needed \u2192 `videodraft avatar create` \u2192 `videodraft avatar render --resolution 720p`. For a one-off portrait, use `videodraft avatar fabric <portrait> --text \\"...\\"` or `--audio <file>`, or `videodraft avatar h3-lipsync <portrait> --audio <file>` for a short high-resolution clip from existing audio. For an existing video plus replacement audio, use `videodraft avatar lipsync <video> --audio <file>`. Managed script/creation is bundled/free; direct Fabric, H3 Max Lip Sync, Sync, the managed Fabric render, and optional portrait generation/upscaling are paid. Confirm expensive steps first.\\n\\n## Working with hosted project data\\n\\nA hosted project is one JSON blob (script, storyboard scenes, shot cards, visual assets, production timeline). To inspect: `videodraft projects get <id>`. To edit: fetch `--raw`, modify, then `videodraft call update_project` \u2014 objects deep-merge, **arrays replace wholesale** (send the complete `storyboard.scenes` array to change one scene). Snapshot first with `videodraft checkpoint create <id>` before risky edits. Schema reference: `videodraft call get_project_schema`. This does not replace native editor tools when `videodraft_editor` is available for the production itself.\\n\\n## More\\n\\n- [references/pipeline.md](references/pipeline.md) \u2014 hosted fallback data model and production workflow\\n- [references/editor.md](references/editor.md) \u2014 native headless editor routing, project selection, import, timeline edits, verification, and export\\n- [references/models.md](references/models.md) \u2014 choosing image/video models, pricing patterns, voices and styles\\n- [references/examples.md](references/examples.md) \u2014 recipes: batch product videos from a CSV, talking-head from a script, changelog video in CI\\n","references/3d.md":"# Standalone 3D assets through MCP and CLI\\n\\nUse this lane for downloadable meshes, textured 3D models, and humanoid rigging. These are independent of AI Studio sessions and do not require a storyboard or web viewer. Never create a project just to generate a mesh. Native editor media import does not establish support for GLB/FBX assets; use external 3D software unless the live editor schema explicitly supports them.\\n\\n## Discover and quote\\n\\n```bash\\nvideodraft models 3d --json\\nvideodraft generate 3d \\"a worn brass telescope\\" --model meshy-7 --estimate\\nvideodraft generate 3d --model tripo-h3.1 --ref ./reference.png --options @options.json --estimate\\nvideodraft rig 3d --estimate\\n```\\n\\nMCP tools: `get_3d_models`, `estimate_3d`, `generate_3d`, `rig_3d`, `check_generation_status`, `list_3d_assets`, `get_3d_asset`, `create_3d_upload`, and `finalize_3d_upload`.\\n\\nThe generator catalog is deliberately limited to `meshy-7` and `tripo-h3.1`. Both expose text/image/multi-image modes; use the live schemas for option availability, view order, limits, topology, texture/PBR, and pose controls. Do not invent common provider option names. Server defaults prioritize quality. Rigging is a separate Meshy operation for compatible textured humanoid GLBs; it is not an arbitrary creature or facial rig service.\\n\\nEvery option is accessible through `--options` JSON (`@path.json` supported), plus repeatable `--option key=value` overrides. Values parse as JSON when valid. Options and explicit input mode affect pricing, so obtain a fresh quote for the exact recipe. Estimates perform no file upload or provider generation. VideoDraft bills whole credits at 100 credits per dollar using the server\'s rounding rules. BYOK costs zero VideoDraft credits and runs on the connected user Fal key. Do not infer the user\'s Fal-account charges from a zero VideoDraft-credit quote.\\n\\n## Generate and retrieve\\n\\n```bash\\nvideodraft generate 3d \\"a worn brass telescope\\" --model meshy-7 --download --json\\nvideodraft generate 3d --model meshy-7 --ref ./hero.png --no-wait --json\\nvideodraft generate 3d --model tripo-h3.1 --input-mode multi_image --ref ./front.png --ref ./left.png --ref ./back.png --ref ./right.png --no-wait --json\\nvideodraft status JOB_ID --json\\nvideodraft wait JOB_ID --download --json\\nvideodraft assets 3d list --status completed --json\\nvideodraft assets 3d get ASSET_ID --download ./media/3d --json\\nvideodraft rig 3d --asset ASSET_ID --download --json\\nvideodraft rig 3d ./textured-humanoid.glb --download --json\\n```\\n\\nLocal image references upload automatically. Local rigging inputs must be self-contained `.glb` files and upload via the dedicated 3D route. Input mode is inferred from reference count unless `--input-mode text|image|multi_image` is supplied. A text prompt requires no images; image mode requires exactly one and no geometry prompt. Meshy multi-image accepts 1-4 views, and Tripo accepts 2-4 in front, left, back, right order. Use Meshy\'s `texture_prompt` option when texturing guidance is needed with image inputs. Do not mix unrelated images as if they were views of the same object.\\n\\nGeneration and rigging wait by default. `--no-wait` returns a persisted job that continues server-side. For many jobs use one `wait ID1 ID2 ... --download`; do not run parallel polling processes. Recover timeout with the same job ID, not another paid generation.\\n\\nSubmission returns a `request_id` UUID. A same-request retry must use `--request-id UUID` and the original arguments. The CLI journals resolved uploads in its private config directory so retries reuse the same remote files. Changed inputs under that UUID are rejected. Submission errors include the UUID in JSON `details.request_id` where available. Never retry a failed request with a new UUID until the prior job status is known.\\n\\n## Complete artifact packages\\n\\n`--download` defaults to `media/3d/<asset_id>/`. A user directory is respected and receives a per-asset folder. `{job_id}` and `{asset_id}` directory templates are accepted. Do not pass a `.glb` filename or a media `{index}.{ext}` template: a 3D result can contain multiple formats and texture/material dependencies.\\n\\nDownloads preserve every returned artifact. The manifest records provider URLs, source filenames, local paths, role, content type, and actual downloaded sizes. glTF/OBJ/MTL references are rewritten to saved dependency paths; binary GLB/FBX and provider ZIPs are retained as returned. Supplied dependency aliases become local copies at the paths needed by original FBX files. Unsafe or colliding aliases produce warnings. Existing packages are never overwritten. A successful package has `manifest.json`; read its `complete` field and `warnings`, plus CLI `package_warnings`. Do not claim a package with unresolved dependencies is ready for offline use.\\n\\nMachine results use `type: \\"model3d\\"` with `output_files` for files and `output_media` only for ordinary preview media. Texture maps and models are not image/video gallery entries. Show a supplied preview via a local image Markdown link and deliver the original mesh/package files. Do not imply an interactive 3D viewer exists in AI Studio or VideoDraft ADE.\\n","references/editor.md":"# Native VideoDraft Editor reference\\n\\nUse this reference when the user wants to assemble, cut, caption, mix, lay out, inspect, or export a local VideoDraft Editor project. The native editor is deterministic and local. Cloud generation remains in the `videodraft` CLI or hosted MCP.\\n\\n## VideoDraft ADE preference rule\\n\\nWhen `videodraft_editor` tools are exposed, treat the native editor as available and make it the default surface for production, timeline assembly, and final export. It is headless by design, so a hidden window or an untouched Open Editor button does not justify using hosted production instead.\\n\\nUse cloud tools for asset generation and optional script/storyboard work, then import the results. Do not call hosted `produce_project` / `videodraft produce` or `export_video` / `videodraft export` unless the user explicitly requests an editable web production or the native editor tools are unavailable. If a native tool call fails after the editor was available, report or recover that native failure rather than silently switching surfaces.\\n\\n## Choose the correct surface\\n\\n- `videodraft` and the hosted VideoDraft MCP generate assets and can manage hosted web projects. They use the user\'s VideoDraft account and credits. In VideoDraft ADE, use them mainly as the source of generated media and optional storyboards for the native production.\\n- `videodraft_editor` edits local `.vdproject` packages. It has no generation, account, model, or credit tools.\\n- Inside VideoDraft ADE on a supported Mac, the editor MCP is injected automatically for Claude, Codex, OpenCode and Grok in both Code and VideoDraft modes. It starts headlessly before the chat opens. The user does not need to click Open Editor, and closing or hiding the editor window does not stop headless editing.\\n- Outside that environment, use the editor only if `videodraft_editor` MCP tools are already exposed or the `videodraft-editor` executable is on PATH. Do not confuse the public `videodraft` cloud CLI with the separate native editor executable.\\n\\nPrefer the direct MCP tools when they are available. The terminal bridge is useful for scripts, diagnostics, or an agent session where the MCP was not injected.\\n\\n## Which editor tools you have\\n\\nCurrent editors expose eight workflow tools: `project_manage`, `edit_snapshot`, `edit_apply`, `edit_undo`, `library_manage`, `inspect`, `speech_apply`, and `delivery_manage`. Use them whenever `edit_apply` is available. Everything below describes them.\\n\\nEarlier editors expose a different set of tools. If `edit_apply` is not in your catalog, follow the live tool descriptions; the working rules below still apply.\\n\\nThe live tool descriptions and schemas are authoritative. When they differ from this reference, follow them.\\n\\nService tools (`project_manage`, `library_manage`, `inspect`, `speech_apply`, `delivery_manage`) take `operation` and `parameters`, plus an optional `context` (see [Keep a reliable editing model](#keep-a-reliable-editing-model)):\\n\\n```json\\n{\\"operation\\": \\"import\\", \\"parameters\\": {\\"source\\": {\\"path\\": \\"/Users/me/clips\\"}}}\\n```\\n\\n### Parameters at a glance\\n\\nEach operation\'s schema has the full list. These are the fields that matter most, and the ones most often confused:\\n\\n| Tool and operation | Key `parameters` |\\n| --- | --- |\\n| `project_manage` `project` | `action` (`list`, `open`, `create`, `close`); `id`, `name` or `path` to open; `name`, `fps`, `aspectRatio`, `quality` to create |\\n| `edit_snapshot` (no `parameters`) | `includeLibrary`, a `startFrame`/`endFrame` window, `offset`/`limit` paging, `context` for later pages |\\n| `edit_apply` (no `parameters`) | `actions`; optional `context`, `requestId`, `previewOnly` |\\n| `library_manage` `import` | `source` with one of `path`, `url`, `bytes` + `mimeType`, or `matte`; optional `name`, `folder`. A `.srt` or `.vtt` file becomes captions |\\n| `library_manage` `list` | `ids` to poll imports, `pending`, `folder` |\\n| `inspect` `timeline` | a `startFrame`/`endFrame` window; there is no clip filter |\\n| `inspect` `frame` | `startFrame` for one frame; add `endFrame` and `maxFrames` to sample a range |\\n| `inspect` `media` | `mediaRef` (a library asset, not a file path), `startSeconds`/`endSeconds` in source seconds, `wordTimestamps`, `overview` |\\n| `inspect` `transcript` | a `startFrame`/`endFrame` window, `granularity` (`words` or `segments`), `clipId` |\\n| `inspect` `color` | `clipId` with `atFrame`, or `mediaRef`; optional `reference` |\\n| `speech_apply` `words` | `words`: transcript word indices, each one index or a `[first, last]` pair; or `matches`: exact words to remove everywhere; `pacing` |\\n| `speech_apply` `silence` | none |\\n| `speech_apply` `captions` | caption `style`, `position` and look fields; it finds the speech itself. `style` `plain` (sentences without a preset) takes `positionY` or `transform.centerY`, not `position` or `punctuation` |\\n| `delivery_manage` `submit` | `mode`, `codec`, `resolution`, `outputPath` (its folder must already exist); `captionGroupId` and `wordTiming` for `srt` and `vtt`; `check: false` skips a video\'s export check |\\n| `delivery_manage` `jobs` | `action` `list` (every job with its progress, warnings, output path and check findings; with a `jobId`, that job with all its findings), `cancel` with a `jobId`, or `dismiss` with a `jobId` and optional `findingIds` |\\n\\n`previewOnly` exists only on `edit_apply`. Service operations apply immediately.\\n\\n### `edit_apply` actions at a glance\\n\\nEvery action puts its fields beside `kind`, for example `{\\"kind\\": \\"adjust\\", \\"clipId\\": \\"c1\\", \\"speed\\": 2}`. `remove`, `adjust`, `replace`, `transition`, `mask`, `text`, `grade` and `effects` take `clipId` for one clip or `clipIds` for several; `animate`, `extract`, and the rows of `move` and `split` take a single `clipId`. Advanced actions cannot take `as`, but their clip ids accept `@name`. The one exception is `transition`, whose own `kind` field names the style, so its fields go inside `parameters`: `{\\"kind\\": \\"transition\\", \\"parameters\\": {\\"clipId\\": \\"c2\\", \\"kind\\": \\"dipToBlack\\"}}`. These `actions` place a clip, warm it and add a vignette:\\n\\n```json\\n[{\\"kind\\": \\"place\\", \\"assetId\\": \\"\u2026\\", \\"atFrame\\": 0, \\"as\\": \\"shot\\"},\\n {\\"kind\\": \\"grade\\", \\"clipId\\": \\"@shot\\", \\"adjustments\\": {\\"temperature\\": 7500}},\\n {\\"kind\\": \\"effects\\", \\"clipId\\": \\"@shot\\", \\"effects\\": [{\\"type\\": \\"finish.vignette\\", \\"params\\": {\\"strength\\": 35}}]}]\\n```\\n\\n| Advanced `kind` | Key fields |\\n| --- | --- |\\n| `place_batch` | `entries: [{mediaRef, startFrame, endFrame or source, trackIndex}]`, not the concise `assetId`, `atFrame`, `durationFrames`; leave `trackIndex` off every entry for a new track |\\n| `insert_batch` | `trackIndex`, `atFrame`, `entries: [{mediaRef, durationFrames or source}]`; later clips move right |\\n| `move` | `moves: [{clipId, toFrame, toTrack}]` |\\n| `remove` | `clipId` or `clipIds`; leaves a gap |\\n| `split` | `splits: [{clipId, atFrame}]`, or `trackIndex` with `frames` |\\n| `extract` | `trackIndex` with `ranges: [[start, end]]` in frames, or `clipId` with `ranges` and `units` (`frames`, or source `seconds`); closes the gap |\\n| `adjust` | `clipId` or `clipIds` plus any of `durationFrames`, `trimStartFrame`, `trimEndFrame`, `speed`, `speedCurve` (`{preset}`, or `{points: [{t, rate}]}` with `t` from 0 to 1 along the source), `preservesPitch`, `volume`, `opacity`, `transform` (`centerX`, `centerY`, `width`, `height`, `flipHorizontal`, `flipVertical`), `blendMode` |\\n| `animate` | `clipId`, `property` (`volume`, `opacity`, `rotation`, `position`, `scale`, `crop`), `keyframes: [[frame, ...values]]` with frames counted from the clip\'s start; `position` is the top-left corner |\\n| `arrange` | `layout`, `slots: [{slot, clipIds or mediaRef, anchor}]`, `fit` (`fill` or `fit`); `mediaRef` slots also need `endFrame` |\\n| `transition` (fields inside `parameters`) | `clipId` or `clipIds` (the clip after each cut), `kind` (`crossDissolve`, `dipToBlack`, `dipToWhite`, `blurDissolve`, `push`, `linearWipe`, `whipPan`, `crossZoom`, or `none` to remove), `durationFrames`, `params` (`direction` 0 to 3 for left, right, up, down; `feather`; `blur`; `intensity`) |\\n| `mask` | `clipId` or `clipIds`, `shape` (`rectangle`, `ellipse`, `none`), `centerX`, `centerY`, `width`, `height` (0 to 1 of the clip\'s own frame), `feather`, `strength`, `inverted` |\\n| `replace` | `clipId` or `clipIds`, `mediaRef` (a library asset of the same kind), `trim` (`keep`, the default, or `reset`), `linkedAudio` (`follow`, the default, or `keep`) |\\n| `tracks` | `reorder: [{trackId, to}]`, `set: [{trackId, muted, hidden, syncLocked}]`, `remove: [{trackId}]`; there is no add, since placing without a track makes one |\\n| `titles` | `entries: [{startFrame, endFrame, content}]`, optionally with `trackIndex` (on every entry or none), `style`, `animation`, `transform` and typography (`fontName`, `fontSize`, `color`) |\\n| `text` | `clipId`, `clipIds` or `captionGroupId`, with `content`, `style`, `position`, `animation` or typography |\\n| `grade` | `clipId` or `clipIds`, `adjustments`, `wheels`, `curves`, `hueCurves`, `lut`, `reset` |\\n| `effects` | `clipId` or `clipIds`, `effects: [{type, params, enabled}]`, `remove: [type]` |\\n\\n- `grade`: `adjustments` holds `exposure` (EV, -4 to 4), `temperature` (kelvin, 1800 to 15000, 6500 neutral, higher is warmer), `tint` (-150 magenta to 150 green); `contrast`, `highlights`, `shadows`, `whites`, `blacks`, `vibrance` and `saturation` are signed percents (-100 to 100, 0 neutral), not factors like 1.2. `wheels`: `shadows`, `midtones`, `highlights`, each `{hue, strength, brightness}`. `curves`: `luma`, `red`, `green`, `blue` as `[x, y]` points from 0 to 1. `hueCurves`: `targets: [{hue, rotate, saturation, lightness}]`. `lut`: `{path, mix}`, with the `storedPath` from `prepare_look`. Out-of-range values are refused.\\n- Effect `type` ids and their knobs, which go inside `params` and run 0 to 100 unless noted. Defaults leave the picture unchanged, so send the knob you want to see:\\n - `finish.vignette`: `strength` and `curvature` (-100 to 100; positive `strength` darkens the edges), `size`, `falloff`\\n - `finish.grain`: `strength`, `grainSize` (0.5 to 6 px); `finish.glow`: `strength`, `haloRadius` (px), `cutoff`, `halation`\\n - `defocus.gaussian`: `blurRadius` (px); `defocus.motion`: `streakLength` (px), `streakAngle` (-180 to 180 degrees)\\n - `texture.clarity`: `localContrast`, `hazeRemoval` (-100 to 100); `texture.sharpen`: `strength` (0 to 200); `texture.denoise`: `strength`\\n - `matte.chroma`: `screenHue` (0 to 360 degrees, 120 is green), `range`, `edgeSoftness`, `spillSuppression`\\n- `arrange` layouts and their slots (fill every slot): `fullscreen` (`stage`); `split` (`left`, `right`); `stack` (`top`, `bottom`); `corner_top_left`, `corner_top_right`, `corner_bottom_left`, `corner_bottom_right` (`stage`, `corner`); `quad` (`top_left`, `top_right`, `bottom_left`, `bottom_right`); `side_panel` (`stage`, `panel`); `thirds` (`left`, `middle`, `right`). The layout picks the corner; `anchor` only biases the crop.\\n- `trackIndex` and `toTrack` are an existing track\'s `order` from `edit_snapshot` (0 draws on top), counted at that point in the recipe: a track an earlier action adds shifts them. `tracks` takes `trackId` handles instead.\\n\\n## Start with the intended project\\n\\nAn MCP session can begin without a project selected. Project selection belongs to the session, not to whichever editor window happens to be frontmost.\\n\\n1. If the user named an existing project but its identity is unclear, call `project_manage` with `operation: \\"project\\"` and `parameters.action: \\"list\\"`.\\n2. Open the exact project with `action: \\"open\\"` and the returned `id`, an unambiguous `name`, or the `.vdproject` `path`.\\n3. Create only when the user wants a new local edit. `action: \\"create\\"` accepts optional `name`, `fps`, `aspectRatio`, and `quality`.\\n4. Treat `isActive` as this session\'s target and `isVisible` as the project shown in the UI. Headless editing only needs the session target.\\n5. Use `action: \\"close\\"` only when closing is part of the task. It saves first and never deletes the project. Afterwards other calls answer `no_project` until you open or create a project again.\\n\\nOther `project_manage` operations: `create_timeline` and `select_timeline` for additional timelines, and `configure` for project settings (a frame-rate change applies to every timeline).\\n\\nDo not substitute a hosted project ID for a native project. A hosted project can supply scripts, storyboards, and generated media, but the native edit is a separate `.vdproject` package.\\n\\n## Keep a reliable editing model\\n\\n- Opening or creating a project, and creating or selecting a timeline, returns `data.snapshot` (the `edit_snapshot` view with the library: format, tracks, clips and assets) and a `context`, so you can edit right away. Call `edit_snapshot` only after an out-of-band user edit, to page a long timeline with `offset` and `limit`, or for handles you lack.\\n- `context` (`epoch`, `timelineId`, `revision`) is optional on every write except `edit_undo` and `project_manage` `project`, which take none. Without it, a write applies to the current state. With it, the editor refuses the write if the project changed since that context, which is what you want when a person may be editing at the same time. Every reply returns a fresh `context`. After `context_expired` (a reopen or restart), take a new snapshot.\\n- Timeline positions and durations are frames at `format.fps`. `sourceSeconds` values are seconds in the source file. Pass values as returned; do not multiply or divide by fps yourself.\\n- IDs are short handles. Pass them back exactly as returned; do not derive them from UUIDs. Tracks keep stable handles; indexes can change.\\n- When the user says \\"this clip\\", \\"these captions\\", or \\"here\\", take a fresh `edit_snapshot` (a one-frame window is enough). Its `selection` names what they selected in the project\'s window, in the snapshot\'s handles: `clipIds`, `captionGroupIds`, a `gap`, the marked `range`, and library `mediaIds`; no `selection` means nothing is selected. `currentFrame` is the playhead. `visible: false` means the user is not looking at this project.\\n- Send writes serially. Parallel writes against one project race each other\'s context.\\n- A refused call answers `status: \\"rejected\\"`: nothing changed, so fix what the message names and send it again. A service write that answers `failed` may have applied before a save or connection failed; inspect before repeating it.\\n- Use `inspect` for detail and verification: `timeline` for exact clip and track properties, `frame` for rendered frames of the composited result, `media` before describing source content, `color` for scopes, and `transcript` to locate spoken words.\\n- Volume values, including volume keyframes, are linear from `0` to `1`.\\n\\n## Edit with recipes\\n\\n`edit_apply` takes an ordered list of `actions`, plus an optional `context` and `requestId`. The editor validates the whole recipe before changing anything, then applies it as one undo step. If any action fails, nothing changes and the error names the action. An applied reply lists the resulting `clips` (position, length, source window, text); use it to confirm the edit instead of re-reading.\\n\\n- Concise actions: `place` (an `assetId` at `atFrame`, optionally on a `trackId`, with `durationFrames` or `sourceSeconds`, `mode` `overwrite` or `insert`), `trim` (a `clipId` to a `sourceSeconds` window), and `title` (`text` at `atFrame` for `durationFrames`, optional `look` and `motion`). Name an action with `as` and address its clip in later actions as `@name`.\\n- Advanced actions put their fields beside `kind` too: `place_batch`, `insert_batch`, `move`, `remove`, `split`, `extract`, `adjust`, `animate`, `arrange`, `transition` (fields inside `parameters`), `mask`, `replace`, `tracks`, `titles`, `text`, `grade`, and `effects`. Their field help lists every supported effect, grading control, mask, layout, speed curve, and caption control.\\n- Use `arrange` for split screens, picture-in-picture, grids, and canvas placement, and `tracks` to fix stacking. Do not synthesize layouts from generic transforms or keyframes.\\n- Use `replace` to swap a clip\'s media and keep its timing, effects, keyframes, and links, for example a re-render of the same shot. `trim: \\"keep\\"` holds the clip\'s source offsets; `trim: \\"reset\\"` plays the new media from its first frame and keeps linked audio in sync.\\n- Put every edit a request needs into one recipe: placing, trimming, titling, grading and effects together are one call and one undo step. Use `previewOnly: true` to validate a large or risky recipe without editing.\\n- Send a `requestId` when a retry must not edit twice: repeating the identical request with the same `requestId` returns the original receipt, even after a dropped connection. Without one, a repeated request applies again. Use a new `requestId` for a different edit.\\n- `applied_unsaved` means the edit applied but saving failed. If you sent your own `requestId`, repeat the identical request to retry the save, not the edit. Without one, do not resend: a repeat would apply the edit again.\\n- `edit_undo` reverses this connection\'s latest edit while it is still the top undo step. It never undoes the user\'s own edits. Take a new snapshot afterward.\\n\\nEdits are undoable. Do not ask for confirmation before each ordinary edit. Ask one focused question only when the user\'s creative direction is materially ambiguous.\\n\\n## Bring generated or local media into the editor\\n\\nFor b-roll and establishing shots, search free stock first (`videodraft stock search \\"<query>\\"`), import the pick with `videodraft stock import <ref>`, and feed the returned CDN URL to `library_manage` `import` as `source.url`. It costs no credits.\\n\\nOtherwise use cloud generation for new assets, save or download the outputs, then call `library_manage` with `operation: \\"import\\"`:\\n\\n- `source.path`: absolute local file or directory. A directory imports recursively and preserves its folder structure.\\n- `source.url`: HTTPS asset URL. Set `mimeType` when a signed URL has no usable extension.\\n- `source.bytes`: small base64 media with a required `mimeType`.\\n- `source.matte`: generated solid-color image.\\n\\nReadiness differs by source, and so does the poll that detects it:\\n\\n- **URL and single-file path** imports return `status: \\"downloading\\"` with one `mediaRef`. Poll `library_manage` `list` with `parameters.ids: [mediaRef]` until `generationStatus` is absent.\\n- **Directory** imports also return `status: \\"downloading\\"` with one placeholder `mediaRef` for the batch. Poll it the same way; the folder\'s assets appear when `generationStatus` clears. (`pending: true` remains a fallback that lists every unresolved import.)\\n- **Inline bytes and matte** imports finish inline and come back `status: \\"ready\\"`; no polling is needed.\\n\\nAn import\'s `mediaRef` is the `assetId` that recipe actions take. Never place a pending asset on the timeline. `generationStatus` is the signal: `preparing`, `generating`, `downloading`, and `rendering` mean keep polling, absent means usable, and **`failed` is terminal**. Report it or retry the import explicitly; never poll on. Do not treat \\"not downloading\\" as ready.\\n\\nA `.srt` or `.vtt` caption file (by `path`, `url`, or `bytes` with `mimeType` `application/x-subrip`, `text/srt`, or `text/vtt`) is not added to the library. Its captions go onto a new caption track at the times the file gives, as one undo step, and the reply names the new `captionGroupId` to restyle with a `text` action.\\n\\nFor a batch of local outputs, download them into one workspace directory and import that directory once when practical. This is safer and faster than racing many import calls.\\n\\nTo apply a LUT, first store it with `library_manage` `prepare_look` (`parameters.path` to a `.cube` file), then use the returned `storedPath` in a `grade` action.\\n\\n## Speech edits\\n\\n`speech_apply` runs speech work as separate operations, not recipe actions: `words` removes transcript words, `silence` trims dead air, and `captions` generates styled captions. Read a fresh transcript with `inspect` `transcript` after any speech edit, because word positions change.\\n\\n## Service operations and timeouts\\n\\n`project_manage`, `library_manage`, `speech_apply`, and `delivery_manage` writes have no retry receipts. After a timeout, inspect the outcome (a snapshot, a library `list`, or delivery `jobs`) before repeating a write. A failed service write may have applied an edit before saving failed, so read its `data` and the current state.\\n\\n## Edit and verify\\n\\nA dependable sequence:\\n\\n1. `project_manage` to select or create the local project; its reply carries the snapshot.\\n2. `inspect` `media` when content selection matters.\\n3. One `edit_apply` recipe with every clip, track, layout, text, audio, color, and effect change the request needs; `speech_apply` for word cuts, silence, and captions.\\n4. Check the reply\'s `clips`; use `inspect` `frame` only when visual composition or layer order matters.\\n5. `edit_undo` if the result is wrong and the next recipe would not cleanly correct it.\\n\\n## Export\\n\\nCall `delivery_manage` with `operation: \\"submit\\"`. Submission is not completion: it returns a job, destination, and `started` or `queued` status.\\n\\n- Use `mode: \\"video\\"` for H.264, H.265, or ProRes.\\n- Use `mode: \\"xml\\"` (XMEML) for Premiere Pro **and DaVinci Resolve**. Resolve reads XMEML natively. Use `fcpxml` only for Final Cut Pro; sending Resolve an FCPXML produces a package it cannot open cleanly.\\n- Use `mode: \\"videodraft\\"` for a self-contained project package.\\n- Use `mode: \\"srt\\"` or `mode: \\"vtt\\"` for a subtitle file of the captions: words and timing only, without styling. `captionGroupId` picks a caption group; `wordTiming: true` adds each word\'s start to a `vtt`.\\n- Omit `outputPath` unless the user named a destination; the default is `~/Downloads`. The editor does not create folders, so a named destination\'s folder must already exist.\\n- Use `operation: \\"jobs\\"` with `parameters.action` `list` to follow progress, warnings, and results. Cancel only when the user asks or the just-queued settings were wrong. Do not infer that an export is stuck from elapsed time alone.\\n- A finished video export is then checked in the background for black picture, gaps, sound that drops out or clips, and missing media. `list` shows each job\'s `findingCount` and its first three findings; `list` with that `jobId` returns every finding. Findings are warnings on a completed export: tell the user what was found and where, rather than treating the export as failed.\\n\\n## Terminal bridge\\n\\nVideoDraft desktop terminals expose `videodraft-editor`, which controls the same process and tools:\\n\\n```bash\\nvideodraft-editor status\\nvideodraft-editor list-tools\\nvideodraft-editor tool project_manage --json \'{\\"operation\\":\\"project\\",\\"parameters\\":{\\"action\\":\\"list\\"}}\'\\nvideodraft-editor tool edit_snapshot --project \\"/path/to/My Video.vdproject\\" --json \'{\\"includeLibrary\\":true}\'\\nvideodraft-editor show\\nvideodraft-editor hide\\n```\\n\\nEach `videodraft-editor` invocation is a separate connection, so a project opened by one call is NOT remembered by the next. Pass `--project <path>` on every `tool` call that operates on a project; without it, a follow-up call answers `no_project` (no project is open for this session) even though the open succeeded. For the same reason, `edit_undo` has nothing to undo from the terminal: it only reverses an edit made on its own connection.\\n\\nControl and tool commands auto-start a headless editor if none is running. `show` only reveals the already-running UI. Use `videodraft-editor tool <name> --json -` to read a JSON object from stdin when shell quoting would be fragile. Never read, copy, or expose the editor\'s rotating local authentication secret.\\n","references/examples.md":"# Recipes\\n\\nWorking patterns for common asks. All assume auth (`videodraft login` once, or `VIDEODRAFT_API_KEY` in the environment) and use `--json` for parsing.\\n\\nInside VideoDraft ADE, if a native editor is available, use these recipes for asset generation and\\noptional script/storyboard stages. Hand the results to the editor for production and export ONLY\\nwhen the deliverable the user asked for is a composed production. A standalone output \u2014 the batch\\nproduct clips in recipe 1, the upscale in recipe 7 \u2014 is finished when it is generated; importing it\\ninto a project and exporting a timeline builds an edit nobody asked for. Recipes below that call `videodraft produce` or `videodraft export` are hosted fallbacks only. Do not choose them over the available native editor unless the user explicitly asks for the hosted web workflow.\\n\\n## 1. Batch product videos from a CSV\\n\\nOne 9:16 product clip per row of `products.csv` (`name,image_url,tagline`):\\n\\n```bash\\n#!/usr/bin/env bash\\nset -euo pipefail\\nmkdir -p outputs\\n\\nwhile IFS=, read -r name image tagline; do\\n job=$(videodraft generate video \\\\\\n \\"Premium product shot of ${name}: ${tagline}. Slow orbit, studio lighting.\\" \\\\\\n --model gemini-omni-1.1-flash --ar 9:16 --duration 6 \\\\\\n --start-image \\"$image\\" \\\\\\n --no-wait --json | jq -r .job_id)\\n echo \\"$name,$job\\" >> outputs/jobs.csv\\ndone < <(tail -n +2 products.csv)\\n\\n# Collect ALL results with ONE process (batched polling \u2014 one request per tick)\\nvideodraft wait $(cut -d, -f2 outputs/jobs.csv) \\\\\\n --download \\"outputs/{job_id}_{index}.{ext}\\" --json > outputs/results.json\\n# map job ids back to product names via outputs/jobs.csv\\n```\\n\\nSubmit-then-collect parallelizes server-side generation; the single multi-id `wait` keeps it to one local process and one batched poll request per tick no matter how many jobs. Gemini Omni 1.1 Flash is selected because these are six-second first-frame product clips. Estimate first: `videodraft costs gemini-omni-1.1-flash --type video --duration 6 --resolution 720p --audio` \xD7 rows, and confirm with the user.\\n\\nExtend an uploaded clip or conversationally edit an official Google interaction with Gemini Omni 1.1 Flash:\\n\\nEach extension appends 3-10 seconds at the end. Edit and extend accept EXACTLY ONE input video, so `--ref-video` cannot accompany `--source-video` or `--previous-interaction-id`; only `--ref` images can. New dialogue is allowed only when the source video is silent, so a chain of spoken beats must carry its speech in the first turn. A continuation is submitted as the prior turn\'s output, so 40 seconds total is reachable only by extending from a source of 30s or less, and a ladder dead-ends past 30s.\\n\\n```bash\\n# Uploaded-video extension. Reference IMAGES are fine here; reference videos are not.\\nvideodraft generate video --model gemini-omni-1.1-flash \\\\\\n --source-video ./ending.mp4 --ref ./wardrobe.png \\\\\\n --extend --duration 6 --resolution 1080p \\\\\\n --download ./media/extended.mp4\\n\\n# Conversational edit from the interaction_id of an earlier generation. Add --extend to lengthen instead.\\nvideodraft generate video --model gemini-omni-1.1-flash \\\\\\n --previous-interaction-id \\"$INTERACTION_ID\\" \\\\\\n --resolution 720p \\\\\\n --download ./media/continued.mp4\\n```\\n\\nFal BYOK supports the currently callable Gemini Omni 1.1 generation and basic source-edit endpoints at zero VideoDraft credits. Its published callable v1.1 endpoints do not expose continuation, extension, or separate creative references on a source edit. Do not infer a Fal route or retry on paid Google while Fal BYOK is active.\\n\\n## 2. Hosted full marketing video from one idea (fallback)\\n\\n```bash\\nvideodraft create \\"30-second launch video for Solace, a sleep-tracking ring. Calm, premium, dark palette.\\" \\\\\\n --ar 9:16 --style cinematic --json > project.json\\nPROJECT=$(jq -r .project_id project.json)\\n\\nvideodraft shots \\"$PROJECT\\" --grid --estimate # show the user the cost; get a go-ahead\\nvideodraft shots \\"$PROJECT\\" --grid\\nvideodraft produce \\"$PROJECT\\"\\nvideodraft generate music \\"minimal ambient, warm pads, 60 BPM\\" --attach \\"$PROJECT\\"\\nvideodraft generate audio \\"Extend @Audio1 into a 20-second transition\\" --ref-audio ./intro.wav --format wav --download ./transition.wav\\nvideodraft export \\"$PROJECT\\" --download solace-launch.mp4\\n```\\n\\nUse this complete hosted path only when the user requested a web project or the native editor is unavailable. Otherwise stop after the storyboard/assets, import them into the native `.vdproject`, and export with `delivery_manage`. The hosted project stays editable at the URL in `project.json` (`.urls`).\\n\\n## 3. Talking-head (avatar) video\\n\\nWhen the user has no portrait, generate a clear front-facing avatar image first. Skip this step when they supplied one or an existing character should be reused.\\n\\n```bash\\nvideodraft generate image \\\\\\n \\"Front-facing head-and-shoulders portrait of a friendly coffee expert, direct eye contact, natural expression, clean studio background\\" \\\\\\n --model nano-banana-2 --ar 9:16 --download ./media/avatar.png\\n\\nSCRIPT=$(videodraft avatar script \\"why our espresso subscription saves you money\\" --style ad-style --json | jq -r .script)\\nAVATAR=$(videodraft avatar create ./media/avatar.png --script \\"$SCRIPT\\" --voice elevenlabs-kPzsL2i3teMYv0FxEYQ6 --ar 9:16 --json | jq -r .avatar_video_id)\\nvideodraft avatar render \\"$AVATAR\\" --resolution 720p # VEED Fabric paid step; confirm cost first (~20 credits/sec)\\n```\\n\\n`avatar script` and `avatar create` (including speech) are bundled/free. In this example only the optional portrait generation and Fabric render spend credits.\\n\\nIf the portrait is low resolution, enhance it before `avatar create`:\\n\\n```bash\\nvideodraft upscale image ./founder-small.jpg --scale 2x --download ./media/founder-upscaled.png\\n```\\n\\nFor a one-off portrait animation without creating a managed avatar record:\\n\\n```bash\\nvideodraft avatar fabric ./founder.jpg \\\\\\n --text \\"Welcome to the weekly product update.\\" \\\\\\n --voice-description \\"warm, confident American presenter\\" \\\\\\n --resolution 720p --download ./media/presenter.mp4\\n```\\n\\nFor a short, sharp clip from speech or a song the user already has (5-14.8s of audio is used):\\n\\n```bash\\nvideodraft avatar h3-lipsync ./singer.png --audio ./chorus.mp3 \\\\\\n --resolution 1080P --audio-duration 12 --estimate\\nvideodraft avatar h3-lipsync ./singer.png --audio ./chorus.mp3 \\\\\\n --resolution 1080P --download ./media/singer-lipsync.mp4\\n```\\n\\nWhen the user already has both the video and replacement speech:\\n\\n```bash\\nvideodraft avatar lipsync ./presenter.mp4 \\\\\\n --audio ./localized-voiceover.mp3 \\\\\\n --sync-mode loop --download ./media/presenter-localized.mp4\\n```\\n\\nEdit an existing video with a dedicated edit model:\\n\\n```bash\\nvideodraft models video --category video_edit\\nvideodraft edit video ./product-demo.mp4 \\\\\\n \\"Preserve the product but follow the reference camera rhythm\\" \\\\\\n --model gemini-omni-1.1-flash --ref-video ./camera-rhythm.mp4 \\\\\\n --ref-video-duration 2.5 --resolution 1080p \\\\\\n --download ./media/product-demo-reframed.mp4\\n\\nvideodraft edit video ./product-demo.mp4 \\\\\\n \\"Turn the room into a warm evening scene while preserving the product and camera motion\\" \\\\\\n --model kling-o3-video-ref-edit --ref ./evening-style.jpg \\\\\\n --preserve-audio --download ./media/product-demo-evening.mp4\\n```\\n\\nTransfer motion from a reference clip onto a character image:\\n\\n```bash\\nvideodraft edit motion ./character.png \\\\\\n \\"Apply the dancer\'s movement to this character while preserving identity\\" \\\\\\n --motion-video ./dance-reference.mp4 \\\\\\n --model kling-v3-motion-control --quality pro \\\\\\n --download ./media/character-dance.mp4\\n```\\n\\n## 4. Hosted changelog video in CI\\n\\nIn a GitHub Action with `VIDEODRAFT_API_KEY` set as a secret:\\n\\n```bash\\nNOTES=$(git log --oneline v1.2.0..HEAD | head -20)\\nvideodraft create \\"Weekly product update video. Energetic, 20 seconds. Changes: ${NOTES}\\" --ar 16:9 --json > p.json\\nPROJECT=$(jq -r .project_id p.json)\\nvideodraft shots \\"$PROJECT\\" && videodraft produce \\"$PROJECT\\"\\nvideodraft export \\"$PROJECT\\" --download changelog.mp4 --wait-timeout 30m\\n```\\n\\n## 5. Variations and picking a winner\\n\\n```bash\\nvideodraft generate image \\"logo concept: minimalist fox, geometric\\" --num 4 --download \\"./concepts/{job_id}_{index}.{ext}\\" --json\\n# Show all 4 to the user; regenerate the chosen one at higher res:\\nvideodraft generate image \\"<same prompt>\\" --model nano-banana-pro --resolution 4K\\n```\\n\\n## 6. Reaching tools without a curated command\\n\\n```bash\\nvideodraft tools list --json | jq -r \'.[].name\'\\nvideodraft tools schema attach_media_to_shot --json\\nvideodraft call attach_media_to_shot --args \'{\\"project_id\\":\\"...\\",\\"scene_index\\":0,\\"shot_index\\":1,\\"media_url\\":\\"https://...\\",\\"media_type\\":\\"video\\",\\"duration_seconds\\":6}\'\\n```\\n\\nAnything the hosted VideoDraft MCP exposes, including character studio, product studio, and hosted project data, is reachable this way even before it gets a curated command. Native `.vdproject` editing uses the separate `videodraft_editor` MCP described in SKILL.md.\\n\\n## 7. Enhance an existing asset without changing it\\n\\n```bash\\n# Light image cleanup, no enlargement\\nvideodraft upscale image ./poster.png --scale 1x --download ./media/poster-enhanced.png\\n\\n# General image and video enlargement\\nvideodraft upscale image ./frame.png --scale 2x --download ./media/frame-2x.png\\nvideodraft upscale video ./clip.mp4 --resolution 1080p --download ./media/clip-1080p.mp4\\n\\n# Rebuild detail in a tiny/blurry source (generative), or reimagine it (creative)\\nvideodraft upscale image ./tiny.jpg --scale 4x --mode generative --model \\"Wonder 3.5\\" --download ./media/tiny-4x.png\\nvideodraft upscale video ./ai-clip.mp4 --resolution 4k --mode generative --download ./media/ai-clip-4k.mp4\\n\\n# Smoother motion or slow motion (resolution unchanged)\\nvideodraft interpolate ./clip.mp4 --fps 60 --download ./media/clip-60fps.mp4\\nvideodraft interpolate ./clip.mp4 --model Chronos --slowdown 4 --download ./media/clip-slowmo.mp4\\n```\\n\\nUse these when the content is correct and only quality or resolution needs improvement. If the poster text, composition, subject, or motion is wrong, edit or regenerate instead.\\n","references/models.md":"# Choosing models (and predicting cost)\\n\\nAlways consult the live catalog instead of memorizing this page \u2014 models change weekly:\\n\\n```bash\\nvideodraft models image --json # every image model + inputs (aspect ratios, resolutions, max refs)\\nvideodraft models video --json # every video model + inputs + per-second pricing metadata\\nvideodraft models audio --json # standalone audio/media models + pricing inputs\\nvideodraft models voices --json # TTS voices\\nvideodraft models styles --json # visual style presets\\n```\\n\\n## Task-based model selection\\n\\nFollow this order every time:\\n\\n1. The user named a model: use it, never substitute.\\n2. Otherwise pick a **Tier 1** model from the task\'s inputs, duration, audio, quality, speed, and cost, and pass it explicitly instead of relying on a blind platform fallback.\\n3. Use a **Tier 2** model only when no Tier 1 model supports the request, and say why. Never pick Tier 2 on your own for price, speed, novelty or variety.\\n\\nTier 1 is the rows marked **T1** below. Every other catalog model is Tier 2 and stays available by name. The catalogs carry this themselves: every `videodraft models image|video|audio --json` entry has `tier` (1 or 2) plus `recommended` / `recommended_for`, and the top-level `recommended` array is Tier 1, best first. Trust the catalog over this page when they disagree.\\n\\n### Images\\n\\n| Need | Choose | Why |\\n| ------------------------------------------------------------------------------------------ | ------------------------ | ----------------------------------------------------------------- |\\n| T1: Most generation, editing, character consistency, or reference work | `nano-banana-2` | Best general default; 1K/2K/4K and up to 14 reference images |\\n| T1: Highest-quality complex generation or reasoning | `nano-banana-pro` | Premium Nano Banana quality and reasoning |\\n| T1: Fast, inexpensive drafts and iteration | `nano-banana-2-lite` | Fastest/cheapest Nano Banana option; 1K only, up to 14 references |\\n| T1: Posters, title cards, signs, logos, or any image with important readable text | `gpt-image-2.5-flare` | Fast OpenAI option; 16 references, 1K/2K/4K and PNG output |\\n| T1: Complex multi-image composition, precise editing, or a strong alternate interpretation | `gpt-image-2.5-sunburst` | More fidelity/detail; 16 references, quality through max |\\n| T1: Vector / SVG output (logos, icons, illustrations that must scale) | `recraft-v4` | Recraft V4.1 text-to-vector; SVG only, no reference images |\\n| T2: By name, or an explicitly requested xAI look | `grok-imagine` | 2 cr flat (3 with a reference); 1 reference image |\\n| T2: xAI at 2K or with a quality tier, up to 3 edit references | `grok-imagine-2.0` | 1K 4 (low) / 6 (medium), 2K 6 / 8, +1 cr per reference image |\\n\\nGPT Image 2.5 has two IDs: `gpt-image-2.5-flare` replaces the former GPT Image 2 recommendation; `gpt-image-2.5-sunburst` is the precision/detail alternative. Other model recommendations are unchanged. Explicit `gpt-image-2` selections continue to use the previous model.\\n\\nUse the same basic image controls as GPT Image 2: `--ref`, `--ar`, `--resolution 1K|2K|4K`, `--quality`, and `--num 1..4`. GPT Image 2.5 quality supports auto/low/medium/high/xhigh/max. Auto is priced at Max. Generation runs directly on OpenAI, or exclusively on the user\'s Fal key when active.\\n\\n```bash\\nvideodraft generate image \'A blue glass bird\' --model gpt-image-2.5-flare --ar 1:1 --resolution 2K --quality high --estimate\\nvideodraft generate image \'A blue glass bird\' --model gpt-image-2.5-flare --ar 1:1 --resolution 2K --quality high --num 2\\n```\\n\\n`videodraft shots` forwards resolution and quality for both normal and grid generation. Estimates use the project\'s model and aspect when omitted; grid canvas costs may differ from ordinary shot costs. Use `videodraft costs --model <id> --ar <ratio> --resolution <tier> --quality <tier>` for a specific per-image quote.\\n\\n### Videos\\n\\n| Need | Choose | Important limits |\\n| -------------------------------------------------------------------------------------------------------------- | ------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |\\n| T1: Most generation, first/last-frame, mixed-reference, source-edit, or extension requests | `gemini-omni-1.1-flash` | 3-10s output; 360p/720p/1080p/4K at 3/10/15/30 cr/s; audio always; up to 10 image inputs, and 3 reference videos <=3s on generate only; edit source <=10s; continuation or 3-10s extension of a 1-30s source, 40s total reachable only by extending from <=30s |\\n| T2: Grok 1.5 text, first-frame, or 1-7 image-reference clips with native audio and optional 1080p | `grok-imagine-video-1.5` | 1-15s; 480p/720p/1080p for text/first-frame; references are 480p/720p only; no last frame |\\n| T2: Document or webpage reference generation; by name otherwise | `wan-3.0` | 2-30s or auto; 480p/720p/1080p at 7/14/28 cr/s; 10 image, 5 video, 5 audio refs, 20 media files total; document/web refs require thinking |\\n| T2: 480p/768p/2K/4K with native stereo audio, first/last frames, or mixed image/video/audio references | `minimax-h3` | 5-15s; 5/6/13/16 cr/s by resolution; up to 9 image, 3 video, 3 audio refs, 12 files total; first 5 reference images free then 8 cr each; reference video/audio each total <=15s |\\n| T1: Text, first/last-frame, or mixed-reference video with native audio and stronger prompt adherence | `minimax-h3-max` | 5-15s; 5/8 cr/s at 480p/768p plus pooled reference tokens; up to 9 image, 3 video, 3 audio refs, 12 files total; seed and disabled/balanced/quality prompt expansion; provider safety checker off by default |\\n| T2: Images pinned to specific moments (keyframes); by name otherwise | `flux-3` | 5-20s (auto for text/first-frame only); 720p/1080p; up to 10 keyframes; `--quality draft` is 720p-only at ~1/3 the cost |\\n| T1: Video/audio references, mixed reference media, broad aspect ratios, frame-mode first+last frame, or 11-15s | `seedance-2` | 4-15s or auto; up to 9 image, 3 video, and 3 audio refs; audio toggle; Mini/Fast are 480p/720p only |\\n| T1: Single takes past 15s, or more references than Seedance 2.0 allows | `seedance-2.5` | 4-30s or auto; up to 30 image, 10 video, and 10 audio refs (50 files total); one quality tier; 480p/720p/1080p |\\n| T2: Fast polished 3-15s video with first frame, multi-prompt, and native audio | `kling-v3-turbo` | Audio always on; Pro default; no end frame, reference-media mode, or elements |\\n| T1: Cinematic 3-15s with reference images, image/video elements, first+last frame, audio, or 4K | `kling-o3` | 7 combined image refs/elements, reduced to 4 with a video element; Standard/Pro/4K |\\n| T1: Kling 3-15s image-to-video with first+last frame, image/video elements, multi-prompt, audio, or 4K | `kling-3.0` | Elements require a start image; image or video elements can bind voice_id; Standard/Pro/4K |\\n| T2: User explicitly requests Veo, or the selected workflow specifically needs Veo | `google-veo3.1` | Good fallback, but not the preferred general model |\\n\\nRouting rules:\\n\\n- Inexpensive pass or draft at 480p/768p with native audio: use MiniMax H3 Max (5/8 cr/s plus reference tokens).\\n- Grok 1.5 reference mode accepts 1-7 images only. Address them in array order as `<IMAGE_0>` through `<IMAGE_6>`. Do not combine reference images with `--start-image`, `--end-image`, `--ref-video`, or `--ref-audio`.\\n- Grok 1.5 first-frame mode accepts one `--start-image`, derives the output aspect ratio from that image, and does not support `--end-image`. Text and first-frame modes support 480p, 720p, or 1080p. Reference mode supports 480p or 720p.\\n- Grok 1.5 always generates native audio. Do not pass `--no-audio`, `--seed`, `--negative`, or `--quality`.\\n- Around 11-15 seconds with native audio: use Seedance, Kling 3.0 / O3, or MiniMax H3 Max, not Gemini. Past 15 seconds: Seedance 2.5.\\n- Dialogue: write the line in the prompt, in quotes, with who says it and how. `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` and `kling-o3` speak it natively; add `--ref-audio` on Seedance or a bound Kling voice for a specific voice. Do not generate speech separately and lip-sync it on.\\n- One existing source video that should be edited: use `videodraft edit video`, which auto-selects Gemini Omni 1.1 Flash for a source up to 10s. Generic `generate video --ref-video` retains source-edit behavior when `--video-task` is omitted, but the dedicated edit command is clearer.\\n- Video supplied as a creative reference: Gemini Omni 1.1 Flash accepts up to 3 videos of at most 3 seconds each, including mixed image and video input. Creative video references are accepted on `--video-task generate` ONLY: edit and extend take exactly one input video, so `--ref-video` cannot accompany `--source-video` or `--previous-interaction-id`. Use Seedance 2.5 for more or longer video/audio references, or Seedance 2.0 when quality-tier control matters.\\n- First and last frame control: every Tier 1 video model supports it (Gemini Omni 1.1 Flash, Seedance, Kling O3, Kling 3.0, MiniMax H3 Max). For Gemini, every start/end frame counts toward the 10-image total, leaving up to 9 references with a start frame or 8 with both frames.\\n- Extension with a 1-30s uploaded source uses Gemini Omni 1.1 Flash plus `--source-video` and `--video-task extend` (or `--extend`). Pass an explicit 3-10s output duration; the model appends it at the end, up to 40 seconds total. Creative `--ref-video` clips CANNOT accompany the source: edit and extend take exactly one input video. New dialogue is allowed only when the source video is silent; adding speech on top of a source that already has speech is refused with \\"the model is currently unable to process speech edits\\". Continuation uses `--previous-interaction-id <id>` and defaults to a conversational edit; `--video-task extend` lengthens it. A continuation is submitted as the prior turn\'s output video, so it obeys the same source limits (<=30s to extend, <=10s to edit) and the same silent-source dialogue rule. 40s total is reachable only by extending from a source of 30s or less, so a ladder dead-ends past 30s. Fal returns interaction IDs, but its currently published callable v1.1 endpoints do not expose continuation/extension or mixed source-edit references; never infer a route or silently fall back to paid Google.\\n- Wan 3.0 reference mode and frame mode are separate. It accepts 10 images, 5 videos, and 5 audio clips, at most 20 media files total. Video and audio each total at most 15 seconds. `--file-url` and `--web-url` require `--thinking`. `--auto-duration` cannot be combined with `--duration`.\\n- MiniMax H3 and H3 Max reference mode and first-plus-last-frame mode are separate. Audio cannot be the only reference. Address references as `Image 1`, `Video 1`, and `Audio 1` in array order.\\n- Seedance reference mode and first-plus-last-frame mode are separate. Do not promise reference video/audio plus a last frame in one generation.\\n- Multi-prompt sequencing: use Kling 3.0 Turbo, Kling O3, or Kling 3.0.\\n- Kling 3.0 image-to-video and Kling O3 reference-to-video accept repeatable structured `--element` JSON. Each element is image-backed (`frontal_image_url` plus 1-3 `reference_image_urls`) or video-backed (`video_url`). Either form may include `voice_id`. Keep the voice ID inside the same element so the association is preserved. Reference them as `@Element1`, `@Element2`. Kling V3 Turbo rejects elements.\\n- Kling 2.6 Pro image-to-video accepts one or two repeatable `--voice-id` values. Cite them as `<<<voice_1>>>` and `<<<voice_2>>>` in the prompt. Voice control costs 17 cr/s. Use `videodraft kling-voices list|create|delete` to manage the separate Kling video-control voice library. Creating a voice requires a clean 5-30 second, up-to-50MB single-speaker `.mp3`, `.wav`, `.mp4`, or `.mov` sample plus `--confirm-consent`. Creation costs 1 VideoDraft credit on the platform Fal account and 0 credits with Fal BYOK; use the create command\'s `--estimate` flag to check without creating.\\n- Kling V3 voice control costs 16 cr/s for Standard and 20 cr/s for Pro.\\n- Seedance quality: `mini` for the lowest cost, `fast` for speed, `standard` for maximum quality and for 1080p/4K. Seedance 2.5 has a single tier and ignores `--quality`.\\n- Longer than 15 seconds, or more than 9 image / 3 video / 3 audio references: use Seedance 2.5. It reaches 30s and 30/10/10 references (50 files total) at 480p/720p/1080p, but has no 4K.\\n- Seedance 2.x allows real people by default, the same as AI Studio: Byteplus first, a submit-time Fal fallback, and Fal\'s higher tier-specific rate. Turn it off with MCP `allow_real_people: false` or CLI `--no-allow-real-people` for the lower Byteplus-only rate, only when the user wants the cheapest run and nothing in the job is a real identifiable person (text-only requests, non-people, anime, clearly synthetic or stylized characters).\\n- If a Seedance request made with the option off fails with `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, estimate the higher rate, follow the user\'s spend policy, and retry once with the option on. The response also carries `retry_with: { allow_real_people: true }`, `cli_flag: \\"--allow-real-people\\"`, and `retry_policy: \\"once\\"`. If the option was already on, do not repeat the same request. Byteplus may accept a task and reject the output later. VideoDraft refunds that failed generation, but it cannot reroute the asynchronous failure to Fal. Rephrase or change the references instead.\\n\\n### Video edit and motion-control categories\\n\\nUse `videodraft models video --category video_edit` for existing-video transforms and `--category motion_control` for motion transfer.\\n\\n| Need | Command/model | Important limits |\\n| ------------------------------------------ | ---------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------- |\\n| Most edits of a source up to 10s (DEFAULT) | `videodraft edit video <video> \\"...\\"` (auto-selects `gemini-omni-1.1-flash`) | Source <=10s or it REFUSES; up to 10 image refs and 3 creative video refs <=3s each; 360p/720p/1080p/4K; audio always regenerated (no `--preserve-audio`) |\\n| Cheapest simple prompt edit | `videodraft edit video <video> \\"...\\" --model grok-imagine-video-edit` | No image refs; source silently truncated to 8s; auto/480p/720p |\\n| Edit with several image references | `--model happy-horse-video-edit --ref ...` | Up to 5 refs; 720p/1080p; source silently truncated to 15s; highest rate (28-56 cr/s) |\\n| Controlled Kling edit | `--model kling-o3-video-ref-edit --ref ...` | Up to 4 refs; Standard/Pro; source silently clamped to 3-10s |\\n| Transfer reference motion to an image | `videodraft edit motion <image> [direction] --motion-video <video>` | Prompt/direction is optional; Kling V3 default; optional one image-only `--element`; element requires video orientation |\\n\\nIf the user explicitly names one of these models, preserve it. The CLI uploads local source videos and reference images automatically. Editing returns an async job and waits by default.\\n\\n**Omit `--model` and the SERVER chooses**, because only it can measure the source. It picks `gemini-omni-1.1-flash` for any source up to 10s. When that is not safe (source longer than 10s, unmeasurable duration, `--preserve-audio`, more than 10 refs, or Fal BYOK with refs) it spends nothing and returns a priced menu: per model, the seconds it would edit, the seconds it would drop, and the credit cost. The CLI prints that table and exits 2. Show it to the user, then re-run with `--model`.\\n\\n**Truncation is the trap here.** Only Gemini refuses a source it cannot fully consume. Every other edit model accepts a 30s clip and returns an edit of its first 8-15 seconds with no error. `videodraft edit video` now warns when this will happen and the tool response carries `source.truncated` / `source.dropped_seconds`; relay it to the user rather than letting them discover it in the output.\\n\\nTo edit a longer source without losing its tail, cut it into <=10s pieces with `videodraft_editor`, edit each with Gemini, then reassemble and export there. VideoDraft has no server-side split/concat, so this needs the native editor.\\n\\nKling O3 is also exposed for reference generation. `videodraft generate video --model kling-o3-video-ref-edit` requires exactly one `--ref-video` and generates a new reference-guided clip. Wan 3.0 handles new text, frame, and mixed-reference generation, but is not an existing-source edit model.\\n\\n### Reference-first video workflow\\n\\n- Prefer a start frame or reference image whenever a specific character, product, location, style, composition, or brand identity must stay recognizable.\\n- If the user gives a reference, pass it. Never silently replace it with a text description.\\n- If no reference exists and continuity matters, generate a still first with the user\'s explicitly requested compatible image model, otherwise use Nano Banana 2. Wait for the image URL, then animate it with the selected video model. Confirm the combined image plus video cost before starting.\\n- For multi-shot scenes, generate shot images with `videodraft shots <project_id> --model <selected-image-model> --grid`. Preserve an explicitly requested compatible image model; otherwise use `nano-banana-2`. The grid establishes the scene and characters together, then decodes into individual shot images.\\n- Animate the decoded shot images as per-shot start frames or references. Do not independently text-generate each video clip when the shots need to match.\\n- Pure text-to-video remains appropriate for generic one-off footage where no subject, composition, or continuity needs to be preserved.\\n\\n### Audio\\n\\n- **Seed Audio 1.0**: use `videodraft generate audio` for open-ended speech, sound, music, or prompt-driven audio editing. It accepts up to three audio references or one image. Address audio references as `@Audio1`, `@Audio2`, and `@Audio3`. Preset and custom cloned voice IDs are supported. Output is up to 120 seconds. There is no requested-duration input. The CLI automatically retries transient responses with one idempotency key. To recover after the CLI process itself is interrupted, set `--idempotency-key <uuid>` on the original command and reuse it.\\n- **Voiceover/TTS**: prefer ElevenLabs. Brittney is the platform default voice; under ElevenLabs BYOK, use a compatible voice from the user\'s account. Honor another supported voice/provider when the user explicitly selects it.\\n- **Dialogue, voice changing, and dubbing**: ElevenLabs only.\\n- **Sound effects**: ElevenLabs Sound Effects only.\\n- **Music**: use `lyria-3.5` for all music, short or long (up to ~3 minutes), with vocals/lyrics or instrumental music. When `--model` is omitted the CLI sends no model and the server default applies, which is `lyria-3.5` wherever the backend supports it. If the server answers that `lyria-3.5` is an unknown model, that backend predates it: rerun without `--model` (the server default then applies), or use `lyria-3-pro-preview` for a track longer than 30 seconds. `lyria-3-clip-preview` (fixed ~30s) and `lyria-3-pro-preview` remain available for explicit requests. Lyria length and structure are prompt-guided, not exact; for an instrumental, include \\"instrumental only, no vocals\\" in the prompt. `--length` and `--instrumental` only apply to ElevenLabs. Use `elevenlabs-music-v2.5` when a specified 3-300 second length, composition plan, or style reference track matters. Lyria references: up to 10 images on Google, 1 on Fal BYOK; 3.5 Fal prompts are 1-5000 characters (an image-only request gets a neutral prompt). `elevenlabs-music` is an alias for v2.5. Use `elevenlabs-music-v1` only when the user asks for v1 by name; it takes a prompt, `--length` and `--instrumental` only.\\n- **ElevenLabs Music v2.5 inputs**: either a prompt (`--length`, `--instrumental`) or a composition plan, never both. A plan is 1-30 sections of 3-120 seconds each, up to 300 seconds in total, and `--plan`/`--section` pick v2.5 when `--model` is omitted. Empty section parts default to 20 seconds and an instrumental part. Build one with repeatable `--section \\"<seconds>|<style, style>|<text>\\"` (use `\\\\n` for line breaks) or pass `--plan plan.json` with `{\\"chunks\\":[{\\"text\\",\\"duration_ms\\",\\"positive_styles\\",\\"negative_styles\\",\\"context_adherence\\",\\"audio_reference\\"}]}`. Section text is an optional `[Section name]`, lyric lines, and `{inline directions}`. Put 6-7 specific English styles on the first section; it sets the genre. `--ref-audio <url|file>` adds a style reference clip to the first section (window up to 30 seconds via `--ref-start`/`--ref-end` in ms, `--ref-strength low|medium|high|xhigh`); local files are uploaded, including relative `audio_reference.audio_url` paths in a plan file. `--seed` works only with a plan. `--format` picks the output (`mp3_48000_192` default on v2.5; `pcm_*`, `ulaw_8000` and `alaw_8000` arrive as stereo WAV files). Style references don\'t run on the user\'s own ElevenLabs key; with that key connected, drop the reference or ask the user to turn the key off. The CLI retries transient responses with one idempotency key; set `--idempotency-key <uuid>` to recover after an interruption.\\n- Voice Changer and Dubbing require the source media duration and currently accept source media up to 300 seconds.\\n\\n**User-supplied ElevenLabs voice IDs:** voiceover, dialogue, and voice changing accept raw IDs (16-64 alphanumeric characters) or `elevenlabs-<id>`. `videodraft models voices` / MCP `list_available_voices` is for discovery, not an allowlist. Pass a supplied ID directly even when it is absent from the catalog; do not reject it or substitute a catalog voice. The voice must be accessible to the provider account used for generation. Private or cloned voices may require the user\'s connected ElevenLabs key. A failed voice listing does not prove that a supplied ID cannot generate; a connected key with generation access may still work. If the provider rejects generation access, report that error rather than silently changing the voice.\\n\\nUse `--voice <id>` for voiceover and voice changing, and repeat `--line \\"<id>:Text\\"` for dialogue. These examples show both ID forms; replace the sample IDs with the user\'s supplied IDs:\\n\\n```bash\\nvideodraft generate voiceover \\"Hello there.\\" --voice kPzsL2i3teMYv0FxEYQ6 --download ./media/voiceover.mp3\\nvideodraft generate dialogue --line \\"kPzsL2i3teMYv0FxEYQ6:Hello.\\" --line \\"elevenlabs-kmSVBPu7loj4ayNinwWM:Welcome back.\\" --download ./media/dialogue.mp3\\nvideodraft generate voice-changer ./media/source.wav --voice elevenlabs-kPzsL2i3teMYv0FxEYQ6 --duration 12 --download ./media/changed-voice.mp3\\n```\\n\\nFor MCP, use `generate_voiceover.voice_id`, `generate_dialogue.lines[].voice_id`, or `change_voice.voice_id` with either form. ElevenLabs IDs are separate from Kling video-control voice IDs and MiniMax `custom-*` cloned-voice IDs. Do not convert IDs from those systems into ElevenLabs IDs.\\n\\n### Avatar / talking head\\n\\n**First decide the framing, not just \\"someone talks\\".** This lane animates a PORTRAIT facing the lens. A character delivering a line inside a real scene, with blocking, framing or camera movement, belongs to `generate video` instead: write the dialogue in the prompt and `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` or `kling-o3` voices it natively (add a Seedance `--ref-audio` clip or a Kling voice bound per element for a specific voice). That is the default for any talking character; never plan silent video + TTS + lip-sync. Sending a cinematic shot to Fabric returns a head-on talking headshot, not the shot that was asked for. `videodraft models video --json` lists these under `recommended.in_scene_dialogue`.\\n\\nFor a presenter, spokesperson or explainer speaking to camera, VEED Fabric is the preferred model. Choose the dedicated path from the media the user already has:\\n\\n| Starting media | Command | Use |\\n| ------------------------------------------ | ------------------------------------------------------------------------- | ---------------------------------------------------------- |\\n| Portrait + script, reusable avatar record | `videodraft avatar create <portrait> --script \\"...\\"` then `avatar render` | Managed avatar flow with bundled speech preparation |\\n| Portrait + text | `videodraft avatar fabric <portrait> --text \\"...\\"` | One-off direct VEED Fabric text mode |\\n| Portrait + existing audio | `videodraft avatar fabric <portrait> --audio <audio>` | One-off direct VEED Fabric audio lip sync |\\n| Portrait + existing audio, short and sharp | `videodraft avatar h3-lipsync <portrait> --audio <audio>` | MiniMax H3 Max Lip Sync, 5-14.8s at up to 2K |\\n| Existing video + existing audio | `videodraft avatar lipsync <video> --audio <audio>` | Sync Labs Lipsync 2 (Tier 2): re-dub existing footage only |\\n\\nThe managed renderer is VEED Fabric Fast (`veed/fabric-1.0/fast`). Direct Fabric, MiniMax H3 Max Lip Sync, and Sync Labs are paid AI Studio generations and return async job IDs.\\n\\n1. Obtain the avatar image. Prefer the user\'s supplied portrait or an existing character. If none exists, use the user\'s explicitly requested compatible image model, otherwise generate a front-facing head-and-shoulders portrait with `nano-banana-2`, direct eye contact, a natural expression, and a clean background. Match the intended video aspect ratio when practical.\\n2. If the portrait is visibly soft or too small, run Topaz image enhancement/upscaling before animation.\\n3. Generate a script only if needed: `videodraft avatar script \\"<idea>\\"`.\\n4. Create the avatar record and speech: `videodraft avatar create <portrait-url-or-file> --script \\"...\\" --voice <id> --ar 9:16`. Prefer ElevenLabs when unspecified, but honor another explicitly selected supported voice/provider.\\n5. Render with VEED Fabric: `videodraft avatar render <avatar_video_id> --resolution 720p`.\\n\\nThe portrait is passed as the avatar\'s character image, not as a generic video\'s start frame. Prefer rendering directly at 720p. Use 480p only when the user prioritizes lower cost. Avatar script generation and `avatar create` (including speech) are bundled/free. Confirm the Fabric render cost, plus portrait generation or upscaling when needed.\\n\\nDirect Fabric text/audio, H3 Max Lip Sync, and Sync Labs do not use the managed avatar record. The CLI uploads local portrait, video, and audio files automatically. `avatar fabric --speed fast` applies only to audio mode. `avatar h3-lipsync` takes no prompt: pick `--resolution 480P|768P|1080P|2K` (default 768P), an optional `--seed`, `--no-transcription` to sync without transcribing the audio (transcription is on by default), and `--safety-checker` only when the user asks for Fal\'s checker. The portrait\'s width / height must be 0.4-2.5, and audio shorter than 5 seconds is refused before any charge. Sync costs 5 credits per verified audio second; under Fal BYOK, `sync_mode` remains available but `temperature` and `active_speaker` are ignored by the provider.\\n\\n### Upscaling / enhancement\\n\\n- **Images**: Topaz via `videodraft upscale image <url-or-file> --scale 1x|2x|4x [--mode precision|generative|creative] [--model <name>]`. Modes: `generative` (default; Wonder 3.5, Topaz\'s recommended model for AI-generated images; `Redefine` takes `--prompt`, `--creativity 1-6`, `--texture 1-5`), `precision` (faithful and cheapest; Standard V2, High Fidelity V3, Low Resolution V2, CGI, Text Refine; use for clean real photos), `creative` (Bloom 2; artistic, `--prompt`, `--creativity 1-9`). Extra knobs: `--no-face-enhance`, `--face-strength`, `--sharpen`, `--denoise`, `--fix-compression`, `--format png`. Use 1x for light enhancement without enlargement, 2x as the general default, and 4x only when the source quality and target size justify it. The server must verify dimensions from a readable image of at most 50 MB; `--width` and `--height` are compatibility hints and cannot bypass a failed probe. The result is synchronous. Cost: 8 credits per started 24 MP of output in precision, per 8 MP (Wonder 3/3.5) or 4 MP in generative, per 2 MP in creative.\\n- **Videos**: Topaz via `videodraft upscale video <url-or-file> --resolution 720p|1080p|4k` (preferred) or `--scale 2x`, plus `--mode precision|generative|creative` and `--model <name>`. `generative` (default; Starlight Precise 2.6, Topaz\'s recommended model for AI-generated footage; Starlight Fast 2 at half price) costs 12 credits/s up to 1080p and 26 at 4K. `precision` (Proteus, Proteus Natural, Iris, Gaia 2 for animation, Rhea, Theia, Artemis, Dione) is 6x cheaper at 1 / 2 / 6 credits per second for 720p / 1080p / 4K output; use it for real footage or when cost matters. `creative` (Astra 2, `--prompt`, `--creativity`, `--realism`, `--sharp`) always renders 4K at 50 credits/s. `--fps 60` delivers 60fps on the same pass; any output above 30fps (including a 50/60fps source) doubles every rate; the server must verify duration, dimensions, and frame rate from an MP4/MOV source of at most 100 MB. `--duration`, `--width`, `--height`, and `--source-fps` are compatibility hints and cannot override billing or bypass a failed probe. The same source-verification requirement applies to frame interpolation. `--scale` also accepts intermediate factors such as `1.5x`. Max source length 5 minutes. The job is asynchronous; the CLI waits by default, while MCP callers poll `check_generation_status`. MCP video input must be VideoDraft-hosted, so upload local or external sources first.\\n- **Frame interpolation / slow motion**: `videodraft interpolate <url-or-file> --fps 60 [--model Apollo|Chronos|Aion] [--slowdown 1-8]` (MCP `interpolate_video`). Resolution is unchanged. Apollo (default) for smooth general conversion, Chronos for natural slow motion, Aion for extreme slow motion. Apollo/Chronos cost 3 credits per output second up to 1080p (6 at 4K); Aion 5 / 17. These rates cover targets up to 60fps; above 60fps multiply by target FPS / 60 (120fps doubles the rate). Output seconds = source seconds \xD7 slowdown. The final charge rounds up to a whole credit.\\n- Use upscaling to preserve the image/video while improving detail, resolution, or cleanup. It cannot fix the wrong subject, misspelled text, bad framing, unwanted objects, broken continuity, or incorrect motion. Use an edit or regeneration for those problems. Keep `generative` for AI-generated sources; switch to `precision` for clean real photos/footage or a cheap pass, and `creative` only when the user wants an artistic reinterpretation.\\n- For a new Fabric avatar, render directly at 720p instead of rendering at 480p and then upscaling. Upscale the source portrait first only when the portrait itself is low quality.\\n\\n## Capability gotchas\\n\\n- Each model\'s `inputs` block is authoritative: supported `aspect_ratios`, `resolutions`, `quality_options`, `start_frame`/`end_frame`, `max_reference_images/videos/audio`, `multi_prompt`, `audio_toggle`. Passing an unsupported input fails with a clear error \u2014 check first, don\'t trial-and-error paid calls.\\n- Most video models support only 16:9 / 9:16 / 1:1. A 3:4 request hard-fails on most.\\n- `--seed` reproduces a specific output on models that support it (e.g. Flux, Ideogram V4). GPT Image 2.5 rejects any explicit seed; older models may ignore unsupported seeds. You do not need a seed for variation \u2014 `--num` already varies.\\n- `--rendering-speed` applies to Ideogram (V3: `Default`/`Turbo`/`Quality`; V4: `Turbo`/`Balanced`/`Quality`) and affects image cost \u2014 pass it to `videodraft costs ... --rendering-speed <tier>` for an accurate estimate. Always trust `videodraft models image --json` over this list; new models and tiers appear there the moment the platform ships them, with no CLI update.\\n- `seedream-v5-pro` supports unified text-to-image and reference-image editing with up to 10 image references. Use `--resolution 1K` for 7 credits/image or `--resolution 2K` for 14 credits/image.\\n- Reference inputs: `--ref <img>` (images, including up to 10 total for Gemini Omni 1.1 Flash and Wan 3.0, and 7 for Grok 1.5), `--source-video <v>` (Gemini uploaded edit/extension source), `--ref-video <v>` (up to 3 creative videos <=3s each for Gemini Omni 1.1 Flash; also Wan 3.0, MiniMax H3, MiniMax H3 Max, and Seedance 2), `--ref-audio <a>` (Wan 3.0, MiniMax H3, MiniMax H3 Max, Seedance 2), and `--element \'<json>\'` or `--element @elements.json` for Kling V3/O3. For an exact Seedance 2.x reference-video `--estimate`, add `--ref-video-seconds <combined-seconds>`; Seedance bills input seconds alongside output. Wan 3.0 and MiniMax H3 do not bill input reference seconds; MiniMax H3 Max bills them as pooled reference tokens. The CLI uploads local media references and every structured element without flattening the source/reference roles. `--segment \\"<prompt>:<seconds>\\"` (repeatable) drives Kling 3.0, Kling 3.0 Turbo, and O3 multi-prompt generation. Use 1-6 segments of 1-15 whole seconds each, with 3-15 seconds total. `generate image --video-ref` is the nano-banana-2 video reference.\\n- The top-level prompt is OPTIONAL for Gemini Omni 1.1 Flash media-input or continuation calls, Wan 3.0 frame/reference modes, `generate video` with multi-prompt models, and Kling 3.0 Turbo (`--model kling-v3-turbo`) image-to-video. Text-only Gemini and Wan calls still require a prompt unless their live schema says otherwise.\\n- Hosted AI Production fallback: `videodraft produce <project> --mode full_video` generates one Seedance 2 video per scene, with real people allowed by default at the higher Fal-tier rate on every submitted scene segment; add `--no-allow-real-people` for the lower Byteplus-only rate when no scene shows a real identifiable person. The MCP equivalent is `produce_project` with `mode: \\"full_video\\"` (and `allow_real_people: false` to opt out). If a partial opted-out run returns `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, rerun the same project once with an explicit `--allow-real-people` after cost confirmation. The server reconciles asynchronous results first, preserves running/completed jobs, and retries only failed placeholders carrying that exact code. Poll with `videodraft generations`, then `videodraft finalize <project>` swaps them into the hosted timeline before `export`. In VideoDraft ADE, do not choose this path while `videodraft_editor` is available unless the user explicitly requests hosted production. Generate or download the scene assets, import them, and assemble/export with the native editor instead. If the user explicitly requests another compatible video model for a hosted production, do not use this fixed Seedance path; generate the project shots manually with the requested model and attach them to the hosted timeline.\\n\\n## Cost model\\n\\n- Images: per image (\xD7 `--num`). Matrix-priced models (GPT-Image, Nano Banana Pro, Seedream v5 Pro) vary by resolution/quality.\\n- Video: usually credits/second \xD7 duration; rate depends on model + resolution + quality + native audio on/off.\\n- Gemini Omni 1.1 Flash: 3 / 10 / 15 / 30 credits per output second at 360p / 720p / 1080p / 4K (720p default). Extend appends an explicit 3-10 seconds to a 1-30s source, whether uploaded or resolved from a prior interaction; 40 seconds total is reachable only while the source stays at or under 30s. Fal BYOK generation and basic source edits cost zero VideoDraft credits; continuation, extension, and separate creative references on a source edit are unavailable under Fal BYOK.\\n- MiniMax H3: 5 / 6 / 13 / 16 credits per output second at 480p / 768p / 2K / 4K (768p default). The first 5 reference images are included, then 8 credits for each additional image. Reference video and reference audio are NOT billed.\\n- MiniMax H3 Max: 5 credits per output second at 480p or 8 credits per second at 768p (default), plus pooled reference tokens. The first 4,096 reference tokens are free and each additional 1,000 costs 2 credits. An image is `(width x height) / 1024` tokens (a 1024x1024 reference is 1,024, so four of them are free); a reference video is 2,886 tokens per second at 480p or 7,459 at 768p; reference audio is about 2,121 per second. Pass `--ref-audio-seconds` for an exact estimate with audio references. The server measures each reference image, so a quote that assumes 1024x1024 is a floor.\\n- Wan 3.0: 7 / 14 / 28 credits per output second at 480p / 720p / 1080p. Auto duration reserves 30 seconds and reconciles unused credits from the provider-reported output duration. Fal BYOK charges zero VideoDraft credits.\\n- Grok Imagine Video 1.5: 8 credits/output second at 480p, 14 at 720p, or 25 at 1080p, plus 1 credit for each first-frame or reference image. Native generated audio is part of every output.\\n- Kling voice control: 17 credits/output second for Kling 2.6 Pro, 16 at Kling V3 Standard, and 20 at Kling V3 Pro.\\n- Shot-image batches: one image per shot (+1 grid image per scene in `--grid` mode) \u2014 the largest single spend in the pipeline.\\n- VEED Fabric avatar renders: ~10 credits/sec at 480p, ~20/sec at 720p. Avatar creation and its speech are bundled/free; only optional portrait generation/upscaling adds cost before the render.\\n- Direct VEED Fabric: text or normal audio is 8 credits/sec at 480p and 15/sec at 720p; fast audio is 10/sec at 480p and 20/sec at 720p.\\n- MiniMax H3 Max Lip Sync: 5 / 8 / 16 / 32 credits per output second at 480P / 768P (default) / 1080P / 2K. The video runs as long as the audio (at least 5s; only the first 14.8s is used), billed on the server-measured length rounded up, so at most 15 seconds. Use MP3, WAV, M4A (AAC), or AAC audio; other formats run only on the user\'s own Fal key.\\n- Sync Labs Lipsync 2: 5 credits per verified audio second.\\n- Voiceover TTS: 10 credits per 1000 characters for standard voices, 30 per 1000 for cloned `custom-*` voices (min 1, pro-rated); applies to standalone voiceovers AND per-scene narration during `produce`. Silent tracks are free. Voice cloning itself is a flat 150 credits per clone.\\n- Lyria music: flat per track, 4 credits (Clip) / 10 credits (3.5) / 8 credits (legacy Pro). Fal BYOK uses zero VideoDraft credits.\\n- Seed Audio 1.0: 19 credits per actual output minute, prorated and rounded up to a whole credit. VideoDraft reserves the 120-second maximum of 38 credits and refunds the unused portion after generation. Fal BYOK is free.\\n- ElevenLabs audio: sound effects are per second, dialogue is per character, music/voice-changer/dubbing are per started minute. ElevenLabs Music v2.5 and v1 both cost 60 credits per started output minute; estimate a composition plan with its total length. Voice changer and dubbing reject source media above 300s in the current synchronous flow.\\n- Seedance 2.0 / 2.5 real people: allowed by default, which keeps Byteplus first, permits a submit-time Fal fallback, and prices at Fal\'s rate for that tier: 2.0 Mini 8/16 cr/s, Fast 11/25, Standard 14/31/69/156, 2.5 23/48/114 for 480p/720p/1080p. `--no-allow-real-people` uses the Byteplus-priced path instead (2.0 Mini 4/8, Fast 6/13, Standard 7/16/38/78, 2.5 11/24/57); Byteplus refuses real-person likenesses, so a likeness-policy failure then does not fall back. The gap is roughly 2x but not exactly: the Seedance 2.0 1080p pair is 38/69, or about 1.82x. If Byteplus accepts the task and later rejects the generated output, VideoDraft refunds the failure but does not resubmit it to Fal. Turn the option off only for the cheapest run when nothing in the job is a real identifiable person.\\n- Grok Imagine images: `grok-imagine` is a flat 2 cr (3 with a reference). `grok-imagine-2.0` is a separate, newer model priced by resolution and quality: 1K 4 (low) / 6 (medium), 2K 6 / 8, plus 1 cr per reference image (up to 3). v1 is NOT superseded \u2014 pick it when cost matters more than 2K.\\n- xAI bills refused requests, so failed Grok generations are not refunded.\\n- Upscales: priced by output size, mode, and (video) duration/fps; see the Upscaling section above.\\n\\nQuote before spending:\\n\\n```bash\\nvideodraft costs gemini-omni-1.1-flash --type video --duration 8 --resolution 720p --audio\\nvideodraft costs topaz-upscale-video --type video --duration 10 --width 1920 --height 1080 --resolution 4k --mode generative\\nvideodraft costs topaz-upscale --type image --width 2048 --height 2048 --scale 2x\\nvideodraft costs topaz-interpolate-video --type video --duration 10 --width 1920 --height 1080 --fps 60 --topaz-model Apollo\\nvideodraft costs minimax-h3 --type video --duration 10 --resolution 2K --ref-images 7\\nvideodraft costs minimax-h3-max --type video --duration 10 --resolution 768p --ref-images 2\\nvideodraft costs grok-imagine-video-1.5 --type video --duration 8 --resolution 720p --ref-images 4\\nvideodraft costs seedance-2 --type video --duration 15 --resolution 720p --quality standard --audio\\nvideodraft costs grok-imagine-2.0 --type image --resolution 2K --quality medium --num 2\\nvideodraft costs seedance-2 --type video --duration 15 --resolution 720p --quality standard --audio --no-allow-real-people # lower Byteplus-only rate\\nvideodraft costs elevenlabs-dubbing --type audio --duration 60\\nvideodraft costs seed-audio-1.0 --type audio --duration 60 # scenario only; model controls actual length\\nvideodraft costs elevenlabs-dialogue --type audio --chars 350\\nvideodraft costs voiceover --type audio --chars 800 # TTS: 10 cr / 1000 chars\\nvideodraft generate video \\"...\\" --model gemini-omni-1.1-flash --estimate # same quote, inline\\nvideodraft generate video \\"...\\" --model minimax-h3-max --duration 8 --resolution 768p --prompt-expansion-mode balanced\\n```\\n","references/pipeline.md":"# VideoDraft pipeline reference\\n\\nEverything here describes the hosted fallback pipeline through the CLI (`videodraft <command>` / `videodraft call <tool>`) or hosted MCP connector (tool names in backticks). When the local `videodraft_editor` MCP is available, do not use hosted production or export by default. Use hosted tools only for asset generation and optional script/storyboard stages, then import the results and finish with the native editor reference linked from SKILL.md. Continue through `produce_project` and `export_video` only when the user explicitly requests a hosted web production or the native editor is unavailable.\\n\\nUse direct asset tools for standalone images, clips, audio, upscales, and descriptions. Use a hosted project when the user explicitly wants the editable web project, when a hosted storyboard stage is useful, or when the native editor is unavailable. Script-only uses a script-stage project and stops at the script. In VideoDraft ADE with editor tools present, stop before hosted production, import the generated assets, and build/export the native project.\\n\\n## Stages and their tools\\n\\n| Stage | CLI | Underlying tool |\\n| ---------------------------------------- | --------------------------------------------------------------------------- | ------------------------------------------- |\\n| Idea \u2192 full storyboard project | `videodraft create \\"<idea>\\"` | `generate_storyboard_from_idea` |\\n| Idea \u2192 script only (stop there) | `videodraft create \\"<idea>\\" --script-only` | `generate_script_from_idea` |\\n| Footage IS the video | `videodraft call generate_storyboard_from_media` | `generate_storyboard_from_media` |\\n| Batch shot images | `videodraft shots <project>` | `generate_shot_images` |\\n| One shot image | `videodraft generate image --project <id> --scene N --shot M` | `generate_image` |\\n| Produce (voiceover, captions, timeline) | `videodraft produce <project>` | `produce_project` |\\n| Seedance full-video production | `videodraft produce <project> --mode full_video` | `produce_project` with `mode: \\"full_video\\"` |\\n| Per-shot motion prompts | `videodraft video-prompts <project>` | `generate_video_prompts` |\\n| Motion clip for a shot | `videodraft generate video --project <id>` | `generate_video` |\\n| Attach a finished clip to the timeline | `videodraft attach <project> --scene N --shot M --media <url> --type video` | `attach_media_to_shot` |\\n| Find free b-roll (no credits) | `videodraft stock search \\"<query>\\"` | `search_stock_media` |\\n| Copy stock media onto the CDN | `videodraft stock import <ref>` | `import_stock_media` |\\n| Background music | `videodraft generate music --attach <project>` | `generate_music` / `set_background_music` |\\n| Song with vocals, lyrics or sections | `videodraft generate music --model elevenlabs-music-v2.5 --section \\"...\\"` | `generate_music` with `composition_plan` |\\n| General or reference-driven audio | `videodraft generate audio \\"...\\"` | `generate_audio` |\\n| Sound effect | `videodraft generate sound-effect \\"...\\"` | `generate_sound_effect` |\\n| Dialogue audio | `videodraft generate dialogue --line \\"voice:text\\"` | `generate_dialogue` |\\n| Voice changer | `videodraft generate voice-changer <audio>` | `change_voice` |\\n| Dubbing | `videodraft generate dub <audio_or_video>` | `dub_media` |\\n| Scene voiceover | `videodraft generate voiceover --project <id> --scene N` | `generate_voiceover` |\\n| Avatar script | `videodraft avatar script \\"<idea>\\"` | `generate_avatar_script` |\\n| Avatar + speech | `videodraft avatar create <portrait> --script \\"...\\"` | `create_avatar_video` |\\n| Talking-head render | `videodraft avatar render <avatar_video_id>` | `render_avatar_video` + `get_avatar_video` |\\n| Direct portrait + text/audio | `videodraft avatar fabric <portrait> --text \\"...\\"` or `--audio <file>` | `generate_veed_fabric_video` |\\n| Short portrait clip from audio, up to 2K | `videodraft avatar h3-lipsync <portrait> --audio <file>` | `generate_minimax_h3_lipsync_video` |\\n| Existing video + replacement audio | `videodraft avatar lipsync <video> --audio <file>` | `generate_sync_lipsync_video` |\\n| Existing-video AI edit | `videodraft edit video <video> \\"<change>\\" --model <video-edit-model>` | `edit_video` |\\n| Motion transfer | `videodraft edit motion <image> [direction] --motion-video <video>` | `generate_motion_control_video` |\\n| Image enhancement/upscale | `videodraft upscale image <image>` | `upscale_image` |\\n| Video enhancement/upscale | `videodraft upscale video <video>` | `upscale_video` |\\n| Frame rate / slow motion | `videodraft interpolate <video> --fps 60 [--slowdown 4]` | `interpolate_video` |\\n| Final MP4 | `videodraft export <project>` | `export_video` + `check_export_status` |\\n\\n## Rules that prevent broken results\\n\\n- **The storyboard is generated FROM the script**, never from the raw idea. `videodraft create` runs the whole chain correctly. Don\'t call `generate_storyboard_scenes` with a raw idea as the \\"script\\".\\n- **Visual consistency**: never generate a storyboard shot in isolation. Shot prompts carry `[[asset:Name]]` / `[[shot:X-Y]]` tags that `generate_shot_images` resolves against the project\'s visual assets and prior shots. When generating a single shot whose prompt has no tags, pass `--ref` images yourself (the project\'s visual assets and/or the previous shot\'s image; `projects get` exposes both). For scenes with multiple shots or recurring characters, prefer `videodraft shots <project> --model <selected-image-model> --grid`: preserve an explicitly requested compatible image model, otherwise use `nano-banana-2`. It creates one coherent scene grid, then decodes it into individual shot images.\\n- **Reference-first video**: when identity, styling, or composition matters, do not generate each motion clip from text alone. Generate or select the shot still first, then pass the decoded shot image as `--start-image` or `--ref` to the selected video model. AI Production already composes scene grids and sends them to Seedance as references. If the user explicitly requests another compatible video model, bypass fixed Seedance full-video mode and generate the per-shot clips with the requested model, using the individual decoded shot images as anchors.\\n- **Seedance full-video real people**: hosted `full_video` allows real people by default, which applies Fal-tier pricing to every submitted segment and permits the Byteplus-to-Fal fallback. For the lower Byteplus-only rate when no scene grid shows a real identifiable person (non-people, anime, clearly synthetic or stylized characters), use `videodraft produce <project> --mode full_video --no-allow-real-people`, or MCP `produce_project` with `mode: \\"full_video\\", allow_real_people: false`. If a partial opted-out run returns `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, re-estimate, follow the user\'s spend-confirmation preference, and rerun the same project once with an explicit `--allow-real-people` / `allow_real_people: true`. The server reconciles asynchronous results first, preserves running/completed jobs, and retries only failed placeholders carrying that exact code. Do not loop when it was already on. VideoDraft refunds a Byteplus task rejected after asynchronous acceptance, but cannot reroute it; rephrase or change the scene references instead.\\n- **Hold off generating shot images while the user is still iterating** on storyboard structure.\\n- **produce \u2192 export ordering**: `export` requires a produced project where every production scene has timeline media. If `produce` returns `generating_shot_images`, poll the job ids it returns, then re-run produce.\\n- **Do not attach motion clips before production exists**: run `produce` successfully first, then attach finished motion clips to the production timeline. Attaching before `production_data` exists cannot place them in the final timeline.\\n- **Generated motion clips do not auto-attach**: after `generate video` completes, attach the clip with `attach_media_to_shot` (`media_type:\\"video\\"`, include `duration_seconds`) \u2014 it replaces the production timeline clip while keeping the storyboard still.\\n- **Talking heads use dedicated avatar tools**: do not use `generate video`. Use managed `avatar create` and `avatar render` for reusable avatars, direct `avatar fabric` for a portrait plus text/audio, `avatar h3-lipsync` for a short (5-14.8s) portrait clip from existing audio at up to 2K, and `avatar lipsync` for an existing video plus replacement audio. Reuse a supplied person image or generate a clear front-facing portrait with the explicitly requested compatible image model, otherwise Nano Banana 2. Managed avatar creation and speech are bundled/free; direct Fabric, H3 Max Lip Sync, Sync, and render are paid.\\n- **Existing-video edits use their own category**: call `edit_video` or `videodraft edit video` with a `video_edit` model when transforming the source itself. Kling O3 also has a reference-generation mode that creates a new guided clip. Wan 3.0 is a text/frame/reference generation model, not a source-video editor. Motion transfer similarly uses `generate_motion_control_video` or `videodraft edit motion` with a `motion_control` model.\\n- **Upscaling preserves rather than redesigns**: use Topaz when resolution, detail, or cleanup is the problem (generative mode by default: Wonder 3.5 / Starlight Precise 2.6, Topaz\'s pick for AI sources; precision for real photos/footage or the cheapest pass; creative only for an artistic reinterpretation; `interpolate` for frame rate or slow motion). Regenerate or edit when the subject, text, framing, continuity, or motion is wrong. Upscale a low-quality avatar portrait before Fabric; do not render a new avatar at 480p just to upscale the result.\\n- **Timeouts on the one-shot create**: if `create` times out at the transport layer, the project was still created server-side \u2014 `videodraft projects list`, take the most recent, and resume with its id. Don\'t start a duplicate.\\n\\n## User-attached media: classify roles first\\n\\nFor EACH attached file decide:\\n\\n- **visual_asset** \u2014 recurring reference (character / product / location / style). Upload, then pass in `visual_assets` of `generate_storyboard_from_idea` (via `videodraft call`), or add to an existing project with `add_visual_assets`. Type must be one of `character | object | location | style | custom` with a short name + concrete description.\\n- **shot** \u2014 the media IS footage for the video. Whole video = footage \u2192 `generate_storyboard_from_media`. Idea + footage \u2192 `generate_storyboard_from_idea` with `shot_media`. Existing storyboard \u2192 `attach_media_to_shots`.\\n- **reference** \u2014 inspiration only \u2192 fold a description into the idea/instructions; don\'t place it as a shot or asset.\\n\\nAmbiguous (e.g. a person holding a product)? Ask the user.\\n\\nUploads persist in the media library \u2014 recall later with `videodraft media list`.\\n\\n## Editing project data safely\\n\\n1. `videodraft call get_project_schema` \u2014 read the structure once per session.\\n2. `videodraft projects get <id> --raw` \u2014 the exact editable blob.\\n3. Modify; then `videodraft call update_project --stdin` with `{\\"project_id\\": \\"...\\", \\"data\\": {...}}`.\\n - Objects deep-merge key-by-key; **arrays replace wholesale** \u2014 send the complete array you\'re changing (e.g. all of `storyboard.scenes`).\\n - Scene shot arrays (`image_prompt` / `shot_types` / `shot_actions` / `search_prompt` / `preview_media`) are auto-aligned; fix-ups come back as warnings.\\n4. Snapshot before risky edits: `videodraft checkpoint create <id> --name \\"before re-script\\"`. Restore with `videodraft checkpoint restore <id> <version>`.\\n\\n## AI Studio sessions (standalone generations)\\n\\nProject generations group automatically, and standalone work is grouped per conversation (MCP) or per working directory (CLI) by the connection session \u2014 see the \\"AI Studio sessions\\" section of SKILL.md. Once the task is clear, give that automatic session a concise 3-6 word title before the first generation:\\n\\n```bash\\nvideodraft sessions name \\"Fox Brand Explorations\\"\\nvideodraft generate image \\"...\\"\\n```\\n\\nNaming creates the session with that title. If generation, a user, or an earlier agent created it first, the existing name is preserved. The command refuses to run while `VIDEODRAFT_SESSION` is set, because that override would send later generations to a different session. Use `sessions create` plus `--session` only to continue or deliberately create a separate group.\\n"}');
7759
+ if ('{"SKILL.md":"---\\nname: videodraft\\ndescription: Create and edit AI videos, images, 3D meshes, Seed Audio, voiceovers, music, sound effects, dialogue, dubbing, storyboards, avatar videos, media upscales, and product/ad videos with VideoDraft. Use whenever the user mentions VideoDraft; asks to generate a video, image, 3D asset, audio asset, ad, explainer, storyboard, avatar, upscale, or batch/CI workflow; wants to download a video from YouTube, Instagram, TikTok, X or another site; or wants to assemble, cut, caption, mix, lay out, inspect, or export a native VideoDraft Editor timeline. Covers the cloud `videodraft` CLI/MCP and local headless `videodraft_editor` MCP. When the editor MCP is exposed, prefer it for production, timeline assembly, and export; use cloud production/export only when explicitly requested or the editor is unavailable.\\n---\\n\\n# VideoDraft\\n\\nVideoDraft is an AI video creation platform where asset generation is the priority lane:\\n\\n- **Asset generation**: standalone images, video clips, Seed Audio, voiceovers, music, sound effects, dialogue, voice-changed audio, dubbed media, upscales, and image descriptions. This is the fastest and most important lane. Treat these as complete deliverables when the user asks for assets.\\n- **Asset I/O**: upload local files, download outputs, auto-upload local references, and save generated media where the user can see it.\\n- **3D assets**: standalone Meshy 7 and Tripo H3.1 meshes, multi-view generation, complete artifact downloads, and Meshy humanoid rigging through MCP/CLI. Read [references/3d.md](references/3d.md), or `videodraft skills show 3d`. These assets have their own saved library and do not require an AI Studio session or web viewer.\\n- **Native editing**: local `.vdproject` timelines, cuts, layouts, captions, effects, audio, and exports through the headless VideoDraft Editor. Inside VideoDraft ADE, this is the default production and export lane whenever `videodraft_editor` is available.\\n- **Hosted project production**: idea \u2192 script \u2192 storyboard \u2192 hosted production timeline \u2192 exported MP4. Use the early stages for scripts, storyboards, and generated assets when useful. Treat hosted production and export as a fallback when the native editor is unavailable, or as an explicit destination when the user asks for an editable web project or hosted workflow.\\n\\n## How to connect\\n\\nGeneration billing is selected in VideoDraft Settings \u2192 API keys. Where enabled,\\n**VideoDraft routing** chooses an equivalent provider internally without changing\\nthe quoted customer credit price. A selected personal key is exclusive: never\\nswitch to VideoDraft credits or another key after an unsupported-model or\\nprovider error. Keep the canonical VideoDraft model ID in CLI/MCP calls. Poll\\nthe original generation after a timeout instead of starting another paid job.\\nNew provider connections can support only a subset of models and settings;\\nfollow the server\'s availability response. ElevenLabs remains audio-only.\\nFor Pika GPT Image 2.5, supply an explicit quality (`low`, `medium`, `high`,\\n`xhigh`, or `max`); `auto` is not supported by that connection. Nano Banana 2\\nLite is 1K only. Never remove requested controls or substitute a different\\nmodel merely to make a personal connection accept the job.\\n\\nCloud generation has two equivalent surfaces (same backend, credits, and hosted projects). Native timeline editing is a separate local surface:\\n\\n1. **CLI** (preferred when you have a shell): run `videodraft` if it\'s on PATH; otherwise `npx -y videodraft@latest` runs it with no install (needs Node \u226520; the `-y` skips npx\'s install prompt so it runs non-interactively; the package is fetched on first use and cached). For heavy use, `npm install -g videodraft`. If there\'s no Node/shell here but the MCP connector below is available, use that instead; if neither works, tell the user how to install (https://videodraft.ai/cli).\\n - Auth \u2014 pick by context, don\'t guess:\\n \u2022 INTERACTIVE (a human is in the session, e.g. Claude Code / Codex): on exit code 3 (\\"not authenticated\\"), tell the user to run `videodraft login` in their terminal \u2014 it opens their browser for a one-click VideoDraft sign-in (OAuth), no key to copy. Wait for them to confirm it succeeded, then retry the command. This is the preferred path when the user is present.\\n \u2022 HEADLESS / CI (no browser): set `VIDEODRAFT_API_KEY=vd_mcp_...` (a token the user mints at https://app.videodraft.ai/mcp-keys).\\n \u2022 SECURITY: never ask the user to paste a `vd_mcp_...` token into the chat \u2014 use browser `login` or the env var so the token never lands in the transcript.\\n - Every command accepts `--json` (parse this, don\'t scrape text). Exit codes: 0 ok, 1 error, 2 usage, 3 auth (see Auth above), 4 insufficient credits (\u2192 tell the user, don\'t retry).\\n - Tool discovery: start with `videodraft tools list` for the grouped catalog, then narrow with `videodraft tools list --lane assets`, `--lane asset_io`, `--lane project_data`, or `--lane production`.\\n - Asset lane: `videodraft generate ...`, `videodraft edit video|motion`, `videodraft avatar ...`, `videodraft upscale ...`, `videodraft interpolate ...`, `videodraft upload`, and `videodraft download`.\\n - Full API access: `videodraft tools schema <name>`, `videodraft call <tool> --args \'<json>\'`.\\n2. **MCP connector**: if VideoDraft MCP tools (e.g. `generate_storyboard_from_idea`) are available, call them directly \u2014 the CLI\'s curated commands map 1:1 onto these tools.\\n3. **Native editor MCP** (`videodraft_editor`): prefer this for project production, timeline assembly, cutting, layouts, transitions, captions, audio placement, and final export. Inside VideoDraft ADE on a supported Mac, Claude, Codex, OpenCode and Grok receive it automatically in both Code and VideoDraft modes. It runs headlessly, so an Open Editor click is not required. Start with `project_manage` (`operation: \\"project\\"`, action `list`, `open`, or `create`); standalone asset generation remains in the cloud CLI or MCP.\\n\\nSend native editor edits serially, batching everything a request needs into one `edit_apply` recipe. Pass the latest `context` when a person may be editing at the same time, so a stale write is refused. See [references/editor.md](references/editor.md) for project selection, media import, timing units, edit recipes, verification, export, and the `videodraft-editor` terminal bridge.\\n\\nIf you are reading this skill through `videodraft skills show skill`, run `videodraft skills show editor` before native editor work to load that reference.\\n\\n**VideoDraft ADE routing rule:** the presence of `videodraft_editor` means the native editor is ready, even when no editor window is visible. Use cloud tools to generate or source assets and, when helpful, scripts or storyboards. Do not call hosted `produce_project` / `videodraft produce` or `export_video` / `videodraft export` by default. Import the assets into the native project, assemble there, and export with native `delivery_manage` `submit`. Use hosted production/export only when the user explicitly asks for the web workflow or the native editor tools are unavailable. Do not silently fall back to hosted production after a native tool error.\\n\\n## First decision: asset, hosted project, or native edit?\\n\\n- **One standalone asset** (image, clip, voiceover, music track, sound effect, dialogue track, voice-changed file, dubbed media file, upscale, or description): generate it directly. Do NOT create a project.\\n - `videodraft generate image \\"a red fox in snow, cinematic\\" --ar 16:9 --download ./out/`\\n - `videodraft generate video \\"slow dolly over a misty lake\\" --model gemini-omni-1.1-flash --duration 6 --download ./out/`\\n- **Any final video, production timeline, existing footage, local `.vdproject`, or hands-on edit**: use `videodraft_editor` when available. List or open the intended local project, or create a native project for a new production. The editor can work without showing its UI.\\n- **A small set of related assets**: still stay in the asset lane. They are grouped automatically (see [AI Studio sessions](#ai-studio-sessions)); give the current group one useful task-specific name with `videodraft sessions name \\"<name>\\"`. Switch to a project only when the deliverable matches the project criteria below or the user asks to attach the assets to one.\\n- **A generated multi-scene video / ad / explainer**: when the editor is available, use hosted tools only for any needed script, storyboard, shot planning, or generated assets; stop before hosted production, import the assets, and build/export the native timeline. A hosted project is optional unless the user wants the web project or its storyboard workflow.\\n- **A hosted web project or hosted export**: use the hosted pipeline only when the user explicitly asks for it or the native editor is unavailable.\\n- **Just a script** (no video asked for): A script-only request creates a script-stage project but stops at the script. Use `videodraft create \\"...\\" --script-only`; do not build a storyboard the user didn\'t ask for.\\n- **Iterating on existing work**: identify the surface first. Use `project_manage` `project` with `action: \\"list\\"` for native projects and `videodraft projects list` only for hosted work. Never create a replacement project just to change an existing one.\\n\\n## Choose the model from the task\\n\\nIf the user names a model, use it when compatible. If it cannot handle the request, explain why and recommend alternatives instead of silently switching. Otherwise inspect the inputs, duration, audio, quality, speed, and cost, check the live catalog, and pass an explicit model.\\n\\n### Seedance 2.x real-person rule\\n\\nSeedance 2.0 and 2.5 allow real people by default, the same as AI Studio:\\n\\n- The default is true. Generation tries Byteplus first and can fall back to Fal, which allows real-person likenesses, so a person in the prompt, a start frame, end frame, reference image, or reference video does not hard-fail. The request is charged at Fal\'s higher tier-specific rate even if Byteplus serves it.\\n- Turn it off for the cheapest run, only when the user wants that and nothing in the job is a real identifiable person (text-only requests, non-people, anime, clearly synthetic or stylized characters). For MCP use `allow_real_people: false`. For the CLI use `--no-allow-real-people`. That pins the job to the lower-priced Byteplus path, and a Byteplus likeness-policy refusal does not fall back to Fal. Pass the same value to `get_model_costs` or `videodraft costs` so the estimate matches the charge.\\n- If a request made with the option off fails with code `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, estimate the higher rate, follow the user\'s spend-confirmation preference, and retry exactly once with the option on. The structured recovery fields are `retryable: true`, `retry_with: { allow_real_people: true }`, `cli_flag: \\"--allow-real-people\\"`, and `retry_policy: \\"once\\"`. CLI `--json` submit errors expose them under `details`; `status` and `wait` include them on the failed job result. Do not treat an unrelated moderation or provider error as that signal.\\n- Do not loop when the option was on. Byteplus can accept a task and reject its generated output later. VideoDraft refunds that failed generation, but the late asynchronous failure cannot be rerouted to Fal. Rephrase the prompt or use different references before trying again.\\n- For hosted AI Production, the same default applies to every Seedance scene segment: `produce_project` with `mode: \\"full_video\\"` and `videodraft produce <project> --mode full_video` allow real people unless you pass `allow_real_people: false` or `--no-allow-real-people`. If an earlier run with the option off partially submitted and returns the opt-in code, rerun that same project once with an explicit `allow_real_people: true` / `--allow-real-people`. The server reconciles asynchronous results first, preserves running/completed jobs, and resubmits only failed scene-video placeholders carrying the exact opt-in signal. Keep the native-first VideoDraft ADE routing rule above: hosted full-video production is still explicit/fallback-only when the local editor is available.\\n\\n**How to pick a model. Follow this order every time:**\\n\\n1. The user named a model: use it, never substitute.\\n2. Otherwise pick a **Tier 1** model.\\n3. Use a **Tier 2** model only when no Tier 1 model supports the request, and say why. Never pick Tier 2 on your own for price, speed, novelty or variety.\\n\\nTier 1: images `nano-banana-2`, `nano-banana-pro`, `nano-banana-2-lite`, `gpt-image-2.5-flare`, `gpt-image-2.5-sunburst`, and `recraft-v4` for vector/SVG only; videos `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0`, `kling-o3`, `minimax-h3-max`; video edits `gemini-omni-1.1-flash`; talking-head portraits `veed-fabric`, `minimax-h3-max-lipsync`; motion transfer `kling-v3-motion-control`; audio ElevenLabs for speech tools, Lyria 3.5 for music at any length.\\n\\nTier 2 is every other catalog model (`wan-3.0`, `flux-3`, `minimax-h3`, `kling-v3-turbo`, `kling-2.6-pro`, `grok-imagine-video-1.5`, `happy-horse`, Veo 3.1, Sora, Seedream, Grok Imagine images, FLUX images, `sync-lipsync-2` and the rest). All of them stay available by name. Real Tier 2-only jobs: document or web-page references (`wan-3.0`), keyframes pinned to timestamps (`flux-3`), re-syncing existing footage (`sync-lipsync-2`).\\n\\nEvery `videodraft models image|video|audio --json` entry carries `tier` (1 or 2), and the top-level `recommended` array is Tier 1, best first. That catalog beats this page when they disagree.\\n\\n**Speech in video:** write the dialogue in the video prompt. `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` and `kling-o3` speak it natively. For a specific voice add a Seedance reference audio clip (`--ref-audio`) or a Kling bound voice. Do not plan silent video + TTS + lip-sync. Lip-sync tools are only for a portrait presenter talking to camera, or for re-syncing footage that already exists. Off-screen narration and voiceover are unaffected: keep using `generate voiceover` for those.\\n\\n**Images:**\\n\\n- `nano-banana-2`: general default, editing, consistency, and references.\\n- `nano-banana-pro`: maximum quality. `nano-banana-2-lite`: fast, inexpensive drafts.\\n- `gpt-image-2.5-flare`: replaces the GPT Image 2 recommendation for posters, logos, signs, title cards, readable text and composition. Use `gpt-image-2.5-sunburst` for extra precision/detail. Both use the same basic options as GPT Image 2: references (up to 16), aspect ratio, 1K/2K/4K resolution, image count (1-4), and quality. Quality adds xhigh/max; auto is charged at the max tier. Other model recommendations are unchanged; explicit `gpt-image-2` requests still use GPT Image 2.\\n\\n**Videos:**\\n\\n- `gemini-omni-1.1-flash`: general default for 3-10s generation, image-to-video, uploaded source edits up to 10s, and silent or voiceover-backed shots (audio is always on, so describe quiet ambience with no speech and mute or mix it on the timeline; only a file with no audio track at all needs `kling-3.0`, `kling-o3` or Seedance with `--no-audio`). It supports first and last frames, up to 10 total image inputs, up to 3 creative reference videos of at most 3 seconds each on `--video-task generate` ONLY, uploaded-video extension, and continuation of an earlier generation through `--previous-interaction-id`. Output is 360p/720p/1080p/4K at 3/10/15/30 cr/s with audio always on. Use `--source-video` for the uploaded edit/extension source; extension sources must be 1-30s. Edit and extend accept EXACTLY ONE input video, so `--ref-video` cannot accompany `--source-video` or `--previous-interaction-id`; only `--ref` images can. One `--ref-video` with no separate source remains a legacy source edit. A previous interaction defaults to conversational edit; add `--video-task extend` or `--extend` with an explicit 3-10s duration to append at the end. A continuation resolves to the prior turn\'s output and is submitted as an ordinary source, so the same limits apply to it: at most 30s to extend, at most 10s to edit. 40 seconds total is reachable, but only by extending from a source of 30s or less, so a ladder dead-ends once it passes 30s. New dialogue can only be added when the SOURCE video is silent; adding speech on top of a source that already contains speech is refused with \\"the model is currently unable to process speech edits\\". The server safely measures creative-reference durations, or you can repeat `--ref-video-duration` when a host blocks metadata probing. Fal BYOK supports its currently callable v1.1 generation and basic-edit endpoints at zero VideoDraft credits, but Fal does not expose continuation/extension or mixed source-edit references as callable endpoints and VideoDraft must never fall back to paid Google.\\n- `grok-imagine-video-1.5` (Tier 2): 1-15s text, first-frame, or 1-7 reference-image generation with native audio. Text/first-frame modes support 480p, 720p, and 1080p; reference mode supports 480p/720p. Cite references as `<IMAGE_0>` through `<IMAGE_6>`. It has no last frame, seed, negative prompt, quality tier, reference video, or reference audio.\\n- `minimax-h3` (Tier 2): 480p/768p/2K/4K (5/6/13/16 cr/s, 768p default), native stereo audio, and 5-15s text, first/last-frame, or mixed-reference generation. Reference mode accepts up to 9 images, 3 videos, and 3 audio clips, with at most 12 files total. Cite them as `Image 1`, `Video 1`, and `Audio 1` in array order.\\n- `minimax-h3-max`: 480p/768p pricing (5/8 cr/s, 768p default), native audio, and 5-15s text, first/last-frame, or mixed-reference generation. Reference mode accepts up to 9 images, 3 videos, and 3 audio clips, with at most 12 files total, cited as `Image 1`, `Video 1`, and `Audio 1` in array order. It supports a reproducibility seed and `disabled` / `balanced` / `quality` prompt expansion.\\n- `wan-3.0` (Tier 2): unified 2-30s text, first/last-frame, or ordered mixed-reference generation at 480p/720p/1080p (7/14/28 cr/s), with optional native audio. Reference mode accepts up to 10 images, 5 videos, and 5 audio clips, at most 20 media files total; video and audio each total at most 15 seconds. `--auto-duration` reserves 30 seconds and reconciles to the provider-reported output length. Document/web references use `--file-url` or `--web-url` and require `--thinking`.\\n- `flux-3` (Tier 2): Black Forest Labs FLUX 3. 5-20s at 720p/1080p with 24fps native audio, from a prompt, a first frame, first + last frames, or up to 10 keyframes pinned to specific moments (`--keyframe shot.png@2.5`, repeatable). `--quality draft` renders the same shot at 720p for roughly a third of the cost. Auto duration is text/first-frame only.\\n- `seedance-2`: 11-15s, video/audio/mixed references, wider ratios, selectable audio, or first/last frames. `mini` is the economical default; move to `standard` when the user asks for high quality, when a Mini result is not good enough, or for 1080p/4K; `fast` for speed.\\n- `seedance-2.5`: 4-30s single takes and up to 50 references (30 image, 10 video, 10 audio). Same modes as 2.0, one quality tier, 480p/720p/1080p. Reach for it when a shot must run past 15s or carry more references than 2.0 allows.\\n- `kling-o3`: reference images plus structured image/video elements, first/last frames, multi-prompt, audio control, or 4K. O3 allows 7 combined image references and image-backed elements, reduced to 4 combined items when a video-backed element is present. `kling-3.0`: image-to-video can use structured image/video elements and bind a custom Kling voice ID to either element form. Tier 2, by name: `kling-v3-turbo` (fast polished 3-15s with first frame, multi-prompt, and audio, but no elements) and Kling 2.6 Pro (top-level voice IDs cited as `<<<voice_1>>>` and `<<<voice_2>>>`).\\n- Existing-video edits use `videodraft edit video`, not generic generation. `gemini-omni-1.1-flash` is the preferred edit model and is chosen automatically when you omit `--model`: source up to 10s, up to 10 reference images, 360p/720p/1080p/4K with audio. Creative `--ref-video` inputs are NOT accepted on an edit, because an edit takes exactly one input video; use `generate video --video-task generate` to guide a new clip with video references instead. Omitting `--model` on a longer or unmeasurable source spends nothing and prints a priced menu (what each model edits, what it drops, what it costs) so you can put the choice to the user. **Truncation:** only Gemini refuses an over-length source. Happy Horse silently edits just the first 15s, Kling O3 the first 10s, and Grok the first 8s. The command warns you when that will happen; always relay it to the user. Gemini regenerates the audio track, so use Happy Horse or Kling O3 with `--preserve-audio` when the source audio must survive. Choose Grok for the cheapest prompt-only edit, Happy Horse for up to 5 image references, or Kling O3 for controlled reference-image edits.\\n- To edit a source longer than 10s without losing its tail, cut it into <=10s pieces in the native editor, edit each with Gemini, and reassemble. VideoDraft has no server-side split/concat, so this path needs `videodraft_editor`.\\n- Kling O3 also has a reference-generation mode. Use `videodraft generate video --model kling-o3-video-ref-edit` with exactly one `--ref-video` to generate a new guided clip; use `videodraft edit video` when changing the source itself. For new mixed-reference generation use Seedance 2 / 2.5.\\n- Motion transfer uses `videodraft edit motion` with Kling V3 by default, or Kling 2.6 when explicitly requested or lower cost matters. It requires a subject image and a motion-reference video.\\n- Veo 3.1, Sora, Grok, Happy Horse and the other Tier 2 video models: only when explicitly requested or when no Tier 1 model fits.\\n\\n**Audio and utilities:**\\n\\n- Use Seed Audio 1.0 for open-ended text-to-audio, speech/music/sound synthesis, voice conditioning, or prompt-driven editing with up to three audio references or one image. Use `videodraft generate audio`. Reference clips are `@Audio1`, `@Audio2`, and `@Audio3` in array order. There is no duration input. Output is up to two minutes and settles at 19 credits per actual minute, with up to 38 credits reserved during generation. The CLI automatically retries transient responses with one operation key. To recover after the CLI process itself is interrupted, set `--idempotency-key <uuid>` on the original command and reuse it.\\n- Prefer ElevenLabs for voiceover, dialogue, voice changing, dubbing, and sound effects. Honor an explicitly selected supported TTS voice/provider. Use `lyria-3.5` for all music, short or long (up to ~3 minutes, 10 credits flat), with vocals/lyrics or instrumental arrangements. When `--model` is omitted the CLI sends no model and the server default applies, which is `lyria-3.5` wherever the backend supports it. If the server answers that `lyria-3.5` is an unknown model, that backend predates it: rerun without `--model` (the server default then applies), or use `lyria-3-pro-preview` for a track longer than 30 seconds. Ask for the length in the prompt. Keep `lyria-3-clip-preview` (fixed ~30s, 4 credits) and `lyria-3-pro-preview` (8 credits) for explicit requests. For Lyria, put desired length, lyrics and \\"instrumental only, no vocals\\" in the prompt; `--length` and `--instrumental` are ElevenLabs-only. Use ElevenLabs Music for exact timing, composition plans, or a style reference track. Lyria allows 10 reference images on Google, 1 on Fal BYOK; Fal 3.5 prompts are limited to 5000 characters. BYOK uses zero VideoDraft credits. ElevenLabs Music means `elevenlabs-music-v2.5` (`elevenlabs-music` is an alias for it); use `elevenlabs-music-v1` only when the user asks for v1 by name.\\n- ElevenLabs voiceover runs on Eleven v4 in two modes. Standard is the default and the best quality, at 10 credits per 1000 characters. Turbo (Eleven v4 Turbo) is faster, at 5 credits per 1000 characters. Keep Standard unless the user wants faster or cheaper speech. Pick it with `generate voiceover --mode standard|turbo`, `produce --voice-mode standard|turbo`, or MCP `generate_voiceover.mode` / `produce_project.voice_mode`, and quote Turbo with `costs voiceover --chars <n> --mode turbo`. Google, OpenAI and cloned `custom-*` voices ignore the mode (cloned voices cost 30 per 1000). v4 has no style or speed settings and no SSML `<break>` tags; audio tags such as `[whispers]` work. Dialogue stays on Eleven v3.\\n- For ElevenLabs voiceover, dialogue, and voice changing, accept a supplied raw voice ID (16-64 alphanumeric characters) or `elevenlabs-<id>`. The voice catalog is for discovery, not an allowlist. Do not reject or substitute a supplied ID because it is absent from `videodraft models voices` / `list_available_voices`. Use `--voice <id>` for voiceover and voice changing, or repeat `--line \\"<id>:Text\\"` for dialogue; for example, `--voice kPzsL2i3teMYv0FxEYQ6` and `--line \\"elevenlabs-kPzsL2i3teMYv0FxEYQ6:Hello.\\"` use the same voice. The voice must be accessible to the provider account used for generation. Private or cloned voices may require the user\'s connected ElevenLabs key. Kling video-control IDs and MiniMax `custom-*` IDs are separate voice systems.\\n- A character who needs to TALK:\\n - **Speaking inside a scene**, or any ordinary clip with dialogue: `generate video` with the line in the prompt, in quotes, with who says it and how. `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` and `kling-o3` all voice it natively. For a specific voice use a Seedance `--ref-audio` clip or a Kling voice bound per element. This is the default path; do not generate speech separately and lip-sync it on.\\n - **Talking to camera from a portrait** (presenter, spokesperson, explainer): the avatar lane, and VEED Fabric is preferred. Use managed `avatar create` then `avatar render` for a reusable avatar record with bundled speech, `avatar fabric` for a one-off portrait plus text or existing audio, and `avatar h3-lipsync` for a short (5-14.8s) portrait clip from existing audio at up to 2K. Not for in-scene shots: Fabric animates a portrait facing the lens, so a cinematic request comes back as a head-on talking headshot.\\n - **Footage that already exists plus replacement audio** (re-dub, translation): `avatar lipsync`. This is the only job for Sync Labs unless the user names it.\\n- Enhancement: use Topaz image/video upscaling only when the content is already correct. Use image 1x for cleanup, 2x by default, 4x when justified; use video 2x by default. Edit or regenerate creative errors.\\n\\nSee [references/models.md](references/models.md) for the detailed routing table and exact capability limits.\\n\\nProvider routing uses a maintained local price catalog, with exact setting compatibility and account rules. Platform requests prefer Google\'s direct image/video APIs and BytePlus for Seedance 2.0/2.5 before comparing public prices. Seedance 1.5 retains its existing Replicate primary; other models remain price-based. Agents should keep using normal model IDs and `get_model_costs` / `--estimate` for customer credits. Do not choose a vendor from a marketing starting price or retry an accepted generation to chase a lower price. Unknown or expired comparable costs retain the existing route; a selected personal provider key stays exclusive and overrides platform-account preferences.\\n\\nTemporary provider promotions are excluded from routing prices. With an Atlas personal key, H3 Max supports text or first/last-frame generation at 480P/768P only with `prompt_expansion_mode: \\"disabled\\"` and no seed. Balanced/quality expansion, reference generation and H3 Max Lip Sync require the existing Fal path. Atlas H3 Max is not selected automatically while its published price units remain inconsistent.\\n\\n## Matching AI Studio controls\\n\\nUse `models image --json` / `models video --json` for the accepted options. Nano Banana Pro/2 expose `--temperature 0..2` and `--google-search-grounding true|false`. Qwen uses one `--ref` plus `--horizontal-angle 0..360`, `--vertical-angle -30..90`, and `--zoom 0..10`; its text prompt is optional. Recraft generates SVG and accepts `--quality Normal|Pro`, repeatable `--recraft-color \'#RRGGBB\'`, and `--recraft-background \'#RRGGBB\'`. Original Nano Banana and VideoDraft Image use fixed 1K output.\\n\\nKling 3.0 / 2.5 Pro accept `--cfg-scale 0..1` and `--negative \\"\\"` to clear a negative prompt. Seedance 2/2.5, Wan 3.0, FLUX 3 text/first-frame and Gemini Omni generation accept `--auto-duration`; do not combine it with `--duration`. FLUX 3 auto reserves 20 seconds, so its default estimate includes that ceiling. `--model sora-2 --quality pro --resolution 1080p` selects Sora Pro. Kling O3 reference/edit modes accept image-only `--element` objects, at most four combined with `--ref`. Grok v1 accepts image references for clips up to 10 seconds.\\n\\nTopaz image results may be immediate or queued. The CLI waits by default and downloads either result with `--download`; `--no-wait` returns a queued job ID. Raw MCP callers must poll `check_generation_status` when `upscale_image` returns `job_id`.\\n\\n## Prefer references when continuity matters\\n\\nPure text-to-image or text-to-video is fine for a generic one-off asset. When a specific character, product, location, style, composition, or brand identity must survive generation, use references instead of hoping the prompt recreates it.\\n\\n- If the user supplies reference media, preserve and pass it. Never reduce the request to text alone.\\n- When continuity matters, generate/select a strong still first with the selected image model (`nano-banana-2` by default), wait for its URL, then animate it as a start frame/reference. Confirm the combined image and video cost.\\n- When using a hosted storyboard stage for multiple shots, use `videodraft shots <project_id> --model <selected-image-model> --grid`, then animate the decoded shots. Preserve explicit models. In VideoDraft ADE, import the resulting assets into the native editor instead of continuing into hosted production. A requested non-Seedance video model must use manual per-shot generation instead of Seedance full-video mode.\\n\\n## Cost and credits\\n\\nDo not call `videodraft credits` before routine generations. Paid endpoints validate and deduct atomically; if the balance is insufficient, the request is rejected before the provider job starts (CLI exit code 4). Check the balance only when the user asks, gives a credit budget, or a large workflow needs budget planning.\\n\\nFor expensive work, estimate with `--estimate` or `videodraft costs`, state the selected model/settings/cost, and get a go-ahead. This matters most for shot-image batches, long or high-resolution video, AI Production, and paid audio batches. Honor the user\'s confirmation preference for the session.\\n\\nKling voice creation costs 1 VideoDraft credit on the platform Fal account and 0 credits with Fal BYOK. Preview it with `videodraft kling-voices create <sample> --name <name> --estimate`; the estimate does not create a voice or require consent confirmation. Actual creation requires `--confirm-consent`.\\n\\nMiniMax H3 costs 5 / 6 / 13 / 16 credits per output second at 480p / 768p / 2K / 4K (768p by default). In reference mode the first 5 images are included and each additional image costs 8 credits. Reference video and reference audio are NOT billed.\\n\\nMiniMax H3 Max costs 5 credits per output second at 480p and 8 credits per second at 768p (default), plus pooled reference tokens in reference mode: the first 4,096 are free and every 1,000 after that costs 2 credits, where an image is `(width x height) / 1024` tokens, a reference video is 2,886 tokens per second at 480p or 7,459 at 768p, and reference audio is about 2,121 per second. Use `--prompt-expansion-mode disabled|balanced|quality`; balanced is the default. Fal\'s provider safety checker is OFF by default and is not exposed in the app \u2014 only pass `--safety-checker true` if the user explicitly asks for it.\\n\\nWan 3.0 costs 7 / 14 / 28 credits per output second at 480p / 720p / 1080p. Auto duration reserves the 30-second maximum and refunds the unused reserve after the provider reports the actual whole-second output length. Fal BYOK runs on the connected user key and charges zero VideoDraft credits.\\n\\nGrok Imagine Video 1.5 costs 8 credits per output second at 480p, 14 at 720p, or 25 at 1080p, plus 1 credit for each first-frame or reference image. Native audio is always generated.\\n\\n`videodraft models image|video` lists the live image and video catalogs with supported inputs. Video entries are grouped as `generation`, `video_edit`, `motion_control`, `avatar_lipsync`, and `upscale`, and each reports the exact tool. Use `videodraft models video --category video_edit` to narrow the list. `videodraft models audio` lists Seed Audio, Google Lyria, and ElevenLabs audio/media tools, while `videodraft models voices` helps discover TTS voices. Consult them for capabilities; a supplied ElevenLabs voice ID does not need to appear in the voice catalog.\\n\\n## Async jobs\\n\\nImage/video generation is asynchronous: commands submit a job and **wait by default**, printing output URLs (and saving files with `--download`). Large downloaded images also get a downscaled copy in `previews/` next to them (the `preview` field / \\"inspect via preview\\" line in the output) \u2014 **look at the preview, deliver the original**; viewing full-resolution images bloats the chat permanently. In scripts/CI prefer explicit control:\\n\\n```bash\\nJOB=$(videodraft generate image \\"...\\" --no-wait --json | jq -r .job_id)\\nvideodraft wait \\"$JOB\\" --download \\"./outputs/{job_id}_{index}.{ext}\\" --json\\n```\\n\\nFor MANY jobs: submit each with `--no-wait`, collect ALL with one command \u2014 `videodraft wait <id1> <id2> ...` polls every job from one process with one batched request per tick. Do NOT spawn parallel `wait`/`generate --wait` processes for a batch.\\n\\nIf a wait times out, the job is still running server-side \u2014 `videodraft status <job_id>` later. Never re-submit just because a wait timed out (that double-spends credits).\\n\\nFor completed Wan 3.0 jobs, MCP `check_generation_status` and CLI `status`/`wait --json` include `outputMetadata` with Fal\'s returned `seed`, `duration`, and `actual_prompt` when present.\\n\\n## AI Studio sessions\\n\\nStandalone image/video/audio generations are filed into an AI Studio session in the web app. 3D assets use `assets 3d list/get` separately. You do not have to create an AI Studio session:\\n\\n- **MCP hosts** (Claude Code, claude.ai, Codex, VideoDraft ADE): on 2025-era MCP the server mints an `Mcp-Session-Id` on `initialize`; your host echoes it, and this conversation\'s generations land in their own session. On stateless MCP (2026-07-28) hosts that send a conversation id (ChatGPT) get the same. Otherwise `name_current_ai_studio_session` returns `pass_session_id: true` with a `session_id`: pass that `session_id` to every later standalone generation in the conversation. Tool results echo the session as `ai_studio_session_id`.\\n- **CLI**: the same handshake runs once per (profile, server, working directory) and is cached for 12 idle hours, so everything generated from one directory shares one session. `videodraft sessions current` shows it; `videodraft sessions reset` starts a new one.\\n- Project generations (`--project <id>` / `project_id`) always go to that project\'s session.\\n\\nOnce you understand the creative task, name the current automatic session before the first standalone generation:\\n\\n```bash\\nvideodraft sessions name \\"Purple Seal Rescue Short\\"\\n```\\n\\nChoose a concise, specific 3-6 word title for the intended work. Do not copy the client name, date, or exact chat title. Name it once: the operation creates the session with that title. If generation, a user, or an earlier agent created the session first, its existing name is preserved.\\n\\nApart from the `pass_session_id: true` case above, pass `--session <id>` / `session_id` only to **continue earlier work** or create an explicit separate group:\\n\\n```bash\\nSESSION=$(videodraft sessions create \\"Fox brand explorations\\" --json | jq -r \'.session.id\')\\nvideodraft generate image \\"a red fox in snow, cinematic\\" --session \\"$SESSION\\"\\nvideodraft generations --session \\"$SESSION\\" # what is in it\\nvideodraft sessions list --name fox # find it again later\\n```\\n\\n`VIDEODRAFT_SESSION=<id>` sets the default for every command (ignored when `--project` is given). While it is set, `sessions name` refuses to run because that command names the current automatic connection session, not the pinned override; unset it first or rename the pinned session in AI Studio. `VIDEODRAFT_SESSION_SCOPE=<label>` groups several directories into one connection session; `VIDEODRAFT_CLIENT_NAME=<host>` labels the fallback placeholder used when generation creates the session before `sessions name`; `VIDEODRAFT_NO_SESSION=1` disables the handshake. Generations that reach the server with no session at all fall back to the account-wide \\"Agent (MCP)\\" session; if you see work landing there, pass `--session` explicitly.\\n\\n## Generation history\\n\\nPast work is queryable \u2014 reuse a previous setup instead of guessing. `videodraft generations` lists recent generations; scope with `--session <id>` or `--project <id>` (includes collaborators\' rows in shared scopes; pass one, project wins), and filter with `--type`, `--model`, `--favorites`. `--full --json` returns each row\'s exact parameters (aspect ratio, resolution, duration, references) \u2014 the human table stays compact, so pair `--full` with `--json`. `videodraft generation <id>` prints one generation\'s complete recipe (prompt, input image, parameters, outputs); `--favorite` / `--unfavorite` stars it. `videodraft sessions list` shows AI Studio sessions (owned + shared) with `--name` search \u2014 take a session id from there to read its history.\\n\\n## Free stock footage and photos\\n\\nPexels and Pixabay are wired in through `search_stock_media` / `import_stock_media` (`videodraft stock search \\"<query>\\"`, `videodraft stock import <ref>`). Both libraries are free, watermark-free and cost **zero credits**, so use stock for b-roll, establishing shots, backgrounds and ordinary real-world footage, and spend credits on the shots that must be specific: the product, the character, the scripted action.\\n\\n```bash\\nvideodraft stock search \\"city skyline night\\" --min-duration 5 --orientation landscape\\nURL=$(videodraft stock import pexels:video:35379336 --quality hd --json | jq -r .url)\\n```\\n\\n- Always two steps. A search result\'s `preview_url` is a thumbnail for judging the shot; never place it on a timeline, attach it to a shot or send it to a model. `import_stock_media` copies the file onto the VideoDraft CDN and returns the URL every other surface accepts, including the native editor\'s `library_manage` import.\\n- `--quality hd` (default) caps video at 1080p; `4k` caps at 2160p, `best` takes the largest the provider has, `sd` suits rough cuts. Stills ignore quality and import at full size. Use `--orientation portrait` for 9:16, and `--min-resolution 1920` to drop anything below HD on its long edge.\\n- Photos come from Pexels. Pixabay contributes video only, because its full-size image host refuses server-side downloads.\\n- Credit the creator and link the provider page when you show results or deliver the finished work; both come back on every result. Skip clips that imply a person or brand endorses the product, and avoid recognisable logos in ads.\\n- Search and import are rate limited per user (30 searches and 15 imports a minute) and searches are cached for 24h, so a burst of searches while planning a montage is fine. Bulk downloading a stock library is prohibited by both providers. A note saying Pixabay was skipped means its provider budget for that minute is spent; the Pexels results still stand.\\n\\n## Downloading from YouTube, Instagram, TikTok and other sites\\n\\nVideoDraft ships no downloader. When the user asks to pull a video from YouTube, Instagram, TikTok, X, LinkedIn or anywhere else, `yt-dlp` on the user\'s own machine does the work. Check for it with `command -v yt-dlp` before promising anything.\\n\\n**Present:** pin the format. On its defaults yt-dlp takes the best stream, usually VP9 or AV1 in WebM, which the native editor\'s import rejects (mp4, mov and m4v only). This returns one pre-muxed H.264 + AAC MP4 and needs no ffmpeg:\\n\\n```bash\\nyt-dlp -f \\"b[ext=mp4]\\" -o \\"media/%(title)s.%(ext)s\\" \\"<url>\\"\\n```\\n\\n**Missing:** say so in one line, then offer `brew install yt-dlp` only where `brew` exists, and only after asking. With no Homebrew, say the tool is unavailable and stop.\\n\\n- Never install Homebrew, never `sudo`, never write to `/usr/local/bin` or `/opt`, never pipe a script into a shell.\\n- `ffmpeg` is optional, matters only above 720p, and the selector above never needs it.\\n- Auth-gated content needs `--cookies-from-browser`; raise it only after a download fails on auth.\\n- Fetch only what was asked for, never a channel, playlist or back catalogue.\\n\\n## Local files and reference images\\n\\nReference inputs must be public URLs. The CLI uploads local files automatically wherever a URL is expected (`--ref photo.jpg`, `--start-image frame.png`), or explicitly:\\n\\n```bash\\nURL=$(videodraft upload ./product.png --json | jq -r .url)\\n```\\n\\nNever silently drop a reference you couldn\'t upload \u2014 stop and tell the user. Never upload a user\'s file to a third-party host.\\n\\nWhen the user attaches media for a native production, import actual footage into the editor by default. For hosted generation/storyboarding, classify each item before acting: a recurring **visual asset** (character/product/location/style), actual **footage to place as shots**, or **inspiration only**. See [references/pipeline.md](references/pipeline.md) for the hosted role mapping.\\n\\n## Showing media to the user\\n\\nGenerated media is **not** displayed in the chat automatically \u2014 you decide what to show. To preview an asset inline, save it locally (use `--download` so it lands under `media/`) and reference its **local path** as a Markdown link with a **leading `./`**:\\n\\n```\\n[ferrari shot](./media/ferrari_01.png) \u2190 image card\\n[the clip](./media/clip.mp4) \u2190 video player\\n[voiceover](./media/vo.mp3) \u2190 audio player\\n```\\n\\nPut the Markdown link **in your message text** \u2014 video and audio embed exactly like images. Do **not** use `SendUserFile` (or other file-send tools) to display media: that renders inside a collapsible tool card and gets buried in the tool list. The Markdown link in your prose is what produces the inline card.\\n\\nUse the path you saved to: a **workspace-relative** path (`./media/clip.mp4`, or `./<any-folder>/clip.mp4` \u2014 any folder in the workspace works), or the **absolute** path for a file outside the workspace (e.g. `/Users/you/Desktop/clip.mp4` or another workspace\'s path). Both render. Show the finished results worth showing (and only those \u2014 not every intermediate job). A bare CDN URL or a JSON dump of output URLs does **not** render; the local-path Markdown link is what produces an inline card.\\n\\n## Native-first VideoDraft ADE pipeline (idea \u2192 MP4)\\n\\nWhen `videodraft_editor` is present:\\n\\n1. Generate or source the script, storyboard, shot images, clips, voiceovers, music, and other assets through the cloud CLI/MCP as needed.\\n2. Call native `project_manage` (`project`, action `open` or `create`) to open or create the `.vdproject`.\\n3. Call native `library_manage` `import`, wait for imports to become ready, then assemble and refine the timeline with `edit_apply` recipes.\\n4. Call native `delivery_manage` `submit` and use its `jobs` operation for progress and results.\\n\\nDo not run the hosted production or export steps in this path unless the user explicitly asks for a web production.\\n\\n## Hosted fallback pipeline (idea \u2192 MP4)\\n\\nUse this only when there is NO native editor at all, or the user explicitly requests the hosted web\\nworkflow. The native surface is not only the injected `videodraft_editor` MCP: a `videodraft-editor`\\nexecutable on PATH is the same editor reached through its terminal bridge, and\\n[references/editor.md](references/editor.md) covers driving it that way. Treating a missing MCP as\\n\\"no editor\\" sends sessions that have the binary into hosted production for no reason.\\n\\n```bash\\nvideodraft create \\"<idea>\\" --ar 9:16 # project: script \u2192 visual assets \u2192 storyboard\\nvideodraft shots <project_id> --grid --estimate # cost preview, confirm with user\\nvideodraft shots <project_id> --grid # batch shot images (waits, writes onto shot cards)\\nvideodraft produce <project_id> # voiceovers + captions + production timeline\\nvideodraft export <project_id> --download final.mp4\\n```\\n\\nOptional between produce and export: per-shot motion clips (`videodraft generate video ... --project <id>` then place it with `videodraft attach <project> --scene N --shot M --media <url|file> --type video --duration <s>`), music (`videodraft generate music \\"...\\" --attach <project_id>`), and standalone audio assets (`generate audio`, `generate sound-effect`, `generate dialogue`, `generate voice-changer`, `generate dub`). Details, per-step tools and editing rules: [references/pipeline.md](references/pipeline.md).\\n\\n## Avatar and talking-head videos (both surfaces)\\n\\nAvatar generation is cloud-only \u2014 the native editor has no avatar or lipsync tools \u2014 so this applies whether or not `videodraft_editor` is present. Generate the avatar in the cloud; in VideoDraft ADE, import the rendered clip and cut it on the native timeline like any other footage.\\n\\nAvatar/talking-head videos use dedicated commands. For a reusable managed avatar, obtain or generate a clear portrait \u2192 `videodraft avatar script` when needed \u2192 `videodraft avatar create` \u2192 `videodraft avatar render --resolution 720p`. For a one-off portrait, use `videodraft avatar fabric <portrait> --text \\"...\\"` or `--audio <file>`, or `videodraft avatar h3-lipsync <portrait> --audio <file>` for a short high-resolution clip from existing audio. For an existing video plus replacement audio, use `videodraft avatar lipsync <video> --audio <file>`. Managed script/creation is bundled/free; direct Fabric, H3 Max Lip Sync, Sync, the managed Fabric render, and optional portrait generation/upscaling are paid. Confirm expensive steps first.\\n\\n## Working with hosted project data\\n\\nA hosted project is one JSON blob (script, storyboard scenes, shot cards, visual assets, production timeline). To inspect: `videodraft projects get <id>`. To edit: fetch `--raw`, modify, then `videodraft call update_project` \u2014 objects deep-merge, **arrays replace wholesale** (send the complete `storyboard.scenes` array to change one scene). Snapshot first with `videodraft checkpoint create <id>` before risky edits. Schema reference: `videodraft call get_project_schema`. This does not replace native editor tools when `videodraft_editor` is available for the production itself.\\n\\n## More\\n\\n- [references/pipeline.md](references/pipeline.md) \u2014 hosted fallback data model and production workflow\\n- [references/editor.md](references/editor.md) \u2014 native headless editor routing, project selection, import, timeline edits, verification, and export\\n- [references/models.md](references/models.md) \u2014 choosing image/video models, pricing patterns, voices and styles\\n- [references/examples.md](references/examples.md) \u2014 recipes: batch product videos from a CSV, talking-head from a script, changelog video in CI\\n","references/3d.md":"# Standalone 3D assets through MCP and CLI\\n\\nUse this lane for downloadable meshes, textured 3D models, and humanoid rigging. These are independent of AI Studio sessions and do not require a storyboard or web viewer. Never create a project just to generate a mesh. Native editor media import does not establish support for GLB/FBX assets; use external 3D software unless the live editor schema explicitly supports them.\\n\\n## Discover and quote\\n\\n```bash\\nvideodraft models 3d --json\\nvideodraft generate 3d \\"a worn brass telescope\\" --model meshy-7 --estimate\\nvideodraft generate 3d --model tripo-h3.1 --ref ./reference.png --options @options.json --estimate\\nvideodraft rig 3d --estimate\\n```\\n\\nMCP tools: `get_3d_models`, `estimate_3d`, `generate_3d`, `rig_3d`, `check_generation_status`, `list_3d_assets`, `get_3d_asset`, `create_3d_upload`, and `finalize_3d_upload`.\\n\\nThe generator catalog is deliberately limited to `meshy-7` and `tripo-h3.1`. Both expose text/image/multi-image modes; use the live schemas for option availability, view order, limits, topology, texture/PBR, and pose controls. Do not invent common provider option names. Server defaults prioritize quality. Rigging is a separate Meshy operation for compatible textured humanoid GLBs; it is not an arbitrary creature or facial rig service.\\n\\nEvery option is accessible through `--options` JSON (`@path.json` supported), plus repeatable `--option key=value` overrides. Values parse as JSON when valid. Options and explicit input mode affect pricing, so obtain a fresh quote for the exact recipe. Estimates perform no file upload or provider generation. VideoDraft bills whole credits at 100 credits per dollar using the server\'s rounding rules. BYOK costs zero VideoDraft credits and runs on the connected user Fal key. Do not infer the user\'s Fal-account charges from a zero VideoDraft-credit quote.\\n\\n## Generate and retrieve\\n\\n```bash\\nvideodraft generate 3d \\"a worn brass telescope\\" --model meshy-7 --download --json\\nvideodraft generate 3d --model meshy-7 --ref ./hero.png --no-wait --json\\nvideodraft generate 3d --model tripo-h3.1 --input-mode multi_image --ref ./front.png --ref ./left.png --ref ./back.png --ref ./right.png --no-wait --json\\nvideodraft status JOB_ID --json\\nvideodraft wait JOB_ID --download --json\\nvideodraft assets 3d list --status completed --json\\nvideodraft assets 3d get ASSET_ID --download ./media/3d --json\\nvideodraft rig 3d --asset ASSET_ID --download --json\\nvideodraft rig 3d ./textured-humanoid.glb --download --json\\n```\\n\\nLocal image references upload automatically. Local rigging inputs must be self-contained `.glb` files and upload via the dedicated 3D route. Input mode is inferred from reference count unless `--input-mode text|image|multi_image` is supplied. A text prompt requires no images; image mode requires exactly one and no geometry prompt. Meshy multi-image accepts 1-4 views, and Tripo accepts 2-4 in front, left, back, right order. Use Meshy\'s `texture_prompt` option when texturing guidance is needed with image inputs. Do not mix unrelated images as if they were views of the same object.\\n\\nGeneration and rigging wait by default. `--no-wait` returns a persisted job that continues server-side. For many jobs use one `wait ID1 ID2 ... --download`; do not run parallel polling processes. Recover timeout with the same job ID, not another paid generation.\\n\\nSubmission returns a `request_id` UUID. A same-request retry must use `--request-id UUID` and the original arguments. The CLI journals resolved uploads in its private config directory so retries reuse the same remote files. Changed inputs under that UUID are rejected. Submission errors include the UUID in JSON `details.request_id` where available. Never retry a failed request with a new UUID until the prior job status is known.\\n\\n## Complete artifact packages\\n\\n`--download` defaults to `media/3d/<asset_id>/`. A user directory is respected and receives a per-asset folder. `{job_id}` and `{asset_id}` directory templates are accepted. Do not pass a `.glb` filename or a media `{index}.{ext}` template: a 3D result can contain multiple formats and texture/material dependencies.\\n\\nDownloads preserve every returned artifact. The manifest records provider URLs, source filenames, local paths, role, content type, and actual downloaded sizes. glTF/OBJ/MTL references are rewritten to saved dependency paths; binary GLB/FBX and provider ZIPs are retained as returned. Supplied dependency aliases become local copies at the paths needed by original FBX files. Unsafe or colliding aliases produce warnings. Existing packages are never overwritten. A successful package has `manifest.json`; read its `complete` field and `warnings`, plus CLI `package_warnings`. Do not claim a package with unresolved dependencies is ready for offline use.\\n\\nMachine results use `type: \\"model3d\\"` with `output_files` for files and `output_media` only for ordinary preview media. Texture maps and models are not image/video gallery entries. Show a supplied preview via a local image Markdown link and deliver the original mesh/package files. Do not imply an interactive 3D viewer exists in AI Studio or VideoDraft ADE.\\n","references/editor.md":"# Native VideoDraft Editor reference\\n\\nUse this reference when the user wants to assemble, cut, caption, mix, lay out, inspect, or export a local VideoDraft Editor project. The native editor is deterministic and local. Cloud generation remains in the `videodraft` CLI or hosted MCP.\\n\\n## VideoDraft ADE preference rule\\n\\nWhen `videodraft_editor` tools are exposed, treat the native editor as available and make it the default surface for production, timeline assembly, and final export. It is headless by design, so a hidden window or an untouched Open Editor button does not justify using hosted production instead.\\n\\nUse cloud tools for asset generation and optional script/storyboard work, then import the results. Do not call hosted `produce_project` / `videodraft produce` or `export_video` / `videodraft export` unless the user explicitly requests an editable web production or the native editor tools are unavailable. If a native tool call fails after the editor was available, report or recover that native failure rather than silently switching surfaces.\\n\\n## Choose the correct surface\\n\\n- `videodraft` and the hosted VideoDraft MCP generate assets and can manage hosted web projects. They use the user\'s VideoDraft account and credits. In VideoDraft ADE, use them mainly as the source of generated media and optional storyboards for the native production.\\n- `videodraft_editor` edits local `.vdproject` packages. It has no generation, account, model, or credit tools.\\n- Inside VideoDraft ADE on a supported Mac, the editor MCP is injected automatically for Claude, Codex, OpenCode and Grok in both Code and VideoDraft modes. It starts headlessly before the chat opens. The user does not need to click Open Editor, and closing or hiding the editor window does not stop headless editing.\\n- Outside that environment, use the editor only if `videodraft_editor` MCP tools are already exposed or the `videodraft-editor` executable is on PATH. Do not confuse the public `videodraft` cloud CLI with the separate native editor executable.\\n\\nPrefer the direct MCP tools when they are available. The terminal bridge is useful for scripts, diagnostics, or an agent session where the MCP was not injected.\\n\\n## Which editor tools you have\\n\\nCurrent editors expose eight workflow tools: `project_manage`, `edit_snapshot`, `edit_apply`, `edit_undo`, `library_manage`, `inspect`, `speech_apply`, and `delivery_manage`. Use them whenever `edit_apply` is available. Everything below describes them.\\n\\nEarlier editors expose a different set of tools. If `edit_apply` is not in your catalog, follow the live tool descriptions; the working rules below still apply.\\n\\nThe live tool descriptions and schemas are authoritative. When they differ from this reference, follow them.\\n\\nService tools (`project_manage`, `library_manage`, `inspect`, `speech_apply`, `delivery_manage`) take `operation` and `parameters`, plus an optional `context` (see [Keep a reliable editing model](#keep-a-reliable-editing-model)):\\n\\n```json\\n{\\"operation\\": \\"import\\", \\"parameters\\": {\\"source\\": {\\"path\\": \\"/Users/me/clips\\"}}}\\n```\\n\\n### Parameters at a glance\\n\\nEach operation\'s schema has the full list. These are the fields that matter most, and the ones most often confused:\\n\\n| Tool and operation | Key `parameters` |\\n| --- | --- |\\n| `project_manage` `project` | `action` (`list`, `open`, `create`, `close`); `id`, `name` or `path` to open; `name`, `fps`, `aspectRatio`, `quality` to create |\\n| `edit_snapshot` (no `parameters`) | `includeLibrary`, a `startFrame`/`endFrame` window, `offset`/`limit` paging, `context` for later pages |\\n| `edit_apply` (no `parameters`) | `actions`; optional `context`, `requestId`, `previewOnly` |\\n| `library_manage` `import` | `source` with one of `path`, `url`, `bytes` + `mimeType`, or `matte`; optional `name`, `folder`. A `.srt` or `.vtt` file becomes captions |\\n| `library_manage` `list` | `ids` to poll imports, `pending`, `folder` |\\n| `inspect` `timeline` | a `startFrame`/`endFrame` window; there is no clip filter |\\n| `inspect` `frame` | `startFrame` for one frame; add `endFrame` and `maxFrames` to sample a range |\\n| `inspect` `media` | `mediaRef` (a library asset, not a file path), `startSeconds`/`endSeconds` in source seconds, `wordTimestamps`, `overview` |\\n| `inspect` `transcript` | a `startFrame`/`endFrame` window, `granularity` (`words` or `segments`), `clipId` |\\n| `inspect` `color` | `clipId` with `atFrame`, or `mediaRef`; optional `reference` |\\n| `speech_apply` `words` | `words`: transcript word indices, each one index or a `[first, last]` pair; or `matches`: exact words to remove everywhere; `pacing` |\\n| `speech_apply` `silence` | none |\\n| `speech_apply` `captions` | caption `style`, `position` and look fields; it finds the speech itself. `style` `plain` (sentences without a preset) takes `positionY` or `transform.centerY`, not `position` or `punctuation` |\\n| `delivery_manage` `submit` | `mode`, `codec`, `resolution`, `outputPath` (its folder must already exist); `captionGroupId` and `wordTiming` for `srt` and `vtt`; `check: false` skips a video\'s export check |\\n| `delivery_manage` `jobs` | `action` `list` (every job with its progress, warnings, output path and check findings; with a `jobId`, that job with all its findings), `cancel` with a `jobId`, or `dismiss` with a `jobId` and optional `findingIds` |\\n\\n`previewOnly` exists only on `edit_apply`. Service operations apply immediately.\\n\\n### `edit_apply` actions at a glance\\n\\nEvery action puts its fields beside `kind`, for example `{\\"kind\\": \\"adjust\\", \\"clipId\\": \\"c1\\", \\"speed\\": 2}`. `remove`, `adjust`, `replace`, `transition`, `mask`, `text`, `grade` and `effects` take `clipId` for one clip or `clipIds` for several; `animate`, `extract`, and the rows of `move` and `split` take a single `clipId`. Advanced actions cannot take `as`, but their clip ids accept `@name`. The one exception is `transition`, whose own `kind` field names the style, so its fields go inside `parameters`: `{\\"kind\\": \\"transition\\", \\"parameters\\": {\\"clipId\\": \\"c2\\", \\"kind\\": \\"dipToBlack\\"}}`. These `actions` place a clip, warm it and add a vignette:\\n\\n```json\\n[{\\"kind\\": \\"place\\", \\"assetId\\": \\"\u2026\\", \\"atFrame\\": 0, \\"as\\": \\"shot\\"},\\n {\\"kind\\": \\"grade\\", \\"clipId\\": \\"@shot\\", \\"adjustments\\": {\\"temperature\\": 7500}},\\n {\\"kind\\": \\"effects\\", \\"clipId\\": \\"@shot\\", \\"effects\\": [{\\"type\\": \\"finish.vignette\\", \\"params\\": {\\"strength\\": 35}}]}]\\n```\\n\\n| Advanced `kind` | Key fields |\\n| --- | --- |\\n| `place_batch` | `entries: [{mediaRef, startFrame, endFrame or source, trackIndex}]`, not the concise `assetId`, `atFrame`, `durationFrames`; leave `trackIndex` off every entry for a new track |\\n| `insert_batch` | `trackIndex`, `atFrame`, `entries: [{mediaRef, durationFrames or source}]`; later clips move right |\\n| `move` | `moves: [{clipId, toFrame, toTrack}]` |\\n| `remove` | `clipId` or `clipIds`; leaves a gap |\\n| `split` | `splits: [{clipId, atFrame}]`, or `trackIndex` with `frames` |\\n| `extract` | `trackIndex` with `ranges: [[start, end]]` in frames, or `clipId` with `ranges` and `units` (`frames`, or source `seconds`); closes the gap |\\n| `adjust` | `clipId` or `clipIds` plus any of `durationFrames`, `trimStartFrame`, `trimEndFrame`, `speed`, `speedCurve` (`{preset}`, or `{points: [{t, rate}]}` with `t` from 0 to 1 along the source), `preservesPitch`, `volume`, `opacity`, `transform` (`centerX`, `centerY`, `width`, `height`, `flipHorizontal`, `flipVertical`), `blendMode` |\\n| `animate` | `clipId`, `property` (`volume`, `opacity`, `rotation`, `position`, `scale`, `crop`), `keyframes: [[frame, ...values]]` with frames counted from the clip\'s start; `position` is the top-left corner |\\n| `arrange` | `layout`, `slots: [{slot, clipIds or mediaRef, anchor}]`, `fit` (`fill` or `fit`); `mediaRef` slots also need `endFrame` |\\n| `transition` (fields inside `parameters`) | `clipId` or `clipIds` (the clip after each cut), `kind` (`crossDissolve`, `dipToBlack`, `dipToWhite`, `blurDissolve`, `push`, `linearWipe`, `whipPan`, `crossZoom`, or `none` to remove), `durationFrames`, `params` (`direction` 0 to 3 for left, right, up, down; `feather`; `blur`; `intensity`) |\\n| `mask` | `clipId` or `clipIds`, `shape` (`rectangle`, `ellipse`, `none`), `centerX`, `centerY`, `width`, `height` (0 to 1 of the clip\'s own frame), `feather`, `strength`, `inverted` |\\n| `replace` | `clipId` or `clipIds`, `mediaRef` (a library asset of the same kind), `trim` (`keep`, the default, or `reset`), `linkedAudio` (`follow`, the default, or `keep`) |\\n| `tracks` | `reorder: [{trackId, to}]`, `set: [{trackId, muted, hidden, syncLocked}]`, `remove: [{trackId}]`; there is no add, since placing without a track makes one |\\n| `titles` | `entries: [{startFrame, endFrame, content}]`, optionally with `trackIndex` (on every entry or none), `style`, `animation`, `transform` and typography (`fontName`, `fontSize`, `color`) |\\n| `text` | `clipId`, `clipIds` or `captionGroupId`, with `content`, `style`, `position`, `animation` or typography |\\n| `grade` | `clipId` or `clipIds`, `adjustments`, `wheels`, `curves`, `hueCurves`, `lut`, `reset` |\\n| `effects` | `clipId` or `clipIds`, `effects: [{type, params, enabled}]`, `remove: [type]` |\\n\\n- `grade`: `adjustments` holds `exposure` (EV, -4 to 4), `temperature` (kelvin, 1800 to 15000, 6500 neutral, higher is warmer), `tint` (-150 magenta to 150 green); `contrast`, `highlights`, `shadows`, `whites`, `blacks`, `vibrance` and `saturation` are signed percents (-100 to 100, 0 neutral), not factors like 1.2. `wheels`: `shadows`, `midtones`, `highlights`, each `{hue, strength, brightness}`. `curves`: `luma`, `red`, `green`, `blue` as `[x, y]` points from 0 to 1. `hueCurves`: `targets: [{hue, rotate, saturation, lightness}]`. `lut`: `{path, mix}`, with the `storedPath` from `prepare_look`. Out-of-range values are refused.\\n- Effect `type` ids and their knobs, which go inside `params` and run 0 to 100 unless noted. Defaults leave the picture unchanged, so send the knob you want to see:\\n - `finish.vignette`: `strength` and `curvature` (-100 to 100; positive `strength` darkens the edges), `size`, `falloff`\\n - `finish.grain`: `strength`, `grainSize` (0.5 to 6 px); `finish.glow`: `strength`, `haloRadius` (px), `cutoff`, `halation`\\n - `defocus.gaussian`: `blurRadius` (px); `defocus.motion`: `streakLength` (px), `streakAngle` (-180 to 180 degrees)\\n - `texture.clarity`: `localContrast`, `hazeRemoval` (-100 to 100); `texture.sharpen`: `strength` (0 to 200); `texture.denoise`: `strength`\\n - `matte.chroma`: `screenHue` (0 to 360 degrees, 120 is green), `range`, `edgeSoftness`, `spillSuppression`\\n- `arrange` layouts and their slots (fill every slot): `fullscreen` (`stage`); `split` (`left`, `right`); `stack` (`top`, `bottom`); `corner_top_left`, `corner_top_right`, `corner_bottom_left`, `corner_bottom_right` (`stage`, `corner`); `quad` (`top_left`, `top_right`, `bottom_left`, `bottom_right`); `side_panel` (`stage`, `panel`); `thirds` (`left`, `middle`, `right`). The layout picks the corner; `anchor` only biases the crop.\\n- `trackIndex` and `toTrack` are an existing track\'s `order` from `edit_snapshot` (0 draws on top), counted at that point in the recipe: a track an earlier action adds shifts them. `tracks` takes `trackId` handles instead.\\n\\n## Start with the intended project\\n\\nAn MCP session can begin without a project selected. Project selection belongs to the session, not to whichever editor window happens to be frontmost.\\n\\n1. If the user named an existing project but its identity is unclear, call `project_manage` with `operation: \\"project\\"` and `parameters.action: \\"list\\"`.\\n2. Open the exact project with `action: \\"open\\"` and the returned `id`, an unambiguous `name`, or the `.vdproject` `path`.\\n3. Create only when the user wants a new local edit. `action: \\"create\\"` accepts optional `name`, `fps`, `aspectRatio`, and `quality`.\\n4. Treat `isActive` as this session\'s target and `isVisible` as the project shown in the UI. Headless editing only needs the session target.\\n5. Use `action: \\"close\\"` only when closing is part of the task. It saves first and never deletes the project. Afterwards other calls answer `no_project` until you open or create a project again.\\n\\nOther `project_manage` operations: `create_timeline` and `select_timeline` for additional timelines, and `configure` for project settings (a frame-rate change applies to every timeline).\\n\\nDo not substitute a hosted project ID for a native project. A hosted project can supply scripts, storyboards, and generated media, but the native edit is a separate `.vdproject` package.\\n\\n## Keep a reliable editing model\\n\\n- Opening or creating a project, and creating or selecting a timeline, returns `data.snapshot` (the `edit_snapshot` view with the library: format, tracks, clips and assets) and a `context`, so you can edit right away. Call `edit_snapshot` only after an out-of-band user edit, to page a long timeline with `offset` and `limit`, or for handles you lack.\\n- `context` (`epoch`, `timelineId`, `revision`) is optional on every write except `edit_undo` and `project_manage` `project`, which take none. Without it, a write applies to the current state. With it, the editor refuses the write if the project changed since that context, which is what you want when a person may be editing at the same time. Every reply returns a fresh `context`. After `context_expired` (a reopen or restart), take a new snapshot.\\n- Timeline positions and durations are frames at `format.fps`. `sourceSeconds` values are seconds in the source file. Pass values as returned; do not multiply or divide by fps yourself.\\n- IDs are short handles. Pass them back exactly as returned; do not derive them from UUIDs. Tracks keep stable handles; indexes can change.\\n- When the user says \\"this clip\\", \\"these captions\\", or \\"here\\", take a fresh `edit_snapshot` (a one-frame window is enough). Its `selection` names what they selected in the project\'s window, in the snapshot\'s handles: `clipIds`, `captionGroupIds`, a `gap`, the marked `range`, and library `mediaIds`; no `selection` means nothing is selected. `currentFrame` is the playhead. `visible: false` means the user is not looking at this project.\\n- Send writes serially. Parallel writes against one project race each other\'s context.\\n- A refused call answers `status: \\"rejected\\"`: nothing changed, so fix what the message names and send it again. A service write that answers `failed` may have applied before a save or connection failed; inspect before repeating it.\\n- Use `inspect` for detail and verification: `timeline` for exact clip and track properties, `frame` for rendered frames of the composited result, `media` before describing source content, `color` for scopes, and `transcript` to locate spoken words.\\n- Volume values, including volume keyframes, are linear from `0` to `1`.\\n\\n## Edit with recipes\\n\\n`edit_apply` takes an ordered list of `actions`, plus an optional `context` and `requestId`. The editor validates the whole recipe before changing anything, then applies it as one undo step. If any action fails, nothing changes and the error names the action. An applied reply lists the resulting `clips` (position, length, source window, text); use it to confirm the edit instead of re-reading.\\n\\n- Concise actions: `place` (an `assetId` at `atFrame`, optionally on a `trackId`, with `durationFrames` or `sourceSeconds`, `mode` `overwrite` or `insert`), `trim` (a `clipId` to a `sourceSeconds` window), and `title` (`text` at `atFrame` for `durationFrames`, optional `look` and `motion`). Name an action with `as` and address its clip in later actions as `@name`.\\n- Advanced actions put their fields beside `kind` too: `place_batch`, `insert_batch`, `move`, `remove`, `split`, `extract`, `adjust`, `animate`, `arrange`, `transition` (fields inside `parameters`), `mask`, `replace`, `tracks`, `titles`, `text`, `grade`, and `effects`. Their field help lists every supported effect, grading control, mask, layout, speed curve, and caption control.\\n- Use `arrange` for split screens, picture-in-picture, grids, and canvas placement, and `tracks` to fix stacking. Do not synthesize layouts from generic transforms or keyframes.\\n- Use `replace` to swap a clip\'s media and keep its timing, effects, keyframes, and links, for example a re-render of the same shot. `trim: \\"keep\\"` holds the clip\'s source offsets; `trim: \\"reset\\"` plays the new media from its first frame and keeps linked audio in sync.\\n- Put every edit a request needs into one recipe: placing, trimming, titling, grading and effects together are one call and one undo step. Use `previewOnly: true` to validate a large or risky recipe without editing.\\n- Send a `requestId` when a retry must not edit twice: repeating the identical request with the same `requestId` returns the original receipt, even after a dropped connection. Without one, a repeated request applies again. Use a new `requestId` for a different edit.\\n- `applied_unsaved` means the edit applied but saving failed. If you sent your own `requestId`, repeat the identical request to retry the save, not the edit. Without one, do not resend: a repeat would apply the edit again.\\n- `edit_undo` reverses this connection\'s latest edit while it is still the top undo step. It never undoes the user\'s own edits. Take a new snapshot afterward.\\n\\nEdits are undoable. Do not ask for confirmation before each ordinary edit. Ask one focused question only when the user\'s creative direction is materially ambiguous.\\n\\n## Bring generated or local media into the editor\\n\\nFor b-roll and establishing shots, search free stock first (`videodraft stock search \\"<query>\\"`), import the pick with `videodraft stock import <ref>`, and feed the returned CDN URL to `library_manage` `import` as `source.url`. It costs no credits.\\n\\nOtherwise use cloud generation for new assets, save or download the outputs, then call `library_manage` with `operation: \\"import\\"`:\\n\\n- `source.path`: absolute local file or directory. A directory imports recursively and preserves its folder structure.\\n- `source.url`: HTTPS asset URL. Set `mimeType` when a signed URL has no usable extension.\\n- `source.bytes`: small base64 media with a required `mimeType`.\\n- `source.matte`: generated solid-color image.\\n\\nReadiness differs by source, and so does the poll that detects it:\\n\\n- **URL and single-file path** imports return `status: \\"downloading\\"` with one `mediaRef`. Poll `library_manage` `list` with `parameters.ids: [mediaRef]` until `generationStatus` is absent.\\n- **Directory** imports also return `status: \\"downloading\\"` with one placeholder `mediaRef` for the batch. Poll it the same way; the folder\'s assets appear when `generationStatus` clears. (`pending: true` remains a fallback that lists every unresolved import.)\\n- **Inline bytes and matte** imports finish inline and come back `status: \\"ready\\"`; no polling is needed.\\n\\nAn import\'s `mediaRef` is the `assetId` that recipe actions take. Never place a pending asset on the timeline. `generationStatus` is the signal: `preparing`, `generating`, `downloading`, and `rendering` mean keep polling, absent means usable, and **`failed` is terminal**. Report it or retry the import explicitly; never poll on. Do not treat \\"not downloading\\" as ready.\\n\\nA `.srt` or `.vtt` caption file (by `path`, `url`, or `bytes` with `mimeType` `application/x-subrip`, `text/srt`, or `text/vtt`) is not added to the library. Its captions go onto a new caption track at the times the file gives, as one undo step, and the reply names the new `captionGroupId` to restyle with a `text` action.\\n\\nFor a batch of local outputs, download them into one workspace directory and import that directory once when practical. This is safer and faster than racing many import calls.\\n\\nTo apply a LUT, first store it with `library_manage` `prepare_look` (`parameters.path` to a `.cube` file), then use the returned `storedPath` in a `grade` action.\\n\\n## Speech edits\\n\\n`speech_apply` runs speech work as separate operations, not recipe actions: `words` removes transcript words, `silence` trims dead air, and `captions` generates styled captions. Read a fresh transcript with `inspect` `transcript` after any speech edit, because word positions change.\\n\\n## Service operations and timeouts\\n\\n`project_manage`, `library_manage`, `speech_apply`, and `delivery_manage` writes have no retry receipts. After a timeout, inspect the outcome (a snapshot, a library `list`, or delivery `jobs`) before repeating a write. A failed service write may have applied an edit before saving failed, so read its `data` and the current state.\\n\\n## Edit and verify\\n\\nA dependable sequence:\\n\\n1. `project_manage` to select or create the local project; its reply carries the snapshot.\\n2. `inspect` `media` when content selection matters.\\n3. One `edit_apply` recipe with every clip, track, layout, text, audio, color, and effect change the request needs; `speech_apply` for word cuts, silence, and captions.\\n4. Check the reply\'s `clips`; use `inspect` `frame` only when visual composition or layer order matters.\\n5. `edit_undo` if the result is wrong and the next recipe would not cleanly correct it.\\n\\n## Export\\n\\nCall `delivery_manage` with `operation: \\"submit\\"`. Submission is not completion: it returns a job, destination, and `started` or `queued` status.\\n\\n- Use `mode: \\"video\\"` for H.264, H.265, or ProRes.\\n- Use `mode: \\"xml\\"` (XMEML) for Premiere Pro **and DaVinci Resolve**. Resolve reads XMEML natively. Use `fcpxml` only for Final Cut Pro; sending Resolve an FCPXML produces a package it cannot open cleanly.\\n- Use `mode: \\"videodraft\\"` for a self-contained project package.\\n- Use `mode: \\"srt\\"` or `mode: \\"vtt\\"` for a subtitle file of the captions: words and timing only, without styling. `captionGroupId` picks a caption group; `wordTiming: true` adds each word\'s start to a `vtt`.\\n- Omit `outputPath` unless the user named a destination; the default is `~/Downloads`. The editor does not create folders, so a named destination\'s folder must already exist.\\n- Use `operation: \\"jobs\\"` with `parameters.action` `list` to follow progress, warnings, and results. Cancel only when the user asks or the just-queued settings were wrong. Do not infer that an export is stuck from elapsed time alone.\\n- A finished video export is then checked in the background for black picture, gaps, sound that drops out or clips, and missing media. `list` shows each job\'s `findingCount` and its first three findings; `list` with that `jobId` returns every finding. Findings are warnings on a completed export: tell the user what was found and where, rather than treating the export as failed.\\n\\n## Terminal bridge\\n\\nVideoDraft desktop terminals expose `videodraft-editor`, which controls the same process and tools:\\n\\n```bash\\nvideodraft-editor status\\nvideodraft-editor list-tools\\nvideodraft-editor tool project_manage --json \'{\\"operation\\":\\"project\\",\\"parameters\\":{\\"action\\":\\"list\\"}}\'\\nvideodraft-editor tool edit_snapshot --project \\"/path/to/My Video.vdproject\\" --json \'{\\"includeLibrary\\":true}\'\\nvideodraft-editor show\\nvideodraft-editor hide\\n```\\n\\nEach `videodraft-editor` invocation is a separate connection, so a project opened by one call is NOT remembered by the next. Pass `--project <path>` on every `tool` call that operates on a project; without it, a follow-up call answers `no_project` (no project is open for this session) even though the open succeeded. For the same reason, `edit_undo` has nothing to undo from the terminal: it only reverses an edit made on its own connection.\\n\\nControl and tool commands auto-start a headless editor if none is running. `show` only reveals the already-running UI. Use `videodraft-editor tool <name> --json -` to read a JSON object from stdin when shell quoting would be fragile. Never read, copy, or expose the editor\'s rotating local authentication secret.\\n","references/examples.md":"# Recipes\\n\\nWorking patterns for common asks. All assume auth (`videodraft login` once, or `VIDEODRAFT_API_KEY` in the environment) and use `--json` for parsing.\\n\\nInside VideoDraft ADE, if a native editor is available, use these recipes for asset generation and\\noptional script/storyboard stages. Hand the results to the editor for production and export ONLY\\nwhen the deliverable the user asked for is a composed production. A standalone output \u2014 the batch\\nproduct clips in recipe 1, the upscale in recipe 7 \u2014 is finished when it is generated; importing it\\ninto a project and exporting a timeline builds an edit nobody asked for. Recipes below that call `videodraft produce` or `videodraft export` are hosted fallbacks only. Do not choose them over the available native editor unless the user explicitly asks for the hosted web workflow.\\n\\n## 1. Batch product videos from a CSV\\n\\nOne 9:16 product clip per row of `products.csv` (`name,image_url,tagline`):\\n\\n```bash\\n#!/usr/bin/env bash\\nset -euo pipefail\\nmkdir -p outputs\\n\\nwhile IFS=, read -r name image tagline; do\\n job=$(videodraft generate video \\\\\\n \\"Premium product shot of ${name}: ${tagline}. Slow orbit, studio lighting.\\" \\\\\\n --model gemini-omni-1.1-flash --ar 9:16 --duration 6 \\\\\\n --start-image \\"$image\\" \\\\\\n --no-wait --json | jq -r .job_id)\\n echo \\"$name,$job\\" >> outputs/jobs.csv\\ndone < <(tail -n +2 products.csv)\\n\\n# Collect ALL results with ONE process (batched polling \u2014 one request per tick)\\nvideodraft wait $(cut -d, -f2 outputs/jobs.csv) \\\\\\n --download \\"outputs/{job_id}_{index}.{ext}\\" --json > outputs/results.json\\n# map job ids back to product names via outputs/jobs.csv\\n```\\n\\nSubmit-then-collect parallelizes server-side generation; the single multi-id `wait` keeps it to one local process and one batched poll request per tick no matter how many jobs. Gemini Omni 1.1 Flash is selected because these are six-second first-frame product clips. Estimate first: `videodraft costs gemini-omni-1.1-flash --type video --duration 6 --resolution 720p --audio` \xD7 rows, and confirm with the user.\\n\\nExtend an uploaded clip or conversationally edit an official Google interaction with Gemini Omni 1.1 Flash:\\n\\nEach extension appends 3-10 seconds at the end. Edit and extend accept EXACTLY ONE input video, so `--ref-video` cannot accompany `--source-video` or `--previous-interaction-id`; only `--ref` images can. New dialogue is allowed only when the source video is silent, so a chain of spoken beats must carry its speech in the first turn. A continuation is submitted as the prior turn\'s output, so 40 seconds total is reachable only by extending from a source of 30s or less, and a ladder dead-ends past 30s.\\n\\n```bash\\n# Uploaded-video extension. Reference IMAGES are fine here; reference videos are not.\\nvideodraft generate video --model gemini-omni-1.1-flash \\\\\\n --source-video ./ending.mp4 --ref ./wardrobe.png \\\\\\n --extend --duration 6 --resolution 1080p \\\\\\n --download ./media/extended.mp4\\n\\n# Conversational edit from the interaction_id of an earlier generation. Add --extend to lengthen instead.\\nvideodraft generate video --model gemini-omni-1.1-flash \\\\\\n --previous-interaction-id \\"$INTERACTION_ID\\" \\\\\\n --resolution 720p \\\\\\n --download ./media/continued.mp4\\n```\\n\\nFal BYOK supports the currently callable Gemini Omni 1.1 generation and basic source-edit endpoints at zero VideoDraft credits. Its published callable v1.1 endpoints do not expose continuation, extension, or separate creative references on a source edit. Do not infer a Fal route or retry on paid Google while Fal BYOK is active.\\n\\n## 2. Hosted full marketing video from one idea (fallback)\\n\\n```bash\\nvideodraft create \\"30-second launch video for Solace, a sleep-tracking ring. Calm, premium, dark palette.\\" \\\\\\n --ar 9:16 --style cinematic --json > project.json\\nPROJECT=$(jq -r .project_id project.json)\\n\\nvideodraft shots \\"$PROJECT\\" --grid --estimate # show the user the cost; get a go-ahead\\nvideodraft shots \\"$PROJECT\\" --grid\\nvideodraft produce \\"$PROJECT\\"\\nvideodraft generate music \\"minimal ambient, warm pads, 60 BPM\\" --attach \\"$PROJECT\\"\\nvideodraft generate audio \\"Extend @Audio1 into a 20-second transition\\" --ref-audio ./intro.wav --format wav --download ./transition.wav\\nvideodraft export \\"$PROJECT\\" --download solace-launch.mp4\\n```\\n\\nUse this complete hosted path only when the user requested a web project or the native editor is unavailable. Otherwise stop after the storyboard/assets, import them into the native `.vdproject`, and export with `delivery_manage`. The hosted project stays editable at the URL in `project.json` (`.urls`).\\n\\n## 3. Talking-head (avatar) video\\n\\nWhen the user has no portrait, generate a clear front-facing avatar image first. Skip this step when they supplied one or an existing character should be reused.\\n\\n```bash\\nvideodraft generate image \\\\\\n \\"Front-facing head-and-shoulders portrait of a friendly coffee expert, direct eye contact, natural expression, clean studio background\\" \\\\\\n --model nano-banana-2 --ar 9:16 --download ./media/avatar.png\\n\\nSCRIPT=$(videodraft avatar script \\"why our espresso subscription saves you money\\" --style ad-style --json | jq -r .script)\\nAVATAR=$(videodraft avatar create ./media/avatar.png --script \\"$SCRIPT\\" --voice elevenlabs-kPzsL2i3teMYv0FxEYQ6 --ar 9:16 --json | jq -r .avatar_video_id)\\nvideodraft avatar render \\"$AVATAR\\" --resolution 720p # VEED Fabric paid step; confirm cost first (~20 credits/sec)\\n```\\n\\n`avatar script` and `avatar create` (including speech) are bundled/free. In this example only the optional portrait generation and Fabric render spend credits.\\n\\nIf the portrait is low resolution, enhance it before `avatar create`:\\n\\n```bash\\nvideodraft upscale image ./founder-small.jpg --scale 2x --download ./media/founder-upscaled.png\\n```\\n\\nFor a one-off portrait animation without creating a managed avatar record:\\n\\n```bash\\nvideodraft avatar fabric ./founder.jpg \\\\\\n --text \\"Welcome to the weekly product update.\\" \\\\\\n --voice-description \\"warm, confident American presenter\\" \\\\\\n --resolution 720p --download ./media/presenter.mp4\\n```\\n\\nFor a short, sharp clip from speech or a song the user already has (5-14.8s of audio is used):\\n\\n```bash\\nvideodraft avatar h3-lipsync ./singer.png --audio ./chorus.mp3 \\\\\\n --resolution 1080P --audio-duration 12 --estimate\\nvideodraft avatar h3-lipsync ./singer.png --audio ./chorus.mp3 \\\\\\n --resolution 1080P --download ./media/singer-lipsync.mp4\\n```\\n\\nWhen the user already has both the video and replacement speech:\\n\\n```bash\\nvideodraft avatar lipsync ./presenter.mp4 \\\\\\n --audio ./localized-voiceover.mp3 \\\\\\n --sync-mode loop --download ./media/presenter-localized.mp4\\n```\\n\\nEdit an existing video with a dedicated edit model:\\n\\n```bash\\nvideodraft models video --category video_edit\\nvideodraft edit video ./product-demo.mp4 \\\\\\n \\"Preserve the product but follow the reference camera rhythm\\" \\\\\\n --model gemini-omni-1.1-flash --ref-video ./camera-rhythm.mp4 \\\\\\n --ref-video-duration 2.5 --resolution 1080p \\\\\\n --download ./media/product-demo-reframed.mp4\\n\\nvideodraft edit video ./product-demo.mp4 \\\\\\n \\"Turn the room into a warm evening scene while preserving the product and camera motion\\" \\\\\\n --model kling-o3-video-ref-edit --ref ./evening-style.jpg \\\\\\n --preserve-audio --download ./media/product-demo-evening.mp4\\n```\\n\\nTransfer motion from a reference clip onto a character image:\\n\\n```bash\\nvideodraft edit motion ./character.png \\\\\\n \\"Apply the dancer\'s movement to this character while preserving identity\\" \\\\\\n --motion-video ./dance-reference.mp4 \\\\\\n --model kling-v3-motion-control --quality pro \\\\\\n --download ./media/character-dance.mp4\\n```\\n\\n## 4. Hosted changelog video in CI\\n\\nIn a GitHub Action with `VIDEODRAFT_API_KEY` set as a secret:\\n\\n```bash\\nNOTES=$(git log --oneline v1.2.0..HEAD | head -20)\\nvideodraft create \\"Weekly product update video. Energetic, 20 seconds. Changes: ${NOTES}\\" --ar 16:9 --json > p.json\\nPROJECT=$(jq -r .project_id p.json)\\nvideodraft shots \\"$PROJECT\\" && videodraft produce \\"$PROJECT\\"\\nvideodraft export \\"$PROJECT\\" --download changelog.mp4 --wait-timeout 30m\\n```\\n\\n## 5. Variations and picking a winner\\n\\n```bash\\nvideodraft generate image \\"logo concept: minimalist fox, geometric\\" --num 4 --download \\"./concepts/{job_id}_{index}.{ext}\\" --json\\n# Show all 4 to the user; regenerate the chosen one at higher res:\\nvideodraft generate image \\"<same prompt>\\" --model nano-banana-pro --resolution 4K\\n```\\n\\n## 6. Reaching tools without a curated command\\n\\n```bash\\nvideodraft tools list --json | jq -r \'.[].name\'\\nvideodraft tools schema attach_media_to_shot --json\\nvideodraft call attach_media_to_shot --args \'{\\"project_id\\":\\"...\\",\\"scene_index\\":0,\\"shot_index\\":1,\\"media_url\\":\\"https://...\\",\\"media_type\\":\\"video\\",\\"duration_seconds\\":6}\'\\n```\\n\\nAnything the hosted VideoDraft MCP exposes, including character studio, product studio, and hosted project data, is reachable this way even before it gets a curated command. Native `.vdproject` editing uses the separate `videodraft_editor` MCP described in SKILL.md.\\n\\n## 7. Enhance an existing asset without changing it\\n\\n```bash\\n# Light image cleanup, no enlargement\\nvideodraft upscale image ./poster.png --scale 1x --download ./media/poster-enhanced.png\\n\\n# General image and video enlargement\\nvideodraft upscale image ./frame.png --scale 2x --download ./media/frame-2x.png\\nvideodraft upscale video ./clip.mp4 --resolution 1080p --download ./media/clip-1080p.mp4\\n\\n# Rebuild detail in a tiny/blurry source (generative), or reimagine it (creative)\\nvideodraft upscale image ./tiny.jpg --scale 4x --mode generative --model \\"Wonder 3.5\\" --download ./media/tiny-4x.png\\nvideodraft upscale video ./ai-clip.mp4 --resolution 4k --mode generative --download ./media/ai-clip-4k.mp4\\n\\n# Smoother motion or slow motion (resolution unchanged)\\nvideodraft interpolate ./clip.mp4 --fps 60 --download ./media/clip-60fps.mp4\\nvideodraft interpolate ./clip.mp4 --model Chronos --slowdown 4 --download ./media/clip-slowmo.mp4\\n```\\n\\nUse these when the content is correct and only quality or resolution needs improvement. If the poster text, composition, subject, or motion is wrong, edit or regenerate instead.\\n","references/models.md":"# Choosing models (and predicting cost)\\n\\nAlways consult the live catalog instead of memorizing this page \u2014 models change weekly:\\n\\n```bash\\nvideodraft models image --json # every image model + inputs (aspect ratios, resolutions, max refs)\\nvideodraft models video --json # every video model + inputs + per-second pricing metadata\\nvideodraft models audio --json # standalone audio/media models + pricing inputs\\nvideodraft models voices --json # TTS voices\\nvideodraft models styles --json # visual style presets\\n```\\n\\n## Task-based model selection\\n\\nFollow this order every time:\\n\\n1. The user named a model: use it, never substitute.\\n2. Otherwise pick a **Tier 1** model from the task\'s inputs, duration, audio, quality, speed, and cost, and pass it explicitly instead of relying on a blind platform fallback.\\n3. Use a **Tier 2** model only when no Tier 1 model supports the request, and say why. Never pick Tier 2 on your own for price, speed, novelty or variety.\\n\\nTier 1 is the rows marked **T1** below. Every other catalog model is Tier 2 and stays available by name. The catalogs carry this themselves: every `videodraft models image|video|audio --json` entry has `tier` (1 or 2) plus `recommended` / `recommended_for`, and the top-level `recommended` array is Tier 1, best first. Trust the catalog over this page when they disagree.\\n\\n### Images\\n\\n| Need | Choose | Why |\\n| ------------------------------------------------------------------------------------------ | ------------------------ | ----------------------------------------------------------------- |\\n| T1: Most generation, editing, character consistency, or reference work | `nano-banana-2` | Best general default; 1K/2K/4K and up to 14 reference images |\\n| T1: Highest-quality complex generation or reasoning | `nano-banana-pro` | Premium Nano Banana quality and reasoning |\\n| T1: Fast, inexpensive drafts and iteration | `nano-banana-2-lite` | Fastest/cheapest Nano Banana option; 1K only, up to 14 references |\\n| T1: Posters, title cards, signs, logos, or any image with important readable text | `gpt-image-2.5-flare` | Fast OpenAI option; 16 references, 1K/2K/4K and PNG output |\\n| T1: Complex multi-image composition, precise editing, or a strong alternate interpretation | `gpt-image-2.5-sunburst` | More fidelity/detail; 16 references, quality through max |\\n| T1: Vector / SVG output (logos, icons, illustrations that must scale) | `recraft-v4` | Recraft V4.1 text-to-vector; SVG only, no reference images |\\n| T2: By name, or an explicitly requested xAI look | `grok-imagine` | 2 cr flat (3 with a reference); 1 reference image |\\n| T2: xAI at 2K or with a quality tier, up to 3 edit references | `grok-imagine-2.0` | 1K 4 (low) / 6 (medium), 2K 6 / 8, +1 cr per reference image |\\n\\nGPT Image 2.5 has two IDs: `gpt-image-2.5-flare` replaces the former GPT Image 2 recommendation; `gpt-image-2.5-sunburst` is the precision/detail alternative. Other model recommendations are unchanged. Explicit `gpt-image-2` selections continue to use the previous model.\\n\\nUse the same basic image controls as GPT Image 2: `--ref`, `--ar`, `--resolution 1K|2K|4K`, `--quality`, and `--num 1..4`. GPT Image 2.5 quality supports auto/low/medium/high/xhigh/max. Auto is priced at Max in customer-credit estimates. The default platform path uses OpenAI; a selected personal connection stays exclusive. A Pika personal connection requires an explicit quality from low/medium/high/xhigh/max because its upstream API has no auto value. Other unsupported connection settings return an error instead of switching payer or changing the request.\\n\\n```bash\\nvideodraft generate image \'A blue glass bird\' --model gpt-image-2.5-flare --ar 1:1 --resolution 2K --quality high --estimate\\nvideodraft generate image \'A blue glass bird\' --model gpt-image-2.5-flare --ar 1:1 --resolution 2K --quality high --num 2\\n```\\n\\n`videodraft shots` forwards resolution and quality for both normal and grid generation. Estimates use the project\'s model and aspect when omitted; grid canvas costs may differ from ordinary shot costs. Use `videodraft costs --model <id> --ar <ratio> --resolution <tier> --quality <tier>` for a specific per-image quote.\\n\\n### Videos\\n\\n| Need | Choose | Important limits |\\n| -------------------------------------------------------------------------------------------------------------- | ------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |\\n| T1: Most generation, first/last-frame, mixed-reference, source-edit, or extension requests | `gemini-omni-1.1-flash` | 3-10s output; 360p/720p/1080p/4K at 3/10/15/30 cr/s; audio always; up to 10 image inputs, and 3 reference videos <=3s on generate only; edit source <=10s; continuation or 3-10s extension of a 1-30s source, 40s total reachable only by extending from <=30s |\\n| T2: Grok 1.5 text, first-frame, or 1-7 image-reference clips with native audio and optional 1080p | `grok-imagine-video-1.5` | 1-15s; 480p/720p/1080p for text/first-frame; references are 480p/720p only; no last frame |\\n| T2: Document or webpage reference generation; by name otherwise | `wan-3.0` | 2-30s or auto; 480p/720p/1080p at 7/14/28 cr/s; 10 image, 5 video, 5 audio refs, 20 media files total; document/web refs require thinking |\\n| T2: 480p/768p/2K/4K with native stereo audio, first/last frames, or mixed image/video/audio references | `minimax-h3` | 5-15s; 5/6/13/16 cr/s by resolution; up to 9 image, 3 video, 3 audio refs, 12 files total; first 5 reference images free then 8 cr each; reference video/audio each total <=15s |\\n| T1: Text, first/last-frame, or mixed-reference video with native audio and stronger prompt adherence | `minimax-h3-max` | 5-15s; 5/8 cr/s at 480p/768p plus pooled reference tokens; up to 9 image, 3 video, 3 audio refs, 12 files total; seed and disabled/balanced/quality prompt expansion; provider safety checker off by default |\\n| T2: Images pinned to specific moments (keyframes); by name otherwise | `flux-3` | 5-20s (auto for text/first-frame only); 720p/1080p; up to 10 keyframes; `--quality draft` is 720p-only at ~1/3 the cost |\\n| T1: Video/audio references, mixed reference media, broad aspect ratios, frame-mode first+last frame, or 11-15s | `seedance-2` | 4-15s or auto; up to 9 image, 3 video, and 3 audio refs; audio toggle; Mini/Fast are 480p/720p only |\\n| T1: Single takes past 15s, or more references than Seedance 2.0 allows | `seedance-2.5` | 4-30s or auto; up to 30 image, 10 video, and 10 audio refs (50 files total); one quality tier; 480p/720p/1080p |\\n| T2: Fast polished 3-15s video with first frame, multi-prompt, and native audio | `kling-v3-turbo` | Audio always on; Pro default; no end frame, reference-media mode, or elements |\\n| T1: Cinematic 3-15s with reference images, image/video elements, first+last frame, audio, or 4K | `kling-o3` | 7 combined image refs/elements, reduced to 4 with a video element; Standard/Pro/4K |\\n| T1: Kling 3-15s image-to-video with first+last frame, image/video elements, multi-prompt, audio, or 4K | `kling-3.0` | Elements require a start image; image or video elements can bind voice_id; Standard/Pro/4K |\\n| T2: User explicitly requests Veo, or the selected workflow specifically needs Veo | `google-veo3.1` | Good fallback, but not the preferred general model |\\n\\nRouting rules:\\n\\n- Inexpensive pass or draft at 480p/768p with native audio: use MiniMax H3 Max (5/8 cr/s plus reference tokens).\\n- Grok 1.5 reference mode accepts 1-7 images only. Address them in array order as `<IMAGE_0>` through `<IMAGE_6>`. Do not combine reference images with `--start-image`, `--end-image`, `--ref-video`, or `--ref-audio`.\\n- Grok 1.5 first-frame mode accepts one `--start-image`, derives the output aspect ratio from that image, and does not support `--end-image`. Text and first-frame modes support 480p, 720p, or 1080p. Reference mode supports 480p or 720p.\\n- Grok 1.5 always generates native audio. Do not pass `--no-audio`, `--seed`, `--negative`, or `--quality`.\\n- Around 11-15 seconds with native audio: use Seedance, Kling 3.0 / O3, or MiniMax H3 Max, not Gemini. Past 15 seconds: Seedance 2.5.\\n- Dialogue: write the line in the prompt, in quotes, with who says it and how. `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` and `kling-o3` speak it natively; add `--ref-audio` on Seedance or a bound Kling voice for a specific voice. Do not generate speech separately and lip-sync it on.\\n- One existing source video that should be edited: use `videodraft edit video`, which auto-selects Gemini Omni 1.1 Flash for a source up to 10s. Generic `generate video --ref-video` retains source-edit behavior when `--video-task` is omitted, but the dedicated edit command is clearer.\\n- Video supplied as a creative reference: Gemini Omni 1.1 Flash accepts up to 3 videos of at most 3 seconds each, including mixed image and video input. Creative video references are accepted on `--video-task generate` ONLY: edit and extend take exactly one input video, so `--ref-video` cannot accompany `--source-video` or `--previous-interaction-id`. Use Seedance 2.5 for more or longer video/audio references, or Seedance 2.0 when quality-tier control matters.\\n- First and last frame control: every Tier 1 video model supports it (Gemini Omni 1.1 Flash, Seedance, Kling O3, Kling 3.0, MiniMax H3 Max). For Gemini, every start/end frame counts toward the 10-image total, leaving up to 9 references with a start frame or 8 with both frames.\\n- Extension with a 1-30s uploaded source uses Gemini Omni 1.1 Flash plus `--source-video` and `--video-task extend` (or `--extend`). Pass an explicit 3-10s output duration; the model appends it at the end, up to 40 seconds total. Creative `--ref-video` clips CANNOT accompany the source: edit and extend take exactly one input video. New dialogue is allowed only when the source video is silent; adding speech on top of a source that already has speech is refused with \\"the model is currently unable to process speech edits\\". Continuation uses `--previous-interaction-id <id>` and defaults to a conversational edit; `--video-task extend` lengthens it. A continuation is submitted as the prior turn\'s output video, so it obeys the same source limits (<=30s to extend, <=10s to edit) and the same silent-source dialogue rule. 40s total is reachable only by extending from a source of 30s or less, so a ladder dead-ends past 30s. Fal returns interaction IDs, but its currently published callable v1.1 endpoints do not expose continuation/extension or mixed source-edit references; never infer a route or silently fall back to paid Google.\\n- Wan 3.0 reference mode and frame mode are separate. It accepts 10 images, 5 videos, and 5 audio clips, at most 20 media files total. Video and audio each total at most 15 seconds. `--file-url` and `--web-url` require `--thinking`. `--auto-duration` cannot be combined with `--duration`.\\n- MiniMax H3 and H3 Max reference mode and first-plus-last-frame mode are separate. Audio cannot be the only reference. Address references as `Image 1`, `Video 1`, and `Audio 1` in array order.\\n- Seedance reference mode and first-plus-last-frame mode are separate. Do not promise reference video/audio plus a last frame in one generation.\\n- Multi-prompt sequencing: use Kling 3.0 Turbo, Kling O3, or Kling 3.0.\\n- Kling 3.0 image-to-video and Kling O3 reference-to-video accept repeatable structured `--element` JSON. Each element is image-backed (`frontal_image_url` plus 1-3 `reference_image_urls`) or video-backed (`video_url`). Either form may include `voice_id`. Keep the voice ID inside the same element so the association is preserved. Reference them as `@Element1`, `@Element2`. Kling V3 Turbo rejects elements.\\n- Kling 2.6 Pro image-to-video accepts one or two repeatable `--voice-id` values. Cite them as `<<<voice_1>>>` and `<<<voice_2>>>` in the prompt. Voice control costs 17 cr/s. Use `videodraft kling-voices list|create|delete` to manage the separate Kling video-control voice library. Creating a voice requires a clean 5-30 second, up-to-50MB single-speaker `.mp3`, `.wav`, `.mp4`, or `.mov` sample plus `--confirm-consent`. Creation costs 1 VideoDraft credit on the platform Fal account and 0 credits with Fal BYOK; use the create command\'s `--estimate` flag to check without creating.\\n- Kling V3 voice control costs 16 cr/s for Standard and 20 cr/s for Pro.\\n- Seedance quality: `mini` for the lowest cost, `fast` for speed, `standard` for maximum quality and for 1080p/4K. Seedance 2.5 has a single tier and ignores `--quality`.\\n- Longer than 15 seconds, or more than 9 image / 3 video / 3 audio references: use Seedance 2.5. It reaches 30s and 30/10/10 references (50 files total) at 480p/720p/1080p, but has no 4K.\\n- Seedance 2.x allows real people by default, the same as AI Studio: Byteplus first, a submit-time Fal fallback, and Fal\'s higher tier-specific rate. Turn it off with MCP `allow_real_people: false` or CLI `--no-allow-real-people` for the lower Byteplus-only rate, only when the user wants the cheapest run and nothing in the job is a real identifiable person (text-only requests, non-people, anime, clearly synthetic or stylized characters).\\n- If a Seedance request made with the option off fails with `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, estimate the higher rate, follow the user\'s spend policy, and retry once with the option on. The response also carries `retry_with: { allow_real_people: true }`, `cli_flag: \\"--allow-real-people\\"`, and `retry_policy: \\"once\\"`. If the option was already on, do not repeat the same request. Byteplus may accept a task and reject the output later. VideoDraft refunds that failed generation, but it cannot reroute the asynchronous failure to Fal. Rephrase or change the references instead.\\n\\n### Video edit and motion-control categories\\n\\nUse `videodraft models video --category video_edit` for existing-video transforms and `--category motion_control` for motion transfer.\\n\\n| Need | Command/model | Important limits |\\n| ------------------------------------------ | ---------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------- |\\n| Most edits of a source up to 10s (DEFAULT) | `videodraft edit video <video> \\"...\\"` (auto-selects `gemini-omni-1.1-flash`) | Source <=10s or it REFUSES; up to 10 image refs and 3 creative video refs <=3s each; 360p/720p/1080p/4K; audio always regenerated (no `--preserve-audio`) |\\n| Cheapest simple prompt edit | `videodraft edit video <video> \\"...\\" --model grok-imagine-video-edit` | No image refs; source silently truncated to 8s; auto/480p/720p |\\n| Edit with several image references | `--model happy-horse-video-edit --ref ...` | Up to 5 refs; 720p/1080p; source silently truncated to 15s; highest rate (28-56 cr/s) |\\n| Controlled Kling edit | `--model kling-o3-video-ref-edit --ref ...` | Up to 4 refs; Standard/Pro; source silently clamped to 3-10s |\\n| Transfer reference motion to an image | `videodraft edit motion <image> [direction] --motion-video <video>` | Prompt/direction is optional; Kling V3 default; optional one image-only `--element`; element requires video orientation |\\n\\nIf the user explicitly names one of these models, preserve it. The CLI uploads local source videos and reference images automatically. Editing returns an async job and waits by default.\\n\\n**Omit `--model` and the SERVER chooses**, because only it can measure the source. It picks `gemini-omni-1.1-flash` for any source up to 10s. When that is not safe (source longer than 10s, unmeasurable duration, `--preserve-audio`, more than 10 refs, or Fal BYOK with refs) it spends nothing and returns a priced menu: per model, the seconds it would edit, the seconds it would drop, and the credit cost. The CLI prints that table and exits 2. Show it to the user, then re-run with `--model`.\\n\\n**Truncation is the trap here.** Only Gemini refuses a source it cannot fully consume. Every other edit model accepts a 30s clip and returns an edit of its first 8-15 seconds with no error. `videodraft edit video` now warns when this will happen and the tool response carries `source.truncated` / `source.dropped_seconds`; relay it to the user rather than letting them discover it in the output.\\n\\nTo edit a longer source without losing its tail, cut it into <=10s pieces with `videodraft_editor`, edit each with Gemini, then reassemble and export there. VideoDraft has no server-side split/concat, so this needs the native editor.\\n\\nKling O3 is also exposed for reference generation. `videodraft generate video --model kling-o3-video-ref-edit` requires exactly one `--ref-video` and generates a new reference-guided clip. Wan 3.0 handles new text, frame, and mixed-reference generation, but is not an existing-source edit model.\\n\\n### Reference-first video workflow\\n\\n- Prefer a start frame or reference image whenever a specific character, product, location, style, composition, or brand identity must stay recognizable.\\n- If the user gives a reference, pass it. Never silently replace it with a text description.\\n- If no reference exists and continuity matters, generate a still first with the user\'s explicitly requested compatible image model, otherwise use Nano Banana 2. Wait for the image URL, then animate it with the selected video model. Confirm the combined image plus video cost before starting.\\n- For multi-shot scenes, generate shot images with `videodraft shots <project_id> --model <selected-image-model> --grid`. Preserve an explicitly requested compatible image model; otherwise use `nano-banana-2`. The grid establishes the scene and characters together, then decodes into individual shot images.\\n- Animate the decoded shot images as per-shot start frames or references. Do not independently text-generate each video clip when the shots need to match.\\n- Pure text-to-video remains appropriate for generic one-off footage where no subject, composition, or continuity needs to be preserved.\\n\\n### Audio\\n\\n- **Seed Audio 1.0**: use `videodraft generate audio` for open-ended speech, sound, music, or prompt-driven audio editing. It accepts up to three audio references or one image. Address audio references as `@Audio1`, `@Audio2`, and `@Audio3`. Preset and custom cloned voice IDs are supported. Output is up to 120 seconds. There is no requested-duration input. The CLI automatically retries transient responses with one idempotency key. To recover after the CLI process itself is interrupted, set `--idempotency-key <uuid>` on the original command and reuse it.\\n- **Voiceover/TTS**: prefer ElevenLabs. Brittney is the platform default voice; under ElevenLabs BYOK, use a compatible voice from the user\'s account. Honor another supported voice/provider when the user explicitly selects it. ElevenLabs voices run on Eleven v4 in one of two modes: Standard (the default, best quality) or Turbo (Eleven v4 Turbo, faster and half the price). Keep Standard unless the user wants faster or cheaper speech. Pick the mode with `generate voiceover --mode standard|turbo` or `produce --voice-mode standard|turbo`; over MCP, pass `mode` to `generate_voiceover` or `voice_mode` to `produce_project`. Google, OpenAI and cloned `custom-*` voices ignore it. v4 takes no style or speed settings and no SSML `<break>` tags; audio tags such as `[whispers]` work. Dialogue stays on Eleven v3.\\n- **Dialogue, voice changing, and dubbing**: ElevenLabs only.\\n- **Sound effects**: ElevenLabs Sound Effects only.\\n- **Music**: use `lyria-3.5` for all music, short or long (up to ~3 minutes), with vocals/lyrics or instrumental music. When `--model` is omitted the CLI sends no model and the server default applies, which is `lyria-3.5` wherever the backend supports it. If the server answers that `lyria-3.5` is an unknown model, that backend predates it: rerun without `--model` (the server default then applies), or use `lyria-3-pro-preview` for a track longer than 30 seconds. `lyria-3-clip-preview` (fixed ~30s) and `lyria-3-pro-preview` remain available for explicit requests. Lyria length and structure are prompt-guided, not exact; for an instrumental, include \\"instrumental only, no vocals\\" in the prompt. `--length` and `--instrumental` only apply to ElevenLabs. Use `elevenlabs-music-v2.5` when a specified 3-300 second length, composition plan, or style reference track matters. Lyria references: up to 10 images on Google, 1 on Fal BYOK; 3.5 Fal prompts are 1-5000 characters (an image-only request gets a neutral prompt). `elevenlabs-music` is an alias for v2.5. Use `elevenlabs-music-v1` only when the user asks for v1 by name; it takes a prompt, `--length` and `--instrumental` only.\\n- **ElevenLabs Music v2.5 inputs**: either a prompt (`--length`, `--instrumental`) or a composition plan, never both. A plan is 1-30 sections of 3-120 seconds each, up to 300 seconds in total, and `--plan`/`--section` pick v2.5 when `--model` is omitted. Empty section parts default to 20 seconds and an instrumental part. Build one with repeatable `--section \\"<seconds>|<style, style>|<text>\\"` (use `\\\\n` for line breaks) or pass `--plan plan.json` with `{\\"chunks\\":[{\\"text\\",\\"duration_ms\\",\\"positive_styles\\",\\"negative_styles\\",\\"context_adherence\\",\\"audio_reference\\"}]}`. Section text is an optional `[Section name]`, lyric lines, and `{inline directions}`. Put 6-7 specific English styles on the first section; it sets the genre. `--ref-audio <url|file>` adds a style reference clip to the first section (window up to 30 seconds via `--ref-start`/`--ref-end` in ms, `--ref-strength low|medium|high|xhigh`); local files are uploaded, including relative `audio_reference.audio_url` paths in a plan file. `--seed` works only with a plan. `--format` picks the output (`mp3_48000_192` default on v2.5; `pcm_*`, `ulaw_8000` and `alaw_8000` arrive as stereo WAV files). Style references don\'t run on the user\'s own ElevenLabs key; with that key connected, drop the reference or ask the user to turn the key off. The CLI retries transient responses with one idempotency key; set `--idempotency-key <uuid>` to recover after an interruption.\\n- Voice Changer and Dubbing require the source media duration and currently accept source media up to 300 seconds.\\n\\n**User-supplied ElevenLabs voice IDs:** voiceover, dialogue, and voice changing accept raw IDs (16-64 alphanumeric characters) or `elevenlabs-<id>`. `videodraft models voices` / MCP `list_available_voices` is for discovery, not an allowlist. Pass a supplied ID directly even when it is absent from the catalog; do not reject it or substitute a catalog voice. The voice must be accessible to the provider account used for generation. Private or cloned voices may require the user\'s connected ElevenLabs key. A failed voice listing does not prove that a supplied ID cannot generate; a connected key with generation access may still work. If the provider rejects generation access, report that error rather than silently changing the voice.\\n\\nUse `--voice <id>` for voiceover and voice changing, and repeat `--line \\"<id>:Text\\"` for dialogue. These examples show both ID forms; replace the sample IDs with the user\'s supplied IDs:\\n\\n```bash\\nvideodraft generate voiceover \\"Hello there.\\" --voice kPzsL2i3teMYv0FxEYQ6 --download ./media/voiceover.mp3\\nvideodraft generate dialogue --line \\"kPzsL2i3teMYv0FxEYQ6:Hello.\\" --line \\"elevenlabs-kmSVBPu7loj4ayNinwWM:Welcome back.\\" --download ./media/dialogue.mp3\\nvideodraft generate voice-changer ./media/source.wav --voice elevenlabs-kPzsL2i3teMYv0FxEYQ6 --duration 12 --download ./media/changed-voice.mp3\\n```\\n\\nFor MCP, use `generate_voiceover.voice_id`, `generate_dialogue.lines[].voice_id`, or `change_voice.voice_id` with either form. ElevenLabs IDs are separate from Kling video-control voice IDs and MiniMax `custom-*` cloned-voice IDs. Do not convert IDs from those systems into ElevenLabs IDs.\\n\\n### Avatar / talking head\\n\\n**First decide the framing, not just \\"someone talks\\".** This lane animates a PORTRAIT facing the lens. A character delivering a line inside a real scene, with blocking, framing or camera movement, belongs to `generate video` instead: write the dialogue in the prompt and `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` or `kling-o3` voices it natively (add a Seedance `--ref-audio` clip or a Kling voice bound per element for a specific voice). That is the default for any talking character; never plan silent video + TTS + lip-sync. Sending a cinematic shot to Fabric returns a head-on talking headshot, not the shot that was asked for. `videodraft models video --json` lists these under `recommended.in_scene_dialogue`.\\n\\nFor a presenter, spokesperson or explainer speaking to camera, VEED Fabric is the preferred model. Choose the dedicated path from the media the user already has:\\n\\n| Starting media | Command | Use |\\n| ------------------------------------------ | ------------------------------------------------------------------------- | ---------------------------------------------------------- |\\n| Portrait + script, reusable avatar record | `videodraft avatar create <portrait> --script \\"...\\"` then `avatar render` | Managed avatar flow with bundled speech preparation |\\n| Portrait + text | `videodraft avatar fabric <portrait> --text \\"...\\"` | One-off direct VEED Fabric text mode |\\n| Portrait + existing audio | `videodraft avatar fabric <portrait> --audio <audio>` | One-off direct VEED Fabric audio lip sync |\\n| Portrait + existing audio, short and sharp | `videodraft avatar h3-lipsync <portrait> --audio <audio>` | MiniMax H3 Max Lip Sync, 5-14.8s at up to 2K |\\n| Existing video + existing audio | `videodraft avatar lipsync <video> --audio <audio>` | Sync Labs Lipsync 2 (Tier 2): re-dub existing footage only |\\n\\nThe managed renderer is VEED Fabric Fast (`veed/fabric-1.0/fast`). Direct Fabric, MiniMax H3 Max Lip Sync, and Sync Labs are paid AI Studio generations and return async job IDs.\\n\\n1. Obtain the avatar image. Prefer the user\'s supplied portrait or an existing character. If none exists, use the user\'s explicitly requested compatible image model, otherwise generate a front-facing head-and-shoulders portrait with `nano-banana-2`, direct eye contact, a natural expression, and a clean background. Match the intended video aspect ratio when practical.\\n2. If the portrait is visibly soft or too small, run Topaz image enhancement/upscaling before animation.\\n3. Generate a script only if needed: `videodraft avatar script \\"<idea>\\"`.\\n4. Create the avatar record and speech: `videodraft avatar create <portrait-url-or-file> --script \\"...\\" --voice <id> --ar 9:16`. Prefer ElevenLabs when unspecified, but honor another explicitly selected supported voice/provider.\\n5. Render with VEED Fabric: `videodraft avatar render <avatar_video_id> --resolution 720p`.\\n\\nThe portrait is passed as the avatar\'s character image, not as a generic video\'s start frame. Prefer rendering directly at 720p. Use 480p only when the user prioritizes lower cost. Avatar script generation and `avatar create` (including speech) are bundled/free. Confirm the Fabric render cost, plus portrait generation or upscaling when needed.\\n\\nDirect Fabric text/audio, H3 Max Lip Sync, and Sync Labs do not use the managed avatar record. The CLI uploads local portrait, video, and audio files automatically. `avatar fabric --speed fast` applies only to audio mode. `avatar h3-lipsync` takes no prompt: pick `--resolution 480P|768P|1080P|2K` (default 768P), an optional `--seed`, `--no-transcription` to sync without transcribing the audio (transcription is on by default), and `--safety-checker` only when the user asks for Fal\'s checker. The portrait\'s width / height must be 0.4-2.5, and audio shorter than 5 seconds is refused before any charge. Sync costs 5 credits per verified audio second; under Fal BYOK, `sync_mode` remains available but `temperature` and `active_speaker` are ignored by the provider.\\n\\n### Upscaling / enhancement\\n\\n- **Images**: Topaz via `videodraft upscale image <url-or-file> --scale 1x|2x|4x [--mode precision|generative|creative] [--model <name>]`. Modes: `generative` (default; Wonder 3.5, Topaz\'s recommended model for AI-generated images; `Redefine` takes `--prompt`, `--creativity 1-6`, `--texture 1-5`), `precision` (faithful and cheapest; Standard V2, High Fidelity V3, Low Resolution V2, CGI, Text Refine; use for clean real photos), `creative` (Bloom 2; artistic, `--prompt`, `--creativity 1-9`). Extra knobs: `--no-face-enhance`, `--face-strength`, `--sharpen`, `--denoise`, `--fix-compression`, `--format png`. Use 1x for light enhancement without enlargement, 2x as the general default, and 4x only when the source quality and target size justify it. The server must verify dimensions from a readable image of at most 50 MB; `--width` and `--height` are compatibility hints and cannot bypass a failed probe. The result may be immediate or queued; the CLI waits by default. Use `--no-wait` for an async job ID, and poll it with `status` or `wait`. Cost: 8 credits per started 24 MP of output in precision, per 8 MP (Wonder 3/3.5) or 4 MP in generative, per 2 MP in creative.\\n- **Videos**: Topaz via `videodraft upscale video <url-or-file> --resolution 720p|1080p|4k` (preferred) or `--scale 2x`, plus `--mode precision|generative|creative` and `--model <name>`. `generative` (default; Starlight Precise 2.6, Topaz\'s recommended model for AI-generated footage; Starlight Fast 2 at half price) costs 12 credits/s up to 1080p and 26 at 4K. `precision` (Proteus, Proteus Natural, Iris, Gaia 2 for animation, Rhea, Theia, Artemis, Dione) is 6x cheaper at 1 / 2 / 6 credits per second for 720p / 1080p / 4K output; use it for real footage or when cost matters. `creative` (Astra 2, `--prompt`, `--creativity`, `--realism`, `--sharp`) always renders 4K at 50 credits/s. `--fps 60` delivers 60fps on the same pass; any output above 30fps (including a 50/60fps source) doubles every rate; the server must verify duration, dimensions, and frame rate from an MP4/MOV source of at most 100 MB. `--duration`, `--width`, `--height`, and `--source-fps` are compatibility hints and cannot override billing or bypass a failed probe. The same source-verification requirement applies to frame interpolation. `--scale` also accepts intermediate factors such as `1.5x`. Max source length 5 minutes. The job is asynchronous; the CLI waits by default, while MCP callers poll `check_generation_status`. MCP video input must be VideoDraft-hosted, so upload local or external sources first.\\n- **Frame interpolation / slow motion**: `videodraft interpolate <url-or-file> --fps 60 [--model Apollo|Chronos|Aion] [--slowdown 1-8]` (MCP `interpolate_video`). Resolution is unchanged. Apollo (default) for smooth general conversion, Chronos for natural slow motion, Aion for extreme slow motion. Apollo/Chronos cost 3 credits per output second up to 1080p (6 at 4K); Aion 5 / 17. These rates cover targets up to 60fps; above 60fps multiply by target FPS / 60 (120fps doubles the rate). Output seconds = source seconds \xD7 slowdown. The final charge rounds up to a whole credit.\\n- Use upscaling to preserve the image/video while improving detail, resolution, or cleanup. It cannot fix the wrong subject, misspelled text, bad framing, unwanted objects, broken continuity, or incorrect motion. Use an edit or regeneration for those problems. Keep `generative` for AI-generated sources; switch to `precision` for clean real photos/footage or a cheap pass, and `creative` only when the user wants an artistic reinterpretation.\\n- For a new Fabric avatar, render directly at 720p instead of rendering at 480p and then upscaling. Upscale the source portrait first only when the portrait itself is low quality.\\n\\n## Capability gotchas\\n\\n- Each model\'s `inputs` block is authoritative: supported `aspect_ratios`, `resolutions`, `quality_options`, `start_frame`/`end_frame`, `max_reference_images/videos/audio`, `multi_prompt`, `audio_toggle`. Passing an unsupported input fails with a clear error \u2014 check first, don\'t trial-and-error paid calls.\\n- Most video models support only 16:9 / 9:16 / 1:1. A 3:4 request hard-fails on most.\\n- `--seed` reproduces a specific output on models that support it (e.g. Flux, Ideogram V4). GPT Image 2.5 rejects any explicit seed; older models may ignore unsupported seeds. You do not need a seed for variation \u2014 `--num` already varies.\\n- `--rendering-speed` applies to Ideogram (V3: `Default`/`Turbo`/`Quality`; V4: `Turbo`/`Balanced`/`Quality`) and affects image cost \u2014 pass it to `videodraft costs ... --rendering-speed <tier>` for an accurate estimate. Always trust `videodraft models image --json` over this list; new models and tiers appear there the moment the platform ships them, with no CLI update.\\n- `seedream-v5-pro` supports unified text-to-image and reference-image editing with up to 10 image references. Use `--resolution 1K` for 7 credits/image or `--resolution 2K` for 14 credits/image.\\n- Reference inputs: `--ref <img>` (images, including up to 10 total for Gemini Omni 1.1 Flash and Wan 3.0, and 7 for Grok 1.5), `--source-video <v>` (Gemini uploaded edit/extension source), `--ref-video <v>` (up to 3 creative videos <=3s each for Gemini Omni 1.1 Flash; also Wan 3.0, MiniMax H3, MiniMax H3 Max, and Seedance 2), `--ref-audio <a>` (Wan 3.0, MiniMax H3, MiniMax H3 Max, Seedance 2), and `--element \'<json>\'` or `--element @elements.json` for Kling V3/O3. For an exact Seedance 2.x reference-video `--estimate`, add `--ref-video-seconds <combined-seconds>`; Seedance bills input seconds alongside output. Wan 3.0 and MiniMax H3 do not bill input reference seconds; MiniMax H3 Max bills them as pooled reference tokens. The CLI uploads local media references and every structured element without flattening the source/reference roles. `--segment \\"<prompt>:<seconds>\\"` (repeatable) drives Kling 3.0, Kling 3.0 Turbo, and O3 multi-prompt generation. Use 1-6 segments of 1-15 whole seconds each, with 3-15 seconds total. `generate image --video-ref` is the nano-banana-2 video reference.\\n- The top-level prompt is OPTIONAL for Gemini Omni 1.1 Flash media-input or continuation calls, Wan 3.0 frame/reference modes, `generate video` with multi-prompt models, and Kling 3.0 Turbo (`--model kling-v3-turbo`) image-to-video. Text-only Gemini and Wan calls still require a prompt unless their live schema says otherwise.\\n- Hosted AI Production fallback: `videodraft produce <project> --mode full_video` generates one Seedance 2 video per scene, with real people allowed by default at the higher Fal-tier rate on every submitted scene segment; add `--no-allow-real-people` for the lower Byteplus-only rate when no scene shows a real identifiable person. The MCP equivalent is `produce_project` with `mode: \\"full_video\\"` (and `allow_real_people: false` to opt out). If a partial opted-out run returns `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, rerun the same project once with an explicit `--allow-real-people` after cost confirmation. The server reconciles asynchronous results first, preserves running/completed jobs, and retries only failed placeholders carrying that exact code. Poll with `videodraft generations`, then `videodraft finalize <project>` swaps them into the hosted timeline before `export`. In VideoDraft ADE, do not choose this path while `videodraft_editor` is available unless the user explicitly requests hosted production. Generate or download the scene assets, import them, and assemble/export with the native editor instead. If the user explicitly requests another compatible video model for a hosted production, do not use this fixed Seedance path; generate the project shots manually with the requested model and attach them to the hosted timeline.\\n\\n## Cost model\\n\\n- Images: per image (\xD7 `--num`). Matrix-priced models (GPT-Image, Nano Banana Pro, Seedream v5 Pro) vary by resolution/quality.\\n- Video: usually credits/second \xD7 duration; rate depends on model + resolution + quality + native audio on/off.\\n- Gemini Omni 1.1 Flash: 3 / 10 / 15 / 30 credits per output second at 360p / 720p / 1080p / 4K (720p default). Extend appends an explicit 3-10 seconds to a 1-30s source, whether uploaded or resolved from a prior interaction; 40 seconds total is reachable only while the source stays at or under 30s. Fal BYOK generation and basic source edits cost zero VideoDraft credits; continuation, extension, and separate creative references on a source edit are unavailable under Fal BYOK.\\n- MiniMax H3: 5 / 6 / 13 / 16 credits per output second at 480p / 768p / 2K / 4K (768p default). The first 5 reference images are included, then 8 credits for each additional image. Reference video and reference audio are NOT billed.\\n- MiniMax H3 Max: 5 credits per output second at 480p or 8 credits per second at 768p (default), plus pooled reference tokens. The first 4,096 reference tokens are free and each additional 1,000 costs 2 credits. An image is `(width x height) / 1024` tokens (a 1024x1024 reference is 1,024, so four of them are free); a reference video is 2,886 tokens per second at 480p or 7,459 at 768p; reference audio is about 2,121 per second. Pass `--ref-audio-seconds` for an exact estimate with audio references. The server measures each reference image, so a quote that assumes 1024x1024 is a floor.\\n- Wan 3.0: 7 / 14 / 28 credits per output second at 480p / 720p / 1080p. Auto duration reserves 30 seconds and reconciles unused credits from the provider-reported output duration. Fal BYOK charges zero VideoDraft credits.\\n- Grok Imagine Video 1.5: 8 credits/output second at 480p, 14 at 720p, or 25 at 1080p, plus 1 credit for each first-frame or reference image. Native generated audio is part of every output.\\n- Kling voice control: 17 credits/output second for Kling 2.6 Pro, 16 at Kling V3 Standard, and 20 at Kling V3 Pro.\\n- Shot-image batches: one image per shot (+1 grid image per scene in `--grid` mode) \u2014 the largest single spend in the pipeline.\\n- VEED Fabric avatar renders: ~10 credits/sec at 480p, ~20/sec at 720p. Avatar creation and its speech are bundled/free; only optional portrait generation/upscaling adds cost before the render.\\n- Direct VEED Fabric: text or normal audio is 8 credits/sec at 480p and 15/sec at 720p; fast audio is 10/sec at 480p and 20/sec at 720p.\\n- MiniMax H3 Max Lip Sync: 5 / 8 / 16 / 32 credits per output second at 480P / 768P (default) / 1080P / 2K. The video runs as long as the audio (at least 5s; only the first 14.8s is used), billed on the server-measured length rounded up, so at most 15 seconds. Use MP3, WAV, M4A (AAC), or AAC audio; other formats run only on the user\'s own Fal key.\\n- Sync Labs Lipsync 2: 5 credits per verified audio second.\\n- Voiceover TTS: 10 credits per 1000 characters for standard voices (including ElevenLabs Standard, Eleven v4), 5 per 1000 for ElevenLabs Turbo (Eleven v4 Turbo), and 30 per 1000 for cloned `custom-*` voices (min 1, pro-rated); applies to standalone voiceovers AND per-scene narration during `produce`. Quote Turbo with model id `voiceover-turbo` (CLI: `costs voiceover --mode turbo`) and cloned voices with `voiceover-cloned`. Silent tracks are free. Voice cloning itself is a flat 150 credits per clone.\\n- Lyria music: flat per track, 4 credits (Clip) / 10 credits (3.5) / 8 credits (legacy Pro). Fal BYOK uses zero VideoDraft credits.\\n- Seed Audio 1.0: 19 credits per actual output minute, prorated and rounded up to a whole credit. VideoDraft reserves the 120-second maximum of 38 credits and refunds the unused portion after generation. Fal BYOK is free.\\n- ElevenLabs audio: sound effects are per second, dialogue is per character, music/voice-changer/dubbing are per started minute. ElevenLabs Music v2.5 and v1 both cost 60 credits per started output minute; estimate a composition plan with its total length. Voice changer and dubbing reject source media above 300s in the current synchronous flow.\\n- Seedance 2.0 / 2.5 real people: allowed by default, which keeps Byteplus first, permits a submit-time Fal fallback, and prices at Fal\'s rate for that tier: 2.0 Mini 8/16 cr/s, Fast 11/25, Standard 14/31/69/156, 2.5 23/48/114 for 480p/720p/1080p. `--no-allow-real-people` uses the Byteplus-priced path instead (2.0 Mini 4/8, Fast 6/13, Standard 7/16/38/78, 2.5 11/24/57); Byteplus refuses real-person likenesses, so a likeness-policy failure then does not fall back. The gap is roughly 2x but not exactly: the Seedance 2.0 1080p pair is 38/69, or about 1.82x. If Byteplus accepts the task and later rejects the generated output, VideoDraft refunds the failure but does not resubmit it to Fal. Turn the option off only for the cheapest run when nothing in the job is a real identifiable person.\\n- Grok Imagine images: `grok-imagine` is a flat 2 cr (3 with a reference). `grok-imagine-2.0` is a separate, newer model priced by resolution and quality: 1K 4 (low) / 6 (medium), 2K 6 / 8, plus 1 cr per reference image (up to 3). v1 is NOT superseded \u2014 pick it when cost matters more than 2K.\\n- xAI bills refused requests, so failed Grok generations are not refunded.\\n- Upscales: priced by output size, mode, and (video) duration/fps; see the Upscaling section above.\\n\\nQuote before spending:\\n\\n```bash\\nvideodraft costs gemini-omni-1.1-flash --type video --duration 8 --resolution 720p --audio\\nvideodraft costs topaz-upscale-video --type video --duration 10 --width 1920 --height 1080 --resolution 4k --mode generative\\nvideodraft costs topaz-upscale --type image --width 2048 --height 2048 --scale 2x\\nvideodraft costs topaz-interpolate-video --type video --duration 10 --width 1920 --height 1080 --fps 60 --topaz-model Apollo\\nvideodraft costs minimax-h3 --type video --duration 10 --resolution 2K --ref-images 7\\nvideodraft costs minimax-h3-max --type video --duration 10 --resolution 768p --ref-images 2\\nvideodraft costs grok-imagine-video-1.5 --type video --duration 8 --resolution 720p --ref-images 4\\nvideodraft costs seedance-2 --type video --duration 15 --resolution 720p --quality standard --audio\\nvideodraft costs grok-imagine-2.0 --type image --resolution 2K --quality medium --num 2\\nvideodraft costs seedance-2 --type video --duration 15 --resolution 720p --quality standard --audio --no-allow-real-people # lower Byteplus-only rate\\nvideodraft costs elevenlabs-dubbing --type audio --duration 60\\nvideodraft costs seed-audio-1.0 --type audio --duration 60 # scenario only; model controls actual length\\nvideodraft costs elevenlabs-dialogue --type audio --chars 350\\nvideodraft costs voiceover --type audio --chars 800 # TTS: 10 cr / 1000 chars\\nvideodraft costs voiceover --type audio --chars 800 --mode turbo # ElevenLabs Turbo: 5 cr / 1000 chars\\nvideodraft generate video \\"...\\" --model gemini-omni-1.1-flash --estimate # same quote, inline\\nvideodraft generate video \\"...\\" --model minimax-h3-max --duration 8 --resolution 768p --prompt-expansion-mode balanced\\n```\\n","references/pipeline.md":"# VideoDraft pipeline reference\\n\\nEverything here describes the hosted fallback pipeline through the CLI (`videodraft <command>` / `videodraft call <tool>`) or hosted MCP connector (tool names in backticks). When the local `videodraft_editor` MCP is available, do not use hosted production or export by default. Use hosted tools only for asset generation and optional script/storyboard stages, then import the results and finish with the native editor reference linked from SKILL.md. Continue through `produce_project` and `export_video` only when the user explicitly requests a hosted web production or the native editor is unavailable.\\n\\nUse direct asset tools for standalone images, clips, audio, upscales, and descriptions. Use a hosted project when the user explicitly wants the editable web project, when a hosted storyboard stage is useful, or when the native editor is unavailable. Script-only uses a script-stage project and stops at the script. In VideoDraft ADE with editor tools present, stop before hosted production, import the generated assets, and build/export the native project.\\n\\n## Stages and their tools\\n\\n| Stage | CLI | Underlying tool |\\n| ---------------------------------------- | --------------------------------------------------------------------------- | ------------------------------------------- |\\n| Idea \u2192 full storyboard project | `videodraft create \\"<idea>\\"` | `generate_storyboard_from_idea` |\\n| Idea \u2192 script only (stop there) | `videodraft create \\"<idea>\\" --script-only` | `generate_script_from_idea` |\\n| Footage IS the video | `videodraft call generate_storyboard_from_media` | `generate_storyboard_from_media` |\\n| Batch shot images | `videodraft shots <project>` | `generate_shot_images` |\\n| One shot image | `videodraft generate image --project <id> --scene N --shot M` | `generate_image` |\\n| Produce (voiceover, captions, timeline) | `videodraft produce <project>` | `produce_project` |\\n| Seedance full-video production | `videodraft produce <project> --mode full_video` | `produce_project` with `mode: \\"full_video\\"` |\\n| Per-shot motion prompts | `videodraft video-prompts <project>` | `generate_video_prompts` |\\n| Motion clip for a shot | `videodraft generate video --project <id>` | `generate_video` |\\n| Attach a finished clip to the timeline | `videodraft attach <project> --scene N --shot M --media <url> --type video` | `attach_media_to_shot` |\\n| Find free b-roll (no credits) | `videodraft stock search \\"<query>\\"` | `search_stock_media` |\\n| Copy stock media onto the CDN | `videodraft stock import <ref>` | `import_stock_media` |\\n| Background music | `videodraft generate music --attach <project>` | `generate_music` / `set_background_music` |\\n| Song with vocals, lyrics or sections | `videodraft generate music --model elevenlabs-music-v2.5 --section \\"...\\"` | `generate_music` with `composition_plan` |\\n| General or reference-driven audio | `videodraft generate audio \\"...\\"` | `generate_audio` |\\n| Sound effect | `videodraft generate sound-effect \\"...\\"` | `generate_sound_effect` |\\n| Dialogue audio | `videodraft generate dialogue --line \\"voice:text\\"` | `generate_dialogue` |\\n| Voice changer | `videodraft generate voice-changer <audio>` | `change_voice` |\\n| Dubbing | `videodraft generate dub <audio_or_video>` | `dub_media` |\\n| Scene voiceover | `videodraft generate voiceover --project <id> --scene N` | `generate_voiceover` |\\n| Avatar script | `videodraft avatar script \\"<idea>\\"` | `generate_avatar_script` |\\n| Avatar + speech | `videodraft avatar create <portrait> --script \\"...\\"` | `create_avatar_video` |\\n| Talking-head render | `videodraft avatar render <avatar_video_id>` | `render_avatar_video` + `get_avatar_video` |\\n| Direct portrait + text/audio | `videodraft avatar fabric <portrait> --text \\"...\\"` or `--audio <file>` | `generate_veed_fabric_video` |\\n| Short portrait clip from audio, up to 2K | `videodraft avatar h3-lipsync <portrait> --audio <file>` | `generate_minimax_h3_lipsync_video` |\\n| Existing video + replacement audio | `videodraft avatar lipsync <video> --audio <file>` | `generate_sync_lipsync_video` |\\n| Existing-video AI edit | `videodraft edit video <video> \\"<change>\\" --model <video-edit-model>` | `edit_video` |\\n| Motion transfer | `videodraft edit motion <image> [direction] --motion-video <video>` | `generate_motion_control_video` |\\n| Image enhancement/upscale | `videodraft upscale image <image>` | `upscale_image` |\\n| Video enhancement/upscale | `videodraft upscale video <video>` | `upscale_video` |\\n| Frame rate / slow motion | `videodraft interpolate <video> --fps 60 [--slowdown 4]` | `interpolate_video` |\\n| Final MP4 | `videodraft export <project>` | `export_video` + `check_export_status` |\\n\\n## Rules that prevent broken results\\n\\n- **The storyboard is generated FROM the script**, never from the raw idea. `videodraft create` runs the whole chain correctly. Don\'t call `generate_storyboard_scenes` with a raw idea as the \\"script\\".\\n- **Visual consistency**: never generate a storyboard shot in isolation. Shot prompts carry `[[asset:Name]]` / `[[shot:X-Y]]` tags that `generate_shot_images` resolves against the project\'s visual assets and prior shots. When generating a single shot whose prompt has no tags, pass `--ref` images yourself (the project\'s visual assets and/or the previous shot\'s image; `projects get` exposes both). For scenes with multiple shots or recurring characters, prefer `videodraft shots <project> --model <selected-image-model> --grid`: preserve an explicitly requested compatible image model, otherwise use `nano-banana-2`. It creates one coherent scene grid, then decodes it into individual shot images.\\n- **Reference-first video**: when identity, styling, or composition matters, do not generate each motion clip from text alone. Generate or select the shot still first, then pass the decoded shot image as `--start-image` or `--ref` to the selected video model. AI Production already composes scene grids and sends them to Seedance as references. If the user explicitly requests another compatible video model, bypass fixed Seedance full-video mode and generate the per-shot clips with the requested model, using the individual decoded shot images as anchors.\\n- **Seedance full-video real people**: hosted `full_video` allows real people by default, which applies Fal-tier pricing to every submitted segment and permits the Byteplus-to-Fal fallback. For the lower Byteplus-only rate when no scene grid shows a real identifiable person (non-people, anime, clearly synthetic or stylized characters), use `videodraft produce <project> --mode full_video --no-allow-real-people`, or MCP `produce_project` with `mode: \\"full_video\\", allow_real_people: false`. If a partial opted-out run returns `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, re-estimate, follow the user\'s spend-confirmation preference, and rerun the same project once with an explicit `--allow-real-people` / `allow_real_people: true`. The server reconciles asynchronous results first, preserves running/completed jobs, and retries only failed placeholders carrying that exact code. Do not loop when it was already on. VideoDraft refunds a Byteplus task rejected after asynchronous acceptance, but cannot reroute it; rephrase or change the scene references instead.\\n- **Narration voice mode**: ElevenLabs narration runs on Eleven v4. Standard is the default and the best quality, at 10 credits per 1000 characters. `videodraft produce <project> --voice-mode turbo` (MCP `produce_project` with `voice_mode: \\"turbo\\"`) uses Eleven v4 Turbo, which is faster at 5 credits per 1000 characters. For one scene, `videodraft generate voiceover --project <id> --scene N --mode turbo` (MCP `generate_voiceover` with `mode: \\"turbo\\"`) does the same. Google, OpenAI and cloned `custom-*` voices ignore the mode.\\n- **Hold off generating shot images while the user is still iterating** on storyboard structure.\\n- **produce \u2192 export ordering**: `export` requires a produced project where every production scene has timeline media. If `produce` returns `generating_shot_images`, poll the job ids it returns, then re-run produce.\\n- **Do not attach motion clips before production exists**: run `produce` successfully first, then attach finished motion clips to the production timeline. Attaching before `production_data` exists cannot place them in the final timeline.\\n- **Generated motion clips do not auto-attach**: after `generate video` completes, attach the clip with `attach_media_to_shot` (`media_type:\\"video\\"`, include `duration_seconds`) \u2014 it replaces the production timeline clip while keeping the storyboard still.\\n- **Talking heads use dedicated avatar tools**: do not use `generate video`. Use managed `avatar create` and `avatar render` for reusable avatars, direct `avatar fabric` for a portrait plus text/audio, `avatar h3-lipsync` for a short (5-14.8s) portrait clip from existing audio at up to 2K, and `avatar lipsync` for an existing video plus replacement audio. Reuse a supplied person image or generate a clear front-facing portrait with the explicitly requested compatible image model, otherwise Nano Banana 2. Managed avatar creation and speech are bundled/free; direct Fabric, H3 Max Lip Sync, Sync, and render are paid.\\n- **Existing-video edits use their own category**: call `edit_video` or `videodraft edit video` with a `video_edit` model when transforming the source itself. Kling O3 also has a reference-generation mode that creates a new guided clip. Wan 3.0 is a text/frame/reference generation model, not a source-video editor. Motion transfer similarly uses `generate_motion_control_video` or `videodraft edit motion` with a `motion_control` model.\\n- **Upscaling preserves rather than redesigns**: use Topaz when resolution, detail, or cleanup is the problem (generative mode by default: Wonder 3.5 / Starlight Precise 2.6, Topaz\'s pick for AI sources; precision for real photos/footage or the cheapest pass; creative only for an artistic reinterpretation; `interpolate` for frame rate or slow motion). Regenerate or edit when the subject, text, framing, continuity, or motion is wrong. Upscale a low-quality avatar portrait before Fabric; do not render a new avatar at 480p just to upscale the result.\\n- **Timeouts on the one-shot create**: if `create` times out at the transport layer, the project was still created server-side \u2014 `videodraft projects list`, take the most recent, and resume with its id. Don\'t start a duplicate.\\n\\n## User-attached media: classify roles first\\n\\nFor EACH attached file decide:\\n\\n- **visual_asset** \u2014 recurring reference (character / product / location / style). Upload, then pass in `visual_assets` of `generate_storyboard_from_idea` (via `videodraft call`), or add to an existing project with `add_visual_assets`. Type must be one of `character | object | location | style | custom` with a short name + concrete description.\\n- **shot** \u2014 the media IS footage for the video. Whole video = footage \u2192 `generate_storyboard_from_media`. Idea + footage \u2192 `generate_storyboard_from_idea` with `shot_media`. Existing storyboard \u2192 `attach_media_to_shots`.\\n- **reference** \u2014 inspiration only \u2192 fold a description into the idea/instructions; don\'t place it as a shot or asset.\\n\\nAmbiguous (e.g. a person holding a product)? Ask the user.\\n\\nUploads persist in the media library \u2014 recall later with `videodraft media list`.\\n\\n## Editing project data safely\\n\\n1. `videodraft call get_project_schema` \u2014 read the structure once per session.\\n2. `videodraft projects get <id> --raw` \u2014 the exact editable blob.\\n3. Modify; then `videodraft call update_project --stdin` with `{\\"project_id\\": \\"...\\", \\"data\\": {...}}`.\\n - Objects deep-merge key-by-key; **arrays replace wholesale** \u2014 send the complete array you\'re changing (e.g. all of `storyboard.scenes`).\\n - Scene shot arrays (`image_prompt` / `shot_types` / `shot_actions` / `search_prompt` / `preview_media`) are auto-aligned; fix-ups come back as warnings.\\n4. Snapshot before risky edits: `videodraft checkpoint create <id> --name \\"before re-script\\"`. Restore with `videodraft checkpoint restore <id> <version>`.\\n\\n## AI Studio sessions (standalone generations)\\n\\nProject generations group automatically, and standalone work is grouped per conversation (MCP) or per working directory (CLI) by the connection session \u2014 see the \\"AI Studio sessions\\" section of SKILL.md. Once the task is clear, give that automatic session a concise 3-6 word title before the first generation:\\n\\n```bash\\nvideodraft sessions name \\"Fox Brand Explorations\\"\\nvideodraft generate image \\"...\\"\\n```\\n\\nNaming creates the session with that title. If generation, a user, or an earlier agent created it first, the existing name is preserved. The command refuses to run while `VIDEODRAFT_SESSION` is set, because that override would send later generations to a different session. Use `sessions create` plus `--session` only to continue or deliberately create a separate group.\\n"}') {
7760
+ return JSON.parse('{"SKILL.md":"---\\nname: videodraft\\ndescription: Create and edit AI videos, images, 3D meshes, Seed Audio, voiceovers, music, sound effects, dialogue, dubbing, storyboards, avatar videos, media upscales, and product/ad videos with VideoDraft. Use whenever the user mentions VideoDraft; asks to generate a video, image, 3D asset, audio asset, ad, explainer, storyboard, avatar, upscale, or batch/CI workflow; wants to download a video from YouTube, Instagram, TikTok, X or another site; or wants to assemble, cut, caption, mix, lay out, inspect, or export a native VideoDraft Editor timeline. Covers the cloud `videodraft` CLI/MCP and local headless `videodraft_editor` MCP. When the editor MCP is exposed, prefer it for production, timeline assembly, and export; use cloud production/export only when explicitly requested or the editor is unavailable.\\n---\\n\\n# VideoDraft\\n\\nVideoDraft is an AI video creation platform where asset generation is the priority lane:\\n\\n- **Asset generation**: standalone images, video clips, Seed Audio, voiceovers, music, sound effects, dialogue, voice-changed audio, dubbed media, upscales, and image descriptions. This is the fastest and most important lane. Treat these as complete deliverables when the user asks for assets.\\n- **Asset I/O**: upload local files, download outputs, auto-upload local references, and save generated media where the user can see it.\\n- **3D assets**: standalone Meshy 7 and Tripo H3.1 meshes, multi-view generation, complete artifact downloads, and Meshy humanoid rigging through MCP/CLI. Read [references/3d.md](references/3d.md), or `videodraft skills show 3d`. These assets have their own saved library and do not require an AI Studio session or web viewer.\\n- **Native editing**: local `.vdproject` timelines, cuts, layouts, captions, effects, audio, and exports through the headless VideoDraft Editor. Inside VideoDraft ADE, this is the default production and export lane whenever `videodraft_editor` is available.\\n- **Hosted project production**: idea \u2192 script \u2192 storyboard \u2192 hosted production timeline \u2192 exported MP4. Use the early stages for scripts, storyboards, and generated assets when useful. Treat hosted production and export as a fallback when the native editor is unavailable, or as an explicit destination when the user asks for an editable web project or hosted workflow.\\n\\n## How to connect\\n\\nGeneration billing is selected in VideoDraft Settings \u2192 API keys. Where enabled,\\n**VideoDraft routing** chooses an equivalent provider internally without changing\\nthe quoted customer credit price. A selected personal key is exclusive: never\\nswitch to VideoDraft credits or another key after an unsupported-model or\\nprovider error. Keep the canonical VideoDraft model ID in CLI/MCP calls. Poll\\nthe original generation after a timeout instead of starting another paid job.\\nNew provider connections can support only a subset of models and settings;\\nfollow the server\'s availability response. ElevenLabs remains audio-only.\\nFor Pika GPT Image 2.5, supply an explicit quality (`low`, `medium`, `high`,\\n`xhigh`, or `max`); `auto` is not supported by that connection. Nano Banana 2\\nLite is 1K only. Never remove requested controls or substitute a different\\nmodel merely to make a personal connection accept the job.\\n\\nCloud generation has two equivalent surfaces (same backend, credits, and hosted projects). Native timeline editing is a separate local surface:\\n\\n1. **CLI** (preferred when you have a shell): run `videodraft` if it\'s on PATH; otherwise `npx -y videodraft@latest` runs it with no install (needs Node \u226520; the `-y` skips npx\'s install prompt so it runs non-interactively; the package is fetched on first use and cached). For heavy use, `npm install -g videodraft`. If there\'s no Node/shell here but the MCP connector below is available, use that instead; if neither works, tell the user how to install (https://videodraft.ai/cli).\\n - Auth \u2014 pick by context, don\'t guess:\\n \u2022 INTERACTIVE (a human is in the session, e.g. Claude Code / Codex): on exit code 3 (\\"not authenticated\\"), tell the user to run `videodraft login` in their terminal \u2014 it opens their browser for a one-click VideoDraft sign-in (OAuth), no key to copy. Wait for them to confirm it succeeded, then retry the command. This is the preferred path when the user is present.\\n \u2022 HEADLESS / CI (no browser): set `VIDEODRAFT_API_KEY=vd_mcp_...` (a token the user mints at https://app.videodraft.ai/mcp-keys).\\n \u2022 SECURITY: never ask the user to paste a `vd_mcp_...` token into the chat \u2014 use browser `login` or the env var so the token never lands in the transcript.\\n - Every command accepts `--json` (parse this, don\'t scrape text). Exit codes: 0 ok, 1 error, 2 usage, 3 auth (see Auth above), 4 insufficient credits (\u2192 tell the user, don\'t retry).\\n - Tool discovery: start with `videodraft tools list` for the grouped catalog, then narrow with `videodraft tools list --lane assets`, `--lane asset_io`, `--lane project_data`, or `--lane production`.\\n - Asset lane: `videodraft generate ...`, `videodraft edit video|motion`, `videodraft avatar ...`, `videodraft upscale ...`, `videodraft interpolate ...`, `videodraft upload`, and `videodraft download`.\\n - Full API access: `videodraft tools schema <name>`, `videodraft call <tool> --args \'<json>\'`.\\n2. **MCP connector**: if VideoDraft MCP tools (e.g. `generate_storyboard_from_idea`) are available, call them directly \u2014 the CLI\'s curated commands map 1:1 onto these tools.\\n3. **Native editor MCP** (`videodraft_editor`): prefer this for project production, timeline assembly, cutting, layouts, transitions, captions, audio placement, and final export. Inside VideoDraft ADE on a supported Mac, Claude, Codex, OpenCode and Grok receive it automatically in both Code and VideoDraft modes. It runs headlessly, so an Open Editor click is not required. Start with `project_manage` (`operation: \\"project\\"`, action `list`, `open`, or `create`); standalone asset generation remains in the cloud CLI or MCP.\\n\\nSend native editor edits serially, batching everything a request needs into one `edit_apply` recipe. Pass the latest `context` when a person may be editing at the same time, so a stale write is refused. See [references/editor.md](references/editor.md) for project selection, media import, timing units, edit recipes, verification, export, and the `videodraft-editor` terminal bridge.\\n\\nIf you are reading this skill through `videodraft skills show skill`, run `videodraft skills show editor` before native editor work to load that reference.\\n\\n**VideoDraft ADE routing rule:** the presence of `videodraft_editor` means the native editor is ready, even when no editor window is visible. Use cloud tools to generate or source assets and, when helpful, scripts or storyboards. Do not call hosted `produce_project` / `videodraft produce` or `export_video` / `videodraft export` by default. Import the assets into the native project, assemble there, and export with native `delivery_manage` `submit`. Use hosted production/export only when the user explicitly asks for the web workflow or the native editor tools are unavailable. Do not silently fall back to hosted production after a native tool error.\\n\\n## First decision: asset, hosted project, or native edit?\\n\\n- **One standalone asset** (image, clip, voiceover, music track, sound effect, dialogue track, voice-changed file, dubbed media file, upscale, or description): generate it directly. Do NOT create a project.\\n - `videodraft generate image \\"a red fox in snow, cinematic\\" --ar 16:9 --download ./out/`\\n - `videodraft generate video \\"slow dolly over a misty lake\\" --model gemini-omni-1.1-flash --duration 6 --download ./out/`\\n- **Any final video, production timeline, existing footage, local `.vdproject`, or hands-on edit**: use `videodraft_editor` when available. List or open the intended local project, or create a native project for a new production. The editor can work without showing its UI.\\n- **A small set of related assets**: still stay in the asset lane. They are grouped automatically (see [AI Studio sessions](#ai-studio-sessions)); give the current group one useful task-specific name with `videodraft sessions name \\"<name>\\"`. Switch to a project only when the deliverable matches the project criteria below or the user asks to attach the assets to one.\\n- **A generated multi-scene video / ad / explainer**: when the editor is available, use hosted tools only for any needed script, storyboard, shot planning, or generated assets; stop before hosted production, import the assets, and build/export the native timeline. A hosted project is optional unless the user wants the web project or its storyboard workflow.\\n- **A hosted web project or hosted export**: use the hosted pipeline only when the user explicitly asks for it or the native editor is unavailable.\\n- **Just a script** (no video asked for): A script-only request creates a script-stage project but stops at the script. Use `videodraft create \\"...\\" --script-only`; do not build a storyboard the user didn\'t ask for.\\n- **Iterating on existing work**: identify the surface first. Use `project_manage` `project` with `action: \\"list\\"` for native projects and `videodraft projects list` only for hosted work. Never create a replacement project just to change an existing one.\\n\\n## Choose the model from the task\\n\\nIf the user names a model, use it when compatible. If it cannot handle the request, explain why and recommend alternatives instead of silently switching. Otherwise inspect the inputs, duration, audio, quality, speed, and cost, check the live catalog, and pass an explicit model.\\n\\n### Seedance 2.x real-person rule\\n\\nSeedance 2.0 and 2.5 allow real people by default, the same as AI Studio:\\n\\n- The default is true. Generation tries Byteplus first and can fall back to Fal, which allows real-person likenesses, so a person in the prompt, a start frame, end frame, reference image, or reference video does not hard-fail. The request is charged at Fal\'s higher tier-specific rate even if Byteplus serves it.\\n- Turn it off for the cheapest run, only when the user wants that and nothing in the job is a real identifiable person (text-only requests, non-people, anime, clearly synthetic or stylized characters). For MCP use `allow_real_people: false`. For the CLI use `--no-allow-real-people`. That pins the job to the lower-priced Byteplus path, and a Byteplus likeness-policy refusal does not fall back to Fal. Pass the same value to `get_model_costs` or `videodraft costs` so the estimate matches the charge.\\n- If a request made with the option off fails with code `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, estimate the higher rate, follow the user\'s spend-confirmation preference, and retry exactly once with the option on. The structured recovery fields are `retryable: true`, `retry_with: { allow_real_people: true }`, `cli_flag: \\"--allow-real-people\\"`, and `retry_policy: \\"once\\"`. CLI `--json` submit errors expose them under `details`; `status` and `wait` include them on the failed job result. Do not treat an unrelated moderation or provider error as that signal.\\n- Do not loop when the option was on. Byteplus can accept a task and reject its generated output later. VideoDraft refunds that failed generation, but the late asynchronous failure cannot be rerouted to Fal. Rephrase the prompt or use different references before trying again.\\n- For hosted AI Production, the same default applies to every Seedance scene segment: `produce_project` with `mode: \\"full_video\\"` and `videodraft produce <project> --mode full_video` allow real people unless you pass `allow_real_people: false` or `--no-allow-real-people`. If an earlier run with the option off partially submitted and returns the opt-in code, rerun that same project once with an explicit `allow_real_people: true` / `--allow-real-people`. The server reconciles asynchronous results first, preserves running/completed jobs, and resubmits only failed scene-video placeholders carrying the exact opt-in signal. Keep the native-first VideoDraft ADE routing rule above: hosted full-video production is still explicit/fallback-only when the local editor is available.\\n\\n**How to pick a model. Follow this order every time:**\\n\\n1. The user named a model: use it, never substitute.\\n2. Otherwise pick a **Tier 1** model.\\n3. Use a **Tier 2** model only when no Tier 1 model supports the request, and say why. Never pick Tier 2 on your own for price, speed, novelty or variety.\\n\\nTier 1: images `nano-banana-2`, `nano-banana-pro`, `nano-banana-2-lite`, `gpt-image-2.5-flare`, `gpt-image-2.5-sunburst`, and `recraft-v4` for vector/SVG only; videos `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0`, `kling-o3`, `minimax-h3-max`; video edits `gemini-omni-1.1-flash`; talking-head portraits `veed-fabric`, `minimax-h3-max-lipsync`; motion transfer `kling-v3-motion-control`; audio ElevenLabs for speech tools, Lyria 3.5 for music at any length.\\n\\nTier 2 is every other catalog model (`wan-3.0`, `flux-3`, `minimax-h3`, `kling-v3-turbo`, `kling-2.6-pro`, `grok-imagine-video-1.5`, `happy-horse`, Veo 3.1, Sora, Seedream, Grok Imagine images, FLUX images, `sync-lipsync-2` and the rest). All of them stay available by name. Real Tier 2-only jobs: document or web-page references (`wan-3.0`), keyframes pinned to timestamps (`flux-3`), re-syncing existing footage (`sync-lipsync-2`).\\n\\nEvery `videodraft models image|video|audio --json` entry carries `tier` (1 or 2), and the top-level `recommended` array is Tier 1, best first. That catalog beats this page when they disagree.\\n\\n**Speech in video:** write the dialogue in the video prompt. `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` and `kling-o3` speak it natively. For a specific voice add a Seedance reference audio clip (`--ref-audio`) or a Kling bound voice. Do not plan silent video + TTS + lip-sync. Lip-sync tools are only for a portrait presenter talking to camera, or for re-syncing footage that already exists. Off-screen narration and voiceover are unaffected: keep using `generate voiceover` for those.\\n\\n**Images:**\\n\\n- `nano-banana-2`: general default, editing, consistency, and references.\\n- `nano-banana-pro`: maximum quality. `nano-banana-2-lite`: fast, inexpensive drafts.\\n- `gpt-image-2.5-flare`: replaces the GPT Image 2 recommendation for posters, logos, signs, title cards, readable text and composition. Use `gpt-image-2.5-sunburst` for extra precision/detail. Both use the same basic options as GPT Image 2: references (up to 16), aspect ratio, 1K/2K/4K resolution, image count (1-4), and quality. Quality adds xhigh/max; auto is charged at the max tier. Other model recommendations are unchanged; explicit `gpt-image-2` requests still use GPT Image 2.\\n\\n**Videos:**\\n\\n- `gemini-omni-1.1-flash`: general default for 3-10s generation, image-to-video, uploaded source edits up to 10s, and silent or voiceover-backed shots (audio is always on, so describe quiet ambience with no speech and mute or mix it on the timeline; only a file with no audio track at all needs `kling-3.0`, `kling-o3` or Seedance with `--no-audio`). It supports first and last frames, up to 10 total image inputs, up to 3 creative reference videos of at most 3 seconds each on `--video-task generate` ONLY, uploaded-video extension, and continuation of an earlier generation through `--previous-interaction-id`. Output is 360p/720p/1080p/4K at 3/10/15/30 cr/s with audio always on. Use `--source-video` for the uploaded edit/extension source; extension sources must be 1-30s. Edit and extend accept EXACTLY ONE input video, so `--ref-video` cannot accompany `--source-video` or `--previous-interaction-id`; only `--ref` images can. One `--ref-video` with no separate source remains a legacy source edit. A previous interaction defaults to conversational edit; add `--video-task extend` or `--extend` with an explicit 3-10s duration to append at the end. A continuation resolves to the prior turn\'s output and is submitted as an ordinary source, so the same limits apply to it: at most 30s to extend, at most 10s to edit. 40 seconds total is reachable, but only by extending from a source of 30s or less, so a ladder dead-ends once it passes 30s. New dialogue can only be added when the SOURCE video is silent; adding speech on top of a source that already contains speech is refused with \\"the model is currently unable to process speech edits\\". The server safely measures creative-reference durations, or you can repeat `--ref-video-duration` when a host blocks metadata probing. Fal BYOK supports its currently callable v1.1 generation and basic-edit endpoints at zero VideoDraft credits, but Fal does not expose continuation/extension or mixed source-edit references as callable endpoints and VideoDraft must never fall back to paid Google.\\n- `grok-imagine-video-1.5` (Tier 2): 1-15s text, first-frame, or 1-7 reference-image generation with native audio. Text/first-frame modes support 480p, 720p, and 1080p; reference mode supports 480p/720p. Cite references as `<IMAGE_0>` through `<IMAGE_6>`. It has no last frame, seed, negative prompt, quality tier, reference video, or reference audio.\\n- `minimax-h3` (Tier 2): 480p/768p/2K/4K (5/6/13/16 cr/s, 768p default), native stereo audio, and 5-15s text, first/last-frame, or mixed-reference generation. Reference mode accepts up to 9 images, 3 videos, and 3 audio clips, with at most 12 files total. Cite them as `Image 1`, `Video 1`, and `Audio 1` in array order.\\n- `minimax-h3-max`: 480p/768p pricing (5/8 cr/s, 768p default), native audio, and 5-15s text, first/last-frame, or mixed-reference generation. Reference mode accepts up to 9 images, 3 videos, and 3 audio clips, with at most 12 files total, cited as `Image 1`, `Video 1`, and `Audio 1` in array order. It supports a reproducibility seed and `disabled` / `balanced` / `quality` prompt expansion.\\n- `wan-3.0` (Tier 2): unified 2-30s text, first/last-frame, or ordered mixed-reference generation at 480p/720p/1080p (7/14/28 cr/s), with optional native audio. Reference mode accepts up to 10 images, 5 videos, and 5 audio clips, at most 20 media files total; video and audio each total at most 15 seconds. `--auto-duration` reserves 30 seconds and reconciles to the provider-reported output length. Document/web references use `--file-url` or `--web-url` and require `--thinking`.\\n- `flux-3` (Tier 2): Black Forest Labs FLUX 3. 5-20s at 720p/1080p with 24fps native audio, from a prompt, a first frame, first + last frames, or up to 10 keyframes pinned to specific moments (`--keyframe shot.png@2.5`, repeatable). `--quality draft` renders the same shot at 720p for roughly a third of the cost. Auto duration is text/first-frame only.\\n- `seedance-2`: 11-15s, video/audio/mixed references, wider ratios, selectable audio, or first/last frames. `mini` is the economical default; move to `standard` when the user asks for high quality, when a Mini result is not good enough, or for 1080p/4K; `fast` for speed.\\n- `seedance-2.5`: 4-30s single takes and up to 50 references (30 image, 10 video, 10 audio). Same modes as 2.0, one quality tier, 480p/720p/1080p. Reach for it when a shot must run past 15s or carry more references than 2.0 allows.\\n- `kling-o3`: reference images plus structured image/video elements, first/last frames, multi-prompt, audio control, or 4K. O3 allows 7 combined image references and image-backed elements, reduced to 4 combined items when a video-backed element is present. `kling-3.0`: image-to-video can use structured image/video elements and bind a custom Kling voice ID to either element form. Tier 2, by name: `kling-v3-turbo` (fast polished 3-15s with first frame, multi-prompt, and audio, but no elements) and Kling 2.6 Pro (top-level voice IDs cited as `<<<voice_1>>>` and `<<<voice_2>>>`).\\n- Existing-video edits use `videodraft edit video`, not generic generation. `gemini-omni-1.1-flash` is the preferred edit model and is chosen automatically when you omit `--model`: source up to 10s, up to 10 reference images, 360p/720p/1080p/4K with audio. Creative `--ref-video` inputs are NOT accepted on an edit, because an edit takes exactly one input video; use `generate video --video-task generate` to guide a new clip with video references instead. Omitting `--model` on a longer or unmeasurable source spends nothing and prints a priced menu (what each model edits, what it drops, what it costs) so you can put the choice to the user. **Truncation:** only Gemini refuses an over-length source. Happy Horse silently edits just the first 15s, Kling O3 the first 10s, and Grok the first 8s. The command warns you when that will happen; always relay it to the user. Gemini regenerates the audio track, so use Happy Horse or Kling O3 with `--preserve-audio` when the source audio must survive. Choose Grok for the cheapest prompt-only edit, Happy Horse for up to 5 image references, or Kling O3 for controlled reference-image edits.\\n- To edit a source longer than 10s without losing its tail, cut it into <=10s pieces in the native editor, edit each with Gemini, and reassemble. VideoDraft has no server-side split/concat, so this path needs `videodraft_editor`.\\n- Kling O3 also has a reference-generation mode. Use `videodraft generate video --model kling-o3-video-ref-edit` with exactly one `--ref-video` to generate a new guided clip; use `videodraft edit video` when changing the source itself. For new mixed-reference generation use Seedance 2 / 2.5.\\n- Motion transfer uses `videodraft edit motion` with Kling V3 by default, or Kling 2.6 when explicitly requested or lower cost matters. It requires a subject image and a motion-reference video.\\n- Veo 3.1, Sora, Grok, Happy Horse and the other Tier 2 video models: only when explicitly requested or when no Tier 1 model fits.\\n\\n**Audio and utilities:**\\n\\n- Use Seed Audio 1.0 for open-ended text-to-audio, speech/music/sound synthesis, voice conditioning, or prompt-driven editing with up to three audio references or one image. Use `videodraft generate audio`. Reference clips are `@Audio1`, `@Audio2`, and `@Audio3` in array order. There is no duration input. Output is up to two minutes and settles at 19 credits per actual minute, with up to 38 credits reserved during generation. The CLI automatically retries transient responses with one operation key. To recover after the CLI process itself is interrupted, set `--idempotency-key <uuid>` on the original command and reuse it.\\n- Prefer ElevenLabs for voiceover, dialogue, voice changing, dubbing, and sound effects. Honor an explicitly selected supported TTS voice/provider. Use `lyria-3.5` for all music, short or long (up to ~3 minutes, 10 credits flat), with vocals/lyrics or instrumental arrangements. When `--model` is omitted the CLI sends no model and the server default applies, which is `lyria-3.5` wherever the backend supports it. If the server answers that `lyria-3.5` is an unknown model, that backend predates it: rerun without `--model` (the server default then applies), or use `lyria-3-pro-preview` for a track longer than 30 seconds. Ask for the length in the prompt. Keep `lyria-3-clip-preview` (fixed ~30s, 4 credits) and `lyria-3-pro-preview` (8 credits) for explicit requests. For Lyria, put desired length, lyrics and \\"instrumental only, no vocals\\" in the prompt; `--length` and `--instrumental` are ElevenLabs-only. Use ElevenLabs Music for exact timing, composition plans, or a style reference track. Lyria allows 10 reference images on Google, 1 on Fal BYOK; Fal 3.5 prompts are limited to 5000 characters. BYOK uses zero VideoDraft credits. ElevenLabs Music means `elevenlabs-music-v2.5` (`elevenlabs-music` is an alias for it); use `elevenlabs-music-v1` only when the user asks for v1 by name.\\n- ElevenLabs voiceover runs on Eleven v4 in two modes. Standard is the default and the best quality, at 10 credits per 1000 characters. Turbo (Eleven v4 Turbo) is faster, at 5 credits per 1000 characters. Keep Standard unless the user wants faster or cheaper speech. Pick it with `generate voiceover --mode standard|turbo`, `produce --voice-mode standard|turbo`, or MCP `generate_voiceover.mode` / `produce_project.voice_mode`, and quote Turbo with `costs voiceover --chars <n> --mode turbo`. Google, OpenAI and cloned `custom-*` voices ignore the mode (cloned voices cost 30 per 1000). v4 has no style or speed settings and no SSML `<break>` tags; audio tags such as `[whispers]` work. Dialogue stays on Eleven v3.\\n- For ElevenLabs voiceover, dialogue, and voice changing, accept a supplied raw voice ID (16-64 alphanumeric characters) or `elevenlabs-<id>`. The voice catalog is for discovery, not an allowlist. Do not reject or substitute a supplied ID because it is absent from `videodraft models voices` / `list_available_voices`. Use `--voice <id>` for voiceover and voice changing, or repeat `--line \\"<id>:Text\\"` for dialogue; for example, `--voice kPzsL2i3teMYv0FxEYQ6` and `--line \\"elevenlabs-kPzsL2i3teMYv0FxEYQ6:Hello.\\"` use the same voice. The voice must be accessible to the provider account used for generation. Private or cloned voices may require the user\'s connected ElevenLabs key. Kling video-control IDs and MiniMax `custom-*` IDs are separate voice systems.\\n- A character who needs to TALK:\\n - **Speaking inside a scene**, or any ordinary clip with dialogue: `generate video` with the line in the prompt, in quotes, with who says it and how. `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` and `kling-o3` all voice it natively. For a specific voice use a Seedance `--ref-audio` clip or a Kling voice bound per element. This is the default path; do not generate speech separately and lip-sync it on.\\n - **Talking to camera from a portrait** (presenter, spokesperson, explainer): the avatar lane, and VEED Fabric is preferred. Use managed `avatar create` then `avatar render` for a reusable avatar record with bundled speech, `avatar fabric` for a one-off portrait plus text or existing audio, and `avatar h3-lipsync` for a short (5-14.8s) portrait clip from existing audio at up to 2K. Not for in-scene shots: Fabric animates a portrait facing the lens, so a cinematic request comes back as a head-on talking headshot.\\n - **Footage that already exists plus replacement audio** (re-dub, translation): `avatar lipsync`. This is the only job for Sync Labs unless the user names it.\\n- Enhancement: use Topaz image/video upscaling only when the content is already correct. Use image 1x for cleanup, 2x by default, 4x when justified; use video 2x by default. Edit or regenerate creative errors.\\n\\nSee [references/models.md](references/models.md) for the detailed routing table and exact capability limits.\\n\\nProvider routing uses a maintained local price catalog, with exact setting compatibility and account rules. Platform requests prefer Google\'s direct image/video APIs and BytePlus for Seedance 2.0/2.5 before comparing public prices. Seedance 1.5 retains its existing Replicate primary; other models remain price-based. Agents should keep using normal model IDs and `get_model_costs` / `--estimate` for customer credits. Do not choose a vendor from a marketing starting price or retry an accepted generation to chase a lower price. Unknown or expired comparable costs retain the existing route; a selected personal provider key stays exclusive and overrides platform-account preferences.\\n\\nTemporary provider promotions are excluded from routing prices. With an Atlas personal key, H3 Max supports text or first/last-frame generation at 480P/768P only with `prompt_expansion_mode: \\"disabled\\"` and no seed. Balanced/quality expansion, reference generation and H3 Max Lip Sync require the existing Fal path. Atlas H3 Max is not selected automatically while its published price units remain inconsistent.\\n\\n## Matching AI Studio controls\\n\\nUse `models image --json` / `models video --json` for the accepted options. Nano Banana Pro/2 expose `--temperature 0..2` and `--google-search-grounding true|false`. Qwen uses one `--ref` plus `--horizontal-angle 0..360`, `--vertical-angle -30..90`, and `--zoom 0..10`; its text prompt is optional. Recraft generates SVG and accepts `--quality Normal|Pro`, repeatable `--recraft-color \'#RRGGBB\'`, and `--recraft-background \'#RRGGBB\'`. Original Nano Banana and VideoDraft Image use fixed 1K output.\\n\\nKling 3.0 / 2.5 Pro accept `--cfg-scale 0..1` and `--negative \\"\\"` to clear a negative prompt. Seedance 2/2.5, Wan 3.0, FLUX 3 text/first-frame and Gemini Omni generation accept `--auto-duration`; do not combine it with `--duration`. FLUX 3 auto reserves 20 seconds, so its default estimate includes that ceiling. `--model sora-2 --quality pro --resolution 1080p` selects Sora Pro. Kling O3 reference/edit modes accept image-only `--element` objects, at most four combined with `--ref`. Grok v1 accepts image references for clips up to 10 seconds.\\n\\nTopaz image results may be immediate or queued. The CLI waits by default and downloads either result with `--download`; `--no-wait` returns a queued job ID. Raw MCP callers must poll `check_generation_status` when `upscale_image` returns `job_id`.\\n\\n## Prefer references when continuity matters\\n\\nPure text-to-image or text-to-video is fine for a generic one-off asset. When a specific character, product, location, style, composition, or brand identity must survive generation, use references instead of hoping the prompt recreates it.\\n\\n- If the user supplies reference media, preserve and pass it. Never reduce the request to text alone.\\n- When continuity matters, generate/select a strong still first with the selected image model (`nano-banana-2` by default), wait for its URL, then animate it as a start frame/reference. Confirm the combined image and video cost.\\n- When using a hosted storyboard stage for multiple shots, use `videodraft shots <project_id> --model <selected-image-model> --grid`, then animate the decoded shots. Preserve explicit models. In VideoDraft ADE, import the resulting assets into the native editor instead of continuing into hosted production. A requested non-Seedance video model must use manual per-shot generation instead of Seedance full-video mode.\\n\\n## Cost and credits\\n\\nDo not call `videodraft credits` before routine generations. Paid endpoints validate and deduct atomically; if the balance is insufficient, the request is rejected before the provider job starts (CLI exit code 4). Check the balance only when the user asks, gives a credit budget, or a large workflow needs budget planning.\\n\\nFor expensive work, estimate with `--estimate` or `videodraft costs`, state the selected model/settings/cost, and get a go-ahead. This matters most for shot-image batches, long or high-resolution video, AI Production, and paid audio batches. Honor the user\'s confirmation preference for the session.\\n\\nKling voice creation costs 1 VideoDraft credit on the platform Fal account and 0 credits with Fal BYOK. Preview it with `videodraft kling-voices create <sample> --name <name> --estimate`; the estimate does not create a voice or require consent confirmation. Actual creation requires `--confirm-consent`.\\n\\nMiniMax H3 costs 5 / 6 / 13 / 16 credits per output second at 480p / 768p / 2K / 4K (768p by default). In reference mode the first 5 images are included and each additional image costs 8 credits. Reference video and reference audio are NOT billed.\\n\\nMiniMax H3 Max costs 5 credits per output second at 480p and 8 credits per second at 768p (default), plus pooled reference tokens in reference mode: the first 4,096 are free and every 1,000 after that costs 2 credits, where an image is `(width x height) / 1024` tokens, a reference video is 2,886 tokens per second at 480p or 7,459 at 768p, and reference audio is about 2,121 per second. Use `--prompt-expansion-mode disabled|balanced|quality`; balanced is the default. Fal\'s provider safety checker is OFF by default and is not exposed in the app \u2014 only pass `--safety-checker true` if the user explicitly asks for it.\\n\\nWan 3.0 costs 7 / 14 / 28 credits per output second at 480p / 720p / 1080p. Auto duration reserves the 30-second maximum and refunds the unused reserve after the provider reports the actual whole-second output length. Fal BYOK runs on the connected user key and charges zero VideoDraft credits.\\n\\nGrok Imagine Video 1.5 costs 8 credits per output second at 480p, 14 at 720p, or 25 at 1080p, plus 1 credit for each first-frame or reference image. Native audio is always generated.\\n\\n`videodraft models image|video` lists the live image and video catalogs with supported inputs. Video entries are grouped as `generation`, `video_edit`, `motion_control`, `avatar_lipsync`, and `upscale`, and each reports the exact tool. Use `videodraft models video --category video_edit` to narrow the list. `videodraft models audio` lists Seed Audio, Google Lyria, and ElevenLabs audio/media tools, while `videodraft models voices` helps discover TTS voices. Consult them for capabilities; a supplied ElevenLabs voice ID does not need to appear in the voice catalog.\\n\\n## Async jobs\\n\\nImage/video generation is asynchronous: commands submit a job and **wait by default**, printing output URLs (and saving files with `--download`). Large downloaded images also get a downscaled copy in `previews/` next to them (the `preview` field / \\"inspect via preview\\" line in the output) \u2014 **look at the preview, deliver the original**; viewing full-resolution images bloats the chat permanently. In scripts/CI prefer explicit control:\\n\\n```bash\\nJOB=$(videodraft generate image \\"...\\" --no-wait --json | jq -r .job_id)\\nvideodraft wait \\"$JOB\\" --download \\"./outputs/{job_id}_{index}.{ext}\\" --json\\n```\\n\\nFor MANY jobs: submit each with `--no-wait`, collect ALL with one command \u2014 `videodraft wait <id1> <id2> ...` polls every job from one process with one batched request per tick. Do NOT spawn parallel `wait`/`generate --wait` processes for a batch.\\n\\nIf a wait times out, the job is still running server-side \u2014 `videodraft status <job_id>` later. Never re-submit just because a wait timed out (that double-spends credits).\\n\\nFor completed Wan 3.0 jobs, MCP `check_generation_status` and CLI `status`/`wait --json` include `outputMetadata` with Fal\'s returned `seed`, `duration`, and `actual_prompt` when present.\\n\\n## AI Studio sessions\\n\\nStandalone image/video/audio generations are filed into an AI Studio session in the web app. 3D assets use `assets 3d list/get` separately. You do not have to create an AI Studio session:\\n\\n- **MCP hosts** (Claude Code, claude.ai, Codex, VideoDraft ADE): on 2025-era MCP the server mints an `Mcp-Session-Id` on `initialize`; your host echoes it, and this conversation\'s generations land in their own session. On stateless MCP (2026-07-28) hosts that send a conversation id (ChatGPT) get the same. Otherwise `name_current_ai_studio_session` returns `pass_session_id: true` with a `session_id`: pass that `session_id` to every later standalone generation in the conversation. Tool results echo the session as `ai_studio_session_id`.\\n- **CLI**: the same handshake runs once per (profile, server, working directory) and is cached for 12 idle hours, so everything generated from one directory shares one session. `videodraft sessions current` shows it; `videodraft sessions reset` starts a new one.\\n- Project generations (`--project <id>` / `project_id`) always go to that project\'s session.\\n\\nOnce you understand the creative task, name the current automatic session before the first standalone generation:\\n\\n```bash\\nvideodraft sessions name \\"Purple Seal Rescue Short\\"\\n```\\n\\nChoose a concise, specific 3-6 word title for the intended work. Do not copy the client name, date, or exact chat title. Name it once: the operation creates the session with that title. If generation, a user, or an earlier agent created the session first, its existing name is preserved.\\n\\nApart from the `pass_session_id: true` case above, pass `--session <id>` / `session_id` only to **continue earlier work** or create an explicit separate group:\\n\\n```bash\\nSESSION=$(videodraft sessions create \\"Fox brand explorations\\" --json | jq -r \'.session.id\')\\nvideodraft generate image \\"a red fox in snow, cinematic\\" --session \\"$SESSION\\"\\nvideodraft generations --session \\"$SESSION\\" # what is in it\\nvideodraft sessions list --name fox # find it again later\\n```\\n\\n`VIDEODRAFT_SESSION=<id>` sets the default for every command (ignored when `--project` is given). While it is set, `sessions name` refuses to run because that command names the current automatic connection session, not the pinned override; unset it first or rename the pinned session in AI Studio. `VIDEODRAFT_SESSION_SCOPE=<label>` groups several directories into one connection session; `VIDEODRAFT_CLIENT_NAME=<host>` labels the fallback placeholder used when generation creates the session before `sessions name`; `VIDEODRAFT_NO_SESSION=1` disables the handshake. Generations that reach the server with no session at all fall back to the account-wide \\"Agent (MCP)\\" session; if you see work landing there, pass `--session` explicitly.\\n\\n## Generation history\\n\\nPast work is queryable \u2014 reuse a previous setup instead of guessing. `videodraft generations` lists recent generations; scope with `--session <id>` or `--project <id>` (includes collaborators\' rows in shared scopes; pass one, project wins), and filter with `--type`, `--model`, `--favorites`. `--full --json` returns each row\'s exact parameters (aspect ratio, resolution, duration, references) \u2014 the human table stays compact, so pair `--full` with `--json`. `videodraft generation <id>` prints one generation\'s complete recipe (prompt, input image, parameters, outputs); `--favorite` / `--unfavorite` stars it. `videodraft sessions list` shows AI Studio sessions (owned + shared) with `--name` search \u2014 take a session id from there to read its history.\\n\\n## Free stock footage and photos\\n\\nPexels and Pixabay are wired in through `search_stock_media` / `import_stock_media` (`videodraft stock search \\"<query>\\"`, `videodraft stock import <ref>`). Both libraries are free, watermark-free and cost **zero credits**, so use stock for b-roll, establishing shots, backgrounds and ordinary real-world footage, and spend credits on the shots that must be specific: the product, the character, the scripted action.\\n\\n```bash\\nvideodraft stock search \\"city skyline night\\" --min-duration 5 --orientation landscape\\nURL=$(videodraft stock import pexels:video:35379336 --quality hd --json | jq -r .url)\\n```\\n\\n- Always two steps. A search result\'s `preview_url` is a thumbnail for judging the shot; never place it on a timeline, attach it to a shot or send it to a model. `import_stock_media` copies the file onto the VideoDraft CDN and returns the URL every other surface accepts, including the native editor\'s `library_manage` import.\\n- `--quality hd` (default) caps video at 1080p; `4k` caps at 2160p, `best` takes the largest the provider has, `sd` suits rough cuts. Stills ignore quality and import at full size. Use `--orientation portrait` for 9:16, and `--min-resolution 1920` to drop anything below HD on its long edge.\\n- Photos come from Pexels. Pixabay contributes video only, because its full-size image host refuses server-side downloads.\\n- Credit the creator and link the provider page when you show results or deliver the finished work; both come back on every result. Skip clips that imply a person or brand endorses the product, and avoid recognisable logos in ads.\\n- Search and import are rate limited per user (30 searches and 15 imports a minute) and searches are cached for 24h, so a burst of searches while planning a montage is fine. Bulk downloading a stock library is prohibited by both providers. A note saying Pixabay was skipped means its provider budget for that minute is spent; the Pexels results still stand.\\n\\n## Downloading from YouTube, Instagram, TikTok and other sites\\n\\nVideoDraft ships no downloader. When the user asks to pull a video from YouTube, Instagram, TikTok, X, LinkedIn or anywhere else, `yt-dlp` on the user\'s own machine does the work. Check for it with `command -v yt-dlp` before promising anything.\\n\\n**Present:** pin the format. On its defaults yt-dlp takes the best stream, usually VP9 or AV1 in WebM, which the native editor\'s import rejects (mp4, mov and m4v only). This returns one pre-muxed H.264 + AAC MP4 and needs no ffmpeg:\\n\\n```bash\\nyt-dlp -f \\"b[ext=mp4]\\" -o \\"media/%(title)s.%(ext)s\\" \\"<url>\\"\\n```\\n\\n**Missing:** say so in one line, then offer `brew install yt-dlp` only where `brew` exists, and only after asking. With no Homebrew, say the tool is unavailable and stop.\\n\\n- Never install Homebrew, never `sudo`, never write to `/usr/local/bin` or `/opt`, never pipe a script into a shell.\\n- `ffmpeg` is optional, matters only above 720p, and the selector above never needs it.\\n- Auth-gated content needs `--cookies-from-browser`; raise it only after a download fails on auth.\\n- Fetch only what was asked for, never a channel, playlist or back catalogue.\\n\\n## Local files and reference images\\n\\nReference inputs must be public URLs. The CLI uploads local files automatically wherever a URL is expected (`--ref photo.jpg`, `--start-image frame.png`), or explicitly:\\n\\n```bash\\nURL=$(videodraft upload ./product.png --json | jq -r .url)\\n```\\n\\nNever silently drop a reference you couldn\'t upload \u2014 stop and tell the user. Never upload a user\'s file to a third-party host.\\n\\nWhen the user attaches media for a native production, import actual footage into the editor by default. For hosted generation/storyboarding, classify each item before acting: a recurring **visual asset** (character/product/location/style), actual **footage to place as shots**, or **inspiration only**. See [references/pipeline.md](references/pipeline.md) for the hosted role mapping.\\n\\n## Showing media to the user\\n\\nGenerated media is **not** displayed in the chat automatically \u2014 you decide what to show. To preview an asset inline, save it locally (use `--download` so it lands under `media/`) and reference its **local path** as a Markdown link with a **leading `./`**:\\n\\n```\\n[ferrari shot](./media/ferrari_01.png) \u2190 image card\\n[the clip](./media/clip.mp4) \u2190 video player\\n[voiceover](./media/vo.mp3) \u2190 audio player\\n```\\n\\nPut the Markdown link **in your message text** \u2014 video and audio embed exactly like images. Do **not** use `SendUserFile` (or other file-send tools) to display media: that renders inside a collapsible tool card and gets buried in the tool list. The Markdown link in your prose is what produces the inline card.\\n\\nUse the path you saved to: a **workspace-relative** path (`./media/clip.mp4`, or `./<any-folder>/clip.mp4` \u2014 any folder in the workspace works), or the **absolute** path for a file outside the workspace (e.g. `/Users/you/Desktop/clip.mp4` or another workspace\'s path). Both render. Show the finished results worth showing (and only those \u2014 not every intermediate job). A bare CDN URL or a JSON dump of output URLs does **not** render; the local-path Markdown link is what produces an inline card.\\n\\n## Native-first VideoDraft ADE pipeline (idea \u2192 MP4)\\n\\nWhen `videodraft_editor` is present:\\n\\n1. Generate or source the script, storyboard, shot images, clips, voiceovers, music, and other assets through the cloud CLI/MCP as needed.\\n2. Call native `project_manage` (`project`, action `open` or `create`) to open or create the `.vdproject`.\\n3. Call native `library_manage` `import`, wait for imports to become ready, then assemble and refine the timeline with `edit_apply` recipes.\\n4. Call native `delivery_manage` `submit` and use its `jobs` operation for progress and results.\\n\\nDo not run the hosted production or export steps in this path unless the user explicitly asks for a web production.\\n\\n## Hosted fallback pipeline (idea \u2192 MP4)\\n\\nUse this only when there is NO native editor at all, or the user explicitly requests the hosted web\\nworkflow. The native surface is not only the injected `videodraft_editor` MCP: a `videodraft-editor`\\nexecutable on PATH is the same editor reached through its terminal bridge, and\\n[references/editor.md](references/editor.md) covers driving it that way. Treating a missing MCP as\\n\\"no editor\\" sends sessions that have the binary into hosted production for no reason.\\n\\n```bash\\nvideodraft create \\"<idea>\\" --ar 9:16 # project: script \u2192 visual assets \u2192 storyboard\\nvideodraft shots <project_id> --grid --estimate # cost preview, confirm with user\\nvideodraft shots <project_id> --grid # batch shot images (waits, writes onto shot cards)\\nvideodraft produce <project_id> # voiceovers + captions + production timeline\\nvideodraft export <project_id> --download final.mp4\\n```\\n\\nOptional between produce and export: per-shot motion clips (`videodraft generate video ... --project <id>` then place it with `videodraft attach <project> --scene N --shot M --media <url|file> --type video --duration <s>`), music (`videodraft generate music \\"...\\" --attach <project_id>`), and standalone audio assets (`generate audio`, `generate sound-effect`, `generate dialogue`, `generate voice-changer`, `generate dub`). Details, per-step tools and editing rules: [references/pipeline.md](references/pipeline.md).\\n\\n## Avatar and talking-head videos (both surfaces)\\n\\nAvatar generation is cloud-only \u2014 the native editor has no avatar or lipsync tools \u2014 so this applies whether or not `videodraft_editor` is present. Generate the avatar in the cloud; in VideoDraft ADE, import the rendered clip and cut it on the native timeline like any other footage.\\n\\nAvatar/talking-head videos use dedicated commands. For a reusable managed avatar, obtain or generate a clear portrait \u2192 `videodraft avatar script` when needed \u2192 `videodraft avatar create` \u2192 `videodraft avatar render --resolution 720p`. For a one-off portrait, use `videodraft avatar fabric <portrait> --text \\"...\\"` or `--audio <file>`, or `videodraft avatar h3-lipsync <portrait> --audio <file>` for a short high-resolution clip from existing audio. For an existing video plus replacement audio, use `videodraft avatar lipsync <video> --audio <file>`. Managed script/creation is bundled/free; direct Fabric, H3 Max Lip Sync, Sync, the managed Fabric render, and optional portrait generation/upscaling are paid. Confirm expensive steps first.\\n\\n## Working with hosted project data\\n\\nA hosted project is one JSON blob (script, storyboard scenes, shot cards, visual assets, production timeline). To inspect: `videodraft projects get <id>`. To edit: fetch `--raw`, modify, then `videodraft call update_project` \u2014 objects deep-merge, **arrays replace wholesale** (send the complete `storyboard.scenes` array to change one scene). Snapshot first with `videodraft checkpoint create <id>` before risky edits. Schema reference: `videodraft call get_project_schema`. This does not replace native editor tools when `videodraft_editor` is available for the production itself.\\n\\n## More\\n\\n- [references/pipeline.md](references/pipeline.md) \u2014 hosted fallback data model and production workflow\\n- [references/editor.md](references/editor.md) \u2014 native headless editor routing, project selection, import, timeline edits, verification, and export\\n- [references/models.md](references/models.md) \u2014 choosing image/video models, pricing patterns, voices and styles\\n- [references/examples.md](references/examples.md) \u2014 recipes: batch product videos from a CSV, talking-head from a script, changelog video in CI\\n","references/3d.md":"# Standalone 3D assets through MCP and CLI\\n\\nUse this lane for downloadable meshes, textured 3D models, and humanoid rigging. These are independent of AI Studio sessions and do not require a storyboard or web viewer. Never create a project just to generate a mesh. Native editor media import does not establish support for GLB/FBX assets; use external 3D software unless the live editor schema explicitly supports them.\\n\\n## Discover and quote\\n\\n```bash\\nvideodraft models 3d --json\\nvideodraft generate 3d \\"a worn brass telescope\\" --model meshy-7 --estimate\\nvideodraft generate 3d --model tripo-h3.1 --ref ./reference.png --options @options.json --estimate\\nvideodraft rig 3d --estimate\\n```\\n\\nMCP tools: `get_3d_models`, `estimate_3d`, `generate_3d`, `rig_3d`, `check_generation_status`, `list_3d_assets`, `get_3d_asset`, `create_3d_upload`, and `finalize_3d_upload`.\\n\\nThe generator catalog is deliberately limited to `meshy-7` and `tripo-h3.1`. Both expose text/image/multi-image modes; use the live schemas for option availability, view order, limits, topology, texture/PBR, and pose controls. Do not invent common provider option names. Server defaults prioritize quality. Rigging is a separate Meshy operation for compatible textured humanoid GLBs; it is not an arbitrary creature or facial rig service.\\n\\nEvery option is accessible through `--options` JSON (`@path.json` supported), plus repeatable `--option key=value` overrides. Values parse as JSON when valid. Options and explicit input mode affect pricing, so obtain a fresh quote for the exact recipe. Estimates perform no file upload or provider generation. VideoDraft bills whole credits at 100 credits per dollar using the server\'s rounding rules. BYOK costs zero VideoDraft credits and runs on the connected user Fal key. Do not infer the user\'s Fal-account charges from a zero VideoDraft-credit quote.\\n\\n## Generate and retrieve\\n\\n```bash\\nvideodraft generate 3d \\"a worn brass telescope\\" --model meshy-7 --download --json\\nvideodraft generate 3d --model meshy-7 --ref ./hero.png --no-wait --json\\nvideodraft generate 3d --model tripo-h3.1 --input-mode multi_image --ref ./front.png --ref ./left.png --ref ./back.png --ref ./right.png --no-wait --json\\nvideodraft status JOB_ID --json\\nvideodraft wait JOB_ID --download --json\\nvideodraft assets 3d list --status completed --json\\nvideodraft assets 3d get ASSET_ID --download ./media/3d --json\\nvideodraft rig 3d --asset ASSET_ID --download --json\\nvideodraft rig 3d ./textured-humanoid.glb --download --json\\n```\\n\\nLocal image references upload automatically. Local rigging inputs must be self-contained `.glb` files and upload via the dedicated 3D route. Input mode is inferred from reference count unless `--input-mode text|image|multi_image` is supplied. A text prompt requires no images; image mode requires exactly one and no geometry prompt. Meshy multi-image accepts 1-4 views, and Tripo accepts 2-4 in front, left, back, right order. Use Meshy\'s `texture_prompt` option when texturing guidance is needed with image inputs. Do not mix unrelated images as if they were views of the same object.\\n\\nGeneration and rigging wait by default. `--no-wait` returns a persisted job that continues server-side. For many jobs use one `wait ID1 ID2 ... --download`; do not run parallel polling processes. Recover timeout with the same job ID, not another paid generation.\\n\\nSubmission returns a `request_id` UUID. A same-request retry must use `--request-id UUID` and the original arguments. The CLI journals resolved uploads in its private config directory so retries reuse the same remote files. Changed inputs under that UUID are rejected. Submission errors include the UUID in JSON `details.request_id` where available. Never retry a failed request with a new UUID until the prior job status is known.\\n\\n## Complete artifact packages\\n\\n`--download` defaults to `media/3d/<asset_id>/`. A user directory is respected and receives a per-asset folder. `{job_id}` and `{asset_id}` directory templates are accepted. Do not pass a `.glb` filename or a media `{index}.{ext}` template: a 3D result can contain multiple formats and texture/material dependencies.\\n\\nDownloads preserve every returned artifact. The manifest records provider URLs, source filenames, local paths, role, content type, and actual downloaded sizes. glTF/OBJ/MTL references are rewritten to saved dependency paths; binary GLB/FBX and provider ZIPs are retained as returned. Supplied dependency aliases become local copies at the paths needed by original FBX files. Unsafe or colliding aliases produce warnings. Existing packages are never overwritten. A successful package has `manifest.json`; read its `complete` field and `warnings`, plus CLI `package_warnings`. Do not claim a package with unresolved dependencies is ready for offline use.\\n\\nMachine results use `type: \\"model3d\\"` with `output_files` for files and `output_media` only for ordinary preview media. Texture maps and models are not image/video gallery entries. Show a supplied preview via a local image Markdown link and deliver the original mesh/package files. Do not imply an interactive 3D viewer exists in AI Studio or VideoDraft ADE.\\n","references/editor.md":"# Native VideoDraft Editor reference\\n\\nUse this reference when the user wants to assemble, cut, caption, mix, lay out, inspect, or export a local VideoDraft Editor project. The native editor is deterministic and local. Cloud generation remains in the `videodraft` CLI or hosted MCP.\\n\\n## VideoDraft ADE preference rule\\n\\nWhen `videodraft_editor` tools are exposed, treat the native editor as available and make it the default surface for production, timeline assembly, and final export. It is headless by design, so a hidden window or an untouched Open Editor button does not justify using hosted production instead.\\n\\nUse cloud tools for asset generation and optional script/storyboard work, then import the results. Do not call hosted `produce_project` / `videodraft produce` or `export_video` / `videodraft export` unless the user explicitly requests an editable web production or the native editor tools are unavailable. If a native tool call fails after the editor was available, report or recover that native failure rather than silently switching surfaces.\\n\\n## Choose the correct surface\\n\\n- `videodraft` and the hosted VideoDraft MCP generate assets and can manage hosted web projects. They use the user\'s VideoDraft account and credits. In VideoDraft ADE, use them mainly as the source of generated media and optional storyboards for the native production.\\n- `videodraft_editor` edits local `.vdproject` packages. It has no generation, account, model, or credit tools.\\n- Inside VideoDraft ADE on a supported Mac, the editor MCP is injected automatically for Claude, Codex, OpenCode and Grok in both Code and VideoDraft modes. It starts headlessly before the chat opens. The user does not need to click Open Editor, and closing or hiding the editor window does not stop headless editing.\\n- Outside that environment, use the editor only if `videodraft_editor` MCP tools are already exposed or the `videodraft-editor` executable is on PATH. Do not confuse the public `videodraft` cloud CLI with the separate native editor executable.\\n\\nPrefer the direct MCP tools when they are available. The terminal bridge is useful for scripts, diagnostics, or an agent session where the MCP was not injected.\\n\\n## Which editor tools you have\\n\\nCurrent editors expose eight workflow tools: `project_manage`, `edit_snapshot`, `edit_apply`, `edit_undo`, `library_manage`, `inspect`, `speech_apply`, and `delivery_manage`. Use them whenever `edit_apply` is available. Everything below describes them.\\n\\nEarlier editors expose a different set of tools. If `edit_apply` is not in your catalog, follow the live tool descriptions; the working rules below still apply.\\n\\nThe live tool descriptions and schemas are authoritative. When they differ from this reference, follow them.\\n\\nService tools (`project_manage`, `library_manage`, `inspect`, `speech_apply`, `delivery_manage`) take `operation` and `parameters`, plus an optional `context` (see [Keep a reliable editing model](#keep-a-reliable-editing-model)):\\n\\n```json\\n{\\"operation\\": \\"import\\", \\"parameters\\": {\\"source\\": {\\"path\\": \\"/Users/me/clips\\"}}}\\n```\\n\\n### Parameters at a glance\\n\\nEach operation\'s schema has the full list. These are the fields that matter most, and the ones most often confused:\\n\\n| Tool and operation | Key `parameters` |\\n| --- | --- |\\n| `project_manage` `project` | `action` (`list`, `open`, `create`, `close`); `id`, `name` or `path` to open; `name`, `fps`, `aspectRatio`, `quality` to create |\\n| `edit_snapshot` (no `parameters`) | `includeLibrary`, a `startFrame`/`endFrame` window, `offset`/`limit` paging, `context` for later pages |\\n| `edit_apply` (no `parameters`) | `actions`; optional `context`, `requestId`, `previewOnly` |\\n| `library_manage` `import` | `source` with one of `path`, `url`, `bytes` + `mimeType`, or `matte`; optional `name`, `folder`. A `.srt` or `.vtt` file becomes captions |\\n| `library_manage` `list` | `ids` to poll imports, `pending`, `folder` |\\n| `inspect` `timeline` | a `startFrame`/`endFrame` window; there is no clip filter |\\n| `inspect` `frame` | `startFrame` for one frame; add `endFrame` and `maxFrames` to sample a range |\\n| `inspect` `media` | `mediaRef` (a library asset, not a file path), `startSeconds`/`endSeconds` in source seconds, `wordTimestamps`, `overview` |\\n| `inspect` `transcript` | a `startFrame`/`endFrame` window, `granularity` (`words` or `segments`), `clipId` |\\n| `inspect` `color` | `clipId` with `atFrame`, or `mediaRef`; optional `reference` |\\n| `speech_apply` `words` | `words`: transcript word indices, each one index or a `[first, last]` pair; or `matches`: exact words to remove everywhere; `pacing` |\\n| `speech_apply` `silence` | none |\\n| `speech_apply` `captions` | caption `style`, `position` and look fields; it finds the speech itself. `style` `plain` (sentences without a preset) takes `positionY` or `transform.centerY`, not `position` or `punctuation` |\\n| `delivery_manage` `submit` | `mode`, `codec`, `resolution`, `outputPath` (its folder must already exist); `captionGroupId` and `wordTiming` for `srt` and `vtt`; `check: false` skips a video\'s export check |\\n| `delivery_manage` `jobs` | `action` `list` (every job with its progress, warnings, output path and check findings; with a `jobId`, that job with all its findings), `cancel` with a `jobId`, or `dismiss` with a `jobId` and optional `findingIds` |\\n\\n`previewOnly` exists only on `edit_apply`. Service operations apply immediately.\\n\\n### `edit_apply` actions at a glance\\n\\nEvery action puts its fields beside `kind`, for example `{\\"kind\\": \\"adjust\\", \\"clipId\\": \\"c1\\", \\"speed\\": 2}`. `remove`, `adjust`, `replace`, `transition`, `mask`, `text`, `grade` and `effects` take `clipId` for one clip or `clipIds` for several; `animate`, `extract`, and the rows of `move` and `split` take a single `clipId`. Advanced actions cannot take `as`, but their clip ids accept `@name`. The one exception is `transition`, whose own `kind` field names the style, so its fields go inside `parameters`: `{\\"kind\\": \\"transition\\", \\"parameters\\": {\\"clipId\\": \\"c2\\", \\"kind\\": \\"dipToBlack\\"}}`. These `actions` place a clip, warm it and add a vignette:\\n\\n```json\\n[{\\"kind\\": \\"place\\", \\"assetId\\": \\"\u2026\\", \\"atFrame\\": 0, \\"as\\": \\"shot\\"},\\n {\\"kind\\": \\"grade\\", \\"clipId\\": \\"@shot\\", \\"adjustments\\": {\\"temperature\\": 7500}},\\n {\\"kind\\": \\"effects\\", \\"clipId\\": \\"@shot\\", \\"effects\\": [{\\"type\\": \\"finish.vignette\\", \\"params\\": {\\"strength\\": 35}}]}]\\n```\\n\\n| Advanced `kind` | Key fields |\\n| --- | --- |\\n| `place_batch` | `entries: [{mediaRef, startFrame, endFrame or source, trackIndex}]`, not the concise `assetId`, `atFrame`, `durationFrames`; leave `trackIndex` off every entry for a new track |\\n| `insert_batch` | `trackIndex`, `atFrame`, `entries: [{mediaRef, durationFrames or source}]`; later clips move right |\\n| `move` | `moves: [{clipId, toFrame, toTrack}]` |\\n| `remove` | `clipId` or `clipIds`; leaves a gap |\\n| `split` | `splits: [{clipId, atFrame}]`, or `trackIndex` with `frames` |\\n| `extract` | `trackIndex` with `ranges: [[start, end]]` in frames, or `clipId` with `ranges` and `units` (`frames`, or source `seconds`); closes the gap |\\n| `adjust` | `clipId` or `clipIds` plus any of `durationFrames`, `trimStartFrame`, `trimEndFrame`, `speed`, `speedCurve` (`{preset}`, or `{points: [{t, rate}]}` with `t` from 0 to 1 along the source), `preservesPitch`, `volume`, `opacity`, `transform` (`centerX`, `centerY`, `width`, `height`, `flipHorizontal`, `flipVertical`), `blendMode` |\\n| `animate` | `clipId`, `property` (`volume`, `opacity`, `rotation`, `position`, `scale`, `crop`), `keyframes: [[frame, ...values]]` with frames counted from the clip\'s start; `position` is the top-left corner |\\n| `arrange` | `layout`, `slots: [{slot, clipIds or mediaRef, anchor}]`, `fit` (`fill` or `fit`); `mediaRef` slots also need `endFrame` |\\n| `transition` (fields inside `parameters`) | `clipId` or `clipIds` (the clip after each cut), `kind` (`crossDissolve`, `dipToBlack`, `dipToWhite`, `blurDissolve`, `push`, `linearWipe`, `whipPan`, `crossZoom`, or `none` to remove), `durationFrames`, `params` (`direction` 0 to 3 for left, right, up, down; `feather`; `blur`; `intensity`) |\\n| `mask` | `clipId` or `clipIds`, `shape` (`rectangle`, `ellipse`, `none`), `centerX`, `centerY`, `width`, `height` (0 to 1 of the clip\'s own frame), `feather`, `strength`, `inverted` |\\n| `replace` | `clipId` or `clipIds`, `mediaRef` (a library asset of the same kind), `trim` (`keep`, the default, or `reset`), `linkedAudio` (`follow`, the default, or `keep`) |\\n| `tracks` | `reorder: [{trackId, to}]`, `set: [{trackId, muted, hidden, syncLocked}]`, `remove: [{trackId}]`; there is no add, since placing without a track makes one |\\n| `titles` | `entries: [{startFrame, endFrame, content}]`, optionally with `trackIndex` (on every entry or none), `style`, `animation`, `transform` and typography (`fontName`, `fontSize`, `color`) |\\n| `text` | `clipId`, `clipIds` or `captionGroupId`, with `content`, `style`, `position`, `animation` or typography |\\n| `grade` | `clipId` or `clipIds`, `adjustments`, `wheels`, `curves`, `hueCurves`, `lut`, `reset` |\\n| `effects` | `clipId` or `clipIds`, `effects: [{type, params, enabled}]`, `remove: [type]` |\\n\\n- `grade`: `adjustments` holds `exposure` (EV, -4 to 4), `temperature` (kelvin, 1800 to 15000, 6500 neutral, higher is warmer), `tint` (-150 magenta to 150 green); `contrast`, `highlights`, `shadows`, `whites`, `blacks`, `vibrance` and `saturation` are signed percents (-100 to 100, 0 neutral), not factors like 1.2. `wheels`: `shadows`, `midtones`, `highlights`, each `{hue, strength, brightness}`. `curves`: `luma`, `red`, `green`, `blue` as `[x, y]` points from 0 to 1. `hueCurves`: `targets: [{hue, rotate, saturation, lightness}]`. `lut`: `{path, mix}`, with the `storedPath` from `prepare_look`. Out-of-range values are refused.\\n- Effect `type` ids and their knobs, which go inside `params` and run 0 to 100 unless noted. Defaults leave the picture unchanged, so send the knob you want to see:\\n - `finish.vignette`: `strength` and `curvature` (-100 to 100; positive `strength` darkens the edges), `size`, `falloff`\\n - `finish.grain`: `strength`, `grainSize` (0.5 to 6 px); `finish.glow`: `strength`, `haloRadius` (px), `cutoff`, `halation`\\n - `defocus.gaussian`: `blurRadius` (px); `defocus.motion`: `streakLength` (px), `streakAngle` (-180 to 180 degrees)\\n - `texture.clarity`: `localContrast`, `hazeRemoval` (-100 to 100); `texture.sharpen`: `strength` (0 to 200); `texture.denoise`: `strength`\\n - `matte.chroma`: `screenHue` (0 to 360 degrees, 120 is green), `range`, `edgeSoftness`, `spillSuppression`\\n- `arrange` layouts and their slots (fill every slot): `fullscreen` (`stage`); `split` (`left`, `right`); `stack` (`top`, `bottom`); `corner_top_left`, `corner_top_right`, `corner_bottom_left`, `corner_bottom_right` (`stage`, `corner`); `quad` (`top_left`, `top_right`, `bottom_left`, `bottom_right`); `side_panel` (`stage`, `panel`); `thirds` (`left`, `middle`, `right`). The layout picks the corner; `anchor` only biases the crop.\\n- `trackIndex` and `toTrack` are an existing track\'s `order` from `edit_snapshot` (0 draws on top), counted at that point in the recipe: a track an earlier action adds shifts them. `tracks` takes `trackId` handles instead.\\n\\n## Start with the intended project\\n\\nAn MCP session can begin without a project selected. Project selection belongs to the session, not to whichever editor window happens to be frontmost.\\n\\n1. If the user named an existing project but its identity is unclear, call `project_manage` with `operation: \\"project\\"` and `parameters.action: \\"list\\"`.\\n2. Open the exact project with `action: \\"open\\"` and the returned `id`, an unambiguous `name`, or the `.vdproject` `path`.\\n3. Create only when the user wants a new local edit. `action: \\"create\\"` accepts optional `name`, `fps`, `aspectRatio`, and `quality`.\\n4. Treat `isActive` as this session\'s target and `isVisible` as the project shown in the UI. Headless editing only needs the session target.\\n5. Use `action: \\"close\\"` only when closing is part of the task. It saves first and never deletes the project. Afterwards other calls answer `no_project` until you open or create a project again.\\n\\nOther `project_manage` operations: `create_timeline` and `select_timeline` for additional timelines, and `configure` for project settings (a frame-rate change applies to every timeline).\\n\\nDo not substitute a hosted project ID for a native project. A hosted project can supply scripts, storyboards, and generated media, but the native edit is a separate `.vdproject` package.\\n\\n## Keep a reliable editing model\\n\\n- Opening or creating a project, and creating or selecting a timeline, returns `data.snapshot` (the `edit_snapshot` view with the library: format, tracks, clips and assets) and a `context`, so you can edit right away. Call `edit_snapshot` only after an out-of-band user edit, to page a long timeline with `offset` and `limit`, or for handles you lack.\\n- `context` (`epoch`, `timelineId`, `revision`) is optional on every write except `edit_undo` and `project_manage` `project`, which take none. Without it, a write applies to the current state. With it, the editor refuses the write if the project changed since that context, which is what you want when a person may be editing at the same time. Every reply returns a fresh `context`. After `context_expired` (a reopen or restart), take a new snapshot.\\n- Timeline positions and durations are frames at `format.fps`. `sourceSeconds` values are seconds in the source file. Pass values as returned; do not multiply or divide by fps yourself.\\n- IDs are short handles. Pass them back exactly as returned; do not derive them from UUIDs. Tracks keep stable handles; indexes can change.\\n- When the user says \\"this clip\\", \\"these captions\\", or \\"here\\", take a fresh `edit_snapshot` (a one-frame window is enough). Its `selection` names what they selected in the project\'s window, in the snapshot\'s handles: `clipIds`, `captionGroupIds`, a `gap`, the marked `range`, and library `mediaIds`; no `selection` means nothing is selected. `currentFrame` is the playhead. `visible: false` means the user is not looking at this project.\\n- Send writes serially. Parallel writes against one project race each other\'s context.\\n- A refused call answers `status: \\"rejected\\"`: nothing changed, so fix what the message names and send it again. A service write that answers `failed` may have applied before a save or connection failed; inspect before repeating it.\\n- Use `inspect` for detail and verification: `timeline` for exact clip and track properties, `frame` for rendered frames of the composited result, `media` before describing source content, `color` for scopes, and `transcript` to locate spoken words.\\n- Volume values, including volume keyframes, are linear from `0` to `1`.\\n\\n## Edit with recipes\\n\\n`edit_apply` takes an ordered list of `actions`, plus an optional `context` and `requestId`. The editor validates the whole recipe before changing anything, then applies it as one undo step. If any action fails, nothing changes and the error names the action. An applied reply lists the resulting `clips` (position, length, source window, text); use it to confirm the edit instead of re-reading.\\n\\n- Concise actions: `place` (an `assetId` at `atFrame`, optionally on a `trackId`, with `durationFrames` or `sourceSeconds`, `mode` `overwrite` or `insert`), `trim` (a `clipId` to a `sourceSeconds` window), and `title` (`text` at `atFrame` for `durationFrames`, optional `look` and `motion`). Name an action with `as` and address its clip in later actions as `@name`.\\n- Advanced actions put their fields beside `kind` too: `place_batch`, `insert_batch`, `move`, `remove`, `split`, `extract`, `adjust`, `animate`, `arrange`, `transition` (fields inside `parameters`), `mask`, `replace`, `tracks`, `titles`, `text`, `grade`, and `effects`. Their field help lists every supported effect, grading control, mask, layout, speed curve, and caption control.\\n- Use `arrange` for split screens, picture-in-picture, grids, and canvas placement, and `tracks` to fix stacking. Do not synthesize layouts from generic transforms or keyframes.\\n- Use `replace` to swap a clip\'s media and keep its timing, effects, keyframes, and links, for example a re-render of the same shot. `trim: \\"keep\\"` holds the clip\'s source offsets; `trim: \\"reset\\"` plays the new media from its first frame and keeps linked audio in sync.\\n- Put every edit a request needs into one recipe: placing, trimming, titling, grading and effects together are one call and one undo step. Use `previewOnly: true` to validate a large or risky recipe without editing.\\n- Send a `requestId` when a retry must not edit twice: repeating the identical request with the same `requestId` returns the original receipt, even after a dropped connection. Without one, a repeated request applies again. Use a new `requestId` for a different edit.\\n- `applied_unsaved` means the edit applied but saving failed. If you sent your own `requestId`, repeat the identical request to retry the save, not the edit. Without one, do not resend: a repeat would apply the edit again.\\n- `edit_undo` reverses this connection\'s latest edit while it is still the top undo step. It never undoes the user\'s own edits. Take a new snapshot afterward.\\n\\nEdits are undoable. Do not ask for confirmation before each ordinary edit. Ask one focused question only when the user\'s creative direction is materially ambiguous.\\n\\n## Bring generated or local media into the editor\\n\\nFor b-roll and establishing shots, search free stock first (`videodraft stock search \\"<query>\\"`), import the pick with `videodraft stock import <ref>`, and feed the returned CDN URL to `library_manage` `import` as `source.url`. It costs no credits.\\n\\nOtherwise use cloud generation for new assets, save or download the outputs, then call `library_manage` with `operation: \\"import\\"`:\\n\\n- `source.path`: absolute local file or directory. A directory imports recursively and preserves its folder structure.\\n- `source.url`: HTTPS asset URL. Set `mimeType` when a signed URL has no usable extension.\\n- `source.bytes`: small base64 media with a required `mimeType`.\\n- `source.matte`: generated solid-color image.\\n\\nReadiness differs by source, and so does the poll that detects it:\\n\\n- **URL and single-file path** imports return `status: \\"downloading\\"` with one `mediaRef`. Poll `library_manage` `list` with `parameters.ids: [mediaRef]` until `generationStatus` is absent.\\n- **Directory** imports also return `status: \\"downloading\\"` with one placeholder `mediaRef` for the batch. Poll it the same way; the folder\'s assets appear when `generationStatus` clears. (`pending: true` remains a fallback that lists every unresolved import.)\\n- **Inline bytes and matte** imports finish inline and come back `status: \\"ready\\"`; no polling is needed.\\n\\nAn import\'s `mediaRef` is the `assetId` that recipe actions take. Never place a pending asset on the timeline. `generationStatus` is the signal: `preparing`, `generating`, `downloading`, and `rendering` mean keep polling, absent means usable, and **`failed` is terminal**. Report it or retry the import explicitly; never poll on. Do not treat \\"not downloading\\" as ready.\\n\\nA `.srt` or `.vtt` caption file (by `path`, `url`, or `bytes` with `mimeType` `application/x-subrip`, `text/srt`, or `text/vtt`) is not added to the library. Its captions go onto a new caption track at the times the file gives, as one undo step, and the reply names the new `captionGroupId` to restyle with a `text` action.\\n\\nFor a batch of local outputs, download them into one workspace directory and import that directory once when practical. This is safer and faster than racing many import calls.\\n\\nTo apply a LUT, first store it with `library_manage` `prepare_look` (`parameters.path` to a `.cube` file), then use the returned `storedPath` in a `grade` action.\\n\\n## Speech edits\\n\\n`speech_apply` runs speech work as separate operations, not recipe actions: `words` removes transcript words, `silence` trims dead air, and `captions` generates styled captions. Read a fresh transcript with `inspect` `transcript` after any speech edit, because word positions change.\\n\\n## Service operations and timeouts\\n\\n`project_manage`, `library_manage`, `speech_apply`, and `delivery_manage` writes have no retry receipts. After a timeout, inspect the outcome (a snapshot, a library `list`, or delivery `jobs`) before repeating a write. A failed service write may have applied an edit before saving failed, so read its `data` and the current state.\\n\\n## Edit and verify\\n\\nA dependable sequence:\\n\\n1. `project_manage` to select or create the local project; its reply carries the snapshot.\\n2. `inspect` `media` when content selection matters.\\n3. One `edit_apply` recipe with every clip, track, layout, text, audio, color, and effect change the request needs; `speech_apply` for word cuts, silence, and captions.\\n4. Check the reply\'s `clips`; use `inspect` `frame` only when visual composition or layer order matters.\\n5. `edit_undo` if the result is wrong and the next recipe would not cleanly correct it.\\n\\n## Export\\n\\nCall `delivery_manage` with `operation: \\"submit\\"`. Submission is not completion: it returns a job, destination, and `started` or `queued` status.\\n\\n- Use `mode: \\"video\\"` for H.264, H.265, or ProRes.\\n- Use `mode: \\"xml\\"` (XMEML) for Premiere Pro **and DaVinci Resolve**. Resolve reads XMEML natively. Use `fcpxml` only for Final Cut Pro; sending Resolve an FCPXML produces a package it cannot open cleanly.\\n- Use `mode: \\"videodraft\\"` for a self-contained project package.\\n- Use `mode: \\"srt\\"` or `mode: \\"vtt\\"` for a subtitle file of the captions: words and timing only, without styling. `captionGroupId` picks a caption group; `wordTiming: true` adds each word\'s start to a `vtt`.\\n- Omit `outputPath` unless the user named a destination; the default is `~/Downloads`. The editor does not create folders, so a named destination\'s folder must already exist.\\n- Use `operation: \\"jobs\\"` with `parameters.action` `list` to follow progress, warnings, and results. Cancel only when the user asks or the just-queued settings were wrong. Do not infer that an export is stuck from elapsed time alone.\\n- A finished video export is then checked in the background for black picture, gaps, sound that drops out or clips, and missing media. `list` shows each job\'s `findingCount` and its first three findings; `list` with that `jobId` returns every finding. Findings are warnings on a completed export: tell the user what was found and where, rather than treating the export as failed.\\n\\n## Terminal bridge\\n\\nVideoDraft desktop terminals expose `videodraft-editor`, which controls the same process and tools:\\n\\n```bash\\nvideodraft-editor status\\nvideodraft-editor list-tools\\nvideodraft-editor tool project_manage --json \'{\\"operation\\":\\"project\\",\\"parameters\\":{\\"action\\":\\"list\\"}}\'\\nvideodraft-editor tool edit_snapshot --project \\"/path/to/My Video.vdproject\\" --json \'{\\"includeLibrary\\":true}\'\\nvideodraft-editor show\\nvideodraft-editor hide\\n```\\n\\nEach `videodraft-editor` invocation is a separate connection, so a project opened by one call is NOT remembered by the next. Pass `--project <path>` on every `tool` call that operates on a project; without it, a follow-up call answers `no_project` (no project is open for this session) even though the open succeeded. For the same reason, `edit_undo` has nothing to undo from the terminal: it only reverses an edit made on its own connection.\\n\\nControl and tool commands auto-start a headless editor if none is running. `show` only reveals the already-running UI. Use `videodraft-editor tool <name> --json -` to read a JSON object from stdin when shell quoting would be fragile. Never read, copy, or expose the editor\'s rotating local authentication secret.\\n","references/examples.md":"# Recipes\\n\\nWorking patterns for common asks. All assume auth (`videodraft login` once, or `VIDEODRAFT_API_KEY` in the environment) and use `--json` for parsing.\\n\\nInside VideoDraft ADE, if a native editor is available, use these recipes for asset generation and\\noptional script/storyboard stages. Hand the results to the editor for production and export ONLY\\nwhen the deliverable the user asked for is a composed production. A standalone output \u2014 the batch\\nproduct clips in recipe 1, the upscale in recipe 7 \u2014 is finished when it is generated; importing it\\ninto a project and exporting a timeline builds an edit nobody asked for. Recipes below that call `videodraft produce` or `videodraft export` are hosted fallbacks only. Do not choose them over the available native editor unless the user explicitly asks for the hosted web workflow.\\n\\n## 1. Batch product videos from a CSV\\n\\nOne 9:16 product clip per row of `products.csv` (`name,image_url,tagline`):\\n\\n```bash\\n#!/usr/bin/env bash\\nset -euo pipefail\\nmkdir -p outputs\\n\\nwhile IFS=, read -r name image tagline; do\\n job=$(videodraft generate video \\\\\\n \\"Premium product shot of ${name}: ${tagline}. Slow orbit, studio lighting.\\" \\\\\\n --model gemini-omni-1.1-flash --ar 9:16 --duration 6 \\\\\\n --start-image \\"$image\\" \\\\\\n --no-wait --json | jq -r .job_id)\\n echo \\"$name,$job\\" >> outputs/jobs.csv\\ndone < <(tail -n +2 products.csv)\\n\\n# Collect ALL results with ONE process (batched polling \u2014 one request per tick)\\nvideodraft wait $(cut -d, -f2 outputs/jobs.csv) \\\\\\n --download \\"outputs/{job_id}_{index}.{ext}\\" --json > outputs/results.json\\n# map job ids back to product names via outputs/jobs.csv\\n```\\n\\nSubmit-then-collect parallelizes server-side generation; the single multi-id `wait` keeps it to one local process and one batched poll request per tick no matter how many jobs. Gemini Omni 1.1 Flash is selected because these are six-second first-frame product clips. Estimate first: `videodraft costs gemini-omni-1.1-flash --type video --duration 6 --resolution 720p --audio` \xD7 rows, and confirm with the user.\\n\\nExtend an uploaded clip or conversationally edit an official Google interaction with Gemini Omni 1.1 Flash:\\n\\nEach extension appends 3-10 seconds at the end. Edit and extend accept EXACTLY ONE input video, so `--ref-video` cannot accompany `--source-video` or `--previous-interaction-id`; only `--ref` images can. New dialogue is allowed only when the source video is silent, so a chain of spoken beats must carry its speech in the first turn. A continuation is submitted as the prior turn\'s output, so 40 seconds total is reachable only by extending from a source of 30s or less, and a ladder dead-ends past 30s.\\n\\n```bash\\n# Uploaded-video extension. Reference IMAGES are fine here; reference videos are not.\\nvideodraft generate video --model gemini-omni-1.1-flash \\\\\\n --source-video ./ending.mp4 --ref ./wardrobe.png \\\\\\n --extend --duration 6 --resolution 1080p \\\\\\n --download ./media/extended.mp4\\n\\n# Conversational edit from the interaction_id of an earlier generation. Add --extend to lengthen instead.\\nvideodraft generate video --model gemini-omni-1.1-flash \\\\\\n --previous-interaction-id \\"$INTERACTION_ID\\" \\\\\\n --resolution 720p \\\\\\n --download ./media/continued.mp4\\n```\\n\\nFal BYOK supports the currently callable Gemini Omni 1.1 generation and basic source-edit endpoints at zero VideoDraft credits. Its published callable v1.1 endpoints do not expose continuation, extension, or separate creative references on a source edit. Do not infer a Fal route or retry on paid Google while Fal BYOK is active.\\n\\n## 2. Hosted full marketing video from one idea (fallback)\\n\\n```bash\\nvideodraft create \\"30-second launch video for Solace, a sleep-tracking ring. Calm, premium, dark palette.\\" \\\\\\n --ar 9:16 --style cinematic --json > project.json\\nPROJECT=$(jq -r .project_id project.json)\\n\\nvideodraft shots \\"$PROJECT\\" --grid --estimate # show the user the cost; get a go-ahead\\nvideodraft shots \\"$PROJECT\\" --grid\\nvideodraft produce \\"$PROJECT\\"\\nvideodraft generate music \\"minimal ambient, warm pads, 60 BPM\\" --attach \\"$PROJECT\\"\\nvideodraft generate audio \\"Extend @Audio1 into a 20-second transition\\" --ref-audio ./intro.wav --format wav --download ./transition.wav\\nvideodraft export \\"$PROJECT\\" --download solace-launch.mp4\\n```\\n\\nUse this complete hosted path only when the user requested a web project or the native editor is unavailable. Otherwise stop after the storyboard/assets, import them into the native `.vdproject`, and export with `delivery_manage`. The hosted project stays editable at the URL in `project.json` (`.urls`).\\n\\n## 3. Talking-head (avatar) video\\n\\nWhen the user has no portrait, generate a clear front-facing avatar image first. Skip this step when they supplied one or an existing character should be reused.\\n\\n```bash\\nvideodraft generate image \\\\\\n \\"Front-facing head-and-shoulders portrait of a friendly coffee expert, direct eye contact, natural expression, clean studio background\\" \\\\\\n --model nano-banana-2 --ar 9:16 --download ./media/avatar.png\\n\\nSCRIPT=$(videodraft avatar script \\"why our espresso subscription saves you money\\" --style ad-style --json | jq -r .script)\\nAVATAR=$(videodraft avatar create ./media/avatar.png --script \\"$SCRIPT\\" --voice elevenlabs-kPzsL2i3teMYv0FxEYQ6 --ar 9:16 --json | jq -r .avatar_video_id)\\nvideodraft avatar render \\"$AVATAR\\" --resolution 720p # VEED Fabric paid step; confirm cost first (~20 credits/sec)\\n```\\n\\n`avatar script` and `avatar create` (including speech) are bundled/free. In this example only the optional portrait generation and Fabric render spend credits.\\n\\nIf the portrait is low resolution, enhance it before `avatar create`:\\n\\n```bash\\nvideodraft upscale image ./founder-small.jpg --scale 2x --download ./media/founder-upscaled.png\\n```\\n\\nFor a one-off portrait animation without creating a managed avatar record:\\n\\n```bash\\nvideodraft avatar fabric ./founder.jpg \\\\\\n --text \\"Welcome to the weekly product update.\\" \\\\\\n --voice-description \\"warm, confident American presenter\\" \\\\\\n --resolution 720p --download ./media/presenter.mp4\\n```\\n\\nFor a short, sharp clip from speech or a song the user already has (5-14.8s of audio is used):\\n\\n```bash\\nvideodraft avatar h3-lipsync ./singer.png --audio ./chorus.mp3 \\\\\\n --resolution 1080P --audio-duration 12 --estimate\\nvideodraft avatar h3-lipsync ./singer.png --audio ./chorus.mp3 \\\\\\n --resolution 1080P --download ./media/singer-lipsync.mp4\\n```\\n\\nWhen the user already has both the video and replacement speech:\\n\\n```bash\\nvideodraft avatar lipsync ./presenter.mp4 \\\\\\n --audio ./localized-voiceover.mp3 \\\\\\n --sync-mode loop --download ./media/presenter-localized.mp4\\n```\\n\\nEdit an existing video with a dedicated edit model:\\n\\n```bash\\nvideodraft models video --category video_edit\\nvideodraft edit video ./product-demo.mp4 \\\\\\n \\"Preserve the product but follow the reference camera rhythm\\" \\\\\\n --model gemini-omni-1.1-flash --ref-video ./camera-rhythm.mp4 \\\\\\n --ref-video-duration 2.5 --resolution 1080p \\\\\\n --download ./media/product-demo-reframed.mp4\\n\\nvideodraft edit video ./product-demo.mp4 \\\\\\n \\"Turn the room into a warm evening scene while preserving the product and camera motion\\" \\\\\\n --model kling-o3-video-ref-edit --ref ./evening-style.jpg \\\\\\n --preserve-audio --download ./media/product-demo-evening.mp4\\n```\\n\\nTransfer motion from a reference clip onto a character image:\\n\\n```bash\\nvideodraft edit motion ./character.png \\\\\\n \\"Apply the dancer\'s movement to this character while preserving identity\\" \\\\\\n --motion-video ./dance-reference.mp4 \\\\\\n --model kling-v3-motion-control --quality pro \\\\\\n --download ./media/character-dance.mp4\\n```\\n\\n## 4. Hosted changelog video in CI\\n\\nIn a GitHub Action with `VIDEODRAFT_API_KEY` set as a secret:\\n\\n```bash\\nNOTES=$(git log --oneline v1.2.0..HEAD | head -20)\\nvideodraft create \\"Weekly product update video. Energetic, 20 seconds. Changes: ${NOTES}\\" --ar 16:9 --json > p.json\\nPROJECT=$(jq -r .project_id p.json)\\nvideodraft shots \\"$PROJECT\\" && videodraft produce \\"$PROJECT\\"\\nvideodraft export \\"$PROJECT\\" --download changelog.mp4 --wait-timeout 30m\\n```\\n\\n## 5. Variations and picking a winner\\n\\n```bash\\nvideodraft generate image \\"logo concept: minimalist fox, geometric\\" --num 4 --download \\"./concepts/{job_id}_{index}.{ext}\\" --json\\n# Show all 4 to the user; regenerate the chosen one at higher res:\\nvideodraft generate image \\"<same prompt>\\" --model nano-banana-pro --resolution 4K\\n```\\n\\n## 6. Reaching tools without a curated command\\n\\n```bash\\nvideodraft tools list --json | jq -r \'.[].name\'\\nvideodraft tools schema attach_media_to_shot --json\\nvideodraft call attach_media_to_shot --args \'{\\"project_id\\":\\"...\\",\\"scene_index\\":0,\\"shot_index\\":1,\\"media_url\\":\\"https://...\\",\\"media_type\\":\\"video\\",\\"duration_seconds\\":6}\'\\n```\\n\\nAnything the hosted VideoDraft MCP exposes, including character studio, product studio, and hosted project data, is reachable this way even before it gets a curated command. Native `.vdproject` editing uses the separate `videodraft_editor` MCP described in SKILL.md.\\n\\n## 7. Enhance an existing asset without changing it\\n\\n```bash\\n# Light image cleanup, no enlargement\\nvideodraft upscale image ./poster.png --scale 1x --download ./media/poster-enhanced.png\\n\\n# General image and video enlargement\\nvideodraft upscale image ./frame.png --scale 2x --download ./media/frame-2x.png\\nvideodraft upscale video ./clip.mp4 --resolution 1080p --download ./media/clip-1080p.mp4\\n\\n# Rebuild detail in a tiny/blurry source (generative), or reimagine it (creative)\\nvideodraft upscale image ./tiny.jpg --scale 4x --mode generative --model \\"Wonder 3.5\\" --download ./media/tiny-4x.png\\nvideodraft upscale video ./ai-clip.mp4 --resolution 4k --mode generative --download ./media/ai-clip-4k.mp4\\n\\n# Smoother motion or slow motion (resolution unchanged)\\nvideodraft interpolate ./clip.mp4 --fps 60 --download ./media/clip-60fps.mp4\\nvideodraft interpolate ./clip.mp4 --model Chronos --slowdown 4 --download ./media/clip-slowmo.mp4\\n```\\n\\nUse these when the content is correct and only quality or resolution needs improvement. If the poster text, composition, subject, or motion is wrong, edit or regenerate instead.\\n","references/models.md":"# Choosing models (and predicting cost)\\n\\nAlways consult the live catalog instead of memorizing this page \u2014 models change weekly:\\n\\n```bash\\nvideodraft models image --json # every image model + inputs (aspect ratios, resolutions, max refs)\\nvideodraft models video --json # every video model + inputs + per-second pricing metadata\\nvideodraft models audio --json # standalone audio/media models + pricing inputs\\nvideodraft models voices --json # TTS voices\\nvideodraft models styles --json # visual style presets\\n```\\n\\n## Task-based model selection\\n\\nFollow this order every time:\\n\\n1. The user named a model: use it, never substitute.\\n2. Otherwise pick a **Tier 1** model from the task\'s inputs, duration, audio, quality, speed, and cost, and pass it explicitly instead of relying on a blind platform fallback.\\n3. Use a **Tier 2** model only when no Tier 1 model supports the request, and say why. Never pick Tier 2 on your own for price, speed, novelty or variety.\\n\\nTier 1 is the rows marked **T1** below. Every other catalog model is Tier 2 and stays available by name. The catalogs carry this themselves: every `videodraft models image|video|audio --json` entry has `tier` (1 or 2) plus `recommended` / `recommended_for`, and the top-level `recommended` array is Tier 1, best first. Trust the catalog over this page when they disagree.\\n\\n### Images\\n\\n| Need | Choose | Why |\\n| ------------------------------------------------------------------------------------------ | ------------------------ | ----------------------------------------------------------------- |\\n| T1: Most generation, editing, character consistency, or reference work | `nano-banana-2` | Best general default; 1K/2K/4K and up to 14 reference images |\\n| T1: Highest-quality complex generation or reasoning | `nano-banana-pro` | Premium Nano Banana quality and reasoning |\\n| T1: Fast, inexpensive drafts and iteration | `nano-banana-2-lite` | Fastest/cheapest Nano Banana option; 1K only, up to 14 references |\\n| T1: Posters, title cards, signs, logos, or any image with important readable text | `gpt-image-2.5-flare` | Fast OpenAI option; 16 references, 1K/2K/4K and PNG output |\\n| T1: Complex multi-image composition, precise editing, or a strong alternate interpretation | `gpt-image-2.5-sunburst` | More fidelity/detail; 16 references, quality through max |\\n| T1: Vector / SVG output (logos, icons, illustrations that must scale) | `recraft-v4` | Recraft V4.1 text-to-vector; SVG only, no reference images |\\n| T2: By name, or an explicitly requested xAI look | `grok-imagine` | 2 cr flat (3 with a reference); 1 reference image |\\n| T2: xAI at 2K or with a quality tier, up to 3 edit references | `grok-imagine-2.0` | 1K 4 (low) / 6 (medium), 2K 6 / 8, +1 cr per reference image |\\n\\nGPT Image 2.5 has two IDs: `gpt-image-2.5-flare` replaces the former GPT Image 2 recommendation; `gpt-image-2.5-sunburst` is the precision/detail alternative. Other model recommendations are unchanged. Explicit `gpt-image-2` selections continue to use the previous model.\\n\\nUse the same basic image controls as GPT Image 2: `--ref`, `--ar`, `--resolution 1K|2K|4K`, `--quality`, and `--num 1..4`. GPT Image 2.5 quality supports auto/low/medium/high/xhigh/max. Auto is priced at Max in customer-credit estimates. The default platform path uses OpenAI; a selected personal connection stays exclusive. A Pika personal connection requires an explicit quality from low/medium/high/xhigh/max because its upstream API has no auto value. Other unsupported connection settings return an error instead of switching payer or changing the request.\\n\\n```bash\\nvideodraft generate image \'A blue glass bird\' --model gpt-image-2.5-flare --ar 1:1 --resolution 2K --quality high --estimate\\nvideodraft generate image \'A blue glass bird\' --model gpt-image-2.5-flare --ar 1:1 --resolution 2K --quality high --num 2\\n```\\n\\n`videodraft shots` forwards resolution and quality for both normal and grid generation. Estimates use the project\'s model and aspect when omitted; grid canvas costs may differ from ordinary shot costs. Use `videodraft costs --model <id> --ar <ratio> --resolution <tier> --quality <tier>` for a specific per-image quote.\\n\\n### Videos\\n\\n| Need | Choose | Important limits |\\n| -------------------------------------------------------------------------------------------------------------- | ------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |\\n| T1: Most generation, first/last-frame, mixed-reference, source-edit, or extension requests | `gemini-omni-1.1-flash` | 3-10s output; 360p/720p/1080p/4K at 3/10/15/30 cr/s; audio always; up to 10 image inputs, and 3 reference videos <=3s on generate only; edit source <=10s; continuation or 3-10s extension of a 1-30s source, 40s total reachable only by extending from <=30s |\\n| T2: Grok 1.5 text, first-frame, or 1-7 image-reference clips with native audio and optional 1080p | `grok-imagine-video-1.5` | 1-15s; 480p/720p/1080p for text/first-frame; references are 480p/720p only; no last frame |\\n| T2: Document or webpage reference generation; by name otherwise | `wan-3.0` | 2-30s or auto; 480p/720p/1080p at 7/14/28 cr/s; 10 image, 5 video, 5 audio refs, 20 media files total; document/web refs require thinking |\\n| T2: 480p/768p/2K/4K with native stereo audio, first/last frames, or mixed image/video/audio references | `minimax-h3` | 5-15s; 5/6/13/16 cr/s by resolution; up to 9 image, 3 video, 3 audio refs, 12 files total; first 5 reference images free then 8 cr each; reference video/audio each total <=15s |\\n| T1: Text, first/last-frame, or mixed-reference video with native audio and stronger prompt adherence | `minimax-h3-max` | 5-15s; 5/8 cr/s at 480p/768p plus pooled reference tokens; up to 9 image, 3 video, 3 audio refs, 12 files total; seed and disabled/balanced/quality prompt expansion; provider safety checker off by default |\\n| T2: Images pinned to specific moments (keyframes); by name otherwise | `flux-3` | 5-20s (auto for text/first-frame only); 720p/1080p; up to 10 keyframes; `--quality draft` is 720p-only at ~1/3 the cost |\\n| T1: Video/audio references, mixed reference media, broad aspect ratios, frame-mode first+last frame, or 11-15s | `seedance-2` | 4-15s or auto; up to 9 image, 3 video, and 3 audio refs; audio toggle; Mini/Fast are 480p/720p only |\\n| T1: Single takes past 15s, or more references than Seedance 2.0 allows | `seedance-2.5` | 4-30s or auto; up to 30 image, 10 video, and 10 audio refs (50 files total); one quality tier; 480p/720p/1080p |\\n| T2: Fast polished 3-15s video with first frame, multi-prompt, and native audio | `kling-v3-turbo` | Audio always on; Pro default; no end frame, reference-media mode, or elements |\\n| T1: Cinematic 3-15s with reference images, image/video elements, first+last frame, audio, or 4K | `kling-o3` | 7 combined image refs/elements, reduced to 4 with a video element; Standard/Pro/4K |\\n| T1: Kling 3-15s image-to-video with first+last frame, image/video elements, multi-prompt, audio, or 4K | `kling-3.0` | Elements require a start image; image or video elements can bind voice_id; Standard/Pro/4K |\\n| T2: User explicitly requests Veo, or the selected workflow specifically needs Veo | `google-veo3.1` | Good fallback, but not the preferred general model |\\n\\nRouting rules:\\n\\n- Inexpensive pass or draft at 480p/768p with native audio: use MiniMax H3 Max (5/8 cr/s plus reference tokens).\\n- Grok 1.5 reference mode accepts 1-7 images only. Address them in array order as `<IMAGE_0>` through `<IMAGE_6>`. Do not combine reference images with `--start-image`, `--end-image`, `--ref-video`, or `--ref-audio`.\\n- Grok 1.5 first-frame mode accepts one `--start-image`, derives the output aspect ratio from that image, and does not support `--end-image`. Text and first-frame modes support 480p, 720p, or 1080p. Reference mode supports 480p or 720p.\\n- Grok 1.5 always generates native audio. Do not pass `--no-audio`, `--seed`, `--negative`, or `--quality`.\\n- Around 11-15 seconds with native audio: use Seedance, Kling 3.0 / O3, or MiniMax H3 Max, not Gemini. Past 15 seconds: Seedance 2.5.\\n- Dialogue: write the line in the prompt, in quotes, with who says it and how. `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` and `kling-o3` speak it natively; add `--ref-audio` on Seedance or a bound Kling voice for a specific voice. Do not generate speech separately and lip-sync it on.\\n- One existing source video that should be edited: use `videodraft edit video`, which auto-selects Gemini Omni 1.1 Flash for a source up to 10s. Generic `generate video --ref-video` retains source-edit behavior when `--video-task` is omitted, but the dedicated edit command is clearer.\\n- Video supplied as a creative reference: Gemini Omni 1.1 Flash accepts up to 3 videos of at most 3 seconds each, including mixed image and video input. Creative video references are accepted on `--video-task generate` ONLY: edit and extend take exactly one input video, so `--ref-video` cannot accompany `--source-video` or `--previous-interaction-id`. Use Seedance 2.5 for more or longer video/audio references, or Seedance 2.0 when quality-tier control matters.\\n- First and last frame control: every Tier 1 video model supports it (Gemini Omni 1.1 Flash, Seedance, Kling O3, Kling 3.0, MiniMax H3 Max). For Gemini, every start/end frame counts toward the 10-image total, leaving up to 9 references with a start frame or 8 with both frames.\\n- Extension with a 1-30s uploaded source uses Gemini Omni 1.1 Flash plus `--source-video` and `--video-task extend` (or `--extend`). Pass an explicit 3-10s output duration; the model appends it at the end, up to 40 seconds total. Creative `--ref-video` clips CANNOT accompany the source: edit and extend take exactly one input video. New dialogue is allowed only when the source video is silent; adding speech on top of a source that already has speech is refused with \\"the model is currently unable to process speech edits\\". Continuation uses `--previous-interaction-id <id>` and defaults to a conversational edit; `--video-task extend` lengthens it. A continuation is submitted as the prior turn\'s output video, so it obeys the same source limits (<=30s to extend, <=10s to edit) and the same silent-source dialogue rule. 40s total is reachable only by extending from a source of 30s or less, so a ladder dead-ends past 30s. Fal returns interaction IDs, but its currently published callable v1.1 endpoints do not expose continuation/extension or mixed source-edit references; never infer a route or silently fall back to paid Google.\\n- Wan 3.0 reference mode and frame mode are separate. It accepts 10 images, 5 videos, and 5 audio clips, at most 20 media files total. Video and audio each total at most 15 seconds. `--file-url` and `--web-url` require `--thinking`. `--auto-duration` cannot be combined with `--duration`.\\n- MiniMax H3 and H3 Max reference mode and first-plus-last-frame mode are separate. Audio cannot be the only reference. Address references as `Image 1`, `Video 1`, and `Audio 1` in array order.\\n- Seedance reference mode and first-plus-last-frame mode are separate. Do not promise reference video/audio plus a last frame in one generation.\\n- Multi-prompt sequencing: use Kling 3.0 Turbo, Kling O3, or Kling 3.0.\\n- Kling 3.0 image-to-video and Kling O3 reference-to-video accept repeatable structured `--element` JSON. Each element is image-backed (`frontal_image_url` plus 1-3 `reference_image_urls`) or video-backed (`video_url`). Either form may include `voice_id`. Keep the voice ID inside the same element so the association is preserved. Reference them as `@Element1`, `@Element2`. Kling V3 Turbo rejects elements.\\n- Kling 2.6 Pro image-to-video accepts one or two repeatable `--voice-id` values. Cite them as `<<<voice_1>>>` and `<<<voice_2>>>` in the prompt. Voice control costs 17 cr/s. Use `videodraft kling-voices list|create|delete` to manage the separate Kling video-control voice library. Creating a voice requires a clean 5-30 second, up-to-50MB single-speaker `.mp3`, `.wav`, `.mp4`, or `.mov` sample plus `--confirm-consent`. Creation costs 1 VideoDraft credit on the platform Fal account and 0 credits with Fal BYOK; use the create command\'s `--estimate` flag to check without creating.\\n- Kling V3 voice control costs 16 cr/s for Standard and 20 cr/s for Pro.\\n- Seedance quality: `mini` for the lowest cost, `fast` for speed, `standard` for maximum quality and for 1080p/4K. Seedance 2.5 has a single tier and ignores `--quality`.\\n- Longer than 15 seconds, or more than 9 image / 3 video / 3 audio references: use Seedance 2.5. It reaches 30s and 30/10/10 references (50 files total) at 480p/720p/1080p, but has no 4K.\\n- Seedance 2.x allows real people by default, the same as AI Studio: Byteplus first, a submit-time Fal fallback, and Fal\'s higher tier-specific rate. Turn it off with MCP `allow_real_people: false` or CLI `--no-allow-real-people` for the lower Byteplus-only rate, only when the user wants the cheapest run and nothing in the job is a real identifiable person (text-only requests, non-people, anime, clearly synthetic or stylized characters).\\n- If a Seedance request made with the option off fails with `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, estimate the higher rate, follow the user\'s spend policy, and retry once with the option on. The response also carries `retry_with: { allow_real_people: true }`, `cli_flag: \\"--allow-real-people\\"`, and `retry_policy: \\"once\\"`. If the option was already on, do not repeat the same request. Byteplus may accept a task and reject the output later. VideoDraft refunds that failed generation, but it cannot reroute the asynchronous failure to Fal. Rephrase or change the references instead.\\n\\n### Video edit and motion-control categories\\n\\nUse `videodraft models video --category video_edit` for existing-video transforms and `--category motion_control` for motion transfer.\\n\\n| Need | Command/model | Important limits |\\n| ------------------------------------------ | ---------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------- |\\n| Most edits of a source up to 10s (DEFAULT) | `videodraft edit video <video> \\"...\\"` (auto-selects `gemini-omni-1.1-flash`) | Source <=10s or it REFUSES; up to 10 image refs and 3 creative video refs <=3s each; 360p/720p/1080p/4K; audio always regenerated (no `--preserve-audio`) |\\n| Cheapest simple prompt edit | `videodraft edit video <video> \\"...\\" --model grok-imagine-video-edit` | No image refs; source silently truncated to 8s; auto/480p/720p |\\n| Edit with several image references | `--model happy-horse-video-edit --ref ...` | Up to 5 refs; 720p/1080p; source silently truncated to 15s; highest rate (28-56 cr/s) |\\n| Controlled Kling edit | `--model kling-o3-video-ref-edit --ref ...` | Up to 4 refs; Standard/Pro; source silently clamped to 3-10s |\\n| Transfer reference motion to an image | `videodraft edit motion <image> [direction] --motion-video <video>` | Prompt/direction is optional; Kling V3 default; optional one image-only `--element`; element requires video orientation |\\n\\nIf the user explicitly names one of these models, preserve it. The CLI uploads local source videos and reference images automatically. Editing returns an async job and waits by default.\\n\\n**Omit `--model` and the SERVER chooses**, because only it can measure the source. It picks `gemini-omni-1.1-flash` for any source up to 10s. When that is not safe (source longer than 10s, unmeasurable duration, `--preserve-audio`, more than 10 refs, or Fal BYOK with refs) it spends nothing and returns a priced menu: per model, the seconds it would edit, the seconds it would drop, and the credit cost. The CLI prints that table and exits 2. Show it to the user, then re-run with `--model`.\\n\\n**Truncation is the trap here.** Only Gemini refuses a source it cannot fully consume. Every other edit model accepts a 30s clip and returns an edit of its first 8-15 seconds with no error. `videodraft edit video` now warns when this will happen and the tool response carries `source.truncated` / `source.dropped_seconds`; relay it to the user rather than letting them discover it in the output.\\n\\nTo edit a longer source without losing its tail, cut it into <=10s pieces with `videodraft_editor`, edit each with Gemini, then reassemble and export there. VideoDraft has no server-side split/concat, so this needs the native editor.\\n\\nKling O3 is also exposed for reference generation. `videodraft generate video --model kling-o3-video-ref-edit` requires exactly one `--ref-video` and generates a new reference-guided clip. Wan 3.0 handles new text, frame, and mixed-reference generation, but is not an existing-source edit model.\\n\\n### Reference-first video workflow\\n\\n- Prefer a start frame or reference image whenever a specific character, product, location, style, composition, or brand identity must stay recognizable.\\n- If the user gives a reference, pass it. Never silently replace it with a text description.\\n- If no reference exists and continuity matters, generate a still first with the user\'s explicitly requested compatible image model, otherwise use Nano Banana 2. Wait for the image URL, then animate it with the selected video model. Confirm the combined image plus video cost before starting.\\n- For multi-shot scenes, generate shot images with `videodraft shots <project_id> --model <selected-image-model> --grid`. Preserve an explicitly requested compatible image model; otherwise use `nano-banana-2`. The grid establishes the scene and characters together, then decodes into individual shot images.\\n- Animate the decoded shot images as per-shot start frames or references. Do not independently text-generate each video clip when the shots need to match.\\n- Pure text-to-video remains appropriate for generic one-off footage where no subject, composition, or continuity needs to be preserved.\\n\\n### Audio\\n\\n- **Seed Audio 1.0**: use `videodraft generate audio` for open-ended speech, sound, music, or prompt-driven audio editing. It accepts up to three audio references or one image. Address audio references as `@Audio1`, `@Audio2`, and `@Audio3`. Preset and custom cloned voice IDs are supported. Output is up to 120 seconds. There is no requested-duration input. The CLI automatically retries transient responses with one idempotency key. To recover after the CLI process itself is interrupted, set `--idempotency-key <uuid>` on the original command and reuse it.\\n- **Voiceover/TTS**: prefer ElevenLabs. Brittney is the platform default voice; under ElevenLabs BYOK, use a compatible voice from the user\'s account. Honor another supported voice/provider when the user explicitly selects it. ElevenLabs voices run on Eleven v4 in one of two modes: Standard (the default, best quality) or Turbo (Eleven v4 Turbo, faster and half the price). Keep Standard unless the user wants faster or cheaper speech. Pick the mode with `generate voiceover --mode standard|turbo` or `produce --voice-mode standard|turbo`; over MCP, pass `mode` to `generate_voiceover` or `voice_mode` to `produce_project`. Google, OpenAI and cloned `custom-*` voices ignore it. v4 takes no style or speed settings and no SSML `<break>` tags; audio tags such as `[whispers]` work. Dialogue stays on Eleven v3.\\n- **Dialogue, voice changing, and dubbing**: ElevenLabs only.\\n- **Sound effects**: ElevenLabs Sound Effects only.\\n- **Music**: use `lyria-3.5` for all music, short or long (up to ~3 minutes), with vocals/lyrics or instrumental music. When `--model` is omitted the CLI sends no model and the server default applies, which is `lyria-3.5` wherever the backend supports it. If the server answers that `lyria-3.5` is an unknown model, that backend predates it: rerun without `--model` (the server default then applies), or use `lyria-3-pro-preview` for a track longer than 30 seconds. `lyria-3-clip-preview` (fixed ~30s) and `lyria-3-pro-preview` remain available for explicit requests. Lyria length and structure are prompt-guided, not exact; for an instrumental, include \\"instrumental only, no vocals\\" in the prompt. `--length` and `--instrumental` only apply to ElevenLabs. Use `elevenlabs-music-v2.5` when a specified 3-300 second length, composition plan, or style reference track matters. Lyria references: up to 10 images on Google, 1 on Fal BYOK; 3.5 Fal prompts are 1-5000 characters (an image-only request gets a neutral prompt). `elevenlabs-music` is an alias for v2.5. Use `elevenlabs-music-v1` only when the user asks for v1 by name; it takes a prompt, `--length` and `--instrumental` only.\\n- **ElevenLabs Music v2.5 inputs**: either a prompt (`--length`, `--instrumental`) or a composition plan, never both. A plan is 1-30 sections of 3-120 seconds each, up to 300 seconds in total, and `--plan`/`--section` pick v2.5 when `--model` is omitted. Empty section parts default to 20 seconds and an instrumental part. Build one with repeatable `--section \\"<seconds>|<style, style>|<text>\\"` (use `\\\\n` for line breaks) or pass `--plan plan.json` with `{\\"chunks\\":[{\\"text\\",\\"duration_ms\\",\\"positive_styles\\",\\"negative_styles\\",\\"context_adherence\\",\\"audio_reference\\"}]}`. Section text is an optional `[Section name]`, lyric lines, and `{inline directions}`. Put 6-7 specific English styles on the first section; it sets the genre. `--ref-audio <url|file>` adds a style reference clip to the first section (window up to 30 seconds via `--ref-start`/`--ref-end` in ms, `--ref-strength low|medium|high|xhigh`); local files are uploaded, including relative `audio_reference.audio_url` paths in a plan file. `--seed` works only with a plan. `--format` picks the output (`mp3_48000_192` default on v2.5; `pcm_*`, `ulaw_8000` and `alaw_8000` arrive as stereo WAV files). Style references don\'t run on the user\'s own ElevenLabs key; with that key connected, drop the reference or ask the user to turn the key off. The CLI retries transient responses with one idempotency key; set `--idempotency-key <uuid>` to recover after an interruption.\\n- Voice Changer and Dubbing require the source media duration and currently accept source media up to 300 seconds.\\n\\n**User-supplied ElevenLabs voice IDs:** voiceover, dialogue, and voice changing accept raw IDs (16-64 alphanumeric characters) or `elevenlabs-<id>`. `videodraft models voices` / MCP `list_available_voices` is for discovery, not an allowlist. Pass a supplied ID directly even when it is absent from the catalog; do not reject it or substitute a catalog voice. The voice must be accessible to the provider account used for generation. Private or cloned voices may require the user\'s connected ElevenLabs key. A failed voice listing does not prove that a supplied ID cannot generate; a connected key with generation access may still work. If the provider rejects generation access, report that error rather than silently changing the voice.\\n\\nUse `--voice <id>` for voiceover and voice changing, and repeat `--line \\"<id>:Text\\"` for dialogue. These examples show both ID forms; replace the sample IDs with the user\'s supplied IDs:\\n\\n```bash\\nvideodraft generate voiceover \\"Hello there.\\" --voice kPzsL2i3teMYv0FxEYQ6 --download ./media/voiceover.mp3\\nvideodraft generate dialogue --line \\"kPzsL2i3teMYv0FxEYQ6:Hello.\\" --line \\"elevenlabs-kmSVBPu7loj4ayNinwWM:Welcome back.\\" --download ./media/dialogue.mp3\\nvideodraft generate voice-changer ./media/source.wav --voice elevenlabs-kPzsL2i3teMYv0FxEYQ6 --duration 12 --download ./media/changed-voice.mp3\\n```\\n\\nFor MCP, use `generate_voiceover.voice_id`, `generate_dialogue.lines[].voice_id`, or `change_voice.voice_id` with either form. ElevenLabs IDs are separate from Kling video-control voice IDs and MiniMax `custom-*` cloned-voice IDs. Do not convert IDs from those systems into ElevenLabs IDs.\\n\\n### Avatar / talking head\\n\\n**First decide the framing, not just \\"someone talks\\".** This lane animates a PORTRAIT facing the lens. A character delivering a line inside a real scene, with blocking, framing or camera movement, belongs to `generate video` instead: write the dialogue in the prompt and `gemini-omni-1.1-flash`, `seedance-2.5`, `seedance-2`, `kling-3.0` or `kling-o3` voices it natively (add a Seedance `--ref-audio` clip or a Kling voice bound per element for a specific voice). That is the default for any talking character; never plan silent video + TTS + lip-sync. Sending a cinematic shot to Fabric returns a head-on talking headshot, not the shot that was asked for. `videodraft models video --json` lists these under `recommended.in_scene_dialogue`.\\n\\nFor a presenter, spokesperson or explainer speaking to camera, VEED Fabric is the preferred model. Choose the dedicated path from the media the user already has:\\n\\n| Starting media | Command | Use |\\n| ------------------------------------------ | ------------------------------------------------------------------------- | ---------------------------------------------------------- |\\n| Portrait + script, reusable avatar record | `videodraft avatar create <portrait> --script \\"...\\"` then `avatar render` | Managed avatar flow with bundled speech preparation |\\n| Portrait + text | `videodraft avatar fabric <portrait> --text \\"...\\"` | One-off direct VEED Fabric text mode |\\n| Portrait + existing audio | `videodraft avatar fabric <portrait> --audio <audio>` | One-off direct VEED Fabric audio lip sync |\\n| Portrait + existing audio, short and sharp | `videodraft avatar h3-lipsync <portrait> --audio <audio>` | MiniMax H3 Max Lip Sync, 5-14.8s at up to 2K |\\n| Existing video + existing audio | `videodraft avatar lipsync <video> --audio <audio>` | Sync Labs Lipsync 2 (Tier 2): re-dub existing footage only |\\n\\nThe managed renderer is VEED Fabric Fast (`veed/fabric-1.0/fast`). Direct Fabric, MiniMax H3 Max Lip Sync, and Sync Labs are paid AI Studio generations and return async job IDs.\\n\\n1. Obtain the avatar image. Prefer the user\'s supplied portrait or an existing character. If none exists, use the user\'s explicitly requested compatible image model, otherwise generate a front-facing head-and-shoulders portrait with `nano-banana-2`, direct eye contact, a natural expression, and a clean background. Match the intended video aspect ratio when practical.\\n2. If the portrait is visibly soft or too small, run Topaz image enhancement/upscaling before animation.\\n3. Generate a script only if needed: `videodraft avatar script \\"<idea>\\"`.\\n4. Create the avatar record and speech: `videodraft avatar create <portrait-url-or-file> --script \\"...\\" --voice <id> --ar 9:16`. Prefer ElevenLabs when unspecified, but honor another explicitly selected supported voice/provider.\\n5. Render with VEED Fabric: `videodraft avatar render <avatar_video_id> --resolution 720p`.\\n\\nThe portrait is passed as the avatar\'s character image, not as a generic video\'s start frame. Prefer rendering directly at 720p. Use 480p only when the user prioritizes lower cost. Avatar script generation and `avatar create` (including speech) are bundled/free. Confirm the Fabric render cost, plus portrait generation or upscaling when needed.\\n\\nDirect Fabric text/audio, H3 Max Lip Sync, and Sync Labs do not use the managed avatar record. The CLI uploads local portrait, video, and audio files automatically. `avatar fabric --speed fast` applies only to audio mode. `avatar h3-lipsync` takes no prompt: pick `--resolution 480P|768P|1080P|2K` (default 768P), an optional `--seed`, `--no-transcription` to sync without transcribing the audio (transcription is on by default), and `--safety-checker` only when the user asks for Fal\'s checker. The portrait\'s width / height must be 0.4-2.5, and audio shorter than 5 seconds is refused before any charge. Sync costs 5 credits per verified audio second; under Fal BYOK, `sync_mode` remains available but `temperature` and `active_speaker` are ignored by the provider.\\n\\n### Upscaling / enhancement\\n\\n- **Images**: Topaz via `videodraft upscale image <url-or-file> --scale 1x|2x|4x [--mode precision|generative|creative] [--model <name>]`. Modes: `generative` (default; Wonder 3.5, Topaz\'s recommended model for AI-generated images; `Redefine` takes `--prompt`, `--creativity 1-6`, `--texture 1-5`), `precision` (faithful and cheapest; Standard V2, High Fidelity V3, Low Resolution V2, CGI, Text Refine; use for clean real photos), `creative` (Bloom 2; artistic, `--prompt`, `--creativity 1-9`). Extra knobs: `--no-face-enhance`, `--face-strength`, `--sharpen`, `--denoise`, `--fix-compression`, `--format png`. Use 1x for light enhancement without enlargement, 2x as the general default, and 4x only when the source quality and target size justify it. The server must verify dimensions from a readable image of at most 50 MB; `--width` and `--height` are compatibility hints and cannot bypass a failed probe. The result may be immediate or queued; the CLI waits by default. Use `--no-wait` for an async job ID, and poll it with `status` or `wait`. Cost: 8 credits per started 24 MP of output in precision, per 8 MP (Wonder 3/3.5) or 4 MP in generative, per 2 MP in creative.\\n- **Videos**: Topaz via `videodraft upscale video <url-or-file> --resolution 720p|1080p|4k` (preferred) or `--scale 2x`, plus `--mode precision|generative|creative` and `--model <name>`. `generative` (default; Starlight Precise 2.6, Topaz\'s recommended model for AI-generated footage; Starlight Fast 2 at half price) costs 12 credits/s up to 1080p and 26 at 4K. `precision` (Proteus, Proteus Natural, Iris, Gaia 2 for animation, Rhea, Theia, Artemis, Dione) is 6x cheaper at 1 / 2 / 6 credits per second for 720p / 1080p / 4K output; use it for real footage or when cost matters. `creative` (Astra 2, `--prompt`, `--creativity`, `--realism`, `--sharp`) always renders 4K at 50 credits/s. `--fps 60` delivers 60fps on the same pass; any output above 30fps (including a 50/60fps source) doubles every rate; the server must verify duration, dimensions, and frame rate from an MP4/MOV source of at most 100 MB. `--duration`, `--width`, `--height`, and `--source-fps` are compatibility hints and cannot override billing or bypass a failed probe. The same source-verification requirement applies to frame interpolation. `--scale` also accepts intermediate factors such as `1.5x`. Max source length 5 minutes. The job is asynchronous; the CLI waits by default, while MCP callers poll `check_generation_status`. MCP video input must be VideoDraft-hosted, so upload local or external sources first.\\n- **Frame interpolation / slow motion**: `videodraft interpolate <url-or-file> --fps 60 [--model Apollo|Chronos|Aion] [--slowdown 1-8]` (MCP `interpolate_video`). Resolution is unchanged. Apollo (default) for smooth general conversion, Chronos for natural slow motion, Aion for extreme slow motion. Apollo/Chronos cost 3 credits per output second up to 1080p (6 at 4K); Aion 5 / 17. These rates cover targets up to 60fps; above 60fps multiply by target FPS / 60 (120fps doubles the rate). Output seconds = source seconds \xD7 slowdown. The final charge rounds up to a whole credit.\\n- Use upscaling to preserve the image/video while improving detail, resolution, or cleanup. It cannot fix the wrong subject, misspelled text, bad framing, unwanted objects, broken continuity, or incorrect motion. Use an edit or regeneration for those problems. Keep `generative` for AI-generated sources; switch to `precision` for clean real photos/footage or a cheap pass, and `creative` only when the user wants an artistic reinterpretation.\\n- For a new Fabric avatar, render directly at 720p instead of rendering at 480p and then upscaling. Upscale the source portrait first only when the portrait itself is low quality.\\n\\n## Capability gotchas\\n\\n- Each model\'s `inputs` block is authoritative: supported `aspect_ratios`, `resolutions`, `quality_options`, `start_frame`/`end_frame`, `max_reference_images/videos/audio`, `multi_prompt`, `audio_toggle`. Passing an unsupported input fails with a clear error \u2014 check first, don\'t trial-and-error paid calls.\\n- Most video models support only 16:9 / 9:16 / 1:1. A 3:4 request hard-fails on most.\\n- `--seed` reproduces a specific output on models that support it (e.g. Flux, Ideogram V4). GPT Image 2.5 rejects any explicit seed; older models may ignore unsupported seeds. You do not need a seed for variation \u2014 `--num` already varies.\\n- `--rendering-speed` applies to Ideogram (V3: `Default`/`Turbo`/`Quality`; V4: `Turbo`/`Balanced`/`Quality`) and affects image cost \u2014 pass it to `videodraft costs ... --rendering-speed <tier>` for an accurate estimate. Always trust `videodraft models image --json` over this list; new models and tiers appear there the moment the platform ships them, with no CLI update.\\n- `seedream-v5-pro` supports unified text-to-image and reference-image editing with up to 10 image references. Use `--resolution 1K` for 7 credits/image or `--resolution 2K` for 14 credits/image.\\n- Reference inputs: `--ref <img>` (images, including up to 10 total for Gemini Omni 1.1 Flash and Wan 3.0, and 7 for Grok 1.5), `--source-video <v>` (Gemini uploaded edit/extension source), `--ref-video <v>` (up to 3 creative videos <=3s each for Gemini Omni 1.1 Flash; also Wan 3.0, MiniMax H3, MiniMax H3 Max, and Seedance 2), `--ref-audio <a>` (Wan 3.0, MiniMax H3, MiniMax H3 Max, Seedance 2), and `--element \'<json>\'` or `--element @elements.json` for Kling V3/O3. For an exact Seedance 2.x reference-video `--estimate`, add `--ref-video-seconds <combined-seconds>`; Seedance bills input seconds alongside output. Wan 3.0 and MiniMax H3 do not bill input reference seconds; MiniMax H3 Max bills them as pooled reference tokens. The CLI uploads local media references and every structured element without flattening the source/reference roles. `--segment \\"<prompt>:<seconds>\\"` (repeatable) drives Kling 3.0, Kling 3.0 Turbo, and O3 multi-prompt generation. Use 1-6 segments of 1-15 whole seconds each, with 3-15 seconds total. `generate image --video-ref` is the nano-banana-2 video reference.\\n- The top-level prompt is OPTIONAL for Gemini Omni 1.1 Flash media-input or continuation calls, Wan 3.0 frame/reference modes, `generate video` with multi-prompt models, and Kling 3.0 Turbo (`--model kling-v3-turbo`) image-to-video. Text-only Gemini and Wan calls still require a prompt unless their live schema says otherwise.\\n- Hosted AI Production fallback: `videodraft produce <project> --mode full_video` generates one Seedance 2 video per scene, with real people allowed by default at the higher Fal-tier rate on every submitted scene segment; add `--no-allow-real-people` for the lower Byteplus-only rate when no scene shows a real identifiable person. The MCP equivalent is `produce_project` with `mode: \\"full_video\\"` (and `allow_real_people: false` to opt out). If a partial opted-out run returns `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, rerun the same project once with an explicit `--allow-real-people` after cost confirmation. The server reconciles asynchronous results first, preserves running/completed jobs, and retries only failed placeholders carrying that exact code. Poll with `videodraft generations`, then `videodraft finalize <project>` swaps them into the hosted timeline before `export`. In VideoDraft ADE, do not choose this path while `videodraft_editor` is available unless the user explicitly requests hosted production. Generate or download the scene assets, import them, and assemble/export with the native editor instead. If the user explicitly requests another compatible video model for a hosted production, do not use this fixed Seedance path; generate the project shots manually with the requested model and attach them to the hosted timeline.\\n\\n## Cost model\\n\\n- Images: per image (\xD7 `--num`). Matrix-priced models (GPT-Image, Nano Banana Pro, Seedream v5 Pro) vary by resolution/quality.\\n- Video: usually credits/second \xD7 duration; rate depends on model + resolution + quality + native audio on/off.\\n- Gemini Omni 1.1 Flash: 3 / 10 / 15 / 30 credits per output second at 360p / 720p / 1080p / 4K (720p default). Extend appends an explicit 3-10 seconds to a 1-30s source, whether uploaded or resolved from a prior interaction; 40 seconds total is reachable only while the source stays at or under 30s. Fal BYOK generation and basic source edits cost zero VideoDraft credits; continuation, extension, and separate creative references on a source edit are unavailable under Fal BYOK.\\n- MiniMax H3: 5 / 6 / 13 / 16 credits per output second at 480p / 768p / 2K / 4K (768p default). The first 5 reference images are included, then 8 credits for each additional image. Reference video and reference audio are NOT billed.\\n- MiniMax H3 Max: 5 credits per output second at 480p or 8 credits per second at 768p (default), plus pooled reference tokens. The first 4,096 reference tokens are free and each additional 1,000 costs 2 credits. An image is `(width x height) / 1024` tokens (a 1024x1024 reference is 1,024, so four of them are free); a reference video is 2,886 tokens per second at 480p or 7,459 at 768p; reference audio is about 2,121 per second. Pass `--ref-audio-seconds` for an exact estimate with audio references. The server measures each reference image, so a quote that assumes 1024x1024 is a floor.\\n- Wan 3.0: 7 / 14 / 28 credits per output second at 480p / 720p / 1080p. Auto duration reserves 30 seconds and reconciles unused credits from the provider-reported output duration. Fal BYOK charges zero VideoDraft credits.\\n- Grok Imagine Video 1.5: 8 credits/output second at 480p, 14 at 720p, or 25 at 1080p, plus 1 credit for each first-frame or reference image. Native generated audio is part of every output.\\n- Kling voice control: 17 credits/output second for Kling 2.6 Pro, 16 at Kling V3 Standard, and 20 at Kling V3 Pro.\\n- Shot-image batches: one image per shot (+1 grid image per scene in `--grid` mode) \u2014 the largest single spend in the pipeline.\\n- VEED Fabric avatar renders: ~10 credits/sec at 480p, ~20/sec at 720p. Avatar creation and its speech are bundled/free; only optional portrait generation/upscaling adds cost before the render.\\n- Direct VEED Fabric: text or normal audio is 8 credits/sec at 480p and 15/sec at 720p; fast audio is 10/sec at 480p and 20/sec at 720p.\\n- MiniMax H3 Max Lip Sync: 5 / 8 / 16 / 32 credits per output second at 480P / 768P (default) / 1080P / 2K. The video runs as long as the audio (at least 5s; only the first 14.8s is used), billed on the server-measured length rounded up, so at most 15 seconds. Use MP3, WAV, M4A (AAC), or AAC audio; other formats run only on the user\'s own Fal key.\\n- Sync Labs Lipsync 2: 5 credits per verified audio second.\\n- Voiceover TTS: 10 credits per 1000 characters for standard voices (including ElevenLabs Standard, Eleven v4), 5 per 1000 for ElevenLabs Turbo (Eleven v4 Turbo), and 30 per 1000 for cloned `custom-*` voices (min 1, pro-rated); applies to standalone voiceovers AND per-scene narration during `produce`. Quote Turbo with model id `voiceover-turbo` (CLI: `costs voiceover --mode turbo`) and cloned voices with `voiceover-cloned`. Silent tracks are free. Voice cloning itself is a flat 150 credits per clone.\\n- Lyria music: flat per track, 4 credits (Clip) / 10 credits (3.5) / 8 credits (legacy Pro). Fal BYOK uses zero VideoDraft credits.\\n- Seed Audio 1.0: 19 credits per actual output minute, prorated and rounded up to a whole credit. VideoDraft reserves the 120-second maximum of 38 credits and refunds the unused portion after generation. Fal BYOK is free.\\n- ElevenLabs audio: sound effects are per second, dialogue is per character, music/voice-changer/dubbing are per started minute. ElevenLabs Music v2.5 and v1 both cost 60 credits per started output minute; estimate a composition plan with its total length. Voice changer and dubbing reject source media above 300s in the current synchronous flow.\\n- Seedance 2.0 / 2.5 real people: allowed by default, which keeps Byteplus first, permits a submit-time Fal fallback, and prices at Fal\'s rate for that tier: 2.0 Mini 8/16 cr/s, Fast 11/25, Standard 14/31/69/156, 2.5 23/48/114 for 480p/720p/1080p. `--no-allow-real-people` uses the Byteplus-priced path instead (2.0 Mini 4/8, Fast 6/13, Standard 7/16/38/78, 2.5 11/24/57); Byteplus refuses real-person likenesses, so a likeness-policy failure then does not fall back. The gap is roughly 2x but not exactly: the Seedance 2.0 1080p pair is 38/69, or about 1.82x. If Byteplus accepts the task and later rejects the generated output, VideoDraft refunds the failure but does not resubmit it to Fal. Turn the option off only for the cheapest run when nothing in the job is a real identifiable person.\\n- Grok Imagine images: `grok-imagine` is a flat 2 cr (3 with a reference). `grok-imagine-2.0` is a separate, newer model priced by resolution and quality: 1K 4 (low) / 6 (medium), 2K 6 / 8, plus 1 cr per reference image (up to 3). v1 is NOT superseded \u2014 pick it when cost matters more than 2K.\\n- xAI bills refused requests, so failed Grok generations are not refunded.\\n- Upscales: priced by output size, mode, and (video) duration/fps; see the Upscaling section above.\\n\\nQuote before spending:\\n\\n```bash\\nvideodraft costs gemini-omni-1.1-flash --type video --duration 8 --resolution 720p --audio\\nvideodraft costs topaz-upscale-video --type video --duration 10 --width 1920 --height 1080 --resolution 4k --mode generative\\nvideodraft costs topaz-upscale --type image --width 2048 --height 2048 --scale 2x\\nvideodraft costs topaz-interpolate-video --type video --duration 10 --width 1920 --height 1080 --fps 60 --topaz-model Apollo\\nvideodraft costs minimax-h3 --type video --duration 10 --resolution 2K --ref-images 7\\nvideodraft costs minimax-h3-max --type video --duration 10 --resolution 768p --ref-images 2\\nvideodraft costs grok-imagine-video-1.5 --type video --duration 8 --resolution 720p --ref-images 4\\nvideodraft costs seedance-2 --type video --duration 15 --resolution 720p --quality standard --audio\\nvideodraft costs grok-imagine-2.0 --type image --resolution 2K --quality medium --num 2\\nvideodraft costs seedance-2 --type video --duration 15 --resolution 720p --quality standard --audio --no-allow-real-people # lower Byteplus-only rate\\nvideodraft costs elevenlabs-dubbing --type audio --duration 60\\nvideodraft costs seed-audio-1.0 --type audio --duration 60 # scenario only; model controls actual length\\nvideodraft costs elevenlabs-dialogue --type audio --chars 350\\nvideodraft costs voiceover --type audio --chars 800 # TTS: 10 cr / 1000 chars\\nvideodraft costs voiceover --type audio --chars 800 --mode turbo # ElevenLabs Turbo: 5 cr / 1000 chars\\nvideodraft generate video \\"...\\" --model gemini-omni-1.1-flash --estimate # same quote, inline\\nvideodraft generate video \\"...\\" --model minimax-h3-max --duration 8 --resolution 768p --prompt-expansion-mode balanced\\n```\\n","references/pipeline.md":"# VideoDraft pipeline reference\\n\\nEverything here describes the hosted fallback pipeline through the CLI (`videodraft <command>` / `videodraft call <tool>`) or hosted MCP connector (tool names in backticks). When the local `videodraft_editor` MCP is available, do not use hosted production or export by default. Use hosted tools only for asset generation and optional script/storyboard stages, then import the results and finish with the native editor reference linked from SKILL.md. Continue through `produce_project` and `export_video` only when the user explicitly requests a hosted web production or the native editor is unavailable.\\n\\nUse direct asset tools for standalone images, clips, audio, upscales, and descriptions. Use a hosted project when the user explicitly wants the editable web project, when a hosted storyboard stage is useful, or when the native editor is unavailable. Script-only uses a script-stage project and stops at the script. In VideoDraft ADE with editor tools present, stop before hosted production, import the generated assets, and build/export the native project.\\n\\n## Stages and their tools\\n\\n| Stage | CLI | Underlying tool |\\n| ---------------------------------------- | --------------------------------------------------------------------------- | ------------------------------------------- |\\n| Idea \u2192 full storyboard project | `videodraft create \\"<idea>\\"` | `generate_storyboard_from_idea` |\\n| Idea \u2192 script only (stop there) | `videodraft create \\"<idea>\\" --script-only` | `generate_script_from_idea` |\\n| Footage IS the video | `videodraft call generate_storyboard_from_media` | `generate_storyboard_from_media` |\\n| Batch shot images | `videodraft shots <project>` | `generate_shot_images` |\\n| One shot image | `videodraft generate image --project <id> --scene N --shot M` | `generate_image` |\\n| Produce (voiceover, captions, timeline) | `videodraft produce <project>` | `produce_project` |\\n| Seedance full-video production | `videodraft produce <project> --mode full_video` | `produce_project` with `mode: \\"full_video\\"` |\\n| Per-shot motion prompts | `videodraft video-prompts <project>` | `generate_video_prompts` |\\n| Motion clip for a shot | `videodraft generate video --project <id>` | `generate_video` |\\n| Attach a finished clip to the timeline | `videodraft attach <project> --scene N --shot M --media <url> --type video` | `attach_media_to_shot` |\\n| Find free b-roll (no credits) | `videodraft stock search \\"<query>\\"` | `search_stock_media` |\\n| Copy stock media onto the CDN | `videodraft stock import <ref>` | `import_stock_media` |\\n| Background music | `videodraft generate music --attach <project>` | `generate_music` / `set_background_music` |\\n| Song with vocals, lyrics or sections | `videodraft generate music --model elevenlabs-music-v2.5 --section \\"...\\"` | `generate_music` with `composition_plan` |\\n| General or reference-driven audio | `videodraft generate audio \\"...\\"` | `generate_audio` |\\n| Sound effect | `videodraft generate sound-effect \\"...\\"` | `generate_sound_effect` |\\n| Dialogue audio | `videodraft generate dialogue --line \\"voice:text\\"` | `generate_dialogue` |\\n| Voice changer | `videodraft generate voice-changer <audio>` | `change_voice` |\\n| Dubbing | `videodraft generate dub <audio_or_video>` | `dub_media` |\\n| Scene voiceover | `videodraft generate voiceover --project <id> --scene N` | `generate_voiceover` |\\n| Avatar script | `videodraft avatar script \\"<idea>\\"` | `generate_avatar_script` |\\n| Avatar + speech | `videodraft avatar create <portrait> --script \\"...\\"` | `create_avatar_video` |\\n| Talking-head render | `videodraft avatar render <avatar_video_id>` | `render_avatar_video` + `get_avatar_video` |\\n| Direct portrait + text/audio | `videodraft avatar fabric <portrait> --text \\"...\\"` or `--audio <file>` | `generate_veed_fabric_video` |\\n| Short portrait clip from audio, up to 2K | `videodraft avatar h3-lipsync <portrait> --audio <file>` | `generate_minimax_h3_lipsync_video` |\\n| Existing video + replacement audio | `videodraft avatar lipsync <video> --audio <file>` | `generate_sync_lipsync_video` |\\n| Existing-video AI edit | `videodraft edit video <video> \\"<change>\\" --model <video-edit-model>` | `edit_video` |\\n| Motion transfer | `videodraft edit motion <image> [direction] --motion-video <video>` | `generate_motion_control_video` |\\n| Image enhancement/upscale | `videodraft upscale image <image>` | `upscale_image` |\\n| Video enhancement/upscale | `videodraft upscale video <video>` | `upscale_video` |\\n| Frame rate / slow motion | `videodraft interpolate <video> --fps 60 [--slowdown 4]` | `interpolate_video` |\\n| Final MP4 | `videodraft export <project>` | `export_video` + `check_export_status` |\\n\\n## Rules that prevent broken results\\n\\n- **The storyboard is generated FROM the script**, never from the raw idea. `videodraft create` runs the whole chain correctly. Don\'t call `generate_storyboard_scenes` with a raw idea as the \\"script\\".\\n- **Visual consistency**: never generate a storyboard shot in isolation. Shot prompts carry `[[asset:Name]]` / `[[shot:X-Y]]` tags that `generate_shot_images` resolves against the project\'s visual assets and prior shots. When generating a single shot whose prompt has no tags, pass `--ref` images yourself (the project\'s visual assets and/or the previous shot\'s image; `projects get` exposes both). For scenes with multiple shots or recurring characters, prefer `videodraft shots <project> --model <selected-image-model> --grid`: preserve an explicitly requested compatible image model, otherwise use `nano-banana-2`. It creates one coherent scene grid, then decodes it into individual shot images.\\n- **Reference-first video**: when identity, styling, or composition matters, do not generate each motion clip from text alone. Generate or select the shot still first, then pass the decoded shot image as `--start-image` or `--ref` to the selected video model. AI Production already composes scene grids and sends them to Seedance as references. If the user explicitly requests another compatible video model, bypass fixed Seedance full-video mode and generate the per-shot clips with the requested model, using the individual decoded shot images as anchors.\\n- **Seedance full-video real people**: hosted `full_video` allows real people by default, which applies Fal-tier pricing to every submitted segment and permits the Byteplus-to-Fal fallback. For the lower Byteplus-only rate when no scene grid shows a real identifiable person (non-people, anime, clearly synthetic or stylized characters), use `videodraft produce <project> --mode full_video --no-allow-real-people`, or MCP `produce_project` with `mode: \\"full_video\\", allow_real_people: false`. If a partial opted-out run returns `SEEDANCE_REAL_PERSON_OPT_IN_REQUIRED`, re-estimate, follow the user\'s spend-confirmation preference, and rerun the same project once with an explicit `--allow-real-people` / `allow_real_people: true`. The server reconciles asynchronous results first, preserves running/completed jobs, and retries only failed placeholders carrying that exact code. Do not loop when it was already on. VideoDraft refunds a Byteplus task rejected after asynchronous acceptance, but cannot reroute it; rephrase or change the scene references instead.\\n- **Narration voice mode**: ElevenLabs narration runs on Eleven v4. Standard is the default and the best quality, at 10 credits per 1000 characters. `videodraft produce <project> --voice-mode turbo` (MCP `produce_project` with `voice_mode: \\"turbo\\"`) uses Eleven v4 Turbo, which is faster at 5 credits per 1000 characters. For one scene, `videodraft generate voiceover --project <id> --scene N --mode turbo` (MCP `generate_voiceover` with `mode: \\"turbo\\"`) does the same. Google, OpenAI and cloned `custom-*` voices ignore the mode.\\n- **Hold off generating shot images while the user is still iterating** on storyboard structure.\\n- **produce \u2192 export ordering**: `export` requires a produced project where every production scene has timeline media. If `produce` returns `generating_shot_images`, poll the job ids it returns, then re-run produce.\\n- **Do not attach motion clips before production exists**: run `produce` successfully first, then attach finished motion clips to the production timeline. Attaching before `production_data` exists cannot place them in the final timeline.\\n- **Generated motion clips do not auto-attach**: after `generate video` completes, attach the clip with `attach_media_to_shot` (`media_type:\\"video\\"`, include `duration_seconds`) \u2014 it replaces the production timeline clip while keeping the storyboard still.\\n- **Talking heads use dedicated avatar tools**: do not use `generate video`. Use managed `avatar create` and `avatar render` for reusable avatars, direct `avatar fabric` for a portrait plus text/audio, `avatar h3-lipsync` for a short (5-14.8s) portrait clip from existing audio at up to 2K, and `avatar lipsync` for an existing video plus replacement audio. Reuse a supplied person image or generate a clear front-facing portrait with the explicitly requested compatible image model, otherwise Nano Banana 2. Managed avatar creation and speech are bundled/free; direct Fabric, H3 Max Lip Sync, Sync, and render are paid.\\n- **Existing-video edits use their own category**: call `edit_video` or `videodraft edit video` with a `video_edit` model when transforming the source itself. Kling O3 also has a reference-generation mode that creates a new guided clip. Wan 3.0 is a text/frame/reference generation model, not a source-video editor. Motion transfer similarly uses `generate_motion_control_video` or `videodraft edit motion` with a `motion_control` model.\\n- **Upscaling preserves rather than redesigns**: use Topaz when resolution, detail, or cleanup is the problem (generative mode by default: Wonder 3.5 / Starlight Precise 2.6, Topaz\'s pick for AI sources; precision for real photos/footage or the cheapest pass; creative only for an artistic reinterpretation; `interpolate` for frame rate or slow motion). Regenerate or edit when the subject, text, framing, continuity, or motion is wrong. Upscale a low-quality avatar portrait before Fabric; do not render a new avatar at 480p just to upscale the result.\\n- **Timeouts on the one-shot create**: if `create` times out at the transport layer, the project was still created server-side \u2014 `videodraft projects list`, take the most recent, and resume with its id. Don\'t start a duplicate.\\n\\n## User-attached media: classify roles first\\n\\nFor EACH attached file decide:\\n\\n- **visual_asset** \u2014 recurring reference (character / product / location / style). Upload, then pass in `visual_assets` of `generate_storyboard_from_idea` (via `videodraft call`), or add to an existing project with `add_visual_assets`. Type must be one of `character | object | location | style | custom` with a short name + concrete description.\\n- **shot** \u2014 the media IS footage for the video. Whole video = footage \u2192 `generate_storyboard_from_media`. Idea + footage \u2192 `generate_storyboard_from_idea` with `shot_media`. Existing storyboard \u2192 `attach_media_to_shots`.\\n- **reference** \u2014 inspiration only \u2192 fold a description into the idea/instructions; don\'t place it as a shot or asset.\\n\\nAmbiguous (e.g. a person holding a product)? Ask the user.\\n\\nUploads persist in the media library \u2014 recall later with `videodraft media list`.\\n\\n## Editing project data safely\\n\\n1. `videodraft call get_project_schema` \u2014 read the structure once per session.\\n2. `videodraft projects get <id> --raw` \u2014 the exact editable blob.\\n3. Modify; then `videodraft call update_project --stdin` with `{\\"project_id\\": \\"...\\", \\"data\\": {...}}`.\\n - Objects deep-merge key-by-key; **arrays replace wholesale** \u2014 send the complete array you\'re changing (e.g. all of `storyboard.scenes`).\\n - Scene shot arrays (`image_prompt` / `shot_types` / `shot_actions` / `search_prompt` / `preview_media`) are auto-aligned; fix-ups come back as warnings.\\n4. Snapshot before risky edits: `videodraft checkpoint create <id> --name \\"before re-script\\"`. Restore with `videodraft checkpoint restore <id> <version>`.\\n\\n## AI Studio sessions (standalone generations)\\n\\nProject generations group automatically, and standalone work is grouped per conversation (MCP) or per working directory (CLI) by the connection session \u2014 see the \\"AI Studio sessions\\" section of SKILL.md. Once the task is clear, give that automatic session a concise 3-6 word title before the first generation:\\n\\n```bash\\nvideodraft sessions name \\"Fox Brand Explorations\\"\\nvideodraft generate image \\"...\\"\\n```\\n\\nNaming creates the session with that title. If generation, a user, or an earlier agent created it first, the existing name is preserved. The command refuses to run while `VIDEODRAFT_SESSION` is set, because that override would send later generations to a different session. Use `sessions create` plus `--session` only to continue or deliberately create a separate group.\\n"}');
7612
7761
  }
7613
7762
  const root = bundledSkillDir();
7614
7763
  const files = {};
@@ -7981,6 +8130,14 @@ function registerEditCommands(program) {
7981
8130
  edit.command("video <video_url_or_file> <prompt...>").description("Edit an existing video with a dedicated video-edit model").option(
7982
8131
  "--model <id>",
7983
8132
  "gemini-omni-1.1-flash (preferred, auto-selected for sources up to 10s) | grok-imagine-video-edit | kling-o3-video-ref-edit | happy-horse-video-edit"
8133
+ ).option(
8134
+ "--element <json|@file>",
8135
+ "Kling O3 image-only element (repeatable)",
8136
+ collect2,
8137
+ []
8138
+ ).option("--seed <n>", "Happy Horse Video Edit reproducibility seed").option(
8139
+ "--safety-checker <true|false>",
8140
+ "Happy Horse Video Edit provider safety checker"
7984
8141
  ).option("--ref <url|file>", "reference image (repeatable)", collect2, []).option(
7985
8142
  "--ref-video <url|file>",
7986
8143
  "deprecated: Gemini Omni 1.1 Flash edits accept exactly one input video, so a creative reference video cannot accompany the source",
@@ -8002,6 +8159,24 @@ function registerEditCommands(program) {
8002
8159
  const ctx = buildContext(this);
8003
8160
  const opts = this.opts();
8004
8161
  const model = validateEditModel(opts.model);
8162
+ const rawElements = parseKlingElements(opts.element ?? []);
8163
+ if (rawElements.length && (model !== "kling-o3-video-ref-edit" || rawElements.some((e) => e.video_url || e.voice_id)))
8164
+ throw new UsageError(
8165
+ "--element requires Kling O3 Video Ref/Edit and image-only elements without voice_id."
8166
+ );
8167
+ if (rawElements.length + (opts.ref?.length ?? 0) > 4 && model === "kling-o3-video-ref-edit")
8168
+ throw new UsageError(
8169
+ "Kling O3 accepts at most 4 combined elements and reference images."
8170
+ );
8171
+ const editSeed = opts.seed === void 0 ? void 0 : Number(opts.seed);
8172
+ if (editSeed !== void 0 && (model !== "happy-horse-video-edit" || !Number.isSafeInteger(editSeed)))
8173
+ throw new UsageError(
8174
+ "--seed requires Happy Horse Video Edit and a safe integer."
8175
+ );
8176
+ if (opts.safetyChecker !== void 0 && (model !== "happy-horse-video-edit" || !["true", "false"].includes(opts.safetyChecker)))
8177
+ throw new UsageError(
8178
+ "--safety-checker requires Happy Horse Video Edit and true or false."
8179
+ );
8005
8180
  const duration = positiveNumber2(opts.duration, "--duration");
8006
8181
  const sceneIndex = nonNegativeInteger(opts.scene, "--scene");
8007
8182
  const shotIndex = nonNegativeInteger(opts.shot, "--shot");
@@ -8082,10 +8257,11 @@ function registerEditCommands(program) {
8082
8257
  "--duration is estimate-only for video edits. Submitted edits follow the source and model duration."
8083
8258
  );
8084
8259
  }
8085
- const [[videoUrl], referenceImages, referenceVideos] = await Promise.all([
8260
+ const [[videoUrl], referenceImages, referenceVideos, elements] = await Promise.all([
8086
8261
  resolveRefs(ctx, [videoSource]),
8087
8262
  resolveRefs(ctx, opts.ref ?? []),
8088
- resolveRefs(ctx, opts.refVideo ?? [])
8263
+ resolveRefs(ctx, opts.refVideo ?? []),
8264
+ resolveKlingElements(ctx, rawElements)
8089
8265
  ]);
8090
8266
  capture("cli_edit", {
8091
8267
  kind: "video",
@@ -8095,6 +8271,9 @@ function registerEditCommands(program) {
8095
8271
  const submitted = await ctx.client.callTool(
8096
8272
  "edit_video",
8097
8273
  compact({
8274
+ ...elements.length ? { elements } : {},
8275
+ seed: editSeed,
8276
+ enable_safety_checker: opts.safetyChecker === void 0 ? void 0 : opts.safetyChecker === "true",
8098
8277
  model,
8099
8278
  prompt: promptWords.join(" ").trim(),
8100
8279
  video_url: videoUrl,