venice-video-harness 2.18.0 → 2.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/skills/venice-video-model-routing/SKILL.md +66 -0
- package/AGENTS.md +42 -0
- package/CHANGELOG.md +106 -0
- package/README.md +91 -1
- package/capabilities.json +202 -2
- package/dist/mini-drama/choices.d.ts.map +1 -1
- package/dist/mini-drama/choices.js +2 -0
- package/dist/mini-drama/choices.js.map +1 -1
- package/dist/mini-drama/cli.d.ts.map +1 -1
- package/dist/mini-drama/cli.js +165 -3
- package/dist/mini-drama/cli.js.map +1 -1
- package/dist/mini-drama/loop-engine.d.ts +187 -0
- package/dist/mini-drama/loop-engine.d.ts.map +1 -0
- package/dist/mini-drama/loop-engine.js +636 -0
- package/dist/mini-drama/loop-engine.js.map +1 -0
- package/dist/mini-drama/montage.d.ts.map +1 -1
- package/dist/mini-drama/montage.js +4 -3
- package/dist/mini-drama/montage.js.map +1 -1
- package/dist/mini-drama/prompt-builder.d.ts +0 -11
- package/dist/mini-drama/prompt-builder.d.ts.map +1 -1
- package/dist/mini-drama/prompt-builder.js +72 -12
- package/dist/mini-drama/prompt-builder.js.map +1 -1
- package/dist/mini-drama/video-generator.d.ts +88 -1
- package/dist/mini-drama/video-generator.d.ts.map +1 -1
- package/dist/mini-drama/video-generator.js +63 -28
- package/dist/mini-drama/video-generator.js.map +1 -1
- package/dist/series/types.d.ts +32 -1
- package/dist/series/types.d.ts.map +1 -1
- package/dist/series/types.js +65 -3
- package/dist/series/types.js.map +1 -1
- package/dist/venice/models.d.ts +22 -0
- package/dist/venice/models.d.ts.map +1 -1
- package/dist/venice/models.js +85 -0
- package/dist/venice/models.js.map +1 -1
- package/dist/web/jobs.js +1 -1
- package/dist/web/jobs.js.map +1 -1
- package/dist/web/server.d.ts +19 -0
- package/dist/web/server.d.ts.map +1 -1
- package/dist/web/server.js +73 -1
- package/dist/web/server.js.map +1 -1
- package/dist/web/state.d.ts +3 -0
- package/dist/web/state.d.ts.map +1 -1
- package/dist/web/state.js +2 -0
- package/dist/web/state.js.map +1 -1
- package/dist/web/ui/dist/assets/index-BM-V1s_N.css +1 -0
- package/dist/web/ui/dist/assets/index-BQPyylPl.js +41 -0
- package/dist/web/ui/dist/assets/index-BbATN-Iz.js +41 -0
- package/dist/web/ui/dist/assets/index-CGIShJAr.js +41 -0
- package/dist/web/ui/dist/assets/index-CbErOQAn.js +41 -0
- package/dist/web/ui/dist/index.html +2 -2
- package/package.json +1 -1
|
@@ -318,6 +318,64 @@ Generate consistent storyboard panels using two Venice models in sequence:
|
|
|
318
318
|
|
|
319
319
|
Construct prompts differently depending on the resolved model's capabilities:
|
|
320
320
|
|
|
321
|
+
### Simple-Prompt Models (MiniMax H3 Max, H3 Max Turbo)
|
|
322
|
+
|
|
323
|
+
Read this before anything else in this section: everything below assumes a model that
|
|
324
|
+
renders what it is told and drifts when it is not. The H3 Max pair is the opposite, and
|
|
325
|
+
the registry marks it with `promptStyle: 'simple'` (`modelWantsSimplePrompt(modelId)`).
|
|
326
|
+
These models compose their own coverage — framing, cutting, beat rhythm — from one plain
|
|
327
|
+
statement of intent, and the directorial stack fights the shot the model would otherwise
|
|
328
|
+
have chosen.
|
|
329
|
+
|
|
330
|
+
`buildVideoPrompt()` and `buildMontagePrompt()` already drop the heavy blocks for them, so
|
|
331
|
+
the adaptation is mostly a matter of not writing them back in by hand:
|
|
332
|
+
|
|
333
|
+
- **Dropped:** the authored `Blocking:` restatement, the location description and
|
|
334
|
+
`spatialAnchors` "Fixed layout (never rearrange)" line, the geography-hold /
|
|
335
|
+
no-mirroring lecture, and the full aesthetic string (the compact one is used instead).
|
|
336
|
+
- **Kept:** `@ImageN` identity declarations and role clauses on the R2V lane, the beat
|
|
337
|
+
description, dialogue with delivery, and the audio-exclusion suffix. Identity and look
|
|
338
|
+
still have to be bound; only the staging instructions go.
|
|
339
|
+
- **Dialogue is IMPROVISED, not scripted.** These models carry natural, continuous speech
|
|
340
|
+
across a whole generation and degrade when handed an exact line to recite. So in
|
|
341
|
+
native-dialogue mode the prompt gives the scripted line as INTENT — `[@Image1, voice,
|
|
342
|
+
delivery] conveys: "…"` plus a one-line "improvise naturally in character, keep the intent
|
|
343
|
+
and tone, don't recite word for word" note — instead of the directorial `[…]: "exact line"`
|
|
344
|
+
quote. `buildVideoPrompt` / `buildMontagePrompt` do this automatically via
|
|
345
|
+
`shouldImproviseDialogue(modelId, series)` (gated on `modelWantsSimplePrompt` +
|
|
346
|
+
`audioStrategy !== 'lip-sync'`). Do NOT hand-write exact quotes for these models expecting
|
|
347
|
+
them verbatim. The exception is **exact-lip-sync**, where the `audio_url` drives the exact
|
|
348
|
+
words, so the line stays verbatim. One consequence: captions must come from transcribing
|
|
349
|
+
the rendered audio, not from `script.json` (the model won't say it word for word).
|
|
350
|
+
- **Don't** add camera terms, shot lists, or `Lens switch.` lines per beat. State the
|
|
351
|
+
sequence in a sentence or two and let the model cut it — that instinct is why this is
|
|
352
|
+
the montage family.
|
|
353
|
+
- Resolution is pinned to **768P** by default (2K is a hard 400 — the inverse of plain
|
|
354
|
+
MiniMax H3, which is 2K-only). **480P is the draft tier** and is only selected via an
|
|
355
|
+
explicit `resolution` override on `renderVideoFile`. Durations run the 5-15s ladder.
|
|
356
|
+
`audio` is not configurable, so the field is omitted from the body entirely.
|
|
357
|
+
- Turbo ships **no R2V lane**; identity and lip-sync shots cross to
|
|
358
|
+
`minimax-h3-max-reference-to-video`.
|
|
359
|
+
|
|
360
|
+
**Loop mode uses this family, with two modes.** `venice-video loop` renders the whole shot
|
|
361
|
+
script continuously and plays it as a live browser loop that hot-swaps in fresh takes:
|
|
362
|
+
|
|
363
|
+
- `--mode watch` (default) — `minimax-h3-max-turbo-text-to-video` (or `-image-to-video` off
|
|
364
|
+
an existing panel) at 480P, ~$0.012/s. The cheapest lane; a disposable draft that does NOT
|
|
365
|
+
lock identity (Turbo has no R2V lane).
|
|
366
|
+
- `--mode create` — the real reference-first routing on the **non-Turbo** H3 Max family at
|
|
367
|
+
768P (~$0.024/s): character shots on `minimax-h3-max-reference-to-video` with the full
|
|
368
|
+
`@Image` reference stack + voice-donor audio (identity locked, takes usable), atmosphere on
|
|
369
|
+
Max i2v/t2v. Shots with missing references degrade to i2v/t2v.
|
|
370
|
+
|
|
371
|
+
Both skip the storyboard/QA gates and write only under `loop/`. See `LoopEngine`
|
|
372
|
+
(`src/mini-drama/loop-engine.ts`) and AGENTS.md rule 58. Watch is a preview, not a production
|
|
373
|
+
path; create is the "keep the good takes" path.
|
|
374
|
+
|
|
375
|
+
The Creator app mirrors this: `VideoModelCapabilities.wantsSimplePrompt(id:)` gates the
|
|
376
|
+
same trimming in `ShotPromptBuilder`, and relaxes the `produce_shots` motion/length gate
|
|
377
|
+
so a correctly short H3 Max prompt isn't rejected as thin.
|
|
378
|
+
|
|
321
379
|
### Image-Tag R2V Models (Seedance 2.0 R2V Enhanced — default for all lanes)
|
|
322
380
|
|
|
323
381
|
- Replace character names in descriptions with `@Image1`, `@Image2` tokens via regex
|
|
@@ -454,6 +512,9 @@ Seedance 2.0 (now the default for both atmosphere and character shots) accepts *
|
|
|
454
512
|
- **Omitting `aspect_ratio` from R2V models:** Both Seedance R2V and Kling O3 R2V require `aspect_ratio`. If omitted, it defaults to `16:9` in code. Always pass `aspect_ratio` explicitly.
|
|
455
513
|
- **Sending `image_references`/`image_1` to `nano-banana-pro`:** Returns 400. The generation model does not accept reference payloads at all.
|
|
456
514
|
- **Sending invalid durations:** Seedance 2.0 accepts 4s/5s/8s/10s/12s/15s. Veo 3.1 accepts 4s/6s/8s. Duration auto-snap corrects this.
|
|
515
|
+
- **Sending `2K` to MiniMax H3 Max, or `768P` to plain MiniMax H3:** The two families share a name and invert on resolution. H3 is 2K-only; H3 Max and H3 Max Turbo top out at 768P and reject 2K. `video-generator.ts` pins each family, and the `-max` branch has to stay above the `minimax-h3` substring match or every H3 Max render 400s.
|
|
516
|
+
- **Reaching for `minimax-h3-max-turbo-reference-to-video`:** It doesn't exist ("Specified model not found"). Turbo has no R2V lane; identity shots route to `minimax-h3-max-reference-to-video`.
|
|
517
|
+
- **Sending `audio` to any MiniMax H3 family model:** `audioConfigurable: false` — audio is always generated and the field must be omitted from the body, not set to `true`.
|
|
457
518
|
- **Reference images below 300x300:** R2V models reject `reference_image_urls` and `elements` images smaller than 300x300 pixels. Never downscale character references below this threshold.
|
|
458
519
|
- **Seedance + non-seedream face images (no longer an issue, 2026-07):** Venice removed the restriction that Seedance 2.0 only accepts face-bearing input images from `seedream-v5-lite` / `seedream-v5-lite-edit`. Any image family now works for face-bearing inputs, so there's nothing to pair, reroute, or launder — the pre-flight gate is a no-op.
|
|
459
520
|
|
|
@@ -471,6 +532,11 @@ Seedance 2.0 (now the default for both atmosphere and character shots) accepts *
|
|
|
471
532
|
- **Sequential action in image descriptions:** Causes comic-panel layouts instead of single frames. Separate the single-frame panel description from the full video action description.
|
|
472
533
|
- **Vague body orientation:** Produces twisted poses. Always specify full-body direction explicitly (e.g., "seen entirely from behind", "facing camera directly").
|
|
473
534
|
|
|
535
|
+
### Prompt-Style Mismatches
|
|
536
|
+
|
|
537
|
+
- **Over-directing a simple-prompt model:** Hand-writing blocking, per-beat camera terms, `Lens switch.` lines, or geography-hold clauses into a MiniMax H3 Max prompt flattens the result — it stages its own coverage and the clauses fight it. The prompt builder strips these automatically; don't re-add them via authored shot fields expecting them to help.
|
|
538
|
+
- **Padding an H3 Max prompt to clear the motion gate:** The Creator's `produce_shots` money gate normally demands 12+ words and motion vocabulary. For simple-prompt models it only asks for a stated subject and setting. Inflating a short, correct prompt to satisfy the old bar is the failure, not the fix.
|
|
539
|
+
|
|
474
540
|
### Style Consistency Failures
|
|
475
541
|
|
|
476
542
|
- **Aesthetic description buried at end of prompt:** The model commits to a rendering style before reaching the style instructions, causing inconsistency between angles/shots. Always front-load style with a `STYLE:` prefix and add a `STYLE REMINDER:` suffix.
|
package/AGENTS.md
CHANGED
|
@@ -123,6 +123,44 @@ The full model registry lives in `src/venice/models.ts` with typed specs for eve
|
|
|
123
123
|
- **R2V is pure-reference-only.** Sending `image_url` (or `end_image_url`) alongside `reference_image_urls` is a hard 400: *"image_url and end_image_url cannot be combined with reference media for this model."* `minimax-h3-reference-to-video` is therefore in `MODELS_USING_IMAGE_TAGS`, which is what puts the generator in pure reference mode. It honors `@ImageN` tags — verified by paid render, both tagged characters landed on their assigned `@Image1` / `@Image2` slots.
|
|
124
124
|
- **Reference aspect influences output orientation, so keep a 16:9 plate in the stack.** With the harness's normal slot plan (1:1 character sheets + the 16:9 storyboard blocking plate) and `aspect_ratio: '16:9'`, a paid render returned a true 2560×1440. But a stack of uniformly portrait references returned 1440×1920 *despite* `aspect_ratio: '16:9'` — the requested ratio did not override them. Character-only H3 shots with no blocking plate are the orientation risk; check the first-frame contact sheet before assembling.
|
|
125
125
|
|
|
126
|
+
**MiniMax H3 Max / H3 Max Turbo (added 2026-09-03):**
|
|
127
|
+
- `minimax-h3-max-text-to-video` / `-image-to-video` / `-reference-to-video`, and
|
|
128
|
+
`minimax-h3-max-turbo-text-to-video` / `-image-to-video`.
|
|
129
|
+
- **Related to MiniMax H3 in name only.** Four differences, each of which costs
|
|
130
|
+
a render if you assume H3 behavior:
|
|
131
|
+
- **768P, and 2K is a hard 400** (`Expected '480P' | '768P'`) — the exact
|
|
132
|
+
inverse of base H3. The generator's resolution pin matches
|
|
133
|
+
`minimax-h3-max` *before* `minimax-h3` for this reason; do not reorder
|
|
134
|
+
those branches. 480P is the draft tier, 768P the finish.
|
|
135
|
+
- **They want plain prompts** (`promptStyle: 'simple'` in the registry).
|
|
136
|
+
These models stage their own framing, coverage, and cutting from a stated
|
|
137
|
+
intent, and the directorial stack overrides that instinct. `buildVideoPrompt`
|
|
138
|
+
and `buildMontagePrompt` drop blocking, the locked location description, and
|
|
139
|
+
the geography-hold paragraphs for them; identity (`@ImageN`), the beat, the
|
|
140
|
+
line, the sound, and a compact look survive. Use `modelWantsSimplePrompt(id)`
|
|
141
|
+
rather than an id check when adding new behavior.
|
|
142
|
+
- **`private` tier** (H3 is `anonymized`), and uncensored. Prompt cap 10000
|
|
143
|
+
chars, though the useful prompt is a couple of sentences.
|
|
144
|
+
- **Price.** $0.024/s for H3 Max and $0.012/s for Turbo at 768P, against
|
|
145
|
+
$0.10/s for base H3 — Turbo is the cheapest lane in the registry, cheap
|
|
146
|
+
enough that a 15s take is disposable: render several and pick.
|
|
147
|
+
- **Best used for montages and single-take storytelling.** This is where the
|
|
148
|
+
simple-prompt instinct pays: describe the sequence and let the model cut it.
|
|
149
|
+
Note the montage window now derives from the montage model's own ladder, so
|
|
150
|
+
H3 Max montages plan at 5-15s rather than Seedance's 30s.
|
|
151
|
+
- Shared with H3: the 5-15s ladder (4s is a hard 400) and native audio that is
|
|
152
|
+
**not** toggleable, so the generator omits the `audio` field entirely.
|
|
153
|
+
- **Turbo has no R2V lane.** `minimax-h3-max-turbo-reference-to-video` is
|
|
154
|
+
"Specified model not found", so the `minimax-h3-max-turbo` family routes
|
|
155
|
+
character-consistency and lip-sync shots to `minimax-h3-max-reference-to-video`.
|
|
156
|
+
- R2V is treated as pure-reference (in `MODELS_USING_IMAGE_TAGS`) like H3 R2V.
|
|
157
|
+
Note the difference from H3: `/video/quote` *accepted* `image_url` alongside
|
|
158
|
+
`reference_image_urls` here, but quote validates less than queue, and
|
|
159
|
+
pure-reference is the right mode regardless — it keeps compositional
|
|
160
|
+
authority with the reference stack and is what makes `@ImageN` resolve.
|
|
161
|
+
- i2v lanes inherit aspect from the start image and expose no `aspect_ratios`;
|
|
162
|
+
t2v and R2V accept `16:9 / 21:9 / 4:3 / 1:1 / 3:4 / 9:16`.
|
|
163
|
+
|
|
126
164
|
**Long Duration:**
|
|
127
165
|
- `longcat-image-to-video` / `longcat-distilled-image-to-video` (up to **30s**, no audio)
|
|
128
166
|
- `ltx-2-fast-image-to-video` / `ltx-2-v2-3-fast-image-to-video` (up to **20s**, up to 4K)
|
|
@@ -440,6 +478,10 @@ Use `POST /video/quote` (via `quoteVideo()`) to estimate costs before committing
|
|
|
440
478
|
|
|
441
479
|
57. **The browser UI is the built-in node web app — default to `venice-video web`.** When the operator asks for a browser UI, a dashboard, or to "open the harness in a browser," start the bundled local web app: `venice-video web` (browser dashboard + a whitelisted command runner over the workspace; binds localhost only, `http://127.0.0.1:3000` by default). Do NOT scaffold a separate/ad-hoc UI or point them at anything else. Per the workspace dev-server rule, kill existing node processes first and run on port 3000. The compiled front-end ships inside the npm package (`dist/web/ui/dist`); `npm run web:build` rebuilds it. Command lives in `src/mini-drama/cli.ts` (`web`), server in `src/web/server.ts`.
|
|
442
480
|
|
|
481
|
+
58. **Loop mode has two modes — a disposable Turbo draft (watch) and an identity-locked Max R2V create — both gate-skipping and money-capped (2026-09-04).** `venice-video loop -p <project> -e <n> [--mode watch|create]` boots the web UI (Loop tab) plus an in-process `LoopEngine` (`src/mini-drama/loop-engine.ts`) that renders every shot into `episodes/episode-NNN/loop/` and keeps regenerating fresh takes so the browser can watch the whole plan on repeat while it evolves. Both modes **bypass the references(only for watch) / storyboard / QA gates** — the only hard precondition is a shot script with ≥1 shot — and write ONLY under `loop/` (`shot-NNN--takeK.mp4` + `loop-manifest.json`), never touching canonical `scene-001/` renders or `series.json`, so a loop can run alongside real production. **The mode is the session's first, REQUIRED decision** — the `loop` command asks "is this for LOOPING (creative flow, lower quality) or PRODUCTION (gather usable shots, higher quality)?" It is a deliberate quality-vs-flow tradeoff, never a silent default: interactive `promptChoice` in a TTY, and a **hard error** in a non-interactive run with no `--mode` (agents MUST pass it). `--mode` accepts natural words — `looping`/`loop`/`fun`/`creative` → watch, `production`/`prod`/`gather` → create (`normalizeLoopMode`). Internally the modes are still `watch`/`create`. **The two modes are the point:** (a) **watch** (enjoyment / creative flow) = MiniMax H3 Max **Turbo** at 480P (~$0.012/s): the first generation is **t2v**, every shot after it **chains i2v off the previous last frame**, and it **NEVER uses R2V** (R2V renders are too slow for a loop, and Turbo has none anyway). It ignores panels and the reference stack — identity is NOT locked; it is a fast fun loop, never production-fidelity. (b) **create** (gather good shots for a project) = the real reference-first routing on the **non-Turbo** H3 Max family at 768P (~$0.024/s): character shots render on `minimax-h3-max-reference-to-video` with the full `@Image` reference stack + voice-donor audio (identity **locked**, takes usable), atmosphere shots on Max i2v/t2v, **each shot rendered independently (chaining OFF by default** — R2V and a start frame can't combine on MiniMax). Create degrades a character shot to i2v/t2v only when its references are missing on disk — so for create, generate character/location references first (rule 54). Both share the render primitive: create mode reuses `resolveShotReferenceInputs` + `ensureVoiceReferenceForShot` (exported from `video-generator.ts`, the SAME resolution `renderSingleShotUnit` uses), so its reference stack can't drift from the real pipeline. Other invariants: (c) **chaining default follows the mode:** watch chains (shot 1 t2v, every later shot i2v off the previous shot's current-take LAST frame via `extractLastFrame`, so the loop plays as one piece); create does NOT chain (each shot is independent R2V — chaining and R2V are mutually exclusive on MiniMax). `--no-chain` forces it off. Chaining uses a lean prompt because the start frame, not a reference stack, drives the render. (d) **Every take renders the model's full length (15s default)** — not the shot's scripted duration — for maximum footage/playback per render; override with `--duration`. (e) **The loop regenerates continuously** and does NOT settle after a fixed number of takes: `--max-takes` is a **ring buffer** (candidate takes kept per shot; older non-current takes are pruned and their files deleted so an infinite run can't fill the disk), NOT a stop condition. It stops only on Pause, `--once` (one pass), or the budget. (f) **Money:** billed at queue time; `--budget` (default $2) pauses the loop, and the UI **Resume** button (or a per-shot regenerate) authorizes another budget's worth via `engine.start()` — so "Start/Resume" always does something. `--unbounded` removes the budget cap (spends until stopped) but keeps the ring buffer. (g) Resolution is reachable because `renderVideoFile` honors an optional `resolution` override validated against the model's `resolutions` (it otherwise force-pins `768P` for every `minimax-h3-max*` id); **watch defaults 480P (the infinite-loop tier), create 768P**. (h) The engine shares the web server's `EventHub` and broadcasts `loop-updated` for instant hot-swap; the manifest (with `mode` + `chain`) also feeds `collectEpisodeState` so a plain `venice-video web` shows the last loop state. When changing loop behavior, keep `loop-engine.ts`, the `loop` command in `cli.ts`, the `/api/projects/:slug/loop/*` endpoints + `WebServerOptions.loop` in `src/web/server.ts`, `resolveShotReferenceInputs`/`extractLastFrame` in `video-generator.ts`, and the `LoopView` UI in sync.
|
|
482
|
+
|
|
483
|
+
59. **MiniMax H3 Max (simple-prompt) models IMPROVISE dialogue — the scripted line is intent, not a script (2026-09-04).** Unlike Seedance/Wan, which render the exact line you give them, the H3 Max family (`promptStyle: 'simple'`) performs markedly better carrying natural, continuous speech across a whole generation than reciting a verbatim quote — a fixed line fights the model the same way the directorial blocks do. So in **native-dialogue mode** the prompt builders render a speaker's line as INTENT — `[@ImageN, voice, delivery] conveys: "…"` plus a one-line "improvise naturally in character, keep the meaning and tone, don't recite word for word" note — instead of the directorial `[…]: "exact line"` quote. This is automatic via `shouldImproviseDialogue(modelId, series)` in `prompt-builder.ts` (`modelWantsSimplePrompt(modelId)` AND `videoDefaults.audioStrategy !== 'lip-sync'`), applied in both `buildVideoPrompt` (singles) and `buildMontagePrompt` (montage) — the two paths the H3 Max family renders through. Directorial models keep the exact quote; the legacy Seedance-native / Kling multi-shot builders (`buildMultiShotPrompt`) are untouched because those lanes are never simple-prompt. **The exception is exact-lip-sync**: there the `audio_url` drives the exact spoken words, so the line stays verbatim (the improv gate excludes `audioStrategy === 'lip-sync'`). **Consequence:** when a simple-prompt model improvises, burned/exported captions must be derived by transcribing the rendered audio (the editing pipeline's `silencedetect`/whisper path), NOT from `script.json` — the model will not say it word for word. When changing this, keep `shouldImproviseDialogue` / `formatDialogueLine` / `IMPROV_DIALOGUE_NOTE` in `prompt-builder.ts` and this rule in sync.
|
|
484
|
+
|
|
443
485
|
## Learned Anti-Patterns (Production Issues Log)
|
|
444
486
|
|
|
445
487
|
Issues discovered during production and their fixes. The agent should internalize these to avoid repeating them.
|
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,111 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 2.19.0 — 2026-09-04
|
|
4
|
+
|
|
5
|
+
### Added
|
|
6
|
+
|
|
7
|
+
- **`venice-video loop` — infinite loop mode (watch + create).** Boots the local
|
|
8
|
+
web UI (a new **Loop** tab) plus an in-process `LoopEngine`
|
|
9
|
+
(`src/mini-drama/loop-engine.ts`) that renders the approved shot script
|
|
10
|
+
continuously and plays the whole plan as a live browser loop, hot-swapping each
|
|
11
|
+
shot in as its take finishes and regenerating fresh takes while it plays. Two
|
|
12
|
+
modes, **chosen by a required decision at session start** — "is this for LOOPING
|
|
13
|
+
(creative flow, lower quality) or PRODUCTION (gather usable shots, higher
|
|
14
|
+
quality)?" — asked interactively in a terminal and a hard error in a
|
|
15
|
+
non-interactive run with no `--mode` (never a silent default). `--mode` accepts
|
|
16
|
+
natural words (`looping`/`loop`/`fun`, `production`/`prod`/`gather`). The modes:
|
|
17
|
+
**watch** (looping) renders
|
|
18
|
+
**MiniMax H3 Max Turbo at 480P** — the first generation is t2v, every later shot
|
|
19
|
+
chains i2v off the previous last frame, and it **never uses R2V** (too slow for
|
|
20
|
+
a loop); identity is NOT locked. **create** renders the **non-Turbo** H3 Max
|
|
21
|
+
family at 768P using the real reference-first routing — character shots on
|
|
22
|
+
`minimax-h3-max-reference-to-video` with the full `@Image` reference stack +
|
|
23
|
+
voice-donor audio (identity locked, takes usable), each shot rendered
|
|
24
|
+
independently, degrading to i2v/t2v when references are missing.
|
|
25
|
+
- **Continuous by default:** the loop regenerates forever (it does not stop
|
|
26
|
+
after N takes); `--max-takes` is a ring buffer (candidate takes kept per shot,
|
|
27
|
+
older ones pruned + deleted), not a stop condition. It stops only on Pause,
|
|
28
|
+
`--once`, or the budget.
|
|
29
|
+
- **Last-frame chaining** defaults to the mode (on for watch, off for create,
|
|
30
|
+
since R2V and a start frame can't combine on MiniMax); `--no-chain` forces it
|
|
31
|
+
off. Shot 1 renders normally, every later chained shot renders i2v off the
|
|
32
|
+
previous shot's last frame so the loop plays as one continuous piece.
|
|
33
|
+
- **Full-length takes:** every generation renders the model max (15s default),
|
|
34
|
+
override with `--duration`.
|
|
35
|
+
- **Budget as pause, not a hard stop:** `--budget` (default $2) pauses the loop;
|
|
36
|
+
the UI **Resume** button (and per-shot regenerate) authorizes another budget's
|
|
37
|
+
worth, so "Start/Resume" always does something. `--unbounded` removes the cap.
|
|
38
|
+
Both modes skip the storyboard/QA gates, write only under
|
|
39
|
+
`episodes/episode-NNN/loop/` (per-take mp4s + `loop-manifest.json`), and never
|
|
40
|
+
touch canonical renders or `series.json`. Resumable across restarts; pins and
|
|
41
|
+
per-shot regenerate from the UI. New
|
|
42
|
+
`/api/projects/:slug/loop/{state,start,stop,pin,regenerate}` endpoints, a shared
|
|
43
|
+
`EventHub` `loop-updated` event, and loop state surfaced through
|
|
44
|
+
`collectEpisodeState`. See AGENTS.md rule 58.
|
|
45
|
+
- **MiniMax H3 Max (simple-prompt) models now improvise dialogue.** In
|
|
46
|
+
native-dialogue mode, `buildVideoPrompt` and `buildMontagePrompt` render a
|
|
47
|
+
speaker's scripted line as INTENT for these models — `[@ImageN, voice,
|
|
48
|
+
delivery] conveys: "…"` plus a "improvise naturally in character, keep the
|
|
49
|
+
meaning and tone, don't recite word for word" note — instead of the
|
|
50
|
+
directorial `[…]: "exact line"` quote. Gated on
|
|
51
|
+
`shouldImproviseDialogue(modelId, series)` (`modelWantsSimplePrompt` AND
|
|
52
|
+
`audioStrategy !== 'lip-sync'`). Seedance/Wan/Kling keep the exact quote, and
|
|
53
|
+
exact-lip-sync keeps the exact line (the `audio_url` drives the words). See
|
|
54
|
+
AGENTS.md rule 59. Note: captions for these shots should come from
|
|
55
|
+
transcribing the rendered audio, not `script.json`.
|
|
56
|
+
- **`resolveShotReferenceInputs` / `ensureVoiceReferenceForShot` exported from
|
|
57
|
+
`video-generator.ts`.** The per-shot reference/scene/voice resolution block was
|
|
58
|
+
extracted from `renderSingleShotUnit` into a shared `resolveShotReferenceInputs`
|
|
59
|
+
helper (behavior-identical; `renderSingleShotUnit` now calls it) so loop create
|
|
60
|
+
mode resolves the exact same reference stack the real pipeline does instead of a
|
|
61
|
+
divergent copy.
|
|
62
|
+
- **`resolution` override on `renderVideoFile`.** Honored only when the model
|
|
63
|
+
lists it (validated against the registry, else the family default applies), so
|
|
64
|
+
loop mode can pin H3 Max Turbo to its 480P draft tier without disturbing the
|
|
65
|
+
`768P`/`2K`/`720p` auto-pins every other path relies on.
|
|
66
|
+
- **MiniMax H3 Max + H3 Max Turbo** (probe-verified 2026-09-03):
|
|
67
|
+
`minimax-h3-max-text-to-video` / `-image-to-video` / `-reference-to-video`
|
|
68
|
+
and `minimax-h3-max-turbo-text-to-video` / `-image-to-video`. 768P/480P,
|
|
69
|
+
5-15s, native non-toggleable audio, `private`, uncensored. $0.024/s and
|
|
70
|
+
$0.012/s at 768P (base H3 is $0.10/s). Turbo ships no R2V lane. Two new
|
|
71
|
+
video families — `minimax-h3-max` and `minimax-h3-max-turbo` — both routing
|
|
72
|
+
identity to `minimax-h3-max-reference-to-video`, which is the only lane in
|
|
73
|
+
the pair with `audio_input: true` and therefore also the family lip-sync model.
|
|
74
|
+
- **`promptStyle` on `VideoModelSpec`, and a simple-prompt path in the prompt
|
|
75
|
+
builder.** H3 Max models are `promptStyle: 'simple'`: they stage their own
|
|
76
|
+
framing, coverage, and cutting from a plain statement of intent, and the
|
|
77
|
+
directorial stack flattens that. `buildVideoPrompt` and `buildMontagePrompt`
|
|
78
|
+
now drop spatial blocking, the locked location description, and the
|
|
79
|
+
geography-hold paragraphs for these models, and use the compact aesthetic
|
|
80
|
+
instead of the full one. Identity declarations, reference role clauses, the
|
|
81
|
+
beat, dialogue, sound, and the hard-cut instruction are kept — those are not
|
|
82
|
+
inferable. Gate new behavior on `modelWantsSimplePrompt(id)`, not id checks.
|
|
83
|
+
Every other family is unchanged (`'directorial'` is the default).
|
|
84
|
+
|
|
85
|
+
### Fixed
|
|
86
|
+
|
|
87
|
+
- **The resolution pin no longer sends H3 Max to 2K.** `renderVideoFile` matched
|
|
88
|
+
`minimax-h3` by substring, so every `minimax-h3-max-*` render would have been
|
|
89
|
+
pinned to `2K` — a hard 400 on those models. The `minimax-h3-max` branch now
|
|
90
|
+
precedes it and pins `768P`.
|
|
91
|
+
- **Montage windows are bounded by the montage model, not a flat 30s.**
|
|
92
|
+
`resolveMontageMaxDurationSec` accepted a `montageModel` and ignored it,
|
|
93
|
+
always returning Seedance 2.5's 30s ceiling. Pointing `montageModel` at any
|
|
94
|
+
shorter-ladder model (H3 Max tops out at 15s) therefore planned 30s units
|
|
95
|
+
that all failed `assertShotDurationsValid` *after* the plan was written. The
|
|
96
|
+
ceiling now comes from the model's `maxDurationSec` and an explicit
|
|
97
|
+
`montageMaxDurationSec` is clamped to it. New `resolveMontageMinDurationSec`
|
|
98
|
+
does the same for the floor, so H3 Max montages respect its 5s ladder start
|
|
99
|
+
instead of Seedance's 4s.
|
|
100
|
+
- **Registry resolution ORDER is now load-bearing, and the Creator app honors
|
|
101
|
+
it.** The app defaults to the first allowed resolution whenever a plan's own
|
|
102
|
+
value isn't offered, and Venice's live `/models` lists H3 Max as
|
|
103
|
+
`["480P", "768P"]` — so the app would have rendered every H3 Max shot at its
|
|
104
|
+
draft tier. H3 Max entries list `['768P', '480P']` deliberately, and the app
|
|
105
|
+
reorders the live list to follow the manifest
|
|
106
|
+
(`VideoModelCapabilities.preferredResolutionOrder`). When editing a
|
|
107
|
+
`resolutions` array, treat position 0 as the default, not as arbitrary.
|
|
108
|
+
|
|
3
109
|
## 2.18.0 — 2026-08-21
|
|
4
110
|
|
|
5
111
|
### Added
|
package/README.md
CHANGED
|
@@ -552,6 +552,8 @@ Live catalog (synced against `GET /api/v1/models?type=video` — 103 entries). F
|
|
|
552
552
|
| **HappyHorse 1.1** | i2v, R2V (up to 9 refs) | t2v | 15s | Yes (joint single-pass, 7-lang phoneme lip-sync) | **#1 blind-preference T2V + I2V** (Alibaba 15B). 3-15s, 720p/1080p, nine aspect ratios. Best for talking characters + multilingual localization; SFW/commercial-leaning. The `happyhorse` video-family now routes here. |
|
|
553
553
|
| **HappyHorse 1.0** | i2v, R2V | t2v | 15s | Yes | Prior line, kept for back-compat. Livelier hand-camera realism / cinematic grain vs Seedance. |
|
|
554
554
|
| **MiniMax H3** | i2v, R2V (up to 9 refs) | t2v | 15s (**5s floor**) | Yes (native stereo, not toggleable) | Open-weight omni-modal model — one net covers T2V/I2V/reference. **2K is the only resolution** (no draft tier) at ~1/3 the per-second cost of other families; 24fps, 2500-char prompts. The `minimax-h3` video-family routes here. Sub-5s durations are a hard 400. |
|
|
555
|
+
| **MiniMax H3 Max** | i2v, R2V (up to 9 refs) | t2v | 15s (**5s floor**) | Yes (native, not toggleable) | **Simple prompts — the model stages its own coverage.** Registry `promptStyle: 'simple'`, so the prompt builder strips blocking, locked location descriptions, and geography-hold clauses; say the intent in a sentence or two. Best for montages and beats where the model telling its own story is the point. **768P max — 2K is a hard 400**, the inverse of base H3 (480P is the draft tier). `private` tier, uncensored, 10000-char prompts. $0.024/s. The `minimax-h3-max` video-family routes here. |
|
|
556
|
+
| **MiniMax H3 Max Turbo** | i2v | t2v | 15s (**5s floor**) | Yes (native, not toggleable) | Same model and constraints at **$0.012/s — the cheapest lane in the registry**, which makes 15s takes cheap enough to render several and pick. **No R2V lane** (`-turbo-reference-to-video` does not exist), so the `minimax-h3-max-turbo` family routes identity shots to `minimax-h3-max-reference-to-video`. |
|
|
555
557
|
| **Wan 3.0** | i2v, R2V (up to 9 refs), Enhanced | t2v | **30s** | Yes (always on, not toggleable) | **Longest shots on Venice** — 5/10/15/20/25/30s at 480p/720p/1080p, five aspect ratios plus adaptive, 5000-char prompts. The `wan-3-0` video-family routes here. No audio input anywhere in the family, so it can't lip-sync to a supplied recording. `*-enhanced-*` variants are beta. |
|
|
556
558
|
| **Wan 2.7** | i2v, R2V, V2V, Spicy | t2v | 15s | Wan i2v has no audio; lip-syncs via `audio_url` input | **The audio-driven fallback for exact lip-sync.** R2V exposes per-element `audio_url` for multi-speaker. Spicy = uncensored i2v variant. Seedance 2.x R2V and MiniMax H3 R2V also accept a top-level `audio_url`, so those families never route here. |
|
|
557
559
|
| **Wan 2.6** | Standard, Flash, R2V | Standard | 15s | Yes (i2v/t2v); R2V capped at 10s | Now has R2V variant with `audio_url` input. 1080p. |
|
|
@@ -930,6 +932,94 @@ a persistent history file; `Ctrl-C` cancels the running command without killing
|
|
|
930
932
|
the session (`Ctrl-D` or `/exit` leaves); `/help`, `/status`, `/jobs`, `/cd`, and
|
|
931
933
|
`/pwd` are shell meta-commands; `!<cmd>` runs something in your system shell.
|
|
932
934
|
|
|
935
|
+
### Loop mode — watch the whole plan while it renders, or iterate on real shots
|
|
936
|
+
|
|
937
|
+
Once a plan exists (an approved shot script), you can play the entire film as a
|
|
938
|
+
live browser loop while the harness renders it, instead of waiting for the full
|
|
939
|
+
gated pipeline:
|
|
940
|
+
|
|
941
|
+
```bash
|
|
942
|
+
venice-video loop -p ~/VeniceVideos/my-film -e 1 # asks the purpose
|
|
943
|
+
venice-video loop -p ~/VeniceVideos/my-film -e 1 --mode looping # or state it
|
|
944
|
+
venice-video loop -p ~/VeniceVideos/my-film -e 1 --mode production
|
|
945
|
+
```
|
|
946
|
+
|
|
947
|
+
Loop mode starts with one **required, deliberate decision** — **is this for
|
|
948
|
+
LOOPING or for PRODUCTION?** — because it is a real quality-vs-flow tradeoff, not
|
|
949
|
+
a default to fall through. In a terminal it asks; non-interactively you must pass
|
|
950
|
+
`--mode` (it errors otherwise). You can state it in plain words —
|
|
951
|
+
`--mode looping` / `loop` / `fun` / `creative`, or `--mode production` / `prod` /
|
|
952
|
+
`gather`:
|
|
953
|
+
|
|
954
|
+
- **Looping — creative flow, lower quality.** The first generation is **t2v**,
|
|
955
|
+
every shot after it **chains i2v off the previous shot's last frame**, and it
|
|
956
|
+
**never uses R2V** (those renders are too slow for a loop). Turbo, 480P, fast.
|
|
957
|
+
Not final-quality; it's for watching and riffing.
|
|
958
|
+
- **Production — gather usable shots, higher quality.** **Max R2V + references**
|
|
959
|
+
at 768P, identity locked, each shot rendered independently. Slower, but the
|
|
960
|
+
takes you pin are keepers.
|
|
961
|
+
|
|
962
|
+
Either way it boots the local web UI, opens the browser to a **Loop** tab, and
|
|
963
|
+
**auto-starts** a background engine that renders each shot into the episode's
|
|
964
|
+
`loop/` directory and **keeps regenerating fresh takes continuously** (it does
|
|
965
|
+
not stop after a fixed number of takes — only a Pause or the budget stops it).
|
|
966
|
+
The plan plays on repeat and each shot hot-swaps in as its take finishes;
|
|
967
|
+
because the render outruns playback, the video keeps evolving. Pin the keepers,
|
|
968
|
+
regenerate the ones you don't, and watch a running spend meter. Both modes
|
|
969
|
+
**skip the storyboard/QA gates** and write **only** under `loop/` — canonical
|
|
970
|
+
`scene-001/shot-NNN.mp4` renders and `series.json` are never touched, so a loop
|
|
971
|
+
can run alongside real production.
|
|
972
|
+
|
|
973
|
+
Two behaviors make the loop play as one continuous piece:
|
|
974
|
+
|
|
975
|
+
- **Last-frame chaining (default on).** Shot 1 renders normally; **every shot
|
|
976
|
+
after the first renders i2v using the previous shot's last frame as its first
|
|
977
|
+
frame**, so the clips flow into each other. Turn it off with `--no-chain` to
|
|
978
|
+
render each shot independently (in create mode that keeps per-shot R2V identity
|
|
979
|
+
locking).
|
|
980
|
+
- **Full-length takes.** Every generation renders the model's full length
|
|
981
|
+
(**15s** by default — MiniMax H3 Max's max), for maximum footage and playback
|
|
982
|
+
per render. Override with `--duration`.
|
|
983
|
+
|
|
984
|
+
Because the engine auto-starts, the **Loop** tab shows **Pause** while it's
|
|
985
|
+
running. It regenerates until you Pause or the budget is reached; when the budget
|
|
986
|
+
is reached it pauses and the button becomes **Resume**, which authorizes another
|
|
987
|
+
budget's worth and continues. (`--max-takes` is a ring buffer — the number of
|
|
988
|
+
candidate takes kept per shot — not a stop condition; older takes are pruned so
|
|
989
|
+
an infinite run can't fill the disk.)
|
|
990
|
+
|
|
991
|
+
The two modes differ in what they render:
|
|
992
|
+
|
|
993
|
+
| Purpose (`--mode`) | Model | Resolution | Identity | Use it to… |
|
|
994
|
+
|---|---|---|---|---|
|
|
995
|
+
| **looping** | MiniMax H3 Max **Turbo** t2v/i2v (~$0.012/s) | 480P | **not** locked (Turbo has no R2V lane) | keep a fast, continuous loop going for creative flow |
|
|
996
|
+
| **production** | MiniMax H3 Max **R2V** for character shots, i2v/t2v otherwise (~$0.024/s) | 768P | **locked** via the project's reference stack | gather real, usable shots and pin keepers |
|
|
997
|
+
|
|
998
|
+
Production mode uses the same reference-first routing as the real pipeline:
|
|
999
|
+
character shots render on `minimax-h3-max-reference-to-video` with the full
|
|
1000
|
+
`@Image` reference stack (character sheets, location angles, blocking plates)
|
|
1001
|
+
plus voice-donor audio, so identity holds. Shots with no references on disk
|
|
1002
|
+
degrade to i2v (off a panel) or t2v, so generate your character/location
|
|
1003
|
+
references first for the full effect.
|
|
1004
|
+
|
|
1005
|
+
Continuous regeneration spends money, so it is capped by default:
|
|
1006
|
+
|
|
1007
|
+
```bash
|
|
1008
|
+
venice-video loop -p <dir> -e 1 \
|
|
1009
|
+
--mode production \ # looping | production (required; also accepts loop/fun, prod/gather)
|
|
1010
|
+
--resolution 768P \ # defaults: 480P (looping) / 768P (production)
|
|
1011
|
+
--duration 15s \ # per-take length, snapped to the 5-15s ladder (default 15s)
|
|
1012
|
+
--budget 2 \ # pause after ~$2; Resume/regenerate authorizes another budget
|
|
1013
|
+
--max-takes 3 \ # candidate takes kept per shot (ring buffer, not a stop)
|
|
1014
|
+
--no-chain \ # render shots independently instead of i2v last-frame chaining
|
|
1015
|
+
--once # or: render one take per shot, then stop
|
|
1016
|
+
# --unbounded # remove the budget cap (spends until you Ctrl-C)
|
|
1017
|
+
```
|
|
1018
|
+
|
|
1019
|
+
The loop is resumable: takes, pins, and spend are recorded in
|
|
1020
|
+
`loop/loop-manifest.json`, so re-running `loop` picks up where it left off.
|
|
1021
|
+
`Ctrl-C` stops the engine and the server.
|
|
1022
|
+
|
|
933
1023
|
### Interrupted renders are resumable
|
|
934
1024
|
|
|
935
1025
|
The harness records each Venice `queue_id` to disk *before* it starts polling, so
|
|
@@ -1018,7 +1108,7 @@ These defaults are overridable per-project via `series.json` → `videoDefaults`
|
|
|
1018
1108
|
|
|
1019
1109
|
### Picking a family at project creation
|
|
1020
1110
|
|
|
1021
|
-
`venice-video new` asks which family to use, and `venice-video new-series` asks too when it's run on a terminal without `--video-family`. Both write the answer to `series.json` → `videoDefaults.videoFamilyPreference` and swap the action / atmosphere / character-consistency models to match. The wizard orders the families as Automatic, Seedance, Wan 3.0, MiniMax H3, HappyHorse, Grok Imagine, then Kling O3.
|
|
1111
|
+
`venice-video new` asks which family to use, and `venice-video new-series` asks too when it's run on a terminal without `--video-family`. Both write the answer to `series.json` → `videoDefaults.videoFamilyPreference` and swap the action / atmosphere / character-consistency models to match. The wizard orders the families as Automatic, Seedance, Wan 3.0, MiniMax H3, MiniMax H3 Max, MiniMax H3 Max Turbo, HappyHorse, Grok Imagine, then Kling O3.
|
|
1022
1112
|
|
|
1023
1113
|
### Choosing dialogue audio
|
|
1024
1114
|
|
package/capabilities.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"schemaVersion": 1,
|
|
3
|
-
"harnessVersion": "2.
|
|
4
|
-
"generatedAt": "2026-08-
|
|
3
|
+
"harnessVersion": "2.19.0",
|
|
4
|
+
"generatedAt": "2026-08-21T10:20:08-03:00",
|
|
5
5
|
"videoModels": [
|
|
6
6
|
{
|
|
7
7
|
"id": "wan-2.6-image-to-video",
|
|
@@ -1767,6 +1767,202 @@
|
|
|
1767
1767
|
"privacy": "anonymized",
|
|
1768
1768
|
"offline": false
|
|
1769
1769
|
},
|
|
1770
|
+
{
|
|
1771
|
+
"id": "minimax-h3-max-text-to-video",
|
|
1772
|
+
"name": "MiniMax H3 Max",
|
|
1773
|
+
"type": "text-to-video",
|
|
1774
|
+
"durations": [
|
|
1775
|
+
"5s",
|
|
1776
|
+
"6s",
|
|
1777
|
+
"7s",
|
|
1778
|
+
"8s",
|
|
1779
|
+
"9s",
|
|
1780
|
+
"10s",
|
|
1781
|
+
"11s",
|
|
1782
|
+
"12s",
|
|
1783
|
+
"13s",
|
|
1784
|
+
"14s",
|
|
1785
|
+
"15s"
|
|
1786
|
+
],
|
|
1787
|
+
"resolutions": [
|
|
1788
|
+
"768P",
|
|
1789
|
+
"480P"
|
|
1790
|
+
],
|
|
1791
|
+
"aspectRatios": [
|
|
1792
|
+
"16:9",
|
|
1793
|
+
"21:9",
|
|
1794
|
+
"4:3",
|
|
1795
|
+
"1:1",
|
|
1796
|
+
"3:4",
|
|
1797
|
+
"9:16"
|
|
1798
|
+
],
|
|
1799
|
+
"audio": true,
|
|
1800
|
+
"audioConfigurable": false,
|
|
1801
|
+
"audioInput": false,
|
|
1802
|
+
"videoInput": false,
|
|
1803
|
+
"supportsElements": false,
|
|
1804
|
+
"supportsReferenceImages": false,
|
|
1805
|
+
"supportsSceneImages": false,
|
|
1806
|
+
"supportsEndImage": false,
|
|
1807
|
+
"maxDurationSec": 15,
|
|
1808
|
+
"promptStyle": "simple",
|
|
1809
|
+
"privacy": "private",
|
|
1810
|
+
"offline": false
|
|
1811
|
+
},
|
|
1812
|
+
{
|
|
1813
|
+
"id": "minimax-h3-max-image-to-video",
|
|
1814
|
+
"name": "MiniMax H3 Max",
|
|
1815
|
+
"type": "image-to-video",
|
|
1816
|
+
"durations": [
|
|
1817
|
+
"5s",
|
|
1818
|
+
"6s",
|
|
1819
|
+
"7s",
|
|
1820
|
+
"8s",
|
|
1821
|
+
"9s",
|
|
1822
|
+
"10s",
|
|
1823
|
+
"11s",
|
|
1824
|
+
"12s",
|
|
1825
|
+
"13s",
|
|
1826
|
+
"14s",
|
|
1827
|
+
"15s"
|
|
1828
|
+
],
|
|
1829
|
+
"resolutions": [
|
|
1830
|
+
"768P",
|
|
1831
|
+
"480P"
|
|
1832
|
+
],
|
|
1833
|
+
"aspectRatios": [],
|
|
1834
|
+
"audio": true,
|
|
1835
|
+
"audioConfigurable": false,
|
|
1836
|
+
"audioInput": false,
|
|
1837
|
+
"videoInput": false,
|
|
1838
|
+
"supportsElements": false,
|
|
1839
|
+
"supportsReferenceImages": false,
|
|
1840
|
+
"supportsSceneImages": false,
|
|
1841
|
+
"supportsEndImage": false,
|
|
1842
|
+
"maxDurationSec": 15,
|
|
1843
|
+
"promptStyle": "simple",
|
|
1844
|
+
"privacy": "private",
|
|
1845
|
+
"offline": false
|
|
1846
|
+
},
|
|
1847
|
+
{
|
|
1848
|
+
"id": "minimax-h3-max-reference-to-video",
|
|
1849
|
+
"name": "MiniMax H3 Max R2V",
|
|
1850
|
+
"type": "image-to-video",
|
|
1851
|
+
"durations": [
|
|
1852
|
+
"5s",
|
|
1853
|
+
"6s",
|
|
1854
|
+
"7s",
|
|
1855
|
+
"8s",
|
|
1856
|
+
"9s",
|
|
1857
|
+
"10s",
|
|
1858
|
+
"11s",
|
|
1859
|
+
"12s",
|
|
1860
|
+
"13s",
|
|
1861
|
+
"14s",
|
|
1862
|
+
"15s"
|
|
1863
|
+
],
|
|
1864
|
+
"resolutions": [
|
|
1865
|
+
"768P",
|
|
1866
|
+
"480P"
|
|
1867
|
+
],
|
|
1868
|
+
"aspectRatios": [
|
|
1869
|
+
"16:9",
|
|
1870
|
+
"21:9",
|
|
1871
|
+
"4:3",
|
|
1872
|
+
"1:1",
|
|
1873
|
+
"3:4",
|
|
1874
|
+
"9:16"
|
|
1875
|
+
],
|
|
1876
|
+
"audio": true,
|
|
1877
|
+
"audioConfigurable": false,
|
|
1878
|
+
"audioInput": true,
|
|
1879
|
+
"videoInput": false,
|
|
1880
|
+
"supportsElements": false,
|
|
1881
|
+
"supportsReferenceImages": true,
|
|
1882
|
+
"supportsSceneImages": false,
|
|
1883
|
+
"supportsEndImage": false,
|
|
1884
|
+
"maxDurationSec": 15,
|
|
1885
|
+
"promptStyle": "simple",
|
|
1886
|
+
"privacy": "private",
|
|
1887
|
+
"offline": false
|
|
1888
|
+
},
|
|
1889
|
+
{
|
|
1890
|
+
"id": "minimax-h3-max-turbo-text-to-video",
|
|
1891
|
+
"name": "MiniMax H3 Max Turbo",
|
|
1892
|
+
"type": "text-to-video",
|
|
1893
|
+
"durations": [
|
|
1894
|
+
"5s",
|
|
1895
|
+
"6s",
|
|
1896
|
+
"7s",
|
|
1897
|
+
"8s",
|
|
1898
|
+
"9s",
|
|
1899
|
+
"10s",
|
|
1900
|
+
"11s",
|
|
1901
|
+
"12s",
|
|
1902
|
+
"13s",
|
|
1903
|
+
"14s",
|
|
1904
|
+
"15s"
|
|
1905
|
+
],
|
|
1906
|
+
"resolutions": [
|
|
1907
|
+
"768P",
|
|
1908
|
+
"480P"
|
|
1909
|
+
],
|
|
1910
|
+
"aspectRatios": [
|
|
1911
|
+
"16:9",
|
|
1912
|
+
"21:9",
|
|
1913
|
+
"4:3",
|
|
1914
|
+
"1:1",
|
|
1915
|
+
"3:4",
|
|
1916
|
+
"9:16"
|
|
1917
|
+
],
|
|
1918
|
+
"audio": true,
|
|
1919
|
+
"audioConfigurable": false,
|
|
1920
|
+
"audioInput": false,
|
|
1921
|
+
"videoInput": false,
|
|
1922
|
+
"supportsElements": false,
|
|
1923
|
+
"supportsReferenceImages": false,
|
|
1924
|
+
"supportsSceneImages": false,
|
|
1925
|
+
"supportsEndImage": false,
|
|
1926
|
+
"maxDurationSec": 15,
|
|
1927
|
+
"promptStyle": "simple",
|
|
1928
|
+
"privacy": "private",
|
|
1929
|
+
"offline": false
|
|
1930
|
+
},
|
|
1931
|
+
{
|
|
1932
|
+
"id": "minimax-h3-max-turbo-image-to-video",
|
|
1933
|
+
"name": "MiniMax H3 Max Turbo",
|
|
1934
|
+
"type": "image-to-video",
|
|
1935
|
+
"durations": [
|
|
1936
|
+
"5s",
|
|
1937
|
+
"6s",
|
|
1938
|
+
"7s",
|
|
1939
|
+
"8s",
|
|
1940
|
+
"9s",
|
|
1941
|
+
"10s",
|
|
1942
|
+
"11s",
|
|
1943
|
+
"12s",
|
|
1944
|
+
"13s",
|
|
1945
|
+
"14s",
|
|
1946
|
+
"15s"
|
|
1947
|
+
],
|
|
1948
|
+
"resolutions": [
|
|
1949
|
+
"768P",
|
|
1950
|
+
"480P"
|
|
1951
|
+
],
|
|
1952
|
+
"aspectRatios": [],
|
|
1953
|
+
"audio": true,
|
|
1954
|
+
"audioConfigurable": false,
|
|
1955
|
+
"audioInput": false,
|
|
1956
|
+
"videoInput": false,
|
|
1957
|
+
"supportsElements": false,
|
|
1958
|
+
"supportsReferenceImages": false,
|
|
1959
|
+
"supportsSceneImages": false,
|
|
1960
|
+
"supportsEndImage": false,
|
|
1961
|
+
"maxDurationSec": 15,
|
|
1962
|
+
"promptStyle": "simple",
|
|
1963
|
+
"privacy": "private",
|
|
1964
|
+
"offline": false
|
|
1965
|
+
},
|
|
1770
1966
|
{
|
|
1771
1967
|
"id": "kling-v3-pro-text-to-video",
|
|
1772
1968
|
"name": "Kling V3 Pro",
|
|
@@ -3466,6 +3662,7 @@
|
|
|
3466
3662
|
"kling-o3-pro-reference-to-video",
|
|
3467
3663
|
"kling-o3-standard-reference-to-video",
|
|
3468
3664
|
"kling-v3-4k-reference-to-video",
|
|
3665
|
+
"minimax-h3-max-reference-to-video",
|
|
3469
3666
|
"minimax-h3-reference-to-video",
|
|
3470
3667
|
"pixverse-c1-reference-to-video",
|
|
3471
3668
|
"seedance-2-0-enhanced-reference-to-video",
|
|
@@ -3503,6 +3700,7 @@
|
|
|
3503
3700
|
"imageTags": [
|
|
3504
3701
|
"grok-imagine-reference-to-video",
|
|
3505
3702
|
"happyhorse-1-1-reference-to-video",
|
|
3703
|
+
"minimax-h3-max-reference-to-video",
|
|
3506
3704
|
"minimax-h3-reference-to-video",
|
|
3507
3705
|
"seedance-2-0-enhanced-reference-to-video",
|
|
3508
3706
|
"seedance-2-0-fast-reference-to-video",
|
|
@@ -3510,6 +3708,7 @@
|
|
|
3510
3708
|
"seedance-2-5-reference-to-video"
|
|
3511
3709
|
],
|
|
3512
3710
|
"audioInput": [
|
|
3711
|
+
"minimax-h3-max-reference-to-video",
|
|
3513
3712
|
"minimax-h3-reference-to-video",
|
|
3514
3713
|
"seedance-2-0-enhanced-reference-to-video",
|
|
3515
3714
|
"seedance-2-0-fast-reference-to-video",
|
|
@@ -3545,6 +3744,7 @@
|
|
|
3545
3744
|
"seedance-2-0-fast-reference-to-video": 9,
|
|
3546
3745
|
"happyhorse-1-1-reference-to-video": 9,
|
|
3547
3746
|
"minimax-h3-reference-to-video": 9,
|
|
3747
|
+
"minimax-h3-max-reference-to-video": 9,
|
|
3548
3748
|
"wan-3-0-reference-to-video": 9,
|
|
3549
3749
|
"wan-3-0-enhanced-reference-to-video": 9
|
|
3550
3750
|
},
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"choices.d.ts","sourceRoot":"","sources":["../../src/mini-drama/choices.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,qBAAqB,EAAE,MAAM,oBAAoB,CAAC;AAGhE,eAAO,MAAM,oBAAoB,EAAE,aAAa,CAAC;IAC/C,KAAK,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,qBAAqB,CAAC;IAAC,WAAW,CAAC,EAAE,MAAM,CAAC;CACnE,
|
|
1
|
+
{"version":3,"file":"choices.d.ts","sourceRoot":"","sources":["../../src/mini-drama/choices.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,qBAAqB,EAAE,MAAM,oBAAoB,CAAC;AAGhE,eAAO,MAAM,oBAAoB,EAAE,aAAa,CAAC;IAC/C,KAAK,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,qBAAqB,CAAC;IAAC,WAAW,CAAC,EAAE,MAAM,CAAC;CACnE,CAUA,CAAC;AAEF;;;;;;;;;;;;;;GAcG;AACH,MAAM,MAAM,WAAW,GAAG,SAAS,GAAG,UAAU,CAAC;AAEjD,eAAO,MAAM,oBAAoB,EAAE,aAAa,CAAC;IAC/C,KAAK,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,WAAW,CAAC;IAAC,WAAW,CAAC,EAAE,MAAM,CAAC;CACzD,CAWA,CAAC;AAEF,eAAO,MAAM,sBAAsB;;;;;;;;;;;;EAIzB,CAAC;AAEX;;;;;;;GAOG;AACH,eAAO,MAAM,oBAAoB,EAAE,aAAa,CAAC;IAC/C,KAAK,EAAE,MAAM,CAAC;IAAC,KAAK,EAAE,MAAM,CAAC;IAAC,WAAW,CAAC,EAAE,MAAM,CAAC;CACpD,CAaG,CAAC"}
|
|
@@ -4,6 +4,8 @@ export const VIDEO_FAMILY_CHOICES = [
|
|
|
4
4
|
{ label: 'Seedance 2.0', value: 'seedance', description: 'Reference-first identity anchoring, native dialogue, 720p drafts, 4-15s' },
|
|
5
5
|
{ label: 'Wan 3.0', value: 'wan-3-0', description: 'Shots up to 30s, 480p drafts through 1080p, native audio always on' },
|
|
6
6
|
{ label: 'MiniMax H3', value: 'minimax-h3', description: 'Open-weight omni-modal, 2K native audio, 5-15s' },
|
|
7
|
+
{ label: 'MiniMax H3 Max', value: 'minimax-h3-max', description: 'Simple prompts, model stages its own coverage; 768P, private, 5-15s' },
|
|
8
|
+
{ label: 'MiniMax H3 Max Turbo', value: 'minimax-h3-max-turbo', description: 'Same at half the price ($0.012/s), no R2V lane — identity crosses to H3 Max R2V' },
|
|
7
9
|
{ label: 'HappyHorse 1.1', value: 'happyhorse', description: 'Native multilingual lip-sync, 720p/1080p, 3-15s' },
|
|
8
10
|
{ label: 'Grok Imagine', value: 'grok-imagine', description: 'Atmosphere-forward look; R2V durations stepped at 5s/8s/10s' },
|
|
9
11
|
{ label: 'Kling O3', value: 'kling-o3', description: 'Stylized and illustrated aesthetics; structured character elements' },
|