@koda-sl/baker-cli 0.123.0-dev.70bf43ce4 → 0.123.0-dev.8e4328629
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +33 -16
- package/dist/{chunk-VSVGPYJK.js → chunk-43KBQLP5.js} +143 -34
- package/dist/chunk-43KBQLP5.js.map +1 -0
- package/dist/cli.js +556 -136
- package/dist/cli.js.map +1 -1
- package/dist/engine/index.d.ts +33 -0
- package/dist/engine/index.js +1 -1
- package/package.json +1 -1
- package/dist/chunk-VSVGPYJK.js.map +0 -1
package/README.md
CHANGED
|
@@ -1142,7 +1142,7 @@ Notes:
|
|
|
1142
1142
|
- All write commands take `--file <json>` payloads; explicit flags override file keys. `baker schema ads.linkedin.campaigns.create` for exact args.
|
|
1143
1143
|
- Money flags (`--bid`, `--daily-budget`, `--total-budget`) require `--currency`.
|
|
1144
1144
|
- Creative media comes from the Baker library (`--image-id`/`--video-id` from `baker images`/`baker videos` — uploaded to LinkedIn at publish) or as LinkedIn URNs (`--image-urn`/`--video-urn`). Formats: `image|video|text|spotlight|follower|document|carousel|conversation|tla|jobs`; complex formats take `--file` with the full content object; conversation ads take `--file` with the message flow (`{message: {subject, body, senderName?, buttons[]}}` — buttons `NESTED` (with `nestedMessage`) or `LANDING_PAGE` (with `landingPageUrl`), ≤25 messages, bodies ≤500 chars, labels ≤25). Limits: headline ≤70, text-ad 25/75, intro soft-truncates at 600 chars. TLA sponsors an existing post via `--post-urn`.
|
|
1145
|
-
- Lead forms are file-first (`lead-forms create --file form.json`). Required: name, headline (≤60), privacyPolicyUrl, questions[] (≤12; playbook: ≤4 for completion). Each question is a predefined profile field (`{ name, predefinedField: "EMAIL" }` — Contact/Work/Company/Education/Demographic library) or a custom question (`{ name, questionType: "SINGLE_LINE_TEXT" | "MULTIPLE_CHOICE", options?: [...] }`; ≤3 custom, MULTIPLE_CHOICE needs 2–30 options). Also supported: `locale {country,language}`, `formImageId`/`formImageUrn` (banner), `
|
|
1145
|
+
- Lead forms are file-first (`lead-forms create --file form.json`). Required: name, headline (≤60), privacyPolicyUrl, questions[] (≤12; playbook: ≤4 for completion). Each question is a predefined profile field (`{ name, predefinedField: "EMAIL" }` — Contact/Work/Company/Education/Demographic library) or a custom question (`{ name, questionType: "SINGLE_LINE_TEXT" | "MULTIPLE_CHOICE", options?: [...] }`; ≤3 custom, MULTIPLE_CHOICE needs 2–30 options). Also supported: `locale {country,language}`, `formImageId`/`formImageUrn` (banner), `consents[]` (≤5 disclosure checkboxes `{text, required}`), `hiddenFields[]` (≤20 `{name, value}` tracking fields), `legalDisclaimer`, `thankYou {message, cta, landingUrl | appointmentUrl}` (Calendly/Chili Piper booking link). The staged preview emits non-blocking best-practice warnings when a form has no qualifying question, no confirmation message/action, or no consent checkbox.
|
|
1146
1146
|
|
|
1147
1147
|
#### `audit` — playbook diagnostic
|
|
1148
1148
|
|
|
@@ -2672,6 +2672,18 @@ baker canvas run my-canvas.json --parallel 8
|
|
|
2672
2672
|
# browsable outputs — pass --no-record too if you want nothing persisted.
|
|
2673
2673
|
baker canvas run my-canvas.json --remote-cache off --no-record
|
|
2674
2674
|
|
|
2675
|
+
# 2d. Regenerate a node whose prompt is fine (force a fresh roll). The engine is
|
|
2676
|
+
# content-addressed: re-running an UNCHANGED node returns the identical cached
|
|
2677
|
+
# render, never a new draw. To re-roll a node without editing its prompt (a
|
|
2678
|
+
# color drifted, a face came out wrong), force it fresh two ways — NEVER
|
|
2679
|
+
# restructure the canvas (repointing output / deleting nodes) to trick the cache:
|
|
2680
|
+
# • One-shot flag — forces the named nodes + everything downstream fresh this
|
|
2681
|
+
# run, leaving every other node cached (unknown ids fail loudly before billing):
|
|
2682
|
+
baker canvas run my-canvas.json --regenerate gen_4x5,gen_9x16
|
|
2683
|
+
# • Persistent — add/bump a node's `regenerate` field in the canvas JSON
|
|
2684
|
+
# (e.g. "regenerate": 2) and re-run; the fresh render is reproducible in any
|
|
2685
|
+
# later session. Bump it again (3, 4, …) for each additional draw.
|
|
2686
|
+
|
|
2675
2687
|
# 3. Inspect a finished run (per-node timing, file list, optional video thumbs)
|
|
2676
2688
|
baker canvas inspect <run_id>
|
|
2677
2689
|
|
|
@@ -3366,9 +3378,9 @@ Accepted ref-image MIMEs vary by model — see per-model sections below.
|
|
|
3366
3378
|
|
|
3367
3379
|
###### Model: `bytedance/seedance-2.0`
|
|
3368
3380
|
|
|
3369
|
-
Production-quality ad-creative model. Routed via **
|
|
3381
|
+
Production-quality ad-creative model. Routed via **Replicate** (`bytedance/seedance-2.0`). NOTE: ByteDance's upstream "real person" likeness filter still blocks photorealistic human reference frames on **any** reseller — the escape is a synthetic/AI presenter face or routing real faces to Veo, not the provider.
|
|
3370
3382
|
|
|
3371
|
-
Ref-image MIMEs: `image/png`, `image/jpeg`, `image/webp
|
|
3383
|
+
Ref-image MIMEs: `image/png`, `image/jpeg`, `image/webp`.
|
|
3372
3384
|
|
|
3373
3385
|
| Name | Type | Required | Notes |
|
|
3374
3386
|
|---|---|---|---|
|
|
@@ -3422,7 +3434,7 @@ Ref-image MIMEs: `image/png`, `image/jpeg`, `image/webp`, `image/gif` (via OpenR
|
|
|
3422
3434
|
>
|
|
3423
3435
|
> A scaffolded canvas carries this table inline at `metadata.todo.model_constraints`.
|
|
3424
3436
|
|
|
3425
|
-
> **Content-policy blocks are deterministic, not flaky.**
|
|
3437
|
+
> **Content-policy blocks are deterministic, not flaky.** ByteDance's Seedance rejects any first/last frame that reads as a real-person likeness (even an AI-generated face) — surfaced as `content_policy_blocked` (HTTP 422, **non-retryable**). This is ByteDance's upstream filter, so it fires on **any** reseller (Replicate or otherwise) — a provider swap does **not** route around it. Retrying **never** succeeds and wastes credits. Fix the cause: use a **synthetic/AI-generated** (non-identifiable) presenter face, or route a real face to **Veo** (`person_generation: allow_adult`), or make the source frame less photorealistic.
|
|
3426
3438
|
|
|
3427
3439
|
---
|
|
3428
3440
|
|
|
@@ -3473,7 +3485,7 @@ None.
|
|
|
3473
3485
|
|
|
3474
3486
|
##### `video_lipsync`
|
|
3475
3487
|
|
|
3476
|
-
Lip-sync a video to an audio track via
|
|
3488
|
+
Lip-sync a video to an audio track via Sync Labs `sync/lipsync-2` (Replicate).
|
|
3477
3489
|
|
|
3478
3490
|
**Inputs**
|
|
3479
3491
|
|
|
@@ -3500,7 +3512,7 @@ Lip-sync a video to an audio track via VEED (fal.ai).
|
|
|
3500
3512
|
|
|
3501
3513
|
##### `video_background_remove`
|
|
3502
3514
|
|
|
3503
|
-
Strip a video's background → alpha WebM
|
|
3515
|
+
Strip a video's background → transparent alpha WebM (VP9) or MOV (ProRes 4444). Powered by `sprited/birefnet-video` (Replicate).
|
|
3504
3516
|
|
|
3505
3517
|
**Inputs**
|
|
3506
3518
|
|
|
@@ -3676,7 +3688,7 @@ Place and mix several audio clips onto one timeline — a music bed plus timed v
|
|
|
3676
3688
|
|
|
3677
3689
|
##### `image_background_remove`
|
|
3678
3690
|
|
|
3679
|
-
Strip background → transparent PNG
|
|
3691
|
+
Strip background → transparent PNG. Powered by `men1scus/birefnet` (Replicate). (Note: mask-only output is no longer produced.)
|
|
3680
3692
|
|
|
3681
3693
|
**Inputs**
|
|
3682
3694
|
|
|
@@ -3927,20 +3939,20 @@ Validate, then execute the graph. Blocks until done. Logs one line per node. Ret
|
|
|
3927
3939
|
|
|
3928
3940
|
#### `baker canvas scaffold-video <video> [flags]`
|
|
3929
3941
|
|
|
3930
|
-
Turn a reference video into a **runnable, self-validated reproduction canvas** in one command — the video counterpart of `scaffold-static-ad`. It runs **billed passes** up front:
|
|
3942
|
+
Turn a reference video into a **runnable, self-validated reproduction canvas** in one command — the video counterpart of `scaffold-static-ad`. The `<video>` positional is a **local path OR an http(s) URL** (a `baker winning-ads` `media_url`, a library URL, any reel link) — a URL is downloaded for you (no manual `curl` first), so pass `--slug`/`--out` with it to give the canvas a home. It runs **billed passes** up front:
|
|
3931
3943
|
|
|
3932
|
-
1. **`video_deconstruct`** (`~google/gemini-pro-latest`, full mode) — reverse-engineers the video into a scene-by-scene blueprint + word-level transcript, written next to the canvas as **`prompt.json
|
|
3944
|
+
1. **`video_deconstruct`** (`~google/gemini-pro-latest`, full mode) — reverse-engineers the video into a scene-by-scene blueprint + word-level transcript, written next to the canvas as **`prompt.json`** (the human-editable source of truth). Each scene's `start_frame_prompt`/`end_frame_prompt` are inlined into the frame nodes (see below); the shared **global style reference** every frame reads via `target_blueprint` is a **slim projection** written alongside as **`prompt.style.json`** (`global` cast/palette/brand + `reference_elements` only, no per-scene array). A 33-scene blueprint is ~200 KB — inlining it into every one of a dozen frame prompts was pure waste and let a frame blend in another scene's content; the slim is ~5 KB. Edit `prompt.json` and re-scaffold to refresh the style projection.
|
|
3933
3945
|
2. **recurring-element selection** (`~google/gemini-flash-latest`) — picks only the **recurring, identity-critical** elements (each `global.cast` person, a recurring animal, a showcased product, the brand logo) and the scene indices each appears in. One real reference image grounds each element across **every** frame it appears in, so the same actor stays consistent the whole video. This selection runs as a **second pass over a slimmed blueprint** (cast/branding + each scene's frame prompts only) — a long ad's full blueprint can exceed the engine's inline-prompt limit, so the heavy per-scene detail (dialogue, overlays, transcript) the selector never reads is dropped before the prompt.
|
|
3934
3946
|
|
|
3935
3947
|
Before the deconstruct it runs a **local shot-cut pass** on the source file with **[PySceneDetect](https://www.scenedetect.com)** (`scenedetect` CLI, `detect-content` — the battle-tested HSV content detector, installed in the canvas sandbox) and passes the cut timestamps as `video_deconstruct`'s `shot_cuts`. The deconstruct snaps its scene boundaries onto those real cuts and **splits any scene that spans one**, so a scene's frames can never straddle a hard cut (the failure where a scene's start frame was the couch and its end frame the b-roll). Two knobs tuned for fast social ads: the content **threshold defaults to 18** (PySceneDetect's own default of 27 misses soft reframes) and the **minimum scene length is dropped to 0.25s** (its default ~0.6s merges away rapid montage flashes) — so super-fast cuts survive and become cheap still-holds downstream. The threshold is **adaptive**: if the first pass looks like a continuous shot shredded into many close micro-cuts (a talking-head selfie's natural motion), it re-runs at PySceneDetect's own default of 27 and **merges the two passes** — the high-threshold set is the base, and the low pass's *isolated* extras (real soft blur-morph transitions that vanish at 27) are added back while clustered extras (motion shred) stay dropped. Pinning **`--shot-threshold N`** disables the re-check (lower = more cuts). The backend snap window is likewise **adaptive** (up to 1s onto an unambiguous nearest cut, shrinking around dense cut pairs so a boundary never jumps past the wrong cut), any scene spanning an interior cut is split, and the residual-sliver coalesce is **cut-aware**: a drift sliver folds backward across its non-cut edge and never re-merges across a real cut. If `scenedetect` is unavailable it warns loudly and degrades to LLM-only boundaries.
|
|
3936
3948
|
|
|
3937
3949
|
A shot longer than the video model's per-clip ceiling (Seedance's 15s, passed as `video_deconstruct`'s `max_clip_s`) is split into equal **continuation sub-scenes** that share their splice boundary exactly — so a long shot is reproduced in **full** (no truncation) and joins seamlessly. Each sub-scene carries `continues_previous`.
|
|
3938
3950
|
|
|
3939
|
-
It then scaffolds the full pipeline like an **editing timeline**: each clip gets a **static-ad-grade start AND end keyframe** (`image_generate`, each with its **own self-contained `params.prompt`** — edit a frame node to change only that frame; `prompt.json` wired as the **
|
|
3951
|
+
It then scaffolds the full pipeline like an **editing timeline**: each clip gets a **static-ad-grade start AND end keyframe** (`image_generate`, each with its **own self-contained `params.prompt`** — edit a frame node to change only that frame; the slim `prompt.style.json` wired as the **shared `target_blueprint`** style reference, plus a per-element reference legend). Each keyframe is **fully recast** to the dropped `el_*` reference images. The original extracted frame is kept LAST as a **pure composition anchor** (framing / camera angle / shot size / pose) whenever identity is safely locked — i.e. a frame with no person/animal, OR every cast member present is **sheet-backed** (a multi-view turnaround owns identity, so the anchor can reproduce the source's framing without dictating the face). Since every base element is now sheet-backed by default, cast frames keep their framing anchor too — this is what reproduces the source's composition (a side-profile stays a side-profile, the camera angle holds scene to scene) instead of drifting to a fresh guess. The anchor's legend forbids taking identity/text/palette from it. It is dropped only when a cast member rests on a weak lone-snapshot reference (e.g. a `same_as` second-look slot), where the original frame could re-leak the source actor. Both keyframes feed `video_generate` (`first_frame`+`last_frame`, so Seedance interpolates real in-shot motion; ultra-detailed motion brief; duration snapped to the nearest allowed clip length). Every keyframe grounds **only on its own extracted frame + `el_*` slots** — no reference to any other generated frame — so all images render **in parallel** (no cascade). Source-frame URLs are **deduped** (each ingested once). `--frames reuse` wires the real source frame straight in.
|
|
3940
3952
|
|
|
3941
3953
|
**Composited scenes (split-screen / picture-in-picture / keyed presenter).** Real ads aren't always one full-frame shot — a frame can be **persistently divided** (b-roll on top, a presenter talking on the bottom) or **layer a presenter** over background footage (boxed in a corner, or green-screen keyed). The deconstruct now reports this per scene as `scene.composition` (`layout: split_screen | pip | keyed_overlay`, with one `region` per stream — each its own clean-plate frame + motion brief, the talking-head region flagged `is_presenter`). The scaffold reproduces a composited scene by building **one clip per region** (`s<i>_r0_*`, `s<i>_r1_*`, …) and compositing them with ffmpeg: a split-screen `vstack`/`hstack` (stack direction read from the region **panels**, so a top/bottom split always stacks vertically), or a picture-in-picture `overlay` of the presenter inset at its corner. A **keyed** presenter is first cut to transparency by `video_background_remove` (`s<i>_key`), then overlaid. The presenter region carries the native lip-synced voice; b-roll/render panels stay silent. To change a layout, edit `composition` in `prompt.json` and re-scaffold, or hand-edit the `s<i>_composite` ffmpeg args. Plain full-frame scenes (the default) are unaffected.
|
|
3942
3954
|
|
|
3943
|
-
**Typed region kinds & real screen surfaces.** Each composition region now carries a `kind` — `camera` (filmed footage, re-generated), `screen_capture` (app/site/document screen recording), `static_graphic` (designed text/graphic panel), or `generated` (3D/motion graphics) — plus an optional `nested` list for video-in-video (a Loom-style camera bubble inside a screen share). `kind` is authoritative for routing (prose keywords remain the fallback for older blueprints): `screen_capture`/`static_graphic` regions are **never generated by the video model** — the scene renders as a clean background plate (its clip prompt is scrubbed of all screen narration and forbids rendering UI) and the real surface is composited on the overlay layer. The route is decided **once per persistent layout run** (consecutive scenes sharing one composition signature), so a layout that runs unbroken across many scenes can't flip between pipelines on wording differences. A persistent surface seeds **ONE grouped stub** in `video-overlay-composition/index.html` spanning its whole window, with a per-scene **state timeline** — build one continuous screen recording/mockup, not one screenshot per scene. A `screen_capture` region also carries `surface_id`: a source video routinely **splices two unrelated screen recordings** under one persistent layout (a live app-processing capture, then an unrelated pre-made demo note) — the deconstruct assigns a stable id while the SAME recording continues and a new one when the on-screen content genuinely changes, so the run splits into **separate stubs** at the splice instead of asking for one screenshot that can't cover both. `baker canvas validate` additionally warns (`VIDEO_UI_IN_PROMPT`) if any clip prompt still narrates a screen surface, and (`VIDEO_BRANDMARK_IN_PROMPT`) if a generate prompt asks the model to paint a brand logo/wordmark (generation garbles marks; source the real one with `baker images logo` and composite it on the overlay layer).
|
|
3955
|
+
**Typed region kinds & real screen surfaces.** Each composition region now carries a `kind` — `camera` (filmed footage, re-generated), `screen_capture` (app/site/document screen recording), `static_graphic` (designed text/graphic panel), or `generated` (3D/motion graphics) — plus an optional `nested` list for video-in-video (a Loom-style camera bubble inside a screen share). `kind` is authoritative for routing (prose keywords remain the fallback for older blueprints): `screen_capture`/`static_graphic` regions are **never generated by the video model** — the scene renders as a clean background plate (its clip prompt is scrubbed of all screen narration and forbids rendering UI) and the real surface is composited on the overlay layer. The route is decided **once per persistent layout run** (consecutive scenes sharing one composition signature), so a layout that runs unbroken across many scenes can't flip between pipelines on wording differences. A persistent surface seeds **ONE grouped stub** in `video-overlay-composition/index.html` spanning its whole window, with a per-scene **state timeline** — build one continuous screen recording/mockup, not one screenshot per scene. A `screen_capture` region also carries `surface_id`: a source video routinely **splices two unrelated screen recordings** under one persistent layout (a live app-processing capture, then an unrelated pre-made demo note) — the deconstruct assigns a stable id while the SAME recording continues and a new one when the on-screen content genuinely changes, so the run splits into **separate stubs** at the splice instead of asking for one screenshot that can't cover both. Full-frame screen scenes reuse the same `surface_id`: consecutive full-frame UI beats of one screen (e.g. a 3-scene import flow) share **ONE `s<i>_screen_ref` ingest** — the operator supplies that screenshot once instead of dropping the same capture into a dozen identical `[TODO]`s (distinct surfaces stay distinct). `baker canvas validate` additionally warns (`VIDEO_UI_IN_PROMPT`) if any clip prompt still narrates a screen surface, and (`VIDEO_BRANDMARK_IN_PROMPT`) if a generate prompt asks the model to paint a brand logo/wordmark (generation garbles marks; source the real one with `baker images logo` and composite it on the overlay layer).
|
|
3944
3956
|
|
|
3945
3957
|
**Designed graphics are rebuilt, not generated.** A `static_graphic` surface (a newspaper-collage panel, a meme card, a marketing composition) seeds a **GRAPHIC PANEL** stub — rebuild it as brand HTML or drop the design asset; it never gets the "screenshot the live page" instruction (there is no live page). A **full-frame** designed-graphic scene (the deconstruct emits one full-frame `static_graphic` region for meme/collage/motion-graphic beats) routes to a real design plate the same way screens do — no `image_generate`/`video_generate` — and dialogue over an all-graphic scene is voiceover by definition (nobody is on screen to lip-sync). A region typed `generated` whose own prose reads like a UI/designed panel is treated as a surface candidate too (the frame-grounded continuity checker delivers the verdict and corrects the kind), so one mistyped kind can't re-open the Seedance-paints-UI hole. Floating FX elements (hearts, sparkles, badges) ride the overlay layer: their narration is **scrubbed from clip briefs** and a categorical no-decorations directive is added, so the model can't bake a second, uneditable copy under the real composited one.
|
|
3946
3958
|
|
|
@@ -3952,13 +3964,13 @@ It then scaffolds the full pipeline like an **editing timeline**: each clip gets
|
|
|
3952
3964
|
|
|
3953
3965
|
**Montage flashes held as stills — unless the picture really moves.** A rapid-cut beat shorter than ~2s with no spoken line is a **flash** — Seedance's shortest clip is 4s, so generating one (then trimming away most of it) burns credits for motion no viewer perceives. The scaffold instead **holds one keyframe as a still** for the scene length (a cheap ffmpeg loop, no billed `video_generate`), same look at a fraction of the cost. The deconstruct now stamps each scene's **`motion_level`** (`static` / `subtle` / `dynamic`): a **dynamic** flash (pouring chocolate, hands working, walking) keeps a **real trimmed clip** — freezing a moving montage turns it into a slideshow — while genuinely static beats (a logo card, a pinned photo, a product still) keep the cheap hold. Talking/ambient beats always keep a real clip (they need motion + native audio). The deconstruct also stamps each dialogue line's **`on_camera`** flag — a voice playing over b-roll, a graphic, or a mere *photo* of the speaker stays voiceover, so the scaffold never lip-syncs a scene with no speaking face (the polaroid close-up failure).
|
|
3954
3966
|
|
|
3955
|
-
**
|
|
3967
|
+
**One clip per shot — separated at complete breaks.** A video is a sequence of clear **shots** with **complete breaks** (hard cuts) between them, and that is what the scaffold separates by: **two adjacent presenter shots at a hard cut become TWO clips**, never glued into one invented take just because the speech runs continuously across the cut. Each presenter shot is one Seedance clip (`s<anchor>_clip`, native lip-sync + audio) re-voiced to the brand voice. What is **NOT** split: a **voiceover** narration stays ONE ElevenLabs `tts` read across the b-roll it plays over, and a **b-roll cutaway** between two on-camera moments leaves the presenter shot continuous — the shown scenes aren't adjacent (the insert sits between them), so the clip covers both on-camera windows (sliced as `s<i>_seg`, an ffmpeg `-ss`/`-t` cut — video+audio from the *same* clip so lip-sync holds) while the cutaway plays its own silent clip over the continuing voice. "Shown" is decided by the **presenter element's per-scene presence**, not just who's speaking — a scene where a cast member narrates over b-roll (their element absent) is a cutaway, so the talking head never appears where the original cut away. A single shot longer than the **gateway-safe ~10s clip ceiling** (Seedance's *API* max is 15s, but the gateway often times out — **HTTP 524** — past ~10s) **splits into contiguous takes joined by a shared boundary frame**; the spine then **seam-dedups** that duplicated frame so the concat doesn't freeze on it (`--seam-dedup head|tail|off`, default `head` = drop the second clip's first frame). Timbre stays consistent across all the separate shot clips because every clip's native audio is re-voiced in **one merged per-speaker pass** (not per clip). A b-roll cutaway *inside* a phrase lands at an **approximate** time (Seedance exposes no word timing) — nudge the scene boundary if it's off its beat.
|
|
3956
3968
|
|
|
3957
3969
|
**A starting point, not a locked render.** The canvas mirrors the reference's structure to give you a faithful scaffold, but `metadata.todo.full_flexibility` makes explicit that the agent has **full editing freedom**: add / delete / reorder / split / merge scenes, re-prompt any frame or motion brief, change a scene's layout (full-frame ↔ composite), or rewrite any line — the content-addressed cache re-bills only what changes, and `baker canvas validate` re-checks timing/lip-sync after any edit.
|
|
3958
3970
|
|
|
3959
3971
|
**Sequenced audio.** Dialogue is a back-and-forth on one absolute timeline, so each **contiguous same-speaker turn** becomes its own `tts` placed at its real `start_s` — turns alternate and never stack (the earlier design concatenated each speaker's whole monologue at their earliest timestamp, so two voices played in parallel for the entire video). Each speaker is locked to one shared `voice_select` voice; a `sound_effect` per SFX and a `music` bed (conditioned on the **ad's own script + emotional arc** so the bed supports the message, styled after the AudD-identified track when available, ducked under the voices, and started at the reference's `music.starts_at_s` rather than always at 0) round out the mix (`audio_timeline`). The final mux normalizes the soundtrack to **−14 LUFS (stereo)** so the output plays loud in every player — the raw mix is quiet mono, which reads as "no sound."
|
|
3960
3972
|
|
|
3961
|
-
**Native talking heads + one voice per person (no post-hoc lip-sync).** Seedance 2.0 generates lip-synced speech **natively** — a presenter phrase puts the full phrase in the clip's prompt with `generate_audio`, so lips and voice are generated together (no `video_lipsync`/veed). Each presenter phrase's audio is extracted and re-voiced through a **
|
|
3973
|
+
**Native talking heads + one voice per person (no post-hoc lip-sync).** Seedance 2.0 generates lip-synced speech **natively** — a presenter phrase puts the full phrase in the clip's prompt with `generate_audio`, so lips and voice are generated together (no `video_lipsync`/veed). Each presenter phrase's audio is extracted (the spoken window only) and the extracts are merged **per speaker** onto one timeline, then re-voiced through a **single** `audio_voice_convert` (`<voice>_conv`, ElevenLabs Voice Changer) to the brand voice — one STS pass over the whole track instead of a convert node per clip, so it's fewer nodes, fewer calls, and a more consistent brand timbre (composite scenes already share this path); timing is preserved so the lips stay matched. There is **ONE voice per person**: a single `voice_select` is reused for all that person's phrases, and the deconstruct's `voiceover` label folds into the sole on-camera presenter (so on-camera and off-camera narration are the same voice, not two). A scene with **two speakers both on screen** can't be one clip — both lines become `tts` over a plain scene clip. But a scene with **one on-camera speaker trading lines with an OFF-camera voice** (an interviewer, a heard-but-not-shown assistant) keeps the on-camera speaker **native** (lip-synced) and reads the off-camera line as `tts` — "on screen" is decided by the speaker's element presence, so a heard-but-unshown voice no longer drops the whole scene to a silent clip. Every `tts` node is stamped with the spoken track's **`language_code`** when the blueprint states a language (cast localization note / voiceover persona / voice description), so numbers and units are read in the target tongue instead of ElevenLabs' English default (the "6900 read in English" bug). For **NATIVE (Seedance) lines** — which carry no language tag — the scaffold additionally **spells numerals into target-language words** across every part of the clip prompt Seedance can vocalize (the spoken line, the scene summary/action/motion, the transcript), so a French "6930 ?" becomes "six mille neuf cent trente ?" and is never read as English digits. Spelling covers **every language the blueprint can resolve** (fr, es, en, de, it, pt, nl, pl, ar, ja, ko, hi — via `n2words`); a language outside that set leaves digits (the `tts` path still localizes them via `language_code`).
|
|
3962
3974
|
|
|
3963
3975
|
**Same-shot lip-sync caution.** A single held shot can carry only ONE lip-synced clip (voiceover turns must not overlap, and Seedance generates one clip per shot), so when the on-camera speaker has further turns in that shot (a rapid "3000? … 4000?" with an off-camera "Plus" between), the first turn is native and the rest play as `tts` over the same clip — where the mouth no longer matches those words. This is inherent to reproducing sparse same-shot dialogue, not a wiring fault; the scaffold lists the affected scenes/lines in **`metadata.video.lip_sync_caution`** (advisory, never gated) so you can cut away to b-roll over those lines or rely on the burned-in captions that already show them.
|
|
3964
3976
|
|
|
@@ -3974,7 +3986,9 @@ It then scaffolds the full pipeline like an **editing timeline**: each clip gets
|
|
|
3974
3986
|
|
|
3975
3987
|
**Re-craft the script — the hook is the #1 decision.** A reproduction is *inspiration* from a proven ad, not a clone: its structure (hook → body → CTA) carries the persuasion, and the hook is *targeting*, so a competitor's hook often does **not** transfer. `metadata.todo.script_recraft` tags each scene with its `narrative_role` (from the deconstruct, else inferred) and carries the original line **flagged** so it is never shipped as-is — and the per-scene `recraft` instruction is **role-aware**: the **hook** scene's entry carries the diagnose → decide (keep/adapt/rebuild) → criteria (statement not question, benefit by ~2s, first frame legible **sound-off** in ~1s, no bait-and-switch) inline and routes to the skill's `references/hook-craft.md`. A dedicated top-level **`metadata.todo.hook`** key foregrounds it as the highest-leverage beat, mapped onto scene-0's artifacts (`s0_start` first frame, scene-0 overlay text, `s0_clip` line, micro-hook, hook-ramp).
|
|
3976
3988
|
|
|
3977
|
-
The
|
|
3989
|
+
**The inspiration video is preserved.** Like `scaffold-static-ad` keeps its reference image, the video command now auto-writes a **`_definition.md`** (so the creative joins the `creatives` collection) recording the source it was built from: `sourceKind: video`, `sourceAdvertiser` (the brand the deconstruct identified, or `--advertiser`), `platform` (`--platform`, default `meta`), and **`sourceReferenceUrl`** — the **durable, content-addressed R2 URL** the deconstruct already uploaded the source to (`prompt.json`'s `source.url`), which the dashboard's Inspiration card plays inline. Unlike the static flow it does **not** commit the video into `references/`: a reference clip can be up to 2 GiB and the video canvas never re-ingests the source at run time (it uses the extracted frame URLs), so a git copy would be pure bloat — the durable R2 URL is the reference. The `_definition.md` is preserved on re-scaffold, and the same `sourceReferenceUrl` is synced to the backend so the creative shows "built from this ad."
|
|
3990
|
+
|
|
3991
|
+
The emitted canvas is validated (`validateCanvasDeep`) before it's written, so it always runs. It also carries a **`metadata.video`** timing plan that `baker canvas validate` proves **statically, before any billed render**: no two voiceover turns overlap, the audio length ≈ the video length, every single-on-camera-speaker scene is a native talking head (its clip carries `generate_audio` and is wired to an `audio_voice_convert` node), **no re-crafted line physically overruns its clip** (`VIDEO_SPEECH_OVERRUN` — est. speech > ~1.6× the clip duration fails validate, since Seedance crams or dies on it), and **every clip agrees on one aspect ratio** (`VIDEO_ASPECT_MISMATCH`). When a **photoreal on-camera cast** generates on **Seedance**, the checklist carries a **`content_policy_risk`** note: ByteDance's real-person-likeness filter can reject a photoreal AI face with a **non-retryable 422** (`content_policy_blocked`) that **no prompt reframe clears** — the escapes are regenerating on Veo (`--video-model google/veo-3.1-fast`) or a less-photoreal frame. Surfaced before the billed run so a face-heavy ad isn't discovered broken mid-render. The full editable checklist is embedded as **`metadata.todo`** (with a step-by-step guide in `metadata.description`). stdout returns `{ ok, canvas_path, prompt_path, models, stats, checklist }`.
|
|
3978
3992
|
|
|
3979
3993
|
```bash
|
|
3980
3994
|
baker canvas scaffold-video ./reference-ad.mp4 --focus "competitor UGC ad for <brand>"
|
|
@@ -3990,6 +4004,7 @@ baker canvas run ./reference-ad.video.canvas.json
|
|
|
3990
4004
|
| `--slug <slug>` | — | Creative slug (lowercase kebab): writes the canvas to `src/creatives/<slug>/<slug>.canvas.json` — the repo convention that attaches every run to the creative's dashboard generation history. `--out` wins over `--slug`. |
|
|
3991
4005
|
| `--frames <mode>` | `generate` | `generate` emits ONE recast keyframe per scene (the original frame is dropped so the dropped `el_*` assets drive identity); `reuse` wires the real extracted first+last frames straight into the clips (faithful, cheaper, no recast). |
|
|
3992
4006
|
| `--ambient` | off | Give silent **b-roll** scenes native diegetic ambient (Seedance `generate_audio`), mixed deep under the music bed. Talking scenes already carry voice; check levels don't muddy the mix before keeping it. |
|
|
4007
|
+
| `--seam-dedup <mode>` | `head` | How to dedup the boundary frame two clips SHARE when a long shot is split for length (the second clip's first frame IS the first clip's last frame, so a plain concat freezes on it for a frame). `head` drops the second clip's first frame, `tail` drops the first clip's last frame, `off` keeps both. Only touches shared-frame continuation joins — a hard cut between two shots shares no frame. |
|
|
3993
4008
|
| `--max-scenes <n>` | all source scenes | **Cost lever that reduces fidelity** — caps the deconstruct, MERGING away every scene beyond the cap (fewer cuts, lost beats). Prints a warning when set; omit it to reproduce every scene. |
|
|
3994
4009
|
| `--language <code>` | auto | Transcript/dialogue language hint (e.g. `fr`, `en`). |
|
|
3995
4010
|
| `--focus <text>` | — | Known provenance/emphasis to ground the deconstruct. |
|
|
@@ -4012,10 +4027,10 @@ The two scaffold passes are billed (the full `video_deconstruct` is the heavy on
|
|
|
4012
4027
|
Turn a source/inspiration image into a **runnable, self-validated static-ad canvas** — the static counterpart of `scaffold-video`. Like the video scaffold, this runs **billed Gemini passes** up front:
|
|
4013
4028
|
|
|
4014
4029
|
1. **`image_describe`** (`~google/gemini-pro-latest`) — reverse-engineers the image into a blueprint JSON, written next to the canvas as **`prompt.json`**. This is the editable "prompt": you rewrite it by hand into the ad you want (palette, copy, claims, subjects). It feeds the generator directly — there is **no automatic brand-transform step**. The blueprint also names the **`winning_mechanisms`** — the special sauce that makes the ad a candidate winner, each tagged `kind` (verbal: rhyme/pun/rhythm; visual: unexpected crop, visual gag, juxtaposition, pattern interrupt, before/after; structural: hook order/reveal) with a `device` and `why_it_works` — so your rewrite rebuilds the mechanism that makes the ad win instead of adapting only the surface and losing it.
|
|
4015
|
-
2. **element selection** (`~google/gemini-flash-latest`) — picks the **main, identity-critical** elements (the brand logo, a showcased product, a trust badge) **plus any foreground/hero person or animal** — the emotional focal point — even a generic one, because a free-generated face/muzzle reads as AI and grows artifacts; the emotional hero always gets a real-reference slot. Background extras are dropped. Each is stamped back onto its blueprint entry as a `reference_image` label so the JSON self-documents which slot grounds which subject.
|
|
4030
|
+
2. **element selection** (`~google/gemini-flash-latest`) — picks the **main, identity-critical** elements (the brand logo, a showcased product, a trust badge) **plus any foreground/hero person or animal** — the emotional focal point — even a generic one, because a free-generated face/muzzle reads as AI and grows artifacts; the emotional hero always gets a real-reference slot. When the advertiser's logo appears in **more than one lockup** (a square/icon **mark** and a horizontal **wordmark**), each is emitted as its **own** element (e.g. `LOGO_MARK`, `LOGO_WORDMARK`) so you drop the right file in each slot instead of stretching one logo to cover both. The describe pass also records the ad's **typography** under a `fonts` block (each typeface's classification, a best-guess family, and its weight/case) so you know exactly what to drop at the brand-font slot. Background extras are dropped. Each element is stamped back onto its blueprint entry as a `reference_image` label so the JSON self-documents which slot grounds which subject.
|
|
4016
4031
|
3. **global layout** (`~google/gemini-flash-latest`) — produces a structured `layout` block in `prompt.json`: the column/row grid, each region's `x_pct`/`y_pct` bounds, panel splits, background/shape, and every text block's relative size/weight/case/alignment. This is what gives the generator a precise composition to rebuild.
|
|
4017
4032
|
|
|
4018
|
-
It then scaffolds a canvas that ingests `prompt.json`, wires **one `[TODO]` ingest slot per detected element** (plus an optional brand-font → type-specimen) into `image_generate`, and wires the original image in for composition only. The canvas is validated before it's written. stdout returns `{ ok, canvas_path, prompt_path, models, layout_regions, stats, checklist }` — the **checklist** lists every real asset to drop in.
|
|
4033
|
+
It then scaffolds a canvas that ingests `prompt.json`, wires **one `[TODO]` ingest slot per detected element** (plus an optional brand-font → type-specimen) into `image_generate`, and wires the original image in for composition only. Each **person/animal hero** is additionally fused into a generated **multi-view reference sheet** (`image_reference_sheet`, a turnaround built from the one dropped photo) that the render grounds on instead of the lone flat snapshot — the same identity lock the video scaffold uses, so the face/muzzle stays consistent and artifact-free from a single reference. Pass `--skip-actor-sheets` to ground straight on the dropped photo. The canvas is validated before it's written. stdout returns `{ ok, canvas_path, prompt_path, models, layout_regions, stats, checklist }` — the **checklist** lists every real asset to drop in (and which heroes get a sheet).
|
|
4019
4034
|
|
|
4020
4035
|
```bash
|
|
4021
4036
|
baker canvas scaffold-static-ad ./reference-ad.png --context "competitor ad for <brand>, <category>, <market>"
|
|
@@ -4036,6 +4051,7 @@ baker canvas run ./static-ad.canvas.json
|
|
|
4036
4051
|
| `--gen-model <id>` | registry default (`openai/gpt-5.4-image-2`) | Override the `image_generate` model. |
|
|
4037
4052
|
| `--aspect <ratio>` | inferred from the image, else `9:16` | Force the output aspect ratio. |
|
|
4038
4053
|
| `--skip-font` | off | Skip the brand-font → type-specimen slot. |
|
|
4054
|
+
| `--skip-actor-sheets` | off | Ground each person/animal on its lone dropped photo instead of a generated multi-view reference sheet. |
|
|
4039
4055
|
|
|
4040
4056
|
Scaffolding runs (and bills) the two vision passes; **running** the result generates a billed image. `baker canvas validate` does not check that the `[TODO]` paths exist — supply the real files before `run`.
|
|
4041
4057
|
|
|
@@ -4559,6 +4575,7 @@ This CLI is designed for AI agent consumption. Key patterns:
|
|
|
4559
4575
|
- **0.105.0**: `baker images ...`, `baker videos ...`, and `baker testimonials ...` commands now type their `/api/{images,videos,testimonials}/...` request/response payloads from the shared `@baker/api` contract package instead of hand-written local interfaces. No command, flag, or output-shape changes.
|
|
4560
4576
|
- **0.106.0**: `baker ads linkedin` gains staged write commands — `campaign-groups`/`campaigns`/`creatives` create|update|pause|resume|(archive|)duplicate, `audiences create|upload`, `conversions create|update`, `lead-forms create|update`, plus `draft [remove|clear]` for review/undo. Ops validate at stage time, apply on chat publish, and run simulated (`urn:li:simulated:*`) unless LinkedIn writes are enabled for the company.
|
|
4561
4577
|
- **0.116.0**: `lead-forms create` models the full Campaign Manager form — `locale`, form banner image (`formImageId`/`formImageUrn`), predefined profile-field questions (validated enum) vs custom questions (`SINGLE_LINE_TEXT`/`MULTIPLE_CHOICE` with `options`, ≤3 custom), `privacyPolicyText`, disclosure `consents[]` (≤5), tracking `hiddenFields[]` (≤20), and `thankYou` confirmation CTA + landing/appointment link. Staged preview surfaces best-practice warnings (no qualifying question, no confirmation, no consent). No breaking flag changes.
|
|
4578
|
+
- **0.121.0**: `lead-forms create` drops the `privacyPolicyText` field — LinkedIn's versioned lead-form API has no privacy-policy-text slot, so it was silently discarded on publish. Use `legalDisclaimer` (shown under the form) or `consents[]` (disclosure checkboxes) instead. (Companion backend fix: staged lead-form questions were serialized in a shape LinkedIn dropped — they now publish correctly, and the staged preview lists each question.)
|
|
4562
4579
|
- **0.119.0**: `draft amend`/`draft show` land on both `baker ads google` and `baker ads linkedin` — a generic JSON-merge-patch to update any staged op in place plus a full-payload receipt, replacing remove+recreate as the correction path. Google gains `assets update` and `asset-groups create|update` (Performance Max asset groups are now their own entity — `ads create --format performanceMaxAssetGroup` never worked and is gone); `ads create --format video` moves from a bare YouTube id to `--video-assets` refs (**breaking flag change** — stage the video as an asset first); `--format demandGen` gains `--image-assets`/`--square-image-assets`/`--logo-image-assets` and flag-building for headlines/descriptions. LinkedIn's `draft list` now renders a readable Campaign group ▸ Campaign ▸ Creative tree by default (`--json` for raw), `creatives update` gains `--campaign` (re-parent while staged), and `campaigns update` passes create-only fields (`--group`/`--type`/`--locale`/`--associated-entity`) through when amending a `li_temp_*` staged create instead of always stripping them.
|
|
4563
4580
|
|
|
4564
4581
|
## Publishing
|
|
@@ -780,7 +780,9 @@ function failedJobError(error) {
|
|
|
780
780
|
retryable: error.retryable ?? false
|
|
781
781
|
});
|
|
782
782
|
}
|
|
783
|
-
|
|
783
|
+
function pollInterval(attempt) {
|
|
784
|
+
return attempt < 15 ? 1e3 : 3e3;
|
|
785
|
+
}
|
|
784
786
|
var JOB_POLL_MAX_MS = 20 * 60 * 1e3;
|
|
785
787
|
var BackendClient = class {
|
|
786
788
|
http;
|
|
@@ -797,7 +799,7 @@ var BackendClient = class {
|
|
|
797
799
|
async pollJob(jobId, signal) {
|
|
798
800
|
const deadline = Date.now() + JOB_POLL_MAX_MS;
|
|
799
801
|
const path16 = `/api/canvas/jobs/${encodeURIComponent(jobId)}`;
|
|
800
|
-
|
|
802
|
+
for (let attempt = 0; ; attempt++) {
|
|
801
803
|
if (signal?.aborted) {
|
|
802
804
|
throw new BackendHttpError({ kind: "network", cause: signal.reason ?? new Error("aborted") });
|
|
803
805
|
}
|
|
@@ -807,7 +809,7 @@ var BackendClient = class {
|
|
|
807
809
|
if (Date.now() > deadline) {
|
|
808
810
|
throw new BackendHttpError({ kind: "timeout", message: `job ${jobId} did not finish in time` });
|
|
809
811
|
}
|
|
810
|
-
await sleep(
|
|
812
|
+
await sleep(pollInterval(attempt));
|
|
811
813
|
}
|
|
812
814
|
}
|
|
813
815
|
presignAssetUpload(sha256, mime, signal) {
|
|
@@ -838,6 +840,14 @@ var BackendClient = class {
|
|
|
838
840
|
async recordRun(payload, signal) {
|
|
839
841
|
await this.http.postJson("/api/canvas/runs", payload, signal);
|
|
840
842
|
}
|
|
843
|
+
/**
|
|
844
|
+
* Chat-scoped blueprint sync — POST /api/creatives/definition. Lets the
|
|
845
|
+
* dashboard draw a scaffolded creative's workflow graph BEFORE the first run.
|
|
846
|
+
* Additive on the backend (never archives siblings, never sets definitionPath).
|
|
847
|
+
*/
|
|
848
|
+
async syncCreativeDefinition(payload, signal) {
|
|
849
|
+
await this.http.postJson("/api/creatives/definition", payload, signal);
|
|
850
|
+
}
|
|
841
851
|
getArtifact(kind, name, version, signal) {
|
|
842
852
|
const path16 = version ? `/api/canvas/artifacts/${encodeURIComponent(kind)}/${encodeURIComponent(name)}/${encodeURIComponent(version)}` : `/api/canvas/artifacts/${encodeURIComponent(kind)}/${encodeURIComponent(name)}`;
|
|
843
853
|
return this.http.getJson(path16, signal);
|
|
@@ -986,10 +996,10 @@ var ELEVENLABS_OUTPUT_FORMATS = [
|
|
|
986
996
|
var ELEVENLABS_MAX_TEXT_CHARS = 45454;
|
|
987
997
|
var ELEVENLABS_MAX_MUSIC_LENGTH_MS = 454545;
|
|
988
998
|
var OPENROUTER_IMAGE_MIMES = ["image/png", "image/jpeg", "image/webp", "image/gif"];
|
|
989
|
-
var
|
|
990
|
-
var
|
|
999
|
+
var REPLICATE_IMAGE_MIMES = ["image/png", "image/jpeg", "image/webp"];
|
|
1000
|
+
var REPLICATE_VIDEO_MIMES = ["video/mp4", "video/webm", "video/quicktime"];
|
|
991
1001
|
var DECONSTRUCT_VIDEO_MIMES = ["video/mp4", "video/webm", "video/quicktime"];
|
|
992
|
-
var
|
|
1002
|
+
var REPLICATE_AUDIO_MIMES = ["audio/wav", "audio/mpeg", "audio/mp3"];
|
|
993
1003
|
var IMAGE_GENERATE_MODELS = [
|
|
994
1004
|
"openai/gpt-5.4-image-2",
|
|
995
1005
|
"google/gemini-3.5-flash",
|
|
@@ -1207,20 +1217,23 @@ var MODEL_REGISTRY = {
|
|
|
1207
1217
|
},
|
|
1208
1218
|
video_generate: {
|
|
1209
1219
|
"bytedance/seedance-2.0": {
|
|
1210
|
-
// Routed via
|
|
1211
|
-
//
|
|
1212
|
-
//
|
|
1220
|
+
// Routed via Replicate's official `bytedance/seedance-2.0` model. NOTE:
|
|
1221
|
+
// ByteDance's upstream "real person" likeness filter still blocks photoreal
|
|
1222
|
+
// human reference frames on ANY reseller — the escape is a synthetic/AI
|
|
1223
|
+
// presenter face or routing real faces to Veo, not the provider choice.
|
|
1213
1224
|
label: "ByteDance Seedance 2.0",
|
|
1214
1225
|
inputs: [],
|
|
1215
|
-
optional_inputs: [{ kind: "image", mimes:
|
|
1226
|
+
optional_inputs: [{ kind: "image", mimes: REPLICATE_IMAGE_MIMES }],
|
|
1216
1227
|
required: ["prompt"],
|
|
1217
1228
|
params: {
|
|
1218
|
-
prompt
|
|
1229
|
+
// Replicate's Seedance wrapper hard-caps the prompt at 4000 chars; gate
|
|
1230
|
+
// it here so an over-length prompt fails validate (free) not the billed call.
|
|
1231
|
+
prompt: { kind: "string", maxLength: 4e3 },
|
|
1219
1232
|
aspect_ratio: {
|
|
1220
1233
|
kind: "string",
|
|
1221
1234
|
enum: ["1:1", "3:4", "9:16", "4:3", "16:9", "21:9", "9:21"]
|
|
1222
1235
|
},
|
|
1223
|
-
resolution: { kind: "string", enum: ["480p", "720p", "1080p"] },
|
|
1236
|
+
resolution: { kind: "string", enum: ["480p", "720p", "1080p", "4k"] },
|
|
1224
1237
|
duration: { kind: "number", enum: SEEDANCE_DURATIONS },
|
|
1225
1238
|
seed: { kind: "number" },
|
|
1226
1239
|
generate_audio: { kind: "boolean" }
|
|
@@ -1241,7 +1254,10 @@ var MODEL_REGISTRY = {
|
|
|
1241
1254
|
duration: { kind: "number", enum: [4, 6, 8] },
|
|
1242
1255
|
seed: { kind: "number" },
|
|
1243
1256
|
generate_audio: { kind: "boolean" },
|
|
1244
|
-
|
|
1257
|
+
// Image-to-video and EU/UK/CH/MENA regions cap this at `allow_adult`;
|
|
1258
|
+
// `allow_all` is text-to-video only. Allow both so an image-conditioned
|
|
1259
|
+
// Veo clip (the real-face fallback) validates.
|
|
1260
|
+
person_generation: { kind: "string", enum: ["allow_all", "allow_adult"] },
|
|
1245
1261
|
enhance_prompt: { kind: "boolean" },
|
|
1246
1262
|
conditioning_scale: { kind: "number" }
|
|
1247
1263
|
}
|
|
@@ -1292,8 +1308,8 @@ var MODEL_REGISTRY = {
|
|
|
1292
1308
|
"fal/veed-lipsync": {
|
|
1293
1309
|
label: "VEED Lipsync (fal.ai)",
|
|
1294
1310
|
inputs: [
|
|
1295
|
-
{ kind: "video", mimes:
|
|
1296
|
-
{ kind: "audio", mimes:
|
|
1311
|
+
{ kind: "video", mimes: REPLICATE_VIDEO_MIMES },
|
|
1312
|
+
{ kind: "audio", mimes: REPLICATE_AUDIO_MIMES }
|
|
1297
1313
|
],
|
|
1298
1314
|
required: [],
|
|
1299
1315
|
params: {}
|
|
@@ -1329,7 +1345,7 @@ var MODEL_REGISTRY = {
|
|
|
1329
1345
|
// TARGET voice, preserving timing/prosody. Used to normalize a talking-head
|
|
1330
1346
|
// clip's native (generator-chosen) voice into ONE consistent brand voice.
|
|
1331
1347
|
label: "ElevenLabs Voice Changer (multilingual STS v2)",
|
|
1332
|
-
inputs: [{ kind: "audio", mimes:
|
|
1348
|
+
inputs: [{ kind: "audio", mimes: REPLICATE_AUDIO_MIMES }],
|
|
1333
1349
|
required: ["voice"],
|
|
1334
1350
|
params: {
|
|
1335
1351
|
voice: { kind: "string" },
|
|
@@ -1356,7 +1372,7 @@ var MODEL_REGISTRY = {
|
|
|
1356
1372
|
},
|
|
1357
1373
|
"elevenlabs/video-background-music-v1": {
|
|
1358
1374
|
label: "ElevenLabs Video Background Music v1",
|
|
1359
|
-
inputs: [{ kind: "video", mimes:
|
|
1375
|
+
inputs: [{ kind: "video", mimes: REPLICATE_VIDEO_MIMES }],
|
|
1360
1376
|
required: [],
|
|
1361
1377
|
params: {
|
|
1362
1378
|
description: { kind: "string" },
|
|
@@ -1772,7 +1788,14 @@ var NodeDecl = z.object({
|
|
|
1772
1788
|
version: z.string().min(1).optional(),
|
|
1773
1789
|
inputs: z.record(z.string(), z.unknown()).optional(),
|
|
1774
1790
|
params: z.record(z.string(), z.unknown()).optional(),
|
|
1775
|
-
when: z.unknown().optional()
|
|
1791
|
+
when: z.unknown().optional(),
|
|
1792
|
+
// Regenerate knob. The engine is content-addressed: identical params + inputs
|
|
1793
|
+
// return the cached render, so re-running an unchanged node NEVER re-bills or
|
|
1794
|
+
// produces a new result. Bump this token (any string/number — a `2`, a `"v3"`,
|
|
1795
|
+
// a note) and re-run to force THIS node to render fresh; because its new output
|
|
1796
|
+
// changes downstream input hashes, everything depending on it regenerates too.
|
|
1797
|
+
// This is the declarative "change a value, re-run, get a new render" affordance.
|
|
1798
|
+
regenerate: z.union([z.string(), z.number()]).optional()
|
|
1776
1799
|
}).strict();
|
|
1777
1800
|
var OutputRef = z.object({
|
|
1778
1801
|
node: z.string(),
|
|
@@ -3055,6 +3078,7 @@ var Engine = class {
|
|
|
3055
3078
|
const counters = { cachedNodes: 0, totalCredits: 0 };
|
|
3056
3079
|
const nodeRuns = [];
|
|
3057
3080
|
const graph = this.pruneToOutput(canvas, buildGraph(canvas));
|
|
3081
|
+
const needsBytes = computeNeedsLocalBytes(canvas, graph, this.registry);
|
|
3058
3082
|
this.emitProgress(opts, {
|
|
3059
3083
|
kind: "plan",
|
|
3060
3084
|
nodes: [...graph.entries()].map(([id, deps]) => {
|
|
@@ -3062,7 +3086,7 @@ var Engine = class {
|
|
|
3062
3086
|
return { node_id: id, node_type: node?.type ?? "unknown", deps: [...deps], params: node?.params };
|
|
3063
3087
|
})
|
|
3064
3088
|
});
|
|
3065
|
-
await this.runLayers(canvas, graph, outputs, runId, writer, opts, counters, nodeRuns);
|
|
3089
|
+
await this.runLayers(canvas, graph, outputs, runId, writer, opts, counters, nodeRuns, needsBytes);
|
|
3066
3090
|
const output = pickFinalOutput(canvas, outputs);
|
|
3067
3091
|
const stats = {
|
|
3068
3092
|
total_nodes: canvas.nodes.length,
|
|
@@ -3087,13 +3111,13 @@ var Engine = class {
|
|
|
3087
3111
|
this.log(`outputs in: ${writer.runDir}`);
|
|
3088
3112
|
return { run_id: runId, output, outputs_by_node: outputs, stats, outputs_dir: writer.runDir, node_runs: nodeRuns };
|
|
3089
3113
|
}
|
|
3090
|
-
async runLayers(canvas, graph, outputs, runId, writer, opts, counters, nodeRuns) {
|
|
3114
|
+
async runLayers(canvas, graph, outputs, runId, writer, opts, counters, nodeRuns, needsBytes) {
|
|
3091
3115
|
const layers = topologicalLayers(graph);
|
|
3092
3116
|
const limit = resolveConcurrency(opts.concurrency);
|
|
3093
3117
|
for (const layer of layers) {
|
|
3094
3118
|
const settled = await mapWithConcurrency(layer, limit, (nodeId) => {
|
|
3095
3119
|
this.emitProgress(opts, { kind: "node_start", node_id: nodeId });
|
|
3096
|
-
return this.executeOne(canvas, nodeId, outputs, runId, writer, opts).then((r) => {
|
|
3120
|
+
return this.executeOne(canvas, nodeId, outputs, runId, writer, opts, needsBytes.has(nodeId)).then((r) => {
|
|
3097
3121
|
if (r.cached) counters.cachedNodes++;
|
|
3098
3122
|
counters.totalCredits += r.credits;
|
|
3099
3123
|
const node = canvas.nodes.find((n) => n.id === nodeId);
|
|
@@ -3160,12 +3184,13 @@ var Engine = class {
|
|
|
3160
3184
|
}
|
|
3161
3185
|
await writer.writeManifest("_final", output);
|
|
3162
3186
|
}
|
|
3163
|
-
async executeOne(canvas, nodeId, outputs, runId, writer, opts) {
|
|
3187
|
+
async executeOne(canvas, nodeId, outputs, runId, writer, opts, downloadOutputs) {
|
|
3164
3188
|
const node = canvas.nodes.find((n) => n.id === nodeId);
|
|
3165
3189
|
if (!node) throw new Error(`executor: missing node ${nodeId}`);
|
|
3166
3190
|
const def = this.registry.get(node.type);
|
|
3167
3191
|
if (!def) throw new Error(`executor: missing registry entry for type ${node.type}`);
|
|
3168
|
-
const
|
|
3192
|
+
const regenerateToken = resolveRegenerateToken(node, opts.regenerate, runId);
|
|
3193
|
+
const prepared = await prepareForExecution(node, outputs, def, canvas.cache_salt, regenerateToken, this.assets);
|
|
3169
3194
|
const policy = opts.cache_policy ?? "read_write";
|
|
3170
3195
|
if (policy !== "bypass") {
|
|
3171
3196
|
const cacheT0 = Date.now();
|
|
@@ -3183,12 +3208,14 @@ var Engine = class {
|
|
|
3183
3208
|
nodeId: node.id,
|
|
3184
3209
|
nodeType: node.type,
|
|
3185
3210
|
cacheKey: prepared.cacheKey,
|
|
3211
|
+
downloadOutputs,
|
|
3186
3212
|
client: this.client,
|
|
3187
3213
|
assets: this.assets,
|
|
3188
3214
|
log: this.log,
|
|
3189
3215
|
signal: opts.signal
|
|
3190
3216
|
};
|
|
3191
|
-
const
|
|
3217
|
+
const preparedForExec = def.location === "local" ? { ...prepared, resolvedInputs: await this.materializeLocalInputs(prepared.resolvedInputs) } : prepared;
|
|
3218
|
+
const { parsedInputs, parsedParams } = parseNodeArgs(def, preparedForExec, node.id, node.type);
|
|
3192
3219
|
const result = await invokeExecute(def, parsedInputs, parsedParams, ctx, node.id, node.type);
|
|
3193
3220
|
const elapsed = Date.now() - t0;
|
|
3194
3221
|
const credits = def.cost ? def.cost({ params: parsedParams }).credits : 0;
|
|
@@ -3230,8 +3257,42 @@ var Engine = class {
|
|
|
3230
3257
|
}
|
|
3231
3258
|
}
|
|
3232
3259
|
}
|
|
3260
|
+
/**
|
|
3261
|
+
* Download any URL-only asset ref reachable in a local node's inputs so the
|
|
3262
|
+
* bytes are on disk before the local runner stages them. Returns a copy —
|
|
3263
|
+
* refs are replaced, never mutated in place, so the producer's cached output
|
|
3264
|
+
* (shared object) keeps its URL-only shape.
|
|
3265
|
+
*/
|
|
3266
|
+
async materializeLocalInputs(inputs) {
|
|
3267
|
+
const fix = async (value) => {
|
|
3268
|
+
if (Array.isArray(value)) return Promise.all(value.map(fix));
|
|
3269
|
+
if (value && typeof value === "object") {
|
|
3270
|
+
const v = value;
|
|
3271
|
+
if (typeof v.kind === "string" && typeof v.url === "string" && typeof v.sha256 === "string" && typeof v.mime === "string" && typeof v.path !== "string") {
|
|
3272
|
+
this.log(`[warn ] materializing URL-only input on demand (${v.kind}/${v.mime}) \u2014 missed graph edge`);
|
|
3273
|
+
return this.assets.ingestRemote({
|
|
3274
|
+
kind: v.kind,
|
|
3275
|
+
url: v.url,
|
|
3276
|
+
sha256: v.sha256,
|
|
3277
|
+
mime: v.mime,
|
|
3278
|
+
metadata: v.metadata
|
|
3279
|
+
});
|
|
3280
|
+
}
|
|
3281
|
+
const out = {};
|
|
3282
|
+
for (const [k, val] of Object.entries(v)) out[k] = await fix(val);
|
|
3283
|
+
return out;
|
|
3284
|
+
}
|
|
3285
|
+
return value;
|
|
3286
|
+
};
|
|
3287
|
+
return await fix(inputs);
|
|
3288
|
+
}
|
|
3233
3289
|
};
|
|
3234
|
-
|
|
3290
|
+
function resolveRegenerateToken(node, forced, runId) {
|
|
3291
|
+
if (forced?.has(node.id)) return `run:${runId}`;
|
|
3292
|
+
if (node.regenerate !== void 0) return `node:${String(node.regenerate)}`;
|
|
3293
|
+
return void 0;
|
|
3294
|
+
}
|
|
3295
|
+
async function prepareForExecution(node, outputs, def, cacheSalt, regenerateToken, assets) {
|
|
3235
3296
|
const resolvedInputs = resolveRefs(node.inputs ?? {}, { outputs }) ?? {};
|
|
3236
3297
|
const resolvedParams = resolveRefs(node.params ?? {}, { outputs }) ?? {};
|
|
3237
3298
|
const slotValues = await hydrateTextSlots(resolvedInputs, assets, node.id, node.type);
|
|
@@ -3244,6 +3305,9 @@ async function prepareForExecution(node, outputs, def, cacheSalt, assets) {
|
|
|
3244
3305
|
throw new NodeExecutionError(node.id, node.type, { kind: "local", cause: e });
|
|
3245
3306
|
}
|
|
3246
3307
|
}
|
|
3308
|
+
if (regenerateToken !== void 0) {
|
|
3309
|
+
extras = { ...extras ?? {}, __regenerate__: regenerateToken };
|
|
3310
|
+
}
|
|
3247
3311
|
const cacheKey = computeCacheKey({
|
|
3248
3312
|
node_id: node.type,
|
|
3249
3313
|
node_version: def.version,
|
|
@@ -3282,6 +3346,16 @@ function pickFinalOutput(canvas, outputs) {
|
|
|
3282
3346
|
const lastOut = outputs[last.id];
|
|
3283
3347
|
return lastOut ? Object.values(lastOut)[0] : void 0;
|
|
3284
3348
|
}
|
|
3349
|
+
function computeNeedsLocalBytes(canvas, graph, registry) {
|
|
3350
|
+
const typeById = new Map(canvas.nodes.map((n) => [n.id, n.type]));
|
|
3351
|
+
const needs = /* @__PURE__ */ new Set();
|
|
3352
|
+
for (const [consumerId, deps] of graph) {
|
|
3353
|
+
const def = registry.get(typeById.get(consumerId) ?? "");
|
|
3354
|
+
if (def?.location !== "local") continue;
|
|
3355
|
+
for (const dep of deps) needs.add(dep);
|
|
3356
|
+
}
|
|
3357
|
+
return needs;
|
|
3358
|
+
}
|
|
3285
3359
|
function buildGraph(canvas) {
|
|
3286
3360
|
const graph = /* @__PURE__ */ new Map();
|
|
3287
3361
|
for (const n of canvas.nodes) graph.set(n.id, /* @__PURE__ */ new Set());
|
|
@@ -3412,7 +3486,16 @@ async function hydrateSlotValue(value, assets, nodeId, nodeType) {
|
|
|
3412
3486
|
try {
|
|
3413
3487
|
bytes = await assets.readBytes(value.sha256, value.mime);
|
|
3414
3488
|
} catch (e) {
|
|
3415
|
-
|
|
3489
|
+
if (value.url) {
|
|
3490
|
+
try {
|
|
3491
|
+
await assets.ingestRemote({ kind: value.kind, url: value.url, sha256: value.sha256, mime: value.mime });
|
|
3492
|
+
bytes = await assets.readBytes(value.sha256, value.mime);
|
|
3493
|
+
} catch (e2) {
|
|
3494
|
+
throw new NodeExecutionError(nodeId, nodeType, { kind: "local", cause: e2 });
|
|
3495
|
+
}
|
|
3496
|
+
} else {
|
|
3497
|
+
throw new NodeExecutionError(nodeId, nodeType, { kind: "local", cause: e });
|
|
3498
|
+
}
|
|
3416
3499
|
}
|
|
3417
3500
|
if (bytes.length > MAX_INLINE_TEXT_BYTES) {
|
|
3418
3501
|
throw new NodeExecutionError(nodeId, nodeType, {
|
|
@@ -3547,7 +3630,9 @@ async function callBackendExec(args) {
|
|
|
3547
3630
|
nodeVersion: args.nodeVersion,
|
|
3548
3631
|
params: args.params,
|
|
3549
3632
|
inputs: serialized,
|
|
3550
|
-
idempotency_key: idempotencyKey
|
|
3633
|
+
idempotency_key: idempotencyKey,
|
|
3634
|
+
canvas_run_id: args.ctx.canvasRunId,
|
|
3635
|
+
node_id: args.ctx.nodeId
|
|
3551
3636
|
},
|
|
3552
3637
|
args.ctx.signal
|
|
3553
3638
|
);
|
|
@@ -3599,6 +3684,9 @@ async function ingestValue(value, ctx, declaredKind) {
|
|
|
3599
3684
|
}
|
|
3600
3685
|
if (isRawAsset(value)) {
|
|
3601
3686
|
const kind = value.kind ?? declaredKind ?? "json";
|
|
3687
|
+
if (ctx.downloadOutputs === false) {
|
|
3688
|
+
return buildRef({ kind, sha: value.sha256, mime: value.mime, url: value.url, metadata: value.metadata });
|
|
3689
|
+
}
|
|
3602
3690
|
return ctx.assets.ingestRemote({
|
|
3603
3691
|
kind,
|
|
3604
3692
|
url: value.url,
|
|
@@ -3836,7 +3924,15 @@ var EXT_TO_MIME = {
|
|
|
3836
3924
|
jpeg: "image/jpeg",
|
|
3837
3925
|
webp: "image/webp",
|
|
3838
3926
|
gif: "image/gif",
|
|
3927
|
+
// Non-model-safe rasters `toModelSafeImage` transcodes to PNG at ingest — they
|
|
3928
|
+
// must resolve to an image mime here or the kind-check rejects the local file
|
|
3929
|
+
// before normalization ever runs.
|
|
3839
3930
|
avif: "image/avif",
|
|
3931
|
+
heic: "image/heic",
|
|
3932
|
+
heif: "image/heif",
|
|
3933
|
+
tif: "image/tiff",
|
|
3934
|
+
tiff: "image/tiff",
|
|
3935
|
+
bmp: "image/bmp",
|
|
3840
3936
|
mp4: "video/mp4",
|
|
3841
3937
|
webm: "video/webm",
|
|
3842
3938
|
mov: "video/quicktime",
|
|
@@ -3906,15 +4002,26 @@ async function toModelSafeImage(bytes) {
|
|
|
3906
4002
|
throw new Error(`bytes are not a decodable image (${e.message})`);
|
|
3907
4003
|
}
|
|
3908
4004
|
}
|
|
4005
|
+
function hasAscii(buf, offset, sig) {
|
|
4006
|
+
return buf.length >= offset + sig.length && buf.toString("ascii", offset, offset + sig.length) === sig;
|
|
4007
|
+
}
|
|
4008
|
+
var HEIC_BRANDS = /* @__PURE__ */ new Set(["heic", "heix", "heim", "heis", "hevc", "hevx", "mif1", "msf1", "heif"]);
|
|
4009
|
+
function sniffIsoBmff(buf) {
|
|
4010
|
+
if (!hasAscii(buf, 4, "ftyp")) return null;
|
|
4011
|
+
const brand = buf.subarray(8, 12).toString("ascii");
|
|
4012
|
+
if (brand === "avif" || brand === "avis") return "image/avif";
|
|
4013
|
+
if (HEIC_BRANDS.has(brand)) return "image/heic";
|
|
4014
|
+
return null;
|
|
4015
|
+
}
|
|
3909
4016
|
function sniffImageMime(buf) {
|
|
3910
4017
|
if (buf.length < 4) return null;
|
|
3911
|
-
if (buf[0] === 137 && buf
|
|
4018
|
+
if (buf[0] === 137 && hasAscii(buf, 1, "PNG")) return "image/png";
|
|
3912
4019
|
if (buf[0] === 255 && buf[1] === 216 && buf[2] === 255) return "image/jpeg";
|
|
3913
|
-
if (buf
|
|
3914
|
-
if (buf
|
|
3915
|
-
|
|
3916
|
-
|
|
3917
|
-
return
|
|
4020
|
+
if (hasAscii(buf, 0, "GIF")) return "image/gif";
|
|
4021
|
+
if (hasAscii(buf, 0, "RIFF") && hasAscii(buf, 8, "WEBP")) return "image/webp";
|
|
4022
|
+
if (hasAscii(buf, 0, "II*\0") || hasAscii(buf, 0, "MM\0*")) return "image/tiff";
|
|
4023
|
+
if (hasAscii(buf, 0, "BM")) return "image/bmp";
|
|
4024
|
+
return sniffIsoBmff(buf);
|
|
3918
4025
|
}
|
|
3919
4026
|
function findBoxPayload(buf, start, end, type) {
|
|
3920
4027
|
let offset = start;
|
|
@@ -6792,6 +6899,8 @@ export {
|
|
|
6792
6899
|
ulid,
|
|
6793
6900
|
isPersistedAssetRef,
|
|
6794
6901
|
collectAssetRefLikes,
|
|
6902
|
+
REF_PREFIX,
|
|
6903
|
+
parseRefExpr,
|
|
6795
6904
|
sha256Hex,
|
|
6796
6905
|
elementMentionKeywords,
|
|
6797
6906
|
toModelSafeImage,
|
|
@@ -6805,4 +6914,4 @@ export {
|
|
|
6805
6914
|
defaultRegistry,
|
|
6806
6915
|
createEngineFromEnv
|
|
6807
6916
|
};
|
|
6808
|
-
//# sourceMappingURL=chunk-
|
|
6917
|
+
//# sourceMappingURL=chunk-43KBQLP5.js.map
|