@slatesvideo/shared 0.7.1 → 0.7.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/clients/cloud.d.ts +4 -0
- package/dist/clients/cloud.js +11 -3
- package/dist/index.d.ts +2 -1
- package/dist/index.js +2 -1
- package/dist/manual/content.d.ts +1 -1
- package/dist/manual/content.js +1 -1
- package/dist/manual/index.d.ts +11 -2
- package/dist/manual/index.js +178 -14
- package/dist/operations/index.d.ts +292 -94
- package/dist/operations/index.js +870 -199
- package/dist/operations/surface.d.ts +6 -2
- package/dist/operations/surface.js +35 -5
- package/dist/prompts/agent-doctrine.d.ts +4 -4
- package/dist/prompts/agent-doctrine.js +18 -29
- package/dist/prompts/generation-policy.d.ts +1 -1
- package/dist/prompts/guide-discovery.d.ts +23 -0
- package/dist/prompts/guide-discovery.js +39 -0
- package/dist/prompts/guide-retrieval.js +1 -1
- package/dist/prompts/model-capabilities.d.ts +8 -9
- package/dist/prompts/model-capabilities.js +11 -51
- package/dist/prompts/model-facts.d.ts +2 -2
- package/dist/prompts/model-facts.js +15 -26
- package/dist/prompts/partials.generated.js +6 -3
- package/dist/prompts/prompting-tips.d.ts +1 -1
- package/dist/prompts/prompting-tips.js +21 -63
- package/dist/prompts/search-terms.d.ts +3 -0
- package/dist/prompts/search-terms.js +24 -0
- package/dist/skills/content.js +36 -37
- package/dist/skills/metadata.d.ts +7 -0
- package/dist/skills/metadata.js +29 -0
- package/exports/slates-chatgpt-images/generated/SKILL.md +7 -1
- package/exports/slates-chatgpt-images/generated/slates-chatgpt-images.skill +0 -0
- package/exports/slates-prompt-builder/generated/SKILL.md +28 -16
- package/exports/slates-prompt-builder/generated/reference-character.md +12 -13
- package/exports/slates-prompt-builder/generated/reference-content-policy.md +2 -2
- package/exports/slates-prompt-builder/generated/reference-gpt-image-2-5.md +191 -0
- package/exports/slates-prompt-builder/generated/reference-kling.md +32 -11
- package/exports/slates-prompt-builder/generated/reference-nano-banana.md +24 -6
- package/exports/slates-prompt-builder/generated/reference-omni-flash.md +65 -0
- package/exports/slates-prompt-builder/generated/reference-seedance-2-5.md +362 -0
- package/exports/slates-prompt-builder/generated/reference-seedance.md +34 -4
- package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +77 -23
- package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
- package/package.json +2 -1
- package/skills/_partials/blender-action-curves.md +24 -0
- package/skills/_partials/cinematic-card.md +1 -1
- package/skills/_partials/iteration-diagnosis.md +5 -0
- package/skills/_partials/model-routing.md +35 -0
- package/skills/_partials/seedance-25-timestamps.md +2 -2
- package/skills/_partials/still-gate.md +2 -2
- package/skills/_partials/thresholds.md +1 -1
- package/skills/slates-blocking-to-prompt.md +15 -13
- package/skills/slates-camera-language.md +45 -7
- package/skills/slates-character-identity.md +8 -6
- package/skills/slates-chatgpt-images.md +7 -1
- package/skills/slates-cinematic-look.md +1 -1
- package/skills/slates-content-policy.md +4 -6
- package/skills/slates-cost-discipline.md +18 -12
- package/skills/slates-dialogue-blocking.md +6 -6
- package/skills/slates-direct-response-ad.md +1 -1
- package/skills/slates-edit-and-iterate.md +12 -4
- package/skills/slates-model-selection.md +82 -90
- package/skills/slates-one-prompt-film.md +1 -1
- package/skills/slates-previs-blocking.md +44 -13
- package/skills/slates-project-organization.md +2 -2
- package/skills/slates-prompting-elevenlabs.md +4 -4
- package/skills/slates-prompting-flux-2-max.md +2 -3
- package/skills/slates-prompting-gpt-image-2-5.md +2 -2
- package/skills/slates-prompting-inworld-tts.md +1 -1
- package/skills/slates-prompting-kling-v3.md +11 -9
- package/skills/slates-prompting-lip-sync.md +15 -15
- package/skills/slates-prompting-ltx-2-5.md +5 -6
- package/skills/slates-prompting-minimax-h3.md +11 -11
- package/skills/slates-prompting-motion-transfer.md +8 -8
- package/skills/slates-prompting-nano-banana-2.md +8 -4
- package/skills/slates-prompting-omni-flash.md +9 -9
- package/skills/slates-prompting-seed-audio.md +24 -4
- package/skills/slates-prompting-seedance-2-5.md +40 -30
- package/skills/slates-prompting-seedance.md +4 -4
- package/skills/slates-prompting-seedream-5-lite.md +6 -6
- package/skills/slates-restyle-from-blocking.md +2 -2
- package/skills/slates-script-craft.md +1 -1
- package/skills/slates-shot-variety.md +1 -1
- package/skills/slates-storyboard-from-script.md +1 -1
- package/skills/slates-style-prompting.md +8 -6
- package/skills/slates-ugc-influencer-ad.md +1 -1
- package/skills/slates-vision-feedback-loop.md +118 -110
- package/skills/slates-prompting-veo-3.md +0 -224
package/dist/skills/content.js
CHANGED
|
@@ -1,42 +1,41 @@
|
|
|
1
1
|
// GENERATED — do not edit. Source: packages/shared/skills/*.md
|
|
2
2
|
// Regenerated by scripts/embed-skills.mjs on every build.
|
|
3
3
|
export const SKILLS = {
|
|
4
|
-
"slates-blocking-to-prompt": "---\nname: slates-blocking-to-prompt\ndescription: Write the generation prompt that matches a blocking clip second by second, so the reference video and the text agree instead of fighting. Use after rendering a previs blocking pass, when a generated shot ignores the reference video, when timings drift, or when the model invents shots and camera angles that are not in the blocking.\n---\n\n# Blocking → prompt\n\nYou have a blocking clip. This is how you write the prompt that goes with it.\n\n## The one idea\n\nThe clip already contains the camera, the cuts and the timing. **The prompt's job is to say what everything looks like — and, where the grey boxes are ambiguous, to disambiguate them.** It is not a second, competing description of the motion.\n\nState that contract inside the prompt, because the model needs it as much as you do:\n\n> Where a timeline line below names a camera position or move, it is a restatement of what the reference video already does at that timestamp — a disambiguation, never a new instruction.\n\nAnd give it a tie-break, because ambiguity is guaranteed:\n\n> If any text in this prompt appears to disagree with the reference video about camera, framing, direction, motion, timing or object placement, the reference video wins.\n\nThose two sentences do more work than any other part of the prompt.\n\n## Get the real numbers first\n\n```\nslates_blender_scene\n```\n\nWrite timings from `cutSeconds`, never from the shot list you intended to build. It resolves to the marker frames on a multi-camera edit and to the camera's own keyframes otherwise, so it is the one field that is never empty on a rig that has cuts. At 24fps cuts land on frame boundaries and the honest values are not round — `7.79s`, `9.33s`, `19.875s`. **Use the exact ones.** Rounding to `7.8` is a tenth of drift you are handing the model for free.\n\n## Structure\n\nOrder matters — contract, then globals, then timeline, then the re-assertion.\n\n```\nTITLE — one line: what this is, how long, that it is video-to-video\n\nLOGLINE — 2-4 sentences. The whole piece in plain language.\n\nACTIVE REFERENCES\n <one entry per reference: what it defines, and what is NOT inherited>\n\nTECHNICAL BLOCK (format, grade, lens, and the blanket negatives — see below)\n\nSTYLE / LOOK\nLIGHTING\nCOLOR\nCAMERA\nPHYSICS (only if things move, collide or deform)\n\nRULES (numbered — the invariants, see below)\n\nACTION TIMING (beat by beat, against the clip's real timestamps)\n\nAUDIO\n [Sound design] [Timed accents] [Dialogue] [Music]\n\nENDING LOCK (one line — where the film stops)\n\nHOLD FOR THE FULL TIMELINE\n <the 5-6 constraints most likely to drift, compressed>\n```\n\n## ACTIVE REFERENCES — every entry has an exclusion\n\nThe single highest-leverage format in this whole workflow. Each reference is a **positive claim plus an exclusion list**, because a reference the model over-reads is as damaging as one it ignores.\n\nLabel each by the badge code Slates echoes back (`IMG-A8`, `VID-C2`) or by an unmistakable role name, and use that same label everywhere below.\n\n**The blocking clip:**\n\n> VID-C2 = the blocking previz (30s, 720 frames, 24fps) — the MASTER for everything that moves and everything that stands. It defines the full edit one-to-one: every cut point, every camera position, angle, move and framing, all action timing, screen direction, and the geometry of the world. Its untextured grey surfaces, flat colours and viewport grid are NOT inherited — every grey proxy is dressed into a real object in the exact position the previz puts it. Proxies give position, angle, scale and motion only; never surface, shape detail or design.\n\nThat last sentence is the **placement-only clause** and it is not optional. Without it the model renders grey boxes.\n\n**A character sheet:**\n\n> IMG-A8 = the driver — defines his face, hair and wardrobe. Identity 100% consistent at every distance and through every motion blur. Background, lighting and pose are NOT inherited.\n\n**A location/style reference:**\n\n> IMG-B3 = the tunnel — defines location geometry, look and grade. Camera angle, framing and any people in it are NOT inherited; the camera comes exclusively from VID-C2.\n\n**An atmosphere or style master — a reference that is never a shot:**\n\n> IMG-D9 = ATMOSPHERE MASTER — NOT a keyframe, NOT a location to reproduce, NOT a frame that ever appears in the film: its own subject, framing and composition are never seen in any shot. It defines ONLY the weather, light, colour and grade: deep clean night just after rain, wet asphalt as a dark mirror, cool white-cyan lamps as the ambient key, teal-and-amber grade, deep clean blacks. Every shot is lit and graded in this regime for all 30 seconds.\n\nWithout those three NOTs the model reproduces the reference's composition as an actual shot — you get its street corner in your film. The same wording covers a rendering-style master; see `slates-restyle-from-blocking`.\n\n**References can be scheduled.** If something is only true for part of the timeline, say so: *the hooded panel applies only to 0–7.0s and 27.5–30s*, or per-reference: *Active for 00:03.3–00:06.7 only.* On a piece that travels through several locations, every location still carries its own window and the model stops blending two sets into one shot.\n\n## Translate the blocking's artifacts\n\nYour grey-box render contains things that are *notation*, not content. Every one needs an explicit reinterpretation or it gets rendered literally:\n\n| In the blocking | Say in the prompt |\n|---|---|\n| Colour-coded bodies | `red = the boss, green = the kid, blue = the driver` |\n| A marked face on a proxy | `RED face = the direction he faces, BLACK = his back` |\n| A checkered floor or wall | `the checkerboard is a scale reference, not a surface — it becomes <the real material>` |\n| Flat black background | `a PLACEHOLDER — replace with the location assigned below` |\n| A floor grid | `a motion-tracking aid — render as light on the surface, never as wireframe or tiles` |\n| A deliberate black gap | `CUT 7 (14.5-17.0, black gap in the reference) — <what fills it>` |\n| Frame goes dark mid-move | `the camera is passing through the ground — a doorway to the NEXT location, never back to a previous one` |\n| The source hard-resets mid-move | `each reset begins a NEW, completely different room — never a replay of one already seen` |\n| A proxy that is a PROP or VEHICLE | `the low-poly flying model in SHOT 18 is THE HELICOPTER · blocks on the rear bench are the luggage · the small dark block in his hand IS the pistol` |\n| A blocky proxy limb in a tight insert | `the blocky low-poly leg is a stand-in and must NOT be replicated — generate complete human anatomy: a real boot, a real trouser leg, a correct ankle at this exact camera angle` |\n| A stray object at the frame edge | `ignore it completely — never blend two sets into one shot` |\n\n## TECHNICAL BLOCK — format, lens, and the blanket negatives\n\nOne paragraph, before the timeline. It carries the things that are true of every frame and that no beat should have to repeat:\n\n> Cinematic, photoreal. 21:9. 30s. SFX only, no music. Kodak 500T film look, natural 35mm grain, organic colour, soft highlight roll-off, anamorphic lens character with oval bokeh and gentle barrel distortion at the edges, chromatic aberration creeping in at the frame edges, natural motion blur on every fast move, faint bloom on hot speculars. Every location well exposed — night interiors bright and readable, open shadows, no crushed blacks, no murk. NO CGI. NON-IP, no brand badges or logos anywhere, no text, no watermark.\n\nThree parts worth naming:\n\n- **Lens realism is a list, not an adjective.** Aberration, motion blur, depth of field, barrel distortion, bloom, grain. \"Cinematic\" buys you nothing; these buy you the look.\n- **Exposure needs saying on dark work.** Models crush night scenes into murk. *Night interiors bright and readable, open shadows, no crushed blacks* is what keeps a scene legible.\n- **The blanket negatives go here once** — `NON-IP`, no logos, no on-screen text, no subtitles, no watermark — rather than being scattered through the beats.\n\n## RULES — the invariants\n\nNumbered, short, absolute. These are the things that must hold in every frame, and they are where you put anything that has already gone wrong once.\n\nTwo patterns worth stealing outright:\n\n**Countable state.** Give the model arithmetic it can check itself against:\n\n> At every second: standing + fallen + on the lintel = 6. Never a seventh figure — no extras, no duplicates, no distant silhouettes, no half-bodies at frame edges.\n\n> Bodies on the ground count exactly: 0 before 12s → 1 → 2 → 3 → 4 at 12/13/14/16s → 5 at 20s → 6 at 26s. Never more.\n\n**Every mass is dressed, and nothing is invented.** The blocking is authority over what EXISTS, not just what moves — otherwise the model deletes the masses it finds boring and adds architecture you never blocked:\n\n> Every lamppost, guardrail, road and terrain mass visible in the reference exists in the output in the same place, at the same scale, in the same position in frame — the opening blocks are dark-brick warehouse facades, the roadside masses are the waterfront skyline, the finale rocks are the city's tower walls. Nothing is deleted, and no structure is invented where the reference shows none.\n\n**Anatomy is never inherited from a proxy.** Tight inserts on hands and feet are where blocking leaks straight into the render:\n\n> The reference shows only WHERE hands and feet are. In the output they are always complete human anatomy — a five-fingered gloved hand with natural knuckles, a real leg in wool trousers, a real foot in a leather shoe — never the proxy's blocky shape.\n\n**A ledger for anything that happens a countable number of times.** A ritual, a reload, a set of falls: state it as a linear sequence, each step exactly once, and close with the tally:\n\n> Strictly linear, six steps in fixed order, each happening EXACTLY ONCE and never repeating; once a step is done it is done for good, and the sequence only ever moves FORWARD, never backward. Count of weapon events in the entire video: one draw, one magazine insertion, one slide rack, one shot.\n\nWithout the ledger the model loops the most cinematic beat — it will rack the slide four times because racking looks good.\n\n**Named misreads.** When a generation gets something specifically wrong, do not rewrite the description — **name the wrong reading and kill it**:\n\n> The lamp is a man-made steel structure — NOT an animal, NOT a snake, NOT any living or organic shape.\n\n> The rear of the car and its tail lights are NOT visible in this shot.\n\nThis is the highest-value edit available after a failed roll, and it is why the prompt grows rather than changes between takes.\n\n## ACTION TIMING — the beats\n\nOne block per shot or beat. Two notations; pick one and hold it.\n\n**For a continuous take**, ranges with a camera note and a closing state audit:\n\n```\n8-12s — THE SWEEP (per VID-C2: elevated rear push, swinging to profile by 12s):\n<what happens, in prose, with sub-beats on tenths and → chaining cause to effect>\nEND 12s: bodies 1 (behind him as he steps past) · standing — four ahead, holding.\n```\n\n**For a cut edit**, numbered shots ending on their cut:\n\n```\n9.33-10.33s — SHOT 10 — Interior over the centre console as in VID-C2: <what the\nframe contains>. Hard cut at 10.33s.\n```\n\nThree habits that separate a beat that works from one that does not:\n\n- **Declare the frame's contents as a closed set** when the shot is tight: *the frame holds exactly the console, the lever, his hand, and the edges of both seats.* An open description invites additions.\n- **Chain cause to effect inside one sentence** with `→`. `he overcommits a lunge → the Hero drops low and sweeps his standing leg → he hits the earth at 12s`.\n- **Put events on tenths.** `11.7s`, `19.5s`, `22.5s`. Vague beats generate vague timing.\n\nDensity: roughly 60–130 words per second of screen time is what these prompts actually run at. That is much denser than a normal video prompt, and it is the point.\n\n## AUDIO\n\n`[Timed accents]` uses the same timestamps as the beats:\n\n> 3.1s tyres light up into the burnout squeal · 7.0s drift-entry screech · 10.1s hard mechanical shifter clack · 20.3s full-speed pass-by whoosh\n\n`[Dialogue]` is a closed list — count the lines, give each a window, quote it verbatim, and forbid everything else:\n\n> Exactly TWO vocal events in the entire 30 seconds, both screamed, in English, VERBATIM: 1. 17.3-18.6s \"STOOOOOP!!\" 2. 23.4-24.0s \"You crazy!\" Nothing else is ever spoken.\n\n**Then forbid the lines it will invent anyway.** A closed list is a rule; an enumerated blacklist is enforcement, and the phrases to list are the clichés the scene invites:\n\n> FORBIDDEN — she never says any of these and no one else says anything: \"they're behind us\", \"cops\", \"go go go\", \"are you crazy\", \"you're insane\", or ANY other invented phrase. All other human voice is wordless screaming or laughing.\n\n**When lines are lip-synced, give each one a timestamp** in the same list, and say that they change nothing else:\n\n> Timing: \"You wind up for this one?\" ~18.2s · \"Three full turns.\" ~19.0s · \"Company.\" ~24.3s. Every line lip-synced; the lines never change the camera.\n\nTwo rules that stop dialogue from breaking the edit:\n\n> DIALOGUE NEVER CREATES SHOTS: spoken lines happen inside the reference's takes exactly as blocked — no cutaways to a speaker, no reverse shots, no added close-ups. If a line plays while the camera is elsewhere, the line stays off-screen audio.\n\n> A line marked off-screen must STAY off-screen — never show the speaker, never move him into frame, never route the camera to him because he spoke.\n\nModel note: dialogue direction as separate layers is minimax-h3's seat; native synced audio is Veo's. Route per `slates-model-selection` and read the model's own prompting skill before writing the audio block.\n\n## ENDING LOCK\n\nOne line, and it is the cheapest fix in the document. Models drift at the end — they hold a frame too long, add a beat after the last one, or fade somewhere the reference does not:\n\n> The film ends exactly where the reference ends: the final take runs unbroken to its last frame, and that source frame IS the final frame of the film. Nothing follows. The last five seconds follow the source exactly as strictly as the first five.\n\n## HOLD FOR THE FULL TIMELINE\n\nClose with a terminal re-assertion of only the constraints most prone to drift — five or six lines, compressed, no new information:\n\n```\nHOLD FOR THE FULL TIMELINE\n- VID-C2 camera path 1:1 — any deviation = failure.\n- Six and only six figures; the count above holds at every second.\n- IMG-A8 identity constant at every distance and through motion blur.\n- The IMG-B3 location in every frame; no subtitles, no watermarks.\n```\n\nRestating is not redundancy here. It is the last thing the model reads.\n\n## Checklist before you generate\n\n- [ ] Timings taken from `slates_blender_scene`'s `cutSeconds`, frame-exact, not rounded\n- [ ] Every reference has an explicit \"NOT inherited\"\n- [ ] The placement-only clause is present\n- [ ] The tie-break clause is present\n- [ ] The disambiguation clause is present\n- [ ] Every blocking artifact is translated — colours, marked faces, checkers, grid, black background, gaps, dark dips, resets, prop proxies, proxy limbs, strays\n- [ ] Any style/weather reference is declared NOT a keyframe and never a shot\n- [ ] The every-mass-is-dressed / nothing-invented rule is present\n- [ ] Tight inserts on hands or feet demand complete anatomy\n- [ ] Counts are stated where anything is countable, and repeatable actions carry a ledger\n- [ ] A TECHNICAL BLOCK carries format, lens realism, exposure and the blanket negatives\n- [ ] `[Dialogue]` is a closed list with a FORBIDDEN blacklist\n- [ ] An ENDING LOCK says where the film stops\n- [ ] `videoReferenceSecondsEach` matches the clip's real duration\n- [ ] A HOLD block closes it\n\n## Related\n\n`slates-previs-blocking` (producing the clip) · `slates-camera-language` (the moves being described) · `slates-dialogue-blocking` (multi-character continuity) · `slates-restyle-from-blocking` (reusing this prompt across styles) · `slates-prompting-seedance-2-5` / `slates-prompting-minimax-h3` (model-specific rules)\n",
|
|
5
|
-
"slates-camera-language": "---\nname: slates-camera-language\ndescription: Turn director vocabulary into real Blender camera rigs — orbits, floor rises, robo-arm whips, handheld, speed ramps, over-the-shoulder cuts — as bpy code. Use when building or refining the camera on a previs blocking pass, when a move needs to accelerate/hold/snap, or when someone asks for a \"cinematic\" camera and you need to convert that into an actual shot list.\n---\n\n# Camera language — from a shot list to a rig\n\nCompanion to `slates-previs-blocking`. That skill owns the workflow; this one owns the camera.\n\n## The first rule\n\n🚨 **Never build \"a cinematic camera move.\" Brief the camera the way you would brief an operator:** rails, target, height, lens, and the frame each move starts and ends on. \"Cinematic\" is not a specification, and asking for one produces the drifting slop the whole blocking workflow exists to avoid.\n\nBad: *a dynamic cinematic orbit around the subject.*\nGood: *a 3/4 orbit starting rear-left at 1.6m, ending front-right at 0.9m, frames 1–96, 35mm, easing out of the start and holding hard on the last 8 frames.*\n\nIf the user gives you the first, convert it to the second and say what you assumed.\n\n## Look it up, don't recall it\n\nBefore any constraint or operator you are not certain of, call `slates_blender_docs` (e.g. `bpy.types.FollowPathConstraint`) or `slates_blender_search_docs`. Invented enum values are the most common failure here and they often fail *quietly* — the constraint gets added, the axis is wrong, and the camera points at nothing.\n\n## The two primitives everything is built from\n\n### Target-based aiming\n\nAlmost every move in this skill is *position on a path* plus *aim at a target*. Separating them is what lets framing vary while the subject stays in frame.\n\n```python\nimport bpy\n\ntarget = bpy.data.objects.new(\"CAM_TARGET\", None) # an Empty\ntarget.empty_display_type = 'PLAIN_AXES'\nbpy.context.collection.objects.link(target)\n\ncon = cam.constraints.new('TRACK_TO')\ncon.target = target\ncon.track_axis = 'TRACK_NEGATIVE_Z' # cameras look down -Z\ncon.up_axis = 'UP_Y'\n```\n\n**Aim at a separate target, not at the subject's head.** A camera locked to the head produces dead, centred framing. Offset the target beside or ahead of the subject and the shot breathes — that small offset is most of what reads as \"real operator.\"\n\nFor a subject that should notice the camera, keyframe the target's follow with a **few frames of lag** behind each camera move. Heads catch up; they don't teleport.\n\n### Path-based movement\n\n```python\ncurve = bpy.data.curves.new(\"CAM_PATH\", 'CURVE')\ncurve.dimensions = '3D'\nspline = curve.splines.new('BEZIER')\nspline.bezier_points.add(len(points) - 1)\nfor bp, co in zip(spline.bezier_points, points):\n bp.co = co\n bp.handle_left_type = bp.handle_right_type = 'AUTO'\n\npath = bpy.data.objects.new(\"CAM_PATH\", curve)\nbpy.context.collection.objects.link(path)\n\ncon = cam.constraints.new('FOLLOW_PATH')\ncon.target = path\ncon.use_curve_follow = False # aiming is the Track To constraint's job\n\n# Animate progress explicitly rather than relying on the default path animation.\ncurve.path_duration = 96\ncurve.eval_time = 0\ncurve.keyframe_insert(\"eval_time\", frame=1)\ncurve.eval_time = 96\ncurve.keyframe_insert(\"eval_time\", frame=96)\n```\n\n**Why a path and not raw location keys:** the user can drag a control point to retime or reshape the move without you regenerating anything. That is the difference between \"re-prompt and hope\" and \"nudge it.\"\n\n## Speed — the part that reads as production value\n\nMovement at one constant speed is the tell of a machine. Real moves accelerate, hold, and snap.\n\nSpeed lives in the **f-curve handles** of `eval_time` (or of location, if you keyed it directly):\n\n- **Long, near-horizontal handle** at a key → slow near that key.\n- **Short, steep handle** → fast.\n- `interpolation = 'CONSTANT'` → no movement at all until the next key. This is how you get an absolute dead stop.\n\n```python\nfc = curve.animation_data.action.fcurves.find(\"eval_time\")\nfor kp in fc.keyframe_points:\n kp.interpolation = 'BEZIER'\n kp.handle_left_type = kp.handle_right_type = 'FREE'\n\na, b, c = fc.keyframe_points # start, middle, end\n# Speed ramp: fast in, sag in the middle, accelerate out.\na.handle_right = (a.co.x + 2, a.co.y + 18) # steep = launches fast\nb.handle_left = (b.co.x - 14, b.co.y) # flat = holds\nb.handle_right = (b.co.x + 14, b.co.y)\nc.handle_left = (c.co.x - 2, c.co.y - 18) # steep = arrives fast\n```\n\n**Zero drift at a stop.** If a move is supposed to be locked off, it must be *actually* locked — a slow crawl at a \"stop\" reads as a mistake. Hold with `CONSTANT` interpolation, or duplicate the key so the segment is genuinely flat.\n\n## The moves\n\n### Orbit\n\nCircle the subject on a path, target at subject height. Vary radius and height across the move so it does not read as a turntable. Half-orbits and 3/4 orbits look more intentional than full ones.\n\nFor a multi-scene continuous orbit, keep one unbroken `eval_time` curve and move the *world* under it — the camera never cuts, the set changes.\n\n### Floor rise\n\nPure vertical translation, **no rotation**, smooth acceleration with a slow middle. Each \"floor\" is a different set stacked on Z at a fixed interval; the subject sits centre-frame at each pass.\n\n```python\nFLOOR_H = 4.0\nfor i in range(4):\n cam.location = (0, -6, i * FLOOR_H)\n cam.keyframe_insert(\"location\", frame=1 + i * 48)\n```\n\nRotation during a rise destroys the effect. Leave it out.\n\n### Robo-arm\n\nThe whip-and-lock commercial move: a fast flight along a curved arc, an **absolute** dead stop at a completely different angle, repeat. Each relocation is roughly a third of a second; each stop is a distinct, readable frame.\n\nBuild it as a path with a control point per stop, then make the stops real:\n\n```python\nHOLD_FRAMES = 10\nfor kp in fc.keyframe_points:\n kp.interpolation = 'CONSTANT' # hold dead still between flights\n```\n\nKeep the target separate and slightly offset per stop, so each lock-off is a different composition of the same subject rather than six centred portraits.\n\n### Handheld\n\nApplied **last**, on top of a finished move. Slow organic sway, not jitter: long waves plus a barely-perceptible tremor.\n\n```python\nfor path in (\"location\", \"rotation_euler\"):\n for i in range(3):\n fc = cam.animation_data.action.fcurves.find(path, index=i)\n if fc is None:\n continue\n n = fc.modifiers.new('NOISE')\n n.scale = 120 # large scale = long lazy waves (4-6s at 24fps)\n n.strength = 0.035 # small; raise for rotation, lower for location\n n.phase = i * 7.3 # decorrelate the axes or it reads as a slide\n```\n\n**Never fast jitter, wobble or snap corrections.** Wrong-flavour handheld is more damaging than none.\n\n### Over-the-shoulder cuts\n\nPer cut: a camera position below shoulder height, a near-foreground body mass, and a target on the far face. Move barely — a slow sideways crawl. See `slates-dialogue-blocking` for who may occupy the foreground and why it matters.\n\n### Lens\n\nAnimate focal length like any other channel; a slow lens breath under a move adds a lot for nothing.\n\n```python\ncam.data.lens = 35\ncam.data.keyframe_insert(\"lens\", frame=1)\ncam.data.lens = 50\ncam.data.keyframe_insert(\"lens\", frame=96)\n```\n\n**But the lens must not drift across a cut.** Within a cut it can animate; on the cut frame it changes instantly with everything else.\n\n## Multiple cameras and cuts\n\nTwo ways, and only one of them survives contact with a 19-shot edit:\n\n- **One camera, jump-cut it.** Keyframe location/rotation/lens with `CONSTANT` interpolation on the cut frames. Fine up to a handful of cuts.\n- **A camera per setup, bound to timeline markers.** Correct for anything bigger, because each setup stays independently editable.\n\n```python\nmarker = bpy.context.scene.timeline_markers.new(\"SHOT_04\", frame=188)\nmarker.camera = cam_shot_04\n```\n\nEither way the invariant is the same: **camera, target and lens all change on the cut frame, and no frame between two setups is interpolated.** One transition frame reads as a whip-pan, and the model will reproduce it faithfully.\n\n## Verify before you render\n\n```\nslates_blender_scene\n```\n\n`cutSeconds` is your actual cut list in seconds. Check it against the shot list you were given — mismatches here become mismatched prompt timings, and the prompt is what you write next.\n\n⚠️ **On a marker-bound rig, read `cutSeconds` or `markers`, never `camera.keyframeSeconds`.** The per-setup cameras are usually static (a Track To constraint does the aiming), so the active camera's action is empty and that field reads `[]` on a perfectly good three-cut edit. `cutSeconds` resolves to whichever the scene actually used.\n\n## Related\n\n`slates-previs-blocking` (the workflow) · `slates-blocking-to-prompt` (turning these moves into prompt text) · `slates-dialogue-blocking` (OTS and eyelines)\n",
|
|
6
|
-
"slates-character-identity": "---\nname: slates-character-identity\ndescription: Build a Slates character from a reference image — generate one identity sheet and bind it to the character so the card updates live. Use when the user wants to create a character, build a character from an image, or starts a storyboard flow that needs consistent character references.\n---\n\n# Character identity sheet — Slates workflow\n\nA character's identity sheet is attached to **every** downstream generation that mentions it, so a flaw in the sheet becomes a flaw in every shot made from it. Building it well is the highest-leverage thing you can do for a project.\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n## The shape: ONE sheet, three panels\n\nSlates generates **one identity sheet per character**, bound as the character's canonical reference:\n\n| Panel | What it carries |\n|---|---|\n| **Chest-up portrait, three-quarter angle, largest panel (~25–30% of the sheet)** | The face. **This is the only place the model reads facial identity from** — every detail it will ever know comes from those pixels, so it gets the resolution. Off-frontal, never dead-on: an angled head reads its volume instantly. |\n| **Full-body front, relaxed A-pose — cropped at the collarbone, just the face cropped out** | Build, proportion, wardrobe. The face is cropped off on purpose: a front-facing body panel renders a ~40px face that can't match the portrait's, so the sheet would carry two competing identities and the model averages them. **Only the face** — neck, arms and hands render as skin. |\n| **Full-body back, head and hair visible** | Hair fall and the back of the outfit — the only panel where either reads. Keeps its head because there's no face to compete with. |\n\nThe rule is **kill every competing rendering of the FACE, not every head** — which is why exactly one body panel is headless.\n\nOn a deep neutral-grey plate (hex `3a3a3c`, emitted without the `#` — see the sigil warning in Don'ts), flat and shadowless, with catchlights in the eyes, irises never crushed to black, surface texture at the medium's own natural level of detail, broken symmetry, and no over-clean 3D-game-model look. Expression is **a slight natural smile with the teeth just visible** — a closed mouth carries no dental information, so every downstream smiling shot invents teeth, and teeth are person-specific.\n\n**Two carve-outs, scoped differently on purpose.** Non-human characters get a natural neutral expression instead of a smile — that one is scoped by *having a human mouth*, so a bipedal robot or humanoid alien is covered. Quadrupeds and non-bipedal characters get a natural standing stance with the head shown on both body panels — that one is *anatomical*. **Both are conditionals the image model evaluates against your reference; neither is a code branch, because the op has no character-kind input.**\n\n**The sheet inherits the source's medium** — photo, anime, illustration, painterly, 3D render — unless the user explicitly asks for a transform. None of the craft clauses above override that: they ask for *readable* eyes and *material-looking* surfaces within whatever medium the character is in, not for photorealism.\n\n**Why one sheet.** Every `@character` mention attaches that character's canonical identity image, so each character costs one reference slot. It also reduces competing facial renderings to **one** — with the front panel headless and the back panel turned away, the portrait is the only face on the sheet, so there is nothing left to average.\n\n## Workflow\n\n### Get the reference\nThe user has either:\n- Pasted/uploaded an image of the character (real person, drawing, AI render).\n- Described the character in text only.\n\nIf image: upload it as a reference<!-- slates-only --> (`slates_upload_reference_image`)<!-- /slates-only -->.\nIf text only: generate from prompt-only — less consistent, so warn the user.\n\n<!-- slates-only -->\n### Create the character record\n`slates_create_character` with:\n- `name` (ask if not given)\n- `description` — 1-2 sentences, *visual* only (\"tall, dark hair, scar over left eye\"), not personality.\n- `style` — leave as the source's own medium by default. Only name a transform if the user wants one (e.g. anime → realistic).\n<!-- /slates-only -->\n\n### Generate the sheet\n<!-- slates-only -->\n`slates_generate_character_identity` with `characterId`, `projectId`, and `baseAssetId` (the source portrait).\n\n**Do not hand-write the sheet prompt.** Slates builds it from the canonical template in `@slatesvideo/shared/prompts` (`buildCharacterIdentityPrompt`) — panels, plate, lighting and craft clauses included — and appends your `userNotes`. Use `userNotes` for what the template can't know: *\"use the woman on the left\"*, *\"keep the scar on the right cheek\"*. A hand-written prompt is a fork of the template and will drift from it.\n\n- Estimate cost first with `slates_estimate_generation_cost` and announce in **credits** — never quote a price from memory.\n<!-- /slates-only -->\n\n<!-- @inject:sheet-tool-defaults -->\n**What the sheet tools render on** (you do not pick these; omit `model`):\n\n- **Character identity sheet:** `gpt-image-2-5-sunburst` at 3k, quality `high`, one 16:9 image.\n- **Establishing image:** `gpt-image-2-5-sunburst` at 3k, quality `high`, one 16:9 image.\n\nPrice a sheet for that model at 16:9, with resolution and quality left at their defaults. **Never 4K** — no identity gain at sheet scale, wasted spend.\n<!-- @end:sheet-tool-defaults -->\n\n- When the result returns inline, **evaluate it before binding**:\n - Is the portrait clearly the largest panel, and is it off-frontal?\n - **Is the front body panel cleanly headless** — an empty collar above a normally rendered body, no partial face, no floating jaw, no smeared neck stump? A botched crop is worse than no crop.\n - **Is the body still there?** Neck, forearms and hands rendered as skin, not an empty outfit floating on nothing. A hollow garment means the invisible-mannequin genre ran unbounded.\n - Do the body panels read as the same build, wardrobe and hair as the portrait?\n - Catchlights present, irises readable rather than black holes?\n - Is it in the source's medium, and does it read as *that* medium done well — or has it drifted toward the over-clean game-model look?\n - Plate a flat deep grey, not white and not black?\n- If off: one focused refinement, then regenerate. The sheet is upstream of everything — it is worth a re-roll that a scene frame is not.\n<!-- slates-only -->\n- The op binds the result as the canonical identity automatically.\n\n### Hand back\n> \"Character {name} ready — identity sheet bound. Use `@{name}` in any prompt and Slates attaches it and names it inline, so the face stays consistent.\"\n<!-- /slates-only -->\n\n## How the reference gets used at scene time\n\nSlates cites the sheet inline under the character's name — `{name} (image N)` — in the exact order it sends references. That **name** is the anti-averaging lever, and it is each model's own official mechanism (NB2: \"assign a distinct name\"; Seedance: `Reference <Subject_N> in <Image_N>`; Kling: reuse a fixed label verbatim).\n\nCritically, the app injects **no** wardrobe, expression, or lighting directive. The user's scene prompt owns all of that — which is why `@{name}` dropped into a movie-still injection keeps the still's own clothing and lighting instead of dragging the sheet's.\n\n## Anti-patterns\n\n- **Don't** studio-light, white-background, or black-background the sheet. White bleeds into the video and washes out the location; black eats edge detail. Flat, even, shadowless light on a deep neutral grey.\n- **Don't** hand-write the sheet prompt when the op will build it — that is how the template and the shipped prompt fork.\n- **Don't** create a second character image. One canonical identity is what the storyboard pipeline reads.\n- **Don't** skip binding. An unbound asset doesn't help downstream.\n- **Don't** invent character details. Stick to what's in the reference image and the user's description.\n- **Don't** describe the front panel's crop as an absent head — in `userNotes` or any hand-written variant. The template asks for it as *framing*: **\"cropped at the collarbone, an invisible-mannequin presentation with just the face cropped out\"**, a standard e-commerce genre with deep training data. **\"the head not shown\" is a hard 422 on GPT Image** (measured on `gpt-image-2`, the model 2.5 replaced; the classifier is OpenAI's, not the version's, so the rule carries — but nobody has re-run it on Flare or Sunburst) — fal returns `content_policy_violation` with `loc: [\"body\",\"prompt\"]`, so the text is rejected before any image is read, because an anatomical absence reads as gore to OpenAI's classifier. It passed NB2, which is why the original receipt looked safe: **it was model-scoped.** State an exclusion as a framing choice, never as a missing body part.\n- **Don't** invoke the invisible-mannequin genre without bounding it to the face. **\"an invisible-mannequin presentation where the clothing holds its own shape\" removed all the skin** — no neck, no hands, no forearms, a garment floating on nothing — because that *is* the e-commerce genre in full: an empty outfit. **\"with just the face cropped out\"** keeps the anchor and bounds it. Generalises: a genre anchor imports the whole genre, so name what STAYS, not only what goes.\n- **Don't** put `#` or `@` anywhere in prompt text. Both are reference-token sigils in the desktop prompt composer and an unresolved one is **silently deleted** — no error, no log, just missing words. `#3a3a3c` reached fal as `background ()` on a real 2026-07-30 request, meaning the plate value had never been delivered to any model since the composer shipped. Write hex values bare.\n- **Don't** use 4K — wastes credits, no quality gain at sheet scale.\n- **Don't** feed a multi-view sheet into a Seedance shot that has **several characters in frame** without binding each character to its image and appending the anti-twin constraint — ByteDance documents multi-view assets as a cause of duplicate characters. See `slates-prompting-seedance`.\n",
|
|
7
|
-
"slates-chatgpt-images": "---\nname: slates-chatgpt-images\ndescription: Generate images
|
|
8
|
-
"slates-cinematic-look": "---\nname: slates-cinematic-look\ndescription: Use when a frame should look filmed, not generated (light, exposure, grade, lens, atmosphere, imperfection), or when an image came back too clean or studio-lit. The organized technique catalogue for every image model, and the rule for picking only what the shot needs.\n---\n\n# Cinematic look — make a generated frame read as filmed\n\n<!-- @card:start -->\n**Image models default to clean, evenly lit and fully exposed.** Real film frames can be dark, flat, murky or burned out. Describe what the camera sees; a mood word or look reference alone does not get you there.\n\n**Look at every reference first.** Write a look reference's grade and imperfections into the prompt: darkness, contrast, black level, colour cast/saturation, softness/noise and subject separation. Never grade cleaner, brighter or higher-contrast than that reference unless asked. Inspect the character sheet's garments. Cite references inline: `the woman from image 1`, `lit and graded like image 2`; never open with a reference-role paragraph. References are optional; these techniques work from words alone.\n\nFor a new photographic frame:\n- **One physical light system:** source, position, effect on the subject; no light without a source. If using atmosphere, put it in front too.\n- **Exposure as it looks:** `close to a silhouette, features only just readable`, not a stop under.\n- **Lens name plus effect, every time:** `200mm telephoto`, `the peaks loom huge behind her and melt into soft shapes`.\n- **Name every garment and close the foreground:** say exactly what is there; omissions invite reference leakage or invented props.\n- **Only what the shot needs:** one light system, at most one exposure, atmosphere and colour choice, one or two composition moves and one moment. Add the imperfections the scene calls for.\n\nA scene reference owns the grade; a look-only reference does not own the new scene's light. For an owned-frame edit, describe only the change and what stays. Never use a released film frame as the edit base; use it as an art-direction brief for a new scene. Use positive descriptions first; one targeted negative is enough where supported. FLUX needs positive wording.\n\nUse query with a technique ID or section, depth \"index\" to browse, or depth \"full\" for the catalogue and examples.\n<!-- @card:end -->\n\n**Evidence scope:** most Slates receipts come from single generations of one scene on GPT Image 2.5 Sunburst. They show useful directions to test, not reliable success rates or a universal ranking. The research record owns prompts, source links and observations: `business/projects/slates/research/cinematic-look-research.md` in the vault. This skill owns active technique wording. The catalogue checker compares IDs, evidence tags and worked-example bytes.\n\nImage models default to a clean, evenly lit, fully exposed picture. Real film frames are often graded \"wrong\": underexposed, backlit, silhouetted, burned out, flat or murky. A model only goes there when the prompt says what that looks like. A look reference alone does not get you there; describing the frame does.\n\n## Two routes to a filmed frame\n\n1. **Describe a new frame.** Write the scene and reference roles inline, then the light, exposure, camera and texture choices needed to realize it.\n2. **Edit a frame you own.** When a Slates plate or sheet, your own photo or footage, or a Blender render already has the composition and look, describe only the requested changes and what must remain. Example: \"Take image 1 and change only the character to the character in image 2.\" Never use a frame from a released film as the base; a film still is an art-direction brief for route 1.\n\n## The rules\n\n1. **Describe what the camera sees, not the camera setting.** \"Her face sits a stop under\" was ignored; \"close to a silhouette, her features only just readable\" landed. Mood words carry little on their own.\n2. **Build one physical light system first.** Name the source and its position, what it does to the subject, and rule out light with no source. If the shot needs atmosphere, place it in front of the subject as well as behind. Every source and shadow must agree.\n3. **Name the lens and describe its effect, every time you describe a new photographic frame.** A lens named alone changed nothing visible in the summit test; named with its effect, it produced compression and blur.\n4. **Use only what the shot needs.** One light system (its consequence clauses count as one), at most one pick each from exposure, atmosphere and colour, one or two composition moves and one moment. Then stop adding. These are drafting limits, not model capability limits.\n5. **Name everything a reference could fill.** Every garment, and a closed list of what is in the foreground. Anything left out can come from a reference or be invented. In route 2, preserve the existing inventory and describe only the change.\n6. **One targeted negative after a positive description is fine; a list is not.** Follow the model's grammar: FLUX needs positive wording.\n7. **A scene reference owns the grade; a look-only reference does not own the light.** With a look-only reference, still write the light direction and visible exposure for the new scene. Name references inline where used, never in an opening role paragraph.\n8. **Look at every reference before writing.** First describe a look reference's own grade and imperfections: darkness, contrast, true or muddy blacks, colour cast and saturation, softness and noise, and how much the subject separates from the background. Never grade cleaner, brighter or higher-contrast than the reference unless the user asks. Inspect a character sheet's garments so the prompt names or replaces every one. Without a reference, the same techniques work from words alone.\n\n## Build order\n\nInspect references first. For a new frame: time/weather → light source and direction → effect on the subject → exposure as it looks → atmosphere in front, if needed → colour → composition and camera placement (lens plus effect) → the moment → kind of picture → inventory (every garment and the closed foreground). Carry the observed reference grade through those choices. Adapt examples to the actual scene; never inherit their props, wardrobe or light by accident.\n\n## Evidence tags\n\n`receipt` measured on Slates generations · `vendor` stated in the model maker's own guide · `practitioner` a published third-party guide or test · `canon` established cinematography, never measured on a model · `untested` reasoning only: try it, then record the result in the research doc and change the tag.\n\n## The catalogue\n\nEach row: the technique, its evidence, what it does to the frame, when to reach for it and when to skip it, and wording written as what the frame looks like.\n\n<!-- @catalogue:start -->\n\n### 1. Exposure and tone — the \"graded wrong\" frame\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `flat-underexposure` | receipt | The whole frame sits in a narrow, dark tonal range; the subject barely separates | A dark, flat look reference or fading light · skip hard contrast, glowing practicals and crushed blacks | \"Everything sits in dark, muddy navy blue; nothing is bright or truly black. Her dim face barely separates from the trees; even the focused face is slightly soft, with fine noise in the dark blues.\" |\n| `subject-under-key` | untested | The face is clearly darker than the brightest part of the frame; features read, nothing lights them | Backlit exteriors, sunrise and sunset, window interiors · skip beauty, product, lip-sync close-ups | \"Her face is in shadow, clearly darker than the sky behind her; her features are readable but nothing lights them.\" |\n| `near-silhouette` | receipt | A dark shape with edge detail against a bright field | Wides, entrances and exits, solitude · skip shots that depend on recognising the face | \"She is close to a silhouette: her face and jacket fall into deep shadow, her features only just readable.\" |\n| `clipped-highlights` | receipt | The sky near the sun, windows and practicals go pure white | Backlit shots, windows, night practicals · skip skies that carry the story, white packaging | \"The sky around the sun burns out to white.\" |\n| `dense-shadows-with-ramp` | receipt | Shadows sink to near-black but fall off gradually, and one detail survives in the dark | Night, interiors, a backlit foreground · skip video dark work that already turns to murk | \"The shadows are dense and slightly crushed, the foreground rock nearly black, and the light fades into them gradually.\" |\n| `low-key-ratio` | vendor | One side of the face lit, the other falls away with nothing filling it | Interiors, night, close-ups that carry mood · skip bright comedy or commercial register | \"Light reaches only the left side of his face; the right side falls into shadow and nothing fills it.\" |\n| `flare-washed-contrast` | receipt | Stray light lifts the blacks and flattens contrast over part of the frame | Shooting toward the sun or a hard practical · skip crisp thriller, neon noir | \"Flare washes across the upper half of the frame and lowers the contrast.\" |\n| `lifted-matte-blacks` | canon | The darkest tones sit at charcoal, like a faded print | Daytime melancholy, period looks, overcast · never with `dense-shadows-with-ramp` | \"The darkest parts of the frame are a soft charcoal rather than black.\" |\n| `uneven-exposure-across-frame` | untested | Exposure is right for one zone only; one side runs hot, the far side goes dark | Mixed-light interiors, night streets, documentary register · skip clean product shots | \"The lamp side of the room is overexposed and the far corner goes nearly black.\" |\n\n### 2. Light source and direction — one physical system\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `name-the-one-source` | receipt | A dominant source with shadows consistent with its distance | When the light needs control · skip when preserving existing light | \"The sun is low behind her and off her right shoulder, just above the far peaks.\" |\n| `forbid-the-phantom-key` | receipt | Removes the soft front light models invent on faces | Any backlit, side-lit or practical-lit person · skip flash, frontal sun, beauty work | \"There is no light in front of her: no frontal key, no fill.\" |\n| `hard-rim-backlight` | receipt | A bright edge on hair and shoulder while the face stays in ambient light | Golden hour, practicals behind the subject · skip overcast and blue hour, which have no hard source | \"A hard orange rim of light traces her hair, the edge of her cheek and one shoulder.\" |\n| `face-in-bounce` | receipt | The unlit side is filled only by coloured light reflected from the surroundings | Backlit exteriors, rooms with coloured walls · skip when the face must read cleanly | \"Her face sits in soft, cooler bounce light off the rock.\" |\n| `raking-side-light` | receipt | Light skims a surface so pores, weave and grain show | Close-ups, skin realism, materials · skip when the brief is flattery | \"Daylight rakes across her face from one side, so texture catches along the cheekbone and the other side sits in soft shadow.\" |\n| `practicals-only` | vendor | Lights inside the frame are the only light: pools with dark gaps between | Night interiors, cars, bars, kitchens at night · skip when the room must read | \"The only light is the open fridge she is standing in; the kitchen behind her is dark.\" |\n| `window-light-falloff` | canon | Bright near the window, then a fast drop across the room | Day interiors, quiet drama · skip big evenly lit spaces | \"Grey daylight from the single window on the left; a few steps away the room drops into shadow.\" |\n| `mixed-colour-temperatures` | receipt | Warm and cool sources side by side, uncorrected | Dusk interiors, night streets, cars at night · skip clean commercial | \"Warm lamplight on his face; cold blue daylight from the window on the wall behind him.\" |\n\n### 3. Atmosphere and optics — what sits between the lens and the subject\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `haze-in-front` | receipt | Haze, dust or smoke between camera and subject softens the subject's edges | Exteriors, sunbeams, dusty sets · skip crisp product or text shots | \"Thin haze crosses the frame in front of her, so her edges are no sharper than the rock beside her.\" |\n| `source-flare` | vendor | Flare from a bright source in the frame | Facing the sun, headlights, stage lights · skip jargon stacks and scenes with no hard source | \"The low sun at the edge of the frame throws a flare that crosses in front of her.\" |\n| `halation-on-highlights` | canon | A soft reddish glow bleeds past bright edges | Night practicals, candles, neon, film looks · skip clean digital register | \"Bright lights have a soft reddish glow bleeding past their edges.\" |\n| `defocus-as-outcome` | receipt | Planes separate: only the subject is sharp | Close and medium shots · skip wides where the place must read | \"On a 200mm telephoto lens only she and the pan are sharp; the background melts into soft shapes and the foreground rock edge falls out of focus.\" |\n| `soft-overall-no-sharpening` | receipt | Lower microcontrast, no halos along edges | Photoreal people, film register · skip product detail and dense text | \"The image is slightly soft overall; edges carry no crisp outline.\" |\n| `grain-in-shadows` | vendor | One texture note | Film or low-light register · never stacked with other noise words | \"Fine grain is visible in the shadows.\" |\n| `edge-falloff` | canon | The corners sit darker than the centre | Film looks, night · skip flat graphic compositions | \"The corners of the frame are darker than the centre.\" |\n| `glass-or-weather-between` | receipt | Rain, smudges or condensation between the lens and the subject | Cars, cafés, storms, observed framing · skip when the face must be sharp | \"Seen through a rain-streaked car window; the drops are sharp and she is soft behind them.\" |\n\n### 4. Colour and grade — preserve the reference unless a change is requested\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `warm-muddy` | receipt | Warm but desaturated; whites read cream, greens olive | Dusty or earthy exteriors, fatigue, nostalgia · skip crisp commercial | \"The colour is warm and a little muddy rather than clean.\" |\n| `restrained-desaturated` | canon | Muted overall; at most one colour stays strong | Drama, cold or bleak moods · skip joyful or brand-colour work | \"Colours are muted and low in saturation; only the red of her jacket holds its colour.\" |\n| `uncorrected-white-balance` | receipt | The colour cast is left in | Phone or documentary register, shade, fluorescents · skip product colour accuracy | \"White balance left a little cool and uncorrected; the white mug looks bluish.\" |\n| `tungsten-in-daylight` | canon | The daylight scene renders blue-cyan | Cold mornings, alienation · skip warm romance | \"Daylight renders cold and blue, as if the camera were set for indoor lamps; skin looks pale.\" |\n| `sodium-vapour-mono` | canon | Street light collapses colour to amber | Urban night, industrial areas, parking lots · skip scenes that need colour separation | \"Orange streetlight flattens every colour to amber and brown; shadows go brown-black.\" |\n| `bleach-bypass-look` | canon | High contrast, drained colour, silvery skin | War, grit, harsh drama · skip warm or intimate scenes | \"Colour almost drained out, contrast harsh, skin grey-silver, heavy shadows.\" |\n| `teal-orange` | vendor | Warm skin against teal shadows; the generic blockbuster look | Only when the brief asks for blockbuster register · never as a default | \"Skin stays warm while the shadows and sky are pushed toward teal.\" |\n| `era-or-device-register` | vendor | One phrase shifts colour, grain and flash together | Period or amateur looks · one per prompt | \"As if shot on 1980s colour film, slightly grainy.\" |\n| `cross-processed` | vendor | Hard colour shifts: cyan shadows, magenta highlights | Music video, fashion, 90s editorial · skip naturalism | \"Colours shifted hard, cyan in the shadows and magenta in the highlights.\" |\n\n### 5. Time and weather presets — bundles of the above, and mutually exclusive\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `golden-hour-backlit` | receipt | Sun low behind, long shadows toward camera, rim light, face in bounce, sky near the sun white, warm haze | Warm exteriors · never with noon shadows or overcast softness | \"The sun sits just above the ridge behind her; long shadows run toward the camera; a hard rim on her hair; her face in cool bounce; the sky around the sun burns white.\" |\n| `after-sunset` | canon | No direct sun, soft shadowless light, sky fading pink to blue, practicals just on | Quiet exteriors · never with rim light or hard shadows | \"The sun has just set; soft shadowless light; the sky fades from pink to deep blue; porch lights have just come on.\" |\n| `blue-hour` | receipt | Cool even light; warm practicals run hot against it; faces dim | Streets and cafés at dusk · never with a warm key on the face | \"Deep blue dusk light, almost no shadows; the café windows glow hot orange; her face is dim and blue.\" |\n| `hard-noon` | practitioner | Bleached sky, short hard shadows, squinting | Heat, desert, exhaustion · skip anything that must flatter | \"Midday sun straight overhead; tiny hard shadows under her brows and chin; she squints.\" |\n| `overcast-flat` | practitioner | No shadows, low contrast, honest skin | Plain daylight, street realism · never with rim light or flare | \"Flat grey overcast light, no shadows, low contrast, colours slightly dull.\" |\n| `night-practicals` | canon | Pools of light, dark gaps, colour casts, blooming highlights | Night streets and interiors · never with \"evenly lit\" | \"The street is dark between the orange streetlights; he is lit only when he passes under one.\" |\n| `firelight` | canon | Warm flicker from below, fast falloff, black past a few metres | Campfires, candles · never with daylight fill | \"The campfire is the only light: warm and flickering from below, their faces half lit, everything beyond the circle black.\" |\n| `moonlight-day-for-night` | canon | Cool, low saturation, one faint hard shadow direction, sky darker than the ground | Night exteriors · skip when warm practicals dominate | \"Cold blue moonlight from one side; faint hard shadows; colour almost gone; faces only just readable.\" |\n| `rain-wet-night` | canon | Wet surfaces smear reflections; rain shows only where it passes a light | Urban night · rain across the whole frame, never in one corner | \"Rain shows as bright streaks where it passes the streetlight; the wet asphalt reflects the lights as long smears.\" |\n| `fog-or-dust` | canon | Depth falls off to grey; figures layer by distance | Mystery, scale · never with crisp distant detail | \"Fog swallows everything past the second streetlight; the far figures are pale grey shapes.\" |\n\n### 6. Composition and camera placement\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `dirty-foreground-occlusion` | receipt | An out-of-focus object near the lens covers part of the frame | Anything that should feel observed rather than staged · only objects that belong in the space | \"The edge of a pine branch crosses the left third of the frame, close to the lens and completely out of focus.\" |\n| `off-centre-cropped-subject` | practitioner | The subject sits near an edge, partly cut by the frame | Candid and documentary register · skip deliberate symmetry | \"She sits in the right third of the frame; her elbow is cut off by the edge.\" |\n| `negative-space` | canon | A small subject in a large empty area | Isolation, scale · in a video plate give the empty area texture, or it moves with the foreground | \"She is a small figure at the bottom right; the rest of the frame is pale, cloud-streaked sky.\" |\n| `frame-within-frame` | canon | A doorway, window or vehicle surrounds the subject | Observed feel, confinement · skip when it hides the action | \"Seen through the open barn door; the dark door frame surrounds her on three sides.\" |\n| `over-the-shoulder-foreground` | vendor | A foreground head or shoulder, dark and soft | Dialogue, two-person scenes | \"The back of his head and shoulder fill the left edge, dark and out of focus; she faces him, sharp.\" |\n| `compression-as-outcome` | receipt | The long-lens look: the background looms huge and close | Making a background loom, crowds, heat haze · skip intimate interiors | \"Shot from far away on a 200mm telephoto lens, the peaks loom huge and close behind her, stacked right up against her shoulders.\" |\n| `camera-height` | receipt | Ground level, hip height or overhead, picked on purpose | Every shot · eye level only by choice | \"The camera sits on the ground by the stove, looking up at her past the pan.\" |\n| `grabbed-framing` | untested | Horizon slightly off, framing a beat late | Documentary or UGC register · skip composed cinema | \"The horizon tilts slightly and the framing is a little late; her head is near the top edge.\" |\n| `reflection-partial` | receipt | The subject seen in glass, steel or a puddle | Night, cities, variety across a set · never a bathroom mirror | \"We see her only as a reflection in the dark shop window, overlapped by the street behind the glass.\" |\n\n### 7. Texture, wardrobe and set\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `name-the-capture-context` | receipt | Says what kind of picture this is, instead of listing flaws | Every photoreal shot · never a flaw inventory, which reads as tokens and turns plastic | \"A frame from a film shot on location, not a studio portrait.\" |\n| `skin-under-real-conditions` | vendor | Skin carries the environment: wind, sun, sweat | People in real conditions · never a stack of pore and blemish words | \"Wind-chapped cheeks and a sunburnt nose after a day on the mountain; skin shiny with sweat at the hairline.\" |\n| `hair-state` | receipt | Flyaways, wind, strands stuck to the forehead | Weather, action, fatigue · vary the state, never the style that carries identity | \"Wind has pulled strands loose across her face.\" |\n| `wardrobe-wear` | vendor | Creases, dust, fading | Lived-in characters · skip fashion hero shots | \"Her jacket is creased at the elbows and faded at the seams; dust on the knees.\" |\n| `full-wardrobe-spec` | receipt | Names top, legwear and footwear so a reference cannot fill the gap | Any shot built from a character reference · not optional | \"Grey hiking jacket zipped up, dark hiking trousers, scuffed brown boots.\" |\n| `closed-prop-list` | receipt | A positive, closed inventory of what is in the frame | Any frame where the model adds junk · never a \"no extra props\" list | \"The only things on the rock are the stove and the pan.\" |\n| `lived-in-wear-on-named-things` | practitioner | Wear goes on objects already named, never as new objects | Sets that should feel used · never \"add clutter\" | \"The pan is blackened underneath and the rock around the stove is stained with old soot.\" |\n| `material-specificity` | vendor | Names the physical material | Hero objects · never a generic noun | \"A dented enamel mug, a scratched aluminium pot, a waxed-cotton jacket going pale at the seams.\" |\n\n### 8. Moment, performance and motion\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `caught-mid-action` | receipt | The action is under way and the camera goes unnoticed | People, by default · skip deliberate portraits and address-the-lens beats | \"Her right hand works a wooden spatula in the pan mid-stir.\" |\n| `eyeline-off-lens` | receipt | The eyes are on the task, another person or out of frame | Images and B-roll · never a lip-sync beat, where eyes on the lens are the point | \"She looks down at the pan.\" |\n| `unresolved-expression` | receipt | Not a stock smile: squinting, chewing, tired, mid-thought | Realism · skip brand-joy beats | \"She squints against the glare, jaw set, tired.\" |\n| `motion-blur-on-the-mover` | practitioner | Only the moving part blurs | Hands, tools, hair, passing vehicles · never on text or a face that must read | \"Her hand with the spoon is a slight blur of movement; the pan and the rock are sharp.\" |\n| `focus-slightly-missed` | untested | Focus landed just behind the subject | Documentary register · skip identity-critical shots | \"Focus landed on the rock just behind her; her face is a touch soft.\" |\n| `handheld-operator-body` | receipt | For video: write the body holding the camera, not the path | Documentary, UGC, tension · skip locked-off formal shots | \"Handheld, the operator breathing; the frame sways slightly, drifts off her and corrects back.\" |\n| `weight-and-consequence` | practitioner | For video: mass moves through the body and the world reacts | Any physical action · skip static dialogue | \"She shifts her weight onto her back foot as she lifts the heavy pan; the stove wobbles.\" |\n| `frame-zero` | receipt | In a still that feeds video, the event has not happened yet | Every image-to-video plate · never an aftermath plate | Describe the moment just before the event: the pole still straight, the glass still whole. |\n\n<!-- @catalogue:end -->\n\n## Techniques that clash\n\n- **`near-silhouette` against a recognisable face or lip-sync.** On identity-critical shots use `subject-under-key` with `face-in-bounce`; keep silhouettes for wides.\n- **`dense-shadows-with-ramp` against murk.** Always say the light falls off gradually and keep one detail readable in the dark. On video dark work keep the subject readable; `slates-blocking-to-prompt`'s \"no crushed blacks\" is scoped to that lane.\n- **`lifted-matte-blacks` against `dense-shadows-with-ramp`.** Pick one.\n- **One colour-temperature story.** `warm-muddy` does not go with `tungsten-in-daylight` or a cool `uncorrected-white-balance`.\n- **Presets exclude each other.** Golden hour has no noon shadows; overcast has no rim light or flare; `practicals-only` has no even room light.\n- **Haze or flare against readable text or product.** Keep the atmosphere away from the text, or drop it.\n- **`defocus-as-outcome` against a place that must read.** Keep it for close and medium shots.\n- **Motion blur, handheld sway or missed focus against text, lip-sync or identity.** Keep them off the face and off the text.\n- **`closed-prop-list` against `lived-in-wear-on-named-things`.** Wear goes on the named objects, never as new ones.\n\n## On video\n\n- **The grade lives in the still.** A motion prompt describes how the light behaves as things move — the flare slides as the camera turns, haze drifts across the foreground, her face stays in shadow as she turns — and preserves the still's grade unless the user requests a change.\n- **Lens and film-stock names translate; they do not paste.** The rule and ByteDance's own wording: `slates-prompting-seedance` → \"Don't cross-pollinate image-model syntax\".\n- **A negative-prompt field can cancel a technique.** Never suppress in `negative_prompt` something the prompt asks for (Kling: `slates-prompting-kling-v3`).\n\n## Model notes\n\n- **GPT Image 2.5.** Ask for a real photograph or film still outright. In the recorded summit tests, it tended to brighten faces, clean up colour and add props; visible-outcome wording gave more control than gear names alone. References route through its edit endpoint.\n- **Nano Banana 2 and Pro.** Google recommends named cameras, film eras, chiaroscuro lighting and positive framing, so gear names are a sanctioned lever here: still add what they do to the picture. The recorded Nano Banana Pro edits changed more of the frame than intended; scope each requested change and name what should stay.\n- **FLUX.2 Max.** No negative prompt at all, and word order is weight, so put the light system early. Its own examples use crushed shadows, blown highlights and era looks.\n- **Seedream 5 Lite.** Keep it short: pick fewer techniques to fit its prompting guide's roughly 100-word ceiling. See `slates-prompting-seedream-5-lite`.\n\n## Worked examples\n\n**Route 1, IMG-A197 (Sunburst, 2026-09-15).** Image 1 is the character identity sheet, image 2 a look reference. One light system, one exposure decision, the wardrobe and the foreground named. Eric: *\"just so well done.\"*\n\n<!-- @example:img-a197:start -->\n```text\nA film still shot on location, lit and graded like image 2. The woman from image 1 cooks on a rocky summit high in the Rockies at golden hour. She stands facing camera behind a stainless steel frying pan of sliced vegetables on a camp stove on the rock in the lower foreground, framed from mid-thigh up. Her right hand works a wooden spatula in the pan mid-stir, her left rests on the pan handle. She looks down at the pan with a slight smile. Long wavy blonde hair down. Grey hiking jacket zipped up, the word \"SLATES\" once in small plain letters on the left chest, and dark hiking trousers. The only things on the rock are the stove and the pan. Behind her: pine tops, a deep valley and snow-capped peaks running to the horizon.\n\nThe sun sits just above the far peaks behind her right shoulder and the whole frame is exposed for that sky. She is close to a silhouette: her face and jacket fall into deep shadow, her features only just readable, lit by nothing but a faint warm bounce off the rock. A hard orange rim of light traces her hair, the edge of her cheek and one shoulder. The sky around the sun burns out to white, flare washes across the upper half of the frame and lowers the contrast, and steam off the pan glows where the sun comes through it. The shadows are dense and slightly crushed, the rock in the foreground is nearly black, and the colour is warm and a little muddy rather than clean. No other text in the image.\n```\n<!-- @example:img-a197:end -->\n\n**Route 1 with a lens, IMG-A198 (Sunburst, 2026-09-15).** The same frame, with the lens named and its effect described. Real compression and depth of field came back.\n\n<!-- @example:img-a198:start -->\n```text\nA film still shot on location from far away on a 200mm telephoto lens, lit and graded like image 2. The woman from image 1 cooks on a rocky summit high in the Rockies at golden hour. She stands facing camera behind a stainless steel frying pan of sliced vegetables on a camp stove on the rock in the lower foreground, framed from mid-thigh up. Her right hand works a wooden spatula in the pan mid-stir, her left rests on the pan handle. She looks down at the pan with a slight smile. Long wavy blonde hair down. Grey hiking jacket zipped up, the word \"SLATES\" once in small plain letters on the left chest, and dark hiking trousers. The only things on the rock are the stove and the pan.\n\nThe long lens compresses the distance: the snow-capped peaks loom huge and close behind her, stacked right up against her shoulders, and they melt into soft out-of-focus shapes. Only she and the pan are sharp. The pine tops between her and the peaks are a smear of dark green, and the foreground rock edge falls out of focus too.\n\nThe sun sits just above the far peaks behind her right shoulder and the whole frame is exposed for that sky. She is close to a silhouette: her face and jacket fall into deep shadow, her features only just readable, lit by nothing but a faint warm bounce off the rock. A hard orange rim of light traces her hair, the edge of her cheek and one shoulder. The sky around the sun burns out to white, flare washes across the upper half of the frame and lowers the contrast, and steam off the pan glows where the sun comes through it. The shadows are dense and slightly crushed, the rock in the foreground is nearly black, and the colour is warm and a little muddy rather than clean. No other text in the image.\n```\n<!-- @example:img-a198:end -->\n\n**Route 1, platform revision (IMG-A200 and IMG-A204).** IMG-A199 followed its written hot lamp and crushed blacks, but Eric wanted the reference's flatter, darker, softer grade. This exact revision subsequently produced A200 and A204 using neutral identity sheets and the original look reference. Eric preferred A204 to the scene-reference remixes A201–A203; A203 and A204 used the same model and quality settings. Prompt and references changed together, so their individual contributions are not isolated.\n\n<!-- @example:platform-flat-untested:start -->\n```text\nA film still shot on location from a distance on an 85mm lens, lit and graded like image 2. The young woman from image 1 waits alone at the far end of an empty country train platform at blue hour. She stands side-on to the camera in the right third of the frame, framed from the chest up, looking down the empty track toward where a train would come from, her lips slightly parted. A dark wool coat hangs open over a grey hooded sweatshirt, hood down. The wind has pulled a few strands of hair loose across her cheek. The only things near her are the edge of the concrete platform and a single old lamp post, its lamp not yet switched on.\n\nThe sun is gone and the light is almost gone with it. The whole frame is underexposed and flat: everything sits in a narrow range of dark, muddy navy blue, nothing in it is bright and nothing is truly black. The brightest thing in the picture is the dull grey-blue sky above the trees. Her face is lit only by that weak sky from behind and to her left, so it is dim and blue and barely separates from the dark trees behind her; her features are there, but you have to look for them. There is no light in front of her, no rim of light on her hair, no catchlight in her eyes, and no contrast anywhere to make her stand out.\n\nThe long lens compresses the distance: the trees sit close behind her as soft dark shapes, and only her face is in focus, and even that is slightly soft. It looks like a camera pushed to its limit in low light: low contrast, a little murky, fine noise in the dark blues, and colour drained to a cold blue-grey. No text anywhere in the image.\n```\n<!-- @example:platform-flat-untested:end -->\n\n**Route 2, IMG-A184 (Sunburst, 2026-09-09).** Image 1 was a finished blue-hour frame, image 2 a character. The whole prompt: *\"take this image 1 and just change the character to the character in image 2\"*. Composition, grade, light and depth of field held exactly. That base frame was not one Slates owns, so the receipt proves the technique, not a shippable workflow: use your own frame.\n\n## Adding or changing a technique\n\n1. Record the evidence first: a row in the research doc's technique table, with its tag and source.\n2. Add or change the row here, with the same id and the same tag.\n3. Run the build. The catalogue check fails until both files agree.\n",
|
|
9
|
-
"slates-content-policy": "---\nname: slates-content-policy\ndescription:
|
|
10
|
-
"slates-cost-discipline": "---\nname: slates-cost-discipline\ndescription: Mandatory pre-flight discipline before ANY generation call (image or video) — estimate cost, announce in credits, get confirmation, aggregate batches. Use its rules when planning generation; reload only when the rules are needed. Skipping this risks burning the user's credits on guesses.\n---\n\n# Slates cost discipline — read before every generation\n\nGeneration costs real money. Every call is on the user's credits. The user can't see what you're about to spend until you tell them. **Tell them first, generate second.**\n\n## The 4 rules\n\n### 1. Pre-flight estimate — never call generate without one\n\nBefore ANY `slates_generate_*` call, run `slates_estimate_generation_cost` first. Inputs you must lock before estimating:\n\n- **Model** — the id you are about to pass, whatever it is. `slates_estimate_generation_cost` takes the same base ids the generate ops take and resolves the billing key itself; do not build one by hand.\n- **Resolution** — use the selected model default unless the user or delivery requires another size. Estimate and generate with the same settings.\n- **Aspect ratio** — never let the op default to 1:1. Pick from the use case (cinematic → 16:9, mobile vertical → 9:16, square feed → 1:1).\n- **Count** — explicit. Don't generate 4 when 1 will tell you if the prompt works.\n\nIf the aspect ratio cannot be inferred from the intended delivery, ask. A missing resolution uses the model default; it does not require another question.\n\n### 2. Announce in credits, plainly, before spending\n\nSlates bills abstract **credits** (they never expire). Announce the credit total the estimate returns — never dollars.\n\nFormat: `About to spend N credits on M image(s) at [resolution] [aspect ratio]. Proceed?`\n\nExamples:\n- `About to spend 4 credits on 1 image at 1k 16:9. Proceed?`\n- `About to spend 24 credits on 4 images at 2k 9:16 (variants). Proceed?`\n\nAnnounce the cost once, then proceed for anything small. For anything the user would notice on their balance, wait for an explicit yes. Where the LINE is, exactly:\n\n<!-- @inject:thresholds -->\n<!-- GENERATED from @slatesvideo/shared — do not edit between the markers.\n Source: CONFIRM_CREDITS, DEVIATION_FACTOR and the audio bounds in\n packages/shared/src/operations/index.ts. Every number here is REFUSED by an\n op when a prompt gets it wrong, which is why none of them is typed by hand\n any more: this block replaced four claims that contradicted the code. -->\n\n**The thresholds, from the code that enforces them:**\n\n- **Confirm gate:** above **17 credits** an op returns `requires_confirm` and will not\n proceed until you re-call with `confirm: true`. Below it, announce the cost once and go.\n- **Deviation pause:** the desktop Studio Agent stops and re-asks when projected generation spend\n exceeds the approved plan by more than **20%**. You do not trigger this; the app does.\n- **Seed Audio duration:** **3–120 seconds.** There is no duration\n parameter on the model — the number you pass is written into the prompt AND is what the user is\n billed. Outside that range the op refuses rather than clamping.\n- **Sound Effects duration:** **1–22 seconds**, billed per second, never left for the\n model to pick.\n\nNever quote a credit figure from memory: `slates_estimate_generation_cost` returns the real one.\n<!-- @end:thresholds -->\n\n### 3. Aggregate batches into ONE upfront announcement\n\nIf you're planning a multi-call workflow (5 storyboard frames, 3 character variants, a grid of options), **announce the total before the first call**, not five small announcements after the fact.\n\nFormat: `Plan: N generations totaling C credits. [Brief description of the sequence.] Proceed with the batch?`\n\nExample: `Plan: 6 frame generations at 1k 16:9 totaling 24 credits — establishing wide, push-in, two-shot, reverse, OTS, insert. Proceed?`\n\n### 3b. Batch authorization — one approval covers the enumerated batch\n\nWhen the user approves a batch plan with one aggregated cost total up front (\"8 scenes, ~$X total — go\"), that single approval authorizes `confirm=true` on **each enumerated call in that batch** — and nothing beyond it. You do not need to re-ask per call; that's the point of the upfront announcement. Hands-off multi-scene runs depend on this.\n\nBoundaries that re-trigger confirmation:\n\n- Any call's actual estimate exceeds what the announced plan implied → stop, surface the delta, get a fresh OK. The app enforces its own ceiling on top of this (see the thresholds above); do not wait for it to catch you.\n- New calls are added that weren't in the enumerated plan (extra variants, retries beyond the plan, a new scene) → those are NOT covered. Announce and confirm separately.\n- The batch scope changes (different model, resolution, or duration than announced) → re-announce, re-confirm.\n\nOne approval = that plan, as enumerated, at those prices. Nothing else.\n\n### 4. Track the running total\n\nAfter each generation completes, the response includes `cost_credits` (when available). Keep a running tally in your context. Surface it every 3 generations or whenever the user asks \"how much have we spent?\"\n\n## Resolution decision rules\n\nUse the model defaults below for ordinary work. For cheap exploration, choose a supported lower setting and compare the quote. For final delivery, use the size the output needs. When comparing prompts, hold model, quality, size and references constant.\n\n<!-- @inject:image-defaults -->\n**Image default:** gpt-image-2-5-sunburst, quality `high`, 3k. User overrides take priority. Without a project, generation uses the headless Nano Banana 2 seat.\n\n| Model | Default resolution |\n|---|---|\n| nano-banana-2 | 2k |\n| nano-banana-2-lite | 1k |\n| nano-banana-pro | 2k |\n| gpt-image-2-5-flare | 2k |\n| gpt-image-2-5-sunburst | 3k |\n| flux-2-max | 1k |\n| seedream-5-lite | 2k |\n<!-- @end:image-defaults -->\n\n**4K VIDEO is Pro-only (2026-07-07).** The ladder above is for IMAGES (open at every tier). For VIDEO — Kling, Seedance, Veo — 4K requires a Slates Pro account; a base-tier 4K video gen is rejected server-side with `PRO_REQUIRED`. Default video to 1080p or lower and only reach for 4K when the user is on Pro and explicitly asks. 4K *images* are never gated.\n\n## Aspect ratio decision rules\n\nAsk the user when ambiguous. Otherwise:\n\n| Context cue | Aspect ratio |\n|---|---|\n| \"cinematic\", \"film\", \"movie\", \"wide\" | 16:9 |\n| \"TikTok\", \"Reels\", \"Story\", \"mobile vertical\", \"phone\" | 9:16 |\n| \"square\", \"Instagram feed\", \"thumbnail\" | 1:1 |\n| \"ultra-wide\", \"anamorphic\", \"cinemascope\" | 21:9 |\n| \"portrait\", \"magazine cover\", \"vertical\" | 4:5 or 2:3 |\n| \"landscape photo\", \"horizontal\" | 3:2 or 4:3 |\n\nIf the user prompt mixes signals (e.g. \"cinematic Instagram post\"), ask. Don't guess.\n\n## When the gate fires\n\nThe server returns `requires_clarification` when required composition inputs are missing, and `requires_confirm` when total spend crosses the gate above. In both cases:\n\n1. Surface the gate response to the user\n2. Get a clean answer\n3. Re-call with the explicit values + `confirm: true` if applicable\n\nDon't bypass the gate by silently filling in defaults. The gates exist because defaults waste money.\n\n## Video is slow + async — a timeout is NOT a failure\n\nVideo gens take minutes (Seedance 4K can run far longer). A client/CLI timeout or a slow, empty-looking response is **not** a failed generation — the job is still running on the provider.\n\n- **Never re-submit a video gen because it \"timed out.\"** That double-charges the user for one video. Re-rolling a slow gen is the single most expensive mistake here.\n- **Poll, don't re-roll.** Use `background: true` on `slates_generate_video`, then poll `slates_get_generation_status` (free, read-only) until it reports `completed` or `failed`. In-flight jobs survive app restarts and are recovered.\n- A gen has only failed when the status comes back `failed` — and a provider *rejection* **refunds** the credits, so failed isolation tests are ~free. Until you see a terminal status, the job is in flight. Wait.\n\n## 🔴 The still-gate — the most expensive mistake in the pipeline\n\n<!-- @inject:still-gate -->\n**A visible defect in the still is already a STOP.** Do not animate it. Fix the frame first, then move to motion — and go to motion only when the crop passes the still scan and you genuinely need movement to confirm an uncertain edge, reflection, or object.\n\nThis is a **cost** rule as much as a craft rule: a 1080p/10s premium video generation costs many multiples of an image re-roll, and video is where a defect stops being fixable. Anything wrong in the still gets worse in motion — soft geometry mushes, broken-but-plausible objects fall apart, oily textures start crawling. **Animating a known-bad frame is the single most expensive mistake in the pipeline.** Re-rolling the image is the cheap move; re-rolling the video is not.\n<!-- @end:still-gate -->\n\nThe check itself lives in `slates-vision-feedback-loop` (the four slop tells and the per-model accents). The **stop** is a cost rule and belongs here: before every image→video call, confirm the source frame passed the still scan. If it didn't, spending video credits on it is not iteration — it is buying a more expensive copy of a defect you already found.\n\n## The 3-strike rule\n\nStop after 3 iterations on the same prompt. Hand back to the user with what you tried and what's not working. The slot machine doesn't converge — if it's not landing, the prompt structure is wrong, not the seed.\n",
|
|
11
|
-
"slates-dialogue-blocking": "---\nname: slates-dialogue-blocking\ndescription:
|
|
12
|
-
"slates-direct-response-ad": "---\nname: slates-direct-response-ad\ndescription: Develop a product-led direct-response ad
|
|
13
|
-
"slates-edit-and-iterate": "---\nname: slates-edit-and-iterate\ndescription:
|
|
14
|
-
"slates-model-selection": "---\nname: slates-model-selection\ndescription: Which model to pick for a given job — the routing doctrine. Read BEFORE choosing any video or image model, before quoting a plan, and before defaulting anywhere. Seedance 2.5 is the DEFAULT video model (Eric, 2026-09-13: the best in the world — physics, effects, scale, 30s takes, 30 references, timestamps); Seedance 2.0 is the 4K seat and the cheaper one at every shared resolution; Kling 3.0 is the cost-effective seat for performances, start-frame animation and lip-sync; MiniMax H3 is the AUTHORED-AUDIO seat (three directable sound layers in one pass, declared reference relationships, 480p-4K) with MiniMax H3 Max beside it as a faster premium with omni-references and MiniMax H3 Max Turbo as its half-price, frames-only sibling; Veo 3.1 is a narrow niche (native synced audio in one gen, 16:9 or 9:16, 4/6/8s) and never the default.\n---\n\n# Model selection — the routing doctrine\n\nPick the model FIRST, deliberately, before writing a prompt or quoting a plan. Model routing is a core part of the intelligence users are paying for: the agent knows what each model is good at and which ones underperform for a job — defaulting to the wrong model burns the user's credits on a weaker result.\n\n## 🔑 The meta-rule — above the table\n\nThe tables below are a snapshot. This roster churns constantly (NB2 Lite, Omni Flash, Seedream 5 Lite, GPT Image 2.5 all landed recently) — **a table rots; a rule doesn't.** When the tables and this rule disagree, or when a model appears that the tables don't cover, run the rule:\n\n> **Name ONE must-preserve requirement for the shot.** Not a vibe — the single thing that, if it breaks, makes the shot unusable: this face stays this face · the fluid behaves like fluid · the text stays legible · the take stays one unbroken move.\n>\n> **Inspect the output at its intended crop.** A frame that holds up as a thumbnail can fall apart at the size it will actually be watched. For a location, look at atmosphere, material texture, and anchor objects; for a character, identity, skin, pose, and gradients.\n>\n> **Choose the model that PROVES that requirement** and leaves only failures you can afford to rerun or mask.\n>\n> **When the roster changes, repeat the evidence test.** Do not carry today's ranking forward on reputation.\n\n## Video routing\n\n| Job | Model | Why |\n|---|---|---|\n| **General-purpose — the default for most shots** | **Seedance 2.5** | The strongest seat in the catalogue: physics, effects, scale and hero shots, 4–30s in one take, 30 image + 10 video + 10 audio references, audio-only refs, and the only Seedance seat that acts on timestamps. 480p / 720p / 1080p, no 4K. LENGTH is the price dial — quote any take over ~10s. |\n| **Cost matters and the shot is a performance or a start-frame animation** | **Kling 3.0 std** | Cost-effective workhorse. Strong image-to-video: preserves identity, layout, and text from the start frame. 16:9 / 9:16 / 1:1, 3–15s. |\n| Higher visual polish, no physics demands | Kling 3.0 pro | Mid-price fidelity bump on the same strengths. |\n| Multi-character dialogue / audio co-generation | Kling 3.0 omni | Dialogue syntax, voice direction, language codes, `@element` refs. |\n| **4K delivery**, or the same resolution cheaper than 2.5 | **Seedance 2.0** | The only Seedance with native 4K (4K video is Pro-only) and cheaper than 2.5 at every shared resolution (720p $0.15/s vs $0.231/s). Same physics and effects strengths; 15s takes, 15 references, no timestamps. |\n| **One take longer than 15 seconds**, more than 15 references, an AUDIO-ONLY reference, or **beats that have to land at a named second** | **Seedance 2.5** | Only 2.5 does these (rules in `slates-prompting-seedance-2-5` § Timestamps); it is the default anyway. 🚨 Two live hazards: (a) with references attached, the words *add / remove / replace / change / extend / continue* make it reclassify the request as a video EDIT and fail AFTER the job queues — describe the finished frame, or use `seedance-2.5-edit`; (b) LENGTH is the price dial, not resolution — a 30s 720p face gen is 489 credits and a 30s 1080p faceless gen is 853, against a 1,000-credit welcome grant. Quote before any take over ~10s. |\n| **The SOUND has to be directed, not just present** — a specific line delivered a specific way, scene sound that has to sit under it, and score that must stay out of the characters' world | **MiniMax H3** | The only seat where audio is authored in three separate layers in ONE pass (synchronised events in the body, ambience in a soundscape section, audience-only score in its own) rather than toggled on. 5–15s, 480p / 768p / 2K / 4K, 24fps, 32kHz stereo, 11 languages. Rules in `slates-prompting-minimax-h3`. |\n| **A reference has to keep a DECLARED amount of itself** — especially moving one subject's characteristic onto a *different* subject | **MiniMax H3** | The only seat that understands a stated retention relationship (kept whole / kept in part / transferred onto another subject / loose echo). 9 images + 3 video + 3 audio, 12 files total. 🚨 The first 5 reference images are free and every one after that costs 4 credits — pass `referenceImages` to `slates_estimate_generation_cost` before a reference-heavy job. |\n| **Turnaround is the requirement** on a text-to-video or start-frame shot at 480p to 1080p | **MiniMax H3 Max** | fal's self-hosted post-train of H3. **Measured 2026-08-27: a 5s 768p clip finished in 4.8s against 57s on base H3 — about 12x faster**, same prompt, queue to file. When turnaround is the requirement this is not a marginal win. 🚨 It is the PREMIUM seat, not a cheap H3 — $0.080/s at 768p against base H3's $0.060/s, 33% more, and it tops out at a 1080p refinement of its 768p render. It still animates a start frame and an end frame — image-to-video is one of the two things it is for — and since 2026-09-09 it takes the full omni-reference set too (9 images + 3 video + 3 audio), so the seats now differ on ladder and price rather than on what they accept. Never the default; never reach for it to save money. |\n| **Drafts and volume** on a text-to-video or start-frame shot, where the credit budget binds and no reference is needed | **MiniMax H3 Max Turbo** | A second fal post-train of H3 with Max's ladder at **half Max's rate at every tier** ($0.040/s at 768p). It takes a start frame and an end frame but has **no reference endpoint**: a shot that needs references goes to H3 Max or base H3. Its 1080p, like Max's, is a refinement of the native 768p render. Re-run the keeper on a hero seat. |\n| Native synchronized audio (dialogue + SFX generated WITH the video in one gen), 16:9, ≤8s | Veo 3.1 | Narrow, and now narrower: if the sound needs DIRECTING rather than merely existing, MiniMax H3 is the better seat. |\n\n### Named Seedance escalation triggers\n\n\"Physics matter\" is an abstract category and it under-fires. These are the beats Seedance is **observably** good at — if the shot contains one, escalate without deliberating:\n\n- **Real-time → slow-motion contrast.** The signature beat; nearly every strong clip rides it.\n- **The camera moving while debris, meteors, sparks or particles crash around the subject.** Distinctly a feature of this model, not just a thing it survives.\n- **Massive scale that has to read as genuinely huge** — not \"a big thing\", a thing whose size is the point of the shot.\n- **One continuous unbroken take.**\n\nConcrete beats route better than an abstract category. Cost stays a tiebreaker, never the router (see below).\n\n## Video EDIT routing (changing an existing clip)\n\n| Job | Tool | Why |\n|---|---|---|\n| **Footage-synced VFX on real footage** — add/remove an effect, prop, or lighting change while the take stays the take (incl. talking heads) | **Omni Flash Edit** (`slates_edit_video`, `omni-flash-edit`) | **The edit-fidelity winner** (head-to-head receipt 2026-07-09, WITH a short prompt): lip movement held perfectly, audio near-identical, effect landed and released on cue — where Kling missed an action beat and drifted lips. Prompt-only, 3–10s clips, 720p out, ~6.4 cr/s (cheapest). Quirk: occasional tail jitter / doubled final speech beat — trim the tail on the timeline. Fidelity is EARNED by prompt discipline: one short line + \"Keep everything else the same\"; long prompts destroy it (see below). |\n| **Identity swap needing reference images** — put @marcus into the clip, lock a style from refs | **Kling O3 Edit** (`slates_edit_video`) | The only edit engine that takes element/style reference images (frontal + angles lock identity). ~19¢/s. |\n| **Spoken words must be bit-exact** (VO, legal copy, music) | **Kling O3 Edit** with `keepAudio` (default true) — or segment-splice | Kling keeps the ORIGINAL audio track verbatim — but re-synthesizes the video, so lips can drift slightly against it (7/09 receipt). Omni Flash regenerates audio (voice editing unsupported): on the 7/09 receipt it came back near-identical with perfect lips, but \"near-identical\" is not a guarantee. Zero-risk path for critical audio: segment-splice — edit only the non-talking seconds and keep the original track under the cut. |\n| Style-transfer-heavy re-imagining, full relocate of the scene, or edit quality worth a premium at 1080p+ | Seedance edit/relocate (`videoReferenceAssetId` on `slates_generate_video`) | Seedance's strength is transfer intensity; it re-generates rather than surgically edits. Head-to-head receipt 2026-07-09 (photoreal-insert job, same clip): at 720p it LOST to Omni Flash edit on result while costing ~3× (vref bills input+output seconds; face-lane rates when people are in frame). Route here for its strengths or at 1080p/4K where its ceiling is higher — never as the cheap default. (2.5's relocate lane reaches 1080p too as of 2026-08-24, at $0.3412/s of combined input+output.) Takes long descriptive prompts fine (no Omni-style hard-fail on timing phrasing). |\n| **A clip LONGER THAN 15 SECONDS** | **Seedance 2.5 Edit** (`slates_edit_video`, `seedance-2.5-edit`) | The only edit engine that takes a 4–30s clip — length is the whole reason to route here. 480p/720p/1080p out, native audio, prompt + clip only (no reference images). Output length AND aspect ratio follow the source, so the billed key is the ceiled source length; an edit bills roughly DOUBLE a plain 2.5 generation of the same length because every provider charges an edit on input + output seconds. Set `seedanceFace: true` when a face is visible — the faceless provider blocks faces outright. No consented-real-face route for editing. Inside 15s, choose on fidelity instead. |\n| AI-edit the user's OWN footage | Omni Flash Edit (3–10s), Kling O3 Edit (3–15s, 720–3840px) or Seedance 2.5 Edit (4–30s) | Both take any MP4/MOV — not just Slates gens. Phone footage MUST be rotation-normalized first (players honor the rotation flag; models don't — raw portrait phone clips come back SIDEWAYS). |\n\n- **Edit before re-roll.** A re-roll gambles away the parts the user already likes; an edit changes only what the prompt names. Quote the edit first when a clip is mostly right.\n- **Ship via segment-splice.** Every edit model re-synthesizes the whole clip, so fidelity risk scales with clip length. For real deliverables: trim out ONLY the seconds where the change happens, edit that segment, splice it back over the original on the timeline with the ORIGINAL audio underneath. Most of the final video stays the untouched original — that's how the polished split-screen demos going around actually work, plus gesture-only beats with voiceover laid over in post.\n- **One change per pass, short prompts.** On Omni Flash this is documented law (\"overly descriptive prompts can lead to unintended changes\" — long identity-lock preambles make drift WORSE, receipt 7/09); on Kling multi-beat instructions get dropped. Chain passes instead.\n- Edited clips are themselves editable clips — chain passes; lineage links each output to its parent.\n\n## Motion Transfer & Lip Sync routing (Kling-only tools)\n\nBoth tools are **Kling-only**. Every entry in them is a real Kling endpoint that bolts motion or lip movement onto a finished source as a dedicated post-process.\n\n| Job | Tool | Why |\n|---|---|---|\n| Motion retarget onto a still character | Kling MC std/pro (`slates_generate_motion_transfer`) | Structured skeleton/depth retarget, ~32–42 credits / 5s, takes up to 30s driving clips. |\n| Re-voice a clip, or animate a still portrait | Kling lip-sync / avatar (`slates_generate_lip_sync`) | ~4–29 credits / 5s blocks. |\n\n**Want the Seedance version of either?** It is not a switch on these tools — it is a normal `slates_generate_video` on `seedance-2` with the clip attached as a **video reference** and the motion or dialogue written into the prompt (\"the character from image 1 performs the exact motion from video 1\"). That routes to the same endpoint the tool would have called, with the prompt visible and editable instead of ghost-written. Single-pass conditioning genuinely beats post-hoc retargeting on fast choreography, contact, cloth and hair — and it carries native audio — so escalate there whenever fidelity matters.\n\n- Seedance video-reference gens bill COMBINED input+output seconds (`seedance-2*-vref-*` keys) — pass the clip duration and quote before confirming. Driving clips must be 2–15s on Seedance 2.0 and up to 30s on 2.5; past that it is Kling MC's lane. On Seedance 2.5's AI-face route (EvoLink) the input side counts as at least the output's length: max(input, output) + output.\n- Faces on that route go through the normal cascade: `seedanceFace` for a character, `[REAL_FACE_DETECTED]` → `seedanceRealFace` + `realFaceConsent` for a real person (premium realface pricing).\n\n**Rules:**\n\n- **Default video = Seedance 2.5.** Route to Seedance 2.0 for 4K or when the same resolution must be cheaper, and to Kling 3.0 std when the budget matters and the shot is a performance or a start-frame animation — and say why in the plan (\"4K delivery, routing to 2.0\"; \"budget dialogue shot, routing to Kling\").\n- **Veo is never the default.** 16:9 or 9:16 only, 4/6/8s only (and 8s only at 1080p/4K, or with reference images), and it is not the quality pick — treat it as a single-purpose tool for native-synced-audio shots. If audio can be added after (Kling lip-sync, edit stage), prefer Kling or Seedance + audio in post.\n- **9:16 vertical → Kling or Seedance by preference**, not by necessity: Veo does take 9:16 on the route Slates uses. Route away from it because it is the niche seat, not because it can't.\n- **Ratios and durations are enforced before submit.** `slates_generate_video` validates the aspect ratio, resolution and duration against the model you picked and refuses out-of-set values with the legal list — it will not silently ignore or downgrade them. The authoritative per-model sets are in the op's own param descriptions, which are generated from the capability SSOT; prefer those over any list written in prose here.\n- **Image-to-video from an NB2 start frame** (the standard pipeline) → Seedance 2.5 by default, Kling when the budget matters and the motion is a performance. Not Veo.\n- **User names a model explicitly → use it.** But if it's a mismatch for the job (crazy physics on Kling std, a 30s take on anything but Seedance 2.5, 4K on Seedance 2.5 which has none), say so in one line and offer the right route before generating.\n\n## Image routing\n\n**Video models (Kling, Seedance, Veo) cannot generate standalone images — ever.** A \"premium hero reference image\" is still an image job: it routes to an image model below, never to Seedance.\n\n<!-- @inject:image-defaults -->\n**Image default:** gpt-image-2-5-sunburst, quality `high`, 3k. User overrides take priority. Without a project, generation uses the headless Nano Banana 2 seat.\n\n| Model | Default resolution |\n|---|---|\n| nano-banana-2 | 2k |\n| nano-banana-2-lite | 1k |\n| nano-banana-pro | 2k |\n| gpt-image-2-5-flare | 2k |\n| gpt-image-2-5-sunburst | 3k |\n| flux-2-max | 1k |\n| seedream-5-lite | 2k |\n<!-- @end:image-defaults -->\n\nUse `slates_estimate_generation_cost` for the selected model's current price and craft card. Routing reasons live in the model facts returned by `slates_list_available_models`; use the model's guide for its particular strengths and limits. Choose a different seat when the brief supplies a reason, such as speed, supported output shape, or an edit that failed on the default.\n\n**Historical photoreal receipt:** the 2026-08-24 comparison favored GPT Image 2 on one skin-realism task at its old high tier. That is evidence about that comparison, not proof that 2.5 requires its most expensive tier. Raise quality only to address a specific observed shortfall and compare at the delivery crop.\n\n## Audio routing\n\n**Image and video models cannot generate standalone audio, and neither audio model can generate images or video.** A shot that needs synced audio generated WITH the picture is still a video job (Kling omni / Veo / Omni Flash / Seedance all carry native audio); the models below produce audio *as its own asset*, to lay on the timeline.\n\n| Job | Model | Why |\n|---|---|---|\n| **Default — a whole audio scene in one pass**: room tone, ambience beds, crowds, nature, layered dialogue + effects, spoken lines inside a scene | **Seed Audio 1.0** (`seed-audio`) | One plain sentence in, a complete scene out. The continuity-bed workhorse; dialogue is performed inside the room, not cast. |\n| **One named voice saying one line** — a character's own voice, a narrator, a clean VO to lip-sync against | **Inworld TTS-2** (`inworld-tts-2`) | The prompt IS the words, spoken verbatim and billed per character. Voice = the character's clip (cloned for the take), a description, or a preset. No room tone — mix it on the timeline. |\n| **One effect that lands on a known frame**, or a seamless loop | **Sound Effects v2** (`eleven-sfx`) | The only surface with an exact duration control and a real loop mode. |\n\n**There is no music model.** A song is imported (Slates reads audio files and puts them on the timeline), not generated. A line that has to be spoken in a SPECIFIC voice is generated on Inworld TTS-2 and lip-synced against; a line that belongs to a scene is performed by Seed Audio inside it.\n\n### Named audio escalation triggers\n\n- **\"It needs to sound like a place\"** → Seed Audio. Three separate SFX generations layered on the timeline is the wrong shape and costs more.\n- **\"Read this line\"** → Seed Audio, with the line in quotes inside the scene sentence. Re-roll until the take is right, then lip-sync against it.\n- **\"That needs a thump right there\"** → Sound Effects, with the duration set to roughly the length of the event.\n- **\"Give it a track\"** → there is no music generation. Say so and offer to lay an imported track on an audio track.\n\n**Rules:**\n\n- **🚨 Seed Audio has NO duration parameter.** Length comes from the prompt text, so Slates writes the requested duration into the prompt and **bills what you asked for**. Choose the duration deliberately and never write a second, different length into the sentence. Full doctrine: `slates-prompting-seed-audio`.\n- **Kling's audio syntax does not transfer.** `SFX:` / `Ambient noise:` / `Background music:` prefixes are Kling 3.0 *video* prompt syntax. Seed Audio reads them as literal words and the result degrades.\n- **Beds outlast the cut.** Always ask for more seconds than the clip needs so the edit has fade handles — and remember those extra seconds are billed on both surfaces.\n- **Audio inside the video vs audio as an asset.** If the sound must be locked to what happens on screen, generate it with the video (Kling omni / Seedance / Omni Flash / Veo). If it needs to be moved, trimmed, re-used, or layered, generate it here and drop it on an audio track.\n- Per-model prompting: `slates-prompting-seed-audio`, `slates-prompting-elevenlabs`.\n\n## Cost is a tiebreaker, not the router\n\nRoute by capability first, then pick the cheapest tier that serves the job (per `slates-cost-discipline`). Never pick a model because its per-second price looked lowest — a cheap clip that has to be regenerated on the right model costs more than routing correctly once.\n",
|
|
15
|
-
"slates-one-prompt-film": "---\nname: slates-one-prompt-film\ndescription:
|
|
16
|
-
"slates-previs-blocking": "---\nname: slates-previs-blocking\ndescription: Build a 3D blocking pass in Blender, render it grey-box, and use it as a reference video so the generated shot follows a camera path you designed instead of one the model invented. Use when the user wants precise camera control, a multi-cut sequence, a one-take move, spatial consistency across shots, or says the camera keeps drifting / they keep burning credits re-rolling.\n---\n\n# Previs blocking — design the shot, then generate it\n\nThe spine of the whole workflow. Read this first; the other four previs skills are branches off it.\n\n## The mechanism (why this works at all)\n\nA text prompt asks the model to *invent* camera motion, so it invents differently every roll. You cannot iterate on a variable you do not control, so you re-roll and pay again.\n\nA **reference video** removes the invention. You build the shot in Blender as untextured proxies — a neutral grey set with colour-coded figures, free, instant, deterministic — render the camera's path to mp4, and hand the model that clip alongside the prompt. **Blender locks the motion; the model builds the world.** Iteration moves to the free half, and the paid half usually lands first try.\n\nTwo halves, and keeping them separate is the whole discipline:\n\n| Half | Lives in | Changes when |\n|---|---|---|\n| **Structure** — cuts, camera, timing, who is where | the blocking clip | you re-block |\n| **Style** — what any of it looks like | references + prompt text | you restyle (see `slates-restyle-from-blocking`) |\n\n## Before you start\n\n1. `slates_blender_status` — confirms the bridge is up and returns fps, frame range, existing camera. If it reports `connected: false`, relay its hint and stop; nothing else here works.\n2. Settle **format first**, because the blocking render *is* the film's format: fps, aspect, duration. 24fps is the default and makes cut times land on clean frames. Duration ≤ 30s (seedance-2.5's reference-video ceiling; 15s on the others).\n3. Know the shot count. \"One take\" and \"19 cuts\" are different builds.\n\n## Build order\n\nDo these in order. Each stage is verifiable on its own, and a camera built before the geometry has nothing to frame.\n\n### 1. Set the format\n\n```python\nscene = bpy.context.scene\nscene.render.fps = 24\nscene.render.fps_base = 1.0\nscene.render.resolution_x, scene.render.resolution_y = 1920, 1080\nscene.frame_start, scene.frame_end = 1, 720 # 30s at 24fps\nresult = {\"seconds\": 720 / 24}\n```\n\nFrame maths, stated once so you never redo it in your head: **frame = seconds × fps + 1**. A cut at 7.79s is frame 188.\n\n### 2. Geometry and light — grey set, coded figures, named\n\nProxies only. A person is a box or a capsule with a sphere head. A car is a stretched cube. A can is a cylinder. **The SET is neutral grey — one light, a floor and enough wall that the space reads.** Colour is reserved for the figures, where it carries meaning (below); a grey set is what makes those few colours legible as notation rather than décor. Anything you spend on materials here you pay for twice, because the model repaints every surface anyway.\n\n**Name every object for what it *is* in the story**, not `Cube.003`. The name is how you refer to it later, and it is how you keep your own timeline honest.\n\nTwo conventions that cost nothing now and save a re-roll later:\n\n- **Colour is identity.** Give each character a distinct viewport colour and *write the mapping down* — `red = the boss, green = the kid, blue = the driver`. The generation prompt will restate that mapping so the model knows which grey body is which person across cuts. Without it, characters swap.\n- **Encode facing on featureless proxies.** A box has no front. Mark one face red, the back black, the sides green, and say so in the prompt: `RED face = the direction he faces`. Otherwise the model guesses which way people are looking.\n- **Checker a surface when SCALE or SPEED has to read.** Flat grey gives a model no parallax cue, so a fast move over a featureless floor reads as slow, and a big room reads as a small one. A black-and-white checker on the ground (or the wall a camera races past) gives it something to measure against. ⚠️ **Build it as GEOMETRY, never as a Checker Texture node.** The blocking render is Workbench, which draws one flat colour per material and never evaluates a shader node tree — a `TEX_CHECKER` comes out flat grey and you lose the cue without being told. Subdivide the plane and alternate `material_index` per face. Like every other colour here it is notation, so it goes in the translation list and gets dressed over.\n\n```python\n# Two materials, alternated per face. `TILE` is the square size in metres.\ndark = bpy.data.materials.new(\"Checker_Dark\")\ndark.diffuse_color = (0.05, 0.05, 0.05, 1.0)\nlight = bpy.data.materials.new(\"Checker_Light\")\nlight.diffuse_color = (0.80, 0.80, 0.80, 1.0)\nfloor.data.materials.append(dark) # material_index 0\nfloor.data.materials.append(light) # material_index 1\n# Subdivide first (edit mode or a Subdivide modifier applied) so there ARE\n# faces to alternate — a 2-triangle plane can only ever be one colour.\nfor face in floor.data.polygons:\n cx, cy = face.center.x, face.center.y\n face.material_index = (int(cx // TILE) + int(cy // TILE)) % 2\n```\n\nAnd the identity colour on each proxy:\n\n```python\nmat = bpy.data.materials.new(\"ID_Red\")\nmat.diffuse_color = (0.8, 0.1, 0.1, 1.0) # what the blocking render draws\nobj.data.materials.append(mat)\nobj.color = (0.8, 0.1, 0.1, 1.0) # same value, for viewport parity\n```\n\nThe blocking render pins Workbench to `MATERIAL` shading, so **`mat.diffuse_color` is the value that reaches the clip** — and an object with no material at all falls back to a neutral grey, which is why an unpainted set still reads correctly. Set `obj.color` to the same value anyway: it costs one line, it makes the user's viewport match what renders, and keeping the two equal means you never have to remember which one is authoritative.\n\n### 3. Camera\n\nThe whole of `slates-camera-language`. Build the rig, then keyframe it. Then **read back what you built** with `slates_blender_scene` — its `cutSeconds` is your cut list, and it is the number you will write timings against. That field is the authoritative one on EITHER rig — marker frames when cameras are bound to markers, the active camera's own keyframes when they are not. `camera.keyframeSeconds` is empty on a marker-bound edit, which is the rig `slates-camera-language` recommends for anything past a handful of cuts.\n\n### 4. Handheld, last\n\nAdd it after the moves are right, never before — noise on top of a wrong path just hides the wrong path.\n\n### 5. Verify the cuts\n\nThe one check that catches the most damage: on a multi-cut blocking, camera position, target and focal length must all change **exactly on the cut frame, with no transition frame between**. One interpolated frame reads as a whip-pan the model will faithfully reproduce.\n\n```python\n# Every camera f-curve keyframe on a cut frame must be CONSTANT out of the\n# previous key, or the cut smears.\nfor fc in cam.animation_data.action.fcurves:\n for kp in fc.keyframe_points:\n if int(kp.co[0]) in CUT_FRAMES:\n kp.interpolation = 'CONSTANT'\n```\n\nAlso check nothing interpenetrates — proxies through floors, clones through the hero object, letters through each other. The model renders intersections as faithfully as it renders everything else.\n\n### 6. Save a backup after every stage\n\nCheap, and blocking is iterative by nature.\n\n```python\nbpy.ops.wm.save_as_mainfile(filepath=path, copy=True)\n```\n\n## Render and generate\n\n```\nslates_blender_render_blocking { projectId, fps: 24 }\n```\n\nRenders the **scene camera** through scene settings — never the user's viewport, so the result does not depend on where they left their mouse — imports the mp4 into the project, and returns `assetId` + `durationSeconds`.\n\nThen:\n\n```\nslates_generate_video {\n model: \"seedance-2.5\",\n videoReferenceAssetIds: [<the blocking asset>],\n videoReferenceSecondsEach: [<durationSeconds>],\n characterAssetIds: [...], environmentAssetIds: [...], styleAssetIds: [...],\n prompt: <written per slates-blocking-to-prompt>\n}\n```\n\n**Four inputs, and that is the entire stack:** a character sheet each, one location/style reference, the blocking clip, and a prompt written against the blocking. Resist adding a fifth.\n\nModel note: seedance-2.5 is the seat for this — 10 reference videos at up to 30s each. seedance-2 and minimax-h3 take 3 at 15s. Route per `slates-model-selection`.\n\n## Leaving holes on purpose\n\nWhere the model outperforms any blockout you could build — liquid, smoke, fire, cloth — **block a black gap instead** and say so in the prompt: `CUT 7 (14.5-17.0, black gap in the reference)`. You are reserving a slot, not forgetting one.\n\n## What not to do\n\n- **Don't texture, light or material the blocking.** Grey is the specification. The reference supplies motion; the references supply look.\n- **Don't animate what you don't need.** Heads especially — a proxy head turning wrong is worse than one that never turns.\n- **Don't build the camera before the geometry.** It has nothing to aim at, and every value you set gets redone.\n- **Don't skip reading the scene back.** Write timings from `slates_blender_scene`'s `cutSeconds`, never from what you intended to build.\n- **Don't exceed the model's reference-video ceiling.** A 40s blocking against a 30s cap silently truncates.\n\n## Related\n\n`slates-camera-language` (rigs and moves) · `slates-blocking-to-prompt` (writing the prompt against the clip) · `slates-dialogue-blocking` (multi-character continuity) · `slates-restyle-from-blocking` (one blocking, many worlds) · `slates-model-selection` (routing)\n",
|
|
17
|
-
"slates-project-organization": "---\nname: slates-project-organization\ndescription:
|
|
18
|
-
"slates-prompting-elevenlabs": "---\nname: slates-prompting-elevenlabs\ndescription:
|
|
19
|
-
"slates-prompting-flux-2-max": "---\nname: slates-prompting-flux-2-max\ndescription: How to prompt FLUX.2 Max (Black Forest Labs image model). Read before calling slates_generate_image with model flux-2-max, or slates_edit_image with editModel flux-2-max. FLUX.2 wants front-loaded structure, real camera vocabulary, and positive-only phrasing — no negative prompts, no tag soup.\n---\n\n# FLUX.2 Max — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — FLUX.2 Max.** Word order is weight: it attends hardest to the start. Structure: `Subject + Action + Style + Context`, then secondary detail. Length 10-30 words for a concept test, 30-80 for most work, 80+ only for a genuinely complex scene.\n\n**The five levers**\n1. **Front-load the subject and the one action.** Anything after the first clause is a modifier, and it is read as one.\n2. **Name real gear** — `Shot on Hasselblad X2D, 80mm, f/2.8, natural light`, `Kodak Portra 400, natural grain`. This is the single biggest realism lever.\n3. **Era cues as a package** — `early digital camera, slight noise, flash photography, candid` reads 2000s; `film grain, warm cast, soft focus` reads 80s.\n4. **Bind every hex colour to an object.** `a #1B4D3E enamel mug` lands; an unbound colour does not.\n5. **For portraits add texture words** — `natural skin texture, realistic pores, subtle imperfections, soft diffused lighting`.\n\n<!-- @inject:cinematic-card -->\n**For a photographic look, use only what this frame needs.** Image models default to clean, evenly lit and fully exposed. Describe what the camera sees, not just gear or mood:\n- **Inspect every reference first.** Write its grade and imperfections in words: darkness, contrast, muddy or true blacks, colour, softness/noise, subject separation. Never grade cleaner or brighter than the look reference unless asked.\n- **One light system** — `low sun behind her`, `her face falls into deep shadow`, `no light in front of her`.\n- **Visible exposure** — `the sky burns out to white`, `dense, slightly crushed shadows`.\n- **Lens name plus effect** — `200mm telephoto`, `peaks loom huge behind her and melt into soft shapes`.\n- **Name every garment and close the foreground.** Omissions invite reference leakage or invented props.\nBind references inline. A scene reference owns the grade; for a look-only reference, write the new scene's light. References are optional. For owned-frame edits, describe only the change and what stays.\n<!-- slates-only -->Use `slates-cinematic-look` with a technique ID or section query for more.<!-- /slates-only -->\n<!-- @end:cinematic-card -->\n\n**Hard constraint:** no negative prompting. Every \"no X\" must be rewritten as the positive state — `no blur` becomes `sharp focus throughout`, `no people` becomes `empty scene`, `no harsh shadows` becomes `soft, diffused lighting`.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — there is no negative prompting, so each of these has a positive form:\n- `no blur` (say `sharp focus throughout`), `no people` (say `empty scene`), `no harsh shadows` (say `soft, diffused lighting`)\n- an unbound hex colour — bind it to an object or it lands inconsistently\n- `masterpiece`, `best quality`, `trending on artstation`, `8k`\n<!-- @banned:end -->\n\n**Examples**\n- `A chef plating in a steel kitchen pass. Shot on Hasselblad X2D, 80mm, f/2.8. Overhead fluorescents plus warm spill from the line. Natural skin texture, subtle imperfections. Muted steel and #7A3B2E copper.`\n- `An empty municipal pool at dusk, 35mm, deep focus, early digital camera with slight noise and flash falloff. Cracked #4A7C8C tiles. Candid, unstaged.`\n\nBlack Forest Labs' top image model, routed via fal.ai. In Slates: `slates_generate_image` with `model: flux-2-max` (REQUIRES projectId — no headless path), priced per resolution (1k/2k/4k — call `slates_estimate_generation_cost` for current numbers, never quote from memory). Strengths vs Nano Banana 2: photoreal texture, less censored, precise hex-color control, strong typography. Reference images route through FLUX's edit endpoint and carry a lower per-model cap than NB2's 14.\n\n## Core structure — front-load what matters\n\n```\nSubject + Action + Style + Context\n```\n\nWord order is weight. FLUX.2 attends hardest to the start of the prompt: main subject → key action → critical style → essential context → secondary details.\n\n**Length:** 10-30 words for concept tests, 30-80 words for most work, 80+ only for genuinely complex scenes.\n\n## Photorealism: name real gear, not \"professional photo\"\n\nThe single biggest realism lever is concrete camera vocabulary:\n\n```\nShot on Hasselblad X2D, 80mm lens, f/2.8, natural lighting\nShot on Sony A7IV, 35mm, golden hour, shallow depth of field\nKodak Portra 400, natural grain, organic colors\n```\n\nEra cues work the same way: \"early digital camera, slight noise, flash photography, candid\" reads 2000s digicam; \"film grain, warm color cast, soft focus\" reads 80s.\n\nFor portraits add: natural skin texture, realistic pores, subtle imperfections, soft diffused lighting.\n\n## No negative prompts — reframe positively\n\nFLUX.2 has no negative prompt support. Describe the presence you want, not the absence:\n\n- ❌ \"no blur\" → ✅ \"sharp focus throughout\"\n- ❌ \"no people\" → ✅ \"empty scene\"\n- ❌ \"no harsh shadows\" → ✅ \"soft, diffused lighting\"\n\n## Hex colors — bind them to objects\n\nFLUX.2 matches hex codes, but only when each code is attached to a specific object:\n\n```\nwalls in hex #C4725A, sofa in #1B6B6F, accent pillows #E8A847\ngradient starting with color #02eb3c and finishing with color #edfa3c\n```\n\n❌ \"use #FF0000 somewhere\" — unbound colors land inconsistently.\n\n## Text rendering\n\nQuote the exact text, then place and style it:\n\n```\nThe text 'OPEN' appears in red neon letters above the door\nLogo text 'ACME' in color #FF5733, ultra-bold decorative serif, centered\n```\n\nSpecify placement relative to other elements, font family feel (serif / sans / script), and relative size (\"large headline,\" \"small body copy\").\n\n## JSON prompting for production work\n\nFor multi-element scenes that must come out exactly right (product shots, infographics, brand work), FLUX.2 parses structured JSON prompts:\n\n```json\n{\n \"scene\": \"Professional studio product photography on polished concrete\",\n \"subjects\": [{ \"description\": \"matte black ceramic mug with steam\", \"position\": \"center foreground\" }],\n \"style\": \"commercial product photography\",\n \"color_palette\": [\"#1B1B1B\", \"#E8A847\"],\n \"lighting\": \"three-point softbox, soft diffused highlights\",\n \"camera\": { \"lens-mm\": 85, \"f-number\": \"f/5.6\" }\n}\n```\n\nUse natural language for exploration, JSON when the layout is locked and you're matching a spec.\n\n## Reference images (edit path)\n\nIn Slates, pass `referenceAssetIds` on `slates_generate_image` — FLUX routes them through its edit endpoint. Slates names each reference inline in the prompt (\"the subject (image 1), the style (image 2)\") in the order it sends them, so you don't hand-write role labels; the name carries the role and unnamed-by-position blending is avoided. For surgical changes to one existing image use `slates_edit_image` with `editModel: flux-2-max` (note: FLUX edits ignore extra referenceAssetIds — that's NB2-only).\n\n### Reference rules (the verified ones)\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n<!-- @inject:reference-rules-core -->\nIdentity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.\n<!-- @end:reference-rules-core -->\n\n### For FLUX.2 Max specifically\n\n- **FLUX caps references well below NB2's 14, so rule 1's \"2-4\" is a ceiling here, not a starting point.** Be deliberate about which roles earn a slot.\n- **Rule 9 has a hard edge on this model:** `slates_edit_image` with `editModel: flux-2-max` ignores extra `referenceAssetIds` — that is NB2-only. A FLUX edit sees the source image and the prompt, nothing else.\n- **FLUX has no memory between generations, so rule 7 is enforced by repetition.** Define the character exhaustively once and repeat those exact descriptors verbatim in every subsequent prompt — see Character consistency across a series below.\n\n## Character consistency across a series\n\nDefine the character exhaustively once, then repeat those exact descriptors verbatim in every subsequent prompt. FLUX has no memory between generations — the repeated description IS the consistency mechanism.\n\n## Common failure modes + fixes\n\n| Failure | Fix |\n|---|---|\n| Generic \"AI look\" on photoreal | Name a camera body + lens + f-stop instead of \"professional photo\" |\n| Colors drift from brand spec | Bind each hex code to a named object |\n| Text garbled | Quote the exact string, specify font feel + placement + size |\n| Multi-reference blend chaos | Name each reference inline (Slates does this from your @mentions/referenceAssetIds) — the same name for one entity, distinct names per role |\n| Wanted element missing | Move it earlier in the prompt — order is weight |\n\n## Pre-flight: references arrive inline, refer by code\n\nWhen you pass `referenceAssetIds`, the first call returns the references **inline as image content blocks** with a cost estimate and `requires_confirm: true`. Look at them — revise the prompt if they suggest a different composition or style — then re-call with `confirm=true`. Refer to each asset by its short code (`IMG-A12 — Beach Sunset`) when talking to the user; it matches the badge on their gallery thumbnail.\n\n## Sources\n\n- [Black Forest Labs — FLUX.2 Prompting Guide](https://docs.bfl.ml/guides/prompting_guide_flux2)\n- [fal.ai — FLUX.2 [max] Prompt Guide](https://fal.ai/learn/devs/flux-2-max-prompt-guide)\n",
|
|
20
|
-
"slates-prompting-gpt-image-2-5": "---\nname: slates-prompting-gpt-image-2-5\ndescription: Prompt and edit images with GPT Image 2.5 Flare or Sunburst. Covers reference roles, realistic lighting, text, grids, quality choices and targeted edits. Use with slates_generate_image or slates_edit_image on these models.\n---\n\n# GPT Image 2.5 — sheets, grids, and text that actually reads\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — GPT Image 2.5.** The photoreal front-runner for people, and the readable-text, ordered-panel engine. Structure: subject and action with each reference named where it is used, then any exact copy in quotes, then layout, then light.\n\n**Pick the tier.** `flare` is Faster, quality comparable to GPT Image 2: drafts and volume. `sunburst` is Better quality, the most capable: finals, hero frames, photoreal people, multi-reference edits. Use the product default; choose Flare when speed is a stated priority.\n\n**The levers**\n1. **Name each reference inline** — `the woman from image 1`, `lit and graded like image 2`. Never an opening paragraph about what the references are.\n2. **Quote every string that must render verbatim** — `the jacket reads \"SLATES\"`. Describe a font's feel, never its name; keep on-image text under about 30 words.\n3. **Name the layout as a grid** for sheets and panels — `a 3x2 grid of panels, reading left to right, equal gutters`.\n4. **Set `quality` deliberately.** `high` is the everyday tier; `max` is 4× its price, `xhigh` about 1.8×. Coming from GPT Image 2 the names moved one rung: its `medium` is this `high`.\n\n<!-- @inject:cinematic-card -->\n**For a photographic look, use only what this frame needs.** Image models default to clean, evenly lit and fully exposed. Describe what the camera sees, not just gear or mood:\n- **Inspect every reference first.** Write its grade and imperfections in words: darkness, contrast, muddy or true blacks, colour, softness/noise, subject separation. Never grade cleaner or brighter than the look reference unless asked.\n- **One light system** — `low sun behind her`, `her face falls into deep shadow`, `no light in front of her`.\n- **Visible exposure** — `the sky burns out to white`, `dense, slightly crushed shadows`.\n- **Lens name plus effect** — `200mm telephoto`, `peaks loom huge behind her and melt into soft shapes`.\n- **Name every garment and close the foreground.** Omissions invite reference leakage or invented props.\nBind references inline. A scene reference owns the grade; for a look-only reference, write the new scene's light. References are optional. For owned-frame edits, describe only the change and what stays.\n<!-- slates-only -->Use `slates-cinematic-look` with a technique ID or section query for more.<!-- /slates-only -->\n<!-- @end:cinematic-card -->\n\n**Hard constraint:** its own content filter, distinct from Gemini's. Never describe a reference as a photograph of a real person.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use:**\n- a font NAME — describe the feel instead, as in: clean geometric sans, high contrast\n- a reference described as a photograph of a real person (`is a photograph of a woman`), or any up-front essay about what each reference is for — name the subject inline where it is used instead, as in: the woman from image 1\n- `8k`, `masterpiece`, `best quality`, `highly detailed` — quality incantations do nothing here either\n<!-- @banned:end -->\n\nGPT Image's edge is **character-level text accuracy** (~99% on English), ordered panels, and exact element placement — the jobs where every other model garbles a word or shuffles a layout. 2.5 inherits all of it and is better at each.\n\n## Which variant\n\n**Speed → Flare. Quality → Sunburst.** That is OpenAI's own routing rule, quoted from its image-prompting guide: *\"start with GPT Image 2.5 Flare when speed is the priority, or GPT Image 2.5 Sunburst when demanding quality requirements are the priority.\"* Same price either way, so the trade is purely latency against quality.\n\n🚨 **FLARE IS NOT AN UPGRADE OVER GPT IMAGE 2 — IT IS THE FAST ONE.** OpenAI, verbatim: *\"GPT Image 2.5 Flare is the small model, optimized for speed, with image quality **comparable to** GPT Image 2. GPT Image 2.5 Sunburst is the base model, optimized for quality, with **higher image quality than** GPT Image 2.\"* Their model pages agree: Flare is *\"our fastest model for high-quality, everyday image generation\"*, Sunburst *\"our most capable model for image generation and editing.\"* **Sunburst is the seat that beats what we had; Flare is the one that holds it at half the latency.** An earlier revision of this file called Flare \"better than GPT Image 2\" and sent Sunburst only to multi-reference edits — both wrong, corrected 2026-09-09 against the vendor docs.\n\n**Choose for the task.** Use the product default for ordinary work. Flare is an option when speed matters; changing model is not a mandatory draft stage.\n\n**Sunburst's widest lead is multi-reference editing** — several references all surviving into one frame, the character-consistency-across-shots problem. Reach for it there first, but that is not the only place it belongs.\n\n⚠️ **The LMArena receipt, scoped.** At launch Arena had Sunburst #1 and Flare #2 across text-to-image, single-image edit and multi-image edit, with margins over GPT Image 2 of **+81 / +47** on multi-image edit (Image Edit Arena: Sunburst 1520, Flare 1491, GPT Image 2 1461). Two caveats were missing and both matter: the baseline is **GPT Image 2 at `medium`, which is this model's `high`** — not its top tier — and the boards were **preliminary, a few thousand votes each**. Arena says Flare beats GPT Image 2; OpenAI says comparable. Route on OpenAI's wording and treat the board as a tiebreaker, not a spec.\n\n🚨 **The GPT Image line is ALSO the photoreal front-runner, and this file said the opposite until 2026-08-24.** **Receipts:** Eric's direct call, plus a head-to-head on the Higgsfield rail where GPT Image 2 at `quality: high`, 2K beat both Nano Banana rails on skin realism for photoreal people — that result is why the whole AI-influencer ad lane generates its plates here. **Route photoreal to this line, not away from it.**\n\n**Historical receipt, not a tier recommendation:** the photoreal comparison above used GPT Image 2 at its old `high` tier. It has not been repeated on 2.5 under matched conditions. Start with the product default and test a higher tier only against an unmet requirement; the old comparison does not establish a minimum tier for this model.\n\n**What the Banana line still owns:** edit-heavy work, and holding many subjects coherently in one frame. **Not the reference ceiling any more** — that line was true until 2026-09-09, when GPT Image went to its documented 16 against Banana's 14. Route on which model keeps them all recognisable, not on the count.\n\n**What would kill this:** a head-to-head at the intended crop going the other way. Per `slates-model-selection` § The meta-rule, re-run the evidence test when the roster changes — never carry a ranking forward on reputation. That rule is exactly what the 2026-08-24 correction failed, and exactly what the two ⚠️ notes above are honouring.\n\n## Quality tiers — always set explicitly\n\nAll five rungs are exposed, and they span ~36× end to end (2k class: $0.0044 → $0.158), which makes this the single biggest cost lever on the model. **The steps are UNEVEN — do not reason about them as a constant multiplier:** ~2.3× `low`→`medium`, ~3.9× `medium`→`high`, ~1.8× `high`→`xhigh`, ~2.25× `xhigh`→`max`. The same ratios hold at every OFFERED resolution class (2k/3k/4k); unoffered 1k differs slightly.\n\n| Tier | Use it for |\n|---|---|\n| `low` | Roughest pass — layout and composition checks, throwaway comps. |\n| `medium` | The draft tier. Cheaper than NB2 Lite and available up to 4K, which is why the draft lane moved here. |\n| `high` | General-purpose quality tier. Blind benchmarks on GPT Image 2 put this rung — which it called `medium` — within a hair of `max` (which it called `high`) at a quarter of the cost. Inherited from the old ladder, never re-run on 2.5, and it says nothing about `xhigh`. |\n| `xhigh` | One rung short of the top at about half its price (2k: 4 cr against `max`'s 8). Worth trying before `max`. |\n| `max` | Top of the ladder. Tiny type, dense diagrams, many labelled elements. |\n\n⚠️ **A tier label means different things on different models.** OpenAI: *\"The same quality label does not imply the same image quality or response time across models.\"* Flare at `max` and Sunburst at `max` are not the same picture, and neither matches Nano Banana's idea of \"high\".\n\n🚨 **The tier NAMES moved between versions and the strings did not.** GPT Image 2's `medium` is this model's `high`; its `high` is this model's `max` — same money, one rung of renaming. For a recipe explicitly written for GPT Image 2, map the old tier before reusing it on 2.5. A current user request for `medium` still means `medium`. Getting this backwards costs picture quality silently: nothing errors, the bill is correct for what was asked, and the image is just worse.\n\nNever rely on the provider default. fal's default is `high`, which is correct today — but it is the third rung of five rather than the top of two, so leaning on it means a fal-side change silently reprices you. Slates sends its configured quality explicitly; current defaults live in `slates-model-selection`. Your explicit choice overrides them.\n\n**Start at the default and change tiers for an unmet requirement.** OpenAI's own procedure: *\"If the output falls short, test a higher quality setting. Once it meets your requirements, test lower settings to see whether they preserve acceptable quality while reducing latency. Use `xhigh` or `max` only when they improve an unmet quality requirement within your latency budget.\"* A higher rung does **not** guarantee a better result on a given prompt. Compare `medium` against `high` when the job is small or dense text; that is where the rungs separate most visibly.\n\n## Resolution classes\n\n`1k` = 1024²-class · `2k` = 1920×1080-class · `3k` = 2560×1440-class · `4k` = 3840×2160-class. Pick 2k for most sheets/panels; 4k for print-density grids. 4K exists at every tier and is API-only — even paid ChatGPT can't render it.\n\n`1k` is not offered, and the reason is not its price: it is strictly dominated. At 1k you pay more for fewer pixels than at 2k, at **all five tiers**. Don't ask for it.\n\n⚠️ **Above 2560×1440 you are on a path OpenAI marks EXPERIMENTAL.** Verbatim: *\"Outputs with more than 3,686,400 total pixels ('2560x1440') are experimental.\"* That is the whole **4k** class (≈8.0 MP) plus 3k at 4:3/3:4 (≈3.70 MP). It bills normally and it works — but prove the shot at 2k or 3k 16:9 first, and do not be surprised by an odd frame at 4k.\n\n**Hard size bounds**, from fal's schema verbatim: each edge ≤ 3840 px, both edges multiples of 16, longer:shorter ratio ≤ 3:1, total pixels between 655,360 and 8,294,400. **The pixel ceiling is the one that actually bites** — the multiple-of-16 rule is documented but NOT enforced, and we have the receipt: 1920×1080 fails it (1080 = 67.5 × 16), is one of fal's own six priced canonical sizes, and metered clean. Slates picks sizes that respect the ceiling; these matter only if you hand-build a request.\n\n🚨 **THE ASPECT RATIO CHANGES THE PRICE ON THIS MODEL, and on no other image model.** OpenAI bills image OUTPUT TOKENS and the count tracks the frame's SHAPE, so at the same resolution class **`1:1` costs about 1.8× and `4:3`/`3:4` about 1.37× what `16:9` costs**; `9:16` costs the same as `16:9`. Metered 2026-09-09 and priced into the cost key, so the quote you get before generating is the real number — but if you are choosing between shapes and the budget is tight, **16:9 or 9:16 is the cheap one.** Every other image model charges the same whatever the shape.\n\n## Reference images — give every one a role, inline, where it is used\n\n**Assign a role to every reference image: subject, style, clothing, or background.** This is new emphasis in 2.5 and the highest-leverage change for the 16-reference character lane. An unroled pile of references makes the model guess what each one is for, and it guesses differently every run — which is the drift people mistake for a consistency failure.\n\n**The role rides a clause in the scene, not a paragraph in front of it.** *The woman from image 1 cooks on a rocky summit…*, *lit and graded like image 2*. Never open with sentences about what each reference is and what to take or ignore from it: that is the role essay the shared reference rules below forbid, and it drags the sheet's studio light into the scene.\n\n**Receipt, 2026-09-15, Sunburst, IMG-A192–A198.** The up-front version returned the studio look; the inline versions were never refused and never came back as a sheet. Two costs, both fixed in words: anything the prompt does not describe is taken from the reference (name every garment), and props nobody asked for appear (say what is in the foreground and that nothing else is). One sheet-only plate kept its described location, which narrows the two-reference rule in `slates-ugc-influencer-ad`. A look reference did far less than a described light. The full ladder is the vault's `cinematic-look-research.md`; the techniques are `slates-cinematic-look`.\n\nReference images route through the edit endpoint, **up to 16** — fal's documented `maxItems`, and the highest reference ceiling of any image seat in Slates (the Banana line takes 14). It was capped at 10 until 2026-09-09, which was never anybody's limit, just a number nobody had checked. The composed \"image N\" naming applies as everywhere else. Mask-based inpainting exists at the API level but is not surfaced: a mask is something the user has to paint, and there is no painting surface — describe the change instead.\n\n## Editing — separate the change from the constraints\n\n**State the change, then list what must survive.** \"Change only X,\" then name the invariants explicitly: identity, geometry, lighting, labels. For precise local edits also pin saturation, contrast, camera angle and surrounding objects — anything you do not pin is fair game for the model to move.\n\n**One change per iteration, and restate the constraints every turn.** Cross-turn drift is the named failure mode in OpenAI's own guidance: constraints do not persist across turns by themselves, so a multi-turn refinement that stops restating them will slowly rewrite the frame. This applies directly to multi-turn shot refinement.\n\n## Prompting for text accuracy\n\n- **Quote every string that must render verbatim**: `the sign reads \"OPEN 24 HOURS\"` — quoted strings render most reliably.\n- Say the text appears **once**, and give its position and typography.\n- Spell unusual words letter-by-letter.\n- Add `no extra text, no watermarks`.\n- Specify font *feel*, not font names: \"clean geometric sans, high contrast\", \"hand-painted brush lettering\".\n- For dense text (posters, UI mocks), list the copy as ordered lines: `Line 1: \"...\" Line 2: \"...\"` — it respects ordering.\n- **Don't bundle unrelated instructions into a text-rendering request.** A prompt that also redesigns the scene competes with the text for attention.\n- Keep total on-image text under ~30 words for perfect accuracy; beyond that, accuracy degrades gracefully but degrades.\n\n## Transparent backgrounds\n\nIf you need a cut-out rather than a scene, **ask for it explicitly and check the alpha**. OpenAI: request `background=transparent` and use PNG or WebP, then *\"check the decoded image's alpha channel, including hair, glass, shadows, and object edges\"* — a painted-white backdrop is the common failure and it is not transparency. Say what must NOT appear: *\"no solid backdrop, no checkerboard, no scenery, no watermark\"*, and do not let the product get restyled while the background is removed. **On every follow-up edit, repeat the transparency requirement** or it gets dropped. (Slates always requests PNG, so the format half is handled for you. **`background` IS surfaced now** — the Background control on the prompt bar, and `backgroundMode` on `slates_generate_image` / `slates_edit_image`. It is free: fal prices this family on size × quality alone.)\n\n## When an edit must not touch a region at all\n\nPrompting alone cannot guarantee pixel-identical pixels. OpenAI's own instruction: if a region must stay exactly as it was, **composite the approved edit back into the original image** rather than asking the model to preserve it. Treat \"preserve\" language as a strong bias, never a lock.\n\n## Structure a complex prompt in labeled sections\n\nFor anything with several requirements, OpenAI recommends organising the prompt as **scene, subject, details, constraints** with labeled sections. Same content, easier to read and to change one part without disturbing the rest — which is what makes the one-change-per-iteration rule practical.\n\n**Say \"photorealistic\" or \"real photograph\" when that is the goal.** It is not inferred from a detailed description; ask for it directly, then describe framing and texture.\n\n## Concrete visuals beat mood words\n\nName materials, lighting, colour and medium. Mood words are cues only — \"cinematic\", \"moody\", \"epic\" tell the model almost nothing on their own. Give scale, atmosphere and colour instead. Camera specs (`85mm`, `f/1.4`) are appearance hints, not a physical simulation; they bias the look, they do not compute optics.\n\n**Name the lens and describe its effect, every time.** A lens named alone changed nothing visible (IMG-A195, 2026-09-15); named together with what it does to the picture, it produced real compression and depth of field (IMG-A198). Wording: `slates-cinematic-look` → `compression-as-outcome`, `defocus-as-outcome`.\n\n**For people, state body framing and scale**: \"full body visible, feet included\", \"hands naturally gripping the handlebars\". This is also the safest way to phrase a crop — see the blocked-phrasings section below.\n\n**No special syntax is required.** Prose, JSON and tagged blocks all work equally well, so pick whatever stays maintainable in the caller.\n\n## Panels, sheets, and grids\n\n- State the grid explicitly and number the cells: \"a 2×3 grid of panels, numbered 1–6, reading left-to-right, top-to-bottom\".\n- Give each cell ONE content clause: \"Panel 3: the character mid-jump, side view\".\n- Character identity sheets: GPT Image holds both the structured panel layout AND photoreal skin, which is why the influencer-ad lane builds its sheets here. Reach for NB2/NB Pro when it is an edit of an existing sheet, or when many subjects have to stay recognisable at once — not for the reference count, which GPT Image now leads at 16.\n\n## 🚨 WHAT GETS YOU BLOCKED — read before writing a prompt with a person in it\n\n**Receipt: 24 consecutive attempts on one character, 2026-08-24, same project and same rail.** Eleven were refused with `content_policy_violation` on the fal edit endpoint. The refusals were never about the scene — one of the blocked prompts was a woman standing at a kitchen counter with her hand on it. **Two phrasings were hard blocks, 5 for 5 each, and neither ever passed:**\n\n**1. Never describe the reference as a photograph of a real person.**\n\n> ❌ `Reference image 1 is a photograph of a woman. Use that exact woman.`\n> ✅ `Reference image 1 is a character identity sheet showing one woman across several panels — the face in the large portrait panel is the authority for her identity. Use that exact woman.`\n\nThe first reads to the filter as *recreate this real person's likeness*, which is a hard refusal regardless of what the rest of the prompt says. The second signals a fictional character and passes. **This is a wording change only — the reference image can be the same file either way.** One plate flipped from refused to accepted on this single sentence with nothing else altered.\n\n**Inline naming sidesteps the question and is now the default:** never describe the reference at all, and name her where she is used (*the woman from image 1*). Six of six Sunburst plates written that way passed on 2026-09-15. Keep the sheet sentence above as the fallback if a refusal appears.\n\n**2. Never attach a reference sheet containing a headless body panel.** A sheet whose full-body panels are cropped above the neck is refused every time, even with the correct opener. Regenerate the sheet with the head visible in every panel. Related, and already in this file's sheet guidance: phrase a cropped panel as *framing* (`cropped at the collarbone`), never as *absence* (`the head not shown`).\n\n⚠️ **These refusals were measured on GPT Image 2, not on 2.5.** The classifier belongs to OpenAI rather than to a model version, so the phrasing rules carry — but they are inherited, not re-measured. If Flare or Sunburst accepts one of the blocked phrasings, that is a new receipt to write down here, not a reason to delete this one.\n\n**On top of those, ordinary content triggers still apply** and they stack independently — a correct opener does not rescue them:\n\n| Refused | Why, and the fix |\n|---|---|\n| A woman sitting on a bed in a bedroom | Domestic + bed reads as intimate. Move her to a chair, a rug, another room. |\n| A knife, even lying flat on a chopping board next to a lemon | The object is the trigger, not the framing. Swap it — a cast-iron pan cleared instantly. |\n\n**🚨 Refusals are PROBABILISTIC. Retry once before rewriting a word.** In the same session an identical prompt, identical reference, identical params was refused and then accepted on a straight re-fire. A rejected job returns no file and costs nothing, so a retry is free and a rewrite is not — rewriting first is how you end up changing four variables and learning nothing. **Only redesign after two or three refusals.**\n\n**And change ONE thing at a time.** The eleven refusals above took far longer to diagnose than they should have because a reference swap and an opener rewrite shipped in the same call. Isolate on the prompt you actually want, so a pass leaves you with a usable asset instead of a data point.\n\n## Filter regime\n\nOpenAI moderate — a third regime distinct from Gemini (NB family) and ByteDance (Seedream). Real-face references pass more readily than Gemini; violence/brand rules are similar. `slates-content-policy` applies unchanged.\n",
|
|
21
|
-
"slates-prompting-inworld-tts": "---\nname: slates-prompting-inworld-tts\ndescription: How to use Inworld Realtime TTS-2, the VOICE seat. Read before calling slates_generate_audio with model inworld-tts-2. Speech in a SPECIFIC voice, billed per character - the prompt is the words spoken, verbatim. Covers the identity-versus-acoustics rule (what a reference clip does and does not carry), how to write a line so it is performed rather than read, when to reach for seed-audio instead, and the voice-consent rule.\n---\n\n# Inworld Realtime TTS-2 — the voice seat\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Keep it under 2,400 characters (the build fails above that) and keep\n the rationale and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Inworld TTS-2.** Speech in a SPECIFIC voice. The prompt is the words spoken, verbatim — not a description of them. Text length determines the bill.\n\n**IDENTITY, NOT ACOUSTICS — the rule that decides whether cloning works**\nA reference carries WHO is speaking: timbre, pitch, accent, age, vowel shape. It does NOT carry WHERE they are — room tone, distance, phone EQ, reverb and mic character are *acoustics*, and this model reproduces the identity while discarding the room. So:\n1. **A noisy reference does not give a noisy read — it gives a WORSE identity.** Music, a second speaker or heavy reverb corrupt what is being extracted. Use a clean single-speaker recording.\n2. **You cannot get \"on a payphone\" by cloning a payphone recording.** Acoustics come from the MIX, or from `seed-audio` which renders a room.\n\n**DIRECTION GOES IN SQUARE BRACKETS. PARENTHESES ARE SPOKEN ALOUD.** `[whispering] I hope nobody notices` is whispered; `(quietly) I hope nobody notices` says the word \"quietly\" out loud. Verified by ear — the easiest way to ruin a take.\n\n- **Plain English works inside them** — it is natural-language steering, not a fixed vocabulary: `[very quiet]`, `[whisper in a hushed style]`, `[very slow]`, `[say excitedly]`. Non-verbals are their own tags: `[laugh]`, `[sigh]`, `[breathe]`, `[clear throat]`.\n- **A tag it does not recognise is still consumed, and still changes the read.** Never spoken, never an error — so a mistyped tag fails SILENTLY and only listening catches it.\n- **Tags persist across sentences** until changed; `[reset]` returns to normal.\n- **Punctuation is the timing.** `Wait. Stop.` differs from `Wait, stop.`\n- **One line, one take.** Split a paragraph so a bad clause costs one re-roll.\n- **Spell numbers and titles aloud:** `twenty twenty-six`, `Doctor Reyes`.\n\n**Route elsewhere when:** the scene needs dialogue mixed with effects and room tone in one pass (`seed-audio`), or it is a single non-speech sound (`eleven-sfx`). This surface makes ONE voice saying ONE thing, cleanly.\n\n**Hard constraints:** no duration parameter — length falls out of the text. Exactly one voice source: a preset `voiceId` from `slates_list_voices`, a clip as `voiceReferenceAssetId` (a character's voice clip to speak AS the character, or any clean clip of one speaker), or `voiceDescription`.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — the prompt on this surface is SPOKEN ALOUD, so anything that describes the audio instead of being the audio gets read out as words:\n\n- `SFX`, `Ambient noise`, `Background music` as labels — this model speaks; it does not render a scene. Use `seed-audio` for those.\n- shot language: `wide shot`, `slow push in`, `warm tungsten` — video-prompt words, and here they would literally be said aloud\n- `voiceover`, `narrator says`, `he says` as stage directions wrapping the line — write only the words that should come out of the speaker\n<!-- @banned:end -->\n\n## Why the prompt is not a prompt\n\nOn every other surface in Slates the prompt DESCRIBES what you want and the model interprets it. Here the prompt IS the deliverable: each character is spoken aloud and each character is billed. `a gravelly man says he is tired` produces a voice saying the words \"a gravelly man says he is tired\".\n\nThat also means the two numbers a user cares about are the same number. The text length sets the price (in 250-character buckets) and sets the length of the audio. There is nothing to choose and nothing to reconcile.\n\n## Steering the delivery\n\n`VERIFIED BY EAR, 2026-09-05.` Every claim in this section was listened to, not\ninferred — an earlier draft of this skill documented tag forms that had only been\nprobed for an HTTP 200, which proves the request was accepted and nothing about\nwhether it was obeyed.\n\n**Square brackets are consumed. Parentheses are read aloud.** That is the whole\nrule, and getting it wrong is not a subtle degradation — the audience hears a\nnarrator say the word \"quietly\" in the middle of your line.\n\n| Written | What comes out |\n|---|---|\n| `[whispering] I really hope nobody notices that.` | whispered, tag not spoken ✅ |\n| `[very quiet] I really hope nobody notices that.` | very quiet, tag not spoken ✅ |\n| `[whisper in a hushed style] …` | hushed, tag not spoken ✅ |\n| `[very slow] …` | slowed right down, tag not spoken ✅ |\n| `[laugh] …` | an actual laugh, then the line ✅ |\n| `(quietly, under his breath) …` | 🚨 **the words \"quietly, under his breath\" are SPOKEN** |\n\n**Plain English works — it is natural-language steering, not a fixed vocabulary.**\nBoth the documented phrasings (`[whisper in a hushed style]`) and ordinary adverbs\n(`[whispering]`) were obeyed. Write the direction the way you would say it to an\nactor.\n\nThe eight dimensions the model steers on, with a working example of each:\n\n| Dimension | Example |\n|---|---|\n| Emotion | `[say excitedly]`, `[sound sad]`, `[sound terrified]` |\n| Articulation | `[say with force]`, `[articulate clearly]` |\n| Intonation | `[say with a rising pitch]` |\n| Volume | `[very quiet]`, `[very loud]` |\n| Pitch | `[say in a low tone]` |\n| Range | `[say playfully]`, `[say with no pitch variation]` |\n| Speed | `[very fast]`, `[very slow]` |\n| Vocal style | `[whisper in a hushed style]`, `[give a nasal quality]` |\n\nNon-verbals sit inline where they happen: `[laugh]`, `[sigh]`, `[cough]`,\n`[breathe]`, `[yawn]`, `[clear throat]`.\n\n### Four rules that are not obvious\n\n1. 🚨 **A tag it does not recognise is still consumed, and still changes the read.**\n `[zzzqqq]` is not spoken and does not error — it produces a different, arbitrary\n delivery. So a typo in a tag is SILENT: there is no rejection, no warning, and no\n way to catch it except listening to the take. Treat an unexpected performance as\n a possible misspelled tag before you blame the voice.\n2. **Tags persist across sentences.** A `[very slow]` at the top governs everything\n after it until something changes it. Use `[reset]` to go back to normal rather\n than assuming the next sentence starts clean.\n3. **Do not stack opposing directions.** `[whisper in a hushed style]` together with\n `[very loud]` produces unpredictable results — the model is resolving a\n contradiction, and which side wins is not something you can rely on.\n4. **Tags COUNT toward the billed characters**, even though they are never spoken.\n They are part of the text sent to the vendor, so the vendor charges for them and\n so do we — billing what was actually sent is the only honest basis. It rarely\n matters (a 13-character tag inside a 250-character bucket), but a line sitting\n just under a bucket boundary can be pushed into the next one by a long\n direction. Prefer `[very slow]` over `[say this one very slowly please]`.\n\n## Identity versus acoustics, at length\n\nThis is the distinction that decides whether the feature feels good, and it is worth being precise about because the failure is quiet — you get a usable clip that is subtly not the person.\n\n**What a reference clip transfers:** vocal timbre, pitch range, accent and regional vowels, apparent age, speech rate tendencies, and the particular rasp or breathiness of the source speaker.\n\n**What it does not transfer:** the room, the microphone, the codec, the distance from the mic, any processing on the source, and any other sound present in it.\n\nSo the ideal reference is boring: one person, close to a microphone, no music, no second speaker, no heavy reverb, five to fifteen seconds, speaking normally rather than performing. A phone voice memo in a quiet room beats a beautifully produced clip with a music bed underneath it.\n\n**Two failure modes, both common:**\n\n- *\"I cloned my podcast intro and it doesn't sound like me.\"* The intro had music under it. The model averaged the music into the identity. Re-clone from a clean stretch.\n- *\"I want the line to sound like it's coming through a car radio.\"* Clone the clean voice, then EQ and process the returned clip on the timeline. A radio-sounding reference makes a worse voice, not a radio effect.\n\n## Getting the voice onto the call\n\nExactly one source per call, and none of them requires a character to exist first:\n\n- **A preset:** `slates_list_voices` lists stock voices with gender, age, accent and tags — filter by any of them, or search the descriptions (\"gravelly\", \"narration\"). Pass the chosen `voiceId`. Presets clone nothing, so they are the fastest path and avoid the clone-creation rate ceiling.\n- **Speak AS a character:** `voiceReferenceAssetId: <its voiceAssetId>` (the clip on the row `slates_list_characters` returns). The seat clones the clip for that take and discards the vendor voice afterwards, so there is nothing to reconcile — but cloning shares a ceiling of two new voices a minute across every Slates user, so a run of lines in one cloned voice pauses between takes rather than failing. Send each line once; do not re-send one that already came back. Any other clean clip of one speaker works the same way.\n- **A voice with no recording:** `voiceDescription` (7–1000 characters of words). If it will be used again, keep the returned clip on a character with `slates_update_character` (`voiceAssetId`) so later lines clone the same clip instead of designing a new voice each time — a convenience, never a requirement.\n\n## Consent\n\nCloning a real person's voice needs that person's explicit, documented permission, scoped to what you are making. Clone from original human recordings only — never from another model's output. This is the same gate the real-face route applies to likeness, and it applies here for the same reason.\n\n## Worked examples\n\n**A line with a direction**\n\n```\n[very quiet] I heard what you said in there. I'm not going to pretend I didn't.\n```\n\n**A line that needs its numbers spoken**\n\n```\nThe vote was three hundred and twelve to eighty-nine. It carried at four minutes past midnight.\n```\n\n**A paragraph, split into three takes** — so one bad clause costs one re-roll:\n\n```\n1. You keep asking me why I stayed.\n2. It wasn't loyalty. It wasn't even fear, not by the end.\n3. [very slow] It was that I couldn't picture the version of me that left.\n```\n\n**What NOT to send**\n\n```\n(gravelly, tired) a tired old man narrates the opening of the film, wide shot, warm tungsten\n```\n\nEvery word of that is spoken aloud — **including the parenthetical**, which is the\ntrap: it looks like a stage direction and is treated as dialogue. Describe the voice when you are CHOOSING one (`voiceDescription`, or the desktop's voice picker); the prompt is only ever the words.\n",
|
|
22
|
-
"slates-prompting-kling-v3": "---\nname: slates-prompting-kling-v3\ndescription: How to prompt Kling V3.0 (Kuaishou). Read before calling slates_generate_video with kling-v3.0-std, kling-v3.0-pro, or kling-v3.0-omni. Kling has dialogue + SFX + ambient native syntax (Omni adds multi-character dialogue and language codes). Multi-shot rules differ from Seedance/Veo — don't cross syntaxes.\n---\n\n# Kling V3.0 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Kling V3.0.** The general default. Define the core subjects clearly at the START and keep those descriptions identical across shots. Up to 15s, up to 6 cuts, and the strongest image-to-video identity hold in the catalogue.\n\n**The five levers**\n1. **Dialogue in quotes** — `Character says, \"exact words here\"`. On Omni, direct the voice with `Gender + Age + Voice quality + Speech rate + Emotional tone + Language`: `[Character A: Detective, mid-40s, raspy, slow cadence, weary]: \"I've seen this before.\"`\n2. **Unique speaker labels, no pronouns after the introduction.** `he`, `the agent`, any synonym causes voice drift.\n3. **Sound has real syntax** — `SFX: heavy boots on wet pavement, distant siren wailing`, `Ambient noise: city traffic`, `Background music: low cello`. Always physical-cause specific; `SFX: footsteps` is not enough.\n4. **Motion adverbs modulate energy directly** — `slowly`, `rapidly`, `gently`, `explosively`. One primary camera move per shot, never stacked.\n5. **On image-to-video, do NOT re-describe the image.** It is an anchor; prompt how the scene EVOLVES from it — movement, camera, environmental change.\n\n**Examples**\n- `A detective in a wet grey overcoat stands under a stairwell light. He steps forward slowly as the light flickers. [Character A: Detective, mid-40s, raspy voice, slow cadence, weary]: \"I've seen this before.\" SFX: heavy boots on wet concrete, distant siren wailing. Ambient noise: rain on metal.`\n- `Camera tracks right alongside a cyclist crossing a bridge at dusk. She rises out of the saddle rapidly as the grade steepens. Ambient noise: wind, tyres on wet asphalt, distant traffic.`\n\n**Hard constraint:** `Immediately` (Omni only) removes the natural conversational beat between speakers — use it when timing matters and leave it out when it does not. Kling has a real `negativePrompt` field, unlike Seedance; start from the standard block and layer scene-specific suppressions.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use:**\n- `SFX: footsteps` and any label-only effect — physical-cause specificity or nothing\n- a pronoun or synonym for a speaker after the first introduction (`he`, `the agent`) — it causes voice drift; repeat the full label\n- `single continuous take` — Seedance's phrase, and it fights Kling's multi-shot\n<!-- @banned:end -->\n\nKuaishou's video model. Three tiers: `kling-v3.0-std` (general use, no audio), `kling-v3.0-pro` (higher visual quality, no audio), `kling-v3.0-omni` (multi-character dialogue + audio-visual co-generation).\n\nUp to 15s. Multi-shot supported (up to 6 cuts in 15s total). Strong on image-to-video — preserves identity, layout, and text from the input image well.\n\n## Subject definition rule (verbatim, fal blog)\n\n> \"Define your core subjects clearly at the beginning of the prompt and keep descriptions consistent across shots.\"\n\n## Dialogue syntax\n\n```\nCharacter says, \"exact words here\"\n```\n\nUse quotation marks for precise speech. Languages (Omni only): EN, ZH, JA, KO, ES.\n\n## Voice direction formula (Omni)\n\n```\nGender + Age Range + Voice Quality + Speech Rate + Emotional Tone + Language\n```\n\nExample:\n```\n[Character A: Detective, mid-40s, raspy voice, slow cadence, weary]: \"I've seen this before.\"\n```\n\nTone phrases that fire:\n- `speaking in a hushed, trembling whisper`\n- `shouting with commanding authority`\n- `clear, fearful voice`\n- `with a trembling voice, \"I'm scared\"`\n\n## The `Immediately` keyword (Omni only)\n\nWithout `Immediately`, Kling adds a natural conversational beat between speakers. With it, dialogue is back-to-back. Use when timing matters.\n\n```\n[Alice]: \"Get down!\" Immediately, [Bob]: \"Where?\"\n```\n\n## Speaker label discipline\n\nUnique labels per character. **No pronouns or synonyms after first introduction** — they cause voice drift.\n\n✅ `[Character A: Black-suited Agent]` ... `[Character A: Black-suited Agent]: \"Stop.\"`\n❌ `[Agent]... then he says...`\n\n## Multi-character dialogue (Omni)\n\n```\nAlice says in English, \"Hello!\" Then Bob replies in Spanish, \"¡Hola!\"\n```\n\n## Sound effects, ambient noise, music\n\n```\nSFX: thunder cracks, footsteps approaching\nAmbient noise: city traffic, birds chirping, ocean waves\nBackground music: tense orchestral strings, low cello\n```\n\nSFX accepts physical-cause specificity:\n- ✅ `SFX: heavy boots on wet pavement, distant siren wailing`\n- ❌ `SFX: footsteps`\n\n## Image-to-video guidance\n\n**Verbatim (fal blog):**\n> \"Treat the input image as an anchor. Kling 3.0 excels at preserving the identity, layout, and text details. Focus prompts on how the scene evolves *from* the image: subtle movements, camera motion, or environmental changes.\"\n\n**Don't re-describe what's already in the image.** Focus on motion, changes, evolution.\n\n## Multi-shot — what makes them hit\n\n**Hard cap: total duration ≤ 15s across all shots. Max 6 cuts.**\n\nHit conditions:\n- Shot labels are explicit: `Shot 1:`, `Shot 2:`\n- One primary action per shot\n- Subject described identically in each shot block\n- Camera move per shot is **one verb**, not a chain\n- Per-shot blocks: 30-60 words\n\nMiss conditions:\n- Compressing narrative into one paragraph\n- Pronoun-only references after the first shot\n- Mixing camera moves within a shot (\"pan then orbit then push in\")\n- Extreme wide → extreme close in adjacent shots without reference images\n\n## Element references (Omni)\n\nUpload 2-4 multi-angle reference photos per character/object. Tag inline:\n\n```\n@element1 is the protagonist (refs: front, side, back angles).\n@element2 is the antagonist.\n```\n\n## Reference discipline (character / environment refs)\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n<!-- @inject:reference-rules-core -->\nIdentity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.\n<!-- @end:reference-rules-core -->\n\n### For Kling specifically\n\n- **Kling's consistency lever is \"lock the subject with a fixed label reused verbatim.\"** That is Kling's phrasing for rules 2 and 3, and it is stricter than the others: **pronoun and synonym drift breaks it**, so the exact same label must appear on every single mention — not \"he\", not \"the detective\" after you named him. Reusing the label verbatim is the whole game. Slates composes this for you from `@mentions`.\n- **Element references are the transport for rule 1** — 2-4 multi-angle photos per character/object, tagged `@element1` / `@element2` (see Element references above). The cap is 4 combined refs on the edit path.\n\n## Negative prompting — has a real field\n\nKling exposes `negative_prompt` on the fal endpoint (different from Seedance which has none). Default block to start from:\n\n```\nblurry, low quality, watermark, text overlay, distorted hands, extra fingers,\nduplicate limbs, unnatural skin texture, overly saturated colors,\nfloating objects, inconsistent shadows, jittery, flickering, morphing face\n```\n\nLayer scene-specific suppressions on top, and never suppress something the prompt asks for. This block carried `lens flare` until 2026-09-15, which silently cancelled every flare a prompt described (`slates-cinematic-look` → `source-flare`); add it back only for a shot that must have none.\n\n## Cinematic tactics\n\n- **Motion adverb precision** modulates motion energy directly: `slowly`, `rapidly`, `gently`, `explosively`\n- **Camera vocabulary that registers as instructions:** profile shot, tracking, following, freezing, panning, \"moving in sync with the subject\"\n- **One primary camera move per shot** — never stack\n\n## Tier choice\n\n- **Standard**: general use, no audio\n- **Pro**: higher visual quality, no audio\n- **Omni**: multi-character dialogue, audio-visual co-gen, language codes, `@elementN` references\n\nPick by capability: need dialogue/audio → Omni; need maximum visual quality silent → Pro; everything else → Standard. Prices change — check current numbers before choosing a tier<!-- slates-only -->; call `slates_estimate_generation_cost` or `slates_list_available_models`<!-- /slates-only -->.\n\n## Benchmark prompt structure\n\n```\n[Character A: <role>, <voice quality>]: \"<line>.\" Immediately, [Character B: <role>, <voice quality>]: \"<reply>.\"\nAmbient noise: <soundscape>.\nCamera <single move>.\n```\n\nCinematic example (paraphrasing fal blog patterns):\n> \"Shot 1: Wide establishing shot of a neon-lit alleyway in heavy rain, steam rising from grates. Camera slowly tracks forward.\n> Shot 2: Medium shot of a detective in a trench coat ducking under an awning, water dripping from his hat brim. [Detective: weary, raspy]: 'I knew she'd come back.' Ambient noise: distant traffic, rain on metal.\n> Shot 3: Close-up on his eyes, narrowing as headlights flash across his face.\"\n\n<!-- slates-only -->\n## Pre-flight: references arrive inline, refer by code\n\nWhen you call `slates_generate_video` with `firstFrameAssetId` or `ingredientAssetIds`, the first call returns those references **inline as image content blocks** alongside cost + `requires_confirm: true`. Look at them, revise prompt if needed, then re-call with `confirm=true`. Kling Omni multi-character with several ingredient images especially benefits — confirm each character image lands cleanly before spending.\n\nWhen talking to the user about the gen, refer to each reference by its short code: `IMG-A12 — Detective Closeup`. The user sees that code as a gallery badge.\n\n- ✅ \"I'm anchoring on **IMG-A12** as the detective and **IMG-A18** as the alleyway environment — Omni will handle the line delivery in EN.\"\n- ❌ \"I'm using the detective image and the alley one...\" (which alley? Three exist.)\n<!-- /slates-only -->\n\n## Video-to-video EDIT<!-- slates-only --> (`slates_edit_video`)<!-- /slates-only --> — @Video1 / @ElementN / @ImageN\n\nKling O3 edit takes an EXISTING 3-15s clip and changes only what the prompt names — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved by default. Its notation is Kling's own, different from the \"image N\" naming used everywhere else:\n\n- **`@Video1`** — the source clip (always; the transport anchors the instruction to it).\n- **`@Element1..`** — subjects to swap IN. Each element = one frontal image + up to 3 angle images<!-- slates-only --> (pass as `characterAssetIds`; @mention names in the prompt compile to @ElementN automatically)<!-- /slates-only -->.\n- **`@Image1..`** — style/appearance references<!-- slates-only --> (pass as `styleAssetIds`)<!-- /slates-only -->.\n- Max **4 combined** element + image refs per edit.\n\n**Prompt shape — the change, not the whole scene:**\n\n```\nReplace the man in @Video1 with @Element1, keeping his walk cycle, the camera move, and the rain unchanged.\n```\n\n```\nEdit @Video1: turn the daytime street into a neon-lit Tokyo alley at night, wet asphalt reflections. Apply the visual style of @Image1. Keep the subject and camera motion exactly as they are.\n```\n\nRules:\n- Name what CHANGES; explicitly state what stays (\"keep the motion / camera / everything else unchanged\") — the model preserves better when told to.\n- One edit intent per pass. Chain passes for compound changes (each output is itself an editable clip, linked to its parent).\n- Billing is per second of OUTPUT ≈ the clip length, rounded UP to the next second. A 7.3s clip bills as 8s.\n- Clip constraints: 3-15s, 720-3840px, MP4/MOV. Agents can pre-trim on the timeline when a clip runs long.\n- Routing: Kling edit is the default edit tool (element lock + audio intact); Seedance edit/relocate wins style-transfer-heavy re-imaginings<!-- slates-only --> — see `slates-model-selection`<!-- /slates-only -->.\n\n## Sources\n\n- [fal.ai — Kling 3.0 Prompting Guide](https://blog.fal.ai/kling-3-0-prompting-guide/)\n- [Vidguru — Kling 3.0 Omni Guide](https://www.vidguru.ai/blog/kling-3.0-omni-guide.html)\n- [AcceptPrompt — Kling 3 Prompt Guide](https://www.acceptprompt.com/blog/kling-3-prompt-guide)\n- [DataCamp — Kling 3.0 Tutorial](https://www.datacamp.com/tutorial/kling-3-0)\n",
|
|
23
|
-
"slates-prompting-lip-sync": "---\nname: slates-prompting-lip-sync\ndescription: How to set up lip-sync — Kling-only (dedicated lip-sync and avatar endpoints, 5-second outputs). Read before calling slates_generate_lip_sync. Two flows — video→video re-dub and image→video avatar — with different inputs, pricing, and gotchas. Voice catalog, framing rules, audio file constraints, and which tier to pick. Also covers the Seedance alternative, which is a normal video generation rather than a mode of this tool.\n---\n\n# Lip-sync — setup guide\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Lip-sync (Kling only).** Two different flows with different inputs and different prices; every output is 5 seconds.\n\n**The five levers**\n1. **Pick `sourceType` deliberately** — `video` re-dubs an existing talking head (cheapest); `image` animates a still portrait (avatar-standard, then avatar-pro only on the final selected take).\n2. **The `prompt` on the avatar flows is SCENE CONTEXT, not motion direction.** Ambience, lighting, micro-expression: `Soft rim light`, `warm office`, `cool blue evening light through a window`, `gentle confident smile between sentences`, `focused intent expression`.\n3. **Clean the audio before uploading** — `noise-reduced`, `levelled`. Lip detection is sensitive, and a raw recording is the most common cause of a bad take.\n4. **Iterate on the SOURCE or the AUDIO, never on a refinement prompt** — there is not one. If the output is wrong, change the input.\n5. **Use avatar-standard for first-pass dialogue takes**, and switch to pro only once the line is locked. Facial fidelity is not visible until then.\n\n**Examples**\n- `Soft rim light, warm office, gentle confident smile between sentences.`\n- `Cool blue evening light through a window, focused intent expression.` (Or `.` — an empty prompt is fine when you have nothing to add.)\n\n**Hard constraint:** it is Kling-only and always 5 seconds. For a generated PERFORMANCE instead — head movement, gesture, delivery energy, with the dialogue as a native conditioning signal — that is a normal Seedance video generation with the clip attached as a video reference, not a mode of this tool. A real recording, or a cloned/cast voice rendered on `inworld-tts-2`, for production; this tool's built-in TTS is for scratch.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — the avatar prompt is scene context and motion verbs are ignored:\n- `turns her head`, `raises an eyebrow`, `hand gestures`, `nods`, `walks`\n- `reader_en_m-v1` — listed in fal's docs, returns \"Voice id not found\" in production\n<!-- @banned:end -->\n\n**This tool is Kling-only.** It wraps Kling's dedicated lip-sync and avatar endpoints; every entry is a real endpoint and every output is 5 seconds.\n\n| Flow | Source | Model | Cost | Use case |\n|------|--------|-------|-----------|----------|\n| Re-dub | video clip | kling-lip-sync-video | ~4 credits / 5s | Replace dialogue on an existing talking head |\n| Avatar standard | still image | ai-avatar/v2/standard | ~14 credits / 5s | Animate a portrait into a talking avatar |\n| Avatar pro | still image | ai-avatar/v2/pro | ~29 credits / 5s | Higher facial fidelity for hero shots |\n\nPick `sourceType` deliberately — it decides the pricing tier and the underlying endpoint.\n\n## Want Seedance instead? That is a video generation, not a mode here\n\nSeedance can generate the performance rather than bolting a mouth onto finished pixels — head movement, gesture, delivery energy, with the dialogue as a native conditioning signal, and a video source keeps its own voice. **It is not an engine switch on this tool.** Run a normal `slates_generate_video` on `seedance-2` with the clip (or portrait) attached as a video/ingredient reference and the dialogue written into the prompt yourself.\n\nThat is the same endpoint the old `engine=seedance-2` branch called — it just built the sentence for you, invisibly, and it presupposed a \"video 1\" that might not exist. Writing the prompt is the whole difference, and it is the part you want control of.\n\n- Driving clips must be 2–15s; output duration is whatever you set (4–15s).\n- Video references bill COMBINED input+output seconds (`seedance-2*-vref-*` keys) — pass the clip duration and quote before confirming. On Seedance 2.5's AI-face route (EvoLink) the input side counts as at least the output's length: max(input, output) + output.\n- Faces go through the normal cascade: `seedanceFace` for a character, `[REAL_FACE_DETECTED]` → `seedanceRealFace` + `realFaceConsent` for a real person.\n\nEverything below is about the Kling tool.\n\n## Choosing video vs avatar\n\nUse **video** (re-dub) when:\n- A talking-head clip already exists (Slates-generated, recorded, or imported)\n- The mouth/face is already moving and only the audio needs to change\n- ~4 credits is hard to beat for short dialogue replacement\n\nUse **avatar** when:\n- Only a still portrait exists\n- The character needs to come alive from a single image\n- Identity + face fidelity matter (avatar-pro for hero shots, standard for everything else)\n\n## Source asset constraints\n\n### Video flow (`sourceType: 'video'`)\n- Format: mp4 or mov\n- Duration: 2–10s (lip-sync output is always 5s — long videos get trimmed)\n- Resolution: 720p or 1080p (480p will be rejected)\n- Max file size: 100MB\n- Face must be visible and roughly facing camera. Profile shots fail.\n- Existing audio is replaced.\n\n### Avatar flow (`sourceType: 'image'`)\n- Min 512×512, PNG/JPG/WebP\n- **Face occupies 60–70% of frame.** This is the single biggest avatar quality lever.\n- Eyes open, mouth neutral, looking near-camera. Side profile = bad output.\n- Single subject, clean background. Group photos confuse the face anchor.\n\n## Audio source\n\nTwo ways to drive the lips:\n\n### TTS (`audioMethod: 'tts'`)\n- Pass `ttsText` (the words spoken)\n- Optional: `ttsVoice` (default `oversea_male1`), `ttsLanguage` (default EN), `ttsSpeed` (default 1.0)\n- **Hard cap: 120 characters of text.** Longer = silently truncated.\n- Languages: EN, ZH, JA, KO, ES\n\n### Upload (`audioMethod: 'upload'`)\n- Pass `audioFilePath` — absolute path to an audio file on the user's machine\n- Format: mp3, wav, m4a, ogg, aac\n- Max 5MB\n- Duration: 2–60s (output is 5s — longer audio gets trimmed)\n- Single clean voice. Music underneath, multiple speakers, or noisy mics produce garbage lips.\n\nPrefer upload for production-quality voice. TTS for fast iteration / placeholder dialogue.\n\n## Voice catalog (TTS)\n\nReliable English voices (verified working on the fal endpoint as of 2026):\n\n| Voice ID | Description |\n|----------|-------------|\n| `oversea_male1` | Male, English — default, stable |\n| `commercial_lady_en_f-v1` | Female commercial English |\n| `uk_boy1` | Young man, UK accent |\n| `uk_man2` | Man, UK accent |\n| `uk_oldman3` | Older man, UK accent |\n| `calm_story1` | Storyteller / narrator |\n\nAvoid `reader_en_m-v1` — listed in fal.ai docs but returns \"Voice id not found\" in production.\n\nFull 48-voice list (ZH, JA, KO included): https://fal.ai/models/fal-ai/kling-video/lipsync/text-to-video/api\n\n## Speech-rate notes\n\n`ttsSpeed` range 0.5–2.0:\n- 0.8–1.0: natural conversational\n- 1.1–1.3: punchy ad delivery\n- 1.4+: rushed, clips consonants\n- 0.6–0.7: slow, weighty (good for dramatic lines)\n\nDefault 1.0 unless the line specifically calls for slower or faster cadence.\n\n## Avatar prompt usage\n\nThe `prompt` parameter on avatar-v2 (standard + pro) is **scene context**, not motion direction. The mouth animation comes from the audio — the prompt sets ambiance, lighting, micro-expression.\n\nGood:\n- `Soft rim light, warm office, gentle confident smile between sentences.`\n- `Cool blue evening light through a window, focused intent expression.`\n\nBad (the model ignores motion verbs):\n- ❌ `She turns her head, raises an eyebrow, then speaks.`\n- ❌ `Hand gestures while talking.`\n\nDefault `\".\"` is fine if you have nothing useful to add.\n\n## Tier selection — avatar standard vs pro\n\n**Use standard** when:\n- Drafts, A/B testing voices, internal review reels\n- Wide / medium shots where face isn't the focal point\n- Cost matters more than micro-expression fidelity\n\n**Use pro** when:\n- Final ads where the avatar's face fills the screen\n- The character is named / branded — identity drift kills the take\n- You're already paying tens of credits for the surrounding video pipeline\n\nDon't default to pro. The ~15-credit delta per take adds up across iteration.\n\n## Common failure modes\n\n| Symptom | Likely cause | Fix |\n|---------|--------------|-----|\n| Lip movement looks \"rubber\" / disconnected | Source face <60% of frame | Re-crop the still tighter |\n| Voice doesn't match character age/gender | Default voice id used | Pick from voice catalog |\n| Output truncated mid-word | TTS text >120 chars | Shorten or chain two takes |\n| Garbled mouth on uploaded audio | Background music / multi-voice | Use clean dialogue-only audio |\n| \"Voice id not found\" 422 | Hit `reader_en_m-v1` | Switch to `oversea_male1` |\n| Avatar eyes drift / cross | Source had closed/angled eyes | Pick a frame with neutral open eyes |\n| Generation completes but lips don't move | Profile shot / face >70° off-axis | Use a near-frontal portrait |\n\n## Cost discipline\n\n- Video re-dub at ~4 credits is the cheapest dialogue iteration in the entire Slates stack — use it for voice A/B testing\n- Avatar standard at ~14 credits is fine for medium use\n- Avatar pro at ~29 credits trips the confirm gate — explicit user OK required every time\n- All 5s. There is no shorter option.\n\n## Workflow patterns\n\n**Voice A/B test (cheap):**\n1. Generate one base talking-head video clip with Veo or Seedance (~40 credits)\n2. Run `slates_generate_lip_sync` with `sourceType: 'video'` against 3–5 different `ttsVoice` values\n3. Total cost: ~40 + (5 × ~4) ≈ 60 credits to compare voices\n\n**Brand avatar from a single portrait:**\n1. Generate or upload the hero portrait (face fills frame, eyes open, neutral mouth)\n2. Avatar standard for first-pass dialogue takes\n3. Avatar pro only on the final selected take\n\n**Avoid:**\n- Avatar pro on first iteration (waste — facial fidelity isn't visible until you've locked the line)\n- TTS for final ads (production should use real voice or cloned voice — the upload flow)\n- Uploading raw recordings — clean noise + level the file first, lip detection is sensitive\n\n## Confirm gate: cost + codes, no inline preview\n\nLip-sync is mechanical — the model re-syncs the chosen source to the chosen audio. The confirm response carries the source asset's code so you can announce it in chat.\n\n- ✅ \"Lip-syncing **IMG-A12 — Founder Portrait** to the new line. ~29 credits on avatar-pro. Confirm?\"\n- ❌ \"Using the founder image...\" (which? Three exist.)\n\nDon't second-guess the source. If the output is wrong, iterate on source choice or audio, not on a refinement prompt (there isn't one).\n\n## Sources\n\n- [fal.ai — Kling LipSync API](https://fal.ai/models/fal-ai/kling-video/lipsync/text-to-video/api)\n- [fal.ai — AI Avatar v2 Standard](https://fal.ai/models/fal-ai/kling-video/ai-avatar/v2/standard/api)\n- [fal.ai — AI Avatar v2 Pro](https://fal.ai/models/fal-ai/kling-video/ai-avatar/v2/pro/api)\n",
|
|
24
|
-
"slates-prompting-ltx-2-5": "---\nname: slates-prompting-ltx-2-5\ndescription: How to prompt LTX-2.5 and LTX-2.5 Pro. Read before calling slates_generate_video with model ltx-2-5 or ltx-2-5-pro. LTX scores the picture on the same pass that draws it, so SOUND IS THE FIRST THING YOU WRITE — Lightricks ranks the prompt sound, camera, character detail, shot type and scene, then scene dressing, all in one flowing paragraph. It is also the catalogue's native MULTISHOT seat: one generation carries two to four connected shots holding character, light and voice across the cuts. Base ltx-2-5 is the distilled build — 720p/1080p/1440p/4K, clips of 6 to 20 seconds in EVEN steps, and the cheapest native 1080p second in Slates; ltx-2-5-pro is the full diffusion build and is NOT a superset, reaching only 1080p and 10 seconds for about a third more money. Three hazards live here: durations are even numbers only from six (there is no 5s or 7s clip), the model has NO reference endpoint at all so identity references are unavailable, and any sound not anchored to something in frame gets invented for you.\n---\n\n# LTX-2.5 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — LTX-2.5.** It scores the picture on the same pass that draws it, so SOUND IS THE FIRST THING YOU WRITE. Lightricks' own priority order: sound, camera, character detail, shot type and scene, then scene dressing. One flowing paragraph, not labelled sections. When a prompt sprawls, cut from the bottom.\n\n**The five levers**\n1. **Lead with sound, and anchor every sound to something in frame.** The test is \"visible, or at least locatable\" — `the rope creaks against the cleat`, `the hull knocking hollow against the fenders`, `rain on the awning`. A distant whistle is fine IF you have named the post it comes from.\n2. **Camera second**, because framing decides the visual weight of the shot — `low camera at the gunwale`, `slow drift right`, `static medium behind the counter`.\n3. **Character detail as physical ACTION**, not as adjectives about a person — `she braces a boot on the rail and hauls`, `his hands counting notes`.\n4. **It is the native MULTISHOT seat** — one generation carries two to four connected shots holding character, light and voice across the cuts. Write the cuts.\n5. **Quote dialogue and name the language and accent** — `in English with a slight German accent` — `\"We should not have come back,\" in English with a slight German accent.`\n\n**Examples**\n- `The rope creaks against the cleat as she leans back, gulls calling somewhere off the port bow, the hull knocking hollow against the fenders. Low camera at the gunwale, slow drift right. She braces a boot on the rail and hauls, twice, then stops.`\n- `A till drawer bangs shut, a fan ticks against its cage, rain on the awning outside. Static medium behind the counter, then cut to a close-up of his hands counting notes, then cut wide as he looks up at the door.`\n\n**Hard constraint:** durations are EVEN numbers from six — there is no 5s or 7s clip. There is NO reference endpoint at all, so identity references are unavailable; use MiniMax H3 or Kling when a character must hold across shots. Any sound not anchored to something in frame gets invented for you. And never write mood adjectives as sound: \"tense atmosphere\", \"a sense of dread\" and \"ominous ambience\" produce nothing usable — the fix is one more moving object in frame with a sound attached to it.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — mood adjectives standing in for sound produce nothing usable:\n- `tense atmosphere`, `a sense of dread`, `ominous ambience`, `eerie silence`\n- an unanchored sound: name the thing in frame it comes from, or cut it\n<!-- @banned:end -->\n\nLTX-2.5 generates picture and sound **in a single pass**, with a Gemma-4 12B text encoder reading\none flowing paragraph. That single fact drives everything below: the prompt is not a shot\ndescription with audio bolted on, it is **a scene where the sound is load-bearing** — and\nLightricks' own priority order puts sound first, ahead of the camera.\n\nTwo seats, and the naming is a trap:\n\n| | `ltx-2-5` (base) | `ltx-2-5-pro` |\n|---|---|---|\n| Build | Distilled, 8-step | Full diffusion (\"Diffusion Fidelity Rendering\") |\n| Resolutions | 720p / 1080p / **1440p** / 4K | 720p / 1080p |\n| Durations | 6–20s, even steps | 6 / 8 / 10s |\n| Price | $0.09–$0.30 per second | $0.12–$0.17 per second |\n| Reach for it when | iterating, long takes, 4K delivery, batch volume | one dense final render inside 1080p and 10s |\n\n**Pro is not \"base plus more.\"** It buys picture quality on a *narrower* envelope — it cannot make\na 1440p frame and it cannot make a 12-second clip. Reaching for it out of habit costs a third more\n*and* takes away the reach.\n\n---\n\n## 1. The six parts, in priority order, in one paragraph\n\nLightricks ranks the elements of an LTX prompt like this. When a prompt sprawls, **cut from the\nbottom.**\n\n1. **Sound** — highest priority; the model scores the picture as it draws it.\n2. **Camera** — framing decides visual weight and the feel of the shot.\n3. **Character detail** — expressed as physical action.\n4. **Shot type and scene** — the action itself.\n5. **Scene dressing** — the first thing to trim.\n\nWrite it as **one flowing paragraph**, not a list of labelled sections. LTX is not Seedance (eight\nengineering slots) and not H3 (three separate audio layers) — it wants continuous prose.\n\n---\n\n## 2. Sound: anchor it or it gets invented\n\n**Write the audio line last, then go back and check every cue has a source you could point at.**\nAnything unanchored, the model invents for you.\n\nThe test is **\"visible, or at least locatable.\"** A distant whistle is fine *if* you have named the\nmarshal's post it comes from. A \"distant whistle\" with nothing to attach to is a coin flip.\n\n> the rope creaks against the cleat as she leans back, gulls calling somewhere off the port bow,\n> the hull knocking hollow against the fenders\n\n**Never write mood adjectives as sound.** \"Tense atmosphere\", \"a sense of dread\" and \"ominous\nambience\" produce nothing usable. If a scene feels thin, the fix is **one more moving object in\nframe with a sound attached to it** — never another adjective.\n\n### Dialogue\n\nQuote it, and name the language and accent:\n\n> \"We should not have come back,\" in English with a slight German accent.\n\nTwo rules that decide whether the lip sync lands:\n\n- **Give the character a beat of stillness before they speak.** The sync needs something to lock\n against; a character already mid-motion when the line starts drifts.\n- **Describe the beat structure** — when they look, how long they wait, when they speak, where they\n look afterwards.\n\nSlates pins the frame rate at 25fps, which is also what Lightricks recommends for dialogue: at 50fps\nthe performance \"pulls toward a video look.\"\n\n---\n\n## 3. Character emotion is physical\n\nThe model renders actions. It does not render adjectives.\n\n| Instead of | Write |\n|---|---|\n| she looks anxious | her jaw sets, she turns the ring on her finger twice |\n| he seems exhausted | he blinks slowly and lets his shoulder take the doorframe |\n| a tense standoff | neither moves; his thumb finds the strap and stays there |\n\n---\n\n## 4. Multishot — the thing this model is uniquely for\n\n**One LTX generation can carry several connected shots**, holding character, environment, lighting,\nvoice and style across every cut. Nothing else in the catalogue does this natively; everywhere else\nyou generate separate clips and stitch them, and identity drifts between them.\n\n**Working range is two to four shots.** Three is the comfortable stopping point.\n\nAt **every** transition you must supply four things:\n\n1. **Name the edit in the prose** — \"hard cut\", \"dissolve\", \"match cut\".\n2. **Re-establish the shot completely** — scale, angle, lens and light all reset at a cut. A cut is\n not a continuation.\n3. **Re-identify recurring characters by their original descriptor.** \"The woman in the bronze\n gown\", never \"she\". Pronouns lose the character across a cut — this is the single most common\n multishot failure.\n4. **State what the sound does at the cut.** Silence is not assumed; if the room tone should drop\n out, say so.\n\nA shape that works:\n\n> Wide establishing shot of the workshop, dust in the window light, a lathe turning somewhere off\n> frame — hard cut — macro close-up of the brass fitting as it seats, the turning noise gone,\n> replaced by a single dry click — match cut — medium shot of the woman in the bronze gown stepping\n> back, the room tone returning underneath her.\n\n---\n\n## 5. Camera: write it, don't enumerate it\n\nfal exposes a `camera_motion` enum (dolly in/out/left/right, jib up/down, static, focus shift).\n**Slates does not surface it, deliberately** — and prose is the better instrument anyway:\n\n- **A written move can be tied to a specific moment.** \"A slow push-in that settles as she reaches\n the door, then holds\" is not expressible as an enum value.\n- **For multishot it would be actively wrong** — one enum value would impose a single camera\n behaviour on three shots that each want their own.\n\nSo name the lens, the framing, the move, and **the moment the move resolves**.\n\n---\n\n## 6. The hard constraints\n\n### Durations are even numbers only, starting at six\n\n**6, 8, 10, 12, 14, 16, 18, 20.** There is no 5-second LTX clip and no odd duration of any length.\nAsking for 7s is not a rounding matter — that generation does not exist.\n\n**And the long end is 1080p-and-below only.** At 1440p and 4K the ceiling drops to **6, 8 or 10**.\n\nfal's own default is `auto`, which lets the model pick the length from the described action.\n**Slates always sends an explicit length instead**, so what you choose is what you are billed for.\nChoose the length the beat needs.\n\n### Aspect ratios: 16:9 and 9:16, and nothing else\n\nThe narrowest set in the catalogue alongside Veo. Square, 4:5 and 21:9 are not available on this\nmodel at any resolution.\n\n### Frames, not references\n\nLTX takes a **start frame** and an **optional end frame** (which generates a transition between the\ntwo). It has **no reference-to-video endpoint at all** — no identity references, no style\nreferences, no environment references, no reference video, no reference audio.\n\n**For character consistency across separate shots, use MiniMax H3 or Kling.** Within a single LTX\ngeneration, use multishot instead — that is precisely the gap it fills.\n\nIn image-to-video, **do not cut away from the opening frame too early.** You have paid for that\nframe; let it play before the first move.\n\n### Do not ask for text on screen\n\nNeither the spelling nor its stability from frame to frame can be relied on. Signage, labels,\ncaptions and lower-thirds belong in post.\n\n---\n\n## 7. Audio is free here, and that changes the routing\n\nNative synchronised audio is **included at every resolution on both seats**, with no surcharge and\nno toggle that costs money — unlike Kling, where sound is a paid dimension. A 6-second 1080p LTX\nclip **with sound** is 39 credits.\n\nCombined with 1080p at $0.13/s — the cheapest native 1080p second in Slates — this makes LTX **the\ncoverage seat**: the one to reach for when the job is many takes rather than one hero shot, when a\nsequence needs its own sound, or when the credit budget is the binding constraint.\n\nRoute away from it when you need identity references (H3, Kling), a ratio other than 16:9 or 9:16\n(Seedance, Kling), or authored multi-layer audio direction (H3).\n",
|
|
25
|
-
"slates-prompting-minimax-h3": "---\nname: slates-prompting-minimax-h3\ndescription: How to prompt MiniMax H3, H3 Max and H3 Max Turbo. Read before calling slates_generate_video with model minimax-h3, minimax-h3-max or minimax-h3-max-turbo. H3 is the only Slates video seat where AUDIO IS AUTHORED rather than toggled — synchronised dialogue, scene sound and an audience-only score are three separate sections of the prompt, generated in one pass — and the only one where a reference carries a DECLARED RELATIONSHIP (kept whole, partly kept, transferred onto a different subject, or a loose echo). Base minimax-h3 runs 480p/768p/2K/4K and reads 9 images + 3 video + 3 audio references; minimax-h3-max is fal's faster post-train, runs 480p/768p plus a 1080p refinement of its 768p render, and costs MORE than base H3 at 768p — a deliberate speed pick, never the default and never the cheap one; it animates start and end frames AND takes the same 9+3+3 omni-reference set (corrected 2026-09-09). minimax-h3-max-turbo is a second fal post-train with Max's ladder at half Max's rate; it takes start and end frames but has NO reference endpoint. Two hazards live here: reference images past the free allowance are billed (5 free then +4 credits on base H3; pooled media tokens on Max), and audio written into the wrong section is dropped or duplicated.\n---\n\n# MiniMax H3 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — MiniMax H3.** The only seat where audio is AUTHORED rather than toggled: dialogue, scene sound and score are three separate sections of the prompt, generated in one pass, and putting a sound in the wrong section drops or doubles it.\n\n**The five levers**\n1. **Write the three audio layers separately** — `Scene sound:` for what is in the room, `Score:` for what only the audience hears, and the dialogue quoted inline. Section decides attribution.\n2. **Quote dialogue and name the language** — `says in English`, `speaks in Spanish`. Eleven languages are stably supported; the language is part of the instruction, not an afterthought.\n3. **Declare the reference RELATIONSHIP**, which no other seat has: `kept whole`, `partly kept`, `transferred`, or `a loose echo`. An undeclared reference is a guess.\n4. **Give a beat of stillness before a line** — `sits still for a beat, then looks up`. The sync needs something to lock against; a character already mid-motion when the line starts drifts.\n5. **Describe the beat structure** — `waits`, `then speaks`, `under the last three seconds`. H3 is a timeline, so write one.\n\n**Examples**\n- `A woman sits still at a kitchen table for a beat, then looks up. She says in English, \"You said Tuesday.\" Scene sound: a fridge hum, a spoon set down on formica. Score: none.`\n- `Two mechanics either side of an open bonnet. The younger one wipes his hands, waits, then speaks in Spanish, \"No es el alternador.\" Scene sound: a socket wrench, a radio two bays over. Score: a low sustained cello under the last three seconds, audience only.`\n\n**Hard constraint:** the three seats differ in what the ENDPOINT accepts, not in grammar. Base H3 reaches 2K/4K and takes references; `minimax-h3-max` tops out at 1080p, takes the same 9+3+3 references, and costs MORE at the tier they share — a speed pick, never the cheap one; `minimax-h3-max-turbo` has Max's ladder at half its rate and takes frames only, NO references. Every tier above 768p is built from the native 768p render: judge at native. Reference inputs affect the quote; include every attached modality when estimating.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use:**\n- a sound written into the wrong audio section — it is dropped, doubled, or attributed to the wrong layer\n- `background music` as a bare instruction: the score is its own authored layer, audience-only, and it is named as such\n- an undeclared reference relationship — say kept whole, partly kept, transferred, or a loose echo\n<!-- @banned:end -->\n\nH3 is an **omni transformer**: it generates picture and sound in the same pass, at 24fps with\n32kHz stereo, 5–15 seconds, in 11 stably-supported languages (Arabic, Chinese, English, French,\nGerman, Italian, Japanese, Korean, Portuguese, Russian, Spanish). That single fact drives\neverything below — the prompt is not a shot description with sound bolted on, it is a **timeline\nwith three audio layers you author separately**.\n\n**Three seats, one grammar.** Everything in this file applies to all three. They differ only in\nwhat the endpoint accepts:\n\n| | `minimax-h3` | `minimax-h3-max` | `minimax-h3-max-turbo` |\n|---|---|---|---|\n| Resolution | 480p / 768p / **2K / 4K** | 480p / 768p / 1080p | 480p / 768p / 1080p |\n| References | 9 images + 3 video + 3 audio (12 files) | 9 images + 3 video + 3 audio (12 files) | **none** (no reference endpoint) |\n| Frames | start and/or end | start and/or end | start and/or end |\n| Price at 768p | **$0.060/s** | $0.080/s | $0.040/s |\n| Why pick it | resolution, references, and the cheaper second | **speed** — a 5s 768p clip in **4.8s** vs **57s** (measured) | **price** — half Max's rate at every tier |\n\n**Max is the premium seat, not the budget one.** It is 33% dearer at the one tier they share and it\ntops out lower. Route there when a fast turnaround on a text-to-video or start-frame shot is worth\npaying for; route to base H3 for anything needing resolution, references, or the same tier cheaper.\n\n**Turbo is the budget seat.** Same grammar and Max's ladder at half Max's rate, with no reference\nendpoint: attach a reference and Slates refuses the call rather than dropping it. Route there for\ndrafts, volume and start-frame coverage, then re-run the keeper on Max or base H3 when it needs\nreferences.\n\n**1080p on Max and Turbo is a refinement, not a native render.** fal's schema, verbatim: *\"1080P\nlatent refinement from a native 768P source.\"* It is a different stage from base H3's 2K/4K\nupscaler, and it costs double the 768p second. Judge a 1080p take against the same shot at 768p\nbefore paying for it across a batch.\n\n**The speed is measured, not claimed** (2026-08-27, same prompt and params on both rows): a 5-second\n768p text-to-video finished in **4.8 seconds** on Max against **57 seconds** on base H3 — roughly\n**12x**, queue to finished file. fal advertises \"under 3 seconds\"; the literal claim did not hold at\n4.8s wall-clock, but the order of magnitude did. For iteration loops and client-present work that gap\nis the entire reason the seat exists.\n\n🚨 **Max's known weakness: colour banding in low light (Eric, 2026-09-09).** Certain shots —\nespecially dark or low-key ones — come back with low-bitrate-looking banding across gradients (skies,\nwalls, shadow falloff). It is the one place the seat visibly gives something up. If a shot is dark\nand gradient-heavy, either light it up in the prompt or route to base H3 at 768p; do not fix it by\nreaching for 2K, which adds its own artifacting on top.\n\n---\n\n## The one thing that makes H3 different: audio is a THREE-LAYER instruction\n\nEvery other video seat treats sound as on or off. H3 splits it, and the split is enforced by where\nyou write each thing. Get the section wrong and the sound is dropped, doubled, or attributed to the\nwrong source.\n\n| Layer | What belongs in it | Where it goes |\n|---|---|---|\n| **Synchronised events** | dialogue, singing, and any sound tied to a specific shot or action | the **body** of the prompt, on the beat it lands |\n| **Scene sound** | ambience and physical sounds that run across the whole clip — room tone, rain, traffic, a ventilation hum | the **soundscape** section |\n| **Score** | music the characters cannot hear; audience-only | the **music** section |\n\n**Three rules, all from MiniMax's own guide:**\n\n1. **Dialogue and singing NEVER go in the soundscape section.** They are synchronised events; they\n belong in the body, at the moment they happen.\n2. **Diegetic music — music the characters can hear** (a radio in the scene, a busker) — also\n belongs in the **body**, not in the score section. The score section is audience-only.\n3. **Write the score in instrumental terms, not mood words.** Name the instruments, the tempo, and\n how it develops. *\"A restrained solo-piano score at a slow tempo, sustained low cello underneath,\n no swell\"* — not *\"emotional music\"*.\n\nUse **N/A** for a section only when silence or absence is genuinely what the shot wants. An empty\nscore section is a real choice; a vague one is a wasted layer.\n\n### The shape, in the one prompt field\n\nSlates sends one prompt string, so write the three layers as labelled paragraphs in this order:\n\n```\n[Shot 1] Live-action, cinematic. A medium-wide shot frames a baker opening the shutters of a\nsmall street bakery before sunrise. The camera pushes in with small amplitude at slow speed as\nthe middle-aged baker with a calm, slightly raspy voice places a fresh loaf on the counter and\nsays: \"First batch of the morning.\" [Shot 2] At 00:05.000, the camera cuts to a close-up of\nsteam rising from the sliced bread while his final words carry over from the previous shot.\n\nSoundscape: wooden shutters scrape open over a quiet street, trays clink softly inside, a\ndoorbell rings once, then light footsteps and the crisp sound of bread being sliced.\n\nScore: a soft acoustic-guitar pattern at a moderate tempo, joined by sparse upright-bass notes,\ngentle fade at the end.\n```\n\n**Body target: 350–500 words** for a reference-carrying shot. Dialogue-heavy content prioritises\nfitting the complete spoken timeline over hitting a word count.\n\n🚨 **Slates disables the provider's prompt expander.** H3's API can rewrite your prompt before\ngeneration; Slates turns that off, because a model rewriting the user's words invisibly is banned\noutright (prompt transparency: what the composer shows is what the model gets). The practical\nconsequence is on you: **nothing will pad a thin prompt.** Write the whole body.\n\n---\n\n## Shots and timing\n\nThe first shot carries **no timestamp**. Every later shot opens with the bracket and a cut time\nthat increases and stays inside the clip length:\n\n```\n[Shot 2] At 00:03.500, the camera cuts to ...\n```\n\nTransition verbs the model knows: **cuts to · transitions to · changes to · switches to**.\n\n**Dialogue that continues across a cut** needs the continuity said out loud — *\"his final words\ncarry over from the previous shot\"* — or the line restarts. **Speech that ends abruptly** should be\ndescribed as cut off rather than trailed off.\n\n---\n\n## Camera — write the move into the sentence\n\nThe model has a named motion vocabulary:\n\n> Zoom In / Zoom Out · Push In / Pull Out · Pan Left / Pan Right · Truck Left / Truck Right ·\n> Tilt Up / Tilt Down · Pedestal Up / Pedestal Down · Arc Shot · Tracking Shot · Static Shot ·\n> Shake Slightly / Shake Strongly · POV · Roll Clockwise / Roll Counterclockwise\n\nModify with **amplitude** (`with small amplitude` / `with large amplitude`) and **speed**\n(`at slow speed` / `at fast speed`).\n\n🚨 **Integrate the motion into the sentence — never stack labels.** MiniMax's own example:\n*\"The camera pushes in with small amplitude at slow speed toward the folded letter in her hands.\"*\nNot *\"Push In. Small amplitude. Slow.\"*\n\n---\n\n## Speakers and dialogue\n\nGive each speaking character a stable identity in the prose and keep it: describe the voice once\n(*\"a young woman with a quiet, breathy voice\"*), then refer back to the same description at every\nline. Identification, delivery and action sit **outside** the quoted line; the line itself is only\nthe words.\n\n```\nThe young woman with a quiet, breathy voice says: \"I get off at the next station.\"\n```\n\n**Voiceover** needs two things — the phrase *\"says in an off-screen voiceover\"* **and** an explicit\nstatement that the lips stay closed. Without the second half the model animates a mouth.\n\n```\nThe man says in an off-screen voiceover: \"I still remember that road.\" — his lips remain\ncompletely closed.\n```\n\n**On-screen text** — signs, banners, labels, subtitles, neon — goes in double quotes with the\noriginal wording preserved exactly: *A red neon sign reading \"Open Late\" glows above the doorway.*\n\n---\n\n## References — H3's real differentiator is the declared RELATIONSHIP\n\n*(`minimax-h3` and `minimax-h3-max`. Max gained the reference set on 2026-09-09; its free allowance\nis FOUR images rather than the base row's five. `minimax-h3-max-turbo` takes no references.)*\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n<!-- @inject:reference-rules-core -->\nIdentity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.\n<!-- @end:reference-rules-core -->\n\n### Cite references by number — Slates already does it for you\n\nH3 on fal takes references as **typed slots** and expects the prompt to name them by modality and\norder: **`image 1`, `image 2`, `video 1`, `audio 1`**. That is exactly what the Slates composer\nemits from your `@mentions` and `#tags` (`Marcus (image 1) in the workshop (image 2)`), in the\nexact order it sends them.\n\n🚨 **Do NOT hand-write angle-bracket reference tags.** MiniMax's own model-card grammar uses\n`<Subject N>` / `<Picture N>` / `<Video N>` / `<Audio N>` labels; the fal endpoints Slates calls do\nnot — they build the binding from the typed slots and ask for plain numbered prose. Typing the tags\nyourself puts literal angle brackets in the prompt the model reads.\n\n### State how much of each reference survives\n\nThis is the lever no other model in the catalogue gives you. Say, in plain words, what each\nreference is FOR and how much of it should carry through:\n\n| Intent | Say something like |\n|---|---|\n| Keep it whole | *\"Keep the woman in image 1 exactly as she appears — hair, cardigan, necklace.\"* |\n| Keep part of it | *\"Use the café in image 2 for the brick wall and the sofa; the lighting is late evening, not daylight.\"* |\n| **Move a trait onto someone else** | *\"Give the man in image 3 the weathered leather texture of the jacket in image 4.\"* |\n| Loose echo | *\"Match the general palette and grain of image 5; nothing else from it.\"* |\n\nThe third row is the one with no equivalent anywhere else in Slates: **transferring a characteristic\nonto a different subject** is a first-class thing H3 understands. Reach for H3 when that is the job.\n\n**Audio references** bind a voice or a texture without copying the words. Say which speaker an\naudio reference is for (*\"the woman in image 1 speaks in the voice timbre of audio 1\"*), and when\nyou are referencing only the timbre, **do not carry the reference clip's original dialogue into your\nprompt** — write the new line. When you genuinely want the same words re-performed, quote them\nexactly and say so.\n\n**An audio reference cannot travel alone** — H3 refuses a reference set that is audio only. Pair it\nwith at least one image or video reference.\n\n### 💸 Reference images past the free allowance are billed — and the two reference rows differ\n\nOn `minimax-h3` the first **5** are free and each additional image adds **4 credits**.\nMax pools image pixels, reference-video seconds and reference-audio seconds into one token\nallowance. Include `referenceImages`, `videoRefSeconds` and `audioRefSeconds` when estimating;\ncharacter voices count as audio. The generation preflight resolves the actual attached media.\n\nFour extra images on a 10s\n768p clip add 16 credits to a 30-credit generation: **more than half again**, for references that\noften make the output worse rather than better (see the 2–4 rule above).\n\nAttach the references the shot needs, not the ceiling. Call\n`slates_estimate_generation_cost` with `referenceImages` set to the real count before a\nreference-heavy job — a quote that omits it under-reports the bill.\n\n---\n\n## Frames\n\nAll three rows take a **start frame**, an **end frame**, or both. With an\nend frame, land it explicitly: describe the final pose, spacing and composition as the thing the\nshot **settles into** at the end, rather than hoping the model finds it.\n\n> *\"…she rotates the handle into the final angle and settles into the pose, spacing and composition\n> of image 2 at the end of the shot.\"*\n\n**Frames and references are mutually exclusive** on the two reference rows — they are different endpoints, and\nthe reference endpoint has no frame slots at all. Slates refuses the combination rather than\ndropping one side.\n\n---\n\n## Cost discipline\n\n| Combination | Credits |\n|---|---:|\n| `minimax-h3` · 768p · 5s | 15 |\n| `minimax-h3` · 768p · 10s | 30 |\n| `minimax-h3` · 2K · 10s | 65 |\n| `minimax-h3` · 4K · 10s | 80 |\n| `minimax-h3-max` · 768p · 10s | 40 |\n| `minimax-h3-max` · 1080p · 10s | 80 |\n| `minimax-h3-max-turbo` · 768p · 10s | 20 |\n| `minimax-h3-max-turbo` · 1080p · 10s | 40 |\n| `minimax-h3` — every reference image past the **fifth** | **+4** |\n\n**768p is the default for a reason.** It is the tier the model natively generates.\n\n🚨 **2K and 4K are UPSCALES of a 768p render, not larger generations.** fal's own schema says so:\n*\"480P and 768P are native generation modes; 2K and 4K upscale a 768P base result.\"* The upscaler\n(H3-Regenerate-2K) is a separate stage bolted onto a finished take — it can enlarge detail but it\ncannot add information.\n\n**In our own test (2026-08-27, same prompt, same seed) the 2K pass came back with MORE artifacting\nthan the 768p original it was built from**, while costing 33 credits for a 5-second take against 15,\nand taking nearly twice as long to return.\n\n🚨 **Confirmed independently (Eric, 2026-09-09): 2K and 4K carry visible AI noise artifacting and\n\"just look bad\".** That is now TWO separate observations, months apart, pointing the same way — it is\nno longer a single-shot warning. The tiers stay available because a delivery spec sometimes demands\nthe pixels, but **do not route to 2K/4K for quality**: you are paying more, waiting longer, and\nadding artifacts to a 768p render. Upscale in post from a clean 768p master instead.\n\n**So: generate at 768p and judge it at 768p.** Reach for 2K or 4K only when a delivery spec demands\nthe pixels, and expect to be paying for size rather than quality — a post-production upscale from a\nclean 768p master is very often the better result. **4K video is Pro-only** (the server returns\n`PRO_REQUIRED` for a base account); 2K is open to every tier.\n\n---\n\n## Quick checklist\n\n- Body written as a timeline, first shot untimestamped, later shots on `[Shot N] At MM:SS.mmm`.\n- Camera motion written **into** a sentence with amplitude and speed.\n- Dialogue and diegetic music in the body; ambience in the soundscape section; audience-only score\n in the score section, described by instrument and tempo.\n- Voiceover carries both the off-screen phrase and the closed-lips statement.\n- References cited as `image 1` / `video 1` / `audio 1`, each with a stated job and a stated degree\n of retention. No angle-bracket tags.\n- Reference count is deliberate — you are paying 4 credits for each one past the fifth.\n- Frames **or** references, never both.\n- The prompt is the prompt: no expander will fill it out for you.\n",
|
|
26
|
-
"slates-prompting-motion-transfer": "---\nname: slates-prompting-motion-transfer\ndescription: How to set up motion transfer — Kling Motion Control only (std and pro tiers, 5-second outputs). Read before calling slates_generate_motion_transfer. Reference image (character) + driving video (motion source) → new video of the character performing the motion. Asset selection rules, character_orientation, tiers, and prompt usage. Also covers the Seedance alternative, which is a normal video generation rather than a mode of this tool.\n---\n\n# Motion transfer — setup guide\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Motion transfer (Kling Motion Control only).** A target IMAGE (your character) plus a source VIDEO (the motion) produces your character performing that motion. Always 5 seconds.\n\n**The five levers**\n1. **The target image must show body proportions clearly** and the character must occupy more than about 5% of the frame. A tiny figure in a wide shot has nothing to drive.\n2. **Single character in the target.** A group image breaks the identity anchor.\n3. **Choose `characterOrientation` on purpose** — `video` takes the source clip's framing, `image` preserves the portrait's. It is the most-missed choice here.\n\n4. **The prompt is atmosphere only** — `Soft afternoon sunlight, dust motes in the air, vintage warm color grade.` Motion verbs are ignored; the motion is already in the driving video.\n5. **Pick the best 5 seconds of the source up front**, and write only atmosphere: `soft afternoon sunlight`, `vintage warm color grade`, `clean studio backdrop`. The output is 5s regardless, so a long driving clip just wastes the choice.\n\n**Examples**\n- `Soft afternoon sunlight, dust motes in the air, vintage warm color grade.`\n- `Clean studio backdrop, sharp focus on the character.` (Or leave it empty.)\n\n**Hard constraint:** cartoon driving videos fail, and a cropped or partial target character drifts. std is fine while the motion-and-framing combination is still moving; switch to pro once it is locked. For a REGENERATED shot instead — physical contact, cloth and hair, camera motion, native audio — that is a normal Seedance video generation with the driving clip as a video reference, not a mode of this tool.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — the motion is already in the driving video, so motion verbs are ignored:\n- `spins faster`, `jumps higher`, `add more energy`, `moves quicker`\n<!-- @banned:end -->\n\nTake a still **target image** (your character) and a **source video** (the motion you want), produce a new video of your character performing the source video's motion. **This tool is Kling-only** — it wraps Kling Motion Control and nothing else.\n\n| Tier | Cost | Use case |\n|------|-----------|----------|\n| Kling std (`kling-mc-std-5s`) | ~32 credits / 5s | General motion transfer, budget lane |\n| Kling pro (`kling-mc-pro-5s`) | ~42 credits / 5s | Cleaner anatomy, better identity preservation |\n\nBoth tiers trip the confirm gate. User OK required every time. (Prices are approximate — `slates_estimate_generation_cost` returns the exact credit total.)\n\n## Want Seedance instead? That is a video generation, not a mode here\n\nKling MC retargets a skeleton onto a finished image; Seedance *generates* the shot with the motion as a conditioning input — the difference shows on fast choreography, physical contact, cloth/hair, and camera motion, and the output carries native audio. **It is not an engine switch on this tool.** Run a normal `slates_generate_video` on `seedance-2` with the driving clip attached as a video reference and the character image as an ingredient, then write the prompt yourself:\n\n```\nThe character from image 1 performs the exact motion, choreography, and camera\nmovement from video 1. Preserve the character's identity, appearance, and outfit.\n```\n\nThat is the same endpoint the old `motionModel=seedance-2` branch called — it just wrote that sentence for you, invisibly. Add style/setting/camera direction freely; Seedance re-generates the whole shot.\n\n- **Driving clip must be 2–15s** (all providers cap reference video at 15s). Longer clips: trim first, or use Kling MC (`characterOrientation: 'video'` takes up to 30s).\n- **Billing = combined input+output seconds** (the vref keys). The server probes the clip and corrects an understated key — quote via the confirm gate before spending. On Seedance 2.5's AI-face route (EvoLink) the input side counts as at least the output's length: max(input, output) + output.\n- **Faces route through the face cascade**: `seedanceFace` for a character, `[REAL_FACE_DETECTED]` → confirm consent → `seedanceRealFace=true, realFaceConsent=true` (premium realface vref pricing).\n- `characterOrientation` has no Seedance equivalent; framing follows the prompt + `aspectRatio`.\n\nEverything below is about the Kling tool.\n\n## Inputs\n\n- `sourceVideoAssetId` — driving video. **Must be a realistic human** with clear proportions. Anime/cartoon/CG driving videos fail.\n- `targetImageAssetId` — character to be animated. Can be any style (cartoon, anime, realistic, painted).\n- Both must already exist as assets in the project. Use `slates_list_assets` to find them or upload first.\n\n## Source video constraints\n\n- Realistic human (not animated, not CG)\n- Entire body OR upper body visible — head must not be obstructed\n- Subject occupies a clear share of the frame\n- Single primary subject. Multi-person driving videos confuse the motion anchor.\n- Clean motion — choppy / cut-edited driving videos produce jittery output\n\nGood driving video sources:\n- Reference dance footage with one subject\n- Walking / gesture / posing clips\n- Talking-head footage when paired with character_orientation: 'video'\n\nBad driving video sources:\n- Music videos with multi-shot edits\n- Anime / animation clips\n- Heavily stylized footage with smoke / particles obscuring the body\n- Footage where the subject's head leaves frame mid-clip\n\n## Target image constraints\n\n- Character body proportions clearly visible\n- Character occupies >5% of image area (not a tiny figure in a wide shot)\n- Single character. Group images break the identity anchor.\n- Any artistic style works — cartoon, anime, painted, realistic, 3D render\n\nAvoid:\n- Extreme close-up of just the face (no body to drive)\n- Character partially cropped at the waist when the driving video is full-body\n- Multiple characters\n\n## character_orientation — the most-missed choice\n\nThis single parameter changes the output dramatically. Pick deliberately.\n\n| Value | Output framing | Max source duration | Best for |\n|-------|----------------|---------------------|----------|\n| `video` | Matches driving video framing | Up to 30s source | Complex full-body motion (dance, action, athletics) |\n| `image` | Matches target image framing | Up to 10s source | Camera moves, simpler motion, preserving original composition |\n\n**Default `video`** when the driving video has the look you want (most cases).\n\nSwitch to `image` when the target image's composition is the brand asset and the motion is secondary (e.g., a hero shot of a character that needs subtle gesture, not a full performance).\n\n## Tier choice — std vs pro\n\n**std (~32 credits)** for:\n- Drafts, motion exploration, blocking\n- Group scenes where the character isn't a hero shot\n- When the budget is tight and the motion is the focus\n\n**pro (~42 credits)** for:\n- Final hero takes\n- Branded characters where identity drift = unacceptable\n- Anatomically complex motion (limbs crossing, fast direction changes)\n- Anime / cartoon target images — pro handles non-realistic styles better\n\nDon't default to pro. The ~10-credit delta compounds fast across iteration.\n\n## Prompt usage (optional)\n\nThe `prompt` field is **scene/style refinement**, not motion direction. The motion comes from the driving video — the prompt sets ambiance, lighting, additional detail.\n\nGood:\n- `Soft afternoon sunlight, dust motes in the air, vintage warm color grade.`\n- `Clean studio backdrop, sharp focus on the character.`\n\nBad (model ignores motion verbs — they're already in the driving video):\n- ❌ `She spins faster and jumps higher.`\n- ❌ `Add more energy to the dance.`\n\nLeave it empty if you don't have a specific atmospheric note.\n\n## Common failure modes\n\n| Symptom | Likely cause | Fix |\n|---------|--------------|-----|\n| Limbs distort / extra fingers | std tier, complex motion | Switch to pro |\n| Character identity drifts | Target image cropped too tight | Use a fuller-body target |\n| Output looks \"stuck\" / minimal motion | Driving video subject too small in frame | Pick a driving video where the subject fills more of the frame |\n| Cartoon target turns realistic | std tier on stylized art | Switch to pro — handles non-realistic styles better |\n| Garbled output entirely | Anime / CG driving video | Use realistic human driving footage |\n| Wrong framing on output | character_orientation set wrong | Try the other value |\n| Background bleeds through character | Target image had complex background | Use a target with cleaner background separation |\n\n## Workflow patterns\n\n**Reference dance to brand character:**\n1. Generate or upload the brand character as a still image (clean background, full body, single subject)\n2. Find driving footage — a clean reference video of the dance you want\n3. Upload both as project assets\n4. Run motion transfer with `motionModel: 'kling-mc-pro'`, `characterOrientation: 'video'`\n5. Total cost: ~42 credits per 5s take\n\n**Subtle motion on a hero portrait:**\n1. Use the locked hero portrait as the target image\n2. Pick a driving video with subtle gesture (head turn, slight posture shift)\n3. `characterOrientation: 'image'` to preserve the portrait's framing\n4. std tier is fine for this case — motion isn't dramatic\n\n**Avoid:**\n- Pro tier on first iteration — waste, switch to it once the motion + framing combo is locked\n- Cartoon driving videos — guaranteed failure\n- Cropped or partial target characters — identity will drift\n- Long driving videos when output is 5s — pick the best 5s of the source upfront\n\n## Cost discipline\n\n- 5 seconds, no shorter option\n- Both tiers trip the confirm gate — every call needs explicit user OK\n- Iteration is expensive: 4 takes at pro ≈ 168 credits. Lock framing + driving video before tier-up to pro.\n- Always run a single std take first to validate the motion + framing combo before committing to pro\n\n## Confirm gate: cost + codes, no inline preview\n\nMotion transfer is mechanical — the model deterministically applies source motion to target image. Both tiers trip the confirm gate; the response includes the asset codes for source and target so you can announce them in chat.\n\n- ✅ \"Transferring motion from **VID-V3** onto **IMG-A12 — Detective Closeup**. ~42 credits, confirm?\"\n- ❌ \"Using the walk video and the detective image...\" (multiple of each in the project.)\n\nDon't second-guess the assets the user picked — the model executes the transfer. If the output is wrong, iterate on motion source or target choice, not on a refinement prompt.\n\n## Sources\n\n- [fal.ai — Kling Motion Control V3 Standard](https://fal.ai/models/fal-ai/kling-video/v3/standard/motion-control)\n- [fal.ai — Kling Motion Control V3 Pro](https://fal.ai/models/fal-ai/kling-video/v3/pro/motion-control)\n",
|
|
27
|
-
"slates-prompting-nano-banana-2": "---\nname: slates-prompting-nano-banana-2\ndescription: How to write prompts that produce cinematic, photorealistic results from Nano Banana 2 (Google Gemini 3.1 Flash Image, accessed via fal-ai/nano-banana-2). Read this before calling slates_generate_image when the user wants film-quality, real-world, or cinematic output. Skip for stylized / illustrated / cartoon work — the rules differ.\n---\n\n# Nano Banana 2 — cinematic & photorealistic prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Nano Banana 2 (Gemini 3.1 Flash Image).** Brief it like a creative director, not a tag list. Structure: `Film still from [director] [genre]. Shot on [camera] with [lens]. [Subject and action]. [3-5 specific visual details]. [Lighting — direction + quality]. [Color palette]. [Film stock]. [1-2 word tone].`\n\n**The five levers**\n1. **Named lens + aperture** beats \"shallow depth of field\" — `85mm f/1.4`, `135mm f/2.8`, `Panavision anamorphic`, `400mm telephoto`.\n2. **Light by direction and quality**, never \"good lighting\" — `hard sidelight from a single window, deep falloff`, `overcast north light`, `practical tungsten spill`.\n3. **A named film stock or sensor** carries a whole palette — `Kodak Portra 400`, `Cinestill 800T`, `ARRI Alexa 65`.\n4. **Composition as a shot** — `low angle`, `aerial view`, `rule of thirds with the subject camera-left`, `foreground occlusion`.\n5. **Positive framing only.** Describe what is there. \"Empty street\", never \"no cars\"; \"unstaged documentary photography\", never \"not anime\".\n\n<!-- @inject:cinematic-card -->\n**For a photographic look, use only what this frame needs.** Image models default to clean, evenly lit and fully exposed. Describe what the camera sees, not just gear or mood:\n- **Inspect every reference first.** Write its grade and imperfections in words: darkness, contrast, muddy or true blacks, colour, softness/noise, subject separation. Never grade cleaner or brighter than the look reference unless asked.\n- **One light system** — `low sun behind her`, `her face falls into deep shadow`, `no light in front of her`.\n- **Visible exposure** — `the sky burns out to white`, `dense, slightly crushed shadows`.\n- **Lens name plus effect** — `200mm telephoto`, `peaks loom huge behind her and melt into soft shapes`.\n- **Name every garment and close the foreground.** Omissions invite reference leakage or invented props.\nBind references inline. A scene reference owns the grade; for a look-only reference, write the new scene's light. References are optional. For owned-frame edits, describe only the change and what stays.\n<!-- slates-only -->Use `slates-cinematic-look` with a technique ID or section query for more.<!-- /slates-only -->\n<!-- @end:cinematic-card -->\n\n**Hard constraint:** there is no `negativePrompt` field. Suppress by reframing positively, or inline `without` / `free of`. Knowledge cutoff January 2025 — anything later needs reference images.\n<!-- @card:end -->\n\nNano Banana 2 is **Gemini 3.1 Flash Image**.<!-- slates-only --> It is the headless image model used when `projectId` is omitted — the op also exposes `flux-2-max` and `seedream-5-lite`, each with its own prompting skill.<!-- /slates-only --> It is **not** Gemini 3 Pro Image; that is Nano Banana **Pro** (`nano-banana-pro`), a separate model with its own seat.<!-- slates-only --> Verified against the runtime slug map in `slate/src/main/api/google.ts`.<!-- /slates-only --> NB2 is a language model that outputs pixels — brief it like a creative director, not like a Stable-Diffusion tag-soup tool. The single biggest lever for realism: **specificity that mimics how real photographers and cinematographers describe their work**.\n\nKnowledge cutoff: January 2025. Anything after needs explicit reference images.\n\n## Google's 4 official rules (verbatim)\n\n1. **Be specific.** Provide concrete details on subject, lighting, and composition.\n2. **Use positive framing.** Describe what you want, not what you don't want.\n3. **Control the camera.** Use photographic and cinematic terms like \"low angle\" and \"aerial view.\"\n4. **Iterate.** Refine images with follow-up prompts in a conversational manner.\n\n## Official prompt formula\n\n```\n[Subject] + [Action] + [Location/context] + [Composition] + [Style]\n```\n\nFor the cinematic / photoreal use case, expand to:\n\n```\nFilm still from [DIRECTOR] [GENRE]. Shot on [CAMERA] with [LENS]. [SUBJECT and action]. [3-5 specific visual details]. [LIGHTING — direction + quality]. [COLOR PALETTE]. [FILM STOCK or sensor language]. [1-2 word emotional tone].\n```\n\n## Photorealism positives — what consistently works\n\n⚠️ **This vocabulary is correct here and does not carry into a video prompt.** The leak happens one way: you write an NB2 start frame, then carry its look description straight into the prompt that animates it.\n\n<!-- @inject:lens-video-split -->\nNamed lenses, apertures, film stocks and camera bodies (`85mm f/1.4`, `Kodak Portra 400`, `ARRI Alexa 65`) are an image-model lever. On a video model, translate the look instead of pasting the gear list: `85mm f/1.4, Portra 400` becomes `close-up, shallow depth of field, warm natural colors, cinematic texture, film-grain texture`. ByteDance's Seedance 2.0 guide never mentions fps, shutter angle, f-stop or lens millimetres. Its Seedance 2.5 guide does, once: the visual-style line of its own storyboard example names one camera body and one 35 mm cinema lens. On 2.5 a single line like that is vendor-sanctioned; a stacked gear list still is not.\n<!-- @end:lens-video-split -->\n\n**Named lenses + apertures** beat generic \"shallow depth of field\":\n- `85mm f/1.4`, `135mm f/2.8`, `50mm f/1.2`, `35mm f/2`\n- `Panavision anamorphic` for horizontal flares + cinematic width\n- `400mm telephoto` for compression + isolation\n- `24mm` for environmental interiors\n\n**Named cameras / sensors:**\n- `ARRI Alexa 65`, `Hasselblad X2D`, `Canon EOS R5`, `Sony A7III`, `Fujifilm X-T5`\n- \"Specific gear\" beats \"DSLR\"\n\n**Named film stocks** (one per prompt — never mix):\n- `Kodak Portra 400` — natural skin, warm\n- `Fuji Velvia 50` — saturated, landscape\n- `Ilford HP5 Plus` — black and white, gritty grain\n- `CineStill 800T` — tungsten night, halation\n\n**Physics-based lighting** (direction + quality):\n- `Single key light at 45 degrees from upper left`\n- `Late afternoon sun at 15 degrees above horizon`\n- `Color temperature 4500K` beats `slightly warm`\n- `Practicals only — no fill` for Deakins-style realism\n\n**Imperfection vocabulary** (forces away from AI-clean):\n- `visible pores`, `natural skin grain`, `peach fuzz`, `slight hyperpigmentation`\n- `unretouched raw photography`, `ISO noise`, `sweat beading`\n- `crisp catchlights in the eyes`, `skin micro-detail`\n- Lead with the kind of photograph and the conditions on the skin (sun, wind, sweat), then add one or two of these. A bare list of flaw words read as tokens and produced plastic skin on GPT Image 2 (2026-08-24).<!-- slates-only --> Technique: `slates-cinematic-look` → `name-the-capture-context`.<!-- /slates-only -->\n\n**Director references** (use when locking style):\n| Director | Tone | Visual signature |\n|---|---|---|\n| Denis Villeneuve | Cold, vast, existential | Desaturated, overwhelming scale |\n| Roger Deakins | Precise motivated light | Single source, deep shadows, practicals |\n| Emmanuel Lubezki | Natural, spiritual | Available light, golden hour |\n| Bradford Young | Warm darkness | Underexposed, rich shadows, skin tones |\n\n**Genre cues that move the model:**\n- `unstaged documentary photography style`\n- `fashion magazine editorial, shot on medium-format analog film, pronounced grain`\n- `Film still from [Director] [genre]`\n\n## The anti-list — phrases that DEGRADE realism\n\nThese are Stable-Diffusion-era tag soup. The model treats them as low-signal noise. Measured success rate: ~60-70% with these vs ~95%+ with positive description.\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is extracted\n by src/prompts/banned-tokens.ts, inlined verbatim into the slates_generate_image\n op description (always in context on both surfaces), and matched against every\n submitted prompt. Editing this list changes what the agent is told AND what it\n is warned about — keep every entry backticked, and keep prose outside the\n backticks. -->\n<!-- /slates-only -->\n**Never use:**\n- `8k`, `4k` (as a quality token)\n- `hyperrealistic`, `ultra-realistic`, `photorealistic` standing alone\n- `masterpiece`, `best quality`, `highly detailed`, `ultra-detailed`\n- `trending on ArtStation`, `award-winning`\n- `perfect skin`, `flawless`, `airbrushed`, `smooth skin`\n- `cinematic` standing alone — always specify *which cinema* (director, lens, era, stock)\n- `not anime, not cartoon, not 3D` — negation tag soup, replace with a positive style cue\n<!-- @banned:end -->\n\n**Examples**\n- `Film still from a Denis Villeneuve thriller. Shot on ARRI Alexa 65, 85mm f/1.4. A woman in a charcoal wool coat stands at a rain-slick bus stop, breath visible. Hard sodium light from a single overhead lamp, deep falloff into blue night. Kodak Vision3 500T. Isolated.`\n\n## Negative prompting — there is no field\n\nNano Banana 2 has **no `negativePrompt` parameter**. Three patterns to suppress unwanted content:\n\n1. **Positive reframing (preferred):** \"empty street\" not \"no cars\". \"Unstaged documentary photography\" not \"not anime.\"\n2. **Inline `without` / `free of`:** \"without any people, vehicles, or man-made structures\", \"free of text overlays, logos, or watermarks.\"\n3. **Constraint clauses for anatomy/quality:** \"accurate anatomy with five fingers per hand, symmetrical features, natural proportions\"; \"sharp, well-exposed, free of blur or JPEG artifacts.\"\n\nDefault to #1. Reach for #2 only when positive framing can't suppress the unwanted element.\n\n## Reference images\n\n- **Hard limit: 14 images** (10 object-fidelity + 4 character-consistency). Categories don't trade — you can't use 14 object slots even if no characters are referenced.\n- **Name each reference inline — Slates does this for you.** When you `@mention` a subject/environment or `#mention` a style<!-- slates-only --> (or pass `referenceAssetIds`)<!-- /slates-only -->, Slates composes the prompt so each reference is named inline as \"image N\" — e.g. `Marcus (image 1) sits across from the woman (image 2) in the cafe (image 3)`, or `lit and graded like image 4` where you placed the style mention. An unmentioned style attachment gets a short fallback clause The model does NOT infer a reference's role from its position; the NAME carries it. NB2's own consistency lever is literally **\"assign a distinct name to each character/object\"**. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render the scene's expression\") — that drags the sheet's wardrobe + studio lighting into the scene. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n\n### Reference rules (the verified ones)\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n<!-- @inject:reference-rules-core -->\nIdentity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.\n<!-- @end:reference-rules-core -->\n\n### For Nano Banana 2 specifically\n\n- **NB2's own consistency lever is \"assign a distinct name to each character/object.\"** That is Google's phrasing for rule 3 — cite each canonical identity inline by name.\n- **Rule 8 is a job you do, not one you delegate.** NB2 *is* the start-frame model — when a downstream video shot needs legible text, render it here and animate from this frame.\n- **Character consistency is officially \"not 100% perfect\"** per Google. Test before bulk generations. High-resolution, front-facing reference images help most.\n- **Injection is stochastic — budget 3-5 re-rolls per shot; re-roll, don't re-engineer.** First rolls miss faces/hands; the same prompt lands a clean one within a few tries.\n\n## Common failure modes + fixes\n\n**Hands:** Append `accurate anatomy with five fingers per hand, symmetrical features, natural proportions, relaxed open palm`. Avoid heavy jewelry, props intersecting fingers, motion blur in references.\n\n**Text in images:** Quote-wrap target text. Specify font (`Century Gothic, 12pt`). Long phrases work; small text degrades. Two-step works best — generate text concepts conversationally first, then ask for the image.\n\n**Left/right confusion:** Default is **viewer's perspective**, not subject's. Append `left and right are from the character's perspective, NOT the camera's` when scene-blocking matters.\n\n**Surreal / absurd prompts trip uncanny valley:** The model drags toward realism. If you want surrealism, lean hard into stylization keywords (`painted`, `illustrated`, `stop-motion`).\n\n**Soft faces / dead eyes:** Add `crisp catchlights in the eyes`, `skin micro-detail`, `peach fuzz visible`. Don't stack quality enhancers — single clean prompt beats multiple re-interpretations.\n\n**Post-cutoff content (anything after Jan 2025):** Use reference images. The model has no knowledge of recent franchises, products, events.\n\n## Resolution tactics\n\n- Resolution is priced: NB2 4k costs roughly 2x 1k. Prices change — check current numbers<!-- slates-only --> by calling `slates_estimate_generation_cost`<!-- /slates-only -->. Pick the cheapest resolution that serves the use case.\n- **At 2K and above, the model allocates more tokens to surface detail** — explicit texture vocabulary (pores, fabric weave, grain) compounds at higher resolution.\n- 1k for fast iteration / drafts; 2k for hero shots; 4k only when you need print-grade detail.\n- 2K generations vary 20-60s+. Don't time-budget tightly.\n\n## Boring vs cinema — examples\n\n❌ **Boring:** \"Wide shot of a man on a dock looking at the forest.\"\n\n✅ **Cinema:** \"Direct overhead drone shot on weathered dock surface. Single figure standing center frame, climbing up from frame bottom. Boot prints leading away from him toward shore. Pale winter light. Anamorphic lens flare from low sun. Desaturated blue and slate grey palette. Kodak Portra 400 grain. The path already walked by someone else. Map of threat.\"\n\n❌ **Boring:** \"Close up of a woman looking scared.\"\n\n✅ **Cinema:** \"Extreme close on subject's mouth and nose, 135mm f/2.8, shallow depth of field. Breath pluming out, catching cold light from upper-left key. Lips slightly parted, peach fuzz visible. The breath holds. CineStill 800T halation around catchlights. Waiting.\"\n\n## The 3-strike rule\n\nIf three iterations on the same prompt haven't produced what the user wants, stop. Hand back to the user with what you tried and what isn't working. The slot machine doesn't converge — the prompt structure is wrong, not the seed.\n\n## Family variants — Lite and Pro\n\nEverything in this skill applies to the whole Nano Banana family; two variants trade speed/ceiling around NB2 full:\n\n- **nano-banana-2-lite** — ~half the price, ~2.7× faster, **1K output only**, max 4 refs. The draft/iteration seat: explore compositions here, then re-run the winner on NB2 full at 2K/4K. Same Gemini filter.\n- **nano-banana-pro** — the hero-frame/typography ceiling (~2× NB2, 4K native). NB2 ≈ 95% of Pro; escalate only when spatial composition, cinematic lighting/skin, fine typography-in-scene, or deep multi-element frames must be perfect. Up to 14 refs — it takes a full subject library in one call.\n\n<!-- slates-only -->\nRouting between them (and vs GPT Image 2.5 / FLUX / Seedream): `slates-model-selection`.\n<!-- /slates-only -->\n",
|
|
28
|
-
"slates-prompting-omni-flash": "---\nname: slates-prompting-omni-flash\ndescription:
|
|
29
|
-
"slates-prompting-seed-audio": "---\nname: slates-prompting-seed-audio\ndescription: How to prompt Seed Audio 1.0 (ByteDance, via fal). Read before calling slates_generate_audio with model seed-audio. The one-pass audio SCENE model — dialogue, SFX and ambience together from ONE plain sentence. CRITICAL - it has NO duration parameter, so length must be named IN THE PROMPT TEXT and Slates bills the duration you request. Covers the one-sentence doctrine, the crowd-size rule, why Kling \"SFX:\" syntax hurts here, and the audio-refs-XOR-image input rule.\n---\n\n# Seed Audio 1.0 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Seed Audio 1.0.** One plain sentence describing a SCENE, and it returns dialogue, effects and ambience together in one pass. For ONE named voice saying ONE line, `inworld-tts-2` is the seat instead; this one renders the whole room.\n\n**The five levers**\n1. **Write one sentence in plain language.** Describe the room and what is happening in it; the model casts and performs it.\n2. **Name the crowd size, the room size and the distance.** This is the highest-leverage single edit on any bed — unqualified nouns default BIG. \"Tiny applause of two or three people at an open mic\", not \"applause\".\n3. **Put the length in the prompt** and make it deliberate.\n4. **Dialogue is performed inside the scene** — write the line as spoken in the room, then re-roll until a take is right and lip-sync against it.\n5. **Describe sounds directly** — \"one coffee machine hissing, cutlery somewhere behind the counter\" — with the object, the action and where it is.\n\n**Examples**\n- `a diner at 2am, one coffee machine hissing, cutlery somewhere behind the counter, one man says quietly \"you're late again\". 15 seconds`\n- `nature soundscape, a wide open field, cicadas near, birds mid-distance, one loon far off across water. 20 seconds`\n\n**Hard constraint:** it has NO duration parameter — the length you request is written into the prompt AND is what you are billed, whatever comes back, so choose it deliberately. Kling's `SFX:` / `Ambient noise:` / `Background music:` labels have no parser here and measurably worsen the output. Shot language, camera moves and lighting are video-prompt words the model must ignore. It is not a music model and it cannot produce pixels.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — Kling video syntax has no parser here and measurably worsens the output:\n- `SFX`, `Ambient noise`, `Background music` as labels — describe the sounds directly\n- shot language: `wide shot`, `slow push in`, `warm tungsten` — camera and lighting words are video-prompt words the model has to ignore\n<!-- @banned:end -->\n\nByteDance's one-pass audio scene model, carried on fal (`bytedance/seed-audio-1.0`). It generates dialogue, sound effects and ambience **together**, from a single plain sentence. 1–120 seconds. It is the default audio model in Slates and the workhorse for continuity beds.\n\n## Where it routes\n\n- **Scene audio, room tone, ambience beds, crowd/nature soundscapes** — anything where several sounds share a space. One generation, not three layered ones.\n- **Dialogue and scratch VO inside a scene.** The line is performed in the room, by a voice the scene casts. When WHO is speaking matters — a character's own voice, a clean narrator — that is `inworld-tts-2` instead. Lock the read by re-rolling until a take is right, then lip-sync against it with `slates_generate_lip_sync`.\n- **NOT** a single effect that must land on a known frame — that is `eleven-sfx`, which takes an exact duration.\n- **NOT** music. Slates has no music model; import a track and drop it on an audio track.\n- **AUDIO-ONLY.** It cannot produce images or video.\n\n## THE RULES\n\n### 1. 🚨 There is no duration parameter — the words set the length\n\nThis is the single most important fact about this model. Output length is driven by the prompt text (\"… 15 seconds\"), capped at 120s.\n\n**Slates handles this for you:** the `durationSeconds` param appends the duration to the prompt and **bills that number of seconds**. So:\n\n- Set `durationSeconds` to what you actually want.\n- **Do not also write a different length into your sentence.** Two numbers fight, and you pay for the one you selected, not the one you got.\n- If the returned clip is shorter than requested you still paid for the request — that is the deal that keeps the displayed price equal to the charge. Ask for what you need.\n\n<!-- slates-only -->\nThe server re-derives the billed key from `durationSeconds` (a client cannot under-bill), probes the returned `audio.duration` after completion, and logs `SEED AUDIO BILLING DRIFT` if the model overshot. No auto-charge, no refund — the request is the contract.\n<!-- /slates-only -->\n\n### 2. One plain sentence. No production jargon.\n\nField-proven (Higgsfield sprint, 2026-07-27/28). Working prompts look like this:\n\n```\ntiny applause of 2 or 3 people at an open mic. 15 seconds\nnature soundscape, wide open field cicadas and birds and a loon.\na diner at 2am, one coffee machine hissing, cutlery somewhere behind the counter\n```\n\nNot this:\n\n```\n✗ AMBIENCE: interior diner, night. SFX: espresso machine (hiss, 2s), cutlery.\n✗ Wide shot of a diner. Slow push in. Warm tungsten. Ambient noise: ...\n```\n\nShot language, camera moves and lighting belong to video prompts. Here they are just words the model has to ignore.\n\n### 3. Never bring Kling's audio syntax to this model\n\n`SFX:` and `Ambient noise:` prefixes and `Background music:` labels are **Kling 3.0 video** syntax. Seed Audio has no parser for them — it reads them as text in the scene and the output gets measurably worse. Describe the sounds directly instead.\n\n### 4. Name the crowd size, the room size, the distance\n\nThe highest-leverage single edit on any bed. Unqualified nouns default big:\n\n| Vague | What it returns | Fixed |\n|---|---|---|\n| `applause` | a full auditorium | `tiny applause of 2 or 3 people` |\n| `traffic` | a highway | `one car passing on a wet residential street` |\n| `crowd` | a stadium | `four people talking at the next table` |\n\nDistance words (`far off`, `muffled through a wall`, `right next to the mic`) work the same way and are how you build depth in one sentence.\n\n### 5. Beds must outlast the cut\n\nAsk for a few seconds more than the clip needs so the edit has handles to fade through. A bed that ends exactly on the cut always sounds clipped. This is a product requirement, not a preference — it is why the duration control exists at all.\n\n### 6. Dialogue goes in quotes, inside the same sentence as the room\n\n```\na tired bartender says, \"we closed twenty minutes ago\", glasses clinking behind him\n```\n\nPick a preset voice when a specific speaker matters. Leave `voice` unset and the scene casts itself — which is usually right for crowd and background dialogue.\n\nPreset voices (20): `vivi_mixed_en_zh_ja_es_id`, `mindy_en_es_id_pt_zh`, `kian_en_zh`, `cedric_en_zh`, `sophie_en_zh`, `jean_en_zh`, `magnus_en_zh`, `mabel_en_zh`, `nadia_en_zh`, `opal_en_zh`, `pearl_en_zh`, `quentin_en_zh`, `corinne_mixed_en_zh`, `esther_mixed_en_zh`, `lyla_mixed_en_zh`, `tracy_es_zh`, `sandy_es_mixed_en_zh`, `felix_zh`, `celeste_zh`, `monkey_king_zh`.\n\nSet `multilingual: true` for non-English or mixed-language lines.\n\n### 7. Inputs: up to 3 audio clips **XOR** one image. Never both.\n\n- **Audio references** — up to 3 clips, each ≤30s and ≤10MB (wav/mp3/pcm/ogg_opus). Refer to them in the prompt as `@Audio1`, `@Audio2`, `@Audio3`: *\"match the room tone of @Audio1\"*.\n- **Image reference** — one image (jpeg/png/webp ≤10MB). The model scores what it sees.\n- Sending both is rejected by the API. Pick the one that carries the intent.\n\n### 8. The knobs, and when to touch them\n\n| Param | Range | Reach for it when |\n|---|---|---|\n| `speed` | 0.5–2.0 | Dialogue is racing or dragging against picture. |\n| `volume` | 0.5–2.0 | Rarely — normalize on the timeline instead. |\n| `pitch` | −12…+12 semitones | Ageing or shifting a voice. Small moves only; ±3 is already a lot. |\n| `multilingual` | bool | Non-English or code-switched lines. |\n| `sampleRate` | 8k–48k | Leave at 24000 unless you are matching an existing stem. |\n| `outputFormat` | mp3 / wav / pcm / ogg_opus | wav when this is going into a mix; mp3 otherwise. |\n\n## Iterating\n\n- A bed that came back wrong is almost always a **scale** problem (crowd/room too big) or a **jargon** problem (the sentence reads like a spec). Fix those two before touching `speed`/`pitch`.\n- Three failed takes on the same sentence means the sentence is wrong, not the seed. Rewrite it the way you would say it out loud.\n- Generations are cheap enough at short durations that auditioning two phrasings beats agonizing over one.\n\n## Content notes\n\nProvider-side moderation applies to voices and to recognizable real people. See slates-content-policy.\n",
|
|
30
|
-
"slates-prompting-seedance-2-5": "---\nname: slates-prompting-seedance-2-5\ndescription: How to prompt Seedance 2.5 and Seedance 2.5 Edit. Read before calling slates_generate_video with model seedance-2.5, or slates_edit_video with model seedance-2.5-edit. 2.5 is the DEFAULT video model (Eric, 2026-09-13) — against 2.0 it buys 30-second takes, 30 image references, audio-only references and INTEGER-SECOND TIMESTAMPS, and it gives up native 4K and costs more than 2.0 at every resolution they share. Timestamps are the one grammar difference that matters: 2.0 ignores them and answers only to shot numbers, 2.5 acts on them. Otherwise it shares 2.0's grammar (read slates-prompting-seedance for subject binding, camera and constraint vocabulary); this file covers what is different, plus the two hazards unique to 2.5 — the prompt-intent task classifier and the cost trap that comes with 30-second takes.\n---\n\n# Seedance 2.5 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Seedance 2.5.** Shares 2.0's grammar exactly (subject binding, camera vocabulary, externalised emotion, inline constraints — read `slates-prompting-seedance` for those). Two things are different, and both matter.\n\n**The five levers**\n1. **Timestamps work here** — integer seconds, and the model acts on them: `[0-4] she reads the letter. [4-9] she folds it and looks up.` 2.0 ignores exactly this syntax.\n2. **Length is the reason to be here** — takes up to 30 seconds, where 2.0 stops at 15. Write the beats as `[0-6]`, `[6-12]`, `[12-18]`; do not hope for them.\n3. **Up to 30 image references**, and a multi-view image can serve as ONE subject reference (up to 5 subjects). 2.0 cannot do either.\n4. **Audio-only references are accepted** without an image or video alongside — the only Seedance seat that takes one.\n5. **Keep the 2.0 discipline**: one camera move per beat (`slow track right`, `handheld follow`), physical action instead of stated emotion, and quality asked for in the image-quality slot vocabulary — `rich details`, `natural colors`, `cinematic texture`, `soft lighting`.\n\n**Examples**\n- `[0-6] Wide shot, <Subject_1>@<Image_1> crosses an empty car park toward a idling van, slow track right. [6-12] Medium, she stops as the driver's window comes down. [12-18] Close-up, she looks off past the lens and does not answer. Rich details, natural colors. Keep it subtitle-free.`\n- `[0-10] A single continuous handheld follow behind a courier climbing a fire escape, rain. [10-20] She reaches the landing, turns, and the city opens behind her. Cinematic texture, soft lighting.`\n\n**Hard constraint:** it is the default AND the dearer seat, and it has NO 4K — 480p/720p/1080p only, dearer than 2.0 at every resolution they share. Long takes multiply cost linearly: quote a 30-second take before you fire it.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** (2.5 reclassifies the task and fails a fresh generation on these):\n- `edit`, `extend`, `continue the video`, `same video but` — they make the provider read a fresh generation as an edit\n- `f/1.4`, `Portra 400` and any other aperture or film-stock token, or a stacked list of gear — image-model vocabulary. The 2.5 guide's own example names one camera body and one 35 mm lens in a single style line, so a lone lens there is not on this list\n<!-- @banned:end -->\n\n**Read `slates-prompting-seedance` first.** The prompt GRAMMAR is the same model family: the\n8-slot advanced formula, subject binding by `<Subject_N>@<Image_N>`, camera vocabulary,\nexternalised emotion, inline constraint words, the anti-twin fix. None of it is restated here.\nThis file is only what 2.5 changes — and the biggest change is that **2.5 acts on timestamps\nwhere 2.0 ignores them.**\n\n---\n\n## The one fact that decides whether you use it at all\n\n**Seedance 2.5 is the EXPENSIVE seat, not the cheap one — and it has no 4K.**\n\nIt runs at 480p, 720p or 1080p (1080p landed on all three routes on 2026-08-24), and at every\nresolution the two seats share it costs MORE than 2.0 — 720p $0.231/s against $0.15/s, **54% more**.\nSo 2.5 does not replace 2.0; it sits beside it, and you pay for what it buys:\n\n| | Seedance 2.0 | Seedance 2.5 |\n|---|---|---|\n| Resolution | 480p / 720p / 1080p / **native 4K** | 480p / 720p / 1080p — **no 4K** |\n| Price at 720p (faceless) | **$0.15/s** | $0.231/s |\n| Length | 4–15s | **4–30s in one take** |\n| Reference budget | 15 (9 image + 3 video + 3 audio) | **50 (30 image + 10 video + 10 audio)** |\n| Combined reference video/audio | ≤15s | **≤30s** |\n| Audio-only reference | ✗ (needs an image or video alongside) | **✓** |\n| **Timestamps in the prompt** | **✗ — ignored; shot numbers only** | **✓ — integer seconds, acted on** |\n| Multi-view image as ONE subject reference | ✗ (not recommended) | **✓ (up to 5 subjects)** |\n| Video edit as its own task type | ✗ | **✓ (`seedance-2.5-edit`)** |\n| Default video model | no | **yes** (since 2026-09-13) |\n\n**2.5 is the default. Route to 2.0 for 4K delivery, or when the same resolution has to be\ncheaper** — its 720p is $0.15/s against 2.5's $0.231/s.\n\n---\n\n## 🚨 Hazard 1 — the prompt-intent task classifier\n\nThis is the one that costs money and time, and it has no equivalent on 2.0.\n\n**Seedance 2.5 sorts every request into one of five task types** — text-to-video,\nreference-to-video, first/last-frame, **video edit**, **video extend** — from the reference roles\nattached **plus the intent of your sentence**. Each type then has its own parameter constraints,\nand a violation comes back **asynchronously**: the task queues, credits are reserved, and only then\ndoes it fail.\n\nThe trigger words are ordinary English:\n\n| Reclassified as | Words that do it (ByteDance's own list) |\n|---|---|\n| **video edit** | `edit video` · `add` · `insert` · `remove` · `delete` · `modify` · `replace` · `change to` |\n| **video extend** | `extend forward` · `extend backward` · `continue` · `continue from` · `extend the story` |\n\nSo a perfectly legitimate reference-to-video prompt — *\"a wide shot of the workshop, **remove** the\ntripod from frame\"* — gets classified as an edit and fails on constraints it never set.\n\n**What to do:**\n\n1. **If you mean to edit an existing clip, say so with the MODEL, not the sentence.** Call\n `slates_edit_video` with `model: 'seedance-2.5-edit'`. That routes to a dedicated\n task-typed endpoint and the classifier never has to guess.\n2. **If you mean a fresh shot, describe the finished frame rather than an instruction to change\n one.** Not *\"remove the tripod\"* → *\"the workshop bench, clear and uncluttered\"*. Not\n *\"add rain\"* → *\"heavy rain falling through the streetlight\"*. This is better prompting anyway:\n the model renders what you describe, it does not take edits to an imagined draft.\n3. The trigger only fires when **references are attached**. A plain text-to-video prompt is safe\n however it is worded.\n\n**Slates will warn you, and it will never rewrite your prompt.** When a 2.5 reference generation's\nprompt contains one of these words, the composer shows a warning that NAMES the words and the agent\nroute returns the same string. Silently editing the user's sentence to dodge a provider classifier\nis forbidden — the words that reach the model are always the words the user can see.\n\n---\n\n## 🚨 Hazard 2 — resolution is not the price dial here. LENGTH is.\n\nEvery other model in Slates trains the habit that lower resolution means lower cost. 2.5 breaks it,\nbecause the thing that moves the bill is **length**, and 2.5's length ceiling is double 2.0's.\n\nWorked, at the shipped rates:\n\n| Generation | Credits |\n|---|---|\n| 2.5 · 480p · 5s · faceless | 26 |\n| 2.5 · 720p · 5s · faceless | 58 |\n| 2.5 · 1080p · 5s · faceless | 142 |\n| 2.5 · 720p · 30s · faceless | 347 |\n| 2.5 · 720p · 30s · AI-face route | **489** |\n| 2.5 · 720p · 30s · consented real-face route | **710** |\n| 2.5 · 1080p · 30s · faceless | **853** |\n| 2.5 · 1080p · 30s · consented real-face route | **1,749** |\n| *(for scale)* 2.0 · 1080p · 15s · AI-face route | 411 |\n\n**A 30-second 720p clip can cost more than a 15-second 1080p one** — and a base licence starts\nwith 1,000 credits. Someone who reads \"720p\" as \"cheap\" and asks for a 30-second take on the\nreal-face route has spent 71% of their welcome grant on one clip; **on the real-face route a single\n30-second 1080p take is more than the whole grant.**\n\n**Discipline:**\n\n- **Always quote with `slates_estimate_generation_cost` before a take over ~10 seconds,** and say\n the number out loud before generating.\n- **Find the shot at short LENGTH, not at low resolution.** Length is what moves the price, so cut\n seconds while you are still exploring — 4–8s — and stay at the resolution you actually want.\n **A 480p pass does not de-risk a 720p or 1080p render.** Generation is stochastic: the higher-\n resolution run is a different take, not the same shot rendered better. So a 480p draft that looks\n right buys you no guarantee, and one that looks wrong may have been fine at 720p — you paid 26\n credits to learn nothing, when 58 would have bought a real candidate.\n- **Length is a creative decision, not a default.** 30 seconds is available; it is rarely the right\n answer for a single shot. Multi-shot storyboards inside one 30s generation are what the length is\n actually for.\n- Read `slates-cost-discipline` — all of it applies, more sharply here.\n\n---\n\n## Timestamps — the one grammar change\n\n<!-- @inject:seedance-25-timestamps -->\n**2.0 does not respond to timestamps and answers only to shot numbers. 2.5 responds to\ninteger-second timestamps.** That is ByteDance's own first line under \"Differences from Seedance\n2.0\", and it is why a 30-second take is usable at all: the length is only worth buying if you can\nsay *when* things happen inside it.\n\nBoth formats are valid on 2.5, and you can mix them — `Shot N` blocks for a storyboard whose\npacing you are happy to leave to the model, timestamps when a beat has to land at a moment.\n\n**Three ways to control time, all first-party:**\n\n| Form | Write it like |\n|---|---|\n| **Interval** | `0-3 seconds… 3-7 seconds… 7-15 seconds` or `[1s-4s]… [4s-8s]… [8s-12s]` |\n| **Time point** | *\"Quick left sideways transition at the 5-second mark.\"* |\n| **Relative** | *\"After 3 seconds, everyone around him shakes their head.\"* · *\"The frame freezes for 1 second after he presses the shutter.\"* |\n\n**The rules that come with them:**\n\n- **One second is the smallest unit.** Integers only — no `2.5s`, no frames.\n- **No gaps in the timeline.** `0-3s… 5-6s…` leaves 3-5s unspecified and the model fills it however\n it likes. Intervals must abut: `0-3s`, `3-7s`, `7-15s`.\n- **Budget the plot to the seconds.** Too little content in a range and the model improvises to\n fill it; too much and you get extra cuts or dropped beats. This is the actual craft of a 30s take.\n- **Never time-code a high-frequency action.** *\"Shake your head three times per second\"* is\n explicitly called out as a misuse — timestamps schedule beats, they don't choreograph frames.\n- **Transitions want both halves:** the moment AND the method — *\"At the 5-second mark, the camera\n transitions leftward with a left wipe into a natural dissolve.\"*\n- **Timestamps work on an EDIT too**, and that is where they earn the most: they scope a change in\n time as well as in content — *\"Change the man's action from drinking coffee to mopping the floor\n from 4-6 seconds in Video 1, and leave the rest of the content unchanged.\"* Without a range, a\n whole-clip instruction is applied to the whole clip.\n\nDo **not** carry this back to 2.0, and do not carry Veo's `[00:00-00:02]` bracket syntax into\neither — 2.0 ignores time entirely, and the cross-model syntax swap is its own known failure.\n<!-- @end:seedance-25-timestamps -->\n\n---\n\n## What the extra reference budget is actually for\n\n30 image references (up from 9) does **not** mean \"attach 30 images\". Every rule in\n`slates-prompting-seedance` about references still holds — 2–4 strong references beat both\nextremes, and one reference per role.\n\n**Where 2.5 moves the ceiling, per ByteDance's own input recommendations:**\n\n| | Stable | Works, but expect re-rolls |\n|---|---|---|\n| Subjects bound by IMAGE reference | 1–8 | 9–12 |\n| Subjects bound by VIDEO or AUDIO reference | 1–5 | 6–10 |\n| Reference clip length, per subject | 5–10s | longer drops stability |\n\n**Multi-view images of one subject are supported on 2.5** — a turnaround sheet can be a single\nreference image, where 2.0 wanted one authoritative rendering per subject. Past **5 subjects**,\ngo back to single-view images, one per view, rather than one image carrying several viewpoints.\n\nThe larger budget earns its keep in exactly two places:\n\n- **A long multi-shot take** where different shots need different subjects and locations bound —\n the budget is spread across the storyboard, not stacked on one frame.\n- **Video and audio references alongside images**, which is where 2.5's 10 + 10 matters far more\n than the image count.\n\n### Audio-only references — the genuinely new input\n\n2.0 required an image or video alongside any audio reference. **2.5 accepts audio on its own.**\nThat makes one recipe possible that was not before: drive a scene's timing, voice or ambience from\na recording with no visual anchor at all — a voice line, a music bed, a room tone — and let the\nmodel build the picture to it. Cite it the same way as any other reference\n(`Reference the timbre in <Audio_N> to generate…`), and remember that audio references carry **no\nbilling dimension** on any Seedance route: audio is included.\n\n### Video references\n\nUp to 10 clips, ≤30s combined (2.0: 3 clips, ≤15s). A reference VIDEO switches the cost key to\n`seedance-2.5*-vref-{res}-{T}s`, where **T = Σ input seconds + output seconds** — the sum is across\n**every** clip attached, not just the longest. Three 6-second references on a 12-second output bills\n30 seconds, not 12 and not 18. Quote before confirming.\n\n**The cap is a refusal, not a trim.** Attach an eleventh clip, or push past 30 combined seconds, and\nthe composition is rejected before anything uploads. That asymmetry is deliberate: reference images\nwarn-and-trim because dropping one doesn't change the price, and a dropped reference VIDEO would be\none you were quoted for and the model never saw.\n\n### Mixing all three in one call\n\n50 files total (30 image + 10 video + 10 audio) is a shared budget. Everything is cited positionally\nby type — `image 1`, `video 2`, `audio 1` — in attachment order, so reordering the attachments\nrenumbers the citations. Write the prompt against those numbers:\n\n```\nMarcus (image 1) performs the motion from video 1 in the workshop from image 2,\nusing the voice timbre from audio 1. Preserve his identity, appearance and outfit.\n```\n\n🚨 **SAY WHAT AN AUDIO REFERENCE IS FOR.** It can mean music, dialogue, voice, tone or timbre — five roles on one attachment — so an unroled clip falls back to **dialogue**: the model re-transcribes it and speaks ITS words. A real take came back as *\"a map called Slates\"* for *\"an app called Slates\"*. Name it as the voice timbre and the clip carries the voice while the prompt carries the words. ByteDance's own sentence: *\"Image 1 depicts the protagonist John and uses the voice timbre from Audio 1.\"* Bind each speaker in a sentence, never by attachment order — position carries nothing.\n\nFrames and reference media stay mutually exclusive, in every combination — the reference endpoint\nhas no first/last-frame parameters at all, so this is a shape mismatch rather than a preference.\n\n---\n\n## Sound: four bracket types, and they are the vendor's syntax\n\nByteDance's 2.5 API tutorial states this as a **prompt rule**, not a suggestion — verbatim: *\"Use\nspecial characters to distinguish sounds: `()` for music, `<>` for sound effects, `{}` for dialogue,\nand `【】` for subtitles. For non-Chinese dialogue, it is recommended to specify the language before\nthe dialogue.\"*\n\n```\nShe sets the cup down {English: \"We open in ten minutes.\"} <ceramic clink on wood>\n(low piano, unhurried)\n```\n\n- `()` **music** · `<>` **sound effects** · `{}` **dialogue** · `【】` **on-screen subtitles**\n- **Name the language before non-Chinese dialogue.** `{English: \"...\"}`.\n- Unbracketed sound description still works — this is a disambiguator, not a required wrapper. Reach\n for it when one sentence carries more than one kind of sound and you need the model to tell them\n apart, which is exactly where an unmarked prompt puts a line of dialogue into the score.\n\n⚠️ **These four are SEEDANCE 2.5's.** MiniMax H3 has its own three-layer scheme (body / soundscape /\nscore) and its angle brackets are documentation notation that must never be typed. Do not carry\neither grammar onto the other model.\n\n## Say what a reference is NOT for\n\nThe same rule adds a half nobody uses: *\"Specify what each asset provides, such as appearance,\naction, or timbre, **and what should not be referenced**.\"* Negative scoping is a first-class part of\nthe citation, not a fallback — *\"use her face and wardrobe from image 1, not its lighting or\nbackground\"* is a stronger instruction than naming the positive alone, because an unscoped reference\nbrings its whole frame with it.\n\n🚨 **The vendor writes `@Image 1`; Slates writes `image 1`, and that difference is deliberate.**\nBytePlus's API tutorial says *\"Use `@Image 1`, `@Video 1`, and `@Audio 1`\"*, while its own 2.5 prompt\nguide states the bare form (`Image 1 / Video 1 / Audio 1`) in the one normative sentence it has.\n**Two first-party docs, two forms** — the disagreement is recorded, not resolved, in\n`second-brain/business/projects/slates/research/model-prompting-research.md`. What settles it FOR US\nis neither: **`@` is a reference-token sigil in the Slates prompt composer, and an unresolved one is\nsilently deleted from the prompt before it is sent.** Typing `@Image 1` here does not produce\n`@Image 1`, it produces nothing. The bare form is confirmed working on both models. Never hand-type\nthe sigil.\n\n## Seedance 2.5 Edit (`slates_edit_video`, `model: 'seedance-2.5-edit'`)\n\nIts own picker row and its own op call, deliberately: the task type is **the model you chose**,\nnever something inferred from your sentence.\n\n**Why route here at all:** it is the **only edit engine in Slates that accepts a clip longer than 15\nseconds** (4–30s, versus Kling O3 Edit's 3–15s and Omni Flash Edit's 3–10s). For a clip inside the\nothers' range, choose on fidelity instead — Omni Flash Edit won the prompt-only head-to-head, and\nKling O3 Edit is the one that takes element and style reference images.\n\n**How it behaves:**\n\n- **Output length follows the SOURCE clip**, and the bill is the ceiled source length. The provider\n requires an automatic duration on this task type, so there is no length knob — the clip you attach\n is the quote. The returned clip can differ from the source by up to ~0.3s, which only compresses\n transition frames; a clip that 2.5 itself generated comes back at exactly its input length.\n- **Source clips under 20 seconds edit more reliably.** 4–30s is what the task type accepts;\n ByteDance's own recommendation is to stay inside 20 for quality. A 28-second source is legal and\n will need more attempts.\n- **The aspect ratio follows the source clip too.** No ratio control; the frame is the clip's frame.\n- **480p, 720p or 1080p output**, native audio — and an edit bills the video-reference tier ×2,\n so 1080p on this row is the most expensive second in the app. Quote it.\n- **Prompt and source clip only** on this op. The MODEL takes reference images on an edit\n (ByteDance recommends 1–5 — *\"replace the man in dark clothing in @Video 1 with @Image 2\"*);\n **Slates has not wired that path**, so today an edit that must lock an identity from a photo\n goes to Kling O3 Edit. Constraint of our build, not of the model — worth revisiting.\n- **An edit bills roughly DOUBLE a plain 2.5 generation of the same length**, because every provider\n charges an edit on input + output seconds. Read the confirm gate's number; do not reason from the\n generation rate.\n- **Set `seedanceFace: true` when a character's face is visible in the clip.** The faceless provider\n blocks faces outright — this is not a price optimisation, it is whether the job runs at all.\n- **There is no consented-real-face route for editing.** Real-person footage that the AI-face route\n rejects has to go to Kling O3 Edit.\n\n**Prompting an edit** — the same discipline as every other edit engine: **describe only what\nchanges.** The source already carries its composition, motion, timing and performance; re-describing\nthem fights the model. Use Seedance's own edit grammar from `slates-prompting-seedance`\n(*\"Strictly edit `<Video_1>`, and modify `<Original_Characteristic>` to `<New_Characteristic>`\"*) and\n**never** write *\"reference video 1\"* in an edit — the official guide is explicit that this phrasing\ngets the request reclassified as a reference task, which is the same landmine as Hazard 1.\n\nTwo things sharpen an edit prompt, both first-party:\n\n- **Say it as A → B, not as an outcome.** *\"Change the man's action from drinking coffee to mopping\n the floor\"* beats *\"the man mops the floor\"* — naming what it currently is tells the model what\n to overwrite.\n- **Timestamp a partial edit** — the edit task type reads the same integer-second timestamps the\n generation path does. Rules and forms are in § Timestamps above; this is the single most useful\n thing they buy.\n\n**Audio is editable too, and it is the least obvious use of this row.** The same op rewrites what\nis heard while the picture stays put: change a spoken line, change the accent, translate the\ndialogue and re-fit the lip movement, strip or replace the BGM or a sound effect. *\"Only edit the\nman's dialogue in Video 1: change it to 'Don't come over here,' in an American accent\"* is an edit,\nnot a lip-sync job. Bill it like any other edit — on the source clip's length.\n\n---\n\n## Faces, and what does NOT change\n\nThe three-tier face routing is identical to 2.0 — faceless → default route, an AI character's face →\n`seedanceFace: true` (the relaxed provider, a real cost premium), a real person's photo → the\nconsent-gated real-person route after a `[REAL_FACE_DETECTED]` rejection, with `realFaceConsent: true`\nset **only** after the user explicitly confirms they hold the rights to the likeness. The full rules,\nincluding why the real-vs-AI call is the provider's and not yours, are in\n`slates-prompting-seedance`.\n\nAlso unchanged, and worth restating because 2.5's length makes each one more expensive to get wrong:\n\n- **One primary camera move per shot.**\n- **No stacked lens / aperture / film-stock vocabulary.** One camera-and-lens style line is the\n most the 2.5 guide itself uses; the full rule is in `slates-prompting-seedance` → \"Don't\n cross-pollinate image-model syntax\".\n- **No `negativePrompt` field** — constraints go inline, and 2.5 acts on negative phrasing in\n exactly two dimensions: subtitles (*\"no subtitles\"*) and audio (*\"no BGM; environmental and\n action sounds only\"*, *\"no audio\"*). Everywhere else, describe what you want, not what you don't.\n- **Legible in-shot text still belongs in a baked start frame**, not in the video prompt.\n",
|
|
31
|
-
"slates-prompting-seedance": "---\nname: slates-prompting-seedance\ndescription: How to prompt Seedance 2.0 (ByteDance video model). Read before calling slates_generate_video with model seedance-2. Seedance 2.0 structures multi-beat prompts as a \"Shot 1 / Shot 2 / Shot 3\" storyboard against an 8-slot advanced formula — never per-second time stamps, which 2.0 does not respond to (Seedance 2.5 does; see slates-prompting-seedance-2-5). Its syntax differs from Kling, Veo and the image models; don't cross-pollinate (in particular, no lens / aperture / film-stock vocabulary).\n---\n\n# Seedance 2.0 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Seedance 2.0.** Not copywriting — an ENGINEERING instruction to a spatial layer and a temporal layer: who, in what scene, doing what, how the camera moves, and in what order. Multi-beat work is a `Shot 1 / Shot 2 / Shot 3` storyboard.\n\n**The five levers**\n1. **Bind every subject to its reference** — `<Subject_1>@<Image_1>` — and keep the descriptions identical across shots. Unbound subjects are where twins come from.\n2. **Shot sizes and camera MOVES, not lens data** — `medium close-up`, `slow push in`, `handheld follow`, `whip pan`. One primary move per shot.\n3. **Externalise emotion as physical action.** Not \"she is nervous\": `she turns the ring on her finger twice, then stops`.\n4. **Use the image-quality slot vocabulary** for quality — `HD`, `rich details`, `cinematic texture`, `natural colors`, `soft lighting`. That is the officially sanctioned way to ask.\n5. **Constraints go INLINE**, led by the official templates — `keep it subtitle-free`, `do not generate a watermark`, `avoid jitter and bent limbs`, `avoid temporal flicker`.\n\n**Examples**\n- `Shot 1: medium shot, <Subject_1>@<Image_1> steps out of the freight lift into a wet loading bay, slow push in. Shot 2: close-up, she turns the ring on her finger twice and stops, handheld. Rich details, cinematic texture, natural colors. Keep it subtitle-free.`\n- `Single continuous take. Wide shot of a fishing skiff crossing a grey swell, camera tracks from the starboard rail. Spray hits the lens once. Soft lighting, natural colors, film-grain texture. Avoid jitter and bent limbs.`\n\n**Hard constraint:** NO timestamps — 2.0 ignores them and answers only to shot numbers (2.5 acts on them). No lens, aperture, film stock or camera body: that is image-model vocabulary and a Seedance anti-pattern. There is no negativePrompt field.\n<!-- @card:end -->\n\nByteDance's video model — first-party via **BytePlus ModelArk** (credits only, no BYOK). Audio always generated alongside the video. Single model `seedance-2` across the full resolution ladder (480p / 720p / 1080p / native 4K — 4K video is Pro-only, default 1080p), 4–15s, first+last frame, and up to 9 reference images / 3 videos / 3 audio clips.\n\n> **How to read this file.**\n> **[official :NNNN]** — ByteDance's own BytePlus ModelArk prompting guide, line `NNNN` of the archived doc dump (`research/byteplus-seedance-2-0-api-docs.md`). Receipt-grade; treat as law.\n> **[community]** — third-party guides and our own field notes. Useful, but an `[official]` block always wins.\n> **[slates]** — how the Slates app composes or bills this; not ByteDance doctrine.\n>\n> The split is load-bearing. A community-sourced \"narrative timing beats\" doctrine shipped in this file for months teaching the **exact inverse** of ByteDance's published guidance. Never merge the two registers again.\n\n---\n\n# Part 1 — Official ByteDance doctrine\n\n## What Seedance actually is `[official :1450-1452]`\n\nSeedance 2.0 is a multimodal AI director. It reads text, images, video and audio **simultaneously** and internally decomposes them into two dimensions:\n\n- the **spatial layer** — what is in the frame\n- the **temporal layer** — how things change over time\n\nSo a good prompt is **not \"copywriting-style description\" but an \"engineering-style instruction\"**: who, in what scene, doing what action, how the camera moves, and in what chronological order events occur — delivered respectively to the spatial layer and the temporal layer.\n\n## The advanced formula — 8 slots `[official :1455]`\n\n```text\nprecise subject + action details + scene/environment + lighting & color tone\n+ camera movement + visual style + image quality + constraints\n```\n\n⚠️ There is **no official \"6-step formula.\"** `Subject + Action + Environment + Camera + Style + Constraints` is community branding with no ByteDance source, and it silently drops the **lighting & color tone** and **image quality** slots. Use the 8 slots above.\n\n## Task-type sentence patterns `[official :1389-1425]`\n\nSeedance classifies your request from the phrasing. Use the pattern that matches the task:\n\n| Task | Pattern |\n|---|---|\n| **Image reference** | ``Reference `<Subject_N>` in `<Image_N>` to generate…`` |\n| **Video reference** | ``Reference `<Action / Camera_movement / Style / Sound_effect>` in `<Video_N>` to generate…`` |\n| **Audio reference** | ``Reference the timbre in `<Audio_N>` to generate…`` |\n| **Video edit — modify** | ``Strictly edit `<Video_N>`, and modify `<Original_Characteristic>` in it to `<New_Characteristic>``` |\n| **Video edit — add** | ``<Element_Features>` + `<Timing>` + `<Location>`` |\n| **Video edit — delete** | Name what to delete; for anything that must stay, say so explicitly |\n| **Video extend** | ``Extend `<Video_N>` forward/backward to generate…`` |\n| **Combined** | ``Reference `[Dimension]` of `<Image/Video_N>`, strictly edit `<Video_X>`, `[Specific_Edits]``` |\n\n### ⚠️ Edit / extend phrasing landmine `[official :1431]`\n\n> *\"For edit / extend video tasks, directly use `<Video_N>` to refer to the video. **Do not use \"reference `<Video_N>`\"**, to avoid being incorrectly identified as a reference task.\"*\n\nThis is easy to trip: Slates has an edit lane<!-- slates-only --> (`slates_generate_video` with `videoReferenceAssetId`, plus the Seedance edit/relocate routes)<!-- /slates-only -->. Writing *\"reference video 1 and change the jacket to red\"* gets classified as a **reference** task — the model generates a brand-new clip inspired by the source instead of editing it. Write *\"Strictly edit video 1, and modify the blue jacket to red.\"*\n\n## Shot structure — \"Shot 1 / Shot 2 / Shot 3\" `[official :1563-1598]`\n\n> *\"Use shot order, write a simple 'Shot 1 / Shot 2 / Shot 3' storyboard for each segment of the video, and then merge them into a complete prompt.\"*\n\n**❌ Never second-stamp.** No `0:00–0:03`, no \"At 4 seconds\", no per-segment durations.\n\n> *\"Do not impose strict limits on the duration of each segment; prioritize allowing the model to naturally generate the pacing based on the plot.\"* `[:1580]`\n>\n> *\"The model's support for precise timing (such as 0–3 seconds) is **unstable**, and forcibly limiting duration may lead to **abnormal generation results**.\"* `[:1586]`\n\nOrder shots by when events occur — primary first, secondary later. Let the plot set the pacing.\n\n**Per-shot internal order** `[official :1590-1598]` — organize each shot in exactly this sequence:\n\n1. **Camera movement or shot transition** — \"slowly push in from a wide shot\", \"fixed camera position\", \"cut to…\"\n2. **Subject actions and expressions** — the key actions and expression changes of the core character/object\n3. **Position or spatial change** — where the subject is, and how that relationship shifts\n4. **Audio** — sound effects, voices, background music for that shot\n\n**One primary camera move per shot** — see Camera below. `[official :1648]`\n\n## Subject binding — names + image indexes `[official :1488-1556]`\n\nEvery time a subject appears, it must be **explicitly referred to**. Two supported forms:\n\n- **Undefined subjects** — bind inline every mention: `<Subject_N>@<Image_N>`. Official example: **`Zhang San@Image 1`**. `[:1540]`\n- **Pre-declared subjects** — define once, then reuse the same label verbatim: *\"Define the tall man in **Video 1** as **police officer**, and define the other short man as **thief**\"*, then say \"police officer\" every time after. `[:1514]`\n\n**One subject spread across several assets** — bind them together: *\"Define `[…]` in **Image 1** and `[…]` in **Image 2** as `<Subject N>`.\"* `[:1514]`\n\n⚠️ **An Asset ID must never substitute for `<Image/Video_N>`.** `[:1546]` *\"the model cannot directly associate the Asset ID with the reference content.\"* Always cite by index.\n\nAlso official: keep descriptions concise, avoid redundancy, avoid semantic conflicts (contradictory traits for one subject), and prefer expressing spatial relationships through reference images rather than dense text. `[:1550-1556]`\n\n**`[slates]`** — the app composes this for you. `composeReferences()` cites each canonical character or environment reference inline as `Name (image N)` in the exact order it sends them, which is ByteDance's own duplicate-character format (*\"Zhang San (corresponding to image 1)\"* `[:1976]`). You never hand-write role labels or index numbers.\n\n## Action description `[official :1602-1621]`\n\n- **Body-part specificity + quantified degree.** Name hands, legs, head, shoulders, back — and supplement **range, speed, and force**. *\"slowly raise a hand\", \"quickly turn the head\", \"push hard off the ground\", \"slightly lower the head.\"*\n- **Prioritize slow, gentle, continuous small movements.** Avoid high-burst, large-dynamic actions — sprinting, big jumps, violent rolls. *\"walk slowly\", \"gently raise a hand\", \"sit down naturally with the motion.\"* **This is the official basis for the folk rule that \"fast\" degrades quality** — it is not a banned token, it is a class of motion the model handles badly.\n- **Supplement transitions between actions.** Specify inertia and continuity between consecutive beats so movement reads coherent: *\"use the inertia of turning around to naturally raise a hand\", \"naturally transition from a pause into raising a hand.\"*\n\n## Externalize emotion `[official :1623-1636]`\n\nReplace abstract emotion words (\"very sad\", \"extremely angry\") with **specific physical detail**. This is the highest-leverage single habit in the official guide:\n\n| Abstract | Externalized as actions and details |\n|---|---|\n| **Sadness** | head lowering, shoulders trembling slightly, eyes reddening, fingers unconsciously clutching the corner of clothing, tears welling but not falling |\n| **Joy** | corners of the mouth rising uncontrollably, brows and eyes relaxing, steps becoming light, unconsciously humming a tune |\n| **Nervousness / anxiety** | frequently checking the watch, fingers constantly tapping the tabletop, rapid breathing, eyes darting away |\n| **Anger** | both fists clenched, jawline tense, chest heaving, eyes sharp, squeezing words out through gritted teeth |\n| **Relief** | letting out a long breath, tense shoulders completely relaxing, a faint smile appearing, looking up toward the distance |\n\n## Camera `[official :1643-1648]`\n\n> *\"The model has a **strong understanding of camera movement terms**, so you can **directly use standard camera movement terminology**, such as 'medium shot, close-up, wide shot, slow push-in, smooth lateral tracking, fixed shot.'\"*\n\nThis is an **open vocabulary, not a fixed list** — and it explicitly includes **shot size** (close-up / medium / wide / long shot), which is as much a camera instruction as the move itself.\n\n> ⚠️ *\"Try to specify only 1 type of camera movement in a single shot. Do not require push, pull, pan, and move at the same time, as this will increase image instability.\"* `[:1648]`\n\n## Image quality, style, and constraints `[official :1656-1679]`\n\nThese three slots \"define creative boundaries for the model, unify image quality and artistic tone, and avoid visual flaws and random deviations.\"\n\n**1. Image quality** — define clarity, texture detail, and lighting quality. Official vocabulary: `HD` · `rich details` · `cinematic texture` · `natural colors` · `soft lighting`.\n\n> ⚠️ This is a **real slot with real vocabulary** — do not confuse it with Stable-Diffusion-era quality incantations. `8K` / `masterpiece` / `trending on artstation` remain banned slop tokens (see Part 3); *\"cinematic texture, rich details, natural colors\"* is the officially sanctioned way to ask for the same thing.\n\n**2. Style** — the overall art style and visual tone: `cyberpunk cool blue-purple tone` · `retro film` · `fresh Japanese style`.\n\n**3. Constraint words** — *\"Constraint words are very important. They can effectively avoid visual flaws, deformities, breakdowns, and unreasonable elements.\"* Official templates, verbatim:\n\n- **No subtitles** — \"keep it subtitle-free\" / \"avoid generating any text or subtitles\"\n- **No logo** — \"do not generate a logo\"\n- **No watermark** — \"do not generate a watermark\"\n\nSeedance has **no `negativePrompt` field** — constraints go inline in this slot. See Part 3 for the wider inline-negative kit.\n\n## 🔴 Duplicated characters — the twin problem `[official :1948-1994]`\n\n**Symptom:** in frames with **many characters**, where **three-view / multi-view character images** are supplied as references, two identical characters appear in the same generated frame.\n\n**Root causes** `[:1954-1959]`:\n1. Character subjects are not clearly defined in the prompt, so the model cannot distinguish roles.\n2. *\"When character **three-view / multi-view images** are used as reference assets, it is easy to confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"*\n\n**Official fixes, in their order** `[:1971-1994]` — ByteDance is explicit that these *reduce probability*, not eliminate it:\n\n1. **Bind each character to its image explicitly**, in a consistent format. Official example: *\"Zhang San (corresponding to image 1) throws the green passbook toward Li Si (corresponding to image 2), who is standing.\"*\n2. **Append the global constraint verbatim** at the end of the prompt `[:1982]`:\n > *\"Throughout the video, characters with completely identical appearance, clothing, and accessories are prohibited. Do not generate duplicate avatars or a twin effect. Keep only a single corresponding character in the same frame, and do not reproduce repeated copies of characters.\"*\n3. **Optimize reference assets** `[:1988]` — *\"For character reference images, prioritize independent single-person photos. Three-view or multi-view assets are not recommended.\"*\n4. **Simplify the prompt** — do not paste a whole script; redundant copy confuses the model.\n\n**Scope this honestly.** This is troubleshooting for the twin problem in **multi-character frames**, not a blanket verdict on identity sheets. Practical rule for Slates:\n\n- **Multi-character Seedance shot** → bind every character to its image, append the anti-twin constraint, and prefer single-person / dominant-portrait references over multi-view sheets.\n- **Single-character shot** → the standard character-sheet flow is fine.\n\n**Too many reference people** `[official :2048-2052]` — past **4 reference people**, output stability drops (wrong headcount, duplicates). Official workaround: group the cast into images of ≤4 people each, generate those stills first, then drive the video from them.\n\n## Worked examples `[official :1689-1745]`\n\nThese are ByteDance's own end-to-end cases. Note the shape: an asset-binding preamble, then `Shot N` blocks in event order, then a trailing style + stability paragraph. No time stamps anywhere — 2.0 does not respond to them at all, which is version-scoped and reverses on 2.5.\n\n**Example 1 — dormitory emotional short drama (dialogue-focused).** Assets: `@Image 1` half-body photo of the female lead · `@Image 2` dormitory scene reference · `@Video 1` camera-movement reference · `@Audio 1` indoor ambience.\n\n> Use the girl in @Image 1 as the main character, use @Image 2 as the dormitory scene style reference, and refer to the camera movement in @Video 1.\n>\n> **Shot 1**: At dusk, **girl @Image 1** walks briskly to the **dormitory entrance @Image 2**. The camera follows steadily in a medium shot. Warm yellow sunlight spills into the hallway from the window. She pauses at the doorway, takes a deep breath, and looks slightly nervous.\n>\n> **Shot 2**: **Girl @Image 1** pushes the door open and enters the dormitory. The camera cuts to an indoor medium shot. Her roommates look up at her while organizing their books. One of them smiles and asks {How did the exam go? Did you pass?}. The camera slowly cuts between half-body close-ups of several people.\n>\n> **Shot 3**: **Girl @Image 1** first lowers her head with a dejected expression. The camera gives her a close-up. Then she raises her head, unable to hold back a smile, laughs out loud, and says {I was kidding}. Her roommates start chasing and play-fighting with her. The camera slowly pulls back and freezes on a wide shot of the dormitory filled with laughter.\n>\n> The entire video should have a high-definition cinematic documentary style, with warm tones and soft lighting. The character's face remains stable without deformation; movements are natural and smooth, with no stutter or flicker. The ambient sound blends naturally with @Audio 1.\n\n**Example 2 — ancient-style cliff confrontation (action/atmosphere-focused).** Assets: `@Image 1` female lead in red · `@Image 2` assassin in black · `@Image 3` cliff and bamboo forest · `@Video 1` martial-arts camera reference · `@Audio 1` drum beats.\n\n> Use the woman in red from @Image 1 as the female lead, use the woman in black from @Image 2 as the opponent, use the cliff and bamboo forest environment in @Image 3 as the scene reference, refer to the overall camera movement and action rhythm in @Video 1, and synchronize the background sound effects with @Audio 1.\n>\n> **Shot 1**: At dusk, the camera slowly pushes in from a side medium shot of **woman in red @Image 1**. She stands at the edge of the cliff and lifts a wine flask to drink. Her sleeves and robe hem sway gently in the mountain wind. The camera circles halfway around her, moving from the front to her back. In the distance, a figure in black is faintly visible in the bamboo forest.\n>\n> **Shot 2**: The camera zooms and fades into a long shot. From a drone perspective, it overlooks the entire cliff and bamboo forest. The two characters stand at opposite ends of the cliff. The mountain wind lifts their robe hems and dust, and the rhythm slightly accelerates with the drum beats.\n>\n> **Shot 3**: The camera cuts back to a ground-level close shot. The two slowly draw their swords and face off. **Woman in red @Image 1** shifts from a careless expression to a cold gaze. **Woman in black @Image 2** looks determined, and the sword tip trembles slightly. The camera steadily follows the two as they circle each other, finally freezing on a close-up of the instant before the two swords meet.\n>\n> The overall visual style should feel like a cinematic wuxia world in misty rain, with cool tones, low saturation, a film-grain texture, and rich light-and-shadow layers. The characters' faces and body proportions remain stable without deformation. Movements are continuous and natural, not stiff, with no clipping or stutter.\n\n## Other official notes\n\n- **On-screen text** `[official :1758]` — Seedance can render common text (ad slogans, subtitles, speech bubbles) and will auto-match style/colour from context, or take an explicit colour / style / timing / position. Prefer **common characters**; avoid rare glyphs and special symbols. (For *guaranteed* legible text, the start-frame route in Part 3 is still safer.)\n- **Extension degrades quality** `[official :2004-2024]` — using a generated video as the input for extension compounds degradation, with mottled colour blocks in face regions. Limit repeated continuations; prefer HD assets as input.\n- **Special effects that miss** `[official :2031-2044]` — when a described effect comes out wrong (a countdown that scrolls randomly), define it with a **reference video** instead of words: *\"the way the number '2999' appears should reference video 1.\"*\n\n---\n\n# Part 2 — Slates-specific `[slates]`\n\n## Reference media — caps and transport\n\nReference-to-video accepts up to **9 reference images, 3 reference videos, 3 audio clips** `[official :275-281]`. Text+audio-only and audio-only inputs are not supported.\n\n**Mutually exclusive:** first-frame/last-frame mode CANNOT be combined with reference images. The error reads `\"first/last frame content cannot be mixed with reference media content.\"` Pick one or the other. *(Official note `[:284]`: you can approximate first/last frames via prompt wording inside a multimodal call, but if the frames must be exact, use the dedicated first/last-frame route.)* The same rule covers reference VIDEO and AUDIO: they ride the reference endpoint, which has no frame parameters at all.\n\n### All three modalities go in ONE call\n\nThe caps are a shared budget, not three separate features: **12 files total on 2.0** (9 image + 3 video + 3 audio), **15 seconds of reference video combined**, **15 seconds of audio combined**. On 2.0 an audio reference needs at least one image or video alongside it; 2.5 accepts audio on its own.\n\nCite each by type and index, in the order they were attached — `image 1`, `video 1`, `audio 1`. The index is positional: reorder the attachments and the numbers move with them.\n\n```\nMarcus (image 1) performs the motion from video 1, in the workshop from image 2,\nusing the voice timbre from audio 1. Preserve his identity, appearance and outfit.\n```\n\n🚨 **SAY WHAT AN AUDIO REFERENCE IS FOR.** It can mean music, dialogue, voice, tone or timbre — five roles on one attachment — so an unroled clip falls back to **dialogue**: the model re-transcribes it and speaks ITS words. A real take came back as *\"a map called Slates\"* for *\"an app called Slates\"*. Name it as the voice timbre and the clip carries the voice while the prompt carries the words. ByteDance's own sentence: *\"Image 1 depicts the protagonist John and uses the voice timbre from Audio 1.\"* Bind each speaker in a sentence, never by attachment order — position carries nothing.\n<!-- slates-only -->\n**Attaching a clip is NOT the same as editing it.** \"Add as reference\" puts it in the composer alongside everything else and wipes nothing; \"Edit with AI\" makes the clip the canvas and clears the tray for a fresh instruction. Two different jobs, two different menu entries — never infer one from the other.\n\n**Over the cap is REFUSED, never trimmed.** A reference video is priced into the quote before it is sent, so a clip silently dropped after the quote would be a clip you paid for and the model never saw. Remove one and retry.\n<!-- /slates-only -->\n\n### Motion transfer & lip-sync recipes (reference video / audio)\n\nThese aren't separate Seedance features — they're prompting strategies over reference media.<!-- slates-only --> The Slates tools (`slates_generate_motion_transfer` / `slates_generate_lip_sync` with the seedance engine) compose them for you. When driving them by hand through `slates_generate_video`:<!-- /slates-only -->\n\n- **Motion transfer:** subject image as a reference + the driving clip<!-- slates-only --> via `videoReferenceAssetId`<!-- /slates-only --> (2–15s) + `The character from image 1 performs the exact motion, choreography, and camera movement from video 1. Preserve the character's identity, appearance, and outfit.`\n- **Lip-sync / dialogue:** write the line in the prompt — `The person in video 1 says: \"…\"` — with audio generation on (always on in Slates). A **video** source's own voice is cloned natively; an **audio** reference (≤15s) drives speech from an existing recording: `…speaks the dialogue from audio 1 with accurate lip sync.`\n- **Voice + face from one clip (the talking-head recipe):** ONE unedited 2–15s clip of the person speaking (clear voice, no music, no cuts) as the video reference + prompt with the new script → their likeness AND voice deliver the new line.\n<!-- slates-only -->\n- **Billing:** a reference VIDEO switches the cost key to `seedance-2*-vref-{res}-{T}s` where T = clip seconds + output seconds — quote before confirming. Audio references are free (audio is included on every route).\n<!-- /slates-only -->\n\n<!-- slates-only -->\n## Faces — set `seedanceFace` for AI-character faces\n\nSeedance routes through **three tiers** depending on the face in the reference, exposed as the \"Face in Reference\" toggle plus the real-face params on `slates_generate_video`:\n\n- **Faceless / object / environment refs → default route (cheapest).** Leave `seedanceFace` off.\n- **An AI-character's FACE in a reference → `seedanceFace: true`.** The default route's baseline moderation rejects or degrades faces, so this reroutes to the face-capable provider. It costs **~45% more** — the cost key becomes `seedance-2-face-{res}-{N}s`, so the pre-flight quote already reflects it. Announce the face-route price, not the faceless one.\n- **A REAL person's photo (the user themselves, an actor) → the consent-gated real-person route.** If a `seedanceFace` gen fails with `[REAL_FACE_DETECTED]`, the provider classified the reference as a real person: confirm with the user that (a) they hold the rights/consent to the likeness and (b) they accept the higher price (cost key `seedance-2-realface-{res}-{N}s`, roughly 2× the AI-face rate — quote via `slates_estimate_generation_cost`), then retry with `seedanceRealFace: true` + `realFaceConsent: true`. Never set `realFaceConsent` without the user's explicit confirmation.\n\nRules:\n- **The real-vs-AI call is the PROVIDER'S, not yours.** ByteDance's classifier is probabilistic — some real photos pass the standard face route (billed at the cheap rate; fine), others get rejected with `[REAL_FACE_DETECTED]` (auto-refunded). Don't preemptively route to the real-face tier just because a photo looks real; try `seedanceFace: true` first and escalate only on the marked rejection. Public figures / celebrities fail on every route.\n- It's about the **reference, not the output.** If your character identity or generated portrait shows a face, turn it on. A product shot with no person stays off.\n- Don't toggle it on \"just in case\" — a faceless gen on the face route burns ~45% extra for nothing.\n<!-- /slates-only -->\n\n## Reference rules (the verified ones)\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n<!-- @inject:reference-rules-core -->\nIdentity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.\n<!-- @end:reference-rules-core -->\n\n### For Seedance specifically\n\n- **Describe the ACTION, never the reference's content.** With refs attached, prompt only what is *happening* — motion, change, camera. Never re-describe what's in the reference, and never say \"still / scene / from a movie / from the image.\" The model already sees the refs; narrating them wastes tokens and induces drift. Injection is stochastic — if a roll misses, **re-roll, don't re-engineer** (and a slow gen is not a failed one<!-- slates-only --> — see slates-cost-discipline<!-- /slates-only -->).\n- **Seedance's own idiom for rule 2 is `Reference <Subject_N> in <Image_N>`** `[official :1389]` — `Image_N` indexes the order the refs are attached, so the name plus the index carries the role. The full binding grammar is in Part 1 (Subject binding).\n- **Rule 3 has an official ceiling here.** The trend is MORE references (video and audio into Seedance), all addressed by name — but for **multi-character frames** see the twin-problem section above: bind every character to its image, append the anti-twin constraint, and prefer single-person references. Past 4 reference people, stability drops `[official :2048-2052]`.\n- **Rule 8 holds even though Seedance can render common text natively** `[official :1758]`. A baked NB2 start frame is still the reliable route for text that must be legible.\n- **Rule 5 pairs with the first/last-frame exclusion** — frames and reference images are mutually exclusive on this model (see Reference media above), so an environment you must match exactly costs you the frame lane.\n\n<!-- slates-only -->\n## Pre-flight: references arrive inline, refer by code\n\nWhen you call `slates_generate_video` with reference asset IDs (firstFrameAssetId, lastFrameAssetId, ingredientAssetIds), the first call returns those references **inline as image content blocks** alongside a cost estimate and `requires_confirm: true`. **Look at the references** — if they suggest a different framing, lighting, or motion than your current prompt captures, revise the prompt before re-calling with `confirm=true`.\n\nWhen talking to the user about the gen, refer to each reference by its short code: `IMG-A12 — Beach Sunset`. The user sees that code as a badge on the gallery thumbnail, so they can match what you're saying to what they're looking at.\n\n- ✅ \"I'm using **IMG-A12** as the first frame and **IMG-A15** as the last frame — the camera move is going to be a slow dolly forward through the gap.\"\n- ❌ \"I'm using the first beach image and the last one...\" (which? They have four.)\n<!-- /slates-only -->\n\n---\n\n# Part 3 — Community field notes `[community]`\n\nThird-party guides and Slates field experience. Useful heuristics — but if one of these ever appears to contradict Part 1, **Part 1 wins**.\n\n## Length\n\n**Sweet spot 60-150 words** for a single shot (not 150-300 — that's the upper bound). Multi-shot storyboards run longer; official Example 1 above is ~230 words across three shots.\n\n## Pin the subject in the first 20-30 words\n\nThe opening sentence is the **identity anchor**. If the subject isn't locked early, the model hallucinates new subjects mid-clip. (Compatible with Part 1: the binding preamble comes before `Shot 1`.)\n\n```\nA matte black earbud case sits on a polished obsidian surface...\n```\n\n## Lighting is a top quality lever\n\nLighting has an outsized impact on output quality — which is why it has its own slot in the official 8-slot formula. Describe it before or alongside the subject.\n\n```\nA cool-white diagonal beam from upper left, dust particles drifting through.\nSoft golden hour lighting from low west angle.\nDramatic rim light against dark background.\n```\n\n## Camera and subject motion — separate sentences\n\nMixing them is a common cause of glitchy / shaky output.\n\n❌ \"The camera speed ramps as the earbud rises.\"\n✅ \"The earbud rises smoothly. The camera tracks upward.\"\n\n## Slow-motion works; \"fast\" is a known bad token\n\nSpeed ramps and slow-motion are supported in natural language, and `fast` is widely reported as a quality-degrading keyword. **The official version of this rule is stronger and better founded** — prioritize slow, gentle, continuous small movements and avoid high-burst action (Part 1, Action description `[:1611-1615]`). Prompt the motion class, not the adjective.\n\n```\nthe lid opens in slow-motion · the blade whips through the air\n```\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ — same contract as the anti-list in slates-prompting-nano-banana-2.\n Extracted by src/prompts/banned-tokens.ts into the slates_generate_video op\n description and matched against submitted prompts. The RECOMMENDED vocabulary\n below sits OUTSIDE the markers on purpose — it is backticked too. -->\n<!-- /slates-only -->\n**Slop tokens to avoid:** `epic`, `amazing`, `beautiful`, `lots of movement`, `8K`, `masterpiece`, `trending on artstation`.\n<!-- @banned:end -->\nThese are quality *incantations* — the officially sanctioned way to ask for quality is the image-quality slot vocabulary in Part 1 (`HD`, `rich details`, `cinematic texture`, `natural colors`, `soft lighting`).\n\n## Style block at the end\n\nOne primary anchor + 2-3 supporting details, as the trailing paragraph (both official examples do exactly this). End with `Single continuous take` if you want one shot with no cuts. **Never** write `no cut` or `seamless transition` — not in the training vocabulary.\n\n## ⚠️ Don't cross-pollinate image-model syntax\n\n<!-- @inject:lens-video-split -->\nNamed lenses, apertures, film stocks and camera bodies (`85mm f/1.4`, `Kodak Portra 400`, `ARRI Alexa 65`) are an image-model lever. On a video model, translate the look instead of pasting the gear list: `85mm f/1.4, Portra 400` becomes `close-up, shallow depth of field, warm natural colors, cinematic texture, film-grain texture`. ByteDance's Seedance 2.0 guide never mentions fps, shutter angle, f-stop or lens millimetres. Its Seedance 2.5 guide does, once: the visual-style line of its own storyboard example names one camera body and one 35 mm cinema lens. On 2.5 a single line like that is vendor-sanctioned; a stacked gear list still is not.\n<!-- @end:lens-video-split -->\n\n## Negative prompting — inline only\n\nSeedance has **no `negativePrompt` field**. Put negatives in the constraints slot, led by the three official templates (Part 1):\n\n```\nkeep it subtitle-free · do not generate a logo · do not generate a watermark\navoid jitter and bent limbs\navoid temporal flicker\navoid identity drift\nno distortion, no stretching\n```\n\nAlso fine: positive reframing (\"empty street\" not \"no cars\").\n\n## Image-to-video / first-frame guidance\n\n**Describe motion, not image.** The model already sees the visual; tokens spent re-describing appearance are wasted.\n\nStability phrases that help:\n- `preserve composition and colors`\n- `maintain exact appearance from reference image`\n- `consistent character throughout, no deformation or drift`\n\n**Cap I2V prompts under 60 words** when possible. Over 100 words frequently triggers silent generation failure.\n\n## Common failure modes + fixes\n\n| Failure | Fix |\n|---|---|\n| Hallucinated subject mid-clip | First 20-30 words = identity anchor |\n| Bent limbs / extra fingers | `avoid jitter and bent limbs` in Constraints |\n| Identity drift across multi-shot | Re-name the bound subject in **every** `Shot N` block `[official :1537]` |\n| Two identical characters in one frame | The twin fix in Part 1 — bind each character to its image + append the global anti-twin constraint |\n| Silent generation failure on I2V | Cut prompt under 100 words, single primary camera move |\n| Speech / motion conflict | Limit dialogue to one line per action shot |\n| Erratic/random pacing | You second-stamped. Remove all time markers and use `Shot N` `[official :1586]` |\n\n## Sources\n\n**Official (authoritative):**\n- BytePlus ModelArk — Seedance 2.0 prompting guide, archived at `research/byteplus-seedance-2-0-api-docs.md` (all `:NNNN` refs above)\n\n**Community (secondary):**\n- [fal.ai — How to Use Seedance 2.0](https://fal.ai/learn/tools/how-to-use-seedance-2-0)\n- [apiyi.com — Seedance 2.0 Prompt Guide](https://help.apiyi.com/en/seedance-2-0-prompt-guide-video-generation-camera-style-tips-en.html)\n- [atlabs.ai — Ultimate Seedance 2.0 Prompting Guide](https://www.atlabs.ai/blog/the-ultimate-seedance-2.0-prompting-guide-47-prompts-2026)\n",
|
|
32
|
-
"slates-prompting-seedream-5-lite": "---\nname: slates-prompting-seedream-5-lite\ndescription:
|
|
33
|
-
"slates-prompting-veo-3": "---\nname: slates-prompting-veo-3\ndescription: How to prompt Veo 3.1 (Google). Read before calling slates_generate_video with veo-3.1-fast or veo-3.1-standard. Veo is a NICHE pick, never the default (route per slates-model-selection — Kling is the general default, Seedance the premium tier) — reach for it only when native synchronized audio must generate WITH the video in one gen. 16:9 or 9:16, 4/6/8s. Different cinematography formula than Seedance/Kling. (no subtitles) is mandatory after every dialogue line.\n---\n\n# Veo 3.1 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Veo 3.1.** Google's formula, in order: `[Cinematography] + [Subject] + [Action] + [Context] + [Style & Ambiance]`. Sweet spot 50-150 words; the official benchmark is about 50.\n\n**The five levers**\n1. **Open with the cinematography** — `Medium shot`, `low angle`, `aerial`, `dolly in`, `rack focus`, `vertigo effect`. Veo reads the first clause as the camera.\n2. **Texture words counter the AI-plastic look** — `fine skin pores`, `visible fabric weave`, `subtle contrast, no gloss or sharpening`. Name materials concretely: `charcoal cotton hoodie`, `matte concrete`, `silk lapel`.\n3. **Weight verbs stop floaty motion** — `trudges`, `drops heavily`, and a ground contact: `boots crunch on gravel`.\n4. **Terse voice direction only** — `says in a weary voice`, `whispers`, `mutters`. Veo is far less responsive to long voice blocks than Kling.\n5. **Always include an ambience line.** Without one the mix feels dead. `Soft office ambience.` `Wind on the open ridge.` And SFX always carries a cause: `SFX: thunder cracks in the distance`, never `SFX: thunder`.\n\n**Examples**\n- `Medium shot, a tired founder rubbing her temples in front of a bulky monitor in a cluttered office late at night. Harsh fluorescent overheads and the green glow of the screen. Fine skin pores, visible fabric weave on a charcoal cotton hoodie. Soft office ambience, a fan hum. Retro, slightly grainy.`\n- `Low angle, a farrier trudges across a wet yard carrying a shoeing box, boots crunching on gravel. Overcast north light, matte concrete and wet steel. Wind and distant livestock. He says in a weary voice, \"One more and we're done.\" (no subtitles).`\n\n**Hard constraint:** `(no subtitles)` after EVERY dialogue line you do not want burned in as text. Negatives are NOUNS, not instructions — `wall, frame`, never `no walls`. And do not cross syntaxes: Seedance's `single continuous take` suppresses Veo's cuts, and Veo timestamps in a Seedance prompt cause drift.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** (each one is spelled out above):\n- `single continuous take` — Seedance's phrase; in a Veo prompt it suppresses the cuts you asked for\n- `no subtitles` is REQUIRED after dialogue, but instruction-shaped negatives are not: `no walls`, `no man-made structures`, `don't show` — negatives are NOUNS here\n- `SFX: thunder` and any label-only effect — every effect carries a cause and a distance\n<!-- @banned:end -->\n\nGoogle DeepMind's video model. Two tiers: `veo-3.1-fast` (cheaper, quick) and `veo-3.1-standard` (higher quality). 4k variants exist for both (4K video requires Slates Pro).\n\n**Native single-shot duration: 4, 6, or 8 seconds** — and **8s only** at 1080p or 4K, or whenever you attach reference images (that endpoint is 8s-fixed). 4s and 6s exist at 720p, text-to-video or single-start-frame only. Longer durations require chaining clips via Extend / last-frame reuse — quality degrades if naively requested past 8s in a single generation. Aspect ratio: **16:9 or 9:16** on the route Slates uses. `slates_generate_video` REFUSES anything outside these before submit and names the legal set — nothing is silently ignored or downgraded.\n\nNative synchronized audio at 48kHz: dialogue, SFX, ambient — generated WITH video, not added after.\n\n## Official Google formula\n\n```\n[Cinematography] + [Subject] + [Action] + [Context] + [Style & Ambiance]\n```\n\nSweet spot length: 50-150 words. Cloud's official benchmark is ~50 words.\n\nVerbatim official benchmark:\n> \"Medium shot, a tired corporate worker, rubbing his temples in exhaustion, in front of a bulky 1980s computer in a cluttered office late at night. The scene is lit by the harsh fluorescent overhead lights and the green glow of the monochrome monitor. Retro aesthetic, shot as if on 1980s color film, slightly grainy.\"\n\n## Cinematography vocabulary (Vertex AI docs)\n\n**Lenses:** wide-angle, telephoto, fisheye, anamorphic, 35mm, 85mm, shallow/deep depth of field\n\n**Lighting:** Rembrandt lighting, volumetric lighting, backlighting, golden hour glow, lens flare, rack focus, **vertigo effect** (dolly zoom)\n\n**Camera moves:** dolly (in/out), truck (left/right), pan, tilt, crane, aerial/drone, handheld, whip pan, arc shot, zoom\n\n## Texture-realism phrases (counter the AI-plastic look)\n\n```\nfine skin pores · visible fabric weave · subtle contrast, no gloss or sharpening\n```\n\nSpecify materials concretely: `charcoal cotton hoodie`, `matte concrete`, `silk lapel`. Generic \"smooth, beautiful\" rendering is the failure mode you're avoiding.\n\n## Dialogue — `(no subtitles)` is mandatory\n\nEvery dialogue line you don't want burned in as text overlay needs `(no subtitles)`. Verbatim from the founder talking-head benchmark:\n\n```\nThe founder says, \"This update cuts setup time in half, helping teams get started faster.\" (no subtitles).\n```\n\nWithout this, Veo will overlay subtitle text on top of your generation.\n\n## Voice direction — keep it terse\n\nVeo is less responsive to long voice-direction blocks than Kling. Use brief modifiers:\n\n```\nsays in a weary voice\nwhispers\nshouts\nmutters\n```\n\nMulti-character: handles 2-3 speakers natively. Past 3, sync degrades — use first-frame/last-frame chaining for 4+.\n\n## SFX with cause\n\n```\n✅ SFX: thunder cracks in the distance\n❌ SFX: thunder\n```\n\nAlways specify direction or distance.\n\n## Ambient is mandatory\n\nAlways include an ambience line per scene. Without it, the audio mix feels dead.\n\n```\nSoft office ambience.\nWind on the open ridge.\nDistant city hum.\n```\n\n## First-frame + last-frame workflow (Veo's strength)\n\n1. Generate start frame (Gemini 2.5 Flash Image is the recommended pair — Slates' Nano Banana 2 works)\n2. Generate end frame\n3. Animate with both frames as anchors\n\n**Motion-Lock hack:** Keep ~60% of the same background pixels between start and end frames. Prevents latent drift across the clip.\n\nVerbatim arc-shot example:\n> \"The camera performs a smooth 180-degree arc shot, starting with the front-facing view of the singer and circling around her to seamlessly end on the POV shot from behind her on stage. The singer sings 'when you look me in the eyes, I can see a million stars.'\"\n\n## Ingredients-to-Video (multiple references)\n\nVerbatim example:\n> \"Using the provided images for the detective, the woman, and the office setting, create a medium shot of the detective behind his desk. He looks up at the woman and says in a weary voice, 'Of all the offices in this town, you had to walk into mine.'\"\n\n## Reference discipline (character / environment refs)\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n<!-- @inject:reference-rules-core -->\nIdentity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.\n<!-- @end:reference-rules-core -->\n\n### For Veo specifically\n\n- **Veo's idiom for rule 2 is plain-English role naming in the sentence itself** — *\"Using the provided images for the detective, the woman, and the office setting, create a medium shot of…\"* (see Ingredients-to-Video above). The role rides in the noun phrase, not in a separate label block.\n- **Rule 8 has a second reason to matter here:** Veo bakes subtitle text into the frame unless every dialogue line carries `(no subtitles)`. Text you did not ask for is the failure mode, not just text you did.\n\n## Negative prompting — nouns, not instructions\n\nVeo has a `negativePrompt` field. **Verbatim Vertex AI rule:**\n> \"Describe unwanted elements as nouns rather than instructions. Use 'wall, frame' instead of 'no walls' or 'don't show walls.'\"\n\nInline: positive reframing in the body too.\n- ✅ `\"a desolate landscape with no buildings or roads\"`\n- ❌ `\"no man-made structures\"`\n\n## Common failure modes + fixes\n\n| Failure | Fix |\n|---|---|\n| Subject identity shifts mid-clip | Front-load identity at prompt start; use material cues (`charcoal canvas`, `cotton`, `silk`) to stabilize |\n| Floaty / weightless motion | Weight verbs (`trudges`, `drops heavily`), ground contact (`boots crunch on gravel`) |\n| AI-plastic look | `fine skin pores`, `visible fabric weave`, `subtle contrast` |\n| Subtitles baked into video | `(no subtitles)` after every dialogue line |\n| Rushed dialogue | Lines fit one natural breath in 8s |\n| Mismatched ambience | Always include an ambience line |\n| Warped geometry | `photorealistic stability` |\n\n## Timestamp shot syntax (for chained / multi-beat scenes)\n\nVeo accepts `[00:00-00:02]` brackets for timed sequences within an 8s clip. **Do NOT cross syntaxes** — Veo timestamps in a Seedance prompt cause subject drift; Seedance \"single continuous take\" in a Veo prompt suppresses cuts.\n\nVerbatim multi-beat:\n> \"[00:00-00:02] Medium shot from behind a young female explorer with a leather satchel and messy brown hair in a ponytail, as she pushes aside a large jungle vine to reveal a hidden path.\n> [00:02-00:04] Reverse shot of the explorer's freckled face, her expression filled with awe as she gazes upon ancient, moss-covered ruins. SFX: The rustle of dense leaves, distant exotic bird calls.\n> [00:04-00:06] Tracking shot following the explorer as she steps into the clearing and runs her hand over the intricate carvings on a crumbling stone wall.\n> [00:06-00:08] Wide, high-angle crane shot, revealing the lone explorer standing small in the center of the vast, forgotten temple complex, half-swallowed by the jungle. SFX: A swelling, gentle orchestral score begins to play.\"\n\n## Benchmark prompt — founder talking head (full)\n\n> \"Camera locked at eye level, medium close-up on a 35mm lens: a startup founder in his late 30s with short black hair and light stubble, wearing a charcoal cotton hoodie, speaking directly to camera, leaning slightly forward as he speaks, lifting one hand to emphasize a point, then relaxing back to neutral, in a quiet office during late afternoon, with blurred monitors glowing faintly in the background, lit by soft daylight from a side window with gentle fill on the opposite side and natural falloff across his face. Style: fine skin pores, visible fabric weave, subtle contrast, no gloss or sharpening. Audio: The founder says, 'This update cuts setup time in half, helping teams get started faster.' (no subtitles). Soft office ambience.\"\n\n## Pre-flight: references arrive inline, refer by code\n\nWhen you call `slates_generate_video` with `firstFrameAssetId` / `lastFrameAssetId` / `ingredientAssetIds`, the first call returns those references **inline as image content blocks** alongside a cost estimate and `requires_confirm: true`. Veo's strongest move is first-frame + last-frame; the pre-flight is where you confirm the two frames actually anchor the motion you wrote. Revise the prompt before `confirm=true` if needed.\n\nWhen talking to the user about the gen, refer to each reference by its short code: `IMG-A12 — Founder Headshot`. The user sees that code as a badge on the gallery thumbnail, so they can match what you're saying to what they're looking at.\n\n- ✅ \"I'm anchoring on **IMG-A12** as the open shot and **IMG-A18** as the close — the 180° arc lands on her looking offscreen left.\"\n- ❌ \"I'm using two of the founder shots...\" (which two? They have six.)\n\n## Sources\n\n- [Google Cloud — Ultimate Prompting Guide for Veo 3.1](https://cloud.google.com/blog/products/ai-machine-learning/ultimate-prompting-guide-for-veo-3-1)\n- [Google DeepMind — Veo Prompt Guide](https://deepmind.google/models/veo/prompt-guide/)\n- [Google Cloud Docs — Vertex AI Video Generation Prompt Guide](https://docs.cloud.google.com/vertex-ai/generative-ai/docs/video/video-gen-prompt-guide)\n- [Atlas Cloud — Veo 3.1 Master Guide](https://www.atlascloud.ai/blog/guides/google-veo-3-1-guide-master-image-to-video-ai-with-native-sound-and-4k-realism)\n- [Invideo — Veo 3.1 Prompt Guide](https://invideo.io/blog/google-veo-prompt-guide/)\n",
|
|
34
|
-
"slates-
|
|
35
|
-
"slates-
|
|
36
|
-
"slates-
|
|
37
|
-
"slates-
|
|
38
|
-
"slates-
|
|
39
|
-
"slates-ugc-influencer-ad": "---\nname: slates-ugc-influencer-ad\ndescription: Direct a creator-style spoken performance when the brief calls for an ordinary camera-facing person or exchange. Use for performance and phone-camera craft, not as a universal rule for ads.\n---\n\n# Creator-style performance\n\nRead `slates-script-craft` for the script and opening/bridge versions. Match the requested creator, audience and reference register. Phone footage, quiet polish and cinematic treatment are choices; no evidence here establishes one as universally highest-converting.\n\n## Direct a person in a place\n\nWrite an observable activity and a speaking intention: showing a worn handle, answering a friend, demonstrating a catch, interrupting a task. Small actions can make a close shot legible: a weight shift, a glance toward the other person, a hand taking an object's weight. Larger actions need space and framing that accommodates them.\n\nFor an ordinary phone register, describe concrete light and surroundings: a window from one side, a practical lamp, a room with ordinary possessions. Avoid generic praise words. Preserve the supplied person's appearance and product details through references, not invented claims.\n\nDescribe camera handling physically when it matters: a hand supports the phone against the table edge; the frame sags and is corrected once. Static framing is also valid. Do not force handheld motion or imperfections into a brief that asks for something else.\n\n## Speech, interaction and sound\n\nOne person or several may carry the piece. Give reactions and answers their antecedents. A continuing sentence may cross an edit; an independent module should establish its own subject. Repetition can be a callback.\n\nSpeech may point at what the viewer sees when demonstrating a claim. It need not compete with the picture for novelty on every line. Keep pauses, listening and the intended register; a fixed words-per-second formula cannot establish the performance's duration.\n\nMake audio intent explicit: dialogue, room sound, effects, score or silence. A silent demo is valid. Use current model guidance for audio support and reference syntax. Plan captions or text only where the actual editing surface supports them; do not promise an unverified rendering feature.\n\n## References and production choices\n\nSelect the identity, voice, location and direct images deliberately. A replacement presenter need not replace a separately retained voice. A changed location does not automatically replace a first-frame image. Inspect the composed request to see which references remain.\n\nA plate-first approach is useful when the composition must be approved before motion, but it is not a prerequisite for all video. Choose single-take or multiple-cut production according to the intended performance and current capabilities. Do not copy vendor duration, resolution or reference limits into this guide.\n\nBefore spending, follow `slates-cost-discipline` for the selected set. Inspect errors and existing job state; a refusal or timeout does not authorize another charge. Check the actual clip's picture and sound against the brief, preserve earlier takes, and trim or rearrange when that solves the issue without regenerating. Exported playback, not a successful process exit, establishes the result.\n",
|
|
40
|
-
"slates-vision-feedback-loop": "---\nname: slates-vision-feedback-loop\ndescription: Lower-level utility skill for any Slates workflow that needs to \"generate, look at the result, refine, regenerate.\" Defines the standard inline-vision pattern. Other Slates skills compose this. Use when generating images and you need to confirm they match the brief before moving on, or when the user asks to \"iterate\" on an image.\n---\n\n# Vision feedback loop — Slates utility skill\n\nSlates returns generated images inline as base64. You see the actual pixels. Use that — don't trust prompt-following blindly.\n\n## Asset codes are your shared vocabulary with the user\n\nEvery asset in Slates has a short stable code (e.g. `IMG-A12`, `VID-V3`, `AUD-S1`) and a label derived from its prompt (e.g. `Beach Sunset`). These are visible in the gallery as a corner badge on each thumbnail. **Always refer to assets by their code in chat** so the user can match what you're saying to a specific card in their gallery.\n\n- ✅ \"I'm using **IMG-A12 — Beach Sunset** as the first frame. The second-frame candidate **IMG-A15** has the right composition but warmer light — want me to use that one instead?\"\n- ❌ \"I'm using the beach sunset image...\" (user has four beach sunset variants — which one?)\n- ❌ \"I'm using asset `7a3f9e4b-...`\" (UUIDs aren't readable; user can't match to a badge)\n\nThe code is the FORMAL reference. The label is human texture. Use both: `IMG-A12 — Beach Sunset`.\n\n## Vision tools at your disposal\n\n- `slates_get_asset_image` — pull one image into context. Returns its code+label.\n- `slates_get_assets_batch` — pull up to 8 images in one call. Use when picking from a candidate set; cheaper than N individual fetches.\n- `slates_get_asset_video_frames` — extract N keyframes (default 3) from a video and inline them as JPEGs. You can't see video natively; this is how you \"look at\" a clip before refining its motion prompt.\n\n## Pre-flight is automatic on the gen tools\n\n`slates_generate_video`, `slates_generate_motion_transfer`, and `slates_generate_lip_sync` now show you their reference assets **inline** on the confirm response. You don't need to fetch them yourself — but you DO need to look at what comes back, revise the prompt if the references suggest a different motion/framing, and only then re-call with `confirm=true`.\n\n## 🔴 The still-gate — never animate a bad frame\n\n<!-- @inject:still-gate -->\n**A visible defect in the still is already a STOP.** Do not animate it. Fix the frame first, then move to motion — and go to motion only when the crop passes the still scan and you genuinely need movement to confirm an uncertain edge, reflection, or object.\n\nThis is a **cost** rule as much as a craft rule: a 1080p/10s premium video generation costs many multiples of an image re-roll, and video is where a defect stops being fixable. Anything wrong in the still gets worse in motion — soft geometry mushes, broken-but-plausible objects fall apart, oily textures start crawling. **Animating a known-bad frame is the single most expensive mistake in the pipeline.** Re-rolling the image is the cheap move; re-rolling the video is not.\n<!-- @end:still-gate -->\n\n## The pattern\n\n1. **Generate.** Call `slates_generate_image` with a prompt. The result is in your context as an image content block.\n2. **Evaluate on TWO axes — they are different questions:**\n - **Brief-conformance** — what did the user actually want? Are the elements right? Composition? Lighting? Subject identity?\n - **Defects** — run the slop rubric below. *A frame can match the brief perfectly and still be slop that mushes the moment it moves.* Checking only the first axis is how a bad frame reaches an expensive video call.\n3. **One of three outcomes:**\n - **Right** → save it (bind to a frame, character slot, etc.) and move on.\n - **Close, but adjustable** → refine with a specific delta, regenerate **once**.\n - **Wrong direction** → ask the user before regenerating. Don't burn credits on prompt-thrashing.\n\n## The defect rubric — five slop tells\n\n| Tell | What it looks like | Why it matters downstream |\n|---|---|---|\n| **Light with no transitions** | Flat-black pits instead of a shadow ramp; light that stops rather than falls off | Transfers onto every character or object added into that plate later |\n| **Broken-but-plausible objects** | Crates, railings, hardware, mechanisms you can *almost* read but that don't resolve | Turn to mush in motion, and the model multiplies them |\n| **Local logic breaks** | An effect present in only part of the frame — rain scratching one corner, wet ground under one figure | The video model's physical logic breaks along with it |\n| **Oily textures** | Soapy, licked-smooth surfaces that have lost their material identity | Reflections crawl in motion; the plate can't hold continuity |\n| **Too perfect** | A soft light on the face that nothing in the scene could cast, the subject sharper and cleaner than everything around them, every region exposed to be readable, colour pushed warm and saturated | It reads as a subject pasted onto a location, and every shot built from the plate inherits the studio look. Fix it in words: `slates-cinematic-look` |\n\n### Per-model accents — check the one you actually used\n\n- **Nano Banana Pro** (`nano-banana-pro`) — ruler-straight symmetry, everything parallel and square, flat even light, pretty but staged/stock, textures reading as 3D render rather than photograph. **It hyperbolizes every edit**: ask for graffiti on one wall and the whole location gets tagged.\n- **GPT Image** (`gpt-image-2-5-flare`, `gpt-image-2-5-sunburst`) — microcontrast to the ceiling, hard halos on every edge, no depth or bokeh, white balance pulled warm until the frame yellows, plastic licked-smooth materials. Worst tell: **one sickly texture pattern laid over the entire frame**. ⚠️ Catalogued on `gpt-image-2`, which 2.5 replaced on 2026-09-09 — an accent is a per-model observation, so treat this as a prior to check rather than a finding, and correct it here the first time a 2.5 frame disagrees.\n\n> ⚠️ These are accents for **`nano-banana-pro`** and the **GPT Image** line specifically. `nano-banana-2` is a **different model** (Gemini 3.1 Flash Image vs NB Pro's Gemini 3 Pro Image) and we have **no evidence** about its accent. Do not inherit one — say nothing rather than warn about a failure mode you can't substantiate. That caution applies to the GPT Image entry above too: it was measured on `gpt-image-2`, not on either 2.5 seat.\n\n## Where the fault lives — triage before you change anything\n\nWe say \"one specific delta per regeneration\" but that only helps once you know *which* variable to move. Diagnose first:\n\n| Visible pattern | Diagnosis | Fix |\n|---|---|---|\n| The defect exists in the source asset, or stays tied to the same feature when the direction changes | **Source asset** | Fix the sheet / plate, not the prompt |\n| Source is clean, and the defect changes when only the suspect motion clause changes | **Motion direction** | Fix the prompt |\n| Controls conflict, or the failure follows neither variable | **Inconclusive** | Narrow the test — change less, not more |\n\n**Review routes; it is not pass/fail.** Geography melts → fix the location. Identity drifts → fix the character sheet. Assets are sound but the action is wrong → fix the video direction. Wrong idea entirely → reopen the brief with the user.\n\n**Correct the earliest broken handoff.** Polishing a downstream symptom hides the source and guarantees it resurfaces in the next shot built from the same asset.\n\n## Baseline hygiene — isolate the variable you're testing\n\nWhen the **character** is the question, keep the location out of it: test on a plate that already holds its own geometry, depth, materials, and light. **A broken plate gives every character failure a second plausible cause**, and you will spend re-rolls deciding which one you're looking at. The same applies in reverse — test a plate empty before you populate it.\n\n## Refinement rules\n\n- **One specific delta per regeneration.** Don't change five things at once — you won't know what helped.\n- **Rewrite the FULL prompt on every iteration — never a diff, never a fragment.** Change one decision, then re-emit the whole prompt so every slot still agrees with every other slot. This composes with the rule above rather than replacing it: *one delta* governs **what changes**, *full rewrite* governs **how you re-emit it**. A patched fragment leaves the old slots stale and silently contradicting the new one.\n - On **Seedance**, a re-emit must keep the `Shot N` structure intact — see `slates-prompting-seedance`.\n - **Exception — Omni Flash Edit.** Long prompts documentedly destroy its fidelity. There the rule inverts: one short instruction plus *\"Keep everything else the same.\"*\n- **Anchor with references.** If the result drifted from the user's intent, attach the *previous best* generation as a reference image alongside the original brief.\n- **Use `slates_get_asset_image`** to pull a previously-generated image back into context if you need to compare against a fresh generation.\n- **Use `slates_edit_image`** for surgical tweaks instead of full regeneration when ~90% of the image is right — `sourceAssetId` = the asset, `prompt` = the change only. Edits preserve composition and identity; full regen rolls the dice. Recipe: `slates-edit-and-iterate`.\n\n## Cost discipline\n\n- Track total credits spent across the loop. Surface to the user every 3 iterations.\n- Stop after 3 failed iterations on the same prompt — escalate to the user with what you tried and what's not working. The slot machine never converges.\n- For a high-cost generation — anything past the confirm gate in `slates-cost-discipline` — confirm before *every* attempt, not just the first.\n\n## When to break the loop\n\n- The user said \"good enough\" or \"ship it.\" Stop iterating.\n- You've burned >5 generations on one frame. Hand back and ask.\n- The user changes brief mid-loop. Treat it as a new brief, not a continuation.\n\n## Voice when narrating to the user\n\nTight, observational, no editorializing.\n- ✅ \"Frame 2 has the wrong lighting direction — back-lit instead of side. Regenerating with side light.\"\n- ❌ \"I notice that the lighting in frame 2 isn't quite what we were going for. I'll go ahead and try again with a different approach.\"\n",
|
|
4
|
+
"slates-blocking-to-prompt": "---\nname: slates-blocking-to-prompt\ndescription: \"Translate a rendered blocking clip into a video prompt that preserves its camera, cuts, timing and spatial relationships. Use for previs-guided generation or when output ignores the blocking.\"\n---\n\n# Blocking → prompt\n\nYou have a blocking clip. This is how you write the prompt that goes with it.\n\n## The one idea\n\nThe clip already contains the camera, the cuts and the timing. **The prompt's job is to say what everything looks like — and, where the grey boxes are ambiguous, to disambiguate them.** It is not a second, competing description of the motion.\n\nState that contract inside the prompt, because the model needs it as much as you do:\n\n> Where a timeline line below names a camera position or move, it is a restatement of what the reference video already does at that timestamp — a disambiguation, never a new instruction.\n\nAnd give it a tie-break, because ambiguity is guaranteed:\n\n> If any text in this prompt appears to disagree with the reference video about camera, framing, direction, motion, timing or object placement, the reference video wins.\n\nThose two sentences do more work than any other part of the prompt.\n\n## Get the real numbers first\n\n```\nslates_blender_scene\n```\n\nRead `cutSeconds` from the rendered scene, rather than using the shot list you intended to build. It resolves to markers on a multi-camera edit and camera keyframes otherwise. Preserve the exact frame boundaries in the blocking and edit record: at 24fps they can be `7.79s`, `9.33s` or `19.875s`.\n\nTranslate those measurements into the selected model's timing grammar. Seedance 2.5 accepts whole-second timestamps; use those for its prompt while the reference clip carries the exact cuts. Seedance 2.0 uses shot numbers instead. The fractional examples below describe the measured blocking; they are not a universal request syntax. If frame-exact output is a delivery requirement, inspect the result and finish the timing in the edit rather than promising the model reproduces every frame.\n\n## Structure\n\nOrder matters — contract, then globals, then timeline, then the re-assertion.\n\n```\nTITLE — one line: what this is, how long, that it is video-to-video\n\nLOGLINE — 2-4 sentences. The whole piece in plain language.\n\nACTIVE REFERENCES\n <one entry per reference: what it defines, and what is NOT inherited>\n\nTECHNICAL BLOCK (format, grade, lens, and the blanket negatives — see below)\n\nSTYLE / LOOK\nLIGHTING\nCOLOR\nCAMERA\nPHYSICS (only if things move, collide or deform)\n\nRULES (numbered — the invariants, see below)\n\nACTION TIMING (beat by beat, against the clip's real timestamps)\n\nAUDIO\n [Sound design] [Timed accents] [Dialogue] [Music]\n\nENDING LOCK (one line — where the film stops)\n\nHOLD FOR THE FULL TIMELINE\n <the 5-6 constraints most likely to drift, compressed>\n```\n\n## ACTIVE REFERENCES — every entry has an exclusion\n\nThe single highest-leverage format in this whole workflow. Each reference is a **positive claim plus an exclusion list**, because a reference the model over-reads is as damaging as one it ignores.\n\nLabel each by the badge code Slates echoes back (`IMG-A8`, `VID-V2`) or by an unmistakable role name, and use that same label everywhere below.\n\n**The blocking clip:**\n\n> VID-V2 = the blocking previz (30s, 720 frames, 24fps), the MASTER for everything that moves and everything that stands. It defines the full edit one-to-one: every cut point, every camera position, angle, move and framing, all action timing, screen direction, and the geometry of the world. Its untextured grey surfaces, flat colours and viewport grid are NOT inherited; every grey proxy is dressed into a real object in the exact position the previz puts it. Proxies give position, angle, scale and motion only; never surface, shape detail or design.\n\nThat last sentence is the **placement-only clause** and it is not optional. Without it the model renders grey boxes.\n\n**A character sheet:**\n\n> IMG-A8 = the driver — defines his face, hair and wardrobe. Identity 100% consistent at every distance and through every motion blur. Background, lighting and pose are NOT inherited.\n\n**A location/style reference:**\n\n> IMG-A3 = the tunnel, defines location geometry, look and grade. Camera angle, framing and any people in it are NOT inherited; the camera comes exclusively from VID-V2.\n\n**An atmosphere or style master — a reference that is never a shot:**\n\n> IMG-A9 = ATMOSPHERE MASTER, NOT a keyframe, NOT a location to reproduce, NOT a frame that ever appears in the film: its own subject, framing and composition are never seen in any shot. It defines ONLY the weather, light, colour and grade: deep clean night just after rain, wet asphalt as a dark mirror, cool white-cyan lamps as the ambient key, teal-and-amber grade, deep clean blacks. Every shot is lit and graded in this regime for all 30 seconds.\n\nWithout those three NOTs the model reproduces the reference's composition as an actual shot — you get its street corner in your film. The same wording covers a rendering-style master; see `slates-restyle-from-blocking`.\n\n**References can be scheduled.** If something is only true for part of the timeline, say so: *the hooded panel applies only to 0–7.0s and 27.5–30s*, or per-reference: *Active for 00:03.3–00:06.7 only.* On a piece that travels through several locations, every location still carries its own window and the model stops blending two sets into one shot.\n\n## Translate the blocking's artifacts\n\nYour grey-box render contains things that are *notation*, not content. Every one needs an explicit reinterpretation or it gets rendered literally:\n\n| In the blocking | Say in the prompt |\n|---|---|\n| Colour-coded bodies | `red = the boss, green = the kid, blue = the driver` |\n| A marked face on a proxy | `RED face = the direction he faces, BLACK = his back` |\n| A checkered floor or wall | `the checkerboard is a scale reference, not a surface — it becomes <the real material>` |\n| Flat black background | `a PLACEHOLDER — replace with the location assigned below` |\n| A floor grid | `a motion-tracking aid — render as light on the surface, never as wireframe or tiles` |\n| A deliberate black gap | `CUT 7 (14.5-17.0, black gap in the reference) — <what fills it>` |\n| Frame goes dark mid-move | `the camera is passing through the ground — a doorway to the NEXT location, never back to a previous one` |\n| The source hard-resets mid-move | `each reset begins a NEW, completely different room — never a replay of one already seen` |\n| A proxy that is a PROP or VEHICLE | `the low-poly flying model in SHOT 18 is THE HELICOPTER · blocks on the rear bench are the luggage · the small dark block in his hand IS the pistol` |\n| A blocky proxy limb in a tight insert | `the blocky low-poly leg is a stand-in and must NOT be replicated — generate complete human anatomy: a real boot, a real trouser leg, a correct ankle at this exact camera angle` |\n| A stray object at the frame edge | `ignore it completely — never blend two sets into one shot` |\n\n## TECHNICAL BLOCK — format, lens, and the blanket negatives\n\nOne paragraph, before the timeline. It carries the things that are true of every frame and that no beat should have to repeat:\n\n> Cinematic, photoreal. 21:9. 30s. SFX only, no music. Kodak 500T film look, natural 35mm grain, organic colour, soft highlight roll-off, anamorphic lens character with oval bokeh and gentle barrel distortion at the edges, chromatic aberration creeping in at the frame edges, natural motion blur on every fast move, faint bloom on hot speculars. Every location well exposed — night interiors bright and readable, open shadows, no crushed blacks, no murk. NO CGI. NON-IP, no brand badges or logos anywhere, no text, no watermark.\n\nThree parts worth naming:\n\n- **Lens realism is a list, not an adjective.** Aberration, motion blur, depth of field, barrel distortion, bloom, grain. \"Cinematic\" buys you nothing; these buy you the look.\n- **Exposure needs saying on dark work.** Models crush night scenes into murk. *Night interiors bright and readable, open shadows, no crushed blacks* is what keeps a scene legible.\n- **The blanket negatives go here once** — `NON-IP`, no logos, no on-screen text, no subtitles, no watermark — rather than being scattered through the beats.\n\n## RULES — the invariants\n\nNumbered, short, absolute. These are the things that must hold in every frame, and they are where you put anything that has already gone wrong once.\n\nTwo patterns worth stealing outright:\n\n**Countable state.** Give the model arithmetic it can check itself against:\n\n> At every second: standing + fallen + on the lintel = 6. Never a seventh figure — no extras, no duplicates, no distant silhouettes, no half-bodies at frame edges.\n\n> Bodies on the ground count exactly: 0 before 12s → 1 → 2 → 3 → 4 at 12/13/14/16s → 5 at 20s → 6 at 26s. Never more.\n\n**Every mass is dressed, and nothing is invented.** The blocking is authority over what EXISTS, not just what moves — otherwise the model deletes the masses it finds boring and adds architecture you never blocked:\n\n> Every lamppost, guardrail, road and terrain mass visible in the reference exists in the output in the same place, at the same scale, in the same position in frame — the opening blocks are dark-brick warehouse facades, the roadside masses are the waterfront skyline, the finale rocks are the city's tower walls. Nothing is deleted, and no structure is invented where the reference shows none.\n\n**Anatomy is never inherited from a proxy.** Tight inserts on hands and feet are where blocking leaks straight into the render:\n\n> The reference shows only WHERE hands and feet are. In the output they are always complete human anatomy — a five-fingered gloved hand with natural knuckles, a real leg in wool trousers, a real foot in a leather shoe — never the proxy's blocky shape.\n\n**A ledger for anything that happens a countable number of times.** A ritual, a reload, a set of falls: state it as a linear sequence, each step exactly once, and close with the tally:\n\n> Strictly linear, six steps in fixed order, each happening EXACTLY ONCE and never repeating; once a step is done it is done for good, and the sequence only ever moves FORWARD, never backward. Count of weapon events in the entire video: one draw, one magazine insertion, one slide rack, one shot.\n\nWithout the ledger the model loops the most cinematic beat — it will rack the slide four times because racking looks good.\n\n**Named misreads.** When a generation gets something specifically wrong, do not rewrite the description — **name the wrong reading and kill it**:\n\n> The lamp is a man-made steel structure — NOT an animal, NOT a snake, NOT any living or organic shape.\n\n> The rear of the car and its tail lights are NOT visible in this shot.\n\nThis is the highest-value edit available after a failed roll, and it is why the prompt grows rather than changes between takes.\n\n## ACTION TIMING — the beats\n\nOne block per shot or beat. Two notations; pick one and hold it.\n\n**For a continuous take**, ranges with a camera note and a closing state audit:\n\n```\n8-12s: THE SWEEP (per VID-V2: elevated rear push, swinging to profile by 12s):\n<what happens, in prose, with sub-beats on tenths and → chaining cause to effect>\nEND 12s: bodies 1 (behind him as he steps past) · standing — four ahead, holding.\n```\n\n**For a cut edit**, numbered shots ending on their cut:\n\n```\n9.33-10.33s: SHOT 10, Interior over the centre console as in VID-V2: <what the\nframe contains>. Hard cut at 10.33s.\n```\n\nThree habits that separate a beat that works from one that does not:\n\n- **Declare the frame's contents as a closed set** when the shot is tight: *the frame holds exactly the console, the lever, his hand, and the edges of both seats.* An open description invites additions.\n- **Chain cause to effect inside one sentence** with `→`. `he overcommits a lunge → the Hero drops low and sweeps his standing leg → he hits the earth at 12s`.\n- **Pin every event to a moment**, in the finest unit the selected model accepts: whole seconds on Seedance 2.5 (`12s`, `19s`), shot numbers on 2.0. Vague beats generate vague timing. The production prompts behind this guide carried tenths (`11.7s`, `19.5s`, `22.5s`); whether 2.5 acts on the fraction is untested, and ByteDance documents integers only.\n\nDensity: roughly 60–130 words per second of screen time is what these prompts actually run at. That is much denser than a normal video prompt, and it is the point.\n\n## AUDIO\n\n`[Timed accents]` uses the same timestamps as the beats:\n\n> 3.1s tyres light up into the burnout squeal · 7.0s drift-entry screech · 10.1s hard mechanical shifter clack · 20.3s full-speed pass-by whoosh\n\n`[Dialogue]` is a closed list — count the lines, give each a window, quote it verbatim, and forbid everything else:\n\n> Exactly TWO vocal events in the entire 30 seconds, both screamed, in English, VERBATIM: 1. 17.3-18.6s \"STOOOOOP!!\" 2. 23.4-24.0s \"You crazy!\" Nothing else is ever spoken.\n\n**Then forbid the lines it will invent anyway.** A closed list is a rule; an enumerated blacklist is enforcement, and the phrases to list are the clichés the scene invites:\n\n> FORBIDDEN — she never says any of these and no one else says anything: \"they're behind us\", \"cops\", \"go go go\", \"are you crazy\", \"you're insane\", or ANY other invented phrase. All other human voice is wordless screaming or laughing.\n\n**When lines are lip-synced, give each one a timestamp** in the same list, and say that they change nothing else:\n\n> Timing: \"You wind up for this one?\" ~18.2s · \"Three full turns.\" ~19.0s · \"Company.\" ~24.3s. Every line lip-synced; the lines never change the camera.\n\nTwo rules that stop dialogue from breaking the edit:\n\n> DIALOGUE NEVER CREATES SHOTS: spoken lines happen inside the reference's takes exactly as blocked — no cutaways to a speaker, no reverse shots, no added close-ups. If a line plays while the camera is elsewhere, the line stays off-screen audio.\n\n> A line marked off-screen must STAY off-screen — never show the speaker, never move him into frame, never route the camera to him because he spoke.\n\nModel note: dialogue direction as separate layers is minimax-h3's seat. Route per `slates-model-selection` and read the model's own prompting skill before writing the audio block.\n\n## ENDING LOCK\n\nOne line, and it is the cheapest fix in the document. Models drift at the end — they hold a frame too long, add a beat after the last one, or fade somewhere the reference does not:\n\n> The film ends exactly where the reference ends: the final take runs unbroken to its last frame, and that source frame IS the final frame of the film. Nothing follows. The last five seconds follow the source exactly as strictly as the first five.\n\n## HOLD FOR THE FULL TIMELINE\n\nClose with a terminal re-assertion of only the constraints most prone to drift — five or six lines, compressed, no new information:\n\n```\nHOLD FOR THE FULL TIMELINE\n- VID-V2 camera path 1:1; any deviation = failure.\n- Six and only six figures; the count above holds at every second.\n- IMG-A8 identity constant at every distance and through motion blur.\n- The IMG-A3 location in every frame; no subtitles, no watermarks.\n```\n\nRestating is not redundancy here. It is the last thing the model reads.\n\n## Checklist before you generate\n\n- [ ] Frame-exact timings retained from `slates_blender_scene`'s `cutSeconds`; model-facing timing translated to the selected guide's syntax\n- [ ] Every reference has an explicit \"NOT inherited\"\n- [ ] The placement-only clause is present\n- [ ] The tie-break clause is present\n- [ ] The disambiguation clause is present\n- [ ] Every blocking artifact is translated — colours, marked faces, checkers, grid, black background, gaps, dark dips, resets, prop proxies, proxy limbs, strays\n- [ ] Any style/weather reference is declared NOT a keyframe and never a shot\n- [ ] The every-mass-is-dressed / nothing-invented rule is present\n- [ ] Tight inserts on hands or feet demand complete anatomy\n- [ ] Counts are stated where anything is countable, and repeatable actions carry a ledger\n- [ ] A TECHNICAL BLOCK carries format, lens realism, exposure and the blanket negatives\n- [ ] `[Dialogue]` is a closed list with a FORBIDDEN blacklist\n- [ ] An ENDING LOCK says where the film stops\n- [ ] `videoReferenceSecondsEach` matches the clip's real duration\n- [ ] A HOLD block closes it\n\n## Related\n\n`slates-previs-blocking` (producing the clip) · `slates-camera-language` (the moves being described) · `slates-dialogue-blocking` (multi-character continuity) · `slates-restyle-from-blocking` (reusing this prompt across styles) · `slates-prompting-seedance-2-5` / `slates-prompting-minimax-h3` (model-specific rules)\n",
|
|
5
|
+
"slates-camera-language": "---\nname: slates-camera-language\ndescription: \"Build or refine Blender camera rigs for a previs pass: orbits, floor rises, whip-and-lock moves, handheld, speed ramps and shot transitions. Use when the intended camera path needs deterministic control.\"\n---\n\n# Camera language — from a shot list to a rig\n\nCompanion to `slates-previs-blocking`. That skill owns the workflow; this one owns the camera.\n\n## The first rule\n\n🚨 **Never build \"a cinematic camera move.\" Brief the camera the way you would brief an operator:** rails, target, height, lens, and the frame each move starts and ends on. \"Cinematic\" is not a specification, and asking for one produces the drifting slop the whole blocking workflow exists to avoid.\n\nBad: *a dynamic cinematic orbit around the subject.*\nGood: *a 3/4 orbit starting rear-left at 1.6m, ending front-right at 0.9m, frames 1–96, 35mm, easing out of the start and holding hard on the last 8 frames.*\n\nIf the user gives you the first, convert it to the second and say what you assumed.\n\n## Look it up, don't recall it\n\nBefore any constraint or operator you are not certain of, call `slates_blender_docs` (e.g. `bpy.types.FollowPathConstraint`) or `slates_blender_search_docs`. Invented enum values are the most common failure here and they often fail *quietly* — the constraint gets added, the axis is wrong, and the camera points at nothing.\n\n## The two primitives everything is built from\n\n### Target-based aiming\n\nAlmost every move in this skill is *position on a path* plus *aim at a target*. Separating them is what lets framing vary while the subject stays in frame.\n\n```python\nimport bpy\n\ntarget = bpy.data.objects.new(\"CAM_TARGET\", None) # an Empty\ntarget.empty_display_type = 'PLAIN_AXES'\nbpy.context.collection.objects.link(target)\n\ncon = cam.constraints.new('TRACK_TO')\ncon.target = target\ncon.track_axis = 'TRACK_NEGATIVE_Z' # cameras look down -Z\ncon.up_axis = 'UP_Y'\n```\n\n**Aim at a separate target, not at the subject's head.** A camera locked to the head produces dead, centred framing. Offset the target beside or ahead of the subject and the shot breathes — that small offset is most of what reads as \"real operator.\"\n\nFor a subject that should notice the camera, keyframe the target's follow with a **few frames of lag** behind each camera move. Heads catch up; they don't teleport.\n\n### Path-based movement\n\n```python\ncurve = bpy.data.curves.new(\"CAM_PATH\", 'CURVE')\ncurve.dimensions = '3D'\nspline = curve.splines.new('BEZIER')\nspline.bezier_points.add(len(points) - 1)\nfor bp, co in zip(spline.bezier_points, points):\n bp.co = co\n bp.handle_left_type = bp.handle_right_type = 'AUTO'\n\npath = bpy.data.objects.new(\"CAM_PATH\", curve)\nbpy.context.collection.objects.link(path)\n\ncon = cam.constraints.new('FOLLOW_PATH')\ncon.target = path\ncon.use_curve_follow = False # aiming is the Track To constraint's job\n\n# Animate progress explicitly rather than relying on the default path animation.\ncurve.path_duration = 96\ncurve.eval_time = 0\ncurve.keyframe_insert(\"eval_time\", frame=1)\ncurve.eval_time = 96\ncurve.keyframe_insert(\"eval_time\", frame=96)\n```\n\n**Why a path and not raw location keys:** the user can drag a control point to retime or reshape the move without you regenerating anything. That is the difference between \"re-prompt and hope\" and \"nudge it.\"\n\n<!-- @inject:blender-action-curves -->\n## Read animation curves from the active action layout\n\nBlender 5 uses layered actions: curves belong to the channelbag for `animation_data.action_slot`, inside each layer's strips. A direct `action.fcurves` lookup failed on Blender 5.2.1 in the 2026-08-28 blocking run. Feature-detect the layout before changing interpolation or noise; an unanimated object can legitimately have no curves.\n\nThe snippets below use this small Blender-side iterator. It runs inside Blender; no add-on code is imported into the MCP package.\n\n```python\ndef action_curves(datablock):\n anim = getattr(datablock, \"animation_data\", None)\n action = getattr(anim, \"action\", None)\n if action is None:\n return\n if hasattr(action, \"fcurves\"):\n yield from action.fcurves\n elif getattr(anim, \"action_slot\", None) is not None:\n for layer in action.layers:\n for strip in layer.strips:\n if hasattr(strip, \"channelbag\"):\n bag = strip.channelbag(anim.action_slot)\n if bag is not None:\n yield from bag.fcurves\n```\n\nUse the datablock that owns the keyed property: the curve data for `eval_time`, the object for location and rotation, the camera data for lens. Confirm a named channel exists before assuming a keyframe operation created it.\n<!-- @end:blender-action-curves -->\n\n## Speed — the part that reads as production value\n\nMovement at one constant speed is the tell of a machine. Real moves accelerate, hold, and snap.\n\nSpeed lives in the **f-curve handles** of `eval_time` (or of location, if you keyed it directly):\n\n- **Long, near-horizontal handle** at a key → slow near that key.\n- **Short, steep handle** → fast.\n- `interpolation = 'CONSTANT'` → no movement at all until the next key. This is how you get an absolute dead stop.\n\n```python\n# Add the middle key to the two-key path example above.\ncurve.eval_time = 48\ncurve.keyframe_insert(\"eval_time\", frame=48)\nfc = next((f for f in action_curves(curve) if f.data_path == \"eval_time\"), None)\nassert fc is not None, \"Key eval_time before shaping its motion\"\nfor kp in fc.keyframe_points:\n kp.interpolation = 'BEZIER'\n kp.handle_left_type = kp.handle_right_type = 'FREE'\n\na, b, c = fc.keyframe_points # start, middle, end\n# Speed ramp: fast in, sag in the middle, accelerate out.\na.handle_right = (a.co.x + 2, a.co.y + 18) # steep = launches fast\nb.handle_left = (b.co.x - 14, b.co.y) # flat = holds\nb.handle_right = (b.co.x + 14, b.co.y)\nc.handle_left = (c.co.x - 2, c.co.y - 18) # steep = arrives fast\n```\n\n**Zero drift at a stop.** If a move is supposed to be locked off, it must be *actually* locked — a slow crawl at a \"stop\" reads as a mistake. Hold with `CONSTANT` interpolation, or duplicate the key so the segment is genuinely flat.\n\n## The moves\n\n### Orbit\n\nCircle the subject on a path, target at subject height. Vary radius and height across the move so it does not read as a turntable. Half-orbits and 3/4 orbits look more intentional than full ones.\n\nFor a multi-scene continuous orbit, keep one unbroken `eval_time` curve and move the *world* under it — the camera never cuts, the set changes.\n\n### Floor rise\n\nPure vertical translation, **no rotation**, smooth acceleration with a slow middle. Each \"floor\" is a different set stacked on Z at a fixed interval; the subject sits centre-frame at each pass.\n\n```python\nFLOOR_H = 4.0\nfor i in range(4):\n cam.location = (0, -6, i * FLOOR_H)\n cam.keyframe_insert(\"location\", frame=1 + i * 48)\n```\n\nRotation during a rise destroys the effect. Leave it out.\n\n### Robo-arm\n\nThe whip-and-lock commercial move: a fast flight along a curved arc, an **absolute** dead stop at a completely different angle, repeat. Each relocation is roughly a third of a second; each stop is a distinct, readable frame.\n\nBuild it as a path with a control point per stop. On a fresh path action, key both ends of each hold and leave the flight between holds animated. At 24 fps, these eight-frame flights last a third of a second:\n\n```python\nfor frame, progress in [(1, 0), (11, 0), (19, 32), (29, 32), (37, 64), (47, 64)]:\n curve.eval_time = progress\n curve.keyframe_insert(\"eval_time\", frame=frame)\nfc = next(f for f in action_curves(curve) if f.data_path == \"eval_time\")\nassert [int(k.co[0]) for k in fc.keyframe_points] == [1, 11, 19, 29, 37, 47]\nHOLD_START_FRAMES = {1, 19, 37}\nfor kp in fc.keyframe_points:\n kp.interpolation = 'CONSTANT' if int(kp.co[0]) in HOLD_START_FRAMES else 'BEZIER'\n kp.handle_left_type = kp.handle_right_type = 'AUTO_CLAMPED'\n```\n\nKeep the target separate and slightly offset per stop, so each lock-off is a different composition of the same subject rather than six centred portraits.\n\n### Handheld\n\nApplied **last**, on top of a finished move. Slow organic sway, not jitter: long waves plus a barely-perceptible tremor.\n\n```python\nfor path in (\"location\", \"rotation_euler\"):\n for i in range(3):\n fc = next((f for f in action_curves(cam)\n if f.data_path == path and f.array_index == i), None)\n if fc is None:\n continue\n n = fc.modifiers.new('NOISE')\n n.scale = 120 # large scale = long lazy waves (4-6s at 24fps)\n n.strength = 0.035 # small; raise for rotation, lower for location\n n.phase = i * 7.3 # decorrelate the axes or it reads as a slide\n```\n\n**For a restrained handheld register, avoid fast jitter, wobble or snap corrections.** Those motions can serve a deliberately frantic or phone-camera brief; choose them intentionally rather than adding them as a generic cinematic finish.\n\n### Over-the-shoulder cuts\n\nPer cut: a camera position below shoulder height, a near-foreground body mass, and a target on the far face. Move barely — a slow sideways crawl. See `slates-dialogue-blocking` for who may occupy the foreground and why it matters.\n\n### Lens\n\nAnimate focal length like any other channel; a slow lens breath under a move adds a lot for nothing.\n\n```python\ncam.data.lens = 35\ncam.data.keyframe_insert(\"lens\", frame=1)\ncam.data.lens = 50\ncam.data.keyframe_insert(\"lens\", frame=96)\n```\n\n**But the lens must not drift across a cut.** Within a cut it can animate; on the cut frame it changes instantly with everything else.\n\n## Multiple cameras and cuts\n\nTwo ways, and only one of them survives contact with a 19-shot edit:\n\n- **One camera, jump-cut it.** Keyframe location/rotation/lens with `CONSTANT` interpolation on the cut frames. Fine up to a handful of cuts.\n- **A camera per setup, bound to timeline markers.** Correct for anything bigger, because each setup stays independently editable.\n\n```python\nmarker = bpy.context.scene.timeline_markers.new(\"SHOT_04\", frame=188)\nmarker.camera = cam_shot_04\n```\n\nEither way the invariant is the same: **camera, target and lens all change on the cut frame, and no frame between two setups is interpolated.** One transition frame reads as a whip-pan, and the model will reproduce it faithfully.\n\n## Verify before you render\n\n```\nslates_blender_scene\n```\n\n`cutSeconds` is your actual cut list in seconds. Check it against the shot list you were given — mismatches here become mismatched prompt timings, and the prompt is what you write next.\n\n⚠️ **On a marker-bound rig, read `cutSeconds` or `markers`, never `camera.keyframeSeconds`.** The per-setup cameras are usually static (a Track To constraint does the aiming), so the active camera's action is empty and that field reads `[]` on a perfectly good three-cut edit. `cutSeconds` resolves to whichever the scene actually used.\n\n## Related\n\n`slates-previs-blocking` (the workflow) · `slates-blocking-to-prompt` (turning these moves into prompt text) · `slates-dialogue-blocking` (OTS and eyelines)\n",
|
|
6
|
+
"slates-character-identity": "---\nname: slates-character-identity\ndescription: \"Prepare and bind a reusable character identity sheet from an image or description. Use when creating cast or preserving the same character across shots.\"\n---\n\n# Character identity sheet — Slates workflow\n\nA character's identity sheet is attached to **every** downstream generation that mentions it, so a flaw in the sheet becomes a flaw in every shot made from it. Building it well is the highest-leverage thing you can do for a project.\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n## The shape: ONE sheet, three panels\n\nSlates generates **one identity sheet per character**, bound as the character's canonical reference:\n\n| Panel | What it carries |\n|---|---|\n| **Chest-up portrait, three-quarter angle, largest panel (~25–30% of the sheet)** | The face. **This is the only place the model reads facial identity from** — every detail it will ever know comes from those pixels, so it gets the resolution. Off-frontal, never dead-on: an angled head reads its volume instantly. |\n| **Full-body front, relaxed A-pose — cropped at the collarbone, just the face cropped out** | Build, proportion, wardrobe. The face is cropped off on purpose: a front-facing body panel renders a ~40px face that can't match the portrait's, so the sheet would carry two competing identities and the model averages them. **Only the face** — neck, arms and hands render as skin. |\n| **Full-body back, head and hair visible** | Hair fall and the back of the outfit — the only panel where either reads. Keeps its head because there's no face to compete with. |\n\nThe rule is **kill every competing rendering of the FACE, not every head** — which is why exactly one body panel is headless.\n\nOn a deep neutral-grey plate (hex `3a3a3c`, written bare; since the composer fix `#3a3a3c` reads the same — see the resolved composer hazard in Don'ts), flat and shadowless, with catchlights in the eyes, irises never crushed to black, surface texture at the medium's own natural level of detail, broken symmetry, and no over-clean 3D-game-model look. Expression is **a slight natural smile with the teeth just visible** — a closed mouth carries no dental information, so every downstream smiling shot invents teeth, and teeth are person-specific.\n\n**Two carve-outs, scoped differently on purpose.** Non-human characters get a natural neutral expression instead of a smile — that one is scoped by *having a human mouth*, so a bipedal robot or humanoid alien is covered. Quadrupeds and non-bipedal characters get a natural standing stance with the head shown on both body panels — that one is *anatomical*. **Both are conditionals the image model evaluates against your reference; neither is a code branch, because the op has no character-kind input.**\n\n**The sheet inherits the source's medium** — photo, anime, illustration, painterly, 3D render — unless the user explicitly asks for a transform. None of the craft clauses above override that: they ask for *readable* eyes and *material-looking* surfaces within whatever medium the character is in, not for photorealism.\n\n**Why one sheet.** Every `@character` mention attaches that character's canonical identity image, so each character costs one reference slot. It also reduces competing facial renderings to **one** — with the front panel headless and the back panel turned away, the portrait is the only face on the sheet, so there is nothing left to average.\n\n## Workflow\n\n### Get the reference\nThe user has either:\n- Pasted/uploaded an image of the character (real person, drawing, AI render).\n- Described the character in text only.\n\nIf image: upload it as a reference<!-- slates-only --> (`slates_upload_reference_image`)<!-- /slates-only -->.\nIf text only: generate from prompt-only — less consistent, so warn the user.\n\n<!-- slates-only -->\n### Create the character record\n`slates_create_character` with:\n- `name` (ask if not given)\n- `description` — 1-2 sentences, *visual* only (\"tall, dark hair, scar over left eye\"), not personality.\n- `style` — leave as the source's own medium by default. Only name a transform if the user wants one (e.g. anime → realistic).\n<!-- /slates-only -->\n\n### Generate the sheet\n<!-- slates-only -->\n`slates_generate_character_identity` with `characterId`, `projectId`, and `baseAssetId` (the source portrait).\n\n**Do not hand-write the sheet prompt.** Slates builds it from the canonical template in `@slatesvideo/shared/prompts` (`buildCharacterIdentityPrompt`) — panels, plate, lighting and craft clauses included — and appends your `userNotes`. Use `userNotes` for what the template can't know: *\"use the woman on the left\"*, *\"keep the scar on the right cheek\"*. A hand-written prompt is a fork of the template and will drift from it.\n\n- Estimate cost first with `slates_estimate_generation_cost` and announce in **credits** — never quote a price from memory.\n<!-- /slates-only -->\n\n<!-- slates-only -->\n<!-- @inject:sheet-tool-defaults -->\n**What the sheet tools render on** (you do not pick these; omit `model`):\n\n- **Character identity sheet:** `gpt-image-2-5-sunburst` at 3k, quality `high`, one 16:9 image.\n- **Establishing image:** `gpt-image-2-5-sunburst` at 3k, quality `high`, one 16:9 image.\n\nPrice a sheet for that model at 16:9, with resolution and quality left at their defaults. **Never 4K** — no identity gain at sheet scale, wasted spend.\n<!-- @end:sheet-tool-defaults -->\n<!-- /slates-only -->\n\n- When the result returns inline, **evaluate it before binding**:\n - Is the portrait clearly the largest panel, and is it off-frontal?\n - **Is the front body panel cleanly headless** — an empty collar above a normally rendered body, no partial face, no floating jaw, no smeared neck stump? A botched crop is worse than no crop.\n - **Is the body still there?** Neck, forearms and hands rendered as skin, not an empty outfit floating on nothing. A hollow garment means the invisible-mannequin genre ran unbounded.\n - Do the body panels read as the same build, wardrobe and hair as the portrait?\n - Catchlights present, irises readable rather than black holes?\n - Is it in the source's medium, and does it read as *that* medium done well — or has it drifted toward the over-clean game-model look?\n - Plate a flat deep grey, not white and not black?\n- If off: one focused refinement, then regenerate. The sheet is upstream of everything — it is worth a re-roll that a scene frame is not.\n<!-- slates-only -->\n- The op binds the result as the canonical identity automatically.\n\n### Hand back\n> \"Character {name} ready — identity sheet bound. Use `@{name}` in any prompt and Slates attaches it and names it inline, so the face stays consistent.\"\n<!-- /slates-only -->\n\n## How the reference gets used at scene time\n\nSlates cites the sheet inline under the character's name — `{name} (image N)` — in the exact order it sends references. That **name** is the anti-averaging lever, and it is each model's own official mechanism (NB2: \"assign a distinct name\"; Seedance: `Reference <Subject_N> in <Image_N>`; Kling: reuse a fixed label verbatim).\n\nCritically, the app injects **no** wardrobe, expression, or lighting directive. The user's scene prompt owns all of that — which is why `@{name}` dropped into a movie-still injection keeps the still's own clothing and lighting instead of dragging the sheet's.\n\n## Anti-patterns\n\n- **Don't** studio-light, white-background, or black-background the sheet. White bleeds into the video and washes out the location; black eats edge detail. Flat, even, shadowless light on a deep neutral grey.\n<!-- slates-only -->- **Don't** hand-write the sheet prompt when the op will build it — that is how the template and the shipped prompt fork.<!-- /slates-only -->\n- **Don't** create a second character image. One canonical identity is what the storyboard pipeline reads.\n<!-- slates-only -->- **Don't** skip binding. An unbound asset doesn't help downstream.<!-- /slates-only -->\n- **Don't** invent character details. Stick to what's in the reference image and the user's description.\n- **Don't** describe the front panel's crop as an absent head — in `userNotes` or any hand-written variant. The template asks for it as *framing*: **\"cropped at the collarbone, an invisible-mannequin presentation with just the face cropped out\"**, a standard e-commerce genre with deep training data. **\"the head not shown\" is a hard 422 on GPT Image** (measured on `gpt-image-2`, the model 2.5 replaced; the classifier is OpenAI's, not the version's, so the rule carries — but nobody has re-run it on Flare or Sunburst) — fal returns `content_policy_violation` with `loc: [\"body\",\"prompt\"]`, so the text is rejected before any image is read, because an anatomical absence reads as gore to OpenAI's classifier. It passed NB2, which is why the original receipt looked safe: **it was model-scoped.** State an exclusion as a framing choice, never as a missing body part.\n- **Don't** invoke the invisible-mannequin genre without bounding it to the face. **\"an invisible-mannequin presentation where the clothing holds its own shape\" removed all the skin** — no neck, no hands, no forearms, a garment floating on nothing — because that *is* the e-commerce genre in full: an empty outfit. **\"with just the face cropped out\"** keeps the anchor and bounds it. Generalises: a genre anchor imports the whole genre, so name what STAYS, not only what goes.\n**Resolved composer hazard:** on 2026-07-30, `#3a3a3c` reached fal as `background ()` because unresolved sigils were deleted. The composer now preserves unresolved `#` and `@` text byte-for-byte; a token binds a reference only when it resolves. Literal hex colours and handles are safe. The sheet template keeps its bare hex as a wording choice, not a workaround.\n- **Don't** use 4K — wastes credits, no quality gain at sheet scale.\n- **Don't** feed a multi-view sheet into a Seedance 2.0 shot that has **several characters in frame** without binding each character to its image and appending the anti-twin constraint; ByteDance documents multi-view assets as a cause of duplicate characters on 2.0. See `slates-prompting-seedance`. Seedance 2.5 supports multi-view subject references; see `slates-prompting-seedance-2-5`.\n",
|
|
7
|
+
"slates-chatgpt-images": "---\nname: slates-chatgpt-images\ndescription: \"Generate and save images through a connected ChatGPT account or the host image tool, preserving Slates prompts, references and project context. Use when ChatGPT generation is requested.\"\n---\n\n# ChatGPT images in Slates\n\n## Prerequisites\n\nSaving into a Slates project requires the Slates desktop app open and a connected Slates MCP server or CLI. Follow the [connection guide](https://slates.video/docs/connect-claude); CLI onboarding is `npx -y @slatesvideo/cli setup`. The host image tool alone does not expose Slates project tools. If the required tools or desktop connection are unavailable, report the missing connection and retain the prompt and references; do not invent operations.\n\nThe connected generation path also needs the optional ChatGPT images add-on enabled and authenticated as described below. A host-tool path needs an actual image generator exposed by the host. These are distinct capabilities; a paid account alone establishes neither.\n\nResolve the project with `slates_list_projects` and references with\n`slates_get_selection` or `slates_list_assets`. Badge codes are project-specific.\nInspect the selected images before generating. Keep the ordered reference IDs\nalongside the exact prompt submitted for every output.\n\nRetrieve relevant craft with `slates_get_prompting_guide`: use `cinematic-look`\nor `style-prompting` for those requests, with section/query retrieval when useful.\nDo not apply another API model's settings or capabilities to the host generator.\n\n## Connected desktop path\n\nThis add-on is off by default. The user enables Settings → AI tools → ChatGPT images in\nSlates before connecting. It requires an installed Codex host and an eligible\nChatGPT account; a subscription alone does not install or connect the host.\nSettings offers installation instructions, Connect ChatGPT and Check again in\nplace, and reports the same connection state as the image picker.\nNever install software or enable the add-on silently. Ordinary Slates use needs\nneither this add-on nor Codex.\n\nCall `slates_get_chatgpt_status`. If connected, use\n`slates_generate_chatgpt_image` with projectId, a fresh UUID requestId, the prompt\nand ordered referenceAssetIds. This is also the path used by Slates' prompt bar\nand Studio Agent. Use background mode for long calls and inspect\n`slates_get_generation_status` with the returned generationId.\nThe desktop bridge removes only its own temporary thread's original image after\nsaving and byte-verifying the project copy. Failed cleanup keeps the original.\nThis does not authorize deletion of files from ordinary host conversations or\nfiles supplied to `slates_save_external_image`.\n\nOn timeout or an uncertain response, reuse the SAME requestId. Do not create\na new request to check the old one. Failed or unavailable connections preserve\nthe user's prompt and refs. Start sign-in with `slates_connect_chatgpt` only when\nthe user requests connecting; give the returned URL to the user to complete it.\n\n## Host-tool path\n\nIf this conversation exposes a built-in image generator and the user chooses\nthat host workflow, retrieve originals through `slates_get_asset_image` with\nfullRes, or use the returned local paths when the host can read them. Inspect\nlocal images using the host's image viewer before editing. Supply every selected\nreference in its intended order using the generator's actual schema.\n\nUse the built-in generator. No API key, paid API, browser automation, invented\nmodel identifier, hidden quality setting or silent fallback. If unavailable,\nreport that limitation and retain the prepared prompt and references.\n\nSave EACH returned image through `slates_save_external_image`, using its actual\nfilePath or image dataUrl, exact submitted prompt, observed generator label and\nthe IDs of the references actually sent. Omit model unless the host reports it.\nRequested settings are requests, not output facts. Text-only images have no\nreferences. Never substitute a screenshot of the result for the original bytes.\n\nFor an uncertain save, inspect the project's new assets and compare prompt,\nlineage and output bytes before retrying. The external save operation is not\nidempotent for new imports. Reuse assetId to annotate a confirmed existing\nupload; do not import the same file again. If identification is ambiguous, stop\nand report the uncertainty. A missing output or host failure saves nothing and\ndoes not authorize another generation.\n\n## Framing and quality requests\n\nThe connected operation accepts optional `aspectRatio`, with presets supplied\nby its schema. Slates appends that request in words through the same composer\nused by the UI's What gets sent preview. Do not also append a second ratio\ninstruction yourself. Saved metadata preserves the original text, requested\nratio and exact submitted prompt; actual dimensions remain measured separately.\n\nFor a new prompt, choose only an aspect-ratio preset supported by the current\nflagship GPT image model in Slates' capability SSOT. Resolve the current model\nthrough `slates_list_available_models` and `slates_get_prompting_guide` for\nmodel-selection; read the generated `slates_generate_image` aspectRatio schema\nfor its allowed presets. Do not call that paid operation. Do not maintain a\nsecond hard-coded ratio list here. If the current presets cannot be retrieved,\nretain the prepared prompt and resolve them before adding a framing request.\nAn explicit user request takes precedence; never silently rewrite their prompt.\n\nState the chosen ratio and orientation in the submitted text. This is a prompt\nrequest, not a host size parameter or an assertion of the host's limits. Measure\nthe returned file and report requested versus actual framing; never silently\ncrop or stretch it to make the numbers match.\n\nDescribe desired detail, legibility, materials and lighting concretely. Words\nsuch as “higher quality” do not establish a quality enum or select a backend.\nOnly record an actual model or quality tier if the host reports it. The built-in\ntool and App Server receipt inspected for this workflow do not expose those\nfields. Preserve revisedPrompt separately when supplied.\n\nOpenAI's [image prompting guide](https://developers.openai.com/api/docs/guides/image-prompting)\nseparates API parameters from prompt language. Its quality and pixel controls\nare not automatically controls of the built-in tool. Product launch names and\nthe `chatgpt-image-latest` API alias are not per-result model receipts.\n\n## Verify the saved result\n\nRead back the returned asset(s) and inspect the images. Report project, badge\ncode, generator, reference codes and measured dimensions. Distinguish submitted\nprompt from any host-reported revised prompt. Do not claim a model/quality tier\nfrom appearance. Reuse Prompt restores text and references; check the displayed\ngeneration destination before another generation. No plugin publication or\ndirectory listing is required for this workflow.\n",
|
|
8
|
+
"slates-cinematic-look": "---\nname: slates-cinematic-look\ndescription: \"Choose lighting, exposure, grade, lens, atmosphere and imperfection techniques for a filmed look. Use for photographic direction or an image that looks too clean, evenly exposed or studio-lit.\"\n---\n\n# Cinematic look — make a generated frame read as filmed\n\n<!-- @card:start -->\n**Image models default to clean, evenly lit and fully exposed.** Real film frames can be dark, flat, murky or burned out. Describe what the camera sees; a mood word or look reference alone does not get you there.\n\n**Look at every reference first.** Write a look reference's grade and imperfections into the prompt: darkness, contrast, black level, colour cast/saturation, softness/noise and subject separation. Never grade cleaner, brighter or higher-contrast than that reference unless asked. Inspect the character sheet's garments. Cite references inline: `the woman from image 1`, `lit and graded like image 2`; never open with a reference-role paragraph. References are optional; these techniques work from words alone.\n\nFor a new photographic frame:\n- **One physical light system:** source, position, effect on the subject; no light without a source. If using atmosphere, put it in front too.\n- **Exposure as it looks:** `close to a silhouette, features only just readable`, not a stop under.\n- **Lens name plus effect, every time:** `200mm telephoto`, `the peaks loom huge behind her and melt into soft shapes`.\n- **Name every garment and close the foreground:** say exactly what is there; omissions invite reference leakage or invented props.\n- **Only what the shot needs:** one light system, at most one exposure, atmosphere and colour choice, one or two composition moves and one moment. Add the imperfections the scene calls for.\n\nA scene reference owns the grade; a look-only reference does not own the new scene's light. For an owned-frame edit, describe only the change and what stays. Never use a released film frame as the edit base; use it as an art-direction brief for a new scene. Use positive descriptions first; one targeted negative is enough where supported. FLUX needs positive wording.\n\nUse query with a technique ID or section, depth \"index\" to browse, or depth \"full\" for the catalogue and examples.\n<!-- @card:end -->\n\n**Evidence scope:** most Slates receipts come from single generations of one scene on GPT Image 2.5 Sunburst. They show useful directions to test, not reliable success rates or a universal ranking. The research record owns prompts, source links and observations: `business/projects/slates/research/cinematic-look-research.md` in the vault. This skill owns active technique wording. The catalogue checker compares IDs, evidence tags and worked-example bytes.\n\nImage models default to a clean, evenly lit, fully exposed picture. Real film frames are often graded \"wrong\": underexposed, backlit, silhouetted, burned out, flat or murky. A model only goes there when the prompt says what that looks like. A look reference alone does not get you there; describing the frame does.\n\n## Two routes to a filmed frame\n\n1. **Describe a new frame.** Write the scene and reference roles inline, then the light, exposure, camera and texture choices needed to realize it.\n2. **Edit a frame you own.** When a Slates plate or sheet, your own photo or footage, or a Blender render already has the composition and look, describe only the requested changes and what must remain. Example: \"Take image 1 and change only the character to the character in image 2.\" Never use a frame from a released film as the base; a film still is an art-direction brief for route 1.\n\n## The rules\n\n1. **Describe what the camera sees, not the camera setting.** \"Her face sits a stop under\" was ignored; \"close to a silhouette, her features only just readable\" landed. Mood words carry little on their own.\n2. **Build one physical light system first.** Name the source and its position, what it does to the subject, and rule out light with no source. If the shot needs atmosphere, place it in front of the subject as well as behind. Every source and shadow must agree.\n3. **Name the lens and describe its effect, every time you describe a new photographic frame.** A lens named alone changed nothing visible in the summit test; named with its effect, it produced compression and blur.\n4. **Use only what the shot needs.** One light system (its consequence clauses count as one), at most one pick each from exposure, atmosphere and colour, one or two composition moves and one moment. Then stop adding. These are drafting limits, not model capability limits.\n5. **Name everything a reference could fill.** Every garment, and a closed list of what is in the foreground. Anything left out can come from a reference or be invented. In route 2, preserve the existing inventory and describe only the change.\n6. **One targeted negative after a positive description is fine; a list is not.** Follow the model's grammar: FLUX needs positive wording.\n7. **A scene reference owns the grade; a look-only reference does not own the light.** With a look-only reference, still write the light direction and visible exposure for the new scene. Name references inline where used, never in an opening role paragraph.\n8. **Look at every reference before writing.** First describe a look reference's own grade and imperfections: darkness, contrast, true or muddy blacks, colour cast and saturation, softness and noise, and how much the subject separates from the background. Never grade cleaner, brighter or higher-contrast than the reference unless the user asks. Inspect a character sheet's garments so the prompt names or replaces every one. Without a reference, the same techniques work from words alone.\n\n## Build order\n\nInspect references first. For a new frame: time/weather → light source and direction → effect on the subject → exposure as it looks → atmosphere in front, if needed → colour → composition and camera placement (lens plus effect) → the moment → kind of picture → inventory (every garment and the closed foreground). Carry the observed reference grade through those choices. Adapt examples to the actual scene; never inherit their props, wardrobe or light by accident.\n\n## Evidence tags\n\n`receipt` measured on Slates generations · `vendor` stated in the model maker's own guide · `practitioner` a published third-party guide or test · `canon` established cinematography, never measured on a model · `untested` reasoning only: try it, then record the result in the research doc and change the tag.\n\n## The catalogue\n\nEach row: the technique, its evidence, what it does to the frame, when to reach for it and when to skip it, and wording written as what the frame looks like.\n\n<!-- @catalogue:start -->\n\n### 1. Exposure and tone — the \"graded wrong\" frame\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `flat-underexposure` | receipt | The whole frame sits in a narrow, dark tonal range; the subject barely separates | A dark, flat look reference or fading light · skip hard contrast, glowing practicals and crushed blacks | \"Everything sits in dark, muddy navy blue; nothing is bright or truly black. Her dim face barely separates from the trees; even the focused face is slightly soft, with fine noise in the dark blues.\" |\n| `subject-under-key` | untested | The face is clearly darker than the brightest part of the frame; features read, nothing lights them | Backlit exteriors, sunrise and sunset, window interiors · skip beauty, product, lip-sync close-ups | \"Her face is in shadow, clearly darker than the sky behind her; her features are readable but nothing lights them.\" |\n| `near-silhouette` | receipt | A dark shape with edge detail against a bright field | Wides, entrances and exits, solitude · skip shots that depend on recognising the face | \"She is close to a silhouette: her face and jacket fall into deep shadow, her features only just readable.\" |\n| `clipped-highlights` | receipt | The sky near the sun, windows and practicals go pure white | Backlit shots, windows, night practicals · skip skies that carry the story, white packaging | \"The sky around the sun burns out to white.\" |\n| `dense-shadows-with-ramp` | receipt | Shadows sink to near-black but fall off gradually, and one detail survives in the dark | Night, interiors, a backlit foreground · skip video dark work that already turns to murk | \"The shadows are dense and slightly crushed, the foreground rock nearly black, and the light fades into them gradually.\" |\n| `low-key-ratio` | vendor | One side of the face lit, the other falls away with nothing filling it | Interiors, night, close-ups that carry mood · skip bright comedy or commercial register | \"Light reaches only the left side of his face; the right side falls into shadow and nothing fills it.\" |\n| `flare-washed-contrast` | receipt | Stray light lifts the blacks and flattens contrast over part of the frame | Shooting toward the sun or a hard practical · skip crisp thriller, neon noir | \"Flare washes across the upper half of the frame and lowers the contrast.\" |\n| `lifted-matte-blacks` | canon | The darkest tones sit at charcoal, like a faded print | Daytime melancholy, period looks, overcast · never with `dense-shadows-with-ramp` | \"The darkest parts of the frame are a soft charcoal rather than black.\" |\n| `uneven-exposure-across-frame` | untested | Exposure is right for one zone only; one side runs hot, the far side goes dark | Mixed-light interiors, night streets, documentary register · skip clean product shots | \"The lamp side of the room is overexposed and the far corner goes nearly black.\" |\n\n### 2. Light source and direction — one physical system\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `name-the-one-source` | receipt | A dominant source with shadows consistent with its distance | When the light needs control · skip when preserving existing light | \"The sun is low behind her and off her right shoulder, just above the far peaks.\" |\n| `forbid-the-phantom-key` | receipt | Removes the soft front light models invent on faces | Any backlit, side-lit or practical-lit person · skip flash, frontal sun, beauty work | \"There is no light in front of her: no frontal key, no fill.\" |\n| `hard-rim-backlight` | receipt | A bright edge on hair and shoulder while the face stays in ambient light | Golden hour, practicals behind the subject · skip overcast and blue hour, which have no hard source | \"A hard orange rim of light traces her hair, the edge of her cheek and one shoulder.\" |\n| `face-in-bounce` | receipt | The unlit side is filled only by coloured light reflected from the surroundings | Backlit exteriors, rooms with coloured walls · skip when the face must read cleanly | \"Her face sits in soft, cooler bounce light off the rock.\" |\n| `raking-side-light` | receipt | Light skims a surface so pores, weave and grain show | Close-ups, skin realism, materials · skip when the brief is flattery | \"Daylight rakes across her face from one side, so texture catches along the cheekbone and the other side sits in soft shadow.\" |\n| `practicals-only` | vendor | Lights inside the frame are the only light: pools with dark gaps between | Night interiors, cars, bars, kitchens at night · skip when the room must read | \"The only light is the open fridge she is standing in; the kitchen behind her is dark.\" |\n| `window-light-falloff` | canon | Bright near the window, then a fast drop across the room | Day interiors, quiet drama · skip big evenly lit spaces | \"Grey daylight from the single window on the left; a few steps away the room drops into shadow.\" |\n| `mixed-colour-temperatures` | receipt | Warm and cool sources side by side, uncorrected | Dusk interiors, night streets, cars at night · skip clean commercial | \"Warm lamplight on his face; cold blue daylight from the window on the wall behind him.\" |\n\n### 3. Atmosphere and optics — what sits between the lens and the subject\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `haze-in-front` | receipt | Haze, dust or smoke between camera and subject softens the subject's edges | Exteriors, sunbeams, dusty sets · skip crisp product or text shots | \"Thin haze crosses the frame in front of her, so her edges are no sharper than the rock beside her.\" |\n| `source-flare` | vendor | Flare from a bright source in the frame | Facing the sun, headlights, stage lights · skip jargon stacks and scenes with no hard source | \"The low sun at the edge of the frame throws a flare that crosses in front of her.\" |\n| `halation-on-highlights` | canon | A soft reddish glow bleeds past bright edges | Night practicals, candles, neon, film looks · skip clean digital register | \"Bright lights have a soft reddish glow bleeding past their edges.\" |\n| `defocus-as-outcome` | receipt | Planes separate: only the subject is sharp | Close and medium shots · skip wides where the place must read | \"On a 200mm telephoto lens only she and the pan are sharp; the background melts into soft shapes and the foreground rock edge falls out of focus.\" |\n| `soft-overall-no-sharpening` | receipt | Lower microcontrast, no halos along edges | Photoreal people, film register · skip product detail and dense text | \"The image is slightly soft overall; edges carry no crisp outline.\" |\n| `grain-in-shadows` | vendor | One texture note | Film or low-light register · never stacked with other noise words | \"Fine grain is visible in the shadows.\" |\n| `edge-falloff` | canon | The corners sit darker than the centre | Film looks, night · skip flat graphic compositions | \"The corners of the frame are darker than the centre.\" |\n| `glass-or-weather-between` | receipt | Rain, smudges or condensation between the lens and the subject | Cars, cafés, storms, observed framing · skip when the face must be sharp | \"Seen through a rain-streaked car window; the drops are sharp and she is soft behind them.\" |\n\n### 4. Colour and grade — preserve the reference unless a change is requested\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `warm-muddy` | receipt | Warm but desaturated; whites read cream, greens olive | Dusty or earthy exteriors, fatigue, nostalgia · skip crisp commercial | \"The colour is warm and a little muddy rather than clean.\" |\n| `restrained-desaturated` | canon | Muted overall; at most one colour stays strong | Drama, cold or bleak moods · skip joyful or brand-colour work | \"Colours are muted and low in saturation; only the red of her jacket holds its colour.\" |\n| `uncorrected-white-balance` | receipt | The colour cast is left in | Phone or documentary register, shade, fluorescents · skip product colour accuracy | \"White balance left a little cool and uncorrected; the white mug looks bluish.\" |\n| `tungsten-in-daylight` | canon | The daylight scene renders blue-cyan | Cold mornings, alienation · skip warm romance | \"Daylight renders cold and blue, as if the camera were set for indoor lamps; skin looks pale.\" |\n| `sodium-vapour-mono` | canon | Street light collapses colour to amber | Urban night, industrial areas, parking lots · skip scenes that need colour separation | \"Orange streetlight flattens every colour to amber and brown; shadows go brown-black.\" |\n| `bleach-bypass-look` | canon | High contrast, drained colour, silvery skin | War, grit, harsh drama · skip warm or intimate scenes | \"Colour almost drained out, contrast harsh, skin grey-silver, heavy shadows.\" |\n| `teal-orange` | vendor | Warm skin against teal shadows; the generic blockbuster look | Only when the brief asks for blockbuster register · never as a default | \"Skin stays warm while the shadows and sky are pushed toward teal.\" |\n| `era-or-device-register` | vendor | One phrase shifts colour, grain and flash together | Period or amateur looks · one per prompt | \"As if shot on 1980s colour film, slightly grainy.\" |\n| `cross-processed` | vendor | Hard colour shifts: cyan shadows, magenta highlights | Music video, fashion, 90s editorial · skip naturalism | \"Colours shifted hard, cyan in the shadows and magenta in the highlights.\" |\n\n### 5. Time and weather presets — bundles of the above, and mutually exclusive\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `golden-hour-backlit` | receipt | Sun low behind, long shadows toward camera, rim light, face in bounce, sky near the sun white, warm haze | Warm exteriors · never with noon shadows or overcast softness | \"The sun sits just above the ridge behind her; long shadows run toward the camera; a hard rim on her hair; her face in cool bounce; the sky around the sun burns white.\" |\n| `after-sunset` | canon | No direct sun, soft shadowless light, sky fading pink to blue, practicals just on | Quiet exteriors · never with rim light or hard shadows | \"The sun has just set; soft shadowless light; the sky fades from pink to deep blue; porch lights have just come on.\" |\n| `blue-hour` | receipt | Cool even light; warm practicals run hot against it; faces dim | Streets and cafés at dusk · never with a warm key on the face | \"Deep blue dusk light, almost no shadows; the café windows glow hot orange; her face is dim and blue.\" |\n| `hard-noon` | practitioner | Bleached sky, short hard shadows, squinting | Heat, desert, exhaustion · skip anything that must flatter | \"Midday sun straight overhead; tiny hard shadows under her brows and chin; she squints.\" |\n| `overcast-flat` | practitioner | No shadows, low contrast, honest skin | Plain daylight, street realism · never with rim light or flare | \"Flat grey overcast light, no shadows, low contrast, colours slightly dull.\" |\n| `night-practicals` | canon | Pools of light, dark gaps, colour casts, blooming highlights | Night streets and interiors · never with \"evenly lit\" | \"The street is dark between the orange streetlights; he is lit only when he passes under one.\" |\n| `firelight` | canon | Warm flicker from below, fast falloff, black past a few metres | Campfires, candles · never with daylight fill | \"The campfire is the only light: warm and flickering from below, their faces half lit, everything beyond the circle black.\" |\n| `moonlight-day-for-night` | canon | Cool, low saturation, one faint hard shadow direction, sky darker than the ground | Night exteriors · skip when warm practicals dominate | \"Cold blue moonlight from one side; faint hard shadows; colour almost gone; faces only just readable.\" |\n| `rain-wet-night` | canon | Wet surfaces smear reflections; rain shows only where it passes a light | Urban night · rain across the whole frame, never in one corner | \"Rain shows as bright streaks where it passes the streetlight; the wet asphalt reflects the lights as long smears.\" |\n| `fog-or-dust` | canon | Depth falls off to grey; figures layer by distance | Mystery, scale · never with crisp distant detail | \"Fog swallows everything past the second streetlight; the far figures are pale grey shapes.\" |\n\n### 6. Composition and camera placement\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `dirty-foreground-occlusion` | receipt | An out-of-focus object near the lens covers part of the frame | Anything that should feel observed rather than staged · only objects that belong in the space | \"The edge of a pine branch crosses the left third of the frame, close to the lens and completely out of focus.\" |\n| `off-centre-cropped-subject` | practitioner | The subject sits near an edge, partly cut by the frame | Candid and documentary register · skip deliberate symmetry | \"She sits in the right third of the frame; her elbow is cut off by the edge.\" |\n| `negative-space` | canon | A small subject in a large empty area | Isolation, scale · in a video plate give the empty area texture, or it moves with the foreground | \"She is a small figure at the bottom right; the rest of the frame is pale, cloud-streaked sky.\" |\n| `frame-within-frame` | canon | A doorway, window or vehicle surrounds the subject | Observed feel, confinement · skip when it hides the action | \"Seen through the open barn door; the dark door frame surrounds her on three sides.\" |\n| `over-the-shoulder-foreground` | vendor | A foreground head or shoulder, dark and soft | Dialogue, two-person scenes | \"The back of his head and shoulder fill the left edge, dark and out of focus; she faces him, sharp.\" |\n| `compression-as-outcome` | receipt | The long-lens look: the background looms huge and close | Making a background loom, crowds, heat haze · skip intimate interiors | \"Shot from far away on a 200mm telephoto lens, the peaks loom huge and close behind her, stacked right up against her shoulders.\" |\n| `camera-height` | receipt | Ground level, hip height or overhead, picked on purpose | Every shot · eye level only by choice | \"The camera sits on the ground by the stove, looking up at her past the pan.\" |\n| `grabbed-framing` | untested | Horizon slightly off, framing a beat late | Documentary or UGC register · skip composed cinema | \"The horizon tilts slightly and the framing is a little late; her head is near the top edge.\" |\n| `reflection-partial` | receipt | The subject seen in glass, steel or a puddle | Night, cities, variety across a set · never a bathroom mirror | \"We see her only as a reflection in the dark shop window, overlapped by the street behind the glass.\" |\n\n### 7. Texture, wardrobe and set\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `name-the-capture-context` | receipt | Says what kind of picture this is, instead of listing flaws | Every photoreal shot · never a flaw inventory, which reads as tokens and turns plastic | \"A frame from a film shot on location, not a studio portrait.\" |\n| `skin-under-real-conditions` | vendor | Skin carries the environment: wind, sun, sweat | People in real conditions · never a stack of pore and blemish words | \"Wind-chapped cheeks and a sunburnt nose after a day on the mountain; skin shiny with sweat at the hairline.\" |\n| `hair-state` | receipt | Flyaways, wind, strands stuck to the forehead | Weather, action, fatigue · vary the state, never the style that carries identity | \"Wind has pulled strands loose across her face.\" |\n| `wardrobe-wear` | vendor | Creases, dust, fading | Lived-in characters · skip fashion hero shots | \"Her jacket is creased at the elbows and faded at the seams; dust on the knees.\" |\n| `full-wardrobe-spec` | receipt | Names top, legwear and footwear so a reference cannot fill the gap | Any shot built from a character reference · not optional | \"Grey hiking jacket zipped up, dark hiking trousers, scuffed brown boots.\" |\n| `closed-prop-list` | receipt | A positive, closed inventory of what is in the frame | Any frame where the model adds junk · never a \"no extra props\" list | \"The only things on the rock are the stove and the pan.\" |\n| `lived-in-wear-on-named-things` | practitioner | Wear goes on objects already named, never as new objects | Sets that should feel used · never \"add clutter\" | \"The pan is blackened underneath and the rock around the stove is stained with old soot.\" |\n| `material-specificity` | vendor | Names the physical material | Hero objects · never a generic noun | \"A dented enamel mug, a scratched aluminium pot, a waxed-cotton jacket going pale at the seams.\" |\n\n### 8. Moment, performance and motion\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `caught-mid-action` | receipt | The action is under way and the camera goes unnoticed | People, by default · skip deliberate portraits and address-the-lens beats | \"Her right hand works a wooden spatula in the pan mid-stir.\" |\n| `eyeline-off-lens` | receipt | The eyes are on the task, another person or out of frame | Images and B-roll · never a lip-sync beat, where eyes on the lens are the point | \"She looks down at the pan.\" |\n| `unresolved-expression` | receipt | Not a stock smile: squinting, chewing, tired, mid-thought | Realism · skip brand-joy beats | \"She squints against the glare, jaw set, tired.\" |\n| `motion-blur-on-the-mover` | practitioner | Only the moving part blurs | Hands, tools, hair, passing vehicles · never on text or a face that must read | \"Her hand with the spoon is a slight blur of movement; the pan and the rock are sharp.\" |\n| `focus-slightly-missed` | untested | Focus landed just behind the subject | Documentary register · skip identity-critical shots | \"Focus landed on the rock just behind her; her face is a touch soft.\" |\n| `handheld-operator-body` | receipt | For video: write the body holding the camera, not the path | Documentary, UGC, tension · skip locked-off formal shots | \"Handheld, the operator breathing; the frame sways slightly, drifts off her and corrects back.\" |\n| `weight-and-consequence` | practitioner | For video: mass moves through the body and the world reacts | Any physical action · skip static dialogue | \"She shifts her weight onto her back foot as she lifts the heavy pan; the stove wobbles.\" |\n| `frame-zero` | receipt | In a still that feeds video, the event has not happened yet | Every image-to-video plate · never an aftermath plate | Describe the moment just before the event: the pole still straight, the glass still whole. |\n\n<!-- @catalogue:end -->\n\n## Techniques that clash\n\n- **`near-silhouette` against a recognisable face or lip-sync.** On identity-critical shots use `subject-under-key` with `face-in-bounce`; keep silhouettes for wides.\n- **`dense-shadows-with-ramp` against murk.** Always say the light falls off gradually and keep one detail readable in the dark. On video dark work keep the subject readable; `slates-blocking-to-prompt`'s \"no crushed blacks\" is scoped to that lane.\n- **`lifted-matte-blacks` against `dense-shadows-with-ramp`.** Pick one.\n- **One colour-temperature story.** `warm-muddy` does not go with `tungsten-in-daylight` or a cool `uncorrected-white-balance`.\n- **Presets exclude each other.** Golden hour has no noon shadows; overcast has no rim light or flare; `practicals-only` has no even room light.\n- **Haze or flare against readable text or product.** Keep the atmosphere away from the text, or drop it.\n- **`defocus-as-outcome` against a place that must read.** Keep it for close and medium shots.\n- **Motion blur, handheld sway or missed focus against text, lip-sync or identity.** Keep them off the face and off the text.\n- **`closed-prop-list` against `lived-in-wear-on-named-things`.** Wear goes on the named objects, never as new ones.\n\n## On video\n\n- **The grade lives in the still.** A motion prompt describes how the light behaves as things move — the flare slides as the camera turns, haze drifts across the foreground, her face stays in shadow as she turns — and preserves the still's grade unless the user requests a change.\n- **Lens and film-stock names translate; they do not paste.** The rule and ByteDance's own wording: `slates-prompting-seedance` → \"Don't cross-pollinate image-model syntax\".\n- **A negative-prompt field can cancel a technique.** Never suppress in `negative_prompt` something the prompt asks for (Kling: `slates-prompting-kling-v3`).\n\n## Model notes\n\n- **GPT Image 2.5.** Ask for a real photograph or film still outright. In the recorded summit tests, it tended to brighten faces, clean up colour and add props; visible-outcome wording gave more control than gear names alone. References route through its edit endpoint.\n- **Nano Banana 2 and Pro.** Google recommends named cameras, film eras, chiaroscuro lighting and positive framing, so gear names are a sanctioned lever here: still add what they do to the picture. The recorded Nano Banana Pro edits changed more of the frame than intended; scope each requested change and name what should stay.\n- **FLUX.2 Max.** No negative prompt at all, and word order is weight, so put the light system early. Its own examples use crushed shadows, blown highlights and era looks.\n- **Seedream 5 Lite.** Keep it short: pick fewer techniques to fit its prompting guide's roughly 100-word ceiling. See `slates-prompting-seedream-5-lite`.\n\n## Worked examples\n\n**Route 1, IMG-A197 (Sunburst, 2026-09-15).** Image 1 is the character identity sheet, image 2 a look reference. One light system, one exposure decision, the wardrobe and the foreground named. Eric: *\"just so well done.\"*\n\n<!-- @example:img-a197:start -->\n```text\nA film still shot on location, lit and graded like image 2. The woman from image 1 cooks on a rocky summit high in the Rockies at golden hour. She stands facing camera behind a stainless steel frying pan of sliced vegetables on a camp stove on the rock in the lower foreground, framed from mid-thigh up. Her right hand works a wooden spatula in the pan mid-stir, her left rests on the pan handle. She looks down at the pan with a slight smile. Long wavy blonde hair down. Grey hiking jacket zipped up, the word \"SLATES\" once in small plain letters on the left chest, and dark hiking trousers. The only things on the rock are the stove and the pan. Behind her: pine tops, a deep valley and snow-capped peaks running to the horizon.\n\nThe sun sits just above the far peaks behind her right shoulder and the whole frame is exposed for that sky. She is close to a silhouette: her face and jacket fall into deep shadow, her features only just readable, lit by nothing but a faint warm bounce off the rock. A hard orange rim of light traces her hair, the edge of her cheek and one shoulder. The sky around the sun burns out to white, flare washes across the upper half of the frame and lowers the contrast, and steam off the pan glows where the sun comes through it. The shadows are dense and slightly crushed, the rock in the foreground is nearly black, and the colour is warm and a little muddy rather than clean. No other text in the image.\n```\n<!-- @example:img-a197:end -->\n\n**Route 1 with a lens, IMG-A198 (Sunburst, 2026-09-15).** The same frame, with the lens named and its effect described. Real compression and depth of field came back.\n\n<!-- @example:img-a198:start -->\n```text\nA film still shot on location from far away on a 200mm telephoto lens, lit and graded like image 2. The woman from image 1 cooks on a rocky summit high in the Rockies at golden hour. She stands facing camera behind a stainless steel frying pan of sliced vegetables on a camp stove on the rock in the lower foreground, framed from mid-thigh up. Her right hand works a wooden spatula in the pan mid-stir, her left rests on the pan handle. She looks down at the pan with a slight smile. Long wavy blonde hair down. Grey hiking jacket zipped up, the word \"SLATES\" once in small plain letters on the left chest, and dark hiking trousers. The only things on the rock are the stove and the pan.\n\nThe long lens compresses the distance: the snow-capped peaks loom huge and close behind her, stacked right up against her shoulders, and they melt into soft out-of-focus shapes. Only she and the pan are sharp. The pine tops between her and the peaks are a smear of dark green, and the foreground rock edge falls out of focus too.\n\nThe sun sits just above the far peaks behind her right shoulder and the whole frame is exposed for that sky. She is close to a silhouette: her face and jacket fall into deep shadow, her features only just readable, lit by nothing but a faint warm bounce off the rock. A hard orange rim of light traces her hair, the edge of her cheek and one shoulder. The sky around the sun burns out to white, flare washes across the upper half of the frame and lowers the contrast, and steam off the pan glows where the sun comes through it. The shadows are dense and slightly crushed, the rock in the foreground is nearly black, and the colour is warm and a little muddy rather than clean. No other text in the image.\n```\n<!-- @example:img-a198:end -->\n\n**Route 1, platform revision (IMG-A200 and IMG-A204).** IMG-A199 followed its written hot lamp and crushed blacks, but Eric wanted the reference's flatter, darker, softer grade. This exact revision subsequently produced A200 and A204 using neutral identity sheets and the original look reference. Eric preferred A204 to the scene-reference remixes A201–A203; A203 and A204 used the same model and quality settings. Prompt and references changed together, so their individual contributions are not isolated.\n\n<!-- @example:platform-flat-untested:start -->\n```text\nA film still shot on location from a distance on an 85mm lens, lit and graded like image 2. The young woman from image 1 waits alone at the far end of an empty country train platform at blue hour. She stands side-on to the camera in the right third of the frame, framed from the chest up, looking down the empty track toward where a train would come from, her lips slightly parted. A dark wool coat hangs open over a grey hooded sweatshirt, hood down. The wind has pulled a few strands of hair loose across her cheek. The only things near her are the edge of the concrete platform and a single old lamp post, its lamp not yet switched on.\n\nThe sun is gone and the light is almost gone with it. The whole frame is underexposed and flat: everything sits in a narrow range of dark, muddy navy blue, nothing in it is bright and nothing is truly black. The brightest thing in the picture is the dull grey-blue sky above the trees. Her face is lit only by that weak sky from behind and to her left, so it is dim and blue and barely separates from the dark trees behind her; her features are there, but you have to look for them. There is no light in front of her, no rim of light on her hair, no catchlight in her eyes, and no contrast anywhere to make her stand out.\n\nThe long lens compresses the distance: the trees sit close behind her as soft dark shapes, and only her face is in focus, and even that is slightly soft. It looks like a camera pushed to its limit in low light: low contrast, a little murky, fine noise in the dark blues, and colour drained to a cold blue-grey. No text anywhere in the image.\n```\n<!-- @example:platform-flat-untested:end -->\n\n**Route 2, IMG-A184 (Sunburst, 2026-09-09).** Image 1 was a finished blue-hour frame, image 2 a character. The whole prompt: *\"take this image 1 and just change the character to the character in image 2\"*. Composition, grade, light and depth of field held exactly. That base frame was not one Slates owns, so the receipt proves the technique, not a shippable workflow: use your own frame.\n\n## Adding or changing a technique\n\n1. Record the evidence first: a row in the research doc's technique table, with its tag and source.\n2. Add or change the row here, with the same id and the same tag.\n3. Run the build. The catalogue check fails until both files agree.\n",
|
|
9
|
+
"slates-content-policy": "---\nname: slates-content-policy\ndescription: \"Check provider-sensitive content before prompting a brief with conflict, creatures, crowds, violence, destruction, weapons, real likenesses or young characters. Use its scoped refusal receipts and scene alternatives alongside the selected model guide.\"\n---\n\n# Content-policy-safe construction — read before any risk-surface prompt\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- Content-policy guidance. Read this list alongside the selected model's\n own prompting guide. -->\n<!-- /slates-only -->\n**Never use**: each one is a filter tripwire with a substitution in the table below:\n- `civilians in panic`, `crowds fleeing`, `blood`, `gore`, `corpse`\n- `ignite`, `catch fire`, `on fire` applied to a person — frame body-contact effects as magical or harmless VFX\n- `candle-like`, `flame-like` and any real object used as a metaphor for an effect\n- a real, named public figure\n<!-- @banned:end -->\n\nDon't depict the harm — depict the energy, the aftermath, the threat, or the scale. Build the scene safe from the first word.\n\nWrite scenes that hit full cinematic impact without ever *needing* to depict prohibited content. This is a craft move, not a compromise — the substitutions below usually read as more cinematic, not less, and they keep your generation from getting silently rejected or degraded by the model's filter. A standoff is more tense than a massacre; an evacuated city is eerier than a crowd in panic; a roar lands harder than a kill. Load this whenever a prompt involves conflict, creatures, crowds, destruction, weapons, or young characters.\n\n## Substitution table\n\n| Avoid | Use instead |\n|---|---|\n| Civilians in panic, crowds fleeing under debris | An evacuated / empty city; abandoned streets; a lone figure for scale |\n| Weapons firing into buildings or at people | Energy-discharge standoffs, searchlights sweeping, charged auras, shockwaves with no muzzle fire |\n| Creatures tearing into each other, gore | A grapple / standoff — roars, near-misses, circling, an energy clash; combat that stays contained (in/on the water, never lifting into the air) |\n| Destruction with people in harm's way | Destruction in uninhabited terrain — glaciers, deserts, ruins, open sea, evacuated zones |\n| Realistic guns as the focus | Stylized / fantasy implements, weapons slung-not-fired, the weapon as silhouette or prop only |\n| Blood, wounds, death | Impact light, dust, debris, buckling and collapse, a silhouette dropping out of frame |\n| Real, named public figures | Original / anonymous characters |\n| Real brand logos | Original or generalized branding — except the user's own product, which is the whole point of a brand film |\n\n## The safe benchmark\n\nWhen a scene starts drifting risky, pull it back toward this shape: **one original creature, in a generalized monument or amphitheatre, in daylight, no weapons present, performing expressive action** (rising, roaring, spreading wings). Original design, generalized location, daylight, expressive rather than violent. That's the confirmed-safe envelope — most epic ideas re-stage into it without losing the punch.\n\n## Containment rule — it doubles as a physics win\n\nGive any creature or combat scene a **containment rule** that grounds the physics at the same time:\n- \"the fight STAYS at the sea surface — they breach, dive, grapple, submerge, but never fly or get carried into the air\"\n- \"boss scale locked ~2.5 human-heights, NOT kaiju-giant\"\n- \"destruction stays in the evacuated valley\"\n\nThis improves coherence (the model isn't inventing absurd escalation) AND keeps the scene inside policy — same clause buys both.\n\n## Scale and stakes without harm\n\nEpic stakes come from environmental danger and reaction, not depicted victims: tiny figures diving clear of *collapsing* terrain (not being crushed), a war-horn over an *empty* field, an army *scrambling* across a frozen valley as a titan tears free of a glacier. The danger is the environment; the figures are reacting, not dying. Snow plumes, glowing runes, splintering ice, shockwaves, and dust carry the chaos.\n\n## Minors — hard rule\n\nNever write romantic, sexual, or suggestive content involving or directed at minors, and never anything that sexualizes a young-presenting character. Any scene with children stays wholesome and age-appropriate. Non-negotiable — it overrides every stylistic goal.\n\n## Pre-flight (run before delivering any risk-surface prompt)\n\n- [ ] No civilians depicted in panic/harm; crowds are evacuated or absent.\n- [ ] No weapons firing at people/buildings; threat is energy / searchlight / silhouette.\n- [ ] No creature-on-creature or creature-on-person gore; combat is grapple / standoff / roar, contained.\n- [ ] Destruction is in uninhabited / evacuated terrain.\n- [ ] Creatures are original (\"not based on any franchise\"); no real public figures; no real brand logos except the user's own product.\n- [ ] Anything with children is wholesome and age-appropriate.\n\nIf a box fails, apply the substitution table before writing the prompt.\n\n## Editing real footage (Kling O3 edit / Omni Flash edit) — real people in the SOURCE\n\nVideo edit takes the user's own footage, which often contains real people. Rules:\n\n- The user must hold rights/consent for any real person's likeness in footage they edit — ask once when it's clearly someone other than the user, then proceed.\n- Kling's video-to-video filter behavior on real faces is **not yet verified** (unlike Seedance, where the consent-gated real-face route is confirmed). If an edit of real-person footage is rejected by the provider, do NOT retry-spam variations — tell the user the filter blocked it and offer a no-face crop/segment or an AI-character swap instead.\n- **Omni Flash: own-footage editing of the uploader's own face PASSED live 2026-07-09** (real talking-head clip, edited on our fal route) despite Google's documented \"recognizable people\" restriction — treat that restriction as aimed at third-party/public figures, but expect probabilistic refusals and never promise passage.\n- Never use edit to put a real, named public figure into a scene, or to make someone appear to say/do something they didn't. Faceless b-roll (hands, products, landscapes, crowds-from-behind) edits freely.\n\n## Gemini / Omni Flash filter regime (video gen + edit) — receipts 2026-07-09\n\nGoogle's filter is its own regime (stricter than fal-hosted Kling about harm-to-a-person, looser than BytePlus about faces). Live receipts:\n\n| Blocked (`content_policy_violation`) | Passed |\n|---|---|\n| \"his fingertips **ignite** with a small real flame\" (fire ON a body part = harm) | \"small **magical** flames appear on his fingertips … vanish when he blows on them\" |\n\n- **Harm-to-person framing is the tripwire**, not the effect itself. Reframe body-contact effects as magical / supernatural / harmless VFX: \"magical flames\", \"a glowing aura\", \"sparks of light dance on\". Avoid ignite / burn / on fire / catch fire applied to a person.\n- **Never use a real object as a metaphor** — \"candle-like flame\" rendered a literal candle in the subject's hand. Describe the effect, not an object that resembles it.\n- The block is a 422 refund (no credits lost) and arrives mid-generation — one reframe per the substitution mindset above, don't retry-spam.\n",
|
|
10
|
+
"slates-cost-discipline": "---\nname: slates-cost-discipline\ndescription: \"Estimate and announce generation spend, aggregate batches, follow existing consent and inspect uncertain jobs before retrying. Use when planning or submitting paid media generation.\"\n---\n\n# Slates cost discipline — read before every generation\n\nGeneration costs real money. Every call is on the user's credits. The user can't see what you're about to spend until you tell them. **Tell them first, generate second.**\n\n## The 4 rules\n\n### 1. Pre-flight estimate — never call generate without one\n\nBefore ANY `slates_generate_*` call, run `slates_estimate_generation_cost` first. Inputs you must lock before estimating:\n\n- **Model** — the id you are about to pass, whatever it is. `slates_estimate_generation_cost` takes the same base ids the generate ops take and resolves the billing key itself; do not build one by hand.\n- **Resolution** — use the selected model default unless the user or delivery requires another size. Estimate and generate with the same settings.\n- **Aspect ratio** — never let the op default to 1:1. Pick from the use case (cinematic → 16:9, mobile vertical → 9:16, square feed → 1:1).\n- **Count** — explicit. Don't generate 4 when 1 will tell you if the prompt works.\n\nIf the aspect ratio cannot be inferred from the intended delivery, ask. A missing resolution uses the model default; it does not require another question.\n\n### 2. Announce in credits, plainly, before spending\n\nSlates bills abstract **credits** (they never expire). Announce the credit total the estimate returns — never dollars.\n\nFormat: `About to spend N credits on M image(s) at [resolution] [aspect ratio]. Proceed?`\n\nExamples:\n- `About to spend 4 credits on 1 image at 1k 16:9. Proceed?`\n- `About to spend 24 credits on 4 images at 2k 9:16 (variants). Proceed?`\n\nFollow the user's and host's generation approval policy before spending. A cost estimate or a small charge does not override a required prompt approval. An approved enumerated batch covers its calls; added calls or changed inputs need the confirmation described below. The code's separate confirm threshold is:\n\n<!-- @inject:thresholds -->\n<!-- GENERATED from @slatesvideo/shared — do not edit between the markers.\n Source: CONFIRM_CREDITS, DEVIATION_FACTOR and the audio bounds in\n packages/shared/src/operations/index.ts. Every number here is REFUSED by an\n op when a prompt gets it wrong, which is why none of them is typed by hand\n any more: this block replaced four claims that contradicted the code. -->\n\n**The thresholds, from the code that enforces them:**\n\n- **Confirm gate:** above **17 credits** an op returns `requires_confirm` and will not\n proceed until you re-call with `confirm: true`. This is a code gate, not permission to spend: every generation still needs the user-approved plan or quote.\n- **Deviation pause:** the desktop Studio Agent stops and re-asks when projected generation spend\n exceeds the approved plan by more than **20%**. You do not trigger this; the app does.\n- **Seed Audio duration:** **3–120 seconds.** There is no duration\n parameter on the model — the number you pass is written into the prompt AND is what the user is\n billed. Outside that range the op refuses rather than clamping.\n- **Sound Effects duration:** **1–22 seconds**, billed per second, never left for the\n model to pick.\n\nNever quote a credit figure from memory: `slates_estimate_generation_cost` returns the real one.\n<!-- @end:thresholds -->\n\n### 3. Aggregate batches into ONE upfront announcement\n\nIf you're planning a multi-call workflow (5 storyboard frames, 3 character variants, a grid of options), **announce the total before the first call**, not five small announcements after the fact.\n\nFormat: `Plan: N generations totaling C credits. [Brief description of the sequence.] Proceed with the batch?`\n\nExample: `Plan: 6 frame generations at 1k 16:9 totaling 24 credits — establishing wide, push-in, two-shot, reverse, OTS, insert. Proceed?`\n\n### 3b. Batch authorization — one approval covers the enumerated batch\n\nWhen the user approves a batch plan with one aggregated cost total up front (\"8 scenes, ~$X total — go\"), that single approval authorizes `confirm=true` on **each enumerated call in that batch** — and nothing beyond it. You do not need to re-ask per call; that's the point of the upfront announcement. Hands-off multi-scene runs depend on this.\n\nBoundaries that re-trigger confirmation:\n\n- Any call's actual estimate exceeds what the announced plan implied → stop, surface the delta, get a fresh OK. The app enforces its own ceiling on top of this (see the thresholds above); do not wait for it to catch you.\n- New calls are added that weren't in the enumerated plan (extra variants, retries beyond the plan, a new scene) → those are NOT covered. Announce and confirm separately.\n- The batch scope changes (different model, resolution, or duration than announced) → re-announce, re-confirm.\n\nOne approval = that plan, as enumerated, at those prices. Nothing else.\n\n### 4. Track the running total\n\nAfter each generation completes, the response includes `cost_credits` (when available). Keep a running tally in your context. Surface it every 3 generations or whenever the user asks \"how much have we spent?\"\n\n## Resolution decision rules\n\nUse the model defaults below for ordinary work. For cheap exploration, choose a supported lower setting and compare the quote. For final delivery, use the size the output needs. When comparing prompts, hold model, quality, size and references constant.\n\n<!-- @inject:image-defaults -->\n**Image default:** gpt-image-2-5-sunburst, quality `high`, 3k. User overrides take priority. Without a project, generation uses the headless Nano Banana 2 seat.\n\n| Model | Default resolution |\n|---|---|\n| nano-banana-2 | 2k |\n| nano-banana-2-lite | 1k |\n| nano-banana-pro | 2k |\n| gpt-image-2-5-flare | 2k |\n| gpt-image-2-5-sunburst | 3k |\n| flux-2-max | 1k |\n| seedream-5-lite | 2k |\n<!-- @end:image-defaults -->\n\n**4K VIDEO is Pro-only (2026-07-07).** The ladder above is for IMAGES (open at every tier). For VIDEO, where the selected model supports 4K, it requires a Slates Pro account; a base-tier 4K video gen is rejected server-side with `PRO_REQUIRED`. Default video to 1080p or lower and only reach for 4K when the user is on Pro and explicitly asks. 4K *images* are never gated.\n\n## Aspect ratio decision rules\n\nAsk the user when ambiguous. Otherwise:\n\n| Context cue | Aspect ratio |\n|---|---|\n| \"cinematic\", \"film\", \"movie\", \"wide\" | 16:9 |\n| \"TikTok\", \"Reels\", \"Story\", \"mobile vertical\", \"phone\" | 9:16 |\n| \"square\", \"Instagram feed\", \"thumbnail\" | 1:1 |\n| \"ultra-wide\", \"anamorphic\", \"cinemascope\" | 21:9 if supported; otherwise 16:9 |\n| \"portrait\", \"magazine cover\", \"vertical\" | 3:4 |\n| \"landscape photo\", \"horizontal\" | 4:3 |\n\nPick a ratio the chosen model accepts; check its `aspectRatio` options before submitting.\n\nIf the user prompt mixes signals (e.g. \"cinematic Instagram post\"), ask. Don't guess.\n\n## When the gate fires\n\nThe server returns `requires_clarification` when required composition inputs are missing, and `requires_confirm` when total spend crosses the gate above. In both cases:\n\n1. Surface the gate response to the user\n2. Get a clean answer\n3. Re-call with the explicit values + `confirm: true` if applicable\n\nDon't bypass the gate by silently filling in defaults. The gates exist because defaults waste money.\n\n## Video is slow + async — a timeout is NOT a failure\n\nVideo gens take minutes (Seedance 4K can run far longer). A client/CLI timeout or a slow, empty-looking response is **not** a failed generation — the job is still running on the provider.\n\n- **Never re-submit a video gen because it \"timed out.\"** That double-charges the user for one video. Re-rolling a slow gen is the single most expensive mistake here.\n- **Poll, don't re-roll.** Use `background: true` on `slates_generate_video`, then poll `slates_get_generation_status` (free, read-only) until it reports `completed` or `failed`. In-flight jobs survive app restarts and are recovered.\n- A gen has only failed when the status comes back `failed` — and a provider *rejection* **refunds** the credits, so failed isolation tests are ~free. Until you see a terminal status, the job is in flight. Wait.\n\n## 🔴 The still-gate — the most expensive mistake in the pipeline\n\n<!-- @inject:still-gate -->\n**Inspect a start frame before animating it.** Repair a visible defect that would make the intended crop or performance unusable before spending on motion. A clean frame can be animated whenever the brief calls for movement; this check does not require an image stage for text-to-video.\n\nThis is a cost rule as well as craft: a premium video call can cost many times an image correction. Broken geometry can turn to mush, oily textures can crawl and malformed objects can fall apart in motion. Fix a known source defect at the source instead of buying a more expensive copy. Judge intentional stylisation against the brief, not a universal photoreal standard. Additional image or video requests still follow the existing generation authorization.\n<!-- @end:still-gate -->\n\nThe check itself lives in `slates-vision-feedback-loop` (the five slop tells and the per-model accents). The **stop** is a cost rule and belongs here: before every image→video call, confirm the source frame passed the still scan. If it didn't, spending video credits on it is not iteration; it is buying a more expensive copy of a defect you already found.\n\n<!-- @inject:iteration-diagnosis -->\n## Diagnose repeated failures\n\nAfter three failed attempts at the same requirement, pause unchanged re-rolls and diagnose the source reference, prompt structure, model fit and tool result. Three is a review checkpoint, not a universal limit or proof that the seed cannot matter. Preserve the attempts and name what each test changed.\n\nContinue autonomously when the brief is clear, a specific correction is supported and the next request is already authorized. Hand control back when taste or intent cannot be inferred, the next request needs fresh consent, or the available tool cannot meet the requirement. A failed roll never authorizes an additional charge. Follow the existing batch and per-request cost policy.\n<!-- @end:iteration-diagnosis -->\n",
|
|
11
|
+
"slates-dialogue-blocking": "---\nname: slates-dialogue-blocking\ndescription: \"Block a multi-character conversation (at a table, in a car, walking) in Blender when seating, eyelines or screen direction must hold across cuts, or generation has lost that continuity. Includes performance and off-screen dialogue direction.\"\n---\n\n# Dialogue blocking — six people who stay where you put them\n\nUse this when fixed seating, eyelines or screen direction across cuts are part of the brief, or when generated shots have lost that continuity. A simple conversation can use ordinary shot and reference direction; 3D blocking is a tool for deterministic geography, not a prerequisite for dialogue.\n\n## Why this is hard\n\nEvery cut is an independent guess unless something forces agreement. Prompt a six-person conversation four times and you get four different seating charts: characters swap places, the 180-degree line breaks, and nothing cuts together. The failure is not aesthetic — the shots are simply unusable as an edit, and you find out only after paying for all four.\n\n**The blocking fixes it structurally.** Positions exist in 3D, so every camera sees the same arrangement, and consistency stops being something the model has to remember.\n\n## Build\n\nWhen a blocking pass serves the brief, follow `slates-previs-blocking` and add the relevant controls below. The examples show how to prevent observed seating and performance failures; choose controls for the scene rather than requiring every example in every conversation.\n\n### Seated proxies, colour-coded\n\nSimple seated shapes. **Do not animate the heads** — a proxy head turning the wrong way is worse than one that never turns.\n\nGive each character a distinct viewport colour and write the mapping down. This is the identity channel:\n\n> red = the boss · green = the kid · blue = the driver · yellow = the fixer · purple = the cousin · cyan = the nephew\n\nThat mapping goes verbatim into the generation prompt. It is what lets the model bind a grey body to a character sheet across four cuts.\n\nGive every proxy a material and set `mat.diffuse_color` to its identity colour — the blocking render pins Workbench to `MATERIAL` shading, so **the material's `diffuse_color` is what reaches the clip**. Set `object.color` to the same value too, so the user's viewport matches what renders. See `slates-previs-blocking` for the snippet.\n\n### Fix the geography, then never move it\n\nPlace people once. Write down who sits where relative to the camera's opening position, in words, because that sentence is going into the prompt:\n\n> Across the table, facing camera: yellow dead centre, purple far left, blue and green to the right.\n\n### The camera plan\n\nPer `slates-camera-language`, with two things specific to dialogue:\n\n- **Below shoulder height, slow rail glides** suit the table-conversation register in these examples; higher angles can intentionally read as surveillance. Choose the height and movement that serve the intended performance.\n- **Decide who owns the near foreground in each cut and honour it.** A shoulder in frame is a spatial anchor; a different shoulder in the next cut relocates the whole room.\n\nThe move that earns the most: **a gaze handoff without a cut** — the camera keeps gliding while the target hands off across the table, face to face, slowing on each but never stopping. Build it by keyframing the Track To target's position between subjects.\n\n### Crossing behind someone\n\nA head wiping frame during a move is a strong depth cue. It is also a spatial claim, so pick who gets crossed and say so — *the camera crosses directly behind cyan's back mid-shot and his head wipes the frame once.*\n\n## The prompt\n\nEverything in `slates-blocking-to-prompt`, plus these blocks.\n\n### Geography — restate it as a rule\n\n> TABLE GEOGRAPHY — do not deviate: the camera is never parked behind red. Only in the opening seconds does his dark shoulder hang at the near frame RIGHT edge, and it slides out as the camera travels LEFT. The true near-foreground of this shot is cyan: the camera crosses directly behind him mid-shot. After the opening seconds red is gone from the foreground, and the camera never travels behind anyone except cyan.\n\n### Screen direction, and the mirror that is not a swap\n\nThe 180-degree rule survives on its own in the blocking. What breaks is the model **\"correcting\" a legitimate mirror** — when the camera faces back through a scene, sides invert, and that inversion is correct. Say so explicitly or it gets flipped:\n\n> The driver's seat is on the LEFT for the entire timeline; this layout never mirrors or flips. When a camera faces BACKWARD into the car, screen sides mirror naturally: the driver reads on the RIGHT of frame, the passenger on the LEFT — that is correct left-hand drive, not a swap. They never swap seats or roles anywhere in the timeline.\n\nThen compress it into the HOLD block: *(backward camera mirrors them: he right of frame, she left)*.\n\n### Presence\n\n> A seated person stays drawn even when partially occluded — in every interior frame some part of each seat's owner is visible: a hand, an arm, a shoulder, a head above the bolster. Every occupied seat visibly holds its person.\n\n### Keep everyone alive\n\nThree orthogonal layers. Without them, whoever is not speaking freezes:\n\n- **ONGOING BUSINESS** — a small continuous physical action per character, running whether or not they are speaking. *turns his glass a quarter every few seconds · thumbs a lighter without lighting it.*\n- **BACKGROUND LIFE** — soft-focus, low contrast, never pulls attention, never crosses in front of a speaking face.\n- **SCENE EVENT** — the unnamed thing everyone is playing but nobody says. One line, repeated verbatim in every character's direction: *keep tomorrow sounding like a fishing trip.*\n\n### Acting, per character\n\nSame six slots each. Terse:\n\n```\nACTING TASK — <character>\n SCENE DIRECTION (shared, unspoken): <the same line for everyone>\n MOTIVE (his fuel): <what he wants underneath>\n GOAL: <what he wants in this scene>\n OBSTACLE: <what is in the way>\n TACTIC: <how he goes about it>\n Moment to moment: <2-3 beats keyed to timestamps>\n (Safety: gaze always engaged in the task — never a frozen, glassy,\n unfocused stare; natural blink cadence.)\n```\n\nThat safety line is not filler. Dead eyes are the characteristic failure of generated faces in dialogue, and naming it is what prevents it.\n\n### Split any strong emotion into phases\n\nThe other characteristic failure is a face that strikes one extreme expression and holds it for the whole shot — a mouth stuck open for three seconds. Give the beat two phases with a hinge, and name the failure you are excluding:\n\n> PHASE 1 (17.3-18.6s) — she SCREAMS at him, mouth wide, eyes huge, hand clamped on the grab handle. PHASE 2 (18.6-19.9s) — the scream breaks off: she shuts her eyes tight and CLOSES her mouth, both hands now on the handle, head ducked, braced. Scream, then brace — never one frozen open mouth held through the whole shot.\n\nThe hinge timestamp is what makes it a performance instead of a pose.\n\n### Dialogue must not restructure the edit\n\nWhen the blocking owns the cut list, include both constraints to prevent dialogue from inventing coverage:\n\n> DIALOGUE NEVER CREATES SHOTS: spoken lines happen inside the reference's takes exactly as blocked — no cutaways to a speaker, no reverse shots, no added close-ups. If a line plays while the camera is elsewhere, the line stays off-screen audio.\n\n> OFF-SCREEN VOICES RULE: a line marked off-screen must STAY off-screen — never show the speaker, never move him into frame, never route the camera behind him because he spoke.\n\nA sentence may cross a cut. Say so where it does: *the sentence does not pause for the edit.*\n\n## Model routing\n\nDialogue directed as separate layers (voices, scene sound, score) is **minimax-h3**'s seat; it also takes declared reference relationships, which suits a colour-coded cast. seedance-2.5 carries the reference-video capacity. Route per `slates-model-selection` and read the chosen model's prompting skill before writing the audio block.\n\n## Checklist\n\n- [ ] Colour→character mapping written down and pasted into the prompt\n- [ ] Heads not animated in the blocking\n- [ ] Seating stated as a geography rule\n- [ ] Foreground owner named per cut\n- [ ] Mirror-is-not-a-swap clause present if any camera faces back through the scene\n- [ ] Presence rule present\n- [ ] Ongoing business, background life and scene event all specified\n- [ ] Acting task per character, safety line included\n- [ ] Any strong emotion split into phases with a hinge timestamp\n- [ ] Dialogue-never-creates-shots and off-screen-voices rules present\n\n## Related\n\n`slates-previs-blocking` · `slates-camera-language` · `slates-blocking-to-prompt` · `slates-character-identity` (the sheets) · `slates-prompting-minimax-h3`\n",
|
|
12
|
+
"slates-direct-response-ad": "---\nname: slates-direct-response-ad\ndescription: \"Develop a product-led direct-response ad from an offer and references. Use for demonstrations, proof, argument and the next action; combine with script craft and presenter direction as needed.\"\n---\n\n# Product-led direct response\n\nUse `slates-script-craft` for the words, evidence and hook/bridge variations. This guide supplies an optional product-led approach. A person can appear; a presenter, fixed duration, fixed frame count or hyper-motion treatment is not required.\n\n## Choose the product's job in the picture\n\nRead the brief and references. Identify the intended audience situation, the supplied claim, what can demonstrate it, and the actual next action. A close product detail, use in context, a visible problem/solution and a final product view are possible ingredients. Omit or reorder any that do not serve the piece.\n\nFor example, a keys tray can interrupt a sliding key, show where it lands, then hold on the result. A quiet demonstration may work without narration. A technical product may need an explanatory exchange. Do not invent product finishes, guarantees or quantified benefits from a reference picture.\n\n## Keep the work editable\n\nUse the current project unless another destination is requested. Save the script as document text, with non-spoken direction separate. Create shots only for the requested production units and file them into explicit scenes. Alternative openings live beside the shared body as saved versions of a section. Preserve manual edits and reference identity.\n\nRead the composed requests before any media call. Model selection, reference slots, durations and resolution come from current capabilities and `slates-model-selection`, never a copied ad recipe. Every prompt stays visible on its shot. A preset imports ordinary editable content and does not authorize generation.\n\n## Production when requested\n\nFollow `slates-cost-discipline` and the user's authorization for the exact selected requests. Estimate the set through the existing quote operation. An extra stochastic take remains an extra requested take; compatible existing media can be deliberately reused.\n\nInspect results against the demonstration and supplied references. A failed job is not authorization for an unchanged reroll. Keep earlier takes accessible. Assemble into an explicitly named cut when comparing variations, inspect playback, and export the selected cuts with their results. `slates-one-prompt-film` covers this delivery task when a finished video is requested.\n\nNo conversion outcome is promised by this format. Creative clarity, observed distribution and measured purchases are different evidence.\n",
|
|
13
|
+
"slates-edit-and-iterate": "---\nname: slates-edit-and-iterate\ndescription: \"Refine an existing Slates image with a targeted edit or revised generation. Use for changes such as warmer light, removing a figure, reframing or a different look; preserve the original and its lineage.\"\n---\n\n# Edit and iterate — Slates workflow\n\nThe user already has a generated image in Slates and wants to refine it. The vision-feedback-loop skill defines the general pattern; this skill is the specific recipe for \"I have asset X, here's what's wrong with it.\"\n\n## 🔴 The master rule — an edit is a LEAF, not a node\n\n**Never re-edit an edit. Always go back and re-edit the master.**\n\nEvery edit model silently re-renders the **whole frame**, not just the region you named. So the parts you didn't ask to change come back slightly different every pass — softer texture, drifted colour, mushier fine detail. It is barely visible after one edit and obvious by the second. Chaining edits compounds the damage and there is no way to undo it, because each generation *is* the new source.\n\nThe fix is structural, not a matter of care:\n\n- **Want two changes?** Make them in ONE edit off the master, or make them as two separate edits **both taken from the master**, then keep whichever you prefer.\n- **An edit came back wrong?** Do NOT edit the result to fix it. Discard it and re-edit the master with a better instruction.\n- **Only the changed region is worth keeping?** That is a compositing job — the edit supplies the new region, the untouched master supplies everything else.\n\nSlates records this: an edit result carries `sourceAssetIds` pointing at the asset it was made from, so **you can tell whether the thing you are about to edit is itself an edit.** Check before you edit — `[Edit]`-prefixed prompts and a populated source lineage both say \"this is a leaf; go back to its parent.\"\n\n## Workflow\n\n### 1. Pull the current asset back into context\n- The user references an asset by id, frame number, or \"the latest one.\"\n- Resolve to an asset id (`slates_list_assets` if needed).\n- `slates_get_asset_image` with that id to load it inline. **You see the image.**\n\n### 2. Identify the delta\nThe user's request is one of:\n- **Surgical** — \"remove the second figure\", \"make the sword red\", \"swap the background to a bamboo forest\".\n- **Aesthetic** — \"warmer light\", \"more dramatic\", \"softer focus\".\n- **Compositional** — \"wider shot\", \"lower angle\", \"centered subject\".\n- **Wholesale** — \"actually let's try a totally different look.\"\n\n### 3. Pick the right tool\n| Delta type | Approach |\n|---|---|\n| Surgical | `slates_edit_image` — `sourceAssetId` = the original, `prompt` = the change only (\"remove the second figure\"), not a re-description of the whole image. |\n| Aesthetic / compositional | `slates_generate_image` with the original in `referenceAssetIds` + a refined prompt. Don't re-roll from scratch. |\n| Wholesale | New prompt, no reference, fresh generation. Treat as a new brief. |\n\n**`slates_edit_image` shape:** `projectId` + `sourceAssetId` + `prompt` (the edit instruction). Omit `editModel` for the app's Edit seat (the default image model). Every edit model also takes `referenceAssetIds`, up to its reference cap less one: the source is image 1. The result lands as a NEW asset (prompt prefixed `[Edit]`); the source is untouched. Follow the current estimate and returned confirmation gate; an endpoint threshold does not grant consent to spend.\n\n### 4. Generate, evaluate, decide\n- Estimate cost first.\n- After generation, the result is inline. Compare side-by-side with the original (`slates_get_asset_image` again).\n- If the delta is correct: bind to the same role (frame, character identity, etc.) the original was bound to.\n- If the delta missed: diagnose the visible mismatch, refine the instruction and re-edit the master within the approved request or batch.\n\n<!-- @inject:iteration-diagnosis -->\n## Diagnose repeated failures\n\nAfter three failed attempts at the same requirement, pause unchanged re-rolls and diagnose the source reference, prompt structure, model fit and tool result. Three is a review checkpoint, not a universal limit or proof that the seed cannot matter. Preserve the attempts and name what each test changed.\n\nContinue autonomously when the brief is clear, a specific correction is supported and the next request is already authorized. Hand control back when taste or intent cannot be inferred, the next request needs fresh consent, or the available tool cannot meet the requirement. A failed roll never authorizes an additional charge. Follow the existing batch and per-request cost policy.\n<!-- @end:iteration-diagnosis -->\n\n### 5. Hand back\n- \"Asset updated. Frame 3 now uses {new_asset_id}.\"\n- Always note what changed and what didn't, so the user can see the surgery worked: \"Lighting shifted to warmer, composition unchanged.\"\n\n## Anti-patterns\n\n- **Don't** delete the original asset until the user confirms the new one. Slates keeps both; the user picks.\n- **Don't** mix surgical and wholesale changes in one regeneration. The user said \"make it warmer\" — don't also reframe the shot.\n- **Don't** re-generate when `slates_edit_image` would work. Edits preserve composition and identity; full regen rolls the dice.\n- **Don't** edit an edit — ever. Not once, not \"just a small one.\" Go back to the master (see the master rule above). Every attempt re-renders the full frame and the degradation is cumulative and permanent.\n- **Don't** keep re-rolling the same failed edit. Inspect whether the reference, edit instruction or chosen model caused the miss; use the repeated-failure checkpoint above.\n",
|
|
14
|
+
"slates-model-selection": "---\nname: slates-model-selection\ndescription: \"Choose image, video, edit and audio models for the brief, including animating photos, making films or directing voices. Load before choosing or defaulting a model or quoting a plan; retrieve model-specific craft after selection.\"\n---\n\n# Model selection for the intended piece\n\nChoose from the current catalogue using the brief, existing media and delivery requirements. The agent makes this production choice; the user supplies the vision, explicit preferences and approval. A named model takes priority when it can do the requested job. If it cannot, explain the specific conflict and choose a supported route within the brief.\n\n## Decide from constraints and evidence\n\nName the primary must-preserve requirement and any other hard constraints: this face stays this face, the fluid behaves like fluid, text remains legible, the voice is specific, or the take remains unbroken. Budget and delivery format can be binding requirements rather than afterthoughts.\n\nChoose a seat whose capability and observed craft fit those requirements. Inspect at the intended delivery crop: atmosphere, material texture and anchor objects for a location; identity, skin, pose and gradients for a character. A thumbnail is insufficient evidence for a large final frame. When the roster changes, repeat relevant comparisons rather than inheriting reputation.\n\nThe catalogue below is generated from the same model facts used by the tools. Current allowed settings and reference inputs come from the tool schemas and `slates_list_available_models`; current prices come from `slates_estimate_generation_cost`. Retrieve the selected model's craft card, then only the sections needed for the shot. Historical measurements below explain a choice; they are not current price quotes or permanent rankings.\n\n## Current catalogue\n\n<!-- @inject:model-routing -->\n**Current model routing, generated from the operation routing source:**\n\n### image generate\n\nNano Banana 2 (Gemini 3.1 Flash Image): The all-rounder and the only image seat with a headless path: holds many subjects coherently in one frame, and the start-frame for legible in-scene text. Knowledge cutoff Jan 2025: anything later needs reference images.\nNano Banana 2 Lite: FAST/DRAFT image tier — markedly cheaper and faster than NB2 full, at draft quality. Route here for iteration volume, then re-run the winner on NB2 full. Same Gemini content filter as NB2.\nNano Banana Pro: HERO-FRAME / typography PREMIUM image tier. NB2 is about 95% of Pro — escalate only when spatial composition, cinematic lighting/skin, fine typography-in-scene or deep multi-element reasoning must be perfect, and say why.\nGPT Image 2.5 Flare: THE FAST GPT IMAGE SEAT — OpenAI's small model, optimized for SPEED, quality COMPARABLE to GPT Image 2 (not better) at roughly half the latency. Route here when speed matters: drafts, exploration, volume. TEXT / DIAGRAM / PANEL work — character sheets, shot grids, text-bearing panels. When quality outranks speed, escalate to Sunburst. Own content filter, distinct from Gemini's. Killed by a head-to-head at the intended crop going the other way.\nGPT Image 2.5 Sunburst: THE QUALITY GPT IMAGE SEAT — OpenAI's most capable image model, higher quality than GPT Image 2, same price as Flare, deliberately SLOWER. Route here unless speed is the point: finals, hero frames, photoreal people, and multi-reference edits where every reference must survive into one frame — its widest lead. Explore on Flare, finish on Sunburst.\nFLUX.2 Max: Photoreal image seat, less censored than the Gemini rails. Auto-routes to its edit endpoint when references are present.\nSeedream 5 Lite: Cheapest flat-priced image seat (GPT Image 2.5 at low quality costs less per image). Less censored. Routes to its edit endpoint when references are present.\n\n### video generate\n\nSeedance 2.0: THE 4K AND VALUE SEAT beside the 2.5 default — the only Seedance with native 4K (Pro-gated; base accounts get PRO_REQUIRED) and cheaper than 2.5 at every resolution they share, with the same physics, effects and scale strengths; shorter takes, fewer references, no timestamps. VIDEO-ONLY. A bare \"seedance\" still resolves here for older CLIs that expect 4K.\nSeedance 2.5: DEFAULT VIDEO MODEL — the strongest seat for physics, effects, scale and hero shots, and the only Seedance that takes long single takes, many references, audio-only references and integer-second timestamps. No 4K, and dearer than 2.0 at every shared resolution: go to 2.0 for 4K or the same resolution cheaper. LENGTH is the price dial — quote long takes first. VIDEO-ONLY. Timestamp grammar and the edit/extend words that make the provider reclassify and fail a generation are in slates-prompting-seedance-2-5.\nKling 3.0: THE COST-EFFECTIVE SEAT — strong start-frame adherence (identity, layout, text), acting, dialogue and lip-sync; pick it when the budget matters and the shot is a performance or a start-frame animation. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools.\nGemini Omni Flash: 720p seat with native synced audio included. Route here for drafts with sound in one pass and reference-to-video character-consistency trials; LTX, H3 and H3 Max Turbo cost less per second. VIDEO-ONLY. Quality against Kling/Seedance is unproven — do not route hero shots here.\nMiniMax H3: THE AUTHORED-AUDIO SEAT — reach for H3 when the sound is part of the shot rather than a switch on it: synchronised dialogue, scene sound and an audience-only score directed as three separate layers in ONE pass, across eleven languages. Kling and Seedance treat audio as on/off. Only H3 also carries a DECLARED REFERENCE RELATIONSHIP (kept whole, partly kept, transferred, or a loose echo). VIDEO-ONLY. Its top two resolution tiers are UPSCALES of the native render, not larger generations — judge at native and upscale in post. Reference images past the fifth are a PAID key dimension: pass referenceImages when quoting.\nMiniMax H3 Max: THE SPEED SEAT, dearer than base H3 at 768p and equal at 480p — never the cheap H3 and never the default. fal's post-train of the H3 weights: MEASURED 2026-08-27 at about 12x faster than base H3 on the same prompt and params, queue to finished file, plus a thin vendor-reported quality edge. It tops out at a 1080p refinement of its 768p render. It takes the same omni-reference set as base H3 and animates start and end frames — but not both in one call, the same as base H3: frames and references go to different endpoints. Never describe this row as taking no image or reference input. Route here when a fast turnaround on text-to-video or a start-frame shot is worth the premium.\nMiniMax H3 Max Turbo: THE BUDGET SEAT of the MiniMax family: a second fal post-train of the H3 weights, billed at half H3 Max's rate at every tier. Its 1080p is a refinement of the native 768p render, not a native 1080p generation. INPUTS ARE FRAMES, NOT REFERENCES: text-to-video and start/end frames only, with no reference endpoint, so reference-driven consistency goes to H3 Max or base H3. Route here for drafts, volume and cheap coverage, then re-run the keeper on H3 Max or a hero seat.\nLTX-2.5: THE VOLUME SEAT — the cheapest 1080p second with sound included, and the row for MANY takes rather than one hero shot. Native synced audio is included free at every tier, unlike Kling where sound is a paid key dimension. It supports native high-resolution output and longer takes than most seats; use the capability surface for its resolution-dependent duration limits. VIDEO-ONLY. INPUTS ARE FRAMES, NOT REFERENCES: start frame plus an optional end frame, and no reference endpoint at all — for character consistency across shots use H3 or Kling. Route here for batch coverage, long takes, and anything where the credit budget is the binding constraint.\nLTX-2.5 Pro: THE FIDELITY SEAT of the LTX pair — the full diffusion build against the base row's distilled one. 🚨 IT IS NOT A SUPERSET OF THE BASE ROW, which is the opposite of every other Pro seat here: it reaches a SHORTER resolution ladder and makes SHORTER clips, and it costs more at both tiers they share. Reaching for it because the name says Pro costs more AND takes away reach. Everything else matches the base row. Route here only when a specific shot needs the fidelity and fits inside its narrower envelope.\n\n### video edit\n\nSeedance 2.5 Edit: VIDEO-TO-VIDEO EDIT via slates_edit_video, and the only edit engine that takes a clip longer than the other two reach — that length is the whole reason to route here. Inside their range, compare on fidelity instead: Omni Flash edit won the prompt-only head-to-head, and Kling edit is the one that takes reference images. Edits audio on the same row (re-voice, re-accent, translate with re-fitted lips, replace BGM). Costs about 1.2x a plain 2.5 generation of the same length: an edit bills at twice the reduced video-reference rate.\nKling O3 Video Edit: VIDEO-TO-VIDEO EDIT, the REF-DRIVEN one: it is the only edit seat that takes element/style reference images to lock subject identity, and its keepAudio preserves the original audio verbatim. Route here when an edit NEEDS reference images or bit-exact audio; for prompt-only footage-synced VFX, omni-flash-edit won the fidelity head-to-head. One instruction beat per pass — multi-beat prompts get under-executed.\nOmni Flash Edit: VIDEO-TO-VIDEO EDIT, prompt-only — THE EDIT-FIDELITY WINNER (head-to-head vs Kling edit on real talking footage: lips held, audio near-identical, both action beats landed), priced level with Kling O3 Edit Standard. Footage-synced prop, effect, environment and lighting swaps. Takes NO reference images — identity swaps needing refs go to Kling edit. Fidelity is EARNED by prompt discipline; the exact form is in slates-prompting-omni-flash.\n\n### audio generate\n\nSeed Audio 1.0: DEFAULT audio model — the one-pass SCENE workhorse: dialogue, SFX and ambience together from ONE plain sentence. Route here for continuity beds, room tone, crowd and nature soundscapes, and quick scratch VO. AUDIO-ONLY. Takes one image XOR up to three audio clips as references, never both. Prompt form and the length rule are in slates-prompting-seed-audio.\nElevenLabs Sound Effects v2: ONE-SHOT SOUND EFFECT with an EXACT duration — route here for a single hit that must land on a frame (door slam, whoosh, impact, UI blip) or for a seamless loop. AUDIO-ONLY. For layered scenes with dialogue or room tone, seed-audio does it in one pass instead.\nInworld Realtime TTS-2: THE VOICE SEAT — one named voice saying one line, billed per CHARACTER not per second. Route here when WHO is speaking matters. NOT scene audio — that is seed-audio; a single effect is eleven-sfx.\n<!-- @end:model-routing -->\n\n## Seedance craft triggers\n\nSlates field experience identifies these useful beats:\n\n- Real-time to slow-motion contrast.\n- A moving camera while debris, meteors, sparks or particles move around the subject.\n- Massive scale whose size is the point of the shot.\n- A continuous unbroken take.\n\nThese concrete cues are more useful than an abstract label such as “physics.” Compare them with the user's budget, sound, reference and delivery constraints. They suggest a candidate; they do not override an explicitly chosen model or establish that every other seat fails.\n\n## Dated production comparisons\n\n| Receipt | What was observed | How to use it |\n|---|---|---|\n| MiniMax turnaround, 2026-08-27 | Same prompt and parameters: a five-second 768p H3 Max clip reached the finished file in 4.8 seconds, against 57 seconds on base H3, about twelve times faster. | Evidence for a turnaround requirement. The then-observed rates were $0.080/s on Max against $0.060/s on base, a premium rather than a saving. Re-quote current settings; queue conditions and provider revisions can change the result. |\n| Prompt-only footage VFX, 2026-07-09 | Omni Flash Edit preserved lip movement, returned near-identical audio and landed both action beats; Kling missed a beat and drifted the lips. Omni sometimes doubled a final speech beat or jittered at the tail. | Use the short change-only prompt demonstrated in `slates-prompting-omni-flash`; trim a defective tail where that solves it. This comparison does not prove exact audio preservation. |\n| Style-heavy relocation, 2026-07-09 | Seedance video-reference regeneration lost to Omni Flash Edit on the tested photoreal insert at 720p, while costing about three times as much in that comparison. | Transfer intensity and reconstruction are different jobs from surgical edits. Choose for the required change and current endpoint, not the historical label “premium.” The video-reference lane regenerates, bills input plus output seconds (at face-lane rates when people are in frame) and takes long descriptive prompts without Omni's hard-fail on timing phrasing; 2.5's lane reached 1080p on 2026-08-24. Route there for transfer intensity or a higher resolution ceiling, never as the cheap default. |\n| Photoreal skin, 2026-08-24 | One comparison favoured GPT Image 2 at its then-high tier. | This is evidence about that old model, crop and comparison. It does not establish the quality tier either GPT Image 2.5 variant needs. Raise quality to address an observed shortfall. |\n\n## Existing footage and edit fidelity\n\nRead the current video-edit catalogue before choosing an engine. Reference-driven identity changes, exact original audio, clip length and output size are distinct constraints; an engine that handles one may not handle the others. Use `slates_edit_video` for an edit endpoint, and the selected guide for its request syntax.\n\nWhen a clip is mostly right, compare an edit with a new generation before gambling away the useful parts. Every video edit engine can re-synthesise the whole clip: “change only this” describes the intention, not a pixel-level guarantee. For a critical deliverable, consider segment-splicing: edit the affected seconds, retain the original outside the change, and preserve the original audio underneath when needed. Phone footage must be rotation-normalised because players can honour a rotation flag that a model ignores.\n\nThe July comparison found Kling's original audio track retained verbatim with `keepAudio`, while regenerated lips could drift against it. Near-identical Omni audio was an observation, not a guarantee. For exact legal copy, narration or music, retain the original track and inspect the assembled playback.\n\nOmni Flash Edit needs a short change instruction plus “Keep everything else the same”; long identity-lock preambles worsened fidelity in that receipt. Kling multi-beat edits can drop an instruction, so a focused pass can be useful. Video edits can form a lineage of passes when that serves the work; this is distinct from the image master-edit rule in `slates-edit-and-iterate`. Every additional pass still needs the existing generation consent.\n\n## Motion transfer and lip sync\n\n`slates_generate_motion_transfer` and `slates_generate_lip_sync` expose dedicated Kling endpoints. The former retargets a driving clip onto a character image; the latter re-voices a clip or animates a portrait. Consult their schemas and guides for the current input and output limits, tiers and cost.\n\nA Seedance alternative is a normal `slates_generate_video` call with a video reference and explicit motion or dialogue direction, for example “the character from image 1 performs the exact motion from video 1.” This preserves an editable prompt and conditions the generation in one pass. Field experience favours it for fast choreography, contact, cloth and hair where post-hoc retargeting loses fidelity; compare for the specific performance rather than promising a universal win.\n\nVideo-reference calls bill from both input and output duration. Pass the actual reference durations when quoting; do not assume the output length is the whole charge. Face flags and any consented real-face route follow the selected endpoint's requirements and returned gate. A provider's face rejection is not permission for a more expensive retry.\n\n## Image and style production\n\nImage, video and audio are separate output lanes. A hero reference still is an image request, even when its final destination is video. Use the current image default for ordinary work and choose another seat when speed, supported shape, reference fidelity or an observed shortfall supplies a reason.\n\n<!-- @inject:image-defaults -->\n**Image default:** gpt-image-2-5-sunburst, quality `high`, 3k. User overrides take priority. Without a project, generation uses the headless Nano Banana 2 seat.\n\n| Model | Default resolution |\n|---|---|\n| nano-banana-2 | 2k |\n| nano-banana-2-lite | 1k |\n| nano-banana-pro | 2k |\n| gpt-image-2-5-flare | 2k |\n| gpt-image-2-5-sunburst | 3k |\n| flux-2-max | 1k |\n| seedream-5-lite | 2k |\n<!-- @end:image-defaults -->\n\nA styled start frame is useful when composition, exact in-scene text or an approved look must hold. It is optional for a video brief; direct text-to-video, imported footage and reference-video direction are other valid entries. `slates-style-prompting` supplies model-specific style craft after the route is chosen.\n\n## Sound as a production choice\n\nDetermine whether sound must generate with the picture, or become a separate editable asset. Native video sound can lock to visible action; a separate voice, effect or ambience bed can be moved, trimmed and reused on the timeline. The generated catalogue owns which models support each job.\n\n- “It needs to sound like a place” calls for a scene, not automatically several separately billed effects. Seed Audio can render dialogue, effects and room tone together from a plain sentence.\n- “Read this line” needs a voice decision: Inworld TTS-2 for a specified voice or clean narration; Seed Audio when the line belongs inside a scene. Measure and listen to the take before lip sync.\n- “That needs a thump right there” calls for a physical cause, the event's length and an exact placement in the cut. A dedicated effect can serve that job.\n- “Give it a track” requires an imported song: there is no standalone music-generation model in Slates. Video models' scene scores are a different capability.\n\nSeed Audio has no model duration parameter: Slates writes the requested length into the prompt and bills the requested duration. Never add a conflicting second duration to the sentence. Kling video labels such as `SFX:` and `Ambient noise:` do not transfer to Seed Audio; describe the sound directly. For fade handles, request extra bed length only when useful and include it in the quote.\n\nUse `slates-prompting-seed-audio`, `slates-prompting-inworld-tts` and `slates-prompting-elevenlabs` for their distinct sound and voice grammar. Return to the selected video guide for sound generated with video.\n\n## Spend and delivery\n\nRoute by the requirements, then compare quotes for settings that satisfy them. A cheaper unusable render costs more after correction, but a binding budget is itself part of the brief. Do not launch paid head-to-head comparisons merely because a table could be fresher. Follow `slates-cost-discipline` and the current generation authorization for the exact requested set.\n",
|
|
15
|
+
"slates-one-prompt-film": "---\nname: slates-one-prompt-film\ndescription: \"Turn a video brief into a finished, editable Slates production and verified export. Use for requests to make a film, short, trailer, music video, ad or other complete video, entering through writing, imported footage or an existing edit.\"\n---\n\n# Idea to finished video\n\nCarry the requested piece through to an exported file. Use existing work whenever it serves the brief. The creator can enter at any point: writing, importing footage, comparing takes or changing an edit. No mandatory stage sequence, shot count or number of approval checkpoints follows from this guide.\n\n## Make the intended piece visible\n\nUse the current project and document unless another destination is requested. Save words through the revision-checked document tools; `slates-script-craft` covers writing and versions. Add production bindings only where needed. A shot needs no image, and a cut can use imported footage with no script.\n\nPreserve fixed passages, explicit creative choices and custom prompt bytes. Record production choices in editable shots. Explain only consequential judgments not already visible there. Recurring cast, repeated framing, silence and dependent scenes are valid when they serve the piece.\n\n## Inspect the actual requests and estimate\n\nUse `slates_get_shot` to inspect the composed prompt, settings and references. Model choices and supported settings come from `slates-model-selection`, the current capability surface and the selected model's guide. Do not carry limits or prices from an old example.\n\nFollow `slates-cost-discipline` and the user's generation policy. Quote the exact requested set with `slates_generate_from_shots` before confirming it. Existing authorization covers its enumerated requests, not extra takes or changed inputs. Editing, choosing versions, importing and building a cut do not spend generation credits.\n\nKeep reusable historical media separate from new requests. Matching words alone do not prove matching voice, references or settings. An explicitly requested extra take is never deduplicated away.\n\n## Generate only the authorized material\n\nSubmit the chosen requests and inspect each returned state. On an uncertain timeout, read the shot's generation IDs and job status before any retry. Diagnose a failure and follow the existing consent policy for added requests. Never discard other takes merely because a new one was selected.\n\nInspect image composition and reference fidelity. Inspect video performance, motion and sound across playback; frame samples alone cannot establish speech or motion quality. If an edit can resolve dead air or order, use the existing take rather than assuming another generation is needed.\n\n## Arrange and deliver\n\nRead the available timelines. Name the destination cut explicitly for a variation; independent comparisons use independent cuts. Add selected media in the intended order, preserve trim/level/transform choices and inspect the actual timeline.\n\nUse supported video/XML exports and their stated fidelity limits. For selected named cuts, `slates_export_cuts` records distinct outputs and a manifest; retry unfinished outputs with the same manifest identity. Verify the returned files and playback before reporting success. Describe the completed piece, actual spend where available and output paths. Do not label a render complete merely because its submission succeeded.\n\n<!-- @inject:decision-log -->\nRecord production choices in the editable shot fields. Explain only consequential judgments the user did not specify and no field already records: for example, why a particular light or performance register supports the brief. Do not repeat the shot list in prose or turn this explanation into an approval gate. Follow the separate generation authorization policy before spending.\n<!-- @end:decision-log -->\n",
|
|
16
|
+
"slates-previs-blocking": "---\nname: slates-previs-blocking\ndescription: \"Build and render a Blender blocking pass for precise camera paths, cut timing or spatial continuity, then guide video generation with the clip. Use when those controls are required or prompting has failed to hold them.\"\n---\n\n# Previs blocking — design the shot, then generate it\n\nThe spine of the whole workflow. Read this first; the other four previs skills are branches off it.\n\n## The mechanism (why this works at all)\n\nA text prompt asks the model to *invent* camera motion, so it invents differently every roll. You cannot iterate on a variable you do not control, so you re-roll and pay again.\n\nA **reference video** removes the invention. You build the shot in Blender as untextured proxies — a neutral grey set with colour-coded figures, free, instant, deterministic — render the camera's path to mp4, and hand the model that clip alongside the prompt. **Blender locks the motion; the model builds the world.** Iteration moves to the free half, and the paid half usually lands first try.\n\nTwo halves, and keeping them separate is the whole discipline:\n\n| Half | Lives in | Changes when |\n|---|---|---|\n| **Structure** — cuts, camera, timing, who is where | the blocking clip | you re-block |\n| **Style** — what any of it looks like | references + prompt text | you restyle (see `slates-restyle-from-blocking`) |\n\n## Before you start\n\n1. `slates_blender_status` — confirms the bridge is up and returns fps, frame range, existing camera. If it reports `connected: false`, relay its hint and stop; nothing else here works.\n2. Settle **format first**, because the blocking render *is* the film's format: fps, aspect, duration. 24fps is the default and makes cut times land on clean frames. Duration ≤ 30s (seedance-2.5's reference-video ceiling; 15s on the others).\n3. Know the shot count. \"One take\" and \"19 cuts\" are different builds.\n\n## Build order\n\nDo these in order. Each stage is verifiable on its own, and a camera built before the geometry has nothing to frame.\n\n### 1. Set the format\n\n```python\nscene = bpy.context.scene\nscene.render.fps = 24\nscene.render.fps_base = 1.0\nscene.render.resolution_x, scene.render.resolution_y = 1920, 1080\nscene.frame_start, scene.frame_end = 1, 720 # 30s at 24fps\nresult = {\"seconds\": 720 / 24}\n```\n\nFrame maths, stated once so you never redo it in your head: **frame = seconds × fps + 1**. A cut at 7.79s is frame 188.\n\n### 2. Geometry and light — grey set, coded figures, named\n\nProxies only. A person is a box or a capsule with a sphere head. A car is a stretched cube. A can is a cylinder. **The SET is neutral grey — one light, a floor and enough wall that the space reads.** Colour is reserved for the figures, where it carries meaning (below); a grey set is what makes those few colours legible as notation rather than décor. Anything you spend on materials here you pay for twice, because the model repaints every surface anyway.\n\n**Name every object for what it *is* in the story**, not `Cube.003`. The name is how you refer to it later, and it is how you keep your own timeline honest.\n\nTwo conventions that cost nothing now and save a re-roll later:\n\n- **Colour is identity.** Give each character a distinct viewport colour and *write the mapping down* — `red = the boss, green = the kid, blue = the driver`. The generation prompt will restate that mapping so the model knows which grey body is which person across cuts. Without it, characters swap.\n- **Encode facing on featureless proxies.** A box has no front. Mark one face red, the back black, the sides green, and say so in the prompt: `RED face = the direction he faces`. Otherwise the model guesses which way people are looking.\n- **Checker a surface when SCALE or SPEED has to read.** Flat grey gives a model no parallax cue, so a fast move over a featureless floor reads as slow, and a big room reads as a small one. A black-and-white checker on the ground (or the wall a camera races past) gives it something to measure against. ⚠️ **Build it as GEOMETRY, never as a Checker Texture node.** The blocking render is Workbench, which draws one flat colour per material and never evaluates a shader node tree — a `TEX_CHECKER` comes out flat grey and you lose the cue without being told. Subdivide the plane and alternate `material_index` per face. Like every other colour here it is notation, so it goes in the translation list and gets dressed over.\n\n```python\n# Two materials, alternated per face. `TILE` is the square size in metres.\ndark = bpy.data.materials.new(\"Checker_Dark\")\ndark.diffuse_color = (0.05, 0.05, 0.05, 1.0)\nlight = bpy.data.materials.new(\"Checker_Light\")\nlight.diffuse_color = (0.80, 0.80, 0.80, 1.0)\nfloor.data.materials.append(dark) # material_index 0\nfloor.data.materials.append(light) # material_index 1\n# Subdivide first (edit mode or a Subdivide modifier applied) so there ARE\n# faces to alternate — a 2-triangle plane can only ever be one colour.\nfor face in floor.data.polygons:\n cx, cy = face.center.x, face.center.y\n face.material_index = (int(cx // TILE) + int(cy // TILE)) % 2\n```\n\nAnd the identity colour on each proxy:\n\n```python\nmat = bpy.data.materials.new(\"ID_Red\")\nmat.diffuse_color = (0.8, 0.1, 0.1, 1.0) # what the blocking render draws\nobj.data.materials.append(mat)\nobj.color = (0.8, 0.1, 0.1, 1.0) # same value, for viewport parity\n```\n\nThe blocking render pins Workbench to `MATERIAL` shading, so **`mat.diffuse_color` is the value that reaches the clip** — and an object with no material at all falls back to a neutral grey, which is why an unpainted set still reads correctly. Set `obj.color` to the same value anyway: it costs one line, it makes the user's viewport match what renders, and keeping the two equal means you never have to remember which one is authoritative.\n\n### 3. Camera\n\nThe whole of `slates-camera-language`. Build the rig, then keyframe it. Then **read back what you built** with `slates_blender_scene` — its `cutSeconds` is your cut list, and it is the number you will write timings against. That field is the authoritative one on EITHER rig — marker frames when cameras are bound to markers, the active camera's own keyframes when they are not. `camera.keyframeSeconds` is empty on a marker-bound edit, which is the rig `slates-camera-language` recommends for anything past a handful of cuts.\n\n### 4. Handheld, last\n\nAdd it after the moves are right, never before — noise on top of a wrong path just hides the wrong path.\n\n<!-- @inject:blender-action-curves -->\n## Read animation curves from the active action layout\n\nBlender 5 uses layered actions: curves belong to the channelbag for `animation_data.action_slot`, inside each layer's strips. A direct `action.fcurves` lookup failed on Blender 5.2.1 in the 2026-08-28 blocking run. Feature-detect the layout before changing interpolation or noise; an unanimated object can legitimately have no curves.\n\nThe snippets below use this small Blender-side iterator. It runs inside Blender; no add-on code is imported into the MCP package.\n\n```python\ndef action_curves(datablock):\n anim = getattr(datablock, \"animation_data\", None)\n action = getattr(anim, \"action\", None)\n if action is None:\n return\n if hasattr(action, \"fcurves\"):\n yield from action.fcurves\n elif getattr(anim, \"action_slot\", None) is not None:\n for layer in action.layers:\n for strip in layer.strips:\n if hasattr(strip, \"channelbag\"):\n bag = strip.channelbag(anim.action_slot)\n if bag is not None:\n yield from bag.fcurves\n```\n\nUse the datablock that owns the keyed property: the curve data for `eval_time`, the object for location and rotation, the camera data for lens. Confirm a named channel exists before assuming a keyframe operation created it.\n<!-- @end:blender-action-curves -->\n\n### 5. Verify the cuts\n\nThe one check that catches the most damage: on a multi-cut blocking, camera position, target and focal length must all change **exactly on the cut frame, with no transition frame between**. One interpolated frame reads as a whip-pan the model will faithfully reproduce.\n\n```python\n# On a jump-cut camera, key the final pre-cut pose at cut_frame - 1.\n# CONSTANT belongs to that preceding key: interpolation controls its OUTGOING segment.\n# Check object transforms and camera data (including lens). Repeat for a keyed target.\nfor owner in (cam, cam.data):\n for fc in action_curves(owner):\n keys = list(fc.keyframe_points)\n for previous, current in zip(keys, keys[1:]):\n if current.co[0] in CUT_FRAMES:\n assert previous.co[0] == current.co[0] - 1, \"Key the final pre-cut state first\"\n previous.interpolation = 'CONSTANT'\n```\n\nAlso check nothing interpenetrates — proxies through floors, clones through the hero object, letters through each other. The model renders intersections as faithfully as it renders everything else.\n\n### 6. Save a backup after every stage\n\nCheap, and blocking is iterative by nature.\n\n```python\nbpy.ops.wm.save_as_mainfile(filepath=path, copy=True)\n```\n\n## Render and generate\n\n```\nslates_blender_render_blocking { projectId, fps: 24 }\n```\n\nRenders the **scene camera** through scene settings and imports the mp4 into the project. The result does not depend on where the user left their viewport or mouse. It returns `asset.id` + `durationSeconds`.\n\nThen:\n\n```\nslates_generate_video {\n model: \"seedance-2.5\",\n videoReferenceAssetIds: [<asset.id>],\n videoReferenceSecondsEach: [<durationSeconds>],\n characterAssetIds: [...], environmentAssetIds: [...], styleAssetIds: [...],\n prompt: <written per slates-blocking-to-prompt>\n}\n```\n\n**A focused reference stack:** one identity sheet per character, any location or look reference the brief needs, the blocking clip, and a prompt written against it. Add a reference only for a distinct requirement; more competing references add variables rather than guaranteeing fidelity. An audio reference is valid when voice or sound continuity needs it and the selected endpoint supports it.\n\nChoose a model that accepts video references using `slates-model-selection`, then read its current reference caps. Video duration limits apply to the combined reference clips, not to each clip independently; quote each actual input duration.\n\n## Leaving holes on purpose\n\nWhere the model outperforms any blockout you could build — liquid, smoke, fire, cloth — **block a black gap instead** and say so in the prompt: `CUT 7 (14.5-17.0, black gap in the reference)`. You are reserving a slot, not forgetting one. Keep these exact times in the blocking record, then translate model-facing time cues through `slates-blocking-to-prompt`; not every endpoint accepts fractional timestamps.\n\n## What not to do\n\n- **Don't texture, light or material the blocking.** Grey is the specification. The reference supplies motion; the references supply look.\n- **Don't animate what you don't need.** Heads especially — a proxy head turning wrong is worse than one that never turns.\n- **Don't build the camera before the geometry.** It has nothing to aim at, and every value you set gets redone.\n- **Don't skip reading the scene back.** Write timings from `slates_blender_scene`'s `cutSeconds`, never from what you intended to build.\n- **Don't exceed the model's reference-video ceiling.** A 40s blocking against a 30s cap is rejected; trim it first.\n\n## Related\n\n`slates-camera-language` (rigs and moves) · `slates-blocking-to-prompt` (writing the prompt against the clip) · `slates-dialogue-blocking` (multi-character continuity) · `slates-restyle-from-blocking` (one blocking, many worlds) · `slates-model-selection` (routing)\n",
|
|
17
|
+
"slates-project-organization": "---\nname: slates-project-organization\ndescription: \"Navigate and organize Slates projects, asset codes such as IMG-A12 or VID-V3, folders, Library references and templates. Use when locating media, preparing a reusable production or keeping a project legible.\"\n---\n\n# Organizing a Slates project\n\nSlates already gives every REUSABLE reference a home — the **Library**, in categories the user names (Characters, Locations, Products, Looks…; `slates_list_library`), each item used in a prompt as `@name`, or `#name` for a look. Do NOT recreate those as folders, and never invent a category the user did not ask for. Folders are for **structure**, never type.\n\n**Folders = where an asset sits in the FILM.** They are in-app structure only. Use them for work product, not references.\n\nCreate with `slates_create_folder`; file assets with `slates_move_assets_to_folder`. Generations land in the project's active folder, so set it before a batch.\n\nA favorite is a keeper, not a folder: `slates_set_asset_favorite` marks one asset (the heart on its card, `isFavorite` in `slates_list_assets`) without moving it. To hand files out of Slates, `slates_export_assets` copies the ORIGINALS of the assets you name into a directory (named by code, never overwriting) — the way to deliver an image-only job that never needed a shot or a timeline. Use it to flag the takes worth a second look; file with folders once the pick is made.\n\nTo reuse a whole piece rather than one reference, use a TEMPLATE: `slates_export_template` saves a board, a scene or one Shot (recipes, script words, references and the Library items they mention; never takes) as a file, `slates_get_template` reads what a file holds and its swap slots, and `slates_import_template` adds it to a project, optionally swapping a slot for one of that project's own assets. An import generates nothing; quote and fire the returned Shots as usual.\n\nConventions by project type:\n- **Short film / narrative:** `Shots` (scene stills) · `Clips` (generated video) · `Final` (the export). Use one folder per scene (`Scene 1`, `Scene 2`, …) instead when the piece has distinct locations/beats.\n- **Ad / UGC:** `Hooks` · `B-roll` · `Talking-head` · `Final`.\n\nRules of thumb:\n- Reusable cast / sets / products / look → leave in the Library. Don't fold them.\n- Scene stills, clips, and the final cut → file into the structural folder they belong to, as you make them.\n- One folder per asset (folders are structure). Cross-cutting status (hero take, reject, variant) is a tag concern, not a folder.\n- Keep the gallery legible: work product lives in folders; the reference scaffolding (sheets, plates, style images) stays in the Library.\n\n## Asset codes — the shared vocabulary (IMG-A12 / VID-V3 / AUD-S1)\n\nEvery asset gets a short, stable code the moment it lands in a project, and the user sees it as the badge in the **top-left corner of every image and video card** in the gallery. This is the shared vocabulary between you and the user — it exists so neither of you ever has to quote a UUID.\n\n**The scheme:**\n- `IMG-A{n}` = images · `VID-V{n}` = videos · `AUD-S{n}` = audio.\n- Numbering is **per project, per type**, counts up from 1, and **numbers are never reused** — deleting IMG-A12 doesn't renumber anything, so a code always means the same asset forever.\n- Each asset also carries a **label**: the first ~4 meaningful words of its prompt, title-cased. Chat format is code + label: `IMG-A12 — Beach Sunset`.\n\n**How to use it:**\n- **User names a code** (\"use IMG-A36 as the reference\", \"animate VID-V3's last frame\") → resolve it via `slates_list_assets` (match the `code` field) to get the assetId, confirm back in the same vocabulary: \"Got it — IMG-A36 — Marcus Rooftop Close-Up as the first frame.\"\n- **You name assets** → ALWAYS code + label, never UUID, never \"the beach one\" (which of three?). The user matches your words to the badge by eye.\n- **User seems confused** about what a code is or how to point you at an image → explain it in one line: \"Every image and video in your gallery has a code badge in its top-left corner — like IMG-A36. Just say that code and I'll know exactly which one you mean.\"\n- **Ambiguity** (\"the sunset image\" when several exist) → pull candidates with `slates_get_assets_batch` and offer the codes: \"I see IMG-A12, IMG-A19, and IMG-A24 with sunsets — which one?\"\n",
|
|
18
|
+
"slates-prompting-elevenlabs": "---\nname: slates-prompting-elevenlabs\ndescription: \"Prompt ElevenLabs Sound Effects v2 (eleven-sfx) for a single effect or loop. Use with slates_generate_audio on this model; covers physical causes, duration, material, space and prompt influence.\"\n---\n\n# ElevenLabs Sound Effects v2 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card: ElevenLabs Sound Effects v2.** ONE short sound with an exact length, or a seamless loop. A Slates audio surface with a real duration control and a real loop mode.\n\n**The five levers**\n1. **Describe the physical CAUSE, not the label** — `heavy oak door slams shut`, `boot scuffs on grit`, `a latch drops home`.\n2. **Name the material and the space.** The material decides the timbre and the space decides the tail: `on wet concrete`, `in a tiled stairwell`, `across an empty warehouse`.\n3. **One sound per generation.** A room with dialogue AND clatter AND ambience is one Seed Audio pass, not three effects.\n4. **Pick the duration from the cut**, not from a feeling: roughly 0.5-1s for an `impact`, 2-4s for a `whoosh`, 8-22s for a `loopable bed`.\n5. **Ask for a loop explicitly** — `seamless loop` — when the sound has to lie under a whole scene, and keep it featureless enough to survive the seam.\n\n**Examples**\n- `A heavy oak door slams shut in a stone hallway, brief reverberant tail.` (1.5s)\n- `Steady rain on a tin awning, no thunder, no wind gusts, seamless loop.` (18s)\n\n**Hard constraint:** it is billed per second and the duration is never left for the model to pick — that would make the charge non-deterministic. It is NOT a speech surface: a line in a specific voice is `inworld-tts-2`, and dialogue inside a scene is Seed Audio, which casts and performs the line in the room.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — a label is not a sound; describe the physical cause:\n- `door sound`, `whoosh`, `footsteps`, `impact`, `ambience` standing alone\n<!-- @banned:end -->\n\nOne short sound with an exact length, carried on fal (`fal-ai/elevenlabs/sound-effects/v2`). This Slates audio surface has a real duration control and a real loop mode.\n\n## Where it routes\n\n- **A single hit that has to land on a known frame** — door slam, whoosh, impact, UI blip, riser.\n- **A seamless loop** you can lay under a whole scene — rain, engine hum, crowd murmur, machine noise.\n- **NOT** layered scenes. A room with dialogue *and* clatter *and* ambience is one `seed-audio` pass, not three SFX generations.\n- **NOT** speech. Dialogue, narration and scratch VO are `seed-audio` — it casts and performs the line inside the scene.\n- **AUDIO-ONLY.** It cannot produce images or video.\n\n## THE RULES\n\n### 1. Describe the physical CAUSE, not the label\n\n```\n✗ door sound\n✓ heavy oak door slams shut in a stone hallway\n\n✗ whoosh\n✓ a thick rope swung fast past a microphone, low air displacement\n\n✗ footsteps\n✓ boots on wet gravel, slow, one person\n```\n\nMaterial + weight + surface + room. Naming all four is the difference between a usable effect and a stock-library shrug. Cap is 450 characters — you will not need them.\n\n### 2. One sound per generation\n\nThis surface makes a single event. A door, then footsteps, then a siren is three generations layered on the timeline — or one `seed-audio` scene, which is usually cheaper and always more coherent.\n\n### 3. Duration is always explicit, and it is the price\n\nSlates **always sends** `durationSeconds`. (Left null the model picks, which makes the charge non-deterministic — so it is never left null.) The window it must fall in:\n\n<!-- @inject:thresholds -->\n<!-- GENERATED from @slatesvideo/shared — do not edit between the markers.\n Source: CONFIRM_CREDITS, DEVIATION_FACTOR and the audio bounds in\n packages/shared/src/operations/index.ts. Every number here is REFUSED by an\n op when a prompt gets it wrong, which is why none of them is typed by hand\n any more: this block replaced four claims that contradicted the code. -->\n\n**The thresholds, from the code that enforces them:**\n\n- **Confirm gate:** above **17 credits** an op returns `requires_confirm` and will not\n proceed until you re-call with `confirm: true`. This is a code gate, not permission to spend: every generation still needs the user-approved plan or quote.\n- **Deviation pause:** the desktop Studio Agent stops and re-asks when projected generation spend\n exceeds the approved plan by more than **20%**. You do not trigger this; the app does.\n- **Seed Audio duration:** **3–120 seconds.** There is no duration\n parameter on the model — the number you pass is written into the prompt AND is what the user is\n billed. Outside that range the op refuses rather than clamping.\n- **Sound Effects duration:** **1–22 seconds**, billed per second, never left for the\n model to pick.\n\nNever quote a credit figure from memory: `slates_estimate_generation_cost` returns the real one.\n<!-- @end:thresholds -->\n\n| Kind of sound | Ask for |\n|---|---|\n| impact, hit, click | 0.5–1s |\n| whoosh, riser, transition | 2–4s |\n| loopable bed | 8–22s + `loop: true` |\n\nOver-asking pads the tail with room tone you then trim. Under-asking clips the decay.\n\n### 4. Loops\n\n`loop: true` tiles without a seam — rain, engine hum, crowd murmur, machine noise. Combine with a longer duration so the loop point is not obvious.\n\nFor a bed longer than 22s, this is the wrong surface: `seed-audio` runs to 120s in one pass.\n\n### 5. Prompt influence\n\n`promptInfluence` 0–1, default 0.3. Higher hugs your wording with less variation between takes; lower explores. Raise it when a re-roll keeps wandering off the brief; lower it when every take sounds like the same take.\n\n## Iterating\n\n- Re-rolls that keep missing = the prompt named a **label** instead of a **cause**. Rewrite it as a physical event.\n- A hit that lands but sounds wrong in the scene is usually a *room* problem — name the space (\"in a stone hallway\", \"in a padded studio\", \"outdoors, no reflections\").\n- Three failed takes means the prompt is wrong, not the seed.\n\n## Content notes\n\nElevenLabs applies its own moderation. See slates-content-policy.\n",
|
|
19
|
+
"slates-prompting-flux-2-max": "---\nname: slates-prompting-flux-2-max\ndescription: \"Prompt or edit images with FLUX.2 Max (flux-2-max). Use with slates_generate_image or slates_edit_image on this model; covers word order, camera vocabulary, materials, colours and positive phrasing.\"\n---\n\n# FLUX.2 Max — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — FLUX.2 Max.** Word order is weight: it attends hardest to the start. Structure: `Subject + Action + Style + Context`, then secondary detail. Length 10-30 words for a concept test, 30-80 for most work, 80+ only for a genuinely complex scene.\n\n**The five levers**\n1. **Front-load the subject and the one action.** Anything after the first clause is a modifier, and it is read as one.\n2. **Name real gear** — `Shot on Hasselblad X2D, 80mm, f/2.8, natural light`, `Kodak Portra 400, natural grain`. This is the single biggest realism lever.\n3. **Era cues as a package** — `early digital camera, slight noise, flash photography, candid` reads 2000s; `film grain, warm cast, soft focus` reads 80s.\n4. **Bind every hex colour to an object.** `a #1B4D3E enamel mug` lands; an unbound colour does not.\n5. **For portraits add texture words** — `natural skin texture, realistic pores, subtle imperfections, soft diffused lighting`.\n\n<!-- @inject:cinematic-card -->\n**For a photographic look, use only what this frame needs.** Image models default to clean, evenly lit and fully exposed. Describe what the camera sees, not just gear or mood:\n- **Inspect every reference first.** Write its grade and imperfections in words: darkness, contrast, muddy or true blacks, colour, softness/noise, subject separation. Never grade cleaner or brighter than the look reference unless asked.\n- **One light system** — `low sun behind her`, `her face falls into deep shadow`, `no light in front of her`.\n- **Visible exposure** — `the sky burns out to white`, `dense, slightly crushed shadows`.\n- **Lens name plus effect** — `200mm telephoto`, `peaks loom huge behind her and melt into soft shapes`.\n- **Name every garment and close the foreground.** Omissions invite reference leakage or invented props.\nBind references inline. A scene reference owns the grade; for a look-only reference, write the new scene's light. References are optional. For owned-frame edits, describe only the change and what stays.\n<!-- slates-only -->Use `slates-cinematic-look` with a technique ID or section query for more.<!-- /slates-only -->\n<!-- @end:cinematic-card -->\n\n**Hard constraint:** no negative prompting. Every \"no X\" must be rewritten as the positive state — `no blur` becomes `sharp focus throughout`, `no people` becomes `empty scene`, `no harsh shadows` becomes `soft, diffused lighting`.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — there is no negative prompting, so each of these has a positive form:\n- `no blur` (say `sharp focus throughout`), `no people` (say `empty scene`), `no harsh shadows` (say `soft, diffused lighting`)\n- an unbound hex colour — bind it to an object or it lands inconsistently\n- `masterpiece`, `best quality`, `trending on artstation`, `8k`\n<!-- @banned:end -->\n\n**Examples**\n- `A chef plating in a steel kitchen pass. Shot on Hasselblad X2D, 80mm, f/2.8. Overhead fluorescents plus warm spill from the line. Natural skin texture, subtle imperfections. Muted steel and #7A3B2E copper.`\n- `An empty municipal pool at dusk, 35mm, deep focus, early digital camera with slight noise and flash falloff. Cracked #4A7C8C tiles. Candid, unstaged.`\n\nBlack Forest Labs' top image model, routed via fal.ai. In Slates: `slates_generate_image` with `model: flux-2-max` (REQUIRES projectId — no headless path), priced per resolution (1k/2k/4k — call `slates_estimate_generation_cost` for current numbers, never quote from memory). Strengths vs Nano Banana 2: photoreal texture, less censored, precise hex-color control, strong typography. Reference images route through FLUX's edit endpoint and carry a lower per-model cap than NB2's 14.\n\n## Core structure — front-load what matters\n\n```\nSubject + Action + Style + Context\n```\n\nWord order is weight. FLUX.2 attends hardest to the start of the prompt: main subject → key action → critical style → essential context → secondary details.\n\n**Length:** 10-30 words for concept tests, 30-80 words for most work, 80+ only for genuinely complex scenes.\n\n## Photorealism: name real gear, not \"professional photo\"\n\nThe single biggest realism lever is concrete camera vocabulary:\n\n```\nShot on Hasselblad X2D, 80mm lens, f/2.8, natural lighting\nShot on Sony A7IV, 35mm, golden hour, shallow depth of field\nKodak Portra 400, natural grain, organic colors\n```\n\nEra cues work the same way: \"early digital camera, slight noise, flash photography, candid\" reads 2000s digicam; \"film grain, warm color cast, soft focus\" reads 80s.\n\nFor portraits add: natural skin texture, realistic pores, subtle imperfections, soft diffused lighting.\n\n## No negative prompts — reframe positively\n\nFLUX.2 has no negative prompt support. Describe the presence you want, not the absence:\n\n- ❌ \"no blur\" → ✅ \"sharp focus throughout\"\n- ❌ \"no people\" → ✅ \"empty scene\"\n- ❌ \"no harsh shadows\" → ✅ \"soft, diffused lighting\"\n\n## Hex colors — bind them to objects\n\nFLUX.2 matches hex codes, but only when each code is attached to a specific object:\n\n```\nwalls in hex #C4725A, sofa in #1B6B6F, accent pillows #E8A847\ngradient starting with color #02eb3c and finishing with color #edfa3c\n```\n\n❌ \"use #FF0000 somewhere\" — unbound colors land inconsistently.\n\n## Text rendering\n\nQuote the exact text, then place and style it:\n\n```\nThe text 'OPEN' appears in red neon letters above the door\nLogo text 'ACME' in color #FF5733, ultra-bold decorative serif, centered\n```\n\nSpecify placement relative to other elements, font family feel (serif / sans / script), and relative size (\"large headline,\" \"small body copy\").\n\n## JSON prompting for production work\n\nFor multi-element scenes that must come out exactly right (product shots, infographics, brand work), FLUX.2 parses structured JSON prompts:\n\n```json\n{\n \"scene\": \"Professional studio product photography on polished concrete\",\n \"subjects\": [{ \"description\": \"matte black ceramic mug with steam\", \"position\": \"center foreground\" }],\n \"style\": \"commercial product photography\",\n \"color_palette\": [\"#1B1B1B\", \"#E8A847\"],\n \"lighting\": \"three-point softbox, soft diffused highlights\",\n \"camera\": { \"lens-mm\": 85, \"f-number\": \"f/5.6\" }\n}\n```\n\nUse natural language for exploration, JSON when the layout is locked and you're matching a spec.\n\n## Reference images (edit path)\n\nIn Slates, pass `referenceAssetIds` on `slates_generate_image` — FLUX routes them through its edit endpoint. Slates names each reference inline in the prompt (\"the subject (image 1), the style (image 2)\") in the order it sends them, so you don't hand-write role labels; the name carries the role and unnamed-by-position blending is avoided. For surgical changes to one existing image use `slates_edit_image` with `editModel: flux-2-max`. Extra `referenceAssetIds` are supported within the current edit-reference cap, with the source occupying one slot; read the tool schema for that cap. An older desktop without this capability refuses the request rather than silently omitting references.\n\n### Reference rules (the verified ones)\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n<!-- @inject:reference-rules-core -->\nIdentity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.\n<!-- @end:reference-rules-core -->\n\n### For FLUX.2 Max specifically\n\n- **FLUX caps references well below NB2's 14, so rule 1's \"2-4\" is a ceiling here, not a starting point.** Be deliberate about which roles earn a slot.\n- **FLUX has no memory between generations, so rule 7 is enforced by repetition.** Define the character exhaustively once and repeat those exact descriptors verbatim in every subsequent prompt — see Character consistency across a series below.\n\n## Character consistency across a series\n\nDefine the character exhaustively once, then repeat those exact descriptors verbatim in every subsequent prompt. FLUX has no memory between generations — the repeated description IS the consistency mechanism.\n\n## Common failure modes + fixes\n\n| Failure | Fix |\n|---|---|\n| Generic \"AI look\" on photoreal | Name a camera body + lens + f-stop instead of \"professional photo\" |\n| Colors drift from brand spec | Bind each hex code to a named object |\n| Text garbled | Quote the exact string, specify font feel + placement + size |\n| Multi-reference blend chaos | Name each reference inline (Slates does this from your @mentions/referenceAssetIds) — the same name for one entity, distinct names per role |\n| Wanted element missing | Move it earlier in the prompt — order is weight |\n\n## Pre-flight: references arrive inline, refer by code\n\nWhen you pass `referenceAssetIds`, the first call returns the references **inline as image content blocks** with a cost estimate and `requires_confirm: true`. Look at them — revise the prompt if they suggest a different composition or style — then re-call with `confirm=true`. Refer to each asset by its short code (`IMG-A12 — Beach Sunset`) when talking to the user; it matches the badge on their gallery thumbnail.\n\n## Sources\n\n- [Black Forest Labs — FLUX.2 Prompting Guide](https://docs.bfl.ml/guides/prompting_guide_flux2)\n- [fal.ai — FLUX.2 [max] Prompt Guide](https://fal.ai/learn/devs/flux-2-max-prompt-guide)\n",
|
|
20
|
+
"slates-prompting-gpt-image-2-5": "---\nname: slates-prompting-gpt-image-2-5\ndescription: \"Prompt or edit images with GPT Image 2.5 Flare or Sunburst. Use with slates_generate_image or slates_edit_image on these models; covers reference roles, lighting, text, panels, quality choices and edits.\"\n---\n\n# GPT Image 2.5 — sheets, grids, and text that actually reads\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — GPT Image 2.5.** The photoreal front-runner for people, and the readable-text, ordered-panel engine. Structure: subject and action with each reference named where it is used, then any exact copy in quotes, then layout, then light.\n\n**Pick the tier.** `flare` is Faster, quality comparable to GPT Image 2: drafts and volume. `sunburst` is Better quality, the most capable: finals, hero frames, photoreal people, multi-reference edits. Use the product default; choose Flare when speed is a stated priority.\n\n**The levers**\n1. **Name each reference inline** — `the woman from image 1`, `lit and graded like image 2`. Never an opening paragraph about what the references are.\n2. **Quote every string that must render verbatim** — `the jacket reads \"SLATES\"`. Describe a font's feel, never its name; keep on-image text under about 30 words.\n3. **Name the layout as a grid** for sheets and panels — `a 3x2 grid of panels, reading left to right, equal gutters`.\n4. **Set `quality` deliberately.** `high` is the everyday tier; `max` is 4× its price, `xhigh` about 1.8×. Coming from GPT Image 2 the names moved one rung: its `medium` is this `high`.\n\n<!-- @inject:cinematic-card -->\n**For a photographic look, use only what this frame needs.** Image models default to clean, evenly lit and fully exposed. Describe what the camera sees, not just gear or mood:\n- **Inspect every reference first.** Write its grade and imperfections in words: darkness, contrast, muddy or true blacks, colour, softness/noise, subject separation. Never grade cleaner or brighter than the look reference unless asked.\n- **One light system** — `low sun behind her`, `her face falls into deep shadow`, `no light in front of her`.\n- **Visible exposure** — `the sky burns out to white`, `dense, slightly crushed shadows`.\n- **Lens name plus effect** — `200mm telephoto`, `peaks loom huge behind her and melt into soft shapes`.\n- **Name every garment and close the foreground.** Omissions invite reference leakage or invented props.\nBind references inline. A scene reference owns the grade; for a look-only reference, write the new scene's light. References are optional. For owned-frame edits, describe only the change and what stays.\n<!-- slates-only -->Use `slates-cinematic-look` with a technique ID or section query for more.<!-- /slates-only -->\n<!-- @end:cinematic-card -->\n\n**Hard constraint:** its own content filter, distinct from Gemini's. Never describe a reference as a photograph of a real person.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use:**\n- a font NAME — describe the feel instead, as in: clean geometric sans, high contrast\n- a reference described as a photograph of a real person (`is a photograph of a woman`), or any up-front essay about what each reference is for — name the subject inline where it is used instead, as in: the woman from image 1\n- `8k`, `masterpiece`, `best quality`, `highly detailed` — quality incantations do nothing here either\n<!-- @banned:end -->\n\nGPT Image's edge is **character-level text accuracy** (~99% on English), ordered panels, and exact element placement — the jobs where every other model garbles a word or shuffles a layout. 2.5 inherits all of it and is better at each.\n\n## Which variant\n\n**Speed → Flare. Quality → Sunburst.** That is OpenAI's own routing rule, quoted from its image-prompting guide: *\"start with GPT Image 2.5 Flare when speed is the priority, or GPT Image 2.5 Sunburst when demanding quality requirements are the priority.\"* Same price either way, so the trade is purely latency against quality.\n\n🚨 **FLARE IS NOT AN UPGRADE OVER GPT IMAGE 2 — IT IS THE FAST ONE.** OpenAI, verbatim: *\"GPT Image 2.5 Flare is the small model, optimized for speed, with image quality **comparable to** GPT Image 2. GPT Image 2.5 Sunburst is the base model, optimized for quality, with **higher image quality than** GPT Image 2.\"* Their model pages agree: Flare is *\"our fastest model for high-quality, everyday image generation\"*, Sunburst *\"our most capable model for image generation and editing.\"* **Sunburst is the seat that beats what we had; Flare is the one that holds it at half the latency.** An earlier revision of this file called Flare \"better than GPT Image 2\" and sent Sunburst only to multi-reference edits — both wrong, corrected 2026-09-09 against the vendor docs.\n\n**Choose for the task.** Use the product default for ordinary work. Flare is an option when speed matters; changing model is not a mandatory draft stage.\n\n**Sunburst's widest lead is multi-reference editing** — several references all surviving into one frame, the character-consistency-across-shots problem. Reach for it there first, but that is not the only place it belongs.\n\n⚠️ **The LMArena receipt, scoped.** At launch Arena had Sunburst #1 and Flare #2 across text-to-image, single-image edit and multi-image edit, with margins over GPT Image 2 of **+81 / +47** on multi-image edit (Image Edit Arena: Sunburst 1520, Flare 1491, GPT Image 2 1461). Two caveats were missing and both matter: the baseline is **GPT Image 2 at `medium`, which is this model's `high`** — not its top tier — and the boards were **preliminary, a few thousand votes each**. Arena says Flare beats GPT Image 2; OpenAI says comparable. Route on OpenAI's wording and treat the board as a tiebreaker, not a spec.\n\n🚨 **The GPT Image line is ALSO the photoreal front-runner, and this file said the opposite until 2026-08-24.** **Receipts:** Eric's direct call, plus a head-to-head on the Higgsfield rail where GPT Image 2 at `quality: high`, 2K beat both Nano Banana rails on skin realism for photoreal people — that result is why the whole AI-influencer ad lane generates its plates here. **Route photoreal to this line, not away from it.**\n\n**Historical receipt, not a tier recommendation:** the photoreal comparison above used GPT Image 2 at its old `high` tier. It has not been repeated on 2.5 under matched conditions. Start with the product default and test a higher tier only against an unmet requirement; the old comparison does not establish a minimum tier for this model.\n\n**What the Banana line still owns:** edit-heavy work, and holding many subjects coherently in one frame. **Not the reference ceiling any more** — that line was true until 2026-09-09, when GPT Image went to its documented 16 against Banana's 14. Route on which model keeps them all recognisable, not on the count.\n\n**What would kill this:** a head-to-head at the intended crop going the other way. Per `slates-model-selection` § The meta-rule, re-run the evidence test when the roster changes — never carry a ranking forward on reputation. That rule is exactly what the 2026-08-24 correction failed, and exactly what the two ⚠️ notes above are honouring.\n\n## Quality tiers — always set explicitly\n\nAll five rungs are exposed, and they span ~36× end to end (2k class: $0.0044 → $0.158), which makes this the single biggest cost lever on the model. **The steps are UNEVEN — do not reason about them as a constant multiplier:** ~2.3× `low`→`medium`, ~3.9× `medium`→`high`, ~1.8× `high`→`xhigh`, ~2.25× `xhigh`→`max`. The same ratios hold at every OFFERED resolution class (2k/3k/4k); unoffered 1k differs slightly.\n\n| Tier | Use it for |\n|---|---|\n| `low` | Roughest pass — layout and composition checks, throwaway comps. |\n| `medium` | The draft tier. Cheaper than NB2 Lite and available up to 4K, which is why the draft lane moved here. |\n| `high` | General-purpose quality tier. Blind benchmarks on GPT Image 2 put this rung — which it called `medium` — within a hair of `max` (which it called `high`) at a quarter of the cost. Inherited from the old ladder, never re-run on 2.5, and it says nothing about `xhigh`. |\n| `xhigh` | One rung short of the top at about half its price (2k: 4 cr against `max`'s 8). Worth trying before `max`. |\n| `max` | Top of the ladder. Tiny type, dense diagrams, many labelled elements. |\n\n⚠️ **A tier label means different things on different models.** OpenAI: *\"The same quality label does not imply the same image quality or response time across models.\"* Flare at `max` and Sunburst at `max` are not the same picture, and neither matches Nano Banana's idea of \"high\".\n\n🚨 **The tier NAMES moved between versions and the strings did not.** GPT Image 2's `medium` is this model's `high`; its `high` is this model's `max` — same money, one rung of renaming. For a recipe explicitly written for GPT Image 2, map the old tier before reusing it on 2.5. A current user request for `medium` still means `medium`. Getting this backwards costs picture quality silently: nothing errors, the bill is correct for what was asked, and the image is just worse.\n\nNever rely on the provider default. fal's default is `high`, which is correct today — but it is the third rung of five rather than the top of two, so leaning on it means a fal-side change silently reprices you. Slates sends its configured quality explicitly; current defaults live in `slates-model-selection`. Your explicit choice overrides them.\n\n**Start at the default and change tiers for an unmet requirement.** OpenAI's own procedure: *\"If the output falls short, test a higher quality setting. Once it meets your requirements, test lower settings to see whether they preserve acceptable quality while reducing latency. Use `xhigh` or `max` only when they improve an unmet quality requirement within your latency budget.\"* A higher rung does **not** guarantee a better result on a given prompt. Compare `medium` against `high` when the job is small or dense text; that is where the rungs separate most visibly.\n\n## Resolution classes\n\n`1k` = 1024²-class · `2k` = 1920×1080-class · `3k` = 2560×1440-class · `4k` = 3840×2160-class. Pick 2k for most sheets/panels; 4k for print-density grids. 4K exists at every tier and is API-only — even paid ChatGPT can't render it.\n\n`1k` is not offered, and the reason is not its price: it is strictly dominated. At 1k you pay more for fewer pixels than at 2k, at **all five tiers**. Don't ask for it.\n\n⚠️ **Above 2560×1440 you are on a path OpenAI marks EXPERIMENTAL.** Verbatim: *\"Outputs with more than 3,686,400 total pixels ('2560x1440') are experimental.\"* That is the whole **4k** class (≈8.0 MP) plus 3k at 4:3/3:4 (≈3.70 MP). It bills normally and it works — but prove the shot at 2k or 3k 16:9 first, and do not be surprised by an odd frame at 4k.\n\n**Hard size bounds**, from fal's schema verbatim: each edge ≤ 3840 px, both edges multiples of 16, longer:shorter ratio ≤ 3:1, total pixels between 655,360 and 8,294,400. **The pixel ceiling is the one that actually bites** — the multiple-of-16 rule is documented but NOT enforced, and we have the receipt: 1920×1080 fails it (1080 = 67.5 × 16), is one of fal's own six priced canonical sizes, and metered clean. Slates picks sizes that respect the ceiling; these matter only if you hand-build a request.\n\n🚨 **THE ASPECT RATIO CHANGES THE PRICE ON THIS MODEL, and on no other image model.** OpenAI bills image OUTPUT TOKENS and the count tracks the frame's SHAPE, so at the same resolution class **`1:1` costs about 1.8× and `4:3`/`3:4` about 1.37× what `16:9` costs**; `9:16` costs the same as `16:9`. Metered 2026-09-09 and priced into the cost key, so the quote you get before generating is the real number — but if you are choosing between shapes and the budget is tight, **16:9 or 9:16 is the cheap one.** Every other image model charges the same whatever the shape.\n\n## Reference images — give every one a role, inline, where it is used\n\n**Assign a role to every reference image: subject, style, clothing, or background.** This is new emphasis in 2.5 and the highest-leverage change for the 16-reference character lane. An unroled pile of references makes the model guess what each one is for, and it guesses differently every run — which is the drift people mistake for a consistency failure.\n\n**The role rides a clause in the scene, not a paragraph in front of it.** *The woman from image 1 cooks on a rocky summit…*, *lit and graded like image 2*. Never open with sentences about what each reference is and what to take or ignore from it: that is the role essay the shared reference rules below forbid, and it drags the sheet's studio light into the scene.\n\n**Receipt, 2026-09-15, Sunburst, IMG-A192–A198.** The up-front version returned the studio look; the inline versions were never refused and never came back as a sheet. Two costs, both fixed in words: anything the prompt does not describe is taken from the reference (name every garment), and props nobody asked for appear (say what is in the foreground and that nothing else is). One sheet-only plate kept its described location, which narrows the two-reference rule in `slates-ugc-influencer-ad`. A look reference did far less than a described light. The full ladder is the vault's `cinematic-look-research.md`; the techniques are `slates-cinematic-look`.\n\nReference images route through the edit endpoint, **up to 16** — fal's documented `maxItems`, and the highest reference ceiling of any image seat in Slates (the Banana line takes 14). It was capped at 10 until 2026-09-09, which was never anybody's limit, just a number nobody had checked. The composed \"image N\" naming applies as everywhere else. Mask-based inpainting exists at the API level but is not surfaced: a mask is something the user has to paint, and there is no painting surface — describe the change instead.\n\n## Editing — separate the change from the constraints\n\n**State the change, then list what must survive.** \"Change only X,\" then name the invariants explicitly: identity, geometry, lighting, labels. For precise local edits also pin saturation, contrast, camera angle and surrounding objects — anything you do not pin is fair game for the model to move.\n\n**One change per iteration, and restate the constraints every turn.** Cross-turn drift is the named failure mode in OpenAI's own guidance: constraints do not persist across turns by themselves, so a multi-turn refinement that stops restating them will slowly rewrite the frame. This applies directly to multi-turn shot refinement.\n\n## Prompting for text accuracy\n\n- **Quote every string that must render verbatim**: `the sign reads \"OPEN 24 HOURS\"` — quoted strings render most reliably.\n- Say the text appears **once**, and give its position and typography.\n- Spell unusual words letter-by-letter.\n- Add `no extra text, no watermarks`.\n- Specify font *feel*, not font names: \"clean geometric sans, high contrast\", \"hand-painted brush lettering\".\n- For dense text (posters, UI mocks), list the copy as ordered lines: `Line 1: \"...\" Line 2: \"...\"` — it respects ordering.\n- **Don't bundle unrelated instructions into a text-rendering request.** A prompt that also redesigns the scene competes with the text for attention.\n- Keep total on-image text under ~30 words for perfect accuracy; beyond that, accuracy degrades gracefully but degrades.\n\n## Transparent backgrounds\n\nIf you need a cut-out rather than a scene, **ask for it explicitly and check the alpha**. OpenAI: request `background=transparent` and use PNG or WebP, then *\"check the decoded image's alpha channel, including hair, glass, shadows, and object edges\"* — a painted-white backdrop is the common failure and it is not transparency. Say what must NOT appear: *\"no solid backdrop, no checkerboard, no scenery, no watermark\"*, and do not let the product get restyled while the background is removed. **On every follow-up edit, repeat the transparency requirement** or it gets dropped. <!-- slates-only -->(Slates always requests PNG, so the format half is handled for you. **`background` IS surfaced now** — the Background control on the prompt bar, and `backgroundMode` on `slates_generate_image` / `slates_edit_image`. It is free: fal prices this family on size × quality alone.)<!-- /slates-only -->\n\n## When an edit must not touch a region at all\n\nPrompting alone cannot guarantee pixel-identical pixels. OpenAI's own instruction: if a region must stay exactly as it was, **composite the approved edit back into the original image** rather than asking the model to preserve it. Treat \"preserve\" language as a strong bias, never a lock.\n\n## Structure a complex prompt in labeled sections\n\nFor anything with several requirements, OpenAI recommends organising the prompt as **scene, subject, details, constraints** with labeled sections. Same content, easier to read and to change one part without disturbing the rest — which is what makes the one-change-per-iteration rule practical.\n\n**Say \"photorealistic\" or \"real photograph\" when that is the goal.** It is not inferred from a detailed description; ask for it directly, then describe framing and texture.\n\n## Concrete visuals beat mood words\n\nName materials, lighting, colour and medium. Mood words are cues only — \"cinematic\", \"moody\", \"epic\" tell the model almost nothing on their own. Give scale, atmosphere and colour instead. Camera specs (`85mm`, `f/1.4`) are appearance hints, not a physical simulation; they bias the look, they do not compute optics.\n\n**Name the lens and describe its effect, every time.** A lens named alone changed nothing visible (IMG-A195, 2026-09-15); named together with what it does to the picture, it produced real compression and depth of field (IMG-A198). Wording: `slates-cinematic-look` → `compression-as-outcome`, `defocus-as-outcome`.\n\n**For people, state body framing and scale**: \"full body visible, feet included\", \"hands naturally gripping the handlebars\". This is also the safest way to phrase a crop — see the blocked-phrasings section below.\n\n**No special syntax is required.** Prose, JSON and tagged blocks all work equally well, so pick whatever stays maintainable in the caller.\n\n## Panels, sheets, and grids\n\n- State the grid explicitly and number the cells: \"a 2×3 grid of panels, numbered 1–6, reading left-to-right, top-to-bottom\".\n- Give each cell ONE content clause: \"Panel 3: the character mid-jump, side view\".\n- Character identity sheets: GPT Image holds both the structured panel layout AND photoreal skin, which is why the influencer-ad lane builds its sheets here. Reach for NB2/NB Pro when it is an edit of an existing sheet, or when many subjects have to stay recognisable at once — not for the reference count, which GPT Image now leads at 16.\n\n## 🚨 WHAT GETS YOU BLOCKED — read before writing a prompt with a person in it\n\n**Receipt: 24 consecutive attempts on one character, 2026-08-24, same project and same rail.** Eleven were refused with `content_policy_violation` on the fal edit endpoint. The refusals were never about the scene — one of the blocked prompts was a woman standing at a kitchen counter with her hand on it. **Two phrasings were hard blocks, 5 for 5 each, and neither ever passed:**\n\n**1. Never describe the reference as a photograph of a real person.**\n\n> ❌ `Reference image 1 is a photograph of a woman. Use that exact woman.`\n> ✅ `Reference image 1 is a character identity sheet showing one woman across several panels — the face in the large portrait panel is the authority for her identity. Use that exact woman.`\n\nThe first reads to the filter as *recreate this real person's likeness*, which is a hard refusal regardless of what the rest of the prompt says. The second signals a fictional character and passes. **This is a wording change only — the reference image can be the same file either way.** One plate flipped from refused to accepted on this single sentence with nothing else altered.\n\n**Inline naming sidesteps the question and is now the default:** never describe the reference at all, and name her where she is used (*the woman from image 1*). Six of six Sunburst plates written that way passed on 2026-09-15. Keep the sheet sentence above as the fallback if a refusal appears.\n\n**2. Never attach a reference sheet containing a headless body panel.** A sheet whose full-body panels are cropped above the neck is refused every time, even with the correct opener. Regenerate the sheet with the head visible in every panel. Related, and already in this file's sheet guidance: phrase a cropped panel as *framing* (`cropped at the collarbone`), never as *absence* (`the head not shown`).\n\n⚠️ **These refusals were measured on GPT Image 2, not on 2.5.** The classifier belongs to OpenAI rather than to a model version, so the phrasing rules carry — but they are inherited, not re-measured. If Flare or Sunburst accepts one of the blocked phrasings, that is a new receipt to write down here, not a reason to delete this one.\n\n**On top of those, ordinary content triggers still apply** and they stack independently — a correct opener does not rescue them:\n\n| Refused | Why, and the fix |\n|---|---|\n| A woman sitting on a bed in a bedroom | Domestic + bed reads as intimate. Move her to a chair, a rug, another room. |\n| A knife, even lying flat on a chopping board next to a lemon | The object is the trigger, not the framing. Swap it — a cast-iron pan cleared instantly. |\n\n**🚨 Refusals are PROBABILISTIC. Retry once before rewriting a word.** In the same session an identical prompt, identical reference, identical params was refused and then accepted on a straight re-fire. A rejected job returns no file and costs nothing, so a retry is free and a rewrite is not — rewriting first is how you end up changing four variables and learning nothing. **Only redesign after two or three refusals.**\n\n**And change ONE thing at a time.** The eleven refusals above took far longer to diagnose than they should have because a reference swap and an opener rewrite shipped in the same call. Isolate on the prompt you actually want, so a pass leaves you with a usable asset instead of a data point.\n\n## Filter regime\n\nOpenAI moderate — a third regime distinct from Gemini (NB family) and ByteDance (Seedream). Real-face references pass more readily than Gemini; violence/brand rules are similar. `slates-content-policy` applies unchanged.\n",
|
|
21
|
+
"slates-prompting-inworld-tts": "---\nname: slates-prompting-inworld-tts\ndescription: \"Direct speech with Inworld Realtime TTS-2 (inworld-tts-2). Use with slates_generate_audio on this model; covers voice identity, reference acoustics, delivery tags, punctuation and voice consent.\"\n---\n\n# Inworld Realtime TTS-2 — the voice seat\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Keep it under 2,400 characters (the build fails above that) and keep\n the rationale and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Inworld TTS-2.** Speech in a SPECIFIC voice. The prompt is the words spoken, verbatim — not a description of them. Text length determines the bill.\n\n**IDENTITY, NOT ACOUSTICS — the rule that decides whether cloning works**\nA reference carries WHO is speaking: timbre, pitch, accent, age, vowel shape. It does NOT carry WHERE they are — room tone, distance, phone EQ, reverb and mic character are *acoustics*, and this model reproduces the identity while discarding the room. So:\n1. **A noisy reference does not give a noisy read — it gives a WORSE identity.** Music, a second speaker or heavy reverb corrupt what is being extracted. Use a clean single-speaker recording.\n2. **You cannot get \"on a payphone\" by cloning a payphone recording.** Acoustics come from the MIX, or from `seed-audio` which renders a room.\n\n**DIRECTION GOES IN SQUARE BRACKETS. PARENTHESES ARE SPOKEN ALOUD.** `[whispering] I hope nobody notices` is whispered; `(quietly) I hope nobody notices` says the word \"quietly\" out loud. Verified by ear — the easiest way to ruin a take.\n\n- **Plain English works inside them** — it is natural-language steering, not a fixed vocabulary: `[very quiet]`, `[whisper in a hushed style]`, `[very slow]`, `[say excitedly]`. Non-verbals are their own tags: `[laugh]`, `[sigh]`, `[breathe]`, `[clear throat]`.\n- **A tag it does not recognise is still consumed, and still changes the read.** Never spoken, never an error — so a mistyped tag fails SILENTLY and only listening catches it.\n- **Tags persist across sentences** until changed; `[reset]` returns to normal.\n- **Punctuation is the timing.** `Wait. Stop.` differs from `Wait, stop.`\n- **One line, one take.** Split a paragraph so a bad clause costs one re-roll.\n- **Spell numbers and titles aloud:** `twenty twenty-six`, `Doctor Reyes`.\n\n**Route elsewhere when:** the scene needs dialogue mixed with effects and room tone in one pass (`seed-audio`), or it is a single non-speech sound (`eleven-sfx`). This surface makes ONE voice saying ONE thing, cleanly.\n\n**Hard constraints:** no duration parameter — length falls out of the text. Exactly one voice source: a preset `voiceId` from `slates_list_voices`, a clip as `voiceReferenceAssetId` (a character's voice clip to speak AS the character, or any clean clip of one speaker), or `voiceDescription`.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — the prompt on this surface is SPOKEN ALOUD, so anything that describes the audio instead of being the audio gets read out as words:\n\n- `SFX`, `Ambient noise`, `Background music` as labels — this model speaks; it does not render a scene. Use `seed-audio` for those.\n- shot language: `wide shot`, `slow push in`, `warm tungsten` — video-prompt words, and here they would literally be said aloud\n- `voiceover`, `narrator says`, `he says` as stage directions wrapping the line — write only the words that should come out of the speaker\n<!-- @banned:end -->\n\n## Why the prompt is not a prompt\n\nOn every other surface in Slates the prompt DESCRIBES what you want and the model interprets it. Here the prompt IS the deliverable: each character is spoken aloud and each character is billed. `a gravelly man says he is tired` produces a voice saying the words \"a gravelly man says he is tired\".\n\nThat also means the two numbers a user cares about are the same number. The text length sets the price (in 250-character buckets) and sets the length of the audio. There is nothing to choose and nothing to reconcile.\n\n## Steering the delivery\n\n`VERIFIED BY EAR, 2026-09-05.` Every claim in this section was listened to, not\ninferred — an earlier draft of this skill documented tag forms that had only been\nprobed for an HTTP 200, which proves the request was accepted and nothing about\nwhether it was obeyed.\n\n**Square brackets are consumed. Parentheses are read aloud.** That is the whole\nrule, and getting it wrong is not a subtle degradation — the audience hears a\nnarrator say the word \"quietly\" in the middle of your line.\n\n| Written | What comes out |\n|---|---|\n| `[whispering] I really hope nobody notices that.` | whispered, tag not spoken ✅ |\n| `[very quiet] I really hope nobody notices that.` | very quiet, tag not spoken ✅ |\n| `[whisper in a hushed style] …` | hushed, tag not spoken ✅ |\n| `[very slow] …` | slowed right down, tag not spoken ✅ |\n| `[laugh] …` | an actual laugh, then the line ✅ |\n| `(quietly, under his breath) …` | 🚨 **the words \"quietly, under his breath\" are SPOKEN** |\n\n**Plain English works — it is natural-language steering, not a fixed vocabulary.**\nBoth the documented phrasings (`[whisper in a hushed style]`) and ordinary adverbs\n(`[whispering]`) were obeyed. Write the direction the way you would say it to an\nactor.\n\nThe eight dimensions the model steers on, with a working example of each:\n\n| Dimension | Example |\n|---|---|\n| Emotion | `[say excitedly]`, `[sound sad]`, `[sound terrified]` |\n| Articulation | `[say with force]`, `[articulate clearly]` |\n| Intonation | `[say with a rising pitch]` |\n| Volume | `[very quiet]`, `[very loud]` |\n| Pitch | `[say in a low tone]` |\n| Range | `[say playfully]`, `[say with no pitch variation]` |\n| Speed | `[very fast]`, `[very slow]` |\n| Vocal style | `[whisper in a hushed style]`, `[give a nasal quality]` |\n\nNon-verbals sit inline where they happen: `[laugh]`, `[sigh]`, `[cough]`,\n`[breathe]`, `[yawn]`, `[clear throat]`.\n\n### Four rules that are not obvious\n\n1. 🚨 **A tag it does not recognise is still consumed, and still changes the read.**\n `[zzzqqq]` is not spoken and does not error — it produces a different, arbitrary\n delivery. So a typo in a tag is SILENT: there is no rejection, no warning, and no\n way to catch it except listening to the take. Treat an unexpected performance as\n a possible misspelled tag before you blame the voice.\n2. **Tags persist across sentences.** A `[very slow]` at the top governs everything\n after it until something changes it. Use `[reset]` to go back to normal rather\n than assuming the next sentence starts clean.\n3. **Do not stack opposing directions.** `[whisper in a hushed style]` together with\n `[very loud]` produces unpredictable results — the model is resolving a\n contradiction, and which side wins is not something you can rely on.\n4. **Tags COUNT toward the billed characters**, even though they are never spoken.\n They are part of the text sent to the vendor, so the vendor charges for them and\n so do we — billing what was actually sent is the only honest basis. It rarely\n matters (a 13-character tag inside a 250-character bucket), but a line sitting\n just under a bucket boundary can be pushed into the next one by a long\n direction. Prefer `[very slow]` over `[say this one very slowly please]`.\n\n## Identity versus acoustics, at length\n\nThis is the distinction that decides whether the feature feels good, and it is worth being precise about because the failure is quiet — you get a usable clip that is subtly not the person.\n\n**What a reference clip transfers:** vocal timbre, pitch range, accent and regional vowels, apparent age, speech rate tendencies, and the particular rasp or breathiness of the source speaker.\n\n**What it does not transfer:** the room, the microphone, the codec, the distance from the mic, any processing on the source, and any other sound present in it.\n\nSo the ideal reference is boring: one person, close to a microphone, no music, no second speaker, no heavy reverb, five to fifteen seconds, speaking normally rather than performing. A phone voice memo in a quiet room beats a beautifully produced clip with a music bed underneath it.\n\n**Two failure modes, both common:**\n\n- *\"I cloned my podcast intro and it doesn't sound like me.\"* The intro had music under it. The model averaged the music into the identity. Re-clone from a clean stretch.\n- *\"I want the line to sound like it's coming through a car radio.\"* Clone the clean voice, then EQ and process the returned clip on the timeline. A radio-sounding reference makes a worse voice, not a radio effect.\n\n## Getting the voice onto the call\n\nExactly one source per call, and none of them requires a character to exist first:\n\n- **A preset:** `slates_list_voices` lists stock voices with gender, age, accent and tags — filter by any of them, or search the descriptions (\"gravelly\", \"narration\"). Pass the chosen `voiceId`. Presets clone nothing, so they are the fastest path and avoid the clone-creation rate ceiling.\n- **Speak AS a character:** `voiceReferenceAssetId: <its voiceAssetId>` (the clip on the row `slates_list_characters` returns). The seat clones the clip for that take and discards the vendor voice afterwards, so there is nothing to reconcile — but cloning shares a ceiling of two new voices a minute across every Slates user, so a run of lines in one cloned voice pauses between takes rather than failing. Send each line once; do not re-send one that already came back. Any other clean clip of one speaker works the same way.\n- **A voice with no recording:** `voiceDescription` (7–1000 characters of words). If it will be used again, keep the returned clip on a character with `slates_update_character` (`voiceAssetId`) so later lines clone the same clip instead of designing a new voice each time — a convenience, never a requirement.\n\n## Consent\n\nCloning a real person's voice needs that person's explicit, documented permission, scoped to what you are making. Clone from original human recordings only — never from another model's output. This is the same gate the real-face route applies to likeness, and it applies here for the same reason.\n\n## Worked examples\n\n**A line with a direction**\n\n```\n[very quiet] I heard what you said in there. I'm not going to pretend I didn't.\n```\n\n**A line that needs its numbers spoken**\n\n```\nThe vote was three hundred and twelve to eighty-nine. It carried at four minutes past midnight.\n```\n\n**A paragraph, split into three takes** — so one bad clause costs one re-roll:\n\n```\n1. You keep asking me why I stayed.\n2. It wasn't loyalty. It wasn't even fear, not by the end.\n3. [very slow] It was that I couldn't picture the version of me that left.\n```\n\n**What NOT to send**\n\n```\n(gravelly, tired) a tired old man narrates the opening of the film, wide shot, warm tungsten\n```\n\nEvery word of that is spoken aloud — **including the parenthetical**, which is the\ntrap: it looks like a stage direction and is treated as dialogue. Describe the voice when you are CHOOSING one (`voiceDescription`, or the desktop's voice picker); the prompt is only ever the words.\n",
|
|
22
|
+
"slates-prompting-kling-v3": "---\nname: slates-prompting-kling-v3\ndescription: \"Prompt Kling V3.0 video generation and Kling O3 video edits. Use with Kling models on slates_generate_video or slates_edit_video; covers subjects, dialogue, sound syntax, multi-shot direction and edit fidelity.\"\n---\n\n# Kling V3.0 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Kling V3.0.** Define the core subjects clearly at the START and keep those descriptions identical across shots. Strong image-to-video identity hold; use the current capability surface for duration and multi-shot limits, and the model catalogue for routing.\n\n**The five levers**\n1. **Dialogue in quotes** — `Character says, \"exact words here\"`. On Omni, direct the voice with `Gender + Age + Voice quality + Speech rate + Emotional tone + Language`: `[Character A: Detective, mid-40s, raspy, slow cadence, weary]: \"I've seen this before.\"`\n2. **Unique speaker labels, no pronouns after the introduction.** `he`, `the agent`, any synonym causes voice drift.\n3. **Sound has real syntax** — `SFX: heavy boots on wet pavement, distant siren wailing`, `Ambient noise: city traffic`, `Background music: low cello`. Always physical-cause specific; `SFX: footsteps` is not enough.\n4. **Motion adverbs modulate energy directly** — `slowly`, `rapidly`, `gently`, `explosively`. One primary camera move per shot, never stacked.\n5. **On image-to-video, do NOT re-describe the image.** It is an anchor; prompt how the scene EVOLVES from it — movement, camera, environmental change.\n\n**Examples**\n- `A detective in a wet grey overcoat stands under a stairwell light. He steps forward slowly as the light flickers. [Character A: Detective, mid-40s, raspy voice, slow cadence, weary]: \"I've seen this before.\" SFX: heavy boots on wet concrete, distant siren wailing. Ambient noise: rain on metal.`\n- `Camera tracks right alongside a cyclist crossing a bridge at dusk. She rises out of the saddle rapidly as the grade steepens. Ambient noise: wind, tyres on wet asphalt, distant traffic.`\n\n**Hard constraint:** `Immediately` (Omni only) removes the natural conversational beat between speakers — use it when timing matters and leave it out when it does not. Kling has a real `negativePrompt` field, unlike Seedance; start from the standard block and layer scene-specific suppressions.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use:**\n- `SFX: footsteps` and any label-only effect — physical-cause specificity or nothing\n- a pronoun or synonym for a speaker after the first introduction (`he`, `the agent`) — it causes voice drift; repeat the full label\n- `single continuous take` — Seedance's phrase, and it fights Kling's multi-shot\n<!-- @banned:end -->\n\nKuaishou's video model. Three tiers: `kling-v3.0-std` (general use, sound supported), `kling-v3.0-pro` (higher visual quality, sound supported), `kling-v3.0-omni` (multi-character dialogue + audio-visual co-generation).\n\nUp to 15s. Multi-shot supported (up to 6 cuts in 15s total). Strong on image-to-video — preserves identity, layout, and text from the input image well.\n\n## Subject definition rule (verbatim, fal blog)\n\n> \"Define your core subjects clearly at the beginning of the prompt and keep descriptions consistent across shots.\"\n\n## Dialogue syntax\n\n```\nCharacter says, \"exact words here\"\n```\n\nUse quotation marks for precise speech. Languages (Omni only): EN, ZH, JA, KO, ES.\n\n## Voice direction formula (Omni)\n\n```\nGender + Age Range + Voice Quality + Speech Rate + Emotional Tone + Language\n```\n\nExample:\n```\n[Character A: Detective, mid-40s, raspy voice, slow cadence, weary]: \"I've seen this before.\"\n```\n\nTone phrases that fire:\n- `speaking in a hushed, trembling whisper`\n- `shouting with commanding authority`\n- `clear, fearful voice`\n- `with a trembling voice, \"I'm scared\"`\n\n## The `Immediately` keyword (Omni only)\n\nWithout `Immediately`, Kling adds a natural conversational beat between speakers. With it, dialogue is back-to-back. Use when timing matters.\n\n```\n[Alice]: \"Get down!\" Immediately, [Bob]: \"Where?\"\n```\n\n## Speaker label discipline\n\nUnique labels per character. **No pronouns or synonyms after first introduction** — they cause voice drift.\n\n✅ `[Character A: Black-suited Agent]` ... `[Character A: Black-suited Agent]: \"Stop.\"`\n❌ `[Agent]... then he says...`\n\n## Multi-character dialogue (Omni)\n\n```\nAlice says in English, \"Hello!\" Then Bob replies in Spanish, \"¡Hola!\"\n```\n\n## Sound effects, ambient noise, music\n\n```\nSFX: thunder cracks, footsteps approaching\nAmbient noise: city traffic, birds chirping, ocean waves\nBackground music: tense orchestral strings, low cello\n```\n\nSFX accepts physical-cause specificity:\n- ✅ `SFX: heavy boots on wet pavement, distant siren wailing`\n- ❌ `SFX: footsteps`\n\n## Image-to-video guidance\n\n**Verbatim (fal blog):**\n> \"Treat the input image as an anchor. Kling 3.0 excels at preserving the identity, layout, and text details. Focus prompts on how the scene evolves *from* the image: subtle movements, camera motion, or environmental changes.\"\n\n**Don't re-describe what's already in the image.** Focus on motion, changes, evolution.\n\n## Multi-shot — what makes them hit\n\n**Hard cap: total duration ≤ 15s across all shots. Max 6 cuts.**\n\nHit conditions:\n- Shot labels are explicit: `Shot 1:`, `Shot 2:`\n- One primary action per shot\n- Subject described identically in each shot block\n- Camera move per shot is **one verb**, not a chain\n- Per-shot blocks: 30-60 words\n\nMiss conditions:\n- Compressing narrative into one paragraph\n- Pronoun-only references after the first shot\n- Mixing camera moves within a shot (\"pan then orbit then push in\")\n- Extreme wide → extreme close in adjacent shots without reference images\n\n## Element references\n\nStandard and Pro take element references with a first frame; Omni also takes references without one. 4K refuses reference images.\n\nUpload 2-4 multi-angle reference photos per character/object. Tag inline:\n\n```\n@element1 is the protagonist (refs: front, side, back angles).\n@element2 is the antagonist.\n```\n\n## Reference discipline (character / environment refs)\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n<!-- @inject:reference-rules-core -->\nIdentity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.\n<!-- @end:reference-rules-core -->\n\n### For Kling specifically\n\n- **Kling's consistency lever is \"lock the subject with a fixed label reused verbatim.\"** That is Kling's phrasing for rules 2 and 3, and it is stricter than the others: **pronoun and synonym drift breaks it**, so the exact same label must appear on every single mention — not \"he\", not \"the detective\" after you named him. Reusing the label verbatim is the whole game. Slates composes this for you from `@mentions`.\n- **Element references are the transport for rule 1** — 2-4 multi-angle photos per character/object, tagged `@element1` / `@element2` (see Element references above). The cap is 4 combined refs on the edit path.\n\n## Negative prompting — has a real field\n\nKling exposes `negative_prompt` on the fal endpoint (different from Seedance which has none). Default block to start from:\n\n```\nblurry, low quality, watermark, text overlay, distorted hands, extra fingers,\nduplicate limbs, unnatural skin texture, overly saturated colors,\nfloating objects, inconsistent shadows, jittery, flickering, morphing face\n```\n\nLayer scene-specific suppressions on top, and never suppress something the prompt asks for. This block carried `lens flare` until 2026-09-15, which silently cancelled every flare a prompt described (`slates-cinematic-look` → `source-flare`); add it back only for a shot that must have none.\n\n## Cinematic tactics\n\n- **Motion adverb precision** modulates motion energy directly: `slowly`, `rapidly`, `gently`, `explosively`\n- **Camera vocabulary that registers as instructions:** profile shot, tracking, following, freezing, panning, \"moving in sync with the subject\"\n- **One primary camera move per shot** — never stack\n\n## Tier choice\n\n- **Standard**: general use, sound supported\n- **Pro**: higher visual quality, sound supported\n- **Omni**: multi-character dialogue, audio-visual co-gen, language codes, references without a first frame\n\nEvery tier can generate dialogue and sound. Sound is on unless `sound: false` is passed; below 4K it bills the audio key, while 4K includes audio. Pick by visual quality and reference needs. Prices change; check current numbers before choosing a tier<!-- slates-only -->; call `slates_estimate_generation_cost` or `slates_list_available_models`<!-- /slates-only -->.\n\n## Benchmark prompt structure\n\n```\n[Character A: <role>, <voice quality>]: \"<line>.\" Immediately, [Character B: <role>, <voice quality>]: \"<reply>.\"\nAmbient noise: <soundscape>.\nCamera <single move>.\n```\n\nCinematic example (paraphrasing fal blog patterns):\n> \"Shot 1: Wide establishing shot of a neon-lit alleyway in heavy rain, steam rising from grates. Camera slowly tracks forward.\n> Shot 2: Medium shot of a detective in a trench coat ducking under an awning, water dripping from his hat brim. [Detective: weary, raspy]: 'I knew she'd come back.' Ambient noise: distant traffic, rain on metal.\n> Shot 3: Close-up on his eyes, narrowing as headlights flash across his face.\"\n\n<!-- slates-only -->\n## Pre-flight: references arrive inline, refer by code\n\nWhen you call `slates_generate_video` with `firstFrameAssetId` or `ingredientAssetIds`, the first call returns those references **inline as image content blocks** alongside cost + `requires_confirm: true`. Look at them, revise prompt if needed, then re-call with `confirm=true`. Kling Omni multi-character with several ingredient images especially benefits — confirm each character image lands cleanly before spending.\n\nWhen talking to the user about the gen, refer to each reference by its short code: `IMG-A12 — Detective Closeup`. The user sees that code as a gallery badge.\n\n- ✅ \"I'm anchoring on **IMG-A12** as the detective and **IMG-A18** as the alleyway environment — Omni will handle the line delivery in EN.\"\n- ❌ \"I'm using the detective image and the alley one...\" (which alley? Three exist.)\n<!-- /slates-only -->\n\n## Video-to-video EDIT<!-- slates-only --> (`slates_edit_video`)<!-- /slates-only --> — @Video1 / @ElementN / @ImageN\n\nKling O3 edit takes an EXISTING 3-15s clip and changes only what the prompt names — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved by default. Its notation is Kling's own, different from the \"image N\" naming used everywhere else:\n\n- **`@Video1`** — the source clip (always; the transport anchors the instruction to it).\n- **`@Element1..`** — subjects to swap IN. Each element = one frontal image + up to 3 angle images<!-- slates-only --> (pass as `characterAssetIds`; @mention names in the prompt compile to @ElementN automatically)<!-- /slates-only -->.\n- **`@Image1..`** — style/appearance references<!-- slates-only --> (pass as `styleAssetIds`)<!-- /slates-only -->.\n- Max **4 combined** element + image refs per edit.\n\n**Prompt shape — the change, not the whole scene:**\n\n```\nReplace the man in @Video1 with @Element1, keeping his walk cycle, the camera move, and the rain unchanged.\n```\n\n```\nEdit @Video1: turn the daytime street into a neon-lit Tokyo alley at night, wet asphalt reflections. Apply the visual style of @Image1. Keep the subject and camera motion exactly as they are.\n```\n\nRules:\n- Name what CHANGES; explicitly state what stays (\"keep the motion / camera / everything else unchanged\") — the model preserves better when told to.\n- One edit intent per pass. Chain passes for compound changes (each output is itself an editable clip, linked to its parent).\n- Billing is per second of OUTPUT ≈ the clip length, rounded UP to the next second. A 7.3s clip bills as 8s.\n- Clip constraints: 3-15s, 720-3840px, MP4/MOV. Agents can pre-trim on the timeline when a clip runs long.\n- Route by the required change: this edit seat supports element/style-reference control and original-audio retention. Read the current catalogue for defaults and competing seats<!-- slates-only --> — see `slates-model-selection`<!-- /slates-only -->.\n\n## Sources\n\n- [fal.ai — Kling 3.0 Prompting Guide](https://blog.fal.ai/kling-3-0-prompting-guide/)\n- [Vidguru — Kling 3.0 Omni Guide](https://www.vidguru.ai/blog/kling-3.0-omni-guide.html)\n- [AcceptPrompt — Kling 3 Prompt Guide](https://www.acceptprompt.com/blog/kling-3-prompt-guide)\n- [DataCamp — Kling 3.0 Tutorial](https://www.datacamp.com/tutorial/kling-3-0)\n",
|
|
23
|
+
"slates-prompting-lip-sync": "---\nname: slates-prompting-lip-sync\ndescription: \"Prepare Kling lip-sync or avatar generation with slates_generate_lip_sync. Covers source selection, voice, framing, audio constraints and the separate Seedance video-reference alternative.\"\n---\n\n# Lip-sync — setup guide\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card: Lip-sync (Kling only).** Two different flows with different inputs and different prices; output follows the source clip for video or the voice track for a still, billed per 5s block.\n\n**The five levers**\n1. **Pick `sourceType` deliberately** — `video` re-dubs an existing talking head (cheapest); `image` animates a still portrait (avatar-standard, then avatar-pro only on the final selected take).\n2. **The `prompt` on the avatar flows is SCENE CONTEXT, not motion direction.** Ambience, lighting, micro-expression: `Soft rim light`, `warm office`, `cool blue evening light through a window`, `gentle confident smile between sentences`, `focused intent expression`.\n3. **Clean the audio before uploading** — `noise-reduced`, `levelled`. Lip detection is sensitive, and a raw recording is the most common cause of a bad take.\n4. **Iterate on the SOURCE or the AUDIO, never on a refinement prompt** — there is not one. If the output is wrong, change the input.\n5. **Use avatar-standard for first-pass dialogue takes**, and switch to pro only once the line is locked. Facial fidelity is not visible until then.\n\n**Examples**\n- `Soft rim light, warm office, gentle confident smile between sentences.`\n- `Cool blue evening light through a window, focused intent expression.` (Or `.` — an empty prompt is fine when you have nothing to add.)\n\n**Hard constraint:** it is Kling-only; output follows the media and bills per 5s block. For a generated PERFORMANCE instead (head movement, gesture, delivery energy, with the dialogue as a native conditioning signal), that is a normal Seedance video generation with the clip attached as a video reference, not a mode of this tool. A real recording, or a cloned/cast voice rendered on `inworld-tts-2`, for production; this tool's built-in TTS is for scratch.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — the avatar prompt is scene context and motion verbs are ignored:\n- `turns her head`, `raises an eyebrow`, `hand gestures`, `nods`, `walks`\n- `reader_en_m-v1` — listed in fal's docs, returns \"Voice id not found\" in production\n<!-- @banned:end -->\n\n**This tool is Kling-only.** It wraps Kling's dedicated lip-sync and avatar endpoints; output follows the source clip for video or the voice track for a still, billed per 5s block.\n\n| Flow | Source | Model | Cost | Use case |\n|------|--------|-------|-----------|----------|\n| Re-dub | video clip | kling-lip-sync-video | ~4 credits / 5s block | Replace dialogue on an existing talking head |\n| Avatar standard | still image | ai-avatar/v2/standard | ~14 credits / 5s block with uploaded audio; typed text adds one flat voice block | Animate a portrait into a talking avatar |\n| Avatar pro | still image | ai-avatar/v2/pro | ~29 credits / 5s block with uploaded audio; typed text adds one flat voice block | Higher facial fidelity for hero shots |\n\nPick `sourceType` deliberately — it decides the pricing tier and the underlying endpoint.\n\n## Want Seedance instead? That is a video generation, not a mode here\n\nSeedance can generate the performance rather than bolting a mouth onto finished pixels — head movement, gesture, delivery energy, with the dialogue as a native conditioning signal, and a video source keeps its own voice. **It is not an engine switch on this tool.** Run a normal `slates_generate_video` on `seedance-2` with the clip (or portrait) attached as a video/ingredient reference and the dialogue written into the prompt yourself.\n\nThat is the same endpoint the old `engine=seedance-2` branch called — it just built the sentence for you, invisibly, and it presupposed a \"video 1\" that might not exist. Writing the prompt is the whole difference, and it is the part you want control of.\n\n- Driving clips must be 2–15s; output duration is whatever you set (4–15s).\n- Video references bill COMBINED input+output seconds (`seedance-2*-vref-*` keys); pass the clip duration and quote before confirming. On both Seedance 2.0 and 2.5's AI-face route (EvoLink) the input side counts as at least the output's length: max(input, output) + output.\n- Faces go through the normal cascade: `seedanceFace` for a character, `[REAL_FACE_DETECTED]` → `seedanceRealFace` + `realFaceConsent` for a real person.\n\nEverything below is about the Kling tool.\n\n## Choosing video vs avatar\n\nUse **video** (re-dub) when:\n- A talking-head clip already exists (Slates-generated, recorded, or imported)\n- The mouth/face is already moving and only the audio needs to change\n- ~4 credits is hard to beat for short dialogue replacement\n\nUse **avatar** when:\n- Only a still portrait exists\n- The character needs to come alive from a single image\n- Identity + face fidelity matter (avatar-pro for hero shots, standard for everything else)\n\n## Source asset constraints\n\n### Video flow (`sourceType: 'video'`)\n- Format: mp4 or mov\n- Duration: 2–10s (output follows the source clip)\n- Resolution: 720p or 1080p (480p will be rejected)\n- Max file size: 100MB\n- Face must be visible and roughly facing camera. Profile shots fail.\n- Existing audio is replaced.\n\n### Avatar flow (`sourceType: 'image'`)\n- Min 512×512, PNG/JPG/WebP\n- **Face occupies 60–70% of frame.** This is the single biggest avatar quality lever.\n- Eyes open, mouth neutral, looking near-camera. Side profile = bad output.\n- Single subject, clean background. Group photos confuse the face anchor.\n\n## Audio source\n\nTwo ways to drive the lips:\n\n### TTS (`audioMethod: 'tts'`)\n- Pass `ttsText` (the words spoken)\n- Optional: `ttsVoice` (default `oversea_male1`), `ttsLanguage` (default EN), `ttsSpeed` (default 1.0)\n- **Hard cap: 120 characters of text.** Longer = silently truncated.\n- Languages: EN, ZH, JA, KO, ES\n\n### Upload (`audioMethod: 'upload'`)\n- Pass `audioFilePath` — absolute path to an audio file on the user's machine\n- Format: mp3, wav, m4a, ogg, aac\n- Max 5MB\n- Duration: 2–60s (avatar output follows the voice track)\n- Single clean voice. Music underneath, multiple speakers, or noisy mics produce garbage lips.\n\nPrefer upload for production-quality voice. TTS for fast iteration / placeholder dialogue.\n\n## Voice catalog (TTS)\n\nReliable English voices (verified working on the fal endpoint as of 2026):\n\n| Voice ID | Description |\n|----------|-------------|\n| `oversea_male1` | Male, English — default, stable |\n| `commercial_lady_en_f-v1` | Female commercial English |\n| `uk_boy1` | Young man, UK accent |\n| `uk_man2` | Man, UK accent |\n| `uk_oldman3` | Older man, UK accent |\n| `calm_story1` | Storyteller / narrator |\n\nAvoid `reader_en_m-v1` — listed in fal.ai docs but returns \"Voice id not found\" in production.\n\nFull 48-voice list (ZH, JA, KO included): https://fal.ai/models/fal-ai/kling-video/lipsync/text-to-video/api\n\n## Speech-rate notes\n\n`ttsSpeed` range 0.5–2.0:\n- 0.8–1.0: natural conversational\n- 1.1–1.3: punchy ad delivery\n- 1.4+: rushed, clips consonants\n- 0.6–0.7: slow, weighty (good for dramatic lines)\n\nDefault 1.0 unless the line specifically calls for slower or faster cadence.\n\n## Avatar prompt usage\n\nThe `prompt` parameter on avatar-v2 (standard + pro) is **scene context**, not motion direction. The mouth animation comes from the audio — the prompt sets ambiance, lighting, micro-expression.\n\nGood:\n- `Soft rim light, warm office, gentle confident smile between sentences.`\n- `Cool blue evening light through a window, focused intent expression.`\n\nBad (the model ignores motion verbs):\n- ❌ `She turns her head, raises an eyebrow, then speaks.`\n- ❌ `Hand gestures while talking.`\n\nDefault `\".\"` is fine if you have nothing useful to add.\n\n## Tier selection — avatar standard vs pro\n\n**Use standard** when:\n- Drafts, A/B testing voices, internal review reels\n- Wide / medium shots where face isn't the focal point\n- Cost matters more than micro-expression fidelity\n\n**Use pro** when:\n- Final ads where the avatar's face fills the screen\n- The character is named / branded — identity drift kills the take\n- You're already paying tens of credits for the surrounding video pipeline\n\nDon't default to pro. The ~15-credit delta per take adds up across iteration.\n\n## Common failure modes\n\n| Symptom | Likely cause | Fix |\n|---------|--------------|-----|\n| Lip movement looks \"rubber\" / disconnected | Source face <60% of frame | Re-crop the still tighter |\n| Voice doesn't match character age/gender | Default voice id used | Pick from voice catalog |\n| Output truncated mid-word | TTS text >120 chars | Shorten or chain two takes |\n| Garbled mouth on uploaded audio | Background music / multi-voice | Use clean dialogue-only audio |\n| \"Voice id not found\" 422 | Hit `reader_en_m-v1` | Switch to `oversea_male1` |\n| Avatar eyes drift / cross | Source had closed/angled eyes | Pick a frame with neutral open eyes |\n| Generation completes but lips don't move | Profile shot / face >70° off-axis | Use a near-frontal portrait |\n\n## Cost discipline\n\n- Video re-dub at ~4 credits per 5s block is the cheapest dialogue iteration in the entire Slates stack; use it for voice A/B testing\n- Avatar standard at ~14 credits per 5s block is fine for medium use; typed text on a still adds one flat voice block\n- Avatar pro at ~29 credits per 5s block trips the confirm gate; explicit user OK required every time\n- Output follows the media; billing rounds up to whole 5s blocks.\n\n## Workflow patterns\n\n**Voice A/B test (cheap):**\n1. Generate one base talking-head video clip with Seedance (~40 credits)\n2. Run `slates_generate_lip_sync` with `sourceType: 'video'` against 3–5 different `ttsVoice` values\n3. Total cost: ~40 + (5 × ~4) ≈ 60 credits to compare voices\n\n**Brand avatar from a single portrait:**\n1. Generate or upload the hero portrait (face fills frame, eyes open, neutral mouth)\n2. Avatar standard for first-pass dialogue takes\n3. Avatar pro only on the final selected take\n\n**Avoid:**\n- Avatar pro on first iteration (waste — facial fidelity isn't visible until you've locked the line)\n- TTS for final ads (production should use real voice or cloned voice — the upload flow)\n- Uploading raw recordings — clean noise + level the file first, lip detection is sensitive\n\n## Confirm gate: cost + codes, no inline preview\n\nLip-sync is mechanical — the model re-syncs the chosen source to the chosen audio. The confirm response carries the source asset's code so you can announce it in chat.\n\n- ✅ \"Lip-syncing **IMG-A12 — Founder Portrait** to the new line. ~29 credits on avatar-pro. Confirm?\"\n- ❌ \"Using the founder image...\" (which? Three exist.)\n\nDon't second-guess the source. If the output is wrong, iterate on source choice or audio, not on a refinement prompt (there isn't one).\n\n## Sources\n\n- [fal.ai — Kling LipSync API](https://fal.ai/models/fal-ai/kling-video/lipsync/text-to-video/api)\n- [fal.ai — AI Avatar v2 Standard](https://fal.ai/models/fal-ai/kling-video/ai-avatar/v2/standard/api)\n- [fal.ai — AI Avatar v2 Pro](https://fal.ai/models/fal-ai/kling-video/ai-avatar/v2/pro/api)\n",
|
|
24
|
+
"slates-prompting-ltx-2-5": "---\nname: slates-prompting-ltx-2-5\ndescription: \"Prompt LTX-2.5 or LTX-2.5 Pro (ltx-2-5, ltx-2-5-pro) with slates_generate_video. Covers sound-first direction, connected shots, frame inputs, variant differences and duration constraints.\"\n---\n\n# LTX-2.5 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — LTX-2.5.** It scores the picture on the same pass that draws it, so SOUND IS THE FIRST THING YOU WRITE. Lightricks' own priority order: sound, camera, character detail, shot type and scene, then scene dressing. One flowing paragraph, not labelled sections. When a prompt sprawls, cut from the bottom.\n\n**The five levers**\n1. **Lead with sound, and anchor every sound to something in frame.** The test is \"visible, or at least locatable\" — `the rope creaks against the cleat`, `the hull knocking hollow against the fenders`, `rain on the awning`. A distant whistle is fine IF you have named the post it comes from.\n2. **Camera second**, because framing decides the visual weight of the shot — `low camera at the gunwale`, `slow drift right`, `static medium behind the counter`.\n3. **Character detail as physical ACTION**, not as adjectives about a person — `she braces a boot on the rail and hauls`, `his hands counting notes`.\n4. **It is the native MULTISHOT seat** — one generation carries two to four connected shots holding character, light and voice across the cuts. Write the cuts.\n5. **Quote dialogue and name the language and accent** — `in English with a slight German accent` — `\"We should not have come back,\" in English with a slight German accent.`\n\n**Examples**\n- `The rope creaks against the cleat as she leans back, gulls calling somewhere off the port bow, the hull knocking hollow against the fenders. Low camera at the gunwale, slow drift right. She braces a boot on the rail and hauls, twice, then stops.`\n- `A till drawer bangs shut, a fan ticks against its cage, rain on the awning outside. Static medium behind the counter, then cut to a close-up of his hands counting notes, then cut wide as he looks up at the door.`\n\n**Hard constraint:** durations are EVEN numbers from six — there is no 5s or 7s clip. There is NO reference endpoint at all, so identity references are unavailable; use MiniMax H3 or Kling when a character must hold across shots. Any sound not anchored to something in frame gets invented for you. And never write mood adjectives as sound: \"tense atmosphere\", \"a sense of dread\" and \"ominous ambience\" produce nothing usable — the fix is one more moving object in frame with a sound attached to it.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — mood adjectives standing in for sound produce nothing usable:\n- `tense atmosphere`, `a sense of dread`, `ominous ambience`, `eerie silence`\n- an unanchored sound: name the thing in frame it comes from, or cut it\n<!-- @banned:end -->\n\nLTX-2.5 generates picture and sound **in a single pass**, with a Gemma-4 12B text encoder reading\none flowing paragraph. That single fact drives everything below: the prompt is not a shot\ndescription with audio bolted on, it is **a scene where the sound is load-bearing** — and\nLightricks' own priority order puts sound first, ahead of the camera.\n\nTwo seats, and the naming is a trap:\n\n| | `ltx-2-5` (base) | `ltx-2-5-pro` |\n|---|---|---|\n| Build | Distilled, 8-step | Full diffusion (\"Diffusion Fidelity Rendering\") |\n| Resolutions | 720p / 1080p / **1440p** / 4K | 720p / 1080p |\n| Durations | 6–20s, even steps | 6 / 8 / 10s |\n| Price | $0.09–$0.30 per second | $0.12–$0.17 per second |\n| Reach for it when | iterating, long takes, 4K delivery, batch volume | one dense final render inside 1080p and 10s |\n\n**Pro is not \"base plus more.\"** It buys picture quality on a *narrower* envelope — it cannot make\na 1440p frame and it cannot make a 12-second clip. Reaching for it out of habit costs a third more\n*and* takes away the reach.\n\n---\n\n## 1. The six parts, in priority order, in one paragraph\n\nLightricks ranks the elements of an LTX prompt like this. When a prompt sprawls, **cut from the\nbottom.**\n\n1. **Sound** — highest priority; the model scores the picture as it draws it.\n2. **Camera** — framing decides visual weight and the feel of the shot.\n3. **Character detail** — expressed as physical action.\n4. **Shot type and scene** — the action itself.\n5. **Scene dressing** — the first thing to trim.\n\nWrite it as **one flowing paragraph**, not a list of labelled sections. LTX is not Seedance (eight\nengineering slots) and not H3 (three separate audio layers) — it wants continuous prose.\n\n---\n\n## 2. Sound: anchor it or it gets invented\n\n**Write the audio line last, then go back and check every cue has a source you could point at.**\nAnything unanchored, the model invents for you.\n\nThe test is **\"visible, or at least locatable.\"** A distant whistle is fine *if* you have named the\nmarshal's post it comes from. A \"distant whistle\" with nothing to attach to is a coin flip.\n\n> the rope creaks against the cleat as she leans back, gulls calling somewhere off the port bow,\n> the hull knocking hollow against the fenders\n\n**Never write mood adjectives as sound.** \"Tense atmosphere\", \"a sense of dread\" and \"ominous\nambience\" produce nothing usable. If a scene feels thin, the fix is **one more moving object in\nframe with a sound attached to it** — never another adjective.\n\n### Dialogue\n\nQuote it, and name the language and accent:\n\n> \"We should not have come back,\" in English with a slight German accent.\n\nTwo rules that decide whether the lip sync lands:\n\n- **Give the character a beat of stillness before they speak.** The sync needs something to lock\n against; a character already mid-motion when the line starts drifts.\n- **Describe the beat structure** — when they look, how long they wait, when they speak, where they\n look afterwards.\n\nSlates pins the frame rate at 25fps, which is also what Lightricks recommends for dialogue: at 50fps\nthe performance \"pulls toward a video look.\"\n\n---\n\n## 3. Character emotion is physical\n\nThe model renders actions. It does not render adjectives.\n\n| Instead of | Write |\n|---|---|\n| she looks anxious | her jaw sets, she turns the ring on her finger twice |\n| he seems exhausted | he blinks slowly and lets his shoulder take the doorframe |\n| a tense standoff | neither moves; his thumb finds the strap and stays there |\n\n---\n\n## 4. Multishot\n\n**One LTX generation can carry several connected shots**, holding character, environment, lighting,\nvoice and style across every cut. It is one of the seats that carry several shots in one generation.\n\n**Working range is two to four shots.** Three is the comfortable stopping point.\n\nAt **every** transition you must supply four things:\n\n1. **Name the edit in the prose** — \"hard cut\", \"dissolve\", \"match cut\".\n2. **Re-establish the shot completely** — scale, angle, lens and light all reset at a cut. A cut is\n not a continuation.\n3. **Re-identify recurring characters by their original descriptor.** \"The woman in the bronze\n gown\", never \"she\". Pronouns lose the character across a cut — this is the single most common\n multishot failure.\n4. **State what the sound does at the cut.** Silence is not assumed; if the room tone should drop\n out, say so.\n\nA shape that works:\n\n> Wide establishing shot of the workshop, dust in the window light, a lathe turning somewhere off\n> frame — hard cut — macro close-up of the brass fitting as it seats, the turning noise gone,\n> replaced by a single dry click — match cut — medium shot of the woman in the bronze gown stepping\n> back, the room tone returning underneath her.\n\n---\n\n## 5. Camera: write it, don't enumerate it\n\nfal exposes a `camera_motion` enum (dolly in/out/left/right, jib up/down, static, focus shift).\n**Slates does not surface it, deliberately** — and prose is the better instrument anyway:\n\n- **A written move can be tied to a specific moment.** \"A slow push-in that settles as she reaches\n the door, then holds\" is not expressible as an enum value.\n- **For multishot it would be actively wrong** — one enum value would impose a single camera\n behaviour on three shots that each want their own.\n\nSo name the lens, the framing, the move, and **the moment the move resolves**.\n\n---\n\n## 6. The hard constraints\n\n### Durations are even numbers only, starting at six\n\n**6, 8, 10, 12, 14, 16, 18, 20.** There is no 5-second LTX clip and no odd duration of any length.\nAsking for 7s is not a rounding matter — that generation does not exist.\n\n**And the long end is 1080p-and-below only.** At 1440p and 4K the ceiling drops to **6, 8 or 10**.\n\nfal's own default is `auto`, which lets the model pick the length from the described action.\n**Slates always sends an explicit length instead**, so what you choose is what you are billed for.\nChoose the length the beat needs.\n\n### Aspect ratios: 16:9 and 9:16, and nothing else\n\nThe narrowest set in the catalogue. Square, 4:5 and 21:9 are not available on this\nmodel at any resolution.\n\n### Frames, not references\n\nLTX takes a **start frame** and an **optional end frame** (which generates a transition between the\ntwo). It has **no reference-to-video endpoint at all** — no identity references, no style\nreferences, no environment references, no reference video, no reference audio.\n\n**For character consistency across separate shots, use MiniMax H3 or Kling.** Within a single LTX\ngeneration, use multishot instead — that is precisely the gap it fills.\n\nIn image-to-video, **do not cut away from the opening frame too early.** You have paid for that\nframe; let it play before the first move.\n\n### Do not ask for text on screen\n\nNeither the spelling nor its stability from frame to frame can be relied on. Signage, labels,\ncaptions and lower-thirds belong in post.\n\n---\n\n## 7. Audio is free here, and that changes the routing\n\nNative synchronised audio is **included at every resolution on both seats**, with no surcharge and\nno toggle that costs money — unlike Kling, where sound is a paid dimension. A 6-second 1080p LTX\nclip **with sound** is 39 credits.\n\nCombined with 1080p at $0.13/s, the cheapest native 1080p second with sound included in Slates, this makes LTX **the\ncoverage seat**: the one to reach for when the job is many takes rather than one hero shot, when a\nsequence needs its own sound, or when the credit budget is the binding constraint.\n\nRoute away from it when you need identity references (H3, Kling), a ratio other than 16:9 or 9:16\n(Seedance, Kling), or authored multi-layer audio direction (H3).\n",
|
|
25
|
+
"slates-prompting-minimax-h3": "---\nname: slates-prompting-minimax-h3\ndescription: \"Prompt MiniMax H3, H3 Max or H3 Max Turbo with slates_generate_video. Covers separately authored dialogue, scene sound and score, declared reference relationships, frame inputs and variant-specific constraints.\"\n---\n\n# MiniMax H3 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card: MiniMax H3.** The only seat with three separately authored audio layers: dialogue, scene sound and score are three separate sections of the prompt, generated in one pass, and putting a sound in the wrong section drops or doubles it.\n\n**The five levers**\n1. **Write the three audio layers separately**: `overall_soundscape:` for what is in the room, `non_diegetic_music:` for what only the audience hears, and the dialogue quoted inline. Section decides attribution.\n2. **Quote dialogue and name the language** — `says in English`, `speaks in Spanish`. Eleven languages are stably supported; the language is part of the instruction, not an afterthought.\n3. **Declare the reference RELATIONSHIP**, which no other seat has: `kept whole`, `partly kept`, `transferred`, or `a loose echo`. An undeclared reference is a guess.\n4. **Give a beat of stillness before a line** — `sits still for a beat, then looks up`. The sync needs something to lock against; a character already mid-motion when the line starts drifts.\n5. **Describe the beat structure** — `waits`, `then speaks`, `under the last three seconds`. H3 is a timeline, so write one.\n\n**Examples**\n- `A woman sits still at a kitchen table for a beat, then looks up. She says in English, \"You said Tuesday.\" overall_soundscape: a fridge hum, a spoon set down on formica. non_diegetic_music: N/A.`\n- `Two mechanics either side of an open bonnet. The younger one wipes his hands, waits, then speaks in Spanish, \"No es el alternador.\" overall_soundscape: a socket wrench, a radio two bays over. non_diegetic_music: a low sustained cello under the last three seconds, audience only.`\n\n**Hard constraint:** the three seats differ in what the ENDPOINT accepts, not in grammar. Base H3 reaches 2K/4K and takes references; `minimax-h3-max` tops out at 1080p, takes the same 9+3+3 references, and costs MORE at the tier they share — a speed pick, never the cheap one; `minimax-h3-max-turbo` has Max's ladder at half its rate and takes frames only, NO references. Every tier above 768p is built from the native 768p render: judge at native. Reference inputs affect the quote; include every attached modality when estimating.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use:**\n- a sound written into the wrong audio section — it is dropped, doubled, or attributed to the wrong layer\n- `background music` as a bare instruction: the score is its own authored layer, audience-only, and it is named as such\n- an undeclared reference relationship — say kept whole, partly kept, transferred, or a loose echo\n<!-- @banned:end -->\n\nH3 is an **omni transformer**: it generates picture and sound in the same pass, at 24fps with\n32kHz stereo, 5–15 seconds, in 11 stably-supported languages (Arabic, Chinese, English, French,\nGerman, Italian, Japanese, Korean, Portuguese, Russian, Spanish). That single fact drives\neverything below — the prompt is not a shot description with sound bolted on, it is a **timeline\nwith three audio layers you author separately**.\n\n**Three seats, one grammar.** Everything in this file applies to all three. They differ only in\nwhat the endpoint accepts:\n\n| | `minimax-h3` | `minimax-h3-max` | `minimax-h3-max-turbo` |\n|---|---|---|---|\n| Resolution | 480p / 768p / **2K / 4K** | 480p / 768p / 1080p | 480p / 768p / 1080p |\n| References | 9 images + 3 video + 3 audio (12 files) | 9 images + 3 video + 3 audio (12 files) | **none** (no reference endpoint) |\n| Frames | start and/or end | start and/or end | start and/or end |\n| Price at 768p | **$0.060/s** | $0.080/s | $0.040/s |\n| Why pick it | resolution, references, and the cheaper second | **speed** — a 5s 768p clip in **4.8s** vs **57s** (measured) | **price** — half Max's rate at every tier |\n\n**Max is the premium seat, not the budget one.** It is 33% dearer at 768p, equal at 480p, and it\ntops out lower. Route there when a fast turnaround on a text-to-video or start-frame shot is worth\npaying for; route to base H3 for anything needing resolution, references, or the same tier cheaper.\n\n**Turbo is the budget seat.** Same grammar and Max's ladder at half Max's rate, with no reference\nendpoint: attach a reference and Slates refuses the call rather than dropping it. Route there for\ndrafts, volume and start-frame coverage, then re-run the keeper on Max or base H3 when it needs\nreferences.\n\n**1080p on Max and Turbo is a refinement, not a native render.** fal's schema, verbatim: *\"1080P\nlatent refinement from a native 768P source.\"* It is a different stage from base H3's 2K/4K\nupscaler, and it costs double the 768p second. Judge a 1080p take against the same shot at 768p\nbefore paying for it across a batch.\n\n**The speed is measured, not claimed** (2026-08-27, same prompt and params on both rows): a 5-second\n768p text-to-video finished in **4.8 seconds** on Max against **57 seconds** on base H3 — roughly\n**12x**, queue to finished file. fal advertises \"under 3 seconds\"; the literal claim did not hold at\n4.8s wall-clock, but the order of magnitude did. For iteration loops and client-present work that gap\nis the entire reason the seat exists.\n\n🚨 **Max's known weakness: colour banding in low light (Eric, 2026-09-09).** Certain shots —\nespecially dark or low-key ones — come back with low-bitrate-looking banding across gradients (skies,\nwalls, shadow falloff). It is the one place the seat visibly gives something up. If a shot is dark\nand gradient-heavy, either light it up in the prompt or route to base H3 at 768p; do not fix it by\nreaching for 2K, which adds its own artifacting on top.\n\n---\n\n## The one thing that makes H3 different: audio is a THREE-LAYER instruction\n\nKling and Seedance have their own sound syntax. H3 splits audio into three separately authored layers, enforced by where\nyou write each thing. Get the section wrong and the sound is dropped, doubled, or attributed to the\nwrong source.\n\n| Layer | What belongs in it | Where it goes |\n|---|---|---|\n| **Synchronised events** | dialogue, singing, and any sound tied to a specific shot or action | the **body** of the prompt, on the beat it lands |\n| **Scene sound** | ambience and physical sounds that run across the whole clip, room tone, rain, traffic, a ventilation hum | the **overall_soundscape** section |\n| **Score** | music the characters cannot hear; audience-only | the **non_diegetic_music** section |\n\n**Three rules, all from MiniMax's own guide:**\n\n1. **Dialogue and singing NEVER go in the soundscape section.** They are synchronised events; they\n belong in the body, at the moment they happen.\n2. **Diegetic music — music the characters can hear** (a radio in the scene, a busker) — also\n belongs in the **body**, not in the score section. The score section is audience-only.\n3. **Write the score in instrumental terms, not mood words.** Name the instruments, the tempo, and\n how it develops. *\"A restrained solo-piano score at a slow tempo, sustained low cello underneath,\n no swell\"* — not *\"emotional music\"*.\n\nUse **N/A** for a section only when silence or absence is genuinely what the shot wants. An empty\nscore section is a real choice; a vague one is a wasted layer.\n\n### The shape, in the one prompt field\n\nSlates sends one prompt string, so write the three layers as labelled paragraphs in this order:\n\n```\n[Shot 1] Live-action, cinematic. A medium-wide shot frames a baker opening the shutters of a\nsmall street bakery before sunrise. The camera pushes in with small amplitude at slow speed as\nthe middle-aged baker with a calm, slightly raspy voice places a fresh loaf on the counter and\nsays: \"First batch of the morning.\" [Shot 2] At 00:05.000, the camera cuts to a close-up of\nsteam rising from the sliced bread while his final words carry over from the previous shot.\n\noverall_soundscape: wooden shutters scrape open over a quiet street, trays clink softly inside, a\ndoorbell rings once, then light footsteps and the crisp sound of bread being sliced.\n\nnon_diegetic_music: a soft acoustic-guitar pattern at a moderate tempo, joined by sparse upright-bass notes,\ngentle fade at the end.\n```\n\n**Body target: 350–500 words** for a reference-carrying shot. Dialogue-heavy content prioritises\nfitting the complete spoken timeline over hitting a word count.\n\n🚨 **Slates disables the provider's prompt expander.** H3's API can rewrite your prompt before\ngeneration; Slates turns that off, because a model rewriting the user's words invisibly is banned\noutright (prompt transparency: what the composer shows is what the model gets). The practical\nconsequence is on you: **nothing will pad a thin prompt.** Write the whole body.\n\n---\n\n## Shots and timing\n\nThe first shot carries **no timestamp**. Every later shot opens with the bracket and a cut time\nthat increases and stays inside the clip length:\n\n```\n[Shot 2] At 00:03.500, the camera cuts to ...\n```\n\nTransition verbs the model knows: **cuts to · transitions to · changes to · switches to**.\n\n**Dialogue that continues across a cut** needs the continuity said out loud — *\"his final words\ncarry over from the previous shot\"* — or the line restarts. **Speech that ends abruptly** should be\ndescribed as cut off rather than trailed off.\n\n---\n\n## Camera — write the move into the sentence\n\nThe model has a named motion vocabulary:\n\n> Zoom In / Zoom Out · Push In / Pull Out · Pan Left / Pan Right · Truck Left / Truck Right ·\n> Tilt Up / Tilt Down · Pedestal Up / Pedestal Down · Arc Shot · Tracking Shot · Static Shot ·\n> Shake Slightly / Shake Strongly · POV · Roll Clockwise / Roll Counterclockwise\n\nModify with **amplitude** (`with small amplitude` / `with large amplitude`) and **speed**\n(`at slow speed` / `at fast speed`).\n\n🚨 **Integrate the motion into the sentence — never stack labels.** MiniMax's own example:\n*\"The camera pushes in with small amplitude at slow speed toward the folded letter in her hands.\"*\nNot *\"Push In. Small amplitude. Slow.\"*\n\n---\n\n## Speakers and dialogue\n\nGive each speaking character a stable identity in the prose and keep it: describe the voice once\n(*\"a young woman with a quiet, breathy voice\"*), then refer back to the same description at every\nline. Identification, delivery and action sit **outside** the quoted line; the line itself is only\nthe words.\n\n```\nThe young woman with a quiet, breathy voice says: \"I get off at the next station.\"\n```\n\n**Voiceover** needs two things — the phrase *\"says in an off-screen voiceover\"* **and** an explicit\nstatement that the lips stay closed. Without the second half the model animates a mouth.\n\n```\nThe man says in an off-screen voiceover: \"I still remember that road.\" — his lips remain\ncompletely closed.\n```\n\n**On-screen text** — signs, banners, labels, subtitles, neon — goes in double quotes with the\noriginal wording preserved exactly: *A red neon sign reading \"Open Late\" glows above the doorway.*\n\n---\n\n## References — H3's real differentiator is the declared RELATIONSHIP\n\n*(`minimax-h3` and `minimax-h3-max`. Max gained the reference set on 2026-09-09; its free allowance\nis FOUR images rather than the base row's five. `minimax-h3-max-turbo` takes no references.)*\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n<!-- @inject:reference-rules-core -->\nIdentity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.\n<!-- @end:reference-rules-core -->\n\n### Cite references by number — Slates already does it for you\n\nH3 on fal takes references as **typed slots** and expects the prompt to name them by modality and\norder: **`image 1`, `image 2`, `video 1`, `audio 1`**. That is exactly what the Slates composer\nemits from your `@mentions` and `#tags` (`Marcus (image 1) in the workshop (image 2)`), in the\nexact order it sends them.\n\n🚨 **Do NOT hand-write angle-bracket reference tags.** MiniMax's own model-card grammar uses\n`<Subject N>` / `<Picture N>` / `<Video N>` / `<Audio N>` labels; the fal endpoints Slates calls do\nnot — they build the binding from the typed slots and ask for plain numbered prose. Typing the tags\nyourself puts literal angle brackets in the prompt the model reads.\n\n### State how much of each reference survives\n\nThis is the lever no other model in the catalogue gives you. Say, in plain words, what each\nreference is FOR and how much of it should carry through:\n\n| Intent | Say something like |\n|---|---|\n| Keep it whole | *\"Keep the woman in image 1 exactly as she appears — hair, cardigan, necklace.\"* |\n| Keep part of it | *\"Use the café in image 2 for the brick wall and the sofa; the lighting is late evening, not daylight.\"* |\n| **Move a trait onto someone else** | *\"Give the man in image 3 the weathered leather texture of the jacket in image 4.\"* |\n| Loose echo | *\"Match the general palette and grain of image 5; nothing else from it.\"* |\n\nThe third row is the one with no equivalent anywhere else in Slates: **transferring a characteristic\nonto a different subject** is a first-class thing H3 understands. Reach for H3 when that is the job.\n\n**Audio references** bind a voice or a texture without copying the words. Say which speaker an\naudio reference is for (*\"the woman in image 1 speaks in the voice timbre of audio 1\"*), and when\nyou are referencing only the timbre, **do not carry the reference clip's original dialogue into your\nprompt** — write the new line. When you genuinely want the same words re-performed, quote them\nexactly and say so.\n\n**An audio reference cannot travel alone** — H3 refuses a reference set that is audio only. Pair it\nwith at least one image or video reference.\n\n### 💸 Reference images past the free allowance are billed — and the two reference rows differ\n\nOn `minimax-h3` the first **5** are free and each additional image adds **4 credits**.\nMax pools image pixels, reference-video seconds and reference-audio seconds into one token\nallowance. Include `referenceImages`, `videoRefSeconds` and `audioRefSeconds` when estimating;\ncharacter voices count as audio. The generation preflight resolves the actual attached media.\n\nFour extra images on a 10s\n768p clip add 16 credits to a 30-credit generation: **more than half again**, for references that\noften make the output worse rather than better (see the 2–4 rule above).\n\nAttach the references the shot needs, not the ceiling. Call\n`slates_estimate_generation_cost` with `referenceImages` set to the real count before a\nreference-heavy job — a quote that omits it under-reports the bill.\n\n---\n\n## Frames\n\nAll three rows take a **start frame**, an **end frame**, or both. With an\nend frame, land it explicitly: describe the final pose, spacing and composition as the thing the\nshot **settles into** at the end, rather than hoping the model finds it.\n\n> *\"…she rotates the handle into the final angle and settles into the pose, spacing and composition\n> of image 2 at the end of the shot.\"*\n\n**Frames and references are mutually exclusive** on the two reference rows — they are different endpoints, and\nthe reference endpoint has no frame slots at all. Slates refuses the combination rather than\ndropping one side.\n\n---\n\n## Cost discipline\n\n| Combination | Credits |\n|---|---:|\n| `minimax-h3` · 768p · 5s | 15 |\n| `minimax-h3` · 768p · 10s | 30 |\n| `minimax-h3` · 2K · 10s | 65 |\n| `minimax-h3` · 4K · 10s | 80 |\n| `minimax-h3-max` · 768p · 10s | 40 |\n| `minimax-h3-max` · 1080p · 10s | 80 |\n| `minimax-h3-max-turbo` · 768p · 10s | 20 |\n| `minimax-h3-max-turbo` · 1080p · 10s | 40 |\n| `minimax-h3` — every reference image past the **fifth** | **+4** |\n\n**768p is the default for a reason.** It is the tier the model natively generates.\n\n🚨 **2K and 4K are UPSCALES of a 768p render, not larger generations.** fal's own schema says so:\n*\"480P and 768P are native generation modes; 2K and 4K upscale a 768P base result.\"* The upscaler\n(H3-Regenerate-2K) is a separate stage bolted onto a finished take — it can enlarge detail but it\ncannot add information.\n\n**In our own test (2026-08-27, same prompt, same seed) the 2K pass came back with MORE artifacting\nthan the 768p original it was built from**, while costing 33 credits for a 5-second take against 15,\nand taking nearly twice as long to return.\n\n🚨 **Confirmed independently (Eric, 2026-09-09): 2K and 4K carry visible AI noise artifacting and\n\"just look bad\".** That is now TWO separate observations, months apart, pointing the same way — it is\nno longer a single-shot warning. The tiers stay available because a delivery spec sometimes demands\nthe pixels, but **do not route to 2K/4K for quality**: you are paying more, waiting longer, and\nadding artifacts to a 768p render. Upscale in post from a clean 768p master instead.\n\n**So: generate at 768p and judge it at 768p.** Reach for 2K or 4K only when a delivery spec demands\nthe pixels, and expect to be paying for size rather than quality — a post-production upscale from a\nclean 768p master is very often the better result. **4K video is Pro-only** (the server returns\n`PRO_REQUIRED` for a base account); 2K is open to every tier.\n\n---\n\n## Quick checklist\n\n- Body written as a timeline, first shot untimestamped, later shots on `[Shot N] At MM:SS.mmm`.\n- Camera motion written **into** a sentence with amplitude and speed.\n- Dialogue and diegetic music in the body; ambience in the soundscape section; audience-only score\n in the score section, described by instrument and tempo.\n- Voiceover carries both the off-screen phrase and the closed-lips statement.\n- References cited as `image 1` / `video 1` / `audio 1`, each with a stated job and a stated degree\n of retention. No angle-bracket tags.\n- Reference count is deliberate — you are paying 4 credits for each one past the fifth.\n- Frames **or** references, never both.\n- The prompt is the prompt: no expander will fill it out for you.\n",
|
|
26
|
+
"slates-prompting-motion-transfer": "---\nname: slates-prompting-motion-transfer\ndescription: \"Prepare Kling Motion Control with slates_generate_motion_transfer: a character image plus a driving clip. Covers orientation, source selection, tiers and the separate Seedance video-reference alternative.\"\n---\n\n# Motion transfer — setup guide\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card: Motion transfer (Kling Motion Control only).** A target IMAGE (your character) plus a source VIDEO (the motion) produces your character performing that motion. Output follows the driving clip, up to 30s with video orientation or 10s with image orientation, billed per 5s block.\n\n**The five levers**\n1. **The target image must show body proportions clearly** and the character must occupy more than about 5% of the frame. A tiny figure in a wide shot has nothing to drive.\n2. **Single character in the target.** A group image breaks the identity anchor.\n3. **Choose `characterOrientation` on purpose** — `video` takes the source clip's framing, `image` preserves the portrait's. It is the most-missed choice here.\n\n4. **The prompt is atmosphere only** — `Soft afternoon sunlight, dust motes in the air, vintage warm color grade.` Motion verbs are ignored; the motion is already in the driving video.\n5. **Pick the source section up front**, and write only atmosphere: `soft afternoon sunlight`, `vintage warm color grade`, `clean studio backdrop`. The output follows that section, up to the orientation's limit; longer clips cost more blocks.\n\n**Examples**\n- `Soft afternoon sunlight, dust motes in the air, vintage warm color grade.`\n- `Clean studio backdrop, sharp focus on the character.` (Or leave it empty.)\n\n**Hard constraint:** cartoon driving videos fail, and a cropped or partial target character drifts. std is fine while the motion-and-framing combination is still moving; switch to pro once it is locked. For a REGENERATED shot instead — physical contact, cloth and hair, camera motion, native audio — that is a normal Seedance video generation with the driving clip as a video reference, not a mode of this tool.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — the motion is already in the driving video, so motion verbs are ignored:\n- `spins faster`, `jumps higher`, `add more energy`, `moves quicker`\n<!-- @banned:end -->\n\nTake a still **target image** (your character) and a **source video** (the motion you want), produce a new video of your character performing the source video's motion. **This tool is Kling-only** — it wraps Kling Motion Control and nothing else.\n\n| Tier | Cost | Use case |\n|------|-----------|----------|\n| Kling std (`kling-mc-std-5s`) | ~32 credits / 5s | General motion transfer, budget lane |\n| Kling pro (`kling-mc-pro-5s`) | ~42 credits / 5s | Cleaner anatomy, better identity preservation |\n\nBoth tiers trip the confirm gate. User OK required every time. (Prices are approximate — `slates_estimate_generation_cost` returns the exact credit total.)\n\n## Want Seedance instead? That is a video generation, not a mode here\n\nKling MC retargets a skeleton onto a finished image; Seedance *generates* the shot with the motion as a conditioning input — the difference shows on fast choreography, physical contact, cloth/hair, and camera motion, and the output carries native audio. **It is not an engine switch on this tool.** Run a normal `slates_generate_video` on `seedance-2` with the driving clip attached as a video reference and the character image as an ingredient, then write the prompt yourself:\n\n```\nThe character from image 1 performs the exact motion, choreography, and camera\nmovement from video 1. Preserve the character's identity, appearance, and outfit.\n```\n\nThat is the same endpoint the old `motionModel=seedance-2` branch called — it just wrote that sentence for you, invisibly. Add style/setting/camera direction freely; Seedance re-generates the whole shot.\n\n- **Reference videos must total 2–15s on 2.0, or 2–30s on 2.5.** Longer clips: trim first. Kling MC (`characterOrientation: 'video'`) takes up to 30s.\n- **Billing = combined input+output seconds** (the vref keys). The server probes the clip and corrects an understated key; quote via the confirm gate before spending. On both Seedance 2.0 and 2.5's AI-face route (EvoLink) the input side counts as at least the output's length: max(input, output) + output.\n- **Faces route through the face cascade**: `seedanceFace` for a character, `[REAL_FACE_DETECTED]` → confirm consent → `seedanceRealFace=true, realFaceConsent=true` (premium realface vref pricing).\n- `characterOrientation` has no Seedance equivalent; framing follows the prompt + `aspectRatio`.\n\nEverything below is about the Kling tool.\n\n## Inputs\n\n- `sourceVideoAssetId` — driving video. **Must be a realistic human** with clear proportions. Anime/cartoon/CG driving videos fail.\n- `targetImageAssetId` — character to be animated. Can be any style (cartoon, anime, realistic, painted).\n- Both must already exist as assets in the project. Use `slates_list_assets` to find them or upload first.\n\n## Source video constraints\n\n- Realistic human (not animated, not CG)\n- Entire body OR upper body visible — head must not be obstructed\n- Subject occupies a clear share of the frame\n- Single primary subject. Multi-person driving videos confuse the motion anchor.\n- Clean motion — choppy / cut-edited driving videos produce jittery output\n\nGood driving video sources:\n- Reference dance footage with one subject\n- Walking / gesture / posing clips\n- Talking-head footage when paired with character_orientation: 'video'\n\nBad driving video sources:\n- Music videos with multi-shot edits\n- Anime / animation clips\n- Heavily stylized footage with smoke / particles obscuring the body\n- Footage where the subject's head leaves frame mid-clip\n\n## Target image constraints\n\n- Character body proportions clearly visible\n- Character occupies >5% of image area (not a tiny figure in a wide shot)\n- Single character. Group images break the identity anchor.\n- Any artistic style works — cartoon, anime, painted, realistic, 3D render\n\nAvoid:\n- Extreme close-up of just the face (no body to drive)\n- Character partially cropped at the waist when the driving video is full-body\n- Multiple characters\n\n## character_orientation — the most-missed choice\n\nThis single parameter changes the output dramatically. Pick deliberately.\n\n| Value | Output framing | Max source duration | Best for |\n|-------|----------------|---------------------|----------|\n| `video` | Matches driving video framing | Up to 30s source | Complex full-body motion (dance, action, athletics) |\n| `image` | Matches target image framing | Up to 10s source | Camera moves, simpler motion, preserving original composition |\n\n**Default `video`** when the driving video has the look you want (most cases).\n\nSwitch to `image` when the target image's composition is the brand asset and the motion is secondary (e.g., a hero shot of a character that needs subtle gesture, not a full performance).\n\n## Tier choice — std vs pro\n\n**std (~32 credits)** for:\n- Drafts, motion exploration, blocking\n- Group scenes where the character isn't a hero shot\n- When the budget is tight and the motion is the focus\n\n**pro (~42 credits)** for:\n- Final hero takes\n- Branded characters where identity drift = unacceptable\n- Anatomically complex motion (limbs crossing, fast direction changes)\n- Anime / cartoon target images — pro handles non-realistic styles better\n\nDon't default to pro. The ~10-credit delta compounds fast across iteration.\n\n## Prompt usage (optional)\n\nThe `prompt` field is **scene/style refinement**, not motion direction. The motion comes from the driving video — the prompt sets ambiance, lighting, additional detail.\n\nGood:\n- `Soft afternoon sunlight, dust motes in the air, vintage warm color grade.`\n- `Clean studio backdrop, sharp focus on the character.`\n\nBad (model ignores motion verbs — they're already in the driving video):\n- ❌ `She spins faster and jumps higher.`\n- ❌ `Add more energy to the dance.`\n\nLeave it empty if you don't have a specific atmospheric note.\n\n## Common failure modes\n\n| Symptom | Likely cause | Fix |\n|---------|--------------|-----|\n| Limbs distort / extra fingers | std tier, complex motion | Switch to pro |\n| Character identity drifts | Target image cropped too tight | Use a fuller-body target |\n| Output looks \"stuck\" / minimal motion | Driving video subject too small in frame | Pick a driving video where the subject fills more of the frame |\n| Cartoon target turns realistic | std tier on stylized art | Switch to pro — handles non-realistic styles better |\n| Garbled output entirely | Anime / CG driving video | Use realistic human driving footage |\n| Wrong framing on output | character_orientation set wrong | Try the other value |\n| Background bleeds through character | Target image had complex background | Use a target with cleaner background separation |\n\n## Workflow patterns\n\n**Reference dance to brand character:**\n1. Generate or upload the brand character as a still image (clean background, full body, single subject)\n2. Find driving footage — a clean reference video of the dance you want\n3. Upload both as project assets\n4. Run motion transfer with `motionModel: 'kling-mc-pro'`, `characterOrientation: 'video'`\n5. Total cost: ~42 credits per 5s take\n\n**Subtle motion on a hero portrait:**\n1. Use the locked hero portrait as the target image\n2. Pick a driving video with subtle gesture (head turn, slight posture shift)\n3. `characterOrientation: 'image'` to preserve the portrait's framing\n4. std tier is fine for this case — motion isn't dramatic\n\n**Avoid:**\n- Pro tier on first iteration — waste, switch to it once the motion + framing combo is locked\n- Cartoon driving videos — guaranteed failure\n- Cropped or partial target characters — identity will drift\n- Driving videos longer than the needed motion; pick the source section upfront, within the orientation's limit\n\n## Cost discipline\n\n- Output follows the driving clip, up to 30s with video orientation or 10s with image orientation; billed per 5s block\n- Both tiers trip the confirm gate — every call needs explicit user OK\n- Iteration is expensive: 4 five-second takes at pro ≈ 168 credits. Lock framing + driving video before tier-up to pro.\n- Always run a single std take first to validate the motion + framing combo before committing to pro\n\n## Confirm gate: cost + codes, no inline preview\n\nMotion transfer is mechanical — the model deterministically applies source motion to target image. Both tiers trip the confirm gate; the response includes the asset codes for source and target so you can announce them in chat.\n\n- ✅ \"Transferring motion from **VID-V3** onto **IMG-A12 — Detective Closeup**. ~42 credits, confirm?\"\n- ❌ \"Using the walk video and the detective image...\" (multiple of each in the project.)\n\nDon't second-guess the assets the user picked — the model executes the transfer. If the output is wrong, iterate on motion source or target choice, not on a refinement prompt.\n\n## Sources\n\n- [fal.ai — Kling Motion Control V3 Standard](https://fal.ai/models/fal-ai/kling-video/v3/standard/motion-control)\n- [fal.ai — Kling Motion Control V3 Pro](https://fal.ai/models/fal-ai/kling-video/v3/pro/motion-control)\n",
|
|
27
|
+
"slates-prompting-nano-banana-2": "---\nname: slates-prompting-nano-banana-2\ndescription: \"Prompt Nano Banana 2, Lite or Pro image generation and edits. Use on these models; covers photographic craft, subject and reference binding, typography, prompt structure and family differences.\"\n---\n\n# Nano Banana 2 — cinematic & photorealistic prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Nano Banana 2 (Gemini 3.1 Flash Image).** Brief it like a creative director, not a tag list. Structure: `Film still from [director] [genre]. Shot on [camera] with [lens]. [Subject and action]. [3-5 specific visual details]. [Lighting — direction + quality]. [Color palette]. [Film stock]. [1-2 word tone].`\n\n**The five levers**\n1. **Named lens + aperture** beats \"shallow depth of field\" — `85mm f/1.4`, `135mm f/2.8`, `Panavision anamorphic`, `400mm telephoto`.\n2. **Light by direction and quality**, never \"good lighting\" — `hard sidelight from a single window, deep falloff`, `overcast north light`, `practical tungsten spill`.\n3. **A named film stock or sensor** carries a whole palette — `Kodak Portra 400`, `Cinestill 800T`, `ARRI Alexa 65`.\n4. **Composition as a shot** — `low angle`, `aerial view`, `rule of thirds with the subject camera-left`, `foreground occlusion`.\n5. **Positive framing only.** Describe what is there. \"Empty street\", never \"no cars\"; \"unstaged documentary photography\", never \"not anime\".\n\n<!-- @inject:cinematic-card -->\n**For a photographic look, use only what this frame needs.** Image models default to clean, evenly lit and fully exposed. Describe what the camera sees, not just gear or mood:\n- **Inspect every reference first.** Write its grade and imperfections in words: darkness, contrast, muddy or true blacks, colour, softness/noise, subject separation. Never grade cleaner or brighter than the look reference unless asked.\n- **One light system** — `low sun behind her`, `her face falls into deep shadow`, `no light in front of her`.\n- **Visible exposure** — `the sky burns out to white`, `dense, slightly crushed shadows`.\n- **Lens name plus effect** — `200mm telephoto`, `peaks loom huge behind her and melt into soft shapes`.\n- **Name every garment and close the foreground.** Omissions invite reference leakage or invented props.\nBind references inline. A scene reference owns the grade; for a look-only reference, write the new scene's light. References are optional. For owned-frame edits, describe only the change and what stays.\n<!-- slates-only -->Use `slates-cinematic-look` with a technique ID or section query for more.<!-- /slates-only -->\n<!-- @end:cinematic-card -->\n\n**Hard constraint:** there is no `negativePrompt` field. Suppress by reframing positively, or inline `without` / `free of`. Knowledge cutoff January 2025 — anything later needs reference images.\n<!-- @card:end -->\n\nNano Banana 2 is **Gemini 3.1 Flash Image**.<!-- slates-only --> It is the headless image model used when `projectId` is omitted — the op also exposes `flux-2-max` and `seedream-5-lite`, each with its own prompting skill.<!-- /slates-only --> It is **not** Gemini 3 Pro Image; that is Nano Banana **Pro** (`nano-banana-pro`), a separate model with its own seat.<!-- slates-only --> Verified against the runtime slug map in `slate/src/main/api/google.ts`.<!-- /slates-only --> NB2 is a language model that outputs pixels — brief it like a creative director, not like a Stable-Diffusion tag-soup tool. The single biggest lever for realism: **specificity that mimics how real photographers and cinematographers describe their work**.\n\nKnowledge cutoff: January 2025. Anything after needs explicit reference images.\n\n## Google's 4 official rules (verbatim)\n\n1. **Be specific.** Provide concrete details on subject, lighting, and composition.\n2. **Use positive framing.** Describe what you want, not what you don't want.\n3. **Control the camera.** Use photographic and cinematic terms like \"low angle\" and \"aerial view.\"\n4. **Iterate.** Refine images with follow-up prompts in a conversational manner.\n\n## Official prompt formula\n\n```\n[Subject] + [Action] + [Location/context] + [Composition] + [Style]\n```\n\nFor the cinematic / photoreal use case, expand to:\n\n```\nFilm still from [DIRECTOR] [GENRE]. Shot on [CAMERA] with [LENS]. [SUBJECT and action]. [3-5 specific visual details]. [LIGHTING — direction + quality]. [COLOR PALETTE]. [FILM STOCK or sensor language]. [1-2 word emotional tone].\n```\n\n## Photorealism positives — what consistently works\n\n⚠️ **This vocabulary is correct here and does not carry into a video prompt.** The leak happens one way: you write an NB2 start frame, then carry its look description straight into the prompt that animates it.\n\n<!-- @inject:lens-video-split -->\nNamed lenses, apertures, film stocks and camera bodies (`85mm f/1.4`, `Kodak Portra 400`, `ARRI Alexa 65`) are an image-model lever. On a video model, translate the look instead of pasting the gear list: `85mm f/1.4, Portra 400` becomes `close-up, shallow depth of field, warm natural colors, cinematic texture, film-grain texture`. ByteDance's Seedance 2.0 guide never mentions fps, shutter angle, f-stop or lens millimetres. Its Seedance 2.5 guide does, once: the visual-style line of its own storyboard example names one camera body and one 35 mm cinema lens. On 2.5 a single line like that is vendor-sanctioned; a stacked gear list still is not.\n<!-- @end:lens-video-split -->\n\n**Named lenses + apertures** beat generic \"shallow depth of field\":\n- `85mm f/1.4`, `135mm f/2.8`, `50mm f/1.2`, `35mm f/2`\n- `Panavision anamorphic` for horizontal flares + cinematic width\n- `400mm telephoto` for compression + isolation\n- `24mm` for environmental interiors\n\n**Named cameras / sensors:**\n- `ARRI Alexa 65`, `Hasselblad X2D`, `Canon EOS R5`, `Sony A7III`, `Fujifilm X-T5`\n- \"Specific gear\" beats \"DSLR\"\n\n**Named film stocks** (one per prompt — never mix):\n- `Kodak Portra 400` — natural skin, warm\n- `Fuji Velvia 50` — saturated, landscape\n- `Ilford HP5 Plus` — black and white, gritty grain\n- `CineStill 800T` — tungsten night, halation\n\n**Physics-based lighting** (direction + quality):\n- `Single key light at 45 degrees from upper left`\n- `Late afternoon sun at 15 degrees above horizon`\n- `Color temperature 4500K` beats `slightly warm`\n- `Practicals only — no fill` for Deakins-style realism\n\n**Imperfection vocabulary** (forces away from AI-clean):\n- `visible pores`, `natural skin grain`, `peach fuzz`, `slight hyperpigmentation`\n- `unretouched raw photography`, `ISO noise`, `sweat beading`\n- `crisp catchlights in the eyes`, `skin micro-detail`\n- Lead with the kind of photograph and the conditions on the skin (sun, wind, sweat), then add one or two of these. A bare list of flaw words read as tokens and produced plastic skin on GPT Image 2 (2026-08-24).<!-- slates-only --> Technique: `slates-cinematic-look` → `name-the-capture-context`.<!-- /slates-only -->\n\n**Director references** (use when locking style):\n| Director | Tone | Visual signature |\n|---|---|---|\n| Denis Villeneuve | Cold, vast, existential | Desaturated, overwhelming scale |\n| Roger Deakins | Precise motivated light | Single source, deep shadows, practicals |\n| Emmanuel Lubezki | Natural, spiritual | Available light, golden hour |\n| Bradford Young | Warm darkness | Underexposed, rich shadows, skin tones |\n\n**Genre cues that move the model:**\n- `unstaged documentary photography style`\n- `fashion magazine editorial, shot on medium-format analog film, pronounced grain`\n- `Film still from [Director] [genre]`\n\n## The anti-list — phrases that DEGRADE realism\n\nThese are Stable-Diffusion-era tag soup. The model treats them as low-signal noise. Measured success rate: ~60-70% with these vs ~95%+ with positive description.\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is extracted\n by src/prompts/banned-tokens.ts, inlined verbatim into the slates_generate_image\n op description (always in context on both surfaces), and matched against every\n submitted prompt. Editing this list changes what the agent is told AND what it\n is warned about — keep every entry backticked, and keep prose outside the\n backticks. -->\n<!-- /slates-only -->\n**Never use:**\n- `8k`, `4k` (as a quality token)\n- `hyperrealistic`, `ultra-realistic`, `photorealistic` standing alone\n- `masterpiece`, `best quality`, `highly detailed`, `ultra-detailed`\n- `trending on ArtStation`, `award-winning`\n- `perfect skin`, `flawless`, `airbrushed`, `smooth skin`\n- `cinematic` standing alone — always specify *which cinema* (director, lens, era, stock)\n- `not anime, not cartoon, not 3D` — negation tag soup, replace with a positive style cue\n<!-- @banned:end -->\n\n**Examples**\n- `Film still from a Denis Villeneuve thriller. Shot on ARRI Alexa 65, 85mm f/1.4. A woman in a charcoal wool coat stands at a rain-slick bus stop, breath visible. Hard sodium light from a single overhead lamp, deep falloff into blue night. Kodak Vision3 500T. Isolated.`\n\n## Negative prompting — there is no field\n\nNano Banana 2 has **no `negativePrompt` parameter**. Three patterns to suppress unwanted content:\n\n1. **Positive reframing (preferred):** \"empty street\" not \"no cars\". \"Unstaged documentary photography\" not \"not anime.\"\n2. **Inline `without` / `free of`:** \"without any people, vehicles, or man-made structures\", \"free of text overlays, logos, or watermarks.\"\n3. **Constraint clauses for anatomy/quality:** \"accurate anatomy with five fingers per hand, symmetrical features, natural proportions\"; \"sharp, well-exposed, free of blur or JPEG artifacts.\"\n\nDefault to #1. Reach for #2 only when positive framing can't suppress the unwanted element.\n\n## Reference images\n\n- **Hard limit: 14 images** (10 object-fidelity + 4 character-consistency). Categories don't trade — you can't use 14 object slots even if no characters are referenced.\n- **Name each reference inline — Slates does this for you.** When you `@mention` a subject/environment or `#mention` a style<!-- slates-only --> (or pass `referenceAssetIds`)<!-- /slates-only -->, Slates composes the prompt so each reference is named inline as \"image N\" — e.g. `Marcus (image 1) sits across from the woman (image 2) in the cafe (image 3)`, or `lit and graded like image 4` where you placed the style mention. An unmentioned style attachment gets a short fallback clause The model does NOT infer a reference's role from its position; the NAME carries it. NB2's own consistency lever is literally **\"assign a distinct name to each character/object\"**. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render the scene's expression\") — that drags the sheet's wardrobe + studio lighting into the scene. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n\n### Reference rules (the verified ones)\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n<!-- @inject:reference-rules-core -->\nIdentity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.\n<!-- @end:reference-rules-core -->\n\n### For Nano Banana 2 specifically\n\n- **NB2's own consistency lever is \"assign a distinct name to each character/object.\"** That is Google's phrasing for rule 3 — cite each canonical identity inline by name.\n- **Rule 8 is a job you do, not one you delegate.** NB2 *is* the start-frame model — when a downstream video shot needs legible text, render it here and animate from this frame.\n- **Character consistency is officially \"not 100% perfect\"** per Google. Test before bulk generations. High-resolution, front-facing reference images help most.\n- **Injection is stochastic — budget 3-5 re-rolls per shot; re-roll, don't re-engineer.** First rolls miss faces/hands; the same prompt lands a clean one within a few tries.\n\n## Common failure modes + fixes\n\n**Hands:** Append `accurate anatomy with five fingers per hand, symmetrical features, natural proportions, relaxed open palm`. Avoid heavy jewelry, props intersecting fingers, motion blur in references.\n\n**Text in images:** Quote-wrap target text. Specify font (`Century Gothic, 12pt`). Long phrases work; small text degrades. Two-step works best — generate text concepts conversationally first, then ask for the image.\n\n**Left/right confusion:** Default is **viewer's perspective**, not subject's. Append `left and right are from the character's perspective, NOT the camera's` when scene-blocking matters.\n\n**Surreal / absurd prompts trip uncanny valley:** The model drags toward realism. If you want surrealism, lean hard into stylization keywords (`painted`, `illustrated`, `stop-motion`).\n\n**Soft faces / dead eyes:** Add `crisp catchlights in the eyes`, `skin micro-detail`, `peach fuzz visible`. Don't stack quality enhancers — single clean prompt beats multiple re-interpretations.\n\n**Post-cutoff content (anything after Jan 2025):** Use reference images. The model has no knowledge of recent franchises, products, events.\n\n## Resolution tactics\n\n- Resolution is priced: NB2 4k costs roughly 2x 1k. Prices change — check current numbers<!-- slates-only --> by calling `slates_estimate_generation_cost`<!-- /slates-only -->. Pick the cheapest resolution that serves the use case.\n- **At 2K and above, the model allocates more tokens to surface detail** — explicit texture vocabulary (pores, fabric weave, grain) compounds at higher resolution.\n- 1k for fast iteration / drafts; 2k for hero shots; 4k only when you need print-grade detail.\n- 2K generations vary 20-60s+. Don't time-budget tightly.\n\n## Boring vs cinema — examples\n\n❌ **Boring:** \"Wide shot of a man on a dock looking at the forest.\"\n\n✅ **Cinema:** \"Direct overhead drone shot on weathered dock surface. Single figure standing center frame, climbing up from frame bottom. Boot prints leading away from him toward shore. Pale winter light. Anamorphic lens flare from low sun. Desaturated blue and slate grey palette. Kodak Portra 400 grain. The path already walked by someone else. Map of threat.\"\n\n❌ **Boring:** \"Close up of a woman looking scared.\"\n\n✅ **Cinema:** \"Extreme close on subject's mouth and nose, 135mm f/2.8, shallow depth of field. Breath pluming out, catching cold light from upper-left key. Lips slightly parted, peach fuzz visible. The breath holds. CineStill 800T halation around catchlights. Waiting.\"\n\n<!-- @inject:iteration-diagnosis -->\n## Diagnose repeated failures\n\nAfter three failed attempts at the same requirement, pause unchanged re-rolls and diagnose the source reference, prompt structure, model fit and tool result. Three is a review checkpoint, not a universal limit or proof that the seed cannot matter. Preserve the attempts and name what each test changed.\n\nContinue autonomously when the brief is clear, a specific correction is supported and the next request is already authorized. Hand control back when taste or intent cannot be inferred, the next request needs fresh consent, or the available tool cannot meet the requirement. A failed roll never authorizes an additional charge. Follow the existing batch and per-request cost policy.\n<!-- @end:iteration-diagnosis -->\n\n## Family variants — Lite and Pro\n\nEverything in this skill applies to the whole Nano Banana family; two variants trade speed/ceiling around NB2 full:\n\n- **nano-banana-2-lite** — ~half the price, ~2.7× faster, **1K output only**, max 4 refs. The draft/iteration seat: explore compositions here, then re-run the winner on NB2 full at 2K/4K. Same Gemini filter.\n- **nano-banana-pro**: the hero-frame/typography ceiling (2× NB2 at 1K, 1.33× at 2K, about 1.9× at 4K; 4K native). NB2 ≈ 95% of Pro; escalate only when spatial composition, cinematic lighting/skin, fine typography-in-scene, or deep multi-element frames must be perfect. Up to 14 refs; it takes a full subject library in one call.\n\n<!-- slates-only -->\nRouting between them (and vs GPT Image 2.5 / FLUX / Seedream): `slates-model-selection`.\n<!-- /slates-only -->\n",
|
|
28
|
+
"slates-prompting-omni-flash": "---\nname: slates-prompting-omni-flash\ndescription: \"Prompt Gemini Omni Flash video generation or Omni Flash edits (omni-flash, omni-flash-edit). Covers native sound, reference inputs and short change-only prompts that preserve edit fidelity.\"\n---\n\n# Gemini Omni Flash — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Gemini Omni Flash.** Two different jobs with OPPOSITE prompt rules, and getting them the wrong way round is the whole failure mode.\n\n**The five levers**\n1. **Editing: short prompt, ONE change, nothing else.** Google's own doc says so and a 2026-07-09 receipt confirms it — a long \"keep every frame identical\" preamble produced WORSE drift than two sentences.\n2. **Editing: always end with `Keep everything else the same.`** — the one documented preservation lever.\n\n3. **Editing: describe the EFFECT, never a real object as a metaphor.** \"Candle-like flame\" rendered a literal candle in the subject's hand.\n4. **Editing: no chained stage directions.** Several beats cued to moments (\"…flies onto his shoulder when he calls it, and perches as he walks…\") hard-failed with `invalid_request`. One effect tied to an action already in the footage (\"…when he snaps his fingers…\") passed. Collapse to one continuous action; the model syncs to the footage's own motion.\n5. **Generation: the opposite — describe fully.** Subject, action, setting, `camera tracking alongside`, `overcast flat light`, tone. Audio is prompt-driven with no parameters: dialogue in quotes, sound in plain language — `rain patters on the tin roof`, `spray from tyres`, `a horn somewhere behind`.\n\n**Examples**\n- Edit: `Small magical flames appear on his fingertips when he snaps his fingers, and vanish when he blows on them. Keep everything else the same.`\n- Generate: `A courier in a yellow shell jacket weaves between stalled cars on a wet arterial road, camera tracking alongside at shoulder height. Overcast flat light, spray from tyres. Rain patters on car roofs, a horn somewhere behind.`\n\n**Hard constraint:** it is a CHEAP DRAFT seat for generation and the EDIT-fidelity winner for footage-synced VFX — never a hero generation shot. Expect a possible jitter or doubled speech beat in the last half second of an edit: trim the tail rather than burning a re-roll.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use in an EDIT prompt** (each one has a receipt above):\n- a long preservation preamble — it produces WORSE drift than `Keep everything else the same.`\n- a real object as a metaphor: `candle-like`, `flame-like`, `laser-like`\n- several staged beats cued to moments (`when he calls it, and perches as he walks`): these hard-fail, they do not merely drift. One effect tied to an action already in the clip (`when he snaps his fingers`) passed\n- harm-to-person framing: `ignite`, `catch fire`, `on fire` applied to a person trips the safety filter\n<!-- @banned:end -->\n\nGoogle's fast video generation + editing model (\"Nano Banana Pro for video\" in creator slang — a nickname; it is NOT the NB Pro image model). Carried on fal (`google/gemini-omni-flash*`). 720p only, 24fps, 3–10 second clips, 16:9 or 9:16. **Audio is native and included** — dialogue, SFX, and ambient generate WITH the video at no extra cost.\n\n## Where it routes\n\n- **Video editing (`omni-flash-edit`) — its headline strength and the edit-lane default** for footage-synced VFX: verified 2026-07-09 head-to-head vs Kling O3 Edit on real phone footage (fire-on-fingertips on a talking take) — Omni Flash held lip movement perfectly, audio near-identical, and executed both action beats; Kling kept audio verbatim but drifted lips and missed the second beat. Full routing: slates-model-selection.\n- **Drafts and iteration volume with sound**: an audio-native video seat (~6.4 cr/s at 720p).\n- **Hero-generation quality is still unproven.** Use the current model-routing guide for the final generation seat; the edit-fidelity receipt does not establish generation quality.\n\n## Editing (<!-- slates-only -->`slates_edit_video`, <!-- /slates-only -->model `omni-flash-edit`) — THE RULES (receipts, not theory)\n\n1. **SHORT PROMPT. One change. Nothing else.** Google's own doc: *\"Simple prompts work best for video editing. Overly descriptive prompts can lead to unintended changes.\"* Live receipt 2026-07-09: a long \"keep every frame/word/movement identical…\" preamble produced WORSE drift (re-synthesized performance, wrong timing); the winning prompt was two sentences: *\"Small magical flames appear on his fingertips when he snaps his fingers, and vanish when he blows on them. Keep everything else the same.\"*\n2. **Always end with \"Keep everything else the same.\"** — the one documented preservation lever.\n3. **Never name a real-world object as a metaphor.** \"Candle-like flame\" rendered a literal candle in his hand. Describe the effect itself (\"small magical flames on his fingertips\").\n3b. **No chained stage directions: they HARD-FAIL, not drift.** Receipt 2026-07-09: \"a dragon appears behind him, flies onto his shoulder WHEN HE CALLS IT, and perches AS HE WALKS…\" → deterministic `invalid_request` (2×, \"could not generate with the given inputs\"); collapsing to one continuous action — \"A small photorealistic dragon flies in and perches on his shoulder, puffing a small breath of flame and smoke.\" — succeeded first try. The model syncs the change to the footage's own motion; it cannot take beat-by-beat stage directions cued to moments in the video. The winning flames prompt above (\"when he snaps his fingers\") shows one effect tied to an action the footage already contains is fine; which part of the dragon prompt triggered the refusal is untested beyond that.\n4. **Safety filter (Google's, strict about harm-to-person):** \"fingertips ignite / catch fire\" → `content_policy_violation`. Frame effects as magical/harmless VFX: \"small magical flames appear on his fingertips\" passed. See slates-content-policy §Gemini for the substitution patterns.\n5. **Expect a possible tail artifact** — jitter or a doubled final speech beat in the last ~0.5s. Plan to trim the tail on the timeline; don't burn a re-roll on it.\n6. **Prompt + source clip ONLY.** No element/style reference images — identity swaps that need refs go to `kling-v3.0-omni-edit`.\n7. Source clip 3–10s (trim longer clips first). Output length follows the source; billing per output second, rounded up. Voice editing unsupported — never ask it to change dialogue.\n8. **Ship via segment-splice** (the workflow, not the model): edit only the seconds where the change happens, splice back over the original on the timeline with the original audio underneath. Most of the deliverable stays untouched original footage — this is how the pro demos are actually assembled (gesture-only edited beats + voiceover in post).\n9. Chain edits one change at a time — each edit saves as a new asset linked to its parent.\n\n## Generation (<!-- slates-only -->`slates_generate_video`, <!-- /slates-only -->model `omni-flash`)\n\n- **Inputs:** prompt only (t2v), prompt + ONE start frame (<!-- slates-only -->`firstFrameAssetId`, <!-- /slates-only -->i2v), or prompt + up to **7 reference images** (ingredient/character/environment/style asset params — they merge into one reference list). No last frame, no video/audio references — the op rejects them.\n- Descriptive prompts are fine for GENERATION (the short-prompt law above is edit-specific). Structure like a shot brief: subject + action + setting + camera + lighting + tone.\n- **Name references inline** the standard Slates way (\"Marcus (image 1) walks…\"). The endpoint also accepts explicit `<IMAGE_REF_0>`-style binding tags (zero-indexed) — useful when a specific image must bind to a specific role.\n- **Audio is prompt-driven** — no audio parameters. Dialogue in quotes; direct sound in plain language (\"rain patters on the tin roof\"). Negative direction as plain instructions (\"Do not show text\").\n- Duration is an explicit 3–10s integer param; cost scales linearly per second.\n\n## Input conditioning (Slates handles this — know it exists)\n\nPhone footage stores rotation as a metadata flag; models ignore it and edit the raw sideways pixels. Clips must be rotation-normalized (and oversized sources downscaled) before upload — receipt 2026-07-09: a portrait Pixel clip came back sideways until conditioned. If an edit output comes back rotated, the source wasn't normalized.\n\n## Content notes\n\n- Google applies its own safety filters to input images/clips and output. Uploads containing recognizable real people are restricted by Google's policy — though own-footage editing of the uploader passed on our route 2026-07-09. See slates-content-policy.\n- Output carries an invisible SynthID watermark (Google-side, programmatic detection only).\n",
|
|
29
|
+
"slates-prompting-seed-audio": "---\nname: slates-prompting-seed-audio\ndescription: \"Prompt Seed Audio 1.0 (seed-audio) with slates_generate_audio. Covers scene sentences, dialogue, crowd size, spatial sound, duration in prompt text and mutually exclusive audio or image references.\"\n---\n\n# Seed Audio 1.0 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Seed Audio 1.0.** One plain sentence describing a SCENE, and it returns dialogue, effects and ambience together in one pass. For ONE named voice saying ONE line, `inworld-tts-2` is the seat instead; this one renders the whole room.\n\n**The five levers**\n1. **Write one sentence in plain language.** Describe the room and what is happening in it; the model casts and performs it.\n2. **Name the crowd size, the room size and the distance.** This is the highest-leverage single edit on any bed — unqualified nouns default BIG. \"Tiny applause of two or three people at an open mic\", not \"applause\".\n3. **Put the length in the prompt** and make it deliberate.\n4. **Dialogue is performed inside the scene** — write the line as spoken in the room, then re-roll until a take is right and lip-sync against it.\n5. **Describe sounds directly** — \"one coffee machine hissing, cutlery somewhere behind the counter\" — with the object, the action and where it is.\n\n**Examples**\n- `a diner at 2am, one coffee machine hissing, cutlery somewhere behind the counter, one man says quietly \"you're late again\". 15 seconds`\n- `nature soundscape, a wide open field, cicadas near, birds mid-distance, one loon far off across water. 20 seconds`\n\n**Hard constraint:** it has NO duration parameter — the length you request is written into the prompt AND is what you are billed, whatever comes back, so choose it deliberately. Kling's `SFX:` / `Ambient noise:` / `Background music:` labels have no parser here and measurably worsen the output. Shot language, camera moves and lighting are video-prompt words the model must ignore. It is not a music model and it cannot produce pixels.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — Kling video syntax has no parser here and measurably worsens the output:\n- `SFX`, `Ambient noise`, `Background music` as labels — describe the sounds directly\n- shot language: `wide shot`, `slow push in`, `warm tungsten` — camera and lighting words are video-prompt words the model has to ignore\n<!-- @banned:end -->\n\nByteDance's one-pass audio scene model, carried on fal (`bytedance/seed-audio-1.0`). It generates dialogue, sound effects and ambience **together**, from a single plain sentence. Use the generated duration bounds below and the current catalogue for routing; it is useful for continuity beds.\n\n<!-- @inject:thresholds -->\n<!-- GENERATED from @slatesvideo/shared — do not edit between the markers.\n Source: CONFIRM_CREDITS, DEVIATION_FACTOR and the audio bounds in\n packages/shared/src/operations/index.ts. Every number here is REFUSED by an\n op when a prompt gets it wrong, which is why none of them is typed by hand\n any more: this block replaced four claims that contradicted the code. -->\n\n**The thresholds, from the code that enforces them:**\n\n- **Confirm gate:** above **17 credits** an op returns `requires_confirm` and will not\n proceed until you re-call with `confirm: true`. This is a code gate, not permission to spend: every generation still needs the user-approved plan or quote.\n- **Deviation pause:** the desktop Studio Agent stops and re-asks when projected generation spend\n exceeds the approved plan by more than **20%**. You do not trigger this; the app does.\n- **Seed Audio duration:** **3–120 seconds.** There is no duration\n parameter on the model — the number you pass is written into the prompt AND is what the user is\n billed. Outside that range the op refuses rather than clamping.\n- **Sound Effects duration:** **1–22 seconds**, billed per second, never left for the\n model to pick.\n\nNever quote a credit figure from memory: `slates_estimate_generation_cost` returns the real one.\n<!-- @end:thresholds -->\n\n## Where it routes\n\n- **Scene audio, room tone, ambience beds, crowd/nature soundscapes** — anything where several sounds share a space. One generation, not three layered ones.\n- **Dialogue and scratch VO inside a scene.** The line is performed in the room, by a voice the scene casts. When WHO is speaking matters — a character's own voice, a clean narrator — that is `inworld-tts-2` instead. Lock the read by re-rolling until a take is right, then lip-sync against it with `slates_generate_lip_sync`.\n- **NOT** a single effect that must land on a known frame — that is `eleven-sfx`, which takes an exact duration.\n- **NOT** music. Slates has no music model; import a track and drop it on an audio track.\n- **AUDIO-ONLY.** It cannot produce images or video.\n\n## THE RULES\n\n### 1. 🚨 There is no duration parameter — the words set the length\n\nThis is the single most important fact about this model. Output length is driven by the prompt text (\"… 15 seconds\"), capped at 120s.\n\n**Slates handles this for you:** the `durationSeconds` param appends the duration to the prompt and **bills that number of seconds**. So:\n\n- Set `durationSeconds` to what you actually want.\n- **Do not also write a different length into your sentence.** Two numbers fight, and you pay for the one you selected, not the one you got.\n- If the returned clip is shorter than requested you still paid for the request — that is the deal that keeps the displayed price equal to the charge. Ask for what you need.\n\n<!-- slates-only -->\nThe server re-derives the billed key from `durationSeconds` (a client cannot under-bill), probes the returned `audio.duration` after completion, and logs `SEED AUDIO BILLING DRIFT` if the model overshot. No auto-charge, no refund — the request is the contract.\n<!-- /slates-only -->\n\n### 2. One plain sentence. No production jargon.\n\nField-proven (Higgsfield sprint, 2026-07-27/28). Working prompts look like this:\n\n```\ntiny applause of 2 or 3 people at an open mic. 15 seconds\nnature soundscape, wide open field cicadas and birds and a loon.\na diner at 2am, one coffee machine hissing, cutlery somewhere behind the counter\n```\n\nNot this:\n\n```\n✗ AMBIENCE: interior diner, night. SFX: espresso machine (hiss, 2s), cutlery.\n✗ Wide shot of a diner. Slow push in. Warm tungsten. Ambient noise: ...\n```\n\nShot language, camera moves and lighting belong to video prompts. Here they are just words the model has to ignore.\n\n### 3. Never bring Kling's audio syntax to this model\n\n`SFX:` and `Ambient noise:` prefixes and `Background music:` labels are **Kling 3.0 video** syntax. Seed Audio has no parser for them — it reads them as text in the scene and the output gets measurably worse. Describe the sounds directly instead.\n\n### 4. Name the crowd size, the room size, the distance\n\nThe highest-leverage single edit on any bed. Unqualified nouns default big:\n\n| Vague | What it returns | Fixed |\n|---|---|---|\n| `applause` | a full auditorium | `tiny applause of 2 or 3 people` |\n| `traffic` | a highway | `one car passing on a wet residential street` |\n| `crowd` | a stadium | `four people talking at the next table` |\n\nDistance words (`far off`, `muffled through a wall`, `right next to the mic`) work the same way and are how you build depth in one sentence.\n\n### 5. Beds must outlast the cut\n\nAsk for a few seconds more than the clip needs so the edit has handles to fade through. A bed that ends exactly on the cut always sounds clipped. This is a product requirement, not a preference — it is why the duration control exists at all.\n\n### 6. Dialogue goes in quotes, inside the same sentence as the room\n\n```\na tired bartender says, \"we closed twenty minutes ago\", glasses clinking behind him\n```\n\nPick a preset voice when a specific speaker matters. Leave `voice` unset and the scene casts itself — which is usually right for crowd and background dialogue.\n\nPreset voices (20): `vivi_mixed_en_zh_ja_es_id`, `mindy_en_es_id_pt_zh`, `kian_en_zh`, `cedric_en_zh`, `sophie_en_zh`, `jean_en_zh`, `magnus_en_zh`, `mabel_en_zh`, `nadia_en_zh`, `opal_en_zh`, `pearl_en_zh`, `quentin_en_zh`, `corinne_mixed_en_zh`, `esther_mixed_en_zh`, `lyla_mixed_en_zh`, `tracy_es_zh`, `sandy_es_mixed_en_zh`, `felix_zh`, `celeste_zh`, `monkey_king_zh`.\n\nSet `multilingual: true` for non-English or mixed-language lines.\n\n### 7. Inputs: up to 3 audio clips **XOR** one image. Never both.\n\n- **Audio references** — up to 3 clips, each ≤30s and ≤10MB (wav/mp3/pcm/ogg_opus). Refer to them in the prompt as `@Audio1`, `@Audio2`, `@Audio3`: *\"match the room tone of @Audio1\"*.\n- **Image reference** — one image (jpeg/png/webp ≤10MB). The model scores what it sees.\n- Sending both is rejected by the API. Pick the one that carries the intent.\n\n### 8. The knobs, and when to touch them\n\n| Param | Range | Reach for it when |\n|---|---|---|\n| `speed` | 0.5–2.0 | Dialogue is racing or dragging against picture. |\n| `volume` | 0.5–2.0 | Rarely — normalize on the timeline instead. |\n| `pitch` | −12…+12 semitones | Ageing or shifting a voice. Small moves only; ±3 is already a lot. |\n| `multilingual` | bool | Non-English or code-switched lines. |\n\n## Iterating\n\n- A bed that came back wrong is almost always a **scale** problem (crowd/room too big) or a **jargon** problem (the sentence reads like a spec). Fix those two before touching `speed`/`pitch`.\n- Three failed takes on the same sentence means the sentence is wrong, not the seed. Rewrite it the way you would say it out loud.\n- Generations are cheap enough at short durations that auditioning two phrasings beats agonizing over one.\n\n## Content notes\n\nProvider-side moderation applies to voices and to recognizable real people. See slates-content-policy.\n",
|
|
30
|
+
"slates-prompting-seedance-2-5": "---\nname: slates-prompting-seedance-2-5\ndescription: \"Prompt Seedance 2.5 generation (seedance-2.5) or edits (seedance-2.5-edit). Covers whole-second timing, shared Seedance craft, reference inputs, task-classifier hazards and long-take spend.\"\n---\n\n# Seedance 2.5 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Seedance 2.5.** Shares 2.0's grammar exactly (subject binding, camera vocabulary, externalised emotion, inline constraints — read `slates-prompting-seedance` for those). Two things are different, and both matter.\n\n**The five levers**\n1. **Timestamps work here**: integer seconds, and the model acts on them: `[0s-4s] she reads the letter. [4s-9s] she folds it and looks up.` 2.0 ignores exactly this syntax.\n2. **Length is the reason to be here**: takes up to 30 seconds, where 2.0 stops at 15. Write the beats as `[0s-6s]`, `[6s-12s]`, `[12s-18s]`; do not hope for them.\n3. **Up to 30 image references**, and a multi-view image can serve as ONE subject reference (up to 5 subjects). 2.0 cannot do either.\n4. **Audio-only references are accepted** without an image or video alongside — the only Seedance seat that takes one.\n5. **Keep the 2.0 discipline**: one camera move per beat (`slow track right`, `handheld follow`), physical action instead of stated emotion, and quality asked for in the image-quality slot vocabulary — `rich details`, `natural colors`, `cinematic texture`, `soft lighting`.\n\n**Examples**\n- `[0s-6s] Wide shot, <Subject_1>@<Image_1> crosses an empty car park toward a idling van, slow track right. [6s-12s] Medium, she stops as the driver's window comes down. [12s-18s] Close-up, she looks off past the lens and does not answer. Rich details, natural colors. Keep it subtitle-free.`\n- `[0s-10s] A single continuous handheld follow behind a courier climbing a fire escape, rain. [10s-20s] She reaches the landing, turns, and the city opens behind her. Cinematic texture, soft lighting.`\n\n**Hard constraint:** it is the default AND the dearer seat, and it has NO 4K — 480p/720p/1080p only, dearer than 2.0 at every resolution they share. Long takes multiply cost linearly: quote a 30-second take before you fire it.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** (with a reference video, 2.5 can reclassify the task and fail a fresh generation on these):\n- `edit`, `extend`, `continue the video`, `same video but` — they make the provider read a fresh generation as an edit\n- `f/1.4`, `Portra 400` and any other aperture or film-stock token, or a stacked list of gear — image-model vocabulary. The 2.5 guide's own example names one camera body and one 35 mm lens in a single style line, so a lone lens there is not on this list\n<!-- @banned:end -->\n\n**Read `slates-prompting-seedance` first.** The prompt GRAMMAR is the same model family: the\n8-slot advanced formula, subject binding by `<Subject_N>@<Image_N>`, camera vocabulary,\nexternalised emotion, inline constraint words, the anti-twin fix. None of it is restated here.\nThis file is only what 2.5 changes — and the biggest change is that **2.5 acts on timestamps\nwhere 2.0 ignores them.**\n\n---\n\n## The one fact that decides whether you use it at all\n\n**Seedance 2.5 is the EXPENSIVE seat, not the cheap one — and it has no 4K.**\n\nIt runs at 480p, 720p or 1080p (1080p landed on all three routes on 2026-08-24), and at every\nresolution the two seats share it costs MORE than 2.0 — 720p $0.231/s against $0.15/s, **54% more**.\nSo 2.5 does not replace 2.0; it sits beside it, and you pay for what it buys:\n\n| | Seedance 2.0 | Seedance 2.5 |\n|---|---|---|\n| Resolution | 480p / 720p / 1080p / **native 4K** | 480p / 720p / 1080p — **no 4K** |\n| Price at 720p (faceless) | **$0.15/s** | $0.231/s |\n| Length | 4–15s | **4–30s in one take** |\n| Reference budget | 12 files total (9 image / 3 video / 3 audio caps) | **50 (30 image + 10 video + 10 audio)** |\n| Combined reference video/audio | ≤15s | **≤30s** |\n| Audio-only reference | ✗ (needs an image or video alongside) | **✓** |\n| **Timestamps in the prompt** | **✗ — ignored; shot numbers only** | **✓ — integer seconds, acted on** |\n| Multi-view image as ONE subject reference | ✗ (not recommended) | **✓ (up to 5 subjects)** |\n| Video edit as its own task type | ✗ | **✓ (`seedance-2.5-edit`)** |\n| Default video model | no | **yes** (since 2026-09-13) |\n\n**2.5 is the default. Route to 2.0 for 4K delivery, or when the same resolution has to be\ncheaper** — its 720p is $0.15/s against 2.5's $0.231/s.\n\n---\n\n## 🚨 Hazard 1 — the prompt-intent task classifier\n\nThis is the one that costs money and time, and it has no equivalent on 2.0.\n\n**Seedance 2.5 sorts every request into one of five task types** — text-to-video,\nreference-to-video, first/last-frame, **video edit**, **video extend** — from the reference roles\nattached **plus the intent of your sentence**. Each type then has its own parameter constraints,\nand a violation comes back **asynchronously**: the task queues, credits are reserved, and only then\ndoes it fail.\n\nThe trigger words are ordinary English:\n\n| Reclassified as | Words that do it (ByteDance's own list) |\n|---|---|\n| **video edit** | `edit video` · `add` · `insert` · `remove` · `delete` · `modify` · `replace` · `change to` |\n| **video extend** | `extend forward` · `extend backward` · `continue` · `continue from` · `extend the story` |\n\nSo a perfectly legitimate prompt with a reference video, *\"a wide shot of the workshop, **remove** the\ntripod from frame\"* — gets classified as an edit and fails on constraints it never set.\n\n**What to do:**\n\n1. **If you mean to edit an existing clip, choose its dedicated video-edit endpoint.** The\n task-typed endpoint removes the classifier's ambiguity. <!-- slates-only -->In Slates, call\n `slates_edit_video` with `model: 'seedance-2.5-edit'`.<!-- /slates-only -->\n2. **If you mean a fresh shot, describe the finished frame rather than an instruction to change\n one.** Not *\"remove the tripod\"* → *\"the workshop bench, clear and uncluttered\"*. Not\n *\"add rain\"* → *\"heavy rain falling through the streetlight\"*. This is better prompting anyway:\n the model renders what you describe, it does not take edits to an imagined draft.\n3. The trigger needs **a reference video plus edit or extend intent**. Image references alone do not trigger it. A plain text-to-video prompt is safe\n however it is worded.\n\n**Slates will warn you, and it will never rewrite your prompt.** When a 2.5 reference generation's\nprompt contains one of these words, the composer shows a warning that NAMES the words and the agent\nroute returns the same string. Silently editing the user's sentence to dodge a provider classifier\nis forbidden — the words that reach the model are always the words the user can see.\n\n---\n\n## 🚨 Hazard 2 — resolution is not the price dial here. LENGTH is.\n\nEvery other model in Slates trains the habit that lower resolution means lower cost. 2.5 breaks it,\nbecause the thing that moves the bill is **length**, and 2.5's length ceiling is double 2.0's.\n\nWorked, at the shipped rates:\n\n| Generation | Credits |\n|---|---|\n| 2.5 · 480p · 5s · faceless | 26 |\n| 2.5 · 720p · 5s · faceless | 58 |\n| 2.5 · 1080p · 5s · faceless | 142 |\n| 2.5 · 720p · 30s · faceless | 347 |\n| 2.5 · 720p · 30s · AI-face route | **489** |\n| 2.5 · 720p · 30s · consented real-face route | **710** |\n| 2.5 · 1080p · 30s · faceless | **853** |\n| 2.5 · 1080p · 30s · consented real-face route | **1,749** |\n| *(for scale)* 2.0 · 1080p · 15s · AI-face route | 411 |\n\n**A 30-second 720p clip can cost more than a 15-second 1080p one** — and a base licence starts\nwith 1,000 credits. Someone who reads \"720p\" as \"cheap\" and asks for a 30-second take on the\nreal-face route has spent 71% of their welcome grant on one clip; **on the real-face route a single\n30-second 1080p take is more than the whole grant.**\n\n**Discipline:**\n\n<!-- slates-only -->\n- **Always quote with `slates_estimate_generation_cost` before a take over ~10 seconds,** and say\n the number out loud before generating.\n<!-- /slates-only -->\n- **Find the shot at short LENGTH, not at low resolution.** Length is what moves the price, so cut\n seconds while you are still exploring — 4–8s — and stay at the resolution you actually want.\n **A 480p pass does not de-risk a 720p or 1080p render.** Generation is stochastic: the higher-\n resolution run is a different take, not the same shot rendered better. So a 480p draft that looks\n right buys you no guarantee, and one that looks wrong may have been fine at 720p — you paid 26\n credits to learn nothing, when 58 would have bought a real candidate.\n- **Length is a creative decision, not a default.** 30 seconds is available; it is rarely the right\n answer for a single shot. Multi-shot storyboards inside one 30s generation are what the length is\n actually for.\n<!-- slates-only -->- Read `slates-cost-discipline` — all of it applies, more sharply here.<!-- /slates-only -->\n\n---\n\n## Timestamps — the one grammar change\n\n<!-- @inject:seedance-25-timestamps -->\n**2.0 does not respond to timestamps and answers only to shot numbers. 2.5 responds to\ninteger-second timestamps.** That is ByteDance's own first line under \"Differences from Seedance\n2.0\", and it is why a 30-second take is usable at all: the length is only worth buying if you can\nsay *when* things happen inside it.\n\nBoth formats are valid on 2.5, and you can mix them — `Shot N` blocks for a storyboard whose\npacing you are happy to leave to the model, timestamps when a beat has to land at a moment.\n\n**Three ways to control time, all first-party:**\n\n| Form | Write it like |\n|---|---|\n| **Interval** | `0-3 seconds… 3-7 seconds… 7-15 seconds` or `[1s-4s]… [4s-8s]… [8s-12s]` |\n| **Time point** | *\"Quick left sideways transition at the 5-second mark.\"* |\n| **Relative** | *\"After 3 seconds, everyone around him shakes their head.\"* · *\"The frame freezes for 1 second after he presses the shutter.\"* |\n\n**The rules that come with them:**\n\n- **One second is the smallest unit.** Integers only — no `2.5s`, no frames.\n- **No gaps in the timeline.** `0-3s… 5-6s…` leaves 3-5s unspecified and the model fills it however\n it likes. Intervals must abut: `0-3s`, `3-7s`, `7-15s`.\n- **Budget the plot to the seconds.** Too little content in a range and the model improvises to\n fill it; too much and you get extra cuts or dropped beats. This is the actual craft of a 30s take.\n- **Never time-code a high-frequency action.** *\"Shake your head three times per second\"* is\n explicitly called out as a misuse — timestamps schedule beats, they don't choreograph frames.\n- **Transitions want both halves:** the moment AND the method — *\"At the 5-second mark, the camera\n transitions leftward with a left wipe into a natural dissolve.\"*\n- **Timestamps work on an EDIT too**, and that is where they earn the most: they scope a change in\n time as well as in content — *\"Change the man's action from drinking coffee to mopping the floor\n from 4-6 seconds in Video 1, and leave the rest of the content unchanged.\"* Without a range, a\n whole-clip instruction is applied to the whole clip.\n\nDo **not** carry this back to 2.0, and do not write `[00:00-00:02]` minute-second brackets (another\nvendor's syntax) into either — 2.0 ignores time entirely, and the cross-model syntax swap is its own known failure.\n<!-- @end:seedance-25-timestamps -->\n\n---\n\n## What the extra reference budget is actually for\n\n30 image references (up from 9) does **not** mean \"attach 30 images\". Every rule in\n`slates-prompting-seedance` about references still holds — 2–4 strong references beat both\nextremes, and one reference per role.\n\n**Where 2.5 moves the ceiling, per ByteDance's own input recommendations:**\n\n| | Stable | Works, but expect re-rolls |\n|---|---|---|\n| Subjects bound by IMAGE reference | 1–8 | 9–12 |\n| Subjects bound by VIDEO or AUDIO reference | 1–5 | 6–10 |\n| Reference clip length, per subject | 5–10s | longer drops stability |\n\n**Multi-view images of one subject are supported on 2.5** — a turnaround sheet can be a single\nreference image, where 2.0 wanted one authoritative rendering per subject. Past **5 subjects**,\ngo back to single-view images, one per view, rather than one image carrying several viewpoints.\n\nThe larger budget earns its keep in exactly two places:\n\n- **A long multi-shot take** where different shots need different subjects and locations bound —\n the budget is spread across the storyboard, not stacked on one frame.\n- **Video and audio references alongside images**, which is where 2.5's 10 + 10 matters far more\n than the image count.\n\n### Audio-only references — the genuinely new input\n\n2.0 required an image or video alongside any audio reference. **2.5 accepts audio on its own.**\nThat makes one recipe possible that was not before: drive a scene's timing, voice or ambience from\na recording with no visual anchor at all — a voice line, a music bed, a room tone — and let the\nmodel build the picture to it. Cite it the same way as any other reference\n(`Reference the timbre in <Audio_N> to generate…`), and remember that audio references carry **no\nbilling dimension** on any Seedance route: audio is included.\n\n### Video references\n\nUp to 10 clips, ≤30s combined (2.0: 3 clips, ≤15s). A reference VIDEO switches the cost key to\n`seedance-2.5*-vref-{res}-{T}s`, where **T = Σ input seconds + output seconds** on faceless and real-face routes;\non the AI-face route, **T = max(Σ input seconds, output seconds) + output seconds** on both 2.0 and 2.5. The sum is across\n**every** clip attached, not just the longest. Three 6-second references on a 12-second output bills\n30 seconds, not 12 and not 18. Quote before confirming.\n\n**The cap is a refusal, not a trim.** Attach an eleventh clip, or push past 30 combined seconds, and\nthe composition is rejected before anything uploads. That asymmetry is deliberate: reference images\nwarn-and-trim because dropping one doesn't change the price, and a dropped reference VIDEO would be\none you were quoted for and the model never saw.\n\n### Mixing all three in one call\n\n50 files total (30 image + 10 video + 10 audio) is a shared budget. Everything is cited positionally\nby type — `image 1`, `video 2`, `audio 1` — in attachment order, so reordering the attachments\nrenumbers the citations. Write the prompt against those numbers:\n\n```\nMarcus (image 1) performs the motion from video 1 in the workshop from image 2,\nusing the voice timbre from audio 1. Preserve his identity, appearance and outfit.\n```\n\n🚨 **SAY WHAT AN AUDIO REFERENCE IS FOR.** It can mean music, dialogue, voice, tone or timbre — five roles on one attachment — so an unroled clip falls back to **dialogue**: the model re-transcribes it and speaks ITS words. A real take came back as *\"a map called Slates\"* for *\"an app called Slates\"*. Name it as the voice timbre and the clip carries the voice while the prompt carries the words. ByteDance's own sentence: *\"Image 1 depicts the protagonist John and uses the voice timbre from Audio 1.\"* Bind each speaker in a sentence, never by attachment order — position carries nothing.\n\nFrames and reference media stay mutually exclusive, in every combination — the reference endpoint\nhas no first/last-frame parameters at all, so this is a shape mismatch rather than a preference.\n\n---\n\n## Sound: four bracket types, and they are the vendor's syntax\n\nByteDance's 2.5 API tutorial states this as a **prompt rule**, not a suggestion — verbatim: *\"Use\nspecial characters to distinguish sounds: `()` for music, `<>` for sound effects, `{}` for dialogue,\nand `【】` for subtitles. For non-Chinese dialogue, it is recommended to specify the language before\nthe dialogue.\"*\n\n```\nShe sets the cup down {English: \"We open in ten minutes.\"} <ceramic clink on wood>\n(low piano, unhurried)\n```\n\n- `()` **music** · `<>` **sound effects** · `{}` **dialogue** · `【】` **on-screen subtitles**\n- **Name the language before non-Chinese dialogue.** `{English: \"...\"}`.\n- Unbracketed sound description still works — this is a disambiguator, not a required wrapper. Reach\n for it when one sentence carries more than one kind of sound and you need the model to tell them\n apart, which is exactly where an unmarked prompt puts a line of dialogue into the score.\n\n⚠️ **These four are SEEDANCE 2.5's.** MiniMax H3 has its own three-layer scheme (body / soundscape /\nscore) and its angle brackets are documentation notation that must never be typed. Do not carry\neither grammar onto the other model.\n\n## Say what a reference is NOT for\n\nThe same rule adds a half nobody uses: *\"Specify what each asset provides, such as appearance,\naction, or timbre, **and what should not be referenced**.\"* Negative scoping is a first-class part of\nthe citation, not a fallback — *\"use her face and wardrobe from image 1, not its lighting or\nbackground\"* is a stronger instruction than naming the positive alone, because an unscoped reference\nbrings its whole frame with it.\n\n**BytePlus documents disagree on reference sigils.** Its API tutorial says *\"Use `@Image 1`,\n`@Video 1`, and `@Audio 1`\"*, while its 2.5 prompt guide uses the bare form\n(`Image 1 / Video 1 / Audio 1`) in its normative sentence. Both are first-party sources; the\nbare form is confirmed working on both Seedance models. Use the syntax accepted by the endpoint\nyou are calling, and preserve each asset's role and scope.\n\n<!-- slates-only -->\nThe disagreement is recorded in\n`second-brain/business/projects/slates/research/model-prompting-research.md`. Slates composes\nresolved reference tokens for the chosen model and now preserves unresolved sigils verbatim.\nThe earlier composer silently deleted unresolved `@Image 1` tokens; that receipt explained the\nold bare-form workaround, and the fix removes its blanket prohibition. Prefer actual bound\nreferences when available so the composer supplies the canonical citation.\n<!-- /slates-only -->\n\n## Seedance 2.5 Edit (<!-- slates-only -->`slates_edit_video`, <!-- /slates-only -->`model: 'seedance-2.5-edit'`)\n\nIts own picker row and its own op call, deliberately: the task type is **the model you chose**,\nnever something inferred from your sentence.\n\n**Why route here at all:** it is the **only edit engine in Slates that accepts a clip longer than 15\nseconds** (4–30s, versus Kling O3 Edit's 3–15s and Omni Flash Edit's 3–10s). For a clip inside the\nothers' range, choose on fidelity instead — Omni Flash Edit won the prompt-only head-to-head, and\nKling O3 Edit is the one that takes element and style reference images.\n\n**How it behaves:**\n\n- **Output length follows the SOURCE clip**, and the bill is the ceiled source length. The provider\n requires an automatic duration on this task type, so there is no length knob — the clip you attach\n is the quote. The returned clip can differ from the source by up to ~0.3s, which only compresses\n transition frames; a clip that 2.5 itself generated comes back at exactly its input length.\n- **Source clips under 20 seconds edit more reliably.** 4–30s is what the task type accepts;\n ByteDance's own recommendation is to stay inside 20 for quality. A 28-second source is legal and\n will need more attempts.\n- **The aspect ratio follows the source clip too.** No ratio control; the frame is the clip's frame.\n- **480p, 720p or 1080p output**, native audio — and an edit bills the video-reference tier ×2,\n so quote the 1080p edit before confirming.\n- **Prompt and source clip only** on this op. The MODEL takes reference images on an edit\n (ByteDance recommends 1–5 — *\"replace the man in dark clothing in @Video 1 with @Image 2\"*);\n **Slates has not wired that path**, so today an edit that must lock an identity from a photo\n goes to Kling O3 Edit. Constraint of our build, not of the model — worth revisiting.\n- **An edit costs about 1.2x a plain 2.5 generation of the same length**: every Seedance provider\n charges an edit on input + output seconds, at the reduced video-reference rate. Read the confirm\n gate's number; do not reason from the generation rate.\n- **Set `seedanceFace: true` when a character's face is visible in the clip.** The faceless provider\n blocks faces outright — this is not a price optimisation, it is whether the job runs at all.\n- **There is no consented-real-face route for editing.** Real-person footage that the AI-face route\n rejects has to go to Kling O3 Edit.\n\n**Prompting an edit** — the same discipline as every other edit engine: **describe only what\nchanges.** The source already carries its composition, motion, timing and performance; re-describing\nthem fights the model. Use Seedance's own edit grammar from `slates-prompting-seedance`\n(*\"Strictly edit `<Video_1>`, and modify `<Original_Characteristic>` to `<New_Characteristic>`\"*) and\n**never** write *\"reference video 1\"* in an edit — the official guide is explicit that this phrasing\ngets the request reclassified as a reference task, which is the same landmine as Hazard 1.\n\nTwo things sharpen an edit prompt, both first-party:\n\n- **Say it as A → B, not as an outcome.** *\"Change the man's action from drinking coffee to mopping\n the floor\"* beats *\"the man mops the floor\"* — naming what it currently is tells the model what\n to overwrite.\n- **Timestamp a partial edit** — the edit task type reads the same integer-second timestamps the\n generation path does. Rules and forms are in § Timestamps above; this is the single most useful\n thing they buy.\n\n**Audio is editable too, and it is the least obvious use of this row.** The same op rewrites what\nis heard while the picture stays put: change a spoken line, change the accent, translate the\ndialogue and re-fit the lip movement, strip or replace the BGM or a sound effect. *\"Only edit the\nman's dialogue in Video 1: change it to 'Don't come over here,' in an American accent\"* is an edit,\nnot a lip-sync job. Bill it like any other edit — on the source clip's length.\n\n---\n\n## Faces, and what does NOT change\n\n<!-- slates-only -->\nThe three-tier face routing is identical to 2.0 — faceless → default route, an AI character's face →\n`seedanceFace: true` (the relaxed provider, a real cost premium), a real person's photo → the\nconsent-gated real-person route after a `[REAL_FACE_DETECTED]` rejection, with `realFaceConsent: true`\nset **only** after the user explicitly confirms they hold the rights to the likeness. The full rules,\nincluding why the real-vs-AI call is the provider's and not yours, are in\n`slates-prompting-seedance`.\n<!-- /slates-only -->\n\nAlso unchanged, and worth restating because 2.5's length makes each one more expensive to get wrong:\n\n- **One primary camera move per shot.**\n- **No stacked lens / aperture / film-stock vocabulary.** One camera-and-lens style line is the\n most the 2.5 guide itself uses; the full rule is in `slates-prompting-seedance` → \"Don't\n cross-pollinate image-model syntax\".\n- **No `negativePrompt` field** — constraints go inline, and 2.5 acts on negative phrasing in\n exactly two dimensions: subtitles (*\"no subtitles\"*) and audio (*\"no BGM; environmental and\n action sounds only\"*, *\"no audio\"*). Everywhere else, describe what you want, not what you don't.\n- **Legible in-shot text still belongs in a baked start frame**, not in the video prompt.\n",
|
|
31
|
+
"slates-prompting-seedance": "---\nname: slates-prompting-seedance\ndescription: \"Prompt Seedance 2.0 (seedance-2), and retrieve the shared Seedance craft used by 2.5. Covers subject binding, shot-number storyboards, camera moves, physical action and inline constraints.\"\n---\n\n# Seedance 2.0 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Seedance 2.0.** Not copywriting — an ENGINEERING instruction to a spatial layer and a temporal layer: who, in what scene, doing what, how the camera moves, and in what order. Multi-beat work is a `Shot 1 / Shot 2 / Shot 3` storyboard.\n\n**The five levers**\n1. **Bind every subject to its reference** — `<Subject_1>@<Image_1>` — and keep the descriptions identical across shots. Unbound subjects are where twins come from.\n2. **Shot sizes and camera MOVES, not lens data** — `medium close-up`, `slow push in`, `handheld follow`, `whip pan`. One primary move per shot.\n3. **Externalise emotion as physical action.** Not \"she is nervous\": `she turns the ring on her finger twice, then stops`.\n4. **Use the image-quality slot vocabulary** for quality — `HD`, `rich details`, `cinematic texture`, `natural colors`, `soft lighting`. That is the officially sanctioned way to ask.\n5. **Constraints go INLINE**, led by the official templates — `keep it subtitle-free`, `do not generate a watermark`, `avoid jitter and bent limbs`, `avoid temporal flicker`.\n\n**Examples**\n- `Shot 1: medium shot, <Subject_1>@<Image_1> steps out of the freight lift into a wet loading bay, slow push in. Shot 2: close-up, she turns the ring on her finger twice and stops, handheld. Rich details, cinematic texture, natural colors. Keep it subtitle-free.`\n- `Single continuous take. Wide shot of a fishing skiff crossing a grey swell, camera tracks from the starboard rail. Spray hits the lens once. Soft lighting, natural colors, film-grain texture. Avoid jitter and bent limbs.`\n\n**Hard constraint:** NO timestamps — 2.0 ignores them and answers only to shot numbers (2.5 acts on them). No lens, aperture, film stock or camera body: that is image-model vocabulary and a Seedance anti-pattern. There is no negativePrompt field.\n<!-- @card:end -->\n\nByteDance's video model — first-party via **BytePlus ModelArk** (credits only, no BYOK). Audio always generated alongside the video. Single model `seedance-2` across the full resolution ladder (480p / 720p / 1080p / native 4K — 4K video is Pro-only, default 1080p), 4–15s, first+last frame, and up to 9 reference images / 3 videos / 3 audio clips.\n\n> **How to read this file.**\n> **[official :NNNN]** — ByteDance's own BytePlus ModelArk prompting guide, line `NNNN` of the archived doc dump (`research/byteplus-seedance-2-0-api-docs.md`). Receipt-grade; treat as law.\n> **[community]** — third-party guides and our own field notes. Useful, but an `[official]` block always wins.\n> **[slates]** — how the Slates app composes or bills this; not ByteDance doctrine.\n>\n> The split is load-bearing. A community-sourced \"narrative timing beats\" doctrine shipped in this file for months teaching the **exact inverse** of ByteDance's published guidance. Never merge the two registers again.\n\n---\n\n# Part 1 — Official ByteDance doctrine\n\n## What Seedance actually is `[official :1450-1452]`\n\nSeedance 2.0 is a multimodal AI director. It reads text, images, video and audio **simultaneously** and internally decomposes them into two dimensions:\n\n- the **spatial layer** — what is in the frame\n- the **temporal layer** — how things change over time\n\nSo a good prompt is **not \"copywriting-style description\" but an \"engineering-style instruction\"**: who, in what scene, doing what action, how the camera moves, and in what chronological order events occur — delivered respectively to the spatial layer and the temporal layer.\n\n## The advanced formula — 8 slots `[official :1455]`\n\n```text\nprecise subject + action details + scene/environment + lighting & color tone\n+ camera movement + visual style + image quality + constraints\n```\n\n⚠️ There is **no official \"6-step formula.\"** `Subject + Action + Environment + Camera + Style + Constraints` is community branding with no ByteDance source, and it silently drops the **lighting & color tone** and **image quality** slots. Use the 8 slots above.\n\n## Task-type sentence patterns `[official :1389-1425]`\n\nSeedance classifies your request from the phrasing. Use the pattern that matches the task:\n\n| Task | Pattern |\n|---|---|\n| **Image reference** | ``Reference `<Subject_N>` in `<Image_N>` to generate…`` |\n| **Video reference** | ``Reference `<Action / Camera_movement / Style / Sound_effect>` in `<Video_N>` to generate…`` |\n| **Audio reference** | ``Reference the timbre in `<Audio_N>` to generate…`` |\n| **Video edit — modify** | ``Strictly edit `<Video_N>`, and modify `<Original_Characteristic>` in it to `<New_Characteristic>``` |\n| **Video edit — add** | ``<Element_Features>` + `<Timing>` + `<Location>`` |\n| **Video edit — delete** | Name what to delete; for anything that must stay, say so explicitly |\n| **Video extend** | ``Extend `<Video_N>` forward/backward to generate…`` |\n| **Combined** | ``Reference `[Dimension]` of `<Image/Video_N>`, strictly edit `<Video_X>`, `[Specific_Edits]``` |\n\n### ⚠️ Edit / extend phrasing landmine `[official :1431]`\n\n> *\"For edit / extend video tasks, directly use `<Video_N>` to refer to the video. **Do not use \"reference `<Video_N>`\"**, to avoid being incorrectly identified as a reference task.\"*\n\nThis is easy to trip: Slates has an edit lane<!-- slates-only --> (`slates_generate_video` with `videoReferenceAssetId`, plus the Seedance edit/relocate routes)<!-- /slates-only -->. Writing *\"reference video 1 and change the jacket to red\"* gets classified as a **reference** task — the model generates a brand-new clip inspired by the source instead of editing it. Write *\"Strictly edit video 1, and modify the blue jacket to red.\"*\n\n## Shot structure — \"Shot 1 / Shot 2 / Shot 3\" `[official :1563-1598]`\n\n> *\"Use shot order, write a simple 'Shot 1 / Shot 2 / Shot 3' storyboard for each segment of the video, and then merge them into a complete prompt.\"*\n\n**❌ Never second-stamp.** No `0:00–0:03`, no \"At 4 seconds\", no per-segment durations.\n\n> *\"Do not impose strict limits on the duration of each segment; prioritize allowing the model to naturally generate the pacing based on the plot.\"* `[:1580]`\n>\n> *\"The model's support for precise timing (such as 0–3 seconds) is **unstable**, and forcibly limiting duration may lead to **abnormal generation results**.\"* `[:1586]`\n\nOrder shots by when events occur — primary first, secondary later. Let the plot set the pacing.\n\n**Per-shot internal order** `[official :1590-1598]` — organize each shot in exactly this sequence:\n\n1. **Camera movement or shot transition** — \"slowly push in from a wide shot\", \"fixed camera position\", \"cut to…\"\n2. **Subject actions and expressions** — the key actions and expression changes of the core character/object\n3. **Position or spatial change** — where the subject is, and how that relationship shifts\n4. **Audio** — sound effects, voices, background music for that shot\n\n**One primary camera move per shot** — see Camera below. `[official :1648]`\n\n## Subject binding — names + image indexes `[official :1488-1556]`\n\nEvery time a subject appears, it must be **explicitly referred to**. Two supported forms:\n\n- **Undefined subjects** — bind inline every mention: `<Subject_N>@<Image_N>`. Official example: **`Zhang San@Image 1`**. `[:1540]`\n- **Pre-declared subjects** — define once, then reuse the same label verbatim: *\"Define the tall man in **Video 1** as **police officer**, and define the other short man as **thief**\"*, then say \"police officer\" every time after. `[:1514]`\n\n**One subject spread across several assets** — bind them together: *\"Define `[…]` in **Image 1** and `[…]` in **Image 2** as `<Subject N>`.\"* `[:1514]`\n\n⚠️ **An Asset ID must never substitute for `<Image/Video_N>`.** `[:1546]` *\"the model cannot directly associate the Asset ID with the reference content.\"* Always cite by index.\n\nAlso official: keep descriptions concise, avoid redundancy, avoid semantic conflicts (contradictory traits for one subject), and prefer expressing spatial relationships through reference images rather than dense text. `[:1550-1556]`\n\n**`[slates]`** — the app composes this for you. `composeReferences()` cites each canonical character or environment reference inline as `Name (image N)` in the exact order it sends them, which is ByteDance's own duplicate-character format (*\"Zhang San (corresponding to image 1)\"* `[:1976]`). You never hand-write role labels or index numbers.\n\n## Action description `[official :1602-1621]`\n\n- **Body-part specificity + quantified degree.** Name hands, legs, head, shoulders, back — and supplement **range, speed, and force**. *\"slowly raise a hand\", \"quickly turn the head\", \"push hard off the ground\", \"slightly lower the head.\"*\n- **Prioritize slow, gentle, continuous small movements.** Avoid high-burst, large-dynamic actions — sprinting, big jumps, violent rolls. *\"walk slowly\", \"gently raise a hand\", \"sit down naturally with the motion.\"* **This is the official basis for the folk rule that \"fast\" degrades quality** — it is not a banned token, it is a class of motion the model handles badly.\n- **Supplement transitions between actions.** Specify inertia and continuity between consecutive beats so movement reads coherent: *\"use the inertia of turning around to naturally raise a hand\", \"naturally transition from a pause into raising a hand.\"*\n\n## Externalize emotion `[official :1623-1636]`\n\nReplace abstract emotion words (\"very sad\", \"extremely angry\") with **specific physical detail**. This is the highest-leverage single habit in the official guide:\n\n| Abstract | Externalized as actions and details |\n|---|---|\n| **Sadness** | head lowering, shoulders trembling slightly, eyes reddening, fingers unconsciously clutching the corner of clothing, tears welling but not falling |\n| **Joy** | corners of the mouth rising uncontrollably, brows and eyes relaxing, steps becoming light, unconsciously humming a tune |\n| **Nervousness / anxiety** | frequently checking the watch, fingers constantly tapping the tabletop, rapid breathing, eyes darting away |\n| **Anger** | both fists clenched, jawline tense, chest heaving, eyes sharp, squeezing words out through gritted teeth |\n| **Relief** | letting out a long breath, tense shoulders completely relaxing, a faint smile appearing, looking up toward the distance |\n\n## Camera `[official :1643-1648]`\n\n> *\"The model has a **strong understanding of camera movement terms**, so you can **directly use standard camera movement terminology**, such as 'medium shot, close-up, wide shot, slow push-in, smooth lateral tracking, fixed shot.'\"*\n\nThis is an **open vocabulary, not a fixed list** — and it explicitly includes **shot size** (close-up / medium / wide / long shot), which is as much a camera instruction as the move itself.\n\n> ⚠️ *\"Try to specify only 1 type of camera movement in a single shot. Do not require push, pull, pan, and move at the same time, as this will increase image instability.\"* `[:1648]`\n\n## Image quality, style, and constraints `[official :1656-1679]`\n\nThese three slots \"define creative boundaries for the model, unify image quality and artistic tone, and avoid visual flaws and random deviations.\"\n\n**1. Image quality** — define clarity, texture detail, and lighting quality. Official vocabulary: `HD` · `rich details` · `cinematic texture` · `natural colors` · `soft lighting`.\n\n> ⚠️ This is a **real slot with real vocabulary** — do not confuse it with Stable-Diffusion-era quality incantations. `8K` / `masterpiece` / `trending on artstation` remain banned slop tokens (see Part 3); *\"cinematic texture, rich details, natural colors\"* is the officially sanctioned way to ask for the same thing.\n\n**2. Style** — the overall art style and visual tone: `cyberpunk cool blue-purple tone` · `retro film` · `fresh Japanese style`.\n\n**3. Constraint words** — *\"Constraint words are very important. They can effectively avoid visual flaws, deformities, breakdowns, and unreasonable elements.\"* Official templates, verbatim:\n\n- **No subtitles** — \"keep it subtitle-free\" / \"avoid generating any text or subtitles\"\n- **No logo** — \"do not generate a logo\"\n- **No watermark** — \"do not generate a watermark\"\n\nSeedance has **no `negativePrompt` field** — constraints go inline in this slot. See Part 3 for the wider inline-negative kit.\n\n## 🔴 Duplicated characters — the twin problem `[official :1948-1994]`\n\n**Symptom:** in frames with **many characters**, where **three-view / multi-view character images** are supplied as references, two identical characters appear in the same generated frame.\n\n**Root causes** `[:1954-1959]`:\n1. Character subjects are not clearly defined in the prompt, so the model cannot distinguish roles.\n2. *\"When character **three-view / multi-view images** are used as reference assets, it is easy to confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"*\n\n**Official fixes, in their order** `[:1971-1994]` — ByteDance is explicit that these *reduce probability*, not eliminate it:\n\n1. **Bind each character to its image explicitly**, in a consistent format. Official example: *\"Zhang San (corresponding to image 1) throws the green passbook toward Li Si (corresponding to image 2), who is standing.\"*\n2. **Append the global constraint verbatim** at the end of the prompt `[:1982]`:\n > *\"Throughout the video, characters with completely identical appearance, clothing, and accessories are prohibited. Do not generate duplicate avatars or a twin effect. Keep only a single corresponding character in the same frame, and do not reproduce repeated copies of characters.\"*\n3. **Optimize reference assets** `[:1988]` — *\"For character reference images, prioritize independent single-person photos. Three-view or multi-view assets are not recommended.\"*\n4. **Simplify the prompt** — do not paste a whole script; redundant copy confuses the model.\n\n**Scope this honestly.** This is troubleshooting for the twin problem in **multi-character frames**, not a blanket verdict on identity sheets. Practical rule for Slates:\n\n- **Multi-character Seedance shot** → bind every character to its image, append the anti-twin constraint, and prefer single-person / dominant-portrait references over multi-view sheets.\n- **Single-character shot** → the standard character-sheet flow is fine.\n\n**Too many reference people** `[official :2048-2052]` — past **4 reference people**, output stability drops (wrong headcount, duplicates). Official workaround: group the cast into images of ≤4 people each, generate those stills first, then drive the video from them.\n\n## Worked examples `[official :1689-1745]`\n\nThese are ByteDance's own end-to-end cases. Note the shape: an asset-binding preamble, then `Shot N` blocks in event order, then a trailing style + stability paragraph. No time stamps anywhere — 2.0 does not respond to them at all, which is version-scoped and reverses on 2.5.\n\n**Example 1 — dormitory emotional short drama (dialogue-focused).** Assets: `@Image 1` half-body photo of the female lead · `@Image 2` dormitory scene reference · `@Video 1` camera-movement reference · `@Audio 1` indoor ambience.\n\n> Use the girl in @Image 1 as the main character, use @Image 2 as the dormitory scene style reference, and refer to the camera movement in @Video 1.\n>\n> **Shot 1**: At dusk, **girl @Image 1** walks briskly to the **dormitory entrance @Image 2**. The camera follows steadily in a medium shot. Warm yellow sunlight spills into the hallway from the window. She pauses at the doorway, takes a deep breath, and looks slightly nervous.\n>\n> **Shot 2**: **Girl @Image 1** pushes the door open and enters the dormitory. The camera cuts to an indoor medium shot. Her roommates look up at her while organizing their books. One of them smiles and asks {How did the exam go? Did you pass?}. The camera slowly cuts between half-body close-ups of several people.\n>\n> **Shot 3**: **Girl @Image 1** first lowers her head with a dejected expression. The camera gives her a close-up. Then she raises her head, unable to hold back a smile, laughs out loud, and says {I was kidding}. Her roommates start chasing and play-fighting with her. The camera slowly pulls back and freezes on a wide shot of the dormitory filled with laughter.\n>\n> The entire video should have a high-definition cinematic documentary style, with warm tones and soft lighting. The character's face remains stable without deformation; movements are natural and smooth, with no stutter or flicker. The ambient sound blends naturally with @Audio 1.\n\n**Example 2 — ancient-style cliff confrontation (action/atmosphere-focused).** Assets: `@Image 1` female lead in red · `@Image 2` assassin in black · `@Image 3` cliff and bamboo forest · `@Video 1` martial-arts camera reference · `@Audio 1` drum beats.\n\n> Use the woman in red from @Image 1 as the female lead, use the woman in black from @Image 2 as the opponent, use the cliff and bamboo forest environment in @Image 3 as the scene reference, refer to the overall camera movement and action rhythm in @Video 1, and synchronize the background sound effects with @Audio 1.\n>\n> **Shot 1**: At dusk, the camera slowly pushes in from a side medium shot of **woman in red @Image 1**. She stands at the edge of the cliff and lifts a wine flask to drink. Her sleeves and robe hem sway gently in the mountain wind. The camera circles halfway around her, moving from the front to her back. In the distance, a figure in black is faintly visible in the bamboo forest.\n>\n> **Shot 2**: The camera zooms and fades into a long shot. From a drone perspective, it overlooks the entire cliff and bamboo forest. The two characters stand at opposite ends of the cliff. The mountain wind lifts their robe hems and dust, and the rhythm slightly accelerates with the drum beats.\n>\n> **Shot 3**: The camera cuts back to a ground-level close shot. The two slowly draw their swords and face off. **Woman in red @Image 1** shifts from a careless expression to a cold gaze. **Woman in black @Image 2** looks determined, and the sword tip trembles slightly. The camera steadily follows the two as they circle each other, finally freezing on a close-up of the instant before the two swords meet.\n>\n> The overall visual style should feel like a cinematic wuxia world in misty rain, with cool tones, low saturation, a film-grain texture, and rich light-and-shadow layers. The characters' faces and body proportions remain stable without deformation. Movements are continuous and natural, not stiff, with no clipping or stutter.\n\n## Other official notes\n\n- **On-screen text** `[official :1758]` — Seedance can render common text (ad slogans, subtitles, speech bubbles) and will auto-match style/colour from context, or take an explicit colour / style / timing / position. Prefer **common characters**; avoid rare glyphs and special symbols. (For *guaranteed* legible text, the start-frame route in Part 3 is still safer.)\n- **Extension degrades quality** `[official :2004-2024]` — using a generated video as the input for extension compounds degradation, with mottled colour blocks in face regions. Limit repeated continuations; prefer HD assets as input.\n- **Special effects that miss** `[official :2031-2044]` — when a described effect comes out wrong (a countdown that scrolls randomly), define it with a **reference video** instead of words: *\"the way the number '2999' appears should reference video 1.\"*\n\n---\n\n# Part 2 — Slates-specific `[slates]`\n\n## Reference media — caps and transport\n\nReference-to-video accepts up to **9 reference images, 3 reference videos, 3 audio clips** `[official :275-281]`. Text+audio-only and audio-only inputs are not supported.\n\n**Mutually exclusive:** first-frame/last-frame mode CANNOT be combined with reference images. The error reads `\"first/last frame content cannot be mixed with reference media content.\"` Pick one or the other. *(Official note `[:284]`: you can approximate first/last frames via prompt wording inside a multimodal call, but if the frames must be exact, use the dedicated first/last-frame route.)* The same rule covers reference VIDEO and AUDIO: they ride the reference endpoint, which has no frame parameters at all.\n\n### All three modalities go in ONE call\n\nThe caps are a shared budget, not three separate features: **12 files total on 2.0** (9 image + 3 video + 3 audio), **15 seconds of reference video combined**, **15 seconds of audio combined**. On 2.0 an audio reference needs at least one image or video alongside it; 2.5 accepts audio on its own.\n\nCite each by type and index, in the order they were attached — `image 1`, `video 1`, `audio 1`. The index is positional: reorder the attachments and the numbers move with them.\n\n```\nMarcus (image 1) performs the motion from video 1, in the workshop from image 2,\nusing the voice timbre from audio 1. Preserve his identity, appearance and outfit.\n```\n\n🚨 **SAY WHAT AN AUDIO REFERENCE IS FOR.** It can mean music, dialogue, voice, tone or timbre — five roles on one attachment — so an unroled clip falls back to **dialogue**: the model re-transcribes it and speaks ITS words. A real take came back as *\"a map called Slates\"* for *\"an app called Slates\"*. Name it as the voice timbre and the clip carries the voice while the prompt carries the words. ByteDance's own sentence: *\"Image 1 depicts the protagonist John and uses the voice timbre from Audio 1.\"* Bind each speaker in a sentence, never by attachment order — position carries nothing.\n<!-- slates-only -->\n**Attaching a clip is NOT the same as editing it.** \"Add as reference\" puts it in the composer alongside everything else and wipes nothing; \"Edit with AI\" makes the clip the canvas and clears the tray for a fresh instruction. Two different jobs, two different menu entries — never infer one from the other.\n\n**Over the cap is REFUSED, never trimmed.** A reference video is priced into the quote before it is sent, so a clip silently dropped after the quote would be a clip you paid for and the model never saw. Remove one and retry.\n<!-- /slates-only -->\n\n### Motion transfer & lip-sync recipes (reference video / audio)\n\nThese aren't separate Seedance features; they're prompting strategies over reference media.<!-- slates-only --> Run `slates_generate_video` with the clip as a video reference and write the motion or dialogue into the prompt:<!-- /slates-only -->\n\n- **Motion transfer:** subject image as a reference + the driving clip<!-- slates-only --> via `videoReferenceAssetId`<!-- /slates-only --> (2–15s) + `The character from image 1 performs the exact motion, choreography, and camera movement from video 1. Preserve the character's identity, appearance, and outfit.`\n- **Lip-sync / dialogue:** write the line in the prompt — `The person in video 1 says: \"…\"` — with audio generation on (always on in Slates). A **video** source's own voice is cloned natively; an **audio** reference (≤15s) drives speech from an existing recording: `…speaks the dialogue from audio 1 with accurate lip sync.`\n- **Voice + face from one clip (the talking-head recipe):** ONE unedited 2–15s clip of the person speaking (clear voice, no music, no cuts) as the video reference + prompt with the new script → their likeness AND voice deliver the new line.\n<!-- slates-only -->\n- **Billing:** a reference VIDEO switches the cost key to `seedance-2*-vref-{res}-{T}s` where T = combined clip seconds + output seconds on faceless and real-face routes. The AI-face route bills max(combined input, output) + output on both 2.0 and 2.5; quote before confirming. Audio references are free (audio is included on every route).\n<!-- /slates-only -->\n\n<!-- slates-only -->\n## Faces — set `seedanceFace` for AI-character faces\n\nSeedance routes through **three tiers** depending on the face in the reference, exposed as the \"Face in Reference\" toggle plus the real-face params on `slates_generate_video`:\n\n- **Faceless / object / environment refs → default route (cheapest).** Leave `seedanceFace` off.\n- **An AI-character's FACE in a reference → `seedanceFace: true`.** The default route's baseline moderation rejects or degrades faces, so this reroutes to the face-capable provider. It costs **~45% more** — the cost key becomes `seedance-2-face-{res}-{N}s`, so the pre-flight quote already reflects it. Announce the face-route price, not the faceless one.\n- **A REAL person's photo (the user themselves, an actor) → the consent-gated real-person route.** If a `seedanceFace` gen fails with `[REAL_FACE_DETECTED]`, the provider classified the reference as a real person: confirm with the user that (a) they hold the rights/consent to the likeness and (b) they accept the higher price (cost key `seedance-2-realface-{res}-{N}s`, about 1.4× the AI-face rate, about 2× faceless; quote via `slates_estimate_generation_cost`), then retry with `seedanceRealFace: true` + `realFaceConsent: true`. Never set `realFaceConsent` without the user's explicit confirmation.\n\nRules:\n- **The real-vs-AI call is the PROVIDER'S, not yours.** ByteDance's classifier is probabilistic — some real photos pass the standard face route (billed at the cheap rate; fine), others get rejected with `[REAL_FACE_DETECTED]` (auto-refunded). Don't preemptively route to the real-face tier just because a photo looks real; try `seedanceFace: true` first and escalate only on the marked rejection. Public figures / celebrities fail on every route.\n- It's about the **reference, not the output.** If your character identity or generated portrait shows a face, turn it on. A product shot with no person stays off.\n- Don't toggle it on \"just in case\" — a faceless gen on the face route burns ~45% extra for nothing.\n<!-- /slates-only -->\n\n## Reference rules (the verified ones)\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n<!-- @inject:reference-rules-core -->\nIdentity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.\n<!-- @end:reference-rules-core -->\n\n### For Seedance specifically\n\n- **Describe the ACTION, never the reference's content.** With refs attached, prompt only what is *happening* — motion, change, camera. Never re-describe what's in the reference, and never say \"still / scene / from a movie / from the image.\" The model already sees the refs; narrating them wastes tokens and induces drift. Injection is stochastic — if a roll misses, **re-roll, don't re-engineer** (and a slow gen is not a failed one<!-- slates-only --> — see slates-cost-discipline<!-- /slates-only -->).\n- **Seedance's own idiom for rule 2 is `Reference <Subject_N> in <Image_N>`** `[official :1389]` — `Image_N` indexes the order the refs are attached, so the name plus the index carries the role. The full binding grammar is in Part 1 (Subject binding).\n- **Rule 3 has an official ceiling here.** The trend is MORE references (video and audio into Seedance), all addressed by name — but for **multi-character frames** see the twin-problem section above: bind every character to its image, append the anti-twin constraint, and prefer single-person references. Past 4 reference people, stability drops `[official :2048-2052]`.\n- **Rule 8 holds even though Seedance can render common text natively** `[official :1758]`. A baked NB2 start frame is still the reliable route for text that must be legible.\n- **Rule 5 pairs with the first/last-frame exclusion** — frames and reference images are mutually exclusive on this model (see Reference media above), so an environment you must match exactly costs you the frame lane.\n\n<!-- slates-only -->\n## Pre-flight: references arrive inline, refer by code\n\nWhen you call `slates_generate_video` with reference asset IDs (firstFrameAssetId, lastFrameAssetId, ingredientAssetIds), the first call returns those references **inline as image content blocks** alongside a cost estimate and `requires_confirm: true`. **Look at the references** — if they suggest a different framing, lighting, or motion than your current prompt captures, revise the prompt before re-calling with `confirm=true`.\n\nWhen talking to the user about the gen, refer to each reference by its short code: `IMG-A12 — Beach Sunset`. The user sees that code as a badge on the gallery thumbnail, so they can match what you're saying to what they're looking at.\n\n- ✅ \"I'm using **IMG-A12** as the first frame and **IMG-A15** as the last frame — the camera move is going to be a slow dolly forward through the gap.\"\n- ❌ \"I'm using the first beach image and the last one...\" (which? They have four.)\n<!-- /slates-only -->\n\n---\n\n# Part 3 — Community field notes `[community]`\n\nThird-party guides and Slates field experience. Useful heuristics — but if one of these ever appears to contradict Part 1, **Part 1 wins**.\n\n## Length\n\n**Sweet spot 60-150 words** for a single shot (not 150-300 — that's the upper bound). Multi-shot storyboards run longer; official Example 1 above is ~230 words across three shots.\n\n## Pin the subject in the first 20-30 words\n\nThe opening sentence is the **identity anchor**. If the subject isn't locked early, the model hallucinates new subjects mid-clip. (Compatible with Part 1: the binding preamble comes before `Shot 1`.)\n\n```\nA matte black earbud case sits on a polished obsidian surface...\n```\n\n## Lighting is a top quality lever\n\nLighting has an outsized impact on output quality — which is why it has its own slot in the official 8-slot formula. Describe it before or alongside the subject.\n\n```\nA cool-white diagonal beam from upper left, dust particles drifting through.\nSoft golden hour lighting from low west angle.\nDramatic rim light against dark background.\n```\n\n## Camera and subject motion — separate sentences\n\nMixing them is a common cause of glitchy / shaky output.\n\n❌ \"The camera speed ramps as the earbud rises.\"\n✅ \"The earbud rises smoothly. The camera tracks upward.\"\n\n## Slow-motion works; \"fast\" is a known bad token\n\nSpeed ramps and slow-motion are supported in natural language, and `fast` is widely reported as a quality-degrading keyword. **The official version of this rule is stronger and better founded** — prioritize slow, gentle, continuous small movements and avoid high-burst action (Part 1, Action description `[:1611-1615]`). Prompt the motion class, not the adjective.\n\n```\nthe lid opens in slow-motion · the blade whips through the air\n```\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ — same contract as the anti-list in slates-prompting-nano-banana-2.\n Extracted by src/prompts/banned-tokens.ts into the slates_generate_video op\n description and matched against submitted prompts. The RECOMMENDED vocabulary\n below sits OUTSIDE the markers on purpose — it is backticked too. -->\n<!-- /slates-only -->\n**Slop tokens to avoid:** `epic`, `amazing`, `beautiful`, `lots of movement`, `8K`, `masterpiece`, `trending on artstation`.\n<!-- @banned:end -->\nThese are quality *incantations* — the officially sanctioned way to ask for quality is the image-quality slot vocabulary in Part 1 (`HD`, `rich details`, `cinematic texture`, `natural colors`, `soft lighting`).\n\n## Style block at the end\n\nOne primary anchor + 2-3 supporting details, as the trailing paragraph (both official examples do exactly this). End with `Single continuous take` if you want one shot with no cuts. **Never** write `no cut` or `seamless transition` — not in the training vocabulary.\n\n## ⚠️ Don't cross-pollinate image-model syntax\n\n<!-- @inject:lens-video-split -->\nNamed lenses, apertures, film stocks and camera bodies (`85mm f/1.4`, `Kodak Portra 400`, `ARRI Alexa 65`) are an image-model lever. On a video model, translate the look instead of pasting the gear list: `85mm f/1.4, Portra 400` becomes `close-up, shallow depth of field, warm natural colors, cinematic texture, film-grain texture`. ByteDance's Seedance 2.0 guide never mentions fps, shutter angle, f-stop or lens millimetres. Its Seedance 2.5 guide does, once: the visual-style line of its own storyboard example names one camera body and one 35 mm cinema lens. On 2.5 a single line like that is vendor-sanctioned; a stacked gear list still is not.\n<!-- @end:lens-video-split -->\n\n## Negative prompting — inline only\n\nSeedance has **no `negativePrompt` field**. Put negatives in the constraints slot, led by the three official templates (Part 1):\n\n```\nkeep it subtitle-free · do not generate a logo · do not generate a watermark\navoid jitter and bent limbs\navoid temporal flicker\navoid identity drift\nno distortion, no stretching\n```\n\nAlso fine: positive reframing (\"empty street\" not \"no cars\").\n\n## Image-to-video / first-frame guidance\n\n**Describe motion, not image.** The model already sees the visual; tokens spent re-describing appearance are wasted.\n\nStability phrases that help:\n- `preserve composition and colors`\n- `maintain exact appearance from reference image`\n- `consistent character throughout, no deformation or drift`\n\n**Cap I2V prompts under 60 words** when possible. Over 100 words frequently triggers silent generation failure.\n\n## Common failure modes + fixes\n\n| Failure | Fix |\n|---|---|\n| Hallucinated subject mid-clip | First 20-30 words = identity anchor |\n| Bent limbs / extra fingers | `avoid jitter and bent limbs` in Constraints |\n| Identity drift across multi-shot | Re-name the bound subject in **every** `Shot N` block `[official :1537]` |\n| Two identical characters in one frame | The twin fix in Part 1 — bind each character to its image + append the global anti-twin constraint |\n| Silent generation failure on I2V | Cut prompt under 100 words, single primary camera move |\n| Speech / motion conflict | Limit dialogue to one line per action shot |\n| Erratic/random pacing | You second-stamped. Remove all time markers and use `Shot N` `[official :1586]` |\n\n## Sources\n\n**Official (authoritative):**\n- BytePlus ModelArk — Seedance 2.0 prompting guide, archived at `research/byteplus-seedance-2-0-api-docs.md` (all `:NNNN` refs above)\n\n**Community (secondary):**\n- [fal.ai — How to Use Seedance 2.0](https://fal.ai/learn/tools/how-to-use-seedance-2-0)\n- [apiyi.com — Seedance 2.0 Prompt Guide](https://help.apiyi.com/en/seedance-2-0-prompt-guide-video-generation-camera-style-tips-en.html)\n- [atlabs.ai — Ultimate Seedance 2.0 Prompting Guide](https://www.atlabs.ai/blog/the-ultimate-seedance-2.0-prompting-guide-47-prompts-2026)\n",
|
|
32
|
+
"slates-prompting-seedream-5-lite": "---\nname: slates-prompting-seedream-5-lite\ndescription: \"Prompt or edit images with Seedream 5 Lite (seedream-5-lite). Use with slates_generate_image or slates_edit_image on this model; covers attention order, focused descriptions, layout and quoted text.\"\n---\n\n# Seedream 5 Lite — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Seedream 5 Lite.** The cheap volume seat: flat-priced at every resolution, which makes it useful for storyboard passes, variant grids and look-dev when volume is the requirement. Structure, most important first: `Subject + Style + Composition + Lighting/Atmosphere + Technical`.\n\n**The five levers**\n1. **Lead with the subject.** Earlier words weigh more; close with the camera and technical detail.\n2. **Keep it 30-100 words.** Unlike models that reward verbosity, this one gets confused by very long prompts. Focused beats exhaustive.\n3. **Name the composition** — `symmetrical composition`, `rule of thirds`, `foreground detail with blurred background`, `overhead perspective`.\n4. **Name the light as a named condition** — `golden hour`, `dramatic side lighting`, `soft diffused light`, `moody low-key`, `bright high-key`.\n5. **Quote in-image text.** It takes quoted strings for posters and layouts, which is half of why it is the drafting seat.\n\n<!-- @inject:cinematic-card -->\n**For a photographic look, use only what this frame needs.** Image models default to clean, evenly lit and fully exposed. Describe what the camera sees, not just gear or mood:\n- **Inspect every reference first.** Write its grade and imperfections in words: darkness, contrast, muddy or true blacks, colour, softness/noise, subject separation. Never grade cleaner or brighter than the look reference unless asked.\n- **One light system** — `low sun behind her`, `her face falls into deep shadow`, `no light in front of her`.\n- **Visible exposure** — `the sky burns out to white`, `dense, slightly crushed shadows`.\n- **Lens name plus effect** — `200mm telephoto`, `peaks loom huge behind her and melt into soft shapes`.\n- **Name every garment and close the foreground.** Omissions invite reference leakage or invented props.\nBind references inline. A scene reference owns the grade; for a look-only reference, write the new scene's light. References are optional. For owned-frame edits, describe only the change and what stays.\n<!-- slates-only -->Use `slates-cinematic-look` with a technique ID or section query for more.<!-- /slates-only -->\n<!-- @end:cinematic-card -->\n\n**Selection:** it is a volume option. Re-render a keeper on another image model only when an observed shortfall or the delivery brief justifies it; use the current catalogue for that choice.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use:**\n- `masterpiece`, `best quality`, `highly detailed`, `8k`, `award-winning` — quality incantations, not description\n- a prompt past about 100 words: this model gets confused by very long prompts, and focused beats exhaustive\n<!-- @banned:end -->\n\n**Examples**\n- `Professional headshot of a female CEO, short blonde hair, confident expression, navy suit, neutral office background. Studio lighting, shallow depth of field, high-end corporate photography, shot on 85mm.`\n- `A rain-soaked night market stall, cinematic, rule of thirds with the vendor camera-right, foreground steam blurred, moody low-key lighting with practical neon, shot on 35mm.`\n\nByteDance's Seedream image model, Lite tier, routed via fal.ai. In Slates: `slates_generate_image` with `model: seedream-5-lite` (REQUIRES projectId; no headless path). **Flat-priced regardless of resolution**, the cheapest flat-priced seat in Slates, which makes it a useful choice for high-volume drafting, storyboard exploration, and variant grids. Call `slates_estimate_generation_cost` for the current number; never quote prices from memory. Less censored than Nano Banana 2.\n\n**When to pick it:** lots of frames cheap (storyboard passes, 3-4 variant exploration), posters/layouts with text, quick look-dev. Step up to NB2 or FLUX.2 Max for the locked hero shot.\n\n## Core structure — five components, most important first\n\n```\nSubject + Style + Composition + Lighting/Atmosphere + Technical parameters\n```\n\nSeedream weights concepts mentioned **earlier in the prompt** more heavily. Lead with the subject; close with camera/technical details.\n\n**Length sweet spot: 30-100 words.** Unlike models that reward verbosity, Seedream gets confused by very long prompts. Focused beats exhaustive.\n\n## Style, composition, lighting vocabulary it responds to\n\n- **Style:** portrait photography, macro photography, cinematic, photorealistic, minimalist, oil painting, watercolor, digital art\n- **Composition:** symmetrical composition, rule of thirds, foreground detail with blurred background, wide-angle view, overhead perspective, medium shot, close-up\n- **Lighting:** golden hour lighting, dramatic side lighting, soft diffused light, moody low-key lighting, bright high-key lighting\n- **Technical:** shot on 85mm lens, shallow depth of field, high resolution\n\n## Worked examples\n\n**Portrait:**\n> \"Professional headshot of a female CEO with short blonde hair, confident expression, wearing a navy blue suit, neutral office background, studio lighting, shallow depth of field, high-end corporate photography style\"\n\n**Product:**\n> \"Modern smartphone floating in space, dark background with subtle blue gradient, product photography, studio lighting highlighting the glossy screen, ultra-detailed, commercial quality, photorealistic rendering\"\n\n## In-image text: double-quote it\n\nPut the exact string in double quotation marks — Seedream treats quoted text as render-this-verbatim:\n\n```\nA minimalist poster with the headline \"SUMMER SALE\" in bold sans-serif, centered\n```\n\nSeedream is one of the stronger models for layout-heavy work (posters, mockups, diagrams): call out the layout explicitly — \"centered headline, subtitle beneath, clean margins.\"\n\n## Edits: change one thing, lock the rest\n\nVia `slates_edit_image` with `editModel: seedream-5-lite`. Seedream edits respond well to instructions that name the change AND the preserved elements:\n\n```\nChange the bag to brown leather. Keep the person's face, pose, and the room unchanged.\n```\n\nSeedream edits accept extra `referenceAssetIds` within the current edit-reference cap, with the source occupying the first image slot. Read the tool schema for the cap. An older desktop without that capability refuses the request: update the desktop rather than claiming the extra references were sent.\n\n## Common failure modes + fixes\n\n| Failure | Fix |\n|---|---|\n| Subject inconsistent / mutates | Put the subject description first; break complex subjects into clear components |\n| Style drift | Reinforce the aesthetic with 2-3 related terms (\"cinematic, photorealistic, shallow depth of field\") |\n| Compositional confusion | Use photography terms (\"medium shot,\" \"overhead view\"); simplify the scene |\n| Garbled text | Double-quote the exact string; keep it short; state placement |\n| Mushy long-prompt output | Cut to under 100 words — Seedream rewards focus, not volume |\n\n## Iterate cheap, lock expensive\n\nFlat pricing supports iteration: draft, evaluate inline, diagnose one specific delta, then generate only within the authorized set. Repeated failures trigger diagnosis rather than unchanged re-rolls; only re-render the winning composition on another model when the delivery requires it. Cost rules live in `slates-cost-discipline` — the batch-authorization pattern applies when generating variant grids.\n\n## Sources\n\n- [fal.ai — Seedream Prompt Guide](https://fal.ai/learn/devs/seedream-v4-5-prompt-guide)\n- [BytePlus ModelArk — Seedream Prompt Guide](https://docs.byteplus.com/en/docs/ModelArk/1829186)\n",
|
|
33
|
+
"slates-restyle-from-blocking": "---\nname: slates-restyle-from-blocking\ndescription: \"Generate different visual treatments from one blocking pass while preserving its camera, cuts and choreography. Use when comparing looks or restyling an approved structure without rebuilding it.\"\n---\n\n# Restyle — one edit, many worlds\n\nThe commercial payoff of the whole previs workflow, and the reason a blocking file is an asset rather than a step.\n\n## The idea\n\nEvery prompt has two halves:\n\n- **Structure** — cuts, camera, timing, who is where. Lives in the blocking clip. **Never changes.**\n- **Style** — what any of it looks like. Lives in the references and the prompt text. **Changes freely.**\n\nHold the structure, swap the style, and the same edit comes back as live action, painted 2.5D, ink on paper or a toybox — **matching frame for frame across all of them.** Cuts land on the same frames, the car drifts at the same moment, the same head turns at the same beat.\n\nFor anyone pitching work: three visual worlds in a day, off one edit the client has already approved. The foundation is not up for renegotiation, so the conversation is only about look.\n\n## Before you restyle\n\nYou need a blocking clip whose structure you are happy with, and a finished prompt for at least one style (per `slates-blocking-to-prompt`). The first style is the expensive one; every later style is an edit of its text.\n\n## What stays fixed\n\nCopy these across every style **verbatim**. Changing them is what desynchronises the outputs:\n\n- The blocking reference's own contract — that it is the master for all movement, the placement-only clause, the tie-break clause, the disambiguation clause\n- The shot count and measured cut boundaries; translate prompt timestamps once to the chosen model's syntax and keep that translation across treatments\n- Every shot's camera position, angle, framing and cut point\n- Screen direction and seating\n- The `HOLD FOR THE FULL TIMELINE` block\n- `videoReferenceAssetIds` and `videoReferenceSecondsEach`\n\nLead each style's prompt with a lock so the style layer cannot leak into the structure:\n\n> VIDEO LOCK — the dominant rule of this prompt: the reference defines 100% of the motion, editing and object choreography. The text below defines only look, materials, locations and effects layered onto that motion. Wherever the text and the video could be read differently about motion, the video decides.\n\n## What changes\n\n| Layer | What you swap |\n|---|---|\n| Rendering style | photoreal · painted 2.5D · 2D ink · miniature/toybox |\n| Characters | different sheets entirely — a couple, grandparents, a robot and a cat |\n| Locations | the same four beats set in a different world |\n| Time of day / weather | night after rain · golden hour · hard noon |\n| Lighting and colour | per style |\n| Audio | SFX-only, or scored, or lip-synced dialogue |\n\nCharacters can change species and still land, because the blocking only supplies where a body is and how it moves.\n\n## Dummy mapping — the mechanism that makes it work\n\nEach style needs its own explicit mapping from grey proxy to real object. The proxy is a slot; the style fills it:\n\n> DUMMY MAPPING: the front-LEFT sphere-head dummy (with its grey arm at the shifter and grey leg at the pedals) is THE GRANDPA; the front-RIGHT sphere-head dummy is THE GRANDMA; a front-seat dummy together with its loose blocks is that ONE whole person. Blocks on the rear bench are the luggage. The low-poly flying model in SHOT 18 is THE HELICOPTER. The two vehicles behind the hero car in SHOT 19 are THE POLICE CARS.\n\nSame clause per style, different right-hand side. And restate the placement-only rule in style terms:\n\n> The source defines only placement and motion, never appearance: every placeholder becomes the real object its position implies — spheres are always people, cabin blocks are always cases and bags, fully drawn.\n\n## Location continuity\n\nIf the piece travels, name the places and pin each shot to one. Reusing labels across styles keeps the four prompts diffable:\n\n```\nLOCATION CONTINUITY — one journey through four fixed places; each looks\nidentical in every shot where it appears:\n LOC-A <opening> LOC-B <middle> LOC-C <turn> LOC-D <finale>\n```\n\nThen tag every beat: `SHOT 9 — 7.79-9.33s — LOCKED, LOC-B: <description>`.\n\n**On a piece that visits many places, make the map absolute and countable** — otherwise the model reuses a room it liked and you get the same interior three times:\n\n> The location map is absolute — SEVEN locations, each appearing EXACTLY ONCE, in this exact order: 1) yard 00:00-00:03.3 … 7) rooftop 00:20-00:30. No location ever appears twice, and the three interiors are three COMPLETELY DIFFERENT rooms — different walls, furniture, people and light — never the same room repeated.\n\n## Style references\n\nA style reference is **not a keyframe**, and saying so prevents the model reproducing its composition as a shot:\n\n> STYLE MASTER — defines the painting and rendering style only: hand-painted look with visible brushstrokes, sculpted painterly volumes, textured matte surfaces, dramatic coloured rim light, deep moody shadows. NOT a keyframe, NOT a location to reproduce, NOT a frame that ever appears in the film. Its own subject, framing and composition are never seen in any shot.\n\nA style can also be **text-only** — no reference image at all. Ink and toybox looks usually specify better in words than they match from a still.\n\n## Keep performance inside the existing shots\n\nStyle changes tempt the model to earn new coverage. Refuse it:\n\n> ACTING — inside the existing shots only: performance is visible only at the size and distance the reference already gives it, only where the source already shows a face; everywhere else it reads through posture and hands alone. The performance NEVER earns a new shot, a new angle or a closer framing.\n\n## Text-free worlds\n\nStylised worlds are where invented signage and garbled lettering appear. One clause kills it:\n\n> TEXT-FREE WORLD: every sign is a blank painted shape, every gauge face carries tick marks only, every licence plate is a blank plate.\n\n## Running it\n\nGenerate each style as its own `slates_generate_video` call against the **same** `videoReferenceAssetIds`. Keep them in one project so they sit side by side; name assets by style so the comparison reads at a glance.\n\nQuote the whole set before firing — `slates_estimate_generation_cost` per style — and confirm. Four styles is four generations, not one.\n\n🚨 Never fire a batch of style variants without showing the user the prompts and the total cost first.\n\n## Checklist per style\n\n- [ ] Same blocking asset, same `videoReferenceSecondsEach`\n- [ ] VIDEO LOCK leads the prompt\n- [ ] Every timestamp and shot count identical to style 1\n- [ ] Dummy mapping written for this style's cast\n- [ ] Style reference declared as style-only, or none used\n- [ ] Location labels reused; on a travelling piece the map is absolute and countable\n- [ ] Acting-inside-existing-shots clause present\n- [ ] HOLD block copied verbatim\n- [ ] Cost quoted and confirmed\n\n## Related\n\n`slates-previs-blocking` · `slates-blocking-to-prompt` · `slates-style-prompting` (style vocabulary per model) · `slates-cost-discipline` (batch quoting)\n",
|
|
34
|
+
"slates-script-craft": "---\nname: slates-script-craft\ndescription: \"Write or revise script passages, openings, bridges and saved variations while preserving voice, format and fixed material. Use for writing within a production brief or a writing-only request.\"\n---\n\n# Script craft and variations\n\nWork in the user's document. Read its revision, requested passage and neighboring context. Keep supplied facts, deliberate cadence and fixed sections intact. A script may be silent, a conversation, one continuous sentence, independent scenes, or any mixture. These are tools to choose from, not required stages.\n\n<!-- @evidence: script-craft-20260922 sc-event sc-proof sc-exchange sc-callback sc-modular sc-bridge sc-offer sc-flow -->\n\n## Opening, argument and payoff\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `sc-event` | Observed creative pattern; conversion unmeasured | Start with an event or consequence, including sound or silence. | Useful when the product can participate; skip spectacle unrelated to its promise. | `Keys slide toward the table edge; the tray catches them.` |\n| `sc-proof` | Observed demonstration pattern | Show the specific claim being tested. Speech may direct attention to the visible evidence. | Useful for observable behavior; skip claims the demonstration cannot establish. | `Watch the rim.` |\n| `sc-exchange` | Observed multi-speaker pattern | Let another speaker question, react or misunderstand. Preserve the answering context. | Useful for objections and comedy; do not isolate a dependent answer. | `A: You bought a tray for that? B: Look where my keys used to land.` |\n| `sc-callback` | Observed repeated-character comedy | Repeat deliberately, escalate, then resolve or change the meaning. | Useful for recognition and payoff; skip repetition without a purpose. | `The same searching hand finally reaches straight for the tray.` |\n| `sc-modular` | Scoped house-format technique | Make selected passages self-contained so they can move independently. | Useful for reorderable demonstrations; do not flatten continuing dialogue. | `At the door, it catches the keys. On the desk, it holds the loose change.` |\n| `sc-bridge` | Variation craft synthesis | Vary an opening together with any transition it requires. | Check pronouns, promise, reveal order and offer; preserve the chosen body. | `Where do your keys land? Mine used to land wherever my hand stopped. Now they land here.` |\n| `sc-offer` | Claim-control synthesis | Make the next action understandable and supported by the brief. | Use supplied destinations and terms; never invent price, savings, scarcity or guarantees. | `See the available finishes.` |\n| `sc-flow` | Spoken-writing synthesis | Clarify subject, action and causal connection before removing stylistic patterns. | Keep intentional rhythm and jokes; skip mechanical fragmenting. | `Put your keys here when you come in.` |\n\n## Distinct openings and compatible bridges\n\nChange the idea: an event, question, objection, proof, audience situation or reveal. Merely swapping adjectives is not a useful comparison. Name what stays fixed for this operation. A dependency belongs in the selected passage: if an opening changes what “that” means, include its bridge in the version.\n\nRead each candidate as a complete piece with the same body. Check unanswered promises, introduced speakers, incompatible offers and repeated reveals. Suggestions remain editable; no required Hook/Body/CTA fields.\n\nUse `slates_get_script_document`, `slates_get_script_sections` and revision-checked `slates_update_script_document` / `slates_update_script_section`. Save versions before switching. Preview one requested combination before materializing it; never expand every possible combination automatically. Reference substitutions are explicit IDs, not name replacements in prose. Keep voice retention deliberate.\n\n## Spoken flow and pacing\n\nPrefer a concrete actor doing something over abstract benefit language. “Seamlessly elevate your daily carry” becomes “Put your keys here when you come in.” Connect causes where needed: “I put them down, then forget where” is clearer than mechanically shortening it to “Keys. Gone. Again.”\n\nPreserve the user's or reference's cadence when it carries character, comedy or comprehension. A repeated sentence or triplet is not inherently an error. Personal voice preferences apply only to the person who supplied them. Never invent testimonials or measurable results.\n\nRead the canonical fit analysis supplied with the shots. Its corpus estimate uses total ad runtime, including silence, and an opt-in register sample. It is not measured articulation speed; the observed maximum is not a universal human limit. Plan for pauses, reactions and sound. Once a voice/video take exists, its measured performance governs the cut. Model clip duration, estimated script duration and actual speech duration are separate facts.\n\n## Apply the requested scope\n\nFor suggestions, propose each replacement with `slates_update_script_suggestions` (action `create`), quoting the exact words it replaces at the revision you read; the creator accepts or dismisses it in the document, and `slates_get_script_suggestions` reports what became of it. For an explicit edit request, apply the scoped edit and read it back; do not add an approval ceremony. On a stale revision, reread and preserve both authors' changes. Do not replace the whole script to change one opening.\n\nHeadings and directions are non-spoken metadata. Shots are optional production bindings. Script-driven recipes compile the active words; custom prompts retain their bytes and need a visible alignment review. Existing takes remain historical media. Writing, switching versions and importing templates do not generate anything. Load production and cost guidance only when production is requested.\n",
|
|
35
|
+
"slates-shot-variety": "---\nname: slates-shot-variety\ndescription: \"Shape visual rhythm across a shot sequence or diagnose unintended sameness. Use while planning or reviewing cuts; preserve deliberate repetition, continuing performance and the chosen format.\"\n---\n\n# Visual rhythm across a sequence\n\n`slates_list_shots` supplies distributions and repeated runs from authored shot fields. Read across the sequence, then decide whether repetition serves the intended effect. A dominant bucket is a question, not a defect or generation barrier.\n\n## Compare neighboring cuts\n\nLook at framing, camera behavior, duration, subject distance, location and cast. Change the dimension that carries the meaning of the next beat. A wider view may reveal geography; a close view may make a small action legible. Do not add camera motion simply because another shot is static.\n\nRepeated frames can establish a joke, a comparison or a calm observational register. Recurring people and locations can carry a conversation. A later change often works because the earlier pattern held. State that purpose briefly when a count flags an intentional choice.\n\n## Re-cut only for a reason\n\nMerge when performance and picture should continue together. Split when the image needs to change while speech continues, or when the intended read needs another placement. Preserve sentence continuity and references across the split. Price the resulting requests; do not assume splitting is free or merging is cheaper.\n\nThe script fit signal derives from an opt-in ad corpus measured over whole runtime. Above-sample pace deserves inspection; it does not prove a line impossible. Measure the actual spoken take when available, including pauses and reactions. No fixed cut length or shot count is a universal rule.\n\nMulti-shot generations can contain several visual cuts. Compare their internal rhythm as well as the boundaries between generated clips. The app counts only authored information: an unknown framing bucket is missing description, not evidence about the pixels.\n\nThis guide improves deliberate visual decisions. It does not measure taste, conversion or the quality of a finished performance. Use `slates-script-craft` for the argument, exchanges and setup/payoff that the picture supports.\n",
|
|
36
|
+
"slates-storyboard-from-script": "---\nname: slates-storyboard-from-script\ndescription: \"Save a supplied script or treatment as an editable Slates document and bind production passages to shots. Use when preparing a storyboard while preserving the words, structure and requested scope.\"\n---\n\n# Script into editable production\n\nRead the existing document and its revision before writing. Preserve supplied words, speaker context, headings, non-spoken direction and any explicit shot list. A heading formats a document; creating a production scene is a separate choice. Paragraph count does not determine shot count.\n\n## Save the words once\n\nUse `slates_get_script_document` and `slates_update_script_document` for ordered, revision-checked text and structure edits. Scene strings own spoken words. Paragraph blocks hold offsets and marks; headings/directions own only their non-spoken text. Do not keep an independently editable master body beside the document.\n\nCreate a board or scene only when needed for the requested destination. Use the current project unless the user asks for another. Writing a script needs no image, character record or generation.\n\n## Bind production where wanted\n\nSelect an intended production passage and use the script-to-shot operation. It can make, attach, extend, split or merge according to the existing bindings. Read back the resulting shots and ranges. A silent shot is equally valid and needs no fabricated dialogue.\n\nA new document-created recipe is script-driven: its prompt compiles from the active passage. Text with no speaker and no delivery goes to the model as written (most script text is action); a speaker, VO included, or a delivery note makes it quoted speech. Keep action, delivery, framing and references in their own controls. Do not write the dialogue a second time in a custom prompt. When the creator explicitly chooses a custom prompt, preserve its bytes and review alignment after script changes.\n\nShots file through the existing filing service. Pass the scene or frame destination when known and use the returned shot codes. Keep recurring identities in existing Library references; a working speaker name does not require a placeholder character.\n\n## Review without imposing a format\n\nRead composed requests, actual reference roles and the current quote. Explain only consequential decisions not already visible in the document or shot. Variety counts are suggestions: intentional repeated frames, continuing sentences and recurring cast may be exactly right. `slates-script-craft` covers passages and versions; `slates-shot-variety` covers deliberate visual rhythm.\n\nIf generation is requested, follow `slates-cost-discipline` for the exact set. On an uncertain timeout inspect existing generation IDs before retrying. Preserve takes and inspect the landed results. Named cuts keep independent edits separate; writing alone does not require a cut, export or paid call.\n\n<!-- @inject:decision-log -->\nRecord production choices in the editable shot fields. Explain only consequential judgments the user did not specify and no field already records: for example, why a particular light or performance register supports the brief. Do not repeat the shot list in prose or turn this explanation into an approval gate. Follow the separate generation authorization policy before spending.\n<!-- @end:decision-log -->\n",
|
|
37
|
+
"slates-style-prompting": "---\nname: slates-style-prompting\ndescription: \"Translate a visual-style brief into model-specific image or video direction and maintain the look across shots. Covers photoreal, anime, painterly and 3D styles, references and optional start-frame control.\"\n---\n\n# Per-style prompting (photoreal · anime · painterly · 3d-render)\n\nThe style library (`slates_create_style` / the app's style ids) defines what each style IS. This guide is how to PROMPT each style per model. Derived from `research/style-prompting-research.md` (second-brain) — claims marked *(hypothesis)* are untested; don't present them to users as fact.\n\n## The four ground rules (all styles)\n\n1. **Assign references where they contribute.** Describe the scene with inline bindings, such as \"the woman from image 1, lit and graded like image 2.\" Preserve an existing scene reference when its look should stay. A look-only reference may need light and exposure described for the new scene; prose and references can work together.\n2. **Use each model’s language, without imposing a fixed prompt template:**\n - **Nano Banana 2** — narrative prose; the style is the opening framing of the sentence (\"A hand-drawn 2D anime cel illustration of…\"), never a comma tag.\n - **Seedance 2.0** — the 8-part formula reserves \"visual style\" (slot 6) and \"image quality\" (slot 7). One clause each. Don't scatter style words through the action text.\n - **Kling V3** — prose scene direction; style rides the lighting/style tail of Scene → Subject → Action → Camera → Lighting/Style. Tag soup underperforms badly.\n3. **Keep the intended look consistent across shots.** Reuse relevant references and stable descriptions, adapting the wording to each scene.\n4. **A styled start frame is one video control.** Generate it with the image seat suited to the brief, then describe the motion. Preserve its look unless the user wants the light or grade to change.\n\nNever stack style buzzwords (\"ARRI ALEXA, 35mm, film grain, depth-of-field mastery…\"). One or two register tokens maximum — piles of specs dull the image.\n\n## Photoreal\n\n- **NB2:** never the literal word \"photorealistic\". Describe *a real photograph*: natural skin texture and imperfection, motivated lighting, one lens/film register (\"shot on a 50mm, soft window light\"). Photographic composition terms: wide-angle / macro / low-angle.\n- **Seedance:** put \"sharp focus, natural color, high detail\" in the image-quality slot and always include a lighting clause. Keep motion slow and coherent — fast/burst action is the #1 quality killer and reads most fake in photoreal.\n- **Kling:** the photoreal-PEOPLE lane — convincing acting, dialogue, lip-sync. It breaks on close-up hands, fine fluids, and crowds beyond ~5 faces: route those beats to Seedance or reframe.\n- **Faces on Seedance:** photoreal humans trigger the face-tier routing (AI face vs consented real face — see slates-prompting-seedance §Faces). Set the face flags honestly; never skip them to save credits.\n\n## Anime\n\n- **NB2:** open with the medium — \"A hand-drawn 2D anime cel illustration of…\" — then normal narrative Subject/Setting/Action. Clean line art, flat-shaded color, expressive eyes. NB2 has no negative prompt: phrase exclusions positively (\"flat cel shading with uniform focus\", not \"no depth of field\").\n- **Seedance:** visual-style slot = \"2D anime style, clean line art, flat cel shading\". The slow/coherent-motion preference still applies — burst sakuga actions are the same instability trap as in photoreal.\n- **Kling:** weakest anime lane (its strength is live-action-like acting); expect style drift on long prose-only shots. Prefer ground rule 4: NB2 anime start-frame → i2v with a motion-only prompt. *(hypothesis: refs hold Kling's anime better than prose — verify before promising.)*\n- Anime faces drift under multiple references faster than photoreal; the named-entity one-sheet doctrine applies unchanged.\n\n## Painterly\n\n- **NB2:** medium + technique in the style framing: \"digital concept-art painting, visible brushwork, painted edges\". At most ONE school/era register (\"classic gouache illustration\") — a register, not an artist-name pile.\n- **Video:** the least-supported style lane. Use ground rule 4 (painterly NB2 frame → i2v, motion-only prompt) and expect some cleanup of painterliness over the clip *(hypothesis — set user expectations, don't promise a perfectly painterly clip)*.\n- Camera language still applies — painterly ≠ static; \"slow push-in\" works the same.\n\n## 3D render\n\n- **NB2:** name the lineage register in the style framing: \"stylized 3D render, soft global illumination, subsurface skin\". Lighting vocabulary (GI, rim light) is unusually load-bearing for the 3D read.\n- **Seedance:** the physics/effects lane flatters 3D content — visual-style slot \"stylized 3D animation\", image-quality slot \"clean render, high detail\".\n- **Kling:** same start-frame preference as anime.\n- *(hypothesis)* An engine token (\"Unreal Engine 5 render\") may help NB2; if used, ONE token, style slot only — never on Seedance where spec-stuffing hurts.\n\n## Routing recipe (what to actually do)\n\n1. Style reference available → inspect it, bind its look role, and describe any light or exposure needed in the new scene.\n2. Image request → use the current image default unless the brief supplies a reason for another seat; load that model's craft. The NB2 examples above apply when NB2 is selected, not to every image model.\n3. Video request → choose a styled start frame when composition, exact text or an approved look must hold. Use the image seat suited to that job, then the chosen video model's motion and sound grammar. Direct styled text-to-video is also valid when it serves the brief; an image pass is not mandatory.\n4. Multi-shot run → retain stable style references and descriptors, adapting action, light and model-specific wording to each scene.\n\n`slates-model-selection` and the current catalogue own routing. These style techniques supply craft after the production choice; no user needs to select a skill or workflow.\n",
|
|
38
|
+
"slates-ugc-influencer-ad": "---\nname: slates-ugc-influencer-ad\ndescription: \"Direct a creator-style spoken ad when the brief calls for a camera-facing person or exchange. Covers activity, performance, phone-camera handling, speech, interaction and sound.\"\n---\n\n# Creator-style performance\n\nRead `slates-script-craft` for the script and opening/bridge versions. Match the requested creator, audience and reference register. Phone footage, quiet polish and cinematic treatment are choices; no evidence here establishes one as universally highest-converting.\n\n## Direct a person in a place\n\nWrite an observable activity and a speaking intention: showing a worn handle, answering a friend, demonstrating a catch, interrupting a task. Small actions can make a close shot legible: a weight shift, a glance toward the other person, a hand taking an object's weight. Larger actions need space and framing that accommodates them.\n\nFor an ordinary phone register, describe concrete light and surroundings: a window from one side, a practical lamp, a room with ordinary possessions. Avoid generic praise words. Preserve the supplied person's appearance and product details through references, not invented claims.\n\nDescribe camera handling physically when it matters: a hand supports the phone against the table edge; the frame sags and is corrected once. Static framing is also valid. Do not force handheld motion or imperfections into a brief that asks for something else.\n\n## Speech, interaction and sound\n\nOne person or several may carry the piece. Give reactions and answers their antecedents. A continuing sentence may cross an edit; an independent module should establish its own subject. Repetition can be a callback.\n\nSpeech may point at what the viewer sees when demonstrating a claim. It need not compete with the picture for novelty on every line. Keep pauses, listening and the intended register; a fixed words-per-second formula cannot establish the performance's duration.\n\nMake audio intent explicit: dialogue, room sound, effects, score or silence. A silent demo is valid. Use current model guidance for audio support and reference syntax. Plan captions or text only where the actual editing surface supports them; do not promise an unverified rendering feature.\n\n## References and production choices\n\nSelect the identity, voice, location and direct images deliberately. A replacement presenter need not replace a separately retained voice. A changed location does not automatically replace a first-frame image. Inspect the composed request to see which references remain.\n\nA plate-first approach is useful when the composition must be approved before motion, but it is not a prerequisite for all video. Choose single-take or multiple-cut production according to the intended performance and current capabilities. Do not copy vendor duration, resolution or reference limits into this guide.\n\nBefore spending, follow `slates-cost-discipline` for the selected set. Inspect errors and existing job state; a refusal or timeout does not authorize another charge. Check the actual clip's picture and sound against the brief, preserve earlier takes, and trim or rearrange when that solves the issue without regenerating. Exported playback, not a successful process exit, establishes the result.\n",
|
|
39
|
+
"slates-vision-feedback-loop": "---\r\nname: slates-vision-feedback-loop\r\ndescription: \"Inspect generated media against the brief, diagnose defects and choose a targeted correction. Use during production or iteration; covers reference review, image defects and model-specific failure receipts.\"\r\n---\r\n\r\n# Vision feedback loop — Slates utility skill\r\n\r\nSlates returns generated images inline as base64. You see the actual pixels. Use that — don't trust prompt-following blindly.\r\n\r\n## Asset codes are your shared vocabulary with the user\r\n\r\nEvery asset in Slates has a short stable code (e.g. `IMG-A12`, `VID-V3`, `AUD-S1`) and a label derived from its prompt (e.g. `Beach Sunset`). These are visible in the gallery as a corner badge on each thumbnail. **Always refer to assets by their code in chat** so the user can match what you're saying to a specific card in their gallery.\r\n\r\n- ✅ \"I'm using **IMG-A12 — Beach Sunset** as the first frame. The second-frame candidate **IMG-A15** has the right composition but warmer light — want me to use that one instead?\"\r\n- ❌ \"I'm using the beach sunset image...\" (user has four beach sunset variants — which one?)\r\n- ❌ \"I'm using asset `7a3f9e4b-...`\" (UUIDs aren't readable; user can't match to a badge)\r\n\r\nThe code is the FORMAL reference. The label is human texture. Use both: `IMG-A12 — Beach Sunset`.\r\n\r\n## Vision tools at your disposal\r\n\r\n- `slates_get_asset_image` — pull one image into context. Returns its code+label.\r\n- `slates_get_assets_batch` — pull up to 8 images in one call. Use when picking from a candidate set; cheaper than N individual fetches.\r\n- `slates_get_asset_video_frames` — extract N keyframes (default 3) from a video and inline them as JPEGs. These sampled stills support appearance, framing and identity checks. They do not verify continuous motion, lip sync or sound. Use actual playback or an audio-capable host for those claims when available; otherwise report them unreviewed and retain the saved asset.\r\n\r\n## Pre-flight is automatic on the gen tools\r\n\r\n`slates_generate_video` and `slates_generate_image` show you their reference assets **inline** on the confirm response. You don't need to fetch them yourself, but you DO need to look at what comes back, revise the prompt if the references suggest a different motion/framing, and only then re-call with `confirm=true`.\n\r\n## 🔴 The still-gate — never animate a bad frame\r\n\r\n<!-- @inject:still-gate -->\r\n**Inspect a start frame before animating it.** Repair a visible defect that would make the intended crop or performance unusable before spending on motion. A clean frame can be animated whenever the brief calls for movement; this check does not require an image stage for text-to-video.\r\n\r\nThis is a cost rule as well as craft: a premium video call can cost many times an image correction. Broken geometry can turn to mush, oily textures can crawl and malformed objects can fall apart in motion. Fix a known source defect at the source instead of buying a more expensive copy. Judge intentional stylisation against the brief, not a universal photoreal standard. Additional image or video requests still follow the existing generation authorization.\r\n<!-- @end:still-gate -->\r\n\r\n## The pattern\r\n\r\n1. **Generate.** Call `slates_generate_image` with a prompt. The result is in your context as an image content block.\r\n2. **Evaluate on TWO axes — they are different questions:**\r\n - **Brief-conformance** — what did the user actually want? Are the elements right? Composition? Lighting? Subject identity?\r\n - **Defects** — run the slop rubric below. *A frame can match the brief perfectly and still be slop that mushes the moment it moves.* Checking only the first axis is how a bad frame reaches an expensive video call.\r\n3. **One of three outcomes:**\r\n - **Right** → save it (bind to a frame, character slot, etc.) and move on.\r\n - **Close, but adjustable** → refine with a specific delta, regenerate **once**.\r\n - **Wrong direction** → diagnose the failed requirement. Refine within the supplied brief and existing authorization; ask only when the creative intent is unresolved or the next request needs fresh consent.\r\n\r\n## The defect rubric — five slop tells\r\n\r\n| Tell | What it looks like | Why it matters downstream |\r\n|---|---|---|\r\n| **Light with no transitions** | Flat-black pits instead of a shadow ramp; light that stops rather than falls off | Transfers onto every character or object added into that plate later |\r\n| **Broken-but-plausible objects** | Crates, railings, hardware, mechanisms you can *almost* read but that don't resolve | Turn to mush in motion, and the model multiplies them |\r\n| **Local logic breaks** | An effect present in only part of the frame — rain scratching one corner, wet ground under one figure | The video model's physical logic breaks along with it |\r\n| **Oily textures** | Soapy, licked-smooth surfaces that have lost their material identity | Reflections crawl in motion; the plate can't hold continuity |\r\n| **Too perfect** | A soft light on the face that nothing in the scene could cast, the subject sharper and cleaner than everything around them, every region exposed to be readable, colour pushed warm and saturated | It reads as a subject pasted onto a location, and every shot built from the plate inherits the studio look. Fix it in words: `slates-cinematic-look` |\r\n\r\n### Per-model accents — check the one you actually used\r\n\r\n- **Nano Banana Pro** (`nano-banana-pro`) — ruler-straight symmetry, everything parallel and square, flat even light, pretty but staged/stock, textures reading as 3D render rather than photograph. **It hyperbolizes every edit**: ask for graffiti on one wall and the whole location gets tagged.\r\n- **GPT Image** (`gpt-image-2-5-flare`, `gpt-image-2-5-sunburst`) — microcontrast to the ceiling, hard halos on every edge, no depth or bokeh, white balance pulled warm until the frame yellows, plastic licked-smooth materials. Worst tell: **one sickly texture pattern laid over the entire frame**. ⚠️ Catalogued on `gpt-image-2`, which 2.5 replaced on 2026-09-09 — an accent is a per-model observation, so treat this as a prior to check rather than a finding, and correct it here the first time a 2.5 frame disagrees.\r\n\r\n> ⚠️ These are accents for **`nano-banana-pro`** and the **GPT Image** line specifically. `nano-banana-2` is a **different model** (Gemini 3.1 Flash Image vs NB Pro's Gemini 3 Pro Image) and we have **no evidence** about its accent. Do not inherit one — say nothing rather than warn about a failure mode you can't substantiate. That caution applies to the GPT Image entry above too: it was measured on `gpt-image-2`, not on either 2.5 seat.\r\n\r\n## Where the fault lives — triage before you change anything\r\n\r\nWe say \"one specific delta per regeneration\" but that only helps once you know *which* variable to move. Diagnose first:\r\n\r\n| Visible pattern | Diagnosis | Fix |\r\n|---|---|---|\r\n| The defect exists in the source asset, or stays tied to the same feature when the direction changes | **Source asset** | Fix the sheet / plate, not the prompt |\r\n| Source is clean, and the defect changes when only the suspect motion clause changes | **Motion direction** | Fix the prompt |\r\n| Controls conflict, or the failure follows neither variable | **Inconclusive** | Narrow the test — change less, not more |\r\n\r\n**Review routes; it is not pass/fail.** Geography melts → fix the location. Identity drifts → fix the character sheet. Assets are sound but the action is wrong → fix the video direction. Wrong idea entirely → reopen the brief with the user.\r\n\r\n**Correct the earliest broken handoff.** Polishing a downstream symptom hides the source and guarantees it resurfaces in the next shot built from the same asset.\r\n\r\n## Baseline hygiene — isolate the variable you're testing\r\n\r\nWhen the **character** is the question, keep the location out of it: test on a plate that already holds its own geometry, depth, materials, and light. **A broken plate gives every character failure a second plausible cause**, and you will spend re-rolls deciding which one you're looking at. The same applies in reverse — test a plate empty before you populate it.\r\n\r\n## Refinement rules\r\n\r\n- **One specific delta per regeneration.** Don't change five things at once — you won't know what helped.\r\n- **A fresh generation needs a complete coherent prompt.** Change one decision and retain the unchanged requirements so old and new clauses do not conflict. An edit request uses its model's change-only grammar instead; do not turn a surgical edit into a full scene re-description.\r\n - On **Seedance**, retain the selected model's structure: shot numbers for 2.0, whole-second timing where used for 2.5. See the matching model guide.\r\n - **Exception — Omni Flash Edit.** Long prompts documentedly destroy its fidelity. There the rule inverts: one short instruction plus *\"Keep everything else the same.\"*\r\n- **Anchor with references.** If the result drifted from the user's intent, attach the *previous best* generation as a reference image alongside the original brief.\r\n- **Use `slates_get_asset_image`** to pull a previously-generated image back into context if you need to compare against a fresh generation.\r\n- **Use `slates_edit_image`** for surgical tweaks instead of full regeneration when ~90% of the image is right — `sourceAssetId` = the asset, `prompt` = the change only. Edits preserve composition and identity; full regen rolls the dice. Recipe: `slates-edit-and-iterate`.\r\n\r\n## Cost discipline\r\n\r\n- Track total credits spent across the loop. Surface to the user every 3 iterations.\r\n- Use the repeated-failure checkpoint below; do not keep submitting an unchanged failed request.\r\n- **Follow the existing consent for every attempt.** An approved enumerated batch covers its listed calls. A retry or changed input outside that batch needs a new quote and the applicable confirmation; a timeout requires a status check before another submission.\r\n\r\n<!-- @inject:iteration-diagnosis -->\r\n## Diagnose repeated failures\r\n\r\nAfter three failed attempts at the same requirement, pause unchanged re-rolls and diagnose the source reference, prompt structure, model fit and tool result. Three is a review checkpoint, not a universal limit or proof that the seed cannot matter. Preserve the attempts and name what each test changed.\r\n\r\nContinue autonomously when the brief is clear, a specific correction is supported and the next request is already authorized. Hand control back when taste or intent cannot be inferred, the next request needs fresh consent, or the available tool cannot meet the requirement. A failed roll never authorizes an additional charge. Follow the existing batch and per-request cost policy.\r\n<!-- @end:iteration-diagnosis -->\r\n\r\n## When to break the loop\r\n\r\n- The user said \"good enough\" or \"ship it.\" Stop iterating.\r\n- Repeated attempts show no progress. Diagnose the source, prompt or model before spending again; apply the scoped retry rule above.\r\n- The user changes brief mid-loop. Treat it as a new brief, not a continuation.\r\n\r\n## Voice when narrating to the user\r\n\r\nTight, observational, no editorializing.\r\n- ✅ \"Frame 2 has the wrong lighting direction — back-lit instead of side. Regenerating with side light.\"\r\n- ❌ \"I notice that the lighting in frame 2 isn't quite what we were going for. I'll go ahead and try again with a different approach.\"\r\n",
|
|
41
40
|
};
|
|
42
41
|
//# sourceMappingURL=content.js.map
|