@slatesvideo/shared 0.7.1 → 0.7.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +1 -1
- package/dist/manual/content.d.ts +1 -1
- package/dist/manual/content.js +1 -1
- package/dist/manual/index.d.ts +11 -2
- package/dist/manual/index.js +178 -14
- package/dist/operations/index.d.ts +286 -87
- package/dist/operations/index.js +715 -69
- package/dist/operations/surface.js +6 -0
- package/dist/prompts/agent-doctrine.js +1 -1
- package/dist/prompts/generation-policy.d.ts +1 -1
- package/dist/skills/content.js +2 -2
- package/package.json +1 -1
- package/skills/_partials/cinematic-card.md +1 -1
- package/skills/slates-prompting-inworld-tts.md +174 -174
- package/skills/slates-style-prompting.md +54 -54
|
@@ -145,6 +145,9 @@ export const OPERATION_GROUPS = {
|
|
|
145
145
|
'slates_relocate_project',
|
|
146
146
|
'slates_undo_relocate_project',
|
|
147
147
|
'slates_reveal_file',
|
|
148
|
+
'slates_reorder_folders',
|
|
149
|
+
'slates_reorder_pins',
|
|
150
|
+
'slates_link_asset_source',
|
|
148
151
|
],
|
|
149
152
|
// The cut and the export. A generation session never touches these.
|
|
150
153
|
script: ['slates_get_script_document', 'slates_update_script_document', 'slates_get_script_sections', 'slates_update_script_section', 'slates_get_script_suggestions', 'slates_update_script_suggestions', 'slates_get_script_uses', 'slates_preview_script_variation', 'slates_create_script_variation', 'slates_get_shot_inputs', 'slates_reuse_shot_take'],
|
|
@@ -182,6 +185,9 @@ export const OPERATION_GROUPS = {
|
|
|
182
185
|
'slates_duplicate_shot',
|
|
183
186
|
'slates_split_shot',
|
|
184
187
|
'slates_merge_shots',
|
|
188
|
+
'slates_get_usage',
|
|
189
|
+
'slates_get_app_settings',
|
|
190
|
+
'slates_set_app_settings',
|
|
185
191
|
],
|
|
186
192
|
// A third transport nobody without Blender installed can reach.
|
|
187
193
|
blender: [
|
|
@@ -96,7 +96,7 @@ const PREAMBLE = fork(`You are the Slates Studio Agent — a production assistan
|
|
|
96
96
|
- Speak in the app's words and name assets by code and label (IMG-A12), never by tool name or UUID.`);
|
|
97
97
|
// ── The working method ─────────────────────────────────────────────
|
|
98
98
|
export const WORKING_METHOD = [
|
|
99
|
-
both(`For HOW/WHERE questions, load slates_get_prompting_guide with topic "app-manual" and relevant query keywords. Teach the documented buttons and tabs, preserving model-specific conditions; do not invent UI paths or mutate the project when the user only asks for instructions. Slates is a sandbox of optional tools, not a required pipeline.`),
|
|
99
|
+
both(`For HOW/WHERE questions, load slates_get_prompting_guide with topic "app-manual" and relevant query keywords. Teach the documented buttons and tabs, preserving model-specific conditions; do not invent UI paths or mutate the project when the user only asks for instructions. When the control is on screen, offer to point at it (slates_highlight_control); show a picture of a screen (slates_get_manual_picture) only when the user cannot find something or asks what it looks like. Slates is a sandbox of optional tools, not a required pipeline.`),
|
|
100
100
|
both(`1. UNDERSTAND the outcome the user wants. If intent is clear, act with sane defaults — don't interrogate. If genuinely ambiguous, batch every question into ONE message.`),
|
|
101
101
|
both(`2. ORIENT: call slates_get_workspace_state once at the start of a workflow. Work in the user's CURRENT project — this chat lives inside it. NEVER create a new project unless explicitly asked; if there's no current project, ask which to use.`),
|
|
102
102
|
// FORKED: the desktop sends core tools plus what `slates_load_tools` loaded,
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
* This is NOT a model capability or provider batch limit. Desktop mirror is
|
|
3
3
|
* generated by slate/scripts/sync-generation-policy.mjs; never edit it there. */
|
|
4
4
|
export declare const IMAGE_QUANTITIES: readonly [1, 2, 3, 4, 5, 6, 7, 8, 9, 10];
|
|
5
|
-
export declare const MAX_IMAGE_VARIATIONS: 1 | 2 |
|
|
5
|
+
export declare const MAX_IMAGE_VARIATIONS: 1 | 2 | 4 | 5 | 3 | 10 | 8 | 7 | 6 | 9;
|
|
6
6
|
export type ImageQuantity = typeof IMAGE_QUANTITIES[number];
|
|
7
7
|
export declare const clampImageQuantity: (count?: number) => ImageQuantity;
|
|
8
8
|
/** Default saved-recipe framing, shared by composer restore, quote and dispatch. */
|
package/dist/skills/content.js
CHANGED
|
@@ -18,7 +18,7 @@ export const SKILLS = {
|
|
|
18
18
|
"slates-prompting-elevenlabs": "---\nname: slates-prompting-elevenlabs\ndescription: How to prompt ElevenLabs Sound Effects v2 in Slates. Read before calling slates_generate_audio with model eleven-sfx — ONE short effect with an EXACT duration, or a seamless loop, billed per second. Covers describing an effect by its physical cause, the one-sound-per-generation rule, picking a duration, loops, prompt_influence, and when to use Seed Audio instead.\n---\n\n# ElevenLabs Sound Effects v2 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — ElevenLabs Sound Effects v2.** ONE short sound with an exact length, or a seamless loop. The only Slates audio surface with a real duration control and a real loop mode.\n\n**The five levers**\n1. **Describe the physical CAUSE, not the label** — `heavy oak door slams shut`, `boot scuffs on grit`, `a latch drops home`.\n2. **Name the material and the space.** The material decides the timbre and the space decides the tail: `on wet concrete`, `in a tiled stairwell`, `across an empty warehouse`.\n3. **One sound per generation.** A room with dialogue AND clatter AND ambience is one Seed Audio pass, not three effects.\n4. **Pick the duration from the cut**, not from a feeling: roughly 0.5-1s for an `impact`, 2-4s for a `whoosh`, 8-22s for a `loopable bed`.\n5. **Ask for a loop explicitly** — `seamless loop` — when the sound has to lie under a whole scene, and keep it featureless enough to survive the seam.\n\n**Examples**\n- `A heavy oak door slams shut in a stone hallway, brief reverberant tail.` (1.5s)\n- `Steady rain on a tin awning, no thunder, no wind gusts, seamless loop.` (18s)\n\n**Hard constraint:** it is billed per second and the duration is never left for the model to pick — that would make the charge non-deterministic. It is NOT a speech surface: a line in a specific voice is `inworld-tts-2`, and dialogue inside a scene is Seed Audio, which casts and performs the line in the room.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — a label is not a sound; describe the physical cause:\n- `door sound`, `whoosh`, `footsteps`, `impact`, `ambience` standing alone\n<!-- @banned:end -->\n\nOne short sound with an exact length, carried on fal (`fal-ai/elevenlabs/sound-effects/v2`). This is the only Slates audio surface with a real duration control and a real loop mode.\n\n## Where it routes\n\n- **A single hit that has to land on a known frame** — door slam, whoosh, impact, UI blip, riser.\n- **A seamless loop** you can lay under a whole scene — rain, engine hum, crowd murmur, machine noise.\n- **NOT** layered scenes. A room with dialogue *and* clatter *and* ambience is one `seed-audio` pass, not three SFX generations.\n- **NOT** speech. Dialogue, narration and scratch VO are `seed-audio` — it casts and performs the line inside the scene.\n- **AUDIO-ONLY.** It cannot produce images or video.\n\n## THE RULES\n\n### 1. Describe the physical CAUSE, not the label\n\n```\n✗ door sound\n✓ heavy oak door slams shut in a stone hallway\n\n✗ whoosh\n✓ a thick rope swung fast past a microphone, low air displacement\n\n✗ footsteps\n✓ boots on wet gravel, slow, one person\n```\n\nMaterial + weight + surface + room. Naming all four is the difference between a usable effect and a stock-library shrug. Cap is 450 characters — you will not need them.\n\n### 2. One sound per generation\n\nThis surface makes a single event. A door, then footsteps, then a siren is three generations layered on the timeline — or one `seed-audio` scene, which is usually cheaper and always more coherent.\n\n### 3. Duration is always explicit, and it is the price\n\nSlates **always sends** `durationSeconds`. (Left null the model picks, which makes the charge non-deterministic — so it is never left null.) The window it must fall in:\n\n<!-- @inject:thresholds -->\n<!-- GENERATED from @slatesvideo/shared — do not edit between the markers.\n Source: CONFIRM_CREDITS, DEVIATION_FACTOR and the audio bounds in\n packages/shared/src/operations/index.ts. Every number here is REFUSED by an\n op when a prompt gets it wrong, which is why none of them is typed by hand\n any more: this block replaced four claims that contradicted the code. -->\n\n**The thresholds, from the code that enforces them:**\n\n- **Confirm gate:** above **17 credits** an op returns `requires_confirm` and will not\n proceed until you re-call with `confirm: true`. Below it, announce the cost once and go.\n- **Deviation pause:** the desktop Studio Agent stops and re-asks when projected generation spend\n exceeds the approved plan by more than **20%**. You do not trigger this; the app does.\n- **Seed Audio duration:** **3–120 seconds.** There is no duration\n parameter on the model — the number you pass is written into the prompt AND is what the user is\n billed. Outside that range the op refuses rather than clamping.\n- **Sound Effects duration:** **1–22 seconds**, billed per second, never left for the\n model to pick.\n\nNever quote a credit figure from memory: `slates_estimate_generation_cost` returns the real one.\n<!-- @end:thresholds -->\n\n| Kind of sound | Ask for |\n|---|---|\n| impact, hit, click | 0.5–1s |\n| whoosh, riser, transition | 2–4s |\n| loopable bed | 8–22s + `loop: true` |\n\nOver-asking pads the tail with room tone you then trim. Under-asking clips the decay.\n\n### 4. Loops\n\n`loop: true` tiles without a seam — rain, engine hum, crowd murmur, machine noise. Combine with a longer duration so the loop point is not obvious.\n\nFor a bed longer than 22s, this is the wrong surface: `seed-audio` runs to 120s in one pass.\n\n### 5. Prompt influence\n\n`promptInfluence` 0–1, default 0.3. Higher hugs your wording with less variation between takes; lower explores. Raise it when a re-roll keeps wandering off the brief; lower it when every take sounds like the same take.\n\n## Iterating\n\n- Re-rolls that keep missing = the prompt named a **label** instead of a **cause**. Rewrite it as a physical event.\n- A hit that lands but sounds wrong in the scene is usually a *room* problem — name the space (\"in a stone hallway\", \"in a padded studio\", \"outdoors, no reflections\").\n- Three failed takes means the prompt is wrong, not the seed.\n\n## Content notes\n\nElevenLabs applies its own moderation. See slates-content-policy.\n",
|
|
19
19
|
"slates-prompting-flux-2-max": "---\nname: slates-prompting-flux-2-max\ndescription: How to prompt FLUX.2 Max (Black Forest Labs image model). Read before calling slates_generate_image with model flux-2-max, or slates_edit_image with editModel flux-2-max. FLUX.2 wants front-loaded structure, real camera vocabulary, and positive-only phrasing — no negative prompts, no tag soup.\n---\n\n# FLUX.2 Max — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — FLUX.2 Max.** Word order is weight: it attends hardest to the start. Structure: `Subject + Action + Style + Context`, then secondary detail. Length 10-30 words for a concept test, 30-80 for most work, 80+ only for a genuinely complex scene.\n\n**The five levers**\n1. **Front-load the subject and the one action.** Anything after the first clause is a modifier, and it is read as one.\n2. **Name real gear** — `Shot on Hasselblad X2D, 80mm, f/2.8, natural light`, `Kodak Portra 400, natural grain`. This is the single biggest realism lever.\n3. **Era cues as a package** — `early digital camera, slight noise, flash photography, candid` reads 2000s; `film grain, warm cast, soft focus` reads 80s.\n4. **Bind every hex colour to an object.** `a #1B4D3E enamel mug` lands; an unbound colour does not.\n5. **For portraits add texture words** — `natural skin texture, realistic pores, subtle imperfections, soft diffused lighting`.\n\n<!-- @inject:cinematic-card -->\n**For a photographic look, use only what this frame needs.** Image models default to clean, evenly lit and fully exposed. Describe what the camera sees, not just gear or mood:\n- **Inspect every reference first.** Write its grade and imperfections in words: darkness, contrast, muddy or true blacks, colour, softness/noise, subject separation. Never grade cleaner or brighter than the look reference unless asked.\n- **One light system** — `low sun behind her`, `her face falls into deep shadow`, `no light in front of her`.\n- **Visible exposure** — `the sky burns out to white`, `dense, slightly crushed shadows`.\n- **Lens name plus effect** — `200mm telephoto`, `peaks loom huge behind her and melt into soft shapes`.\n- **Name every garment and close the foreground.** Omissions invite reference leakage or invented props.\nBind references inline. A scene reference owns the grade; for a look-only reference, write the new scene's light. References are optional. For owned-frame edits, describe only the change and what stays.\n<!-- slates-only -->Use `slates-cinematic-look` with a technique ID or section query for more.<!-- /slates-only -->\n<!-- @end:cinematic-card -->\n\n**Hard constraint:** no negative prompting. Every \"no X\" must be rewritten as the positive state — `no blur` becomes `sharp focus throughout`, `no people` becomes `empty scene`, `no harsh shadows` becomes `soft, diffused lighting`.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — there is no negative prompting, so each of these has a positive form:\n- `no blur` (say `sharp focus throughout`), `no people` (say `empty scene`), `no harsh shadows` (say `soft, diffused lighting`)\n- an unbound hex colour — bind it to an object or it lands inconsistently\n- `masterpiece`, `best quality`, `trending on artstation`, `8k`\n<!-- @banned:end -->\n\n**Examples**\n- `A chef plating in a steel kitchen pass. Shot on Hasselblad X2D, 80mm, f/2.8. Overhead fluorescents plus warm spill from the line. Natural skin texture, subtle imperfections. Muted steel and #7A3B2E copper.`\n- `An empty municipal pool at dusk, 35mm, deep focus, early digital camera with slight noise and flash falloff. Cracked #4A7C8C tiles. Candid, unstaged.`\n\nBlack Forest Labs' top image model, routed via fal.ai. In Slates: `slates_generate_image` with `model: flux-2-max` (REQUIRES projectId — no headless path), priced per resolution (1k/2k/4k — call `slates_estimate_generation_cost` for current numbers, never quote from memory). Strengths vs Nano Banana 2: photoreal texture, less censored, precise hex-color control, strong typography. Reference images route through FLUX's edit endpoint and carry a lower per-model cap than NB2's 14.\n\n## Core structure — front-load what matters\n\n```\nSubject + Action + Style + Context\n```\n\nWord order is weight. FLUX.2 attends hardest to the start of the prompt: main subject → key action → critical style → essential context → secondary details.\n\n**Length:** 10-30 words for concept tests, 30-80 words for most work, 80+ only for genuinely complex scenes.\n\n## Photorealism: name real gear, not \"professional photo\"\n\nThe single biggest realism lever is concrete camera vocabulary:\n\n```\nShot on Hasselblad X2D, 80mm lens, f/2.8, natural lighting\nShot on Sony A7IV, 35mm, golden hour, shallow depth of field\nKodak Portra 400, natural grain, organic colors\n```\n\nEra cues work the same way: \"early digital camera, slight noise, flash photography, candid\" reads 2000s digicam; \"film grain, warm color cast, soft focus\" reads 80s.\n\nFor portraits add: natural skin texture, realistic pores, subtle imperfections, soft diffused lighting.\n\n## No negative prompts — reframe positively\n\nFLUX.2 has no negative prompt support. Describe the presence you want, not the absence:\n\n- ❌ \"no blur\" → ✅ \"sharp focus throughout\"\n- ❌ \"no people\" → ✅ \"empty scene\"\n- ❌ \"no harsh shadows\" → ✅ \"soft, diffused lighting\"\n\n## Hex colors — bind them to objects\n\nFLUX.2 matches hex codes, but only when each code is attached to a specific object:\n\n```\nwalls in hex #C4725A, sofa in #1B6B6F, accent pillows #E8A847\ngradient starting with color #02eb3c and finishing with color #edfa3c\n```\n\n❌ \"use #FF0000 somewhere\" — unbound colors land inconsistently.\n\n## Text rendering\n\nQuote the exact text, then place and style it:\n\n```\nThe text 'OPEN' appears in red neon letters above the door\nLogo text 'ACME' in color #FF5733, ultra-bold decorative serif, centered\n```\n\nSpecify placement relative to other elements, font family feel (serif / sans / script), and relative size (\"large headline,\" \"small body copy\").\n\n## JSON prompting for production work\n\nFor multi-element scenes that must come out exactly right (product shots, infographics, brand work), FLUX.2 parses structured JSON prompts:\n\n```json\n{\n \"scene\": \"Professional studio product photography on polished concrete\",\n \"subjects\": [{ \"description\": \"matte black ceramic mug with steam\", \"position\": \"center foreground\" }],\n \"style\": \"commercial product photography\",\n \"color_palette\": [\"#1B1B1B\", \"#E8A847\"],\n \"lighting\": \"three-point softbox, soft diffused highlights\",\n \"camera\": { \"lens-mm\": 85, \"f-number\": \"f/5.6\" }\n}\n```\n\nUse natural language for exploration, JSON when the layout is locked and you're matching a spec.\n\n## Reference images (edit path)\n\nIn Slates, pass `referenceAssetIds` on `slates_generate_image` — FLUX routes them through its edit endpoint. Slates names each reference inline in the prompt (\"the subject (image 1), the style (image 2)\") in the order it sends them, so you don't hand-write role labels; the name carries the role and unnamed-by-position blending is avoided. For surgical changes to one existing image use `slates_edit_image` with `editModel: flux-2-max` (note: FLUX edits ignore extra referenceAssetIds — that's NB2-only).\n\n### Reference rules (the verified ones)\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n<!-- @inject:reference-rules-core -->\nIdentity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.\n<!-- @end:reference-rules-core -->\n\n### For FLUX.2 Max specifically\n\n- **FLUX caps references well below NB2's 14, so rule 1's \"2-4\" is a ceiling here, not a starting point.** Be deliberate about which roles earn a slot.\n- **Rule 9 has a hard edge on this model:** `slates_edit_image` with `editModel: flux-2-max` ignores extra `referenceAssetIds` — that is NB2-only. A FLUX edit sees the source image and the prompt, nothing else.\n- **FLUX has no memory between generations, so rule 7 is enforced by repetition.** Define the character exhaustively once and repeat those exact descriptors verbatim in every subsequent prompt — see Character consistency across a series below.\n\n## Character consistency across a series\n\nDefine the character exhaustively once, then repeat those exact descriptors verbatim in every subsequent prompt. FLUX has no memory between generations — the repeated description IS the consistency mechanism.\n\n## Common failure modes + fixes\n\n| Failure | Fix |\n|---|---|\n| Generic \"AI look\" on photoreal | Name a camera body + lens + f-stop instead of \"professional photo\" |\n| Colors drift from brand spec | Bind each hex code to a named object |\n| Text garbled | Quote the exact string, specify font feel + placement + size |\n| Multi-reference blend chaos | Name each reference inline (Slates does this from your @mentions/referenceAssetIds) — the same name for one entity, distinct names per role |\n| Wanted element missing | Move it earlier in the prompt — order is weight |\n\n## Pre-flight: references arrive inline, refer by code\n\nWhen you pass `referenceAssetIds`, the first call returns the references **inline as image content blocks** with a cost estimate and `requires_confirm: true`. Look at them — revise the prompt if they suggest a different composition or style — then re-call with `confirm=true`. Refer to each asset by its short code (`IMG-A12 — Beach Sunset`) when talking to the user; it matches the badge on their gallery thumbnail.\n\n## Sources\n\n- [Black Forest Labs — FLUX.2 Prompting Guide](https://docs.bfl.ml/guides/prompting_guide_flux2)\n- [fal.ai — FLUX.2 [max] Prompt Guide](https://fal.ai/learn/devs/flux-2-max-prompt-guide)\n",
|
|
20
20
|
"slates-prompting-gpt-image-2-5": "---\nname: slates-prompting-gpt-image-2-5\ndescription: Prompt and edit images with GPT Image 2.5 Flare or Sunburst. Covers reference roles, realistic lighting, text, grids, quality choices and targeted edits. Use with slates_generate_image or slates_edit_image on these models.\n---\n\n# GPT Image 2.5 — sheets, grids, and text that actually reads\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — GPT Image 2.5.** The photoreal front-runner for people, and the readable-text, ordered-panel engine. Structure: subject and action with each reference named where it is used, then any exact copy in quotes, then layout, then light.\n\n**Pick the tier.** `flare` is Faster, quality comparable to GPT Image 2: drafts and volume. `sunburst` is Better quality, the most capable: finals, hero frames, photoreal people, multi-reference edits. Use the product default; choose Flare when speed is a stated priority.\n\n**The levers**\n1. **Name each reference inline** — `the woman from image 1`, `lit and graded like image 2`. Never an opening paragraph about what the references are.\n2. **Quote every string that must render verbatim** — `the jacket reads \"SLATES\"`. Describe a font's feel, never its name; keep on-image text under about 30 words.\n3. **Name the layout as a grid** for sheets and panels — `a 3x2 grid of panels, reading left to right, equal gutters`.\n4. **Set `quality` deliberately.** `high` is the everyday tier; `max` is 4× its price, `xhigh` about 1.8×. Coming from GPT Image 2 the names moved one rung: its `medium` is this `high`.\n\n<!-- @inject:cinematic-card -->\n**For a photographic look, use only what this frame needs.** Image models default to clean, evenly lit and fully exposed. Describe what the camera sees, not just gear or mood:\n- **Inspect every reference first.** Write its grade and imperfections in words: darkness, contrast, muddy or true blacks, colour, softness/noise, subject separation. Never grade cleaner or brighter than the look reference unless asked.\n- **One light system** — `low sun behind her`, `her face falls into deep shadow`, `no light in front of her`.\n- **Visible exposure** — `the sky burns out to white`, `dense, slightly crushed shadows`.\n- **Lens name plus effect** — `200mm telephoto`, `peaks loom huge behind her and melt into soft shapes`.\n- **Name every garment and close the foreground.** Omissions invite reference leakage or invented props.\nBind references inline. A scene reference owns the grade; for a look-only reference, write the new scene's light. References are optional. For owned-frame edits, describe only the change and what stays.\n<!-- slates-only -->Use `slates-cinematic-look` with a technique ID or section query for more.<!-- /slates-only -->\n<!-- @end:cinematic-card -->\n\n**Hard constraint:** its own content filter, distinct from Gemini's. Never describe a reference as a photograph of a real person.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use:**\n- a font NAME — describe the feel instead, as in: clean geometric sans, high contrast\n- a reference described as a photograph of a real person (`is a photograph of a woman`), or any up-front essay about what each reference is for — name the subject inline where it is used instead, as in: the woman from image 1\n- `8k`, `masterpiece`, `best quality`, `highly detailed` — quality incantations do nothing here either\n<!-- @banned:end -->\n\nGPT Image's edge is **character-level text accuracy** (~99% on English), ordered panels, and exact element placement — the jobs where every other model garbles a word or shuffles a layout. 2.5 inherits all of it and is better at each.\n\n## Which variant\n\n**Speed → Flare. Quality → Sunburst.** That is OpenAI's own routing rule, quoted from its image-prompting guide: *\"start with GPT Image 2.5 Flare when speed is the priority, or GPT Image 2.5 Sunburst when demanding quality requirements are the priority.\"* Same price either way, so the trade is purely latency against quality.\n\n🚨 **FLARE IS NOT AN UPGRADE OVER GPT IMAGE 2 — IT IS THE FAST ONE.** OpenAI, verbatim: *\"GPT Image 2.5 Flare is the small model, optimized for speed, with image quality **comparable to** GPT Image 2. GPT Image 2.5 Sunburst is the base model, optimized for quality, with **higher image quality than** GPT Image 2.\"* Their model pages agree: Flare is *\"our fastest model for high-quality, everyday image generation\"*, Sunburst *\"our most capable model for image generation and editing.\"* **Sunburst is the seat that beats what we had; Flare is the one that holds it at half the latency.** An earlier revision of this file called Flare \"better than GPT Image 2\" and sent Sunburst only to multi-reference edits — both wrong, corrected 2026-09-09 against the vendor docs.\n\n**Choose for the task.** Use the product default for ordinary work. Flare is an option when speed matters; changing model is not a mandatory draft stage.\n\n**Sunburst's widest lead is multi-reference editing** — several references all surviving into one frame, the character-consistency-across-shots problem. Reach for it there first, but that is not the only place it belongs.\n\n⚠️ **The LMArena receipt, scoped.** At launch Arena had Sunburst #1 and Flare #2 across text-to-image, single-image edit and multi-image edit, with margins over GPT Image 2 of **+81 / +47** on multi-image edit (Image Edit Arena: Sunburst 1520, Flare 1491, GPT Image 2 1461). Two caveats were missing and both matter: the baseline is **GPT Image 2 at `medium`, which is this model's `high`** — not its top tier — and the boards were **preliminary, a few thousand votes each**. Arena says Flare beats GPT Image 2; OpenAI says comparable. Route on OpenAI's wording and treat the board as a tiebreaker, not a spec.\n\n🚨 **The GPT Image line is ALSO the photoreal front-runner, and this file said the opposite until 2026-08-24.** **Receipts:** Eric's direct call, plus a head-to-head on the Higgsfield rail where GPT Image 2 at `quality: high`, 2K beat both Nano Banana rails on skin realism for photoreal people — that result is why the whole AI-influencer ad lane generates its plates here. **Route photoreal to this line, not away from it.**\n\n**Historical receipt, not a tier recommendation:** the photoreal comparison above used GPT Image 2 at its old `high` tier. It has not been repeated on 2.5 under matched conditions. Start with the product default and test a higher tier only against an unmet requirement; the old comparison does not establish a minimum tier for this model.\n\n**What the Banana line still owns:** edit-heavy work, and holding many subjects coherently in one frame. **Not the reference ceiling any more** — that line was true until 2026-09-09, when GPT Image went to its documented 16 against Banana's 14. Route on which model keeps them all recognisable, not on the count.\n\n**What would kill this:** a head-to-head at the intended crop going the other way. Per `slates-model-selection` § The meta-rule, re-run the evidence test when the roster changes — never carry a ranking forward on reputation. That rule is exactly what the 2026-08-24 correction failed, and exactly what the two ⚠️ notes above are honouring.\n\n## Quality tiers — always set explicitly\n\nAll five rungs are exposed, and they span ~36× end to end (2k class: $0.0044 → $0.158), which makes this the single biggest cost lever on the model. **The steps are UNEVEN — do not reason about them as a constant multiplier:** ~2.3× `low`→`medium`, ~3.9× `medium`→`high`, ~1.8× `high`→`xhigh`, ~2.25× `xhigh`→`max`. The same ratios hold at every OFFERED resolution class (2k/3k/4k); unoffered 1k differs slightly.\n\n| Tier | Use it for |\n|---|---|\n| `low` | Roughest pass — layout and composition checks, throwaway comps. |\n| `medium` | The draft tier. Cheaper than NB2 Lite and available up to 4K, which is why the draft lane moved here. |\n| `high` | General-purpose quality tier. Blind benchmarks on GPT Image 2 put this rung — which it called `medium` — within a hair of `max` (which it called `high`) at a quarter of the cost. Inherited from the old ladder, never re-run on 2.5, and it says nothing about `xhigh`. |\n| `xhigh` | One rung short of the top at about half its price (2k: 4 cr against `max`'s 8). Worth trying before `max`. |\n| `max` | Top of the ladder. Tiny type, dense diagrams, many labelled elements. |\n\n⚠️ **A tier label means different things on different models.** OpenAI: *\"The same quality label does not imply the same image quality or response time across models.\"* Flare at `max` and Sunburst at `max` are not the same picture, and neither matches Nano Banana's idea of \"high\".\n\n🚨 **The tier NAMES moved between versions and the strings did not.** GPT Image 2's `medium` is this model's `high`; its `high` is this model's `max` — same money, one rung of renaming. For a recipe explicitly written for GPT Image 2, map the old tier before reusing it on 2.5. A current user request for `medium` still means `medium`. Getting this backwards costs picture quality silently: nothing errors, the bill is correct for what was asked, and the image is just worse.\n\nNever rely on the provider default. fal's default is `high`, which is correct today — but it is the third rung of five rather than the top of two, so leaning on it means a fal-side change silently reprices you. Slates sends its configured quality explicitly; current defaults live in `slates-model-selection`. Your explicit choice overrides them.\n\n**Start at the default and change tiers for an unmet requirement.** OpenAI's own procedure: *\"If the output falls short, test a higher quality setting. Once it meets your requirements, test lower settings to see whether they preserve acceptable quality while reducing latency. Use `xhigh` or `max` only when they improve an unmet quality requirement within your latency budget.\"* A higher rung does **not** guarantee a better result on a given prompt. Compare `medium` against `high` when the job is small or dense text; that is where the rungs separate most visibly.\n\n## Resolution classes\n\n`1k` = 1024²-class · `2k` = 1920×1080-class · `3k` = 2560×1440-class · `4k` = 3840×2160-class. Pick 2k for most sheets/panels; 4k for print-density grids. 4K exists at every tier and is API-only — even paid ChatGPT can't render it.\n\n`1k` is not offered, and the reason is not its price: it is strictly dominated. At 1k you pay more for fewer pixels than at 2k, at **all five tiers**. Don't ask for it.\n\n⚠️ **Above 2560×1440 you are on a path OpenAI marks EXPERIMENTAL.** Verbatim: *\"Outputs with more than 3,686,400 total pixels ('2560x1440') are experimental.\"* That is the whole **4k** class (≈8.0 MP) plus 3k at 4:3/3:4 (≈3.70 MP). It bills normally and it works — but prove the shot at 2k or 3k 16:9 first, and do not be surprised by an odd frame at 4k.\n\n**Hard size bounds**, from fal's schema verbatim: each edge ≤ 3840 px, both edges multiples of 16, longer:shorter ratio ≤ 3:1, total pixels between 655,360 and 8,294,400. **The pixel ceiling is the one that actually bites** — the multiple-of-16 rule is documented but NOT enforced, and we have the receipt: 1920×1080 fails it (1080 = 67.5 × 16), is one of fal's own six priced canonical sizes, and metered clean. Slates picks sizes that respect the ceiling; these matter only if you hand-build a request.\n\n🚨 **THE ASPECT RATIO CHANGES THE PRICE ON THIS MODEL, and on no other image model.** OpenAI bills image OUTPUT TOKENS and the count tracks the frame's SHAPE, so at the same resolution class **`1:1` costs about 1.8× and `4:3`/`3:4` about 1.37× what `16:9` costs**; `9:16` costs the same as `16:9`. Metered 2026-09-09 and priced into the cost key, so the quote you get before generating is the real number — but if you are choosing between shapes and the budget is tight, **16:9 or 9:16 is the cheap one.** Every other image model charges the same whatever the shape.\n\n## Reference images — give every one a role, inline, where it is used\n\n**Assign a role to every reference image: subject, style, clothing, or background.** This is new emphasis in 2.5 and the highest-leverage change for the 16-reference character lane. An unroled pile of references makes the model guess what each one is for, and it guesses differently every run — which is the drift people mistake for a consistency failure.\n\n**The role rides a clause in the scene, not a paragraph in front of it.** *The woman from image 1 cooks on a rocky summit…*, *lit and graded like image 2*. Never open with sentences about what each reference is and what to take or ignore from it: that is the role essay the shared reference rules below forbid, and it drags the sheet's studio light into the scene.\n\n**Receipt, 2026-09-15, Sunburst, IMG-A192–A198.** The up-front version returned the studio look; the inline versions were never refused and never came back as a sheet. Two costs, both fixed in words: anything the prompt does not describe is taken from the reference (name every garment), and props nobody asked for appear (say what is in the foreground and that nothing else is). One sheet-only plate kept its described location, which narrows the two-reference rule in `slates-ugc-influencer-ad`. A look reference did far less than a described light. The full ladder is the vault's `cinematic-look-research.md`; the techniques are `slates-cinematic-look`.\n\nReference images route through the edit endpoint, **up to 16** — fal's documented `maxItems`, and the highest reference ceiling of any image seat in Slates (the Banana line takes 14). It was capped at 10 until 2026-09-09, which was never anybody's limit, just a number nobody had checked. The composed \"image N\" naming applies as everywhere else. Mask-based inpainting exists at the API level but is not surfaced: a mask is something the user has to paint, and there is no painting surface — describe the change instead.\n\n## Editing — separate the change from the constraints\n\n**State the change, then list what must survive.** \"Change only X,\" then name the invariants explicitly: identity, geometry, lighting, labels. For precise local edits also pin saturation, contrast, camera angle and surrounding objects — anything you do not pin is fair game for the model to move.\n\n**One change per iteration, and restate the constraints every turn.** Cross-turn drift is the named failure mode in OpenAI's own guidance: constraints do not persist across turns by themselves, so a multi-turn refinement that stops restating them will slowly rewrite the frame. This applies directly to multi-turn shot refinement.\n\n## Prompting for text accuracy\n\n- **Quote every string that must render verbatim**: `the sign reads \"OPEN 24 HOURS\"` — quoted strings render most reliably.\n- Say the text appears **once**, and give its position and typography.\n- Spell unusual words letter-by-letter.\n- Add `no extra text, no watermarks`.\n- Specify font *feel*, not font names: \"clean geometric sans, high contrast\", \"hand-painted brush lettering\".\n- For dense text (posters, UI mocks), list the copy as ordered lines: `Line 1: \"...\" Line 2: \"...\"` — it respects ordering.\n- **Don't bundle unrelated instructions into a text-rendering request.** A prompt that also redesigns the scene competes with the text for attention.\n- Keep total on-image text under ~30 words for perfect accuracy; beyond that, accuracy degrades gracefully but degrades.\n\n## Transparent backgrounds\n\nIf you need a cut-out rather than a scene, **ask for it explicitly and check the alpha**. OpenAI: request `background=transparent` and use PNG or WebP, then *\"check the decoded image's alpha channel, including hair, glass, shadows, and object edges\"* — a painted-white backdrop is the common failure and it is not transparency. Say what must NOT appear: *\"no solid backdrop, no checkerboard, no scenery, no watermark\"*, and do not let the product get restyled while the background is removed. **On every follow-up edit, repeat the transparency requirement** or it gets dropped. (Slates always requests PNG, so the format half is handled for you. **`background` IS surfaced now** — the Background control on the prompt bar, and `backgroundMode` on `slates_generate_image` / `slates_edit_image`. It is free: fal prices this family on size × quality alone.)\n\n## When an edit must not touch a region at all\n\nPrompting alone cannot guarantee pixel-identical pixels. OpenAI's own instruction: if a region must stay exactly as it was, **composite the approved edit back into the original image** rather than asking the model to preserve it. Treat \"preserve\" language as a strong bias, never a lock.\n\n## Structure a complex prompt in labeled sections\n\nFor anything with several requirements, OpenAI recommends organising the prompt as **scene, subject, details, constraints** with labeled sections. Same content, easier to read and to change one part without disturbing the rest — which is what makes the one-change-per-iteration rule practical.\n\n**Say \"photorealistic\" or \"real photograph\" when that is the goal.** It is not inferred from a detailed description; ask for it directly, then describe framing and texture.\n\n## Concrete visuals beat mood words\n\nName materials, lighting, colour and medium. Mood words are cues only — \"cinematic\", \"moody\", \"epic\" tell the model almost nothing on their own. Give scale, atmosphere and colour instead. Camera specs (`85mm`, `f/1.4`) are appearance hints, not a physical simulation; they bias the look, they do not compute optics.\n\n**Name the lens and describe its effect, every time.** A lens named alone changed nothing visible (IMG-A195, 2026-09-15); named together with what it does to the picture, it produced real compression and depth of field (IMG-A198). Wording: `slates-cinematic-look` → `compression-as-outcome`, `defocus-as-outcome`.\n\n**For people, state body framing and scale**: \"full body visible, feet included\", \"hands naturally gripping the handlebars\". This is also the safest way to phrase a crop — see the blocked-phrasings section below.\n\n**No special syntax is required.** Prose, JSON and tagged blocks all work equally well, so pick whatever stays maintainable in the caller.\n\n## Panels, sheets, and grids\n\n- State the grid explicitly and number the cells: \"a 2×3 grid of panels, numbered 1–6, reading left-to-right, top-to-bottom\".\n- Give each cell ONE content clause: \"Panel 3: the character mid-jump, side view\".\n- Character identity sheets: GPT Image holds both the structured panel layout AND photoreal skin, which is why the influencer-ad lane builds its sheets here. Reach for NB2/NB Pro when it is an edit of an existing sheet, or when many subjects have to stay recognisable at once — not for the reference count, which GPT Image now leads at 16.\n\n## 🚨 WHAT GETS YOU BLOCKED — read before writing a prompt with a person in it\n\n**Receipt: 24 consecutive attempts on one character, 2026-08-24, same project and same rail.** Eleven were refused with `content_policy_violation` on the fal edit endpoint. The refusals were never about the scene — one of the blocked prompts was a woman standing at a kitchen counter with her hand on it. **Two phrasings were hard blocks, 5 for 5 each, and neither ever passed:**\n\n**1. Never describe the reference as a photograph of a real person.**\n\n> ❌ `Reference image 1 is a photograph of a woman. Use that exact woman.`\n> ✅ `Reference image 1 is a character identity sheet showing one woman across several panels — the face in the large portrait panel is the authority for her identity. Use that exact woman.`\n\nThe first reads to the filter as *recreate this real person's likeness*, which is a hard refusal regardless of what the rest of the prompt says. The second signals a fictional character and passes. **This is a wording change only — the reference image can be the same file either way.** One plate flipped from refused to accepted on this single sentence with nothing else altered.\n\n**Inline naming sidesteps the question and is now the default:** never describe the reference at all, and name her where she is used (*the woman from image 1*). Six of six Sunburst plates written that way passed on 2026-09-15. Keep the sheet sentence above as the fallback if a refusal appears.\n\n**2. Never attach a reference sheet containing a headless body panel.** A sheet whose full-body panels are cropped above the neck is refused every time, even with the correct opener. Regenerate the sheet with the head visible in every panel. Related, and already in this file's sheet guidance: phrase a cropped panel as *framing* (`cropped at the collarbone`), never as *absence* (`the head not shown`).\n\n⚠️ **These refusals were measured on GPT Image 2, not on 2.5.** The classifier belongs to OpenAI rather than to a model version, so the phrasing rules carry — but they are inherited, not re-measured. If Flare or Sunburst accepts one of the blocked phrasings, that is a new receipt to write down here, not a reason to delete this one.\n\n**On top of those, ordinary content triggers still apply** and they stack independently — a correct opener does not rescue them:\n\n| Refused | Why, and the fix |\n|---|---|\n| A woman sitting on a bed in a bedroom | Domestic + bed reads as intimate. Move her to a chair, a rug, another room. |\n| A knife, even lying flat on a chopping board next to a lemon | The object is the trigger, not the framing. Swap it — a cast-iron pan cleared instantly. |\n\n**🚨 Refusals are PROBABILISTIC. Retry once before rewriting a word.** In the same session an identical prompt, identical reference, identical params was refused and then accepted on a straight re-fire. A rejected job returns no file and costs nothing, so a retry is free and a rewrite is not — rewriting first is how you end up changing four variables and learning nothing. **Only redesign after two or three refusals.**\n\n**And change ONE thing at a time.** The eleven refusals above took far longer to diagnose than they should have because a reference swap and an opener rewrite shipped in the same call. Isolate on the prompt you actually want, so a pass leaves you with a usable asset instead of a data point.\n\n## Filter regime\n\nOpenAI moderate — a third regime distinct from Gemini (NB family) and ByteDance (Seedream). Real-face references pass more readily than Gemini; violence/brand rules are similar. `slates-content-policy` applies unchanged.\n",
|
|
21
|
-
"slates-prompting-inworld-tts": "---\nname: slates-prompting-inworld-tts\ndescription: How to use Inworld Realtime TTS-2, the VOICE seat. Read before calling slates_generate_audio with model inworld-tts-2. Speech in a SPECIFIC voice, billed per character - the prompt is the words spoken, verbatim. Covers the identity-versus-acoustics rule (what a reference clip does and does not carry), how to write a line so it is performed rather than read, when to reach for seed-audio instead, and the voice-consent rule.\n---\n\n# Inworld Realtime TTS-2 — the voice seat\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Keep it under 2,400 characters (the build fails above that) and keep\n the rationale and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Inworld TTS-2.** Speech in a SPECIFIC voice. The prompt is the words spoken, verbatim — not a description of them. Text length determines the bill.\n\n**IDENTITY, NOT ACOUSTICS — the rule that decides whether cloning works**\nA reference carries WHO is speaking: timbre, pitch, accent, age, vowel shape. It does NOT carry WHERE they are — room tone, distance, phone EQ, reverb and mic character are *acoustics*, and this model reproduces the identity while discarding the room. So:\n1. **A noisy reference does not give a noisy read — it gives a WORSE identity.** Music, a second speaker or heavy reverb corrupt what is being extracted. Use a clean single-speaker recording.\n2. **You cannot get \"on a payphone\" by cloning a payphone recording.** Acoustics come from the MIX, or from `seed-audio` which renders a room.\n\n**DIRECTION GOES IN SQUARE BRACKETS. PARENTHESES ARE SPOKEN ALOUD.** `[whispering] I hope nobody notices` is whispered; `(quietly) I hope nobody notices` says the word \"quietly\" out loud. Verified by ear — the easiest way to ruin a take.\n\n- **Plain English works inside them** — it is natural-language steering, not a fixed vocabulary: `[very quiet]`, `[whisper in a hushed style]`, `[very slow]`, `[say excitedly]`. Non-verbals are their own tags: `[laugh]`, `[sigh]`, `[breathe]`, `[clear throat]`.\n- **A tag it does not recognise is still consumed, and still changes the read.** Never spoken, never an error — so a mistyped tag fails SILENTLY and only listening catches it.\n- **Tags persist across sentences** until changed; `[reset]` returns to normal.\n- **Punctuation is the timing.** `Wait. Stop.` differs from `Wait, stop.`\n- **One line, one take.** Split a paragraph so a bad clause costs one re-roll.\n- **Spell numbers and titles aloud:** `twenty twenty-six`, `Doctor Reyes`.\n\n**Route elsewhere when:** the scene needs dialogue mixed with effects and room tone in one pass (`seed-audio`), or it is a single non-speech sound (`eleven-sfx`). This surface makes ONE voice saying ONE thing, cleanly.\n\n**Hard constraints:** no duration parameter — length falls out of the text. Exactly one voice source: a preset `voiceId` from `slates_list_voices`, a clip as `voiceReferenceAssetId` (a character's voice clip to speak AS the character, or any clean clip of one speaker), or `voiceDescription`.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — the prompt on this surface is SPOKEN ALOUD, so anything that describes the audio instead of being the audio gets read out as words:\n\n- `SFX`, `Ambient noise`, `Background music` as labels — this model speaks; it does not render a scene. Use `seed-audio` for those.\n- shot language: `wide shot`, `slow push in`, `warm tungsten` — video-prompt words, and here they would literally be said aloud\n- `voiceover`, `narrator says`, `he says` as stage directions wrapping the line — write only the words that should come out of the speaker\n<!-- @banned:end -->\n\n## Why the prompt is not a prompt\n\nOn every other surface in Slates the prompt DESCRIBES what you want and the model interprets it. Here the prompt IS the deliverable: each character is spoken aloud and each character is billed. `a gravelly man says he is tired` produces a voice saying the words \"a gravelly man says he is tired\".\n\nThat also means the two numbers a user cares about are the same number. The text length sets the price (in 250-character buckets) and sets the length of the audio. There is nothing to choose and nothing to reconcile.\n\n## Steering the delivery\n\n`VERIFIED BY EAR, 2026-09-05.` Every claim in this section was listened to, not\ninferred — an earlier draft of this skill documented tag forms that had only been\nprobed for an HTTP 200, which proves the request was accepted and nothing about\nwhether it was obeyed.\n\n**Square brackets are consumed. Parentheses are read aloud.** That is the whole\nrule, and getting it wrong is not a subtle degradation — the audience hears a\nnarrator say the word \"quietly\" in the middle of your line.\n\n| Written | What comes out |\n|---|---|\n| `[whispering] I really hope nobody notices that.` | whispered, tag not spoken ✅ |\n| `[very quiet] I really hope nobody notices that.` | very quiet, tag not spoken ✅ |\n| `[whisper in a hushed style] …` | hushed, tag not spoken ✅ |\n| `[very slow] …` | slowed right down, tag not spoken ✅ |\n| `[laugh] …` | an actual laugh, then the line ✅ |\n| `(quietly, under his breath) …` | 🚨 **the words \"quietly, under his breath\" are SPOKEN** |\n\n**Plain English works — it is natural-language steering, not a fixed vocabulary.**\nBoth the documented phrasings (`[whisper in a hushed style]`) and ordinary adverbs\n(`[whispering]`) were obeyed. Write the direction the way you would say it to an\nactor.\n\nThe eight dimensions the model steers on, with a working example of each:\n\n| Dimension | Example |\n|---|---|\n| Emotion | `[say excitedly]`, `[sound sad]`, `[sound terrified]` |\n| Articulation | `[say with force]`, `[articulate clearly]` |\n| Intonation | `[say with a rising pitch]` |\n| Volume | `[very quiet]`, `[very loud]` |\n| Pitch | `[say in a low tone]` |\n| Range | `[say playfully]`, `[say with no pitch variation]` |\n| Speed | `[very fast]`, `[very slow]` |\n| Vocal style | `[whisper in a hushed style]`, `[give a nasal quality]` |\n\nNon-verbals sit inline where they happen: `[laugh]`, `[sigh]`, `[cough]`,\n`[breathe]`, `[yawn]`, `[clear throat]`.\n\n### Four rules that are not obvious\n\n1. 🚨 **A tag it does not recognise is still consumed, and still changes the read.**\n `[zzzqqq]` is not spoken and does not error — it produces a different, arbitrary\n delivery. So a typo in a tag is SILENT: there is no rejection, no warning, and no\n way to catch it except listening to the take. Treat an unexpected performance as\n a possible misspelled tag before you blame the voice.\n2. **Tags persist across sentences.** A `[very slow]` at the top governs everything\n after it until something changes it. Use `[reset]` to go back to normal rather\n than assuming the next sentence starts clean.\n3. **Do not stack opposing directions.** `[whisper in a hushed style]` together with\n `[very loud]` produces unpredictable results — the model is resolving a\n contradiction, and which side wins is not something you can rely on.\n4. **Tags COUNT toward the billed characters**, even though they are never spoken.\n They are part of the text sent to the vendor, so the vendor charges for them and\n so do we — billing what was actually sent is the only honest basis. It rarely\n matters (a 13-character tag inside a 250-character bucket), but a line sitting\n just under a bucket boundary can be pushed into the next one by a long\n direction. Prefer `[very slow]` over `[say this one very slowly please]`.\n\n## Identity versus acoustics, at length\n\nThis is the distinction that decides whether the feature feels good, and it is worth being precise about because the failure is quiet — you get a usable clip that is subtly not the person.\n\n**What a reference clip transfers:** vocal timbre, pitch range, accent and regional vowels, apparent age, speech rate tendencies, and the particular rasp or breathiness of the source speaker.\n\n**What it does not transfer:** the room, the microphone, the codec, the distance from the mic, any processing on the source, and any other sound present in it.\n\nSo the ideal reference is boring: one person, close to a microphone, no music, no second speaker, no heavy reverb, five to fifteen seconds, speaking normally rather than performing. A phone voice memo in a quiet room beats a beautifully produced clip with a music bed underneath it.\n\n**Two failure modes, both common:**\n\n- *\"I cloned my podcast intro and it doesn't sound like me.\"* The intro had music under it. The model averaged the music into the identity. Re-clone from a clean stretch.\n- *\"I want the line to sound like it's coming through a car radio.\"* Clone the clean voice, then EQ and process the returned clip on the timeline. A radio-sounding reference makes a worse voice, not a radio effect.\n\n## Getting the voice onto the call\n\nExactly one source per call, and none of them requires a character to exist first:\n\n- **A preset:** `slates_list_voices` lists stock voices with gender, age, accent and tags — filter by any of them, or search the descriptions (\"gravelly\", \"narration\"). Pass the chosen `voiceId`. Presets clone nothing, so they are the fastest path and avoid the clone-creation rate ceiling.\n- **Speak AS a character:** `voiceReferenceAssetId: <its voiceAssetId>` (the clip on the row `slates_list_characters` returns). The seat clones the clip for that take and discards the vendor voice afterwards, so there is nothing to reconcile — but cloning shares a ceiling of two new voices a minute across every Slates user, so a run of lines in one cloned voice pauses between takes rather than failing. Send each line once; do not re-send one that already came back. Any other clean clip of one speaker works the same way.\n- **A voice with no recording:** `voiceDescription` (7–1000 characters of words). If it will be used again, keep the returned clip on a character with `slates_update_character` (`voiceAssetId`) so later lines clone the same clip instead of designing a new voice each time — a convenience, never a requirement.\n\n## Consent\n\nCloning a real person's voice needs that person's explicit, documented permission, scoped to what you are making. Clone from original human recordings only — never from another model's output. This is the same gate the real-face route applies to likeness, and it applies here for the same reason.\n\n## Worked examples\n\n**A line with a direction**\n\n```\n[very quiet] I heard what you said in there. I'm not going to pretend I didn't.\n```\n\n**A line that needs its numbers spoken**\n\n```\nThe vote was three hundred and twelve to eighty-nine. It carried at four minutes past midnight.\n```\n\n**A paragraph, split into three takes** — so one bad clause costs one re-roll:\n\n```\n1. You keep asking me why I stayed.\n2. It wasn't loyalty. It wasn't even fear, not by the end.\n3. [very slow] It was that I couldn't picture the version of me that left.\n```\n\n**What NOT to send**\n\n```\n(gravelly, tired) a tired old man narrates the opening of the film, wide shot, warm tungsten\n```\n\nEvery word of that is spoken aloud — **including the parenthetical**, which is the\ntrap: it looks like a stage direction and is treated as dialogue. Describe the voice when you are CHOOSING one (`voiceDescription`, or the desktop's voice picker); the prompt is only ever the words.\n",
|
|
21
|
+
"slates-prompting-inworld-tts": "---\r\nname: slates-prompting-inworld-tts\r\ndescription: How to use Inworld Realtime TTS-2, the VOICE seat. Read before calling slates_generate_audio with model inworld-tts-2. Speech in a SPECIFIC voice, billed per character - the prompt is the words spoken, verbatim. Covers the identity-versus-acoustics rule (what a reference clip does and does not carry), how to write a line so it is performed rather than read, when to reach for seed-audio instead, and the voice-consent rule.\r\n---\r\n\r\n# Inworld Realtime TTS-2 — the voice seat\r\n\r\n<!-- @card:start -->\r\n<!-- slates-only -->\r\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\r\n src/prompts/craft-cards.ts and returned on every cost estimate for this\r\n model, so it is the ONE piece of positive craft guidance the agent cannot\r\n skip. Keep it under 2,400 characters (the build fails above that) and keep\r\n the rationale and the worked examples in the body below. -->\r\n<!-- /slates-only -->\r\n**Card — Inworld TTS-2.** Speech in a SPECIFIC voice. The prompt is the words spoken, verbatim — not a description of them. Text length determines the bill.\r\n\r\n**IDENTITY, NOT ACOUSTICS — the rule that decides whether cloning works**\r\nA reference carries WHO is speaking: timbre, pitch, accent, age, vowel shape. It does NOT carry WHERE they are — room tone, distance, phone EQ, reverb and mic character are *acoustics*, and this model reproduces the identity while discarding the room. So:\r\n1. **A noisy reference does not give a noisy read — it gives a WORSE identity.** Music, a second speaker or heavy reverb corrupt what is being extracted. Use a clean single-speaker recording.\r\n2. **You cannot get \"on a payphone\" by cloning a payphone recording.** Acoustics come from the MIX, or from `seed-audio` which renders a room.\r\n\r\n**DIRECTION GOES IN SQUARE BRACKETS. PARENTHESES ARE SPOKEN ALOUD.** `[whispering] I hope nobody notices` is whispered; `(quietly) I hope nobody notices` says the word \"quietly\" out loud. Verified by ear — the easiest way to ruin a take.\r\n\r\n- **Plain English works inside them** — it is natural-language steering, not a fixed vocabulary: `[very quiet]`, `[whisper in a hushed style]`, `[very slow]`, `[say excitedly]`. Non-verbals are their own tags: `[laugh]`, `[sigh]`, `[breathe]`, `[clear throat]`.\r\n- **A tag it does not recognise is still consumed, and still changes the read.** Never spoken, never an error — so a mistyped tag fails SILENTLY and only listening catches it.\r\n- **Tags persist across sentences** until changed; `[reset]` returns to normal.\r\n- **Punctuation is the timing.** `Wait. Stop.` differs from `Wait, stop.`\r\n- **One line, one take.** Split a paragraph so a bad clause costs one re-roll.\r\n- **Spell numbers and titles aloud:** `twenty twenty-six`, `Doctor Reyes`.\r\n\r\n**Route elsewhere when:** the scene needs dialogue mixed with effects and room tone in one pass (`seed-audio`), or it is a single non-speech sound (`eleven-sfx`). This surface makes ONE voice saying ONE thing, cleanly.\r\n\r\n**Hard constraints:** no duration parameter — length falls out of the text. Exactly one voice source: a preset `voiceId` from `slates_list_voices`, a clip as `voiceReferenceAssetId` (a character's voice clip to speak AS the character, or any clean clip of one speaker), or `voiceDescription`.\r\n<!-- @card:end -->\r\n\r\n<!-- @banned:start -->\r\n<!-- slates-only -->\r\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\r\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\r\n estimate, and every submitted prompt is matched against it. Keep entries\r\n backticked and prose outside the backticks. -->\r\n<!-- /slates-only -->\r\n**Never use** — the prompt on this surface is SPOKEN ALOUD, so anything that describes the audio instead of being the audio gets read out as words:\r\n\r\n- `SFX`, `Ambient noise`, `Background music` as labels — this model speaks; it does not render a scene. Use `seed-audio` for those.\r\n- shot language: `wide shot`, `slow push in`, `warm tungsten` — video-prompt words, and here they would literally be said aloud\r\n- `voiceover`, `narrator says`, `he says` as stage directions wrapping the line — write only the words that should come out of the speaker\r\n<!-- @banned:end -->\r\n\r\n## Why the prompt is not a prompt\r\n\r\nOn every other surface in Slates the prompt DESCRIBES what you want and the model interprets it. Here the prompt IS the deliverable: each character is spoken aloud and each character is billed. `a gravelly man says he is tired` produces a voice saying the words \"a gravelly man says he is tired\".\r\n\r\nThat also means the two numbers a user cares about are the same number. The text length sets the price (in 250-character buckets) and sets the length of the audio. There is nothing to choose and nothing to reconcile.\r\n\r\n## Steering the delivery\r\n\r\n`VERIFIED BY EAR, 2026-09-05.` Every claim in this section was listened to, not\r\ninferred — an earlier draft of this skill documented tag forms that had only been\r\nprobed for an HTTP 200, which proves the request was accepted and nothing about\r\nwhether it was obeyed.\r\n\r\n**Square brackets are consumed. Parentheses are read aloud.** That is the whole\r\nrule, and getting it wrong is not a subtle degradation — the audience hears a\r\nnarrator say the word \"quietly\" in the middle of your line.\r\n\r\n| Written | What comes out |\r\n|---|---|\r\n| `[whispering] I really hope nobody notices that.` | whispered, tag not spoken ✅ |\r\n| `[very quiet] I really hope nobody notices that.` | very quiet, tag not spoken ✅ |\r\n| `[whisper in a hushed style] …` | hushed, tag not spoken ✅ |\r\n| `[very slow] …` | slowed right down, tag not spoken ✅ |\r\n| `[laugh] …` | an actual laugh, then the line ✅ |\r\n| `(quietly, under his breath) …` | 🚨 **the words \"quietly, under his breath\" are SPOKEN** |\r\n\r\n**Plain English works — it is natural-language steering, not a fixed vocabulary.**\r\nBoth the documented phrasings (`[whisper in a hushed style]`) and ordinary adverbs\r\n(`[whispering]`) were obeyed. Write the direction the way you would say it to an\r\nactor.\r\n\r\nThe eight dimensions the model steers on, with a working example of each:\r\n\r\n| Dimension | Example |\r\n|---|---|\r\n| Emotion | `[say excitedly]`, `[sound sad]`, `[sound terrified]` |\r\n| Articulation | `[say with force]`, `[articulate clearly]` |\r\n| Intonation | `[say with a rising pitch]` |\r\n| Volume | `[very quiet]`, `[very loud]` |\r\n| Pitch | `[say in a low tone]` |\r\n| Range | `[say playfully]`, `[say with no pitch variation]` |\r\n| Speed | `[very fast]`, `[very slow]` |\r\n| Vocal style | `[whisper in a hushed style]`, `[give a nasal quality]` |\r\n\r\nNon-verbals sit inline where they happen: `[laugh]`, `[sigh]`, `[cough]`,\r\n`[breathe]`, `[yawn]`, `[clear throat]`.\r\n\r\n### Four rules that are not obvious\r\n\r\n1. 🚨 **A tag it does not recognise is still consumed, and still changes the read.**\r\n `[zzzqqq]` is not spoken and does not error — it produces a different, arbitrary\r\n delivery. So a typo in a tag is SILENT: there is no rejection, no warning, and no\r\n way to catch it except listening to the take. Treat an unexpected performance as\r\n a possible misspelled tag before you blame the voice.\r\n2. **Tags persist across sentences.** A `[very slow]` at the top governs everything\r\n after it until something changes it. Use `[reset]` to go back to normal rather\r\n than assuming the next sentence starts clean.\r\n3. **Do not stack opposing directions.** `[whisper in a hushed style]` together with\r\n `[very loud]` produces unpredictable results — the model is resolving a\r\n contradiction, and which side wins is not something you can rely on.\r\n4. **Tags COUNT toward the billed characters**, even though they are never spoken.\r\n They are part of the text sent to the vendor, so the vendor charges for them and\r\n so do we — billing what was actually sent is the only honest basis. It rarely\r\n matters (a 13-character tag inside a 250-character bucket), but a line sitting\r\n just under a bucket boundary can be pushed into the next one by a long\r\n direction. Prefer `[very slow]` over `[say this one very slowly please]`.\r\n\r\n## Identity versus acoustics, at length\r\n\r\nThis is the distinction that decides whether the feature feels good, and it is worth being precise about because the failure is quiet — you get a usable clip that is subtly not the person.\r\n\r\n**What a reference clip transfers:** vocal timbre, pitch range, accent and regional vowels, apparent age, speech rate tendencies, and the particular rasp or breathiness of the source speaker.\r\n\r\n**What it does not transfer:** the room, the microphone, the codec, the distance from the mic, any processing on the source, and any other sound present in it.\r\n\r\nSo the ideal reference is boring: one person, close to a microphone, no music, no second speaker, no heavy reverb, five to fifteen seconds, speaking normally rather than performing. A phone voice memo in a quiet room beats a beautifully produced clip with a music bed underneath it.\r\n\r\n**Two failure modes, both common:**\r\n\r\n- *\"I cloned my podcast intro and it doesn't sound like me.\"* The intro had music under it. The model averaged the music into the identity. Re-clone from a clean stretch.\r\n- *\"I want the line to sound like it's coming through a car radio.\"* Clone the clean voice, then EQ and process the returned clip on the timeline. A radio-sounding reference makes a worse voice, not a radio effect.\r\n\r\n## Getting the voice onto the call\r\n\r\nExactly one source per call, and none of them requires a character to exist first:\r\n\r\n- **A preset:** `slates_list_voices` lists stock voices with gender, age, accent and tags — filter by any of them, or search the descriptions (\"gravelly\", \"narration\"). Pass the chosen `voiceId`. Presets clone nothing, so they are the fastest path and avoid the clone-creation rate ceiling.\r\n- **Speak AS a character:** `voiceReferenceAssetId: <its voiceAssetId>` (the clip on the row `slates_list_characters` returns). The seat clones the clip for that take and discards the vendor voice afterwards, so there is nothing to reconcile — but cloning shares a ceiling of two new voices a minute across every Slates user, so a run of lines in one cloned voice pauses between takes rather than failing. Send each line once; do not re-send one that already came back. Any other clean clip of one speaker works the same way.\r\n- **A voice with no recording:** `voiceDescription` (7–1000 characters of words). If it will be used again, keep the returned clip on a character with `slates_update_character` (`voiceAssetId`) so later lines clone the same clip instead of designing a new voice each time — a convenience, never a requirement.\r\n\r\n## Consent\r\n\r\nCloning a real person's voice needs that person's explicit, documented permission, scoped to what you are making. Clone from original human recordings only — never from another model's output. This is the same gate the real-face route applies to likeness, and it applies here for the same reason.\r\n\r\n## Worked examples\r\n\r\n**A line with a direction**\r\n\r\n```\r\n[very quiet] I heard what you said in there. I'm not going to pretend I didn't.\r\n```\r\n\r\n**A line that needs its numbers spoken**\r\n\r\n```\r\nThe vote was three hundred and twelve to eighty-nine. It carried at four minutes past midnight.\r\n```\r\n\r\n**A paragraph, split into three takes** — so one bad clause costs one re-roll:\r\n\r\n```\r\n1. You keep asking me why I stayed.\r\n2. It wasn't loyalty. It wasn't even fear, not by the end.\r\n3. [very slow] It was that I couldn't picture the version of me that left.\r\n```\r\n\r\n**What NOT to send**\r\n\r\n```\r\n(gravelly, tired) a tired old man narrates the opening of the film, wide shot, warm tungsten\r\n```\r\n\r\nEvery word of that is spoken aloud — **including the parenthetical**, which is the\r\ntrap: it looks like a stage direction and is treated as dialogue. Describe the voice when you are CHOOSING one (`voiceDescription`, or the desktop's voice picker); the prompt is only ever the words.\r\n",
|
|
22
22
|
"slates-prompting-kling-v3": "---\nname: slates-prompting-kling-v3\ndescription: How to prompt Kling V3.0 (Kuaishou). Read before calling slates_generate_video with kling-v3.0-std, kling-v3.0-pro, or kling-v3.0-omni. Kling has dialogue + SFX + ambient native syntax (Omni adds multi-character dialogue and language codes). Multi-shot rules differ from Seedance/Veo — don't cross syntaxes.\n---\n\n# Kling V3.0 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Kling V3.0.** The general default. Define the core subjects clearly at the START and keep those descriptions identical across shots. Up to 15s, up to 6 cuts, and the strongest image-to-video identity hold in the catalogue.\n\n**The five levers**\n1. **Dialogue in quotes** — `Character says, \"exact words here\"`. On Omni, direct the voice with `Gender + Age + Voice quality + Speech rate + Emotional tone + Language`: `[Character A: Detective, mid-40s, raspy, slow cadence, weary]: \"I've seen this before.\"`\n2. **Unique speaker labels, no pronouns after the introduction.** `he`, `the agent`, any synonym causes voice drift.\n3. **Sound has real syntax** — `SFX: heavy boots on wet pavement, distant siren wailing`, `Ambient noise: city traffic`, `Background music: low cello`. Always physical-cause specific; `SFX: footsteps` is not enough.\n4. **Motion adverbs modulate energy directly** — `slowly`, `rapidly`, `gently`, `explosively`. One primary camera move per shot, never stacked.\n5. **On image-to-video, do NOT re-describe the image.** It is an anchor; prompt how the scene EVOLVES from it — movement, camera, environmental change.\n\n**Examples**\n- `A detective in a wet grey overcoat stands under a stairwell light. He steps forward slowly as the light flickers. [Character A: Detective, mid-40s, raspy voice, slow cadence, weary]: \"I've seen this before.\" SFX: heavy boots on wet concrete, distant siren wailing. Ambient noise: rain on metal.`\n- `Camera tracks right alongside a cyclist crossing a bridge at dusk. She rises out of the saddle rapidly as the grade steepens. Ambient noise: wind, tyres on wet asphalt, distant traffic.`\n\n**Hard constraint:** `Immediately` (Omni only) removes the natural conversational beat between speakers — use it when timing matters and leave it out when it does not. Kling has a real `negativePrompt` field, unlike Seedance; start from the standard block and layer scene-specific suppressions.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use:**\n- `SFX: footsteps` and any label-only effect — physical-cause specificity or nothing\n- a pronoun or synonym for a speaker after the first introduction (`he`, `the agent`) — it causes voice drift; repeat the full label\n- `single continuous take` — Seedance's phrase, and it fights Kling's multi-shot\n<!-- @banned:end -->\n\nKuaishou's video model. Three tiers: `kling-v3.0-std` (general use, no audio), `kling-v3.0-pro` (higher visual quality, no audio), `kling-v3.0-omni` (multi-character dialogue + audio-visual co-generation).\n\nUp to 15s. Multi-shot supported (up to 6 cuts in 15s total). Strong on image-to-video — preserves identity, layout, and text from the input image well.\n\n## Subject definition rule (verbatim, fal blog)\n\n> \"Define your core subjects clearly at the beginning of the prompt and keep descriptions consistent across shots.\"\n\n## Dialogue syntax\n\n```\nCharacter says, \"exact words here\"\n```\n\nUse quotation marks for precise speech. Languages (Omni only): EN, ZH, JA, KO, ES.\n\n## Voice direction formula (Omni)\n\n```\nGender + Age Range + Voice Quality + Speech Rate + Emotional Tone + Language\n```\n\nExample:\n```\n[Character A: Detective, mid-40s, raspy voice, slow cadence, weary]: \"I've seen this before.\"\n```\n\nTone phrases that fire:\n- `speaking in a hushed, trembling whisper`\n- `shouting with commanding authority`\n- `clear, fearful voice`\n- `with a trembling voice, \"I'm scared\"`\n\n## The `Immediately` keyword (Omni only)\n\nWithout `Immediately`, Kling adds a natural conversational beat between speakers. With it, dialogue is back-to-back. Use when timing matters.\n\n```\n[Alice]: \"Get down!\" Immediately, [Bob]: \"Where?\"\n```\n\n## Speaker label discipline\n\nUnique labels per character. **No pronouns or synonyms after first introduction** — they cause voice drift.\n\n✅ `[Character A: Black-suited Agent]` ... `[Character A: Black-suited Agent]: \"Stop.\"`\n❌ `[Agent]... then he says...`\n\n## Multi-character dialogue (Omni)\n\n```\nAlice says in English, \"Hello!\" Then Bob replies in Spanish, \"¡Hola!\"\n```\n\n## Sound effects, ambient noise, music\n\n```\nSFX: thunder cracks, footsteps approaching\nAmbient noise: city traffic, birds chirping, ocean waves\nBackground music: tense orchestral strings, low cello\n```\n\nSFX accepts physical-cause specificity:\n- ✅ `SFX: heavy boots on wet pavement, distant siren wailing`\n- ❌ `SFX: footsteps`\n\n## Image-to-video guidance\n\n**Verbatim (fal blog):**\n> \"Treat the input image as an anchor. Kling 3.0 excels at preserving the identity, layout, and text details. Focus prompts on how the scene evolves *from* the image: subtle movements, camera motion, or environmental changes.\"\n\n**Don't re-describe what's already in the image.** Focus on motion, changes, evolution.\n\n## Multi-shot — what makes them hit\n\n**Hard cap: total duration ≤ 15s across all shots. Max 6 cuts.**\n\nHit conditions:\n- Shot labels are explicit: `Shot 1:`, `Shot 2:`\n- One primary action per shot\n- Subject described identically in each shot block\n- Camera move per shot is **one verb**, not a chain\n- Per-shot blocks: 30-60 words\n\nMiss conditions:\n- Compressing narrative into one paragraph\n- Pronoun-only references after the first shot\n- Mixing camera moves within a shot (\"pan then orbit then push in\")\n- Extreme wide → extreme close in adjacent shots without reference images\n\n## Element references (Omni)\n\nUpload 2-4 multi-angle reference photos per character/object. Tag inline:\n\n```\n@element1 is the protagonist (refs: front, side, back angles).\n@element2 is the antagonist.\n```\n\n## Reference discipline (character / environment refs)\n\n<!-- @inject:references-read-literally -->\n> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.\n<!-- @end:references-read-literally -->\n\n<!-- @inject:reference-rules-core -->\nIdentity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.\n<!-- @end:reference-rules-core -->\n\n### For Kling specifically\n\n- **Kling's consistency lever is \"lock the subject with a fixed label reused verbatim.\"** That is Kling's phrasing for rules 2 and 3, and it is stricter than the others: **pronoun and synonym drift breaks it**, so the exact same label must appear on every single mention — not \"he\", not \"the detective\" after you named him. Reusing the label verbatim is the whole game. Slates composes this for you from `@mentions`.\n- **Element references are the transport for rule 1** — 2-4 multi-angle photos per character/object, tagged `@element1` / `@element2` (see Element references above). The cap is 4 combined refs on the edit path.\n\n## Negative prompting — has a real field\n\nKling exposes `negative_prompt` on the fal endpoint (different from Seedance which has none). Default block to start from:\n\n```\nblurry, low quality, watermark, text overlay, distorted hands, extra fingers,\nduplicate limbs, unnatural skin texture, overly saturated colors,\nfloating objects, inconsistent shadows, jittery, flickering, morphing face\n```\n\nLayer scene-specific suppressions on top, and never suppress something the prompt asks for. This block carried `lens flare` until 2026-09-15, which silently cancelled every flare a prompt described (`slates-cinematic-look` → `source-flare`); add it back only for a shot that must have none.\n\n## Cinematic tactics\n\n- **Motion adverb precision** modulates motion energy directly: `slowly`, `rapidly`, `gently`, `explosively`\n- **Camera vocabulary that registers as instructions:** profile shot, tracking, following, freezing, panning, \"moving in sync with the subject\"\n- **One primary camera move per shot** — never stack\n\n## Tier choice\n\n- **Standard**: general use, no audio\n- **Pro**: higher visual quality, no audio\n- **Omni**: multi-character dialogue, audio-visual co-gen, language codes, `@elementN` references\n\nPick by capability: need dialogue/audio → Omni; need maximum visual quality silent → Pro; everything else → Standard. Prices change — check current numbers before choosing a tier<!-- slates-only -->; call `slates_estimate_generation_cost` or `slates_list_available_models`<!-- /slates-only -->.\n\n## Benchmark prompt structure\n\n```\n[Character A: <role>, <voice quality>]: \"<line>.\" Immediately, [Character B: <role>, <voice quality>]: \"<reply>.\"\nAmbient noise: <soundscape>.\nCamera <single move>.\n```\n\nCinematic example (paraphrasing fal blog patterns):\n> \"Shot 1: Wide establishing shot of a neon-lit alleyway in heavy rain, steam rising from grates. Camera slowly tracks forward.\n> Shot 2: Medium shot of a detective in a trench coat ducking under an awning, water dripping from his hat brim. [Detective: weary, raspy]: 'I knew she'd come back.' Ambient noise: distant traffic, rain on metal.\n> Shot 3: Close-up on his eyes, narrowing as headlights flash across his face.\"\n\n<!-- slates-only -->\n## Pre-flight: references arrive inline, refer by code\n\nWhen you call `slates_generate_video` with `firstFrameAssetId` or `ingredientAssetIds`, the first call returns those references **inline as image content blocks** alongside cost + `requires_confirm: true`. Look at them, revise prompt if needed, then re-call with `confirm=true`. Kling Omni multi-character with several ingredient images especially benefits — confirm each character image lands cleanly before spending.\n\nWhen talking to the user about the gen, refer to each reference by its short code: `IMG-A12 — Detective Closeup`. The user sees that code as a gallery badge.\n\n- ✅ \"I'm anchoring on **IMG-A12** as the detective and **IMG-A18** as the alleyway environment — Omni will handle the line delivery in EN.\"\n- ❌ \"I'm using the detective image and the alley one...\" (which alley? Three exist.)\n<!-- /slates-only -->\n\n## Video-to-video EDIT<!-- slates-only --> (`slates_edit_video`)<!-- /slates-only --> — @Video1 / @ElementN / @ImageN\n\nKling O3 edit takes an EXISTING 3-15s clip and changes only what the prompt names — character swap, environment change, style transfer — in one pass, no masking. Original motion, camera, and audio are preserved by default. Its notation is Kling's own, different from the \"image N\" naming used everywhere else:\n\n- **`@Video1`** — the source clip (always; the transport anchors the instruction to it).\n- **`@Element1..`** — subjects to swap IN. Each element = one frontal image + up to 3 angle images<!-- slates-only --> (pass as `characterAssetIds`; @mention names in the prompt compile to @ElementN automatically)<!-- /slates-only -->.\n- **`@Image1..`** — style/appearance references<!-- slates-only --> (pass as `styleAssetIds`)<!-- /slates-only -->.\n- Max **4 combined** element + image refs per edit.\n\n**Prompt shape — the change, not the whole scene:**\n\n```\nReplace the man in @Video1 with @Element1, keeping his walk cycle, the camera move, and the rain unchanged.\n```\n\n```\nEdit @Video1: turn the daytime street into a neon-lit Tokyo alley at night, wet asphalt reflections. Apply the visual style of @Image1. Keep the subject and camera motion exactly as they are.\n```\n\nRules:\n- Name what CHANGES; explicitly state what stays (\"keep the motion / camera / everything else unchanged\") — the model preserves better when told to.\n- One edit intent per pass. Chain passes for compound changes (each output is itself an editable clip, linked to its parent).\n- Billing is per second of OUTPUT ≈ the clip length, rounded UP to the next second. A 7.3s clip bills as 8s.\n- Clip constraints: 3-15s, 720-3840px, MP4/MOV. Agents can pre-trim on the timeline when a clip runs long.\n- Routing: Kling edit is the default edit tool (element lock + audio intact); Seedance edit/relocate wins style-transfer-heavy re-imaginings<!-- slates-only --> — see `slates-model-selection`<!-- /slates-only -->.\n\n## Sources\n\n- [fal.ai — Kling 3.0 Prompting Guide](https://blog.fal.ai/kling-3-0-prompting-guide/)\n- [Vidguru — Kling 3.0 Omni Guide](https://www.vidguru.ai/blog/kling-3.0-omni-guide.html)\n- [AcceptPrompt — Kling 3 Prompt Guide](https://www.acceptprompt.com/blog/kling-3-prompt-guide)\n- [DataCamp — Kling 3.0 Tutorial](https://www.datacamp.com/tutorial/kling-3-0)\n",
|
|
23
23
|
"slates-prompting-lip-sync": "---\nname: slates-prompting-lip-sync\ndescription: How to set up lip-sync — Kling-only (dedicated lip-sync and avatar endpoints, 5-second outputs). Read before calling slates_generate_lip_sync. Two flows — video→video re-dub and image→video avatar — with different inputs, pricing, and gotchas. Voice catalog, framing rules, audio file constraints, and which tier to pick. Also covers the Seedance alternative, which is a normal video generation rather than a mode of this tool.\n---\n\n# Lip-sync — setup guide\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — Lip-sync (Kling only).** Two different flows with different inputs and different prices; every output is 5 seconds.\n\n**The five levers**\n1. **Pick `sourceType` deliberately** — `video` re-dubs an existing talking head (cheapest); `image` animates a still portrait (avatar-standard, then avatar-pro only on the final selected take).\n2. **The `prompt` on the avatar flows is SCENE CONTEXT, not motion direction.** Ambience, lighting, micro-expression: `Soft rim light`, `warm office`, `cool blue evening light through a window`, `gentle confident smile between sentences`, `focused intent expression`.\n3. **Clean the audio before uploading** — `noise-reduced`, `levelled`. Lip detection is sensitive, and a raw recording is the most common cause of a bad take.\n4. **Iterate on the SOURCE or the AUDIO, never on a refinement prompt** — there is not one. If the output is wrong, change the input.\n5. **Use avatar-standard for first-pass dialogue takes**, and switch to pro only once the line is locked. Facial fidelity is not visible until then.\n\n**Examples**\n- `Soft rim light, warm office, gentle confident smile between sentences.`\n- `Cool blue evening light through a window, focused intent expression.` (Or `.` — an empty prompt is fine when you have nothing to add.)\n\n**Hard constraint:** it is Kling-only and always 5 seconds. For a generated PERFORMANCE instead — head movement, gesture, delivery energy, with the dialogue as a native conditioning signal — that is a normal Seedance video generation with the clip attached as a video reference, not a mode of this tool. A real recording, or a cloned/cast voice rendered on `inworld-tts-2`, for production; this tool's built-in TTS is for scratch.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — the avatar prompt is scene context and motion verbs are ignored:\n- `turns her head`, `raises an eyebrow`, `hand gestures`, `nods`, `walks`\n- `reader_en_m-v1` — listed in fal's docs, returns \"Voice id not found\" in production\n<!-- @banned:end -->\n\n**This tool is Kling-only.** It wraps Kling's dedicated lip-sync and avatar endpoints; every entry is a real endpoint and every output is 5 seconds.\n\n| Flow | Source | Model | Cost | Use case |\n|------|--------|-------|-----------|----------|\n| Re-dub | video clip | kling-lip-sync-video | ~4 credits / 5s | Replace dialogue on an existing talking head |\n| Avatar standard | still image | ai-avatar/v2/standard | ~14 credits / 5s | Animate a portrait into a talking avatar |\n| Avatar pro | still image | ai-avatar/v2/pro | ~29 credits / 5s | Higher facial fidelity for hero shots |\n\nPick `sourceType` deliberately — it decides the pricing tier and the underlying endpoint.\n\n## Want Seedance instead? That is a video generation, not a mode here\n\nSeedance can generate the performance rather than bolting a mouth onto finished pixels — head movement, gesture, delivery energy, with the dialogue as a native conditioning signal, and a video source keeps its own voice. **It is not an engine switch on this tool.** Run a normal `slates_generate_video` on `seedance-2` with the clip (or portrait) attached as a video/ingredient reference and the dialogue written into the prompt yourself.\n\nThat is the same endpoint the old `engine=seedance-2` branch called — it just built the sentence for you, invisibly, and it presupposed a \"video 1\" that might not exist. Writing the prompt is the whole difference, and it is the part you want control of.\n\n- Driving clips must be 2–15s; output duration is whatever you set (4–15s).\n- Video references bill COMBINED input+output seconds (`seedance-2*-vref-*` keys) — pass the clip duration and quote before confirming. On Seedance 2.5's AI-face route (EvoLink) the input side counts as at least the output's length: max(input, output) + output.\n- Faces go through the normal cascade: `seedanceFace` for a character, `[REAL_FACE_DETECTED]` → `seedanceRealFace` + `realFaceConsent` for a real person.\n\nEverything below is about the Kling tool.\n\n## Choosing video vs avatar\n\nUse **video** (re-dub) when:\n- A talking-head clip already exists (Slates-generated, recorded, or imported)\n- The mouth/face is already moving and only the audio needs to change\n- ~4 credits is hard to beat for short dialogue replacement\n\nUse **avatar** when:\n- Only a still portrait exists\n- The character needs to come alive from a single image\n- Identity + face fidelity matter (avatar-pro for hero shots, standard for everything else)\n\n## Source asset constraints\n\n### Video flow (`sourceType: 'video'`)\n- Format: mp4 or mov\n- Duration: 2–10s (lip-sync output is always 5s — long videos get trimmed)\n- Resolution: 720p or 1080p (480p will be rejected)\n- Max file size: 100MB\n- Face must be visible and roughly facing camera. Profile shots fail.\n- Existing audio is replaced.\n\n### Avatar flow (`sourceType: 'image'`)\n- Min 512×512, PNG/JPG/WebP\n- **Face occupies 60–70% of frame.** This is the single biggest avatar quality lever.\n- Eyes open, mouth neutral, looking near-camera. Side profile = bad output.\n- Single subject, clean background. Group photos confuse the face anchor.\n\n## Audio source\n\nTwo ways to drive the lips:\n\n### TTS (`audioMethod: 'tts'`)\n- Pass `ttsText` (the words spoken)\n- Optional: `ttsVoice` (default `oversea_male1`), `ttsLanguage` (default EN), `ttsSpeed` (default 1.0)\n- **Hard cap: 120 characters of text.** Longer = silently truncated.\n- Languages: EN, ZH, JA, KO, ES\n\n### Upload (`audioMethod: 'upload'`)\n- Pass `audioFilePath` — absolute path to an audio file on the user's machine\n- Format: mp3, wav, m4a, ogg, aac\n- Max 5MB\n- Duration: 2–60s (output is 5s — longer audio gets trimmed)\n- Single clean voice. Music underneath, multiple speakers, or noisy mics produce garbage lips.\n\nPrefer upload for production-quality voice. TTS for fast iteration / placeholder dialogue.\n\n## Voice catalog (TTS)\n\nReliable English voices (verified working on the fal endpoint as of 2026):\n\n| Voice ID | Description |\n|----------|-------------|\n| `oversea_male1` | Male, English — default, stable |\n| `commercial_lady_en_f-v1` | Female commercial English |\n| `uk_boy1` | Young man, UK accent |\n| `uk_man2` | Man, UK accent |\n| `uk_oldman3` | Older man, UK accent |\n| `calm_story1` | Storyteller / narrator |\n\nAvoid `reader_en_m-v1` — listed in fal.ai docs but returns \"Voice id not found\" in production.\n\nFull 48-voice list (ZH, JA, KO included): https://fal.ai/models/fal-ai/kling-video/lipsync/text-to-video/api\n\n## Speech-rate notes\n\n`ttsSpeed` range 0.5–2.0:\n- 0.8–1.0: natural conversational\n- 1.1–1.3: punchy ad delivery\n- 1.4+: rushed, clips consonants\n- 0.6–0.7: slow, weighty (good for dramatic lines)\n\nDefault 1.0 unless the line specifically calls for slower or faster cadence.\n\n## Avatar prompt usage\n\nThe `prompt` parameter on avatar-v2 (standard + pro) is **scene context**, not motion direction. The mouth animation comes from the audio — the prompt sets ambiance, lighting, micro-expression.\n\nGood:\n- `Soft rim light, warm office, gentle confident smile between sentences.`\n- `Cool blue evening light through a window, focused intent expression.`\n\nBad (the model ignores motion verbs):\n- ❌ `She turns her head, raises an eyebrow, then speaks.`\n- ❌ `Hand gestures while talking.`\n\nDefault `\".\"` is fine if you have nothing useful to add.\n\n## Tier selection — avatar standard vs pro\n\n**Use standard** when:\n- Drafts, A/B testing voices, internal review reels\n- Wide / medium shots where face isn't the focal point\n- Cost matters more than micro-expression fidelity\n\n**Use pro** when:\n- Final ads where the avatar's face fills the screen\n- The character is named / branded — identity drift kills the take\n- You're already paying tens of credits for the surrounding video pipeline\n\nDon't default to pro. The ~15-credit delta per take adds up across iteration.\n\n## Common failure modes\n\n| Symptom | Likely cause | Fix |\n|---------|--------------|-----|\n| Lip movement looks \"rubber\" / disconnected | Source face <60% of frame | Re-crop the still tighter |\n| Voice doesn't match character age/gender | Default voice id used | Pick from voice catalog |\n| Output truncated mid-word | TTS text >120 chars | Shorten or chain two takes |\n| Garbled mouth on uploaded audio | Background music / multi-voice | Use clean dialogue-only audio |\n| \"Voice id not found\" 422 | Hit `reader_en_m-v1` | Switch to `oversea_male1` |\n| Avatar eyes drift / cross | Source had closed/angled eyes | Pick a frame with neutral open eyes |\n| Generation completes but lips don't move | Profile shot / face >70° off-axis | Use a near-frontal portrait |\n\n## Cost discipline\n\n- Video re-dub at ~4 credits is the cheapest dialogue iteration in the entire Slates stack — use it for voice A/B testing\n- Avatar standard at ~14 credits is fine for medium use\n- Avatar pro at ~29 credits trips the confirm gate — explicit user OK required every time\n- All 5s. There is no shorter option.\n\n## Workflow patterns\n\n**Voice A/B test (cheap):**\n1. Generate one base talking-head video clip with Veo or Seedance (~40 credits)\n2. Run `slates_generate_lip_sync` with `sourceType: 'video'` against 3–5 different `ttsVoice` values\n3. Total cost: ~40 + (5 × ~4) ≈ 60 credits to compare voices\n\n**Brand avatar from a single portrait:**\n1. Generate or upload the hero portrait (face fills frame, eyes open, neutral mouth)\n2. Avatar standard for first-pass dialogue takes\n3. Avatar pro only on the final selected take\n\n**Avoid:**\n- Avatar pro on first iteration (waste — facial fidelity isn't visible until you've locked the line)\n- TTS for final ads (production should use real voice or cloned voice — the upload flow)\n- Uploading raw recordings — clean noise + level the file first, lip detection is sensitive\n\n## Confirm gate: cost + codes, no inline preview\n\nLip-sync is mechanical — the model re-syncs the chosen source to the chosen audio. The confirm response carries the source asset's code so you can announce it in chat.\n\n- ✅ \"Lip-syncing **IMG-A12 — Founder Portrait** to the new line. ~29 credits on avatar-pro. Confirm?\"\n- ❌ \"Using the founder image...\" (which? Three exist.)\n\nDon't second-guess the source. If the output is wrong, iterate on source choice or audio, not on a refinement prompt (there isn't one).\n\n## Sources\n\n- [fal.ai — Kling LipSync API](https://fal.ai/models/fal-ai/kling-video/lipsync/text-to-video/api)\n- [fal.ai — AI Avatar v2 Standard](https://fal.ai/models/fal-ai/kling-video/ai-avatar/v2/standard/api)\n- [fal.ai — AI Avatar v2 Pro](https://fal.ai/models/fal-ai/kling-video/ai-avatar/v2/pro/api)\n",
|
|
24
24
|
"slates-prompting-ltx-2-5": "---\nname: slates-prompting-ltx-2-5\ndescription: How to prompt LTX-2.5 and LTX-2.5 Pro. Read before calling slates_generate_video with model ltx-2-5 or ltx-2-5-pro. LTX scores the picture on the same pass that draws it, so SOUND IS THE FIRST THING YOU WRITE — Lightricks ranks the prompt sound, camera, character detail, shot type and scene, then scene dressing, all in one flowing paragraph. It is also the catalogue's native MULTISHOT seat: one generation carries two to four connected shots holding character, light and voice across the cuts. Base ltx-2-5 is the distilled build — 720p/1080p/1440p/4K, clips of 6 to 20 seconds in EVEN steps, and the cheapest native 1080p second in Slates; ltx-2-5-pro is the full diffusion build and is NOT a superset, reaching only 1080p and 10 seconds for about a third more money. Three hazards live here: durations are even numbers only from six (there is no 5s or 7s clip), the model has NO reference endpoint at all so identity references are unavailable, and any sound not anchored to something in frame gets invented for you.\n---\n\n# LTX-2.5 — prompting\n\n<!-- @card:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Everything between the @card markers is extracted by\n src/prompts/craft-cards.ts and returned on every cost estimate for this\n model, so it is the ONE piece of positive craft guidance the agent cannot\n skip. Measured 2026-08-30: a fact inlined where it cannot be skipped moved\n compliance 0/8 to 30/32; the same guidance behind a fetch moved nothing.\n Keep it under 2,400 characters (the build fails above that) and keep the\n rationale, the receipts and the worked examples in the body below. -->\n<!-- /slates-only -->\n**Card — LTX-2.5.** It scores the picture on the same pass that draws it, so SOUND IS THE FIRST THING YOU WRITE. Lightricks' own priority order: sound, camera, character detail, shot type and scene, then scene dressing. One flowing paragraph, not labelled sections. When a prompt sprawls, cut from the bottom.\n\n**The five levers**\n1. **Lead with sound, and anchor every sound to something in frame.** The test is \"visible, or at least locatable\" — `the rope creaks against the cleat`, `the hull knocking hollow against the fenders`, `rain on the awning`. A distant whistle is fine IF you have named the post it comes from.\n2. **Camera second**, because framing decides the visual weight of the shot — `low camera at the gunwale`, `slow drift right`, `static medium behind the counter`.\n3. **Character detail as physical ACTION**, not as adjectives about a person — `she braces a boot on the rail and hauls`, `his hands counting notes`.\n4. **It is the native MULTISHOT seat** — one generation carries two to four connected shots holding character, light and voice across the cuts. Write the cuts.\n5. **Quote dialogue and name the language and accent** — `in English with a slight German accent` — `\"We should not have come back,\" in English with a slight German accent.`\n\n**Examples**\n- `The rope creaks against the cleat as she leans back, gulls calling somewhere off the port bow, the hull knocking hollow against the fenders. Low camera at the gunwale, slow drift right. She braces a boot on the rail and hauls, twice, then stops.`\n- `A till drawer bangs shut, a fan ticks against its cage, rain on the awning outside. Static medium behind the counter, then cut to a close-up of his hands counting notes, then cut wide as he looks up at the door.`\n\n**Hard constraint:** durations are EVEN numbers from six — there is no 5s or 7s clip. There is NO reference endpoint at all, so identity references are unavailable; use MiniMax H3 or Kling when a character must hold across shots. Any sound not anchored to something in frame gets invented for you. And never write mood adjectives as sound: \"tense atmosphere\", \"a sense of dread\" and \"ominous ambience\" produce nothing usable — the fix is one more moving object in frame with a sound attached to it.\n<!-- @card:end -->\n\n<!-- @banned:start -->\n<!-- slates-only -->\n<!-- MACHINE-READ. Every `backticked` token between the @banned markers is\n extracted by src/prompts/banned-tokens.ts and returned on this model's cost\n estimate, and every submitted prompt is matched against it. Keep entries\n backticked and prose outside the backticks. -->\n<!-- /slates-only -->\n**Never use** — mood adjectives standing in for sound produce nothing usable:\n- `tense atmosphere`, `a sense of dread`, `ominous ambience`, `eerie silence`\n- an unanchored sound: name the thing in frame it comes from, or cut it\n<!-- @banned:end -->\n\nLTX-2.5 generates picture and sound **in a single pass**, with a Gemma-4 12B text encoder reading\none flowing paragraph. That single fact drives everything below: the prompt is not a shot\ndescription with audio bolted on, it is **a scene where the sound is load-bearing** — and\nLightricks' own priority order puts sound first, ahead of the camera.\n\nTwo seats, and the naming is a trap:\n\n| | `ltx-2-5` (base) | `ltx-2-5-pro` |\n|---|---|---|\n| Build | Distilled, 8-step | Full diffusion (\"Diffusion Fidelity Rendering\") |\n| Resolutions | 720p / 1080p / **1440p** / 4K | 720p / 1080p |\n| Durations | 6–20s, even steps | 6 / 8 / 10s |\n| Price | $0.09–$0.30 per second | $0.12–$0.17 per second |\n| Reach for it when | iterating, long takes, 4K delivery, batch volume | one dense final render inside 1080p and 10s |\n\n**Pro is not \"base plus more.\"** It buys picture quality on a *narrower* envelope — it cannot make\na 1440p frame and it cannot make a 12-second clip. Reaching for it out of habit costs a third more\n*and* takes away the reach.\n\n---\n\n## 1. The six parts, in priority order, in one paragraph\n\nLightricks ranks the elements of an LTX prompt like this. When a prompt sprawls, **cut from the\nbottom.**\n\n1. **Sound** — highest priority; the model scores the picture as it draws it.\n2. **Camera** — framing decides visual weight and the feel of the shot.\n3. **Character detail** — expressed as physical action.\n4. **Shot type and scene** — the action itself.\n5. **Scene dressing** — the first thing to trim.\n\nWrite it as **one flowing paragraph**, not a list of labelled sections. LTX is not Seedance (eight\nengineering slots) and not H3 (three separate audio layers) — it wants continuous prose.\n\n---\n\n## 2. Sound: anchor it or it gets invented\n\n**Write the audio line last, then go back and check every cue has a source you could point at.**\nAnything unanchored, the model invents for you.\n\nThe test is **\"visible, or at least locatable.\"** A distant whistle is fine *if* you have named the\nmarshal's post it comes from. A \"distant whistle\" with nothing to attach to is a coin flip.\n\n> the rope creaks against the cleat as she leans back, gulls calling somewhere off the port bow,\n> the hull knocking hollow against the fenders\n\n**Never write mood adjectives as sound.** \"Tense atmosphere\", \"a sense of dread\" and \"ominous\nambience\" produce nothing usable. If a scene feels thin, the fix is **one more moving object in\nframe with a sound attached to it** — never another adjective.\n\n### Dialogue\n\nQuote it, and name the language and accent:\n\n> \"We should not have come back,\" in English with a slight German accent.\n\nTwo rules that decide whether the lip sync lands:\n\n- **Give the character a beat of stillness before they speak.** The sync needs something to lock\n against; a character already mid-motion when the line starts drifts.\n- **Describe the beat structure** — when they look, how long they wait, when they speak, where they\n look afterwards.\n\nSlates pins the frame rate at 25fps, which is also what Lightricks recommends for dialogue: at 50fps\nthe performance \"pulls toward a video look.\"\n\n---\n\n## 3. Character emotion is physical\n\nThe model renders actions. It does not render adjectives.\n\n| Instead of | Write |\n|---|---|\n| she looks anxious | her jaw sets, she turns the ring on her finger twice |\n| he seems exhausted | he blinks slowly and lets his shoulder take the doorframe |\n| a tense standoff | neither moves; his thumb finds the strap and stays there |\n\n---\n\n## 4. Multishot — the thing this model is uniquely for\n\n**One LTX generation can carry several connected shots**, holding character, environment, lighting,\nvoice and style across every cut. Nothing else in the catalogue does this natively; everywhere else\nyou generate separate clips and stitch them, and identity drifts between them.\n\n**Working range is two to four shots.** Three is the comfortable stopping point.\n\nAt **every** transition you must supply four things:\n\n1. **Name the edit in the prose** — \"hard cut\", \"dissolve\", \"match cut\".\n2. **Re-establish the shot completely** — scale, angle, lens and light all reset at a cut. A cut is\n not a continuation.\n3. **Re-identify recurring characters by their original descriptor.** \"The woman in the bronze\n gown\", never \"she\". Pronouns lose the character across a cut — this is the single most common\n multishot failure.\n4. **State what the sound does at the cut.** Silence is not assumed; if the room tone should drop\n out, say so.\n\nA shape that works:\n\n> Wide establishing shot of the workshop, dust in the window light, a lathe turning somewhere off\n> frame — hard cut — macro close-up of the brass fitting as it seats, the turning noise gone,\n> replaced by a single dry click — match cut — medium shot of the woman in the bronze gown stepping\n> back, the room tone returning underneath her.\n\n---\n\n## 5. Camera: write it, don't enumerate it\n\nfal exposes a `camera_motion` enum (dolly in/out/left/right, jib up/down, static, focus shift).\n**Slates does not surface it, deliberately** — and prose is the better instrument anyway:\n\n- **A written move can be tied to a specific moment.** \"A slow push-in that settles as she reaches\n the door, then holds\" is not expressible as an enum value.\n- **For multishot it would be actively wrong** — one enum value would impose a single camera\n behaviour on three shots that each want their own.\n\nSo name the lens, the framing, the move, and **the moment the move resolves**.\n\n---\n\n## 6. The hard constraints\n\n### Durations are even numbers only, starting at six\n\n**6, 8, 10, 12, 14, 16, 18, 20.** There is no 5-second LTX clip and no odd duration of any length.\nAsking for 7s is not a rounding matter — that generation does not exist.\n\n**And the long end is 1080p-and-below only.** At 1440p and 4K the ceiling drops to **6, 8 or 10**.\n\nfal's own default is `auto`, which lets the model pick the length from the described action.\n**Slates always sends an explicit length instead**, so what you choose is what you are billed for.\nChoose the length the beat needs.\n\n### Aspect ratios: 16:9 and 9:16, and nothing else\n\nThe narrowest set in the catalogue alongside Veo. Square, 4:5 and 21:9 are not available on this\nmodel at any resolution.\n\n### Frames, not references\n\nLTX takes a **start frame** and an **optional end frame** (which generates a transition between the\ntwo). It has **no reference-to-video endpoint at all** — no identity references, no style\nreferences, no environment references, no reference video, no reference audio.\n\n**For character consistency across separate shots, use MiniMax H3 or Kling.** Within a single LTX\ngeneration, use multishot instead — that is precisely the gap it fills.\n\nIn image-to-video, **do not cut away from the opening frame too early.** You have paid for that\nframe; let it play before the first move.\n\n### Do not ask for text on screen\n\nNeither the spelling nor its stability from frame to frame can be relied on. Signage, labels,\ncaptions and lower-thirds belong in post.\n\n---\n\n## 7. Audio is free here, and that changes the routing\n\nNative synchronised audio is **included at every resolution on both seats**, with no surcharge and\nno toggle that costs money — unlike Kling, where sound is a paid dimension. A 6-second 1080p LTX\nclip **with sound** is 39 credits.\n\nCombined with 1080p at $0.13/s — the cheapest native 1080p second in Slates — this makes LTX **the\ncoverage seat**: the one to reach for when the job is many takes rather than one hero shot, when a\nsequence needs its own sound, or when the credit budget is the binding constraint.\n\nRoute away from it when you need identity references (H3, Kling), a ratio other than 16:9 or 9:16\n(Seedance, Kling), or authored multi-layer audio direction (H3).\n",
|
|
@@ -35,7 +35,7 @@ export const SKILLS = {
|
|
|
35
35
|
"slates-script-craft": "---\nname: slates-script-craft\ndescription: Write or revise script passages, develop distinct openings and bridges, and compare section variations while preserving the user's format, voice and fixed material. This is writing craft, not a request to generate media.\n---\n\n# Script craft and variations\n\nWork in the user's document. Read its revision, requested passage and neighboring context. Keep supplied facts, deliberate cadence and fixed sections intact. A script may be silent, a conversation, one continuous sentence, independent scenes, or any mixture. These are tools to choose from, not required stages.\n\n<!-- @evidence: script-craft-20260922 sc-event sc-proof sc-exchange sc-callback sc-modular sc-bridge sc-offer sc-flow -->\n\n## Opening, argument and payoff\n\n| Technique | Evidence | What it does | Reach for · skip | Say |\n|---|---|---|---|---|\n| `sc-event` | Observed creative pattern; conversion unmeasured | Start with an event or consequence, including sound or silence. | Useful when the product can participate; skip spectacle unrelated to its promise. | `Keys slide toward the table edge; the tray catches them.` |\n| `sc-proof` | Observed demonstration pattern | Show the specific claim being tested. Speech may direct attention to the visible evidence. | Useful for observable behavior; skip claims the demonstration cannot establish. | `Watch the rim.` |\n| `sc-exchange` | Observed multi-speaker pattern | Let another speaker question, react or misunderstand. Preserve the answering context. | Useful for objections and comedy; do not isolate a dependent answer. | `A: You bought a tray for that? B: Look where my keys used to land.` |\n| `sc-callback` | Observed repeated-character comedy | Repeat deliberately, escalate, then resolve or change the meaning. | Useful for recognition and payoff; skip repetition without a purpose. | `The same searching hand finally reaches straight for the tray.` |\n| `sc-modular` | Scoped house-format technique | Make selected passages self-contained so they can move independently. | Useful for reorderable demonstrations; do not flatten continuing dialogue. | `At the door, it catches the keys. On the desk, it holds the loose change.` |\n| `sc-bridge` | Variation craft synthesis | Vary an opening together with any transition it requires. | Check pronouns, promise, reveal order and offer; preserve the chosen body. | `Where do your keys land? Mine used to land wherever my hand stopped. Now they land here.` |\n| `sc-offer` | Claim-control synthesis | Make the next action understandable and supported by the brief. | Use supplied destinations and terms; never invent price, savings, scarcity or guarantees. | `See the available finishes.` |\n| `sc-flow` | Spoken-writing synthesis | Clarify subject, action and causal connection before removing stylistic patterns. | Keep intentional rhythm and jokes; skip mechanical fragmenting. | `Put your keys here when you come in.` |\n\n## Distinct openings and compatible bridges\n\nChange the idea: an event, question, objection, proof, audience situation or reveal. Merely swapping adjectives is not a useful comparison. Name what stays fixed for this operation. A dependency belongs in the selected passage: if an opening changes what “that” means, include its bridge in the version.\n\nRead each candidate as a complete piece with the same body. Check unanswered promises, introduced speakers, incompatible offers and repeated reveals. Suggestions remain editable; no required Hook/Body/CTA fields.\n\nUse `slates_get_script_document`, `slates_get_script_sections` and revision-checked `slates_update_script_document` / `slates_update_script_section`. Save versions before switching. Preview one requested combination before materializing it; never expand every possible combination automatically. Reference substitutions are explicit IDs, not name replacements in prose. Keep voice retention deliberate.\n\n## Spoken flow and pacing\n\nPrefer a concrete actor doing something over abstract benefit language. “Seamlessly elevate your daily carry” becomes “Put your keys here when you come in.” Connect causes where needed: “I put them down, then forget where” is clearer than mechanically shortening it to “Keys. Gone. Again.”\n\nPreserve the user's or reference's cadence when it carries character, comedy or comprehension. A repeated sentence or triplet is not inherently an error. Personal voice preferences apply only to the person who supplied them. Never invent testimonials or measurable results.\n\nRead the canonical fit analysis supplied with the shots. Its corpus estimate uses total ad runtime, including silence, and an opt-in register sample. It is not measured articulation speed; the observed maximum is not a universal human limit. Plan for pauses, reactions and sound. Once a voice/video take exists, its measured performance governs the cut. Model clip duration, estimated script duration and actual speech duration are separate facts.\n\n## Apply the requested scope\n\nFor suggestions, propose each replacement with `slates_update_script_suggestions` (action `create`), quoting the exact words it replaces at the revision you read; the creator accepts or dismisses it in the document, and `slates_get_script_suggestions` reports what became of it. For an explicit edit request, apply the scoped edit and read it back; do not add an approval ceremony. On a stale revision, reread and preserve both authors' changes. Do not replace the whole script to change one opening.\n\nHeadings and directions are non-spoken metadata. Shots are optional production bindings. Script-driven recipes compile the active words; custom prompts retain their bytes and need a visible alignment review. Existing takes remain historical media. Writing, switching versions and importing templates do not generate anything. Load production and cost guidance only when production is requested.\n",
|
|
36
36
|
"slates-shot-variety": "---\nname: slates-shot-variety\ndescription: Diagnose unintended visual sameness across a shot sequence while preserving deliberate repetition, continuing performance and the user's chosen format.\n---\n\n# Visual rhythm across a sequence\n\n`slates_list_shots` supplies distributions and repeated runs from authored shot fields. Read across the sequence, then decide whether repetition serves the intended effect. A dominant bucket is a question, not a defect or generation barrier.\n\n## Compare neighboring cuts\n\nLook at framing, camera behavior, duration, subject distance, location and cast. Change the dimension that carries the meaning of the next beat. A wider view may reveal geography; a close view may make a small action legible. Do not add camera motion simply because another shot is static.\n\nRepeated frames can establish a joke, a comparison or a calm observational register. Recurring people and locations can carry a conversation. A later change often works because the earlier pattern held. State that purpose briefly when a count flags an intentional choice.\n\n## Re-cut only for a reason\n\nMerge when performance and picture should continue together. Split when the image needs to change while speech continues, or when the intended read needs another placement. Preserve sentence continuity and references across the split. Price the resulting requests; do not assume splitting is free or merging is cheaper.\n\nThe script fit signal derives from an opt-in ad corpus measured over whole runtime. Above-sample pace deserves inspection; it does not prove a line impossible. Measure the actual spoken take when available, including pauses and reactions. No fixed cut length or shot count is a universal rule.\n\nMulti-shot generations can contain several visual cuts. Compare their internal rhythm as well as the boundaries between generated clips. The app counts only authored information: an unknown framing bucket is missing description, not evidence about the pixels.\n\nThis guide improves deliberate visual decisions. It does not measure taste, conversion or the quality of a finished performance. Use `slates-script-craft` for the argument, exchanges and setup/payoff that the picture supports.\n",
|
|
37
37
|
"slates-storyboard-from-script": "---\nname: slates-storyboard-from-script\ndescription: Put supplied script or treatment into an editable Slates document and bind requested passages to production shots. Preserve the words and structure; generate media only within the user's requested scope.\n---\n\n# Script into editable production\n\nRead the existing document and its revision before writing. Preserve supplied words, speaker context, headings, non-spoken direction and any explicit shot list. A heading formats a document; creating a production scene is a separate choice. Paragraph count does not determine shot count.\n\n## Save the words once\n\nUse `slates_get_script_document` and `slates_update_script_document` for ordered, revision-checked text and structure edits. Scene strings own spoken words. Paragraph blocks hold offsets and marks; headings/directions own only their non-spoken text. Do not keep an independently editable master body beside the document.\n\nCreate a board or scene only when needed for the requested destination. Use the current project unless the user asks for another. Writing a script needs no image, character record or generation.\n\n## Bind production where wanted\n\nSelect an intended production passage and use the script-to-shot operation. It can make, attach, extend, split or merge according to the existing bindings. Read back the resulting shots and ranges. A silent shot is equally valid and needs no fabricated dialogue.\n\nA new document-created recipe is script-driven: its prompt compiles from the active passage. Text with no speaker and no delivery goes to the model as written (most script text is action); a speaker, VO included, or a delivery note makes it quoted speech. Keep action, delivery, framing and references in their own controls. Do not write the dialogue a second time in a custom prompt. When the creator explicitly chooses a custom prompt, preserve its bytes and review alignment after script changes.\n\nShots file through the existing filing service. Pass the scene or frame destination when known and use the returned shot codes. Keep recurring identities in existing Library references; a working speaker name does not require a placeholder character.\n\n## Review without imposing a format\n\nRead composed requests, actual reference roles and the current quote. Explain only consequential decisions not already visible in the document or shot. Variety counts are suggestions: intentional repeated frames, continuing sentences and recurring cast may be exactly right. `slates-script-craft` covers passages and versions; `slates-shot-variety` covers deliberate visual rhythm.\n\nIf generation is requested, follow `slates-cost-discipline` for the exact set. On an uncertain timeout inspect existing generation IDs before retrying. Preserve takes and inspect the landed results. Named cuts keep independent edits separate; writing alone does not require a cut, export or paid call.\n\n<!-- @inject:decision-log -->\nRecord production choices in the editable shot fields. Explain only consequential judgments the user did not specify and no field already records: for example, why a particular light or performance register supports the brief. Do not repeat the shot list in prose or turn this explanation into an approval gate. Follow the separate generation authorization policy before spending.\n<!-- @end:decision-log -->\n",
|
|
38
|
-
"slates-style-prompting": "---\nname: slates-style-prompting\ndescription: Use when the user asks for a visual style (\"make it anime\", \"painterly look\", \"like a Pixar film\"), or when a style has to hold across several shots. Covers how photoreal, anime, painterly and 3d-render are prompted DIFFERENTLY per model, and the style-routing recipe (reference-first, styled start-frame → i2v).\n---\n\n# Per-style prompting (photoreal · anime · painterly · 3d-render)\n\nThe style library (`slates_create_style` / the app's style ids) defines what each style IS. This guide is how to PROMPT each style per model. Derived from `research/style-prompting-research.md` (second-brain) — claims marked *(hypothesis)* are untested; don't present them to users as fact.\n\n## The four ground rules (all styles)\n\n1. **Assign references where they contribute.** Describe the scene with inline bindings, such as \"the woman from image 1, lit and graded like image 2.\" Preserve an existing scene reference when its look should stay. A look-only reference may need light and exposure described for the new scene; prose and references can work together.\n2. **Use each model’s language, without imposing a fixed prompt template:**\n - **Nano Banana 2** — narrative prose; the style is the opening framing of the sentence (\"A hand-drawn 2D anime cel illustration of…\"), never a comma tag.\n - **Seedance 2.0** — the 8-part formula reserves \"visual style\" (slot 6) and \"image quality\" (slot 7). One clause each. Don't scatter style words through the action text.\n - **Kling V3** — prose scene direction; style rides the lighting/style tail of Scene → Subject → Action → Camera → Lighting/Style. Tag soup underperforms badly.\n3. **Keep the intended look consistent across shots.** Reuse relevant references and stable descriptions, adapting the wording to each scene.\n4. **A styled start frame is one video control.** Generate it with the image seat suited to the brief, then describe the motion. Preserve its look unless the user wants the light or grade to change.\n\nNever stack style buzzwords (\"ARRI ALEXA, 35mm, film grain, depth-of-field mastery…\"). One or two register tokens maximum — piles of specs dull the image.\n\n## Photoreal\n\n- **NB2:** never the literal word \"photorealistic\". Describe *a real photograph*: natural skin texture and imperfection, motivated lighting, one lens/film register (\"shot on a 50mm, soft window light\"). Photographic composition terms: wide-angle / macro / low-angle.\n- **Seedance:** put \"sharp focus, natural color, high detail\" in the image-quality slot and always include a lighting clause. Keep motion slow and coherent — fast/burst action is the #1 quality killer and reads most fake in photoreal.\n- **Kling:** the photoreal-PEOPLE lane — convincing acting, dialogue, lip-sync. It breaks on close-up hands, fine fluids, and crowds beyond ~5 faces: route those beats to Seedance or reframe.\n- **Faces on Seedance:** photoreal humans trigger the face-tier routing (AI face vs consented real face — see slates-prompting-seedance §Faces). Set the face flags honestly; never skip them to save credits.\n\n## Anime\n\n- **NB2:** open with the medium — \"A hand-drawn 2D anime cel illustration of…\" — then normal narrative Subject/Setting/Action. Clean line art, flat-shaded color, expressive eyes. NB2 has no negative prompt: phrase exclusions positively (\"flat cel shading with uniform focus\", not \"no depth of field\").\n- **Seedance:** visual-style slot = \"2D anime style, clean line art, flat cel shading\". The slow/coherent-motion preference still applies — burst sakuga actions are the same instability trap as in photoreal.\n- **Kling:** weakest anime lane (its strength is live-action-like acting); expect style drift on long prose-only shots. Prefer ground rule 4: NB2 anime start-frame → i2v with a motion-only prompt. *(hypothesis: refs hold Kling's anime better than prose — verify before promising.)*\n- Anime faces drift under multiple references faster than photoreal — the named-entity two-sheet doctrine applies unchanged.\n\n## Painterly\n\n- **NB2:** medium + technique in the style framing: \"digital concept-art painting, visible brushwork, painted edges\". At most ONE school/era register (\"classic gouache illustration\") — a register, not an artist-name pile.\n- **Video:** the least-supported style lane. Use ground rule 4 (painterly NB2 frame → i2v, motion-only prompt) and expect some cleanup of painterliness over the clip *(hypothesis — set user expectations, don't promise a perfectly painterly clip)*.\n- Camera language still applies — painterly ≠ static; \"slow push-in\" works the same.\n\n## 3D render\n\n- **NB2:** name the lineage register in the style framing: \"stylized 3D render, soft global illumination, subsurface skin\". Lighting vocabulary (GI, rim light) is unusually load-bearing for the 3D read.\n- **Seedance:** the physics/effects lane flatters 3D content — visual-style slot \"stylized 3D animation\", image-quality slot \"clean render, high detail\".\n- **Kling:** same start-frame preference as anime.\n- *(hypothesis)* An engine token (\"Unreal Engine 5 render\") may help NB2; if used, ONE token, style slot only — never on Seedance where spec-stuffing hurts.\n\n## Routing recipe (what to actually do)\n\n1. Style reference available → attach it, rely on inherit. Done.\n2. No reference, image request → styled NB2 prose per the section above.\n3. No reference, video request → NB2 styled start-frame first, then i2v with motion-only prompt. Direct styled text-to-video is the fallback when a start frame doesn't fit (e.g. dialogue-first Kling shots).\n4. Multi-shot run → byte-identical style clause per shot + shared references.\n",
|
|
38
|
+
"slates-style-prompting": "---\r\nname: slates-style-prompting\r\ndescription: Use when the user asks for a visual style (\"make it anime\", \"painterly look\", \"like a Pixar film\"), or when a style has to hold across several shots. Covers how photoreal, anime, painterly and 3d-render are prompted DIFFERENTLY per model, and the style-routing recipe (reference-first, styled start-frame → i2v).\r\n---\r\n\r\n# Per-style prompting (photoreal · anime · painterly · 3d-render)\r\n\r\nThe style library (`slates_create_style` / the app's style ids) defines what each style IS. This guide is how to PROMPT each style per model. Derived from `research/style-prompting-research.md` (second-brain) — claims marked *(hypothesis)* are untested; don't present them to users as fact.\r\n\r\n## The four ground rules (all styles)\r\n\r\n1. **Assign references where they contribute.** Describe the scene with inline bindings, such as \"the woman from image 1, lit and graded like image 2.\" Preserve an existing scene reference when its look should stay. A look-only reference may need light and exposure described for the new scene; prose and references can work together.\r\n2. **Use each model’s language, without imposing a fixed prompt template:**\r\n - **Nano Banana 2** — narrative prose; the style is the opening framing of the sentence (\"A hand-drawn 2D anime cel illustration of…\"), never a comma tag.\r\n - **Seedance 2.0** — the 8-part formula reserves \"visual style\" (slot 6) and \"image quality\" (slot 7). One clause each. Don't scatter style words through the action text.\r\n - **Kling V3** — prose scene direction; style rides the lighting/style tail of Scene → Subject → Action → Camera → Lighting/Style. Tag soup underperforms badly.\r\n3. **Keep the intended look consistent across shots.** Reuse relevant references and stable descriptions, adapting the wording to each scene.\r\n4. **A styled start frame is one video control.** Generate it with the image seat suited to the brief, then describe the motion. Preserve its look unless the user wants the light or grade to change.\r\n\r\nNever stack style buzzwords (\"ARRI ALEXA, 35mm, film grain, depth-of-field mastery…\"). One or two register tokens maximum — piles of specs dull the image.\r\n\r\n## Photoreal\r\n\r\n- **NB2:** never the literal word \"photorealistic\". Describe *a real photograph*: natural skin texture and imperfection, motivated lighting, one lens/film register (\"shot on a 50mm, soft window light\"). Photographic composition terms: wide-angle / macro / low-angle.\r\n- **Seedance:** put \"sharp focus, natural color, high detail\" in the image-quality slot and always include a lighting clause. Keep motion slow and coherent — fast/burst action is the #1 quality killer and reads most fake in photoreal.\r\n- **Kling:** the photoreal-PEOPLE lane — convincing acting, dialogue, lip-sync. It breaks on close-up hands, fine fluids, and crowds beyond ~5 faces: route those beats to Seedance or reframe.\r\n- **Faces on Seedance:** photoreal humans trigger the face-tier routing (AI face vs consented real face — see slates-prompting-seedance §Faces). Set the face flags honestly; never skip them to save credits.\r\n\r\n## Anime\r\n\r\n- **NB2:** open with the medium — \"A hand-drawn 2D anime cel illustration of…\" — then normal narrative Subject/Setting/Action. Clean line art, flat-shaded color, expressive eyes. NB2 has no negative prompt: phrase exclusions positively (\"flat cel shading with uniform focus\", not \"no depth of field\").\r\n- **Seedance:** visual-style slot = \"2D anime style, clean line art, flat cel shading\". The slow/coherent-motion preference still applies — burst sakuga actions are the same instability trap as in photoreal.\r\n- **Kling:** weakest anime lane (its strength is live-action-like acting); expect style drift on long prose-only shots. Prefer ground rule 4: NB2 anime start-frame → i2v with a motion-only prompt. *(hypothesis: refs hold Kling's anime better than prose — verify before promising.)*\r\n- Anime faces drift under multiple references faster than photoreal — the named-entity two-sheet doctrine applies unchanged.\r\n\r\n## Painterly\r\n\r\n- **NB2:** medium + technique in the style framing: \"digital concept-art painting, visible brushwork, painted edges\". At most ONE school/era register (\"classic gouache illustration\") — a register, not an artist-name pile.\r\n- **Video:** the least-supported style lane. Use ground rule 4 (painterly NB2 frame → i2v, motion-only prompt) and expect some cleanup of painterliness over the clip *(hypothesis — set user expectations, don't promise a perfectly painterly clip)*.\r\n- Camera language still applies — painterly ≠ static; \"slow push-in\" works the same.\r\n\r\n## 3D render\r\n\r\n- **NB2:** name the lineage register in the style framing: \"stylized 3D render, soft global illumination, subsurface skin\". Lighting vocabulary (GI, rim light) is unusually load-bearing for the 3D read.\r\n- **Seedance:** the physics/effects lane flatters 3D content — visual-style slot \"stylized 3D animation\", image-quality slot \"clean render, high detail\".\r\n- **Kling:** same start-frame preference as anime.\r\n- *(hypothesis)* An engine token (\"Unreal Engine 5 render\") may help NB2; if used, ONE token, style slot only — never on Seedance where spec-stuffing hurts.\r\n\r\n## Routing recipe (what to actually do)\r\n\r\n1. Style reference available → attach it, rely on inherit. Done.\r\n2. No reference, image request → styled NB2 prose per the section above.\r\n3. No reference, video request → NB2 styled start-frame first, then i2v with motion-only prompt. Direct styled text-to-video is the fallback when a start frame doesn't fit (e.g. dialogue-first Kling shots).\r\n4. Multi-shot run → byte-identical style clause per shot + shared references.\r\n",
|
|
39
39
|
"slates-ugc-influencer-ad": "---\nname: slates-ugc-influencer-ad\ndescription: Direct a creator-style spoken performance when the brief calls for an ordinary camera-facing person or exchange. Use for performance and phone-camera craft, not as a universal rule for ads.\n---\n\n# Creator-style performance\n\nRead `slates-script-craft` for the script and opening/bridge versions. Match the requested creator, audience and reference register. Phone footage, quiet polish and cinematic treatment are choices; no evidence here establishes one as universally highest-converting.\n\n## Direct a person in a place\n\nWrite an observable activity and a speaking intention: showing a worn handle, answering a friend, demonstrating a catch, interrupting a task. Small actions can make a close shot legible: a weight shift, a glance toward the other person, a hand taking an object's weight. Larger actions need space and framing that accommodates them.\n\nFor an ordinary phone register, describe concrete light and surroundings: a window from one side, a practical lamp, a room with ordinary possessions. Avoid generic praise words. Preserve the supplied person's appearance and product details through references, not invented claims.\n\nDescribe camera handling physically when it matters: a hand supports the phone against the table edge; the frame sags and is corrected once. Static framing is also valid. Do not force handheld motion or imperfections into a brief that asks for something else.\n\n## Speech, interaction and sound\n\nOne person or several may carry the piece. Give reactions and answers their antecedents. A continuing sentence may cross an edit; an independent module should establish its own subject. Repetition can be a callback.\n\nSpeech may point at what the viewer sees when demonstrating a claim. It need not compete with the picture for novelty on every line. Keep pauses, listening and the intended register; a fixed words-per-second formula cannot establish the performance's duration.\n\nMake audio intent explicit: dialogue, room sound, effects, score or silence. A silent demo is valid. Use current model guidance for audio support and reference syntax. Plan captions or text only where the actual editing surface supports them; do not promise an unverified rendering feature.\n\n## References and production choices\n\nSelect the identity, voice, location and direct images deliberately. A replacement presenter need not replace a separately retained voice. A changed location does not automatically replace a first-frame image. Inspect the composed request to see which references remain.\n\nA plate-first approach is useful when the composition must be approved before motion, but it is not a prerequisite for all video. Choose single-take or multiple-cut production according to the intended performance and current capabilities. Do not copy vendor duration, resolution or reference limits into this guide.\n\nBefore spending, follow `slates-cost-discipline` for the selected set. Inspect errors and existing job state; a refusal or timeout does not authorize another charge. Check the actual clip's picture and sound against the brief, preserve earlier takes, and trim or rearrange when that solves the issue without regenerating. Exported playback, not a successful process exit, establishes the result.\n",
|
|
40
40
|
"slates-vision-feedback-loop": "---\nname: slates-vision-feedback-loop\ndescription: Lower-level utility skill for any Slates workflow that needs to \"generate, look at the result, refine, regenerate.\" Defines the standard inline-vision pattern. Other Slates skills compose this. Use when generating images and you need to confirm they match the brief before moving on, or when the user asks to \"iterate\" on an image.\n---\n\n# Vision feedback loop — Slates utility skill\n\nSlates returns generated images inline as base64. You see the actual pixels. Use that — don't trust prompt-following blindly.\n\n## Asset codes are your shared vocabulary with the user\n\nEvery asset in Slates has a short stable code (e.g. `IMG-A12`, `VID-V3`, `AUD-S1`) and a label derived from its prompt (e.g. `Beach Sunset`). These are visible in the gallery as a corner badge on each thumbnail. **Always refer to assets by their code in chat** so the user can match what you're saying to a specific card in their gallery.\n\n- ✅ \"I'm using **IMG-A12 — Beach Sunset** as the first frame. The second-frame candidate **IMG-A15** has the right composition but warmer light — want me to use that one instead?\"\n- ❌ \"I'm using the beach sunset image...\" (user has four beach sunset variants — which one?)\n- ❌ \"I'm using asset `7a3f9e4b-...`\" (UUIDs aren't readable; user can't match to a badge)\n\nThe code is the FORMAL reference. The label is human texture. Use both: `IMG-A12 — Beach Sunset`.\n\n## Vision tools at your disposal\n\n- `slates_get_asset_image` — pull one image into context. Returns its code+label.\n- `slates_get_assets_batch` — pull up to 8 images in one call. Use when picking from a candidate set; cheaper than N individual fetches.\n- `slates_get_asset_video_frames` — extract N keyframes (default 3) from a video and inline them as JPEGs. You can't see video natively; this is how you \"look at\" a clip before refining its motion prompt.\n\n## Pre-flight is automatic on the gen tools\n\n`slates_generate_video`, `slates_generate_motion_transfer`, and `slates_generate_lip_sync` now show you their reference assets **inline** on the confirm response. You don't need to fetch them yourself — but you DO need to look at what comes back, revise the prompt if the references suggest a different motion/framing, and only then re-call with `confirm=true`.\n\n## 🔴 The still-gate — never animate a bad frame\n\n<!-- @inject:still-gate -->\n**A visible defect in the still is already a STOP.** Do not animate it. Fix the frame first, then move to motion — and go to motion only when the crop passes the still scan and you genuinely need movement to confirm an uncertain edge, reflection, or object.\n\nThis is a **cost** rule as much as a craft rule: a 1080p/10s premium video generation costs many multiples of an image re-roll, and video is where a defect stops being fixable. Anything wrong in the still gets worse in motion — soft geometry mushes, broken-but-plausible objects fall apart, oily textures start crawling. **Animating a known-bad frame is the single most expensive mistake in the pipeline.** Re-rolling the image is the cheap move; re-rolling the video is not.\n<!-- @end:still-gate -->\n\n## The pattern\n\n1. **Generate.** Call `slates_generate_image` with a prompt. The result is in your context as an image content block.\n2. **Evaluate on TWO axes — they are different questions:**\n - **Brief-conformance** — what did the user actually want? Are the elements right? Composition? Lighting? Subject identity?\n - **Defects** — run the slop rubric below. *A frame can match the brief perfectly and still be slop that mushes the moment it moves.* Checking only the first axis is how a bad frame reaches an expensive video call.\n3. **One of three outcomes:**\n - **Right** → save it (bind to a frame, character slot, etc.) and move on.\n - **Close, but adjustable** → refine with a specific delta, regenerate **once**.\n - **Wrong direction** → ask the user before regenerating. Don't burn credits on prompt-thrashing.\n\n## The defect rubric — five slop tells\n\n| Tell | What it looks like | Why it matters downstream |\n|---|---|---|\n| **Light with no transitions** | Flat-black pits instead of a shadow ramp; light that stops rather than falls off | Transfers onto every character or object added into that plate later |\n| **Broken-but-plausible objects** | Crates, railings, hardware, mechanisms you can *almost* read but that don't resolve | Turn to mush in motion, and the model multiplies them |\n| **Local logic breaks** | An effect present in only part of the frame — rain scratching one corner, wet ground under one figure | The video model's physical logic breaks along with it |\n| **Oily textures** | Soapy, licked-smooth surfaces that have lost their material identity | Reflections crawl in motion; the plate can't hold continuity |\n| **Too perfect** | A soft light on the face that nothing in the scene could cast, the subject sharper and cleaner than everything around them, every region exposed to be readable, colour pushed warm and saturated | It reads as a subject pasted onto a location, and every shot built from the plate inherits the studio look. Fix it in words: `slates-cinematic-look` |\n\n### Per-model accents — check the one you actually used\n\n- **Nano Banana Pro** (`nano-banana-pro`) — ruler-straight symmetry, everything parallel and square, flat even light, pretty but staged/stock, textures reading as 3D render rather than photograph. **It hyperbolizes every edit**: ask for graffiti on one wall and the whole location gets tagged.\n- **GPT Image** (`gpt-image-2-5-flare`, `gpt-image-2-5-sunburst`) — microcontrast to the ceiling, hard halos on every edge, no depth or bokeh, white balance pulled warm until the frame yellows, plastic licked-smooth materials. Worst tell: **one sickly texture pattern laid over the entire frame**. ⚠️ Catalogued on `gpt-image-2`, which 2.5 replaced on 2026-09-09 — an accent is a per-model observation, so treat this as a prior to check rather than a finding, and correct it here the first time a 2.5 frame disagrees.\n\n> ⚠️ These are accents for **`nano-banana-pro`** and the **GPT Image** line specifically. `nano-banana-2` is a **different model** (Gemini 3.1 Flash Image vs NB Pro's Gemini 3 Pro Image) and we have **no evidence** about its accent. Do not inherit one — say nothing rather than warn about a failure mode you can't substantiate. That caution applies to the GPT Image entry above too: it was measured on `gpt-image-2`, not on either 2.5 seat.\n\n## Where the fault lives — triage before you change anything\n\nWe say \"one specific delta per regeneration\" but that only helps once you know *which* variable to move. Diagnose first:\n\n| Visible pattern | Diagnosis | Fix |\n|---|---|---|\n| The defect exists in the source asset, or stays tied to the same feature when the direction changes | **Source asset** | Fix the sheet / plate, not the prompt |\n| Source is clean, and the defect changes when only the suspect motion clause changes | **Motion direction** | Fix the prompt |\n| Controls conflict, or the failure follows neither variable | **Inconclusive** | Narrow the test — change less, not more |\n\n**Review routes; it is not pass/fail.** Geography melts → fix the location. Identity drifts → fix the character sheet. Assets are sound but the action is wrong → fix the video direction. Wrong idea entirely → reopen the brief with the user.\n\n**Correct the earliest broken handoff.** Polishing a downstream symptom hides the source and guarantees it resurfaces in the next shot built from the same asset.\n\n## Baseline hygiene — isolate the variable you're testing\n\nWhen the **character** is the question, keep the location out of it: test on a plate that already holds its own geometry, depth, materials, and light. **A broken plate gives every character failure a second plausible cause**, and you will spend re-rolls deciding which one you're looking at. The same applies in reverse — test a plate empty before you populate it.\n\n## Refinement rules\n\n- **One specific delta per regeneration.** Don't change five things at once — you won't know what helped.\n- **Rewrite the FULL prompt on every iteration — never a diff, never a fragment.** Change one decision, then re-emit the whole prompt so every slot still agrees with every other slot. This composes with the rule above rather than replacing it: *one delta* governs **what changes**, *full rewrite* governs **how you re-emit it**. A patched fragment leaves the old slots stale and silently contradicting the new one.\n - On **Seedance**, a re-emit must keep the `Shot N` structure intact — see `slates-prompting-seedance`.\n - **Exception — Omni Flash Edit.** Long prompts documentedly destroy its fidelity. There the rule inverts: one short instruction plus *\"Keep everything else the same.\"*\n- **Anchor with references.** If the result drifted from the user's intent, attach the *previous best* generation as a reference image alongside the original brief.\n- **Use `slates_get_asset_image`** to pull a previously-generated image back into context if you need to compare against a fresh generation.\n- **Use `slates_edit_image`** for surgical tweaks instead of full regeneration when ~90% of the image is right — `sourceAssetId` = the asset, `prompt` = the change only. Edits preserve composition and identity; full regen rolls the dice. Recipe: `slates-edit-and-iterate`.\n\n## Cost discipline\n\n- Track total credits spent across the loop. Surface to the user every 3 iterations.\n- Stop after 3 failed iterations on the same prompt — escalate to the user with what you tried and what's not working. The slot machine never converges.\n- For a high-cost generation — anything past the confirm gate in `slates-cost-discipline` — confirm before *every* attempt, not just the first.\n\n## When to break the loop\n\n- The user said \"good enough\" or \"ship it.\" Stop iterating.\n- You've burned >5 generations on one frame. Hand back and ask.\n- The user changes brief mid-loop. Treat it as a new brief, not a continuation.\n\n## Voice when narrating to the user\n\nTight, observational, no editorializing.\n- ✅ \"Frame 2 has the wrong lighting direction — back-lit instead of side. Regenerating with side light.\"\n- ❌ \"I notice that the lighting in frame 2 isn't quite what we were going for. I'll go ahead and try again with a different approach.\"\n",
|
|
41
41
|
};
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@slatesvideo/shared",
|
|
3
|
-
"version": "0.7.
|
|
3
|
+
"version": "0.7.2",
|
|
4
4
|
"description": "Shared operations layer for the Slates MCP server and CLI: auth, cloud/desktop clients, and the single tool surface both consume. Most users want @slatesvideo/mcp-server or @slatesvideo/cli instead.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|
|
@@ -5,4 +5,4 @@
|
|
|
5
5
|
- **Lens name plus effect** — `200mm telephoto`, `peaks loom huge behind her and melt into soft shapes`.
|
|
6
6
|
- **Name every garment and close the foreground.** Omissions invite reference leakage or invented props.
|
|
7
7
|
Bind references inline. A scene reference owns the grade; for a look-only reference, write the new scene's light. References are optional. For owned-frame edits, describe only the change and what stays.
|
|
8
|
-
<!-- slates-only -->Use `slates-cinematic-look` with a technique ID or section query for more.<!-- /slates-only -->
|
|
8
|
+
<!-- slates-only -->Use `slates-cinematic-look` with a technique ID or section query for more.<!-- /slates-only -->
|