@slatesvideo/shared 0.7.2 → 0.7.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/clients/cloud.d.ts +4 -0
- package/dist/clients/cloud.js +11 -3
- package/dist/index.d.ts +1 -0
- package/dist/index.js +1 -0
- package/dist/manual/content.d.ts +1 -1
- package/dist/manual/content.js +1 -1
- package/dist/operations/index.d.ts +12 -13
- package/dist/operations/index.js +158 -133
- package/dist/operations/surface.d.ts +6 -2
- package/dist/operations/surface.js +29 -5
- package/dist/prompts/agent-doctrine.d.ts +4 -4
- package/dist/prompts/agent-doctrine.js +17 -28
- package/dist/prompts/guide-discovery.d.ts +23 -0
- package/dist/prompts/guide-discovery.js +39 -0
- package/dist/prompts/guide-retrieval.js +1 -1
- package/dist/prompts/model-capabilities.d.ts +8 -9
- package/dist/prompts/model-capabilities.js +11 -51
- package/dist/prompts/model-facts.d.ts +2 -2
- package/dist/prompts/model-facts.js +15 -26
- package/dist/prompts/partials.generated.js +6 -3
- package/dist/prompts/prompting-tips.d.ts +1 -1
- package/dist/prompts/prompting-tips.js +21 -63
- package/dist/prompts/search-terms.d.ts +3 -0
- package/dist/prompts/search-terms.js +24 -0
- package/dist/skills/content.js +36 -37
- package/dist/skills/metadata.d.ts +7 -0
- package/dist/skills/metadata.js +29 -0
- package/exports/slates-chatgpt-images/generated/SKILL.md +7 -1
- package/exports/slates-chatgpt-images/generated/slates-chatgpt-images.skill +0 -0
- package/exports/slates-prompt-builder/generated/SKILL.md +28 -16
- package/exports/slates-prompt-builder/generated/reference-character.md +12 -13
- package/exports/slates-prompt-builder/generated/reference-content-policy.md +2 -2
- package/exports/slates-prompt-builder/generated/reference-gpt-image-2-5.md +191 -0
- package/exports/slates-prompt-builder/generated/reference-kling.md +32 -11
- package/exports/slates-prompt-builder/generated/reference-nano-banana.md +24 -6
- package/exports/slates-prompt-builder/generated/reference-omni-flash.md +65 -0
- package/exports/slates-prompt-builder/generated/reference-seedance-2-5.md +362 -0
- package/exports/slates-prompt-builder/generated/reference-seedance.md +34 -4
- package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +77 -23
- package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
- package/package.json +2 -1
- package/skills/_partials/blender-action-curves.md +24 -0
- package/skills/_partials/iteration-diagnosis.md +5 -0
- package/skills/_partials/model-routing.md +35 -0
- package/skills/_partials/seedance-25-timestamps.md +2 -2
- package/skills/_partials/still-gate.md +2 -2
- package/skills/_partials/thresholds.md +1 -1
- package/skills/slates-blocking-to-prompt.md +15 -13
- package/skills/slates-camera-language.md +45 -7
- package/skills/slates-character-identity.md +8 -6
- package/skills/slates-chatgpt-images.md +7 -1
- package/skills/slates-cinematic-look.md +1 -1
- package/skills/slates-content-policy.md +4 -6
- package/skills/slates-cost-discipline.md +18 -12
- package/skills/slates-dialogue-blocking.md +6 -6
- package/skills/slates-direct-response-ad.md +1 -1
- package/skills/slates-edit-and-iterate.md +12 -4
- package/skills/slates-model-selection.md +82 -90
- package/skills/slates-one-prompt-film.md +1 -1
- package/skills/slates-previs-blocking.md +44 -13
- package/skills/slates-project-organization.md +2 -2
- package/skills/slates-prompting-elevenlabs.md +4 -4
- package/skills/slates-prompting-flux-2-max.md +2 -3
- package/skills/slates-prompting-gpt-image-2-5.md +2 -2
- package/skills/slates-prompting-inworld-tts.md +174 -174
- package/skills/slates-prompting-kling-v3.md +11 -9
- package/skills/slates-prompting-lip-sync.md +15 -15
- package/skills/slates-prompting-ltx-2-5.md +5 -6
- package/skills/slates-prompting-minimax-h3.md +11 -11
- package/skills/slates-prompting-motion-transfer.md +8 -8
- package/skills/slates-prompting-nano-banana-2.md +8 -4
- package/skills/slates-prompting-omni-flash.md +9 -9
- package/skills/slates-prompting-seed-audio.md +24 -4
- package/skills/slates-prompting-seedance-2-5.md +40 -30
- package/skills/slates-prompting-seedance.md +4 -4
- package/skills/slates-prompting-seedream-5-lite.md +6 -6
- package/skills/slates-restyle-from-blocking.md +2 -2
- package/skills/slates-script-craft.md +1 -1
- package/skills/slates-shot-variety.md +1 -1
- package/skills/slates-storyboard-from-script.md +1 -1
- package/skills/slates-style-prompting.md +56 -54
- package/skills/slates-ugc-influencer-ad.md +1 -1
- package/skills/slates-vision-feedback-loop.md +118 -110
- package/skills/slates-prompting-veo-3.md +0 -224
|
@@ -5,19 +5,22 @@
|
|
|
5
5
|
// per-model skills, so the TS consumers and the markdown consumers can
|
|
6
6
|
// no longer disagree. Edit the partial, not this file, not the skills.
|
|
7
7
|
export const PARTIALS = {
|
|
8
|
+
"blender-action-curves": "## Read animation curves from the active action layout\n\nBlender 5 uses layered actions: curves belong to the channelbag for `animation_data.action_slot`, inside each layer's strips. A direct `action.fcurves` lookup failed on Blender 5.2.1 in the 2026-08-28 blocking run. Feature-detect the layout before changing interpolation or noise; an unanimated object can legitimately have no curves.\n\nThe snippets below use this small Blender-side iterator. It runs inside Blender; no add-on code is imported into the MCP package.\n\n```python\ndef action_curves(datablock):\n anim = getattr(datablock, \"animation_data\", None)\n action = getattr(anim, \"action\", None)\n if action is None:\n return\n if hasattr(action, \"fcurves\"):\n yield from action.fcurves\n elif getattr(anim, \"action_slot\", None) is not None:\n for layer in action.layers:\n for strip in layer.strips:\n if hasattr(strip, \"channelbag\"):\n bag = strip.channelbag(anim.action_slot)\n if bag is not None:\n yield from bag.fcurves\n```\n\nUse the datablock that owns the keyed property: the curve data for `eval_time`, the object for location and rotation, the camera data for lens. Confirm a named channel exists before assuming a keyframe operation created it.",
|
|
8
9
|
"cinematic-card": "**For a photographic look, use only what this frame needs.** Image models default to clean, evenly lit and fully exposed. Describe what the camera sees, not just gear or mood:\n- **Inspect every reference first.** Write its grade and imperfections in words: darkness, contrast, muddy or true blacks, colour, softness/noise, subject separation. Never grade cleaner or brighter than the look reference unless asked.\n- **One light system** — `low sun behind her`, `her face falls into deep shadow`, `no light in front of her`.\n- **Visible exposure** — `the sky burns out to white`, `dense, slightly crushed shadows`.\n- **Lens name plus effect** — `200mm telephoto`, `peaks loom huge behind her and melt into soft shapes`.\n- **Name every garment and close the foreground.** Omissions invite reference leakage or invented props.\nBind references inline. A scene reference owns the grade; for a look-only reference, write the new scene's light. References are optional. For owned-frame edits, describe only the change and what stays.\n<!-- slates-only -->Use `slates-cinematic-look` with a technique ID or section query for more.<!-- /slates-only -->",
|
|
9
10
|
"cinematic-routes-short": "Two routes: describe a new frame, or change a frame you own. For a new scene, name references where you use them, write the look reference's grade and imperfections in plain words, then describe one light system, visible exposure, lens plus effect, every garment and a closed foreground. Use only what the shot needs. For your own plate, sheet, photo, footage or Blender render, say only what changes and what stays. Never use a released film frame as the edit base; use it as an art-direction brief for a new scene.",
|
|
10
11
|
"cinematic-tips-short": "Image models tend toward clean, evenly lit, fully exposed pictures. For a filmed look, describe what you see: one light source and its effect, a face almost in silhouette, a sky burned white. Name the lens and its visible effect together. Look at every reference first and describe its own darkness, contrast, colour, softness, noise and subject separation. Keep muddy blacks muddy; never clean up or brighten the reference's grade unless that is the change you want. Name every garment and exactly what is in the foreground. Use only what the shot needs; references are optional.",
|
|
11
12
|
"decision-log": "Record production choices in the editable shot fields. Explain only consequential judgments the user did not specify and no field already records: for example, why a particular light or performance register supports the brief. Do not repeat the shot list in prose or turn this explanation into an approval gate. Follow the separate generation authorization policy before spending.",
|
|
12
13
|
"image-defaults": "**Image default:** gpt-image-2-5-sunburst, quality `high`, 3k. User overrides take priority. Without a project, generation uses the headless Nano Banana 2 seat.\n\n| Model | Default resolution |\n|---|---|\n| nano-banana-2 | 2k |\n| nano-banana-2-lite | 1k |\n| nano-banana-pro | 2k |\n| gpt-image-2-5-flare | 2k |\n| gpt-image-2-5-sunburst | 3k |\n| flux-2-max | 1k |\n| seedream-5-lite | 2k |",
|
|
14
|
+
"iteration-diagnosis": "## Diagnose repeated failures\n\nAfter three failed attempts at the same requirement, pause unchanged re-rolls and diagnose the source reference, prompt structure, model fit and tool result. Three is a review checkpoint, not a universal limit or proof that the seed cannot matter. Preserve the attempts and name what each test changed.\n\nContinue autonomously when the brief is clear, a specific correction is supported and the next request is already authorized. Hand control back when taste or intent cannot be inferred, the next request needs fresh consent, or the available tool cannot meet the requirement. A failed roll never authorizes an additional charge. Follow the existing batch and per-request cost policy.",
|
|
13
15
|
"lens-video-split": "Named lenses, apertures, film stocks and camera bodies (`85mm f/1.4`, `Kodak Portra 400`, `ARRI Alexa 65`) are an image-model lever. On a video model, translate the look instead of pasting the gear list: `85mm f/1.4, Portra 400` becomes `close-up, shallow depth of field, warm natural colors, cinematic texture, film-grain texture`. ByteDance's Seedance 2.0 guide never mentions fps, shutter angle, f-stop or lens millimetres. Its Seedance 2.5 guide does, once: the visual-style line of its own storyboard example names one camera body and one 35 mm cinema lens. On 2.5 a single line like that is vendor-sanctioned; a stacked gear list still is not.",
|
|
16
|
+
"model-routing": "**Current model routing, generated from the operation routing source:**\n\n### image generate\n\nNano Banana 2 (Gemini 3.1 Flash Image): The all-rounder and the only image seat with a headless path: holds many subjects coherently in one frame, and the start-frame for legible in-scene text. Knowledge cutoff Jan 2025: anything later needs reference images.\nNano Banana 2 Lite: FAST/DRAFT image tier — markedly cheaper and faster than NB2 full, at draft quality. Route here for iteration volume, then re-run the winner on NB2 full. Same Gemini content filter as NB2.\nNano Banana Pro: HERO-FRAME / typography PREMIUM image tier. NB2 is about 95% of Pro — escalate only when spatial composition, cinematic lighting/skin, fine typography-in-scene or deep multi-element reasoning must be perfect, and say why.\nGPT Image 2.5 Flare: THE FAST GPT IMAGE SEAT — OpenAI's small model, optimized for SPEED, quality COMPARABLE to GPT Image 2 (not better) at roughly half the latency. Route here when speed matters: drafts, exploration, volume. TEXT / DIAGRAM / PANEL work — character sheets, shot grids, text-bearing panels. When quality outranks speed, escalate to Sunburst. Own content filter, distinct from Gemini's. Killed by a head-to-head at the intended crop going the other way.\nGPT Image 2.5 Sunburst: THE QUALITY GPT IMAGE SEAT — OpenAI's most capable image model, higher quality than GPT Image 2, same price as Flare, deliberately SLOWER. Route here unless speed is the point: finals, hero frames, photoreal people, and multi-reference edits where every reference must survive into one frame — its widest lead. Explore on Flare, finish on Sunburst.\nFLUX.2 Max: Photoreal image seat, less censored than the Gemini rails. Auto-routes to its edit endpoint when references are present.\nSeedream 5 Lite: Cheapest flat-priced image seat (GPT Image 2.5 at low quality costs less per image). Less censored. Routes to its edit endpoint when references are present.\n\n### video generate\n\nSeedance 2.0: THE 4K AND VALUE SEAT beside the 2.5 default — the only Seedance with native 4K (Pro-gated; base accounts get PRO_REQUIRED) and cheaper than 2.5 at every resolution they share, with the same physics, effects and scale strengths; shorter takes, fewer references, no timestamps. VIDEO-ONLY. A bare \"seedance\" still resolves here for older CLIs that expect 4K.\nSeedance 2.5: DEFAULT VIDEO MODEL — the strongest seat for physics, effects, scale and hero shots, and the only Seedance that takes long single takes, many references, audio-only references and integer-second timestamps. No 4K, and dearer than 2.0 at every shared resolution: go to 2.0 for 4K or the same resolution cheaper. LENGTH is the price dial — quote long takes first. VIDEO-ONLY. Timestamp grammar and the edit/extend words that make the provider reclassify and fail a generation are in slates-prompting-seedance-2-5.\nKling 3.0: THE COST-EFFECTIVE SEAT — strong start-frame adherence (identity, layout, text), acting, dialogue and lip-sync; pick it when the budget matters and the shot is a performance or a start-frame animation. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools.\nGemini Omni Flash: 720p seat with native synced audio included. Route here for drafts with sound in one pass and reference-to-video character-consistency trials; LTX, H3 and H3 Max Turbo cost less per second. VIDEO-ONLY. Quality against Kling/Seedance is unproven — do not route hero shots here.\nMiniMax H3: THE AUTHORED-AUDIO SEAT — reach for H3 when the sound is part of the shot rather than a switch on it: synchronised dialogue, scene sound and an audience-only score directed as three separate layers in ONE pass, across eleven languages. Kling and Seedance treat audio as on/off. Only H3 also carries a DECLARED REFERENCE RELATIONSHIP (kept whole, partly kept, transferred, or a loose echo). VIDEO-ONLY. Its top two resolution tiers are UPSCALES of the native render, not larger generations — judge at native and upscale in post. Reference images past the fifth are a PAID key dimension: pass referenceImages when quoting.\nMiniMax H3 Max: THE SPEED SEAT, dearer than base H3 at 768p and equal at 480p — never the cheap H3 and never the default. fal's post-train of the H3 weights: MEASURED 2026-08-27 at about 12x faster than base H3 on the same prompt and params, queue to finished file, plus a thin vendor-reported quality edge. It tops out at a 1080p refinement of its 768p render. It takes the same omni-reference set as base H3 and animates start and end frames — but not both in one call, the same as base H3: frames and references go to different endpoints. Never describe this row as taking no image or reference input. Route here when a fast turnaround on text-to-video or a start-frame shot is worth the premium.\nMiniMax H3 Max Turbo: THE BUDGET SEAT of the MiniMax family: a second fal post-train of the H3 weights, billed at half H3 Max's rate at every tier. Its 1080p is a refinement of the native 768p render, not a native 1080p generation. INPUTS ARE FRAMES, NOT REFERENCES: text-to-video and start/end frames only, with no reference endpoint, so reference-driven consistency goes to H3 Max or base H3. Route here for drafts, volume and cheap coverage, then re-run the keeper on H3 Max or a hero seat.\nLTX-2.5: THE VOLUME SEAT — the cheapest 1080p second with sound included, and the row for MANY takes rather than one hero shot. Native synced audio is included free at every tier, unlike Kling where sound is a paid key dimension. It supports native high-resolution output and longer takes than most seats; use the capability surface for its resolution-dependent duration limits. VIDEO-ONLY. INPUTS ARE FRAMES, NOT REFERENCES: start frame plus an optional end frame, and no reference endpoint at all — for character consistency across shots use H3 or Kling. Route here for batch coverage, long takes, and anything where the credit budget is the binding constraint.\nLTX-2.5 Pro: THE FIDELITY SEAT of the LTX pair — the full diffusion build against the base row's distilled one. 🚨 IT IS NOT A SUPERSET OF THE BASE ROW, which is the opposite of every other Pro seat here: it reaches a SHORTER resolution ladder and makes SHORTER clips, and it costs more at both tiers they share. Reaching for it because the name says Pro costs more AND takes away reach. Everything else matches the base row. Route here only when a specific shot needs the fidelity and fits inside its narrower envelope.\n\n### video edit\n\nSeedance 2.5 Edit: VIDEO-TO-VIDEO EDIT via slates_edit_video, and the only edit engine that takes a clip longer than the other two reach — that length is the whole reason to route here. Inside their range, compare on fidelity instead: Omni Flash edit won the prompt-only head-to-head, and Kling edit is the one that takes reference images. Edits audio on the same row (re-voice, re-accent, translate with re-fitted lips, replace BGM). Costs about 1.2x a plain 2.5 generation of the same length: an edit bills at twice the reduced video-reference rate.\nKling O3 Video Edit: VIDEO-TO-VIDEO EDIT, the REF-DRIVEN one: it is the only edit seat that takes element/style reference images to lock subject identity, and its keepAudio preserves the original audio verbatim. Route here when an edit NEEDS reference images or bit-exact audio; for prompt-only footage-synced VFX, omni-flash-edit won the fidelity head-to-head. One instruction beat per pass — multi-beat prompts get under-executed.\nOmni Flash Edit: VIDEO-TO-VIDEO EDIT, prompt-only — THE EDIT-FIDELITY WINNER (head-to-head vs Kling edit on real talking footage: lips held, audio near-identical, both action beats landed), priced level with Kling O3 Edit Standard. Footage-synced prop, effect, environment and lighting swaps. Takes NO reference images — identity swaps needing refs go to Kling edit. Fidelity is EARNED by prompt discipline; the exact form is in slates-prompting-omni-flash.\n\n### audio generate\n\nSeed Audio 1.0: DEFAULT audio model — the one-pass SCENE workhorse: dialogue, SFX and ambience together from ONE plain sentence. Route here for continuity beds, room tone, crowd and nature soundscapes, and quick scratch VO. AUDIO-ONLY. Takes one image XOR up to three audio clips as references, never both. Prompt form and the length rule are in slates-prompting-seed-audio.\nElevenLabs Sound Effects v2: ONE-SHOT SOUND EFFECT with an EXACT duration — route here for a single hit that must land on a frame (door slam, whoosh, impact, UI blip) or for a seamless loop. AUDIO-ONLY. For layered scenes with dialogue or room tone, seed-audio does it in one pass instead.\nInworld Realtime TTS-2: THE VOICE SEAT — one named voice saying one line, billed per CHARACTER not per second. Route here when WHO is speaking matters. NOT scene audio — that is seed-audio; a single effect is eleven-sfx.",
|
|
14
17
|
"reference-rules-core": "Identity = a few flat-lit neutral angles; one reference per role, named inline; 2-4 refs not 12; describe environments instead of feeding a grid.\n\n1. **2-4 strong references beat both extremes.** Not 1 (warps toward itself), not 12 (averages worse). Start with 2-3 focused refs — each one adds context AND another variable to balance.\n2. **One reference per ROLE, named in the prompt** — identity / style-grade / environment. The model does **not** infer a reference's role from its position in the list; the inline name carries it. Same-role competitors drift (two \"identity\" refs of different people blend into a third face). Slates resolves `@mentions` / `#tags` into numbered citations. You can also bind references directly in scene prose, naming what each image supplies.\n3. **One identity sheet per character, named inline.** A character's identity is a single asset (dominant portrait + body panels), so attach that one asset rather than a pile of views: **fewer competing renderings of a face is better, because the model cannot tell which one is authoritative and averages them.** Slates cites it as `Marcus (image 1)`. **Do NOT hand-write a \"Reference Image Instructions\" block or role essays** (\"use for identity, ignore the outfit, render a neutral expression\") — that drags the sheet's studio lighting and wardrobe into a scene that asked for neither. The prompt leads; the user's words own wardrobe, expression, lighting, and action.\n4. **Flat-light identity refs.** Prep identity references with flat, even, shadowless lighting on a plain neutral background. A studio-lit or scene-lit character sheet bleeds its lighting into every generation — the failure looks like the subject was green-screen-pasted in front of the location. Reference prep beats prompting here.\n5. **Environment: describe it, don't feed a grid.** Default to describing the location in words and let the model build a space that fits the shot. Reserve an environment reference for a mandatory exact-match, and then use ONE clean establishing image with natural ambient light that reads as the location's real light — never a multi-panel grid fed whole.\n6. **Grids: explore, don't input.** Use grids to explore compositions cheaply, then pick a cell. Never feed a grid back in as a reference — the cells share a split detail budget and were generated jointly, so their flaws propagate.\n7. **Reuse the same refs across every shot** in a sequence. Lock a set and keep it; swapping references mid-sequence causes drift, because the model adapts each reference to the current prompt rather than copying it.\n8. **Legible in-shot text → bake it into a still start frame, never trust text-to-video.** Have an image model render the text, then animate from that locked frame. Video models smear type.\n9. **Working from existing media — describe ONLY what changes.** The source already carries its composition, motion, timing, and performance; re-describing them fights the model. Narrate the delta. (Video lane: restyle your own clip while keeping the performance; delayed-VFX on \"video one\"; marker-object insertion; video-as-reference for a series.)\n10. **Style transforms happen in natural language.** By default the source's artistic medium and visual style are inherited. To change it, add a plain-text instruction (\"anime → real person\"). There are no preset pickers, and there is no style slider.",
|
|
15
18
|
"reference-tips-short": "Name each reference inline; never write role essays. Slates does this for you: `@mention` a subject or environment and it composes `Marcus (image 1) in the cafe (image 2)`, citing them in the exact order it sends them. One canonical identity image avoids competing facial renderings; a \"Reference Image Instructions\" block drags reference lighting into your scene. Start with 2-3 focused refs.",
|
|
16
19
|
"references-read-literally": "> **The general law: the model reads a reference literally.**\n> A reference image is not a suggestion. Whatever is baked into it — lighting, medium, texture, symmetry, competing identities — is read as a **property of the subject** and reproduced downstream. A baked rim light tints every shot made from that sheet. A sheet that looks like a 3D game render gets animated like game footage. Two competing renderings of one face get averaged into a third face.\n\nEvery reference rule below is a corollary of that one sentence, which is why \"prep the reference\" beats \"prompt around the reference\" every time:\n\n- **Flat, plain identity refs** — because scene lighting in the sheet becomes scene lighting in the output (Slates' own receipt: a studio-lit sheet produced a subject that looked green-screen-pasted in front of mountains).\n- **One authoritative rendering per subject** — because the model cannot tell which panel is the real one. ByteDance documents this failure directly: multi-view character assets \"confuse the model's character recognition, causing it to generate duplicate characters of the same appearance.\"\n- **No 3D-game-render look in a reference** — the model recognizes the render mood and inherits its motion character, so the *animation* comes out looking like game footage. This is not a taste rule; it is the same literal-reading mechanism applied to the temporal layer.\n- **Break perfect symmetry** — mirrored faces and dead-square framing read as synthetic, and the model preserves that reading rather than correcting it.\n\n**What this means in practice:** when output is wrong in a way that tracks the *subject* rather than the *scene* — the lighting is wrong the same way in every shot, the face drifts, the material looks synthetic everywhere — fix the reference, not the prompt. Prompting around a baked-in property is the expensive way to lose.",
|
|
17
|
-
"seedance-25-timestamps": "**2.0 does not respond to timestamps and answers only to shot numbers. 2.5 responds to\ninteger-second timestamps.** That is ByteDance's own first line under \"Differences from Seedance\n2.0\", and it is why a 30-second take is usable at all: the length is only worth buying if you can\nsay *when* things happen inside it.\n\nBoth formats are valid on 2.5, and you can mix them — `Shot N` blocks for a storyboard whose\npacing you are happy to leave to the model, timestamps when a beat has to land at a moment.\n\n**Three ways to control time, all first-party:**\n\n| Form | Write it like |\n|---|---|\n| **Interval** | `0-3 seconds… 3-7 seconds… 7-15 seconds` or `[1s-4s]… [4s-8s]… [8s-12s]` |\n| **Time point** | *\"Quick left sideways transition at the 5-second mark.\"* |\n| **Relative** | *\"After 3 seconds, everyone around him shakes their head.\"* · *\"The frame freezes for 1 second after he presses the shutter.\"* |\n\n**The rules that come with them:**\n\n- **One second is the smallest unit.** Integers only — no `2.5s`, no frames.\n- **No gaps in the timeline.** `0-3s… 5-6s…` leaves 3-5s unspecified and the model fills it however\n it likes. Intervals must abut: `0-3s`, `3-7s`, `7-15s`.\n- **Budget the plot to the seconds.** Too little content in a range and the model improvises to\n fill it; too much and you get extra cuts or dropped beats. This is the actual craft of a 30s take.\n- **Never time-code a high-frequency action.** *\"Shake your head three times per second\"* is\n explicitly called out as a misuse — timestamps schedule beats, they don't choreograph frames.\n- **Transitions want both halves:** the moment AND the method — *\"At the 5-second mark, the camera\n transitions leftward with a left wipe into a natural dissolve.\"*\n- **Timestamps work on an EDIT too**, and that is where they earn the most: they scope a change in\n time as well as in content — *\"Change the man's action from drinking coffee to mopping the floor\n from 4-6 seconds in Video 1, and leave the rest of the content unchanged.\"* Without a range, a\n whole-clip instruction is applied to the whole clip.\n\nDo **not** carry this back to 2.0, and do not
|
|
20
|
+
"seedance-25-timestamps": "**2.0 does not respond to timestamps and answers only to shot numbers. 2.5 responds to\ninteger-second timestamps.** That is ByteDance's own first line under \"Differences from Seedance\n2.0\", and it is why a 30-second take is usable at all: the length is only worth buying if you can\nsay *when* things happen inside it.\n\nBoth formats are valid on 2.5, and you can mix them — `Shot N` blocks for a storyboard whose\npacing you are happy to leave to the model, timestamps when a beat has to land at a moment.\n\n**Three ways to control time, all first-party:**\n\n| Form | Write it like |\n|---|---|\n| **Interval** | `0-3 seconds… 3-7 seconds… 7-15 seconds` or `[1s-4s]… [4s-8s]… [8s-12s]` |\n| **Time point** | *\"Quick left sideways transition at the 5-second mark.\"* |\n| **Relative** | *\"After 3 seconds, everyone around him shakes their head.\"* · *\"The frame freezes for 1 second after he presses the shutter.\"* |\n\n**The rules that come with them:**\n\n- **One second is the smallest unit.** Integers only — no `2.5s`, no frames.\n- **No gaps in the timeline.** `0-3s… 5-6s…` leaves 3-5s unspecified and the model fills it however\n it likes. Intervals must abut: `0-3s`, `3-7s`, `7-15s`.\n- **Budget the plot to the seconds.** Too little content in a range and the model improvises to\n fill it; too much and you get extra cuts or dropped beats. This is the actual craft of a 30s take.\n- **Never time-code a high-frequency action.** *\"Shake your head three times per second\"* is\n explicitly called out as a misuse — timestamps schedule beats, they don't choreograph frames.\n- **Transitions want both halves:** the moment AND the method — *\"At the 5-second mark, the camera\n transitions leftward with a left wipe into a natural dissolve.\"*\n- **Timestamps work on an EDIT too**, and that is where they earn the most: they scope a change in\n time as well as in content — *\"Change the man's action from drinking coffee to mopping the floor\n from 4-6 seconds in Video 1, and leave the rest of the content unchanged.\"* Without a range, a\n whole-clip instruction is applied to the whole clip.\n\nDo **not** carry this back to 2.0, and do not write `[00:00-00:02]` minute-second brackets (another\nvendor's syntax) into either — 2.0 ignores time entirely, and the cross-model syntax swap is its own known failure.",
|
|
18
21
|
"seedance-25-timestamps-short": "Seedance 2.0 ignores timing and answers only to \"Shot 1 / Shot 2\"; 2.5 acts on whole-second timestamps, and that is what makes a 30-second take controllable rather than just long. Three forms work: intervals (\"0-3 seconds…3-7 seconds\"), a point (\"at the 5-second mark\"), or relative (\"after 3 seconds\"). Whole seconds only, no gaps between intervals, and never to choreograph fast repeated motion. They work on edits too, where a range scopes the change: \"…from 4-6 seconds…\".",
|
|
19
22
|
"sheet-tool-defaults": "**What the sheet tools render on** (you do not pick these; omit `model`):\n\n- **Character identity sheet:** `gpt-image-2-5-sunburst` at 3k, quality `high`, one 16:9 image.\n- **Establishing image:** `gpt-image-2-5-sunburst` at 3k, quality `high`, one 16:9 image.\n\nPrice a sheet for that model at 16:9, with resolution and quality left at their defaults. **Never 4K** — no identity gain at sheet scale, wasted spend.",
|
|
20
|
-
"still-gate": "**
|
|
21
|
-
"thresholds": "<!-- GENERATED from @slatesvideo/shared — do not edit between the markers.\n Source: CONFIRM_CREDITS, DEVIATION_FACTOR and the audio bounds in\n packages/shared/src/operations/index.ts. Every number here is REFUSED by an\n op when a prompt gets it wrong, which is why none of them is typed by hand\n any more: this block replaced four claims that contradicted the code. -->\n\n**The thresholds, from the code that enforces them:**\n\n- **Confirm gate:** above **17 credits** an op returns `requires_confirm` and will not\n proceed until you re-call with `confirm: true`.
|
|
23
|
+
"still-gate": "**Inspect a start frame before animating it.** Repair a visible defect that would make the intended crop or performance unusable before spending on motion. A clean frame can be animated whenever the brief calls for movement; this check does not require an image stage for text-to-video.\n\nThis is a cost rule as well as craft: a premium video call can cost many times an image correction. Broken geometry can turn to mush, oily textures can crawl and malformed objects can fall apart in motion. Fix a known source defect at the source instead of buying a more expensive copy. Judge intentional stylisation against the brief, not a universal photoreal standard. Additional image or video requests still follow the existing generation authorization.",
|
|
24
|
+
"thresholds": "<!-- GENERATED from @slatesvideo/shared — do not edit between the markers.\n Source: CONFIRM_CREDITS, DEVIATION_FACTOR and the audio bounds in\n packages/shared/src/operations/index.ts. Every number here is REFUSED by an\n op when a prompt gets it wrong, which is why none of them is typed by hand\n any more: this block replaced four claims that contradicted the code. -->\n\n**The thresholds, from the code that enforces them:**\n\n- **Confirm gate:** above **17 credits** an op returns `requires_confirm` and will not\n proceed until you re-call with `confirm: true`. This is a code gate, not permission to spend: every generation still needs the user-approved plan or quote.\n- **Deviation pause:** the desktop Studio Agent stops and re-asks when projected generation spend\n exceeds the approved plan by more than **20%**. You do not trigger this; the app does.\n- **Seed Audio duration:** **3–120 seconds.** There is no duration\n parameter on the model — the number you pass is written into the prompt AND is what the user is\n billed. Outside that range the op refuses rather than clamping.\n- **Sound Effects duration:** **1–22 seconds**, billed per second, never left for the\n model to pick.\n\nNever quote a credit figure from memory: `slates_estimate_generation_cost` returns the real one.",
|
|
22
25
|
};
|
|
23
26
|
//# sourceMappingURL=partials.generated.js.map
|
|
@@ -16,7 +16,7 @@ export interface PromptingTipsEntry {
|
|
|
16
16
|
/** Footer callout paragraphs. */
|
|
17
17
|
footer?: string[];
|
|
18
18
|
}
|
|
19
|
-
export type PromptingTipsKey = 'seedance' | 'seedance-2-5' | 'seedance-2-5-edit' | 'kling' | 'kling-edit' | '
|
|
19
|
+
export type PromptingTipsKey = 'seedance' | 'seedance-2-5' | 'seedance-2-5-edit' | 'kling' | 'kling-edit' | 'omni-flash' | 'omni-flash-edit' | 'minimax-h3' | 'ltx-2-5' | 'nano-banana' | 'nano-banana-lite' | 'gpt-image-2-5' | 'seed-audio' | 'eleven-sfx' | 'inworld-tts-2';
|
|
20
20
|
export declare const PROMPTING_TIPS: Record<PromptingTipsKey, PromptingTipsEntry>;
|
|
21
21
|
/** Null when no tips exist for the key — callers render an honest fallback. */
|
|
22
22
|
export declare function getPromptingTips(key: string): PromptingTipsEntry | null;
|
|
@@ -183,14 +183,14 @@ const SEEDANCE_25_EDIT = {
|
|
|
183
183
|
critical: true,
|
|
184
184
|
},
|
|
185
185
|
{
|
|
186
|
-
heading: 'An edit costs about
|
|
187
|
-
note: '
|
|
186
|
+
heading: 'An edit costs about 1.2x a generation',
|
|
187
|
+
note: 'A Seedance edit bills at twice the reduced video-reference rate, so a 20-second edit costs about 1.2x a 20-second generation. Read the number on the Generate button rather than reasoning from the generation rate.',
|
|
188
188
|
},
|
|
189
189
|
],
|
|
190
190
|
[
|
|
191
191
|
{
|
|
192
192
|
heading: 'When to use it instead of the others',
|
|
193
|
-
note: 'Length is the reason: it is the only engine that accepts a clip over 15 seconds. Inside the others\' range, choose on fidelity — Omni Flash Edit is the prompt-only fidelity winner
|
|
193
|
+
note: 'Length is the reason: it is the only engine that accepts a clip over 15 seconds. Inside the others\' range, choose on fidelity — Omni Flash Edit is the prompt-only fidelity winner, priced level with Kling O3 Edit Standard, and Kling O3 Edit is the one that takes subject and style reference images.',
|
|
194
194
|
},
|
|
195
195
|
{
|
|
196
196
|
heading: 'It edits the audio too',
|
|
@@ -269,53 +269,10 @@ const KLING_EDIT = {
|
|
|
269
269
|
],
|
|
270
270
|
],
|
|
271
271
|
};
|
|
272
|
-
const VEO = {
|
|
273
|
-
label: 'Veo 3.1',
|
|
274
|
-
intro: [
|
|
275
|
-
'Veo 3.1 generates synchronized audio directly with video. Aspect ratio: 16:9 only (for 9:16 vertical, use Kling or Seedance). Native single-clip duration: 4, 6, or 8 seconds — longer durations require chaining clips via last-frame reuse.',
|
|
276
|
-
'Official Cloud formula: [Cinematography] + [Subject] + [Action] + [Context] + [Style & Ambiance]. Sweet spot ~50-150 words.',
|
|
277
|
-
],
|
|
278
|
-
columns: [
|
|
279
|
-
[
|
|
280
|
-
{
|
|
281
|
-
heading: 'Dialogue',
|
|
282
|
-
example: 'Character says, "exact words"',
|
|
283
|
-
note: 'Use quotation marks for exact speech. Keep voice direction terse: "says in a weary voice", "whispers", "shouts". 2-3 speakers max — sync degrades past that.',
|
|
284
|
-
},
|
|
285
|
-
{
|
|
286
|
-
heading: 'Sound Effects — with cause',
|
|
287
|
-
example: 'SFX: thunder cracks in the distance',
|
|
288
|
-
note: 'Always specify direction or distance — "SFX: thunder" alone is too vague.',
|
|
289
|
-
},
|
|
290
|
-
{
|
|
291
|
-
heading: 'Ambient is mandatory',
|
|
292
|
-
example: 'Soft office ambience. · Wind on the open ridge.',
|
|
293
|
-
note: 'Include an ambience line in every scene — without it the audio mix feels dead.',
|
|
294
|
-
},
|
|
295
|
-
],
|
|
296
|
-
[
|
|
297
|
-
{
|
|
298
|
-
heading: 'No subtitles — MANDATORY',
|
|
299
|
-
example: 'The founder says, "..." (no subtitles). Soft office ambience.',
|
|
300
|
-
note: 'Without (no subtitles) after every dialogue line, Veo bakes subtitle text into the video. This is genuinely critical and underspecified in most guides.',
|
|
301
|
-
critical: true,
|
|
302
|
-
},
|
|
303
|
-
{
|
|
304
|
-
heading: 'Cinematography vocabulary',
|
|
305
|
-
example: '85mm · shallow depth of field · Rembrandt lighting · dolly in · whip pan',
|
|
306
|
-
note: 'Veo responds to real lens, lighting, and camera-move terms — lead the prompt with them.',
|
|
307
|
-
},
|
|
308
|
-
],
|
|
309
|
-
],
|
|
310
|
-
footer: [
|
|
311
|
-
"First-frame + last-frame is Veo's strongest workflow. Generate a start frame, generate an end frame, then animate with both as anchors. Motion-Lock hack: keep ~60% of the same background pixels between start and end to prevent latent drift.",
|
|
312
|
-
'Keep dialogue under one natural breath — lines fit the 8s clip ceiling. Texture-realism phrases: fine skin pores, visible fabric weave, subtle contrast, no gloss or sharpening.',
|
|
313
|
-
],
|
|
314
|
-
};
|
|
315
272
|
const OMNI_FLASH = {
|
|
316
273
|
label: 'Gemini Omni Flash',
|
|
317
274
|
intro: [
|
|
318
|
-
'Gemini Omni Flash is the
|
|
275
|
+
'Gemini Omni Flash is the 720p tier with native synced audio included — dialogue, SFX, and ambient generate WITH the video at no extra cost. 3-10s, 16:9 or 9:16. Text-to-video, one start frame, or up to 7 reference images. No last frame, no video/audio references.',
|
|
319
276
|
],
|
|
320
277
|
columns: [
|
|
321
278
|
[
|
|
@@ -348,7 +305,7 @@ const OMNI_FLASH = {
|
|
|
348
305
|
},
|
|
349
306
|
{
|
|
350
307
|
heading: 'Know its role',
|
|
351
|
-
note: '
|
|
308
|
+
note: 'Drafts and audio in one pass; LTX, H3 and H3 Max Turbo cost less per second. For hero shots, Seedance 2.5 (the default) or Seedance 2.0 (4K, cheaper) still win.',
|
|
352
309
|
},
|
|
353
310
|
],
|
|
354
311
|
],
|
|
@@ -512,7 +469,7 @@ const NANO_BANANA = {
|
|
|
512
469
|
],
|
|
513
470
|
footer: [
|
|
514
471
|
'Boring vs Cinema. Boring: "Wide shot of man on dock looking at forest." Cinema: "Direct overhead drone shot on weathered dock. Single figure climbing up frame bottom. Boot prints leading toward shore. Pale winter light. Anamorphic flare. Desaturated blue/slate palette. Kodak Portra 400 grain. Map of threat."',
|
|
515
|
-
"
|
|
472
|
+
"Three-attempt checkpoint. If three iterations on the same prompt haven't landed, stop re-rolling and diagnose the reference, prompt structure and model fit before the next spend. Usually the structure is wrong, not the seed.",
|
|
516
473
|
],
|
|
517
474
|
};
|
|
518
475
|
const NANO_BANANA_LITE = {
|
|
@@ -520,13 +477,15 @@ const NANO_BANANA_LITE = {
|
|
|
520
477
|
label: 'Nano Banana 2 Lite',
|
|
521
478
|
columns: [
|
|
522
479
|
NANO_BANANA.columns[0],
|
|
523
|
-
NANO_BANANA.columns[1].map((card) => card.heading === '
|
|
524
|
-
? {
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
480
|
+
NANO_BANANA.columns[1].map((card) => card.heading === 'Reference images — name them, never label roles'
|
|
481
|
+
? { ...card, note: `Up to 4 refs on Lite (the edit endpoint caps input images at 4). ${PARTIALS['reference-tips-short']}` }
|
|
482
|
+
: card.heading === 'Resolution tactics'
|
|
483
|
+
? {
|
|
484
|
+
heading: 'Resolution tactics',
|
|
485
|
+
example: '1k only on Lite',
|
|
486
|
+
note: 'Lite outputs 1K only — use it for iteration volume and drafts, then switch to Nano Banana 2 for 2K/4K finals.',
|
|
487
|
+
}
|
|
488
|
+
: card),
|
|
530
489
|
],
|
|
531
490
|
};
|
|
532
491
|
// ── Audio lane ──────────────────────────────────────────────────
|
|
@@ -591,7 +550,7 @@ const ELEVEN_SFX = {
|
|
|
591
550
|
label: 'ElevenLabs Sound Effects',
|
|
592
551
|
intro: [
|
|
593
552
|
'Sound Effects makes one short sound with an exact length — the lane for a hit that has to land on a specific frame, or a seamless loop you can lay under a whole scene.',
|
|
594
|
-
'Duration is always sent explicitly (
|
|
553
|
+
'Duration is always sent explicitly (1–22s). Billing is per second, so the length you pick is the price you pay.',
|
|
595
554
|
],
|
|
596
555
|
columns: [
|
|
597
556
|
[
|
|
@@ -607,7 +566,7 @@ const ELEVEN_SFX = {
|
|
|
607
566
|
},
|
|
608
567
|
{
|
|
609
568
|
heading: 'Duration is the edit',
|
|
610
|
-
example: '
|
|
569
|
+
example: '1s for an impact · 4s for a whoosh · 22s for a bed',
|
|
611
570
|
note: 'Ask for roughly the length you need. A 4-second request for a door slam pads the tail with room tone you then have to trim.',
|
|
612
571
|
},
|
|
613
572
|
],
|
|
@@ -633,13 +592,13 @@ const MINIMAX_H3 = {
|
|
|
633
592
|
label: 'MiniMax H3',
|
|
634
593
|
intro: [
|
|
635
594
|
'MiniMax H3 generates picture and sound in one pass — 24fps, 32kHz stereo, 5-15 seconds, 11 stably-supported languages. It is the only video model in Slates where audio is AUTHORED rather than switched on: synchronised dialogue and action sounds go in the body of the prompt, ambience goes in a soundscape section, and audience-only music goes in a score section. Put a sound in the wrong section and it is dropped, doubled, or attributed to the wrong source.',
|
|
636
|
-
'Three models. Base H3 runs 480p / 768p / 2K / 4K; H3 Max is fal\'s faster post-train and runs 480p / 768p / 1080p, dearer than base H3 at
|
|
595
|
+
'Three models. Base H3 runs 480p / 768p / 2K / 4K; H3 Max is fal\'s faster post-train and runs 480p / 768p / 1080p, dearer than base H3 at 768p and equal at 480p - a deliberate speed pick, never the cheap one; H3 Max Turbo has Max\'s ladder at half Max\'s rate and takes NO references. Base H3 and Max read up to 9 reference images plus 3 video and 3 audio clips (12 files total, and audio never travels alone); all three animate a start frame and an end frame. 768p is the default on all three because it is the tier the model natively generates; base H3\'s 2K and 4K are upscales of a 768p base, and 1080p on Max and Turbo is a refinement of it. Reference images past the free allowance are billed and the allowances DIFFER: 5 free on base H3, 4 on Max.',
|
|
637
596
|
],
|
|
638
597
|
columns: [
|
|
639
598
|
[
|
|
640
599
|
{
|
|
641
600
|
heading: 'Three audio layers, three places',
|
|
642
|
-
example: 'body: "First batch of the morning."\
|
|
601
|
+
example: 'body: "First batch of the morning."\noverall_soundscape: shutters scrape, trays clink\nnon_diegetic_music: solo piano, slow, no swell',
|
|
643
602
|
note: 'Dialogue, singing and diegetic music (a radio in the scene) go in the BODY on the beat they land. Ambience goes in the soundscape. The score is audience-only — name instruments and tempo, not moods.',
|
|
644
603
|
critical: true,
|
|
645
604
|
},
|
|
@@ -672,7 +631,7 @@ const MINIMAX_H3 = {
|
|
|
672
631
|
},
|
|
673
632
|
{
|
|
674
633
|
heading: 'Reference images past the fifth cost extra',
|
|
675
|
-
note: '
|
|
634
|
+
note: 'On base H3 the first 5 are free; each one after that adds 4 credits at every resolution and length, and the model takes 9 (H3 Max prices references as a token pool instead). Four extra images on a 10s 768p clip add 16 credits to a 30-credit generation. Attach what the shot needs, not the ceiling.',
|
|
676
635
|
critical: true,
|
|
677
636
|
},
|
|
678
637
|
{
|
|
@@ -694,7 +653,7 @@ const LTX_2_5 = {
|
|
|
694
653
|
label: 'LTX-2.5',
|
|
695
654
|
intro: [
|
|
696
655
|
'LTX-2.5 scores the picture on the same pass that draws it, so SOUND IS THE FIRST THING YOU WRITE, not the last. Lightricks ranks the six parts of a prompt in this order: sound, camera, character detail, shot type and scene, then scene dressing — and scene dressing is the first thing to cut when a prompt sprawls. Everything goes in ONE flowing paragraph, not a list of labelled sections.',
|
|
697
|
-
'Two models. Base LTX-2.5 is the distilled build: 720p / 1080p / 1440p / 4K and clips from 6 to 20 seconds, and it is the cheapest
|
|
656
|
+
'Two models. Base LTX-2.5 is the distilled build: 720p / 1080p / 1440p / 4K and clips from 6 to 20 seconds, and it is the cheapest 1080p second with sound included in Slates. LTX-2.5 Pro is the full diffusion build ("Diffusion Fidelity Rendering" spends extra compute on busy frames) but reaches a SHORTER ladder — 1080p and 10 seconds maximum — while costing about a third more. Pro is for a dense final render; base is for iteration, long takes and 4K.',
|
|
698
657
|
],
|
|
699
658
|
columns: [
|
|
700
659
|
[
|
|
@@ -817,7 +776,6 @@ export const PROMPTING_TIPS = {
|
|
|
817
776
|
'seedance-2-5-edit': SEEDANCE_25_EDIT,
|
|
818
777
|
kling: KLING,
|
|
819
778
|
'kling-edit': KLING_EDIT,
|
|
820
|
-
veo: VEO,
|
|
821
779
|
'omni-flash': OMNI_FLASH,
|
|
822
780
|
'omni-flash-edit': OMNI_FLASH_EDIT,
|
|
823
781
|
'minimax-h3': MINIMAX_H3,
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
const STOP_WORDS = new Set('a an and are as at be by can do for from have he her him his i in into is it its me my of on or our please she that the their them these they this those to was want we were with you your slates'.split(' '));
|
|
2
|
+
/** Shared task vocabulary, without substring matches or grammatical filler. */
|
|
3
|
+
export function searchTerms(text, additionalStopWords = []) {
|
|
4
|
+
const omitted = new Set([...STOP_WORDS, ...additionalStopWords]);
|
|
5
|
+
const normalize = (word) => {
|
|
6
|
+
if (/^movies?$/.test(word))
|
|
7
|
+
return 'film';
|
|
8
|
+
if (/^animat(?:e|ed|ing|ion|ions)$/.test(word))
|
|
9
|
+
return 'animate';
|
|
10
|
+
if (/^photograph(?:s)?$/.test(word))
|
|
11
|
+
return 'photo';
|
|
12
|
+
if (/^(voiceover|narration)s?$/.test(word))
|
|
13
|
+
return 'voice';
|
|
14
|
+
if (word.length > 4 && word.endsWith('ies'))
|
|
15
|
+
return word.slice(0, -3) + 'y';
|
|
16
|
+
if (word.length > 3 && word.endsWith('s') && !/(ss|us|ics)$/.test(word))
|
|
17
|
+
return word.slice(0, -1);
|
|
18
|
+
return word;
|
|
19
|
+
};
|
|
20
|
+
const words = text.toLowerCase().normalize('NFKC').match(/[\p{L}\p{N}]+/gu) ?? [];
|
|
21
|
+
// Filler is checked before and after normalizing, so "this" never survives as "thi".
|
|
22
|
+
return [...new Set(words.filter(word => !omitted.has(word)).map(normalize))].filter(word => word.length > 1 && !omitted.has(word));
|
|
23
|
+
}
|
|
24
|
+
//# sourceMappingURL=search-terms.js.map
|