@slatesvideo/shared 0.7.2 → 0.7.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/dist/clients/cloud.d.ts +4 -0
  2. package/dist/clients/cloud.js +11 -3
  3. package/dist/index.d.ts +1 -0
  4. package/dist/index.js +1 -0
  5. package/dist/manual/content.d.ts +1 -1
  6. package/dist/manual/content.js +1 -1
  7. package/dist/operations/index.d.ts +12 -13
  8. package/dist/operations/index.js +158 -133
  9. package/dist/operations/surface.d.ts +6 -2
  10. package/dist/operations/surface.js +29 -5
  11. package/dist/prompts/agent-doctrine.d.ts +4 -4
  12. package/dist/prompts/agent-doctrine.js +17 -28
  13. package/dist/prompts/guide-discovery.d.ts +23 -0
  14. package/dist/prompts/guide-discovery.js +39 -0
  15. package/dist/prompts/guide-retrieval.js +1 -1
  16. package/dist/prompts/model-capabilities.d.ts +8 -9
  17. package/dist/prompts/model-capabilities.js +11 -51
  18. package/dist/prompts/model-facts.d.ts +2 -2
  19. package/dist/prompts/model-facts.js +15 -26
  20. package/dist/prompts/partials.generated.js +6 -3
  21. package/dist/prompts/prompting-tips.d.ts +1 -1
  22. package/dist/prompts/prompting-tips.js +21 -63
  23. package/dist/prompts/search-terms.d.ts +3 -0
  24. package/dist/prompts/search-terms.js +24 -0
  25. package/dist/skills/content.js +36 -37
  26. package/dist/skills/metadata.d.ts +7 -0
  27. package/dist/skills/metadata.js +29 -0
  28. package/exports/slates-chatgpt-images/generated/SKILL.md +7 -1
  29. package/exports/slates-chatgpt-images/generated/slates-chatgpt-images.skill +0 -0
  30. package/exports/slates-prompt-builder/generated/SKILL.md +28 -16
  31. package/exports/slates-prompt-builder/generated/reference-character.md +12 -13
  32. package/exports/slates-prompt-builder/generated/reference-content-policy.md +2 -2
  33. package/exports/slates-prompt-builder/generated/reference-gpt-image-2-5.md +191 -0
  34. package/exports/slates-prompt-builder/generated/reference-kling.md +32 -11
  35. package/exports/slates-prompt-builder/generated/reference-nano-banana.md +24 -6
  36. package/exports/slates-prompt-builder/generated/reference-omni-flash.md +65 -0
  37. package/exports/slates-prompt-builder/generated/reference-seedance-2-5.md +362 -0
  38. package/exports/slates-prompt-builder/generated/reference-seedance.md +34 -4
  39. package/exports/slates-prompt-builder/generated/slates-prompt-builder-manifest.json +77 -23
  40. package/exports/slates-prompt-builder/generated/slates-prompt-builder.skill +0 -0
  41. package/package.json +2 -1
  42. package/skills/_partials/blender-action-curves.md +24 -0
  43. package/skills/_partials/iteration-diagnosis.md +5 -0
  44. package/skills/_partials/model-routing.md +35 -0
  45. package/skills/_partials/seedance-25-timestamps.md +2 -2
  46. package/skills/_partials/still-gate.md +2 -2
  47. package/skills/_partials/thresholds.md +1 -1
  48. package/skills/slates-blocking-to-prompt.md +15 -13
  49. package/skills/slates-camera-language.md +45 -7
  50. package/skills/slates-character-identity.md +8 -6
  51. package/skills/slates-chatgpt-images.md +7 -1
  52. package/skills/slates-cinematic-look.md +1 -1
  53. package/skills/slates-content-policy.md +4 -6
  54. package/skills/slates-cost-discipline.md +18 -12
  55. package/skills/slates-dialogue-blocking.md +6 -6
  56. package/skills/slates-direct-response-ad.md +1 -1
  57. package/skills/slates-edit-and-iterate.md +12 -4
  58. package/skills/slates-model-selection.md +82 -90
  59. package/skills/slates-one-prompt-film.md +1 -1
  60. package/skills/slates-previs-blocking.md +44 -13
  61. package/skills/slates-project-organization.md +2 -2
  62. package/skills/slates-prompting-elevenlabs.md +4 -4
  63. package/skills/slates-prompting-flux-2-max.md +2 -3
  64. package/skills/slates-prompting-gpt-image-2-5.md +2 -2
  65. package/skills/slates-prompting-inworld-tts.md +174 -174
  66. package/skills/slates-prompting-kling-v3.md +11 -9
  67. package/skills/slates-prompting-lip-sync.md +15 -15
  68. package/skills/slates-prompting-ltx-2-5.md +5 -6
  69. package/skills/slates-prompting-minimax-h3.md +11 -11
  70. package/skills/slates-prompting-motion-transfer.md +8 -8
  71. package/skills/slates-prompting-nano-banana-2.md +8 -4
  72. package/skills/slates-prompting-omni-flash.md +9 -9
  73. package/skills/slates-prompting-seed-audio.md +24 -4
  74. package/skills/slates-prompting-seedance-2-5.md +40 -30
  75. package/skills/slates-prompting-seedance.md +4 -4
  76. package/skills/slates-prompting-seedream-5-lite.md +6 -6
  77. package/skills/slates-restyle-from-blocking.md +2 -2
  78. package/skills/slates-script-craft.md +1 -1
  79. package/skills/slates-shot-variety.md +1 -1
  80. package/skills/slates-storyboard-from-script.md +1 -1
  81. package/skills/slates-style-prompting.md +56 -54
  82. package/skills/slates-ugc-influencer-ad.md +1 -1
  83. package/skills/slates-vision-feedback-loop.md +118 -110
  84. package/skills/slates-prompting-veo-3.md +0 -224
@@ -0,0 +1,7 @@
1
+ export interface SkillMetadata {
2
+ name: string;
3
+ description: string;
4
+ }
5
+ /** Portable discovery metadata, validated before embedding or choosing an install path. */
6
+ export declare function parseSkillMetadata(markdown: string, expectedName?: string): SkillMetadata;
7
+ //# sourceMappingURL=metadata.d.ts.map
@@ -0,0 +1,29 @@
1
+ import { parseDocument } from 'yaml';
2
+ /** Portable discovery metadata, validated before embedding or choosing an install path. */
3
+ export function parseSkillMetadata(markdown, expectedName) {
4
+ const label = expectedName ?? 'skill';
5
+ const fail = (message) => { throw new Error(`[skill-metadata] ${label}: ${message}`); };
6
+ const frontmatter = /^\uFEFF?---\r?\n([\s\S]*?)\r?\n---(?:\r?\n|$)/.exec(markdown);
7
+ if (!frontmatter)
8
+ return fail('missing YAML frontmatter');
9
+ const document = parseDocument(frontmatter[1], { uniqueKeys: true });
10
+ if (document.errors.length)
11
+ return fail(`invalid YAML: ${document.errors[0].message}`);
12
+ const fields = document.toJS({ maxAliasCount: 0 });
13
+ if (!fields || typeof fields !== 'object' || Array.isArray(fields))
14
+ return fail('frontmatter must be a mapping');
15
+ const { name, description } = fields;
16
+ if (typeof name !== 'string' || !/^[a-z0-9]+(?:-[a-z0-9]+)*$/.test(name) || name.length > 64) {
17
+ return fail('name must be 1-64 lowercase letters, digits and single hyphens');
18
+ }
19
+ if (expectedName && name !== expectedName)
20
+ return fail(`name "${name}" must match its source key`);
21
+ if (typeof description !== 'string' || !description.trim() || [...description].length > 1024) {
22
+ return fail('description must be a nonempty string of at most 1024 characters');
23
+ }
24
+ if (fields.compatibility !== undefined && (typeof fields.compatibility !== 'string' || [...fields.compatibility].length > 500)) {
25
+ return fail('compatibility must be a string of at most 500 characters');
26
+ }
27
+ return { name, description: description.trim() };
28
+ }
29
+ //# sourceMappingURL=metadata.js.map
@@ -1,10 +1,16 @@
1
1
  ---
2
2
  name: slates-chatgpt-images
3
- description: Generate images using a connected ChatGPT account or the desktop host's built-in image tool, preserving Slates project context, exact prompts and reference lineage. Use when the user requests ChatGPT generation rather than Slates credits.
3
+ description: "Generate and save images through a connected ChatGPT account or the host image tool, preserving Slates prompts, references and project context. Use when ChatGPT generation is requested."
4
4
  ---
5
5
 
6
6
  # ChatGPT images in Slates
7
7
 
8
+ ## Prerequisites
9
+
10
+ Saving into a Slates project requires the Slates desktop app open and a connected Slates MCP server or CLI. Follow the [connection guide](https://slates.video/docs/connect-claude); CLI onboarding is `npx -y @slatesvideo/cli setup`. The host image tool alone does not expose Slates project tools. If the required tools or desktop connection are unavailable, report the missing connection and retain the prompt and references; do not invent operations.
11
+
12
+ The connected generation path also needs the optional ChatGPT images add-on enabled and authenticated as described below. A host-tool path needs an actual image generator exposed by the host. These are distinct capabilities; a paid account alone establishes neither.
13
+
8
14
  Resolve the project with `slates_list_projects` and references with
9
15
  `slates_get_selection` or `slates_list_assets`. Badge codes are project-specific.
10
16
  Inspect the selected images before generating. Keep the ordered reference IDs
@@ -1,6 +1,7 @@
1
1
  ---
2
2
  name: slates-prompt-builder
3
- description: Turn a plain-language idea into a paste-ready AI video or image prompt using the production Slates prompting guides for Seedance 2.0, Kling 3.0, and Nano Banana 2. Use for video prompts, image prompts, shot planning, character-reference preparation, ads, brand films, product videos, talking heads, or any request that needs generation-ready visual direction.
3
+ description: Builds paste-ready image and video prompts from an ordinary-language brief using model-specific Slates craft. Use for visual prompt writing, clip edits, shot planning or recurring-character references.
4
+ compatibility: Standalone prompt preparation. Generation requires the user's chosen image or video tool; check that tool's current schema and reference support.
4
5
  ---
5
6
 
6
7
  <!-- Generated from the Slates production prompting guides. Do not edit — this file is rebuilt from source. -->
@@ -9,32 +10,43 @@ description: Turn a plain-language idea into a paste-ready AI video or image pro
9
10
 
10
11
  Turn the user's idea into the exact prompt to paste into their generation tool. The deliverable is the prompt, not a lecture about prompting.
11
12
 
12
- This portable skill is deliberately thin. Its reference files are generated directly from the same production skills used by the Slates MCP server, CLI-installed Claude skills, and Studio Agent. Treat those references as authoritative; never recreate their rules from memory.
13
+ This portable skill is deliberately thin. Its reference files are generated directly from the same production skills used by the Slates MCP server, CLI-installed skills, and Studio Agent. Treat those references as authoritative; never recreate their rules from memory.
13
14
 
14
- ## The curated stack
15
+ ## Models and routing
15
16
 
16
17
  <!-- @generated:model-routing -->
17
- | Model | Canonical route | Guide |
18
- |---|---|---|
19
- | **Kling 3.0** | THE COST-EFFECTIVE SEAT — strong start-frame adherence (identity, layout, text), acting, dialogue, lip-sync and the widest aspect-ratio set; pick it when the budget matters and the shot is a performance or a start-frame animation. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools. | `reference-kling.md` |
20
- | **Seedance 2.0** | THE 4K AND VALUE SEAT beside the 2.5 default — the only Seedance with native 4K (Pro-gated; base accounts get PRO_REQUIRED) and cheaper than 2.5 at every resolution they share, with the same physics, effects and scale strengths; shorter takes, fewer references, no timestamps. VIDEO-ONLY. A bare "seedance" still resolves here for older CLIs that expect 4K. | `reference-seedance.md` |
21
- | **Nano Banana 2 (Gemini 3.1 Flash Image)** | The all-rounder and the only image seat with a headless path: holds many subjects coherently in one frame, and the start-frame for legible in-scene text. Knowledge cutoff Jan 2025: anything later needs reference images. | `reference-nano-banana.md` |
18
+ | Model | Lane | Canonical route | Guide |
19
+ |---|---|---|---|
20
+ | **GPT Image 2.5 Sunburst** | image generate; default | THE QUALITY GPT IMAGE SEAT — OpenAI's most capable image model, higher quality than GPT Image 2, same price as Flare, deliberately SLOWER. Route here unless speed is the point: finals, hero frames, photoreal people, and multi-reference edits where every reference must survive into one frame — its widest lead. Explore on Flare, finish on Sunburst. | `reference-gpt-image-2-5.md` |
21
+ | **GPT Image 2.5 Flare** | image generate; specialist | THE FAST GPT IMAGE SEAT — OpenAI's small model, optimized for SPEED, quality COMPARABLE to GPT Image 2 (not better) at roughly half the latency. Route here when speed matters: drafts, exploration, volume. TEXT / DIAGRAM / PANEL work — character sheets, shot grids, text-bearing panels. When quality outranks speed, escalate to Sunburst. Own content filter, distinct from Gemini's. Killed by a head-to-head at the intended crop going the other way. | `reference-gpt-image-2-5.md` |
22
+ | **Nano Banana 2 (Gemini 3.1 Flash Image)** | image generate; specialist | The all-rounder and the only image seat with a headless path: holds many subjects coherently in one frame, and the start-frame for legible in-scene text. Knowledge cutoff Jan 2025: anything later needs reference images. | `reference-nano-banana.md` |
23
+ | **Seedance 2.5** | video generate; default | DEFAULT VIDEO MODEL — the strongest seat for physics, effects, scale and hero shots, and the only Seedance that takes long single takes, many references, audio-only references and integer-second timestamps. No 4K, and dearer than 2.0 at every shared resolution: go to 2.0 for 4K or the same resolution cheaper. LENGTH is the price dial — quote long takes first. VIDEO-ONLY. Timestamp grammar and the edit/extend words that make the provider reclassify and fail a generation are in reference-seedance-2-5.md. | `reference-seedance-2-5.md` |
24
+ | **Seedance 2.0** | video generate; specialist | THE 4K AND VALUE SEAT beside the 2.5 default — the only Seedance with native 4K (Pro required in Slates) and cheaper than 2.5 at every resolution they share, with the same physics, effects and scale strengths; shorter takes, fewer references, no timestamps. VIDEO-ONLY. | `reference-seedance.md` |
25
+ | **Kling 3.0** | video generate; specialist | THE COST-EFFECTIVE SEAT — strong start-frame adherence (identity, layout, text), acting, dialogue and lip-sync; pick it when the budget matters and the shot is a performance or a start-frame animation. Kling is also the ONLY engine behind the Motion Transfer and Lip Sync tools. | `reference-kling.md` |
26
+ | **Gemini Omni Flash** | video generate; specialist | 720p seat with native synced audio included. Route here for drafts with sound in one pass and reference-to-video character-consistency trials; LTX, H3 and H3 Max Turbo cost less per second. VIDEO-ONLY. Quality against Kling/Seedance is unproven — do not route hero shots here. | `reference-omni-flash.md` |
27
+ | **Omni Flash Edit** | video edit; default | VIDEO-TO-VIDEO EDIT, prompt-only — THE EDIT-FIDELITY WINNER (head-to-head vs Kling edit on real talking footage: lips held, audio near-identical, both action beats landed), priced level with Kling O3 Edit Standard. Footage-synced prop, effect, environment and lighting swaps. Takes NO reference images — identity swaps needing refs go to Kling edit. Fidelity is EARNED by prompt discipline; the exact form is in reference-omni-flash.md. | `reference-omni-flash.md` |
28
+ | **Kling O3 Video Edit** | video edit; specialist | VIDEO-TO-VIDEO EDIT, the REF-DRIVEN one: it is the only edit seat that takes element/style reference images to lock subject identity, and its keepAudio preserves the original audio verbatim. Route here when an edit NEEDS reference images or bit-exact audio; for prompt-only footage-synced VFX, omni-flash-edit won the fidelity head-to-head. One instruction beat per pass — multi-beat prompts get under-executed. | `reference-kling.md` |
29
+ | **Seedance 2.5 Edit** | video edit; specialist | VIDEO-TO-VIDEO EDIT, and the only edit engine that takes a clip longer than the other two reach — that length is the whole reason to route here. Inside their range, compare on fidelity instead: Omni Flash edit won the prompt-only head-to-head, and Kling edit is the one that takes reference images. Edits audio on the same row (re-voice, re-accent, translate with re-fitted lips, replace BGM). Costs about 1.2x a plain 2.5 generation of the same length: an edit bills at twice the reduced video-reference rate. | `reference-seedance-2-5.md` |
22
30
  <!-- @end:model-routing -->
23
31
 
24
- If the user names a model, use it. Otherwise route by the generated table above.
32
+ This export includes the current Slates image, video and video-edit defaults plus the specialists listed above. If the user names a model, use it when its guide is included. For another model, obtain its current prompting guide rather than adapting unrelated syntax. Otherwise choose from this table for the brief. A model choice does not authorize generation.
25
33
 
26
34
  ## Workflow
27
35
 
28
36
  1. Read the brief. A sentence or a full storyboard is enough.
29
37
  2. If intent is clear, take the fast path: choose the model and write the prompt immediately. Do not interrogate the user for optional detail.
30
38
  3. Load the matching generated reference file before writing:
31
- - `reference-seedance.md`
32
- - `reference-kling.md`
33
- - `reference-nano-banana.md`
39
+ - `reference-seedance-2-5.md` for Seedance 2.5; its grammar points directly to `reference-seedance.md`.
40
+ - `reference-seedance.md` for Seedance 2.0 and shared Seedance grammar.
41
+ - `reference-kling.md` for Kling generation or reference-driven editing.
42
+ - `reference-gpt-image-2-5.md` for GPT Image 2.5.
43
+ - `reference-nano-banana.md` for Nano Banana 2.
44
+ - `reference-omni-flash.md` for Omni Flash generation or prompt-only editing.
45
+ Load only the model and mode being used; the contents list identifies the relevant sections.
34
46
  4. For recurring characters, identity consistency, or character-sheet preparation, also load `reference-character.md`. It owns the exact sheet architecture, background plate, lighting, and evaluation gate.
35
47
  5. For conflict, creatures, crowds, destruction, weapons, public figures, or young characters, also load `reference-content-policy.md` and construct the scene safely from the first word.
36
- 6. If a referenced production guide mentions Slates operations or billing and those tools are not available, use its prompting doctrine and ignore only the transport-specific instruction. Never invent a tool call.
37
- 7. Return one paste-ready prompt. If the concept genuinely requires multiple generations, return the smallest ordered chain (for example: Nano Banana 2 start frame, then Kling motion prompt).
48
+ 6. Check how the target generator accepts prompts and ordered references. Slates-specific billing, default settings and historical measurements describe Slates' endpoints; they are not guarantees about another tool. Prepare the prompt without running a generation unless the user requests one.
49
+ 7. Return one paste-ready prompt. If the concept requires separate generations, return the smallest useful chain of prompts. Preserve the user's visual choices and use model-specific craft to realize them.
38
50
 
39
51
  ## Output
40
52
 
@@ -52,8 +64,8 @@ Then give no more than three short notes covering only decisions the user needs
52
64
  - Never hand-invent a character-sheet prompt when `reference-character.md` already defines the canonical one.
53
65
  - Never silently add weather, props, style, or camera movement the user did not request. If you apply a sane default, name it briefly in the notes.
54
66
  - Never carry image-model lens, aperture, film-stock, or camera-body syntax into Seedance. Follow the model reference's translation rule.
55
- - Never second-stamp Seedance shots. Follow its official `Shot 1 / Shot 2 / Shot 3` structure.
67
+ - Use the chosen Seedance version's timing grammar: 2.0 uses `Shot 1 / Shot 2 / Shot 3`; 2.5 accepts integer-second timestamps. Do not transfer one version's restriction to the other.
56
68
 
57
69
  ## Provenance
58
70
 
59
- Every `reference-*.md` file in this package is generated from `@slatesvideo/shared`. If a generated reference and this router appear to disagree, the generated reference wins and the router must be corrected at its canonical source.
71
+ Every reference is generated from the production Slates skills. The routing table comes from the model registry; character-sheet wording comes from the same builder Slates calls. A craft guide's measurement keeps its recorded scope and date. Routing comes from the table, while the matching guide owns the prompt grammar.
@@ -1,6 +1,6 @@
1
1
  <!-- Generated from the Slates production prompting guides. Do not edit — this file is rebuilt from source. -->
2
2
 
3
- > **This is the real thing.** Every rule below is the working doctrine Slates runs in production against this model — not a summary written for a handout. Slates automates it end to end; the doctrine works by hand too.
3
+ > Generated from the production Slates guide. Model-specific syntax and measured examples apply to the endpoints named below. For another generation tool, check its current schema and reference handling; its limits, billing and defaults may differ.
4
4
 
5
5
  # Character identity sheet — Slates workflow
6
6
 
@@ -32,7 +32,7 @@ Slates generates **one identity sheet per character**, bound as the character's
32
32
 
33
33
  The rule is **kill every competing rendering of the FACE, not every head** — which is why exactly one body panel is headless.
34
34
 
35
- On a deep neutral-grey plate (hex `3a3a3c`, emitted without the `#` — see the sigil warning in Don'ts), flat and shadowless, with catchlights in the eyes, irises never crushed to black, surface texture at the medium's own natural level of detail, broken symmetry, and no over-clean 3D-game-model look. Expression is **a slight natural smile with the teeth just visible** — a closed mouth carries no dental information, so every downstream smiling shot invents teeth, and teeth are person-specific.
35
+ On a deep neutral-grey plate (hex `3a3a3c`, written bare; since the composer fix `#3a3a3c` reads the same — see the resolved composer hazard in Don'ts), flat and shadowless, with catchlights in the eyes, irises never crushed to black, surface texture at the medium's own natural level of detail, broken symmetry, and no over-clean 3D-game-model look. Expression is **a slight natural smile with the teeth just visible** — a closed mouth carries no dental information, so every downstream smiling shot invents teeth, and teeth are person-specific.
36
36
 
37
37
  **Two carve-outs, scoped differently on purpose.** Non-human characters get a natural neutral expression instead of a smile — that one is scoped by *having a human mouth*, so a bipedal robot or humanoid alien is covered. Quadrupeds and non-bipedal characters get a natural standing stance with the head shown on both body panels — that one is *anatomical*. **Both are conditionals the image model evaluates against your reference; neither is a code branch, because the op has no character-kind input.**
38
38
 
@@ -52,16 +52,17 @@ If text only: generate from prompt-only — less consistent, so warn the user.
52
52
 
53
53
  ### Generate the sheet
54
54
 
55
- <!-- @inject:sheet-tool-defaults -->
56
- **What the sheet tools render on** (you do not pick these; omit `model`):
55
+ Attach the source portrait to your image generator and submit this canonical sheet prompt. For a text-only character, add the user's visual description.
57
56
 
58
- - **Character identity sheet:** `gpt-image-2-5-sunburst` at 3k, quality `high`, one 16:9 image.
59
- - **Establishing image:** `gpt-image-2-5-sunburst` at 3k, quality `high`, one 16:9 image.
57
+ ```text
58
+ A single character identity reference sheet of one character, three panels side by side on one plate: a large chest-up portrait on the left at a three-quarter angle (never dead-on), a full-body front view in a relaxed A-pose in the centre, cropped at the collarbone — an invisible-mannequin presentation with just the face cropped out, and a full-body back view on the right with the head and hair fully visible. The portrait is the largest panel and occupies roughly a quarter to a third of the sheet — it is the sole authority for the face, so render it at maximum facial detail. No second rendering of the face anywhere on the sheet. A slight natural smile with the teeth just visible, and identical appearance, wardrobe and hair across all three panels. Preserve the artistic medium and visual style of the reference image (photograph, anime, illustration, 3D render, painterly, etc.). Render on a plain, deep neutral-grey background (hex 3a3a3c) with flat, even, shadowless lighting so the sheet captures the character's identity, not scene lighting. Crisp catchlights in the eyes and open, readable irises — never crushed to black. Render surface texture at the medium's own natural level of detail — skin, hair and fabric should read as material, not airbrushed or plastic. Break perfect symmetry — avoid a mirrored face or dead-square framing. Whatever the medium, avoid the over-clean 3D-game-model look. For non-human characters, use a natural neutral expression instead of a smile. For quadruped or non-bipedal characters, replace the A-pose with a natural standing stance, show the whole animal including the head on both body panels, and keep the same three-panel layout. No text, no labels, no captions, no panel borders.
59
+ ```
60
60
 
61
- Price a sheet for that model at 16:9, with resolution and quality left at their defaults. **Never 4K** — no identity gain at sheet scale, wasted spend.
62
- <!-- @end:sheet-tool-defaults -->
61
+ If the user requests a style transform, replace `Preserve the artistic medium and visual style of the reference image (photograph, anime, illustration, 3D render, painterly, etc.).` with that explicit transform; do not ask for both preservation and transformation. Append only user-specific identity details.
63
62
 
64
- - When the result returns inline, **evaluate it before binding**:
63
+ After inspection, save one approved sheet under the character's name. Attach that same sheet to every shot containing the character and name its reference inline using the selected model's syntax. Keep the attachment order stable and inspect the target tool's reference limits.
64
+
65
+ - When the result returns inline, **evaluate the sheet before reuse**:
65
66
  - Is the portrait clearly the largest panel, and is it off-frontal?
66
67
  - **Is the front body panel cleanly headless** — an empty collar above a normally rendered body, no partial face, no floating jaw, no smeared neck stump? A botched crop is worse than no crop.
67
68
  - **Is the body still there?** Neck, forearms and hands rendered as skin, not an empty outfit floating on nothing. A hollow garment means the invisible-mannequin genre ran unbounded.
@@ -80,12 +81,10 @@ Critically, the app injects **no** wardrobe, expression, or lighting directive.
80
81
  ## Anti-patterns
81
82
 
82
83
  - **Don't** studio-light, white-background, or black-background the sheet. White bleeds into the video and washes out the location; black eats edge detail. Flat, even, shadowless light on a deep neutral grey.
83
- - **Don't** hand-write the sheet prompt when the op will build it — that is how the template and the shipped prompt fork.
84
84
  - **Don't** create a second character image. One canonical identity is what the storyboard pipeline reads.
85
- - **Don't** skip binding. An unbound asset doesn't help downstream.
86
85
  - **Don't** invent character details. Stick to what's in the reference image and the user's description.
87
86
  - **Don't** describe the front panel's crop as an absent head — in `userNotes` or any hand-written variant. The template asks for it as *framing*: **"cropped at the collarbone, an invisible-mannequin presentation with just the face cropped out"**, a standard e-commerce genre with deep training data. **"the head not shown" is a hard 422 on GPT Image** (measured on `gpt-image-2`, the model 2.5 replaced; the classifier is OpenAI's, not the version's, so the rule carries — but nobody has re-run it on Flare or Sunburst) — fal returns `content_policy_violation` with `loc: ["body","prompt"]`, so the text is rejected before any image is read, because an anatomical absence reads as gore to OpenAI's classifier. It passed NB2, which is why the original receipt looked safe: **it was model-scoped.** State an exclusion as a framing choice, never as a missing body part.
88
87
  - **Don't** invoke the invisible-mannequin genre without bounding it to the face. **"an invisible-mannequin presentation where the clothing holds its own shape" removed all the skin** — no neck, no hands, no forearms, a garment floating on nothing — because that *is* the e-commerce genre in full: an empty outfit. **"with just the face cropped out"** keeps the anchor and bounds it. Generalises: a genre anchor imports the whole genre, so name what STAYS, not only what goes.
89
- - **Don't** put `#` or `@` anywhere in prompt text. Both are reference-token sigils in the desktop prompt composer and an unresolved one is **silently deleted** — no error, no log, just missing words. `#3a3a3c` reached fal as `background ()` on a real 2026-07-30 request, meaning the plate value had never been delivered to any model since the composer shipped. Write hex values bare.
88
+ **Resolved composer hazard:** on 2026-07-30, `#3a3a3c` reached fal as `background ()` because unresolved sigils were deleted. The composer now preserves unresolved `#` and `@` text byte-for-byte; a token binds a reference only when it resolves. Literal hex colours and handles are safe. The sheet template keeps its bare hex as a wording choice, not a workaround.
90
89
  - **Don't** use 4K — wastes credits, no quality gain at sheet scale.
91
- - **Don't** feed a multi-view sheet into a Seedance shot that has **several characters in frame** without binding each character to its image and appending the anti-twin constraint — ByteDance documents multi-view assets as a cause of duplicate characters. See `reference-seedance.md`.
90
+ - **Don't** feed a multi-view sheet into a Seedance 2.0 shot that has **several characters in frame** without binding each character to its image and appending the anti-twin constraint; ByteDance documents multi-view assets as a cause of duplicate characters on 2.0. See `reference-seedance.md`. Seedance 2.5 supports multi-view subject references; see `reference-seedance-2-5.md`.
@@ -1,10 +1,10 @@
1
1
  <!-- Generated from the Slates production prompting guides. Do not edit — this file is rebuilt from source. -->
2
2
 
3
- > **This is the real thing.** Every rule below is the working doctrine Slates runs in production against this model — not a summary written for a handout. Slates automates it end to end; the doctrine works by hand too.
3
+ > Generated from the production Slates guide. Model-specific syntax and measured examples apply to the endpoints named below. For another generation tool, check its current schema and reference handling; its limits, billing and defaults may differ.
4
4
 
5
5
  # Content-policy-safe construction — read before any risk-surface prompt
6
6
 
7
- **Never use** — each one is a filter tripwire with a substitution in the table above:
7
+ **Never use**: each one is a filter tripwire with a substitution in the table below:
8
8
  - `civilians in panic`, `crowds fleeing`, `blood`, `gore`, `corpse`
9
9
  - `ignite`, `catch fire`, `on fire` applied to a person — frame body-contact effects as magical or harmless VFX
10
10
  - `candle-like`, `flame-like` and any real object used as a metaphor for an effect
@@ -0,0 +1,191 @@
1
+ <!-- Generated from the Slates production prompting guides. Do not edit — this file is rebuilt from source. -->
2
+
3
+ > Generated from the production Slates guide. Model-specific syntax and measured examples apply to the endpoints named below. For another generation tool, check its current schema and reference handling; its limits, billing and defaults may differ.
4
+
5
+ # GPT Image 2.5 — sheets, grids, and text that actually reads
6
+
7
+ ## Contents
8
+
9
+ - [Which variant](#which-variant)
10
+ - [Quality tiers — always set explicitly](#quality-tiers--always-set-explicitly)
11
+ - [Resolution classes](#resolution-classes)
12
+ - [Reference images — give every one a role, inline, where it is used](#reference-images--give-every-one-a-role-inline-where-it-is-used)
13
+ - [Editing — separate the change from the constraints](#editing--separate-the-change-from-the-constraints)
14
+ - [Prompting for text accuracy](#prompting-for-text-accuracy)
15
+ - [Transparent backgrounds](#transparent-backgrounds)
16
+ - [When an edit must not touch a region at all](#when-an-edit-must-not-touch-a-region-at-all)
17
+ - [Structure a complex prompt in labeled sections](#structure-a-complex-prompt-in-labeled-sections)
18
+ - [Concrete visuals beat mood words](#concrete-visuals-beat-mood-words)
19
+ - [Panels, sheets, and grids](#panels-sheets-and-grids)
20
+ - [🚨 WHAT GETS YOU BLOCKED — read before writing a prompt with a person in it](#-what-gets-you-blocked--read-before-writing-a-prompt-with-a-person-in-it)
21
+ - [Filter regime](#filter-regime)
22
+
23
+ **Card — GPT Image 2.5.** The photoreal front-runner for people, and the readable-text, ordered-panel engine. Structure: subject and action with each reference named where it is used, then any exact copy in quotes, then layout, then light.
24
+
25
+ **Pick the tier.** `flare` is Faster, quality comparable to GPT Image 2: drafts and volume. `sunburst` is Better quality, the most capable: finals, hero frames, photoreal people, multi-reference edits. Use the product default; choose Flare when speed is a stated priority.
26
+
27
+ **The levers**
28
+ 1. **Name each reference inline** — `the woman from image 1`, `lit and graded like image 2`. Never an opening paragraph about what the references are.
29
+ 2. **Quote every string that must render verbatim** — `the jacket reads "SLATES"`. Describe a font's feel, never its name; keep on-image text under about 30 words.
30
+ 3. **Name the layout as a grid** for sheets and panels — `a 3x2 grid of panels, reading left to right, equal gutters`.
31
+ 4. **Set `quality` deliberately.** `high` is the everyday tier; `max` is 4× its price, `xhigh` about 1.8×. Coming from GPT Image 2 the names moved one rung: its `medium` is this `high`.
32
+
33
+ <!-- @inject:cinematic-card -->
34
+ **For a photographic look, use only what this frame needs.** Image models default to clean, evenly lit and fully exposed. Describe what the camera sees, not just gear or mood:
35
+ - **Inspect every reference first.** Write its grade and imperfections in words: darkness, contrast, muddy or true blacks, colour, softness/noise, subject separation. Never grade cleaner or brighter than the look reference unless asked.
36
+ - **One light system** — `low sun behind her`, `her face falls into deep shadow`, `no light in front of her`.
37
+ - **Visible exposure** — `the sky burns out to white`, `dense, slightly crushed shadows`.
38
+ - **Lens name plus effect** — `200mm telephoto`, `peaks loom huge behind her and melt into soft shapes`.
39
+ - **Name every garment and close the foreground.** Omissions invite reference leakage or invented props.
40
+ Bind references inline. A scene reference owns the grade; for a look-only reference, write the new scene's light. References are optional. For owned-frame edits, describe only the change and what stays.
41
+ <!-- @end:cinematic-card -->
42
+
43
+ **Hard constraint:** its own content filter, distinct from Gemini's. Never describe a reference as a photograph of a real person.
44
+
45
+ **Never use:**
46
+ - a font NAME — describe the feel instead, as in: clean geometric sans, high contrast
47
+ - a reference described as a photograph of a real person (`is a photograph of a woman`), or any up-front essay about what each reference is for — name the subject inline where it is used instead, as in: the woman from image 1
48
+ - `8k`, `masterpiece`, `best quality`, `highly detailed` — quality incantations do nothing here either
49
+
50
+ GPT Image's edge is **character-level text accuracy** (~99% on English), ordered panels, and exact element placement — the jobs where every other model garbles a word or shuffles a layout. 2.5 inherits all of it and is better at each.
51
+
52
+ ## Which variant
53
+
54
+ **Speed → Flare. Quality → Sunburst.** That is OpenAI's own routing rule, quoted from its image-prompting guide: *"start with GPT Image 2.5 Flare when speed is the priority, or GPT Image 2.5 Sunburst when demanding quality requirements are the priority."* Same price either way, so the trade is purely latency against quality.
55
+
56
+ 🚨 **FLARE IS NOT AN UPGRADE OVER GPT IMAGE 2 — IT IS THE FAST ONE.** OpenAI, verbatim: *"GPT Image 2.5 Flare is the small model, optimized for speed, with image quality **comparable to** GPT Image 2. GPT Image 2.5 Sunburst is the base model, optimized for quality, with **higher image quality than** GPT Image 2."* Their model pages agree: Flare is *"our fastest model for high-quality, everyday image generation"*, Sunburst *"our most capable model for image generation and editing."* **Sunburst is the seat that beats what we had; Flare is the one that holds it at half the latency.** An earlier revision of this file called Flare "better than GPT Image 2" and sent Sunburst only to multi-reference edits — both wrong, corrected 2026-09-09 against the vendor docs.
57
+
58
+ **Choose for the task.** Use the product default for ordinary work. Flare is an option when speed matters; changing model is not a mandatory draft stage.
59
+
60
+ **Sunburst's widest lead is multi-reference editing** — several references all surviving into one frame, the character-consistency-across-shots problem. Reach for it there first, but that is not the only place it belongs.
61
+
62
+ ⚠️ **The LMArena receipt, scoped.** At launch Arena had Sunburst #1 and Flare #2 across text-to-image, single-image edit and multi-image edit, with margins over GPT Image 2 of **+81 / +47** on multi-image edit (Image Edit Arena: Sunburst 1520, Flare 1491, GPT Image 2 1461). Two caveats were missing and both matter: the baseline is **GPT Image 2 at `medium`, which is this model's `high`** — not its top tier — and the boards were **preliminary, a few thousand votes each**. Arena says Flare beats GPT Image 2; OpenAI says comparable. Route on OpenAI's wording and treat the board as a tiebreaker, not a spec.
63
+
64
+ 🚨 **The GPT Image line is ALSO the photoreal front-runner, and this file said the opposite until 2026-08-24.** **Receipts:** Eric's direct call, plus a head-to-head on the Higgsfield rail where GPT Image 2 at `quality: high`, 2K beat both Nano Banana rails on skin realism for photoreal people — that result is why the whole AI-influencer ad lane generates its plates here. **Route photoreal to this line, not away from it.**
65
+
66
+ **Historical receipt, not a tier recommendation:** the photoreal comparison above used GPT Image 2 at its old `high` tier. It has not been repeated on 2.5 under matched conditions. Start with the product default and test a higher tier only against an unmet requirement; the old comparison does not establish a minimum tier for this model.
67
+
68
+ **What the Banana line still owns:** edit-heavy work, and holding many subjects coherently in one frame. **Not the reference ceiling any more** — that line was true until 2026-09-09, when GPT Image went to its documented 16 against Banana's 14. Route on which model keeps them all recognisable, not on the count.
69
+
70
+ **What would kill this:** a head-to-head at the intended crop going the other way. Per `SKILL.md` § The meta-rule, re-run the evidence test when the roster changes — never carry a ranking forward on reputation. That rule is exactly what the 2026-08-24 correction failed, and exactly what the two ⚠️ notes above are honouring.
71
+
72
+ ## Quality tiers — always set explicitly
73
+
74
+ All five rungs are exposed, and they span ~36× end to end (2k class: $0.0044 → $0.158), which makes this the single biggest cost lever on the model. **The steps are UNEVEN — do not reason about them as a constant multiplier:** ~2.3× `low`→`medium`, ~3.9× `medium`→`high`, ~1.8× `high`→`xhigh`, ~2.25× `xhigh`→`max`. The same ratios hold at every OFFERED resolution class (2k/3k/4k); unoffered 1k differs slightly.
75
+
76
+ | Tier | Use it for |
77
+ |---|---|
78
+ | `low` | Roughest pass — layout and composition checks, throwaway comps. |
79
+ | `medium` | The draft tier. Cheaper than NB2 Lite and available up to 4K, which is why the draft lane moved here. |
80
+ | `high` | General-purpose quality tier. Blind benchmarks on GPT Image 2 put this rung — which it called `medium` — within a hair of `max` (which it called `high`) at a quarter of the cost. Inherited from the old ladder, never re-run on 2.5, and it says nothing about `xhigh`. |
81
+ | `xhigh` | One rung short of the top at about half its price (2k: 4 cr against `max`'s 8). Worth trying before `max`. |
82
+ | `max` | Top of the ladder. Tiny type, dense diagrams, many labelled elements. |
83
+
84
+ ⚠️ **A tier label means different things on different models.** OpenAI: *"The same quality label does not imply the same image quality or response time across models."* Flare at `max` and Sunburst at `max` are not the same picture, and neither matches Nano Banana's idea of "high".
85
+
86
+ 🚨 **The tier NAMES moved between versions and the strings did not.** GPT Image 2's `medium` is this model's `high`; its `high` is this model's `max` — same money, one rung of renaming. For a recipe explicitly written for GPT Image 2, map the old tier before reusing it on 2.5. A current user request for `medium` still means `medium`. Getting this backwards costs picture quality silently: nothing errors, the bill is correct for what was asked, and the image is just worse.
87
+
88
+ Never rely on the provider default. fal's default is `high`, which is correct today — but it is the third rung of five rather than the top of two, so leaning on it means a fal-side change silently reprices you. Slates sends its configured quality explicitly; current defaults live in `SKILL.md`. Your explicit choice overrides them.
89
+
90
+ **Start at the default and change tiers for an unmet requirement.** OpenAI's own procedure: *"If the output falls short, test a higher quality setting. Once it meets your requirements, test lower settings to see whether they preserve acceptable quality while reducing latency. Use `xhigh` or `max` only when they improve an unmet quality requirement within your latency budget."* A higher rung does **not** guarantee a better result on a given prompt. Compare `medium` against `high` when the job is small or dense text; that is where the rungs separate most visibly.
91
+
92
+ ## Resolution classes
93
+
94
+ `1k` = 1024²-class · `2k` = 1920×1080-class · `3k` = 2560×1440-class · `4k` = 3840×2160-class. Pick 2k for most sheets/panels; 4k for print-density grids. 4K exists at every tier and is API-only — even paid ChatGPT can't render it.
95
+
96
+ `1k` is not offered, and the reason is not its price: it is strictly dominated. At 1k you pay more for fewer pixels than at 2k, at **all five tiers**. Don't ask for it.
97
+
98
+ ⚠️ **Above 2560×1440 you are on a path OpenAI marks EXPERIMENTAL.** Verbatim: *"Outputs with more than 3,686,400 total pixels ('2560x1440') are experimental."* That is the whole **4k** class (≈8.0 MP) plus 3k at 4:3/3:4 (≈3.70 MP). It bills normally and it works — but prove the shot at 2k or 3k 16:9 first, and do not be surprised by an odd frame at 4k.
99
+
100
+ **Hard size bounds**, from fal's schema verbatim: each edge ≤ 3840 px, both edges multiples of 16, longer:shorter ratio ≤ 3:1, total pixels between 655,360 and 8,294,400. **The pixel ceiling is the one that actually bites** — the multiple-of-16 rule is documented but NOT enforced, and we have the receipt: 1920×1080 fails it (1080 = 67.5 × 16), is one of fal's own six priced canonical sizes, and metered clean. Slates picks sizes that respect the ceiling; these matter only if you hand-build a request.
101
+
102
+ 🚨 **THE ASPECT RATIO CHANGES THE PRICE ON THIS MODEL, and on no other image model.** OpenAI bills image OUTPUT TOKENS and the count tracks the frame's SHAPE, so at the same resolution class **`1:1` costs about 1.8× and `4:3`/`3:4` about 1.37× what `16:9` costs**; `9:16` costs the same as `16:9`. Metered 2026-09-09 and priced into the cost key, so the quote you get before generating is the real number — but if you are choosing between shapes and the budget is tight, **16:9 or 9:16 is the cheap one.** Every other image model charges the same whatever the shape.
103
+
104
+ ## Reference images — give every one a role, inline, where it is used
105
+
106
+ **Assign a role to every reference image: subject, style, clothing, or background.** This is new emphasis in 2.5 and the highest-leverage change for the 16-reference character lane. An unroled pile of references makes the model guess what each one is for, and it guesses differently every run — which is the drift people mistake for a consistency failure.
107
+
108
+ **The role rides a clause in the scene, not a paragraph in front of it.** *The woman from image 1 cooks on a rocky summit…*, *lit and graded like image 2*. Never open with sentences about what each reference is and what to take or ignore from it: that is the role essay the shared reference rules below forbid, and it drags the sheet's studio light into the scene.
109
+
110
+ **Receipt, 2026-09-15, Sunburst, IMG-A192–A198.** The up-front version returned the studio look; the inline versions were never refused and never came back as a sheet. Two costs, both fixed in words: anything the prompt does not describe is taken from the reference (name every garment), and props nobody asked for appear (say what is in the foreground and that nothing else is). One sheet-only plate kept its described location, which narrows the two-reference rule in `slates-ugc-influencer-ad`. A look reference did far less than a described light. The full ladder is the vault's `cinematic-look-research.md`; the techniques are `slates-cinematic-look`.
111
+
112
+ Reference images route through the edit endpoint, **up to 16** — fal's documented `maxItems`, and the highest reference ceiling of any image seat in Slates (the Banana line takes 14). It was capped at 10 until 2026-09-09, which was never anybody's limit, just a number nobody had checked. The composed "image N" naming applies as everywhere else. Mask-based inpainting exists at the API level but is not surfaced: a mask is something the user has to paint, and there is no painting surface — describe the change instead.
113
+
114
+ ## Editing — separate the change from the constraints
115
+
116
+ **State the change, then list what must survive.** "Change only X," then name the invariants explicitly: identity, geometry, lighting, labels. For precise local edits also pin saturation, contrast, camera angle and surrounding objects — anything you do not pin is fair game for the model to move.
117
+
118
+ **One change per iteration, and restate the constraints every turn.** Cross-turn drift is the named failure mode in OpenAI's own guidance: constraints do not persist across turns by themselves, so a multi-turn refinement that stops restating them will slowly rewrite the frame. This applies directly to multi-turn shot refinement.
119
+
120
+ ## Prompting for text accuracy
121
+
122
+ - **Quote every string that must render verbatim**: `the sign reads "OPEN 24 HOURS"` — quoted strings render most reliably.
123
+ - Say the text appears **once**, and give its position and typography.
124
+ - Spell unusual words letter-by-letter.
125
+ - Add `no extra text, no watermarks`.
126
+ - Specify font *feel*, not font names: "clean geometric sans, high contrast", "hand-painted brush lettering".
127
+ - For dense text (posters, UI mocks), list the copy as ordered lines: `Line 1: "..." Line 2: "..."` — it respects ordering.
128
+ - **Don't bundle unrelated instructions into a text-rendering request.** A prompt that also redesigns the scene competes with the text for attention.
129
+ - Keep total on-image text under ~30 words for perfect accuracy; beyond that, accuracy degrades gracefully but degrades.
130
+
131
+ ## Transparent backgrounds
132
+
133
+ If you need a cut-out rather than a scene, **ask for it explicitly and check the alpha**. OpenAI: request `background=transparent` and use PNG or WebP, then *"check the decoded image's alpha channel, including hair, glass, shadows, and object edges"* — a painted-white backdrop is the common failure and it is not transparency. Say what must NOT appear: *"no solid backdrop, no checkerboard, no scenery, no watermark"*, and do not let the product get restyled while the background is removed. **On every follow-up edit, repeat the transparency requirement** or it gets dropped.
134
+
135
+ ## When an edit must not touch a region at all
136
+
137
+ Prompting alone cannot guarantee pixel-identical pixels. OpenAI's own instruction: if a region must stay exactly as it was, **composite the approved edit back into the original image** rather than asking the model to preserve it. Treat "preserve" language as a strong bias, never a lock.
138
+
139
+ ## Structure a complex prompt in labeled sections
140
+
141
+ For anything with several requirements, OpenAI recommends organising the prompt as **scene, subject, details, constraints** with labeled sections. Same content, easier to read and to change one part without disturbing the rest — which is what makes the one-change-per-iteration rule practical.
142
+
143
+ **Say "photorealistic" or "real photograph" when that is the goal.** It is not inferred from a detailed description; ask for it directly, then describe framing and texture.
144
+
145
+ ## Concrete visuals beat mood words
146
+
147
+ Name materials, lighting, colour and medium. Mood words are cues only — "cinematic", "moody", "epic" tell the model almost nothing on their own. Give scale, atmosphere and colour instead. Camera specs (`85mm`, `f/1.4`) are appearance hints, not a physical simulation; they bias the look, they do not compute optics.
148
+
149
+ **Name the lens and describe its effect, every time.** A lens named alone changed nothing visible (IMG-A195, 2026-09-15); named together with what it does to the picture, it produced real compression and depth of field (IMG-A198). Wording: `slates-cinematic-look` → `compression-as-outcome`, `defocus-as-outcome`.
150
+
151
+ **For people, state body framing and scale**: "full body visible, feet included", "hands naturally gripping the handlebars". This is also the safest way to phrase a crop — see the blocked-phrasings section below.
152
+
153
+ **No special syntax is required.** Prose, JSON and tagged blocks all work equally well, so pick whatever stays maintainable in the caller.
154
+
155
+ ## Panels, sheets, and grids
156
+
157
+ - State the grid explicitly and number the cells: "a 2×3 grid of panels, numbered 1–6, reading left-to-right, top-to-bottom".
158
+ - Give each cell ONE content clause: "Panel 3: the character mid-jump, side view".
159
+ - Character identity sheets: GPT Image holds both the structured panel layout AND photoreal skin, which is why the influencer-ad lane builds its sheets here. Reach for NB2/NB Pro when it is an edit of an existing sheet, or when many subjects have to stay recognisable at once — not for the reference count, which GPT Image now leads at 16.
160
+
161
+ ## 🚨 WHAT GETS YOU BLOCKED — read before writing a prompt with a person in it
162
+
163
+ **Receipt: 24 consecutive attempts on one character, 2026-08-24, same project and same rail.** Eleven were refused with `content_policy_violation` on the fal edit endpoint. The refusals were never about the scene — one of the blocked prompts was a woman standing at a kitchen counter with her hand on it. **Two phrasings were hard blocks, 5 for 5 each, and neither ever passed:**
164
+
165
+ **1. Never describe the reference as a photograph of a real person.**
166
+
167
+ > ❌ `Reference image 1 is a photograph of a woman. Use that exact woman.`
168
+ > ✅ `Reference image 1 is a character identity sheet showing one woman across several panels — the face in the large portrait panel is the authority for her identity. Use that exact woman.`
169
+
170
+ The first reads to the filter as *recreate this real person's likeness*, which is a hard refusal regardless of what the rest of the prompt says. The second signals a fictional character and passes. **This is a wording change only — the reference image can be the same file either way.** One plate flipped from refused to accepted on this single sentence with nothing else altered.
171
+
172
+ **Inline naming sidesteps the question and is now the default:** never describe the reference at all, and name her where she is used (*the woman from image 1*). Six of six Sunburst plates written that way passed on 2026-09-15. Keep the sheet sentence above as the fallback if a refusal appears.
173
+
174
+ **2. Never attach a reference sheet containing a headless body panel.** A sheet whose full-body panels are cropped above the neck is refused every time, even with the correct opener. Regenerate the sheet with the head visible in every panel. Related, and already in this file's sheet guidance: phrase a cropped panel as *framing* (`cropped at the collarbone`), never as *absence* (`the head not shown`).
175
+
176
+ ⚠️ **These refusals were measured on GPT Image 2, not on 2.5.** The classifier belongs to OpenAI rather than to a model version, so the phrasing rules carry — but they are inherited, not re-measured. If Flare or Sunburst accepts one of the blocked phrasings, that is a new receipt to write down here, not a reason to delete this one.
177
+
178
+ **On top of those, ordinary content triggers still apply** and they stack independently — a correct opener does not rescue them:
179
+
180
+ | Refused | Why, and the fix |
181
+ |---|---|
182
+ | A woman sitting on a bed in a bedroom | Domestic + bed reads as intimate. Move her to a chair, a rug, another room. |
183
+ | A knife, even lying flat on a chopping board next to a lemon | The object is the trigger, not the framing. Swap it — a cast-iron pan cleared instantly. |
184
+
185
+ **🚨 Refusals are PROBABILISTIC. Retry once before rewriting a word.** In the same session an identical prompt, identical reference, identical params was refused and then accepted on a straight re-fire. A rejected job returns no file and costs nothing, so a retry is free and a rewrite is not — rewriting first is how you end up changing four variables and learning nothing. **Only redesign after two or three refusals.**
186
+
187
+ **And change ONE thing at a time.** The eleven refusals above took far longer to diagnose than they should have because a reference swap and an opener rewrite shipped in the same call. Isolate on the prompt you actually want, so a pass leaves you with a usable asset instead of a data point.
188
+
189
+ ## Filter regime
190
+
191
+ OpenAI moderate — a third regime distinct from Gemini (NB family) and ByteDance (Seedream). Real-face references pass more readily than Gemini; violence/brand rules are similar. `reference-content-policy.md` applies unchanged.
@@ -1,11 +1,31 @@
1
1
  <!-- Generated from the Slates production prompting guides. Do not edit — this file is rebuilt from source. -->
2
2
 
3
- > **This is the real thing.** Every rule below is the working doctrine Slates runs in production against this model — not a summary written for a handout. Slates automates it end to end; the doctrine works by hand too.
3
+ > Generated from the production Slates guide. Model-specific syntax and measured examples apply to the endpoints named below. For another generation tool, check its current schema and reference handling; its limits, billing and defaults may differ.
4
4
 
5
5
  # Kling V3.0 — prompting
6
6
 
7
- <!-- @card:start -->
8
- **Card — Kling V3.0.** The general default. Define the core subjects clearly at the START and keep those descriptions identical across shots. Up to 15s, up to 6 cuts, and the strongest image-to-video identity hold in the catalogue.
7
+ ## Contents
8
+
9
+ - [Subject definition rule (verbatim, fal blog)](#subject-definition-rule-verbatim-fal-blog)
10
+ - [Dialogue syntax](#dialogue-syntax)
11
+ - [Voice direction formula (Omni)](#voice-direction-formula-omni)
12
+ - [The Immediately keyword (Omni only)](#the-immediately-keyword-omni-only)
13
+ - [Speaker label discipline](#speaker-label-discipline)
14
+ - [Multi-character dialogue (Omni)](#multi-character-dialogue-omni)
15
+ - [Sound effects, ambient noise, music](#sound-effects-ambient-noise-music)
16
+ - [Image-to-video guidance](#image-to-video-guidance)
17
+ - [Multi-shot — what makes them hit](#multi-shot--what-makes-them-hit)
18
+ - [Element references](#element-references)
19
+ - [Reference discipline (character / environment refs)](#reference-discipline-character--environment-refs)
20
+ - [For Kling specifically](#for-kling-specifically)
21
+ - [Negative prompting — has a real field](#negative-prompting--has-a-real-field)
22
+ - [Cinematic tactics](#cinematic-tactics)
23
+ - [Tier choice](#tier-choice)
24
+ - [Benchmark prompt structure](#benchmark-prompt-structure)
25
+ - [Video-to-video EDIT — @Video1 / @ElementN / @ImageN](#video-to-video-edit--video1--elementn--imagen)
26
+ - [Sources](#sources)
27
+
28
+ **Card — Kling V3.0.** Define the core subjects clearly at the START and keep those descriptions identical across shots. Strong image-to-video identity hold; use the current capability surface for duration and multi-shot limits, and the model catalogue for routing.
9
29
 
10
30
  **The five levers**
11
31
  1. **Dialogue in quotes** — `Character says, "exact words here"`. On Omni, direct the voice with `Gender + Age + Voice quality + Speech rate + Emotional tone + Language`: `[Character A: Detective, mid-40s, raspy, slow cadence, weary]: "I've seen this before."`
@@ -19,14 +39,13 @@
19
39
  - `Camera tracks right alongside a cyclist crossing a bridge at dusk. She rises out of the saddle rapidly as the grade steepens. Ambient noise: wind, tyres on wet asphalt, distant traffic.`
20
40
 
21
41
  **Hard constraint:** `Immediately` (Omni only) removes the natural conversational beat between speakers — use it when timing matters and leave it out when it does not. Kling has a real `negativePrompt` field, unlike Seedance; start from the standard block and layer scene-specific suppressions.
22
- <!-- @card:end -->
23
42
 
24
43
  **Never use:**
25
44
  - `SFX: footsteps` and any label-only effect — physical-cause specificity or nothing
26
45
  - a pronoun or synonym for a speaker after the first introduction (`he`, `the agent`) — it causes voice drift; repeat the full label
27
46
  - `single continuous take` — Seedance's phrase, and it fights Kling's multi-shot
28
47
 
29
- Kuaishou's video model. Three tiers: `kling-v3.0-std` (general use, no audio), `kling-v3.0-pro` (higher visual quality, no audio), `kling-v3.0-omni` (multi-character dialogue + audio-visual co-generation).
48
+ Kuaishou's video model. Three tiers: `kling-v3.0-std` (general use, sound supported), `kling-v3.0-pro` (higher visual quality, sound supported), `kling-v3.0-omni` (multi-character dialogue + audio-visual co-generation).
30
49
 
31
50
  Up to 15s. Multi-shot supported (up to 6 cuts in 15s total). Strong on image-to-video — preserves identity, layout, and text from the input image well.
32
51
 
@@ -116,7 +135,9 @@ Miss conditions:
116
135
  - Mixing camera moves within a shot ("pan then orbit then push in")
117
136
  - Extreme wide → extreme close in adjacent shots without reference images
118
137
 
119
- ## Element references (Omni)
138
+ ## Element references
139
+
140
+ Standard and Pro take element references with a first frame; Omni also takes references without one. 4K refuses reference images.
120
141
 
121
142
  Upload 2-4 multi-angle reference photos per character/object. Tag inline:
122
143
 
@@ -181,11 +202,11 @@ Layer scene-specific suppressions on top, and never suppress something the promp
181
202
 
182
203
  ## Tier choice
183
204
 
184
- - **Standard**: general use, no audio
185
- - **Pro**: higher visual quality, no audio
186
- - **Omni**: multi-character dialogue, audio-visual co-gen, language codes, `@elementN` references
205
+ - **Standard**: general use, sound supported
206
+ - **Pro**: higher visual quality, sound supported
207
+ - **Omni**: multi-character dialogue, audio-visual co-gen, language codes, references without a first frame
187
208
 
188
- Pick by capability: need dialogue/audio → Omni; need maximum visual quality silent → Pro; everything else → Standard. Prices change — check current numbers before choosing a tier.
209
+ Every tier can generate dialogue and sound. Sound is on unless `sound: false` is passed; below 4K it bills the audio key, while 4K includes audio. Pick by visual quality and reference needs. Prices change; check current numbers before choosing a tier.
189
210
 
190
211
  ## Benchmark prompt structure
191
212
 
@@ -224,7 +245,7 @@ Rules:
224
245
  - One edit intent per pass. Chain passes for compound changes (each output is itself an editable clip, linked to its parent).
225
246
  - Billing is per second of OUTPUT ≈ the clip length, rounded UP to the next second. A 7.3s clip bills as 8s.
226
247
  - Clip constraints: 3-15s, 720-3840px, MP4/MOV. Agents can pre-trim on the timeline when a clip runs long.
227
- - Routing: Kling edit is the default edit tool (element lock + audio intact); Seedance edit/relocate wins style-transfer-heavy re-imaginings.
248
+ - Route by the required change: this edit seat supports element/style-reference control and original-audio retention. Read the current catalogue for defaults and competing seats.
228
249
 
229
250
  ## Sources
230
251
 
@@ -1,10 +1,25 @@
1
1
  <!-- Generated from the Slates production prompting guides. Do not edit — this file is rebuilt from source. -->
2
2
 
3
- > **This is the real thing.** Every rule below is the working doctrine Slates runs in production against this model — not a summary written for a handout. Slates automates it end to end; the doctrine works by hand too.
3
+ > Generated from the production Slates guide. Model-specific syntax and measured examples apply to the endpoints named below. For another generation tool, check its current schema and reference handling; its limits, billing and defaults may differ.
4
4
 
5
5
  # Nano Banana 2 — cinematic & photorealistic prompting
6
6
 
7
- <!-- @card:start -->
7
+ ## Contents
8
+
9
+ - [Google's 4 official rules (verbatim)](#googles-4-official-rules-verbatim)
10
+ - [Official prompt formula](#official-prompt-formula)
11
+ - [Photorealism positives — what consistently works](#photorealism-positives--what-consistently-works)
12
+ - [The anti-list — phrases that DEGRADE realism](#the-anti-list--phrases-that-degrade-realism)
13
+ - [Negative prompting — there is no field](#negative-prompting--there-is-no-field)
14
+ - [Reference images](#reference-images)
15
+ - [Reference rules (the verified ones)](#reference-rules-the-verified-ones)
16
+ - [For Nano Banana 2 specifically](#for-nano-banana-2-specifically)
17
+ - [Common failure modes + fixes](#common-failure-modes--fixes)
18
+ - [Resolution tactics](#resolution-tactics)
19
+ - [Boring vs cinema — examples](#boring-vs-cinema--examples)
20
+ - [Diagnose repeated failures](#diagnose-repeated-failures)
21
+ - [Family variants — Lite and Pro](#family-variants--lite-and-pro)
22
+
8
23
  **Card — Nano Banana 2 (Gemini 3.1 Flash Image).** Brief it like a creative director, not a tag list. Structure: `Film still from [director] [genre]. Shot on [camera] with [lens]. [Subject and action]. [3-5 specific visual details]. [Lighting — direction + quality]. [Color palette]. [Film stock]. [1-2 word tone].`
9
24
 
10
25
  **The five levers**
@@ -25,7 +40,6 @@ Bind references inline. A scene reference owns the grade; for a look-only refere
25
40
  <!-- @end:cinematic-card -->
26
41
 
27
42
  **Hard constraint:** there is no `negativePrompt` field. Suppress by reframing positively, or inline `without` / `free of`. Knowledge cutoff January 2025 — anything later needs reference images.
28
- <!-- @card:end -->
29
43
 
30
44
  Nano Banana 2 is **Gemini 3.1 Flash Image**. It is **not** Gemini 3 Pro Image; that is Nano Banana **Pro** (`nano-banana-pro`), a separate model with its own seat. NB2 is a language model that outputs pixels — brief it like a creative director, not like a Stable-Diffusion tag-soup tool. The single biggest lever for realism: **specificity that mimics how real photographers and cinematographers describe their work**.
31
45
 
@@ -199,13 +213,17 @@ Identity = a few flat-lit neutral angles; one reference per role, named inline;
199
213
 
200
214
  ✅ **Cinema:** "Extreme close on subject's mouth and nose, 135mm f/2.8, shallow depth of field. Breath pluming out, catching cold light from upper-left key. Lips slightly parted, peach fuzz visible. The breath holds. CineStill 800T halation around catchlights. Waiting."
201
215
 
202
- ## The 3-strike rule
216
+ <!-- @inject:iteration-diagnosis -->
217
+ ## Diagnose repeated failures
218
+
219
+ After three failed attempts at the same requirement, pause unchanged re-rolls and diagnose the source reference, prompt structure, model fit and tool result. Three is a review checkpoint, not a universal limit or proof that the seed cannot matter. Preserve the attempts and name what each test changed.
203
220
 
204
- If three iterations on the same prompt haven't produced what the user wants, stop. Hand back to the user with what you tried and what isn't working. The slot machine doesn't converge — the prompt structure is wrong, not the seed.
221
+ Continue autonomously when the brief is clear, a specific correction is supported and the next request is already authorized. Hand control back when taste or intent cannot be inferred, the next request needs fresh consent, or the available tool cannot meet the requirement. A failed roll never authorizes an additional charge. Follow the existing batch and per-request cost policy.
222
+ <!-- @end:iteration-diagnosis -->
205
223
 
206
224
  ## Family variants — Lite and Pro
207
225
 
208
226
  Everything in this skill applies to the whole Nano Banana family; two variants trade speed/ceiling around NB2 full:
209
227
 
210
228
  - **nano-banana-2-lite** — ~half the price, ~2.7× faster, **1K output only**, max 4 refs. The draft/iteration seat: explore compositions here, then re-run the winner on NB2 full at 2K/4K. Same Gemini filter.
211
- - **nano-banana-pro** — the hero-frame/typography ceiling (~2× NB2, 4K native). NB2 ≈ 95% of Pro; escalate only when spatial composition, cinematic lighting/skin, fine typography-in-scene, or deep multi-element frames must be perfect. Up to 14 refs — it takes a full subject library in one call.
229
+ - **nano-banana-pro**: the hero-frame/typography ceiling (2× NB2 at 1K, 1.33× at 2K, about 1.9× at 4K; 4K native). NB2 ≈ 95% of Pro; escalate only when spatial composition, cinematic lighting/skin, fine typography-in-scene, or deep multi-element frames must be perfect. Up to 14 refs; it takes a full subject library in one call.