@nodaro/prompts 1.5.0 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +670 -11
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +609 -2
- package/dist/index.d.ts +609 -2
- package/dist/index.js +660 -13
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
- package/src/__tests__/assemble-suno-input.test.ts +20 -0
- package/src/__tests__/doctrine-roster-completeness.test.ts +59 -0
- package/src/__tests__/location-convergence-image.test.ts +6 -2
- package/src/__tests__/picker-analyzer-registry.test.ts +11 -1
- package/src/__tests__/provider-prompt-doctrine.test.ts +5 -3
- package/src/__tests__/reference-rules.test.ts +240 -0
- package/src/__tests__/video-reference-features.test.ts +1 -0
- package/src/assemble-suno-input.ts +8 -0
- package/src/factory-snippets/catalog.ts +19 -1
- package/src/index.ts +2 -0
- package/src/picker-analyzer-registry.ts +190 -0
- package/src/picker-wiring.ts +336 -0
- package/src/prompt-wizard-categories.ts +4 -0
- package/src/provider-prompt-doctrine.ts +279 -6
- package/src/reference-rules.ts +226 -0
|
@@ -19,6 +19,12 @@
|
|
|
19
19
|
* - KIE API docs https://docs.kie.ai/market/kling/kling-3-0
|
|
20
20
|
* - Live KIE-path dialogue probe 2026-07-16 (scripted lines spoken verbatim,
|
|
21
21
|
* lip-synced, on both kling-2.6 and kling-3.0 with sound=true)
|
|
22
|
+
* MiniMax Hailuo 3:
|
|
23
|
+
* - KIE API docs https://docs.kie.ai/market/minimax-h3 (text-to-video /
|
|
24
|
+
* image-to-video / reference-to-video — the API contract is the doctrine
|
|
25
|
+
* source: limits, modes, aspect rules, pricing dimensions). No live probe
|
|
26
|
+
* yet — structure guidance mirrors the platform's ordinal-reference
|
|
27
|
+
* conventions rather than vendor style claims.
|
|
22
28
|
*/
|
|
23
29
|
export interface ProviderPromptDoctrine {
|
|
24
30
|
/** MODEL_CATALOG ids this doctrine covers. */
|
|
@@ -32,14 +38,16 @@ export interface ProviderPromptDoctrine {
|
|
|
32
38
|
}
|
|
33
39
|
|
|
34
40
|
const SEEDANCE_2_DOCTRINE: ProviderPromptDoctrine = {
|
|
35
|
-
providers: ["seedance-2", "seedance-2-fast", "seedance-2-mini"],
|
|
36
|
-
heading: "Seedance 2
|
|
41
|
+
providers: ["seedance-2", "seedance-2-fast", "seedance-2-mini", "seedance-2-5"],
|
|
42
|
+
heading: "Seedance 2 (seedance-2, seedance-2-fast, seedance-2-mini, seedance-2-5)",
|
|
37
43
|
tips: [
|
|
38
44
|
"Storyboard complex videos as 'Shot 1: … Shot 2: …' WITHOUT timestamps — timed shots like '(0-3s)' are officially unstable and can break generation.",
|
|
39
45
|
"One camera movement per shot; describe actions per body part with degree ('slowly raises a hand'); express emotion as physical detail, never abstract words.",
|
|
40
46
|
"Native multi-track audio — cue it inline: (background music), <sound effects>, and quoted dialogue.",
|
|
41
47
|
"References go by ordinal (@Image 1, Video 2) in attachment order; earlier = higher priority. Identity = ONE headshot + ONE full-body (multi-view sheets cause ID drift). 4-5 assets total beats maxing the 9/3/3 caps.",
|
|
42
48
|
"No negative-prompt parameter — put constraints in the prompt: 'keep it subtitle-free, do not generate a watermark, do not generate a logo'.",
|
|
49
|
+
"seedance-2-5 only: one shot runs to 30s (the 2.0 SKUs stop at 15s), so storyboard a whole beat instead of planning a stitch. Ref caps are wider (30/10/10), but 4-5 assets still gives the best identity fidelity.",
|
|
50
|
+
"Auto-path formula: Subject → Action → Environment → Camera → Style → Constraints in 60-100 words; ONE camera instruction (chain with 'then'); separate camera motion from subject motion; always add one lighting phrase.",
|
|
43
51
|
],
|
|
44
52
|
doctrine: `Prompt structure (front-load what matters most):
|
|
45
53
|
precise subject → action details → scene/environment → lighting & color tone → camera movement → visual style → image quality → constraints.
|
|
@@ -51,6 +59,12 @@ precise subject → action details → scene/environment → lighting & color to
|
|
|
51
59
|
- Prefer slow, gentle, continuous movements over high-burst action (sprints, big jumps, violent rolls morph). Describe actions per body part with quantified degree: "slowly raises a hand", "pushes hard off the ground". Chain actions with inertia: "uses the momentum of the turn to naturally raise an arm".
|
|
52
60
|
- Express emotion as externalized physical detail, never abstract words: not "very sad" but "lowering the head, shoulders trembling slightly, eyes reddening, fingers clutching the corner of clothing".
|
|
53
61
|
|
|
62
|
+
**Generation differences (seedance-2-5 vs the 2.0 SKUs)**
|
|
63
|
+
- A single 2.5 shot runs to 30s, where every 2.0 SKU stops at 15s. Plan a complete 4-6 shot beat inside ONE generation instead of splitting it into two clips and stitching — no seam to hide, and continuity holds because it never leaves the model.
|
|
64
|
+
- 2.5 also takes far more reference material (30 images / 10 videos / 10 audio vs 9/3/3). Treat that as room for COVERAGE — more distinct characters, locations and props in one shot — not as licence to pile refs onto one identity. The "ONE headshot + ONE full-body, 4-5 assets total" rule above still produces the best likeness on 2.5.
|
|
65
|
+
- 2.5 renders at 480p/720p only: there is no 1080p or 4K tier, so route a job that needs one to seedance-2 (which has both) or upscale afterwards.
|
|
66
|
+
- With a start frame, 2.5 always derives the output aspect from that frame — an explicit aspect ratio is rejected outright, so compose the frame at the ratio you want.
|
|
67
|
+
|
|
54
68
|
**References (when reference media is attached)**
|
|
55
69
|
- Refer to assets by ordinal in attachment order: "@Image 1", "Video 2", "Audio 1". Asset ORDER is priority — put the most identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding — \`{image:1:person}\` resolves to "the person from @image_1" — so a wired reference and its mention stay in sync.)
|
|
56
70
|
- Define each subject once, then reuse the label consistently: 'Define the woman in the red dress in Image 1 as the courier' … 'the courier opens the door'. In multi-character scenes bind every character to its image ("the man from Image 1 hands the box to the woman from Image 2") and append: "do not generate duplicate copies of the same character".
|
|
@@ -71,12 +85,35 @@ precise subject → action details → scene/environment → lighting & color to
|
|
|
71
85
|
**Known weaknesses → workarounds**
|
|
72
86
|
- Text rendering is weak: keep on-screen text to short common words; for exact text or logos, attach the artwork as a reference image and instruct "the logo from Image N stays in the corner unchanged".
|
|
73
87
|
- More than 4 referenced people gets unstable: group people into composite images of ≤4 first (image generation), then reference those composites.
|
|
74
|
-
- Repeated extension degrades quality: prefer high-definition reference assets and avoid stacking many continuations
|
|
88
|
+
- Repeated extension degrades quality: prefer high-definition reference assets and avoid stacking many continuations.
|
|
89
|
+
|
|
90
|
+
**Auto-path formula (community-sourced enrichment — apiyi.com Seedance 2.0 prompt guide,
|
|
91
|
+
higgsfield.ai 4K breakdown; captured 2026-08-09)**
|
|
92
|
+
- Six steps IN ORDER, 60-100 words total (longer measurably degrades): Subject → Action → Environment → Camera → Style → Constraints.
|
|
93
|
+
- ONE primary camera instruction per shot. Compound moves chain with "then": "camera slow tracking then subtle rise" — never two competing verbs. The 8 reliable camera types: push-in, pull-out, pan, tracking, orbit/arc, aerial, handheld, locked-off.
|
|
94
|
+
- SEPARATE camera movement from subject movement — the single biggest quality lever: "The dancer spins slowly. Camera holds fixed framing." — never "spinning camera around a dancing person".
|
|
95
|
+
- Pace with human words (slow / gentle / gradual / smooth / controlled) — never fps numbers or f-stops in the basic path.
|
|
96
|
+
- ALWAYS add one lighting phrase (highest-impact single addition): golden hour / rim light / neon glow / backlit / overcast.
|
|
97
|
+
- Bake stability constraints in: "avoid jitter and bent limbs", "avoid temporal flicker", "avoid identity drift".
|
|
98
|
+
- Ban vague adjectives standing alone ("epic", "amazing", "beautiful", bare "cinematic") — every adjective needs a concrete noun.
|
|
99
|
+
- Mode notes: i2v — skip subject description (the frame has it), focus on motion, append "preserve composition and colors". v2v — describe the style TRANSFORM, keep motion + identity.
|
|
100
|
+
- Advanced (pro path): focal angles in degrees ("47° normal", "29° telephoto", "107° wide"); "180° shutter" for filmic motion blur; handheld texture as "organic shake, micro-drift, subtle dutch"; "white balance locked 5200K"; explicit POSITIVE LOCKS section + "100% matches the reference" for identity-critical shots.
|
|
101
|
+
|
|
102
|
+
**Camera-path control — the magenta-line method (STORYBOARD community technique; the
|
|
103
|
+
manual pro path for precise trajectories, NOT the auto path)**
|
|
104
|
+
1. Duplicate the start frame; on the COPY draw a thick magenta line + arrowhead — the line is the camera's flight path, the arrow its end point. Keep the clean original.
|
|
105
|
+
2. Attach BOTH frames and declare the guide: "Image N contains a magenta line and arrow — a hidden camera trajectory guide, NOT part of the scene. Completely remove it: no line, no arrow, no paint, no trail, no reflection." Skipping the removal order RENDERS the line.
|
|
106
|
+
3. Command the path: "one continuous FPV drone glide following the S-shaped curve as closely as possible — do not shortcut. Camera motion is the priority." Lock the clean frame as first frame + scene reference; lock the destination frame if wired.
|
|
107
|
+
4. Pace with timing blocks ("[00:00-00:02] rise over the rooftop … [00:07-00:09] settle on the doorway") and keep any dialogue SHORT — long lines fight the move.
|
|
108
|
+
5. Assign image-input roles explicitly: first-frame/scene-ref · destination frame · path-guide · 3-6 character-identity refs — and bind identities with @-mentions exactly like the platform's reference pills.`,
|
|
75
109
|
}
|
|
76
110
|
|
|
77
111
|
const KLING_AUDIO_DOCTRINE: ProviderPromptDoctrine = {
|
|
78
|
-
|
|
79
|
-
|
|
112
|
+
// kling-turbo (2.5 Turbo Pro) + kling-master (2.1 Master) are SILENT tiers of
|
|
113
|
+
// the same engine: the structure/motion guidance applies, the Audio block
|
|
114
|
+
// does not (variant note in the doctrine body).
|
|
115
|
+
providers: ["kling", "kling-3.0", "kling-3-omni", "kling-turbo", "kling-master"],
|
|
116
|
+
heading: "Kling 2.1 / 2.5 / 2.6 / 3.0 / 3 Omni (kling, kling-3.0, kling-3-omni, kling-turbo, kling-master)",
|
|
80
117
|
tips: [
|
|
81
118
|
"Kling speaks scripted dialogue natively with lip sync — quote the line and enable sound: [Anna: warm calm voice]: \"We made it.\" On kling/kling-3.0 audio raises the credit cost; kling-3-omni includes it.",
|
|
82
119
|
"Structure prompts as Scene → character/element → Motion → Audio → style. Put ALL sound in one 'Audio:' block: dialogue in quotes, then SFX and ambience described plainly ('rain tapping on glass, no music').",
|
|
@@ -108,12 +145,248 @@ const KLING_AUDIO_DOCTRINE: ProviderPromptDoctrine = {
|
|
|
108
145
|
|
|
109
146
|
**Limits**
|
|
110
147
|
- Kling 2.6 prompts cap at 1000 characters — front-load scene + dialogue and trim style tails first. kling-3.0 accepts long prompts.
|
|
111
|
-
- Durations: 2.6 = 5/10s; 3.0/omni = 3-15s. A spoken line needs roughly 1s per 2-3 words — don't script more dialogue than the clip can hold
|
|
148
|
+
- Durations: 2.6 = 5/10s; 3.0/omni = 3-15s. A spoken line needs roughly 1s per 2-3 words — don't script more dialogue than the clip can hold.
|
|
149
|
+
|
|
150
|
+
**Variant note — kling-turbo (2.5 Turbo Pro) & kling-master (2.1 Master)**
|
|
151
|
+
- SILENT tiers: no audio parameter, so the entire Audio block above does not apply — skip dialogue/SFX cues; the Scene → Character → Motion → Style structure and motion guidance carry over unchanged.
|
|
152
|
+
- Durations 5/10s; kling-turbo takes an end frame (tail_image_url); kling-master is single-image i2v.`,
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
const MINIMAX_H3_DOCTRINE: ProviderPromptDoctrine = {
|
|
156
|
+
providers: ["minimax-h3"],
|
|
157
|
+
heading: "MiniMax Hailuo 3 (minimax-h3)",
|
|
158
|
+
tips: [
|
|
159
|
+
"Natural-language prompts up to 7000 chars. Front-load what matters: subject → action → scene/environment → lighting → camera move → style. One camera movement per shot.",
|
|
160
|
+
"Output 2K (default) or 768P (cheaper tier). Aspect: pure text-to-video needs a concrete ratio (21:9/16:9/4:3/1:1/3:4/9:16); reference runs default to adaptive (match the input).",
|
|
161
|
+
"References go by ordinal in attachment order (@Image 1, Video 1) — earlier = higher priority. Caps: 9 images, 3 videos (2-15s each, ≤15s total), 3 audio clips (≤15s total).",
|
|
162
|
+
"Reference audio drives speech/lip-sync but never rides alone — pair it with an image or video reference. Audio input is free; the first 5 input images are free, extras bill per image.",
|
|
163
|
+
"No negative-prompt parameter — put constraints in the prompt text: 'keep it subtitle-free, do not generate a watermark, do not generate a logo'.",
|
|
164
|
+
],
|
|
165
|
+
doctrine: `Prompt structure (front-load what matters most):
|
|
166
|
+
precise subject → action details → scene/environment → lighting & color tone → camera movement → visual style → image quality → constraints. Prompts are natural language, 1-7000 characters, across all three modes.
|
|
167
|
+
|
|
168
|
+
**Modes (picked automatically from the wired inputs)**
|
|
169
|
+
- First frame and/or last frame connected, nothing else → exact frame mode (image-to-video): the output opens on the first frame and/or closes on the last. The clip's aspect is inferred from the frame — there is no aspect parameter in this mode.
|
|
170
|
+
- ANY reference connected (image, video, or audio) → reference mode (reference-to-video): frames ride along as reference images with a prompt directive binding them to the opening/closing position. Aspect defaults to adaptive (matches the input); a concrete ratio can be forced.
|
|
171
|
+
- Nothing visual connected → text-to-video. A concrete aspect ratio is required (21:9 / 16:9 / 4:3 / 1:1 / 3:4 / 9:16 — no adaptive); Nodaro renders 16:9 unless one is picked.
|
|
172
|
+
|
|
173
|
+
**References (when reference media is attached)**
|
|
174
|
+
- Refer to assets by ordinal in attachment order: "@Image 1", "Video 1", "Audio 1". Put the identity-critical asset first. (In the editor, the \`{image:N:label}\` / \`{video:N}\` / \`{audio:N}\` prompt tokens auto-emit this binding, so a wired reference and its mention stay in sync.)
|
|
175
|
+
- Caps: 9 reference images; 3 reference videos, each 2-15s and ≤15s combined; 3 reference audio clips, ≤15s combined. Reference audio cannot be used alone — it must accompany an image or video reference.
|
|
176
|
+
- Define each subject once, then reuse the label consistently ("the woman from @Image 1 … the woman opens the door"). A focused set of 4-5 assets beats maxing every cap.
|
|
177
|
+
- Billing note: generated seconds AND reference-video input seconds bill at the same per-second rate; the first 5 input images are free and each extra image adds a small surcharge; audio input is free.
|
|
178
|
+
|
|
179
|
+
**Audio**
|
|
180
|
+
- Audio is always generated — there is no on/off toggle. With reference audio attached, the model syncs speech to the supplied track (the platform's lip-sync surface routes image + voice line through this mode automatically).
|
|
181
|
+
- Quoted dialogue in the prompt gives the model the line to perform; describe the voice in words when no reference audio is supplied.
|
|
182
|
+
|
|
183
|
+
**Duration & pacing**
|
|
184
|
+
- 4-15 seconds, integer, default 6. Per-second pricing — a 15s clip costs ~3.7× a 4s clip, so pick the shortest duration that serves the shot.
|
|
185
|
+
- One camera movement type per shot; chain actions with physical, quantified detail ("slowly raises a hand", "pushes hard off the ground") rather than abstract emotion words.
|
|
186
|
+
|
|
187
|
+
**Constraints**
|
|
188
|
+
- There is NO negative-prompt parameter — all constraints belong in the prompt text itself: "keep it subtitle-free, do not generate a watermark, do not generate a logo, stable picture".`,
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
const VEO_31_DOCTRINE: ProviderPromptDoctrine = {
|
|
192
|
+
providers: ["veo3", "veo3.1", "veo3_lite", "veo-1080p", "veo-4k", "veo-extend"],
|
|
193
|
+
heading: "VEO 3.1 — Quality / Fast / Lite (veo3, veo3.1, veo3_lite)",
|
|
194
|
+
tips: [
|
|
195
|
+
"Structure prompts as [Cinematography] + [Subject] + [Action] + [Context] + [Style & Ambiance] — lead with the camera, not the subject (Google's official formula).",
|
|
196
|
+
"Dialogue: quote the exact line with attribution — A woman says, \"We have to leave now.\" (no subtitles). Cue sound as separate lines: SFX: thunder cracks; Ambient noise: quiet hum of a starship bridge.",
|
|
197
|
+
"Multi-shot pacing via timestamp blocks: [00:00-00:02] medium shot… [00:02-00:04] reverse shot… — VEO honors per-window actions inside one 8s generation.",
|
|
198
|
+
"Negative prompting is positive phrasing: not 'no buildings' but 'a desolate landscape with no buildings or roads'. Keep prompts under ~175 words — longer overloads the generation.",
|
|
199
|
+
"Start+end frame: pass both and describe the transition move ('smooth 180-degree arc ending on the POV behind her'). References (ingredients) keep characters/objects consistent and DO generate audio.",
|
|
200
|
+
],
|
|
201
|
+
doctrine: `Prompt structure (Google's official Veo 3.1 formula — lead with the camera):
|
|
202
|
+
[Cinematography] + [Subject] + [Action] + [Context] + [Style & Ambiance].
|
|
203
|
+
Example: "Medium shot, a tired corporate worker, rubbing his temples in exhaustion, in front of a bulky 1980s computer in a cluttered office late at night, lit by harsh fluorescents and the green monitor glow. Retro aesthetic, 1980s color film, slightly grainy."
|
|
204
|
+
|
|
205
|
+
**Camera vocabulary (use the exact terms)**
|
|
206
|
+
- Movement: dolly shot, tracking shot, crane shot, aerial view, slow pan, POV shot, 180-degree arc shot.
|
|
207
|
+
- Composition: wide shot, medium shot, close-up, extreme close-up, two-shot, low angle, high angle.
|
|
208
|
+
- Lens/focus: shallow depth of field, deep focus, wide-angle lens, macro lens, soft focus.
|
|
209
|
+
|
|
210
|
+
**Audio (native, multi-track — dialogue / SFX / ambience)**
|
|
211
|
+
- Dialogue: quote the exact line with attribution: The detective says in a weary voice, "Of all the offices in this town, you had to walk into mine." Append "(no subtitles)" — VEO otherwise tends to burn captions in.
|
|
212
|
+
- Sound effects on their own line: "SFX: a crystal wine glass shatters on the marble floor". Ambient bed: "Ambient noise: rain against the window, distant traffic".
|
|
213
|
+
- Sound can drive the visual ("the sound reverberating through the empty ballroom") — VEO syncs audio-visual timing.
|
|
214
|
+
|
|
215
|
+
**Multi-shot timestamp prompting (inside one generation)**
|
|
216
|
+
- Split the clip into [mm:ss-mm:ss] windows, one action per window:
|
|
217
|
+
[00:00-00:02] Medium shot from behind a young explorer walking toward a clearing.
|
|
218
|
+
[00:02-00:04] Reverse shot of her freckled face, eyes widening.
|
|
219
|
+
[00:04-00:08] Wide, high-angle crane shot revealing the ruins below.
|
|
220
|
+
- 4 / 6 / 8 second clips; budget ~2s per window.
|
|
221
|
+
|
|
222
|
+
**Frames & references**
|
|
223
|
+
- Start + end frame: wire both (imageUrls [start, end]) and describe the camera path between them — "a smooth 180-degree arc shot, starting front-facing and circling to end on the POV from behind her".
|
|
224
|
+
- Reference images (ingredients): attach character/object/scene refs and name them in the prompt ("using the provided images for the detective and the office, …"). Reference runs DO generate audio.
|
|
225
|
+
|
|
226
|
+
**Constraints**
|
|
227
|
+
- Negative prompting works by positive description: write "a desolate landscape with no buildings or roads", not "no buildings".
|
|
228
|
+
- Keep prompts ≤ ~175 words — beyond that instructions conflict and adherence drops. Resolution 720p/1080p; aspect 16:9 / 9:16.
|
|
229
|
+
|
|
230
|
+
Sources: Google Cloud "Ultimate prompting guide for Veo 3.1"
|
|
231
|
+
(cloud.google.com/blog/products/ai-machine-learning/ultimate-prompting-guide-for-veo-3-1),
|
|
232
|
+
KIE VEO API docs (docs.kie.ai/veo3-api/generate-veo-3-video). Captured 2026-08-09.`,
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
const GEMINI_OMNI_DOCTRINE: ProviderPromptDoctrine = {
|
|
236
|
+
providers: ["gemini-omni-video"],
|
|
237
|
+
heading: "Gemini Omni Video (gemini-omni-video)",
|
|
238
|
+
tips: [
|
|
239
|
+
"Multimodal Google video with native audio: text-to-video, image-to-video, and video-edit through the same prompt surface. 4/6/8/10s; 720p/1080p or 4K tier.",
|
|
240
|
+
"Structure like the platform default: subject → action → scene → lighting → camera → style. Quote dialogue lines to have them spoken; describe SFX/ambience plainly in the prompt.",
|
|
241
|
+
"Text-to-video REQUIRES a concrete aspect ratio (the API hard-rejects a missing one); image runs infer aspect from the input.",
|
|
242
|
+
"Reference images ride along as additional imageUrls — bind them in the prompt ('the woman from the first image'). Video-edit: wire a source clip and describe the change, not the whole scene.",
|
|
243
|
+
],
|
|
244
|
+
doctrine: `Prompt structure (no public Google prompt guide exists for the Omni video endpoint —
|
|
245
|
+
the API contract is the doctrine source, like MiniMax H3; structure guidance mirrors the
|
|
246
|
+
platform's ordinal-reference conventions):
|
|
247
|
+
subject → action → scene/environment → lighting → camera movement → style → constraints.
|
|
248
|
+
|
|
249
|
+
**Modes (picked from the wired inputs)**
|
|
250
|
+
- Nothing visual → text-to-video. A concrete aspect ratio is REQUIRED — the API hard-rejects a missing one (Nodaro sends the node's ratio; there is no adaptive).
|
|
251
|
+
- Image(s) wired → image-to-video: the first image anchors the scene; extra images are references — bind each in the prompt ("the woman from the first image", "the interior from the second image").
|
|
252
|
+
- Source video wired → video-edit (served through the same handle): describe the CHANGE ("replace the daylight with dusk, keep the motion and framing"), not a full re-description.
|
|
253
|
+
|
|
254
|
+
**Audio (native)**
|
|
255
|
+
- Audio is generated with the clip. Quote dialogue to have it spoken; describe SFX and ambience plainly ("rain on glass, low synth bed"). State exclusions ("no music") or a bed may be invented.
|
|
256
|
+
|
|
257
|
+
**Duration & tiers**
|
|
258
|
+
- 4 / 6 / 8 / 10 seconds. 720p/1080p tier or the pricier 4K tier — pick 4K only when the deliverable needs it (nearly 2× the credits).
|
|
259
|
+
|
|
260
|
+
Source: KIE gemini-omni-video market contract (parameters + live behavior probed for the
|
|
261
|
+
aspect-ratio hard-reject, see providers/kie/video.ts). Captured 2026-08-09.`,
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
const GROK_IMAGINE_DOCTRINE: ProviderPromptDoctrine = {
|
|
265
|
+
providers: ["grok-i2v", "grok-imagine-video-1.5"],
|
|
266
|
+
heading: "Grok Imagine (grok-i2v, grok-imagine-video-1.5)",
|
|
267
|
+
tips: [
|
|
268
|
+
"Keep prompts simple and direct — Subject + Action + Setting + Camera + Mood. Grok expands the prompt itself; over-specification fights the expander.",
|
|
269
|
+
"Image-to-video: the input image IS the first frame (composition, identity, and style are preserved) — describe the MOTION, don't re-describe the still.",
|
|
270
|
+
"Video 1.5: 1-15s (default 8), 480p/720p/1080p, up to 7 input images (1080p allows only one). Native audio incl. music, SFX, and lip-synced dialogue — quote the line to have it spoken.",
|
|
271
|
+
"Aspect ratio applies to text runs (1:1/16:9/9:16/3:2/2:3/auto); a single input image locks the output to the image's own aspect.",
|
|
272
|
+
],
|
|
273
|
+
doctrine: `Prompt structure (xAI's guidance is minimal by design — the model auto-expands prompts):
|
|
274
|
+
Subject + Action + Setting + Camera + Lighting/Mood, written simply and directly. Reduce
|
|
275
|
+
descriptions of static/unchanged parts — spend the words on what MOVES.
|
|
276
|
+
|
|
277
|
+
**Image-to-video (the primary mode)**
|
|
278
|
+
- The input image is the FIRST FRAME, not a loose reference: composition, subject identity, and visual style carry over. Describe motion and camera only ("she turns toward the window as the camera slowly pushes in"); re-describing the still wastes adherence.
|
|
279
|
+
- grok-imagine-video-1.5 accepts up to 7 images (identity/scene references beyond the first frame); at 1080p only ONE image is allowed.
|
|
280
|
+
|
|
281
|
+
**Audio (video-1.5)**
|
|
282
|
+
- Native audio generates with the clip — background music, SFX, and lip-synced dialogue. Quote the spoken line; describe the music/SFX plainly. There is no audio toggle on the KIE contract — cue (or exclude) sound in the prompt text.
|
|
283
|
+
|
|
284
|
+
**Durations / tiers**
|
|
285
|
+
- grok-i2v: 6 or 10 seconds. grok-imagine-video-1.5: 1-15 seconds in 1s steps (default 8), 480p (default) / 720p / 1080p. Prompt cap 4096 chars — but shorter is better here.
|
|
286
|
+
|
|
287
|
+
Sources: KIE Grok Imagine contracts (docs.kie.ai/market/grok-imagine/image-to-video,
|
|
288
|
+
docs.kie.ai/market/grok-imagine/1-5-preview), xAI Grok Imagine 1.5 release notes
|
|
289
|
+
(x.ai/news/grok-imagine-1-5). Captured 2026-08-09.`,
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
const WAN_DOCTRINE: ProviderPromptDoctrine = {
|
|
293
|
+
providers: ["wan", "wan-i2v", "wan-turbo", "wan-flash", "wan-2.7", "wan-2.7-i2v", "wan-2.7-t2v", "wan-2.7-pro", "wan-videoedit"],
|
|
294
|
+
heading: "Wan 2.x (wan, wan-i2v, wan-turbo, wan-2.7 family)",
|
|
295
|
+
tips: [
|
|
296
|
+
"Alibaba's official formula: Entity + Scene + Motion (basic) → add Aesthetic control + Stylization (advanced). Image-to-video: Motion + Camera only — the image already defines entity and scene.",
|
|
297
|
+
"Sound (2.5+): append a sound description block — voice / sound effects / background music. Avoid scripting EXACT lip-synced lines (official anti-pattern); describe the voice and intent instead.",
|
|
298
|
+
"Multi-shot (2.6/2.7): Overall description + shot number + timestamp + per-shot content. For ONE continuous take write 'Generate single shot' (the shot_type parameter is gone in 2.7).",
|
|
299
|
+
"References go by 'Image 1' / 'Video 1' (capitalized, with a space). Anti-patterns: naming real people, demanding exact legible text, rapid scene changes in one clip, very long choreography.",
|
|
300
|
+
"Style words are strong levers: cyberpunk, claymation, pixel style, felt style, tilt-shift, time-lapse. wan-videoedit: describe the transform, keep motion + identity.",
|
|
301
|
+
],
|
|
302
|
+
doctrine: `Prompt structure (Alibaba Model Studio's official formulas):
|
|
303
|
+
- Basic: Entity + Scene + Motion.
|
|
304
|
+
- Advanced: Entity (description) + Scene (description) + Motion (description) + Aesthetic control + Stylization.
|
|
305
|
+
- Image-to-video: Motion + Camera movement ONLY — the wired image already defines entity and scene; re-describing it fights the frame.
|
|
306
|
+
- Sound (2.5/2.6/2.7): … + Sound description (voice / sound effects / background music).
|
|
307
|
+
- Multi-shot (2.6/2.7): Overall description + Shot number + Timestamp + Shot content.
|
|
308
|
+
- Reference-to-video (2.6/2.7): Reference identifier + Action + Scene + optional Lines + optional BGM.
|
|
309
|
+
|
|
310
|
+
**Camera vocabulary**
|
|
311
|
+
push-in (intimacy/tension), pull-out (scale/isolation), tracking shot, orbit, fixed camera, and compound movements chained sequentially for epic scale.
|
|
312
|
+
|
|
313
|
+
**Single-shot control (2.7)**
|
|
314
|
+
- The shot_type parameter no longer exists — write "Generate single shot" in the prompt to force one continuous take; otherwise 2.7's planner may cut.
|
|
315
|
+
|
|
316
|
+
**References**
|
|
317
|
+
- English format is "Image 1" / "Video 1" (capitalized, space-separated) — bind every wired asset by that name or it may be ignored.
|
|
318
|
+
|
|
319
|
+
**Official anti-patterns (from Alibaba's guide)**
|
|
320
|
+
- Do NOT name specific real people.
|
|
321
|
+
- Do NOT script exact lip-synced dialogue — describe the voice and intent ("she murmurs a reassurance, warm and low") instead of demanding word-perfect lips.
|
|
322
|
+
- Avoid rapid scene changes inside a single clip, very long choreographed sequences, and demands for exactly legible on-screen text.
|
|
323
|
+
|
|
324
|
+
**Stylization**
|
|
325
|
+
- Style words are strong levers: cyberpunk, line-art illustration, felt style, 3D cartoon, pixel style, puppet animation, claymation, black-and-white animation, tilt-shift, time-lapse.
|
|
326
|
+
|
|
327
|
+
Source: Alibaba Cloud Model Studio — "Text-to-video / image-to-video prompt guide"
|
|
328
|
+
(alibabacloud.com/help/en/model-studio/text-to-video-prompt). Captured 2026-08-09.`,
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
const HAPPYHORSE_DOCTRINE: ProviderPromptDoctrine = {
|
|
332
|
+
providers: ["happyhorse", "happyhorse-i2v", "happyhorse-ref2v", "happyhorse-edit"],
|
|
333
|
+
heading: "HappyHorse 1.1 (happyhorse, happyhorse-i2v, happyhorse-ref2v)",
|
|
334
|
+
tips: [
|
|
335
|
+
"Any-language prompts up to 5000 chars (2500 Chinese) — excess is silently truncated, so front-load subject → action → scene → camera → style.",
|
|
336
|
+
"3-15 seconds per second of billing; 720p or 1080p; ratios 16:9 / 9:16 / 1:1 / 4:3 / 3:4. Pick the shortest duration that serves the shot.",
|
|
337
|
+
"ref2v is one of the few true REFERENCE modes on the roster: wired refs keep identity across the clip — bind each reference explicitly in the prompt.",
|
|
338
|
+
"No published vendor style guide — the platform's standard structure applies; keep one camera move per shot and quantify motion physically.",
|
|
339
|
+
],
|
|
340
|
+
doctrine: `Prompt structure (no public HappyHorse prompt guide exists — the KIE API contract is the
|
|
341
|
+
doctrine source; platform-standard structure applies):
|
|
342
|
+
subject → action → scene/environment → lighting → camera movement → style → constraints.
|
|
343
|
+
|
|
344
|
+
**Contract facts (KIE, per-mode pages)**
|
|
345
|
+
- Prompts: any language, up to 5000 non-Chinese / 2500 Chinese characters — excess is TRUNCATED silently, so put the load-bearing content first.
|
|
346
|
+
- Duration 3-15s (default 5), billed per second. Resolution 720p / 1080p (default). Aspect 16:9 (default) / 9:16 / 1:1 / 4:3 / 3:4.
|
|
347
|
+
- Modes: text-to-video (happyhorse), image-to-video (happyhorse-i2v), reference-to-video (happyhorse-ref2v) — ref2v preserves wired identities; name each reference in the prompt so the binding is explicit.
|
|
348
|
+
|
|
349
|
+
**Style guidance (platform-standard, honestly generic)**
|
|
350
|
+
- One camera movement per shot; physical, quantified action ("slowly raises a hand") over abstract emotion words; state exclusions ("no on-screen text, no watermark") in the prompt.
|
|
351
|
+
|
|
352
|
+
Source: KIE HappyHorse 1.1 contracts (docs.kie.ai/market/happyhorse/text-to-video,
|
|
353
|
+
…/happyhorse-1-1/image-to-video, …/happyhorse-1-1/reference-to-video). Captured 2026-08-09.`,
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
const RUNWAY_KIE_DOCTRINE: ProviderPromptDoctrine = {
|
|
357
|
+
providers: ["runway-kie", "runway-extend", "runway-aleph"],
|
|
358
|
+
heading: "Runway via KIE (runway-kie)",
|
|
359
|
+
tips: [
|
|
360
|
+
"Prompt cap is 1800 chars; KIE's own guidance: be specific about subject, action, style, and setting. No native audio — plan sound as a separate pass.",
|
|
361
|
+
"Durations 5 or 10s with a hard trade-off: 10s cannot be 1080p, 1080p cannot exceed 5s — pick per deliverable.",
|
|
362
|
+
"Text runs REQUIRE an aspect ratio (16:9/4:3/1:1/3:4/9:16); image runs IGNORE it — the input image dictates output dimensions.",
|
|
363
|
+
"Image-to-video treats the image as the anchor frame: describe motion and camera, not the still.",
|
|
364
|
+
],
|
|
365
|
+
doctrine: `Prompt structure (KIE contract guidance): "be specific about subject, action, style, and
|
|
366
|
+
setting" — subject → action → scene → camera → style, within the 1800-character cap.
|
|
367
|
+
|
|
368
|
+
**Contract facts (KIE Runway endpoint)**
|
|
369
|
+
- Duration 5 or 10 seconds; quality 720p or 1080p — 10s@1080p does NOT exist (10s forces 720p; 1080p forces 5s). Choose by deliverable: crisp hero shot → 5s/1080p; longer beat → 10s/720p.
|
|
370
|
+
- Text-to-video REQUIRES aspectRatio (16:9 / 4:3 / 1:1 / 3:4 / 9:16). Image-to-video IGNORES aspectRatio — the input image dictates output dimensions.
|
|
371
|
+
- No audio is generated — score/SFX are a separate pass (merge-video-audio / video-sfx downstream).
|
|
372
|
+
|
|
373
|
+
**Style guidance**
|
|
374
|
+
- The image input anchors composition and identity — describe the motion ("she pushes the door open as the camera tracks left"), not the still.
|
|
375
|
+
- Keep one continuous camera idea per clip; front-load the subject and action.
|
|
376
|
+
|
|
377
|
+
Source: KIE Runway contract (docs.kie.ai/runway-api/generate-ai-video). Captured 2026-08-09.`,
|
|
112
378
|
}
|
|
113
379
|
|
|
114
380
|
export const PROVIDER_PROMPT_DOCTRINES: readonly ProviderPromptDoctrine[] = [
|
|
115
381
|
SEEDANCE_2_DOCTRINE,
|
|
116
382
|
KLING_AUDIO_DOCTRINE,
|
|
383
|
+
MINIMAX_H3_DOCTRINE,
|
|
384
|
+
VEO_31_DOCTRINE,
|
|
385
|
+
GEMINI_OMNI_DOCTRINE,
|
|
386
|
+
GROK_IMAGINE_DOCTRINE,
|
|
387
|
+
WAN_DOCTRINE,
|
|
388
|
+
HAPPYHORSE_DOCTRINE,
|
|
389
|
+
RUNWAY_KIE_DOCTRINE,
|
|
117
390
|
]
|
|
118
391
|
|
|
119
392
|
const DOCTRINE_BY_PROVIDER: ReadonlyMap<string, ProviderPromptDoctrine> = new Map(
|
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* REFERENCE RULES — the two short blocks that decide whether a multi-reference
|
|
3
|
+
* image obeys its references, and whether it looks like a film frame or a
|
|
4
|
+
* posed photograph.
|
|
5
|
+
*
|
|
6
|
+
* Both were measured on gpt-image-2 at 2K, against one deliberately hard brief:
|
|
7
|
+
* four references (two people, a street-fashion shot, and a composite holding a
|
|
8
|
+
* wine glass, a smartphone, a hotel suite AND two other women's faces), with
|
|
9
|
+
* wardrobe swapped BETWEEN the two people. 36 draws, 2026-08-06. Scored on five
|
|
10
|
+
* criteria: both faces correct and unblended, each person's wardrobe, the
|
|
11
|
+
* props, the location, and one coherent composition.
|
|
12
|
+
*
|
|
13
|
+
* ─── REFERENCE_RULES ───────────────────────────────────────────────────────
|
|
14
|
+
*
|
|
15
|
+
* Two wordings already existed in this codebase and NEITHER was best. Each was
|
|
16
|
+
* written without knowledge of the other:
|
|
17
|
+
*
|
|
18
|
+
* the `reference-lock` snippet — default-deny + likeness + compose.
|
|
19
|
+
* **0 of 4** on the hardest criterion (move
|
|
20
|
+
* a garment from one reference to another).
|
|
21
|
+
* Flawless on the other four, which is what
|
|
22
|
+
* made it feel like it worked.
|
|
23
|
+
* gvp's `referenceRules` — default-deny + face rules + likeness +
|
|
24
|
+
* "expression, gaze and pose follow the
|
|
25
|
+
* scene". **1 of 4**.
|
|
26
|
+
* the two MERGED — **4 of 4** (Fisher exact vs the snippet,
|
|
27
|
+
* p ≈ 0.03).
|
|
28
|
+
*
|
|
29
|
+
* The arms isolate: swapping the performance clause for the compose clause took
|
|
30
|
+
* it 1/4 → 4/4, and adding the face rules took 0/4 → 4/4. Each half does real
|
|
31
|
+
* work and neither is sufficient alone. The performance clause is dropped here
|
|
32
|
+
* because a composite has no scene to perform — it belongs to gvp's SCENE lane,
|
|
33
|
+
* where a subject must act the beat.
|
|
34
|
+
*
|
|
35
|
+
* ─── SCENE_FRAME_RULE ──────────────────────────────────────────────────────
|
|
36
|
+
*
|
|
37
|
+
* The separate failure Tal reported: everyone faces the lens and the result
|
|
38
|
+
* reads as an artfully posed photograph rather than a frame out of a film.
|
|
39
|
+
* Four fixes were measured against the block above (which is 0/4 on gaze by
|
|
40
|
+
* itself):
|
|
41
|
+
*
|
|
42
|
+
* "This image is a scene start frame of a video." 0/4 gaze, **0/3 identity**
|
|
43
|
+
* "Film still from a feature film." gaze ✗ (sampled)
|
|
44
|
+
* "A candid moment, unposed, nobody aware of…" identity ✗ (sampled)
|
|
45
|
+
* rewriting the verbs (drinking, not holding) 4/4 gaze, 3/4 identity
|
|
46
|
+
* **"Nobody looks at the camera."** **4/4 gaze, 4/4 identity,
|
|
47
|
+
* 4/4 wardrobe**
|
|
48
|
+
*
|
|
49
|
+
* Only the short negative is free. Everything longer — a medium declaration, a
|
|
50
|
+
* genre label, a mood sentence — competes with the reference bindings and the
|
|
51
|
+
* references lose: the "scene start frame" arm put a woman from INSIDE a
|
|
52
|
+
* reference into the lead role in every draw. This is the same effect gvp's
|
|
53
|
+
* block already recorded on a different brief ("more instruction bought LESS
|
|
54
|
+
* compliance"), now measured twice.
|
|
55
|
+
*
|
|
56
|
+
* KEPT SEPARATE FROM THE RULES, deliberately. A portrait, a piece to camera, or
|
|
57
|
+
* a product shot WANTS the eyeline; suppressing it is a creative choice, not a
|
|
58
|
+
* correctness rule. They are two controls, defaulting independently.
|
|
59
|
+
*/
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Default-deny + likeness + compose. Goes ahead of the scene it governs.
|
|
63
|
+
*
|
|
64
|
+
* OPT-IN, NOT INJECTED. An earlier cut of this work defaulted it on for every
|
|
65
|
+
* image node; Tal's call was that the snippets are enough, and he is right for
|
|
66
|
+
* a wording still being learned — a platform-wide default would apply the
|
|
67
|
+
* current best guess to every job at once, and the evidence for WHICH block is
|
|
68
|
+
* best is still split (see REFERENCE_RULES_MULTI_PERSON).
|
|
69
|
+
*
|
|
70
|
+
* ONE STRING, TWO CONSUMERS — the `reference-lock` factory snippet and
|
|
71
|
+
* gvp/recast's own grounding. They drifted apart once already (the snippet and
|
|
72
|
+
* gvp each carried a different version, and neither knew the other existed),
|
|
73
|
+
* which is the whole reason this is a constant rather than two literals.
|
|
74
|
+
*/
|
|
75
|
+
export const REFERENCE_RULES =
|
|
76
|
+
"Do not use anything from reference images unless specified explicitly. " +
|
|
77
|
+
"All elements taken from reference images must preserve likeness. " +
|
|
78
|
+
"Compose them naturally into a single image."
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* The same rules PLUS the two face clauses — for briefs that move elements
|
|
82
|
+
* BETWEEN people.
|
|
83
|
+
*
|
|
84
|
+
* NOT THE DEFAULT, and the reason is a genuine conflict in the evidence that
|
|
85
|
+
* should not be quietly resolved by whoever edits this next.
|
|
86
|
+
*
|
|
87
|
+
* The controlled comparison: on a four-reference brief with two faces and a
|
|
88
|
+
* garment crossing from one person to the other, this block moved that garment
|
|
89
|
+
* 4 times in 4 draws and {@link REFERENCE_RULES} moved it 0 in 4 (Fisher exact,
|
|
90
|
+
* p ≈ 0.03). On the other four criteria — faces correct, own wardrobe, props,
|
|
91
|
+
* location, composition — the two were identical, 12/12 both ways.
|
|
92
|
+
*
|
|
93
|
+
* Tal's counter-evidence: across his own volume of real jobs, the shorter block
|
|
94
|
+
* works better IN GENERAL. Both hold. The face clauses earn their place exactly
|
|
95
|
+
* when two faces are in play and elements cross between them; on a single
|
|
96
|
+
* subject, a product or a landscape, "do not alter face structure" is dead
|
|
97
|
+
* weight, and this codebase has already measured twice that more instruction
|
|
98
|
+
* buys less compliance.
|
|
99
|
+
*
|
|
100
|
+
* So the DEFAULT follows the population (the short block) and this is the tool
|
|
101
|
+
* you reach for on the composition case. Settling "in general" properly needs
|
|
102
|
+
* the same method applied across brief TYPES, not more draws of one brief.
|
|
103
|
+
*/
|
|
104
|
+
export const REFERENCE_RULES_MULTI_PERSON =
|
|
105
|
+
"Do not take anything from the reference images unless specified explicitly. " +
|
|
106
|
+
"Do not alter face structure. Do not blend faces. " +
|
|
107
|
+
"Preserve the likeness of every element taken. " +
|
|
108
|
+
"Compose them naturally into a single image."
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* THE FRAMING PREFIX — a sentence fragment that swallows the scene after it.
|
|
112
|
+
*
|
|
113
|
+
* "Medium wide film still of" + "The person from reference image A wears…"
|
|
114
|
+
* reads as one phrase, and that is the whole trick. Lead with the SHOT SIZE
|
|
115
|
+
* (see {@link filmStillPrefix}).
|
|
116
|
+
*
|
|
117
|
+
* NO "CINEMATIC", and that word was in here for about ten minutes. "Film still"
|
|
118
|
+
* describes the KIND of picture — a frame lifted out of moving footage, which
|
|
119
|
+
* is what a UGC clip, a product video and a documentary all are too. "Cinematic"
|
|
120
|
+
* describes a REGISTER, and imposing one is wrong for most briefs. It is also
|
|
121
|
+
* the exact category of word every measured arm punished: a genre claim. The same idea as a standalone SENTENCE
|
|
122
|
+
* ("Film still from a feature film." / "This image is a scene start frame of a
|
|
123
|
+
* video.") measured badly-to-catastrophically: the sentence competes with the
|
|
124
|
+
* reference bindings and the references lose — the "scene start frame" arm put
|
|
125
|
+
* a face from INSIDE a reference into the lead role in 3 of 3 draws. The prefix
|
|
126
|
+
* costs nothing because it never makes a separate claim.
|
|
127
|
+
*
|
|
128
|
+
* What it buys is not just the eyeline. Tal's distinction, which is sharper
|
|
129
|
+
* than the binary this was first scored on: a subject may be turned toward the
|
|
130
|
+
* lens and still be IN the scene rather than presenting to the viewer. The
|
|
131
|
+
* prefix also moves staging, depth and the quality of light — things an
|
|
132
|
+
* eyeline rule cannot reach.
|
|
133
|
+
*/
|
|
134
|
+
export const FILM_STILL_PREFIX = "Film still of"
|
|
135
|
+
|
|
136
|
+
/**
|
|
137
|
+
* The prefix with a SHOT SIZE in front — "Extreme wide cinematic film still of
|
|
138
|
+
* …", "Medium close-up cinematic film still of …".
|
|
139
|
+
*
|
|
140
|
+
* Leading with the framing is standard practice and Tal reports it better
|
|
141
|
+
* again. HONEST STATUS: the POSITION is measured (a prefix costs nothing where
|
|
142
|
+
* a standalone claim cost the lead's identity 3 of 3); the shot-size word is
|
|
143
|
+
* his experience, not a controlled arm. The shot size is safe in a way a genre
|
|
144
|
+
* label is not — it says how the picture is FRAMED, which the scene needs
|
|
145
|
+
* anyway, rather than what kind of production it belongs to.
|
|
146
|
+
*/
|
|
147
|
+
export function filmStillPrefix(shotSize?: string): string {
|
|
148
|
+
const shot = shotSize?.trim()
|
|
149
|
+
return shot ? `${shot} film still of` : FILM_STILL_PREFIX
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/**
|
|
153
|
+
* The eyeline suppressor. Five words, measured free — it costs nothing on
|
|
154
|
+
* identity, wardrobe or composition, which none of the longer phrasings
|
|
155
|
+
* managed.
|
|
156
|
+
*
|
|
157
|
+
* DO NOT EXPAND THIS. Every attempt to say more about what the picture IS cost
|
|
158
|
+
* reference fidelity; the sentence works BECAUSE it constrains exactly one
|
|
159
|
+
* thing and claims nothing about medium, genre or mood.
|
|
160
|
+
*/
|
|
161
|
+
export const SCENE_FRAME_RULE = "Nobody looks at the camera."
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* A LOOK TAIL — film stock, lens, light, palette — appended AFTER everything.
|
|
165
|
+
*
|
|
166
|
+
* An example to edit, not a universal: a different film wants a different
|
|
167
|
+
* stock. What generalises is the POSITION.
|
|
168
|
+
*
|
|
169
|
+
* THIS one is allowed to say "cinematic" — imposing a register is its entire
|
|
170
|
+
* job, and a user opts in by name. {@link filmStillPrefix} may not, because it
|
|
171
|
+
* is a DEFAULT: "film still" describes the kind of picture (a frame out of
|
|
172
|
+
* moving footage — true of a UGC clip and a documentary too), while
|
|
173
|
+
* "cinematic" describes a register most briefs did not ask for.
|
|
174
|
+
*
|
|
175
|
+
* ─── THE ONE RULE ALL OF THIS TURNED OUT TO BE ─────────────────────────────
|
|
176
|
+
*
|
|
177
|
+
* Position decides whether an instruction helps or fights the references:
|
|
178
|
+
*
|
|
179
|
+
* 1. RULES FIRST — default-deny, ahead of the bindings it governs.
|
|
180
|
+
* 2. FRAMING AS A PREFIX that swallows the subject ("Film still of …").
|
|
181
|
+
* 3. LOOK LAST — stock, lens, lighting, palette.
|
|
182
|
+
* 4. A separate CLAIM in the middle competes with the reference bindings,
|
|
183
|
+
* and the references lose.
|
|
184
|
+
*
|
|
185
|
+
* That is why "This image is a scene start frame of a video." (a standalone
|
|
186
|
+
* claim, mid-prompt) cost the lead's identity in 3 of 3 draws while "Film still
|
|
187
|
+
* of" (a prefix) costs nothing, and why this tail is free at the end. It
|
|
188
|
+
* matches the published guidance independently — Subject → Action →
|
|
189
|
+
* Surroundings → Camera/Lighting → Atmosphere — and gvp's engine had already
|
|
190
|
+
* found rule 3 the hard way: "THE MEDIUM GOES LAST… a rendering directive at
|
|
191
|
+
* the end has nothing after it to argue with."
|
|
192
|
+
*
|
|
193
|
+
* RECAST DOES NOT NEED THIS SNIPPET. `lookDirective` already appends a tail
|
|
194
|
+
* built from the look the ANALYSER observed in the source film — stock, grade,
|
|
195
|
+
* lens, lighting — which beats a hand-written one because it is that film's
|
|
196
|
+
* own look. This exists for the platform's image nodes, which have no analyser.
|
|
197
|
+
*/
|
|
198
|
+
export const CINEMATIC_LOOK_TAIL =
|
|
199
|
+
"Shot on Super 16mm Kodak 7298 with Canon K35 lenses, soft naturalistic window light mixed with " +
|
|
200
|
+
"dim tungsten practicals and muted fluorescent spill, earthy muted palette with faded greens, " +
|
|
201
|
+
"warm skin tones and gentle shadow fall-off"
|
|
202
|
+
|
|
203
|
+
/**
|
|
204
|
+
* Compose the blocks a caller wants, in a stable order, ready to prepend.
|
|
205
|
+
* Returns `""` when everything is off, so a caller can prepend blind.
|
|
206
|
+
*
|
|
207
|
+
* No route calls this today — the platform ships these as SNIPPETS a user
|
|
208
|
+
* inserts. It exists for gvp/recast (which builds its grounding block in code)
|
|
209
|
+
* and for whatever calls this next, so the ordering rule lives in one place:
|
|
210
|
+
* rules first, eyeline rule after them, scene after both.
|
|
211
|
+
*/
|
|
212
|
+
export function referenceRulesBlock(opts?: {
|
|
213
|
+
/** Default-deny + likeness + compose. Absent = ON. */
|
|
214
|
+
referenceRules?: boolean
|
|
215
|
+
/** Add the two face clauses — for briefs moving elements between people. */
|
|
216
|
+
multiPerson?: boolean
|
|
217
|
+
/** "Nobody looks at the camera." Absent = OFF (a portrait wants the eyeline). */
|
|
218
|
+
sceneFrame?: boolean
|
|
219
|
+
}): string {
|
|
220
|
+
const parts: string[] = []
|
|
221
|
+
if (opts?.referenceRules !== false) {
|
|
222
|
+
parts.push(opts?.multiPerson === true ? REFERENCE_RULES_MULTI_PERSON : REFERENCE_RULES)
|
|
223
|
+
}
|
|
224
|
+
if (opts?.sceneFrame === true) parts.push(SCENE_FRAME_RULE)
|
|
225
|
+
return parts.join(" ")
|
|
226
|
+
}
|