@nodaro/prompts 1.6.0 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +576 -8
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +396 -2
- package/dist/index.d.ts +396 -2
- package/dist/index.js +573 -10
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
- package/src/__tests__/doctrine-roster-completeness.test.ts +59 -0
- package/src/__tests__/picker-analyzer-registry.test.ts +11 -1
- package/src/__tests__/provider-prompt-doctrine.test.ts +5 -3
- package/src/index.ts +1 -0
- package/src/picker-analyzer-registry.ts +190 -0
- package/src/picker-wiring.ts +336 -0
- package/src/provider-prompt-doctrine.ts +227 -4
|
@@ -0,0 +1,336 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Parameter-picker WIRING — the single source of truth for how each picker
|
|
3
|
+
* node type binds to its catalog: value field(s), defaults, entries, grouping,
|
|
4
|
+
* and per-field option lists for multi-dim pickers.
|
|
5
|
+
*
|
|
6
|
+
* Pure data, no React. Extracted from the app's parameter-picker-registry so
|
|
7
|
+
* THREE consumers share one definition and cannot drift:
|
|
8
|
+
* 1. The app's community fallback registry (chip pickers, no rich previews).
|
|
9
|
+
* 2. `@nodaroai/picker-ui`'s rich registry (attaches preview/Picker renderers).
|
|
10
|
+
* 3. Nodaro Cine's builder panels.
|
|
11
|
+
*
|
|
12
|
+
* Renderers (preview components, multi-dim Picker components) deliberately do
|
|
13
|
+
* NOT live here — presentation is the private package's concern; this file is
|
|
14
|
+
* the public vocabulary.
|
|
15
|
+
*/
|
|
16
|
+
import {
|
|
17
|
+
ANIMALS,
|
|
18
|
+
ANIMAL_SUBCATEGORY_LABELS,
|
|
19
|
+
ANIMAL_SUBCATEGORY_ORDER,
|
|
20
|
+
VEHICLES,
|
|
21
|
+
VEHICLE_SUBCATEGORY_LABELS,
|
|
22
|
+
VEHICLE_SUBCATEGORY_ORDER,
|
|
23
|
+
WEAPONS,
|
|
24
|
+
WEAPON_SUBCATEGORY_LABELS,
|
|
25
|
+
WEAPON_SUBCATEGORY_ORDER,
|
|
26
|
+
FURNITURE,
|
|
27
|
+
FURNITURE_SUBCATEGORY_LABELS,
|
|
28
|
+
FURNITURE_SUBCATEGORY_ORDER,
|
|
29
|
+
type I18nCatalogId,
|
|
30
|
+
} from "@nodaro/shared"
|
|
31
|
+
import { SETTINGS, SETTING_CATEGORY_LABELS } from "./setting.js"
|
|
32
|
+
import { MATERIALS, MATERIAL_CATEGORY_LABELS, MATERIAL_CATEGORY_ORDER } from "./materials.js"
|
|
33
|
+
import { ATMOSPHERES } from "./atmosphere.js"
|
|
34
|
+
import { STYLES } from "./style.js"
|
|
35
|
+
import { MOODS, MOOD_CATEGORY_LABELS, MOOD_CATEGORY_ORDER } from "./mood.js"
|
|
36
|
+
import { POSES, POSE_CATEGORY_LABELS, POSE_CATEGORY_ORDER } from "./pose.js"
|
|
37
|
+
import { CAMERA_MOTIONS, CAMERA_MOTION_CATEGORY_LABELS, CAMERA_MOTION_CATEGORY_ORDER } from "./camera-motions.js"
|
|
38
|
+
import { LENSES } from "./lens.js"
|
|
39
|
+
import { CAMERA_FORMATS } from "./camera-format.js"
|
|
40
|
+
import { COLOR_LOOKS, COLOR_LOOK_CATEGORY_LABELS, COLOR_LOOK_CATEGORY_ORDER } from "./color-look.js"
|
|
41
|
+
import { PHOTOGRAPHERS, PHOTOGRAPHER_CATEGORY_LABELS, PHOTOGRAPHER_CATEGORY_ORDER } from "./photographer.js"
|
|
42
|
+
import { AESTHETICS, AESTHETIC_CATEGORY_LABELS, AESTHETIC_CATEGORY_ORDER } from "./aesthetic.js"
|
|
43
|
+
import { ERAS, ERA_CATEGORY_LABELS, ERA_CATEGORY_ORDER } from "./era.js"
|
|
44
|
+
import { PHOTO_GENRES, PHOTO_GENRE_CATEGORY_LABELS, PHOTO_GENRE_CATEGORY_ORDER } from "./photo-genre.js"
|
|
45
|
+
import { BACKDROPS, BACKDROP_CATEGORY_LABELS, BACKDROP_CATEGORY_ORDER } from "./backdrop.js"
|
|
46
|
+
import { HELD_PROPS, HELD_PROP_CATEGORY_LABELS, HELD_PROP_CATEGORY_ORDER } from "./held-prop.js"
|
|
47
|
+
import { RENDER_QUALITIES } from "./render-quality.js"
|
|
48
|
+
import { COMPOSITION_EFFECTS } from "./composition-effects.js"
|
|
49
|
+
import { POST_PROCESS_EFFECTS } from "./post-process-effects.js"
|
|
50
|
+
import { ACTION_FX, ACTION_FX_CATEGORY_LABELS, ACTION_FX_CATEGORY_ORDER } from "./action-fx.js"
|
|
51
|
+
import { LOOP_SUBJECTS, LOOP_SUBJECT_CATEGORY_LABELS, LOOP_SUBJECT_CATEGORY_ORDER } from "./loop-subject.js"
|
|
52
|
+
import { TRANSITIONS, TRANSITION_CATEGORY_LABELS, TRANSITION_CATEGORY_ORDER } from "./transitions.js"
|
|
53
|
+
import { CHARACTER_FX, CHARACTER_FX_CATEGORY_LABELS, CHARACTER_FX_CATEGORY_ORDER } from "./character-fx.js"
|
|
54
|
+
import { FRAMINGS, FRAMING_CATEGORY_ORDER, FRAMING_FIELD_BY_CATEGORY } from "./framing.js"
|
|
55
|
+
import { LIGHTINGS, LIGHTING_CATEGORY_ORDER, LIGHTING_FIELD_BY_CATEGORY } from "./lighting.js"
|
|
56
|
+
import { STYLINGS, STYLING_DIMENSION_ORDER, STYLING_FIELD_BY_DIMENSION } from "./styling.js"
|
|
57
|
+
import { PEOPLE, PERSON_DIMENSION_ORDER, PERSON_FIELD_BY_DIMENSION } from "./person.js"
|
|
58
|
+
import { TEMPORALS } from "./temporal.js"
|
|
59
|
+
import { EXPOSURE_SETTINGS } from "./exposure-settings.js"
|
|
60
|
+
import { MUSIC_GENRES, MUSIC_ERAS } from "./music-genre.js"
|
|
61
|
+
import { MUSIC_ENERGIES, MUSIC_EMOTIONS, MUSIC_VIBES } from "./music-mood.js"
|
|
62
|
+
import { INSTRUMENTS, PRODUCTION_STYLES, VOCAL_PRESENCE, SINGING_STYLES } from "./instrumentation.js"
|
|
63
|
+
import { VOICE_AGES, VOICE_GENDERS, VOICE_LANGUAGES, VOICE_ACCENTS, VOICE_TIMBRES } from "./voice-character.js"
|
|
64
|
+
import { VOICE_PACES, VOICE_EMOTIONS, VOICE_ARCHETYPES } from "./voice-delivery.js"
|
|
65
|
+
|
|
66
|
+
/** One selectable catalog entry as picker UIs consume it. */
|
|
67
|
+
export interface PickerWiringEntry {
|
|
68
|
+
readonly id: string
|
|
69
|
+
readonly label: string
|
|
70
|
+
readonly description: string
|
|
71
|
+
/** Optional group key for category headers. */
|
|
72
|
+
readonly group?: string
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
export interface SingleDimPickerWiring {
|
|
76
|
+
readonly kind: "single"
|
|
77
|
+
readonly nodeType: string
|
|
78
|
+
readonly label: string
|
|
79
|
+
readonly valueField: string
|
|
80
|
+
readonly defaultValue: string
|
|
81
|
+
readonly catalogId: I18nCatalogId
|
|
82
|
+
readonly entries: ReadonlyArray<PickerWiringEntry>
|
|
83
|
+
readonly groupOrder?: ReadonlyArray<string>
|
|
84
|
+
readonly groupLabels?: Readonly<Record<string, string>>
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
export interface MultiDimPickerWiring {
|
|
88
|
+
readonly kind: "multi"
|
|
89
|
+
readonly nodeType: string
|
|
90
|
+
readonly label: string
|
|
91
|
+
/** Data fields the picker reads/writes — used to slice node.data into a value object. */
|
|
92
|
+
readonly fields: ReadonlyArray<string>
|
|
93
|
+
readonly catalogId: I18nCatalogId
|
|
94
|
+
/** Flat id→label list — used to resolve ids into labels for summary chips. */
|
|
95
|
+
readonly catalogEntries: ReadonlyArray<{ readonly id: string; readonly label: string }>
|
|
96
|
+
/** Per-field option lists — lets a data-only consumer (community fallback,
|
|
97
|
+
* Cine simple mode) render one select per field without knowing the
|
|
98
|
+
* catalog's discriminator scheme. */
|
|
99
|
+
readonly fieldOptions: Readonly<Record<string, ReadonlyArray<{ readonly id: string; readonly label: string }>>>
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
export type PickerWiring = SingleDimPickerWiring | MultiDimPickerWiring
|
|
103
|
+
|
|
104
|
+
function mapCat<T extends { id: string; label: string; description: string }>(
|
|
105
|
+
arr: ReadonlyArray<T>,
|
|
106
|
+
groupKey?: keyof T,
|
|
107
|
+
): ReadonlyArray<PickerWiringEntry> {
|
|
108
|
+
return arr.map((e) => ({
|
|
109
|
+
id: e.id,
|
|
110
|
+
label: e.label,
|
|
111
|
+
description: e.description,
|
|
112
|
+
group: groupKey ? (e[groupKey] as unknown as string) : undefined,
|
|
113
|
+
}))
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
function flatCat<T extends { id: string; label: string }>(
|
|
117
|
+
arr: ReadonlyArray<T>,
|
|
118
|
+
): ReadonlyArray<{ id: string; label: string }> {
|
|
119
|
+
return arr.map((e) => ({ id: e.id, label: e.label }))
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/** Group a discriminated catalog into per-field option lists via key→field map. */
|
|
123
|
+
function optionsByDiscriminator<T extends { id: string; label: string }>(
|
|
124
|
+
arr: ReadonlyArray<T>,
|
|
125
|
+
keyOf: (e: T) => string,
|
|
126
|
+
fieldByKey: Readonly<Record<string, string>>,
|
|
127
|
+
): Record<string, Array<{ id: string; label: string }>> {
|
|
128
|
+
const out: Record<string, Array<{ id: string; label: string }>> = {}
|
|
129
|
+
for (const e of arr) {
|
|
130
|
+
const field = fieldByKey[keyOf(e)]
|
|
131
|
+
if (!field) continue
|
|
132
|
+
;(out[field] ??= []).push({ id: e.id, label: e.label })
|
|
133
|
+
}
|
|
134
|
+
return out
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
export const SINGLE_PICKER_WIRING: ReadonlyArray<SingleDimPickerWiring> = [
|
|
138
|
+
// -------- "Look" family --------
|
|
139
|
+
{ kind: "single", nodeType: "setting", label: "Setting", valueField: "setting", defaultValue: "forest", catalogId: "setting", entries: mapCat(SETTINGS, "category"), groupOrder: ["indoor", "urban", "nature", "fantastical"], groupLabels: SETTING_CATEGORY_LABELS },
|
|
140
|
+
{ kind: "single", nodeType: "atmosphere", label: "Atmosphere", valueField: "atmosphere", defaultValue: "clear", catalogId: "atmosphere", entries: mapCat(ATMOSPHERES) },
|
|
141
|
+
{ kind: "single", nodeType: "style", label: "Style", valueField: "style", defaultValue: "cinematic", catalogId: "style", entries: mapCat(STYLES) },
|
|
142
|
+
{ kind: "single", nodeType: "color-look", label: "Color / Look", valueField: "colorLook", defaultValue: "warm", catalogId: "color-look", entries: mapCat(COLOR_LOOKS, "category"), groupOrder: COLOR_LOOK_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: COLOR_LOOK_CATEGORY_LABELS as Record<string, string> },
|
|
143
|
+
{ kind: "single", nodeType: "mood", label: "Mood", valueField: "mood", defaultValue: "calm", catalogId: "mood", entries: mapCat(MOODS, "category"), groupOrder: MOOD_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: MOOD_CATEGORY_LABELS },
|
|
144
|
+
{ kind: "single", nodeType: "photographer", label: "Photographer / Artist", valueField: "photographer", defaultValue: "tim-walker", catalogId: "photographer", entries: mapCat(PHOTOGRAPHERS, "category"), groupOrder: PHOTOGRAPHER_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: PHOTOGRAPHER_CATEGORY_LABELS },
|
|
145
|
+
{ kind: "single", nodeType: "aesthetic", label: "Aesthetic / Microtrend", valueField: "aesthetic", defaultValue: "y2k", catalogId: "aesthetic", entries: mapCat(AESTHETICS, "category"), groupOrder: AESTHETIC_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: AESTHETIC_CATEGORY_LABELS },
|
|
146
|
+
{ kind: "single", nodeType: "era", label: "Era / Period", valueField: "era", defaultValue: "1990s-mall", catalogId: "era", entries: mapCat(ERAS, "category"), groupOrder: ERA_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: ERA_CATEGORY_LABELS },
|
|
147
|
+
{ kind: "single", nodeType: "photo-genre", label: "Photo Genre", valueField: "photoGenre", defaultValue: "fashion-editorial", catalogId: "photo-genre", entries: mapCat(PHOTO_GENRES, "category"), groupOrder: PHOTO_GENRE_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: PHOTO_GENRE_CATEGORY_LABELS },
|
|
148
|
+
{ kind: "single", nodeType: "backdrop", label: "Backdrop", valueField: "backdrop", defaultValue: "white-seamless", catalogId: "backdrop", entries: mapCat(BACKDROPS, "category"), groupOrder: BACKDROP_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: BACKDROP_CATEGORY_LABELS },
|
|
149
|
+
{ kind: "single", nodeType: "render-quality", label: "Render Quality", valueField: "renderQuality", defaultValue: "raytracing", catalogId: "render-quality", entries: mapCat(RENDER_QUALITIES) },
|
|
150
|
+
{ kind: "single", nodeType: "composition-effects", label: "Composition Effect", valueField: "compositionEffect", defaultValue: "bursting-through-frame", catalogId: "composition-effects", entries: mapCat(COMPOSITION_EFFECTS) },
|
|
151
|
+
{ kind: "single", nodeType: "action-fx", label: "Action FX", valueField: "actionFx", defaultValue: "earthquake-tremor", catalogId: "action-fx", entries: mapCat(ACTION_FX, "category"), groupOrder: ACTION_FX_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: ACTION_FX_CATEGORY_LABELS as Record<string, string> },
|
|
152
|
+
{ kind: "single", nodeType: "loop-subject", label: "Loop Subject", valueField: "loopSubject", defaultValue: "tunnel", catalogId: "loop-subject", entries: mapCat(LOOP_SUBJECTS, "category"), groupOrder: LOOP_SUBJECT_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: LOOP_SUBJECT_CATEGORY_LABELS as Record<string, string> },
|
|
153
|
+
{ kind: "single", nodeType: "post-process-effects", label: "Post-Process Effect", valueField: "postProcess", defaultValue: "vignette-soft", catalogId: "post-process-effects", entries: mapCat(POST_PROCESS_EFFECTS) },
|
|
154
|
+
|
|
155
|
+
// -------- "Camera" family --------
|
|
156
|
+
{ kind: "single", nodeType: "camera-motion", label: "Camera Motion", valueField: "cameraMotion", defaultValue: "static", catalogId: "camera-motions", entries: mapCat(CAMERA_MOTIONS, "category"), groupOrder: CAMERA_MOTION_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: CAMERA_MOTION_CATEGORY_LABELS },
|
|
157
|
+
{ kind: "single", nodeType: "lens", label: "Lens", valueField: "lens", defaultValue: "normal-50mm", catalogId: "lens", entries: mapCat(LENSES) },
|
|
158
|
+
{ kind: "single", nodeType: "camera-format", label: "Camera / Film", valueField: "cameraFormat", defaultValue: "35mm-film", catalogId: "camera-format", entries: mapCat(CAMERA_FORMATS) },
|
|
159
|
+
{ kind: "single", nodeType: "transition", label: "Transition", valueField: "transition", defaultValue: "auto", catalogId: "transitions", entries: mapCat(TRANSITIONS, "category"), groupOrder: TRANSITION_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: TRANSITION_CATEGORY_LABELS },
|
|
160
|
+
{ kind: "single", nodeType: "character-fx", label: "Character FX", valueField: "characterFx", defaultValue: "auto", catalogId: "character-fx", entries: mapCat(CHARACTER_FX, "category"), groupOrder: CHARACTER_FX_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: CHARACTER_FX_CATEGORY_LABELS },
|
|
161
|
+
|
|
162
|
+
// -------- "Subject / Object" family --------
|
|
163
|
+
{ kind: "single", nodeType: "pose", label: "Pose", valueField: "pose", defaultValue: "standing-upright", catalogId: "pose", entries: mapCat(POSES, "category"), groupOrder: POSE_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: POSE_CATEGORY_LABELS },
|
|
164
|
+
{ kind: "single", nodeType: "material", label: "Material", valueField: "material", defaultValue: "silk", catalogId: "materials", entries: mapCat(MATERIALS, "category"), groupOrder: MATERIAL_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: MATERIAL_CATEGORY_LABELS },
|
|
165
|
+
{ kind: "single", nodeType: "animal", label: "Animal", valueField: "animal", defaultValue: "dog-golden-retriever", catalogId: "animals", entries: mapCat(ANIMALS, "subcategory"), groupOrder: ANIMAL_SUBCATEGORY_ORDER as ReadonlyArray<string>, groupLabels: ANIMAL_SUBCATEGORY_LABELS },
|
|
166
|
+
{ kind: "single", nodeType: "vehicle", label: "Vehicle", valueField: "vehicle", defaultValue: "sedan", catalogId: "vehicles", entries: mapCat(VEHICLES, "subcategory"), groupOrder: VEHICLE_SUBCATEGORY_ORDER as ReadonlyArray<string>, groupLabels: VEHICLE_SUBCATEGORY_LABELS },
|
|
167
|
+
{ kind: "single", nodeType: "weapon", label: "Weapon", valueField: "weapon", defaultValue: "katana", catalogId: "weapons", entries: mapCat(WEAPONS, "subcategory"), groupOrder: WEAPON_SUBCATEGORY_ORDER as ReadonlyArray<string>, groupLabels: WEAPON_SUBCATEGORY_LABELS },
|
|
168
|
+
{ kind: "single", nodeType: "furniture", label: "Furniture", valueField: "furniture", defaultValue: "sofa", catalogId: "furniture", entries: mapCat(FURNITURE, "subcategory"), groupOrder: FURNITURE_SUBCATEGORY_ORDER as ReadonlyArray<string>, groupLabels: FURNITURE_SUBCATEGORY_LABELS },
|
|
169
|
+
{ kind: "single", nodeType: "held-prop", label: "Held Prop", valueField: "heldProp", defaultValue: "smartphone", catalogId: "held-prop", entries: mapCat(HELD_PROPS, "category"), groupOrder: HELD_PROP_CATEGORY_ORDER as ReadonlyArray<string>, groupLabels: HELD_PROP_CATEGORY_LABELS },
|
|
170
|
+
]
|
|
171
|
+
|
|
172
|
+
// Derived from the canonical dimension/category order + field map (same source
|
|
173
|
+
// the picker components and the describe-to-picker analyzer use). Deriving
|
|
174
|
+
// (not hand-listing) means these can never drift from the node-data shape.
|
|
175
|
+
const STYLING_FIELDS = STYLING_DIMENSION_ORDER.map((d) => STYLING_FIELD_BY_DIMENSION[d])
|
|
176
|
+
const PERSON_FIELDS = PERSON_DIMENSION_ORDER.map((d) => PERSON_FIELD_BY_DIMENSION[d])
|
|
177
|
+
const LIGHTING_FIELDS = LIGHTING_CATEGORY_ORDER.map((c) => LIGHTING_FIELD_BY_CATEGORY[c])
|
|
178
|
+
|
|
179
|
+
// Literal category→field maps for the two catalogs whose discriminators don't
|
|
180
|
+
// ship a shared FIELD_BY map (kept tiny + local; a wrong key = the field
|
|
181
|
+
// silently missing from fieldOptions, which the wiring guard test catches).
|
|
182
|
+
const TEMPORAL_FIELD_BY_CATEGORY: Readonly<Record<string, string>> = {
|
|
183
|
+
speed: "temporalSpeed",
|
|
184
|
+
freeze: "temporalFreeze",
|
|
185
|
+
direction: "temporalDirection",
|
|
186
|
+
shutter: "temporalShutter",
|
|
187
|
+
}
|
|
188
|
+
const EXPOSURE_FIELD_BY_CATEGORY: Readonly<Record<string, string>> = {
|
|
189
|
+
aperture: "aperture",
|
|
190
|
+
"shutter-speed": "shutterSpeed",
|
|
191
|
+
iso: "isoValue",
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
export const MULTI_PICKER_WIRING: ReadonlyArray<MultiDimPickerWiring> = [
|
|
195
|
+
{
|
|
196
|
+
kind: "multi",
|
|
197
|
+
nodeType: "framing",
|
|
198
|
+
label: "Framing",
|
|
199
|
+
fields: ["shotSize", "angle", "coverage", "composition", "vantage"],
|
|
200
|
+
catalogId: "framing",
|
|
201
|
+
catalogEntries: flatCat(FRAMINGS),
|
|
202
|
+
fieldOptions: optionsByDiscriminator(FRAMINGS, (e) => e.category, FRAMING_FIELD_BY_CATEGORY as Readonly<Record<string, string>>),
|
|
203
|
+
},
|
|
204
|
+
{
|
|
205
|
+
kind: "multi",
|
|
206
|
+
nodeType: "lighting",
|
|
207
|
+
label: "Lighting",
|
|
208
|
+
fields: LIGHTING_FIELDS,
|
|
209
|
+
catalogId: "lighting",
|
|
210
|
+
catalogEntries: flatCat(LIGHTINGS),
|
|
211
|
+
fieldOptions: optionsByDiscriminator(LIGHTINGS, (e) => e.category, LIGHTING_FIELD_BY_CATEGORY as Readonly<Record<string, string>>),
|
|
212
|
+
},
|
|
213
|
+
{
|
|
214
|
+
kind: "multi",
|
|
215
|
+
nodeType: "person",
|
|
216
|
+
label: "Person",
|
|
217
|
+
fields: PERSON_FIELDS,
|
|
218
|
+
catalogId: "person",
|
|
219
|
+
catalogEntries: flatCat(PEOPLE),
|
|
220
|
+
fieldOptions: optionsByDiscriminator(PEOPLE, (e) => e.dimension, PERSON_FIELD_BY_DIMENSION as Readonly<Record<string, string>>),
|
|
221
|
+
},
|
|
222
|
+
{
|
|
223
|
+
kind: "multi",
|
|
224
|
+
nodeType: "styling",
|
|
225
|
+
label: "Styling",
|
|
226
|
+
fields: STYLING_FIELDS,
|
|
227
|
+
catalogId: "styling",
|
|
228
|
+
catalogEntries: flatCat(STYLINGS),
|
|
229
|
+
fieldOptions: optionsByDiscriminator(STYLINGS, (e) => e.dimension, STYLING_FIELD_BY_DIMENSION as Readonly<Record<string, string>>),
|
|
230
|
+
},
|
|
231
|
+
{
|
|
232
|
+
kind: "multi",
|
|
233
|
+
nodeType: "temporal",
|
|
234
|
+
label: "Temporal",
|
|
235
|
+
fields: ["temporalSpeed", "temporalFreeze", "temporalDirection", "temporalShutter"],
|
|
236
|
+
catalogId: "temporal",
|
|
237
|
+
catalogEntries: flatCat(TEMPORALS),
|
|
238
|
+
fieldOptions: optionsByDiscriminator(TEMPORALS, (e) => e.category, TEMPORAL_FIELD_BY_CATEGORY),
|
|
239
|
+
},
|
|
240
|
+
{
|
|
241
|
+
kind: "multi",
|
|
242
|
+
nodeType: "exposure-settings",
|
|
243
|
+
label: "Exposure Settings",
|
|
244
|
+
fields: ["aperture", "shutterSpeed", "isoValue"],
|
|
245
|
+
catalogId: "exposure-settings",
|
|
246
|
+
catalogEntries: flatCat(EXPOSURE_SETTINGS),
|
|
247
|
+
fieldOptions: optionsByDiscriminator(EXPOSURE_SETTINGS, (e) => e.category, EXPOSURE_FIELD_BY_CATEGORY),
|
|
248
|
+
},
|
|
249
|
+
// -------- "Sound" family --------
|
|
250
|
+
// Music Genre catalog is hierarchical: flatten genres + every subgenre +
|
|
251
|
+
// eras so summary chips can resolve any selected id back to a human label.
|
|
252
|
+
{
|
|
253
|
+
kind: "multi",
|
|
254
|
+
nodeType: "music-genre",
|
|
255
|
+
label: "Music Genre",
|
|
256
|
+
fields: ["genre", "subgenre", "era"],
|
|
257
|
+
catalogId: "music-genre",
|
|
258
|
+
catalogEntries: [
|
|
259
|
+
...flatCat(MUSIC_GENRES),
|
|
260
|
+
...MUSIC_GENRES.flatMap((g) => g.subgenres.map((s) => ({ id: s.id, label: s.label }))),
|
|
261
|
+
...flatCat(MUSIC_ERAS),
|
|
262
|
+
],
|
|
263
|
+
fieldOptions: {
|
|
264
|
+
genre: flatCat(MUSIC_GENRES),
|
|
265
|
+
subgenre: MUSIC_GENRES.flatMap((g) => g.subgenres.map((s) => ({ id: s.id, label: s.label }))),
|
|
266
|
+
era: flatCat(MUSIC_ERAS),
|
|
267
|
+
},
|
|
268
|
+
},
|
|
269
|
+
{
|
|
270
|
+
kind: "multi",
|
|
271
|
+
nodeType: "music-mood",
|
|
272
|
+
label: "Music Mood",
|
|
273
|
+
fields: ["energy", "emotion", "vibe"],
|
|
274
|
+
catalogId: "music-mood",
|
|
275
|
+
catalogEntries: [...flatCat(MUSIC_ENERGIES), ...flatCat(MUSIC_EMOTIONS), ...flatCat(MUSIC_VIBES)],
|
|
276
|
+
fieldOptions: {
|
|
277
|
+
energy: flatCat(MUSIC_ENERGIES),
|
|
278
|
+
emotion: flatCat(MUSIC_EMOTIONS),
|
|
279
|
+
vibe: flatCat(MUSIC_VIBES),
|
|
280
|
+
},
|
|
281
|
+
},
|
|
282
|
+
{
|
|
283
|
+
kind: "multi",
|
|
284
|
+
nodeType: "instrumentation",
|
|
285
|
+
label: "Instrumentation",
|
|
286
|
+
fields: ["instruments", "production", "vocalPresence", "singingStyle"],
|
|
287
|
+
catalogId: "instrumentation",
|
|
288
|
+
catalogEntries: [...flatCat(INSTRUMENTS), ...flatCat(PRODUCTION_STYLES), ...flatCat(VOCAL_PRESENCE), ...flatCat(SINGING_STYLES)],
|
|
289
|
+
fieldOptions: {
|
|
290
|
+
instruments: flatCat(INSTRUMENTS),
|
|
291
|
+
production: flatCat(PRODUCTION_STYLES),
|
|
292
|
+
vocalPresence: flatCat(VOCAL_PRESENCE),
|
|
293
|
+
singingStyle: flatCat(SINGING_STYLES),
|
|
294
|
+
},
|
|
295
|
+
},
|
|
296
|
+
{
|
|
297
|
+
kind: "multi",
|
|
298
|
+
nodeType: "voice-character",
|
|
299
|
+
label: "Voice Character",
|
|
300
|
+
fields: ["age", "gender", "language", "accent", "timbre"],
|
|
301
|
+
catalogId: "voice-character",
|
|
302
|
+
catalogEntries: [...flatCat(VOICE_AGES), ...flatCat(VOICE_GENDERS), ...flatCat(VOICE_LANGUAGES), ...flatCat(VOICE_ACCENTS), ...flatCat(VOICE_TIMBRES)],
|
|
303
|
+
fieldOptions: {
|
|
304
|
+
age: flatCat(VOICE_AGES),
|
|
305
|
+
gender: flatCat(VOICE_GENDERS),
|
|
306
|
+
language: flatCat(VOICE_LANGUAGES),
|
|
307
|
+
accent: flatCat(VOICE_ACCENTS),
|
|
308
|
+
timbre: flatCat(VOICE_TIMBRES),
|
|
309
|
+
},
|
|
310
|
+
},
|
|
311
|
+
{
|
|
312
|
+
kind: "multi",
|
|
313
|
+
nodeType: "voice-delivery",
|
|
314
|
+
label: "Voice Delivery",
|
|
315
|
+
fields: ["pace", "emotion", "archetype"],
|
|
316
|
+
catalogId: "voice-delivery",
|
|
317
|
+
catalogEntries: [...flatCat(VOICE_PACES), ...flatCat(VOICE_EMOTIONS), ...flatCat(VOICE_ARCHETYPES)],
|
|
318
|
+
fieldOptions: {
|
|
319
|
+
pace: flatCat(VOICE_PACES),
|
|
320
|
+
emotion: flatCat(VOICE_EMOTIONS),
|
|
321
|
+
archetype: flatCat(VOICE_ARCHETYPES),
|
|
322
|
+
},
|
|
323
|
+
},
|
|
324
|
+
]
|
|
325
|
+
|
|
326
|
+
export const ALL_PICKER_WIRING: ReadonlyArray<PickerWiring> = [
|
|
327
|
+
...SINGLE_PICKER_WIRING,
|
|
328
|
+
...MULTI_PICKER_WIRING,
|
|
329
|
+
]
|
|
330
|
+
|
|
331
|
+
const WIRING_MAP = new Map<string, PickerWiring>(ALL_PICKER_WIRING.map((w) => [w.nodeType, w]))
|
|
332
|
+
|
|
333
|
+
export function getPickerWiring(nodeType: string | undefined | null): PickerWiring | undefined {
|
|
334
|
+
if (!nodeType) return undefined
|
|
335
|
+
return WIRING_MAP.get(nodeType)
|
|
336
|
+
}
|
|
@@ -47,6 +47,7 @@ const SEEDANCE_2_DOCTRINE: ProviderPromptDoctrine = {
|
|
|
47
47
|
"References go by ordinal (@Image 1, Video 2) in attachment order; earlier = higher priority. Identity = ONE headshot + ONE full-body (multi-view sheets cause ID drift). 4-5 assets total beats maxing the 9/3/3 caps.",
|
|
48
48
|
"No negative-prompt parameter — put constraints in the prompt: 'keep it subtitle-free, do not generate a watermark, do not generate a logo'.",
|
|
49
49
|
"seedance-2-5 only: one shot runs to 30s (the 2.0 SKUs stop at 15s), so storyboard a whole beat instead of planning a stitch. Ref caps are wider (30/10/10), but 4-5 assets still gives the best identity fidelity.",
|
|
50
|
+
"Auto-path formula: Subject → Action → Environment → Camera → Style → Constraints in 60-100 words; ONE camera instruction (chain with 'then'); separate camera motion from subject motion; always add one lighting phrase.",
|
|
50
51
|
],
|
|
51
52
|
doctrine: `Prompt structure (front-load what matters most):
|
|
52
53
|
precise subject → action details → scene/environment → lighting & color tone → camera movement → visual style → image quality → constraints.
|
|
@@ -84,12 +85,35 @@ precise subject → action details → scene/environment → lighting & color to
|
|
|
84
85
|
**Known weaknesses → workarounds**
|
|
85
86
|
- Text rendering is weak: keep on-screen text to short common words; for exact text or logos, attach the artwork as a reference image and instruct "the logo from Image N stays in the corner unchanged".
|
|
86
87
|
- More than 4 referenced people gets unstable: group people into composite images of ≤4 first (image generation), then reference those composites.
|
|
87
|
-
- Repeated extension degrades quality: prefer high-definition reference assets and avoid stacking many continuations
|
|
88
|
+
- Repeated extension degrades quality: prefer high-definition reference assets and avoid stacking many continuations.
|
|
89
|
+
|
|
90
|
+
**Auto-path formula (community-sourced enrichment — apiyi.com Seedance 2.0 prompt guide,
|
|
91
|
+
higgsfield.ai 4K breakdown; captured 2026-08-09)**
|
|
92
|
+
- Six steps IN ORDER, 60-100 words total (longer measurably degrades): Subject → Action → Environment → Camera → Style → Constraints.
|
|
93
|
+
- ONE primary camera instruction per shot. Compound moves chain with "then": "camera slow tracking then subtle rise" — never two competing verbs. The 8 reliable camera types: push-in, pull-out, pan, tracking, orbit/arc, aerial, handheld, locked-off.
|
|
94
|
+
- SEPARATE camera movement from subject movement — the single biggest quality lever: "The dancer spins slowly. Camera holds fixed framing." — never "spinning camera around a dancing person".
|
|
95
|
+
- Pace with human words (slow / gentle / gradual / smooth / controlled) — never fps numbers or f-stops in the basic path.
|
|
96
|
+
- ALWAYS add one lighting phrase (highest-impact single addition): golden hour / rim light / neon glow / backlit / overcast.
|
|
97
|
+
- Bake stability constraints in: "avoid jitter and bent limbs", "avoid temporal flicker", "avoid identity drift".
|
|
98
|
+
- Ban vague adjectives standing alone ("epic", "amazing", "beautiful", bare "cinematic") — every adjective needs a concrete noun.
|
|
99
|
+
- Mode notes: i2v — skip subject description (the frame has it), focus on motion, append "preserve composition and colors". v2v — describe the style TRANSFORM, keep motion + identity.
|
|
100
|
+
- Advanced (pro path): focal angles in degrees ("47° normal", "29° telephoto", "107° wide"); "180° shutter" for filmic motion blur; handheld texture as "organic shake, micro-drift, subtle dutch"; "white balance locked 5200K"; explicit POSITIVE LOCKS section + "100% matches the reference" for identity-critical shots.
|
|
101
|
+
|
|
102
|
+
**Camera-path control — the magenta-line method (STORYBOARD community technique; the
|
|
103
|
+
manual pro path for precise trajectories, NOT the auto path)**
|
|
104
|
+
1. Duplicate the start frame; on the COPY draw a thick magenta line + arrowhead — the line is the camera's flight path, the arrow its end point. Keep the clean original.
|
|
105
|
+
2. Attach BOTH frames and declare the guide: "Image N contains a magenta line and arrow — a hidden camera trajectory guide, NOT part of the scene. Completely remove it: no line, no arrow, no paint, no trail, no reflection." Skipping the removal order RENDERS the line.
|
|
106
|
+
3. Command the path: "one continuous FPV drone glide following the S-shaped curve as closely as possible — do not shortcut. Camera motion is the priority." Lock the clean frame as first frame + scene reference; lock the destination frame if wired.
|
|
107
|
+
4. Pace with timing blocks ("[00:00-00:02] rise over the rooftop … [00:07-00:09] settle on the doorway") and keep any dialogue SHORT — long lines fight the move.
|
|
108
|
+
5. Assign image-input roles explicitly: first-frame/scene-ref · destination frame · path-guide · 3-6 character-identity refs — and bind identities with @-mentions exactly like the platform's reference pills.`,
|
|
88
109
|
}
|
|
89
110
|
|
|
90
111
|
const KLING_AUDIO_DOCTRINE: ProviderPromptDoctrine = {
|
|
91
|
-
|
|
92
|
-
|
|
112
|
+
// kling-turbo (2.5 Turbo Pro) + kling-master (2.1 Master) are SILENT tiers of
|
|
113
|
+
// the same engine: the structure/motion guidance applies, the Audio block
|
|
114
|
+
// does not (variant note in the doctrine body).
|
|
115
|
+
providers: ["kling", "kling-3.0", "kling-3-omni", "kling-turbo", "kling-master"],
|
|
116
|
+
heading: "Kling 2.1 / 2.5 / 2.6 / 3.0 / 3 Omni (kling, kling-3.0, kling-3-omni, kling-turbo, kling-master)",
|
|
93
117
|
tips: [
|
|
94
118
|
"Kling speaks scripted dialogue natively with lip sync — quote the line and enable sound: [Anna: warm calm voice]: \"We made it.\" On kling/kling-3.0 audio raises the credit cost; kling-3-omni includes it.",
|
|
95
119
|
"Structure prompts as Scene → character/element → Motion → Audio → style. Put ALL sound in one 'Audio:' block: dialogue in quotes, then SFX and ambience described plainly ('rain tapping on glass, no music').",
|
|
@@ -121,7 +145,11 @@ const KLING_AUDIO_DOCTRINE: ProviderPromptDoctrine = {
|
|
|
121
145
|
|
|
122
146
|
**Limits**
|
|
123
147
|
- Kling 2.6 prompts cap at 1000 characters — front-load scene + dialogue and trim style tails first. kling-3.0 accepts long prompts.
|
|
124
|
-
- Durations: 2.6 = 5/10s; 3.0/omni = 3-15s. A spoken line needs roughly 1s per 2-3 words — don't script more dialogue than the clip can hold
|
|
148
|
+
- Durations: 2.6 = 5/10s; 3.0/omni = 3-15s. A spoken line needs roughly 1s per 2-3 words — don't script more dialogue than the clip can hold.
|
|
149
|
+
|
|
150
|
+
**Variant note — kling-turbo (2.5 Turbo Pro) & kling-master (2.1 Master)**
|
|
151
|
+
- SILENT tiers: no audio parameter, so the entire Audio block above does not apply — skip dialogue/SFX cues; the Scene → Character → Motion → Style structure and motion guidance carry over unchanged.
|
|
152
|
+
- Durations 5/10s; kling-turbo takes an end frame (tail_image_url); kling-master is single-image i2v.`,
|
|
125
153
|
}
|
|
126
154
|
|
|
127
155
|
const MINIMAX_H3_DOCTRINE: ProviderPromptDoctrine = {
|
|
@@ -160,10 +188,205 @@ precise subject → action details → scene/environment → lighting & color to
|
|
|
160
188
|
- There is NO negative-prompt parameter — all constraints belong in the prompt text itself: "keep it subtitle-free, do not generate a watermark, do not generate a logo, stable picture".`,
|
|
161
189
|
}
|
|
162
190
|
|
|
191
|
+
const VEO_31_DOCTRINE: ProviderPromptDoctrine = {
|
|
192
|
+
providers: ["veo3", "veo3.1", "veo3_lite", "veo-1080p", "veo-4k", "veo-extend"],
|
|
193
|
+
heading: "VEO 3.1 — Quality / Fast / Lite (veo3, veo3.1, veo3_lite)",
|
|
194
|
+
tips: [
|
|
195
|
+
"Structure prompts as [Cinematography] + [Subject] + [Action] + [Context] + [Style & Ambiance] — lead with the camera, not the subject (Google's official formula).",
|
|
196
|
+
"Dialogue: quote the exact line with attribution — A woman says, \"We have to leave now.\" (no subtitles). Cue sound as separate lines: SFX: thunder cracks; Ambient noise: quiet hum of a starship bridge.",
|
|
197
|
+
"Multi-shot pacing via timestamp blocks: [00:00-00:02] medium shot… [00:02-00:04] reverse shot… — VEO honors per-window actions inside one 8s generation.",
|
|
198
|
+
"Negative prompting is positive phrasing: not 'no buildings' but 'a desolate landscape with no buildings or roads'. Keep prompts under ~175 words — longer overloads the generation.",
|
|
199
|
+
"Start+end frame: pass both and describe the transition move ('smooth 180-degree arc ending on the POV behind her'). References (ingredients) keep characters/objects consistent and DO generate audio.",
|
|
200
|
+
],
|
|
201
|
+
doctrine: `Prompt structure (Google's official Veo 3.1 formula — lead with the camera):
|
|
202
|
+
[Cinematography] + [Subject] + [Action] + [Context] + [Style & Ambiance].
|
|
203
|
+
Example: "Medium shot, a tired corporate worker, rubbing his temples in exhaustion, in front of a bulky 1980s computer in a cluttered office late at night, lit by harsh fluorescents and the green monitor glow. Retro aesthetic, 1980s color film, slightly grainy."
|
|
204
|
+
|
|
205
|
+
**Camera vocabulary (use the exact terms)**
|
|
206
|
+
- Movement: dolly shot, tracking shot, crane shot, aerial view, slow pan, POV shot, 180-degree arc shot.
|
|
207
|
+
- Composition: wide shot, medium shot, close-up, extreme close-up, two-shot, low angle, high angle.
|
|
208
|
+
- Lens/focus: shallow depth of field, deep focus, wide-angle lens, macro lens, soft focus.
|
|
209
|
+
|
|
210
|
+
**Audio (native, multi-track — dialogue / SFX / ambience)**
|
|
211
|
+
- Dialogue: quote the exact line with attribution: The detective says in a weary voice, "Of all the offices in this town, you had to walk into mine." Append "(no subtitles)" — VEO otherwise tends to burn captions in.
|
|
212
|
+
- Sound effects on their own line: "SFX: a crystal wine glass shatters on the marble floor". Ambient bed: "Ambient noise: rain against the window, distant traffic".
|
|
213
|
+
- Sound can drive the visual ("the sound reverberating through the empty ballroom") — VEO syncs audio-visual timing.
|
|
214
|
+
|
|
215
|
+
**Multi-shot timestamp prompting (inside one generation)**
|
|
216
|
+
- Split the clip into [mm:ss-mm:ss] windows, one action per window:
|
|
217
|
+
[00:00-00:02] Medium shot from behind a young explorer walking toward a clearing.
|
|
218
|
+
[00:02-00:04] Reverse shot of her freckled face, eyes widening.
|
|
219
|
+
[00:04-00:08] Wide, high-angle crane shot revealing the ruins below.
|
|
220
|
+
- 4 / 6 / 8 second clips; budget ~2s per window.
|
|
221
|
+
|
|
222
|
+
**Frames & references**
|
|
223
|
+
- Start + end frame: wire both (imageUrls [start, end]) and describe the camera path between them — "a smooth 180-degree arc shot, starting front-facing and circling to end on the POV from behind her".
|
|
224
|
+
- Reference images (ingredients): attach character/object/scene refs and name them in the prompt ("using the provided images for the detective and the office, …"). Reference runs DO generate audio.
|
|
225
|
+
|
|
226
|
+
**Constraints**
|
|
227
|
+
- Negative prompting works by positive description: write "a desolate landscape with no buildings or roads", not "no buildings".
|
|
228
|
+
- Keep prompts ≤ ~175 words — beyond that instructions conflict and adherence drops. Resolution 720p/1080p; aspect 16:9 / 9:16.
|
|
229
|
+
|
|
230
|
+
Sources: Google Cloud "Ultimate prompting guide for Veo 3.1"
|
|
231
|
+
(cloud.google.com/blog/products/ai-machine-learning/ultimate-prompting-guide-for-veo-3-1),
|
|
232
|
+
KIE VEO API docs (docs.kie.ai/veo3-api/generate-veo-3-video). Captured 2026-08-09.`,
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
const GEMINI_OMNI_DOCTRINE: ProviderPromptDoctrine = {
|
|
236
|
+
providers: ["gemini-omni-video"],
|
|
237
|
+
heading: "Gemini Omni Video (gemini-omni-video)",
|
|
238
|
+
tips: [
|
|
239
|
+
"Multimodal Google video with native audio: text-to-video, image-to-video, and video-edit through the same prompt surface. 4/6/8/10s; 720p/1080p or 4K tier.",
|
|
240
|
+
"Structure like the platform default: subject → action → scene → lighting → camera → style. Quote dialogue lines to have them spoken; describe SFX/ambience plainly in the prompt.",
|
|
241
|
+
"Text-to-video REQUIRES a concrete aspect ratio (the API hard-rejects a missing one); image runs infer aspect from the input.",
|
|
242
|
+
"Reference images ride along as additional imageUrls — bind them in the prompt ('the woman from the first image'). Video-edit: wire a source clip and describe the change, not the whole scene.",
|
|
243
|
+
],
|
|
244
|
+
doctrine: `Prompt structure (no public Google prompt guide exists for the Omni video endpoint —
|
|
245
|
+
the API contract is the doctrine source, like MiniMax H3; structure guidance mirrors the
|
|
246
|
+
platform's ordinal-reference conventions):
|
|
247
|
+
subject → action → scene/environment → lighting → camera movement → style → constraints.
|
|
248
|
+
|
|
249
|
+
**Modes (picked from the wired inputs)**
|
|
250
|
+
- Nothing visual → text-to-video. A concrete aspect ratio is REQUIRED — the API hard-rejects a missing one (Nodaro sends the node's ratio; there is no adaptive).
|
|
251
|
+
- Image(s) wired → image-to-video: the first image anchors the scene; extra images are references — bind each in the prompt ("the woman from the first image", "the interior from the second image").
|
|
252
|
+
- Source video wired → video-edit (served through the same handle): describe the CHANGE ("replace the daylight with dusk, keep the motion and framing"), not a full re-description.
|
|
253
|
+
|
|
254
|
+
**Audio (native)**
|
|
255
|
+
- Audio is generated with the clip. Quote dialogue to have it spoken; describe SFX and ambience plainly ("rain on glass, low synth bed"). State exclusions ("no music") or a bed may be invented.
|
|
256
|
+
|
|
257
|
+
**Duration & tiers**
|
|
258
|
+
- 4 / 6 / 8 / 10 seconds. 720p/1080p tier or the pricier 4K tier — pick 4K only when the deliverable needs it (nearly 2× the credits).
|
|
259
|
+
|
|
260
|
+
Source: KIE gemini-omni-video market contract (parameters + live behavior probed for the
|
|
261
|
+
aspect-ratio hard-reject, see providers/kie/video.ts). Captured 2026-08-09.`,
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
const GROK_IMAGINE_DOCTRINE: ProviderPromptDoctrine = {
|
|
265
|
+
providers: ["grok-i2v", "grok-imagine-video-1.5"],
|
|
266
|
+
heading: "Grok Imagine (grok-i2v, grok-imagine-video-1.5)",
|
|
267
|
+
tips: [
|
|
268
|
+
"Keep prompts simple and direct — Subject + Action + Setting + Camera + Mood. Grok expands the prompt itself; over-specification fights the expander.",
|
|
269
|
+
"Image-to-video: the input image IS the first frame (composition, identity, and style are preserved) — describe the MOTION, don't re-describe the still.",
|
|
270
|
+
"Video 1.5: 1-15s (default 8), 480p/720p/1080p, up to 7 input images (1080p allows only one). Native audio incl. music, SFX, and lip-synced dialogue — quote the line to have it spoken.",
|
|
271
|
+
"Aspect ratio applies to text runs (1:1/16:9/9:16/3:2/2:3/auto); a single input image locks the output to the image's own aspect.",
|
|
272
|
+
],
|
|
273
|
+
doctrine: `Prompt structure (xAI's guidance is minimal by design — the model auto-expands prompts):
|
|
274
|
+
Subject + Action + Setting + Camera + Lighting/Mood, written simply and directly. Reduce
|
|
275
|
+
descriptions of static/unchanged parts — spend the words on what MOVES.
|
|
276
|
+
|
|
277
|
+
**Image-to-video (the primary mode)**
|
|
278
|
+
- The input image is the FIRST FRAME, not a loose reference: composition, subject identity, and visual style carry over. Describe motion and camera only ("she turns toward the window as the camera slowly pushes in"); re-describing the still wastes adherence.
|
|
279
|
+
- grok-imagine-video-1.5 accepts up to 7 images (identity/scene references beyond the first frame); at 1080p only ONE image is allowed.
|
|
280
|
+
|
|
281
|
+
**Audio (video-1.5)**
|
|
282
|
+
- Native audio generates with the clip — background music, SFX, and lip-synced dialogue. Quote the spoken line; describe the music/SFX plainly. There is no audio toggle on the KIE contract — cue (or exclude) sound in the prompt text.
|
|
283
|
+
|
|
284
|
+
**Durations / tiers**
|
|
285
|
+
- grok-i2v: 6 or 10 seconds. grok-imagine-video-1.5: 1-15 seconds in 1s steps (default 8), 480p (default) / 720p / 1080p. Prompt cap 4096 chars — but shorter is better here.
|
|
286
|
+
|
|
287
|
+
Sources: KIE Grok Imagine contracts (docs.kie.ai/market/grok-imagine/image-to-video,
|
|
288
|
+
docs.kie.ai/market/grok-imagine/1-5-preview), xAI Grok Imagine 1.5 release notes
|
|
289
|
+
(x.ai/news/grok-imagine-1-5). Captured 2026-08-09.`,
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
const WAN_DOCTRINE: ProviderPromptDoctrine = {
|
|
293
|
+
providers: ["wan", "wan-i2v", "wan-turbo", "wan-flash", "wan-2.7", "wan-2.7-i2v", "wan-2.7-t2v", "wan-2.7-pro", "wan-videoedit"],
|
|
294
|
+
heading: "Wan 2.x (wan, wan-i2v, wan-turbo, wan-2.7 family)",
|
|
295
|
+
tips: [
|
|
296
|
+
"Alibaba's official formula: Entity + Scene + Motion (basic) → add Aesthetic control + Stylization (advanced). Image-to-video: Motion + Camera only — the image already defines entity and scene.",
|
|
297
|
+
"Sound (2.5+): append a sound description block — voice / sound effects / background music. Avoid scripting EXACT lip-synced lines (official anti-pattern); describe the voice and intent instead.",
|
|
298
|
+
"Multi-shot (2.6/2.7): Overall description + shot number + timestamp + per-shot content. For ONE continuous take write 'Generate single shot' (the shot_type parameter is gone in 2.7).",
|
|
299
|
+
"References go by 'Image 1' / 'Video 1' (capitalized, with a space). Anti-patterns: naming real people, demanding exact legible text, rapid scene changes in one clip, very long choreography.",
|
|
300
|
+
"Style words are strong levers: cyberpunk, claymation, pixel style, felt style, tilt-shift, time-lapse. wan-videoedit: describe the transform, keep motion + identity.",
|
|
301
|
+
],
|
|
302
|
+
doctrine: `Prompt structure (Alibaba Model Studio's official formulas):
|
|
303
|
+
- Basic: Entity + Scene + Motion.
|
|
304
|
+
- Advanced: Entity (description) + Scene (description) + Motion (description) + Aesthetic control + Stylization.
|
|
305
|
+
- Image-to-video: Motion + Camera movement ONLY — the wired image already defines entity and scene; re-describing it fights the frame.
|
|
306
|
+
- Sound (2.5/2.6/2.7): … + Sound description (voice / sound effects / background music).
|
|
307
|
+
- Multi-shot (2.6/2.7): Overall description + Shot number + Timestamp + Shot content.
|
|
308
|
+
- Reference-to-video (2.6/2.7): Reference identifier + Action + Scene + optional Lines + optional BGM.
|
|
309
|
+
|
|
310
|
+
**Camera vocabulary**
|
|
311
|
+
push-in (intimacy/tension), pull-out (scale/isolation), tracking shot, orbit, fixed camera, and compound movements chained sequentially for epic scale.
|
|
312
|
+
|
|
313
|
+
**Single-shot control (2.7)**
|
|
314
|
+
- The shot_type parameter no longer exists — write "Generate single shot" in the prompt to force one continuous take; otherwise 2.7's planner may cut.
|
|
315
|
+
|
|
316
|
+
**References**
|
|
317
|
+
- English format is "Image 1" / "Video 1" (capitalized, space-separated) — bind every wired asset by that name or it may be ignored.
|
|
318
|
+
|
|
319
|
+
**Official anti-patterns (from Alibaba's guide)**
|
|
320
|
+
- Do NOT name specific real people.
|
|
321
|
+
- Do NOT script exact lip-synced dialogue — describe the voice and intent ("she murmurs a reassurance, warm and low") instead of demanding word-perfect lips.
|
|
322
|
+
- Avoid rapid scene changes inside a single clip, very long choreographed sequences, and demands for exactly legible on-screen text.
|
|
323
|
+
|
|
324
|
+
**Stylization**
|
|
325
|
+
- Style words are strong levers: cyberpunk, line-art illustration, felt style, 3D cartoon, pixel style, puppet animation, claymation, black-and-white animation, tilt-shift, time-lapse.
|
|
326
|
+
|
|
327
|
+
Source: Alibaba Cloud Model Studio — "Text-to-video / image-to-video prompt guide"
|
|
328
|
+
(alibabacloud.com/help/en/model-studio/text-to-video-prompt). Captured 2026-08-09.`,
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
const HAPPYHORSE_DOCTRINE: ProviderPromptDoctrine = {
|
|
332
|
+
providers: ["happyhorse", "happyhorse-i2v", "happyhorse-ref2v", "happyhorse-edit"],
|
|
333
|
+
heading: "HappyHorse 1.1 (happyhorse, happyhorse-i2v, happyhorse-ref2v)",
|
|
334
|
+
tips: [
|
|
335
|
+
"Any-language prompts up to 5000 chars (2500 Chinese) — excess is silently truncated, so front-load subject → action → scene → camera → style.",
|
|
336
|
+
"3-15 seconds per second of billing; 720p or 1080p; ratios 16:9 / 9:16 / 1:1 / 4:3 / 3:4. Pick the shortest duration that serves the shot.",
|
|
337
|
+
"ref2v is one of the few true REFERENCE modes on the roster: wired refs keep identity across the clip — bind each reference explicitly in the prompt.",
|
|
338
|
+
"No published vendor style guide — the platform's standard structure applies; keep one camera move per shot and quantify motion physically.",
|
|
339
|
+
],
|
|
340
|
+
doctrine: `Prompt structure (no public HappyHorse prompt guide exists — the KIE API contract is the
|
|
341
|
+
doctrine source; platform-standard structure applies):
|
|
342
|
+
subject → action → scene/environment → lighting → camera movement → style → constraints.
|
|
343
|
+
|
|
344
|
+
**Contract facts (KIE, per-mode pages)**
|
|
345
|
+
- Prompts: any language, up to 5000 non-Chinese / 2500 Chinese characters — excess is TRUNCATED silently, so put the load-bearing content first.
|
|
346
|
+
- Duration 3-15s (default 5), billed per second. Resolution 720p / 1080p (default). Aspect 16:9 (default) / 9:16 / 1:1 / 4:3 / 3:4.
|
|
347
|
+
- Modes: text-to-video (happyhorse), image-to-video (happyhorse-i2v), reference-to-video (happyhorse-ref2v) — ref2v preserves wired identities; name each reference in the prompt so the binding is explicit.
|
|
348
|
+
|
|
349
|
+
**Style guidance (platform-standard, honestly generic)**
|
|
350
|
+
- One camera movement per shot; physical, quantified action ("slowly raises a hand") over abstract emotion words; state exclusions ("no on-screen text, no watermark") in the prompt.
|
|
351
|
+
|
|
352
|
+
Source: KIE HappyHorse 1.1 contracts (docs.kie.ai/market/happyhorse/text-to-video,
|
|
353
|
+
…/happyhorse-1-1/image-to-video, …/happyhorse-1-1/reference-to-video). Captured 2026-08-09.`,
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
const RUNWAY_KIE_DOCTRINE: ProviderPromptDoctrine = {
|
|
357
|
+
providers: ["runway-kie", "runway-extend", "runway-aleph"],
|
|
358
|
+
heading: "Runway via KIE (runway-kie)",
|
|
359
|
+
tips: [
|
|
360
|
+
"Prompt cap is 1800 chars; KIE's own guidance: be specific about subject, action, style, and setting. No native audio — plan sound as a separate pass.",
|
|
361
|
+
"Durations 5 or 10s with a hard trade-off: 10s cannot be 1080p, 1080p cannot exceed 5s — pick per deliverable.",
|
|
362
|
+
"Text runs REQUIRE an aspect ratio (16:9/4:3/1:1/3:4/9:16); image runs IGNORE it — the input image dictates output dimensions.",
|
|
363
|
+
"Image-to-video treats the image as the anchor frame: describe motion and camera, not the still.",
|
|
364
|
+
],
|
|
365
|
+
doctrine: `Prompt structure (KIE contract guidance): "be specific about subject, action, style, and
|
|
366
|
+
setting" — subject → action → scene → camera → style, within the 1800-character cap.
|
|
367
|
+
|
|
368
|
+
**Contract facts (KIE Runway endpoint)**
|
|
369
|
+
- Duration 5 or 10 seconds; quality 720p or 1080p — 10s@1080p does NOT exist (10s forces 720p; 1080p forces 5s). Choose by deliverable: crisp hero shot → 5s/1080p; longer beat → 10s/720p.
|
|
370
|
+
- Text-to-video REQUIRES aspectRatio (16:9 / 4:3 / 1:1 / 3:4 / 9:16). Image-to-video IGNORES aspectRatio — the input image dictates output dimensions.
|
|
371
|
+
- No audio is generated — score/SFX are a separate pass (merge-video-audio / video-sfx downstream).
|
|
372
|
+
|
|
373
|
+
**Style guidance**
|
|
374
|
+
- The image input anchors composition and identity — describe the motion ("she pushes the door open as the camera tracks left"), not the still.
|
|
375
|
+
- Keep one continuous camera idea per clip; front-load the subject and action.
|
|
376
|
+
|
|
377
|
+
Source: KIE Runway contract (docs.kie.ai/runway-api/generate-ai-video). Captured 2026-08-09.`,
|
|
378
|
+
}
|
|
379
|
+
|
|
163
380
|
export const PROVIDER_PROMPT_DOCTRINES: readonly ProviderPromptDoctrine[] = [
|
|
164
381
|
SEEDANCE_2_DOCTRINE,
|
|
165
382
|
KLING_AUDIO_DOCTRINE,
|
|
166
383
|
MINIMAX_H3_DOCTRINE,
|
|
384
|
+
VEO_31_DOCTRINE,
|
|
385
|
+
GEMINI_OMNI_DOCTRINE,
|
|
386
|
+
GROK_IMAGINE_DOCTRINE,
|
|
387
|
+
WAN_DOCTRINE,
|
|
388
|
+
HAPPYHORSE_DOCTRINE,
|
|
389
|
+
RUNWAY_KIE_DOCTRINE,
|
|
167
390
|
]
|
|
168
391
|
|
|
169
392
|
const DOCTRINE_BY_PROVIDER: ReadonlyMap<string, ProviderPromptDoctrine> = new Map(
|