@nodaro/prompts 1.13.0 → 1.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +86 -7
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +11 -1
- package/dist/index.d.ts +11 -1
- package/dist/index.js +87 -8
- package/dist/index.js.map +1 -1
- package/package.json +2 -2
- package/src/__tests__/multi-picker-spec.test.ts +21 -1
- package/src/__tests__/person-regional-aesthetic.test.ts +2 -1
- package/src/__tests__/provider-prompt-doctrine.test.ts +39 -0
- package/src/gemini-omni-inputs.ts +11 -3
- package/src/held-prop.ts +1 -0
- package/src/person.ts +4 -0
- package/src/picker-analyzer-registry.ts +37 -0
- package/src/prompt-wizard-categories.ts +6 -0
- package/src/provider-prompt-doctrine.ts +51 -2
- package/src/setting.ts +1 -0
- package/src/style.ts +1 -0
- package/src/styling.ts +5 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@nodaro/prompts",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.14.0",
|
|
4
4
|
"description": "Nodaro's prompt-engineering layer — person/picker catalogs with prompt hints, identity-lock clauses, entity prompt builders, brand presets, and prompt/reference assembly shared by the Nodaro platform and SDK.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "FSL-1.1-Apache-2.0",
|
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
"test": "vitest run"
|
|
21
21
|
},
|
|
22
22
|
"dependencies": {
|
|
23
|
-
"@nodaro/shared": "^2.
|
|
23
|
+
"@nodaro/shared": "^2.19.0"
|
|
24
24
|
},
|
|
25
25
|
"devDependencies": {
|
|
26
26
|
"tsup": "^8.5.0",
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { describe, it, expect } from "vitest"
|
|
2
|
-
import { buildMultiPickerAnalyzerSpec, buildPickerAnalyzerSpec, pickerFanoutTargets } from "../index.js"
|
|
2
|
+
import { buildMultiPickerAnalyzerSpec, buildPickerAnalyzerSpec, pickerFanoutTargets, PICKER_TYPES } from "../index.js"
|
|
3
3
|
|
|
4
4
|
describe("buildMultiPickerAnalyzerSpec", () => {
|
|
5
5
|
it("composes one section per wired picker plus gaps, order-independent", () => {
|
|
@@ -23,6 +23,26 @@ describe("buildMultiPickerAnalyzerSpec", () => {
|
|
|
23
23
|
expect(withGaps.gaps.missingItems).toHaveLength(1)
|
|
24
24
|
})
|
|
25
25
|
|
|
26
|
+
it("otherPickersLegend names the non-wired pickers by key, excluding the wired ones", () => {
|
|
27
|
+
const { otherPickersLegend } = buildMultiPickerAnalyzerSpec(["person", "styling"])
|
|
28
|
+
expect(otherPickersLegend.length).toBeGreaterThan(0)
|
|
29
|
+
// non-wired pickers are attributable by their exact key
|
|
30
|
+
expect(otherPickersLegend).toContain("- era:")
|
|
31
|
+
expect(otherPickersLegend).toContain("- setting:")
|
|
32
|
+
expect(otherPickersLegend).toContain("- held-prop:")
|
|
33
|
+
// wired pickers must NOT appear as their own bullet lines
|
|
34
|
+
expect(otherPickersLegend).not.toContain("- person:")
|
|
35
|
+
expect(otherPickersLegend).not.toContain("- styling:")
|
|
36
|
+
// discriminated pickers carry their dimension labels; flat ones carry the label
|
|
37
|
+
expect(otherPickersLegend).toContain("Era / Period") // flat `era` label
|
|
38
|
+
expect(otherPickersLegend).toMatch(/- exposure-settings: Exposure Settings — .*Aperture/) // title-cased key + dims
|
|
39
|
+
})
|
|
40
|
+
|
|
41
|
+
it("otherPickersLegend is empty when every picker is wired", () => {
|
|
42
|
+
const { otherPickersLegend } = buildMultiPickerAnalyzerSpec(PICKER_TYPES)
|
|
43
|
+
expect(otherPickersLegend).toBe("")
|
|
44
|
+
})
|
|
45
|
+
|
|
26
46
|
it("FUZZ: arbitrary gaps content never alters the enum-validated picker sections", () => {
|
|
27
47
|
const { schema } = buildMultiPickerAnalyzerSpec(["person"])
|
|
28
48
|
const personSpec = buildPickerAnalyzerSpec("person")
|
|
@@ -26,7 +26,7 @@ describe("regional-aesthetic dimension — wiring", () => {
|
|
|
26
26
|
|
|
27
27
|
describe("regional-aesthetic catalog — coverage and structure", () => {
|
|
28
28
|
it("ships the expected total entry count", () => {
|
|
29
|
-
expect(REGIONAL_ENTRIES.length).toBe(
|
|
29
|
+
expect(REGIONAL_ENTRIES.length).toBe(88)
|
|
30
30
|
})
|
|
31
31
|
|
|
32
32
|
it.each([
|
|
@@ -34,6 +34,7 @@ describe("regional-aesthetic catalog — coverage and structure", () => {
|
|
|
34
34
|
["USA — African-American", 6],
|
|
35
35
|
["Europe", 16],
|
|
36
36
|
["Asia", 12],
|
|
37
|
+
["Central Asia", 2],
|
|
37
38
|
["Latin America", 7],
|
|
38
39
|
["Middle East", 7],
|
|
39
40
|
["North Africa", 3],
|
|
@@ -66,4 +66,43 @@ describe("PROVIDER_PROMPT_DOCTRINES", () => {
|
|
|
66
66
|
expect(tips).toMatch(/2\.0 SKUs ignore timestamps/)
|
|
67
67
|
expect(tips).toMatch(/seedance-2-5 honours integer-second timestamps/)
|
|
68
68
|
})
|
|
69
|
+
|
|
70
|
+
it("Wan 3.0 has its OWN doctrine — the 2.x token format must not leak onto it", () => {
|
|
71
|
+
// WAN_DOCTRINE teaches "Image 1" / "Video 1" (capitalised, WITH a space),
|
|
72
|
+
// which is the Wan 2.x format. The Wan 3.0 KIE contract binds Image1 /
|
|
73
|
+
// Video1 / Audio1 (no space). Folding 3.0 into the 2.x entry would ship a
|
|
74
|
+
// token format the model does not use — a prompt-quality regression that
|
|
75
|
+
// costs real generations, so the split is the point of this test.
|
|
76
|
+
const d = getPromptDoctrine("wan-3")!
|
|
77
|
+
expect(d.providers).toEqual(["wan-3", "wan-3-prime"])
|
|
78
|
+
expect(getPromptDoctrine("wan-3-prime")).toBe(d)
|
|
79
|
+
expect(d.doctrine).toContain("Image1")
|
|
80
|
+
expect(d.doctrine).toContain("Audio1")
|
|
81
|
+
// The 2.x spelling may appear ONLY as an explicit contrast, never as the
|
|
82
|
+
// instruction — so the body must say the format has no space.
|
|
83
|
+
expect(d.doctrine).toMatch(/NO space/)
|
|
84
|
+
expect(d.doctrine).toMatch(/Wan 2\.x's "Image 1"/)
|
|
85
|
+
|
|
86
|
+
// The 2.x doctrine must stay the 2.x doctrine.
|
|
87
|
+
const wan2 = getPromptDoctrine("wan-i2v")!
|
|
88
|
+
expect(wan2).not.toBe(d)
|
|
89
|
+
expect(wan2.providers).not.toContain("wan-3")
|
|
90
|
+
expect(wan2.doctrine).toContain("Image 1")
|
|
91
|
+
|
|
92
|
+
// KIE contract facts the body is required to carry.
|
|
93
|
+
expect(d.doctrine).toMatch(/mutually exclusive|CANNOT be combined/i)
|
|
94
|
+
expect(d.doctrine).toMatch(/20,000 characters/)
|
|
95
|
+
expect(d.doctrine).toMatch(/≤ 30 seconds/)
|
|
96
|
+
// Prime is the SPEED tier — never described as higher quality.
|
|
97
|
+
expect(d.doctrine).toMatch(/faster turnaround/i)
|
|
98
|
+
expect(d.doctrine).not.toMatch(/higher quality/i)
|
|
99
|
+
})
|
|
100
|
+
|
|
101
|
+
it("Gemini Omni Flash rides the sibling's doctrine (identical request surface)", () => {
|
|
102
|
+
const d = getPromptDoctrine("gemini-omni-flash")!
|
|
103
|
+
expect(d).toBe(getPromptDoctrine("gemini-omni-video"))
|
|
104
|
+
expect(d.providers).toEqual(["gemini-omni-video", "gemini-omni-flash"])
|
|
105
|
+
expect(d.heading).toContain("gemini-omni-flash")
|
|
106
|
+
expect(d.doctrine).toMatch(/gemini-omni-flash is the faster\/cheaper tier/)
|
|
107
|
+
})
|
|
69
108
|
})
|
|
@@ -40,6 +40,11 @@ export interface GeminiOmniI2vInputsArgs {
|
|
|
40
40
|
refImageUrls?: Array<string | undefined>
|
|
41
41
|
/** A connected source video occupies 2 of the 7 input slots (KIE quota). */
|
|
42
42
|
videoConnected?: boolean
|
|
43
|
+
/** The Omni SKU this run targets — defaults to `gemini-omni-video` for
|
|
44
|
+
* back-compat. Both SKUs cap at 7 today, so passing it is behaviour-neutral;
|
|
45
|
+
* it stops the flash path from silently reading the pro model's quota if the
|
|
46
|
+
* two ever diverge. */
|
|
47
|
+
provider?: string
|
|
43
48
|
}
|
|
44
49
|
|
|
45
50
|
export interface GeminiOmniI2vInputsResult {
|
|
@@ -51,13 +56,16 @@ export interface GeminiOmniI2vInputsResult {
|
|
|
51
56
|
droppedRefImages: number
|
|
52
57
|
}
|
|
53
58
|
|
|
54
|
-
/** The catalog-declared cap (7) — read from the shared limits map so the
|
|
59
|
+
/** The catalog-declared cap (7) — read PER SKU from the shared limits map so the
|
|
55
60
|
* wire-contract number has one home; the literal is only the safety net. */
|
|
56
|
-
const
|
|
61
|
+
const DEFAULT_GEMINI_OMNI_PROVIDER = "gemini-omni-video"
|
|
62
|
+
function geminiOmniInputSlots(provider: string | undefined): number {
|
|
63
|
+
return VIDEO_REF_LIMITS_BY_PROVIDER[provider ?? DEFAULT_GEMINI_OMNI_PROVIDER]?.images ?? 7
|
|
64
|
+
}
|
|
57
65
|
|
|
58
66
|
export function resolveGeminiOmniI2vInputs(args: GeminiOmniI2vInputsArgs): GeminiOmniI2vInputsResult {
|
|
59
67
|
const refs = (args.refImageUrls ?? []).filter((u): u is string => typeof u === "string" && u.length > 0)
|
|
60
|
-
const slots =
|
|
68
|
+
const slots = geminiOmniInputSlots(args.provider) - (args.videoConnected ? 2 : 0)
|
|
61
69
|
const refSlots = Math.max(0, slots - 1)
|
|
62
70
|
const kept = refs.slice(0, refSlots)
|
|
63
71
|
const droppedRefImages = refs.length - kept.length
|
package/src/held-prop.ts
CHANGED
|
@@ -135,6 +135,7 @@ export const HELD_PROPS: ReadonlyArray<HeldProp> = [
|
|
|
135
135
|
{ id: "compass", label: "Compass", category: "occupational", description: "Vintage handheld nautical compass", promptHint: "holding a vintage brass nautical compass open in one cupped palm at chest height, the needle clearly visible as the eyes drift down to read the bearing", term: "holding an open brass nautical compass" },
|
|
136
136
|
{ id: "bow-and-arrow", label: "Bow and Arrow", category: "occupational", description: "Drawn archery bow with arrow nocked", promptHint: "holding an archery bow drawn at full tension with one hand on the grip and the other pulling the string back to the cheek, an arrow nocked and aimed forward", term: "drawing an archery bow with a nocked arrow" },
|
|
137
137
|
{ id: "shield", label: "Shield", category: "occupational", description: "Handheld medieval shield", promptHint: "holding a medieval shield raised across the body with one arm strapped through the back, the front face angled forward in a defensive stance", term: "holding a raised medieval shield" },
|
|
138
|
+
{ id: "work-gloves", label: "Work Gloves", category: "occupational", description: "Worn leather work gloves held in hand", promptHint: "holding a worn pair of tan leather work gloves in both hands at waist height, the thick weathered leather clearly visible", term: "holding a pair of leather work gloves" },
|
|
138
139
|
] as const
|
|
139
140
|
|
|
140
141
|
const heldPropById = new Map<string, HeldProp>(HELD_PROPS.map((p) => [p.id, p]))
|
package/src/person.ts
CHANGED
|
@@ -809,6 +809,10 @@ export const PEOPLE: ReadonlyArray<Person> = [
|
|
|
809
809
|
{ id: "south-india-traditional", label: "South India Traditional", group: "Asia", dimension: "regional-aesthetic", description: "Tamil / Kerala temple-town classical aesthetic", promptHint: "a South Indian traditional aesthetic — Tamil / Kerala temple-town vibe, classical refinement", term: "south indian traditional aesthetic" },
|
|
810
810
|
{ id: "bangkok-street", label: "Bangkok Street", group: "Asia", dimension: "regional-aesthetic", description: "Bangkok Thai night-market neon-urban aesthetic", promptHint: "a Bangkok street aesthetic — Thai night-market energy, neon-and-warmth urban vibe", term: "bangkok street aesthetic" },
|
|
811
811
|
|
|
812
|
+
// ----- Central Asia -----
|
|
813
|
+
{ id: "samarkand-silk-road", label: "Samarkand Silk Road", group: "Central Asia", dimension: "regional-aesthetic", description: "Uzbek Silk Road bazaar aesthetic (Samarkand / Bukhara)", promptHint: "a Central Asian Silk Road aesthetic — Samarkand / Bukhara bazaar vibe, ikat-and-suzani textiles, sun-baked adobe and blue-tiled madrasa mood", term: "samarkand silk road aesthetic" },
|
|
814
|
+
{ id: "tashkent-modern", label: "Tashkent Modern", group: "Central Asia", dimension: "regional-aesthetic", description: "Contemporary Uzbek metropolitan Central Asian aesthetic", promptHint: "a modern Tashkent aesthetic — contemporary Uzbek metropolitan vibe, Soviet-modern-meets-Silk-Road blend, warm steppe-city confidence", term: "tashkent modern aesthetic" },
|
|
815
|
+
|
|
812
816
|
// ----- Latin America -----
|
|
813
817
|
{ id: "carioca-rio", label: "Carioca (Rio)", group: "Latin America", dimension: "regional-aesthetic", description: "Rio de Janeiro beach-and-favela-music Brazilian aesthetic", promptHint: "a Carioca aesthetic — Rio de Janeiro beach-and-favela-music vibe, sun-warmed Brazilian energy", term: "carioca aesthetic" },
|
|
814
818
|
{ id: "paulista", label: "Paulista (São Paulo)", group: "Latin America", dimension: "regional-aesthetic", description: "São Paulo metropolitan Brazilian creative-class aesthetic", promptHint: "a Paulista aesthetic — São Paulo metropolitan vibe, urban-Brazilian creative-class polish", term: "paulista aesthetic" },
|
|
@@ -454,10 +454,46 @@ export interface MultiPickerAnalyzerSpec {
|
|
|
454
454
|
readonly schema: z.ZodType<Record<string, unknown>, unknown>
|
|
455
455
|
readonly toolName: string
|
|
456
456
|
readonly legend: string
|
|
457
|
+
/** Compact bullet list of the pickers NOT wired into this spec (PICKER_TYPES
|
|
458
|
+
* minus `types`), keyed by picker-type key so the LLM can ATTRIBUTE a gap to
|
|
459
|
+
* the right picker even when it was not wired. Names + dimension labels only,
|
|
460
|
+
* never catalog ids. Empty string when every picker is already wired. */
|
|
461
|
+
readonly otherPickersLegend: string
|
|
457
462
|
}
|
|
458
463
|
|
|
459
464
|
const MULTI_CACHE = new Map<string, MultiPickerAnalyzerSpec>()
|
|
460
465
|
|
|
466
|
+
/** Title-case a picker-type key for display, e.g. "person" → "Person",
|
|
467
|
+
* "exposure-settings" → "Exposure Settings". */
|
|
468
|
+
function pickerDisplayName(type: string): string {
|
|
469
|
+
return type
|
|
470
|
+
.split("-")
|
|
471
|
+
.map((w) => (w.length > 0 ? w[0].toUpperCase() + w.slice(1) : w))
|
|
472
|
+
.join(" ")
|
|
473
|
+
}
|
|
474
|
+
|
|
475
|
+
/** One compact bullet per non-wired picker so the LLM can attribute a gap to a
|
|
476
|
+
* picker it wasn't handed. Flat pickers show their registry `label` (one axis);
|
|
477
|
+
* discriminated pickers show a title-cased name plus their dimension labels.
|
|
478
|
+
* Never lists catalog ids. Empty string when every PICKER_TYPES member is
|
|
479
|
+
* wired. */
|
|
480
|
+
function buildOtherPickersLegend(sorted: ReadonlyArray<PickerType>): string {
|
|
481
|
+
const otherTypes = PICKER_TYPES.filter((t) => !sorted.includes(t))
|
|
482
|
+
if (otherTypes.length === 0) return ""
|
|
483
|
+
const lines = otherTypes.map((type) => {
|
|
484
|
+
const descriptor = PICKER_ANALYZER_REGISTRY[type as PickerType] as PickerAnalyzerDescriptor
|
|
485
|
+
if (descriptor.kind === "flat") {
|
|
486
|
+
return `- ${type}: ${descriptor.label}`
|
|
487
|
+
}
|
|
488
|
+
const dims = descriptor.order
|
|
489
|
+
.map((k) => descriptor.labels[k])
|
|
490
|
+
.filter(Boolean)
|
|
491
|
+
.join(", ")
|
|
492
|
+
return `- ${type}: ${pickerDisplayName(type)}${dims ? ` — ${dims}` : ""}`
|
|
493
|
+
})
|
|
494
|
+
return `Non-wired pickers — use one of these keys in a gap's \`picker\` when an attribute belongs to it:\n${lines.join("\n")}`
|
|
495
|
+
}
|
|
496
|
+
|
|
461
497
|
/** Build ONE forced-tool schema spanning the given pickers (each section
|
|
462
498
|
* optional so an omitted picker doesn't trigger a validation retry) plus the
|
|
463
499
|
* capped `gaps` sidecar. Memoized by the sorted picker-set key. */
|
|
@@ -479,6 +515,7 @@ export function buildMultiPickerAnalyzerSpec(types: ReadonlyArray<PickerType>):
|
|
|
479
515
|
schema: z.object(shape).strict() as unknown as MultiPickerAnalyzerSpec["schema"],
|
|
480
516
|
toolName: "emit_pickers",
|
|
481
517
|
legend: legendParts.join("\n\n"),
|
|
518
|
+
otherPickersLegend: buildOtherPickersLegend(sorted),
|
|
482
519
|
}
|
|
483
520
|
MULTI_CACHE.set(key, result)
|
|
484
521
|
return result
|
|
@@ -256,6 +256,9 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
|
|
|
256
256
|
"ltx-2.3-pro": "Lightricks LTX 2.3 Pro — text/image/audio→video, 6–10s, up to 4K",
|
|
257
257
|
"ltx-2.3-fast": "Lightricks LTX 2.3 Fast — text/image→video, 6–20s, up to 4K",
|
|
258
258
|
"gemini-omni-video": "Google Gemini Omni — multimodal video with native audio, 4–10s, up to 4K.",
|
|
259
|
+
"gemini-omni-flash": "Google Gemini Omni Flash — faster, cheaper Omni tier; multimodal video with native audio, 4–10s, up to 4K.",
|
|
260
|
+
"wan-3": "Wan 3.0 — multimodal refs (10 images / 5 videos / 5 audio) or first+last frame, native audio, 2–30s, 480p/720p/1080p",
|
|
261
|
+
"wan-3-prime": "Wan 3.0 Prime — high-speed Wan 3.0 tier; same surface, faster turnaround at a higher rate",
|
|
259
262
|
"grok-imagine-video-1.5": "Grok Imagine 1.5 — image-to-video only; requires an input image",
|
|
260
263
|
},
|
|
261
264
|
"image-to-video": {
|
|
@@ -289,6 +292,9 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
|
|
|
289
292
|
"ltx-2.3-pro": "Lightricks LTX 2.3 Pro — start/end frame i2v + audio→video, 6–10s, up to 4K",
|
|
290
293
|
"ltx-2.3-fast": "Lightricks LTX 2.3 Fast — start/end frame i2v, 6–20s, up to 4K",
|
|
291
294
|
"gemini-omni-video": "Google Gemini Omni — multimodal video with native audio, 4–10s, up to 4K.",
|
|
295
|
+
"gemini-omni-flash": "Google Gemini Omni Flash — faster, cheaper Omni tier; multimodal video with native audio, 4–10s, up to 4K.",
|
|
296
|
+
"wan-3": "Wan 3.0 — multimodal refs (10 images / 5 videos / 5 audio) or first+last frame, native audio, 2–30s, 480p/720p/1080p",
|
|
297
|
+
"wan-3-prime": "Wan 3.0 Prime — high-speed Wan 3.0 tier; same surface, faster turnaround at a higher rate",
|
|
292
298
|
"grok-imagine-video-1.5": "Grok Imagine 1.5 — stylized animation, 1–15s, 480p/720p (image required)",
|
|
293
299
|
},
|
|
294
300
|
"video-to-video": {
|
|
@@ -239,8 +239,8 @@ KIE VEO API docs (docs.kie.ai/veo3-api/generate-veo-3-video). Captured 2026-08-0
|
|
|
239
239
|
}
|
|
240
240
|
|
|
241
241
|
const GEMINI_OMNI_DOCTRINE: ProviderPromptDoctrine = {
|
|
242
|
-
providers: ["gemini-omni-video"],
|
|
243
|
-
heading: "Gemini Omni
|
|
242
|
+
providers: ["gemini-omni-video", "gemini-omni-flash"],
|
|
243
|
+
heading: "Gemini Omni (gemini-omni-video, gemini-omni-flash)",
|
|
244
244
|
tips: [
|
|
245
245
|
"Multimodal Google video with native audio: text-to-video, image-to-video, and video-edit through the same prompt surface. 4/6/8/10s; 720p/1080p or 4K tier.",
|
|
246
246
|
"Structure like the platform default: subject → action → scene → lighting → camera → style. Quote dialogue lines to have them spoken; describe SFX/ambience plainly in the prompt.",
|
|
@@ -262,6 +262,7 @@ subject → action → scene/environment → lighting → camera movement → st
|
|
|
262
262
|
|
|
263
263
|
**Duration & tiers**
|
|
264
264
|
- 4 / 6 / 8 / 10 seconds. 720p/1080p tier or the pricier 4K tier — pick 4K only when the deliverable needs it (nearly 2× the credits).
|
|
265
|
+
- gemini-omni-flash is the faster/cheaper tier with the identical request surface — same 4/6/8/10s, same 720p/1080p and 4K tiers, same video-edit path. Everything above applies verbatim.
|
|
265
266
|
|
|
266
267
|
Source: KIE gemini-omni-video market contract (parameters + live behavior probed for the
|
|
267
268
|
aspect-ratio hard-reject, see providers/kie/video.ts). Captured 2026-08-09.`,
|
|
@@ -334,6 +335,53 @@ Source: Alibaba Cloud Model Studio — "Text-to-video / image-to-video prompt gu
|
|
|
334
335
|
(alibabacloud.com/help/en/model-studio/text-to-video-prompt). Captured 2026-08-09.`,
|
|
335
336
|
}
|
|
336
337
|
|
|
338
|
+
// Wan 3.0 is a SEPARATE doctrine from WAN_DOCTRINE on purpose: Wan 2.x binds
|
|
339
|
+
// references as "Image 1" / "Video 1" (capitalised, WITH a space) while the Wan
|
|
340
|
+
// 3.0 contract uses "Image1" / "Video1" / "Audio1" (no space), and 3.0's surface
|
|
341
|
+
// (adaptive aspect, boolean audio, 30s, mutually-exclusive frame vs reference
|
|
342
|
+
// modes) is different. Folding them together would ship a token format the model
|
|
343
|
+
// does not use. There is no published Wan 3.0 prompt guide, so the body below is
|
|
344
|
+
// KIE contract facts only — no invented vendor style claims.
|
|
345
|
+
const WAN_3_DOCTRINE: ProviderPromptDoctrine = {
|
|
346
|
+
providers: ["wan-3", "wan-3-prime"],
|
|
347
|
+
heading: "Wan 3.0 (wan-3, wan-3-prime)",
|
|
348
|
+
tips: [
|
|
349
|
+
"Two INPUT MODES, exclusive on the wire: first/last frame, OR reference mode (images + videos + audio). With any reference wired the platform folds the frame into the references and names it in the prompt.",
|
|
350
|
+
"References bind by ordinal token in array order: Image1, Image2, Video1, Audio1 — no space, unlike Wan 2.x's \"Image 1\". Name every wired asset or it may be ignored.",
|
|
351
|
+
"Reference caps: 10 images / 5 videos / 5 audio clips; each video and each audio clip 1-15s, with ≤15s combined per array. With reference videos, input seconds + output duration ≤ 30.",
|
|
352
|
+
"2-30 seconds (default 5); 480p/720p/1080p; aspect adaptive (default, matches the input media) or 16:9 / 4:3 / 1:1 / 3:4 / 9:16. Prompt cap 20,000 chars — excess is truncated silently.",
|
|
353
|
+
"`audio` is a boolean, ON by default: the clip comes back with an ambient/SFX track. Cue the sound you want in the prompt, or state the exclusion (\"no music\") — it is not a dialogue guarantee.",
|
|
354
|
+
"wan-3-prime is the HIGH-SPEED tier: identical surface and limits, faster turnaround at a higher per-second rate. It is not a quality upgrade — choose it for latency, not for looks.",
|
|
355
|
+
],
|
|
356
|
+
doctrine: `Prompt structure (no public Wan 3.0 prompt guide exists — the KIE API contract is the
|
|
357
|
+
doctrine source, like MiniMax H3 and HappyHorse; platform-standard structure applies):
|
|
358
|
+
subject → action → scene/environment → lighting → camera movement → style → constraints.
|
|
359
|
+
|
|
360
|
+
**Modes (mutually exclusive at the provider)**
|
|
361
|
+
- Frame mode: first_frame_url, optionally with last_frame_url, and NO references — the frames anchor the shot exactly, so describe MOTION and camera, not the still.
|
|
362
|
+
- Reference mode: image / video / audio reference arrays. The provider CANNOT take these together with the first/last frame parameters, so when both are wired the platform folds — the frame is appended to the reference images (after the caller's own, ordinals unchanged) and bound in the prompt as the opening/closing frame. Write for reference mode whenever a reference is attached.
|
|
363
|
+
- Text-only runs are supported and are the model's default mode.
|
|
364
|
+
|
|
365
|
+
**Reference binding**
|
|
366
|
+
- Assets bind by ORDINAL TOKEN in array order: Image1, Image2, …, Video1, …, Audio1, …. Note the format has NO space — Wan 2.x's "Image 1" is a different generation and does not apply here.
|
|
367
|
+
- Write the binding into the prompt explicitly ("Image1 walks into the room described in Image2"); an unnamed reference may simply be ignored.
|
|
368
|
+
- Caps: up to 10 images, 5 videos, 5 audio clips. Each video and each audio clip must be 1-15s with ≤15s combined per array. Audio should not be the only media input — pair it with an image or a video.
|
|
369
|
+
|
|
370
|
+
**Duration, resolution, aspect**
|
|
371
|
+
- 2-30 seconds (provider default 5). With reference videos there is an extra ceiling: input video duration + output duration ≤ 30 seconds.
|
|
372
|
+
- 480p / 720p / 1080p. Aspect "adaptive" (the default — the model selects the ratio from the input media and intent) or 16:9 / 4:3 / 1:1 / 3:4 / 9:16. There is no 21:9.
|
|
373
|
+
- Prompts accept Chinese and English, up to 20,000 characters; anything beyond is truncated silently, so front-load the load-bearing content.
|
|
374
|
+
|
|
375
|
+
**Audio**
|
|
376
|
+
- The "audio" boolean defaults ON and produces an ambient/SFX track with the clip. Describe the soundscape you want plainly ("rain on glass, distant traffic"), or state the exclusion, or turn the toggle off. The contract documents no lip-synced dialogue guarantee — plan spoken lines as a separate TTS + lip-sync pass.
|
|
377
|
+
|
|
378
|
+
**Tiers**
|
|
379
|
+
- wan-3 and wan-3-prime take identical inputs. Prime trades a higher per-second rate for faster turnaround; it is not documented as a quality tier.
|
|
380
|
+
|
|
381
|
+
Source: KIE Wan 3.0 market contract (docs.kie.ai/market/wan/3-0-video,
|
|
382
|
+
docs.kie.ai/market/wan/3-0-video-prime). Captured 2026-09-01.`,
|
|
383
|
+
}
|
|
384
|
+
|
|
337
385
|
const HAPPYHORSE_DOCTRINE: ProviderPromptDoctrine = {
|
|
338
386
|
providers: ["happyhorse", "happyhorse-i2v", "happyhorse-ref2v", "happyhorse-edit"],
|
|
339
387
|
heading: "HappyHorse 1.1 (happyhorse, happyhorse-i2v, happyhorse-ref2v)",
|
|
@@ -391,6 +439,7 @@ export const PROVIDER_PROMPT_DOCTRINES: readonly ProviderPromptDoctrine[] = [
|
|
|
391
439
|
GEMINI_OMNI_DOCTRINE,
|
|
392
440
|
GROK_IMAGINE_DOCTRINE,
|
|
393
441
|
WAN_DOCTRINE,
|
|
442
|
+
WAN_3_DOCTRINE,
|
|
394
443
|
HAPPYHORSE_DOCTRINE,
|
|
395
444
|
RUNWAY_KIE_DOCTRINE,
|
|
396
445
|
]
|
package/src/setting.ts
CHANGED
|
@@ -85,6 +85,7 @@ export const SETTINGS: ReadonlyArray<Setting> = [
|
|
|
85
85
|
{ id: "parking-lot", label: "Parking Lot", category: "urban", description: "Suburban parking lot at dusk", promptHint: "set in an empty suburban parking lot at dusk with sodium-vapor lamps casting orange pools, scattered shopping carts and painted lane lines" },
|
|
86
86
|
{ id: "penthouse", label: "Penthouse", category: "urban", description: "Luxury penthouse with skyline view", promptHint: "set in a luxury penthouse interior with panoramic skyline views, marble floors, modernist furniture and low warm ambient light" },
|
|
87
87
|
{ id: "gas-station", label: "Gas Station", category: "urban", description: "Lonely highway gas station at night", promptHint: "set at a lonely highway gas station at night with a fluorescent canopy, bug-swarmed sodium lamps and cracked asphalt" },
|
|
88
|
+
{ id: "open-air-market", label: "Open-Air Market", category: "urban", description: "Bustling market of vendor stalls under canopies", term: "open-air market", promptHint: "set in a bustling open-air market — rows of vendor stalls under thatched and canvas canopies, produce piled high, warm dusty light and crowds moving between the stalls" },
|
|
88
89
|
|
|
89
90
|
// -------------------- Nature --------------------
|
|
90
91
|
{ id: "forest", label: "Forest Clearing", category: "nature", description: "Sunlit mossy clearing", promptHint: "set in a sunlit forest clearing with moss-covered stones, dappled light through tall trees and a soft carpet of fallen leaves" },
|
package/src/style.ts
CHANGED
|
@@ -89,6 +89,7 @@ export const STYLES: ReadonlyArray<Style> = [
|
|
|
89
89
|
{ id: "acrylic-paint", label: "Acrylic Paint", description: "Fast-drying opaque acrylic on canvas", promptHint: "rendered as an acrylic painting on canvas, fast-drying opaque pigment with crisp sharp edges, confident quick brush strokes, high-key saturated color and a flatter more graphic finish than traditional oil paint", term: "acrylic painting" },
|
|
90
90
|
{ id: "mixed-media", label: "Mixed Media", description: "Collage + paint + ink hybrid", promptHint: "rendered as a mixed-media artwork combining torn paper collage, acrylic paint, ink and graphite on a layered substrate, heterogeneous textures, visible tape and stitching, and an exuberant hand-assembled studio-art quality" },
|
|
91
91
|
{ id: "manga", label: "Manga", description: "Inked B&W Japanese comic panel", promptHint: "rendered as inked manga panel art, crisp black ink on white with confident line weight variation, screen-tone dot patterns for shading, dramatic speed lines and the distinctly Japanese black-and-white comic aesthetic — separate from full-color anime" },
|
|
92
|
+
{ id: "early-color-photo", label: "Early Color Photo", description: "Prokudin-Gorsky / autochrome early-1900s color", promptHint: "rendered as an early-1900s color photograph in the Prokudin-Gorsky / autochrome tradition — soft three-colour-separation registration, muted dye-toned palette, fine grain and a gentle antique warmth", term: "early autochrome color photograph" },
|
|
92
93
|
] as const
|
|
93
94
|
|
|
94
95
|
const styleById = new Map<string, Style>(STYLES.map((s) => [s.id, s]))
|
package/src/styling.ts
CHANGED
|
@@ -300,6 +300,9 @@ export const STYLINGS: ReadonlyArray<Styling> = [
|
|
|
300
300
|
{ id: "outfit-fairy", label: "Fairy", dimension: "outfit", description: "Fantasy fairy: gauzy wings, flower crown, ethereal dress", promptHint: "wearing a fantasy fairy costume — gauzy translucent wings, a flower crown, and an ethereal flowing dress", term: "fairy costume with wings" },
|
|
301
301
|
{ id: "outfit-mermaid", label: "Mermaid", dimension: "outfit", description: "Fantasy mermaid: scaled tail/skirt, shell top, flowing hair", promptHint: "wearing a fantasy mermaid costume — a scaled tail or fitted scaled skirt, a shell top, and long flowing hair", term: "mermaid costume" },
|
|
302
302
|
{ id: "outfit-pharaoh", label: "Pharaoh Regalia", dimension: "outfit", description: "Ancient Egyptian royalty: usekh collar, pectoral, pleated kilt", promptHint: "wearing ancient Egyptian pharaoh regalia — a broad beaded usekh collar, a jeweled falcon pectoral, and a pleated linen shendyt kilt with golden arm cuffs" },
|
|
303
|
+
{ id: "outfit-workwear-overalls", label: "Workwear Overalls", dimension: "outfit", description: "Denim bib overalls over a plaid flannel shirt", promptHint: "dressed in a farmer's workwear outfit — denim bib overalls over a checked plaid flannel shirt, sturdy and worn-in", term: "denim overalls and plaid shirt" },
|
|
304
|
+
{ id: "outfit-chapan", label: "Chapan Robe", dimension: "outfit", description: "Central Asian long quilted ikat robe", promptHint: "dressed in a traditional Central Asian chapan — a long quilted robe with an ikat-striped weave, tied at the waist with a sash" },
|
|
305
|
+
{ id: "outfit-caftan", label: "Caftan", dimension: "outfit", description: "Long flowing Middle-Eastern / North-African robe", promptHint: "dressed in a long flowing caftan robe, a full-length garment worn across the Middle East and North Africa" },
|
|
303
306
|
|
|
304
307
|
// -------------------- Top (upper-body garment) --------------------
|
|
305
308
|
{ id: "top-tshirt", label: "T-Shirt", dimension: "top", description: "Plain crewneck t-shirt", promptHint: "wearing a fitted plain crewneck t-shirt with short sleeves" },
|
|
@@ -507,6 +510,7 @@ export const STYLING_FIELD_BY_DIMENSION: Record<
|
|
|
507
510
|
* (necklace + earrings + rings), wardrobe-state: 3 (oversized + wet + ripped),
|
|
508
511
|
* hair-state: 2 (wet + windswept). All others single-select (absent → 1). */
|
|
509
512
|
export const MAX_SELECTED_BY_STYLING_DIMENSION: Partial<Record<StylingDimension, number>> = {
|
|
513
|
+
headwear: 2, // a hat layered over a wrap/turban
|
|
510
514
|
jewelry: 3,
|
|
511
515
|
"wardrobe-state": 3,
|
|
512
516
|
"hair-state": 2,
|
|
@@ -520,7 +524,7 @@ export function getStylingDimensionLimit(dimension: StylingDimension): number {
|
|
|
520
524
|
export interface StylingValue {
|
|
521
525
|
makeup?: string
|
|
522
526
|
eyewear?: string
|
|
523
|
-
headwear?: string
|
|
527
|
+
headwear?: string | ReadonlyArray<string>
|
|
524
528
|
/** Hair cut / styling choice — bob, wolf cut, braids, ponytail, etc.
|
|
525
529
|
* Pairs with Person.hair-base (texture + length). */
|
|
526
530
|
hairCut?: string
|