@nodaro/prompts 1.12.0 → 1.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/dist/index.cjs +389 -194
  2. package/dist/index.cjs.map +1 -1
  3. package/dist/index.d.cts +244 -29
  4. package/dist/index.d.ts +244 -29
  5. package/dist/index.js +377 -196
  6. package/dist/index.js.map +1 -1
  7. package/package.json +2 -2
  8. package/src/__tests__/assemble-image-input-cap.test.ts +37 -13
  9. package/src/__tests__/assemble-image-input.test.ts +100 -19
  10. package/src/__tests__/assemble-video-input-cap.test.ts +101 -15
  11. package/src/__tests__/assemble-video-input.test.ts +167 -33
  12. package/src/__tests__/direction-hint-token-safety.test.ts +21 -0
  13. package/src/__tests__/multi-picker-spec.test.ts +21 -1
  14. package/src/__tests__/person-regional-aesthetic.test.ts +2 -1
  15. package/src/__tests__/prompt-style-section.test.ts +345 -0
  16. package/src/__tests__/provider-prompt-doctrine.test.ts +39 -0
  17. package/src/__tests__/style-section-boundary.test.ts +179 -0
  18. package/src/__tests__/subject-fold.test.ts +32 -13
  19. package/src/assemble-image-input.ts +51 -26
  20. package/src/assemble-video-input.ts +54 -34
  21. package/src/direction-registry.ts +108 -37
  22. package/src/gemini-omni-inputs.ts +11 -3
  23. package/src/held-prop.ts +1 -0
  24. package/src/hint-shedding.ts +23 -4
  25. package/src/index.ts +1 -0
  26. package/src/person.ts +4 -0
  27. package/src/picker-analyzer-registry.ts +37 -0
  28. package/src/prompt-builder.ts +84 -27
  29. package/src/prompt-hint-join.ts +9 -0
  30. package/src/prompt-style-section.ts +256 -0
  31. package/src/prompt-wizard-categories.ts +6 -0
  32. package/src/provider-prompt-doctrine.ts +51 -2
  33. package/src/setting.ts +1 -0
  34. package/src/style.ts +1 -0
  35. package/src/styling.ts +5 -1
  36. package/src/video-reference-resolver.ts +5 -2
@@ -454,10 +454,46 @@ export interface MultiPickerAnalyzerSpec {
454
454
  readonly schema: z.ZodType<Record<string, unknown>, unknown>
455
455
  readonly toolName: string
456
456
  readonly legend: string
457
+ /** Compact bullet list of the pickers NOT wired into this spec (PICKER_TYPES
458
+ * minus `types`), keyed by picker-type key so the LLM can ATTRIBUTE a gap to
459
+ * the right picker even when it was not wired. Names + dimension labels only,
460
+ * never catalog ids. Empty string when every picker is already wired. */
461
+ readonly otherPickersLegend: string
457
462
  }
458
463
 
459
464
  const MULTI_CACHE = new Map<string, MultiPickerAnalyzerSpec>()
460
465
 
466
+ /** Title-case a picker-type key for display, e.g. "person" → "Person",
467
+ * "exposure-settings" → "Exposure Settings". */
468
+ function pickerDisplayName(type: string): string {
469
+ return type
470
+ .split("-")
471
+ .map((w) => (w.length > 0 ? w[0].toUpperCase() + w.slice(1) : w))
472
+ .join(" ")
473
+ }
474
+
475
+ /** One compact bullet per non-wired picker so the LLM can attribute a gap to a
476
+ * picker it wasn't handed. Flat pickers show their registry `label` (one axis);
477
+ * discriminated pickers show a title-cased name plus their dimension labels.
478
+ * Never lists catalog ids. Empty string when every PICKER_TYPES member is
479
+ * wired. */
480
+ function buildOtherPickersLegend(sorted: ReadonlyArray<PickerType>): string {
481
+ const otherTypes = PICKER_TYPES.filter((t) => !sorted.includes(t))
482
+ if (otherTypes.length === 0) return ""
483
+ const lines = otherTypes.map((type) => {
484
+ const descriptor = PICKER_ANALYZER_REGISTRY[type as PickerType] as PickerAnalyzerDescriptor
485
+ if (descriptor.kind === "flat") {
486
+ return `- ${type}: ${descriptor.label}`
487
+ }
488
+ const dims = descriptor.order
489
+ .map((k) => descriptor.labels[k])
490
+ .filter(Boolean)
491
+ .join(", ")
492
+ return `- ${type}: ${pickerDisplayName(type)}${dims ? ` — ${dims}` : ""}`
493
+ })
494
+ return `Non-wired pickers — use one of these keys in a gap's \`picker\` when an attribute belongs to it:\n${lines.join("\n")}`
495
+ }
496
+
461
497
  /** Build ONE forced-tool schema spanning the given pickers (each section
462
498
  * optional so an omitted picker doesn't trigger a validation retry) plus the
463
499
  * capped `gaps` sidecar. Memoized by the sorted picker-set key. */
@@ -479,6 +515,7 @@ export function buildMultiPickerAnalyzerSpec(types: ReadonlyArray<PickerType>):
479
515
  schema: z.object(shape).strict() as unknown as MultiPickerAnalyzerSpec["schema"],
480
516
  toolName: "emit_pickers",
481
517
  legend: legendParts.join("\n\n"),
518
+ otherPickersLegend: buildOtherPickersLegend(sorted),
482
519
  }
483
520
  MULTI_CACHE.set(key, result)
484
521
  return result
@@ -7,6 +7,13 @@
7
7
  import { resolveTemplate, applyTemplate } from "./prompt-templates.js"
8
8
  import { NATIVE_NEGATIVE_PROMPT_MODELS, MODELS_WITH_REFERENCE_IMAGE_SUPPORT, imageReferenceLimit, getMaxImagePromptChars, getMaxNegativePromptChars } from "@nodaro/shared"
9
9
  import { getStylePromptHint } from "./style.js"
10
+ import {
11
+ STYLE_SECTION_GAP,
12
+ STYLE_SECTION_HEADER,
13
+ endsInsideStyleSection,
14
+ insertBeforeStyleSection,
15
+ splitStyleSection,
16
+ } from "./prompt-style-section.js"
10
17
  import { findCharacterMentionTokens, type CharacterMentionTokenInfo } from "@nodaro/shared"
11
18
  import { usageModeDirective, DEFAULT_USAGE_MODE, type UsageMode } from "@nodaro/shared"
12
19
  import { roleToPhrase, defaultRoleForSource, REFERENCE_ROLE_PRESETS, normalizeRoleSlug, resolveDefaultRole } from "@nodaro/shared"
@@ -2025,6 +2032,32 @@ function reconcileBodySegments(body: string, bodySegments: readonly PromptSegmen
2025
2032
  return [{ text: body, origin: "user" }]
2026
2033
  }
2027
2034
 
2035
+ /**
2036
+ * The `Style:` / `Avoid:` control lines with the separator each one needs.
2037
+ *
2038
+ * They are self-labeling and stay at the very END of the prompt, which makes
2039
+ * them the only text that can still land under an open `[style]` header — the
2040
+ * section has no terminator, so a blank line is what closes its scope. `Avoid:`
2041
+ * reads the body WITH `Style:` already on it, so once the first control line has
2042
+ * closed the section the second rejoins on a single newline, exactly as before.
2043
+ *
2044
+ * Derived from the body they are appended to, because the cap's tail cut moves
2045
+ * that boundary. The cut can only NARROW the separator — the section is the last
2046
+ * block, so a cut either drops it entirely or lands inside it — which is what
2047
+ * lets the caller reserve on the pre-cut body and re-derive afterwards.
2048
+ */
2049
+ function controlSuffixes(
2050
+ body: string,
2051
+ styleLine: string,
2052
+ avoidLine: string,
2053
+ ): { styleSuffix: string; avoidSuffix: string } {
2054
+ const separator = (text: string): string =>
2055
+ endsInsideStyleSection(text) ? STYLE_SECTION_GAP : "\n"
2056
+ const styleSuffix = styleLine ? `${separator(body)}${styleLine}` : ""
2057
+ const avoidSuffix = avoidLine ? `${separator(body + styleSuffix)}${avoidLine}` : ""
2058
+ return { styleSuffix, avoidSuffix }
2059
+ }
2060
+
2028
2061
  /**
2029
2062
  * Build the final image generation prompt from config.
2030
2063
  * Handles character description wrapping, style appending, negative prompt routing,
@@ -2585,8 +2618,9 @@ function buildImagePromptInternal(config: BuildImagePromptConfig, marks?: Assemb
2585
2618
  ...extrasRendered.elementDirectives,
2586
2619
  ]
2587
2620
  const lockBlock = allLockLines.length > 0 ? `${allLockLines.join("\n")}\n\n` : ""
2588
- const trailingBlock = trailingLines.length > 0 ? `\n${trailingLines.join("\n")}` : ""
2589
- promptForNext = `${lockBlock}${promptForNext}${trailingBlock}`
2621
+ // Scene content, so it extends the BODY: appended flat it would land
2622
+ // under a `[style]` header the composer left open.
2623
+ promptForNext = `${lockBlock}${insertBeforeStyleSection(promptForNext, trailingLines)}`
2590
2624
  }
2591
2625
 
2592
2626
  // Mutate the config locals (NOT the original passed config).
@@ -2731,8 +2765,10 @@ function buildImagePromptInternal(config: BuildImagePromptConfig, marks?: Assemb
2731
2765
  ...locCanon.phrases, ...objCanon.phrases,
2732
2766
  ...locCanon.elementDirectives, ...objCanon.elementDirectives,
2733
2767
  ]
2734
- const canonTrailingBlock = canonTrailingLines.length > 0 ? `\n${canonTrailingLines.join("\n")}` : ""
2735
- const composedScene = `${canonLockBlock}${scene}${canonTrailingBlock}`
2768
+ // Role phrases and element injections are scene content → they extend the
2769
+ // BODY, ahead of the `[style]` section (which has no terminator, so a flat
2770
+ // append would read as one more look clause).
2771
+ const composedScene = `${canonLockBlock}${insertBeforeStyleSection(scene, canonTrailingLines)}`
2736
2772
  prompt = config.referenceLockSnippet
2737
2773
  ? `${config.referenceLockSnippet}\n${composedScene}`
2738
2774
  : composedScene
@@ -2776,24 +2812,20 @@ function buildImagePromptInternal(config: BuildImagePromptConfig, marks?: Assemb
2776
2812
  }
2777
2813
 
2778
2814
  const styleText = style?.trim()
2779
- const styleSuffix = styleText ? `\nStyle: ${getStylePromptHint(styleText) || styleText}` : ""
2815
+ const styleLine = styleText ? `Style: ${getStylePromptHint(styleText) || styleText}` : ""
2780
2816
 
2781
2817
  const negPrompt = negativePrompt?.trim()
2782
2818
  let nativeNegativePrompt: string | undefined
2783
- let avoidSuffix = ""
2819
+ let avoidLine = ""
2784
2820
  if (negPrompt) {
2785
2821
  if (NATIVE_NEGATIVE_PROMPT_MODELS.has(provider)) {
2786
2822
  // Clamp native negatives to the provider's verified cap (e.g. ideogram /
2787
2823
  // qwen = 500) so an over-long negative can't trigger a provider reject.
2788
2824
  nativeNegativePrompt = negPrompt.slice(0, getMaxNegativePromptChars(provider))
2789
2825
  } else {
2790
- avoidSuffix = `\nAvoid: ${negPrompt}`
2826
+ avoidLine = `Avoid: ${negPrompt}`
2791
2827
  }
2792
2828
  }
2793
- if (marks) {
2794
- marks.styleSuffix = styleSuffix
2795
- marks.avoidSuffix = avoidSuffix
2796
- }
2797
2829
 
2798
2830
  // Cap the assembled prompt at the PROVIDER's max (default IMAGE_PROMPT_MAX =
2799
2831
  // 5000, what the image routes already accept) — never the old hardcoded
@@ -2804,11 +2836,19 @@ function buildImagePromptInternal(config: BuildImagePromptConfig, marks?: Assemb
2804
2836
  // The cut is ORDER-BLIND — record how much it removed so a cap-aware caller
2805
2837
  // (`assembleImageInput`) can shed its own lowest-value text and re-assemble.
2806
2838
  const maxLen = getMaxImagePromptChars(provider)
2807
- const reserved = styleSuffix.length + avoidSuffix.length
2839
+ // Reserved on the PRE-cut body; the cut can only NARROW the separator (see
2840
+ // `controlSuffixes`), so the re-derived suffixes always fit the reservation.
2841
+ const preCut = controlSuffixes(prompt, styleLine, avoidLine)
2842
+ const reserved = preCut.styleSuffix.length + preCut.avoidSuffix.length
2808
2843
  if (prompt.length + reserved > maxLen) {
2809
2844
  if (marks) marks.overflowChars += prompt.length + reserved - maxLen
2810
2845
  prompt = prompt.slice(0, Math.max(0, maxLen - reserved - 3)) + "..."
2811
2846
  }
2847
+ const { styleSuffix, avoidSuffix } = controlSuffixes(prompt, styleLine, avoidLine)
2848
+ if (marks) {
2849
+ marks.styleSuffix = styleSuffix
2850
+ marks.avoidSuffix = avoidSuffix
2851
+ }
2812
2852
  // Body span = everything after the captured directive prefix, taken from the
2813
2853
  // possibly-truncated body so the segment join still reconstructs (empty in the
2814
2854
  // Phase-0 consolidation branch → collapses via the fallback, the documented
@@ -2878,37 +2918,38 @@ function buildImagePromptInternal(config: BuildImagePromptConfig, marks?: Assemb
2878
2918
  })
2879
2919
  })
2880
2920
 
2881
- // Assemble prompt
2921
+ // Assemble prompt. The wrapper template composes the BODY only — a `[style]`
2922
+ // section is lifted off first and re-attached after, so a description can
2923
+ // never land under its header (and a user-overridden template, which may put
2924
+ // the descriptions anywhere, still only ever rearranges the body).
2882
2925
  let prompt = config.prompt
2883
2926
  if (charDescs.length > 0) {
2884
2927
  const wrapperTemplate = resolveTemplate("generate-image-wrapper", userTemplates, flowTemplates)
2885
- prompt = applyTemplate(wrapperTemplate, {
2886
- userPrompt: prompt,
2928
+ const { body, section } = splitStyleSection(prompt)
2929
+ const wrapped = applyTemplate(wrapperTemplate, {
2930
+ userPrompt: body,
2887
2931
  assetDescriptions: charDescs.join(" "),
2888
2932
  })
2933
+ prompt = section.length > 0 ? `${wrapped}${STYLE_SECTION_GAP}${section}` : wrapped
2889
2934
  }
2890
2935
 
2891
2936
  // Append style — if the inline `style` is a known STYLES catalog id, inject
2892
2937
  // the richer promptHint; otherwise fall back to the raw text (covers custom
2893
2938
  // free-text styles that don't match a preset).
2894
2939
  const styleText = style?.trim()
2895
- const styleSuffix = styleText ? `\nStyle: ${getStylePromptHint(styleText) || styleText}` : ""
2940
+ const styleLine = styleText ? `Style: ${getStylePromptHint(styleText) || styleText}` : ""
2896
2941
 
2897
2942
  // Handle negative prompt: native support vs prompt-appended
2898
2943
  const negPrompt = negativePrompt?.trim()
2899
2944
  let nativeNegativePrompt: string | undefined
2900
- let avoidSuffix = ""
2945
+ let avoidLine = ""
2901
2946
  if (negPrompt) {
2902
2947
  if (NATIVE_NEGATIVE_PROMPT_MODELS.has(provider)) {
2903
2948
  nativeNegativePrompt = negPrompt.slice(0, getMaxNegativePromptChars(provider))
2904
2949
  } else {
2905
- avoidSuffix = `\nAvoid: ${negPrompt}`
2950
+ avoidLine = `Avoid: ${negPrompt}`
2906
2951
  }
2907
2952
  }
2908
- if (marks) {
2909
- marks.styleSuffix = styleSuffix
2910
- marks.avoidSuffix = avoidSuffix
2911
- }
2912
2953
 
2913
2954
  // Cap at the provider max (default IMAGE_PROMPT_MAX = 5000), reserving the
2914
2955
  // style/avoid suffixes so a long body never severs the appended control text
@@ -2916,11 +2957,17 @@ function buildImagePromptInternal(config: BuildImagePromptConfig, marks?: Assemb
2916
2957
  // BODY, THEN append the suffixes. Same order-blind cut as the
2917
2958
  // connectedReferences path above → same `overflowChars` bookkeeping.
2918
2959
  const maxLen = getMaxImagePromptChars(provider)
2919
- const reserved = styleSuffix.length + avoidSuffix.length
2960
+ const preCut = controlSuffixes(prompt, styleLine, avoidLine)
2961
+ const reserved = preCut.styleSuffix.length + preCut.avoidSuffix.length
2920
2962
  if (prompt.length + reserved > maxLen) {
2921
2963
  if (marks) marks.overflowChars += prompt.length + reserved - maxLen
2922
2964
  prompt = prompt.slice(0, Math.max(0, maxLen - reserved - 3)) + "..."
2923
2965
  }
2966
+ const { styleSuffix, avoidSuffix } = controlSuffixes(prompt, styleLine, avoidLine)
2967
+ if (marks) {
2968
+ marks.styleSuffix = styleSuffix
2969
+ marks.avoidSuffix = avoidSuffix
2970
+ }
2924
2971
  // Legacy path has no directive prefix; the body is the char-desc-wrapped
2925
2972
  // (possibly-truncated) prompt right before the style/avoid suffixes.
2926
2973
  if (marks) marks.bodyBeforeSuffixes = prompt
@@ -3394,15 +3441,25 @@ function capitalizeLineInitial(line: string): string {
3394
3441
 
3395
3442
  /** Render the user prompt as the hybrid scene: each `{image:N:label}` token
3396
3443
  * expanded to its uniform lettered phrase, each line's first letter
3397
- * capitalized. No per-role special-casing — the label drives the phrase. */
3444
+ * capitalized. No per-role special-casing — the label drives the phrase.
3445
+ *
3446
+ * The capitalizer STOPS at the `[style]` section and everything after it: the
3447
+ * header would become `[Style]:`, and every catalog clause under it would gain
3448
+ * a capital it was not written with. The header line is matched EXACTLY — the
3449
+ * composer never indents it, and the whitespace collapses on this path are
3450
+ * horizontal-only (`[^\S\r\n]`), so nothing can pad it before this runs. */
3398
3451
  function buildHybridScene(
3399
3452
  prompt: string,
3400
3453
  refs: readonly ConnectedReference[],
3401
3454
  finalIndexByUrl: ReadonlyMap<string, number>,
3402
3455
  ): string {
3403
- return prompt
3404
- .split("\n")
3405
- .map((line) => capitalizeLineInitial(expandImageRefTokensHybrid(line, refs, finalIndexByUrl)))
3456
+ const lines = prompt.split("\n")
3457
+ const sectionAt = lines.indexOf(STYLE_SECTION_HEADER)
3458
+ return lines
3459
+ .map((line, i) => {
3460
+ const expanded = expandImageRefTokensHybrid(line, refs, finalIndexByUrl)
3461
+ return sectionAt >= 0 && i >= sectionAt ? expanded : capitalizeLineInitial(expanded)
3462
+ })
3406
3463
  .join("\n")
3407
3464
  }
3408
3465
 
@@ -10,6 +10,15 @@
10
10
  * ("prompt. hint", not "prompt . hint"), and drop a blank body so the result
11
11
  * never starts with ". ".
12
12
  *
13
+ * WHAT THIS JOIN COVERS NOW: the prompt BODY only. A LOOK clause no longer
14
+ * reaches this function from either composer — it lifts into the trailing
15
+ * `[style]` section (`prompt-style-section.ts`), which is `". "`-joined WITHIN a
16
+ * line but hung off the body by a blank line. So "every folded clause is one
17
+ * `". "` further along the same string" stopped being true for a look-carrying
18
+ * call, deliberately; `composeSectionedPrompt` is the whole-prompt shape and
19
+ * this is the piece of it that assembles the body. The zero-hint no-op branch is
20
+ * untouched and still the thing the routes' `composed !== prompt` guard reads.
21
+ *
13
22
  * Never mutates its inputs.
14
23
  */
15
24
 
@@ -0,0 +1,256 @@
1
+ /**
2
+ * THE `[style]` SECTION — the trailing block an assembled prompt carries when a
3
+ * run selects any LOOK dimension, shared by the image (`assembleImageInput`)
4
+ * and video (`composeVideoPromptText`) composers so the two surfaces render one
5
+ * shape.
6
+ *
7
+ * <body>
8
+ *
9
+ * [style]:
10
+ * <film line>
11
+ * <scene line>
12
+ *
13
+ * WHY THE LOOK CLAUSES MOVED: folded inline, a broad direction buried the shot
14
+ * — a dozen grade/lighting/era sentences between the user's prose and the
15
+ * structured fields, all in the same register, with nothing telling the model
16
+ * which sentences describe the ACTION and which describe the LOOK. The section
17
+ * says it structurally instead.
18
+ *
19
+ * WHAT STAYS IN THE BODY: the user's prose, the subject fold, the whole MOTION
20
+ * family, and the structured fragment last. Camera motion is part of the shot
21
+ * prose, not the look — so the body/section boundary IS the registry's `family`
22
+ * column, the same column the video verbosity policy splits on. Coupling them
23
+ * is deliberate: one row cannot be shot-prose for the verbosity policy and look
24
+ * for the section.
25
+ *
26
+ * THE SECTION HAS NO TERMINATOR, so "after the section" is not a shape a caller
27
+ * can reach by appending: every assembler downstream of the composer (both
28
+ * reference resolvers, the legacy character-description wrapper, `Style:` /
29
+ * `Avoid:`) joins its text with a single `\n`, which lands it UNDER the header.
30
+ * Two helpers below are how they stay out — `insertBeforeStyleSection` for body
31
+ * content, `endsInsideStyleSection` for the self-labeling control lines that
32
+ * must stay last. A terminator instead would dangle whenever nothing follows,
33
+ * and would break the byte parity below.
34
+ *
35
+ * THE ZERO-CLAUSE CONTRACT (load-bearing): with no look clause — none selected,
36
+ * all shed, or all deduped away — there is NO header and NO extra newline, and
37
+ * the output is byte-identical to the plain hint join. That keeps the
38
+ * verbatim-and-untrimmed no-op alive (zero hints AND zero section → the user's
39
+ * prompt back byte-for-byte, `undefined` included), which is what the routes'
40
+ * `composed !== prompt` guard reads to decide whether to pin
41
+ * `input_data.userPrompt`.
42
+ *
43
+ * NO INDENTATION ANYWHERE: the video reference resolver collapses 2+ horizontal
44
+ * spaces unanchored, so an indented section line would come back flattened. The
45
+ * section is written flush-left rather than relying on that collapse to be
46
+ * harmless.
47
+ */
48
+ import { joinPromptHints, PROMPT_HINT_SEPARATOR } from "./prompt-hint-join.js"
49
+ import {
50
+ renderDirectionHintClauses,
51
+ type DirectionFamily,
52
+ type DirectionFields,
53
+ type DirectionHintMode,
54
+ type DirectionStyleGroup,
55
+ } from "./direction-registry.js"
56
+
57
+ /**
58
+ * The section header, verbatim. Lowercase and bracketed so it reads as
59
+ * structure rather than as a sentence — and so it can be found again by
60
+ * `buildImagePrompt`'s hybrid line-capitalizer, which must stop here
61
+ * (`[Style]:` would be a different token, and capitalized clause lines would
62
+ * corrupt the lowercase catalog wording).
63
+ */
64
+ export const STYLE_SECTION_HEADER = "[style]:"
65
+
66
+ /** The blank line between the body and the section. Omitted for an empty body. */
67
+ export const STYLE_SECTION_GAP = "\n\n"
68
+
69
+ /** The section's opening bytes — the gap, the header and the newline before its
70
+ * first clause line (the section is never emitted without one). */
71
+ const STYLE_SECTION_OPENING = `${STYLE_SECTION_GAP}${STYLE_SECTION_HEADER}\n`
72
+
73
+ /**
74
+ * Split a composed prompt into the BODY and the `[style]` section it ends with
75
+ * (`section: ""` when it carries none, the body then being the whole string;
76
+ * the section comes back WITHOUT the gap).
77
+ *
78
+ * Matched from the RIGHT: the composer always emits the section last, and a
79
+ * user's own prose is free to contain the same characters. A blank body drops
80
+ * the gap with it, so the section-only form is matched on its own.
81
+ */
82
+ export function splitStyleSection(prompt: string): { body: string; section: string } {
83
+ const at = prompt.lastIndexOf(STYLE_SECTION_OPENING)
84
+ if (at >= 0) {
85
+ return { body: prompt.slice(0, at), section: prompt.slice(at + STYLE_SECTION_GAP.length) }
86
+ }
87
+ return prompt.startsWith(`${STYLE_SECTION_HEADER}\n`)
88
+ ? { body: "", section: prompt }
89
+ : { body: prompt, section: "" }
90
+ }
91
+
92
+ /**
93
+ * Extend a composed prompt's BODY with more lines, AHEAD of the `[style]`
94
+ * section — what every assembler downstream of the composer needs, because the
95
+ * section has no terminator: a plain append lands under the header and reads as
96
+ * one more look clause. Reference bindings, element directives and character
97
+ * descriptions are scene content that belongs with the prose, and leaving the
98
+ * look clauses last is where a look tail was measured to cost nothing.
99
+ *
100
+ * With no section this IS the plain `"\n"` join every caller emitted before —
101
+ * the byte-parity path, down to the leading newline an empty prompt produces.
102
+ */
103
+ export function insertBeforeStyleSection(prompt: string, lines: readonly string[]): string {
104
+ if (lines.length === 0) return prompt
105
+ const block = lines.join("\n")
106
+ const { body, section } = splitStyleSection(prompt)
107
+ if (section.length === 0) return `${prompt}\n${block}`
108
+ return `${body.length > 0 ? `${body}\n${block}` : block}${STYLE_SECTION_GAP}${section}`
109
+ }
110
+
111
+ /**
112
+ * True when `prompt` ends INSIDE the section — its last `\n\n`-delimited block
113
+ * opens with the header. What the self-labeling control lines (`Style:`,
114
+ * `Avoid:`) read: they stay at the END of the prompt by design, so a blank line
115
+ * of their own is the only thing that can close the header's scope ahead of
116
+ * them.
117
+ */
118
+ export function endsInsideStyleSection(prompt: string): boolean {
119
+ const at = prompt.lastIndexOf(STYLE_SECTION_GAP)
120
+ const lastBlock = at >= 0 ? prompt.slice(at + STYLE_SECTION_GAP.length) : prompt
121
+ return lastBlock.startsWith(`${STYLE_SECTION_HEADER}\n`)
122
+ }
123
+
124
+ /** Where a rendered clause reads in the assembled prompt. */
125
+ export type PromptClauseSlot = "body" | "film" | "scene"
126
+
127
+ /** A rendered clause plus the line it belongs on. */
128
+ export interface SlottedPromptClause {
129
+ readonly text: string
130
+ readonly slot: PromptClauseSlot
131
+ }
132
+
133
+ /**
134
+ * The slot a direction row's clause takes: motion stays in the body, the five
135
+ * `styleGroup: "film"` rows lead the section, every other look row follows on
136
+ * the scene line.
137
+ */
138
+ export function styleSlotFor(
139
+ row: { readonly family: DirectionFamily; readonly styleGroup?: DirectionStyleGroup },
140
+ ): PromptClauseSlot {
141
+ if (row.family === "motion") return "body"
142
+ return row.styleGroup === "film" ? "film" : "scene"
143
+ }
144
+
145
+ /**
146
+ * The SUBJECT channel's clauses: always body. Who is in the shot is the noun
147
+ * phrase the look modifies, not part of the look.
148
+ */
149
+ export function asBodyClauses(texts: readonly string[]): SlottedPromptClause[] {
150
+ return texts.map((text) => ({ text, slot: "body" as const }))
151
+ }
152
+
153
+ /**
154
+ * Fold a direction bag and tag each clause with its slot, in registry table
155
+ * order. The ORDER is the fold/survival order, not the string order — the
156
+ * composers slice this list from the tail when a cap forces a shed, and only
157
+ * then hand the surviving prefix to `composeSectionedPrompt`.
158
+ */
159
+ export function partitionStyleClauses(
160
+ direction: DirectionFields | undefined,
161
+ opts: { surface: "image" | "video"; mode?: DirectionHintMode },
162
+ ): SlottedPromptClause[] {
163
+ return renderDirectionHintClauses(direction, opts).map((clause) => ({
164
+ text: clause.text,
165
+ slot: styleSlotFor(clause),
166
+ }))
167
+ }
168
+
169
+ /**
170
+ * The section block for a set of clauses — `""` when none of them is a look
171
+ * clause. Each line is omitted entirely when its half of the split is empty, so
172
+ * a scene-only fold never emits a blank film line.
173
+ */
174
+ export function styleSectionFromClauses(clauses: readonly SlottedPromptClause[]): string {
175
+ const line = (slot: PromptClauseSlot) =>
176
+ clauses
177
+ .filter((c) => c.slot === slot)
178
+ .map((c) => c.text)
179
+ .join(PROMPT_HINT_SEPARATOR)
180
+ const lines = [line("film"), line("scene")].filter((l) => l.length > 0)
181
+ return lines.length === 0 ? "" : `${STYLE_SECTION_HEADER}\n${lines.join("\n")}`
182
+ }
183
+
184
+ /**
185
+ * The section for a raw direction bag — the entry point a client renders its
186
+ * preview through, so the preview and the server emit the same bytes.
187
+ */
188
+ export function renderStyleSection(
189
+ direction: DirectionFields | undefined,
190
+ opts: { surface: "image" | "video"; mode?: DirectionHintMode },
191
+ ): string {
192
+ return styleSectionFromClauses(partitionStyleClauses(direction, opts))
193
+ }
194
+
195
+ /**
196
+ * The assembled prompt for a set of clauses: body clauses `". "`-joined onto the
197
+ * user's prompt, the structured fragment last in the body, then the section.
198
+ *
199
+ * TRIMMING follows `joinPromptHints`: the prompt is trimmed whenever ANYTHING
200
+ * folded — a section counts, so a look-only fold trims too, or the blank line
201
+ * would inherit the prompt's trailing whitespace. With nothing folded at all the
202
+ * prompt is returned VERBATIM AND UNTRIMMED (`undefined` passes straight
203
+ * through), which is the byte-parity contract every existing caller rests on.
204
+ *
205
+ * A blank body drops the gap with it, so the result never opens on a newline —
206
+ * the same reason `joinPromptHints` filters a blank prompt out of its join.
207
+ */
208
+ export function composeSectionedPrompt<T extends string | undefined>(
209
+ userPrompt: T,
210
+ clauses: readonly SlottedPromptClause[],
211
+ structuredFragment: string,
212
+ ): string | T {
213
+ const bodyHints = [
214
+ ...clauses.filter((c) => c.slot === "body").map((c) => c.text),
215
+ structuredFragment,
216
+ ].filter((p) => p.length > 0)
217
+ const section = styleSectionFromClauses(clauses)
218
+ if (section.length === 0) {
219
+ return bodyHints.length === 0 ? userPrompt : joinPromptHints(userPrompt ?? "", bodyHints)
220
+ }
221
+ const body =
222
+ bodyHints.length > 0 ? joinPromptHints(userPrompt ?? "", bodyHints) : (userPrompt ?? "").trim()
223
+ return body.length > 0 ? `${body}${STYLE_SECTION_GAP}${section}` : section
224
+ }
225
+
226
+ /**
227
+ * What each clause costs the assembled prompt, as the EXACT composed-length
228
+ * delta of adding it to the prefix below it. Feeds `keepableDirectionHints`,
229
+ * which walks the list tail-first.
230
+ *
231
+ * Exact deltas rather than "clause + separator" because the section's own bytes
232
+ * are not evenly distributed: the FIRST look clause carries the whole
233
+ * `"\n\n[style]:\n"` header (11 characters that only come back when the section
234
+ * disappears), the second look clause of a line carries a `". "`, the first of
235
+ * the scene line carries a `"\n"`. Flat costs would under-price the header, so
236
+ * the walk would cover a deficit with more clauses than it needs and over-shed
237
+ * — visible as a fold that drops two clauses where one would have fit. The
238
+ * deltas do not change between shed iterations (the walk only ever shortens the
239
+ * prefix), so they are computed once, before the loop.
240
+ */
241
+ export function sectionedClauseCosts(
242
+ userPrompt: string | undefined,
243
+ clauses: readonly SlottedPromptClause[],
244
+ structuredFragment: string,
245
+ ): number[] {
246
+ const lengthAt = (kept: number) =>
247
+ composeSectionedPrompt(userPrompt, clauses.slice(0, kept), structuredFragment)?.length ?? 0
248
+ const costs: number[] = []
249
+ let below = lengthAt(0)
250
+ for (let i = 0; i < clauses.length; i++) {
251
+ const at = lengthAt(i + 1)
252
+ costs.push(at - below)
253
+ below = at
254
+ }
255
+ return costs
256
+ }
@@ -256,6 +256,9 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
256
256
  "ltx-2.3-pro": "Lightricks LTX 2.3 Pro — text/image/audio→video, 6–10s, up to 4K",
257
257
  "ltx-2.3-fast": "Lightricks LTX 2.3 Fast — text/image→video, 6–20s, up to 4K",
258
258
  "gemini-omni-video": "Google Gemini Omni — multimodal video with native audio, 4–10s, up to 4K.",
259
+ "gemini-omni-flash": "Google Gemini Omni Flash — faster, cheaper Omni tier; multimodal video with native audio, 4–10s, up to 4K.",
260
+ "wan-3": "Wan 3.0 — multimodal refs (10 images / 5 videos / 5 audio) or first+last frame, native audio, 2–30s, 480p/720p/1080p",
261
+ "wan-3-prime": "Wan 3.0 Prime — high-speed Wan 3.0 tier; same surface, faster turnaround at a higher rate",
259
262
  "grok-imagine-video-1.5": "Grok Imagine 1.5 — image-to-video only; requires an input image",
260
263
  },
261
264
  "image-to-video": {
@@ -289,6 +292,9 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
289
292
  "ltx-2.3-pro": "Lightricks LTX 2.3 Pro — start/end frame i2v + audio→video, 6–10s, up to 4K",
290
293
  "ltx-2.3-fast": "Lightricks LTX 2.3 Fast — start/end frame i2v, 6–20s, up to 4K",
291
294
  "gemini-omni-video": "Google Gemini Omni — multimodal video with native audio, 4–10s, up to 4K.",
295
+ "gemini-omni-flash": "Google Gemini Omni Flash — faster, cheaper Omni tier; multimodal video with native audio, 4–10s, up to 4K.",
296
+ "wan-3": "Wan 3.0 — multimodal refs (10 images / 5 videos / 5 audio) or first+last frame, native audio, 2–30s, 480p/720p/1080p",
297
+ "wan-3-prime": "Wan 3.0 Prime — high-speed Wan 3.0 tier; same surface, faster turnaround at a higher rate",
292
298
  "grok-imagine-video-1.5": "Grok Imagine 1.5 — stylized animation, 1–15s, 480p/720p (image required)",
293
299
  },
294
300
  "video-to-video": {
@@ -239,8 +239,8 @@ KIE VEO API docs (docs.kie.ai/veo3-api/generate-veo-3-video). Captured 2026-08-0
239
239
  }
240
240
 
241
241
  const GEMINI_OMNI_DOCTRINE: ProviderPromptDoctrine = {
242
- providers: ["gemini-omni-video"],
243
- heading: "Gemini Omni Video (gemini-omni-video)",
242
+ providers: ["gemini-omni-video", "gemini-omni-flash"],
243
+ heading: "Gemini Omni (gemini-omni-video, gemini-omni-flash)",
244
244
  tips: [
245
245
  "Multimodal Google video with native audio: text-to-video, image-to-video, and video-edit through the same prompt surface. 4/6/8/10s; 720p/1080p or 4K tier.",
246
246
  "Structure like the platform default: subject → action → scene → lighting → camera → style. Quote dialogue lines to have them spoken; describe SFX/ambience plainly in the prompt.",
@@ -262,6 +262,7 @@ subject → action → scene/environment → lighting → camera movement → st
262
262
 
263
263
  **Duration & tiers**
264
264
  - 4 / 6 / 8 / 10 seconds. 720p/1080p tier or the pricier 4K tier — pick 4K only when the deliverable needs it (nearly 2× the credits).
265
+ - gemini-omni-flash is the faster/cheaper tier with the identical request surface — same 4/6/8/10s, same 720p/1080p and 4K tiers, same video-edit path. Everything above applies verbatim.
265
266
 
266
267
  Source: KIE gemini-omni-video market contract (parameters + live behavior probed for the
267
268
  aspect-ratio hard-reject, see providers/kie/video.ts). Captured 2026-08-09.`,
@@ -334,6 +335,53 @@ Source: Alibaba Cloud Model Studio — "Text-to-video / image-to-video prompt gu
334
335
  (alibabacloud.com/help/en/model-studio/text-to-video-prompt). Captured 2026-08-09.`,
335
336
  }
336
337
 
338
+ // Wan 3.0 is a SEPARATE doctrine from WAN_DOCTRINE on purpose: Wan 2.x binds
339
+ // references as "Image 1" / "Video 1" (capitalised, WITH a space) while the Wan
340
+ // 3.0 contract uses "Image1" / "Video1" / "Audio1" (no space), and 3.0's surface
341
+ // (adaptive aspect, boolean audio, 30s, mutually-exclusive frame vs reference
342
+ // modes) is different. Folding them together would ship a token format the model
343
+ // does not use. There is no published Wan 3.0 prompt guide, so the body below is
344
+ // KIE contract facts only — no invented vendor style claims.
345
+ const WAN_3_DOCTRINE: ProviderPromptDoctrine = {
346
+ providers: ["wan-3", "wan-3-prime"],
347
+ heading: "Wan 3.0 (wan-3, wan-3-prime)",
348
+ tips: [
349
+ "Two INPUT MODES, exclusive on the wire: first/last frame, OR reference mode (images + videos + audio). With any reference wired the platform folds the frame into the references and names it in the prompt.",
350
+ "References bind by ordinal token in array order: Image1, Image2, Video1, Audio1 — no space, unlike Wan 2.x's \"Image 1\". Name every wired asset or it may be ignored.",
351
+ "Reference caps: 10 images / 5 videos / 5 audio clips; each video and each audio clip 1-15s, with ≤15s combined per array. With reference videos, input seconds + output duration ≤ 30.",
352
+ "2-30 seconds (default 5); 480p/720p/1080p; aspect adaptive (default, matches the input media) or 16:9 / 4:3 / 1:1 / 3:4 / 9:16. Prompt cap 20,000 chars — excess is truncated silently.",
353
+ "`audio` is a boolean, ON by default: the clip comes back with an ambient/SFX track. Cue the sound you want in the prompt, or state the exclusion (\"no music\") — it is not a dialogue guarantee.",
354
+ "wan-3-prime is the HIGH-SPEED tier: identical surface and limits, faster turnaround at a higher per-second rate. It is not a quality upgrade — choose it for latency, not for looks.",
355
+ ],
356
+ doctrine: `Prompt structure (no public Wan 3.0 prompt guide exists — the KIE API contract is the
357
+ doctrine source, like MiniMax H3 and HappyHorse; platform-standard structure applies):
358
+ subject → action → scene/environment → lighting → camera movement → style → constraints.
359
+
360
+ **Modes (mutually exclusive at the provider)**
361
+ - Frame mode: first_frame_url, optionally with last_frame_url, and NO references — the frames anchor the shot exactly, so describe MOTION and camera, not the still.
362
+ - Reference mode: image / video / audio reference arrays. The provider CANNOT take these together with the first/last frame parameters, so when both are wired the platform folds — the frame is appended to the reference images (after the caller's own, ordinals unchanged) and bound in the prompt as the opening/closing frame. Write for reference mode whenever a reference is attached.
363
+ - Text-only runs are supported and are the model's default mode.
364
+
365
+ **Reference binding**
366
+ - Assets bind by ORDINAL TOKEN in array order: Image1, Image2, …, Video1, …, Audio1, …. Note the format has NO space — Wan 2.x's "Image 1" is a different generation and does not apply here.
367
+ - Write the binding into the prompt explicitly ("Image1 walks into the room described in Image2"); an unnamed reference may simply be ignored.
368
+ - Caps: up to 10 images, 5 videos, 5 audio clips. Each video and each audio clip must be 1-15s with ≤15s combined per array. Audio should not be the only media input — pair it with an image or a video.
369
+
370
+ **Duration, resolution, aspect**
371
+ - 2-30 seconds (provider default 5). With reference videos there is an extra ceiling: input video duration + output duration ≤ 30 seconds.
372
+ - 480p / 720p / 1080p. Aspect "adaptive" (the default — the model selects the ratio from the input media and intent) or 16:9 / 4:3 / 1:1 / 3:4 / 9:16. There is no 21:9.
373
+ - Prompts accept Chinese and English, up to 20,000 characters; anything beyond is truncated silently, so front-load the load-bearing content.
374
+
375
+ **Audio**
376
+ - The "audio" boolean defaults ON and produces an ambient/SFX track with the clip. Describe the soundscape you want plainly ("rain on glass, distant traffic"), or state the exclusion, or turn the toggle off. The contract documents no lip-synced dialogue guarantee — plan spoken lines as a separate TTS + lip-sync pass.
377
+
378
+ **Tiers**
379
+ - wan-3 and wan-3-prime take identical inputs. Prime trades a higher per-second rate for faster turnaround; it is not documented as a quality tier.
380
+
381
+ Source: KIE Wan 3.0 market contract (docs.kie.ai/market/wan/3-0-video,
382
+ docs.kie.ai/market/wan/3-0-video-prime). Captured 2026-09-01.`,
383
+ }
384
+
337
385
  const HAPPYHORSE_DOCTRINE: ProviderPromptDoctrine = {
338
386
  providers: ["happyhorse", "happyhorse-i2v", "happyhorse-ref2v", "happyhorse-edit"],
339
387
  heading: "HappyHorse 1.1 (happyhorse, happyhorse-i2v, happyhorse-ref2v)",
@@ -391,6 +439,7 @@ export const PROVIDER_PROMPT_DOCTRINES: readonly ProviderPromptDoctrine[] = [
391
439
  GEMINI_OMNI_DOCTRINE,
392
440
  GROK_IMAGINE_DOCTRINE,
393
441
  WAN_DOCTRINE,
442
+ WAN_3_DOCTRINE,
394
443
  HAPPYHORSE_DOCTRINE,
395
444
  RUNWAY_KIE_DOCTRINE,
396
445
  ]
package/src/setting.ts CHANGED
@@ -85,6 +85,7 @@ export const SETTINGS: ReadonlyArray<Setting> = [
85
85
  { id: "parking-lot", label: "Parking Lot", category: "urban", description: "Suburban parking lot at dusk", promptHint: "set in an empty suburban parking lot at dusk with sodium-vapor lamps casting orange pools, scattered shopping carts and painted lane lines" },
86
86
  { id: "penthouse", label: "Penthouse", category: "urban", description: "Luxury penthouse with skyline view", promptHint: "set in a luxury penthouse interior with panoramic skyline views, marble floors, modernist furniture and low warm ambient light" },
87
87
  { id: "gas-station", label: "Gas Station", category: "urban", description: "Lonely highway gas station at night", promptHint: "set at a lonely highway gas station at night with a fluorescent canopy, bug-swarmed sodium lamps and cracked asphalt" },
88
+ { id: "open-air-market", label: "Open-Air Market", category: "urban", description: "Bustling market of vendor stalls under canopies", term: "open-air market", promptHint: "set in a bustling open-air market — rows of vendor stalls under thatched and canvas canopies, produce piled high, warm dusty light and crowds moving between the stalls" },
88
89
 
89
90
  // -------------------- Nature --------------------
90
91
  { id: "forest", label: "Forest Clearing", category: "nature", description: "Sunlit mossy clearing", promptHint: "set in a sunlit forest clearing with moss-covered stones, dappled light through tall trees and a soft carpet of fallen leaves" },