@nodaro/prompts 1.1.1 → 1.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@nodaro/prompts",
3
- "version": "1.1.1",
3
+ "version": "1.2.1",
4
4
  "description": "Nodaro's prompt-engineering layer — person/picker catalogs with prompt hints, identity-lock clauses, entity prompt builders, brand presets, and prompt/reference assembly shared by the Nodaro platform and SDK.",
5
5
  "type": "module",
6
6
  "license": "FSL-1.1-Apache-2.0",
@@ -22,7 +22,7 @@
22
22
  */
23
23
 
24
24
  import { describe, it, expect } from "vitest"
25
- import { PARAMETER_NODE_TYPES, getParameterValue } from "@nodaro/shared"
25
+ import { PARAMETER_NODE_TYPES, HINT_EXEMPT_PARAMETER_TYPES, getParameterValue } from "@nodaro/shared"
26
26
  import { getParameterPromptHint } from "../parameter-prompt-hint.js"
27
27
 
28
28
  // Catalog-driven types need real ids — the prompt-hint builders look them up.
@@ -155,16 +155,11 @@ const SAMPLE_DATA_BY_TYPE: Record<string, Record<string, unknown>> = {
155
155
  }
156
156
 
157
157
  // Types that intentionally do NOT inject a prompt hint via
158
- // getParameterPromptHint. They carry pure runtime parameters (counts,
159
- // durations, aspect ratios, motion intensity) consumed by the executor
160
- // directly, not appended to a downstream prompt.
161
- const HINT_EXEMPT: ReadonlySet<string> = new Set([
162
- "motion",
163
- "style-guide",
164
- "scene-count",
165
- "duration",
166
- "aspect-ratio",
167
- ])
158
+ // getParameterPromptHint. Canonical set lives in @nodaro/shared next to
159
+ // PARAMETER_NODE_TYPES — Test 4 below keeps it honest by asserting each
160
+ // member REALLY returns "" (so a type can't be quietly exempted to dodge a
161
+ // missing-case failure).
162
+ const HINT_EXEMPT = HINT_EXEMPT_PARAMETER_TYPES
168
163
 
169
164
  // =============================================================================
170
165
  // Test 1 — every type in the set has a sample (forces the developer who adds
@@ -21,11 +21,12 @@ describe("seedance-2-extend shared wiring", () => {
21
21
  expect(m.pricing[0]!.identifier).toBe("seedance-2-extend")
22
22
  })
23
23
 
24
- it("stitch constants match the spike-validated recipe (4 tail / 3 head / 0.15s fades)", () => {
24
+ it("stitch constants match the spike-validated recipe (4 tail / 3 head / 0.15s fades / 2s ref tail)", () => {
25
25
  expect(SEEDANCE_2_EXTEND_STITCH).toEqual({
26
26
  trimTailFrames: 4,
27
27
  trimHeadFrames: 3,
28
28
  audioFadeSec: 0.15,
29
+ referenceTailSeconds: 2,
29
30
  })
30
31
  })
31
32
 
@@ -273,6 +273,12 @@ export function getParameterPromptHint(
273
273
 
274
274
  case "tone":
275
275
  return asStr(data.tone).trim()
276
+ case "style-guide":
277
+ // The node's whole purpose is injecting its style text into consumer
278
+ // prompts; without this case, {Style Guide} refs stayed as literal
279
+ // brace text and direct wires injected nothing (only fieldMappings
280
+ // worked, via getParameterValue).
281
+ return asStr(data.text).trim()
276
282
  case "text-prompt":
277
283
  return asStr(data.text).trim()
278
284
  default:
@@ -162,6 +162,7 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
162
162
  "nano-banana": "Fast generation, style flexibility, reference image support",
163
163
  "nano-banana-pro": "Higher quality Nano Banana with better detail",
164
164
  "nano-banana-2": "Latest Nano Banana with resolution options (1K/2K/4K)",
165
+ "nano-banana-2-lite": "Fast, low-cost 1K generation for drafts and iteration",
165
166
  "gpt-image": "Creative concepts, illustration, variable quality tiers",
166
167
  "gpt-image-2": "Latest GPT Image — sharp text, photorealism, 1K/2K/4K resolution",
167
168
  "grok": "General purpose, good text understanding",
@@ -172,6 +173,7 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
172
173
  "qwen": "Versatile, good prompt adherence",
173
174
  "seedream": "Artistic, painterly styles, creative interpretation",
174
175
  "seedream-5-lite": "Lighter Seedream, faster artistic generation",
176
+ "seedream-5-pro": "Flagship Seedream — strongest instruction following, 2K via high quality",
175
177
  "z-image": "Experimental, novel generation approaches",
176
178
  "wan-2.7": "Wan 2.7 T2I — 1K/2K/4K, up to 9 ref images",
177
179
  "wan-2.7-pro": "Wan 2.7 Pro T2I — higher quality, 1K/2K/4K",
@@ -194,6 +196,7 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
194
196
  "qwen-edit": "Instruction-based editing",
195
197
  "seedream-edit": "Artistic style editing",
196
198
  "seedream-5-lite-i2i": "Light artistic transformation",
199
+ "seedream-5-pro-i2i": "Pro-grade instruction-based edits, multi-reference",
197
200
  "flux-kontext": "Character-consistent edits with reference awareness",
198
201
  "flux-kontext-max": "Premium character-consistent editing",
199
202
  "kontext-multi": "Multi-image Kontext via Replicate — up to 4 refs, no safety filter",
@@ -216,6 +219,7 @@ export const PROVIDER_CAPABILITIES: Record<string, Record<string, string>> = {
216
219
  "qwen-edit": "Instruction-based editing",
217
220
  "seedream-edit": "Artistic style editing",
218
221
  "seedream-5-lite-i2i": "Light artistic transformation",
222
+ "seedream-5-pro-i2i": "Pro-grade instruction-based edits, multi-reference",
219
223
  "flux-kontext": "Character-consistent edits with reference awareness",
220
224
  "flux-kontext-max": "Premium character-consistent editing",
221
225
  "kontext-multi": "Multi-image Kontext via Replicate — up to 4 refs, no safety filter",
@@ -7,10 +7,18 @@
7
7
  * 4. compact recipes in MCP tool descriptions point here via get_node_skill
8
8
  *
9
9
  * Sources, in precedence order (conflicts resolve top-down):
10
+ * Seedance 2.0:
10
11
  * - Official BytePlus ModelArk "Dreamina Seedance 2.0 series prompt guide"
11
12
  * https://docs.byteplus.com/en/docs/ModelArk/2222480
12
13
  * - Official launch post https://seed.bytedance.com/en/blog/official-launch-of-seedance-2-0
13
14
  * - KIE API docs https://docs.kie.ai/market/bytedance/seedance-2
15
+ * Kling:
16
+ * - Official "Kling Video 2.6 Audio User Guide"
17
+ * https://kling.ai/quickstart/klingai-video-26-audio-user-guide
18
+ * - fal.ai "Kling 3.0 Prompting Guide" https://blog.fal.ai/kling-3-0-prompting-guide/
19
+ * - KIE API docs https://docs.kie.ai/market/kling/kling-3-0
20
+ * - Live KIE-path dialogue probe 2026-07-16 (scripted lines spoken verbatim,
21
+ * lip-synced, on both kling-2.6 and kling-3.0 with sound=true)
14
22
  */
15
23
  export interface ProviderPromptDoctrine {
16
24
  /** MODEL_CATALOG ids this doctrine covers. */
@@ -66,8 +74,46 @@ precise subject → action details → scene/environment → lighting & color to
66
74
  - Repeated extension degrades quality: prefer high-definition reference assets and avoid stacking many continuations.`,
67
75
  }
68
76
 
77
+ const KLING_AUDIO_DOCTRINE: ProviderPromptDoctrine = {
78
+ providers: ["kling", "kling-3.0", "kling-3-omni"],
79
+ heading: "Kling 2.6 / 3.0 / 3 Omni (kling, kling-3.0, kling-3-omni)",
80
+ tips: [
81
+ "Kling speaks scripted dialogue natively with lip sync — quote the line and enable sound: [Anna: warm calm voice]: \"We made it.\" On kling/kling-3.0 audio raises the credit cost; kling-3-omni includes it.",
82
+ "Structure prompts as Scene → character/element → Motion → Audio → style. Put ALL sound in one 'Audio:' block: dialogue in quotes, then SFX and ambience described plainly ('rain tapping on glass, no music').",
83
+ "Give each speaker a stable label + voice description and reuse it exactly — [Detective: low raspy voice, tired] — pronouns or renamed speakers break voice binding in multi-character scenes.",
84
+ "Tone words in the bracket steer delivery (whispering, crying, fast urgent voice); pace with 'Immediately' / 'after a pause'. 2.6 voices are English/Chinese only; 3.0 adds dialects and code-switching.",
85
+ "kling-3.0 wired references become @element_name mentions (handled automatically by the editor); multi-shot mode forces sound ON. Kling 2.6 prompts cap at 1000 chars — keep the Audio block tight.",
86
+ "Say what should NOT sound: 'no background music, no other sounds' — otherwise Kling invents a music bed under dialogue.",
87
+ ],
88
+ doctrine: `Prompt structure: Scene (setting, light) → Character/Element (who, appearance) → Motion (action, camera) → Audio (dialogue / SFX / ambience / music) → Others (style, emotion).
89
+
90
+ **Dialogue (native speech + lip sync — verified on the KIE path 2026-07-16)**
91
+ - Quote the spoken line and enable the sound toggle; the model bakes the voice AND matching lip movement: the woman says "The quick brown fox jumps over the lazy dog."
92
+ - Prefer labeled dialogue with a voice description: [Character label: voice/tone description]: "line". Example: [Exhausted Partner: trembling frustrated voice]: "You never listen to me."
93
+ - Keep character labels unique and reuse them verbatim — never switch to pronouns mid-prompt; the label is what binds a voice to a speaker across lines. Kling 2.6 additionally supports [Character@VoiceName] platform-voice binding.
94
+ - Tone words inside the bracket steer delivery: whispering, crying voice, controlled serious voice, fast urgent voice. Sequence speech with temporal markers ("Immediately", "after a pause") when two lines must not overlap.
95
+ - Languages: Kling 2.6 outputs English/Chinese voices only (other languages are auto-translated to English). Kling 3.0 supports multiple languages, dialects, accents, and code-switching within one scene — mark the language explicitly ("says in Japanese …").
96
+
97
+ **SFX / ambience / music**
98
+ - Put them in the same Audio block, described plainly: "Rain tapping softly on the window, distant thunder, no music."
99
+ - State exclusions explicitly — "no background music, no other sounds" — or the model tends to add a bed under dialogue.
100
+
101
+ **Toggle + cost**
102
+ - The audio lever is the node's sound toggle (KIE \`sound\` param). On kling (2.6) and kling-3.0 enabling audio raises the credit cost (the \`:audio\` composite); kling-3.0 generates audio by DEFAULT — pass sound: false for the cheaper silent tier. kling-3-omni (Replicate) includes audio in its flat per-duration rate.
103
+ - Multi-shot kling-3.0 (\`multi_shots\`) forces sound ON — budget for the audio rate.
104
+
105
+ **References & elements (kling-3.0 / omni)**
106
+ - Wired references are injected as \`kling_elements\` and MUST be mentioned as @element_name in the prompt — the editor's {image:N} tokens and the server prefixer handle this automatically; when hand-writing prompts, mention every element or it is silently ignored.
107
+ - kling-3-omni is image-to-video only (start frame required) and accepts up to 7 reference images; element voice references (element_input_audio_urls, 5-30s clips) bind a voice to an element.
108
+
109
+ **Limits**
110
+ - Kling 2.6 prompts cap at 1000 characters — front-load scene + dialogue and trim style tails first. kling-3.0 accepts long prompts.
111
+ - Durations: 2.6 = 5/10s; 3.0/omni = 3-15s. A spoken line needs roughly 1s per 2-3 words — don't script more dialogue than the clip can hold.`,
112
+ }
113
+
69
114
  export const PROVIDER_PROMPT_DOCTRINES: readonly ProviderPromptDoctrine[] = [
70
115
  SEEDANCE_2_DOCTRINE,
116
+ KLING_AUDIO_DOCTRINE,
71
117
  ]
72
118
 
73
119
  const DOCTRINE_BY_PROVIDER: ReadonlyMap<string, ProviderPromptDoctrine> = new Map(