@sogni-ai/sogni-protocol 1.0.0-alpha.2 → 1.0.0-alpha.21

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/README.md +10 -1
  2. package/catalogs/audio-models.json +68 -7
  3. package/catalogs/quality-presets.json +3 -3
  4. package/catalogs/seedance-reference-limits.json +35 -0
  5. package/enums/tool-names.json +2 -0
  6. package/manifests/composition-tools.json +3 -3
  7. package/manifests/generation-tools.json +153 -77
  8. package/manifests/openai-tools.json +140 -65
  9. package/package.json +1 -1
  10. package/prompts/tools/animate_photo.json +1 -1
  11. package/prompts/tools/compose_script.json +1 -1
  12. package/prompts/tools/compose_workflow.json +1 -1
  13. package/prompts/tools/compose_workflow_template.json +1 -1
  14. package/prompts/tools/edit_image.json +1 -1
  15. package/prompts/tools/enhance_prompt.json +1 -1
  16. package/prompts/tools/extend_video.json +2 -2
  17. package/prompts/tools/generate_image.json +2 -2
  18. package/prompts/tools/generate_video.json +1 -1
  19. package/prompts/tools/map_assets_for_model.json +1 -1
  20. package/prompts/tools/replace_video_segment.json +2 -2
  21. package/prompts/tools/resolve_personas.json +1 -1
  22. package/prompts/tools/sound_to_video.json +3 -2
  23. package/prompts/tools/video_to_video.json +2 -2
  24. package/schemas/agent/intent-input.schema.json +128 -0
  25. package/schemas/agent/turn-analysis.schema.json +75 -0
  26. package/schemas/artifacts/artifact-graph.schema.json +42 -0
  27. package/schemas/artifacts/artifact-node.schema.json +137 -0
  28. package/schemas/billing/spend-gate.schema.json +151 -0
  29. package/schemas/billing/workflow-authorization.schema.json +83 -0
  30. package/schemas/events/run-event.schema.json +122 -0
  31. package/schemas/tools/animate_photo.schema.json +23 -12
  32. package/schemas/tools/compose_script.schema.json +1 -1
  33. package/schemas/tools/compose_workflow.schema.json +2 -2
  34. package/schemas/tools/compose_workflow_template.schema.json +2 -2
  35. package/schemas/tools/edit_image.schema.json +8 -7
  36. package/schemas/tools/enhance_prompt.schema.json +1 -1
  37. package/schemas/tools/extend_video.schema.json +8 -5
  38. package/schemas/tools/generate_image.schema.json +12 -9
  39. package/schemas/tools/generate_music.schema.json +3 -2
  40. package/schemas/tools/generate_video.schema.json +24 -14
  41. package/schemas/tools/replace_video_segment.schema.json +5 -2
  42. package/schemas/tools/sound_to_video.schema.json +16 -8
  43. package/schemas/tools/tool-metadata.schema.json +78 -0
  44. package/schemas/tools/upscale_image.schema.json +31 -0
  45. package/schemas/tools/video_to_video.schema.json +13 -10
  46. package/schemas/workflows/durable-workflow-run.schema.json +1 -0
  47. package/version.json +1 -1
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@sogni-ai/sogni-protocol",
3
- "version": "1.0.0-alpha.2",
3
+ "version": "1.0.0-alpha.21",
4
4
  "description": "Language-neutral protocol artifacts for the Sogni ecosystem: tool schemas, prompts, OpenAI tool manifests, and enums. Consumed by every Sogni SDK (TypeScript, Swift, and future Python/Kotlin/Rust SDKs) so contracts stay in lockstep across languages.",
5
5
  "keywords": [
6
6
  "sogni",
@@ -2,7 +2,7 @@
2
2
  "contractId": "animate_photo_v1",
3
3
  "version": "1.0.0",
4
4
  "toolName": "animate_photo",
5
- "baseDescription": "animate_photo produces video from one or more source images using LTX 2.3.\n\nVIDEO PROMPT QUOTING: In video prompts, ONLY use double quotes for spoken dialogue.\nSpeaker tags are allowed outside the quotes for screenplay-style dialogue, e.g.\nCHARACTER: \"We made it.\" Never put on-screen text, overlay text, titles, captions, signs,\nwatermarks, or any visual text in quotes — describe them without quotes (e.g. bold white text\nreading CONGRATULATIONS overlays the lower third). Quotes signal speech to the model;\nquoting non-speech text confuses audio generation.\n\nDIALOGUE DURATION: Spoken dialogue in video prompts must fit the clip duration. Estimate\nat 2.5 words per second for natural cinematic delivery, plus ~1 second per acting beat\n(pauses, gestures, glances between lines). If the user did NOT explicitly request a specific\nduration (using default 5s), extend the duration to fit the dialogue (max 20s). If the user\nexplicitly requested a specific duration, condense the dialogue to fit while preserving meaning.\nAlways check: total dialogue words ÷ 2.5 + beat count ≤ clip duration.\n\nLATEST GENERATED IMAGE FOLLOW-UP: When the newest user turn asks to animate, make a video,\nor make a clip from a generated image/result (for example \"the apple\", \"this one\",\n\"the latest image\"), use animate_photo with that latest generated image. Do not inherit an\nolder Seedance model, resolution, or duration from an unrelated prior turn unless the newest\nuser turn explicitly says Seedance or confirms an immediately suggested Seedance video stage.\nLTX supports exact 2-20s durations, so honor requests like 3s exactly.\n\nWORD BUDGET PER CLIP: The handler REJECTS clips whose spoken dialogue exceeds the budget\n— there is NO auto-trim, so plan dialogue lengths up-front. Hard maximum is 3.75 spoken\nwords per second. Ceilings: 5s = 18 words, 6s = 22 words, 8s = 30 words, 10s = 37 words,\n15s = 56 words, 20s = 75 words. Aim below these ceilings. If a scene's dialogue won't fit,\ntighten the lines, raise the per-clip duration, or split into two segments — do NOT submit\nand hope it works. Spoken words inside double quotes count toward the budget; speaker tags\nand visual/action prose are free.\n\nBATCH VIDEO PER-CLIP DURATION: For a multi-segment animate_photo batch\n(sourceImageIndices + prompts) when the user states a TOTAL video length but NO per-clip\nlength, target 15 seconds per clip when dialogue is involved, and pass that duration\nexplicitly. Example: 60s total → 4 segments × 15s, NOT 6×10s or 12×5s. There is NO 3-clip\nbatch cap: sourceImageIndices supports up to 16 clips, so never split one planned batch into\n\"first 3\" and \"remaining clips\" calls. Do NOT split a planned 15s dialogue scene into multiple\nshorter clips just because a retry complains about word budget; keep duration=15 and tighten\nthe line. Use 5s clips only for single short motion beats or one very short spoken phrase.\nIf the user explicitly specifies a per-clip duration, honor that instead.\n\nN-VERSIONS-OF-A-VIDEO PATTERN: NEVER call animate_photo N times sequentially — ALWAYS\nuse sourceImageIndices in ONE call so all N projects run in parallel. Two flavors:\n(A) SHARED CONTENT — one edit_image/generate_image call with numberOfVariations=N + {|}\nDynamic Prompts to make N distinct source images, then ONE animate_photo call with\nsourceImageIndices=[start..start+N-1] and a single shared prompt.\n(B) PER-CLIP CONTENT — when each clip has DIFFERENT dialogue, jokes, narration, or motion,\npass BOTH sourceImageIndices AND prompts (array of N strings, one per clip) in the SAME\nsingle animate_photo call. The top-level prompt is still required — pass a brief batch summary.\n\nCRITICAL: sourceImageIndices values MUST be read from the latest edit_image/generate_image\ntool result's startIndex field — if startIndex=3 and 4 images were generated, pass\nsourceImageIndices=[3,4,5,6], NOT [0,1,2,3]. Negative indices refer to uploaded images:\n-1 first upload, -2 second upload, -3 third upload. Use repeated -1 entries only when\nintentionally reusing the primary uploaded image. When prompts is supplied, prompts.length\nMUST equal sourceImageIndices.length.\n\nSEEDANCE UPLOADED STORYBOARD DEFAULT: If the user uploaded a storyboard, shot sheet,\nor visual trailer board and asks to make a trailer/video/movie/clip from it, do NOT use\nanimate_photo on the board image and do NOT split it into four LTX clips. Use generate_video\nwith Seedance referenceImageIndices for one continuous clip unless the user explicitly asks\nfor separate LTX clips or first-frame/last-frame animation.\n\nSCREENPLAY / STORYBOARD ANIMATE RULE: For full storyboard projects, use one\nanimate_photo batch with sourceImageIndices + prompts so each clip keeps its own exact\nscene text, stable cast anchors, and screenplay-style speaker-tagged dialogue, and all video\nclips render in parallel. Every speaking clip's video prompt must include that clip's actual\nquoted dialogue, not placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\",\nor \"final line lands\". If each generated scene keyframe should be both the first and last frame\nof its own stitched segment, call animate_photo with sourceImageIndices=[start..end],\nframeRole=\"both\", prompts=[...], and OMIT endImageIndex/endImageIndices so the handler\nuses each source as its own end frame.\n\nUPLOADED REFERENCE LOOPED SKITS: When the user supplies one uploaded reference image and\nasks for several scripted/storyboard/dialogue segments to reuse that same image as BOTH the\nfirst frame and last frame of each segment before stitching, do it in ONE animate_photo call:\nsourceImageIndices=[-1,-1,...], frameRole=\"both\", endImageIndex=-1 (or matching\nendImageIndices=[-1,-1,...]), duration equal to the requested per-segment duration, and\nprompts=[one full scene prompt per segment]. Each prompt must preserve the exact screenplay\nspeaker tags and quoted dialogue from that scene, e.g. HOST: \"...\" GUEST: \"...\". Do not\ndrop speaker tags, convert them to generic narration, omit the last-frame contract, analyze\nthe image first, generate new keyframes first, or split the batch into serial calls. After\nthe single animate_photo batch completes, call stitch_video with the returned video indices.\n\nFor adjacent transition chains: N images create N-1 clips — call animate_photo with\nframeRole=\"both\", sourceImageIndices=[start..end-1], endImageIndices=[start+1..end],\nprompts=[one transition prompt per adjacent pair], then stitch_video. If 5 uploaded images\nare the keyframe sequence, use sourceImageIndices=[-1,-2,-3,-4],\nendImageIndices=[-2,-3,-4,-5], frameRole=\"both\", prompts length 4, then stitch_video.\nDo NOT set endImageIndex=-1 in generated-keyframe patterns — that means every clip ends\non the primary uploaded image.\n\nUPLOADED FIRST-FRAME/LAST-FRAME TRANSITION CHAINS: If the user uploads multiple images\nand asks for a video that transitions from image to image, changes country/version every\nN seconds, or says to use first-frame/last-frame for each pair, call animate_photo directly.\nDo not call edit_image, generate_image, analyze_image, or map_assets_for_model first — the\nuploaded images are already the keyframes. For N uploaded images, create N-1 adjacent clips\nunless the user explicitly asks for a loop back to the first image. Use per-clip duration\nfrom \"every N seconds\" when present; otherwise divide the requested total by the number of\nadjacent clips. After animate_photo returns the batch videos, always call stitch_video with\nthose video indices before finalizing.",
5
+ "baseDescription": "animate_photo produces video from one or more source images using LTX 2.5 by default, with LTX 2.3 retained as a rollback path.\n\nVIDEO PROMPT QUOTING: In video prompts, ONLY use double quotes for spoken dialogue.\nSpeaker tags are allowed outside the quotes for screenplay-style dialogue, e.g.\nCHARACTER: \"We made it.\" Never put on-screen text, overlay text, titles, captions, signs,\nwatermarks, or any visual text in quotes — describe them without quotes (e.g. bold white text\nreading CONGRATULATIONS overlays the lower third). Quotes signal speech to the model;\nquoting non-speech text confuses audio generation.\n\nDIALOGUE DURATION: Spoken dialogue in video prompts must fit the clip duration. Estimate\nat 2.5 words per second for natural cinematic delivery, plus ~1 second per acting beat\n(pauses, gestures, glances between lines). If the user did NOT explicitly request a specific\nduration (using default 5s), extend the duration to fit the dialogue (max 20s). If the user\nexplicitly requested a specific duration, condense the dialogue to fit while preserving meaning.\nAlways check: total dialogue words ÷ 2.5 + beat count ≤ clip duration.\n\nLATEST GENERATED IMAGE FOLLOW-UP: When the newest user turn asks to animate, make a video,\nor make a clip from a generated image/result (for example \"the apple\", \"this one\",\n\"the latest image\"), use animate_photo with that latest generated image. Do not inherit an\nolder Seedance model, resolution, or duration from an unrelated prior turn unless the newest\nuser turn explicitly says Seedance or confirms an immediately suggested Seedance video stage.\nLTX supports exact 2-20s durations, so honor requests like 3s exactly.\n\nWORD BUDGET PER CLIP: The handler REJECTS clips whose spoken dialogue exceeds the budget\n— there is NO auto-trim, so plan dialogue lengths up-front. Hard maximum is 3.75 spoken\nwords per second. Ceilings: 5s = 18 words, 6s = 22 words, 8s = 30 words, 10s = 37 words,\n15s = 56 words, 20s = 75 words. Aim below these ceilings. If a scene's dialogue won't fit,\ntighten the lines, raise the per-clip duration, or split into two segments — do NOT submit\nand hope it works. Spoken words inside double quotes count toward the budget; speaker tags\nand visual/action prose are free.\n\nBATCH VIDEO PER-CLIP DURATION: For a multi-segment animate_photo batch\n(sourceImageIndices + prompts) when the user states a TOTAL video length but NO per-clip\nlength, target 15 seconds per clip when dialogue is involved, and pass that duration\nexplicitly. Example: 60s total → 4 segments × 15s, NOT 6×10s or 12×5s. There is NO 3-clip\nbatch cap: sourceImageIndices supports up to 16 clips, so never split one planned batch into\n\"first 3\" and \"remaining clips\" calls. Do NOT split a planned 15s dialogue scene into multiple\nshorter clips just because a retry complains about word budget; keep duration=15 and tighten\nthe line. Use 5s clips only for single short motion beats or one very short spoken phrase.\nIf the user explicitly specifies a per-clip duration, honor that instead.\n\nN-VERSIONS-OF-A-VIDEO PATTERN: NEVER call animate_photo N times sequentially — ALWAYS\nuse sourceImageIndices in ONE call so all N projects run in parallel. Two flavors:\n(A) SHARED CONTENT — one edit_image/generate_image call with numberOfVariations=N + {|}\nDynamic Prompts to make N distinct source images, then ONE animate_photo call with\nsourceImageIndices=[start..start+N-1] and a single shared prompt.\n(B) PER-CLIP CONTENT — when each clip has DIFFERENT dialogue, jokes, narration, or motion,\npass BOTH sourceImageIndices AND prompts (array of N strings, one per clip) in the SAME\nsingle animate_photo call. The top-level prompt is still required — pass a brief batch summary.\n\nCRITICAL: sourceImageIndices values MUST be read from the latest edit_image/generate_image\ntool result's startIndex field — if startIndex=3 and 4 images were generated, pass\nsourceImageIndices=[3,4,5,6], NOT [0,1,2,3]. Negative indices refer to uploaded images:\n-1 first upload, -2 second upload, -3 third upload. Use repeated -1 entries only when\nintentionally reusing the primary uploaded image. When prompts is supplied, prompts.length\nMUST equal sourceImageIndices.length.\n\nSEEDANCE UPLOADED STORYBOARD DEFAULT: If the user uploaded a storyboard, shot sheet,\nor visual trailer board and asks to make a trailer/video/movie/clip from it, do NOT use\nanimate_photo on the board image and do NOT split it into four LTX clips. Use generate_video\nwith Seedance referenceImageIndices for one continuous clip unless the user explicitly asks\nfor separate LTX clips or first-frame/last-frame animation.\n\nSCREENPLAY / STORYBOARD ANIMATE RULE: For full storyboard projects, use one\nanimate_photo batch with sourceImageIndices + prompts so each clip keeps its own exact\nscene text, stable cast anchors, and screenplay-style speaker-tagged dialogue, and all video\nclips render in parallel. Every speaking clip's video prompt must include that clip's actual\nquoted dialogue, not placeholders such as \"while speaking\", \"dialogue begins\", \"explaining\",\nor \"final line lands\". If each generated scene keyframe should be both the first and last frame\nof its own stitched segment, call animate_photo with sourceImageIndices=[start..end],\nframeRole=\"both\", prompts=[...], and OMIT endImageIndex/endImageIndices so the handler\nuses each source as its own end frame.\n\nUPLOADED REFERENCE LOOPED SKITS: When the user supplies one uploaded reference image and\nasks for several scripted/storyboard/dialogue segments to reuse that same image as BOTH the\nfirst frame and last frame of each segment before stitching, do it in ONE animate_photo call:\nsourceImageIndices=[-1,-1,...], frameRole=\"both\", endImageIndex=-1 (or matching\nendImageIndices=[-1,-1,...]), duration equal to the requested per-segment duration, and\nprompts=[one full scene prompt per segment]. Each prompt must preserve the exact screenplay\nspeaker tags and quoted dialogue from that scene, e.g. HOST: \"...\" GUEST: \"...\". Do not\ndrop speaker tags, convert them to generic narration, omit the last-frame contract, analyze\nthe image first, generate new keyframes first, or split the batch into serial calls. After\nthe single animate_photo batch completes, call stitch_video with the returned video indices.\n\nFor adjacent transition chains: N images create N-1 clips — call animate_photo with\nframeRole=\"both\", sourceImageIndices=[start..end-1], endImageIndices=[start+1..end],\nprompts=[one transition prompt per adjacent pair], then stitch_video. If 5 uploaded images\nare the keyframe sequence, use sourceImageIndices=[-1,-2,-3,-4],\nendImageIndices=[-2,-3,-4,-5], frameRole=\"both\", prompts length 4, then stitch_video.\nDo NOT set endImageIndex=-1 in generated-keyframe patterns — that means every clip ends\non the primary uploaded image.\n\nUPLOADED FIRST-FRAME/LAST-FRAME TRANSITION CHAINS: If the user uploads multiple images\nand asks for a video that transitions from image to image, changes country/version every\nN seconds, or says to use first-frame/last-frame for each pair, call animate_photo directly.\nDo not call edit_image, generate_image, analyze_image, or map_assets_for_model first — the\nuploaded images are already the keyframes. For N uploaded images, create N-1 adjacent clips\nunless the user explicitly asks for a loop back to the first image. Use per-clip duration\nfrom \"every N seconds\" when present; otherwise divide the requested total by the number of\nadjacent clips. After animate_photo returns the batch videos, always call stitch_video with\nthose video indices before finalizing.",
6
6
  "parameterDocs": {
7
7
  "sourceImageIndices": "Batch source image indices. Read startIndex from prior generate_image/edit_image result. Negative = uploaded images (-1 = first upload).",
8
8
  "prompts": "Per-clip prompt array. Length MUST equal sourceImageIndices.length when both are set.",
@@ -6,7 +6,7 @@
6
6
  "parameterDocs": {
7
7
  "brief": "The creative writing brief, story idea, product concept, video idea, or revision request.",
8
8
  "script_type": "The kind of script or creative writing artifact to produce. One of video_prompt, screenplay, storyboard, ad_script, trailer, social_short, talking_head, campaign, or revision.",
9
- "destination_model": "Optional destination video model selector, such as ltx23, wan22, or seedance2.",
9
+ "destination_model": "Optional destination video model selector, such as ltx25, ltx23, wan22, or seedance2.",
10
10
  "destination_tool": "Optional downstream tool, such as generate_video, animate_photo, sound_to_video, or video_to_video.",
11
11
  "duration_seconds": "Requested runtime in seconds (1-300) for a video prompt, social short, ad, or talking-head script.",
12
12
  "scene_count": "Requested number of scenes, shots, beats, or storyboard panels (1-12).",
@@ -9,7 +9,7 @@
9
9
  "duration_seconds": "Target total duration in seconds for video-bearing plans (1-120).",
10
10
  "aspect_ratio": "Output aspect ratio. One of 1:1, 4:3, 3:4, 16:9, 9:16, 21:9.",
11
11
  "style": "Optional stylistic guidance such as \"cinematic, neon, low-key\" or \"whiteboard illustration\".",
12
- "destination_models": "Optional preferred image/video/music model selectors (e.g. flux2, ltx23). Each subkey is optional.",
12
+ "destination_models": "Optional preferred image/video/music model selectors (e.g. gpt-image-2, ltx25, ltx23). Each subkey is optional.",
13
13
  "max_estimated_capacity_units": "Optional coarse capacity budget. If set, the planner tries to keep total estimated cost at or below this value; the API returns fits_budget=false if it cannot.",
14
14
  "include_audio": "When true, include a generate_music step in the plan. Defaults to false.",
15
15
  "return_format": "Reserved for future use; only \"json\" is supported in Phase 1. Omit if unsure."
@@ -14,7 +14,7 @@
14
14
  "duration_seconds": "Target total duration in seconds for video-bearing plans (1-120).",
15
15
  "aspect_ratio": "Output aspect ratio. One of 1:1, 4:3, 3:4, 16:9, 9:16, 21:9.",
16
16
  "style": "Optional stylistic guidance such as \"cinematic, neon, low-key\" or \"whiteboard illustration\".",
17
- "destination_models": "Optional preferred image/video/music model selectors (e.g. flux2, ltx23). Each subkey is optional.",
17
+ "destination_models": "Optional preferred image/video/music model selectors (e.g. gpt-image-2, ltx25, ltx23). Each subkey is optional.",
18
18
  "max_estimated_capacity_units": "Optional coarse capacity budget. The planner returns fits_budget=false if it cannot fit under the cap.",
19
19
  "include_audio": "When true, include a generate_music step in the example plan. Defaults to false.",
20
20
  "return_format": "Reserved for future use; only \"json\" is supported in Phase 2. Omit if unsure.",
@@ -2,7 +2,7 @@
2
2
  "contractId": "edit_image_v1",
3
3
  "version": "1.0.0",
4
4
  "toolName": "edit_image",
5
- "baseDescription": "edit_image applies instruction-based edits to uploaded or generated images. Use when\nuploaded or reference images must guide identity or likeness.\n\nImage-to-Image prompt order: [IDENTITY LOCK] → [REQUESTED EDIT] → [REFERENCE ROLE\nMAPPING] → [POSE/COMPOSITION] → [STYLE] → [LIGHTING/REALISM] → [PRESERVE ALL\nUNMENTIONED DETAILS]. GOLDEN RULE: When editing a person, always state which image owns\nidentity — never leave identity ambiguous. Describe only the DELTA — what changes. Don't\nrewrite the entire image; the base image already contains most of the truth. Default to minimal\nchange. For multi-image edits, assign ONE primary role per reference image (identity, pose,\noutfit, style, environment). Never let a style/pose/clothing reference silently override the face.\nUse positive constraints — \"preserve exact facial likeness, face structure, eye shape, nose\nshape, mouth shape, jawline, skin tone, hairline, apparent age, and overall recognizability\"\n— not vague negatives like \"don't mess up the face\".\n\nUPLOADED IMAGE VARIANT SETS: When the user supplies a photo/portrait/reference image and\nasks for N distinct generated images deriving from that source while changing paired\nper-output details, call edit_image exactly once with sourceImageIndex=-1,\nnumberOfVariations=N, and ONE Dynamic Prompt branch with N complete options. Each option\nmust be a full concrete image prompt for one output, including the uploaded subject/reference\nanchor, requested pose or placement preservation, the specific changed appearance/style/role,\nclothing or surface details when relevant, setting/background, and any requested label text or\nvisual symbol. If one option is a remade original/preserved source and the rest are themed\nvariants, the original option must explicitly say to preserve the original clothing/wardrobe/outfit\nand background/setting, plus any requested added label, flag, logo, symbol, or prop.\nDo not call generate_image, analyze_image, or multiple serial edit_image calls first.\n\nSELECTION-GATED IMAGE STAGES: If the user asks for N image options and says they will pick\none before a later dance/video/animation, call edit_image exactly once with numberOfVariations=N\nand one Dynamic Prompt branch. After images are created, stop and ask the user to choose;\ndo not call dance_montage, animate_photo, or generate_video until they select.\n\nMULTI-PERSONA (COMBINED): When multiple personas must appear in the SAME scene, make\nONE edit_image call with ALL persona faces in one prompt and DO NOT pass personaName.\nPer-persona splits (one call each with personaName set) are RARE — only when the user\nexplicitly asks for solo images of each person individually.\n\nSTORYBOARD IMAGE BATCH RULE: When rendering scene keyframes from a screenplay/storyboard,\nnumberOfVariations is only the count; the prompt MUST be one Dynamic Prompt branch with one\nfull keyframe prompt per scene:\n{scene 1 full keyframe prompt|scene 2 full keyframe prompt|...|scene N full keyframe prompt}.\nNEVER set numberOfVariations=N with only the first scene prompt — that creates N versions of\nscene 1. For full project requests, one edit_image batch for all scene keyframes, then one\nanimate_photo batch for all video clips in parallel.\nException: if the storyboard/shot sheet is already uploaded and the user asks to make a\ntrailer/video/movie/clip from that uploaded board, do not extract panels or redraw keyframes.\nUse generate_video with Seedance references for one continuous clip unless the user explicitly\nasks for separate image keyframes or a storyboard sheet output.\n\nDIRECT UPLOADED GPT IMAGE 2 STORYBOARD SHEETS: If the user uploaded reference images and\nasks for one finished GPT Image 2 storyboard/keyframe sheet now, call edit_image directly\nwith sourceImageIndex=-1, model=\"gpt-image-2\", numberOfVariations=1, and the requested\ncanvas/aspect settings. If the user did not explicitly specify a storyboard page/canvas/sheet\nshape, default the GPT Image 2 storyboard sheet pixel dimensions to a balanced grid that hosts\nthe target cell aspect ratio natively (e.g., 12 cells with 9:16 portrait video target -> ~3:4\nportrait sheet around 1728x2304; 12 cells with 16:9 landscape video target -> ~4:3 landscape\nsheet around 2304x1728; 6 cells with 9:16 target -> ~27:32 portrait sheet around 1840x2176). Do\nNOT default the sheet to 2560x1440 landscape when cells are portrait — a landscape sheet with\na portrait-cell grid physically forces cells to ~4:3 landscape and the model will not render\n9:16 portrait rectangles inside it. Keep individual scene-cell/frame areas at the target video\naspect ratio. Do not call map_assets_for_model, analyze_image, generate_image, or a separate\nplanning tool first. The uploaded files are already available as references; describe their\nroles plainly in the edit_image prompt and generate the sheet in that call.\n\nDO NOT USE edit_image FOR UPLOADED REFERENCE LOOPED VIDEO SEGMENTS: If the user says the\nsame uploaded image/reference should be reused as the first frame and last frame of each\nscripted segment/scene/clip before stitching, they are explicitly asking to animate the\nuploaded image, not to generate new storyboard keyframes. Do not call edit_image for that\nrequest. Call animate_photo once with repeated uploaded source indices and per-scene prompts.",
5
+ "baseDescription": "edit_image applies instruction-based edits to uploaded or generated images. Use when\nuploaded or reference images must guide identity or likeness.\n\nImage-to-Image prompt order: [IDENTITY LOCK] → [REQUESTED EDIT] → [REFERENCE ROLE\nMAPPING] → [POSE/COMPOSITION] → [STYLE] → [LIGHTING/REALISM] → [PRESERVE ALL\nUNMENTIONED DETAILS]. GOLDEN RULE: When editing a person, always state which image owns\nidentity — never leave identity ambiguous. Describe only the DELTA — what changes. Don't\nrewrite the entire image; the base image already contains most of the truth. Default to minimal\nchange. For multi-image edits, assign ONE primary role per reference image (identity, pose,\noutfit, style, environment). Never let a style/pose/clothing reference silently override the face.\nUse positive constraints — \"preserve exact facial likeness, face structure, eye shape, nose\nshape, mouth shape, jawline, skin tone, hairline, apparent age, and overall recognizability\"\n— not vague negatives like \"don't mess up the face\".\n\nUPLOADED IMAGE VARIANT SETS: When the user supplies a photo/portrait/reference image and\nasks for N distinct generated images deriving from that source while changing paired\nper-output details, call edit_image exactly once with sourceImageIndex=-1,\nnumberOfVariations=N, and ONE Dynamic Prompt branch with N complete options. Each option\nmust be a full concrete image prompt for one output, including the uploaded subject/reference\nanchor, requested pose or placement preservation, the specific changed appearance/style/role,\nclothing or surface details when relevant, setting/background, and any requested label text or\nvisual symbol. If one option is a remade original/preserved source and the rest are themed\nvariants, the original option must explicitly say to preserve the original clothing/wardrobe/outfit\nand background/setting, plus any requested added label, flag, logo, symbol, or prop.\nDo not call generate_image, analyze_image, or multiple serial edit_image calls first.\n\nSELECTION-GATED IMAGE STAGES: If the user asks for N image options and says they will pick\none before a later dance/video/animation, call edit_image exactly once with numberOfVariations=N\nand one Dynamic Prompt branch. After images are created, stop and ask the user to choose;\ndo not call dance_montage, animate_photo, or generate_video until they select.\n\nMULTI-PERSONA (COMBINED): When multiple personas must appear in the SAME scene, make\nONE edit_image call with ALL persona faces in one prompt and DO NOT pass personaName.\nPer-persona splits (one call each with personaName set) are RARE — only when the user\nexplicitly asks for solo images of each person individually.\n\nSTORYBOARD IMAGE BATCH RULE: When rendering scene keyframes from a screenplay/storyboard,\nnumberOfVariations is only the count; the prompt MUST be one Dynamic Prompt branch with one\nfull keyframe prompt per scene:\n{scene 1 full keyframe prompt|scene 2 full keyframe prompt|...|scene N full keyframe prompt}.\nNEVER set numberOfVariations=N with only the first scene prompt — that creates N versions of\nscene 1. For full project requests, one edit_image batch for all scene keyframes, then one\nanimate_photo batch for all video clips in parallel.\nException: if the storyboard/shot sheet is already uploaded and the user asks to make a\ntrailer/video/movie/clip from that uploaded board, do not extract panels or redraw keyframes.\nUse generate_video with Seedance references for one continuous clip unless the user explicitly\nasks for separate image keyframes or a storyboard sheet output.\n\nDIRECT UPLOADED GPT IMAGE 2 STORYBOARD SHEETS: If the user uploaded reference images and\nasks for one finished GPT Image 2 storyboard/keyframe sheet now, call edit_image directly\nwith sourceImageIndex=-1, model=\"gpt-image-2\", numberOfVariations=1, and the requested\ncanvas/aspect settings. If the user did not explicitly specify a storyboard page/canvas/sheet\nshape, default the GPT Image 2 storyboard sheet pixel dimensions to a balanced grid that hosts\nthe target cell aspect ratio natively (e.g., 12 cells with 9:16 portrait video target -> ~3:4\nportrait sheet around 1728x2304; 12 cells with 16:9 landscape video target -> ~4:3 landscape\nsheet around 2304x1728; 6 cells with 9:16 target -> ~27:32 portrait sheet around 1840x2176). Do\nNOT default the sheet to 2560x1440 landscape when cells are portrait — a landscape sheet with\na portrait-cell grid physically forces cells to ~4:3 landscape and the model will not render\n9:16 portrait rectangles inside it. Keep individual scene-cell/frame areas at the target video\naspect ratio. Do not call map_assets_for_model, analyze_image, generate_image, or a separate\nplanning tool first. The uploaded files are already available as references; describe their\nroles plainly in the edit_image prompt and generate the sheet in that call.\n\nDO NOT USE edit_image FOR UPLOADED REFERENCE LOOPED VIDEO SEGMENTS: If the user says the\nsame uploaded image/reference should be reused as the first frame and last frame of each\nscripted segment/scene/clip before stitching, they are explicitly asking to animate the\nuploaded image, not to generate new storyboard keyframes. Do not call edit_image for that\nrequest. Call animate_photo once with repeated uploaded source indices and per-scene prompts.\n\nKREA IDENTITY EDIT: Use model=\"krea-identity-edit\" whenever an edit of a referenced person or character must keep likeness or character identity while changing clothing, hair or makeup, pose or position, face/head/body, background, lighting, or visual style. Infer that semantic intent in any language; never route from keyword or regex matching. Also use it for a non-Pro single-character sheet. This default applies even when the user did not name Krea; an explicitly requested model always wins. Use model=\"dark-beast-krea2-identity-edit\" only when the user explicitly requests that model, its uncensored/community variant, or dark_beast_krea2_identity_edit_v1_2. These models require one reference image, accept up to two context images, and work best at 512-2048 px. Let the current model tier and worker choose steps, guidance, sampler, scheduler, grounding, and reference-boost defaults; do not send a negative prompt. Put the primary scene/base image first and an optional person/detail reference second. Write a concise 1-4 sentence delta instruction rather than restating the whole image; reference context_image_0 and context_image_1 when model_ref tokens are needed.",
6
6
  "parameterDocs": {
7
7
  "sourceImageIndex": "Index of uploaded/generated image. Use -1 for the first uploaded image.",
8
8
  "numberOfVariations": "Number of output variants. When > 1, use a Dynamic Prompt branch with one complete prompt per output.",
@@ -6,7 +6,7 @@
6
6
  "parameterDocs": {
7
7
  "prompt": "The source prompt, rough idea, or prompt revision request to enhance.",
8
8
  "target_output": "The kind of prompt artifact to produce. One of image_prompt, video_prompt, music_prompt, edit_prompt, model_prompt, or general_prompt.",
9
- "destination_model": "Optional destination model selector, such as seedance2, ltx23, wan22, flux2, gpt-image-2, or sdxl.",
9
+ "destination_model": "Optional destination model selector, such as seedance2, ltx25, ltx23, wan22, gpt-image-2, or sdxl.",
10
10
  "destination_tool": "Optional downstream generation tool, such as generate_image, edit_image, generate_video, animate_photo, sound_to_video, video_to_video, or generate_music.",
11
11
  "prompting_type": "Optional image-prompting family when producing an image prompt. One of flux, sdxl, sd15, pony, fast, sd3, editing, or video.",
12
12
  "model_title": "Optional human-readable target model name for image prompt guidance.",
@@ -2,10 +2,10 @@
2
2
  "contractId": "extend_video_v1",
3
3
  "version": "1.0.0",
4
4
  "toolName": "extend_video",
5
- "baseDescription": "Use extend_video when a video already exists in the session — whether previously rendered OR\nuploaded — and the user asks to make it longer, add a segment, or append a bumper, outro,\nintro, tag, or sting to the end (or start). Do NOT call generate_video, animate_photo, or build\na new bumper from scratch via edit_image+animate_photo+stitch_video — those render fresh\nclips and either waste the previous render or ignore the uploaded base.\n\nFor uploaded base videos, set videoIndex to a negative number (-1 for first uploaded video).\nSet duration to the ADDITIONAL seconds (not the new total).\n\nTrigger phrases: \"make it longer\", \"extend the video\", \"add another N seconds\", \"continue the\nscene\", \"add a bumper/outro/intro/tag/sting to the end (or start)\".\n\nBoth extend_video and replace_video_segment auto-detect the base video's model (Seedance\nsource → Seedance continuation; LTX source → LTX continuation; Wan source → Wan continuation), so OMIT videoModel unless\nthe user explicitly demands a different model. This applies regardless of whether the prior\nrender came from generate_video, animate_photo, sound_to_video, or video_to_video.",
5
+ "baseDescription": "Use extend_video when a video already exists in the session — whether previously rendered OR\nuploaded — and the user asks to make it longer, add a segment, or append a bumper, outro,\nintro, tag, or sting to the end (or start). Do NOT call generate_video, animate_photo, or build\na new bumper from scratch via edit_image+animate_photo+stitch_video — those render fresh\nclips and either waste the previous render or ignore the uploaded base.\n\nFor uploaded base videos, set videoIndex to a negative number (-1 for first uploaded video).\nSet duration to the ADDITIONAL seconds (not the new total).\n\nTrigger phrases: \"make it longer\", \"extend the video\", \"add another N seconds\", \"continue the\nscene\", \"add a bumper/outro/intro/tag/sting to the end (or start)\".\n\nBoth extend_video and replace_video_segment auto-detect the base video's model (Seedance\nsource → Seedance continuation; LTX source → matching LTX continuation; Wan source → Wan continuation). New non-Seedance continuations default to LTX 2.5; use LTX 2.3 only for explicit rollback. OMIT videoModel unless the user explicitly demands a different model. This applies regardless of whether the prior render came from generate_video, animate_photo, sound_to_video, or video_to_video.",
6
6
  "parameterDocs": {
7
7
  "videoIndex": "Index of the existing video. Use -1 for first uploaded video, non-negative for generated videos.",
8
8
  "duration": "ADDITIONAL seconds to add — not the new total length.",
9
- "videoModel": "Omit to auto-detect from source video. Only set if user explicitly requests a different model."
9
+ "videoModel": "Omit to auto-detect from source video. New non-Seedance continuations default to ltx25; use ltx23 only for explicit rollback."
10
10
  }
11
11
  }
@@ -2,9 +2,9 @@
2
2
  "contractId": "generate_image_v1",
3
3
  "version": "1.1.0",
4
4
  "toolName": "generate_image",
5
- "baseDescription": "generate_image creates images from text descriptions. Use for text-only image generation;\nuse edit_image when uploaded or reference images must guide identity/likeness.\nException: Z-image and Z-image Turbo image-to-image/enhancement requests use generate_image\nwith model=\"z-turbo\" or model=\"z-image\", sourceImageIndex=-1, and starting_image_strength;\ndo not route explicit Z-image Turbo uploaded-image enhancement to edit_image because\nedit_image does not expose Z-image models.\n\nFLUX.2 PROMPT ORDER: [SUBJECT] → [ATTRIBUTES] → [ACTION/POSE] → [CAMERA/FRAMING]\n→ [ENVIRONMENT] → [LIGHTING] → [STYLE/MEDIUM] → [MATERIALS/TEXTURES] →\n[SECONDARY DETAILS]. Always start with the main subject, never mood or atmosphere.\nUse concrete nouns and observable adjectives — \"soft overcast daylight\" not \"nice lighting\".\nGood defaults when user is underspecified: medium shot for portraits, wide shot for\nenvironments, eye-level angle, soft natural light for realism.\n\nDYNAMIC PROMPTS: When numberOfVariations > 1, use Dynamic Prompt syntax to make each\nvariation meaningfully different — not just seed-different. Syntax: {a|b|c} cycles\nsequentially, {@a|b|c} picks randomly, {~a|b} paired cycling across groups. Rules: (1) Vary\nONLY what the user left unspecified — lock in everything they specified. (2) Match option\ncount to numberOfVariations so every result is unique. (3) Briefly tell the user what you're\nvarying — never show raw {|} syntax. (4) Skip when: user wants consistency, prompt is fully\nspecified, user typed their own {|} syntax, or iterating on a specific result. (5) NEVER put\nthe count or the word \"versions\"/\"variations\" inside the prompt — the prompt always describes\na single image. The multiplicity comes ONLY from numberOfVariations + the {|} syntax.\nLINKED VARIANTS: when multiple attributes must stay paired per result, use ONE top-level\nDynamic Prompt branch with one complete self-contained prompt per output. Do NOT split\nlinked fields into separate Dynamic Prompt groups.\n\nSELECTION-GATED IMAGE STAGES: If the user asks for N image options and says they will pick\none before a later dance/video/animation, call generate_image once with numberOfVariations=N.\nAfter images are created, stop and ask the user to choose; do not call dance_montage,\nanimate_photo, or generate_video until they select.\n\nIMAGE→VIDEO DIMENSION RULE: When generating an image that will feed into a video tool\n(animate_photo, sound_to_video, etc.), the image MUST be generated at the SAME aspect\nratio and dimensions as the target video. Default video aspect ratio is 16:9 landscape —\npass aspectRatio=\"16:9\" (or the user's specified/reference ratio) so the source image\nmatches the video output. Never generate a square image for a widescreen video. Exception:\na composite GPT Image 2 storyboard/keyframe sheet for a later Seedance video is a board,\nnot a single source frame; unless the user explicitly specifies a storyboard page/canvas/sheet\nshape, default the sheet image dimensions to a balanced grid that hosts the target\nscene-cell/frame aspect natively (portrait video target -> portrait or square sheet whose\ncolumns x rows grid produces ~9:16 cells; landscape video target -> landscape sheet whose\nrows x columns grid produces ~16:9 cells). Each scene-cell/frame area preserves the target\nvideo aspect ratio.\n\nSTORYBOARD IMAGE BATCH RULE: When rendering scene keyframes from a screenplay/storyboard,\nnumberOfVariations is only the count; the prompt MUST be one Dynamic Prompt branch with one\nfull keyframe prompt per scene:\n{scene 1 full keyframe prompt|scene 2 full keyframe prompt|...|scene N full keyframe prompt}.\nNEVER set numberOfVariations=N with only the first scene prompt — that creates N versions of\nscene 1. For full project requests, one generate_image batch for all scene keyframes, then\none animate_photo batch for all video clips in parallel.\n\nSTORYTELLING / BRAND / SOCIAL IMAGE PROMPTS: If generating a storyboard, ad concept,\ntrailer sheet, meme, creator post, or provocative social concept, make the first frame or\npanel immediately legible. Preserve the user's requested tone and audience. Use concrete\ncomposition, persona, product/brand role, caption placement, readable required text, and a\nclear visual transformation or punchline. For provocative adult social content, keep subjects\nclearly adult and consensual, PG-13/non-explicit, and avoid minor-coded styling or school-coded\nsettings while still optimizing visual magnet, persona, caption bait, and replay/comment value.\n\nGPT IMAGE 2 STORYBOARD SHEET → SEEDANCE AUTO-PROCEED: If the user asks to run the whole\nGPT Image 2 storyboard/keyframe sheet plus Seedance workflow without approval, the FIRST\ngenerate_image call must create ONE composite storyboard/keyframe sheet, not loose concept\nart and not separate keyframes. Use model=\"gpt-image-2\", numberOfVariations=1, and a\ncompiled storyboard prompt that literally includes: \"Create exactly N sequential video\nstoryboard frames as one composite storyboard image\", \"Target final video aspect ratio: X\",\na `SCENES:` section, and exactly N concrete scene entries named `SCENE_01`, `SCENE_02`,\netc. Each scene entry must include `Visual/Action:`, `Camera/Motion:`, `Dialogue/VO:`\n(use `[no dialogue]` when silent), `Audio/SFX:`, and any reference/visible-text notes\nneeded for that scene. Do not send only a source brief, storyboard concept, or generic\nlayout instructions as the prompt; malformed compiled storyboard prompts are blocked by\nquality audit instead of being repaired at runtime. Unless the user explicitly specifies another\nstoryboard page/canvas/sheet shape, default the GPT Image 2 storyboard sheet pixel dimensions\nto a balanced grid that hosts the target cell aspect natively: for a 9:16 portrait video,\npick a portrait-leaning sheet whose columns x rows grid produces ~9:16 cells (e.g., 12 cells\n-> ~3:4 sheet around 1728x2304, 6 cells -> ~27:32 around 1840x2176, 9 cells -> ~9:16 around\n1504x2672); for a 16:9 landscape video, pick a landscape sheet whose rows x columns grid\nproduces ~16:9 cells (e.g., 12 cells -> ~4:3 sheet around 2304x1728). Do not force landscape\n2560x1440 when cells are portrait — a landscape sheet with a portrait-cell grid cannot host\n9:16 cells without crushing them. Preserve the requested final video aspect ratio for every\nframe area. After\nthat image completes, call generate_video once using the generated storyboard board as\n@Image1/referenceImageIndices=[0], with skipPromptProcessing=false only when the user\nexplicitly wants the storyboard text rewritten; otherwise preserve the compiled shot guide\nand use skipPromptProcessing=true, expandPrompt=false.\n\nDO NOT USE generate_image FOR UPLOADED REFERENCE LOOPED VIDEO SEGMENTS: If the user says\nthe same uploaded image/reference should be reused as the first frame and last frame of each\nscripted segment/scene/clip before stitching, they are explicitly asking to animate the\nuploaded image, not to generate new storyboard keyframes. Do not call generate_image for\nthat request. Call animate_photo once with repeated uploaded source indices and per-scene\nprompts.\n\nREUSING RESULTS: When the user asks to redo, retry, or revise (e.g., \"try a new version\",\n\"redo the video with X\"), reuse the existing source images — do NOT regenerate them unless\nthe user explicitly asks for new images or describes changes to the images themselves.\nReference the existing result indices from the prior generation. If unsure whether the user\nwants new images, ask — don't regenerate by default.",
5
+ "baseDescription": "generate_image creates images from text descriptions. Use for text-only image generation;\nuse edit_image when uploaded or reference images must guide identity/likeness.\nException: Z-image and Z-image Turbo image-to-image/enhancement requests use generate_image\nwith model=\"z-turbo\" or model=\"z-image\", sourceImageIndex=-1, and starting_image_strength;\ndo not route explicit Z-image Turbo uploaded-image enhancement to edit_image because\nedit_image does not expose Z-image models.\n\nIMAGE PROMPT ORDER: [SUBJECT] → [ATTRIBUTES] → [ACTION/POSE] → [CAMERA/FRAMING]\n→ [ENVIRONMENT] → [LIGHTING] → [STYLE/MEDIUM] → [MATERIALS/TEXTURES] →\n[SECONDARY DETAILS]. Always start with the main subject, never mood or atmosphere.\nUse concrete nouns and observable adjectives — \"soft overcast daylight\" not \"nice lighting\".\nGood defaults when user is underspecified: medium shot for portraits, wide shot for\nenvironments, eye-level angle, soft natural light for realism.\n\nDYNAMIC PROMPTS: When numberOfVariations > 1, use Dynamic Prompt syntax to make each\nvariation meaningfully different — not just seed-different. Syntax: {a|b|c} cycles\nsequentially, {@a|b|c} picks randomly, {~a|b} paired cycling across groups. Rules: (1) Vary\nONLY what the user left unspecified — lock in everything they specified. (2) Match option\ncount to numberOfVariations so every result is unique. (3) Briefly tell the user what you're\nvarying — never show raw {|} syntax. (4) Skip when: user wants consistency, prompt is fully\nspecified, user typed their own {|} syntax, or iterating on a specific result. (5) NEVER put\nthe count or the word \"versions\"/\"variations\" inside the prompt — the prompt always describes\na single image. The multiplicity comes ONLY from numberOfVariations + the {|} syntax.\nLINKED VARIANTS: when multiple attributes must stay paired per result, use ONE top-level\nDynamic Prompt branch with one complete self-contained prompt per output. Do NOT split\nlinked fields into separate Dynamic Prompt groups.\n\nSELECTION-GATED IMAGE STAGES: If the user asks for N image options and says they will pick\none before a later dance/video/animation, call generate_image once with numberOfVariations=N.\nAfter images are created, stop and ask the user to choose; do not call dance_montage,\nanimate_photo, or generate_video until they select.\n\nIMAGE→VIDEO DIMENSION RULE: When generating an image that will feed into a video tool\n(animate_photo, sound_to_video, etc.), the image MUST be generated at the SAME aspect\nratio and dimensions as the target video. Default video aspect ratio is 16:9 landscape —\npass aspectRatio=\"16:9\" (or the user's specified/reference ratio) so the source image\nmatches the video output. Never generate a square image for a widescreen video. Exception:\na composite GPT Image 2 storyboard/keyframe sheet for a later Seedance video is a board,\nnot a single source frame; unless the user explicitly specifies a storyboard page/canvas/sheet\nshape, default the sheet image dimensions to a balanced grid that hosts the target\nscene-cell/frame aspect natively (portrait video target -> portrait or square sheet whose\ncolumns x rows grid produces ~9:16 cells; landscape video target -> landscape sheet whose\nrows x columns grid produces ~16:9 cells). Each scene-cell/frame area preserves the target\nvideo aspect ratio.\n\nSTORYBOARD IMAGE BATCH RULE: When rendering scene keyframes from a screenplay/storyboard,\nnumberOfVariations is only the count; the prompt MUST be one Dynamic Prompt branch with one\nfull keyframe prompt per scene:\n{scene 1 full keyframe prompt|scene 2 full keyframe prompt|...|scene N full keyframe prompt}.\nNEVER set numberOfVariations=N with only the first scene prompt — that creates N versions of\nscene 1. For full project requests, one generate_image batch for all scene keyframes, then\none animate_photo batch for all video clips in parallel.\n\nSTORYTELLING / BRAND / SOCIAL IMAGE PROMPTS: If generating a storyboard, ad concept,\ntrailer sheet, meme, creator post, or provocative social concept, make the first frame or\npanel immediately legible. Preserve the user's requested tone and audience. Use concrete\ncomposition, persona, product/brand role, caption placement, readable required text, and a\nclear visual transformation or punchline. For provocative adult social content, keep subjects\nclearly adult and consensual, PG-13/non-explicit, and avoid minor-coded styling or school-coded\nsettings while still optimizing visual magnet, persona, caption bait, and replay/comment value.\n\nGPT IMAGE 2 STORYBOARD SHEET → SEEDANCE AUTO-PROCEED: If the user asks to run the whole\nGPT Image 2 storyboard/keyframe sheet plus Seedance workflow without approval, the FIRST\ngenerate_image call must create ONE composite storyboard/keyframe sheet, not loose concept\nart and not separate keyframes. Use model=\"gpt-image-2\", numberOfVariations=1, and a\ncompiled storyboard prompt that literally includes: \"Create exactly N sequential video\nstoryboard frames as one composite storyboard image\", \"Target final video aspect ratio: X\",\na `SCENES:` section, and exactly N concrete scene entries named `SCENE_01`, `SCENE_02`,\netc. Each scene entry must include `Visual/Action:`, `Camera/Motion:`, `Dialogue/VO:`\n(use `[no dialogue]` when silent), `Audio/SFX:`, and any reference/visible-text notes\nneeded for that scene. Do not send only a source brief, storyboard concept, or generic\nlayout instructions as the prompt; malformed compiled storyboard prompts are blocked by\nquality audit instead of being repaired at runtime. Unless the user explicitly specifies another\nstoryboard page/canvas/sheet shape, default the GPT Image 2 storyboard sheet pixel dimensions\nto a balanced grid that hosts the target cell aspect natively: for a 9:16 portrait video,\npick a portrait-leaning sheet whose columns x rows grid produces ~9:16 cells (e.g., 12 cells\n-> ~3:4 sheet around 1728x2304, 6 cells -> ~27:32 around 1840x2176, 9 cells -> ~9:16 around\n1504x2672); for a 16:9 landscape video, pick a landscape sheet whose rows x columns grid\nproduces ~16:9 cells (e.g., 12 cells -> ~4:3 sheet around 2304x1728). Do not force landscape\n2560x1440 when cells are portrait — a landscape sheet with a portrait-cell grid cannot host\n9:16 cells without crushing them. Preserve the requested final video aspect ratio for every\nframe area. After\nthat image completes, call generate_video once using the generated storyboard board as\n@Image1/referenceImageIndices=[0], with skipPromptProcessing=false only when the user\nexplicitly wants the storyboard text rewritten; otherwise preserve the compiled shot guide\nand use skipPromptProcessing=true, expandPrompt=false.\n\nDO NOT USE generate_image FOR UPLOADED REFERENCE LOOPED VIDEO SEGMENTS: If the user says\nthe same uploaded image/reference should be reused as the first frame and last frame of each\nscripted segment/scene/clip before stitching, they are explicitly asking to animate the\nuploaded image, not to generate new storyboard keyframes. Do not call generate_image for\nthat request. Call animate_photo once with repeated uploaded source indices and per-scene\nprompts.\n\nREUSING RESULTS: When the user asks to redo, retry, or revise (e.g., \"try a new version\",\n\"redo the video with X\"), reuse the existing source images — do NOT regenerate them unless\nthe user explicitly asks for new images or describes changes to the images themselves.\nReference the existing result indices from the prior generation. If unsure whether the user\nwants new images, ask — don't regenerate by default.\n\nMODEL CHOICE: If the user asks for Krea 2 Turbo, choose model=\"krea-2-turbo\". If they ask for Dark Beast Krea 2, choose model=\"dark-beast-krea2\". If they ask for Dark Beast Z-Image Turbo, choose model=\"dark-beast-z-turbo\". If they ask for Chroma 1 HD, choose model=\"chroma1-hd\". If they ask for anime or anime-style output without naming a model, choose model=\"one-obsession-v22\". Base Z-image and Krea 2 Turbo image-to-image/enhancement requests stay on generate_image with sourceImageIndex and starting_image_strength; identity-preserving Krea edits with reference photos use edit_image instead.",
6
6
  "parameterDocs": {
7
- "prompt": "Text description. Follow FLUX.2 prompt order: subject first. Use Dynamic Prompt syntax when numberOfVariations > 1.",
7
+ "prompt": "Text description. Follow image prompt order: subject first. Use Dynamic Prompt syntax when numberOfVariations > 1.",
8
8
  "numberOfVariations": "Number of distinct outputs. Use Dynamic Prompt {|} syntax to vary one attribute per image. Never put the count in the prompt itself.",
9
9
  "aspectRatio": "For ordinary images feeding a video tool, set to match the target video aspect ratio. For composite GPT Image 2 storyboard sheets, default the sheet pixel dimensions to a balanced grid that hosts the target cell aspect natively (portrait video target -> portrait/square sheet, landscape video target -> landscape sheet) unless the user explicitly specifies a storyboard page/canvas/sheet shape; keep the target video ratio inside each frame area."
10
10
  }
@@ -2,7 +2,7 @@
2
2
  "contractId": "generate_video_v1",
3
3
  "version": "1.1.0",
4
4
  "toolName": "generate_video",
5
- "baseDescription": "generate_video produces text-to-video clips and Seedance multimodal reference videos.\nUse for text-only video generation with no source image input. For Seedance, also use this\ntool when uploaded/generated images, videos, or audio are loose references. Use animate_photo\nonly when a non-Seedance source image must become the first frame of an LTX/WAN animation.\n\nSEEDANCE UPLOADED STORYBOARD DEFAULT: When the user uploads a storyboard, shot sheet,\nmood board, or trailer concept image and asks to make a movie trailer/video/clip from it,\ndefault to one Seedance generate_video call with referenceImageIndices=[-1]. Do not first\nextract panels with edit_image, do not generate replacement keyframes, and do not make four\nseparate LTX animate_photo clips unless the user explicitly asks for separate clips or LTX.\nUse seedance2 when premium Spark access is available; if premium access is unavailable,\nexplain the limitation or use the best non-Seedance fallback the user accepts.\n\nSTORYTELLING / COMMERCIAL / TRAILER PROMPTS: For creative video requests, turn the brief\ninto timed, causally connected visual beats before writing the final prompt. Default social\nvideo is 15s 9:16 with a strong first 1-2s, visible escalation, payoff, and brand/CTA/final\nimage. Commercials should show audience desire/problem, transformation, proof/benefit, and\nCTA. Trailers should follow hook → world → disruption → escalation → reveal → title/CTA.\nEvery beat must be generatable: subject, setting, action, camera, lighting, audio, and text\nrole where relevant. Avoid vague \"cinematic\" filler, feature dumps, and beautiful images with\nno visible change.\n\nVIDEO PROMPT QUOTING: ONLY use double quotes for spoken dialogue in video prompts. Never\nquote on-screen text, titles, captions, or visual text elements — describe them without\nquotes. Quotes signal speech to the model and confuse audio generation.\n\nSTORYBOARD TEXT: Structural headings, section numbers, slide titles, panel titles, and\ncaptions in storyboard references may become short audio-only narration/VO or\nkey-message beats, but they are not subtitles, title cards, lower thirds, or visible\noverlays unless the user explicitly asks for visible text, on-screen text, a title\ncard, subtitle, lower third, signage, or CTA. Keep narration as separate brief phrases\nwith pauses; do not concatenate storyboard labels into run-on voiceover.\n\nDIALOGUE DURATION: Spoken dialogue must fit the clip. Estimate 2.5 words per second\nnatural delivery plus ~1s per acting beat. Hard maximum 3.75 words/second.\nCheck: dialogue words ÷ 2.5 + beats ≤ duration. Do not submit oversized dialogue.\n\nLATEST USER DURATION WINS: In follow-up turns, use the newest duration the user states,\neven if a previous assistant message mentioned a longer script/runtime. For example, if\nhistory says \"the full script is 66 seconds\" but the user now says \"do a 30 second version\",\ngenerate the 30 second version. Do not ask a clarification question just because history\ncontains another duration; treat the latest user request as the override.\n\nSEEDANCE SHORT-DURATION LIMIT: Seedance supports 4-15s clips. If the user explicitly asks\nfor Seedance below 4s, do not silently round up. Ask whether they prefer a 4s Seedance clip\nor an exact-duration LTX clip. If the user did not explicitly ask for Seedance, choose the\nmodel/tool that can satisfy the requested duration exactly.",
5
+ "baseDescription": "generate_video produces text-to-video clips and Seedance multimodal reference videos.\nUse for text-only video generation with no source image input. For Seedance, also use this\ntool when uploaded/generated images, videos, or audio are loose references. Use animate_photo\nonly when a non-Seedance source image must become the first frame of an LTX/WAN animation.\n\nSEEDANCE UPLOADED STORYBOARD DEFAULT: When the user uploads a storyboard, shot sheet,\nmood board, or trailer concept image and asks to make a movie trailer/video/clip from it,\ndefault to one Seedance generate_video call with referenceImageIndices=[-1]. Do not first\nextract panels with edit_image, do not generate replacement keyframes, and do not make four\nseparate LTX animate_photo clips unless the user explicitly asks for separate clips or LTX.\nUse seedance2 when premium Spark access is available; if premium access is unavailable,\nexplain the limitation or use the best non-Seedance fallback the user accepts.\n\nSTORYTELLING / COMMERCIAL / TRAILER PROMPTS: For creative video requests, turn the brief\ninto timed, causally connected visual beats before writing the final prompt. Default social\nvideo is 15s 9:16 with a strong first 1-2s, visible escalation, payoff, and brand/CTA/final\nimage. Commercials should show audience desire/problem, transformation, proof/benefit, and\nCTA. Trailers should follow hook → world → disruption → escalation → reveal → title/CTA.\nEvery beat must be generatable: subject, setting, action, camera, lighting, audio, and text\nrole where relevant. Avoid vague \"cinematic\" filler, feature dumps, and beautiful images with\nno visible change.\n\nVIDEO PROMPT QUOTING: ONLY use double quotes for spoken dialogue in video prompts. Never\nquote on-screen text, titles, captions, or visual text elements — describe them without\nquotes. Quotes signal speech to the model and confuse audio generation.\n\nSTORYBOARD TEXT: Structural headings, section numbers, slide titles, panel titles, and\ncaptions in storyboard references may become short audio-only narration/VO or\nkey-message beats, but they are not subtitles, title cards, lower thirds, or visible\noverlays unless the user explicitly asks for visible text, on-screen text, a title\ncard, subtitle, lower third, signage, or CTA. Keep narration as separate brief phrases\nwith pauses; do not concatenate storyboard labels into run-on voiceover.\n\nDIALOGUE DURATION: Spoken dialogue must fit the clip. Estimate 2.5 words per second\nnatural delivery plus ~1s per acting beat. Hard maximum 3.75 words/second.\nCheck: dialogue words ÷ 2.5 + beats ≤ duration. Do not submit oversized dialogue.\n\nLATEST USER DURATION WINS: In follow-up turns, use the newest duration the user states,\neven if a previous assistant message mentioned a longer script/runtime. For example, if\nhistory says \"the full script is 66 seconds\" but the user now says \"do a 30 second version\",\ngenerate the 30 second version. Do not ask a clarification question just because history\ncontains another duration; treat the latest user request as the override.\n\nSEEDANCE DURATION LIMITS: Seedance 2.0/Mini/Fast support 4-15s clips; Seedance 2.5 supports 4-30s clips. If the user explicitly asks\nfor Seedance below 4s, do not silently round up. Ask whether they prefer a 4s Seedance clip\nor an exact-duration LTX clip. If the user did not explicitly ask for Seedance, choose the\nmodel/tool that can satisfy the requested duration exactly.",
6
6
  "parameterDocs": {
7
7
  "prompt": "Video prompt. Use double quotes ONLY for spoken dialogue. Describe visual text without quotes.",
8
8
  "duration": "Clip duration in seconds. Plan dialogue word count against the 3.75 words/second ceiling."
@@ -2,7 +2,7 @@
2
2
  "contractId": "map_assets_for_model_v1",
3
3
  "version": "1.0.0",
4
4
  "toolName": "map_assets_for_model",
5
- "baseDescription": "map_assets_for_model is an inspection helper for previously generated asset-manifest entries\nwhen you need exact model_ref tokens for a later prompt.\n\nDo NOT call this for ordinary uploaded image references. If the user uploaded images and\nasks GPT Image 2 to use all uploaded assets as visual references, call edit_image directly\nwith sourceImageIndex=-1 and describe the uploaded assets/roles in the prompt. Uploaded\nreference images are already available to edit_image/generate_image; mapping them first\nwastes a tool round and may violate direct-generation requests.\n\nUse this helper only when a previous tool result produced assets in the manifest and the\nnext prompt must name those prior generated assets with provider-specific tokens.",
5
+ "baseDescription": "map_assets_for_model is an inspection helper for previously generated asset-manifest entries\nwhen you need exact model_ref tokens for a later prompt.\n\nDo NOT call this for ordinary uploaded image references. If the user uploaded images and\nasks GPT Image 2 to use all uploaded assets as visual references, call edit_image directly\nwith sourceImageIndex=-1 and describe the uploaded assets/roles in the prompt. Uploaded\nreference images are already available to edit_image/generate_image; mapping them first\nwastes a tool round and may violate direct-generation requests.\n\nUse this helper only when a previous tool result produced assets in the manifest and the\nnext prompt must name those prior generated assets with provider-specific tokens.\n\nContext-conditioned image/edit models such as LTX 2.3, Wan, Qwen Image Edit, Krea 2 Identity Edit, and Dark Beast Krea 2 Identity Edit use zero-indexed model_ref tokens like context_image_0 and context_image_1. Use the helper output instead of hand-formatting those tokens.",
6
6
  "parameterDocs": {
7
7
  "model_id": "Target model for prior generated manifest refs. Do not use for plain uploaded references."
8
8
  }
@@ -2,13 +2,13 @@
2
2
  "contractId": "replace_video_segment_v1",
3
3
  "version": "1.0.0",
4
4
  "toolName": "replace_video_segment",
5
- "baseDescription": "Use replace_video_segment when the user wants to regenerate a specific time range of an\nexisting video: \"regenerate from Xs to Ys\", \"redo the last N seconds\", \"swap out the middle\",\n\"fix the [start/middle/end] of the video\", or \"replace the [bumper/intro/outro/end card/\ntag/sting] at the [start/end] of the video\". Use explicit startSeconds and endSeconds; use\n-1 sentinels when exact base duration is unknown — the handler probes and resolves.\n\nWhen the replacement is already another uploaded or generated video clip, still use\nreplace_video_segment but pass replacementVideoIndex. Example: \"splice video 2 into video 1\nat 5s\" means videoIndex=-1, replacementVideoIndex=-2, startSeconds=5, endSeconds=5.\nUse endSeconds=startSeconds for insertion; use a wider endSeconds only when the user says to\nreplace/remove that base-video range. Do not use stitch_video for \"into the middle\"/\"insert\"\nrequests, because stitch_video only concatenates full clips end-to-end.\n\nFor time-sliced interleaving from existing videos — \"alternate 1s from each video\", \"weave\none-second clips from video 1 and video 2\", \"cut back and forth every N seconds\" — do NOT\nuse stitch_video and do NOT omit replacementVideoIndex. Start with the first requested video\nas the base, then call replace_video_segment once for each window that should come from the\nother video. Set replacementVideoIndex to that other existing video and set\nreplacementStartSeconds/replacementEndSeconds to the next source slice from that\nreplacement video. For ordinary\nalternation, preserve the base duration: set endSeconds=startSeconds+sliceDuration, not\nendSeconds=startSeconds insertion, unless the user explicitly asks to lengthen the output by\ninserting extra slices. Skip no-op windows that already come from the base video; only splice\nwindows that should come from a different source. Example for two 10s uploads alternating every 1s starting with video\n1: replace base windows 1..2, 3..4,\n5..6, 7..8, and 9..10 with slices 0..1, 1..2, 2..3, 3..4, and 4..5 from video 2. After\neach successful splice, target the newest composite video index for the next splice.\nThe -1 time sentinel applies only to base startSeconds/endSeconds when the base duration is\nunknown. Never use -1 for replacementStartSeconds or replacementEndSeconds; source windows\nmust use concrete non-negative seconds. For uploaded/generated videos with duration metadata,\nuse that known duration directly; do not call analyze_video just to learn the clip length for\nroutine alternating slices. Do not add a final tail splice with an unknown source end — stop at\nthe known clip duration or skip a no-op tail window.\n\nDo NOT call generate_video or animate_photo to re-render an existing video just to change\npart of it (the bumper, the intro, the end card, a single scene, the last few seconds, etc.).\nUse replace_video_segment — it preserves the unchanged portion, keeps the original audio\noutside the replaced window, and costs far less.\n\nAuto-detects the base video's model, so OMIT videoModel unless the user explicitly demands\na different model. Short requested windows are supported by rendering with model-specific\nhandles and trimming the rendered clip before splicing, so still pass the user's exact\nstartSeconds/endSeconds.",
5
+ "baseDescription": "Use replace_video_segment when the user wants to regenerate a specific time range of an\nexisting video: \"regenerate from Xs to Ys\", \"redo the last N seconds\", \"swap out the middle\",\n\"fix the [start/middle/end] of the video\", or \"replace the [bumper/intro/outro/end card/\ntag/sting] at the [start/end] of the video\". Use explicit startSeconds and endSeconds; use\n-1 sentinels when exact base duration is unknown — the handler probes and resolves.\n\nWhen the replacement is already another uploaded or generated video clip, still use\nreplace_video_segment but pass replacementVideoIndex. Example: \"splice video 2 into video 1\nat 5s\" means videoIndex=-1, replacementVideoIndex=-2, startSeconds=5, endSeconds=5.\nUse endSeconds=startSeconds for insertion; use a wider endSeconds only when the user says to\nreplace/remove that base-video range. Do not use stitch_video for \"into the middle\"/\"insert\"\nrequests, because stitch_video only concatenates full clips end-to-end.\n\nFor time-sliced interleaving from existing videos — \"alternate 1s from each video\", \"weave\none-second clips from video 1 and video 2\", \"cut back and forth every N seconds\" — do NOT\nuse stitch_video and do NOT omit replacementVideoIndex. Start with the first requested video\nas the base, then call replace_video_segment once for each window that should come from the\nother video. Set replacementVideoIndex to that other existing video and set\nreplacementStartSeconds/replacementEndSeconds to the next source slice from that\nreplacement video. For ordinary\nalternation, preserve the base duration: set endSeconds=startSeconds+sliceDuration, not\nendSeconds=startSeconds insertion, unless the user explicitly asks to lengthen the output by\ninserting extra slices. Skip no-op windows that already come from the base video; only splice\nwindows that should come from a different source. Example for two 10s uploads alternating every 1s starting with video\n1: replace base windows 1..2, 3..4,\n5..6, 7..8, and 9..10 with slices 0..1, 1..2, 2..3, 3..4, and 4..5 from video 2. After\neach successful splice, target the newest composite video index for the next splice.\nThe -1 time sentinel applies only to base startSeconds/endSeconds when the base duration is\nunknown. Never use -1 for replacementStartSeconds or replacementEndSeconds; source windows\nmust use concrete non-negative seconds. For uploaded/generated videos with duration metadata,\nuse that known duration directly; do not call analyze_video just to learn the clip length for\nroutine alternating slices. Do not add a final tail splice with an unknown source end — stop at\nthe known clip duration or skip a no-op tail window.\n\nDo NOT call generate_video or animate_photo to re-render an existing video just to change\npart of it (the bumper, the intro, the end card, a single scene, the last few seconds, etc.).\nUse replace_video_segment — it preserves the unchanged portion, keeps the original audio\noutside the replaced window, and costs far less.\n\nAuto-detects the base video's model. New non-Seedance/WAN replacements default to LTX 2.5; use LTX 2.3 only for explicit rollback. OMIT videoModel unless the user explicitly demands a different model. Short requested windows are supported by rendering with model-specific handles and trimming the rendered clip before splicing, so still pass the user's exact startSeconds/endSeconds.",
6
6
  "parameterDocs": {
7
7
  "startSeconds": "Start of segment to replace in seconds. Use -1 sentinel if exact base duration is unknown.",
8
8
  "endSeconds": "End of segment to replace in seconds. Use the same value as startSeconds for insertion with replacementVideoIndex.",
9
9
  "replacementVideoIndex": "Existing uploaded/generated replacement clip. Use negative uploaded-video indices, e.g. -2 for the second uploaded video.",
10
10
  "replacementStartSeconds": "Optional start time inside replacementVideoIndex. Use with replacementEndSeconds for time-sliced interleaving. Must be concrete and >= 0; never use -1 here.",
11
11
  "replacementEndSeconds": "Optional end time inside replacementVideoIndex. Must be concrete, >= 0, and greater than replacementStartSeconds; never use -1 here.",
12
- "videoModel": "Omit to auto-detect from source. Only set if user explicitly requests a different model."
12
+ "videoModel": "Omit to auto-detect from source. New non-Seedance/WAN replacements default to ltx25; use ltx23 only for explicit rollback."
13
13
  }
14
14
  }
@@ -2,7 +2,7 @@
2
2
  "contractId": "resolve_personas_v1",
3
3
  "version": "1.0.0",
4
4
  "toolName": "resolve_personas",
5
- "baseDescription": "resolve_personas is the required first step when the user explicitly names a saved Persona\nor says to use a Persona Image, Persona reference photo, Persona Voice, registered voice,\nor voice clone. Do not answer in prose, ask a follow-up, or finalize before calling this\ntool when a listed Persona name is present.\n\nDIRECT PERSONA IMAGE / VOICE VIDEO: If the user says to use the Persona image/reference\ndirectly/originally, call resolve_personas first, then call animate_photo using the injected\npersona photo as an uploaded image index. For one named Persona, use sourceImageIndex=-1\nor sourceImageIndices=[-1,...] for a multi-clip batch. If Persona Voice was explicitly\nrequested, set voicePersonaName to the exact resolved Persona name and use an LTX model.\nDo not call generate_video for Persona image/voice videos. Do not generate a new image first\nwhen the user explicitly requested the existing Persona image directly.\n\nMULTI-CLIP PERSONA BATCHES: If the user asks for several separate clips from the same\nPersona Image, make one animate_photo call after resolve_personas with repeated persona\nsource indices, one prompt per clip, and the requested per-clip duration. If the user asks\nto stitch the clips, call stitch_video with the returned video indices after animate_photo.",
5
+ "baseDescription": "resolve_personas is the required first step when the user explicitly names a saved Persona\nor says to use a Persona Image, Persona reference photo, Persona Voice, registered voice,\nor voice clone. Do not answer in prose, ask a follow-up, or finalize before calling this\ntool when a listed Persona name is present.\n\nDIRECT PERSONA IMAGE / VOICE VIDEO: If the user says to use the Persona image/reference\ndirectly/originally, call resolve_personas first, then call animate_photo using the injected\npersona photo as an uploaded image index. For one named Persona, use sourceImageIndex=-1\nor sourceImageIndices=[-1,...] for a multi-clip batch. If Persona Voice was explicitly\nrequested, set voicePersonaName to the exact resolved Persona name and use an LTX model.\nDo not call generate_video for Persona image/voice videos. Do not generate a new image first\nwhen the user explicitly requested the existing Persona image directly.\n\nMULTI-CLIP PERSONA BATCHES: If the user asks for several separate clips from the same\nPersona Image, make one animate_photo call after resolve_personas with repeated persona\nsource indices, one prompt per clip, and the requested per-clip duration. If the user asks\nto stitch the clips, call stitch_video with the returned video indices after animate_photo.\n\nPERSONA IMAGE GENERATION: After resolving persona reference photos for a new persona image, use edit_image rather than generate_image. Krea 2 Identity Edit and Dark Beast Krea 2 Identity Edit are available edit_image model choices when the request needs strong identity preservation from one or two reference photos; use concise identity locks and keep within the two-reference limit.",
6
6
  "parameterDocs": {
7
7
  "names": "Persona names to load. Use the exact listed Persona name; call this before any Persona image/voice video or image generation."
8
8
  }
@@ -2,8 +2,9 @@
2
2
  "contractId": "sound_to_video_v1",
3
3
  "version": "1.0.0",
4
4
  "toolName": "sound_to_video",
5
- "baseDescription": "sound_to_video creates audio-synced video from an audio source. Works with uploaded audio\nfiles (mp3, m4a, wav) OR previously generated music from generate_music (auto-detected).\n\nWhen the user asks to \"turn that song/music into a video\" after generate_music, use\nsound_to_video — it will automatically find the generated audio.\n\nFor music visualization (syncing video to a specific song or audio track), use the\ngenerate_music → sound_to_video pipeline. Do NOT use animate_photo or generate_video for\naudio-driven visualization.\n\nanimate_photo and generate_video produce audio natively via LTX 2.3 — never pre-generate\naudio for those tools. sound_to_video is only for when the audio IS the primary creative\ninput driving the video output.",
5
+ "baseDescription": "sound_to_video creates audio-synced video from an audio source. Works with uploaded audio\nfiles (mp3, m4a, wav) OR previously generated music from generate_music (auto-detected).\n\nWhen the user asks to \"turn that song/music into a video\" after generate_music, use\nsound_to_video — it will automatically find the generated audio.\n\nFor music visualization (syncing video to a specific song or audio track), use the\ngenerate_music → sound_to_video pipeline. Do NOT use animate_photo or generate_video for\naudio-driven visualization.\n\nanimate_photo and generate_video produce audio natively via LTX 2.5 by default, with LTX 2.3 retained as rollback — never pre-generate\naudio for those tools. sound_to_video is only for when the audio IS the primary creative\ninput driving the video output.",
6
6
  "parameterDocs": {
7
- "audioSource": "Uploaded audio file or reference to a prior generate_music result. Auto-detected when omitted after generate_music."
7
+ "audioSource": "Uploaded audio file or reference to a prior generate_music result. Auto-detected when omitted after generate_music.",
8
+ "negativePrompt": "Advanced LTX 2.5/LTX 2.3/WAN only. The LTX A2V and IA2V workflows accept a separate negative prompt; do not set it for Seedance."
8
9
  }
9
10
  }
@@ -2,11 +2,11 @@
2
2
  "contractId": "video_to_video_v1",
3
3
  "version": "1.0.0",
4
4
  "toolName": "video_to_video",
5
- "baseDescription": "video_to_video transforms an uploaded video. Use for uploaded-video restyling, enhancement,\nupscaling/remastering, motion transfer from video to image, subject replacement, edge/pose/\ndepth-guided restyle, or explicit Seedance V2V transforms.\n\nThis tool requires an uploaded video source. Do not use it for generated video indices. For\ngenerated or uploaded partial edits use replace_video_segment; for appended time use\nextend_video; for logos/text overlays use overlay_video; for stitching use stitch_video.\n\nChoose controlMode by intent. Use detailer for quality-only enhancement without restyling.\nUse seedance-v2v only when the user asks to transform/enhance/remaster an uploaded video\nwith Seedance. For detailer, describe the original scene plus quality terms, not new content.",
5
+ "baseDescription": "video_to_video transforms an uploaded video. LTX 2.5 is the default for canny, pose, depth, detailer, inpaint, and outpaint; use LTX 2.3 only for explicit rollback. LTX 2.5 Distilled supports all six controls, while Dev + Speed LoRA supports canny, pose, depth, and detailer. Use WAN Animate for motion transfer or subject replacement, and Seedance V2V only when explicitly requested.\n\nThis tool requires an uploaded video source. Do not use it for generated video indices. For generated or uploaded partial edits use replace_video_segment; for appended time use extend_video; for logos/text overlays use overlay_video; for stitching use stitch_video.\n\nChoose controlMode by intent. Use detailer for quality-only enhancement without restyling. For detailer, describe the original scene plus quality terms, not new content.",
6
6
  "parameterDocs": {
7
7
  "prompt": "Describe the target appearance in present tense. For detailer, describe the original content plus quality qualifiers only.",
8
8
  "videoSourceIndex": "Uploaded video index. Omit when there is one uploaded video; use 0 for first uploaded video or -1 if using negative upload notation.",
9
- "controlMode": "Pick from intent: detailer for enhance, seedance-v2v for explicit Seedance V2V, canny/depth/pose for control-net restyles, animate-move/replace for WAN Animate.",
9
+ "controlMode": "Pick from intent: detailer for enhance, inpaint/outpaint for masked or canvas edits, seedance-v2v for explicit Seedance V2V, canny/depth/pose for control-guided restyles, animate-move/replace for WAN Animate.",
10
10
  "sourceImageIndex": "Required for animate-move and animate-replace. Ignored by canny, depth, and detailer.",
11
11
  "duration": "Set only when the user requests a different output length; otherwise let the tool match/cap the source duration."
12
12
  }
@@ -0,0 +1,128 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://schemas.sogni.ai/creative-agent/2026-05-20.1/agent/intent-input.schema.json",
4
+ "title": "L1 IntentClassifier input packet",
5
+ "schemaVersion": "2026-05-20.1",
6
+ "description": "Compact context packet handed to the L1 IntentClassifier at the start of every user turn. Qwen3 256k context window comfortably holds 16-reference-image generations and multi-segment storyboard frames; the runtime never silently drops artifacts. Trim only when measured budget pressure forces it. This contract is built deterministically by the host (browser or cloud runner) from explicit runtime state; it never re-derives semantics from transcript regex.",
7
+ "type": "object",
8
+ "additionalProperties": false,
9
+ "properties": {
10
+ "currentMessage": {
11
+ "type": "string",
12
+ "description": "Raw user message text for this turn. Verbatim; sanitization belongs to boundary code, not this contract."
13
+ },
14
+ "currentMessageDetails": {
15
+ "type": "object",
16
+ "additionalProperties": false,
17
+ "description": "Optional structured form of the user's latest message. Producers MAY emit either currentMessage or this; consumers MUST handle BOTH. `text` is required so consumers can always derive a string-level view.",
18
+ "properties": {
19
+ "id": { "type": "string" },
20
+ "role": { "type": "string", "enum": ["user", "system"] },
21
+ "text": { "type": "string" },
22
+ "createdAt": { "type": "string", "format": "date-time" },
23
+ "localeHint": { "type": "string" }
24
+ },
25
+ "required": ["text"]
26
+ },
27
+ "runtimeFlags": {
28
+ "type": "object",
29
+ "additionalProperties": false,
30
+ "description": "Optional runtime feature flags the planner reads to decide tool surface and confirmation policy.",
31
+ "properties": {
32
+ "surface": {
33
+ "type": "string",
34
+ "enum": ["browser", "hosted_chat", "durable_chat", "workflow", "native", "skill"],
35
+ "description": "Originating producer surface."
36
+ },
37
+ "allowPaidTools": { "type": "boolean" },
38
+ "allowMutatingTools": { "type": "boolean" },
39
+ "durableRequired": { "type": "boolean" }
40
+ }
41
+ },
42
+ "activeState": {
43
+ "type": "object",
44
+ "additionalProperties": false,
45
+ "description": "Currently-resolved focus state owned by the runtime (artifact graph + workflow runner). Populated from explicit runtime references — never from prose extraction.",
46
+ "properties": {
47
+ "activeArtifactId": { "type": "string" },
48
+ "activeArtifactType": {
49
+ "type": "string",
50
+ "enum": ["image", "video", "audio", "text", "workflow", "collection"]
51
+ },
52
+ "pendingAction": {
53
+ "type": "object",
54
+ "description": "Opaque reference to a runtime-tracked pending action (e.g. proposed plan awaiting selection). Shape is owned by the runtime, not this schema.",
55
+ "additionalProperties": true
56
+ },
57
+ "awaitingConfirmation": { "type": "boolean" },
58
+ "lastToolResult": {
59
+ "type": "object",
60
+ "additionalProperties": false,
61
+ "properties": {
62
+ "toolName": { "type": "string" },
63
+ "toolCallId": { "type": "string" },
64
+ "status": { "type": "string" }
65
+ },
66
+ "required": ["toolName", "toolCallId", "status"]
67
+ },
68
+ "activeWorkflowRunId": { "type": "string" }
69
+ }
70
+ },
71
+ "artifactState": {
72
+ "type": "object",
73
+ "additionalProperties": false,
74
+ "description": "Stable artifact identifiers owned by the ArtifactGraph. Qwen3 256k context window comfortably holds 16-reference-image generations and multi-segment storyboard frames; the runtime never silently drops artifacts. Trim only when measured budget pressure forces it.",
75
+ "properties": {
76
+ "selectedArtifactIds": {
77
+ "type": "array",
78
+ "description": "Artifacts the user or planner has explicitly focused (e.g. multi-select for batch edit).",
79
+ "items": { "type": "string" }
80
+ },
81
+ "artifactIds": {
82
+ "type": "array",
83
+ "description": "Full conversation artifact id list. UNBOUNDED by default. Qwen3 256k context window comfortably holds 16-reference-image generations and multi-segment storyboard frames; the runtime never silently drops artifacts. Trim only when measured budget pressure forces it.",
84
+ "items": { "type": "string" }
85
+ },
86
+ "lastGeneratedArtifactId": { "type": "string" },
87
+ "lastEditedArtifactId": { "type": "string" }
88
+ },
89
+ "required": ["selectedArtifactIds", "artifactIds"]
90
+ },
91
+ "recentTurns": {
92
+ "type": "array",
93
+ "description": "Recent transcript window. UNBOUNDED by default. Qwen3 256k context window comfortably holds 16-reference-image generations and multi-segment storyboard frames; the runtime never silently drops artifacts. Trim only when measured budget pressure forces it. Sliding-window trimming from contextWindow.ts only fires after measured token budget pressure, and writes the trimmed prefix into conversationSummary.",
94
+ "items": {
95
+ "type": "object",
96
+ "additionalProperties": false,
97
+ "properties": {
98
+ "role": { "type": "string", "enum": ["user", "assistant", "tool", "system"] },
99
+ "content": { "type": "string" },
100
+ "sequence": { "type": "integer", "minimum": 0 }
101
+ },
102
+ "required": ["role", "content", "sequence"]
103
+ }
104
+ },
105
+ "conversationSummary": {
106
+ "type": "string",
107
+ "description": "Rolling summary of trimmed prefix. Empty string until the first measured-budget-pressure trim event."
108
+ },
109
+ "userPreferences": {
110
+ "type": "object",
111
+ "description": "Free-form preference payload (e.g. preferred model tier, default aspect ratio). Deliberate exception: additionalProperties is true here, mirroring the data/metadata pattern in other contracts, because preferences evolve faster than the schema.",
112
+ "additionalProperties": true
113
+ },
114
+ "availableCapabilitiesSummary": {
115
+ "type": "array",
116
+ "description": "Short human-readable capability strings the planner can quote when answering capability questions without invoking any tool.",
117
+ "items": { "type": "string" }
118
+ }
119
+ },
120
+ "required": [
121
+ "currentMessage",
122
+ "activeState",
123
+ "artifactState",
124
+ "recentTurns",
125
+ "conversationSummary",
126
+ "availableCapabilitiesSummary"
127
+ ]
128
+ }
@@ -0,0 +1,75 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://schemas.sogni.ai/creative-agent/2026-05-20.1/agent/turn-analysis.schema.json",
4
+ "title": "L1 IntentClassifier output",
5
+ "schemaVersion": "2026-05-20.1",
6
+ "description": "Structured output of the L1 IntentClassifier. Produced by an LLM-backed classifier, a typed planner, runtime state inspection, the artifact graph, or an explicit user signal — never by regex. Per the Codex reconciliation in v2 plan §1.A, the legacy SignalSource value 'regex' is intentionally absent from SignalProvenance; regex is demoted to bounded fact extraction (dimensions, durations, file types) which does not produce semantic intent.",
7
+ "type": "object",
8
+ "additionalProperties": false,
9
+ "$defs": {
10
+ "SignalProvenance": {
11
+ "type": "string",
12
+ "description": "Which layer authored this TurnAnalysis. The value 'regex' is intentionally absent — regex extracts bounded facts only and never decides intent, tool surface, or routing.",
13
+ "enum": [
14
+ "classifier",
15
+ "planner",
16
+ "runtime_state",
17
+ "artifact_graph",
18
+ "user_explicit"
19
+ ]
20
+ }
21
+ },
22
+ "properties": {
23
+ "domain": {
24
+ "type": "string",
25
+ "enum": ["chat", "image", "video", "audio", "analysis", "workflow", "memory", "settings", "unknown"]
26
+ },
27
+ "intent": {
28
+ "type": "string",
29
+ "enum": ["generate", "edit", "analyze", "transform", "question", "capability", "continue", "reference", "configure", "clarify", "unknown"]
30
+ },
31
+ "executionMode": {
32
+ "type": "string",
33
+ "enum": ["none", "tool", "multi_tool", "workflow"]
34
+ },
35
+ "userWantsExecution": { "type": "boolean" },
36
+ "isCapabilityQuestion": { "type": "boolean" },
37
+ "isFutureInstruction": { "type": "boolean" },
38
+ "isReferenceOnly": { "type": "boolean" },
39
+ "needsPriorContext": { "type": "boolean" },
40
+ "needsClarification": { "type": "boolean" },
41
+ "referencedArtifacts": {
42
+ "type": "array",
43
+ "items": { "type": "string" },
44
+ "description": "Artifact ids the classifier resolved from the user's message via the artifact graph (positional references like 'the second image' resolve to stable ids before this list is emitted)."
45
+ },
46
+ "requiredCapabilities": {
47
+ "type": "array",
48
+ "items": { "type": "string" },
49
+ "description": "Free-form capability tags the planner must satisfy (e.g. 'image_generation', 'video_stitch')."
50
+ },
51
+ "confidence": {
52
+ "type": "number",
53
+ "minimum": 0,
54
+ "maximum": 1
55
+ },
56
+ "provenance": {
57
+ "$ref": "#/$defs/SignalProvenance"
58
+ }
59
+ },
60
+ "required": [
61
+ "domain",
62
+ "intent",
63
+ "executionMode",
64
+ "userWantsExecution",
65
+ "isCapabilityQuestion",
66
+ "isFutureInstruction",
67
+ "isReferenceOnly",
68
+ "needsPriorContext",
69
+ "referencedArtifacts",
70
+ "requiredCapabilities",
71
+ "needsClarification",
72
+ "confidence",
73
+ "provenance"
74
+ ]
75
+ }
@@ -0,0 +1,42 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://schemas.sogni.ai/creative-agent/2026-05-20.1/artifacts/artifact-graph.schema.json",
4
+ "title": "Artifact graph serialization",
5
+ "schemaVersion": "2026-05-20.1",
6
+ "description": "Durable serialization of the in-memory ArtifactGraph. The same shape mirrors into ChatRunRecord.artifacts[] and WorkflowRunRecord.artifacts[] so cloud and hosted runners can reconstitute the graph on resume. The runtime Map<string, ArtifactNode> serializes as a positional nodes[] array; artifactId uniqueness is a runtime invariant, not expressible cheaply in JSON Schema.",
7
+ "type": "object",
8
+ "additionalProperties": false,
9
+ "properties": {
10
+ "nodes": {
11
+ "type": "array",
12
+ "description": "All artifact nodes in the graph. Uniqueness on artifactId is enforced by the runtime, not the schema.",
13
+ "items": {
14
+ "$ref": "./artifact-node.schema.json"
15
+ }
16
+ },
17
+ "selectedId": {
18
+ "type": "string",
19
+ "description": "Optional currently-selected artifact id. When present, MUST match an artifactId in nodes."
20
+ },
21
+ "compatibilityProjections": {
22
+ "type": "object",
23
+ "additionalProperties": false,
24
+ "description": "Read-only legacy projections derived from nodes[]. Populated during the Phase 1-4 transition so existing chat-side code that still reads positional URL arrays keeps working. Deleted after plan Phase 5.",
25
+ "properties": {
26
+ "resultUrls": {
27
+ "type": "array",
28
+ "items": { "type": "string" }
29
+ },
30
+ "videoResultUrls": {
31
+ "type": "array",
32
+ "items": { "type": "string" }
33
+ },
34
+ "audioResultUrls": {
35
+ "type": "array",
36
+ "items": { "type": "string" }
37
+ }
38
+ }
39
+ }
40
+ },
41
+ "required": ["nodes"]
42
+ }