@sogni-ai/sogni-protocol 1.0.0-alpha.6 → 1.0.0-alpha.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/manifests/generation-tools.json +6 -6
- package/manifests/openai-tools.json +6 -6
- package/package.json +1 -1
- package/schemas/billing/spend-gate.schema.json +58 -37
- package/schemas/events/run-event.schema.json +17 -6
- package/schemas/tools/animate_photo.schema.json +1 -1
- package/schemas/tools/generate_video.schema.json +2 -2
- package/schemas/tools/sound_to_video.schema.json +2 -2
- package/schemas/tools/video_to_video.schema.json +1 -1
- package/version.json +1 -1
|
@@ -165,7 +165,7 @@
|
|
|
165
165
|
"seedance2",
|
|
166
166
|
"seedance2-fast"
|
|
167
167
|
],
|
|
168
|
-
"description": "Video model. \"ltx23\" (default): LTX 2.3 with native audio; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx23\".
|
|
168
|
+
"description": "Video model. \"ltx23\" (default): LTX 2.3 with native audio; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx23\". Seedance quality is selected only by model: use \"seedance2-fast\" when the user asks for Seedance Fast / seedance-fast / faster draft iteration, and use \"seedance2\" for the full Seedance 2.0 model, explicit non-fast/full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft or the fast model. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. Seedance supports multimodal loose reference assets: images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices."
|
|
169
169
|
},
|
|
170
170
|
"generateAudio": {
|
|
171
171
|
"type": "boolean",
|
|
@@ -202,7 +202,7 @@
|
|
|
202
202
|
},
|
|
203
203
|
"targetResolution": {
|
|
204
204
|
"type": "number",
|
|
205
|
-
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
205
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This is resolution only, not a Seedance quality tier: Seedance quality is selected by videoModel (\"seedance2\" vs \"seedance2-fast\"). Seedance 2.0 full supports 4K; Seedance Fast supports 480p/720p only. Do not set targetResolution from Default Media Quality Fast/HQ/Pro. If omitted for Seedance, the host uses the selected model default. This preserves/inherits the current video shape instead of forcing landscape. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact width/height/aspectRatio instead."
|
|
206
206
|
},
|
|
207
207
|
"numberOfVariations": {
|
|
208
208
|
"type": "number",
|
|
@@ -531,7 +531,7 @@
|
|
|
531
531
|
},
|
|
532
532
|
"targetResolution": {
|
|
533
533
|
"type": "number",
|
|
534
|
-
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
534
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This preserves the source/reference aspect ratio. Do NOT set exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact-pixel aspectRatio instead."
|
|
535
535
|
},
|
|
536
536
|
"sourceImageIndex": {
|
|
537
537
|
"type": "number",
|
|
@@ -682,7 +682,7 @@
|
|
|
682
682
|
},
|
|
683
683
|
"targetResolution": {
|
|
684
684
|
"type": "number",
|
|
685
|
-
"description": "Seedance V2V only. Short-side output resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
685
|
+
"description": "Seedance V2V only. Short-side output resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact dimensions. Seedance V2V full supports 4K; Seedance V2V fast supports 480p and 720p. Preserve the source video shape instead of forcing landscape pixels."
|
|
686
686
|
},
|
|
687
687
|
"sourceImageIndex": {
|
|
688
688
|
"type": "number",
|
|
@@ -923,7 +923,7 @@
|
|
|
923
923
|
"ltx23-ia2v",
|
|
924
924
|
"ltx23-a2v"
|
|
925
925
|
],
|
|
926
|
-
"description": "Video model. \"ltx23-ia2v\" (default when image available): LTX 2.3 image+audio to video, audio-reactive with a reference image; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"ltx23-a2v\" (default when no image): LTX 2.3 audio-only to video, no image needed, creates video purely from text prompt + audio with the same quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-fast\": Seedance 2.0 Fast
|
|
926
|
+
"description": "Video model. \"ltx23-ia2v\" (default when image available): LTX 2.3 image+audio to video, audio-reactive with a reference image; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"ltx23-a2v\" (default when no image): LTX 2.3 audio-only to video, no image needed, creates video purely from text prompt + audio with the same quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-fast\": Seedance 2.0 Fast for faster/lower-cost iteration. Seedance quality is selected only by this model value: pick \"seedance2-fast\" when the user says Seedance Fast / seedance-fast / faster draft, and pick \"seedance2\" for full/non-fast Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present."
|
|
927
927
|
},
|
|
928
928
|
"generateAudio": {
|
|
929
929
|
"type": "boolean",
|
|
@@ -937,7 +937,7 @@
|
|
|
937
937
|
},
|
|
938
938
|
"targetResolution": {
|
|
939
939
|
"type": "number",
|
|
940
|
-
"description": "Short-side video resolution target in pixels. Use
|
|
940
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This preserves the source/reference aspect ratio. Do NOT set exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact-pixel aspectRatio instead."
|
|
941
941
|
},
|
|
942
942
|
"aspectRatio": {
|
|
943
943
|
"type": "string",
|
|
@@ -146,7 +146,7 @@
|
|
|
146
146
|
"seedance2",
|
|
147
147
|
"seedance2-fast"
|
|
148
148
|
],
|
|
149
|
-
"description": "Video model. \"ltx23\" (default): LTX 2.3 with native audio; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx23\".
|
|
149
|
+
"description": "Video model. \"ltx23\" (default): LTX 2.3 with native audio; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx23\". Seedance quality is selected only by model: use \"seedance2-fast\" when the user asks for Seedance Fast / seedance-fast / faster draft iteration, and use \"seedance2\" for the full Seedance 2.0 model, explicit non-fast/full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft or the fast model. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. Seedance supports multimodal loose reference assets: images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices."
|
|
150
150
|
},
|
|
151
151
|
"generateAudio": {
|
|
152
152
|
"type": "boolean",
|
|
@@ -183,7 +183,7 @@
|
|
|
183
183
|
},
|
|
184
184
|
"targetResolution": {
|
|
185
185
|
"type": "number",
|
|
186
|
-
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
186
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This is resolution only, not a Seedance quality tier: Seedance quality is selected by videoModel (\"seedance2\" vs \"seedance2-fast\"). Seedance 2.0 full supports 4K; Seedance Fast supports 480p/720p only. Do not set targetResolution from Default Media Quality Fast/HQ/Pro. If omitted for Seedance, the host uses the selected model default. This preserves/inherits the current video shape instead of forcing landscape. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact width/height/aspectRatio instead."
|
|
187
187
|
},
|
|
188
188
|
"numberOfVariations": {
|
|
189
189
|
"type": "number",
|
|
@@ -512,7 +512,7 @@
|
|
|
512
512
|
},
|
|
513
513
|
"targetResolution": {
|
|
514
514
|
"type": "number",
|
|
515
|
-
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
515
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This preserves the source/reference aspect ratio. Do NOT set exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact-pixel aspectRatio instead."
|
|
516
516
|
},
|
|
517
517
|
"sourceImageIndex": {
|
|
518
518
|
"type": "number",
|
|
@@ -663,7 +663,7 @@
|
|
|
663
663
|
},
|
|
664
664
|
"targetResolution": {
|
|
665
665
|
"type": "number",
|
|
666
|
-
"description": "Seedance V2V only. Short-side output resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
666
|
+
"description": "Seedance V2V only. Short-side output resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact dimensions. Seedance V2V full supports 4K; Seedance V2V fast supports 480p and 720p. Preserve the source video shape instead of forcing landscape pixels."
|
|
667
667
|
},
|
|
668
668
|
"sourceImageIndex": {
|
|
669
669
|
"type": "number",
|
|
@@ -904,7 +904,7 @@
|
|
|
904
904
|
"ltx23-ia2v",
|
|
905
905
|
"ltx23-a2v"
|
|
906
906
|
],
|
|
907
|
-
"description": "Video model. \"ltx23-ia2v\" (default when image available): LTX 2.3 image+audio to video, audio-reactive with a reference image; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"ltx23-a2v\" (default when no image): LTX 2.3 audio-only to video, no image needed, creates video purely from text prompt + audio with the same quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-fast\": Seedance 2.0 Fast
|
|
907
|
+
"description": "Video model. \"ltx23-ia2v\" (default when image available): LTX 2.3 image+audio to video, audio-reactive with a reference image; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"ltx23-a2v\" (default when no image): LTX 2.3 audio-only to video, no image needed, creates video purely from text prompt + audio with the same quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-fast\": Seedance 2.0 Fast for faster/lower-cost iteration. Seedance quality is selected only by this model value: pick \"seedance2-fast\" when the user says Seedance Fast / seedance-fast / faster draft, and pick \"seedance2\" for full/non-fast Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present."
|
|
908
908
|
},
|
|
909
909
|
"generateAudio": {
|
|
910
910
|
"type": "boolean",
|
|
@@ -918,7 +918,7 @@
|
|
|
918
918
|
},
|
|
919
919
|
"targetResolution": {
|
|
920
920
|
"type": "number",
|
|
921
|
-
"description": "Short-side video resolution target in pixels. Use
|
|
921
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This preserves the source/reference aspect ratio. Do NOT set exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact-pixel aspectRatio instead."
|
|
922
922
|
},
|
|
923
923
|
"aspectRatio": {
|
|
924
924
|
"type": "string",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@sogni-ai/sogni-protocol",
|
|
3
|
-
"version": "1.0.0-alpha.
|
|
3
|
+
"version": "1.0.0-alpha.7",
|
|
4
4
|
"description": "Language-neutral protocol artifacts for the Sogni ecosystem: tool schemas, prompts, OpenAI tool manifests, and enums. Consumed by every Sogni SDK (TypeScript, Swift, and future Python/Kotlin/Rust SDKs) so contracts stay in lockstep across languages.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"sogni",
|
|
@@ -3,9 +3,8 @@
|
|
|
3
3
|
"$id": "https://schemas.sogni.ai/creative-agent/2026-05-20.1/billing/spend-gate.schema.json",
|
|
4
4
|
"title": "Spend gate request and state",
|
|
5
5
|
"schemaVersion": "2026-05-20.1",
|
|
6
|
-
"description": "Single shared spend-approval state machine for
|
|
6
|
+
"description": "Single shared spend-approval state machine for atomic tool calls (scope='tool_call'), grouped concurrent dispatches (scope='parallel_batch'), and workflow authorizations (scope='workflow_run'). One envelope, one state enum, one transition log. Per-job settlement remains on the sogni-socket project+N path — this gate authorizes spend, it does not move funds. Note: additionalProperties:false is applied per oneOf branch, not at the root — a root-level additionalProperties has no sibling properties to whitelist against and would reject every payload.",
|
|
7
7
|
"type": "object",
|
|
8
|
-
"additionalProperties": false,
|
|
9
8
|
"$defs": {
|
|
10
9
|
"SpendGateState": {
|
|
11
10
|
"type": "string",
|
|
@@ -23,12 +22,13 @@
|
|
|
23
22
|
},
|
|
24
23
|
"SpendGateDecision": {
|
|
25
24
|
"type": "string",
|
|
26
|
-
"
|
|
25
|
+
"description": "Decision recorded when the gate leaves waiting_for_user. Three historical vocabularies are accepted: 'confirm'/'cancel' (canonical), 'approved'/'rejected' (pre-2026-05-20 aliases still emitted by sogni-api and sogni-creative-agent durable runs), and 'cancelled' (alternate past-tense spelling collapsing to 'cancel'). Consumers normalize via normalizeSpendDecision so business logic only sees the canonical pair.",
|
|
26
|
+
"enum": ["confirm", "cancel", "approved", "rejected", "cancelled"]
|
|
27
27
|
},
|
|
28
28
|
"SpendEstimateLineItem": {
|
|
29
29
|
"type": "object",
|
|
30
30
|
"additionalProperties": false,
|
|
31
|
-
"description": "One row of the spend estimate breakdown. Sum of (units * model price) across line items yields the
|
|
31
|
+
"description": "One row of the spend estimate breakdown. Sum of (units * model price) across line items yields the estimate's capacityUnits.",
|
|
32
32
|
"properties": {
|
|
33
33
|
"model": { "type": "string" },
|
|
34
34
|
"units": { "type": "number", "minimum": 0 },
|
|
@@ -39,13 +39,33 @@
|
|
|
39
39
|
},
|
|
40
40
|
"required": ["model", "units", "tokenType"]
|
|
41
41
|
},
|
|
42
|
+
"SpendGateEstimate": {
|
|
43
|
+
"type": "object",
|
|
44
|
+
"additionalProperties": false,
|
|
45
|
+
"description": "Estimate payload carried on the gate itself. capacityUnits = sum of breakdown units; tokenType = denomination the breakdown is in; maxAcceptableUnits = optional caller-supplied cap.",
|
|
46
|
+
"properties": {
|
|
47
|
+
"capacityUnits": { "type": "number", "minimum": 0 },
|
|
48
|
+
"breakdown": {
|
|
49
|
+
"type": "array",
|
|
50
|
+
"minItems": 1,
|
|
51
|
+
"items": { "$ref": "#/$defs/SpendEstimateLineItem" }
|
|
52
|
+
},
|
|
53
|
+
"tokenType": {
|
|
54
|
+
"type": "string",
|
|
55
|
+
"enum": ["spark", "sogni"]
|
|
56
|
+
},
|
|
57
|
+
"maxAcceptableUnits": { "type": "number", "minimum": 0 }
|
|
58
|
+
},
|
|
59
|
+
"required": ["capacityUnits", "breakdown", "tokenType"]
|
|
60
|
+
},
|
|
42
61
|
"PendingToolCall": {
|
|
43
62
|
"type": "object",
|
|
44
63
|
"additionalProperties": false,
|
|
45
|
-
"description": "Minimal reference to a tool call this gate covers (when scope='tool_call' the array has one entry; when scope='workflow_run' the gate authorizes the umbrella and pendingToolCalls may be empty).",
|
|
64
|
+
"description": "Minimal reference to a tool call this gate covers (when scope='tool_call' the array has one entry; when scope='parallel_batch' it carries the constituent calls of the fan-out; when scope='workflow_run' the gate authorizes the umbrella and pendingToolCalls may be empty).",
|
|
46
65
|
"properties": {
|
|
47
66
|
"toolCallId": { "type": "string" },
|
|
48
|
-
"toolName": { "type": "string" }
|
|
67
|
+
"toolName": { "type": "string" },
|
|
68
|
+
"estimateUnits": { "type": "number", "minimum": 0 }
|
|
49
69
|
},
|
|
50
70
|
"required": ["toolCallId", "toolName"]
|
|
51
71
|
},
|
|
@@ -73,27 +93,39 @@
|
|
|
73
93
|
"type": "array",
|
|
74
94
|
"items": { "$ref": "#/$defs/PendingToolCall" }
|
|
75
95
|
},
|
|
76
|
-
"estimate": {
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
96
|
+
"estimate": { "$ref": "#/$defs/SpendGateEstimate" },
|
|
97
|
+
"state": { "$ref": "#/$defs/SpendGateState" },
|
|
98
|
+
"reason": { "type": "string" },
|
|
99
|
+
"decision": { "$ref": "#/$defs/SpendGateDecision" },
|
|
100
|
+
"createdAt": { "type": "string", "format": "date-time" },
|
|
101
|
+
"decidedAt": { "type": "string", "format": "date-time" },
|
|
102
|
+
"updatedAt": { "type": "string", "format": "date-time" },
|
|
103
|
+
"failureReason": { "type": "string" }
|
|
104
|
+
},
|
|
105
|
+
"required": ["gateId", "scope", "toolCallId", "estimate", "state", "updatedAt"]
|
|
106
|
+
},
|
|
107
|
+
{
|
|
108
|
+
"type": "object",
|
|
109
|
+
"additionalProperties": false,
|
|
110
|
+
"properties": {
|
|
111
|
+
"gateId": { "type": "string" },
|
|
112
|
+
"runId": { "type": "string" },
|
|
113
|
+
"scope": { "const": "parallel_batch" },
|
|
114
|
+
"pendingToolCalls": {
|
|
115
|
+
"type": "array",
|
|
116
|
+
"minItems": 1,
|
|
117
|
+
"items": { "$ref": "#/$defs/PendingToolCall" }
|
|
89
118
|
},
|
|
119
|
+
"estimate": { "$ref": "#/$defs/SpendGateEstimate" },
|
|
90
120
|
"state": { "$ref": "#/$defs/SpendGateState" },
|
|
91
121
|
"reason": { "type": "string" },
|
|
92
122
|
"decision": { "$ref": "#/$defs/SpendGateDecision" },
|
|
123
|
+
"createdAt": { "type": "string", "format": "date-time" },
|
|
93
124
|
"decidedAt": { "type": "string", "format": "date-time" },
|
|
94
|
-
"
|
|
125
|
+
"updatedAt": { "type": "string", "format": "date-time" },
|
|
126
|
+
"failureReason": { "type": "string" }
|
|
95
127
|
},
|
|
96
|
-
"required": ["gateId", "scope", "
|
|
128
|
+
"required": ["gateId", "scope", "pendingToolCalls", "estimate", "state", "updatedAt"]
|
|
97
129
|
},
|
|
98
130
|
{
|
|
99
131
|
"type": "object",
|
|
@@ -104,27 +136,16 @@
|
|
|
104
136
|
"scope": { "const": "workflow_run" },
|
|
105
137
|
"workflowRunId": { "type": "string" },
|
|
106
138
|
"pendingWorkflowPlan": { "$ref": "#/$defs/PendingWorkflowPlan" },
|
|
107
|
-
"estimate": {
|
|
108
|
-
"type": "object",
|
|
109
|
-
"additionalProperties": false,
|
|
110
|
-
"properties": {
|
|
111
|
-
"totalEstimatedCapacityUnits": { "type": "number", "minimum": 0 },
|
|
112
|
-
"breakdown": {
|
|
113
|
-
"type": "array",
|
|
114
|
-
"minItems": 1,
|
|
115
|
-
"items": { "$ref": "#/$defs/SpendEstimateLineItem" }
|
|
116
|
-
},
|
|
117
|
-
"maxAcceptableUnits": { "type": "number", "minimum": 0 }
|
|
118
|
-
},
|
|
119
|
-
"required": ["totalEstimatedCapacityUnits", "breakdown"]
|
|
120
|
-
},
|
|
139
|
+
"estimate": { "$ref": "#/$defs/SpendGateEstimate" },
|
|
121
140
|
"state": { "$ref": "#/$defs/SpendGateState" },
|
|
122
141
|
"reason": { "type": "string" },
|
|
123
142
|
"decision": { "$ref": "#/$defs/SpendGateDecision" },
|
|
143
|
+
"createdAt": { "type": "string", "format": "date-time" },
|
|
124
144
|
"decidedAt": { "type": "string", "format": "date-time" },
|
|
125
|
-
"
|
|
145
|
+
"updatedAt": { "type": "string", "format": "date-time" },
|
|
146
|
+
"failureReason": { "type": "string" }
|
|
126
147
|
},
|
|
127
|
-
"required": ["gateId", "scope", "workflowRunId", "estimate", "state", "
|
|
148
|
+
"required": ["gateId", "scope", "workflowRunId", "estimate", "state", "updatedAt"]
|
|
128
149
|
}
|
|
129
150
|
]
|
|
130
151
|
}
|
|
@@ -9,10 +9,12 @@
|
|
|
9
9
|
"$defs": {
|
|
10
10
|
"RunEventType": {
|
|
11
11
|
"type": "string",
|
|
12
|
-
"description": "All event types. Clusters: lifecycle (run_queued, run_started, run_completed, run_partial_failure, run_failed, run_cancelled), LLM (llm_round_started, llm_token, llm_round_completed), tools (tool_call_proposed, tool_call_dispatched, tool_call_progress, tool_call_resolved), artifacts (artifact_created, artifact_updated, artifact_referenced), waiting (run_waiting_for_user), spend (spend_preview_emitted, spend_confirmed, spend_cancelled, spend_insufficient
|
|
12
|
+
"description": "All event types — superset of every type emitted by any current consumer (sogni-creative-agent-v2, sogni-chat, sogni-api) plus the schema-mandated set; mirrors RUN_EVENT_TYPES in sogni-intelligence-client src/events/runEvent.ts. Clusters: lifecycle (run_created, run_queued, run_started, run_resumed, run_completed, run_partial_failure, run_failed, run_cancelled), LLM (llm_round_started, llm_token, llm_round_completed, assistant_message_delta, assistant_message_completed), tools (tool_call_proposed, tool_call_dispatched, tool_call_progress, tool_call_resolved), artifacts (artifact_created, artifact_updated, artifact_referenced), media context (media_context_updated, media_turn_intent_classified, asset_manifest_updated), waiting (run_waiting_for_user), spend/billing (billing_preview_updated, spend_gate_opened, spend_preview_emitted, spend_confirmed, spend_cancelled, spend_insufficient, run_awaiting_cost_confirmation, run_cost_confirmation_resolved), workflow-stage (stage_started, stage_completed, stage_failed, stage_waiting_for_user), audit (audit_evaluated, repair_requested).",
|
|
13
13
|
"enum": [
|
|
14
|
+
"run_created",
|
|
14
15
|
"run_queued",
|
|
15
16
|
"run_started",
|
|
17
|
+
"run_resumed",
|
|
16
18
|
"run_completed",
|
|
17
19
|
"run_partial_failure",
|
|
18
20
|
"run_failed",
|
|
@@ -20,6 +22,8 @@
|
|
|
20
22
|
"llm_round_started",
|
|
21
23
|
"llm_token",
|
|
22
24
|
"llm_round_completed",
|
|
25
|
+
"assistant_message_delta",
|
|
26
|
+
"assistant_message_completed",
|
|
23
27
|
"tool_call_proposed",
|
|
24
28
|
"tool_call_dispatched",
|
|
25
29
|
"tool_call_progress",
|
|
@@ -27,17 +31,24 @@
|
|
|
27
31
|
"artifact_created",
|
|
28
32
|
"artifact_updated",
|
|
29
33
|
"artifact_referenced",
|
|
34
|
+
"media_context_updated",
|
|
35
|
+
"media_turn_intent_classified",
|
|
36
|
+
"asset_manifest_updated",
|
|
30
37
|
"run_waiting_for_user",
|
|
38
|
+
"billing_preview_updated",
|
|
39
|
+
"spend_gate_opened",
|
|
31
40
|
"spend_preview_emitted",
|
|
32
41
|
"spend_confirmed",
|
|
33
42
|
"spend_cancelled",
|
|
34
43
|
"spend_insufficient",
|
|
35
|
-
"
|
|
36
|
-
"
|
|
44
|
+
"run_awaiting_cost_confirmation",
|
|
45
|
+
"run_cost_confirmation_resolved",
|
|
37
46
|
"stage_started",
|
|
38
47
|
"stage_completed",
|
|
39
48
|
"stage_failed",
|
|
40
|
-
"stage_waiting_for_user"
|
|
49
|
+
"stage_waiting_for_user",
|
|
50
|
+
"audit_evaluated",
|
|
51
|
+
"repair_requested"
|
|
41
52
|
]
|
|
42
53
|
},
|
|
43
54
|
"RunStatus": {
|
|
@@ -71,8 +82,8 @@
|
|
|
71
82
|
"runId": { "type": "string" },
|
|
72
83
|
"runKind": {
|
|
73
84
|
"type": "string",
|
|
74
|
-
"enum": ["chat", "workflow"],
|
|
75
|
-
"description": "Substrate discriminator (plan §13.5). chat and workflow runs share persistence, lease, heartbeat, event log, waiting semantics, cancellation, cost confirmation, and resume."
|
|
85
|
+
"enum": ["chat", "workflow", "tool_batch"],
|
|
86
|
+
"description": "Substrate discriminator (plan §13.5). chat and workflow runs share persistence, lease, heartbeat, event log, waiting semantics, cancellation, cost confirmation, and resume. tool_batch is used by sogni-creative-agent-v2 for fan-out tool-call runs that share the same event substrate without being a full chat or workflow run."
|
|
76
87
|
},
|
|
77
88
|
"sequence": {
|
|
78
89
|
"type": "integer",
|
|
@@ -37,7 +37,7 @@
|
|
|
37
37
|
},
|
|
38
38
|
"targetResolution": {
|
|
39
39
|
"type": "number",
|
|
40
|
-
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
40
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This preserves the source/reference aspect ratio. Do NOT set exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact-pixel aspectRatio instead."
|
|
41
41
|
},
|
|
42
42
|
"sourceImageIndex": {
|
|
43
43
|
"type": "number",
|
|
@@ -37,7 +37,7 @@
|
|
|
37
37
|
"seedance2",
|
|
38
38
|
"seedance2-fast"
|
|
39
39
|
],
|
|
40
|
-
"description": "Video model. \"ltx23\" (default): LTX 2.3 with native audio; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx23\".
|
|
40
|
+
"description": "Video model. \"ltx23\" (default): LTX 2.3 with native audio; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"wan22\": Fast 4-step, simple motion, no audio. Default: \"ltx23\". Seedance quality is selected only by model: use \"seedance2-fast\" when the user asks for Seedance Fast / seedance-fast / faster draft iteration, and use \"seedance2\" for the full Seedance 2.0 model, explicit non-fast/full-quality requests, 1080p/4K requests, or generated/uploaded storyboard images unless the user explicitly asks for a draft or the fast model. Do not use Default Media Quality Fast/HQ/Pro or targetResolution to represent Seedance quality. Seedance supports multimodal loose reference assets: images (up to 9), videos (up to 3), and audios (up to 3), with no more than 12 asset files total. Use @Image1/@Video1/@Audio1 style references in creative briefs when assigning roles. Assign every useful reference asset a role and prefer positive preservation constraints. If an uploaded video is the source clip to transform, upscale, enhance, restyle, or remaster, use video_to_video with controlMode=\"seedance-v2v\" instead of generate_video referenceVideoIndices."
|
|
41
41
|
},
|
|
42
42
|
"generateAudio": {
|
|
43
43
|
"type": "boolean",
|
|
@@ -74,7 +74,7 @@
|
|
|
74
74
|
},
|
|
75
75
|
"targetResolution": {
|
|
76
76
|
"type": "number",
|
|
77
|
-
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
77
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This is resolution only, not a Seedance quality tier: Seedance quality is selected by videoModel (\"seedance2\" vs \"seedance2-fast\"). Seedance 2.0 full supports 4K; Seedance Fast supports 480p/720p only. Do not set targetResolution from Default Media Quality Fast/HQ/Pro. If omitted for Seedance, the host uses the selected model default. This preserves/inherits the current video shape instead of forcing landscape. Do NOT set width, height, or exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact width/height/aspectRatio instead."
|
|
78
78
|
},
|
|
79
79
|
"numberOfVariations": {
|
|
80
80
|
"type": "number",
|
|
@@ -43,7 +43,7 @@
|
|
|
43
43
|
"ltx23-ia2v",
|
|
44
44
|
"ltx23-a2v"
|
|
45
45
|
],
|
|
46
|
-
"description": "Video model. \"ltx23-ia2v\" (default when image available): LTX 2.3 image+audio to video, audio-reactive with a reference image; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"ltx23-a2v\" (default when no image): LTX 2.3 audio-only to video, no image needed, creates video purely from text prompt + audio with the same quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-fast\": Seedance 2.0 Fast
|
|
46
|
+
"description": "Video model. \"ltx23-ia2v\" (default when image available): LTX 2.3 image+audio to video, audio-reactive with a reference image; Fast/HQ use the distilled 8-step worker and Default Media Quality Pro uses the non-distilled dev worker. \"ltx23-a2v\" (default when no image): LTX 2.3 audio-only to video, no image needed, creates video purely from text prompt + audio with the same quality-tier routing. \"wan-s2v\": WAN 2.2 sound-to-video, best for lip-sync with a face image, fast 4-step. \"seedance2\": full Seedance 2.0 audio-reference video, 4-15s; this tool supplies the audio plus a required reference image because Seedance text+audio without image/video is unsupported. \"seedance2-fast\": Seedance 2.0 Fast for faster/lower-cost iteration. Seedance quality is selected only by this model value: pick \"seedance2-fast\" when the user says Seedance Fast / seedance-fast / faster draft, and pick \"seedance2\" for full/non-fast Seedance or 1080p/4K. Do not infer the Seedance model from Default Media Quality Fast/HQ/Pro. For Seedance audio-reference prompts, preserve exact spoken dialogue when the user supplied it, and assign @Image1/@Audio1 roles. If the user asks for speech without words, describe the vocal performance without inventing quoted dialogue. Treat lip-sync, voice cloning, and real-human reference behavior as provider-sensitive rather than guaranteed. Omit to auto-select based on whether an image is present."
|
|
47
47
|
},
|
|
48
48
|
"generateAudio": {
|
|
49
49
|
"type": "boolean",
|
|
@@ -57,7 +57,7 @@
|
|
|
57
57
|
},
|
|
58
58
|
"targetResolution": {
|
|
59
59
|
"type": "number",
|
|
60
|
-
"description": "Short-side video resolution target in pixels. Use
|
|
60
|
+
"description": "Short-side video resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact pixels or an output orientation. This preserves the source/reference aspect ratio. Do NOT set exact-pixel aspectRatio for bare named resolution requests. If the user says \"720p portrait\", \"720p landscape\", \"4K portrait\", or \"4K landscape\", use exact-pixel aspectRatio instead."
|
|
61
61
|
},
|
|
62
62
|
"aspectRatio": {
|
|
63
63
|
"type": "string",
|
|
@@ -52,7 +52,7 @@
|
|
|
52
52
|
},
|
|
53
53
|
"targetResolution": {
|
|
54
54
|
"type": "number",
|
|
55
|
-
"description": "Seedance V2V only. Short-side output resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", or \"
|
|
55
|
+
"description": "Seedance V2V only. Short-side output resolution target in pixels. Use when the user asks for a bare named resolution such as \"480p\", \"720p\", \"1080p\", \"2160p\", or \"4K\" without exact dimensions. Seedance V2V full supports 4K; Seedance V2V fast supports 480p and 720p. Preserve the source video shape instead of forcing landscape pixels."
|
|
56
56
|
},
|
|
57
57
|
"sourceImageIndex": {
|
|
58
58
|
"type": "number",
|
package/version.json
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
1
|
{
|
|
2
|
-
"protocolVersion": "1.2.
|
|
2
|
+
"protocolVersion": "1.2.2",
|
|
3
3
|
"description": "Sogni protocol artifact version. SDKs may refuse to operate against a protocolVersion they were not built for. Bump the major when removing or renaming any schema / enum / manifest field; bump the minor when adding new optional fields or new tools; bump the patch for description / prose changes only."
|
|
4
4
|
}
|