@sogni-ai/sogni-protocol 1.0.0-alpha.2 → 1.0.0-alpha.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -1
- package/catalogs/audio-models.json +68 -7
- package/catalogs/quality-presets.json +3 -3
- package/catalogs/seedance-reference-limits.json +35 -0
- package/manifests/composition-tools.json +3 -3
- package/manifests/generation-tools.json +120 -77
- package/manifests/openai-tools.json +108 -65
- package/package.json +1 -1
- package/prompts/tools/animate_photo.json +1 -1
- package/prompts/tools/compose_script.json +1 -1
- package/prompts/tools/compose_workflow.json +1 -1
- package/prompts/tools/compose_workflow_template.json +1 -1
- package/prompts/tools/edit_image.json +1 -1
- package/prompts/tools/enhance_prompt.json +1 -1
- package/prompts/tools/extend_video.json +2 -2
- package/prompts/tools/generate_image.json +2 -2
- package/prompts/tools/generate_video.json +1 -1
- package/prompts/tools/map_assets_for_model.json +1 -1
- package/prompts/tools/replace_video_segment.json +2 -2
- package/prompts/tools/resolve_personas.json +1 -1
- package/prompts/tools/sound_to_video.json +3 -2
- package/prompts/tools/video_to_video.json +2 -2
- package/schemas/agent/intent-input.schema.json +128 -0
- package/schemas/agent/turn-analysis.schema.json +75 -0
- package/schemas/artifacts/artifact-graph.schema.json +42 -0
- package/schemas/artifacts/artifact-node.schema.json +137 -0
- package/schemas/billing/spend-gate.schema.json +151 -0
- package/schemas/billing/workflow-authorization.schema.json +83 -0
- package/schemas/events/run-event.schema.json +122 -0
- package/schemas/tools/animate_photo.schema.json +23 -12
- package/schemas/tools/compose_script.schema.json +1 -1
- package/schemas/tools/compose_workflow.schema.json +2 -2
- package/schemas/tools/compose_workflow_template.schema.json +2 -2
- package/schemas/tools/edit_image.schema.json +8 -7
- package/schemas/tools/enhance_prompt.schema.json +1 -1
- package/schemas/tools/extend_video.schema.json +8 -5
- package/schemas/tools/generate_image.schema.json +12 -9
- package/schemas/tools/generate_music.schema.json +3 -2
- package/schemas/tools/generate_video.schema.json +24 -14
- package/schemas/tools/replace_video_segment.schema.json +5 -2
- package/schemas/tools/sound_to_video.schema.json +16 -8
- package/schemas/tools/tool-metadata.schema.json +78 -0
- package/schemas/tools/video_to_video.schema.json +13 -10
- package/version.json +1 -1
package/README.md
CHANGED
|
@@ -24,7 +24,7 @@ This package ships pure data — JSON Schemas, OpenAI tool manifests, prompt con
|
|
|
24
24
|
| `enums/chat-run-waiting-reasons.json` | Why a run is paused for user input. | All SDKs. |
|
|
25
25
|
| `enums/token-types.json` | Sogni billing token types (`sogni`, `spark`). | All SDKs. |
|
|
26
26
|
| `catalogs/quality-presets.json` | Quality tier presets (fast/hq/pro) -> model + sampling parameters. | UI SDKs rendering quality toggles. |
|
|
27
|
-
| `catalogs/audio-models.json` | ACE-Step audio model capability table + duration/BPM/time-signature constraints. | SDKs validating music-gen input. |
|
|
27
|
+
| `catalogs/audio-models.json` | ACE-Step + MiniMax Music 3 audio model capability table + duration/BPM/time-signature constraints. | SDKs validating music-gen input. |
|
|
28
28
|
| `version.json` | Protocol version for compatibility checks. | All SDKs. |
|
|
29
29
|
|
|
30
30
|
## Why a separate package
|
|
@@ -78,3 +78,12 @@ Tool prompt prose lives in `prompts/tools/*.json`. Each file is one [`PromptCont
|
|
|
78
78
|
## Editing schemas
|
|
79
79
|
|
|
80
80
|
The JSON Schema files under `schemas/` are the authoritative wire-spec shape. Any change here is a protocol bump (minor or major depending on compatibility). Consumers regenerate their language-specific types via their codegen step.
|
|
81
|
+
|
|
82
|
+
## v2 contract docs
|
|
83
|
+
|
|
84
|
+
This branch carries the additive schemas for the Sogni Platform v2 execution architecture. Two markdown documents under [`docs/`](./docs/) describe the new surface for downstream consumers:
|
|
85
|
+
|
|
86
|
+
- [`docs/v2-changes-summary.md`](./docs/v2-changes-summary.md) — concise map of the 8 new schemas (IntentInput, TurnAnalysis, ToolMetadata, ArtifactNode, ArtifactGraph, SpendGate, WorkflowAuthorization, RunEvent) and per-consumer impact.
|
|
87
|
+
- [`docs/v2-consumer-contract.md`](./docs/v2-consumer-contract.md) — full integration contract for native `sogni` (Mac/iOS via SogniKit codegen) and `sogni-creative-agent-skill`. Covers transport choice, classifier/regex replacement plan, artifact-graph projection, tool-surface budget, preserved public API params, and the workflow charging model.
|
|
88
|
+
|
|
89
|
+
v2 does not implement migration in the native or skill repos; their teams own implementation timing against this contract.
|
|
@@ -1,20 +1,76 @@
|
|
|
1
1
|
{
|
|
2
|
-
"description": "Audio / music generation model catalog. ACE-Step 1.5 family. Each entry lists per-parameter ranges so SDKs can validate user input and render sliders without round-tripping to the server. Cross-cutting audio constraints (duration, BPM, time signatures) live in the top-level `constraints` block.",
|
|
2
|
+
"description": "Audio / music generation model catalog. ACE-Step 1.5 family plus MiniMax Music 3. Each entry lists per-parameter ranges so SDKs can validate user input and render sliders without round-tripping to the server. Cross-cutting audio constraints (duration, BPM, time signatures) live in the top-level `constraints` block; models with different limits carry a per-model `constraints` override.",
|
|
3
3
|
"default": "turbo",
|
|
4
4
|
"models": {
|
|
5
5
|
"turbo": {
|
|
6
6
|
"id": "ace_step_1.5_turbo",
|
|
7
7
|
"name": "ACE-Step 1.5 Turbo",
|
|
8
|
-
"steps": {
|
|
9
|
-
|
|
8
|
+
"steps": {
|
|
9
|
+
"min": 4,
|
|
10
|
+
"max": 16,
|
|
11
|
+
"default": 8
|
|
12
|
+
},
|
|
13
|
+
"shift": {
|
|
14
|
+
"min": 1,
|
|
15
|
+
"max": 5,
|
|
16
|
+
"default": 3
|
|
17
|
+
},
|
|
10
18
|
"guidance": null
|
|
11
19
|
},
|
|
12
20
|
"sft": {
|
|
13
21
|
"id": "ace_step_1.5_sft",
|
|
14
22
|
"name": "ACE-Step 1.5 SFT",
|
|
15
|
-
"steps": {
|
|
16
|
-
|
|
17
|
-
|
|
23
|
+
"steps": {
|
|
24
|
+
"min": 10,
|
|
25
|
+
"max": 200,
|
|
26
|
+
"default": 50
|
|
27
|
+
},
|
|
28
|
+
"shift": {
|
|
29
|
+
"min": 1,
|
|
30
|
+
"max": 5,
|
|
31
|
+
"default": 3
|
|
32
|
+
},
|
|
33
|
+
"guidance": {
|
|
34
|
+
"min": 1,
|
|
35
|
+
"max": 15,
|
|
36
|
+
"default": 5
|
|
37
|
+
}
|
|
38
|
+
},
|
|
39
|
+
"music3": {
|
|
40
|
+
"id": "minimax_music3",
|
|
41
|
+
"name": "MiniMax Music 3",
|
|
42
|
+
"steps": {
|
|
43
|
+
"min": 10,
|
|
44
|
+
"max": 100,
|
|
45
|
+
"default": 30
|
|
46
|
+
},
|
|
47
|
+
"shift": null,
|
|
48
|
+
"guidance": {
|
|
49
|
+
"min": 1,
|
|
50
|
+
"max": 5,
|
|
51
|
+
"default": 1.7
|
|
52
|
+
},
|
|
53
|
+
"promptStrength": {
|
|
54
|
+
"min": 0,
|
|
55
|
+
"max": 10,
|
|
56
|
+
"default": 1.7
|
|
57
|
+
},
|
|
58
|
+
"topK": {
|
|
59
|
+
"min": 1,
|
|
60
|
+
"max": 16384,
|
|
61
|
+
"default": 50
|
|
62
|
+
},
|
|
63
|
+
"constraints": {
|
|
64
|
+
"duration": {
|
|
65
|
+
"min": 10,
|
|
66
|
+
"max": 300,
|
|
67
|
+
"default": 60,
|
|
68
|
+
"unit": "seconds",
|
|
69
|
+
"note": "Maximum duration; the planner composes an ending and may stop earlier."
|
|
70
|
+
},
|
|
71
|
+
"bpm": null,
|
|
72
|
+
"timeSignatures": null
|
|
73
|
+
}
|
|
18
74
|
}
|
|
19
75
|
},
|
|
20
76
|
"constraints": {
|
|
@@ -29,6 +85,11 @@
|
|
|
29
85
|
"max": 300,
|
|
30
86
|
"default": 120
|
|
31
87
|
},
|
|
32
|
-
"timeSignatures": [
|
|
88
|
+
"timeSignatures": [
|
|
89
|
+
2,
|
|
90
|
+
3,
|
|
91
|
+
4,
|
|
92
|
+
6
|
|
93
|
+
]
|
|
33
94
|
}
|
|
34
95
|
}
|
|
@@ -19,12 +19,12 @@
|
|
|
19
19
|
"description": "More detail"
|
|
20
20
|
},
|
|
21
21
|
"pro": {
|
|
22
|
-
"model": "
|
|
23
|
-
"steps":
|
|
22
|
+
"model": "qwen_image_edit_2511_fp8",
|
|
23
|
+
"steps": 25,
|
|
24
24
|
"guidance": 4.0,
|
|
25
25
|
"outputFormat": "jpg",
|
|
26
26
|
"label": "Pro",
|
|
27
|
-
"description": "
|
|
27
|
+
"description": "Maximum Qwen detail"
|
|
28
28
|
}
|
|
29
29
|
}
|
|
30
30
|
}
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
{
|
|
2
|
+
"description": "Maximum loose-reference assets accepted per Seedance video generation request. SDKs use this catalog to validate user input client-side and to render UI affordances (e.g. disable an upload button once the per-modality cap or the combined-total cap is reached). The numbers mirror the BytePlus Dreamina Seedance platform limits and are the same values previously hard-coded in sogni-chat's orchestration. Per-video request: at most `images` image refs (start frame + end frame + loose `referenceImageUrls`), `videos` video refs, and `audios` audio refs, with at most `assets` total ref files across all three modalities. The caps are per-model: model-aware consumers should read `perModel[<canonical model id>]` and reject unknown model IDs rather than guessing. `limits` intentionally remains the backwards-compatible default for consumers that do not select a model and carries the Seedance 2.0 family numbers so an SDK built against an older protocol under-permits rather than over-permits.",
|
|
3
|
+
"limits": {
|
|
4
|
+
"images": 9,
|
|
5
|
+
"videos": 3,
|
|
6
|
+
"audios": 3,
|
|
7
|
+
"assets": 12
|
|
8
|
+
},
|
|
9
|
+
"perModel": {
|
|
10
|
+
"seedance-2-0": {
|
|
11
|
+
"images": 9,
|
|
12
|
+
"videos": 3,
|
|
13
|
+
"audios": 3,
|
|
14
|
+
"assets": 12
|
|
15
|
+
},
|
|
16
|
+
"seedance-2-0-mini": {
|
|
17
|
+
"images": 9,
|
|
18
|
+
"videos": 3,
|
|
19
|
+
"audios": 3,
|
|
20
|
+
"assets": 12
|
|
21
|
+
},
|
|
22
|
+
"seedance-2-0-fast": {
|
|
23
|
+
"images": 9,
|
|
24
|
+
"videos": 3,
|
|
25
|
+
"audios": 3,
|
|
26
|
+
"assets": 12
|
|
27
|
+
},
|
|
28
|
+
"seedance-2-5": {
|
|
29
|
+
"images": 30,
|
|
30
|
+
"videos": 10,
|
|
31
|
+
"audios": 10,
|
|
32
|
+
"assets": 30
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
}
|
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
"properties": {
|
|
21
21
|
"prompt": { "type": "string", "description": "The source prompt, rough idea, or prompt revision request to enhance." },
|
|
22
22
|
"target_output": { "type": "string", "enum": ["image_prompt", "video_prompt", "music_prompt", "edit_prompt", "model_prompt", "general_prompt"], "description": "The kind of prompt artifact to produce." },
|
|
23
|
-
"destination_model": { "type": "string", "description": "Optional destination model selector, such as seedance2, ltx23, wan22,
|
|
23
|
+
"destination_model": { "type": "string", "description": "Optional destination model selector, such as seedance2, ltx23, wan22, gpt-image-2, or sdxl." },
|
|
24
24
|
"destination_tool": { "type": "string", "description": "Optional downstream generation tool, such as generate_image, edit_image, generate_video, animate_photo, sound_to_video, video_to_video, or generate_music." },
|
|
25
25
|
"prompting_type": { "type": "string", "enum": ["flux", "sdxl", "sd15", "pony", "fast", "sd3", "editing", "video"], "description": "Optional image-prompting family when producing an image prompt." },
|
|
26
26
|
"model_title": { "type": "string", "description": "Optional human-readable target model name for image prompt guidance." },
|
|
@@ -131,7 +131,7 @@
|
|
|
131
131
|
"type": "object",
|
|
132
132
|
"additionalProperties": false,
|
|
133
133
|
"properties": {
|
|
134
|
-
"image": { "type": "string", "description": "Preferred image model (e.g., '
|
|
134
|
+
"image": { "type": "string", "description": "Preferred image model (e.g., 'gpt-image-2', 'qwen')." },
|
|
135
135
|
"video": { "type": "string", "description": "Preferred video model (e.g., 'ltx23', 'wan22', 'seedance2')." },
|
|
136
136
|
"music": { "type": "string", "description": "Preferred music model." }
|
|
137
137
|
}
|
|
@@ -207,7 +207,7 @@
|
|
|
207
207
|
"type": "object",
|
|
208
208
|
"additionalProperties": false,
|
|
209
209
|
"properties": {
|
|
210
|
-
"image": { "type": "string", "description": "Preferred image model (e.g., '
|
|
210
|
+
"image": { "type": "string", "description": "Preferred image model (e.g., 'gpt-image-2', 'qwen')." },
|
|
211
211
|
"video": { "type": "string", "description": "Preferred video model (e.g., 'ltx23', 'wan22', 'seedance2')." },
|
|
212
212
|
"music": { "type": "string", "description": "Preferred music model." }
|
|
213
213
|
}
|